Skip to content

Commit 6545510

Browse files
committed
native: 汇编器针对ARM调整注释语法
1 parent aca3c33 commit 6545510

6 files changed

Lines changed: 221 additions & 7 deletions

File tree

‎docs/nasm.md‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -93,7 +93,9 @@ main:
9393

9494
### 2.1. 注释
9595

96-
`#` 开头的一行为注释
96+
`#` 和 `//` 开头的一行为注释.
97+
98+
注意, ARM 平台不支持 `#` 风格的注释(被用作了立即数前缀).
9799

98100
### 2.2. 标识符
99101

‎internal/native/arm64/lookup.go‎

Lines changed: 36 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,36 @@
1+
package arm64
2+
3+
import "wa-lang.org/wa/internal/native/abi"
4+
5+
// 寄存器名字列表
6+
var _Register = []string{}
7+
8+
// 指令的名字
9+
// 保持和指令定义相同的顺序
10+
var _Anames = []string{}
11+
12+
// 根据名字查找寄存器(忽略大小写, 忽略下划线和点的区别)
13+
func LookupRegister(regName string) (r abi.RegType, ok bool) {
14+
if regName == "" {
15+
return
16+
}
17+
for i, s := range _Register {
18+
if strEqualFold(s, regName) {
19+
return abi.RegType(i), true
20+
}
21+
}
22+
return 0, false
23+
}
24+
25+
// 根据名字查找汇编指令(忽略大小写, 忽略下划线和点的区别)
26+
func LookupAs(asName string) (as abi.As, ok bool) {
27+
if asName == "" {
28+
return
29+
}
30+
for i, s := range _Anames {
31+
if strEqualFold(s, asName) {
32+
return abi.As(i), true
33+
}
34+
}
35+
return 0, false
36+
}

‎internal/native/arm64/utils.go‎

Lines changed: 112 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,112 @@
1+
// Copyright (C) 2026 武汉凹语言科技有限公司
2+
// SPDX-License-Identifier: AGPL-3.0-or-later
3+
4+
package arm64
5+
6+
import (
7+
"fmt"
8+
"unicode"
9+
"unicode/utf8"
10+
)
11+
12+
func assert(ok bool, message ...interface{}) {
13+
if !ok {
14+
if len(message) != 0 {
15+
panic(fmt.Sprint(append([]interface{}{"assert failed:"}, message...)...))
16+
} else {
17+
panic("assert failed")
18+
}
19+
}
20+
}
21+
22+
// 忽略大小写
23+
// 下划线和"."视作相同
24+
func strEqualFold(s, t string) bool {
25+
// ASCII fast path
26+
i := 0
27+
for ; i < len(s) && i < len(t); i++ {
28+
sr := s[i]
29+
tr := t[i]
30+
if sr|tr >= utf8.RuneSelf {
31+
goto hasUnicode
32+
}
33+
34+
// Easy case.
35+
if tr == sr {
36+
continue
37+
}
38+
39+
// Make sr < tr to simplify what follows.
40+
if tr < sr {
41+
tr, sr = sr, tr
42+
}
43+
// ASCII only, sr/tr must be upper/lower case
44+
if 'A' <= sr && sr <= 'Z' && tr == sr+'a'-'A' {
45+
continue
46+
}
47+
// '_' 和 '.' 视作相等
48+
if (sr == '_' && tr == '.') || (sr == '.' && tr == '_') {
49+
continue
50+
}
51+
return false
52+
}
53+
// Check if we've exhausted both strings.
54+
return len(s) == len(t)
55+
56+
hasUnicode:
57+
s = s[i:]
58+
t = t[i:]
59+
for _, sr := range s {
60+
// If t is exhausted the strings are not equal.
61+
if len(t) == 0 {
62+
return false
63+
}
64+
65+
// Extract first rune from second string.
66+
var tr rune
67+
if t[0] < utf8.RuneSelf {
68+
tr, t = rune(t[0]), t[1:]
69+
} else {
70+
r, size := utf8.DecodeRuneInString(t)
71+
tr, t = r, t[size:]
72+
}
73+
74+
// If they match, keep going; if not, return false.
75+
76+
// Easy case.
77+
if tr == sr {
78+
continue
79+
}
80+
81+
// Make sr < tr to simplify what follows.
82+
if tr < sr {
83+
tr, sr = sr, tr
84+
}
85+
// Fast check for ASCII.
86+
if tr < utf8.RuneSelf {
87+
// ASCII only, sr/tr must be upper/lower case
88+
if 'A' <= sr && sr <= 'Z' && tr == sr+'a'-'A' {
89+
continue
90+
}
91+
// '_' 和 '.' 视作相等
92+
if (sr == '_' && tr == '.') || (sr == '.' && tr == '_') {
93+
continue
94+
}
95+
return false
96+
}
97+
98+
// General case. SimpleFold(x) returns the next equivalent rune > x
99+
// or wraps around to smaller values.
100+
r := unicode.SimpleFold(sr)
101+
for r != sr && r < tr {
102+
r = unicode.SimpleFold(r)
103+
}
104+
if r == tr {
105+
continue
106+
}
107+
return false
108+
}
109+
110+
// First string is empty, so check if the second one is also empty.
111+
return len(t) == 0
112+
}

‎internal/native/parser/parser.go‎

Lines changed: 24 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -8,6 +8,7 @@ import (
88
"strconv"
99

1010
"wa-lang.org/wa/internal/native/abi"
11+
"wa-lang.org/wa/internal/native/arm64"
1112
"wa-lang.org/wa/internal/native/ast"
1213
"wa-lang.org/wa/internal/native/loong64"
1314
"wa-lang.org/wa/internal/native/riscv"
@@ -63,6 +64,11 @@ func newParser(cpu abi.CPUType, fset *token.FileSet, filename string, src []byte
6364
p.gasSectionAlign = make(map[string]int)
6465
p.gasGlobl = make(map[string]bool)
6566

67+
useArmStyleComment := false
68+
if cpu == abi.ARM64 {
69+
useArmStyleComment = true
70+
}
71+
6672
switch cpu {
6773
case abi.LOONG64:
6874
p.scanner = scanner.NewScanner(
@@ -77,6 +83,7 @@ func newParser(cpu abi.CPUType, fset *token.FileSet, filename string, src []byte
7783
}
7884
return token.NONE
7985
},
86+
useArmStyleComment,
8087
)
8188
case abi.RISCV32, abi.RISCV64:
8289
p.scanner = scanner.NewScanner(
@@ -91,6 +98,7 @@ func newParser(cpu abi.CPUType, fset *token.FileSet, filename string, src []byte
9198
}
9299
return token.NONE
93100
},
101+
useArmStyleComment,
94102
)
95103
case abi.X64Unix, abi.X64Windows:
96104
p.scanner = scanner.NewScanner(
@@ -105,6 +113,22 @@ func newParser(cpu abi.CPUType, fset *token.FileSet, filename string, src []byte
105113
}
106114
return token.NONE
107115
},
116+
useArmStyleComment,
117+
)
118+
case abi.ARM64:
119+
p.scanner = scanner.NewScanner(
120+
func(ident string) token.Token {
121+
// 将原始的寄存器映射到 token.Token 编码
122+
if reg, ok := arm64.LookupRegister(ident); ok {
123+
return token.REG_ARM64_BEGIN + token.Token(reg)
124+
}
125+
// 将原始的指令映射到 token.Token 编码
126+
if as, ok := arm64.LookupAs(ident); ok {
127+
return token.A_ARM64_BEGIN + token.Token(as)
128+
}
129+
return token.NONE
130+
},
131+
useArmStyleComment,
108132
)
109133
default:
110134
panic(fmt.Errorf("unknown cpu: %v", cpu))

‎internal/native/scanner/scanner.go‎

Lines changed: 35 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -30,6 +30,8 @@ type ErrorHandler func(pos token.Position, msg string)
3030
type Scanner struct {
3131
// 查询特定平台的指令和寄存器
3232
lookupRegisterOrAs func(ident string) token.Token
33+
// 是否为ARM64风格的注释
34+
useArmStyleComment bool
3335

3436
// immutable state
3537
file *token.File // source file handle
@@ -52,9 +54,10 @@ type Scanner struct {
5254

5355
const bom = 0xFEFF // byte order mark, only permitted as very first character
5456

55-
func NewScanner(lookupRegisterOrAs func(ident string) token.Token) *Scanner {
57+
func NewScanner(lookupRegisterOrAs func(ident string) token.Token, useArmStyleComment bool) *Scanner {
5658
s := &Scanner{}
5759
s.lookupRegisterOrAs = lookupRegisterOrAs
60+
s.useArmStyleComment = useArmStyleComment
5861
return s
5962
}
6063

@@ -747,16 +750,43 @@ scanAgain:
747750
insertSemi = true
748751
tok = token.RBRACK
749752
case '#':
750-
// #-style comment
751-
if s.insertSemi && s.findLineEnd('#') {
753+
if s.useArmStyleComment {
754+
// ARM语法中 #123 为立即数
755+
insertSemi = true
756+
tok, lit = s.scanNumber()
757+
} else {
758+
// #-style comment
759+
if s.insertSemi && s.findLineEnd('#') {
760+
// reset position to the beginning of the comment
761+
s.ch = '#'
762+
s.offset = s.file.Offset(pos)
763+
s.rdOffset = s.offset + 1
764+
s.insertSemi = false // newline consumed
765+
return pos, token.SEMICOLON, "\n"
766+
}
767+
comment := s.scanComment('#')
768+
if s.mode&ScanComments == 0 {
769+
// skip comment
770+
s.insertSemi = false // newline consumed
771+
goto scanAgain
772+
}
773+
tok = token.COMMENT
774+
lit = comment
775+
}
776+
777+
case '/':
778+
// comment
779+
// 对于 ARM64, # 被用于立即数前缀, 因此需要提供不同的注释语法
780+
// 对于格式化时, 也需要做特别处理
781+
if s.insertSemi && s.findLineEnd('/') {
752782
// reset position to the beginning of the comment
753-
s.ch = '#'
783+
s.ch = '/'
754784
s.offset = s.file.Offset(pos)
755785
s.rdOffset = s.offset + 1
756786
s.insertSemi = false // newline consumed
757787
return pos, token.SEMICOLON, "\n"
758788
}
759-
comment := s.scanComment('#')
789+
comment := s.scanComment('/')
760790
if s.mode&ScanComments == 0 {
761791
// skip comment
762792
s.insertSemi = false // newline consumed

‎internal/native/token/token.go‎

Lines changed: 11 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -106,12 +106,14 @@ const (
106106
REG_LOONG_BEGIN Token = 1000 + 100*iota
107107
REG_RISCV_BEGIN
108108
REG_X64_BEGIN
109+
REG_ARM64_BEGIN
109110

110111
REG_BEGIN = REG_LOONG_BEGIN
111112
REG_LOONG_END = REG_LOONG_BEGIN + 100
112113
REG_RISCV_END = REG_RISCV_BEGIN + 100
113114
REG_X64_END = REG_X64_BEGIN + 100
114-
REG_END = REG_X64_END
115+
REG_ARM64_END = REG_ARM64_BEGIN + 100
116+
REG_END = REG_ARM64_END
115117
)
116118

117119
// 指令到 Token 空间的映射
@@ -121,12 +123,14 @@ const (
121123
A_LOONG_BEGIN Token = 2000 + 2000*iota
122124
A_RISCV_BEGIN
123125
A_X64_BEGIN
126+
A_ARM64_BEGIN
124127
A_END
125128

126129
A_BEGIN = A_LOONG_BEGIN
127130
A_LOONG_END = A_LOONG_BEGIN + 2000
128131
A_RISCV_END = A_RISCV_BEGIN + 2000
129132
A_X64_END = A_X64_BEGIN + 2000
133+
A_ARM64_END = A_ARM64_BEGIN + 2000
130134
)
131135

132136
var tokens = [...]string{
@@ -287,6 +291,9 @@ func (tok Token) RawReg() abi.RegType {
287291
if REG_X64_BEGIN <= tok && tok < REG_X64_END {
288292
return abi.RegType(tok - REG_X64_BEGIN)
289293
}
294+
if REG_ARM64_BEGIN <= tok && tok < REG_ARM64_END {
295+
return abi.RegType(tok - REG_ARM64_BEGIN)
296+
}
290297
return 0
291298
}
292299

@@ -301,6 +308,9 @@ func (tok Token) RawAs() abi.As {
301308
if A_X64_BEGIN <= tok && tok < A_X64_END {
302309
return abi.As(tok - A_X64_BEGIN)
303310
}
311+
if A_ARM64_BEGIN <= tok && tok < A_ARM64_END {
312+
return abi.As(tok - A_ARM64_BEGIN)
313+
}
304314
return 0
305315
}
306316

0 commit comments

Comments
 (0)