386 lines
8.6 KiB
NASM
386 lines
8.6 KiB
NASM
section .bss
|
|
token_buf resb 0x100
|
|
buf resb 0x100
|
|
labels resb 0x1000
|
|
|
|
section .text
|
|
extern panic_abort
|
|
;; returns the next character from the source without advancing.
|
|
extern peekc
|
|
;; returns the next character from the source and advances.
|
|
extern getc
|
|
;; advances past the next character and peeks the next.
|
|
extern consuming_peekc
|
|
|
|
;; calculates a quick hash of a byte sequence given by rdi (pointer) and rsi (length)
|
|
fasthash:
|
|
mov eax, 0x811c9dc5 ; FNV offset basis
|
|
.loop:
|
|
test rsi, rsi
|
|
jz .done
|
|
movzx ecx, byte [rdi] ; get next byte
|
|
xor eax, ecx
|
|
imul eax, 0x1000193 ; FNV prime
|
|
inc rdi
|
|
dec rsi
|
|
jmp .loop
|
|
.done:
|
|
ret
|
|
|
|
;; @param lhs: (rdi, rsi)
|
|
;; @param rhs: (rdx, rcx)
|
|
;; @return al
|
|
strcmp:
|
|
cmp rcx, rsi
|
|
cmovb rsi, rcx ; if rhs is shorter, use its length for the loop
|
|
xor eax, eax
|
|
.loop:
|
|
cmp rsi, rax
|
|
je .equal
|
|
movzx ecx, byte [rdx + rax]
|
|
cmp byte [rdi + rax], cl
|
|
lea rax, [rax + 1]
|
|
je .loop
|
|
seta al ; al = lhs > rhs
|
|
sbb al, 0 ; al = al - CF
|
|
ret
|
|
.equal:
|
|
xor eax, eax
|
|
ret
|
|
|
|
memcpy:
|
|
xor rax, rax
|
|
.loop:
|
|
cmp rax, rdx
|
|
je .done
|
|
mov al, byte [rdi + rax]
|
|
mov byte [rsi + rax], al
|
|
lea rax, [rax + 1]
|
|
jmp .loop
|
|
.done:
|
|
ret
|
|
|
|
;; returns 1 if the result of `peekc()` is $sil
|
|
;; treats all characters less than ' ' as spaces.
|
|
is_ch:
|
|
push rsi
|
|
call peekc
|
|
pop rsi
|
|
cmp al, ' '
|
|
setbe cl ; cl = al <= ' '
|
|
movzx ecx, cl
|
|
mov edx, ' '
|
|
mul ecx, edx ; cl = (al < ' ') ? ' ' : 0
|
|
cmp sil, ' '
|
|
cmove ax, cx ; al = (sil < ' ') ? cl : al
|
|
cmp al, sil
|
|
setz al ; al = al == sil
|
|
ret
|
|
|
|
;; struct packed LabelEntry {
|
|
;; hash: u64,
|
|
;; file_offset: Option<NonMaxU32>,
|
|
;; }
|
|
;; push the label pointed at by rdi with length rsi, and the current file offset onto the labels array.
|
|
push_lbl:
|
|
push r12
|
|
lea r12, [rel labels]
|
|
mov rax, dword [r12 + 0x1000 - 4] ; get current label count
|
|
cmp rax, 341
|
|
jge panic_abort
|
|
lea rax, [rax + rax*2]
|
|
shl rax, 2 ; $rax = num_labels + size_of::<label_entry>()
|
|
add r12, rax
|
|
call fasthash
|
|
mov qword [r12], rax ; store hash in labels array
|
|
mov dword [r12 + 8], -1 ; initialize file offset to -1 (invalid)
|
|
inc dword [r12 + 0x1000 - 4] ; increment label count
|
|
pop r12
|
|
ret
|
|
|
|
;; searches for the offset of the label with the name pointed at by rdi with length rsi, and returns it in rax. If the label is not found, returns 0 in rax.
|
|
find_lbl_ptr:
|
|
call fasthash
|
|
lea rdx, [r15 + 0x5010] ; rdx = pointer to start of labels array
|
|
mov ebx, dword [r15 + 0x6000]
|
|
.loop:
|
|
test rbx, rbx
|
|
jz .not_found
|
|
dec rbx
|
|
cmp qword [rdx], rax
|
|
je .found
|
|
add rdx, 12 ; move to next label entry
|
|
jmp .loop
|
|
.found:
|
|
mov rax, rdx
|
|
ret
|
|
.not_found:
|
|
xor rax, rax
|
|
ret
|
|
|
|
push_or_find_lbl:
|
|
call fasthash
|
|
push rax
|
|
mov rdi, rax
|
|
call find_lbl_ptr
|
|
test rax, rax
|
|
jz .push
|
|
add rsp, 8
|
|
ret
|
|
.push:
|
|
pop rdi
|
|
call push_lbl
|
|
ret
|
|
|
|
find_lbl_offset:
|
|
call find_lbl_ptr
|
|
test rax, rax
|
|
jz .not_found
|
|
mov eax, dword [rax + 8] ; get file offset of label
|
|
ret
|
|
.not_found:
|
|
xor rdi, rdi
|
|
call panic_abort
|
|
|
|
set_lbl_offset:
|
|
push rsi
|
|
call find_lbl_ptr
|
|
pop rsi
|
|
mov dword [rax + 8], esi ; set file offset of label
|
|
ret
|
|
|
|
is_alpha:
|
|
movzx edi, dil
|
|
and edi, 0x1fffdf ; ignore bit 5 (case)
|
|
add edi, -65 ; subtract 'A'
|
|
cmp edi, 26 ; check if in range 0-25
|
|
setb al
|
|
ret
|
|
|
|
is_digit:
|
|
movzx edi, dil
|
|
lea eax, [rdi - 48] ; subtract '0'
|
|
cmp eax, 10 ; check if in range 0-9
|
|
setb al
|
|
ret
|
|
|
|
;; returns 1 if $dil is a valid identifier character (alphanumeric, '-' or '_'), and 0 otherwise.
|
|
;; clobbers rdi, rax, rcx
|
|
is_id_cont:
|
|
movzx rdi, dil
|
|
; is_digit {
|
|
lea eax, [rdi - 48]
|
|
cmp eax, 10
|
|
setb al
|
|
; } || is_alpha {
|
|
mov ecx, edi
|
|
and ecx, 0x1fffdf
|
|
add ecx, -65
|
|
cmp ecx, 26
|
|
setb cl
|
|
; } is_alpha
|
|
or al, cl
|
|
cmp edi, 45 ; || '-'
|
|
sete cl
|
|
or al, cl
|
|
cmp edi, 95 ; || '_'
|
|
sete cl
|
|
or al, cl
|
|
ret
|
|
|
|
;; converts char $dil to a digit with radix $rsi, returning it in $edx.
|
|
;; $al is set to 1 if the char is a valid digit, and 0 otherwise.
|
|
to_digit:
|
|
lea eax, [rsi - 2]
|
|
cmp eax, 35
|
|
jae .invalid
|
|
movzx rdi, dil
|
|
lea edx, [rdi - 65] ; 'A' = 65
|
|
and edx, -33 ; convert to uppercase
|
|
add edx, 10 ; 'A' should map to 10
|
|
lea eax, [rdi - 48] ; '0' = 48
|
|
cmp esi, 11
|
|
cmovb edx, eax ; if radix <= 10, then take the difference from '0'
|
|
cmp edi, 58
|
|
cmovb edx, eax ; or if char < '9', then take the difference from '0'
|
|
xor eax, eax
|
|
cmp edx, esi
|
|
setb al ; al = edx < radix
|
|
ret
|
|
.invalid:
|
|
xor eax, eax
|
|
ret
|
|
|
|
;; reads from $rdi and parses digits with radix $rsi until a non-digit is encountered.
|
|
parse_digits:
|
|
sub rsp, 24
|
|
mov qword [rsp + 16], rsi ; save the radix
|
|
mov qword [rsp + 8], rdi ; save the source iterator
|
|
mov qword [rsp], 0
|
|
.loop:
|
|
mov rdi, qword [rsp + 8] ; restore the source iterator
|
|
call peekc
|
|
mov dil, al
|
|
mov rsi, qword [rsp + 16] ; restore the radix
|
|
call to_digit
|
|
test al, al
|
|
jz .done
|
|
mov rax, qword [rsp]
|
|
mov rsi, qword [rsp + 16] ; restore the radix
|
|
imul rax, rsi
|
|
add rax, rdx
|
|
mov qword [rsp], rax
|
|
mov rdi, qword [rsp + 8] ; restore the source iterator
|
|
call getc
|
|
jmp .loop
|
|
|
|
.done:
|
|
mov rax, [rsp]
|
|
add rsp, 24
|
|
ret
|
|
|
|
parse_num:
|
|
sub rsp, 24
|
|
mov qword [rsp], 0 ; acc
|
|
mov qword [rsp + 8], rdi ; source iterator
|
|
mov dword [rsp + 16], 10 ; radix
|
|
call peekc
|
|
cmp al, `-`
|
|
jne .skip_sign
|
|
mov qword [rsp], -1 ; acc = -1
|
|
call consuming_peekc
|
|
.skip_sign:
|
|
cmp al, `0`
|
|
jne .skip_radix
|
|
call consuming_peekc
|
|
cmp al, `x`
|
|
jne .skip_radix
|
|
mov dword [rsp + 16], 16 ; radix = 16
|
|
call consuming_peekc
|
|
.skip_radix:
|
|
mov dil, al
|
|
call to_digit
|
|
test al, al
|
|
jz .done
|
|
mov rax, qword [rsp] ; acc
|
|
mov rsi, qword [rsp + 16] ; radix
|
|
mov rcx, rdx
|
|
imul rax, rsi
|
|
add rax, rcx
|
|
mov qword [rsp], rax ; acc = acc * radix + digit
|
|
mov rdi, qword [rsp + 8] ; source iterator
|
|
call consuming_peekc
|
|
jmp .skip_radix
|
|
.done:
|
|
mov rax, qword [rsp] ; move the result into rax
|
|
add rsp, 24
|
|
ret
|
|
|
|
parse_label:
|
|
push r12
|
|
lea r12, [rel buf]
|
|
sub rsp, 8
|
|
mov qword [rsp], rdi ; source iterator
|
|
call consuming_peekc
|
|
mov dil, al
|
|
call is_alpha
|
|
test al, al
|
|
jz .done
|
|
mov byte [r12], al
|
|
inc r12
|
|
mov rdi, qword [rsp] ; restore source iterator
|
|
call consuming_peekc
|
|
mov dil, al
|
|
call is_id_cont
|
|
test al, al
|
|
jz .done
|
|
jmp .loop
|
|
.done:
|
|
lea rdi, [rel buf]
|
|
mov rsi, r12
|
|
sub rsi, rdi
|
|
call push_or_find_lbl
|
|
add rsp, 8
|
|
pop r12
|
|
ret
|
|
|
|
;; reads the next token into `token_buf`
|
|
next_token:
|
|
sub rsp, 8
|
|
mov qword [rsp], 0
|
|
.skip_spaces:
|
|
mov rsi, ' '
|
|
call is_ch
|
|
test al, al
|
|
jz .skipped_spaces:
|
|
call getc
|
|
jmp .skip_spaces
|
|
.skipped_spaces:
|
|
; leading [ is a memory access
|
|
mov rsi, `[`
|
|
call is_ch
|
|
test al, al
|
|
jnz .mem_access
|
|
; leading @ is label-def
|
|
mov rsi, '@'
|
|
call is_ch
|
|
test al, al
|
|
jnz .label_def
|
|
; leading ' is a label-ref
|
|
mov rsi, `'`
|
|
call is_ch
|
|
test al, al
|
|
jnz .label_ref
|
|
; leading digit is a number
|
|
call peekc
|
|
mov rdi, rax
|
|
call is_digit
|
|
test al, al
|
|
jnz .number
|
|
; leading ; or # is a comment
|
|
mov rsi, `;`
|
|
call is_ch
|
|
test al, al
|
|
jnz .comment
|
|
mov rsi, `#`
|
|
call is_ch
|
|
test al, al
|
|
jnz .comment
|
|
; leading , is a separator
|
|
mov rsi, `,`
|
|
call is_ch
|
|
test al, al
|
|
jz .ident
|
|
; just skip the comma since its not important for parsing
|
|
call getc
|
|
jmp .skip_spaces
|
|
.ident:
|
|
; now we have some identifier
|
|
|
|
; (mov rax (byte rdi))
|
|
|
|
|
|
|
|
;; registers:
|
|
;; r0/rax
|
|
;; r1/rcx
|
|
;; r2/rdx
|
|
;; r3/rbx
|
|
;; r4/rsp
|
|
;; r5/rbp
|
|
;; r6/rsi
|
|
;; r7/rdi
|
|
;; r8
|
|
;; r9
|
|
;; r10
|
|
;; r11
|
|
;; r12
|
|
;; r13
|
|
;; r14
|
|
;; r15
|
|
;;
|
|
;; registers of the form `rNUM` may have an optional suffix b, w, d, or q
|
|
;; legacy registers may also be referenced by their 8-bit, 16-bit, or 32-bit names (e.g. al, ax, eax, rax or spl, sp, esp, rsp)
|
|
;; high-byte registers (ah, bh, ch, dh) are not supported
|
|
|