from-scratch/stages/hex1/hex1.asm
2026-06-05 04:32:19 +02:00

320 lines
8.4 KiB
NASM

;; This is the 2nd stage of the hex compiler.
;; It still requires hex input, but it can automatically calculate rip-relative offsets to labels
;; we need the following variables:
;; +0 = buf: [u8; 0x1000]
;; +0x1000 = write_buf: [u8; 0x4000]
;; +0x5000 = read_cursor: u32
;; +0x5004 write_cursor: u32
;; +0x5008 read_end: u32
;; +0x500c file_offset: u32 (the offset of the current file in the overall input, used for calculating label offsets)
;; +0x5010 labels: [(u64, u32); 340]
;; +0x6000 num_labels: u32
;; +0x6004 scratch space
section .data
_start_lbl db "_start", 0
section .text
;; exits the program
panic:
mov rax, 60 ; syscall: exit
mov rdi, 1 ; status: 1
syscall
;; exits the program
die:
mov rax, 60 ; syscall: exit
mov rdi, 0 ; status: 0
syscall
;; calculates a quick hash of a byte sequence given by rdi (pointer) and rsi (length)
fasthash:
mov eax, 0x811c9dc5 ; FNV offset basis
.fashhash_loop:
test rsi, rsi
jz .fashhash_exit
movzx ecx, byte [rdi] ; get next byte
xor eax, ecx
imul eax, 0x1000193 ; FNV prime
inc rdi
dec rsi
jmp .fashhash_loop
.fashhash_exit:
ret
;; push the label pointed at by rdi with length rsi, and the current file offset onto the labels array.
push_lbl:
mov eax, dword [r15 + 0x6000]
cmp eax, 340
jge panic
lea edx, [rax + rax*2]
shl rdx, 2
push rdx
call fasthash
pop rdx ; rdx = offset of label entry in labels array
lea rdi, [r15 + 0x5010]
mov qword [rdi + rdx], rax ; store hash in labels array
mov eax, dword [r15 + 0x500c] ; get current file offset
mov dword [rdi + rdx + 8], eax ; store file offset in labels array
inc dword [r15 + 0x6000]
ret
;; searches for the offset of the label with the name pointed at by rdi with length rsi, and returns it in rax. If the label is not found, dies.
find_lbl:
call fasthash
lea rdx, [r15 + 0x5010] ; rdx = pointer to start of labels array
mov ebx, dword [r15 + 0x6000]
.find_lbl_loop:
test rbx, rbx
jz .find_lbl_not_found
dec rbx
cmp qword [rdx], rax
je .find_lbl_found
add rdx, 12 ; move to next label entry
jmp .find_lbl_loop
.find_lbl_not_found:
call panic
.find_lbl_found:
mov eax, dword [rdx + 8] ; get file offset of label
ret
;; write up to write_cursor to stdout, and then reset the write cursor to 0
wrt_stdout:
mov rax, 1 ; syscall: write
mov rdi, 1 ; fd: stdout
lea rsi, [r15 + 0x1000] ; buf: pointer to write_buf
mov edx, dword [r15 + 0x5004] ; count: write_cursor
syscall
mov dword [r15 + 0x5004], 0
ret
die_jmp_start:
;; @_start_lbl: 5F 73 74 61 72 74 00 00
;; lea rdi, [@_start_lbl]
mov rdi, _start_lbl
mov rsi, 6
call find_lbl
add dword [r15 + 0x500c], 5 ; advance the file offset by 5
sub eax, dword [r15 + 0x500c] ; add current file offset
mov edi, dword [r15 + 0x5004] ; get current write cursor
lea rdi, [r15 + rdi + 0x1000] ; rdi = pointer to write location in write_buf
mov byte [rdi], 0xe9 ; write the opcode for jmp rel32
mov dword [rdi + 1], eax ; write the offset of the label reference
add dword [r15 + 0x5004], 5 ; advance the write cursor by 5
call wrt_stdout
call die
;; reads up to 0x1000 bytes from stdin into buf, and sets read_end to the number of bytes read, and resets read_cursor to 0
;; dies if no bytes were read (if it returns, there were bytes read)
rd_stdin:
mov rax, 0 ; syscall: read
mov rdi, 0 ; fd: stdin
mov rsi, r15
mov rdx, 0x1000 ; count: size of buffer
syscall
test eax, eax
jle die_jmp_start
mov dword [r15 + 0x5008], eax ; store number of bytes read in read_end
mov dword [r15 + 0x5000], 0 ; reset read_cursor to 0
ret
;; return the next byte from the input
;; automatically writes to stdout and reads from stdin if we are at the end of the buffer
rd_buf:
mov eax, dword [r15 + 0x5000] ; get read_cursor
cmp eax, dword [r15 + 0x5008] ; compare to read_end
jl .rd_buf_end
call wrt_stdout
call rd_stdin
jmp .rd_buf_end
.rd_buf_end:
mov edx, dword [r15 + 0x5000] ; get read_cursor
xor rax, rax
mov al, byte [r15 + rdx] ; get next byte from buf
inc dword [r15 + 0x5000] ; advance read_cursor
ret
;; The grammar is slightly updated from hex0, as follows:
;; program = (label-def / instruction-seq / comment)*
;; comment = ";" [^\n]* "\n"
;; label-def = "@" label-name ":" comment?
;; instruction-seq = instruction+ comment?
;; instruction = label-ref / opcode
;; label-ref = "'" label-name
;; opcode = hex hex
;; label-name = [a-zA-Z0-9_-]+
;; hex = [0-9a-f][0-9a-f]
;;
;; whitespaces may occur between any tokens and are ignored.
;; whitespace = [ \t\n]+
skip_line:
cmp al, `\n`
jne .skip_line_read
call rd_buf
ret
.skip_line_read:
call rd_buf
jmp skip_line
skip_whitespaces:
cmp al, ' '
je .skip_whitespace_read
cmp al, `\t`
je .skip_whitespace_read
cmp al, `\n`
je .skip_whitespace_read
ret
.skip_whitespace_read:
call rd_buf
jmp skip_whitespaces
skip_comment:
cmp al, ";"
jne .skip_comment_ret
call skip_line
ret
.skip_comment_ret:
ret
get_next_token:
cmp al, ";"
je .get_next_token_skip
cmp al, ' '
je .get_next_token_skip
cmp al, `\t`
je .get_next_token_skip
cmp al, `\n`
je .get_next_token_skip
ret
.get_next_token_skip:
call skip_whitespaces
call skip_comment
jmp get_next_token
char_to_u4:
cmp al, '0'
jb .invalid_hex
cmp al, '9'
jbe .is_digit
cmp al, 'a'
jb .invalid_hex
cmp al, 'f'
jbe .is_lower_hex
cmp al, 'A'
jb .invalid_hex
cmp al, 'F'
jbe .is_upper_hex
jmp .invalid_hex
.invalid_hex:
call panic
.is_digit:
sub al, '0'
ret
.is_lower_hex:
sub al, 'a' - 10
ret
.is_upper_hex:
sub al, 'A' - 10
ret
octet:
sub rsp, 8
call char_to_u4
mov byte [rsp], al
call rd_buf
call char_to_u4
mov bl, byte [rsp]
shl bl, 4
or al, bl
mov edi, dword [r15 + 0x5004] ; get current write cursor
lea rdx, [r15 + 0x1000]
mov byte [rdi + rdx], al ; write the byte to the write_buf
inc dword [r15 + 0x5004] ; advance the write cursor by 1
inc dword [r15 + 0x500c] ; advance file offset by 1
add rsp, 8
ret
label_ref:
cmp al, "'"
jne .not_label_ref
xor rsi, rsi ; rsi = length of label
.label_ref_loop:
push rsi
call rd_buf
pop rsi
lea rdi, [r15 + 0x6004] ; rdi = pointer to scratch
cmp al, ' '
je .label_ref_ret
cmp al, `\t`
je .label_ref_ret
cmp al, `\n`
je .label_ref_ret
mov byte [rdi + rsi], al
inc rsi
jmp .label_ref_loop
.label_ref_ret:
lea rdi, [r15 + 0x6004]
call find_lbl
add dword [r15 + 0x500c], 4 ; advance the file offset by 4
sub eax, dword [r15 + 0x500c] ; add current file offset
mov edi, dword [r15 + 0x5004] ; get current write cursor
mov dword [r15 + 0x1000 + rdi], eax ; write the opcode for jmp rel32
add dword [r15 + 0x5004], 4 ; advance the write cursor by 4
ret
.not_label_ref:
call octet
ret
;; read the label into scratch, call push_lbl
label:
cmp al, '@'
jne .not_label
xor rsi, rsi ; rsi = length of label
.label_loop:
push rsi
call rd_buf
pop rsi
lea rdi, [r15 + 0x6004] ; rdi = pointer to scratch
cmp al, ':'
je .label_ret
mov byte [rdi + rsi], al
inc rsi
jmp .label_loop
.label_ret:
lea rdi, [r15 + 0x6004] ; rdi = pointer to scratch
call push_lbl
ret
.not_label:
call label_ref
ret
global _start
_start:
push rbp
mov rbp, rsp
sub rsp, 0x8000 ; allocate space for our variables + some scratch space
mov r15, rsp ; r15 will point to the start of our variables
mov dword [r15 + 0x6000], 0 ; num_labels = 0
mov dword [r15 + 0x500c], 0 ; file_offset = 0
mov dword [r15 + 0x5008], 0 ; read_end = 0
mov dword [r15 + 0x5004], 0 ; write_cursor = 0
mov dword [r15 + 0x5000], 0 ; read_cursor = 0
.loop:
call rd_buf
call get_next_token
call label
jmp .loop