;; This assembler is a 2-pass compiler, permitting forward and backward label references. ;; It's also the first stage in this bootstrapping experiment that will ;; consume proper mnemonic assembly, rather than hex input. section .bss buf resb 0x100 labels resb 0x1000 section .data _start_lbl db "_start", 0 ;; supported instructions: ;; mov ;; test ;; xor ;; and ;; shl ;; shr ;; inc ;; dec ;; neg ;; ret ;; not ;; add ;; sub ;; imul ;; cmp ;; lea ;; push, pop ;; cmovcc ;; xchg ;; bts ;; bsf ;; bt ;; syscall ;; jcc rel32 ;; jmp rel32 ;; call rel32 section .text instr_table: ; mnemonic, num_operands, handler _mov: dq "mov" db 2 dq instr_table - _mov_handler ;; calculates a quick hash of a byte sequence given by rdi (pointer) and rsi (length) fasthash: mov eax, 0x811c9dc5 ; FNV offset basis .fashhash_loop: test rsi, rsi jz .fashhash_exit movzx ecx, byte [rdi] ; get next byte xor eax, ecx imul eax, 0x1000193 ; FNV prime inc rdi dec rsi jmp .fashhash_loop .fashhash_exit: ret ;; push the label pointed at by rdi with length rsi, and the current file offset onto the labels array. push_lbl: push r12 lea r12, [rel labels] mov rax, dword [r12 + 0x1000 - 4] ; get current label count cmp rax, 341 jge panic lea rax, [rax + rax*2] shl rax, 2 ; $rax = num_labels + size_of::() add r12, rax call fasthash mov qword [r12], rax ; store hash in labels array mov dword [r12 + 8], -1 ; initialize file offset to -1 (invalid) inc dword [r12 + 0x1000 - 4] ; increment label count pop r12 ret ;; searches for the offset of the label with the name pointed at by rdi with length rsi, and returns it in rax. If the label is not found, dies. find_lbl_ptr: call fasthash lea rdx, [r15 + 0x5010] ; rdx = pointer to start of labels array mov ebx, dword [r15 + 0x6000] .find_lbl_loop: test rbx, rbx jz .find_lbl_not_found dec rbx cmp qword [rdx], rax je .find_lbl_found add rdx, 12 ; move to next label entry jmp .find_lbl_loop push_or_find_lbl: call fasthash push rax mov rdi, rax call find_lbl_ptr test rax, rax jz .push add rsp, 8 ret .push: pop rdi call push_lbl ret .find_lbl_not_found: xor rax, rax ret .find_lbl_found: mov rax, rdx ret find_lbl_offset: call find_lbl_ptr test rax, rax jz .not_found mov eax, dword [rax + 8] ; get file offset of label ret .not_found: xor rdi, rdi call panic_abort set_lbl_offset: push rsi call find_lbl_ptr pop rsi mov dword [rax + 8], esi ; set file offset of label ret is_alpha: movzx edi, dil and edi, 2097119 add edi, -65 cmp edi, 26 setb al ret ;; returns 1 if $dil is a valid identifier character (alphanumeric, '-' or '_'), and 0 otherwise. ;; clobbers rdi, rax, rcx is_id_cont: movzx rdi, dil lea eax, [rdi - 48] cmp eax, 10 setb al mov ecx, edi and ecx, 2097119 add ecx, -65 cmp ecx, 26 setb cl or al, cl cmp edi, 45 ; '-' sete cl or al, cl cmp edi, 95 ; '_' sete cl or al, cl ret is_digit: movzx edi, dil add edi, -48 cmp edi, 10 setb al ret ;; converts char $dil to a digit with radix $rsi, returning it in $edx. $al is set to 1 if the char is a valid digit, and 0 otherwise. to_digit: lea eax, [rsi - 2] cmp eax, 35 jae .invalid movzx rdi, dil lea edx, [rdi - 65] ; 'A' = 65 and edx, -33 ; convert to uppercase add edx, 10 ; 'A' should map to 10 lea eax, [rdi - 48] ; '0' = 48 cmp esi, 11 cmovb edx, eax ; if radix <= 10, then take the difference from '0' cmp edi, 58 cmovb edx, eax ; or if char < '9', then take the difference from '0' xor eax, eax cmp edx, esi setb al ; al = edx < radix ret .invalid: xor eax, eax ret parse_num: sub rsp, 24 mov qword [rsp], 0 ; acc mov qword [rsp + 8], rdi ; source iterator mov dword [rsp + 16], 10 ; radix call peekc cmp al, `-` jne .skip_sign mov qword [rsp], -1 ; acc = -1 call consuming_peekc .skip_sign: cmp al, `0` jne .skip_radix call consuming_peekc cmp al, `x` jne .skip_radix mov dword [rsp + 16], 16 ; radix = 16 call consuming_peekc .skip_radix: mov dil, al call to_digit test al, al jz .done mov rax, qword [rsp] ; acc mov rsi, qword [rsp + 16] ; radix mov rcx, rdx imul rax, rsi add rax, rcx mov qword [rsp], rax ; acc = acc * radix + digit mov rdi, qword [rsp + 8] ; source iterator call consuming_peekc jmp .skip_radix .done: mov rax, qword [rsp] ; move the result into rax add rsp, 24 ret parse_label: push r12 lea r12, [rel buf] sub rsp, 8 mov qword [rsp], rdi ; source iterator call consuming_peekc mov dil, al call is_alpha test al, al jz .done mov byte [r12], al inc r12 mov rdi, qword [rsp] ; restore source iterator call consuming_peekc mov dil, al call is_id_cont test al, al jz .done jmp .loop .done: lea rdi, [rel buf] mov rsi, r12 sub rsi, rdi call push_or_find_lbl add rsp, 8 pop r12 ret ;; @param lhs: (rdi, rsi) ;; @param rhs: (rdx, rcx) ;; @return al strcmp: cmp rcx, rsi cmovb rsi, rcx ; if rhs is shorter, use its length for the loop xor eax, eax .strcmp_loop: cmp rsi, rax je .strcmp_equal movzx ecx, byte [rdx + rax] cmp byte [rdi + rax], cl lea rax, [rax + 1] je .strcmp_loop seta al ; al = lhs > rhs sbb al, 0 ; al = al - CF ret .strcmp_equal: xor eax, eax ret memcpy: xor rax, rax .memcpy_loop: cmp rax, rdx je .memcpy_done mov al, byte [rdi + rax] mov byte [rsi + rax], al lea rax, [rax + 1] jmp .memcpy_loop .memcpy_done: ret ;; read from input file read_file: ; ;; returns the next byte in the input stream without advancing the read position. ; peekc: ; ;; returns the next byte in the input stream and advances the read position. ; getc: extern panic_abort extern peekc extern getc extern consuming_peekc skip_whitespaces: push rdi call peekc .loop: cmp al, ' ' je .read cmp al, `\t` je .read cmp al, `\n` je .read pop rdi ret .read: call consuming_peekc jmp .loop ;; converts char $dil to a digit with radix $rsi, returning it in $edx. $al is set to 1 if the char is a valid digit, and 0 otherwise. to_digit: lea eax, [rsi - 2] cmp eax, 35 jae .invalid movzx rdi, dil lea edx, [rdi - 65] ; 'A' = 65 and edx, -33 ; convert to uppercase add edx, 10 ; 'A' should map to 10 lea eax, [rdi - 48] ; '0' = 48 cmp esi, 11 cmovb edx, eax ; if radix <= 10, then take the difference from '0' cmp edi, 58 cmovb edx, eax ; or if char < '9', then take the difference from '0' xor eax, eax cmp edx, esi setb al ; al = edx < radix ret .invalid: xor eax, eax ret ;; reads from $rdi and parses digits with radix $rsi until a non-digit is encountered. parse_digits: sub rsp, 24 mov qword [rsp + 16], rsi ; save the radix mov qword [rsp + 8], rdi ; save the source iterator mov qword [rsp], 0 .loop: mov rdi, qword [rsp + 8] ; restore the source iterator call peekc mov dil, al mov rsi, qword [rsp + 16] ; restore the radix call to_digit test al, al jz .done mov rax, qword [rsp] mov rsi, qword [rsp + 16] ; restore the radix imul rax, rsi add rax, rdx mov qword [rsp], rax mov rdi, qword [rsp + 8] ; restore the source iterator call getc jmp .loop .done: mov rax, [rsp] add rsp, 24 ret ;; enum Register { ; A = 0, ; B, ; C, ; D, ; Src, ; Dst, ; Sp, ; Bp, ; R8, ; R9, ; R10, ; R11, ; R12, ; R13, ; R14, ; R15, ;; } ;; parses a non-extended GPR operand (e.g. a register name that isn't r8-r15) ;; returns the register number in rax and the width in rdx ;; if $rdx == 0, then parsing failed (invalid register name) ;; $rdi = source ;; $rsi = prefix char (e.g. 'r' or 'e' or 0) global parse_gpr parse_gpr: sub rsp, 24 mov qword [rsp + 16], -1 ; register number, initialized to -1 (invalid) mov qword [rsp + 8], rdi ; source iterator mov qword [rsp], rsi ; prefix char call getc ; check for a,b,c,d,s movzx rax, al sub rax, 'a' ; convert 'a' to 0, 'b' to 1, ..., 's' to 4 cmp rax, 's' - 'a' ja .invalid ; if it's greater than 's', it's invalid mov rcx, 4 cmovz rax, rcx ; if it's 's', set rax to 4 cmp rax, 4 ja .invalid ; if it's greater than 4, it's invalid lea rcx, [rel .jt] ; jump table for a,b,c,d movsxd rdx, dword [rcx + rax * 4] ; get the offset of the handler if a-d add rcx, rdx ; calculate the address of the handler jmp rcx .suffix: mov byte [rsp + 16], al ; save the register number for a,b,c,d,s mov rdi, qword [rsp + 8] ; restore the source iterator call peekc cmp al, 'l' jne .not_l mov rdi, qword [rsp + 8] ; restore the source iterator call getc ; consume the suffix mov rdx, 1 ; width = 1 for al, bl, cl, dl, sil, dil jmp .done .not_l: ; x is only valid for a,b,c,d, not sp, bp, si or di mov rdx, 2 cmp byte [rsp + 16], 4 jae .done ; for sp, bp, si, di, the width is 2 if not 'l' cmp al, 'x' jne .not_x mov rdi, qword [rsp + 8] ; restore the source iterator call getc ; consume the suffix mov rdx, 2 ; width = 2 for ax, bx, cx, dx jmp .done .not_x: ; x is only valid for a,b,c,d cmp al, 'h' jne .invalid mov rdx, 3 ; width = 3 for ah, bh, ch, dh .done: movzx rax, byte [rsp + 16] ; move the register number into rax add rsp, 24 ret .invalid: xor rax, rax ; rax = 0 indicates invalid register xor rdx, rdx ; rdx = 0 indicates invalid register jmp .done .a: mov rax, 0 jmp .suffix .b: mov rdi, qword [rsp + 8] ; restore the source iterator call peekc cmp al, 'p' mov rax, 1 jne .suffix mov rdi, qword [rsp + 8] ; restore the source iterator call getc mov rax, 5 jmp .suffix .c: mov rax, 2 jmp .suffix .d: mov rdi, qword [rsp + 8] ; restore the source iterator call peekc cmp al, 'i' mov rax, 3 jne .suffix mov rdi, qword [rsp + 8] ; restore the source iterator call getc mov rax, 7 jmp .suffix .s: mov rdi, qword [rsp + 8] ; restore the source iterator call getc mov cl, al cmp cl, 'i' mov rax, 6 je .suffix cmp cl, 'p' mov rax, 4 je .suffix jmp .invalid .jt: dd .a-.jt dd .b-.jt dd .c-.jt dd .d-.jt dd .s-.jt ;; parses an extended GPR operand (e.g. r8-r15) ;; $rdi = source, leading r is already consumed parse_egpr: sub rsp, 16 mov qword [rsp + 8], rdi ; source iterator mov rsi, 10 call parse_digits cmp rax, 8 jb .invalid cmp rax, 15 ja .invalid mov qword [rsp], rax .suffix: mov rdi, qword [rsp + 8] ; restore the source iterator call peekc mov rcx, 1 cmp al, 'b' je .done_getc mov rcx, 2 cmp al, 'w' je .done_getc mov rcx, 4 cmp al, 'd' je .done_getc mov rcx, 8 ; default width is 8 for r8-r15 jmp .done .done_getc: mov rdi, qword [rsp + 8] ; restore the source iterator push rcx call getc pop rcx .done: mov rdx, rcx mov rax, qword [rsp] ; move the register number into rax add rsp, 16 ret .invalid: mov qword [rsp], 0 xor rcx, rcx ; rdx = 0 indicates invalid register jmp .done ;; (rdi, rsi): source string global parse_reg parse_reg: sub rsp, 16 mov qword [rsp + 8], rdi ; source iterator mov byte [rsp], 0 ; x1 call peekc cmp al, 'r' mov rdi, qword [rsp + 8] ; restore the source iterator je .r cmp al, 'e' jne .gpr mov byte [rsp], 1 ; x2 call getc ; consume the 'e' mov rdi, qword [rsp + 8] ; restore the source iterator jmp .gpr .r: mov byte [rsp], 2 ; x4 call getc ; consume the 'r' mov rdi, qword [rsp + 8] ; restore the source iterator call peekc mov dil, al mov rsi, 10 call to_digit test al, al mov rdi, qword [rsp + 8] ; restore the source iterator jz .gpr call parse_egpr jmp .done .gpr: call parse_gpr mov cl, byte [rsp] shl rdx, cl .done: add rsp, 16 ret global parse_packed_reg parse_packed_reg: call parse_reg test rdx, rdx jz .invalid shl rdx, 4 or rax, rdx ret .invalid: xor rax, rax ret ;; enum Operand { ;; Reg(Reg, width: u8), ;; Imm(u32, width: u8), ;; Mem(width: u8, base: Option, index: Option<(Reg, scale: u8)>, disp: Option), ;; } ;; operands can be registers, memory operands, immediates or labels ;; labels start with a ' ;; immediates start with a digit or a '-' ;; memory operands may have the forms: [