Tag: assembly

  • chastdin calculator for RISC-V Assembly

    I really did it this time. I translated all of my Intel Assembly functions into RISC-V Assembly and rebuilt my calculator. It works flawlessly running under the rars simulator.

    To run this example, you need the RARS java archive and a java runtime environment installed on your machine.

    You can get RARS here:

    https://github.com/rarsm/rars

    However, once you do, you can run my program with a command like the following.

    java -jar ~/rars.jar main.s
    
    # chastelib test suite for RISC-V Assembly in RARS simulator
    
    # this program tests the stdin extension of chastelib
    
    # The same library of functions I commonly use in my Intel Assembly code
    # have now been translated to RISC-V.
    # All assembly code seen here is for the RARS simulator written in Java.
    
    .data
    
    ##################################################################
    # chastelib core specific variables                              #
    #                                                                #
    # These variables are used by the intstr function to convert an  #
    # integer to a string and what radix and widthshould be used     #
    # width means how many minimum digits including leading zeros    #
    ##################################################################
    
    int_string: .space 32 #reserve space for 32 bytes for up to 32 bits if printed in binary
    int_end: .byte 0 #the terminating zero of the integer string
    radix: .byte 2   #the radix the number will be shown in
    int_width: .byte 1 #by default
    
    # These variables are for outputting special strings
    # such as a newline, space, or a single character based on s0
    
    space: .byte 0x20, 0
    line:  .byte 0x0A, 0
    char:  .byte 0, 0 
    
    ##################################################################
    # chastdin specific variables                                    #
    #                                                                #
    # these variables are used as the default controllers            #
    # for the getstr and getline functions                           #
    # buf stores keyboard input during those functions               #
    # count stores how many bytes were read during system read calls #
    # last_char stores the last character read                       #
    # usually this will be a space, tab, or newline                  #
    ##################################################################
    
    buf: .space 0x100
    count: .word 0
    last_char: .byte 0
    
    # program specific variables
    # These variables are for outputting specific messages
    # or to simulate user input as integers in the strint function
    
    string0: .ascii "calculator for RISC-V Assembly\n"
    string1: .asciz "chastdin (Chastity's STanDard INput) extension\n\n"
    
    string_add: .asciz "add"
    string_sub: .asciz "sub"
    string_mul: .asciz "mul"
    string_div: .asciz "div"
    string_rem: .asciz "rem"
    string_setradix: .asciz "setradix"
    
    string_help: .asciz "help"
    string_exit: .asciz "exit"
    string_putstack: .asciz "?"
    string_clear: .asciz "clear"
    
    string_prompt: .asciz "->"
    
    string_err: .asciz "Error: invalid number or command: "
    string_err1: .asciz "Error: need one number on stack for command: "
    string_err2: .asciz "Error: need two numbers on stack for command: "
    
    chastdin_help: .ascii "chastdin is a stack based interactive calculator\n"
                  .ascii "that reads stdin for numbers and commands.\n"
                  .ascii "Numbers are pushed on the stack for all math.\n"
                  .ascii "Each line can contain multiple numbers or commands.\n\n"
                  .ascii "Arithmetic commands are add,sub,mul,div,rem\n"
                  .ascii "The exit command ends the program\n"
                  .ascii "The ? command prints the entire stack\n"
                  .asciz "The setradix command changes the radix for input and output\n"
    
    .align 2  # Aligns the next item to a 4-byte (2^2) word boundary
    chastack: .space 0x400 #reserve space for RPN calculator stack
    
    .text
    
    la s0, string0
    jal putstr
    
    # change radix for this program
    li t0, 10    #load t0 register with the new radix
    la t1, radix #load t1 register with the address the radix will go to
    sb t0, 0(t1) #save t0 register (byte) to address t1
    
    la s11, chastack #s11 will be used as the virtual stack pointer for this program
    
    #print the help message at the beginning of the program
    la s0, chastdin_help
    jal putstr
    
    #print the initial arrow prompt
    la s0, string_prompt
    jal putstr
    
    main_loop:
    
    la t1, last_char #load address of last_char
    lb t0, 0(t1)     #get the last character
    
    #show the arrow indicating we wait for the user to enter something
    #but only show it when the last character is a newline
    #otherwise it will print too many if multiple commands were entered on the same line
    li t1, 0xA
    bne t0, t1, skip_prompt
    la s0, string_prompt
    jal putstr
    skip_prompt:
    
    jal getstr  # read the string from standard input
    
    #load the length of string just entered from (count)
    la t1, count            #load address of count into t1
    lw t0, 0(t1)            #load number of chars read at (count) address
    beq t0, zero, main_loop #restart main_loop on empty string
    
    #jal putline # print extra line for readability
    #jal putstr # echo it to standard output
    #jal putline
    
    #s0 already contains string that was input
    #s1 will be loaded with address of exit string
    la s1, string_exit
    jal strcmp
    # end program if the string entered is equal to string_exit
    beq t0, zero, exit
    
    la s1, string_putstack
    jal strcmp
    beq t0, zero, command_putstack
    
    la s1, string_clear
    jal strcmp
    beq t0, zero, command_clear
    
    la s1, string_help
    jal strcmp
    beq t0, zero, command_help
    
    #next we begin checking for actual math commands of arithmetic
    
    la s1, string_add
    jal strcmp
    beq t0, zero, command_add
    
    la s1, string_sub
    jal strcmp
    beq t0, zero, command_sub
    
    la s1, string_mul
    jal strcmp
    beq t0, zero, command_mul
    
    la s1, string_div
    jal strcmp
    beq t0, zero, command_div
    
    la s1, string_rem
    jal strcmp
    beq t0, zero, command_rem
    
    la s1, string_setradix
    jal strcmp
    beq t0, zero, command_setradix
    
    
    #if the last string entered was not exit or a math command then
    #The default command is to turn the argument into a number and push to stack
    command_num:
    
    mv s1, s0              #back up this string address to s1 register
    jal strint             #try to get a number from the string pointed to by s0 register
    beq a0, zero, num_push #branch to number push if zero errors in integer string
    
    la s0, string_err    #load error message
    jal putstr           #print error message
    mv s0, s1            #load original command string
    jal putstr           #print which command failed
    jal putline
    j num_push_end       #skip the push because this can't be used
    
    num_push:            #push the number to the fake stack
    addi s11, s11, 4     #increment the pointer by the size of the native int for this mode
    sw s0, 0(s11)        #store the value we converted from the string with strint to this stack space
    num_push_end:
    j main_loop          #once value is pushed, continue the program
    
    exit:
    li a0, 0  #status
    li a7, 93 #exit
    ecall     #environment call
    
    #################################################################################
    # The following functions are used in the calculator program                    #
    # The all jump back to the main_loop after they are done                        #
    #                                                                               #
    #################################################################################
    
    #check if the stack has enough space for the last command
    #this will print an error if less than two numbers were on the stack
    #when using one of the math commands above
    
    memory_check:
    
    la s10, chastack        #load s10 with chastack address for branch comparison
    blt s10, s11, memory_ok # if s10 is less than s11, no errors
    
    print_stack_error:   #otherwise we print error message
    la s0, string_err2   #get error message for less than 2 numbers on stack
    jal putstr           #print error message
    mv s0, s1            #get name of the command used
    jal putstr           #print which command failed
    jal putline
    addi s11, s11, 4     #increment the pointer to what it was before the failed command
    j main_loop          #now go back to main loop after error was printed
    
    memory_ok:
    sw zero, 4(sp)       #if no error, erase the old top of stack by storing zero
    j main_loop          #and continue main_loop as normal
    
    
    command_putstack: #print all numbers on the stack
    la s9, chastack #load s9 with address of chastack
    mv s10, s11     #copy value of s11 to s10
    command_putstack_loop:
    
    #is s10 equal to the address of stack start?
    #if so, end the putstack loop
    beq s9, s10 command_putstack_end
    lw s0, 0(s10) #load the word at s10 into s0 for printing integer 
    addi s10, s10, -4 #subtract the word size from this temp stack index
    jal putint
    jal putline
    j command_putstack_loop
    command_putstack_end:
    j main_loop
    
    
    
    
    command_clear: #erase all numbers on the stack
    la s9, chastack #load s9 with address of chastack
    command_clear_loop:
    
    #is s11 equal to the address of stack start?
    #if so, end the clear loop
    beq s9, s11 command_clear_end
    sw zero, 0(s11) #store zero into the word at 0(s11) to erase it
    addi s11, s11, -4 #subtract the word size from this temp stack index
    j command_clear_loop
    command_clear_end:
    j main_loop
    
    
    command_help:
    la s0, chastdin_help
    jal putstr
    j main_loop
    
    #add number on top of stack to the one below it
    command_add:
    lw t1, 0(s11)     #load the word at this chastack address
    addi s11, s11, -4 #subtract the word size from s11
    lw t0, 0(s11)     #load the word at this chastack address
    add t0, t0, t1    #t0 = t0 + t1
    sw t0, 0(s11)     #save the word at this chastack address
    j memory_check    #check stack for errors after this command
    
    #add number on top of stack to the one below it
    command_sub:
    lw t1, 0(s11)     #load the word at this chastack address
    addi s11, s11, -4 #subtract the word size from s11
    lw t0, 0(s11)     #load the word at this chastack address
    sub t0, t0, t1    #t0 = t0 - t1
    sw t0, 0(s11)     #save the word at this chastack address
    j memory_check    #check stack for errors after this command
    
    #mul number on top of stack to the one below it
    command_mul:
    lw t1, 0(s11)     #load the word at this chastack address
    addi s11, s11, -4 #subtract the word size from s11
    lw t0, 0(s11)     #load the word at this chastack address
    mul t0, t0, t1    #t0 = t0 * t1
    sw t0, 0(s11)     #save the word at this chastack address
    j memory_check    #check stack for errors after this command
    
    #divide and store quotient on stack
    command_div:
    lw t1, 0(s11)     #load the word at this chastack address
    addi s11, s11, -4 #subtract the word size from s11
    lw t0, 0(s11)     #load the word at this chastack address
    divu t0, t0, t1    #t0 = t0 / t1
    sw t0, 0(s11)     #save the word at this chastack address
    j memory_check    #check stack for errors after this command
    
    #divide and store remainder on stack
    command_rem:
    lw t1, 0(s11)     #load the word at this chastack address
    addi s11, s11, -4 #subtract the word size from s11
    lw t0, 0(s11)     #load the word at this chastack address
    remu t0, t0, t1   #t0 = t0 % t1
    sw t0, 0(s11)     #save the word at this chastack address
    j memory_check    #check stack for errors after this command
    
    
    
    #pop top of stack and set the current radix to it
    #it has error checking and leaves the radix as is
    #unless at least one number is on the stack
    command_setradix:
    
    
    la s10, chastack     #load s10 with chastack address for branch comparison
    ble s11, s10, change_radix_no # if s11 is less than or equal to chastack address, branch to radix error
    change_radix_yes:
    lw t0, 0(s11)        #load t0 register with the new radix
    la t1, radix         #load t1 register with the address the radix will go to
    sb t0, 0(t1)         #save t0 register (byte) to address t1
    sw zero, 0(s11)      #erase the old top of stack by storing zero
    addi s11, s11, -4
    j main_loop          #and continue main_loop as normal
    change_radix_no:
    la s0,string_err1    #get error message for less than 1 numbers on stack
    jal putstr           #print error message
    mv s0, s1            #get name of the command used
    jal putstr           #print which command failed
    jal putline
    
    addi s11, s11, 4     #increment the pointer to what it was before the failed command
    j main_loop          #now go back to main loop after error was printed
    
    #################################################################################
    # The following functions are independent of a specific RISC-V Operating System #
    #                                                                               #
    # intstr = convert integer into a string ready for printing                     #
    # putint = prints integer using intstr and the OS specific putstr function      #
    # strint = convert string into an integer                                       #
    #                                                                               #
    # The s0 register is used for pass data in or out of these functions            #
    # See comments above those specific functions for full details                  #
    #################################################################################
    
    # The intstr function does several things at once and is the foundation for all integer output.
    # It uses the global radix variable to know which radix or number base to use when turning the integer to a string
    # It also uses the global int_width variable to determine how many leading zeros should be used for the string
    # The purpose of this is to make numbers look good when lined up when they are printed in a list.
    # radices 2 to 36 are supported. Digits higher than 9 will be capital letters
    
    intstr:
    
    la t1, radix     #load address of radix into t1
    lb t2, 0(t1)     #load value of radix into t2
    la t1, int_width #load address of width into t1
    lb t4, 0(t1)     #load value of int_width into t4
    li t3, 1         #load current number of digits, always 1
    
    la t1, int_end   #t1=address of terminating zero in string
    addi t1, t1, -1  #t1-- to go to lowest digit
    
    digits_start:
    
    remu t0, s0, t2  #t0=remainder of the previous division
    divu s0, s0, t2  #s0=s0/t2 (divide s0 by the radix value in t2)
    
    li t5, 10        #load t5 with 10 because RISC-V does not allow constants for branches
    
    blt t0, t5, decimal_digit
    bge t0, t5, hexadecimal_digit
    
    decimal_digit:   #we go here if it is only a digit 0 to 9
    
    addi t0, t0, 0x30
    
    j save_digit
    
    hexadecimal_digit:
    addi t0, t0, -10
    addi t0, t0, 0x41
    
    save_digit:
    sb t0, 0(t1)     #store byte from t0 at address t1
    beq s0, zero, intstr_end
    addi t1, t1, -1
    addi t3, t3, 1
    j digits_start
    
    intstr_end:
    
    li t0, 0x30
    prefix_zeros:
    bge t3, t4, end_zeros
    addi t1, t1, -1
    sb t0, 0(t1) # store byte from t0 at address t1
    addi t3, t3, 1
    j prefix_zeros
    end_zeros:
    
    mv s0, t1
    
    ret
    
    # this function calls intstr to convert the s0 register into a string
    # then it uses the system specific putstr call to print the string
    # it also uses the stack to save the value of s0 and ra (return address)
    # this way, s0 is restored to the value it had before this function
    # restoring ra is required because it is modified during calls to other functions
    
    putint:
    
    addi sp, sp, -8
    sw ra, 0(sp)
    sw s0, 4(sp)
    
    jal intstr
    jal putstr
    
    lw ra, 0(sp)
    lw s0, 4(sp)
    addi sp, sp, 8
    
    ret
    
    # strint takes the string at address pointed to by s0 register
    # and then loads the s0 register with an integer equivalent value
    # the a0 register is returned with the number of errors that happened
    # programs can use this to find if a user entered a valid number
    # number is intepreted according to the current radix
    
    strint:
    
    li a0, 0         #load zero into register for error counting
    
    la t1, radix     #load address of radix into t1
    lb t2, 0(t1)     #load value of radix into t2
    
    mv t1, s0        #copy string address from s0 to t1
    li s0, 0
    
    read_strint:
    lb t0, 0(t1)
    addi t1, t1, 1
    beq t0, zero, strint_end
    
    #if char is below '0' or above '9', it is outside the range of these and is not a digit
    li t5, 0x30
    blt t0, t5, not_digit
    li t5, 0x39
    blt t5, t0, not_digit
    
    #but if it is a digit, then correct and process the character
    is_digit:
    andi t0, t0, 0xF
    j process_char
    
    not_digit:
    #it isn't a digit, but it could be an alphabet character
    #which counts as a digit in a higher base
    
    # if char is below 'A' or above 'Z', it is outside the range of these and is not capital letter
    li t5, 0x41
    blt t0, t5, not_upper
    li t5, 0x5A
    blt t5, t0, not_upper
    
    is_upper:
    li t5, 0x41
    sub t0, t0, t5
    addi t0, t0, 10
    j process_char
    
    not_upper:
    
    # if char is below 'a' or above 'z', it is outside the range of these and is not lowercase letter
    li t5, 0x61
    blt t0, t5, not_lower
    li t5, 0x7A
    blt t5, t0, not_lower
    
    is_lower:
    li t5, 0x61
    sub t0, t0, t5
    addi t0, t0, 10
    j process_char
    
    not_lower:
    
    # if we have reached this point, result invalid and end function
    # this is only reached if the byte was not a valid digit or alphabet character
    j strint_end_error
    
    process_char:
    
    blt t2, t0 strint_end_error #if this value is above or equal to radix, it is too high despite being a valid digit/alpha
    
    mul s0, s0, t2 # multiply s0 by the radix
    add s0, s0, t0 # add the correct value of this digit
    
    j read_strint # jump back and continue the loop if nothing has exited it
    
    strint_end_error:  #we jump here if there was an error with one of the chars
    addi a0, a0, 1 #add 1 to the a0 register indicating an error occurred
    
    strint_end: #we jump here when no errors happened
    ret
    
    ###############################################################################
    # This putstr function is my most portable function for RISC-V simulators     #
    # It calculates the length of a zero terminated string before printing it     #
    # This is the same way used in my Intel Assembly programs for DOS and Linux   #
    # This function was written to operate the same in both RARS and riscemu      #
    ###############################################################################
    
    putstr:
    
    mv t1, s0                       # t1 will be used as an index register
    
    putstr_strlen_start:
    lb t0, 0(t1)                    # load byte into t0 from address of t1
    beq t0, zero, putstr_strlen_end # if t0==0, then we jump to the end of the loop.
    addi t1, t1, 1                  # go to next byte
    j putstr_strlen_start           # jump to start of the loop
    putstr_strlen_end:              
    
    li a0, 1                        # STDOUT file number
    mv a1, s0                       # address of string 
    sub a2, t1, s0                  # length of string
    li a7, 64                       # write call number
    ecall                           # environment call
    
    ret
    
    #############################################################################
    # The next four 3 functions print things to standard output                 #
    # All of them use the putstr function above to achieve the output           #
    # They use the stack to preserve the values of the s0 and t1 registers used #
    # They also use global variables in the data section                        #
    #############################################################################
    
    #the putchar function, which is named after the C language function of the same name
    #prints the lowest byte of the s0 register as a byte or character to standard output
    
    putchar:
    
    addi sp, sp, -12
    sw ra, 0(sp)
    sw s0, 4(sp)
    sw t1, 8(sp)
    
    la t1, char
    sb s0, 0(t1)
    la s0, char
    jal putstr
    
    lw ra, 0(sp)
    lw s0, 4(sp)
    lw t1, 8(sp)
    addi sp, sp, 12
    
    ret
    
    # the putspace function prints a space to standard output
    
    putspace:
    
    addi sp, sp, -8
    sw ra, 0(sp)
    sw s0, 4(sp)
    
    la s0, space
    jal putstr
    
    lw ra, 0(sp)
    lw s0, 4(sp)
    addi sp, sp, 8
    
    ret
    
    # the putline function prints a newline to standard output
    
    putline:
    
    addi sp, sp, -8
    sw ra, 0(sp)
    sw s0, 4(sp)
    
    la s0, line
    jal putstr
    
    lw ra, 0(sp)
    lw s0, 4(sp)
    addi sp, sp, 8
    
    ret
    
    ##########################################################################
    # chastdin extension functions                                           #
    #                                                                        #
    # all functions that deal with getting strings and characters from stdin #
    ##########################################################################
    
    # the getstr function will read a string into a buffer from stdin
    # and return it in the s0 register for printing with the putstr function
    # the (count) variable will also return the number of characters
    
    getstr:
    
    li t0, 0                        # use t0 register to track chars read
    la a1, buf                      # load address of buffer for read string
    li a2, 1                        # read only 1 byte for each env call
    
    getstr_chars:
    
    li a0, 0                        # STDIN file number
    li a7, 63                       # read call number
    ecall                           # environment call
    
    # Branch to label getstr_end if a0 is less than a2
    # a0 is the return value of this environment read call
    # as will be -1 on error or 1 if successful
    # because we read 1 character at a time
    
    blt a0, a2, getstr_end
    
    # if no error, test range of the last byte
    
    lb t1, 0(a1)      #load byte at address (a1) into t1 register
    
    # if t1 is less than 0x21
    # or t1 is more than 0x7E
    # branch to function end because it is outside of print range
    
    li t2, 0x21
    blt t1, t2, getstr_end
    li t2, 0x7E
    blt t2, t1, getstr_end
    
    # otherwise, proceed to read more characters
    add t0, t0, a0    # add to read counter
    addi a1, a1, 1    # add 1 to buffer pointer register a1
    j getstr_chars # unconditional jump to getstr_chars
    
    getstr_end:
    
    la t2, count       #load address of count into t2
    sw t0, 0(t2)       #store number of chars read at (count) address
    la t2, last_char   #load address of last_char into t2
    sb t1, 0(t2)       #store last byte at (last_char) address
    sb zero, 0(a1)     #store byte zero to terminate string
    la s0, buf         #return address of buf in s0 register
    
    ret
    
    
    
    
    # the getline function will read a string into a buffer from stdin
    # and return it in the s0 register for printing with the putstr function
    # the (count) variable will also return the number of characters
    # this function will get the whole line including spaces
    
    getline:
    
    li t0, 0                        # use t0 register to track chars read
    la a1, buf                      # load address of buffer for read string
    li a2, 1                        # read only 1 byte for each env call
    
    getline_chars:
    
    li a0, 0                        # STDIN file number
    li a7, 63                       # read call number
    ecall                           # environment call
    
    # Branch to label getline_end if a0 is less than a2
    # a0 is the return value of this environment read call
    # as will be -1 on error or 1 if successful
    # because we read 1 character at a time
    
    blt a0, a2, getline_end
    
    # if no error, test range of the last byte
    
    lb t1, 0(a1)      #load byte at address (a1) into t1 register
    
    # if t1 is less than 0x20
    # or t1 is more than 0x7E
    # branch to function end because it is outside of print range
    
    li t2, 0x20
    blt t1, t2, getline_end
    li t2, 0x7E
    blt t2, t1, getline_end
    
    # otherwise, proceed to read more characters
    add t0, t0, a0    # add to read counter
    addi a1, a1, 1    # add 1 to buffer pointer register a1
    j getline_chars # unconditional jump to getline_chars
    
    getline_end:
    
    la t2, count       #load address of count into t2
    sw t0, 0(t2)       #store number of chars read at (count) address
    la t2, last_char   #load address of last_char into t2
    sb t1, 0(t2)       #store last byte at (last_char) address
    sb zero, 0(a1)     #store byte zero to terminate string
    la s0, buf         #return address of buf in s0 register
    
    ret
    
    
    
    # Short Description of strlen:
    # The strlen function gets the length of string in s0 and returns it in s0
    # This is the same algorithm used in my putstr function but is independent of an operating system.
    
    strlen:
    
    mv t1, s0                       # t1 will be used as an index register
    
    strlen_start:
    lb t0, 0(t1)                    # load byte into t0 from address of t1
    beq t0, zero, strlen_end        # if t0==0, then we jump to the end of the loop.
    addi t1, t1, 1                  # go to next byte
    j strlen_start                  # jump to start of the loop
    strlen_end:              
    
    sub s0, t1, s0                  # return length of string in s0
    
    ret
    
    
    # Short Description of strcmp:
    # strcmp compares the string at s0 to the one at s1
    # t0 returns 0 if the strings are the same and non zero if different
    # the algorithm is simple but I will explain it for those who are confused
    
    # Long Description of strcmp:
    # each byte from each string is loaded into the t0 and t1 registers
    # the bytes are compared. if they are different, then we jump to the end
    # However, if they are the same, then we check if one of them is zero
    # if it is zero, this also jumps to the end of the function
    # If neither jump took place, then we jump to the start of the loop
    # but when the function finally ends t1 will be subtracted from t0
    # this ensures that the t0 register returns zero if the final characters are the same
    # a zero result in t0 also guarantees that both strings are equal
    
    strcmp:
    
    mv a0, s0 # move pointer s0 to a0
    mv a1, s1 # move pointer s1 to a1
    
    strcmp_start:
    
    #read a byte from each string
    lb t0, 0(a0) 
    lb t1, 0(a1) 
    #if the two bytes are not equal end comparison
    bne t0, t1, strcmp_end
    
    #but if they are equal, test for zero
    #if one of them is zero, also end the loop
    beq t0, zero, strcmp_end
    
    addi a0, a0, 1                  # go to next byte
    addi a1, a1, 1                  # go to next byte
    
    j strcmp_start
    
    strcmp_end:
    
    #subtract t1 from t0
    #if t0 is still zero after the function returns
    #it means that the strings are equal
    sub t0, t0, t1
    
    ret
    
    
  • Learning POSIX System Calls

    I have been doing Assembly language programming for some time now, and yet only today did I take the time to read the documentation and some online examples to help me learn how to use the system calls from C programs.

    On Linux systems like the Debian one I use, there are documentation pages already installed. There are hundreds of them, and yet only 6 are required to create all of my command-line tools.

    The following commands can be used to read each one of the 6 fundamental system calls available on Linux and Unix systems for C programmers.

    Six Supreme System Calls

    man 2 open
    man 2 close
    man 2 read
    man 2 write
    man 2 lseek
    man 2 exit
    

    My command line utilities, I have been creating such as chastehex, chastecmp, and chastext used these system calls in their assembly versions. However, I traditionally used the C standard library for the C versions of these programs.

    But after reading about the system calls and seeing some examples, I realized that I could make copies and rewrite these tools by calling only system calls. During this process, I learned how much easier they are to use compared to the C library functions.

    Here is a summary of each of these functions and how they are used in my programs.

    All of my tools use “open” to open a file and then “close” to close it when I am done with it. There can be no confusion as to what these functions do because of their names.

    Similarly, “read” and “write” do exactly what their names imply. They operate the same as fread and fwrite do in stdio. But they take only 3 arguments instead of 4, which makes a lot of sense. You give them a pointer, and then you tell them exactly how many bytes you want to read from or write to a file descriptor previously assigned with “open”.

    The “lseek” function stands for long seek and is capable of moving to a different position in a file before the next read or write operation. Not every program needs this, but chastehex does because one of the arguments is an address in hexadecimal to read or write. Jumping around in a file is sometimes necessary if you are working with large files or the address matters a lot.

    The final call to any program is “exit” because it ends the program. There isn’t much to say about it except that it also lets you return a number to the operating system. Usually, 0 means no errors happened, and a value of anything else indicates a specific type of error you have defined in your program. All of my programs return 1 if a file could not be opened.

    Each of these functions has various arguments that have clearly defined meanings. The return values are also specified in their manual pages.

    Interestingly, these calls are available on every operating system that I know about except for Windows. However, considering how easy these are to implement using the C standard library, it would be possible to write Windows versions of these. In fact, some people have already done this.

    See the Cygwin and MinGW projects for more information about how to use these calls on Windows. For all other operating systems: Linux, Unix (FreeBSD, OpenBSD, NetBSD, Minix, and ChromiumOS) These calls are already available if you have a working C compiler.

    You might wonder why I spent the time learning and explaining this. It is because having a super small library of functions that I can memorize allows faster programming and less time spent looking at my references when I have forgotten which order the arguments go in.

    This knowledge gives me an alternative library of functions I can use that is easier than the C standard library. However, I am keeping both versions of every program I have written.

    But the final point I want to make is that because these are the same calls used in my Assembly programs, I can make C programs that map 1 to 1 when comparing and teaching Assembly in the books I write!

  • chastehex 1280 byte edition for Linux

    The following source code is a major update to chastehex for 32-bit Assembly source code for Linux. The behavior of the program hasn’t changed. It is still the great command line hex editor. However, the executable is a lot smaller than it previously was. I found some optimizations to reduce function calls and also removed some of the text while still having the messages say the same basic idea. This may not mean much to the average person but this is the best hand written assembly I have ever achieved and I made some extensions to chastelib that will be helpful for future programs.

    main.asm

    ;Linux 32-bit Assembly Source for chastehex
    ;a special tool originally written in C
    format ELF executable
    entry main
    
    start:
    
    include 'chastelib32.asm'
    
    main:
    
    ;radix will be 16 because this whole program is about hexadecimal
    mov dword [radix],16 ; can choose radix for integer input/output!
    
    pop eax
    mov [argc],eax ;save the argument count for later
    
    ;first arg is the name of the program. we skip past it
    pop eax
    dec dword [argc]
    
    ;before we try to get the first argument as a filename, we must check if it exists
    cmp dword [argc],0
    jnz arg_open_file
    
    help:
    mov eax,help_message
    call putstring
    jmp main_end
    
    arg_open_file:
    
    pop eax
    dec dword [argc]
    mov [filename],eax ; save the name of the file we will open to read
    call putstr_and_line
    
    ;Linux system call to open a file
    
    mov ecx,2   ;open file in read and write mode 
    mov ebx,eax ;filename should be in eax before this function was called
    mov eax,5   ;invoke SYS_OPEN (kernel opcode 5)
    int 80h     ;call the kernel
    
    cmp eax,0
    jns file_open_no_errors ;if eax is not negative/signed there was no error
    
    ;Otherwise, if it was signed, then this code will display an error message.
    
    neg eax
    call putint_and_space
    mov eax,open_error_message
    call putstr_and_line
    
    jmp main_end ;end the program because we failed at opening the file
    
    file_open_no_errors:
    
    mov [filedesc],eax ; save the file descriptor number for later use
    mov dword [file_offset],0 ;assume the offset is 0,beginning of file
    
    ;check next arg
    cmp dword [argc],0 ;if there are no more args after filename, just hexdump it
    jnz next_arg_address ;but if there are more, jump to the next argument to process it as address
    
    hexdump:
    
    mov edx,0x10         ;number of bytes to read
    mov ecx,byte_array   ;address to store the bytes
    mov ebx,[filedesc]   ;move the opened file descriptor into EBX
    mov eax,3            ;invoke SYS_READ (kernel opcode 3)
    int 80h              ;call the kernel
    
    mov [bytes_read],eax
    
    cmp eax,0
    jnz file_success ;if more than zero bytes read, proceed to display
    
    ;display EOF to indicate we have reached the end of file
    
    mov eax,end_of_file_string
    call putstr_and_line
    
    jmp main_end
    
    ; this point is reached if file was read from successfully
    
    file_success:
    
    call print_bytes_row
    
    cmp dword [bytes_read],1 
    jl main_end ;if less than one bytes read, there is an error
    jmp hexdump
    
    ;address argument section
    next_arg_address:
    
    ;if there is at least one more arg
    pop eax ;pop the argument into eax and process it as a hex number
    dec dword [argc]
    call strint
    
    ;use the hex number as an address to seek to in the file
    mov edx,0          ;whence argument (SEEK_SET)
    mov ecx,eax        ;move the file cursor to this address
    mov ebx,[filedesc] ;move the opened file descriptor into EBX
    mov eax,19         ;invoke SYS_LSEEK (kernel opcode 19)
    int 80h            ;call the kernel
    
    mov [file_offset],eax ;move the new offset
    
    ;check the number of args still remaining
    cmp dword [argc],0
    jnz next_arg_write ; if there are still arguments, skip this read section and enter writing mode
    
    read_one_byte:
    mov edx,1          ;number of bytes to read
    mov ecx,byte_array ;address to store the bytes
    mov ebx,[filedesc] ;move the opened file descriptor into EBX
    mov eax,3          ;invoke SYS_READ (kernel opcode 3)
    int 80h            ;call the kernel
    
    ;eax will have the number of bytes read after system call
    cmp eax,1
    jz print_byte_read ;if exactly 1 byte was read, proceed to print info
    
    call show_eof
    
    jmp main_end ;go to end of program
    
    ;print the address and the byte at that address
    print_byte_read:
    call print_byte_info
    
    ;this section interprets the rest of the args as bytes to write
    next_arg_write:
    cmp dword [argc],0
    jz main_end
    
    pop eax
    dec dword [argc]
    call strint ;try to convert string to a hex number
    
    ;write that number as a byte value to the file
    
    mov [byte_array],al
    
    mov eax,4          ;invoke SYS_WRITE (kernel opcode 4 on 32 bit systems)
    mov ebx,[filedesc] ;write to the file (not STDOUT)
    mov ecx,byte_array ;pointer to temporary byte address
    mov edx,1          ;write 1 byte
    int 80h            ;system call to write the message
    
    call print_byte_info
    inc dword [file_offset]
    
    jmp next_arg_write
    
    main_end:
    
    ;this is the end of the program
    ;we close the open file and then use the exit call
    
    ;Linux system call to close a file
    
    mov ebx,[filedesc] ;file number to close
    mov eax,6          ;invoke SYS_CLOSE (kernel opcode 6)
    int 80h            ;call the kernel
    
    mov eax, 1  ; invoke SYS_EXIT (kernel opcode 1)
    mov ebx, 0  ; return 0 status on exit - 'No Errors'
    int 80h
    
    
    ;this function prints a row of hex bytes
    ;each row is 16 bytes
    print_bytes_row:
    mov eax,[file_offset]
    mov dword [int_width],8
    call putint_and_space
    
    mov ebx,byte_array
    mov ecx,[bytes_read]
    add [file_offset],ecx
    next_byte:
    mov eax,0
    mov al,[ebx]
    mov dword [int_width],2
    call putint_and_space
    
    inc ebx
    dec ecx
    cmp ecx,0
    jnz next_byte
    
    mov ecx,[bytes_read]
    pad_spaces:
    cmp ecx,0x10
    jz pad_spaces_end
    mov eax,space_three
    call putstring
    inc ecx
    jmp pad_spaces
    pad_spaces_end:
    
    ;optionally, print chars after hex bytes
    call print_bytes_row_text
    call putline
    
    ret
    
    space_three db '   ',0
    
    print_bytes_row_text:
    mov ebx,byte_array
    mov ecx,[bytes_read]
    next_char:
    mov eax,0
    mov al,[ebx]
    
    ;if char is below '0' or above '9', it is outside the range of these and is not a digit
    cmp al,0x20
    jb not_printable
    cmp al,0x7E
    ja not_printable
    
    printable:
    ;if char is in printable range,keep as is and proceed to next index
    jmp next_index
    
    not_printable:
    mov al,'.' ;otherwise replace with placeholder value
    
    next_index:
    mov [ebx],al
    inc ebx
    dec ecx
    cmp ecx,0
    jnz next_char
    mov [ebx],byte 0 ;make sure string is zero terminated
    
    mov eax,byte_array
    call putstring
    
    ret
    
    
    ;function to display EOF with address
    show_eof:
    
    mov eax,[file_offset]
    mov dword [int_width],8
    call putint_and_space
    mov eax,end_of_file_string
    call putstr_and_line
    
    ret
    
    ;print the address and the byte at that address
    print_byte_info:
    mov eax,[file_offset]
    mov dword [int_width],8
    call putint_and_space
    mov eax,0
    mov al,[byte_array]
    mov dword [int_width],2
    call putint_and_line
    
    ret
    
    end_of_file_string db 'EOF',0
    
    help_message db 'chastehex by Chastity White Rose',0Ah,0Ah
    db 'hexdump a file:',0Ah,0Ah,9,'chastehex file',0Ah,0Ah
    db 'read a byte:',0Ah,0Ah,9,'chastehex file address',0Ah,0Ah
    db 'write a byte:',0Ah,0Ah,9,'chastehex file address value',0Ah,0Ah
    db 'The file must exist',0Ah,0
    
    ;variables for managing arguments and files
    argc dd 0
    filename dd 0 ; name of the file to be opened
    filedesc dd 0 ; file descriptor
    bytes_read dd 0
    file_offset dd 0
    open_error_message db 'error while opening file',0
    
    ;where we will store data from the file
    byte_array db 17 dup '?'
    

    chastelib32.asm

    ; chastelib assembly header file for 32 bit Linux
    ; This file is where I keep the source of my most important Assembly functions
    ; These are my string and integer output and conversion routines.
    
    ; To simplify documentation. The Accumulator/Arithmetic register
    ; (ax,ebx,rax) depending on bit size shall be referred to as register A
    ; for the description of these core functions because the A register
    ; is treated special both by the Intel company and my code;
    
    ; putstring; Prints a zero terminated string from the address pointer to by A register.
    ; intstr;    Converts the number in A into a zero terminated string and points A to that address
    ; putint;    Prints the integer in A by calling intstr and then putstring.
    ; strint;    Converts the zero terminated string into an integer and sets A to that value
       
    ; Now, the source of the functions begins, with comments included for parts that I felt needed explanation.
    
    stdout dd 1 ; variable for standard output so that it can theoretically be redirected
    
    putstring:
    
    push eax
    push ebx
    push ecx
    push edx
    
    mov ebx,eax ; copy eax to ebx. ebx will be used as index to the string
    
    putstring_strlen_start: ; this loop finds the length of the string as part of the putstring function
    
    cmp [ebx],byte 0 ; compare byte at address ebx with 0
    jz putstring_strlen_end ; if comparison was zero, jump to loop end because we have found the length
    inc ebx
    jmp putstring_strlen_start
    
    putstring_strlen_end:
    sub ebx,eax ;subtract start pointer from current pointer to get length of string
    
    ;Write string using Linux Write system call.
    ;Reference for 32 bit x86 syscalls is below.
    ;https://www.chromium.org/chromium-os/developer-library/reference/linux-constants/syscalls/#x86-32-bit
    
    mov edx,ebx      ;number of bytes to write
    mov ecx,eax      ;pointer/address of string to write
    mov ebx,[stdout] ;write to the STDOUT file
    mov eax, 4       ;invoke SYS_WRITE (kernel opcode 4 on 32 bit systems)
    int 80h          ;system call to write the message
    
    pop edx
    pop ecx
    pop ebx
    pop eax
    
    ret ; this is the end of the putstring function return to calling location
    
    ; This is the location in memory where digits are written to by the intstr function
    ; The string of bytes and settings such as the radix and width are global variables defined below.
    
    int_string db 32 dup '?' ;enough bytes to hold maximum size 32-bit binary integer
    
    int_string_end db 0 ;zero byte terminator for the integer string
    
    radix dd 2 ;radix or base for integer output. 2=binary, 8=octal, 10=decimal, 16=hexadecimal
    int_width dd 8
    
    ;this function creates a string of the integer in eax
    ;it uses the above radix variable to determine base from 2 to 36
    ;it then loads eax with the address of the string
    ;this means that it can be used with the putstring function
    
    intstr:
    
    mov ebx,int_string_end-1 ;find address of lowest digit(just before the newline 0Ah)
    mov ecx,1
    
    digits_start:
    
    mov edx,0;
    div dword [radix]
    cmp edx,10
    jb decimal_digit
    jae hexadecimal_digit
    
    decimal_digit: ;we go here if it is only a digit 0 to 9
    add edx,'0'
    jmp save_digit
    
    hexadecimal_digit:
    sub edx,10
    add edx,'A'
    
    save_digit:
    
    mov [ebx],dl
    cmp eax,0
    jz intstr_end
    dec ebx
    inc ecx
    jmp digits_start
    
    intstr_end:
    
    prefix_zeros:
    cmp ecx,[int_width]
    jnb end_zeros
    dec ebx
    mov [ebx],byte '0'
    inc ecx
    jmp prefix_zeros
    end_zeros:
    
    mov eax,ebx ; now that the digits have been written to the string, display it!
    
    ret
    
    ; function to print string form of whatever integer is in eax
    ; The radix determines which number base the string form takes.
    ; Anything from 2 to 36 is a valid radix
    ; in practice though, only bases 2,8,10,and 16 will make sense to other programmers
    ; this function does not process anything by itself but calls the combination of my other
    ; functions in the order I intended them to be used.
    
    putint: 
    
    push eax
    push ebx
    push ecx
    push edx
    
    call intstr
    
    call putstring
    
    pop edx
    pop ecx
    pop ebx
    pop eax
    
    ret
    
    ;this function converts a string pointed to by eax into an integer returned in eax instead
    ;it is a little complicated because it has to account for whether the character in
    ;a string is a decimal digit 0 to 9, or an alphabet character for bases higher than ten
    ;it also checks for both uppercase and lowercase letters for bases 11 to 36
    ;finally, it checks if that letter makes sense for the base.
    ;For example, G to Z cannot be used in hexadecimal, only A to F can
    ;The purpose of writing this function was to be able to accept user input as integers
    
    strint:
    
    mov ebx,eax ;copy string address from eax to ebx because eax will be replaced soon!
    mov eax,0
    
    read_strint:
    mov ecx,0 ; zero ecx so only lower 8 bits are used
    mov cl,[ebx]
    inc ebx
    cmp cl,0 ; compare byte at address edx with 0
    jz strint_end ; if comparison was zero, this is the end of string
    
    ;if char is below '0' or above '9', it is outside the range of these and is not a digit
    cmp cl,'0'
    jb not_digit
    cmp cl,'9'
    ja not_digit
    
    ;but if it is a digit, then correct and process the character
    is_digit:
    sub cl,'0'
    jmp process_char
    
    not_digit:
    ;it isn't a digit, but it could an alphabet character which is a digit in a higher base
    
    ;if char is below 'A' or above 'Z', it is outside the range of these and is not capital letter
    cmp cl,'A'
    jb not_upper
    cmp cl,'Z'
    ja not_upper
    
    is_upper:
    sub cl,'A'
    add cl,10
    jmp process_char
    
    not_upper:
    
    ;if char is below 'a' or above 'z', it is outside the range of these and is not lowercase letter
    cmp cl,'a'
    jb not_lower
    cmp cl,'z'
    ja not_lower
    
    is_lower:
    sub cl,'a'
    add cl,10
    jmp process_char
    
    not_lower:
    
    ;if we have reached this point, result invalid and end function
    jmp strint_end
    
    process_char:
    
    cmp ecx,[radix] ;compare char with radix
    jae strint_end ;if this value is above or equal to radix, it is too high despite being a valid digit/alpha
    
    mov edx,0 ;zero edx because it is used in mul sometimes
    mul  dword [radix] ;mul eax with radix
    add eax,ecx
    
    jmp read_strint ;jump back and continue the loop if nothing has exited it
    
    strint_end:
    
    ret
    
    ;The utility functions below simply print a space or a newline.
    ;these help me save code when printing lots of strings and integers.
    
    space db ' ',0
    line db 0Dh,0Ah,0
    
    putspace:
    push eax
    mov eax,space
    call putstring
    pop eax
    ret
    
    putline:
    push eax
    mov eax,line
    call putstring
    pop eax
    ret
    
    ;a function for printing a single character that is the value of al
    
    char: db 0,0
    
    putchar:
    push eax
    mov [char],al
    mov eax,char
    call putstring
    pop eax
    ret
    
    ;a small function just for the common operation
    ;printing an integer followed by a space
    ;this saves a few bytes in the assembled code
    ;by reducing the number of function calls in the main program
    
    putint_and_space:
    call putint
    call putspace
    ret
    
    ;a small function just for the common operation
    ;printing an integer followed by a line feed
    ;this saves a few bytes in the assembled code
    ;by reducing the number of function calls in the main program
    
    putint_and_line:
    call putint
    call putline
    ret
    
    ;a small function just for the common operation
    ;printing a string followed by a line feed
    ;this saves a few bytes in the assembled code
    ;by reducing the number of function calls in the main program
    ;it also means we don't need to include a newline in every string!
    
    putstr_and_line:
    call putstring
    call putline
    ret
    

  • AAA DOS: Chapter 8: Going from DOS to Linux or Windows

    In the unlikely event that you have read the first 7 chapters of this book, I am going to assume you are a pretty hard core computer user. What I can say for sure is that you are the type of person who reads books or blog posts about technical details. DOS is an operating system that tends to only be used by nerds who love reading text and efficient operations at the command line.

    Sadly to say, our kind is dying out. At the time of writing this I am 38 years old and there are few people who remember the old way computers were used. DOS is mostly seen as a dead platform and it is not usually used except by programmers and hard core gamers who still run their favorite games in a DOS emulator. Though I cannot fail to mention that FreeDOS is available as a real DOS system.

    https://www.freedos.org/

    But most people know nothing about DOS because the popular operating systems available today are Windows, MacOS and Linux.

    If you have enjoyed programming in Assembly, I do have some helpful tips on how you can apply most of the same information to start Assembly in Linux.

    As far as Windows or MacOS go, I cannot help you much with that because I don’t use proprietary operating systems if I have a choice. These operating systems don’t allow you to simply load registers and call interrupts to print things on the screen.

    Linux, however, works very much like DOS does. If you know how to load the registers correctly and use a system call, you can print strings of text just like in DOS except MUCH faster because you will be running natively instead of in an emulator as in the DOS examples from the rest of this book.

    I cannot cover the details of installing a Linux operating system because there are many choices. However I recommend Debian because it has been my main distro for years. Therefore, the following two programs that I will show you in this chapter have both been tested to work on my 64 bit Intel PC running Debian 12 (bookworm).

    Remember, although DOS was a 16 bit system, modern Linux processors and distros usually support 32 or 64 bit code. Therefore, I will be showing you a small program using the FASM assembler that prints text using a Linux version of the putstring function. It behaves the same as the DOS version behaves in chapter 2.

    main.asm (32 bit)

    format ELF executable
    entry main
    
    main:
    
    mov eax,main_string
    call putstring
    
    mov eax, 1  ; invoke SYS_EXIT (kernel opcode 1)
    mov ebx, 0  ; return 0 status on exit - 'No Errors'
    int 80h
    
    ;A string to test if output works
    main_string db 'This program runs in Linux!',0Ah,0
    
    putstring:
    
    push eax
    push ebx
    push ecx
    push edx
    
    mov ebx,eax ; copy eax to ebx. ebx will be used as index to the string
    
    putstring_strlen_start: ; this loop finds the length of the string as part of the putstring function
    
    cmp [ebx],byte 0 ; compare byte at address ebx with 0
    jz putstring_strlen_end ; if comparison was zero, jump to loop end because we have found the length
    inc ebx
    jmp putstring_strlen_start
    
    putstring_strlen_end:
    sub ebx,eax ;By subtracting the start of the string with the current address, we have the length of the string.
    
    ; Write string using Linux Write system call. Reference for 32 bit x86 syscalls is below.
    ; https://www.chromium.org/chromium-os/developer-library/reference/linux-constants/syscalls/#x86-32-bit
    
    mov edx,ebx      ;number of bytes to write
    mov ecx,eax      ;pointer/address of string to write
    mov ebx,1        ;write to the STDOUT file
    mov eax,4        ;invoke SYS_WRITE (kernel opcode 4 on 32 bit systems)
    int 80h          ;system call to write the message
    
    pop edx
    pop ecx
    pop ebx
    pop eax
    
    ret ; this is the end of the putstring function return to calling location
    
    ; This Assembly source file has been formatted for the FASM assembler.
    ; The following 3 commands assemble, give executable permissions, and run the program
    ;
    ;	fasm main.asm
    ;	chmod +x main
    ;	./main
    

    The program above uses only two system calls. One is the call to exit the program. The other is the write call which is the same as the DOS function 0x40 of interrupt 0x21; However, the usage of the registers is not in the same order. However, these registers: eax,ebx,ecx,edx are the same registers except that they are extended to 32 bits. That is why they have an e in their name.

    But if you take the time to study it, you will see that it does the exact same process of finding the length of the string by the terminating zero and then loading the registers in such a way that the operating system knows what function we care calling, which handle we are writing to, how many bytes to write, and where the data is in memory which will be written.

    Next I will show you the 64-bit equivalent that works the same way but uses different numbers for the system calls.

    main.asm 64 bit

    format ELF64 executable
    entry main
    
    main: ; the main function of our assembly function, just as if I were writing C.
    
    mov rax,main_string ; move the address of main_string into rax register
    call putstring
    
    mov rax, 60 ; invoke SYS_EXIT (kernel opcode 60 on 64 bit systems)
    mov rdi,0   ; return 0 status on exit - 'No Errors'
    syscall
    
    ;A string to test if output works
    main_string db 'This program runs in Linux!',0Ah,0
    
    putstring:
    
    push rax
    push rbx
    push rcx
    push rdx
    
    mov rbx,rax ; copy rax to rbx as well. Now both registers have the address of the main_string
    
    putstring_strlen_start: ; this loop finds the lenge of the string as part of the putstring function
    
    cmp [rbx],byte 0 ; compare byte at address rdx with 0
    jz putstring_strlen_end ; if comparison was zero, jump to loop end because we have found the length
    inc rbx
    jmp putstring_strlen_start
    
    putstring_strlen_end:
    sub rbx,rax ;rbx will now have correct number of bytes
    
    ;write string using Linux Write system call
    ;https://www.chromium.org/chromium-os/developer-library/reference/linux-constants/syscalls/#x86_64-64-bit
    
    mov rdx,rbx      ;number of bytes to write
    mov rsi,rax      ;pointer/address of string to write
    mov rdi,1        ;write to the STDOUT file
    mov rax,1        ;invoke SYS_WRITE (kernel opcode 1 on 64 bit systems)
    syscall          ;system call to write the message
    
    pop rdx
    pop rcx
    pop rbx
    pop rax
    
    ret ; this is the end of the putstring function return to calling location
    
    
    ; This Assembly source file has been formatted for the FASM assembler.
    ; The following 3 commands assemble, give executable permissions, and run the program
    ;
    ;	fasm main.asm
    ;	chmod +x main
    ;	./main
    
    

    You may notice that the 64-bit program also uses the syscall instruction rather than interrupt 0x80. On my machine both programs behave identically because both calling conventions are valid. There are executables that run in 32 bit mode and others that run in 64 bit mode. They are not usually compatible and the FASM assembler has to be told which format is being assembled.

    FASM has been my preferred assembler for a long time because unlike NASM, it has everything it needs to create executables without depending on a linker.

    “What is a linker?” You might be asking. You see, the developers of Linux never really expected for people to be writing applications entirely in assembly. Usually they are written in C and then GCC compiles it to assembly that only the Gnu assembler (informally called Gas) can assemble and then link with the standard library. There is a linker program called “ld” that GCC automatically uses.

    However, through some research and experimentation, I have converted the previous 64 bit FASM program into the Gas syntax. As you read it, remember that the AT&T phone company made this weird alternative syntax. The source and destination have been flipped so you will see the register receiving data on the right side instead of the left.

    main.s (GNU Assembler 64 bit)

    # Using Linux System calls for 64-bit
    # Tested with GNU Assembler on Debian 12 (bookworm)
    # It uses Chastity's putstring function for output
    
    .global _start
    
    .text
    
    _start:
    
    mov $main_string,%rax # move address of string into rax register
    call   putstring      # call the putstring function Chastity wrote
    mov    $0x3c,%eax     # system call 60 is exit
    mov    $0x0,%edi      # we want to return code 0
    syscall               # end program with system call
    
    main_string:
    .string	"This program runs in Linux!\n"
    
    putstring:            # the start of the putstring function
    push   %rax
    push   %rbx
    push   %rcx
    push   %rdx
    mov    %rax,%rbx
    
    putstring_strlen_start:
    cmpb   $0x0,(%rbx)
    je     putstring_strlen_end
    inc    %rbx
    jmp    putstring_strlen_start
    
    putstring_strlen_end:
    sub    %rax,%rbx # subtract rax from rbx for number of bytes to write
    mov    %rbx,%rdx # copy number of bytes from rbx to rdx
    mov    %rax,%rsi # address of string to output
    mov    $0x1,%edi # file handler 1 is stdout
    mov    $0x1,%rax # system call 1 is write
    syscall
    pop    %rdx
    pop    %rcx
    pop    %rbx
    pop    %rax
    ret
    
    # This Assembly source file has been formatted for the GNU assembler.
    # The following makefile rule has commands to assemble, link, and run the program
    #
    #main-gas:
    #	gcc -nostdlib -nostartfiles -nodefaultlibs -static main.s -o main
    #	strip main
    #	./main
    

    Although I find the GNU Assembler syntax hard to read, the fact that this assembler exists as part of the GNU Compiler Collection means that it is usually available even on systems that don’t have FASM or NASM available.

    It is possible to use NASM also but it can’t create executables and requires linking with “ld” anyway. It is better to just write directly for the GNU Assembler or stick with FASM if you prefer intel syntax.

    However, the beauty is that the machine code bytes from both types of assembly are identical! In fact that is how I got the GAS version. I had to assemble the other version and then disassemble it with objdump to get the equivalent syntax.

    The programs you saw in this chapter only work on Linux, but Linux is Free both in terms of Software Freedom and Free in price too because anyone with an internet connection can download the ISO of a new operating system and install it on their computer as long as they take the time to read directions from the makers of that distribution. In fact Debian, Arch, Gentoo, and FreeBSD (not Linux but very similar) all have great instruction manuals. If you have managed to read this book, then you will have no problem following their stuff.

  • Writing to video RAM

    One of the reasons DOS is the best platform for learning Intel Assembly language in my opinion is that it doesn’t prevent you from writing directly to video memory. People are unaware how much operating systems like Windows, and, to a lesser extent, Linux place limits on what you are allowed to do. The idea is that most people are not smart enough to be trusted with arbitrarily writing to video RAM, devices, etc.

    But that is where DOSBox comes in. Since it runs DOS inside an emulator, there is no danger. The program I am going to show you today is long and complicated but it does something that I can’t even do on Linux, write directly to video memory in multiple colors.

    Here is the source code that made all of that possible! I tested it to assemble with either FASM or NASM. If you assemble it and run it in DOSBox, or even a real DOS system, you will get something that looks like the picture above.

    org 100h
    
    main:
    
    ;set up the extra segment at the beginning of the program
    ;to point to the video memory segment in DOS text mode
    mov ax, 0xB800
    mov es, ax ; Or mov ds, ax
    
    
    ;80 columes times 25 rows is 2000 chars
    ;but since each character is two bytes
    ;4000 is the number of bytes to erase
    ;whatever character we write in this loop will fill the whole screen!
    
    mov bx,0
    screen_clear:
    mov [es:bx],word 0x0403
    add bx,2
    cmp bx,4000
    jnz screen_clear
    
    mov ax,title  ;the string we intend to write to video RAM
    mov ch,0x0F   ;the character attribute
    mov dx,0x0218
    call putstring_vram
    
    mov ax,v_str  ;the string we intend to write to video RAM
    mov ch,0x70   ;the character attribute
    mov dx,0x0401
    call putstring_vram
    
    ;set the starting attribute for characters and location
    mov ch,0x01   ;the character attribute
    mov dx,0x0501 ;x,y position of where text should start on screen
    
    loop_vram:
    cmp ch,0x10
    jz loop_vram_end
    mov ax,v_str  ;the string we intend to write to video RAM
    call putstring_vram
    add dx,0x100
    inc ch
    jmp loop_vram
    loop_vram_end:
    
    mov ax,4C00h
    int 21h
    
    title db 'Chastity Video RAM Demonstration!',0
    v_str db 'Hello World! This string will be written to video RAM using Assembly language!',0
    
    ;Unlike previous functions I wrote that use DOS interrupts to write text to the screen
    ;this one makes use of several registers which are not meant to be preserved
    ;registers ax,cx,and dx must be set before calling this function
    
    ;ax = address of string to write
    ;bx = copied from ax and used to index the string
    ;cx = used for character attribute(ch) and value(cl)
    ;dx = column(x pos) and row(y pos) of where string should be printed
    
    ;For this routine, I chose to copy the dx register to memory locations for clarity
    ;Yes, it wastes some bytes but at least I can read it as I am familiar with x,y coordinates
    ;Most importantly, the dx register is never modified in this function
    ;This is important because the main program may need to modify it in a loop
    ;For writing data in consecutive rows (e.g. integer sequences)
    
    x db 0
    y db 0
    
    putstring_vram:
    
    mov bx,ax             ;copy ax to bx for use as index register
    
    ;get x and y positions from each byte of dx register
    mov [x],dl
    mov [y],dh
    
    mov ax,80  ;set ax to 80 because there are 80 chars per row in text mode
    mul byte [y]    ;multiply with the y value
    mov byte [y],0  ;zero the y byte so we can add a 16 bit x value to ax
    add ax, word [x]
    
    shl ax,1 ;shift left once to account for two bytes per character
    
    mov di,ax ;we will use di as our starting output location
    
    putstring_vram_strlen_start:    ;this loop finds the length of the string as part of the putstring function
    
    cmp [bx],byte 0                 ;compare this byte with 0
    jz putstring_vram_strlen_end    ;if comparison was zero, jump to loop end because we have found the length/end of string
    mov cl,[bx]                     ;mov this character to cl
    mov [es:di],cx                  ;mov character and attribute set in ch(before calling this function) to extra_segment+di
    add di,2                        ;each character contains two bytes (ASCII+Attribute). We must add two here.
    inc bx                          ;increment bx to point to next character
    jmp putstring_vram_strlen_start ;jump to the start of the loop and keep trying until we find a zero
    
    putstring_vram_strlen_end:
    
    ret