forked from TCShenanigans/symphony_stdlib
Compare commits
36
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ad8fd7e68e | ||
|
|
50f7040f59 | ||
|
|
57934309c2 | ||
|
|
32a157e87d | ||
|
|
a00c7f2a84 | ||
|
|
ae0684f98a | ||
|
|
257735d776 | ||
|
|
809f6e35d5 | ||
|
|
c83269fcf7 | ||
|
|
6bf0c0a4fc | ||
|
|
697445feec | ||
|
|
7c49c5212e | ||
|
|
6ff1a43c32 | ||
|
|
b5249a43b9 | ||
|
|
655ce0237e | ||
|
|
a1e19112c5 | ||
|
|
3d874cebbd | ||
|
|
4aca17aa40 | ||
|
|
0188969dab | ||
|
|
5c9a33c24d | ||
|
|
1e4ad8f64c | ||
|
|
7b43e75a9d | ||
|
|
20e1f17181 | ||
|
|
625f01166a | ||
|
|
e6a0739e95 | ||
|
|
1885f304ad | ||
|
|
63b37ab094 | ||
|
|
6981ff027d | ||
|
|
88c91a8eec | ||
|
|
a02f0646b4 | ||
|
|
7abfcabd73 | ||
|
|
b476b8aaa3 | ||
|
|
bddb153d9b | ||
|
|
29103d4335 | ||
|
|
837e8ba0a6 | ||
|
|
6ca77970b8 |
@@ -3,7 +3,7 @@
|
|||||||
This is a standard library for symphony.
|
This is a standard library for symphony.
|
||||||
It is both intended as a practical toolkit to develop more complex software as well as a teaching resource.
|
It is both intended as a practical toolkit to develop more complex software as well as a teaching resource.
|
||||||
|
|
||||||
If you just want to use the standard library [[stdlib.asm]] is your main header, include it after your code.
|
If you just want to use the standard library [[src/stdlib.asm]] is your main header, include it after your code.
|
||||||
|
|
||||||
If you are using it as a learning resource have a look at the [teaching folder](teaching).
|
If you are using it as a learning resource have a look at the [teaching folder](teaching).
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,3 @@
|
|||||||
|
# Examples
|
||||||
|
|
||||||
|
Examples of how to use the standard library to accomplish a task.
|
||||||
@@ -0,0 +1,77 @@
|
|||||||
|
; Returns the index of the first element matching the provided predicate function (or -1 if not found)
|
||||||
|
; Arguments:
|
||||||
|
; r1 - The array pointer
|
||||||
|
; r2 - The array length (number of items)
|
||||||
|
; r3 - The stride (size of one item) - either 1, 2 or 4 (bytes)
|
||||||
|
; r4 - The predicate
|
||||||
|
; r5 - Predicate context
|
||||||
|
; Result:
|
||||||
|
; r1 - The index of the first element matching the provided predicate function (or -1 if not found)
|
||||||
|
; Clobbers: r2, r3, r4, r5, r6, + what the predicate clobbers
|
||||||
|
; Info:
|
||||||
|
; The predicate function should follow the stdlib calling convention
|
||||||
|
; The predicate receives two arguments (the value and the predicate context) and should return either a zero when the value is not the one we search for
|
||||||
|
; , or any other result if it is the searched-for item.
|
||||||
|
pub find_index:
|
||||||
|
push r12 ; We will store the predicate pointer here
|
||||||
|
push r11 ; We will store the current pointer here
|
||||||
|
push r10 ; We will store the stride here
|
||||||
|
push r9 ; We will store the final address here
|
||||||
|
push r8 ; We will store the mask here
|
||||||
|
|
||||||
|
mov r12, r4
|
||||||
|
mov r11, r1
|
||||||
|
mov r10, r3
|
||||||
|
mov r9, r2
|
||||||
|
lsr r6, r3, 1 ; We turn the stride into a byte shift
|
||||||
|
lsl r9, r9, r6 ; We calculate bytes left
|
||||||
|
add r9, r9, r1 ; We add the start address to get the final address
|
||||||
|
|
||||||
|
push r13 ; We save up the return address because we will provide our own to the predicate
|
||||||
|
push r1 ; We need the array pointer to calculate the item index
|
||||||
|
counter r13
|
||||||
|
add r13, r13, 52 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles
|
||||||
|
|
||||||
|
nand r8, zr, zr ; We create a mask of 0xFFFFFFFF
|
||||||
|
mov r6, 4
|
||||||
|
sub r6, r6, r3 ; We create a "negative stride", e.g. 4 -> 0, 2 -> 2, 1 -> 3
|
||||||
|
lsl r6, r6, 3
|
||||||
|
lsr r8, r8, r6 ; We shift the mask by the negative stride to obtain the proper mask for a value
|
||||||
|
; e.g. stride 4 -> mask is 0xFFFFFFFF
|
||||||
|
; stride 2 -> mask is 0x0000FFFF
|
||||||
|
; stride 1 -> mask is 0x000000FF
|
||||||
|
|
||||||
|
push r5 ; We save the predicate context on the stack
|
||||||
|
find_index_loop:
|
||||||
|
load_32 r1, [r11] ; We load the element
|
||||||
|
and r1, r1, r8 ; We mask it to handle stride 2 and 1 cases
|
||||||
|
load_32 r2, [sp] ; We load the predicate context into r2
|
||||||
|
jmp r12 ; We call the predicate
|
||||||
|
cmp r1, zr
|
||||||
|
jne find_index_found_item ; If we found the item, we jump out
|
||||||
|
; If we didn't, move to next item
|
||||||
|
add r11, r11, r10 ; We add the stride to the pointer
|
||||||
|
cmp r11, r9 ; We compare with the final address
|
||||||
|
jne find_index_loop ; If we did not reach the end we jump back into the loop
|
||||||
|
|
||||||
|
find_index_not_found:
|
||||||
|
add sp, sp, 8 ; The predicate context and old array pointer are not useful
|
||||||
|
pop r13 ; We get our return address
|
||||||
|
nand r1, zr, zr ; We put -1 in r1
|
||||||
|
jmp find_index_postamble
|
||||||
|
|
||||||
|
find_index_found_item:
|
||||||
|
add sp, sp, 4 ; The predicate context is not useful
|
||||||
|
pop r1 ; We get the array pointer
|
||||||
|
pop r13 ; We get our return address
|
||||||
|
sub r1, r11, r1 ; We calculate the bytes from the start
|
||||||
|
lsr r10, r10, 1 ; We shift the stride to get the amount to shift the bytes for
|
||||||
|
lsr r1, r1, r10 ; We shift to get the index of the item
|
||||||
|
|
||||||
|
find_index_postamble:
|
||||||
|
pop r8
|
||||||
|
pop r9
|
||||||
|
pop r10
|
||||||
|
pop r11
|
||||||
|
pop r12
|
||||||
|
jmp r13 ; Return
|
||||||
@@ -0,0 +1,256 @@
|
|||||||
|
include errno
|
||||||
|
|
||||||
|
; Performs a Bitblock transfer, copying a section of the Source to the Destination, applying a given Mode operation
|
||||||
|
; Arguments:
|
||||||
|
; r1 - Destination Buffer pointer
|
||||||
|
; r2 - Source Buffer pointer
|
||||||
|
; r3 - X destination coordinate
|
||||||
|
; r4 - Y destination coordinate
|
||||||
|
; r5 - X source coordinate
|
||||||
|
; r6 - Y source coordinate
|
||||||
|
; r7 - Mode:
|
||||||
|
; 000 - SRCCOPY (copies source over destination)
|
||||||
|
; 001 - XOR
|
||||||
|
; 010 - AND
|
||||||
|
; 011 - NOT
|
||||||
|
; 100 - OR
|
||||||
|
; 101 - reserved
|
||||||
|
; 110 - reserved
|
||||||
|
; 111 - SRCALPHA (copies source over destination if: source != 0 (for 8bpp), source.alpha != 0 (for 32bpp))
|
||||||
|
; Stack argument 1: Width (in bytes)
|
||||||
|
; Stack argument 2: Height (in bytes)
|
||||||
|
; Result: None
|
||||||
|
; Clobbers: r1, r2, r3, r4, r5, r6, r7
|
||||||
|
; Errors: BUFFER_DEPTH_MISMATCH - if the buffers have different bit depths
|
||||||
|
pub bitblit:
|
||||||
|
push r8
|
||||||
|
push r9
|
||||||
|
; Checking if the bit depths are correct
|
||||||
|
add r8, r1, 6 ; r8 = pointer to destination depth
|
||||||
|
load_16 r8, [r8] ; r8 = dest depth
|
||||||
|
add r9, r2, 6 ; r9 = pointer to source depth
|
||||||
|
load_16 r9, [r9] ; r9 = src depth
|
||||||
|
cmp r8, r9
|
||||||
|
je bitblit_depth_good
|
||||||
|
; Bit depths differ - error
|
||||||
|
pop r9
|
||||||
|
pop r8
|
||||||
|
add sp, sp, 8 ; Removing the stack Arguments
|
||||||
|
mov flags, errno.BUFFER_DEPTH_MISMATCH
|
||||||
|
jmp r13
|
||||||
|
bitblit_depth_good:
|
||||||
|
push r8
|
||||||
|
add sp, sp, 12
|
||||||
|
load_32 r8, [sp] ; r8 = height
|
||||||
|
add sp, sp, 4
|
||||||
|
load_32 r9, [sp] ; r9 = width
|
||||||
|
|
||||||
|
sub sp, sp, 16
|
||||||
|
push r10
|
||||||
|
push r11
|
||||||
|
push r12
|
||||||
|
push r13
|
||||||
|
|
||||||
|
add r10, r1, 4 ; r10 = pointer to destination stride
|
||||||
|
load_16 r10, [r10] ; r10 = destination stride
|
||||||
|
|
||||||
|
add r11, r2, 4 ; r11 = pointer to source stride
|
||||||
|
load_16 r11, [r11] ; r11 = source stride
|
||||||
|
|
||||||
|
; Calculating destination index
|
||||||
|
push r10
|
||||||
|
mov r12, 0
|
||||||
|
bitblit_destination_index:
|
||||||
|
cmp r4, zr
|
||||||
|
je bitblit_after_destination_index
|
||||||
|
mov flags, r4
|
||||||
|
jne bitblit_destination_index_afteradd
|
||||||
|
add r12, r12, r10
|
||||||
|
bitblit_destination_index_afteradd:
|
||||||
|
lsr r4, r4, 1
|
||||||
|
lsl r10, r10, 1
|
||||||
|
jmp bitblit_destination_index
|
||||||
|
bitblit_after_destination_index:
|
||||||
|
add r4, r4, r12 ; Now r4 is dest_y * dest_stride
|
||||||
|
load_32 r1, [r1] ; r1 = destination data pointer
|
||||||
|
add r1, r1, r4 ; Now r1 is dest_pointer + dest_y * dest_stride
|
||||||
|
add r1, r1, r3 ; Now r1 is dest_pointer + dest_y * dest_stride + dest_x
|
||||||
|
pop r10
|
||||||
|
|
||||||
|
push r11
|
||||||
|
; Calculating source index
|
||||||
|
mov r12, 0
|
||||||
|
bitblit_source_index:
|
||||||
|
cmp r6, zr
|
||||||
|
je bitblit_after_source_index
|
||||||
|
mov flags, r6
|
||||||
|
jne bitblit_source_index_afteradd
|
||||||
|
add r12, r12, r11
|
||||||
|
bitblit_source_index_afteradd:
|
||||||
|
lsr r6, r6, 1
|
||||||
|
lsl r11, r11, 1
|
||||||
|
jmp bitblit_source_index
|
||||||
|
bitblit_after_source_index:
|
||||||
|
add r6, r6, r12 ; Now r6 is src_y * src_stride
|
||||||
|
load_32 r2, [r2] ; r2 = source data pointer
|
||||||
|
add r2, r2, r6 ; Now r2 is src_pointer + src_y * src_stride
|
||||||
|
add r2, r2, r5 ; Now r2 is src_pointer + src_y * src_stride + src_x
|
||||||
|
pop r11
|
||||||
|
|
||||||
|
; r3, r4, r5, r6 are now free
|
||||||
|
|
||||||
|
mov r3, bitblit_mode
|
||||||
|
mov r12, bitblit_end_mode
|
||||||
|
lsl r7, r7, 3 ; 2 instructions per mode
|
||||||
|
add r12, r7, r12
|
||||||
|
add r7, r7, r3
|
||||||
|
|
||||||
|
; r1 is line dest pointer (or bit depth)
|
||||||
|
; r2 is line src pointer (or temp)
|
||||||
|
; r3 is width iterator
|
||||||
|
; r4 is current dest pointer
|
||||||
|
; r5 is current src pointer
|
||||||
|
; r6 is current src value
|
||||||
|
; r7 is 32bit mode jump table destination
|
||||||
|
; r8 is height iterator
|
||||||
|
; r9 is width of bitblit
|
||||||
|
; r10 is dest stride
|
||||||
|
; r11 is src stride
|
||||||
|
; r12 is 8bit mode jump table destination
|
||||||
|
; r13 is current dest value
|
||||||
|
|
||||||
|
bitblit_copy_height:
|
||||||
|
cmp r8, zr
|
||||||
|
je bitblit_finish
|
||||||
|
|
||||||
|
bitblit_copy_line:
|
||||||
|
mov r3, 0
|
||||||
|
mov r4, r1 ; r4 = dest
|
||||||
|
mov r5, r2 ; r5 = src
|
||||||
|
push r1
|
||||||
|
add r1, sp, 20 ; Pointing to bit depth saved on stack
|
||||||
|
load_32 r1, [r1] ; r1 = bit depth
|
||||||
|
push r2
|
||||||
|
bitblit_copy_line_loop: ; Main loop that copies 4 bytes at a time
|
||||||
|
add r3, r3, 4
|
||||||
|
cmp r9, r3
|
||||||
|
je bitblit_copy_line_finish
|
||||||
|
jb bitblit_copy_line_end
|
||||||
|
load_32 r6, [r5] ; r6 = [src]
|
||||||
|
load_32 r13, [r4] ; r13 = [dest]
|
||||||
|
jmp r7
|
||||||
|
bitblit_mode:
|
||||||
|
nop ; SRCCOPY
|
||||||
|
jmp bitblit_save_bits
|
||||||
|
xor r6, r6, r13 ; XOR
|
||||||
|
jmp bitblit_save_bits
|
||||||
|
and r6, r6, r13 ; AND
|
||||||
|
jmp bitblit_save_bits
|
||||||
|
not r6, r6 ; NOT
|
||||||
|
jmp bitblit_save_bits
|
||||||
|
or r6, r6, r13 ; OR
|
||||||
|
jmp bitblit_save_bits
|
||||||
|
nop ; reserved
|
||||||
|
jmp bitblit_save_bits
|
||||||
|
nop ; reserved
|
||||||
|
jmp bitblit_save_bits
|
||||||
|
cmp r1, 8 ; SRCALPHA
|
||||||
|
jne bitblit_srcalpha_32 ; If bit depth is 32, we jump to alpha checking
|
||||||
|
; If bit depth is 8, we need to repack this value
|
||||||
|
mov r2, 0 ; our mask
|
||||||
|
|
||||||
|
lsr flags, r6, 24
|
||||||
|
cmp flags, zr
|
||||||
|
jne bitblit_srcalpha_8_1
|
||||||
|
or r2, r2, 0xFF
|
||||||
|
bitblit_srcalpha_8_1:
|
||||||
|
lsl r2, r2, 8
|
||||||
|
|
||||||
|
lsr flags, r6, 16
|
||||||
|
and flags, flags, 0xFF
|
||||||
|
cmp flags, zr
|
||||||
|
jne bitblit_srcalpha_8_2
|
||||||
|
or r2, r2, 0xFF
|
||||||
|
bitblit_srcalpha_8_2:
|
||||||
|
lsl r2, r2, 8
|
||||||
|
|
||||||
|
lsr flags, r6, 8
|
||||||
|
and flags, flags, 0xFF
|
||||||
|
cmp flags, zr
|
||||||
|
jne bitblit_srcalpha_8_3
|
||||||
|
or r2, r2, 0xFF
|
||||||
|
bitblit_srcalpha_8_3:
|
||||||
|
lsl r2, r2, 8
|
||||||
|
|
||||||
|
and flags, r6, 0xFF
|
||||||
|
cmp flags, zr
|
||||||
|
jne bitblit_srcalpha_8_4
|
||||||
|
or r2, r2, 0xFF
|
||||||
|
bitblit_srcalpha_8_4:
|
||||||
|
|
||||||
|
and r13, r13, r2
|
||||||
|
not r2, r2
|
||||||
|
and r6, r6, r2
|
||||||
|
or r6, r6, r13
|
||||||
|
|
||||||
|
jmp bitblit_save_bits
|
||||||
|
bitblit_srcalpha_32:
|
||||||
|
and flags, r6, 0xFF
|
||||||
|
cmp flags, zr
|
||||||
|
jne bitblit_save_bits ; If alpha channel is not zero, we save the value
|
||||||
|
mov r6, r13 ; If it was zero, we take [dest], i.e. don't overwrite
|
||||||
|
bitblit_save_bits:
|
||||||
|
store_32 [r4], r6
|
||||||
|
add r4, r4, 4
|
||||||
|
add r5, r5, 4
|
||||||
|
jmp bitblit_copy_line_loop
|
||||||
|
bitblit_copy_line_end: ; Now we need to handle up to 3 bytes that were not copied
|
||||||
|
sub r3, r3, 4
|
||||||
|
bitblit_copy_line_end_loop:
|
||||||
|
load_8 r6, [r5] ; r6 = [src]
|
||||||
|
load_8 r13, [r4] ; r13 = [dest]
|
||||||
|
jmp r12
|
||||||
|
bitblit_end_mode:
|
||||||
|
nop ; SRCCOPY
|
||||||
|
jmp bitblit_save_bits_end
|
||||||
|
xor r6, r6, r13 ; XOR
|
||||||
|
jmp bitblit_save_bits_end
|
||||||
|
and r6, r6, r13 ; AND
|
||||||
|
jmp bitblit_save_bits_end
|
||||||
|
not r6, r6 ; NOT
|
||||||
|
jmp bitblit_save_bits_end
|
||||||
|
or r6, r6, r13 ; OR
|
||||||
|
jmp bitblit_save_bits_end
|
||||||
|
nop ; reserved
|
||||||
|
jmp bitblit_save_bits_end
|
||||||
|
nop ; reserved
|
||||||
|
jmp bitblit_save_bits_end
|
||||||
|
cmp r6, zr ; SRCALPHA (will happen only in 8bpp)
|
||||||
|
jne bitblit_save_bits_end
|
||||||
|
mov r6, r13 ; [src] = 0, so we take [dest], i.e. don't overwrite
|
||||||
|
bitblit_save_bits_end:
|
||||||
|
store_8 [r4], r6
|
||||||
|
add r3, r3, 1
|
||||||
|
add r4, r4, 1
|
||||||
|
add r5, r5, 1
|
||||||
|
cmp r9, r3
|
||||||
|
ja bitblit_copy_line_end_loop
|
||||||
|
|
||||||
|
bitblit_copy_line_finish:
|
||||||
|
pop r2
|
||||||
|
pop r1
|
||||||
|
add r1, r1, r10
|
||||||
|
add r2, r2, r11 ; Moving dest & src to next line
|
||||||
|
sub r8, r8, 1
|
||||||
|
jmp bitblit_copy_height
|
||||||
|
bitblit_finish:
|
||||||
|
pop r13
|
||||||
|
pop r12
|
||||||
|
pop r11
|
||||||
|
pop r10
|
||||||
|
add sp, sp, 4 ; Bit depth skip
|
||||||
|
pop r9
|
||||||
|
pop r8
|
||||||
|
add sp, sp, 8
|
||||||
|
mov flags, errno.OK
|
||||||
|
jmp r13
|
||||||
+20
-11
@@ -6,20 +6,29 @@
|
|||||||
; r1 - The lower 32 bits of the result
|
; r1 - The lower 32 bits of the result
|
||||||
; Clobbers: r2, r3, r4, r5
|
; Clobbers: r2, r3, r4, r5
|
||||||
pub mul_low:
|
pub mul_low:
|
||||||
mov r3, 0 ; result
|
add r3, r2, r2 ; r3 = r2 * 2
|
||||||
mov r4, 31 ; loop counter
|
mov r4, 0 ; r4 has the result
|
||||||
|
|
||||||
mul_low_loop:
|
mul_low_loop:
|
||||||
asr r5, r2, 31
|
|
||||||
and r5, r5, r1
|
|
||||||
lsl r5, r5, r4
|
|
||||||
add r3, r3, r5
|
|
||||||
lsl r2, r2, 1
|
|
||||||
sub r4, r4, 1
|
|
||||||
cmp r4, 0
|
|
||||||
jge mul_low_loop
|
|
||||||
mov r1, r3
|
|
||||||
|
|
||||||
|
and r5, r1, 1 ; a0
|
||||||
|
neg r5, r5 ; mask
|
||||||
|
and r5, r2, r5 ; a0 ? r2 : 0
|
||||||
|
add r4, r4, r5
|
||||||
|
|
||||||
|
and r5, r1, 2 ; a1 is now 0 or 2
|
||||||
|
lsr r5, r5, 1 ; normalize to 0/1
|
||||||
|
neg r5, r5 ; mask
|
||||||
|
and r5, r3, r5 ; a1 ? r2 * 2 : 0
|
||||||
|
add r4, r4, r5
|
||||||
|
|
||||||
|
lsl r2, r2, 2
|
||||||
|
lsl r3, r3, 2
|
||||||
|
lsr r1, r1, 2
|
||||||
|
cmp r1, zr
|
||||||
|
jne mul_low_loop
|
||||||
|
|
||||||
|
mov r1, r4
|
||||||
jmp r13
|
jmp r13
|
||||||
|
|
||||||
; Calculates the absolute value of the value provided in the r1 register
|
; Calculates the absolute value of the value provided in the r1 register
|
||||||
+174
@@ -0,0 +1,174 @@
|
|||||||
|
; int compare(uint8_t* a, uint8_t* b, size_t count);
|
||||||
|
; Compares two memory segment of equal length lexicographically.
|
||||||
|
;
|
||||||
|
; Arguments:
|
||||||
|
; - `r1`: A pointer to the first memory segment.
|
||||||
|
; - `r2`: A pointer to the second memory segment.
|
||||||
|
; - `r3`: The size of both memory segments.
|
||||||
|
; Results:
|
||||||
|
; - `r1`:
|
||||||
|
; - `0` if both segments are equal.
|
||||||
|
; - `<0` if the first segment is less than the second segment.
|
||||||
|
; - `>0` if the first segment is greater than the second segment.
|
||||||
|
;
|
||||||
|
pub compare:
|
||||||
|
; Exclusive end point of the first segment.
|
||||||
|
add r3, r3, r1
|
||||||
|
sub r3, r3, 4
|
||||||
|
_compare__loop:
|
||||||
|
load_32 r4, [r1]
|
||||||
|
add r1, r1, 4
|
||||||
|
load_32 r5, [r2]
|
||||||
|
add r2, r2, 4
|
||||||
|
; Comparing two sequences of 4 bytes lexicographically is equivalent to
|
||||||
|
; comparing the corresponding big endian 32 bit words.
|
||||||
|
cmp r4, r5
|
||||||
|
jne _compare__break
|
||||||
|
; Check if there are enough bytes left to continue with the vectorized loop.
|
||||||
|
cmp r1, r3
|
||||||
|
jbe _compare__loop
|
||||||
|
; `r3 + 4 - r1 = <remaining byte count> = r3 - r1 mod 4`
|
||||||
|
sub flags, r3, r1
|
||||||
|
; Check if one of the lowest 2 bits is non-zero
|
||||||
|
jbe _compare__rem
|
||||||
|
; If not, we are done. Both segments are equal.
|
||||||
|
mov r1, 0
|
||||||
|
jmp r13
|
||||||
|
_compare__break:
|
||||||
|
; `flags` is the comparison result in the format of `cmp`. Convert it to the desired format.
|
||||||
|
; 00 => 0x40000000 > 0
|
||||||
|
; 01 => 0x00000000 = 0
|
||||||
|
; 10 => 0xC0000000 < 0
|
||||||
|
xor r1, flags, 1
|
||||||
|
lsl r1, r1, 30
|
||||||
|
jmp r13
|
||||||
|
_compare__rem:
|
||||||
|
; Compute `S = 8*(4 - <remaining byte count>)` and
|
||||||
|
; [r1] >> S, [r2] >> S
|
||||||
|
mov r3, 8
|
||||||
|
load_32 r4, [r1]
|
||||||
|
sub r3, r3, flags
|
||||||
|
load_32 r5, [r2]
|
||||||
|
lsl r3, r3, 3
|
||||||
|
lsr r4, r4, r3
|
||||||
|
lsr r5, r5, r3
|
||||||
|
; Compare both values, now with garbage bytes removed.
|
||||||
|
cmp r4, r5
|
||||||
|
jmp _compare__break
|
||||||
|
|
||||||
|
|
||||||
|
; void copy(void* src, void* dest, size_t count);
|
||||||
|
; Copies `count` bytes from `src` to `dest`. The two memory segments must not overlap.
|
||||||
|
;
|
||||||
|
; Arguments:
|
||||||
|
; - `r1`: Pointer to the memory segment to be copied.
|
||||||
|
; - `r2`: Pointer to the memory segment to be copied into.
|
||||||
|
; - `r3`: Byte size of both the `src` and `dest` segments.
|
||||||
|
;
|
||||||
|
pub copy:
|
||||||
|
; Exclusive end point of the source segment.
|
||||||
|
add r3, r1, r3
|
||||||
|
; Last index from where we can safely copy 8 bytes per loop iteration.
|
||||||
|
sub r3, r3, 8
|
||||||
|
jmp _copy__loop_entry
|
||||||
|
_copy__loop:
|
||||||
|
; Copy 8 bytes from `src` to `dest`.
|
||||||
|
load_32 flags, [r1]
|
||||||
|
add r1, r1, 4
|
||||||
|
store_32 [r2], flags
|
||||||
|
add r2, r2, 4
|
||||||
|
load_32 flags, [r1]
|
||||||
|
add r1, r1, 4
|
||||||
|
store_32 [r2], flags
|
||||||
|
add r2, r2, 4
|
||||||
|
_copy__loop_entry:
|
||||||
|
; Check if we can process more data in the vectorized loop.
|
||||||
|
cmp r1, r3
|
||||||
|
jbe _copy__loop
|
||||||
|
; The remaining amount of bytes `R` is `R = r3 + 8 - r1 = r3 - r1 mod 8`.
|
||||||
|
sub flags, r3, r1
|
||||||
|
; Test if `R` is not a multiple of `4`, i.e. the lowest 2 bits are non-zero.
|
||||||
|
jbe _copy__rem
|
||||||
|
; `R` is a multiple of `4`. Special case this.
|
||||||
|
; Check if `R` is `0`, i.e. the third bit is also 0. In that case, we are already done.
|
||||||
|
; There are no conditional indirect jumps, so we can't return immediately.
|
||||||
|
jge _copy__ret
|
||||||
|
; `R = 4`. No need to update `r1` or `r2`, we don't need them anymore.
|
||||||
|
load_32 flags, [r1]
|
||||||
|
store_32 [r2], flags
|
||||||
|
_copy__ret:
|
||||||
|
; Return
|
||||||
|
jmp r13
|
||||||
|
_copy__rem:
|
||||||
|
; Optimize the remaining cases for code size.
|
||||||
|
; End point of the source segment.
|
||||||
|
add r3, r3, 8
|
||||||
|
; We already handled the case `R = 0` earlier,
|
||||||
|
; so no bounds check needed for the first iteration.
|
||||||
|
_copy__rem_loop:
|
||||||
|
; Copy 1 byte.
|
||||||
|
load_8 flags, [r1]
|
||||||
|
add r1, r1, 1
|
||||||
|
store_8 [r2], flags
|
||||||
|
add r2, r2, 1
|
||||||
|
; Check if we are still within the bounds.
|
||||||
|
cmp r1, r3
|
||||||
|
jb _copy__rem_loop
|
||||||
|
jmp r13
|
||||||
|
|
||||||
|
; void fill32(uint8_t* dest, size_t count, uint32_t value);
|
||||||
|
; Fills `count` bytes in `dest` with `value`. If `count` is not a multiple of 4,
|
||||||
|
; the least significant bytes of `value` are cut off for the last entry.
|
||||||
|
;
|
||||||
|
; Arguments:
|
||||||
|
; - `r1`: A pointer to the destination segment.
|
||||||
|
; - `r2`: The size of the destination segment.
|
||||||
|
; - `r3`: The 32 bit value that the segment is filled with.
|
||||||
|
;
|
||||||
|
pub fill32:
|
||||||
|
; Exclusive end point of the destination segment.
|
||||||
|
add r2, r1, r2
|
||||||
|
; Last index from where we can safely write 8 bytes per loop iteration.
|
||||||
|
sub r2, r2, 8
|
||||||
|
jmp _fill32__entry
|
||||||
|
_fill32__loop:
|
||||||
|
; Set 8 bytes per loop iteraion.
|
||||||
|
store_32 [r1], r3
|
||||||
|
add r1, r1, 4
|
||||||
|
store_32 [r1], r3
|
||||||
|
add r1, r1, 4
|
||||||
|
_fill32__entry:
|
||||||
|
; Check if we can process more data in the vectorized loop.
|
||||||
|
cmp r1, r2
|
||||||
|
jbe _fill32__loop
|
||||||
|
; The remaining amount of bytes `R` is `R = r2 + 8 - r1 = r2 - r1 mod 8`.
|
||||||
|
sub flags, r2, r1
|
||||||
|
; Check if the third bit of the remainder is cleared.
|
||||||
|
jge _fill32__r4
|
||||||
|
; Otherwise set 4 bytes.
|
||||||
|
store_32 [r1], r3
|
||||||
|
add r1, r1, 4
|
||||||
|
_fill32__r4:
|
||||||
|
; Check if the two least significant bits of the remainder are zero.
|
||||||
|
ja _fill32__ret
|
||||||
|
; Handle the remaining bytes `R` individually, in reverse order.
|
||||||
|
add r2, r2, 4
|
||||||
|
; `r1 + 4 - r2 = 4 - R`.
|
||||||
|
sub flags, r1, r2
|
||||||
|
; Exclusive end point of the destination segment.
|
||||||
|
add r2, r2, 4
|
||||||
|
; Shift out the least significant `8*(4 - R)` bits of the value.
|
||||||
|
lsl flags, flags, 3
|
||||||
|
lsr r3, r3, flags
|
||||||
|
jmp _fill32__loop2_entry
|
||||||
|
_fill32__loop2:
|
||||||
|
sub r2, r2, 1
|
||||||
|
; Write the least significant byte of the value...
|
||||||
|
store_8 [r2], r3
|
||||||
|
; and then shift it out.
|
||||||
|
lsr r3, r3, 8
|
||||||
|
_fill32__loop2_entry:
|
||||||
|
cmp r1, r2
|
||||||
|
jb _fill32__loop2
|
||||||
|
_fill32__ret:
|
||||||
|
jmp r13
|
||||||
@@ -1,6 +1,8 @@
|
|||||||
pub include bit
|
pub include bit
|
||||||
pub include imath
|
pub include imath
|
||||||
|
pub include array
|
||||||
pub include console
|
pub include console
|
||||||
|
pub include mem
|
||||||
|
|
||||||
; Needs to be last!
|
; Needs to be last!
|
||||||
pub include LUTs
|
pub include LUTs
|
||||||
@@ -0,0 +1,77 @@
|
|||||||
|
; int compare(uint8_t* a, uint8_t* b, size_t count);
|
||||||
|
; Compares two memory segment of equal length lexicographically.
|
||||||
|
; Temporarily modifies the byte at address `a + count`.
|
||||||
|
;
|
||||||
|
; Arguments:
|
||||||
|
; - `r1`: A pointer to the first memory segment.
|
||||||
|
; - `r2`: A pointer to the second memory segment.
|
||||||
|
; - `r3`: The size of both memory segments.
|
||||||
|
; Results:
|
||||||
|
; - `r1`:
|
||||||
|
; - `0` if both segments are equal.
|
||||||
|
; - `<0` if the first segment is less than the second segment.
|
||||||
|
; - `>0` if the first segment is greater than the second segment.
|
||||||
|
pub compare:
|
||||||
|
cmp r1, r2
|
||||||
|
je _compare__is_eq
|
||||||
|
; Exclusive end point of the second segment.
|
||||||
|
add r4, r3, r2
|
||||||
|
; Exclusive end point of the first segment.
|
||||||
|
add r3, r3, r1
|
||||||
|
load_8 r6, [r3]
|
||||||
|
load_8 flags, [r4]
|
||||||
|
; Check if the first bytes behind the sequences are equal.
|
||||||
|
cmp flags, r6
|
||||||
|
jne _compare__loop
|
||||||
|
; Change the byte directly behind the first segment.
|
||||||
|
xor r4, r6, 1
|
||||||
|
; This would be problematic if someone calls compare with a first segment
|
||||||
|
; whose end point overlaps the program memory of this function.
|
||||||
|
store_8 [r3], r4
|
||||||
|
_compare__loop:
|
||||||
|
load_32 r4, [r1]
|
||||||
|
add r1, r1, 4
|
||||||
|
load_32 r5, [r2]
|
||||||
|
add r2, r2, 4
|
||||||
|
; Comparing two sequences of 4 bytes lexicographically is equivalent to
|
||||||
|
; comparing the corresponding big endian 32 bit words.
|
||||||
|
cmp r4, r5
|
||||||
|
je _compare__loop
|
||||||
|
; We overshot in the loop; decrement r1 again. (Only by 2, we backtrack the rest if necessary later)
|
||||||
|
sub r1, r1, 2
|
||||||
|
; Restore the byte we changed.
|
||||||
|
store_8 [r3], r6
|
||||||
|
; We encountered two different words. Figure out what byte they differ on.
|
||||||
|
xor r4, r4, r5
|
||||||
|
; Store the flags for later, to figure out the return value.
|
||||||
|
mov r5, flags
|
||||||
|
; Check if at least one of the two most significant bytes is not 0.
|
||||||
|
cmp r4, 0xffff
|
||||||
|
jbe _compare__low2
|
||||||
|
; If it is, backtrack the remaining 2 indices.
|
||||||
|
; Shift the most significant bytes to the least significant ones.
|
||||||
|
sub r1, r1, 2
|
||||||
|
lsr r4, r4, 16
|
||||||
|
_compare__low2:
|
||||||
|
; r1 now points to a non-zero 16 bit value.
|
||||||
|
; If the 16 bit value at r1-2 is in-bounds, then it is 0.
|
||||||
|
; Check if the most significant byte of the 16 bit value is 0.
|
||||||
|
cmp r4, 0xff
|
||||||
|
ja _compare__low1
|
||||||
|
; If it is, our target is the least significant byte.
|
||||||
|
add r1, r1, 1
|
||||||
|
_compare__low1:
|
||||||
|
; Otherwise, the target is that non-zero byte.
|
||||||
|
; Check if the target is out of bounds, i.e. the loop terminated through the "bounds check".
|
||||||
|
cmp r1, r3
|
||||||
|
jae _compare__is_eq
|
||||||
|
; r5 is the comparison result in the format of `cmp`. Convert it to the desired format.
|
||||||
|
; 00 => 0x40000000 > 0
|
||||||
|
; 01 => 0x00000000 = 0
|
||||||
|
; 10 => 0xC0000000 < 0
|
||||||
|
xor r1, r5, 1
|
||||||
|
lsl r1, r1, 30
|
||||||
|
jmp r13
|
||||||
|
_compare__is_eq:
|
||||||
|
mov r1, 0
|
||||||
|
jmp r13
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
# Tests
|
||||||
|
|
||||||
|
Tests for the standard library go here, tests are allowed to depend on the recommended spec.isa changes.
|
||||||
Reference in New Issue
Block a user