4 changed files with 76 additions and 69 deletions
+74
View File
@@ -0,0 +1,74 @@
; Returns the index of the first element matching the provided predicate function (or -1 if not found)
; r1 - The array pointer
; r2 - The array length (number of items)
; r3 - The stride (size of one item) - either 1, 2 or 4 (bytes)
; r4 - The predicate
; r5 - Predicate context
; The predicate function should follow the stdlib calling convention
; The predicate receives two arguments (the value and the predicate context) and should return either a zero when the value is not the one we search for
; , or any other result if it is the searched-for item.
pub find_index:
push r12 ; We will store the predicate pointer here
push r11 ; We will store the current pointer here
push r10 ; We will store the stride here
push r9 ; We will store the final address here
push r8 ; We will store the mask here
push r1 ; We need the array pointer to calculate the item index
mov r12, r4
mov r11, r1
mov r10, r3
mov r9, r2
lsr r6, r3, 1 ; We turn the stride into a byte shift
lsl r9, r9, r6 ; We calculate bytes left
add r9, r9, r1 ; We add the start address to get the final address
push r13 ; We save up the return address because we will provide our own to the predicate
counter r13
add r13, r13, 52 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles
nand r8, zr, zr ; We create a mask of 0xFFFFFFFF
mov r6, 4
sub r6, r6, r3 ; We create a "negative stride", e.g. 4 -> 0, 2 -> 2, 1 -> 3
lsl r6, r6, 3
lsr r8, r8, r6 ; We shift the mask by the negative stride to obtain the proper mask for a value
; e.g. stride 4 -> mask is 0xFFFFFFFF
; stride 2 -> mask is 0x0000FFFF
; stride 1 -> mask is 0x000000FF
push r5 ; We save the predicate context on the stack
find_index_loop:
load_32 r1, [r11] ; We load the element
and r1, r1, r8 ; We mask it to handle stride 2 and 1 cases
load_32 r2, [sp] ; We load the predicate context into r2
jmp r12 ; We call the predicate
cmp r1, zr
jne find_index_found_item ; If we found the item, we jump out
; If we didn't, move to next item
add r11, r11, r10 ; We add the stride to the pointer
cmp r11, r9 ; We compare with the final address
je find_index_not_found ; If we reached the end we're done
jmp find_index_loop
find_index_not_found:
add sp, sp, 4 ; The predicate context is not useful
pop r13 ; We get our return address
nand r1, zr, zr ; We put -1 in r1
add sp, sp, 4 ; The old array pointer are not useful
jmp find_index_postamble
find_index_found_item:
add sp, sp, 4 ; The predicate context is not useful
pop r13 ; We get our return address
pop r1 ; We get the array pointer
sub r1, r11, r1 ; We calculate the bytes from the start
lsr r10, r10, 1 ; We shift the stride to get the amount to shift the bytes for
lsr r1, r1, r10 ; We shift to get the index of the item
find_index_postamble:
pop r8
pop r9
pop r10
pop r11
pop r12
jmp r13 ; Return
-22
View File
@@ -70,25 +70,3 @@ pub popc:
add r4, r4, r2 add r4, r4, r2
and r1, r4, 0xFF ; mask out end result in lowest byte and r1, r4, 0xFF ; mask out end result in lowest byte
jmp r13 jmp r13
; Calculates the parity of the value provided in the r1 register
; i.e. 0 means even bits set, 1 means odd bits set
; Based on Stanford's BitHacks
; Clobbers r2
pub parity:
; v ^= v >> 16;
lsr r2, r1, 16
xor r1, r1, r2
; v ^= v >> 8;
lsr r2, r1, 8
xor r1, r1, r2
; v ^= v >> 4;
lsr r2, r1, 4
xor r1, r1, r2
; v &= 0xf;
and r1, r1, 0x0F
; return (0x6996 >> v) & 1;
mov r2, 0x6996
lsr r1, r2, r1
and r1, r1, 0x01
jmp r13
-46
View File
@@ -13,50 +13,4 @@ pub mul_low:
jge mull_loop jge mull_loop
mov r1, r3 mov r1, r3
jmp r13
; Calculates the absolute value of the value provided in the r1 register
; Based on Stanford's BitHacks
; Clobbers r2
pub abs: ; SHOULD BE INLINED
; mask = v >> 31
asr r2, r1, 31
; v + mask
add r1, r1, r2
; return (v + mask) ^ mask
xor r1, r1, r2
jmp r13
; Calculates the minimum value of the two values provided in the r1 and r2 registers
; Based on Stanford's BitHacks
; Clobbers flags
pub min: ; SHOULD BE INLINED
; x < y
cmp r1, r2
lsr flags, flags, 2
; -(x < y)
neg flags, flags
; x ^ y
xor r1, r1, r2
; (x ^ y) & -(x < y)
and r1, r1, flags
; return y ^ ((x ^ y) & -(x < y))
xor r1, r2, r1
jmp r13
; Calculates the maximum value of the two values provided in the r1 and r2 registers
; Based on Stanford's BitHacks
; Clobbers r2 and flags
pub max: ; SHOULD BE INLINED
; x < y
cmp r1, r2
lsr flags, flags, 2
; -(x < y)
neg flags, flags
; x ^ y
xor r2, r1, r2
; (x ^ y) & -(x < y)
and r2, r2, flags
; return x ^ ((x ^ y) & -(x < y))
xor r1, r1, r2
jmp r13 jmp r13
+2 -1
View File
@@ -22,4 +22,5 @@
; ===== TYPES ===== ; ===== TYPES =====
pub include bit pub include bit
pub include imath pub include imath
pub include array