From b99532b980561a0ed9e0ddb759ea4e64e8ee7dba Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Isalski?= Date: Mon, 31 Aug 2026 01:17:50 +0200 Subject: [PATCH 1/6] Added abs, max, min and parity functions --- bit.asm | 22 ++++++++++++++++++++++ imath.asm | 46 ++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 68 insertions(+) diff --git a/bit.asm b/bit.asm index f015315..3cba7a6 100644 --- a/bit.asm +++ b/bit.asm @@ -70,3 +70,25 @@ pub popc: add r4, r4, r2 and r1, r4, 0xFF ; mask out end result in lowest byte jmp r13 + +; Calculates the parity of the value provided in the r1 register +; i.e. 0 means even bits set, 1 means odd bits set +; Based on Stanford's BitHacks +; Clobbers r2 +pub parity: + ; v ^= v >> 16; + lsr r2, r1, 16 + xor r1, r1, r2 + ; v ^= v >> 8; + lsr r2, r1, 8 + xor r1, r1, r2 + ; v ^= v >> 4; + lsr r2, r1, 4 + xor r1, r1, r2 + ; v &= 0xf; + and r1, r1, 0x0F + ; return (0x6996 >> v) & 1; + mov r2, 0x6996 + lsr r1, r2, r1 + and r1, r1, 0x01 + jmp r13 \ No newline at end of file diff --git a/imath.asm b/imath.asm index 3262321..90056f0 100644 --- a/imath.asm +++ b/imath.asm @@ -13,4 +13,50 @@ pub mul_low: jge mull_loop mov r1, r3 + jmp r13 + +; Calculates the absolute value of the value provided in the r1 register +; Based on Stanford's BitHacks +; Clobbers r2 +pub abs: ; SHOULD BE INLINED + ; mask = v >> 31 + asr r2, r1, 31 + ; v + mask + add r1, r1, r2 + ; return (v + mask) ^ mask + xor r1, r1, r2 + jmp r13 + +; Calculates the minimum value of the two values provided in the r1 and r2 registers +; Based on Stanford's BitHacks +; Clobbers flags +pub min: ; SHOULD BE INLINED + ; x < y + cmp r1, r2 + lsr flags, flags, 2 + ; -(x < y) + neg flags, flags + ; x ^ y + xor r1, r1, r2 + ; (x ^ y) & -(x < y) + and r1, r1, flags + ; return y ^ ((x ^ y) & -(x < y)) + xor r1, r2, r1 + jmp r13 + +; Calculates the maximum value of the two values provided in the r1 and r2 registers +; Based on Stanford's BitHacks +; Clobbers r2 and flags +pub max: ; SHOULD BE INLINED + ; x < y + cmp r1, r2 + lsr flags, flags, 2 + ; -(x < y) + neg flags, flags + ; x ^ y + xor r2, r1, r2 + ; (x ^ y) & -(x < y) + and r2, r2, flags + ; return x ^ ((x ^ y) & -(x < y)) + xor r1, r1, r2 jmp r13 \ No newline at end of file From 7abfcabd73246a6eea6a5c92b81690f256ebbfe1 Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Mon, 31 Aug 2026 13:44:07 +0200 Subject: [PATCH 2/6] Added find_index array function --- array.asm | 76 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ stdlib.asm | 3 ++- 2 files changed, 78 insertions(+), 1 deletion(-) create mode 100644 array.asm diff --git a/array.asm b/array.asm new file mode 100644 index 0000000..798d8d8 --- /dev/null +++ b/array.asm @@ -0,0 +1,76 @@ +; Returns the index of the first element matching the provided predicate function (or -1 if not found) +; r1 - The array pointer +; r2 - The array length (number of items) +; r3 - The stride (size of one item) - either 1, 2 or 4 (bytes) +; r4 - The predicate +; r5 - Predicate context +; The predicate function should follow the stdlib calling convention +; and receive one argument and return either a zero when the value is not the one we search for +; , or any other result if it is the searched-for item. +pub find_index: + push r12 ; We will store the predicate pointer here + push r11 ; We will store the current pointer here + push r10 ; We will store the stride here + push r9 ; We will store the bytes left here + push r8 ; We will store the mask here + push r1 ; We need the array pointer to calculate the item index + + mov r12, r4 + mov r11, r1 + mov r10, r3 + mov r9, r2 + lsl r9, r9, r3 + + push r13 ; We save up the return address because we will provide our own to the predicate + counter r13 + add r13, r13, 40 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles + + mov r8, 0 + not r8, r8 ; We create a mask of 0xFFFFFFFF + mov r6, 4 + sub r6, r6, r3 ; We create a "negative stride", e.g. 4 -> 0, 2 -> 2, 1 -> 3 + lsl r6, r6, 3 + lsr r8, r8, r6 ; We shift the mask by the negative stride to obtain the proper mask for a value + ; e.g. stride 4 -> mask is 0xFFFFFFFF + ; stride 2 -> mask is 0x0000FFFF + ; stride 1 -> mask is 0x000000FF + + push r5 ; We save the predicate context on the stack + find_index_loop: + load_32 r1, [r11] ; We load the element + and r1, r1, r8 ; We mask it to handle stride 2 and 1 cases + load_32 r2, [sp] ; We load the predicate context into r2 + jmp r12 ; We call the predicate + cmp r1, zr + jneq find_index_found_item ; If we found the item, we jump out + ; If we didn't, move to next item + add r11, r11, r10 ; We add the stride to the pointer + sub r9, r9, r10 ; We decrease bytes left + cmp r9, zr + jeq find_index_not_found ; If we reached the end we're done + jmp find_index_loop + + find_index_not_found: + add sp, sp, 4 ; The predicate context is not useful + pop r13 ; We get our return address + mov r1, 0 + not r1, r1 ; We put -1 in r1 + add sp, sp, 4 ; The old array pointer are not useful + jmp find_index_postamble + + find_index_found_item: + add sp, sp, 4 ; The predicate context is not useful + pop r13 ; We get our return address + pop r1 ; We get the array pointer + sub r1, r11, r1 ; We calculate the bytes from the start + lsr r10, r10, 1 ; We shift the stride to get the amount to shift the bytes for + lsr r1, r1, r10 ; We shift to get the index of the item + + find_index_postamble: + pop r8 + pop r9 + pop r10 + pop r11 + pop r12 + jmp r13 ; Return + diff --git a/stdlib.asm b/stdlib.asm index 5852f04..5b29575 100644 --- a/stdlib.asm +++ b/stdlib.asm @@ -22,4 +22,5 @@ ; ===== TYPES ===== pub include bit -pub include imath \ No newline at end of file +pub include imath +pub include array \ No newline at end of file From a02f0646b49be3323c6393e82fb09b6cbea35801 Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Mon, 31 Aug 2026 13:52:27 +0200 Subject: [PATCH 3/6] Fixed the predicate return address --- array.asm | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/array.asm b/array.asm index 798d8d8..bfcb3ed 100644 --- a/array.asm +++ b/array.asm @@ -23,7 +23,7 @@ pub find_index: push r13 ; We save up the return address because we will provide our own to the predicate counter r13 - add r13, r13, 40 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles + add r13, r13, 52 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles mov r8, 0 not r8, r8 ; We create a mask of 0xFFFFFFFF From 88c91a8eecfc118f57f1946d364e44b3d903e596 Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Mon, 31 Aug 2026 15:30:11 +0200 Subject: [PATCH 4/6] Reduced operations to get -1 in register --- array.asm | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/array.asm b/array.asm index bfcb3ed..3d06c91 100644 --- a/array.asm +++ b/array.asm @@ -11,7 +11,7 @@ pub find_index: push r12 ; We will store the predicate pointer here push r11 ; We will store the current pointer here push r10 ; We will store the stride here - push r9 ; We will store the bytes left here + push r9 ; We will store the final address here push r8 ; We will store the mask here push r1 ; We need the array pointer to calculate the item index @@ -19,14 +19,14 @@ pub find_index: mov r11, r1 mov r10, r3 mov r9, r2 - lsl r9, r9, r3 + lsl r9, r9, r3 ; We calculate bytes left + add r9, r9, r1 ; We add the start address to get the final address push r13 ; We save up the return address because we will provide our own to the predicate counter r13 add r13, r13, 52 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles - mov r8, 0 - not r8, r8 ; We create a mask of 0xFFFFFFFF + nand r8, zr, zr ; We create a mask of 0xFFFFFFFF mov r6, 4 sub r6, r6, r3 ; We create a "negative stride", e.g. 4 -> 0, 2 -> 2, 1 -> 3 lsl r6, r6, 3 @@ -45,16 +45,14 @@ pub find_index: jneq find_index_found_item ; If we found the item, we jump out ; If we didn't, move to next item add r11, r11, r10 ; We add the stride to the pointer - sub r9, r9, r10 ; We decrease bytes left - cmp r9, zr + cmp r11, r9 ; We compare with the final address jeq find_index_not_found ; If we reached the end we're done jmp find_index_loop find_index_not_found: add sp, sp, 4 ; The predicate context is not useful pop r13 ; We get our return address - mov r1, 0 - not r1, r1 ; We put -1 in r1 + nand r1, zr, zr ; We put -1 in r1 add sp, sp, 4 ; The old array pointer are not useful jmp find_index_postamble From 6981ff027df7dc2a6631bca2acd415332ab471ff Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Isalski?= Date: Mon, 31 Aug 2026 23:03:23 +0200 Subject: [PATCH 5/6] Tested and fixed stride->shift conversion missing --- array.asm | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/array.asm b/array.asm index 3d06c91..c6e34d5 100644 --- a/array.asm +++ b/array.asm @@ -19,7 +19,8 @@ pub find_index: mov r11, r1 mov r10, r3 mov r9, r2 - lsl r9, r9, r3 ; We calculate bytes left + lsr r6, r3, 1 ; We turn the stride into a byte shift + lsl r9, r9, r6 ; We calculate bytes left add r9, r9, r1 ; We add the start address to get the final address push r13 ; We save up the return address because we will provide our own to the predicate @@ -42,11 +43,11 @@ pub find_index: load_32 r2, [sp] ; We load the predicate context into r2 jmp r12 ; We call the predicate cmp r1, zr - jneq find_index_found_item ; If we found the item, we jump out + jne find_index_found_item ; If we found the item, we jump out ; If we didn't, move to next item add r11, r11, r10 ; We add the stride to the pointer cmp r11, r9 ; We compare with the final address - jeq find_index_not_found ; If we reached the end we're done + je find_index_not_found ; If we reached the end we're done jmp find_index_loop find_index_not_found: @@ -71,4 +72,3 @@ pub find_index: pop r11 pop r12 jmp r13 ; Return - From 63b37ab0947dba2931c02ca868ce93f96e83c3a8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Isalski?= Date: Mon, 31 Aug 2026 23:06:40 +0200 Subject: [PATCH 6/6] Added comment about predicate context --- array.asm | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/array.asm b/array.asm index c6e34d5..13dfc4b 100644 --- a/array.asm +++ b/array.asm @@ -5,7 +5,7 @@ ; r4 - The predicate ; r5 - Predicate context ; The predicate function should follow the stdlib calling convention -; and receive one argument and return either a zero when the value is not the one we search for +; The predicate receives two arguments (the value and the predicate context) and should return either a zero when the value is not the one we search for ; , or any other result if it is the searched-for item. pub find_index: push r12 ; We will store the predicate pointer here