From 6ca77970b808240f99b91f84e441d8997fa931d1 Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Mon, 31 Aug 2026 13:44:07 +0200 Subject: [PATCH 01/13] Added find_index array function --- array.asm | 76 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ stdlib.asm | 3 ++- 2 files changed, 78 insertions(+), 1 deletion(-) create mode 100644 array.asm diff --git a/array.asm b/array.asm new file mode 100644 index 0000000..798d8d8 --- /dev/null +++ b/array.asm @@ -0,0 +1,76 @@ +; Returns the index of the first element matching the provided predicate function (or -1 if not found) +; r1 - The array pointer +; r2 - The array length (number of items) +; r3 - The stride (size of one item) - either 1, 2 or 4 (bytes) +; r4 - The predicate +; r5 - Predicate context +; The predicate function should follow the stdlib calling convention +; and receive one argument and return either a zero when the value is not the one we search for +; , or any other result if it is the searched-for item. +pub find_index: + push r12 ; We will store the predicate pointer here + push r11 ; We will store the current pointer here + push r10 ; We will store the stride here + push r9 ; We will store the bytes left here + push r8 ; We will store the mask here + push r1 ; We need the array pointer to calculate the item index + + mov r12, r4 + mov r11, r1 + mov r10, r3 + mov r9, r2 + lsl r9, r9, r3 + + push r13 ; We save up the return address because we will provide our own to the predicate + counter r13 + add r13, r13, 40 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles + + mov r8, 0 + not r8, r8 ; We create a mask of 0xFFFFFFFF + mov r6, 4 + sub r6, r6, r3 ; We create a "negative stride", e.g. 4 -> 0, 2 -> 2, 1 -> 3 + lsl r6, r6, 3 + lsr r8, r8, r6 ; We shift the mask by the negative stride to obtain the proper mask for a value + ; e.g. stride 4 -> mask is 0xFFFFFFFF + ; stride 2 -> mask is 0x0000FFFF + ; stride 1 -> mask is 0x000000FF + + push r5 ; We save the predicate context on the stack + find_index_loop: + load_32 r1, [r11] ; We load the element + and r1, r1, r8 ; We mask it to handle stride 2 and 1 cases + load_32 r2, [sp] ; We load the predicate context into r2 + jmp r12 ; We call the predicate + cmp r1, zr + jneq find_index_found_item ; If we found the item, we jump out + ; If we didn't, move to next item + add r11, r11, r10 ; We add the stride to the pointer + sub r9, r9, r10 ; We decrease bytes left + cmp r9, zr + jeq find_index_not_found ; If we reached the end we're done + jmp find_index_loop + + find_index_not_found: + add sp, sp, 4 ; The predicate context is not useful + pop r13 ; We get our return address + mov r1, 0 + not r1, r1 ; We put -1 in r1 + add sp, sp, 4 ; The old array pointer are not useful + jmp find_index_postamble + + find_index_found_item: + add sp, sp, 4 ; The predicate context is not useful + pop r13 ; We get our return address + pop r1 ; We get the array pointer + sub r1, r11, r1 ; We calculate the bytes from the start + lsr r10, r10, 1 ; We shift the stride to get the amount to shift the bytes for + lsr r1, r1, r10 ; We shift to get the index of the item + + find_index_postamble: + pop r8 + pop r9 + pop r10 + pop r11 + pop r12 + jmp r13 ; Return + diff --git a/stdlib.asm b/stdlib.asm index 5852f04..5b29575 100644 --- a/stdlib.asm +++ b/stdlib.asm @@ -22,4 +22,5 @@ ; ===== TYPES ===== pub include bit -pub include imath \ No newline at end of file +pub include imath +pub include array \ No newline at end of file From 837e8ba0a6757cd089e282d3350dc151cc1d241d Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Mon, 31 Aug 2026 13:52:27 +0200 Subject: [PATCH 02/13] Fixed the predicate return address --- array.asm | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/array.asm b/array.asm index 798d8d8..bfcb3ed 100644 --- a/array.asm +++ b/array.asm @@ -23,7 +23,7 @@ pub find_index: push r13 ; We save up the return address because we will provide our own to the predicate counter r13 - add r13, r13, 40 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles + add r13, r13, 52 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles mov r8, 0 not r8, r8 ; We create a mask of 0xFFFFFFFF From 29103d43354db2b254fe93834a67a88936684c87 Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Mon, 31 Aug 2026 15:30:11 +0200 Subject: [PATCH 03/13] Reduced operations to get -1 in register --- array.asm | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/array.asm b/array.asm index bfcb3ed..3d06c91 100644 --- a/array.asm +++ b/array.asm @@ -11,7 +11,7 @@ pub find_index: push r12 ; We will store the predicate pointer here push r11 ; We will store the current pointer here push r10 ; We will store the stride here - push r9 ; We will store the bytes left here + push r9 ; We will store the final address here push r8 ; We will store the mask here push r1 ; We need the array pointer to calculate the item index @@ -19,14 +19,14 @@ pub find_index: mov r11, r1 mov r10, r3 mov r9, r2 - lsl r9, r9, r3 + lsl r9, r9, r3 ; We calculate bytes left + add r9, r9, r1 ; We add the start address to get the final address push r13 ; We save up the return address because we will provide our own to the predicate counter r13 add r13, r13, 52 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles - mov r8, 0 - not r8, r8 ; We create a mask of 0xFFFFFFFF + nand r8, zr, zr ; We create a mask of 0xFFFFFFFF mov r6, 4 sub r6, r6, r3 ; We create a "negative stride", e.g. 4 -> 0, 2 -> 2, 1 -> 3 lsl r6, r6, 3 @@ -45,16 +45,14 @@ pub find_index: jneq find_index_found_item ; If we found the item, we jump out ; If we didn't, move to next item add r11, r11, r10 ; We add the stride to the pointer - sub r9, r9, r10 ; We decrease bytes left - cmp r9, zr + cmp r11, r9 ; We compare with the final address jeq find_index_not_found ; If we reached the end we're done jmp find_index_loop find_index_not_found: add sp, sp, 4 ; The predicate context is not useful pop r13 ; We get our return address - mov r1, 0 - not r1, r1 ; We put -1 in r1 + nand r1, zr, zr ; We put -1 in r1 add sp, sp, 4 ; The old array pointer are not useful jmp find_index_postamble From bddb153d9be9712247b586f3b42b4718aa8ea05d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Isalski?= Date: Mon, 31 Aug 2026 23:03:23 +0200 Subject: [PATCH 04/13] Tested and fixed stride->shift conversion missing --- array.asm | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/array.asm b/array.asm index 3d06c91..c6e34d5 100644 --- a/array.asm +++ b/array.asm @@ -19,7 +19,8 @@ pub find_index: mov r11, r1 mov r10, r3 mov r9, r2 - lsl r9, r9, r3 ; We calculate bytes left + lsr r6, r3, 1 ; We turn the stride into a byte shift + lsl r9, r9, r6 ; We calculate bytes left add r9, r9, r1 ; We add the start address to get the final address push r13 ; We save up the return address because we will provide our own to the predicate @@ -42,11 +43,11 @@ pub find_index: load_32 r2, [sp] ; We load the predicate context into r2 jmp r12 ; We call the predicate cmp r1, zr - jneq find_index_found_item ; If we found the item, we jump out + jne find_index_found_item ; If we found the item, we jump out ; If we didn't, move to next item add r11, r11, r10 ; We add the stride to the pointer cmp r11, r9 ; We compare with the final address - jeq find_index_not_found ; If we reached the end we're done + je find_index_not_found ; If we reached the end we're done jmp find_index_loop find_index_not_found: @@ -71,4 +72,3 @@ pub find_index: pop r11 pop r12 jmp r13 ; Return - From b476b8aaa3b7640a956e4472340a87d90219645b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Isalski?= Date: Mon, 31 Aug 2026 23:06:40 +0200 Subject: [PATCH 05/13] Added comment about predicate context --- array.asm | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/array.asm b/array.asm index c6e34d5..13dfc4b 100644 --- a/array.asm +++ b/array.asm @@ -5,7 +5,7 @@ ; r4 - The predicate ; r5 - Predicate context ; The predicate function should follow the stdlib calling convention -; and receive one argument and return either a zero when the value is not the one we search for +; The predicate receives two arguments (the value and the predicate context) and should return either a zero when the value is not the one we search for ; , or any other result if it is the searched-for item. pub find_index: push r12 ; We will store the predicate pointer here From 7abfcabd73246a6eea6a5c92b81690f256ebbfe1 Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Mon, 31 Aug 2026 13:44:07 +0200 Subject: [PATCH 06/13] Added find_index array function --- array.asm | 76 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ stdlib.asm | 3 ++- 2 files changed, 78 insertions(+), 1 deletion(-) create mode 100644 array.asm diff --git a/array.asm b/array.asm new file mode 100644 index 0000000..798d8d8 --- /dev/null +++ b/array.asm @@ -0,0 +1,76 @@ +; Returns the index of the first element matching the provided predicate function (or -1 if not found) +; r1 - The array pointer +; r2 - The array length (number of items) +; r3 - The stride (size of one item) - either 1, 2 or 4 (bytes) +; r4 - The predicate +; r5 - Predicate context +; The predicate function should follow the stdlib calling convention +; and receive one argument and return either a zero when the value is not the one we search for +; , or any other result if it is the searched-for item. +pub find_index: + push r12 ; We will store the predicate pointer here + push r11 ; We will store the current pointer here + push r10 ; We will store the stride here + push r9 ; We will store the bytes left here + push r8 ; We will store the mask here + push r1 ; We need the array pointer to calculate the item index + + mov r12, r4 + mov r11, r1 + mov r10, r3 + mov r9, r2 + lsl r9, r9, r3 + + push r13 ; We save up the return address because we will provide our own to the predicate + counter r13 + add r13, r13, 40 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles + + mov r8, 0 + not r8, r8 ; We create a mask of 0xFFFFFFFF + mov r6, 4 + sub r6, r6, r3 ; We create a "negative stride", e.g. 4 -> 0, 2 -> 2, 1 -> 3 + lsl r6, r6, 3 + lsr r8, r8, r6 ; We shift the mask by the negative stride to obtain the proper mask for a value + ; e.g. stride 4 -> mask is 0xFFFFFFFF + ; stride 2 -> mask is 0x0000FFFF + ; stride 1 -> mask is 0x000000FF + + push r5 ; We save the predicate context on the stack + find_index_loop: + load_32 r1, [r11] ; We load the element + and r1, r1, r8 ; We mask it to handle stride 2 and 1 cases + load_32 r2, [sp] ; We load the predicate context into r2 + jmp r12 ; We call the predicate + cmp r1, zr + jneq find_index_found_item ; If we found the item, we jump out + ; If we didn't, move to next item + add r11, r11, r10 ; We add the stride to the pointer + sub r9, r9, r10 ; We decrease bytes left + cmp r9, zr + jeq find_index_not_found ; If we reached the end we're done + jmp find_index_loop + + find_index_not_found: + add sp, sp, 4 ; The predicate context is not useful + pop r13 ; We get our return address + mov r1, 0 + not r1, r1 ; We put -1 in r1 + add sp, sp, 4 ; The old array pointer are not useful + jmp find_index_postamble + + find_index_found_item: + add sp, sp, 4 ; The predicate context is not useful + pop r13 ; We get our return address + pop r1 ; We get the array pointer + sub r1, r11, r1 ; We calculate the bytes from the start + lsr r10, r10, 1 ; We shift the stride to get the amount to shift the bytes for + lsr r1, r1, r10 ; We shift to get the index of the item + + find_index_postamble: + pop r8 + pop r9 + pop r10 + pop r11 + pop r12 + jmp r13 ; Return + diff --git a/stdlib.asm b/stdlib.asm index 5852f04..5b29575 100644 --- a/stdlib.asm +++ b/stdlib.asm @@ -22,4 +22,5 @@ ; ===== TYPES ===== pub include bit -pub include imath \ No newline at end of file +pub include imath +pub include array \ No newline at end of file From a02f0646b49be3323c6393e82fb09b6cbea35801 Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Mon, 31 Aug 2026 13:52:27 +0200 Subject: [PATCH 07/13] Fixed the predicate return address --- array.asm | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/array.asm b/array.asm index 798d8d8..bfcb3ed 100644 --- a/array.asm +++ b/array.asm @@ -23,7 +23,7 @@ pub find_index: push r13 ; We save up the return address because we will provide our own to the predicate counter r13 - add r13, r13, 40 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles + add r13, r13, 52 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles mov r8, 0 not r8, r8 ; We create a mask of 0xFFFFFFFF From 88c91a8eecfc118f57f1946d364e44b3d903e596 Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Mon, 31 Aug 2026 15:30:11 +0200 Subject: [PATCH 08/13] Reduced operations to get -1 in register --- array.asm | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/array.asm b/array.asm index bfcb3ed..3d06c91 100644 --- a/array.asm +++ b/array.asm @@ -11,7 +11,7 @@ pub find_index: push r12 ; We will store the predicate pointer here push r11 ; We will store the current pointer here push r10 ; We will store the stride here - push r9 ; We will store the bytes left here + push r9 ; We will store the final address here push r8 ; We will store the mask here push r1 ; We need the array pointer to calculate the item index @@ -19,14 +19,14 @@ pub find_index: mov r11, r1 mov r10, r3 mov r9, r2 - lsl r9, r9, r3 + lsl r9, r9, r3 ; We calculate bytes left + add r9, r9, r1 ; We add the start address to get the final address push r13 ; We save up the return address because we will provide our own to the predicate counter r13 add r13, r13, 52 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles - mov r8, 0 - not r8, r8 ; We create a mask of 0xFFFFFFFF + nand r8, zr, zr ; We create a mask of 0xFFFFFFFF mov r6, 4 sub r6, r6, r3 ; We create a "negative stride", e.g. 4 -> 0, 2 -> 2, 1 -> 3 lsl r6, r6, 3 @@ -45,16 +45,14 @@ pub find_index: jneq find_index_found_item ; If we found the item, we jump out ; If we didn't, move to next item add r11, r11, r10 ; We add the stride to the pointer - sub r9, r9, r10 ; We decrease bytes left - cmp r9, zr + cmp r11, r9 ; We compare with the final address jeq find_index_not_found ; If we reached the end we're done jmp find_index_loop find_index_not_found: add sp, sp, 4 ; The predicate context is not useful pop r13 ; We get our return address - mov r1, 0 - not r1, r1 ; We put -1 in r1 + nand r1, zr, zr ; We put -1 in r1 add sp, sp, 4 ; The old array pointer are not useful jmp find_index_postamble From 6981ff027df7dc2a6631bca2acd415332ab471ff Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Isalski?= Date: Mon, 31 Aug 2026 23:03:23 +0200 Subject: [PATCH 09/13] Tested and fixed stride->shift conversion missing --- array.asm | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/array.asm b/array.asm index 3d06c91..c6e34d5 100644 --- a/array.asm +++ b/array.asm @@ -19,7 +19,8 @@ pub find_index: mov r11, r1 mov r10, r3 mov r9, r2 - lsl r9, r9, r3 ; We calculate bytes left + lsr r6, r3, 1 ; We turn the stride into a byte shift + lsl r9, r9, r6 ; We calculate bytes left add r9, r9, r1 ; We add the start address to get the final address push r13 ; We save up the return address because we will provide our own to the predicate @@ -42,11 +43,11 @@ pub find_index: load_32 r2, [sp] ; We load the predicate context into r2 jmp r12 ; We call the predicate cmp r1, zr - jneq find_index_found_item ; If we found the item, we jump out + jne find_index_found_item ; If we found the item, we jump out ; If we didn't, move to next item add r11, r11, r10 ; We add the stride to the pointer cmp r11, r9 ; We compare with the final address - jeq find_index_not_found ; If we reached the end we're done + je find_index_not_found ; If we reached the end we're done jmp find_index_loop find_index_not_found: @@ -71,4 +72,3 @@ pub find_index: pop r11 pop r12 jmp r13 ; Return - From 63b37ab0947dba2931c02ca868ce93f96e83c3a8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Isalski?= Date: Mon, 31 Aug 2026 23:06:40 +0200 Subject: [PATCH 10/13] Added comment about predicate context --- array.asm | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/array.asm b/array.asm index c6e34d5..13dfc4b 100644 --- a/array.asm +++ b/array.asm @@ -5,7 +5,7 @@ ; r4 - The predicate ; r5 - Predicate context ; The predicate function should follow the stdlib calling convention -; and receive one argument and return either a zero when the value is not the one we search for +; The predicate receives two arguments (the value and the predicate context) and should return either a zero when the value is not the one we search for ; , or any other result if it is the searched-for item. pub find_index: push r12 ; We will store the predicate pointer here From e6a0739e95df68f6d647fe0a749e98c68b07d914 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Isalski?= Date: Mon, 31 Aug 2026 23:57:55 +0200 Subject: [PATCH 11/13] pleegwat's code review fixes --- array.asm | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/array.asm b/array.asm index 13dfc4b..0ba4e6b 100644 --- a/array.asm +++ b/array.asm @@ -13,7 +13,6 @@ pub find_index: push r10 ; We will store the stride here push r9 ; We will store the final address here push r8 ; We will store the mask here - push r1 ; We need the array pointer to calculate the item index mov r12, r4 mov r11, r1 @@ -24,6 +23,7 @@ pub find_index: add r9, r9, r1 ; We add the start address to get the final address push r13 ; We save up the return address because we will provide our own to the predicate + push r1 ; We need the array pointer to calculate the item index counter r13 add r13, r13, 52 ; Point to just after the predicate call - we can set this up now so we don't waste loop cycles @@ -47,14 +47,12 @@ pub find_index: ; If we didn't, move to next item add r11, r11, r10 ; We add the stride to the pointer cmp r11, r9 ; We compare with the final address - je find_index_not_found ; If we reached the end we're done - jmp find_index_loop + jne find_index_loop ; If we did not reach the end we jump back into the loop find_index_not_found: - add sp, sp, 4 ; The predicate context is not useful + add sp, sp, 8 ; The predicate context and old array pointer are not useful pop r13 ; We get our return address nand r1, zr, zr ; We put -1 in r1 - add sp, sp, 4 ; The old array pointer are not useful jmp find_index_postamble find_index_found_item: From 625f01166af76b8900fd4d692e075ba45cca6741 Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Tue, 1 Sep 2026 13:52:26 +0200 Subject: [PATCH 12/13] Made find_index conform to new doc guidelines --- array.asm | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/array.asm b/array.asm index 0ba4e6b..582ca7b 100644 --- a/array.asm +++ b/array.asm @@ -1,9 +1,14 @@ ; Returns the index of the first element matching the provided predicate function (or -1 if not found) +; Arguments: ; r1 - The array pointer ; r2 - The array length (number of items) ; r3 - The stride (size of one item) - either 1, 2 or 4 (bytes) ; r4 - The predicate ; r5 - Predicate context +; Result: +; r1 - The index of the first element matching the provided predicate function (or -1 if not found) +; Clobbers: r2, r3, r4, r5, r6, + what the predicate clobbers +; Info: ; The predicate function should follow the stdlib calling convention ; The predicate receives two arguments (the value and the predicate context) and should return either a zero when the value is not the one we search for ; , or any other result if it is the searched-for item. From 20e1f171810bb5efedf4bb330b8f9db769478aef Mon Sep 17 00:00:00 2001 From: Michal Isalski Date: Tue, 1 Sep 2026 13:52:45 +0200 Subject: [PATCH 13/13] Fixed swapped pop instructions --- array.asm | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/array.asm b/array.asm index 582ca7b..c90ba38 100644 --- a/array.asm +++ b/array.asm @@ -62,8 +62,8 @@ pub find_index: find_index_found_item: add sp, sp, 4 ; The predicate context is not useful - pop r13 ; We get our return address pop r1 ; We get the array pointer + pop r13 ; We get our return address sub r1, r11, r1 ; We calculate the bytes from the start lsr r10, r10, 1 ; We shift the stride to get the amount to shift the bytes for lsr r1, r1, r10 ; We shift to get the index of the item