From 7b43e75a9dad253275fcaf1f694665954e0de825 Mon Sep 17 00:00:00 2001 From: MutexRaceCondition Date: Wed, 2 Sep 2026 19:41:56 +0200 Subject: [PATCH 1/5] Added memcmp, memcpy, memset32, unsafe.memcmp --- stdlib.asm | 177 ++++++++++++++++++++++++++++++++++++++++++++++++++++- unsafe.asm | 75 +++++++++++++++++++++++ 2 files changed, 251 insertions(+), 1 deletion(-) create mode 100644 unsafe.asm diff --git a/stdlib.asm b/stdlib.asm index 5852f04..a072fc7 100644 --- a/stdlib.asm +++ b/stdlib.asm @@ -22,4 +22,179 @@ ; ===== TYPES ===== pub include bit -pub include imath \ No newline at end of file +pub include imath + +; int memcmp(uint8_t* a, uint8_t* b, size_t count); +; Compares two memory segment of equal length lexicographically. +; +; Arguments: +; - `r1`: A pointer to the first memory segment. +; - `r2`: A pointer to the second memory segment. +; - `r3`: The size of both memory segments. +; Results: +; - `r1`: +; - `0` if both segments are equal. +; - `<0` if the first segment is less than the second segment. +; - `>0` if the first segment is greater than the second segment. +; +pub fn_memcmp: + ; Exclusive end point of the first segment. + add r3, r3, r1 + sub r3, r3, 4 +_memcmp__loop: + load_32 r4, [r1] + add r1, r1, 4 + load_32 r5, [r2] + add r2, r2, 4 + ; Comparing two sequences of 4 bytes lexicographically is equivalent to + ; comparing the corresponding big endian 32 bit words. + cmp r4, r5 + jne _memcmp__break + ; Check if there are enough bytes left to continue with the vectorized loop. + cmp r1, r3 + jbe _memcmp__loop + ; `r3 + 4 - r1 = = r3 - r1 mod 4` + sub flags, r3, r1 + ; Check if one of the lowest 2 bits is non-zero + jbe _memcmp__rem + ; If not, we are done. Both segments are equal. + mov r1, 0 + jmp r13 +_memcmp__break: + ; `flags` is the comparison result in the format of `cmp`. Convert it to the desired format. + ; 00 => 0x40000000 > 0 + ; 01 => 0x00000000 = 0 + ; 10 => 0xC0000000 < 0 + xor r1, flags, 1 + lsl r1, r1, 30 + jmp r13 +_memcmp__rem: + ; Compute `S = 8*(4 - )` and + ; [r1] >> S, [r2] >> S + mov r3, 8 + load_32 r4, [r1] + sub r3, r3, flags + load_32 r5, [r2] + lsl r3, r3, 3 + lsr r4, r4, r3 + lsr r5, r5, r3 + ; Compare both values, now with garbage bytes removed. + cmp r4, r5 + jmp _memcmp__break + + +; void memcpy(void* src, void* dest, size_t count); +; Copies `count` bytes from `src` to `dest`. The two memory segments must not overlap. +; +; Arguments: +; - `r1`: Pointer to the memory segment to be copied. +; - `r2`: Pointer to the memory segment to be copied into. +; - `r3`: Byte size of both the `src` and `dest` segments. +; +pub fn_memcpy: + ; Exclusive end point of the source segment. + add r3, r1, r3 + ; Last index from where we can safely copy 8 bytes per loop iteration. + sub r3, r3, 8 + jmp _memcpy__loop_entry +_memcpy__loop: + ; Copy 8 bytes from `src` to `dest`. + load_32 flags, [r1] + add r1, r1, 4 + store_32 [r2], flags + add r2, r2, 4 + load_32 flags, [r1] + add r1, r1, 4 + store_32 [r2], flags + add r2, r2, 4 +_memcpy__loop_entry: + ; Check if we can process more data in the vectorized loop. + cmp r1, r3 + jbe _memcpy__loop + ; The remaining amount of bytes `R` is `R = r3 + 8 - r1 = r3 - r1 mod 8`. + sub flags, r3, r1 + ; Test if `R` is not a multiple of `4`, i.e. the lowest 2 bits are non-zero. + jbe _memcpy__rem + ; `R` is a multiple of `4`. Special case this. + ; Check if `R` is `0`, i.e. the third bit is also 0. In that case, we are already done. + ; There are no conditional indirect jumps, so we can't return immediately. + jge _memcpy__ret + ; `R = 4`. No need to update `r1` or `r2`, we don't need them anymore. + load_32 flags, [r1] + store_32 [r2], flags +_memcpy__ret: + ; Return + jmp r13 +_memcpy__rem: + ; Optimize the remaining cases for code size. + ; End point of the source segment. + add r3, r3, 8 + ; We already handled the case `R = 0` earlier, + ; so no bounds check needed for the first iteration. +_memcpy__rem_loop: + ; Copy 1 byte. + load_8 flags, [r1] + add r1, r1, 1 + store_8 [r2], flags + add r2, r2, 1 + ; Check if we are still within the bounds. + cmp r1, r3 + jb _memcpy__rem_loop + jmp r13 + +; void memset32(uint8_t* dest, size_t count, uint32_t value); +; Fills `count` bytes in `dest` with `value`. If `count` is not a multiple of 4, +; the least significant bytes of `value` are cut off for the last entry. +; +; Arguments: +; - `r1`: A pointer to the destination segment. +; - `r2`: The size of the destination segment. +; - `r3`: The 32 bit value that the segment is filled with. +; +pub fn_memset32: + ; Exclusive end point of the destination segment. + add r2, r1, r2 + ; Last index from where we can safely write 8 bytes per loop iteration. + sub r2, r2, 8 + jmp _memset32__entry +_memset32__loop: + ; Set 8 bytes per loop iteraion. + store_32 [r1], r3 + add r1, r1, 4 + store_32 [r1], r3 + add r1, r1, 4 +_memset32__entry: + ; Check if we can process more data in the vectorized loop. + cmp r1, r2 + jbe _memset32__loop + ; The remaining amount of bytes `R` is `R = r2 + 8 - r1 = r2 - r1 mod 8`. + sub flags, r2, r1 + ; Check if the third bit of the remainder is cleared. + jge _memset32__r4 + ; Otherwise set 4 bytes. + store_32 [r1], r3 + add r1, r1, 4 +_memset32__r4: + ; Check if the two least significant bits of the remainder are zero. + ja _memset32__ret + ; Handle the remaining bytes `R` individually, in reverse order. + add r2, r2, 4 + ; `r1 + 4 - r2 = 4 - R`. + sub flags, r1, r2 + ; Exclusive end point of the destination segment. + add r2, r2, 4 + ; Shift out the least significant `8*(4 - R)` bits of the value. + lsl flags, flags, 3 + lsr r3, r3, flags + jmp _memset32__loop2_entry +_memset32__loop2: + sub r2, r2, 1 + ; Write the least significant byte of the value... + store_8 [r2], r3 + ; and then shift it out. + lsr r3, r3, 8 +_memset32__loop2_entry: + cmp r1, r2 + jb _memset32__loop2 +_memset32__ret: + jmp r13 diff --git a/unsafe.asm b/unsafe.asm new file mode 100644 index 0000000..e2ce635 --- /dev/null +++ b/unsafe.asm @@ -0,0 +1,75 @@ +; int memcmp(uint8_t* a, uint8_t* b, size_t count); +; Compares two memory segment of equal length lexicographically. +; Temporarily modifies the byte at address `a + count`. +; +; Arguments: +; - `r1`: A pointer to the first memory segment. +; - `r2`: A pointer to the second memory segment. +; - `r3`: The size of both memory segments. +; Results: +; - `r1`: +; - `0` if both segments are equal. +; - `<0` if the first segment is less than the second segment. +; - `>0` if the first segment is greater than the second segment. +pub fn_memcmp: + ; Exclusive end point of the second segment. + add r4, r3, r2 + ; Exclusive end point of the first segment. + add r3, r3, r1 + load_8 r6, [r3] + load_8 flags, [r4] + ; Check if the first bytes behind the sequences are equal. + cmp flags, r6 + jne _memcmp__loop + ; Change the byte directly behind the first segment. + xor r4, r6, 1 + ; This would be problematic if someone calls memcmp with a first segment + ; whose end point overlaps the program memory of this function. + store_8 [r3], r4 +_memcmp__loop: + load_32 r4, [r1] + add r1, r1, 4 + load_32 r5, [r2] + add r2, r2, 4 + ; Comparing two sequences of 4 bytes lexicographically is equivalent to + ; comparing the corresponding big endian 32 bit words. + cmp r4, r5 + je _memcmp__loop + ; We overshot in the loop; decrement r1 again. (Only by 2, we backtrack the rest if necessary later) + sub r1, r1, 2 + ; Restore the byte we changed. + store_8 [r3], r6 + ; We encountered two different words. Figure out what byte they differ on. + xor r4, r4, r5 + ; Store the flags for later, to figure out the return value. + mov r5, flags + ; Check if at least one of the two most significant bytes is not 0. + cmp r4, 0xffff + jbe _memcmp__low2 + ; If it is, backtrack the remaining 2 indices. + ; Shift the most significant bytes to the least significant ones. + sub r1, r1, 2 + lsr r4, r4, 16 +_memcmp__low2: + ; r1 now points to a non-zero 16 bit value. + ; If the 16 bit value at r1-2 is in-bounds, then it is 0. + ; Check if the most significant byte of the 16 bit value is 0. + cmp r4, 0xff + ja _memcmp__low1 + ; If it is, our target is the least significant byte. + add r1, r1, 1 +_memcmp__low1: + ; Otherwise, the target is that non-zero byte. + ; Check if the target is out of bounds, i.e. the loop terminated through the "bounds check". + cmp r1, r3 + jae _memcmp__oob + ; r5 is the comparison result in the format of `cmp`. Convert it to the desired format. + ; 00 => 0x40000000 > 0 + ; 01 => 0x00000000 = 0 + ; 10 => 0xC0000000 < 0 + xor r1, r5, 1 + lsl r1, r1, 30 + jmp r13 +_memcmp__oob: + mov r1, 0 + jmp r13 From a1e19112c559a841539840f4579286465c311a8c Mon Sep 17 00:00:00 2001 From: MutexRaceCondition Date: Thu, 3 Sep 2026 15:22:40 +0200 Subject: [PATCH 2/5] Specialised unsafe.asm to unsafe_mem.asm, and renamed unsafe.memcmp to unsafe.compare --- unsafe.asm => unsafe_mem.asm | 24 ++++++++++++------------ 1 file changed, 12 insertions(+), 12 deletions(-) rename unsafe.asm => unsafe_mem.asm (88%) diff --git a/unsafe.asm b/unsafe_mem.asm similarity index 88% rename from unsafe.asm rename to unsafe_mem.asm index e2ce635..7108db0 100644 --- a/unsafe.asm +++ b/unsafe_mem.asm @@ -1,4 +1,4 @@ -; int memcmp(uint8_t* a, uint8_t* b, size_t count); +; int compare(uint8_t* a, uint8_t* b, size_t count); ; Compares two memory segment of equal length lexicographically. ; Temporarily modifies the byte at address `a + count`. ; @@ -11,7 +11,7 @@ ; - `0` if both segments are equal. ; - `<0` if the first segment is less than the second segment. ; - `>0` if the first segment is greater than the second segment. -pub fn_memcmp: +pub compare: ; Exclusive end point of the second segment. add r4, r3, r2 ; Exclusive end point of the first segment. @@ -20,13 +20,13 @@ pub fn_memcmp: load_8 flags, [r4] ; Check if the first bytes behind the sequences are equal. cmp flags, r6 - jne _memcmp__loop + jne _compare__loop ; Change the byte directly behind the first segment. xor r4, r6, 1 - ; This would be problematic if someone calls memcmp with a first segment + ; This would be problematic if someone calls compare with a first segment ; whose end point overlaps the program memory of this function. store_8 [r3], r4 -_memcmp__loop: +_compare__loop: load_32 r4, [r1] add r1, r1, 4 load_32 r5, [r2] @@ -34,7 +34,7 @@ _memcmp__loop: ; Comparing two sequences of 4 bytes lexicographically is equivalent to ; comparing the corresponding big endian 32 bit words. cmp r4, r5 - je _memcmp__loop + je _compare__loop ; We overshot in the loop; decrement r1 again. (Only by 2, we backtrack the rest if necessary later) sub r1, r1, 2 ; Restore the byte we changed. @@ -45,24 +45,24 @@ _memcmp__loop: mov r5, flags ; Check if at least one of the two most significant bytes is not 0. cmp r4, 0xffff - jbe _memcmp__low2 + jbe _compare__low2 ; If it is, backtrack the remaining 2 indices. ; Shift the most significant bytes to the least significant ones. sub r1, r1, 2 lsr r4, r4, 16 -_memcmp__low2: +_compare__low2: ; r1 now points to a non-zero 16 bit value. ; If the 16 bit value at r1-2 is in-bounds, then it is 0. ; Check if the most significant byte of the 16 bit value is 0. cmp r4, 0xff - ja _memcmp__low1 + ja _compare__low1 ; If it is, our target is the least significant byte. add r1, r1, 1 -_memcmp__low1: +_compare__low1: ; Otherwise, the target is that non-zero byte. ; Check if the target is out of bounds, i.e. the loop terminated through the "bounds check". cmp r1, r3 - jae _memcmp__oob + jae _compare__oob ; r5 is the comparison result in the format of `cmp`. Convert it to the desired format. ; 00 => 0x40000000 > 0 ; 01 => 0x00000000 = 0 @@ -70,6 +70,6 @@ _memcmp__low1: xor r1, r5, 1 lsl r1, r1, 30 jmp r13 -_memcmp__oob: +_compare__oob: mov r1, 0 jmp r13 From 655ce0237eddaad69a9603175227075022b42aec Mon Sep 17 00:00:00 2001 From: MutexRaceCondition Date: Thu, 3 Sep 2026 18:48:57 +0200 Subject: [PATCH 3/5] Fixed unsafe_mem.compare not working correctly when both input pointers are the same --- unsafe_mem.asm | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/unsafe_mem.asm b/unsafe_mem.asm index 7108db0..446bfc5 100644 --- a/unsafe_mem.asm +++ b/unsafe_mem.asm @@ -12,6 +12,8 @@ ; - `<0` if the first segment is less than the second segment. ; - `>0` if the first segment is greater than the second segment. pub compare: + cmp r1, r2 + je _compare__is_eq ; Exclusive end point of the second segment. add r4, r3, r2 ; Exclusive end point of the first segment. @@ -62,7 +64,7 @@ _compare__low1: ; Otherwise, the target is that non-zero byte. ; Check if the target is out of bounds, i.e. the loop terminated through the "bounds check". cmp r1, r3 - jae _compare__oob + jae _compare__is_eq ; r5 is the comparison result in the format of `cmp`. Convert it to the desired format. ; 00 => 0x40000000 > 0 ; 01 => 0x00000000 = 0 @@ -70,6 +72,6 @@ _compare__low1: xor r1, r5, 1 lsl r1, r1, 30 jmp r13 -_compare__oob: +_compare__is_eq: mov r1, 0 jmp r13 From 257735d77628562f411511998e3a6cb2d6476d88 Mon Sep 17 00:00:00 2001 From: MutexRaceCondition Date: Thu, 3 Sep 2026 21:01:07 +0200 Subject: [PATCH 4/5] Remove leftover files --- mem.asm | 174 ----------------------------------------------------- unsafe.asm | 75 ----------------------- 2 files changed, 249 deletions(-) delete mode 100644 mem.asm delete mode 100644 unsafe.asm diff --git a/mem.asm b/mem.asm deleted file mode 100644 index bf82a2e..0000000 --- a/mem.asm +++ /dev/null @@ -1,174 +0,0 @@ -; int compare(uint8_t* a, uint8_t* b, size_t count); -; Compares two memory segment of equal length lexicographically. -; -; Arguments: -; - `r1`: A pointer to the first memory segment. -; - `r2`: A pointer to the second memory segment. -; - `r3`: The size of both memory segments. -; Results: -; - `r1`: -; - `0` if both segments are equal. -; - `<0` if the first segment is less than the second segment. -; - `>0` if the first segment is greater than the second segment. -; -pub compare: - ; Exclusive end point of the first segment. - add r3, r3, r1 - sub r3, r3, 4 -_compare__loop: - load_32 r4, [r1] - add r1, r1, 4 - load_32 r5, [r2] - add r2, r2, 4 - ; Comparing two sequences of 4 bytes lexicographically is equivalent to - ; comparing the corresponding big endian 32 bit words. - cmp r4, r5 - jne _compare__break - ; Check if there are enough bytes left to continue with the vectorized loop. - cmp r1, r3 - jbe _compare__loop - ; `r3 + 4 - r1 = = r3 - r1 mod 4` - sub flags, r3, r1 - ; Check if one of the lowest 2 bits is non-zero - jbe _compare__rem - ; If not, we are done. Both segments are equal. - mov r1, 0 - jmp r13 -_compare__break: - ; `flags` is the comparison result in the format of `cmp`. Convert it to the desired format. - ; 00 => 0x40000000 > 0 - ; 01 => 0x00000000 = 0 - ; 10 => 0xC0000000 < 0 - xor r1, flags, 1 - lsl r1, r1, 30 - jmp r13 -_compare__rem: - ; Compute `S = 8*(4 - )` and - ; [r1] >> S, [r2] >> S - mov r3, 8 - load_32 r4, [r1] - sub r3, r3, flags - load_32 r5, [r2] - lsl r3, r3, 3 - lsr r4, r4, r3 - lsr r5, r5, r3 - ; Compare both values, now with garbage bytes removed. - cmp r4, r5 - jmp _compare__break - - -; void copy(void* src, void* dest, size_t count); -; Copies `count` bytes from `src` to `dest`. The two memory segments must not overlap. -; -; Arguments: -; - `r1`: Pointer to the memory segment to be copied. -; - `r2`: Pointer to the memory segment to be copied into. -; - `r3`: Byte size of both the `src` and `dest` segments. -; -pub copy: - ; Exclusive end point of the source segment. - add r3, r1, r3 - ; Last index from where we can safely copy 8 bytes per loop iteration. - sub r3, r3, 8 - jmp _copy__loop_entry -_copy__loop: - ; Copy 8 bytes from `src` to `dest`. - load_32 flags, [r1] - add r1, r1, 4 - store_32 [r2], flags - add r2, r2, 4 - load_32 flags, [r1] - add r1, r1, 4 - store_32 [r2], flags - add r2, r2, 4 -_copy__loop_entry: - ; Check if we can process more data in the vectorized loop. - cmp r1, r3 - jbe _copy__loop - ; The remaining amount of bytes `R` is `R = r3 + 8 - r1 = r3 - r1 mod 8`. - sub flags, r3, r1 - ; Test if `R` is not a multiple of `4`, i.e. the lowest 2 bits are non-zero. - jbe _copy__rem - ; `R` is a multiple of `4`. Special case this. - ; Check if `R` is `0`, i.e. the third bit is also 0. In that case, we are already done. - ; There are no conditional indirect jumps, so we can't return immediately. - jge _copy__ret - ; `R = 4`. No need to update `r1` or `r2`, we don't need them anymore. - load_32 flags, [r1] - store_32 [r2], flags -_copy__ret: - ; Return - jmp r13 -_copy__rem: - ; Optimize the remaining cases for code size. - ; End point of the source segment. - add r3, r3, 8 - ; We already handled the case `R = 0` earlier, - ; so no bounds check needed for the first iteration. -_copy__rem_loop: - ; Copy 1 byte. - load_8 flags, [r1] - add r1, r1, 1 - store_8 [r2], flags - add r2, r2, 1 - ; Check if we are still within the bounds. - cmp r1, r3 - jb _copy__rem_loop - jmp r13 - -; void fill32(uint8_t* dest, size_t count, uint32_t value); -; Fills `count` bytes in `dest` with `value`. If `count` is not a multiple of 4, -; the least significant bytes of `value` are cut off for the last entry. -; -; Arguments: -; - `r1`: A pointer to the destination segment. -; - `r2`: The size of the destination segment. -; - `r3`: The 32 bit value that the segment is filled with. -; -pub fill32: - ; Exclusive end point of the destination segment. - add r2, r1, r2 - ; Last index from where we can safely write 8 bytes per loop iteration. - sub r2, r2, 8 - jmp _fill32__entry -_fill32__loop: - ; Set 8 bytes per loop iteraion. - store_32 [r1], r3 - add r1, r1, 4 - store_32 [r1], r3 - add r1, r1, 4 -_fill32__entry: - ; Check if we can process more data in the vectorized loop. - cmp r1, r2 - jbe _fill32__loop - ; The remaining amount of bytes `R` is `R = r2 + 8 - r1 = r2 - r1 mod 8`. - sub flags, r2, r1 - ; Check if the third bit of the remainder is cleared. - jge _fill32__r4 - ; Otherwise set 4 bytes. - store_32 [r1], r3 - add r1, r1, 4 -_fill32__r4: - ; Check if the two least significant bits of the remainder are zero. - ja _fill32__ret - ; Handle the remaining bytes `R` individually, in reverse order. - add r2, r2, 4 - ; `r1 + 4 - r2 = 4 - R`. - sub flags, r1, r2 - ; Exclusive end point of the destination segment. - add r2, r2, 4 - ; Shift out the least significant `8*(4 - R)` bits of the value. - lsl flags, flags, 3 - lsr r3, r3, flags - jmp _fill32__loop2_entry -_fill32__loop2: - sub r2, r2, 1 - ; Write the least significant byte of the value... - store_8 [r2], r3 - ; and then shift it out. - lsr r3, r3, 8 -_fill32__loop2_entry: - cmp r1, r2 - jb _fill32__loop2 -_fill32__ret: - jmp r13 diff --git a/unsafe.asm b/unsafe.asm deleted file mode 100644 index e2ce635..0000000 --- a/unsafe.asm +++ /dev/null @@ -1,75 +0,0 @@ -; int memcmp(uint8_t* a, uint8_t* b, size_t count); -; Compares two memory segment of equal length lexicographically. -; Temporarily modifies the byte at address `a + count`. -; -; Arguments: -; - `r1`: A pointer to the first memory segment. -; - `r2`: A pointer to the second memory segment. -; - `r3`: The size of both memory segments. -; Results: -; - `r1`: -; - `0` if both segments are equal. -; - `<0` if the first segment is less than the second segment. -; - `>0` if the first segment is greater than the second segment. -pub fn_memcmp: - ; Exclusive end point of the second segment. - add r4, r3, r2 - ; Exclusive end point of the first segment. - add r3, r3, r1 - load_8 r6, [r3] - load_8 flags, [r4] - ; Check if the first bytes behind the sequences are equal. - cmp flags, r6 - jne _memcmp__loop - ; Change the byte directly behind the first segment. - xor r4, r6, 1 - ; This would be problematic if someone calls memcmp with a first segment - ; whose end point overlaps the program memory of this function. - store_8 [r3], r4 -_memcmp__loop: - load_32 r4, [r1] - add r1, r1, 4 - load_32 r5, [r2] - add r2, r2, 4 - ; Comparing two sequences of 4 bytes lexicographically is equivalent to - ; comparing the corresponding big endian 32 bit words. - cmp r4, r5 - je _memcmp__loop - ; We overshot in the loop; decrement r1 again. (Only by 2, we backtrack the rest if necessary later) - sub r1, r1, 2 - ; Restore the byte we changed. - store_8 [r3], r6 - ; We encountered two different words. Figure out what byte they differ on. - xor r4, r4, r5 - ; Store the flags for later, to figure out the return value. - mov r5, flags - ; Check if at least one of the two most significant bytes is not 0. - cmp r4, 0xffff - jbe _memcmp__low2 - ; If it is, backtrack the remaining 2 indices. - ; Shift the most significant bytes to the least significant ones. - sub r1, r1, 2 - lsr r4, r4, 16 -_memcmp__low2: - ; r1 now points to a non-zero 16 bit value. - ; If the 16 bit value at r1-2 is in-bounds, then it is 0. - ; Check if the most significant byte of the 16 bit value is 0. - cmp r4, 0xff - ja _memcmp__low1 - ; If it is, our target is the least significant byte. - add r1, r1, 1 -_memcmp__low1: - ; Otherwise, the target is that non-zero byte. - ; Check if the target is out of bounds, i.e. the loop terminated through the "bounds check". - cmp r1, r3 - jae _memcmp__oob - ; r5 is the comparison result in the format of `cmp`. Convert it to the desired format. - ; 00 => 0x40000000 > 0 - ; 01 => 0x00000000 = 0 - ; 10 => 0xC0000000 < 0 - xor r1, r5, 1 - lsl r1, r1, 30 - jmp r13 -_memcmp__oob: - mov r1, 0 - jmp r13 From b97e17fa84235a81d81744d29c71027fd2e4d793 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Isalski?= Date: Fri, 4 Sep 2026 23:08:06 +0200 Subject: [PATCH 5/5] Changed mul_low algorithm to radix-16-1 algorithm with swap --- src/imath.asm | 64 ++++++++++++++++++++++++++++++++++----------------- 1 file changed, 43 insertions(+), 21 deletions(-) diff --git a/src/imath.asm b/src/imath.asm index 5d8fbd2..a246a2c 100644 --- a/src/imath.asm +++ b/src/imath.asm @@ -4,29 +4,51 @@ ; r2 - The second value ; Result: ; r1 - The lower 32 bits of the result -; Clobbers: r2, r3, r4, r5 +; Clobbers: r2, r3, r4, r5, r6 pub mul_low: - add r3, r2, r2 ; r3 = r2 * 2 + cmp r1, r2 + jbe mul_low_noswap + xor r1, r2, r1 + xor r2, r1, r2 + xor r1, r2, r1 + ; Passthrough + +; Multiplies r1 and r2, returning the lower part of the result +; This method assumes r1 is smaller than r2, which results in faster execution +; Arguments: +; r1 - The first value +; r2 - The second value +; Result: +; r1 - The lower 32 bits of the result +; Clobbers: r2, r3, r4, r5, r6 +pub mul_low_noswap: mov r4, 0 ; r4 has the result - - mul_low_loop: - - and r5, r1, 1 ; a0 - neg r5, r5 ; mask - and r5, r2, r5 ; a0 ? r2 : 0 - add r4, r4, r5 - - and r5, r1, 2 ; a1 is now 0 or 2 - lsr r5, r5, 1 ; normalize to 0/1 - neg r5, r5 ; mask - and r5, r3, r5 ; a1 ? r2 * 2 : 0 - add r4, r4, r5 - - lsl r2, r2, 2 - lsl r3, r3, 2 - lsr r1, r1, 2 - cmp r1, zr - jne mul_low_loop + mov r6, mul_loop_end + add r3, r2, r2 + + mul_loop: + and r5, r1, 14 ; Taking the first four bits, discarding the odd + lsl r5, r5, 1 ; 0000 - 0, 0001 - 2, 0010 - 4, 0011 - 8, etc. + sub r5, r6, r5 + jmp r5 + + add r4, r4, r3 + add r4, r4, r3 + add r4, r4, r3 + add r4, r4, r3 + add r4, r4, r3 + add r4, r4, r3 + add r4, r4, r3 + mul_loop_end: + mov flags, r1 + jne mul_skip_one + add r4, r4, r2 + mul_skip_one: + lsl r2, r2, 4 + lsl r3, r3, 4 + lsr r1, r1, 4 + cmp r1, zr + jne mul_loop mov r1, r4 jmp r13