From b97e17fa84235a81d81744d29c71027fd2e4d793 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Isalski?= Date: Fri, 4 Sep 2026 23:08:06 +0200 Subject: [PATCH] Changed mul_low algorithm to radix-16-1 algorithm with swap --- src/imath.asm | 64 ++++++++++++++++++++++++++++++++++----------------- 1 file changed, 43 insertions(+), 21 deletions(-) diff --git a/src/imath.asm b/src/imath.asm index 5d8fbd2..a246a2c 100644 --- a/src/imath.asm +++ b/src/imath.asm @@ -4,29 +4,51 @@ ; r2 - The second value ; Result: ; r1 - The lower 32 bits of the result -; Clobbers: r2, r3, r4, r5 +; Clobbers: r2, r3, r4, r5, r6 pub mul_low: - add r3, r2, r2 ; r3 = r2 * 2 + cmp r1, r2 + jbe mul_low_noswap + xor r1, r2, r1 + xor r2, r1, r2 + xor r1, r2, r1 + ; Passthrough + +; Multiplies r1 and r2, returning the lower part of the result +; This method assumes r1 is smaller than r2, which results in faster execution +; Arguments: +; r1 - The first value +; r2 - The second value +; Result: +; r1 - The lower 32 bits of the result +; Clobbers: r2, r3, r4, r5, r6 +pub mul_low_noswap: mov r4, 0 ; r4 has the result - - mul_low_loop: - - and r5, r1, 1 ; a0 - neg r5, r5 ; mask - and r5, r2, r5 ; a0 ? r2 : 0 - add r4, r4, r5 - - and r5, r1, 2 ; a1 is now 0 or 2 - lsr r5, r5, 1 ; normalize to 0/1 - neg r5, r5 ; mask - and r5, r3, r5 ; a1 ? r2 * 2 : 0 - add r4, r4, r5 - - lsl r2, r2, 2 - lsl r3, r3, 2 - lsr r1, r1, 2 - cmp r1, zr - jne mul_low_loop + mov r6, mul_loop_end + add r3, r2, r2 + + mul_loop: + and r5, r1, 14 ; Taking the first four bits, discarding the odd + lsl r5, r5, 1 ; 0000 - 0, 0001 - 2, 0010 - 4, 0011 - 8, etc. + sub r5, r6, r5 + jmp r5 + + add r4, r4, r3 + add r4, r4, r3 + add r4, r4, r3 + add r4, r4, r3 + add r4, r4, r3 + add r4, r4, r3 + add r4, r4, r3 + mul_loop_end: + mov flags, r1 + jne mul_skip_one + add r4, r4, r2 + mul_skip_one: + lsl r2, r2, 4 + lsl r3, r3, 4 + lsr r1, r1, 4 + cmp r1, zr + jne mul_loop mov r1, r4 jmp r13