diff --git a/src/mem.asm b/src/mem.asm new file mode 100644 index 0000000..bf82a2e --- /dev/null +++ b/src/mem.asm @@ -0,0 +1,174 @@ +; int compare(uint8_t* a, uint8_t* b, size_t count); +; Compares two memory segment of equal length lexicographically. +; +; Arguments: +; - `r1`: A pointer to the first memory segment. +; - `r2`: A pointer to the second memory segment. +; - `r3`: The size of both memory segments. +; Results: +; - `r1`: +; - `0` if both segments are equal. +; - `<0` if the first segment is less than the second segment. +; - `>0` if the first segment is greater than the second segment. +; +pub compare: + ; Exclusive end point of the first segment. + add r3, r3, r1 + sub r3, r3, 4 +_compare__loop: + load_32 r4, [r1] + add r1, r1, 4 + load_32 r5, [r2] + add r2, r2, 4 + ; Comparing two sequences of 4 bytes lexicographically is equivalent to + ; comparing the corresponding big endian 32 bit words. + cmp r4, r5 + jne _compare__break + ; Check if there are enough bytes left to continue with the vectorized loop. + cmp r1, r3 + jbe _compare__loop + ; `r3 + 4 - r1 = = r3 - r1 mod 4` + sub flags, r3, r1 + ; Check if one of the lowest 2 bits is non-zero + jbe _compare__rem + ; If not, we are done. Both segments are equal. + mov r1, 0 + jmp r13 +_compare__break: + ; `flags` is the comparison result in the format of `cmp`. Convert it to the desired format. + ; 00 => 0x40000000 > 0 + ; 01 => 0x00000000 = 0 + ; 10 => 0xC0000000 < 0 + xor r1, flags, 1 + lsl r1, r1, 30 + jmp r13 +_compare__rem: + ; Compute `S = 8*(4 - )` and + ; [r1] >> S, [r2] >> S + mov r3, 8 + load_32 r4, [r1] + sub r3, r3, flags + load_32 r5, [r2] + lsl r3, r3, 3 + lsr r4, r4, r3 + lsr r5, r5, r3 + ; Compare both values, now with garbage bytes removed. + cmp r4, r5 + jmp _compare__break + + +; void copy(void* src, void* dest, size_t count); +; Copies `count` bytes from `src` to `dest`. The two memory segments must not overlap. +; +; Arguments: +; - `r1`: Pointer to the memory segment to be copied. +; - `r2`: Pointer to the memory segment to be copied into. +; - `r3`: Byte size of both the `src` and `dest` segments. +; +pub copy: + ; Exclusive end point of the source segment. + add r3, r1, r3 + ; Last index from where we can safely copy 8 bytes per loop iteration. + sub r3, r3, 8 + jmp _copy__loop_entry +_copy__loop: + ; Copy 8 bytes from `src` to `dest`. + load_32 flags, [r1] + add r1, r1, 4 + store_32 [r2], flags + add r2, r2, 4 + load_32 flags, [r1] + add r1, r1, 4 + store_32 [r2], flags + add r2, r2, 4 +_copy__loop_entry: + ; Check if we can process more data in the vectorized loop. + cmp r1, r3 + jbe _copy__loop + ; The remaining amount of bytes `R` is `R = r3 + 8 - r1 = r3 - r1 mod 8`. + sub flags, r3, r1 + ; Test if `R` is not a multiple of `4`, i.e. the lowest 2 bits are non-zero. + jbe _copy__rem + ; `R` is a multiple of `4`. Special case this. + ; Check if `R` is `0`, i.e. the third bit is also 0. In that case, we are already done. + ; There are no conditional indirect jumps, so we can't return immediately. + jge _copy__ret + ; `R = 4`. No need to update `r1` or `r2`, we don't need them anymore. + load_32 flags, [r1] + store_32 [r2], flags +_copy__ret: + ; Return + jmp r13 +_copy__rem: + ; Optimize the remaining cases for code size. + ; End point of the source segment. + add r3, r3, 8 + ; We already handled the case `R = 0` earlier, + ; so no bounds check needed for the first iteration. +_copy__rem_loop: + ; Copy 1 byte. + load_8 flags, [r1] + add r1, r1, 1 + store_8 [r2], flags + add r2, r2, 1 + ; Check if we are still within the bounds. + cmp r1, r3 + jb _copy__rem_loop + jmp r13 + +; void fill32(uint8_t* dest, size_t count, uint32_t value); +; Fills `count` bytes in `dest` with `value`. If `count` is not a multiple of 4, +; the least significant bytes of `value` are cut off for the last entry. +; +; Arguments: +; - `r1`: A pointer to the destination segment. +; - `r2`: The size of the destination segment. +; - `r3`: The 32 bit value that the segment is filled with. +; +pub fill32: + ; Exclusive end point of the destination segment. + add r2, r1, r2 + ; Last index from where we can safely write 8 bytes per loop iteration. + sub r2, r2, 8 + jmp _fill32__entry +_fill32__loop: + ; Set 8 bytes per loop iteraion. + store_32 [r1], r3 + add r1, r1, 4 + store_32 [r1], r3 + add r1, r1, 4 +_fill32__entry: + ; Check if we can process more data in the vectorized loop. + cmp r1, r2 + jbe _fill32__loop + ; The remaining amount of bytes `R` is `R = r2 + 8 - r1 = r2 - r1 mod 8`. + sub flags, r2, r1 + ; Check if the third bit of the remainder is cleared. + jge _fill32__r4 + ; Otherwise set 4 bytes. + store_32 [r1], r3 + add r1, r1, 4 +_fill32__r4: + ; Check if the two least significant bits of the remainder are zero. + ja _fill32__ret + ; Handle the remaining bytes `R` individually, in reverse order. + add r2, r2, 4 + ; `r1 + 4 - r2 = 4 - R`. + sub flags, r1, r2 + ; Exclusive end point of the destination segment. + add r2, r2, 4 + ; Shift out the least significant `8*(4 - R)` bits of the value. + lsl flags, flags, 3 + lsr r3, r3, flags + jmp _fill32__loop2_entry +_fill32__loop2: + sub r2, r2, 1 + ; Write the least significant byte of the value... + store_8 [r2], r3 + ; and then shift it out. + lsr r3, r3, 8 +_fill32__loop2_entry: + cmp r1, r2 + jb _fill32__loop2 +_fill32__ret: + jmp r13 diff --git a/src/stdlib.asm b/src/stdlib.asm index e444aad..512c6c3 100644 --- a/src/stdlib.asm +++ b/src/stdlib.asm @@ -2,6 +2,7 @@ pub include bit pub include imath pub include array pub include console +pub include mem ; Needs to be last! pub include LUTs diff --git a/src/unsafe_mem.asm b/src/unsafe_mem.asm new file mode 100644 index 0000000..446bfc5 --- /dev/null +++ b/src/unsafe_mem.asm @@ -0,0 +1,77 @@ +; int compare(uint8_t* a, uint8_t* b, size_t count); +; Compares two memory segment of equal length lexicographically. +; Temporarily modifies the byte at address `a + count`. +; +; Arguments: +; - `r1`: A pointer to the first memory segment. +; - `r2`: A pointer to the second memory segment. +; - `r3`: The size of both memory segments. +; Results: +; - `r1`: +; - `0` if both segments are equal. +; - `<0` if the first segment is less than the second segment. +; - `>0` if the first segment is greater than the second segment. +pub compare: + cmp r1, r2 + je _compare__is_eq + ; Exclusive end point of the second segment. + add r4, r3, r2 + ; Exclusive end point of the first segment. + add r3, r3, r1 + load_8 r6, [r3] + load_8 flags, [r4] + ; Check if the first bytes behind the sequences are equal. + cmp flags, r6 + jne _compare__loop + ; Change the byte directly behind the first segment. + xor r4, r6, 1 + ; This would be problematic if someone calls compare with a first segment + ; whose end point overlaps the program memory of this function. + store_8 [r3], r4 +_compare__loop: + load_32 r4, [r1] + add r1, r1, 4 + load_32 r5, [r2] + add r2, r2, 4 + ; Comparing two sequences of 4 bytes lexicographically is equivalent to + ; comparing the corresponding big endian 32 bit words. + cmp r4, r5 + je _compare__loop + ; We overshot in the loop; decrement r1 again. (Only by 2, we backtrack the rest if necessary later) + sub r1, r1, 2 + ; Restore the byte we changed. + store_8 [r3], r6 + ; We encountered two different words. Figure out what byte they differ on. + xor r4, r4, r5 + ; Store the flags for later, to figure out the return value. + mov r5, flags + ; Check if at least one of the two most significant bytes is not 0. + cmp r4, 0xffff + jbe _compare__low2 + ; If it is, backtrack the remaining 2 indices. + ; Shift the most significant bytes to the least significant ones. + sub r1, r1, 2 + lsr r4, r4, 16 +_compare__low2: + ; r1 now points to a non-zero 16 bit value. + ; If the 16 bit value at r1-2 is in-bounds, then it is 0. + ; Check if the most significant byte of the 16 bit value is 0. + cmp r4, 0xff + ja _compare__low1 + ; If it is, our target is the least significant byte. + add r1, r1, 1 +_compare__low1: + ; Otherwise, the target is that non-zero byte. + ; Check if the target is out of bounds, i.e. the loop terminated through the "bounds check". + cmp r1, r3 + jae _compare__is_eq + ; r5 is the comparison result in the format of `cmp`. Convert it to the desired format. + ; 00 => 0x40000000 > 0 + ; 01 => 0x00000000 = 0 + ; 10 => 0xC0000000 < 0 + xor r1, r5, 1 + lsl r1, r1, 30 + jmp r13 +_compare__is_eq: + mov r1, 0 + jmp r13