30 Commits
Author SHA1 Message Date
MutexRaceCondition 3d874cebbd Merged with main 2026-09-03 15:17:00 +02:00
MutexRaceCondition 4aca17aa40 Moved memory operations into mem.asm (and renamed them) 2026-09-03 15:15:54 +02:00
ShatteredMINT 0188969dab Merge pull request 'Added read_line console function' (#10) from Micha_i/symphony_stdlib:console-functions into main
Reviewed-on: TCShenanigans/symphony_stdlib#10
2026-09-03 14:56:12 +02:00
Micha_i e4edb56d08 Merge branch 'main' into console-functions 2026-09-03 09:04:03 +02:00
Michał Isalski aa8eef7b1f Fixed at-address 2026-09-02 22:56:24 +02:00
ShatteredMINT 37d0e556d2 Merge pull request 'clean up confusion about function template' (#14) from meta-documentation into main
Reviewed-on: TCShenanigans/symphony_stdlib#14
2026-09-02 20:49:38 +02:00
ShatteredMINT 26d3a4d4f4 clean up confusion about function template 2026-09-02 20:49:22 +02:00
Michał Isalski 870b6ebc39 Merge branch 'main' into console-functions 2026-09-02 20:41:38 +02:00
Michal Isalski ca1ba67fd8 Added a shift conversion LUT and finished read_line (except Writeback) 2026-09-02 13:20:55 +02:00
ShatteredMINT 786adfdb1c Merge pull request 'Changed docs of math functions to conform to new guidelines' (#12) from Micha_i/symphony_stdlib:math-functions-docs into main
Reviewed-on: TCShenanigans/symphony_stdlib#12
2026-09-02 11:46:14 +02:00
Michał Isalski 84636799b6 Changed docs of math functions to conform to new guidelines 2026-09-02 11:44:30 +02:00
ShatteredMINT aa2cbe4ad8 Merge pull request 'Add basic meta documentation' (#11) from meta-documentation into main
Documentation efforts are ongoing but this should be good enough for now
2026-09-02 11:39:25 +02:00
ShatteredMINT 8301d5988a create teaching directory 2026-09-02 11:32:46 +02:00
ShatteredMINT dbd4865f14 change stack start 2026-09-02 11:31:52 +02:00
ShatteredMINT 8f1b970a2f remove mention of non existent file 2026-09-02 11:31:33 +02:00
ShatteredMINT ddca02d8a8 relax r7 requirement for result stack 2026-09-02 11:21:03 +02:00
ShatteredMINT a3d6cb64b5 remove duplicate documentation from stdlib.asm 2026-09-02 11:17:18 +02:00
Michał Isalski f9bb69f1f5 Added storing shift status 2026-09-02 02:03:00 +02:00
ShatteredMINT 9816111f38 clarify stack arguments 2026-09-01 14:08:20 +02:00
ShatteredMINT f2e5c830b3 explain inline comment 2026-09-01 13:48:12 +02:00
Michal Isalski 65caf8aff9 Made read_line compliant to new doc guidelines 2026-09-01 13:47:09 +02:00
ShatteredMINT 60e7b6209f basic contribution guidelines 2026-09-01 13:21:17 +02:00
ShatteredMINT 33b7886e70 add arrays to readme 2026-09-01 13:07:13 +02:00
Michal Isalski 1dca814388 One instruction less by using another register 2026-09-01 10:54:29 +02:00
Michal Isalski 9dfb8248a6 Added handling for skipping non-renderable characters
Added handling for backspace character
2026-09-01 10:51:31 +02:00
Michał Isalski 4b962a1f87 Added read_line function 2026-09-01 00:38:37 +02:00
ShatteredMINT daf4168129 format calling convention 2026-08-31 10:17:25 +02:00
ShatteredMINT 4bce47209a fix remaining links 2026-08-31 10:12:57 +02:00
ShatteredMINT af5a750031 link test 2026-08-31 10:10:10 +02:00
ShatteredMINT 8e0fdf0d28 start readme 2026-08-31 10:08:44 +02:00
8 changed files with 451 additions and 207 deletions
+25
View File
@@ -0,0 +1,25 @@
# Contributing
## Code of Conduct
Be nice, we are all just doing this to have fun
## General rules
- All text (names, comments, etc.) has to be in English
- You are responsible for ensuring that you have the rights for us to use the code you contribute to the project
- follow the guidelines, for code, documentation, etc.
- all code has to work with the standard symphony ISA
## Documenting Functions
All functions in the standard library should follow the following outline:
```
; <description>
; Arguments: <which register contains what argument>
; Result: <what is the result, and where is it stored>
; Clobbers: <list of registers that are clobbered>
<label>: <;SHOULD BE INLINED>
<CODE>
```
Functions should be in the appropriate asm file, if you are unsure where functionality fits make a seperate file and ask in the pull request
Functions that are provided for convenience/reference but should be inlined in production code should be marked with `;SHOULD BE INLINED` after their label
+100
View File
@@ -0,0 +1,100 @@
@0x10000 ; Example address until we get a proper memory map for this
; Shift conversion table
; It stores the mapping from value 32-127 of the ASCII table to their shifted equivalents (both ways) in the standard US keyboard layout
; e.g. 1 -> !
U8 32 ; Space -> Space
U8 49 ; ! -> 1
U8 39 ; " -> '
U8 51 ; # -> 3
U8 52 ; $ -> 4
U8 53 ; % -> 5
U8 55 ; & -> 7
U8 34 ; ' -> "
U8 57 ; ( -> 9
U8 48 ; ) -> 0
U8 56 ; * -> 8
U8 61 ; + -> =
U8 60 ; , -> <
U8 95 ; - -> _
U8 62 ; . -> >
U8 63 ; / -> ?
U8 41 ; 0 -> )
U8 33 ; 1 -> !
U8 64 ; 2 -> @
U8 35 ; 3 -> #
U8 36 ; 4 -> $
U8 37 ; 5 -> %
U8 94 ; 6 -> ^
U8 38 ; 7 -> &
U8 42 ; 8 -> *
U8 40 ; 9 -> (
U8 59 ; : -> ;
U8 58 ; ; -> :
U8 44 ; < -> ,
U8 43 ; = -> +
U8 46 ; > -> .
U8 47 ; ? -> /
U8 50 ; @ -> 2
U8 97 ; A -> a
U8 98 ; B -> b
U8 99 ; C -> c
U8 100; D -> d
U8 101; E -> e
U8 102; F -> f
U8 103; G -> g
U8 104; H -> h
U8 105; I -> i
U8 106; J -> j
U8 107; K -> k
U8 108; L -> l
U8 109; M -> m
U8 110; N -> n
U8 111; O -> o
U8 112; P -> p
U8 113; Q -> q
U8 114; R -> r
U8 115; S -> s
U8 116; T -> t
U8 117; U -> u
U8 118; V -> v
U8 119; W -> w
U8 120; X -> x
U8 121; Y -> y
U8 122; Z -> z
U8 123; [ -> {
U8 124; \ -> |
U8 125; ] -> }
U8 125; ^ -> 6
U8 45 ; _ -> -
U8 126; ` -> ~
U8 65 ; a -> A
U8 66 ; b -> B
U8 67 ; c -> C
U8 68 ; d -> D
U8 69 ; e -> E
U8 70 ; f -> F
U8 71 ; g -> G
U8 72 ; h -> H
U8 73 ; i -> I
U8 74 ; j -> J
U8 75 ; k -> K
U8 76 ; l -> L
U8 77 ; m -> M
U8 78 ; n -> N
U8 79 ; o -> O
U8 80 ; p -> P
U8 81 ; q -> Q
U8 82 ; r -> R
U8 83 ; s -> S
U8 84 ; t -> T
U8 85 ; u -> U
U8 86 ; v -> V
U8 87 ; w -> W
U8 88 ; x -> X
U8 89 ; y -> Y
U8 90 ; z -> Z
U8 91 ; { -> [
U8 92 ; | -> \
U8 93 ; } -> ]
U8 96 ; ~ -> `
+45 -2
View File
@@ -1,3 +1,46 @@
# symphony_stdlib
# Symphony Stdlib
standard library for symphony
This is a standard library for symphony.
It is both intended as a practical toolkit to develop more complex software as well as a teaching resource.
If you just want to use the standard library [[stdlib.asm]] is your main header, include it after your code.
If you are using it as a learning resource have a look at the [teaching folder](teaching).
If you are intersted in contributing have a look at [[CONTRIBUTING.md]]
---
## ABI
### Calling Convention
| class | registers |
| ----- | --------- |
| n.a. | zr |
| preserved | sp, r8 - r12 |
| scratch | flags, r1 - r7 |
| arguments | r1 - r7 |
| result | r1-r7 |
| return address | r13 |
Arguments not fitting into the 7 registers should be passed on the top of the stack, meaning they should be the last values pushed before the function call.
Arguments are passed in reverse order with the stack so:
lowest address = 1st stack arg
highest address = last stack arg
Should a function return more values than fit into the 7 registers, the caller has to allocate space on the stack for them, and pass the pointer to that space in the next free argument register.
This reduces the amount of argument registers to 6 and all arguments above that shall go on the stack, the pointer to the result stack shall **always** be passed in an argument register.
This register points at the highest available address for results, with the 8th result being stored there, the 9th below it and so on.
### Stack
Grows downwards from 0xXXFE_0000 (so top of memory -0x1_0000).
### Types
#### String
Strings are stored in memory as null terminated sequences of bytes encoding ascii characters.
They should be passed by reference.
#### Array
Arrays are stored in memory with a reference to them being the tuple (pointer, length) stored in a register pair.
Array elements may only have a size of 8/16/32 bits
+69
View File
@@ -0,0 +1,69 @@
; Reads a line from the keyboard and fills the specified buffer with it
; Does not support Shift or any other special keys
; Arguments:
; r1 - pointer to the buffer
; Result:
; r1 - pointer to the same buffer
; Clobbers: r2, r3, r4, r5, r6, r7
pub read_line:
mov r6, 1
lsl r6, r6, 16
sub r6, r6, 32 ; Calculating the address to the shift LUT
mov r2, 0 ; Storing the shift status here
mov r4, 0 ; Storing the last key here, so we don't repeat the same key
mov r3, r1 ; The pointer to after the last character
read_line_keyloop:
keyboard r5
cmp r5, r4
je read_line_keyloop ; If the current key is same as previous, we loop
mov r4, r5 ; Storing current key as previous
cmp r5, 0x120 ; Is the key renderable or special?
jb read_line_special ; If the key was special, we handle it separately
xor r5, r5, 0x100 ; Clearing the "down" bit
cmp r2, zr ; Checking for shift status
je read_line_store ; If shift is up, we skip conversion
;;; converting from shift-down to shift-up keys
add r7, r6, r5 ; Calculating the index of the shift conversion
load_8 r5, [r7] ; Loading the shifted value
;;;
read_line_store:
store_8 [r3], r5 ; Else, we store the key in the buffer
add r3, r3, 1 ; We advance forward
; TODO: Writeback
jmp read_line_keyloop
read_line_special:
cmp r5, 13 ; Was the key Backspace?
je read_line_backspace ; If yes we need to move one character back
cmp r5, 10 ; Was the key Enter?
je read_line_finished ; If so, we're finished
and r5, r5, 0x1FB ; Mask out the left/right shift direction bit
cmp r5, 0x110 ; Was the key Shift Down?
and flags, flags, 0x1 ; We care only about equality bit
or r2, r2, flags ; If shift was down before, it still is. If it was pressed now, it is down now
cmp r5, 0x010 ; Was the key Shift Up?
and flags, flags, 0x1 ; We care only about equality bit
xor flags, flags, 0x1 ; We invert it, i.e. "if it's not up"
and r2, r2, flags ; The shift can be kept down if it's not currently up
jmp read_line_keyloop ; If no special handling, we loop back
read_line_backspace:
cmp r3, r1 ; Compare the current pointer to start of buffer
je read_line_keyloop ; If we are at the start, we loop
sub r3, r3, 1 ; We move back one character
; TODO: Writeback
jmp read_line_keyloop
read_line_finished:
store_8 [r3], zr ; We store null at the end so the string is finished
jmp r13
+29 -8
View File
@@ -1,8 +1,15 @@
; Multiplies r1 and r2, returning the lower part of the result
; Arguments:
; r1 - The first value
; r2 - The second value
; Result:
; r1 - The lower 32 bits of the result
; Clobbers: r2, r3, r4, r5
pub mul_low:
mov r3, 0 ; result
mov r4, 31 ; loop counter
mull_loop:
mul_low_loop:
asr r5, r2, 31
and r5, r5, r1
lsl r5, r5, r4
@@ -10,14 +17,18 @@ pub mul_low:
lsl r2, r2, 1
sub r4, r4, 1
cmp r4, 0
jge mull_loop
jge mul_low_loop
mov r1, r3
jmp r13
; Calculates the absolute value of the value provided in the r1 register
; Based on Stanford's BitHacks
; Clobbers r2
; Arguments:
; r1 - The value for which we want the absolute value
; Result:
; r1 - The calculated absolute value
; Clobbers: r2
; Info: Based on Stanford's BitHacks
pub abs: ; SHOULD BE INLINED
; mask = v >> 31
asr r2, r1, 31
@@ -28,8 +39,13 @@ pub abs: ; SHOULD BE INLINED
jmp r13
; Calculates the minimum value of the two values provided in the r1 and r2 registers
; Based on Stanford's BitHacks
; Clobbers flags
; Arguments:
; r1 - The first value
; r2 - The second value
; Result:
; r1 - The smaller value
; Clobbers: Nothing
; Info: Based on Stanford's BitHacks
pub min: ; SHOULD BE INLINED
; x < y
cmp r1, r2
@@ -45,8 +61,13 @@ pub min: ; SHOULD BE INLINED
jmp r13
; Calculates the maximum value of the two values provided in the r1 and r2 registers
; Based on Stanford's BitHacks
; Clobbers r2 and flags
; Arguments:
; r1 - The first value
; r2 - The second value
; Result:
; r1 - The smaller value
; Clobbers: r2
; Info: Based on Stanford's BitHacks
pub max: ; SHOULD BE INLINED
; x < y
cmp r1, r2
+174
View File
@@ -0,0 +1,174 @@
; int compare(uint8_t* a, uint8_t* b, size_t count);
; Compares two memory segment of equal length lexicographically.
;
; Arguments:
; - `r1`: A pointer to the first memory segment.
; - `r2`: A pointer to the second memory segment.
; - `r3`: The size of both memory segments.
; Results:
; - `r1`:
; - `0` if both segments are equal.
; - `<0` if the first segment is less than the second segment.
; - `>0` if the first segment is greater than the second segment.
;
pub compare:
; Exclusive end point of the first segment.
add r3, r3, r1
sub r3, r3, 4
_compare__loop:
load_32 r4, [r1]
add r1, r1, 4
load_32 r5, [r2]
add r2, r2, 4
; Comparing two sequences of 4 bytes lexicographically is equivalent to
; comparing the corresponding big endian 32 bit words.
cmp r4, r5
jne _compare__break
; Check if there are enough bytes left to continue with the vectorized loop.
cmp r1, r3
jbe _compare__loop
; `r3 + 4 - r1 = <remaining byte count> = r3 - r1 mod 4`
sub flags, r3, r1
; Check if one of the lowest 2 bits is non-zero
jbe _compare__rem
; If not, we are done. Both segments are equal.
mov r1, 0
jmp r13
_compare__break:
; `flags` is the comparison result in the format of `cmp`. Convert it to the desired format.
; 00 => 0x40000000 > 0
; 01 => 0x00000000 = 0
; 10 => 0xC0000000 < 0
xor r1, flags, 1
lsl r1, r1, 30
jmp r13
_compare__rem:
; Compute `S = 8*(4 - <remaining byte count>)` and
; [r1] >> S, [r2] >> S
mov r3, 8
load_32 r4, [r1]
sub r3, r3, flags
load_32 r5, [r2]
lsl r3, r3, 3
lsr r4, r4, r3
lsr r5, r5, r3
; Compare both values, now with garbage bytes removed.
cmp r4, r5
jmp _compare__break
; void copy(void* src, void* dest, size_t count);
; Copies `count` bytes from `src` to `dest`. The two memory segments must not overlap.
;
; Arguments:
; - `r1`: Pointer to the memory segment to be copied.
; - `r2`: Pointer to the memory segment to be copied into.
; - `r3`: Byte size of both the `src` and `dest` segments.
;
pub copy:
; Exclusive end point of the source segment.
add r3, r1, r3
; Last index from where we can safely copy 8 bytes per loop iteration.
sub r3, r3, 8
jmp _copy__loop_entry
_copy__loop:
; Copy 8 bytes from `src` to `dest`.
load_32 flags, [r1]
add r1, r1, 4
store_32 [r2], flags
add r2, r2, 4
load_32 flags, [r1]
add r1, r1, 4
store_32 [r2], flags
add r2, r2, 4
_copy__loop_entry:
; Check if we can process more data in the vectorized loop.
cmp r1, r3
jbe _copy__loop
; The remaining amount of bytes `R` is `R = r3 + 8 - r1 = r3 - r1 mod 8`.
sub flags, r3, r1
; Test if `R` is not a multiple of `4`, i.e. the lowest 2 bits are non-zero.
jbe _copy__rem
; `R` is a multiple of `4`. Special case this.
; Check if `R` is `0`, i.e. the third bit is also 0. In that case, we are already done.
; There are no conditional indirect jumps, so we can't return immediately.
jge _copy__ret
; `R = 4`. No need to update `r1` or `r2`, we don't need them anymore.
load_32 flags, [r1]
store_32 [r2], flags
_copy__ret:
; Return
jmp r13
_copy__rem:
; Optimize the remaining cases for code size.
; End point of the source segment.
add r3, r3, 8
; We already handled the case `R = 0` earlier,
; so no bounds check needed for the first iteration.
_copy__rem_loop:
; Copy 1 byte.
load_8 flags, [r1]
add r1, r1, 1
store_8 [r2], flags
add r2, r2, 1
; Check if we are still within the bounds.
cmp r1, r3
jb _copy__rem_loop
jmp r13
; void fill32(uint8_t* dest, size_t count, uint32_t value);
; Fills `count` bytes in `dest` with `value`. If `count` is not a multiple of 4,
; the least significant bytes of `value` are cut off for the last entry.
;
; Arguments:
; - `r1`: A pointer to the destination segment.
; - `r2`: The size of the destination segment.
; - `r3`: The 32 bit value that the segment is filled with.
;
pub fill32:
; Exclusive end point of the destination segment.
add r2, r1, r2
; Last index from where we can safely write 8 bytes per loop iteration.
sub r2, r2, 8
jmp _fill32__entry
_fill32__loop:
; Set 8 bytes per loop iteraion.
store_32 [r1], r3
add r1, r1, 4
store_32 [r1], r3
add r1, r1, 4
_fill32__entry:
; Check if we can process more data in the vectorized loop.
cmp r1, r2
jbe _fill32__loop
; The remaining amount of bytes `R` is `R = r2 + 8 - r1 = r2 - r1 mod 8`.
sub flags, r2, r1
; Check if the third bit of the remainder is cleared.
jge _fill32__r4
; Otherwise set 4 bytes.
store_32 [r1], r3
add r1, r1, 4
_fill32__r4:
; Check if the two least significant bits of the remainder are zero.
ja _fill32__ret
; Handle the remaining bytes `R` individually, in reverse order.
add r2, r2, 4
; `r1 + 4 - r2 = 4 - R`.
sub flags, r1, r2
; Exclusive end point of the destination segment.
add r2, r2, 4
; Shift out the least significant `8*(4 - R)` bits of the value.
lsl flags, flags, 3
lsr r3, r3, flags
jmp _fill32__loop2_entry
_fill32__loop2:
sub r2, r2, 1
; Write the least significant byte of the value...
store_8 [r2], r3
; and then shift it out.
lsr r3, r3, 8
_fill32__loop2_entry:
cmp r1, r2
jb _fill32__loop2
_fill32__ret:
jmp r13
+4 -197
View File
@@ -1,200 +1,7 @@
; ===== INTRODUCTION =====
; This is supposed to provide some standard library functionality for stock symphony.
; In particular its supposed to work with an unmodified ISA, that means some choices are not
; optimal (RA being stored in flags for example)
; ===== ABI =====
; ----- CALLING CONVENTION -----
; n.a. zr
; preserved: sp, r8 - r12
; scratch: flags, r1 - r7
; arguments: r1 - r7 (r1 = 1st argument, r6 = 6th arg/stack args, r7 = 7th arg/stack res)
; result: r1, r2 (r1 = low word, r2 = high word)
; return address: r13
; ----- STACK -----
; grows downwards from top of memory
; arguments are passed in reverse order with the stack so:
; lowest address = 1st stack arg
; highest address = last stack arg
; ===== TYPES =====
pub include bit
pub include imath
pub include mem
pub include console
; int memcmp(uint8_t* a, uint8_t* b, size_t count);
; Compares two memory segment of equal length lexicographically.
;
; Arguments:
; - `r1`: A pointer to the first memory segment.
; - `r2`: A pointer to the second memory segment.
; - `r3`: The size of both memory segments.
; Results:
; - `r1`:
; - `0` if both segments are equal.
; - `<0` if the first segment is less than the second segment.
; - `>0` if the first segment is greater than the second segment.
;
pub fn_memcmp:
; Exclusive end point of the first segment.
add r3, r3, r1
sub r3, r3, 4
_memcmp__loop:
load_32 r4, [r1]
add r1, r1, 4
load_32 r5, [r2]
add r2, r2, 4
; Comparing two sequences of 4 bytes lexicographically is equivalent to
; comparing the corresponding big endian 32 bit words.
cmp r4, r5
jne _memcmp__break
; Check if there are enough bytes left to continue with the vectorized loop.
cmp r1, r3
jbe _memcmp__loop
; `r3 + 4 - r1 = <remaining byte count> = r3 - r1 mod 4`
sub flags, r3, r1
; Check if one of the lowest 2 bits is non-zero
jbe _memcmp__rem
; If not, we are done. Both segments are equal.
mov r1, 0
jmp r13
_memcmp__break:
; `flags` is the comparison result in the format of `cmp`. Convert it to the desired format.
; 00 => 0x40000000 > 0
; 01 => 0x00000000 = 0
; 10 => 0xC0000000 < 0
xor r1, flags, 1
lsl r1, r1, 30
jmp r13
_memcmp__rem:
; Compute `S = 8*(4 - <remaining byte count>)` and
; [r1] >> S, [r2] >> S
mov r3, 8
load_32 r4, [r1]
sub r3, r3, flags
load_32 r5, [r2]
lsl r3, r3, 3
lsr r4, r4, r3
lsr r5, r5, r3
; Compare both values, now with garbage bytes removed.
cmp r4, r5
jmp _memcmp__break
; void memcpy(void* src, void* dest, size_t count);
; Copies `count` bytes from `src` to `dest`. The two memory segments must not overlap.
;
; Arguments:
; - `r1`: Pointer to the memory segment to be copied.
; - `r2`: Pointer to the memory segment to be copied into.
; - `r3`: Byte size of both the `src` and `dest` segments.
;
pub fn_memcpy:
; Exclusive end point of the source segment.
add r3, r1, r3
; Last index from where we can safely copy 8 bytes per loop iteration.
sub r3, r3, 8
jmp _memcpy__loop_entry
_memcpy__loop:
; Copy 8 bytes from `src` to `dest`.
load_32 flags, [r1]
add r1, r1, 4
store_32 [r2], flags
add r2, r2, 4
load_32 flags, [r1]
add r1, r1, 4
store_32 [r2], flags
add r2, r2, 4
_memcpy__loop_entry:
; Check if we can process more data in the vectorized loop.
cmp r1, r3
jbe _memcpy__loop
; The remaining amount of bytes `R` is `R = r3 + 8 - r1 = r3 - r1 mod 8`.
sub flags, r3, r1
; Test if `R` is not a multiple of `4`, i.e. the lowest 2 bits are non-zero.
jbe _memcpy__rem
; `R` is a multiple of `4`. Special case this.
; Check if `R` is `0`, i.e. the third bit is also 0. In that case, we are already done.
; There are no conditional indirect jumps, so we can't return immediately.
jge _memcpy__ret
; `R = 4`. No need to update `r1` or `r2`, we don't need them anymore.
load_32 flags, [r1]
store_32 [r2], flags
_memcpy__ret:
; Return
jmp r13
_memcpy__rem:
; Optimize the remaining cases for code size.
; End point of the source segment.
add r3, r3, 8
; We already handled the case `R = 0` earlier,
; so no bounds check needed for the first iteration.
_memcpy__rem_loop:
; Copy 1 byte.
load_8 flags, [r1]
add r1, r1, 1
store_8 [r2], flags
add r2, r2, 1
; Check if we are still within the bounds.
cmp r1, r3
jb _memcpy__rem_loop
jmp r13
; void memset32(uint8_t* dest, size_t count, uint32_t value);
; Fills `count` bytes in `dest` with `value`. If `count` is not a multiple of 4,
; the least significant bytes of `value` are cut off for the last entry.
;
; Arguments:
; - `r1`: A pointer to the destination segment.
; - `r2`: The size of the destination segment.
; - `r3`: The 32 bit value that the segment is filled with.
;
pub fn_memset32:
; Exclusive end point of the destination segment.
add r2, r1, r2
; Last index from where we can safely write 8 bytes per loop iteration.
sub r2, r2, 8
jmp _memset32__entry
_memset32__loop:
; Set 8 bytes per loop iteraion.
store_32 [r1], r3
add r1, r1, 4
store_32 [r1], r3
add r1, r1, 4
_memset32__entry:
; Check if we can process more data in the vectorized loop.
cmp r1, r2
jbe _memset32__loop
; The remaining amount of bytes `R` is `R = r2 + 8 - r1 = r2 - r1 mod 8`.
sub flags, r2, r1
; Check if the third bit of the remainder is cleared.
jge _memset32__r4
; Otherwise set 4 bytes.
store_32 [r1], r3
add r1, r1, 4
_memset32__r4:
; Check if the two least significant bits of the remainder are zero.
ja _memset32__ret
; Handle the remaining bytes `R` individually, in reverse order.
add r2, r2, 4
; `r1 + 4 - r2 = 4 - R`.
sub flags, r1, r2
; Exclusive end point of the destination segment.
add r2, r2, 4
; Shift out the least significant `8*(4 - R)` bits of the value.
lsl flags, flags, 3
lsr r3, r3, flags
jmp _memset32__loop2_entry
_memset32__loop2:
sub r2, r2, 1
; Write the least significant byte of the value...
store_8 [r2], r3
; and then shift it out.
lsr r3, r3, 8
_memset32__loop2_entry:
cmp r1, r2
jb _memset32__loop2
_memset32__ret:
jmp r13
; Needs to be last!
pub include LUTs
+5
View File
@@ -0,0 +1,5 @@
# Teaching
This is a collection of teaching advice regarding the stdlib.
**WIP**