2 Commits
2 changed files with 43 additions and 277 deletions
-256
View File
@@ -1,256 +0,0 @@
include errno
; Performs a Bitblock transfer, copying a section of the Source to the Destination, applying a given Mode operation
; Arguments:
; r1 - Destination Buffer pointer
; r2 - Source Buffer pointer
; r3 - X destination coordinate
; r4 - Y destination coordinate
; r5 - X source coordinate
; r6 - Y source coordinate
; r7 - Mode:
; 000 - SRCCOPY (copies source over destination)
; 001 - XOR
; 010 - AND
; 011 - NOT
; 100 - OR
; 101 - reserved
; 110 - reserved
; 111 - SRCALPHA (copies source over destination if: source != 0 (for 8bpp), source.alpha != 0 (for 32bpp))
; Stack argument 1: Width (in bytes)
; Stack argument 2: Height (in bytes)
; Result: None
; Clobbers: r1, r2, r3, r4, r5, r6, r7
; Errors: BUFFER_DEPTH_MISMATCH - if the buffers have different bit depths
pub bitblit:
push r8
push r9
; Checking if the bit depths are correct
add r8, r1, 6 ; r8 = pointer to destination depth
load_16 r8, [r8] ; r8 = dest depth
add r9, r2, 6 ; r9 = pointer to source depth
load_16 r9, [r9] ; r9 = src depth
cmp r8, r9
je bitblit_depth_good
; Bit depths differ - error
pop r9
pop r8
add sp, sp, 8 ; Removing the stack Arguments
mov flags, errno.BUFFER_DEPTH_MISMATCH
jmp r13
bitblit_depth_good:
push r8
add sp, sp, 12
load_32 r8, [sp] ; r8 = height
add sp, sp, 4
load_32 r9, [sp] ; r9 = width
sub sp, sp, 16
push r10
push r11
push r12
push r13
add r10, r1, 4 ; r10 = pointer to destination stride
load_16 r10, [r10] ; r10 = destination stride
add r11, r2, 4 ; r11 = pointer to source stride
load_16 r11, [r11] ; r11 = source stride
; Calculating destination index
push r10
mov r12, 0
bitblit_destination_index:
cmp r4, zr
je bitblit_after_destination_index
mov flags, r4
jne bitblit_destination_index_afteradd
add r12, r12, r10
bitblit_destination_index_afteradd:
lsr r4, r4, 1
lsl r10, r10, 1
jmp bitblit_destination_index
bitblit_after_destination_index:
add r4, r4, r12 ; Now r4 is dest_y * dest_stride
load_32 r1, [r1] ; r1 = destination data pointer
add r1, r1, r4 ; Now r1 is dest_pointer + dest_y * dest_stride
add r1, r1, r3 ; Now r1 is dest_pointer + dest_y * dest_stride + dest_x
pop r10
push r11
; Calculating source index
mov r12, 0
bitblit_source_index:
cmp r6, zr
je bitblit_after_source_index
mov flags, r6
jne bitblit_source_index_afteradd
add r12, r12, r11
bitblit_source_index_afteradd:
lsr r6, r6, 1
lsl r11, r11, 1
jmp bitblit_source_index
bitblit_after_source_index:
add r6, r6, r12 ; Now r6 is src_y * src_stride
load_32 r2, [r2] ; r2 = source data pointer
add r2, r2, r6 ; Now r2 is src_pointer + src_y * src_stride
add r2, r2, r5 ; Now r2 is src_pointer + src_y * src_stride + src_x
pop r11
; r3, r4, r5, r6 are now free
mov r3, bitblit_mode
mov r12, bitblit_end_mode
lsl r7, r7, 3 ; 2 instructions per mode
add r12, r7, r12
add r7, r7, r3
; r1 is line dest pointer (or bit depth)
; r2 is line src pointer (or temp)
; r3 is width iterator
; r4 is current dest pointer
; r5 is current src pointer
; r6 is current src value
; r7 is 32bit mode jump table destination
; r8 is height iterator
; r9 is width of bitblit
; r10 is dest stride
; r11 is src stride
; r12 is 8bit mode jump table destination
; r13 is current dest value
bitblit_copy_height:
cmp r8, zr
je bitblit_finish
bitblit_copy_line:
mov r3, 0
mov r4, r1 ; r4 = dest
mov r5, r2 ; r5 = src
push r1
add r1, sp, 20 ; Pointing to bit depth saved on stack
load_32 r1, [r1] ; r1 = bit depth
push r2
bitblit_copy_line_loop: ; Main loop that copies 4 bytes at a time
add r3, r3, 4
cmp r9, r3
je bitblit_copy_line_finish
jb bitblit_copy_line_end
load_32 r6, [r5] ; r6 = [src]
load_32 r13, [r4] ; r13 = [dest]
jmp r7
bitblit_mode:
nop ; SRCCOPY
jmp bitblit_save_bits
xor r6, r6, r13 ; XOR
jmp bitblit_save_bits
and r6, r6, r13 ; AND
jmp bitblit_save_bits
not r6, r6 ; NOT
jmp bitblit_save_bits
or r6, r6, r13 ; OR
jmp bitblit_save_bits
nop ; reserved
jmp bitblit_save_bits
nop ; reserved
jmp bitblit_save_bits
cmp r1, 8 ; SRCALPHA
jne bitblit_srcalpha_32 ; If bit depth is 32, we jump to alpha checking
; If bit depth is 8, we need to repack this value
mov r2, 0 ; our mask
lsr flags, r6, 24
cmp flags, zr
jne bitblit_srcalpha_8_1
or r2, r2, 0xFF
bitblit_srcalpha_8_1:
lsl r2, r2, 8
lsr flags, r6, 16
and flags, flags, 0xFF
cmp flags, zr
jne bitblit_srcalpha_8_2
or r2, r2, 0xFF
bitblit_srcalpha_8_2:
lsl r2, r2, 8
lsr flags, r6, 8
and flags, flags, 0xFF
cmp flags, zr
jne bitblit_srcalpha_8_3
or r2, r2, 0xFF
bitblit_srcalpha_8_3:
lsl r2, r2, 8
and flags, r6, 0xFF
cmp flags, zr
jne bitblit_srcalpha_8_4
or r2, r2, 0xFF
bitblit_srcalpha_8_4:
and r13, r13, r2
not r2, r2
and r6, r6, r2
or r6, r6, r13
jmp bitblit_save_bits
bitblit_srcalpha_32:
and flags, r6, 0xFF
cmp flags, zr
jne bitblit_save_bits ; If alpha channel is not zero, we save the value
mov r6, r13 ; If it was zero, we take [dest], i.e. don't overwrite
bitblit_save_bits:
store_32 [r4], r6
add r4, r4, 4
add r5, r5, 4
jmp bitblit_copy_line_loop
bitblit_copy_line_end: ; Now we need to handle up to 3 bytes that were not copied
sub r3, r3, 4
bitblit_copy_line_end_loop:
load_8 r6, [r5] ; r6 = [src]
load_8 r13, [r4] ; r13 = [dest]
jmp r12
bitblit_end_mode:
nop ; SRCCOPY
jmp bitblit_save_bits_end
xor r6, r6, r13 ; XOR
jmp bitblit_save_bits_end
and r6, r6, r13 ; AND
jmp bitblit_save_bits_end
not r6, r6 ; NOT
jmp bitblit_save_bits_end
or r6, r6, r13 ; OR
jmp bitblit_save_bits_end
nop ; reserved
jmp bitblit_save_bits_end
nop ; reserved
jmp bitblit_save_bits_end
cmp r6, zr ; SRCALPHA (will happen only in 8bpp)
jne bitblit_save_bits_end
mov r6, r13 ; [src] = 0, so we take [dest], i.e. don't overwrite
bitblit_save_bits_end:
store_8 [r4], r6
add r3, r3, 1
add r4, r4, 1
add r5, r5, 1
cmp r9, r3
ja bitblit_copy_line_end_loop
bitblit_copy_line_finish:
pop r2
pop r1
add r1, r1, r10
add r2, r2, r11 ; Moving dest & src to next line
sub r8, r8, 1
jmp bitblit_copy_height
bitblit_finish:
pop r13
pop r12
pop r11
pop r10
add sp, sp, 4 ; Bit depth skip
pop r9
pop r8
add sp, sp, 8
mov flags, errno.OK
jmp r13
+43 -21
View File
@@ -4,29 +4,51 @@
; r2 - The second value
; Result:
; r1 - The lower 32 bits of the result
; Clobbers: r2, r3, r4, r5
; Clobbers: r2, r3, r4, r5, r6
pub mul_low:
add r3, r2, r2 ; r3 = r2 * 2
cmp r1, r2
jbe mul_low_noswap
xor r1, r2, r1
xor r2, r1, r2
xor r1, r2, r1
; Passthrough
; Multiplies r1 and r2, returning the lower part of the result
; This method assumes r1 is smaller than r2, which results in faster execution
; Arguments:
; r1 - The first value
; r2 - The second value
; Result:
; r1 - The lower 32 bits of the result
; Clobbers: r2, r3, r4, r5, r6
pub mul_low_noswap:
mov r4, 0 ; r4 has the result
mul_low_loop:
and r5, r1, 1 ; a0
neg r5, r5 ; mask
and r5, r2, r5 ; a0 ? r2 : 0
add r4, r4, r5
and r5, r1, 2 ; a1 is now 0 or 2
lsr r5, r5, 1 ; normalize to 0/1
neg r5, r5 ; mask
and r5, r3, r5 ; a1 ? r2 * 2 : 0
add r4, r4, r5
lsl r2, r2, 2
lsl r3, r3, 2
lsr r1, r1, 2
cmp r1, zr
jne mul_low_loop
mov r6, mul_loop_end
add r3, r2, r2
mul_loop:
and r5, r1, 14 ; Taking the first four bits, discarding the odd
lsl r5, r5, 1 ; 0000 - 0, 0001 - 2, 0010 - 4, 0011 - 8, etc.
sub r5, r6, r5
jmp r5
add r4, r4, r3
add r4, r4, r3
add r4, r4, r3
add r4, r4, r3
add r4, r4, r3
add r4, r4, r3
add r4, r4, r3
mul_loop_end:
mov flags, r1
jne mul_skip_one
add r4, r4, r2
mul_skip_one:
lsl r2, r2, 4
lsl r3, r3, 4
lsr r1, r1, 4
cmp r1, zr
jne mul_loop
mov r1, r4
jmp r13