Compare commits

..

View file

@ -558,6 +558,8 @@ input_max:
; input: arg1, arg2 as fixed4.12 ; input: arg1, arg2 as fixed4.12
; output: dest as fixed8.24 ; output: dest as fixed8.24
; patch point jsr at 16 bytes in
imul16_patch_offset = 16
.macro imul16 dest, arg1, arg2 .macro imul16 dest, arg1, arg2
copy16 FR0, arg1 ; 12 cyc copy16 FR0, arg1 ; 12 cyc
copy16 FR1, arg2 ; 12 cyc copy16 FR1, arg2 ; 12 cyc
@ -567,11 +569,13 @@ input_max:
; input: arg as fixed4.12 ; input: arg as fixed4.12
; output: dest as fixed8.24 ; output: dest as fixed8.24
;.macro sqr16 dest, arg ; patch point jsr at 8 bytes in
; copy16 FR0, arg ; 12 cyc sqr16_patch_offset = 8
; jsr sqr16_func ; ? cyc .macro sqr16 dest, arg
; copy32 dest, FR2 ; 24 cyc copy16 FR0, arg ; 12 cyc
;.endmacro jsr sqr16_func ; ? cyc
copy32 dest, FR2 ; 24 cyc
.endmacro
; input: arg as u8 ; input: arg as u8
; output: dest as u16 ; output: dest as u16
@ -795,15 +799,15 @@ inner_loop:
.endproc .endproc
; min base mem: 81 * 4 + 22 * 2 + 10 = 378 cyc ; min base mem: 81 * 4 + 22 * 2 + 10 + 6 = 384 cyc
; max base mem: 92 * 4 + 27 * 2 + 50 = 472 cyc ; max base mem: 92 * 4 + 27 * 2 + 50 + 6 = 478 cyc
; min ext-mem: 50 * 4 + 22 * 2 + 10 = 254 cyc ; min ext-mem: 50 * 4 + 22 * 2 + 10 + 6 = 260 cyc
; max ext-mem: 55 * 4 + 27 * 2 + 50 = 324 cyc ; max ext-mem: 55 * 4 + 27 * 2 + 50 + 6 = 330 cyc
.macro imul16_impl result, arg1, arg2, xe .macro imul16_impl xe
.scope .scope
;arg1 ; 16-bit arg arg1 = FR0 ; 16-bit arg
;arg2 ; 16-bit arg arg2 = FR1 ; 16-bit arg
;result ; 32-bit result (output) result = FR2 ; 32-bit result (output)
inter = temp2 ; 16-bit temporary (clobbered) inter = temp2 ; 16-bit temporary (clobbered)
; h1l1 * h2l2 ; h1l1 * h2l2
@ -839,6 +843,8 @@ inner_loop:
bpl arg2_pos ; 2 cyc bpl arg2_pos ; 2 cyc
sub16 result + 2, result + 2, arg1 ; 20 cyc sub16 result + 2, result + 2, arg1 ; 20 cyc
arg2_pos: arg2_pos:
rts ; 6 cyc
.endscope .endscope
.endmacro .endmacro
@ -871,15 +877,15 @@ inner_loop:
.endmacro .endmacro
; min base ram: 5 + 163 + 3 = 171 ; min base ram: 5 + 163 + 6 = 174
; max base ram: 7 + 184 + 18 = 209 ; max base ram: 7 + 184 + 18 + 6 = 215
; min ext ram: 5 + 132 + 3 = 140 ; min ext ram: 5 + 132 + 6 = 143
; max ext ram: 7 + 147 + 18 = 172 ; max ext ram: 7 + 147 + 18 + 6 = 178
.macro sqr16_impl result, arg, xe .macro sqr16_impl xe
.scope .scope
; arg ; 16-bit arg arg = FR0 ; 16-bit arg
negated = FR1 negated = FR1
; result ; 32-bit result result = FR2 ; 32-bit result
; 5-7 cycles ; 5-7 cycles
lda arg + 1 ; 3 cyc lda arg + 1 ; 3 cyc
@ -888,35 +894,30 @@ inner_loop:
arg_pos: arg_pos:
sqr16_impl_inner result, arg, xe ; 163-184 / 132-147 cyc sqr16_impl_inner result, arg, xe ; 163-184 / 132-147 cyc
jmp done ; 3 cyc rts ; 6 cyc
arg_neg: arg_neg:
copy_neg16 negated, arg ; 18 cyc copy_neg16 negated, arg ; 18 cyc
sqr16_impl_inner result, negated, xe ; 163-184 / 132-147 cyc sqr16_impl_inner result, negated, xe ; 163-184 / 132-147 cyc
rts ; 6 cyc
done:
.endscope .endscope
.endmacro .endmacro
.proc imul16_func .proc imul16_func
imul16_impl FR2, FR0, FR1, 0 imul16_impl 0
rts
.endproc .endproc
.proc imul16xe_func .proc imul16xe_func
imul16_impl FR2, FR0, FR1, 1 imul16_impl 1
rts
.endproc .endproc
;.proc sqr16_func .proc sqr16_func
; sqr16_impl FR2, FR0, 0 sqr16_impl 0
; rts .endproc
;.endproc
;.proc sqr16xe_func .proc sqr16xe_func
; sqr16_impl FR2, FR0, 1 sqr16_impl 1
; rts .endproc
;.endproc
; 11-27 cycles ; 11-27 cycles
.macro round16 arg .macro round16 arg
@ -1156,11 +1157,16 @@ keep_going:
shift_round_16 zy, 3 shift_round_16 zy, 3
; zx_2 = zx * zx ; zx_2 = zx * zx
fixup_sqr16_1:
sqr16 zx_2, zx + 2
; zy_2 = zy * zy ; zy_2 = zy * zy
fixup_sqr16_2:
sqr16 zy_2, zy + 2
; zx_zy = zx * zy ; zx_zy = zx * zy
fixup_mandelbrot_hotspot: fixup_imul16_1:
jmp mandelbrot_hotspot imul16 zx_zy, zx + 2, zy + 2
after_mandelbrot_hotspot:
; dist = zx_2 + zy_2 ; dist = zx_2 + zy_2
add32 dist, zx_2, zy_2 add32 dist, zx_2, zy_2
@ -1256,27 +1262,6 @@ next:
.endproc .endproc
.macro mandelbrot_hotspot_impl xe
; zx_2 = zx * zx
sqr16_impl zx_2, zx + 2, xe
; zy_2 = zy * zy
sqr16_impl zy_2, zy + 2, xe
; zx_zy = zx * zy
imul16_impl zx_zy, zx + 2, zy + 2, xe
jmp mandelbrot::after_mandelbrot_hotspot
.endmacro
.proc mandelbrot_hotspot
mandelbrot_hotspot_impl 0
.endproc
.proc mandelbrot_hotspot_xe
mandelbrot_hotspot_impl 1
.endproc
.macro scale_zoom dest .macro scale_zoom dest
; clobbers X, flags ; clobbers X, flags
.local cont .local cont
@ -2204,21 +2189,22 @@ init:
sta imul16_func sta imul16_func
lda #.lobyte(imul16xe_func) lda #.lobyte(imul16xe_func)
sta imul16_func + 1 sta imul16_func + 1
sta mandelbrot::fixup_imul16_1 + imul16_patch_offset + 1
lda #.hibyte(imul16xe_func) lda #.hibyte(imul16xe_func)
sta imul16_func + 2 sta imul16_func + 2
sta mandelbrot::fixup_imul16_1 + imul16_patch_offset + 2
; ditto for sqr16_func -> sqr16xe_func ; ditto for sqr16_func -> sqr16xe_func
;lda #$4c ; 'jmp' opcode lda #$4c ; 'jmp' opcode
;sta sqr16_func sta sqr16_func
;lda #.lobyte(sqr16xe_func) lda #.lobyte(sqr16xe_func)
;sta sqr16_func + 1 sta sqr16_func + 1
;lda #.hibyte(sqr16xe_func) sta mandelbrot::fixup_sqr16_1 + sqr16_patch_offset + 1
;sta sqr16_func + 2 sta mandelbrot::fixup_sqr16_2 + sqr16_patch_offset + 1
lda #.hibyte(sqr16xe_func)
lda #.lobyte(mandelbrot_hotspot_xe) sta sqr16_func + 2
sta mandelbrot::fixup_mandelbrot_hotspot + 1 sta mandelbrot::fixup_sqr16_1 + sqr16_patch_offset + 2
lda #.hibyte(mandelbrot_hotspot_xe) sta mandelbrot::fixup_sqr16_2 + sqr16_patch_offset + 2
sta mandelbrot::fixup_mandelbrot_hotspot + 2
; create the lookup table ; create the lookup table