Add some timing notes

This commit is contained in:
Brooke Vibber 2026-08-22 13:01:23 -07:00
commit 3ce4c1e580

View file

@ -563,12 +563,13 @@ sqr16_patch_offset = 8
; input: arg as u8 ; input: arg as u8
; output: dest as u16 ; output: dest as u16
; clobbers a, x ; clobbers a, x
; 19 cyc
.macro sqr8 dest, arg .macro sqr8 dest, arg
ldx arg ldx arg ; 3 cyc
lda sqr_lobyte,x lda sqr_lobyte,x ; 5 cyc
sta dest sta dest ; 3 cyc
lda sqr_hibyte,x lda sqr_hibyte,x ; 5 cyc
sta dest + 1 sta dest + 1 ; 3 cyc
.endmacro .endmacro
.segment "TABLES" .segment "TABLES"
@ -770,6 +771,10 @@ inner_loop:
.endproc .endproc
; min low-mem: 81 * 4 + 28 * 2 + 10 + 6 = 396 cyc
; max low-mem: 92 * 4 + 28 * 2 + 50 + 6 = 480 cyc
; min ext-mem: 51 * 4 + 28 * 2 + 10 + 6 = 276 cyc
; max low-mem: 70 * 4 + 28 * 2 + 50 + 6 = 392 cyc
.macro imul16_impl xe .macro imul16_impl xe
.local arg1 .local arg1
.local arg2 .local arg2
@ -787,32 +792,37 @@ inner_loop:
; h1*256*(h2*256 + l2) + l1*(h2*256 + l2) ; h1*256*(h2*256 + l2) + l1*(h2*256 + l2)
; h1*h2*256*256 + h1*l2*256 + h2*l1*256 + l1*l2 ; h1*h2*256*256 + h1*l2*256 + h2*l1*256 + l1*l2
imul8 result, arg1, arg2, xe imul8 result, arg1, arg2, xe ; 81-92 or 51-70
imul8 result + 2, arg1 + 1, arg2 + 1, xe imul8 result + 2, arg1 + 1, arg2 + 1, xe ; 81-92 or 51-70
imul8 inter, arg1 + 1, arg2, xe imul8 inter, arg1 + 1, arg2, xe ; 81-92 or 51-70
add16 result + 1, result + 1, inter add16 result + 1, result + 1, inter ; 20 cyc
add_carry result + 3 add_carry result + 3 ; 8 cyc
imul8 inter, arg1, arg2 + 1, xe imul8 inter, arg1, arg2 + 1, xe ; 81-92 or 51-70
add16 result + 1, result + 1, inter add16 result + 1, result + 1, inter ; 20 cyc
add_carry result + 3 add_carry result + 3 ; 8 cyc
; In case of negative inputs, adjust high word ; In case of negative inputs, adjust high word
; https://stackoverflow.com/a/28827013 ; https://stackoverflow.com/a/28827013
lda arg1 + 1 ; 10-50 cycles
bpl arg1_pos lda arg1 + 1 ; 3 cyc
sub16 result + 2, result + 2, arg2 bpl arg1_pos ; 2 cyc
sub16 result + 2, result + 2, arg2 ; 20 cyc
arg1_pos: arg1_pos:
lda arg2 + 1 lda arg2 + 1 ; 3 cyc
bpl arg2_pos bpl arg2_pos ; 2 cyc
sub16 result + 2, result + 2, arg1 sub16 result + 2, result + 2, arg1 ; 20 cyc
arg2_pos: arg2_pos:
rts ; 6 cyc rts ; 6 cyc
.endmacro .endmacro
; min base ram: 19 * 2 + 81 + 28 * 2 + 6 = 181
; max base ram: 19 * 2 + 92 + 28 * 2 + 6 = 192
; min ext ram: 19 * 2 + 51 + 28 * 2 + 6 = 151
; max ext ram: 19 * 2 + 70 + 28 * 2 + 6 = 170
.macro sqr16_impl xe .macro sqr16_impl xe
.scope .scope
arg = FR0 ; 16-bit arg (clobbered) arg = FR0 ; 16-bit arg (clobbered)
@ -820,9 +830,9 @@ arg2_pos:
;inter = temp2 ;inter = temp2
inter = FR1 inter = FR1
lda arg + 1 lda arg + 1 ; 3 cyc
bpl arg_pos bpl arg_pos ; 2 cyc
neg16 arg neg16 arg ; 18 cyc
arg_pos: arg_pos:
; hl * hl ; hl * hl
@ -830,15 +840,15 @@ arg2_pos:
; h*256*(h*256 + l) + l*(h*256 + l) ; h*256*(h*256 + l) + l*(h*256 + l)
; h*h*256*256 + h*l*256 + h*l*256 + l*l ; h*h*256*256 + h*l*256 + h*l*256 + l*l
sqr8 result, arg sqr8 result, arg ; 19 cyc
sqr8 result + 2, arg + 1 sqr8 result + 2, arg + 1 ; 19 cyc
imul8 inter, arg + 1, arg, xe imul8 inter, arg + 1, arg, xe ; 81-92 / 51-70 cyc
add16 result + 1, result + 1, inter add16 result + 1, result + 1, inter ; 20 cyc
add_carry result + 3 add_carry result + 3 ; 8 cyc
add16 result + 1, result + 1, inter add16 result + 1, result + 1, inter ; 20 cyc
add_carry result + 3 add_carry result + 3 ; 8 cyc
rts ; 6 cyc rts ; 6 cyc
.endscope .endscope