diff --git a/mandel-core.s b/mandel-core.s index f021749..a153ff5 100644 --- a/mandel-core.s +++ b/mandel-core.s @@ -18,7 +18,6 @@ z_buffer_start: .res 1 ; u8: index into z_buffer z_buffer_end: .res 1 ; u8: index into z_buffer iter: .res 1 ; u8: iteration count ptr: .res 2 ; u16 -pixel_ptr: .res 2 ; u16 temp: .res 2 ; u16 temp2: .res 2 ; u16 @@ -593,7 +592,7 @@ bank_switch_table: .macro imul8 dest, arg1, arg2, xe .if xe ; using 64KB lookup table - ; 52-69 cycles + ; 51-70 cycles ; clobbers x, y, dest, ptr .scope output = dest @@ -637,7 +636,9 @@ bank_switch_table: txa ; 2 cyc adc output ; 3 cyc sta output ; 3 cyc - add_carry output + 1 ; 2-7 cyc + lda #0 ; 2 cyc + adc output+1 ; 3 cyc + sta output+1 ; 3 cyc done: .endscope @@ -774,8 +775,8 @@ inner_loop: ; min base mem: 81 * 4 + 22 * 2 + 10 + 6 = 384 cyc ; max base mem: 92 * 4 + 27 * 2 + 50 + 6 = 478 cyc -; min ext-mem: 52 * 4 + 22 * 2 + 10 + 6 = 268 cyc -; max ext-mem: 69 * 4 + 27 * 2 + 50 + 6 = 386 cyc +; min ext-mem: 51 * 4 + 22 * 2 + 10 + 6 = 264 cyc +; max ext-mem: 70 * 4 + 27 * 2 + 50 + 6 = 390 cyc .macro imul16_impl xe .local arg1 .local arg2 @@ -793,15 +794,15 @@ inner_loop: ; h1*256*(h2*256 + l2) + l1*(h2*256 + l2) ; h1*h2*256*256 + h1*l2*256 + h2*l1*256 + l1*l2 - imul8 result, arg1, arg2, xe ; 81-92 or 52-69 + imul8 result, arg1, arg2, xe ; 81-92 or 51-70 - imul8 result + 2, arg1 + 1, arg2 + 1, xe ; 81-92 or 52-69 + imul8 result + 2, arg1 + 1, arg2 + 1, xe ; 81-92 or 51-70 - imul8 inter, arg1 + 1, arg2, xe ; 81-92 or 52-69 + imul8 inter, arg1 + 1, arg2, xe ; 81-92 or 51-70 add16 result + 1, result + 1, inter ; 20 cyc add_carry result + 3 ; 2-7 cyc - imul8 inter, arg1, arg2 + 1, xe ; 81-92 or 52-69 + imul8 inter, arg1, arg2 + 1, xe ; 81-92 or 51-70 add16 result + 1, result + 1, inter ; 20 cyc add_carry result + 3 ; 2-7 cyc @@ -822,8 +823,8 @@ arg2_pos: ; min base ram: 5 + 19 * 2 + 81 + 22 * 2 + 6 = 174 ; max base ram: 23 + 19 * 2 + 92 + 27 * 2 + 6 = 213 -; min ext ram: 5 + 19 * 2 + 52 + 22 * 2 + 6 = 145 -; max ext ram: 23 + 19 * 2 + 69 + 27 * 2 + 6 = 190 +; min ext ram: 5 + 19 * 2 + 51 + 22 * 2 + 6 = 144 +; max ext ram: 23 + 19 * 2 + 70 + 27 * 2 + 6 = 191 .macro sqr16_impl xe .scope arg = FR0 ; 16-bit arg (clobbered) @@ -846,7 +847,7 @@ arg2_pos: sqr8 result + 2, arg + 1 ; 19 cyc - imul8 inter, arg + 1, arg, xe ; 81-92 / 52-69 cyc + imul8 inter, arg + 1, arg, xe ; 81-92 / 51-70 cyc add16 result + 1, result + 1, inter ; 20 cyc add_carry result + 3 ; 2-7 cyc add16 result + 1, result + 1, inter ; 20 cyc @@ -1266,21 +1267,21 @@ enough: negative: ; temp1 = top half lda #.lobyte(framebuffer_top + stride * half_height) - sta pixel_ptr + sta ptr lda #.hibyte(framebuffer_top + stride * half_height) - sta pixel_ptr + 1 + sta ptr + 1 jmp point positive: lda #.lobyte(framebuffer_bottom) - sta pixel_ptr + sta ptr lda #.hibyte(framebuffer_bottom) - sta pixel_ptr + 1 + sta ptr + 1 point: - ; pixel_ptr += sy * stride + ; ptr += sy * stride ; temp * 40 ; = temp * 32 + temp * 8 ; = (temp << 5) + (temp << 3) @@ -1288,10 +1289,10 @@ point: shl16 temp shl16 temp shl16 temp - add16 pixel_ptr, pixel_ptr, temp + add16 ptr, ptr, temp shl16 temp shl16 temp - add16 pixel_ptr, pixel_ptr, temp + add16 ptr, ptr, temp ; Ok so temp1 points to the start of the line, which is 40 bytes. ; Get the byte and bit offsets @@ -1331,20 +1332,20 @@ shift_done: draw_pixel: ; read, mask, or, write - lda (pixel_ptr),y + lda (ptr),y and pixel_mask ora pixel_color - sta (pixel_ptr),y + sta (ptr),y dex beq done clc lda #40 - adc pixel_ptr - sta pixel_ptr + adc ptr + sta ptr lda #0 - adc pixel_ptr + 1 - sta pixel_ptr + 1 + adc ptr + 1 + sta ptr + 1 jmp draw_pixel done: