joeylib2/src/port/iigs/joeyDraw.s

4500 lines
134 KiB
ArmAsm

; joeyDraw.s - Apple IIgs ASM inner loops for graphics primitives.
;
; GAS syntax for the llvm-mos w65816 toolchain. Each function is the
; FAST INNER LOOP only;
; the C wrappers in src/core/draw.c and src/core/tile.c handle NULL /
; range validation, clipping, and dirty-rect marking. The inner loops
; trust their inputs (pre-clipped, byte-aligned, non-NULL).
;
; Calling convention (llvm-mos w65816 cdecl), as probed from clang -S:
; - arg0 in registers: A:X for a 32-bit pointer/long, A for a 16-bit
; word. (clang puts arg0 in registers, which is why the
; per-function stack offsets below are small.)
; - args 1.. on the stack, pushed right-to-left, first stack arg at
; 4,s (after the 3-byte JSL return), each word +2, each ptr/long +4.
; - After PHP every stack offset shifts by +1.
; - uint16_t return in A; 32-bit return in A:X.
;
; Memory model (llvm-mos small model): DBR = 0 for C code, and globals
; live in bank 0, so 16-bit absolute reaches both globals and bank-0
; soft switches ($00Cxxx) -- no long prefix is needed on those.
; The SHR stage at $01:2000 is written with full 24-bit literals, which
; the assembler selects long addressing for automatically (value >64K).
;
; llvm-mc defaults to M=I=16; the explicit .a8/.a16/.i8/.i16 directives
; below pair with sep/rep so immediates are sized correctly.
.text
; ----------------------------------------------------------------
; void iigsInitRowLut(void)
;
; One-shot init for the y*160 lookup table at gRowOffsetLut. C calls
; this once from halInit; afterwards every primitive that needs y ->
; row byte offset can do `lda gRowOffsetLut,x` (X = y*2) in a few cyc
; instead of a shift-add chain. 200 entries x 2 bytes = 400 bytes.
; ----------------------------------------------------------------
.section .text.iigsInitRowLut,"ax"
.globl iigsInitRowLut
iigsInitRowLut:
php
rep #0x30 ; M=16, X=16
.a16
.i16
ldx #0 ; X = byte offset into LUT
lda #0 ; A = current y*160
initLutLoop:
sta gRowOffsetLut, x
clc
adc #160
inx
inx
cpx #400
bcc initLutLoop
.a8
.i8
plp
rtl
; ----------------------------------------------------------------
; void iigsSurfaceClearInner(uint8_t *pixels, uint16_t fillWord)
;
; Fast constant fill of the 32000-byte stage at $01:$2000-$9CFF with
; fillWord (both bytes the same). Unrolled run of STA long-absolute
; indexed stores (`sta >$0120xx,x`, 6 cyc / 2 bytes). 40 stores cover
; 80 bytes; X steps by 80 over 400 rows. Long addressing targets bank
; $01 explicitly, so no DBR / AUXWRITE / shadow manipulation and no SEI.
;
; `pixels` (arg0) arrives in A:X and is ignored; the stage is hardcoded
; at $01:$2000 (the C wrapper guarantees s == jlStageGet()). fillWord
; (arg1) is at 5,s after PHP.
; ----------------------------------------------------------------
.section .text.iigsSurfaceClearInner,"ax"
.globl iigsSurfaceClearInner
iigsSurfaceClearInner:
php
rep #0x30 ; M=16, X=16
.a16
.i16
ldx #0 ; X = row byte base (0,80,...,31920)
clrRowLoop:
lda 5, s ; A = fillWord (S never moves; 5,s stays valid)
sta 0x012000, x
sta 0x012002, x
sta 0x012004, x
sta 0x012006, x
sta 0x012008, x
sta 0x01200a, x
sta 0x01200c, x
sta 0x01200e, x
sta 0x012010, x
sta 0x012012, x
sta 0x012014, x
sta 0x012016, x
sta 0x012018, x
sta 0x01201a, x
sta 0x01201c, x
sta 0x01201e, x
sta 0x012020, x
sta 0x012022, x
sta 0x012024, x
sta 0x012026, x
sta 0x012028, x
sta 0x01202a, x
sta 0x01202c, x
sta 0x01202e, x
sta 0x012030, x
sta 0x012032, x
sta 0x012034, x
sta 0x012036, x
sta 0x012038, x
sta 0x01203a, x
sta 0x01203c, x
sta 0x01203e, x
sta 0x012040, x
sta 0x012042, x
sta 0x012044, x
sta 0x012046, x
sta 0x012048, x
sta 0x01204a, x
sta 0x01204c, x
sta 0x01204e, x
txa
clc
adc #80
tax
cpx #32000
bcs clrExit ; X >= 32000 -> done
brl clrRowLoop ; loop back is >127 bytes -> needs brl
clrExit:
.a8
.i8
plp
rtl
; ----------------------------------------------------------------
; void iigsSurfaceClearFastInner(uint8_t *pixels, uint16_t fillWord)
;
; Post-init fast clear: PHA stack-slam (4 cyc / 2 bytes) vs the STA
; long,X path (6 cyc / 2 bytes). SP is relocated to the top of the $01
; stage region ($2000..$9CFF); AUXWRITE ($C005) redirects the bank-$00
; stack writes to the $01 stage bank, and SHR shadow is already
; inhibited by halInit. ~1.5x faster.
;
; SEI guards the AUXWRITE/SP-hijack window; SP is restored before CLI.
; SAFE ONLY POST-halInit. Soft switches use LONG addressing to bank $E0
; ($E0C005), NOT `sta $C005`: under the GS/OS Loader DBR=$08, a DBR-rel
; `sta $C0xx` hits bank-$08 RAM and the switch silently never toggles, so
; AUXWRITE stays off and the slam corrupts bank 0 instead of the $01
; stage. `pixels` (arg0) in A:X is ignored; fillWord (arg1) at 5,s.
; ----------------------------------------------------------------
.section .text.iigsSurfaceClearFastInner,"ax"
.globl iigsSurfaceClearFastInner
iigsSurfaceClearFastInner:
php
rep #0x30
.a16
.i16
lda 5, s ; A = fillWord (read before SP/AUXWRITE change)
sta sclrFillCap
sei
tsc
sta sclrSavedSp ; save original SP
sep #0x20
.a8
sta 0xE0C005 ; AUXWRITE on: $00:0200-BFFF writes -> bank $01
rep #0x20
.a16
lda #0x9cff
tcs ; SP = top of $01 stage region
lda sclrFillCap ; A = fillWord = the pushed value
ldx #200 ; 200 blocks x 80 PHA = 16000 pushes = 32000 bytes
clrFastBlock:
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
pha
dex
bne clrFastBlock
lda sclrSavedSp
tcs ; restore original SP
sep #0x20
.a8
sta 0xE0C004 ; AUXWRITE off
rep #0x20
.a16
.a8
.i8
plp ; restores I (CLI)
rtl
; ----- DRAWPRIMS data (uninitialized; bank 0 BSS) -----
.section .bss.gRowOffsetLut,"aw"
.globl gRowOffsetLut
gRowOffsetLut:
.zero 400
.section .bss.sclrFillCap,"aw"
.globl sclrFillCap
sclrFillCap:
.zero 2
.section .bss.sclrSavedSp,"aw"
.globl sclrSavedSp
sclrSavedSp:
.zero 2
; Flood-fill seed/match scratch (asm-only data; no C definition exists).
; iigsFloodWalkAndScansInner writes the seed match result and the left/
; right span extents here for the C wrapper to read back.
.section .bss.gFloodSeedMatch,"aw"
.globl gFloodSeedMatch
gFloodSeedMatch:
.zero 2
.section .bss.gFloodLeftX,"aw"
.globl gFloodLeftX
gFloodLeftX:
.zero 2
.section .bss.gFloodRightX,"aw"
.globl gFloodRightX
gFloodRightX:
.zero 2
; ===== ported: tile primitives =====
; ====================================================================
; frag_tiles.s - tile inner loops (llvm-mc w65816)
;
; Shared ABI fact (clang w65816 cdecl, per PORT_SPEC.md):
; - arg0 (pointer/long) arrives in A=low16, X=high16 (bank byte =
; low byte of X). It is NOT on the stack.
; - arg1, arg2, ... are pushed right-to-left and live on the stack;
; on JSL entry arg1 byte0 is at 4,s. A word arg is 2 bytes, a
; pointer/long arg is 4 bytes.
; - Each prologue push shifts every stack offset up: php=+1, phb=+1,
; phd=+2, pha/phx=+2.
; ====================================================================
; --------------------------------------------------------------------
; STAGING pointer block for the two-pointer tile routines (Copy / Paste
; / Snap / CopyMasked). The clang ABI hands us dst in A:X and src on the
; stack, but the copy loops assume a D-frame with dst at [D+0] and src
; at [D+4]. We assemble the 8 bytes here first, then COPY them onto the
; hardware stack and point D there -- D (the direct page) is ALWAYS bank
; 0, but this BSS block lives in the LOAD bank, so D must NOT point at it
; directly (that was the historical bug: [0],y/[4],y then read bank 0
; while the stores wrote the load bank). The stack is bank 0, so the
; copied frame matches.
; tileScratchPtrs+0..2 = dst (low16 + bank) (staged, then -> [0],y)
; tileScratchPtrs+4..6 = src (low16 + bank) (staged, then -> [4],y)
; (bytes +3 / +7 are unused padding, mirroring the 4-byte ptr layout)
; --------------------------------------------------------------------
.section .bss.tileScratchPtrs,"aw"
.globl tileScratchPtrs
tileScratchPtrs:
.zero 8
; ====================================================================
; iigsTileFillInner(uint8_t *dstRow0, uint16_t fillWord)
;
; Original: 16 STA abs,X stores at fixed offsets along a 160-byte
; stride (DBR set to the dst bank via PHB/PLB so abs,X rode the right
; bank). ~120 cyc per call vs the C version's ~300.
;
; ABI: arg0 dstRow0 in A:X (A = dst low16, X = dst high16 incl. bank).
; arg1 fillWord (word) on the stack: @4,s on JSL entry; after
; PHP+PHB+PHD(+4) it sits @8,s. Only fillWord is on the stack and
; dst comes in A:X.
;
; PORTING NOTE (addressing mode): the original used `sta |off,x`
; (16-bit absolute,X, DBR = dst bank). llvm-mc here unconditionally
; shrinks `sta off,x` with a small constant `off` to DIRECT-PAGE,X
; (opcode 0x95), which is bank-0 / D-relative and would target the
; wrong memory; and this build accepts none of the abs-forcing
; modifiers (`!`/`|`/`mos16()`). To stay bank-correct AND assemble, we
; keep the identical 16 store offsets and the identical fill word but
; reach the destination via DP-indirect-long `[dst],y` with Y = the
; offset -- the same mechanism the two-pointer tile routines use. We
; stash dst (A:X) in tileScratchPtrs and point D there ([0] = dst).
; D-frame: tileScratchPtrs+0..2 = dst (low16 + bank) -> [0],y
; ====================================================================
.section .text.iigsTileFillInner,"ax"
.globl iigsTileFillInner
iigsTileFillInner:
php
phb
phd ; save caller D
rep #0x30
.a16
.i16
; Build the dst D-frame on the STACK (bank 0). A BSS scratch
; (tileScratchPtrs) lives in the LOAD bank, but D (the direct
; page) is ALWAYS bank 0, so the old `tcd #tileScratchPtrs`
; pointed [0],y at bank-0:offset while the stores wrote
; loadbank:offset -- the indirect read the wrong bank entirely.
; The hardware stack is bank 0, so a stack D-frame matches.
; A = dst low16, X = dst high16 (bank in low byte).
phx ; dst high16 -> 3,s
pha ; dst low16 -> 1,s
tsc
inc a ; A = SP+1
tcd ; D -> stacked dst; [0] = dst
lda 12,s ; fillWord (8,s + 4 just pushed)
ldy #0
sta [0], y
ldy #2
sta [0], y
ldy #160
sta [0], y
ldy #162
sta [0], y
ldy #320
sta [0], y
ldy #322
sta [0], y
ldy #480
sta [0], y
ldy #482
sta [0], y
ldy #640
sta [0], y
ldy #642
sta [0], y
ldy #800
sta [0], y
ldy #802
sta [0], y
ldy #960
sta [0], y
ldy #962
sta [0], y
ldy #1120
sta [0], y
ldy #1122
sta [0], y
.a8
.i8
rep #0x30
.a16
.i16
tsc
clc
adc #4 ; drop the dst D-frame (phx+pha)
tcs
pld ; restore caller D
plb
plp
rtl
; ====================================================================
; iigsTileCopyInner(uint8_t *dst, const uint8_t *src)
;
; Opaque copy of an 8x8 tile region (32 bytes) between two 4bpp
; surfaces with 160-byte row stride. Dst and src may be in any banks.
; ~340 cyc per tile vs the C version's ~900.
;
; ABI: arg0 dst in A:X (A=low16, X=high16). arg1 src (4-byte ptr) on
; the stack: @4,s on JSL entry; after PHP+PHB+PHD(+4) src low16
; @8,s, src bank @10,s. We point D at tileScratchPtrs, store dst
; (A:X) at +0..2 and src (from the stack) at +4..6, then run the
; original [0],y (dst) / [4],y (src) loads unchanged. PLD restores
; the caller's D (saved by our PHD).
; ====================================================================
.section .text.iigsTileCopyInner,"ax"
.globl iigsTileCopyInner
iigsTileCopyInner:
php
phb
phd ; save caller D
rep #0x30
.a16
.i16
; Build the dst/src D-frame in bank-0 scratch.
sta tileScratchPtrs+0 ; dst low16
stx tileScratchPtrs+2 ; dst high16 (bank + pad)
lda 8,s ; src low16 (post +4 push)
sta tileScratchPtrs+4
lda 10,s ; src bank
sta tileScratchPtrs+6
; Copy the staged dst/src into a STACK (bank-0) D-frame: the
; tileScratchPtrs BSS scratch is in the LOAD bank, but D (the
; direct page) is ALWAYS bank 0, so [0],y/[4],y via
; `tcd #tileScratchPtrs` indirected through the wrong bank.
lda tileScratchPtrs+6
pha ; src bank -> 7,s
lda tileScratchPtrs+4
pha ; src low16 -> 5,s
lda tileScratchPtrs+2
pha ; dst high16 -> 3,s
lda tileScratchPtrs+0
pha ; dst low16 -> 1,s
tsc
inc a ; A = SP+1
tcd ; D -> stacked {dst,src}; [0]=dst [4]=src
ldy #0
lda [4], y
sta [0], y
ldy #2
lda [4], y
sta [0], y
ldy #160
lda [4], y
sta [0], y
ldy #162
lda [4], y
sta [0], y
ldy #320
lda [4], y
sta [0], y
ldy #322
lda [4], y
sta [0], y
ldy #480
lda [4], y
sta [0], y
ldy #482
lda [4], y
sta [0], y
ldy #640
lda [4], y
sta [0], y
ldy #642
lda [4], y
sta [0], y
ldy #800
lda [4], y
sta [0], y
ldy #802
lda [4], y
sta [0], y
ldy #960
lda [4], y
sta [0], y
ldy #962
lda [4], y
sta [0], y
ldy #1120
lda [4], y
sta [0], y
ldy #1122
lda [4], y
sta [0], y
.a8
.i8
rep #0x30
.a16
.i16
tsc
clc
adc #8 ; drop the {dst,src} D-frame
tcs
pld ; restore caller D
plb
plp
rtl
; ====================================================================
; iigsTilePasteInner(uint8_t *dst, const uint8_t *src)
;
; Paste a packed jlTileT buffer (4 bytes/row * 8 rows = 32 bytes tight)
; onto a surface row at `dst`. Src stride 4, dst stride 160. Both args
; are 4-byte large-model pointers.
;
; ABI: identical to iigsTileCopyInner -- arg0 dst in A:X, arg1 src on
; the stack (low16 @8,s / bank @10,s after PHP+PHB+PHD). D points
; at tileScratchPtrs; [0],y = dst, [4],y = src.
; ====================================================================
.section .text.iigsTilePasteInner,"ax"
.globl iigsTilePasteInner
iigsTilePasteInner:
php
phb
phd ; save caller D
rep #0x30
.a16
.i16
sta tileScratchPtrs+0 ; dst low16
stx tileScratchPtrs+2 ; dst high16 (bank + pad)
lda 8,s ; src low16
sta tileScratchPtrs+4
lda 10,s ; src bank
sta tileScratchPtrs+6
; Copy the staged dst/src into a STACK (bank-0) D-frame: the
; tileScratchPtrs BSS scratch is in the LOAD bank, but D (the
; direct page) is ALWAYS bank 0, so [0],y/[4],y via
; `tcd #tileScratchPtrs` indirected through the wrong bank.
lda tileScratchPtrs+6
pha ; src bank -> 7,s
lda tileScratchPtrs+4
pha ; src low16 -> 5,s
lda tileScratchPtrs+2
pha ; dst high16 -> 3,s
lda tileScratchPtrs+0
pha ; dst low16 -> 1,s
tsc
inc a ; A = SP+1
tcd ; D -> stacked {dst,src}; [0]=dst [4]=src
; Row 0: src word 0 = 0, dst word 0 = 0
ldy #0
lda [4], y
sta [0], y
ldy #2
lda [4], y
sta [0], y
; Row 1: src 4/6, dst 160/162
ldy #4
lda [4], y
ldy #160
sta [0], y
ldy #6
lda [4], y
ldy #162
sta [0], y
; Row 2: src 8/10, dst 320/322
ldy #8
lda [4], y
ldy #320
sta [0], y
ldy #10
lda [4], y
ldy #322
sta [0], y
; Row 3: src 12/14, dst 480/482
ldy #12
lda [4], y
ldy #480
sta [0], y
ldy #14
lda [4], y
ldy #482
sta [0], y
; Row 4: src 16/18, dst 640/642
ldy #16
lda [4], y
ldy #640
sta [0], y
ldy #18
lda [4], y
ldy #642
sta [0], y
; Row 5: src 20/22, dst 800/802
ldy #20
lda [4], y
ldy #800
sta [0], y
ldy #22
lda [4], y
ldy #802
sta [0], y
; Row 6: src 24/26, dst 960/962
ldy #24
lda [4], y
ldy #960
sta [0], y
ldy #26
lda [4], y
ldy #962
sta [0], y
; Row 7: src 28/30, dst 1120/1122
ldy #28
lda [4], y
ldy #1120
sta [0], y
ldy #30
lda [4], y
ldy #1122
sta [0], y
.a8
.i8
rep #0x30
.a16
.i16
tsc
clc
adc #8 ; drop the {dst,src} D-frame
tcs
pld ; restore caller D
plb
plp
rtl
; ====================================================================
; iigsTileSnapInner(uint8_t *dstTilePixels, const uint8_t *srcRow0)
;
; Snapshot an 8x8 region of a surface into a packed jlTileT buffer.
; Src stride 160, dst stride 4. Mirrors jlTilePaste with stride values
; swapped.
;
; ABI: arg0 dst (the packed tile buffer) in A:X, arg1 src (the surface
; row) on the stack (low16 @8,s / bank @10,s after PHP+PHB+PHD).
; D points at tileScratchPtrs; [0],y = dst, [4],y = src.
; ====================================================================
.section .text.iigsTileSnapInner,"ax"
.globl iigsTileSnapInner
iigsTileSnapInner:
php
phb
phd ; save caller D
rep #0x30
.a16
.i16
sta tileScratchPtrs+0 ; dst low16
stx tileScratchPtrs+2 ; dst high16 (bank + pad)
lda 8,s ; src low16
sta tileScratchPtrs+4
lda 10,s ; src bank
sta tileScratchPtrs+6
; Copy the staged dst/src into a STACK (bank-0) D-frame: the
; tileScratchPtrs BSS scratch is in the LOAD bank, but D (the
; direct page) is ALWAYS bank 0, so [0],y/[4],y via
; `tcd #tileScratchPtrs` indirected through the wrong bank.
lda tileScratchPtrs+6
pha ; src bank -> 7,s
lda tileScratchPtrs+4
pha ; src low16 -> 5,s
lda tileScratchPtrs+2
pha ; dst high16 -> 3,s
lda tileScratchPtrs+0
pha ; dst low16 -> 1,s
tsc
inc a ; A = SP+1
tcd ; D -> stacked {dst,src}; [0]=dst [4]=src
; Row 0: src 0/2, dst 0/2
ldy #0
lda [4], y
sta [0], y
ldy #2
lda [4], y
sta [0], y
; Row 1: src 160/162, dst 4/6
ldy #160
lda [4], y
ldy #4
sta [0], y
ldy #162
lda [4], y
ldy #6
sta [0], y
; Row 2: src 320/322, dst 8/10
ldy #320
lda [4], y
ldy #8
sta [0], y
ldy #322
lda [4], y
ldy #10
sta [0], y
; Row 3: src 480/482, dst 12/14
ldy #480
lda [4], y
ldy #12
sta [0], y
ldy #482
lda [4], y
ldy #14
sta [0], y
; Row 4: src 640/642, dst 16/18
ldy #640
lda [4], y
ldy #16
sta [0], y
ldy #642
lda [4], y
ldy #18
sta [0], y
; Row 5: src 800/802, dst 20/22
ldy #800
lda [4], y
ldy #20
sta [0], y
ldy #802
lda [4], y
ldy #22
sta [0], y
; Row 6: src 960/962, dst 24/26
ldy #960
lda [4], y
ldy #24
sta [0], y
ldy #962
lda [4], y
ldy #26
sta [0], y
; Row 7: src 1120/1122, dst 28/30
ldy #1120
lda [4], y
ldy #28
sta [0], y
ldy #1122
lda [4], y
ldy #30
sta [0], y
.a8
.i8
rep #0x30
.a16
.i16
tsc
clc
adc #8 ; drop the {dst,src} D-frame
tcs
pld ; restore caller D
plb
plp
rtl
; ====================================================================
; iigsTileCopyMaskedInner(uint8_t *dst, const uint8_t *src,
; uint16_t transparent)
;
; Same shape as jlTileCopy (160/160 strides, 32 bytes), but source
; nibbles equal to `transparent` are skipped. Per-byte fast path: if
; both src nibbles == transparent the whole byte is skipped; otherwise
; build the output by mixing src/dst nibbles.
;
; ABI: arg0 dst in A:X. arg1 src (4-byte ptr) and arg2 transparent
; (word) on the stack. JSL entry: src @4,s, transparent @8,s.
; After PHP+PHB+PHD(+4): src low16 @8,s, src bank @10,s,
; transparent @12,s. The original read `trans` via the D-frame
; ([D+8]); here D points at tileScratchPtrs instead, so we read
; transparent's low byte from the stack (@12,s in M=8) during
; setup. [0],y = dst, [4],y = src in the per-byte subroutine.
; ====================================================================
.section .text.iigsTileCopyMaskedInner,"ax"
.globl iigsTileCopyMaskedInner
iigsTileCopyMaskedInner:
php
phb
phd ; save caller D
rep #0x30
.a16
.i16
; Build the dst/src D-frame in bank-0 scratch.
sta tileScratchPtrs+0 ; dst low16
stx tileScratchPtrs+2 ; dst high16 (bank + pad)
lda 8,s ; src low16
sta tileScratchPtrs+4
lda 10,s ; src bank
sta tileScratchPtrs+6
; Copy the staged dst/src into a STACK (bank-0) D-frame: the
; tileScratchPtrs BSS scratch is in the LOAD bank, but D (the
; direct page) is ALWAYS bank 0, so [0],y/[4],y via
; `tcd #tileScratchPtrs` indirected through the wrong bank.
lda tileScratchPtrs+6
pha ; src bank -> 7,s
lda tileScratchPtrs+4
pha ; src low16 -> 5,s
lda tileScratchPtrs+2
pha ; dst high16 -> 3,s
lda tileScratchPtrs+0
pha ; dst low16 -> 1,s
tsc
inc a ; A = SP+1
tcd ; D -> stacked {dst,src}; [0]=dst [4]=src
; Pre-compute scratch values from `transparent`.
; transparent sits @12,s after the prologue; the 4-word stack
; D-frame pushed just above adds 8 more, so it is now @20,s.
sep #0x20 ; M=8
.a8
lda 20,s ; transparent (low byte)
and #0x0F
sta tmaskTLo ; tLo = T
asl a
asl a
asl a
asl a
sta tmaskTHi ; tHi = T<<4
ora tmaskTLo
sta tmaskDoubled ; doubled = (T<<4)|T
; Walk 8 rows of 4 bytes. Y holds the byte offset; after
; the 3 in-row INYs Y is at row*160+3, so +157 lands at
; next row's start (row*160+160).
ldx #8 ; row counter
ldy #0 ; byte offset
tmaskRowLoop:
jsr tmaskByte ; byte 0
iny
jsr tmaskByte ; byte 1
iny
jsr tmaskByte ; byte 2
iny
jsr tmaskByte ; byte 3
rep #0x20 ; M=16 for arithmetic on Y
.a16
tya
clc
adc #157 ; +156 stride, +1 (we're at row+3)
tay
sep #0x20 ; back to M=8
.a8
dex
bne tmaskRowLoop
rep #0x20 ; M=16 before epilogue
.a16
.i16
tsc
clc
adc #8 ; drop the {dst,src} D-frame
tcs
pld ; restore caller D
plb
plp
rtl
; tmaskByte: in M=8 X=16, Y holds byte offset.
; - load src byte
; - if both nibbles transparent: skip
; - else: assemble output from src/dst nibbles per nibble's
; non-transparency, store
tmaskByte:
; Entered (via JSR from the M=8 row loop) with M=8, X=16.
; Tell the assembler so the immediates below size as 1
; byte. RTS returns to the still-M=8 row loop, so no
; width restore is needed here.
.a8
.i16
lda [4], y ; A = src byte
cmp tmaskDoubled
beq tmaskSkip ; both nibbles == transparent
sta tmaskSrc
; Decide hi nibble.
and #0xF0
cmp tmaskTHi
beq tmaskUseDstHi
sta tmaskOutHi ; src hi (kept in $F0 form)
bra tmaskHiDone
tmaskUseDstHi:
lda [0], y
and #0xF0
sta tmaskOutHi
tmaskHiDone:
; Decide lo nibble.
lda tmaskSrc
and #0x0F
cmp tmaskTLo
beq tmaskUseDstLo
sta tmaskOutLo ; src lo
bra tmaskLoDone
tmaskUseDstLo:
lda [0], y
and #0x0F
sta tmaskOutLo
tmaskLoDone:
lda tmaskOutHi
ora tmaskOutLo
sta [0], y
tmaskSkip:
rts
; --------------------------------------------------------------------
; Scratch for the masked-tile inner loop. Lives outside the
; primitive's stack frame because the per-byte subroutine call would
; otherwise need to thread state through registers. (bank-0 globals)
; --------------------------------------------------------------------
.section .bss.tmaskTLo,"aw"
.globl tmaskTLo
tmaskTLo:
.zero 1
.section .bss.tmaskTHi,"aw"
.globl tmaskTHi
tmaskTHi:
.zero 1
.section .bss.tmaskDoubled,"aw"
.globl tmaskDoubled
tmaskDoubled:
.zero 1
.section .bss.tmaskSrc,"aw"
.globl tmaskSrc
tmaskSrc:
.zero 1
.section .bss.tmaskOutHi,"aw"
.globl tmaskOutHi
tmaskOutHi:
.zero 1
.section .bss.tmaskOutLo,"aw"
.globl tmaskOutLo
tmaskOutLo:
.zero 1
; ===== ported: fillrect / pixel / line =====
; ====================================================================
; JoeyLib IIgs draw primitives (llvm-mc GAS, w65816)
; Functions: iigsFillRectInner, iigsDrawPixelInner, iigsDrawLineInner
; ====================================================================
; ====================================================================
; iigsFillRectInner(uint8_t *pixels, uint16_t x, uint16_t y,
; uint16_t w, uint16_t h, uint16_t nibble)
;
; Fill an axis-aligned rectangle of 4bpp pixels into the IIgs SHR
; stage at bank $01. FillRect hard-codes bank 1 and derives the row
; address from y via gRowOffsetLut, so the `pixels` arg (A:X) is unused.
;
; ABI: clang cdecl. pixels arrives in A:X (UNUSED). x,y,w,h,nibble are
; on the stack (right-to-left). On JSL entry: x@4,s y@6,s w@8,s h@10,s
; nib@12,s. After php+phb+phd (+4) and tsc+adc#8+tcd (D = SP+8), the
; equates are:
; fri_x=0 fri_y=2 fri_w=4 fri_h=6 fri_nib=8
; Prologue (php/phb/phd, tsc+adc#8+tcd) otherwise unchanged.
; ====================================================================
.section .text.iigsFillRectInner,"ax"
.globl iigsFillRectInner
iigsFillRectInner:
fri_x = 0
fri_y = 2
fri_w = 4
fri_h = 6
fri_nib = 8
php
phb
phd
rep #0x30
.a16
.i16
tsc
clc
adc #8
tcd
; pxStart = x; pxEnd = x + w.
lda fri_x
sta friPxStart
clc
adc fri_w
sta friPxEnd
; leadingByte = x >> 1; hasLeading = x & 1.
lda fri_x
lsr a
sta friLeadingByte
lda fri_x
and #1
sta friHasLeading
beq friStartReady
lda friPxStart
inc a
sta friPxStart
friStartReady:
; midStart = pxStart >> 1.
lda friPxStart
lsr a
sta friMidStart
; spanLen = pxEnd - pxStart; hasTrailing = spanLen & 1; midBytes = spanLen >> 1.
lda friPxEnd
sec
sbc friPxStart
sta friMidBytes ; temporary hold spanLen
and #1
sta friHasTrailing
lda friMidBytes
lsr a
sta friMidBytes
; trailingByte = midStart + midBytes.
lda friMidStart
clc
adc friMidBytes
sta friTrailingByte
; nibble decoding: friNibLo = nib (low half), friNibHi = nib<<4
; (high half), friDoubled = (nib<<4)|nib (full byte fill).
sep #0x20
.a8
lda fri_nib
and #0x0F
sta friNibLo
asl a
asl a
asl a
asl a
sta friNibHi
ora friNibLo
sta friDoubled
rep #0x20
.a16
; curRow = $2000 + y*160 (LUT-driven).
lda fri_y
asl a
tax
lda gRowOffsetLut,x
clc
adc #0x2000
sta friCurRow
lda fri_h
sta friRowsLeft
; friFillWord = friDoubled replicated into both bytes (16-bit pattern).
lda friDoubled
and #0x00FF
sta friFillWord
xba ; A = byte in high half
ora friFillWord
sta friFillWord
friRowLoop:
lda friRowsLeft
bne friDoRow
brl friExit
friDoRow:
; Leading partial-byte RMW.
lda friHasLeading
beq friNoLead
lda friCurRow
clc
adc friLeadingByte
tax
sep #0x20
.a8
lda 0x010000,x
and #0xF0
ora friNibLo
sta 0x010000,x
rep #0x20
.a16
friNoLead:
; Middle bytes: if midBytes == 0, skip; if midBytes == 1, just seed
; one byte; if midBytes >= 2, seed first byte then MVN propagates the
; seed across the rest of the row.
lda friMidBytes
bne friMidNonZero
brl friNoMid
friMidNonZero:
; Computed-jump slam for ANY middle width. nWords = midBytes>>1 words are
; slammed through the unrolled STA long,X table by entering it at store
; #(80-nWords), with X biased so those stores cover
; [curRow+midStart .. +2*nWords-1] at 3 cyc/byte. Replaces the old
; friFullWidth-only slam plus seed+MVN-for-partials (7 cyc/byte). The entry
; is reached with an RTS computed-goto: it stays in the program bank and
; uses the normal stack, so there is no bank-0 indirect-vector hazard.
; nWords==0 (a 1-byte middle) lands just past the table on the odd-byte
; writer. An odd trailing middle byte is written after the table.
lda friMidBytes
lsr a
sta friSlamWords ; nWords = midBytes >> 1
; Push (entry - 1); entry = friMidSlamTbl + (80 - nWords) * 4 (each store is 4 bytes).
lda #80
sec
sbc friSlamWords
asl a
asl a
clc
adc #friMidSlamTbl-1
pha
; X = curRow + midStart + 2*nWords - 160, so store #(80-nWords) hits curRow+midStart.
lda friSlamWords
asl a
clc
adc friCurRow
clc
adc friMidStart
sec
sbc #160
tax
lda friFillWord
rts ; computed-goto into friMidSlamTbl
friMidSlamTbl:
sta 0x010000,x
sta 0x010002,x
sta 0x010004,x
sta 0x010006,x
sta 0x010008,x
sta 0x01000A,x
sta 0x01000C,x
sta 0x01000E,x
sta 0x010010,x
sta 0x010012,x
sta 0x010014,x
sta 0x010016,x
sta 0x010018,x
sta 0x01001A,x
sta 0x01001C,x
sta 0x01001E,x
sta 0x010020,x
sta 0x010022,x
sta 0x010024,x
sta 0x010026,x
sta 0x010028,x
sta 0x01002A,x
sta 0x01002C,x
sta 0x01002E,x
sta 0x010030,x
sta 0x010032,x
sta 0x010034,x
sta 0x010036,x
sta 0x010038,x
sta 0x01003A,x
sta 0x01003C,x
sta 0x01003E,x
sta 0x010040,x
sta 0x010042,x
sta 0x010044,x
sta 0x010046,x
sta 0x010048,x
sta 0x01004A,x
sta 0x01004C,x
sta 0x01004E,x
sta 0x010050,x
sta 0x010052,x
sta 0x010054,x
sta 0x010056,x
sta 0x010058,x
sta 0x01005A,x
sta 0x01005C,x
sta 0x01005E,x
sta 0x010060,x
sta 0x010062,x
sta 0x010064,x
sta 0x010066,x
sta 0x010068,x
sta 0x01006A,x
sta 0x01006C,x
sta 0x01006E,x
sta 0x010070,x
sta 0x010072,x
sta 0x010074,x
sta 0x010076,x
sta 0x010078,x
sta 0x01007A,x
sta 0x01007C,x
sta 0x01007E,x
sta 0x010080,x
sta 0x010082,x
sta 0x010084,x
sta 0x010086,x
sta 0x010088,x
sta 0x01008A,x
sta 0x01008C,x
sta 0x01008E,x
sta 0x010090,x
sta 0x010092,x
sta 0x010094,x
sta 0x010096,x
sta 0x010098,x
sta 0x01009A,x
sta 0x01009C,x
sta 0x01009E,x
; Odd trailing middle byte: the slam wrote 2*nWords bytes; if midBytes is
; odd, write the last middle byte at curRow + midStart + 2*nWords.
lda friMidBytes
and #1
beq friMidNoOdd
lda friSlamWords
asl a
clc
adc friCurRow
clc
adc friMidStart
tax
sep #0x20
.a8
lda friDoubled
sta 0x010000,x
rep #0x20
.a16
friMidNoOdd:
friNoMid:
; Trailing partial-byte RMW.
lda friHasTrailing
beq friNoTrail
lda friCurRow
clc
adc friTrailingByte
tax
sep #0x20
.a8
lda 0x010000,x
and #0x0F
ora friNibHi
sta 0x010000,x
rep #0x20
.a16
friNoTrail:
; Advance to next row.
lda friCurRow
clc
adc #160
sta friCurRow
lda friRowsLeft
dec a
sta friRowsLeft
brl friRowLoop
friExit:
.a8
.i8
pld
plb
plp
rtl
.section .bss.friPxStart,"aw"
.globl friPxStart
friPxStart:
.zero 2
.section .bss.friPxEnd,"aw"
.globl friPxEnd
friPxEnd:
.zero 2
.section .bss.friLeadingByte,"aw"
.globl friLeadingByte
friLeadingByte:
.zero 2
.section .bss.friHasLeading,"aw"
.globl friHasLeading
friHasLeading:
.zero 2
.section .bss.friMidStart,"aw"
.globl friMidStart
friMidStart:
.zero 2
.section .bss.friMidBytes,"aw"
.globl friMidBytes
friMidBytes:
.zero 2
.section .bss.friHasTrailing,"aw"
.globl friHasTrailing
friHasTrailing:
.zero 2
.section .bss.friTrailingByte,"aw"
.globl friTrailingByte
friTrailingByte:
.zero 2
.section .bss.friNibLo,"aw"
.globl friNibLo
friNibLo:
.zero 2
.section .bss.friNibHi,"aw"
.globl friNibHi
friNibHi:
.zero 2
.section .bss.friDoubled,"aw"
.globl friDoubled
friDoubled:
.zero 2
.section .bss.friCurRow,"aw"
.globl friCurRow
friCurRow:
.zero 2
.section .bss.friRowsLeft,"aw"
.globl friRowsLeft
friRowsLeft:
.zero 2
.section .bss.friFillWord,"aw"
.globl friFillWord
friFillWord:
.zero 2
.section .bss.friSlamWords,"aw"
.globl friSlamWords
friSlamWords:
.zero 2
; ====================================================================
; iigsDrawPixelInner(uint8_t *pixels, uint16_t x, uint16_t y,
; uint16_t nibble)
;
; Plot one 4bpp pixel at (x,y) into a 320x200 surface. Caller has
; already validated bounds and stripped the high nibbles of `nibble`
; (only low 4 bits used). pixels is a 4-byte large-model pointer that
; IS dereferenced via DP-indirect-long.
;
; Offset math: byte = y*160 + (x>>1). Nibble half: x&1 == 0 picks high
; nibble (left pixel), else low.
;
; ABI: clang cdecl. pixels arrives in A:X; x,y,nibble are on the stack.
; llvm-mc only encodes DP-indirect-long
; ([dp],y) for an operand that fits the direct page, so we cannot use a
; 16-bit bss symbol as the pointer slot. Instead, faithful to the spec's
; "pointer arg0 that IS used" recipe, we repoint D at a bank-0 scratch
; struct (dpxlScratch): store pixels (A:X) at dpxlScratch+0..2 and use
; [pix],y with pix=0. Because D no longer overlays the stack frame, the
; x,y,nibble args are read stack-relative.
; Prologue pushes: php(+1)+phb(+1)+phd(+2)=+4. On JSL entry x@4,s y@6,s
; nib@8,s; after the 4 pushes: x@8,s y@10,s nib@12,s.
; pix=0 (D-relative, = dpxlScratch+0)
; xx=8,s yy=10,s nib=12,s
; pix is a 3-byte far pointer (A low16 + X bank byte); we store A then X
; as words -> a harmless 4th filler byte in the 4-byte scratch.
; ====================================================================
.section .text.iigsDrawPixelInner,"ax"
.globl iigsDrawPixelInner
iigsDrawPixelInner:
pix = 0 ; pixels far ptr (D+0..2 = dpxlScratch)
php
phb
phd
rep #0x30
.a16
.i16
; Stash the pixels far pointer into the DP scratch via D-RELATIVE stores
; (AFTER tcd): a plain `sta dpxlScratch` writes the LOAD bank (DBR) while
; [pix],y reads through D, which is ALWAYS bank 0 -- the two disagreed and the
; plot pointer was garbage. phx/pha preserve pix (A:X survive php/phb/phd)
; across the `lda #dpxlScratch` that clobbers A; pulled back so the N,s arg
; offsets below keep their values.
phx ; save pix bank
pha ; save pix offset
lda #dpxlScratch
tcd ; D = dpxlScratch (bank 0); [pix],y == [0],y
pla ; pix offset
sta pix+0 ; D-relative -> bank0:dpxlScratch+0
pla ; pix bank
sta pix+2 ; D-relative -> bank0:dpxlScratch+2
; Compute byte offset = y*160 + (x>>1) into A. Use the LUT.
lda 10,s ; y
asl a ; A = y*2 (LUT byte offset)
tax
lda gRowOffsetLut,x ; A = y*160
sta dpxlTmp
lda 8,s ; x
lsr a ; A = x >> 1
clc
adc dpxlTmp ; A = byte offset within surface
tay
sep #0x20 ; M=8
.a8
lda 8,s ; A = x low byte (parity in LSB)
and #1
bne dpxlOdd
; Even x -> high nibble.
lda 12,s ; nibble
asl a
asl a
asl a
asl a
sta dpxlNibPart
lda [pix],y
and #0x0F
ora dpxlNibPart
sta [pix],y
bra dpxlDone
dpxlOdd:
; Odd x -> low nibble.
lda 12,s ; nibble
and #0x0F
sta dpxlNibPart
lda [pix],y
and #0xF0
ora dpxlNibPart
sta [pix],y
dpxlDone:
rep #0x20 ; back to M=16 for epilogue
.a16
.a8
.i8
pld
plb
plp
rtl
.section .bss.dpxlTmp,"aw"
.globl dpxlTmp
dpxlTmp:
.zero 2
.section .bss.dpxlNibPart,"aw"
.globl dpxlNibPart
dpxlNibPart:
.zero 1
.section .bss.dpxlScratch,"aw"
.globl dpxlScratch
dpxlScratch:
.zero 4
; ====================================================================
; iigsDrawLineInner(uint8_t *pixels, uint16_t x0, uint16_t y0,
; uint16_t x1, uint16_t y1, uint16_t nibble)
;
; Bresenham line plot. Caller has clipped both endpoints into
; [0..319] x [0..199], so no per-pixel bounds check.
;
; Walks (dx, dy, sx, sy, err) and plots inline (no per-pixel
; function call). Each plot does the same parity-aware nibble RMW as
; iigsDrawPixel. State stored in DRAWPRIMS scratch because registers
; are too few for the full Bresenham state plus current pixel address.
;
; ABI: clang cdecl. pixels arrives in A:X (dereferenced via [pix],y);
; x0,y0,x1,y1,nibble on the stack. Same hybrid as DrawPixel: llvm-mc
; cannot encode [dp],y from a 16-bit bss symbol, so we repoint D at a
; bank-0 scratch struct (dlnScratch), stash pixels (A:X) at
; dlnScratch+0..2, and use [pix],y with pix=0 in the loop. The five
; word args are copied into DRAWPRIMS scratch up front (stack-relative),
; so the loop never touches the stack frame again.
; Prologue pushes: php(+1)+phb(+1)+phd(+2)=+4. On JSL entry x0@4,s
; y0@6,s x1@8,s y1@10,s nib@12,s; after the 4 pushes:
; x0@8,s y0@10,s x1@12,s y1@14,s nib@16,s
; pix=0 (D-relative, = dlnScratch+0)
; ====================================================================
.section .text.iigsDrawLineInner,"ax"
.globl iigsDrawLineInner
iigsDrawLineInner:
pix = 0 ; pixels far ptr (D+0..2 = dlnScratch)
; Stash pixels far pointer (A:X) into bank-0 scratch before any push.
rep #0x30
.a16
.i16
sta dlnScratch ; low 16 bits of pixels
stx dlnScratch+2 ; bank byte (+ filler)
php
phb
phd
rep #0x30
.a16
.i16
; Repoint D at the scratch struct so [pix],y (pix=0) is DP-indirect-long.
lda #dlnScratch
tcd
; dx = |x1 - x0|, sx = sign(x1 - x0)
lda 12,s ; x1
sec
sbc 8,s ; x0
bpl dlnDxPos
eor #0xFFFF
clc
adc #1 ; A = x0 - x1 (positive |dx|)
sta dlnDx
lda #0xFFFF
sta dlnSx ; sx = -1
bra dlnDxDone
dlnDxPos:
sta dlnDx
lda #1
sta dlnSx
dlnDxDone:
; dy = -|y1 - y0|, sy = sign(y1 - y0). Bresenham uses negative dy.
lda 14,s ; y1
sec
sbc 10,s ; y0
bpl dlnDyPos
sta dlnDy ; A is negative; that's dy with the sign already
lda #0xFFFF
sta dlnSy ; sy = -1
bra dlnDyDone
dlnDyPos:
; y1 >= y0 -> raw dy positive; negate for Bresenham (-|y1-y0|).
eor #0xFFFF
clc
adc #1
sta dlnDy
lda #1
sta dlnSy
dlnDyDone:
; err = dx + dy
lda dlnDx
clc
adc dlnDy
sta dlnErr
; Copy x0, y0 to running state slots.
lda 8,s ; x0
sta dlnX
lda 10,s ; y0
sta dlnY
; Cache x1, y1 so the loop's compare is fast.
lda 12,s ; x1
sta dlnX1
lda 14,s ; y1
sta dlnY1
; Precompute both 16-bit nibble OR-values ONCE (stays in M=16 the
; whole loop -- no per-pixel SEP/REP flip). The nibble RMW reads the
; target byte plus its adjacent high byte (16-bit), masks so only the
; target nibble changes, and writes both bytes back; the high byte is
; preserved by the mask + zero OR-high-byte, so the result is identical
; to the old 8-bit single-byte RMW. dlnNibHi/dlnNibLo have low byte =
; (nib<<4) / nib and high byte = 0.
lda 16,s ; nibble
and #0x000F
sta dlnNibLo ; low byte = nib, high byte = 0
asl a
asl a
asl a
asl a
sta dlnNibHi ; low byte = nib<<4, high byte = 0
dlnLoop:
; --- Plot pixel at (dlnX, dlnY) --- LUT path. Stays in M=16.
lda dlnY
asl a ; A = y*2 (LUT byte offset)
tax
lda gRowOffsetLut,x ; A = y*160
sta dlnTmp
lda dlnX
lsr a ; x >> 1
clc
adc dlnTmp ; byte offset
tay
lda dlnX ; x parity
and #1
bne dlnPlotOdd
; Even x -> high nibble. Mask keeps low nibble of target byte and the
; whole adjacent high byte; OR sets nib<<4 into the high nibble.
lda [pix],y
and #0xFF0F
ora dlnNibHi
sta [pix],y
bra dlnPlotDone
dlnPlotOdd:
; Odd x -> low nibble. Mask keeps high nibble of target byte and the
; whole adjacent high byte; OR sets nib into the low nibble.
lda [pix],y
and #0xF0FF
ora dlnNibLo
sta [pix],y
dlnPlotDone:
; --- Loop test: if X==X1 and Y==Y1, done ---
lda dlnX
cmp dlnX1
bne dlnStep
lda dlnY
cmp dlnY1
bne dlnStep
brl dlnExit
dlnStep:
; e2 = err << 1 (signed). Compare e2 against dy and dx.
lda dlnErr
asl a ; e2 = err * 2
sta dlnE2
; if (e2 >= dy) { err += dy; X += sx; }
; dy is negative, so signed-compare with BMI/BPL needed.
; Test e2 - dy >= 0 via SEC; SBC dy; BPL.
sec
sbc dlnDy
bmi dlnSkipX ; e2 < dy
lda dlnErr
clc
adc dlnDy
sta dlnErr
lda dlnX
clc
adc dlnSx
sta dlnX
dlnSkipX:
; if (e2 <= dx) { err += dx; Y += sy; }
; Test dx - e2 >= 0 via dx - e2 (BPL means e2 <= dx).
lda dlnDx
sec
sbc dlnE2
bmi dlnSkipY ; dx < e2
lda dlnErr
clc
adc dlnDx
sta dlnErr
lda dlnY
clc
adc dlnSy
sta dlnY
dlnSkipY:
brl dlnLoop
dlnExit:
.a8
.i8
pld
plb
plp
rtl
.section .bss.dlnDx,"aw"
.globl dlnDx
dlnDx:
.zero 2
.section .bss.dlnDy,"aw"
.globl dlnDy
dlnDy:
.zero 2
.section .bss.dlnSx,"aw"
.globl dlnSx
dlnSx:
.zero 2
.section .bss.dlnSy,"aw"
.globl dlnSy
dlnSy:
.zero 2
.section .bss.dlnErr,"aw"
.globl dlnErr
dlnErr:
.zero 2
.section .bss.dlnE2,"aw"
.globl dlnE2
dlnE2:
.zero 2
.section .bss.dlnX,"aw"
.globl dlnX
dlnX:
.zero 2
.section .bss.dlnY,"aw"
.globl dlnY
dlnY:
.zero 2
.section .bss.dlnX1,"aw"
.globl dlnX1
dlnX1:
.zero 2
.section .bss.dlnY1,"aw"
.globl dlnY1
dlnY1:
.zero 2
.section .bss.dlnNibLo,"aw"
.globl dlnNibLo
dlnNibLo:
.zero 2
.section .bss.dlnNibHi,"aw"
.globl dlnNibHi
dlnNibHi:
.zero 2
.section .bss.dlnTmp,"aw"
.globl dlnTmp
dlnTmp:
.zero 2
.section .bss.dlnScratch,"aw"
.globl dlnScratch
dlnScratch:
.zero 4
; ===== ported: circles =====
; ================================================================
; iigsDrawCircleInner(uint8_t *pixels, uint16_t cx, uint16_t cy,
; uint16_t r, uint16_t nibble)
;
; Bresenham midpoint circle outline. Caller has verified that the
; whole bounding box (cx-r..cx+r, cy-r..cy+r) fits inside the
; surface, so the inner loop plots all 8 octants without per-pixel
; clip checks. Each iteration computes 4 row-base byte offsets
; (cy +/- y)*160, (cy +/- x)*160 then plots 8 pixels inline.
;
; ABI (llvm-mos w65816 cdecl):
; pix -> arg0 in A:X (A=offset low16, X low byte=bank). USED via
; [pix],y, so it must live in the direct page. We stash it
; into bank-0 scratch dcPixPtr (offset@+0, bank@+2) BEFORE
; setting D, then point D at dcPixPtr so the original
; `[pix],y` (pix equ 0 -> [D+0],y) loads are byte-identical.
; php/phb/phd do not touch A/X, so A:X still hold pix when
; we capture them after the pushes.
; cx,cy,r,nibble -> on the stack. On JSL entry cx@4,s cy@6,s r@8,s
; nib@10,s. After php(+1)+phb(+1)+phd(+2)=+4: cx@8,s
; cy@10,s r@12,s nib@14,s. S never moves in the body
; (no pushes in the loop), so these N,s reads stay valid.
; They are stack reads because D hosts the pix pointer.
; ================================================================
.section .text.iigsDrawCircleInner,"ax"
.globl iigsDrawCircleInner
pix = 0 ; D-relative: D points at dcPixPtr
dcArgCx = 8 ; cx @ 8,s after +4 prologue
dcArgCy = 10 ; cy @ 10,s
dcArgR = 12 ; r @ 12,s
dcArgNib = 14 ; nibble @ 14,s
iigsDrawCircleInner:
php
phb
phd
rep #0x30
.a16
.i16
; Stash the pix far pointer (still in A:X) into bank-0 scratch and
; point D at it so [pix],y works unchanged.
; Store pix into the DP scratch via D-RELATIVE stores (AFTER tcd): a
; plain `sta dcPixPtr` writes the LOAD bank (DBR) while [pix],y reads
; through D, which is ALWAYS bank 0 -- the two disagreed, so the plot
; pointer was garbage and nothing drew. phx/pha preserve pix across the
; `lda #dcPixPtr` that clobbers A, then are pulled back so the N,s arg
; offsets below do not shift.
phx ; save pix bank
pha ; save pix offset
lda #dcPixPtr
tcd ; D = &dcPixPtr (bank 0); [pix],y == [0],y
pla ; pix offset
sta pix+0 ; D-relative -> bank0:dcPixPtr+0
pla ; pix bank
sta pix+2 ; D-relative -> bank0:dcPixPtr+2
; Cache the nibble in 8-bit form: dcNibLo = nib, dcNibHi = nib<<4.
lda dcArgNib, s
and #0x000F
sta dcNibLo ; low byte = nib, high byte = 0
asl a
asl a
asl a
asl a
sta dcNibHi ; low byte = nib<<4
; x = r, y = 0, err = 1 - x.
lda dcArgR, s
sta dcX
lda #0
sta dcY
lda #1
sec
sbc dcArgR, s
sta dcErr
; Initialize the 4 octant row bases once, then maintain them
; incrementally in the Bresenham update (instead of 4 LUT lookups per
; iteration). At entry x=r, y=0: RowYP=RowYN=acy*160,
; RowXP=(acy+r)*160, RowXN=(acy-r)*160.
lda dcArgCy, s
asl a
tax
lda gRowOffsetLut, x
sta dcRowYP
sta dcRowYN
lda dcArgCy, s
clc
adc dcArgR, s
asl a
tax
lda gRowOffsetLut, x
sta dcRowXP
lda dcArgCy, s
sec
sbc dcArgR, s
asl a
tax
lda gRowOffsetLut, x
sta dcRowXN
dcLoop:
; Loop guard: if x < y we are done.
lda dcX
cmp dcY
bcs dcLoopBody ; x >= y -> continue
brl dcExit
dcLoopBody:
; Row bases (dcRowYP/YN/XP/XN) are maintained incrementally in the
; Bresenham update -- no per-iteration LUT lookups here.
;
; 8 octant plots, fully inlined. Each plot:
; 1. col = (acx +/- dcX|dcY) -> A
; 2. stash col in X (free across the plots); the parity test reads it
; back with txa instead of a memory round-trip.
; 3. byteIdx = col >> 1; byte addr = byteIdx + rowBase -> Y
; 4. test col & 1 (via txa); do high or low nibble RMW
;
; All 8 octant plots stay in M=16 -- no per-plot SEP/REP flip. The
; nibble RMW reads the target byte plus its adjacent high byte (16-bit
; load), masks so only the target nibble changes ($FF0F keeps low
; nibble + whole high byte for even/high-nibble x, $F0FF keeps high
; nibble + whole high byte for odd/low-nibble x), then writes both
; bytes back.
; Octants 1-4: y-row pair (cx +/- x, cy +/- y).
; Plot 1: (cx+x, cy+y)
lda dcArgCx, s
clc
adc dcX
tax
lsr a
clc
adc dcRowYP
tay
txa
and #1
bne dcOdd1
lda [pix], y
and #0xFF0F
ora dcNibHi
sta [pix], y
bra dcDone1
dcOdd1:
lda [pix], y
and #0xF0FF
ora dcNibLo
sta [pix], y
dcDone1:
; Plot 2: (cx-x, cy+y)
lda dcArgCx, s
sec
sbc dcX
tax
lsr a
clc
adc dcRowYP
tay
txa
and #1
bne dcOdd2
lda [pix], y
and #0xFF0F
ora dcNibHi
sta [pix], y
bra dcDone2
dcOdd2:
lda [pix], y
and #0xF0FF
ora dcNibLo
sta [pix], y
dcDone2:
; Plot 3: (cx+x, cy-y)
lda dcArgCx, s
clc
adc dcX
tax
lsr a
clc
adc dcRowYN
tay
txa
and #1
bne dcOdd3
lda [pix], y
and #0xFF0F
ora dcNibHi
sta [pix], y
bra dcDone3
dcOdd3:
lda [pix], y
and #0xF0FF
ora dcNibLo
sta [pix], y
dcDone3:
; Plot 4: (cx-x, cy-y)
lda dcArgCx, s
sec
sbc dcX
tax
lsr a
clc
adc dcRowYN
tay
txa
and #1
bne dcOdd4
lda [pix], y
and #0xFF0F
ora dcNibHi
sta [pix], y
bra dcDone4
dcOdd4:
lda [pix], y
and #0xF0FF
ora dcNibLo
sta [pix], y
dcDone4:
; Octants 5-8: x-row pair (cx +/- y, cy +/- x).
; Plot 5: (cx+y, cy+x)
lda dcArgCx, s
clc
adc dcY
tax
lsr a
clc
adc dcRowXP
tay
txa
and #1
bne dcOdd5
lda [pix], y
and #0xFF0F
ora dcNibHi
sta [pix], y
bra dcDone5
dcOdd5:
lda [pix], y
and #0xF0FF
ora dcNibLo
sta [pix], y
dcDone5:
; Plot 6: (cx-y, cy+x)
lda dcArgCx, s
sec
sbc dcY
tax
lsr a
clc
adc dcRowXP
tay
txa
and #1
bne dcOdd6
lda [pix], y
and #0xFF0F
ora dcNibHi
sta [pix], y
bra dcDone6
dcOdd6:
lda [pix], y
and #0xF0FF
ora dcNibLo
sta [pix], y
dcDone6:
; Plot 7: (cx+y, cy-x)
lda dcArgCx, s
clc
adc dcY
tax
lsr a
clc
adc dcRowXN
tay
txa
and #1
bne dcOdd7
lda [pix], y
and #0xFF0F
ora dcNibHi
sta [pix], y
bra dcDone7
dcOdd7:
lda [pix], y
and #0xF0FF
ora dcNibLo
sta [pix], y
dcDone7:
; Plot 8: (cx-y, cy-x)
lda dcArgCx, s
sec
sbc dcY
tax
lsr a
clc
adc dcRowXN
tay
txa
and #1
bne dcOdd8
lda [pix], y
and #0xFF0F
ora dcNibHi
sta [pix], y
bra dcDone8
dcOdd8:
lda [pix], y
and #0xF0FF
ora dcNibLo
sta [pix], y
dcDone8:
; Update Bresenham: y++; if err<=0: err+=2y+1; else x--; err+=2(y-x)+1.
lda dcY
inc a
sta dcY ; y now = old y + 1
; y increased by 1: RowYP moves down a row (+160), RowYN moves up (-160).
lda dcRowYP
clc
adc #160
sta dcRowYP
lda dcRowYN
sec
sbc #160
sta dcRowYN
lda dcErr
bmi dcErrLE ; err < 0 -> err <= 0
bne dcErrGT ; err > 0
; err == 0: take the LE branch (err += 2y+1).
dcErrLE:
lda dcY
asl a
clc
adc #1
clc
adc dcErr
sta dcErr
brl dcLoop
dcErrGT:
lda dcX
dec a
sta dcX
; x decreased by 1: RowXP moves up a row (-160), RowXN moves down (+160).
lda dcRowXP
sec
sbc #160
sta dcRowXP
lda dcRowXN
clc
adc #160
sta dcRowXN
; err += 2*(y-x) + 1
lda dcY
sec
sbc dcX
asl a
clc
adc #1
clc
adc dcErr
sta dcErr
brl dcLoop
dcExit:
.a8
.i8
pld
plb
plp
rtl
; ================================================================
; iigsFillCircleInner(uint8_t *pixels, uint16_t cx, uint16_t cy,
; uint16_t r, uint16_t fillWord)
;
; Filled circle with horizontal-span scanline output. Caller has
; verified the bbox fits inside the surface, so the inner loop fills
; every span unconditionally. Maintains xx=x*x and yy=y*y
; incrementally so the hot path uses only 16-bit add/sub/cmp -- no
; 65816 multiply at all (other than one r*r at setup).
;
; fillWord is the doubled byte (low byte) replicated, e.g. nibble 3
; -> $3333 (we only use the low byte for nibble RMW).
;
; ABI (llvm-mos w65816 cdecl):
; pix -> arg0 in A:X. USED via [fpix],y, so stashed into bank-0
; scratch fcPixPtr (offset@+0, bank@+2) and D pointed at
; it; fpix equ 0 -> [D+0],y.
; cx,cy,r,fillWord -> on the stack. JSL entry cx@4,s cy@6,s r@8,s
; fill@10,s. After php+phb+phd (+4): cx@8,s cy@10,s
; r@12,s fill@14,s. S is constant in the outer body; the
; jsr-called helpers (fcMul16/fcDoSpan) only touch
; scratch globals + [fpix],y, never N,s, so the inner
; jsr-pushed return PC never disturbs an arg read.
; ================================================================
.section .text.iigsFillCircleInner,"ax"
.globl iigsFillCircleInner
fpix = 0 ; D-relative: D points at fcPixPtr
fcArgCx = 8 ; cx @ 8,s after +4 prologue
fcArgCy = 10 ; cy @ 10,s
fcArgR = 12 ; r @ 12,s
fcArgFill = 14 ; fillWord @ 14,s
iigsFillCircleInner:
php
phb
phd
rep #0x30
.a16
.i16
; Stash the pix far pointer (A:X) into bank-0 scratch and point D at it.
sta fcPixPtr ; offset low 16 bits
txa
sta fcPixPtr+2 ; low byte = bank
lda #fcPixPtr
tcd ; D = &fcPixPtr ; [fpix],y == [0],y
; Cache fill nibble bytes. ffill low byte = doubled byte (e.g., $33).
; fcFillByte = doubled low byte (full byte fills + low nibble).
; fcFillHi = (nibble << 4) only (high nibble RMW).
; fcFillLo = nibble only (low nibble RMW).
lda fcArgFill, s
and #0x00FF
sta fcFillByte ; doubled byte e.g. $33
lsr a
lsr a
lsr a
lsr a ; A = nibble
sta fcFillLo
asl a
asl a
asl a
asl a ; A = nibble << 4
sta fcFillHi
; Compute r*r via 16-bit shift-and-add. Only done once.
lda fcArgR, s
sta fcMulA
lda fcArgR, s
sta fcMulB
jsr fcMul16
sta fcR2
sta fcXX ; xx = r*r initially
; yy = 0, x = r, y = 0.
lda #0
sta fcYY
sta fcY
lda fcArgR, s
sta fcX
fcOuterLoop:
; Inner: while (xx + yy) > r2 then xx -= 2x - 1; x--.
fcInnerLoop:
lda fcXX
clc
adc fcYY
cmp fcR2
bcc fcInnerDone ; xx+yy < r2
beq fcInnerDone ; xx+yy == r2 (acceptable)
; xx + yy > r2 -> decrement
lda fcX
asl a ; A = 2x
sec
sbc #1 ; A = 2x - 1
sta fcDec
lda fcXX
sec
sbc fcDec
sta fcXX
lda fcX
dec a
sta fcX
bra fcInnerLoop
fcInnerDone:
; leftCol = cx - x, rightCol = cx + x for both lower and (if y>0) upper.
lda fcArgCx, s
sec
sbc fcX
sta fcLeftCol
lda fcArgCx, s
clc
adc fcX
sta fcRightCol
; Lower row: rowBase = (cy + y) * 160
lda fcArgCy, s
clc
adc fcY
asl a ; A = y*2 (LUT byte offset)
tax
lda gRowOffsetLut, x ; A = y*160
sta fcSpanRowBase
jsr fcDoSpan
; If y > 0: upper row at (cy - y).
lda fcY
beq fcSkipUpper
lda fcArgCy, s
sec
sbc fcY
asl a ; A = y*2 (LUT byte offset)
tax
lda gRowOffsetLut, x ; A = y*160
sta fcSpanRowBase
jsr fcDoSpan
fcSkipUpper:
; yy += 2*y + 1
lda fcY
asl a
clc
adc #1
clc
adc fcYY
sta fcYY
; y++
lda fcY
inc a
sta fcY
; if y > r: exit, else loop
cmp fcArgR, s
bcc fcContinue
beq fcContinue
brl fcExit
fcContinue:
brl fcOuterLoop
fcExit:
.a8
.i8
pld
plb
plp
rtl
; ----------------------------------------------------------------
; fcMul16: A = fcMulA * fcMulB (16-bit unsigned, low 16 of result).
; Standard shift-and-add. 16 iterations. Caller has M=16, X=16.
; ----------------------------------------------------------------
fcMul16:
.a16
.i16
lda #0
sta fcMulRes
ldx #16
fcMulLoop:
; result <<= 1
lda fcMulRes
asl a
sta fcMulRes
; multiplier <<= 1, MSB -> C
lda fcMulA
asl a
sta fcMulA
bcc fcMulSkip
lda fcMulRes
clc
adc fcMulB
sta fcMulRes
fcMulSkip:
dex
bne fcMulLoop
lda fcMulRes
rts
; ----------------------------------------------------------------
; fcDoSpan: fill horizontal span from fcLeftCol to fcRightCol on row
; whose byte base is in fcSpanRowBase. Byte-aligned fill for the
; middle bytes and per-pixel RMW for the partial-nibble leading /
; trailing bytes.
;
; Inputs: fcLeftCol, fcRightCol, fcSpanRowBase (M=16 values).
; Trashes A, X, Y, P. Preserves D and B (relies on D = &fcPixPtr).
; ----------------------------------------------------------------
fcDoSpan:
.a16
.i16
; leftByte = leftCol >> 1, leftPartial = leftCol & 1
lda fcLeftCol
lsr a
sta fcLeftByte
lda fcLeftCol
and #1
sta fcLeftPartial
; rightByte = rightCol >> 1, rightPartial = !(rightCol & 1)
lda fcRightCol
lsr a
sta fcRightByte
lda fcRightCol
and #1
eor #1
sta fcRightPartial
; If leftByte == rightByte: single-byte case
lda fcLeftByte
cmp fcRightByte
bne fcSpanMulti
brl fcSpanSingle
fcSpanMulti:
; Multi-byte case.
; Leading partial (low nibble of leftByte) if leftPartial.
lda fcLeftPartial
beq fcSkipLP
lda fcSpanRowBase
clc
adc fcLeftByte
tay
sep #0x20
.a8
lda [fpix], y
and #0xF0
ora fcFillLo
sta [fpix], y
rep #0x20
.a16
fcSkipLP:
; Trailing partial (high nibble of rightByte) if rightPartial.
lda fcRightPartial
beq fcSkipRP
lda fcSpanRowBase
clc
adc fcRightByte
tay
sep #0x20
.a8
lda [fpix], y
and #0x0F
ora fcFillHi
sta [fpix], y
rep #0x20
.a16
fcSkipRP:
; Mid bytes: (leftByte+leftPartial) .. (rightByte-rightPartial) incl.
lda fcLeftByte
clc
adc fcLeftPartial
sta fcMidStart
lda fcRightByte
sec
sbc fcRightPartial
sta fcMidEnd
cmp fcMidStart
bcs fcMidGo
rts ; midEnd < midStart
fcMidGo:
lda fcMidEnd
sec
sbc fcMidStart
inc a
sta fcMidCount
lda fcSpanRowBase
clc
adc fcMidStart
tay ; Y = byte offset
lda fcMidCount
tax ; X = byte counter
sep #0x20
.a8
lda fcFillByte ; doubled byte
fcFillLoop:
sta [fpix], y
iny
dex
bne fcFillLoop
rep #0x20
.a16
rts
fcSpanSingle:
; Single-byte case: leftByte == rightByte. Three sub-cases:
; leftPartial=1 -> low nibble only
; rightPartial=1 -> high nibble only
; neither -> full byte (two adjacent pixels)
lda fcSpanRowBase
clc
adc fcLeftByte
tay
sep #0x20
.a8
lda fcLeftPartial
beq fcSbCheckRP
lda [fpix], y
and #0xF0
ora fcFillLo
sta [fpix], y
rep #0x20
.a16
rts
fcSbCheckRP:
.a8 ; reached via BEQ from M=8 path
lda fcRightPartial
beq fcSbFull
lda [fpix], y
and #0x0F
ora fcFillHi
sta [fpix], y
rep #0x20
.a16
rts
fcSbFull:
.a8 ; reached via BEQ from M=8 path
lda fcFillByte
sta [fpix], y
rep #0x20
.a16
rts
; ================================================================
; Scratch slots (bank-0 BSS). dcSavedCol is kept for byte-faithful
; parity with the original DRAWPRIMS layout even though the inlined
; plots no longer round-trip through it. dcPixPtr / fcPixPtr hold the
; stashed arg0 far pointer for the [pix],y / [fpix],y direct-page
; indirect-long loads.
; ================================================================
.section .bss.dcPixPtr,"aw"
.globl dcPixPtr
dcPixPtr:
.zero 4
.section .bss.dcX,"aw"
.globl dcX
dcX:
.zero 2
.section .bss.dcY,"aw"
.globl dcY
dcY:
.zero 2
.section .bss.dcErr,"aw"
.globl dcErr
dcErr:
.zero 2
.section .bss.dcNibLo,"aw"
.globl dcNibLo
dcNibLo:
.zero 2
.section .bss.dcNibHi,"aw"
.globl dcNibHi
dcNibHi:
.zero 2
.section .bss.dcRowYP,"aw"
.globl dcRowYP
dcRowYP:
.zero 2
.section .bss.dcRowYN,"aw"
.globl dcRowYN
dcRowYN:
.zero 2
.section .bss.dcRowXP,"aw"
.globl dcRowXP
dcRowXP:
.zero 2
.section .bss.dcRowXN,"aw"
.globl dcRowXN
dcRowXN:
.zero 2
.section .bss.dcSavedCol,"aw"
.globl dcSavedCol
dcSavedCol:
.zero 2
.section .bss.fcPixPtr,"aw"
.globl fcPixPtr
fcPixPtr:
.zero 4
.section .bss.fcX,"aw"
.globl fcX
fcX:
.zero 2
.section .bss.fcY,"aw"
.globl fcY
fcY:
.zero 2
.section .bss.fcXX,"aw"
.globl fcXX
fcXX:
.zero 2
.section .bss.fcYY,"aw"
.globl fcYY
fcYY:
.zero 2
.section .bss.fcR2,"aw"
.globl fcR2
fcR2:
.zero 2
.section .bss.fcDec,"aw"
.globl fcDec
fcDec:
.zero 2
.section .bss.fcMulA,"aw"
.globl fcMulA
fcMulA:
.zero 2
.section .bss.fcMulB,"aw"
.globl fcMulB
fcMulB:
.zero 2
.section .bss.fcMulRes,"aw"
.globl fcMulRes
fcMulRes:
.zero 2
.section .bss.fcLeftCol,"aw"
.globl fcLeftCol
fcLeftCol:
.zero 2
.section .bss.fcRightCol,"aw"
.globl fcRightCol
fcRightCol:
.zero 2
.section .bss.fcSpanRowBase,"aw"
.globl fcSpanRowBase
fcSpanRowBase:
.zero 2
.section .bss.fcLeftByte,"aw"
.globl fcLeftByte
fcLeftByte:
.zero 2
.section .bss.fcRightByte,"aw"
.globl fcRightByte
fcRightByte:
.zero 2
.section .bss.fcLeftPartial,"aw"
.globl fcLeftPartial
fcLeftPartial:
.zero 2
.section .bss.fcRightPartial,"aw"
.globl fcRightPartial
fcRightPartial:
.zero 2
.section .bss.fcMidStart,"aw"
.globl fcMidStart
fcMidStart:
.zero 2
.section .bss.fcMidEnd,"aw"
.globl fcMidEnd
fcMidEnd:
.zero 2
.section .bss.fcMidCount,"aw"
.globl fcMidCount
fcMidCount:
.zero 2
.section .bss.fcFillByte,"aw"
.globl fcFillByte
fcFillByte:
.zero 2
.section .bss.fcFillLo,"aw"
.globl fcFillLo
fcFillLo:
.zero 2
.section .bss.fcFillHi,"aw"
.globl fcFillHi
fcFillHi:
.zero 2
; ===== ported: blit / dirty / input / flood =====
; ====================================================================
; frag_blitflood.s - llvm-mc GAS (w65816):
; iigsBlitStageToShr, iigsMarkDirtyRowsInner, iigsPollJoystickInner,
; iigsInputSnapshot, iigsFloodWalkAndScansInner
;
; Shared ABI facts (clang w65816 cdecl, per PORT_SPEC.md):
; - arg0 (pointer/long) arrives in A=low16, X=high16; word arg0 in A.
; It is NOT on the stack.
; - arg1, arg2, ... are pushed right-to-left; on JSL entry arg1 byte0
; is at 4,s. A word arg is 2 bytes, a pointer/long arg is 4 bytes.
; - Each prologue push shifts every stack offset up: php=+1, phb=+1,
; phd=+2, pha/phx=+2.
; - Memory model: DBR=0 for C-called code; all globals + bank-0 soft
; switches reachable via 16-bit absolute (no long prefix needed;
; `$00Cxxx` operands are plain 16-bit abs).
; ====================================================================
; ====================================================================
; iigsBlitStageToShr(uint8_t *scbPtr, uint16_t *palettePtr,
; uint16_t uploadFlags)
;
; SHR upload: optional SCB (200 bytes) and palette (512 bytes) uploads
; via direct-$E1 MVN, then a per-row dirty-skipped, chunked-SEI PEI
; slam of the 32000-byte pixel buffer from the bank-$01 back buffer to
; the $E1 SHR frame buffer. WIDE dirty rows take the unrolled 80-PEI
; full-row slam; NARROW dirty rows take an exact-extent MVN.
;
; Original args (D-frame, D=SP+8, every arg on stack):
; scbPtr D+0..3 (long ptr)
; palettePtr D+4..7 (long ptr)
; uploadFlags D+8..9 (uint16_t bitmask)
;
; ABI: under clang scbPtr (arg0, ptr) arrives in A:X and is NOT on the
; stack; palettePtr (arg1, ptr) and uploadFlags (arg2, word) are on the
; stack. JSL entry: palettePtr @4,s, uploadFlags @8,s; after
; php+phb+phd (+4): palettePtr @8,s, uploadFlags @12,s.
;
; The original read scbPtr/palettePtr/uploadFlags via a D-frame
; (bscb=D+0, bpal=D+4, bflags=D+8). We rebuild that exact frame in a
; private bank-0 scratch block (bsFrame) -- D+0..3 must be writable
; scratch and cannot overlap the saved registers / return address the
; way the original's stack args did -- copy scbPtr from A:X to +0..3,
; palettePtr from the stack to +4..7, uploadFlags to +8, then point D
; at bsFrame so all `bscb`/`bpal`/`bflags` reads work unchanged. Phase
; 3 (the slam) clobbers D freely via TCD to row bases (exactly as the
; original); the final PLD restores the caller's D (saved by our PHD)
; AND therefore restores DBR=0 expectations -- DBR is restored by PLB.
; ====================================================================
.section .text.iigsBlitStageToShr,"ax"
.globl iigsBlitStageToShr
iigsBlitStageToShr:
bscb = 0
bpal = 4
bflags = 8
; A dirty row with (maxWord-minWord) < PEI_NARROW_MAX uses an exact MVN
; instead of the full-row PEI slam. 32 keeps a safe margin below the
; ~34-word MVN/PEI break-even.
PEI_NARROW_MAX = 32
php
phb
phd
rep #0x30
.a16
.i16
; DBR must be the CALLER's data bank (= the bank our globals /
; D-frame live in). crt0 set it and a JSL preserves it across
; the cross-seg call, so at entry DBR already equals that bank.
; We must NOT force DBR := PBR (`phk`): under MULTI-SEGMENT this
; asm can land in a different bank (seg2) than the globals
; (seg1), so PBR != globals bank and phk would point DBR at the
; wrong bank -> garbage dirty-band / gPei* reads -> corrupted
; present. (Two earlier GS/OS traps this path also fixes: args
; were read back via a DP frame under DBR=$08 -> wrong bank, now
; read via the SAME absolute addressing; and each MVN leaves
; DBR=$E1, undone after every MVN -- see below.)
;
; Stash args into absolute scratch (DBR = caller/globals bank).
; scbPtr in A:X; palettePtr/flags on stack (post php+phb+phd
; +4: @8,s @12,s).
sta bsFrame+0 ; scbPtr.lo (A)
stx bsFrame+2 ; scbPtr.hi (X, bank+pad)
lda 8,s ; palettePtr.lo
sta bsFrame+4
lda 10,s ; palettePtr.hi (bank+pad)
sta bsFrame+6
lda 12,s ; uploadFlags
sta bsFrame+8
; Capture the caller's DBR (= globals bank, saved by phb at
; 3,s) into a FIXED bank-$E1 byte. Each MVN below leaves
; DBR=$E1; we reload this byte (long-addressed, so it is both
; DBR- and SP-independent -- the PEI slam relocates SP) to put
; DBR back at the globals bank. $E1:9DFE is in the unused SCB
; tail ($9DC8-$9DFF), never touched by the SCB/palette/pixel
; MVNs.
sep #0x20
.a8
lda 3,s ; caller DBR (globals bank)
sta 0xE19DFE ; SP/DBR-independent stash
rep #0x20
.a16
; 1. SCB upload (200 bytes) via MVN, only when the SCB dirty bit is
; set. Source bank is runtime-patched into the MVN instruction
; (encoding: $54 dst src, so byte +2 is src).
lda bsFrame+8 ; uploadFlags (absolute, DBR=PBR)
and #1
beq bsSkipScb
sep #0x20
.a8
lda bsFrame+2 ; scbPtr bank byte (DBR = globals)
; Self-mod the MVN src-bank byte. The store must reach the MVN
; instruction in THIS code's bank (PBR), not DBR (globals):
; under multi-seg this asm can land in seg2 while DBR=seg1, so a
; DBR-relative `sta mvnScbInst+2` would patch the wrong bank and
; leave the MVN src=$00 -> garbage SCB upload. phk/plb makes
; DBR:=PBR for the store, then we restore DBR := globals.
phk
plb
sta mvnScbInst+2
lda 0xE19DFE ; globals bank (long: DBR/SP-indep)
pha
plb
rep #0x20
.a16
lda bsFrame+0 ; scbPtr offset
tax
ldy #0x9D00
lda #199
mvnScbInst: mvn 0xE1, 0x00 ; mvn dst,src (src bank runtime-patched)
; MVN left DBR=$E1; restore DBR := globals bank (caller's,
; stashed at $E1:9DFE). NOT phk -- under multi-seg PBR is this
; asm's bank, not the globals bank. A is dead here.
sep #0x20
.a8
lda 0xE19DFE
pha
plb
rep #0x20
.a16
bsSkipScb:
; 2. Palette upload (512 bytes) via MVN, only when the palette dirty
; bit is set. Same trick.
lda bsFrame+8 ; uploadFlags (absolute, DBR=PBR)
and #2
beq bsSkipPal
sep #0x20
.a8
lda bsFrame+6 ; palettePtr bank byte (DBR = globals)
; Self-mod the MVN src-bank byte via DBR:=PBR (see the SCB MVN
; above) so the patch reaches the instruction in THIS code's
; bank, then restore DBR := globals.
phk
plb
sta mvnPalInst+2
lda 0xE19DFE ; globals bank (long: DBR/SP-indep)
pha
plb
rep #0x20
.a16
lda bsFrame+4 ; palettePtr offset
tax
ldy #0x9E00
lda #511
mvnPalInst: mvn 0xE1, 0x00
; MVN left DBR=$E1; restore DBR := globals bank (see SCB MVN).
sep #0x20
.a8
lda 0xE19DFE
pha
plb
rep #0x20
.a16
bsSkipPal:
; 3. Pixel blit via PEI-slam, with per-row dirty skip and chunked SEI.
tsc
sta gPeiOrigSp
sep #0x20
.a8
; Soft switches are accessed via LONG addressing to bank $E0
; ($E0Cxxx), NOT `sta $C035`. Under the GS/OS Loader DBR=$08,
; a DBR-relative `lda/sta $C0xx` hits bank-$08 RAM, not the
; I/O soft switches -- so shadow/RAMRD/RAMWRT silently never
; toggle and the slam writes nothing to $E1. Bank $E0 always
; holds the real I/O. (Technique from the prior joeylib's
; asmSlam: >$E0C035 for shadow, >$E0C0xx for RAMRD/RAMWRT.)
lda 0xE0C035
sta gPeiOrigShadow
rep #0x20
.a16
ldx #0 ; X = absolute row counter
peiChunkBegin:
ldy #0 ; Y = in-chunk row counter
sei
; Shadow/AUXWRITE/RAMRD are scoped to each PEI burst below
; (peiSlamRow), NOT the whole chunk. The per-row control
; logic reads/writes bank-0 globals (dirty bands, row LUT,
; gPei* scratch, gPeiOrigSp); RAMRD/AUXWRITE on would
; misdirect those to aux bank $01 -- reading garbage dirty
; decisions and a garbage saved SP that ran RTL off to
; $FFFE. MVN rows need no switches (explicit src/dst banks).
peiRowLoop:
cpx #200
bcc peiNotAllDone ; X < 200 -> still rows to do
brl peiChunkEnd
peiNotAllDone:
cpy #40
bcc peiCheckDirty ; Y < 40 -> chunk has room
brl peiChunkEnd
peiCheckDirty:
sep #0x20
.a8
lda gStageMinWord,x
cmp gStageMaxWord,x
rep #0x20
.a16
bcc peiRowDirty ; min < max -> dirty
beq peiRowDirty ; min == max -> dirty (1 word)
inx ; clean row, skip
iny
brl peiRowLoop
; Dirty row: narrow spans take the exact-extent MVN; wide spans take
; the full-row PEI slam ($01 stack-push, shadowed $01->$E1). The slam is
; the GS fast path (the whole reason the stage lives at $01/2000 and the
; screen at $E1/2000); re-enabled now that the DBR/D-frame bugs that were
; corrupting its dirty-band reads are fixed, plus the shadow mask clears
; bit 4 (aux-bank shadow inhibit) so $01->$E1 shadowing is actually live.
peiRowDirty:
sep #0x20
.a8
lda gStageMaxWord,x
sec
sbc gStageMinWord,x ; maxWord - minWord (>= 0)
rep #0x20
.a16
and #0x00FF
cmp #PEI_NARROW_MAX
bcs peiSlamRow ; wide span -> full-row PEI slam
brl peiMvnRow ; narrow span -> exact-extent MVN
peiSlamRow:
; Save X and Y into long-mode scratch (stack is hijacked into $E1).
txa
sta gPeiCurRow
tya
sta gPeiChunkRow
lda gPeiCurRow
asl a ; A = y*2 (LUT byte offset)
tax
lda gRowOffsetLut,x ; A = y*160
clc
adc #0x2000 ; A = row_start
tcd ; D = row_start (PEI dp base)
clc
adc #159
tcs ; SP = row_start + 159
; Enable shadow+AUXWRITE+RAMRD for the PEI burst ONLY.
; gPeiOrigShadow is read here while RAMRD is still off, so
; it comes from bank 0 correctly.
sep #0x20
.a8
lda gPeiOrigShadow
and #0xF7 ; clear bit 3 -> SHR shadow ON. SHR
; shadowing mirrors the $2000-9FFF
; region for BOTH banks ($00->$E0 and
; $01->$E1); the burst pushes into
; aux $01 (RAMWRT) so it lands at $E1.
sta 0xE0C035
lda #0
sta 0xE0C005 ; RAMWRT on: $00:0200-BFFF writes -> $01
sta 0xE0C003 ; RAMRD on: PEI reads aux $01 stage
rep #0x20
.a16
; 80 PEIs from DP+$9E down to DP+$00.
pei 0x9E
pei 0x9C
pei 0x9A
pei 0x98
pei 0x96
pei 0x94
pei 0x92
pei 0x90
pei 0x8E
pei 0x8C
pei 0x8A
pei 0x88
pei 0x86
pei 0x84
pei 0x82
pei 0x80
pei 0x7E
pei 0x7C
pei 0x7A
pei 0x78
pei 0x76
pei 0x74
pei 0x72
pei 0x70
pei 0x6E
pei 0x6C
pei 0x6A
pei 0x68
pei 0x66
pei 0x64
pei 0x62
pei 0x60
pei 0x5E
pei 0x5C
pei 0x5A
pei 0x58
pei 0x56
pei 0x54
pei 0x52
pei 0x50
pei 0x4E
pei 0x4C
pei 0x4A
pei 0x48
pei 0x46
pei 0x44
pei 0x42
pei 0x40
pei 0x3E
pei 0x3C
pei 0x3A
pei 0x38
pei 0x36
pei 0x34
pei 0x32
pei 0x30
pei 0x2E
pei 0x2C
pei 0x2A
pei 0x28
pei 0x26
pei 0x24
pei 0x22
pei 0x20
pei 0x1E
pei 0x1C
pei 0x1A
pei 0x18
pei 0x16
pei 0x14
pei 0x12
pei 0x10
pei 0x0E
pei 0x0C
pei 0x0A
pei 0x08
pei 0x06
pei 0x04
pei 0x02
pei 0x00
; Restore: AUXWRITE/RAMRD off FIRST, then read gPeiOrigShadow
; from bank 0 to restore the shadow register. After this the
; bank-0 control reads below (and at peiChunkEnd) hit main.
sep #0x20
.a8
lda #0
sta 0xE0C004 ; RAMWRT off
sta 0xE0C002 ; RAMRD off
lda gPeiOrigShadow
sta 0xE0C035 ; shadow restored
rep #0x20
.a16
lda gPeiCurRow
tax
inx
lda gPeiChunkRow
tay
iny
brl peiRowLoop
; Narrow dirty span: exact-extent MVN of words [minWord..maxWord] from
; bank-$01 back buffer to $E1 SHR frame buffer. src $01 -> dst $E1;
; GAS spells MVN dst,src so `mvn 0xE1, 0x01`.
peiMvnRow:
txa
sta gPeiCurRow
tya
sta gPeiChunkRow
; rowStart = $2000 + y*160
lda gPeiCurRow
asl a
tax
lda gRowOffsetLut,x
clc
adc #0x2000
sta gPeiMvnSrc
; src/dst offset = rowStart + minWord*2
lda gPeiCurRow
tax
sep #0x20
.a8
lda gStageMinWord,x
rep #0x20
.a16
and #0x00FF
asl a
clc
adc gPeiMvnSrc
sta gPeiMvnSrc
; count = (maxWord - minWord)*2 + 1
sep #0x20
.a8
lda gStageMaxWord,x
sec
sbc gStageMinWord,x
rep #0x20
.a16
and #0x00FF
asl a
inc a
sta gPeiMvnCount
; Block move: X=src off, Y=dst off (same), A=count-1; src $01, dst $E1.
lda gPeiMvnSrc
tax
tay
lda gPeiMvnCount
mvn 0xE1, 0x01
; MVN left DBR=$E1; restore DBR := globals bank (caller's,
; stashed at $E1:9DFE) before the absolute gPei*/dirty reads
; below and on the next pass. A is dead (reloaded just below).
sep #0x20
.a8
lda 0xE19DFE
pha
plb
rep #0x20
.a16
; Restore row counters and advance.
lda gPeiCurRow
tax
inx
lda gPeiChunkRow
tay
iny
brl peiRowLoop
peiChunkEnd:
; Per-chunk teardown: restore SP, shadow, AUXWRITE/RAMRD, then CLI.
lda gPeiOrigSp
tcs
sep #0x20
.a8
lda gPeiOrigShadow
sta 0xE0C035
lda #0
sta 0xE0C004 ; RAMWRT off
sta 0xE0C002 ; RAMRD off
rep #0x20
.a16
cli
cpx #200
bcs peiAllDone
brl peiChunkBegin
peiAllDone:
.a8
.i8
pld
plb
plp ; restores I (pre-SEI value)
rtl
; ====================================================================
; iigsMarkDirtyRowsInner(uint16_t yStart, uint16_t yEnd,
; uint16_t minWord, uint16_t maxWord)
;
; For each row in [yStart, yEnd) widen gStageMinWord[row] DOWN to
; minWord (if smaller) and gStageMaxWord[row] UP to maxWord (if larger).
; Caller clips yStart/yEnd to [0,200); no internal bounds check.
;
; Original args (D-frame, D=SP+8, all on stack):
; yStart D+0..1, yEnd D+2..3, minWord D+4..5, maxWord D+6..7
;
; ABI: ALL FOUR ARGS ARE WORDS. arg0 yStart arrives in A; arg1 yEnd,
; arg2 minWord, arg3 maxWord are on the stack. JSL entry: yEnd @4,s,
; minWord @6,s, maxWord @8,s; after php+phb+phd (+4): yEnd @8,s.
; The original D-frame had yStart at D+0; under clang yStart is in A
; and is only ever loaded ONCE into X (the loop never re-reads it), so
; we just TAX it -- no D+0 slot is needed. The three stack args slid
; DOWN by 2 (one word), so set D = SP+6 (adc #6 not #8): then
; yEnd@8,s = D+2, minWord = D+4, maxWord = D+6 (original equates hold).
; D+0..1 would overlap the return address, but mdrYStart is never
; touched in the frame, so that is harmless.
; ====================================================================
.section .text.iigsMarkDirtyRowsInner,"ax"
.globl iigsMarkDirtyRowsInner
iigsMarkDirtyRowsInner:
mdrYEnd = 2
mdrMinWord = 4
mdrMaxWord = 6
php
phb
phd
rep #0x30
.a16
.i16
tax ; X = yStart (arg0 in A); never re-read
tsc
clc
adc #6
tcd ; D = SP+6 (yEnd@D+2, min@D+4, max@D+6)
cpx mdrYEnd
bcs mdrExit ; empty range -> nothing to do
sep #0x20
.a8
mdrLoop:
lda mdrMinWord
cmp gStageMinWord,x
bcs mdrNoMin ; A >= existing -> already <= minWord
sta gStageMinWord,x ; widen down
mdrNoMin:
lda mdrMaxWord
cmp gStageMaxWord,x
bcc mdrNoMax ; A < existing -> already >= maxWord
beq mdrNoMax
sta gStageMaxWord,x ; widen up
mdrNoMax:
inx
cpx mdrYEnd
bcc mdrLoop
rep #0x20
.a16
mdrExit:
.a8
.i8
pld
plb
plp
rtl
; ====================================================================
; iigsPollJoystickInner(void)
;
; Reads the IIgs paddle ports (PDL0/PDL1) at 1 MHz. No args; returns the
; packed result in registers (see ABI below) -- no output globals.
;
; ABI: no args. Returns a uint32_t in A:X (A=low16, X=high16):
; A = (JoyX << 8) | resolved , X = JoyY
; i.e. resolved = ret & 0xFF (3 = stick present, 0 = none),
; px = (ret >> 8) & 0xFF (JoyX 0..255),
; py = (ret >> 16) & 0xFF (JoyY 0..255).
; Register return is used instead of output globals because an asm WRITE
; of a global does not reach the C-read address in this toolchain.
;
; Faithful port of John Brooks' 1 MHz single-pass GetJoyXY. KEY trick:
; DP is set to $C000 so the paddle/speed I/O is read DP-relative
; (`lda $64` -> $00C064). DP is ALWAYS bank 0, so this is DBR-INDEPENDENT
; -- no DBR forcing, so DBR stays at the caller's bank ($08 under GS/OS)
; and the gJoy* globals (absolute, DBR-relative) are written in the right
; bank with no bank juggling. The dual/solo counter reads JoyX ($C064) and
; JoyY ($C065) together (`lda $64 / and $65`) every ~11 cyc; whichever
; axis trips first, it drops to a solo loop for the other. A settle-wait
; (with timeout) up front avoids a mid-discharge trigger and avoids
; hanging when no stick is attached.
; ====================================================================
.section .text.iigsPollJoystickInner,"ax"
.globl iigsPollJoystickInner
iigsPollJoystickInner:
php
sep #0x34 ; sei + 8-bit M/X
.a8
.i8
phd
pea 0xC000
pld ; DP = $C000 (I/O via DP, bank 0)
; Wait for both paddles to settle (bit7=0), with a timeout so a missing
; stick can't hang the poll.
ldy #0
gjWaitBad:
lda 0x64 ; $C064 (JoyX)
ora 0x65 ; | $C065 (JoyY)
bpl gjSettled ; both bit7=0 -> settled
iny
bne gjWaitBad ; Y<256 -> keep waiting
brl gjNoJoy ; settle timeout -> no stick
gjSettled:
bit 0x70 ; PTRIG: start X,Y timers
lsr 0x36 ; force 1 MHz (clear bit 7)
ldx #1 ; dual X,Y counter init
xba
xba
gjDualXY0:
lda 0x64
and 0x65
bpl gjToSolo
inx
gjDualXY1:
lda 0x64
and 0x65
bpl gjToSolo
inx
gjDualXY2:
lda 0x64
and 0x65
bpl gjToSolo
inx
bne gjDualXY0
gjDualXY3:
lda 0x64
and 0x65
bmi gjSameXY
dex
gjSameXY:
dex
txy
bra gjGotXY
gjSoloX:
bit 0x64
bmi gjSoloXOk
dey
bra gjGotXY
gjSoloXOk:
inx
bne gjSoloX
gjSoloY:
bit 0x65
bmi gjSoloYOk
dex
bra gjGotXY
gjToSolo:
txy
bit 0x64
bmi gjSoloXOk
bit 0x65
bpl gjSameXY
gjSoloYOk:
iny
bne gjSoloY
dey
gjGotXY:
rol 0x36 ; restore CPU speed (8-bit mode)
; X = JoyX (px), Y = JoyY (py), both 0..255. Hand the result back in
; REGISTERS as the C ABI's uint32_t (A = low16, X = high16) -- NOT via
; globals: an asm WRITE of a global does not reach the C-read address in
; this toolchain (DBR/bank mismatch), whereas register return is proven
; (same path as iigsGetTickWord / iigsReadHzParam). Packing:
; A (low16) = (px << 8) | resolved X (high16) = py
; -> C sees: resolved = r&0xFF, px = (r>>8)&0xFF, py = (r>>16)&0xFF.
pld ; restore caller DP
plp ; back to caller M=I=16; X=px, Y=py
.a16
.i16
phx ; save px (16-bit)
tya
tax ; X = py -> high word of result
pla ; A = px
xba ; A = px << 8 (XBA is 16-bit reg-wide)
ora #0x0003 ; A = (px << 8) | resolved(=3)
rtl ; uint32_t in A:X
gjNoJoy:
; Settle timed out (speed unchanged, so no `rol`). Report no stick
; (resolved=0) with centred axes (px=py=128).
pld ; restore caller DP
plp ; back to caller M=I=16
.a16
.i16
ldx #128 ; py = 128 (high word)
lda #0x8000 ; (px=128 << 8) | resolved(=0)
rtl ; uint32_t in A:X
; ====================================================================
; iigsInputSnapshot(void)
;
; Snapshot gKeyState->gKeyPrev (KEY_COUNT=60 bytes) plus
; gMouseButtonState/gJoyButtonState (4 bytes each). No args.
;
; ABI: void; no register/stack args. DBR stays 0; all globals via
; 16-bit abs.
; ====================================================================
.section .text.iigsInputSnapshot,"ax"
.globl iigsInputSnapshot
iigsInputSnapshot:
php
rep #0x30
.a16
.i16
sep #0x20
.a8
; Snapshot gKeyState -> gKeyPrev (60 bytes), long-mode loop.
ldx #59
isnKeyLoop:
lda gKeyState,x
sta gKeyPrev,x
dex
bpl isnKeyLoop
; Snapshot gMouseButtonState -> gMouseButtonPrev (4 bytes inline).
lda gMouseButtonState
sta gMouseButtonPrev
lda gMouseButtonState+1
sta gMouseButtonPrev+1
lda gMouseButtonState+2
sta gMouseButtonPrev+2
lda gMouseButtonState+3
sta gMouseButtonPrev+3
; Snapshot gJoyButtonState -> gJoyButtonPrev (4 bytes inline).
lda gJoyButtonState
sta gJoyButtonPrev
lda gJoyButtonState+1
sta gJoyButtonPrev+1
lda gJoyButtonState+2
sta gJoyButtonPrev+2
lda gJoyButtonState+3
sta gJoyButtonPrev+3
rep #0x30
.a16
.i16
plp
rtl
; ====================================================================
; iigsFloodWalkAndScansInner(uint8_t *pixels, uint16_t x, uint16_t y,
; uint16_t matchColor, uint16_t newColor,
; uint16_t matchEqual,
; int16_t *stackX, int16_t *stackY,
; uint16_t *spInOut, uint16_t maxSp)
;
; Combined per-popped-seed flood work: seed test + walk-left + walk-
; right + span fill + scan-above + scan-below (each scan pushes new
; seeds onto the explicit stackX/stackY scan stack via *spInOut/maxSp).
; Outputs gFloodSeedMatch / gFloodLeftX / gFloodRightX.
;
; Original args (D-frame, D=SP+8, all on stack):
; pixels D+0..3, x D+4..5, y D+6..7, matchColor D+8..9,
; newColor D+10..11, matchEqual D+12..13, stackX D+14..17,
; stackY D+18..21, spInOut D+22..25, maxSp D+26..27
;
; ABI: pixels (arg0, ptr) arrives in A:X and is NOT on the stack; the
; other 9 args are on the stack. The body needs ALL of D+0..D+27 to be
; writable DP scratch -- it overwrites the pixels slot (wsRow) with the
; computed row pointer and uses `[wsRow],y`, `[wsScanRow],y`,
; `[wsStackX],y`, `[wsStackY],y`, `[wsSpInOut],y` DP-indirect-long.
; The clang stack offers no such writable 28-byte window above the
; saved regs, so we point D at a private bank-0 scratch frame (wsFrame)
; and copy EVERY arg into it at the ORIGINAL D-offsets, leaving all the
; ws* equates and indirect loads unchanged.
;
; Stack source offsets (after php+phb+phd, +4):
; x@8,s y@10,s matchColor@12,s newColor@14,s matchEqual@16,s
; stackX@18,s(4) stackY@22,s(4) spInOut@26,s(4) maxSp@30,s
; Copied to: wsFrame+4 / +6 / +8 / +10 / +12 / +14..17 / +18..21 /
; +22..25 / +26 ; pixels(A:X) -> wsFrame+0..3. PLD restores caller D.
; ====================================================================
.section .text.iigsFloodWalkAndScansInner,"ax"
.globl iigsFloodWalkAndScansInner
iigsFloodWalkAndScansInner:
; wsRow occupies D+0..3 (was pixels arg): row = pixels + y*160 is stored
; over the pixels slot so `lda [wsRow],y` reads the row directly.
; wsScanRow overlaps wsMatch+wsNew (D+8..11): those args are cached to
; wsMatchByte/wsNewByte first, then the slot is reused.
wsRow = 0 ; D+0..3 (was wsPixels arg)
wsX = 4
wsY = 6
wsScanRow = 8 ; D+8..11 (was wsMatch+wsNew args)
wsMatch = 8 ; alias of wsScanRow during initial cache
wsNew = 10 ; alias of wsScanRow+2 during initial cache
wsEq = 12
wsStackX = 14
wsStackY = 18
wsSpInOut = 22
wsMaxSp = 26
wsPixels = 0 ; alias of wsRow for the initial row-addr compute
wsMidEnd = 4 ; alias of wsX, used by inline fill (post-walk)
; DP scratch for the rewritten scan inner loop (DP-relative reads).
wsScanCurX = 4 ; alias wsX/wsMidEnd, 16-bit
wsScanByte = 12 ; alias wsEq.lo, 8-bit
wsScanPrevHit = 26 ; alias wsMaxSp.lo, 8-bit
wsScanCurHit = 27 ; alias wsMaxSp.hi, 8-bit
php
phb
phd
rep #0x30
.a16
.i16
; Rebuild the full D-frame in bank-0 scratch.
; pixels in A:X; the 9 stack args at +8/+10/.../+30,s.
sta wsFrame+0 ; pixels.lo (A)
stx wsFrame+2 ; pixels.hi (X)
lda 8,s ; x
sta wsFrame+4
lda 10,s ; y
sta wsFrame+6
lda 12,s ; matchColor
sta wsFrame+8
lda 14,s ; newColor
sta wsFrame+10
lda 16,s ; matchEqual
sta wsFrame+12
lda 18,s ; stackX.lo
sta wsFrame+14
lda 20,s ; stackX.hi
sta wsFrame+16
lda 22,s ; stackY.lo
sta wsFrame+18
lda 24,s ; stackY.hi
sta wsFrame+20
lda 26,s ; spInOut.lo
sta wsFrame+22
lda 28,s ; spInOut.hi
sta wsFrame+24
lda 30,s ; maxSp
sta wsFrame+26
; Copy the staged 32-byte frame into a STACK (bank-0) D-frame:
; wsFrame is a BSS global in the LOAD bank, but D (the direct
; page) is ALWAYS bank 0, so `tcd #wsFrame` aimed every
; D-relative read and [wsRow],y / [wsSpInOut],y indirect at
; bank-0:offset while the stores above wrote loadbank:offset.
; The hardware stack is bank 0, so a stack D-frame matches.
lda wsFrame+30
pha
lda wsFrame+28
pha
lda wsFrame+26
pha
lda wsFrame+24
pha
lda wsFrame+22
pha
lda wsFrame+20
pha
lda wsFrame+18
pha
lda wsFrame+16
pha
lda wsFrame+14
pha
lda wsFrame+12
pha
lda wsFrame+10
pha
lda wsFrame+8
pha
lda wsFrame+6
pha
lda wsFrame+4
pha
lda wsFrame+2
pha
lda wsFrame+0
pha
tsc
inc a ; A = SP+1
tcd ; D -> stacked frame (bank 0)
; Cache 8-bit constants used across walk + scan + fill.
sep #0x20
.a8
lda wsMatch
and #0x0F
sta wsMatchByte
lda wsNew
and #0x0F
sta wsNewByte
ora #0
pha ; save newByte for doubled compute
asl a
asl a
asl a
asl a ; A = newByte << 4
sta wsNewHigh
ora 1,s ; A = (newByte<<4) | newByte
sta wsDoubledByte
pla ; clean stack
lda wsEq
sta wsEqByte
rep #0x20
.a16
; Compute rowAddr (long ptr) = pixels + lut[y*2]; store over wsRow.
lda wsY
asl a ; A = y*2 (LUT byte offset)
tax
lda gRowOffsetLut,x ; A = y*160
clc
adc wsPixels ; A = pixels.lo + y*160 = row.lo
tax ; X = row.lo (carry preserved)
lda wsPixels+2 ; A = pixels.hi
adc #0 ; A = row.hi (carry from prev add)
sta wsRow+2 ; D+2..3 = row.hi
txa
sta wsRow ; D+0..1 = row.lo
; === SEED TEST + WALK LEFT + WALK RIGHT ===
sep #0x20
.a8
lda wsEqByte
rep #0x20
.a16
bne wsWalkEqEntry
brl wsWalkBndEntry
; ***** EQUAL MODE WALK *****
wsWalkEqEntry:
; --- SEED TEST EQ (inline) ---
lda wsX
lsr a
tay
sep #0x20
.a8
bcs wsSeedEqOdd
lda [wsRow],y
lsr a
lsr a
lsr a
lsr a
bra wsSeedEqHave
wsSeedEqOdd:
lda [wsRow],y
and #0x0F
wsSeedEqHave:
cmp wsMatchByte
rep #0x20
.a16
bne wsSeedEqMiss
lda #1
sta gFloodSeedMatch
bra wsWalkEqSeedOk
wsSeedEqMiss:
lda #0
sta gFloodSeedMatch
brl wsExit
wsWalkEqSeedOk:
; --- WALK LEFT EQ (byte-cached) ---
lda wsX
sta wsScanCurX
lsr a
tay ; Y = byteIdx
bcc wsLEqEvenEntry
sep #0x20
.a8
lda [wsRow],y
sta wsScanByte
rep #0x20
.a16
bra wsLEqOddEntry
wsLEqEvenEntry:
lda wsScanCurX
beq wsLeftEqDone
dey
sep #0x20
.a8
lda [wsRow],y
sta wsScanByte
and #0x0F
cmp wsMatchByte
rep #0x20
.a16
bne wsLeftEqDone
dec wsScanCurX
wsLEqOddEntry:
lda wsScanCurX
beq wsLeftEqDone
sep #0x20
.a8
lda wsScanByte
lsr a
lsr a
lsr a
lsr a
cmp wsMatchByte
rep #0x20
.a16
bne wsLeftEqDone
dec wsScanCurX
bra wsLEqEvenEntry
wsLeftEqDone:
lda wsScanCurX
sta gFloodLeftX
; --- WALK RIGHT EQ (byte-cached) ---
lda wsX
sta wsScanCurX
lsr a
tay
bcs wsREqOddEntry
sep #0x20
.a8
lda [wsRow],y
sta wsScanByte
rep #0x20
.a16
bra wsREqEvenEntry
wsREqEvenEntry:
lda wsScanCurX
cmp #319
bcs wsRightEqDone
sep #0x20
.a8
lda wsScanByte
and #0x0F
cmp wsMatchByte
rep #0x20
.a16
bne wsRightEqDone
inc wsScanCurX
wsREqOddEntry:
lda wsScanCurX
cmp #319
bcs wsRightEqDone
iny
sep #0x20
.a8
lda [wsRow],y
sta wsScanByte
lsr a
lsr a
lsr a
lsr a
cmp wsMatchByte
rep #0x20
.a16
bne wsRightEqDone
inc wsScanCurX
bra wsREqEvenEntry
wsRightEqDone:
lda wsScanCurX
sta gFloodRightX
brl wsAfterWalk
; ***** BOUNDARY MODE WALK *****
wsWalkBndEntry:
; --- SEED TEST BND (inline) ---
lda wsX
lsr a
tay
sep #0x20
.a8
bcs wsSeedBndOdd
lda [wsRow],y
lsr a
lsr a
lsr a
lsr a
bra wsSeedBndHave
wsSeedBndOdd:
lda [wsRow],y
and #0x0F
wsSeedBndHave:
cmp wsMatchByte
beq wsSeedBndMiss
cmp wsNewByte
beq wsSeedBndMiss
rep #0x20
.a16
lda #1
sta gFloodSeedMatch
bra wsWalkBndSeedOk
wsSeedBndMiss:
rep #0x20
.a16
lda #0
sta gFloodSeedMatch
brl wsExit
wsWalkBndSeedOk:
; --- WALK LEFT BND (byte-cached) ---
lda wsX
sta wsScanCurX
lsr a
tay
bcc wsLBndEvenEntry
sep #0x20
.a8
lda [wsRow],y
sta wsScanByte
rep #0x20
.a16
bra wsLBndOddEntry
wsLBndEvenEntry:
lda wsScanCurX
beq wsLeftBndDone
dey
sep #0x20
.a8
lda [wsRow],y
sta wsScanByte
and #0x0F
cmp wsMatchByte
beq wsLBndStop
cmp wsNewByte
beq wsLBndStop
rep #0x20
.a16
dec wsScanCurX
bra wsLBndOddEntry2
wsLBndStop:
rep #0x20
.a16
bra wsLeftBndDone
wsLBndOddEntry2:
wsLBndOddEntry:
lda wsScanCurX
beq wsLeftBndDone
sep #0x20
.a8
lda wsScanByte
lsr a
lsr a
lsr a
lsr a
cmp wsMatchByte
beq wsLBndStop2
cmp wsNewByte
beq wsLBndStop2
rep #0x20
.a16
dec wsScanCurX
bra wsLBndEvenEntry
wsLBndStop2:
rep #0x20
.a16
wsLeftBndDone:
lda wsScanCurX
sta gFloodLeftX
; --- WALK RIGHT BND (byte-cached) ---
lda wsX
sta wsScanCurX
lsr a
tay
bcs wsRBndOddEntry
sep #0x20
.a8
lda [wsRow],y
sta wsScanByte
rep #0x20
.a16
bra wsRBndEvenEntry
wsRBndEvenEntry:
lda wsScanCurX
cmp #319
bcs wsRightBndDone
sep #0x20
.a8
lda wsScanByte
and #0x0F
cmp wsMatchByte
beq wsRBndStop
cmp wsNewByte
beq wsRBndStop
rep #0x20
.a16
inc wsScanCurX
bra wsRBndOddEntry2
wsRBndStop:
rep #0x20
.a16
bra wsRightBndDone
wsRBndOddEntry2:
wsRBndOddEntry:
lda wsScanCurX
cmp #319
bcs wsRightBndDone
iny
sep #0x20
.a8
lda [wsRow],y
sta wsScanByte
lsr a
lsr a
lsr a
lsr a
cmp wsMatchByte
beq wsRBndStop2
cmp wsNewByte
beq wsRBndStop2
rep #0x20
.a16
inc wsScanCurX
bra wsRBndEvenEntry
wsRBndStop2:
rep #0x20
.a16
wsRightBndDone:
lda wsScanCurX
sta gFloodRightX
wsAfterWalk:
; Cache leftX/rightX/spanLen for the scans.
lda gFloodLeftX
sta wsLeftX
lda gFloodRightX
sta wsRightX
sec
sbc wsLeftX
inc a
sta wsSpanLen
; === FILL THE SPAN ===
lda wsLeftX
lsr a ; A = leadingByte; carry = leftX & 1
tay ; Y = leadingByte
bcc wsFillNoLead
; Leading partial byte: low nibble of byte[Y] = newNibble.
sep #0x20
.a8
lda [wsRow],y
and #0xF0
ora wsNewByte
sta [wsRow],y
rep #0x20
.a16
iny ; midStart = leadingByte + 1
wsFillNoLead:
; midEnd = (rightX + 1) >> 1; trailing partial only if rightX even.
lda wsRightX
inc a ; A = pxEnd = rightX + 1
lsr a ; A = midEnd; carry = pxEnd & 1
sta wsMidEnd ; DP store (so DP-direct CPY works)
bcs wsFillSetTrail
lda #0
sta wsHasTrail
bra wsFillMidLoop
wsFillSetTrail:
lda #1
sta wsHasTrail
wsFillMidLoop:
sep #0x20
.a8
lda wsDoubledByte
wsFillMidIter:
cpy wsMidEnd ; DP-direct CPY (X=16-bit reads 2 bytes)
bcs wsFillMidDone
sta [wsRow],y
iny
bra wsFillMidIter
wsFillMidDone:
rep #0x20
.a16
lda wsHasTrail
beq wsFillDone
; Trailing partial byte: high nibble of byte[wsMidEnd] = newNibble.
ldy wsMidEnd ; Y = trailing byte index
sep #0x20
.a8
lda [wsRow],y
and #0x0F
ora wsNewHigh
sta [wsRow],y
rep #0x20
.a16
wsFillDone:
; Load sp from *spInOut, cache maxSp.
ldy #0
lda [wsSpInOut],y
sta wsSp
lda wsMaxSp
sta wsMaxSpCache
; === SCAN ABOVE (if y > 0) ===
lda wsY
bne wsHasAbove
brl wsSkipAbove
wsHasAbove:
lda wsRow
sec
sbc #160
sta wsScanRow
lda wsRow+2
sbc #0
sta wsScanRow+2
lda wsY
dec a
sta wsScanY
jsr wsScanAndPush
wsSkipAbove:
; === SCAN BELOW (if y < 199) ===
lda wsY
cmp #199
bcc wsHasBelow
brl wsSkipBelow
wsHasBelow:
lda wsRow
clc
adc #160
sta wsScanRow
lda wsRow+2
adc #0
sta wsScanRow+2
lda wsY
inc a
sta wsScanY
jsr wsScanAndPush
wsSkipBelow:
; Store updated sp back to *spInOut.
ldy #0
lda wsSp
sta [wsSpInOut],y
wsExit:
rep #0x30 ; M=16/X=16 for the SP arithmetic
.a16
.i16
tsc
clc
adc #32 ; drop the 32-byte stack D-frame
tcs
pld
plb
plp
rtl
; wsScanAndPush: walk wsScanRow[wsLeftX..wsRightX] for run-edge
; transitions. Pushes (curX-1, wsScanY) on falling edges and
; (rightX, wsScanY) at end of trailing run. 2 pixels per byte read.
wsScanAndPush:
sep #0x20
.a8
lda #0
sta wsScanPrevHit
lda wsEqByte
rep #0x20
.a16
bne wsEqEntry2
brl wsBoundEntry2
; ***** EQUAL MODE *****
wsEqEntry2:
lda wsLeftX
sta wsScanCurX ; DP write 16-bit
lsr a ; A = byteIdx; carry = leftX & 1
tay ; Y = byteIdx
bcc wsEqAfterLead2
; LEADING: low nibble of byte at Y (curX = leftX, odd).
sep #0x20
.a8
lda [wsScanRow],y
and #0x0F
cmp wsMatchByte
beq wsEqLeadHit2
lda #0
bra wsEqLeadStore2
wsEqLeadHit2:
lda #1
wsEqLeadStore2:
sta wsScanCurHit
bne wsEqLeadNoFall2
lda wsScanPrevHit
beq wsEqLeadNoFall2
rep #0x20
.a16
lda wsScanCurX
dec a
sta wsPushX
jsr wsPushXY
sep #0x20
.a8
wsEqLeadNoFall2:
lda wsScanCurHit
sta wsScanPrevHit
rep #0x20
.a16
inc wsScanCurX
iny
wsEqAfterLead2:
; Pair loop: while curX+1 <= rightX, process high+low of one byte.
wsEqPairLoop2:
lda wsScanCurX
inc a
cmp wsRightX
bcc wsEqDoPair2
beq wsEqDoPair2
brl wsEqTrailing2
wsEqDoPair2:
sep #0x20
.a8
lda [wsScanRow],y
sta wsScanByte
lsr a
lsr a
lsr a
lsr a ; A = high nibble (pixel at curX)
cmp wsMatchByte
beq wsEqHi2Hit
lda #0
bra wsEqHi2Store
wsEqHi2Hit:
lda #1
wsEqHi2Store:
sta wsScanCurHit
bne wsEqHi2NoFall
lda wsScanPrevHit
beq wsEqHi2NoFall
rep #0x20
.a16
lda wsScanCurX
dec a
sta wsPushX
jsr wsPushXY
sep #0x20
.a8
wsEqHi2NoFall:
lda wsScanCurHit
sta wsScanPrevHit
rep #0x20
.a16
inc wsScanCurX
sep #0x20
.a8
lda wsScanByte
and #0x0F ; A = low nibble (pixel at curX after inc)
cmp wsMatchByte
beq wsEqLo2Hit
lda #0
bra wsEqLo2Store
wsEqLo2Hit:
lda #1
wsEqLo2Store:
sta wsScanCurHit
bne wsEqLo2NoFall
lda wsScanPrevHit
beq wsEqLo2NoFall
rep #0x20
.a16
lda wsScanCurX
dec a
sta wsPushX
jsr wsPushXY
sep #0x20
.a8
wsEqLo2NoFall:
lda wsScanCurHit
sta wsScanPrevHit
rep #0x20
.a16
inc wsScanCurX
iny
brl wsEqPairLoop2
wsEqTrailing2:
lda wsScanCurX
cmp wsRightX
bcc wsEqDoTrail2
beq wsEqDoTrail2
brl wsScanDone
wsEqDoTrail2:
sep #0x20
.a8
lda [wsScanRow],y
lsr a
lsr a
lsr a
lsr a ; A = high nibble
cmp wsMatchByte
beq wsEqTr2Hit
lda #0
bra wsEqTr2Store
wsEqTr2Hit:
lda #1
wsEqTr2Store:
sta wsScanCurHit
bne wsEqTr2NoFall
lda wsScanPrevHit
beq wsEqTr2NoFall
rep #0x20
.a16
lda wsScanCurX
dec a
sta wsPushX
jsr wsPushXY
sep #0x20
.a8
wsEqTr2NoFall:
lda wsScanCurHit
sta wsScanPrevHit
rep #0x20
.a16
brl wsScanDone
; ***** BOUNDARY MODE *****
wsBoundEntry2:
lda wsLeftX
sta wsScanCurX
lsr a
tay
bcc wsBndAfterLead2
; LEADING: low nibble of byte at Y, boundary mode.
sep #0x20
.a8
lda [wsScanRow],y
and #0x0F
cmp wsMatchByte
beq wsBndLeadMiss2
cmp wsNewByte
beq wsBndLeadMiss2
lda #1
bra wsBndLeadStore2
wsBndLeadMiss2:
lda #0
wsBndLeadStore2:
sta wsScanCurHit
bne wsBndLeadNoFall2
lda wsScanPrevHit
beq wsBndLeadNoFall2
rep #0x20
.a16
lda wsScanCurX
dec a
sta wsPushX
jsr wsPushXY
sep #0x20
.a8
wsBndLeadNoFall2:
lda wsScanCurHit
sta wsScanPrevHit
rep #0x20
.a16
inc wsScanCurX
iny
wsBndAfterLead2:
wsBndPairLoop2:
lda wsScanCurX
inc a
cmp wsRightX
bcc wsBndDoPair2
beq wsBndDoPair2
brl wsBndTrailing2
wsBndDoPair2:
sep #0x20
.a8
lda [wsScanRow],y
sta wsScanByte
lsr a
lsr a
lsr a
lsr a ; A = high nibble
cmp wsMatchByte
beq wsBndHi2Miss
cmp wsNewByte
beq wsBndHi2Miss
lda #1
bra wsBndHi2Store
wsBndHi2Miss:
lda #0
wsBndHi2Store:
sta wsScanCurHit
bne wsBndHi2NoFall
lda wsScanPrevHit
beq wsBndHi2NoFall
rep #0x20
.a16
lda wsScanCurX
dec a
sta wsPushX
jsr wsPushXY
sep #0x20
.a8
wsBndHi2NoFall:
lda wsScanCurHit
sta wsScanPrevHit
rep #0x20
.a16
inc wsScanCurX
sep #0x20
.a8
lda wsScanByte
and #0x0F ; A = low nibble
cmp wsMatchByte
beq wsBndLo2Miss
cmp wsNewByte
beq wsBndLo2Miss
lda #1
bra wsBndLo2Store
wsBndLo2Miss:
lda #0
wsBndLo2Store:
sta wsScanCurHit
bne wsBndLo2NoFall
lda wsScanPrevHit
beq wsBndLo2NoFall
rep #0x20
.a16
lda wsScanCurX
dec a
sta wsPushX
jsr wsPushXY
sep #0x20
.a8
wsBndLo2NoFall:
lda wsScanCurHit
sta wsScanPrevHit
rep #0x20
.a16
inc wsScanCurX
iny
brl wsBndPairLoop2
wsBndTrailing2:
lda wsScanCurX
cmp wsRightX
bcc wsBndDoTrail2
beq wsBndDoTrail2
brl wsScanDone
wsBndDoTrail2:
sep #0x20
.a8
lda [wsScanRow],y
lsr a
lsr a
lsr a
lsr a ; A = high nibble
cmp wsMatchByte
beq wsBndTr2Miss
cmp wsNewByte
beq wsBndTr2Miss
lda #1
bra wsBndTr2Store
wsBndTr2Miss:
lda #0
wsBndTr2Store:
sta wsScanCurHit
bne wsBndTr2NoFall
lda wsScanPrevHit
beq wsBndTr2NoFall
rep #0x20
.a16
lda wsScanCurX
dec a
sta wsPushX
jsr wsPushXY
sep #0x20
.a8
wsBndTr2NoFall:
lda wsScanCurHit
sta wsScanPrevHit
rep #0x20
.a16
brl wsScanDone
wsScanDone:
; Trailing run: if prevHit, push (rightX, scanY).
sep #0x20
.a8
lda wsScanPrevHit
rep #0x20
.a16
beq wsScanReturn
lda wsRightX
sta wsPushX
jsr wsPushXY
wsScanReturn:
rts
; wsPushXY: push (wsPushX, wsScanY) at sp if sp<maxSp. Increments sp.
; M=16 X=16 on entry/exit; preserves caller A/Y via PHA/PHY.
wsPushXY:
pha
phy
lda wsSp
cmp wsMaxSpCache
bcs wsPushSkip
asl a
tay
lda wsPushX
sta [wsStackX],y
lda wsScanY
sta [wsStackY],y
lda wsSp
inc a
sta wsSp
wsPushSkip:
ply
pla
rts
; ====================================================================
; Data symbols defined by these functions (DRAWPRIMS scratch globals).
; gRowOffsetLut / gFloodSeedMatch / gFloodLeftX / gFloodRightX,
; gStageMinWord / gStageMaxWord, gPei* and gKey/gMouse/gJoy*Button*
; are defined ELSEWHERE (asm or C); they are left as externs.
; ====================================================================
; Private bank-0 D-frame for iigsBlitStageToShr (scbPtr/palettePtr/flags).
.section .bss.bsFrame,"aw"
.globl bsFrame
bsFrame:
.zero 10
; Private bank-0 D-frame for iigsFloodWalkAndScansInner (28 args bytes;
; pad to a clean 32).
.section .bss.wsFrame,"aw"
.globl wsFrame
wsFrame:
.zero 32
; iigsPollJoystickInner has NO output globals: it returns its result in
; registers (uint32_t in A:X). An asm WRITE of a global does not reach the
; C-read address in this toolchain (asm READING a C-defined global like
; gStageMinWord is fine; asm WRITING one that C reads is not), so the
; register-return ABI is used instead.
; iigsFloodWalkAndScansInner scratch.
.section .bss.wsLeftX,"aw"
.globl wsLeftX
wsLeftX:
.zero 2
.section .bss.wsRightX,"aw"
.globl wsRightX
wsRightX:
.zero 2
.section .bss.wsScanY,"aw"
.globl wsScanY
wsScanY:
.zero 2
.section .bss.wsMatchByte,"aw"
.globl wsMatchByte
wsMatchByte:
.zero 2
.section .bss.wsNewByte,"aw"
.globl wsNewByte
wsNewByte:
.zero 2
.section .bss.wsEqByte,"aw"
.globl wsEqByte
wsEqByte:
.zero 2
.section .bss.wsSp,"aw"
.globl wsSp
wsSp:
.zero 2
.section .bss.wsMaxSpCache,"aw"
.globl wsMaxSpCache
wsMaxSpCache:
.zero 2
.section .bss.wsSpanLen,"aw"
.globl wsSpanLen
wsSpanLen:
.zero 2
.section .bss.wsPushX,"aw"
.globl wsPushX
wsPushX:
.zero 2
.section .bss.wsNewHigh,"aw"
.globl wsNewHigh
wsNewHigh:
.zero 2
.section .bss.wsDoubledByte,"aw"
.globl wsDoubledByte
wsDoubledByte:
.zero 2
.section .bss.wsHasTrail,"aw"
.globl wsHasTrail
wsHasTrail:
.zero 2