// Atari ST HAL for M2 + M2.5. // // M2 scope: // * XBIOS Setscreen to ST low-res (320x200x16, mode 0). // * Word-interleaved ST planar buffer copied to the screen at present. // // M2.5 scope (per-band palette / SCB emulation): // * jlpPresent scans the jlSurfaceT's SCB array and builds a compact // transitions table: each entry is (start_line, palette_index) // for a new palette region. For pattern.c's 8 uniform bands this // is 8 entries; in the worst case it is 200 (one per scanline). // * VBL ISR pre-loads the first band's palette, then programs // MFP Timer B (event-count mode, TBDR = HBL delta to first // transition) to fire at the END of the last scanline before // the next band starts. // * Timer B ISR writes the current band's palette, advances the // transition index, and (stop/reload TBDR/restart) reprograms // Timer B to fire at the next transition. With 8 transitions per // frame the ISR runs 8 times instead of 313 -- well under the // ~147-HBL-fires-per-frame cap Hatari's MFP emulation imposes on // event-count mode, and ~0.2% CPU overhead vs ~60% for per-HBL. // * gLinePalettes is a flat pre-quantized (line, color)->$0RGB // table built in jlpPresent by flattenScbPalettes; the ISR uses // its first row per band as the source of 16 shifter writes. // // Deferred: // * Takeover mode (direct shifter programming without TOS). // * STE's extended palette bits (we drop to STF 9-bit for now). #include #include #include #include #include #include "port.h" #include "surfaceInternal.h" #include "spriteInternal.h" #include "joey/tile.h" // ----- Constants ----- // Word-interleaved ST planar uses the same 160 bytes/scanline as our // chunky source, but organized as 20 groups of 4 words per scanline, // with each word holding the 16 one-bit samples for one bitplane. // (ST_BYTES_PER_ROW and ST_BYTES_PER_GROUP live in stPlanar.h so the // stTile.h inlines can see them; the rest are hal-private.) #define ST_GROUPS_PER_ROW 20 #define ST_PLANE_OFF_BYTES 2 // step between adjacent plane words within a group #define ST_BITPLANES 4 #define ST_PLANAR_SIZE (ST_BYTES_PER_ROW * SURFACE_HEIGHT) #define ST_SCREEN_ALIGN 256 // ----- Per-surface planar storage (project_planar_68k_plan, ST Phase 2) ----- // // StPlanarT and the inline portData accessor moved to stPlanar.h // (NATIVE-PERF Phase 2) so spriteCompile.c's dispatchers can read the // buffer base without a cross-TU call. #include "stPlanar.h" // Shifter palette registers: 16 words at $FFFF8240..$FFFF825F. #define ST_PALETTE_REGS ((volatile uint16_t *)0xFFFF8240L) // Shifter resolution ($FF8260, low 2 bits: 0=low 1=med 2=high/mono) and // sync-mode ($FF820A, bit 1: set=50 Hz PAL, clear=60 Hz NTSC) registers, // read by jlpFrameHz to report the true refresh instead of assuming PAL. #define ST_SHIFTER_RES ((volatile uint8_t *)0xFFFF8260L) #define ST_SYNC_MODE ((volatile uint8_t *)0xFFFF820AL) #define ST_RES_MONO 0x02u // $FF8260 low-2-bits value for high res (SM124) #define ST_SYNC_50HZ_BIT 0x02u // $FF820A bit 1: set = 50 Hz #define ST_HZ_MONO 71u // SM124 mono refresh #define ST_HZ_PAL 50u #define ST_HZ_NTSC 60u // Shifter video base address registers. The shifter latches the base // at the start of each frame's display (during vblank), so a write // here mid-visible-frame takes effect on the NEXT frame -- a deferred, // tear-free page flip with no TOS involvement. STF exposes only the // high and mid bytes; there is no low byte, which is why the ST screen // must be 256-byte aligned (the low 8 bits of the base are hardwired // to zero). Both bytes are odd-addressed byte registers; byte access // to an odd address is legal on 68000 (only word/long faults). #define ST_VID_BASE_HI ((volatile uint8_t *)0xFFFF8201L) // bits 23..16 #define ST_VID_BASE_MID ((volatile uint8_t *)0xFFFF8203L) // bits 15..8 // MFP hardware addresses. #define ST_MFP_TBCR ((volatile uint8_t *)0xFFFFFA1BL) // Timer B control #define ST_MFP_TBDR ((volatile uint8_t *)0xFFFFFA21L) // Timer B data #define ST_MFP_ISRA ((volatile uint8_t *)0xFFFFFA0FL) // In-service A #define MFP_TBCR_STOP 0x00 #define MFP_TBCR_EVENT 0x08 #define MFP_TB_CLEAR 0xFE // clear bit 0 of ISRA (Timer B) // Exception-vector numbers passed to Setexc (= vector offset / 4). #define VEC_VBL (0x70 / 4) // 68k autovector IRQ 4 #define VEC_MFP_TB (0x120 / 4) // MFP Timer B #define INT_TIMER_B 8 // ----- Prototypes ----- // Phase 10: planar primitive helpers must be visible everywhere they // could inline. Defined up here (between StPlanarT and the rest of // the prototype block) so every jlp / fillSpan / circle walker // can fold the 4-plane RMW directly into its body. always_inline // hammers the point home for gcc-mint's conservative inliner. static inline __attribute__((always_inline)) void stApplyMaskToGroup(uint8_t *groupBase, uint16_t mask, uint8_t color) { uint16_t notMask = (uint16_t)~mask; uint16_t *pw = (uint16_t *)groupBase; if (color & 1u) { pw[0] = (uint16_t)(pw[0] | mask); } else { pw[0] = (uint16_t)(pw[0] & notMask); } if (color & 2u) { pw[1] = (uint16_t)(pw[1] | mask); } else { pw[1] = (uint16_t)(pw[1] & notMask); } if (color & 4u) { pw[2] = (uint16_t)(pw[2] | mask); } else { pw[2] = (uint16_t)(pw[2] & notMask); } if (color & 8u) { pw[3] = (uint16_t)(pw[3] | mask); } else { pw[3] = (uint16_t)(pw[3] & notMask); } } static inline __attribute__((always_inline)) void stPlanarSetPixel(StPlanarT *pd, int16_t x, int16_t y, uint8_t color) { uint16_t group = (uint16_t)((uint16_t)x >> 4); uint16_t bitMask = (uint16_t)(1u << (15u - ((uint16_t)x & 15u))); uint16_t notMask = (uint16_t)~bitMask; uint16_t *pw = (uint16_t *)(pd->base + (uint16_t)y * ST_BYTES_PER_ROW + group * ST_BYTES_PER_GROUP); if (color & 1u) { pw[0] = (uint16_t)(pw[0] | bitMask); } else { pw[0] = (uint16_t)(pw[0] & notMask); } if (color & 2u) { pw[1] = (uint16_t)(pw[1] | bitMask); } else { pw[1] = (uint16_t)(pw[1] & notMask); } if (color & 4u) { pw[2] = (uint16_t)(pw[2] | bitMask); } else { pw[2] = (uint16_t)(pw[2] & notMask); } if (color & 8u) { pw[3] = (uint16_t)(pw[3] | bitMask); } else { pw[3] = (uint16_t)(pw[3] & notMask); } } static inline __attribute__((always_inline)) uint8_t stPlanarGetPixel(const StPlanarT *pd, int16_t x, int16_t y) { uint16_t group = (uint16_t)((uint16_t)x >> 4); uint16_t bitMask = (uint16_t)(1u << (15u - ((uint16_t)x & 15u))); const uint16_t *pw = (const uint16_t *)(pd->base + (uint16_t)y * ST_BYTES_PER_ROW + group * ST_BYTES_PER_GROUP); uint8_t c = 0u; if (pw[0] & bitMask) { c = (uint8_t)(c | 1u); } if (pw[1] & bitMask) { c = (uint8_t)(c | 2u); } if (pw[2] & bitMask) { c = (uint8_t)(c | 4u); } if (pw[3] & bitMask) { c = (uint8_t)(c | 8u); } return c; } // Flood-fill group decoder (PERF-AUDIT #4). For one 16-pixel group (4 // interleaved plane words) build the 16-bit mask whose bit // (15 - (x & 15)) is set exactly where pixel x equals color: invert // each plane word whose color bit is 0, then AND the four words. Four // word reads decode 16 pixels, vs 4 reads PER pixel through // stPlanarGetPixel. static inline __attribute__((always_inline)) uint16_t stGroupEqBits(const uint16_t *gp, uint8_t color) { uint16_t p0 = gp[0]; uint16_t p1 = gp[1]; uint16_t p2 = gp[2]; uint16_t p3 = gp[3]; if ((color & 1u) == 0u) { p0 = (uint16_t)~p0; } if ((color & 2u) == 0u) { p1 = (uint16_t)~p1; } if ((color & 4u) == 0u) { p2 = (uint16_t)~p2; } if ((color & 8u) == 0u) { p3 = (uint16_t)~p3; } return (uint16_t)(p0 & p1 & p2 & p3); } // Flood walk-out stop mask for one group: bit set where the walk must // STOP (pixel is not fillable). matchEqual (jlFloodFill): stop where // pix != matchColor. Bounded (jlFloodFillBounded): stop where // pix == matchColor (boundary) or pix == newColor (already filled). static inline __attribute__((always_inline)) uint16_t stFloodStopBits(const uint8_t *rowBase, uint16_t group, uint8_t matchColor, uint8_t newColor, bool matchEqual) { const uint16_t *gp = (const uint16_t *)(rowBase + group * ST_BYTES_PER_GROUP); if (matchEqual) { return (uint16_t)~stGroupEqBits(gp, matchColor); } return (uint16_t)(stGroupEqBits(gp, matchColor) | stGroupEqBits(gp, newColor)); } static uint16_t quantizeColorToSt(uint16_t orgb); static void flattenScbPalettes(const jlSurfaceT *src); static void setShifterVideoBase(const uint8_t *page); static void writeDiagnostics(void); static long writePrevPaletteRegs(void); static __attribute__((interrupt_handler)) void timerBIsr(void); static __attribute__((interrupt_handler)) void vblIsr(void); static void buildTransitions(const jlSurfaceT *src); static void refreshPaletteStateIfNeeded(const jlSurfaceT *src); // ----- Module state ----- // Screen buffer: TWO 320x200x4bpp planar pages for page-flipping, plus // padding for runtime 256-byte alignment. TOS .PRG format only supports // 2-byte object-file alignment, so we overallocate and align the first // page pointer manually in jlpInit; the second page sits exactly one // page (SURFACE_PIXELS_SIZE = 32000, itself a multiple of 256) after // the first, so it stays 256-byte aligned as the shifter requires. static uint8_t gScreenBuffer[(2 * SURFACE_PIXELS_SIZE) + ST_SCREEN_ALIGN]; // Two physical screen pages. gFrontIdx selects the one the shifter is // currently displaying; the OFF-screen page (gFrontIdx ^ 1) is the // flip target written by the next present. Drawing primitives target // the stage's own shadow planar buffer (StPlanarT.base), NOT a page, // so the union(prev,cur) dirty copy at present brings the off-screen // page fully up to date before the flip -- mirrors the Amiga 2-buffer // gPrevStage* staleness pattern (shadow -> off-screen buffer -> flip). static uint8_t *gScreenPage[2] = { NULL, NULL }; static uint8_t gFrontIdx = 0; static void *gPrevPhysbase = NULL; static void *gPrevLogbase = NULL; static int16_t gPrevRez = 0; static uint16_t gPrevPalette[SURFACE_COLORS_PER_PALETTE]; static bool gModeSet = false; // Supervisor-stack token returned by Super(0L) in jlpInit, used to drop // back to user mode in jlpShutdown. The ST maps the entire $FF8000.. // $FFFFFF I/O region (shifter video-base, palette, MFP) as supervisor- // only; a GEMDOS .PRG starts in user mode, so the direct register pokes // in jlpPresent (setShifterVideoBase) bus-error until we enter // supervisor for the program's lifetime. static void *gSavedSsp = NULL; static bool gSuperEntered = false; // Snapshot of the previous frame's dirty bands. With two physical // pages the off-screen page holds content from TWO frames ago, so each // present must refresh the union(prev, current) of dirty rows from the // shadow -- not just this frame's dirty rows -- or stale streaks from // two frames back survive on the page we are about to display. Seeded // to "all dirty" so the first two presents do full-screen refreshes // and converge both pages to the current shadow. After that we pay // only the small per-row union(prev,current) copy. static uint8_t gPrevStageMinWord[SURFACE_HEIGHT]; static uint8_t gPrevStageMaxWord[SURFACE_HEIGHT]; // "The previous frame's snapshot recorded at least one dirty row." // Lets jlpPresent skip the whole 200-row union scan + 200-row snapshot // copy on consecutive idle frames: if neither the previous frame nor // the current frame touched any row, union(prev,current) is empty, the // off-screen page already matches the displayed page (the settling // flip happened on the first idle frame), and the snapshot would only // copy clean->clean. Seeded true to match the "all dirty" bootstrap of // gPrevStage* above so the first two presents still run. static bool gPrevAnyDirty = true; // Per-scanline pre-quantized palette table. Indexed by display line; // each row is a 16-word palette ready to be copied straight into the // shifter registers. Written at present() time, read by the Timer B // ISR with no CPU-side math beyond a counter subtract. static uint16_t gLinePalettes[SURFACE_HEIGHT][SURFACE_COLORS_PER_PALETTE]; // Band-transition table. Each entry is one palette change: at // display line gBandStart[i], load palette indexed by gBandPalIdx[i]. // Built once per jlpPresent from the jlSurfaceT's SCB array. #define MAX_BANDS SURFACE_HEIGHT static uint16_t gBandStart [MAX_BANDS]; static uint8_t gBandPalIdx[MAX_BANDS]; static uint16_t gBandCount = 0; // Index of the band the Timer B ISR is currently scheduling TO. At // VBL this is 0 (band 0 palette pre-loaded, Timer B scheduled to // fire when it's time to transition to band 1). Each ISR fire writes // the palette for gCurrentBand and advances to the next. static volatile uint16_t gCurrentBand = 0; // Diagnostic captures. static volatile int16_t gFrameCount = 0; static volatile uint16_t gLastBandCount = 0; // Saved exception vectors for restore on shutdown. static void (*gOldVblVec)(void) = NULL; static void (*gOldTimerBVec)(void) = NULL; // Palette/SCB change detection is the core gStageScbDirty / // gStagePaletteDirty flags (surfaceInternal.h) -- the mutation APIs // set them and refreshPaletteStateIfNeeded clears them after // rebuilding, replacing a 712-byte cached-mirror memcmp per present // (NATIVE-PERF Phase 2 R9; the IIgs consumed the flags first). // flattenScbPalettes runs 200 * 16 quantize conversions and // buildTransitions rescans the full SCB; neither is cheap on a 7 MHz // 68000, and in the typical game loop SCB and palette never change. // ----- Internal helpers (alphabetical) ----- // Scan the surface's SCB and record one transition entry for each // run of the same palette index. gBandCount is the number of // distinct bands; gBandStart[i] is the display line where band i // begins; gBandPalIdx[i] is the palette index that band uses. static void buildTransitions(const jlSurfaceT *src) { uint16_t line; uint8_t idx; uint8_t prev; gBandCount = 0; prev = 0xFF; for (line = 0; line < SURFACE_HEIGHT; line++) { idx = src->scb[line]; if (idx >= SURFACE_PALETTE_COUNT) { idx = 0; } if (idx != prev) { if (gBandCount < MAX_BANDS) { gBandStart [gBandCount] = line; gBandPalIdx[gBandCount] = idx; gBandCount++; } prev = idx; } } gLastBandCount = gBandCount; } // Pre-quantize every palette row indexed by scanline through the SCB // into gLinePalettes, so the Timer B ISR can do a flat indexed copy // without any surface-level lookups. Called once per jlpPresent. static void flattenScbPalettes(const jlSurfaceT *src) { uint16_t line; uint16_t col; uint8_t idx; for (line = 0; line < SURFACE_HEIGHT; line++) { idx = src->scb[line]; if (idx >= SURFACE_PALETTE_COUNT) { idx = 0; } for (col = 0; col < SURFACE_COLORS_PER_PALETTE; col++) { gLinePalettes[line][col] = quantizeColorToSt(src->palette[idx][col]); } } } // Rebuild the per-line palette table and band-transition table only // when the SCB/palette state has actually changed. Both are hot -- the // flatten pass runs 3200 palette entries through quantization -- so // skipping them on clean frames dominates rect-present timing. `src` // is always gStage (jlpPresent's contract), so the core dirty flags // are the change signal; clear them only after rebuilding. static void refreshPaletteStateIfNeeded(const jlSurfaceT *src) { if (!gStageScbDirty && !gStagePaletteDirty) { return; } flattenScbPalettes(src); buildTransitions(src); gStageScbDirty = false; gStagePaletteDirty = false; } // 12-bit $0RGB to STF 9-bit palette register (drops the low bit of // each 4-bit channel). static uint16_t quantizeColorToSt(uint16_t orgb) { uint16_t r; uint16_t g; uint16_t b; r = (orgb >> 8) & 0x0F; g = (orgb >> 4) & 0x0F; b = orgb & 0x0F; r = r >> 1; g = g >> 1; b = b >> 1; return (uint16_t)((r << 8) | (g << 4) | b); } // Timer B interrupt handler. Fires once at each band transition; // writes the band's palette to the shifter and lets Timer B's // auto-reload keep counting for the next fire. We deliberately do // NOT stop/reload/restart the timer here: that sequence would cost // 1-2 HBL edges each fire, and those losses compound across 7+ // transitions into a visible "last band short" drift. // // MFP event-count auto-reload latches the CURRENT TBDR at each // underflow, so a value written here does not take effect on the // NEXT fire (that interval was already reloaded at the underflow // that triggered this ISR) but on the fire AFTER next. To time // variable-height bands correctly we therefore program one fire // ahead: in the ISR that transitions to `band` we load TBDR with // the band+1 -> band+2 delta (the interval that the band+1 // underflow will reload and use to time the band+2 transition). // Uniform bands (like pattern.c) have all deltas equal, so the // one-ahead value is identical and they stay perfectly aligned. static void timerBIsr(void) { uint16_t band; uint8_t palIdx; const uint16_t *src; volatile uint16_t *dst; uint16_t nextDelta; band = gCurrentBand + 1; gCurrentBand = band; if (band < gBandCount) { palIdx = gBandPalIdx[band]; if (palIdx >= SURFACE_PALETTE_COUNT) { palIdx = 0; } src = &gLinePalettes[gBandStart[band]][0]; dst = ST_PALETTE_REGS; dst[ 0] = src[ 0]; dst[ 1] = src[ 1]; dst[ 2] = src[ 2]; dst[ 3] = src[ 3]; dst[ 4] = src[ 4]; dst[ 5] = src[ 5]; dst[ 6] = src[ 6]; dst[ 7] = src[ 7]; dst[ 8] = src[ 8]; dst[ 9] = src[ 9]; dst[10] = src[10]; dst[11] = src[11]; dst[12] = src[12]; dst[13] = src[13]; dst[14] = src[14]; dst[15] = src[15]; if (band + 1 < gBandCount) { // A transition to band+1 is still pending; keep Timer B // running. The band+1 underflow has already had its reload // value latched (set one fire earlier), so program TBDR // now with the band+1 -> band+2 delta, which the band+1 // underflow will reload and use to time band+2. Don't // stop the timer. if (band + 2 < gBandCount) { nextDelta = gBandStart[band + 2] - gBandStart[band + 1]; if (nextDelta == 0 || nextDelta > 255) { nextDelta = 1; } *ST_MFP_TBDR = (uint8_t)nextDelta; } *ST_MFP_ISRA = MFP_TB_CLEAR; return; } } // No further transitions this frame; stopping Timer B here only // affects the (never-used) next fire, not timing of any band // we've already scheduled. *ST_MFP_TBCR = MFP_TBCR_STOP; *ST_MFP_ISRA = MFP_TB_CLEAR; } // Vertical blank handler. Pre-loads band 0's palette so the first // visible scanline is correct, then programs Timer B to fire at // the HBL delta to the next band transition (if any). static void vblIsr(void) { uint16_t delta; const uint16_t *src; volatile uint16_t *dst; gFrameCount = gFrameCount + 1; gCurrentBand = 0; if (gBandCount == 0) { return; } // Stage band 0's palette into the shifter registers. src = &gLinePalettes[gBandStart[0]][0]; dst = ST_PALETTE_REGS; dst[ 0] = src[ 0]; dst[ 1] = src[ 1]; dst[ 2] = src[ 2]; dst[ 3] = src[ 3]; dst[ 4] = src[ 4]; dst[ 5] = src[ 5]; dst[ 6] = src[ 6]; dst[ 7] = src[ 7]; dst[ 8] = src[ 8]; dst[ 9] = src[ 9]; dst[10] = src[10]; dst[11] = src[11]; dst[12] = src[12]; dst[13] = src[13]; dst[14] = src[14]; dst[15] = src[15]; // Program Timer B for the next band transition. if (gBandCount > 1) { delta = gBandStart[1] - gBandStart[0]; if (delta == 0 || delta > 255) { delta = 1; } *ST_MFP_TBCR = MFP_TBCR_STOP; *ST_MFP_TBDR = (uint8_t)delta; *ST_MFP_ISRA = MFP_TB_CLEAR; *ST_MFP_TBCR = MFP_TBCR_EVENT; // The restart above loads the counter from TBDR (= the band // 0 -> 1 delta), so the first fire is correctly timed. Now // pre-program TBDR with the band 1 -> 2 delta so the band-1 // underflow auto-reloads with the right interval (one fire // ahead, matching timerBIsr). Writing TBDR while counting // only changes the next reload, not the in-flight count. if (gBandCount > 2) { delta = gBandStart[2] - gBandStart[1]; if (delta == 0 || delta > 255) { delta = 1; } *ST_MFP_TBDR = (uint8_t)delta; } } else { *ST_MFP_TBCR = MFP_TBCR_STOP; *ST_MFP_ISRA = MFP_TB_CLEAR; } } // Point the shifter at a screen page by writing the high and mid base // bytes. The low byte does not exist on STF (base is 256-aligned), so // two byte writes fully specify the address. The shifter latches the // base at the next frame start, so calling this mid-visible-frame (as // jlpPresent does, after the app's jlWaitVBL) makes the flip take // effect on the following frame with no tearing. No TOS / Setscreen // involvement, which matters because our VBL ISR replaced TOS's and // does not run TOS's Setscreen-latch path. static void setShifterVideoBase(const uint8_t *page) { uintptr_t addr; addr = (uintptr_t)page; *ST_VID_BASE_HI = (uint8_t)((addr >> 16) & 0xFFu); *ST_VID_BASE_MID = (uint8_t)((addr >> 8) & 0xFFu); } static long writePrevPaletteRegs(void) { uint16_t i; for (i = 0; i < SURFACE_COLORS_PER_PALETTE; i++) { ST_PALETTE_REGS[i] = gPrevPalette[i]; } return 0; } static void writeDiagnostics(void) { FILE *fp; uint16_t i; fp = fopen("diag.txt", "w"); if (fp == NULL) { return; } fprintf(fp, "frames observed: %d\n", (int)gFrameCount); fprintf(fp, "band count: %d\n", (int)gLastBandCount); for (i = 0; i < gLastBandCount && i < 16; i++) { fprintf(fp, " band %2d: start line %3d, palIdx %d\n", (int)i, (int)gBandStart[i], (int)gBandPalIdx[i]); } fclose(fp); } // ----- HAL API (alphabetical) ----- bool jlpInit(const jlConfigT *config) { uintptr_t addr; uint16_t i; (void)config; // Enter supervisor mode for the program's lifetime. The shifter // video-base, palette, and MFP registers live in the supervisor-only // $FF8000..$FFFFFF I/O region; jlpPresent pokes the video-base bytes // directly every flip, and a .PRG launched by GEMDOS starts in user // mode, so without this the first present bus-errors. Staying super // for the whole run (vs a per-present Supexec trap) also keeps the // present hot path free of trap overhead. jlpShutdown restores user // mode via the saved stack token. GEMDOS/XBIOS/BIOS traps (including // the stdio used by joeylog) work fine from supervisor. if (!gSuperEntered) { gSavedSsp = (void *)Super(0L); gSuperEntered = true; } // Align page 0 to 256 bytes inside the static storage; page 1 // follows exactly one 32000-byte (256-aligned) page later, so it // is 256-aligned too. The shifter only displays 256-byte-aligned // bases, so both pages must satisfy that. addr = (uintptr_t)gScreenBuffer; addr = (addr + (ST_SCREEN_ALIGN - 1)) & ~((uintptr_t)ST_SCREEN_ALIGN - 1); gScreenPage[0] = (uint8_t *)addr; gScreenPage[1] = gScreenPage[0] + SURFACE_PIXELS_SIZE; gFrontIdx = 0; memset(gScreenPage[0], 0, SURFACE_PIXELS_SIZE); memset(gScreenPage[1], 0, SURFACE_PIXELS_SIZE); // Bootstrap: pretend the previous frame dirtied EVERY row so the // first two presents do full-page refreshes and bring both pages // into sync with the shadow. After that we converge to the small // per-row union(prev,current) cost. for (i = 0; i < SURFACE_HEIGHT; i++) { gPrevStageMinWord[i] = 0u; gPrevStageMaxWord[i] = STAGE_DIRTY_FULL_MAX; } gPrevPhysbase = Physbase(); gPrevLogbase = Logbase(); gPrevRez = Getrez(); // Capture current palette so we can restore exactly on shutdown. for (i = 0; i < SURFACE_COLORS_PER_PALETTE; i++) { gPrevPalette[i] = (uint16_t)Setcolor((int16_t)i, -1); } // Switch to ST low-res: 320x200x16, mode 0. Display page 0 first; // jlpPresent flips to the off-screen page via the shifter base // registers from here on. Setscreen((long)gScreenPage[0], (long)gScreenPage[0], 0); gModeSet = true; // Force hardware palette entry 0 to black so the overscan border // (which the ST shows in palette[0]) stays black until the app's // first refreshPaletteStateIfNeeded uploads its own palette. Setcolor(0, 0x000); // Save previous VBL + Timer B vectors, install ours. Timer B // is at MFP vector $120; vector installed by Xbtimer below. gOldVblVec = (void (*)(void))Setexc(VEC_VBL, -1L); gOldTimerBVec = (void (*)(void))Setexc(VEC_MFP_TB, -1L); (void)Setexc(VEC_VBL, (long)vblIsr); // Program MFP Timer B: event-count (HBL) mode, initial TBDR=1 // (a placeholder -- VBL ISR reprograms it for the first real // transition per frame), vector=timerBIsr, then enable in IMRA. Xbtimer(1, MFP_TBCR_EVENT, 1, timerBIsr); Jenabint(INT_TIMER_B); return true; } void jlpPresent(const jlSurfaceT *src) { StPlanarT *pd; uint8_t *backPage; uint8_t backIdx; int16_t y; int16_t firstDirty; int16_t lastDirty; int16_t runStart; uint8_t curMin; uint8_t curMax; uint8_t prvMin; uint8_t prvMax; uint8_t minWord; uint8_t maxWord; uint16_t groupStart; uint16_t groupEnd; uint16_t byteStart; uint16_t byteLen; bool fullWidth; if (src == NULL || !gModeSet) { return; } pd = (StPlanarT *)src->portData; if (pd == NULL) { return; } refreshPaletteStateIfNeeded(src); // Idle-frame early-out. When the previous frame's snapshot recorded // no dirty rows (gPrevAnyDirty == false) AND the current frame // touched no row, union(prev,current) is empty: both pages already // match the shadow (the one settling flip after the last activity // ran on the first idle frame), so there is nothing to copy, no // page to flip, and the snapshot below would only copy clean->clean. // Skip the whole 200-row union scan and the 200-row snapshot copy. // A row is dirty iff its min word stayed at or below its max word; // a single byte test per row with early exit on the first dirty row // suffices. Palette/SCB updates already ran above, so display state // stays correct. if (!gPrevAnyDirty) { bool anyDirty = false; for (y = 0; y < SURFACE_HEIGHT; y++) { if (!STAGE_DIRTY_ROW_CLEAN(y)) { anyDirty = true; break; } } if (!anyDirty) { return; } } // PAGE FLIP: draw into the stage's shadow (pd->base), bring the // OFF-screen page up to date with the union(prev,current) dirty // rows, then point the shifter at it. The off-screen page holds // content from TWO frames ago (strict A/B alternation), so any row // dirtied in either of the last two frames may differ from the // current shadow -- the union refresh covers exactly those rows. // The displayed (front) page is never touched, so there is no // tearing regardless of copy duration. backIdx = (uint8_t)(gFrontIdx ^ 1u); backPage = gScreenPage[backIdx]; // WIN 2: bound the present to the union dirty row range. The vast // majority of game frames touch a small vertical band; finding the // first/last union-dirty row up front lets the copy loop skip the // huge clean head/tail of the page, and lets us bail the whole copy // body when nothing is dirty (steady-state idle frames). The scan // is 200 cheap byte compares -- far less than 200 memcpy setups. firstDirty = -1; lastDirty = -1; for (y = 0; y < SURFACE_HEIGHT; y++) { curMin = gStageMinWord[y]; curMax = gStageMaxWord[y]; prvMin = gPrevStageMinWord[y]; prvMax = gPrevStageMaxWord[y]; minWord = (curMin < prvMin) ? curMin : prvMin; maxWord = (curMax > prvMax) ? curMax : prvMax; if (minWord <= maxWord) { if (firstDirty < 0) { firstDirty = y; } lastDirty = y; } } if (firstDirty >= 0) { // Shadow -> off-screen page. Each dirty word covers 4 pixels // (a quarter of an 8-byte group). Round to whole groups for a // simple aligned memcpy, since planar groups are the natural // copy unit. // // WIN 1: backPage and pd->base share the 160-byte row stride and // are both contiguous, so a run of consecutive FULL-WIDTH dirty // rows is one contiguous span of (rows * 160) bytes. Coalesce // such runs into a single memcpy: mintlib memcpy bursts long- // aligned blocks with movem.l/move.l, so one big copy beats many // tiny per-row ones (fewer call setups, longer inner burst). // Partial-width rows keep their own per-row group-aligned memcpy, // so narrow dirty bands stay correct and cheap. runStart < 0 // means "no full-width run is open". The per-row band is the // union(prev,current) band so a row stale on the off-screen page // from two frames ago is fully refreshed, not just this frame's // sub-span. runStart = -1; for (y = firstDirty; y <= lastDirty; y++) { curMin = gStageMinWord[y]; curMax = gStageMaxWord[y]; prvMin = gPrevStageMinWord[y]; prvMax = gPrevStageMaxWord[y]; minWord = (curMin < prvMin) ? curMin : prvMin; maxWord = (curMax > prvMax) ? curMax : prvMax; if (minWord > maxWord) { // Clean row breaks any open full-width run. if (runStart >= 0) { memcpy(&backPage [(uint16_t)runStart * ST_BYTES_PER_ROW], &pd->base [(uint16_t)runStart * ST_BYTES_PER_ROW], (uint16_t)(y - runStart) * (uint16_t)ST_BYTES_PER_ROW); runStart = -1; } continue; } // Defense-in-depth: an upstream clipping bug could leave a // maxWord past the row; clamp so the groupEnd-driven byteLen // can never push the memcpy into the next scanline. if (maxWord >= SURFACE_WORDS_PER_ROW) { maxWord = (uint8_t)(SURFACE_WORDS_PER_ROW - 1); } groupStart = (uint16_t)(minWord >> 2); groupEnd = (uint16_t)((maxWord >> 2) + 1); byteStart = (uint16_t)(groupStart * ST_BYTES_PER_GROUP); byteLen = (uint16_t)((groupEnd - groupStart) * ST_BYTES_PER_GROUP); fullWidth = (groupStart == 0 && byteLen == (uint16_t)ST_BYTES_PER_ROW); if (fullWidth) { // Extend (or open) the contiguous full-width run; defer // the copy until a non-full-width row or the range end // flushes it. if (runStart < 0) { runStart = y; } continue; } // Partial-width row: flush any open full-width run first so // the page stays in row order, then copy just this row's band. if (runStart >= 0) { memcpy(&backPage [(uint16_t)runStart * ST_BYTES_PER_ROW], &pd->base [(uint16_t)runStart * ST_BYTES_PER_ROW], (uint16_t)(y - runStart) * (uint16_t)ST_BYTES_PER_ROW); runStart = -1; } memcpy(&backPage [(uint16_t)y * ST_BYTES_PER_ROW + byteStart], &pd->base [(uint16_t)y * ST_BYTES_PER_ROW + byteStart], byteLen); } // Flush a full-width run that reached the end of the dirty range. if (runStart >= 0) { memcpy(&backPage [(uint16_t)runStart * ST_BYTES_PER_ROW], &pd->base [(uint16_t)runStart * ST_BYTES_PER_ROW], (uint16_t)(lastDirty + 1 - runStart) * (uint16_t)ST_BYTES_PER_ROW); } // The off-screen page now holds the complete current image. // Point the shifter at it; the latch happens at the next frame // start (we are mid-visible-frame here, after the app's // jlWaitVBL), so the swap is tear-free. Only flip when there was // something to refresh -- flipping on a fully idle frame would // oscillate the display between the two pages with no benefit. setShifterVideoBase(backPage); gFrontIdx = backIdx; } // Snapshot this frame's dirty bands as "previous" for the next // present. jlStagePresent clears the live bands right after // jlpPresent returns, so we must grab them now. Next present unions // these with that frame's dirty rows to refresh whatever the other // page is stale on. Record whether any row was dirty so the next // present can take the idle-frame early-out above. { bool anyDirty = false; for (y = 0; y < SURFACE_HEIGHT; y++) { gPrevStageMinWord[y] = gStageMinWord[y]; gPrevStageMaxWord[y] = gStageMaxWord[y]; if (!STAGE_DIRTY_ROW_CLEAN(y)) { anyDirty = true; } } gPrevAnyDirty = anyDirty; } } // Vsync() is XBIOS opcode 37; mintlib exposes it directly. It blocks // until the next 50 Hz (PAL) or 60 Hz (NTSC) vertical blank. void jlpWaitVBL(void) { int16_t before; // Can't use Vsync(): TOS's Vsync increments _vblsem inside its // own VBL ISR, which we replaced (Setexc(VEC_VBL, vblIsr)) with // our SCB-emulating ISR that doesn't chain to the original. // Spin on gFrameCount instead -- it's volatile and bumped every // VBL by our ISR. before = gFrameCount; while (gFrameCount == before) { // wait } } // gFrameCount is already maintained by our VBL ISR; just narrow to // uint16_t for the cross-port HAL contract. uint16_t jlpFrameCount(void) { return (uint16_t)gFrameCount; } uint16_t jlpFrameHz(void) { /* Report the REAL refresh, not a hardcoded 50. The old assumption * silently under-reported ops/sec ~17% under Hatari (which runs the * ST at 60 Hz) and would be wrong on any NTSC/mono machine. The * shifter resolution register selects mono vs color; for color the * sync-mode register's 50/60 bit is authoritative. The program holds * supervisor mode for its lifetime (jlpInit Super(0L)), so these * $FFFF82xx I/O reads are legal. */ if ((*ST_SHIFTER_RES & 0x03u) == ST_RES_MONO) { return ST_HZ_MONO; } if (*ST_SYNC_MODE & ST_SYNC_50HZ_BIT) { return ST_HZ_PAL; } return ST_HZ_NTSC; } // TOS maintains a 200 Hz long counter at $4BA, incremented by the // Timer C VBL chain. Read in supervisor mode (Supexec) so we don't // fault on user-mode access. Resolution is 5 ms, plenty for sound // scheduling at 60 Hz AGI ticks. static volatile uint32_t gMillisBase200 = 0; static volatile bool gMillisBaseSet = false; static long readHz200(void) { return (long)*((volatile uint32_t *)0x4BAUL); } uint32_t jlpMillisElapsed(void) { uint32_t ticks; ticks = (uint32_t)Supexec(readHz200); // 0 is a legal counter value (power-on tick 0 and every ~248-day // wrap), so use an explicit one-shot flag instead of a value-0 // sentinel to capture the base. if (!gMillisBaseSet) { gMillisBase200 = ticks; gMillisBaseSet = true; return 0u; } // 200 Hz counter -> ms = ticks * 5. Unsigned subtraction is wrap-safe. return (uint32_t)((ticks - gMillisBase200) * 5u); } void jlpShutdown(void) { if (!gModeSet) { return; } // Stop the audio Timer A first. The audio HAL has its own // jlpAudioShutdown that disables Timer A and restores the vector, // but cross-platform jlShutdown doesn't call it -- if a sketch // forgets jlAudioShutdown(), Timer A keeps firing after our // code unloads and TOS panics on the first dangling vector hit. // Calling jlpAudioShutdown here is idempotent (gReady guard), // so explicit-shutdown sketches still work. jlpAudioShutdown(); // Disable MFP Timer B and restore the exception vectors before // changing the screen -- a late ISR firing mid-Setscreen would // write palette into whatever buffer TOS remapped. Jdisint(INT_TIMER_B); if (gOldTimerBVec != NULL) { (void)Setexc(VEC_MFP_TB, (long)gOldTimerBVec); } if (gOldVblVec != NULL) { (void)Setexc(VEC_VBL, (long)gOldVblVec); } Setscreen((long)gPrevLogbase, (long)gPrevPhysbase, gPrevRez); Supexec(writePrevPaletteRegs); writeDiagnostics(); gModeSet = false; // Drop back to user mode so control returns to GEMDOS the way it // launched us (mirrors the Super(0L) in jlpInit). if (gSuperEntered) { (void)Super(gSavedSsp); gSuperEntered = false; } } // ST word-interleaved planar 68k fast paths. The surface68kSt* // primitives are ST-specific (plane-major group layout); they live in // src/shared68k/surface68k.s alongside the Amiga-only planar helpers. extern void surface68kStCircleOutline(uint8_t *base, uint16_t cx, uint16_t cy, uint16_t r, uint8_t color); extern void surface68kStDrawLine(uint8_t *base, int16_t x0, int16_t y0, int16_t x1, int16_t y1, uint8_t color); extern void surface68kStFillSpan(uint8_t *base, int16_t left, int16_t right, int16_t y, uint8_t color); extern void surface68kStFillCircle(uint8_t *base, uint16_t cx, uint16_t cy, uint16_t r, uint8_t color); extern void surface68kStFillRectSingleGroup(uint8_t *firstGroupPtr, uint16_t mask, uint16_t h, uint8_t color); extern void surface68kStFillRectMulti(uint8_t *base, int16_t x, int16_t y, uint16_t w, uint16_t h, uint8_t color); extern void surface68kStLongFill(uint8_t *dst, uint16_t numGroups, uint32_t loLong, uint32_t hiLong); extern void surface68kStSprite16x16Save(uint8_t *base, uint16_t x, uint16_t y, uint8_t *dstBuf); extern void surface68kStSprite16x16Restore(uint8_t *base, uint16_t x, uint16_t y, const uint8_t *srcBuf); extern void surface68kStSpriteSaveByteAligned(uint8_t *base, uint16_t x, uint16_t y, uint16_t w, uint16_t h, uint8_t *dstPlaneBytes); extern void surface68kStSpriteRestoreByteAligned(uint8_t *base, uint16_t x, uint16_t y, uint16_t w, uint16_t h, const uint8_t *srcPlaneBytes); // Phase 9: clear the entire planar buffer to a 4-bit color. Build an // 8-byte group template (4 plane words: 0xFFFF or 0x0000 each by // color bit) then stream it across all 4000 groups via long stores. void jlpSurfaceClear(jlSurfaceT *s, uint8_t doubled) { StPlanarT *pd; uint8_t color; uint32_t loLong; uint32_t hiLong; if (s->portData == NULL) { jlpGenericSurfaceClear(s, doubled); return; } pd = (StPlanarT *)s->portData; color = (uint8_t)(doubled & 0x0Fu); /* Per-group: [p0_word][p1_word][p2_word][p3_word] = 8 bytes = 2 longs. * loLong = (p0_word << 16) | p1_word; hiLong = (p2_word << 16) | p3_word. */ loLong = ((color & 1u) ? 0xFFFF0000ul : 0ul) | ((color & 2u) ? 0x0000FFFFul : 0ul); hiLong = ((color & 4u) ? 0xFFFF0000ul : 0ul) | ((color & 8u) ? 0x0000FFFFul : 0ul); surface68kStLongFill(pd->base, (uint16_t)(ST_PLANAR_SIZE / ST_BYTES_PER_GROUP), loLong, hiLong); } // Phase 9: planar-only. Chunky shadow is gone; only the planar buffer // gets the pixel. void jlpDrawPixel(jlSurfaceT *s, uint16_t x, uint16_t y, uint8_t colorIndex) { StPlanarT *pd; if (s->portData == NULL) { jlpGenericDrawPixel(s, x, y, colorIndex); return; } pd = (StPlanarT *)s->portData; stPlanarSetPixel(pd, (int16_t)x, (int16_t)y, (uint8_t)(colorIndex & 0x0Fu)); } // Phase 9 planar walkers. Same Bresenham as the cross-platform // fallback, but writing to the planar buffer via stPlanarSetPixel. // Mirror the Amiga amigaPlanarLine / amigaPlanarCircleOutline / // amigaPlanarCircleFill structure. Phase 10 hand-rolled asm replaces // these (jlDrawCircle.s already exists for Amiga; ST will get its own). static void stPlanarLine(StPlanarT *pd, int16_t x0, int16_t y0, int16_t x1, int16_t y1, uint8_t color) { int16_t dx; int16_t dy; int16_t sx; int16_t sy; int16_t err; int16_t e2; dx = (int16_t)((x1 > x0) ? (x1 - x0) : (x0 - x1)); dy = (int16_t)(-((y1 > y0) ? (y1 - y0) : (y0 - y1))); sx = (int16_t)((x0 < x1) ? 1 : -1); sy = (int16_t)((y0 < y1) ? 1 : -1); err = (int16_t)(dx + dy); while (1) { if (x0 >= 0 && x0 < SURFACE_WIDTH && y0 >= 0 && y0 < SURFACE_HEIGHT) { stPlanarSetPixel(pd, x0, y0, color); } if (x0 == x1 && y0 == y1) { break; } e2 = (int16_t)(2 * err); if (e2 >= dy) { err = (int16_t)(err + dy); x0 = (int16_t)(x0 + sx); } if (e2 <= dx) { err = (int16_t)(err + dx); y0 = (int16_t)(y0 + sy); } } } static void stPlanarCircleOutline(StPlanarT *pd, int16_t cx, int16_t cy, uint16_t r, uint8_t color) { int16_t bx = (int16_t)r; int16_t by = 0; int16_t err = (int16_t)(1 - bx); int16_t px; int16_t py; while (bx >= by) { /* 8 octants. Per-pixel clip since Bresenham can leave the * surface for circles touching the edge. */ #define ST_PLOT(X, Y) do { px = (X); py = (Y); if (px >= 0 && px < SURFACE_WIDTH && py >= 0 && py < SURFACE_HEIGHT) { stPlanarSetPixel(pd, px, py, color); } } while (0) ST_PLOT((int16_t)(cx + bx), (int16_t)(cy + by)); ST_PLOT((int16_t)(cx - bx), (int16_t)(cy + by)); ST_PLOT((int16_t)(cx + bx), (int16_t)(cy - by)); ST_PLOT((int16_t)(cx - bx), (int16_t)(cy - by)); ST_PLOT((int16_t)(cx + by), (int16_t)(cy + bx)); ST_PLOT((int16_t)(cx - by), (int16_t)(cy + bx)); ST_PLOT((int16_t)(cx + by), (int16_t)(cy - bx)); ST_PLOT((int16_t)(cx - by), (int16_t)(cy - bx)); #undef ST_PLOT by++; if (err > 0) { bx--; err = (int16_t)(err + 2 * (by - bx) + 1); } else { err = (int16_t)(err + 2 * by + 1); } } } // Phase 10: group-aware span fill -- the same leading-mask / // full-group / trailing-mask decomposition jlpFillRect uses, // but for one row. Replaces the per-pixel walk that gave jlFillCircle // r=40 ~1 ops/sec. static void stPlanarFillSpan(StPlanarT *pd, int16_t x0, int16_t x1, int16_t y, uint8_t color) { int16_t left; int16_t right; if (y < 0 || y >= SURFACE_HEIGHT) { return; } left = (x0 < x1) ? x0 : x1; right = (x0 > x1) ? x0 : x1; if (left < 0) { left = 0; } if (right >= SURFACE_WIDTH) { right = SURFACE_WIDTH - 1; } if (left > right) { return; } surface68kStFillSpan(pd->base, left, right, y, color); } // Exact-disk span walk, same row coverage as src/core/draw.c // jlFillCircle and the asm surface68kStFillCircle: for each y offset // the largest x with x*x + y*y <= r*r, tracked incrementally. Defensive // fallback only -- draw.c routes every off-surface circle through its // own clipped span loop before jlpFillCircle is ever called, so keep r // modest (< SURFACE_HEIGHT) if a future caller ever lands here. static void stPlanarCircleFill(StPlanarT *pd, int16_t cx, int16_t cy, uint16_t r, uint8_t color) { uint32_t r2 = (uint32_t)r * (uint32_t)r; uint32_t xx = r2; uint32_t yy = 0u; int32_t x = (int32_t)r; uint16_t y; for (y = 0u; y <= r; y++) { uint32_t xxLimit; xxLimit = r2 - yy; while (xx > xxLimit) { xx = xx - ((uint32_t)x + (uint32_t)x - 1u); x--; } stPlanarFillSpan(pd, (int16_t)(cx - (int16_t)x), (int16_t)(cx + (int16_t)x), (int16_t)(cy + (int16_t)y), color); if (y > 0u) { stPlanarFillSpan(pd, (int16_t)(cx - (int16_t)x), (int16_t)(cx + (int16_t)x), (int16_t)(cy - (int16_t)y), color); } yy = yy + ((uint32_t)y + (uint32_t)y + 1u); } } void jlpDrawLine(jlSurfaceT *s, int16_t x0, int16_t y0, int16_t x1, int16_t y1, uint8_t colorIndex) { StPlanarT *pd; if (s->portData == NULL) { return; } pd = (StPlanarT *)s->portData; // Asm walker assumes fully on-surface; partial-clip lines fall // back to the C walker which clips per-pixel. if (x0 >= 0 && x0 < SURFACE_WIDTH && y0 >= 0 && y0 < SURFACE_HEIGHT && x1 >= 0 && x1 < SURFACE_WIDTH && y1 >= 0 && y1 < SURFACE_HEIGHT) { surface68kStDrawLine(pd->base, x0, y0, x1, y1, (uint8_t)(colorIndex & 0x0Fu)); } else { stPlanarLine(pd, x0, y0, x1, y1, (uint8_t)(colorIndex & 0x0Fu)); } } void jlpDrawCircle(jlSurfaceT *s, int16_t cx, int16_t cy, uint16_t r, uint8_t colorIndex) { StPlanarT *pd; if (s->portData == NULL) { return; } pd = (StPlanarT *)s->portData; /* Off-surface circles fall back to the per-pixel C walker which * does the clip per plot; the asm assumes fully-on-surface so it * can drop the clip check from the inner loop. */ if ((int32_t)cx - (int32_t)r < 0 || (int32_t)cx + (int32_t)r >= SURFACE_WIDTH || (int32_t)cy - (int32_t)r < 0 || (int32_t)cy + (int32_t)r >= SURFACE_HEIGHT) { stPlanarCircleOutline(pd, cx, cy, r, (uint8_t)(colorIndex & 0x0Fu)); } else { surface68kStCircleOutline(pd->base, (uint16_t)cx, (uint16_t)cy, r, (uint8_t)(colorIndex & 0x0Fu)); } } bool jlpFillCircle(jlSurfaceT *s, int16_t cx, int16_t cy, uint16_t r, uint8_t colorIndex) { StPlanarT *pd; if (s->portData == NULL) { return false; } pd = (StPlanarT *)s->portData; // Off-surface bounding box falls back to the C span walker, which // clips each span; the asm assumes the whole circle is on-surface. if ((int32_t)cx - (int32_t)r < 0 || (int32_t)cx + (int32_t)r >= SURFACE_WIDTH || (int32_t)cy - (int32_t)r < 0 || (int32_t)cy + (int32_t)r >= SURFACE_HEIGHT) { stPlanarCircleFill(pd, cx, cy, r, (uint8_t)(colorIndex & 0x0Fu)); } else { surface68kStFillCircle(pd->base, (uint16_t)cx, (uint16_t)cy, r, (uint8_t)(colorIndex & 0x0Fu)); } return true; } // Phase 2: allocate a shadow word-interleaved planar buffer per // surface. Both stage and non-stage get their own buffer; the two // displayed pages (gScreenPage[0/1]) are the flip targets, kept // separate from every surface's shadow. // // LONG alignment is required, not just word: the full-row long-fill // path and circle.s both do `move.l` writes on this buffer, and // 68000 address-errors on long access to a word-aligned-but-not- // long-aligned destination. mintlib's malloc usually returns long- // aligned blocks, but TOS heaps can land at odd offsets after a // few allocations -- over-allocate by 4 bytes and align up here. // Symptom of getting this wrong: intermittent return-to-desktop // after the red startup paint as the first long write hits an // odd-by-2 base. void *jlpSurfaceAllocPortData(jlSurfaceT *s, bool isStage) { StPlanarT *pd; uint8_t *raw; uintptr_t addr; (void)s; (void)isStage; pd = (StPlanarT *)calloc(1, sizeof(StPlanarT)); if (pd == NULL) { return NULL; } raw = (uint8_t *)malloc(ST_PLANAR_SIZE + 4u); if (raw == NULL) { free(pd); return NULL; } addr = (uintptr_t)raw; addr = (addr + 3u) & ~(uintptr_t)3u; /* round up to long-aligned */ pd->raw = raw; pd->base = (uint8_t *)addr; pd->ownsBuffer = true; memset(pd->base, 0, ST_PLANAR_SIZE); return pd; } void jlpSurfaceFreePortData(jlSurfaceT *s, bool isStage, void *portData) { StPlanarT *pd; (void)s; (void)isStage; if (portData == NULL) { return; } pd = (StPlanarT *)portData; if (pd->ownsBuffer && pd->raw != NULL) { free(pd->raw); } free(pd); } // Phase 3: dual-write to the word-interleaved planar shadow buffer. // Chunky shadow (s->pixels) is still the source-of-truth for display // (c2p at present); the planar buffer becomes authoritative at // Phase 9 switch flip. // // Per row: split [x, x+w) into a leading partial group (bits // 15..15-bitFirst within the leading word -> mask = (1<<(16-bitFirst)) // - 1), zero or more full groups, and a trailing partial group // (bits 15..15-bitLast -> mask = ~((1<<(15-bitLast)) - 1)). For each // of the 4 plane words within a group, the bit value of the color // index controls OR-with-mask (set) vs AND-with-not-mask (clear). // Single-group case (groupFirst == groupLast) collapses to one word // RMW per plane with the combined mask. // (stApplyMaskToGroup is defined inline near the top of the file.) void jlpFillRect(jlSurfaceT *s, int16_t x, int16_t y, int16_t w, int16_t h, uint8_t colorIndex) { StPlanarT *pd; uint16_t groupFirst; uint16_t groupLast; uint8_t *rowBase; if (s->portData == NULL) { jlpGenericFillRect(s, x, y, w, h, colorIndex); return; } // Callers guarantee s != NULL and positive w/h: the jlFillRect // wrapper validates + clips before fillRectOnSurface, and the // fill-circle span loop emits spanWidth = 2x+1 >= 1 with h = 1. pd = (StPlanarT *)s->portData; if (pd == NULL) { return; } /* Phase 10 fast path: x == 0 AND w == SURFACE_WIDTH means the rect * spans every group on every row with no edge masks. movem.l-based * asm long-fill batches 24 bytes per call. UBER jlFillRect 320x200 * lands here. */ if (x == 0 && w == SURFACE_WIDTH) { uint32_t loLong = ((colorIndex & 1u) ? 0xFFFF0000ul : 0ul) | ((colorIndex & 2u) ? 0x0000FFFFul : 0ul); uint32_t hiLong = ((colorIndex & 4u) ? 0xFFFF0000ul : 0ul) | ((colorIndex & 8u) ? 0x0000FFFFul : 0ul); surface68kStLongFill(pd->base + (uint16_t)y * ST_BYTES_PER_ROW, (uint16_t)((uint16_t)h * ST_GROUPS_PER_ROW), loLong, hiLong); return; } groupFirst = (uint16_t)((uint16_t)x >> 4); groupLast = (uint16_t)(((uint16_t)x + w - 1u) >> 4); if (groupFirst == groupLast) { uint16_t bitFirst = (uint16_t)((uint16_t)x & 15u); uint16_t bitLast = (uint16_t)(((uint16_t)x + w - 1u) & 15u); uint16_t leftMask = (uint16_t)((1ul << (16u - bitFirst)) - 1ul); uint16_t rightMask = (uint16_t)~((1ul << (15u - bitLast)) - 1ul); uint16_t mask = (uint16_t)(leftMask & rightMask); rowBase = pd->base + (uint16_t)y * ST_BYTES_PER_ROW; surface68kStFillRectSingleGroup(rowBase + groupFirst * ST_BYTES_PER_GROUP, mask, h, colorIndex); return; } /* Phase 10.5: multi-group case (groupFirst != groupLast) handled * by 16-way-color-dispatched asm with hoisted mask state. ~3-5x * faster than the C loop with inlined stApplyMaskToGroup. */ surface68kStFillRectMulti(pd->base, x, y, w, h, colorIndex); } void jlpSurfaceCopyPlanes(jlSurfaceT *dst, const jlSurfaceT *src) { StPlanarT *dstPd; StPlanarT *srcPd; if (dst == NULL || src == NULL) { return; } dstPd = (StPlanarT *)dst->portData; srcPd = (StPlanarT *)src->portData; if (dstPd == NULL || srcPd == NULL) { return; } memcpy(dstPd->base, srcPd->base, ST_PLANAR_SIZE); } // ----- Phases 4-7: per-pixel / tile / sprite / blit planar primitives ----- // // These implementations dual-write the planar shadow alongside the // chunky shadow that cross-platform code maintains. They use simple // per-pixel walks for clarity and correctness; Phase 10 will replace // the hot ones (jlFillRect, jlDrawPixel, sprite codegen) with hand-rolled // asm. The emphasis here is "correct first, fast later" -- Phase 9 // flips the read source from chunky to planar and we'll see immediately // (DRAW hash vs IIgs reference) whether each primitive landed bits // in the right place. // stPlanarSetPixel and stPlanarGetPixel are defined inline near the // top of the file (between StPlanarT and the prototype block) so // every callsite folds the 4-plane RMW into its body. // Phase 10 fast path: byte-aligned, fully-on-surface sprite draw. // Builds 4 plane bytes + 1 opacity byte from each tile-column row // in one pass, then does 4 word RMWs per group half. ~7x faster // than the per-pixel walker for the typical (byte-aligned) case. // // Per row of a tile column: 4 chunky bytes -> 8 nibbles -> {plane0 // byte, plane1 byte, plane2 byte, plane3 byte, opacity byte}. The // opacity byte has bits set where the sprite pixel is non-zero; // transparent pixels (color 0) leave the destination plane bits // alone via the (word AND ~opMask) | (planeBits AND opMask) RMW. // // 8 pixels at byte-aligned x always cover exactly one half of one // group: high half if (x mod 16) == 0, low half if (x mod 16) == 8. // We branch once per tile column on (dstX & 8). static void stSpriteDrawByteAligned(StPlanarT *pd, const jlSpriteT *sp, int16_t x, int16_t y) { uint16_t wTiles = sp->widthTiles; int16_t srcH = (int16_t)(sp->heightTiles * 8); uint8_t *rowBase = pd->base + (uint16_t)y * ST_BYTES_PER_ROW; int16_t row; for (row = 0; row < srcH; row++) { int16_t tileY = (int16_t)(row >> 3); int16_t inTileY = (int16_t)(row & 7); const uint8_t *tileRowBase = sp->tileData + (uint32_t)tileY * wTiles * 32u + (uint32_t)inTileY * 4u; int16_t tileCol; for (tileCol = 0; tileCol < (int16_t)wTiles; tileCol++) { const uint8_t *trp = tileRowBase + (uint32_t)tileCol * 32u; uint8_t b0 = trp[0]; uint8_t b1 = trp[1]; uint8_t b2 = trp[2]; uint8_t b3 = trp[3]; uint8_t pb0 = 0u; uint8_t pb1 = 0u; uint8_t pb2 = 0u; uint8_t pb3 = 0u; uint8_t pop = 0u; uint8_t c; /* 8 pixels per tile column: hi(b0),lo(b0),hi(b1),lo(b1), * hi(b2),lo(b2),hi(b3),lo(b3) at bit positions 7..0 * within the eventual plane byte. Walk inline -- no LUT * loop overhead. */ c = (uint8_t)(b0 >> 4); if (c) { pop = (uint8_t)(pop | 0x80u); if (c & 1u) pb0 = (uint8_t)(pb0 | 0x80u); if (c & 2u) pb1 = (uint8_t)(pb1 | 0x80u); if (c & 4u) pb2 = (uint8_t)(pb2 | 0x80u); if (c & 8u) pb3 = (uint8_t)(pb3 | 0x80u); } c = (uint8_t)(b0 & 0x0Fu); if (c) { pop = (uint8_t)(pop | 0x40u); if (c & 1u) pb0 = (uint8_t)(pb0 | 0x40u); if (c & 2u) pb1 = (uint8_t)(pb1 | 0x40u); if (c & 4u) pb2 = (uint8_t)(pb2 | 0x40u); if (c & 8u) pb3 = (uint8_t)(pb3 | 0x40u); } c = (uint8_t)(b1 >> 4); if (c) { pop = (uint8_t)(pop | 0x20u); if (c & 1u) pb0 = (uint8_t)(pb0 | 0x20u); if (c & 2u) pb1 = (uint8_t)(pb1 | 0x20u); if (c & 4u) pb2 = (uint8_t)(pb2 | 0x20u); if (c & 8u) pb3 = (uint8_t)(pb3 | 0x20u); } c = (uint8_t)(b1 & 0x0Fu); if (c) { pop = (uint8_t)(pop | 0x10u); if (c & 1u) pb0 = (uint8_t)(pb0 | 0x10u); if (c & 2u) pb1 = (uint8_t)(pb1 | 0x10u); if (c & 4u) pb2 = (uint8_t)(pb2 | 0x10u); if (c & 8u) pb3 = (uint8_t)(pb3 | 0x10u); } c = (uint8_t)(b2 >> 4); if (c) { pop = (uint8_t)(pop | 0x08u); if (c & 1u) pb0 = (uint8_t)(pb0 | 0x08u); if (c & 2u) pb1 = (uint8_t)(pb1 | 0x08u); if (c & 4u) pb2 = (uint8_t)(pb2 | 0x08u); if (c & 8u) pb3 = (uint8_t)(pb3 | 0x08u); } c = (uint8_t)(b2 & 0x0Fu); if (c) { pop = (uint8_t)(pop | 0x04u); if (c & 1u) pb0 = (uint8_t)(pb0 | 0x04u); if (c & 2u) pb1 = (uint8_t)(pb1 | 0x04u); if (c & 4u) pb2 = (uint8_t)(pb2 | 0x04u); if (c & 8u) pb3 = (uint8_t)(pb3 | 0x04u); } c = (uint8_t)(b3 >> 4); if (c) { pop = (uint8_t)(pop | 0x02u); if (c & 1u) pb0 = (uint8_t)(pb0 | 0x02u); if (c & 2u) pb1 = (uint8_t)(pb1 | 0x02u); if (c & 4u) pb2 = (uint8_t)(pb2 | 0x02u); if (c & 8u) pb3 = (uint8_t)(pb3 | 0x02u); } c = (uint8_t)(b3 & 0x0Fu); if (c) { pop = (uint8_t)(pop | 0x01u); if (c & 1u) pb0 = (uint8_t)(pb0 | 0x01u); if (c & 2u) pb1 = (uint8_t)(pb1 | 0x01u); if (c & 4u) pb2 = (uint8_t)(pb2 | 0x01u); if (c & 8u) pb3 = (uint8_t)(pb3 | 0x01u); } if (pop != 0u) { int16_t dstX = (int16_t)(x + tileCol * 8); uint16_t group = (uint16_t)((uint16_t)dstX >> 4); uint16_t *pw = (uint16_t *)(rowBase + group * ST_BYTES_PER_GROUP); uint16_t opMask; uint16_t notOpMask; uint16_t pv0; uint16_t pv1; uint16_t pv2; uint16_t pv3; if ((dstX & 8) == 0) { opMask = (uint16_t)((uint16_t)pop << 8); pv0 = (uint16_t)((uint16_t)pb0 << 8); pv1 = (uint16_t)((uint16_t)pb1 << 8); pv2 = (uint16_t)((uint16_t)pb2 << 8); pv3 = (uint16_t)((uint16_t)pb3 << 8); } else { opMask = (uint16_t)pop; pv0 = (uint16_t)pb0; pv1 = (uint16_t)pb1; pv2 = (uint16_t)pb2; pv3 = (uint16_t)pb3; } notOpMask = (uint16_t)~opMask; pw[0] = (uint16_t)((pw[0] & notOpMask) | pv0); pw[1] = (uint16_t)((pw[1] & notOpMask) | pv1); pw[2] = (uint16_t)((pw[2] & notOpMask) | pv2); pw[3] = (uint16_t)((pw[3] & notOpMask) | pv3); } } rowBase += ST_BYTES_PER_ROW; } } // Phase 10: sprite walker with hoisted state. rowBase advances by // 160 per row instead of recomputing y*160 per pixel; tile-row // pointer is advanced once per tile column (8 cols) instead of // recomputed per pixel; the per-pixel inner block is the inlined // stPlanarSetPixel body so there's no nested function entry / y*160 // re-derivation. Major rewrite of the dispatcher path that drove // the 0.06x sprite gap before this commit. void jlpSpriteDrawPlanes(jlSurfaceT *s, const jlSpriteT *sp, int16_t x, int16_t y) { StPlanarT *pd; int16_t spritePxStart; int16_t spritePyStart; int16_t srcW; int16_t srcH; int16_t sx; int16_t sy; int16_t row; int16_t col; uint16_t wTiles; uint8_t *rowBase; if (s == NULL || sp == NULL || sp->tileData == NULL) { return; } pd = (StPlanarT *)s->portData; if (pd == NULL) { return; } wTiles = sp->widthTiles; srcW = (int16_t)(wTiles * 8); srcH = (int16_t)(sp->heightTiles * 8); spritePxStart = x; spritePyStart = y; /* Phase 10 fast path: byte-aligned x AND fully on-surface lets * us bulk-write 8 pixels into one group half per tile column, * skipping the per-pixel walker entirely. UBER's jlSpriteDraw test * at (160, 100) lands here. */ if ((x & 7) == 0 && x >= 0 && (x + srcW) <= SURFACE_WIDTH && y >= 0 && (y + srcH) <= SURFACE_HEIGHT) { stSpriteDrawByteAligned(pd, sp, x, y); return; } sx = 0; sy = 0; if (spritePxStart < 0) { sx = (int16_t)(-spritePxStart); srcW = (int16_t)(srcW - sx); spritePxStart = 0; } if (spritePyStart < 0) { sy = (int16_t)(-spritePyStart); srcH = (int16_t)(srcH - sy); spritePyStart = 0; } if (spritePxStart >= SURFACE_WIDTH || spritePyStart >= SURFACE_HEIGHT || srcW <= 0 || srcH <= 0) { return; } if (spritePxStart + srcW > SURFACE_WIDTH) { srcW = (int16_t)(SURFACE_WIDTH - spritePxStart); } if (spritePyStart + srcH > SURFACE_HEIGHT) { srcH = (int16_t)(SURFACE_HEIGHT - spritePyStart); } /* Phase 11 (shared-walker): byte-wide shifted draw. For each 8-px * column chunk of the sprite, build 4 plane bytes + a shape mask * via c2p, shift left/right by (x & 7), and AND-mask + OR the * results onto the dst plane bytes. ST's planar layout stores 4 * plane bytes interleaved at word granularity inside each 16-px * group: byte address = base + group*8 + plane*2 + half_byte * (where half_byte = dst_byte_col & 1). This replaces the old * per-pixel walker which was ~30 cycles/px = ~17K cycles/draw and * murdered the frame rate on shifted sprites (passenger walk, * flame, anything else not on an 8-px boundary). * * Phase / placement is driven from the ORIGINAL signed x so a * sub-byte left clip (x negative and not a multiple of 8) keeps * its true source-to-destination phase. shift = x & 7 picks the * low 3 bits of x (correct two's-complement modulo-8 even when x * is negative); startByteX = x >> 3 (arithmetic) is the signed * destination byte column for source byte column 0. Each source * byte column is walked in full and the per-byte guards drop the * off-surface portions, so the left byte and its right-carry spill * are clipped independently. */ { uint8_t shift = (uint8_t)(x & 7); int16_t startByteX = (int16_t)(x >> 3); int16_t byteCol; rowBase = pd->base + (uint16_t)spritePyStart * ST_BYTES_PER_ROW; for (row = 0; row < srcH; row++) { int16_t spritePy = (int16_t)(sy + row); int16_t tileY = (int16_t)(spritePy >> 3); int16_t inTileY = (int16_t)(spritePy & 7); for (byteCol = 0; byteCol < (int16_t)wTiles; byteCol++) { const uint8_t *tile = sp->tileData + (uint32_t)((tileY * wTiles + byteCol) * 32u); const uint8_t *chunky = tile + inTileY * 4u; /* c2p: 8 pixels (4 chunky bytes) -> 4 plane bytes + mask. */ uint8_t p0 = 0u, p1 = 0u, p2 = 0u, p3 = 0u, mask = 0u; uint8_t bit; for (bit = 0; bit < 8u; bit++) { uint8_t nibble = (bit & 1u) ? (uint8_t)(chunky[bit >> 1] & 0x0Fu) : (uint8_t)(chunky[bit >> 1] >> 4); uint8_t bm = (uint8_t)(0x80u >> bit); if (nibble != 0u) { mask |= bm; } if (nibble & 1u) { p0 |= bm; } if (nibble & 2u) { p1 |= bm; } if (nibble & 4u) { p2 |= bm; } if (nibble & 8u) { p3 |= bm; } } if (mask == 0u) { continue; } /* Compute ST address for the dst byte at byte_col_dst. * Plane bytes are interleaved word-wise inside groups of * 16 pixels = 8 bytes (4 planes x 2 bytes). The left byte * and the right-carry spill are guarded independently so * a negative dstByteCol (sub-byte left clip) still lands * its spill into on-surface byte dstByteCol+1. */ int16_t dstByteCol = (int16_t)(startByteX + byteCol); { uint8_t leftMask = (uint8_t)(mask >> shift); if (leftMask != 0u && dstByteCol >= 0 && dstByteCol < (int16_t)(SURFACE_WIDTH / 8)) { uint16_t group = (uint16_t)(dstByteCol >> 1); uint16_t half = (uint16_t)(dstByteCol & 1); uint8_t notM = (uint8_t)(~leftMask); uint8_t *p = rowBase + group * ST_BYTES_PER_GROUP + half; p[0] = (uint8_t)((p[0] & notM) | (uint8_t)(p0 >> shift)); p[2] = (uint8_t)((p[2] & notM) | (uint8_t)(p1 >> shift)); p[4] = (uint8_t)((p[4] & notM) | (uint8_t)(p2 >> shift)); p[6] = (uint8_t)((p[6] & notM) | (uint8_t)(p3 >> shift)); } if (shift != 0u) { uint8_t rshift = (uint8_t)(8u - shift); uint8_t rightMask = (uint8_t)(mask << rshift); int16_t rdc = (int16_t)(dstByteCol + 1); if (rightMask != 0u && rdc >= 0 && rdc < (int16_t)(SURFACE_WIDTH / 8)) { uint16_t rgroup = (uint16_t)(rdc >> 1); uint16_t rhalf = (uint16_t)(rdc & 1); uint8_t notM = (uint8_t)(~rightMask); uint8_t *p = rowBase + rgroup * ST_BYTES_PER_GROUP + rhalf; p[0] = (uint8_t)((p[0] & notM) | (uint8_t)(p0 << rshift)); p[2] = (uint8_t)((p[2] & notM) | (uint8_t)(p1 << rshift)); p[4] = (uint8_t)((p[4] & notM) | (uint8_t)(p2 << rshift)); p[6] = (uint8_t)((p[6] & notM) | (uint8_t)(p3 << rshift)); } } } } rowBase += ST_BYTES_PER_ROW; } } (void)col; } // Phase 10: hoist y*160 to per-row, fold setPixel/getPixel bodies // inline. Each pixel's group address differs only in (x), so we // can compute base+row*160 once per row and just do per-pixel // (group, bitMask, 4 plane RMW). 2x speedup over the per-pixel // stPlanarSetPixel form. void jlpSpriteSavePlanes(const jlSurfaceT *s, int16_t x, int16_t y, uint16_t w, uint16_t h, uint8_t *dstPlaneBytes) { const StPlanarT *pd; int16_t row; int16_t pair; int16_t pairs; uint8_t *pp; const uint8_t *rowBase; if (s == NULL || dstPlaneBytes == NULL || w == 0u || h == 0u) { return; } pd = (const StPlanarT *)s->portData; if (pd == NULL) { return; } /* NATIVE-PERF Phase 1: group-aligned raw-span format. When x AND w * are both 16-px multiples (the compiled save/restore window * shape) each row is a CONTIGUOUS span of interleaved-group bytes, * so the save is a straight per-row copy -- and, critically, the * SAME byte format the compiled emitters read and write (backup * format is a pure function of the recorded geometry; see * spriteInternal.h). This branch must run BEFORE the byte-aligned * branch: a group-aligned window previously took the byte-col asm, * whose byte ORDER differs -- save and restore switch formats * together, atomically, on the same predicate. */ if (((x | (int16_t)w) & 15) == 0 && x >= 0 && (x + (int16_t)w) <= SURFACE_WIDTH && y >= 0 && (y + (int16_t)h) <= SURFACE_HEIGHT) { uint16_t spanBytes = (uint16_t)(w >> 1); const uint8_t *src = pd->base + (uint16_t)y * ST_BYTES_PER_ROW + (uint16_t)((uint16_t)x >> 4) * ST_BYTES_PER_GROUP; uint8_t *dst = dstPlaneBytes; for (row = 0; row < (int16_t)h; row++) { memcpy(dst, src, (size_t)spanBytes); src += ST_BYTES_PER_ROW; dst += spanBytes; } return; } /* Phase 10.5 fast path: byte-aligned, fully on-surface. * Specialized 16x16 (the UBER ball-sprite size) skips the asm * walker's per-row col-init + col-loop-check overhead. Group- * aligned windows never reach here (raw-span above owns them); * this serves the x % 16 == 8 byte-aligned case. */ if ((x & 7) == 0 && (w & 7) == 0 && x >= 0 && (x + (int16_t)w) <= SURFACE_WIDTH && y >= 0 && (y + (int16_t)h) <= SURFACE_HEIGHT) { if (w == 16u && h == 16u) { surface68kStSprite16x16Save(pd->base, (uint16_t)x, (uint16_t)y, dstPlaneBytes); } else { surface68kStSpriteSaveByteAligned(pd->base, (uint16_t)x, (uint16_t)y, w, h, dstPlaneBytes); } return; } pairs = (int16_t)(w >> 1); rowBase = pd->base + (uint16_t)y * ST_BYTES_PER_ROW; for (row = 0; row < (int16_t)h; row++) { pp = &dstPlaneBytes[(uint16_t)row * (uint16_t)pairs]; for (pair = 0; pair < pairs; pair++) { int16_t px; uint16_t group; uint16_t bitMask; const uint16_t *pw; uint8_t hi; uint8_t lo; px = (int16_t)(x + pair * 2); group = (uint16_t)((uint16_t)px >> 4); bitMask = (uint16_t)(1u << (15u - ((uint16_t)px & 15u))); pw = (const uint16_t *)(rowBase + group * ST_BYTES_PER_GROUP); hi = 0u; if (pw[0] & bitMask) { hi = (uint8_t)(hi | 1u); } if (pw[1] & bitMask) { hi = (uint8_t)(hi | 2u); } if (pw[2] & bitMask) { hi = (uint8_t)(hi | 4u); } if (pw[3] & bitMask) { hi = (uint8_t)(hi | 8u); } px = (int16_t)(x + pair * 2 + 1); group = (uint16_t)((uint16_t)px >> 4); bitMask = (uint16_t)(1u << (15u - ((uint16_t)px & 15u))); pw = (const uint16_t *)(rowBase + group * ST_BYTES_PER_GROUP); lo = 0u; if (pw[0] & bitMask) { lo = (uint8_t)(lo | 1u); } if (pw[1] & bitMask) { lo = (uint8_t)(lo | 2u); } if (pw[2] & bitMask) { lo = (uint8_t)(lo | 4u); } if (pw[3] & bitMask) { lo = (uint8_t)(lo | 8u); } pp[pair] = (uint8_t)((hi << 4) | lo); } rowBase += ST_BYTES_PER_ROW; } } void jlpSpriteRestorePlanes(jlSurfaceT *s, int16_t x, int16_t y, uint16_t w, uint16_t h, const uint8_t *srcPlaneBytes) { StPlanarT *pd; int16_t row; int16_t pair; int16_t pairs; uint8_t b; const uint8_t *pp; uint8_t *rowBase; if (s == NULL || srcPlaneBytes == NULL || w == 0u || h == 0u) { return; } pd = (StPlanarT *)s->portData; if (pd == NULL) { return; } /* Group-aligned raw-span restore: MUST mirror the save-side branch * above exactly (same predicate, same byte order) -- see the save * comment. A raw-span backup restored through the byte-col asm (or * vice versa) silently scrambles planes. */ if (((x | (int16_t)w) & 15) == 0 && x >= 0 && (x + (int16_t)w) <= SURFACE_WIDTH && y >= 0 && (y + (int16_t)h) <= SURFACE_HEIGHT) { uint16_t spanBytes = (uint16_t)(w >> 1); uint8_t *dst = pd->base + (uint16_t)y * ST_BYTES_PER_ROW + (uint16_t)((uint16_t)x >> 4) * ST_BYTES_PER_GROUP; const uint8_t *src = srcPlaneBytes; for (row = 0; row < (int16_t)h; row++) { memcpy(dst, src, (size_t)spanBytes); dst += ST_BYTES_PER_ROW; src += spanBytes; } return; } /* Phase 10.5 fast path: byte-aligned, fully on-surface. * Specialized 16x16 (UBER ball-sprite) skips walker overhead. * Serves only the x % 16 == 8 case (raw-span above owns group- * aligned windows). */ if ((x & 7) == 0 && (w & 7) == 0 && x >= 0 && (x + (int16_t)w) <= SURFACE_WIDTH && y >= 0 && (y + (int16_t)h) <= SURFACE_HEIGHT) { if (w == 16u && h == 16u) { surface68kStSprite16x16Restore(pd->base, (uint16_t)x, (uint16_t)y, srcPlaneBytes); } else { surface68kStSpriteRestoreByteAligned(pd->base, (uint16_t)x, (uint16_t)y, w, h, srcPlaneBytes); } return; } pairs = (int16_t)(w >> 1); rowBase = pd->base + (uint16_t)y * ST_BYTES_PER_ROW; for (row = 0; row < (int16_t)h; row++) { pp = &srcPlaneBytes[(uint16_t)row * (uint16_t)pairs]; for (pair = 0; pair < pairs; pair++) { int16_t px; uint16_t group; uint16_t bitMask; uint16_t notMask; uint16_t *pw; uint8_t color; b = pp[pair]; px = (int16_t)(x + pair * 2); color = (uint8_t)(b >> 4); group = (uint16_t)((uint16_t)px >> 4); bitMask = (uint16_t)(1u << (15u - ((uint16_t)px & 15u))); notMask = (uint16_t)~bitMask; pw = (uint16_t *)(rowBase + group * ST_BYTES_PER_GROUP); if (color & 1u) { pw[0] = (uint16_t)(pw[0] | bitMask); } else { pw[0] = (uint16_t)(pw[0] & notMask); } if (color & 2u) { pw[1] = (uint16_t)(pw[1] | bitMask); } else { pw[1] = (uint16_t)(pw[1] & notMask); } if (color & 4u) { pw[2] = (uint16_t)(pw[2] | bitMask); } else { pw[2] = (uint16_t)(pw[2] & notMask); } if (color & 8u) { pw[3] = (uint16_t)(pw[3] | bitMask); } else { pw[3] = (uint16_t)(pw[3] & notMask); } px = (int16_t)(x + pair * 2 + 1); color = (uint8_t)(b & 0x0Fu); group = (uint16_t)((uint16_t)px >> 4); bitMask = (uint16_t)(1u << (15u - ((uint16_t)px & 15u))); notMask = (uint16_t)~bitMask; pw = (uint16_t *)(rowBase + group * ST_BYTES_PER_GROUP); if (color & 1u) { pw[0] = (uint16_t)(pw[0] | bitMask); } else { pw[0] = (uint16_t)(pw[0] & notMask); } if (color & 2u) { pw[1] = (uint16_t)(pw[1] | bitMask); } else { pw[1] = (uint16_t)(pw[1] & notMask); } if (color & 4u) { pw[2] = (uint16_t)(pw[2] | bitMask); } else { pw[2] = (uint16_t)(pw[2] & notMask); } if (color & 8u) { pw[3] = (uint16_t)(pw[3] | bitMask); } else { pw[3] = (uint16_t)(pw[3] & notMask); } } rowBase += ST_BYTES_PER_ROW; } } // ----- Flood-fill plane hooks (PERF-AUDIT #4) ----- // // floodFillInternal's planar walk-out and row-scan tiers, modeled on // the Amiga versions but against the word-interleaved layout. The // semantics contract with the core C fallback (bit-for-bit): // * walk: seed test, then extend left/right while pixels are // fillable. matchEqual: fillable = (pix == matchColor); bounded: // fillable = (pix != matchColor && pix != newColor). // * scan: markBuf[i] = 1 where row pixel (leftX + i) is fillable, // 0 otherwise, for i in [0 .. rightX - leftX]. // Both decode a 16-pixel group's 4 plane words ONCE into a stop/mark // bit mask, then test single register bits per pixel -- and the walk // skips a whole all-fillable group remainder in one step. bool jlpFloodWalkPlanes(const jlSurfaceT *s, int16_t startX, int16_t y, uint8_t matchColor, uint8_t newColor, bool matchEqual, bool *seedMatched, int16_t *leftXOut, int16_t *rightXOut) { StPlanarT *pd; const uint8_t *rowBase; uint16_t seedGroup; uint16_t seedStopBits; uint16_t curGroup; uint16_t stopBits; uint16_t rem; uint16_t g; int16_t leftX; int16_t rightX; int16_t prevX; int16_t nextX; pd = (StPlanarT *)s->portData; if (pd == NULL) { return false; } matchColor = (uint8_t)(matchColor & 0x0Fu); newColor = (uint8_t)(newColor & 0x0Fu); rowBase = pd->base + (uint16_t)y * ST_BYTES_PER_ROW; seedGroup = (uint16_t)((uint16_t)startX >> 4); seedStopBits = stFloodStopBits(rowBase, seedGroup, matchColor, newColor, matchEqual); if (seedStopBits & (uint16_t)(1u << (15u - ((uint16_t)startX & 15u)))) { *seedMatched = false; return true; } *seedMatched = true; // Walk left. Pixel x sits at bit (15 - (x & 15)), so moving left // means moving toward bit 15: shift the stop mask down so bit 0 is // the candidate pixel, then consume clear bits upward. leftX = startX; curGroup = seedGroup; stopBits = seedStopBits; while (leftX > 0) { prevX = (int16_t)(leftX - 1); g = (uint16_t)((uint16_t)prevX >> 4); if (g != curGroup) { curGroup = g; stopBits = stFloodStopBits(rowBase, g, matchColor, newColor, matchEqual); } rem = (uint16_t)(stopBits >> (15u - ((uint16_t)prevX & 15u))); if (rem == 0u) { // Everything from prevX to the group's first pixel is // fillable: take the rest of the group in one step. leftX = (int16_t)(g << 4); continue; } while ((rem & 1u) == 0u) { rem = (uint16_t)(rem >> 1); leftX--; } break; } // Walk right: mirror image, shifting the stop mask up so bit 15 is // the candidate pixel. rightX = startX; curGroup = seedGroup; stopBits = seedStopBits; while (rightX < SURFACE_WIDTH - 1) { nextX = (int16_t)(rightX + 1); g = (uint16_t)((uint16_t)nextX >> 4); if (g != curGroup) { curGroup = g; stopBits = stFloodStopBits(rowBase, g, matchColor, newColor, matchEqual); } rem = (uint16_t)(stopBits << ((uint16_t)nextX & 15u)); if (rem == 0u) { // Everything from nextX to the group's last pixel is // fillable: take the rest of the group in one step. rightX = (int16_t)((g << 4) + 15); continue; } while ((rem & 0x8000u) == 0u) { rem = (uint16_t)(rem << 1); rightX++; } break; } *leftXOut = leftX; *rightXOut = rightX; return true; } bool jlpFloodScanRowPlanes(const jlSurfaceT *s, int16_t leftX, int16_t rightX, int16_t scanY, uint8_t matchColor, uint8_t newColor, bool matchEqual, uint8_t *markBuf) { StPlanarT *pd; const uint8_t *rowBase; const uint16_t *gp; uint8_t *mb; uint16_t markBits; uint16_t bit; int16_t x; int16_t groupEnd; pd = (StPlanarT *)s->portData; if (pd == NULL) { return false; } matchColor = (uint8_t)(matchColor & 0x0Fu); newColor = (uint8_t)(newColor & 0x0Fu); rowBase = pd->base + (uint16_t)scanY * ST_BYTES_PER_ROW; mb = markBuf; x = leftX; while (x <= rightX) { gp = (const uint16_t *)(rowBase + (uint16_t)((uint16_t)x >> 4) * ST_BYTES_PER_GROUP); if (matchEqual) { markBits = stGroupEqBits(gp, matchColor); } else { markBits = (uint16_t)~(stGroupEqBits(gp, matchColor) | stGroupEqBits(gp, newColor)); } groupEnd = (int16_t)((uint16_t)x | 15u); if (groupEnd > rightX) { groupEnd = rightX; } bit = (uint16_t)(1u << (15u - ((uint16_t)x & 15u))); while (x <= groupEnd) { *mb = (uint8_t)((markBits & bit) ? 1u : 0u); mb++; bit = (uint16_t)(bit >> 1); x++; } } return true; } // Phase 7: pixel reader. Pre-Phase-9 reads from the chunky shadow // (s->pixels) since that's the source-of-truth during transition. // Once Phase 9 sets s->pixels = NULL the planar shadow becomes // authoritative and we walk the 4 plane bits at (x, y). uint8_t jlpSamplePixel(const jlSurfaceT *s, int16_t x, int16_t y) { if (s->pixels != NULL) { uint8_t byte = s->pixels[(uint16_t)y * SURFACE_BYTES_PER_ROW + ((uint16_t)x >> 1)]; if (x & 1) return (uint8_t)(byte & 0x0Fu); return (uint8_t)((byte & 0xF0u) >> 4); } { StPlanarT *pd = (StPlanarT *)s->portData; if (pd == NULL) { return 0u; } return stPlanarGetPixel(pd, x, y); } } // Derive 160 chunky bytes per row from the word-interleaved planar // buffer (20 groups x 4 plane words). Same shape as the Amiga's // amigaPlanesToChunkyRow but per-group instead of per-byte. Used by // jlpSurfaceHash to fold the planar surface into the same byte stream // the chunky ports hash, so cross-port hash comparisons stay valid. static void stPlanarToChunkyRow(const StPlanarT *pd, int16_t y, uint8_t *dstChunkyRow) { uint16_t group; uint16_t p; uint16_t bitMask; uint8_t pix; const uint16_t *gp; for (group = 0; group < ST_GROUPS_PER_ROW; group++) { gp = (const uint16_t *)(pd->base + (uint16_t)y * ST_BYTES_PER_ROW + group * ST_BYTES_PER_GROUP); for (p = 0; p < 16u; p++) { bitMask = (uint16_t)(1u << (15u - p)); pix = 0u; if (gp[0] & bitMask) { pix = (uint8_t)(pix | 1u); } if (gp[1] & bitMask) { pix = (uint8_t)(pix | 2u); } if (gp[2] & bitMask) { pix = (uint8_t)(pix | 4u); } if (gp[3] & bitMask) { pix = (uint8_t)(pix | 8u); } if ((p & 1u) == 0u) { dstChunkyRow[group * 8u + (p >> 1)] = (uint8_t)(pix << 4); } else { dstChunkyRow[group * 8u + (p >> 1)] = (uint8_t)(dstChunkyRow[group * 8u + (p >> 1)] | pix); } } } } uint32_t jlpSurfaceHash(const jlSurfaceT *s) { StPlanarT *pd; uint16_t lo = 0xACE1u; uint16_t hi = 0x1357u; uint16_t n; uint16_t v; int16_t row; uint16_t col; uint8_t b; uint8_t chunkyRow[SURFACE_BYTES_PER_ROW]; const uint16_t *w; pd = (StPlanarT *)s->portData; if (pd == NULL) { return 0u; } /* Pixel hash: derive equivalent chunky bytes from the planar * shadow row by row, fold them into the same SURFACE_HASH_MIX_BYTE * the chunky ports use so cross-port hash comparisons stay valid. */ for (row = 0; row < SURFACE_HEIGHT; row++) { stPlanarToChunkyRow(pd, row, chunkyRow); for (col = 0; col < SURFACE_BYTES_PER_ROW; col++) { b = chunkyRow[col]; SURFACE_HASH_MIX_BYTE(lo, hi, b); } } /* SCB + palette mix unchanged from chunky days. */ { const uint8_t *sp = s->scb; for (n = 0; n < (uint16_t)SURFACE_HEIGHT; n++) { b = *sp++; SURFACE_HASH_MIX_BYTE(lo, hi, b); } } w = &s->palette[0][0]; for (n = 0; n < (uint16_t)SURFACE_PALETTE_ENTRIES; n++) { v = *w++; b = (uint8_t)((v >> 8) & 0xFFu); SURFACE_HASH_MIX_BYTE(lo, hi, b); b = (uint8_t)(v & 0xFFu); SURFACE_HASH_MIX_BYTE(lo, hi, b); } return ((uint32_t)hi << 16) | (uint32_t)lo; } // On-disk format is the ST's native interleaved planar buffer; one // fread fills it directly, no chunky scratch or c2p step. bool jlpSurfaceLoadFile(jlSurfaceT *dst, FILE *fp) { StPlanarT *pd; pd = (StPlanarT *)dst->portData; if (pd == NULL) { return false; } return fread(pd->base, 1, ST_PLANAR_SIZE, fp) == ST_PLANAR_SIZE; } bool jlpSurfaceSaveFile(const jlSurfaceT *src, FILE *fp) { StPlanarT *pd; pd = (StPlanarT *)src->portData; if (pd == NULL) { return false; } return fwrite(pd->base, 1, ST_PLANAR_SIZE, fp) == ST_PLANAR_SIZE; } // Phase 9: no chunky storage on the ST. Cross-platform code treats // NULL as "port has no chunky shadow" (same contract Amiga uses). uint8_t *jlpSurfaceAllocPixels(void) { return NULL; } // Phase 9: stage has no chunky shadow either. Cross-platform stageAlloc // stores NULL in s->pixels and skips the chunky memset. uint8_t *jlpStageAllocPixels(void) { return NULL; }