// Uber demo: exercise every JoeyLib public API and measure throughput // of the per-frame-hot ones. Results are written to joeylog.txt via // jlLogF. A green screen on exit means the run completed. // // Timing model: each test aligns to a VBL boundary via jlWaitVBL, // records the starting jlFrameCount, then runs the op in a tight // loop polling jlFrameCount until UBER_FRAMES frames have elapsed. // Reported metric is ops/sec, computed as iters * jlFrameHz() / // UBER_FRAMES so results are directly comparable across ports // regardless of CPU speed or VBL rate. // // jlFrameCount is wall-clock-based per port, but READING it is not // free: on the IIgs it is a Misc Toolset GetTick call (several // hundred cycles per poll -- an earlier comment here claimed // ~10-30 cyc, which was wrong by more than an order of magnitude // and inflated every fast op's measured cost by a near-constant // ~600-800 cycles/iter). runForFrames therefore polls the clock // once per batch of UBER_BATCH op calls for sub-frame ops // (calibrated by the first call), amortizing the poll to noise. // // One-shot ops (jlSpriteCompile) get one call each, timed by frame // delta -- coarser but representative. #include #include #include #include #include // ----- Timing primitives ----- // 4-frame measurement window. Long enough that loop overhead doesn't // dominate; short enough to keep the full demo run under ~10 sec. /* 16 frames per timed op gives 4x the iter-count resolution of the * earlier 4-frame budget. Exposes the actual per-op cost on slow * ops where 4 frames produced the same iter count on different * framerates -- e.g. jlDrawCircle r=80 read as "4 iters / 4 frames" * on both 60 Hz IIgs (16.7 ms/frame, 67 ms window) and 50 Hz Amiga * (20 ms/frame, 80 ms window) even though per-op cost was equal, * just because 4 ops at 16-17 ms happen to fit both windows. The * 16-frame budget extends the windows to 267 ms / 320 ms; quantum * gap shrinks to ~6%. Total run time scales 4x (~80 sec each). */ #define UBER_FRAMES 16u // Op calls per clock poll for sub-frame ops. The unrolled batch in // runForFrames must contain exactly this many op() calls. Ops slower // than one frame poll every call instead, so a batch cannot overrun // the measurement window by more than one op. #define UBER_BATCH 16u // ----- IIgs progress mailbox (headless-MAME diagnosis) ----- // // The SHR SCB region covers rows 0..199 at $E1:9D00-$E1:9DC7; the bytes // at $E1:9DC8-$E1:9DFF are unused by hardware and by jlpPresent. UBER // mirrors its progress there so a MAME Lua probe can see WHERE a // headless run is (and whether the timing loop and the tick are alive) // even when the joeylog FST flush never lands on disk. Layout: // +0 (9DC8) uint8 current timed-op index (from timeOp); during // phase 5 it is 100 + the correctness-check number // +1 (9DC9) uint8 phase (UBER_PHASE_*) // +2 (9DCA) uint16 heartbeat (one increment per timing-loop pass) // +4 (9DCC) uint16 last jlFrameCount() value the timing loop saw // No-ops on every other port. #define UBER_PHASE_SETUP_SPRITE 1u #define UBER_PHASE_AUDIO_INIT 2u #define UBER_PHASE_SHOWCASE 3u #define UBER_PHASE_TIMED_RUN 4u #define UBER_PHASE_CHECKS 5u #define UBER_PHASE_DONE 6u #ifdef JOEYLIB_PLATFORM_IIGS #define UBER_MB_OP ((volatile uint8_t *)0xE19DC8L) #define UBER_MB_PHASE ((volatile uint8_t *)0xE19DC9L) #define UBER_MB_HEARTBEAT ((volatile uint16_t *)0xE19DCAL) #define UBER_MB_TICK ((volatile uint16_t *)0xE19DCCL) // One-time layout facts for probe correlation: +8 stage pointer value, // +12 address of uber.c's gStage static, +16 log-ring base address, // +20 log-ring head-counter address (see src/core/debug.c's IIgs RAM // ring -- the Lua probe drains the log live through these). #define UBER_MB_MAGIC ((volatile uint16_t *)0xE19DCEL) #define UBER_MB_STAGEPTR ((volatile uint32_t *)0xE19DD0L) #define UBER_MB_GSTAGEAT ((volatile uint32_t *)0xE19DD4L) #define UBER_MB_LOGRING ((volatile uint32_t *)0xE19DD8L) #define UBER_MB_LOGHEAD ((volatile uint32_t *)0xE19DDCL) // TEMP #84: the gSprite pointer value, poked after setupSprite so the // Lua probe can watch the struct bytes get overwritten in place. #define UBER_MB_SPRITEPTR ((volatile uint32_t *)0xE19DEAL) // "JL" -- the Lua probes refuse to trust the layout words until this // appears (the SCB region holds 0x80 fill before the app writes it). #define UBER_MB_MAGIC_VAL 0x4A4Cu extern uint8_t gJoeyLogRing[]; extern uint16_t gJoeyLogRingHead; #define uberMbOp(_i) (*UBER_MB_OP = (uint8_t)(_i)) #define uberMbPhase(_p) (*UBER_MB_PHASE = (uint8_t)(_p)) #define uberMbBeat() (*UBER_MB_HEARTBEAT = (uint16_t)(*UBER_MB_HEARTBEAT + 1u)) #define uberMbTick(_t) (*UBER_MB_TICK = (uint16_t)(_t)) #define uberMbLayout() (*UBER_MB_STAGEPTR = (uint32_t)gStage, *UBER_MB_GSTAGEAT = (uint32_t)&gStage, *UBER_MB_LOGRING = (uint32_t)&gJoeyLogRing[0], *UBER_MB_LOGHEAD = (uint32_t)&gJoeyLogRingHead, *UBER_MB_MAGIC = UBER_MB_MAGIC_VAL) #else #define uberMbOp(_i) ((void)0) #define uberMbPhase(_p) ((void)0) #define uberMbBeat() ((void)0) #define uberMbTick(_t) ((void)0) #define uberMbLayout() ((void)0) #endif typedef void (*OpFn)(void); static const char *gCurName = "(none)"; static jlSurfaceT *gStage = NULL; static jlSpriteT *gSprite = NULL; static jlSpriteBackupT gBackup; static unsigned char gBackupBytes[256]; static jlTileT gTileScratch; // Current timed-op number (1-based); mirrored into the mailbox. Lives // up here because runForFrames re-writes it every loop pass: the SHR // SCB upload in jlStagePresent can overwrite the adjacent mailbox // bytes, so a single write at timeOp start can be stomped mid-window. static uint8_t gOpIndex = 0; // Run `op` in a tight loop until `targetFrames` jlFrameCount ticks // have elapsed. Returns iterations completed. static unsigned long runForFrames(OpFn op, unsigned int targetFrames, uint16_t *actualFramesOut, uint32_t *millisDeltaOut) { unsigned long count; uint16_t startFrame; uint16_t endFrame; uint16_t now; uint32_t startMillis; uint32_t endMillis; bool slowOp; count = 0UL; jlWaitVBL(); startFrame = jlFrameCount(); startMillis = jlMillisElapsed(); // Calibrate: jlWaitVBL aligned us to a tick edge, so a sub-frame op // cannot see the counter advance during a single call. If the first // call DOES advance it, the op is slower than one frame -- poll the // clock every call so a batch cannot overrun the window. op(); count++; slowOp = ((uint16_t)(jlFrameCount() - startFrame) != 0u); for (;;) { now = jlFrameCount(); uberMbTick(now); uberMbBeat(); uberMbOp(gOpIndex); if ((uint16_t)(now - startFrame) >= targetFrames) { break; } if (slowOp) { op(); count++; } else { // Exactly UBER_BATCH calls between clock polls (the poll is // a toolbox call on IIgs; per-op polling dominated fast ops). op(); op(); op(); op(); op(); op(); op(); op(); op(); op(); op(); op(); op(); op(); op(); op(); count += UBER_BATCH; } } /* Capture the actual elapsed frames -- the last iter typically * overruns the target. Using actual instead of target as the * ops/sec divisor stays honest for ops slower than 1 frame * (where count is forced low while real time stretches well * past targetFrames). */ endFrame = jlFrameCount(); endMillis = jlMillisElapsed(); *actualFramesOut = (uint16_t)(endFrame - startFrame); if (*actualFramesOut == 0u) { *actualFramesOut = 1u; /* defensive: avoid div-by-zero */ } *millisDeltaOut = endMillis - startMillis; if (*millisDeltaOut == 0u) { *millisDeltaOut = 1u; /* defensive: avoid div-by-zero */ } return count; } // Time and log one op. Reports iters / N frames AND the derived // ops/sec so per-port results are directly comparable against IIgs // regardless of CPU speed or display refresh rate. Also logs an // FNV-1a hash of the surface state after timing -- this is the // pixel-perfect comparison input for the cross-port validation // harness (tools/diff-uber-hashes.py). Captured against IIgs as the // golden reference; planar 68k rewrites validate by matching it. static void timeOp(const char *name, OpFn op) { unsigned long iters; unsigned long opsPerSec; unsigned long opsPerSecFr; uint16_t actualFrames; uint32_t millisDelta; uint32_t hash; gCurName = name; gOpIndex++; uberMbOp(gOpIndex); iters = runForFrames(op, UBER_FRAMES, &actualFrames, &millisDelta); if (iters == 0UL) { jlLogF("UBER: %s: 0 iters (op too slow?)\n", name); return; } /* Report ops/sec from the MILLISECOND clock, not the frame counter. * jlMillisElapsed is refresh-independent on every port (ST Timer-C * 200 Hz, Amiga audio-ISR tick, DOS PIT, IIgs GetTick-derived so * identical to the old formula there), so it does not depend on * jlFrameHz() matching the emulator's real VBL rate -- the ST used * to hardcode 50 Hz while Hatari runs ~60, under-reporting ~17%. * The frame-derived rate is kept as the UBER-CLK cross-check: after * the jlpFrameHz fix the two agree; a divergence flags a clock bug. */ opsPerSec = (iters * 1000UL) / (unsigned long)millisDelta; opsPerSecFr = (iters * (unsigned long)jlFrameHz()) / (unsigned long)actualFrames; hash = jlSurfaceHash(gStage); jlLogF("UBER: %s: %lu iters / %u frames = %lu ops/sec | hash=%08lX\n", name, iters, actualFrames, opsPerSec, (unsigned long)hash); jlLogF("UBER-CLK: %s: frame=%lu millis=%lu ops/sec (%lu ms)\n", name, opsPerSecFr, opsPerSec, (unsigned long)millisDelta); } // ----- Test ops ----- static void op_drawPixel (void) { jlDrawPixel (gStage, 100, 100, 5); } static void op_drawLineH (void) { jlDrawLine (gStage, 0, 50, 319, 50, 5); } static void op_drawLineV (void) { jlDrawLine (gStage, 50, 0, 50, 199, 5); } static void op_drawLineDiag (void) { jlDrawLine (gStage, 0, 0, 319, 199, 5); } static void op_drawRect (void) { jlDrawRect (gStage, 10, 10, 100, 100, 5); } static void op_drawCircleSmall (void) { jlDrawCircle (gStage, 160, 100, 16, 5); } static void op_drawCircleLarge (void) { jlDrawCircle (gStage, 160, 100, 80, 5); } static void op_fillRectSmall (void) { jlFillRect (gStage, 20, 20, 16, 16, 7); } static void op_fillRectMid (void) { jlFillRect (gStage, 20, 20, 80, 80, 7); } static void op_fillRectFull (void) { jlFillRect (gStage, 0, 0, 320, 200, 7); } static void op_fillCircle (void) { jlFillCircle (gStage, 160, 100, 40, 7); } static void op_samplePixel (void) { (void)jlSamplePixel(gStage, 100, 100); } static void op_surfaceClear (void) { jlSurfaceClear (gStage, 0); } static void op_paletteSet(void) { static uint16_t colors[16] = { 0x000, 0xF00, 0x0F0, 0x00F, 0xFF0, 0xF0F, 0x0FF, 0xFFF, 0x800, 0x080, 0x008, 0x880, 0x808, 0x088, 0x888, 0x444 }; jlPaletteSet(gStage, 0, colors); } static void op_scbSetRange (void) { jlScbSetRange (gStage, 0, 199, 0); } static void op_tileFill (void) { jlTileFill (gStage, 5, 5, 7); } static void op_tileCopy (void) { jlTileCopy (gStage, 6, 6, gStage, 5, 5); } static void op_tileCopyMasked (void) { jlTileCopyMasked (gStage, 7, 7, gStage, 5, 5, 0); } static void op_tilePaste (void) { jlTilePaste (gStage, 8, 8, &gTileScratch); } static void op_tileSnap (void) { jlTileSnap (gStage, 5, 5, &gTileScratch); } static int16_t gSpriteX = 40; static int16_t gSpriteY = 30; static void op_spriteSave (void) { jlSpriteSaveUnder (gStage, gSprite, gSpriteX, gSpriteY, &gBackup); } static void op_spriteDraw (void) { jlSpriteDraw (gStage, gSprite, gSpriteX, gSpriteY); } static void op_spriteRestore (void) { jlSpriteRestoreUnder(gStage, &gBackup); } static void op_spriteSaveAndDraw (void) { jlSpriteSaveAndDraw (gStage, gSprite, gSpriteX, gSpriteY, &gBackup); } static void op_stagePresent (void) { jlStagePresent(); } static void op_inputPoll (void) { jlInputPoll(); } static void op_keyDown (void) { (void)jlKeyDown(KEY_A); } static void op_keyPressed (void) { (void)jlKeyPressed(KEY_A); } static void op_mouseX (void) { (void)jlMouseX(); } static void op_joyConnected (void) { (void)jlJoystickConnected(JOYSTICK_1); } static void op_audioFrameTick (void) { jlAudioFrameTick(); } static void op_audioIsPlaying (void) { (void)jlAudioIsPlayingMod(); } static void op_surfaceMarkDirty(void) { /* jlDrawPixel already marks; use fill instead */ jlFillRect(gStage, 0, 0, 32, 32, 0); } // ----- Build the ball sprite procedurally ----- #define BALL_TILES_X 2 #define BALL_TILES_Y 2 #define BALL_TILE_BYTES (BALL_TILES_X * BALL_TILES_Y * 32u) static const uint8_t gBallAuthored[16 * 8] = { 0x00, 0x00, 0x22, 0x22, 0x22, 0x22, 0x00, 0x00, 0x00, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x00, 0x02, 0x22, 0x32, 0x22, 0x22, 0x22, 0x22, 0x20, 0x02, 0x23, 0x32, 0x22, 0x22, 0x22, 0x22, 0x20, 0x22, 0x33, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x02, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x20, 0x02, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x20, 0x00, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x00, 0x00, 0x00, 0x22, 0x22, 0x22, 0x22, 0x00, 0x00, 0x00, 0x00, 0x00, 0x22, 0x22, 0x00, 0x00, 0x00 }; static uint8_t gBallTiles[BALL_TILE_BYTES]; static void buildBallSprite(void) { uint16_t tx; uint16_t ty; uint16_t row; uint16_t b; uint8_t *dst; for (ty = 0; ty < BALL_TILES_Y; ty++) { for (tx = 0; tx < BALL_TILES_X; tx++) { dst = &gBallTiles[(ty * BALL_TILES_X + tx) * 32u]; for (row = 0; row < 8; row++) { for (b = 0; b < 4; b++) { dst[row * 4 + b] = gBallAuthored[((ty * 8) + row) * 8 + (tx * 4) + b]; } } } } } // ----- Visual showcase ----- // // Before the (visually meaningless) timed benchmark, draw one example // of each primitive into its own grid cell so a viewer can actually SEE // what the library renders instead of a single solid color. The // benchmark that follows never presents, so this showcase stays on // screen for the whole timed run; results go to joeylog.txt. A legend // mapping cell index -> primitive is also logged. #define SC_SCREEN_W 320 #define SC_SCREEN_H 200 #define SC_PAL_COUNT 16 #define SC_PAL_SWATCH_W (SC_SCREEN_W / SC_PAL_COUNT) #define SC_PAL_STRIP_H 12 #define SC_GRID_TOP 16 #define SC_COLS 4 #define SC_ROWS 3 #define SC_CELLS (SC_COLS * SC_ROWS) #define SC_GUTTER 4 #define SC_INSET 5 #define SC_BG_COLOR 10 #define SC_BORDER_COLOR 1 #define SC_HOLD_FRAMES 210 static const uint16_t gShowcasePal[SC_PAL_COUNT] = { 0x000, 0xFFF, 0xF00, 0x0F0, 0x00F, 0xFF0, 0x0FF, 0xF0F, 0xF80, 0xAAA, 0x555, 0x8AF, 0x8F8, 0xF8C, 0x840, 0xACE }; static void showcaseCellRect(uint16_t index, int16_t *outX, int16_t *outY, int16_t *outW, int16_t *outH) { uint16_t col; uint16_t row; int16_t cellW; int16_t cellH; col = (uint16_t)(index % SC_COLS); row = (uint16_t)(index / SC_COLS); cellW = (int16_t)((SC_SCREEN_W - (SC_COLS + 1) * SC_GUTTER) / SC_COLS); cellH = (int16_t)((SC_SCREEN_H - SC_GRID_TOP - (SC_ROWS + 1) * SC_GUTTER) / SC_ROWS); *outX = (int16_t)(SC_GUTTER + (int16_t)col * (cellW + SC_GUTTER)); *outY = (int16_t)(SC_GRID_TOP + SC_GUTTER + (int16_t)row * (cellH + SC_GUTTER)); *outW = cellW; *outH = cellH; } // optnone: this 12-case showcase has too many simultaneously-live locals for // the w65816 register allocator at -O2 (it bails "ran out of registers"). It // renders once at startup, so dropping optimization here costs nothing. // optnone is a clang-only attribute; GCC (DOS/Amiga/ST) rejects it under // -Werror=attributes and does not need it, so guard it to clang. #if defined(__clang__) #define DRAWSHOWCASE_ATTR __attribute__((noinline, optnone)) #else #define DRAWSHOWCASE_ATTR __attribute__((noinline)) #endif static void DRAWSHOWCASE_ATTR drawShowcase(void) { uint16_t cell; int16_t x; int16_t y; int16_t w; int16_t h; int16_t ix; int16_t iy; int16_t iw; int16_t ih; int16_t cx; int16_t cy; int16_t r; int16_t k; int16_t px; int16_t py; int16_t minDim; uint8_t bx; uint8_t by; uint8_t bxStart; uint8_t bxEnd; uint8_t byStart; uint8_t byEnd; uint16_t held; uberMbPhase(UBER_PHASE_SHOWCASE); jlPaletteSet(gStage, 0, gShowcasePal); jlScbSetRange(gStage, 0, 199, 0); jlSurfaceClear(gStage, SC_BG_COLOR); // Top strip: all 16 palette entries as swatches. for (k = 0; k < SC_PAL_COUNT; k++) { jlFillRect(gStage, (int16_t)(k * SC_PAL_SWATCH_W), 0, SC_PAL_SWATCH_W, SC_PAL_STRIP_H, (uint8_t)k); } for (cell = 0; cell < SC_CELLS; cell++) { showcaseCellRect(cell, &x, &y, &w, &h); jlDrawRect(gStage, x, y, (uint16_t)w, (uint16_t)h, SC_BORDER_COLOR); ix = (int16_t)(x + SC_INSET); iy = (int16_t)(y + SC_INSET); iw = (int16_t)(w - 2 * SC_INSET); ih = (int16_t)(h - 2 * SC_INSET); cx = (int16_t)(ix + iw / 2); cy = (int16_t)(iy + ih / 2); minDim = (iw < ih) ? iw : ih; switch (cell) { case 0: // Pixels: a scatter of single plotted pixels. for (py = iy; py < iy + ih; py += 3) { for (px = ix; px < ix + iw; px += 3) { jlDrawPixel(gStage, px, py, (uint8_t)(2 + ((px + py) % 14))); } } break; case 1: // Horizontal lines. for (k = 0; k < ih; k += 4) { jlDrawLine(gStage, ix, (int16_t)(iy + k), (int16_t)(ix + iw - 1), (int16_t)(iy + k), (uint8_t)(2 + (k / 4) % 14)); } break; case 2: // Vertical lines. for (k = 0; k < iw; k += 4) { jlDrawLine(gStage, (int16_t)(ix + k), iy, (int16_t)(ix + k), (int16_t)(iy + ih - 1), (uint8_t)(2 + (k / 4) % 14)); } break; case 3: // Diagonals: an X plus a fan from the center. jlDrawLine(gStage, ix, iy, (int16_t)(ix + iw - 1), (int16_t)(iy + ih - 1), 5); jlDrawLine(gStage, ix, (int16_t)(iy + ih - 1), (int16_t)(ix + iw - 1), iy, 6); for (k = 0; k < iw; k += 8) { jlDrawLine(gStage, cx, cy, (int16_t)(ix + k), iy, (uint8_t)(8 + (k / 8) % 8)); } break; case 4: // Rectangle outlines, concentric. for (k = 0; 2 * k < minDim - 4; k += 5) { jlDrawRect(gStage, (int16_t)(ix + k), (int16_t)(iy + k), (uint16_t)(iw - 2 * k), (uint16_t)(ih - 2 * k), (uint8_t)(2 + (k / 5) % 14)); } break; case 5: // Filled rectangles, overlapping. jlFillRect(gStage, ix, iy, (uint16_t)(iw * 2 / 3), (uint16_t)(ih * 2 / 3), 2); jlFillRect(gStage, (int16_t)(ix + iw / 3), (int16_t)(iy + ih / 3), (uint16_t)(iw * 2 / 3), (uint16_t)(ih * 2 / 3), 4); break; case 6: // Circle outlines, concentric. for (r = (int16_t)(minDim / 2); r > 2; r -= 4) { jlDrawCircle(gStage, cx, cy, (uint16_t)r, (uint8_t)(2 + (r / 4) % 14)); } break; case 7: // Filled circles. jlFillCircle(gStage, cx, cy, (uint16_t)(minDim / 2 - 1), 8); jlFillCircle(gStage, cx, cy, (uint16_t)(minDim / 4), 5); break; case 8: // Tiles: an 8x8-block checkerboard inside the cell. bxStart = (uint8_t)((ix + 7) / 8); bxEnd = (uint8_t)((ix + iw) / 8); byStart = (uint8_t)((iy + 7) / 8); byEnd = (uint8_t)((iy + ih) / 8); for (by = byStart; by < byEnd; by++) { for (bx = bxStart; bx < bxEnd; bx++) { jlTileFill(gStage, bx, by, (uint8_t)(((bx + by) & 1) ? 6 : 8)); } } break; case 9: // Sprite: the compiled ball at a few positions. jlSpriteDraw(gStage, gSprite, ix, iy); jlSpriteDraw(gStage, gSprite, (int16_t)(ix + iw - 16), (int16_t)(iy + ih - 16)); jlSpriteDraw(gStage, gSprite, (int16_t)(cx - 8), (int16_t)(cy - 8)); break; case 10: // Flood fill: outline a circle, then flood its interior. jlDrawCircle(gStage, cx, cy, (uint16_t)(minDim / 2 - 2), 1); jlFloodFill(gStage, cx, cy, 12); break; default: // Mini scene: ground, sun, horizon, ball. jlFillRect(gStage, ix, (int16_t)(iy + ih / 2), (uint16_t)iw, (uint16_t)(ih - ih / 2), 4); jlFillCircle(gStage, cx, (int16_t)(iy + ih / 3), (uint16_t)(ih / 4), 5); jlDrawLine(gStage, ix, (int16_t)(iy + ih / 2), (int16_t)(ix + iw - 1), (int16_t)(iy + ih / 2), 1); jlSpriteDraw(gStage, gSprite, (int16_t)(cx - 8), (int16_t)(iy + ih / 2 - 16)); break; } } jlLogF("UBER: showcase cells: 0=pixels 1=lineH 2=lineV 3=diag 4=rect 5=fillRect 6=circle 7=fillCircle 8=tiles 9=sprite 10=flood 11=scene\n"); jlStagePresent(); // Hold the showcase on screen, then auto-advance to the benchmark so // the headless perf-capture run still completes without a keypress. held = jlFrameCount(); while ((uint16_t)(jlFrameCount() - held) < SC_HOLD_FRAMES) { /* hold */ } } // ----- Non-timed correctness checks (Phase 0 verification harness) ----- // // Each check draws a deterministic scene and logs the surface hash so // tools/diff-uber-hashes can compare ports against each other and against // the frozen golden logs. Several checks intentionally capture the CURRENT // behavior of known bugs (PERF-AUDIT.md #1, #11, #12, #72): their hash // lines are EXPECTED to change when the Phase 1 fixes land -- re-golden // exactly those lines then, nothing else. // // PASS/FAIL lines assert invariants that must hold on every port both // before and after the fixes (round-trips, forced palette color 0, the // PRNG golden sequence, arena bookkeeping). static void __attribute__((noinline)) chkHash(const char *name) { jlLogF("UBER-CHK: %s: hash=%08lX\n", name, (unsigned long)jlSurfaceHash(gStage)); } static void __attribute__((noinline)) chkPassFail(const char *name, bool pass) { jlLogF("UBER-CHK: %s: %s\n", name, pass ? "PASS" : "FAIL"); } // Edge-coordinate draws. The off-surface pixel coords stay modest (within // ~200 rows of the surface) so the pre-fix dirty-band overwrite (finding #1, // non-IIgs) lands inside the paired band arrays instead of unrelated // globals; Phase 1 turns these into true no-ops. static void __attribute__((noinline)) checkEdgeDraws(void) { uberMbOp(101); jlDrawPixel(gStage, -5, 100, 5); jlDrawPixel(gStage, 330, 100, 5); jlDrawPixel(gStage, 100, -3, 5); jlDrawPixel(gStage, 100, 210, 5); chkHash("edge-pixels"); // w=65535 draws phantom edges today (finding #11); h==2 exercises the // zero-height interior-edge fills; the third rect clips normally. jlDrawRect(gStage, 10, 10, 65535u, 100, 5); jlDrawRect(gStage, 50, 20, 30, 2, 6); jlDrawRect(gStage, -10, 150, 340, 40, 7); chkHash("edge-drawRect"); // r=300 clips every span; r=40000 should cover the surface but draws // nothing today (finding #12). jlFillCircle(gStage, 160, 100, 300, 4); jlFillCircle(gStage, 160, 100, 40000u, 9); jlDrawCircle(gStage, 10, 10, 40000u, 3); chkHash("edge-circles"); // Far-off-surface endpoints. The H line should fill the whole visible // row but draws nothing today (finding #72: the H/V fast path clamps // span to 320 BEFORE clipping, so a huge span anchored off-surface // clips away entirely). The diagonal clips per-pixel and is correct. jlDrawLine(gStage, -20000, 50, 20000, 50, 2); jlDrawLine(gStage, 60, -20000, 60, 20000, 2); jlDrawLine(gStage, -300, -200, 620, 400, 8); chkHash("edge-lines"); } // Clipped sprite draws take the interpreted path on every port (the // compiled routines require fully-on-surface): the ONLY cross-port // coverage of that path, which UBER's timed ops (on-surface, compiled) // never touch. Also asserts the save/draw/restore round-trip restores // the exact pre-save pixels at a clipped position. static void __attribute__((noinline)) checkClippedSprites(void) { uint32_t before; uberMbOp(102); jlSpriteDraw(gStage, gSprite, -8, 50); jlSpriteDraw(gStage, gSprite, 312, 50); jlSpriteDraw(gStage, gSprite, 150, -8); jlSpriteDraw(gStage, gSprite, 150, 192); jlSpriteDraw(gStage, gSprite, -8, -8); jlSpriteDraw(gStage, gSprite, 400, 100); chkHash("sprite-clipped"); before = jlSurfaceHash(gStage); jlSpriteSaveUnder(gStage, gSprite, -8, 100, &gBackup); jlSpriteDraw(gStage, gSprite, -8, 100); jlSpriteRestoreUnder(gStage, &gBackup); chkPassFail("sprite-clip-roundtrip", jlSurfaceHash(gStage) == before); } static void __attribute__((noinline)) checkTileMonoFlood(void) { static jlTileT mono; uint8_t i; uberMbOp(103); for (i = 0; i < TILE_BYTES; i++) { mono.pixels[i] = (uint8_t)((i & 1) ? 0x0F : 0xF0); } jlTilePasteMono(gStage, 10, 10, &mono, 5, 9); jlTilePasteMono(gStage, 11, 10, &mono, 14, 0); chkHash("tilePasteMono"); jlDrawRect(gStage, 240, 150, 40, 30, 1); jlFloodFill(gStage, 250, 160, 12); chkHash("floodFill"); } static void __attribute__((noinline)) checkDrawText(void) { static uint16_t asciiMap[256]; uint16_t i; uberMbOp(104); for (i = 0; i < 256u; i++) { asciiMap[i] = TILE_NO_GLYPH; } // 'A' -> tile (5,5), 'B' -> tile (6,6). Seed both glyph tiles here -- // the section-opening surface clear wiped whatever the timed tile ops // left there. jlTileFill(gStage, 5, 5, 9); jlTileFill(gStage, 6, 6, 3); asciiMap['A'] = (uint16_t)(5u | (5u << 8)); asciiMap['B'] = (uint16_t)(6u | (6u << 8)); jlDrawText(gStage, 2, 20, gStage, asciiMap, "ABBA"); chkHash("drawText"); // Out-of-range start: the entry sanitation loop (Phase 5, #22) // reproduces the historical absorb-and-wrap semantics bit-for-bit // (the first glyph is consumed without drawing, the second wraps // to (0,21) and draws) while guaranteeing jlpTileCopyMasked never // sees an out-of-range destination. Separate hash line so any // future semantic change stays isolated. jlDrawText(gStage, 200, 20, gStage, asciiMap, "AB"); chkHash("drawText-offgrid"); } static void __attribute__((noinline)) checkPalette(void) { static const uint16_t conforming[16] = { 0x0ABC, 0x0111, 0x0222, 0x0333, 0x0444, 0x0555, 0x0666, 0x0777, 0x0888, 0x0999, 0x0AAA, 0x0BBB, 0x0CCC, 0x0DDD, 0x0EEE, 0x0123 }; uint16_t readBack[16]; uint16_t i; bool ok; // Contract invariants (stable across the Phase 2 #16 change): color 0 // is forced to $000 even when the caller passes nonzero, and // conforming $0RGB entries 1..15 round-trip exactly. uberMbOp(105); jlPaletteSet(gStage, 2, conforming); jlPaletteGet(gStage, 2, readBack); ok = (readBack[0] == 0x0000u); for (i = 1; i < 16u; i++) { if (readBack[i] != conforming[i]) { ok = false; } } chkPassFail("palette-roundtrip", ok); } static void __attribute__((noinline)) checkRandom(void) { static const uint32_t expected[4] = { 0x87985AA5UL, 0x155B24A3UL, 0x4820F4C4UL, 0x81B3AC98UL }; uint16_t i; bool ok; // xorshift32 golden sequence from a fixed seed; bit-identical on // every port and across the Phase 8 (#43) IIgs rewrite. uberMbOp(106); jlRandomSeed(0x12345678UL); ok = true; for (i = 0; i < 4u; i++) { if (jlRandom() != expected[i]) { ok = false; } } if (jlRandomRange(100u) != 43u) { ok = false; } chkPassFail("random-golden", ok); jlRandomSeed(1u); } // Arena churn: create/compile/destroy so a hole opens and is reused, then // compact and assert the used-byte counter returns to its pre-churn value. // Covers the codegen allocator paths UBER's single long-lived sprite never // exercises (findings #5, #34 land here in Phase 1). Also runs the // owned-tileData path (jlSpriteCreateFromSurface) that finding #31 // re-allocates in Phase 1, and a sprite-bank load if an asset is present. static void __attribute__((noinline)) checkAllocator(void) { jlSpriteT *a; jlSpriteT *b; jlSpriteT *c; uint32_t usedBefore; // Sub-markers 111+ pinpoint the statement that hangs (Phase 1 // stabilization; the coarse marker froze on this check). uberMbOp(111); usedBefore = jlSpriteCodegenBytesUsed(); a = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y); b = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y); if (a == NULL || b == NULL) { chkPassFail("arena-churn", false); return; } uberMbOp(112); (void)jlSpriteCompile(a); (void)jlSpriteCompile(b); uberMbOp(113); jlSpriteDestroy(a); uberMbOp(114); c = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y); if (c == NULL) { jlSpriteDestroy(b); chkPassFail("arena-churn", false); return; } (void)jlSpriteCompile(c); uberMbOp(115); jlSpriteDestroy(b); jlSpriteDestroy(c); uberMbOp(116); jlSpriteCompact(); uberMbOp(117); chkPassFail("arena-churn", jlSpriteCodegenBytesUsed() == usedBefore); uberMbOp(118); a = jlSpriteCreateFromSurface(gStage, 0, 0, 2, 2); if (a != NULL) { (void)jlSpriteCompile(a); uberMbOp(119); jlSpriteDraw(gStage, a, 280, 20); jlSpriteDestroy(a); } chkPassFail("sprite-from-surface", a != NULL); uberMbOp(120); chkHash("sprite-from-surface-draw"); } // Coverage the primary checkTileMonoFlood golden misses (found in the // #3/#4 re-analysis): the alternating 0x0F/0xF0 mono pattern only // selects IIgs #3 combo indices 1 and 2, and jlFloodFill only drives // the matchEqual branch of the ST #4 plane hooks. Runs LAST so it // perturbs no earlier cumulative full-stage hash. static void __attribute__((noinline)) checkTileMonoFloodExtra(void) { // 0x00 -> combo idx 0 (bg:bg), 0xFF -> idx 3 (fg:fg), 0x0F -> idx 1 // (bg:fg), 0xF0 -> idx 2 (fg:bg): all four opacity pairs in one tile. static const uint8_t comboBytes[4] = { 0x00u, 0xFFu, 0x0Fu, 0xF0u }; static jlTileT monoAll; uint8_t i; uberMbOp(121); for (i = 0; i < TILE_BYTES; i++) { monoAll.pixels[i] = comboBytes[i & 3u]; } jlTilePasteMono(gStage, 10, 12, &monoAll, 5, 9); jlTilePasteMono(gStage, 11, 12, &monoAll, 14, 0); chkHash("tilePasteMonoAll"); // Bounded flood: an enclosed color-3 box over a pre-cleared interior // so the fill stays contained and drives the bounded stop mask // (pix == boundary || pix == new) plus the planar group-skip across // several 16px groups. matchColor = boundaryColor = 3, matchEqual = 0. jlFillRect(gStage, 40, 140, 64, 40, 0); jlDrawRect(gStage, 44, 144, 52, 32, 3); jlFloodFillBounded(gStage, 70, 160, 7, 3); chkHash("floodFillBounded"); } static void __attribute__((noinline)) runCorrectnessChecks(void) { uberMbPhase(UBER_PHASE_CHECKS); jlLogF("UBER-CHK: ----- begin -----\n"); jlSurfaceClear(gStage, 0); checkEdgeDraws(); checkClippedSprites(); checkTileMonoFlood(); checkDrawText(); checkPalette(); checkRandom(); checkAllocator(); checkTileMonoFloodExtra(); // jlShutdown -> jlInit -> stale-sprite-destroy (#34) needs a teardown // of the whole library mid-run; it gets a dedicated micro-example in // Phase 1 rather than risking the benchmark's display state here. jlLogF("UBER-CHK: ----- end -----\n"); } // ----- Main ----- static void __attribute__((noinline)) runAllTests(void) { uberMbPhase(UBER_PHASE_TIMED_RUN); jlLogF("UBER: ----- begin -----\n"); // Surface / palette / SCB. timeOp("jlSurfaceClear", op_surfaceClear); timeOp("jlPaletteSet", op_paletteSet); timeOp("jlScbSetRange", op_scbSetRange); // Drawing primitives. timeOp("jlDrawPixel", op_drawPixel); timeOp("jlDrawLine H", op_drawLineH); timeOp("jlDrawLine V", op_drawLineV); timeOp("jlDrawLine diag", op_drawLineDiag); timeOp("jlDrawRect 100x100", op_drawRect); timeOp("jlDrawCircle r=16", op_drawCircleSmall); timeOp("jlDrawCircle r=80", op_drawCircleLarge); timeOp("jlFillRect 16x16", op_fillRectSmall); timeOp("jlFillRect 80x80", op_fillRectMid); timeOp("jlFillRect 320x200", op_fillRectFull); timeOp("jlFillCircle r=40", op_fillCircle); timeOp("jlSamplePixel", op_samplePixel); // Tiles. Seed scratch tile + dest cells with non-zero pixels first. jlFillRect(gStage, 0, 0, 320, 64, 7); jlTileSnap(gStage, 5, 5, &gTileScratch); timeOp("jlTileFill", op_tileFill); timeOp("jlTileCopy", op_tileCopy); timeOp("jlTileCopyMasked", op_tileCopyMasked); timeOp("jlTilePaste", op_tilePaste); timeOp("jlTileSnap", op_tileSnap); // Sprites. Background must be non-empty so save-under has work // to do (otherwise it's a 4 KB memset of zeros, atypical). jlSurfaceClear(gStage, 4); timeOp("jlSpriteSaveUnder", op_spriteSave); timeOp("jlSpriteDraw", op_spriteDraw); timeOp("jlSpriteRestoreUnder", op_spriteRestore); timeOp("jlSpriteSaveAndDraw", op_spriteSaveAndDraw); // Present. One warm-up call before each timed loop primes any // per-port one-time setup (Amiga: copper list rebuild after the // jlPaletteSet / jlScbSetRange tests dirty the cache; without warm-up // the rebuild's MakeScreen + MrgCop + WaitTOF chain consumes the // entire 4-frame measurement window) so we measure steady-state // throughput rather than first-call penalty. jlStagePresent(); timeOp("jlStagePresent full", op_stagePresent); // Input. timeOp("jlInputPoll", op_inputPoll); timeOp("jlKeyDown", op_keyDown); timeOp("jlKeyPressed", op_keyPressed); timeOp("jlMouseX", op_mouseX); timeOp("joeyJoyConnected", op_joyConnected); // Audio. timeOp("jlAudioFrameTick", op_audioFrameTick); timeOp("jlAudioIsPlayingMod", op_audioIsPlaying); // Surface mark dirty (via jlFillRect's mark step). timeOp("surfaceMarkDirtyRect (via jlFillRect 32x32)", op_surfaceMarkDirty); jlLogF("UBER: ----- end -----\n"); } // Extracted from main so main's register pressure stays under the w65816 // allocator's ceiling -- the 16-entry pal[] + loop index was the overflow. // noinline is load-bearing at -O2 (the backend would otherwise re-inline a // single-call static and recreate the pressure). static void __attribute__((noinline)) setupPalette(void) { uint16_t pal[16]; int i; for (i = 0; i < 16; i++) { pal[i] = (uint16_t)((i << 8) | (i << 4) | i); // grey ramp } pal[ 0] = 0x000; pal[ 1] = 0x800; // dark red (running) pal[ 2] = 0x080; // green (done) pal[ 3] = 0x008; // blue pal[ 5] = 0xFF0; // yellow (test pixels) pal[ 7] = 0xFFF; // white (fills) pal[15] = 0xF00; // red jlPaletteSet(gStage, 0, pal); jlScbSetRange(gStage, 0, 199, 0); } // Extracted from main for the same register-pressure reason as setupPalette. static void __attribute__((noinline)) reportElapsed(uint16_t startFrame) { uint16_t endFrame; uint16_t elapsedFrames; unsigned long elapsedMs; endFrame = jlFrameCount(); elapsedFrames = (uint16_t)(endFrame - startFrame); elapsedMs = ((unsigned long)elapsedFrames * 1000UL) / (unsigned long)jlFrameHz(); jlLogF("UBER: total wall time: %lu ms (%u frames @ %u Hz)\n", elapsedMs, elapsedFrames, (unsigned)jlFrameHz()); } // Extracted from main (register pressure). Returns false on sprite-create fail. static bool __attribute__((noinline)) setupSprite(void) { uint16_t before; uberMbPhase(UBER_PHASE_SETUP_SPRITE); buildBallSprite(); gSprite = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y); if (gSprite == NULL) { jlLog("UBER: jlSpriteCreate failed"); return false; } // jlSpriteCompile is a one-shot. Time at frame resolution. jlWaitVBL(); before = jlFrameCount(); if (!jlSpriteCompile(gSprite)) { jlLog("UBER: jlSpriteCompile failed"); } while (jlFrameCount() == before) { /* wait for next VBL edge */ } jlLogF("UBER: jlSpriteCompile: 1 call in <= 1 frame\n"); gBackup.bytes = gBackupBytes; return true; } // Extracted from main (register pressure): the jlConfigT struct on main's // frame, on top of the rest of main, pushed the w65816 allocator over its // limit. noinline keeps it out at -O2. static bool __attribute__((noinline)) initJoeyLib(void) { jlConfigT config; /* 32 KB fits the 8 pre-shifted DRAW variants the Amiga planar * compiled sprite emitter generates. UL on the multiply because * a 16-bit int overflows on 32 * 1024. */ config.codegenBytes = 32UL * 1024; config.audioBytes = 64UL * 1024; return jlInit(&config); } int main(void) { uint16_t startFrame; if (!initJoeyLib()) { return 1; } /* jlFrameCount is VBL-driven, so it only ticks after halInit * installed its VBL ISR -- captured here is "everything from now * to press-any-key". Pre-init setup time is small and not the * cost the user is chasing; runAllTests dominates. */ startFrame = jlFrameCount(); gStage = jlStageGet(); if (gStage == NULL) { jlShutdown(); return 1; } uberMbLayout(); // A simple visible palette so users see SOMETHING during the run. setupPalette(); // Indicate "running": red bar at top of screen. jlSurfaceClear(gStage, 0); jlFillRect(gStage, 0, 0, 320, 8, 1); jlStagePresent(); if (!setupSprite()) { jlShutdown(); return 1; } #ifdef JOEYLIB_PLATFORM_IIGS *UBER_MB_SPRITEPTR = (uint32_t)gSprite; #endif // Audio: only init/shutdown is exercised. Triggering jlAudioPlaySfx // without first calling jlAudioPlayMod leaves NTP's engine in a // half-initialized state -- NTPstreamsound is designed to OVERLAY on // an already-running module. Without NTPprepare/NTPplay first, the // streamer oscillator is fired but no music tick ever advances or // silences it, and you get a stuck high-pitched scream. UBER doesn't // ship a MOD asset, so we skip the SFX exercise. The frame-tick and // isPlayingMod calls below still get timed (both are no-op fast // paths on IIgs). uberMbPhase(UBER_PHASE_AUDIO_INIT); if (jlAudioInit()) { jlLogF("UBER: audioInit OK\n"); } else { jlLogF("UBER: audioInit failed (skipping audio)\n"); } // Visual showcase: render one of each primitive into its own grid // cell so the run shows something legible (the timed benchmark below // never presents, so this stays on screen throughout it). The first // timed op is jlSurfaceClear, so this leaves the benchmark untouched. drawShowcase(); runAllTests(); runCorrectnessChecks(); reportElapsed(startFrame); // Done. Green screen + waitForKey. uberMbPhase(UBER_PHASE_DONE); jlSurfaceClear(gStage, 2); jlStagePresent(); jlLogF("UBER: press any key to exit\n"); // Flush the log to disk BEFORE the blocking key wait, so an automated // (headless) run that kills the process at this prompt still captures // the results -- joeyLog otherwise only flushes at the atexit fclose. jlLogFlush(); jlWaitForAnyKey(); jlSpriteDestroy(gSprite); jlShutdown(); return 0; }