joeylib2/examples/uber/uber.c

1059 lines
41 KiB
C

// Uber demo: exercise every JoeyLib public API and measure throughput
// of the per-frame-hot ones. Results are written to joeylog.txt via
// jlLogF. A green screen on exit means the run completed.
//
// Timing model: each test aligns to a VBL boundary via jlWaitVBL,
// records the starting jlFrameCount, then runs the op in a tight
// loop polling jlFrameCount until UBER_FRAMES frames have elapsed.
// Reported metric is ops/sec, computed as iters * jlFrameHz() /
// UBER_FRAMES so results are directly comparable across ports
// regardless of CPU speed or VBL rate.
//
// jlFrameCount is wall-clock-based per port, but READING it is not
// free: on the IIgs it is a Misc Toolset GetTick call (several
// hundred cycles per poll -- an earlier comment here claimed
// ~10-30 cyc, which was wrong by more than an order of magnitude
// and inflated every fast op's measured cost by a near-constant
// ~600-800 cycles/iter). runForFrames therefore polls the clock
// once per batch of UBER_BATCH op calls for sub-frame ops
// (calibrated by the first call), amortizing the poll to noise.
//
// One-shot ops (jlSpriteCompile) get one call each, timed by frame
// delta -- coarser but representative.
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <stddef.h>
#include <joey/joey.h>
// ----- Timing primitives -----
// 4-frame measurement window. Long enough that loop overhead doesn't
// dominate; short enough to keep the full demo run under ~10 sec.
/* 16 frames per timed op gives 4x the iter-count resolution of the
* earlier 4-frame budget. Exposes the actual per-op cost on slow
* ops where 4 frames produced the same iter count on different
* framerates -- e.g. jlDrawCircle r=80 read as "4 iters / 4 frames"
* on both 60 Hz IIgs (16.7 ms/frame, 67 ms window) and 50 Hz Amiga
* (20 ms/frame, 80 ms window) even though per-op cost was equal,
* just because 4 ops at 16-17 ms happen to fit both windows. The
* 16-frame budget extends the windows to 267 ms / 320 ms; quantum
* gap shrinks to ~6%. Total run time scales 4x (~80 sec each). */
#define UBER_FRAMES 16u
// Op calls per clock poll for sub-frame ops. The unrolled batch in
// runForFrames must contain exactly this many op() calls. Ops slower
// than one frame poll every call instead, so a batch cannot overrun
// the measurement window by more than one op.
#define UBER_BATCH 16u
// ----- IIgs progress mailbox (headless-MAME diagnosis) -----
//
// The SHR SCB region covers rows 0..199 at $E1:9D00-$E1:9DC7; the bytes
// at $E1:9DC8-$E1:9DFF are unused by hardware and by jlpPresent. UBER
// mirrors its progress there so a MAME Lua probe can see WHERE a
// headless run is (and whether the timing loop and the tick are alive)
// even when the joeylog FST flush never lands on disk. Layout:
// +0 (9DC8) uint8 current timed-op index (from timeOp); during
// phase 5 it is 100 + the correctness-check number
// +1 (9DC9) uint8 phase (UBER_PHASE_*)
// +2 (9DCA) uint16 heartbeat (one increment per timing-loop pass)
// +4 (9DCC) uint16 last jlFrameCount() value the timing loop saw
// No-ops on every other port.
#define UBER_PHASE_SETUP_SPRITE 1u
#define UBER_PHASE_AUDIO_INIT 2u
#define UBER_PHASE_SHOWCASE 3u
#define UBER_PHASE_TIMED_RUN 4u
#define UBER_PHASE_CHECKS 5u
#define UBER_PHASE_DONE 6u
#ifdef JOEYLIB_PLATFORM_IIGS
#define UBER_MB_OP ((volatile uint8_t *)0xE19DC8L)
#define UBER_MB_PHASE ((volatile uint8_t *)0xE19DC9L)
#define UBER_MB_HEARTBEAT ((volatile uint16_t *)0xE19DCAL)
#define UBER_MB_TICK ((volatile uint16_t *)0xE19DCCL)
// One-time layout facts for probe correlation: +8 stage pointer value,
// +12 address of uber.c's gStage static, +16 log-ring base address,
// +20 log-ring head-counter address (see src/core/debug.c's IIgs RAM
// ring -- the Lua probe drains the log live through these).
#define UBER_MB_MAGIC ((volatile uint16_t *)0xE19DCEL)
#define UBER_MB_STAGEPTR ((volatile uint32_t *)0xE19DD0L)
#define UBER_MB_GSTAGEAT ((volatile uint32_t *)0xE19DD4L)
#define UBER_MB_LOGRING ((volatile uint32_t *)0xE19DD8L)
#define UBER_MB_LOGHEAD ((volatile uint32_t *)0xE19DDCL)
// TEMP #84: the gSprite pointer value, poked after setupSprite so the
// Lua probe can watch the struct bytes get overwritten in place.
#define UBER_MB_SPRITEPTR ((volatile uint32_t *)0xE19DEAL)
// "JL" -- the Lua probes refuse to trust the layout words until this
// appears (the SCB region holds 0x80 fill before the app writes it).
#define UBER_MB_MAGIC_VAL 0x4A4Cu
extern uint8_t gJoeyLogRing[];
extern uint16_t gJoeyLogRingHead;
#define uberMbOp(_i) (*UBER_MB_OP = (uint8_t)(_i))
#define uberMbPhase(_p) (*UBER_MB_PHASE = (uint8_t)(_p))
#define uberMbBeat() (*UBER_MB_HEARTBEAT = (uint16_t)(*UBER_MB_HEARTBEAT + 1u))
#define uberMbTick(_t) (*UBER_MB_TICK = (uint16_t)(_t))
#define uberMbLayout() (*UBER_MB_STAGEPTR = (uint32_t)gStage, *UBER_MB_GSTAGEAT = (uint32_t)&gStage, *UBER_MB_LOGRING = (uint32_t)&gJoeyLogRing[0], *UBER_MB_LOGHEAD = (uint32_t)&gJoeyLogRingHead, *UBER_MB_MAGIC = UBER_MB_MAGIC_VAL)
#else
#define uberMbOp(_i) ((void)0)
#define uberMbPhase(_p) ((void)0)
#define uberMbBeat() ((void)0)
#define uberMbTick(_t) ((void)0)
#define uberMbLayout() ((void)0)
#endif
typedef void (*OpFn)(void);
static const char *gCurName = "(none)";
static jlSurfaceT *gStage = NULL;
static jlSpriteT *gSprite = NULL;
static jlSpriteBackupT gBackup;
static unsigned char gBackupBytes[256];
static jlTileT gTileScratch;
// Current timed-op number (1-based); mirrored into the mailbox. Lives
// up here because runForFrames re-writes it every loop pass: the SHR
// SCB upload in jlStagePresent can overwrite the adjacent mailbox
// bytes, so a single write at timeOp start can be stomped mid-window.
static uint8_t gOpIndex = 0;
// Run `op` in a tight loop until `targetFrames` jlFrameCount ticks
// have elapsed. Returns iterations completed.
static unsigned long runForFrames(OpFn op, unsigned int targetFrames, uint16_t *actualFramesOut, uint32_t *millisDeltaOut) {
unsigned long count;
uint16_t startFrame;
uint16_t endFrame;
uint16_t now;
uint32_t startMillis;
uint32_t endMillis;
bool slowOp;
count = 0UL;
jlWaitVBL();
startFrame = jlFrameCount();
startMillis = jlMillisElapsed();
// Calibrate: jlWaitVBL aligned us to a tick edge, so a sub-frame op
// cannot see the counter advance during a single call. If the first
// call DOES advance it, the op is slower than one frame -- poll the
// clock every call so a batch cannot overrun the window.
op();
count++;
slowOp = ((uint16_t)(jlFrameCount() - startFrame) != 0u);
for (;;) {
now = jlFrameCount();
uberMbTick(now);
uberMbBeat();
uberMbOp(gOpIndex);
if ((uint16_t)(now - startFrame) >= targetFrames) {
break;
}
if (slowOp) {
op();
count++;
} else {
// Exactly UBER_BATCH calls between clock polls (the poll is
// a toolbox call on IIgs; per-op polling dominated fast ops).
op();
op();
op();
op();
op();
op();
op();
op();
op();
op();
op();
op();
op();
op();
op();
op();
count += UBER_BATCH;
}
}
/* Capture the actual elapsed frames -- the last iter typically
* overruns the target. Using actual instead of target as the
* ops/sec divisor stays honest for ops slower than 1 frame
* (where count is forced low while real time stretches well
* past targetFrames). */
endFrame = jlFrameCount();
endMillis = jlMillisElapsed();
*actualFramesOut = (uint16_t)(endFrame - startFrame);
if (*actualFramesOut == 0u) {
*actualFramesOut = 1u; /* defensive: avoid div-by-zero */
}
*millisDeltaOut = endMillis - startMillis;
if (*millisDeltaOut == 0u) {
*millisDeltaOut = 1u; /* defensive: avoid div-by-zero */
}
return count;
}
// Time and log one op. Reports iters / N frames AND the derived
// ops/sec so per-port results are directly comparable against IIgs
// regardless of CPU speed or display refresh rate. Also logs an
// FNV-1a hash of the surface state after timing -- this is the
// pixel-perfect comparison input for the cross-port validation
// harness (tools/diff-uber-hashes.py). Captured against IIgs as the
// golden reference; planar 68k rewrites validate by matching it.
static void timeOp(const char *name, OpFn op) {
unsigned long iters;
unsigned long opsPerSec;
unsigned long opsPerSecFr;
uint16_t actualFrames;
uint32_t millisDelta;
uint32_t hash;
gCurName = name;
gOpIndex++;
uberMbOp(gOpIndex);
iters = runForFrames(op, UBER_FRAMES, &actualFrames, &millisDelta);
if (iters == 0UL) {
jlLogF("UBER: %s: 0 iters (op too slow?)\n", name);
return;
}
/* Report ops/sec from the MILLISECOND clock, not the frame counter.
* jlMillisElapsed is refresh-independent on every port (ST Timer-C
* 200 Hz, Amiga audio-ISR tick, DOS PIT, IIgs GetTick-derived so
* identical to the old formula there), so it does not depend on
* jlFrameHz() matching the emulator's real VBL rate -- the ST used
* to hardcode 50 Hz while Hatari runs ~60, under-reporting ~17%.
* The frame-derived rate is kept as the UBER-CLK cross-check: after
* the jlpFrameHz fix the two agree; a divergence flags a clock bug. */
opsPerSec = (iters * 1000UL) / (unsigned long)millisDelta;
opsPerSecFr = (iters * (unsigned long)jlFrameHz()) / (unsigned long)actualFrames;
hash = jlSurfaceHash(gStage);
jlLogF("UBER: %s: %lu iters / %u frames = %lu ops/sec | hash=%08lX\n",
name, iters, actualFrames, opsPerSec, (unsigned long)hash);
jlLogF("UBER-CLK: %s: frame=%lu millis=%lu ops/sec (%lu ms)\n",
name, opsPerSecFr, opsPerSec, (unsigned long)millisDelta);
}
// ----- Test ops -----
static void op_drawPixel (void) { jlDrawPixel (gStage, 100, 100, 5); }
static void op_drawLineH (void) { jlDrawLine (gStage, 0, 50, 319, 50, 5); }
static void op_drawLineV (void) { jlDrawLine (gStage, 50, 0, 50, 199, 5); }
static void op_drawLineDiag (void) { jlDrawLine (gStage, 0, 0, 319, 199, 5); }
static void op_drawRect (void) { jlDrawRect (gStage, 10, 10, 100, 100, 5); }
static void op_drawCircleSmall (void) { jlDrawCircle (gStage, 160, 100, 16, 5); }
static void op_drawCircleLarge (void) { jlDrawCircle (gStage, 160, 100, 80, 5); }
static void op_fillRectSmall (void) { jlFillRect (gStage, 20, 20, 16, 16, 7); }
static void op_fillRectMid (void) { jlFillRect (gStage, 20, 20, 80, 80, 7); }
static void op_fillRectFull (void) { jlFillRect (gStage, 0, 0, 320, 200, 7); }
static void op_fillCircle (void) { jlFillCircle (gStage, 160, 100, 40, 7); }
static void op_samplePixel (void) { (void)jlSamplePixel(gStage, 100, 100); }
static void op_surfaceClear (void) { jlSurfaceClear (gStage, 0); }
static void op_paletteSet(void) {
static uint16_t colors[16] = {
0x000, 0xF00, 0x0F0, 0x00F, 0xFF0, 0xF0F, 0x0FF, 0xFFF,
0x800, 0x080, 0x008, 0x880, 0x808, 0x088, 0x888, 0x444
};
jlPaletteSet(gStage, 0, colors);
}
static void op_scbSetRange (void) { jlScbSetRange (gStage, 0, 199, 0); }
static void op_tileFill (void) { jlTileFill (gStage, 5, 5, 7); }
static void op_tileCopy (void) { jlTileCopy (gStage, 6, 6, gStage, 5, 5); }
static void op_tileCopyMasked (void) { jlTileCopyMasked (gStage, 7, 7, gStage, 5, 5, 0); }
static void op_tilePaste (void) { jlTilePaste (gStage, 8, 8, &gTileScratch); }
static void op_tileSnap (void) { jlTileSnap (gStage, 5, 5, &gTileScratch); }
static int16_t gSpriteX = 40;
static int16_t gSpriteY = 30;
static void op_spriteSave (void) { jlSpriteSaveUnder (gStage, gSprite, gSpriteX, gSpriteY, &gBackup); }
static void op_spriteDraw (void) { jlSpriteDraw (gStage, gSprite, gSpriteX, gSpriteY); }
static void op_spriteRestore (void) { jlSpriteRestoreUnder(gStage, &gBackup); }
static void op_spriteSaveAndDraw (void) { jlSpriteSaveAndDraw (gStage, gSprite, gSpriteX, gSpriteY, &gBackup); }
static void op_stagePresent (void) { jlStagePresent(); }
static void op_inputPoll (void) { jlInputPoll(); }
static void op_keyDown (void) { (void)jlKeyDown(KEY_A); }
static void op_keyPressed (void) { (void)jlKeyPressed(KEY_A); }
static void op_mouseX (void) { (void)jlMouseX(); }
static void op_joyConnected (void) { (void)jlJoystickConnected(JOYSTICK_1); }
static void op_audioFrameTick (void) { jlAudioFrameTick(); }
static void op_audioIsPlaying (void) { (void)jlAudioIsPlayingMod(); }
static void op_surfaceMarkDirty(void) { /* jlDrawPixel already marks; use fill instead */
jlFillRect(gStage, 0, 0, 32, 32, 0); }
// ----- Build the ball sprite procedurally -----
#define BALL_TILES_X 2
#define BALL_TILES_Y 2
#define BALL_TILE_BYTES (BALL_TILES_X * BALL_TILES_Y * 32u)
static const uint8_t gBallAuthored[16 * 8] = {
0x00, 0x00, 0x22, 0x22, 0x22, 0x22, 0x00, 0x00,
0x00, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x00,
0x02, 0x22, 0x32, 0x22, 0x22, 0x22, 0x22, 0x20,
0x02, 0x23, 0x32, 0x22, 0x22, 0x22, 0x22, 0x20,
0x22, 0x33, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
0x02, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x20,
0x02, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x20,
0x00, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x00,
0x00, 0x00, 0x22, 0x22, 0x22, 0x22, 0x00, 0x00,
0x00, 0x00, 0x00, 0x22, 0x22, 0x00, 0x00, 0x00
};
static uint8_t gBallTiles[BALL_TILE_BYTES];
static void buildBallSprite(void) {
uint16_t tx;
uint16_t ty;
uint16_t row;
uint16_t b;
uint8_t *dst;
for (ty = 0; ty < BALL_TILES_Y; ty++) {
for (tx = 0; tx < BALL_TILES_X; tx++) {
dst = &gBallTiles[(ty * BALL_TILES_X + tx) * 32u];
for (row = 0; row < 8; row++) {
for (b = 0; b < 4; b++) {
dst[row * 4 + b] =
gBallAuthored[((ty * 8) + row) * 8 + (tx * 4) + b];
}
}
}
}
}
// ----- Visual showcase -----
//
// Before the (visually meaningless) timed benchmark, draw one example
// of each primitive into its own grid cell so a viewer can actually SEE
// what the library renders instead of a single solid color. The
// benchmark that follows never presents, so this showcase stays on
// screen for the whole timed run; results go to joeylog.txt. A legend
// mapping cell index -> primitive is also logged.
#define SC_SCREEN_W 320
#define SC_SCREEN_H 200
#define SC_PAL_COUNT 16
#define SC_PAL_SWATCH_W (SC_SCREEN_W / SC_PAL_COUNT)
#define SC_PAL_STRIP_H 12
#define SC_GRID_TOP 16
#define SC_COLS 4
#define SC_ROWS 3
#define SC_CELLS (SC_COLS * SC_ROWS)
#define SC_GUTTER 4
#define SC_INSET 5
#define SC_BG_COLOR 10
#define SC_BORDER_COLOR 1
#define SC_HOLD_FRAMES 210
static const uint16_t gShowcasePal[SC_PAL_COUNT] = {
0x000, 0xFFF, 0xF00, 0x0F0, 0x00F, 0xFF0, 0x0FF, 0xF0F,
0xF80, 0xAAA, 0x555, 0x8AF, 0x8F8, 0xF8C, 0x840, 0xACE
};
static void showcaseCellRect(uint16_t index, int16_t *outX, int16_t *outY, int16_t *outW, int16_t *outH) {
uint16_t col;
uint16_t row;
int16_t cellW;
int16_t cellH;
col = (uint16_t)(index % SC_COLS);
row = (uint16_t)(index / SC_COLS);
cellW = (int16_t)((SC_SCREEN_W - (SC_COLS + 1) * SC_GUTTER) / SC_COLS);
cellH = (int16_t)((SC_SCREEN_H - SC_GRID_TOP - (SC_ROWS + 1) * SC_GUTTER) / SC_ROWS);
*outX = (int16_t)(SC_GUTTER + (int16_t)col * (cellW + SC_GUTTER));
*outY = (int16_t)(SC_GRID_TOP + SC_GUTTER + (int16_t)row * (cellH + SC_GUTTER));
*outW = cellW;
*outH = cellH;
}
// optnone: this 12-case showcase has too many simultaneously-live locals for
// the w65816 register allocator at -O2 (it bails "ran out of registers"). It
// renders once at startup, so dropping optimization here costs nothing.
// optnone is a clang-only attribute; GCC (DOS/Amiga/ST) rejects it under
// -Werror=attributes and does not need it, so guard it to clang.
#if defined(__clang__)
#define DRAWSHOWCASE_ATTR __attribute__((noinline, optnone))
#else
#define DRAWSHOWCASE_ATTR __attribute__((noinline))
#endif
static void DRAWSHOWCASE_ATTR drawShowcase(void) {
uint16_t cell;
int16_t x;
int16_t y;
int16_t w;
int16_t h;
int16_t ix;
int16_t iy;
int16_t iw;
int16_t ih;
int16_t cx;
int16_t cy;
int16_t r;
int16_t k;
int16_t px;
int16_t py;
int16_t minDim;
uint8_t bx;
uint8_t by;
uint8_t bxStart;
uint8_t bxEnd;
uint8_t byStart;
uint8_t byEnd;
uint16_t held;
uberMbPhase(UBER_PHASE_SHOWCASE);
jlPaletteSet(gStage, 0, gShowcasePal);
jlScbSetRange(gStage, 0, 199, 0);
jlSurfaceClear(gStage, SC_BG_COLOR);
// Top strip: all 16 palette entries as swatches.
for (k = 0; k < SC_PAL_COUNT; k++) {
jlFillRect(gStage, (int16_t)(k * SC_PAL_SWATCH_W), 0, SC_PAL_SWATCH_W, SC_PAL_STRIP_H, (uint8_t)k);
}
for (cell = 0; cell < SC_CELLS; cell++) {
showcaseCellRect(cell, &x, &y, &w, &h);
jlDrawRect(gStage, x, y, (uint16_t)w, (uint16_t)h, SC_BORDER_COLOR);
ix = (int16_t)(x + SC_INSET);
iy = (int16_t)(y + SC_INSET);
iw = (int16_t)(w - 2 * SC_INSET);
ih = (int16_t)(h - 2 * SC_INSET);
cx = (int16_t)(ix + iw / 2);
cy = (int16_t)(iy + ih / 2);
minDim = (iw < ih) ? iw : ih;
switch (cell) {
case 0:
// Pixels: a scatter of single plotted pixels.
for (py = iy; py < iy + ih; py += 3) {
for (px = ix; px < ix + iw; px += 3) {
jlDrawPixel(gStage, px, py, (uint8_t)(2 + ((px + py) % 14)));
}
}
break;
case 1:
// Horizontal lines.
for (k = 0; k < ih; k += 4) {
jlDrawLine(gStage, ix, (int16_t)(iy + k), (int16_t)(ix + iw - 1), (int16_t)(iy + k), (uint8_t)(2 + (k / 4) % 14));
}
break;
case 2:
// Vertical lines.
for (k = 0; k < iw; k += 4) {
jlDrawLine(gStage, (int16_t)(ix + k), iy, (int16_t)(ix + k), (int16_t)(iy + ih - 1), (uint8_t)(2 + (k / 4) % 14));
}
break;
case 3:
// Diagonals: an X plus a fan from the center.
jlDrawLine(gStage, ix, iy, (int16_t)(ix + iw - 1), (int16_t)(iy + ih - 1), 5);
jlDrawLine(gStage, ix, (int16_t)(iy + ih - 1), (int16_t)(ix + iw - 1), iy, 6);
for (k = 0; k < iw; k += 8) {
jlDrawLine(gStage, cx, cy, (int16_t)(ix + k), iy, (uint8_t)(8 + (k / 8) % 8));
}
break;
case 4:
// Rectangle outlines, concentric.
for (k = 0; 2 * k < minDim - 4; k += 5) {
jlDrawRect(gStage, (int16_t)(ix + k), (int16_t)(iy + k), (uint16_t)(iw - 2 * k), (uint16_t)(ih - 2 * k), (uint8_t)(2 + (k / 5) % 14));
}
break;
case 5:
// Filled rectangles, overlapping.
jlFillRect(gStage, ix, iy, (uint16_t)(iw * 2 / 3), (uint16_t)(ih * 2 / 3), 2);
jlFillRect(gStage, (int16_t)(ix + iw / 3), (int16_t)(iy + ih / 3), (uint16_t)(iw * 2 / 3), (uint16_t)(ih * 2 / 3), 4);
break;
case 6:
// Circle outlines, concentric.
for (r = (int16_t)(minDim / 2); r > 2; r -= 4) {
jlDrawCircle(gStage, cx, cy, (uint16_t)r, (uint8_t)(2 + (r / 4) % 14));
}
break;
case 7:
// Filled circles.
jlFillCircle(gStage, cx, cy, (uint16_t)(minDim / 2 - 1), 8);
jlFillCircle(gStage, cx, cy, (uint16_t)(minDim / 4), 5);
break;
case 8:
// Tiles: an 8x8-block checkerboard inside the cell.
bxStart = (uint8_t)((ix + 7) / 8);
bxEnd = (uint8_t)((ix + iw) / 8);
byStart = (uint8_t)((iy + 7) / 8);
byEnd = (uint8_t)((iy + ih) / 8);
for (by = byStart; by < byEnd; by++) {
for (bx = bxStart; bx < bxEnd; bx++) {
jlTileFill(gStage, bx, by, (uint8_t)(((bx + by) & 1) ? 6 : 8));
}
}
break;
case 9:
// Sprite: the compiled ball at a few positions.
jlSpriteDraw(gStage, gSprite, ix, iy);
jlSpriteDraw(gStage, gSprite, (int16_t)(ix + iw - 16), (int16_t)(iy + ih - 16));
jlSpriteDraw(gStage, gSprite, (int16_t)(cx - 8), (int16_t)(cy - 8));
break;
case 10:
// Flood fill: outline a circle, then flood its interior.
jlDrawCircle(gStage, cx, cy, (uint16_t)(minDim / 2 - 2), 1);
jlFloodFill(gStage, cx, cy, 12);
break;
default:
// Mini scene: ground, sun, horizon, ball.
jlFillRect(gStage, ix, (int16_t)(iy + ih / 2), (uint16_t)iw, (uint16_t)(ih - ih / 2), 4);
jlFillCircle(gStage, cx, (int16_t)(iy + ih / 3), (uint16_t)(ih / 4), 5);
jlDrawLine(gStage, ix, (int16_t)(iy + ih / 2), (int16_t)(ix + iw - 1), (int16_t)(iy + ih / 2), 1);
jlSpriteDraw(gStage, gSprite, (int16_t)(cx - 8), (int16_t)(iy + ih / 2 - 16));
break;
}
}
jlLogF("UBER: showcase cells: 0=pixels 1=lineH 2=lineV 3=diag 4=rect 5=fillRect 6=circle 7=fillCircle 8=tiles 9=sprite 10=flood 11=scene\n");
jlStagePresent();
// Hold the showcase on screen, then auto-advance to the benchmark so
// the headless perf-capture run still completes without a keypress.
held = jlFrameCount();
while ((uint16_t)(jlFrameCount() - held) < SC_HOLD_FRAMES) {
/* hold */
}
}
// ----- Non-timed correctness checks (Phase 0 verification harness) -----
//
// Each check draws a deterministic scene and logs the surface hash so
// tools/diff-uber-hashes can compare ports against each other and against
// the frozen golden logs. Several checks intentionally capture the CURRENT
// behavior of known bugs (PERF-AUDIT.md #1, #11, #12, #72): their hash
// lines are EXPECTED to change when the Phase 1 fixes land -- re-golden
// exactly those lines then, nothing else.
//
// PASS/FAIL lines assert invariants that must hold on every port both
// before and after the fixes (round-trips, forced palette color 0, the
// PRNG golden sequence, arena bookkeeping).
static void __attribute__((noinline)) chkHash(const char *name) {
jlLogF("UBER-CHK: %s: hash=%08lX\n", name, (unsigned long)jlSurfaceHash(gStage));
}
static void __attribute__((noinline)) chkPassFail(const char *name, bool pass) {
jlLogF("UBER-CHK: %s: %s\n", name, pass ? "PASS" : "FAIL");
}
// Edge-coordinate draws. The off-surface pixel coords stay modest (within
// ~200 rows of the surface) so the pre-fix dirty-band overwrite (finding #1,
// non-IIgs) lands inside the paired band arrays instead of unrelated
// globals; Phase 1 turns these into true no-ops.
static void __attribute__((noinline)) checkEdgeDraws(void) {
uberMbOp(101);
jlDrawPixel(gStage, -5, 100, 5);
jlDrawPixel(gStage, 330, 100, 5);
jlDrawPixel(gStage, 100, -3, 5);
jlDrawPixel(gStage, 100, 210, 5);
chkHash("edge-pixels");
// w=65535 draws phantom edges today (finding #11); h==2 exercises the
// zero-height interior-edge fills; the third rect clips normally.
jlDrawRect(gStage, 10, 10, 65535u, 100, 5);
jlDrawRect(gStage, 50, 20, 30, 2, 6);
jlDrawRect(gStage, -10, 150, 340, 40, 7);
chkHash("edge-drawRect");
// r=300 clips every span; r=40000 should cover the surface but draws
// nothing today (finding #12).
jlFillCircle(gStage, 160, 100, 300, 4);
jlFillCircle(gStage, 160, 100, 40000u, 9);
jlDrawCircle(gStage, 10, 10, 40000u, 3);
chkHash("edge-circles");
// Far-off-surface endpoints. The H line should fill the whole visible
// row but draws nothing today (finding #72: the H/V fast path clamps
// span to 320 BEFORE clipping, so a huge span anchored off-surface
// clips away entirely). The diagonal clips per-pixel and is correct.
jlDrawLine(gStage, -20000, 50, 20000, 50, 2);
jlDrawLine(gStage, 60, -20000, 60, 20000, 2);
jlDrawLine(gStage, -300, -200, 620, 400, 8);
chkHash("edge-lines");
}
// Clipped sprite draws take the interpreted path on every port (the
// compiled routines require fully-on-surface): the ONLY cross-port
// coverage of that path, which UBER's timed ops (on-surface, compiled)
// never touch. Also asserts the save/draw/restore round-trip restores
// the exact pre-save pixels at a clipped position.
static void __attribute__((noinline)) checkClippedSprites(void) {
uint32_t before;
uberMbOp(102);
jlSpriteDraw(gStage, gSprite, -8, 50);
jlSpriteDraw(gStage, gSprite, 312, 50);
jlSpriteDraw(gStage, gSprite, 150, -8);
jlSpriteDraw(gStage, gSprite, 150, 192);
jlSpriteDraw(gStage, gSprite, -8, -8);
jlSpriteDraw(gStage, gSprite, 400, 100);
chkHash("sprite-clipped");
before = jlSurfaceHash(gStage);
jlSpriteSaveUnder(gStage, gSprite, -8, 100, &gBackup);
jlSpriteDraw(gStage, gSprite, -8, 100);
jlSpriteRestoreUnder(gStage, &gBackup);
chkPassFail("sprite-clip-roundtrip", jlSurfaceHash(gStage) == before);
}
static void __attribute__((noinline)) checkTileMonoFlood(void) {
static jlTileT mono;
uint8_t i;
uberMbOp(103);
for (i = 0; i < TILE_BYTES; i++) {
mono.pixels[i] = (uint8_t)((i & 1) ? 0x0F : 0xF0);
}
jlTilePasteMono(gStage, 10, 10, &mono, 5, 9);
jlTilePasteMono(gStage, 11, 10, &mono, 14, 0);
chkHash("tilePasteMono");
jlDrawRect(gStage, 240, 150, 40, 30, 1);
jlFloodFill(gStage, 250, 160, 12);
chkHash("floodFill");
}
static void __attribute__((noinline)) checkDrawText(void) {
static uint16_t asciiMap[256];
uint16_t i;
uberMbOp(104);
for (i = 0; i < 256u; i++) {
asciiMap[i] = TILE_NO_GLYPH;
}
// 'A' -> tile (5,5), 'B' -> tile (6,6). Seed both glyph tiles here --
// the section-opening surface clear wiped whatever the timed tile ops
// left there.
jlTileFill(gStage, 5, 5, 9);
jlTileFill(gStage, 6, 6, 3);
asciiMap['A'] = (uint16_t)(5u | (5u << 8));
asciiMap['B'] = (uint16_t)(6u | (6u << 8));
jlDrawText(gStage, 2, 20, gStage, asciiMap, "ABBA");
chkHash("drawText");
// Out-of-range start: the entry sanitation loop (Phase 5, #22)
// reproduces the historical absorb-and-wrap semantics bit-for-bit
// (the first glyph is consumed without drawing, the second wraps
// to (0,21) and draws) while guaranteeing jlpTileCopyMasked never
// sees an out-of-range destination. Separate hash line so any
// future semantic change stays isolated.
jlDrawText(gStage, 200, 20, gStage, asciiMap, "AB");
chkHash("drawText-offgrid");
}
static void __attribute__((noinline)) checkPalette(void) {
static const uint16_t conforming[16] = {
0x0ABC, 0x0111, 0x0222, 0x0333, 0x0444, 0x0555, 0x0666, 0x0777,
0x0888, 0x0999, 0x0AAA, 0x0BBB, 0x0CCC, 0x0DDD, 0x0EEE, 0x0123
};
uint16_t readBack[16];
uint16_t i;
bool ok;
// Contract invariants (stable across the Phase 2 #16 change): color 0
// is forced to $000 even when the caller passes nonzero, and
// conforming $0RGB entries 1..15 round-trip exactly.
uberMbOp(105);
jlPaletteSet(gStage, 2, conforming);
jlPaletteGet(gStage, 2, readBack);
ok = (readBack[0] == 0x0000u);
for (i = 1; i < 16u; i++) {
if (readBack[i] != conforming[i]) {
ok = false;
}
}
chkPassFail("palette-roundtrip", ok);
}
static void __attribute__((noinline)) checkRandom(void) {
static const uint32_t expected[4] = {
0x87985AA5UL, 0x155B24A3UL, 0x4820F4C4UL, 0x81B3AC98UL
};
uint16_t i;
bool ok;
// xorshift32 golden sequence from a fixed seed; bit-identical on
// every port and across the Phase 8 (#43) IIgs rewrite.
uberMbOp(106);
jlRandomSeed(0x12345678UL);
ok = true;
for (i = 0; i < 4u; i++) {
if (jlRandom() != expected[i]) {
ok = false;
}
}
if (jlRandomRange(100u) != 43u) {
ok = false;
}
chkPassFail("random-golden", ok);
jlRandomSeed(1u);
}
// Arena churn: create/compile/destroy so a hole opens and is reused, then
// compact and assert the used-byte counter returns to its pre-churn value.
// Covers the codegen allocator paths UBER's single long-lived sprite never
// exercises (findings #5, #34 land here in Phase 1). Also runs the
// owned-tileData path (jlSpriteCreateFromSurface) that finding #31
// re-allocates in Phase 1, and a sprite-bank load if an asset is present.
static void __attribute__((noinline)) checkAllocator(void) {
jlSpriteT *a;
jlSpriteT *b;
jlSpriteT *c;
uint32_t usedBefore;
// Sub-markers 111+ pinpoint the statement that hangs (Phase 1
// stabilization; the coarse marker froze on this check).
uberMbOp(111);
usedBefore = jlSpriteCodegenBytesUsed();
a = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y);
b = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y);
if (a == NULL || b == NULL) {
chkPassFail("arena-churn", false);
return;
}
uberMbOp(112);
(void)jlSpriteCompile(a);
(void)jlSpriteCompile(b);
uberMbOp(113);
jlSpriteDestroy(a);
uberMbOp(114);
c = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y);
if (c == NULL) {
jlSpriteDestroy(b);
chkPassFail("arena-churn", false);
return;
}
(void)jlSpriteCompile(c);
uberMbOp(115);
jlSpriteDestroy(b);
jlSpriteDestroy(c);
uberMbOp(116);
jlSpriteCompact();
uberMbOp(117);
chkPassFail("arena-churn", jlSpriteCodegenBytesUsed() == usedBefore);
uberMbOp(118);
a = jlSpriteCreateFromSurface(gStage, 0, 0, 2, 2);
if (a != NULL) {
(void)jlSpriteCompile(a);
uberMbOp(119);
jlSpriteDraw(gStage, a, 280, 20);
jlSpriteDestroy(a);
}
chkPassFail("sprite-from-surface", a != NULL);
uberMbOp(120);
chkHash("sprite-from-surface-draw");
}
// Coverage the primary checkTileMonoFlood golden misses (found in the
// #3/#4 re-analysis): the alternating 0x0F/0xF0 mono pattern only
// selects IIgs #3 combo indices 1 and 2, and jlFloodFill only drives
// the matchEqual branch of the ST #4 plane hooks. Runs LAST so it
// perturbs no earlier cumulative full-stage hash.
static void __attribute__((noinline)) checkTileMonoFloodExtra(void) {
// 0x00 -> combo idx 0 (bg:bg), 0xFF -> idx 3 (fg:fg), 0x0F -> idx 1
// (bg:fg), 0xF0 -> idx 2 (fg:bg): all four opacity pairs in one tile.
static const uint8_t comboBytes[4] = { 0x00u, 0xFFu, 0x0Fu, 0xF0u };
static jlTileT monoAll;
uint8_t i;
uberMbOp(121);
for (i = 0; i < TILE_BYTES; i++) {
monoAll.pixels[i] = comboBytes[i & 3u];
}
jlTilePasteMono(gStage, 10, 12, &monoAll, 5, 9);
jlTilePasteMono(gStage, 11, 12, &monoAll, 14, 0);
chkHash("tilePasteMonoAll");
// Bounded flood: an enclosed color-3 box over a pre-cleared interior
// so the fill stays contained and drives the bounded stop mask
// (pix == boundary || pix == new) plus the planar group-skip across
// several 16px groups. matchColor = boundaryColor = 3, matchEqual = 0.
jlFillRect(gStage, 40, 140, 64, 40, 0);
jlDrawRect(gStage, 44, 144, 52, 32, 3);
jlFloodFillBounded(gStage, 70, 160, 7, 3);
chkHash("floodFillBounded");
}
static void __attribute__((noinline)) runCorrectnessChecks(void) {
uberMbPhase(UBER_PHASE_CHECKS);
jlLogF("UBER-CHK: ----- begin -----\n");
jlSurfaceClear(gStage, 0);
checkEdgeDraws();
checkClippedSprites();
checkTileMonoFlood();
checkDrawText();
checkPalette();
checkRandom();
checkAllocator();
checkTileMonoFloodExtra();
// jlShutdown -> jlInit -> stale-sprite-destroy (#34) needs a teardown
// of the whole library mid-run; it gets a dedicated micro-example in
// Phase 1 rather than risking the benchmark's display state here.
jlLogF("UBER-CHK: ----- end -----\n");
}
// ----- Main -----
static void __attribute__((noinline)) runAllTests(void) {
uberMbPhase(UBER_PHASE_TIMED_RUN);
jlLogF("UBER: ----- begin -----\n");
// Surface / palette / SCB.
timeOp("jlSurfaceClear", op_surfaceClear);
timeOp("jlPaletteSet", op_paletteSet);
timeOp("jlScbSetRange", op_scbSetRange);
// Drawing primitives.
timeOp("jlDrawPixel", op_drawPixel);
timeOp("jlDrawLine H", op_drawLineH);
timeOp("jlDrawLine V", op_drawLineV);
timeOp("jlDrawLine diag", op_drawLineDiag);
timeOp("jlDrawRect 100x100", op_drawRect);
timeOp("jlDrawCircle r=16", op_drawCircleSmall);
timeOp("jlDrawCircle r=80", op_drawCircleLarge);
timeOp("jlFillRect 16x16", op_fillRectSmall);
timeOp("jlFillRect 80x80", op_fillRectMid);
timeOp("jlFillRect 320x200", op_fillRectFull);
timeOp("jlFillCircle r=40", op_fillCircle);
timeOp("jlSamplePixel", op_samplePixel);
// Tiles. Seed scratch tile + dest cells with non-zero pixels first.
jlFillRect(gStage, 0, 0, 320, 64, 7);
jlTileSnap(gStage, 5, 5, &gTileScratch);
timeOp("jlTileFill", op_tileFill);
timeOp("jlTileCopy", op_tileCopy);
timeOp("jlTileCopyMasked", op_tileCopyMasked);
timeOp("jlTilePaste", op_tilePaste);
timeOp("jlTileSnap", op_tileSnap);
// Sprites. Background must be non-empty so save-under has work
// to do (otherwise it's a 4 KB memset of zeros, atypical).
jlSurfaceClear(gStage, 4);
timeOp("jlSpriteSaveUnder", op_spriteSave);
timeOp("jlSpriteDraw", op_spriteDraw);
timeOp("jlSpriteRestoreUnder", op_spriteRestore);
timeOp("jlSpriteSaveAndDraw", op_spriteSaveAndDraw);
// Present. One warm-up call before each timed loop primes any
// per-port one-time setup (Amiga: copper list rebuild after the
// jlPaletteSet / jlScbSetRange tests dirty the cache; without warm-up
// the rebuild's MakeScreen + MrgCop + WaitTOF chain consumes the
// entire 4-frame measurement window) so we measure steady-state
// throughput rather than first-call penalty.
jlStagePresent();
timeOp("jlStagePresent full", op_stagePresent);
// Input.
timeOp("jlInputPoll", op_inputPoll);
timeOp("jlKeyDown", op_keyDown);
timeOp("jlKeyPressed", op_keyPressed);
timeOp("jlMouseX", op_mouseX);
timeOp("joeyJoyConnected", op_joyConnected);
// Audio.
timeOp("jlAudioFrameTick", op_audioFrameTick);
timeOp("jlAudioIsPlayingMod", op_audioIsPlaying);
// Surface mark dirty (via jlFillRect's mark step).
timeOp("surfaceMarkDirtyRect (via jlFillRect 32x32)", op_surfaceMarkDirty);
jlLogF("UBER: ----- end -----\n");
}
// Extracted from main so main's register pressure stays under the w65816
// allocator's ceiling -- the 16-entry pal[] + loop index was the overflow.
// noinline is load-bearing at -O2 (the backend would otherwise re-inline a
// single-call static and recreate the pressure).
static void __attribute__((noinline)) setupPalette(void) {
uint16_t pal[16];
int i;
for (i = 0; i < 16; i++) {
pal[i] = (uint16_t)((i << 8) | (i << 4) | i); // grey ramp
}
pal[ 0] = 0x000;
pal[ 1] = 0x800; // dark red (running)
pal[ 2] = 0x080; // green (done)
pal[ 3] = 0x008; // blue
pal[ 5] = 0xFF0; // yellow (test pixels)
pal[ 7] = 0xFFF; // white (fills)
pal[15] = 0xF00; // red
jlPaletteSet(gStage, 0, pal);
jlScbSetRange(gStage, 0, 199, 0);
}
// Extracted from main for the same register-pressure reason as setupPalette.
static void __attribute__((noinline)) reportElapsed(uint16_t startFrame) {
uint16_t endFrame;
uint16_t elapsedFrames;
unsigned long elapsedMs;
endFrame = jlFrameCount();
elapsedFrames = (uint16_t)(endFrame - startFrame);
elapsedMs = ((unsigned long)elapsedFrames * 1000UL) / (unsigned long)jlFrameHz();
jlLogF("UBER: total wall time: %lu ms (%u frames @ %u Hz)\n",
elapsedMs, elapsedFrames, (unsigned)jlFrameHz());
}
// Extracted from main (register pressure). Returns false on sprite-create fail.
static bool __attribute__((noinline)) setupSprite(void) {
uint16_t before;
uberMbPhase(UBER_PHASE_SETUP_SPRITE);
buildBallSprite();
gSprite = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y);
if (gSprite == NULL) {
jlLog("UBER: jlSpriteCreate failed");
return false;
}
// jlSpriteCompile is a one-shot. Time at frame resolution.
jlWaitVBL();
before = jlFrameCount();
if (!jlSpriteCompile(gSprite)) {
jlLog("UBER: jlSpriteCompile failed");
}
while (jlFrameCount() == before) {
/* wait for next VBL edge */
}
jlLogF("UBER: jlSpriteCompile: 1 call in <= 1 frame\n");
gBackup.bytes = gBackupBytes;
return true;
}
// Extracted from main (register pressure): the jlConfigT struct on main's
// frame, on top of the rest of main, pushed the w65816 allocator over its
// limit. noinline keeps it out at -O2.
static bool __attribute__((noinline)) initJoeyLib(void) {
jlConfigT config;
/* 32 KB fits the 8 pre-shifted DRAW variants the Amiga planar
* compiled sprite emitter generates. UL on the multiply because
* a 16-bit int overflows on 32 * 1024. */
config.codegenBytes = 32UL * 1024;
config.audioBytes = 64UL * 1024;
return jlInit(&config);
}
int main(void) {
uint16_t startFrame;
if (!initJoeyLib()) {
return 1;
}
/* jlFrameCount is VBL-driven, so it only ticks after halInit
* installed its VBL ISR -- captured here is "everything from now
* to press-any-key". Pre-init setup time is small and not the
* cost the user is chasing; runAllTests dominates. */
startFrame = jlFrameCount();
gStage = jlStageGet();
if (gStage == NULL) {
jlShutdown();
return 1;
}
uberMbLayout();
// A simple visible palette so users see SOMETHING during the run.
setupPalette();
// Indicate "running": red bar at top of screen.
jlSurfaceClear(gStage, 0);
jlFillRect(gStage, 0, 0, 320, 8, 1);
jlStagePresent();
if (!setupSprite()) {
jlShutdown();
return 1;
}
#ifdef JOEYLIB_PLATFORM_IIGS
*UBER_MB_SPRITEPTR = (uint32_t)gSprite;
#endif
// Audio: only init/shutdown is exercised. Triggering jlAudioPlaySfx
// without first calling jlAudioPlayMod leaves NTP's engine in a
// half-initialized state -- NTPstreamsound is designed to OVERLAY on
// an already-running module. Without NTPprepare/NTPplay first, the
// streamer oscillator is fired but no music tick ever advances or
// silences it, and you get a stuck high-pitched scream. UBER doesn't
// ship a MOD asset, so we skip the SFX exercise. The frame-tick and
// isPlayingMod calls below still get timed (both are no-op fast
// paths on IIgs).
uberMbPhase(UBER_PHASE_AUDIO_INIT);
if (jlAudioInit()) {
jlLogF("UBER: audioInit OK\n");
} else {
jlLogF("UBER: audioInit failed (skipping audio)\n");
}
// Visual showcase: render one of each primitive into its own grid
// cell so the run shows something legible (the timed benchmark below
// never presents, so this stays on screen throughout it). The first
// timed op is jlSurfaceClear, so this leaves the benchmark untouched.
drawShowcase();
runAllTests();
runCorrectnessChecks();
reportElapsed(startFrame);
// Done. Green screen + waitForKey.
uberMbPhase(UBER_PHASE_DONE);
jlSurfaceClear(gStage, 2);
jlStagePresent();
jlLogF("UBER: press any key to exit\n");
// Flush the log to disk BEFORE the blocking key wait, so an automated
// (headless) run that kills the process at this prompt still captures
// the results -- joeyLog otherwise only flushes at the atexit fclose.
jlLogFlush();
jlWaitForAnyKey();
jlSpriteDestroy(gSprite);
jlShutdown();
return 0;
}