1197 lines
46 KiB
C
1197 lines
46 KiB
C
// Uber demo: exercise every JoeyLib public API and measure throughput
|
|
// of the per-frame-hot ones. Results are written to joeylog.txt via
|
|
// jlLogF. A green screen on exit means the run completed.
|
|
//
|
|
// Timing model: each test aligns to a VBL boundary via jlWaitVBL,
|
|
// records the starting jlFrameCount, then runs the op in a tight
|
|
// loop polling jlFrameCount until UBER_FRAMES frames have elapsed.
|
|
// Reported metric is ops/sec, computed as iters * jlFrameHz() /
|
|
// UBER_FRAMES so results are directly comparable across ports
|
|
// regardless of CPU speed or VBL rate.
|
|
//
|
|
// jlFrameCount is wall-clock-based per port, but READING it is not
|
|
// free: on the IIgs it is a Misc Toolset GetTick call (several
|
|
// hundred cycles per poll -- an earlier comment here claimed
|
|
// ~10-30 cyc, which was wrong by more than an order of magnitude
|
|
// and inflated every fast op's measured cost by a near-constant
|
|
// ~600-800 cycles/iter). runForFrames therefore polls the clock
|
|
// once per batch of UBER_BATCH op calls for sub-frame ops
|
|
// (calibrated by the first call), amortizing the poll to noise.
|
|
//
|
|
// One-shot ops (jlSpriteCompile) get one call each, timed by frame
|
|
// delta -- coarser but representative.
|
|
|
|
#include <stdio.h>
|
|
#include <stdlib.h>
|
|
#include <string.h>
|
|
#include <stddef.h>
|
|
|
|
#include <joey/joey.h>
|
|
|
|
|
|
// ----- Timing primitives -----
|
|
|
|
// 4-frame measurement window. Long enough that loop overhead doesn't
|
|
// dominate; short enough to keep the full demo run under ~10 sec.
|
|
/* 16 frames per timed op gives 4x the iter-count resolution of the
|
|
* earlier 4-frame budget. Exposes the actual per-op cost on slow
|
|
* ops where 4 frames produced the same iter count on different
|
|
* framerates -- e.g. jlDrawCircle r=80 read as "4 iters / 4 frames"
|
|
* on both 60 Hz IIgs (16.7 ms/frame, 67 ms window) and 50 Hz Amiga
|
|
* (20 ms/frame, 80 ms window) even though per-op cost was equal,
|
|
* just because 4 ops at 16-17 ms happen to fit both windows. The
|
|
* 16-frame budget extends the windows to 267 ms / 320 ms; quantum
|
|
* gap shrinks to ~6%. Total run time scales 4x (~80 sec each). */
|
|
#define UBER_FRAMES 16u
|
|
|
|
|
|
// Op calls per clock poll for sub-frame ops. The unrolled batch in
|
|
// runForFrames must contain exactly this many op() calls. Ops slower
|
|
// than one frame poll every call instead, so a batch cannot overrun
|
|
// the measurement window by more than one op.
|
|
#define UBER_BATCH 16u
|
|
|
|
// ----- IIgs progress mailbox (headless-MAME diagnosis) -----
|
|
//
|
|
// The SHR SCB region covers rows 0..199 at $E1:9D00-$E1:9DC7; the bytes
|
|
// at $E1:9DC8-$E1:9DFF are unused by hardware and by jlpPresent. UBER
|
|
// mirrors its progress there so a MAME Lua probe can see WHERE a
|
|
// headless run is (and whether the timing loop and the tick are alive)
|
|
// even when the joeylog FST flush never lands on disk. Layout:
|
|
// +0 (9DC8) uint8 current timed-op index (from timeOp); during
|
|
// phase 5 it is 100 + the correctness-check number
|
|
// +1 (9DC9) uint8 phase (UBER_PHASE_*)
|
|
// +2 (9DCA) uint16 heartbeat (one increment per timing-loop pass)
|
|
// +4 (9DCC) uint16 last jlFrameCount() value the timing loop saw
|
|
// No-ops on every other port.
|
|
#define UBER_PHASE_SETUP_SPRITE 1u
|
|
#define UBER_PHASE_AUDIO_INIT 2u
|
|
#define UBER_PHASE_SHOWCASE 3u
|
|
#define UBER_PHASE_TIMED_RUN 4u
|
|
#define UBER_PHASE_CHECKS 5u
|
|
#define UBER_PHASE_DONE 6u
|
|
|
|
#ifdef JOEYLIB_PLATFORM_IIGS
|
|
#define UBER_MB_OP ((volatile uint8_t *)0xE19DC8L)
|
|
#define UBER_MB_PHASE ((volatile uint8_t *)0xE19DC9L)
|
|
#define UBER_MB_HEARTBEAT ((volatile uint16_t *)0xE19DCAL)
|
|
#define UBER_MB_TICK ((volatile uint16_t *)0xE19DCCL)
|
|
// One-time layout facts for probe correlation: +8 stage pointer value,
|
|
// +12 address of uber.c's gStage static, +16 log-ring base address,
|
|
// +20 log-ring head-counter address (see src/core/debug.c's IIgs RAM
|
|
// ring -- the Lua probe drains the log live through these).
|
|
#define UBER_MB_MAGIC ((volatile uint16_t *)0xE19DCEL)
|
|
#define UBER_MB_STAGEPTR ((volatile uint32_t *)0xE19DD0L)
|
|
#define UBER_MB_GSTAGEAT ((volatile uint32_t *)0xE19DD4L)
|
|
#define UBER_MB_LOGRING ((volatile uint32_t *)0xE19DD8L)
|
|
#define UBER_MB_LOGHEAD ((volatile uint32_t *)0xE19DDCL)
|
|
// TEMP #84: the gSprite pointer value, poked after setupSprite so the
|
|
// Lua probe can watch the struct bytes get overwritten in place.
|
|
#define UBER_MB_SPRITEPTR ((volatile uint32_t *)0xE19DEAL)
|
|
// "JL" -- the Lua probes refuse to trust the layout words until this
|
|
// appears (the SCB region holds 0x80 fill before the app writes it).
|
|
#define UBER_MB_MAGIC_VAL 0x4A4Cu
|
|
extern uint8_t gJoeyLogRing[];
|
|
extern uint16_t gJoeyLogRingHead;
|
|
#define uberMbOp(_i) (*UBER_MB_OP = (uint8_t)(_i))
|
|
#define uberMbPhase(_p) (*UBER_MB_PHASE = (uint8_t)(_p))
|
|
#define uberMbBeat() (*UBER_MB_HEARTBEAT = (uint16_t)(*UBER_MB_HEARTBEAT + 1u))
|
|
#define uberMbTick(_t) (*UBER_MB_TICK = (uint16_t)(_t))
|
|
#define uberMbLayout() (*UBER_MB_STAGEPTR = (uint32_t)gStage, *UBER_MB_GSTAGEAT = (uint32_t)&gStage, *UBER_MB_LOGRING = (uint32_t)&gJoeyLogRing[0], *UBER_MB_LOGHEAD = (uint32_t)&gJoeyLogRingHead, *UBER_MB_MAGIC = UBER_MB_MAGIC_VAL)
|
|
#else
|
|
#define uberMbOp(_i) ((void)0)
|
|
#define uberMbPhase(_p) ((void)0)
|
|
#define uberMbBeat() ((void)0)
|
|
#define uberMbTick(_t) ((void)0)
|
|
#define uberMbLayout() ((void)0)
|
|
#endif
|
|
|
|
|
|
typedef void (*OpFn)(void);
|
|
|
|
static const char *gCurName = "(none)";
|
|
static jlSurfaceT *gStage = NULL;
|
|
static jlSpriteT *gSprite = NULL;
|
|
static jlSpriteBackupT gBackup;
|
|
static unsigned char gBackupBytes[JOEY_SPRITE_BACKUP_BYTES(2, 2)] __attribute__((aligned(2)));
|
|
|
|
static jlTileT gTileScratch;
|
|
|
|
// Phase 0 workload rows (NATIVE-PERF-PLAN.md item 3): patterned tiles
|
|
// seeded via draw+snap (tile pixels are platform-private; only a snap
|
|
// round-trips portably) and dedicated gameFrame sprite backups so the
|
|
// correctness-check gBackup state stays untouched.
|
|
static jlTileT gTileMapA;
|
|
static jlTileT gTileMapB;
|
|
|
|
#define UBER_FRAME_SPRITES 3
|
|
|
|
static jlSpriteBackupT gFrameBackups[UBER_FRAME_SPRITES];
|
|
static unsigned char gFrameBackupBytes[UBER_FRAME_SPRITES][JOEY_SPRITE_BACKUP_BYTES(2, 2)] __attribute__((aligned(2)));
|
|
|
|
|
|
// Current timed-op number (1-based); mirrored into the mailbox. Lives
|
|
// up here because runForFrames re-writes it every loop pass: the SHR
|
|
// SCB upload in jlStagePresent can overwrite the adjacent mailbox
|
|
// bytes, so a single write at timeOp start can be stomped mid-window.
|
|
static uint8_t gOpIndex = 0;
|
|
|
|
|
|
// Run `op` in a tight loop until `targetFrames` jlFrameCount ticks
|
|
// have elapsed. Returns iterations completed.
|
|
static unsigned long runForFrames(OpFn op, unsigned int targetFrames, uint16_t *actualFramesOut, uint32_t *millisDeltaOut) {
|
|
unsigned long count;
|
|
uint16_t startFrame;
|
|
uint16_t endFrame;
|
|
uint16_t now;
|
|
uint32_t startMillis;
|
|
uint32_t endMillis;
|
|
bool slowOp;
|
|
|
|
count = 0UL;
|
|
|
|
jlWaitVBL();
|
|
startFrame = jlFrameCount();
|
|
startMillis = jlMillisElapsed();
|
|
|
|
// Calibrate: jlWaitVBL aligned us to a tick edge, so a sub-frame op
|
|
// cannot see the counter advance during a single call. If the first
|
|
// call DOES advance it, the op is slower than one frame -- poll the
|
|
// clock every call so a batch cannot overrun the window.
|
|
op();
|
|
count++;
|
|
slowOp = ((uint16_t)(jlFrameCount() - startFrame) != 0u);
|
|
|
|
for (;;) {
|
|
now = jlFrameCount();
|
|
uberMbTick(now);
|
|
uberMbBeat();
|
|
uberMbOp(gOpIndex);
|
|
if ((uint16_t)(now - startFrame) >= targetFrames) {
|
|
break;
|
|
}
|
|
if (slowOp) {
|
|
op();
|
|
count++;
|
|
} else {
|
|
// Exactly UBER_BATCH calls between clock polls (the poll is
|
|
// a toolbox call on IIgs; per-op polling dominated fast ops).
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
op();
|
|
count += UBER_BATCH;
|
|
}
|
|
}
|
|
/* Capture the actual elapsed frames -- the last iter typically
|
|
* overruns the target. Using actual instead of target as the
|
|
* ops/sec divisor stays honest for ops slower than 1 frame
|
|
* (where count is forced low while real time stretches well
|
|
* past targetFrames). */
|
|
endFrame = jlFrameCount();
|
|
endMillis = jlMillisElapsed();
|
|
*actualFramesOut = (uint16_t)(endFrame - startFrame);
|
|
if (*actualFramesOut == 0u) {
|
|
*actualFramesOut = 1u; /* defensive: avoid div-by-zero */
|
|
}
|
|
*millisDeltaOut = endMillis - startMillis;
|
|
if (*millisDeltaOut == 0u) {
|
|
*millisDeltaOut = 1u; /* defensive: avoid div-by-zero */
|
|
}
|
|
return count;
|
|
}
|
|
|
|
|
|
// Time and log one op. Reports iters / N frames AND the derived
|
|
// ops/sec so per-port results are directly comparable against IIgs
|
|
// regardless of CPU speed or display refresh rate. Also logs an
|
|
// FNV-1a hash of the surface state after timing -- this is the
|
|
// pixel-perfect comparison input for the cross-port validation
|
|
// harness (tools/diff-uber-hashes.py). Captured against IIgs as the
|
|
// golden reference; planar 68k rewrites validate by matching it.
|
|
static void timeOp(const char *name, OpFn op) {
|
|
unsigned long iters;
|
|
unsigned long opsPerSec;
|
|
unsigned long opsPerSecFr;
|
|
uint16_t actualFrames;
|
|
uint32_t millisDelta;
|
|
uint32_t hash;
|
|
|
|
gCurName = name;
|
|
gOpIndex++;
|
|
uberMbOp(gOpIndex);
|
|
|
|
iters = runForFrames(op, UBER_FRAMES, &actualFrames, &millisDelta);
|
|
|
|
if (iters == 0UL) {
|
|
jlLogF("UBER: %s: 0 iters (op too slow?)\n", name);
|
|
return;
|
|
}
|
|
|
|
/* Report ops/sec from the MILLISECOND clock, not the frame counter.
|
|
* jlMillisElapsed is refresh-independent on every port (ST Timer-C
|
|
* 200 Hz, Amiga audio-ISR tick, DOS PIT, IIgs GetTick-derived so
|
|
* identical to the old formula there), so it does not depend on
|
|
* jlFrameHz() matching the emulator's real VBL rate -- the ST used
|
|
* to hardcode 50 Hz while Hatari runs ~60, under-reporting ~17%.
|
|
* The frame-derived rate is kept as the UBER-CLK cross-check: after
|
|
* the jlpFrameHz fix the two agree; a divergence flags a clock bug. */
|
|
opsPerSec = (iters * 1000UL) / (unsigned long)millisDelta;
|
|
opsPerSecFr = (iters * (unsigned long)jlFrameHz()) / (unsigned long)actualFrames;
|
|
hash = jlSurfaceHash(gStage);
|
|
jlLogF("UBER: %s: %lu iters / %u frames = %lu ops/sec | hash=%08lX\n",
|
|
name, iters, actualFrames, opsPerSec, (unsigned long)hash);
|
|
jlLogF("UBER-CLK: %s: frame=%lu millis=%lu ops/sec (%lu ms)\n",
|
|
name, opsPerSecFr, opsPerSec, (unsigned long)millisDelta);
|
|
}
|
|
|
|
|
|
|
|
|
|
// ----- Test ops -----
|
|
|
|
static void op_drawPixel (void) { jlDrawPixel (gStage, 100, 100, 5); }
|
|
static void op_drawLineH (void) { jlDrawLine (gStage, 0, 50, 319, 50, 5); }
|
|
static void op_drawLineV (void) { jlDrawLine (gStage, 50, 0, 50, 199, 5); }
|
|
static void op_drawLineDiag (void) { jlDrawLine (gStage, 0, 0, 319, 199, 5); }
|
|
static void op_drawRect (void) { jlDrawRect (gStage, 10, 10, 100, 100, 5); }
|
|
static void op_drawCircleSmall (void) { jlDrawCircle (gStage, 160, 100, 16, 5); }
|
|
static void op_drawCircleLarge (void) { jlDrawCircle (gStage, 160, 100, 80, 5); }
|
|
static void op_fillRectSmall (void) { jlFillRect (gStage, 20, 20, 16, 16, 7); }
|
|
static void op_fillRectMid (void) { jlFillRect (gStage, 20, 20, 80, 80, 7); }
|
|
static void op_fillRectFull (void) { jlFillRect (gStage, 0, 0, 320, 200, 7); }
|
|
static void op_fillCircle (void) { jlFillCircle (gStage, 160, 100, 40, 7); }
|
|
static void op_samplePixel (void) { (void)jlSamplePixel(gStage, 100, 100); }
|
|
static void op_surfaceClear (void) { jlSurfaceClear (gStage, 0); }
|
|
|
|
static void op_paletteSet(void) {
|
|
static uint16_t colors[16] = {
|
|
0x000, 0xF00, 0x0F0, 0x00F, 0xFF0, 0xF0F, 0x0FF, 0xFFF,
|
|
0x800, 0x080, 0x008, 0x880, 0x808, 0x088, 0x888, 0x444
|
|
};
|
|
jlPaletteSet(gStage, 0, colors);
|
|
}
|
|
static void op_scbSetRange (void) { jlScbSetRange (gStage, 0, 199, 0); }
|
|
|
|
static void op_tileFill (void) { jlTileFill (gStage, 5, 5, 7); }
|
|
static void op_tileCopy (void) { jlTileCopy (gStage, 6, 6, gStage, 5, 5); }
|
|
static void op_tileCopyMasked (void) { jlTileCopyMasked (gStage, 7, 7, gStage, 5, 5, 0); }
|
|
static void op_tilePaste (void) { jlTilePaste (gStage, 8, 8, &gTileScratch); }
|
|
static void op_tileSnap (void) { jlTileSnap (gStage, 5, 5, &gTileScratch); }
|
|
|
|
static int16_t gSpriteX = 40;
|
|
static int16_t gSpriteY = 30;
|
|
|
|
static void op_spriteSave (void) { jlSpriteSaveUnder (gStage, gSprite, gSpriteX, gSpriteY, &gBackup); }
|
|
static void op_spriteDraw (void) { jlSpriteDraw (gStage, gSprite, gSpriteX, gSpriteY); }
|
|
static void op_spriteRestore (void) { jlSpriteRestoreUnder(gStage, &gBackup); }
|
|
static void op_spriteSaveAndDraw (void) { jlSpriteSaveAndDraw (gStage, gSprite, gSpriteX, gSpriteY, &gBackup); }
|
|
|
|
static void op_stagePresent (void) { jlStagePresent(); }
|
|
|
|
static void op_inputPoll (void) { jlInputPoll(); }
|
|
static void op_keyDown (void) { (void)jlKeyDown(KEY_A); }
|
|
static void op_keyPressed (void) { (void)jlKeyPressed(KEY_A); }
|
|
static void op_mouseX (void) { (void)jlMouseX(); }
|
|
static void op_joyConnected (void) { (void)jlJoystickConnected(JOYSTICK_1); }
|
|
|
|
static void op_audioFrameTick (void) { jlAudioFrameTick(); }
|
|
static void op_audioIsPlaying (void) { (void)jlAudioIsPlayingMod(); }
|
|
|
|
static void op_surfaceMarkDirty(void) { /* jlDrawPixel already marks; use fill instead */
|
|
jlFillRect(gStage, 0, 0, 32, 32, 0); }
|
|
|
|
|
|
// ----- Phase 0 workload ops (NATIVE-PERF-PLAN.md item 3) -----
|
|
|
|
// Worst-case unaligned x on every port at once: odd -> IIgs/DOS
|
|
// compiled shift-1 (widened RMW row); x % 8 == 7 -> ST shift index 2
|
|
// (never compiled, interpreted jlpSpriteDrawPlanes) and Amiga shift 7
|
|
// (only shift 0 compiled) with the maximal 7-bit spill across three
|
|
// plane bytes per row. Phase 1's all-shift emitters must pull this row
|
|
// to ~parity with the aligned jlSpriteDraw row.
|
|
#define UBER_UNALIGNED_X 71
|
|
#define UBER_UNALIGNED_Y 120
|
|
|
|
static void op_spriteDrawUnaligned(void) { jlSpriteDraw(gStage, gSprite, UBER_UNALIGNED_X, UBER_UNALIGNED_Y); }
|
|
|
|
|
|
// All 16 x phases in ONE call so the op is idempotent: per-port
|
|
// iteration counts and the 16-call unrolled batch would otherwise
|
|
// leave a port-dependent final x and a port-dependent hash. 208..223
|
|
// covers each Amiga shift twice, both ST compiled shifts (208, 216)
|
|
// plus all 14 interpreter phases, and 8 even + 8 odd IIgs/DOS draws.
|
|
// Reported ops/sec is SWEEPS per second: multiply by 16 for draws/sec.
|
|
#define UBER_SWEEP_X0 208
|
|
#define UBER_SWEEP_Y 120
|
|
|
|
static void op_spriteDrawSweep(void) {
|
|
int16_t x;
|
|
|
|
for (x = UBER_SWEEP_X0; x < (int16_t)(UBER_SWEEP_X0 + 16); x++) {
|
|
jlSpriteDraw(gStage, gSprite, x, UBER_SWEEP_Y);
|
|
}
|
|
}
|
|
|
|
|
|
// 50 single-tile pastes = the per-call wrapper tax paid 50 times; the
|
|
// Phase 3 jlTileMapPaste batch API must beat this row by roughly the
|
|
// per-call overhead fraction. Fixed checker of two snapped tiles.
|
|
#define UBER_TILEMAP_X0 12
|
|
#define UBER_TILEMAP_Y0 12
|
|
#define UBER_TILEMAP_W 10
|
|
#define UBER_TILEMAP_H 5
|
|
|
|
static void op_tileMapPaste(void) {
|
|
uint8_t tx;
|
|
uint8_t ty;
|
|
|
|
for (ty = 0; ty < UBER_TILEMAP_H; ty++) {
|
|
for (tx = 0; tx < UBER_TILEMAP_W; tx++) {
|
|
if (((tx + ty) & 1u) == 0u) {
|
|
jlTilePaste(gStage, (uint8_t)(UBER_TILEMAP_X0 + tx), (uint8_t)(UBER_TILEMAP_Y0 + ty), &gTileMapA);
|
|
} else {
|
|
jlTilePaste(gStage, (uint8_t)(UBER_TILEMAP_X0 + tx), (uint8_t)(UBER_TILEMAP_Y0 + ty), &gTileMapB);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
|
|
// One realistic game frame: repaint a play region, paste a HUD tile
|
|
// row, save+draw three sprites, restore them in reverse, present.
|
|
// The fill runs FIRST in the same call, so the saves always capture
|
|
// UBER_FRAME_BG and the surface is net-identical after every call --
|
|
// the post-window hash is iteration-count independent. All sprite x
|
|
// are mod-16-aligned (compiled shift 0 on every port): this row
|
|
// isolates the per-call wrapper tax + present; the rows above own the
|
|
// alignment signal. Presenting a re-dirtied stage every call also
|
|
// fixes the present-row artifact (the timed jlStagePresent row goes
|
|
// idle after its first copy).
|
|
#define UBER_FRAME_RX 32
|
|
#define UBER_FRAME_RY 64
|
|
#define UBER_FRAME_RW 256
|
|
#define UBER_FRAME_RH 96
|
|
#define UBER_FRAME_BG 6
|
|
#define UBER_FRAME_TILES 16
|
|
#define UBER_FRAME_TILE_TX 4
|
|
#define UBER_FRAME_TILE_TY 8
|
|
|
|
static const int16_t gFrameSpriteX[UBER_FRAME_SPRITES] = { 48, 144, 240 };
|
|
static const int16_t gFrameSpriteY[UBER_FRAME_SPRITES] = { 80, 96, 112 };
|
|
|
|
static void op_gameFrame(void) {
|
|
int16_t i;
|
|
uint8_t t;
|
|
|
|
jlFillRect(gStage, UBER_FRAME_RX, UBER_FRAME_RY, UBER_FRAME_RW, UBER_FRAME_RH, UBER_FRAME_BG);
|
|
for (t = 0; t < UBER_FRAME_TILES; t++) {
|
|
if ((t & 1u) == 0u) {
|
|
jlTilePaste(gStage, (uint8_t)(UBER_FRAME_TILE_TX + t), UBER_FRAME_TILE_TY, &gTileMapA);
|
|
} else {
|
|
jlTilePaste(gStage, (uint8_t)(UBER_FRAME_TILE_TX + t), UBER_FRAME_TILE_TY, &gTileMapB);
|
|
}
|
|
}
|
|
for (i = 0; i < UBER_FRAME_SPRITES; i++) {
|
|
jlSpriteSaveAndDraw(gStage, gSprite, gFrameSpriteX[i], gFrameSpriteY[i], &gFrameBackups[i]);
|
|
}
|
|
for (i = UBER_FRAME_SPRITES - 1; i >= 0; i--) {
|
|
jlSpriteRestoreUnder(gStage, &gFrameBackups[i]);
|
|
}
|
|
jlStagePresent();
|
|
}
|
|
|
|
|
|
// ----- Build the ball sprite procedurally -----
|
|
|
|
#define BALL_TILES_X 2
|
|
#define BALL_TILES_Y 2
|
|
#define BALL_TILE_BYTES (BALL_TILES_X * BALL_TILES_Y * 32u)
|
|
|
|
static const uint8_t gBallAuthored[16 * 8] = {
|
|
0x00, 0x00, 0x22, 0x22, 0x22, 0x22, 0x00, 0x00,
|
|
0x00, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x00,
|
|
0x02, 0x22, 0x32, 0x22, 0x22, 0x22, 0x22, 0x20,
|
|
0x02, 0x23, 0x32, 0x22, 0x22, 0x22, 0x22, 0x20,
|
|
0x22, 0x33, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
|
|
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
|
|
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
|
|
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
|
|
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
|
|
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
|
|
0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22,
|
|
0x02, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x20,
|
|
0x02, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x20,
|
|
0x00, 0x22, 0x22, 0x22, 0x22, 0x22, 0x22, 0x00,
|
|
0x00, 0x00, 0x22, 0x22, 0x22, 0x22, 0x00, 0x00,
|
|
0x00, 0x00, 0x00, 0x22, 0x22, 0x00, 0x00, 0x00
|
|
};
|
|
static uint8_t gBallTiles[BALL_TILE_BYTES];
|
|
|
|
static void buildBallSprite(void) {
|
|
uint16_t tx;
|
|
uint16_t ty;
|
|
uint16_t row;
|
|
uint16_t b;
|
|
uint8_t *dst;
|
|
|
|
for (ty = 0; ty < BALL_TILES_Y; ty++) {
|
|
for (tx = 0; tx < BALL_TILES_X; tx++) {
|
|
dst = &gBallTiles[(ty * BALL_TILES_X + tx) * 32u];
|
|
for (row = 0; row < 8; row++) {
|
|
for (b = 0; b < 4; b++) {
|
|
dst[row * 4 + b] =
|
|
gBallAuthored[((ty * 8) + row) * 8 + (tx * 4) + b];
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
|
|
// ----- Visual showcase -----
|
|
//
|
|
// Before the (visually meaningless) timed benchmark, draw one example
|
|
// of each primitive into its own grid cell so a viewer can actually SEE
|
|
// what the library renders instead of a single solid color. The
|
|
// benchmark that follows never presents, so this showcase stays on
|
|
// screen for the whole timed run; results go to joeylog.txt. A legend
|
|
// mapping cell index -> primitive is also logged.
|
|
|
|
#define SC_SCREEN_W 320
|
|
#define SC_SCREEN_H 200
|
|
#define SC_PAL_COUNT 16
|
|
#define SC_PAL_SWATCH_W (SC_SCREEN_W / SC_PAL_COUNT)
|
|
#define SC_PAL_STRIP_H 12
|
|
#define SC_GRID_TOP 16
|
|
#define SC_COLS 4
|
|
#define SC_ROWS 3
|
|
#define SC_CELLS (SC_COLS * SC_ROWS)
|
|
#define SC_GUTTER 4
|
|
#define SC_INSET 5
|
|
#define SC_BG_COLOR 10
|
|
#define SC_BORDER_COLOR 1
|
|
#define SC_HOLD_FRAMES 210
|
|
|
|
|
|
static const uint16_t gShowcasePal[SC_PAL_COUNT] = {
|
|
0x000, 0xFFF, 0xF00, 0x0F0, 0x00F, 0xFF0, 0x0FF, 0xF0F,
|
|
0xF80, 0xAAA, 0x555, 0x8AF, 0x8F8, 0xF8C, 0x840, 0xACE
|
|
};
|
|
|
|
|
|
static void showcaseCellRect(uint16_t index, int16_t *outX, int16_t *outY, int16_t *outW, int16_t *outH) {
|
|
uint16_t col;
|
|
uint16_t row;
|
|
int16_t cellW;
|
|
int16_t cellH;
|
|
|
|
col = (uint16_t)(index % SC_COLS);
|
|
row = (uint16_t)(index / SC_COLS);
|
|
cellW = (int16_t)((SC_SCREEN_W - (SC_COLS + 1) * SC_GUTTER) / SC_COLS);
|
|
cellH = (int16_t)((SC_SCREEN_H - SC_GRID_TOP - (SC_ROWS + 1) * SC_GUTTER) / SC_ROWS);
|
|
*outX = (int16_t)(SC_GUTTER + (int16_t)col * (cellW + SC_GUTTER));
|
|
*outY = (int16_t)(SC_GRID_TOP + SC_GUTTER + (int16_t)row * (cellH + SC_GUTTER));
|
|
*outW = cellW;
|
|
*outH = cellH;
|
|
}
|
|
|
|
|
|
// optnone: this 12-case showcase has too many simultaneously-live locals for
|
|
// the w65816 register allocator at -O2 (it bails "ran out of registers"). It
|
|
// renders once at startup, so dropping optimization here costs nothing.
|
|
// optnone is a clang-only attribute; GCC (DOS/Amiga/ST) rejects it under
|
|
// -Werror=attributes and does not need it, so guard it to clang.
|
|
#if defined(__clang__)
|
|
#define DRAWSHOWCASE_ATTR __attribute__((noinline, optnone))
|
|
#else
|
|
#define DRAWSHOWCASE_ATTR __attribute__((noinline))
|
|
#endif
|
|
static void DRAWSHOWCASE_ATTR drawShowcase(void) {
|
|
uint16_t cell;
|
|
int16_t x;
|
|
int16_t y;
|
|
int16_t w;
|
|
int16_t h;
|
|
int16_t ix;
|
|
int16_t iy;
|
|
int16_t iw;
|
|
int16_t ih;
|
|
int16_t cx;
|
|
int16_t cy;
|
|
int16_t r;
|
|
int16_t k;
|
|
int16_t px;
|
|
int16_t py;
|
|
int16_t minDim;
|
|
uint8_t bx;
|
|
uint8_t by;
|
|
uint8_t bxStart;
|
|
uint8_t bxEnd;
|
|
uint8_t byStart;
|
|
uint8_t byEnd;
|
|
uint16_t held;
|
|
|
|
uberMbPhase(UBER_PHASE_SHOWCASE);
|
|
jlPaletteSet(gStage, 0, gShowcasePal);
|
|
jlScbSetRange(gStage, 0, 199, 0);
|
|
jlSurfaceClear(gStage, SC_BG_COLOR);
|
|
|
|
// Top strip: all 16 palette entries as swatches.
|
|
for (k = 0; k < SC_PAL_COUNT; k++) {
|
|
jlFillRect(gStage, (int16_t)(k * SC_PAL_SWATCH_W), 0, SC_PAL_SWATCH_W, SC_PAL_STRIP_H, (uint8_t)k);
|
|
}
|
|
|
|
for (cell = 0; cell < SC_CELLS; cell++) {
|
|
showcaseCellRect(cell, &x, &y, &w, &h);
|
|
jlDrawRect(gStage, x, y, (uint16_t)w, (uint16_t)h, SC_BORDER_COLOR);
|
|
ix = (int16_t)(x + SC_INSET);
|
|
iy = (int16_t)(y + SC_INSET);
|
|
iw = (int16_t)(w - 2 * SC_INSET);
|
|
ih = (int16_t)(h - 2 * SC_INSET);
|
|
cx = (int16_t)(ix + iw / 2);
|
|
cy = (int16_t)(iy + ih / 2);
|
|
minDim = (iw < ih) ? iw : ih;
|
|
|
|
switch (cell) {
|
|
case 0:
|
|
// Pixels: a scatter of single plotted pixels.
|
|
for (py = iy; py < iy + ih; py += 3) {
|
|
for (px = ix; px < ix + iw; px += 3) {
|
|
jlDrawPixel(gStage, px, py, (uint8_t)(2 + ((px + py) % 14)));
|
|
}
|
|
}
|
|
break;
|
|
case 1:
|
|
// Horizontal lines.
|
|
for (k = 0; k < ih; k += 4) {
|
|
jlDrawLine(gStage, ix, (int16_t)(iy + k), (int16_t)(ix + iw - 1), (int16_t)(iy + k), (uint8_t)(2 + (k / 4) % 14));
|
|
}
|
|
break;
|
|
case 2:
|
|
// Vertical lines.
|
|
for (k = 0; k < iw; k += 4) {
|
|
jlDrawLine(gStage, (int16_t)(ix + k), iy, (int16_t)(ix + k), (int16_t)(iy + ih - 1), (uint8_t)(2 + (k / 4) % 14));
|
|
}
|
|
break;
|
|
case 3:
|
|
// Diagonals: an X plus a fan from the center.
|
|
jlDrawLine(gStage, ix, iy, (int16_t)(ix + iw - 1), (int16_t)(iy + ih - 1), 5);
|
|
jlDrawLine(gStage, ix, (int16_t)(iy + ih - 1), (int16_t)(ix + iw - 1), iy, 6);
|
|
for (k = 0; k < iw; k += 8) {
|
|
jlDrawLine(gStage, cx, cy, (int16_t)(ix + k), iy, (uint8_t)(8 + (k / 8) % 8));
|
|
}
|
|
break;
|
|
case 4:
|
|
// Rectangle outlines, concentric.
|
|
for (k = 0; 2 * k < minDim - 4; k += 5) {
|
|
jlDrawRect(gStage, (int16_t)(ix + k), (int16_t)(iy + k), (uint16_t)(iw - 2 * k), (uint16_t)(ih - 2 * k), (uint8_t)(2 + (k / 5) % 14));
|
|
}
|
|
break;
|
|
case 5:
|
|
// Filled rectangles, overlapping.
|
|
jlFillRect(gStage, ix, iy, (uint16_t)(iw * 2 / 3), (uint16_t)(ih * 2 / 3), 2);
|
|
jlFillRect(gStage, (int16_t)(ix + iw / 3), (int16_t)(iy + ih / 3), (uint16_t)(iw * 2 / 3), (uint16_t)(ih * 2 / 3), 4);
|
|
break;
|
|
case 6:
|
|
// Circle outlines, concentric.
|
|
for (r = (int16_t)(minDim / 2); r > 2; r -= 4) {
|
|
jlDrawCircle(gStage, cx, cy, (uint16_t)r, (uint8_t)(2 + (r / 4) % 14));
|
|
}
|
|
break;
|
|
case 7:
|
|
// Filled circles.
|
|
jlFillCircle(gStage, cx, cy, (uint16_t)(minDim / 2 - 1), 8);
|
|
jlFillCircle(gStage, cx, cy, (uint16_t)(minDim / 4), 5);
|
|
break;
|
|
case 8:
|
|
// Tiles: an 8x8-block checkerboard inside the cell.
|
|
bxStart = (uint8_t)((ix + 7) / 8);
|
|
bxEnd = (uint8_t)((ix + iw) / 8);
|
|
byStart = (uint8_t)((iy + 7) / 8);
|
|
byEnd = (uint8_t)((iy + ih) / 8);
|
|
for (by = byStart; by < byEnd; by++) {
|
|
for (bx = bxStart; bx < bxEnd; bx++) {
|
|
jlTileFill(gStage, bx, by, (uint8_t)(((bx + by) & 1) ? 6 : 8));
|
|
}
|
|
}
|
|
break;
|
|
case 9:
|
|
// Sprite: the compiled ball at a few positions.
|
|
jlSpriteDraw(gStage, gSprite, ix, iy);
|
|
jlSpriteDraw(gStage, gSprite, (int16_t)(ix + iw - 16), (int16_t)(iy + ih - 16));
|
|
jlSpriteDraw(gStage, gSprite, (int16_t)(cx - 8), (int16_t)(cy - 8));
|
|
break;
|
|
case 10:
|
|
// Flood fill: outline a circle, then flood its interior.
|
|
jlDrawCircle(gStage, cx, cy, (uint16_t)(minDim / 2 - 2), 1);
|
|
jlFloodFill(gStage, cx, cy, 12);
|
|
break;
|
|
default:
|
|
// Mini scene: ground, sun, horizon, ball.
|
|
jlFillRect(gStage, ix, (int16_t)(iy + ih / 2), (uint16_t)iw, (uint16_t)(ih - ih / 2), 4);
|
|
jlFillCircle(gStage, cx, (int16_t)(iy + ih / 3), (uint16_t)(ih / 4), 5);
|
|
jlDrawLine(gStage, ix, (int16_t)(iy + ih / 2), (int16_t)(ix + iw - 1), (int16_t)(iy + ih / 2), 1);
|
|
jlSpriteDraw(gStage, gSprite, (int16_t)(cx - 8), (int16_t)(iy + ih / 2 - 16));
|
|
break;
|
|
}
|
|
}
|
|
|
|
jlLogF("UBER: showcase cells: 0=pixels 1=lineH 2=lineV 3=diag 4=rect 5=fillRect 6=circle 7=fillCircle 8=tiles 9=sprite 10=flood 11=scene\n");
|
|
jlStagePresent();
|
|
|
|
// Hold the showcase on screen, then auto-advance to the benchmark so
|
|
// the headless perf-capture run still completes without a keypress.
|
|
held = jlFrameCount();
|
|
while ((uint16_t)(jlFrameCount() - held) < SC_HOLD_FRAMES) {
|
|
/* hold */
|
|
}
|
|
}
|
|
|
|
|
|
// ----- Non-timed correctness checks (Phase 0 verification harness) -----
|
|
//
|
|
// Each check draws a deterministic scene and logs the surface hash so
|
|
// tools/diff-uber-hashes can compare ports against each other and against
|
|
// the frozen golden logs. Several checks intentionally capture the CURRENT
|
|
// behavior of known bugs (PERF-AUDIT.md #1, #11, #12, #72): their hash
|
|
// lines are EXPECTED to change when the Phase 1 fixes land -- re-golden
|
|
// exactly those lines then, nothing else.
|
|
//
|
|
// PASS/FAIL lines assert invariants that must hold on every port both
|
|
// before and after the fixes (round-trips, forced palette color 0, the
|
|
// PRNG golden sequence, arena bookkeeping).
|
|
|
|
static void __attribute__((noinline)) chkHash(const char *name) {
|
|
jlLogF("UBER-CHK: %s: hash=%08lX\n", name, (unsigned long)jlSurfaceHash(gStage));
|
|
}
|
|
|
|
|
|
static void __attribute__((noinline)) chkPassFail(const char *name, bool pass) {
|
|
jlLogF("UBER-CHK: %s: %s\n", name, pass ? "PASS" : "FAIL");
|
|
}
|
|
|
|
|
|
// Edge-coordinate draws. The off-surface pixel coords stay modest (within
|
|
// ~200 rows of the surface) so the pre-fix dirty-band overwrite (finding #1,
|
|
// non-IIgs) lands inside the paired band arrays instead of unrelated
|
|
// globals; Phase 1 turns these into true no-ops.
|
|
static void __attribute__((noinline)) checkEdgeDraws(void) {
|
|
uberMbOp(101);
|
|
jlDrawPixel(gStage, -5, 100, 5);
|
|
jlDrawPixel(gStage, 330, 100, 5);
|
|
jlDrawPixel(gStage, 100, -3, 5);
|
|
jlDrawPixel(gStage, 100, 210, 5);
|
|
chkHash("edge-pixels");
|
|
// w=65535 draws phantom edges today (finding #11); h==2 exercises the
|
|
// zero-height interior-edge fills; the third rect clips normally.
|
|
jlDrawRect(gStage, 10, 10, 65535u, 100, 5);
|
|
jlDrawRect(gStage, 50, 20, 30, 2, 6);
|
|
jlDrawRect(gStage, -10, 150, 340, 40, 7);
|
|
chkHash("edge-drawRect");
|
|
// r=300 clips every span; r=40000 should cover the surface but draws
|
|
// nothing today (finding #12).
|
|
jlFillCircle(gStage, 160, 100, 300, 4);
|
|
jlFillCircle(gStage, 160, 100, 40000u, 9);
|
|
jlDrawCircle(gStage, 10, 10, 40000u, 3);
|
|
chkHash("edge-circles");
|
|
// Far-off-surface endpoints. The H line should fill the whole visible
|
|
// row but draws nothing today (finding #72: the H/V fast path clamps
|
|
// span to 320 BEFORE clipping, so a huge span anchored off-surface
|
|
// clips away entirely). The diagonal clips per-pixel and is correct.
|
|
jlDrawLine(gStage, -20000, 50, 20000, 50, 2);
|
|
jlDrawLine(gStage, 60, -20000, 60, 20000, 2);
|
|
jlDrawLine(gStage, -300, -200, 620, 400, 8);
|
|
chkHash("edge-lines");
|
|
}
|
|
|
|
|
|
// Clipped sprite draws take the interpreted path on every port (the
|
|
// compiled routines require fully-on-surface): the ONLY cross-port
|
|
// coverage of that path, which UBER's timed ops (on-surface, compiled)
|
|
// never touch. Also asserts the save/draw/restore round-trip restores
|
|
// the exact pre-save pixels at a clipped position.
|
|
static void __attribute__((noinline)) checkClippedSprites(void) {
|
|
uint32_t before;
|
|
|
|
uberMbOp(102);
|
|
jlSpriteDraw(gStage, gSprite, -8, 50);
|
|
jlSpriteDraw(gStage, gSprite, 312, 50);
|
|
jlSpriteDraw(gStage, gSprite, 150, -8);
|
|
jlSpriteDraw(gStage, gSprite, 150, 192);
|
|
jlSpriteDraw(gStage, gSprite, -8, -8);
|
|
jlSpriteDraw(gStage, gSprite, 400, 100);
|
|
chkHash("sprite-clipped");
|
|
before = jlSurfaceHash(gStage);
|
|
jlSpriteSaveUnder(gStage, gSprite, -8, 100, &gBackup);
|
|
jlSpriteDraw(gStage, gSprite, -8, 100);
|
|
jlSpriteRestoreUnder(gStage, &gBackup);
|
|
chkPassFail("sprite-clip-roundtrip", jlSurfaceHash(gStage) == before);
|
|
}
|
|
|
|
|
|
static void __attribute__((noinline)) checkTileMonoFlood(void) {
|
|
static jlTileT mono;
|
|
uint8_t i;
|
|
|
|
uberMbOp(103);
|
|
for (i = 0; i < TILE_BYTES; i++) {
|
|
mono.pixels[i] = (uint8_t)((i & 1) ? 0x0F : 0xF0);
|
|
}
|
|
jlTilePasteMono(gStage, 10, 10, &mono, 5, 9);
|
|
jlTilePasteMono(gStage, 11, 10, &mono, 14, 0);
|
|
chkHash("tilePasteMono");
|
|
jlDrawRect(gStage, 240, 150, 40, 30, 1);
|
|
jlFloodFill(gStage, 250, 160, 12);
|
|
chkHash("floodFill");
|
|
}
|
|
|
|
|
|
static void __attribute__((noinline)) checkDrawText(void) {
|
|
static uint16_t asciiMap[256];
|
|
uint16_t i;
|
|
|
|
uberMbOp(104);
|
|
for (i = 0; i < 256u; i++) {
|
|
asciiMap[i] = TILE_NO_GLYPH;
|
|
}
|
|
// 'A' -> tile (5,5), 'B' -> tile (6,6). Seed both glyph tiles here --
|
|
// the section-opening surface clear wiped whatever the timed tile ops
|
|
// left there.
|
|
jlTileFill(gStage, 5, 5, 9);
|
|
jlTileFill(gStage, 6, 6, 3);
|
|
asciiMap['A'] = (uint16_t)(5u | (5u << 8));
|
|
asciiMap['B'] = (uint16_t)(6u | (6u << 8));
|
|
jlDrawText(gStage, 2, 20, gStage, asciiMap, "ABBA");
|
|
chkHash("drawText");
|
|
// Out-of-range start: the entry sanitation loop (Phase 5, #22)
|
|
// reproduces the historical absorb-and-wrap semantics bit-for-bit
|
|
// (the first glyph is consumed without drawing, the second wraps
|
|
// to (0,21) and draws) while guaranteeing jlpTileCopyMasked never
|
|
// sees an out-of-range destination. Separate hash line so any
|
|
// future semantic change stays isolated.
|
|
jlDrawText(gStage, 200, 20, gStage, asciiMap, "AB");
|
|
chkHash("drawText-offgrid");
|
|
}
|
|
|
|
|
|
static void __attribute__((noinline)) checkPalette(void) {
|
|
static const uint16_t conforming[16] = {
|
|
0x0ABC, 0x0111, 0x0222, 0x0333, 0x0444, 0x0555, 0x0666, 0x0777,
|
|
0x0888, 0x0999, 0x0AAA, 0x0BBB, 0x0CCC, 0x0DDD, 0x0EEE, 0x0123
|
|
};
|
|
uint16_t readBack[16];
|
|
uint16_t i;
|
|
bool ok;
|
|
|
|
// Contract invariants (stable across the Phase 2 #16 change): color 0
|
|
// is forced to $000 even when the caller passes nonzero, and
|
|
// conforming $0RGB entries 1..15 round-trip exactly.
|
|
uberMbOp(105);
|
|
jlPaletteSet(gStage, 2, conforming);
|
|
jlPaletteGet(gStage, 2, readBack);
|
|
ok = (readBack[0] == 0x0000u);
|
|
for (i = 1; i < 16u; i++) {
|
|
if (readBack[i] != conforming[i]) {
|
|
ok = false;
|
|
}
|
|
}
|
|
chkPassFail("palette-roundtrip", ok);
|
|
}
|
|
|
|
|
|
static void __attribute__((noinline)) checkRandom(void) {
|
|
static const uint32_t expected[4] = {
|
|
0x87985AA5UL, 0x155B24A3UL, 0x4820F4C4UL, 0x81B3AC98UL
|
|
};
|
|
uint16_t i;
|
|
bool ok;
|
|
|
|
// xorshift32 golden sequence from a fixed seed; bit-identical on
|
|
// every port and across the Phase 8 (#43) IIgs rewrite.
|
|
uberMbOp(106);
|
|
jlRandomSeed(0x12345678UL);
|
|
ok = true;
|
|
for (i = 0; i < 4u; i++) {
|
|
if (jlRandom() != expected[i]) {
|
|
ok = false;
|
|
}
|
|
}
|
|
if (jlRandomRange(100u) != 43u) {
|
|
ok = false;
|
|
}
|
|
chkPassFail("random-golden", ok);
|
|
jlRandomSeed(1u);
|
|
}
|
|
|
|
|
|
// Arena churn: create/compile/destroy so a hole opens and is reused, then
|
|
// compact and assert the used-byte counter returns to its pre-churn value.
|
|
// Covers the codegen allocator paths UBER's single long-lived sprite never
|
|
// exercises (findings #5, #34 land here in Phase 1). Also runs the
|
|
// owned-tileData path (jlSpriteCreateFromSurface) that finding #31
|
|
// re-allocates in Phase 1, and a sprite-bank load if an asset is present.
|
|
static void __attribute__((noinline)) checkAllocator(void) {
|
|
jlSpriteT *a;
|
|
jlSpriteT *b;
|
|
jlSpriteT *c;
|
|
uint32_t usedBefore;
|
|
|
|
// Sub-markers 111+ pinpoint the statement that hangs (Phase 1
|
|
// stabilization; the coarse marker froze on this check).
|
|
uberMbOp(111);
|
|
usedBefore = jlSpriteCodegenBytesUsed();
|
|
a = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y);
|
|
b = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y);
|
|
if (a == NULL || b == NULL) {
|
|
chkPassFail("arena-churn", false);
|
|
return;
|
|
}
|
|
uberMbOp(112);
|
|
(void)jlSpriteCompile(a);
|
|
(void)jlSpriteCompile(b);
|
|
uberMbOp(113);
|
|
jlSpriteDestroy(a);
|
|
uberMbOp(114);
|
|
c = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y);
|
|
if (c == NULL) {
|
|
jlSpriteDestroy(b);
|
|
chkPassFail("arena-churn", false);
|
|
return;
|
|
}
|
|
(void)jlSpriteCompile(c);
|
|
uberMbOp(115);
|
|
jlSpriteDestroy(b);
|
|
jlSpriteDestroy(c);
|
|
uberMbOp(116);
|
|
jlSpriteCompact();
|
|
uberMbOp(117);
|
|
chkPassFail("arena-churn", jlSpriteCodegenBytesUsed() == usedBefore);
|
|
uberMbOp(118);
|
|
a = jlSpriteCreateFromSurface(gStage, 0, 0, 2, 2);
|
|
if (a != NULL) {
|
|
(void)jlSpriteCompile(a);
|
|
uberMbOp(119);
|
|
jlSpriteDraw(gStage, a, 280, 20);
|
|
jlSpriteDestroy(a);
|
|
}
|
|
chkPassFail("sprite-from-surface", a != NULL);
|
|
uberMbOp(120);
|
|
chkHash("sprite-from-surface-draw");
|
|
}
|
|
|
|
|
|
// Coverage the primary checkTileMonoFlood golden misses (found in the
|
|
// #3/#4 re-analysis): the alternating 0x0F/0xF0 mono pattern only
|
|
// selects IIgs #3 combo indices 1 and 2, and jlFloodFill only drives
|
|
// the matchEqual branch of the ST #4 plane hooks. Runs LAST so it
|
|
// perturbs no earlier cumulative full-stage hash.
|
|
static void __attribute__((noinline)) checkTileMonoFloodExtra(void) {
|
|
// 0x00 -> combo idx 0 (bg:bg), 0xFF -> idx 3 (fg:fg), 0x0F -> idx 1
|
|
// (bg:fg), 0xF0 -> idx 2 (fg:bg): all four opacity pairs in one tile.
|
|
static const uint8_t comboBytes[4] = { 0x00u, 0xFFu, 0x0Fu, 0xF0u };
|
|
static jlTileT monoAll;
|
|
uint8_t i;
|
|
|
|
uberMbOp(121);
|
|
for (i = 0; i < TILE_BYTES; i++) {
|
|
monoAll.pixels[i] = comboBytes[i & 3u];
|
|
}
|
|
jlTilePasteMono(gStage, 10, 12, &monoAll, 5, 9);
|
|
jlTilePasteMono(gStage, 11, 12, &monoAll, 14, 0);
|
|
chkHash("tilePasteMonoAll");
|
|
|
|
// Bounded flood: an enclosed color-3 box over a pre-cleared interior
|
|
// so the fill stays contained and drives the bounded stop mask
|
|
// (pix == boundary || pix == new) plus the planar group-skip across
|
|
// several 16px groups. matchColor = boundaryColor = 3, matchEqual = 0.
|
|
jlFillRect(gStage, 40, 140, 64, 40, 0);
|
|
jlDrawRect(gStage, 44, 144, 52, 32, 3);
|
|
jlFloodFillBounded(gStage, 70, 160, 7, 3);
|
|
chkHash("floodFillBounded");
|
|
}
|
|
|
|
|
|
static void __attribute__((noinline)) runCorrectnessChecks(void) {
|
|
uberMbPhase(UBER_PHASE_CHECKS);
|
|
jlLogF("UBER-CHK: ----- begin -----\n");
|
|
jlSurfaceClear(gStage, 0);
|
|
checkEdgeDraws();
|
|
checkClippedSprites();
|
|
checkTileMonoFlood();
|
|
checkDrawText();
|
|
checkPalette();
|
|
checkRandom();
|
|
checkAllocator();
|
|
checkTileMonoFloodExtra();
|
|
// jlShutdown -> jlInit -> stale-sprite-destroy (#34) needs a teardown
|
|
// of the whole library mid-run; it gets a dedicated micro-example in
|
|
// Phase 1 rather than risking the benchmark's display state here.
|
|
jlLogF("UBER-CHK: ----- end -----\n");
|
|
}
|
|
|
|
|
|
// ----- Main -----
|
|
|
|
static void __attribute__((noinline)) runAllTests(void) {
|
|
uberMbPhase(UBER_PHASE_TIMED_RUN);
|
|
jlLogF("UBER: ----- begin -----\n");
|
|
|
|
// Surface / palette / SCB.
|
|
timeOp("jlSurfaceClear", op_surfaceClear);
|
|
timeOp("jlPaletteSet", op_paletteSet);
|
|
timeOp("jlScbSetRange", op_scbSetRange);
|
|
|
|
// Drawing primitives.
|
|
timeOp("jlDrawPixel", op_drawPixel);
|
|
timeOp("jlDrawLine H", op_drawLineH);
|
|
timeOp("jlDrawLine V", op_drawLineV);
|
|
timeOp("jlDrawLine diag", op_drawLineDiag);
|
|
timeOp("jlDrawRect 100x100", op_drawRect);
|
|
timeOp("jlDrawCircle r=16", op_drawCircleSmall);
|
|
timeOp("jlDrawCircle r=80", op_drawCircleLarge);
|
|
timeOp("jlFillRect 16x16", op_fillRectSmall);
|
|
timeOp("jlFillRect 80x80", op_fillRectMid);
|
|
timeOp("jlFillRect 320x200", op_fillRectFull);
|
|
timeOp("jlFillCircle r=40", op_fillCircle);
|
|
timeOp("jlSamplePixel", op_samplePixel);
|
|
|
|
// Tiles. Seed scratch tile + dest cells with non-zero pixels first.
|
|
jlFillRect(gStage, 0, 0, 320, 64, 7);
|
|
jlTileSnap(gStage, 5, 5, &gTileScratch);
|
|
timeOp("jlTileFill", op_tileFill);
|
|
timeOp("jlTileCopy", op_tileCopy);
|
|
timeOp("jlTileCopyMasked", op_tileCopyMasked);
|
|
timeOp("jlTilePaste", op_tilePaste);
|
|
timeOp("jlTileSnap", op_tileSnap);
|
|
|
|
// Sprites. Background must be non-empty so save-under has work
|
|
// to do (otherwise it's a 4 KB memset of zeros, atypical).
|
|
jlSurfaceClear(gStage, 4);
|
|
timeOp("jlSpriteSaveUnder", op_spriteSave);
|
|
timeOp("jlSpriteDraw", op_spriteDraw);
|
|
timeOp("jlSpriteRestoreUnder", op_spriteRestore);
|
|
timeOp("jlSpriteSaveAndDraw", op_spriteSaveAndDraw);
|
|
|
|
// Present. One warm-up call before each timed loop primes any
|
|
// per-port one-time setup (Amiga: copper list rebuild after the
|
|
// jlPaletteSet / jlScbSetRange tests dirty the cache; without warm-up
|
|
// the rebuild's MakeScreen + MrgCop + WaitTOF chain consumes the
|
|
// entire 4-frame measurement window) so we measure steady-state
|
|
// throughput rather than first-call penalty.
|
|
jlStagePresent();
|
|
timeOp("jlStagePresent full", op_stagePresent);
|
|
|
|
// Input.
|
|
timeOp("jlInputPoll", op_inputPoll);
|
|
timeOp("jlKeyDown", op_keyDown);
|
|
timeOp("jlKeyPressed", op_keyPressed);
|
|
timeOp("jlMouseX", op_mouseX);
|
|
timeOp("joeyJoyConnected", op_joyConnected);
|
|
|
|
// Audio.
|
|
timeOp("jlAudioFrameTick", op_audioFrameTick);
|
|
timeOp("jlAudioIsPlayingMod", op_audioIsPlaying);
|
|
|
|
// Surface mark dirty (via jlFillRect's mark step).
|
|
timeOp("surfaceMarkDirtyRect (via jlFillRect 32x32)", op_surfaceMarkDirty);
|
|
|
|
// ----- Phase 0 workload rows (NATIVE-PERF-PLAN.md item 3). Appended
|
|
// last on purpose: timed hashes are cumulative and the correctness
|
|
// checks clear the stage first, so the frozen lines above stay
|
|
// bit-identical (Phase-7 golden-extension precedent).
|
|
timeOp("jlSpriteDraw unaligned", op_spriteDrawUnaligned);
|
|
timeOp("jlSpriteDraw sweep16", op_spriteDrawSweep);
|
|
|
|
// Seed two patterned tiles via portable draws + snap. jlTileT.pixels
|
|
// is platform-private (plane bytes on ST/Amiga), so hand-written
|
|
// bytes would render differently per port; only a snap round-trips.
|
|
jlFillRect(gStage, 0, 72, 4, 4, 3);
|
|
jlFillRect(gStage, 4, 72, 4, 4, 5);
|
|
jlFillRect(gStage, 0, 76, 4, 4, 5);
|
|
jlFillRect(gStage, 4, 76, 4, 4, 3);
|
|
jlTileSnap(gStage, 0, 9, &gTileMapA);
|
|
jlFillRect(gStage, 8, 72, 8, 8, 9);
|
|
jlDrawLine(gStage, 8, 72, 15, 79, 1);
|
|
jlTileSnap(gStage, 1, 9, &gTileMapB);
|
|
gFrameBackups[0].bytes = gFrameBackupBytes[0];
|
|
gFrameBackups[1].bytes = gFrameBackupBytes[1];
|
|
gFrameBackups[2].bytes = gFrameBackupBytes[2];
|
|
|
|
timeOp("jlTilePaste map 10x5", op_tileMapPaste);
|
|
timeOp("gameFrame composite", op_gameFrame);
|
|
|
|
jlLogF("UBER: ----- end -----\n");
|
|
}
|
|
|
|
|
|
// Extracted from main so main's register pressure stays under the w65816
|
|
// allocator's ceiling -- the 16-entry pal[] + loop index was the overflow.
|
|
// noinline is load-bearing at -O2 (the backend would otherwise re-inline a
|
|
// single-call static and recreate the pressure).
|
|
static void __attribute__((noinline)) setupPalette(void) {
|
|
uint16_t pal[16];
|
|
int i;
|
|
|
|
for (i = 0; i < 16; i++) {
|
|
pal[i] = (uint16_t)((i << 8) | (i << 4) | i); // grey ramp
|
|
}
|
|
pal[ 0] = 0x000;
|
|
pal[ 1] = 0x800; // dark red (running)
|
|
pal[ 2] = 0x080; // green (done)
|
|
pal[ 3] = 0x008; // blue
|
|
pal[ 5] = 0xFF0; // yellow (test pixels)
|
|
pal[ 7] = 0xFFF; // white (fills)
|
|
pal[15] = 0xF00; // red
|
|
jlPaletteSet(gStage, 0, pal);
|
|
jlScbSetRange(gStage, 0, 199, 0);
|
|
}
|
|
|
|
|
|
// Extracted from main for the same register-pressure reason as setupPalette.
|
|
static void __attribute__((noinline)) reportElapsed(uint16_t startFrame) {
|
|
uint16_t endFrame;
|
|
uint16_t elapsedFrames;
|
|
unsigned long elapsedMs;
|
|
|
|
endFrame = jlFrameCount();
|
|
elapsedFrames = (uint16_t)(endFrame - startFrame);
|
|
elapsedMs = ((unsigned long)elapsedFrames * 1000UL) / (unsigned long)jlFrameHz();
|
|
jlLogF("UBER: total wall time: %lu ms (%u frames @ %u Hz)\n",
|
|
elapsedMs, elapsedFrames, (unsigned)jlFrameHz());
|
|
}
|
|
|
|
|
|
// Extracted from main (register pressure). Returns false on sprite-create fail.
|
|
static bool __attribute__((noinline)) setupSprite(void) {
|
|
uint16_t before;
|
|
|
|
uberMbPhase(UBER_PHASE_SETUP_SPRITE);
|
|
buildBallSprite();
|
|
gSprite = jlSpriteCreate(gBallTiles, BALL_TILES_X, BALL_TILES_Y);
|
|
if (gSprite == NULL) {
|
|
jlLog("UBER: jlSpriteCreate failed");
|
|
return false;
|
|
}
|
|
// jlSpriteCompile is a one-shot. Time at frame resolution.
|
|
jlWaitVBL();
|
|
before = jlFrameCount();
|
|
if (!jlSpriteCompile(gSprite)) {
|
|
jlLog("UBER: jlSpriteCompile failed");
|
|
}
|
|
while (jlFrameCount() == before) {
|
|
/* wait for next VBL edge */
|
|
}
|
|
jlLogF("UBER: jlSpriteCompile: 1 call in <= 1 frame\n");
|
|
gBackup.bytes = gBackupBytes;
|
|
return true;
|
|
}
|
|
|
|
|
|
// Extracted from main (register pressure): the jlConfigT struct on main's
|
|
// frame, on top of the rest of main, pushed the w65816 allocator over its
|
|
// limit. noinline keeps it out at -O2.
|
|
static bool __attribute__((noinline)) initJoeyLib(void) {
|
|
jlConfigT config;
|
|
|
|
/* 32 KB fits one 16x16 sprite's full compiled set on either 68k
|
|
* port (Amiga: 8 pre-shifted DRAW variants ~15-18 KB; ST: 16-phase
|
|
* thunk+table DRAWs ~12 KB; plus ~1-2 KB of save/restore classes).
|
|
* UL on the multiply because a 16-bit int overflows on 32 * 1024. */
|
|
config.codegenBytes = 32UL * 1024;
|
|
config.audioBytes = 64UL * 1024;
|
|
return jlInit(&config);
|
|
}
|
|
|
|
|
|
int main(void) {
|
|
uint16_t startFrame;
|
|
|
|
if (!initJoeyLib()) {
|
|
return 1;
|
|
}
|
|
/* jlFrameCount is VBL-driven, so it only ticks after halInit
|
|
* installed its VBL ISR -- captured here is "everything from now
|
|
* to press-any-key". Pre-init setup time is small and not the
|
|
* cost the user is chasing; runAllTests dominates. */
|
|
startFrame = jlFrameCount();
|
|
|
|
gStage = jlStageGet();
|
|
if (gStage == NULL) {
|
|
jlShutdown();
|
|
return 1;
|
|
}
|
|
uberMbLayout();
|
|
|
|
// A simple visible palette so users see SOMETHING during the run.
|
|
setupPalette();
|
|
|
|
// Indicate "running": red bar at top of screen.
|
|
jlSurfaceClear(gStage, 0);
|
|
jlFillRect(gStage, 0, 0, 320, 8, 1);
|
|
jlStagePresent();
|
|
|
|
if (!setupSprite()) {
|
|
jlShutdown();
|
|
return 1;
|
|
}
|
|
#ifdef JOEYLIB_PLATFORM_IIGS
|
|
*UBER_MB_SPRITEPTR = (uint32_t)gSprite;
|
|
#endif
|
|
|
|
// Audio: only init/shutdown is exercised. Triggering jlAudioPlaySfx
|
|
// without first calling jlAudioPlayMod leaves NTP's engine in a
|
|
// half-initialized state -- NTPstreamsound is designed to OVERLAY on
|
|
// an already-running module. Without NTPprepare/NTPplay first, the
|
|
// streamer oscillator is fired but no music tick ever advances or
|
|
// silences it, and you get a stuck high-pitched scream. UBER doesn't
|
|
// ship a MOD asset, so we skip the SFX exercise. The frame-tick and
|
|
// isPlayingMod calls below still get timed (both are no-op fast
|
|
// paths on IIgs).
|
|
uberMbPhase(UBER_PHASE_AUDIO_INIT);
|
|
if (jlAudioInit()) {
|
|
jlLogF("UBER: audioInit OK\n");
|
|
} else {
|
|
jlLogF("UBER: audioInit failed (skipping audio)\n");
|
|
}
|
|
|
|
// Visual showcase: render one of each primitive into its own grid
|
|
// cell so the run shows something legible (the timed benchmark below
|
|
// never presents, so this stays on screen throughout it). The first
|
|
// timed op is jlSurfaceClear, so this leaves the benchmark untouched.
|
|
drawShowcase();
|
|
|
|
runAllTests();
|
|
|
|
runCorrectnessChecks();
|
|
|
|
reportElapsed(startFrame);
|
|
|
|
// Done. Green screen + waitForKey.
|
|
uberMbPhase(UBER_PHASE_DONE);
|
|
jlSurfaceClear(gStage, 2);
|
|
jlStagePresent();
|
|
|
|
jlLogF("UBER: press any key to exit\n");
|
|
// Flush the log to disk BEFORE the blocking key wait, so an automated
|
|
// (headless) run that kills the process at this prompt still captures
|
|
// the results -- joeyLog otherwise only flushes at the atexit fclose.
|
|
jlLogFlush();
|
|
jlWaitForAnyKey();
|
|
|
|
jlSpriteDestroy(gSprite);
|
|
jlShutdown();
|
|
return 0;
|
|
}
|