singe/src/scene.c
2026-09-22 18:32:07 -05:00

6365 lines
237 KiB
C

/*
*
* Singe 3
* Copyright (C) 2006-2026 Scott Duensing <scott@kangaroopunch.com>
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License
* as published by the Free Software Foundation; either version 3
* of the License, or (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA
* 02110-1301, USA.
*
*/
// The 3D scene: a layer drawn between the disc video and the 2D overlay. It renders on the
// SDL_GPU device the 2D renderer was created on, into its own colour and depth textures; the colour
// texture is wrapped as an SDL_Texture so the frame loop composites it like any other layer.
//
// Everything in the scene is a node in one tree (node 0 is the root): a node has a transform, and
// optionally a mesh with a material, or a light. Meshes, materials and nodes are addressed from
// Lua by integer handles that index the arrays below; a freed slot is reused. Shaders come
// precompiled from the generated sceneShaders.h (cmake/shaderHeader.cmake builds it from
// src/shaders/scene.hlsl); the limits and codes the uniform blocks share with them live in
// sceneShared.h.
#include <float.h>
#include <math.h>
#include <stdlib.h>
#include <string.h>
#include "util.h"
#include "render.h"
#include "scene.h"
#include "shaders/sceneShaders.h"
#include "particles.h"
#include "gui.h"
#define COLOUR_MAX 255.0f
#define MAX_ANISOTROPY 8.0f // Texture samples along a grazing surface
#define SAMPLER_MAX_LOD 1000.0f // "No clamp", SDL's own spelling of it: walk the whole mipmap chain
#define POST_VERTICES 3 // One triangle covers the screen
#define BLOOM_LEVELS 5 // Half-size chain for the glow
#define BLOOM_LEVELS_MIN 2 // Fewer and the up pass has nothing to add (tiny targets)
#define BLOOM_MIN_SIZE 8 // Smallest level, pixels
#define VIEW_SIZE_MAX 4096 // Pixels per side of a rendered view
#define TEXTURE_SIZES_STEP 64 // Growth of the texture size table
#define DEFAULT_BLOOM_THRESHOLD 1.0f
#define MATERIAL_SAMPLERS 7 // Base, shadows, normal, occlusion, metallic-roughness, emissive, sky
#define FRAME_UNIFORMS 0 // Fragment uniform slots: the pass's FragmentUniformsT ...
#define MATERIAL_UNIFORMS 1 // ... and the batch's MaterialUniformsT
#define DRAW_UNIFORMS 0 // Vertex uniform slots: DrawUniformsT ...
#define SKIN_UNIFORMS 1 // ... and SkinUniformsT
#define CUBE_FACE_MIN 16 // Sky cube face sizes, a power of two from the source's height
#define CUBE_FACE_MAX 1024
#define SH_SAMPLES_ACROSS 128 // Equirect columns sampled for the harmonics
#define HALF_BYTES 2
#define CUBE_CHANNELS 4
#define PIPELINE_COUNT 16 // skinned x blend x double sided x occluder
#define PIPELINE_SKINNED 1
#define PIPELINE_BLEND 2
#define PIPELINE_TWO_SIDED 4
#define PIPELINE_OCCLUDER 8 // Writes depth and no colour: a stand-in for something painted
#define SHADOW_PIPELINES 8 // skinned x double sided x cutout
#define SHADOW_PIPELINE_SKINNED 1
#define SHADOW_PIPELINE_TWO_SIDED 2
#define SHADOW_PIPELINE_CUTOUT 4
#define SAMPLE_SETS 2 // Pipelines per target sample count ...
#define SAMPLE_SET_SINGLE 0 // ... single sample (views, and the window without antialiasing) ...
#define SAMPLE_SET_MULTI 1 // ... and the window's multisampled targets
#define MESH_ATTRIBUTES 6 // Vertex attributes of a SceneVertexT
#define PARTICLE_PIPELINES 2 // PARTICLE_ALPHA, PARTICLE_ADD
#define PARTICLE_VERTICES 6 // Two triangles per particle, unindexed
#define DYNAMIC_BUFFER_MIN 65536 // Bytes: the smallest per-frame vertex buffer
#define PARTICLE_DRAW_MAX 64 // 3D emitters drawn per frame
#define PARTICLE_FRAMES_MAX 16 // Frames (runs) per emitter the run table allows for
#define PARTICLE_RUN_MAX (PARTICLE_DRAW_MAX * PARTICLE_FRAMES_MAX) // Runs beyond this are dropped
#define DEFAULT_FOV 60.0f
#define DEFAULT_NEAR 0.1f
#define DEFAULT_FAR 1000.0f
#define DEFAULT_EYE_Z 5.0f // The default camera, looking at the origin from +Z
#define DEFAULT_ORTHO_HEIGHT 5.0f // World units the default orthographic view spans vertically
#define DEFAULT_ROUGHNESS 0.5f
#define DEFAULT_CONE_INNER 20.0f // A new spot light's cone, degrees
#define DEFAULT_CONE_OUTER 30.0f
#define MIN_SEGMENTS 3
#define NO_HANDLE -1
#define SHADOW_NEAR_MIN 0.01f
#define SHADOW_FAR_MIN 0.02f // A point or spot shadow's frustum, however close its casters
#define SHADOW_NEAR_FRACTION 0.5f // Near no more than this far along a point or spot shadow's frustum
#define SPOT_SHADOW_FOV_MAX 170.0f // A spot shadow's perspective, degrees; wider is unusable
#define SHADOW_DEPTH_BIAS_CONSTANT 2.0f // Rasterizer bias in the shadow passes against self-shadowing
#define SHADOW_DEPTH_BIAS_SLOPE 2.0f
#define UP_PARALLEL_LIMIT 0.99f // |direction.y| above this is straight up or down: use another up
#define BOUNDS_PAD 0.001f // Added to every bounding radius so flat meshes have some
#define SQRT2 1.41421356f
#define CUBE_FACES 6
#define SHADOW_SIZE 1024
#define DEFAULT_AMBIENT 26 // sRGB; the same look as the old 0.1 in gamma space
#define MIN_EXPOSURE -10.0f // Stops
#define TANGENT_EPSILON 1e-8f // Below this a triangle has no UV area to take a tangent from
#define MAX_EXPOSURE 10.0f
#define SHADOW_SIZE_MIN 256
#define SHADOW_SIZE_MAX 4096
#define SHADOW_BIAS 0.0015f
#define SHADOW_MARGIN 1.05f // The fitted light frustum, a little larger than the scene
#define DEFAULT_CASCADES 3
#define DEFAULT_SHADOW_DISTANCE 60.0f // How far from the camera cascaded shadows reach
#define CASCADE_LAMBDA 0.7f // The practical split scheme's mix of log and linear splits
#define CASCADE_NEAR 0.01f
#define SKIN_BOUNDS_GROW 1.5f // A skinned mesh moves beyond its bind pose
#define MORPH_FLOATS 8 // Per target per vertex: position delta xyz + pad, normal delta xyz + pad
#define MORPH_INPUT_FLOATS 6 // Per target per vertex as meshSetMorphTargets takes them: position delta xyz, normal delta xyz
#define FNV_OFFSET 1469598103934665603ULL // FNV-1a, for the shadow cache fingerprints
#define FNV_PRIME 1099511628211ULL
#define BILLBOARD_LANE 15 // The normal matrix's spare lane (the shader's _m33) carrying a draw's BILLBOARD_MODE_*
// Matches DrawUniforms in scene.hlsl.
typedef struct DrawUniformsS {
Mat4T viewProjection;
float morphWeights[MAX_MORPHS];
int32_t morphTargets[MAX_MORPHS];
int32_t morphInfo[4]; // x = active targets, y = vertices per target, z = first instance in the matrix buffer
float billboardRight[4]; // The axes and eye billboards turn to
float billboardUp[4];
float billboardEye[4];
} DrawUniformsT;
// One draw's matrices in the per-frame instance buffer the vertex shaders read.
typedef struct InstanceMatricesS {
Mat4T model;
Mat4T normal; // Inverse transpose, for normals under non-uniform scale; m[BILLBOARD_LANE] the billboard mode
} InstanceMatricesT;
// Matches Light and FragmentUniforms in scene.hlsl.
typedef struct LightUniformS {
float positionType[4];
float directionRange[4];
float color[4];
float cone[4];
} LightUniformT;
// What every draw in a pass shares.
typedef struct FragmentUniformsS {
float cameraPosition[4];
float cameraForward[4]; // xyz; w = 1 when the cascades were fitted to this camera (the window's)
float ambient[4];
float counts[4];
float shadowParams[4];
Mat4T shadowMatrix[MAX_SHADOWS * MAX_CASCADES]; // Per slot, one per cascade
float shadowInfo[MAX_SHADOWS][4];
float cascadeSplits[MAX_SHADOWS][4]; // View depth where each cascade ends
LightUniformT lights[MAX_LIGHTS];
float fog[4]; // rgb, w = on
float fogRange[4]; // x = near, y = far
float environment[4]; // x = lit by the sky, y = the sky's last mip level, z = sky intensity
float sh[SH_COEFFICIENTS][4];
} FragmentUniformsT;
// Matches MaterialUniforms in scene.hlsl: what changes per draw batch.
typedef struct MaterialUniformsS {
float baseColor[4];
float emissive[4];
float material[4]; // x = metallic, y = roughness, z = unlit (1/0), w = TEXTURE_*
float maps[4]; // x = normal map strength (0 = none), y = occlusion strength (0 = none), z = alpha cutoff (0 = none)
float tiling[4]; // x, y = texture repeats across the surface
} MaterialUniformsT;
// One light's shadow for this frame.
typedef struct ShadowS {
int32_t node;
int32_t type; // SHADOW_MAP, SHADOW_CUBE or SHADOW_CASCADE
int32_t layer; // First layer in the shadow array (a point light uses six, a cascaded sun its cascades)
Mat4T matrix; // Map: the light's view-projection
Mat4T faces[CUBE_FACES]; // Cube: one per face; cascades: one per cascade
Mat4T faceViews[CUBE_FACES]; // Each face's or cascade's view alone, for culling
float near;
float far;
int32_t cascades;
float splits[MAX_CASCADES]; // View depth where each cascade ends
float radius[MAX_CASCADES]; // Half the width of each cascade's box
float depth[MAX_CASCADES]; // Each cascade's far plane in its own view
} ShadowT;
typedef struct MeshS {
SDL_GPUBuffer *vertexBuffer;
SDL_GPUBuffer *indexBuffer;
uint32_t indexCount;
Vec3T boundsMin; // Of the vertices, for fitting the shadow map
Vec3T boundsMax;
float *positions; // A CPU copy of the geometry (x, y, z per vertex) for physics shapes
uint32_t *indices;
SceneVertexT *vertices; // The whole vertex array, for meshes that are rewritten (soft bodies)
SDL_GPUTransferBuffer *transfer; // For those rewrites, made on first use
int32_t vertexCount;
SDL_GPUBuffer *morphBuffer; // Morph target deltas, MORPH_FLOATS per vertex per target
char **morphNames;
int32_t morphCount;
float *heights; // A heightmap mesh keeps its samples for the height field body and height queries
int32_t heightColumns;
int32_t heightRows;
float sizeX;
float sizeY;
float sizeZ;
uint32_t version; // Stamped from the scene's mesh counter whenever the geometry changes (_cubeHash)
bool skinned;
bool used;
} MeshT;
// A video player's frames as a texture: the 2D renderer copies the player's (YUV) texture into
// an RGBA target every frame, and materials sample that.
typedef struct FeedS {
int32_t player;
SDL_Texture *target; // Owned; created once the source's size is known
SDL_GPUTexture *gpu; // The target as the scene samples it (not owned)
bool used;
} FeedT;
typedef struct MaterialS {
Vec4T baseColor;
Vec3T emissive;
float metallic;
float roughness;
float cutoff; // glTF alpha masking: texels below this are discarded, 0 = off
SDL_GPUTexture *texture; // Owned; NULL means untextured
SDL_GPUTexture *normalMap; // Owned, each NULL when absent
SDL_GPUTexture *occlusionMap;
SDL_GPUTexture *metallicRoughnessMap;
SDL_GPUTexture *emissiveMap;
float normalStrength;
float occlusionStrength;
float tilingU;
float tilingV;
int32_t feed; // A video feed instead of the texture, NO_HANDLE for none
int32_t view; // A rendered view instead of the texture, NO_HANDLE for none
int32_t gui; // A GUI's texture instead of the texture, NO_HANDLE for none
MaterialFilterE filter;
bool unlit;
bool doubleSided;
bool blend;
bool occluder; // Depth only: hides what is behind it and shows what is behind itself
bool textureBorrowed; // The base texture belongs to a sprite node: never released here
bool used;
} MaterialT;
typedef struct LightS {
LightTypeE type;
Vec3T color;
float intensity;
float range;
float innerDegrees;
float outerDegrees;
} LightT;
typedef struct NodeS {
char *name;
int32_t parent;
int32_t firstChild;
int32_t nextSibling;
Vec3T translation;
QuatT rotation;
Vec3T scale;
Mat4T world; // Rebuilt every frame
int32_t mesh;
int32_t material;
LightT light;
bool castsShadow;
float *morphWeights; // One per target of the node's mesh
int32_t morphCount;
int32_t *skinJoints; // Nodes whose world matrices drive a skinned mesh
Mat4T *skinInverseBind;
int32_t skinCount;
uint32_t generation; // Counts reuses of this slot, so stale handles can be told apart
bool hasLight;
bool visible;
bool worldVisible; // Own flag and every ancestor's
bool shadowCaster; // Drawn into shadow maps (nodeSetShadow)
BillboardE billboard; // Turned to face each camera by the vertex shader
int32_t spriteSlot; // Into the sprite records when the node shows a sprite or text, else NO_HANDLE
bool used;
} NodeT;
// A node showing a picture (or text) on the shared quad: its frames as textures, and a private
// material that borrows the current frame.
typedef struct SpriteNodeS {
SDL_GPUTexture **frames;
int32_t count;
int32_t frame;
int32_t material;
float width;
float height;
bool used;
} SpriteNodeT;
typedef struct DrawS {
int32_t node;
Vec3T centre; // World bounding sphere, filled by _boundDraws
float radius;
int32_t skin; // Into the frame's skin matrices (_fillInstances), NO_HANDLE unskinned
} DrawT;
// Matches SkinUniforms in scene.hlsl.
typedef struct SkinUniformsS {
Mat4T joints[MAX_JOINTS];
} SkinUniformsT;
// One corner of a particle billboard; the vertex shader expands it from the camera's axes.
typedef struct ParticleVertexS {
float centre[3];
float corner[2];
float sizeAngle[2]; // Size in world units, rotation in degrees
float colour[4];
float uv[2];
} ParticleVertexT;
// One end of a debug line: world position and linear colour.
typedef struct LineVertexS {
float position[3];
float colour[4];
} LineVertexT;
// Matches LineUniforms in scene.hlsl.
typedef struct LineUniformsS {
Mat4T viewProjection;
} LineUniformsT;
typedef struct ParticleUniformsS {
Mat4T viewProjection;
float right[4];
float up[4];
float eye[4];
float forward[4];
} ParticleUniformsT;
// Matches ParticleParams in scene.hlsl: what one run of particles needs beyond the frame's lights.
typedef struct ParticleParamsS {
float flags[4]; // x = additive, y = lit, z = softness (0 = none)
float depthParams[4]; // x = near, y = far, z = perspective (1/0)
float targetSize[4]; // x = 1 / width, y = 1 / height
float right[4];
float up[4];
float forward[4];
} ParticleParamsT;
// Matches PostUniforms in scene.hlsl.
typedef struct PostUniformsS {
float params[4]; // x = exposure scale, y = the tone curve, z = bloom strength
} PostUniformsT;
// Matches BloomUniforms in scene.hlsl.
typedef struct BloomUniformsS {
float params[4]; // x, y = the source's texel size, z = threshold, w = first pass (1/0)
} BloomUniformsT;
// One camera's render: where it looks from and what it draws into.
typedef struct CameraFrameS {
Mat4T view;
Mat4T world; // The camera's own transform (view's inverse): its axes and eye
Mat4T viewProjection;
Vec3T eye;
Vec3T right;
Vec3T up;
Vec3T forward;
int32_t width;
int32_t height;
int32_t sampleSet; // SAMPLE_SET_* the pipelines drawing into the targets come from
SDL_GPUTexture *colour; // HDR, resolved
SDL_GPUTexture *multisampled; // Or NULL
SDL_GPUTexture *depth;
SDL_GPUTexture *softDepth; // Or NULL: no soft particles here
SDL_GPUTexture *output; // Display texture the post pass writes
bool main; // The window's camera: stats, bloom
} CameraFrameT;
// A camera rendered to a texture materials can show (viewNew).
typedef struct ViewS {
int32_t camera; // Node, or NO_HANDLE for the default view
int32_t width;
int32_t height;
SDL_GPUTexture *colour;
SDL_GPUTexture *depth;
SDL_GPUTexture *output;
bool used;
} ViewT;
// Matches SkyUniforms in scene.hlsl.
typedef struct SkyUniformsS {
Mat4T inverseViewProjection;
float eye[4];
float params[4]; // x = intensity
} SkyUniformsT;
// GPU textures made from an emitter's frames, kept until the frames change or the emitter goes.
typedef struct ParticleTexturesS {
int32_t id;
uint32_t version;
int32_t count;
SDL_GPUTexture **textures;
} ParticleTexturesT;
// A stretch of the frame's particle vertices drawn with one texture and blend.
typedef struct ParticleRunS {
int32_t first;
int32_t count;
SDL_GPUTexture *texture;
ParticleBlendE blend;
bool lit;
float softness;
} ParticleRunT;
// Sort keys for particles and emitters, far to near.
typedef struct DepthOrderS {
float depth;
int32_t index;
} DepthOrderT;
// A one-off upload in three steps (_stageBegin, _stageCopy, _stageEnd): a transfer buffer mapped
// for filling, a copy pass for the copies out of it, then submit and release.
typedef struct StagingS {
SDL_GPUTransferBuffer *transfer;
SDL_GPUCommandBuffer *commands;
SDL_GPUCopyPass *pass;
void *mapped;
} StagingT;
// What a point light's six faces were last rendered from, so unchanged ones are kept.
typedef struct ShadowCacheS {
int32_t node;
int32_t layer;
uint32_t mapsVersion;
uint64_t hash;
} ShadowCacheT;
typedef struct SceneS {
SDL_GPUDevice *device;
SDL_Renderer *renderer;
SDL_GPUTexture *colour; // What the 2D renderer samples (resolved when multisampled)
SDL_GPUTexture *multisampled; // The colour target while antialiasing, resolved into colour
SDL_GPUTexture *depth;
SDL_GPUTextureFormat depthFormat;
SDL_GPUSampleCount sampleCount;
bool antialias; // Wanted; sampleCount says what the device gave
SDL_Texture *composite; // The colour target as the 2D renderer sees it
SDL_GPUShader *vertexStatic;
SDL_GPUShader *vertexSkinned;
SDL_GPUShader *fragment;
SDL_GPUShader *depthFragment; // Empty; the shadow pass writes depth only
SDL_GPUShader *depthCutoutFragment; // Samples the base texture to mask, for cutout casters
SDL_GPUShader *particleVertex;
SDL_GPUShader *particleFragment;
SDL_GPUShader *lineVertex;
SDL_GPUShader *lineFragment;
SDL_GPUGraphicsPipeline *linePipeline[SAMPLE_SETS];
SDL_GPUShader *postVertex;
SDL_GPUShader *postFragment;
SDL_GPUGraphicsPipeline *postPipeline;
SDL_GPUSampler *postSampler; // Clamped, for the post pass reading the HDR target
SDL_GPUTexture *flatNormal; // 1x1 defaults bound where a material has no map
SDL_GPUTexture *black;
SDL_GPUTexture *blackCube; // Bound as the sky when there is none
SDL_GPUTexture *skyCube; // The sky, six faces with a mip chain, or NULL
int32_t skyLevels;
SDL_GPUSampler *skySampler; // Trilinear, clamped
SDL_GPUShader *skyFragment;
SDL_GPUGraphicsPipeline *skyPipeline[SAMPLE_SETS];
float skyIntensity;
Vec3T sh[SH_COEFFICIENTS]; // The sky's diffuse light
bool environment; // Light the scene from the sky when there is one
Vec3T fogColour;
float fogNear;
float fogFar;
bool fog;
SDL_GPUTextureFormat hdrFormat; // Of the colour target
SDL_GPUTexture *output; // What the post pass writes and the composite wraps
float exposure; // In stops
SceneTonemapE tonemap;
SDL_GPUGraphicsPipeline *particlePipelines[SAMPLE_SETS][PARTICLE_PIPELINES];
InstanceMatricesT *instances; // This frame's matrices, one per draw ...
int32_t instanceRoom; // ... how many the array holds ...
SDL_GPUBuffer *instanceBuffer; // ... and on the GPU
SDL_GPUTransferBuffer *instanceTransfer;
uint32_t instanceCapacity; // Bytes the GPU buffers hold
SkinUniformsT *skins; // This frame's joint matrices, one block per skinned draw ...
int32_t skinRoom; // ... and how many the array holds
bool *skip; // Scratch: a flag per draw for culling ...
int32_t skipRoom; // ... and how many it holds
DepthOrderT *particleOrder; // Scratch: one emitter's particles sorted by depth ...
int32_t particleOrderRoom; // ... and how many it holds
int32_t statTotal; // Last frame: draws collected ...
int32_t statDrawn; // ... inside the view ...
int32_t statBatches; // ... and draw calls they became
ViewT views[MAX_VIEWS];
SpriteNodeT *spriteNodes;
SDL_GPUTexture **sizedTextures; // Every texture uploaded, with its bytes, for the memory total ...
size_t *sizedBytes;
int32_t sizedCount;
int32_t sizedCapacity;
int64_t textureBytes; // ... which is this
Ktx2FormatE compressedFormat;
int32_t spriteNodeCount;
int32_t quadMesh; // The unit quad sprites and text draw on, NO_HANDLE until needed
float bloomThreshold;
float bloomStrength; // 0: no bloom
SDL_GPUTexture *bloomDown[BLOOM_LEVELS]; // The half-size chain down ...
SDL_GPUTexture *bloomUp[BLOOM_LEVELS]; // ... and back up, one texture per level
int32_t bloomLevels;
int32_t bloomWidth; // Of level 0
int32_t bloomHeight;
SDL_GPUShader *bloomDownFragment;
SDL_GPUShader *bloomUpFragment;
SDL_GPUGraphicsPipeline *bloomDownPipeline;
SDL_GPUGraphicsPipeline *bloomUpPipeline;
SDL_GPUTexture *softDepth; // The camera's depth, single sample, for soft particles
SDL_GPUTexture *depthNone; // 1x1 stand-in when no run is soft
bool particleSoft; // Some run this frame wants the depth
bool depthPrepass; // _drawList is filling softDepth: every opaque mesh counts
SDL_GPUBuffer *particleBuffer; // This frame's billboard vertices
SDL_GPUBuffer *lineBuffer; // This frame's debug lines
SDL_GPUTransferBuffer *lineTransfer;
uint32_t lineCapacity; // Bytes the line buffers hold
LineVertexT *lineVertices;
int32_t lineVertexCapacity;
int32_t lineVertexCount;
SDL_GPUTransferBuffer *particleTransfer;
uint32_t particleCapacity; // Bytes in both
ParticleVertexT *particleVertices; // CPU side, grown as needed
int32_t particleVertexCapacity;
int32_t particleVertexCount;
ParticleRunT particleRuns[PARTICLE_RUN_MAX];
int32_t particleRunCount;
ParticleTexturesT *particleTextures;
int32_t particleTextureCount;
SDL_GPUGraphicsPipeline *pipelines[SAMPLE_SETS][PIPELINE_COUNT];
SDL_GPUGraphicsPipeline *shadowPipelines[SHADOW_PIPELINES];
ShadowCacheT shadowCache[MAX_SHADOWS];
uint32_t shadowMapsVersion; // Bumped whenever the map array is (re)made
uint32_t meshVersion; // The last stamp given to a mesh's contents (MeshT.version)
SDL_GPUSampler *sampler;
SDL_GPUSampler *shadowSampler;
SDL_GPUSampler *nearestSampler; // For FILTER_NEAREST materials
SDL_GPUTexture *shadowMaps; // 2D array: a layer per directional or spot shadow, six per point light
SDL_GPUTexture *shadowMapsNone; // 1x1 stand-in bound when nothing casts, so the shader's slot is filled
int32_t shadowMapLayers;
SDL_GPUTextureFormat shadowFormat;
int32_t shadowSize;
int32_t shadowCascades;
float shadowDistance;
ShadowT shadows[MAX_SHADOWS];
int32_t shadowCount;
SDL_GPUTexture *white; // 1x1 stand-in for untextured materials
SDL_GPUBuffer *noMorphs; // Stand-in delta buffer for meshes without morph targets
SDL_FColor background;
Vec3T ambient;
NodeT *nodes;
int32_t nodeCount;
MeshT *meshes;
int32_t meshCount;
MaterialT *materials;
int32_t materialCount;
FeedT *feeds;
int32_t feedCount;
DrawT *draws; // This frame's draws: the opaque ones first, sorted for batching, then the blended ...
int32_t drawCapacity;
int32_t opaqueCount; // ... how many are opaque ...
DepthOrderT *blendedOrder; // ... and the blended ones back to front from the camera being rendered
int32_t blendedOrderRoom;
int32_t cameraNode; // NO_HANDLE for the built-in default view
bool perspective;
float fov;
float orthoHeight;
float near;
float far;
Mat4T viewProjection; // Of the last rendered frame, for sceneProject
int32_t width;
int32_t height;
bool enabled;
} SceneT;
static int32_t _addMesh(const SceneVertexT *vertices, int32_t vertexCount, const uint32_t *indices, int32_t indexCount, bool skinned);
static int32_t _allocFeed(int32_t player);
static int32_t _allocMaterial(void);
static int32_t _allocNode(void);
static void _alphaBlendState(SDL_GPUColorTargetDescription *colour, bool additive);
static void _attach(int32_t node, int32_t parent);
static void _boundDraws(int32_t drawCount);
static void _cameraFrame(int32_t camera, int32_t width, int32_t height, CameraFrameT *frame);
static int32_t _compareDepth(float a, float b);
static int32_t _compareDepthOrder(const void *a, const void *b);
static int32_t _compareOpaque(const void *a, const void *b);
static void _computeSh(const float *rgb, int32_t width, int32_t height);
static bool _createBloomPipelines(void);
static bool _createBloomTargets(int32_t width, int32_t height);
static bool _createLinePipeline(int32_t sampleSet);
static bool _createParticlePipeline(int32_t sampleSet, int32_t blend);
static bool _createPipeline(int32_t sampleSet, int32_t variant);
static bool _createPostPipeline(void);
static SDL_GPUShader *_createShader(const SceneShaderT *shader, SDL_GPUShaderStage stage, uint32_t samplers, uint32_t uniforms, uint32_t storageBuffers);
static bool _createShaders(void);
static SDL_GPUTexture *_createShadowArray(SDL_GPUTextureType type, int32_t layers, int32_t size);
static bool _createShadowMaps(int32_t layers);
static bool _createShadowPipeline(int32_t variant);
static bool _createSkyPipeline(int32_t sampleSet);
static uint64_t _cubeHash(const ShadowT *shadow, int32_t drawCount);
static void _cullCascade(const ShadowT *shadow, int32_t cascade, int32_t drawCount, bool *skip);
static void _cullDraws(const Mat4T *viewProjection, int32_t drawCount, bool *skip);
static void _cullFace(const ShadowT *shadow, int32_t face, int32_t drawCount, bool *skip);
static SDL_GPUTextureFormat _depthFormat(void);
static void _describeMeshVertex(SDL_GPUVertexBufferDescription *buffer, SDL_GPUVertexAttribute *attributes);
static void _destroyBloomTargets(void);
static void _destroyPipelines(void);
static void _destroyShadowMaps(void);
static void _destroyTargets(void);
static void _detach(int32_t node);
static void _drawBloom(SDL_GPUCommandBuffer *commands, const CameraFrameT *frame);
static void _drawLines(SDL_GPUCommandBuffer *commands, SDL_GPURenderPass *pass, const CameraFrameT *frame);
static void _drawList(SDL_GPUCommandBuffer *commands, SDL_GPURenderPass *pass, int32_t drawCount, const Mat4T *viewProjection, const CameraFrameT *axes, bool shadowPass, bool twoSided, const bool *skip, const FragmentUniformsT *fragmentUniforms, int32_t sampleSet);
static void _drawParticles(SDL_GPUCommandBuffer *commands, SDL_GPURenderPass *pass, const CameraFrameT *frame, const FragmentUniformsT *fragmentUniforms);
static void _drawPost(SDL_GPUCommandBuffer *commands, const CameraFrameT *frame);
static void _drawSky(SDL_GPUCommandBuffer *commands, SDL_GPURenderPass *pass, const CameraFrameT *frame);
static Vec3T _faceDirection(int32_t face, float s, float t);
static void _fillInstances(int32_t drawCount);
static void _fillLights(FragmentUniformsT *uniforms);
static void _fillSkin(const NodeT *node, SkinUniformsT *uniforms);
static void _fitShadows(int32_t drawCount, const CameraFrameT *camera);
static void _freeFeed(FeedT *feed);
static void _freeMaterialTexture(MaterialT *material);
static void _freeMorphs(MeshT *mesh);
static void _freeMorphWeights(NodeT *node);
static void _freeSkin(NodeT *node);
static void _freeSpriteNode(int32_t node);
static void _freeView(ViewT *view);
static void _gatherParticles(Vec3T eye, Vec3T forward);
static int32_t _gridMesh(const float *heights, int32_t columns, int32_t rows, float sizeX, float sizeY, float sizeZ, bool firstRowFar);
static uint16_t _half(float value);
static bool _hasMorphs(const NodeT *node, const MeshT *mesh);
static SDL_GPUTextureFormat _hdrFormat(void);
static bool _isSkinned(const NodeT *node, const MeshT *mesh);
static void _lathe(SceneVertexT **vertices, int32_t *vertexCount, uint32_t **indices, int32_t *indexCount, float bottomRadius, float topRadius, float height, int32_t segments);
static float _linear(uint8_t value);
static float _linearF(float value);
static bool _mapIsColour(MaterialMapE map);
static void _matchMorphWeights(NodeT *node);
static void _materialDefaults(MaterialT *material);
static void _materialPlace(MaterialT *material, MaterialMapE map, SDL_GPUTexture *texture, float strength);
static SDL_GPUTexture *_materialTexture(const MaterialT *material);
static uint32_t _mipLevels(int32_t width, int32_t height);
static Mat4T _modelOf(const NodeT *node);
static void _orderBlended(Vec3T eye, int32_t drawCount);
static ParticleTexturesT *_particleTextures(const EmitterViewT *view);
static SDL_GPUTextureFormat _pickDepthFormat(const SDL_GPUTextureFormat *wanted, int32_t count, SDL_GPUTextureUsageFlags usage);
static int32_t _pipelineVariant(int32_t node);
static Mat4T _projectionFor(int32_t width, int32_t height);
static int32_t _quadMesh(float width, float height, Vec3T down, Vec3T normal);
static void _recordTexture(SDL_GPUTexture *texture, size_t bytes);
static void _releaseMaterialBase(MaterialT *material);
static void _releaseParticleTextures(bool all);
static void _releasePipeline(SDL_GPUGraphicsPipeline **pipeline);
static void _releaseTexture(SDL_GPUTexture **texture);
static void _renderCamera(SDL_GPUCommandBuffer *commands, const CameraFrameT *frame, FragmentUniformsT *uniforms, int32_t drawCount);
static void _ribbon(const EmitterViewT *view, int32_t index, Vec3T eye);
static bool _sameBatch(int32_t a, int32_t b, bool shadowPass, const bool *skip);
static SDL_GPUSampleCount _sampleCountOf(int32_t sampleSet);
static Vec3T _sampleEquirect(const float *rgb, int32_t width, int32_t height, Vec3T direction);
static SDL_GPUTextureFormat _shadowFormat(void);
static bool *_skipScratch(int32_t drawCount);
static SDL_GPUTexture *_solidTexture(uint8_t r, uint8_t g, uint8_t b);
static bool _stageBegin(StagingT *staging, uint32_t bytes);
static bool _stageCopy(StagingT *staging);
static void _stageEnd(StagingT *staging, SDL_GPUTexture *mipmaps);
static void _stageTexture(const StagingT *staging, uint32_t offset, SDL_GPUTexture *texture, uint32_t level, uint32_t layer, uint32_t width, uint32_t height);
static void _stampMesh(MeshT *mesh);
static void _updateWorld(int32_t node, const Mat4T *parentWorld, bool parentVisible);
static SDL_GPUBuffer *_uploadBuffer(SDL_GPUBufferUsageFlags usage, const void *data, uint32_t size);
static SDL_GPUTexture *_uploadCompressed(const Ktx2ImageT *image, bool srgb);
static SDL_GPUTexture *_uploadCube(const uint16_t *pixels, int32_t face);
static bool _uploadDynamic(SDL_GPUCommandBuffer *commands, SDL_GPUBufferUsageFlags usage, SDL_GPUBuffer **buffer, SDL_GPUTransferBuffer **transfer, uint32_t *capacity, const void *data, uint32_t bytes, const char *what);
static void _uploadInstances(SDL_GPUCommandBuffer *commands, int32_t drawCount);
static void _uploadLines(SDL_GPUCommandBuffer *commands);
static void _uploadParticles(SDL_GPUCommandBuffer *commands);
static SDL_GPUTexture *_uploadTexture(SDL_Surface *image, bool srgb);
static SceneVertexT _vertex(float x, float y, float z, float nx, float ny, float nz, float u, float v);
static Mat4T _viewOf(int32_t camera);
static SceneT _scene;
// ===== Internal helpers =====
static int32_t _addMesh(const SceneVertexT *vertices, int32_t vertexCount, const uint32_t *indices, int32_t indexCount, bool skinned) {
int32_t x;
MeshT *mesh;
if ((_scene.device == NULL) || (vertexCount <= 0) || (indexCount <= 0)) {
return NO_HANDLE;
}
for (x = 0; x < _scene.meshCount; x++) {
if (!_scene.meshes[x].used) {
break;
}
}
if (x == _scene.meshCount) {
_scene.meshes = SDL_realloc(_scene.meshes, sizeof(MeshT) * (size_t)(_scene.meshCount + 1));
if (_scene.meshes == NULL) {
utilDie("Out of memory allocating a mesh.");
}
_scene.meshCount++;
}
mesh = &_scene.meshes[x];
memset(mesh, 0, sizeof(*mesh));
// The CPU copy of the vertices first: tangents for normal mapping are computed into it (unless
// the source supplied them) and the GPU buffer is uploaded from it.
mesh->vertices = SDL_malloc(sizeof(SceneVertexT) * (size_t)vertexCount);
if (mesh->vertices == NULL) {
utilDie("Out of memory keeping a mesh.");
}
memcpy(mesh->vertices, vertices, sizeof(SceneVertexT) * (size_t)vertexCount);
for (x = 0; x < vertexCount; x++) {
if ((vertices[x].tangent[0] != 0.0f) || (vertices[x].tangent[1] != 0.0f) || (vertices[x].tangent[2] != 0.0f)) {
break;
}
}
if (x == vertexCount) {
sceneComputeTangents(mesh->vertices, vertexCount, indices, indexCount);
}
vertices = mesh->vertices;
mesh->vertexBuffer = _uploadBuffer(SDL_GPU_BUFFERUSAGE_VERTEX, vertices, (uint32_t)(sizeof(SceneVertexT) * (size_t)vertexCount));
mesh->indexBuffer = _uploadBuffer(SDL_GPU_BUFFERUSAGE_INDEX, indices, (uint32_t)(sizeof(uint32_t) * (size_t)indexCount));
if ((mesh->vertexBuffer == NULL) || (mesh->indexBuffer == NULL)) {
mesh->used = true;
meshDelete((int32_t)(mesh - _scene.meshes));
return NO_HANDLE;
}
mesh->indexCount = (uint32_t)indexCount;
mesh->vertexCount = vertexCount;
mesh->skinned = skinned;
mesh->used = true;
mesh->positions = SDL_malloc(sizeof(float) * 3 * (size_t)vertexCount);
mesh->indices = SDL_malloc(sizeof(uint32_t) * (size_t)indexCount);
if ((mesh->positions == NULL) || (mesh->indices == NULL)) {
utilDie("Out of memory keeping a mesh's geometry.");
}
for (x = 0; x < vertexCount; x++) {
mesh->positions[x * 3] = vertices[x].position[0];
mesh->positions[x * 3 + 1] = vertices[x].position[1];
mesh->positions[x * 3 + 2] = vertices[x].position[2];
}
memcpy(mesh->indices, indices, sizeof(uint32_t) * (size_t)indexCount);
mesh->boundsMin = vec3(vertices[0].position[0], vertices[0].position[1], vertices[0].position[2]);
mesh->boundsMax = mesh->boundsMin;
for (x = 1; x < vertexCount; x++) {
mesh->boundsMin.x = SDL_min(mesh->boundsMin.x, vertices[x].position[0]);
mesh->boundsMin.y = SDL_min(mesh->boundsMin.y, vertices[x].position[1]);
mesh->boundsMin.z = SDL_min(mesh->boundsMin.z, vertices[x].position[2]);
mesh->boundsMax.x = SDL_max(mesh->boundsMax.x, vertices[x].position[0]);
mesh->boundsMax.y = SDL_max(mesh->boundsMax.y, vertices[x].position[1]);
mesh->boundsMax.z = SDL_max(mesh->boundsMax.z, vertices[x].position[2]);
}
if (skinned) {
Vec3T centre = vec3Scale(vec3Add(mesh->boundsMin, mesh->boundsMax), 0.5f);
Vec3T half = vec3Scale(vec3Subtract(mesh->boundsMax, mesh->boundsMin), 0.5f * SKIN_BOUNDS_GROW);
mesh->boundsMin = vec3Subtract(centre, half);
mesh->boundsMax = vec3Add(centre, half);
}
_stampMesh(mesh);
return (int32_t)(mesh - _scene.meshes);
}
// One feed per player, shared by every material showing it.
static int32_t _allocFeed(int32_t player) {
int32_t x;
for (x = 0; x < _scene.feedCount; x++) {
if (_scene.feeds[x].used && (_scene.feeds[x].player == player)) {
return x;
}
}
for (x = 0; x < _scene.feedCount; x++) {
if (!_scene.feeds[x].used) {
break;
}
}
if (x == _scene.feedCount) {
_scene.feeds = SDL_realloc(_scene.feeds, sizeof(FeedT) * (size_t)(_scene.feedCount + 1));
if (_scene.feeds == NULL) {
utilDie("Out of memory allocating a video feed.");
}
_scene.feedCount++;
}
memset(&_scene.feeds[x], 0, sizeof(FeedT));
_scene.feeds[x].player = player;
_scene.feeds[x].used = true;
return x;
}
static int32_t _allocMaterial(void) {
int32_t x;
MaterialT *material;
for (x = 0; x < _scene.materialCount; x++) {
if (!_scene.materials[x].used) {
break;
}
}
if (x == _scene.materialCount) {
_scene.materials = SDL_realloc(_scene.materials, sizeof(MaterialT) * (size_t)(_scene.materialCount + 1));
if (_scene.materials == NULL) {
utilDie("Out of memory allocating a material.");
}
_scene.materialCount++;
}
material = &_scene.materials[x];
_materialDefaults(material);
material->used = true;
return x;
}
// A fresh node, detached, at the origin.
static int32_t _allocNode(void) {
int32_t x;
uint32_t generation;
NodeT *node;
for (x = 0; x < _scene.nodeCount; x++) {
if (!_scene.nodes[x].used) {
break;
}
}
if (x == _scene.nodeCount) {
_scene.nodes = SDL_realloc(_scene.nodes, sizeof(NodeT) * (size_t)(_scene.nodeCount + 1));
if (_scene.nodes == NULL) {
utilDie("Out of memory allocating a scene node.");
}
memset(&_scene.nodes[x], 0, sizeof(NodeT));
_scene.nodeCount++;
}
node = &_scene.nodes[x];
generation = node->generation + 1;
memset(node, 0, sizeof(*node));
node->generation = generation;
node->parent = NO_HANDLE;
node->spriteSlot = NO_HANDLE;
node->firstChild = NO_HANDLE;
node->nextSibling = NO_HANDLE;
node->rotation = quatIdentity();
node->scale = vec3(1.0f, 1.0f, 1.0f);
node->world = mat4Identity();
node->mesh = NO_HANDLE;
node->material = NO_HANDLE;
node->visible = true;
node->shadowCaster = true;
node->used = true;
return x;
}
// Source-alpha blending over what is there (or, additive, added to it), alpha kept as coverage.
static void _alphaBlendState(SDL_GPUColorTargetDescription *colour, bool additive) {
colour->blend_state.enable_blend = true;
colour->blend_state.src_color_blendfactor = SDL_GPU_BLENDFACTOR_SRC_ALPHA;
colour->blend_state.dst_color_blendfactor = additive ? SDL_GPU_BLENDFACTOR_ONE : SDL_GPU_BLENDFACTOR_ONE_MINUS_SRC_ALPHA;
colour->blend_state.color_blend_op = SDL_GPU_BLENDOP_ADD;
colour->blend_state.src_alpha_blendfactor = SDL_GPU_BLENDFACTOR_ONE;
colour->blend_state.dst_alpha_blendfactor = SDL_GPU_BLENDFACTOR_ONE_MINUS_SRC_ALPHA;
colour->blend_state.alpha_blend_op = SDL_GPU_BLENDOP_ADD;
}
// Links node under parent as its last child.
static void _attach(int32_t node, int32_t parent) {
int32_t last;
_scene.nodes[node].parent = parent;
_scene.nodes[node].nextSibling = NO_HANDLE;
if (_scene.nodes[parent].firstChild == NO_HANDLE) {
_scene.nodes[parent].firstChild = node;
return;
}
last = _scene.nodes[parent].firstChild;
while (_scene.nodes[last].nextSibling != NO_HANDLE) {
last = _scene.nodes[last].nextSibling;
}
_scene.nodes[last].nextSibling = node;
}
// A world bounding sphere per draw, for culling and for fitting shadows.
static void _boundDraws(int32_t drawCount) {
int32_t x;
int32_t c;
for (x = 0; x < drawCount; x++) {
NodeT *node = &_scene.nodes[_scene.draws[x].node];
MeshT *mesh = &_scene.meshes[node->mesh];
Mat4T model = _modelOf(node);
Vec3T drawMin = vec3(0.0f, 0.0f, 0.0f);
Vec3T drawMax = vec3(0.0f, 0.0f, 0.0f);
for (c = 0; c < 8; c++) {
Vec3T corner = vec3((c & 1) ? mesh->boundsMax.x : mesh->boundsMin.x, (c & 2) ? mesh->boundsMax.y : mesh->boundsMin.y, (c & 4) ? mesh->boundsMax.z : mesh->boundsMin.z);
corner = mat4TransformPoint(model, corner);
if (c == 0) {
drawMin = corner;
drawMax = corner;
} else {
drawMin = vec3(SDL_min(drawMin.x, corner.x), SDL_min(drawMin.y, corner.y), SDL_min(drawMin.z, corner.z));
drawMax = vec3(SDL_max(drawMax.x, corner.x), SDL_max(drawMax.y, corner.y), SDL_max(drawMax.z, corner.z));
}
}
_scene.draws[x].centre = vec3Scale(vec3Add(drawMin, drawMax), 0.5f);
_scene.draws[x].radius = vec3Length(vec3Subtract(drawMax, _scene.draws[x].centre)) + BOUNDS_PAD;
}
}
// A camera's view, projection and axes for a target of the given size (single sample until the
// caller says otherwise). The axes and eye are the columns of the camera's own world matrix, so
// only the view itself costs an inversion.
static void _cameraFrame(int32_t camera, int32_t width, int32_t height, CameraFrameT *frame) {
const float *m;
memset(frame, 0, sizeof(*frame));
frame->view = _viewOf(camera);
frame->world = nodeValid(camera) ? _scene.nodes[camera].world : mat4Compose(vec3(0.0f, 0.0f, DEFAULT_EYE_Z), quatIdentity(), vec3(1.0f, 1.0f, 1.0f));
frame->viewProjection = mat4Multiply(_projectionFor(width, height), frame->view);
frame->width = width;
frame->height = height;
frame->sampleSet = SAMPLE_SET_SINGLE;
m = frame->world.m;
frame->right = vec3Normalize(vec3(m[0], m[1], m[2]));
frame->up = vec3Normalize(vec3(m[4], m[5], m[6]));
frame->forward = vec3Normalize(vec3(-m[8], -m[9], -m[10]));
frame->eye = vec3(m[12], m[13], m[14]);
}
// Far to near.
static int32_t _compareDepth(float a, float b) {
if (a > b) {
return -1;
}
if (a < b) {
return 1;
}
return 0;
}
// Particles and emitters: far to near.
static int32_t _compareDepthOrder(const void *a, const void *b) {
return _compareDepth(((const DepthOrderT *)a)->depth, ((const DepthOrderT *)b)->depth);
}
// Opaque draws sort by mesh then material so that copies of one thing sit together and batch.
static int32_t _compareOpaque(const void *a, const void *b) {
const NodeT *na = &_scene.nodes[((const DrawT *)a)->node];
const NodeT *nb = &_scene.nodes[((const DrawT *)b)->node];
if (na->mesh != nb->mesh) {
return (na->mesh < nb->mesh) ? -1 : 1;
}
if (na->material != nb->material) {
return (na->material < nb->material) ? -1 : 1;
}
return 0;
}
// The sky's diffuse light as nine spherical harmonic coefficients per channel, integrated over
// a coarse sampling of the equirect weighted by each sample's solid angle.
static void _computeSh(const float *rgb, int32_t width, int32_t height) {
int32_t stepX = SDL_max(width / SH_SAMPLES_ACROSS, 1);
int32_t stepY = SDL_max(height / (SH_SAMPLES_ACROSS / 2), 1);
int32_t x;
int32_t y;
int32_t c;
for (c = 0; c < SH_COEFFICIENTS; c++) {
_scene.sh[c] = vec3(0.0f, 0.0f, 0.0f);
}
for (y = stepY / 2; y < height; y += stepY) {
float theta = SDL_PI_F * (y + 0.5f) / (float)height;
float solid = (2.0f * SDL_PI_F / (float)width * stepX) * (SDL_PI_F / (float)height * stepY) * SDL_sinf(theta);
for (x = stepX / 2; x < width; x += stepX) {
float phi = 2.0f * SDL_PI_F * ((x + 0.5f) / (float)width - 0.5f);
Vec3T d = vec3(SDL_sinf(theta) * SDL_sinf(phi), SDL_cosf(theta), -SDL_sinf(theta) * SDL_cosf(phi));
const float *pixel = rgb + ((size_t)y * (size_t)width + (size_t)x) * 3;
Vec3T colour = vec3Scale(vec3(pixel[0], pixel[1], pixel[2]), solid);
float basis[SH_COEFFICIENTS];
basis[0] = 0.282095f;
basis[1] = 0.488603f * d.y;
basis[2] = 0.488603f * d.z;
basis[3] = 0.488603f * d.x;
basis[4] = 1.092548f * d.x * d.y;
basis[5] = 1.092548f * d.y * d.z;
basis[6] = 0.315392f * (3.0f * d.z * d.z - 1.0f);
basis[7] = 1.092548f * d.x * d.z;
basis[8] = 0.546274f * (d.x * d.x - d.y * d.y);
for (c = 0; c < SH_COEFFICIENTS; c++) {
_scene.sh[c] = vec3Add(_scene.sh[c], vec3Scale(colour, basis[c]));
}
}
}
}
// The bloom pipelines: the post vertex shader with the downsample and upsample fragments, into
// 16-bit float levels.
static bool _createBloomPipelines(void) {
SDL_GPUGraphicsPipelineCreateInfo info;
SDL_GPUColorTargetDescription colour;
memset(&info, 0, sizeof(info));
memset(&colour, 0, sizeof(colour));
colour.format = _scene.hdrFormat;
info.vertex_shader = _scene.postVertex;
info.fragment_shader = _scene.bloomDownFragment;
info.primitive_type = SDL_GPU_PRIMITIVETYPE_TRIANGLELIST;
info.rasterizer_state.fill_mode = SDL_GPU_FILLMODE_FILL;
info.rasterizer_state.cull_mode = SDL_GPU_CULLMODE_NONE;
info.rasterizer_state.front_face = SDL_GPU_FRONTFACE_COUNTER_CLOCKWISE;
info.multisample_state.sample_count = SDL_GPU_SAMPLECOUNT_1;
info.target_info.color_target_descriptions = &colour;
info.target_info.num_color_targets = 1;
_scene.bloomDownPipeline = rgpuCreateGraphicsPipeline(_scene.device, &info);
if (_scene.bloomDownPipeline == NULL) {
utilTrace("Scene: bloom pipeline: %s", SDL_GetError());
return false;
}
info.fragment_shader = _scene.bloomUpFragment;
_scene.bloomUpPipeline = rgpuCreateGraphicsPipeline(_scene.device, &info);
if (_scene.bloomUpPipeline == NULL) {
utilTrace("Scene: bloom pipeline: %s", SDL_GetError());
rgpuReleaseGraphicsPipeline(_scene.device, _scene.bloomDownPipeline);
_scene.bloomDownPipeline = NULL;
return false;
}
return true;
}
// The bloom chain for a target of this size: half-size levels halving again down to
// BLOOM_MIN_SIZE, one texture per level each way so no pass reads the texture it writes.
static bool _createBloomTargets(int32_t width, int32_t height) {
SDL_GPUTextureCreateInfo info;
int32_t w = SDL_max(width / 2, 1);
int32_t h = SDL_max(height / 2, 1);
int32_t level;
if ((_scene.bloomWidth == w) && (_scene.bloomHeight == h) && (_scene.bloomLevels > 0)) {
return true;
}
_destroyBloomTargets();
_scene.bloomWidth = w;
_scene.bloomHeight = h;
memset(&info, 0, sizeof(info));
info.type = SDL_GPU_TEXTURETYPE_2D;
info.format = _scene.hdrFormat;
info.usage = SDL_GPU_TEXTUREUSAGE_COLOR_TARGET | SDL_GPU_TEXTUREUSAGE_SAMPLER;
info.layer_count_or_depth = 1;
info.num_levels = 1;
info.sample_count = SDL_GPU_SAMPLECOUNT_1;
for (level = 0; level < BLOOM_LEVELS; level++) {
if ((level > 0) && ((w < BLOOM_MIN_SIZE) || (h < BLOOM_MIN_SIZE))) {
break;
}
info.width = (Uint32)w;
info.height = (Uint32)h;
_scene.bloomDown[level] = rgpuCreateTexture(_scene.device, &info);
_scene.bloomUp[level] = rgpuCreateTexture(_scene.device, &info);
if ((_scene.bloomDown[level] == NULL) || (_scene.bloomUp[level] == NULL)) {
utilTrace("Scene: bloom targets: %s", SDL_GetError());
_destroyBloomTargets();
return false;
}
w = SDL_max(w / 2, 1);
h = SDL_max(h / 2, 1);
}
_scene.bloomLevels = level;
return level > 0;
}
// Debug lines: unlit, alpha blended, depth tested against the scene but never writing it.
static bool _createLinePipeline(int32_t sampleSet) {
SDL_GPUGraphicsPipelineCreateInfo info;
SDL_GPUVertexBufferDescription buffers[1];
SDL_GPUVertexAttribute attributes[2];
SDL_GPUColorTargetDescription colour;
memset(&info, 0, sizeof(info));
memset(buffers, 0, sizeof(buffers));
memset(attributes, 0, sizeof(attributes));
memset(&colour, 0, sizeof(colour));
buffers[0].slot = 0;
buffers[0].pitch = sizeof(LineVertexT);
buffers[0].input_rate = SDL_GPU_VERTEXINPUTRATE_VERTEX;
attributes[0].location = 0;
attributes[0].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT3;
attributes[0].offset = offsetof(LineVertexT, position);
attributes[1].location = 1;
attributes[1].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT4;
attributes[1].offset = offsetof(LineVertexT, colour);
colour.format = _scene.hdrFormat;
_alphaBlendState(&colour, false);
info.vertex_shader = _scene.lineVertex;
info.fragment_shader = _scene.lineFragment;
info.vertex_input_state.vertex_buffer_descriptions = buffers;
info.vertex_input_state.num_vertex_buffers = 1;
info.vertex_input_state.vertex_attributes = attributes;
info.vertex_input_state.num_vertex_attributes = 2;
info.primitive_type = SDL_GPU_PRIMITIVETYPE_LINELIST;
info.rasterizer_state.fill_mode = SDL_GPU_FILLMODE_FILL;
info.rasterizer_state.cull_mode = SDL_GPU_CULLMODE_NONE;
info.rasterizer_state.front_face = SDL_GPU_FRONTFACE_COUNTER_CLOCKWISE;
info.multisample_state.sample_count = _sampleCountOf(sampleSet);
info.depth_stencil_state.compare_op = SDL_GPU_COMPAREOP_LESS_OR_EQUAL;
info.depth_stencil_state.enable_depth_test = true;
info.depth_stencil_state.enable_depth_write = false;
info.target_info.color_target_descriptions = &colour;
info.target_info.num_color_targets = 1;
info.target_info.depth_stencil_format = _scene.depthFormat;
info.target_info.has_depth_stencil_target = true;
_scene.linePipeline[sampleSet] = rgpuCreateGraphicsPipeline(_scene.device, &info);
if (_scene.linePipeline[sampleSet] == NULL) {
utilTrace("Scene: line pipeline: %s", SDL_GetError());
return false;
}
return true;
}
// Billboard pipeline for one blend: camera-facing quads, depth tested, never written, two-sided.
static bool _createParticlePipeline(int32_t sampleSet, int32_t blend) {
SDL_GPUGraphicsPipelineCreateInfo info;
SDL_GPUVertexBufferDescription buffers[1];
SDL_GPUVertexAttribute attributes[5];
SDL_GPUColorTargetDescription colour;
memset(&info, 0, sizeof(info));
memset(buffers, 0, sizeof(buffers));
memset(attributes, 0, sizeof(attributes));
memset(&colour, 0, sizeof(colour));
buffers[0].slot = 0;
buffers[0].pitch = sizeof(ParticleVertexT);
buffers[0].input_rate = SDL_GPU_VERTEXINPUTRATE_VERTEX;
attributes[0].location = 0;
attributes[0].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT3;
attributes[0].offset = offsetof(ParticleVertexT, centre);
attributes[1].location = 1;
attributes[1].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT2;
attributes[1].offset = offsetof(ParticleVertexT, corner);
attributes[2].location = 2;
attributes[2].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT2;
attributes[2].offset = offsetof(ParticleVertexT, sizeAngle);
attributes[3].location = 3;
attributes[3].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT4;
attributes[3].offset = offsetof(ParticleVertexT, colour);
attributes[4].location = 4;
attributes[4].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT2;
attributes[4].offset = offsetof(ParticleVertexT, uv);
colour.format = _scene.hdrFormat;
_alphaBlendState(&colour, blend == PARTICLE_ADD);
info.vertex_shader = _scene.particleVertex;
info.fragment_shader = _scene.particleFragment;
info.vertex_input_state.vertex_buffer_descriptions = buffers;
info.vertex_input_state.num_vertex_buffers = 1;
info.vertex_input_state.vertex_attributes = attributes;
info.vertex_input_state.num_vertex_attributes = 5;
info.primitive_type = SDL_GPU_PRIMITIVETYPE_TRIANGLELIST;
info.rasterizer_state.fill_mode = SDL_GPU_FILLMODE_FILL;
info.rasterizer_state.cull_mode = SDL_GPU_CULLMODE_NONE;
info.rasterizer_state.front_face = SDL_GPU_FRONTFACE_COUNTER_CLOCKWISE;
info.multisample_state.sample_count = _sampleCountOf(sampleSet);
info.depth_stencil_state.compare_op = SDL_GPU_COMPAREOP_LESS_OR_EQUAL;
info.depth_stencil_state.enable_depth_test = true;
info.depth_stencil_state.enable_depth_write = false;
info.target_info.color_target_descriptions = &colour;
info.target_info.num_color_targets = 1;
info.target_info.depth_stencil_format = _scene.depthFormat;
info.target_info.has_depth_stencil_target = true;
_scene.particlePipelines[sampleSet][blend] = rgpuCreateGraphicsPipeline(_scene.device, &info);
if (_scene.particlePipelines[sampleSet][blend] == NULL) {
utilTrace("Scene: particle pipeline %d: %s", blend, SDL_GetError());
return false;
}
return true;
}
// The main pass pipeline for one PIPELINE_* variant, for targets of the sample set's count.
static bool _createPipeline(int32_t sampleSet, int32_t variant) {
SDL_GPUGraphicsPipelineCreateInfo info;
SDL_GPUVertexBufferDescription buffer;
SDL_GPUVertexAttribute attributes[MESH_ATTRIBUTES];
SDL_GPUColorTargetDescription colour;
memset(&info, 0, sizeof(info));
memset(&colour, 0, sizeof(colour));
_describeMeshVertex(&buffer, attributes);
colour.format = _scene.hdrFormat;
if (variant & PIPELINE_BLEND) {
_alphaBlendState(&colour, false);
}
if (variant & PIPELINE_OCCLUDER) {
// The shape goes into the depth buffer and nowhere else, so a painted backdrop stays in
// view where it stands while whatever walks behind it is hidden.
colour.blend_state.enable_color_write_mask = true;
colour.blend_state.color_write_mask = 0;
}
info.vertex_shader = (variant & PIPELINE_SKINNED) ? _scene.vertexSkinned : _scene.vertexStatic;
info.fragment_shader = _scene.fragment;
info.vertex_input_state.vertex_buffer_descriptions = &buffer;
info.vertex_input_state.num_vertex_buffers = 1;
info.vertex_input_state.vertex_attributes = attributes;
info.vertex_input_state.num_vertex_attributes = MESH_ATTRIBUTES;
info.primitive_type = SDL_GPU_PRIMITIVETYPE_TRIANGLELIST;
info.rasterizer_state.fill_mode = SDL_GPU_FILLMODE_FILL;
info.rasterizer_state.cull_mode = (variant & PIPELINE_TWO_SIDED) ? SDL_GPU_CULLMODE_NONE : SDL_GPU_CULLMODE_BACK;
info.rasterizer_state.front_face = SDL_GPU_FRONTFACE_COUNTER_CLOCKWISE;
info.multisample_state.sample_count = _sampleCountOf(sampleSet);
info.depth_stencil_state.compare_op = SDL_GPU_COMPAREOP_LESS_OR_EQUAL;
info.depth_stencil_state.enable_depth_test = true;
info.depth_stencil_state.enable_depth_write = !(variant & PIPELINE_BLEND);
info.target_info.color_target_descriptions = &colour;
info.target_info.num_color_targets = 1;
info.target_info.depth_stencil_format = _scene.depthFormat;
info.target_info.has_depth_stencil_target = true;
_scene.pipelines[sampleSet][variant] = rgpuCreateGraphicsPipeline(_scene.device, &info);
if (_scene.pipelines[sampleSet][variant] == NULL) {
utilTrace("Scene: pipeline %d: %s", variant, SDL_GetError());
return false;
}
return true;
}
// The post pipeline: one triangle from the HDR target to the display texture, no vertex buffer.
static bool _createPostPipeline(void) {
SDL_GPUGraphicsPipelineCreateInfo info;
SDL_GPUColorTargetDescription colour;
memset(&info, 0, sizeof(info));
memset(&colour, 0, sizeof(colour));
colour.format = rgpuGetTextureFormatFromPixelFormat(SDL_PIXELFORMAT_BGRA32);
info.vertex_shader = _scene.postVertex;
info.fragment_shader = _scene.postFragment;
info.primitive_type = SDL_GPU_PRIMITIVETYPE_TRIANGLELIST;
info.rasterizer_state.fill_mode = SDL_GPU_FILLMODE_FILL;
info.rasterizer_state.cull_mode = SDL_GPU_CULLMODE_NONE;
info.rasterizer_state.front_face = SDL_GPU_FRONTFACE_COUNTER_CLOCKWISE;
info.multisample_state.sample_count = SDL_GPU_SAMPLECOUNT_1;
info.target_info.color_target_descriptions = &colour;
info.target_info.num_color_targets = 1;
_scene.postPipeline = rgpuCreateGraphicsPipeline(_scene.device, &info);
if (_scene.postPipeline == NULL) {
utilTrace("Scene: post pipeline: %s", SDL_GetError());
return false;
}
return true;
}
// Picks the blob for the format the device accepts.
static SDL_GPUShader *_createShader(const SceneShaderT *shader, SDL_GPUShaderStage stage, uint32_t samplers, uint32_t uniforms, uint32_t storageBuffers) {
SDL_GPUShaderCreateInfo info;
SDL_GPUShaderFormat formats = rgpuGetShaderFormats(_scene.device);
SDL_GPUShader *result;
memset(&info, 0, sizeof(info));
if (formats & SDL_GPU_SHADERFORMAT_SPIRV) {
info.code = shader->spirv;
info.code_size = shader->spirvSize;
info.format = SDL_GPU_SHADERFORMAT_SPIRV;
} else if (formats & SDL_GPU_SHADERFORMAT_DXIL) {
info.code = shader->dxil;
info.code_size = shader->dxilSize;
info.format = SDL_GPU_SHADERFORMAT_DXIL;
} else if (formats & SDL_GPU_SHADERFORMAT_MSL) {
info.code = shader->msl;
info.code_size = shader->mslSize;
info.format = SDL_GPU_SHADERFORMAT_MSL;
} else if (formats & RGPU_SHADERFORMAT_ESSL) {
info.code = (const Uint8 *)shader->essl;
info.code_size = shader->essl != NULL ? SDL_strlen(shader->essl) : 0;
info.format = RGPU_SHADERFORMAT_ESSL;
} else {
utilTrace("Scene: the device accepts none of SPIR-V, DXIL, MSL or GLSL ES.");
return NULL;
}
info.entrypoint = shader->entryPoint;
info.stage = stage;
info.num_samplers = samplers;
info.num_uniform_buffers = uniforms;
info.num_storage_buffers = storageBuffers;
result = rgpuCreateShader(_scene.device, &info);
if (result == NULL) {
utilTrace("Scene: shader %s: %s", shader->entryPoint, SDL_GetError());
}
return result;
}
static bool _createShaders(void) {
_scene.vertexStatic = _createShader(&sceneShaderVertexStatic, SDL_GPU_SHADERSTAGE_VERTEX, 0, 1, 2);
_scene.vertexSkinned = _createShader(&sceneShaderVertexSkinned, SDL_GPU_SHADERSTAGE_VERTEX, 0, 2, 2);
_scene.fragment = _createShader(&sceneShaderFragmentMain, SDL_GPU_SHADERSTAGE_FRAGMENT, MATERIAL_SAMPLERS, 2, 0);
_scene.depthFragment = _createShader(&sceneShaderDepthMain, SDL_GPU_SHADERSTAGE_FRAGMENT, 0, 0, 0);
_scene.depthCutoutFragment = _createShader(&sceneShaderDepthCutoutMain, SDL_GPU_SHADERSTAGE_FRAGMENT, 1, 2, 0);
_scene.particleVertex = _createShader(&sceneShaderParticleVertex, SDL_GPU_SHADERSTAGE_VERTEX, 0, 1, 0);
_scene.particleFragment = _createShader(&sceneShaderParticleFragment, SDL_GPU_SHADERSTAGE_FRAGMENT, 2, 2, 0);
_scene.lineVertex = _createShader(&sceneShaderLineVertex, SDL_GPU_SHADERSTAGE_VERTEX, 0, 1, 0);
_scene.lineFragment = _createShader(&sceneShaderLineFragment, SDL_GPU_SHADERSTAGE_FRAGMENT, 0, 0, 0);
_scene.skyFragment = _createShader(&sceneShaderSkyFragment, SDL_GPU_SHADERSTAGE_FRAGMENT, 1, 1, 0);
_scene.postVertex = _createShader(&sceneShaderPostVertex, SDL_GPU_SHADERSTAGE_VERTEX, 0, 0, 0);
_scene.postFragment = _createShader(&sceneShaderPostFragment, SDL_GPU_SHADERSTAGE_FRAGMENT, 2, 1, 0);
_scene.bloomDownFragment = _createShader(&sceneShaderBloomDown, SDL_GPU_SHADERSTAGE_FRAGMENT, 1, 1, 0);
_scene.bloomUpFragment = _createShader(&sceneShaderBloomUp, SDL_GPU_SHADERSTAGE_FRAGMENT, 2, 1, 0);
return (_scene.vertexStatic != NULL) && (_scene.vertexSkinned != NULL) && (_scene.fragment != NULL) && (_scene.depthFragment != NULL) && (_scene.depthCutoutFragment != NULL) && (_scene.particleVertex != NULL) && (_scene.particleFragment != NULL) && (_scene.postVertex != NULL) && (_scene.postFragment != NULL) && (_scene.skyFragment != NULL) && (_scene.bloomDownFragment != NULL) && (_scene.bloomUpFragment != NULL) && (_scene.lineVertex != NULL) && (_scene.lineFragment != NULL);
}
// A depth texture array the shadow passes render into and the main pass samples.
// Size 0 means the scene's own width and height.
static SDL_GPUTexture *_createShadowArray(SDL_GPUTextureType type, int32_t layers, int32_t size) {
SDL_GPUTextureCreateInfo info;
SDL_GPUTexture *texture;
memset(&info, 0, sizeof(info));
info.type = type;
info.format = _scene.shadowFormat;
info.usage = SDL_GPU_TEXTUREUSAGE_DEPTH_STENCIL_TARGET | SDL_GPU_TEXTUREUSAGE_SAMPLER;
info.width = (Uint32)((size > 0) ? size : _scene.width);
info.height = (Uint32)((size > 0) ? size : _scene.height);
info.layer_count_or_depth = (Uint32)layers;
info.num_levels = 1;
info.sample_count = SDL_GPU_SAMPLECOUNT_1;
texture = rgpuCreateTexture(_scene.device, &info);
if (texture == NULL) {
utilTrace("Scene: shadow map: %s", SDL_GetError());
}
return texture;
}
// Makes sure the array has room for this frame's shadows; it grows to the most used so far.
static bool _createShadowMaps(int32_t layers) {
if (layers > _scene.shadowMapLayers) {
if (_scene.shadowMaps != NULL) {
rgpuReleaseTexture(_scene.device, _scene.shadowMaps);
}
_scene.shadowMaps = _createShadowArray(SDL_GPU_TEXTURETYPE_2D_ARRAY, layers, _scene.shadowSize);
_scene.shadowMapsVersion++;
_scene.shadowMapLayers = (_scene.shadowMaps != NULL) ? layers : 0;
}
return _scene.shadowMaps != NULL;
}
// Depth-only pipelines for the shadow pass: the scene's vertex shaders with an empty fragment
// shader, no colour target, and a depth bias against self-shadowing.
static bool _createShadowPipeline(int32_t variant) {
SDL_GPUGraphicsPipelineCreateInfo info;
SDL_GPUVertexBufferDescription buffer;
SDL_GPUVertexAttribute attributes[MESH_ATTRIBUTES];
memset(&info, 0, sizeof(info));
_describeMeshVertex(&buffer, attributes);
info.vertex_shader = (variant & SHADOW_PIPELINE_SKINNED) ? _scene.vertexSkinned : _scene.vertexStatic;
info.fragment_shader = (variant & SHADOW_PIPELINE_CUTOUT) ? _scene.depthCutoutFragment : _scene.depthFragment;
info.vertex_input_state.vertex_buffer_descriptions = &buffer;
info.vertex_input_state.num_vertex_buffers = 1;
info.vertex_input_state.vertex_attributes = attributes;
info.vertex_input_state.num_vertex_attributes = MESH_ATTRIBUTES;
info.primitive_type = SDL_GPU_PRIMITIVETYPE_TRIANGLELIST;
info.rasterizer_state.fill_mode = SDL_GPU_FILLMODE_FILL;
info.rasterizer_state.cull_mode = (variant & SHADOW_PIPELINE_TWO_SIDED) ? SDL_GPU_CULLMODE_NONE : SDL_GPU_CULLMODE_BACK;
info.rasterizer_state.front_face = SDL_GPU_FRONTFACE_COUNTER_CLOCKWISE;
info.rasterizer_state.enable_depth_bias = true;
info.rasterizer_state.depth_bias_constant_factor = SHADOW_DEPTH_BIAS_CONSTANT;
info.rasterizer_state.depth_bias_slope_factor = SHADOW_DEPTH_BIAS_SLOPE;
info.multisample_state.sample_count = SDL_GPU_SAMPLECOUNT_1;
info.depth_stencil_state.compare_op = SDL_GPU_COMPAREOP_LESS_OR_EQUAL;
info.depth_stencil_state.enable_depth_test = true;
info.depth_stencil_state.enable_depth_write = true;
info.target_info.num_color_targets = 0;
info.target_info.depth_stencil_format = _scene.shadowFormat;
info.target_info.has_depth_stencil_target = true;
_scene.shadowPipelines[variant] = rgpuCreateGraphicsPipeline(_scene.device, &info);
if (_scene.shadowPipelines[variant] == NULL) {
utilTrace("Scene: shadow pipeline %d: %s", variant, SDL_GetError());
return false;
}
return true;
}
// The sky pipeline: the post pass's screen triangle with the sky fragment shader, drawn first
// into the main pass under everything (no depth test or write), so it must match the pass's
// multisampling and depth format.
static bool _createSkyPipeline(int32_t sampleSet) {
SDL_GPUGraphicsPipelineCreateInfo info;
SDL_GPUColorTargetDescription colour;
memset(&info, 0, sizeof(info));
memset(&colour, 0, sizeof(colour));
colour.format = _scene.hdrFormat;
info.vertex_shader = _scene.postVertex;
info.fragment_shader = _scene.skyFragment;
info.primitive_type = SDL_GPU_PRIMITIVETYPE_TRIANGLELIST;
info.rasterizer_state.fill_mode = SDL_GPU_FILLMODE_FILL;
info.rasterizer_state.cull_mode = SDL_GPU_CULLMODE_NONE;
info.rasterizer_state.front_face = SDL_GPU_FRONTFACE_COUNTER_CLOCKWISE;
info.multisample_state.sample_count = _sampleCountOf(sampleSet);
info.depth_stencil_state.enable_depth_test = false;
info.target_info.color_target_descriptions = &colour;
info.target_info.num_color_targets = 1;
info.target_info.depth_stencil_format = _scene.depthFormat;
info.target_info.has_depth_stencil_target = true;
_scene.skyPipeline[sampleSet] = rgpuCreateGraphicsPipeline(_scene.device, &info);
if (_scene.skyPipeline[sampleSet] == NULL) {
utilTrace("Scene: sky pipeline: %s", SDL_GetError());
return false;
}
return true;
}
// A fingerprint of everything a point light's faces depend on: the light, its range, and every
// caster's transform, mesh and mesh contents; skinned and morphing casters count as always changed.
static uint64_t _cubeHash(const ShadowT *shadow, int32_t drawCount) {
uint64_t hash = FNV_OFFSET;
int32_t x;
int32_t b;
const uint8_t *bytes;
bytes = (const uint8_t *)&_scene.nodes[shadow->node].world;
for (b = 0; b < (int32_t)sizeof(Mat4T); b++) {
hash = (hash ^ bytes[b]) * FNV_PRIME;
}
bytes = (const uint8_t *)&shadow->far;
for (b = 0; b < (int32_t)sizeof(float); b++) {
hash = (hash ^ bytes[b]) * FNV_PRIME;
}
for (x = 0; x < drawCount; x++) {
NodeT *node = &_scene.nodes[_scene.draws[x].node];
MeshT *mesh = &_scene.meshes[node->mesh];
if (!node->shadowCaster) {
continue;
}
if (_isSkinned(node, mesh) || ((mesh->morphBuffer != NULL) && (node->morphCount > 0))) {
return 0;
}
hash = (hash ^ (uint64_t)(uint32_t)_scene.draws[x].node) * FNV_PRIME;
hash = (hash ^ (uint64_t)(uint32_t)node->mesh) * FNV_PRIME;
hash = (hash ^ (uint64_t)mesh->version) * FNV_PRIME;
bytes = (const uint8_t *)&node->world;
for (b = 0; b < (int32_t)sizeof(Mat4T); b++) {
hash = (hash ^ bytes[b]) * FNV_PRIME;
}
}
return (hash == 0) ? 1 : hash;
}
// Draws outside one cascade's box (in the cascade's light view, looking down -Z) are skipped.
static void _cullCascade(const ShadowT *shadow, int32_t cascade, int32_t drawCount, bool *skip) {
int32_t x;
float half = shadow->radius[cascade];
for (x = 0; x < drawCount; x++) {
Vec3T local = mat4TransformPoint(shadow->faceViews[cascade], _scene.draws[x].centre);
float radius = _scene.draws[x].radius;
skip[x] = ((fabsf(local.x) - radius > half) || (fabsf(local.y) - radius > half) || (-local.z - radius > shadow->depth[cascade]) || (-local.z + radius < 0.0f));
}
}
// Draws whose bounding sphere lies wholly outside the camera frustum are skipped in the main pass
// (they still cast shadows). The six planes come straight from the view-projection rows.
static void _cullDraws(const Mat4T *viewProjection, int32_t drawCount, bool *skip) {
const float *m = viewProjection->m;
float planes[6][4];
int32_t p;
int32_t x;
for (p = 0; p < 6; p++) {
int32_t row = (p < 2) ? 0 : ((p < 4) ? 1 : 2);
float sign = (p & 1) ? -1.0f : 1.0f;
float length;
int32_t k;
for (k = 0; k < 4; k++) {
// Row 3 plus or minus row 0, 1, 2 for the sides; the near plane is row 2 alone.
planes[p][k] = ((p == 4) ? 0.0f : m[k * 4 + 3]) + sign * m[k * 4 + row];
}
length = SDL_sqrtf(planes[p][0] * planes[p][0] + planes[p][1] * planes[p][1] + planes[p][2] * planes[p][2]);
if (length > 0.0f) {
for (k = 0; k < 4; k++) {
planes[p][k] /= length;
}
}
}
for (x = 0; x < drawCount; x++) {
const DrawT *draw = &_scene.draws[x];
skip[x] = false;
for (p = 0; p < 6; p++) {
if (planes[p][0] * draw->centre.x + planes[p][1] * draw->centre.y + planes[p][2] * draw->centre.z + planes[p][3] < -draw->radius) {
skip[x] = true;
break;
}
}
}
}
// Marks the draws a point light's face cannot see: bounding sphere against the 90 degree frustum
// (whose side planes are at 45 degrees, hence the root two on the radius).
static void _cullFace(const ShadowT *shadow, int32_t face, int32_t drawCount, bool *skip) {
int32_t x;
for (x = 0; x < drawCount; x++) {
Vec3T local = mat4TransformPoint(shadow->faceViews[face], _scene.draws[x].centre);
float radius = _scene.draws[x].radius;
float ahead = -local.z;
skip[x] = ((ahead + radius < shadow->near) || (ahead - radius > shadow->far) || (fabsf(local.x) - radius * SQRT2 > ahead + radius) || (fabsf(local.y) - radius * SQRT2 > ahead + radius));
}
}
// The best depth format the device offers for the camera's depth target.
static SDL_GPUTextureFormat _depthFormat(void) {
static const SDL_GPUTextureFormat wanted[] = { SDL_GPU_TEXTUREFORMAT_D32_FLOAT, SDL_GPU_TEXTUREFORMAT_D24_UNORM, SDL_GPU_TEXTUREFORMAT_D16_UNORM };
return _pickDepthFormat(wanted, (int32_t)SDL_arraysize(wanted), SDL_GPU_TEXTUREUSAGE_DEPTH_STENCIL_TARGET);
}
// The vertex buffer and attribute layout of a SceneVertexT, for the pipelines that draw meshes.
static void _describeMeshVertex(SDL_GPUVertexBufferDescription *buffer, SDL_GPUVertexAttribute *attributes) {
memset(buffer, 0, sizeof(*buffer));
memset(attributes, 0, sizeof(*attributes) * MESH_ATTRIBUTES);
buffer->slot = 0;
buffer->pitch = sizeof(SceneVertexT);
buffer->input_rate = SDL_GPU_VERTEXINPUTRATE_VERTEX;
attributes[0].location = 0;
attributes[0].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT3;
attributes[0].offset = offsetof(SceneVertexT, position);
attributes[1].location = 1;
attributes[1].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT3;
attributes[1].offset = offsetof(SceneVertexT, normal);
attributes[2].location = 2;
attributes[2].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT2;
attributes[2].offset = offsetof(SceneVertexT, uv);
attributes[3].location = 3;
attributes[3].format = SDL_GPU_VERTEXELEMENTFORMAT_UBYTE4;
attributes[3].offset = offsetof(SceneVertexT, joints);
attributes[4].location = 4;
attributes[4].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT4;
attributes[4].offset = offsetof(SceneVertexT, weights);
attributes[5].location = 5;
attributes[5].format = SDL_GPU_VERTEXELEMENTFORMAT_FLOAT4;
attributes[5].offset = offsetof(SceneVertexT, tangent);
}
static void _destroyBloomTargets(void) {
int32_t level;
for (level = 0; level < BLOOM_LEVELS; level++) {
_releaseTexture(&_scene.bloomDown[level]);
_releaseTexture(&_scene.bloomUp[level]);
}
_scene.bloomLevels = 0;
_scene.bloomWidth = 0;
_scene.bloomHeight = 0;
}
// Pipelines bake in the sample count, so a change in antialiasing drops every one that draws into
// the camera targets (meshes, sky, particles, lines) along with the shadow pipelines; they come
// back on first use.
static void _destroyPipelines(void) {
int32_t set;
int32_t x;
for (set = 0; set < SAMPLE_SETS; set++) {
for (x = 0; x < PIPELINE_COUNT; x++) {
_releasePipeline(&_scene.pipelines[set][x]);
}
for (x = 0; x < PARTICLE_PIPELINES; x++) {
_releasePipeline(&_scene.particlePipelines[set][x]);
}
_releasePipeline(&_scene.skyPipeline[set]);
_releasePipeline(&_scene.linePipeline[set]);
}
for (x = 0; x < SHADOW_PIPELINES; x++) {
_releasePipeline(&_scene.shadowPipelines[x]);
}
}
static void _destroyShadowMaps(void) {
if (_scene.shadowMaps != NULL) {
rgpuReleaseTexture(_scene.device, _scene.shadowMaps);
_scene.shadowMaps = NULL;
}
_scene.shadowMapLayers = 0;
}
static void _destroyTargets(void) {
// The SDL_Texture only wraps the colour target (it was created from it), so it goes first.
if (_scene.composite != NULL) {
SDL_DestroyTexture(_scene.composite);
_scene.composite = NULL;
}
if (_scene.output != NULL) {
rgpuReleaseTexture(_scene.device, _scene.output);
_scene.output = NULL;
}
if (_scene.colour != NULL) {
rgpuReleaseTexture(_scene.device, _scene.colour);
_scene.colour = NULL;
}
if (_scene.multisampled != NULL) {
rgpuReleaseTexture(_scene.device, _scene.multisampled);
_scene.multisampled = NULL;
}
if (_scene.depth != NULL) {
rgpuReleaseTexture(_scene.device, _scene.depth);
_scene.depth = NULL;
}
_releaseTexture(&_scene.softDepth);
_scene.width = 0;
_scene.height = 0;
}
// Unlinks node from its parent's child list.
static void _detach(int32_t node) {
int32_t parent = _scene.nodes[node].parent;
int32_t child;
if (parent == NO_HANDLE) {
return;
}
if (_scene.nodes[parent].firstChild == node) {
_scene.nodes[parent].firstChild = _scene.nodes[node].nextSibling;
} else {
child = _scene.nodes[parent].firstChild;
while ((child != NO_HANDLE) && (_scene.nodes[child].nextSibling != node)) {
child = _scene.nodes[child].nextSibling;
}
if (child != NO_HANDLE) {
_scene.nodes[child].nextSibling = _scene.nodes[node].nextSibling;
}
}
_scene.nodes[node].parent = NO_HANDLE;
_scene.nodes[node].nextSibling = NO_HANDLE;
}
// The glow: the bright part of the HDR frame taken down a half-size chain with a 13-tap filter
// and brought back up with a tent filter, each level adding to the one above, into bloomUp[0]
// for the post pass to add.
static void _drawBloom(SDL_GPUCommandBuffer *commands, const CameraFrameT *frame) {
SDL_GPUColorTargetInfo colour;
SDL_GPURenderPass *pass;
SDL_GPUTextureSamplerBinding samplers[2];
BloomUniformsT uniforms;
int32_t level;
int32_t w;
int32_t h;
if (!_createBloomTargets(frame->width, frame->height) || (_scene.bloomLevels < BLOOM_LEVELS_MIN)) {
return;
}
if (((_scene.bloomDownPipeline == NULL) || (_scene.bloomUpPipeline == NULL)) && !_createBloomPipelines()) {
return;
}
memset(samplers, 0, sizeof(samplers));
samplers[0].sampler = _scene.postSampler;
samplers[1].sampler = _scene.postSampler;
memset(&colour, 0, sizeof(colour));
colour.load_op = SDL_GPU_LOADOP_DONT_CARE;
colour.store_op = SDL_GPU_STOREOP_STORE;
// Down: level 0 from the frame with the threshold, each level after from the one above.
w = frame->width;
h = frame->height;
for (level = 0; level < _scene.bloomLevels; level++) {
memset(&uniforms, 0, sizeof(uniforms));
uniforms.params[0] = 1.0f / (float)w;
uniforms.params[1] = 1.0f / (float)h;
uniforms.params[2] = _scene.bloomThreshold;
uniforms.params[3] = (level == 0) ? 1.0f : 0.0f;
samplers[0].texture = (level == 0) ? frame->colour : _scene.bloomDown[level - 1];
colour.texture = _scene.bloomDown[level];
pass = rgpuBeginRenderPass(commands, &colour, 1, NULL);
rgpuBindGraphicsPipeline(pass, _scene.bloomDownPipeline);
rgpuBindFragmentSamplers(pass, 0, samplers, 1);
rgpuPushFragmentUniformData(commands, 0, &uniforms, sizeof(uniforms));
rgpuDrawPrimitives(pass, POST_VERTICES, 1, 0, 0);
rgpuEndRenderPass(pass);
w = SDL_max(w / 2, 1);
h = SDL_max(h / 2, 1);
}
// Up: each level is its own down level plus the tent-filtered level below.
for (level = _scene.bloomLevels - 2; level >= 0; level--) {
int32_t lowerW = SDL_max(_scene.bloomWidth >> (level + 1), 1);
int32_t lowerH = SDL_max(_scene.bloomHeight >> (level + 1), 1);
memset(&uniforms, 0, sizeof(uniforms));
uniforms.params[0] = 1.0f / (float)lowerW;
uniforms.params[1] = 1.0f / (float)lowerH;
samplers[0].texture = _scene.bloomDown[level];
samplers[1].texture = (level == _scene.bloomLevels - 2) ? _scene.bloomDown[level + 1] : _scene.bloomUp[level + 1];
colour.texture = _scene.bloomUp[level];
pass = rgpuBeginRenderPass(commands, &colour, 1, NULL);
rgpuBindGraphicsPipeline(pass, _scene.bloomUpPipeline);
rgpuBindFragmentSamplers(pass, 0, samplers, 2);
rgpuPushFragmentUniformData(commands, 0, &uniforms, sizeof(uniforms));
rgpuDrawPrimitives(pass, POST_VERTICES, 1, 0, 0);
rgpuEndRenderPass(pass);
}
}
// Draws this frame's debug lines, after everything else in the pass.
static void _drawLines(SDL_GPUCommandBuffer *commands, SDL_GPURenderPass *pass, const CameraFrameT *frame) {
SDL_GPUBufferBinding binding;
LineUniformsT uniforms;
if ((_scene.lineVertexCount == 0) || (_scene.lineBuffer == NULL)) {
return;
}
if ((_scene.linePipeline[frame->sampleSet] == NULL) && !_createLinePipeline(frame->sampleSet)) {
return;
}
uniforms.viewProjection = frame->viewProjection;
memset(&binding, 0, sizeof(binding));
binding.buffer = _scene.lineBuffer;
rgpuBindGraphicsPipeline(pass, _scene.linePipeline[frame->sampleSet]);
rgpuPushVertexUniformData(commands, 0, &uniforms, sizeof(uniforms));
rgpuBindVertexBuffers(pass, 0, &binding, 1);
rgpuDrawPrimitives(pass, (Uint32)_scene.lineVertexCount, 1, 0, 0);
}
// Issues the collected draws into a pass: the shadow pass with the light's view-projection and
// depth-only pipelines (the opaque range only; blended meshes cast nothing), or the main pass with
// the camera's and the full material, from the pipelines of the target's sample set, the blended
// range in the order _orderBlended left. A draw's instance index is its own place in the list
// whatever order it goes in. Billboards turn to the axes given. The frame's fragment uniforms go
// up once per pipeline, the material's per batch.
static void _drawList(SDL_GPUCommandBuffer *commands, SDL_GPURenderPass *pass, int32_t drawCount, const Mat4T *viewProjection, const CameraFrameT *axes, bool shadowPass, bool twoSided, const bool *skip, const FragmentUniformsT *fragmentUniforms, int32_t sampleSet) {
SDL_GPUBufferBinding binding;
SDL_GPUTextureSamplerBinding samplerBindings[MATERIAL_SAMPLERS];
SDL_GPUSampler *materialSampler;
SDL_GPUTexture *baseTexture;
DrawUniformsT drawUniforms;
MaterialUniformsT materialUniforms;
SDL_GPUBuffer *storage[2];
int32_t x;
int32_t end;
int32_t index;
int32_t count = shadowPass ? _scene.opaqueCount : drawCount;
int32_t lastPipeline = NO_HANDLE;
int32_t lastMesh = NO_HANDLE;
int32_t variant;
NodeT *node;
MeshT *mesh;
MaterialT *material;
MaterialT defaultMaterial;
SDL_GPUGraphicsPipeline *pipeline;
_materialDefaults(&defaultMaterial);
memset(&drawUniforms, 0, sizeof(drawUniforms));
drawUniforms.viewProjection = *viewProjection;
drawUniforms.billboardRight[0] = axes->right.x;
drawUniforms.billboardRight[1] = axes->right.y;
drawUniforms.billboardRight[2] = axes->right.z;
drawUniforms.billboardUp[0] = axes->up.x;
drawUniforms.billboardUp[1] = axes->up.y;
drawUniforms.billboardUp[2] = axes->up.z;
drawUniforms.billboardEye[0] = axes->eye.x;
drawUniforms.billboardEye[1] = axes->eye.y;
drawUniforms.billboardEye[2] = axes->eye.z;
for (x = 0; x < count; x = end) {
end = x + 1;
index = (x < _scene.opaqueCount) ? x : _scene.blendedOrder[x - _scene.opaqueCount].index;
node = &_scene.nodes[_scene.draws[index].node];
mesh = &_scene.meshes[node->mesh];
material = (node->material != NO_HANDLE) ? &_scene.materials[node->material] : &defaultMaterial;
variant = _pipelineVariant(_scene.draws[index].node);
if ((skip != NULL) && skip[index]) {
continue;
}
if (shadowPass) {
// An occluder stands for something already painted, shadows included.
if ((!node->shadowCaster && !_scene.depthPrepass) || material->occluder) {
continue;
}
// A bulb inside a closed mesh sees only its back faces; they must still cast.
variant = ((variant & PIPELINE_SKINNED) ? SHADOW_PIPELINE_SKINNED : 0) | ((twoSided || (variant & PIPELINE_TWO_SIDED)) ? SHADOW_PIPELINE_TWO_SIDED : 0) | ((material->cutoff > 0.0f) ? SHADOW_PIPELINE_CUTOUT : 0);
if ((_scene.shadowPipelines[variant] == NULL) && !_createShadowPipeline(variant)) {
continue;
}
pipeline = _scene.shadowPipelines[variant];
} else {
if ((_scene.pipelines[sampleSet][variant] == NULL) && !_createPipeline(sampleSet, variant)) {
continue;
}
pipeline = _scene.pipelines[sampleSet][variant];
}
if (variant != lastPipeline) {
rgpuBindGraphicsPipeline(pass, pipeline);
if (!shadowPass) {
rgpuPushFragmentUniformData(commands, FRAME_UNIFORMS, fragmentUniforms, sizeof(FragmentUniformsT));
}
lastPipeline = variant;
lastMesh = NO_HANDLE;
}
drawUniforms.morphInfo[0] = 0;
drawUniforms.morphInfo[1] = 0;
drawUniforms.morphInfo[2] = index;
// Copies of the same thing after this one ride along as instances: opaque draws only, whose
// instances sit in list order.
if ((x < _scene.opaqueCount) && !_isSkinned(node, mesh) && !_hasMorphs(node, mesh)) {
while ((end < _scene.opaqueCount) && _sameBatch(x, end, shadowPass, skip)) {
end++;
}
}
// The strongest active morph targets, up to the shader's limit.
if ((mesh->morphBuffer != NULL) && (node->morphCount == mesh->morphCount)) {
int32_t active = 0;
int32_t t;
for (t = 0; t < node->morphCount; t++) {
if (node->morphWeights[t] == 0.0f) {
continue;
}
if (active < MAX_MORPHS) {
drawUniforms.morphWeights[active] = node->morphWeights[t];
drawUniforms.morphTargets[active] = t;
active++;
} else {
// Replace the weakest chosen one if this is stronger.
int32_t weakest = 0;
int32_t k;
for (k = 1; k < MAX_MORPHS; k++) {
if (fabsf(drawUniforms.morphWeights[k]) < fabsf(drawUniforms.morphWeights[weakest])) {
weakest = k;
}
}
if (fabsf(node->morphWeights[t]) > fabsf(drawUniforms.morphWeights[weakest])) {
drawUniforms.morphWeights[weakest] = node->morphWeights[t];
drawUniforms.morphTargets[weakest] = t;
}
}
}
drawUniforms.morphInfo[0] = active;
drawUniforms.morphInfo[1] = mesh->vertexCount;
}
rgpuPushVertexUniformData(commands, DRAW_UNIFORMS, &drawUniforms, sizeof(drawUniforms));
// Opaque draws are sorted by mesh, so runs of one mesh keep their buffers bound.
if (node->mesh != lastMesh) {
storage[0] = (mesh->morphBuffer != NULL) ? mesh->morphBuffer : _scene.noMorphs;
storage[1] = _scene.instanceBuffer;
rgpuBindVertexStorageBuffers(pass, 0, storage, 2);
memset(&binding, 0, sizeof(binding));
binding.buffer = mesh->vertexBuffer;
rgpuBindVertexBuffers(pass, 0, &binding, 1);
binding.buffer = mesh->indexBuffer;
rgpuBindIndexBuffer(pass, &binding, SDL_GPU_INDEXELEMENTSIZE_32BIT);
lastMesh = node->mesh;
}
if (_scene.draws[index].skin != NO_HANDLE) {
rgpuPushVertexUniformData(commands, SKIN_UNIFORMS, &_scene.skins[_scene.draws[index].skin], sizeof(SkinUniformsT));
}
// A cutout caster binds what the masking depth shader reads and nothing else: the base
// texture, the colour whose alpha it multiplies, the cutoff and the tiling. The shader
// declares the frame's buffer it never reads, so a zeroed one goes in to fill the slot.
if (shadowPass && ((variant & SHADOW_PIPELINE_CUTOUT) != 0)) {
SDL_GPUTextureSamplerBinding cutoutBinding;
FragmentUniformsT unread;
memset(&unread, 0, sizeof(unread));
memset(&materialUniforms, 0, sizeof(materialUniforms));
baseTexture = _materialTexture(material);
materialUniforms.baseColor[3] = material->baseColor.w;
materialUniforms.material[3] = (float)((baseTexture == NULL) ? TEXTURE_NONE : TEXTURE_SRGB);
materialUniforms.maps[2] = material->cutoff;
materialUniforms.tiling[0] = material->tilingU;
materialUniforms.tiling[1] = material->tilingV;
memset(&cutoutBinding, 0, sizeof(cutoutBinding));
cutoutBinding.texture = (baseTexture != NULL) ? baseTexture : _scene.white;
cutoutBinding.sampler = (material->filter == FILTER_NEAREST) ? _scene.nearestSampler : _scene.sampler;
rgpuPushFragmentUniformData(commands, FRAME_UNIFORMS, &unread, sizeof(unread));
rgpuPushFragmentUniformData(commands, MATERIAL_UNIFORMS, &materialUniforms, sizeof(materialUniforms));
rgpuBindFragmentSamplers(pass, 0, &cutoutBinding, 1);
}
if (!shadowPass) {
baseTexture = _materialTexture(material);
memset(&materialUniforms, 0, sizeof(materialUniforms));
materialUniforms.baseColor[0] = material->baseColor.x;
materialUniforms.baseColor[1] = material->baseColor.y;
materialUniforms.baseColor[2] = material->baseColor.z;
materialUniforms.baseColor[3] = material->baseColor.w;
materialUniforms.emissive[0] = material->emissive.x;
materialUniforms.emissive[1] = material->emissive.y;
materialUniforms.emissive[2] = material->emissive.z;
materialUniforms.emissive[3] = 1.0f;
materialUniforms.material[0] = material->metallic;
materialUniforms.material[1] = material->roughness;
materialUniforms.material[2] = material->unlit ? 1.0f : 0.0f;
materialUniforms.material[3] = (float)((baseTexture == NULL) ? TEXTURE_NONE : (((material->feed != NO_HANDLE) || (material->view != NO_HANDLE)) ? TEXTURE_FEED : TEXTURE_SRGB));
materialUniforms.maps[0] = (material->normalMap != NULL) ? material->normalStrength : 0.0f;
materialUniforms.maps[1] = (material->occlusionMap != NULL) ? material->occlusionStrength : 0.0f;
materialUniforms.maps[2] = material->cutoff;
materialUniforms.tiling[0] = material->tilingU;
materialUniforms.tiling[1] = material->tilingV;
rgpuPushFragmentUniformData(commands, MATERIAL_UNIFORMS, &materialUniforms, sizeof(materialUniforms));
memset(samplerBindings, 0, sizeof(samplerBindings));
materialSampler = (material->filter == FILTER_NEAREST) ? _scene.nearestSampler : _scene.sampler;
samplerBindings[0].texture = (baseTexture != NULL) ? baseTexture : _scene.white;
samplerBindings[0].sampler = materialSampler;
samplerBindings[1].texture = (_scene.shadowMaps != NULL) ? _scene.shadowMaps : _scene.shadowMapsNone;
samplerBindings[1].sampler = _scene.shadowSampler;
samplerBindings[2].texture = (material->normalMap != NULL) ? material->normalMap : _scene.flatNormal;
samplerBindings[2].sampler = materialSampler;
samplerBindings[3].texture = (material->occlusionMap != NULL) ? material->occlusionMap : _scene.white;
samplerBindings[3].sampler = materialSampler;
samplerBindings[4].texture = (material->metallicRoughnessMap != NULL) ? material->metallicRoughnessMap : _scene.white;
samplerBindings[4].sampler = materialSampler;
samplerBindings[5].texture = (material->emissiveMap != NULL) ? material->emissiveMap : _scene.white;
samplerBindings[5].sampler = materialSampler;
samplerBindings[6].texture = (_scene.skyCube != NULL) ? _scene.skyCube : _scene.blackCube;
samplerBindings[6].sampler = _scene.skySampler;
rgpuBindFragmentSamplers(pass, 0, samplerBindings, MATERIAL_SAMPLERS);
_scene.statBatches++;
}
rgpuDrawIndexedPrimitives(pass, mesh->indexCount, (Uint32)(end - x), 0, 0, 0);
}
}
// Draws this frame's particle runs, after every mesh, with the camera's axes for the billboards.
static void _drawParticles(SDL_GPUCommandBuffer *commands, SDL_GPURenderPass *pass, const CameraFrameT *frame, const FragmentUniformsT *fragmentUniforms) {
SDL_GPUBufferBinding binding;
SDL_GPUTextureSamplerBinding samplers[2];
SDL_GPUGraphicsPipeline **pipelines = _scene.particlePipelines[frame->sampleSet];
ParticleUniformsT uniforms;
ParticleParamsT params;
int32_t x;
int32_t lastBlend = NO_HANDLE;
if ((_scene.particleRunCount == 0) || (_scene.particleBuffer == NULL)) {
return;
}
memset(&uniforms, 0, sizeof(uniforms));
uniforms.viewProjection = frame->viewProjection;
uniforms.right[0] = frame->right.x;
uniforms.right[1] = frame->right.y;
uniforms.right[2] = frame->right.z;
uniforms.up[0] = frame->up.x;
uniforms.up[1] = frame->up.y;
uniforms.up[2] = frame->up.z;
uniforms.eye[0] = frame->eye.x;
uniforms.eye[1] = frame->eye.y;
uniforms.eye[2] = frame->eye.z;
uniforms.forward[0] = frame->forward.x;
uniforms.forward[1] = frame->forward.y;
uniforms.forward[2] = frame->forward.z;
memset(&params, 0, sizeof(params));
params.depthParams[0] = _scene.near;
params.depthParams[1] = _scene.far;
params.depthParams[2] = _scene.perspective ? 1.0f : 0.0f;
params.targetSize[0] = (frame->width > 0) ? 1.0f / (float)frame->width : 0.0f;
params.targetSize[1] = (frame->height > 0) ? 1.0f / (float)frame->height : 0.0f;
memcpy(params.right, uniforms.right, sizeof(params.right));
memcpy(params.up, uniforms.up, sizeof(params.up));
memcpy(params.forward, uniforms.forward, sizeof(params.forward));
memset(&binding, 0, sizeof(binding));
binding.buffer = _scene.particleBuffer;
rgpuBindVertexBuffers(pass, 0, &binding, 1);
memset(samplers, 0, sizeof(samplers));
samplers[0].sampler = _scene.sampler;
samplers[1].texture = (_scene.particleSoft && (frame->softDepth != NULL)) ? frame->softDepth : _scene.depthNone;
samplers[1].sampler = _scene.shadowSampler;
for (x = 0; x < _scene.particleRunCount; x++) {
ParticleRunT *run = &_scene.particleRuns[x];
if ((pipelines[run->blend] == NULL) && !_createParticlePipeline(frame->sampleSet, run->blend)) {
continue;
}
if ((int32_t)run->blend != lastBlend) {
rgpuBindGraphicsPipeline(pass, pipelines[run->blend]);
rgpuPushVertexUniformData(commands, 0, &uniforms, sizeof(uniforms));
rgpuPushFragmentUniformData(commands, FRAME_UNIFORMS, fragmentUniforms, sizeof(FragmentUniformsT));
lastBlend = run->blend;
}
params.flags[0] = (run->blend == PARTICLE_ADD) ? 1.0f : 0.0f;
params.flags[1] = run->lit ? 1.0f : 0.0f;
params.flags[2] = (frame->softDepth != NULL) ? run->softness : 0.0f;
rgpuPushFragmentUniformData(commands, 1, &params, sizeof(params));
samplers[0].texture = run->texture;
rgpuBindFragmentSamplers(pass, 0, samplers, 2);
rgpuDrawPrimitives(pass, (Uint32)run->count, 1, (Uint32)run->first, 0);
}
}
// The post pass: exposure, bloom (the main camera's, when on), the tone curve and the sRGB
// encode, from the camera's HDR target into its display texture.
static void _drawPost(SDL_GPUCommandBuffer *commands, const CameraFrameT *frame) {
SDL_GPUColorTargetInfo colour;
SDL_GPURenderPass *pass;
SDL_GPUTextureSamplerBinding samplers[2];
PostUniformsT uniforms;
bool bloom = frame->main && (_scene.bloomStrength > 0.0f);
if ((_scene.postPipeline == NULL) && !_createPostPipeline()) {
return;
}
if (bloom) {
// Under BLOOM_LEVELS_MIN levels (a tiny target) nothing was rendered into bloomUp[0].
_drawBloom(commands, frame);
bloom = _scene.bloomLevels >= BLOOM_LEVELS_MIN;
}
memset(&colour, 0, sizeof(colour));
colour.texture = frame->output;
colour.load_op = SDL_GPU_LOADOP_DONT_CARE;
colour.store_op = SDL_GPU_STOREOP_STORE;
pass = rgpuBeginRenderPass(commands, &colour, 1, NULL);
rgpuBindGraphicsPipeline(pass, _scene.postPipeline);
memset(samplers, 0, sizeof(samplers));
samplers[0].texture = frame->colour;
samplers[0].sampler = _scene.postSampler;
samplers[1].texture = bloom ? _scene.bloomUp[0] : _scene.black;
samplers[1].sampler = _scene.postSampler;
rgpuBindFragmentSamplers(pass, 0, samplers, 2);
memset(&uniforms, 0, sizeof(uniforms));
uniforms.params[0] = SDL_powf(2.0f, _scene.exposure);
uniforms.params[1] = (float)_scene.tonemap;
uniforms.params[2] = bloom ? _scene.bloomStrength : 0.0f;
rgpuPushFragmentUniformData(commands, 0, &uniforms, sizeof(uniforms));
rgpuDrawPrimitives(pass, POST_VERTICES, 1, 0, 0);
rgpuEndRenderPass(pass);
}
// The sky under everything: a screen triangle whose fragments look up the cube along the
// camera ray. Only when a sky is set; otherwise the clear colour stays.
static void _drawSky(SDL_GPUCommandBuffer *commands, SDL_GPURenderPass *pass, const CameraFrameT *frame) {
SDL_GPUTextureSamplerBinding sampler;
SkyUniformsT uniforms;
if (_scene.skyCube == NULL) {
return;
}
if ((_scene.skyPipeline[frame->sampleSet] == NULL) && !_createSkyPipeline(frame->sampleSet)) {
return;
}
memset(&uniforms, 0, sizeof(uniforms));
if (!mat4Invert(frame->viewProjection, &uniforms.inverseViewProjection)) {
return;
}
uniforms.eye[0] = frame->eye.x;
uniforms.eye[1] = frame->eye.y;
uniforms.eye[2] = frame->eye.z;
uniforms.params[0] = _scene.skyIntensity;
rgpuBindGraphicsPipeline(pass, _scene.skyPipeline[frame->sampleSet]);
memset(&sampler, 0, sizeof(sampler));
sampler.texture = _scene.skyCube;
sampler.sampler = _scene.skySampler;
rgpuBindFragmentSamplers(pass, 0, &sampler, 1);
rgpuPushFragmentUniformData(commands, 0, &uniforms, sizeof(uniforms));
rgpuDrawPrimitives(pass, POST_VERTICES, 1, 0, 0);
}
// The world direction through a cube face at s, t (-1 to 1, t down), in the usual cube map
// convention (+X right, +Y up, +Z toward the viewer of a face looking down -Z).
static Vec3T _faceDirection(int32_t face, float s, float t) {
switch (face) {
case 0:
return vec3(1.0f, -t, -s);
case 1:
return vec3(-1.0f, -t, s);
case 2:
return vec3(s, 1.0f, t);
case 3:
return vec3(s, -1.0f, -t);
case 4:
return vec3(s, -t, 1.0f);
default:
return vec3(-s, -t, -1.0f);
}
}
// The frame's model and normal matrices, one pair per draw in draw order, and the joint matrices
// of every skinned draw, posed once here for every pass that draws it. A billboard's pair is the
// node's own with its mode in the normal matrix's spare lane; the vertex shader turns both to
// whichever camera draws it.
static void _fillInstances(int32_t drawCount) {
int32_t x;
int32_t skins = 0;
if (_scene.instanceRoom < drawCount) {
SDL_free(_scene.instances);
_scene.instanceRoom = SDL_max(drawCount, _scene.instanceRoom * 2);
_scene.instances = SDL_malloc(sizeof(InstanceMatricesT) * (size_t)_scene.instanceRoom);
if (_scene.instances == NULL) {
utilDie("Out of memory collecting scene matrices.");
}
}
for (x = 0; x < drawCount; x++) {
const NodeT *node = &_scene.nodes[_scene.draws[x].node];
Mat4T model = _modelOf(node);
// The shader makes a billboard's normal matrix from the axes it turns to.
_scene.instances[x].model = model;
_scene.instances[x].normal = (node->billboard == BILLBOARD_NONE) ? mat4NormalMatrix(model) : mat4Identity();
_scene.instances[x].normal.m[BILLBOARD_LANE] = (float)node->billboard;
_scene.draws[x].skin = NO_HANDLE;
if (_isSkinned(node, &_scene.meshes[node->mesh])) {
if (skins == _scene.skinRoom) {
_scene.skinRoom = SDL_max(skins + 1, _scene.skinRoom * 2);
_scene.skins = SDL_realloc(_scene.skins, sizeof(SkinUniformsT) * (size_t)_scene.skinRoom);
if (_scene.skins == NULL) {
utilDie("Out of memory posing skins.");
}
}
_fillSkin(node, &_scene.skins[skins]);
_scene.draws[x].skin = skins;
skins++;
}
}
}
// The first MAX_LIGHTS visible lights, in world space, and the shadow slots: every one of them
// flagged to cast gets a slot (a map for directional and spot lights, a cube for point lights).
static void _fillLights(FragmentUniformsT *uniforms) {
int32_t x;
int32_t count = 0;
int32_t layers = 0;
NodeT *node;
LightUniformT *light;
Vec3T position;
Vec3T direction;
for (x = 0; (x < _scene.nodeCount) && (count < MAX_LIGHTS); x++) {
node = &_scene.nodes[x];
if (!node->used || !node->hasLight || !node->worldVisible) {
continue;
}
light = &uniforms->lights[count];
position = mat4TransformPoint(node->world, vec3(0.0f, 0.0f, 0.0f));
direction = vec3Normalize(mat4TransformVector(node->world, vec3(0.0f, 0.0f, -1.0f)));
light->positionType[0] = position.x;
light->positionType[1] = position.y;
light->positionType[2] = position.z;
light->positionType[3] = (float)node->light.type;
light->directionRange[0] = direction.x;
light->directionRange[1] = direction.y;
light->directionRange[2] = direction.z;
light->directionRange[3] = node->light.range;
light->color[0] = node->light.color.x * node->light.intensity;
light->color[1] = node->light.color.y * node->light.intensity;
light->color[2] = node->light.color.z * node->light.intensity;
light->color[3] = 1.0f;
light->cone[0] = cosf(DEGREES_TO_RADIANS(node->light.innerDegrees));
light->cone[1] = cosf(DEGREES_TO_RADIANS(node->light.outerDegrees));
light->cone[2] = 0.0f;
light->cone[3] = 0.0f;
if (node->castsShadow && (_scene.shadowCount < MAX_SHADOWS)) {
ShadowT *shadow = &_scene.shadows[_scene.shadowCount];
memset(shadow, 0, sizeof(*shadow));
shadow->node = x;
shadow->layer = layers;
shadow->cascades = 1;
if (node->light.type == LIGHT_POINT) {
shadow->type = SHADOW_CUBE;
layers += CUBE_FACES;
} else if ((node->light.type == LIGHT_DIRECTIONAL) && _scene.perspective && (_scene.shadowCascades > 1)) {
shadow->type = SHADOW_CASCADE;
shadow->cascades = _scene.shadowCascades;
layers += shadow->cascades;
} else {
shadow->type = SHADOW_MAP;
layers++;
}
light->cone[2] = (float)(_scene.shadowCount + 1);
_scene.shadowCount++;
}
count++;
}
uniforms->counts[0] = (float)count;
}
// Joint matrices for one skinned draw: the mesh node's own transform cancels out (glTF says a
// skinned mesh ignores it), so each joint is world * inverseBind brought into the mesh's space.
static void _fillSkin(const NodeT *node, SkinUniformsT *uniforms) {
Mat4T meshInverse;
int32_t x;
if (!mat4Invert(node->world, &meshInverse)) {
meshInverse = mat4Identity();
}
for (x = 0; (x < node->skinCount) && (x < MAX_JOINTS); x++) {
int32_t joint = node->skinJoints[x];
if (nodeValid(joint)) {
uniforms->joints[x] = mat4Multiply(meshInverse, mat4Multiply(_scene.nodes[joint].world, node->skinInverseBind[x]));
} else {
uniforms->joints[x] = mat4Identity();
}
}
for (; x < MAX_JOINTS; x++) {
uniforms->joints[x] = mat4Identity();
}
}
// Every shadow's projection, fitted to what is drawn: a directional light gets a parallel box
// round the scene's bounds, a spot light its own cone, a point light six 90 degree faces out to
// its range or the far edge of the scene.
static void _fitShadows(int32_t drawCount, const CameraFrameT *camera) {
Vec3T corners[8];
Vec3T boundsMin = vec3(0.0f, 0.0f, 0.0f);
Vec3T boundsMax = vec3(0.0f, 0.0f, 0.0f);
Vec3T centre;
float radius;
int32_t x;
int32_t c;
int32_t slot;
bool any = false;
// The six faces of a point light's shadow: the direction each looks and its up. The fragment
// shader rebuilds the same frames to look them up, so these need only match it.
Vec3T faceForward[CUBE_FACES] = { { 1.0f, 0.0f, 0.0f }, { -1.0f, 0.0f, 0.0f }, { 0.0f, 1.0f, 0.0f }, { 0.0f, -1.0f, 0.0f }, { 0.0f, 0.0f, 1.0f }, { 0.0f, 0.0f, -1.0f } };
Vec3T faceUp[CUBE_FACES] = { { 0.0f, 1.0f, 0.0f }, { 0.0f, 1.0f, 0.0f }, { 0.0f, 0.0f, -1.0f }, { 0.0f, 0.0f, 1.0f }, { 0.0f, 1.0f, 0.0f }, { 0.0f, 1.0f, 0.0f } };
// World bounds of everything drawn, from the draws' spheres.
for (x = 0; x < drawCount; x++) {
Vec3T centreX = _scene.draws[x].centre;
float r = _scene.draws[x].radius;
Vec3T lo = vec3(centreX.x - r, centreX.y - r, centreX.z - r);
Vec3T hi = vec3(centreX.x + r, centreX.y + r, centreX.z + r);
if (!any) {
boundsMin = lo;
boundsMax = hi;
any = true;
} else {
boundsMin = vec3(SDL_min(boundsMin.x, lo.x), SDL_min(boundsMin.y, lo.y), SDL_min(boundsMin.z, lo.z));
boundsMax = vec3(SDL_max(boundsMax.x, hi.x), SDL_max(boundsMax.y, hi.y), SDL_max(boundsMax.z, hi.z));
}
}
centre = vec3Scale(vec3Add(boundsMin, boundsMax), 0.5f);
radius = vec3Length(vec3Subtract(boundsMax, centre)) * SHADOW_MARGIN + BOUNDS_PAD;
for (c = 0; c < 8; c++) {
corners[c] = vec3((c & 1) ? boundsMax.x : boundsMin.x, (c & 2) ? boundsMax.y : boundsMin.y, (c & 4) ? boundsMax.z : boundsMin.z);
}
for (slot = 0; slot < _scene.shadowCount; slot++) {
ShadowT *shadow = &_scene.shadows[slot];
NodeT *light = &_scene.nodes[shadow->node];
Vec3T direction = vec3Normalize(mat4TransformVector(light->world, vec3(0.0f, 0.0f, -1.0f)));
Vec3T position = mat4TransformPoint(light->world, vec3(0.0f, 0.0f, 0.0f));
Vec3T up = (fabsf(direction.y) < UP_PARALLEL_LIMIT) ? vec3(0.0f, 1.0f, 0.0f) : vec3(0.0f, 0.0f, 1.0f);
Mat4T view;
float far;
float minX = 0.0f;
float maxX = 0.0f;
float minY = 0.0f;
float maxY = 0.0f;
float minZ = 0.0f;
float maxZ = 0.0f;
if (shadow->type == SHADOW_CASCADE) {
// The camera frustum out to the shadow distance, split by the practical scheme; each
// slice gets a bounding sphere (a stable size as the camera turns), an orthographic box
// round it looking along the light and reaching back past the scene for casters behind,
// snapped to whole texels so edges hold still as the camera moves.
float aspect = (_scene.height > 0) ? (float)_scene.width / (float)_scene.height : 1.0f;
float tanHalf = tanf(DEGREES_TO_RADIANS(_scene.fov) * 0.5f);
float nearPlane = _scene.near;
float farPlane = SDL_min(_scene.shadowDistance, _scene.far);
float sliceNear = nearPlane;
int32_t k;
for (k = 0; k < shadow->cascades; k++) {
float fraction = (float)(k + 1) / (float)shadow->cascades;
float logSplit = nearPlane * powf(farPlane / nearPlane, fraction);
float linSplit = nearPlane + (farPlane - nearPlane) * fraction;
float sliceFar = CASCADE_LAMBDA * logSplit + (1.0f - CASCADE_LAMBDA) * linSplit;
Vec3T slice[8];
Vec3T sliceCentre = vec3(0.0f, 0.0f, 0.0f);
float sliceRadius = 0.0f;
float texel;
Vec3T local;
float snapX;
float snapY;
float reach;
for (c = 0; c < 8; c++) {
float depth = (c & 4) ? sliceFar : sliceNear;
slice[c] = mat4TransformPoint(camera->world, vec3(((c & 1) ? 1.0f : -1.0f) * depth * tanHalf * aspect, ((c & 2) ? 1.0f : -1.0f) * depth * tanHalf, -depth));
sliceCentre = vec3Add(sliceCentre, slice[c]);
}
sliceCentre = vec3Scale(sliceCentre, 1.0f / 8.0f);
for (c = 0; c < 8; c++) {
sliceRadius = SDL_max(sliceRadius, vec3Length(vec3Subtract(slice[c], sliceCentre)));
}
sliceRadius *= SHADOW_MARGIN;
reach = sliceRadius + radius * 2.0f;
view = mat4LookAt(vec3Subtract(sliceCentre, vec3Scale(direction, reach)), sliceCentre, up);
texel = 2.0f * sliceRadius / (float)_scene.shadowSize;
local = mat4TransformPoint(view, sliceCentre);
snapX = floorf(local.x / texel) * texel - local.x;
snapY = floorf(local.y / texel) * texel - local.y;
shadow->faceViews[k] = view;
shadow->faces[k] = mat4Multiply(mat4OrthographicBounds(-sliceRadius + snapX, sliceRadius + snapX, -sliceRadius + snapY, sliceRadius + snapY, CASCADE_NEAR, reach * 2.0f), view);
shadow->splits[k] = sliceFar;
shadow->radius[k] = sliceRadius;
shadow->depth[k] = reach * 2.0f;
sliceNear = sliceFar;
}
} else if (light->light.type == LIGHT_DIRECTIONAL) {
view = mat4LookAt(vec3Subtract(centre, vec3Scale(direction, radius * 2.0f)), centre, up);
for (c = 0; c < 8; c++) {
Vec3T v = mat4TransformPoint(view, corners[c]);
if (c == 0) {
minX = v.x;
maxX = v.x;
minY = v.y;
maxY = v.y;
minZ = v.z;
maxZ = v.z;
} else {
minX = SDL_min(minX, v.x);
maxX = SDL_max(maxX, v.x);
minY = SDL_min(minY, v.y);
maxY = SDL_max(maxY, v.y);
minZ = SDL_min(minZ, v.z);
maxZ = SDL_max(maxZ, v.z);
}
}
// View space looks down -Z: the nearest point has the largest z.
shadow->matrix = mat4Multiply(mat4OrthographicBounds(minX * SHADOW_MARGIN, maxX * SHADOW_MARGIN, minY * SHADOW_MARGIN, maxY * SHADOW_MARGIN, SDL_max(-maxZ / SHADOW_MARGIN, SHADOW_NEAR_MIN), -minZ * SHADOW_MARGIN), view);
} else {
// Near and far from the casters themselves: the nearest caster surface (a shade round the
// bulb) sets near, the range or the farthest caster sets far.
float nearest = FLT_MAX;
float farthest = 0.0f;
for (x = 0; x < drawCount; x++) {
float distance = vec3Length(vec3Subtract(_scene.draws[x].centre, position));
if (!_scene.nodes[_scene.draws[x].node].shadowCaster) {
continue;
}
nearest = SDL_min(nearest, distance - _scene.draws[x].radius);
farthest = SDL_max(farthest, distance + _scene.draws[x].radius);
}
far = (light->light.range > 0.0f) ? light->light.range : farthest;
far = SDL_max(far, SHADOW_FAR_MIN);
shadow->near = SDL_clamp(nearest, SHADOW_NEAR_MIN, far * SHADOW_NEAR_FRACTION);
shadow->far = far;
if (light->light.type == LIGHT_SPOT) {
view = mat4LookAt(position, vec3Add(position, direction), up);
shadow->matrix = mat4Multiply(mat4Perspective(SDL_min(light->light.outerDegrees * 2.0f * SHADOW_MARGIN, SPOT_SHADOW_FOV_MAX), 1.0f, shadow->near, far), view);
continue;
}
for (c = 0; c < CUBE_FACES; c++) {
shadow->faceViews[c] = mat4LookAt(position, vec3Add(position, faceForward[c]), faceUp[c]);
shadow->faces[c] = mat4Multiply(mat4Perspective(90.0f, 1.0f, shadow->near, shadow->far), shadow->faceViews[c]);
}
}
}
}
static void _freeFeed(FeedT *feed) {
if (feed->target != NULL) {
SDL_DestroyTexture(feed->target);
}
memset(feed, 0, sizeof(*feed));
}
// Every texture a material owns.
static void _freeMaterialTexture(MaterialT *material) {
_releaseMaterialBase(material);
_releaseTexture(&material->normalMap);
_releaseTexture(&material->occlusionMap);
_releaseTexture(&material->metallicRoughnessMap);
_releaseTexture(&material->emissiveMap);
}
static void _freeMorphs(MeshT *mesh) {
int32_t x;
if (mesh->morphBuffer != NULL) {
rgpuReleaseBuffer(_scene.device, mesh->morphBuffer);
mesh->morphBuffer = NULL;
}
for (x = 0; x < mesh->morphCount; x++) {
SDL_free(mesh->morphNames[x]);
}
SDL_free(mesh->morphNames);
mesh->morphNames = NULL;
mesh->morphCount = 0;
}
static void _freeMorphWeights(NodeT *node) {
SDL_free(node->morphWeights);
node->morphWeights = NULL;
node->morphCount = 0;
}
static void _freeSkin(NodeT *node) {
SDL_free(node->skinJoints);
SDL_free(node->skinInverseBind);
node->skinJoints = NULL;
node->skinInverseBind = NULL;
node->skinCount = 0;
}
// Drops a node's sprite: its textures, its private material, the quad off the node.
static void _freeSpriteNode(int32_t node) {
NodeT *n = &_scene.nodes[node];
SpriteNodeT *sprite;
int32_t x;
if (n->spriteSlot == NO_HANDLE) {
return;
}
sprite = &_scene.spriteNodes[n->spriteSlot];
for (x = 0; x < sprite->count; x++) {
_releaseTexture(&sprite->frames[x]);
}
SDL_free(sprite->frames);
if (materialValid(sprite->material)) {
_scene.materials[sprite->material].texture = NULL;
_scene.materials[sprite->material].textureBorrowed = false;
materialDelete(sprite->material);
}
memset(sprite, 0, sizeof(*sprite));
n->spriteSlot = NO_HANDLE;
if (n->mesh == _scene.quadMesh) {
n->mesh = NO_HANDLE;
n->material = NO_HANDLE;
}
}
static void _freeView(ViewT *view) {
_releaseTexture(&view->colour);
_releaseTexture(&view->depth);
_releaseTexture(&view->output);
memset(view, 0, sizeof(*view));
}
// Collects every 3D emitter's live particles into this frame's vertex list: emitters far to near,
// alpha-blended particles far to near within each, one run per frame texture.
static void _gatherParticles(Vec3T eye, Vec3T forward) {
EmitterViewT views[PARTICLE_DRAW_MAX];
DepthOrderT order[PARTICLE_DRAW_MAX];
DepthOrderT *particleOrder;
ParticleTexturesT *textures;
ParticleViewT *particle;
ParticleVertexT *vertex;
Vec3T origin;
int32_t emitters;
int32_t total = 0;
int32_t needed;
int32_t e;
int32_t i;
int32_t k;
int32_t frame;
int32_t first;
int32_t c;
// The two triangles of a quad as corner offsets.
static const float corners[PARTICLE_VERTICES][2] = { { -1.0f, -1.0f }, { 1.0f, -1.0f }, { 1.0f, 1.0f }, { -1.0f, -1.0f }, { 1.0f, 1.0f }, { -1.0f, 1.0f } };
_scene.particleRunCount = 0;
_scene.particleVertexCount = 0;
_scene.particleSoft = false;
_releaseParticleTextures(false);
emitters = particlesView3D(views, PARTICLE_DRAW_MAX);
for (e = 0; e < emitters; e++) {
origin = nodeValid(views[e].node) ? nodeGetWorldPosition(views[e].node) : eye;
order[e].depth = vec3Dot(vec3Subtract(origin, eye), forward);
order[e].index = e;
total += views[e].count * (1 + SDL_max(views[e].trailLength - 1, 0));
}
if (total == 0) {
return;
}
needed = total * PARTICLE_VERTICES;
if (_scene.particleVertexCapacity < needed) {
_scene.particleVertices = SDL_realloc(_scene.particleVertices, sizeof(ParticleVertexT) * (size_t)needed);
if (_scene.particleVertices == NULL) {
utilDie("Out of memory for %d particle vertices.", needed);
}
_scene.particleVertexCapacity = needed;
}
qsort(order, (size_t)emitters, sizeof(DepthOrderT), _compareDepthOrder);
for (e = 0; e < emitters; e++) {
EmitterViewT *view = &views[order[e].index];
textures = _particleTextures(view);
if (textures == NULL) {
continue;
}
if (_scene.particleOrderRoom < view->count) {
_scene.particleOrderRoom = SDL_max(view->count, _scene.particleOrderRoom * 2);
_scene.particleOrder = SDL_realloc(_scene.particleOrder, sizeof(DepthOrderT) * (size_t)_scene.particleOrderRoom);
if (_scene.particleOrder == NULL) {
utilDie("Out of memory sorting particles.");
}
}
particleOrder = _scene.particleOrder;
for (i = 0; i < view->count; i++) {
particleOrder[i].depth = vec3Dot(vec3Subtract(view->particles[i].position, eye), forward);
particleOrder[i].index = i;
}
if (view->blend == PARTICLE_ALPHA) {
qsort(particleOrder, (size_t)view->count, sizeof(DepthOrderT), _compareDepthOrder);
}
for (frame = 0; frame < textures->count; frame++) {
first = _scene.particleVertexCount;
for (k = 0; k < view->count; k++) {
particle = &view->particles[particleOrder[k].index];
if ((textures->count > 1) && (particle->frame != frame)) {
continue;
}
for (c = 0; c < PARTICLE_VERTICES; c++) {
vertex = &_scene.particleVertices[_scene.particleVertexCount++];
vertex->centre[0] = particle->position.x;
vertex->centre[1] = particle->position.y;
vertex->centre[2] = particle->position.z;
vertex->corner[0] = corners[c][0];
vertex->corner[1] = corners[c][1];
vertex->sizeAngle[0] = particle->size;
vertex->sizeAngle[1] = particle->angle;
vertex->colour[0] = _linearF(particle->colour[0]);
vertex->colour[1] = _linearF(particle->colour[1]);
vertex->colour[2] = _linearF(particle->colour[2]);
vertex->colour[3] = particle->colour[3];
vertex->uv[0] = corners[c][0] * 0.5f + 0.5f;
vertex->uv[1] = 0.5f - corners[c][1] * 0.5f;
}
_ribbon(view, particleOrder[k].index, eye);
}
if ((_scene.particleVertexCount > first) && (_scene.particleRunCount < PARTICLE_RUN_MAX)) {
_scene.particleRuns[_scene.particleRunCount].first = first;
_scene.particleRuns[_scene.particleRunCount].count = _scene.particleVertexCount - first;
_scene.particleRuns[_scene.particleRunCount].lit = view->lit;
_scene.particleRuns[_scene.particleRunCount].softness = view->softness;
_scene.particleRuns[_scene.particleRunCount].texture = textures->textures[frame];
_scene.particleRuns[_scene.particleRunCount].blend = view->blend;
_scene.particleRunCount++;
if (view->softness > 0.0f) {
_scene.particleSoft = true;
}
}
}
}
}
// A grid of columns x rows cells across sizeX by sizeZ, centred on the origin, each vertex lifted
// by its sample (0 to 1) times sizeY, or flat at y = 0 with NULL heights. Rows run from the far
// (-Z) edge or the near one, with UVs 0 to 1 across the whole either way (v = 1 at +Z, as
// meshPlane has it) and the triangles wound to face +Y. Normals come from the slopes.
static int32_t _gridMesh(const float *heights, int32_t columns, int32_t rows, float sizeX, float sizeY, float sizeZ, bool firstRowFar) {
SceneVertexT *vertices;
uint32_t *indices;
int32_t vertexCount = (columns + 1) * (rows + 1);
int32_t indexCount = columns * rows * 6;
float zStart = firstRowFar ? -sizeZ / 2.0f : sizeZ / 2.0f;
float zStep = firstRowFar ? sizeZ : -sizeZ;
int32_t x;
int32_t y;
int32_t mesh;
if ((columns < 1) || (rows < 1)) {
return NO_HANDLE;
}
vertices = SDL_calloc((size_t)vertexCount, sizeof(SceneVertexT));
indices = SDL_calloc((size_t)indexCount, sizeof(uint32_t));
if ((vertices == NULL) || (indices == NULL)) {
utilDie("Out of memory making a grid.");
}
for (y = 0; y <= rows; y++) {
for (x = 0; x <= columns; x++) {
float u = (float)x / (float)columns;
float v = (float)y / (float)rows;
float z = zStart + zStep * v;
float h = (heights != NULL) ? heights[y * (columns + 1) + x] * sizeY : 0.0f;
vertices[y * (columns + 1) + x] = _vertex(-sizeX / 2.0f + sizeX * u, h, z, 0.0f, 1.0f, 0.0f, u, 0.5f + z / sizeZ);
}
}
for (y = 0; y < rows; y++) {
for (x = 0; x < columns; x++) {
uint32_t a = (uint32_t)(y * (columns + 1) + x);
uint32_t b = a + 1;
uint32_t c = a + (uint32_t)(columns + 1);
uint32_t d = c + 1;
uint32_t *tri = &indices[(y * columns + x) * 6];
// Counter-clockwise seen from above: the winding flips with the row direction.
tri[0] = a;
tri[1] = firstRowFar ? d : b;
tri[2] = firstRowFar ? b : d;
tri[3] = a;
tri[4] = firstRowFar ? c : d;
tri[5] = firstRowFar ? d : c;
}
}
if (heights != NULL) {
sceneComputeNormals(vertices, vertexCount, indices, indexCount);
}
mesh = _addMesh(vertices, vertexCount, indices, indexCount, false);
SDL_free(vertices);
SDL_free(indices);
return mesh;
}
// A float as a 16-bit float (round toward zero; denormals flush to zero).
static uint16_t _half(float value) {
uint32_t bits;
uint32_t sign;
int32_t exponent;
uint32_t mantissa;
memcpy(&bits, &value, sizeof(bits));
sign = (bits >> 16) & 0x8000u;
exponent = (int32_t)((bits >> 23) & 0xFFu) - 127 + 15;
mantissa = bits & 0x7FFFFFu;
if (exponent <= 0) {
return (uint16_t)sign;
}
if (exponent >= 31) {
return (uint16_t)(sign | 0x7C00u);
}
return (uint16_t)(sign | ((uint32_t)exponent << 10) | (mantissa >> 13));
}
// Whether a draw needs morph targets this frame (which keeps it out of instanced batches).
static bool _hasMorphs(const NodeT *node, const MeshT *mesh) {
int32_t t;
if ((mesh->morphBuffer == NULL) || (node->morphCount != mesh->morphCount)) {
return false;
}
for (t = 0; t < node->morphCount; t++) {
if (node->morphWeights[t] != 0.0f) {
return true;
}
}
return false;
}
// The scene renders in linear light with headroom above white: 16-bit float where offered.
static SDL_GPUTextureFormat _hdrFormat(void) {
if (rgpuTextureSupportsFormat(_scene.device, SDL_GPU_TEXTUREFORMAT_R16G16B16A16_FLOAT, SDL_GPU_TEXTURETYPE_2D, SDL_GPU_TEXTUREUSAGE_COLOR_TARGET | SDL_GPU_TEXTUREUSAGE_SAMPLER)) {
return SDL_GPU_TEXTUREFORMAT_R16G16B16A16_FLOAT;
}
utilTrace("Scene: no 16-bit float render target; highlights will clip.");
return rgpuGetTextureFormatFromPixelFormat(SDL_PIXELFORMAT_BGRA32);
}
// Whether a draw is posed by a skin this frame (which keeps it out of instanced batches).
static bool _isSkinned(const NodeT *node, const MeshT *mesh) {
return mesh->skinned && (node->skinCount > 0);
}
// A surface of revolution around Y with flat caps: cylinders and cones. Sides get their own
// vertices so the caps can have flat normals.
static void _lathe(SceneVertexT **vertices, int32_t *vertexCount, uint32_t **indices, int32_t *indexCount, float bottomRadius, float topRadius, float height, int32_t segments) {
int32_t x;
int32_t v = 0;
int32_t i = 0;
float angle;
float c;
float s;
float half = height / 2.0f;
float slope;
float slopeLength;
int32_t bottomCentre;
int32_t topCentre;
SceneVertexT *verts;
uint32_t *idx;
// Sides: two rings of segments + 1 vertices (the seam is doubled for the uv wrap), then a
// centre and ring per cap.
*vertexCount = (segments + 1) * 2 + (segments + 1) * 2;
*indexCount = segments * 6 + segments * 3 * 2;
verts = SDL_calloc((size_t)*vertexCount, sizeof(SceneVertexT));
idx = SDL_calloc((size_t)*indexCount, sizeof(uint32_t));
if ((verts == NULL) || (idx == NULL)) {
utilDie("Out of memory building a mesh.");
}
// A zero-height, equal-radius lathe has no slope to take a normal from; keep the division finite.
slope = bottomRadius - topRadius;
slopeLength = SDL_max(sqrtf(slope * slope + height * height), MATH_EPSILON);
for (x = 0; x <= segments; x++) {
angle = (float)x / (float)segments * 2.0f * PI;
c = cosf(angle);
s = sinf(angle);
verts[v++] = _vertex(c * bottomRadius, -half, s * bottomRadius, c * height / slopeLength, slope / slopeLength, s * height / slopeLength, (float)x / (float)segments, 1.0f);
verts[v++] = _vertex(c * topRadius, half, s * topRadius, c * height / slopeLength, slope / slopeLength, s * height / slopeLength, (float)x / (float)segments, 0.0f);
}
for (x = 0; x < segments; x++) {
// Counter-clockwise seen from outside: the ring runs with the angle, +X toward +Z.
idx[i++] = (uint32_t)(x * 2);
idx[i++] = (uint32_t)(x * 2 + 2);
idx[i++] = (uint32_t)(x * 2 + 1);
idx[i++] = (uint32_t)(x * 2 + 1);
idx[i++] = (uint32_t)(x * 2 + 2);
idx[i++] = (uint32_t)(x * 2 + 3);
}
bottomCentre = v;
verts[v++] = _vertex(0.0f, -half, 0.0f, 0.0f, -1.0f, 0.0f, 0.5f, 0.5f);
for (x = 0; x < segments; x++) {
angle = (float)x / (float)segments * 2.0f * PI;
verts[v++] = _vertex(cosf(angle) * bottomRadius, -half, sinf(angle) * bottomRadius, 0.0f, -1.0f, 0.0f, 0.5f + cosf(angle) / 2.0f, 0.5f + sinf(angle) / 2.0f);
}
for (x = 0; x < segments; x++) {
idx[i++] = (uint32_t)bottomCentre;
idx[i++] = (uint32_t)(bottomCentre + 1 + x);
idx[i++] = (uint32_t)(bottomCentre + 1 + (x + 1) % segments);
}
topCentre = v;
verts[v++] = _vertex(0.0f, half, 0.0f, 0.0f, 1.0f, 0.0f, 0.5f, 0.5f);
for (x = 0; x < segments; x++) {
angle = (float)x / (float)segments * 2.0f * PI;
verts[v++] = _vertex(cosf(angle) * topRadius, half, sinf(angle) * topRadius, 0.0f, 1.0f, 0.0f, 0.5f + cosf(angle) / 2.0f, 0.5f - sinf(angle) / 2.0f);
}
for (x = 0; x < segments; x++) {
idx[i++] = (uint32_t)topCentre;
idx[i++] = (uint32_t)(topCentre + 1 + (x + 1) % segments);
idx[i++] = (uint32_t)(topCentre + 1 + x);
}
*vertices = verts;
*indices = idx;
}
// An sRGB byte as linear light.
static float _linear(uint8_t value) {
return _linearF(value / COLOUR_MAX);
}
// An sRGB fraction as linear light.
static float _linearF(float value) {
if (value <= 0.04045f) {
return value / 12.92f;
}
return SDL_powf((value + 0.055f) / 1.055f, 2.4f);
}
// Colour maps are sRGB (the sampler decodes them); data maps are not.
static bool _mapIsColour(MaterialMapE map) {
return (map == MAP_BASE) || (map == MAP_EMISSIVE);
}
// Sizes the node's weight list to its mesh's targets (weights start at 0).
static void _matchMorphWeights(NodeT *node) {
int32_t count = meshValid(node->mesh) ? _scene.meshes[node->mesh].morphCount : 0;
if (count == node->morphCount) {
return;
}
_freeMorphWeights(node);
if (count > 0) {
node->morphWeights = SDL_calloc((size_t)count, sizeof(float));
if (node->morphWeights == NULL) {
utilDie("Out of memory allocating morph weights.");
}
node->morphCount = count;
}
}
// White, half rough, no texture: the look of a new material and of a node without one.
static void _materialDefaults(MaterialT *material) {
memset(material, 0, sizeof(*material));
material->baseColor.x = 1.0f;
material->baseColor.y = 1.0f;
material->baseColor.z = 1.0f;
material->baseColor.w = 1.0f;
material->roughness = DEFAULT_ROUGHNESS;
material->tilingU = 1.0f;
material->tilingV = 1.0f;
material->feed = NO_HANDLE;
material->view = NO_HANDLE;
material->gui = NO_HANDLE;
}
// Puts an uploaded texture (or NULL, clearing) in one of a material's map slots, releasing what
// was there, with the strength that goes with the map: the normal map's bump scale or the
// occlusion map's blend. A base texture also drops any video feed or view the material showed.
static void _materialPlace(MaterialT *material, MaterialMapE map, SDL_GPUTexture *texture, float strength) {
SDL_GPUTexture **slot;
switch (map) {
case MAP_NORMAL:
slot = &material->normalMap;
material->normalStrength = SDL_max(strength, 0.0f);
break;
case MAP_OCCLUSION:
slot = &material->occlusionMap;
material->occlusionStrength = SDL_clamp(strength, 0.0f, 1.0f);
break;
case MAP_METALLIC_ROUGHNESS:
slot = &material->metallicRoughnessMap;
break;
case MAP_EMISSIVE:
slot = &material->emissiveMap;
break;
default:
slot = &material->texture;
break;
}
if (map == MAP_BASE) {
_releaseMaterialBase(material);
material->feed = NO_HANDLE;
material->view = NO_HANDLE;
} else {
_releaseTexture(slot);
}
*slot = texture;
}
// What the fragment shader samples for a material: its video feed, its rendered view, its image,
// or NULL.
static SDL_GPUTexture *_materialTexture(const MaterialT *material) {
if ((material->feed != NO_HANDLE) && (material->feed < _scene.feedCount) && _scene.feeds[material->feed].used) {
return _scene.feeds[material->feed].gpu;
}
if ((material->view != NO_HANDLE) && (material->view < MAX_VIEWS) && _scene.views[material->view].used) {
return _scene.views[material->view].output;
}
if ((material->gui != NO_HANDLE) && (guiTexture(material->gui) != NULL)) {
return guiTexture(material->gui);
}
return material->texture;
}
// Levels in a full mipmap chain down to 1x1.
static uint32_t _mipLevels(int32_t width, int32_t height) {
uint32_t levels = 1;
int32_t size = SDL_max(width, height);
while (size > 1) {
size >>= 1;
levels++;
}
return levels;
}
// A draw's model matrix: the node's world, with a sprite's size folded in (its quad is a unit
// square, so the node's own scale stays free). Bounds and instances both come from this.
static Mat4T _modelOf(const NodeT *node) {
if (node->spriteSlot != NO_HANDLE) {
const SpriteNodeT *sprite = &_scene.spriteNodes[node->spriteSlot];
return mat4Multiply(node->world, mat4Compose(vec3(0.0f, 0.0f, 0.0f), quatIdentity(), vec3(sprite->width, sprite->height, 1.0f)));
}
return node->world;
}
// The blended draws (those after the opaque ones) back to front from an eye, as the order the
// camera's pass draws them in; the draws and their instances stay where they are.
static void _orderBlended(Vec3T eye, int32_t drawCount) {
int32_t count = drawCount - _scene.opaqueCount;
int32_t x;
if (_scene.blendedOrderRoom < count) {
_scene.blendedOrderRoom = SDL_max(count, _scene.blendedOrderRoom * 2);
_scene.blendedOrder = SDL_realloc(_scene.blendedOrder, sizeof(DepthOrderT) * (size_t)_scene.blendedOrderRoom);
if (_scene.blendedOrder == NULL) {
utilDie("Out of memory ordering blended draws.");
}
}
for (x = 0; x < count; x++) {
const float *m = _scene.nodes[_scene.draws[_scene.opaqueCount + x].node].world.m;
_scene.blendedOrder[x].index = _scene.opaqueCount + x;
_scene.blendedOrder[x].depth = vec3Length(vec3Subtract(vec3(m[12], m[13], m[14]), eye));
}
if (count > 0) {
qsort(_scene.blendedOrder, (size_t)count, sizeof(DepthOrderT), _compareDepthOrder);
}
}
// The GPU textures for an emitter's frames, made on first sight and remade when the frames change.
static ParticleTexturesT *_particleTextures(const EmitterViewT *view) {
ParticleTexturesT *entry = NULL;
int32_t x;
for (x = 0; x < _scene.particleTextureCount; x++) {
if (_scene.particleTextures[x].id == view->id) {
entry = &_scene.particleTextures[x];
break;
}
}
if ((entry != NULL) && (entry->version != view->textureVersion)) {
for (x = 0; x < entry->count; x++) {
rgpuReleaseTexture(_scene.device, entry->textures[x]);
}
SDL_free(entry->textures);
entry->textures = NULL;
entry->count = 0;
entry->version = 0;
}
if ((entry != NULL) && (entry->textures != NULL)) {
return entry;
}
if (entry == NULL) {
_scene.particleTextures = SDL_realloc(_scene.particleTextures, sizeof(ParticleTexturesT) * (size_t)(_scene.particleTextureCount + 1));
if (_scene.particleTextures == NULL) {
utilDie("Out of memory for particle textures.");
}
entry = &_scene.particleTextures[_scene.particleTextureCount++];
memset(entry, 0, sizeof(*entry));
entry->id = view->id;
}
entry->textures = SDL_calloc((size_t)view->frameCount, sizeof(SDL_GPUTexture *));
if (entry->textures == NULL) {
utilDie("Out of memory for particle textures.");
}
for (x = 0; x < view->frameCount; x++) {
entry->textures[x] = _uploadTexture(view->frames[x], true);
if (entry->textures[x] == NULL) {
while (x > 0) {
x--;
rgpuReleaseTexture(_scene.device, entry->textures[x]);
}
SDL_free(entry->textures);
entry->textures = NULL;
return NULL;
}
}
entry->count = view->frameCount;
entry->version = view->textureVersion;
return entry;
}
// The first of the wanted depth formats the device offers for the usage, else 16-bit.
static SDL_GPUTextureFormat _pickDepthFormat(const SDL_GPUTextureFormat *wanted, int32_t count, SDL_GPUTextureUsageFlags usage) {
int32_t x;
for (x = 0; x < count; x++) {
if (rgpuTextureSupportsFormat(_scene.device, wanted[x], SDL_GPU_TEXTURETYPE_2D, usage)) {
return wanted[x];
}
}
return SDL_GPU_TEXTUREFORMAT_D16_UNORM;
}
// The PIPELINE_* variant a node's mesh and material call for.
static int32_t _pipelineVariant(int32_t node) {
int32_t variant = 0;
MaterialT *material;
if (_isSkinned(&_scene.nodes[node], &_scene.meshes[_scene.nodes[node].mesh])) {
variant |= PIPELINE_SKINNED;
}
if (_scene.nodes[node].material != NO_HANDLE) {
material = &_scene.materials[_scene.nodes[node].material];
if (material->blend) {
variant |= PIPELINE_BLEND;
}
if (material->doubleSided) {
variant |= PIPELINE_TWO_SIDED;
}
if (material->occluder) {
variant |= PIPELINE_OCCLUDER;
}
}
return variant;
}
// The camera's projection for a target of the given size.
static Mat4T _projectionFor(int32_t width, int32_t height) {
float aspect = (height > 0) ? (float)width / (float)height : 1.0f;
if (_scene.perspective) {
return mat4Perspective(_scene.fov, aspect, _scene.near, _scene.far);
}
return mat4Orthographic(_scene.orthoHeight * aspect, _scene.orthoHeight, _scene.near, _scene.far);
}
// A two-triangle quad centred on the origin: width along +X, height along down (the edge where
// v = 1, so a picture on it reads upright), facing normal.
static int32_t _quadMesh(float width, float height, Vec3T down, Vec3T normal) {
SceneVertexT vertices[4];
uint32_t indices[6] = { 0, 1, 2, 0, 2, 3 };
float w = width / 2.0f;
float h = height / 2.0f;
Vec3T corner;
int32_t x;
// Counter-clockwise seen from the front: each corner's (right, down) signs and its uv.
static const float corners[4][4] = { { -1.0f, 1.0f, 0.0f, 1.0f }, { 1.0f, 1.0f, 1.0f, 1.0f }, { 1.0f, -1.0f, 1.0f, 0.0f }, { -1.0f, -1.0f, 0.0f, 0.0f } };
for (x = 0; x < 4; x++) {
corner = vec3Add(vec3(corners[x][0] * w, 0.0f, 0.0f), vec3Scale(down, corners[x][1] * h));
vertices[x] = _vertex(corner.x, corner.y, corner.z, normal.x, normal.y, normal.z, corners[x][2], corners[x][3]);
}
return _addMesh(vertices, 4, indices, 6, false);
}
// Keeps a texture's size so releasing it can take it off the total.
static void _recordTexture(SDL_GPUTexture *texture, size_t bytes) {
if (texture == NULL) {
return;
}
if (_scene.sizedCount == _scene.sizedCapacity) {
_scene.sizedCapacity += TEXTURE_SIZES_STEP;
_scene.sizedTextures = SDL_realloc(_scene.sizedTextures, sizeof(SDL_GPUTexture *) * (size_t)_scene.sizedCapacity);
_scene.sizedBytes = SDL_realloc(_scene.sizedBytes, sizeof(size_t) * (size_t)_scene.sizedCapacity);
if ((_scene.sizedTextures == NULL) || (_scene.sizedBytes == NULL)) {
utilDie("Out of memory tracking textures.");
}
}
_scene.sizedTextures[_scene.sizedCount] = texture;
_scene.sizedBytes[_scene.sizedCount] = bytes;
_scene.sizedCount++;
_scene.textureBytes += (int64_t)bytes;
}
// A material's base texture, unless it is a sprite node's (which keeps it).
static void _releaseMaterialBase(MaterialT *material) {
if (material->textureBorrowed) {
material->texture = NULL;
material->textureBorrowed = false;
return;
}
_releaseTexture(&material->texture);
}
// Drops the textures of emitters that no longer exist, or all of them.
static void _releaseParticleTextures(bool all) {
int32_t x = 0;
int32_t f;
while (x < _scene.particleTextureCount) {
ParticleTexturesT *entry = &_scene.particleTextures[x];
if (all || !emitterValid(entry->id)) {
for (f = 0; f < entry->count; f++) {
rgpuReleaseTexture(_scene.device, entry->textures[f]);
}
SDL_free(entry->textures);
_scene.particleTextureCount--;
*entry = _scene.particleTextures[_scene.particleTextureCount];
continue;
}
x++;
}
if (all) {
SDL_free(_scene.particleTextures);
_scene.particleTextures = NULL;
}
}
static void _releasePipeline(SDL_GPUGraphicsPipeline **pipeline) {
if (*pipeline != NULL) {
rgpuReleaseGraphicsPipeline(_scene.device, *pipeline);
*pipeline = NULL;
}
}
static void _releaseTexture(SDL_GPUTexture **texture) {
int32_t x;
if (*texture == NULL) {
return;
}
for (x = 0; x < _scene.sizedCount; x++) {
if (_scene.sizedTextures[x] == *texture) {
_scene.textureBytes -= (int64_t)_scene.sizedBytes[x];
_scene.sizedCount--;
_scene.sizedTextures[x] = _scene.sizedTextures[_scene.sizedCount];
_scene.sizedBytes[x] = _scene.sizedBytes[_scene.sizedCount];
break;
}
}
rgpuReleaseTexture(_scene.device, *texture);
*texture = NULL;
}
// One camera's render: its particles gathered and uploaded, draws outside its view culled and
// the blended ones ordered from its eye, the depth copy for soft particles when wanted, the main
// pass (sky, meshes, particles) into its HDR target and the post pass into its display texture.
static void _renderCamera(SDL_GPUCommandBuffer *commands, const CameraFrameT *frame, FragmentUniformsT *uniforms, int32_t drawCount) {
SDL_GPUColorTargetInfo colour;
SDL_GPUDepthStencilTargetInfo depth;
SDL_GPURenderPass *pass;
bool *culled = _skipScratch(drawCount);
int32_t x;
uniforms->cameraPosition[0] = frame->eye.x;
uniforms->cameraPosition[1] = frame->eye.y;
uniforms->cameraPosition[2] = frame->eye.z;
uniforms->cameraPosition[3] = 1.0f;
uniforms->cameraForward[0] = frame->forward.x;
uniforms->cameraForward[1] = frame->forward.y;
uniforms->cameraForward[2] = frame->forward.z;
uniforms->cameraForward[3] = frame->main ? 1.0f : 0.0f;
_gatherParticles(frame->eye, frame->forward);
_uploadParticles(commands);
_cullDraws(&frame->viewProjection, drawCount, culled);
_orderBlended(frame->eye, drawCount);
if (frame->main) {
_scene.statTotal = drawCount;
_scene.statDrawn = 0;
_scene.statBatches = 0;
for (x = 0; x < drawCount; x++) {
if (!culled[x]) {
_scene.statDrawn++;
}
}
}
// Soft particles need the camera's depth before the main pass: a depth-only pass of every
// opaque mesh into the single-sample copy the particle shader samples. With nothing to draw
// the clear alone matters: the copy must not hold last frame's (or no) depth when sampled.
if (_scene.particleSoft && (frame->softDepth != NULL)) {
memset(&depth, 0, sizeof(depth));
depth.texture = frame->softDepth;
depth.clear_depth = 1.0f;
depth.load_op = SDL_GPU_LOADOP_CLEAR;
depth.store_op = SDL_GPU_STOREOP_STORE;
depth.stencil_load_op = SDL_GPU_LOADOP_DONT_CARE;
depth.stencil_store_op = SDL_GPU_STOREOP_DONT_CARE;
_scene.depthPrepass = true;
pass = rgpuBeginRenderPass(commands, NULL, 0, &depth);
_drawList(commands, pass, drawCount, &frame->viewProjection, frame, true, false, culled, NULL, SAMPLE_SET_SINGLE);
rgpuEndRenderPass(pass);
_scene.depthPrepass = false;
}
// The main pass.
memset(&colour, 0, sizeof(colour));
colour.clear_color = _scene.background;
colour.load_op = SDL_GPU_LOADOP_CLEAR;
if (frame->multisampled != NULL) {
colour.texture = frame->multisampled;
colour.resolve_texture = frame->colour;
colour.store_op = SDL_GPU_STOREOP_RESOLVE;
} else {
colour.texture = frame->colour;
colour.store_op = SDL_GPU_STOREOP_STORE;
}
memset(&depth, 0, sizeof(depth));
depth.texture = frame->depth;
depth.clear_depth = 1.0f;
depth.load_op = SDL_GPU_LOADOP_CLEAR;
depth.store_op = SDL_GPU_STOREOP_DONT_CARE;
depth.stencil_load_op = SDL_GPU_LOADOP_DONT_CARE;
depth.stencil_store_op = SDL_GPU_STOREOP_DONT_CARE;
pass = rgpuBeginRenderPass(commands, &colour, 1, &depth);
_drawSky(commands, pass, frame);
_drawList(commands, pass, drawCount, &frame->viewProjection, frame, false, false, culled, uniforms, frame->sampleSet);
_drawParticles(commands, pass, frame, uniforms);
_drawLines(commands, pass, frame);
rgpuEndRenderPass(pass);
_drawPost(commands, frame);
}
// A particle's trail as a ribbon of quads through its recorded points, each facing the eye and
// fading toward the tail, textured by the middle column of the particle's picture so the disc's
// soft edge becomes the ribbon's. Appended as plain vertices (no corner offset) of the same run.
static void _ribbon(const EmitterViewT *view, int32_t index, Vec3T eye) {
const ParticleViewT *particle = &view->particles[index];
const float *points;
int32_t count;
int32_t j;
int32_t c;
float half = view->trailWidth * 0.5f;
Vec3T previous;
Vec3T sidePrevious = vec3(0.0f, 0.0f, 0.0f);
// Two triangles between successive points: corners of the quad as (point, side) pairs.
static const int32_t quad[PARTICLE_VERTICES][2] = { { 0, -1 }, { 1, -1 }, { 1, 1 }, { 0, -1 }, { 1, 1 }, { 0, 1 } };
if ((view->trailLength < 2) || (view->trailCounts == NULL)) {
return;
}
count = view->trailCounts[index];
if (count < 2) {
return;
}
points = view->trailPoints + (size_t)index * (size_t)view->trailLength * 3;
previous = vec3Add(view->trailOffset, vec3(points[0], points[1], points[2]));
for (j = 1; j < count; j++) {
Vec3T current = vec3Add(view->trailOffset, vec3(points[j * 3], points[j * 3 + 1], points[j * 3 + 2]));
Vec3T along = vec3Subtract(current, previous);
Vec3T side = vec3Cross(along, vec3Subtract(eye, current));
float fade0 = (float)(j - 1) / (float)(count - 1);
float fade1 = (float)j / (float)(count - 1);
if (vec3Length(side) > 0.0f) {
side = vec3Scale(vec3Normalize(side), half);
} else {
side = sidePrevious;
}
if (j == 1) {
sidePrevious = side;
}
for (c = 0; c < PARTICLE_VERTICES; c++) {
ParticleVertexT *vertex = &_scene.particleVertices[_scene.particleVertexCount++];
Vec3T point = quad[c][0] ? current : previous;
Vec3T edge = quad[c][0] ? side : sidePrevious;
Vec3T world = vec3Add(point, vec3Scale(edge, (float)quad[c][1]));
vertex->centre[0] = world.x;
vertex->centre[1] = world.y;
vertex->centre[2] = world.z;
vertex->corner[0] = 0.0f;
vertex->corner[1] = 0.0f;
vertex->sizeAngle[0] = 0.0f;
vertex->sizeAngle[1] = 0.0f;
vertex->colour[0] = _linearF(particle->colour[0]);
vertex->colour[1] = _linearF(particle->colour[1]);
vertex->colour[2] = _linearF(particle->colour[2]);
vertex->colour[3] = particle->colour[3] * (quad[c][0] ? fade1 : fade0);
vertex->uv[0] = 0.5f;
vertex->uv[1] = (quad[c][1] < 0) ? 0.0f : 1.0f;
}
previous = current;
sidePrevious = side;
}
}
// Whether opaque draw b can ride in draw a's instanced batch: the same mesh and material (so the
// same pipeline), nothing per-draw beyond the matrices (no skin, no morphs), not skipped, and in
// a shadow pass a caster.
static bool _sameBatch(int32_t a, int32_t b, bool shadowPass, const bool *skip) {
const NodeT *na = &_scene.nodes[_scene.draws[a].node];
const NodeT *nb = &_scene.nodes[_scene.draws[b].node];
const MeshT *mesh;
if ((na->mesh != nb->mesh) || (na->material != nb->material) || ((skip != NULL) && skip[b])) {
return false;
}
mesh = &_scene.meshes[nb->mesh];
if (_isSkinned(nb, mesh) || _hasMorphs(nb, mesh)) {
return false;
}
if (shadowPass && !nb->shadowCaster && !_scene.depthPrepass) {
return false;
}
return true;
}
// The sample count a pipeline set draws with: the window's targets may be multisampled, a view's
// never are.
static SDL_GPUSampleCount _sampleCountOf(int32_t sampleSet) {
return (sampleSet == SAMPLE_SET_MULTI) ? _scene.sampleCount : SDL_GPU_SAMPLECOUNT_1;
}
// A bilinear sample of an equirectangular image along a direction (the image's centre column
// faces -Z, its top is +Y).
static Vec3T _sampleEquirect(const float *rgb, int32_t width, int32_t height, Vec3T direction) {
Vec3T d = vec3Normalize(direction);
float u = (0.5f + SDL_atan2f(d.x, -d.z) / (2.0f * SDL_PI_F)) * (float)width - 0.5f;
float v = (SDL_acosf(SDL_clamp(d.y, -1.0f, 1.0f)) / SDL_PI_F) * (float)height - 0.5f;
int32_t x0 = (int32_t)SDL_floorf(u);
int32_t y0 = (int32_t)SDL_floorf(v);
float fx = u - (float)x0;
float fy = v - (float)y0;
int32_t x1 = ((x0 + 1) % width + width) % width;
int32_t y1 = SDL_clamp(y0 + 1, 0, height - 1);
Vec3T sum = vec3(0.0f, 0.0f, 0.0f);
int32_t xs[2];
int32_t ys[2];
float ws[2][2];
int32_t i;
int32_t j;
x0 = (x0 % width + width) % width;
y0 = SDL_clamp(y0, 0, height - 1);
xs[0] = x0;
xs[1] = x1;
ys[0] = y0;
ys[1] = y1;
ws[0][0] = (1.0f - fx) * (1.0f - fy);
ws[1][0] = fx * (1.0f - fy);
ws[0][1] = (1.0f - fx) * fy;
ws[1][1] = fx * fy;
for (j = 0; j < 2; j++) {
for (i = 0; i < 2; i++) {
const float *pixel = rgb + ((size_t)ys[j] * (size_t)width + (size_t)xs[i]) * 3;
sum = vec3Add(sum, vec3Scale(vec3(pixel[0], pixel[1], pixel[2]), ws[i][j]));
}
}
return sum;
}
// A depth format the shadow map can be both rendered into and sampled from. 16-bit comes before
// 24-bit here: the shadow compare needs no more, and the smaller map samples faster.
static SDL_GPUTextureFormat _shadowFormat(void) {
static const SDL_GPUTextureFormat wanted[] = { SDL_GPU_TEXTUREFORMAT_D32_FLOAT, SDL_GPU_TEXTUREFORMAT_D16_UNORM, SDL_GPU_TEXTUREFORMAT_D24_UNORM };
return _pickDepthFormat(wanted, (int32_t)SDL_arraysize(wanted), SDL_GPU_TEXTUREUSAGE_DEPTH_STENCIL_TARGET | SDL_GPU_TEXTUREUSAGE_SAMPLER);
}
// The frame's culling flags, one per draw, cleared; grown to the most draws seen.
static bool *_skipScratch(int32_t drawCount) {
if (_scene.skipRoom < drawCount) {
_scene.skipRoom = SDL_max(drawCount, _scene.skipRoom * 2);
_scene.skip = SDL_realloc(_scene.skip, sizeof(bool) * (size_t)_scene.skipRoom);
if (_scene.skip == NULL) {
utilDie("Out of memory culling the scene.");
}
}
if (drawCount > 0) {
memset(_scene.skip, 0, sizeof(bool) * (size_t)drawCount);
}
return _scene.skip;
}
// A 1x1 data texture of one colour, for the maps a material does not have.
static SDL_GPUTexture *_solidTexture(uint8_t r, uint8_t g, uint8_t b) {
SDL_Surface *pixel = SDL_CreateSurface(1, 1, SDL_PIXELFORMAT_RGBA32);
SDL_GPUTexture *texture = NULL;
if (pixel == NULL) {
return NULL;
}
SDL_FillSurfaceRect(pixel, NULL, SDL_MapSurfaceRGBA(pixel, r, g, b, SDL_ALPHA_OPAQUE));
texture = _uploadTexture(pixel, false);
SDL_DestroySurface(pixel);
return texture;
}
// Opens a staged upload of bytes: a transfer buffer mapped at staging->mapped for the caller to
// fill. False, with the error traced, when the GPU refuses.
static bool _stageBegin(StagingT *staging, uint32_t bytes) {
SDL_GPUTransferBufferCreateInfo info;
memset(staging, 0, sizeof(*staging));
memset(&info, 0, sizeof(info));
info.usage = SDL_GPU_TRANSFERBUFFERUSAGE_UPLOAD;
info.size = bytes;
staging->transfer = rgpuCreateTransferBuffer(_scene.device, &info);
if (staging->transfer == NULL) {
utilTrace("Scene: %s", SDL_GetError());
return false;
}
staging->mapped = rgpuMapTransferBuffer(_scene.device, staging->transfer, false);
if (staging->mapped == NULL) {
utilTrace("Scene: %s", SDL_GetError());
rgpuReleaseTransferBuffer(_scene.device, staging->transfer);
staging->transfer = NULL;
return false;
}
return true;
}
// Unmaps the filled transfer buffer and opens the copy pass (staging->pass) the caller issues its
// uploads into. False, with everything released, when no command buffer could be had.
static bool _stageCopy(StagingT *staging) {
rgpuUnmapTransferBuffer(_scene.device, staging->transfer);
staging->mapped = NULL;
staging->commands = rgpuAcquireCommandBuffer(_scene.device);
if (staging->commands == NULL) {
utilTrace("Scene: %s", SDL_GetError());
rgpuReleaseTransferBuffer(_scene.device, staging->transfer);
staging->transfer = NULL;
return false;
}
staging->pass = rgpuBeginCopyPass(staging->commands);
return true;
}
// Ends the copy pass, generates the mip chain of mipmaps (when given), submits and releases.
static void _stageEnd(StagingT *staging, SDL_GPUTexture *mipmaps) {
rgpuEndCopyPass(staging->pass);
if (mipmaps != NULL) {
rgpuGenerateMipmapsForTexture(staging->commands, mipmaps);
}
rgpuSubmitCommandBuffer(staging->commands);
rgpuReleaseTransferBuffer(_scene.device, staging->transfer);
memset(staging, 0, sizeof(*staging));
}
// One upload out of the staged bytes at offset into a level and layer of a texture.
static void _stageTexture(const StagingT *staging, uint32_t offset, SDL_GPUTexture *texture, uint32_t level, uint32_t layer, uint32_t width, uint32_t height) {
SDL_GPUTextureTransferInfo source;
SDL_GPUTextureRegion region;
memset(&source, 0, sizeof(source));
memset(&region, 0, sizeof(region));
source.transfer_buffer = staging->transfer;
source.offset = offset;
region.texture = texture;
region.mip_level = level;
region.layer = layer;
region.w = width;
region.h = height;
region.d = 1;
rgpuUploadToTexture(staging->pass, &source, &region, false);
}
// Marks a mesh's contents as changed, so caches keyed on them (_cubeHash) miss.
static void _stampMesh(MeshT *mesh) {
_scene.meshVersion++;
mesh->version = _scene.meshVersion;
}
// World matrices and inherited visibility, depth first.
static void _updateWorld(int32_t node, const Mat4T *parentWorld, bool parentVisible) {
NodeT *n = &_scene.nodes[node];
int32_t child;
n->world = mat4Multiply(*parentWorld, mat4Compose(n->translation, n->rotation, n->scale));
n->worldVisible = parentVisible && n->visible;
for (child = n->firstChild; child != NO_HANDLE; child = _scene.nodes[child].nextSibling) {
_updateWorld(child, &n->world, n->worldVisible);
}
}
// Copies data into a new GPU buffer through a transfer buffer.
static SDL_GPUBuffer *_uploadBuffer(SDL_GPUBufferUsageFlags usage, const void *data, uint32_t size) {
SDL_GPUBufferCreateInfo info;
SDL_GPUTransferBufferLocation source;
SDL_GPUBufferRegion region;
SDL_GPUBuffer *buffer;
StagingT staging;
memset(&info, 0, sizeof(info));
info.usage = usage;
info.size = size;
buffer = rgpuCreateBuffer(_scene.device, &info);
if (buffer == NULL) {
utilTrace("Scene: %s", SDL_GetError());
return NULL;
}
if (!_stageBegin(&staging, size)) {
rgpuReleaseBuffer(_scene.device, buffer);
return NULL;
}
memcpy(staging.mapped, data, size);
if (!_stageCopy(&staging)) {
rgpuReleaseBuffer(_scene.device, buffer);
return NULL;
}
memset(&source, 0, sizeof(source));
memset(&region, 0, sizeof(region));
source.transfer_buffer = staging.transfer;
region.buffer = buffer;
region.size = size;
rgpuUploadToBuffer(staging.pass, &source, &region, false);
_stageEnd(&staging, NULL);
return buffer;
}
// A block-compressed (or, as the fallback, RGBA) texture from a transcoded KTX2 image, every mip
// level uploaded as it came (no generation: compressed formats cannot be rendered into).
static SDL_GPUTexture *_uploadCompressed(const Ktx2ImageT *image, bool srgb) {
SDL_GPUTextureCreateInfo info;
SDL_GPUTexture *texture;
StagingT staging;
size_t total = 0;
size_t offset = 0;
int32_t level;
for (level = 0; level < image->levelCount; level++) {
total += image->levels[level].size;
}
memset(&info, 0, sizeof(info));
info.type = SDL_GPU_TEXTURETYPE_2D;
switch (image->format) {
case KTX2_BC7:
info.format = srgb ? SDL_GPU_TEXTUREFORMAT_BC7_RGBA_UNORM_SRGB : SDL_GPU_TEXTUREFORMAT_BC7_RGBA_UNORM;
break;
case KTX2_ASTC:
info.format = srgb ? SDL_GPU_TEXTUREFORMAT_ASTC_4x4_UNORM_SRGB : SDL_GPU_TEXTUREFORMAT_ASTC_4x4_UNORM;
break;
case KTX2_BC3:
info.format = srgb ? SDL_GPU_TEXTUREFORMAT_BC3_RGBA_UNORM_SRGB : SDL_GPU_TEXTUREFORMAT_BC3_RGBA_UNORM;
break;
case KTX2_ETC2:
info.format = srgb ? RGPU_TEXTUREFORMAT_ETC2_RGBA8_UNORM_SRGB : RGPU_TEXTUREFORMAT_ETC2_RGBA8_UNORM;
break;
default:
info.format = srgb ? SDL_GPU_TEXTUREFORMAT_R8G8B8A8_UNORM_SRGB : SDL_GPU_TEXTUREFORMAT_R8G8B8A8_UNORM;
break;
}
info.usage = SDL_GPU_TEXTUREUSAGE_SAMPLER;
info.width = (Uint32)image->width;
info.height = (Uint32)image->height;
info.layer_count_or_depth = 1;
info.num_levels = (Uint32)image->levelCount;
info.sample_count = SDL_GPU_SAMPLECOUNT_1;
texture = rgpuCreateTexture(_scene.device, &info);
if (texture == NULL) {
utilTrace("Scene: compressed texture: %s", SDL_GetError());
return NULL;
}
if (!_stageBegin(&staging, (uint32_t)total)) {
rgpuReleaseTexture(_scene.device, texture);
return NULL;
}
for (level = 0; level < image->levelCount; level++) {
memcpy((uint8_t *)staging.mapped + offset, image->levels[level].data, image->levels[level].size);
offset += image->levels[level].size;
}
if (!_stageCopy(&staging)) {
rgpuReleaseTexture(_scene.device, texture);
return NULL;
}
offset = 0;
for (level = 0; level < image->levelCount; level++) {
_stageTexture(&staging, (uint32_t)offset, texture, (uint32_t)level, 0, (uint32_t)image->levels[level].width, (uint32_t)image->levels[level].height);
offset += image->levels[level].size;
}
_stageEnd(&staging, NULL);
_recordTexture(texture, total);
return texture;
}
// A 16-bit float RGBA cube texture from six face-sized images of pixels in +X, -X, +Y, -Y, +Z,
// -Z order, with its mip chain generated on the GPU.
static SDL_GPUTexture *_uploadCube(const uint16_t *pixels, int32_t face) {
SDL_GPUTextureCreateInfo info;
SDL_GPUTexture *texture;
StagingT staging;
uint32_t faceBytes = (uint32_t)face * (uint32_t)face * CUBE_CHANNELS * HALF_BYTES;
uint32_t levels = _mipLevels(face, face);
int32_t f;
memset(&info, 0, sizeof(info));
info.type = SDL_GPU_TEXTURETYPE_CUBE;
info.format = SDL_GPU_TEXTUREFORMAT_R16G16B16A16_FLOAT;
info.usage = SDL_GPU_TEXTUREUSAGE_SAMPLER | ((levels > 1) ? SDL_GPU_TEXTUREUSAGE_COLOR_TARGET : 0);
info.width = (Uint32)face;
info.height = (Uint32)face;
info.layer_count_or_depth = CUBE_FACES;
info.num_levels = levels;
info.sample_count = SDL_GPU_SAMPLECOUNT_1;
texture = rgpuCreateTexture(_scene.device, &info);
if (texture == NULL) {
utilTrace("Scene: cube texture: %s", SDL_GetError());
return NULL;
}
if (!_stageBegin(&staging, faceBytes * CUBE_FACES)) {
rgpuReleaseTexture(_scene.device, texture);
return NULL;
}
memcpy(staging.mapped, pixels, faceBytes * CUBE_FACES);
if (!_stageCopy(&staging)) {
rgpuReleaseTexture(_scene.device, texture);
return NULL;
}
for (f = 0; f < CUBE_FACES; f++) {
_stageTexture(&staging, faceBytes * (uint32_t)f, texture, 0, (uint32_t)f, (uint32_t)face, (uint32_t)face);
}
_stageEnd(&staging, (levels > 1) ? texture : NULL);
_recordTexture(texture, (size_t)faceBytes * CUBE_FACES * ((levels > 1) ? 4 : 3) / 3);
return texture;
}
// A per-frame vertex buffer (and the transfer buffer that fills it), grown when a frame needs more,
// with this frame's data uploaded. False when the GPU refused the buffers.
static bool _uploadDynamic(SDL_GPUCommandBuffer *commands, SDL_GPUBufferUsageFlags usage, SDL_GPUBuffer **buffer, SDL_GPUTransferBuffer **transfer, uint32_t *capacity, const void *data, uint32_t bytes, const char *what) {
SDL_GPUBufferCreateInfo info;
SDL_GPUTransferBufferCreateInfo transferInfo;
SDL_GPUTransferBufferLocation source;
SDL_GPUBufferRegion region;
SDL_GPUCopyPass *pass;
void *mapped;
if (bytes == 0) {
return true;
}
if (bytes > *capacity) {
if (*buffer != NULL) {
rgpuReleaseBuffer(_scene.device, *buffer);
}
if (*transfer != NULL) {
rgpuReleaseTransferBuffer(_scene.device, *transfer);
}
*capacity = SDL_max(bytes * 2, DYNAMIC_BUFFER_MIN);
memset(&info, 0, sizeof(info));
info.usage = usage;
info.size = *capacity;
*buffer = rgpuCreateBuffer(_scene.device, &info);
memset(&transferInfo, 0, sizeof(transferInfo));
transferInfo.usage = SDL_GPU_TRANSFERBUFFERUSAGE_UPLOAD;
transferInfo.size = *capacity;
*transfer = rgpuCreateTransferBuffer(_scene.device, &transferInfo);
if ((*buffer == NULL) || (*transfer == NULL)) {
// Nothing kept, so the next frame tries again rather than mapping a buffer it has not got.
utilTrace("Scene: %s buffer: %s", what, SDL_GetError());
if (*buffer != NULL) {
rgpuReleaseBuffer(_scene.device, *buffer);
}
if (*transfer != NULL) {
rgpuReleaseTransferBuffer(_scene.device, *transfer);
}
*buffer = NULL;
*transfer = NULL;
*capacity = 0;
return false;
}
}
mapped = rgpuMapTransferBuffer(_scene.device, *transfer, true);
if (mapped == NULL) {
utilTrace("Scene: %s buffer: %s", what, SDL_GetError());
return false;
}
memcpy(mapped, data, bytes);
rgpuUnmapTransferBuffer(_scene.device, *transfer);
pass = rgpuBeginCopyPass(commands);
memset(&source, 0, sizeof(source));
memset(&region, 0, sizeof(region));
source.transfer_buffer = *transfer;
region.buffer = *buffer;
region.size = bytes;
rgpuUploadToBuffer(pass, &source, &region, true);
rgpuEndCopyPass(pass);
return true;
}
// This frame's matrices into the storage buffer the vertex shaders index by instance.
static void _uploadInstances(SDL_GPUCommandBuffer *commands, int32_t drawCount) {
if (!_uploadDynamic(commands, SDL_GPU_BUFFERUSAGE_GRAPHICS_STORAGE_READ, &_scene.instanceBuffer, &_scene.instanceTransfer, &_scene.instanceCapacity, _scene.instances, (uint32_t)drawCount * (uint32_t)sizeof(InstanceMatricesT), "instance")) {
utilDie("Unable to upload the scene's matrices.");
}
}
static void _uploadLines(SDL_GPUCommandBuffer *commands) {
if (!_uploadDynamic(commands, SDL_GPU_BUFFERUSAGE_VERTEX, &_scene.lineBuffer, &_scene.lineTransfer, &_scene.lineCapacity, _scene.lineVertices, (uint32_t)_scene.lineVertexCount * (uint32_t)sizeof(LineVertexT), "line")) {
_scene.lineVertexCount = 0;
}
}
static void _uploadParticles(SDL_GPUCommandBuffer *commands) {
if (!_uploadDynamic(commands, SDL_GPU_BUFFERUSAGE_VERTEX, &_scene.particleBuffer, &_scene.particleTransfer, &_scene.particleCapacity, _scene.particleVertices, (uint32_t)_scene.particleVertexCount * (uint32_t)sizeof(ParticleVertexT), "particle")) {
_scene.particleRunCount = 0;
}
}
// An RGBA sampler texture from any surface, with a full mipmap chain generated on the GPU (a
// backend that refuses a texture usable as a render target gets a single level instead). Colour
// textures are sRGB, so the sampler hands the shader linear light; data textures are not.
static SDL_GPUTexture *_uploadTexture(SDL_Surface *image, bool srgb) {
SDL_Surface *rgba;
SDL_GPUTextureCreateInfo info;
SDL_GPUTexture *texture;
StagingT staging;
uint32_t size;
uint32_t levels;
rgba = SDL_ConvertSurface(image, SDL_PIXELFORMAT_RGBA32);
if (rgba == NULL) {
utilTrace("Scene: %s", SDL_GetError());
return NULL;
}
size = (uint32_t)(rgba->w * rgba->h * 4);
levels = _mipLevels(rgba->w, rgba->h);
memset(&info, 0, sizeof(info));
info.type = SDL_GPU_TEXTURETYPE_2D;
info.format = srgb ? SDL_GPU_TEXTUREFORMAT_R8G8B8A8_UNORM_SRGB : SDL_GPU_TEXTUREFORMAT_R8G8B8A8_UNORM;
info.usage = SDL_GPU_TEXTUREUSAGE_SAMPLER | ((levels > 1) ? SDL_GPU_TEXTUREUSAGE_COLOR_TARGET : 0);
info.width = (Uint32)rgba->w;
info.height = (Uint32)rgba->h;
info.layer_count_or_depth = 1;
info.num_levels = levels;
info.sample_count = SDL_GPU_SAMPLECOUNT_1;
texture = rgpuCreateTexture(_scene.device, &info);
if ((texture == NULL) && (levels > 1)) {
utilTrace("Scene: no mipmaps for a %dx%d texture: %s", rgba->w, rgba->h, SDL_GetError());
levels = 1;
info.usage = SDL_GPU_TEXTUREUSAGE_SAMPLER;
info.num_levels = 1;
texture = rgpuCreateTexture(_scene.device, &info);
}
if (texture == NULL) {
utilTrace("Scene: %s", SDL_GetError());
SDL_DestroySurface(rgba);
return NULL;
}
if (!_stageBegin(&staging, size)) {
rgpuReleaseTexture(_scene.device, texture);
SDL_DestroySurface(rgba);
return NULL;
}
if (rgba->pitch == rgba->w * 4) {
memcpy(staging.mapped, rgba->pixels, size);
} else {
int32_t y;
for (y = 0; y < rgba->h; y++) {
memcpy((uint8_t *)staging.mapped + y * rgba->w * 4, (uint8_t *)rgba->pixels + y * rgba->pitch, (size_t)rgba->w * 4);
}
}
if (!_stageCopy(&staging)) {
rgpuReleaseTexture(_scene.device, texture);
SDL_DestroySurface(rgba);
return NULL;
}
_stageTexture(&staging, 0, texture, 0, 0, (uint32_t)rgba->w, (uint32_t)rgba->h);
_stageEnd(&staging, (levels > 1) ? texture : NULL);
SDL_DestroySurface(rgba);
_recordTexture(texture, (levels > 1) ? (size_t)size * 4 / 3 : (size_t)size);
return texture;
}
// A vertex with no tangent (computed on upload) and one full weight on joint 0.
static SceneVertexT _vertex(float x, float y, float z, float nx, float ny, float nz, float u, float v) {
SceneVertexT out;
memset(&out, 0, sizeof(out));
out.position[0] = x;
out.position[1] = y;
out.position[2] = z;
out.normal[0] = nx;
out.normal[1] = ny;
out.normal[2] = nz;
out.uv[0] = u;
out.uv[1] = v;
out.weights[0] = 1.0f;
return out;
}
// The view matrix: the inverse of the camera node's world matrix, or the default view (looking at
// the origin from DEFAULT_EYE_Z) when there is no camera.
static Mat4T _viewOf(int32_t camera) {
Mat4T view;
if (!nodeValid(camera)) {
return mat4LookAt(vec3(0.0f, 0.0f, DEFAULT_EYE_Z), vec3(0.0f, 0.0f, 0.0f), vec3(0.0f, 1.0f, 0.0f));
}
if (!mat4Invert(_scene.nodes[camera].world, &view)) {
return mat4Identity();
}
return view;
}
// ===== Camera =====
// Any node can be the camera: it looks down its own -Z. NO_HANDLE restores the default view.
bool cameraSet(int32_t node) {
if ((node != NO_HANDLE) && !nodeValid(node)) {
return false;
}
_scene.cameraNode = node;
return true;
}
void cameraSetOrthographic(float height, float near, float far) {
_scene.perspective = false;
_scene.orthoHeight = height;
_scene.near = near;
_scene.far = far;
}
void cameraSetPerspective(float fovDegrees, float near, float far) {
_scene.perspective = true;
_scene.fov = fovDegrees;
_scene.near = near;
_scene.far = far;
}
// ===== Lights =====
// Makes an existing node a light. Directional and spot lights shine down the node's -Z.
bool lightAttach(int32_t node, LightTypeE type) {
if (!nodeValid(node)) {
return false;
}
_scene.nodes[node].hasLight = true;
_scene.nodes[node].light.type = type;
_scene.nodes[node].light.color = vec3(1.0f, 1.0f, 1.0f);
_scene.nodes[node].light.intensity = 1.0f;
_scene.nodes[node].light.range = 0.0f;
_scene.nodes[node].light.innerDegrees = DEFAULT_CONE_INNER;
_scene.nodes[node].light.outerDegrees = DEFAULT_CONE_OUTER;
return true;
}
// A new node carrying a light.
int32_t lightNew(LightTypeE type, int32_t parent) {
int32_t node = nodeNew(parent);
if (node != NO_HANDLE) {
lightAttach(node, type);
}
return node;
}
bool lightSetColor(int32_t node, uint8_t r, uint8_t g, uint8_t b) {
if (!nodeValid(node) || !_scene.nodes[node].hasLight) {
return false;
}
_scene.nodes[node].light.color = vec3(_linear(r), _linear(g), _linear(b));
return true;
}
// Linear components, already in the space the shader wants (glTF light colours are linear).
bool lightSetColorLinear(int32_t node, float r, float g, float b) {
if (!nodeValid(node) || !_scene.nodes[node].hasLight) {
return false;
}
_scene.nodes[node].light.color = vec3(r, g, b);
return true;
}
// A spot light's full brightness inside the inner angle, fading to nothing at the outer, which is
// never narrower than the inner (the shader's edge runs from one to the other).
bool lightSetCone(int32_t node, float innerDegrees, float outerDegrees) {
if (!nodeValid(node) || !_scene.nodes[node].hasLight) {
return false;
}
_scene.nodes[node].light.innerDegrees = innerDegrees;
_scene.nodes[node].light.outerDegrees = SDL_max(outerDegrees, innerDegrees);
return true;
}
bool lightSetIntensity(int32_t node, float intensity) {
if (!nodeValid(node) || !_scene.nodes[node].hasLight) {
return false;
}
_scene.nodes[node].light.intensity = intensity;
return true;
}
// 0 means no range limit.
bool lightSetRange(int32_t node, float range) {
if (!nodeValid(node) || !_scene.nodes[node].hasLight) {
return false;
}
_scene.nodes[node].light.range = range;
return true;
}
// Whether this light casts shadows. Every light may; a point light's cost six passes to a
// directional or spot light's one.
bool lightSetShadow(int32_t node, bool shadow) {
if (!nodeValid(node) || !_scene.nodes[node].hasLight) {
return false;
}
_scene.nodes[node].castsShadow = shadow;
return true;
}
// ===== Materials =====
// A sprite node's private material (its base texture is the sprite's) is the node's to delete.
bool materialDelete(int32_t material) {
int32_t x;
if (!materialValid(material) || _scene.materials[material].textureBorrowed) {
return false;
}
_freeMaterialTexture(&_scene.materials[material]);
_scene.materials[material].used = false;
// Nodes that used it fall back to the default look.
for (x = 0; x < _scene.nodeCount; x++) {
if (_scene.nodes[x].used && (_scene.nodes[x].material == material)) {
_scene.nodes[x].material = NO_HANDLE;
}
}
return true;
}
void materialForgetGui(int32_t gui) {
int32_t x;
for (x = 0; x < _scene.materialCount; x++) {
if (_scene.materials[x].used && (_scene.materials[x].gui == gui)) {
_scene.materials[x].gui = NO_HANDLE;
}
}
}
// White, half rough, no texture.
int32_t materialNew(void) {
return _allocMaterial();
}
bool materialSetBlend(int32_t material, bool blend) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].blend = blend;
return true;
}
bool materialSetColor(int32_t material, uint8_t r, uint8_t g, uint8_t b, uint8_t a) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].baseColor.x = _linear(r);
_scene.materials[material].baseColor.y = _linear(g);
_scene.materials[material].baseColor.z = _linear(b);
_scene.materials[material].baseColor.w = a / COLOUR_MAX;
return true;
}
// For factors that are already linear (glTF's).
bool materialSetColorLinear(int32_t material, float r, float g, float b, float a) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].baseColor.x = r;
_scene.materials[material].baseColor.y = g;
_scene.materials[material].baseColor.z = b;
_scene.materials[material].baseColor.w = a;
return true;
}
// glTF's alpha masking: a texel whose base colour alpha falls below the cutoff is discarded, in
// the lit pass and in the shadow pass alike. Zero turns masking off, which is the default and what
// every material that never asked for it keeps.
bool materialSetCutoff(int32_t material, float cutoff) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].cutoff = SDL_clamp(cutoff, 0.0f, 1.0f);
return true;
}
bool materialSetDoubleSided(int32_t material, bool doubleSided) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].doubleSided = doubleSided;
return true;
}
bool materialSetEmissive(int32_t material, uint8_t r, uint8_t g, uint8_t b) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].emissive = vec3(_linear(r), _linear(g), _linear(b));
return true;
}
bool materialSetEmissiveLinear(int32_t material, float r, float g, float b) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].emissive = vec3(r, g, b);
return true;
}
bool materialSetFilter(int32_t material, MaterialFilterE filter) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].filter = filter;
return true;
}
// A GUI's texture as the base colour texture; NO_HANDLE goes back to the material's own texture.
bool materialSetGui(int32_t material, int32_t gui) {
if (!materialValid(material) || ((gui != NO_HANDLE) && !guiValid(gui))) {
return false;
}
_scene.materials[material].gui = gui;
if (gui != NO_HANDLE) {
_scene.materials[material].feed = NO_HANDLE;
_scene.materials[material].view = NO_HANDLE;
}
return true;
}
// One of a material's textures from a transcoded KTX2 image (colour maps sRGB, data maps not);
// NULL clears it. Strength is the normal map's bump scale or the occlusion map's blend.
bool materialSetMap(int32_t material, MaterialMapE map, const Ktx2ImageT *image, float strength) {
SDL_GPUTexture *texture = NULL;
if (!materialValid(material)) {
return false;
}
if (image != NULL) {
texture = _uploadCompressed(image, _mapIsColour(map));
if (texture == NULL) {
return false;
}
}
_materialPlace(&_scene.materials[material], map, texture, strength);
return true;
}
// The same from a surface, copied into a mipmapped texture. The base map is what an untextured
// material shows plain; a normal map is tangent space (flat is 128, 128, 255); occlusion sits in R
// and metallic-roughness in B and G (glTF's packing); the emissive map multiplies the emissive colour.
bool materialSetMapSurface(int32_t material, MaterialMapE map, SDL_Surface *image, float strength) {
SDL_GPUTexture *texture = NULL;
if (!materialValid(material)) {
return false;
}
if (image != NULL) {
texture = _uploadTexture(image, _mapIsColour(map));
if (texture == NULL) {
return false;
}
}
_materialPlace(&_scene.materials[material], map, texture, strength);
return true;
}
bool materialSetMetallic(int32_t material, float metallic) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].metallic = SDL_clamp(metallic, 0.0f, 1.0f);
return true;
}
bool materialSetOccluder(int32_t material, bool occluder) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].occluder = occluder;
return true;
}
bool materialSetRoughness(int32_t material, float roughness) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].roughness = SDL_clamp(roughness, 0.0f, 1.0f);
return true;
}
// How many times the material's textures repeat across a surface's 0 to 1 UV range.
bool materialSetTiling(int32_t material, float u, float v) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].tilingU = u;
_scene.materials[material].tilingV = v;
return true;
}
bool materialSetUnlit(int32_t material, bool unlit) {
if (!materialValid(material)) {
return false;
}
_scene.materials[material].unlit = unlit;
return true;
}
// A video player's frames as the base colour texture; replaces any image. The frames arrive
// through sceneUpdateVideo each frame.
bool materialSetVideo(int32_t material, int32_t player) {
if (!materialValid(material)) {
return false;
}
_releaseMaterialBase(&_scene.materials[material]);
_scene.materials[material].feed = _allocFeed(player);
_scene.materials[material].view = NO_HANDLE;
_scene.materials[material].gui = NO_HANDLE;
return true;
}
// A rendered view as the base colour texture; NO_HANDLE goes back to the material's own texture.
bool materialSetView(int32_t material, int32_t view) {
if (!materialValid(material) || ((view != NO_HANDLE) && !viewValid(view))) {
return false;
}
_scene.materials[material].view = view;
if (view != NO_HANDLE) {
_scene.materials[material].feed = NO_HANDLE;
_scene.materials[material].gui = NO_HANDLE;
}
return true;
}
bool materialValid(int32_t material) {
return (material >= 0) && (material < _scene.materialCount) && _scene.materials[material].used;
}
// ===== Meshes =====
// Six faces with their own vertices so each has a flat normal. Centred on the origin.
int32_t meshBox(float width, float height, float depth) {
SceneVertexT vertices[24];
uint32_t indices[36];
float w = width / 2.0f;
float h = height / 2.0f;
float d = depth / 2.0f;
int32_t face;
int32_t v = 0;
int32_t i = 0;
// Per face: normal, then the four corners counter-clockwise seen from outside.
float faces[6][5][3] = {
{ { 0.0f, 0.0f, 1.0f }, { -w, -h, d }, { w, -h, d }, { w, h, d }, { -w, h, d } }, // Front (+Z)
{ { 0.0f, 0.0f, -1.0f }, { w, -h, -d }, { -w, -h, -d }, { -w, h, -d }, { w, h, -d } }, // Back (-Z)
{ { 1.0f, 0.0f, 0.0f }, { w, -h, d }, { w, -h, -d }, { w, h, -d }, { w, h, d } }, // Right (+X)
{ { -1.0f, 0.0f, 0.0f }, { -w, -h, -d }, { -w, -h, d }, { -w, h, d }, { -w, h, -d } }, // Left (-X)
{ { 0.0f, 1.0f, 0.0f }, { -w, h, d }, { w, h, d }, { w, h, -d }, { -w, h, -d } }, // Top (+Y)
{ { 0.0f, -1.0f, 0.0f }, { -w, -h, -d }, { w, -h, -d }, { w, -h, d }, { -w, -h, d } }, // Bottom (-Y)
};
float uvs[4][2] = { { 0.0f, 1.0f }, { 1.0f, 1.0f }, { 1.0f, 0.0f }, { 0.0f, 0.0f } };
for (face = 0; face < 6; face++) {
int32_t corner;
for (corner = 0; corner < 4; corner++) {
vertices[v++] = _vertex(faces[face][corner + 1][0], faces[face][corner + 1][1], faces[face][corner + 1][2], faces[face][0][0], faces[face][0][1], faces[face][0][2], uvs[corner][0], uvs[corner][1]);
}
indices[i++] = (uint32_t)(face * 4);
indices[i++] = (uint32_t)(face * 4 + 1);
indices[i++] = (uint32_t)(face * 4 + 2);
indices[i++] = (uint32_t)(face * 4);
indices[i++] = (uint32_t)(face * 4 + 2);
indices[i++] = (uint32_t)(face * 4 + 3);
}
return _addMesh(vertices, 24, indices, 36, false);
}
int32_t meshCone(float radius, float height, int32_t segments) {
SceneVertexT *vertices;
uint32_t *indices;
int32_t vertexCount;
int32_t indexCount;
int32_t mesh;
_lathe(&vertices, &vertexCount, &indices, &indexCount, radius, 0.0f, height, SDL_max(segments, MIN_SEGMENTS));
mesh = _addMesh(vertices, vertexCount, indices, indexCount, false);
SDL_free(vertices);
SDL_free(indices);
return mesh;
}
int32_t meshCylinder(float radius, float height, int32_t segments) {
SceneVertexT *vertices;
uint32_t *indices;
int32_t vertexCount;
int32_t indexCount;
int32_t mesh;
_lathe(&vertices, &vertexCount, &indices, &indexCount, radius, radius, height, SDL_max(segments, MIN_SEGMENTS));
mesh = _addMesh(vertices, vertexCount, indices, indexCount, false);
SDL_free(vertices);
SDL_free(indices);
return mesh;
}
bool meshDelete(int32_t mesh) {
int32_t x;
if (!meshValid(mesh)) {
return false;
}
if (_scene.meshes[mesh].vertexBuffer != NULL) {
rgpuReleaseBuffer(_scene.device, _scene.meshes[mesh].vertexBuffer);
}
if (_scene.meshes[mesh].indexBuffer != NULL) {
rgpuReleaseBuffer(_scene.device, _scene.meshes[mesh].indexBuffer);
}
SDL_free(_scene.meshes[mesh].positions);
SDL_free(_scene.meshes[mesh].heights);
SDL_free(_scene.meshes[mesh].indices);
SDL_free(_scene.meshes[mesh].vertices);
if (_scene.meshes[mesh].transfer != NULL) {
rgpuReleaseTransferBuffer(_scene.device, _scene.meshes[mesh].transfer);
}
_freeMorphs(&_scene.meshes[mesh]);
memset(&_scene.meshes[mesh], 0, sizeof(MeshT));
for (x = 0; x < _scene.nodeCount; x++) {
if (_scene.nodes[x].used && (_scene.nodes[x].mesh == mesh)) {
_scene.nodes[x].mesh = NO_HANDLE;
_freeMorphWeights(&_scene.nodes[x]);
}
}
return true;
}
// A morph target by name, or -1.
int32_t meshFindMorph(int32_t mesh, const char *name) {
int32_t x;
if (!meshValid(mesh) || (name == NULL)) {
return NO_HANDLE;
}
for (x = 0; x < _scene.meshes[mesh].morphCount; x++) {
if ((_scene.meshes[mesh].morphNames[x] != NULL) && (strcmp(_scene.meshes[mesh].morphNames[x], name) == 0)) {
return x;
}
}
return NO_HANDLE;
}
// The mesh's geometry as kept on the CPU: x, y, z per vertex and triangle indices.
bool meshGetGeometry(int32_t mesh, const float **positions, int32_t *vertexCount, const uint32_t **indices, int32_t *indexCount) {
if (!meshValid(mesh)) {
return false;
}
*positions = _scene.meshes[mesh].positions;
*vertexCount = _scene.meshes[mesh].vertexCount;
*indices = _scene.meshes[mesh].indices;
*indexCount = (int32_t)_scene.meshes[mesh].indexCount;
return true;
}
// A heightmap mesh's samples and size; false for any other mesh.
bool meshGetHeights(int32_t mesh, const float **heights, int32_t *columns, int32_t *rows, float *sizeX, float *sizeY, float *sizeZ) {
if (!meshValid(mesh) || (_scene.meshes[mesh].heights == NULL)) {
return false;
}
*heights = _scene.meshes[mesh].heights;
*columns = _scene.meshes[mesh].heightColumns;
*rows = _scene.meshes[mesh].heightRows;
*sizeX = _scene.meshes[mesh].sizeX;
*sizeY = _scene.meshes[mesh].sizeY;
*sizeZ = _scene.meshes[mesh].sizeZ;
return true;
}
int32_t meshGetMorphCount(int32_t mesh) {
if (!meshValid(mesh)) {
return 0;
}
return _scene.meshes[mesh].morphCount;
}
const char *meshGetMorphName(int32_t mesh, int32_t target) {
if (!meshValid(mesh) || (target < 0) || (target >= _scene.meshes[mesh].morphCount) || (_scene.meshes[mesh].morphNames[target] == NULL)) {
return "";
}
return _scene.meshes[mesh].morphNames[target];
}
// A plane of columns x rows quads, for cloth and terrain that bends; the first row is the near
// (+Z) edge.
int32_t meshGrid(float width, float depth, int32_t columns, int32_t rows) {
return _gridMesh(NULL, columns, rows, width, 0.0f, depth, false);
}
// A grid of columns x rows cells across sizeX by sizeZ, each vertex lifted by its sample (0 to 1)
// times sizeY, centred on the origin with the first row of samples at the far (-Z) edge and UVs
// 0 to 1 across the whole (materialSetTiling repeats a texture over it). Normals come from the
// slopes. The samples are kept for the height field body and terrainGetHeight.
int32_t meshHeightmap(const float *heights, int32_t columns, int32_t rows, float sizeX, float sizeY, float sizeZ) {
int32_t vertexCount = (columns + 1) * (rows + 1);
int32_t mesh;
MeshT *m;
if (heights == NULL) {
return NO_HANDLE;
}
mesh = _gridMesh(heights, columns, rows, sizeX, sizeY, sizeZ, true);
if (mesh == NO_HANDLE) {
return NO_HANDLE;
}
m = &_scene.meshes[mesh];
m->heights = SDL_malloc(sizeof(float) * (size_t)vertexCount);
if (m->heights == NULL) {
utilDie("Out of memory keeping a heightmap.");
}
memcpy(m->heights, heights, sizeof(float) * (size_t)vertexCount);
m->heightColumns = columns;
m->heightRows = rows;
m->sizeX = sizeX;
m->sizeY = sizeY;
m->sizeZ = sizeZ;
return mesh;
}
// Raw geometry from a script: positions (3 per vertex), normals (3, may be NULL for flat
// shading computed here), uvs (2, may be NULL), and triangle indices.
int32_t meshNew(const float *positions, const float *normals, const float *uvs, int32_t vertexCount, const uint32_t *indices, int32_t indexCount) {
SceneVertexT *vertices;
int32_t x;
int32_t mesh;
if ((positions == NULL) || (indices == NULL) || (vertexCount <= 0) || (indexCount < 3)) {
return NO_HANDLE;
}
for (x = 0; x < indexCount; x++) {
if (indices[x] >= (uint32_t)vertexCount) {
return NO_HANDLE;
}
}
vertices = SDL_calloc((size_t)vertexCount, sizeof(SceneVertexT));
if (vertices == NULL) {
utilDie("Out of memory building a mesh.");
}
for (x = 0; x < vertexCount; x++) {
vertices[x] = _vertex(positions[x * 3], positions[x * 3 + 1], positions[x * 3 + 2], 0.0f, 0.0f, 0.0f, uvs ? uvs[x * 2] : 0.0f, uvs ? uvs[x * 2 + 1] : 0.0f);
if (normals != NULL) {
vertices[x].normal[0] = normals[x * 3];
vertices[x].normal[1] = normals[x * 3 + 1];
vertices[x].normal[2] = normals[x * 3 + 2];
}
}
if (normals == NULL) {
sceneComputeNormals(vertices, vertexCount, indices, indexCount);
}
mesh = _addMesh(vertices, vertexCount, indices, indexCount, false);
SDL_free(vertices);
return mesh;
}
// Geometry a loader has already laid out in GPU form.
int32_t meshNewVertices(const SceneVertexT *vertices, int32_t vertexCount, const uint32_t *indices, int32_t indexCount, bool skinned) {
int32_t x;
if ((vertices == NULL) || (indices == NULL) || (vertexCount <= 0) || (indexCount < 3)) {
return NO_HANDLE;
}
for (x = 0; x < indexCount; x++) {
if (indices[x] >= (uint32_t)vertexCount) {
return NO_HANDLE;
}
}
return _addMesh(vertices, vertexCount, indices, indexCount, skinned);
}
// A quad in the XZ plane facing +Y.
int32_t meshPlane(float width, float depth) {
return _quadMesh(width, depth, vec3(0.0f, 0.0f, 1.0f), vec3(0.0f, 1.0f, 0.0f));
}
// Gives the mesh morph targets: deltas holds, per target, per vertex, a position delta (x, y, z)
// and a normal delta (x, y, z), six floats; names may be NULL or hold NULL entries. Nodes using
// the mesh get a weight per target, all 0.
bool meshSetMorphTargets(int32_t mesh, const float *deltas, int32_t targetCount, const char **names) {
MeshT *m;
float *packed;
size_t count;
size_t v;
int32_t x;
if (!meshValid(mesh) || (deltas == NULL) || (targetCount <= 0)) {
return false;
}
m = &_scene.meshes[mesh];
count = (size_t)targetCount * (size_t)m->vertexCount;
// The GPU buffer is sized in 32 bits.
if (count > UINT32_MAX / (MORPH_FLOATS * sizeof(float))) {
return false;
}
packed = SDL_calloc(count * MORPH_FLOATS, sizeof(float));
if (packed == NULL) {
utilDie("Out of memory packing morph targets.");
}
// float4 pairs for the shader: xyz0 position delta, xyz0 normal delta.
for (v = 0; v < count; v++) {
packed[v * MORPH_FLOATS] = deltas[v * MORPH_INPUT_FLOATS];
packed[v * MORPH_FLOATS + 1] = deltas[v * MORPH_INPUT_FLOATS + 1];
packed[v * MORPH_FLOATS + 2] = deltas[v * MORPH_INPUT_FLOATS + 2];
packed[v * MORPH_FLOATS + 4] = deltas[v * MORPH_INPUT_FLOATS + 3];
packed[v * MORPH_FLOATS + 5] = deltas[v * MORPH_INPUT_FLOATS + 4];
packed[v * MORPH_FLOATS + 6] = deltas[v * MORPH_INPUT_FLOATS + 5];
}
_freeMorphs(m);
m->morphBuffer = _uploadBuffer(SDL_GPU_BUFFERUSAGE_GRAPHICS_STORAGE_READ, packed, (uint32_t)(count * MORPH_FLOATS * sizeof(float)));
SDL_free(packed);
if (m->morphBuffer == NULL) {
return false;
}
m->morphNames = SDL_calloc((size_t)targetCount, sizeof(char *));
if (m->morphNames == NULL) {
utilDie("Out of memory naming morph targets.");
}
for (x = 0; x < targetCount; x++) {
m->morphNames[x] = ((names != NULL) && (names[x] != NULL)) ? SDL_strdup(names[x]) : NULL;
}
m->morphCount = targetCount;
_stampMesh(m);
for (x = 0; x < _scene.nodeCount; x++) {
if (_scene.nodes[x].used && (_scene.nodes[x].mesh == mesh)) {
_matchMorphWeights(&_scene.nodes[x]);
}
}
return true;
}
// Rewrites a mesh's vertex positions (x, y, z per vertex, in the mesh's space), recomputing normals
// and bounds and uploading in place, for meshes a soft body drives.
bool meshSetPositions(int32_t mesh, const float *positions) {
MeshT *m;
SDL_GPUTransferBufferCreateInfo transferInfo;
SDL_GPUTransferBufferLocation source;
SDL_GPUBufferRegion region;
SDL_GPUCommandBuffer *commands;
SDL_GPUCopyPass *pass;
void *mapped;
uint32_t size;
int32_t x;
if (!meshValid(mesh) || (positions == NULL)) {
return false;
}
m = &_scene.meshes[mesh];
size = (uint32_t)(sizeof(SceneVertexT) * (size_t)m->vertexCount);
memcpy(m->positions, positions, sizeof(float) * 3 * (size_t)m->vertexCount);
for (x = 0; x < m->vertexCount; x++) {
m->vertices[x].position[0] = positions[x * 3];
m->vertices[x].position[1] = positions[x * 3 + 1];
m->vertices[x].position[2] = positions[x * 3 + 2];
if (x == 0) {
m->boundsMin = vec3(positions[0], positions[1], positions[2]);
m->boundsMax = m->boundsMin;
} else {
m->boundsMin = vec3(SDL_min(m->boundsMin.x, positions[x * 3]), SDL_min(m->boundsMin.y, positions[x * 3 + 1]), SDL_min(m->boundsMin.z, positions[x * 3 + 2]));
m->boundsMax = vec3(SDL_max(m->boundsMax.x, positions[x * 3]), SDL_max(m->boundsMax.y, positions[x * 3 + 1]), SDL_max(m->boundsMax.z, positions[x * 3 + 2]));
}
}
sceneComputeNormals(m->vertices, m->vertexCount, m->indices, (int32_t)m->indexCount);
_stampMesh(m);
if (_scene.device == NULL) {
return true;
}
if (m->transfer == NULL) {
memset(&transferInfo, 0, sizeof(transferInfo));
transferInfo.usage = SDL_GPU_TRANSFERBUFFERUSAGE_UPLOAD;
transferInfo.size = size;
m->transfer = rgpuCreateTransferBuffer(_scene.device, &transferInfo);
if (m->transfer == NULL) {
utilTrace("Scene: %s", SDL_GetError());
return false;
}
}
mapped = rgpuMapTransferBuffer(_scene.device, m->transfer, true);
if (mapped == NULL) {
utilTrace("Scene: %s", SDL_GetError());
return false;
}
memcpy(mapped, m->vertices, size);
rgpuUnmapTransferBuffer(_scene.device, m->transfer);
commands = rgpuAcquireCommandBuffer(_scene.device);
if (commands == NULL) {
utilTrace("Scene: %s", SDL_GetError());
return false;
}
pass = rgpuBeginCopyPass(commands);
memset(&source, 0, sizeof(source));
memset(&region, 0, sizeof(region));
source.transfer_buffer = m->transfer;
region.buffer = m->vertexBuffer;
region.size = size;
rgpuUploadToBuffer(pass, &source, &region, true);
rgpuEndCopyPass(pass);
rgpuSubmitCommandBuffer(commands);
return true;
}
// Latitude/longitude sphere; segments around, half as many from pole to pole.
int32_t meshSphere(float radius, int32_t segments) {
SceneVertexT *vertices;
uint32_t *indices;
int32_t rings;
int32_t ring;
int32_t seg;
int32_t v = 0;
int32_t i = 0;
int32_t vertexCount;
int32_t indexCount;
int32_t mesh;
segments = SDL_max(segments, MIN_SEGMENTS);
rings = SDL_max(segments / 2, 2);
vertexCount = (rings + 1) * (segments + 1);
indexCount = rings * segments * 6;
vertices = SDL_calloc((size_t)vertexCount, sizeof(SceneVertexT));
indices = SDL_calloc((size_t)indexCount, sizeof(uint32_t));
if ((vertices == NULL) || (indices == NULL)) {
utilDie("Out of memory building a mesh.");
}
for (ring = 0; ring <= rings; ring++) {
float phi = (float)ring / (float)rings * PI;
for (seg = 0; seg <= segments; seg++) {
float theta = (float)seg / (float)segments * 2.0f * PI;
float nx = sinf(phi) * cosf(theta);
float ny = cosf(phi);
float nz = sinf(phi) * sinf(theta);
// u runs the other way so an image reads correctly from outside.
vertices[v++] = _vertex(nx * radius, ny * radius, nz * radius, nx, ny, nz, 1.0f - (float)seg / (float)segments, (float)ring / (float)rings);
}
}
for (ring = 0; ring < rings; ring++) {
for (seg = 0; seg < segments; seg++) {
uint32_t a = (uint32_t)(ring * (segments + 1) + seg);
uint32_t b = a + (uint32_t)segments + 1;
indices[i++] = a;
indices[i++] = a + 1;
indices[i++] = b;
indices[i++] = a + 1;
indices[i++] = b + 1;
indices[i++] = b;
}
}
mesh = _addMesh(vertices, vertexCount, indices, indexCount, false);
SDL_free(vertices);
SDL_free(indices);
return mesh;
}
// A ring around Y.
int32_t meshTorus(float radius, float tubeRadius, int32_t segments) {
SceneVertexT *vertices;
uint32_t *indices;
int32_t tubeSegments;
int32_t x;
int32_t y;
int32_t v = 0;
int32_t i = 0;
int32_t vertexCount;
int32_t indexCount;
int32_t mesh;
segments = SDL_max(segments, MIN_SEGMENTS);
tubeSegments = SDL_max(segments / 2, MIN_SEGMENTS);
vertexCount = (segments + 1) * (tubeSegments + 1);
indexCount = segments * tubeSegments * 6;
vertices = SDL_calloc((size_t)vertexCount, sizeof(SceneVertexT));
indices = SDL_calloc((size_t)indexCount, sizeof(uint32_t));
if ((vertices == NULL) || (indices == NULL)) {
utilDie("Out of memory building a mesh.");
}
for (x = 0; x <= segments; x++) {
float theta = (float)x / (float)segments * 2.0f * PI;
float cx = cosf(theta);
float sx = sinf(theta);
for (y = 0; y <= tubeSegments; y++) {
float phi = (float)y / (float)tubeSegments * 2.0f * PI;
float cy = cosf(phi);
float sy = sinf(phi);
vertices[v++] = _vertex(cx * (radius + cy * tubeRadius), sy * tubeRadius, sx * (radius + cy * tubeRadius), cx * cy, sy, sx * cy, (float)x / (float)segments, (float)y / (float)tubeSegments);
}
}
for (x = 0; x < segments; x++) {
for (y = 0; y < tubeSegments; y++) {
uint32_t a = (uint32_t)(x * (tubeSegments + 1) + y);
uint32_t b = a + (uint32_t)tubeSegments + 1;
indices[i++] = a;
indices[i++] = b;
indices[i++] = a + 1;
indices[i++] = a + 1;
indices[i++] = b;
indices[i++] = b + 1;
}
}
mesh = _addMesh(vertices, vertexCount, indices, indexCount, false);
SDL_free(vertices);
SDL_free(indices);
return mesh;
}
bool meshValid(int32_t mesh) {
return (mesh >= 0) && (mesh < _scene.meshCount) && _scene.meshes[mesh].used;
}
// ===== Nodes =====
// Frees the node and everything under it. The root cannot be deleted.
bool nodeDelete(int32_t node) {
int32_t child;
int32_t next;
if (!nodeValid(node) || (node == SCENE_ROOT_NODE)) {
return false;
}
for (child = _scene.nodes[node].firstChild; child != NO_HANDLE; child = next) {
next = _scene.nodes[child].nextSibling;
nodeDelete(child);
}
_detach(node);
_freeSpriteNode(node);
if (_scene.cameraNode == node) {
_scene.cameraNode = NO_HANDLE;
}
SDL_free(_scene.nodes[node].name);
_freeSkin(&_scene.nodes[node]);
_freeMorphWeights(&_scene.nodes[node]);
_scene.nodes[node].name = NULL;
_scene.nodes[node].used = false;
return true;
}
// Depth-first search below root (root itself included) for a node by name.
int32_t nodeFind(int32_t root, const char *name) {
int32_t child;
int32_t found;
if (!nodeValid(root) || (name == NULL)) {
return NO_HANDLE;
}
if ((_scene.nodes[root].name != NULL) && (strcmp(_scene.nodes[root].name, name) == 0)) {
return root;
}
for (child = _scene.nodes[root].firstChild; child != NO_HANDLE; child = _scene.nodes[child].nextSibling) {
found = nodeFind(child, name);
if (found != NO_HANDLE) {
return found;
}
}
return NO_HANDLE;
}
int32_t nodeGetChild(int32_t node, int32_t index) {
int32_t child;
if (!nodeValid(node) || (index < 0)) {
return NO_HANDLE;
}
child = _scene.nodes[node].firstChild;
while ((child != NO_HANDLE) && (index > 0)) {
child = _scene.nodes[child].nextSibling;
index--;
}
return child;
}
int32_t nodeGetChildCount(int32_t node) {
int32_t child;
int32_t count = 0;
if (!nodeValid(node)) {
return 0;
}
for (child = _scene.nodes[node].firstChild; child != NO_HANDLE; child = _scene.nodes[child].nextSibling) {
count++;
}
return count;
}
// Changes every time the slot is reused, so a handle kept across a delete can be detected.
uint32_t nodeGetGeneration(int32_t node) {
if ((node < 0) || (node >= _scene.nodeCount)) {
return 0;
}
return _scene.nodes[node].generation;
}
// The node's material handle, or -1.
int32_t nodeGetMaterial(int32_t node) {
return nodeValid(node) ? _scene.nodes[node].material : NO_HANDLE;
}
// The node's mesh handle, or -1.
int32_t nodeGetMesh(int32_t node) {
if (!nodeValid(node)) {
return NO_HANDLE;
}
return _scene.nodes[node].mesh;
}
int32_t nodeGetMorphCount(int32_t node) {
if (!nodeValid(node)) {
return 0;
}
return _scene.nodes[node].morphCount;
}
float nodeGetMorphWeight(int32_t node, int32_t target) {
if (!nodeValid(node) || (target < 0) || (target >= _scene.nodes[node].morphCount)) {
return 0.0f;
}
return _scene.nodes[node].morphWeights[target];
}
const char *nodeGetName(int32_t node) {
if (!nodeValid(node) || (_scene.nodes[node].name == NULL)) {
return "";
}
return _scene.nodes[node].name;
}
int32_t nodeGetParent(int32_t node) {
if (!nodeValid(node)) {
return NO_HANDLE;
}
return _scene.nodes[node].parent;
}
Vec3T nodeGetPosition(int32_t node) {
if (!nodeValid(node)) {
return vec3(0.0f, 0.0f, 0.0f);
}
return _scene.nodes[node].translation;
}
QuatT nodeGetRotation(int32_t node) {
if (!nodeValid(node)) {
return quatIdentity();
}
return _scene.nodes[node].rotation;
}
Vec3T nodeGetScale(int32_t node) {
if (!nodeValid(node)) {
return vec3(1.0f, 1.0f, 1.0f);
}
return _scene.nodes[node].scale;
}
// The joint nodes of a skinned node's skin, or 0 with none.
int32_t nodeGetSkinJoints(int32_t node, const int32_t **joints) {
if (!nodeValid(node) || (_scene.nodes[node].skinCount == 0)) {
return 0;
}
if (joints != NULL) {
*joints = _scene.nodes[node].skinJoints;
}
return _scene.nodes[node].skinCount;
}
// From the last rendered frame's matrices (this frame's edits show after the next render).
Vec3T nodeGetWorldPosition(int32_t node) {
if (!nodeValid(node)) {
return vec3(0.0f, 0.0f, 0.0f);
}
return mat4TransformPoint(_scene.nodes[node].world, vec3(0.0f, 0.0f, 0.0f));
}
// The node's world-space position, rotation and scale as of the last transform update.
bool nodeGetWorldTransform(int32_t node, Vec3T *position, QuatT *rotation, Vec3T *scale) {
if (!nodeValid(node)) {
return false;
}
mat4Decompose(_scene.nodes[node].world, position, rotation, scale);
return true;
}
// Points the node's -Z at a world-space target (the camera and lights look down -Z). Only the
// node's own rotation changes, in its parent's space.
bool nodeLookAt(int32_t node, Vec3T target) {
NodeT *n;
Mat4T parentInverse;
Vec3T localTarget;
Vec3T forward;
if (!nodeValid(node)) {
return false;
}
n = &_scene.nodes[node];
if ((n->parent != NO_HANDLE) && mat4Invert(_scene.nodes[n->parent].world, &parentInverse)) {
localTarget = mat4TransformPoint(parentInverse, target);
} else {
localTarget = target;
}
forward = vec3Subtract(localTarget, n->translation);
if (vec3Length(forward) < MATH_EPSILON) {
return true;
}
n->rotation = quatLookRotation(forward, vec3(0.0f, 1.0f, 0.0f));
return true;
}
// Moves along the node's own axes.
bool nodeMove(int32_t node, Vec3T delta) {
if (!nodeValid(node)) {
return false;
}
_scene.nodes[node].translation = vec3Add(_scene.nodes[node].translation, quatRotate(_scene.nodes[node].rotation, delta));
return true;
}
int32_t nodeNew(int32_t parent) {
int32_t node;
if (parent == NO_HANDLE) {
parent = SCENE_ROOT_NODE;
}
if (!nodeValid(parent)) {
return NO_HANDLE;
}
node = _allocNode();
_attach(node, parent);
return node;
}
// Rotates about the node's own axes.
bool nodeRotate(int32_t node, QuatT delta) {
if (!nodeValid(node)) {
return false;
}
_scene.nodes[node].rotation = quatNormalize(quatMultiply(_scene.nodes[node].rotation, delta));
return true;
}
bool nodeSetBillboard(int32_t node, BillboardE mode) {
if (!nodeValid(node)) {
return false;
}
_scene.nodes[node].billboard = mode;
return true;
}
// Changes the material and keeps the mesh.
bool nodeSetMaterial(int32_t node, int32_t material) {
if (!nodeValid(node)) {
return false;
}
if ((material != NO_HANDLE) && !materialValid(material)) {
return false;
}
_scene.nodes[node].material = material;
return true;
}
// mesh NO_HANDLE clears; material NO_HANDLE means the default look.
bool nodeSetMesh(int32_t node, int32_t mesh, int32_t material) {
if (!nodeValid(node)) {
return false;
}
if ((mesh != NO_HANDLE) && !meshValid(mesh)) {
return false;
}
if ((material != NO_HANDLE) && !materialValid(material)) {
return false;
}
_scene.nodes[node].mesh = mesh;
_scene.nodes[node].material = material;
_matchMorphWeights(&_scene.nodes[node]);
return true;
}
// How much of a morph target the node's mesh shows (usually 0 to 1).
bool nodeSetMorphWeight(int32_t node, int32_t target, float weight) {
if (!nodeValid(node) || (target < 0) || (target >= _scene.nodes[node].morphCount)) {
return false;
}
_scene.nodes[node].morphWeights[target] = weight;
return true;
}
bool nodeSetName(int32_t node, const char *name) {
if (!nodeValid(node)) {
return false;
}
SDL_free(_scene.nodes[node].name);
_scene.nodes[node].name = (name != NULL) ? SDL_strdup(name) : NULL;
return true;
}
// Re-parents, keeping the node's local transform (so it moves with the new parent). A node
// cannot be put under itself or its own descendants.
bool nodeSetParent(int32_t node, int32_t parent) {
int32_t ancestor;
if (!nodeValid(node) || (node == SCENE_ROOT_NODE)) {
return false;
}
if (parent == NO_HANDLE) {
parent = SCENE_ROOT_NODE;
}
if (!nodeValid(parent)) {
return false;
}
for (ancestor = parent; ancestor != NO_HANDLE; ancestor = _scene.nodes[ancestor].parent) {
if (ancestor == node) {
return false;
}
}
_detach(node);
_attach(node, parent);
return true;
}
bool nodeSetPosition(int32_t node, Vec3T position) {
if (!nodeValid(node)) {
return false;
}
_scene.nodes[node].translation = position;
return true;
}
bool nodeSetRotation(int32_t node, QuatT rotation) {
if (!nodeValid(node)) {
return false;
}
_scene.nodes[node].rotation = quatNormalize(rotation);
return true;
}
bool nodeSetScale(int32_t node, Vec3T scale) {
if (!nodeValid(node)) {
return false;
}
_scene.nodes[node].scale = scale;
return true;
}
// Whether the node's mesh is drawn into shadow maps; a bulb's own mesh or a glowing sign is not.
bool nodeSetShadow(int32_t node, bool casts) {
if (!nodeValid(node)) {
return false;
}
_scene.nodes[node].shadowCaster = casts;
return true;
}
// Drives the node's skinned mesh from other nodes: joints (up to MAX_JOINTS) and their inverse
// bind matrices, copied. count 0 removes the skin and the mesh draws unskinned.
bool nodeSetSkin(int32_t node, const int32_t *joints, const Mat4T *inverseBind, int32_t count) {
NodeT *n;
if (!nodeValid(node) || (count < 0) || (count > MAX_JOINTS)) {
return false;
}
n = &_scene.nodes[node];
_freeSkin(n);
if (count == 0) {
return true;
}
n->skinJoints = SDL_malloc(sizeof(int32_t) * (size_t)count);
n->skinInverseBind = SDL_malloc(sizeof(Mat4T) * (size_t)count);
if ((n->skinJoints == NULL) || (n->skinInverseBind == NULL)) {
utilDie("Out of memory attaching a skin.");
}
memcpy(n->skinJoints, joints, sizeof(int32_t) * (size_t)count);
memcpy(n->skinInverseBind, inverseBind, sizeof(Mat4T) * (size_t)count);
n->skinCount = count;
return true;
}
// A picture on the shared unit quad (width by height world units, the node's scale on top): the
// frames become textures, a private blended double-sided material shows the current one, and the
// node's mesh becomes the quad. NULL frames clears it.
bool nodeSetSprite(int32_t node, SDL_Surface **frames, int32_t count, float width, float height, bool lit) {
NodeT *n;
SpriteNodeT *sprite;
int32_t slot;
int32_t x;
if (!nodeValid(node) || (_scene.device == NULL)) {
return false;
}
_freeSpriteNode(node);
if ((frames == NULL) || (count <= 0)) {
return true;
}
if (_scene.quadMesh == NO_HANDLE) {
// A unit quad in the XY plane facing +Z.
_scene.quadMesh = _quadMesh(1.0f, 1.0f, vec3(0.0f, -1.0f, 0.0f), vec3(0.0f, 0.0f, 1.0f));
if (_scene.quadMesh == NO_HANDLE) {
return false;
}
}
for (slot = 0; slot < _scene.spriteNodeCount; slot++) {
if (!_scene.spriteNodes[slot].used) {
break;
}
}
if (slot == _scene.spriteNodeCount) {
_scene.spriteNodes = SDL_realloc(_scene.spriteNodes, sizeof(SpriteNodeT) * (size_t)(_scene.spriteNodeCount + 1));
if (_scene.spriteNodes == NULL) {
utilDie("Out of memory for a sprite node.");
}
_scene.spriteNodeCount++;
}
sprite = &_scene.spriteNodes[slot];
memset(sprite, 0, sizeof(*sprite));
sprite->frames = SDL_calloc((size_t)count, sizeof(SDL_GPUTexture *));
if (sprite->frames == NULL) {
utilDie("Out of memory for a sprite node.");
}
for (x = 0; x < count; x++) {
sprite->frames[x] = _uploadTexture(frames[x], true);
if (sprite->frames[x] == NULL) {
while (x-- > 0) {
_releaseTexture(&sprite->frames[x]);
}
SDL_free(sprite->frames);
memset(sprite, 0, sizeof(*sprite));
return false;
}
}
sprite->material = materialNew();
if (sprite->material == NO_HANDLE) {
for (x = 0; x < count; x++) {
_releaseTexture(&sprite->frames[x]);
}
SDL_free(sprite->frames);
memset(sprite, 0, sizeof(*sprite));
return false;
}
_scene.materials[sprite->material].texture = sprite->frames[0];
_scene.materials[sprite->material].textureBorrowed = true;
_scene.materials[sprite->material].blend = true;
_scene.materials[sprite->material].doubleSided = true;
_scene.materials[sprite->material].unlit = !lit;
_scene.materials[sprite->material].roughness = 1.0f;
sprite->count = count;
sprite->width = width;
sprite->height = height;
sprite->used = true;
n = &_scene.nodes[node];
n->spriteSlot = slot;
n->mesh = _scene.quadMesh;
n->material = sprite->material;
_matchMorphWeights(n);
return true;
}
bool nodeSetSpriteFrame(int32_t node, int32_t frame) {
SpriteNodeT *sprite;
if (!nodeValid(node) || (_scene.nodes[node].spriteSlot == NO_HANDLE)) {
return false;
}
sprite = &_scene.spriteNodes[_scene.nodes[node].spriteSlot];
if ((frame < 0) || (frame >= sprite->count)) {
return false;
}
sprite->frame = frame;
_scene.materials[sprite->material].texture = sprite->frames[frame];
return true;
}
// Hides the node and everything under it.
bool nodeSetVisible(int32_t node, bool visible) {
if (!nodeValid(node)) {
return false;
}
_scene.nodes[node].visible = visible;
return true;
}
// Places the node at a world-space position and rotation by converting through its parent's
// world transform (what a physics body needs to drive a node under any parent). Scale is kept.
bool nodeSetWorldTransform(int32_t node, Vec3T position, QuatT rotation) {
NodeT *n;
Mat4T parentInverse;
Vec3T parentPosition;
QuatT parentRotation;
Vec3T parentScale;
if (!nodeValid(node)) {
return false;
}
n = &_scene.nodes[node];
if ((n->parent == NO_HANDLE) || (n->parent == SCENE_ROOT_NODE) || !mat4Invert(_scene.nodes[n->parent].world, &parentInverse)) {
n->translation = position;
n->rotation = quatNormalize(rotation);
return true;
}
mat4Decompose(_scene.nodes[n->parent].world, &parentPosition, &parentRotation, &parentScale);
n->translation = mat4TransformPoint(parentInverse, position);
n->rotation = quatNormalize(quatMultiply(quatInverse(parentRotation), rotation));
return true;
}
bool nodeValid(int32_t node) {
return (node >= 0) && (node < _scene.nodeCount) && _scene.nodes[node].used;
}
// ===== Scene =====
bool sceneAvailable(void) {
return _scene.device != NULL;
}
Ktx2FormatE sceneCompressedFormat(void) {
return _scene.compressedFormat;
}
// Smooth normals from the triangles: face normals accumulated per vertex, then normalised.
void sceneComputeNormals(SceneVertexT *vertices, int32_t vertexCount, const uint32_t *indices, int32_t indexCount) {
int32_t x;
for (x = 0; x < vertexCount; x++) {
vertices[x].normal[0] = 0.0f;
vertices[x].normal[1] = 0.0f;
vertices[x].normal[2] = 0.0f;
}
for (x = 0; x + 2 < indexCount; x += 3) {
SceneVertexT *a = &vertices[indices[x]];
SceneVertexT *b = &vertices[indices[x + 1]];
SceneVertexT *c = &vertices[indices[x + 2]];
SceneVertexT *corners[3] = { a, b, c };
Vec3T pa = vec3(a->position[0], a->position[1], a->position[2]);
Vec3T pb = vec3(b->position[0], b->position[1], b->position[2]);
Vec3T pc = vec3(c->position[0], c->position[1], c->position[2]);
Vec3T n = vec3Cross(vec3Subtract(pb, pa), vec3Subtract(pc, pa));
int32_t k;
for (k = 0; k < 3; k++) {
corners[k]->normal[0] += n.x;
corners[k]->normal[1] += n.y;
corners[k]->normal[2] += n.z;
}
}
for (x = 0; x < vertexCount; x++) {
Vec3T n = vec3Normalize(vec3(vertices[x].normal[0], vertices[x].normal[1], vertices[x].normal[2]));
vertices[x].normal[0] = n.x;
vertices[x].normal[1] = n.y;
vertices[x].normal[2] = n.z;
}
}
// Per-vertex tangents from the triangles' UV gradients (Lengyel's method), made perpendicular to
// the normal, with the bitangent's handedness in w. Triangles without a UV area get any
// perpendicular, which is right for a flat normal map and harmless otherwise.
void sceneComputeTangents(SceneVertexT *vertices, int32_t vertexCount, const uint32_t *indices, int32_t indexCount) {
Vec3T *tangents = SDL_calloc((size_t)vertexCount, sizeof(Vec3T));
Vec3T *bitangents = SDL_calloc((size_t)vertexCount, sizeof(Vec3T));
int32_t x;
if ((tangents == NULL) || (bitangents == NULL)) {
utilDie("Out of memory computing tangents.");
}
for (x = 0; x + 2 < indexCount; x += 3) {
uint32_t ia = indices[x];
uint32_t ib = indices[x + 1];
uint32_t ic = indices[x + 2];
SceneVertexT *a = &vertices[ia];
SceneVertexT *b = &vertices[ib];
SceneVertexT *c = &vertices[ic];
Vec3T e1 = vec3(b->position[0] - a->position[0], b->position[1] - a->position[1], b->position[2] - a->position[2]);
Vec3T e2 = vec3(c->position[0] - a->position[0], c->position[1] - a->position[1], c->position[2] - a->position[2]);
float u1 = b->uv[0] - a->uv[0];
float v1 = b->uv[1] - a->uv[1];
float u2 = c->uv[0] - a->uv[0];
float v2 = c->uv[1] - a->uv[1];
float d = u1 * v2 - u2 * v1;
Vec3T t;
Vec3T bt;
if (SDL_fabsf(d) < TANGENT_EPSILON) {
continue;
}
d = 1.0f / d;
t = vec3Scale(vec3Subtract(vec3Scale(e1, v2), vec3Scale(e2, v1)), d);
bt = vec3Scale(vec3Subtract(vec3Scale(e2, u1), vec3Scale(e1, u2)), d);
tangents[ia] = vec3Add(tangents[ia], t);
tangents[ib] = vec3Add(tangents[ib], t);
tangents[ic] = vec3Add(tangents[ic], t);
bitangents[ia] = vec3Add(bitangents[ia], bt);
bitangents[ib] = vec3Add(bitangents[ib], bt);
bitangents[ic] = vec3Add(bitangents[ic], bt);
}
for (x = 0; x < vertexCount; x++) {
Vec3T n = vec3(vertices[x].normal[0], vertices[x].normal[1], vertices[x].normal[2]);
Vec3T t = tangents[x];
float w = 1.0f;
if (vec3Length(t) < TANGENT_EPSILON) {
t = vec3Cross(n, (SDL_fabsf(n.y) < 0.9f) ? vec3(0.0f, 1.0f, 0.0f) : vec3(1.0f, 0.0f, 0.0f));
}
t = vec3Normalize(vec3Subtract(t, vec3Scale(n, vec3Dot(n, t))));
if (vec3Dot(vec3Cross(n, t), bitangents[x]) < 0.0f) {
w = -1.0f;
}
vertices[x].tangent[0] = t.x;
vertices[x].tangent[1] = t.y;
vertices[x].tangent[2] = t.z;
vertices[x].tangent[3] = w;
}
SDL_free(tangents);
SDL_free(bitangents);
}
void sceneDrawLine(Vec3T from, Vec3T to, uint8_t r, uint8_t g, uint8_t b) {
LineVertexT *vertex;
int32_t needed = _scene.lineVertexCount + 2;
int32_t x;
if (!_scene.enabled) {
return;
}
if (_scene.lineVertexCapacity < needed) {
int32_t capacity = SDL_max(needed, _scene.lineVertexCapacity * 2);
LineVertexT *grown = SDL_realloc(_scene.lineVertices, (size_t)capacity * sizeof(LineVertexT));
if (grown == NULL) {
return;
}
_scene.lineVertices = grown;
_scene.lineVertexCapacity = capacity;
}
vertex = &_scene.lineVertices[_scene.lineVertexCount];
for (x = 0; x < 2; x++) {
Vec3T at = (x == 0) ? from : to;
vertex[x].position[0] = at.x;
vertex[x].position[1] = at.y;
vertex[x].position[2] = at.z;
vertex[x].colour[0] = _linear(r);
vertex[x].colour[1] = _linear(g);
vertex[x].colour[2] = _linear(b);
vertex[x].colour[3] = 1.0f;
}
_scene.lineVertexCount = needed;
}
// Turns the layer on or off. Fails only when the machine has no GPU device.
bool sceneEnable(bool enabled) {
if (enabled && (_scene.device == NULL)) {
return false;
}
_scene.enabled = enabled;
return true;
}
void sceneGetSize(int32_t *width, int32_t *height) {
*width = _scene.width;
*height = _scene.height;
}
// Last frame's draw counts (collected, inside the view, the draw calls they became) and the bytes
// of texture the scene holds on the GPU.
void sceneGetStats(int32_t *total, int32_t *drawn, int32_t *batches, int64_t *textureBytes) {
*total = _scene.statTotal;
*drawn = _scene.statDrawn;
*batches = _scene.statBatches;
*textureBytes = _scene.textureBytes;
}
// The camera's view matrix: world to a frame looking down -Z with +X right and +Y up, which is
// also the listener frame positional sound wants.
Mat4T sceneGetView(void) {
return _viewOf(_scene.cameraNode);
}
// device may be NULL (no GPU backend on this machine); the scene then refuses to be enabled.
// Creates the root node, the shaders, the samplers and the stand-in textures.
bool sceneInit(SDL_GPUDevice *device, SDL_Renderer *renderer) {
SDL_GPUSamplerCreateInfo samplerInfo;
memset(&_scene, 0, sizeof(_scene));
_scene.device = device;
_scene.renderer = renderer;
_scene.cameraNode = NO_HANDLE;
_scene.perspective = true;
_scene.fov = DEFAULT_FOV;
_scene.near = DEFAULT_NEAR;
_scene.far = DEFAULT_FAR;
_scene.orthoHeight = DEFAULT_ORTHO_HEIGHT;
_scene.ambient = vec3(_linear(DEFAULT_AMBIENT), _linear(DEFAULT_AMBIENT), _linear(DEFAULT_AMBIENT));
_scene.tonemap = TONEMAP_NEUTRAL;
_scene.skyIntensity = 1.0f;
_scene.environment = true;
_scene.bloomThreshold = DEFAULT_BLOOM_THRESHOLD;
_scene.quadMesh = NO_HANDLE;
_scene.antialias = true;
_scene.sampleCount = SDL_GPU_SAMPLECOUNT_1;
_scene.shadowSize = SHADOW_SIZE;
_scene.shadowCascades = DEFAULT_CASCADES;
_scene.shadowDistance = DEFAULT_SHADOW_DISTANCE;
// The node tree is plain data and exists on every machine (physics bodies live on nodes);
// everything from here on needs the GPU, and without one the layer cannot be enabled.
_allocNode();
nodeSetName(SCENE_ROOT_NODE, "root");
if (device == NULL) {
return false;
}
_scene.depthFormat = _depthFormat();
_scene.shadowFormat = _shadowFormat();
_scene.hdrFormat = _hdrFormat();
// What KTX2 textures become: the best block format the device has, else plain RGBA. ETC2
// comes before BC3 because every OpenGL ES 3 part has it, and a Mali without ASTC would
// otherwise fall all the way to RGBA on the devices with the least memory.
_scene.compressedFormat = KTX2_RGBA;
if (rgpuTextureSupportsFormat(device, SDL_GPU_TEXTUREFORMAT_BC7_RGBA_UNORM_SRGB, SDL_GPU_TEXTURETYPE_2D, SDL_GPU_TEXTUREUSAGE_SAMPLER)) {
_scene.compressedFormat = KTX2_BC7;
} else if (rgpuTextureSupportsFormat(device, SDL_GPU_TEXTUREFORMAT_ASTC_4x4_UNORM_SRGB, SDL_GPU_TEXTURETYPE_2D, SDL_GPU_TEXTUREUSAGE_SAMPLER)) {
_scene.compressedFormat = KTX2_ASTC;
} else if (rgpuTextureSupportsFormat(device, RGPU_TEXTUREFORMAT_ETC2_RGBA8_UNORM_SRGB, SDL_GPU_TEXTURETYPE_2D, SDL_GPU_TEXTUREUSAGE_SAMPLER)) {
_scene.compressedFormat = KTX2_ETC2;
} else if (rgpuTextureSupportsFormat(device, SDL_GPU_TEXTUREFORMAT_BC3_RGBA_UNORM_SRGB, SDL_GPU_TEXTURETYPE_2D, SDL_GPU_TEXTUREUSAGE_SAMPLER)) {
_scene.compressedFormat = KTX2_BC3;
}
utilTrace("Scene: KTX2 textures as %s", (const char *[]){ "RGBA", "BC7", "ASTC 4x4", "BC3", "ETC2" }[_scene.compressedFormat]);
if (!_createShaders()) {
sceneQuit();
return false;
}
// Textures: trilinear with anisotropy (SDL asks the driver for it and drops it where it is not
// offered), and a nearest-neighbour sampler for pixel art that still walks the mipmap chain.
memset(&samplerInfo, 0, sizeof(samplerInfo));
// max_lod is not optional. SDL hands it straight to Vulkan's maxLod and D3D12's MaxLOD, so the
// zero a memset leaves clamps every lookup to mip level 0: no mipmapping at all, the anisotropy
// below mostly wasted, and the sky cube unable to pick a level by roughness. Every sampler here
// wants the whole chain, so it is set once for all of them.
samplerInfo.max_lod = SAMPLER_MAX_LOD;
samplerInfo.min_filter = SDL_GPU_FILTER_LINEAR;
samplerInfo.mag_filter = SDL_GPU_FILTER_LINEAR;
samplerInfo.mipmap_mode = SDL_GPU_SAMPLERMIPMAPMODE_LINEAR;
samplerInfo.address_mode_u = SDL_GPU_SAMPLERADDRESSMODE_REPEAT;
samplerInfo.address_mode_v = SDL_GPU_SAMPLERADDRESSMODE_REPEAT;
samplerInfo.address_mode_w = SDL_GPU_SAMPLERADDRESSMODE_REPEAT;
samplerInfo.enable_anisotropy = true;
samplerInfo.max_anisotropy = MAX_ANISOTROPY;
_scene.sampler = rgpuCreateSampler(device, &samplerInfo);
samplerInfo.min_filter = SDL_GPU_FILTER_NEAREST;
samplerInfo.mag_filter = SDL_GPU_FILTER_NEAREST;
samplerInfo.mipmap_mode = SDL_GPU_SAMPLERMIPMAPMODE_NEAREST;
samplerInfo.enable_anisotropy = false;
_scene.nearestSampler = rgpuCreateSampler(device, &samplerInfo);
// The shadow map is compared texel by texel, so no filtering, and clamped so its edge holds.
samplerInfo.min_filter = SDL_GPU_FILTER_NEAREST;
samplerInfo.mag_filter = SDL_GPU_FILTER_NEAREST;
samplerInfo.mipmap_mode = SDL_GPU_SAMPLERMIPMAPMODE_NEAREST;
samplerInfo.address_mode_u = SDL_GPU_SAMPLERADDRESSMODE_CLAMP_TO_EDGE;
samplerInfo.address_mode_v = SDL_GPU_SAMPLERADDRESSMODE_CLAMP_TO_EDGE;
samplerInfo.address_mode_w = SDL_GPU_SAMPLERADDRESSMODE_CLAMP_TO_EDGE;
_scene.shadowSampler = rgpuCreateSampler(device, &samplerInfo);
// The post pass reads the HDR target texel for texel.
samplerInfo.min_filter = SDL_GPU_FILTER_LINEAR;
samplerInfo.mag_filter = SDL_GPU_FILTER_LINEAR;
_scene.postSampler = rgpuCreateSampler(device, &samplerInfo);
// The sky cube: trilinear (roughness picks the mip level) and clamped.
samplerInfo.mipmap_mode = SDL_GPU_SAMPLERMIPMAPMODE_LINEAR;
_scene.skySampler = rgpuCreateSampler(device, &samplerInfo);
_scene.white = _solidTexture(255, 255, 255);
_scene.flatNormal = _solidTexture(128, 128, 255);
_scene.black = _solidTexture(0, 0, 0);
{
uint16_t dark[CUBE_FACES * CUBE_CHANNELS] = { 0 };
_scene.blackCube = _uploadCube(dark, 1);
}
{
float zero[MORPH_FLOATS] = { 0.0f };
_scene.noMorphs = _uploadBuffer(SDL_GPU_BUFFERUSAGE_GRAPHICS_STORAGE_READ, zero, sizeof(zero));
}
_scene.shadowMapsNone = _createShadowArray(SDL_GPU_TEXTURETYPE_2D_ARRAY, 1, 1);
_scene.depthNone = _createShadowArray(SDL_GPU_TEXTURETYPE_2D, 1, 1);
if ((_scene.sampler == NULL) || (_scene.nearestSampler == NULL) || (_scene.shadowSampler == NULL) || (_scene.postSampler == NULL) || (_scene.skySampler == NULL) || (_scene.white == NULL) || (_scene.flatNormal == NULL) || (_scene.black == NULL) || (_scene.blackCube == NULL) || (_scene.shadowMapsNone == NULL) || (_scene.depthNone == NULL) || (_scene.noMorphs == NULL)) {
utilTrace("Scene: %s", SDL_GetError());
sceneQuit();
return false;
}
return true;
}
bool sceneIsEnabled(void) {
return _scene.enabled;
}
// The GUI shown on the surface under an overlay point: the ray through that pixel against the
// triangles of every visible node whose material shows a GUI (a screen, a sign), taking the nearest
// hit's texture coordinates as the place on the GUI. False when the ray meets none of them.
bool sceneProbeGui(float x, float y, int32_t *gui, float *u, float *v) {
Vec3T origin = sceneUnproject(x, y, 0.0f);
Vec3T direction = vec3Subtract(sceneUnproject(x, y, 1.0f), origin);
float best = FLT_MAX;
int32_t n;
*gui = NO_HANDLE;
if (_scene.width <= 0) {
return false;
}
for (n = 0; n < _scene.nodeCount; n++) {
const NodeT *node = &_scene.nodes[n];
const MeshT *mesh;
Mat4T inverse;
Vec3T localOrigin;
Vec3T localDirection;
uint32_t i;
if (!node->used || !node->worldVisible || !meshValid(node->mesh) || !materialValid(node->material) || (_scene.materials[node->material].gui == NO_HANDLE)) {
continue;
}
if (!mat4Invert(node->world, &inverse)) {
continue;
}
mesh = &_scene.meshes[node->mesh];
localOrigin = mat4TransformPoint(inverse, origin);
localDirection = vec3Subtract(mat4TransformPoint(inverse, vec3Add(origin, direction)), localOrigin);
for (i = 0; i + 2 < mesh->indexCount; i += 3) {
const SceneVertexT *a = &mesh->vertices[mesh->indices[i]];
const SceneVertexT *b = &mesh->vertices[mesh->indices[i + 1]];
const SceneVertexT *c = &mesh->vertices[mesh->indices[i + 2]];
Vec3T edge1 = vec3Subtract(vec3(b->position[0], b->position[1], b->position[2]), vec3(a->position[0], a->position[1], a->position[2]));
Vec3T edge2 = vec3Subtract(vec3(c->position[0], c->position[1], c->position[2]), vec3(a->position[0], a->position[1], a->position[2]));
Vec3T p = vec3Cross(localDirection, edge2);
float det = vec3Dot(edge1, p);
Vec3T t;
Vec3T q;
float b1;
float b2;
float distance;
// Moller-Trumbore, both faces.
if (fabsf(det) < MATH_EPSILON) {
continue;
}
t = vec3Subtract(localOrigin, vec3(a->position[0], a->position[1], a->position[2]));
b1 = vec3Dot(t, p) / det;
if ((b1 < 0.0f) || (b1 > 1.0f)) {
continue;
}
q = vec3Cross(t, edge1);
b2 = vec3Dot(localDirection, q) / det;
if ((b2 < 0.0f) || (b1 + b2 > 1.0f)) {
continue;
}
distance = vec3Dot(edge2, q) / det;
if ((distance <= 0.0f) || (distance >= best)) {
continue;
}
best = distance;
*gui = _scene.materials[node->material].gui;
*u = (1.0f - b1 - b2) * a->uv[0] + b1 * b->uv[0] + b2 * c->uv[0];
*v = (1.0f - b1 - b2) * a->uv[1] + b1 * b->uv[1] + b2 * c->uv[1];
}
}
return best < FLT_MAX;
}
// World point to overlay coordinates using the last rendered frame's camera. Returns false when
// the point is behind the camera (x and y are still filled in) or there is no target yet (zeros).
bool sceneProject(Vec3T world, float *x, float *y, float *depth) {
float w;
Vec3T clip;
if ((_scene.width <= 0) || (_scene.height <= 0)) {
*x = 0.0f;
*y = 0.0f;
*depth = 0.0f;
return false;
}
clip = mat4Project(_scene.viewProjection, world, &w);
*x = (clip.x + 1.0f) / 2.0f * (float)_scene.width;
*y = (1.0f - clip.y) / 2.0f * (float)_scene.height;
*depth = clip.z;
return w > 0.0f;
}
void sceneQuit(void) {
int32_t x;
if (_scene.device != NULL) {
for (x = 0; x < _scene.materialCount; x++) {
_freeMaterialTexture(&_scene.materials[x]);
}
for (x = 0; x < _scene.feedCount; x++) {
if (_scene.feeds[x].used) {
_freeFeed(&_scene.feeds[x]);
}
}
for (x = 0; x < _scene.meshCount; x++) {
if (_scene.meshes[x].used) {
meshDelete(x);
}
}
_destroyPipelines();
if (_scene.instanceBuffer != NULL) {
rgpuReleaseBuffer(_scene.device, _scene.instanceBuffer);
}
if (_scene.instanceTransfer != NULL) {
rgpuReleaseTransferBuffer(_scene.device, _scene.instanceTransfer);
}
if (_scene.particleVertex != NULL) {
rgpuReleaseShader(_scene.device, _scene.particleVertex);
}
if (_scene.particleFragment != NULL) {
rgpuReleaseShader(_scene.device, _scene.particleFragment);
}
if (_scene.lineVertex != NULL) {
rgpuReleaseShader(_scene.device, _scene.lineVertex);
}
if (_scene.lineFragment != NULL) {
rgpuReleaseShader(_scene.device, _scene.lineFragment);
}
_releasePipeline(&_scene.postPipeline);
if (_scene.postVertex != NULL) {
rgpuReleaseShader(_scene.device, _scene.postVertex);
}
if (_scene.postFragment != NULL) {
rgpuReleaseShader(_scene.device, _scene.postFragment);
}
if (_scene.postSampler != NULL) {
rgpuReleaseSampler(_scene.device, _scene.postSampler);
}
if (_scene.skySampler != NULL) {
rgpuReleaseSampler(_scene.device, _scene.skySampler);
}
if (_scene.skyFragment != NULL) {
rgpuReleaseShader(_scene.device, _scene.skyFragment);
}
if (_scene.bloomDownFragment != NULL) {
rgpuReleaseShader(_scene.device, _scene.bloomDownFragment);
}
if (_scene.bloomUpFragment != NULL) {
rgpuReleaseShader(_scene.device, _scene.bloomUpFragment);
}
_releasePipeline(&_scene.bloomDownPipeline);
_releasePipeline(&_scene.bloomUpPipeline);
_destroyBloomTargets();
for (x = 0; x < MAX_VIEWS; x++) {
_freeView(&_scene.views[x]);
}
for (x = 0; x < _scene.nodeCount; x++) {
if (_scene.nodes[x].used) {
_freeSpriteNode(x);
}
}
_releaseTexture(&_scene.skyCube);
_releaseTexture(&_scene.blackCube);
if (_scene.particleBuffer != NULL) {
rgpuReleaseBuffer(_scene.device, _scene.particleBuffer);
}
if (_scene.particleTransfer != NULL) {
rgpuReleaseTransferBuffer(_scene.device, _scene.particleTransfer);
}
if (_scene.lineBuffer != NULL) {
rgpuReleaseBuffer(_scene.device, _scene.lineBuffer);
}
if (_scene.lineTransfer != NULL) {
rgpuReleaseTransferBuffer(_scene.device, _scene.lineTransfer);
}
_releaseParticleTextures(true);
if (_scene.vertexStatic != NULL) {
rgpuReleaseShader(_scene.device, _scene.vertexStatic);
}
if (_scene.vertexSkinned != NULL) {
rgpuReleaseShader(_scene.device, _scene.vertexSkinned);
}
if (_scene.fragment != NULL) {
rgpuReleaseShader(_scene.device, _scene.fragment);
}
if (_scene.depthCutoutFragment != NULL) {
rgpuReleaseShader(_scene.device, _scene.depthCutoutFragment);
_scene.depthCutoutFragment = NULL;
}
if (_scene.depthFragment != NULL) {
rgpuReleaseShader(_scene.device, _scene.depthFragment);
}
if (_scene.sampler != NULL) {
rgpuReleaseSampler(_scene.device, _scene.sampler);
}
if (_scene.nearestSampler != NULL) {
rgpuReleaseSampler(_scene.device, _scene.nearestSampler);
}
if (_scene.shadowSampler != NULL) {
rgpuReleaseSampler(_scene.device, _scene.shadowSampler);
}
_destroyShadowMaps();
if (_scene.shadowMapsNone != NULL) {
rgpuReleaseTexture(_scene.device, _scene.shadowMapsNone);
}
_releaseTexture(&_scene.depthNone);
_releaseTexture(&_scene.white);
_releaseTexture(&_scene.flatNormal);
_releaseTexture(&_scene.black);
if (_scene.noMorphs != NULL) {
rgpuReleaseBuffer(_scene.device, _scene.noMorphs);
}
_destroyTargets();
}
for (x = 0; x < _scene.nodeCount; x++) {
SDL_free(_scene.nodes[x].name);
_freeSkin(&_scene.nodes[x]);
_freeMorphWeights(&_scene.nodes[x]);
}
SDL_free(_scene.nodes);
SDL_free(_scene.meshes);
SDL_free(_scene.materials);
SDL_free(_scene.feeds);
SDL_free(_scene.draws);
SDL_free(_scene.particleVertices);
SDL_free(_scene.lineVertices);
SDL_free(_scene.instances);
SDL_free(_scene.skins);
SDL_free(_scene.skip);
SDL_free(_scene.particleOrder);
SDL_free(_scene.blendedOrder);
SDL_free(_scene.spriteNodes);
SDL_free(_scene.sizedTextures);
SDL_free(_scene.sizedBytes);
memset(&_scene, 0, sizeof(_scene));
}
// Draws the frame into the colour target and returns it as a texture for compositing, or NULL when
// the layer is off or has no target yet. The command buffer is submitted before the 2D renderer
// flushes its own, so the composite always samples this frame's result.
SDL_Texture *sceneRender(void) {
SDL_GPUDepthStencilTargetInfo depth;
SDL_GPUCommandBuffer *commands;
SDL_GPURenderPass *pass;
FragmentUniformsT fragmentUniforms;
CameraFrameT main;
CameraFrameT frame;
Mat4T identity = mat4Identity();
bool *skip;
int32_t x;
int32_t drawCount;
int32_t opaqueCount = 0;
int32_t blendedStart;
int32_t slot;
int32_t face;
int32_t layers = 0;
NodeT *node;
if (!_scene.enabled || (_scene.colour == NULL)) {
_scene.lineVertexCount = 0;
return NULL;
}
// Transforms, the window's camera and the lights for this frame.
_updateWorld(SCENE_ROOT_NODE, &identity, true);
_cameraFrame(_scene.cameraNode, _scene.width, _scene.height, &main);
main.colour = _scene.colour;
main.multisampled = _scene.multisampled;
main.depth = _scene.depth;
main.softDepth = _scene.softDepth;
main.output = _scene.output;
main.main = true;
main.sampleSet = (_scene.sampleCount == SDL_GPU_SAMPLECOUNT_1) ? SAMPLE_SET_SINGLE : SAMPLE_SET_MULTI;
_scene.viewProjection = main.viewProjection;
memset(&fragmentUniforms, 0, sizeof(fragmentUniforms));
fragmentUniforms.ambient[0] = _scene.ambient.x;
fragmentUniforms.ambient[1] = _scene.ambient.y;
fragmentUniforms.ambient[2] = _scene.ambient.z;
fragmentUniforms.ambient[3] = 1.0f;
fragmentUniforms.fog[0] = _scene.fogColour.x;
fragmentUniforms.fog[1] = _scene.fogColour.y;
fragmentUniforms.fog[2] = _scene.fogColour.z;
fragmentUniforms.fog[3] = _scene.fog ? 1.0f : 0.0f;
fragmentUniforms.fogRange[0] = _scene.fogNear;
fragmentUniforms.fogRange[1] = _scene.fogFar;
fragmentUniforms.environment[0] = (_scene.environment && (_scene.skyCube != NULL)) ? 1.0f : 0.0f;
fragmentUniforms.environment[1] = (float)(_scene.skyLevels - 1);
fragmentUniforms.environment[2] = _scene.skyIntensity;
for (x = 0; x < SH_COEFFICIENTS; x++) {
fragmentUniforms.sh[x][0] = _scene.sh[x].x;
fragmentUniforms.sh[x][1] = _scene.sh[x].y;
fragmentUniforms.sh[x][2] = _scene.sh[x].z;
}
_scene.shadowCount = 0;
_fillLights(&fragmentUniforms);
// Collect what to draw in one walk: opaque draws from the front of the array, sorted for
// batching, blended ones from the back, then moved up behind them; each camera orders those
// back to front from its own eye as it renders (_orderBlended).
if (_scene.drawCapacity < _scene.nodeCount) {
_scene.draws = SDL_realloc(_scene.draws, sizeof(DrawT) * (size_t)_scene.nodeCount);
if (_scene.draws == NULL) {
utilDie("Out of memory collecting scene draws.");
}
_scene.drawCapacity = _scene.nodeCount;
}
blendedStart = _scene.nodeCount;
for (x = 0; x < _scene.nodeCount; x++) {
node = &_scene.nodes[x];
if (!node->used || !node->worldVisible || !meshValid(node->mesh)) {
continue;
}
if ((node->material != NO_HANDLE) && _scene.materials[node->material].blend) {
blendedStart--;
_scene.draws[blendedStart].node = x;
} else {
_scene.draws[opaqueCount].node = x;
opaqueCount++;
}
}
drawCount = opaqueCount + (_scene.nodeCount - blendedStart);
_scene.opaqueCount = opaqueCount;
qsort(_scene.draws, (size_t)opaqueCount, sizeof(DrawT), _compareOpaque);
if (drawCount > opaqueCount) {
memmove(&_scene.draws[opaqueCount], &_scene.draws[blendedStart], sizeof(DrawT) * (size_t)(drawCount - opaqueCount));
}
commands = rgpuAcquireCommandBuffer(_scene.device);
if (commands == NULL) {
utilTrace("Scene: %s", SDL_GetError());
return NULL;
}
// Bounds and matrices for every draw.
_boundDraws(drawCount);
_fillInstances(drawCount);
_uploadInstances(commands, drawCount);
// The shadow passes: depth from every shadow light into its map layer, or its six cube faces,
// fitted to the window's camera (billboards turned to it) and shared by every view this frame.
for (slot = 0; slot < _scene.shadowCount; slot++) {
layers += (_scene.shadows[slot].type == SHADOW_CUBE) ? CUBE_FACES : _scene.shadows[slot].cascades;
}
if ((_scene.shadowCount > 0) && (drawCount > 0) && _createShadowMaps(layers)) {
skip = _skipScratch(drawCount);
_fitShadows(drawCount, &main);
fragmentUniforms.shadowParams[0] = SHADOW_BIAS;
fragmentUniforms.shadowParams[1] = 1.0f / (float)_scene.shadowSize;
fragmentUniforms.shadowParams[2] = (float)_scene.shadowCount;
memset(&depth, 0, sizeof(depth));
depth.clear_depth = 1.0f;
depth.load_op = SDL_GPU_LOADOP_CLEAR;
depth.store_op = SDL_GPU_STOREOP_STORE;
depth.stencil_load_op = SDL_GPU_LOADOP_DONT_CARE;
depth.stencil_store_op = SDL_GPU_STOREOP_DONT_CARE;
for (slot = 0; slot < _scene.shadowCount; slot++) {
ShadowT *shadow = &_scene.shadows[slot];
fragmentUniforms.shadowInfo[slot][0] = (float)shadow->type;
fragmentUniforms.shadowInfo[slot][1] = (float)shadow->layer;
fragmentUniforms.shadowInfo[slot][2] = shadow->near;
fragmentUniforms.shadowInfo[slot][3] = shadow->far;
if (shadow->type == SHADOW_CUBE) {
// Faces rendered last frame from the same light and casters are still good.
ShadowCacheT *cache = &_scene.shadowCache[slot];
uint64_t hash = _cubeHash(shadow, drawCount);
if ((hash != 0) && (cache->node == shadow->node) && (cache->layer == shadow->layer) && (cache->mapsVersion == _scene.shadowMapsVersion) && (cache->hash == hash)) {
continue;
}
cache->node = shadow->node;
cache->layer = shadow->layer;
cache->mapsVersion = _scene.shadowMapsVersion;
cache->hash = hash;
for (face = 0; face < CUBE_FACES; face++) {
_cullFace(shadow, face, drawCount, skip);
depth.texture = _scene.shadowMaps;
depth.layer = (Uint8)(shadow->layer + face);
pass = rgpuBeginRenderPass(commands, NULL, 0, &depth);
_drawList(commands, pass, drawCount, &shadow->faces[face], &main, true, true, skip, NULL, SAMPLE_SET_SINGLE);
rgpuEndRenderPass(pass);
}
} else if (shadow->type == SHADOW_CASCADE) {
int32_t k;
fragmentUniforms.shadowInfo[slot][2] = (float)shadow->cascades;
for (k = 0; k < shadow->cascades; k++) {
fragmentUniforms.shadowMatrix[slot * MAX_CASCADES + k] = shadow->faces[k];
fragmentUniforms.cascadeSplits[slot][k] = shadow->splits[k];
_cullCascade(shadow, k, drawCount, skip);
depth.texture = _scene.shadowMaps;
depth.layer = (Uint8)(shadow->layer + k);
pass = rgpuBeginRenderPass(commands, NULL, 0, &depth);
_drawList(commands, pass, drawCount, &shadow->faces[k], &main, true, false, skip, NULL, SAMPLE_SET_SINGLE);
rgpuEndRenderPass(pass);
}
} else {
fragmentUniforms.shadowMatrix[slot * MAX_CASCADES] = shadow->matrix;
depth.texture = _scene.shadowMaps;
depth.layer = (Uint8)shadow->layer;
pass = rgpuBeginRenderPass(commands, NULL, 0, &depth);
_drawList(commands, pass, drawCount, &shadow->matrix, &main, true, false, NULL, NULL, SAMPLE_SET_SINGLE);
rgpuEndRenderPass(pass);
}
}
} else {
// Lights flagged to cast but nothing to draw into: no slots this frame.
for (x = 0; x < MAX_LIGHTS; x++) {
fragmentUniforms.lights[x].cone[2] = 0.0f;
}
}
_uploadLines(commands);
// The views first, so the window's frame can show them, then the window's camera.
for (x = 0; x < MAX_VIEWS; x++) {
ViewT *view = &_scene.views[x];
if (!view->used || (view->colour == NULL)) {
continue;
}
_cameraFrame(view->camera, view->width, view->height, &frame);
frame.colour = view->colour;
frame.depth = view->depth;
frame.output = view->output;
_renderCamera(commands, &frame, &fragmentUniforms, drawCount);
}
_renderCamera(commands, &main, &fragmentUniforms, drawCount);
rgpuSubmitCommandBuffer(commands);
_scene.lineVertexCount = 0;
return _scene.composite;
}
// (Re)creates the render targets at the overlay's size.
bool sceneResize(int32_t width, int32_t height) {
SDL_GPUTextureCreateInfo info;
if (_scene.device == NULL) {
return false;
}
if ((width == _scene.width) && (height == _scene.height)) {
return true;
}
_destroyTargets();
_scene.width = width;
_scene.height = height;
// The display texture the post pass writes and the 2D renderer composites.
memset(&info, 0, sizeof(info));
info.type = SDL_GPU_TEXTURETYPE_2D;
info.format = rgpuGetTextureFormatFromPixelFormat(SDL_PIXELFORMAT_BGRA32);
info.usage = SDL_GPU_TEXTUREUSAGE_COLOR_TARGET | SDL_GPU_TEXTUREUSAGE_SAMPLER;
info.width = (Uint32)width;
info.height = (Uint32)height;
info.layer_count_or_depth = 1;
info.num_levels = 1;
info.sample_count = SDL_GPU_SAMPLECOUNT_1;
_scene.output = rgpuCreateTexture(_scene.device, &info);
if (_scene.output == NULL) {
utilTrace("Scene: %s", SDL_GetError());
return false;
}
// The scene itself renders in linear light into the HDR target.
info.format = _scene.hdrFormat;
_scene.colour = rgpuCreateTexture(_scene.device, &info);
if (_scene.colour == NULL) {
utilTrace("Scene: %s", SDL_GetError());
_destroyTargets();
return false;
}
// 4x multisampling when wanted and offered: a multisampled colour target resolved into colour,
// and a multisampled depth target to match.
_scene.sampleCount = SDL_GPU_SAMPLECOUNT_1;
if (_scene.antialias && rgpuTextureSupportsSampleCount(_scene.device, info.format, SDL_GPU_SAMPLECOUNT_4) && rgpuTextureSupportsSampleCount(_scene.device, _scene.depthFormat, SDL_GPU_SAMPLECOUNT_4)) {
info.usage = SDL_GPU_TEXTUREUSAGE_COLOR_TARGET;
info.sample_count = SDL_GPU_SAMPLECOUNT_4;
_scene.multisampled = rgpuCreateTexture(_scene.device, &info);
if (_scene.multisampled != NULL) {
_scene.sampleCount = SDL_GPU_SAMPLECOUNT_4;
} else {
utilTrace("Scene: no multisampling: %s", SDL_GetError());
info.sample_count = SDL_GPU_SAMPLECOUNT_1;
}
}
_destroyPipelines();
info.format = _scene.depthFormat;
info.usage = SDL_GPU_TEXTUREUSAGE_DEPTH_STENCIL_TARGET;
_scene.depth = rgpuCreateTexture(_scene.device, &info);
if (_scene.depth == NULL) {
utilTrace("Scene: %s", SDL_GetError());
_destroyTargets();
return false;
}
// The camera's depth for soft particles: single sample in the shadow pipelines' format, so the
// depth-only pipelines can fill it.
_scene.softDepth = _createShadowArray(SDL_GPU_TEXTURETYPE_2D, 1, 0);
_scene.composite = renderWrapTexture(_scene.renderer, _scene.output, SDL_PIXELFORMAT_BGRA32, width, height);
if (_scene.composite == NULL) {
utilTrace("Scene: %s", SDL_GetError());
_destroyTargets();
return false;
}
SDL_SetTextureBlendMode(_scene.composite, SDL_BLENDMODE_BLEND);
return true;
}
void sceneSetAmbient(uint8_t r, uint8_t g, uint8_t b) {
_scene.ambient = vec3(_linear(r), _linear(g), _linear(b));
}
// 4x multisampling on or off (on by default where the device offers it); takes effect on the
// next resize, so the targets are rebuilt here.
void sceneSetAntialias(bool antialias) {
int32_t width = _scene.width;
int32_t height = _scene.height;
_scene.antialias = antialias;
if (width > 0) {
_destroyTargets();
sceneResize(width, height);
}
}
void sceneSetBackground(uint8_t r, uint8_t g, uint8_t b, uint8_t a) {
_scene.background.r = _linear(r);
_scene.background.g = _linear(g);
_scene.background.b = _linear(b);
_scene.background.a = a / COLOUR_MAX;
}
// The glow of everything brighter than the threshold, added back at the given strength (0 off).
void sceneSetBloom(float threshold, float strength) {
_scene.bloomThreshold = SDL_max(threshold, 0.0f);
_scene.bloomStrength = SDL_max(strength, 0.0f);
}
// Whether the sky lights the scene (its diffuse light replacing the flat ambient, its reflections
// on metals and glossy surfaces); on by default when a sky is set.
bool sceneSetEnvironment(bool lit) {
_scene.environment = lit;
return true;
}
void sceneSetExposure(float stops) {
_scene.exposure = SDL_clamp(stops, MIN_EXPOSURE, MAX_EXPOSURE);
}
// Distance fog from near to far; far no greater than near switches it off.
void sceneSetFog(uint8_t r, uint8_t g, uint8_t b, float near, float far) {
_scene.fogColour = vec3(_linear(r), _linear(g), _linear(b));
_scene.fogNear = near;
_scene.fogFar = far;
_scene.fog = far > near;
}
// Cascades for a directional light's shadow, 1 to MAX_CASCADES (1 fits one map to the whole scene).
void sceneSetShadowCascades(int32_t count) {
_scene.shadowCascades = SDL_clamp(count, 1, MAX_CASCADES);
}
// How far from the camera cascaded shadows reach; beyond it nothing is shadowed by the sun.
void sceneSetShadowDistance(float distance) {
_scene.shadowDistance = SDL_max(distance, 1.0f);
}
// The shadow map's size in texels per side (default 1024, clamped to 256..4096); larger is
// sharper and slower. The map is rebuilt on the next frame that needs it.
void sceneSetShadowSize(int32_t size) {
_scene.shadowSize = SDL_clamp(size, SHADOW_SIZE_MIN, SHADOW_SIZE_MAX);
_destroyShadowMaps();
}
// The sky from an equirectangular image of linear RGB floats (NULL removes it): six cube faces
// resampled from it, a mip chain for reflections by roughness, and the spherical harmonics of its
// diffuse light.
bool sceneSetSky(const float *rgb, int32_t width, int32_t height) {
int32_t face;
int32_t f;
int32_t x;
int32_t y;
uint16_t *pixels;
_releaseTexture(&_scene.skyCube);
_scene.skyLevels = 0;
if ((rgb == NULL) || (_scene.device == NULL)) {
return rgb == NULL;
}
face = CUBE_FACE_MIN;
while ((face * 2 <= height / 2) && (face * 2 <= CUBE_FACE_MAX)) {
face *= 2;
}
pixels = SDL_malloc((size_t)CUBE_FACES * (size_t)face * (size_t)face * CUBE_CHANNELS * sizeof(uint16_t));
if (pixels == NULL) {
utilDie("Out of memory building the sky.");
}
for (f = 0; f < CUBE_FACES; f++) {
for (y = 0; y < face; y++) {
for (x = 0; x < face; x++) {
float s = ((float)x + 0.5f) / (float)face * 2.0f - 1.0f;
float t = ((float)y + 0.5f) / (float)face * 2.0f - 1.0f;
Vec3T colour = _sampleEquirect(rgb, width, height, _faceDirection(f, s, t));
uint16_t *out = pixels + (((size_t)f * (size_t)face + (size_t)y) * (size_t)face + (size_t)x) * CUBE_CHANNELS;
out[0] = _half(colour.x);
out[1] = _half(colour.y);
out[2] = _half(colour.z);
out[3] = _half(1.0f);
}
}
}
_scene.skyCube = _uploadCube(pixels, face);
SDL_free(pixels);
if (_scene.skyCube == NULL) {
return false;
}
_scene.skyLevels = (int32_t)_mipLevels(face, face);
_computeSh(rgb, width, height);
return true;
}
void sceneSetSkyIntensity(float intensity) {
_scene.skyIntensity = SDL_max(intensity, 0.0f);
}
void sceneSetTonemap(SceneTonemapE tonemap) {
_scene.tonemap = tonemap;
}
// Overlay coordinates back to a world point: the point on the ray through that pixel at the given
// distance from the camera's near plane, using the last rendered frame's camera. Two distances
// give a ray for picking.
Vec3T sceneUnproject(float x, float y, float distance) {
Mat4T inverse;
Vec3T ndc;
Vec3T near;
Vec3T far;
if ((_scene.width <= 0) || !mat4Invert(_scene.viewProjection, &inverse)) {
return vec3(0.0f, 0.0f, 0.0f);
}
ndc = vec3(x / (float)_scene.width * 2.0f - 1.0f, 1.0f - y / (float)_scene.height * 2.0f, 0.0f);
near = mat4TransformPoint(inverse, ndc);
ndc.z = 1.0f;
far = mat4TransformPoint(inverse, ndc);
return vec3Add(near, vec3Scale(vec3Normalize(vec3Subtract(far, near)), distance));
}
// Rebuilds every node's world matrix now (sceneRender does it too); physics needs them before the
// step, after animation has moved the nodes.
void sceneUpdateTransforms(void) {
Mat4T identity = mat4Identity();
if (_scene.nodeCount > 0) {
_updateWorld(SCENE_ROOT_NODE, &identity, true);
}
}
// Once per frame before sceneRender: copies every player a material shows into that feed's RGBA
// target with the 2D renderer (which converts YUV on the way), then flushes the renderer so the
// copies are queued ahead of the scene's own command buffer.
void sceneUpdateVideo(SceneVideoSourceFn source) {
int32_t x;
bool any = false;
SDL_Texture *previous = SDL_GetRenderTarget(_scene.renderer);
SDL_Texture *frame;
float width;
float height;
for (x = 0; x < _scene.feedCount; x++) {
FeedT *feed = &_scene.feeds[x];
if (!feed->used) {
continue;
}
frame = source(feed->player);
if (frame == NULL) {
continue;
}
if (feed->target == NULL) {
SDL_GetTextureSize(frame, &width, &height);
feed->target = SDL_CreateTexture(_scene.renderer, SDL_PIXELFORMAT_BGRA32, SDL_TEXTUREACCESS_TARGET, (int32_t)width, (int32_t)height);
if (feed->target == NULL) {
utilTrace("Scene: video feed: %s", SDL_GetError());
continue;
}
feed->gpu = renderTextureFor(feed->target);
}
SDL_SetRenderTarget(_scene.renderer, feed->target);
SDL_RenderTexture(_scene.renderer, frame, NULL, NULL);
any = true;
}
if (any) {
// Back to whatever was being drawn into, which is a texture when --rotate is turning the frame.
SDL_SetRenderTarget(_scene.renderer, previous);
SDL_FlushRenderer(_scene.renderer);
}
}
// ===== Terrain =====
// The world height of a heightmap mesh's node at world x, z (the node's position and scale apply;
// its rotation is ignored, terrains lying flat), bilinear between samples; false off the mesh.
bool terrainGetHeight(int32_t node, float x, float z, float *height) {
const float *heights;
int32_t columns;
int32_t rows;
float sizeX;
float sizeY;
float sizeZ;
Vec3T position;
QuatT rotation;
Vec3T scale;
float u;
float v;
int32_t x0;
int32_t y0;
float fx;
float fy;
float h00;
float h10;
float h01;
float h11;
if (!nodeValid(node) || !meshGetHeights(_scene.nodes[node].mesh, &heights, &columns, &rows, &sizeX, &sizeY, &sizeZ) || !nodeGetWorldTransform(node, &position, &rotation, &scale)) {
return false;
}
u = ((x - position.x) / SDL_max(scale.x, 0.0001f) + sizeX / 2.0f) / sizeX;
v = ((z - position.z) / SDL_max(scale.z, 0.0001f) + sizeZ / 2.0f) / sizeZ;
if ((u < 0.0f) || (u > 1.0f) || (v < 0.0f) || (v > 1.0f)) {
return false;
}
u *= (float)columns;
v *= (float)rows;
x0 = SDL_min((int32_t)u, columns - 1);
y0 = SDL_min((int32_t)v, rows - 1);
fx = u - (float)x0;
fy = v - (float)y0;
h00 = heights[y0 * (columns + 1) + x0];
h10 = heights[y0 * (columns + 1) + x0 + 1];
h01 = heights[(y0 + 1) * (columns + 1) + x0];
h11 = heights[(y0 + 1) * (columns + 1) + x0 + 1];
*height = position.y + scale.y * sizeY * ((h00 * (1.0f - fx) + h10 * fx) * (1.0f - fy) + (h01 * (1.0f - fx) + h11 * fx) * fy);
return true;
}
// ===== Views: cameras rendered to textures =====
bool viewDelete(int32_t view) {
int32_t x;
if (!viewValid(view)) {
return false;
}
_freeView(&_scene.views[view]);
for (x = 0; x < _scene.materialCount; x++) {
if (_scene.materials[x].used && (_scene.materials[x].view == view)) {
_scene.materials[x].view = NO_HANDLE;
}
}
return true;
}
// A view of the scene from a camera node (the default view until viewSetCamera), width by height
// pixels, rendered every frame before the window's camera; NO_HANDLE when the slots are used up.
int32_t viewNew(int32_t width, int32_t height) {
SDL_GPUTextureCreateInfo info;
ViewT *view;
int32_t x;
if (_scene.device == NULL) {
return NO_HANDLE;
}
for (x = 0; x < MAX_VIEWS; x++) {
if (!_scene.views[x].used) {
break;
}
}
if (x == MAX_VIEWS) {
return NO_HANDLE;
}
view = &_scene.views[x];
memset(view, 0, sizeof(*view));
view->camera = NO_HANDLE;
view->width = SDL_clamp(width, 1, VIEW_SIZE_MAX);
view->height = SDL_clamp(height, 1, VIEW_SIZE_MAX);
memset(&info, 0, sizeof(info));
info.type = SDL_GPU_TEXTURETYPE_2D;
info.format = _scene.hdrFormat;
info.usage = SDL_GPU_TEXTUREUSAGE_COLOR_TARGET | SDL_GPU_TEXTUREUSAGE_SAMPLER;
info.width = (Uint32)view->width;
info.height = (Uint32)view->height;
info.layer_count_or_depth = 1;
info.num_levels = 1;
info.sample_count = SDL_GPU_SAMPLECOUNT_1;
view->colour = rgpuCreateTexture(_scene.device, &info);
info.format = rgpuGetTextureFormatFromPixelFormat(SDL_PIXELFORMAT_BGRA32);
view->output = rgpuCreateTexture(_scene.device, &info);
info.format = _scene.depthFormat;
info.usage = SDL_GPU_TEXTUREUSAGE_DEPTH_STENCIL_TARGET;
view->depth = rgpuCreateTexture(_scene.device, &info);
if ((view->colour == NULL) || (view->output == NULL) || (view->depth == NULL)) {
utilTrace("Scene: view: %s", SDL_GetError());
_freeView(view);
return NO_HANDLE;
}
view->used = true;
return x;
}
bool viewSetCamera(int32_t view, int32_t camera) {
if (!viewValid(view) || ((camera != NO_HANDLE) && !nodeValid(camera))) {
return false;
}
_scene.views[view].camera = camera;
return true;
}
bool viewValid(int32_t view) {
return (view >= 0) && (view < MAX_VIEWS) && _scene.views[view].used;
}