/* * * Singe 3 * Copyright (C) 2006-2026 Scott Duensing * * This program is free software; you can redistribute it and/or * modify it under the terms of the GNU General Public License * as published by the Free Software Foundation; either version 3 * of the License, or (at your option) any later version. * * This program is distributed in the hope that it will be useful, * but WITHOUT ANY WARRANTY; without even the implied warranty of * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the * GNU General Public License for more details. * * You should have received a copy of the GNU General Public License * along with this program; if not, write to the Free Software * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA * 02110-1301, USA. * */ // Frame exact video playback on libavformat and libavcodec. // // A laserdisc game addresses video by frame number, so the player needs to know every frame's // presentation time before it plays anything. The first load of a file demuxes it once without // decoding, records each video packet's timestamp and keyframe flag, sorts them into display // order, and caches the table next to the game's other data. Seeking to frame N then means: // find the last keyframe at or before N, seek the demuxer there, and decode forward until the // frame whose timestamp matches N comes out. Decoding runs on a thread per player; audio is // demuxed and decoded separately on the main thread and fed to an SDL3_mixer track. #include #include #include #ifdef __linux__ #include #endif #include #include #include #include #include #include #include #include #include #include #include "../thirdparty/uthash/src/uthash.h" #include "util.h" #include "vfs.h" #include "videoPlayer.h" // This is kinda ugly but it lets us use the language // data from VLC without changing their files. #define VLC_API #define N_(x) x typedef struct iso639_lang_t iso639_lang_t; #include "../thirdparty/vlc/vlc_iso_lang.h" #include "../thirdparty/vlc/iso-639_def.h" #undef N_ #undef VLC_API #define AUDIO_CHANNELS_OUT 2 // Everything is mixed as stereo #define AUDIO_DRAIN_MS 400 // Long enough for any sane device queue to empty #define AUDIO_MEASURE_MAX_BUFFERS 64 #define AUDIO_MEASURE_TIMEOUT_MS 2000 #define AUDIO_STREAM_LOW_WATERMARK (24 * 1024) // Bytes queued for the mixer before we stop decoding ahead #define AVIO_BUFFER_BYTES (64 * 1024) // libavformat's read buffer over a vfs stream #define BYTES_PER_PIXEL 4 #define BYTES_PER_SAMPLE 4 // Float samples #define DEFAULT_FPS_NUMERATOR 30 #define DEFAULT_FPS_DENOMINATOR 1 #define ERROR_BUFFER_SIZE 1024 #define INDEX_MAGIC "SINGEIDX" #define INDEX_VERSION 1 #define KEYFRAME_WARN_SECONDS 2.0 // Seeks decode forward from the previous keyframe #define LANGUAGE_CODE_LENGTH 3 #define MS_PER_SECOND 1000.0 #define PERCENT_MAX 100 #define PERCENT_TO_SCALE 0.01f #define PLANE_COUNT 3 // Y, U, V #define SCALER_PLANES 4 // libswscale reads four plane pointers and strides whatever the format #define SEEK_FORWARD_LIMIT 64 // Decode forward rather than seek when the wanted frame is this close #define SEEK_RETRY_MAX 3 // Keyframes to back up when a seek lands past its target #define STRIDE_ALIGNMENT 64 // libswscale stores with aligned vector instructions; rows must start aligned // One video frame in display order. typedef struct FrameInfoS { int64_t pts; // In the video stream's time base bool keyframe; } FrameInfoT; // What the index cache file starts with. typedef struct IndexHeaderS { char magic[8]; int32_t version; int32_t streamIndex; int64_t fileSize; int64_t fileTime; int64_t count; } IndexHeaderT; // One decoded frame owned by the player. typedef struct FrameBufferS { uint8_t *data[SCALER_PLANES]; // Only PLANE_COUNT are ever allocated; the rest stay NULL for libswscale int32_t linesize[SCALER_PLANES]; int64_t frame; // -1 until something has been decoded into it } FrameBufferT; typedef struct AudioTrackS { int32_t streamIndex; char *language; } AudioTrackT; #pragma GCC diagnostic push #pragma GCC diagnostic ignored "-Wpadded" typedef struct VideoPlayerS { int32_t id; bool playing; bool resetTime; bool rgb; // BGRA frames for scripts to read; otherwise YUV for the GPU int32_t width; int32_t height; int32_t volumeLeft; int32_t volumeRight; AVRational fps; // Frame table, immutable after load. FrameInfoT *frames; int64_t frameCount; AVRational videoTimeBase; // Video demuxer and decoder. Owned by the decoder thread once it starts. AVFormatContext *videoFormat; AVCodecContext *videoCodec; AVPacket *videoPacket; AVFrame *videoFrame; struct SwsContext *sws; AVBufferRef *hwDevice; // Hardware decoder context, NULL when decoding in software AVFrame *hwFrame; // Decoded hardware frame transferred to system memory enum AVPixelFormat hwPixelFormat; // What the hardware decoder hands back int32_t videoStream; int64_t nextDecodeFrame; // Frame the decoder will produce next, -1 when unknown int64_t seekKeyframe; // Keyframe index of the last seek, for backing up bool videoDrained; bool packetPending; // videoPacket holds data the decoder refused (EAGAIN) bool hwReported; // First hardware frame has been traced // Audio demuxer, decoder, and resampler. Owned by the main thread. AVFormatContext *audioFormat; AVCodecContext *audioCodec; AVPacket *audioPacket; AVFrame *audioFrame; SwrContext *swr; AudioTrackT *audio; int32_t audioSourceCount; int32_t currentAudioTrack; bool audioEof; int64_t audioSkipUntilMs; // Samples before this time are dropped after a seek int64_t audioNextMs; // Where the next decoded audio frame is expected to start float *audioBuffer; int32_t audioBufferSamples; // Capacity, in sample frames SDL_AudioSpec audioSpec; // What the resampler produces SDL_AudioStream *audioStream; // Feeds the mixer track MIX_Track *track; // Clock, driven by what the mixer has pulled from the track. int64_t startTime; // Video time (ms) at the last play/seek uint64_t startTicks; // Wall clock (ms) at the last play/seek int64_t samplesPlayed; // Track frames mixed since the last reset, at trackRate int32_t trackRate; // Sample rate the mixer pulls this track at uint64_t lastCallbackTicks; bool audioClockValid; // A callback has run since the last reset int64_t frame; // Frame buffers and the decoder thread. FrameBufferT front; // What the texture and the script see FrameBufferT back; // What the decoder thread fills bool backReady; bool threadError; bool quitThread; int64_t requestedFrame; // Waiting for the decoder, -1 when none int64_t pendingFrame; // Last frame asked for, so it is not asked for twice int64_t uploadedFrame; // Frame currently on the texture SDL_Texture *videoTexture; SDL_Thread *thread; SDL_Mutex *lock; SDL_Condition *wake; char threadErrMsg[ERROR_BUFFER_SIZE]; UT_hash_handle hh; } VideoPlayerT; #pragma GCC diagnostic pop static void _allocateFrameBuffer(VideoPlayerT *v, FrameBufferT *buffer); static void _alsaSetQuiet(bool quiet); static int64_t _audioClock(VideoPlayerT *v, uint64_t now); static void _audioCloseTrack(VideoPlayerT *v); static void _audioQueueFrame(VideoPlayerT *v); static void _audioSeek(VideoPlayerT *v, int64_t ms); static void _audioSelectTrack(VideoPlayerT *v, int32_t track); static void _audioSetupResampler(VideoPlayerT *v); static int _avioRead(void *opaque, uint8_t *buffer, int size); static int64_t _avioSeek(void *opaque, int64_t offset, int whence); static void _buildFrameTable(VideoPlayerT *v, const char *filename, const char *indexPath); static int _compareFrames(const void *a, const void *b); // qsort callback. Not changing int. static void _convertFrame(VideoPlayerT *v, FrameBufferT *buffer, const AVFrame *frame); static bool _decodeFrame(VideoPlayerT *v, int64_t want); static int _decoderThread(void *data); // SDL thread entry. Not changing int. static void _feedAudio(VideoPlayerT *v); static int64_t _findFrameIndex(VideoPlayerT *v, int64_t pts); static void _formatClose(AVFormatContext **format); static AVFormatContext *_formatOpen(const char *filename); static int64_t _frameTime(VideoPlayerT *v, int64_t frame); static VideoPlayerT *_getPlayer(int32_t playerHandle, const char *caller); static char *_indexFileName(const char *filename, const char *indexPath); static void _loadAudio(VideoPlayerT *v, const char *filename); static void _measureDeviceQueue(void *udata, MIX_Mixer *mixer, const SDL_AudioSpec *spec, float *pcm, int32_t samples); static void _openHardware(VideoPlayerT *v, const AVCodec *decoder); static void _openVideo(VideoPlayerT *v, const char *filename); static bool _readIndexCache(VideoPlayerT *v, const char *indexName, const IndexHeaderT *expected); static void _reportKeyframes(VideoPlayerT *v, const char *filename); static void _requestFrame(VideoPlayerT *v); static void _resetClock(VideoPlayerT *v, uint64_t now); static void _seekVideo(VideoPlayerT *v, int64_t keyframe); static enum AVPixelFormat _selectPixelFormat(AVCodecContext *codec, const enum AVPixelFormat *formats); // libavcodec callback. static int64_t _streamTimeToMs(int64_t ts, AVRational timeBase); static bool _takeDecodedFrame(VideoPlayerT *v); static void _trackMixed(void *udata, MIX_Track *track, const SDL_AudioSpec *spec, float *pcm, int32_t samples); static void _uploadFrame(VideoPlayerT *v); static void _writeIndexCache(VideoPlayerT *v, const char *indexName, const IndexHeaderT *header); #ifdef __linux__ // libasound prints its own diagnostics to stderr. SDL loads it, so borrow its error hook. typedef void (*AlsaErrorHandlerT)(const char *file, int line, const char *function, int err, const char *fmt, ...); typedef int (*AlsaSetErrorHandlerT)(AlsaErrorHandlerT handler); static void _alsaQuiet(const char *file, int line, const char *function, int err, const char *fmt, ...); static void *_alsaLibrary = NULL; #endif static VideoPlayerT *_videoPlayerHash = NULL; static int32_t _nextId = 0; static MIX_Mixer *_mixer = NULL; static SDL_AudioSpec _mixSpec; // What the mixer feeds the device static int64_t _mixLatencyMs = 0; // Time between handing audio to the device and hearing it static int32_t _audioDelayMs = 0; // Per-game correction on top of the measured latency static int32_t _audioCalibrationMs = 0; // Per-machine correction, from the calibration screen static bool _hardwareDecoding = true; // Try the platform decoder before falling back to software // Startup measurement of the audio device queue (see videoInit). static bool _measuring = false; static int32_t _measureCount = 0; static int64_t _measureFrames = 0; static int64_t _measurePeriodMs = 0; static uint64_t _measureLastTicks = 0; // Planes are allocated the way libavcodec allocates its own: aligned start, aligned stride. // libswscale writes rows with aligned vector stores and faults on anything less. static void _allocateFrameBuffer(VideoPlayerT *v, FrameBufferT *buffer) { int32_t chromaWidth = (v->width + 1) / 2; int32_t chromaHeight = (v->height + 1) / 2; buffer->frame = -1; if (v->rgb) { buffer->linesize[0] = FFALIGN(v->width * BYTES_PER_PIXEL, STRIDE_ALIGNMENT); buffer->data[0] = av_malloc((size_t)buffer->linesize[0] * (size_t)v->height); if (!buffer->data[0]) { utilDie("Unable to allocate frame buffer."); } } else { buffer->linesize[0] = FFALIGN(v->width, STRIDE_ALIGNMENT); buffer->linesize[1] = FFALIGN(chromaWidth, STRIDE_ALIGNMENT); buffer->linesize[2] = buffer->linesize[1]; buffer->data[0] = av_malloc((size_t)buffer->linesize[0] * (size_t)v->height); buffer->data[1] = av_malloc((size_t)buffer->linesize[1] * (size_t)chromaHeight); buffer->data[2] = av_malloc((size_t)buffer->linesize[2] * (size_t)chromaHeight); if (!buffer->data[0] || !buffer->data[1] || !buffer->data[2]) { utilDie("Unable to allocate frame buffer."); } } } #ifdef __linux__ static void _alsaQuiet(const char *file, int line, const char *function, int err, const char *fmt, ...) { (void)file; (void)line; (void)function; (void)err; (void)fmt; } #endif // Silences libasound while the device is drained on purpose, so the expected underrun is not reported. static void _alsaSetQuiet(bool quiet) { #ifdef __linux__ AlsaSetErrorHandlerT setHandler = NULL; void *symbol = NULL; if (quiet) { _alsaLibrary = dlopen("libasound.so.2", RTLD_NOW); } if (_alsaLibrary) { symbol = dlsym(_alsaLibrary, "snd_lib_error_set_handler"); if (symbol) { memcpy(&setHandler, &symbol, sizeof(setHandler)); setHandler(quiet ? _alsaQuiet : NULL); } if (!quiet) { dlclose(_alsaLibrary); _alsaLibrary = NULL; } } #else (void)quiet; #endif } // Presentation time (ms) the listener is hearing right now. Audio is the master clock: // the picture is fitted to what the device has actually consumed, so the two cannot drift. static int64_t _audioClock(VideoPlayerT *v, uint64_t now) { int64_t played = 0; int32_t rate = 0; uint64_t last = 0; bool valid = false; int64_t clock = 0; MIX_LockMixer(_mixer); played = v->samplesPlayed; rate = v->trackRate; last = v->lastCallbackTicks; valid = v->audioClockValid; MIX_UnlockMixer(_mixer); if (valid && (rate > 0)) { // Consumed samples, interpolated since the last callback, less the device buffer still queued. clock = v->startTime + (played * (int64_t)MS_PER_SECOND / rate) + (int64_t)(now - last) - _mixLatencyMs - _audioDelayMs - _audioCalibrationMs; } else { // No callback yet: wall clock, offset the same way so the switch over is seamless. clock = v->startTime + (int64_t)(now - v->startTicks) - _mixLatencyMs - _audioDelayMs - _audioCalibrationMs; } if (clock < v->startTime) { clock = v->startTime; } return clock; } static void _audioCloseTrack(VideoPlayerT *v) { if (v->audioCodec) { avcodec_free_context(&v->audioCodec); } if (v->swr) { swr_free(&v->swr); } } // Resamples one decoded audio frame to the track's format and queues it, dropping whatever lies // before the seek target. static void _audioQueueFrame(VideoPlayerT *v) { AVStream *stream = v->audioFormat->streams[v->audio[v->currentAudioTrack].streamIndex]; int64_t ts = v->audioFrame->best_effort_timestamp; int64_t startMs = (ts == AV_NOPTS_VALUE) ? v->audioNextMs : _streamTimeToMs(ts, stream->time_base); int32_t capacity = swr_get_out_samples(v->swr, v->audioFrame->nb_samples); int32_t produced = 0; int32_t skip = 0; uint8_t *out = NULL; if (capacity < 0) { utilDie("Unable to size the audio resampler output."); } if (capacity > v->audioBufferSamples) { free(v->audioBuffer); v->audioBufferSamples = capacity; v->audioBuffer = malloc((size_t)capacity * AUDIO_CHANNELS_OUT * BYTES_PER_SAMPLE); if (!v->audioBuffer) { utilDie("Unable to allocate audio buffer."); } } out = (uint8_t *)v->audioBuffer; produced = swr_convert(v->swr, &out, capacity, (const uint8_t **)v->audioFrame->extended_data, v->audioFrame->nb_samples); if (produced < 0) { utilDie("Audio resampling failed."); } // After a seek the first frame usually starts before the target; trim it. if (startMs < v->audioSkipUntilMs) { skip = (int32_t)((v->audioSkipUntilMs - startMs) * v->audioSpec.freq / (int64_t)MS_PER_SECOND); if (skip > produced) { skip = produced; } } if (produced > skip) { if (!SDL_PutAudioStreamData(v->audioStream, v->audioBuffer + (size_t)skip * AUDIO_CHANNELS_OUT, (produced - skip) * AUDIO_CHANNELS_OUT * BYTES_PER_SAMPLE)) { utilDie("%s", SDL_GetError()); } } v->audioNextMs = startMs + (int64_t)produced * (int64_t)MS_PER_SECOND / v->audioSpec.freq; } // Positions the audio demuxer so the next samples queued are the ones for video time ms. static void _audioSeek(VideoPlayerT *v, int64_t ms) { AVStream *stream = v->audioFormat->streams[v->audio[v->currentAudioTrack].streamIndex]; int64_t ts = av_rescale(ms, stream->time_base.den, (int64_t)stream->time_base.num * (int64_t)MS_PER_SECOND); if (av_seek_frame(v->audioFormat, stream->index, ts, AVSEEK_FLAG_BACKWARD) < 0) { // Before the first packet, or an unseekable stream: start over. avformat_seek_file(v->audioFormat, stream->index, INT64_MIN, 0, INT64_MAX, 0); } avcodec_flush_buffers(v->audioCodec); _audioSetupResampler(v); if (v->audioPacket->data) { av_packet_unref(v->audioPacket); } v->audioSkipUntilMs = ms; v->audioNextMs = ms; v->audioEof = false; } // Opens the decoder for one audio track and points the mixer track's stream at its output. static void _audioSelectTrack(VideoPlayerT *v, int32_t track) { AVStream *stream = v->audioFormat->streams[v->audio[track].streamIndex]; const AVCodec *decoder = avcodec_find_decoder(stream->codecpar->codec_id); SDL_AudioSpec spec; SDL_AudioStream *fresh = NULL; int32_t x = 0; if (decoder == NULL) { utilDie("No decoder for audio track %d.", track); } _audioCloseTrack(v); v->audioCodec = avcodec_alloc_context3(decoder); if (!v->audioCodec || (avcodec_parameters_to_context(v->audioCodec, stream->codecpar) < 0) || (avcodec_open2(v->audioCodec, decoder, NULL) < 0)) { utilDie("Unable to open audio track %d.", track); } v->currentAudioTrack = track; // Only the chosen track needs demuxing. for (x = 0; x < (int32_t)v->audioFormat->nb_streams; x++) { v->audioFormat->streams[x]->discard = (x == stream->index) ? AVDISCARD_DEFAULT : AVDISCARD_ALL; } // The resampler always produces interleaved float stereo at the track's own rate. spec.format = SDL_AUDIO_F32; spec.channels = AUDIO_CHANNELS_OUT; spec.freq = v->audioCodec->sample_rate; if ((v->audioStream == NULL) || (spec.freq != v->audioSpec.freq)) { fresh = SDL_CreateAudioStream(&spec, &_mixSpec); if (!fresh) { utilDie("%s", SDL_GetError()); } if (!MIX_SetTrackAudioStream(v->track, fresh)) { utilDie("%s", SDL_GetError()); } if (v->audioStream) { SDL_DestroyAudioStream(v->audioStream); } v->audioStream = fresh; } v->audioSpec = spec; _audioSetupResampler(v); } // (Re)creates the resampler for the current track. Also the cheapest way to flush it after a seek. static void _audioSetupResampler(VideoPlayerT *v) { AVChannelLayout stereo = AV_CHANNEL_LAYOUT_STEREO; if (v->swr) { swr_free(&v->swr); } if ((swr_alloc_set_opts2(&v->swr, &stereo, AV_SAMPLE_FMT_FLT, v->audioSpec.freq, &v->audioCodec->ch_layout, v->audioCodec->sample_fmt, v->audioCodec->sample_rate, 0, NULL) < 0) || (swr_init(v->swr) < 0)) { utilDie("Unable to create the audio resampler."); } } // Loads the frame table from the cache, or demuxes the file once to build it and caches the result. static int _avioRead(void *opaque, uint8_t *buffer, int size) { int64_t got = vfsStreamRead((VfsStreamT *)opaque, buffer, size); return (got <= 0) ? AVERROR_EOF : (int)got; } static int64_t _avioSeek(void *opaque, int64_t offset, int whence) { VfsStreamT *stream = (VfsStreamT *)opaque; if (whence & AVSEEK_SIZE) { return vfsStreamSize(stream); } return vfsStreamSeek(stream, offset, whence & ~AVSEEK_FORCE); } static void _buildFrameTable(VideoPlayerT *v, const char *filename, const char *indexPath) { AVFormatContext *format = NULL; AVPacket *packet = NULL; AVStream *stream = NULL; FrameInfoT *grown = NULL; IndexHeaderT header; int64_t fileSize = 0; int64_t fileTime = 0; char *indexName = _indexFileName(filename, indexPath); int64_t capacity = 0; int64_t pts = 0; uint64_t started = SDL_GetTicks(); int64_t lastPts = AV_NOPTS_VALUE; int64_t duration = 1; int64_t x = 0; int64_t kept = 0; if (!vfsStat(filename, &fileSize, &fileTime)) { utilDie("Unable to stat %s", filename); } memset(&header, 0, sizeof(header)); memcpy(header.magic, INDEX_MAGIC, sizeof(header.magic)); header.version = INDEX_VERSION; header.streamIndex = v->videoStream; header.fileSize = fileSize; header.fileTime = fileTime; if (_readIndexCache(v, indexName, &header)) { free(indexName); return; } // Demux the whole file once. Every packet of the video stream is one frame. format = _formatOpen(filename); if (format == NULL) { utilDie("Unable to open %s for indexing.", filename); } if (avformat_find_stream_info(format, NULL) < 0) { utilDie("Unable to read stream information from %s.", filename); } stream = format->streams[v->videoStream]; for (x = 0; x < (int64_t)format->nb_streams; x++) { format->streams[x]->discard = (x == v->videoStream) ? AVDISCARD_DEFAULT : AVDISCARD_ALL; } packet = av_packet_alloc(); if (!packet) { utilDie("Unable to allocate a packet."); } if (stream->avg_frame_rate.num > 0) { duration = av_rescale_q(1, av_inv_q(stream->avg_frame_rate), stream->time_base); if (duration < 1) { duration = 1; } } while (av_read_frame(format, packet) >= 0) { if (packet->stream_index == v->videoStream) { if (v->frameCount == capacity) { capacity = (capacity == 0) ? 4096 : capacity * 2; grown = realloc(v->frames, sizeof(FrameInfoT) * (size_t)capacity); if (!grown) { utilDie("Unable to allocate the frame table."); } v->frames = grown; } // Prefer the presentation time; elementary streams may only have decode times or nothing. pts = (packet->pts != AV_NOPTS_VALUE) ? packet->pts : packet->dts; if (pts == AV_NOPTS_VALUE) { pts = (lastPts == AV_NOPTS_VALUE) ? 0 : lastPts + duration; } lastPts = pts; v->frames[v->frameCount].pts = pts; v->frames[v->frameCount].keyframe = (packet->flags & AV_PKT_FLAG_KEY) != 0; v->frameCount++; } av_packet_unref(packet); } av_packet_free(&packet); _formatClose(&format); if (v->frameCount == 0) { utilDie("%s has no video frames.", filename); } // Display order, one entry per timestamp. qsort(v->frames, (size_t)v->frameCount, sizeof(FrameInfoT), _compareFrames); for (x = 0; x < v->frameCount; x++) { if ((kept == 0) || (v->frames[x].pts != v->frames[kept - 1].pts)) { v->frames[kept++] = v->frames[x]; } else if (v->frames[x].keyframe) { v->frames[kept - 1].keyframe = true; } } v->frameCount = kept; header.count = kept; _writeIndexCache(v, indexName, &header); utilTrace("Indexed %s: %" PRId64 " frames in %" PRIu64 " ms", filename, kept, SDL_GetTicks() - started); free(indexName); } static int _compareFrames(const void *a, const void *b) { const FrameInfoT *fa = (const FrameInfoT *)a; const FrameInfoT *fb = (const FrameInfoT *)b; return (fa->pts > fb->pts) - (fa->pts < fb->pts); } // Puts a decoded frame into our buffer in the player's format, converting only when the decoder's // output differs (10 bit sources, 4:2:2, hardware formats). static void _convertFrame(VideoPlayerT *v, FrameBufferT *buffer, const AVFrame *frame) { enum AVPixelFormat wanted = v->rgb ? AV_PIX_FMT_BGRA : AV_PIX_FMT_YUV420P; int32_t planes = v->rgb ? 1 : PLANE_COUNT; int32_t plane = 0; int32_t rows = 0; int32_t bytes = 0; if ((frame->format == wanted) || (!v->rgb && (frame->format == AV_PIX_FMT_YUVJ420P))) { for (plane = 0; plane < planes; plane++) { rows = (plane == 0) ? v->height : (v->height + 1) / 2; bytes = v->rgb ? v->width * BYTES_PER_PIXEL : ((plane == 0) ? v->width : (v->width + 1) / 2); av_image_copy_plane(buffer->data[plane], buffer->linesize[plane], frame->data[plane], frame->linesize[plane], bytes, rows); } } else { v->sws = sws_getCachedContext(v->sws, frame->width, frame->height, (enum AVPixelFormat)frame->format, v->width, v->height, wanted, SWS_BILINEAR, NULL, NULL, NULL); if (!v->sws) { utilDie("Unable to convert video frames."); } sws_scale(v->sws, (const uint8_t *const *)frame->data, frame->linesize, 0, frame->height, buffer->data, buffer->linesize); } } // Decodes frame "want" into the back buffer. Runs on the decoder thread. static bool _decodeFrame(VideoPlayerT *v, int64_t want) { int64_t keyframe = want; int64_t index = 0; int64_t ts = 0; int32_t retries = 0; int32_t result = 0; // Decide between decoding forward and seeking. while ((keyframe > 0) && !v->frames[keyframe].keyframe) { keyframe--; } if (v->videoDrained || (v->nextDecodeFrame < 0) || (want < v->nextDecodeFrame) || (keyframe > v->nextDecodeFrame) || (want - v->nextDecodeFrame > SEEK_FORWARD_LIMIT)) { _seekVideo(v, keyframe); } for (;;) { result = avcodec_receive_frame(v->videoCodec, v->videoFrame); if (result == 0) { ts = v->videoFrame->best_effort_timestamp; index = (ts == AV_NOPTS_VALUE) ? v->nextDecodeFrame : _findFrameIndex(v, ts); if (index == want) { if (v->hwDevice && (v->videoFrame->format == v->hwPixelFormat)) { // Pull the picture out of the GPU; it arrives as NV12 and is converted like any other format. if (av_hwframe_transfer_data(v->hwFrame, v->videoFrame, 0) < 0) { snprintf(v->threadErrMsg, sizeof(v->threadErrMsg), "Unable to read back a hardware decoded frame."); v->threadError = true; av_frame_unref(v->videoFrame); return false; } if (!v->hwReported) { v->hwReported = true; utilTrace("Video %d: first hardware frame read back as %s %dx%d", v->id, av_get_pix_fmt_name((enum AVPixelFormat)v->hwFrame->format), v->hwFrame->width, v->hwFrame->height); } _convertFrame(v, &v->back, v->hwFrame); av_frame_unref(v->hwFrame); } else { _convertFrame(v, &v->back, v->videoFrame); } av_frame_unref(v->videoFrame); v->nextDecodeFrame = want + 1; return true; } if (index > want) { av_frame_unref(v->videoFrame); // The seek landed past the target (sparse cues); back up one keyframe and try again. if ((retries < SEEK_RETRY_MAX) && (v->seekKeyframe > 0)) { retries++; keyframe = v->seekKeyframe - 1; while ((keyframe > 0) && !v->frames[keyframe].keyframe) { keyframe--; } _seekVideo(v, keyframe); continue; } // Best we can do: show what we have. v->nextDecodeFrame = index + 1; return false; } // Earlier than wanted: keep going. av_frame_unref(v->videoFrame); v->nextDecodeFrame = index + 1; continue; } if (result == AVERROR_EOF) { v->videoDrained = true; return false; } if (result != AVERROR(EAGAIN)) { av_strerror(result, v->threadErrMsg, sizeof(v->threadErrMsg)); v->threadError = true; return false; } // The decoder wants data. if (!v->packetPending) { result = av_read_frame(v->videoFormat, v->videoPacket); if (result < 0) { // End of file: drain the decoder. avcodec_send_packet(v->videoCodec, NULL); continue; } if (v->videoPacket->stream_index != v->videoStream) { av_packet_unref(v->videoPacket); continue; } v->packetPending = true; } result = avcodec_send_packet(v->videoCodec, v->videoPacket); if (result == 0) { av_packet_unref(v->videoPacket); v->packetPending = false; } else if (result != AVERROR(EAGAIN)) { av_strerror(result, v->threadErrMsg, sizeof(v->threadErrMsg)); v->threadError = true; av_packet_unref(v->videoPacket); v->packetPending = false; return false; } } } // Decodes whatever frame was last requested into the back buffer, then waits for the next request. // Seeks decode forward from a keyframe, so this keeps long seeks off the main loop. static int _decoderThread(void *data) { VideoPlayerT *v = (VideoPlayerT *)data; int64_t want = -1; bool got = false; SDL_LockMutex(v->lock); while (!v->quitThread) { // Only one frame is in flight and the main thread must take it before the next one. if ((v->requestedFrame < 0) || v->backReady) { SDL_WaitCondition(v->wake, v->lock); continue; } want = v->requestedFrame; v->requestedFrame = -1; SDL_UnlockMutex(v->lock); got = _decodeFrame(v, want); SDL_LockMutex(v->lock); if (got) { v->back.frame = want; v->backReady = true; } else if (!v->threadError) { // Nothing decodable there (past the end, or a seek that could not be satisfied): let the // main thread ask again for whatever frame it wants next rather than hanging on this one. v->pendingFrame = -1; } } SDL_UnlockMutex(v->lock); return 0; } // Keeps the mixer track's stream topped up from the current audio track. static void _feedAudio(VideoPlayerT *v) { int32_t result = 0; while (!v->audioEof && (SDL_GetAudioStreamQueued(v->audioStream) < AUDIO_STREAM_LOW_WATERMARK)) { result = avcodec_receive_frame(v->audioCodec, v->audioFrame); if (result == 0) { _audioQueueFrame(v); av_frame_unref(v->audioFrame); continue; } if (result == AVERROR_EOF) { v->audioEof = true; break; } if (result != AVERROR(EAGAIN)) { utilDie("Audio decoding failed."); } result = av_read_frame(v->audioFormat, v->audioPacket); if (result < 0) { avcodec_send_packet(v->audioCodec, NULL); continue; } if (v->audioPacket->stream_index == v->audio[v->currentAudioTrack].streamIndex) { if (avcodec_send_packet(v->audioCodec, v->audioPacket) < 0) { utilDie("Audio decoding failed."); } } av_packet_unref(v->audioPacket); } } // Index of the last frame whose timestamp is at or before pts. static int64_t _findFrameIndex(VideoPlayerT *v, int64_t pts) { int64_t low = 0; int64_t high = v->frameCount - 1; int64_t mid = 0; if (pts <= v->frames[0].pts) { return 0; } while (low < high) { mid = (low + high + 1) / 2; if (v->frames[mid].pts <= pts) { low = mid; } else { high = mid - 1; } } return low; } // Closes a demuxer opened by _formatOpen along with the stream behind it. static void _formatClose(AVFormatContext **format) { AVIOContext *io = NULL; VfsStreamT *stream = NULL; if (*format == NULL) { return; } io = (*format)->pb; avformat_close_input(format); if (io != NULL) { stream = (VfsStreamT *)io->opaque; av_freep(&io->buffer); avio_context_free(&io); } vfsStreamClose(stream); } // Opens a demuxer over a vfs stream, so packed and loose videos read alike. NULL when the name resolves nowhere. static AVFormatContext *_formatOpen(const char *filename) { VfsStreamT *stream = vfsStreamOpen(filename); AVFormatContext *format = NULL; AVIOContext *io = NULL; unsigned char *buffer = NULL; if (stream == NULL) { return NULL; } buffer = (unsigned char *)av_malloc(AVIO_BUFFER_BYTES); io = avio_alloc_context(buffer, AVIO_BUFFER_BYTES, 0, stream, _avioRead, NULL, _avioSeek); format = avformat_alloc_context(); format->pb = io; format->flags |= AVFMT_FLAG_CUSTOM_IO; if (avformat_open_input(&format, filename, NULL, NULL) < 0) { av_freep(&io->buffer); avio_context_free(&io); vfsStreamClose(stream); return NULL; } return format; } // Presentation time of a frame in milliseconds. static int64_t _frameTime(VideoPlayerT *v, int64_t frame) { return _streamTimeToMs(v->frames[frame].pts, v->videoTimeBase); } static VideoPlayerT *_getPlayer(int32_t playerHandle, const char *caller) { VideoPlayerT *v = NULL; HASH_FIND_INT(_videoPlayerHash, &playerHandle, v); if (!v) { utilDie("No video player at index %d in %s.", playerHandle, caller); } return v; } // Returns a new string the caller must free. static char *_indexFileName(const char *filename, const char *indexPath) { char *name = utilCreateString("%s%c%s.index", indexPath, utilGetPathSeparator(), utilGetLastPathComponent(filename)); utilFixPathSeparators(&name, false); return name; } // Opens the audio side of a file (the video file, or the separate audio file of an old framefile) and // lists its tracks. static void _loadAudio(VideoPlayerT *v, const char *filename) { AVDictionaryEntry *tag = NULL; int32_t x = 0; int32_t count = 0; v->audioFormat = _formatOpen(filename); if (v->audioFormat == NULL) { utilDie("Unable to open %s for audio.", filename); } if (avformat_find_stream_info(v->audioFormat, NULL) < 0) { utilDie("Unable to read stream information from %s.", filename); } for (x = 0; x < (int32_t)v->audioFormat->nb_streams; x++) { if (v->audioFormat->streams[x]->codecpar->codec_type == AVMEDIA_TYPE_AUDIO) { count++; } } if (count == 0) { _formatClose(&v->audioFormat); return; } v->audio = (AudioTrackT *)calloc((size_t)count, sizeof(AudioTrackT)); if (!v->audio) { utilDie("Unable to allocate audio tracks."); } for (x = 0; x < (int32_t)v->audioFormat->nb_streams; x++) { if (v->audioFormat->streams[x]->codecpar->codec_type == AVMEDIA_TYPE_AUDIO) { v->audio[v->audioSourceCount].streamIndex = x; tag = av_dict_get(v->audioFormat->streams[x]->metadata, "language", NULL, 0); if (tag != NULL) { v->audio[v->audioSourceCount].language = strdup(tag->value); } v->audioSourceCount++; } } v->audioPacket = av_packet_alloc(); v->audioFrame = av_frame_alloc(); if (!v->audioPacket || !v->audioFrame) { utilDie("Unable to allocate audio decoding buffers."); } } // Post-mix hook used once at startup. After the mixer has been stalled long enough for the device // to drain, SDL refills the device as fast as it accepts data: a burst of callbacks a few // microseconds apart, then one per chunk period. The burst is the queue depth between the mixer // and the speaker, which is what the audio clock has to subtract. static void _measureDeviceQueue(void *udata, MIX_Mixer *mixer, const SDL_AudioSpec *spec, float *pcm, int32_t samples) { uint64_t now = SDL_GetTicks(); int64_t frames = samples / spec->channels; (void)udata; (void)mixer; (void)pcm; if (!_measuring) { return; } if (_measureCount == 0) { _measurePeriodMs = frames * (int64_t)MS_PER_SECOND / spec->freq; } else if ((int64_t)(now - _measureLastTicks) >= _measurePeriodMs / 2) { // First callback paced by the device: the burst is over. _measuring = false; return; } _measureCount++; _measureFrames += frames; _measureLastTicks = now; } // Attaches the platform's hardware decoder to the codec context when it offers one for this codec. // Failure of any step leaves the context decoding in software. static void _openHardware(VideoPlayerT *v, const AVCodec *decoder) { static const enum AVHWDeviceType wanted[] = { #if defined(_WIN32) AV_HWDEVICE_TYPE_D3D11VA, AV_HWDEVICE_TYPE_DXVA2, #elif defined(__APPLE__) AV_HWDEVICE_TYPE_VIDEOTOOLBOX, #else AV_HWDEVICE_TYPE_VAAPI, AV_HWDEVICE_TYPE_VDPAU, #endif AV_HWDEVICE_TYPE_NONE }; const AVCodecHWConfig *config = NULL; int32_t x = 0; int32_t i = 0; if (!_hardwareDecoding) { return; } for (x = 0; wanted[x] != AV_HWDEVICE_TYPE_NONE; x++) { for (i = 0; (config = avcodec_get_hw_config(decoder, i)) != NULL; i++) { if ((config->methods & AV_CODEC_HW_CONFIG_METHOD_HW_DEVICE_CTX) && (config->device_type == wanted[x])) { if (av_hwdevice_ctx_create(&v->hwDevice, wanted[x], NULL, NULL, 0) < 0) { utilTrace("Hardware decoding via %s is not available; decoding in software.", av_hwdevice_get_type_name(wanted[x])); break; } v->hwPixelFormat = config->pix_fmt; v->videoCodec->hw_device_ctx = av_buffer_ref(v->hwDevice); v->videoCodec->get_format = _selectPixelFormat; v->videoCodec->opaque = v; v->hwFrame = av_frame_alloc(); if (!v->hwFrame) { utilDie("Unable to allocate the hardware frame."); } // The GPU does the work; frame threads would only add latency to every seek. v->videoCodec->thread_count = 1; utilTrace("Hardware decoding via %s.", av_hwdevice_get_type_name(wanted[x])); return; } } } } // Opens the demuxer and decoder the decoder thread will use. static void _openVideo(VideoPlayerT *v, const char *filename) { const AVCodec *decoder = NULL; AVStream *stream = NULL; int32_t x = 0; v->videoFormat = _formatOpen(filename); if (v->videoFormat == NULL) { utilDie("Unable to open %s.", filename); } if (avformat_find_stream_info(v->videoFormat, NULL) < 0) { utilDie("Unable to read stream information from %s.", filename); } v->videoStream = av_find_best_stream(v->videoFormat, AVMEDIA_TYPE_VIDEO, -1, -1, &decoder, 0); if ((v->videoStream < 0) || (decoder == NULL)) { utilDie("%s has no decodable video.", filename); } for (x = 0; x < (int32_t)v->videoFormat->nb_streams; x++) { v->videoFormat->streams[x]->discard = (x == v->videoStream) ? AVDISCARD_DEFAULT : AVDISCARD_ALL; } stream = v->videoFormat->streams[v->videoStream]; v->videoTimeBase = stream->time_base; v->width = stream->codecpar->width; v->height = stream->codecpar->height; v->fps = av_guess_frame_rate(v->videoFormat, stream, NULL); if ((v->fps.num <= 0) || (v->fps.den <= 0)) { v->fps.num = DEFAULT_FPS_NUMERATOR; v->fps.den = DEFAULT_FPS_DENOMINATOR; } v->videoCodec = avcodec_alloc_context3(decoder); if (!v->videoCodec || (avcodec_parameters_to_context(v->videoCodec, stream->codecpar) < 0)) { utilDie("Unable to set up the video decoder for %s.", filename); } v->videoCodec->thread_count = 0; // Let libavcodec pick _openHardware(v, decoder); if (avcodec_open2(v->videoCodec, decoder, NULL) < 0) { utilDie("Unable to open the video decoder for %s.", filename); } v->videoPacket = av_packet_alloc(); v->videoFrame = av_frame_alloc(); if (!v->videoPacket || !v->videoFrame) { utilDie("Unable to allocate video decoding buffers."); } v->nextDecodeFrame = -1; v->seekKeyframe = 0; } // Loads a cached frame table if it matches this file. Returns false when the table must be rebuilt. static bool _readIndexCache(VideoPlayerT *v, const char *indexName, const IndexHeaderT *expected) { size_t bytes = 0; char *data = utilReadFile(indexName, &bytes); IndexHeaderT header; bool valid = false; if (data && (bytes >= sizeof(IndexHeaderT))) { memcpy(&header, data, sizeof(header)); valid = (memcmp(header.magic, expected->magic, sizeof(header.magic)) == 0) && (header.version == expected->version) && (header.streamIndex == expected->streamIndex) && (header.fileSize == expected->fileSize) && (header.fileTime == expected->fileTime) && (header.count > 0) && (bytes == sizeof(IndexHeaderT) + sizeof(FrameInfoT) * (size_t)header.count); } if (valid) { v->frameCount = header.count; v->frames = malloc(sizeof(FrameInfoT) * (size_t)v->frameCount); if (!v->frames) { utilDie("Unable to allocate the frame table."); } memcpy(v->frames, data + sizeof(IndexHeaderT), sizeof(FrameInfoT) * (size_t)v->frameCount); } free(data); return valid; } // Tells the author how far a seek may have to decode. Keyframe spacing is a property of the file. static void _reportKeyframes(VideoPlayerT *v, const char *filename) { int64_t frame = 0; int64_t lastKeyframe = 0; int64_t keyframes = 0; int64_t longestGap = 0; double seconds = 0.0; for (frame = 0; frame < v->frameCount; frame++) { if (v->frames[frame].keyframe) { if ((keyframes > 0) && (frame - lastKeyframe > longestGap)) { longestGap = frame - lastKeyframe; } lastKeyframe = frame; keyframes++; } } if (v->frameCount - lastKeyframe > longestGap) { longestGap = v->frameCount - lastKeyframe; } seconds = (double)longestGap * (double)v->fps.den / (double)v->fps.num; utilTrace("%s: %" PRId64 " frames, %" PRId64 " keyframes, longest gap %" PRId64 " frames (%.1f seconds)", filename, v->frameCount, keyframes, longestGap, seconds); if (seconds > KEYFRAME_WARN_SECONDS) { utilTrace("Warning: %s has keyframes up to %.1f seconds apart; seeking into that video may stall. Re-encode with a keyframe interval of two seconds or less.", filename, seconds); } } // Asks the decoder thread for the frame the clock says we should be showing, if it is not already coming. static void _requestFrame(VideoPlayerT *v) { SDL_LockMutex(v->lock); if ((v->frame != v->front.frame) && (v->frame != v->pendingFrame)) { v->requestedFrame = v->frame; v->pendingFrame = v->frame; SDL_SignalCondition(v->wake); } SDL_UnlockMutex(v->lock); } // Restart the presentation clock at the current frame and realign audio to it. static void _resetClock(VideoPlayerT *v, uint64_t now) { v->startTicks = now; v->startTime = _frameTime(v, v->frame); v->resetTime = false; if (v->audioSourceCount > 0) { MIX_LockMixer(_mixer); SDL_ClearAudioStream(v->audioStream); v->samplesPlayed = 0; v->lastCallbackTicks = now; v->audioClockValid = false; MIX_UnlockMixer(_mixer); // Video and audio share the container's timeline, so the frame's time is the audio's time. _audioSeek(v, v->startTime); } } // Positions the video demuxer at a keyframe and resets the decoder. Decoder thread only. static void _seekVideo(VideoPlayerT *v, int64_t keyframe) { if (av_seek_frame(v->videoFormat, v->videoStream, v->frames[keyframe].pts, AVSEEK_FLAG_BACKWARD) < 0) { avformat_seek_file(v->videoFormat, v->videoStream, INT64_MIN, 0, INT64_MAX, 0); keyframe = 0; } avcodec_flush_buffers(v->videoCodec); if (v->packetPending) { av_packet_unref(v->videoPacket); v->packetPending = false; } // Streams without timestamps are counted from here; the rest are matched by timestamp. v->nextDecodeFrame = keyframe; v->seekKeyframe = keyframe; v->videoDrained = false; } // libavcodec asks which of the formats it can produce we want; take the hardware one when offered. static enum AVPixelFormat _selectPixelFormat(AVCodecContext *codec, const enum AVPixelFormat *formats) { VideoPlayerT *v = (VideoPlayerT *)codec->opaque; int32_t x = 0; for (x = 0; formats[x] != AV_PIX_FMT_NONE; x++) { if (formats[x] == v->hwPixelFormat) { return formats[x]; } } // The hardware declined this stream: fall back to the decoder's first software format. utilTrace("Video %d: hardware decoder declined the stream; decoding in software.", v->id); return formats[0]; } static int64_t _streamTimeToMs(int64_t ts, AVRational timeBase) { return av_rescale(ts, (int64_t)timeBase.num * (int64_t)MS_PER_SECOND, timeBase.den); } // Swaps in a finished frame from the decoder thread. Returns true when there is a new one. static bool _takeDecodedFrame(VideoPlayerT *v) { FrameBufferT temp; bool taken = false; SDL_LockMutex(v->lock); if (v->threadError) { utilDie("Video decoding failed: %s", v->threadErrMsg); } if (v->backReady) { temp = v->front; v->front = v->back; v->back = temp; v->backReady = false; taken = true; SDL_SignalCondition(v->wake); } SDL_UnlockMutex(v->lock); return taken; } // Runs on the mixer thread each time this track's audio is pulled. Feeds the audio clock. static void _trackMixed(void *udata, MIX_Track *track, const SDL_AudioSpec *spec, float *pcm, int32_t samples) { VideoPlayerT *v = (VideoPlayerT *)udata; (void)track; (void)pcm; // The mixer pulls a track in the track's own rate and resamples afterwards, so count in that rate. if (v->trackRate != spec->freq) { utilTrace("Video %d audio is pulled at %d Hz, %d channels", v->id, spec->freq, spec->channels); } v->samplesPlayed += samples / spec->channels; v->trackRate = spec->freq; v->lastCallbackTicks = SDL_GetTicks(); v->audioClockValid = true; } // Pushes the front buffer to the texture. YUV players convert on the GPU. static void _uploadFrame(VideoPlayerT *v) { if (v->rgb) { SDL_UpdateTexture(v->videoTexture, NULL, v->front.data[0], v->front.linesize[0]); } else { SDL_UpdateYUVTexture(v->videoTexture, NULL, v->front.data[0], v->front.linesize[0], v->front.data[1], v->front.linesize[1], v->front.data[2], v->front.linesize[2]); } v->uploadedFrame = v->front.frame; } static void _writeIndexCache(VideoPlayerT *v, const char *indexName, const IndexHeaderT *header) { FILE *out = fopen(indexName, "wb"); if (!out) { utilSay("Warning: Unable to write %s; the video will be indexed again next time.", indexName); return; } if ((fwrite(header, sizeof(IndexHeaderT), 1, out) != 1) || (fwrite(v->frames, sizeof(FrameInfoT), (size_t)v->frameCount, out) != (size_t)v->frameCount)) { fclose(out); unlink(indexName); utilSay("Warning: Unable to write %s; the video will be indexed again next time.", indexName); return; } fclose(out); } int32_t videoGetAudioCalibration(void) { return _audioCalibrationMs; } int32_t videoGetAudioDelay(void) { return _audioDelayMs; } // The device queue measured at startup, in milliseconds. int32_t videoGetAudioLatency(void) { return (int32_t)_mixLatencyMs; } int32_t videoGetAudioTrack(int32_t playerHandle) { return _getPlayer(playerHandle, "videoGetAudioTrack")->currentAudioTrack; } int32_t videoGetAudioTracks(int32_t playerHandle) { return _getPlayer(playerHandle, "videoGetAudioTracks")->audioSourceCount; } int64_t videoGetFrame(int32_t playerHandle) { return _getPlayer(playerHandle, "videoGetFrame")->frame; } int64_t videoGetFrameCount(int32_t playerHandle) { return _getPlayer(playerHandle, "videoGetFrameCount")->frameCount; } int32_t videoGetHeight(int32_t playerHandle) { return _getPlayer(playerHandle, "videoGetHeight")->height; } const char *videoGetLanguage(int32_t playerHandle, int32_t audioTrack) { VideoPlayerT *v = _getPlayer(playerHandle, "videoGetLanguage"); const char *r = "unk"; // Unknown Language if ((audioTrack < 0) || (audioTrack >= v->audioSourceCount)) { utilDie("Invalid audio track %d in videoGetLanguage.", audioTrack); } if ((v->audio[audioTrack].language != NULL) && (strlen(v->audio[audioTrack].language) == LANGUAGE_CODE_LENGTH)) { r = v->audio[audioTrack].language; } return r; } const char *videoGetLanguageDescription(const char *languageCode) { int32_t i = 0; if (languageCode == NULL) { return "Unknown"; } for (i = 0; p_languages[i].psz_eng_name != NULL; i++) { if ((utilStricmp(languageCode, p_languages[i].psz_iso639_1) == 0) || (utilStricmp(languageCode, p_languages[i].psz_iso639_2T) == 0) || (utilStricmp(languageCode, p_languages[i].psz_iso639_2B) == 0)) { return p_languages[i].psz_eng_name; } } return "Unknown"; } // The mixer every sound and video goes through. MIX_Mixer *videoGetMixer(void) { return _mixer; } // Reads one pixel of the frame being shown. Returns false if there is no frame yet. bool videoGetPixel(int32_t playerHandle, int32_t x, int32_t y, uint8_t *r, uint8_t *g, uint8_t *b) { VideoPlayerT *v = _getPlayer(playerHandle, "videoGetPixel"); const uint8_t *pixel = NULL; int32_t c = 0; int32_t d = 0; int32_t e = 0; int32_t value = 0; if ((v->front.frame < 0) || (x < 0) || (y < 0) || (x >= v->width) || (y >= v->height)) { return false; } if (v->rgb) { pixel = v->front.data[0] + (y * v->front.linesize[0]) + (x * BYTES_PER_PIXEL); *b = pixel[0]; *g = pixel[1]; *r = pixel[2]; } else { // BT.601 limited range, the same conversion SDL applies on the GPU for SD video. c = v->front.data[0][y * v->front.linesize[0] + x] - 16; d = v->front.data[1][(y / 2) * v->front.linesize[1] + (x / 2)] - 128; e = v->front.data[2][(y / 2) * v->front.linesize[2] + (x / 2)] - 128; value = (298 * c + 409 * e + 128) >> 8; *r = (uint8_t)((value < 0) ? 0 : ((value > 255) ? 255 : value)); value = (298 * c - 100 * d - 208 * e + 128) >> 8; *g = (uint8_t)((value < 0) ? 0 : ((value > 255) ? 255 : value)); value = (298 * c + 516 * d + 128) >> 8; *b = (uint8_t)((value < 0) ? 0 : ((value > 255) ? 255 : value)); } return true; } // Exposes the BGRA frame being shown (players loaded with rgb). Valid until the next videoUpdate of this player. bool videoGetPixels(int32_t playerHandle, const uint8_t **pixels, int32_t *pitch) { VideoPlayerT *v = _getPlayer(playerHandle, "videoGetPixels"); if (!v->rgb || (v->front.frame < 0)) { return false; } *pixels = v->front.data[0]; *pitch = v->front.linesize[0]; return true; } void videoGetVolume(int32_t playerHandle, int32_t *leftPercent, int32_t *rightPercent) { VideoPlayerT *v = _getPlayer(playerHandle, "videoGetVolume"); if (leftPercent != NULL) { *leftPercent = v->volumeLeft; } if (rightPercent != NULL) { *rightPercent = v->volumeRight; } } int32_t videoGetWidth(int32_t playerHandle) { return _getPlayer(playerHandle, "videoGetWidth")->width; } void videoInit(MIX_Mixer *mixer) { uint64_t started = 0; // Fetch mixer settings _mixer = mixer; if (!MIX_GetMixerFormat(_mixer, &_mixSpec)) { utilDie("%s", SDL_GetError()); } // Measure the device queue. SDL cannot report how much audio sits between the mixer and the // speaker, but it refills an empty device in a burst, and the burst is that amount. Nothing // but silence is playing yet, so draining the device is inaudible. _alsaSetQuiet(true); MIX_SetPostMixCallback(_mixer, _measureDeviceQueue, NULL); MIX_LockMixer(_mixer); _measuring = true; _measureCount = 0; _measureFrames = 0; SDL_Delay(AUDIO_DRAIN_MS); MIX_UnlockMixer(_mixer); started = SDL_GetTicks(); while (_measuring && ((SDL_GetTicks() - started) < AUDIO_MEASURE_TIMEOUT_MS)) { SDL_Delay(1); } MIX_SetPostMixCallback(_mixer, NULL, NULL); _alsaSetQuiet(false); if (_measuring || (_measureCount < 1) || (_measureCount > AUDIO_MEASURE_MAX_BUFFERS)) { // Unreadable result: assume the two buffers a double-buffered device holds. _measuring = false; _measureFrames = (int64_t)(2.0 * (double)_measurePeriodMs * (double)_mixSpec.freq / MS_PER_SECOND); _measureCount = 2; } _mixLatencyMs = _measureFrames * (int64_t)MS_PER_SECOND / _mixSpec.freq; utilTrace("Audio device queue: %d buffers, %" PRId64 " frames, %" PRId64 " ms at %d Hz", _measureCount, _measureFrames, _mixLatencyMs, _mixSpec.freq); } bool videoIsPlaying(int32_t playerHandle) { return _getPlayer(playerHandle, "videoIsPlaying")->playing; } // The mixer's audio thread runs the sound and video callbacks; hold this to read what they write. void videoLockAudio(void) { MIX_LockMixer(_mixer); } // audioFilename may be NULL when the audio lives in the video file. rgb players decode to BGRA so // scripts can read the pixels; everything else stays YUV and is converted by the GPU. int32_t videoLoad(const char *videoFilename, const char *audioFilename, const char *indexPath, SDL_Renderer *renderer, bool rgb) { VideoPlayerT *v = NULL; SDL_PropertiesID playProps = 0; // Create new videoPlayer v = calloc(1, sizeof(VideoPlayerT)); if (!v) { utilDie("Unable to allocate new video player."); } // Set some starting values (everything else is zero from calloc) v->currentAudioTrack = -1; v->videoStream = -1; v->volumeLeft = VIDEO_VOLUME_MAX; v->volumeRight = VIDEO_VOLUME_MAX; v->rgb = rgb; v->requestedFrame = -1; v->pendingFrame = -1; v->uploadedFrame = -1; v->front.frame = -1; v->back.frame = -1; // Video: demuxer, frame table, decoder. _openVideo(v, videoFilename); _buildFrameTable(v, videoFilename, indexPath); _reportKeyframes(v, videoFilename); // Create video texture v->videoTexture = SDL_CreateTexture(renderer, rgb ? SDL_PIXELFORMAT_BGRA32 : SDL_PIXELFORMAT_IYUV, SDL_TEXTUREACCESS_STREAMING, v->width, v->height); if (v->videoTexture == NULL) { utilDie("%s", SDL_GetError()); } // Video textures scale smoothly. SDL_SetTextureScaleMode(v->videoTexture, SDL_SCALEMODE_LINEAR); // Frame buffers and the decoder thread that fills them _allocateFrameBuffer(v, &v->front); _allocateFrameBuffer(v, &v->back); v->lock = SDL_CreateMutex(); v->wake = SDL_CreateCondition(); if (!v->lock || !v->wake) { utilDie("%s", SDL_GetError()); } v->thread = SDL_CreateThread(_decoderThread, "singeDecoder", v); if (v->thread == NULL) { utilDie("%s", SDL_GetError()); } // Audio: a mixer track that never halts, fed from its own demuxer on the main thread. _loadAudio(v, audioFilename ? audioFilename : videoFilename); if (v->audioSourceCount > 0) { v->track = MIX_CreateTrack(_mixer); if (!v->track) { utilDie("%s", SDL_GetError()); } _audioSelectTrack(v, 0); if (!MIX_SetTrackRawCallback(v->track, _trackMixed, v)) { utilDie("%s", SDL_GetError()); } playProps = SDL_CreateProperties(); SDL_SetBooleanProperty(playProps, MIX_PROP_PLAY_HALT_WHEN_EXHAUSTED_BOOLEAN, false); if (!MIX_PlayTrack(v->track, playProps)) { utilDie("%s", SDL_GetError()); } SDL_DestroyProperties(playProps); // Paused until the video plays. MIX_PauseTrack(v->track); } // Add to player hash v->id = _nextId++; HASH_ADD_INT(_videoPlayerHash, id, v); return v->id; } void videoPause(int32_t playerHandle) { VideoPlayerT *v = _getPlayer(playerHandle, "videoPause"); v->playing = false; if (v->audioSourceCount > 0) { MIX_PauseTrack(v->track); } } void videoPlay(int32_t playerHandle) { VideoPlayerT *v = _getPlayer(playerHandle, "videoPlay"); v->playing = true; v->resetTime = true; if (v->audioSourceCount > 0) { MIX_ResumeTrack(v->track); } } void videoQuit(void) { VideoPlayerT *v = NULL; VideoPlayerT *t = NULL; // Unload any remaining videos HASH_ITER(hh, _videoPlayerHash, v, t) { videoUnload(v->id); } } void videoSeek(int32_t playerHandle, int64_t seekFrame) { VideoPlayerT *v = _getPlayer(playerHandle, "videoSeek"); int64_t count = v->frameCount; // Wrap into range. if (count > 0) { seekFrame = ((seekFrame % count) + count) % count; } else { seekFrame = 0; } v->frame = seekFrame; v->resetTime = true; } void videoSetAudioCalibration(int32_t milliseconds) { _audioCalibrationMs = milliseconds; } void videoSetAudioDelay(int32_t milliseconds) { _audioDelayMs = milliseconds; } // Chosen before any video loads; existing players keep whatever they opened with. void videoSetHardwareDecoding(bool enabled) { _hardwareDecoding = enabled; } void videoSetAudioTrack(int32_t playerHandle, int32_t track) { VideoPlayerT *v = _getPlayer(playerHandle, "videoSetAudioTrack"); if ((track < 0) || (track >= v->audioSourceCount)) { utilDie("Invalid audio track %d in videoSetAudioTrack.", track); } if (track != v->currentAudioTrack) { _audioSelectTrack(v, track); // Drop the queued audio from the old track and realign. v->resetTime = true; } } void videoSetVolume(int32_t playerHandle, int32_t leftPercent, int32_t rightPercent) { VideoPlayerT *v = _getPlayer(playerHandle, "videoSetVolume"); MIX_StereoGains gains; v->volumeLeft = leftPercent; v->volumeRight = rightPercent; if (v->audioSourceCount > 0) { gains.left = (float)leftPercent * PERCENT_TO_SCALE; gains.right = (float)rightPercent * PERCENT_TO_SCALE; MIX_SetTrackStereo(v->track, &gains); } } void videoUnlockAudio(void) { MIX_UnlockMixer(_mixer); } void videoUnload(int32_t playerHandle) { VideoPlayerT *v = _getPlayer(playerHandle, "videoUnload"); int32_t x = 0; // Stop the decoder before touching anything it uses. SDL_LockMutex(v->lock); v->quitThread = true; SDL_SignalCondition(v->wake); SDL_UnlockMutex(v->lock); SDL_WaitThread(v->thread, NULL); SDL_DestroyCondition(v->wake); SDL_DestroyMutex(v->lock); for (x = 0; x < PLANE_COUNT; x++) { av_free(v->front.data[x]); av_free(v->back.data[x]); } sws_freeContext(v->sws); av_frame_free(&v->hwFrame); av_frame_free(&v->videoFrame); av_packet_free(&v->videoPacket); avcodec_free_context(&v->videoCodec); av_buffer_unref(&v->hwDevice); _formatClose(&v->videoFormat); free(v->frames); SDL_DestroyTexture(v->videoTexture); if (v->audioSourceCount > 0) { MIX_DestroyTrack(v->track); SDL_DestroyAudioStream(v->audioStream); _audioCloseTrack(v); av_frame_free(&v->audioFrame); av_packet_free(&v->audioPacket); _formatClose(&v->audioFormat); for (x = 0; x < v->audioSourceCount; x++) { free(v->audio[x].language); } free(v->audio); free(v->audioBuffer); } #pragma GCC diagnostic push #pragma GCC diagnostic ignored "-Wcast-align" HASH_DEL(_videoPlayerHash, v); #pragma GCC diagnostic pop free(v); } // Advances playback to match the audio clock (or the wall clock for silent videos). Returns the frame now on the texture. int64_t videoUpdate(int32_t playerHandle, SDL_Texture **texture) { VideoPlayerT *v = _getPlayer(playerHandle, "videoUpdate"); uint64_t now = SDL_GetTicks(); int64_t elapsed = 0; int64_t count = v->frameCount; int64_t next = 0; int64_t lastDuration = 0; if (v->resetTime) { _resetClock(v, now); } else { if (v->playing) { // Where in the video should we be? Follow the audio when there is any. if (v->audioSourceCount > 0) { elapsed = _audioClock(v, now); } else { elapsed = (int64_t)(now - v->startTicks) + v->startTime; } // Advance to the last frame whose presentation time has arrived. next = v->frame + 1; while ((next < count) && (_frameTime(v, next) <= elapsed)) { v->frame = next; next++; } // Past the end of the last frame? Loop. if (next >= count) { lastDuration = (int64_t)(MS_PER_SECOND * (double)v->fps.den / (double)v->fps.num); if (elapsed >= _frameTime(v, count - 1) + lastDuration) { v->frame = 0; _resetClock(v, now); } } } } // Hand the decoder the frame we want, and show whatever it has finished. _requestFrame(v); if (_takeDecodedFrame(v)) { _uploadFrame(v); } *texture = v->videoTexture; // Handle audio samples if (v->playing && (v->audioSourceCount > 0)) { _feedAudio(v); } // Report what is on screen, so the caller redraws when it changes. return v->uploadedFrame; }