From 4d22797130b2a11c11c11725d5a0c349cbbe6ed4 Mon Sep 17 00:00:00 2001 From: TheM14 Date: Tue, 6 Oct 2026 02:43:44 +0800 Subject: [PATCH] Fix 32-bit dropouts: enable AVX2 and bound the render-ahead --- kernel/joc_kernel.vcxproj | 24 +++++++++--- kernel/src/simd/cpu_probe.cpp | 15 +++++--- kernel/src/stream/stream.cpp | 18 +++++++-- kernel/src/stream/stream.h | 12 +++++- src/joc_decode.cpp | 70 ++++++++++++++++++++++++++--------- src/main.cpp | 2 +- 6 files changed, 107 insertions(+), 34 deletions(-) diff --git a/kernel/joc_kernel.vcxproj b/kernel/joc_kernel.vcxproj index 54de5bc..15ffe88 100644 --- a/kernel/joc_kernel.vcxproj +++ b/kernel/joc_kernel.vcxproj @@ -113,18 +113,19 @@ @@ -203,6 +209,12 @@ + + + /arch:AVX2 %(AdditionalOptions) + + + diff --git a/kernel/src/simd/cpu_probe.cpp b/kernel/src/simd/cpu_probe.cpp index 90cec52..206d973 100644 --- a/kernel/src/simd/cpu_probe.cpp +++ b/kernel/src/simd/cpu_probe.cpp @@ -7,10 +7,10 @@ #include "simd/cpu_probe.h" -#if defined(_M_X64) +#if defined(_M_X64) || defined(_M_IX86) #include #include -#elif defined(__x86_64__) +#elif defined(__x86_64__) || defined(__i386__) #include #endif @@ -21,10 +21,13 @@ namespace joc::simd { namespace { -// ------------------------------------------------------------------- x86-64 -- -#if defined(_M_X64) || defined(__x86_64__) +// ---------------------------------------------------------------------- x86 -- +// Both pointer sizes are probed: AVX2 is not an x86-64-only ISA, and gating this +// on _M_X64 / __x86_64__ left every 32-bit x86 build reporting "no features", +// which pinned the dispatcher to the scalar kernels. +#if defined(_M_X64) || defined(_M_IX86) || defined(__x86_64__) || defined(__i386__) -#if defined(_M_X64) +#if defined(_M_X64) || defined(_M_IX86) // CPUID tells us what the silicon can do; XCR0 tells us whether the OS saves the // state the wider registers need. Both have to agree, or the first AVX @@ -53,7 +56,7 @@ CpuFeatures probe_x86() noexcept { return features; } -#else // GCC/Clang on x86-64 +#else // GCC/Clang on x86 // The compiler runtime performs the same CPUID + XGETBV probe (libgcc's cpuinfo // checks XCR0 before it reports AVX), which keeps this file free of inline diff --git a/kernel/src/stream/stream.cpp b/kernel/src/stream/stream.cpp index f6b7fd8..399d5ea 100644 --- a/kernel/src/stream/stream.cpp +++ b/kernel/src/stream/stream.cpp @@ -221,7 +221,7 @@ Status Stream::push_eac3(const std::uint8_t* data, std::size_t size, std::size_t } metadata_.push_back(entry); } - return process_ready_frames(); + return process_ready_frames(false); } Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::size_t* consumed) { @@ -235,7 +235,7 @@ Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::siz bed_pending_.insert(bed_pending_.end(), interleaved6, interleaved6 + samples * kBedChannels); } - return process_ready_frames(); + return process_ready_frames(false); } Status Stream::push_objects16(const float* planar16, std::size_t samples, std::size_t* consumed) { @@ -267,8 +267,13 @@ Status Stream::push_objects16(const float* planar16, std::size_t samples, std::s return Status::success(); } -Status Stream::process_ready_frames() { +Status Stream::process_ready_frames(bool drain_all) { while (bed_pending_.size() / kBedChannels >= kFrameSamples && !metadata_.empty()) { + // Stop before rendering what the caller is not about to take: the frames + // stay queued, in order, and are rendered by a later push or by flush(). + if (!drain_all && buffered_samples() >= kMaxRenderAheadSamples) { + break; + } const FrameMetadata entry = metadata_.front(); metadata_.pop_front(); @@ -429,6 +434,13 @@ Status Stream::pull(float* destination, std::size_t capacity_samples, std::size_ } Status Stream::flush() { + // Input has ended, so the render-ahead bound has nothing left to wait for: + // every frame still queued has to reach the renderer before its tail is + // drained, or the end of the file would be dropped. + const Status remaining = process_ready_frames(true); + if (!remaining.ok()) { + return remaining; + } if (binaural_ready_) { std::vector tail; const Status drained = diff --git a/kernel/src/stream/stream.h b/kernel/src/stream/stream.h index 99d61d4..e0ad56a 100644 --- a/kernel/src/stream/stream.h +++ b/kernel/src/stream/stream.h @@ -85,8 +85,18 @@ public: : 0u; } + // A push renders every frame it makes ready, and the caller decides how far + // its demuxer runs ahead of playback. Without a bound, a demuxer that runs + // far ahead turns its whole read-ahead burst into latency on whichever pull() + // happens to follow it: the samples are not wasted, but they are rendered at + // the worst possible moment. Rendering therefore stops once this many + // samples are rendered and unpulled; flush() lifts the bound so the frames + // still waiting when the input ends are drained rather than dropped. + static constexpr std::size_t kMaxRenderAheadSamples = 16384; + private: - Status process_ready_frames(); + // `drain_all` ignores kMaxRenderAheadSamples and renders every ready frame. + Status process_ready_frames(bool drain_all); Status render_objects16(const std::vector& objects16); Status render_rosella_objects16(const std::vector& objects16); // Moves at most `limit` rendered stereo samples per channel out of the FIFO diff --git a/src/joc_decode.cpp b/src/joc_decode.cpp index eb64342..218cfd7 100644 --- a/src/joc_decode.cpp +++ b/src/joc_decode.cpp @@ -20,6 +20,16 @@ namespace joc_decode { namespace { constexpr std::size_t kEac3Chunk = 96u * 1024u; // bytes read per push +// The core renders every frame it is handed, and it renders it during the push, +// so a read must not queue more frames than the caller is about to take: a 96 KB +// read is around thirty syncframes, i.e. a second of audio rendered to satisfy +// one 4096-frame read. This is that read (2.67 syncframes) rounded up, so each +// read hands the core about as much as it is about to consume. +constexpr std::uint64_t kEac3FramesPerRead = 3u; +// Rendered audio the caller has not taken yet. Once this much is waiting there +// is nothing to gain from queueing more input: the core would render it now and +// the caller would not ask for it for several more reads. +constexpr std::size_t kMaxRenderedAheadSamples = 4096u; constexpr std::size_t kBedFramesChunk = 8192u; // staging capacity, in frames constexpr std::size_t kBedChannels = 6; // ffmpeg -ac 6 // A read on an anonymous pipe only completes once the whole request is available, @@ -974,7 +984,23 @@ std::size_t Engine::read(float* destination, std::size_t frames, std::string* er impl.eac3_eof = true; } - if (!impl.eac3_eof && impl.frames_queued <= impl.bed_frames_pushed + 2u && + // The core renders during the push, so input is queued only while the + // caller still has less than one read's worth of rendered audio waiting. + // Pushing past that is what turns a single read into a second of work: + // the samples are rendered early rather than wrongly, and the read that + // pays for them overruns its own audio. + std::size_t rendered_ahead = 0; + { + joc_stream_status_info pending{}; + pending.struct_size = sizeof(pending); + pending.struct_version = 1; + if (impl.api.status(impl.stream, &pending) == JOC_OK) { + rendered_ahead = static_cast(pending.buffered_samples); + } + } + + if (!impl.eac3_eof && rendered_ahead < kMaxRenderedAheadSamples && + impl.frames_queued <= impl.bed_frames_pushed + 2u && (impl.settings.input_frame_limit == 0 || impl.frames_queued < impl.settings.input_frame_limit)) { // The tail of a chunk is usually the head of the next syncframe. It @@ -1023,26 +1049,32 @@ std::size_t Engine::read(float* destination, std::size_t frames, std::string* er offset += bytes; ++complete; } - // With an input limit the chunk is cut at a frame boundary: the + // The chunk is cut at a frame boundary for two independent + // reasons: a configured input limit has to stop exactly where it + // says, and a read must not queue more frames than it is about to + // consume. Both cuts land on a frame boundary because the // renderer's output depends on how many frames it was given, so a - // limit that overshoots to the end of the read buffer would not - // reproduce a run that stopped earlier. - std::size_t push_bytes = offset; - std::uint64_t pushed_frames = complete; + // cut that overshoots would not reproduce a run that stopped + // earlier. + std::uint64_t allowed = complete; if (impl.settings.input_frame_limit != 0) { const std::uint64_t room = impl.settings.input_frame_limit - impl.frames_queued; - if (complete > room) { - pushed_frames = room; - std::size_t walk = 0; - for (std::uint64_t index = 0; index < pushed_frames; ++index) { - const std::size_t bytes = joc_eac3::frame_bytes_at( - impl.eac3_buffer.data(), total, walk); - if (bytes == 0 || walk + bytes > total) break; - walk += bytes; - } - push_bytes = walk; + if (allowed > room) allowed = room; + } + if (allowed > kEac3FramesPerRead) allowed = kEac3FramesPerRead; + std::size_t push_bytes = offset; + std::uint64_t pushed_frames = complete; + if (allowed < complete) { + pushed_frames = allowed; + std::size_t walk = 0; + for (std::uint64_t index = 0; index < pushed_frames; ++index) { + const std::size_t bytes = joc_eac3::frame_bytes_at( + impl.eac3_buffer.data(), total, walk); + if (bytes == 0 || walk + bytes > total) break; + walk += bytes; } + push_bytes = walk; } joc_stream_buffer input{}; input.struct_size = sizeof(input); @@ -1064,7 +1096,11 @@ std::size_t Engine::read(float* destination, std::size_t frames, std::string* er std::memmove(impl.eac3_buffer.data(), impl.eac3_buffer.data() + push_bytes, impl.eac3_carry); } - if (got == 0) impl.eac3_eof = true; + // The pipe can end while complete syncframes are still waiting in + // the buffer: they are queued by the next pass, so the input is + // only over once nothing but a partial frame is left. Ending here + // instead would drop them, and with them the end of the file. + if (got == 0 && pushed_frames >= complete) impl.eac3_eof = true; } } diff --git a/src/main.cpp b/src/main.cpp index 69240d4..dfdd67f 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -12,7 +12,7 @@ // Kept in one place: the string reported to foobar2000 and written to the log // must not drift apart. -#define JOC_VERSION "0.2.2" +#define JOC_VERSION "0.3.0" DECLARE_COMPONENT_VERSION("JOC decoder (E-AC-3 JOC)", JOC_VERSION, "Plays E-AC-3 JOC (Dolby Atmos) files: the JOC objects are "