From 30182ac6705532daec15f1d1da480818e1d4c22d Mon Sep 17 00:00:00 2001 From: TheM14 Date: Tue, 6 Oct 2026 02:45:25 +0800 Subject: [PATCH] Bound the render-ahead so a demuxer cannot pace rendering --- CMakeLists.txt | 21 +++++++++++++++++++++ src/simd/cpu_probe.cpp | 15 +++++++++------ src/stream/stream.cpp | 18 +++++++++++++++--- src/stream/stream.h | 12 +++++++++++- 4 files changed, 56 insertions(+), 10 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index e6c96de..b389ebf 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -124,6 +124,8 @@ set(JOC_INTERNAL_SOURCES # The x86-64 baseline is SSE2 and there is no SSE2 unit on purpose: a 128-bit # SSE2 register is the register a scalar double already occupies, so SSE2 cannot # widen double-precision arithmetic and hand-written SSE2 would only add moves. +# 32-bit x86 does take the AVX2 unit: AVX2 is not an x86-64-only ISA, and it is +# the only vector path such a build can have. # AArch64 needs no probe either; ASIMD is architectural, and the NEON unit is # how a vector path gets selected there. # --------------------------------------------------------------------------- @@ -141,12 +143,18 @@ if(CMAKE_SIZEOF_VOID_P EQUAL 8) elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(ARM64|arm64|aarch64|AARCH64)$") set(JOC_ARCH aarch64) endif() +elseif(CMAKE_SIZEOF_VOID_P EQUAL 4) + if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(i[3-6]86|x86|X86|IA32)$") + set(JOC_ARCH x86_32) + endif() endif() if(NOT JOC_ARCH AND DEFINED CMAKE_CXX_COMPILER_ARCHITECTURE_ID) if(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "x64") set(JOC_ARCH x86_64) elseif(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "ARM64") set(JOC_ARCH aarch64) + elseif(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "X86") + set(JOC_ARCH x86_32) endif() endif() @@ -174,6 +182,19 @@ if(JOC_ARCH STREQUAL "x86_64") PROPERTIES COMPILE_OPTIONS "-mavx512f") endif() endif() +elseif(JOC_ARCH STREQUAL "x86_32") + # AVX2 only: 32-bit mode addresses ZMM0-7, so the AVX-512 unit is not a fit. + if(JOC_ENABLE_AVX2) + list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_avx2.cpp) + list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_AVX2=1) + if(MSVC) + set_source_files_properties(src/simd/kernels_intrin_avx2.cpp + PROPERTIES COMPILE_OPTIONS "/arch:AVX2") + else() + set_source_files_properties(src/simd/kernels_intrin_avx2.cpp + PROPERTIES COMPILE_OPTIONS "-mavx2") + endif() + endif() elseif(JOC_ARCH STREQUAL "aarch64") list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_NEON=1) list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_neon.cpp) diff --git a/src/simd/cpu_probe.cpp b/src/simd/cpu_probe.cpp index 90cec52..206d973 100644 --- a/src/simd/cpu_probe.cpp +++ b/src/simd/cpu_probe.cpp @@ -7,10 +7,10 @@ #include "simd/cpu_probe.h" -#if defined(_M_X64) +#if defined(_M_X64) || defined(_M_IX86) #include #include -#elif defined(__x86_64__) +#elif defined(__x86_64__) || defined(__i386__) #include #endif @@ -21,10 +21,13 @@ namespace joc::simd { namespace { -// ------------------------------------------------------------------- x86-64 -- -#if defined(_M_X64) || defined(__x86_64__) +// ---------------------------------------------------------------------- x86 -- +// Both pointer sizes are probed: AVX2 is not an x86-64-only ISA, and gating this +// on _M_X64 / __x86_64__ left every 32-bit x86 build reporting "no features", +// which pinned the dispatcher to the scalar kernels. +#if defined(_M_X64) || defined(_M_IX86) || defined(__x86_64__) || defined(__i386__) -#if defined(_M_X64) +#if defined(_M_X64) || defined(_M_IX86) // CPUID tells us what the silicon can do; XCR0 tells us whether the OS saves the // state the wider registers need. Both have to agree, or the first AVX @@ -53,7 +56,7 @@ CpuFeatures probe_x86() noexcept { return features; } -#else // GCC/Clang on x86-64 +#else // GCC/Clang on x86 // The compiler runtime performs the same CPUID + XGETBV probe (libgcc's cpuinfo // checks XCR0 before it reports AVX), which keeps this file free of inline diff --git a/src/stream/stream.cpp b/src/stream/stream.cpp index f6b7fd8..399d5ea 100644 --- a/src/stream/stream.cpp +++ b/src/stream/stream.cpp @@ -221,7 +221,7 @@ Status Stream::push_eac3(const std::uint8_t* data, std::size_t size, std::size_t } metadata_.push_back(entry); } - return process_ready_frames(); + return process_ready_frames(false); } Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::size_t* consumed) { @@ -235,7 +235,7 @@ Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::siz bed_pending_.insert(bed_pending_.end(), interleaved6, interleaved6 + samples * kBedChannels); } - return process_ready_frames(); + return process_ready_frames(false); } Status Stream::push_objects16(const float* planar16, std::size_t samples, std::size_t* consumed) { @@ -267,8 +267,13 @@ Status Stream::push_objects16(const float* planar16, std::size_t samples, std::s return Status::success(); } -Status Stream::process_ready_frames() { +Status Stream::process_ready_frames(bool drain_all) { while (bed_pending_.size() / kBedChannels >= kFrameSamples && !metadata_.empty()) { + // Stop before rendering what the caller is not about to take: the frames + // stay queued, in order, and are rendered by a later push or by flush(). + if (!drain_all && buffered_samples() >= kMaxRenderAheadSamples) { + break; + } const FrameMetadata entry = metadata_.front(); metadata_.pop_front(); @@ -429,6 +434,13 @@ Status Stream::pull(float* destination, std::size_t capacity_samples, std::size_ } Status Stream::flush() { + // Input has ended, so the render-ahead bound has nothing left to wait for: + // every frame still queued has to reach the renderer before its tail is + // drained, or the end of the file would be dropped. + const Status remaining = process_ready_frames(true); + if (!remaining.ok()) { + return remaining; + } if (binaural_ready_) { std::vector tail; const Status drained = diff --git a/src/stream/stream.h b/src/stream/stream.h index 99d61d4..e0ad56a 100644 --- a/src/stream/stream.h +++ b/src/stream/stream.h @@ -85,8 +85,18 @@ public: : 0u; } + // A push renders every frame it makes ready, and the caller decides how far + // its demuxer runs ahead of playback. Without a bound, a demuxer that runs + // far ahead turns its whole read-ahead burst into latency on whichever pull() + // happens to follow it: the samples are not wasted, but they are rendered at + // the worst possible moment. Rendering therefore stops once this many + // samples are rendered and unpulled; flush() lifts the bound so the frames + // still waiting when the input ends are drained rather than dropped. + static constexpr std::size_t kMaxRenderAheadSamples = 16384; + private: - Status process_ready_frames(); + // `drain_all` ignores kMaxRenderAheadSamples and renders every ready frame. + Status process_ready_frames(bool drain_all); Status render_objects16(const std::vector& objects16); Status render_rosella_objects16(const std::vector& objects16); // Moves at most `limit` rendered stereo samples per channel out of the FIFO