Fix 32-bit dropouts: enable AVX2 and bound the render-ahead
build / windows (push) Has been cancelled
build / release (push) Has been cancelled

This commit is contained in:
2026-10-06 02:43:44 +08:00
parent ea30786711
commit 4d22797130
6 changed files with 107 additions and 34 deletions
+18 -6
View File
@@ -113,18 +113,19 @@
</ItemDefinitionGroup> </ItemDefinitionGroup>
<!-- <!--
SIMD units, mirroring CMakeLists.txt (the JOC_SIMD_* block) exactly. SIMD units, mirroring CMakeLists.txt (the JOC_SIMD_* block).
The ISA in the file name, one translation unit per ISA, each compiled with The ISA in the file name, one translation unit per ISA, each compiled with
its own /arch because MSVC has no function-level ISA attribute; dispatch.cpp its own /arch because MSVC has no function-level ISA attribute; dispatch.cpp
(a baseline unit) picks one at run time from CPUID/XGETBV. The kernel (a baseline unit) picks one at run time from CPUID/XGETBV.
enables SIMD only for x86_64 and aarch64, so:
* Win32 (x86) gets NO SIMD unit and NO SIMD define: the scalar reference
and the probe/dispatch baseline are all it builds, exactly as
CMAKE_SIZEOF_VOID_P EQUAL 8 gates them out upstream;
* x64 gets the AVX2 and AVX-512 units, JOC_SIMD_HAVE_SSE2 / _AVX2 / * x64 gets the AVX2 and AVX-512 units, JOC_SIMD_HAVE_SSE2 / _AVX2 /
_AVX512, and a per-file /arch for those two files only; _AVX512, and a per-file /arch for those two files only;
* Win32 (x86) gets the AVX2 unit and JOC_SIMD_HAVE_AVX2. AVX2 is not an
x86-64-only ISA and MSVC accepts /arch:AVX2 for x86, so gating it on the
pointer size left the 32-bit component on the scalar reference, which is
too slow to hold a 4096-frame read inside its own 85.3 ms of audio.
AVX-512 stays x64-only: 32-bit mode addresses ZMM0-7 only.
* src\simd\kernels_intrin_neon.cpp is excluded everywhere here: the CMake * src\simd\kernels_intrin_neon.cpp is excluded everywhere here: the CMake
build lists it only for aarch64 (CMakeLists.txt lines 177-180), which no build lists it only for aarch64 (CMakeLists.txt lines 177-180), which no
configuration of this project targets. configuration of this project targets.
@@ -134,6 +135,11 @@
<PreprocessorDefinitions>JOC_SIMD_HAVE_SSE2=1;JOC_SIMD_HAVE_AVX2=1;JOC_SIMD_HAVE_AVX512=1;%(PreprocessorDefinitions)</PreprocessorDefinitions> <PreprocessorDefinitions>JOC_SIMD_HAVE_SSE2=1;JOC_SIMD_HAVE_AVX2=1;JOC_SIMD_HAVE_AVX512=1;%(PreprocessorDefinitions)</PreprocessorDefinitions>
</ClCompile> </ClCompile>
</ItemDefinitionGroup> </ItemDefinitionGroup>
<ItemDefinitionGroup Condition="'$(Platform)'=='Win32'">
<ClCompile>
<PreprocessorDefinitions>JOC_SIMD_HAVE_AVX2=1;%(PreprocessorDefinitions)</PreprocessorDefinitions>
</ClCompile>
</ItemDefinitionGroup>
<ItemGroup> <ItemGroup>
<!-- Verbatim copies of the upstream native library (JOC_REUSED_SOURCES). --> <!-- Verbatim copies of the upstream native library (JOC_REUSED_SOURCES). -->
@@ -203,6 +209,12 @@
</ClCompile> </ClCompile>
</ItemGroup> </ItemGroup>
<ItemGroup Condition="'$(Platform)'=='Win32'">
<ClCompile Include="src\simd\kernels_intrin_avx2.cpp">
<AdditionalOptions>/arch:AVX2 %(AdditionalOptions)</AdditionalOptions>
</ClCompile>
</ItemGroup>
<ItemGroup> <ItemGroup>
<ClInclude Include="include\eac3joc_core.h" /> <ClInclude Include="include\eac3joc_core.h" />
<ClInclude Include="include\joc_core.h" /> <ClInclude Include="include\joc_core.h" />
+9 -6
View File
@@ -7,10 +7,10 @@
#include "simd/cpu_probe.h" #include "simd/cpu_probe.h"
#if defined(_M_X64) #if defined(_M_X64) || defined(_M_IX86)
#include <immintrin.h> #include <immintrin.h>
#include <intrin.h> #include <intrin.h>
#elif defined(__x86_64__) #elif defined(__x86_64__) || defined(__i386__)
#include <cpuid.h> #include <cpuid.h>
#endif #endif
@@ -21,10 +21,13 @@
namespace joc::simd { namespace joc::simd {
namespace { namespace {
// ------------------------------------------------------------------- x86-64 -- // ---------------------------------------------------------------------- x86 --
#if defined(_M_X64) || defined(__x86_64__) // Both pointer sizes are probed: AVX2 is not an x86-64-only ISA, and gating this
// on _M_X64 / __x86_64__ left every 32-bit x86 build reporting "no features",
// which pinned the dispatcher to the scalar kernels.
#if defined(_M_X64) || defined(_M_IX86) || defined(__x86_64__) || defined(__i386__)
#if defined(_M_X64) #if defined(_M_X64) || defined(_M_IX86)
// CPUID tells us what the silicon can do; XCR0 tells us whether the OS saves the // CPUID tells us what the silicon can do; XCR0 tells us whether the OS saves the
// state the wider registers need. Both have to agree, or the first AVX // state the wider registers need. Both have to agree, or the first AVX
@@ -53,7 +56,7 @@ CpuFeatures probe_x86() noexcept {
return features; return features;
} }
#else // GCC/Clang on x86-64 #else // GCC/Clang on x86
// The compiler runtime performs the same CPUID + XGETBV probe (libgcc's cpuinfo // The compiler runtime performs the same CPUID + XGETBV probe (libgcc's cpuinfo
// checks XCR0 before it reports AVX), which keeps this file free of inline // checks XCR0 before it reports AVX), which keeps this file free of inline
+15 -3
View File
@@ -221,7 +221,7 @@ Status Stream::push_eac3(const std::uint8_t* data, std::size_t size, std::size_t
} }
metadata_.push_back(entry); metadata_.push_back(entry);
} }
return process_ready_frames(); return process_ready_frames(false);
} }
Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::size_t* consumed) { Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::size_t* consumed) {
@@ -235,7 +235,7 @@ Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::siz
bed_pending_.insert(bed_pending_.end(), interleaved6, bed_pending_.insert(bed_pending_.end(), interleaved6,
interleaved6 + samples * kBedChannels); interleaved6 + samples * kBedChannels);
} }
return process_ready_frames(); return process_ready_frames(false);
} }
Status Stream::push_objects16(const float* planar16, std::size_t samples, std::size_t* consumed) { Status Stream::push_objects16(const float* planar16, std::size_t samples, std::size_t* consumed) {
@@ -267,8 +267,13 @@ Status Stream::push_objects16(const float* planar16, std::size_t samples, std::s
return Status::success(); return Status::success();
} }
Status Stream::process_ready_frames() { Status Stream::process_ready_frames(bool drain_all) {
while (bed_pending_.size() / kBedChannels >= kFrameSamples && !metadata_.empty()) { while (bed_pending_.size() / kBedChannels >= kFrameSamples && !metadata_.empty()) {
// Stop before rendering what the caller is not about to take: the frames
// stay queued, in order, and are rendered by a later push or by flush().
if (!drain_all && buffered_samples() >= kMaxRenderAheadSamples) {
break;
}
const FrameMetadata entry = metadata_.front(); const FrameMetadata entry = metadata_.front();
metadata_.pop_front(); metadata_.pop_front();
@@ -429,6 +434,13 @@ Status Stream::pull(float* destination, std::size_t capacity_samples, std::size_
} }
Status Stream::flush() { Status Stream::flush() {
// Input has ended, so the render-ahead bound has nothing left to wait for:
// every frame still queued has to reach the renderer before its tail is
// drained, or the end of the file would be dropped.
const Status remaining = process_ready_frames(true);
if (!remaining.ok()) {
return remaining;
}
if (binaural_ready_) { if (binaural_ready_) {
std::vector<double> tail; std::vector<double> tail;
const Status drained = const Status drained =
+11 -1
View File
@@ -85,8 +85,18 @@ public:
: 0u; : 0u;
} }
// A push renders every frame it makes ready, and the caller decides how far
// its demuxer runs ahead of playback. Without a bound, a demuxer that runs
// far ahead turns its whole read-ahead burst into latency on whichever pull()
// happens to follow it: the samples are not wasted, but they are rendered at
// the worst possible moment. Rendering therefore stops once this many
// samples are rendered and unpulled; flush() lifts the bound so the frames
// still waiting when the input ends are drained rather than dropped.
static constexpr std::size_t kMaxRenderAheadSamples = 16384;
private: private:
Status process_ready_frames(); // `drain_all` ignores kMaxRenderAheadSamples and renders every ready frame.
Status process_ready_frames(bool drain_all);
Status render_objects16(const std::vector<float>& objects16); Status render_objects16(const std::vector<float>& objects16);
Status render_rosella_objects16(const std::vector<float>& objects16); Status render_rosella_objects16(const std::vector<float>& objects16);
// Moves at most `limit` rendered stereo samples per channel out of the FIFO // Moves at most `limit` rendered stereo samples per channel out of the FIFO
+53 -17
View File
@@ -20,6 +20,16 @@ namespace joc_decode {
namespace { namespace {
constexpr std::size_t kEac3Chunk = 96u * 1024u; // bytes read per push constexpr std::size_t kEac3Chunk = 96u * 1024u; // bytes read per push
// The core renders every frame it is handed, and it renders it during the push,
// so a read must not queue more frames than the caller is about to take: a 96 KB
// read is around thirty syncframes, i.e. a second of audio rendered to satisfy
// one 4096-frame read. This is that read (2.67 syncframes) rounded up, so each
// read hands the core about as much as it is about to consume.
constexpr std::uint64_t kEac3FramesPerRead = 3u;
// Rendered audio the caller has not taken yet. Once this much is waiting there
// is nothing to gain from queueing more input: the core would render it now and
// the caller would not ask for it for several more reads.
constexpr std::size_t kMaxRenderedAheadSamples = 4096u;
constexpr std::size_t kBedFramesChunk = 8192u; // staging capacity, in frames constexpr std::size_t kBedFramesChunk = 8192u; // staging capacity, in frames
constexpr std::size_t kBedChannels = 6; // ffmpeg -ac 6 constexpr std::size_t kBedChannels = 6; // ffmpeg -ac 6
// A read on an anonymous pipe only completes once the whole request is available, // A read on an anonymous pipe only completes once the whole request is available,
@@ -974,7 +984,23 @@ std::size_t Engine::read(float* destination, std::size_t frames, std::string* er
impl.eac3_eof = true; impl.eac3_eof = true;
} }
if (!impl.eac3_eof && impl.frames_queued <= impl.bed_frames_pushed + 2u && // The core renders during the push, so input is queued only while the
// caller still has less than one read's worth of rendered audio waiting.
// Pushing past that is what turns a single read into a second of work:
// the samples are rendered early rather than wrongly, and the read that
// pays for them overruns its own audio.
std::size_t rendered_ahead = 0;
{
joc_stream_status_info pending{};
pending.struct_size = sizeof(pending);
pending.struct_version = 1;
if (impl.api.status(impl.stream, &pending) == JOC_OK) {
rendered_ahead = static_cast<std::size_t>(pending.buffered_samples);
}
}
if (!impl.eac3_eof && rendered_ahead < kMaxRenderedAheadSamples &&
impl.frames_queued <= impl.bed_frames_pushed + 2u &&
(impl.settings.input_frame_limit == 0 || (impl.settings.input_frame_limit == 0 ||
impl.frames_queued < impl.settings.input_frame_limit)) { impl.frames_queued < impl.settings.input_frame_limit)) {
// The tail of a chunk is usually the head of the next syncframe. It // The tail of a chunk is usually the head of the next syncframe. It
@@ -1023,26 +1049,32 @@ std::size_t Engine::read(float* destination, std::size_t frames, std::string* er
offset += bytes; offset += bytes;
++complete; ++complete;
} }
// With an input limit the chunk is cut at a frame boundary: the // The chunk is cut at a frame boundary for two independent
// reasons: a configured input limit has to stop exactly where it
// says, and a read must not queue more frames than it is about to
// consume. Both cuts land on a frame boundary because the
// renderer's output depends on how many frames it was given, so a // renderer's output depends on how many frames it was given, so a
// limit that overshoots to the end of the read buffer would not // cut that overshoots would not reproduce a run that stopped
// reproduce a run that stopped earlier. // earlier.
std::size_t push_bytes = offset; std::uint64_t allowed = complete;
std::uint64_t pushed_frames = complete;
if (impl.settings.input_frame_limit != 0) { if (impl.settings.input_frame_limit != 0) {
const std::uint64_t room = const std::uint64_t room =
impl.settings.input_frame_limit - impl.frames_queued; impl.settings.input_frame_limit - impl.frames_queued;
if (complete > room) { if (allowed > room) allowed = room;
pushed_frames = room; }
std::size_t walk = 0; if (allowed > kEac3FramesPerRead) allowed = kEac3FramesPerRead;
for (std::uint64_t index = 0; index < pushed_frames; ++index) { std::size_t push_bytes = offset;
const std::size_t bytes = joc_eac3::frame_bytes_at( std::uint64_t pushed_frames = complete;
impl.eac3_buffer.data(), total, walk); if (allowed < complete) {
if (bytes == 0 || walk + bytes > total) break; pushed_frames = allowed;
walk += bytes; std::size_t walk = 0;
} for (std::uint64_t index = 0; index < pushed_frames; ++index) {
push_bytes = walk; const std::size_t bytes = joc_eac3::frame_bytes_at(
impl.eac3_buffer.data(), total, walk);
if (bytes == 0 || walk + bytes > total) break;
walk += bytes;
} }
push_bytes = walk;
} }
joc_stream_buffer input{}; joc_stream_buffer input{};
input.struct_size = sizeof(input); input.struct_size = sizeof(input);
@@ -1064,7 +1096,11 @@ std::size_t Engine::read(float* destination, std::size_t frames, std::string* er
std::memmove(impl.eac3_buffer.data(), impl.eac3_buffer.data() + push_bytes, std::memmove(impl.eac3_buffer.data(), impl.eac3_buffer.data() + push_bytes,
impl.eac3_carry); impl.eac3_carry);
} }
if (got == 0) impl.eac3_eof = true; // The pipe can end while complete syncframes are still waiting in
// the buffer: they are queued by the next pass, so the input is
// only over once nothing but a partial frame is left. Ending here
// instead would drop them, and with them the end of the file.
if (got == 0 && pushed_frames >= complete) impl.eac3_eof = true;
} }
} }
+1 -1
View File
@@ -12,7 +12,7 @@
// Kept in one place: the string reported to foobar2000 and written to the log // Kept in one place: the string reported to foobar2000 and written to the log
// must not drift apart. // must not drift apart.
#define JOC_VERSION "0.2.2" #define JOC_VERSION "0.3.0"
DECLARE_COMPONENT_VERSION("JOC decoder (E-AC-3 JOC)", JOC_VERSION, DECLARE_COMPONENT_VERSION("JOC decoder (E-AC-3 JOC)", JOC_VERSION,
"Plays E-AC-3 JOC (Dolby Atmos) files: the JOC objects are " "Plays E-AC-3 JOC (Dolby Atmos) files: the JOC objects are "