diff --git a/kernel/joc_kernel.vcxproj b/kernel/joc_kernel.vcxproj
index 54de5bc..15ffe88 100644
--- a/kernel/joc_kernel.vcxproj
+++ b/kernel/joc_kernel.vcxproj
@@ -113,18 +113,19 @@
@@ -203,6 +209,12 @@
+
+
+ /arch:AVX2 %(AdditionalOptions)
+
+
+
diff --git a/kernel/src/simd/cpu_probe.cpp b/kernel/src/simd/cpu_probe.cpp
index 90cec52..206d973 100644
--- a/kernel/src/simd/cpu_probe.cpp
+++ b/kernel/src/simd/cpu_probe.cpp
@@ -7,10 +7,10 @@
#include "simd/cpu_probe.h"
-#if defined(_M_X64)
+#if defined(_M_X64) || defined(_M_IX86)
#include
#include
-#elif defined(__x86_64__)
+#elif defined(__x86_64__) || defined(__i386__)
#include
#endif
@@ -21,10 +21,13 @@
namespace joc::simd {
namespace {
-// ------------------------------------------------------------------- x86-64 --
-#if defined(_M_X64) || defined(__x86_64__)
+// ---------------------------------------------------------------------- x86 --
+// Both pointer sizes are probed: AVX2 is not an x86-64-only ISA, and gating this
+// on _M_X64 / __x86_64__ left every 32-bit x86 build reporting "no features",
+// which pinned the dispatcher to the scalar kernels.
+#if defined(_M_X64) || defined(_M_IX86) || defined(__x86_64__) || defined(__i386__)
-#if defined(_M_X64)
+#if defined(_M_X64) || defined(_M_IX86)
// CPUID tells us what the silicon can do; XCR0 tells us whether the OS saves the
// state the wider registers need. Both have to agree, or the first AVX
@@ -53,7 +56,7 @@ CpuFeatures probe_x86() noexcept {
return features;
}
-#else // GCC/Clang on x86-64
+#else // GCC/Clang on x86
// The compiler runtime performs the same CPUID + XGETBV probe (libgcc's cpuinfo
// checks XCR0 before it reports AVX), which keeps this file free of inline
diff --git a/kernel/src/stream/stream.cpp b/kernel/src/stream/stream.cpp
index f6b7fd8..399d5ea 100644
--- a/kernel/src/stream/stream.cpp
+++ b/kernel/src/stream/stream.cpp
@@ -221,7 +221,7 @@ Status Stream::push_eac3(const std::uint8_t* data, std::size_t size, std::size_t
}
metadata_.push_back(entry);
}
- return process_ready_frames();
+ return process_ready_frames(false);
}
Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::size_t* consumed) {
@@ -235,7 +235,7 @@ Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::siz
bed_pending_.insert(bed_pending_.end(), interleaved6,
interleaved6 + samples * kBedChannels);
}
- return process_ready_frames();
+ return process_ready_frames(false);
}
Status Stream::push_objects16(const float* planar16, std::size_t samples, std::size_t* consumed) {
@@ -267,8 +267,13 @@ Status Stream::push_objects16(const float* planar16, std::size_t samples, std::s
return Status::success();
}
-Status Stream::process_ready_frames() {
+Status Stream::process_ready_frames(bool drain_all) {
while (bed_pending_.size() / kBedChannels >= kFrameSamples && !metadata_.empty()) {
+ // Stop before rendering what the caller is not about to take: the frames
+ // stay queued, in order, and are rendered by a later push or by flush().
+ if (!drain_all && buffered_samples() >= kMaxRenderAheadSamples) {
+ break;
+ }
const FrameMetadata entry = metadata_.front();
metadata_.pop_front();
@@ -429,6 +434,13 @@ Status Stream::pull(float* destination, std::size_t capacity_samples, std::size_
}
Status Stream::flush() {
+ // Input has ended, so the render-ahead bound has nothing left to wait for:
+ // every frame still queued has to reach the renderer before its tail is
+ // drained, or the end of the file would be dropped.
+ const Status remaining = process_ready_frames(true);
+ if (!remaining.ok()) {
+ return remaining;
+ }
if (binaural_ready_) {
std::vector tail;
const Status drained =
diff --git a/kernel/src/stream/stream.h b/kernel/src/stream/stream.h
index 99d61d4..e0ad56a 100644
--- a/kernel/src/stream/stream.h
+++ b/kernel/src/stream/stream.h
@@ -85,8 +85,18 @@ public:
: 0u;
}
+ // A push renders every frame it makes ready, and the caller decides how far
+ // its demuxer runs ahead of playback. Without a bound, a demuxer that runs
+ // far ahead turns its whole read-ahead burst into latency on whichever pull()
+ // happens to follow it: the samples are not wasted, but they are rendered at
+ // the worst possible moment. Rendering therefore stops once this many
+ // samples are rendered and unpulled; flush() lifts the bound so the frames
+ // still waiting when the input ends are drained rather than dropped.
+ static constexpr std::size_t kMaxRenderAheadSamples = 16384;
+
private:
- Status process_ready_frames();
+ // `drain_all` ignores kMaxRenderAheadSamples and renders every ready frame.
+ Status process_ready_frames(bool drain_all);
Status render_objects16(const std::vector& objects16);
Status render_rosella_objects16(const std::vector& objects16);
// Moves at most `limit` rendered stereo samples per channel out of the FIFO
diff --git a/src/joc_decode.cpp b/src/joc_decode.cpp
index eb64342..218cfd7 100644
--- a/src/joc_decode.cpp
+++ b/src/joc_decode.cpp
@@ -20,6 +20,16 @@ namespace joc_decode {
namespace {
constexpr std::size_t kEac3Chunk = 96u * 1024u; // bytes read per push
+// The core renders every frame it is handed, and it renders it during the push,
+// so a read must not queue more frames than the caller is about to take: a 96 KB
+// read is around thirty syncframes, i.e. a second of audio rendered to satisfy
+// one 4096-frame read. This is that read (2.67 syncframes) rounded up, so each
+// read hands the core about as much as it is about to consume.
+constexpr std::uint64_t kEac3FramesPerRead = 3u;
+// Rendered audio the caller has not taken yet. Once this much is waiting there
+// is nothing to gain from queueing more input: the core would render it now and
+// the caller would not ask for it for several more reads.
+constexpr std::size_t kMaxRenderedAheadSamples = 4096u;
constexpr std::size_t kBedFramesChunk = 8192u; // staging capacity, in frames
constexpr std::size_t kBedChannels = 6; // ffmpeg -ac 6
// A read on an anonymous pipe only completes once the whole request is available,
@@ -974,7 +984,23 @@ std::size_t Engine::read(float* destination, std::size_t frames, std::string* er
impl.eac3_eof = true;
}
- if (!impl.eac3_eof && impl.frames_queued <= impl.bed_frames_pushed + 2u &&
+ // The core renders during the push, so input is queued only while the
+ // caller still has less than one read's worth of rendered audio waiting.
+ // Pushing past that is what turns a single read into a second of work:
+ // the samples are rendered early rather than wrongly, and the read that
+ // pays for them overruns its own audio.
+ std::size_t rendered_ahead = 0;
+ {
+ joc_stream_status_info pending{};
+ pending.struct_size = sizeof(pending);
+ pending.struct_version = 1;
+ if (impl.api.status(impl.stream, &pending) == JOC_OK) {
+ rendered_ahead = static_cast(pending.buffered_samples);
+ }
+ }
+
+ if (!impl.eac3_eof && rendered_ahead < kMaxRenderedAheadSamples &&
+ impl.frames_queued <= impl.bed_frames_pushed + 2u &&
(impl.settings.input_frame_limit == 0 ||
impl.frames_queued < impl.settings.input_frame_limit)) {
// The tail of a chunk is usually the head of the next syncframe. It
@@ -1023,26 +1049,32 @@ std::size_t Engine::read(float* destination, std::size_t frames, std::string* er
offset += bytes;
++complete;
}
- // With an input limit the chunk is cut at a frame boundary: the
+ // The chunk is cut at a frame boundary for two independent
+ // reasons: a configured input limit has to stop exactly where it
+ // says, and a read must not queue more frames than it is about to
+ // consume. Both cuts land on a frame boundary because the
// renderer's output depends on how many frames it was given, so a
- // limit that overshoots to the end of the read buffer would not
- // reproduce a run that stopped earlier.
- std::size_t push_bytes = offset;
- std::uint64_t pushed_frames = complete;
+ // cut that overshoots would not reproduce a run that stopped
+ // earlier.
+ std::uint64_t allowed = complete;
if (impl.settings.input_frame_limit != 0) {
const std::uint64_t room =
impl.settings.input_frame_limit - impl.frames_queued;
- if (complete > room) {
- pushed_frames = room;
- std::size_t walk = 0;
- for (std::uint64_t index = 0; index < pushed_frames; ++index) {
- const std::size_t bytes = joc_eac3::frame_bytes_at(
- impl.eac3_buffer.data(), total, walk);
- if (bytes == 0 || walk + bytes > total) break;
- walk += bytes;
- }
- push_bytes = walk;
+ if (allowed > room) allowed = room;
+ }
+ if (allowed > kEac3FramesPerRead) allowed = kEac3FramesPerRead;
+ std::size_t push_bytes = offset;
+ std::uint64_t pushed_frames = complete;
+ if (allowed < complete) {
+ pushed_frames = allowed;
+ std::size_t walk = 0;
+ for (std::uint64_t index = 0; index < pushed_frames; ++index) {
+ const std::size_t bytes = joc_eac3::frame_bytes_at(
+ impl.eac3_buffer.data(), total, walk);
+ if (bytes == 0 || walk + bytes > total) break;
+ walk += bytes;
}
+ push_bytes = walk;
}
joc_stream_buffer input{};
input.struct_size = sizeof(input);
@@ -1064,7 +1096,11 @@ std::size_t Engine::read(float* destination, std::size_t frames, std::string* er
std::memmove(impl.eac3_buffer.data(), impl.eac3_buffer.data() + push_bytes,
impl.eac3_carry);
}
- if (got == 0) impl.eac3_eof = true;
+ // The pipe can end while complete syncframes are still waiting in
+ // the buffer: they are queued by the next pass, so the input is
+ // only over once nothing but a partial frame is left. Ending here
+ // instead would drop them, and with them the end of the file.
+ if (got == 0 && pushed_frames >= complete) impl.eac3_eof = true;
}
}
diff --git a/src/main.cpp b/src/main.cpp
index 69240d4..dfdd67f 100644
--- a/src/main.cpp
+++ b/src/main.cpp
@@ -12,7 +12,7 @@
// Kept in one place: the string reported to foobar2000 and written to the log
// must not drift apart.
-#define JOC_VERSION "0.2.2"
+#define JOC_VERSION "0.3.0"
DECLARE_COMPONENT_VERSION("JOC decoder (E-AC-3 JOC)", JOC_VERSION,
"Plays E-AC-3 JOC (Dolby Atmos) files: the JOC objects are "