Bound the render-ahead so a demuxer cannot pace rendering
This commit is contained in:
@@ -124,6 +124,8 @@ set(JOC_INTERNAL_SOURCES
|
||||
# The x86-64 baseline is SSE2 and there is no SSE2 unit on purpose: a 128-bit
|
||||
# SSE2 register is the register a scalar double already occupies, so SSE2 cannot
|
||||
# widen double-precision arithmetic and hand-written SSE2 would only add moves.
|
||||
# 32-bit x86 does take the AVX2 unit: AVX2 is not an x86-64-only ISA, and it is
|
||||
# the only vector path such a build can have.
|
||||
# AArch64 needs no probe either; ASIMD is architectural, and the NEON unit is
|
||||
# how a vector path gets selected there.
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -141,12 +143,18 @@ if(CMAKE_SIZEOF_VOID_P EQUAL 8)
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(ARM64|arm64|aarch64|AARCH64)$")
|
||||
set(JOC_ARCH aarch64)
|
||||
endif()
|
||||
elseif(CMAKE_SIZEOF_VOID_P EQUAL 4)
|
||||
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(i[3-6]86|x86|X86|IA32)$")
|
||||
set(JOC_ARCH x86_32)
|
||||
endif()
|
||||
endif()
|
||||
if(NOT JOC_ARCH AND DEFINED CMAKE_CXX_COMPILER_ARCHITECTURE_ID)
|
||||
if(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "x64")
|
||||
set(JOC_ARCH x86_64)
|
||||
elseif(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "ARM64")
|
||||
set(JOC_ARCH aarch64)
|
||||
elseif(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "X86")
|
||||
set(JOC_ARCH x86_32)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -174,6 +182,19 @@ if(JOC_ARCH STREQUAL "x86_64")
|
||||
PROPERTIES COMPILE_OPTIONS "-mavx512f")
|
||||
endif()
|
||||
endif()
|
||||
elseif(JOC_ARCH STREQUAL "x86_32")
|
||||
# AVX2 only: 32-bit mode addresses ZMM0-7, so the AVX-512 unit is not a fit.
|
||||
if(JOC_ENABLE_AVX2)
|
||||
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_avx2.cpp)
|
||||
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_AVX2=1)
|
||||
if(MSVC)
|
||||
set_source_files_properties(src/simd/kernels_intrin_avx2.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "/arch:AVX2")
|
||||
else()
|
||||
set_source_files_properties(src/simd/kernels_intrin_avx2.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "-mavx2")
|
||||
endif()
|
||||
endif()
|
||||
elseif(JOC_ARCH STREQUAL "aarch64")
|
||||
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_NEON=1)
|
||||
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_neon.cpp)
|
||||
|
||||
@@ -7,10 +7,10 @@
|
||||
|
||||
#include "simd/cpu_probe.h"
|
||||
|
||||
#if defined(_M_X64)
|
||||
#if defined(_M_X64) || defined(_M_IX86)
|
||||
#include <immintrin.h>
|
||||
#include <intrin.h>
|
||||
#elif defined(__x86_64__)
|
||||
#elif defined(__x86_64__) || defined(__i386__)
|
||||
#include <cpuid.h>
|
||||
#endif
|
||||
|
||||
@@ -21,10 +21,13 @@
|
||||
namespace joc::simd {
|
||||
namespace {
|
||||
|
||||
// ------------------------------------------------------------------- x86-64 --
|
||||
#if defined(_M_X64) || defined(__x86_64__)
|
||||
// ---------------------------------------------------------------------- x86 --
|
||||
// Both pointer sizes are probed: AVX2 is not an x86-64-only ISA, and gating this
|
||||
// on _M_X64 / __x86_64__ left every 32-bit x86 build reporting "no features",
|
||||
// which pinned the dispatcher to the scalar kernels.
|
||||
#if defined(_M_X64) || defined(_M_IX86) || defined(__x86_64__) || defined(__i386__)
|
||||
|
||||
#if defined(_M_X64)
|
||||
#if defined(_M_X64) || defined(_M_IX86)
|
||||
|
||||
// CPUID tells us what the silicon can do; XCR0 tells us whether the OS saves the
|
||||
// state the wider registers need. Both have to agree, or the first AVX
|
||||
@@ -53,7 +56,7 @@ CpuFeatures probe_x86() noexcept {
|
||||
return features;
|
||||
}
|
||||
|
||||
#else // GCC/Clang on x86-64
|
||||
#else // GCC/Clang on x86
|
||||
|
||||
// The compiler runtime performs the same CPUID + XGETBV probe (libgcc's cpuinfo
|
||||
// checks XCR0 before it reports AVX), which keeps this file free of inline
|
||||
|
||||
+15
-3
@@ -221,7 +221,7 @@ Status Stream::push_eac3(const std::uint8_t* data, std::size_t size, std::size_t
|
||||
}
|
||||
metadata_.push_back(entry);
|
||||
}
|
||||
return process_ready_frames();
|
||||
return process_ready_frames(false);
|
||||
}
|
||||
|
||||
Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::size_t* consumed) {
|
||||
@@ -235,7 +235,7 @@ Status Stream::push_bed(const float* interleaved6, std::size_t samples, std::siz
|
||||
bed_pending_.insert(bed_pending_.end(), interleaved6,
|
||||
interleaved6 + samples * kBedChannels);
|
||||
}
|
||||
return process_ready_frames();
|
||||
return process_ready_frames(false);
|
||||
}
|
||||
|
||||
Status Stream::push_objects16(const float* planar16, std::size_t samples, std::size_t* consumed) {
|
||||
@@ -267,8 +267,13 @@ Status Stream::push_objects16(const float* planar16, std::size_t samples, std::s
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status Stream::process_ready_frames() {
|
||||
Status Stream::process_ready_frames(bool drain_all) {
|
||||
while (bed_pending_.size() / kBedChannels >= kFrameSamples && !metadata_.empty()) {
|
||||
// Stop before rendering what the caller is not about to take: the frames
|
||||
// stay queued, in order, and are rendered by a later push or by flush().
|
||||
if (!drain_all && buffered_samples() >= kMaxRenderAheadSamples) {
|
||||
break;
|
||||
}
|
||||
const FrameMetadata entry = metadata_.front();
|
||||
metadata_.pop_front();
|
||||
|
||||
@@ -429,6 +434,13 @@ Status Stream::pull(float* destination, std::size_t capacity_samples, std::size_
|
||||
}
|
||||
|
||||
Status Stream::flush() {
|
||||
// Input has ended, so the render-ahead bound has nothing left to wait for:
|
||||
// every frame still queued has to reach the renderer before its tail is
|
||||
// drained, or the end of the file would be dropped.
|
||||
const Status remaining = process_ready_frames(true);
|
||||
if (!remaining.ok()) {
|
||||
return remaining;
|
||||
}
|
||||
if (binaural_ready_) {
|
||||
std::vector<double> tail;
|
||||
const Status drained =
|
||||
|
||||
+11
-1
@@ -85,8 +85,18 @@ public:
|
||||
: 0u;
|
||||
}
|
||||
|
||||
// A push renders every frame it makes ready, and the caller decides how far
|
||||
// its demuxer runs ahead of playback. Without a bound, a demuxer that runs
|
||||
// far ahead turns its whole read-ahead burst into latency on whichever pull()
|
||||
// happens to follow it: the samples are not wasted, but they are rendered at
|
||||
// the worst possible moment. Rendering therefore stops once this many
|
||||
// samples are rendered and unpulled; flush() lifts the bound so the frames
|
||||
// still waiting when the input ends are drained rather than dropped.
|
||||
static constexpr std::size_t kMaxRenderAheadSamples = 16384;
|
||||
|
||||
private:
|
||||
Status process_ready_frames();
|
||||
// `drain_all` ignores kMaxRenderAheadSamples and renders every ready frame.
|
||||
Status process_ready_frames(bool drain_all);
|
||||
Status render_objects16(const std::vector<float>& objects16);
|
||||
Status render_rosella_objects16(const std::vector<float>& objects16);
|
||||
// Moves at most `limit` rendered stereo samples per channel out of the FIFO
|
||||
|
||||
Reference in New Issue
Block a user