20 Commits

Author SHA1 Message Date
TheM14 428772eb87 Fix OAMD ramp timing and band-0 DC filter gating
Native builds / linux-x64 (push) Failing after 21s
Native builds / macos-arm64 (push) Has been cancelled
Native builds / macos-x64 (push) Has been cancelled
Native builds / windows-x64 (push) Has been cancelled
Native builds / Publish GitHub Release (push) Has been cancelled
2026-09-27 00:55:13 +08:00
TheM14 6bc2c28856 Corrected the wording in the README.
Native builds / macos-arm64 (push) Has been cancelled
Native builds / macos-x64 (push) Has been cancelled
Native builds / windows-x64 (push) Has been cancelled
Native builds / Publish GitHub Release (push) Has been cancelled
Native builds / linux-x64 (push) Has been cancelled
2026-09-16 11:52:18 +08:00
TheM14 704ea0897b Fix formula errors in math.md.
Native builds / linux-x64 (push) Failing after 12s
Native builds / macos-arm64 (push) Has been cancelled
Native builds / macos-x64 (push) Has been cancelled
Native builds / windows-x64 (push) Has been cancelled
Native builds / Publish GitHub Release (push) Has been cancelled
2026-09-16 03:13:20 +08:00
TheM14 c6859e6b02 Fix formula errors in math.md.
Native builds / linux-x64 (push) Failing after 9s
Native builds / macos-arm64 (push) Has been cancelled
Native builds / macos-x64 (push) Has been cancelled
Native builds / windows-x64 (push) Has been cancelled
Native builds / Publish GitHub Release (push) Has been cancelled
2026-09-16 02:58:55 +08:00
TheM14 afef6d2c52 Add Sparse JOC decoding support.
Native builds / linux-x64 (push) Failing after 18s
Native builds / macos-arm64 (push) Has been cancelled
Native builds / macos-x64 (push) Has been cancelled
Native builds / windows-x64 (push) Has been cancelled
Native builds / Publish GitHub Release (push) Has been cancelled
2026-09-16 02:51:28 +08:00
TheM14 ab7e815a4d Clean up .gitignore rules.
Native builds / linux-x64 (push) Failing after 12s
Native builds / macos-arm64 (push) Has been cancelled
Native builds / macos-x64 (push) Has been cancelled
Native builds / windows-x64 (push) Has been cancelled
Native builds / Publish GitHub Release (push) Has been cancelled
2026-09-11 02:22:36 +08:00
TheM14 b634326b5d Decode E-AC-3 at full dynamic range by default and add -drc-scale/target-level options. 2026-09-11 02:22:36 +08:00
TheM14 cf50beafcb Fix OAMD variant resolution. 2026-09-11 02:22:36 +08:00
TheM14 c8676d7aa3 Fix the syntax in docs 2026-09-11 02:22:36 +08:00
TheM14 b619dee523 Fix the syntax in docs 2026-09-11 02:22:36 +08:00
TheM14 7da3a5eb95 Fix the syntax in math.md 2026-09-11 02:22:36 +08:00
TheM14 128286153e Test-related Markdown syntax corrections 2026-09-11 02:22:36 +08:00
TheM14 85105f21d4 Add binaural rendering support.
Native builds / linux-x64 (push) Failing after 11s
Native builds / macos-arm64 (push) Has been cancelled
Native builds / macos-x64 (push) Has been cancelled
Native builds / windows-x64 (push) Has been cancelled
Native builds / Publish GitHub Release (push) Has been cancelled
2026-09-11 02:21:57 +08:00
TheM14 329445ed25 Add experimental JOC binaural mode controls 2026-09-03 02:59:44 +08:00
TheM14 887b3e317f Align OAMD metadata delay with decoder output 2026-09-02 22:17:09 +08:00
TheM14 35536b0bd4 Ignore nested EMDF payload syncwords 2026-09-02 01:11:21 +08:00
TheM14 9b41fedb14 Support variable-length OAMD element payloads 2026-09-02 00:26:46 +08:00
TheM14 65a08e544d Add License 2026-09-01 20:53:04 +08:00
TheM14 7031b43284 Add GitHub Actions workflow for native builds
Native builds / linux-x64 (push) Failing after 12s
Native builds / macos-arm64 (push) Has been cancelled
Native builds / macos-x64 (push) Has been cancelled
Native builds / windows-x64 (push) Has been cancelled
Native builds / Publish GitHub Release (push) Has been cancelled
2026-09-01 16:50:21 +08:00
TheM14 aa3a519f79 Initial public release of JustOneCacophony 2026-09-01 16:31:38 +08:00
153 changed files with 13416 additions and 31703 deletions
-38
View File
@@ -1,38 +0,0 @@
name: ci
on:
push:
pull_request:
jobs:
build:
name: ${{ matrix.os }}
runs-on: ${{ matrix.os }}
strategy:
fail-fast: false
matrix:
os: [ubuntu-latest, macos-latest, windows-latest]
steps:
- uses: actions/checkout@v4
- name: Configure
run: cmake -S . -B build -DCMAKE_BUILD_TYPE=Release
- name: Build
run: cmake --build build --config Release --parallel
- name: Test
run: ctest --test-dir build --output-on-failure -C Release
# Publish the install tree: the shared library and the CLI.
- name: Stage
run: cmake --install build --config Release --prefix stage
- name: Upload
uses: actions/upload-artifact@v4
with:
name: joc-core-${{ runner.os }}
path: |
stage/bin/*
stage/lib/*
if-no-files-found: error
+99
View File
@@ -0,0 +1,99 @@
name: Native builds
on:
workflow_dispatch:
push:
branches:
- main
tags:
- "v*"
permissions:
contents: read
jobs:
build:
name: ${{ matrix.asset }}
runs-on: ${{ matrix.runner }}
strategy:
fail-fast: false
matrix:
include:
- runner: windows-2022
asset: windows-x64
cmake_args: -A x64
- runner: ubuntu-22.04
asset: linux-x64
cmake_args: ""
- runner: macos-15-intel
asset: macos-x64
cmake_args: -DCMAKE_OSX_DEPLOYMENT_TARGET=12.0
- runner: macos-15
asset: macos-arm64
cmake_args: -DCMAKE_OSX_DEPLOYMENT_TARGET=12.0
steps:
- name: Checkout
uses: actions/checkout@v6
- name: Configure
run: >
cmake
-S native
-B build/native
-DCMAKE_BUILD_TYPE=Release
-DCMAKE_INSTALL_PREFIX="${{ github.workspace }}/stage"
${{ matrix.cmake_args }}
- name: Build
run: cmake --build build/native --config Release --parallel
- name: Install
run: cmake --install build/native --config Release
- name: Package
working-directory: stage
run: >
cmake -E tar
cf "../JustOneCacophony-native-${{ matrix.asset }}.zip"
--format=zip
-- .
- name: Upload workflow artifact
uses: actions/upload-artifact@v4
with:
name: JustOneCacophony-native-${{ matrix.asset }}
path: JustOneCacophony-native-${{ matrix.asset }}.zip
if-no-files-found: error
retention-days: 14
release:
name: Publish GitHub Release
if: startsWith(github.ref, 'refs/tags/v')
needs: build
runs-on: ubuntu-24.04
permissions:
contents: write
steps:
- name: Download native packages
uses: actions/download-artifact@v5
with:
pattern: JustOneCacophony-native-*
path: dist
merge-multiple: true
- name: Create release
run: >
gh release create "$GITHUB_REF_NAME"
dist/*.zip
--verify-tag
--generate-notes
env:
GH_TOKEN: ${{ github.token }}
GH_REPO: ${{ github.repository }}
+22 -11
View File
@@ -1,12 +1,23 @@
/build/
/output/
/out/
/traces/
/vectors/
/testdata/
/devtools/
.vs/
.vscode/
CMakeUserPresets.json
__pycache__/
*.spool.f32
*.py[cod]
.pytest_cache/
.venv/
venv/
build/
output/
tests/
lib/
metadata_cache/
*.metadata.json
*.report.json
*.variant-error.json
*.objects16.f32le
HRTF/
*.sofa
*.personalized_headphone
*.jochrtf
-330
View File
@@ -1,330 +0,0 @@
cmake_minimum_required(VERSION 3.20)
project(joc_core VERSION 0.1.0 LANGUAGES C CXX)
# ---------------------------------------------------------------------------
# Products
# joc_core shared library: the execution core (E-AC-3/EMDF/JOC/OAMD
# bitstream, DSP, speaker and binaural rendering, ADM-BWF/WAV
# writing, file task and the streaming push/pull surface)
# joc_cli command line frontend for file tasks
#
# Public headers: include/joc_core.h engine, telemetry and file task
# include/joc_stream.h embedder-facing streaming surface
#
# Options
# JOC_BUILD_TESTS unit tests (CTest), on by default
# JOC_BUILD_DEVTOOLS in-tree verification tools, off: those sources live in
# devtools/, which is not part of the repository
# JOC_ENABLE_AVX2 build the runtime-dispatched AVX2 kernels, on by default
# JOC_ENABLE_AVX512 build the runtime-dispatched AVX-512 kernels, on by default
#
# Only the MSVC toolchain is validated locally; other platforms are built by CI.
# The floating-point flags of the original native library are preserved on
# purpose: /fp:precise on MSVC, -fno-fast-math elsewhere, and a static CRT so
# that no redistributable is required.
# ---------------------------------------------------------------------------
option(JOC_BUILD_TESTS "Build the unit tests" ON)
option(JOC_BUILD_DEVTOOLS "Build the in-tree verification tools (needs devtools/)" OFF)
# The DSP kernels are dispatched at run time (see src/simd/simd.h): the
# baseline units stay on the ISA every x86-64 CPU has, and these two options
# decide whether the wider units are linked in at all. Both are on by default.
# The instruction set actually executed is chosen from CPUID/XGETBV (or the
# AArch64 baseline) when the library is first used, so a binary carrying the
# AVX-512 unit still runs on a CPU without it, and JOC_SIMD=scalar|sse2|avx2|
# avx512|neon pins one tier for verification.
option(JOC_ENABLE_AVX2 "Build the runtime-dispatched AVX2 kernels" ON)
option(JOC_ENABLE_AVX512 "Build the runtime-dispatched AVX-512 kernels" ON)
set(CMAKE_CXX_STANDARD 20)
set(CMAKE_CXX_STANDARD_REQUIRED ON)
set(CMAKE_CXX_EXTENSIONS OFF)
if(NOT CMAKE_BUILD_TYPE AND NOT CMAKE_CONFIGURATION_TYPES)
set(CMAKE_BUILD_TYPE Release CACHE STRING "Build type" FORCE)
endif()
find_package(Threads REQUIRED)
# Ninja learns MSVC's header dependencies by parsing the compiler's /showIncludes
# notes, and it only recognises the prefix it is told about. A localized MSVC
# prints a translated prefix, and when CMake cannot detect it the notes are
# silently dropped: the build then reuses stale objects after a header changes and
# produces a binary that does not match its sources -- wrong, not just slow. This
# only warns, because supplying the value is not always possible either: it has to
# survive the cache code page to be usable, and a build driver can compensate more
# reliably by dropping objects when a header is newer than they are.
if(MSVC AND CMAKE_GENERATOR MATCHES "Ninja" AND NOT CMAKE_CL_SHOWINCLUDES_PREFIX)
message(WARNING
"No /showIncludes prefix was detected, so Ninja will not track header "
"dependencies and a header change will not rebuild what includes it. "
"Configure with -DCMAKE_CL_SHOWINCLUDES_PREFIX=<the text MSVC prints in "
"front of each included file>, or make sure the compiler emits its "
"messages in the language CMake probes for.")
endif()
# Verbatim copies of the upstream native library; see THIRD_PARTY_NOTICES.md.
# These files are never edited: they are the validated DSP kernels.
set(JOC_REUSED_SOURCES
src/joc_core/eac3joc_core.cpp
src/speaker/speaker_renderer.cpp
src/binaural/binaural_renderer.cpp
)
set(JOC_INTERNAL_SOURCES
src/adm/adm_metadata.cpp
src/adm/adm_tracks.cpp
src/binaural/binaural_runtime.cpp
src/binaural/sofa_binaural_renderer.cpp
src/eac3_transport/eac3_reader.cpp
src/emdf/emdf_parser.cpp
src/foundation/bit_reader.cpp
src/foundation/fft.cpp
src/foundation/fs_utf8.cpp
src/foundation/mini_json.cpp
src/foundation/sha256.cpp
src/hrtf/jochrtf.cpp
src/hrtf/public_filterbank.cpp
src/hrtf/rosella_model.cpp
src/hrtf/rosella_renderer.cpp
src/hrtf/kernel_tables.cpp
src/hrtf/sofa.cpp
src/hrtf/sofa_cache.cpp
src/hrtf/sofa_field.cpp
src/io/adm_writer.cpp
src/io/hdf5.cpp
src/io/inflate.cpp
src/io/npy.cpp
src/io/npy_writer.cpp
src/io/process.cpp
src/io/wav_writer.cpp
src/io/zip_reader.cpp
src/joc_bitstream/joc_parser.cpp
src/joc_core/objects16.cpp
src/oamd/oamd_parser.cpp
src/speaker/speaker_layout_lookup.cpp
src/speaker/speaker_step.cpp
src/stream/stream.cpp
src/task/task.cpp
src/telemetry/event_bus.cpp
src/timeline/position_timeline.cpp
)
# ---------------------------------------------------------------------------
# Runtime-dispatched SIMD kernels (src/simd/simd.h explains the contract).
#
# One flat directory, the ISA in the file name (`kernels_intrin_<isa>.cpp`),
# never in a subdirectory. MSVC has no function-level ISA attribute, so every
# ISA lives in its own translation unit compiled with its own flag, and
# dispatch.cpp -- built for the baseline ISA -- picks one when the library is
# first used. Only a `kernels_intrin_*.cpp` unit ever gets a wider flag, so a
# baseline unit cannot inherit one by accident.
#
# The x86-64 baseline is SSE2 and there is no SSE2 unit on purpose: a 128-bit
# SSE2 register is the register a scalar double already occupies, so SSE2 cannot
# widen double-precision arithmetic and hand-written SSE2 would only add moves.
# AArch64 needs no probe either; ASIMD is architectural, and the NEON unit is
# how a vector path gets selected there.
# ---------------------------------------------------------------------------
set(JOC_SIMD_SOURCES
src/simd/cpu_probe.cpp
src/simd/dispatch.cpp
src/simd/kernels_scalar.cpp
)
set(JOC_SIMD_DEFINES "")
set(JOC_ARCH "")
if(CMAKE_SIZEOF_VOID_P EQUAL 8)
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(AMD64|amd64|x86_64|X86_64|EM64T)$")
set(JOC_ARCH x86_64)
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(ARM64|arm64|aarch64|AARCH64)$")
set(JOC_ARCH aarch64)
endif()
endif()
if(NOT JOC_ARCH AND DEFINED CMAKE_CXX_COMPILER_ARCHITECTURE_ID)
if(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "x64")
set(JOC_ARCH x86_64)
elseif(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "ARM64")
set(JOC_ARCH aarch64)
endif()
endif()
if(JOC_ARCH STREQUAL "x86_64")
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_SSE2=1)
if(JOC_ENABLE_AVX2)
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_avx2.cpp)
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_AVX2=1)
if(MSVC)
set_source_files_properties(src/simd/kernels_intrin_avx2.cpp
PROPERTIES COMPILE_OPTIONS "/arch:AVX2")
else()
set_source_files_properties(src/simd/kernels_intrin_avx2.cpp
PROPERTIES COMPILE_OPTIONS "-mavx2")
endif()
endif()
if(JOC_ENABLE_AVX512)
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_avx512.cpp)
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_AVX512=1)
if(MSVC)
set_source_files_properties(src/simd/kernels_intrin_avx512.cpp
PROPERTIES COMPILE_OPTIONS "/arch:AVX512")
else()
set_source_files_properties(src/simd/kernels_intrin_avx512.cpp
PROPERTIES COMPILE_OPTIONS "-mavx512f")
endif()
endif()
elseif(JOC_ARCH STREQUAL "aarch64")
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_NEON=1)
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_neon.cpp)
endif()
add_library(joc_simd OBJECT ${JOC_SIMD_SOURCES})
target_include_directories(joc_simd PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
target_compile_definitions(joc_simd PRIVATE ${JOC_SIMD_DEFINES})
# The objects are linked into the shared library as well, so they must be
# position independent even though an object library does not inherit that.
set_target_properties(joc_simd PROPERTIES POSITION_INDEPENDENT_CODE ON)
# Static form of the engine, used by the in-tree tools and tests so they can use
# internal modules without exporting them from the shared library.
if(JOC_BUILD_TESTS OR JOC_BUILD_DEVTOOLS)
add_library(joc_core_impl STATIC ${JOC_INTERNAL_SOURCES} ${JOC_REUSED_SOURCES}
$<TARGET_OBJECTS:joc_simd>)
target_include_directories(joc_core_impl
PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/include"
PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src"
)
target_link_libraries(joc_core_impl PUBLIC Threads::Threads)
endif()
add_library(joc_core SHARED
src/api/joc_api.cpp
src/api/joc_stream_api.cpp
src/api/joc_task_api.cpp
${JOC_INTERNAL_SOURCES}
${JOC_REUSED_SOURCES}
$<TARGET_OBJECTS:joc_simd>
)
target_include_directories(joc_core
PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/include"
PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src"
)
target_compile_definitions(joc_core PRIVATE JOC_BUILD_DLL)
target_link_libraries(joc_core PRIVATE Threads::Threads)
set_target_properties(joc_core PROPERTIES
OUTPUT_NAME "joc_core"
CXX_VISIBILITY_PRESET hidden
VISIBILITY_INLINES_HIDDEN YES
POSITION_INDEPENDENT_CODE YES
)
# The frontend compiles the UTF-8 path shim itself: it is small, and the shared
# library keeps its internals unexported.
add_executable(joc_cli src/cli/joc_cli.cpp src/foundation/fs_utf8.cpp)
target_link_libraries(joc_cli PRIVATE joc_core)
target_include_directories(joc_cli PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
# The installed CLI is in bin/ and the shared library in lib/, and ELF/Mach-O
# strip the build rpath on install, so the relative lookup has to be recorded
# here or `bin/joc_cli` cannot find `../lib/libjoc_core.*`.
if(UNIX)
if(APPLE)
set_target_properties(joc_cli PROPERTIES INSTALL_RPATH "@loader_path/../lib")
else()
set_target_properties(joc_cli PROPERTIES INSTALL_RPATH "$ORIGIN/../lib")
endif()
endif()
if(JOC_BUILD_TESTS)
enable_testing()
add_executable(joc_tests tests/test_core.cpp)
target_link_libraries(joc_tests PRIVATE joc_core_impl)
target_include_directories(joc_tests PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
add_test(NAME core COMMAND joc_tests)
# The public headers must stay valid C and C++: these targets exist to prove it.
add_library(joc_headers_c OBJECT tests/test_headers.c)
target_include_directories(joc_headers_c PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/include")
add_library(joc_headers_cpp OBJECT tests/test_headers.cpp)
target_link_libraries(joc_headers_cpp PRIVATE joc_core)
endif()
if(JOC_BUILD_DEVTOOLS)
add_executable(joc_dump devtools/joc_dump/main.cpp)
target_link_libraries(joc_dump PRIVATE joc_core joc_core_impl)
target_include_directories(joc_dump PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
# SOFA reader probe: the C++ side of devtools/checks/check_sofa.py.
add_executable(joc_sofa_field_probe devtools/sofa_field_probe/main.cpp)
add_executable(joc_dictionary_probe devtools/sofa_field_probe/dictionary_main.cpp)
target_link_libraries(joc_dictionary_probe PRIVATE joc_core_impl)
target_include_directories(joc_dictionary_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
target_link_libraries(joc_sofa_field_probe PRIVATE joc_core_impl)
target_include_directories(joc_sofa_field_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
# Rosella model probe: the C++ side of devtools/checks/check_rosella.py.
add_executable(joc_rosella_probe devtools/rosella_probe/main.cpp)
target_link_libraries(joc_rosella_probe PRIVATE joc_core_impl)
target_include_directories(joc_rosella_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
add_executable(joc_sofa_probe devtools/sofa_probe/main.cpp)
target_link_libraries(joc_sofa_probe PRIVATE joc_core_impl)
target_include_directories(joc_sofa_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
# Probe for the streaming surface: the acceptance harness and the usage
# example for joc_stream.h. Not shipped: a player integrates the library.
add_executable(joc_stream_probe devtools/joc_stream/main.cpp)
target_link_libraries(joc_stream_probe PRIVATE joc_core)
endif()
set(JOC_MSVC_TARGETS joc_core joc_cli joc_simd)
if(JOC_BUILD_TESTS OR JOC_BUILD_DEVTOOLS)
list(APPEND JOC_MSVC_TARGETS joc_core_impl)
endif()
if(JOC_BUILD_TESTS)
list(APPEND JOC_MSVC_TARGETS joc_tests joc_headers_c joc_headers_cpp)
endif()
if(JOC_BUILD_DEVTOOLS)
list(APPEND JOC_MSVC_TARGETS joc_rosella_probe joc_dump joc_stream_probe joc_sofa_probe joc_sofa_field_probe joc_dictionary_probe)
endif()
if(MSVC)
set(JOC_MSVC_FLAGS /W4 /permissive- /Zc:__cplusplus /utf-8 /EHsc /GR- /fp:precise)
# The writers use std::fopen for seekable, byte-exact output.
set(JOC_MSVC_DEFINES _CRT_SECURE_NO_WARNINGS)
# C4324 comes from the reused kernel's std::barrier member at /W4; it is
# pre-existing behaviour of a verbatim file, so the warning is silenced
# rather than the file edited.
set_source_files_properties(src/joc_core/eac3joc_core.cpp
PROPERTIES COMPILE_OPTIONS "/wd4324")
foreach(target IN LISTS JOC_MSVC_TARGETS)
set_property(TARGET ${target} PROPERTY
MSVC_RUNTIME_LIBRARY "MultiThreaded$<$<CONFIG:Debug>:Debug>")
# No /arch here on purpose: the baseline units keep the architecture's
# guaranteed ISA and only the src/simd `kernels_intrin_*` units carry a
# wider one (their flags are set per source file above).
target_compile_options(${target} PRIVATE ${JOC_MSVC_FLAGS}
$<$<CONFIG:Release>:/O2>
$<$<CONFIG:Release>:/Oi>)
target_compile_definitions(${target} PRIVATE ${JOC_MSVC_DEFINES})
endforeach()
target_link_options(joc_core PRIVATE /INCREMENTAL:NO /OPT:REF /OPT:ICF)
target_link_options(joc_cli PRIVATE /INCREMENTAL:NO)
else()
foreach(target IN LISTS JOC_MSVC_TARGETS)
target_compile_options(${target} PRIVATE -Wall -Wextra -Wpedantic -fno-fast-math
$<$<CONFIG:Release>:-O3>)
endforeach()
endif()
# Headers listed for IDE visibility.
target_sources(joc_core PRIVATE
include/joc_core.h
include/joc_stream.h
include/eac3joc_core.h
src/joc_bitstream/joc_huffman_tables.h
src/joc_core/qmf_tables.h
src/speaker/speaker_layouts.h
)
install(TARGETS joc_core joc_cli
RUNTIME DESTINATION bin
LIBRARY DESTINATION lib
ARCHIVE DESTINATION lib
)
+288 -144
View File
@@ -1,159 +1,303 @@
# JustOneCacophony — C++ Core
# JustOneCacophony — JOC
[中文](README.md) · [Mathematics](docs/math.en.md) · [Binaural rendering](docs/binaural.en.md) · [SIMD dispatch](docs/simd.en.md)
[中文版](README.md)
The C++ implementation of JustOneCacophony: an execution core for E-AC-3 JOC
bitstream parsing, object reconstruction and rendering. It extracts EMDF, ID14 JOC
parameters and ID11 OAMD metadata from E-AC-3 syncframes, combines them with the
core 5.1 PCM decoded by FFmpeg to rebuild the LFE and 15 object signals, and writes
ADM BWF, a WAV for a chosen speaker layout, or a binaural WAV using a compiled HRTF
directional field.
> JustOneCacophony is an experimental/test implementation of E-AC-3 JOC for studying JOC parsing, reconstruction, rendering, and the associated mathematics.
This is research code, not a complete, standard-conformant or production JOC
decoder. It covers the bitstream forms it implements and reports an explicit error
on unknown variants instead of pretending everything is in harmony.
The project can extract and parse EMDF, ID14 JOC parameters, and ID11 OAMD metadata from common E-AC-3 JOC streams. It combines those data with the core 5.1 PCM decoded by FFmpeg, reconstructs LFE plus 15 object channels, and writes ADM BWF, a WAV file for a selected speaker layout, or direct binaural stereo using a standard SOFA HRTF.
## Building
This is research code, not a complete, standards-compliant, or production-grade JOC decoder. It covers only the stream forms currently implemented. Unknown variants fail explicitly—because when the math goes wrong, all that may remain is the cacophony.
```powershell
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release
cmake --build build
ctest --test-dir build --output-on-failure
```
## Current features
Only MSVC (VS 2022, static CRT) is validated locally; Linux and macOS are built and
unit-tested by `.github/workflows/ci.yml`. Floating-point behaviour is part of the
byte-exact acceptance, so fast-math is never enabled: `/fp:precise` on MSVC,
`-fno-fast-math` elsewhere.
- Scan common contiguous EMDF containers in E-AC-3 sync frames.
- Parse ID14 dense / sparse JOC parameters, Huffman data, differential matrices, and `joc_clipgain`.
- Parse ID11 OAMD position updates and build object trajectories.
- Reconstruct LFE plus 15 object channels through analysis QMF, parameter interpolation, the object matrix, and inverse QMF.
- Write a 25-channel ADM BWF: a 10-channel 7.1.2 bed (silent except for LFE) plus 15 objects.
- Render directly to `2.0`, `3.1`, `5.1`, `7.1`, `5.1.2`, `5.1.4`, `7.1.2`, `7.1.4`, `9.1.4`, or `9.1.6`.
- Run public SOFA binaural rendering directly from `pcm16 + ID11/OAMD`, without a temporary ADM BWF.
- Keep the binaural DSP in float64/complex128, including 961-sample latency compensation, cross-frame state, and the room tail.
- Use a shared float32/PCM24 WAV writer and explicit PCM24 clipping policy for direct outputs.
- Use the NumPy backend or an optional C++20 core through `ctypes`; `auto` falls back to Python when the native library is unavailable.
- Read or write metadata sidecars and produce metadata, timing, and output reports.
Windows Release builds target AVX2 by default (`JOC_ENABLE_AVX2`, ON, see
`CMakeLists.txt`). That switch is itself part of the byte-exact acceptance — the
SHA-256 of every rendered output is identical — and buys 12.4% on the SOFA binaural
kernel and 1.8% on Rosella. The price is a runtime requirement: such a `joc_core.dll`
executes AVX2 instructions and dies on an illegal instruction on pre-2013 x86. There
is no runtime dispatch, so a binary is one or the other:
```powershell
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release -DJOC_ENABLE_AVX2=OFF
```
gives a baseline-ISA (SSE2) build that runs on any x86-64. Non-MSVC builds never
receive the flag.
### Paths and encoding
Every path inside the library is **UTF-8**, converted only at the OS boundary
(`src/foundation/fs_utf8.*`): on Windows through `std::filesystem::path` (UTF-16
inside) into `_wfopen`/`CreateProcessW`, and as plain bytes elsewhere. Command line
arguments are re-parsed from `GetCommandLineW` + `CommandLineToArgvW` on Windows and
the console code page is set to UTF-8, so non-ASCII paths (Japanese, Chinese, ...)
work for the input, the ffmpeg child process and the output files alike; a
non-ASCII path regression case runs in `ctest`.
## Artifacts
| Artifact | Purpose |
|---|---|
| `joc_core.dll` | The engine: bitstream parsing, JOC/OAMD, DSP, speaker and binaural rendering, ADM BWF/WAV writing, file task, streaming surface |
| `joc_cli.exe` | Command line frontend for file tasks |
| `include/joc_core.h` | Engine, telemetry and file-task interface (pure C) |
| `include/joc_stream.h` | Embedder-facing streaming push/pull interface (pure C, self-contained) |
## Command line
The arguments match the reference Python CLI exactly: the input is positional,
**ADM BWF is the default output**, `--speaker-layout` or `--binaural` selects the
other two modes, and without `-o` the result lands in `output/`.
```powershell
# Default: ADM BWF (inherently 24-bit, so there is no format option)
joc_cli "07. Gold Forever (2021 Master).m4a"
# -> output/07. Gold Forever (2021 Master).adm.wav
joc_cli input.m4a -o out/adm.wav # explicit output
# Speaker layout
joc_cli input.m4a --speaker-layout 5.1 # -> output/<name>.5.1.wav
joc_cli input.m4a --speaker-layout 7.1.4 --speaker-output out/714.wav --speaker-format int24
# Binaural (HRTF defaults to <exe>/HRTF/binaural.sofa, then <exe>/HRTF/binaural.personalized_headphone)
joc_cli input.m4a --binaural
joc_cli input.m4a --binaural --sofa-hrtf HRTF/other.sofa # another SOFA
joc_cli input.m4a --binaural --personalized-headphone # Rosella personalisation
# -> output/<name>.binaural.wav
# Other common switches
joc_cli input.m4a --duration 30 --gain-db -3 --trajectory-mode dense64
joc_cli input.eac3 --metadata-only --print-metadata summary # parse and print metadata only
```
`--speaker-format` / `--binaural-format` default to `float32`; an `int24` request that
would clip follows `--clip-action` (default `ask`; a non-interactive terminal must
pass `continue`, `float32` or `abort`). `--duration` is in **seconds**, and
`--object-delay-samples`, `--speaker-metadata-offset`, `--binaural-tail-seconds` and
`--binaural-tail-threshold` (1e-8, the binaural tail trim) keep the reference
defaults. A run always writes `<output>.report.json` (`--report-json` overrides it).
**Differences from the reference:** `--sofa-hrtf`, `--personalized-headphone`,
`--backend python` and the metadata sidecars (`--metadata-dir`, `--metadata-cache`,
`--metadata-backend sidecar`) are unavailable in this build and fail immediately
with an explanation instead of being ignored. This build adds `--bed` (pre-decoded
6-channel float32 PCM, which skips ffmpeg decoding), `--kernels`, `--work-dir`,
`--report-json`, `--dry-run` and `--quiet`.
## Library integration
The engine and the file task are exposed by `joc_core.h`: `joc_task_validate` /
`joc_task_execute` run a file task and report state events through a callback, and
`joc_task_result` carries frame counts, peak, byte count and SHA-256.
Players and decoder components use `joc_stream.h`: the caller pushes E-AC-3 bytes
and the matching core PCM (or already-rebuilt objects16) at its own pace and pulls
rendered PCM. Any chunking is allowed, and the result is byte-identical to the file
task.
```c
joc_stream_config config = {0};
config.struct_size = sizeof(config);
config.input = JOC_STREAM_IN_EAC3; /* or JOC_STREAM_IN_PCM_OBJECTS16 */
config.output = JOC_STREAM_OUT_SPEAKER; /* or BINAURAL / PCM_OBJECTS16 */
config.speaker_layout_name = "5.1";
joc_stream* stream = NULL;
joc_stream_create(&config, &stream);
/* loop: joc_stream_push(...) / joc_stream_pull(...) */
joc_stream_flush(stream);
joc_stream_destroy(stream);
```
Contract: state is instance-private, so streams coexist; push and pull on one
instance must come from the same thread; rendering is stateful, so **this version
offers no seek** - repositioning means decoding from the start of the stream. The
kernel latency is 961 samples for speaker/binaural output and `joc_stream_flush`
drains the binaural room tail.
## Layout
## Processing flow
```text
include/ public C ABI: joc_core.h (engine/file task), joc_stream.h (streaming),
eac3joc_core.h (upstream ABI)
src/ implementation: eac3_transport, emdf, joc_bitstream, joc_core, oamd,
timeline, speaker, binaural, hrtf, adm, io, telemetry, task, stream,
api, cli, simd
tests/ unit tests (CTest, self-contained, no external data)
docs/ mathematics, the binaural rendering flow and SIMD dispatch
M4A / E-AC-3
├─ FFmpeg extracts E-AC-3 and decodes the core 5.1 PCM
├─ EMDF → ID14 JOC parameters → object matrix
├─ core PCM → analysis QMF → parameter interpolation → inverse QMF
├─ ID11 OAMD → object positions and timing
└─ LFE + 15 objects
├─ 25ch ADM BWF
├─ speaker WAV for the selected layout
└─ direct ID11 timeline + SOFA HRTF → binaural WAV
```
Eight files under `src/` are byte-identical copies of the upstream JustOneCacophony
native library and are never edited (see [THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md)).
The Python and C++ backends follow the same mathematics for JOC object reconstruction and speaker rendering. The public SOFA binaural backend currently runs in Python; bitstream parsing, the OAMD timeline, and CLI behavior also remain in Python.
## Compatibility note
## Requirements
`object_delay_samples` defaults to **1473**, preserving the behaviour of the existing
implementation; it is a configurable field and changing it changes the OAMD/ADM time
alignment. Upstream investigation suggests the value should be 0; this project keeps
the current default to stay byte-identical.
- Python 3.10+
- NumPy 1.24+
- h5py 3.8+
- SciPy 1.10+
- A standalone FFmpeg executable; `ffmpeg-python` is not required. FFmpeg is discovered through `PATH` by default or selected with `--ffmpeg`. On startup the decoder options are probed with `ffmpeg -h decoder=eac3`: a missing E-AC-3 decoder or `-drc_scale` is a hard error, while a missing `-target_level` only fails when `--eac3-target-level` is used
- Optional: CMake and a C++20 toolchain to build the native core
## License
Install the Python dependency in a project-specific environment:
MIT, see [LICENSE](LICENSE). Third-party provenance and patent boundaries are in
[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md).
```powershell
python -m pip install -r requirements.txt
```
If FFmpeg is not on `PATH`:
```powershell
python main.py input.m4a --ffmpeg C:\path\to\ffmpeg.exe
```
## Usage
Write a 25-channel ADM BWF by default:
```powershell
python main.py input.m4a
```
Select a backend or output path:
```powershell
python main.py input.eac3 -o output.adm.wav --backend python
python main.py input.m4a --backend native --native-threads 2
python main.py input.m4a --native-library lib/eac3joc_core.dll
```
Write a speaker-layout WAV directly:
```powershell
python main.py input.m4a --speaker-layout 2.0 --speaker-format float32
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24
python main.py input.m4a --speaker-layout 7.1.2 --speaker-output output.7.1.2.wav
```
Write binaural stereo directly (ordinary objects are Near/Mid/Far only; Mid is
the default). The HRTF input accepts three sources:
```powershell
# 1) SOFA (defaults to HRTF/binaural.sofa, or an explicit path)
python main.py input.m4a --binaural
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa
# 2) Rosella .personalized_headphone (defaults to HRTF/binaural.personalized_headphone)
python main.py input.m4a --binaural --personalized-headphone
python main.py input.m4a --binaural --personalized-headphone C:\HRTF\subject.personalized_headphone
# 3) .jochrtf compiled cache
python main.py input.m4a --binaural --compiled-hrtf-cache C:\HRTF\subject.jochrtf
# Common options
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
--binaural-mode near
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
--hrtf-cache-policy disk
python main.py input.m4a --binaural --binaural-output output.binaural.wav
```
With none of the three specified, resolution tries, in order:
`HRTF/binaural.sofa`, the unique `.jochrtf` under `output/hrtf-cache`, then
`HRTF/binaural.personalized_headphone`; if none exist, an error asks for an
explicit path.
- `.sofa` is the portable source of truth; it can hold self-scanned or any
generic HRTF data.
- `.personalized_headphone` is a model produced by Dolby's official
personalization scan; its JSON parsing is implemented by this project
(`src/rosella_model.py`) and does not invoke any Dolby software.
- `.jochrtf` is a project-internal cache compiled from SOFA; it is disposable,
rebuildable, and written to `output/hrtf-cache` by default.
HRTF data lives under `HRTF/` (git-ignored): the default SOFA
`HRTF/binaural.sofa` and the default model
`HRTF/binaural.personalized_headphone`. Because the cache contains transformed
HRTF data, its use and redistribution remain subject to the source dataset's
terms. See [Binaural Rendering](docs/binaural.en.md) and
[Third-party notices](THIRD_PARTY_NOTICES.md) for format boundaries, formulas,
state, timing, and distribution considerations.
Speaker and binaural output share peak analysis, the WAV writer, and clipping policy. When PCM24 may clip in a non-interactive environment, select a policy explicitly:
```powershell
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action abort
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action float32
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action continue
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
--binaural-format int24 --clip-action abort
```
Metadata and diagnostics:
```powershell
python main.py input.m4a --print-metadata summary
python main.py input.m4a --metadata-only --print-metadata frames
python main.py input.m4a --metadata-cache metadata_cache
python main.py input.m4a --metadata-dir metadata_cache
```
### E-AC-3 decode-side dynamic range and level
By default FFmpeg applies the stream `dynrng` dynamic range compression when
decoding E-AC-3 (`-drc_scale 1`). The core 5.1 PCM is the input of JOC object
reconstruction, and `dynrng` is playback-time gain, so it is inherited linearly
by every object and every output (ADM, speaker, binaural). This tool therefore
decodes at **full dynamic range** by default:
```powershell
python main.py input.m4a # default: -drc_scale 0, full range
python main.py input.m4a --eac3-drc-scale 1 # reproduce consumer playback
python main.py input.m4a --eac3-drc-scale 0.5 # apply half of it
python main.py input.m4a --eac3-target-level -27 # dialnorm-referenced level
```
- `--eac3-drc-scale` (`0`–`6`, default `0`) maps to FFmpeg `-drc_scale`: the gain
of each E-AC-3 block is `dynrng factor ^ value`. `0` disables DRC, `1` is the
author's intent, and `>1` is asymmetric (loud parts fully compressed, quiet
parts enhanced).
- `--eac3-target-level` (`-31`–`0`, default `0` = off) maps to FFmpeg
`-target_level`: a static per-frame gain of about `target_level - dialnorm` dB,
independent of and stackable with `--eac3-drc-scale`. dialnorm is a per-stream
property (measured Apple Music Atmos streams are about `-18` to `-19` dB, so
`-27` is roughly `8`–`9` dB of attenuation).
- The level change is expected: compared with the FFmpeg default, measured
tracks move by `0` to `-2.15` dB peak and `0` to `-1.69` dB RMS (direction
depends on the stream `dynrng`), so `output_clip.peak` and the PCM24 clipping
decision in `.report.json` change accordingly.
- `--gain-db` is a static gain applied **after** reconstruction (float64 on the
binaural path) and is not the same thing as decode-side DRC, which is
block-varying; do not use `--gain-db` to cancel it.
- The `ffmpeg` field of `.report.json` records the FFmpeg version and the decode
options that were actually passed (`version`, `eac3_decode_options`).
### Binaural render mode
`--binaural-mode off|near|mid|far` selects the binaural render mode; the default
is `mid`, and both outputs share this single option:
- **Direct binaural rendering** (`--binaural`): `off` is rejected (error);
near/mid/far apply, defaulting to `mid`;
- **ADM BWF**: the low 3 binaural-render-mode bits of the last 15 JOC object
entries in DBMD segment 10 carry `off=0/near=1/far=2/mid=3`, leaving the first
10 bed entries unchanged; the default is `mid`, and `off` explicitly disables
the binaural metadata hint.
```powershell
python main.py input.m4a --binaural-mode mid
python main.py input.m4a --binaural-mode off # ADM BWF only: disable the DBMD hint
```
**The default `mid` is a human-specified rendering hint**; it is not original
binaural metadata extracted or recovered from the input E-AC-3 JOC bitstream,
nor does it represent the original mix's per-object binaural settings. The hint
does not change PCM, object trajectories, or direct speaker rendering. The
adjacent `.report.json` records `binaural_mode` (the mode name) and
`binaural_mode_value` (the ADM code; `null` for direct binaural output).
### OAMD time alignment
Object trajectories and direct speaker rendering both default to a metadata delay of `1473 samples`. This value describes the theoretical mapping between decoder-output PCM and OAMD updates. The speaker renderer retains its existing 32-sample control block, so the default update lands on effective block boundary `1472`:
```text
align32(1473) = 1472
```
Override the two paths with `--object-delay-samples` and `--speaker-metadata-offset`, respectively. The 1473-sample timing offset is distinct from the 640-value inverse-QMF filter/window state; 640 is a QMF state length, not a metadata delay.
The direct binaural path uses `--object-delay-samples`. Each ID11/OAMD event is
placed on an absolute sample timeline from its frame start, outer-subpayload
offset, and block offset, then shifted by that delay. Each 1536-sample input
frame is processed as three consecutive 512-sample blocks; the interpolated
position, direction, and profile are updated at each block's absolute starting
sample.
### Binaural calculation
See [Binaural Rendering Mathematics](docs/binaural.en.md) for QMF, hybrid processing, direction fields, distance, ITD, room processing, the 512-sample parameter updates above, and 961-sample latency compensation.
For all options:
```powershell
python main.py --help
```
Without `-o`, output still goes to `output/` at the repository root. The directory move intentionally preserves this behavior.
## Native core
The repository does not include native binaries by default. Download a prebuilt runtime for the current platform from a project Release, or build one locally, then place the runtime library under `lib/` at the repository root; create the directory if it is absent. To build it yourself, run CMake from the repository root:
```powershell
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
cmake --build build/cmake --config Release
cmake --install build/cmake --config Release
```
The runtime lookup order is:
1. `--native-library`;
2. `EAC3JOC_NATIVE_LIBRARY`;
3. the standard platform library name under `lib/`.
See the [native-core notes](docs/native.en.md) for ABI, state, and precision details.
## Repository layout
```text
JustOneCacophony/
├─ main.py command-line entry point
├─ src/ Python implementation modules
├─ native/ C/C++ acceleration core, C ABI, and required table data
├─ data/ Python runtime table data
├─ lib/ native runtime drop-in directory (create as needed)
├─ HRTF/ user HRTF data directory (create as needed, git-ignored)
├─ output/ output directory (create as needed; the .jochrtf cache defaults to its hrtf-cache subdirectory)
├─ docs/ math and native-core notes in both languages
├─ requirements.txt Python dependency
├─ README.md Chinese documentation
└─ README.en.md English documentation
```
## Mathematical implementation
The main documented stages are:
- dense JOC differential reconstruction and dequantization;
- parameter-band mapping to 64 QMF subbands;
- cross-frame parameter interpolation;
- analysis/inverse QMF, surround delay, and FIR state;
- the 1217-sample LFE delay;
- OAMD Q15 coordinate conversion;
- equal-power panning over target-layout regions;
- layout-dependent position compensation and sample-wise gain ramps;
- float32 and PCM24 output quantization;
- SOFA canonical import, 64-QMF/77-hybrid projection, `36×2×77` fifth-order fields, exactly-once delay/phase, project early/late room behavior, and special LFE.
See the [mathematical notes](docs/math.en.md) for the equations used by the decoding and rendering process.
## Known limitations
- Only the common contiguous EMDF transport is covered. Fragmented transport across multiple audio-block skip fields is not covered.
- The speaker and SOFA binaural paths currently cover ordinary point objects; extent, spread, diffuse, divergence, channel lock, and similar controls are outside the supported scope.
- OAMD trim elements are boundary-checked and skipped; warp, balance, and trim parameters are not applied to raw object trajectories or speaker rendering.
- Multi-data-point streams, uncommon band configurations, and unusual OAMD scheduling have less coverage than common 12-band, single-data-point material.
- A speaker limiter is outside the current primary formula.
- The SOFA importer currently supports the strict `SimpleFreeFieldHRIR` FIR subset; other SOFA conventions require explicit adapters.
- The binaural runtime is fixed at 48 kHz, fifth order, and one measurement-radius shell at a time; the public binaural backend defaults to the native accelerator and falls back to Python when the native library is unavailable.
- ADM output, native binaries, speaker layouts, and binaural models still need broader interoperability checks across platforms, players, and real material.
## Documentation
- [Mathematical notes](docs/math.en.md) · [中文](docs/math.md)
- [Native-core notes](docs/native.en.md) · [中文](docs/native.md)
- [Binaural rendering](docs/binaural.en.md) · [中文](docs/binaural.md)
+258 -123
View File
@@ -1,138 +1,273 @@
# JustOneCacophony — C++ Core
# JustOneCacophony — JOC
[English](README.en.md) · [数学说明](docs/math.md) · [双耳渲染](docs/binaural.md) · [SIMD 派发](docs/simd.md)
[English](README.en.md)
JustOneCacophony 的 C++ 实现:E-AC-3 JOC 码流解析、对象重建与渲染的执行内核。它从 E-AC-3
同步帧中提取 EMDF、ID14 JOC 参数与 ID11 OAMD 元数据,结合 FFmpeg 解码出的核心 5.1 PCM
重建 LFE 与 15 路对象 PCM,并输出 ADM BWF、指定扬声器布局的 WAV,或用编译好的 HRTF 方向场
直接输出双耳 WAV。
> JustOneCacophony 是一个 E-AC-3 JOC 的实验性 / 测试实现,用于研究 JOC 的解析、重建、渲染以及相关数学过程。
这是研究代码,不是完整、标准兼容或生产级的 JOC 解码器。它只覆盖已实现的码流形态,遇到未知
变体时明确报错,而不是假装一切都很和谐。
项目可以从常见 E-AC-3 JOC 码流中提取并解析 EMDF、ID14 JOC 参数和 ID11 OAMD 元数据,结合 FFmpeg 解码出的核心 5.1 PCM 重建 LFE 与 15 路对象 PCM,并输出 ADM BWF、指定扬声器布局的 WAV,或使用标准 SOFA HRTF 直接输出双耳 WAV。
## 构建
这是研究代码,不是完整、标准兼容或生产级的 JOC 解码器。它只覆盖当前已实现的码流形态;遇到未知变体时会明确报错,而不是假装一切都很和谐——如果哪里算错了,它可能就真的只剩 cacophony 了。
```powershell
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release
cmake --build build
ctest --test-dir build --output-on-failure
```
## 当前功能
Windows(MSVC / VS 2022,静态 CRT)、Linux 与 macOS 由 `.github/workflows/ci.yml` 同时构建并跑
单元测试;本地只验证 MSVC。浮点行为是逐字节验收的一部分,因此不启用 fast-math:MSVC 用
`/fp:precise`,其他编译器用 `-fno-fast-math`。
- 扫描 E-AC-3 同步帧中的常见连续 EMDF 容器;
- 解析 ID14 dense / sparse JOC 参数、Huffman 数据、差分矩阵与 `joc_clipgain`;
- 解析 ID11 OAMD 位置更新并生成对象轨迹;
- 通过 analysis QMF、参数插值、对象矩阵和 inverse QMF 重建 LFE + 15 路对象 PCM;
- 输出 25 声道 ADM BWF:10 声道 7.1.2 bed(除 LFE 外静音)+ 15 个对象;
- 直接渲染 `2.0`、`3.1`、`5.1`、`7.1`、`5.1.2`、`5.1.4`、`7.1.2`、`7.1.4`、`9.1.4`、`9.1.6`;
- 从 `pcm16 + ID11/OAMD` 直接运行公开 SOFA 双耳渲染,不生成临时 ADM BWF;
- 双耳 DSP 全程使用 float64/complex128,并保留 961-sample latency compensation、跨帧状态和 room 尾声;
- 直接输出统一支持 float32 或 PCM24 WAV,并在 PCM24 削波前提供明确处理策略;
- 使用 NumPy 后端,或通过 `ctypes` 调用可选的 C++20 原生核;`auto` 模式在原生库不可用时回退到 Python;
- 读取或写入 metadata sidecar,并生成元数据、运行时间和输出摘要。
Windows Release 默认带 `/arch:AVX2`(`JOC_ENABLE_AVX2`,默认 ON,见 `CMakeLists.txt`)。这个
开关是逐字节验收过的:所有渲染产物的 SHA-256 完全一致,换来 SOFA 双耳内核 12.4%、Rosella
1.8% 的提升。代价是运行要求——这样的 `joc_core.dll` 会执行 AVX2 指令,在 2013 年以前的 x86 上
直接非法指令退出。没有运行时派发,一个二进制只能二选一,所以:
```powershell
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release -DJOC_ENABLE_AVX2=OFF
```
得到基线 ISA(SSE2)、任何 x86-64 都能跑的产物。非 MSVC 构建永远不会带上这个开关。
### 路径与编码
库内部所有路径都是 **UTF-8**,只在系统边界转换(`src/foundation/fs_utf8.*`):Windows 上经
`std::filesystem::path`(内部 UTF-16)落到 `_wfopen`/`CreateProcessW`,其他平台直接是字节。
命令行参数在 Windows 上由 `GetCommandLineW` + `CommandLineToArgvW` 重新解析,控制台设为
UTF-8,因此日文/中文等非 ASCII 路径(含 ffmpeg 子进程与输出文件)都能正常工作;单元测试里有
一条非 ASCII 路径的回归用例守在 `ctest` 里。
## 产物
| 产物 | 说明 |
|---|---|
| `joc_core.dll` | 执行内核:码流解析、JOC/OAMD、DSP、扬声器与双耳渲染、ADM BWF/WAV 落盘、文件任务、流式接口 |
| `joc_cli.exe` | 文件任务命令行前端 |
| `include/joc_core.h` | 引擎、遥测与文件任务接口(纯 C) |
| `include/joc_stream.h` | 面向嵌入者的流式 push/pull 接口(纯 C,仅包含它即可) |
## 命令行
参数与上游 Python CLI 完全一致:输入是位置参数,**默认输出 25 通道 ADM BWF**,用
`--speaker-layout` 或 `--binaural` 切换到另外两种模式;未指定 `-o` 时产物落在 `output/`。
```powershell
# 默认:ADM BWF(本身就是 24-bit,没有也不需要格式参数)
joc_cli "07. Gold Forever (2021 Master).m4a"
# -> output/07. Gold Forever (2021 Master).adm.wav
joc_cli input.m4a -o out/adm.wav # 指定输出
# 扬声器布局
joc_cli input.m4a --speaker-layout 5.1 # -> output/<名称>.5.1.wav
joc_cli input.m4a --speaker-layout 7.1.4 --speaker-output out/714.wav --speaker-format int24
# 双耳(HRTF 默认取 <exe>/HRTF/binaural.sofa,其次 <exe>/HRTF/binaural.personalized_headphone)
joc_cli input.m4a --binaural
joc_cli input.m4a --binaural --sofa-hrtf HRTF/other.sofa # 换一个 SOFA
joc_cli input.m4a --binaural --personalized-headphone # Rosella 个性化模型
# -> output/<名称>.binaural.wav
# 其它常用开关
joc_cli input.m4a --duration 30 --gain-db -3 --trajectory-mode dense64
joc_cli input.eac3 --metadata-only --print-metadata summary # 只解析并打印元数据
```
`--speaker-format` / `--binaural-format` 默认 `float32`;`int24` 若会削波按 `--clip-action`
处理(默认 `ask`,非交互终端下需显式给出 `continue`/`float32`/`abort`)。`--duration` 以**秒**
为单位;`--object-delay-samples`、`--speaker-metadata-offset`、`--binaural-tail-seconds`、
`--binaural-tail-threshold`(默认 1e-8,双耳尾音裁切阈值)等默认值与上游一致。命令总是写出
`<输出>.report.json`(`--report-json` 可改路径)。
**与上游参数的差异**:`--sofa-hrtf`、`--personalized-headphone`、`--backend python` 以及
metadata sidecar(`--metadata-dir`/`--metadata-cache`/`--metadata-backend sidecar`)在本构建中
不可用,给出时立刻报错并说明原因,而不是静默忽略。本构建额外提供 `--bed`(已解码的 6 通道
float32 PCM,给出后不调用 ffmpeg 解码)、`--kernels`(滤波器组表路径)、`--work-dir`、
`--report-json`、`--dry-run`、`--quiet`。
## 库集成
引擎与文件任务使用 `joc_core.h`:`joc_task_validate` / `joc_task_execute` 跑一个文件任务并
通过回调返回状态事件,`joc_task_result` 给出帧数、峰值、字节数与 SHA-256。
播放器或解码组件使用 `joc_stream.h`:调用方按自己的节奏推入 E-AC-3 字节与对应的核心 PCM
(或已重建的 objects16),再拉取渲染后的 PCM;分块粒度任意,输出与文件任务逐字节一致。
```c
joc_stream_config config = {0};
config.struct_size = sizeof(config);
config.input = JOC_STREAM_IN_EAC3; /* 或 JOC_STREAM_IN_PCM_OBJECTS16 */
config.output = JOC_STREAM_OUT_SPEAKER; /* 或 BINAURAL / PCM_OBJECTS16 */
config.speaker_layout_name = "5.1";
joc_stream* stream = NULL;
joc_stream_create(&config, &stream);
/* 循环:joc_stream_push(...) / joc_stream_pull(...) */
joc_stream_flush(stream);
joc_stream_destroy(stream);
```
契约:状态实例私有、可并存;同一实例的 push/pull 必须在同一线程;渲染是有状态的,
因此**本版本不提供 seek**——定位需要从流起点重新解码。扬声器/双耳通路的内核延迟为
961 样本,`joc_stream_flush` 负责排空双耳房间尾音。
## 目录
## 处理流程
```text
include/ 公共 C ABI:joc_core.h(引擎/文件任务)、joc_stream.h(流式)、eac3joc_core.h(上游 ABI)
src/ 实现:eac3_transport、emdf、joc_bitstream、joc_core、oamd、timeline、speaker、
binaural、hrtf、adm、io、telemetry、task、stream、api、cli、simd
tests/ 单元测试(CTest,自足,不需要外部素材)
docs/ 数学说明、双耳渲染流程与 SIMD 派发
M4A / E-AC-3
├─ FFmpeg 提取 E-AC-3 并解码核心 5.1 PCM
├─ EMDF → ID14 JOC 参数 → 对象矩阵
├─ 核心 PCM → analysis QMF → 参数插值 → inverse QMF
├─ ID11 OAMD → 对象位置与时间轨迹
└─ LFE + 15 objects
├─ 25ch ADM BWF
├─ 指定布局的扬声器 WAV
└─ ID11 直接时间轴 + SOFA HRTF → 双耳 WAV
```
`src/` 下 8 个文件是 JustOneCacophony 原生库的逐字节副本,永不修改(见
[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md))。
Python 与 C++ 后端在 JOC 对象重建、扬声器渲染和公开 SOFA 双耳渲染中使用同一组
数学过程;native 双耳后端与 Python 参考实现逐值一致(差异 < 1e-9)。位流解析、
OAMD 时间轴和命令行逻辑在 Python 中。
## 兼容性说明
## 环境
`object_delay_samples` 默认 **1473**,与既有实现的行为保持一致;它是可配置字段,改动它会
改变 OAMD/ADM 时间对齐。上游调查认为该值应为 0,本项目为保持逐字节等价暂不改默认值。
- Python 3.10+
- NumPy 1.24+
- h5py 3.8+
- SciPy 1.10+
- 独立的 FFmpeg 可执行程序;不需要 `ffmpeg-python`。默认从 `PATH` 查找,也可通过 `--ffmpeg` 指定可执行文件路径。启动时会探测 `ffmpeg -h decoder=eac3`:缺 E-AC-3 解码器或 `-drc_scale` 直接报错,缺 `-target_level` 只在使用 `--eac3-target-level` 时报错
- 可选:支持 C++20 的 CMake 工具链,用于自行构建原生核
## 许可
建议在项目专用虚拟环境中安装依赖:
MIT,见 [LICENSE](LICENSE);第三方来源与专利边界见
[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md)。
```powershell
python -m pip install -r requirements.txt
```
如果 FFmpeg 不在 `PATH` 中:
```powershell
python main.py input.m4a --ffmpeg C:\path\to\ffmpeg.exe
```
## 使用方法
默认输出 25 声道 ADM BWF:
```powershell
python main.py input.m4a
```
选择后端或输出路径:
```powershell
python main.py input.eac3 -o output.adm.wav --backend python
python main.py input.m4a --backend native --native-threads 2
python main.py input.m4a --native-library lib/eac3joc_core.dll
```
直接输出扬声器 WAV:
```powershell
python main.py input.m4a --speaker-layout 2.0 --speaker-format float32
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24
python main.py input.m4a --speaker-layout 7.1.2 --speaker-output output.7.1.2.wav
```
直接输出双耳渲染 WAV(普通对象仅 Near/Mid/Far,默认 Mid)。HRTF 输入支持三种来源:
```powershell
# 1) SOFA(缺省取 HRTF/binaural.sofa,也可显式指定)
python main.py input.m4a --binaural
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa
# 2) Rosella .personalized_headphone(缺省取 HRTF/binaural.personalized_headphone)
python main.py input.m4a --binaural --personalized-headphone
python main.py input.m4a --binaural --personalized-headphone C:\HRTF\subject.personalized_headphone
# 3) .jochrtf 编译缓存
python main.py input.m4a --binaural --compiled-hrtf-cache C:\HRTF\subject.jochrtf
# 常用选项
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
--binaural-mode near
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
--hrtf-cache-policy disk
python main.py input.m4a --binaural --binaural-output output.binaural.wav
```
三者都不指定时的自动选择顺序:`HRTF/binaural.sofa` → `output/hrtf-cache` 下唯一的
`.jochrtf` → `HRTF/binaural.personalized_headphone`;都没有则报错并提示显式指定。
- `.sofa` 是可移植的 source of truth;可以是自行扫描或任何来源的通用 HRTF 数据。
- `.personalized_headphone` 是杜比官方软件个性化扫描得到的模型,其 JSON 解析由
本项目自行实现(`src/rosella_model.py`),不调用杜比软件。
- `.jochrtf` 是从 SOFA 编译出的项目内部 cache,可删除、可从 SOFA 重建,默认写在
`output/hrtf-cache`。
HRTF 数据统一放在 `HRTF/`(git 忽略):默认 SOFA `HRTF/binaural.sofa`、默认模型
`HRTF/binaural.personalized_headphone`。cache 含有源 HRTF 的变换数据,使用与再分发
仍受源数据许可约束;格式边界、计算公式、状态、时间轴及发布注意事项见
[双耳渲染](docs/binaural.md) 和 [第三方通知](THIRD_PARTY_NOTICES.md)。
扬声器和双耳输出共享峰值检查、writer 与削波策略。在非交互环境请求 PCM24 且可能削波时,需要显式选择处理方式:
```powershell
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action abort
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action float32
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action continue
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
--binaural-format int24 --clip-action abort
```
元数据与诊断:
```powershell
python main.py input.m4a --print-metadata summary
python main.py input.m4a --metadata-only --print-metadata frames
python main.py input.m4a --metadata-cache metadata_cache
python main.py input.m4a --metadata-dir metadata_cache
```
### E-AC-3 解码级动态范围与电平
FFmpeg 解码 E-AC-3 时默认施加码流 `dynrng` 动态范围压缩(`-drc_scale 1`)。核心 5.1 PCM 是 JOC 对象重建的输入,而 `dynrng` 属于回放期增益,会被线性继承到全部对象与成品(ADM/扬声器/双耳),因此本工具默认按**全动态范围**解码:
```powershell
python main.py input.m4a # 默认:-drc_scale 0,全动态范围
python main.py input.m4a --eac3-drc-scale 1 # 复现消费者回放(码流作者意图)
python main.py input.m4a --eac3-drc-scale 0.5 # 施加一半
python main.py input.m4a --eac3-target-level -27 # 按码流 dialnorm 归一化电平
```
- `--eac3-drc-scale`(`0`~`6`,默认 `0`)对应 FFmpeg 的 `-drc_scale`:每个 E-AC-3 block 的增益为 `dynrng 因子 ^ 该值`。`0` 关闭 DRC;`1` 为码流作者意图;`>1` 非对称(响处全压、轻处增强)。
- `--eac3-target-level`(`-31`~`0`,默认 `0` 不施加)对应 FFmpeg 的 `-target_level`:按每帧 dialnorm 施加静态增益,约 `target_level - dialnorm` dB,与 `--eac3-drc-scale` 相互独立、可叠加。dialnorm 是逐码流属性(实测 Apple Music Atmos 流约 `-18`~`-19` dB,故 `-27` 约等于衰减 `8`~`9` dB)。
- 电平变化是预期的:与 FFmpeg 默认值相比,实测曲目峰值变化 `0`~`-2.15` dB、RMS `0`~`-1.69` dB(方向取决于码流 `dynrng`),`.report.json` 的 `output_clip.peak` 与 int24 削波判定会随之变化。
- `--gain-db` 是**重建之后**的静态增益(双耳路径 float64),与解码级 DRC 不是一回事;解码级 DRC 是按 block 时变的,不要用 `--gain-db` 去抵消它。
- `.report.json` 的 `ffmpeg` 字段记录 FFmpeg 版本与实际下发的解码选项(`version`、`eac3_decode_options`)。
### 双耳渲染模式
`--binaural-mode off|near|mid|far` 选择双耳渲染模式,默认 `mid`,两种输出共用这一个选项:
- **直接双耳渲染**(`--binaural`):`off` 不可用(报错),near/mid/far 生效,默认 `mid`;
- **ADM BWF**:DBMD segment 10 中后 15 个 JOC 对象的 binaural render mode 写
`off=0/near=1/far=2/mid=3`,前 10 个 bed 保持不变,默认 `mid`;`off` 用于显式
关闭双耳元数据提示。
```powershell
python main.py input.m4a --binaural-mode mid
python main.py input.m4a --binaural-mode off # 仅 ADM BWF:关闭 DBMD 双耳提示
```
**默认 `mid` 是本工具人为指定的渲染提示**,不是从输入 E-AC-3 JOC 码流中提取或
还原的原始双耳元数据,也不代表原始混音中各对象的双耳设置。该提示不改变 PCM、
对象轨迹或直接扬声器渲染。输出旁的 `.report.json` 用 `binaural_mode`(模式名)
和 `binaural_mode_value`(ADM 编码值,直接双耳输出时为 `null`)记录。
### OAMD 时间对齐
对象轨迹和直接扬声器渲染的 metadata delay 默认均为 `1473 samples`。该值描述 decoder 输出 PCM 与 OAMD 更新之间的理论时间映射;扬声器 renderer 仍使用现有的 32-sample control block,因此默认更新的实际 block boundary 为 `1472`:
```text
align32(1473) = 1472
```
可分别用 `--object-delay-samples` 和 `--speaker-metadata-offset` 覆盖默认值。这里的 1473 不应与 inverse-QMF 的 640 项 filter/window state 混淆;后者是 QMF 状态长度,不是 metadata delay。
直接双耳路径使用 `--object-delay-samples`。每个 ID11/OAMD event 先按 frame start、
outer subpayload offset 与 block offset 落到绝对 sample timeline,再加该 delay;每个
1536-sample 输入帧按三个连续 512-sample block 处理,并在每块的绝对起始 sample
查询插值后的位置、更新方向和 profile。
### 双耳计算
双耳路径的 QMF、hybrid、方向场、距离、ITD、room、上述 512-sample 参数更新和 961-sample 延迟补偿见[双耳渲染数学](docs/binaural.md)。
更多参数可查看:
```powershell
python main.py --help
```
未指定 `-o` 时,输出仍写入仓库根目录的 `output/`。这是文件移动后特意保持的原有行为。
## 原生核
仓库默认不附带原生二进制。可以从项目 Release 下载适合当前平台的预构建运行库,或自行构建,然后把运行库直接放入仓库根目录的 `lib/`;若该目录不存在,创建即可。自行构建时可从仓库根目录使用 CMake:
```powershell
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
cmake --build build/cmake --config Release
cmake --install build/cmake --config Release
```
运行时查找顺序为:
1. `--native-library`;
2. `EAC3JOC_NATIVE_LIBRARY`;
3. `lib/` 下当前平台的标准库文件名。
详细 ABI、状态与精度说明见[原生核说明](docs/native.md)。
## 目录结构
```text
JustOneCacophony/
├─ main.py 命令行启动入口
├─ src/ Python 实现模块
├─ native/ C/C++ 加速核、C ABI 与必要表数据
├─ data/ Python 运行时表数据
├─ lib/ 原生运行库投放目录(按需创建)
├─ HRTF/ 用户 HRTF 数据目录(按需创建,git 忽略)
├─ output/ 输出目录(按需创建;.jochrtf 缓存默认在其 hrtf-cache 子目录)
├─ docs/ 数学与原生核文档(中英文)
├─ requirements.txt Python 依赖
├─ README.md 中文说明
└─ README.en.md English documentation
```
## 数学实现
核心过程包括:
- dense JOC 差分还原与去量化;
- 参数带到 64 个 QMF 子带的映射;
- 跨帧参数插值;
- analysis / inverse QMF、环绕声道延迟与 FIR 状态;
- LFE 1217-sample 延迟;
- OAMD Q15 坐标转换;
- 基于目标布局 region 的等功率声像;
- 布局位置补偿与逐样本增益斜坡;
- float32 与 PCM24 输出量化;
- SOFA canonical importer、64-QMF/77-hybrid 投影、`36×2×77` 五阶方向 field、exactly-once delay/phase、项目 early/late room 与 special LFE。
解码与渲染过程使用的公式见[数学说明](docs/math.md)。
## 已知限制
- 当前只覆盖常见 continuous EMDF transport;跨多个 audio-block skip field 的碎片化 transport 尚未覆盖。
- 扬声器与 SOFA 双耳路径当前只覆盖普通点对象;extent、spread、diffuse、divergence、channel lock 等对象控制不在支持范围内。
- OAMD trim element 会按声明边界校验并跳过;warp、balance 和 trim 参数不应用于当前原始对象轨迹或扬声器渲染。
- 多数据点、少见参数带配置和特殊 OAMD 调度的覆盖度低于常见 12-band、单数据点素材。
- 扬声器 limiter 不属于当前实现的主公式。
- SOFA importer 当前严格支持 `SimpleFreeFieldHRIR` FIR;其它 SOFA convention 需要显式 adapter。
- 双耳 runtime 固定 48 kHz、五阶和一次选择一个 measurement-radius shell;公开双耳默认走 native 加速,原生库不可用时自动回退 Python。
- ADM 输出、原生库、扬声器布局和双耳模型仍需在更多平台、播放器与真实素材上确认互操作性。
## 文档
- [数学说明](docs/math.md) · [English](docs/math.en.md)
- [原生核说明](docs/native.md) · [English](docs/native.en.md)
- [双耳渲染](docs/binaural.md) · [English](docs/binaural.en.md)
+9 -7
View File
@@ -1,24 +1,26 @@
# Third-party notices / 第三方通知
本文件记录 64-QMF / 77-hybrid 滤波器组表(`src/hrtf/public_filterbank.h`、
`src/joc_core/qmf_tables.h`)与 JOC Huffman 表(`src/joc_bitstream/joc_huffman_tables.h`)
本文件记录 `data/rosella_kernels.npz`(`src/public_filterbank.py` 使用的滤波器组表)
的公开标准来源,以及 HRTF 数据与专利的边界说明。
## 公开标准来源
64-QMF → 77-hybrid 结构与 13-tap 低带 prototype 定义于
[3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf)
第 5.2.2 节(Table 1 的 $Q=8$/$Q=4$ 系数,delay 6):
第 5.2.2 节(Table 1 的 $Q=8$/
$Q=4$ 系数,delay 6):
$$G_q^p[n] = g^p[n]\cdot\exp\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr)$$
64-band QMF analysis 即 ISO/IEC 14496-3/AMD1:2003 第 4.B.18.2 节的 MPEG-4
AAC/SBR 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap prototype 的多相重排:
AAC/SBR 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap prototype 的
多相重排:
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t}$$
QMF synthesis 表为 analysis 多相矩阵 $\mathbf{A}$ 的因果左逆
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$($\mathbf{P}$ 为 577-sample 延迟置换;
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$(
$\mathbf{P}$ 为 577-sample 延迟置换;
全链 $961 = 577 + 6\times64$),rank-4 分解存储:
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
@@ -31,8 +33,8 @@ $$Y_p = \sum_{q\in C_p}\Bigl(\mathrm{Re}X_q + j\,s_q\,\mathrm{Im}X_q\Bigr),\qqua
## HRTF 数据与 `.jochrtf`
`.jochrtf` 含有特定源 SOFA/HRTF 数据集的变换系数与 delay;其使用、复制与再分发仍受源
数据集许可约束,权限不明确时应作为私有 cache 保存。本仓库不分发任何 HRTF 数据集。
`.jochrtf` 含有特定源 SOFA/HRTF 数据集的变换系数与 delay;其使用、复制与再分发
仍受源数据集许可约束,权限不明确时应作为私有 cache 保存。
## 专利说明
+83
View File
@@ -0,0 +1,83 @@
# Python runtime tables
[中文](README.md)
This directory contains static production tables. It does not contain user HRTFs.
`tables.npz` contains the JOC core decoding tables:
```text
analysis_window float64[10,64]
qmf5_window float64[640]
joc_huff_code_coarse_generic int64[95,2]
joc_huff_code_fine_generic int64[191,2]
joc_huff_code_coarse_coeff_sparse int64[95,2]
joc_huff_code_fine_coeff_sparse int64[191,2]
joc_huff_code_5ch_pos_index_sparse int64[4,2]
joc_huff_code_7ch_pos_index_sparse int64[6,2]
```
`src/joc_qmf.py` loads the QMF tables, while `src/joc_decode.py` loads the JOC Huffman trees. Python does not read C/C++ headers under `native/`.
The corresponding native data are stored in `native/src/qmf_tables.h` and `native/src/joc_huffman_tables.h`. Changes on either side should update the other and be checked for value-by-value agreement.
## Binaural rendering tables
`rosella_kernels.npz` contains the fixed 64-QMF/77-hybrid tables used by the
public SOFA binaural path:
```text
format_version little-endian int32[1]
qmf_analysis_coefficients float32[64,10]
hybrid_analysis_low_kernel float32[3,2,13,16,2]
hybrid_synthesis_indices int16[154,4]
hybrid_synthesis_values float32[154]
qmf_synthesis_basis float64[64,4,128]
qmf_synthesis_taps float64[64,10,4]
```
The float32 table values are promoted to float64 when loaded.
`src/public_filterbank.py` verifies the archive and every array by SHA-256.
Those hashes, the table version, and the 77 reference band-center values all
participate in the `.jochrtf` cache key. The full analysis/synthesis latency is
961 samples.
The packaged tables implement publicly standardized filter banks, computable
from the following formulas.
The 64-QMF → 77-hybrid structure, the 13-tap low-band prototypes, and their
half-bin complex modulation are defined in
[3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf),
Section 5.2.2 (Table 1 $Q=8$/$Q=4$ coefficients, delay 6):
$$G_q^p[n] = g^p[n]\cdot\exp\!\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr),\qquad n=0,\dots,12$$
The 64-band QMF analysis is the MPEG-4 AAC/SBR 64 complex QMF analysis bank of
ISO/IEC 14496-3/AMD1:2003, subclause 4.B.18.2; the packaged $64\times10$ table
is the polyphase reordering of the public 640-tap prototype $c_0,\dots,c_{639}$:
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t},\qquad r=0,\dots,63,\ t=0,\dots,9$$
The QMF synthesis table is the causal left inverse of the analysis polyphase
matrix $\mathbf{A}$, i.e. the solution of $\mathbf{A}\,\mathbf{W}=\mathbf{P}$
($\mathbf{P}$ is the 577-sample delay permutation; total latency
$961 = 577 + 6\times64$), stored as a rank-4 factorization:
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
The hybrid synthesis table is the 77→64 recombination: identity for the high
bands, $Y_{3+b}=X_{16+b}$, and for the low bands ($C_p$ is the $8+4+4$ child
partition):
$$Y_p = \sum_{q\in C_p}\Bigl(\operatorname{Re}X_q + j\,s_q\,\operatorname{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
The same values also appear in other public implementations of these standards
(for example FFmpeg's `aacps_tablegen.h` and `aacsbrdata.h`).
Public availability of a standard does not by itself grant permission to
practice related patent claims.
SOFA is the user-visible source of truth. A `.jochrtf` file is a disposable JOC
compiled HRTF cache that can be rebuilt from SOFA. The cache contains
transformed source-HRTF data and remains subject to the source SOFA/HRTF
dataset's licence and redistribution restrictions.
+73
View File
@@ -0,0 +1,73 @@
# Python 运行时表
[English](README.en.md)
本目录保存 Python 生产路径使用的静态表数据,不保存用户 HRTF。
`tables.npz` 保存 JOC 核心解码表:
```text
analysis_window float64[10,64]
qmf5_window float64[640]
joc_huff_code_coarse_generic int64[95,2]
joc_huff_code_fine_generic int64[191,2]
joc_huff_code_coarse_coeff_sparse int64[95,2]
joc_huff_code_fine_coeff_sparse int64[191,2]
joc_huff_code_5ch_pos_index_sparse int64[4,2]
joc_huff_code_7ch_pos_index_sparse int64[6,2]
```
`src/joc_qmf.py` 读取 QMF 表,`src/joc_decode.py` 读取 JOC Huffman 树。Python 不读取 `native/` 下的 C/C++ 头文件。
原生侧对应数据分别位于 `native/src/qmf_tables.h` 与 `native/src/joc_huffman_tables.h`。修改任何一侧时,应同步更新另一侧并进行逐值一致性检查。
## 双耳渲染表
`rosella_kernels.npz` 保存公开 SOFA 双耳路径使用的 64-QMF/77-hybrid 固定表:
```text
format_version little-endian int32[1]
qmf_analysis_coefficients float32[64,10]
hybrid_analysis_low_kernel float32[3,2,13,16,2]
hybrid_synthesis_indices int16[154,4]
hybrid_synthesis_values float32[154]
qmf_synthesis_basis float64[64,4,128]
qmf_synthesis_taps float64[64,10,4]
```
float32 表值载入后提升为 float64。`src/public_filterbank.py` 在读取时校验 archive
及每个数组的 SHA-256;这些 hash、table version 和 77 个 band-center 参考值共同进入
`.jochrtf` cache key。analysis/synthesis 全链 latency 为 961 samples。
打包表实现的是公开标准化的滤波器组,各表可由如下公式计算。
64-QMF → 77-hybrid 结构、13-tap 低带 prototype 与半 bin 复调制定义于
[3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf)
第 5.2.2 节(Table 1 的 $Q=8$/$Q=4$ 系数,delay 6):
$$G_q^p[n] = g^p[n]\cdot\exp\!\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr),\qquad n=0,\dots,12$$
64-band QMF analysis 即 ISO/IEC 14496-3/AMD1:2003 第 4.B.18.2 节的 MPEG-4
AAC/SBR 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap prototype
$c_0,\dots,c_{639}$ 的多相重排:
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t},\qquad r=0,\dots,63,\ t=0,\dots,9$$
QMF synthesis 表为上述 analysis 多相矩阵 $\mathbf{A}$ 的因果左逆,即求解
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$($\mathbf{P}$ 为 577-sample 延迟置换;
全链 $961 = 577 + 6\times64$),以 rank-4 分解形式存储:
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
hybrid synthesis 表为 77→64 重组:高频带恒等 $Y_{3+b}=X_{16+b}$;低频带
($C_p$ 为 $8+4+4$ 子带划分):
$$Y_p = \sum_{q\in C_p}\Bigl(\operatorname{Re}X_q + j\,s_q\,\operatorname{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
相同数值可在 FFmpeg(`aacps_tablegen.h`、`aacsbrdata.h`)等公开实现中查到。
标准可公开获取不等于获准实施相关专利。
`.sofa` 是用户可见的 source of truth;`.jochrtf` 是可删除、可从 SOFA 重建的
JOC compiled HRTF cache。cache 含有源 HRTF 的变换数据,仍受源 SOFA/HRTF
数据集的许可与再分发限制约束。
Binary file not shown.
BIN
View File
Binary file not shown.
+59 -21
View File
@@ -24,29 +24,61 @@ SOFA FIR
## Inputs
The binaural backend consumes a compiled directional field (a JOC compiled HRTF cache,
`.jochrtf`) plus the shared filterbank tables. Both are read-only inputs: this library
performs no parsing or conversion of measurement data formats.
The CLI has three mutually exclusive HRTF input sources; with none given, a
default rule resolves the input:
```powershell
joc_cli input.m4a --binaural `
--compiled-hrtf-cache path\to\subject.jochrtf `
--kernels data\rosella_kernels.npz `
--binaural-mode mid --binaural-tail-seconds 5.0
# 1) SOFA: defaults to HRTF/binaural.sofa, or an explicit path
python main.py input.m4a --binaural
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa
# 2) Rosella .personalized_headphone: defaults to HRTF/binaural.personalized_headphone
python main.py input.m4a --binaural --personalized-headphone
python main.py input.m4a --binaural --personalized-headphone C:\HRTF\subject.personalized_headphone
# 3) .jochrtf: explicitly load a compiled cache
python main.py input.m4a --binaural `
--compiled-hrtf-cache C:\HRTF\subject.jochrtf
# Optional: create/reuse a transparent disk cache for SOFA
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
--hrtf-cache-policy disk
```
The library exposes the same fields: `joc_task_config` for a file task,
`joc_stream_config` for streaming, where `hrtf_path`, `kernels_path`,
`binaural_mode` and `binaural_tail_seconds` configure the binaural path.
The default order is `HRTF/binaural.sofa`, then the unique `.jochrtf` under
`output/hrtf-cache`, then `HRTF/binaural.personalized_headphone`; if none of
the three exist, an error asks for an explicit path. Multiple `.jochrtf` files
under `output/hrtf-cache` are also an error requiring an explicit choice.
The `.jochrtf` file is an **input**, not a product of this library: compiling it from
SOFA data or measurements belongs to the toolchain and is decoupled from this
repository. Loading validates the member set, dtypes and shapes, C-contiguity,
CRC-32 and a payload hash recomputed over the members (see `.jochrtf` below).
The `.personalized_headphone` JSON parsing is implemented by this project
(`src/rosella_model.py`) and does not invoke any Dolby software.
`binaural_mode` is `near`, `mid` or `far` and selects one of the backend's three
preset parameter sets; `binaural_tail_seconds` sets the room-tail length drained on
flush (default 5.0 s).
`--hrtf-cache-policy` accepts `none`, `memory`, or `disk`. The default is
`memory`; neither `none` nor `memory` creates a file. `disk` writes to
`output/hrtf-cache` by default, or to `--hrtf-cache-dir`. `--hrtf-radius-m`
selects the nearest measurement-radius shell.
The Python API also uses explicit factories:
```python
from sofa_binaural_backend import SofaBinauralBackend
renderer = SofaBinauralBackend.from_sofa(
"subject.sofa",
source_count=16,
default_profile="mid",
cache_policy="memory",
)
cached = SofaBinauralBackend.from_compiled_cache(
"subject.jochrtf",
source_count=16,
default_profile="mid",
)
```
The factories never guess a format from an unknown suffix: SOFA and `.jochrtf`
always use distinct loaders.
## Binaural render mode
@@ -222,10 +254,16 @@ late sends, the unitary FDN, the 120–180 Hz cosine-squared LFE low-pass, and r
calibration are JOC project-defined behavior, not constants published by SOFA or
Dolby.
The binaural renderer is implemented inside this library: the filterbank, the SH
directional-field evaluation, the per-object early/direct histories and the shared
FDN all run in `joc_core` (the `ejoc_sofa_binaural_*` kernel), consuming a compiled
directional field and 512-sample metadata updates.
The public SOFA binaural renderer defaults to the C++20 native core under
`--backend auto/native` (`ejoc_sofa_binaural_*` in `lib/eac3joc_core.dll`): the
filterbank, the SH direction-field evaluation, the per-object direct/early
histories and the shared FDN all run natively, while Python only compiles the
SOFA source and issues the per-512-sample metadata updates. When the native
library is unavailable the renderer falls back to the Python/NumPy reference
implementation; the two agree to better than 1e-9. `--backend python` forces
the Python backend.
`--backend` still selects native/Python JOC reconstruction and speaker rendering;
native acceleration for the public binaural DSP is outside the current API.
## Technical references and rights boundary
+50 -16
View File
@@ -21,25 +21,57 @@ SOFA FIR
## 输入接口
双耳后端消费一个已编译的方向场(JOC compiled HRTF cache,`.jochrtf`)与共享滤波器组表;
两者都是只读输入,本库不做任何测量数据格式的解析或转换:
CLI 有三个互斥的 HRTF 输入来源;都不指定时按默认规则自动选择:
```powershell
joc_cli input.m4a --binaural `
--compiled-hrtf-cache path\to\subject.jochrtf `
--kernels data\rosella_kernels.npz `
--binaural-mode mid --binaural-tail-seconds 5.0
# 1) SOFA:缺省取 HRTF/binaural.sofa,也可显式指定
python main.py input.m4a --binaural
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa
# 2) Rosella .personalized_headphone:缺省取 HRTF/binaural.personalized_headphone
python main.py input.m4a --binaural --personalized-headphone
python main.py input.m4a --binaural --personalized-headphone C:\HRTF\subject.personalized_headphone
# 3) .jochrtf:显式读取预编译 cache
python main.py input.m4a --binaural `
--compiled-hrtf-cache C:\HRTF\subject.jochrtf
# 可选:SOFA 透明生成/复用磁盘 cache
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
--hrtf-cache-policy disk
```
库接口使用同一组字段:文件任务用 `joc_task_config`,流式用 `joc_stream_config`,
其中 `hrtf_path`、`kernels_path`、`binaural_mode`、`binaural_tail_seconds` 决定双耳通路。
默认选择顺序:`HRTF/binaural.sofa` → `output/hrtf-cache` 下唯一的 `.jochrtf` →
`HRTF/binaural.personalized_headphone`;三者都没有时报错并提示显式指定。
`output/hrtf-cache` 下有多个 `.jochrtf` 时同样报错,要求显式选择。
`.jochrtf` 是**输入**而不是本库的产物:从 SOFA 或测量数据编译该缓存属于工具链的职责,
与本仓库解耦。缓存加载时会校验成员集合、dtype 与形状、C 连续性、CRC-32 以及按成员重算的
载荷哈希(见下文 `.jochrtf` 一节)。
`.personalized_headphone` 的 JSON 解析由本项目自行实现(`src/rosella_model.py`),
不调用任何杜比软件。
`binaural_mode` 取 `near`、`mid`、`far`,选择后端的三组预置参数;`binaural_tail_seconds`
决定 flush 时排空的房间尾音长度(默认 5.0 s)。
`--hrtf-cache-policy` 可取 `none`、`memory`、`disk`。默认是 `memory`;`none` 和
`memory` 都不会创建磁盘文件。`disk` 默认写入 `output/hrtf-cache`,也可用
`--hrtf-cache-dir` 指定。`--hrtf-radius-m` 选择距离目标最近的 measurement shell。
Python API 使用显式 factory:
```python
from sofa_binaural_backend import SofaBinauralBackend
renderer = SofaBinauralBackend.from_sofa(
"subject.sofa",
source_count=16,
default_profile="mid",
cache_policy="memory",
)
cached = SofaBinauralBackend.from_compiled_cache(
"subject.jochrtf",
source_count=16,
default_profile="mid",
)
```
文件工厂不会按“未知后缀”猜格式:SOFA 和 `.jochrtf` 始终走不同 loader。
## 双耳渲染模式
@@ -192,9 +224,11 @@ Near/Mid/Far、equal-power direct level、六面 shoebox 一阶 image source、l
unitary FDN、LFE 120–180 Hz cosine-squared 低通及 room calibration 都是 JOC
项目定义行为,不是 SOFA 或 Dolby 公布常数。
双耳渲染在本库内实现:filterbank、SH 方向场求值、逐对象 early/direct 历史与共享 FDN
全部执行于 `joc_core`(`ejoc_sofa_binaural_*` 内核),对外只消费编译好的方向场与
512-sample 粒度的元数据更新。
公开 SOFA 双耳渲染在 `--backend auto/native` 下默认走 C++20 原生核
(`lib/eac3joc_core.dll` 的 `ejoc_sofa_binaural_*` 接口:filterbank、SH 方向场求值、
逐对象 early/direct 历史与共享 FDN 全部在原生侧执行,Python 只做 SOFA 编译与每
512-sample 的元数据更新);原生库不可用时自动回退 Python/NumPy 参考实现,两者
逐值一致(差异 < 1e-9)。`--backend python` 强制使用 Python 后端。
## 技术引用与权利边界
+178
View File
@@ -0,0 +1,178 @@
# JustOneCacophony native-core notes
[中文](native.md) · [Back to README](../README.en.md)
## 1. Responsibility boundary
`native/` contains only the state-heavy, frequently called DSP and speaker-rendering kernels. High-level EMDF/JOC/OAMD parsing, error reporting, ADM assembly, and the CLI remain in Python.
Python calls a C ABI through the standard-library `ctypes` module. The native core does not use pybind11, Cython, FFTW, MKL, or OpenMP. It is an optional acceleration path and does not expand the set of supported stream variants.
Main files:
```text
native/include/eac3joc_core.h C ABI
native/src/eac3joc_core.cpp JOC/QMF object reconstruction
native/src/speaker_renderer.cpp object-to-speaker rendering
native/src/qmf_tables.h QMF tables
native/src/speaker_layouts.h layout tables
native/src/joc_huffman_tables.h JOC Huffman tables
src/native_renderer.py JOC ctypes bridge
src/speaker_native_renderer.py speaker ctypes bridge
```
## 2. JOC rendering ABI
An opaque renderer owns all cross-frame state. Its main call is:
```c
int ejoc_renderer_process(
ejoc_renderer_handle handle,
const float* bed5_planar, /* [5][1536] */
const float* lfe, /* [1536] or NULL */
uint32_t object_mask,
const uint8_t* n_bands, /* [15] */
const uint8_t* n_dpoints, /* [15] */
const uint8_t* slope_idx, /* [15] */
const uint8_t* offset_ts, /* [15][2] */
const double* dq, /* [15][2][5][23] */
double clipgain,
float phase_new,
float output_scale,
float* output16_planar); /* [16][1536] */
```
Python performs JOC Huffman decoding, differential reconstruction, and dequantization before the call, with the dense and sparse syntaxes sharing one entry point. The native core consumes the already dequantized `dq` in double precision, and both syntaxes have the same layout at that ABI.
Thread control is exposed as:
```c
int ejoc_renderer_set_threads(ejoc_renderer_handle handle, uint32_t total_threads);
uint32_t ejoc_renderer_thread_count(ejoc_renderer_handle handle);
```
`total_threads` includes the calling thread. Frames must be submitted sequentially to one renderer instance; the instance may parallelize work across objects and analysis channels.
## 3. Cross-frame state
Each JOC renderer stores:
- analysis FIFO: `double[5][9][64]`;
- L/R/C analysis delay: `float[3][10][64]`;
- Ls/Rs QMF delay: `complex<double>[2][10][64]`;
- Ls/Rs band-0 FIR history: `complex<double>[2][20]`;
- previous matrix interpolation values: `double[15][5][64]`;
- inverse-QMF state: `double[15][640]`;
- LFE delay: `double[1217]`.
This state belongs to the renderer instance. Processing cannot be arbitrarily segmented or reordered without a corresponding state checkpoint.
## 4. FFT, QMF, and precision
The native core contains a fixed 64-point radix-2 complex FFT:
- analysis QMF uses a forward FFT followed by division by 64;
- inverse QMF uses the fixed reorder, rotation, and 640-value active-window state;
- no external FFT library is called.
The JOC path uses:
- float32 core-PCM input;
- double matrices, complex QMF, FFT, FIR, and cross-frame state;
- float32 phase and final gain;
- float32 16-channel object output.
## 5. Speaker-rendering ABI
The same shared library exports object-to-speaker rendering:
```c
uint32_t ejoc_speaker_layout_channel_count(uint32_t speaker_bitfield);
ejoc_speaker_renderer_handle
ejoc_speaker_renderer_create(uint32_t speaker_bitfield);
int ejoc_speaker_renderer_process(
ejoc_speaker_renderer_handle handle,
const float* objects16_interleaved,
uint32_t sample_count,
uint32_t metadata_count,
const uint32_t* metadata_offsets,
const uint32_t* ramp_durations,
const uint16_t* positions_q15,
const uint8_t* region_indices,
const uint8_t* height_enabled,
const double* object_gains,
double* output_interleaved);
```
Input channel 0 is LFE and channels 1–15 are objects. Each metadata entry is an object-state snapshot. `sample_count` must be a multiple of 32; unfinished gain ramps remain in the handle and continue across calls.
The speaker path uses float32 object input, double coordinates/gains/accumulation, and interleaved double output. Quantization to float32 or PCM24 happens when the WAV is written.
Supported layouts:
```text
2.0 3.1 5.1 7.1 5.1.2 5.1.4 7.1.2 7.1.4 9.1.4 9.1.6
```
## 6. Binaural-rendering ABI
The shared library provides a 512-sample float64 binaural DSP interface:
```c
ejoc_binaural_renderer_handle ejoc_binaural_renderer_create(void);
int ejoc_binaural_renderer_configure_kernels(...);
int ejoc_binaural_renderer_configure_room(...);
int ejoc_binaural_renderer_process(
ejoc_binaural_renderer_handle handle,
const double* input16_interleaved, /* [512][16] */
const double* gains_complex, /* [16][2][77][2] */
const double* room_sends, /* [16] */
double output_gain,
double* output_stereo_interleaved); /* [512][2] */
```
Python parses the model, evaluates the OAMD timeline, and supplies complex gains and room sends every 512 samples. The C++ handle owns QMF, hybrid, recursive-room, and QMF-synthesis state. Inputs, state, accumulation, and output are double/complex double.
## 7. Building
The CMake definition is `native/CMakeLists.txt`. Run from the repository root:
```powershell
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
cmake --build build/cmake --config Release
cmake --install build/cmake --config Release
```
Platform runtime names:
```text
Windows lib/eac3joc_core.dll
Linux lib/libeac3joc_core.so
macOS lib/libeac3joc_core.dylib
```
The MSVC configuration uses the static CRT. Other runtime dependencies depend on the platform and toolchain and should be checked independently before publishing a prebuilt library.
The repository does not include native binaries by default. A prebuilt Release runtime or a locally built runtime can be placed directly under `lib/`.
## 8. Runtime lookup and fallback
Lookup order:
1. explicit `--native-library`;
2. `EAC3JOC_NATIVE_LIBRARY`;
3. the standard platform filename under `lib/`.
`--backend auto` falls back to NumPy when loading fails, and `--backend python` skips native discovery. The current CLI also prints the failure and falls back for `--backend native`; this existing behavior should not be read as successful native execution.
## 9. Implementation boundaries
- The native layer accepts only dense-JOC data already parsed by Python.
- The ABI fixes a 1536-sample JOC frame, at most 15 objects, at most 23 parameter bands, and at most 2 data points.
- The shared library and Python bridge must report the same ABI version.
- Only the ABI and data types are specified across platforms; bit-identical float64 results are not guaranteed.
- Private table headers under `native/src/` serve the native side only. The current repository does not include the scripts that generated those headers.
See the [mathematical notes](math.en.md) for the related formulas.
+191
View File
@@ -0,0 +1,191 @@
# JustOneCacophony 原生核说明
[English](native.en.md) · [返回 README](../README.md)
## 1. 职责边界
`native/` 只承载状态密集、调用频繁的 DSP 与扬声器渲染核。EMDF/JOC/OAMD 高层解析、错误报告、ADM 组装和 CLI 保留在 Python 中。
Python 通过标准库 `ctypes` 调用 C ABI;原生核不使用 pybind11、Cython、FFTW、MKL 或 OpenMP。它是可选加速路径,不扩大项目所支持的码流范围。
主要文件:
```text
native/include/eac3joc_core.h C ABI
native/src/eac3joc_core.cpp JOC/QMF 对象重建
native/src/speaker_renderer.cpp 对象到扬声器渲染
native/src/qmf_tables.h QMF 表
native/src/speaker_layouts.h 布局表
native/src/joc_huffman_tables.h JOC Huffman 表
src/native_renderer.py JOC ctypes 桥
src/speaker_native_renderer.py 扬声器 ctypes 桥
```
## 2. JOC 渲染 ABI
一个 opaque renderer 保存所有跨帧状态。主要调用为:
```c
int ejoc_renderer_process(
ejoc_renderer_handle handle,
const float* bed5_planar, /* [5][1536] */
const float* lfe, /* [1536] or NULL */
uint32_t object_mask,
const uint8_t* n_bands, /* [15] */
const uint8_t* n_dpoints, /* [15] */
const uint8_t* slope_idx, /* [15] */
const uint8_t* offset_ts, /* [15][2] */
const double* dq, /* [15][2][5][23] */
double clipgain,
float phase_new,
float output_scale,
float* output16_planar); /* [16][1536] */
```
JOC 的 Huffman 解码、差分还原和去量化先在 Python 中完成,dense 与 sparse 两条语法共用同一条入口。原生核心消费已去量化的 `dq`(double),两条语法在该 ABI 上布局一致。
线程接口为:
```c
int ejoc_renderer_set_threads(ejoc_renderer_handle handle, uint32_t total_threads);
uint32_t ejoc_renderer_thread_count(ejoc_renderer_handle handle);
```
`total_threads` 包含调用线程。单个 renderer 实例必须顺序提交帧;实例内部可以按对象和 analysis channel 并行。
## 3. 跨帧状态
每个 JOC renderer 独立保存:
- analysis FIFO:`double[5][9][64]`;
- L/R/C analysis delay:`float[3][10][64]`;
- Ls/Rs QMF delay:`complex<double>[2][10][64]`;
- Ls/Rs band-0 FIR history:`complex<double>[2][20]`;
- 矩阵插值 previous:`double[15][5][64]`;
- inverse-QMF state:`double[15][640]`;
- LFE delay:`double[1217]`。
这些状态属于 renderer 实例,不能在无 checkpoint 的情况下任意分段或乱序处理。
## 4. FFT、QMF 与精度
原生核包含固定 64 点 radix-2 complex FFT:
- analysis QMF 使用 forward FFT 后除以 64;
- inverse QMF 使用固定重排、旋转和 640 项有效窗状态;
- 不调用外部 FFT 库。
JOC 路径的数值类型为:
- 核心 PCM 输入:float32;
- 矩阵、复 QMF、FFT、FIR 和跨帧状态:double;
- phase 与最终 gain:float32;
- 16 声道对象输出:float32。
## 5. 扬声器渲染 ABI
同一个共享库还导出对象到扬声器布局的渲染接口:
```c
uint32_t ejoc_speaker_layout_channel_count(uint32_t speaker_bitfield);
ejoc_speaker_renderer_handle
ejoc_speaker_renderer_create(uint32_t speaker_bitfield);
int ejoc_speaker_renderer_process(
ejoc_speaker_renderer_handle handle,
const float* objects16_interleaved,
uint32_t sample_count,
uint32_t metadata_count,
const uint32_t* metadata_offsets,
const uint32_t* ramp_durations,
const uint16_t* positions_q15,
const uint8_t* region_indices,
const uint8_t* height_enabled,
const double* object_gains,
double* output_interleaved);
```
输入声道 0 为 LFE,1–15 为对象。每个 metadata entry 是一份对象状态快照。`sample_count` 必须是 32 的倍数;未完成的增益斜坡保存在 handle 中并跨调用继续。
扬声器路径使用 float32 对象输入、double 坐标/增益/累加与 interleaved double 输出;写 WAV 时才量化为 float32 或 PCM24。
支持的布局为:
```text
2.0 3.1 5.1 7.1 5.1.2 5.1.4 7.1.2 7.1.4 9.1.4 9.1.6
```
## 6. 双耳渲染 ABI
共享库提供 512-sample float64 双耳 DSP:
```c
ejoc_binaural_renderer_handle ejoc_binaural_renderer_create(void);
int ejoc_binaural_renderer_configure_kernels(...);
int ejoc_binaural_renderer_configure_room(...);
int ejoc_binaural_renderer_process(
ejoc_binaural_renderer_handle handle,
const double* input16_interleaved, /* [512][16] */
const double* gains_complex, /* [16][2][77][2] */
const double* room_sends, /* [16] */
double output_gain,
double* output_stereo_interleaved); /* [512][2] */
```
Python 负责模型解析、OAMD 时间轴和每 512 samples 的 complex gains/room sends。C++ handle 保存 QMF、hybrid、递归 room 和 QMF synthesis 状态。全部输入、状态、乘加和输出均为 double/complex double。
## 6.1 公开 SOFA 双耳渲染 ABI
共享库同时提供完整的原生 SOFA 双耳渲染器(`ejoc_sofa_binaural_*`),它镜像
Python `SofaBinauralBackend` 的全部数学:64-QMF/77-hybrid analysis/synthesis、
五阶 ACN/N3D 实球谐方向场求值、whole-QMF-slot 逐对象 delay 历史、六面一阶
image-source early reflections、共享 unitary FDN late room、LFE 120–180 Hz
低通与 961-sample latency 语义。kernel 表、编译好的 HRTF 场与房间常数通过
`configure_kernels/configure_field/configure_room` 一次上传;每 512-sample
block 先 `set_source` 更新 16 个 source,再 `process` 输入 PCM;`process` 返回
裁剪后的 stereo 样本数(首个 961 samples 被丢弃)。`finish` 以 64-sample 对齐的
块排空尾音。Python 桥位于 `src/sofa_native_backend.py`,与 Python 参考实现逐值
一致(差异 < 1e-9);原生库缺失时 `main.py` 自动回退 Python。
## 7. 构建
CMake 定义位于 `native/CMakeLists.txt`。从仓库根目录运行:
```powershell
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
cmake --build build/cmake --config Release
cmake --install build/cmake --config Release
```
平台运行库文件名:
```text
Windows lib/eac3joc_core.dll
Linux lib/libeac3joc_core.so
macOS lib/libeac3joc_core.dylib
```
MSVC 配置使用静态 CRT。其他运行时依赖由平台和工具链决定,发布预构建库前应对产物独立检查。
仓库默认不附带原生二进制。预构建的 Release 运行库或自行构建的运行库均可直接放入 `lib/`。
## 8. 运行时查找与回退
查找顺序为:
1. 显式 `--native-library`;
2. `EAC3JOC_NATIVE_LIBRARY`;
3. `lib/` 下当前平台的标准文件名。
`--backend auto` 在加载失败时回退到 NumPy;`--backend python` 跳过原生探测。`--backend native` 当前也会打印失败原因后回退,这是现有 CLI 行为,不应理解为原生库已成功使用。
## 9. 实现边界
- 原生层只接收 Python 已解析的 dense JOC 数据。
- ABI 固定了 1536-sample JOC 帧、最多 15 个对象、最多 23 个参数带和最多 2 个数据点。
- 共享库与 Python 桥需要 ABI version 一致。
- 跨平台只约定 ABI 与数据类型,不保证 float64 结果逐位一致。
- `native/src/` 中的私有表头只服务于原生侧;当前仓库不包含重新生成这些头文件的脚本。
相关公式见[数学说明](math.md)。
-128
View File
@@ -1,128 +0,0 @@
# SIMD and runtime dispatch
[中文](simd.md) · [Back to README](../README.en.md)
The heaviest loops in the binaural path (QMF analysis and synthesis, the hybrid
analysis low join, hybrid-domain path rendering, the ROOM FFT, the spherical
harmonic alignment) each have a runtime-dispatched vector implementation: one
binary carries several instruction-set variants, asks the CPU once at startup and
runs the widest one. **The output is byte-identical either way** — that is a hard
constraint, not a goal.
```text
JOC_SIMD=auto|scalar|sse2|avx2|avx512|neon pin one tier (used for verification)
JOC_SIMD_LOG=1 report the ISA each kernel actually got
```
## Why the split has to happen per translation unit
MSVC has no function-level attribute like `__attribute__((target("avx2")))`: one
`.cpp` file gets one `/arch`. So every ISA is its own translation unit with its own
`/arch:AVX2` or `/arch:AVX512` (GCC/Clang: `-mavx2` / `-mavx512f`), and
`dispatch.cpp` fills the function table at run time. The baseline units — the
dispatcher itself, the CPU probe and the scalar reference — carry **no** `/arch` at
all and stay on the SSE2 that x86-64 guarantees.
A trap from the history of this tree: `JOC_ENABLE_AVX2` used to be global, so
turning it on put AVX2 instructions into the very code paths that exist for older
CPUs. It now only selects whether the AVX2 unit is compiled in.
## Directory layout
One flat directory, **the instruction set in the file name and never in a
subdirectory** — that is how FLAC does it (`lpc.c` sits next to
`lpc_intrin_sse2.c`, `lpc_intrin_avx2.c` and `lpc_intrin_neon.c`, with the CPU
probe in its own `cpu.c`).
```text
src/simd/
simd.h the contract: Isa / Kernel / Kernels / dimensions
cpu_probe.{h,cpp} "can this machine run ISA X": CPUID+XGETBV / __builtin_cpu_supports / getauxval
dispatch.cpp policy: JOC_SIMD, the fallback ladder, the table, the log
kernels_scalar.cpp the reference (Isa::scalar; always built, always selectable)
kernels_intrin_avx2.cpp /arch:AVX2 -mavx2
kernels_intrin_avx512.cpp /arch:AVX512 -mavx512f
kernels_intrin_neon.cpp AArch64 default
```
The module lives at `src/simd/`, not `src/dsp/simd/`: `foundation/`, `hrtf/` and
`binaural/` all call into these kernels, so it is a cross-cutting layer rather than
a submodule of the DSP code (and `src/dsp/` held nothing else).
The header comment of `simd.h` carries the same module map; keep the two in sync
when the layout changes.
## Three rules
1. **Only `kernels_intrin_*.cpp` gets a wider flag.** `CMakeLists.txt` names those
files explicitly with `set_source_files_properties`; every other target stays on
the architecture's guaranteed ISA. A unit that goes wide without matching that
name fails `devtools/vec/isa_audit.ps1`.
2. **No dynamic initialisation inside an ISA unit.** Those objects are linked into
the same image as the baseline, so a global constructor would execute a wide
instruction before the dispatcher has looked at the CPU. Constant tables are
fine — they land in `.rdata`.
3. **Bit-exactness comes from the lane assignment, not from the ISA.** A lane may
only carry mutually independent outputs; the rounding sequence of a single
output, the separation of multiply and add (never an FMA) and the summation
order all stay exactly as `kernels_scalar.cpp` wrote them. Layout changes that
only reorder stored doubles (rank-minor basis tables, term-ordered tap tables,
stage-contiguous twiddle tables, band-major ROOM planes) are allowed.
## How the choice is made
`dispatch.cpp` parses `JOC_SIMD` first (forcing a tier this build or this machine
does not have prints a diagnostic and falls back, rather than pretending and
crashing), then walks `avx512 → avx2 → sse2 → neon` and picks, per kernel, the
widest implementation that is both compiled into this binary **and** runnable
here, falling back to the scalar reference. An AVX-512 unit therefore costs
nothing on a CPU without AVX-512; it is simply never selected.
On x86 the probe requires CPUID *and* XGETBV to agree: CPUID says the silicon can
do it, XCR0 says the OS saves the registers it needs. Either one alone is not
enough — using AVX without OS state support corrupts other threads across a
context switch. AArch64 needs no probe; ASIMD is the architectural baseline.
`sse2` is a selectable tier with **no unit of its own**, on purpose: a 128-bit SSE2
register is the register a scalar double already occupies, SSE2 cannot widen
double-precision arithmetic, and hand-written SSE2 would only add moves. The tier
resolves to the baseline unit.
## Effect
30-second reference cases, one binary with only `JOC_SIMD` switched (DSP stage,
`t_render_dsp`):
| Case | `scalar` | `auto` | DSP speed-up | End-to-end wall clock |
|---|---|---|---|---|
| Binaural Rosella | 1.256 s | **0.640 s** | **1.96×** | 1.690 → **0.941 s** |
| Binaural SOFA | 1.560 s | **0.654 s** | **2.39×** | 1.859 → **0.849 s** |
| Speaker 5.1 / 9.1.6 / ADM | — | — | 1.00× | no regression (these kernels are not on those paths) |
The vectorised loops themselves gain more: synthesis basis 5.89×, 13-tap low join
4.50×, SOFA QMF synthesis 4.27×, 128-point FFT 2.24×. The whole pipeline stops
short of 8× because a good part of the time goes to parameter setup, straight
copies and file writing — none of which has independent work items — and because
SOFA's 33 M sin/cos calls per sample cannot be vectorised under a byte-exactness
contract.
## Verifying a change
```powershell
$env:JOC_SIMD_LOG='1' # per-kernel ISA on this machine
$env:JOC_SIMD='scalar' # force the reference: hashes must not move
pwsh -NoProfile -File devtools\vec\isa_audit.ps1 # disassemble every .obj: 0 unguarded wide instructions
pwsh -NoProfile -File devtools\vec\sha_matrix.ps1 # 5 tiers x 2 renders against the reference digests
cmd /c devtools\vec\build_kernel_probe.bat # per-kernel byte digests (8 kernels)
```
## Adding an ISA
1. Write `kernels_intrin_<isa>.cpp`, implementing the slots you have and leaving the
rest `nullptr` — the dispatcher falls back per kernel (that is how
`qmf_synthesis_basis` is handled in the AVX-512 unit).
2. Add it to `JOC_SIMD_SOURCES` in `CMakeLists.txt` with `JOC_SIMD_HAVE_<ISA>=1` and
its flag, and extend `Isa`, `isa_rank`, `isa_compiled`, `isa_supported` and the
`JOC_SIMD` name table in `simd.h` / `dispatch.cpp`.
3. Verify: the kernel digests must match the scalar unit byte for byte, and the
reference renders must keep their SHA-256 on every tier.
-107
View File
@@ -1,107 +0,0 @@
# SIMD 与运行时派发
[English](simd.en.md) · [返回 README](../README.md)
双耳通路里最重的那几段循环(QMF 分析/合成、混合分析的低频拼接、混合域路径渲染、
ROOM 的 FFT、球谐对齐)都有一份运行时分派的向量实现:同一份二进制里装多套 ISA 代码,
启动时问一次 CPU,然后选最宽的那套跑。**输出逐字节不变**——这是硬约束,不是目标。
```text
JOC_SIMD=auto|scalar|sse2|avx2|avx512|neon 强制某一层(验收用)
JOC_SIMD_LOG=1 打印每个 kernel 实际生效的 ISA
```
## 为什么必须"按编译单元分 ISA"
MSVC 没有 `__attribute__((target("avx2")))` 这类函数级多版本能力,一个 .cpp 只能有
一个 `/arch`。所以每个 ISA 一个编译单元,各自带自己的 `/arch:AVX2` / `/arch:AVX512`
(GCC/Clang 是 `-mavx2` / `-mavx512f`),由 `dispatch.cpp` 在运行时填函数表。
基线单元(含派发器本身、CPU 探测、标量参考实现)**不带任何 `/arch`**,它们只使用
x86-64 架构保证的 SSE2。
历史坑:早先的 `JOC_ENABLE_AVX2` 是**全局**的,一旦打开,连"给老 CPU 用"的基线路径
都带 AVX2 指令。现在这个选项只决定是否把 AVX2 单元编进二进制。
## 目录布局
一个扁平目录,**ISA 写在文件名里,不写进子目录**——这是 FLAC 的做法
(`src/libFLAC/lpc.c` 旁边就是 `lpc_intrin_sse2.c` / `lpc_intrin_avx2.c` /
`lpc_intrin_neon.c`,CPU 探测单独放在 `cpu.c`)。
```text
src/simd/
simd.h 唯一契约头:Isa / Kernel / Kernels / 维度常量
cpu_probe.{h,cpp} "这台机器能不能跑 ISA X":CPUID+XGETBV / __builtin_cpu_supports / getauxval
dispatch.cpp 策略:JOC_SIMD 解析、回退阶梯、填函数表、日志
kernels_scalar.cpp 参考实现(Isa::scalar,永远编译、永远可选中)
kernels_intrin_avx2.cpp /arch:AVX2 -mavx2
kernels_intrin_avx512.cpp /arch:AVX512 -mavx512f
kernels_intrin_neon.cpp AArch64 默认 -march=armv8-a+simd
```
`simd.h` 的头注释里有一份同样的模块地图,改布局时两处一起改。
模块放在 `src/simd/` 而不是 `src/dsp/simd/`:这些 kernel 被 `foundation/`、`hrtf/`、
`binaural/` 三个模块共用,是横切的一层,不是 DSP 的子模块(更何况 `src/dsp/` 里除了
`simd/` 空无一物)。
## 三条规则
1. **只有 `kernels_intrin_*.cpp` 拿更宽的编译开关。** `CMakeLists.txt` 用
`set_source_files_properties` 逐个点名,其余目标一律留在架构保证的 ISA 上。
任何不属于这个命名却带了宽指令的单元都会被 `devtools/vec/isa_audit.ps1` 判失败。
2. **ISA 单元里不许有动态初始化。** 它们和基线代码链进同一个镜像,全局构造函数会在
派发器看 CPU 之前就跑宽指令。常量表没问题(落在 `.rdata`)。
3. **逐位一致靠的是 lane 的划分,不是 ISA。** lane 里只能放**互相独立**的输出;单个输出
的舍入序列、乘加分离(绝不用 FMA)、求和顺序都保持 `kernels_scalar.cpp` 原样。
只改变 double **存放顺序**的布局改造(基函数表转秩小序、抽头表按项序、蝶形因子表
按级连续化、ROOM 谱平面改频带主序)是允许的。
## 运行时怎么选
`dispatch.cpp` 先解析 `JOC_SIMD`(强制一个本机不支持的层会打印诊断并回退,而不是假装
选中然后崩),再走阶梯 `avx512 → avx2 → sse2 → neon`,每个 kernel 单独挑"已编进本
二进制 **且** 本机可跑"的最宽实现,挑不到就落到标量参考实现。所以 AVX-512 单元在
不支持它的 CPU 上只是不被选中,不影响启动。
x86 的探测要 CPUID 与 XGETBV **同时**成立:CPUID 说明硅片有这个能力,XCR0 说明操作
系统会保存对应寄存器状态,缺一个就不能用(否则上下文切换会踩坏别的线程)。AArch64
不需要探测,ASIMD 是架构基线。
`sse2` 是一个有意保留的档位但**没有单独的单元**:128 位 SSE2 寄存器就是标量 double
已经在用的寄存器,SSE2 加宽不了双精度运算,手写只会多出搬运指令,所以它选中的是基线
单元。
## 效果
30 s 参考用例,同一二进制只切 `JOC_SIMD`(DSP 阶段 `t_render_dsp`):
| 用例 | `scalar` | `auto` | DSP 加速 | 端到端墙钟 |
|---|---|---|---|---|
| 双耳 Rosella | 1.256 s | **0.640 s** | **1.96×** | 1.690 → **0.941 s** |
| 双耳 SOFA | 1.560 s | **0.654 s** | **2.39×** | 1.859 → **0.849 s** |
| 扬声器 5.1 / 9.1.6 / ADM | — | — | 1.00× | 0 回归(不走这些 kernel) |
单看被向量化的循环,红利更大:合成基函数 5.89×、13 抽头低频拼接 4.50×、
SOFA QMF 合成 4.27×、128 点 FFT 2.24×。整条流水线到不了 8×,是因为相当一部分时间在
参数设置、直通拷贝、写盘这些没有独立工作项的代码上,以及 SOFA 每次采样的 33 M 次
sin/cos 按逐位契约不能向量化。
## 验证
```powershell
$env:JOC_SIMD_LOG='1' # 本机每个 kernel 实际选中的 ISA
$env:JOC_SIMD='scalar' # 强制参考路径:哈希必须一动不动
pwsh -NoProfile -File devtools\vec\isa_audit.ps1 # 反汇编全部 .obj:0 个无守卫的宽指令
pwsh -NoProfile -File devtools\vec\sha_matrix.ps1 # 5 档 × 2 渲染,逐字节比对参考摘要
cmd /c devtools\vec\build_kernel_probe.bat # kernel 级逐字节摘要(8 个 kernel)
```
## 加一个新的 ISA
1. 写 `kernels_intrin_<isa>.cpp`,实现能实现的槽位,其余留 `nullptr`——派发器会逐
kernel 回退(AVX-512 单元里的 `qmf_synthesis_basis` 就是这么处理的)。
2. 在 `CMakeLists.txt` 里加进 `JOC_SIMD_SOURCES`、`JOC_SIMD_HAVE_<ISA>=1` 和它的编译
开关,并在 `simd.h` / `dispatch.cpp` 里补上 `Isa`、`isa_rank`、`isa_compiled`、
`isa_supported` 与 `JOC_SIMD` 名字表。
3. 验证:kernel 摘要必须与标量逐字节相同,参考渲染在每个档位上的 SHA-256 都不能变。
-490
View File
@@ -1,490 +0,0 @@
/*
* joc_core.h -- JustOneCacophony C++ Core, public ABI. Pure C.
*
* This header is the single authoritative definition of the Core's public
* parameter/error surface. Frontends (thin Python CLI, joc_dump, foobar2000,
* MPV, FFmpeg) only ever fill these POD structs and read these POD results;
* no audio data, no internal DSP concept, and no Python type crosses this
* boundary.
*
* Stability tiers
* ---------------
* [T1] host tier
* joc_abi_version / joc_version_string / joc_build_info, joc_error,
* joc_error_name / joc_error_stage, joc_last_error_detail.
* Stable, versioned, safe for media hosts.
*
* [T2] bitstream / verification tier
* joc_parse_eac3_frame, joc_parse_id14, joc_frame_params, joc_emdf_info.
* These deliberately expose the dequantized JOC matrix coefficients so the
* bitstream front-end can be validated bit-exactly and driven by tooling
* (joc_dump) and A/B harnesses. Media hosts must NOT use this tier; they
* use the task/stream API (added in later milestones).
*
* Conventions
* -----------
* * every struct's first two fields are struct_size / struct_version;
* * all arrays are fixed size and POD, no pointers, no allocation;
* * every function returns joc_error (JOC_OK == 0);
* * joc_last_error_detail() returns a thread-local message that stays valid
* until the next Core call on the same thread.
*/
#pragma once
#include <stdint.h>
#include <stddef.h>
#define JOC_ABI_VERSION 3u
#define JOC_FRAME_PARAMS_VERSION 1u
#define JOC_EMDF_INFO_VERSION 1u
#define JOC_TASK_CONFIG_VERSION 3u
#define JOC_TASK_RESULT_VERSION 2u
#define JOC_EVENT_VERSION 1u
/* JOC_STATIC: this branch exists for building the sources directly into an application, where nothing is imported or exported. */
#if defined(JOC_STATIC)
#define JOC_API
#define JOC_CALL __cdecl
#elif defined(_WIN32)
#if defined(JOC_BUILD_DLL)
#define JOC_API __declspec(dllexport)
#else
#define JOC_API __declspec(dllimport)
#endif
#define JOC_CALL __cdecl
#else
#define JOC_API __attribute__((visibility("default")))
#define JOC_CALL
#endif
#ifdef __cplusplus
extern "C" {
#endif
/* ------------------------------------------------------------------ */
/* Fixed layout constants (single source of truth for every frontend) */
/* ------------------------------------------------------------------ */
enum {
JOC_FRAME_SAMPLES = 1536, /* E-AC-3 frame = 24 * 64 */
JOC_TIMESLOTS = 24,
JOC_SUBBANDS = 64,
JOC_CORE_CHANNELS = 5, /* L R C Ls Rs (JOC order) */
JOC_MAX_CORE_CHANNELS = 7, /* dmx_config_idx 1/2/4 declare 7 */
JOC_OUTPUT_CHANNELS = 16, /* ch0 = LFE, ch1..15 = objects */
JOC_MAX_OBJECTS = 15,
JOC_MAX_DPOINTS = 2,
JOC_MAX_PARAMETER_BANDS = 23,
JOC_MAX_EMDF_PAYLOADS = 16,
JOC_SPEAKER_BLOCK_SAMPLES = 32,
JOC_BINAURAL_BLOCK_SAMPLES = 512,
JOC_BINAURAL_QMF_BANDS = 64,
JOC_BINAURAL_HYBRID_BANDS = 77,
JOC_QMF_HOP_SAMPLES = 64,
JOC_BINAURAL_LATENCY_SAMPLES = 961,
JOC_LFE_DELAY_SAMPLES = 1217
};
/* ------------------------------------------------------------------ */
/* [T1] library / error surface */
/* ------------------------------------------------------------------ */
typedef enum joc_error {
JOC_OK = 0,
JOC_ERR_INVALID_ARGUMENT,
JOC_ERR_INVALID_CONFIG,
JOC_ERR_OUT_OF_MEMORY,
JOC_ERR_IO,
JOC_ERR_UNSUPPORTED_PLATFORM,
JOC_ERR_LIBRARY_MISSING,
/* input / bitstream */
JOC_ERR_INPUT_NOT_FOUND,
JOC_ERR_INPUT_FORMAT,
JOC_ERR_EAC3_SYNCFRAME,
JOC_ERR_EMDF_TRANSPORT,
JOC_ERR_EMDF_SYNTAX,
JOC_ERR_JOC_SYNTAX,
JOC_ERR_JOC_UNSUPPORTED_VARIANT,
JOC_ERR_OAMD_SYNTAX,
JOC_ERR_OAMD_UNSUPPORTED_VARIANT,
JOC_ERR_BITSTREAM_TRUNCATED,
JOC_ERR_BITSTREAM_PADDING,
/* resources */
JOC_ERR_HRTF_NOT_FOUND,
JOC_ERR_HRTF_FORMAT,
JOC_ERR_HRTF_VERSION,
JOC_ERR_HRTF_HASH,
JOC_ERR_HRTF_UNSUPPORTED_CONVENTION,
/* rendering / output */
JOC_ERR_LAYOUT_UNSUPPORTED,
JOC_ERR_RENDER_FAILED,
JOC_ERR_OUTPUT_OPEN,
JOC_ERR_OUTPUT_WRITE,
JOC_ERR_OUTPUT_CLIP_ABORT,
JOC_ERR_ADM_VALIDATION,
/* task / stream */
JOC_ERR_CANCELLED,
JOC_ERR_STATE,
JOC_ERR_NOT_SUPPORTED,
JOC_ERR_INTERNAL
} joc_error;
JOC_API uint32_t JOC_CALL joc_abi_version(void);
JOC_API const char* JOC_CALL joc_version_string(void);
JOC_API const char* JOC_CALL joc_build_info(void);
/* Struct sizes, so a binding can assert its layout matches the library instead of
* assuming (mismatches are otherwise silent memory corruption). */
JOC_API uint32_t JOC_CALL joc_event_size(void);
JOC_API uint32_t JOC_CALL joc_task_config_size(void);
JOC_API uint32_t JOC_CALL joc_task_result_size(void);
JOC_API const char* JOC_CALL joc_error_name(joc_error code);
JOC_API const char* JOC_CALL joc_error_stage(joc_error code);
/* Thread-local structured detail for the most recent failing call. */
JOC_API const char* JOC_CALL joc_last_error_detail(void);
/* ------------------------------------------------------------------ */
/* [T2] bitstream / verification tier */
/* ------------------------------------------------------------------ */
/* One EMDF payload directory entry. bit_offset is the MSB-first bit
* position of the payload's first byte inside the syncframe, exactly as the
* EMDF container syntax defines it (payloads are not byte aligned in general). */
typedef struct joc_emdf_payload_info {
uint8_t id;
uint8_t reserved[3];
uint16_t sample_offset; /* EMDF outer smpoffst, 11 bits */
uint16_t reserved2;
uint32_t bit_offset;
uint32_t size; /* payload bytes */
} joc_emdf_payload_info;
typedef struct joc_emdf_info {
uint32_t struct_size;
uint32_t struct_version;
uint32_t start_bit; /* container syncword bit position */
uint32_t container_bytes; /* 4 + declared length */
uint32_t payload_count;
uint32_t reserved;
joc_emdf_payload_info payloads[JOC_MAX_EMDF_PAYLOADS];
} joc_emdf_info;
/* Per-object ID14/JOC descriptor plus its dequantized matrix.
* dq[dp][ch][pb] is zero filled outside [0,n_dpoints) x [0,n_channels) x
* [0,n_bands). Absent objects are entirely zero. */
typedef struct joc_object_params {
uint8_t present;
uint8_t num_bands_idx;
uint8_t n_bands;
uint8_t sparse; /* 0 = dense (MTX), 1 = sparse (IDX+VEC) */
uint8_t quant_idx; /* 0 = 96 levels, 1 = 192 levels */
uint8_t slope_idx; /* 0 = interpolate, 1 = step at offset_ts */
uint8_t num_dpoints_bits;
uint8_t n_dpoints;
uint8_t offset_ts[JOC_MAX_DPOINTS];
uint8_t reserved[2];
double dq[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS];
} joc_object_params;
typedef struct joc_frame_params {
uint32_t struct_size;
uint32_t struct_version;
uint8_t dmx_config_idx;
uint8_t num_objects_bits;
uint8_t ext_config_idx;
uint8_t n_objects;
uint8_t n_channels; /* 5 or 7 */
uint8_t clipgain_x_bits;
uint8_t clipgain_y_bits;
uint8_t reserved;
uint32_t seq_count; /* 10-bit JOC sequence counter, parsed only */
uint32_t present_mask; /* bit i == object i present */
uint32_t data_end_bits; /* bit position just after joc_data */
uint32_t trailing_bits; /* bits left after joc_data (padding/ext) */
uint8_t tail_bytes[8]; /* first up to 8 trailing bytes, for A/B */
double clipgain; /* 1 + (y/32) * 2^(x-4), bit-exact */
joc_object_params objects[JOC_MAX_OBJECTS];
} joc_frame_params;
/* Parse an EMDF ID14 (JOC) payload. */
JOC_API joc_error JOC_CALL joc_parse_id14(const uint8_t* payload, size_t payload_size,
joc_frame_params* out_params);
/* Locate the contiguous JOC EMDF container inside one E-AC-3 syncframe and
* parse its ID14 payload. out_emdf may be NULL. */
JOC_API joc_error JOC_CALL joc_parse_eac3_frame(const uint8_t* frame, size_t frame_size,
joc_frame_params* out_params,
joc_emdf_info* out_emdf);
/* Copy one payload's bytes out of a syncframe (MSB-first bit extraction, which
* equals a memcpy for byte-aligned containers). out_size receives the payload
* byte count; pass out == NULL to query only the size. */
JOC_API joc_error JOC_CALL joc_extract_payload(const uint8_t* frame, size_t frame_size,
const joc_emdf_payload_info* payload,
uint8_t* out, size_t out_capacity,
size_t* out_size);
/* E-AC-3 syncframe traversal: given the offset of a frame start, report its
* byte length so a host can walk a bare E-AC-3 stream without duplicating
* frmsiz logic. Rejects resynchronisation (no silent recovery). */
JOC_API joc_error JOC_CALL joc_eac3_frame_bytes(const uint8_t* data, size_t size,
size_t offset, size_t* out_frame_bytes);
/* Strict trailing-bit check for the JOC payload (A/B robustness corpus).
* Returns JOC_ERR_BITSTREAM_PADDING when more than 7 bits are left over or the
* leftover bits are not zero. */
JOC_API joc_error JOC_CALL joc_check_id14_padding(const uint8_t* payload, size_t payload_size,
uint32_t* out_trailing_bits);
/* ================================================================== */
/* [T1] host tier: task, telemetry and control */
/* ================================================================== */
/* ---- 1. events ---------------------------------------------------- */
/*
* Events carry state only: frame/sample counters, stage, progress, statistics,
* warnings, errors and paths. Audio never travels through an event; a fixed
* size POD with no pointers keeps that enforceable (see the static assertion in
* the implementation).
*/
typedef enum joc_event_type {
JOC_EV_TASK_STARTED = 0x0001,
JOC_EV_TASK_STATE_CHANGED = 0x0002,
JOC_EV_TASK_COMPLETED = 0x0003,
JOC_EV_TASK_FAILED = 0x0004,
JOC_EV_TASK_CANCELLED = 0x0005,
JOC_EV_INPUT_OPENED = 0x0101,
JOC_EV_METADATA_INDEXED = 0x0103,
JOC_EV_HRTF_LOADED = 0x0110,
JOC_EV_RENDERER_INITIALIZED = 0x0120,
JOC_EV_PROGRESS = 0x0201,
JOC_EV_STAGE_CHANGED = 0x0202,
JOC_EV_JOC_FRAME_STATS = 0x0301,
JOC_EV_OAMD_STATS = 0x0302,
JOC_EV_OUTPUT_STATS = 0x0303,
JOC_EV_OUTPUT_OPENED = 0x0401,
JOC_EV_OUTPUT_FORMAT_DECIDED = 0x0402,
JOC_EV_OUTPUT_FINALIZED = 0x0403,
JOC_EV_LOG = 0x0501,
JOC_EV_WARNING = 0x0502,
JOC_EV_ERROR = 0x0503
} joc_event_type;
typedef enum joc_stage {
JOC_STAGE_IDLE = 0,
JOC_STAGE_INPUT = 1,
JOC_STAGE_METADATA = 2,
JOC_STAGE_DECODE = 3,
JOC_STAGE_JOC = 4,
JOC_STAGE_RENDER = 5,
JOC_STAGE_OUTPUT = 6,
JOC_STAGE_DONE = 7
} joc_stage;
enum { JOC_LOG_TRACE = 0, JOC_LOG_DEBUG = 1, JOC_LOG_INFO = 2, JOC_LOG_NOTICE = 3,
JOC_LOG_WARNING = 4, JOC_LOG_ERROR = 5, JOC_LOG_FATAL = 6 };
typedef struct joc_event {
uint32_t struct_size;
uint32_t type;
uint64_t sequence;
uint64_t timestamp_us;
uint64_t current_frame;
uint64_t total_frames;
uint64_t current_sample;
uint64_t total_samples;
uint32_t stage;
uint32_t backend;
double progress; /* 0..1, -1 = unknown */
double elapsed_seconds;
double realtime_factor;
uint64_t output_samples;
uint64_t output_bytes;
double output_duration_seconds;
joc_error error_code;
uint32_t log_level;
char stage_name[32];
char message[256];
} joc_event;
typedef void (JOC_CALL *joc_event_fn)(void* user, const joc_event* event);
typedef struct joc_event_sink {
uint32_t struct_size;
joc_event_fn callback;
void* user;
uint32_t min_type; /* 0 = no filter */
uint32_t max_type; /* 0 = no filter */
} joc_event_sink;
/* ---- 2. cancellation ---------------------------------------------- */
typedef struct joc_cancel_token joc_cancel_token;
JOC_API joc_cancel_token* JOC_CALL joc_cancel_token_create(void);
JOC_API void JOC_CALL joc_cancel_token_request(joc_cancel_token* token);
JOC_API int32_t JOC_CALL joc_cancel_token_is_requested(const joc_cancel_token* token);
JOC_API void JOC_CALL joc_cancel_token_destroy(joc_cancel_token* token);
/* ---- 3. task configuration and results ---------------------------- */
typedef enum joc_operation {
JOC_OP_ADM_BWF = 0,
JOC_OP_SPEAKER = 1,
JOC_OP_BINAURAL = 2
} joc_operation;
typedef enum joc_output_format {
JOC_FORMAT_FLOAT32 = 0,
JOC_FORMAT_PCM24 = 1
} joc_output_format;
typedef enum joc_binaural_mode {
JOC_BINAURAL_OFF = 0,
JOC_BINAURAL_NEAR = 1,
JOC_BINAURAL_FAR = 2,
JOC_BINAURAL_MID = 3
} joc_binaural_mode;
typedef enum joc_trajectory_mode {
JOC_TRAJECTORY_COMPACT = 0,
JOC_TRAJECTORY_DENSE64 = 1
} joc_trajectory_mode;
/* What to do when an int24 WAV would clip (peak outside [-1, 1]). Mirrors the
* reference CLI's --clip-action; ADM BWF output is always int24 and does not
* consult this because no alternative format exists there. */
typedef enum joc_clip_action {
JOC_CLIP_ASK = 0, /* prompt on stdin; an error when stdin is not a terminal */
JOC_CLIP_CONTINUE = 1, /* write int24, truncating out-of-range values */
JOC_CLIP_FLOAT32 = 2, /* switch the output to float32 */
JOC_CLIP_ABORT = 3 /* fail the task */
} joc_clip_action;
/* Compiled-HRTF cache policy for a SOFA input; the .jochrtf itself is an
* internal artifact of the compile step. */
typedef enum joc_hrtf_cache_policy {
JOC_HRTF_CACHE_NONE = 0, /* compile and discard */
JOC_HRTF_CACHE_MEMORY = 1, /* compile and keep in this process (default) */
JOC_HRTF_CACHE_DISK = 2 /* compile, reuse and write <cache_dir>/<name>.jochrtf */
} joc_hrtf_cache_policy;
enum { JOC_TASK_F_SKIP_SHA256 = 1u, JOC_TASK_F_KEEP_INTERMEDIATE = 2u,
JOC_TASK_F_QUIET = 4u, JOC_TASK_F_METADATA_ONLY = 8u };
typedef struct joc_task_config {
uint32_t struct_size;
uint32_t struct_version;
/* input */
const char* input_path; /* .eac3/.ec3/.m4a/... */
const char* ffmpeg_path; /* NULL = "ffmpeg" from PATH */
const char* bed_path; /* NULL = decode the core PCM with ffmpeg */
const char* work_dir; /* NULL = a temporary directory */
double eac3_drc_scale; /* 0 = DRC off (reference default) */
int32_t eac3_target_level; /* -31..0, 0 = not applied */
/* output */
uint32_t operation;
const char* output_path;
uint32_t output_format; /* requested format for speaker/binaural */
uint32_t flags;
uint32_t clip_action; /* joc_clip_action; JOC_CLIP_ASK by default */
/* rendering */
const char* speaker_layout_name;
uint32_t speaker_metadata_offset; /* default 1473 */
uint32_t binaural_mode;
const char* hrtf_path; /* .jochrtf */
const char* kernels_path; /* rosella_kernels.npz */
double binaural_tail_seconds; /* default 5.0 */
double binaural_tail_threshold; /* binaural only; 0 disables trimming */
uint32_t binaural_chunk_frames; /* accepted for CLI parity; no effect */
/* ADM */
uint32_t adm_binaural_mode; /* DBMD segment 10 encoding */
uint32_t trajectory_mode;
/* generic */
uint32_t object_delay_samples; /* default 1473 */
double gain_db; /* default 0 */
uint64_t duration_frames; /* 0 = the whole stream */
uint32_t progress_interval_frames; /* default 1000 */
uint32_t native_threads; /* 0 = automatic */
/* diagnostics */
uint32_t print_metadata; /* 0 none, 1 summary, 2 per frame */
const char* metadata_json_path; /* NULL = no JSON summary */
joc_cancel_token* cancel; /* optional */
/* Binaural HRTF input (config version 2): when hrtf_sofa_path is set the
* library compiles it with hrtf_cache_policy / hrtf_cache_dir / hrtf_radius_m
* and hrtf_path is unused. hrtf_path stays the advanced override that reads
* a .jochrtf directly. */
const char* hrtf_sofa_path; /* SOFA SimpleFreeFieldHRIR input */
const char* personalized_headphone_path; /* Rosella .personalized_headphone input */
const char* hrtf_cache_dir; /* disk policy directory */
uint32_t hrtf_cache_policy; /* joc_hrtf_cache_policy */
uint32_t reserved0;
double hrtf_radius_m; /* SOFA measurement-radius shell */
} joc_task_config;
typedef enum joc_task_status {
JOC_TASK_OK = 0,
JOC_TASK_FAILED = 1,
JOC_TASK_CANCELLED = 2
} joc_task_status;
typedef struct joc_task_result {
uint32_t struct_size;
uint32_t struct_version;
uint32_t status;
uint32_t error_code;
char error_message[512];
char error_stage[32];
uint64_t input_frames;
uint64_t output_samples;
double duration_sec;
uint32_t output_format_actual;
double output_peak;
uint64_t output_over_unity_values;
uint64_t output_file_bytes;
char output_sha256[65]; /* empty when skipped */
uint64_t oamd_payloads;
uint64_t oamd_transitions;
double t_decode_bed;
double t_render;
double t_write;
double t_total;
double t_render_dsp; /* speaker/binaural DSP calls only; 0 for ADM */
double t_write_file; /* disk writes including the finalize; >= t_write */
} joc_task_result;
typedef struct joc_validation_issue {
joc_error code;
uint32_t severity; /* 0 = info, 1 = warning, 2 = error */
char field[48];
char message[256];
} joc_validation_issue;
/* ---- 4. entry points ---------------------------------------------- */
/* Fills `issues` (up to `capacity`) and reports how many were produced.
* Returns JOC_OK when no *error*-severity issue was found. */
JOC_API joc_error JOC_CALL joc_task_validate(const joc_task_config* config,
joc_validation_issue* issues, uint32_t capacity,
uint32_t* count);
/* Runs the whole task on the calling thread. `sink` may be NULL; `out` may be
* NULL. On cancellation the partial output is removed and JOC_ERR_CANCELLED is
* returned with out->status = JOC_TASK_CANCELLED. */
JOC_API joc_error JOC_CALL joc_task_execute(const joc_task_config* config,
const joc_event_sink* sink, joc_task_result* out);
/* Serialises the stable subset of the result as JSON (no environment fields -
* those belong to the frontend). `needed` receives the required size including
* the terminator; a NULL buffer queries only the size. */
JOC_API joc_error JOC_CALL joc_task_result_to_json(const joc_task_result* result, char* buffer,
size_t capacity, size_t* needed);
/* The streaming (push/pull) tier lives in "joc_stream.h" so an embedder can
* include the narrow surface without the bitstream/verification tier. */
#ifdef __cplusplus
} /* extern "C" */
#endif
-136
View File
@@ -1,136 +0,0 @@
/*
* joc_stream.h -- streaming (push/pull) interface of the JustOneCacophony core.
*
* This is the narrow, embedder-facing surface: a player or decoder component
* includes only this header. It is the same shared library as joc_core.h, split
* so an integrator never has to see the bitstream/verification tier.
*
* A stream is a stateful instance for hosts that cannot wait for a whole file:
* the caller feeds E-AC-3 bytes and the core PCM of the same frames (or already
* rebuilt objects16) and pulls rendered PCM as soon as it is available. It
* mirrors the two library shapes a decoder library normally offers: this
* push/pull form for host-owned I/O, and joc_task_execute() in joc_core.h for
* library-owned file I/O.
*
* Contract:
* - all state is instance-private, so several streams coexist;
* - a stream is NOT thread safe: push and pull must come from one thread;
* - rendering is stateful (matrix interpolation, gain ramps, room tail), so a
* new position on the timeline requires decoding to continue from the start
* of the stream; there is no seek in this version;
* - the kernel latency is 961 samples for speaker/binaural output: the first
* pull after two 512-sample blocks, and flush() drains the binaural tail.
*/
#pragma once
#include "joc_core.h"
/* JOC_STATIC (building the sources directly into an application): joc_core.h above already installs the empty JOC_API. */
#if defined(JOC_STATIC) && !defined(JOC_API)
#define JOC_API
#define JOC_CALL __cdecl
#endif
#ifdef __cplusplus
extern "C" {
#endif
typedef struct joc_stream joc_stream;
typedef enum joc_stream_input {
JOC_STREAM_IN_EAC3 = 0, /* bare E-AC-3 syncframes (the metadata stream) */
JOC_STREAM_IN_PCM_OBJECTS16 = 1, /* 16-channel objects16, decoded by the host */
JOC_STREAM_IN_CORE_PCM = 3 /* the 5.1 core PCM of the pushed E-AC-3 frames */
} joc_stream_input;
typedef enum joc_stream_output {
JOC_STREAM_OUT_PCM_OBJECTS16 = 0, /* planar [16][samples] float32 */
JOC_STREAM_OUT_SPEAKER = 1, /* interleaved [samples][channels] f32 */
JOC_STREAM_OUT_BINAURAL = 2 /* interleaved [samples][2] f32 */
} joc_stream_output;
typedef struct joc_stream_config {
uint32_t struct_size;
uint32_t struct_version;
uint32_t input;
uint32_t output;
const char* speaker_layout_name;
uint32_t speaker_metadata_offset; /* default 1473 */
uint32_t binaural_mode; /* near|mid|far; 0 means mid */
const char* hrtf_path; /* binaural only */
const char* kernels_path; /* binaural only */
double binaural_tail_seconds; /* default 5.0 */
uint32_t object_delay_samples; /* default 1473 */
double gain_db; /* default 0 */
uint32_t native_threads;
uint32_t reserved;
/* Binaural HRTF input, the same three shapes joc_task_config accepts: when
* hrtf_sofa_path is set the library compiles it with hrtf_cache_policy /
* hrtf_cache_dir / hrtf_radius_m and hrtf_path is unused; when
* personalized_headphone_path is set the Rosella runtime renders instead.
* hrtf_path stays the fallback/advanced input that reads a .jochrtf directly. */
const char* hrtf_sofa_path; /* SOFA SimpleFreeFieldHRIR input */
const char* personalized_headphone_path; /* Rosella .personalized_headphone input */
const char* hrtf_cache_dir; /* disk cache directory for SOFA compilation */
uint32_t hrtf_cache_policy; /* joc_hrtf_cache_policy: 0 none, 1 memory, 2 disk */
double hrtf_radius_m; /* SOFA measurement-radius shell, default 1.0 */
} joc_stream_config;
typedef struct joc_stream_buffer {
uint32_t struct_size;
uint32_t struct_version;
uint32_t kind; /* which joc_stream_input/output this buffer carries */
uint32_t channels;
uint32_t sample_rate;
uint32_t sample_count; /* in: capacity / out: produced (per channel) */
uint32_t byte_count; /* in: capacity / out: consumed or produced bytes */
uint32_t reserved;
const uint8_t* bytes; /* EAC3 input */
const float* pcm; /* PCM input */
uint8_t* out_bytes; /* reserved for encoded outputs */
float* out_pcm; /* PCM output */
} joc_stream_buffer;
typedef struct joc_stream_status_info {
uint32_t struct_size;
uint32_t struct_version;
uint64_t frames_in;
uint64_t frames_out;
uint64_t samples_in;
uint64_t samples_out;
uint64_t bytes_in;
uint64_t buffered_samples; /* rendered but not yet pulled, per channel */
uint64_t oamd_payloads;
uint64_t oamd_transitions;
uint32_t output_channels;
uint32_t ended; /* 1 after flush() */
} joc_stream_status_info;
JOC_API joc_error JOC_CALL joc_stream_create(const joc_stream_config* config, joc_stream** out);
/* Feeds one buffer. Kind selects the path: EAC3 bytes, the core PCM of those
* frames, or objects16. Consumed counts are reported so a caller can resume
* from a partial push. */
JOC_API joc_error JOC_CALL joc_stream_push(joc_stream* stream, const joc_stream_buffer* input,
uint32_t* consumed_samples, uint32_t* consumed_bytes);
/* Copies as many rendered samples as fit into `output` (interleaved, or planar
* for PCM_OBJECTS16) and reports how many were produced; 0 means "push more". */
JOC_API joc_error JOC_CALL joc_stream_pull(joc_stream* stream, joc_stream_buffer* output,
uint32_t* produced_samples);
/* Marks the end of input and drains whatever the renderer still holds (the
* binaural room tail); pull the remaining samples afterwards. */
JOC_API joc_error JOC_CALL joc_stream_flush(joc_stream* stream);
/* Returns the instance to its initial state (kernel, ramps, timeline, room,
* parser, counters) so the same stream can be reused for another pass. */
JOC_API joc_error JOC_CALL joc_stream_reset(joc_stream* stream);
JOC_API joc_error JOC_CALL joc_stream_status(const joc_stream* stream,
joc_stream_status_info* out);
JOC_API joc_error JOC_CALL joc_stream_destroy(joc_stream* stream);
#ifdef __cplusplus
} /* extern "C" */
#endif
+998
View File
@@ -0,0 +1,998 @@
"""JustOneCacophony 的 E-AC-3 JOC 命令行入口。"""
import argparse
import hashlib
import json
import math
import os
from pathlib import Path
import platform
import re
import shutil
import subprocess
import sys
import tempfile
import time
PROJECT_DIR = Path(__file__).resolve().parent
SOURCE_DIR = PROJECT_DIR / "src"
if str(SOURCE_DIR) not in sys.path:
sys.path.insert(0, str(SOURCE_DIR))
import numpy as np
import adm_assemble
import adm_atmos
from adm_validate import validate
from metadata import DirectPayloadIndex, PayloadIndex, write_summary
import oamd_tracks
from renderer import JocRenderer
from native_renderer import NativeBackendUnavailable, NativeJocRenderer
from binaural_renderer import (
DEFAULT_SOFA_HRTF,
SofaBinauralRenderer,
resolve_compiled_hrtf_cache,
resolve_sofa_hrtf,
)
from rosella_binaural_renderer import (
DEFAULT_PERSONALIZED_HEADPHONE,
ROSSELLA_BLOCK_SAMPLES,
ROSSELLA_LATENCY_SAMPLES,
RosellaBinauralRenderer,
resolve_personalized_headphone,
)
from sofa_hrtf_field import DEFAULT_HRTF_CACHE_DIR
from speaker_backend import create_speaker_renderer
from speaker_layouts import (SPEAKER_LAYOUT_CHOICES, get_speaker_layout,
speaker_layout_display_name)
from speaker_wav import BinauralPcmSpool, SpeakerPcmSpool, write_pcm_wav
from variant_error import UnsupportedVariantError, write_variant_report
RATE = 48000
FRAME_SAMPLES = 1536
DEFAULT_OUTPUT_DIR = PROJECT_DIR / "output"
EAC3_DRC_SCALE_MAX = 6.0
EAC3_TARGET_LEVEL_RANGE = (-31, 0)
EAC3_DECODER_OPTION_RE = re.compile(r"(?m)^\s*-([A-Za-z0-9_]+)\s+<")
def resolve_output(source, requested=None, speaker_layout=None, *, binaural=False):
"""解析成品路径;未指定时使用项目内的 ``output`` 目录。"""
source = Path(source)
if requested is not None:
target = Path(requested)
elif speaker_layout is not None:
target = DEFAULT_OUTPUT_DIR / f"{source.stem}.{speaker_layout}.wav"
elif binaural:
target = DEFAULT_OUTPUT_DIR / f"{source.stem}.binaural.wav"
else:
target = DEFAULT_OUTPUT_DIR / (source.stem + ".adm.wav")
return target.expanduser().resolve()
def _find_default_compiled_hrtf_cache():
"""在默认 cache 目录寻找唯一的 .jochrtf;无文件返回 None,多个则报错。"""
directory = DEFAULT_HRTF_CACHE_DIR
if not directory.is_dir():
return None
candidates = sorted(directory.glob("*.jochrtf"))
if not candidates:
return None
if len(candidates) > 1:
listing = ", ".join(path.name for path in candidates[:8])
raise ValueError(
f"{directory} 下有多个 .jochrtf 缓存({listing}…),无法自动选择;"
"请用 --compiled-hrtf-cache PATH 或 --sofa-hrtf PATH 显式指定")
return candidates[0]
def resolve_binaural_hrtf_input(args, *, required):
"""解析 binaural 的 HRTF 输入。
无显式输入时按顺序回退:默认 HRTF/binaural.sofa → 默认 cache 目录下唯一的
.jochrtf → 默认 HRTF/binaural.personalized_headphone → 报错。
只校验路径,不做编译。
"""
sofa = args.sofa_hrtf
compiled = args.compiled_hrtf_cache
private = args.personalized_headphone
cache_policy = args.hrtf_cache_policy
cache_dir = args.hrtf_cache_dir
radius = args.hrtf_radius_m
if compiled is not None and cache_policy is not None:
raise ValueError("显式 .jochrtf 输入不能再指定 --hrtf-cache-policy")
if compiled is not None and radius != 1.0:
raise ValueError("显式 .jochrtf 输入不能再选择 SOFA radius shell")
if private is not None and (cache_policy is not None or cache_dir is not None
or radius != 1.0):
raise ValueError(
"Rosella 模型输入不能使用 "
"--hrtf-cache-policy/--hrtf-cache-dir/--hrtf-radius-m")
if required and sofa is None and compiled is None and private is None:
if DEFAULT_SOFA_HRTF.is_file():
sofa = DEFAULT_SOFA_HRTF
else:
compiled = _find_default_compiled_hrtf_cache()
if compiled is None and DEFAULT_PERSONALIZED_HEADPHONE.is_file():
private = DEFAULT_PERSONALIZED_HEADPHONE
if sofa is None and compiled is None and private is None:
if cache_policy is not None or cache_dir is not None or radius != 1.0:
raise ValueError("HRTF cache/radius 选项需要 --sofa-hrtf")
if required:
raise ValueError(
"--binaural 未找到 HRTF 输入:默认 "
f"{DEFAULT_SOFA_HRTF}、{DEFAULT_PERSONALIZED_HEADPHONE} 与 "
f"{DEFAULT_HRTF_CACHE_DIR} 下的 .jochrtf 缓存都不存在;请用 "
"--sofa-hrtf PATH、--compiled-hrtf-cache PATH 或 "
"--personalized-headphone PATH 指定")
return None
if sofa is None and (cache_policy is not None or cache_dir is not None
or radius != 1.0):
raise ValueError("HRTF cache/radius 选项需要 --sofa-hrtf")
effective_policy = "memory" if cache_policy is None else cache_policy
if cache_dir is not None and (sofa is None or effective_policy != "disk"):
raise ValueError("--hrtf-cache-dir 仅与 SOFA 的 disk cache policy 一起使用")
if sofa is not None:
return {
"kind": "sofa",
"path": resolve_sofa_hrtf(sofa),
"cache_policy": effective_policy,
"cache_dir": (DEFAULT_HRTF_CACHE_DIR if cache_dir is None else
cache_dir.expanduser().resolve()),
}
if compiled is not None:
return {
"kind": "compiled_cache",
"path": resolve_compiled_hrtf_cache(compiled),
"cache_policy": None,
"cache_dir": None,
}
if private is not None:
return {
"kind": "rosella",
"path": resolve_personalized_headphone(private),
"cache_policy": None,
"cache_dir": None,
}
return None
def executable(value, name):
path = shutil.which(value) if value else None
if path is None and value and Path(value).is_file():
path = str(Path(value).resolve())
if path is None:
raise FileNotFoundError(f"找不到 {name}: {value!r}")
return path
def run(command, label):
print(f"[{label}]", flush=True)
result = subprocess.run(command, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE,
text=True, encoding="utf-8", errors="replace")
if result.returncode:
tail = result.stderr[-4000:]
raise RuntimeError(f"{label} 失败(exit {result.returncode})\n{tail}")
def timed_call(timings, name, function, *args, **kwargs):
started = time.perf_counter()
try:
return function(*args, **kwargs)
finally:
timings[name] = time.perf_counter() - started
def probe_eac3_decoder_options(ffmpeg):
"""读取 ``ffmpeg -h decoder=eac3`` 暴露的 AVOption 名。"""
result = subprocess.run(
[ffmpeg, "-hide_banner", "-h", "decoder=eac3"],
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, encoding="utf-8", errors="replace")
options = frozenset(EAC3_DECODER_OPTION_RE.findall(result.stdout or ""))
# decoder 名不存在时 ffmpeg 依然返回 0,因此以“解析不到任何选项”为失败。
if not options:
raise RuntimeError(
"无法读取 FFmpeg 的 eac3 解码器选项(ffmpeg -h decoder=eac3);"
"需要带 E-AC-3 解码器的构建")
return options
def ffmpeg_version(ffmpeg):
"""FFmpeg 版本字符串;探测失败返回空串,不影响渲染。"""
try:
result = subprocess.run(
[ffmpeg, "-hide_banner", "-version"],
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, encoding="utf-8", errors="replace")
except OSError:
return ""
lines = (result.stdout or "").splitlines()
line = lines[0].strip() if lines else ""
prefix = "ffmpeg version "
return line[len(prefix):].strip() if line.startswith(prefix) else line
def eac3_decode_options(drc_scale, target_level, available):
"""构造 ``-i`` 之前的 E-AC-3 解码选项,返回 ``(argv, report 片段)``。"""
if "drc_scale" not in available:
raise RuntimeError(
"FFmpeg 的 eac3 解码器缺少 -drc_scale,无法关闭码流 DRC")
# -drc_scale 始终显式下发:0(全动态范围)不是 ffmpeg 的默认值。
argv = ["-drc_scale", format(float(drc_scale), ".10g")]
if target_level:
if "target_level" not in available:
raise RuntimeError(
"FFmpeg 的 eac3 解码器不支持 -target_level;请升级 FFmpeg "
"或去掉 --eac3-target-level")
argv += ["-target_level", str(int(target_level))]
applied = {"drc_scale": float(drc_scale), "target_level": int(target_level)}
return argv, applied
def extract_eac3(ffmpeg, source, target):
if source.suffix.lower() in (".eac3", ".ec3"):
return source
run([ffmpeg, "-hide_banner", "-loglevel", "error", "-y", "-i", str(source),
"-map", "0:a:0", "-vn", "-c:a", "copy", "-f", "eac3", str(target)],
"FFmpeg 提取 E-AC-3")
return target
def decode_core(ffmpeg, eac3, target, duration_sec=None, *, options=()):
# 5.1(side) 的 f32le 顺序为 FL FR FC LFE SL SR;JOC 使用其中 0,1,2,4,5。
command = [ffmpeg, "-hide_banner", "-loglevel", "error", "-y", *options,
"-i", str(eac3), "-map", "0:a:0", "-vn"]
if duration_sec is not None:
command.extend(["-t", f"{duration_sec:.9f}"])
command.extend(["-ac", "6", "-ar", str(RATE),
"-c:a", "pcm_f32le", "-f", "f32le", str(target)])
run(command, "FFmpeg 解码核心 5.1 PCM")
return target
def sha256(path):
digest = hashlib.sha256()
with Path(path).open("rb") as fp:
for block in iter(lambda: fp.read(16 << 20), b""):
digest.update(block)
return digest.hexdigest()
def choose_pcm_output_format(requested_format, clip_action, peak, clipped_values,
*, input_func=input, interactive=None):
"""Resolve int24 clipping interactively or through an explicit policy."""
if requested_format != "int24" or clipped_values == 0:
return requested_format
print(
f"[clip] int24 将发生削波:peak={peak:.9g},超出 [-1,1] 的样本值={clipped_values}",
file=sys.stderr, flush=True)
action = clip_action
if action == "ask":
if interactive is None:
interactive = bool(getattr(sys.stdin, "isatty", lambda: False)())
if not interactive:
raise RuntimeError(
"检测到 int24 削波,但当前不是交互终端;请使用 "
"--clip-action continue、--clip-action float32 或 --clip-action abort")
while True:
answer = input_func(
"继续写 int24 并截断 [i] / 改为 float32 [f,默认] / 取消 [a]:"
).strip().lower()
if answer in ("", "f", "float", "float32"):
action = "float32"
break
if answer in ("i", "int", "int24", "c", "continue"):
action = "continue"
break
if answer in ("a", "abort", "q", "quit", "n", "no"):
action = "abort"
break
print("请输入 i、f 或 a。", file=sys.stderr, flush=True)
if action == "continue":
print("[clip] 将继续写 int24,超范围值会截断到 [-1,1]。", flush=True)
return "int24"
if action == "float32":
print("[clip] 已切换为 float32 WAV,不执行截断。", flush=True)
return "float32"
if action == "abort":
raise RuntimeError("用户因 int24 削波取消输出")
raise ValueError(f"未知 clip action: {action}")
# Backward-compatible public name used by existing tests and callers.
choose_speaker_output_format = choose_pcm_output_format
def resolve_metadata(args, eac3, temp_dir):
if args.metadata_dir:
directory = Path(args.metadata_dir).resolve()
return PayloadIndex(directory), "sidecar", directory
if args.metadata_backend == "sidecar":
raise ValueError("metadata-backend=sidecar 时必须提供 --metadata-dir")
cache_dir = (args.metadata_cache.expanduser().resolve()
if args.metadata_cache else None)
max_frames = (math.ceil(args.duration * RATE / FRAME_SAMPLES)
if args.duration is not None else None)
index = DirectPayloadIndex.from_eac3(
eac3, max_frames=max_frames, cache_dir=cache_dir)
return index, "python-emdf-memory", cache_dir
def variant_call(output, source, function, *args, **kwargs):
"""执行一个阶段;遇到未知变体时在目标文件旁写结构化报告。"""
try:
return function(*args, **kwargs)
except UnsupportedVariantError as exc:
report_path = Path(str(output) + ".variant-error.json")
write_variant_report(report_path, exc, input_path=source, output_path=output)
print(f"[VARIANT] {exc}", file=sys.stderr, flush=True)
print(f"[VARIANT] 维修报告: {report_path}", file=sys.stderr, flush=True)
raise
def create_renderer(backend, gain, native_library=None, native_threads=None):
"""选择整帧 DSP 后端;auto 优先使用 lib 中当前平台的原生构建。"""
if backend in ("auto", "native"):
try:
decoder = NativeJocRenderer(
output_scale=gain, library_path=native_library, threads=native_threads)
info = {
"name": "native",
"library": str(decoder.library_path),
"build": decoder.build_info,
"threads": decoder.threads,
}
print(f"[backend] native: {info['build']} threads={info['threads']} "
f"({info['library']})", flush=True)
return decoder, info
except (NativeBackendUnavailable, OSError) as exc:
print(f"[backend] native unavailable, falling back to Python: {exc}", flush=True)
decoder = JocRenderer(output_scale=gain)
info = {"name": "python", "library": None, "build": None, "threads": None}
print("[backend] python/numpy", flush=True)
return decoder, info
def render(index, bed_path, frame_count, raw_path, gain, progress_every,
backend="auto", native_library=None, native_threads=None, frame_sink=None,
speaker_renderer=None, speaker_sink=None, speaker_metadata_offset=1473,
binaural_renderer=None, binaural_sink=None, binaural_metadata_offset=1473,
raw_scale=1.0):
values = np.memmap(bed_path, dtype=np.float32, mode="r")
frame_width = FRAME_SAMPLES * 6
if values.size % frame_width:
raise ValueError(f"FFmpeg PCM 长度不是 1536×6 的整数倍: {values.size}")
bed = values.reshape(-1, FRAME_SAMPLES, 6)
if len(bed) < frame_count:
raise ValueError(f"PCM 只有 {len(bed)} 帧,元数据需要 {frame_count} 帧")
output = (np.memmap(raw_path, dtype=np.float32, mode="w+",
shape=(frame_count, FRAME_SAMPLES, 16))
if raw_path is not None else None)
decoder, backend_info = create_renderer(backend, gain, native_library, native_threads)
started = time.perf_counter()
dsp_seconds = 0.0
adm_stream_seconds = 0.0
raw_write_seconds = 0.0
speaker_render_seconds = 0.0
speaker_write_seconds = 0.0
binaural_render_seconds = 0.0
binaural_write_seconds = 0.0
elapsed = 0.0
try:
for frame_number, row in enumerate(index.rows[:frame_count]):
bed6 = np.asarray(bed[frame_number], dtype=np.float32)
subs = index.subpayloads(row)
stage = time.perf_counter()
pcm16, _ = decoder.render_subpayloads(
subs, bed6[:, [0, 1, 2, 4, 5]].T, bed6[:, 3])
dsp_seconds += time.perf_counter() - stage
if output is not None:
stage = time.perf_counter()
output[frame_number] = np.multiply(
pcm16.T, np.float32(raw_scale), dtype=np.float32)
raw_write_seconds += time.perf_counter() - stage
if frame_sink is not None:
stage = time.perf_counter()
frame_sink.write_frame(pcm16)
adm_stream_seconds += time.perf_counter() - stage
if speaker_renderer is not None:
stage = time.perf_counter()
speaker_pcm = speaker_renderer.render_frame(
pcm16.T, subs.get(11), speaker_metadata_offset)
speaker_render_seconds += time.perf_counter() - stage
stage = time.perf_counter()
speaker_sink.write_frame(speaker_pcm)
speaker_write_seconds += time.perf_counter() - stage
if binaural_renderer is not None:
payload = subs.get(11)
outer_offset = (
index.subpayload_sample_offset(row, 11)
if payload is not None and hasattr(index, "subpayload_sample_offset")
else 0
)
stage = time.perf_counter()
binaural_pcm = binaural_renderer.render_frame(
pcm16.T, payload, binaural_metadata_offset,
outer_sample_offset=outer_offset)
binaural_render_seconds += time.perf_counter() - stage
if len(binaural_pcm):
stage = time.perf_counter()
binaural_sink.write_frame(binaural_pcm)
binaural_write_seconds += time.perf_counter() - stage
done = frame_number + 1
if done % progress_every == 0 or done == frame_count:
elapsed = time.perf_counter() - started
speed = done / max(elapsed, 1e-9)
eta = (frame_count - done) / max(speed, 1e-9)
print(f"[JOC:{backend_info['name']}] {done}/{frame_count} "
f"{speed:.1f} frame/s ETA {eta:.1f}s", flush=True)
if binaural_renderer is not None:
stage = time.perf_counter()
binaural_tail = binaural_renderer.finish()
binaural_render_seconds += time.perf_counter() - stage
if len(binaural_tail):
stage = time.perf_counter()
binaural_sink.write_frame(binaural_tail)
binaural_write_seconds += time.perf_counter() - stage
if output is not None:
output.flush()
elapsed = time.perf_counter() - started
finally:
close = getattr(decoder, "close", None)
if close is not None:
close()
close = getattr(speaker_renderer, "close", None)
if close is not None:
close()
close = getattr(binaural_renderer, "close", None)
if close is not None:
close()
breakdown = {
"pipeline_wall_seconds": elapsed,
"dsp_and_joc_parse_seconds": dsp_seconds,
"adm_stream_write_seconds": adm_stream_seconds,
"raw_float_write_seconds": raw_write_seconds,
"speaker_render_seconds": speaker_render_seconds,
"speaker_spool_write_seconds": speaker_write_seconds,
"binaural_render_seconds": binaural_render_seconds,
"binaural_spool_write_seconds": binaural_write_seconds,
}
return dsp_seconds, backend_info, breakdown
def build_parser():
parser = argparse.ArgumentParser(
description=("JustOneCacophony (JOC):E-AC-3 JOC → 25ch ADM BWF、"
"扬声器 WAV 或公开 SOFA 双耳 WAV"))
parser.add_argument("input", type=Path, help="输入 .m4a/.eac3/.ec3")
parser.add_argument("-o", "--output", type=Path, help="输出文件;默认按模式和布局命名")
parser.add_argument("--speaker-output", type=Path,
help="扬声器 WAV 路径;仅与 --speaker-layout 一起使用")
parser.add_argument("--binaural-output", type=Path,
help="双耳 WAV 路径;仅与 --binaural 一起使用")
direct_mode = parser.add_mutually_exclusive_group()
direct_mode.add_argument("--speaker-layout", choices=SPEAKER_LAYOUT_CHOICES,
help="直接扬声器渲染布局,例如 2.0、5.1、7.1.2")
direct_mode.add_argument("--binaural", action="store_true",
help="直接 SOFA 双耳渲染;不生成临时 ADM BWF")
parser.add_argument("--speaker-format", choices=("float32", "int24"), default="float32",
help="扬声器 WAV 格式,默认 float32")
parser.add_argument("--binaural-format", choices=("float32", "int24"), default="float32",
help="双耳 WAV 格式,默认 float32")
parser.add_argument("--clip-action", choices=("ask", "continue", "float32", "abort"),
default="ask",
help="int24 削波处理:交互询问、继续截断、改 float32 或中止")
parser.add_argument("--speaker-metadata-offset", type=int, default=1473,
help="扬声器渲染 metadata 相对帧偏移,默认 1473 samples")
parser.add_argument("--binaural-mode", choices=("off", "near", "mid", "far"),
default="mid",
help="双耳渲染模式,默认 mid(人为指定的渲染提示,非码流 "
"原始元数据);直接双耳渲染与 ADM BWF 的 DBMD 提示共用。"
"off 仅用于 ADM BWF:关闭 DBMD 双耳提示(编码 0)")
hrtf_input = parser.add_mutually_exclusive_group()
hrtf_input.add_argument(
"--sofa-hrtf", type=Path,
help="SimpleFreeFieldHRIR SOFA;缺省时依次尝试 HRTF/binaural.sofa、"
"output/hrtf-cache 下唯一的 .jochrtf、"
"HRTF/binaural.personalized_headphone,均无则报错")
hrtf_input.add_argument(
"--compiled-hrtf-cache", type=Path,
help="高级入口:显式读取 JOC .jochrtf compiled cache")
hrtf_input.add_argument(
"--personalized-headphone", type=Path, nargs="?",
const=DEFAULT_PERSONALIZED_HEADPHONE,
help="Rosella .personalized_headphone 模型;不带路径时默认 "
"HRTF/binaural.personalized_headphone")
parser.add_argument(
"--hrtf-cache-policy", choices=("none", "memory", "disk"), default=None,
help="SOFA 编译缓存;默认 memory,disk 写入可删除的 .jochrtf")
parser.add_argument(
"--hrtf-cache-dir", type=Path,
help="disk cache 目录;默认 output/hrtf-cache")
parser.add_argument(
"--hrtf-radius-m", type=float, default=1.0,
help="选择最近的 SOFA measurement-radius shell,默认 1.0 m")
parser.add_argument("--binaural-tail-seconds", type=float, default=5.0,
help="双耳 room/filterbank flush 上限,默认 5 秒")
parser.add_argument("--binaural-tail-threshold", type=float, default=1.0e-8,
help="双耳尾声裁切阈值,默认 1e-8;主体至少保留原时长")
parser.add_argument("--binaural-chunk-frames", type=int, default=64,
help="双耳内部批处理 E-AC-3 帧数,默认 64")
parser.add_argument("--gain-db", type=float, default=0.0,
help="成品增益 dB,默认 0;双耳路径以 float64 应用")
parser.add_argument("--duration", type=float, help="只处理开头指定秒数")
parser.add_argument("--object-delay-samples", type=int, default=1473,
help="对象 PCM/OAMD 时间补偿;ADM 与双耳默认 1473 samples")
parser.add_argument("--trajectory-mode", choices=("compact", "dense64"), default="compact",
help="ADM 对象轨迹表示;直接双耳路径不序列化 AXML")
parser.add_argument("--ffmpeg", default=os.environ.get("FFMPEG", "ffmpeg"))
parser.add_argument("--eac3-drc-scale", type=float, default=0.0,
help="E-AC-3 解码器 -drc_scale:0=关闭码流 dynrng(全动态范围),"
"1=码流作者意图,>1 非对称;默认 0")
parser.add_argument("--eac3-target-level", type=int, default=0,
help="E-AC-3 解码器 -target_level:按码流 dialnorm 归一化电平,"
"增益约 target_level - dialnorm dB;0=不施加,默认 0")
parser.add_argument("--backend", choices=("auto", "native", "python"), default="auto",
help="JOC/扬声器 DSP 后端;SOFA 双耳 DSP 当前使用 Python")
parser.add_argument("--native-library", type=Path,
help="显式指定原生库;默认从单层 lib 目录选择当前平台文件")
parser.add_argument("--native-threads", type=int,
help="原生 DSP 总线程数;默认在 4 核以上使用 2,可用环境变量 EAC3JOC_NATIVE_THREADS 覆盖")
metadata_source = parser.add_mutually_exclusive_group()
metadata_source.add_argument("--metadata-dir", type=Path,
help="含 frames.csv 和 emdf/ 或 payloads/ 的元数据 sidecar")
metadata_source.add_argument("--metadata-cache", type=Path,
help="把直接 EMDF 扫描或兼容桥结果持久保存到此目录")
parser.add_argument("--metadata-backend", choices=("auto", "emdf", "sidecar"),
default="auto", help="直接扫描连续 EMDF,或读取现有 sidecar")
parser.add_argument("--print-metadata", choices=("none", "summary", "frames"), default="none",
help="诊断元数据输出;默认 none,避免转换前重复完整解析")
parser.add_argument("--metadata-json", type=Path, help="元数据汇总 JSON 路径")
parser.add_argument("--metadata-only", action="store_true", help="解析/打印元数据后退出")
parser.add_argument("--keep-raw", action="store_true", help="额外保留 16ch f32le 对象中间文件")
parser.add_argument("--skip-sha256", action="store_true",
help="跳过最终文件 SHA-256 全量复扫以缩短大文件处理时间")
parser.add_argument("--progress-every", type=int, default=500)
return parser
def main(argv=None):
# Windows 控制台的活动代码页未必能表示日文文件名;保留信息并避免
# UnicodeEncodeError 中断长任务。支持 UTF-8 的终端仍会原样显示。
for stream in (sys.stdout, sys.stderr):
if hasattr(stream, "reconfigure"):
stream.reconfigure(encoding="utf-8", errors="backslashreplace")
args = build_parser().parse_args(argv)
source = args.input.expanduser().resolve()
if not source.is_file():
raise FileNotFoundError(source)
speaker_mode = args.speaker_layout is not None
binaural_mode = bool(args.binaural)
binaural_render_mode = args.binaural_mode
if binaural_mode and binaural_render_mode == "off":
raise ValueError(
"--binaural-mode off 仅用于 ADM BWF 输出(关闭 DBMD 双耳提示);"
"直接双耳渲染请使用 near/mid/far")
if args.speaker_output is not None and not speaker_mode:
raise ValueError("--speaker-output 必须与 --speaker-layout 一起使用")
if args.binaural_output is not None and not binaural_mode:
raise ValueError("--binaural-output 必须与 --binaural 一起使用")
specific_outputs = [value for value in (args.speaker_output, args.binaural_output)
if value is not None]
if args.output is not None and specific_outputs:
raise ValueError("-o/--output 与 --speaker-output/--binaural-output 不能同时使用")
if len(specific_outputs) > 1:
raise ValueError("--speaker-output 与 --binaural-output 不能同时使用")
if args.speaker_metadata_offset < 0:
raise ValueError("speaker-metadata-offset 不能为负数")
hrtf_options_used = any((
args.sofa_hrtf is not None,
args.compiled_hrtf_cache is not None,
args.personalized_headphone is not None,
args.hrtf_cache_policy is not None,
args.hrtf_cache_dir is not None,
args.hrtf_radius_m != 1.0,
))
if hrtf_options_used and not binaural_mode:
raise ValueError("SOFA/HRTF 选项仅与 --binaural 一起使用")
if (not math.isfinite(args.binaural_tail_seconds)
or args.binaural_tail_seconds < 0):
raise ValueError("binaural-tail-seconds 必须是非负有限值")
if (not math.isfinite(args.binaural_tail_threshold)
or args.binaural_tail_threshold < 0):
raise ValueError("binaural-tail-threshold 必须是非负有限值")
if args.binaural_chunk_frames <= 0:
raise ValueError("binaural-chunk-frames 必须大于 0")
if not math.isfinite(args.hrtf_radius_m) or args.hrtf_radius_m <= 0.0:
raise ValueError("hrtf-radius-m 必须是正有限值")
requested_output = (args.speaker_output if args.speaker_output is not None
else args.binaural_output if args.binaural_output is not None
else args.output)
output = resolve_output(
source, requested_output, args.speaker_layout if speaker_mode else None,
binaural=binaural_mode)
output.parent.mkdir(parents=True, exist_ok=True)
if args.duration is not None and args.duration <= 0:
raise ValueError("duration 必须大于 0")
if args.object_delay_samples < 0:
raise ValueError("object-delay-samples 不能为负数")
gain_float64 = 10.0 ** (args.gain_db / 20.0)
gain = np.float32(gain_float64)
if not math.isfinite(gain_float64) or not np.isfinite(gain):
raise ValueError("gain-db 超出支持范围")
if (not math.isfinite(args.eac3_drc_scale)
or not 0.0 <= args.eac3_drc_scale <= EAC3_DRC_SCALE_MAX):
raise ValueError(f"eac3-drc-scale 必须在 0..{EAC3_DRC_SCALE_MAX:g} 之间")
if not (EAC3_TARGET_LEVEL_RANGE[0] <= args.eac3_target_level
<= EAC3_TARGET_LEVEL_RANGE[1]):
raise ValueError("eac3-target-level 必须在 -31..0 之间")
binaural_hrtf_input = resolve_binaural_hrtf_input(
args, required=binaural_mode and not args.metadata_only)
ffmpeg = executable(args.ffmpeg, "FFmpeg")
decode_options = ()
decode_option_info = None
if not args.metadata_only:
available = probe_eac3_decoder_options(ffmpeg)
decode_options, applied = eac3_decode_options(
args.eac3_drc_scale, args.eac3_target_level, available)
decode_option_info = {
"version": ffmpeg_version(ffmpeg),
"eac3_decode_options": applied,
}
print(f"[decode] ffmpeg {decode_option_info['version']} "
f"drc_scale={args.eac3_drc_scale:g} "
f"target_level={args.eac3_target_level}", flush=True)
total_started = time.perf_counter()
timings = {}
with tempfile.TemporaryDirectory(prefix="eac3joc-", dir=output.parent) as temporary:
temp_dir = Path(temporary)
eac3 = timed_call(timings, "extract_eac3", extract_eac3,
ffmpeg, source, temp_dir / "input.eac3")
index, metadata_backend, metadata_cache_dir = timed_call(
timings, "resolve_metadata", variant_call,
output, source, resolve_metadata, args, eac3, temp_dir)
timings["load_metadata_index"] = 0.0
frame_count = len(index)
if args.duration is not None:
frame_count = min(frame_count, math.ceil(args.duration * RATE / FRAME_SAMPLES))
duration_sec = frame_count * FRAME_SAMPLES / RATE
need_metadata_summary = (
args.metadata_only or args.metadata_json is not None or args.print_metadata != "none")
if need_metadata_summary:
metadata_json = (args.metadata_json or Path(str(output) + ".metadata.json")).resolve()
summary = timed_call(
timings, "metadata_summary", variant_call,
output, source, write_summary, index, metadata_json, limit=frame_count,
print_frames=args.print_metadata == "frames")
if args.print_metadata == "summary":
print("[metadata] " + json.dumps(summary, ensure_ascii=False, separators=(",", ":")))
print(f"[metadata] backend={metadata_backend} frames={frame_count} -> {metadata_json}")
else:
metadata_json = None
timings["metadata_summary"] = 0.0
print(f"[metadata] backend={metadata_backend} frames={frame_count} summary=skipped")
if args.metadata_only:
return 0
bed_path = timed_call(
timings, "decode_core", decode_core,
ffmpeg, eac3, temp_dir / "core51_f32le.raw", duration_sec,
options=decode_options)
raw_path = (output.with_name(output.name + ".objects16.f32le")
if args.keep_raw else None)
master = None
speaker_backend_info = None
speaker_wav_info = None
speaker_clip_info = None
speaker_actual_format = None
binaural_backend_info = None
binaural_hrtf_report = None
binaural_wav_info = None
binaural_clip_info = None
binaural_actual_format = None
if speaker_mode:
timings["create_binaural_renderer"] = 0.0
layout = get_speaker_layout(args.speaker_layout)
speaker_name = speaker_layout_display_name(layout)
speaker_decoder, speaker_backend_info = create_speaker_renderer(
layout, backend=args.backend, native_library=args.native_library)
fallback = speaker_backend_info.get("fallback_reason")
if fallback:
print(f"[speaker] native unavailable, falling back to Python: {fallback}",
flush=True)
print(f"[speaker] layout={speaker_name} backend={speaker_backend_info['name']} "
f"channels={layout.channel_count}", flush=True)
spool = SpeakerPcmSpool(
temp_dir / "speaker_interleaved_f32.raw",
frame_count * FRAME_SAMPLES, layout.channel_count)
try:
render_seconds, renderer_backend, render_breakdown = timed_call(
timings, "render_and_stream", variant_call,
output, source, render, index, bed_path, frame_count, raw_path, gain,
max(1, args.progress_every), args.backend, args.native_library,
args.native_threads, None, speaker_decoder, spool,
args.speaker_metadata_offset)
spool.finalize()
speaker_actual_format = choose_pcm_output_format(
args.speaker_format, args.clip_action, spool.peak,
spool.clipped_values)
speaker_wav_info = timed_call(
timings, "write_speaker_wav", write_pcm_wav,
output, spool.values, speaker_actual_format, rate=RATE)
speaker_clip_info = {
"peak": spool.peak,
"over_unity_values": spool.clipped_values,
"requested_format": args.speaker_format,
"actual_format": speaker_actual_format,
"clip_action": args.clip_action,
}
finally:
spool.close()
timings["build_adm_tracks"] = 0.0
timings["finalize_adm"] = 0.0
timings["validate_adm"] = 0.0
info = (f"speaker layout={speaker_name}, format={speaker_actual_format}, "
f"peak={speaker_clip_info['peak']:.9g}")
elif binaural_mode:
hrtf_source = binaural_hrtf_input
common_options = {
"mode": binaural_render_mode,
"object_delay_samples": args.object_delay_samples,
"tail_seconds": args.binaural_tail_seconds,
"output_gain": gain_float64,
"chunk_frames": args.binaural_chunk_frames,
}
if hrtf_source["kind"] == "sofa":
binaural_decoder = None
if args.backend in ("auto", "native"):
try:
from sofa_native_backend import create_native_sofa_renderer
binaural_decoder = timed_call(
timings, "create_binaural_renderer",
create_native_sofa_renderer,
hrtf_source["path"],
cache_policy=hrtf_source["cache_policy"],
cache_dir=hrtf_source["cache_dir"],
shell_radius_m=args.hrtf_radius_m,
**common_options)
except (ImportError, OSError, RuntimeError, ValueError) as exc:
print(
f"[binaural] native SOFA backend unavailable "
f"({exc.__class__.__name__}: {exc}); "
f"falling back to Python", flush=True)
binaural_decoder = None
if binaural_decoder is None:
binaural_decoder = timed_call(
timings, "create_binaural_renderer",
SofaBinauralRenderer.from_sofa,
hrtf_source["path"],
cache_policy=hrtf_source["cache_policy"],
cache_dir=hrtf_source["cache_dir"],
shell_radius_m=args.hrtf_radius_m,
**common_options)
elif hrtf_source["kind"] == "rosella":
binaural_decoder = timed_call(
timings, "create_binaural_renderer",
RosellaBinauralRenderer,
hrtf_source["path"],
mode=binaural_render_mode,
object_delay_samples=args.object_delay_samples,
tail_seconds=args.binaural_tail_seconds,
output_gain=gain_float64,
chunk_frames=args.binaural_chunk_frames,
backend=args.backend,
native_library=args.native_library)
else:
binaural_decoder = None
if args.backend in ("auto", "native"):
try:
from sofa_native_backend import (
create_native_compiled_cache_renderer)
binaural_decoder = timed_call(
timings, "create_binaural_renderer",
create_native_compiled_cache_renderer,
hrtf_source["path"],
**common_options)
except (ImportError, OSError, RuntimeError, ValueError) as exc:
print(
f"[binaural] native SOFA backend unavailable "
f"({exc.__class__.__name__}: {exc}); "
f"falling back to Python", flush=True)
binaural_decoder = None
if binaural_decoder is None:
binaural_decoder = timed_call(
timings, "create_binaural_renderer",
SofaBinauralRenderer.from_compiled_cache,
hrtf_source["path"],
**common_options)
print(
f"[binaural] mode={binaural_render_mode} "
f"backend={binaural_decoder.dsp_backend} "
f"precision=float64/complex128 "
f"hrtf={hrtf_source['kind']}:{hrtf_source['path']}", flush=True)
if hrtf_source["kind"] == "rosella":
flush_samples = math.ceil(
(args.binaural_tail_seconds * RATE
+ ROSSELLA_LATENCY_SAMPLES + ROSSELLA_BLOCK_SAMPLES)
/ ROSSELLA_BLOCK_SAMPLES) * ROSSELLA_BLOCK_SAMPLES
spool_capacity = frame_count * FRAME_SAMPLES + flush_samples
else:
spool_capacity = (
frame_count * FRAME_SAMPLES
+ binaural_decoder.finish_capacity_samples)
spool = BinauralPcmSpool(
temp_dir / "binaural_interleaved_f64.raw",
spool_capacity,
tail_threshold=args.binaural_tail_threshold)
try:
render_seconds, renderer_backend, render_breakdown = timed_call(
timings, "render_and_stream", variant_call,
output, source, render, index, bed_path, frame_count, raw_path,
np.float32(1.0), max(1, args.progress_every),
backend=args.backend, native_library=args.native_library,
native_threads=args.native_threads,
binaural_renderer=binaural_decoder, binaural_sink=spool,
binaural_metadata_offset=args.object_delay_samples, raw_scale=gain)
spool.finalize(minimum_samples=frame_count * FRAME_SAMPLES)
binaural_actual_format = choose_pcm_output_format(
args.binaural_format, args.clip_action, spool.peak,
spool.clipped_values)
binaural_wav_info = timed_call(
timings, "write_binaural_wav", write_pcm_wav,
output, spool.values, binaural_actual_format, rate=RATE)
binaural_clip_info = {
"peak": spool.peak,
"over_unity_values": spool.clipped_values,
"requested_format": args.binaural_format,
"actual_format": binaural_actual_format,
"clip_action": args.clip_action,
"tail_threshold": args.binaural_tail_threshold,
"source_samples": frame_count * FRAME_SAMPLES,
"kept_samples": spool.sample_count,
}
binaural_backend_info = binaural_decoder.backend_info
if hrtf_source["kind"] == "rosella":
binaural_hrtf_report = {
"input_kind": "rosella",
"input_path": str(binaural_decoder.model_path.resolve()),
"model_coefficient_sha256": (
binaural_decoder.model.coefficient_sha256),
"cache_policy": None,
}
else:
binaural_hrtf_report = {
"input_kind": binaural_backend_info["hrtf_input_kind"],
"input_path": binaural_backend_info["hrtf_input_path"],
"source_sha256": (
binaural_backend_info["field"]["source_sha256"]),
"cache_policy": binaural_backend_info["cache_policy"],
"cache_key": (
binaural_backend_info["field"]["cache_key"]),
"format_version": (
binaural_backend_info["field"]["format_version"]),
}
finally:
spool.close()
timings["build_adm_tracks"] = 0.0
timings["finalize_adm"] = 0.0
timings["validate_adm"] = 0.0
info = (f"binaural mode={binaural_render_mode}, "
f"format={binaural_actual_format}, "
f"peak={binaural_clip_info['peak']:.9g}, "
f"samples={binaural_clip_info['kept_samples']}")
else:
timings["create_binaural_renderer"] = 0.0
master = adm_assemble.StreamingMaster(
output, duration_sec, rate=RATE,
joc_binaural_mode=adm_atmos.JOC_BINAURAL_MODES[args.binaural_mode])
try:
render_seconds, renderer_backend, render_breakdown = timed_call(
timings, "render_and_stream", variant_call,
output, source, render, index, bed_path, frame_count, raw_path, gain,
max(1, args.progress_every), args.backend, args.native_library,
args.native_threads, master)
tracks = timed_call(
timings, "build_adm_tracks", variant_call,
output, source, oamd_tracks.build_adm_tracks,
index, index.rows[:frame_count], rate=RATE, frame_samples=FRAME_SAMPLES,
object_delay_samples=args.object_delay_samples,
trajectory_mode=args.trajectory_mode)
timed_call(timings, "finalize_adm", master.finalize, tracks)
except Exception:
master.abort()
raise
errors, info = timed_call(timings, "validate_adm", validate, str(output))
if errors:
raise RuntimeError("ADM 校验失败: " + "; ".join(errors))
# Windows 不允许删除仍被 NumPy memmap 持有的临时 core/raw;显式回收闭包。
import gc
gc.collect()
if args.skip_sha256:
output_sha = None
timings["sha256"] = 0.0
else:
output_sha = timed_call(timings, "sha256", sha256, output)
total_seconds = time.perf_counter() - total_started
mode_name = "speaker" if speaker_mode else "binaural" if binaural_mode else "adm"
report = {
"input": str(source),
"output": str(output),
"mode": mode_name,
"metadata": str(metadata_json) if metadata_json is not None else None,
"metadata_backend": metadata_backend,
"metadata_cache": str(metadata_cache_dir) if metadata_cache_dir is not None else None,
"frames": frame_count,
"duration_sec": duration_sec,
"gain_db": args.gain_db,
"gain_float32": float(gain),
"gain_float64": float(gain_float64),
"object_delay_samples": (None if speaker_mode else args.object_delay_samples),
"trajectory_mode": args.trajectory_mode if mode_name == "adm" else None,
"binaural_mode_value": (
adm_atmos.JOC_BINAURAL_MODES[args.binaural_mode]
if mode_name == "adm" else None),
"render_seconds": render_seconds,
"render_breakdown": render_breakdown,
"renderer_backend": renderer_backend,
"speaker_renderer_backend": speaker_backend_info,
"speaker_layout": args.speaker_layout if speaker_mode else None,
"speaker_metadata_offset": args.speaker_metadata_offset if speaker_mode else None,
"speaker_clip": speaker_clip_info,
"speaker_wav": speaker_wav_info,
"binaural_renderer_backend": binaural_backend_info,
"binaural_mode": (
args.binaural_mode if (binaural_mode or mode_name == "adm") else None),
"binaural_hrtf": (
binaural_hrtf_report if binaural_backend_info else None),
"binaural_clip": binaural_clip_info,
"binaural_wav": binaural_wav_info,
"output_clip": speaker_clip_info if speaker_mode else binaural_clip_info,
"output_wav": speaker_wav_info if speaker_mode else binaural_wav_info,
"streaming_adm": mode_name == "adm",
"kept_raw": str(raw_path) if raw_path is not None else None,
"timings": timings,
"total_seconds": total_seconds,
"adm_validation": info if mode_name == "adm" else None,
"adm_metadata": getattr(master, "metadata_info", None) if master is not None else None,
"sha256": output_sha,
"python": platform.python_version(),
"numpy": np.__version__,
"ffmpeg": decode_option_info,
}
report_path = Path(str(output) + ".report.json")
report_path.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
print(f"[PASS] {output}")
if report["sha256"] is None:
print(f"[PASS] {info}; SHA-256 skipped")
else:
print(f"[PASS] {info}; SHA-256={report['sha256']}")
if speaker_mode:
print(f"[time] JOC-DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
f"speaker={render_breakdown['speaker_render_seconds']:.2f}s "
f"pipeline={render_breakdown['pipeline_wall_seconds']:.2f}s "
f"total={report['total_seconds']:.2f}s")
elif binaural_mode:
print(f"[time] JOC-DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
f"binaural={render_breakdown['binaural_render_seconds']:.2f}s "
f"pipeline={render_breakdown['pipeline_wall_seconds']:.2f}s "
f"total={report['total_seconds']:.2f}s")
else:
print(f"[time] DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
f"render+ADM-stream={render_breakdown['pipeline_wall_seconds']:.2f}s "
f"total={report['total_seconds']:.2f}s")
print(f"[report] {report_path}")
return 0
if __name__ == "__main__":
try:
raise SystemExit(main())
except KeyboardInterrupt:
raise SystemExit(130)
+74
View File
@@ -0,0 +1,74 @@
cmake_minimum_required(VERSION 3.20)
project(eac3joc_core VERSION 1.0.0 LANGUAGES CXX)
include(GNUInstallDirs)
find_package(Threads REQUIRED)
add_library(eac3joc_core SHARED
src/eac3joc_core.cpp
src/speaker_renderer.cpp
src/binaural_renderer.cpp
src/sofa_binaural_renderer.cpp
src/joc_huffman_tables.h
src/qmf_tables.h
src/speaker_layouts.h
)
target_compile_features(eac3joc_core PRIVATE cxx_std_20)
target_include_directories(eac3joc_core PRIVATE
"${CMAKE_CURRENT_SOURCE_DIR}/include"
)
target_link_libraries(eac3joc_core PRIVATE Threads::Threads)
set_target_properties(eac3joc_core PROPERTIES
OUTPUT_NAME "eac3joc_core"
CXX_VISIBILITY_PRESET hidden
VISIBILITY_INLINES_HIDDEN YES
POSITION_INDEPENDENT_CODE YES
)
if(MSVC)
set_property(TARGET eac3joc_core PROPERTY
MSVC_RUNTIME_LIBRARY "MultiThreaded$<$<CONFIG:Debug>:Debug>")
target_compile_options(eac3joc_core PRIVATE
/W4 /permissive- /Zc:__cplusplus /utf-8 /EHsc /GR- /fp:precise
$<$<CONFIG:Release>:/O2>
$<$<CONFIG:Release>:/Oi>
$<$<CONFIG:Release>:/GL>
)
target_link_options(eac3joc_core PRIVATE
$<$<CONFIG:Release>:/LTCG>
/INCREMENTAL:NO /OPT:REF /OPT:ICF
)
else()
target_compile_options(eac3joc_core PRIVATE
-Wall -Wextra -Wpedantic -fno-fast-math
$<$<CONFIG:Release>:-O3>
)
endif()
# Install directly into the chosen prefix. Recommended invocation from the
# repository root:
# cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release \
# -DCMAKE_INSTALL_PREFIX=<repo>/lib
# cmake --build build/cmake --config Release
# cmake --install build/cmake --config Release
#
# Result names are supplied by the platform toolchain:
# Windows: eac3joc_core.dll (+ eac3joc_core.lib import library)
# Linux: libeac3joc_core.so
# macOS: libeac3joc_core.dylib
install(TARGETS eac3joc_core
RUNTIME DESTINATION .
LIBRARY DESTINATION .
ARCHIVE DESTINATION .
)
if(MSVC)
install(FILES "$<TARGET_PDB_FILE:eac3joc_core>"
DESTINATION .
OPTIONAL
)
endif()
@@ -2,11 +2,7 @@
#include <stdint.h>
/* EJOC_STATIC: this branch exists for building the sources directly into an application, where nothing is imported or exported. */
#if defined(EJOC_STATIC)
#define EJOC_API
#define EJOC_CALL __cdecl
#elif defined(_WIN32)
#if defined(_WIN32)
#if defined(EJOC_BUILD_DLL)
#define EJOC_API __declspec(dllexport)
#else
@@ -1,53 +1,3 @@
// Derived from the upstream native/src/sofa_binaural_renderer.cpp (JustOneCacophony
// @ 6bc2c2885666bb151bb66af93472199af9a99b81, sha256 81b485e4c71907672ab308ddc38228
// 4acf29161cee93d7906785a4cce58941c7) and a drop-in replacement for it: the same
// ejoc_sofa_binaural_* C ABI. The changes are table hoisting, result memoisation,
// input validation and a dispatched SIMD cascade -- no arithmetic is reordered,
// reassociated or contracted, so the output is bit-identical to the upstream
// implementation (verified: identical FNV-1a over the exact output bit patterns at
// 8/64/512/2000 blocks -- ef7ca498a6bb19b0 at 2000 -- and identical SHA-256 of the
// joc_dump --binaural-out float64 stream for 1000/2000/4000 frames):
//
// The deviations from the upstream implementation, and why each one keeps the
// output bit-identical:
//
// Hoisted invariants. QmfAnalysis::configure builds the two modulation tables
// (cos/sin of -kPi*p/128 and of -3.0*(b+0.5)*kPi/128) once, instead of
// recomputing 384 transcendental calls per slot and per channel (49,152 per
// 512-sample block); RealSh::evaluate hoists the normalization (which depends
// only on (degree, |order|)) and the six pmm_value(m, x) Legendre seeds out of
// the 36-term loop. In both cases the stored values are std::cos/std::sin of
// the identical double expressions, evaluated once, so they are the same bits.
//
// Memoised results. set_source memoises the seven paths it derives from a
// source position. The wrapper calls it for every source on every 512-sample
// block and the timeline holds a position constant between OAMD updates, so
// most calls were rebuilding an identical result. A hit requires the exact bit
// pattern of every input make_path reads to match the call that produced the
// stored paths; the fade state machine and the late-send envelope are
// unchanged, so the emitted samples are the same bits.
//
// Input validation. configure_room rejects zero FDN / allpass delay-line
// lengths, which would otherwise reach an integer divide-by-zero from the
// public C ABI, and render_paths wraps the history index with an explicit
// conditional that is identical to `% slots` for every index in [0, 2*slots) --
// which is every index the shipped room can form -- turning a negative index
// (a delay longer than the 256-slot early history) into an in-range one instead
// of reading out of bounds. Neither can change an accepted input.
//
// Dispatched SIMD cascade. Fft128::butterflies hands the 128-point cascade,
// QmfSynthesis::process the basis application and HybridAnalysis::process the
// 78-term low join to the runtime-dispatched kernels in src/simd. Only
// mutually independent outputs share a vector lane and every lane keeps the
// corresponding loop's own term order and its own two roundings: each butterfly
// keeps its two, the synthesis keeps the j = 0..127 order with one multiply and
// one add per term, and the hybrid join keeps its 32 independent accumulations.
// The kernels read their tables contiguously, so the constructors materialise
// them in the order the kernel wants (same doubles, same order, same table).
//
// Under JOC_SIMD=scalar and under every forced SIMD tier the output is still
// bit-identical: the joc_dump --binaural-out stream and the rendered WAV keep the
// SHA-256 they had before the change.
#define EJOC_BUILD_DLL
#include "eac3joc_core.h"
@@ -55,15 +5,11 @@
#include <array>
#include <bit>
#include <cmath>
#include <cstddef>
#include <cstdint>
#include <cstdio>
#include <cstring>
#include <new>
#include <vector>
#include "simd/simd.h"
namespace ejoc::sofa_binaural {
constexpr double kPi = 3.141592653589793238462643383279502884;
@@ -112,51 +58,36 @@ public:
const double angle = -2.0 * kPi * k / kFftSize;
twiddle_[k] = {std::cos(angle), std::sin(angle)};
}
// The bit-reversal permutation depends only on the index, so it is
// derived once here instead of by a seven-iteration bit loop inside every
// one of the 256 transforms a 512-sample block runs. Same indices, same
// destinations, same store order.
}
void forward(const Complex* input, Complex* output) const noexcept {
// bit-reversal permutation for the 7-bit index (DIT)
for (int index = 0; index < kFftSize; ++index) {
unsigned int reversed = 0;
for (int bit = 0; bit < 7; ++bit) {
reversed = (reversed << 1)
| ((static_cast<unsigned int>(index) >> bit) & 1u);
}
reverse_[index] = static_cast<int>(reversed);
output[static_cast<int>(reversed)] = input[index];
}
// The dispatched butterfly kernel reads a stage's factors as one
// contiguous run, so the per-stage stride `offset * step` is resolved once
// here. The entries are copies of the same doubles in the same order.
int cursor = 0;
int stage = 0;
for (int size = 2; size <= kFftSize; size <<= 1, ++stage) {
stage_begin_[stage] = static_cast<std::size_t>(cursor);
for (int size = 2; size <= kFftSize; size <<= 1) {
const int half = size >> 1;
const int step = kFftSize / size;
for (int offset = 0; offset < size / 2; ++offset) {
stage_twiddle_[cursor] = twiddle_[offset * step];
++cursor;
for (int base = 0; base < kFftSize; base += size) {
for (int offset = 0; offset < half; ++offset) {
const Complex w = twiddle_[offset * step];
const Complex even = output[base + offset];
const Complex odd = mul(output[base + offset + half], w);
output[base + offset] = add(even, odd);
output[base + offset + half] = {
even.re - odd.re, even.im - odd.im};
}
}
}
}
// The transform is split so a caller that can produce its input already
// bit-reversed does not have to stage it through a second array and permute it
// afterwards. `permuted(index)` is the destination the permutation used, and
// `butterflies` runs the stages in place on a permuted buffer.
int permuted(int index) const noexcept { return reverse_[index]; }
void butterflies(Complex* data) const noexcept {
joc::simd::fft_butterflies(reinterpret_cast<double*>(data), kFftSize,
reinterpret_cast<const double*>(stage_twiddle_),
stage_begin_);
}
private:
Complex twiddle_[kFftSize];
int reverse_[kFftSize];
// 1 + 2 + 4 + ... + 64 factors, one contiguous run per stage.
Complex stage_twiddle_[kFftSize - 1];
std::size_t stage_begin_[7];
};
// ---------------------------------------------------------------------------
@@ -166,16 +97,6 @@ class QmfAnalysis final {
public:
void configure(const double* coefficients) noexcept {
std::memcpy(coeff_, coefficients, sizeof(coeff_));
for (int p = 0; p < kQmf; ++p) {
const double pre_angle = -kPi * p / kFftSize;
pre_cos_[p] = std::cos(pre_angle);
pre_sin_[p] = std::sin(pre_angle);
}
for (int b = 0; b < kQmf; ++b) {
const double post_angle = -3.0 * (b + 0.5) * kPi / kFftSize;
post_cos_[b] = std::cos(post_angle);
post_sin_[b] = std::sin(post_angle);
}
}
void reset() noexcept {
@@ -199,17 +120,11 @@ public:
}
for (int slot = 0; slot < slots; ++slot) {
for (int channel = 0; channel < kChannels; ++channel) {
Complex even_fft[kFftSize];
Complex odd_fft[kFftSize];
// The polyphase sums are written straight to the bit-reversed
// positions the DIT permutation would have put them in, so the two
// staging arrays and the permutation pass over them are gone. Each
// destination is still written exactly once, with the same value, so
// the permuted buffers hold the same bits as before, and only the zero-padded
// upper half still needs clearing.
for (int p = kQmf; p < kFftSize; ++p) {
even_fft[fft_.permuted(p)] = {0.0, 0.0};
odd_fft[fft_.permuted(p)] = {0.0, 0.0};
Complex even[kFftSize];
Complex odd[kFftSize];
for (int p = 0; p < kFftSize; ++p) {
even[p] = {0.0, 0.0};
odd[p] = {0.0, 0.0};
}
// 64 polyphase positions; the 128-point FFT zero-pads the rest
for (int p = 0; p < kQmf; ++p) {
@@ -223,17 +138,20 @@ public:
odd_re += joined_[9 + slot - (2 * lag + 1)][channel][p]
* coeff_[p][2 * lag + 1];
}
even_fft[fft_.permuted(p)] = {even_re * pre_cos_[p],
even_re * pre_sin_[p]};
odd_fft[fft_.permuted(p)] = {odd_re * pre_cos_[p],
odd_re * pre_sin_[p]};
const double pre_angle = -kPi * p / kFftSize;
even[p] = {even_re * std::cos(pre_angle),
even_re * std::sin(pre_angle)};
odd[p] = {odd_re * std::cos(pre_angle), odd_re * std::sin(pre_angle)};
}
fft_.butterflies(even_fft);
fft_.butterflies(odd_fft);
Complex even_fft[kFftSize];
Complex odd_fft[kFftSize];
fft_.forward(even, even_fft);
fft_.forward(odd, odd_fft);
Complex* band = output
+ (static_cast<size_t>(slot) * kChannels + channel) * kQmf;
for (int b = 0; b < kQmf; ++b) {
const Complex post = {post_cos_[b], post_sin_[b]};
const double post_angle = -3.0 * (b + 0.5) * kPi / kFftSize;
const Complex post = {std::cos(post_angle), std::sin(post_angle)};
const Complex even_post = {0.0, (b % 2 == 0) ? 1.0 : -1.0};
// python: (odd_fft + even_fft * even_post) * post
band[b] = mul(post, add(odd_fft[b],
@@ -248,12 +166,6 @@ public:
private:
double coeff_[kQmf][10];
// The two modulation tables depend only on the loop index, so they are
// built once with the identical expressions used in process().
double pre_cos_[kQmf];
double pre_sin_[kQmf];
double post_cos_[kQmf];
double post_sin_[kQmf];
double history_[9][kChannels][64];
double joined_[9 + kMaxBlockSamples / kHop][kChannels][64];
Fft128 fft_;
@@ -266,21 +178,6 @@ class HybridAnalysis final {
public:
void configure(const double* low_kernel) noexcept {
std::memcpy(low_, low_kernel, sizeof(low_));
// The dispatched join walks its terms in the order this class
// accumulates them -- parent, then component, then lag -- and adds the 32
// weights of one term at once, so the table is regrouped once here. Same
// weights, same order.
int cursor = 0;
for (int parent = 0; parent < 3; ++parent) {
for (int comp = 0; comp < 2; ++comp) {
for (int lag = 0; lag < 13; ++lag) {
for (int hb = 0; hb < 16; ++hb) {
low_by_term_[cursor++] = low_[parent][comp][lag][hb][0];
low_by_term_[cursor++] = low_[parent][comp][lag][hb][1];
}
}
}
}
}
void reset() noexcept {
@@ -310,46 +207,24 @@ public:
}
}
}
// The low bands are a 78-term join into 32 outputs, and the 32 outputs
// of a row are independent accumulations over the same terms -- that is what
// the dispatched kernel puts in its lanes, keeping this loop's term order
// (parent, component, lag) and its two roundings per term. Rows are staged
// in blocks so the gathered values stay in the first-level cache.
const std::size_t rows = static_cast<std::size_t>(slots) * kChannels;
const std::size_t block = joc::simd::kHybridJoinBlock;
const std::size_t terms = joc::simd::kHybridTerms;
for (std::size_t first = 0u; first < rows; first += block) {
const std::size_t count = std::min(block, rows - first);
for (std::size_t index = 0u; index < count; ++index) {
const std::size_t row = first + index;
const int slot = static_cast<int>(row / kChannels);
const int channel = static_cast<int>(row % kChannels);
double* staged = low_values_ + index * terms;
for (int parent = 0; parent < 3; ++parent) {
for (int comp = 0; comp < 2; ++comp) {
for (int lag = 0; lag < 13; ++lag) {
staged[(static_cast<std::size_t>(parent) * 2u +
static_cast<std::size_t>(comp)) * 13u +
static_cast<std::size_t>(lag)] =
low_joined_[12 + slot - lag][channel][parent][comp];
}
}
}
}
joc::simd::hybrid_low_join(low_values_, low_by_term_, low_out_, count);
for (std::size_t index = 0u; index < count; ++index) {
Complex* out = output + (first + index) * kHybrid;
const double* values = low_out_ + index * joc::simd::kHybridOutputs;
for (int hb = 0; hb < 16; ++hb) {
out[hb] = {values[static_cast<std::size_t>(hb) * 2u],
values[static_cast<std::size_t>(hb) * 2u + 1u]};
}
}
}
for (int slot = 0; slot < slots; ++slot) {
for (int channel = 0; channel < kChannels; ++channel) {
Complex* out = output
+ (static_cast<size_t>(slot) * kChannels + channel) * kHybrid;
for (int hb = 0; hb < 16; ++hb) {
Complex value = {0.0, 0.0};
for (int parent = 0; parent < 3; ++parent) {
for (int comp = 0; comp < 2; ++comp) {
for (int lag = 0; lag < 13; ++lag) {
const double source =
low_joined_[12 + slot - lag][channel][parent][comp];
value.re += source * low_[parent][comp][lag][hb][0];
value.im += source * low_[parent][comp][lag][hb][1];
}
}
}
out[hb] = value;
}
for (int b = 0; b < 61; ++b) {
out[16 + b] = high_joined_[slot][channel][b];
}
@@ -366,9 +241,6 @@ public:
private:
double low_[3][2][13][16][2];
double low_by_term_[78 * 32]; // S3: the same weights in this class's term order
double low_values_[joc::simd::kHybridJoinBlock * 78]; // scratch
double low_out_[joc::simd::kHybridJoinBlock * 32]; // scratch
double history_[12][kChannels][3][2];
double low_joined_[12 + kMaxBlockSamples / kHop][kChannels][3][2];
Complex high_history_[6][kChannels][61];
@@ -437,18 +309,6 @@ public:
void configure(const double* basis, const double* taps) noexcept {
std::memcpy(basis_, basis, sizeof(basis_));
std::memcpy(taps_, taps, sizeof(taps_));
// The dispatched basis kernel reads the four ranks of one (band, tap)
// as one vector, so the shipped [band][rank][tap] table is reordered once
// here. Same doubles, different order.
int cursor = 0;
for (int b = 0; b < kQmf; ++b) {
for (int j = 0; j < kFftSize; ++j) {
for (int r = 0; r < kRank; ++r) {
basis_by_tap_[cursor] = basis_[b][r][j];
++cursor;
}
}
}
}
void reset() noexcept {
@@ -461,27 +321,25 @@ public:
std::memcpy(joined_[lag], history_[lag], sizeof(joined_[0]));
}
for (int slot = 0; slot < slots; ++slot) {
// Both channels consume the same basis_[b][r][*] row. The
// dispatched kernel does the same thing a vector at a time: the
// four ranks of a band are four independent dot products over one
// channel's 128 values, so they fill the lanes while each lane keeps
// the original j = 0..127 order and its own two roundings.
for (int channel = 0; channel < 2; ++channel) {
const Complex* values = qmf
+ (static_cast<size_t>(slot) * 2 + channel) * kQmf;
double flat[kFftSize];
for (int b = 0; b < kQmf; ++b) {
flat_[slot][channel][2 * b] = values[b].re;
flat_[slot][channel][2 * b + 1] = values[b].im;
flat[2 * b] = values[b].re;
flat[2 * b + 1] = values[b].im;
}
for (int b = 0; b < kQmf; ++b) {
for (int r = 0; r < kRank; ++r) {
double value = 0.0;
for (int j = 0; j < kFftSize; ++j) {
value += flat[j] * basis_[b][r][j];
}
joined_[9 + slot][channel][b][r] = value;
}
}
}
}
// Every slot is handed over at once rather than one at a time: a band's
// accumulation is 128 dependent adds, and the kernel hides that latency by
// running several rows side by side, so it needs more than the two rows of
// a single slot to work with.
joc::simd::qmf_synthesis_basis(&flat_[0][0][0], basis_by_tap_,
&joined_[9][0][0][0],
static_cast<std::size_t>(slots) * 2u);
for (int slot = 0; slot < slots; ++slot) {
for (int channel = 0; channel < 2; ++channel) {
for (int b = 0; b < kQmf; ++b) {
@@ -504,9 +362,7 @@ public:
private:
double basis_[kQmf][kRank][kFftSize];
double basis_by_tap_[kQmf * kFftSize * kRank]; // S2: the same weights, transposed
double taps_[kQmf][10][kRank];
double flat_[kMaxBlockSamples / kHop][2][kFftSize]; // S2: staging for the kernel
double history_[9][2][kQmf][kRank];
double joined_[9 + kMaxBlockSamples / kHop][2][kQmf][kRank];
};
@@ -519,27 +375,15 @@ public:
static void evaluate(const double* direction, double* basis) noexcept {
const double azimuth = std::atan2(direction[1], direction[0]);
const double x = std::max(-1.0, std::min(1.0, direction[2]));
// Normalization depends only on (degree, |order|) and the Legendre
// seed pmm_value(m, x) only on (m, x): 6 seeds, not 36 recomputations.
double pmm[kOrder + 1];
for (int m = 0; m <= kOrder; ++m) {
pmm[m] = pmm_value(m, x);
}
double norm[kOrder + 1][kOrder + 1];
for (int degree = 0; degree <= kOrder; ++degree) {
for (int absolute = 0; absolute <= degree; ++absolute) {
norm[degree][absolute] = std::sqrt(
(2.0 * degree + 1.0) / (4.0 * kPi)
* factorial_ratio(degree, absolute));
}
}
int index = 0;
for (int degree = 0; degree <= kOrder; ++degree) {
for (int order = -degree; order <= degree; ++order) {
const int absolute = std::abs(order);
const double normalization = norm[degree][absolute];
const double normalization = std::sqrt(
(2.0 * degree + 1.0) / (4.0 * kPi)
* factorial_ratio(degree, absolute));
const double legendre = associated_legendre(
degree, absolute, x, pmm[absolute]);
degree, absolute, x, pmm_value(absolute, x));
if (order > 0) {
basis[index] = std::sqrt(2.0) * normalization * legendre
* std::cos(order * azimuth);
@@ -623,20 +467,6 @@ struct SourceState {
double late_target;
int64_t late_fade_position;
int64_t late_fade_total;
// The seven paths are a pure function of (position, profile, effective
// gain, special_lfe) plus the configuration that only configure_* can change.
// The wrapper calls set_source for every source on every 512-sample block and
// the timeline holds a position constant between OAMD updates, so the same
// paths were rebuilt from scratch over and over. The memo stores the last
// result and the exact bit pattern of the arguments that produced it; a hit
// reuses those bits instead of recomputing them. The fade state machine and
// the late-send envelope below run identically either way.
double memo_position[3] = {0.0, 0.0, 0.0};
double memo_effective = 0.0;
int memo_profile = 0;
int memo_special_lfe = 0;
int memo_valid = 0;
std::vector<Path> memo_paths;
};
constexpr double kDistanceM[3] = {1.00000465, 2.19327927, 6.40177584};
@@ -710,18 +540,6 @@ public:
|| !(damping >= 0.0 && damping < 1.0)) {
return fail("invalid sofa binaural room configuration");
}
// A zero delay-line length makes the per-sample `% delay` an integer
// divide by zero. Rejecting it here cannot change any accepted input.
for (int line = 0; line < 4; ++line) {
if (fdn_delays[line] == 0u) {
return fail("invalid sofa binaural room configuration");
}
}
for (int line = 0; line < 2; ++line) {
if (allpass_delays[line] == 0u) {
return fail("invalid sofa binaural room configuration");
}
}
std::memcpy(dims_, dims, sizeof(dims_));
std::memcpy(listener_, listener, sizeof(listener_));
std::memcpy(wall_gain_, wall_gains, sizeof(wall_gain_));
@@ -776,43 +594,23 @@ public:
state.special_lfe = special_lfe != 0;
const double effective = enabled ? gain : 0.0;
// Reuse the memoised paths when the arguments are bit-identical to the
// ones that produced them. Everything make_path reads -- the direction and
// distance derived from `position`, the profile, the effective gain, the LFE
// flag, and the field/room tables -- is covered by this key or fixed by
// configure_*, and the side effect ordinary_paths has on
// maximum_early_delay_ is a maximum of the same values, so reusing the
// stored bits is the same as recomputing them.
const int memo_special_lfe = state.special_lfe ? 1 : 0;
const bool memo_hit = state.memo_valid != 0 && state.memo_profile == state.profile
&& state.memo_special_lfe == memo_special_lfe
&& std::memcmp(state.memo_position, position, sizeof(state.memo_position)) == 0
&& std::memcmp(&state.memo_effective, &effective, sizeof(effective)) == 0;
std::vector<Path>& paths = state.memo_paths;
std::vector<Path> paths;
double late_send = 0.0;
double direction[3];
double radius;
normalize_adm(position, direction, radius);
const double distance =
std::max(kMinimumDistance, radius * kDistanceM[state.profile]);
if (!memo_hit) {
paths.clear();
if (state.special_lfe) {
paths.push_back(lfe_path(effective));
} else {
paths = ordinary_paths(direction, distance, state.profile, effective);
if (state.special_lfe) {
paths.push_back(lfe_path(effective));
} else {
paths = ordinary_paths(direction, distance, state.profile, effective);
if (enabled && enable_late_room_) {
const double base = kLateSend[state.profile];
const double radial = std::min(
std::max(std::sqrt(std::max(radius, 0.0)), 0.25), 1.5);
late_send = effective * kRoomCalibration * base * radial;
}
std::memcpy(state.memo_position, position, sizeof(state.memo_position));
std::memcpy(&state.memo_effective, &effective, sizeof(effective));
state.memo_profile = state.profile;
state.memo_special_lfe = memo_special_lfe;
state.memo_valid = 1;
}
if (!state.special_lfe && enabled && enable_late_room_) {
const double base = kLateSend[state.profile];
const double radial = std::min(
std::max(std::sqrt(std::max(radius, 0.0)), 0.25), 1.5);
late_send = effective * kRoomCalibration * base * radial;
}
set_late_target(source, late_send, fade != 0);
@@ -910,28 +708,21 @@ private:
RealSh::evaluate(listener_direction, basis);
Complex aligned[2][kHybrid];
double delay[2];
double delay_value[2] = {0.0, 0.0};
for (int ear = 0; ear < 2; ++ear) {
double delay_value = 0.0;
for (int band = 0; band < kHybrid; ++band) {
aligned[ear][band] = {0.0, 0.0};
}
}
// The 36 spherical-harmonic terms are independent contributions to the
// same 2 x 77 bands, so the band axis is what the dispatched accumulate
// spreads across its lanes. Each band still adds `field * term` once per
// term, in term order, with the same two roundings; only the delay sum --
// which is a reduction -- stays scalar and keeps its own order.
for (int term = 0; term < kTerms; ++term) {
for (int ear = 0; ear < 2; ++ear) {
delay_value[ear] += basis[term] * delay_coeff_[term][ear];
for (int term = 0; term < kTerms; ++term) {
delay_value += basis[term] * delay_coeff_[term][ear];
for (int band = 0; band < kHybrid; ++band) {
aligned[ear][band].re +=
field_coeff_[term][ear][band].re * basis[term];
aligned[ear][band].im +=
field_coeff_[term][ear][band].im * basis[term];
}
}
joc::simd::complex_axpy(
reinterpret_cast<double*>(&aligned[0][0]),
reinterpret_cast<const double*>(&field_coeff_[term][0][0]), basis[term],
2u * kHybrid);
}
for (int ear = 0; ear < 2; ++ear) {
delay[ear] = std::min(std::max(delay_value[ear], delay_bounds_[ear][0]),
delay[ear] = std::min(std::max(delay_value, delay_bounds_[ear][0]),
delay_bounds_[ear][1]);
}
Path path;
@@ -1040,14 +831,8 @@ private:
void process_internal(const double* input, int slots, double output_gain) noexcept {
const uint32_t sample_count = static_cast<uint32_t>(slots) * kHop;
// These four planes are written in full before they are read, so they
// are reused scratch buffers. `assign` keeps the capacity, so after the
// first block each one is a fill with no allocation, and the fills that are
// load-bearing (the mono and late accumulators, and the direct/early output
// that is only added into) are preserved exactly.
// 1) late send envelopes -> mono
std::vector<double>& mono = mono_;
mono.assign(sample_count, 0.0);
std::vector<double> mono(sample_count, 0.0);
for (int source = 0; source < kChannels; ++source) {
SourceState& state = sources_[source];
const int64_t total = state.late_fade_total;
@@ -1089,15 +874,12 @@ private:
}
// 2) FDN + 961-sample stereo delay
std::vector<double>& late_pcm = late_pcm_;
late_pcm.assign(static_cast<size_t>(sample_count) * 2, 0.0);
std::vector<double> late_pcm(static_cast<size_t>(sample_count) * 2, 0.0);
if (enable_late_room_) {
diffused_.assign(mono.begin(), mono.end());
std::vector<double> diffused = mono;
for (int line = 0; line < 2; ++line) {
allpass(line, diffused_, &allpass_work_);
diffused_.swap(allpass_work_);
diffused = allpass(line, diffused);
}
const std::vector<double>& diffused = diffused_;
for (uint32_t s = 0; s < sample_count; ++s) {
const double value = diffused[s];
double delayed[4];
@@ -1156,8 +938,8 @@ private:
analysis_.process(input ? input : zeros_.data(), slots, qmf_work_.data());
hybrid_analysis_.process(qmf_work_.data(), slots, hybrid_work_.data());
// 4) per-object direct/early with crossfade
std::vector<Complex>& direct_and_early = direct_early_;
direct_and_early.assign(static_cast<size_t>(slots) * 2 * kHybrid, Complex{0.0, 0.0});
std::vector<Complex> direct_and_early(
static_cast<size_t>(slots) * 2 * kHybrid, Complex{0.0, 0.0});
for (int slot = 0; slot < slots; ++slot) {
const Complex* hybrid_slot =
hybrid_work_.data() + static_cast<size_t>(slot) * kChannels * kHybrid;
@@ -1214,32 +996,30 @@ private:
Complex* out) const noexcept {
for (const Path& path : paths) {
const int slots = static_cast<int>(history_slots_);
// The history index: position_ is in [0, slots), so for any delay_slots
// <= slots the raw index lands in [0, 2*slots) and this conditional is
// exactly `% slots`. A longer delay can make the raw index negative,
// where C++ `%` would produce an out-of-bounds negative index; the
// `raw < 0` arm wraps it into range instead.
const int raw0 = static_cast<int>(position_) + slots - path.delay_slots[0];
const int raw1 = static_cast<int>(position_) + slots - path.delay_slots[1];
const int index0 = raw0 >= slots ? raw0 - slots : (raw0 < 0 ? raw0 + slots : raw0);
const int index1 = raw1 >= slots ? raw1 - slots : (raw1 < 0 ? raw1 + slots : raw1);
const int index0 = (static_cast<int>(position_) + slots
- path.delay_slots[0]) % slots;
const int index1 = (static_cast<int>(position_) + slots
- path.delay_slots[1]) % slots;
const Complex* history = history_[source].data();
// The 77 bands of an ear are independent accumulations into
// independent outputs, so they are what the dispatched kernel spreads
// across its lanes; each lane keeps this loop's `h * t * scale` with its
// own two roundings, and the ears read their own history rows.
joc::simd::render_hybrid_path(
reinterpret_cast<double*>(out),
reinterpret_cast<const double*>(path.transfer),
reinterpret_cast<const double*>(history + index0 * kHybrid),
reinterpret_cast<const double*>(history + index1 * kHybrid), scale);
for (int band = 0; band < kHybrid; ++band) {
const Complex source0 = history[index0 * kHybrid + band];
const Complex source1 = history[index1 * kHybrid + band];
out[band].re += source0.re * path.transfer[0][band].re * scale
- source0.im * path.transfer[0][band].im * scale;
out[band].im += source0.re * path.transfer[0][band].im * scale
+ source0.im * path.transfer[0][band].re * scale;
out[kHybrid + band].re +=
source1.re * path.transfer[1][band].re * scale
- source1.im * path.transfer[1][band].im * scale;
out[kHybrid + band].im +=
source1.re * path.transfer[1][band].im * scale
+ source1.im * path.transfer[1][band].re * scale;
}
}
}
void allpass(int line, const std::vector<double>& source,
std::vector<double>* output) noexcept {
// Every element is assigned below, so only the size has to be established.
output->resize(source.size());
std::vector<double> allpass(int line, const std::vector<double>& source) noexcept {
std::vector<double> output(source.size(), 0.0);
const double gain = allpass_gains_[line];
const uint32_t delay = allpass_delays_[line];
for (size_t i = 0; i < source.size(); ++i) {
@@ -1248,8 +1028,9 @@ private:
allpass_buffers_[line][allpass_positions_[line]] =
source[i] + gain * result;
allpass_positions_[line] = (allpass_positions_[line] + 1) % delay;
(*output)[i] = result;
output[i] = result;
}
return output;
}
int fail(const char* message) noexcept {
@@ -1268,13 +1049,6 @@ private:
std::vector<Complex>(
static_cast<size_t>(history_slots_) * kHybrid,
Complex{0.0, 0.0}));
// Re-assigning the source states is also what invalidates the
// set_source path memo, because the memo fields are SourceState members
// and a fresh SourceState starts with memo_valid == 0. Every configure_*
// reaches reset_state() through reset(), so a reconfigured renderer can
// never serve a path derived from the previous configuration. If this
// function is ever changed to reset the fields in place, clear the memo
// explicitly here instead.
sources_.assign(kChannels, SourceState{});
for (int source = 0; source < kChannels; ++source) {
sources_[source].position[0] = 0.0;
@@ -1360,12 +1134,6 @@ private:
std::array<double, kMaxBlockSamples / kHop * 2 * 64> pcm_work_{};
std::array<double, kMaxBlockSamples * 2> output_{};
std::array<double, kMaxBlockSamples * kChannels> zeros_{};
// Reused per-block scratch (sized once by the first block's `assign`).
std::vector<double> mono_;
std::vector<double> late_pcm_;
std::vector<double> diffused_;
std::vector<double> allpass_work_;
std::vector<Complex> direct_early_;
char error_[256];
bool enable_early_reflections_ = true;
+3
View File
@@ -0,0 +1,3 @@
numpy>=1.24
scipy>=1.10
h5py>=3.8
-333
View File
@@ -1,333 +0,0 @@
#include "adm/adm_metadata.h"
#include <algorithm>
#include <cmath>
#include <cstring>
#include "foundation/py_num.h"
namespace joc::adm {
namespace {
constexpr const char* kBedNames[10] = {
"RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter", "RoomCentricLFE",
"RoomCentricLeftSideSurround", "RoomCentricRightSideSurround",
"RoomCentricLeftRearSurround", "RoomCentricRightRearSurround",
"RoomCentricLeftTopSurround", "RoomCentricRightTopSurround"};
constexpr const char* kBedLabels[10] = {"RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss",
"RC_Rss", "RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"};
constexpr double kBedPos[10][3] = {{-1.0, 1.0, 0.0}, {1.0, 1.0, 0.0}, {0.0, 1.0, 0.0},
{-1.0, 1.0, -1.0}, {-1.0, 0.0, 0.0}, {1.0, 0.0, 0.0},
{-1.0, -1.0, 0.0}, {1.0, -1.0, 0.0}, {-1.0, 0.0, 1.0},
{1.0, 0.0, 1.0}};
std::string hex4(std::uint32_t value) {
char buffer[16];
std::snprintf(buffer, sizeof(buffer), "%04x", value);
return std::string(buffer);
}
std::string hex8(std::uint32_t value) {
char buffer[16];
std::snprintf(buffer, sizeof(buffer), "%08x", value);
return std::string(buffer);
}
void put_u16(std::string* out, std::uint16_t value) {
char buffer[2];
std::memcpy(buffer, &value, 2);
out->append(buffer, 2);
}
void put_u32(std::string* out, std::uint32_t value) {
char buffer[4];
std::memcpy(buffer, &value, 4);
out->append(buffer, 4);
}
std::uint8_t checksum(const std::string& segment) {
int sum = static_cast<int>(segment.size());
for (const char raw : segment) {
sum += static_cast<unsigned char>(raw);
}
return static_cast<std::uint8_t>((~sum + 1) & 0xFF);
}
} // namespace
std::string ts(double seconds) {
long long whole = static_cast<long long>(seconds);
long long fraction = pynum::py_round((seconds - static_cast<double>(whole)) * 100000.0);
if (fraction >= 100000) {
whole += 1;
fraction = 0;
}
char buffer[32];
std::snprintf(buffer, sizeof(buffer), "%02lld:%02lld:%02lld.%05lld", whole / 3600,
(whole % 3600) / 60, whole % 60, fraction);
return std::string(buffer);
}
bool binaural_mode_from_name(const char* name, BinauralMode* out) {
if (name == nullptr || out == nullptr) {
return false;
}
if (std::strcmp(name, "off") == 0) { *out = BinauralMode::Off; return true; }
if (std::strcmp(name, "near") == 0) { *out = BinauralMode::Near; return true; }
if (std::strcmp(name, "far") == 0) { *out = BinauralMode::Far; return true; }
if (std::strcmp(name, "mid") == 0) { *out = BinauralMode::Mid; return true; }
if (std::strcmp(name, "unspecified") == 0) { *out = BinauralMode::Unspecified; return true; }
return false;
}
std::string build_chna() {
std::string out;
put_u16(&out, static_cast<std::uint16_t>(kTrackCount));
put_u16(&out, static_cast<std::uint16_t>(kTrackCount));
for (std::uint32_t i = 0; i < 10; ++i) {
put_u16(&out, static_cast<std::uint16_t>(i + 1));
out += "ATU_" + hex8(i + 1);
out += "AT_0001" + hex4(0x1001 + i) + "_01";
out += "AP_00011001";
out.push_back('\0');
}
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
put_u16(&out, static_cast<std::uint16_t>(i + 11));
out += "ATU_" + hex8(i + 11);
out += "AT_0003" + hex4(0x1001 + i) + "_01";
out += "AP_0003" + hex4(0x1001 + i);
out.push_back('\0');
}
return out;
}
Status build_dbmd(std::uint32_t object_count, BinauralMode mode, std::string* out) {
if (out == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput, "null output");
}
const std::uint32_t mode_value = static_cast<std::uint32_t>(mode);
if (mode_value > 4u) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput,
"invalid JOC binaural render mode");
}
out->clear();
put_u32(out, 0x01000006u);
std::string segment7(96, '\0');
segment7[1] = static_cast<char>(0x47);
segment7[5] = static_cast<char>(0x60);
segment7[8] = static_cast<char>(0x24);
segment7[9] = static_cast<char>(0x24);
out->push_back(7);
put_u16(out, 96);
out->append(segment7);
out->push_back(static_cast<char>(checksum(segment7)));
std::string segment9(248, '\0');
const std::string creator = "Created with EAC3JOC";
const std::string renderer = "EAC3JOC Python Renderer";
std::memcpy(&segment9[0], creator.data(), creator.size());
std::memcpy(&segment9[32], renderer.data(), renderer.size());
segment9[96] = 2;
segment9[97] = 1;
segment9[98] = 0;
segment9[103] = 0x03;
segment9[106] = 0x01;
segment9[111] = 0x22;
segment9[112] = static_cast<char>(0xFF);
out->push_back(9);
put_u16(out, 248);
out->append(segment9);
out->push_back(static_cast<char>(checksum(segment9)));
// The reference allocates the body zeroed and then fills only the trailing
// `object_count` bytes with 0x84, so the template region stays zero.
const std::size_t object_body = 5u + 262u + object_count;
std::string segment10(object_body, '\0');
const std::uint32_t sync = 0xF8726FBDu;
std::memcpy(&segment10[0], &sync, 4);
segment10[4] = static_cast<char>(object_count);
for (std::size_t i = 5u + 262u; i < segment10.size(); ++i) {
segment10[i] = static_cast<char>(0x84);
}
const std::size_t object_modes = 4u + 2u + 1u + 9u * 15u + object_count;
for (std::uint32_t i = 10; i < std::min<std::uint32_t>(object_count, 10u + kObjectCount); ++i) {
const std::size_t index = object_modes + i;
if (index >= segment10.size()) {
return Status::fail(JOC_ERR_INTERNAL, stage::kOutput, "dbmd object slot out of range");
}
segment10[index] = static_cast<char>((static_cast<unsigned char>(segment10[index]) & 0xF8u) |
mode_value);
}
out->push_back(10);
put_u16(out, static_cast<std::uint16_t>(segment10.size()));
out->append(segment10);
out->push_back(static_cast<char>(checksum(segment10)));
out->append("\0\0", 2);
return Status::success();
}
std::string build_axml(const std::vector<Track>& tracks, double duration_sec, std::uint32_t rate) {
const double scale = static_cast<double>(rate);
std::string out;
out.reserve(64u * 1024u);
const std::string duration_ts = ts(duration_sec);
out += "<?xml version=\"1.0\" encoding=\"utf-8\"?>";
out += "<ebuCoreMain xsi:schemaLocation=\"urn:ebu:metadata-schema:ebuCore_2016 ebucore.xsd\" "
"lang=\"en\" xmlns:xsi=\"http://www.w3.org/2001/XMLSchema-instance\" "
"xmlns=\"urn:ebu:metadata-schema:ebuCore_2016\">";
out += "<coreMetadata><format><audioFormatExtended>";
out += "<audioProgramme audioProgrammeID=\"APR_1001\" audioProgrammeName=\"EAC3JOC_Export\" "
"start=\"" +
ts(0.0) + "\" end=\"" + duration_ts + "\">";
out += "<audioContentIDRef>ACO_1001</audioContentIDRef>";
out += "<audioContentIDRef>ACO_1002</audioContentIDRef>";
out += "</audioProgramme>";
out += "<audioContent audioContentID=\"ACO_1001\" "
"audioContentName=\"EAC3JOC_Master_Content\">";
out += "<audioObjectIDRef>AO_1001</audioObjectIDRef>";
out += "<dialogue mixedContentKind=\"0\">2</dialogue>";
out += "</audioContent>";
out += "<audioContent audioContentID=\"ACO_1002\" audioContentName=\"Objects\">";
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
out += "<audioObjectIDRef>AO_" + hex4(0x100b + i) + "</audioObjectIDRef>";
}
out += "<dialogue mixedContentKind=\"0\">2</dialogue>";
out += "</audioContent>";
out += "<audioObject audioObjectID=\"AO_1001\" audioObjectName=\"Bed\" start=\"" + ts(0.0) +
"\" duration=\"" + duration_ts + "\">";
out += "<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>";
for (std::uint32_t i = 0; i < 10; ++i) {
out += "<audioTrackUIDRef>ATU_" + hex8(i + 1) + "</audioTrackUIDRef>";
}
out += "</audioObject>";
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
out += "<audioObject audioObjectID=\"AO_" + hex4(0x100b + i) +
"\" audioObjectName=\"Audio Object " + std::to_string(i + 1) + "\" start=\"" +
ts(0.0) + "\" duration=\"" + duration_ts + "\">";
out += "<audioPackFormatIDRef>AP_0003" + hex4(0x1001 + i) + "</audioPackFormatIDRef>";
out += "<audioTrackUIDRef>ATU_" + hex8(11 + i) + "</audioTrackUIDRef>";
out += "</audioObject>";
}
out += "<audioPackFormat audioPackFormatID=\"AP_00011001\" "
"audioPackFormatName=\"EAC3JOCBedPack\" typeDefinition=\"DirectSpeakers\" "
"typeLabel=\"0001\">";
for (std::uint32_t i = 0; i < 10; ++i) {
out += "<audioChannelFormatIDRef>AC_0001" + hex4(0x1001 + i) +
"</audioChannelFormatIDRef>";
}
out += "</audioPackFormat>";
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
out += "<audioPackFormat audioPackFormatID=\"AP_0003" + hex4(0x1001 + i) +
"\" audioPackFormatName=\"JOC_Object_" + std::to_string(i + 1) +
"\" typeDefinition=\"Objects\" typeLabel=\"0003\">";
out += "<audioChannelFormatIDRef>AC_0003" + hex4(0x1001 + i) +
"</audioChannelFormatIDRef>";
out += "</audioPackFormat>";
}
for (std::uint32_t i = 0; i < 10; ++i) {
out += "<audioChannelFormat audioChannelFormatID=\"AC_0001" + hex4(0x1001 + i) +
"\" audioChannelFormatName=\"" + kBedNames[i] +
"\" typeDefinition=\"DirectSpeakers\" typeLabel=\"0001\">";
out += "<audioBlockFormat audioBlockFormatID=\"AB_0001" + hex4(0x1001 + i) +
"_00000001\">";
out += "<cartesian>1</cartesian>";
out += "<position coordinate=\"X\">" + pynum::format_fixed(kBedPos[i][0], 10) +
"</position>";
out += "<position coordinate=\"Y\">" + pynum::format_fixed(kBedPos[i][1], 10) +
"</position>";
if (kBedPos[i][2] != 0.0) {
out += "<position coordinate=\"Z\">" + pynum::format_fixed(kBedPos[i][2], 10) +
"</position>";
}
out += std::string("<speakerLabel>") + kBedLabels[i] + "</speakerLabel>";
out += "</audioBlockFormat>";
out += "</audioChannelFormat>";
}
for (std::size_t i = 0; i < tracks.size(); ++i) {
const Track& track = tracks[i];
out += "<audioChannelFormat audioChannelFormatID=\"AC_0003" +
hex4(0x1001 + static_cast<std::uint32_t>(i)) + "\" audioChannelFormatName=\"" +
track.name + "\" typeDefinition=\"Objects\" typeLabel=\"0003\">";
for (std::size_t k = 0; k < track.blocks.size(); ++k) {
const Keyframe& block = track.blocks[k];
out += "<audioBlockFormat audioBlockFormatID=\"AB_0003" +
hex4(0x1001 + static_cast<std::uint32_t>(i)) + "_" +
hex8(static_cast<std::uint32_t>(k + 1)) + "\" rtime=\"" +
ts(static_cast<double>(block.rtime_samples) / scale) + "\" duration=\"" +
ts(static_cast<double>(block.duration_samples) / scale) + "\">";
out += "<cartesian>1</cartesian>";
out += "<position coordinate=\"X\">" + pynum::format_fixed(block.x, 10) + "</position>";
out += "<position coordinate=\"Y\">" + pynum::format_fixed(block.y, 10) + "</position>";
if (block.z != 0.0) {
out += "<position coordinate=\"Z\">" + pynum::format_fixed(block.z, 10) +
"</position>";
}
out += "<jumpPosition interpolationLength=\"" +
pynum::format_fixed(static_cast<double>(block.interpolation_samples) / scale,
5) +
"\">1</jumpPosition>";
out += "</audioBlockFormat>";
}
out += "</audioChannelFormat>";
}
for (std::uint32_t i = 0; i < 10; ++i) {
out += "<audioTrackUID UID=\"ATU_" + hex8(i + 1) +
"\" bitDepth=\"24\" sampleRate=\"48000\">";
out += "<audioTrackFormatIDRef>AT_0001" + hex4(0x1001 + i) + "_01</audioTrackFormatIDRef>";
out += "<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>";
out += "</audioTrackUID>";
}
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
out += "<audioTrackUID UID=\"ATU_" + hex8(11 + i) +
"\" bitDepth=\"24\" sampleRate=\"48000\">";
out += "<audioTrackFormatIDRef>AT_0003" + hex4(0x1001 + i) + "_01</audioTrackFormatIDRef>";
out += "<audioPackFormatIDRef>AP_0003" + hex4(0x1001 + i) + "</audioPackFormatIDRef>";
out += "</audioTrackUID>";
}
for (std::uint32_t i = 0; i < 10; ++i) {
out += "<audioTrackFormat audioTrackFormatID=\"AT_0001" + hex4(0x1001 + i) +
"_01\" audioTrackFormatName=\"PCM_" + kBedNames[i] +
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
out += "<audioStreamFormatIDRef>AS_0001" + hex4(0x1001 + i) +
"</audioStreamFormatIDRef>";
out += "</audioTrackFormat>";
}
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
out += "<audioTrackFormat audioTrackFormatID=\"AT_0003" + hex4(0x1001 + i) +
"_01\" audioTrackFormatName=\"PCM_JOC_Object_" + std::to_string(i + 1) +
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
out += "<audioStreamFormatIDRef>AS_0003" + hex4(0x1001 + i) +
"</audioStreamFormatIDRef>";
out += "</audioTrackFormat>";
}
for (std::uint32_t i = 0; i < 10; ++i) {
out += "<audioStreamFormat audioStreamFormatID=\"AS_0001" + hex4(0x1001 + i) +
"\" audioStreamFormatName=\"PCM_" + kBedNames[i] +
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
out += "<audioChannelFormatIDRef>AC_0001" + hex4(0x1001 + i) +
"</audioChannelFormatIDRef>";
out += "<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>";
out += "<audioTrackFormatIDRef>AT_0001" + hex4(0x1001 + i) +
"_01</audioTrackFormatIDRef>";
out += "</audioStreamFormat>";
}
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
out += "<audioStreamFormat audioStreamFormatID=\"AS_0003" + hex4(0x1001 + i) +
"\" audioStreamFormatName=\"PCM_JOC_Object_" + std::to_string(i + 1) +
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
out += "<audioChannelFormatIDRef>AC_0003" + hex4(0x1001 + i) +
"</audioChannelFormatIDRef>";
out += "<audioPackFormatIDRef>AP_0003" + hex4(0x1001 + i) + "</audioPackFormatIDRef>";
out += "<audioTrackFormatIDRef>AT_0003" + hex4(0x1001 + i) +
"_01</audioTrackFormatIDRef>";
out += "</audioStreamFormat>";
}
out += "</audioFormatExtended></format></coreMetadata>";
out += "</ebuCoreMain>";
return out;
}
} // namespace joc::adm
-43
View File
@@ -1,43 +0,0 @@
// Port of src/adm_atmos.py.
#pragma once
#include <cstdint>
#include <string>
#include <vector>
#include "foundation/status.h"
namespace joc::adm {
inline constexpr int kObjectCount = 15;
inline constexpr std::uint32_t kTrackCount = 25;
struct Keyframe {
std::int64_t rtime_samples = 0;
double x = 0.0;
double y = 0.0;
double z = 0.0;
std::int64_t duration_samples = 0;
std::int64_t interpolation_samples = 0;
};
struct Track {
std::string name;
std::vector<Keyframe> blocks;
};
// HH:MM:SS.fffff with the reference's truncation + round-half-even carry.
std::string ts(double seconds);
enum class BinauralMode : std::uint32_t { Off = 0, Near = 1, Far = 2, Mid = 3, Unspecified = 4 };
bool binaural_mode_from_name(const char* name, BinauralMode* out);
std::string build_axml(const std::vector<Track>& tracks, double duration_sec, std::uint32_t rate);
std::string build_chna();
Status build_dbmd(std::uint32_t object_count, BinauralMode mode, std::string* out);
} // namespace joc::adm
-281
View File
@@ -1,281 +0,0 @@
#include "adm/adm_tracks.h"
#include <algorithm>
#include <cmath>
#include "foundation/geometry.h"
namespace joc::adm {
namespace {
struct Point {
std::int64_t sample = 0;
double x = 0.0;
double y = 0.0;
double z = 0.0;
std::int64_t interpolation_samples = 0;
};
// otherwise (the reference drops silently).
void append_point(std::vector<Point>* points, std::int64_t sample, double x, double y, double z,
std::int64_t interpolation_samples) {
if (!points->empty() && sample == points->back().sample) {
points->back() = Point{sample, x, y, z, interpolation_samples};
} else if (points->empty() || sample > points->back().sample) {
points->push_back(Point{sample, x, y, z, interpolation_samples});
}
}
void lerp(double ax, double ay, double az, double bx, double by, double bz, double amount,
double* x, double* y, double* z) {
*x = ax + (bx - ax) * amount;
*y = ay + (by - ay) * amount;
*z = az + (bz - az) * amount;
}
void points_to_blocks(const std::vector<Point>& points, std::int64_t total_samples,
std::vector<Keyframe>* out) {
for (std::size_t index = 0; index < points.size(); ++index) {
const Point& point = points[index];
const std::int64_t end =
(index + 1 < points.size()) ? points[index + 1].sample : total_samples;
const std::int64_t duration = std::max<std::int64_t>(0, end - point.sample);
if (duration == 0) {
continue;
}
Keyframe keyframe;
keyframe.rtime_samples = point.sample;
keyframe.x = point.x;
keyframe.y = point.y;
keyframe.z = point.z;
keyframe.duration_samples = duration;
keyframe.interpolation_samples = std::min(point.interpolation_samples, duration);
out->push_back(keyframe);
}
}
Status non_monotonic(const char* name, const char* message, int object_index, std::int64_t sample,
std::int64_t previous_sample) {
return Status::fail(JOC_ERR_OAMD_UNSUPPORTED_VARIANT, stage::kOamd,
std::string(name) + ": " + message + " (object " +
std::to_string(object_index) + ", sample " + std::to_string(sample) +
", previous " + std::to_string(previous_sample) + ")");
}
} // namespace
Status expand_compact(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
int object_index, std::vector<Keyframe>* out) {
out->clear();
const double scale = static_cast<double>(rate);
if (events.empty()) {
Keyframe keyframe;
keyframe.rtime_samples = 0;
keyframe.duration_samples = total_samples;
keyframe.interpolation_samples = 0;
keyframe.x = 0.0;
keyframe.y = 0.0;
keyframe.z = 0.0;
out->push_back(keyframe);
return Status::success();
}
std::vector<Point> points;
double current[3] = {events[0].x, events[0].y, events[0].z};
append_point(&points, 0, current[0], current[1], current[2], 0);
for (std::size_t index = 1; index < events.size(); ++index) {
const OamdEvent& event = events[index];
const std::int64_t event_start = event.sample + object_delay_samples;
if (event_start >= total_samples) {
break;
}
const std::int64_t effective_ramp =
std::max<std::int64_t>(0, event.ramp_samples - update_quantum_samples);
const std::int64_t block_start =
event_start + (effective_ramp != 0 ? update_quantum_samples : 0);
if (block_start >= total_samples) {
break;
}
const std::int64_t ramp_end = block_start + effective_ramp;
if (block_start < points.back().sample) {
return non_monotonic("non_monotonic_compact_position_updates",
"compact object position update moved backwards", object_index,
block_start, points.back().sample);
}
if (index + 1 < events.size()) {
const std::int64_t next_event_start = events[index + 1].sample + object_delay_samples;
const std::int64_t next_effective =
std::max<std::int64_t>(0, events[index + 1].ramp_samples - update_quantum_samples);
const std::int64_t next_block_start =
next_event_start + (next_effective != 0 ? update_quantum_samples : 0);
if (next_block_start < ramp_end) {
return non_monotonic("overlapping_compact_position_ramps",
"a new position update arrived before the previous compact "
"ramp finished",
object_index, block_start, ramp_end);
}
}
double target[3] = {event.x, event.y, event.z};
std::int64_t interpolation = effective_ramp;
const std::int64_t available = total_samples - block_start;
if (effective_ramp > available) {
const double amount =
static_cast<double>(available) / static_cast<double>(effective_ramp);
double x = 0.0;
double y = 0.0;
double z = 0.0;
lerp(current[0], current[1], current[2], event.x, event.y, event.z, amount, &x, &y, &z);
target[0] = x;
target[1] = y;
target[2] = z;
interpolation = available;
}
append_point(&points, block_start, target[0], target[1], target[2], interpolation);
current[0] = event.x;
current[1] = event.y;
current[2] = event.z;
}
points_to_blocks(points, total_samples, out);
(void)scale;
return Status::success();
}
Status expand_dense64(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
int object_index, std::vector<Keyframe>* out) {
out->clear();
(void)rate;
if (events.empty()) {
Keyframe keyframe;
keyframe.duration_samples = total_samples;
out->push_back(keyframe);
return Status::success();
}
std::vector<Point> points;
double current[3] = {events[0].x, events[0].y, events[0].z};
append_point(&points, 0, current[0], current[1], current[2], 0);
for (std::size_t index = 1; index < events.size(); ++index) {
const OamdEvent& event = events[index];
const std::int64_t start = event.sample + object_delay_samples;
if (start >= total_samples) {
break;
}
if (start < points.back().sample) {
return non_monotonic("non_monotonic_position_updates",
"object position update moved backwards", object_index, start,
points.back().sample);
}
if (start > points.back().sample) {
append_point(&points, start, current[0], current[1], current[2], 0);
}
const std::int64_t effective_ramp =
std::max<std::int64_t>(0, event.ramp_samples - update_quantum_samples);
if (effective_ramp == 0) {
append_point(&points, start, event.x, event.y, event.z, 0);
current[0] = event.x;
current[1] = event.y;
current[2] = event.z;
continue;
}
const std::int64_t steps =
(effective_ramp + update_quantum_samples - 1) / update_quantum_samples;
const std::int64_t end = start + steps * update_quantum_samples;
if (index + 1 < events.size()) {
const std::int64_t next_start = events[index + 1].sample + object_delay_samples;
if (next_start < end) {
return non_monotonic("overlapping_position_ramps",
"a new position update arrived before the previous ramp "
"finished",
object_index, start, end);
}
}
std::int64_t future = effective_ramp;
std::int64_t elapsed = 0;
double position[3] = {current[0], current[1], current[2]};
while (future > 0) {
const double amount =
std::min(static_cast<double>(update_quantum_samples) / static_cast<double>(future),
1.0);
double x = 0.0;
double y = 0.0;
double z = 0.0;
lerp(position[0], position[1], position[2], event.x, event.y, event.z, amount, &x, &y,
&z);
position[0] = x;
position[1] = y;
position[2] = z;
elapsed += update_quantum_samples;
const std::int64_t sample = start + elapsed;
if (sample >= total_samples) {
break;
}
append_point(&points, sample, position[0], position[1], position[2],
update_quantum_samples);
future -= update_quantum_samples;
}
current[0] = event.x;
current[1] = event.y;
current[2] = event.z;
}
points_to_blocks(points, total_samples, out);
return Status::success();
}
void TrajectoryBuilder::submit_frame(std::int64_t frame_index, const oamd::OamdUpdate* update,
std::int64_t outer_offset) {
std::int64_t event_sample = frame_index * 1536;
std::int64_t ramp_samples = 0;
if (update != nullptr) {
state_.apply(*update);
event_sample += outer_offset + static_cast<std::int64_t>(update->block_offset_samples);
ramp_samples = static_cast<std::int64_t>(update->ramp_duration_samples);
}
for (int object = 1; object <= kObjectCount; ++object) {
double x = 0.0;
double y = 0.0;
double z = 0.0;
geometry::q_to_adm_xyz(state_.q(object, 0), state_.q(object, 1), state_.q(object, 2), &x,
&y, &z);
const int slot = object - 1;
if (!has_previous_[slot] || previous_[slot][0] != x || previous_[slot][1] != y ||
previous_[slot][2] != z) {
events_[slot].push_back(OamdEvent{event_sample, x, y, z, ramp_samples});
previous_[slot][0] = x;
previous_[slot][1] = y;
previous_[slot][2] = z;
has_previous_[slot] = true;
}
}
}
Status TrajectoryBuilder::build(std::int64_t total_samples, TrajectoryMode mode,
std::vector<Track>* out) const {
out->clear();
out->reserve(kObjectCount);
for (int object = 1; object <= kObjectCount; ++object) {
Track track;
track.name = "JOC_Object_" + std::to_string(object);
const Status status =
(mode == TrajectoryMode::Compact)
? expand_compact(events_[object - 1], total_samples, rate_, quantum_,
object_delay_, object, &track.blocks)
: expand_dense64(events_[object - 1], total_samples, rate_, quantum_,
object_delay_, object, &track.blocks);
if (!status.ok()) {
return status;
}
out->push_back(std::move(track));
}
return Status::success();
}
} // namespace joc::adm
-61
View File
@@ -1,61 +0,0 @@
// Port of src/oamd_tracks.py.
#pragma once
#include <cstdint>
#include <string>
#include <vector>
#include "adm/adm_metadata.h"
#include "foundation/status.h"
#include "oamd/oamd_parser.h"
namespace joc::adm {
struct OamdEvent {
std::int64_t sample = 0;
double x = 0.0;
double y = 0.0;
double z = 0.0;
std::int64_t ramp_samples = 0;
};
enum class TrajectoryMode { Compact, Dense64 };
// Feeds the same per-frame OAMD state machine the reference's build_adm_tracks
// runs, and records one event per object whenever its coordinates change.
class TrajectoryBuilder {
public:
TrajectoryBuilder(std::uint32_t rate = 48000, std::int64_t update_quantum_samples = 64,
std::int64_t object_delay_samples = 1473)
: rate_(rate),
quantum_(update_quantum_samples),
object_delay_(object_delay_samples) {}
void submit_frame(std::int64_t frame_index, const oamd::OamdUpdate* update, std::int64_t outer_offset);
Status build(std::int64_t total_samples, TrajectoryMode mode, std::vector<Track>* out) const;
const std::vector<OamdEvent>& events(int object_index) const { return events_[object_index]; }
std::uint32_t rate() const { return rate_; }
std::int64_t object_delay_samples() const { return object_delay_; }
private:
std::uint32_t rate_;
std::int64_t quantum_;
std::int64_t object_delay_;
oamd::OamdState state_;
std::vector<OamdEvent> events_[kObjectCount];
bool has_previous_[kObjectCount] = {};
double previous_[kObjectCount][3] = {};
};
Status expand_compact(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
int object_index, std::vector<Keyframe>* out);
Status expand_dense64(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
int object_index, std::vector<Keyframe>* out);
} // namespace joc::adm
+145
View File
@@ -0,0 +1,145 @@
"""把 LFE、15 路对象 PCM 和对象轨迹组装为 ADM BWF。
固定输出契约:
EAC3JOC 重放输出 = 16ch(ch0 = LFE + ch1-15 = 15 对象);
最终 ADM BWF = 7.1.2 bed(L R C Ls Rs Lb Rb + LFE + Ltf Rtf = 10ch)
—— 除 LFE 外全部静音;
15 对象 = ch1-15 直接填充对象轨;轨迹 = OAMD(q1/q2/q3 → xyz)。
"""
import os
import numpy as np
import adm_atmos
def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None,
duration_sec=None, rate=48000, joc_binaural_mode=4):
"""16ch f32 交织 raw → 25ch ADM BWF(空 7.1.2 bed + LFE + 15 对象)。
raw16: (n, 16) 交织(ch0 = LFE,ch1-15 = 对象)。
scale: 1.0 = 默认 0 dB,不附加输出缩放。该参数与 joc_clipgain 无关;
主命令行已在渲染阶段应用用户增益,因此这里传 1.0。
kf_tracks: 可选轨迹关键帧(OAMD 输出,格式 [(obj_id, [(t, x, y, z), ...]), ...]);
缺省 = 静止参考位置(adm_atmos 默认)。
"""
raw = np.memmap(raw16_path, dtype=np.float32, mode="r")
n = len(raw) // 16
raw = raw[:n * 16].reshape(-1, 16)
if duration_sec is None:
duration_sec = n / rate
# 惰性视图:adm_atmos 按块读取,避免全片 25ch 在内存中展开。
class BedView:
shape = (n, 10)
def __getitem__(self, key):
src = np.asarray(raw[key], dtype=np.float32)
one = src.ndim == 1
if one:
src = src[None, :]
out = np.zeros((len(src), 10), dtype=np.float32)
out[:, 3] = np.multiply(src[:, 0], np.float32(scale), dtype=np.float32)
return out[0] if one else out
class ObjView:
shape = (n, 16)
def __getitem__(self, key):
return np.multiply(np.asarray(raw[key], dtype=np.float32),
np.float32(scale), dtype=np.float32)
if kf_tracks is None:
kf_tracks = []
for oi in range(15):
# 静止参考位置(q1=q2=q3=0 → 原点;实际坐标按 OAMD 输出填入)
kf_tracks.append(("JOC_Object_%d" % (oi + 1),
[(0.0, 0.0, 0.0, 0.0, max(duration_sec, 1e-6))]))
adm_atmos.build_master(out_path, BedView(), ObjView(), kf_tracks,
duration_sec, rate=rate,
joc_binaural_mode=joc_binaural_mode)
# 及时释放 Windows 文件句柄,允许 TemporaryDirectory 删除中间 raw。
raw._mmap.close()
return out_path
class StreamingMaster:
"""Incrementally write renderer frames into the final 25-channel ADM BWF.
This removes the default 16-channel float32 intermediate file. The mapping
remains identical to :func:`assemble_from_raw`: bed channel 3 receives LFE,
bed channels 0..2/4..9 are silent, and output objects 1..15 map to ADM
channels 10..24.
"""
def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072,
joc_binaural_mode=4):
if block_samples < 1536:
raise ValueError("block_samples must be at least one E-AC-3 frame")
self.out_path = os.fspath(out_path)
self.duration_sec = float(duration_sec)
self.rate = int(rate)
self.joc_binaural_mode = joc_binaural_mode
self._sink = adm_atmos.Sink25(self.out_path, 25, self.rate)
self._buffer = np.empty((int(block_samples), 25), dtype=np.float32)
self._used = 0
self._finalized = False
def _flush(self):
if self._used:
self._sink.write_block(self._buffer[:self._used])
self._used = 0
def write_frame(self, pcm16):
pcm = np.asarray(pcm16, dtype=np.float32)
if pcm.shape != (16, 1536):
raise ValueError(f"renderer frame must be (16,1536), got {pcm.shape}")
source = 0
while source < 1536:
available = len(self._buffer) - self._used
count = min(available, 1536 - source)
target = self._buffer[self._used:self._used + count]
target.fill(0.0)
target[:, 3] = pcm[0, source:source + count]
target[:, 10:25] = pcm[1:16, source:source + count].T
self._used += count
source += count
if self._used == len(self._buffer):
self._flush()
def finalize(self, kf_tracks):
if self._finalized:
raise RuntimeError("StreamingMaster already finalized")
self._flush()
try:
from . import adm_serializer
except ImportError:
import adm_serializer
axml = adm_serializer.build_axml(kf_tracks, self.duration_sec)
chna = adm_atmos.build_chna()
dbmd = adm_atmos.build_dbmd(
25, joc_binaural_mode=self.joc_binaural_mode)
trajectory_blocks = sum(len(track[1]) for track in kf_tracks)
self.metadata_info = {
"axml_bytes": len(axml),
"trajectory_blocks": trajectory_blocks,
"chna_bytes": len(chna),
"dbmd_bytes": len(dbmd),
}
self._sink.finalize(axml, chna, dbmd)
self._finalized = True
print(f"master25 -> {self.out_path} ({self.duration_sec:.2f}s, 25ch, "
f"axml={len(axml)}B, chna={len(chna)}B, dbmd={len(dbmd)}B)")
return self.out_path
def abort(self):
if self._finalized:
return
sink = getattr(self, "_sink", None)
fp = getattr(sink, "fp", None)
if fp is not None and not fp.closed:
fp.close()
def __del__(self):
try:
self.abort()
except Exception:
pass
+300
View File
@@ -0,0 +1,300 @@
"""生成 25 声道 RF64 ADM BWF 及其 axml、chna、dbmd 元数据。
输出由 10 声道 7.1.2 bed 和 15 路对象组成;RF64 尺寸字段在写入完成后回填。
"""
import operator
import struct
import numpy as np
import xml.etree.ElementTree as ET
NS = "urn:ebu:metadata-schema:ebuCore_2016"
XSI = "http://www.w3.org/2001/XMLSchema-instance"
BED_NAMES = ["RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter",
"RoomCentricLFE", "RoomCentricLeftSideSurround",
"RoomCentricRightSideSurround", "RoomCentricLeftRearSurround",
"RoomCentricRightRearSurround", "RoomCentricLeftTopSurround",
"RoomCentricRightTopSurround"]
BED_LABELS = ["RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss", "RC_Rss",
"RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"]
BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0),
(-1.0, 1.0, -1.0), (-1.0, 0.0, 0.0), (1.0, 0.0, 0.0),
(-1.0, -1.0, 0.0), (1.0, -1.0, 0.0), (-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
N_OBJ = 15
JOC_BINAURAL_MODES = {
"off": 0,
"near": 1,
"far": 2,
"mid": 3,
"unspecified": 4,
}
JOC_BINAURAL_MODE_DEFAULT = "unspecified"
def q_to_adm_xyz(q1, q2, q3):
posX = min(1.0, round(q1 * 62 / 32767.0) / 62.0)
posY = min(1.0, round(q2 * 62 / 32767.0) / 62.0)
posZ = round(q3 * 15 / 32767.0) / 15.0
posZ = max(-1.0, min(1.0, posZ))
return posX * 2 - 1, 1 - posY * 2, posZ
def ts(seconds):
s = int(seconds)
frac = int(round((seconds - s) * 100000))
if frac >= 100000:
s += 1; frac = 0
return f"{s // 3600:02d}:{s % 3600 // 60:02d}:{s % 60:02d}.{frac:05d}"
def sub(parent, tag, attrib=None, text=None):
e = ET.SubElement(parent, tag)
if attrib:
for k, v in attrib.items():
e.set(k, v)
if text is not None:
e.text = text
return e
def add_refs(parent, tag, ids):
for i in ids:
sub(parent, tag, text=i)
def obj_block(cf, bid, t, x, y, z, dur, interpolation=0.0):
b = sub(cf, "audioBlockFormat", {
"audioBlockFormatID": bid, "rtime": ts(t), "duration": ts(dur)})
sub(b, "cartesian", text="1")
for c, v in (("X", x), ("Y", y), ("Z", z)):
if c == "Z" and v == 0:
continue
p = sub(b, "position", {"coordinate": c})
p.text = f"{v:.10f}"
sub(b, "jumpPosition", {"interpolationLength": f"{interpolation:.5f}"}, text="1")
def build_axml(obj_tracks, duration_sec):
adm = ET.Element("ebuCoreMain", {
"xmlns": NS, "xmlns:xsi": XSI,
"xsi:schemaLocation": f"{NS} ebucore.xsd", "lang": "en"})
core = sub(adm, "coreMetadata")
fmt = sub(core, "format")
af = sub(fmt, "audioFormatExtended")
prog = sub(af, "audioProgramme", {
"audioProgrammeID": "APR_1001", "audioProgrammeName": "EAC3JOC_Export",
"start": ts(0), "end": ts(duration_sec)})
add_refs(prog, "audioContentIDRef", ("ACO_1001", "ACO_1002"))
bc = sub(af, "audioContent", {"audioContentID": "ACO_1001",
"audioContentName": "EAC3JOC_Master_Content"})
add_refs(bc, "audioObjectIDRef", ["AO_1001"])
sub(bc, "dialogue", {"mixedContentKind": "0"})
oc = sub(af, "audioContent", {"audioContentID": "ACO_1002",
"audioContentName": "Objects"})
add_refs(oc, "audioObjectIDRef", ["AO_%04x" % (0x100b + i) for i in range(N_OBJ)])
sub(oc, "dialogue", {"mixedContentKind": "0"})
bed_o = sub(af, "audioObject", {"audioObjectID": "AO_1001", "audioObjectName": "Bed",
"start": ts(0), "duration": ts(duration_sec)})
sub(bed_o, "audioPackFormatIDRef", text="AP_00011001")
add_refs(bed_o, "audioTrackUIDRef", ["ATU_%08x" % (i + 1) for i in range(10)])
for i in range(N_OBJ):
o = sub(af, "audioObject", {"audioObjectID": "AO_%04x" % (0x100b + i),
"audioObjectName": f"Audio Object {i+1}",
"start": ts(0), "duration": ts(duration_sec)})
sub(o, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
add_refs(o, "audioTrackUIDRef", ["ATU_%08x" % (i + 11)])
bp = sub(af, "audioPackFormat", {"audioPackFormatID": "AP_00011001",
"audioPackFormatName": "EAC3JOCBedPack",
"typeDefinition": "DirectSpeakers", "typeLabel": "0001"})
add_refs(bp, "audioChannelFormatIDRef", ["AC_0001%04x" % (0x1001 + i) for i in range(10)])
for i in range(N_OBJ):
pk = sub(af, "audioPackFormat", {"audioPackFormatID": "AP_0003%04x" % (0x1001 + i),
"audioPackFormatName": f"JOC_Object_{i+1}",
"typeDefinition": "Objects", "typeLabel": "0003"})
add_refs(pk, "audioChannelFormatIDRef", ["AC_0003%04x" % (0x1001 + i)])
for i in range(10):
cf = sub(af, "audioChannelFormat", {"audioChannelFormatID": "AC_0001%04x" % (0x1001 + i),
"audioChannelFormatName": BED_NAMES[i],
"typeDefinition": "DirectSpeakers", "typeLabel": "0001"})
b = sub(cf, "audioBlockFormat", {"audioBlockFormatID": "AB_0001%04x_00000001" % (0x1001 + i)})
sub(b, "cartesian", text="1")
x, y, z = BED_POS[i]
for c, v in (("X", x), ("Y", y), ("Z", z)):
if c == "Z" and v == 0:
continue
p = sub(b, "position", {"coordinate": c})
p.text = f"{v:.10f}"
sub(b, "speakerLabel", text=BED_LABELS[i])
for i, (oname, kfs) in enumerate(obj_tracks):
cf = sub(af, "audioChannelFormat", {"audioChannelFormatID": "AC_0003%04x" % (0x1001 + i),
"audioChannelFormatName": oname,
"typeDefinition": "Objects", "typeLabel": "0003"})
for k, keyframe in enumerate(kfs):
t, x, y, z, dur = keyframe[:5]
interpolation = keyframe[5] if len(keyframe) > 5 else 0.0
obj_block(cf, "AB_0003%04x_%08x" % (0x1001 + i, k + 1),
t, x, y, z, dur, interpolation)
for i in range(10):
t = sub(af, "audioTrackUID", {"UID": "ATU_%08x" % (i + 1),
"bitDepth": "24", "sampleRate": "48000"})
sub(t, "audioTrackFormatIDRef", text="AT_0001%04x_01" % (0x1001 + i))
sub(t, "audioPackFormatIDRef", text="AP_00011001")
for i in range(N_OBJ):
t = sub(af, "audioTrackUID", {"UID": "ATU_%08x" % (i + 11),
"bitDepth": "24", "sampleRate": "48000"})
sub(t, "audioTrackFormatIDRef", text="AT_0003%04x_01" % (0x1001 + i))
sub(t, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
for i in range(10):
tf = sub(af, "audioTrackFormat", {"audioTrackFormatID": "AT_0001%04x_01" % (0x1001 + i),
"audioTrackFormatName": "PCM_" + BED_NAMES[i],
"formatDefinition": "PCM", "formatLabel": "0001"})
sub(tf, "audioStreamFormatIDRef", text="AS_0001%04x" % (0x1001 + i))
for i in range(N_OBJ):
tf = sub(af, "audioTrackFormat", {"audioTrackFormatID": "AT_0003%04x_01" % (0x1001 + i),
"audioTrackFormatName": "PCM_JOC_Object_%d" % (i + 1),
"formatDefinition": "PCM", "formatLabel": "0001"})
sub(tf, "audioStreamFormatIDRef", text="AS_0003%04x" % (0x1001 + i))
for i in range(10):
sf = sub(af, "audioStreamFormat", {"audioStreamFormatID": "AS_0001%04x" % (0x1001 + i),
"audioStreamFormatName": "PCM_" + BED_NAMES[i],
"formatDefinition": "PCM", "formatLabel": "0001"})
sub(sf, "audioChannelFormatIDRef", text="AC_0001%04x" % (0x1001 + i))
sub(sf, "audioPackFormatIDRef", text="AP_00011001")
sub(sf, "audioTrackFormatIDRef", text="AT_0001%04x_01" % (0x1001 + i))
for i in range(N_OBJ):
sf = sub(af, "audioStreamFormat", {"audioStreamFormatID": "AS_0003%04x" % (0x1001 + i),
"audioStreamFormatName": "PCM_JOC_Object_%d" % (i + 1),
"formatDefinition": "PCM", "formatLabel": "0001"})
sub(sf, "audioChannelFormatIDRef", text="AC_0003%04x" % (0x1001 + i))
sub(sf, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
sub(sf, "audioTrackFormatIDRef", text="AT_0003%04x_01" % (0x1001 + i))
return ET.tostring(adm, encoding="utf-8", xml_declaration=True)
def build_chna():
out = bytearray()
out += struct.pack("<HH", 25, 25)
for i in range(10):
out += struct.pack("<H", i + 1)
out += ("ATU_%08x" % (i + 1)).encode()
out += ("AT_0001%04x_01" % (0x1001 + i)).encode()
out += b"AP_00011001" + b"\x00"
for i in range(N_OBJ):
out += struct.pack("<H", i + 11)
out += ("ATU_%08x" % (i + 11)).encode()
out += ("AT_0003%04x_01" % (0x1001 + i)).encode()
out += ("AP_0003%04x" % (0x1001 + i)).encode() + b"\x00"
return bytes(out)
def _checksum(seg):
s = len(seg)
for b in seg:
s += b
return (~s + 1) & 0xFF
def build_dbmd(object_count=25, joc_binaural_mode=4):
"""仅覆盖 segment 10 中 JOC object slots 10..24 的 mode 低 3 bit。"""
mode = operator.index(joc_binaural_mode)
if mode not in JOC_BINAURAL_MODES.values():
raise ValueError(f"invalid JOC binaural render mode: {mode}")
out = bytearray(struct.pack("<I", 0x01000006))
dd = bytearray(96)
dd[1] = 0x47
dd[5] = 0x60
dd[8] = 0x24; dd[9] = 0x24
out.append(7); out += struct.pack("<H", 96); out += bytes(dd)
out.append(_checksum(dd))
at = bytearray(248)
c0 = b"Created with EAC3JOC"; c1 = b"EAC3JOC Python Renderer"
at[0:len(c0)] = c0
at[32:32 + len(c1)] = c1
at[96], at[97], at[98] = 2, 1, 0
at[103] = 0x03
at[106] = 0x01
at[111] = 0x22; at[112] = 0xFF
out.append(9); out += struct.pack("<H", 248); out += bytes(at)
out.append(_checksum(at))
ob = bytearray(5 + 262 + object_count)
ob[0:4] = struct.pack("<I", 0xF8726FBD)
ob[4] = object_count
for i in range(5 + 262, len(ob)):
ob[i] = 0x84
# sync (4), count (2), reserved (1), nine 15-byte config trims,
# then one trim-bypass byte per track before the headphone modes.
# Preserve the existing template's bed fields and trailing bytes.
object_modes = 4 + 2 + 1 + 9 * 15 + object_count
for i in range(10, min(object_count, 10 + N_OBJ)):
ob[object_modes + i] = (ob[object_modes + i] & 0xF8) | mode
out.append(10); out += struct.pack("<H", len(ob)); out += bytes(ob)
out.append(_checksum(ob))
out += b"\x00\x00"
return bytes(out)
class Sink25:
"""RF64 ADM BWF writer.
Header layout is fixed so that sizes can be patched without rereading the
file: RF64+size+WAVE (12) + ds64 chunk (8+28) + fmt chunk (8+16) + data
chunk header (8). Sizes beyond 32 bits follow the RF64 convention: the
chunk size field holds 0xFFFFFFFF and the true value lives in ds64.
"""
_DS64_BODY_OFFSET = 20
_DATA_SIZE_OFFSET = 76
def __init__(self, path, channels, rate):
self.ch = channels; self.rate = rate; self.frames = 0
self.fp = open(path, "wb+")
self.fp.write(b"RF64" + struct.pack("<I", 0xFFFFFFFF) + b"WAVE")
self._chunk(b"ds64", b"\x00" * 28)
self._chunk(b"fmt ", self._fmt())
self._chunk(b"data", b"")
def _chunk(self, cid, body):
self.fp.write(cid + struct.pack("<I", len(body)) + body)
if len(body) & 1:
self.fp.write(b"\x00")
def _fmt(self):
return struct.pack("<HHIIHH", 1, self.ch, self.rate,
self.rate * self.ch * 3, self.ch * 3, 24)
def write_block(self, arr):
arr = arr.reshape(-1, self.ch)
i24 = (np.clip(arr, -1.0, 1.0) * 8388607.0).astype(np.int32)
self.fp.write(i24.view(np.uint8).reshape(-1, 4)[:, :3].tobytes())
self.frames += arr.shape[0]
def finalize(self, axml_bytes, chna_bytes, dbmd_bytes):
data_len = self.frames * self.ch * 3
self._chunk(b"axml", axml_bytes)
self._chunk(b"chna", chna_bytes)
self._chunk(b"dbmd", dbmd_bytes)
self.fp.seek(0, 2); total = self.fp.tell()
# RF64: 超过 32-bit 的 chunk size 字段写 0xFFFFFFFF,真实大小回填 ds64。
self.fp.seek(self._DATA_SIZE_OFFSET)
self.fp.write(struct.pack(
"<I", data_len if data_len <= 0xFFFFFFFF else 0xFFFFFFFF))
self.fp.seek(self._DS64_BODY_OFFSET)
self.fp.write(struct.pack("<QQQI", total - 8, data_len, self.frames, 0))
self.fp.flush()
self.fp.close()
def build_master(out_path, bed_mm, obj_mm, kf_tracks, duration_sec, rate=48000,
block=480000, joc_binaural_mode=4):
n = min(bed_mm.shape[0], obj_mm.shape[0])
try:
from . import adm_serializer
except ImportError:
import adm_serializer
serial_axml = adm_serializer.build_axml
axml = serial_axml(kf_tracks, duration_sec)
chna = build_chna()
dbmd = build_dbmd(25, joc_binaural_mode=joc_binaural_mode)
sink = Sink25(out_path, 25, rate)
for st in range(0, n, block):
en = min(n, st + block)
blk = np.hstack((np.asarray(bed_mm[st:en], dtype=np.float32),
np.asarray(obj_mm[st:en, 1:16], dtype=np.float32)))
sink.write_block(blk)
sink.finalize(axml, chna, dbmd)
print(f"master25 -> {out_path} ({duration_sec:.2f}s, 25ch, axml={len(axml)}B, "
f"chna={len(chna)}B, dbmd={len(dbmd)}B)")
+136
View File
@@ -0,0 +1,136 @@
"""把 7.1.2 bed、15 个对象及其位置轨迹序列化为 ADM axml。
序列化结果采用固定元素顺序、属性顺序和十六进制 ADM 标识符,便于稳定输出和校验。
"""
BED_NAMES = ["RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter", "RoomCentricLFE",
"RoomCentricLeftSideSurround", "RoomCentricRightSideSurround",
"RoomCentricLeftRearSurround", "RoomCentricRightRearSurround",
"RoomCentricLeftTopSurround", "RoomCentricRightTopSurround"]
BED_LABELS = ["RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss", "RC_Rss",
"RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"]
BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0), (-1.0, 1.0, -1.0),
(-1.0, 0.0, 0.0), (1.0, 0.0, 0.0), (-1.0, -1.0, 0.0), (1.0, -1.0, 0.0),
(-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
N_OBJ = 15
def ts(seconds):
s = int(seconds)
frac = int(round((seconds - s) * 100000))
if frac >= 100000:
s += 1; frac = 0
return f"{s // 3600:02d}:{s % 3600 // 60:02d}:{s % 60:02d}.{frac:05d}"
def esc(v):
return (str(v).replace("&", "&amp;").replace("<", "&lt;").replace(">", "&gt;"))
def build_axml(obj_tracks, duration_sec):
"""obj_tracks: [(name, [(rtime, x, y, z, dur), ...]) ×15]"""
w = []
a = w.append
a('<?xml version="1.0" encoding="utf-8"?>')
a('<ebuCoreMain xsi:schemaLocation="urn:ebu:metadata-schema:ebuCore_2016 ebucore.xsd" '
'lang="en" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" '
'xmlns="urn:ebu:metadata-schema:ebuCore_2016">')
a('<coreMetadata><format><audioFormatExtended>')
a(f'<audioProgramme audioProgrammeID="APR_1001" audioProgrammeName="EAC3JOC_Export" '
f'start="{ts(0)}" end="{ts(duration_sec)}">')
a('<audioContentIDRef>ACO_1001</audioContentIDRef>')
a('<audioContentIDRef>ACO_1002</audioContentIDRef>')
a('</audioProgramme>')
a('<audioContent audioContentID="ACO_1001" audioContentName="EAC3JOC_Master_Content">')
a('<audioObjectIDRef>AO_1001</audioObjectIDRef>')
a('<dialogue mixedContentKind="0">2</dialogue>')
a('</audioContent>')
a('<audioContent audioContentID="ACO_1002" audioContentName="Objects">')
for i in range(N_OBJ):
a(f'<audioObjectIDRef>AO_{0x100b + i:04x}</audioObjectIDRef>')
a('<dialogue mixedContentKind="0">2</dialogue>')
a('</audioContent>')
a(f'<audioObject audioObjectID="AO_1001" audioObjectName="Bed" '
f'start="{ts(0)}" duration="{ts(duration_sec)}">')
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
for i in range(10):
a(f'<audioTrackUIDRef>ATU_{i + 1:08x}</audioTrackUIDRef>')
a('</audioObject>')
for i in range(N_OBJ):
a(f'<audioObject audioObjectID="AO_{0x100b + i:04x}" audioObjectName="Audio Object {i+1}" '
f'start="{ts(0)}" duration="{ts(duration_sec)}">')
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
a(f'<audioTrackUIDRef>ATU_{11 + i:08x}</audioTrackUIDRef>')
a('</audioObject>')
a('<audioPackFormat audioPackFormatID="AP_00011001" audioPackFormatName="EAC3JOCBedPack" '
'typeDefinition="DirectSpeakers" typeLabel="0001">')
for i in range(10):
a(f'<audioChannelFormatIDRef>AC_0001{0x1001 + i:04x}</audioChannelFormatIDRef>')
a('</audioPackFormat>')
for i in range(N_OBJ):
a(f'<audioPackFormat audioPackFormatID="AP_0003{0x1001 + i:04x}" '
f'audioPackFormatName="JOC_Object_{i+1}" typeDefinition="Objects" typeLabel="0003">')
a(f'<audioChannelFormatIDRef>AC_0003{0x1001 + i:04x}</audioChannelFormatIDRef>')
a('</audioPackFormat>')
for i in range(10):
a(f'<audioChannelFormat audioChannelFormatID="AC_0001{0x1001 + i:04x}" '
f'audioChannelFormatName="{BED_NAMES[i]}" typeDefinition="DirectSpeakers" typeLabel="0001">')
a(f'<audioBlockFormat audioBlockFormatID="AB_0001{0x1001 + i:04x}_00000001">')
a('<cartesian>1</cartesian>')
x, y, z = BED_POS[i]
a(f'<position coordinate="X">{x:.10f}</position>')
a(f'<position coordinate="Y">{y:.10f}</position>')
if z != 0:
a(f'<position coordinate="Z">{z:.10f}</position>')
a(f'<speakerLabel>{BED_LABELS[i]}</speakerLabel>')
a('</audioBlockFormat>')
a('</audioChannelFormat>')
for i, (oname, kfs) in enumerate(obj_tracks):
a(f'<audioChannelFormat audioChannelFormatID="AC_0003{0x1001 + i:04x}" '
f'audioChannelFormatName="{oname}" typeDefinition="Objects" typeLabel="0003">')
for k, keyframe in enumerate(kfs):
t, x, y, z, dur = keyframe[:5]
interpolation = keyframe[5] if len(keyframe) > 5 else 0.0
a(f'<audioBlockFormat audioBlockFormatID="AB_0003{0x1001 + i:04x}_{k + 1:08x}" '
f'rtime="{ts(t)}" duration="{ts(dur)}">')
a('<cartesian>1</cartesian>')
a(f'<position coordinate="X">{x:.10f}</position>')
a(f'<position coordinate="Y">{y:.10f}</position>')
if z != 0:
a(f'<position coordinate="Z">{z:.10f}</position>')
a(f'<jumpPosition interpolationLength="{interpolation:.5f}">1</jumpPosition>')
a('</audioBlockFormat>')
a('</audioChannelFormat>')
for i in range(10):
a(f'<audioTrackUID UID="ATU_{i + 1:08x}" bitDepth="24" sampleRate="48000">')
a(f'<audioTrackFormatIDRef>AT_0001{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
a('</audioTrackUID>')
for i in range(N_OBJ):
a(f'<audioTrackUID UID="ATU_{11 + i:08x}" bitDepth="24" sampleRate="48000">')
a(f'<audioTrackFormatIDRef>AT_0003{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
a('</audioTrackUID>')
for i in range(10):
a(f'<audioTrackFormat audioTrackFormatID="AT_0001{0x1001 + i:04x}_01" '
f'audioTrackFormatName="PCM_{BED_NAMES[i]}" formatDefinition="PCM" formatLabel="0001">')
a(f'<audioStreamFormatIDRef>AS_0001{0x1001 + i:04x}</audioStreamFormatIDRef>')
a('</audioTrackFormat>')
for i in range(N_OBJ):
a(f'<audioTrackFormat audioTrackFormatID="AT_0003{0x1001 + i:04x}_01" '
f'audioTrackFormatName="PCM_JOC_Object_{i+1}" formatDefinition="PCM" formatLabel="0001">')
a(f'<audioStreamFormatIDRef>AS_0003{0x1001 + i:04x}</audioStreamFormatIDRef>')
a('</audioTrackFormat>')
for i in range(10):
a(f'<audioStreamFormat audioStreamFormatID="AS_0001{0x1001 + i:04x}" '
f'audioStreamFormatName="PCM_{BED_NAMES[i]}" formatDefinition="PCM" formatLabel="0001">')
a(f'<audioChannelFormatIDRef>AC_0001{0x1001 + i:04x}</audioChannelFormatIDRef>')
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
a(f'<audioTrackFormatIDRef>AT_0001{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
a('</audioStreamFormat>')
for i in range(N_OBJ):
a(f'<audioStreamFormat audioStreamFormatID="AS_0003{0x1001 + i:04x}" '
f'audioStreamFormatName="PCM_JOC_Object_{i+1}" formatDefinition="PCM" formatLabel="0001">')
a(f'<audioChannelFormatIDRef>AC_0003{0x1001 + i:04x}</audioChannelFormatIDRef>')
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
a(f'<audioTrackFormatIDRef>AT_0003{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
a('</audioStreamFormat>')
a('</audioFormatExtended></format></coreMetadata>')
a('</ebuCoreMain>')
return ''.join(w).encode('utf-8')
+199
View File
@@ -0,0 +1,199 @@
"""校验 ADM BWF 的 RF64、通道、axml、chna、dbmd 和对象引用结构。
用法:``python src/adm_validate.py <file.wav> [more.wav ...]``,全部通过时退出码为 0。
"""
import struct, sys, re, os
def fail(msgs, m): msgs.append(m)
def walk_chunks(path):
chunks, ds64 = [], {}
with open(path, "rb") as f:
riff = f.read(4); f.read(4); wave = f.read(4)
if riff not in (b"RIFF", b"RF64"):
return None, None, f"File does not have a 'RIFF' or 'RF64' chunk"
if wave != b"WAVE":
return None, None, "File does not have a required 'WAVE' chunk"
while True:
off = f.tell()
cid = f.read(4)
if len(cid) < 4: break
sz = struct.unpack("<I", f.read(4))[0]
if cid == b"ds64":
body = f.read(sz + (sz & 1))
riff64, data64, sample64, _ = struct.unpack("<QQQI", body[:28])
ds64 = dict(riff64=riff64, data64=data64, sample64=sample64)
chunks.append(("ds64", off, sz)); continue
chunks.append((cid.decode("latin1"), off, sz))
eff = ds64.get("data64", sz) if (sz == 0xFFFFFFFF and cid == b"data") else sz
f.seek(off + 8 + eff + (eff & 1))
return chunks, ds64, None
def read_body(path, chunks, cid):
for c, off, sz in chunks:
if c == cid:
with open(path, "rb") as f:
f.seek(off + 8)
return f.read(sz)
return None
def parse_chna(body):
n_track, n_uid = struct.unpack("<HH", body[:4])
rows, p = [], 4
while p + 40 <= len(body):
trk = struct.unpack("<H", body[p:p+2])[0]
uid = body[p+2:p+14].rstrip(b"\x00").decode()
tf = body[p+14:p+28].rstrip(b"\x00").decode()
pk = body[p+28:p+40].rstrip(b"\x00").decode()
rows.append((trk, uid, tf, pk)); p += 40
return n_track, n_uid, rows
def decode_channel_input(ao_id):
"""将 ``AO_xxxx`` 的十六进制标识符解码为低 12 位通道输入号。"""
m = re.fullmatch(r"AO_([0-9a-fA-F]+)", ao_id)
if not m:
return None
v = int(m.group(1), 16)
if v > 0x1FFF: # 超过 12 位通道域
return None
return v & 0x0FFF
def ts_sec(s):
h, m, rest = s.split(":")
return int(h) * 3600 + int(m) * 60 + float(rest)
def validate(path, axml_override=None, chna_override=None):
msgs = []
chunks, ds64, err = walk_chunks(path)
if err:
return [err]
have = {c for c, _, _ in chunks}
for need in ("fmt ", "data", "axml", "chna", "dbmd"):
if need not in have:
fail(msgs, f"File does not have a required '{need.strip()}' chunk")
if msgs:
return msgs
fmt = read_body(path, chunks, "fmt ")
f_tag, f_ch, f_rate, _, _, f_bits = struct.unpack("<HHIIHH", fmt[:16])
chna_body = chna_override if chna_override is not None else read_body(path, chunks, "chna")
n_track, n_uid, rows = parse_chna(chna_body)
if f_ch != n_track:
fail(msgs, f"Mismatched number of audio channels and chna entries "
f"(fmt={f_ch} chna={n_track})")
ax_raw = axml_override if axml_override is not None else read_body(path, chunks, "axml")
ax = ax_raw.decode("utf-8")
# --- audioObjectID 十六进制通道输入号解码 ---
objs = re.findall(r'audioObjectID="(AO_[0-9a-zA-Z]+)"', ax)
bed_ch, obj_ch = [], []
for ao in objs:
cid = decode_channel_input(ao)
if cid is None:
fail(msgs, f"Invalid ADM BWF XML format: cannot decode channel "
f"input ID from AudioObjectID '{ao}'")
continue
if ao == "AO_1001":
bed_ch.append(cid)
else:
if cid <= 10:
fail(msgs, f"Source channel index should be greater than 10 "
f"for objects ('{ao}' -> {cid})")
obj_ch.append(cid)
# UID 十六进制 → 必须与 chna 表一致
uid_map = {uid: trk for trk, uid, tf, pk in rows}
for uid in re.findall(r'UID="(ATU_[0-9a-zA-Z]+)"', ax):
if uid not in uid_map:
fail(msgs, f"'{uid}' is not referenced in 'chna' chunk UID table")
continue
m = re.fullmatch(r"ATU_([0-9a-fA-F]+)", uid)
if m:
v = int(m.group(1), 16)
if v > 128:
fail(msgs, f"Channel index out of range (UID {uid} -> {v})")
# 轨数一致性:axml audioTrackUID 数 == fmt 声道数
n_tu = len(re.findall(r"<audioTrackUID ", ax))
if n_tu != f_ch:
fail(msgs, f"Number of channels declared in ADM ({n_tu}) does not "
f"match 'fmt ' chunk ({f_ch})")
# sampleRate / bitDepth 一致
for sr in set(re.findall(r'sampleRate="(\d+)"', ax)):
if int(sr) != f_rate:
fail(msgs, f"Mismatched track sample rate between ADM and WAV ({sr} vs {f_rate})")
for bd in set(re.findall(r'bitDepth="(\d+)"', ax)):
if int(bd) != f_bits:
fail(msgs, f"Mismatched track bit depth between ADM and WAV ({bd} vs {f_bits})")
# audioProgramme 唯一性 / audioContent ≥1
if ax.count("<audioProgramme ") != 1:
fail(msgs, "ADM has more than one audioProgramme object -- there must be only one"
if ax.count("<audioProgramme ") > 1
else "ADM does not have a required audioProgramme object")
if "<audioContent " not in ax:
fail(msgs, "audioProgramme object does not have a required audioContent object")
# 每个 channelFormat ≥1 blockFormat + 对象块链连续性
cfs = re.findall(r'<audioChannelFormat [^>]*typeLabel="0003".*?</audioChannelFormat>', ax, re.S)
object_block_formats = 0
for seg in cfs:
cf_id = re.search(r'audioChannelFormatID="([^"]+)"', seg).group(1)
block_xml = re.findall(r'<audioBlockFormat [^>]*rtime="[^"]+".*?</audioBlockFormat>',
seg, re.S)
object_block_formats += len(block_xml)
blocks = []
for block_index, block in enumerate(block_xml, 1):
timing = re.search(r'rtime="([^"]+)" duration="([^"]+)"', block)
if timing is None:
continue
rtime, duration = timing.groups()
blocks.append((rtime, duration))
jump = re.search(
r'<jumpPosition interpolationLength="([^"]+)">1</jumpPosition>', block)
if jump is not None and float(jump.group(1)) > ts_sec(duration) + 1e-8:
fail(msgs, f"Interpolation length exceeds duration in block format "
f"{block_index} of {cf_id}: {jump.group(1)} > {duration}")
if not blocks:
fail(msgs, f"AudioChannelFormat {cf_id} is missing audioBlockFormat sub-element")
continue
for i in range(len(blocks) - 1):
end_i = ts_sec(blocks[i][0]) + ts_sec(blocks[i][1])
nxt = ts_sec(blocks[i + 1][0])
if abs(end_i - nxt) > 2e-5:
fail(msgs, f"Time gap between block format {i+1} and {i+2} of {cf_id}: "
f"{end_i:.5f} vs {nxt:.5f}")
return msgs, dict(fmt_ch=f_ch, fmt_rate=f_rate, fmt_bits=f_bits,
chna=n_track, objects=len(obj_ch), bed=len(bed_ch),
trackUIDs=n_tu, axml_bytes=len(ax_raw),
audioBlockFormats=ax.count("<audioBlockFormat "),
objectBlockFormats=object_block_formats)
def main():
import argparse
ap = argparse.ArgumentParser()
ap.add_argument("files", nargs="+")
ap.add_argument("--axml-file", default=None, help="用该文件内容替换 wav 内 axml(对照实验)")
ap.add_argument("--chna-file", default=None, help="用该文件内容替换 wav 内 chna(对照实验)")
a = ap.parse_args()
ax_o = open(a.axml_file, "rb").read() if a.axml_file else None
ch_o = open(a.chna_file, "rb").read() if a.chna_file else None
rc = 0
for p in a.files:
r = validate(p, axml_override=ax_o, chna_override=ch_o)
name = os.path.basename(p)
if isinstance(r, list):
msgs, info = r, {}
else:
msgs, info = r
if msgs:
rc = 1
print(f"[FAIL] {name}")
for m in msgs:
print(" -", m)
else:
print(f"[PASS] {name} {info}")
return rc
if __name__ == "__main__":
sys.exit(main())
-249
View File
@@ -1,249 +0,0 @@
#include "joc_core.h"
#include <cstring>
#include <string>
#include "eac3_transport/eac3_reader.h"
#include "emdf/emdf_parser.h"
#include "foundation/status.h"
#include "joc_bitstream/joc_parser.h"
namespace {
thread_local std::string g_detail;
joc_error finish(const joc::Status& status) {
if (status.ok()) {
g_detail.clear();
return JOC_OK;
}
g_detail.assign(status.stage());
g_detail.append(": ");
g_detail.append(status.message());
return status.code();
}
joc_error arg_fail(const char* message) {
return finish(joc::Status::fail(JOC_ERR_INVALID_ARGUMENT, "core", message));
}
void fill_emdf_info(const joc::emdf::Container& container, joc_emdf_info* out) {
std::memset(out, 0, sizeof(*out));
out->struct_size = sizeof(joc_emdf_info);
out->struct_version = JOC_EMDF_INFO_VERSION;
out->start_bit = static_cast<std::uint32_t>(container.start_bit);
out->container_bytes = static_cast<std::uint32_t>(container.raw_size);
out->payload_count = static_cast<std::uint32_t>(container.payload_count);
for (std::size_t i = 0; i < container.payload_count; ++i) {
out->payloads[i].id = container.payloads[i].id;
out->payloads[i].sample_offset = container.payloads[i].sample_offset;
out->payloads[i].bit_offset = static_cast<std::uint32_t>(container.payloads[i].bit_offset);
out->payloads[i].size = static_cast<std::uint32_t>(container.payloads[i].size);
}
}
} // namespace
extern "C" {
std::uint32_t JOC_CALL joc_abi_version(void) { return JOC_ABI_VERSION; }
const char* JOC_CALL joc_version_string(void) { return "0.1.0-m1"; }
std::uint32_t JOC_CALL joc_event_size(void) { return static_cast<std::uint32_t>(sizeof(joc_event)); }
std::uint32_t JOC_CALL joc_task_config_size(void) {
return static_cast<std::uint32_t>(sizeof(joc_task_config));
}
std::uint32_t JOC_CALL joc_task_result_size(void) {
return static_cast<std::uint32_t>(sizeof(joc_task_result));
}
const char* JOC_CALL joc_build_info(void) {
static const std::string info = [] {
std::string text = "joc_core 0.1.0-m1 (";
#if defined(_MSC_VER)
text += "msvc " + std::to_string(_MSC_VER);
#elif defined(__clang__)
text += std::string("clang ") + __clang_version__;
#elif defined(__GNUC__)
text += "gcc " + std::to_string(__GNUC__) + "." + std::to_string(__GNUC_MINOR__);
#else
text += "unknown-compiler";
#endif
#if defined(_M_AMD64) || defined(__x86_64__)
text += ", x64";
#elif defined(_M_ARM64) || defined(__aarch64__)
text += ", arm64";
#elif defined(_M_IX86) || defined(__i386__)
text += ", x86";
#endif
text += ", c++";
text += std::to_string(static_cast<long long>(__cplusplus / 100 % 100));
text += ")";
return text;
}();
return info.c_str();
}
const char* JOC_CALL joc_error_name(joc_error code) {
switch (code) {
case JOC_OK: return "JOC_OK";
case JOC_ERR_INVALID_ARGUMENT: return "JOC_ERR_INVALID_ARGUMENT";
case JOC_ERR_INVALID_CONFIG: return "JOC_ERR_INVALID_CONFIG";
case JOC_ERR_OUT_OF_MEMORY: return "JOC_ERR_OUT_OF_MEMORY";
case JOC_ERR_IO: return "JOC_ERR_IO";
case JOC_ERR_UNSUPPORTED_PLATFORM: return "JOC_ERR_UNSUPPORTED_PLATFORM";
case JOC_ERR_LIBRARY_MISSING: return "JOC_ERR_LIBRARY_MISSING";
case JOC_ERR_INPUT_NOT_FOUND: return "JOC_ERR_INPUT_NOT_FOUND";
case JOC_ERR_INPUT_FORMAT: return "JOC_ERR_INPUT_FORMAT";
case JOC_ERR_EAC3_SYNCFRAME: return "JOC_ERR_EAC3_SYNCFRAME";
case JOC_ERR_EMDF_TRANSPORT: return "JOC_ERR_EMDF_TRANSPORT";
case JOC_ERR_EMDF_SYNTAX: return "JOC_ERR_EMDF_SYNTAX";
case JOC_ERR_JOC_SYNTAX: return "JOC_ERR_JOC_SYNTAX";
case JOC_ERR_JOC_UNSUPPORTED_VARIANT: return "JOC_ERR_JOC_UNSUPPORTED_VARIANT";
case JOC_ERR_OAMD_SYNTAX: return "JOC_ERR_OAMD_SYNTAX";
case JOC_ERR_OAMD_UNSUPPORTED_VARIANT: return "JOC_ERR_OAMD_UNSUPPORTED_VARIANT";
case JOC_ERR_BITSTREAM_TRUNCATED: return "JOC_ERR_BITSTREAM_TRUNCATED";
case JOC_ERR_BITSTREAM_PADDING: return "JOC_ERR_BITSTREAM_PADDING";
case JOC_ERR_HRTF_NOT_FOUND: return "JOC_ERR_HRTF_NOT_FOUND";
case JOC_ERR_HRTF_FORMAT: return "JOC_ERR_HRTF_FORMAT";
case JOC_ERR_HRTF_VERSION: return "JOC_ERR_HRTF_VERSION";
case JOC_ERR_HRTF_HASH: return "JOC_ERR_HRTF_HASH";
case JOC_ERR_HRTF_UNSUPPORTED_CONVENTION: return "JOC_ERR_HRTF_UNSUPPORTED_CONVENTION";
case JOC_ERR_LAYOUT_UNSUPPORTED: return "JOC_ERR_LAYOUT_UNSUPPORTED";
case JOC_ERR_RENDER_FAILED: return "JOC_ERR_RENDER_FAILED";
case JOC_ERR_OUTPUT_OPEN: return "JOC_ERR_OUTPUT_OPEN";
case JOC_ERR_OUTPUT_WRITE: return "JOC_ERR_OUTPUT_WRITE";
case JOC_ERR_OUTPUT_CLIP_ABORT: return "JOC_ERR_OUTPUT_CLIP_ABORT";
case JOC_ERR_ADM_VALIDATION: return "JOC_ERR_ADM_VALIDATION";
case JOC_ERR_CANCELLED: return "JOC_ERR_CANCELLED";
case JOC_ERR_STATE: return "JOC_ERR_STATE";
case JOC_ERR_NOT_SUPPORTED: return "JOC_ERR_NOT_SUPPORTED";
case JOC_ERR_INTERNAL: return "JOC_ERR_INTERNAL";
default: return "JOC_ERR_UNKNOWN";
}
}
const char* JOC_CALL joc_error_stage(joc_error code) {
switch (code) {
case JOC_ERR_EAC3_SYNCFRAME:
case JOC_ERR_INPUT_NOT_FOUND:
case JOC_ERR_INPUT_FORMAT:
return "eac3_transport";
case JOC_ERR_EMDF_TRANSPORT:
case JOC_ERR_EMDF_SYNTAX:
return "emdf";
case JOC_ERR_JOC_SYNTAX:
case JOC_ERR_JOC_UNSUPPORTED_VARIANT:
return "joc";
case JOC_ERR_OAMD_SYNTAX:
case JOC_ERR_OAMD_UNSUPPORTED_VARIANT:
return "oamd";
case JOC_ERR_BITSTREAM_TRUNCATED:
case JOC_ERR_BITSTREAM_PADDING:
return "bitstream";
case JOC_ERR_HRTF_NOT_FOUND:
case JOC_ERR_HRTF_FORMAT:
case JOC_ERR_HRTF_VERSION:
case JOC_ERR_HRTF_HASH:
case JOC_ERR_HRTF_UNSUPPORTED_CONVENTION:
return "hrtf";
case JOC_ERR_LAYOUT_UNSUPPORTED:
case JOC_ERR_RENDER_FAILED:
return "render";
case JOC_ERR_OUTPUT_OPEN:
case JOC_ERR_OUTPUT_WRITE:
case JOC_ERR_OUTPUT_CLIP_ABORT:
case JOC_ERR_ADM_VALIDATION:
return "output";
case JOC_ERR_CANCELLED:
case JOC_ERR_STATE:
case JOC_ERR_NOT_SUPPORTED:
case JOC_ERR_INTERNAL:
return "task";
default:
return "core";
}
}
const char* JOC_CALL joc_last_error_detail(void) { return g_detail.c_str(); }
joc_error JOC_CALL joc_parse_id14(const std::uint8_t* payload, std::size_t payload_size,
joc_frame_params* out_params) {
if (payload == nullptr || out_params == nullptr || payload_size == 0) {
return arg_fail("joc_parse_id14 requires a non-empty payload and an output struct");
}
return finish(joc::joc::parse_id14(payload, payload_size, out_params, nullptr));
}
joc_error JOC_CALL joc_parse_eac3_frame(const std::uint8_t* frame, std::size_t frame_size,
joc_frame_params* out_params, joc_emdf_info* out_emdf) {
if (frame == nullptr || out_params == nullptr || frame_size == 0) {
return arg_fail("joc_parse_eac3_frame requires a frame and an output struct");
}
joc::emdf::Container container;
const joc::Status status =
joc::joc::parse_eac3_frame(frame, frame_size, out_params, &container, nullptr);
if (!status.ok()) {
return finish(status);
}
if (out_emdf != nullptr) {
fill_emdf_info(container, out_emdf);
}
return JOC_OK;
}
joc_error JOC_CALL joc_extract_payload(const std::uint8_t* frame, std::size_t frame_size,
const joc_emdf_payload_info* payload, std::uint8_t* out,
std::size_t out_capacity, std::size_t* out_size) {
if (frame == nullptr || payload == nullptr || out_size == nullptr) {
return arg_fail("joc_extract_payload requires frame, payload and out_size");
}
*out_size = payload->size;
if (out == nullptr) {
return JOC_OK;
}
if (out_capacity < payload->size) {
return arg_fail("joc_extract_payload output buffer too small");
}
joc::emdf::Payload entry;
entry.id = payload->id;
entry.sample_offset = payload->sample_offset;
entry.bit_offset = payload->bit_offset;
entry.size = payload->size;
std::vector<std::uint8_t> bytes;
const joc::Status status = joc::emdf::extract_payload_bytes(frame, frame_size, entry, &bytes);
if (!status.ok()) {
return finish(status);
}
if (!bytes.empty()) {
std::memcpy(out, bytes.data(), bytes.size());
}
*out_size = bytes.size();
return JOC_OK;
}
joc_error JOC_CALL joc_eac3_frame_bytes(const std::uint8_t* data, std::size_t size, std::size_t offset,
std::size_t* out_frame_bytes) {
if (data == nullptr || out_frame_bytes == nullptr) {
return arg_fail("joc_eac3_frame_bytes requires data and out_frame_bytes");
}
const joc_error code = joc::eac3::FrameReader::frame_bytes(data, size, offset, out_frame_bytes);
if (code != JOC_OK) {
return finish(joc::Status::fail(code, joc::stage::kEac3,
"invalid or truncated E-AC-3 syncframe at byte " +
std::to_string(offset)));
}
return JOC_OK;
}
joc_error JOC_CALL joc_check_id14_padding(const std::uint8_t* payload, std::size_t payload_size,
std::uint32_t* out_trailing_bits) {
if (payload == nullptr || payload_size == 0) {
return arg_fail("joc_check_id14_padding requires a payload");
}
return finish(joc::joc::check_id14_padding(payload, payload_size, out_trailing_bits));
}
} // extern "C"
-162
View File
@@ -1,162 +0,0 @@
#include <cstring>
#include <string>
#include "foundation/status.h"
#include "joc_core.h"
#include "joc_stream.h"
#include "stream/stream.h"
struct joc_stream {
joc::stream::Stream instance;
};
namespace {
joc_error finish_stream(const joc::Status& status) {
return status.code();
}
joc::stream::Config to_config(const joc_stream_config& config) {
joc::stream::Config out;
out.input = config.input;
out.output = config.output;
if (config.speaker_layout_name != nullptr) { out.layout = config.speaker_layout_name; }
if (config.speaker_metadata_offset != 0u) {
out.metadata_offset = config.speaker_metadata_offset;
}
out.binaural_mode = config.binaural_mode != 0u ? config.binaural_mode : JOC_BINAURAL_MID;
if (config.hrtf_path != nullptr) { out.hrtf_path = config.hrtf_path; }
if (config.kernels_path != nullptr) { out.kernels_path = config.kernels_path; }
if (config.binaural_tail_seconds > 0.0) { out.tail_seconds = config.binaural_tail_seconds; }
if (config.object_delay_samples != 0u) {
out.object_delay_samples = config.object_delay_samples;
}
out.gain_db = config.gain_db;
out.native_threads = config.native_threads;
// The binaural HRTF inputs of joc_task_config, copied with the same defaults:
// the policy is taken verbatim (0 is "none", a real choice, not "unset") and
// the radius keeps its documented default of 1.0 when the field is not set.
if (config.hrtf_sofa_path != nullptr) { out.hrtf_sofa_path = config.hrtf_sofa_path; }
if (config.personalized_headphone_path != nullptr) {
out.personalized_headphone_path = config.personalized_headphone_path;
}
if (config.hrtf_cache_dir != nullptr) { out.hrtf_cache_dir = config.hrtf_cache_dir; }
out.hrtf_cache_policy = config.hrtf_cache_policy;
if (config.hrtf_radius_m > 0.0) { out.hrtf_radius_m = config.hrtf_radius_m; }
return out;
}
} // namespace
extern "C" {
joc_error JOC_CALL joc_stream_create(const joc_stream_config* config, joc_stream** out) {
if (config == nullptr || out == nullptr || config->struct_size != sizeof(joc_stream_config)) {
return JOC_ERR_INVALID_ARGUMENT;
}
auto* stream = new (std::nothrow) joc_stream();
if (stream == nullptr) {
return JOC_ERR_OUT_OF_MEMORY;
}
const joc::Status status = stream->instance.create(to_config(*config));
if (!status.ok()) {
delete stream;
return finish_stream(status);
}
*out = stream;
return JOC_OK;
}
joc_error JOC_CALL joc_stream_push(joc_stream* stream, const joc_stream_buffer* input,
std::uint32_t* consumed_samples, std::uint32_t* consumed_bytes) {
if (stream == nullptr || input == nullptr ||
input->struct_size != sizeof(joc_stream_buffer)) {
return JOC_ERR_INVALID_ARGUMENT;
}
const joc::Status status = [&] {
if (input->kind == JOC_STREAM_IN_EAC3) {
std::size_t consumed = 0;
const joc::Status pushed =
stream->instance.push_eac3(input->bytes, input->byte_count, &consumed);
if (consumed_bytes != nullptr) {
*consumed_bytes = static_cast<std::uint32_t>(consumed);
}
return pushed;
}
if (input->kind == JOC_STREAM_IN_PCM_OBJECTS16) {
std::size_t consumed = 0;
const joc::Status pushed =
stream->instance.push_objects16(input->pcm, input->sample_count, &consumed);
if (consumed_samples != nullptr) {
*consumed_samples = static_cast<std::uint32_t>(consumed);
}
return pushed;
}
std::size_t consumed = 0;
const joc::Status pushed =
stream->instance.push_bed(input->pcm, input->sample_count, &consumed);
if (consumed_samples != nullptr) {
*consumed_samples = static_cast<std::uint32_t>(consumed);
}
return pushed; }();
return finish_stream(status);
}
joc_error JOC_CALL joc_stream_pull(joc_stream* stream, joc_stream_buffer* output,
std::uint32_t* produced_samples) {
if (stream == nullptr || output == nullptr || output->out_pcm == nullptr ||
output->struct_size != sizeof(joc_stream_buffer)) {
return JOC_ERR_INVALID_ARGUMENT;
}
std::size_t produced = 0;
const joc::Status status =
stream->instance.pull(output->out_pcm, output->sample_count, &produced);
if (!status.ok()) {
return finish_stream(status);
}
if (produced_samples != nullptr) {
*produced_samples = static_cast<std::uint32_t>(produced);
}
output->sample_count = static_cast<std::uint32_t>(produced);
return JOC_OK;
}
joc_error JOC_CALL joc_stream_flush(joc_stream* stream) {
if (stream == nullptr) {
return JOC_ERR_INVALID_ARGUMENT;
}
return finish_stream(stream->instance.flush());
}
joc_error JOC_CALL joc_stream_reset(joc_stream* stream) {
if (stream == nullptr) {
return JOC_ERR_INVALID_ARGUMENT;
}
return finish_stream(stream->instance.reset());
}
joc_error JOC_CALL joc_stream_status(const joc_stream* stream, joc_stream_status_info* out) {
if (stream == nullptr || out == nullptr ||
out->struct_size != sizeof(joc_stream_status_info)) {
return JOC_ERR_INVALID_ARGUMENT;
}
const joc::stream::Info& info = stream->instance.info();
out->frames_in = info.frames_in;
out->frames_out = info.frames_out;
out->samples_in = info.samples_in;
out->samples_out = info.samples_out;
out->bytes_in = info.bytes_in;
out->buffered_samples = stream->instance.buffered_samples();
out->oamd_payloads = info.oamd_payloads;
out->oamd_transitions = info.oamd_transitions;
out->output_channels = info.output_channels;
out->ended = info.ended;
return JOC_OK;
}
joc_error JOC_CALL joc_stream_destroy(joc_stream* stream) {
delete stream;
return JOC_OK;
}
} // extern "C"
-120
View File
@@ -1,120 +0,0 @@
#include <atomic>
#include <cstring>
#include <new>
#include <string>
#include <vector>
#include "foundation/status.h"
#include "joc_core.h"
#include "task/task.h"
// task and whichever frontend wants to stop it (plan 29.1/29.2).
struct joc_cancel_token {
std::atomic<std::uint32_t> requested{0u};
};
namespace {
thread_local std::string g_task_detail;
joc_error finish_task(const joc::Status& status) {
if (status.ok()) {
g_task_detail.clear();
return JOC_OK;
}
g_task_detail.assign(status.stage());
g_task_detail.append(": ");
g_task_detail.append(status.message());
return status.code();
}
} // namespace
extern "C" {
joc_cancel_token* JOC_CALL joc_cancel_token_create(void) {
return new (std::nothrow) joc_cancel_token();
}
void JOC_CALL joc_cancel_token_request(joc_cancel_token* token) {
if (token != nullptr) {
token->requested.store(1u, std::memory_order_relaxed);
}
}
std::int32_t JOC_CALL joc_cancel_token_is_requested(const joc_cancel_token* token) {
return (token != nullptr && token->requested.load(std::memory_order_relaxed) != 0u) ? 1 : 0;
}
void JOC_CALL joc_cancel_token_destroy(joc_cancel_token* token) { delete token; }
joc_error JOC_CALL joc_task_validate(const joc_task_config* config,
joc_validation_issue* issues, std::uint32_t capacity,
std::uint32_t* count) {
if (config == nullptr) {
return JOC_ERR_INVALID_ARGUMENT;
}
if (config->struct_size != sizeof(joc_task_config)) {
return JOC_ERR_INVALID_ARGUMENT;
}
std::vector<joc_validation_issue> found;
std::uint32_t errors = 0;
const joc::Status status = joc::task::validate(*config, &found, &errors);
if (count != nullptr) {
*count = static_cast<std::uint32_t>(found.size());
}
if (issues != nullptr) {
for (std::uint32_t i = 0; i < capacity && i < found.size(); ++i) {
issues[i] = found[i];
}
}
return status.ok() ? JOC_OK : JOC_ERR_INVALID_CONFIG;
}
joc_error JOC_CALL joc_task_execute(const joc_task_config* config, const joc_event_sink* sink,
joc_task_result* out) {
if (config == nullptr || config->struct_size != sizeof(joc_task_config)) {
return JOC_ERR_INVALID_ARGUMENT;
}
if (out != nullptr) {
std::memset(out, 0, sizeof(*out));
out->struct_size = sizeof(joc_task_result);
out->struct_version = JOC_TASK_RESULT_VERSION;
}
const joc::Status status = joc::task::run(*config, sink, out);
if (!status.ok() && out != nullptr && out->status == 0u) {
out->status = JOC_TASK_FAILED;
out->error_code = static_cast<std::uint32_t>(status.code());
std::snprintf(out->error_stage, sizeof(out->error_stage), "%s", status.stage().c_str());
std::snprintf(out->error_message, sizeof(out->error_message), "%s",
status.message().c_str());
}
g_task_detail = status.ok() ? std::string() : status.message();
return status.code();
}
joc_error JOC_CALL joc_task_result_to_json(const joc_task_result* result, char* buffer,
std::size_t capacity, std::size_t* needed) {
if (result == nullptr) {
return JOC_ERR_INVALID_ARGUMENT;
}
std::string json;
const joc::Status status = joc::task::result_to_json(*result, &json);
if (!status.ok()) {
return status.code();
}
if (needed != nullptr) {
*needed = json.size() + 1u;
}
if (buffer == nullptr) {
return JOC_OK;
}
if (capacity < json.size() + 1u) {
return JOC_ERR_INVALID_ARGUMENT;
}
std::memcpy(buffer, json.c_str(), json.size() + 1u);
return JOC_OK;
}
} // extern "C"
-251
View File
@@ -1,251 +0,0 @@
#include "binaural/binaural_runtime.h"
#include <algorithm>
#include <cmath>
#include <cstring>
#include <string>
namespace joc::binaural {
namespace {
double max_delay_bound(const hrtf::Field& field) {
double maximum = 0.0;
for (const double value : field.delay_bounds) {
maximum = std::max(maximum, value);
}
return maximum;
}
} // namespace
bool profile_from_name(const char* name, Profile* out) {
if (name == nullptr || out == nullptr) {
return false;
}
if (std::strcmp(name, "near") == 0) { *out = Profile::Near; return true; }
if (std::strcmp(name, "mid") == 0) { *out = Profile::Mid; return true; }
if (std::strcmp(name, "far") == 0) { *out = Profile::Far; return true; }
return false;
}
SofaBinauralRuntime::~SofaBinauralRuntime() {
if (handle_ != nullptr) {
ejoc_sofa_binaural_destroy(handle_);
handle_ = nullptr;
}
}
Status SofaBinauralRuntime::open(const hrtf::Field& field, const hrtf::Kernels& kernels,
Profile profile, const RoomConstants& room) {
if (handle_ != nullptr) {
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime already open");
}
if (field.coefficients.size() != static_cast<std::size_t>(hrtf::kShTerms * hrtf::kEars *
hrtf::kHybridBands * 2) ||
field.band_centers_hz.size() != static_cast<std::size_t>(hrtf::kHybridBands)) {
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
"compiled HRTF field has unexpected array sizes");
}
handle_ = ejoc_sofa_binaural_create();
if (handle_ == nullptr) {
return Status::fail(JOC_ERR_OUT_OF_MEMORY, stage::kRender,
"ejoc_sofa_binaural_create failed");
}
profile_ = profile;
if (ejoc_sofa_binaural_configure_kernels(
handle_, kernels.qmf_analysis.data(), kernels.hybrid_low.data(),
kernels.hybrid_indices.data(), kernels.hybrid_values.data(), kernels.hybrid_count,
kernels.qmf_basis.data(), kernels.qmf_taps.data()) != 0) {
const char* message = ejoc_sofa_binaural_last_error(handle_);
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
std::string("configure_kernels failed: ") +
(message != nullptr ? message : "unknown"));
}
if (ejoc_sofa_binaural_configure_field(handle_, field.coefficients.data(),
field.delay_coefficients.data(),
field.delay_bounds.data(), field.band_centers_hz.data(),
field.measurement_radius_m) != 0) {
const char* message = ejoc_sofa_binaural_last_error(handle_);
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
std::string("configure_field failed: ") +
(message != nullptr ? message : "unknown"));
}
if (ejoc_sofa_binaural_configure_room(
handle_, room.dims, room.listener, room.walls, room.speed_of_sound, room.fdn_delays,
room.fdn_feedback, room.damping, room.fdn_output_gain, room.allpass_delays,
room.allpass_gains, room.enable_early_reflections, room.enable_late_room) != 0) {
const char* message = ejoc_sofa_binaural_last_error(handle_);
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
std::string("configure_room failed: ") +
(message != nullptr ? message : "unknown"));
}
maximum_hrtf_delay_ =
static_cast<std::int64_t>(std::ceil(max_delay_bound(field) - 1e-9));
hrtf_history_slots_ = static_cast<std::uint32_t>(std::max<std::int64_t>(1, (maximum_hrtf_delay_ + 63) / 64));
staging_.assign(kBlockSamples * kSourceCount, 0.0);
block_output_.assign(kBlockSamples * 2u, 0.0);
output_.clear();
staged_ = 0;
input_samples_ = 0;
processed_samples_ = 0;
blocks_processed_ = 0;
return Status::success();
}
Status SofaBinauralRuntime::process_block() {
double positions[timeline::kTimelineObjects][3] = {};
const Status queried = timeline_.positions_at(static_cast<std::int64_t>(processed_samples_),
positions);
if (!queried.ok()) {
return queried;
}
// The reference adapter calls set_source without a `fade` argument, so the
// backend default (fade enabled) applies - the per-object path crossfade is
// part of the reference behaviour, not an optional extra.
constexpr std::uint32_t kFade = 1u;
if (ejoc_sofa_binaural_set_source(handle_, 0u, kLfePosition,
static_cast<std::uint32_t>(profile_), 1.0, 1u, 1u,
kFade) != 0) {
const char* message = ejoc_sofa_binaural_last_error(handle_);
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
std::string("set_source(LFE) failed: ") +
(message != nullptr ? message : "unknown"));
}
for (std::uint32_t source = 0; source < timeline::kTimelineObjects; ++source) {
if (ejoc_sofa_binaural_set_source(handle_, source + 1u, positions[source],
static_cast<std::uint32_t>(profile_), 1.0, 1u, 0u,
kFade) != 0) {
const char* message = ejoc_sofa_binaural_last_error(handle_);
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
std::string("set_source(object ") + std::to_string(source + 1u) +
") failed: " + (message != nullptr ? message : "unknown"));
}
}
const int trimmed = ejoc_sofa_binaural_process(handle_, staging_.data(), kBlockSamples, 1.0,
block_output_.data());
if (trimmed < 0) {
const char* message = ejoc_sofa_binaural_last_error(handle_);
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
std::string("sofa process failed: ") +
(message != nullptr ? message : "unknown"));
}
if (trimmed > 0) {
output_.insert(output_.end(), block_output_.begin(),
block_output_.begin() + static_cast<std::ptrdiff_t>(trimmed) * 2);
}
staged_ = 0;
processed_samples_ += kBlockSamples;
++blocks_processed_;
return Status::success();
}
Status SofaBinauralRuntime::submit_frame(const float* objects16_planar,
const oamd::OamdUpdate* update, std::int64_t frame_index,
std::int64_t outer_sample_offset,
std::int64_t object_delay_samples) {
if (handle_ == nullptr) {
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime is not open");
}
if (objects16_planar == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null frame");
}
// A frame without an ID11 payload submits nothing at all (the reference only
if (update != nullptr) {
const Status submitted = timeline_.submit_update(
*update, frame_index * JOC_FRAME_SAMPLES, outer_sample_offset, object_delay_samples,
static_cast<std::int64_t>(input_samples_));
if (!submitted.ok()) {
return submitted;
}
}
std::size_t offset = 0;
while (offset < JOC_FRAME_SAMPLES) {
const std::size_t room = kBlockSamples - staged_;
const std::size_t count = std::min<std::size_t>(room, JOC_FRAME_SAMPLES - offset);
for (std::size_t sample = 0; sample < count; ++sample) {
double* row = staging_.data() + (staged_ + sample) * kSourceCount;
for (std::size_t channel = 0; channel < kSourceCount; ++channel) {
row[channel] = static_cast<double>(
objects16_planar[channel * JOC_FRAME_SAMPLES + offset + sample]);
}
}
staged_ += count;
offset += count;
if (staged_ == kBlockSamples) {
const Status status = process_block();
if (!status.ok()) {
return status;
}
}
}
input_samples_ += JOC_FRAME_SAMPLES;
return Status::success();
}
std::uint32_t SofaBinauralRuntime::finish_capacity(double tail_seconds) const {
std::int64_t requested = tail_samples_;
if (tail_seconds >= 0.0) {
requested = static_cast<std::int64_t>(std::ceil(tail_seconds * 48000.0 - 1e-9));
}
const std::int64_t hrtf_bound = static_cast<std::int64_t>(hrtf_history_slots_) * 64;
const std::int64_t early_bound = hrtf_bound + 2048 + 256 * 64;
std::int64_t drain = std::max(requested, early_bound) + 961;
drain = ((drain + 63) / 64) * 64;
return static_cast<std::uint32_t>(drain);
}
Status SofaBinauralRuntime::reset() {
if (handle_ == nullptr) {
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime is not open");
}
if (ejoc_sofa_binaural_reset(handle_) != 0) {
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender, "sofa reset failed");
}
timeline_ = timeline::OamdPositionTimeline();
output_.clear();
staged_ = 0;
input_samples_ = 0;
processed_samples_ = 0;
blocks_processed_ = 0;
return Status::success();
}
Status SofaBinauralRuntime::finish(std::uint32_t flush_samples, std::vector<double>* out) {
if (handle_ == nullptr) {
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime is not open");
}
if (out == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null output");
}
out->clear();
if (flush_samples == 0u) {
return Status::success();
}
std::vector<double> chunk(static_cast<std::size_t>(kBlockSamples) * 2u, 0.0);
std::uint32_t produced_total = 0;
std::uint32_t remaining = flush_samples;
while (remaining > 0) {
const std::uint32_t request = std::min<std::uint32_t>(remaining, kBlockSamples);
const int produced = ejoc_sofa_binaural_finish(handle_, request, chunk.data(), request);
if (produced < 0) {
const char* message = ejoc_sofa_binaural_last_error(handle_);
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
std::string("sofa finish failed: ") +
(message != nullptr ? message : "unknown"));
}
if (produced == 0) {
break;
}
out->insert(out->end(), chunk.begin(),
chunk.begin() + static_cast<std::ptrdiff_t>(produced) * 2);
produced_total += static_cast<std::uint32_t>(produced);
remaining -= std::min<std::uint32_t>(remaining, static_cast<std::uint32_t>(produced));
}
return Status::success();
}
} // namespace joc::binaural
-96
View File
@@ -1,96 +0,0 @@
#pragma once
#include <cstdint>
#include <vector>
#include "eac3joc_core.h"
#include "foundation/status.h"
#include "hrtf/jochrtf.h"
#include "oamd/oamd_parser.h"
#include "timeline/position_timeline.h"
namespace joc::binaural {
// ADM direction of the LFE source, as the reference passes it.
inline constexpr double kLfePosition[3] = {0.0, 1.0, 0.0};
inline constexpr int kSourceCount = 16;
inline constexpr std::uint32_t kBlockSamples = 512;
// Room constants the reference's native bridge passes for the default shoebox.
struct RoomConstants {
double dims[3] = {18.0, 18.0, 14.0};
double listener[3] = {9.0, 9.0, 7.0};
double walls[6] = {0.62, 0.60, 0.58, 0.61, 0.52, 0.56};
double speed_of_sound = 343.3;
std::uint32_t fdn_delays[4] = {1427u, 1783u, 1973u, 2099u};
double fdn_feedback[4] = {0.7853685923259284, 0.7394299865898056, 0.7160221718631921,
0.7009092068085467};
double damping = 0.32;
double fdn_output_gain = 0.22;
std::uint32_t allpass_delays[2] = {113u, 331u};
double allpass_gains[2] = {0.63, 0.51};
std::uint32_t enable_early_reflections = 1;
std::uint32_t enable_late_room = 1;
};
enum class Profile : std::uint32_t { Near = 0, Mid = 1, Far = 2 };
bool profile_from_name(const char* name, Profile* out);
class SofaBinauralRuntime {
public:
SofaBinauralRuntime() = default;
~SofaBinauralRuntime();
SofaBinauralRuntime(const SofaBinauralRuntime&) = delete;
SofaBinauralRuntime& operator=(const SofaBinauralRuntime&) = delete;
Status open(const hrtf::Field& field, const hrtf::Kernels& kernels, Profile profile,
const RoomConstants& room = RoomConstants{});
Status submit_frame(const float* objects16_planar, const oamd::OamdUpdate* update,
std::int64_t frame_index, std::int64_t outer_sample_offset,
std::int64_t object_delay_samples);
// Resets the kernel, the timeline and the counters (plan 31.2).
Status reset();
// Drains the room tail. `flush_samples` is the drain length; the reference
Status finish(std::uint32_t flush_samples, std::vector<double>* out);
// Program output (input minus the 961-sample kernel latency), interleaved.
const std::vector<double>& output() const { return output_; }
void take_output(std::vector<double>* out) {
out->swap(output_);
output_.clear();
}
std::uint64_t input_samples() const { return input_samples_; }
std::uint64_t blocks_processed() const { return blocks_processed_; }
std::size_t staged_samples() const { return staged_; }
const timeline::OamdPositionTimeline& timeline() const { return timeline_; }
// finish_output_capacity as the reference computes it (plan 21.4).
std::uint32_t finish_capacity(double tail_seconds) const;
private:
Status process_block();
ejoc_sofa_binaural_handle handle_ = nullptr;
Profile profile_ = Profile::Mid;
timeline::OamdPositionTimeline timeline_;
std::vector<double> staging_;
std::size_t staged_ = 0;
std::vector<double> block_output_;
std::vector<double> output_;
std::uint64_t input_samples_ = 0;
std::uint64_t processed_samples_ = 0;
std::uint64_t blocks_processed_ = 0;
std::int64_t maximum_hrtf_delay_ = 0;
std::uint32_t hrtf_history_slots_ = 1;
std::uint32_t tail_samples_ = 61200;
};
} // namespace joc::binaural
+184
View File
@@ -0,0 +1,184 @@
"""Direct ID11/OAMD position scheduling for the binaural render path."""
from __future__ import annotations
from dataclasses import dataclass
import numpy as np
from adm_atmos import q_to_adm_xyz
from oamd_bits import JocFieldState, frame_update
from oamd_tracks import align_metadata_sample
from variant_error import UnsupportedVariantError
@dataclass(frozen=True)
class PositionTransition:
start_sample: int
duration_samples: int
origin: np.ndarray
target: np.ndarray
@property
def end_sample(self) -> int:
return self.start_sample + self.duration_samples
class _ObjectPositionTrack:
def __init__(self):
self.initial = np.zeros(3, dtype=np.float64)
self.last_target = self.initial.copy()
self.transitions: list[PositionTransition] = []
self.cursor = 0
self.last_query_sample = -1
def set_initial(self, position):
target = np.asarray(position, dtype=np.float64)
self.initial = target.copy()
self.last_target = target.copy()
def append(self, start_sample: int, duration_samples: int, target,
object_index: int):
start = int(start_sample)
duration = int(duration_samples)
if start < 0 or duration < 0:
raise ValueError("position transition timing must be non-negative")
target = np.asarray(target, dtype=np.float64)
if self.transitions:
previous = self.transitions[-1]
if start < previous.end_sample:
raise UnsupportedVariantError(
"oamd", "overlapping_binaural_position_ramps",
"同一对象的新位置更新在上一双耳 ramp 完成前到达",
details={
"object": object_index,
"ramp_start_sample": previous.start_sample,
"ramp_end_sample": previous.end_sample,
"next_update_sample": start,
})
if start == previous.start_sample and previous.duration_samples == 0:
self.transitions[-1] = PositionTransition(
start, duration, previous.origin.copy(), target.copy())
self.last_target = target.copy()
return
self.transitions.append(PositionTransition(
start, duration, self.last_target.copy(), target.copy()))
self.last_target = target.copy()
def position_at(self, sample: int) -> np.ndarray:
sample = int(sample)
if sample < self.last_query_sample:
raise ValueError("binaural metadata positions must be queried monotonically")
self.last_query_sample = sample
while self.cursor < len(self.transitions):
transition = self.transitions[self.cursor]
if sample < transition.end_sample:
break
self.initial = transition.target.copy()
self.cursor += 1
if self.cursor >= len(self.transitions):
return self.initial
transition = self.transitions[self.cursor]
if sample < transition.start_sample:
return self.initial
if transition.duration_samples == 0:
return transition.target
amount = (sample - transition.start_sample) / float(transition.duration_samples)
return transition.origin + (transition.target - transition.origin) * amount
class OamdPositionTimeline:
"""Convert OAMD state updates into a sample-timed Cartesian trajectory."""
def __init__(self, object_count: int = 15):
if object_count != 15:
raise ValueError("JOC OAMD currently requires 15 object slots")
self.object_count = int(object_count)
self.state = JocFieldState()
self.tracks = [_ObjectPositionTrack() for _ in range(self.object_count)]
self.initialized = False
self.previous_targets: list[tuple[float, float, float] | None] = [
None] * self.object_count
self.payload_count = 0
self.transition_count = 0
self.last_coded_event_sample = -1
def _targets(self) -> list[tuple[float, float, float]]:
q = self.state.q
return [
q_to_adm_xyz(
q[(object_index, "q1")],
q[(object_index, "q2")],
q[(object_index, "q3")],
)
for object_index in range(1, self.object_count + 1)
]
def submit_update(self, update: dict, *, frame_start_sample: int,
outer_sample_offset: int = 0,
object_delay_samples: int = 1473,
processed_sample: int = 0):
"""Schedule one already-parsed :func:`oamd_bits.frame_update` result."""
frame_start = int(frame_start_sample)
outer_offset = int(outer_sample_offset)
object_delay = int(object_delay_samples)
if min(frame_start, outer_offset, object_delay) < 0:
raise ValueError("OAMD frame, outer offset, and object delay must be non-negative")
self.state.apply(update["values"])
targets = self._targets()
coded_event = (
frame_start + outer_offset + int(update["block_offset_samples"]))
if coded_event < self.last_coded_event_sample:
raise UnsupportedVariantError(
"oamd", "non_monotonic_binaural_updates",
"双耳 OAMD 更新时间倒退",
details={
"event_sample": coded_event,
"previous_event_sample": self.last_coded_event_sample,
})
self.last_coded_event_sample = coded_event
if not self.initialized:
if int(processed_sample) > 0:
raise UnsupportedVariantError(
"oamd", "late_initial_binaural_state",
"首个 OAMD 状态在双耳 PCM 已处理后才出现,无法回填 sample 0",
details={
"processed_sample": int(processed_sample),
"first_event_sample": coded_event,
})
for index, target in enumerate(targets):
self.tracks[index].set_initial(target)
self.previous_targets[index] = target
self.initialized = True
self.payload_count += 1
return
effective_ramp = max(0, int(update["ramp_duration_samples"]))
transition_start = align_metadata_sample(coded_event + object_delay)
for index, target in enumerate(targets):
if self.previous_targets[index] == target:
continue
self.tracks[index].append(
transition_start, effective_ramp, target, index + 1)
self.previous_targets[index] = target
self.transition_count += 1
self.payload_count += 1
def submit_payload(self, payload, *, frame_start_sample: int,
outer_sample_offset: int = 0,
object_delay_samples: int = 1473,
processed_sample: int = 0):
update = frame_update(payload)
self.submit_update(
update,
frame_start_sample=frame_start_sample,
outer_sample_offset=outer_sample_offset,
object_delay_samples=object_delay_samples,
processed_sample=processed_sample,
)
return update
def positions_at(self, sample: int) -> np.ndarray:
return np.stack(
[track.position_at(sample) for track in self.tracks], axis=0
).astype(np.float64, copy=False)
+214
View File
@@ -0,0 +1,214 @@
"""ctypes bridge for the native float64 binaural DSP."""
from __future__ import annotations
import ctypes
from pathlib import Path
import numpy as np
from native_renderer import ABI_VERSION, find_native_library
from rosella_filterbank import DEFAULT_KERNEL_DATA, load_kernel_tables
from rosella_model import RosellaModel
BLOCK_SAMPLES = 512
INPUT_CHANNELS = 16
OUTPUT_CHANNELS = 2
HYBRID_BANDS = 77
class NativeBinauralDsp:
def __init__(self, model: RosellaModel, *, library_path=None,
kernel_data: str | Path = DEFAULT_KERNEL_DATA):
self.library_path = find_native_library(library_path)
self._lib = ctypes.CDLL(str(self.library_path))
self._bind()
version = int(self._lib.ejoc_abi_version())
if version != ABI_VERSION:
raise RuntimeError(
f"native ABI mismatch: expected {ABI_VERSION}, got {version}")
self._handle = self._lib.ejoc_binaural_renderer_create()
if not self._handle:
raise RuntimeError("native binaural renderer creation failed")
try:
self._configure_kernels(kernel_data)
self._configure_room(model)
except Exception:
self.close()
raise
def _bind(self):
void_p = ctypes.c_void_p
f64_p = ctypes.POINTER(ctypes.c_double)
i16_p = ctypes.POINTER(ctypes.c_int16)
u32_p = ctypes.POINTER(ctypes.c_uint32)
self._lib.ejoc_abi_version.argtypes = []
self._lib.ejoc_abi_version.restype = ctypes.c_uint32
self._lib.ejoc_binaural_renderer_create.argtypes = []
self._lib.ejoc_binaural_renderer_create.restype = void_p
self._lib.ejoc_binaural_renderer_destroy.argtypes = [void_p]
self._lib.ejoc_binaural_renderer_destroy.restype = None
self._lib.ejoc_binaural_renderer_reset.argtypes = [void_p]
self._lib.ejoc_binaural_renderer_reset.restype = ctypes.c_int
self._lib.ejoc_binaural_renderer_last_error.argtypes = [void_p]
self._lib.ejoc_binaural_renderer_last_error.restype = ctypes.c_char_p
self._lib.ejoc_binaural_renderer_configure_kernels.argtypes = [
void_p, f64_p, f64_p, i16_p, f64_p, ctypes.c_uint32, f64_p, f64_p]
self._lib.ejoc_binaural_renderer_configure_kernels.restype = ctypes.c_int
self._lib.ejoc_binaural_renderer_configure_room.argtypes = [
void_p, ctypes.c_uint32, ctypes.c_uint32, u32_p, f64_p,
u32_p, f64_p, ctypes.c_uint32, f64_p, f64_p, f64_p,
ctypes.c_uint32, u32_p, f64_p, f64_p]
self._lib.ejoc_binaural_renderer_configure_room.restype = ctypes.c_int
self._lib.ejoc_binaural_renderer_process.argtypes = [
void_p, f64_p, f64_p, f64_p, ctypes.c_double, f64_p]
self._lib.ejoc_binaural_renderer_process.restype = ctypes.c_int
def _raise(self, operation, status):
message = self._lib.ejoc_binaural_renderer_last_error(self._handle)
detail = (message or b"").decode("utf-8", "replace")
raise RuntimeError(
f"native binaural renderer {operation} failed ({status}): {detail}")
@staticmethod
def _f64_pointer(values):
return values.ctypes.data_as(ctypes.POINTER(ctypes.c_double))
def _configure_kernels(self, kernel_data):
tables = load_kernel_tables(kernel_data)
qmf_analysis = np.ascontiguousarray(
tables["qmf_analysis_coefficients"], dtype=np.float64)
hybrid_low = np.ascontiguousarray(
tables["hybrid_analysis_low_kernel"], dtype=np.float64)
hybrid_indices = np.ascontiguousarray(
tables["hybrid_synthesis_indices"], dtype=np.int16)
hybrid_values = np.ascontiguousarray(
tables["hybrid_synthesis_values"], dtype=np.float64)
qmf_basis = np.ascontiguousarray(
tables["qmf_synthesis_basis"], dtype=np.float64)
qmf_taps = np.ascontiguousarray(
tables["qmf_synthesis_taps"], dtype=np.float64)
status = self._lib.ejoc_binaural_renderer_configure_kernels(
self._handle,
self._f64_pointer(qmf_analysis),
self._f64_pointer(hybrid_low),
hybrid_indices.ctypes.data_as(ctypes.POINTER(ctypes.c_int16)),
self._f64_pointer(hybrid_values),
len(hybrid_values),
self._f64_pointer(qmf_basis),
self._f64_pointer(qmf_taps),
)
if status:
self._raise("configure_kernels", status)
def _configure_room(self, model: RosellaModel):
if float(model.table_a_scalar) >= 0.5:
raise NotImplementedError("alternate table-A room mode")
bands = min(64, model.table_a_dimension)
allpass_delays = np.ascontiguousarray(
model.table_a_option_ids, dtype=np.uint32)
allpass_gains = np.ascontiguousarray(
model.table_a_option_values, dtype=np.float64)
fdn_delays = np.ascontiguousarray(
model.table_a_four_integers, dtype=np.uint32)
fdn_matrix = np.ascontiguousarray(
np.asarray(model.table_a_vector16, dtype=np.float64).reshape(
4, 4, order="F"))
filter8 = np.asarray(
model.table_a_filter_8x64_padded, dtype=np.float64).reshape(20, 4, 2, 4)
filter4 = np.asarray(
model.table_a_filter_4x64_padded, dtype=np.float64).reshape(20, 4, 4)
filter16 = np.asarray(
model.table_a_filter_16x64_padded, dtype=np.float64).reshape(20, 4, 4, 4)
feedback = np.empty((64, 4, 2), dtype=np.float64)
output_taps = np.empty((64, 4), dtype=np.float64)
output_matrix = np.empty((2, 64, 4, 2), dtype=np.float64)
for band in range(64):
group, lane = divmod(band, 4)
feedback[band, :, 0] = filter8[group, :, 0, lane]
feedback[band, :, 1] = filter8[group, :, 1, lane]
output_taps[band] = filter4[group, :, lane]
output_matrix[0, band, :, 0] = filter16[group, :, 0, lane]
output_matrix[0, band, :, 1] = filter16[group, :, 1, lane]
output_matrix[1, band, :, 0] = filter16[group, :, 2, lane]
output_matrix[1, band, :, 1] = filter16[group, :, 3, lane]
extra_count = int(model.table_a_extra)
extra_delays = np.ascontiguousarray(
model.table_a_extra_indices, dtype=np.uint32)
extra_fields = np.empty((extra_count, 64, 2), dtype=np.float64)
extra_source = np.asarray(
model.table_a_extra_fields_padded, dtype=np.float64).reshape(
extra_count, 20, 2, 4)
for extra in range(extra_count):
for band in range(64):
group, lane = divmod(band, 4)
extra_fields[extra, band] = extra_source[extra, group, :, lane]
extra_matrices = np.empty((extra_count, 4, 4), dtype=np.float64)
for extra in range(extra_count):
extra_matrices[extra] = np.asarray(
model.table_a_extra_vectors[extra], dtype=np.float64).reshape(
4, 4, order="F")
null_u32 = ctypes.POINTER(ctypes.c_uint32)()
null_f64 = ctypes.POINTER(ctypes.c_double)()
status = self._lib.ejoc_binaural_renderer_configure_room(
self._handle,
bands,
len(allpass_delays),
allpass_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)),
self._f64_pointer(allpass_gains),
fdn_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)),
self._f64_pointer(fdn_matrix),
int(model.table_a_integer),
self._f64_pointer(feedback),
self._f64_pointer(output_taps),
self._f64_pointer(output_matrix),
extra_count,
(extra_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32))
if extra_count else null_u32),
self._f64_pointer(extra_fields) if extra_count else null_f64,
self._f64_pointer(extra_matrices) if extra_count else null_f64,
)
if status:
self._raise("configure_room", status)
def reset(self):
if not self._handle:
raise RuntimeError("native binaural renderer is closed")
status = self._lib.ejoc_binaural_renderer_reset(self._handle)
if status:
self._raise("reset", status)
def process_block(self, pcm16, gains, room_sends, output_gain=1.0):
if not self._handle:
raise RuntimeError("native binaural renderer is closed")
source = np.ascontiguousarray(pcm16, dtype=np.float64)
gain_values = np.asarray(gains)
sends = np.ascontiguousarray(room_sends, dtype=np.float64)
if source.shape != (BLOCK_SAMPLES, INPUT_CHANNELS):
raise ValueError(f"pcm16 block must be (512,16), got {source.shape}")
if gain_values.shape != (INPUT_CHANNELS, OUTPUT_CHANNELS, HYBRID_BANDS):
raise ValueError(f"gains must be (16,2,77), got {gain_values.shape}")
direct = np.ascontiguousarray(
gain_values, dtype=np.complex128).view(np.float64)
if sends.shape != (INPUT_CHANNELS,):
raise ValueError(f"room_sends must be (16,), got {sends.shape}")
output = np.empty((BLOCK_SAMPLES, OUTPUT_CHANNELS), dtype=np.float64)
status = self._lib.ejoc_binaural_renderer_process(
self._handle,
self._f64_pointer(source),
self._f64_pointer(direct),
self._f64_pointer(sends),
float(output_gain),
self._f64_pointer(output),
)
if status:
self._raise("process", status)
return output
def close(self):
handle = getattr(self, "_handle", None)
if handle:
self._lib.ejoc_binaural_renderer_destroy(handle)
self._handle = None
+272
View File
@@ -0,0 +1,272 @@
"""JOC frame adapter for the public SOFA binaural backend."""
from __future__ import annotations
import math
from pathlib import Path
import numpy as np
from binaural_metadata import OamdPositionTimeline
from public_filterbank import ANALYSIS_SYNTHESIS_LATENCY_SAMPLES
from sofa_binaural_backend import SofaBinauralBackend
from sofa_hrtf_field import (
DEFAULT_HRTF_CACHE_DIR,
DEFAULT_PROJECTION_RIDGE,
DEFAULT_SH_RIDGE,
)
SAMPLE_RATE = 48000
FRAME_SAMPLES = 1536
BINAURAL_BLOCK_SAMPLES = 512
QMF_HOP_SAMPLES = 64
BINAURAL_LATENCY_SAMPLES = ANALYSIS_SYNTHESIS_LATENCY_SAMPLES
SOURCE_CHANNELS = 16
OUTPUT_CHANNELS = 2
PROJECT_DIR = Path(__file__).resolve().parent.parent
DEFAULT_HRTF_DIR = PROJECT_DIR / "HRTF"
DEFAULT_SOFA_HRTF = DEFAULT_HRTF_DIR / "binaural.sofa"
def _resolve_hrtf_file(path: str | Path, suffix: str, label: str) -> Path:
target = Path(path).expanduser().resolve()
if target.suffix.lower() != suffix:
raise ValueError(f"{label} must use the {suffix} extension: {target}")
if not target.is_file():
raise FileNotFoundError(f"{label} not found: {target}")
return target
def resolve_sofa_hrtf(path: str | Path) -> Path:
"""Resolve an explicitly selected public SOFA source."""
return _resolve_hrtf_file(path, ".sofa", "SOFA HRTF")
def resolve_compiled_hrtf_cache(path: str | Path) -> Path:
"""Resolve an explicitly selected JOC compiled HRTF cache."""
return _resolve_hrtf_file(path, ".jochrtf", "compiled HRTF cache")
class SofaBinauralRenderer:
"""Render interleaved LFE plus fifteen JOC objects to stereo.
The adapter owns frame buffering and sample-timed OAMD updates. The
backend owns the 64-QMF/77-hybrid state, the 961-sample latency policy,
per-object direct/early state, and the shared late room.
"""
def __init__(
self, backend, *,
mode: str = "mid",
object_delay_samples: int = 1473,
tail_seconds: float = 5.0,
chunk_frames: int = 64):
required_interface = (
"source_count", "default_profile", "set_source", "process",
"finish", "finish_output_capacity", "info")
missing = [name for name in required_interface if not hasattr(backend, name)]
if missing:
raise TypeError(
f"backend must implement the binaural backend interface; "
f"missing: {', '.join(missing)}")
if backend.source_count != SOURCE_CHANNELS:
raise ValueError(f"JOC binaural backend must have {SOURCE_CHANNELS} sources")
if backend.default_profile != str(mode).lower():
raise ValueError("backend default profile does not match renderer mode")
if int(object_delay_samples) < 0:
raise ValueError("object_delay_samples must be non-negative")
if not math.isfinite(float(tail_seconds)) or float(tail_seconds) < 0.0:
raise ValueError("tail_seconds must be finite and non-negative")
if int(chunk_frames) <= 0:
raise ValueError("chunk_frames must be positive")
self.backend = backend
self.mode = str(mode).lower()
self.object_delay_samples = int(object_delay_samples)
self.tail_seconds = float(tail_seconds)
self.chunk_frames = int(chunk_frames)
self.chunk_samples = self.chunk_frames * FRAME_SAMPLES
self.dsp_backend = getattr(backend, "dsp_backend", "python-sofa")
self.timeline = OamdPositionTimeline(15)
self._input_buffer = np.empty(
(self.chunk_samples, SOURCE_CHANNELS), dtype=np.float64)
self._buffer_used = 0
self.input_samples = 0
self.processed_input_samples = 0
self.output_samples = 0
self.finished = False
self.metadata_block_updates = 0
@classmethod
def from_sofa(
cls, sofa: str | Path, *,
mode: str = "mid",
cache_policy: str = "memory",
cache_dir: str | Path | None = DEFAULT_HRTF_CACHE_DIR,
shell_radius_m: float = 1.0,
projection_ridge: float = DEFAULT_PROJECTION_RIDGE,
sh_ridge: float = DEFAULT_SH_RIDGE,
object_delay_samples: int = 1473,
tail_seconds: float = 5.0,
output_gain: float = 1.0,
chunk_frames: int = 64) -> "SofaBinauralRenderer":
source = resolve_sofa_hrtf(sofa)
backend = SofaBinauralBackend.from_sofa(
source,
source_count=SOURCE_CHANNELS,
default_profile=mode,
output_gain=output_gain,
cache_policy=cache_policy,
cache_dir=cache_dir,
shell_radius_m=shell_radius_m,
projection_ridge=projection_ridge,
sh_ridge=sh_ridge)
return cls(
backend,
mode=mode,
object_delay_samples=object_delay_samples,
tail_seconds=tail_seconds,
chunk_frames=chunk_frames)
@classmethod
def from_compiled_cache(
cls, cache: str | Path, *,
mode: str = "mid",
object_delay_samples: int = 1473,
tail_seconds: float = 5.0,
output_gain: float = 1.0,
chunk_frames: int = 64) -> "SofaBinauralRenderer":
source = resolve_compiled_hrtf_cache(cache)
backend = SofaBinauralBackend.from_compiled_cache(
source,
source_count=SOURCE_CHANNELS,
default_profile=mode,
output_gain=output_gain)
return cls(
backend,
mode=mode,
object_delay_samples=object_delay_samples,
tail_seconds=tail_seconds,
chunk_frames=chunk_frames)
@property
def finish_capacity_samples(self) -> int:
return self.backend.finish_output_capacity(self.tail_seconds)
def _append_input(self, samples: np.ndarray) -> list[np.ndarray]:
outputs = []
source = np.asarray(samples, dtype=np.float64)
position = 0
while position < len(source):
count = min(self.chunk_samples - self._buffer_used, len(source) - position)
self._input_buffer[self._buffer_used:self._buffer_used + count] = (
source[position:position + count])
self._buffer_used += count
position += count
if self._buffer_used == self.chunk_samples:
outputs.append(self._process_samples(self._input_buffer))
self._buffer_used = 0
return outputs
def render_frame(self, objects16, payload=None, metadata_offset=None,
*, outer_sample_offset=0) -> np.ndarray:
"""Submit one 1536-sample reconstructed frame and its ID11 payload."""
if self.finished:
raise RuntimeError("binaural renderer is already finished")
source = np.asarray(objects16)
if source.shape != (FRAME_SAMPLES, SOURCE_CHANNELS):
raise ValueError(
f"binaural frame must have shape ({FRAME_SAMPLES},{SOURCE_CHANNELS}), "
f"got {source.shape}")
frame_start = self.input_samples
metadata_delay = (self.object_delay_samples if metadata_offset is None
else int(metadata_offset))
if metadata_delay < 0:
raise ValueError("metadata_offset must be non-negative")
if payload is not None:
self.timeline.submit_payload(
payload,
frame_start_sample=frame_start,
outer_sample_offset=int(outer_sample_offset),
object_delay_samples=metadata_delay,
processed_sample=self.processed_input_samples,
)
self.metadata_block_updates += 1
self.input_samples += FRAME_SAMPLES
chunks = self._append_input(source)
if not chunks:
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
return np.concatenate(chunks, axis=0) if len(chunks) > 1 else chunks[0]
def _set_block_parameters(self, sample: int) -> None:
positions = self.timeline.positions_at(sample)
self.backend.set_source(
0, (0.0, 1.0, 0.0), profile=self.mode,
special_lfe=True)
for object_index in range(15):
self.backend.set_source(
object_index + 1,
positions[object_index],
profile=self.mode)
def _process_samples(self, source: np.ndarray) -> np.ndarray:
values = np.asarray(source, dtype=np.float64)
if values.ndim != 2 or values.shape[1] != SOURCE_CHANNELS:
raise ValueError(f"expected [samples,{SOURCE_CHANNELS}], got {values.shape}")
if len(values) % BINAURAL_BLOCK_SAMPLES:
raise ValueError("binaural input must be divisible by 512 samples")
outputs = []
block_base = self.processed_input_samples
for start in range(0, len(values), BINAURAL_BLOCK_SAMPLES):
sample = block_base + start
self._set_block_parameters(sample)
outputs.append(self.backend.process(
values[start:start + BINAURAL_BLOCK_SAMPLES]))
self.processed_input_samples += len(values)
nonempty = [value for value in outputs if len(value)]
if not nonempty:
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
output = np.concatenate(nonempty, axis=0)
self.output_samples += len(output)
return output
def finish(self) -> np.ndarray:
"""Process pending source samples and drain early/late room state once."""
if self.finished:
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
outputs: list[np.ndarray] = []
if self._buffer_used:
outputs.append(self._process_samples(
self._input_buffer[:self._buffer_used]))
self._buffer_used = 0
outputs.append(self.backend.finish(tail_seconds=self.tail_seconds))
self.finished = True
nonempty = [value for value in outputs if len(value)]
if not nonempty:
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
output = np.concatenate(nonempty, axis=0)
self.output_samples += len(outputs[-1])
return output
def close(self) -> None:
self.finished = True
@property
def backend_info(self) -> dict:
info = self.backend.info()
info.update({
"adapter": "JOC 1536-frame / 512-sample metadata",
"dsp_backend": self.dsp_backend,
"mode": self.mode,
"latency_compensated_samples": BINAURAL_LATENCY_SAMPLES,
"object_delay_samples": self.object_delay_samples,
"tail_seconds": self.tail_seconds,
"metadata_payloads": self.timeline.payload_count,
"metadata_position_transitions": self.timeline.transition_count,
"input_samples": self.input_samples,
"source_samples_processed": self.processed_input_samples,
"output_samples_before_tail_trim": self.output_samples,
"thread_safe": False,
})
return info
-795
View File
@@ -1,795 +0,0 @@
// joc_cli -- command line frontend, argument-compatible with the reference
// Python CLI (main.py): the same positional input, the same mode selection
// (ADM BWF by default, --speaker-layout or --binaural), the same option names,
// choices and defaults, and the same default output naming under output/.
//
// Options that exist only because this build has no Python side or no Rosella
// import chain (--backend python, --sofa-hrtf, --personalized-headphone,
// metadata sidecars) fail with an explicit message instead of being ignored.
#include <algorithm>
#include <cmath>
#include <cstdint>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <filesystem>
#include <stdexcept>
#include <string>
#include <vector>
#include "foundation/fs_utf8.h"
#include "joc_core.h"
namespace {
namespace fs = std::filesystem;
namespace fs_utf8 = joc::fs_utf8;
constexpr double kRate = 48000.0;
constexpr int kFrameSamples = 1536;
struct Options {
std::string input;
std::string output;
std::string speaker_output;
std::string binaural_output;
std::string speaker_layout;
bool binaural = false;
std::string speaker_format = "float32";
std::string binaural_format = "float32";
std::string clip_action = "ask";
int speaker_metadata_offset = 1473;
std::string binaural_mode = "mid";
std::string sofa_hrtf;
std::string compiled_hrtf_cache;
std::string personalized_headphone;
bool personalized_headphone_used = false;
std::string hrtf_cache_policy;
std::string hrtf_cache_dir;
double hrtf_radius_m = 1.0;
double binaural_tail_seconds = 5.0;
double binaural_tail_threshold = 1.0e-8;
int binaural_chunk_frames = 64;
double gain_db = 0.0;
double duration = 0.0;
bool duration_set = false;
int object_delay_samples = 1473;
std::string trajectory_mode = "compact";
std::string ffmpeg;
double eac3_drc_scale = 0.0;
int eac3_target_level = 0;
std::string backend = "auto";
std::string native_library;
int native_threads = 0;
bool native_threads_set = false;
std::string metadata_dir;
std::string metadata_cache;
std::string metadata_backend = "auto";
std::string print_metadata = "none";
std::string metadata_json;
bool metadata_only = false;
bool keep_raw = false;
bool skip_sha256 = false;
int progress_every = 1000;
// C++-side additions (documented as such; the Python CLI has no equivalent).
std::string bed;
std::string kernels;
std::string work_dir;
std::string report_json;
bool report_json_set = false;
bool dry_run = false;
bool quiet = false;
bool help = false;
};
const char* kLayoutChoices =
"2.0 3.0 3.1 4.0 5.0 5.1 5.1.2 5.1.4 6.1 7.0 7.1 7.1.2 7.1.4 9.1.4 9.1.6 22.2";
void print_usage() {
std::printf(
"usage: joc_cli [options] input\n"
"\n"
"JustOneCacophony (JOC):E-AC-3 JOC → 25ch ADM BWF、扬声器 WAV 或双耳 WAV\n"
"\n"
"位置参数:\n"
" input 输入 .m4a/.eac3/.ec3\n"
"\n"
"模式(默认输出 ADM BWF):\n"
" --speaker-layout L 直接扬声器渲染布局,例如 2.0、5.1、7.1.2\n"
" 可选值: %s\n"
" --binaural 直接双耳渲染;不生成临时 ADM BWF\n"
"\n"
"输出:\n"
" -o, --output PATH 输出文件;默认 output/<名称>.adm.wav、\n"
" output/<名称>.<布局>.wav 或 output/<名称>.binaural.wav\n"
" --speaker-output PATH 扬声器 WAV 路径;仅与 --speaker-layout 一起使用\n"
" --binaural-output PATH 双耳 WAV 路径;仅与 --binaural 一起使用\n"
" --speaker-format F 扬声器 WAV 格式 float32|int24,默认 float32\n"
" --binaural-format F 双耳 WAV 格式 float32|int24,默认 float32\n"
" --clip-action A int24 削波处理 ask|continue|float32|abort,默认 ask\n"
"\n"
"渲染:\n"
" --speaker-metadata-offset N 扬声器渲染 metadata 相对帧偏移,默认 1473 samples\n"
" --binaural-mode M 双耳渲染模式 off|near|mid|far,默认 mid;\n"
" off 仅用于 ADM BWF(关闭 DBMD 双耳提示)\n"
" --sofa-hrtf PATH SOFA SimpleFreeFieldHRIR 输入;.jochrtf 由本工具内部编译\n"
" --personalized-headphone [PATH] Rosella 个性化模型,默认 "
"HRTF/binaural.personalized_headphone\n"
" --compiled-hrtf-cache PATH 直接读取 .jochrtf(高级用法,跳过 SOFA 编译)\n"
" --hrtf-cache-policy P SOFA 编译缓存策略 none|memory|disk,默认 memory\n"
" --hrtf-cache-dir DIR disk cache 目录,默认 <exe>/output/hrtf-cache\n"
" --hrtf-radius-m R 选择最近的 SOFA measurement-radius shell,默认 1.0 m\n"
" --binaural-tail-seconds S 双耳 room/filterbank flush 上限,默认 5 秒\n"
" --binaural-tail-threshold T 双耳尾声裁切阈值,默认 1e-8;主体至少保留原时长\n"
" --binaural-chunk-frames N 双耳内部批处理帧数,默认 64(本构建按 512 块渲染,\n"
" 取值不影响输出)\n"
" --gain-db X 成品增益 dB,默认 0;双耳路径以 float64 应用\n"
" --duration S 只处理开头指定秒数\n"
" --object-delay-samples N 对象 PCM/OAMD 时间补偿,默认 1473 samples\n"
" --trajectory-mode M ADM 对象轨迹表示 compact|dense64,默认 compact\n"
"\n"
"输入与解码:\n"
" --ffmpeg PATH ffmpeg 可执行文件,默认取 FFMPEG 环境变量或 PATH\n"
" --eac3-drc-scale X E-AC-3 解码器 -drc_scale,0=关闭码流 dynrng,默认 0\n"
" --eac3-target-level N E-AC-3 解码器 -target_level,0=不施加,默认 0\n"
" --backend B JOC/扬声器 DSP 后端 auto|native;本构建无 python 后端\n"
" --native-threads N 原生 DSP 总线程数;默认在 4 核以上使用 2\n"
"\n"
"诊断:\n"
" --print-metadata M 诊断元数据输出 none|summary|frames,默认 none\n"
" --metadata-json PATH 元数据汇总 JSON 路径\n"
" --metadata-only 解析/打印元数据后退出\n"
" --keep-raw 额外保留 16ch f32le 对象中间文件\n"
" --skip-sha256 跳过最终文件 SHA-256 全量复扫\n"
" --progress-every N 进度输出间隔,默认 1000 帧(渲染与收尾写盘同一节奏)\n"
"\n"
"本构建特有(Python 版没有对应参数):\n"
" --bed PATH 已解码的 6 通道 float32 PCM;给出后不调用 ffmpeg 解码\n"
" --kernels PATH 双耳滤波器组表 rosella_kernels.npz\n"
" --work-dir DIR 临时目录\n"
" --report-json PATH 结果 JSON 路径;默认 <输出>.report.json\n"
" --dry-run 只校验配置\n"
" --quiet 只输出警告与错误\n",
kLayoutChoices);
}
[[noreturn]] void fail(const std::string& message) { throw std::runtime_error(message); }
std::string require_value(const std::vector<std::string>& arguments, int* index) {
if (static_cast<std::size_t>(*index) + 1u >= arguments.size()) {
fail("argument " + arguments[static_cast<std::size_t>(*index)] +
": expected one argument");
}
return arguments[static_cast<std::size_t>(++(*index))];
}
double to_double(const std::string& text, const char* name) {
try {
std::size_t used = 0;
const double value = std::stod(text, &used);
if (used != text.size()) {
fail(std::string(name) + ": invalid float value: " + text);
}
return value;
} catch (const std::exception&) {
fail(std::string(name) + ": invalid float value: " + text);
}
}
long long to_int(const std::string& text, const char* name) {
try {
std::size_t used = 0;
const long long value = std::stoll(text, &used);
if (used != text.size()) {
fail(std::string(name) + ": invalid int value: " + text);
}
return value;
} catch (const std::exception&) {
fail(std::string(name) + ": invalid int value: " + text);
}
}
void check_choice(const std::string& value, const char* name,
std::initializer_list<const char*> allowed) {
for (const char* candidate : allowed) {
if (value == candidate) {
return;
}
}
std::string list;
for (const char* candidate : allowed) {
list += list.empty() ? candidate : (", " + std::string(candidate));
}
fail(std::string(name) + ": invalid choice: '" + value + "' (choose from " + list + ")");
}
void parse_args(const std::vector<std::string>& arguments, Options* options) {
std::vector<std::string> positional;
const int argc = static_cast<int>(arguments.size());
for (int index = 1; index < argc; ++index) {
const std::string arg = arguments[static_cast<std::size_t>(index)];
if (arg == "-h" || arg == "--help") { options->help = true; }
else if (arg == "-o" || arg == "--output") { options->output = require_value(arguments, &index); }
else if (arg == "--speaker-output") { options->speaker_output = require_value(arguments, &index); }
else if (arg == "--binaural-output") { options->binaural_output = require_value(arguments, &index); }
else if (arg == "--speaker-layout") { options->speaker_layout = require_value(arguments, &index); }
else if (arg == "--binaural") { options->binaural = true; }
else if (arg == "--speaker-format") { options->speaker_format = require_value(arguments, &index); }
else if (arg == "--binaural-format") { options->binaural_format = require_value(arguments, &index); }
else if (arg == "--clip-action") { options->clip_action = require_value(arguments, &index); }
else if (arg == "--speaker-metadata-offset") {
options->speaker_metadata_offset = static_cast<int>(
to_int(require_value(arguments, &index), "--speaker-metadata-offset"));
}
else if (arg == "--binaural-mode") { options->binaural_mode = require_value(arguments, &index); }
else if (arg == "--sofa-hrtf") { options->sofa_hrtf = require_value(arguments, &index); }
else if (arg == "--compiled-hrtf-cache") { options->compiled_hrtf_cache = require_value(arguments, &index); }
else if (arg == "--personalized-headphone") {
options->personalized_headphone_used = true;
// nargs="?": the path is optional, so the next token may be the input.
// Without a path the executable-anchored default is resolved later.
if (static_cast<std::size_t>(index) + 1u < arguments.size() &&
arguments[static_cast<std::size_t>(index) + 1u][0] != '-') {
options->personalized_headphone = require_value(arguments, &index);
}
}
else if (arg == "--hrtf-cache-policy") { options->hrtf_cache_policy = require_value(arguments, &index); }
else if (arg == "--hrtf-cache-dir") { options->hrtf_cache_dir = require_value(arguments, &index); }
else if (arg == "--hrtf-radius-m") { options->hrtf_radius_m = to_double(require_value(arguments, &index), "--hrtf-radius-m"); }
else if (arg == "--binaural-tail-seconds") { options->binaural_tail_seconds = to_double(require_value(arguments, &index), "--binaural-tail-seconds"); }
else if (arg == "--binaural-tail-threshold") { options->binaural_tail_threshold = to_double(require_value(arguments, &index), "--binaural-tail-threshold"); }
else if (arg == "--binaural-chunk-frames") { options->binaural_chunk_frames = static_cast<int>(to_int(require_value(arguments, &index), "--binaural-chunk-frames")); }
else if (arg == "--gain-db") { options->gain_db = to_double(require_value(arguments, &index), "--gain-db"); }
else if (arg == "--duration") { options->duration = to_double(require_value(arguments, &index), "--duration"); options->duration_set = true; }
else if (arg == "--object-delay-samples") { options->object_delay_samples = static_cast<int>(to_int(require_value(arguments, &index), "--object-delay-samples")); }
else if (arg == "--trajectory-mode") { options->trajectory_mode = require_value(arguments, &index); }
else if (arg == "--ffmpeg") { options->ffmpeg = require_value(arguments, &index); }
else if (arg == "--eac3-drc-scale") { options->eac3_drc_scale = to_double(require_value(arguments, &index), "--eac3-drc-scale"); }
else if (arg == "--eac3-target-level") { options->eac3_target_level = static_cast<int>(to_int(require_value(arguments, &index), "--eac3-target-level")); }
else if (arg == "--backend") { options->backend = require_value(arguments, &index); }
else if (arg == "--native-library") { options->native_library = require_value(arguments, &index); }
else if (arg == "--native-threads") { options->native_threads = static_cast<int>(to_int(require_value(arguments, &index), "--native-threads")); options->native_threads_set = true; }
else if (arg == "--metadata-dir") { options->metadata_dir = require_value(arguments, &index); }
else if (arg == "--metadata-cache") { options->metadata_cache = require_value(arguments, &index); }
else if (arg == "--metadata-backend") { options->metadata_backend = require_value(arguments, &index); }
else if (arg == "--print-metadata") { options->print_metadata = require_value(arguments, &index); }
else if (arg == "--metadata-json") { options->metadata_json = require_value(arguments, &index); }
else if (arg == "--metadata-only") { options->metadata_only = true; }
else if (arg == "--keep-raw") { options->keep_raw = true; }
else if (arg == "--skip-sha256") { options->skip_sha256 = true; }
else if (arg == "--progress-every") { options->progress_every = static_cast<int>(to_int(require_value(arguments, &index), "--progress-every")); }
else if (arg == "--bed") { options->bed = require_value(arguments, &index); }
else if (arg == "--kernels") { options->kernels = require_value(arguments, &index); }
else if (arg == "--work-dir") { options->work_dir = require_value(arguments, &index); }
else if (arg == "--report-json") { options->report_json = require_value(arguments, &index); options->report_json_set = true; }
else if (arg == "--dry-run") { options->dry_run = true; }
else if (arg == "--quiet") { options->quiet = true; }
else if (!arg.empty() && arg[0] == '-' && arg != "-") { fail("unrecognized argument: " + arg); }
else { positional.push_back(arg); }
}
if (positional.size() > 1u) {
fail("unrecognized extra arguments: " + positional[1] +
(positional.size() > 2u ? " ..." : ""));
}
if (!positional.empty()) {
options->input = positional.front();
}
}
// Mirrors the reference resolve_output(): <project>/output plus a mode-specific
// name. The project directory is the executable's directory, as upstream uses
// the script's directory, so the layout does not depend on the working directory.
std::string resolve_output(const Options& options, const std::string& source,
const std::string& executable_directory) {
const std::string requested = !options.speaker_output.empty() ? options.speaker_output
: !options.binaural_output.empty() ? options.binaural_output
: options.output;
if (!requested.empty()) {
std::error_code error;
const fs::path absolute = fs::absolute(fs_utf8::to_path(requested), error);
return error ? requested : fs_utf8::from_path(absolute);
}
const fs::path directory = fs_utf8::to_path(executable_directory) / "output";
const std::string stem = fs_utf8::from_path(fs_utf8::to_path(source).stem());
if (!options.speaker_layout.empty()) {
return fs_utf8::from_path(directory /
fs_utf8::to_path(stem + "." + options.speaker_layout + ".wav"));
}
if (options.binaural) {
return fs_utf8::from_path(directory / fs_utf8::to_path(stem + ".binaural.wav"));
}
return fs_utf8::from_path(directory / fs_utf8::to_path(stem + ".adm.wav"));
}
// The project directory the reference anchors its defaults at: the directory of
// the running executable, never the working directory.
std::string executable_dir(const std::string& argv0) {
const std::string own_path = fs_utf8::executable_path();
if (!own_path.empty()) {
const fs::path path = fs_utf8::to_path(own_path);
if (path.has_parent_path()) {
return fs_utf8::from_path(path.parent_path());
}
}
if (argv0.empty()) {
return ".";
}
std::error_code error;
const fs::path path = fs::absolute(fs_utf8::to_path(argv0), error);
if (error || path.empty()) {
return ".";
}
return fs_utf8::from_path(path.parent_path());
}
std::string find_kernels(const Options& options, const std::string& argv0) {
(void)argv0;
if (!options.kernels.empty() && !fs_utf8::exists(options.kernels)) {
fail("--kernels 指向的文件不存在: " + options.kernels);
}
// Empty means the tables compiled into the library.
return options.kernels;
}
// Mirrors the reference binaural HRTF resolution (main.py:89-160): the SOFA file
// is the user-facing input and the .jochrtf is only its compiled cache. Paths
// are anchored at the executable directory, as the reference anchors them at the
// project directory.
struct HrtfInput {
std::string sofa_path; // compile this
std::string compiled_path; // or read this .jochrtf directly
std::string cache_dir; // disk policy directory
std::string personalized_path; // Rosella .personalized_headphone
bool disk = false;
};
std::string resolve_compiled_hrtf(const Options& options, const std::string& project_directory) {
if (!options.compiled_hrtf_cache.empty()) {
return options.compiled_hrtf_cache;
}
const std::string directory_utf8 =
options.hrtf_cache_dir.empty()
? fs_utf8::from_path(fs_utf8::to_path(project_directory) / "output" / "hrtf-cache")
: options.hrtf_cache_dir;
if (!fs_utf8::is_directory(directory_utf8)) {
return std::string();
}
const fs::path directory = fs_utf8::to_path(directory_utf8);
std::vector<fs::path> candidates;
for (const fs::directory_entry& entry : fs::directory_iterator(directory)) {
if (entry.is_regular_file() && entry.path().extension() == ".jochrtf") {
candidates.push_back(entry.path());
}
}
std::sort(candidates.begin(), candidates.end());
if (candidates.size() > 1u) {
fail(directory_utf8 +
" 下有多个 .jochrtf 缓存,无法自动选择;请用 --sofa-hrtf PATH 或 "
"--compiled-hrtf-cache PATH 显式指定");
}
return candidates.empty() ? std::string() : fs_utf8::from_path(candidates.front());
}
HrtfInput resolve_hrtf_input(const Options& options, const std::string& project_directory) {
HrtfInput input;
const std::string default_sofa =
fs_utf8::from_path(fs_utf8::to_path(project_directory) / "HRTF" / "binaural.sofa");
const std::string default_private = fs_utf8::from_path(
fs_utf8::to_path(project_directory) / "HRTF" / "binaural.personalized_headphone");
const std::string default_cache_dir =
fs_utf8::from_path(fs_utf8::to_path(project_directory) / "output" / "hrtf-cache");
if (!options.compiled_hrtf_cache.empty() && !options.hrtf_cache_policy.empty()) {
fail("显式 .jochrtf 输入不能再指定 --hrtf-cache-policy");
}
if (!options.compiled_hrtf_cache.empty() && options.hrtf_radius_m != 1.0) {
fail("显式 .jochrtf 输入不能再选择 SOFA radius shell");
}
if (options.personalized_headphone_used &&
(!options.hrtf_cache_policy.empty() || !options.hrtf_cache_dir.empty() ||
options.hrtf_radius_m != 1.0)) {
fail("Rosella 模型输入不能使用 --hrtf-cache-policy/--hrtf-cache-dir/--hrtf-radius-m");
}
std::string sofa = options.sofa_hrtf;
std::string compiled = options.compiled_hrtf_cache;
std::string personalized =
options.personalized_headphone_used ? options.personalized_headphone : std::string();
if (options.personalized_headphone_used && personalized.empty()) {
// "--personalized-headphone" without a path means the project default.
personalized = default_private;
}
if (sofa.empty() && compiled.empty() && personalized.empty()) {
// The reference order: the SOFA file, then the unique compiled cache, then the
// personalized model.
if (fs_utf8::exists(default_sofa)) {
sofa = default_sofa;
} else {
compiled = resolve_compiled_hrtf(options, project_directory);
if (compiled.empty() && fs_utf8::exists(default_private)) {
personalized = default_private;
}
}
}
if (!personalized.empty()) {
if (!fs_utf8::exists(personalized)) {
fail("双耳模型不存在: " + personalized);
}
input.personalized_path = personalized;
return input;
}
if (sofa.empty() && compiled.empty()) {
if (!options.hrtf_cache_policy.empty() || !options.hrtf_cache_dir.empty() ||
options.hrtf_radius_m != 1.0) {
fail("HRTF cache/radius 选项需要 --sofa-hrtf");
}
fail("--binaural 未找到 HRTF 输入:默认 " + default_sofa + "、" + default_private +
" 或 " + default_cache_dir +
" 下的 .jochrtf 都不存在,请用 --sofa-hrtf PATH、--personalized-headphone PATH "
"或 --compiled-hrtf-cache PATH 指定");
}
if (sofa.empty() && (!options.hrtf_cache_policy.empty() || !options.hrtf_cache_dir.empty() ||
options.hrtf_radius_m != 1.0)) {
fail("HRTF cache/radius 选项需要 --sofa-hrtf");
}
if (sofa.empty()) {
input.compiled_path = compiled;
return input;
}
const std::string effective_policy =
options.hrtf_cache_policy.empty() ? "memory" : options.hrtf_cache_policy;
if (!options.hrtf_cache_dir.empty() && effective_policy != "disk") {
fail("--hrtf-cache-dir 需要 SOFA 与 disk cache policy 一起使用");
}
input.sofa_path = sofa;
input.disk = effective_policy == "disk";
input.cache_dir = options.hrtf_cache_dir.empty() ? default_cache_dir : options.hrtf_cache_dir;
if (!fs_utf8::exists(input.sofa_path)) {
fail("SOFA HRTF 不存在: " + input.sofa_path);
}
return input;
}
std::uint32_t binaural_mode_value(const std::string& name) {
if (name == "off") { return JOC_BINAURAL_OFF; }
if (name == "near") { return JOC_BINAURAL_NEAR; }
if (name == "far") { return JOC_BINAURAL_FAR; }
return JOC_BINAURAL_MID;
}
std::uint32_t clip_action_value(const std::string& name) {
if (name == "continue") { return JOC_CLIP_CONTINUE; }
if (name == "float32") { return JOC_CLIP_FLOAT32; }
if (name == "abort") { return JOC_CLIP_ABORT; }
return JOC_CLIP_ASK;
}
std::string format_eta(double seconds) {
if (seconds < 0.0 || seconds > 86400.0) {
return "--";
}
char buffer[64];
std::snprintf(buffer, sizeof(buffer), "%.0fs", seconds);
return buffer;
}
void JOC_CALL on_event(void* user, const joc_event* event) {
const Options* options = static_cast<const Options*>(user);
if (event == nullptr) {
return;
}
switch (event->type) {
case JOC_EV_PROGRESS: {
if (options->quiet) {
return;
}
const double fraction = event->progress >= 0.0 ? event->progress : 0.0;
const double remaining =
fraction > 0.0 ? event->elapsed_seconds * (1.0 - fraction) / fraction : -1.0;
std::printf("[%s] %llu/%llu %.1fx realtime ETA %s\n", event->stage_name,
static_cast<unsigned long long>(event->current_frame),
static_cast<unsigned long long>(event->total_frames),
event->realtime_factor, format_eta(remaining).c_str());
std::fflush(stdout);
return;
}
case JOC_EV_LOG: {
if (options->quiet || event->log_level < JOC_LOG_INFO || event->message[0] == '\0') {
return;
}
std::printf("[%s] %s\n", event->stage_name, event->message);
std::fflush(stdout);
return;
}
case JOC_EV_WARNING:
std::printf("[warning] %s\n", event->message);
return;
case JOC_EV_ERROR:
std::fprintf(stderr, "[error] %s (%s)\n", event->message,
joc_error_name(event->error_code));
return;
default:
if (options->quiet || event->log_level < JOC_LOG_INFO || event->message[0] == '\0') {
return;
}
std::printf("[%s] %s\n", event->stage_name, event->message);
return;
}
}
} // namespace
int main(int argc, char** argv) {
fs_utf8::configure_console();
const std::vector<std::string> arguments = fs_utf8::command_line_arguments(argc, argv);
Options options;
try {
parse_args(arguments, &options);
} catch (const std::exception& error) {
std::fprintf(stderr, "joc_cli: error: %s\n", error.what());
return 2;
}
if (options.help) {
print_usage();
return 0;
}
if (options.input.empty()) {
print_usage();
return 2;
}
try {
check_choice(options.speaker_format, "--speaker-format", {"float32", "int24"});
check_choice(options.binaural_format, "--binaural-format", {"float32", "int24"});
check_choice(options.clip_action, "--clip-action",
{"ask", "continue", "float32", "abort"});
check_choice(options.binaural_mode, "--binaural-mode", {"off", "near", "mid", "far"});
check_choice(options.trajectory_mode, "--trajectory-mode", {"compact", "dense64"});
check_choice(options.backend, "--backend", {"auto", "native", "python"});
check_choice(options.print_metadata, "--print-metadata", {"none", "summary", "frames"});
check_choice(options.metadata_backend, "--metadata-backend", {"auto", "emdf", "sidecar"});
if (!options.hrtf_cache_policy.empty()) {
check_choice(options.hrtf_cache_policy, "--hrtf-cache-policy",
{"none", "memory", "disk"});
}
const bool speaker_mode = !options.speaker_layout.empty();
const bool binaural_mode = options.binaural;
if (speaker_mode && binaural_mode) {
fail("argument --binaural: not allowed with argument --speaker-layout");
}
if (options.binaural_mode == "off" && (speaker_mode || binaural_mode)) {
fail("--binaural-mode off 仅用于 ADM BWF 输出(关闭 DBMD 双耳提示);"
"直接双耳渲染请使用 near/mid/far");
}
if (!options.speaker_output.empty() && !speaker_mode) {
fail("--speaker-output 必须与 --speaker-layout 一起使用");
}
if (!options.binaural_output.empty() && !binaural_mode) {
fail("--binaural-output 必须与 --binaural 一起使用");
}
const bool specific_output =
!options.speaker_output.empty() || !options.binaural_output.empty();
if (!options.output.empty() && specific_output) {
fail("-o/--output 与 --speaker-output/--binaural-output 不能同时使用");
}
if (!options.speaker_output.empty() && !options.binaural_output.empty()) {
fail("--speaker-output 与 --binaural-output 不能同时使用");
}
if (options.speaker_metadata_offset < 0) {
fail("speaker-metadata-offset 不能为负数");
}
const bool hrtf_options_used =
!options.sofa_hrtf.empty() || !options.compiled_hrtf_cache.empty() ||
options.personalized_headphone_used || !options.hrtf_cache_policy.empty() ||
!options.hrtf_cache_dir.empty() || options.hrtf_radius_m != 1.0;
if (hrtf_options_used && !binaural_mode) {
fail("SOFA/HRTF 选项仅与 --binaural 一起使用");
}
if (!std::isfinite(options.binaural_tail_seconds) ||
options.binaural_tail_seconds < 0.0) {
fail("binaural-tail-seconds 必须是非负有限值");
}
if (!std::isfinite(options.binaural_tail_threshold) ||
options.binaural_tail_threshold < 0.0) {
fail("binaural-tail-threshold 必须是非负有限值");
}
if (options.binaural_chunk_frames <= 0) {
fail("binaural-chunk-frames 必须大于 0");
}
if (!std::isfinite(options.hrtf_radius_m) || options.hrtf_radius_m <= 0.0) {
fail("hrtf-radius-m 必须是正有限值");
}
if (options.duration_set && options.duration <= 0.0) {
fail("duration 必须大于 0");
}
if (options.object_delay_samples < 0) {
fail("object-delay-samples 不能为负数");
}
if (options.native_threads_set && options.native_threads < 1) {
fail("native-threads 必须大于 0");
}
if (!std::isfinite(options.gain_db) || std::abs(options.gain_db) > 200.0) {
fail("gain-db 超出支持范围");
}
// Options this build cannot honour: fail loudly instead of ignoring them.
if (options.backend == "python") {
fail("--backend python 在本构建中不可用(已无 Python 后端);请使用 auto 或 native");
}
if (options.metadata_backend == "sidecar" || !options.metadata_dir.empty() ||
!options.metadata_cache.empty()) {
fail("metadata sidecar 在本构建中不可用(始终直接扫描 EMDF)");
}
if (!options.native_library.empty()) {
std::fprintf(stderr, "[info] --native-library 在本构建中忽略(单一 joc_core.dll)\n");
}
if (!fs_utf8::exists(options.input)) {
fail("输入文件不存在: " + options.input);
}
} catch (const std::exception& error) {
std::fprintf(stderr, "joc_cli: error: %s\n", error.what());
return 2;
}
const bool speaker_mode = !options.speaker_layout.empty();
const bool binaural_mode = options.binaural;
const std::string project_directory =
executable_dir(arguments.empty() ? std::string() : arguments.front());
const std::string output_path = resolve_output(options, options.input, project_directory);
std::error_code directory_error;
fs::create_directories(fs_utf8::to_path(output_path).parent_path(), directory_error);
joc_task_config config{};
config.struct_size = sizeof(config);
config.struct_version = JOC_TASK_CONFIG_VERSION;
config.input_path = options.input.c_str();
config.output_path = output_path.c_str();
config.ffmpeg_path = options.ffmpeg.empty() ? nullptr : options.ffmpeg.c_str();
config.bed_path = options.bed.empty() ? nullptr : options.bed.c_str();
config.work_dir = options.work_dir.empty() ? nullptr : options.work_dir.c_str();
config.eac3_drc_scale = options.eac3_drc_scale;
config.eac3_target_level = options.eac3_target_level;
config.operation =
binaural_mode ? JOC_OP_BINAURAL : speaker_mode ? JOC_OP_SPEAKER : JOC_OP_ADM_BWF;
const std::string& requested_format =
binaural_mode ? options.binaural_format : options.speaker_format;
config.output_format =
requested_format == "int24" ? JOC_FORMAT_PCM24 : JOC_FORMAT_FLOAT32;
config.clip_action = clip_action_value(options.clip_action);
config.speaker_layout_name = speaker_mode ? options.speaker_layout.c_str() : nullptr;
config.speaker_metadata_offset = static_cast<std::uint32_t>(options.speaker_metadata_offset);
config.binaural_mode = binaural_mode_value(options.binaural_mode);
config.adm_binaural_mode = binaural_mode_value(options.binaural_mode);
config.binaural_tail_seconds = options.binaural_tail_seconds;
config.binaural_tail_threshold = options.binaural_tail_threshold;
config.binaural_chunk_frames = static_cast<std::uint32_t>(options.binaural_chunk_frames);
config.object_delay_samples = static_cast<std::uint32_t>(options.object_delay_samples);
config.trajectory_mode =
options.trajectory_mode == "dense64" ? JOC_TRAJECTORY_DENSE64 : JOC_TRAJECTORY_COMPACT;
config.gain_db = options.gain_db;
config.progress_interval_frames = static_cast<std::uint32_t>(options.progress_every);
config.native_threads =
options.native_threads_set ? static_cast<std::uint32_t>(options.native_threads) : 0u;
config.print_metadata = options.print_metadata == "frames" ? 2u
: options.print_metadata == "summary" ? 1u
: 0u;
config.metadata_json_path =
options.metadata_json.empty() ? nullptr : options.metadata_json.c_str();
config.duration_frames =
options.duration_set
? static_cast<std::uint64_t>(
std::ceil(options.duration * kRate / static_cast<double>(kFrameSamples)))
: 0u;
config.flags = 0u;
if (options.skip_sha256) { config.flags |= JOC_TASK_F_SKIP_SHA256; }
if (options.keep_raw) { config.flags |= JOC_TASK_F_KEEP_INTERMEDIATE; }
if (options.metadata_only) { config.flags |= JOC_TASK_F_METADATA_ONLY; }
if (options.quiet) { config.flags |= JOC_TASK_F_QUIET; }
std::string hrtf_path;
std::string hrtf_sofa_path;
std::string hrtf_cache_dir;
std::string personalized_path;
std::string kernels_path;
if (binaural_mode) {
try {
const HrtfInput input = resolve_hrtf_input(options, project_directory);
hrtf_path = input.compiled_path;
hrtf_sofa_path = input.sofa_path;
hrtf_cache_dir = input.cache_dir;
personalized_path = input.personalized_path;
kernels_path = find_kernels(options, arguments.empty() ? std::string()
: arguments.front());
} catch (const std::exception& error) {
std::fprintf(stderr, "joc_cli: error: %s\n", error.what());
return 2;
}
}
config.hrtf_path = hrtf_path.empty() ? nullptr : hrtf_path.c_str();
config.hrtf_sofa_path = hrtf_sofa_path.empty() ? nullptr : hrtf_sofa_path.c_str();
config.hrtf_cache_dir = hrtf_cache_dir.empty() ? nullptr : hrtf_cache_dir.c_str();
config.personalized_headphone_path =
personalized_path.empty() ? nullptr : personalized_path.c_str();
config.hrtf_cache_policy = options.hrtf_cache_policy == "disk" ? JOC_HRTF_CACHE_DISK
: options.hrtf_cache_policy == "none" ? JOC_HRTF_CACHE_NONE
: JOC_HRTF_CACHE_MEMORY;
config.hrtf_radius_m = options.hrtf_radius_m;
config.kernels_path = kernels_path.empty() ? nullptr : kernels_path.c_str();
joc_validation_issue issues[32];
std::uint32_t issue_count = 0;
const joc_error validated = joc_task_validate(&config, issues, 32u, &issue_count);
for (std::uint32_t index = 0; index < std::min(issue_count, 32u); ++index) {
if (issues[index].severity >= 2u) {
std::fprintf(stderr, "[error] %s: %s\n", issues[index].field, issues[index].message);
} else if (!options.quiet) {
std::fprintf(stderr, "[warning] %s: %s\n", issues[index].field,
issues[index].message);
}
}
if (validated != JOC_OK) {
std::fprintf(stderr, "joc_cli: error: configuration rejected (%u issue(s))\n", issue_count);
return 2;
}
if (options.dry_run) {
std::printf("configuration accepted (%u issue(s))\n", issue_count);
return 0;
}
joc_event_sink sink{};
sink.struct_size = sizeof(sink);
sink.callback = &on_event;
sink.user = &options;
if (!options.quiet) {
const char* mode_name = binaural_mode ? "binaural" : speaker_mode ? "speaker" : "adm";
std::printf("[cli] %s -> %s (%s)\n", options.input.c_str(), output_path.c_str(),
mode_name);
std::fflush(stdout);
}
joc_task_result result{};
const joc_error status = joc_task_execute(&config, &sink, &result);
const std::string report_path =
options.report_json_set ? options.report_json : (output_path + ".report.json");
{
std::size_t needed = 0;
joc_task_result_to_json(&result, nullptr, 0u, &needed);
std::vector<char> buffer(needed + 1u);
if (joc_task_result_to_json(&result, buffer.data(), buffer.size(), &needed) == JOC_OK) {
if (std::FILE* file = fs_utf8::fopen(report_path, "wb")) {
std::fwrite(buffer.data(), 1, std::strlen(buffer.data()), file);
std::fputc('\n', file);
std::fclose(file);
}
}
}
std::printf("\nresult: %s\n", status == JOC_OK ? "ok" : joc_error_name(status));
std::printf(" output : %s\n", output_path.c_str());
std::printf(" frames : %llu (%.2f s)\n",
static_cast<unsigned long long>(result.input_frames), result.duration_sec);
std::printf(" output samples: %llu\n",
static_cast<unsigned long long>(result.output_samples));
std::printf(" output bytes : %llu\n",
static_cast<unsigned long long>(result.output_file_bytes));
std::printf(" format : %s\n",
result.output_format_actual == JOC_FORMAT_PCM24 ? "int24" : "float32");
std::printf(" peak : %.9g (%llu sample(s) above full scale)\n", result.output_peak,
static_cast<unsigned long long>(result.output_over_unity_values));
std::printf(" sha256 : %s\n",
result.output_sha256[0] != '\0' ? result.output_sha256 : "(skipped)");
std::printf(" report : %s\n", report_path.c_str());
// Each stage time is measured where that stage actually runs, and the three
// stages now overlap (see the pipeline in src/task/task.cpp), so the stage
// times deliberately do not add up to the wall-clock total.
std::printf(" timings : decode %.2fs, joc %.2fs, dsp %.2fs, write %.2fs"
" (stage times, concurrent), total %.2fs\n",
result.t_decode_bed, result.t_render, result.t_render_dsp, result.t_write_file,
result.t_total);
if (status != JOC_OK) {
std::fprintf(stderr, "joc_cli: error: %s: %s\n", result.error_stage, result.error_message);
}
return status == JOC_OK ? 0 : 1;
}
-113
View File
@@ -1,113 +0,0 @@
#include "eac3_transport/eac3_reader.h"
#include <utility>
#include "foundation/status.h"
namespace joc::eac3 {
namespace {
constexpr std::size_t kHeaderBytes = 4;
} // namespace
void FrameReader::push(const std::uint8_t* data, std::size_t size) {
if (failed_ || data == nullptr || size == 0) {
return;
}
if (consumed_ > 0) {
compact();
}
buffer_.insert(buffer_.end(), data, data + size);
}
void FrameReader::compact() {
if (consumed_ == 0) {
return;
}
buffer_.erase(buffer_.begin(), buffer_.begin() + static_cast<std::ptrdiff_t>(consumed_));
base_offset_ += consumed_;
consumed_ = 0;
}
void FrameReader::fail(joc_error code, std::string message) {
failed_ = true;
error_ = code;
message_ = std::move(message);
}
FrameReader::Next FrameReader::next(Frame* out) {
if (failed_) {
return Next::Fail;
}
const std::size_t available = buffer_.size() - consumed_;
if (available == 0) {
return Next::End;
}
const std::uint8_t* p = buffer_.data() + consumed_;
// The reference implementation rejects a frame whose header does not fit,
// rather than silently resynchronising on the next 0x0B77.
if (available < kHeaderBytes) {
if (finished_) {
fail(JOC_ERR_EAC3_SYNCFRAME, "E-AC-3 syncframe header truncated at end of input");
return Next::Fail;
}
return Next::End;
}
const std::uint16_t syncword = static_cast<std::uint16_t>((static_cast<std::uint16_t>(p[0]) << 8) | p[1]);
if (syncword != kSyncword) {
fail(JOC_ERR_EAC3_SYNCFRAME, "invalid E-AC-3 syncword (silent resynchronisation is not allowed)");
return Next::Fail;
}
// frmsiz: 11 bits spread over the low 3 bits of byte 2 and all of byte 3,
const std::size_t words =
static_cast<std::size_t>(((p[2] & 0x07u) << 8) | p[3]) + 1u;
const std::size_t frame_bytes = words * 2u;
if (frame_bytes > available) {
if (!finished_) {
return Next::End;
}
fail(JOC_ERR_BITSTREAM_TRUNCATED,
"last E-AC-3 syncframe extends past end of input (declared " +
std::to_string(frame_bytes) + " bytes, remaining " +
std::to_string(available) + ")");
return Next::Fail;
}
if (out != nullptr) {
out->data = p;
out->size = frame_bytes;
out->offset = base_offset_ + consumed_;
}
consumed_ += frame_bytes;
stream_offset_ = base_offset_ + consumed_;
++frames_emitted_;
return Next::Ok;
}
joc_error FrameReader::frame_bytes(const std::uint8_t* data, std::size_t size, std::size_t offset,
std::size_t* out_frame_bytes) {
if (data == nullptr || out_frame_bytes == nullptr) {
return JOC_ERR_INVALID_ARGUMENT;
}
if (offset + kHeaderBytes > size) {
return JOC_ERR_EAC3_SYNCFRAME;
}
if (static_cast<std::uint16_t>((static_cast<std::uint16_t>(data[offset]) << 8) | data[offset + 1]) !=
kSyncword) {
return JOC_ERR_EAC3_SYNCFRAME;
}
const std::size_t words =
static_cast<std::size_t>(((data[offset + 2] & 0x07u) << 8) | data[offset + 3]) + 1u;
const std::size_t frame_bytes = words * 2u;
if (offset + frame_bytes > size) {
return JOC_ERR_BITSTREAM_TRUNCATED;
}
*out_frame_bytes = frame_bytes;
return JOC_OK;
}
} // namespace joc::eac3
-68
View File
@@ -1,68 +0,0 @@
#pragma once
#include <cstddef>
#include <cstdint>
#include <string>
#include <vector>
#include "joc_core.h"
namespace joc::eac3 {
struct Frame {
const std::uint8_t* data = nullptr;
std::size_t size = 0;
std::size_t offset = 0; // byte offset of the frame start in the fed stream
};
class FrameReader {
public:
enum class Next {
Ok,
End,
Fail
};
FrameReader() = default;
FrameReader(const std::uint8_t* data, std::size_t size) {
push(data, size);
finish();
}
// Appends bytes to the internal buffer (used in incremental mode).
void push(const std::uint8_t* data, std::size_t size);
// Declares that no further bytes will arrive; a frame that is still
void finish() { finished_ = true; }
Next next(Frame* out);
joc_error error() const { return error_; }
const std::string& error_message() const { return message_; }
std::size_t frames_emitted() const { return frames_emitted_; }
std::size_t stream_offset() const { return stream_offset_; }
// report its declared byte length.
static joc_error frame_bytes(const std::uint8_t* data, std::size_t size, std::size_t offset,
std::size_t* out_frame_bytes);
static constexpr std::uint16_t kSyncword = 0x0B77;
private:
void compact();
void fail(joc_error code, std::string message);
std::vector<std::uint8_t> buffer_;
std::size_t consumed_ = 0; // bytes of buffer_ already turned into frames
std::size_t base_offset_ = 0;
bool finished_ = false;
bool failed_ = false;
joc_error error_ = JOC_OK;
std::string message_;
std::size_t frames_emitted_ = 0;
std::size_t stream_offset_ = 0;
};
} // namespace joc::eac3
+307
View File
@@ -0,0 +1,307 @@
"""从常见 E-AC-3 同步帧直接提取连续 EMDF 容器。
扫描器检查八种全局位对齐,定位 ``0x5838`` 同步字,验证容器长度并解析各
payload config,因此不要求 EMDF 在原始 E-AC-3 文件中按字节对齐。
本模块有意只覆盖“完整 EMDF 容器在一个同步帧中连续出现”的常见情形。不解析
E-AC-3 mantissa,也不重组被音频数据隔开的多个 skip-field 碎片;遇到这种输入会
明确报错,让上层决定是否使用兼容桥。
"""
from dataclasses import dataclass
import csv
import hashlib
from pathlib import Path
import numpy as np
from variant_error import UnsupportedVariantError
SYNCWORD = 0x5838
REQUIRED_JOC_IDS = frozenset((11, 14))
class EmdfError(ValueError):
"""EMDF 或其 E-AC-3 传输结构不符合本实现支持的范围。"""
class BitReader:
"""MSB-first 位读取器;位置以源数据的绝对 bit offset 表示。"""
def __init__(self, data, position=0, limit=None):
self.data = memoryview(data)
self.position = int(position)
self.limit = len(self.data) * 8 if limit is None else int(limit)
def read(self, count):
count = int(count)
if count < 0 or self.position + count > self.limit:
raise EmdfError(f"位流越界 @bit{self.position}, need={count}, limit={self.limit}")
value = 0
while count:
byte_pos = self.position >> 3
removed_left = self.position & 7
take = min(count, 8 - removed_left)
shift = 8 - removed_left - take
value = (value << take) | ((self.data[byte_pos] >> shift) & ((1 << take) - 1))
self.position += take
count -= take
return value
def skip(self, count):
self.read(count)
def read_bytes(self, count):
return bytes(self.read(8) for _ in range(count))
def variable_bits(reader, width, max_groups=8):
"""读取 EMDF ``variable_bits(width)`` 变长整数。"""
value = 0
for _ in range(max_groups):
value += reader.read(width)
more = reader.read(1)
if not more:
return value
value = (value + 1) << width
raise EmdfError(f"variable_bits({width}) 延伸组过多")
@dataclass(frozen=True)
class EmdfContainer:
start_bit: int
raw: bytes
payloads: dict
sample_offsets: dict
def _parse_at(data, start_bit):
"""在已知 syncword 的 bit offset 解析一个 EMDF 容器。"""
reader = BitReader(data, start_bit)
if reader.read(16) != SYNCWORD:
raise EmdfError(f"EMDF syncword 不匹配 @bit{start_bit}")
length = reader.read(16)
body_start = reader.position
body_end = body_start + length * 8
if body_end > reader.limit:
raise EmdfError(f"EMDF 容器越界 @bit{start_bit}: length={length}")
reader.limit = body_end
version = reader.read(2)
if version == 3:
version += variable_bits(reader, 2)
key_id = reader.read(3)
if key_id == 7:
key_id += variable_bits(reader, 3)
# TS 103 420 JOC 使用 version=0/key_id=0;严格限制也能排除音频中的伪 marker。
if version != 0 or key_id != 0:
raise EmdfError(f"不支持的 EMDF version/key_id: {version}/{key_id}")
payloads = {}
sample_offsets = {}
terminated = False
while reader.position + 5 <= body_end:
payload_id = reader.read(5)
if payload_id == 0:
terminated = True
break
if payload_id == 0x1F:
payload_id += variable_bits(reader, 5)
if payload_id in payloads:
raise EmdfError(f"同一 EMDF 容器重复 payload id {payload_id}")
has_sample_offset = bool(reader.read(1))
sample_offset = (reader.read(12) >> 1) if has_sample_offset else 0
if reader.read(1):
variable_bits(reader, 11) # duration
if reader.read(1):
variable_bits(reader, 2) # group id
if reader.read(1):
reader.skip(8) # codec data
if not reader.read(1): # discard_unknown_payload
frame_aligned = False
if not has_sample_offset:
frame_aligned = bool(reader.read(1))
if frame_aligned:
reader.skip(2)
if has_sample_offset or frame_aligned:
reader.skip(7)
payload_size = variable_bits(reader, 8)
if reader.position + payload_size * 8 > body_end:
raise EmdfError(
f"payload id {payload_id} 越界: size={payload_size}, @bit{reader.position}")
payloads[payload_id] = reader.read_bytes(payload_size)
sample_offsets[payload_id] = sample_offset
if not terminated:
raise EmdfError("EMDF 容器缺少 payload id 0 终止符")
total_bytes = 4 + length
raw_reader = BitReader(data, start_bit, start_bit + total_bytes * 8)
raw = raw_reader.read_bytes(total_bytes)
return EmdfContainer(start_bit, raw, payloads, sample_offsets)
def _marker_offsets(data):
"""以 NumPy 批量检查八种位移,返回可能的 0x5838 bit offsets。"""
source = np.frombuffer(data, dtype=np.uint8)
if source.size < 4:
return []
offsets = []
for shift in range(8):
if shift == 0:
aligned = source
else:
aligned = np.bitwise_or(
np.left_shift(source[:-1].astype(np.uint16), shift) & 0xFF,
np.right_shift(source[1:].astype(np.uint16), 8 - shift),
).astype(np.uint8)
hits = np.flatnonzero((aligned[:-1] == 0x58) & (aligned[1:] == 0x38))
offsets.extend(int(hit) * 8 + shift for hit in hits)
return sorted(offsets)
def find_joc_emdf(frame):
"""返回同步帧中唯一、顶层连续且包含 ID11/ID14 的 JOC EMDF 容器。
EMDF payload 是不透明字节串,其中可能自然出现另一个 ``0x5838``。若从这个
内嵌 marker 开始的后续随机位恰好也能通过容器语法探测,它仍不是一个独立的
transport 容器。因此,候选的起点一旦落在较早 JOC 容器的声明范围内,就只把
它记作内嵌伪候选,不参与“多个容器”的判定。
"""
matches = []
offsets = _marker_offsets(frame)
parsed_candidates = []
parse_errors = []
for start_bit in offsets:
try:
container = _parse_at(frame, start_bit)
except EmdfError as exc:
if len(parse_errors) < 8:
parse_errors.append({"start_bit": start_bit, "error": str(exc)})
continue
parsed_candidates.append({
"start_bit": start_bit,
"payload_ids": list(container.payloads),
"payload_lengths": {str(k): len(v) for k, v in container.payloads.items()},
})
if REQUIRED_JOC_IDS.issubset(container.payloads):
matches.append(container)
if not matches:
raise UnsupportedVariantError(
"emdf_transport", "no_contiguous_joc_container",
"同步帧中未找到可连续解析且同时包含 ID11/ID14 的 EMDF 容器",
details={
"syncframe_bytes": len(frame),
"marker_bit_offsets": offsets,
"parsed_candidates": parsed_candidates,
"candidate_parse_errors": parse_errors,
"repair_hint": "检查 EMDF 是否跨多个 audio-block skip field 分片,或 payload config 是否变化",
})
top_level_matches = []
nested_matches = []
for container in sorted(matches, key=lambda item: item.start_bit):
parent = next((candidate for candidate in top_level_matches
if candidate.start_bit < container.start_bit <
candidate.start_bit + len(candidate.raw) * 8), None)
if parent is None:
top_level_matches.append(container)
else:
nested_matches.append({
"start_bit": container.start_bit,
"end_bit": container.start_bit + len(container.raw) * 8,
"parent_start_bit": parent.start_bit,
"parent_end_bit": parent.start_bit + len(parent.raw) * 8,
})
if len(top_level_matches) != 1:
starts = [item.start_bit for item in top_level_matches]
raise UnsupportedVariantError(
"emdf_transport", "multiple_joc_containers",
"同步帧中存在多个可用 JOC EMDF,当前无法自动选择",
details={
"syncframe_bytes": len(frame),
"joc_container_start_bits": starts,
"nested_joc_candidates": nested_matches,
})
return top_level_matches[0]
def parse_container(data):
"""解析从 syncword 开始、已经重新按字节对齐保存的 EMDF 容器。"""
container = _parse_at(data, 0)
if len(container.raw) != len(data):
raise EmdfError(f"EMDF 文件尾有额外数据: parsed={len(container.raw)}, file={len(data)}")
return container
def iter_eac3_frames(data):
"""按 E-AC-3 ``frmsiz`` 遍历同步帧,拒绝静默重同步。"""
pos = 0
while pos < len(data):
if pos + 4 > len(data) or data[pos:pos + 2] != b"\x0b\x77":
raise UnsupportedVariantError(
"eac3_transport", "syncframe_header",
"E-AC-3 同步帧头无效或出现了未处理的子流排列",
details={
"byte_offset": pos,
"remaining_bytes": len(data) - pos,
"next_16_bytes_hex": data[pos:pos + 16].hex(),
})
size = ((((data[pos + 2] & 7) << 8) | data[pos + 3]) + 1) * 2
if pos + size > len(data):
raise UnsupportedVariantError(
"eac3_transport", "truncated_syncframe",
"E-AC-3 末帧长度超过输入剩余数据",
details={
"byte_offset": pos,
"declared_frame_bytes": size,
"remaining_bytes": len(data) - pos,
})
yield data[pos:pos + size]
pos += size
def extract_index(eac3_path, output_dir, max_frames=None):
"""将裸 E-AC-3 的连续 EMDF 保存为 ``frames.csv + emdf/``。"""
output_dir = Path(output_dir)
emdf_dir = output_dir / "emdf"
emdf_dir.mkdir(parents=True, exist_ok=True)
frames = iter_eac3_frames(Path(eac3_path).read_bytes())
rows = []
for frame_number, frame in enumerate(frames):
if max_frames is not None and frame_number >= max_frames:
break
try:
container = find_joc_emdf(frame)
except UnsupportedVariantError as exc:
exc.add_context(frame=frame_number, details={"syncframe_bytes": len(frame)})
raise
except EmdfError as exc:
raise UnsupportedVariantError(
"emdf_transport", "container_syntax",
"EMDF 容器语法无法解析",
frame=frame_number,
details={"syncframe_bytes": len(frame), "parser_error": str(exc)}) from exc
digest = hashlib.sha256(container.raw).hexdigest()
target = emdf_dir / f"{digest}.bin"
if not target.is_file():
target.write_bytes(container.raw)
rows.append({
"frame": frame_number,
"emdf_hash": digest,
"emdf_size": len(container.raw),
"emdf_start_bit": container.start_bit,
"payload_ids": ";".join(str(x) for x in container.payloads),
"error": "",
})
if (frame_number + 1) % 1000 == 0:
print(f"[metadata] {frame_number + 1} frames", flush=True)
if not rows:
raise EmdfError("E-AC-3 输入中没有可处理的同步帧")
with (output_dir / "frames.csv").open("w", encoding="utf-8", newline="") as fp:
fields = ("frame", "emdf_hash", "emdf_size", "emdf_start_bit", "payload_ids", "error")
writer = csv.DictWriter(fp, fieldnames=fields)
writer.writeheader()
writer.writerows(rows)
return output_dir
-305
View File
@@ -1,305 +0,0 @@
#include "emdf/emdf_parser.h"
#include <algorithm>
#include <string>
#include "foundation/bit_reader.h"
namespace joc::emdf {
namespace {
Status syntax_fail(const std::string& message) {
return Status::fail(JOC_ERR_EMDF_SYNTAX, stage::kEmdf, message);
}
Status truncated_fail(const bits::BitReader& reader) {
return Status::fail(JOC_ERR_BITSTREAM_TRUNCATED, stage::kEmdf,
std::string("EMDF bitstream truncated: ") + reader.error_message());
}
} // namespace
Status parse_at(const std::uint8_t* data, std::size_t size, std::size_t start_bit, Container* out) {
if (data == nullptr || out == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kEmdf, "null buffer or output");
}
if (start_bit + 16u > size * 8u) {
return syntax_fail("EMDF syncword position beyond buffer");
}
bits::BitReader reader;
reader.reset(data, size, start_bit);
if (reader.read(16) != kSyncword) {
return syntax_fail("EMDF syncword mismatch at bit " + std::to_string(start_bit));
}
const std::uint32_t length = reader.read(16);
const std::size_t body_start = reader.position();
const std::size_t body_end = body_start + static_cast<std::size_t>(length) * 8u;
if (body_end > reader.limit()) {
return syntax_fail("EMDF container length " + std::to_string(length) +
" exceeds buffer at bit " + std::to_string(start_bit));
}
reader.set_limit_bits(body_end);
std::uint32_t version = reader.read(2);
if (version == 3u) {
std::uint32_t extra = 0;
if (!bits::variable_bits(reader, 2, 8, &extra)) {
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
: syntax_fail(reader.error_message());
}
version += extra;
}
std::uint32_t key_id = reader.read(3);
if (key_id == 7u) {
std::uint32_t extra = 0;
if (!bits::variable_bits(reader, 3, 8, &extra)) {
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
: syntax_fail(reader.error_message());
}
key_id += extra;
}
if (reader.failed()) {
return truncated_fail(reader);
}
// TS 103 420 JOC uses version 0 / key_id 0; the strict check also rejects
// false 0x5838 markers that happen to sit inside audio data.
if (version != 0u || key_id != 0u) {
return syntax_fail("unsupported EMDF version/key_id " + std::to_string(version) + "/" +
std::to_string(key_id));
}
Container container;
container.start_bit = start_bit;
bool terminated = false;
while (reader.position() + 5u <= body_end) {
std::uint32_t payload_id = reader.read(5);
if (reader.failed()) {
return truncated_fail(reader);
}
if (payload_id == 0u) {
terminated = true;
break;
}
if (payload_id == 0x1Fu) {
std::uint32_t extra = 0;
if (!bits::variable_bits(reader, 5, 8, &extra)) {
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
: syntax_fail(reader.error_message());
}
payload_id += extra;
}
for (std::size_t i = 0; i < container.payload_count; ++i) {
if (container.payloads[i].id == static_cast<std::uint8_t>(payload_id)) {
return syntax_fail("duplicate EMDF payload id " + std::to_string(payload_id));
}
}
if (container.payload_count >= kMaxPayloads) {
return syntax_fail("EMDF payload count exceeds " + std::to_string(kMaxPayloads));
}
const std::uint32_t has_sample_offset = reader.read(1);
std::uint16_t sample_offset = 0;
if (has_sample_offset != 0u) {
sample_offset = static_cast<std::uint16_t>(reader.read(12) >> 1);
}
if (reader.read(1) != 0u) {
std::uint32_t ignored = 0;
if (!bits::variable_bits(reader, 11, 8, &ignored)) {
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
: syntax_fail(reader.error_message());
}
}
if (reader.read(1) != 0u) {
std::uint32_t ignored = 0;
if (!bits::variable_bits(reader, 2, 8, &ignored)) {
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
: syntax_fail(reader.error_message());
}
}
if (reader.read(1) != 0u) {
if (!reader.skip(8)) {
return truncated_fail(reader);
}
}
if (reader.read(1) == 0u) {
bool frame_aligned = false;
if (has_sample_offset == 0u) {
frame_aligned = reader.read(1) != 0u;
if (frame_aligned) {
if (!reader.skip(2)) {
return truncated_fail(reader);
}
}
}
if (has_sample_offset != 0u || frame_aligned) {
if (!reader.skip(7)) {
return truncated_fail(reader);
}
}
}
if (reader.failed()) {
return truncated_fail(reader);
}
std::uint32_t payload_size = 0;
if (!bits::variable_bits(reader, 8, 8, &payload_size)) {
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
: syntax_fail(reader.error_message());
}
const std::size_t payload_bits = static_cast<std::size_t>(payload_size) * 8u;
if (reader.position() + payload_bits > body_end) {
return syntax_fail("EMDF payload id " + std::to_string(payload_id) +
" extends past container body (size " + std::to_string(payload_size) +
" at bit " + std::to_string(reader.position()) + ")");
}
Payload& entry = container.payloads[container.payload_count++];
entry.id = static_cast<std::uint8_t>(payload_id);
entry.sample_offset = sample_offset;
entry.bit_offset = reader.position();
entry.size = payload_size;
if (!reader.skip(payload_bits)) {
return truncated_fail(reader);
}
}
if (!terminated) {
return syntax_fail("EMDF container has no payload id 0 terminator");
}
container.raw_size = 4u + static_cast<std::size_t>(length);
*out = container;
return Status::success();
}
void marker_offsets(const std::uint8_t* data, std::size_t size, std::vector<std::size_t>* out) {
out->clear();
if (data == nullptr || size < 4u) {
return;
}
// Eight global bit alignments. For shift != 0 the reference builds an
// (n-1)-byte shifted view and only scans pairs inside it, which is what the
// bounds below reproduce exactly.
for (std::size_t shift = 0; shift < 8u; ++shift) {
const std::size_t aligned_len = (shift == 0u) ? size : (size - 1u);
auto aligned_byte = [&](std::size_t index) -> std::uint8_t {
if (shift == 0u) {
return data[index];
}
const std::uint16_t high = static_cast<std::uint16_t>(data[index]) << shift;
const std::uint16_t low = static_cast<std::uint16_t>(data[index + 1u]) >> (8u - shift);
return static_cast<std::uint8_t>((high | low) & 0xFFu);
};
if (aligned_len < 2u) {
continue;
}
for (std::size_t i = 0; i + 1u < aligned_len; ++i) {
if (aligned_byte(i) == 0x58u && aligned_byte(i + 1u) == 0x38u) {
out->push_back(i * 8u + shift);
}
}
}
std::sort(out->begin(), out->end());
}
Status find_joc_emdf(const std::uint8_t* data, std::size_t size, Container* out) {
if (data == nullptr || out == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kEmdf, "null buffer or output");
}
std::vector<std::size_t> offsets;
marker_offsets(data, size, &offsets);
std::vector<Container> matches;
std::size_t parse_errors = 0;
std::string first_parse_error;
for (const std::size_t start_bit : offsets) {
Container candidate;
const Status status = parse_at(data, size, start_bit, &candidate);
if (!status.ok()) {
++parse_errors;
if (first_parse_error.empty()) {
first_parse_error = "@bit" + std::to_string(start_bit) + ": " + status.message();
}
continue;
}
if (candidate.find(kIdOamd) != nullptr && candidate.find(kIdJoc) != nullptr) {
matches.push_back(candidate);
}
}
if (matches.empty()) {
// Classification stays at the transport level (identical to the reference
// implementation, which raises emdf_transport here), but the underlying
std::string message =
"no contiguous EMDF container carrying ID11+ID14 in this syncframe (markers=" +
std::to_string(offsets.size()) + ", parse_failures=" + std::to_string(parse_errors) +
")";
if (!first_parse_error.empty()) {
message += "; first candidate error " + first_parse_error;
}
return Status::fail(JOC_ERR_EMDF_TRANSPORT, stage::kEmdf, message);
}
std::sort(matches.begin(), matches.end(),
[](const Container& a, const Container& b) { return a.start_bit < b.start_bit; });
// A payload may contain bytes that look like another 0x5838 container; a
std::vector<Container> top_level;
for (const Container& candidate : matches) {
bool nested = false;
for (const Container& parent : top_level) {
if (parent.start_bit < candidate.start_bit &&
candidate.start_bit < parent.start_bit + parent.raw_size * 8u) {
nested = true;
break;
}
}
if (!nested) {
top_level.push_back(candidate);
}
}
if (top_level.size() != 1u) {
return Status::fail(JOC_ERR_EMDF_TRANSPORT, stage::kEmdf,
"multiple top-level JOC EMDF containers (" +
std::to_string(top_level.size()) +
"); automatic selection is not defined");
}
*out = top_level.front();
return Status::success();
}
void extract_container_bytes(const std::uint8_t* data, std::size_t size, const Container& container,
std::vector<std::uint8_t>* out) {
out->assign(container.raw_size, 0u);
if (out->empty()) {
return;
}
bits::BitReader reader;
reader.reset(data, size, container.start_bit);
reader.read_bytes(out->data(), out->size());
}
Status extract_payload_bytes(const std::uint8_t* data, std::size_t size, const Payload& payload,
std::vector<std::uint8_t>* out) {
if (data == nullptr || out == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kEmdf, "null buffer or output");
}
out->assign(payload.size, 0u);
if (out->empty()) {
return Status::success();
}
bits::BitReader reader;
reader.reset(data, size, payload.bit_offset);
if (!reader.read_bytes(out->data(), out->size())) {
return Status::fail(JOC_ERR_BITSTREAM_TRUNCATED, stage::kEmdf,
"payload bytes extend past the syncframe");
}
return Status::success();
}
} // namespace joc::emdf
-57
View File
@@ -1,57 +0,0 @@
#pragma once
#include <cstddef>
#include <cstdint>
#include <vector>
#include "joc_core.h"
#include "foundation/status.h"
namespace joc::emdf {
inline constexpr std::uint16_t kSyncword = 0x5838;
inline constexpr std::uint8_t kIdOamd = 11;
inline constexpr std::uint8_t kIdJoc = 14;
inline constexpr std::size_t kMaxPayloads = JOC_MAX_EMDF_PAYLOADS;
struct Payload {
std::uint8_t id = 0;
std::uint16_t sample_offset = 0;
std::size_t bit_offset = 0; // MSB-first bit position of the payload bytes
std::size_t size = 0; // payload byte count
};
struct Container {
std::size_t start_bit = 0;
std::size_t raw_size = 0;
std::size_t payload_count = 0;
Payload payloads[kMaxPayloads] = {};
const Payload* find(std::uint8_t id) const {
for (std::size_t i = 0; i < payload_count; ++i) {
if (payloads[i].id == id) {
return &payloads[i];
}
}
return nullptr;
}
};
Status parse_at(const std::uint8_t* data, std::size_t size, std::size_t start_bit, Container* out);
// All candidate 0x5838 bit offsets over the eight alignments, ascending.
void marker_offsets(const std::uint8_t* data, std::size_t size, std::vector<std::size_t>* out);
Status find_joc_emdf(const std::uint8_t* data, std::size_t size, Container* out);
// Extract the container's bytes exactly as the bit reader sees them (identical
// to a memcpy for byte-aligned containers).
void extract_container_bytes(const std::uint8_t* data, std::size_t size, const Container& container,
std::vector<std::uint8_t>* out);
// Extract one payload's bytes with the same MSB-first semantics.
Status extract_payload_bytes(const std::uint8_t* data, std::size_t size, const Payload& payload,
std::vector<std::uint8_t>* out);
} // namespace joc::emdf
+76
View File
@@ -0,0 +1,76 @@
"""解包 EVO MD-set evolution 载荷,返回各 payload ID 的字节数据和位偏移。"""
_MARK = "1001001000000"
# 各子载荷字段: (id, 头部前缀, 同步标记, 后缀常量, 尺寸域位数, 是否有转义)
# 头部前缀 = 5 位 id 的 MSB 二进制(id11 字段前另有 5 位容器前导 00000)
# 尺寸域单位 = nibble(4 位)。id14 有转义:9 位值=0 → 再读 9 位 = 字节数。
_LAYOUT = [
(11, "0000001011", "010000000000000", 8, False),
(14, "01110", "01000000000000", 9, True),
(2, "00010", "000100", 7, False),
(1, "00001", "1110000000000000000000000000", 4, False),
(30, "11110", "1110000000000000000000000000", 4, False),
]
def _msb_bits(data: bytes):
return [(x >> (7 - i)) & 1 for x in data for i in range(8)]
def _val(bits, off, n):
v = 0
for b in bits[off:off + n]:
v = (v << 1) | b
return v
class _LooseSkip(Exception):
def __init__(self, ident):
self.ident = ident
def unpack_evolution(payload: bytes, loose=False):
"""解包 evolution 载荷 → (subs, offsets)。subs 键为 id 整数。
loose=True 时对每个 id 的 (前缀+标记+后缀) 全模式做位流重同步扫描
(不同编码流的子载荷次序/内部常量可有合法差异,如 kanata 的 id11)。"""
bits = _msb_bits(payload)
pos = 0
subs = {}
offsets = {}
for ident, pref, suff, sbits, escape in _LAYOUT:
pat = pref + _MARK + suff
if loose:
hit = -1
for i in range(pos, len(bits) - len(pat)):
if ''.join(map(str, bits[i:i + len(pat)])) == pat:
hit = i
break
if hit < 0:
continue
pos = hit + len(pat)
else:
for name, const, expect in (("前缀", bits[pos:pos + len(pref)], pref),
("标记", bits[pos + len(pref):pos + len(pref) + len(_MARK)], _MARK),
("后缀", bits[pos + len(pref) + len(_MARK):
pos + len(pref) + len(_MARK) + len(suff)], suff)):
got = ''.join(map(str, const))
if got != expect:
raise ValueError(f"id={ident}: {name}常量不匹配 @bit{pos} got={got} want={expect}")
pos += len(pref) + len(_MARK) + len(suff)
n_nib = _val(bits, pos, sbits)
pos += sbits
if escape and n_nib == 1:
n_nib = 512 + _val(bits, pos, sbits)
pos += sbits
n_bits = n_nib * 4
body = bits[pos:pos + n_bits]
pos += n_bits
b = bytearray(len(body) // 8)
for i in range(0, len(body) // 8 * 8, 8):
v = 0
for x in body[i:i + 8]:
v = (v << 1) | x
b[i // 8] = v
subs[ident] = bytes(b)
offsets[ident] = pos
return subs, offsets
-30
View File
@@ -1,30 +0,0 @@
#include "foundation/bit_reader.h"
namespace joc::bits {
bool variable_bits(BitReader& reader, unsigned width, unsigned max_groups, std::uint32_t* out_value) {
std::uint32_t value = 0;
for (unsigned group = 0; group < max_groups; ++group) {
value += reader.read(width);
if (reader.failed()) {
return false;
}
const std::uint32_t more = reader.read(1);
if (reader.failed()) {
return false;
}
if (more == 0u) {
if (out_value != nullptr) {
*out_value = value;
}
return true;
}
value = (value + 1u) << width;
}
// Same failure mode as the reference implementation: an extension chain
// that never terminates is a syntax error, not a truncation.
reader.fail(JOC_ERR_EMDF_SYNTAX, "variable_bits extension groups exceeded");
return false;
}
} // namespace joc::bits
-125
View File
@@ -1,125 +0,0 @@
#pragma once
#include <cstddef>
#include <cstdint>
#include "joc_core.h"
namespace joc::bits {
class BitReader {
public:
BitReader() = default;
BitReader(const std::uint8_t* data, std::size_t size) { reset(data, size); }
void reset(const std::uint8_t* data, std::size_t size, std::size_t start_bit = 0) {
data_ = data;
size_bits_ = size * 8u;
pos_ = start_bit;
limit_ = size_bits_;
error_ = JOC_OK;
message_ = "";
}
void set_limit_bits(std::size_t limit_bits) {
limit_ = limit_bits < size_bits_ ? limit_bits : size_bits_;
}
std::size_t position() const { return pos_; }
std::size_t limit() const { return limit_; }
std::size_t remaining_bits() const { return pos_ <= limit_ ? limit_ - pos_ : 0; }
const std::uint8_t* data() const { return data_; }
bool failed() const { return error_ != JOC_OK; }
joc_error error() const { return error_; }
const char* error_message() const { return message_; }
std::uint32_t read(unsigned count) {
if (count == 0) {
return 0;
}
if (!can_read(count)) {
fail_truncated(count);
return 0;
}
std::uint32_t value = 0;
if ((pos_ & 7u) == 0u && count >= 8u) {
while (count >= 8u) {
value = (value << 8) | data_[pos_ >> 3];
pos_ += 8u;
count -= 8u;
}
}
while (count-- > 0u) {
const std::uint32_t bit = (data_[pos_ >> 3] >> (7u - (pos_ & 7u))) & 1u;
value = (value << 1) | bit;
++pos_;
}
return value;
}
std::uint64_t read64(unsigned count) {
if (count <= 32u) {
return static_cast<std::uint64_t>(read(count));
}
const std::uint64_t high = static_cast<std::uint64_t>(read(count - 32u));
const std::uint64_t low = static_cast<std::uint64_t>(read(32u));
return (high << 32) | low;
}
bool skip(std::size_t count) {
if (!can_read(count)) {
fail_truncated(count);
return false;
}
pos_ += count;
return true;
}
bool read_bytes(std::uint8_t* out, std::size_t count) {
if (count == 0) {
return true;
}
if (!can_read(count * 8u)) {
fail_truncated(count * 8u);
return false;
}
for (std::size_t i = 0; i < count; ++i) {
out[i] = static_cast<std::uint8_t>(read(8u));
}
return true;
}
bool can_read(std::size_t count) const {
return !failed() && count <= limit_ && pos_ <= limit_ - count;
}
// semantic check fails, so the reader never continues past it).
void fail(joc_error code, const char* message) {
if (!failed()) {
error_ = code;
message_ = message;
}
}
private:
void fail_truncated(std::size_t count) {
fail(JOC_ERR_BITSTREAM_TRUNCATED, "bit read past end of buffer");
last_request_ = count;
}
const std::uint8_t* data_ = nullptr;
std::size_t size_bits_ = 0;
std::size_t pos_ = 0;
std::size_t limit_ = 0;
std::size_t last_request_ = 0;
joc_error error_ = JOC_OK;
const char* message_ = "";
};
// followed by a continuation bit. Mirrors src/emdf.py:variable_bits().
bool variable_bits(BitReader& reader, unsigned width, unsigned max_groups,
std::uint32_t* out_value);
} // namespace joc::bits
-206
View File
@@ -1,206 +0,0 @@
#include "foundation/fft.h"
#include <cmath>
#include "simd/simd.h"
namespace joc::dsp {
// The dispatched kernels read and write the spectrum as interleaved doubles, and
// an array of std::complex<double> is exactly that: two doubles per element, no
// padding, no vtable.
static_assert(sizeof(Complex) == 2u * sizeof(double), "complex layout");
namespace {
constexpr double kPi = 3.14159265358979323846;
template <typename Container>
void fft_in_place(Container* data, bool inverse) {
const std::size_t count = data->size();
if (count < 2u) {
return;
}
for (std::size_t index = 1u, reversed = 0u; index < count; ++index) {
std::size_t bit = count >> 1u;
for (; (reversed & bit) != 0u; bit >>= 1u) {
reversed ^= bit;
}
reversed ^= bit;
if (index < reversed) {
std::swap((*data)[index], (*data)[reversed]);
}
}
for (std::size_t length = 2u; length <= count; length <<= 1u) {
const double angle = (inverse ? 2.0 : -2.0) * kPi / static_cast<double>(length);
const Complex step(std::cos(angle), std::sin(angle));
for (std::size_t start = 0u; start < count; start += length) {
Complex factor(1.0, 0.0);
for (std::size_t offset = 0u; offset < length / 2u; ++offset) {
const Complex even = (*data)[start + offset];
const Complex odd = (*data)[start + offset + length / 2u] * factor;
(*data)[start + offset] = even + odd;
(*data)[start + offset + length / 2u] = even - odd;
factor *= step;
}
}
}
if (inverse) {
for (Complex& value : *data) {
value /= static_cast<double>(count);
}
}
}
bool is_power_of_two(std::size_t value) { return value != 0u && (value & (value - 1u)) == 0u; }
} // namespace
FftPlan::FftPlan(std::size_t size, bool inverse) : size_(size), inverse_(inverse) {
reverse_.resize(size);
for (std::size_t index = 1u, reversed = 0u; index < size; ++index) {
std::size_t bit = size >> 1u;
for (; (reversed & bit) != 0u; bit >>= 1u) {
reversed ^= bit;
}
reversed ^= bit;
reverse_[index] = reversed;
}
for (std::size_t length = 2u; length <= size; length <<= 1u) {
const double angle = (inverse ? 2.0 : -2.0) * kPi / static_cast<double>(length);
const Complex step(std::cos(angle), std::sin(angle));
stage_begin_.push_back(twiddle_.size());
Complex factor(1.0, 0.0);
for (std::size_t offset = 0u; offset < length / 2u; ++offset) {
twiddle_.push_back(factor);
factor *= step;
}
}
}
// Exactly the operations fft_in_place performs, in the same order, with the
// twiddles read from the precomputed recurrence instead of being re-derived.
template <typename Container>
void FftPlan::apply(Container* data) const {
const std::size_t count = data->size();
if (count < 2u) {
return;
}
const std::size_t* reverse = reverse_.data();
for (std::size_t index = 1u; index < count; ++index) {
const std::size_t reversed = reverse[index];
if (index < reversed) {
std::swap((*data)[index], (*data)[reversed]);
}
}
// The cascade is dispatched for every power-of-two size the kernels can pack
// whole groups into a vector (JOC_SIMD pins one tier for verification). A
// kernel only ever puts independent butterflies in the same vector, so every
// output keeps the operation sequence and the roundings written below; small
// transforms -- and the caller's own table -- keep the portable loop.
if (count >= simd::kMinVectorFftSize && (count & (count - 1u)) == 0u) {
simd::fft_butterflies(reinterpret_cast<double*>(data->data()), count,
reinterpret_cast<const double*>(twiddle_.data()),
stage_begin_.data());
} else {
std::size_t stage = 0u;
for (std::size_t length = 2u; length <= count; length <<= 1u, ++stage) {
const Complex* table = twiddle_.data() + stage_begin_[stage];
for (std::size_t start = 0u; start < count; start += length) {
for (std::size_t offset = 0u; offset < length / 2u; ++offset) {
const Complex even = (*data)[start + offset];
const Complex odd = (*data)[start + offset + length / 2u] * table[offset];
(*data)[start + offset] = even + odd;
(*data)[start + offset + length / 2u] = even - odd;
}
}
}
}
if (inverse_) {
for (Complex& value : *data) {
value /= static_cast<double>(count);
}
}
}
void fft_radix2(std::vector<Complex>* data, const FftPlan& plan) { plan.apply(data); }
void fft_radix2(std::array<Complex, kQmfFftSize>* data, const FftPlan& plan) { plan.apply(data); }
void fft_radix2(std::vector<Complex>* data, bool inverse) { fft_in_place(data, inverse); }
void fft_radix2(std::array<Complex, kQmfFftSize>* data, bool inverse) {
fft_in_place(data, inverse);
}
void fft_any(const std::vector<Complex>& input, bool inverse, std::vector<Complex>* output) {
const std::size_t count = input.size();
if (is_power_of_two(count)) {
*output = input;
fft_radix2(output, inverse);
return;
}
std::size_t size = 1u;
while (size < 2u * count + 1u) {
size <<= 1u;
}
const double sign = inverse ? 1.0 : -1.0;
std::vector<Complex> left(size, Complex(0.0, 0.0));
std::vector<Complex> right(size, Complex(0.0, 0.0));
for (std::size_t index = 0u; index < count; ++index) {
const std::size_t wrapped = (index * index) % (2u * count);
const double angle = kPi * static_cast<double>(wrapped) / static_cast<double>(count);
const Complex chirp(std::cos(angle), sign * std::sin(angle));
left[index] = input[index] * chirp;
right[index] = std::conj(chirp);
if (index != 0u) {
right[size - index] = std::conj(chirp);
}
}
fft_radix2(&left, false);
fft_radix2(&right, false);
for (std::size_t index = 0u; index < size; ++index) {
left[index] *= right[index];
}
fft_radix2(&left, true);
output->resize(count);
for (std::size_t index = 0u; index < count; ++index) {
const std::size_t wrapped = (index * index) % (2u * count);
const double angle = kPi * static_cast<double>(wrapped) / static_cast<double>(count);
const Complex chirp(std::cos(angle), sign * std::sin(angle));
(*output)[index] = left[index] * chirp;
if (inverse) {
(*output)[index] /= static_cast<double>(count);
}
}
}
std::size_t next_fast_len(std::size_t value) {
if (value <= 6u) {
return value;
}
std::size_t best = value;
for (std::size_t power2 = 1u; power2 < value * 2u; power2 *= 2u) {
for (std::size_t power3 = power2; power3 < value * 2u; power3 *= 3u) {
std::size_t power5 = power3;
while (power5 < value) {
power5 *= 5u;
}
best = std::min(best, power5);
if (power3 >= value) {
break;
}
}
}
return best;
}
std::size_t next_power_of_two(std::size_t value) {
std::size_t result = 1u;
while (result < value) {
result <<= 1u;
}
return result;
}
} // namespace joc::dsp
-68
View File
@@ -1,68 +0,0 @@
#pragma once
#include <array>
#include <complex>
#include <cstddef>
#include <vector>
// Complex transforms shared by the HRTF and Rosella DSP cores. The convention is
// NumPy's: the forward transform is unnormalised and the inverse scales by 1/N,
// so a ported pipeline keeps the reference's arithmetic bit for bit.
namespace joc::dsp {
using Complex = std::complex<double>;
inline constexpr std::size_t kQmfFftSize = 128;
// Precomputed radix-2 plan for one size and direction.
//
// The transform derives each butterfly's twiddle by multiplying the previous one
// by the stage step, so the twiddle at offset k is `step` multiplied k times in
// that order, independently of the group. Materialising that exact recurrence --
// and the bit-reversal permutation -- removes one complex multiply and a
// (length/2)-deep serial dependency from every stage's inner loop. The table
// entries are the recurrence's own values, so the transform is bit-identical.
//
// The 128-point cascade is executed by the runtime-dispatched SIMD kernel
// (src/simd/simd.h): it computes independent butterflies in parallel lanes,
// which leaves both the table and every output's summation order untouched.
class FftPlan {
public:
FftPlan(std::size_t size, bool inverse);
std::size_t size() const { return size_; }
bool inverse() const { return inverse_; }
private:
template <typename Container>
void apply(Container* data) const;
friend void fft_radix2(std::vector<Complex>* data, const FftPlan& plan);
friend void fft_radix2(std::array<Complex, kQmfFftSize>* data, const FftPlan& plan);
std::size_t size_ = 0;
bool inverse_ = false;
std::vector<std::size_t> reverse_; // bit-reversal permutation, [size]
std::vector<std::size_t> stage_begin_; // twiddle offset of each stage
std::vector<Complex> twiddle_; // per stage, length/2 entries, concatenated
};
// In-place radix-2 transform; the size must be a power of two.
void fft_radix2(std::vector<Complex>* data, bool inverse);
void fft_radix2(std::array<Complex, kQmfFftSize>* data, bool inverse);
// Plan-driven forms: the plan carries the size and the direction, so a caller that
// transforms the same length repeatedly builds it once.
void fft_radix2(std::vector<Complex>* data, const FftPlan& plan);
void fft_radix2(std::array<Complex, kQmfFftSize>* data, const FftPlan& plan);
// Exact-length transform: radix-2 when the size allows it, Bluestein otherwise.
// scipy/numpy use a mixed-radix transform, which is the same transform.
void fft_any(const std::vector<Complex>& input, bool inverse, std::vector<Complex>* output);
// scipy's next_fast_len: the smallest 5-smooth number that is not smaller.
std::size_t next_fast_len(std::size_t value);
std::size_t next_power_of_two(std::size_t value);
} // namespace joc::dsp
-164
View File
@@ -1,164 +0,0 @@
#include "foundation/fs_utf8.h"
#include <cstring>
#include <fstream>
#include <vector>
#if defined(_WIN32)
#define WIN32_LEAN_AND_MEAN
#define NOMINMAX
#include <windows.h>
#include <shellapi.h>
#include <fcntl.h>
#include <io.h>
#else
#include <cstdlib>
#include <unistd.h>
#endif
namespace joc::fs_utf8 {
namespace fs = std::filesystem;
fs::path to_path(const std::string& utf8) {
return fs::path(std::u8string(reinterpret_cast<const char8_t*>(utf8.data()), utf8.size()));
}
std::string from_path(const fs::path& path) {
const std::u8string text = path.u8string();
return std::string(reinterpret_cast<const char*>(text.data()), text.size());
}
std::FILE* fopen(const std::string& utf8_path, const char* mode) {
#if defined(_WIN32)
const std::wstring wide_mode(mode, mode + std::strlen(mode));
return ::_wfopen(to_path(utf8_path).c_str(), wide_mode.c_str());
#else
return std::fopen(utf8_path.c_str(), mode);
#endif
}
std::FILE* fopen_spool(const std::string& utf8_path) {
#if defined(_WIN32)
// Delete-on-close handed to the CRT: if the process is killed the file goes with
// it, which is what stops an aborted run from leaving hundreds of gigabytes.
HANDLE handle = ::CreateFileW(
to_path(utf8_path).c_str(), GENERIC_READ | GENERIC_WRITE | DELETE,
FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, nullptr, CREATE_ALWAYS,
FILE_ATTRIBUTE_TEMPORARY | FILE_FLAG_DELETE_ON_CLOSE, nullptr);
if (handle == INVALID_HANDLE_VALUE) {
return nullptr;
}
const int descriptor = ::_open_osfhandle(reinterpret_cast<std::intptr_t>(handle), 0);
if (descriptor == -1) {
::CloseHandle(handle);
return nullptr;
}
return ::_fdopen(descriptor, "wb+");
#else
return std::fopen(utf8_path.c_str(), "wb+");
#endif
}
int remove(const std::string& utf8_path) {
#if defined(_WIN32)
return ::_wremove(to_path(utf8_path).c_str());
#else
return std::remove(utf8_path.c_str());
#endif
}
bool exists(const std::string& utf8_path) {
std::error_code error;
return fs::exists(to_path(utf8_path), error);
}
bool is_directory(const std::string& utf8_path) {
std::error_code error;
return fs::is_directory(to_path(utf8_path), error);
}
std::uintmax_t file_size(const std::string& utf8_path, std::error_code& error) {
return fs::file_size(to_path(utf8_path), error);
}
std::string temp_directory() {
std::error_code error;
const fs::path directory = fs::temp_directory_path(error);
return error ? std::string(".") : from_path(directory);
}
std::string executable_path() {
#if defined(_WIN32)
std::vector<wchar_t> buffer(MAX_PATH);
while (true) {
const DWORD written =
::GetModuleFileNameW(nullptr, buffer.data(), static_cast<DWORD>(buffer.size()));
if (written == 0) {
return std::string();
}
if (written < buffer.size()) {
return from_path(fs::path(std::wstring(buffer.data(), written)));
}
buffer.resize(buffer.size() * 2u);
}
#elif defined(__linux__)
std::vector<char> buffer(4096u, '\0');
const ssize_t written = ::readlink("/proc/self/exe", buffer.data(), buffer.size() - 1u);
return written > 0 ? std::string(buffer.data(), static_cast<std::size_t>(written))
: std::string();
#else
return std::string();
#endif
}
std::ifstream open_input(const std::string& utf8_path) {
return std::ifstream(to_path(utf8_path), std::ios::binary);
}
std::ofstream open_output(const std::string& utf8_path) {
return std::ofstream(to_path(utf8_path), std::ios::binary);
}
std::vector<std::string> command_line_arguments(int argc, char** argv) {
#if defined(_WIN32)
(void)argc;
(void)argv;
int count = 0;
LPWSTR* wide = ::CommandLineToArgvW(::GetCommandLineW(), &count);
std::vector<std::string> arguments;
if (wide == nullptr) {
return arguments;
}
arguments.reserve(static_cast<std::size_t>(count));
for (int index = 0; index < count; ++index) {
const std::wstring_view text(wide[index]);
const int size = ::WideCharToMultiByte(CP_UTF8, 0, text.data(),
static_cast<int>(text.size()), nullptr, 0, nullptr,
nullptr);
std::string utf8(static_cast<std::size_t>(size), '\0');
if (size > 0) {
::WideCharToMultiByte(CP_UTF8, 0, text.data(), static_cast<int>(text.size()),
utf8.data(), size, nullptr, nullptr);
}
arguments.push_back(std::move(utf8));
}
::LocalFree(wide);
return arguments;
#else
std::vector<std::string> arguments;
arguments.reserve(static_cast<std::size_t>(argc));
for (int index = 0; index < argc; ++index) {
arguments.emplace_back(argv[index]);
}
return arguments;
#endif
}
void configure_console() {
#if defined(_WIN32)
::SetConsoleOutputCP(CP_UTF8);
#endif
}
} // namespace joc::fs_utf8
-47
View File
@@ -1,47 +0,0 @@
#pragma once
#include <cstdint>
#include <cstdio>
#include <filesystem>
#include <fstream>
#include <string>
#include <system_error>
#include <vector>
// Paths inside this library are always UTF-8, on every platform. std::filesystem
// stores UTF-16 on Windows and bytes elsewhere, and the narrow CRT uses the ANSI
// code page on Windows, so every path crosses into the OS through this shim: that
// is what makes non-ASCII names (Japanese, Chinese, ...) work.
namespace joc::fs_utf8 {
std::filesystem::path to_path(const std::string& utf8);
std::string from_path(const std::filesystem::path& path);
// File handles and queries take a UTF-8 path: _wfopen on Windows, plain calls
// elsewhere. Nothing else in the library may call the narrow CRT with a path.
std::FILE* fopen(const std::string& utf8_path, const char* mode);
// Temporary spool handle: on Windows the file is opened delete-on-close, so killing
// the process removes it instead of leaving a multi-gigabyte leftover behind.
std::FILE* fopen_spool(const std::string& utf8_path);
int remove(const std::string& utf8_path);
bool exists(const std::string& utf8_path);
bool is_directory(const std::string& utf8_path);
std::uintmax_t file_size(const std::string& utf8_path, std::error_code& error);
std::string temp_directory();
// The running executable's own path, UTF-8, or empty when the platform cannot
// report it. Defaults are anchored here so they never depend on the CWD.
std::string executable_path();
// Streams: std::ifstream/ofstream accept a std::filesystem::path, which is the
// portable way to open a UTF-8 path.
std::ifstream open_input(const std::string& utf8_path);
std::ofstream open_output(const std::string& utf8_path);
// Command line arguments as UTF-8. Windows hands the process UTF-16 and the
// narrow CRT would convert it through the ANSI code page, so the wide command
// line is re-parsed there; on POSIX argv is already bytes in the user's locale.
std::vector<std::string> command_line_arguments(int argc, char** argv);
void configure_console();
} // namespace joc::fs_utf8
-25
View File
@@ -1,25 +0,0 @@
// Port of the reference's adm_atmos.q_to_adm_xyz: OAMD Q15 coordinates to the ADM
// cartesian triple. It lives in foundation because both the ADM writer and the
// object position timeline need it, and the timeline must not depend on output.
#pragma once
#include <algorithm>
#include "foundation/py_num.h"
namespace joc::geometry {
inline void q_to_adm_xyz(int q1, int q2, int q3, double* x, double* y, double* z) {
const double posX = std::min(1.0, static_cast<double>(pynum::py_round(
static_cast<double>(q1) * 62.0 / 32767.0)) / 62.0);
const double posY = std::min(1.0, static_cast<double>(pynum::py_round(
static_cast<double>(q2) * 62.0 / 32767.0)) / 62.0);
double posZ = static_cast<double>(pynum::py_round(
static_cast<double>(q3) * 15.0 / 32767.0)) / 15.0;
posZ = std::max(-1.0, std::min(1.0, posZ));
*x = posX * 2.0 - 1.0;
*y = 1.0 - posY * 2.0;
*z = posZ;
}
} // namespace joc::geometry
-227
View File
@@ -1,227 +0,0 @@
#include "foundation/mini_json.h"
#include <cmath>
#include <cstdlib>
namespace joc::json {
namespace {
void skip_space(const std::string& text, std::size_t* index) {
while (*index < text.size() &&
(text[*index] == ' ' || text[*index] == '\t' || text[*index] == '\n' ||
text[*index] == '\r')) {
++(*index);
}
}
bool read_string(const std::string& text, std::size_t* index, std::string* out) {
if (*index >= text.size() || text[*index] != '"') {
return false;
}
++(*index);
out->clear();
while (*index < text.size()) {
const char c = text[*index];
if (c == '\\') {
if (*index + 1 >= text.size()) {
return false;
}
const char escape = text[*index + 1];
*index += 2;
switch (escape) {
case '"': out->push_back('"'); break;
case '\\': out->push_back('\\'); break;
case '/': out->push_back('/'); break;
case 'b': out->push_back('\b'); break;
case 'f': out->push_back('\f'); break;
case 'n': out->push_back('\n'); break;
case 'r': out->push_back('\r'); break;
case 't': out->push_back('\t'); break;
case 'u': {
if (*index + 4 > text.size()) {
return false;
}
unsigned code = 0;
for (int i = 0; i < 4; ++i) {
const char digit = text[*index + static_cast<std::size_t>(i)];
code <<= 4;
if (digit >= '0' && digit <= '9') { code |= static_cast<unsigned>(digit - '0'); }
else if (digit >= 'a' && digit <= 'f') { code |= static_cast<unsigned>(digit - 'a' + 10); }
else if (digit >= 'A' && digit <= 'F') { code |= static_cast<unsigned>(digit - 'A' + 10); }
else { return false; }
}
*index += 4;
if (code < 0x80u) {
out->push_back(static_cast<char>(code));
} else if (code < 0x800u) {
out->push_back(static_cast<char>(0xC0u | (code >> 6)));
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
} else {
out->push_back(static_cast<char>(0xE0u | (code >> 12)));
out->push_back(static_cast<char>(0x80u | ((code >> 6) & 0x3Fu)));
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
}
break;
}
default: return false;
}
continue;
}
if (c == '"') {
++(*index);
return true;
}
out->push_back(c);
++(*index);
}
return false;
}
bool read_compound(const std::string& text, std::size_t* index, std::string* out) {
const char open = text[*index];
const char close = open == '{' ? '}' : ']';
int depth = 0;
const std::size_t start = *index;
while (*index < text.size()) {
const char c = text[*index];
if (c == '"') {
std::string ignored;
if (!read_string(text, index, &ignored)) {
return false;
}
continue;
}
if (c == open) {
++depth;
} else if (c == close) {
--depth;
if (depth == 0) {
++(*index);
*out = text.substr(start, *index - start);
return true;
}
}
++(*index);
}
return false;
}
} // namespace
bool parse_object(const std::string& text, std::vector<Member>* out, std::string* error) {
out->clear();
std::size_t index = 0;
skip_space(text, &index);
if (index >= text.size() || text[index] != '{') {
if (error != nullptr) { *error = "metadata is not a JSON object"; }
return false;
}
++index;
for (;;) {
skip_space(text, &index);
if (index < text.size() && text[index] == '}') {
++index;
break;
}
if (index >= text.size() || text[index] == ',') {
if (index >= text.size()) {
if (error != nullptr) { *error = "unterminated JSON object"; }
return false;
}
++index;
continue;
}
Member member;
if (!read_string(text, &index, &member.key)) {
if (error != nullptr) { *error = "expected a JSON key"; }
return false;
}
skip_space(text, &index);
if (index >= text.size() || text[index] != ':') {
if (error != nullptr) { *error = "expected ':' after JSON key " + member.key; }
return false;
}
++index;
skip_space(text, &index);
if (index >= text.size()) {
if (error != nullptr) { *error = "missing JSON value for " + member.key; }
return false;
}
if (text[index] == '"') {
member.is_string = true;
if (!read_string(text, &index, &member.raw)) {
if (error != nullptr) { *error = "bad JSON string for " + member.key; }
return false;
}
} else if (text[index] == '{' || text[index] == '[') {
if (!read_compound(text, &index, &member.raw)) {
if (error != nullptr) { *error = "bad JSON container for " + member.key; }
return false;
}
} else {
const std::size_t start = index;
while (index < text.size() && text[index] != ',' && text[index] != '}') {
++index;
}
member.raw = text.substr(start, index - start);
while (!member.raw.empty() &&
(member.raw.back() == ' ' || member.raw.back() == '\n' ||
member.raw.back() == '\r' || member.raw.back() == '\t')) {
member.raw.pop_back();
}
}
for (const Member& existing : *out) {
if (existing.key == member.key) {
if (error != nullptr) { *error = "duplicate JSON key " + member.key; }
return false;
}
}
out->push_back(std::move(member));
}
return true;
}
const Member* find(const std::vector<Member>& members, const std::string& key) {
for (const Member& member : members) {
if (member.key == key) {
return &member;
}
}
return nullptr;
}
bool as_string(const Member& member, std::string* out) {
if (!member.is_string || out == nullptr) {
return false;
}
*out = member.raw;
return true;
}
bool as_number(const Member& member, double* out) {
if (member.is_string || out == nullptr) {
return false;
}
char* end = nullptr;
const double value = std::strtod(member.raw.c_str(), &end);
if (end == member.raw.c_str() || !std::isfinite(value)) {
return false;
}
*out = value;
return true;
}
bool as_integer(const Member& member, long long* out) {
double value = 0.0;
if (!as_number(member, &value) || out == nullptr) {
return false;
}
if (value != std::floor(value)) {
return false;
}
*out = static_cast<long long>(value);
return true;
}
} // namespace joc::json
-25
View File
@@ -1,25 +0,0 @@
#pragma once
#include <string>
#include <utility>
#include <vector>
namespace joc::json {
struct Member {
std::string key;
std::string raw;
bool is_string = false;
};
// Parses a top-level JSON object. Rejects non-objects and duplicate keys.
bool parse_object(const std::string& text, std::vector<Member>* out, std::string* error);
const Member* find(const std::vector<Member>& members, const std::string& key);
bool as_string(const Member& member, std::string* out);
bool as_number(const Member& member, double* out);
bool as_integer(const Member& member, long long* out);
} // namespace joc::json
-25
View File
@@ -1,25 +0,0 @@
#pragma once
#include <cfenv>
#include <cmath>
#include <cstdio>
#include <string>
namespace joc::pynum {
inline long long py_round(double value) {
return static_cast<long long>(std::nearbyint(value));
}
inline std::string format_fixed(double value, int decimals) {
char buffer[64];
std::snprintf(buffer, sizeof(buffer), "%.*f", decimals, value);
return std::string(buffer);
}
inline long long trunc_to_ll(double value) {
return static_cast<long long>(value);
}
} // namespace joc::pynum
-165
View File
@@ -1,165 +0,0 @@
#include "foundation/sha256.h"
#include <cstring>
namespace joc::crypto {
namespace {
constexpr std::uint32_t kK[64] = {
0x428a2f98u, 0x71374491u, 0xb5c0fbcfu, 0xe9b5dba5u, 0x3956c25bu, 0x59f111f1u, 0x923f82a4u,
0xab1c5ed5u, 0xd807aa98u, 0x12835b01u, 0x243185beu, 0x550c7dc3u, 0x72be5d74u, 0x80deb1feu,
0x9bdc06a7u, 0xc19bf174u, 0xe49b69c1u, 0xefbe4786u, 0x0fc19dc6u, 0x240ca1ccu, 0x2de92c6fu,
0x4a7484aau, 0x5cb0a9dcu, 0x76f988dau, 0x983e5152u, 0xa831c66du, 0xb00327c8u, 0xbf597fc7u,
0xc6e00bf3u, 0xd5a79147u, 0x06ca6351u, 0x14292967u, 0x27b70a85u, 0x2e1b2138u, 0x4d2c6dfcu,
0x53380d13u, 0x650a7354u, 0x766a0abbu, 0x81c2c92eu, 0x92722c85u, 0xa2bfe8a1u, 0xa81a664bu,
0xc24b8b70u, 0xc76c51a3u, 0xd192e819u, 0xd6990624u, 0xf40e3585u, 0x106aa070u, 0x19a4c116u,
0x1e376c08u, 0x2748774cu, 0x34b0bcb5u, 0x391c0cb3u, 0x4ed8aa4au, 0x5b9cca4fu, 0x682e6ff3u,
0x748f82eeu, 0x78a5636fu, 0x84c87814u, 0x8cc70208u, 0x90befffau, 0xa4506cebu, 0xbef9a3f7u,
0xc67178f2u};
inline std::uint32_t rotr(std::uint32_t value, unsigned count) {
return (value >> count) | (value << (32u - count));
}
} // namespace
void Sha256::reset() {
state_[0] = 0x6a09e667u;
state_[1] = 0xbb67ae85u;
state_[2] = 0x3c6ef372u;
state_[3] = 0xa54ff53au;
state_[4] = 0x510e527fu;
state_[5] = 0x9b05688cu;
state_[6] = 0x1f83d9abu;
state_[7] = 0x5be0cd19u;
bit_count_ = 0;
buffer_used_ = 0;
std::memset(buffer_, 0, sizeof(buffer_));
}
void Sha256::transform(const std::uint8_t block[64]) {
std::uint32_t w[64];
for (unsigned i = 0; i < 16; ++i) {
w[i] = (static_cast<std::uint32_t>(block[i * 4]) << 24) |
(static_cast<std::uint32_t>(block[i * 4 + 1]) << 16) |
(static_cast<std::uint32_t>(block[i * 4 + 2]) << 8) |
static_cast<std::uint32_t>(block[i * 4 + 3]);
}
for (unsigned i = 16; i < 64; ++i) {
const std::uint32_t s0 = rotr(w[i - 15], 7) ^ rotr(w[i - 15], 18) ^ (w[i - 15] >> 3);
const std::uint32_t s1 = rotr(w[i - 2], 17) ^ rotr(w[i - 2], 19) ^ (w[i - 2] >> 10);
w[i] = w[i - 16] + s0 + w[i - 7] + s1;
}
std::uint32_t a = state_[0];
std::uint32_t b = state_[1];
std::uint32_t c = state_[2];
std::uint32_t d = state_[3];
std::uint32_t e = state_[4];
std::uint32_t f = state_[5];
std::uint32_t g = state_[6];
std::uint32_t h = state_[7];
for (unsigned i = 0; i < 64; ++i) {
const std::uint32_t s1 = rotr(e, 6) ^ rotr(e, 11) ^ rotr(e, 25);
const std::uint32_t ch = (e & f) ^ (~e & g);
const std::uint32_t temp1 = h + s1 + ch + kK[i] + w[i];
const std::uint32_t s0 = rotr(a, 2) ^ rotr(a, 13) ^ rotr(a, 22);
const std::uint32_t maj = (a & b) ^ (a & c) ^ (b & c);
const std::uint32_t temp2 = s0 + maj;
h = g;
g = f;
f = e;
e = d + temp1;
d = c;
c = b;
b = a;
a = temp1 + temp2;
}
state_[0] += a;
state_[1] += b;
state_[2] += c;
state_[3] += d;
state_[4] += e;
state_[5] += f;
state_[6] += g;
state_[7] += h;
}
void Sha256::update(const void* data, std::size_t size) {
const std::uint8_t* bytes = static_cast<const std::uint8_t*>(data);
bit_count_ += static_cast<std::uint64_t>(size) * 8u;
while (size > 0) {
const std::size_t space = 64u - buffer_used_;
const std::size_t take = size < space ? size : space;
std::memcpy(buffer_ + buffer_used_, bytes, take);
buffer_used_ += take;
bytes += take;
size -= take;
if (buffer_used_ == 64u) {
transform(buffer_);
buffer_used_ = 0;
}
}
}
void Sha256::finish(std::uint8_t out[32]) {
const std::uint64_t total_bits = bit_count_;
const std::uint8_t pad = 0x80u;
update(&pad, 1);
const std::uint8_t zero = 0x00u;
while (buffer_used_ != 56u) {
update(&zero, 1);
}
std::uint8_t length_bytes[8];
for (unsigned i = 0; i < 8; ++i) {
length_bytes[i] = static_cast<std::uint8_t>((total_bits >> (56u - i * 8u)) & 0xFFu);
}
std::memcpy(buffer_ + buffer_used_, length_bytes, 8);
buffer_used_ += 8;
transform(buffer_);
buffer_used_ = 0;
for (unsigned i = 0; i < 8; ++i) {
out[i * 4 + 0] = static_cast<std::uint8_t>((state_[i] >> 24) & 0xFFu);
out[i * 4 + 1] = static_cast<std::uint8_t>((state_[i] >> 16) & 0xFFu);
out[i * 4 + 2] = static_cast<std::uint8_t>((state_[i] >> 8) & 0xFFu);
out[i * 4 + 3] = static_cast<std::uint8_t>(state_[i] & 0xFFu);
}
}
std::string Sha256::finish_hex() {
std::uint8_t digest[32];
finish(digest);
static const char* kHex = "0123456789abcdef";
std::string text;
text.resize(64);
for (unsigned i = 0; i < 32; ++i) {
text[i * 2] = kHex[(digest[i] >> 4) & 0x0Fu];
text[i * 2 + 1] = kHex[digest[i] & 0x0Fu];
}
return text;
}
std::string sha256_hex(const void* data, std::size_t size) {
Sha256 hash;
hash.update(data, size);
return hash.finish_hex();
}
bool sha256_hex_matches(const void* data, std::size_t size, const std::string& expected_hex) {
if (expected_hex.size() != 64) {
return false;
}
std::string actual = sha256_hex(data, size);
for (std::size_t i = 0; i < 64; ++i) {
char expected = expected_hex[i];
if (expected >= 'A' && expected <= 'F') {
expected = static_cast<char>(expected - 'A' + 'a');
}
if (actual[i] != expected) {
return false;
}
}
return true;
}
} // namespace joc::crypto
-31
View File
@@ -1,31 +0,0 @@
#pragma once
#include <cstddef>
#include <cstdint>
#include <string>
namespace joc::crypto {
class Sha256 {
public:
Sha256() { reset(); }
void reset();
void update(const void* data, std::size_t size);
void finish(std::uint8_t out[32]);
std::string finish_hex();
private:
void transform(const std::uint8_t block[64]);
std::uint32_t state_[8] = {};
std::uint64_t bit_count_ = 0;
std::uint8_t buffer_[64] = {};
std::size_t buffer_used_ = 0;
};
std::string sha256_hex(const void* data, std::size_t size);
bool sha256_hex_matches(const void* data, std::size_t size, const std::string& expected_hex);
} // namespace joc::crypto
-48
View File
@@ -1,48 +0,0 @@
#pragma once
#include <string>
#include <utility>
#include "joc_core.h"
namespace joc {
class Status {
public:
Status() = default;
static Status success() { return Status(); }
static Status fail(joc_error code, std::string stage, std::string message) {
Status s;
s.code_ = code;
s.stage_ = std::move(stage);
s.message_ = std::move(message);
return s;
}
bool ok() const { return code_ == JOC_OK; }
joc_error code() const { return code_; }
const std::string& stage() const { return stage_; }
const std::string& message() const { return message_; }
private:
joc_error code_ = JOC_OK;
std::string stage_ = "none";
std::string message_;
};
// Stage names are kept as plain literals so that C++ and the Python frontend
namespace stage {
inline constexpr const char* kFoundation = "foundation";
inline constexpr const char* kEac3 = "eac3_transport";
inline constexpr const char* kEmdf = "emdf";
inline constexpr const char* kJoc = "joc";
inline constexpr const char* kOamd = "oamd";
inline constexpr const char* kDsp = "dsp";
inline constexpr const char* kRender = "render";
inline constexpr const char* kOutput = "output";
} // namespace stage
} // namespace joc
-401
View File
@@ -1,401 +0,0 @@
#include "hrtf/jochrtf.h"
#include <algorithm>
#include <cmath>
#include <cstdio>
#include <cstring>
#include "foundation/mini_json.h"
#include "foundation/sha256.h"
#include "io/npy.h"
#include "io/zip_reader.h"
namespace joc::hrtf {
namespace {
std::string to_upper(std::string text) {
for (char& c : text) {
if (c >= 'a' && c <= 'z') {
c = static_cast<char>(c - 'a' + 'A');
}
}
return text;
}
bool is_sha256_hex(const std::string& text) {
if (text.size() != 64) {
return false;
}
for (const char c : text) {
const bool digit = c >= '0' && c <= '9';
const bool upper = c >= 'A' && c <= 'F';
if (!digit && !upper) {
return false;
}
}
return true;
}
// json.dumps(list(shape)) as the reference writes it, e.g. "[36, 2, 77]".
std::string shape_json(const std::vector<std::int64_t>& shape) {
std::string text = "[";
for (std::size_t i = 0; i < shape.size(); ++i) {
text += (i == 0 ? "" : ", ");
text += std::to_string(shape[i]);
}
text += "]";
return text;
}
Status hrtf_fail(const std::string& message) {
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender, message);
}
std::string payload_sha256(const std::vector<double>& centers,
const std::vector<double>& coefficients,
const std::vector<double>& delay_coefficients,
const std::vector<double>& delay_bounds) {
crypto::Sha256 hash;
const char prefix[] = "JOC-HRTF-CACHE-PAYLOAD-V1";
hash.update(prefix, sizeof(prefix) - 1);
const std::uint8_t zero = 0;
hash.update(&zero, 1);
struct Entry {
const char* name;
const char* dtype;
const std::vector<double>* values;
std::vector<std::int64_t> shape;
};
const Entry entries[4] = {
{"band_center_frequencies_hz", "<f8", &centers, {kHybridBands}},
{"coefficients", "<c16", &coefficients, {kShTerms, kEars, kHybridBands}},
{"delay_coefficients", "<f8", &delay_coefficients, {kShTerms, kEars}},
{"delay_bounds", "<f8", &delay_bounds, {2, 2}},
};
for (const Entry& entry : entries) {
const std::string name(entry.name);
const std::string dtype(entry.dtype);
const std::string shape = shape_json(entry.shape);
hash.update(name.data(), name.size());
hash.update(&zero, 1);
hash.update(dtype.data(), dtype.size());
hash.update(&zero, 1);
hash.update(shape.data(), shape.size());
hash.update(&zero, 1);
hash.update(entry.values->data(), entry.values->size() * sizeof(double));
}
return hash.finish_hex();
}
} // namespace
Status load_jochrtf(const std::string& path, Field* out) {
if (out == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null field");
}
io::ZipArchive archive;
std::string error;
if (!archive.open(path, &error)) {
return Status::fail(JOC_ERR_HRTF_NOT_FOUND, stage::kRender,
"cannot read compiled HRTF " + path + ": " + error);
}
// Member set must be exactly the five expected names.
static const char* kMembers[5] = {"metadata_json.npy", "band_center_frequencies_hz.npy",
"coefficients.npy", "delay_coefficients.npy",
"delay_bounds.npy"};
if (archive.entries().size() != 5u) {
return hrtf_fail("compiled HRTF cache has an invalid member set (" +
std::to_string(archive.entries().size()) + " members)");
}
for (const char* name : kMembers) {
if (archive.find(name) == nullptr) {
return hrtf_fail(std::string("compiled HRTF cache is missing ") + name);
}
}
auto read_member = [&](const char* name, std::vector<std::uint8_t>* raw,
io::NpyArray* array) -> Status {
if (!archive.read_member(name, raw, &error)) {
return hrtf_fail(std::string("compiled HRTF member ") + name + ": " + error);
}
if (!io::parse_npy(raw->data(), raw->size(), array, &error)) {
return hrtf_fail(std::string("compiled HRTF member ") + name + ": " + error);
}
if (array->fortran_order) {
return hrtf_fail(std::string("compiled HRTF member must be C-contiguous: ") + name);
}
return Status::success();
};
std::vector<std::uint8_t> raw;
io::NpyArray array;
Status status = read_member("metadata_json.npy", &raw, &array);
if (!status.ok()) {
return status;
}
std::string metadata_text;
if (!io::npy_unicode_to_utf8(array, &metadata_text, &error)) {
return hrtf_fail("compiled HRTF metadata: " + error);
}
if (metadata_text.size() > 64u * 1024u) {
return hrtf_fail("compiled HRTF metadata is too large");
}
status = read_member("band_center_frequencies_hz.npy", &raw, &array);
if (!status.ok()) {
return status;
}
if (array.descr != "<f8" || !io::npy_shape_is(array, {kHybridBands})) {
return hrtf_fail("band_center_frequencies_hz must be <f8(77,)");
}
std::vector<double> centers;
io::npy_to_double(array, &centers, &error);
status = read_member("coefficients.npy", &raw, &array);
if (!status.ok()) {
return status;
}
if (array.descr != "<c16" || !io::npy_shape_is(array, {kShTerms, kEars, kHybridBands})) {
return hrtf_fail("coefficients must be <c16(36, 2, 77)");
}
std::vector<double> coefficients;
if (!io::npy_to_double(array, &coefficients, &error)) {
return hrtf_fail("coefficients: " + error);
}
status = read_member("delay_coefficients.npy", &raw, &array);
if (!status.ok()) {
return status;
}
if (array.descr != "<f8" || !io::npy_shape_is(array, {kShTerms, kEars})) {
return hrtf_fail("delay_coefficients must be <f8(36, 2)");
}
std::vector<double> delay_coefficients;
io::npy_to_double(array, &delay_coefficients, &error);
status = read_member("delay_bounds.npy", &raw, &array);
if (!status.ok()) {
return status;
}
if (array.descr != "<f8" || !io::npy_shape_is(array, {2, 2})) {
return hrtf_fail("delay_bounds must be <f8(2, 2)");
}
std::vector<double> delay_bounds;
io::npy_to_double(array, &delay_bounds, &error);
std::vector<json::Member> members;
if (!json::parse_object(metadata_text, &members, &error)) {
return hrtf_fail("compiled HRTF metadata: " + error);
}
auto require_string = [&](const char* key, std::string* value) -> Status {
const json::Member* member = json::find(members, key);
if (member == nullptr || !json::as_string(*member, value)) {
return hrtf_fail(std::string("compiled HRTF metadata is missing ") + key);
}
return Status::success();
};
std::string magic;
std::string schema;
std::string source_sha256;
std::string cache_key;
std::string payload_hash;
std::string delay_source;
status = require_string("magic", &magic);
if (!status.ok()) { return status; }
status = require_string("cache_schema", &schema);
if (!status.ok()) { return status; }
status = require_string("source_sha256", &source_sha256);
if (!status.ok()) { return status; }
status = require_string("cache_key", &cache_key);
if (!status.ok()) { return status; }
status = require_string("payload_sha256", &payload_hash);
if (!status.ok()) { return status; }
status = require_string("delay_source", &delay_source);
if (!status.ok()) { return status; }
if (magic != kMagic) {
return hrtf_fail("compiled HRTF magic mismatch: " + magic);
}
if (schema != kCacheSchema) {
return hrtf_fail("compiled HRTF cache schema mismatch: " + schema);
}
const json::Member* version_member = json::find(members, "format_version");
long long version = -1;
if (version_member == nullptr || !json::as_integer(*version_member, &version)) {
return hrtf_fail("compiled HRTF metadata is missing format_version");
}
if (version != kFormatVersion) {
return Status::fail(JOC_ERR_HRTF_VERSION, stage::kRender,
"unsupported .jochrtf version " + std::to_string(version) +
"; rebuild it from the source SOFA");
}
out->source_sha256 = to_upper(source_sha256);
out->cache_key = to_upper(cache_key);
if (!is_sha256_hex(out->source_sha256)) {
return hrtf_fail("compiled HRTF source_sha256 is not a 64-digit digest");
}
if (!is_sha256_hex(out->cache_key)) {
return hrtf_fail("compiled HRTF cache_key is not a 64-digit digest");
}
const std::string expected = payload_sha256(centers, coefficients, delay_coefficients,
delay_bounds);
if (to_upper(payload_hash) != to_upper(expected)) {
return Status::fail(JOC_ERR_HRTF_HASH, stage::kRender,
"compiled HRTF payload hash mismatch");
}
out->payload_sha256 = to_upper(payload_hash);
const json::Member* radius_member = json::find(members, "measurement_radius_m");
double radius = 0.0;
if (radius_member == nullptr || !json::as_number(*radius_member, &radius) || radius <= 0.0) {
return hrtf_fail("compiled HRTF measurement_radius_m must be a positive number");
}
out->measurement_radius_m = radius;
const json::Member* order_member = json::find(members, "order");
long long order = 0;
if (order_member == nullptr || !json::as_integer(*order_member, &order) || order <= 0 ||
order * order > kShTerms) {
return hrtf_fail("compiled HRTF order is out of range");
}
out->order = order;
for (const double value : coefficients) {
if (!std::isfinite(value)) {
return hrtf_fail("compiled HRTF coefficients contain non-finite values");
}
}
for (const double value : delay_coefficients) {
if (!std::isfinite(value) || std::abs(value) > 48000.0 * 64.0) {
return hrtf_fail("compiled HRTF delay coefficients are out of range");
}
}
for (const double value : delay_bounds) {
if (!std::isfinite(value)) {
return hrtf_fail("compiled HRTF delay bounds contain non-finite values");
}
}
if (delay_bounds.size() == 4u && delay_bounds[0] > delay_bounds[1]) {
return hrtf_fail("compiled HRTF delay bounds are inverted");
}
if (const json::Member* member = json::find(members, "compiler_version")) {
json::as_string(*member, &out->compiler_version);
}
if (const json::Member* member = json::find(members, "phase_policy_version")) {
json::as_string(*member, &out->phase_policy_version);
}
if (const json::Member* member = json::find(members, "sh_convention")) {
json::as_string(*member, &out->sh_convention);
}
if (const json::Member* member = json::find(members, "source_display_name")) {
json::as_string(*member, &out->source_display_name);
}
if (const json::Member* member = json::find(members, "projection_ridge")) {
json::as_number(*member, &out->projection_ridge);
}
if (const json::Member* member = json::find(members, "spherical_harmonic_ridge")) {
json::as_number(*member, &out->spherical_harmonic_ridge);
}
if (const json::Member* member = json::find(members, "fit_report")) {
out->fit_report_json = member->raw;
}
if (const json::Member* member = json::find(members, "filterbank")) {
out->filterbank_json = member->raw;
}
out->delay_source = delay_source;
out->metadata_json = metadata_text;
out->coefficients = std::move(coefficients);
out->delay_coefficients = std::move(delay_coefficients);
out->delay_bounds = std::move(delay_bounds);
out->band_centers_hz = std::move(centers);
return Status::success();
}
Status load_kernels(const std::string& npz_path, Kernels* out) {
if (out == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null kernels");
}
io::ZipArchive archive;
std::string error;
if (!archive.open(npz_path, &error)) {
return Status::fail(JOC_ERR_HRTF_NOT_FOUND, stage::kRender,
"cannot read kernel tables " + npz_path + ": " + error);
}
struct Request {
const char* member;
const char* shape_text;
std::vector<std::int64_t> shape;
};
const Request requests[6] = {
{"qmf_analysis_coefficients.npy", "<f4", {64, 10}},
{"hybrid_analysis_low_kernel.npy", "<f4", {3, 2, 13, 16, 2}},
{"hybrid_synthesis_indices.npy", "<i2", {154, 4}},
{"hybrid_synthesis_values.npy", "<f4", {154}},
{"qmf_synthesis_basis.npy", "<f8", {64, 4, 128}},
{"qmf_synthesis_taps.npy", "<f8", {64, 10, 4}},
};
std::vector<std::uint8_t> raw;
std::vector<std::uint8_t> ordered;
for (const Request& request : requests) {
const std::string name = request.member;
if (!archive.read_member(name, &raw, &error)) {
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
"kernel table member " + name + ": " + error);
}
io::NpyArray array;
if (!io::parse_npy(raw.data(), raw.size(), &array, &error)) {
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
"kernel table member " + name + ": " + error);
}
if (array.descr != request.shape_text || !io::npy_shape_is(array, request.shape)) {
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
"kernel table member " + name + " has an unexpected dtype/shape");
}
// Logical C order: required because the reused kernel indexes the hybrid
// synthesis table row-major while the shipped member is Fortran-order.
if (!io::npy_to_c_order(array, &ordered, &error)) {
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
"kernel table member " + name + ": " + error);
}
const std::size_t count = array.element_count();
if (std::strcmp(request.member, "qmf_analysis_coefficients.npy") == 0) {
std::vector<float> values(count);
std::memcpy(values.data(), ordered.data(), count * sizeof(float));
out->qmf_analysis.assign(values.begin(), values.end());
} else if (std::strcmp(request.member, "hybrid_analysis_low_kernel.npy") == 0) {
std::vector<float> values(count);
std::memcpy(values.data(), ordered.data(), count * sizeof(float));
out->hybrid_low.assign(values.begin(), values.end());
} else if (std::strcmp(request.member, "hybrid_synthesis_indices.npy") == 0) {
out->hybrid_indices.resize(count);
std::memcpy(out->hybrid_indices.data(), ordered.data(), count * sizeof(std::int16_t));
} else if (std::strcmp(request.member, "hybrid_synthesis_values.npy") == 0) {
std::vector<float> values(count);
std::memcpy(values.data(), ordered.data(), count * sizeof(float));
out->hybrid_values.assign(values.begin(), values.end());
} else if (std::strcmp(request.member, "qmf_synthesis_basis.npy") == 0) {
std::memcpy(out->qmf_basis.empty() ? (out->qmf_basis.resize(count), out->qmf_basis.data())
: out->qmf_basis.data(),
ordered.data(), count * sizeof(double));
out->qmf_basis.resize(count);
} else {
out->qmf_taps.resize(count);
std::memcpy(out->qmf_taps.data(), ordered.data(), count * sizeof(double));
}
}
out->hybrid_count = static_cast<std::uint32_t>(out->hybrid_values.size());
if (out->hybrid_count == 0u) {
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
"kernel tables contain no hybrid synthesis entries");
}
return Status::success();
}
} // namespace joc::hrtf
-66
View File
@@ -1,66 +0,0 @@
#pragma once
#include <cstdint>
#include <string>
#include <vector>
#include "foundation/status.h"
namespace joc::hrtf {
inline constexpr int kShTerms = 36;
inline constexpr int kEars = 2;
inline constexpr int kHybridBands = 77;
inline constexpr int kFormatVersion = 1;
inline constexpr const char* kMagic = "JOC-HRTF-CACHE";
inline constexpr const char* kCacheSchema = "joc-compiled-hrtf-v1";
struct Field {
std::vector<double> coefficients;
std::vector<double> delay_coefficients;
std::vector<double> delay_bounds;
std::vector<double> band_centers_hz;
double measurement_radius_m = 1.0;
long long order = 5;
std::string source_sha256;
std::string cache_key;
std::string payload_sha256;
std::string delay_source;
std::string compiler_version;
std::string phase_policy_version;
std::string sh_convention;
std::string filterbank_json;
std::string metadata_json;
// Compile-side metadata, needed to write the cache back out unchanged.
std::string source_display_name;
std::string fit_report_json;
double projection_ridge = 0.0;
double spherical_harmonic_ridge = 0.0;
};
Status load_jochrtf(const std::string& path, Field* out);
// Binaural filterbank kernels, as the reused kernel expects them (C order, the
// exact dtypes of the ABI parameters).
struct Kernels {
std::vector<double> qmf_analysis;
std::vector<double> hybrid_low;
std::vector<std::int16_t> hybrid_indices;
std::vector<double> hybrid_values;
std::vector<double> qmf_basis;
std::vector<double> qmf_taps;
std::uint32_t hybrid_count = 0;
};
// Loads a kernel-table archive. The file path is an override for verification;
// the shipped tables are embedded (see builtin_kernels) so no data file is needed.
// The Fortran-order index member is transposed into C order on purpose: the reused
// kernel indexes the hybrid synthesis table row-major.
Status load_kernels(const std::string& npz_path, Kernels* out);
// The public filterbank tables compiled into the library (identical values to the
// archive the file loader accepts; the unit test checks their hashes).
const Kernels& builtin_kernels();
} // namespace joc::hrtf
File diff suppressed because it is too large Load Diff
-610
View File
@@ -1,610 +0,0 @@
#include "hrtf/public_filterbank.h"
#include <algorithm>
#include <array>
#include <cmath>
#include <cstdint>
#include <cstring>
#include <memory>
#include <string>
#include <utility>
#include <vector>
#include "foundation/fft.h"
#include "simd/simd.h"
namespace joc::hrtf {
namespace {
constexpr double kPi = 3.14159265358979323846;
constexpr std::size_t kQmfLength = dsp::kQmfFftSize;
constexpr int kQmfTaps = 10;
constexpr int kSynthesisRank = 4;
constexpr int kSynthesisTaps = 10;
// ----------------------------------------------------------- filterbank -----
// One shared forward plan for the 128-point QMF transform. The analysis bank runs
// it 2 * slots * channels times per chunk, so the twiddle recurrence is built once
// instead of being re-derived inside every butterfly.
const dsp::FftPlan& qmf_fft_plan() {
static const dsp::FftPlan plan(dsp::kQmfFftSize, false);
return plan;
}
// Public 64-band complex QMF analysis (public_filterbank.QmfAnalysis).
class QmfAnalysis {
public:
static_assert(static_cast<std::size_t>(kQmfBands) == simd::kQmfAnalysisBands,
"the dispatched accumulate is written for this band count");
QmfAnalysis(const Kernels& kernels, std::size_t channels)
: channels_(channels), coefficients_(kernels.qmf_analysis) {
history_.assign(9u * channels_ * kQmfBands, 0.0);
// The polyphase MAC consumes one coefficient per band, so the shipped
// [band][tap] layout makes its inner loop a stride-10 gather. Transposing
// once here turns that into a contiguous AXPY. The coefficient values and
// the accumulation order are untouched, so the sums are bit-identical.
coefficients_by_lag_.resize(static_cast<std::size_t>(kQmfTaps) * kQmfBands);
for (int band = 0; band < kQmfBands; ++band) {
for (int tap = 0; tap < kQmfTaps; ++tap) {
coefficients_by_lag_[static_cast<std::size_t>(tap) * kQmfBands +
static_cast<std::size_t>(band)] =
coefficients_[static_cast<std::size_t>(band) * kQmfTaps +
static_cast<std::size_t>(tap)];
}
}
premultiply_.resize(kQmfBands);
post_.resize(kQmfBands);
even_post_.resize(kQmfBands);
for (int band = 0; band < kQmfBands; ++band) {
const double phase = static_cast<double>(band);
premultiply_[static_cast<std::size_t>(band)] =
std::polar(1.0, -kPi * phase / 128.0);
post_[static_cast<std::size_t>(band)] =
std::polar(1.0, -3.0 * (phase + 0.5) * kPi / 128.0);
even_post_[static_cast<std::size_t>(band)] =
Complex(0.0, band % 2 == 0 ? 1.0 : -1.0);
}
}
void reset() { std::fill(history_.begin(), history_.end(), 0.0); }
// samples: [slots*64, channels]; output: [slots, channels, 64] complex.
void process(const std::vector<double>& samples, std::size_t slots,
std::vector<Complex>* output) {
const std::size_t joined_slots = 9u + slots;
const std::size_t history_size = 9u * channels_ * kQmfBands;
const std::size_t joined_size = joined_slots * channels_ * kQmfBands;
// The joined window is filled completely -- the history lands in its first
// 9 * channels * 64 entries and the new samples in the rest -- so it is a
// reusable scratch buffer rather than a fresh zero-filled allocation. The
// history tail is taken by index instead of from end(), because the buffer may
// be longer than the window this call uses.
if (joined_.size() < joined_size) {
joined_.resize(joined_size);
}
std::copy(history_.begin(), history_.end(), joined_.begin());
std::copy(samples.begin(), samples.begin() + static_cast<std::ptrdiff_t>(slots * channels_ * kQmfBands),
joined_.begin() + static_cast<std::ptrdiff_t>(history_size));
// The two polyphase accumulators are read before they are written, so their
// zero fill is load-bearing and stays; only the per-call allocation goes.
const std::size_t accumulator_size = slots * channels_ * kQmfBands;
if (even_.size() < accumulator_size) {
even_.resize(accumulator_size);
}
if (odd_.size() < accumulator_size) {
odd_.resize(accumulator_size);
}
std::fill(even_.begin(), even_.begin() + static_cast<std::ptrdiff_t>(accumulator_size), 0.0);
std::fill(odd_.begin(), odd_.begin() + static_cast<std::ptrdiff_t>(accumulator_size), 0.0);
// The ten lags are ten accumulate passes over the same 64 bands with one
// shared coefficient row; the bands are independent accumulations of a
// single product each, so they are what the dispatched kernel puts in its
// lanes, and every band keeps the caller's own multiply-then-add.
//
// Slots are processed in blocks, with the lag loop inside: one lag pass
// touches every source row once, so running the ten passes over the whole
// chunk re-reads the joined window ten times -- at 1536 slots that is
// hundreds of megabytes per chunk and the loop ends up bound by memory, not
// by arithmetic. A block's ten lag passes instead slide over a window of
// (block + 9) rows that stays in the second-level cache. Lags still run in
// ascending order inside a block, which is the order each output's sum is
// formed in, so nothing about the arithmetic changes.
constexpr std::size_t kSlotBlock = 32;
for (std::size_t first = 0u; first < slots; first += kSlotBlock) {
const std::size_t block = std::min(kSlotBlock, slots - first);
for (int lag = 0; lag < kQmfTaps; ++lag) {
std::vector<double>& target = (lag % 2 == 0) ? even_ : odd_;
const double* row =
coefficients_by_lag_.data() + static_cast<std::size_t>(lag) * kQmfBands;
const std::size_t source_slot = 9u - static_cast<std::size_t>(lag) + first;
simd::qmf_analysis_taps(
target.data() + first * channels_ * kQmfBands,
joined_.data() + source_slot * channels_ * kQmfBands, row,
block * channels_);
}
}
std::copy(joined_.begin() + static_cast<std::ptrdiff_t>(joined_size - history_size),
joined_.begin() + static_cast<std::ptrdiff_t>(joined_size), history_.begin());
// Every output element is assigned below, so the size is all that has to be
// established; a resize of an already correctly sized buffer touches nothing.
output->resize(slots * channels_ * kQmfBands);
std::array<Complex, dsp::kQmfFftSize> even_spectrum{};
std::array<Complex, dsp::kQmfFftSize> odd_spectrum{};
for (std::size_t slot = 0u; slot < slots; ++slot) {
for (std::size_t channel = 0u; channel < channels_; ++channel) {
const double* even_values = even_.data() + (slot * channels_ + channel) * kQmfBands;
const double* odd_values = odd_.data() + (slot * channels_ + channel) * kQmfBands;
transform(even_values, &even_spectrum);
transform(odd_values, &odd_spectrum);
Complex* destination =
output->data() + (slot * channels_ + channel) * kQmfBands;
for (int band = 0; band < kQmfBands; ++band) {
destination[band] = odd_spectrum[static_cast<std::size_t>(band)] +
even_spectrum[static_cast<std::size_t>(band)] *
even_post_[static_cast<std::size_t>(band)];
}
}
}
}
private:
void transform(const double* values, std::array<Complex, dsp::kQmfFftSize>* spectrum) {
for (int band = 0; band < kQmfBands; ++band) {
(*spectrum)[static_cast<std::size_t>(band)] =
Complex(values[band], 0.0) * premultiply_[static_cast<std::size_t>(band)];
}
for (int index = kQmfBands; index < dsp::kQmfFftSize; ++index) {
(*spectrum)[static_cast<std::size_t>(index)] = Complex(0.0, 0.0);
}
dsp::fft_radix2(spectrum, qmf_fft_plan());
for (int band = 0; band < kQmfBands; ++band) {
(*spectrum)[static_cast<std::size_t>(band)] *= post_[static_cast<std::size_t>(band)];
}
}
std::size_t channels_;
std::vector<double> coefficients_; // [64][10]
std::vector<double> coefficients_by_lag_; // [10][64], the same values transposed
std::vector<double> history_; // [9][channels][64]
std::vector<double> joined_; // scratch, [9 + slots][channels][64]
std::vector<double> even_; // scratch, [slots][channels][64], zeroed per call
std::vector<double> odd_; // scratch, [slots][channels][64], zeroed per call
std::vector<Complex> premultiply_;
std::vector<Complex> post_;
std::vector<Complex> even_post_;
};
// Sparse 64-QMF to 77-hybrid analysis (public_filterbank.HybridAnalysis).
class HybridAnalysis {
public:
HybridAnalysis(const Kernels& kernels, std::size_t channels)
: channels_(channels), low_kernel_(kernels.hybrid_low) {
history_.assign(12u * channels_ * 3u * 2u, 0.0);
high_history_.assign(6u * channels_ * 61u, Complex(0.0, 0.0));
// The dispatched join walks one term at a time and adds its 32 weights to
// 32 outputs, so the shipped [tap][band][component] table is regrouped to
// the term order the caller accumulates in. Same weights, same order.
const std::size_t outputs = simd::kHybridOutputs;
low_by_term_.resize(simd::kHybridTerms * outputs);
for (int lag = 0; lag < 13; ++lag) {
for (int point = 0; point < 3; ++point) {
for (int input = 0; input < 2; ++input) {
const std::size_t term =
(static_cast<std::size_t>(lag) * 3u + static_cast<std::size_t>(point)) * 2u +
static_cast<std::size_t>(input);
const std::size_t source = (static_cast<std::size_t>(point) * 2u +
static_cast<std::size_t>(input)) * 13u +
static_cast<std::size_t>(lag);
for (std::size_t output = 0u; output < outputs; ++output) {
low_by_term_[term * outputs + output] =
low_kernel_[source * outputs + output];
}
}
}
}
low_values_.resize(simd::kHybridJoinBlock * simd::kHybridTerms);
low_out_.resize(simd::kHybridJoinBlock * outputs);
}
void reset() {
std::fill(history_.begin(), history_.end(), 0.0);
std::fill(high_history_.begin(), high_history_.end(), Complex(0.0, 0.0));
}
// qmf: [slots, channels, 64]; output: [slots, channels, 77] complex.
void process(const std::vector<Complex>& qmf, std::size_t slots,
std::vector<Complex>* output) {
const std::size_t joined_slots = 12u + slots;
const std::size_t history_size = 12u * channels_ * 6u;
const std::size_t joined_size = joined_slots * channels_ * 3u * 2u;
// Both the joined window and the pending high-band history are written in full
// before they are read, so they are reused scratch buffers; the history tail is
// taken by index because the buffer can be longer than this call's window.
if (joined_.size() < joined_size) {
joined_.resize(joined_size);
}
std::copy(history_.begin(), history_.end(), joined_.begin());
for (std::size_t slot = 0u; slot < slots; ++slot) {
for (std::size_t channel = 0u; channel < channels_; ++channel) {
const Complex* source = qmf.data() + (slot * channels_ + channel) * kQmfBands;
double* destination =
joined_.data() + ((12u + slot) * channels_ + channel) * 6u;
for (int band = 0; band < 3; ++band) {
destination[static_cast<std::size_t>(band) * 2u] = source[band].real();
destination[static_cast<std::size_t>(band) * 2u + 1u] = source[band].imag();
}
}
}
// The low bands are accumulated in a register block and written straight into
// the output, and the high bands are written by the pass below; between them
// every one of the 77 bands is assigned, so only the size has to be set.
output->resize(slots * channels_ * kHybridBands);
// The thirteen taps are summed in a per-output register block and the low
// bands are written straight into the output. Keeping a separate low plane
// and then copying it into the output re-streams tens of megabytes per chunk
// for nothing, and only the first kHybridLow bands are ever touched. The
// join itself is dispatched (see src/simd/simd.h): the 32 outputs of a
// row are 32 independent accumulations over the same 78 terms, which is what
// shares a vector. Every lane keeps the caller's term order -- lag, then
// point, then input -- and its two roundings, and skips exactly the terms
// this loop skips. Rows are staged in blocks so the gathered values do not
// spill out of the first-level cache.
const std::size_t hybrid_rows = slots * channels_;
const std::size_t block = simd::kHybridJoinBlock;
const std::size_t terms = simd::kHybridTerms;
for (std::size_t first = 0u; first < hybrid_rows; first += block) {
const std::size_t count = std::min(block, hybrid_rows - first);
for (std::size_t index = 0u; index < count; ++index) {
const std::size_t row = first + index;
const std::size_t slot = row / channels_;
const std::size_t channel = row % channels_;
double* staged = low_values_.data() + index * terms;
for (int lag = 0; lag < 13; ++lag) {
const std::size_t source_slot = 12u - static_cast<std::size_t>(lag) + slot;
const double* source =
joined_.data() + (source_slot * channels_ + channel) * 6u;
for (int point = 0; point < 3; ++point) {
for (int input = 0; input < 2; ++input) {
staged[(static_cast<std::size_t>(lag) * 3u +
static_cast<std::size_t>(point)) * 2u +
static_cast<std::size_t>(input)] =
source[static_cast<std::size_t>(point) * 2u +
static_cast<std::size_t>(input)];
}
}
}
}
simd::hybrid_low_join(low_values_.data(), low_by_term_.data(),
low_out_.data(), count);
for (std::size_t index = 0u; index < count; ++index) {
Complex* destination = output->data() + (first + index) * kHybridBands;
const double* values = low_out_.data() + index * simd::kHybridOutputs;
for (int band = 0; band < kHybridLow; ++band) {
destination[band] = Complex(values[static_cast<std::size_t>(band) * 2u],
values[static_cast<std::size_t>(band) * 2u + 1u]);
}
}
}
std::copy(joined_.begin() + static_cast<std::ptrdiff_t>(joined_size - history_size),
joined_.begin() + static_cast<std::ptrdiff_t>(joined_size), history_.begin());
// The high bands pass through unchanged but delayed by the six slots of
// history the reference concatenates in front of them. Only the last six
// entries of that concatenation survive into high_history_, so a six-entry
// register replaces the (6 + slots) plane and its full copy. Note the
// output reads the concatenation at index `slot`, not `6 + slot`, so the
// first six output slots come from the history: that offset is part of the
// current output and is preserved verbatim.
for (std::size_t slot = 0u; slot < slots; ++slot) {
for (std::size_t channel = 0u; channel < channels_; ++channel) {
Complex* destination = output->data() +
(slot * channels_ + channel) * kHybridBands + kHybridLow;
if (slot < 6u) {
const Complex* source =
high_history_.data() + (slot * channels_ + channel) * 61u;
for (int band = 0; band < 61; ++band) {
destination[band] = source[band];
}
} else {
const Complex* source =
qmf.data() + ((slot - 6u) * channels_ + channel) * kQmfBands;
for (int band = 3; band < kQmfBands; ++band) {
destination[static_cast<std::size_t>(band - 3)] = source[band];
}
}
}
}
// Every entry of the pending high-band history is written here, so it is a
// reusable scratch buffer; the copy into the live history is kept as it was.
if (next_high_history_.size() < 6u * channels_ * 61u) {
next_high_history_.resize(6u * channels_ * 61u);
}
for (std::size_t entry = 0u; entry < 6u; ++entry) {
const std::size_t combined = slots + entry;
for (std::size_t channel = 0u; channel < channels_; ++channel) {
Complex* destination =
next_high_history_.data() + (entry * channels_ + channel) * 61u;
if (combined < 6u) {
const Complex* source =
high_history_.data() + (combined * channels_ + channel) * 61u;
for (int band = 0; band < 61; ++band) {
destination[band] = source[band];
}
} else {
const Complex* source =
qmf.data() + ((combined - 6u) * channels_ + channel) * kQmfBands;
for (int band = 3; band < kQmfBands; ++band) {
destination[static_cast<std::size_t>(band - 3)] = source[band];
}
}
}
}
std::copy(next_high_history_.begin(), next_high_history_.end(), high_history_.begin());
}
private:
std::size_t channels_;
std::vector<double> low_kernel_; // [3][2][13][16][2]
std::vector<double> low_by_term_; // [78][32], the same weights in the caller's term order
std::vector<double> low_values_; // scratch, [block][78]
std::vector<double> low_out_; // scratch, [block][32]
std::vector<double> history_; // [12][channels][3][2]
std::vector<Complex> high_history_; // [6][channels][61]
std::vector<double> joined_; // scratch, [12 + slots][channels][3][2]
std::vector<Complex> next_high_history_; // scratch, [6][channels][61]
};
// Instantaneous sparse 77-hybrid to 64-QMF synthesis map.
class HybridSynthesis {
public:
explicit HybridSynthesis(const Kernels& kernels) {
const std::size_t rows = kernels.hybrid_indices.size() / 4u;
mapping_.reserve(rows);
for (std::size_t index = 0u; index < rows; ++index) {
Entry entry;
for (int field = 0; field < 4; ++field) {
entry.index[static_cast<std::size_t>(field)] =
kernels.hybrid_indices[index * 4u + static_cast<std::size_t>(field)];
}
entry.gain = kernels.hybrid_values[index];
mapping_.push_back(entry);
}
}
// hybrid: [slots, channels, 77]; output: [slots, channels, 64] complex.
// The sparse map moves a real or imaginary part of one band into a real or
// imaginary part of another, so the two components are accumulated apart.
void process(const std::vector<Complex>& hybrid, std::size_t slots, std::size_t channels,
std::vector<Complex>* output) const {
const std::size_t rows = slots * channels;
std::vector<double> real(rows * kQmfBands, 0.0);
std::vector<double> imaginary(rows * kQmfBands, 0.0);
for (std::size_t slot = 0u; slot < slots; ++slot) {
for (std::size_t channel = 0u; channel < channels; ++channel) {
const std::size_t row = slot * channels + channel;
const Complex* source = hybrid.data() + row * kHybridBands;
for (const Entry& entry : mapping_) {
const double value = entry.index[1] == 0u ? source[entry.index[0]].real()
: source[entry.index[0]].imag();
if (value == 0.0) {
continue;
}
double* destination =
(entry.index[3] == 0u ? real.data() : imaginary.data()) + row * kQmfBands;
destination[entry.index[2]] += value * entry.gain;
}
}
}
// Every output element is assigned from the two accumulators below, so the
// zero fill that `assign` performed was dead; only the size is needed.
output->resize(rows * kQmfBands);
for (std::size_t index = 0u; index < output->size(); ++index) {
(*output)[index] = Complex(real[index], imaginary[index]);
}
}
private:
struct Entry {
std::size_t index[4] = {0u, 0u, 0u, 0u};
double gain = 0.0;
};
std::vector<Entry> mapping_;
};
// Rank-4 64-band synthesis.
class QmfSynthesis {
public:
QmfSynthesis(const Kernels& kernels, std::size_t channels)
: channels_(channels), basis_(kernels.qmf_basis), taps_(kernels.qmf_taps) {
history_.assign(9u * channels_ * kQmfBands * kSynthesisRank, 0.0);
// The dispatched basis kernel reads the four ranks of one (band, tap) as
// one vector, so the shipped [band][rank][tap] table is reordered once
// here. The weights are the same doubles, only their order differs.
const std::size_t bands = static_cast<std::size_t>(kQmfBands);
const std::size_t ranks = static_cast<std::size_t>(kSynthesisRank);
const std::size_t taps = dsp::kQmfFftSize;
basis_by_tap_.resize(bands * taps * ranks);
for (std::size_t band = 0u; band < bands; ++band) {
for (std::size_t tap = 0u; tap < taps; ++tap) {
for (std::size_t rank = 0u; rank < ranks; ++rank) {
basis_by_tap_[(band * taps + tap) * ranks + rank] =
basis_[(band * ranks + rank) * taps + tap];
}
}
}
}
void reset() { std::fill(history_.begin(), history_.end(), 0.0); }
// qmf: [slots, channels, 64]; output: [slots*64, channels] real.
void process(const std::vector<Complex>& qmf, std::size_t slots, std::vector<double>* output) {
const std::size_t rows = slots * channels_;
// [row][band][component] staging for the basis application. Both staging
// planes and the joined window are reusable scratch: every element of each is
// written before it is read, so the buffers are sized once and kept instead of
// being allocated and zero-filled on every call.
const std::size_t flat_size = rows * dsp::kQmfFftSize;
if (flat_.size() < flat_size) {
flat_.resize(flat_size);
}
for (std::size_t row = 0u; row < rows; ++row) {
for (int band = 0; band < kQmfBands; ++band) {
flat_[row * dsp::kQmfFftSize + static_cast<std::size_t>(band) * 2u] =
qmf[row * kQmfBands + static_cast<std::size_t>(band)].real();
flat_[row * dsp::kQmfFftSize + static_cast<std::size_t>(band) * 2u + 1u] =
qmf[row * kQmfBands + static_cast<std::size_t>(band)].imag();
}
}
// The sums are written straight into the joined window: the destination index
// is known up front, the summation order is untouched, and the application
// itself is dispatched -- the four ranks of a band are four independent dot
// products over the same 128 values, so they share a vector while every lane
// keeps the tap order and the two roundings of `sum +=`.
const std::size_t history_size = 9u * channels_ * kQmfBands * kSynthesisRank;
const std::size_t joined_size = history_size + rows * kQmfBands * kSynthesisRank;
if (joined_.size() < joined_size) {
joined_.resize(joined_size);
}
std::copy(history_.begin(), history_.end(), joined_.begin());
simd::qmf_synthesis_basis(flat_.data(), basis_by_tap_.data(),
joined_.data() + history_size, rows);
std::copy(joined_.begin() + static_cast<std::ptrdiff_t>(joined_size - history_size),
joined_.begin() + static_cast<std::ptrdiff_t>(joined_size), history_.begin());
output->assign(rows * kQmfBands, 0.0);
for (int lag = 0; lag < kSynthesisTaps; ++lag) {
for (std::size_t slot = 0u; slot < slots; ++slot) {
const std::size_t source_slot = 9u - static_cast<std::size_t>(lag) + slot;
for (std::size_t channel = 0u; channel < channels_; ++channel) {
const double* source =
joined_.data() +
(source_slot * channels_ + channel) * kQmfBands * kSynthesisRank;
double* destination =
output->data() + (slot * channels_ + channel) * kQmfBands;
for (int band = 0; band < kQmfBands; ++band) {
double sum = 0.0;
for (int rank = 0; rank < kSynthesisRank; ++rank) {
sum += source[static_cast<std::size_t>(band) * kSynthesisRank +
static_cast<std::size_t>(rank)] *
taps_[(static_cast<std::size_t>(band) * kSynthesisTaps +
static_cast<std::size_t>(lag)) * kSynthesisRank +
static_cast<std::size_t>(rank)];
}
destination[band] += sum;
}
}
}
}
}
private:
std::size_t channels_;
std::vector<double> basis_; // [64][4][128]
std::vector<double> basis_by_tap_; // [64][128][4], the same weights transposed
std::vector<double> taps_; // [64][10][4]
std::vector<double> history_; // [9][channels][64][4]
std::vector<double> flat_; // scratch, [rows][128], fully written per call
std::vector<double> joined_; // scratch, [9 + slots][channels][64][4]
};
// public_filterbank.PublicAnalysis77.process: [N, channels] -> [N/64, channels, 77].
void analysis_77(const std::vector<double>& samples, std::size_t slots,
std::size_t channels, QmfAnalysis& qmf,
HybridAnalysis& hybrid_analysis, std::vector<Complex>* hybrid) {
std::vector<double> hops(slots * channels * kQmfHop, 0.0);
for (std::size_t slot = 0u; slot < slots; ++slot) {
for (std::size_t channel = 0u; channel < channels; ++channel) {
for (int index = 0; index < kQmfHop; ++index) {
hops[(slot * channels + channel) * kQmfHop + static_cast<std::size_t>(index)] =
samples[(slot * kQmfHop + static_cast<std::size_t>(index)) * channels + channel];
}
}
}
std::vector<Complex> qmf_bands;
qmf.process(hops, slots, &qmf_bands);
hybrid_analysis.process(qmf_bands, slots, hybrid);
}
// public_filterbank.PublicSynthesis77.process: [slots, channels, 77] -> [slots*64, channels].
void synthesis_77(const std::vector<Complex>& hybrid, std::size_t slots,
std::size_t channels, const HybridSynthesis& synthesis,
QmfSynthesis& qmf, std::vector<double>* time) {
std::vector<Complex> qmf_bands;
synthesis.process(hybrid, slots, channels, &qmf_bands);
std::vector<double> samples;
qmf.process(qmf_bands, slots, &samples);
// The reference transposes (slots, channels, 64) to sample-major output.
time->assign(samples.size(), 0.0);
for (std::size_t slot = 0u; slot < slots; ++slot) {
for (std::size_t channel = 0u; channel < channels; ++channel) {
for (int band = 0; band < kQmfBands; ++band) {
(*time)[(slot * kQmfHop + static_cast<std::size_t>(band)) * channels + channel] =
samples[(slot * channels + channel) * kQmfBands + static_cast<std::size_t>(band)];
}
}
}
}
} // namespace
struct PublicFilterbank::Impl {
Impl(const Kernels& kernels, std::size_t channels)
: channels(channels), qmf(kernels, channels), hybrid_analysis(kernels, channels),
hybrid_synthesis(kernels), qmf_synthesis(kernels, channels) {}
std::size_t channels;
QmfAnalysis qmf;
HybridAnalysis hybrid_analysis;
HybridSynthesis hybrid_synthesis;
QmfSynthesis qmf_synthesis;
};
PublicFilterbank::PublicFilterbank(const Kernels& kernels, std::size_t channels)
: impl_(std::make_unique<Impl>(kernels, channels)) {}
PublicFilterbank::~PublicFilterbank() = default;
void PublicFilterbank::reset() {
impl_->qmf.reset();
impl_->hybrid_analysis.reset();
impl_->qmf_synthesis.reset();
}
void PublicFilterbank::analyze_full_rate(const std::vector<double>& samples, std::size_t slots,
std::vector<Complex>* hybrid) {
analysis_77(samples, slots, impl_->channels, impl_->qmf, impl_->hybrid_analysis, hybrid);
}
void PublicFilterbank::synthesize_full_rate(const std::vector<Complex>& hybrid, std::size_t slots,
std::vector<double>* time) {
synthesis_77(hybrid, slots, impl_->channels, impl_->hybrid_synthesis, impl_->qmf_synthesis,
time);
}
void PublicFilterbank::analyze_qmf(const std::vector<double>& hops, std::size_t slots,
std::vector<Complex>* qmf) {
impl_->qmf.process(hops, slots, qmf);
}
void PublicFilterbank::analyze_hybrid(const std::vector<Complex>& qmf, std::size_t slots,
std::vector<Complex>* hybrid) {
impl_->hybrid_analysis.process(qmf, slots, hybrid);
}
void PublicFilterbank::synthesize_hybrid(const std::vector<Complex>& hybrid, std::size_t slots,
std::vector<Complex>* qmf) {
impl_->hybrid_synthesis.process(hybrid, slots, impl_->channels, qmf);
}
void PublicFilterbank::synthesize_qmf(const std::vector<Complex>& qmf, std::size_t slots,
std::vector<double>* time) {
impl_->qmf_synthesis.process(qmf, slots, time);
}
} // namespace joc::hrtf
-55
View File
@@ -1,55 +0,0 @@
#pragma once
#include <complex>
#include <cstddef>
#include <memory>
#include <vector>
#include "hrtf/jochrtf.h"
// Public 64-QMF / 77-hybrid filterbank, shared by the SOFA field compiler and the
// Rosella renderer (upstream public_filterbank.py and rosella_filterbank.py are
// the same bank). Everything is float64/complex128, as the reference computes it,
// and the stateful half-steps are exposed because Rosella drives them directly.
namespace joc::hrtf {
inline constexpr int kQmfBands = 64;
inline constexpr int kQmfHop = 64;
inline constexpr int kHybridLow = 16;
inline constexpr int kHybridBandCount = 77;
inline constexpr int kLatencySamples = 961;
using Complex = std::complex<double>;
class PublicFilterbank {
public:
PublicFilterbank(const Kernels& kernels, std::size_t channels);
~PublicFilterbank();
PublicFilterbank(const PublicFilterbank&) = delete;
PublicFilterbank& operator=(const PublicFilterbank&) = delete;
void reset();
// Full-rate [slots*64, channels] -> hybrid [slots, channels, 77].
void analyze_full_rate(const std::vector<double>& samples, std::size_t slots,
std::vector<Complex>* hybrid);
// Hybrid [slots, channels, 77] -> full-rate [slots*64, channels].
void synthesize_full_rate(const std::vector<Complex>& hybrid, std::size_t slots,
std::vector<double>* time);
// The stateful half-steps, in the order the reference runs them.
void analyze_qmf(const std::vector<double>& hops, std::size_t slots,
std::vector<Complex>* qmf);
void analyze_hybrid(const std::vector<Complex>& qmf, std::size_t slots,
std::vector<Complex>* hybrid);
void synthesize_hybrid(const std::vector<Complex>& hybrid, std::size_t slots,
std::vector<Complex>* qmf);
void synthesize_qmf(const std::vector<Complex>& qmf, std::size_t slots,
std::vector<double>* time);
private:
struct Impl;
std::unique_ptr<Impl> impl_;
};
} // namespace joc::hrtf
-535
View File
@@ -1,535 +0,0 @@
#include "hrtf/rosella_model.h"
#include <algorithm>
#include <cmath>
#include <cstring>
#include <fstream>
#include <string>
#include <utility>
#include <vector>
#include "foundation/fs_utf8.h"
#include "foundation/mini_json.h"
#include "foundation/sha256.h"
namespace joc::hrtf {
namespace {
// The model's fixed-point lane scale: every stored value is a Q15 integer.
constexpr float kQ15 = 1.0f / 32768.0f;
Status model_fail(joc_error code, const std::string& message) {
return Status::fail(code, stage::kRender, message);
}
float q15(std::int32_t value) { return static_cast<float>(value) * kQ15; }
float q15_exp(std::int32_t value, int exponent) {
return q15(value) * static_cast<float>(std::ldexp(1.0, exponent));
}
std::uint16_t low16(std::int32_t value) {
return static_cast<std::uint16_t>(static_cast<std::uint32_t>(value) & 0xFFFFu);
}
std::string trim(const std::string& text) {
const std::size_t begin = text.find_first_not_of(" \t\r\n");
const std::size_t end = text.find_last_not_of(" \t\r\n");
return begin == std::string::npos ? std::string() : text.substr(begin, end - begin + 1u);
}
// The lane array is read straight out of the JSON text: it is one flat list of
// integers, and building a 15691-node DOM for it would only cost time.
bool parse_int_array(const std::string& raw, std::vector<std::int32_t>* out, std::string* error) {
out->clear();
const char* cursor = raw.c_str();
const char* end = cursor + raw.size();
while (cursor < end && *cursor != '[') {
++cursor;
}
if (cursor == end) {
*error = "rosella_coefficients must be a JSON array";
return false;
}
++cursor;
while (cursor < end) {
while (cursor < end && (*cursor == ' ' || *cursor == '\t' || *cursor == '\r' ||
*cursor == '\n' || *cursor == ',')) {
++cursor;
}
if (cursor >= end) {
break;
}
if (*cursor == ']') {
return true;
}
const bool negative = *cursor == '-';
if (negative) {
++cursor;
}
if (cursor >= end || *cursor < '0' || *cursor > '9') {
*error = "rosella_coefficients contains a non-integer value";
return false;
}
long long value = 0;
while (cursor < end && *cursor >= '0' && *cursor <= '9') {
value = value * 10 + (*cursor - '0');
if (value > (1ll << 40)) {
*error = "rosella_coefficients value is out of range";
return false;
}
++cursor;
}
// A fractional part or an exponent means the value is not an exact integer.
if (cursor < end && (*cursor == '.' || *cursor == 'e' || *cursor == 'E')) {
*error = "rosella_coefficients contains a non-integer value";
return false;
}
if (negative) {
value = -value;
}
if (value < -(1ll << 31) || value > (1ll << 31) - 1) {
*error = "rosella_coefficients value is outside signed int32";
return false;
}
out->push_back(static_cast<std::int32_t>(value));
}
*error = "rosella_coefficients array is truncated";
return false;
}
struct RpHeader {
std::uint16_t stored_checksum = 0;
std::uint16_t computed_checksum = 0;
bool checksum_valid = false;
bool table_a_present = false;
bool table_b_present = false;
bool table_c_present = false;
int table_a_dimension = 0;
int table_a_option = 0;
int table_a_extra = 0;
int table_b_dimension = 0;
int table_b_extra = 0;
int table_b_groups = 0;
int table_c_dimension = 0;
std::size_t active_lanes = 0;
};
Status inspect_rp(const std::vector<std::int32_t>& lanes, RpHeader* out) {
if (lanes.size() < 5u) {
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella rp must contain whole int32 lanes");
}
if (low16(lanes[0]) != 0x7072u) {
return model_fail(JOC_ERR_HRTF_FORMAT, "bad Rosella rp magic");
}
out->stored_checksum = low16(lanes[1]);
out->table_a_present = low16(lanes[2]) != 0u;
out->table_b_present = low16(lanes[3]) != 0u;
out->table_c_present = low16(lanes[4]) != 0u;
std::size_t index = 5u;
if (out->table_a_present) {
out->table_a_dimension = low16(lanes[index]);
out->table_a_option = low16(lanes[index + 1u]);
out->table_a_extra = low16(lanes[index + 2u]);
index += 5u;
} else {
out->table_a_dimension = 77;
}
if (out->table_b_present) {
if (!out->table_a_present) {
return model_fail(JOC_ERR_HRTF_FORMAT,
"Rosella rp table B cannot be present without table A");
}
out->table_b_dimension = low16(lanes[index]);
out->table_b_extra = low16(lanes[index + 1u]);
out->table_b_groups = low16(lanes[index + 2u]);
index += 3u;
}
if (out->table_c_present) {
out->table_c_dimension = low16(lanes[index]);
index += 1u;
}
const long long payload_words =
static_cast<long long>(index) - 2 +
(out->table_b_present ? (out->table_b_dimension + 380 * out->table_b_groups +
out->table_b_extra + 79)
: 0) +
(out->table_a_present ? (171 * out->table_a_extra + 79 +
2 * (out->table_a_option + 14 * out->table_a_dimension))
: 0) +
11 + (out->table_c_present ? (314 * out->table_c_dimension + 1) : 0);
if (payload_words < 0) {
return model_fail(JOC_ERR_HRTF_FORMAT, "malformed Rosella rp header");
}
out->active_lanes = static_cast<std::size_t>(2 + payload_words);
if (lanes.size() < out->active_lanes) {
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella rp is truncated");
}
std::uint32_t computed = 0xA569u;
for (std::size_t lane = 2u; lane < out->active_lanes; ++lane) {
computed ^= low16(lanes[lane]);
}
out->computed_checksum = static_cast<std::uint16_t>(computed & 0xFFFFu);
out->checksum_valid = out->computed_checksum == out->stored_checksum;
return Status::success();
}
// _unpack_field: the serialized 154-per-direction field lanes to the padded grid.
void unpack_field(const std::int32_t* serialized, int directions, int exponent,
std::vector<float>* padded) {
padded->assign(static_cast<std::size_t>(160 * directions), 0.0f);
const int stride8 = 8 * directions;
const int stride2 = 2 * directions;
for (int source = 0; source < 154 * directions; ++source) {
const int group4 = (source % stride8) / stride2;
const int destination = (group4 & 3) + 4 * (source % stride2 +
2 * directions * (source / stride8 +
(group4 >> 2)));
(*padded)[static_cast<std::size_t>(destination)] =
q15_exp(serialized[source], exponent);
}
}
// _unpack_table_a_grid: the serialized table-A rows to the padded lane grid.
void unpack_table_a_grid(const std::int32_t* serialized, int dimension, int serialized_rows,
int padded_rows, int lane_group, std::vector<float>* padded) {
padded->assign(static_cast<std::size_t>(padded_rows) * static_cast<std::size_t>(dimension),
0.0f);
const int group_width = lane_group * 4;
for (int source = 0; source < serialized_rows * dimension; ++source) {
const int remainder = source % group_width;
const int destination = (remainder / lane_group) +
4 * (remainder % lane_group +
group_width / 4 * (source / group_width));
(*padded)[static_cast<std::size_t>(destination)] = q15(serialized[source]);
}
}
void unpack_table_a_extra(const std::int32_t* serialized, std::vector<float>* padded) {
padded->assign(160u, 0.0f);
for (int source = 0; source < 154; ++source) {
const int remainder = source & 7;
const int destination = (remainder >> 1) + 4 * ((source & 1) + 2 * (source >> 3));
(*padded)[static_cast<std::size_t>(destination)] = q15(serialized[source]);
}
}
} // namespace
std::string RosellaModel::summary() const {
std::string name = capture.name.empty() ? std::string("unnamed") : capture.name;
return "Rosella personalized_headphone '" + name + "' (" +
(room_model.empty() ? std::string("unknown room") : room_model) + "), " +
std::to_string(table_a_dimension) + " HQMF / 77 hybrid @ " +
std::to_string(sample_rate) + " Hz";
}
Status load_personalized_headphone(const std::string& path, RosellaModel* out) {
if (out == nullptr) {
return model_fail(JOC_ERR_INVALID_ARGUMENT, "null Rosella model destination");
}
if (!fs_utf8::exists(path)) {
return model_fail(JOC_ERR_HRTF_NOT_FOUND, "personalized headphone model not found: " + path);
}
std::ifstream stream = fs_utf8::open_input(path);
if (!stream.good()) {
return model_fail(JOC_ERR_IO, "cannot open " + path);
}
std::string text((std::istreambuf_iterator<char>(stream)), std::istreambuf_iterator<char>());
if (text.empty()) {
return model_fail(JOC_ERR_HRTF_FORMAT, "empty personalized headphone model: " + path);
}
// The checksum is taken over the coefficient lanes, exactly as upstream hashes
// the int32 image of the array.
const std::size_t first = text.find_first_not_of(" \t\r\n");
if (first == std::string::npos || text[first] != '{') {
return model_fail(JOC_ERR_NOT_SUPPORTED,
"raw rp models are not supported; use a .personalized_headphone JSON");
}
std::vector<json::Member> root;
std::string error;
if (!json::parse_object(text, &root, &error)) {
return model_fail(JOC_ERR_HRTF_FORMAT, "invalid personalized headphone JSON: " + error);
}
const json::Member* personalized = json::find(root, "personalized_hrtf");
if (personalized == nullptr) {
return model_fail(JOC_ERR_HRTF_FORMAT, "personalized_hrtf is missing");
}
std::vector<json::Member> inner;
if (!json::parse_object(personalized->raw, &inner, &error)) {
return model_fail(JOC_ERR_HRTF_FORMAT, "invalid personalized_hrtf object: " + error);
}
const json::Member* virtualizer = json::find(inner, "virtualizer_parameters");
if (virtualizer == nullptr) {
return model_fail(JOC_ERR_HRTF_FORMAT, "virtualizer_parameters is missing");
}
std::vector<json::Member> parameters;
if (!json::parse_object(virtualizer->raw, &parameters, &error)) {
return model_fail(JOC_ERR_HRTF_FORMAT, "invalid virtualizer_parameters: " + error);
}
const json::Member* coefficient_member = json::find(parameters, "rosella_coefficients");
if (coefficient_member == nullptr) {
return model_fail(JOC_ERR_HRTF_FORMAT, "rosella_coefficients is missing");
}
RosellaModel model;
model.source_path = path;
if (const json::Member* member = json::find(parameters, "rosella_coefficients_version")) {
json::as_string(*member, &model.coefficient_version);
}
if (const json::Member* member = json::find(parameters, "room_model")) {
json::as_string(*member, &model.room_model);
}
if (const json::Member* capture = json::find(inner, "phrtf_capture_metadata")) {
std::vector<json::Member> fields;
if (json::parse_object(capture->raw, &fields, &error)) {
const std::pair<const char*, std::string*> mapping[] = {
{"capture_submission_date", &model.capture.capture_submission_date},
{"capture_type", &model.capture.capture_type},
{"label", &model.capture.label},
{"name", &model.capture.name},
{"phrtf_algorithm_version", &model.capture.algorithm_version},
{"phrtf_creation_date", &model.capture.creation_date},
{"uuid", &model.capture.uuid},
{"version", &model.capture.version},
};
for (const auto& entry : mapping) {
if (const json::Member* member = json::find(fields, entry.first)) {
json::as_string(*member, entry.second);
}
}
}
}
std::vector<std::int32_t> lanes;
if (!parse_int_array(coefficient_member->raw, &lanes, &error)) {
return model_fail(JOC_ERR_HRTF_FORMAT, error);
}
{
crypto::Sha256 hash;
hash.update(lanes.data(), lanes.size() * sizeof(std::int32_t));
model.coefficient_sha256 = hash.finish_hex();
}
RpHeader header;
Status status = inspect_rp(lanes, &header);
if (!status.ok()) {
return status;
}
if (!header.checksum_valid || header.active_lanes != lanes.size()) {
return model_fail(JOC_ERR_HRTF_FORMAT,
"invalid or non-active Rosella rp coefficient sequence");
}
if (!header.table_a_present || !header.table_b_present || header.table_c_present) {
return model_fail(JOC_ERR_NOT_SUPPORTED,
"the renderer requires table A+B and no table C");
}
if (header.table_a_dimension != 64 || header.table_a_option != 3) {
return model_fail(JOC_ERR_NOT_SUPPORTED,
"the renderer requires the observed 64-channel HQMF layout");
}
if (header.table_b_dimension != 20 || header.table_b_groups != 36) {
return model_fail(JOC_ERR_NOT_SUPPORTED,
"the renderer requires 20 hybrid groups and 36 direction terms");
}
const std::int32_t* values = lanes.data();
const std::size_t total = lanes.size();
std::size_t position = 13u;
const int extra = header.table_a_extra;
model.table_a_dimension = header.table_a_dimension;
model.table_a_option = header.table_a_option;
model.table_a_extra = extra;
model.table_a_header_field = low16(values[8]);
model.table_a_header_25 = low16(values[9]);
model.table_a_control = low16(values[position]);
model.field_exponent = values[position];
position += 1u;
const int option_count = header.table_a_option;
model.table_a_option_ids.resize(static_cast<std::size_t>(option_count));
for (int index = 0; index < option_count; ++index) {
model.table_a_option_ids[static_cast<std::size_t>(index)] =
low16(values[position + static_cast<std::size_t>(index)]);
}
position += static_cast<std::size_t>(option_count);
model.table_a_option_values.resize(static_cast<std::size_t>(option_count));
for (int index = 0; index < option_count; ++index) {
model.table_a_option_values[static_cast<std::size_t>(index)] =
q15(values[position + static_cast<std::size_t>(index)]);
}
position += static_cast<std::size_t>(option_count);
model.table_a_scalar = q15(values[position]);
position += 1u;
const int dimension = header.table_a_dimension;
unpack_table_a_grid(values + position, dimension, 16, 20, 16,
&model.table_a_filter_16x64_padded);
position += static_cast<std::size_t>(16 * dimension);
for (int index = 0; index < 4; ++index) {
model.table_a_four_integers[static_cast<std::size_t>(index)] =
low16(values[position + static_cast<std::size_t>(index)]);
}
position += 4u;
model.table_a_integer = low16(values[position]);
position += 1u;
unpack_table_a_grid(values + position, dimension, 8, 10, 8,
&model.table_a_filter_8x64_padded);
position += static_cast<std::size_t>(8 * dimension);
model.table_a_vector16.resize(16u);
for (int index = 0; index < 16; ++index) {
model.table_a_vector16[static_cast<std::size_t>(index)] =
q15(values[position + static_cast<std::size_t>(index)]);
}
position += 16u;
unpack_table_a_grid(values + position, dimension, 4, 5, 4,
&model.table_a_filter_4x64_padded);
position += static_cast<std::size_t>(4 * dimension);
model.table_a_extra_indices.resize(static_cast<std::size_t>(extra));
for (int index = 0; index < extra; ++index) {
model.table_a_extra_indices[static_cast<std::size_t>(index)] =
low16(values[position + static_cast<std::size_t>(index)]);
}
position += static_cast<std::size_t>(extra);
model.table_a_extra_fields_padded.assign(static_cast<std::size_t>(extra) * 160u, 0.0f);
std::vector<float> unpacked;
for (int index = 0; index < extra; ++index) {
unpack_table_a_extra(values + position, &unpacked);
std::copy(unpacked.begin(), unpacked.end(),
model.table_a_extra_fields_padded.begin() + static_cast<std::ptrdiff_t>(index) * 160);
position += 154u;
}
model.table_a_extra_vectors.assign(static_cast<std::size_t>(extra) * 16u, 0.0f);
for (int index = 0; index < extra; ++index) {
for (int lane = 0; lane < 16; ++lane) {
model.table_a_extra_vectors[static_cast<std::size_t>(index) * 16u +
static_cast<std::size_t>(lane)] =
q15(values[position + static_cast<std::size_t>(lane)]);
}
position += 16u;
}
const std::size_t table_b_start = position;
if (table_b_start != 13u + 1821u + static_cast<std::size_t>(171 * extra)) {
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella table-A parser lost its place");
}
model.sample_rate = 2 * low16(values[position]);
position += 1u;
model.matrix_exponent = values[position];
position += 1u;
const std::size_t matrix_count = 36u * 36u;
model.matrix_left.resize(matrix_count);
model.matrix_right.resize(matrix_count);
for (std::size_t index = 0; index < matrix_count; ++index) {
model.matrix_left[index] = q15_exp(values[position + index], model.matrix_exponent);
}
position += matrix_count;
for (std::size_t index = 0; index < matrix_count; ++index) {
model.matrix_right[index] = q15_exp(values[position + index], model.matrix_exponent);
}
position += matrix_count;
model.vector_left.resize(36u);
model.vector_right.resize(36u);
for (int index = 0; index < 36; ++index) {
model.vector_left[static_cast<std::size_t>(index)] =
q15_exp(values[position + static_cast<std::size_t>(index)], model.matrix_exponent);
}
position += 36u;
for (int index = 0; index < 36; ++index) {
model.vector_right[static_cast<std::size_t>(index)] =
q15_exp(values[position + static_cast<std::size_t>(index)], model.matrix_exponent);
}
position += 36u;
const std::size_t serialized_count = 154u * 36u;
unpack_field(values + position, 36, model.field_exponent, &model.field_left_padded);
bool odd_zero = true;
for (std::size_t index = 1u; index < serialized_count; index += 2u) {
const float value = q15_exp(values[position + index], model.field_exponent);
if (std::abs(value) > 1.0e-6f) {
odd_zero = false;
break;
}
}
model.field_left_odd_serialized_zero = odd_zero;
position += serialized_count;
unpack_field(values + position, 36, model.field_exponent, &model.field_right_padded);
position += serialized_count;
model.hybrid_flags.resize(20u);
int active_hybrid = 0;
for (int index = 0; index < 20; ++index) {
model.hybrid_flags[static_cast<std::size_t>(index)] =
low16(values[position + static_cast<std::size_t>(index)]);
if (model.hybrid_flags[static_cast<std::size_t>(index)] == 1) {
++active_hybrid;
}
}
position += 20u;
if (active_hybrid != header.table_b_extra) {
return model_fail(JOC_ERR_HRTF_FORMAT, "hybrid value count does not match the header");
}
model.hybrid_values.resize(static_cast<std::size_t>(active_hybrid));
for (int index = 0; index < active_hybrid; ++index) {
model.hybrid_values[static_cast<std::size_t>(index)] =
q15(values[position + static_cast<std::size_t>(index)]);
}
position += static_cast<std::size_t>(active_hybrid);
model.model_scalars.resize(5u);
for (int index = 0; index < 5; ++index) {
model.model_scalars[static_cast<std::size_t>(index)] =
q15(values[position + static_cast<std::size_t>(index)]);
}
position += 5u;
const std::size_t expected_tail =
table_b_start + static_cast<std::size_t>(header.table_b_dimension +
380 * header.table_b_groups +
header.table_b_extra + 79);
if (position != expected_tail) {
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella table-B parser lost its place");
}
model.header_float_scalars[0] = q15(values[position]);
model.header_float_scalars[1] = q15(values[position + 1u]) * 16.0f;
model.header_integer_fields[0] = values[position + 2u];
model.header_integer_fields[1] = low16(values[position + 3u]);
position += 4u;
for (int profile = 0; profile < 4; ++profile) {
RosellaDistanceProfile parsed;
for (int index = 0; index < 6; ++index) {
parsed.bounds[static_cast<std::size_t>(index)] =
q15(values[position + static_cast<std::size_t>(index)]);
}
position += 6u;
parsed.distance_scale_m =
q15_exp(values[position], values[position + 1u]);
position += 2u;
parsed.inverse_distance_per_m = q15(values[position]);
parsed.axis_scales_internal[0] = q15(values[position + 1u]);
parsed.axis_scales_internal[1] = q15(values[position + 2u]);
parsed.axis_scales_internal[2] = q15(values[position + 3u]);
parsed.minimum_normalized_radius = q15(values[position + 4u]);
position += 5u;
model.profiles[static_cast<std::size_t>(profile)] = parsed;
}
model.profile_tail.resize(8u);
for (int index = 0; index < 8; ++index) {
model.profile_tail[static_cast<std::size_t>(index)] =
q15(values[position + static_cast<std::size_t>(index)]);
}
position += 8u;
for (int index = 0; index < 3; ++index) {
model.post_fields[static_cast<std::size_t>(index)] =
values[position + static_cast<std::size_t>(index)];
}
position += 3u;
if (position != total) {
return model_fail(JOC_ERR_HRTF_FORMAT, "unparsed Rosella coefficient lanes");
}
if (model.sample_rate != 48000) {
return model_fail(JOC_ERR_HRTF_FORMAT,
"Rosella model sample rate must be 48000, got " +
std::to_string(model.sample_rate));
}
*out = std::move(model);
return Status::success();
}
} // namespace joc::hrtf
-88
View File
@@ -1,88 +0,0 @@
#pragma once
#include <array>
#include <cstdint>
#include <string>
#include <vector>
#include "foundation/status.h"
// Parser for the Dolby ".personalized_headphone" model (upstream rosella_model.py).
// The file is JSON whose virtualizer_parameters carry the raw "rp" coefficient
// lanes; everything the renderer needs is unpacked here, in the same float32
// arithmetic the reference uses, because those values are part of the model.
namespace joc::hrtf {
struct RosellaDistanceProfile {
std::array<float, 6> bounds{};
float distance_scale_m = 0.0f;
float inverse_distance_per_m = 0.0f;
std::array<float, 3> axis_scales_internal{};
float minimum_normalized_radius = 0.0f;
};
struct RosellaCaptureMetadata {
std::string capture_submission_date;
std::string capture_type;
std::string label;
std::string name;
std::string algorithm_version;
std::string creation_date;
std::string uuid;
std::string version;
};
struct RosellaModel {
std::string source_path;
std::string coefficient_sha256;
std::string coefficient_version;
std::string room_model;
RosellaCaptureMetadata capture;
int table_a_dimension = 0;
int table_a_option = 0;
int table_a_extra = 0;
int table_a_header_field = 0;
int table_a_header_25 = 0;
int table_a_control = 0;
std::vector<int> table_a_option_ids;
std::vector<float> table_a_option_values;
float table_a_scalar = 0.0f;
std::vector<float> table_a_filter_16x64_padded;
std::array<int, 4> table_a_four_integers{};
int table_a_integer = 0;
std::vector<float> table_a_filter_8x64_padded;
std::vector<float> table_a_vector16;
std::vector<float> table_a_filter_4x64_padded;
std::vector<int> table_a_extra_indices;
std::vector<float> table_a_extra_fields_padded;
std::vector<float> table_a_extra_vectors;
int sample_rate = 0;
int matrix_exponent = 0;
int field_exponent = 0;
std::vector<float> matrix_left;
std::vector<float> matrix_right;
std::vector<float> vector_left;
std::vector<float> vector_right;
std::vector<float> field_left_padded;
std::vector<float> field_right_padded;
bool field_left_odd_serialized_zero = false;
std::vector<int> hybrid_flags;
std::vector<float> hybrid_values;
std::vector<float> model_scalars;
std::array<float, 2> header_float_scalars{};
std::array<int, 2> header_integer_fields{};
std::array<RosellaDistanceProfile, 4> profiles{};
std::vector<float> profile_tail;
std::array<int, 3> post_fields{};
// One line for reports and logs: the capture name and room model are the
// model's own strings, followed by the table layout and sample rate, e.g.
// "Rosella personalized_headphone '<name>' (<room>), <N> HQMF / 77 hybrid @ <rate> Hz".
std::string summary() const;
};
Status load_personalized_headphone(const std::string& path, RosellaModel* out);
} // namespace joc::hrtf
File diff suppressed because it is too large Load Diff
-64
View File
@@ -1,64 +0,0 @@
#pragma once
#include <cstdint>
#include <memory>
#include <string>
#include <vector>
#include "foundation/status.h"
#include "hrtf/rosella_model.h"
#include "oamd/oamd_parser.h"
#include "timeline/position_timeline.h"
// Rosella ".personalized_headphone" binaural renderer (upstream rosella_core.py,
// rosella_direct.py, rosella_room.py and rosella_binaural_renderer.py). It takes
// the same pipeline slot as the SOFA runtime: sixteen object channels per frame in,
// interleaved stereo out, with the OAMD timeline driving the per-block parameters.
namespace joc::hrtf {
// rosella_direct.BINAURAL_PROFILE_NAMES.
enum class RosellaProfile : std::int32_t { Near = 1, Far = 2, Mid = 3 };
struct RosellaRenderOptions {
RosellaProfile profile = RosellaProfile::Mid;
std::int64_t object_delay_samples = 1473;
double tail_seconds = 5.0;
double output_gain = 1.0;
int chunk_frames = 64;
int room_impulse_slots = 4096;
};
class RosellaRuntime {
public:
RosellaRuntime();
~RosellaRuntime();
RosellaRuntime(const RosellaRuntime&) = delete;
RosellaRuntime& operator=(const RosellaRuntime&) = delete;
Status open(const RosellaModel& model, const RosellaRenderOptions& options);
// objects16_planar is channel-major: channel * 1536 + sample.
Status submit_frame(const float* objects16_planar, const oamd::OamdUpdate* update,
std::int64_t frame_index, std::int64_t outer_sample_offset,
std::int64_t object_delay_samples);
// Drains the flush tail: the pending partial chunk plus flush_samples of silence.
Status finish(std::uint32_t flush_samples, std::vector<double>* out);
std::uint32_t finish_capacity(double tail_seconds) const;
Status reset();
const std::vector<double>& output() const;
void take_output(std::vector<double>* out);
std::uint64_t input_samples() const;
std::uint64_t processed_input_samples() const;
std::uint64_t metadata_block_updates() const;
const timeline::OamdPositionTimeline& timeline() const;
private:
struct Impl;
std::unique_ptr<Impl> impl_;
};
} // namespace joc::hrtf
-255
View File
@@ -1,255 +0,0 @@
#include "hrtf/sofa.h"
#include <algorithm>
#include <cmath>
#include <cctype>
#include <cstdio>
#include <fstream>
#include <utility>
#include "foundation/fs_utf8.h"
#include "foundation/sha256.h"
#include "io/hdf5.h"
namespace joc::hrtf {
namespace {
Status sofa_fail(joc_error code, const std::string& message) {
return Status::fail(code, stage::kRender, message);
}
std::string format_number(double value) {
if (std::isfinite(value) && value == std::floor(value) && std::fabs(value) < 1.0e15) {
return std::to_string(static_cast<long long>(value));
}
char buffer[32];
std::snprintf(buffer, sizeof(buffer), "%.6g", value);
return std::string(buffer);
}
// Every array is checked against the element count the convention prescribes, so
// a file whose shape disagrees with its metadata is rejected instead of silently
// producing a shifted impulse response.
Status read_doubles(const io::Hdf5File& file, const std::string& path, std::uint64_t expected,
std::vector<double>* out) {
if (!file.has_dataset(path)) {
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA file has no " + path + " dataset");
}
const Status status = file.read_dataset_double(path, out);
if (!status.ok()) {
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA dataset " + path + ": " + status.message());
}
if (out->size() != expected) {
return sofa_fail(JOC_ERR_HRTF_FORMAT,
"SOFA dataset " + path + " holds " + std::to_string(out->size()) +
" values, expected " + std::to_string(expected));
}
return Status::success();
}
Status read_text(const io::Hdf5File& file, const std::string& name, bool required,
std::string* out) {
io::Hdf5Attribute attribute;
const Status status = file.attribute("", name, &attribute);
if (!status.ok()) {
if (required) {
return sofa_fail(JOC_ERR_HRTF_FORMAT,
"SOFA file has no root attribute " + name + ": " + status.message());
}
return Status::success();
}
*out = attribute.text;
return Status::success();
}
// SHA-256 of the whole file: the compiled-cache key is derived from it, so the
// digest is taken over the exact bytes the parse consumed.
std::string file_digest(const std::string& path) {
std::ifstream stream = fs_utf8::open_input(path);
if (!stream.good()) {
return std::string();
}
crypto::Sha256 hash;
std::vector<char> buffer(1u << 20);
while (stream.good()) {
stream.read(buffer.data(), static_cast<std::streamsize>(buffer.size()));
const std::streamsize count = stream.gcount();
if (count > 0) {
hash.update(buffer.data(), static_cast<std::size_t>(count));
}
}
return hash.finish_hex();
}
// The coordinate declaration of one dataset, when the file carries it.
void read_coordinates(const io::Hdf5File& file, const std::string& dataset,
SofaCoordinate* out) {
io::Hdf5Attribute attribute;
if (file.attribute(dataset, "Type", &attribute).ok()) {
out->type = attribute.text;
}
if (file.attribute(dataset, "Units", &attribute).ok()) {
out->units = attribute.text;
}
}
} // namespace
std::string SofaHrir::summary() const {
return "SOFA " + sofa_conventions + ", " + std::to_string(ir_count) + " IRs x " +
std::to_string(ir_length) + " taps @ " + format_number(sample_rate) + " Hz";
}
Status load_sofa(const std::string& path, SofaHrir* out) {
if (out == nullptr) {
return sofa_fail(JOC_ERR_INVALID_ARGUMENT, "null SOFA destination");
}
if (!fs_utf8::exists(path)) {
return sofa_fail(JOC_ERR_HRTF_NOT_FOUND, "SOFA file not found: " + path);
}
io::Hdf5File file;
Status status = file.open(path);
if (!status.ok()) {
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA file " + path + ": " + status.message());
}
SofaHrir sofa;
sofa.source_path = path;
sofa.source_sha256 = file_digest(path);
for (char& character : sofa.source_sha256) {
character = static_cast<char>(std::toupper(static_cast<unsigned char>(character)));
}
status = read_text(file, "Conventions", true, &sofa.conventions);
if (!status.ok()) {
return status;
}
status = read_text(file, "SOFAConventions", true, &sofa.sofa_conventions);
if (!status.ok()) {
return status;
}
if (sofa.conventions != "SOFA" || sofa.sofa_conventions != "SimpleFreeFieldHRIR") {
return sofa_fail(JOC_ERR_HRTF_UNSUPPORTED_CONVENTION,
"SOFA conventions " + sofa.conventions + "/" + sofa.sofa_conventions +
" are not SimpleFreeFieldHRIR");
}
status = read_text(file, "SOFAConventionsVersion", false, &sofa.convention_version);
if (!status.ok()) {
return status;
}
status = read_text(file, "Version", false, &sofa.version);
if (!status.ok()) {
return status;
}
status = read_text(file, "DataType", false, &sofa.data_type);
if (!status.ok()) {
return status;
}
status = read_text(file, "RoomType", false, &sofa.room_type);
if (!status.ok()) {
return status;
}
status = read_text(file, "Title", false, &sofa.title);
if (!status.ok()) {
return status;
}
status = read_text(file, "DatabaseName", false, &sofa.database_name);
if (!status.ok()) {
return status;
}
status = read_text(file, "ListenerShortName", false, &sofa.listener_short_name);
if (!status.ok()) {
return status;
}
status = read_text(file, "Comment", false, &sofa.comment);
if (!status.ok()) {
return status;
}
io::Hdf5DatasetInfo info;
status = file.dataset_info("Data.IR", &info);
if (!status.ok()) {
return sofa_fail(JOC_ERR_HRTF_FORMAT,
"SOFA file has no usable Data.IR dataset: " + status.message());
}
if (info.shape.size() != 3u || info.shape[1] != 2u || info.shape[0] == 0u ||
info.shape[2] == 0u) {
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA Data.IR is not shaped (M, 2, N)");
}
if (info.shape[0] > 0xFFFFFFFFull || info.shape[2] > 0xFFFFFFFFull) {
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA Data.IR is larger than this reader accepts");
}
sofa.ir_count = static_cast<std::uint32_t>(info.shape[0]);
sofa.ir_length = static_cast<std::uint32_t>(info.shape[2]);
const std::uint64_t taps = static_cast<std::uint64_t>(sofa.ir_count) * 2u * sofa.ir_length;
status = read_doubles(file, "Data.IR", taps, &sofa.ir);
if (!status.ok()) {
return status;
}
std::vector<double> scalar;
status = read_doubles(file, "Data.SamplingRate", 1u, &scalar);
if (!status.ok()) {
return status;
}
sofa.sample_rate = scalar[0];
io::Hdf5Attribute attribute;
if (file.attribute("Data.SamplingRate", "Units", &attribute).ok()) {
sofa.sampling_rate_units = attribute.text;
}
// Data.Delay is optional in the wild; absent means "no delay was measured".
if (file.has_dataset("Data.Delay")) {
std::vector<double> delay;
status = read_doubles(file, "Data.Delay", 2u, &delay);
if (!status.ok()) {
return status;
}
sofa.delay[0] = delay[0];
sofa.delay[1] = delay[1];
}
const std::uint64_t measurements = sofa.ir_count;
status = read_doubles(file, "SourcePosition", measurements * 3u, &sofa.source_position);
if (!status.ok()) {
return status;
}
read_coordinates(file, "SourcePosition", &sofa.source_position_coordinates);
std::vector<double> vector;
status = read_doubles(file, "ListenerPosition", 3u, &vector);
if (!status.ok()) {
return status;
}
std::copy(vector.begin(), vector.end(), sofa.listener_position);
read_coordinates(file, "ListenerPosition", &sofa.listener_position_coordinates);
status = read_doubles(file, "ListenerView", 3u, &vector);
if (!status.ok()) {
return status;
}
std::copy(vector.begin(), vector.end(), sofa.listener_view);
read_coordinates(file, "ListenerView", &sofa.listener_view_coordinates);
status = read_doubles(file, "ListenerUp", 3u, &vector);
if (!status.ok()) {
return status;
}
std::copy(vector.begin(), vector.end(), sofa.listener_up);
read_coordinates(file, "ListenerUp", &sofa.listener_up_coordinates);
status = read_doubles(file, "EmitterPosition", 3u, &vector);
if (!status.ok()) {
return status;
}
std::copy(vector.begin(), vector.end(), sofa.emitter_position);
read_coordinates(file, "EmitterPosition", &sofa.emitter_position_coordinates);
status = read_doubles(file, "ReceiverPosition", 6u, &vector);
if (!status.ok()) {
return status;
}
std::copy(vector.begin(), vector.end(), sofa.receiver_position);
read_coordinates(file, "ReceiverPosition", &sofa.receiver_position_coordinates);
*out = std::move(sofa);
return Status::success();
}
} // namespace joc::hrtf
-61
View File
@@ -1,61 +0,0 @@
#pragma once
#include <cstdint>
#include <string>
#include <vector>
#include "foundation/status.h"
namespace joc::hrtf {
// Coordinate declaration of one SOFA variable: Type ("spherical"/"cartesian") and
// Units. Empty when the file does not declare them (ListenerUp inherits).
struct SofaCoordinate {
std::string type;
std::string units;
};
// SOFA SimpleFreeFieldHRIR as this project consumes it: the impulse responses,
// the measurement geometry and the metadata needed to report what was loaded.
// All angles are degrees, all distances metres, exactly as the file stores them.
struct SofaHrir {
double sample_rate = 0.0;
std::uint32_t ir_count = 0; // M: number of measurements
std::uint32_t ir_length = 0; // N: taps per impulse response
std::vector<double> ir; // C order [M][2][N]
double delay[2] = {0.0, 0.0};
std::vector<double> source_position; // M*3
double listener_position[3] = {0.0, 0.0, 0.0};
double listener_view[3] = {1.0, 0.0, 0.0};
double listener_up[3] = {0.0, 0.0, 1.0};
double emitter_position[3] = {0.0, 0.0, 0.0};
double receiver_position[6] = {0.0, 0.0, 0.0, 0.0, 0.0, 0.0};
std::string conventions;
std::string sofa_conventions;
std::string convention_version;
std::string version;
std::string data_type;
std::string room_type;
std::string title;
std::string database_name;
std::string listener_short_name;
std::string comment;
std::string sampling_rate_units;
SofaCoordinate source_position_coordinates;
SofaCoordinate listener_position_coordinates;
SofaCoordinate listener_view_coordinates;
SofaCoordinate listener_up_coordinates;
SofaCoordinate emitter_position_coordinates;
SofaCoordinate receiver_position_coordinates;
// Identity of the file itself, needed for the compiled-cache key.
std::string source_path;
std::string source_sha256;
// One line for reports and logs:
// "SOFA SimpleFreeFieldHRIR, <M> IRs x <N> taps @ <rate> Hz".
std::string summary() const;
};
Status load_sofa(const std::string& path, SofaHrir* out);
} // namespace joc::hrtf
-237
View File
@@ -1,237 +0,0 @@
#include "hrtf/sofa_cache.h"
#include <algorithm>
#include <filesystem>
#include <list>
#include <mutex>
#include <utility>
#include <vector>
#include "foundation/fs_utf8.h"
#include "hrtf/sofa.h"
namespace joc::hrtf {
namespace {
namespace fs = std::filesystem;
// Small process-local cache: the reference keeps the last eight compiled fields.
constexpr std::size_t kMemoryCacheEntries = 8;
struct MemoryEntry {
std::string key;
Field field;
};
std::mutex& memory_mutex() {
static std::mutex mutex;
return mutex;
}
std::list<MemoryEntry>& memory_cache() {
static std::list<MemoryEntry> cache;
return cache;
}
bool memory_cache_get(const std::string& key, Field* out) {
std::lock_guard<std::mutex> lock(memory_mutex());
std::list<MemoryEntry>& cache = memory_cache();
for (auto entry = cache.begin(); entry != cache.end(); ++entry) {
if (entry->key == key) {
*out = entry->field;
cache.splice(cache.begin(), cache, entry);
return true;
}
}
return false;
}
void memory_cache_put(const std::string& key, const Field& field) {
std::lock_guard<std::mutex> lock(memory_mutex());
std::list<MemoryEntry>& cache = memory_cache();
for (auto entry = cache.begin(); entry != cache.end(); ++entry) {
if (entry->key == key) {
entry->field = field;
cache.splice(cache.begin(), cache, entry);
return;
}
}
cache.push_front(MemoryEntry{key, field});
while (cache.size() > kMemoryCacheEntries) {
cache.pop_back();
}
}
Status cache_fail(joc_error code, const std::string& message) {
return Status::fail(code, stage::kRender, message);
}
std::string upper(std::string text) {
std::transform(text.begin(), text.end(), text.begin(), [](unsigned char value) {
return static_cast<char>(std::toupper(value));
});
return text;
}
} // namespace
Status parse_cache_policy(const std::string& text, CachePolicy* out) {
if (out == nullptr) {
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "null cache policy");
}
if (text == "none") {
*out = CachePolicy::None;
return Status::success();
}
if (text == "memory") {
*out = CachePolicy::Memory;
return Status::success();
}
if (text == "disk") {
*out = CachePolicy::Disk;
return Status::success();
}
return cache_fail(JOC_ERR_INVALID_CONFIG, "cache_policy must be none, memory, or disk");
}
Status validate_jochrtf(const std::string& path, const std::string& source_sha256,
const std::string& cache_key, Field* out) {
Field field;
const Status status = load_jochrtf(path, &field);
if (!status.ok()) {
return status;
}
if (!source_sha256.empty() && field.source_sha256 != upper(source_sha256)) {
return cache_fail(JOC_ERR_HRTF_HASH, "compiled HRTF source hash mismatch");
}
if (!cache_key.empty() && field.cache_key != upper(cache_key)) {
return cache_fail(JOC_ERR_HRTF_HASH, "compiled HRTF configuration hash mismatch");
}
if (out != nullptr) {
*out = std::move(field);
}
return Status::success();
}
Status save_jochrtf_atomic(const Field& field, const std::string& path) {
if (path.empty()) {
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "empty compiled HRTF cache path");
}
const fs::path target = fs_utf8::to_path(path);
std::error_code error;
if (target.has_parent_path()) {
fs::create_directories(target.parent_path(), error);
if (error) {
return cache_fail(JOC_ERR_OUTPUT_OPEN,
"cannot create " + fs_utf8::from_path(target.parent_path()));
}
}
const std::string temporary = path + ".tmp";
Status status = write_jochrtf(field, temporary);
if (!status.ok()) {
return status;
}
// The rename is what makes a half-written cache impossible to observe.
fs::rename(fs_utf8::to_path(temporary), target, error);
if (error) {
fs::remove(fs_utf8::to_path(temporary), error);
return cache_fail(JOC_ERR_OUTPUT_WRITE, "cannot replace " + path);
}
return Status::success();
}
Status load_or_compile_sofa_field(const SofaFieldRequest& request, Field* out,
std::string* cache_path) {
if (out == nullptr) {
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "null compiled HRTF destination");
}
if (request.sofa_path.empty()) {
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "no SOFA path for the compiled HRTF field");
}
if (request.policy == CachePolicy::Disk && request.cache_dir.empty()) {
return cache_fail(JOC_ERR_INVALID_CONFIG, "the disk cache policy needs a cache directory");
}
SofaHrir sofa;
Status status = load_sofa(request.sofa_path, &sofa);
if (!status.ok()) {
return status;
}
CanonicalHrtf canonical;
status = canonicalize_sofa(sofa, &canonical);
if (!status.ok()) {
return status;
}
// The key depends on the shell that the radius selects, exactly as upstream.
double actual_radius = request.options.shell_radius_m;
(void)canonical_shell_indices(canonical, request.options.shell_radius_m, &actual_radius);
const std::string key = compiled_hrtf_cache_key(
canonical.source_sha256, canonical.sample_rate_hz, actual_radius, request.options.order,
request.options.projection_ridge, request.options.sh_ridge);
if (cache_path != nullptr) {
cache_path->clear();
}
std::string target;
if (request.policy == CachePolicy::Disk) {
target = request.cache_dir;
if (!target.empty() && target.back() != '/' && target.back() != '\\') {
target += "/";
}
target += cache_file_name(canonical.source_path.empty()
? std::string()
: canonical.source_path,
key);
if (fs_utf8::exists(target)) {
Field cached_field;
const Status cached =
validate_jochrtf(target, canonical.source_sha256, key, &cached_field);
if (cached.ok()) {
memory_cache_put(key, cached_field);
*out = std::move(cached_field);
if (cache_path != nullptr) {
*cache_path = target;
}
return Status::success();
}
}
}
Field field;
if (request.policy != CachePolicy::None && memory_cache_get(key, &field)) {
// A memory hit still materialises the disk cache the caller asked for.
if (request.policy == CachePolicy::Disk) {
status = save_jochrtf_atomic(field, target);
if (!status.ok()) {
return status;
}
if (cache_path != nullptr) {
*cache_path = target;
}
}
*out = std::move(field);
return Status::success();
}
status = compile_canonical_field(canonical, request.options, &field);
if (!status.ok()) {
return status;
}
if (field.cache_key != key) {
return cache_fail(JOC_ERR_INTERNAL, "internal compiled HRTF cache-key mismatch");
}
if (request.policy == CachePolicy::Disk) {
status = save_jochrtf_atomic(field, target);
if (!status.ok()) {
return status;
}
if (cache_path != nullptr) {
*cache_path = target;
}
}
if (request.policy != CachePolicy::None) {
memory_cache_put(key, field);
}
*out = std::move(field);
return Status::success();
}
} // namespace joc::hrtf
-38
View File
@@ -1,38 +0,0 @@
#pragma once
#include <string>
#include "foundation/status.h"
#include "hrtf/jochrtf.h"
#include "hrtf/sofa_field.h"
// Compiled-field cache: the .jochrtf is an internal artifact, so the caller only
// names the SOFA file and the policy. "memory" keeps the compiled field in this
// process, "disk" additionally reuses (and writes) <cache_dir>/<name>.<key>.jochrtf.
namespace joc::hrtf {
enum class CachePolicy { None, Memory, Disk };
struct SofaFieldRequest {
std::string sofa_path;
CompileOptions options;
CachePolicy policy = CachePolicy::Memory;
std::string cache_dir; // required for the disk policy
};
// Parses "none"/"memory"/"disk"; anything else is rejected.
Status parse_cache_policy(const std::string& text, CachePolicy* out);
// Returns the compiled field, reusing a valid cache when the policy allows it.
// `cache_path` (optional) receives the cache file that was read or written.
Status load_or_compile_sofa_field(const SofaFieldRequest& request, Field* out,
std::string* cache_path);
// Writes the field to `path` through a temporary file and an atomic rename.
Status save_jochrtf_atomic(const Field& field, const std::string& path);
// Verifies that a cache file belongs to `source_sha256` and `cache_key`.
Status validate_jochrtf(const std::string& path, const std::string& source_sha256,
const std::string& cache_key, Field* out);
} // namespace joc::hrtf
File diff suppressed because it is too large Load Diff
-103
View File
@@ -1,103 +0,0 @@
#pragma once
#include <cstdint>
#include <string>
#include <vector>
#include "foundation/status.h"
#include "hrtf/jochrtf.h"
#include "hrtf/sofa.h"
// SOFA SimpleFreeFieldHRIR -> compiled directional field, ported from the
// reference chain (sofa_canonical.py + sofa_hrtf_field.py + the public
// filterbank): the measurement shell is selected, one delay representation is
// separated, the FIRs are projected onto the 64-QMF/77-hybrid filterbank and the
// result is fitted with fifth-order ACN/N3D real spherical harmonics.
namespace joc::hrtf {
inline constexpr int kFieldOrder = 5;
inline constexpr int kFieldTerms = 36;
inline constexpr double kFieldSampleRateHz = 48000.0;
inline constexpr double kDefaultShellRadiusM = 1.0;
inline constexpr double kDefaultProjectionRidge = 1.0e-3;
inline constexpr double kDefaultSphericalHarmonicRidge = 1.0e-5;
inline constexpr const char* kCompilerVersion = "joc-sofa-compiler-v1";
inline constexpr const char* kPhasePolicyVersion = "sofa-delay-exactly-once-v1";
inline constexpr const char* kShConvention = "ACN/N3D real";
inline constexpr const char* kFilterbankTableVersion = "joc-public-64qmf-77hybrid-v1";
// SHA-256 of the standard filterbank archive the embedded tables came from. It
// participates in the cache key, so it is part of the file-format contract.
inline constexpr const char* kFilterbankArchiveSha256 =
"C05BEF4D26E96ECBD4694E2572F05DA400255C777BA5047300B9D3B1F81081CD";
struct CompileOptions {
double shell_radius_m = kDefaultShellRadiusM;
int order = kFieldOrder;
double projection_ridge = kDefaultProjectionRidge;
double sh_ridge = kDefaultSphericalHarmonicRidge;
};
// Canonical HRIR set: Data.IR and Data.Delay stay separate, the listener frame is
// applied to the source positions and the ears are ordered left/right.
struct CanonicalHrtf {
std::string source_path;
std::string source_sha256;
std::string convention;
std::string convention_version;
std::string processing_label;
double sample_rate_hz = 0.0;
std::uint32_t measurements = 0;
std::uint32_t taps = 0;
int left_receiver_index = 0;
int right_receiver_index = 1;
std::vector<double> source_position_cartesian_m; // [M,3] listener-local
std::vector<double> unit_directions; // [M,3]
std::vector<double> measurement_radius_m; // [M]
std::vector<double> hrir; // [M,2,N] canonical L/R
std::vector<double> delay_samples; // [M,2], not applied
};
// Port of load_simple_free_field_hrir(): strict SimpleFreeFieldHRIR import.
Status canonicalize_sofa(const SofaHrir& sofa, CanonicalHrtf* out);
// Compiles the canonical set into the runtime field (port of SofaHrtfField.fit).
Status compile_sofa_field(const SofaHrir& sofa, const CompileOptions& options, Field* out);
// The measurements on the shell nearest to radius_m; actual_radius_m receives the
// mean radius of that shell (upstream CanonicalHrtf.shell_indices).
std::vector<std::size_t> canonical_shell_indices(const CanonicalHrtf& canonical, double radius_m,
double* actual_radius_m);
// Compiles an already canonicalized set (used by tests and the cache layer).
Status compile_canonical_field(const CanonicalHrtf& canonical, const CompileOptions& options,
Field* out);
// Configuration hash that names the cache file (upstream compiled_hrtf_cache_key).
std::string compiled_hrtf_cache_key(const std::string& source_sha256, double sample_rate_hz,
double shell_radius_m, int order, double projection_ridge,
double sh_ridge);
// Payload hash over the four arrays (upstream _payload_sha256).
std::string field_payload_sha256(const Field& field);
// "<stem>.<first 20 key digits>.jochrtf", the upstream cache file name.
std::string cache_file_name(const std::string& display_name, const std::string& cache_key);
// Serializes the field as a .jochrtf cache the upstream loader also accepts.
Status write_jochrtf(const Field& field, const std::string& path);
// The analysis/gain/synthesis dictionary the projection solves against (dev check).
std::vector<double> hybrid_gain_synthesis_dictionary_for_check(std::size_t sample_count);
// Shell directions and their spherical Voronoi weights (dev check).
void shell_directions_and_weights_for_check(const SofaHrir& sofa, double radius_m,
std::vector<double>* directions,
std::vector<double>* weights);
// PublicAnalysis77 on a unit impulse, interleaved complex (dev check).
std::vector<double> analysis_impulse_for_check(std::size_t total_samples);
// The 77 hybrid-band centre frequencies at 48 kHz.
const std::vector<double>& hybrid_band_center_frequencies_hz();
} // namespace joc::hrtf
-241
View File
@@ -1,241 +0,0 @@
#include "io/adm_writer.h"
#include "foundation/fs_utf8.h"
#include <cstring>
#include <filesystem>
#include "io/wav_writer.h" // pack_int24 (shared int24 quantisation)
namespace joc::io {
namespace {
constexpr long kDs64BodyOffset = 20;
constexpr long kDataSizeOffset = 76;
void put_u16(std::string* out, std::uint16_t value) {
char buffer[2];
std::memcpy(buffer, &value, 2);
out->append(buffer, 2);
}
void put_u32(std::string* out, std::uint32_t value) {
char buffer[4];
std::memcpy(buffer, &value, 4);
out->append(buffer, 4);
}
void put_u64(std::string* out, std::uint64_t value) {
char buffer[8];
std::memcpy(buffer, &value, 8);
out->append(buffer, 8);
}
} // namespace
AdmBwfWriter::~AdmBwfWriter() { abort(); }
Status AdmBwfWriter::open(const std::string& path, std::size_t block_samples) {
if (block_samples < JOC_FRAME_SAMPLES) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput,
"ADM block size must hold at least one E-AC-3 frame");
}
path_ = path;
block_samples_ = block_samples;
used_ = 0;
frames_ = 0;
finalized_ = false;
buffer_.assign(block_samples * kChannels, 0.0f);
file_ = fs_utf8::fopen(path, "wb+");
if (file_ == nullptr) {
std::error_code ignored;
const std::filesystem::path parent = std::filesystem::path(path).parent_path();
if (!parent.empty()) {
std::filesystem::create_directories(parent, ignored);
}
file_ = fs_utf8::fopen(path, "wb+");
}
if (file_ == nullptr) {
return Status::fail(JOC_ERR_OUTPUT_OPEN, stage::kOutput, "cannot open " + path);
}
std::string header;
header.append("RF64", 4);
put_u32(&header, 0xFFFFFFFFu);
header.append("WAVE", 4);
if (std::fwrite(header.data(), 1, header.size(), file_) != header.size()) {
abort();
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "cannot write " + path);
}
Status status = write_chunk("ds64", std::string(28, '\0'));
if (!status.ok()) {
abort();
return status;
}
std::string fmt;
put_u16(&fmt, 1);
put_u16(&fmt, static_cast<std::uint16_t>(kChannels));
put_u32(&fmt, kRate);
put_u32(&fmt, kRate * kChannels * 3u);
put_u16(&fmt, static_cast<std::uint16_t>(kChannels * 3u));
put_u16(&fmt, 24);
status = write_chunk("fmt ", fmt);
if (!status.ok()) {
abort();
return status;
}
status = write_chunk("data", std::string());
if (!status.ok()) {
abort();
return status;
}
return Status::success();
}
Status AdmBwfWriter::write_chunk(const char id[4], const std::string& body) {
std::string header;
header.append(id, 4);
put_u32(&header, static_cast<std::uint32_t>(body.size()));
if (std::fwrite(header.data(), 1, header.size(), file_) != header.size()) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "chunk header write failed");
}
if (!body.empty() &&
std::fwrite(body.data(), 1, body.size(), file_) != body.size()) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "chunk body write failed");
}
if ((body.size() & 1u) != 0u) {
const char pad = '\0';
if (std::fwrite(&pad, 1, 1, file_) != 1) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "chunk padding write failed");
}
}
return Status::success();
}
Status AdmBwfWriter::flush() {
if (used_ == 0) {
return Status::success();
}
if (file_ == nullptr) {
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer is not open");
}
packed_.clear();
pack_int24(buffer_.data(), used_, kChannels, &packed_);
if (std::fwrite(packed_.data(), 1, packed_.size(), file_) != packed_.size()) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "audio write failed for " + path_);
}
used_ = 0;
return Status::success();
}
Status AdmBwfWriter::write_objects16(const float* planar16) {
if (file_ == nullptr) {
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer is not open");
}
if (planar16 == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput, "null frame");
}
std::size_t source = 0;
while (source < JOC_FRAME_SAMPLES) {
const std::size_t available = block_samples_ - used_;
const std::size_t count =
std::min(available, static_cast<std::size_t>(JOC_FRAME_SAMPLES) - source);
float* target = buffer_.data() + used_ * kChannels;
std::memset(target, 0, count * kChannels * sizeof(float));
for (std::size_t sample = 0; sample < count; ++sample) {
float* row = target + sample * kChannels;
row[3] = planar16[0u * JOC_FRAME_SAMPLES + source + sample];
for (std::size_t object = 0; object < 15u; ++object) {
row[10u + object] =
planar16[(object + 1u) * JOC_FRAME_SAMPLES + source + sample];
}
}
used_ += count;
source += count;
if (used_ == block_samples_) {
const Status status = flush();
if (!status.ok()) {
return status;
}
}
}
frames_ += JOC_FRAME_SAMPLES;
return Status::success();
}
Status AdmBwfWriter::finalize(const std::string& axml, const std::string& chna,
const std::string& dbmd) {
if (file_ == nullptr) {
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer is not open");
}
if (finalized_) {
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer already finalized");
}
Status status = flush();
if (!status.ok()) {
return status;
}
status = write_chunk("axml", axml);
if (!status.ok()) {
return status;
}
status = write_chunk("chna", chna);
if (!status.ok()) {
return status;
}
status = write_chunk("dbmd", dbmd);
if (!status.ok()) {
return status;
}
if (std::fseek(file_, 0, SEEK_END) != 0) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "seek failed for " + path_);
}
const long long total = std::ftell(file_);
if (total < 0) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "tell failed for " + path_);
}
const std::uint64_t data_len = frames_ * kChannels * 3u;
const std::uint32_t data_field =
data_len <= 0xFFFFFFFFull ? static_cast<std::uint32_t>(data_len) : 0xFFFFFFFFu;
if (std::fseek(file_, kDataSizeOffset, SEEK_SET) != 0) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "seek failed for " + path_);
}
char buffer[4];
std::memcpy(buffer, &data_field, 4);
if (std::fwrite(buffer, 1, 4, file_) != 4) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "data size patch failed");
}
std::string ds64;
put_u64(&ds64, static_cast<std::uint64_t>(total) - 8u);
put_u64(&ds64, data_len);
put_u64(&ds64, frames_);
put_u32(&ds64, 0);
if (std::fseek(file_, kDs64BodyOffset, SEEK_SET) != 0) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "seek failed for " + path_);
}
if (std::fwrite(ds64.data(), 1, ds64.size(), file_) != ds64.size()) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "ds64 patch failed");
}
finalized_ = true;
if (std::fclose(file_) != 0) {
file_ = nullptr;
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "close failed for " + path_);
}
file_ = nullptr;
return Status::success();
}
void AdmBwfWriter::abort() {
if (file_ != nullptr) {
std::fclose(file_);
file_ = nullptr;
}
if (!finalized_ && !path_.empty()) {
fs_utf8::remove(path_);
}
}
} // namespace joc::io
-52
View File
@@ -1,52 +0,0 @@
#pragma once
#include <cstdint>
#include <cstdio>
#include <string>
#include <vector>
#include "foundation/status.h"
#include "joc_core.h"
namespace joc::io {
class AdmBwfWriter {
public:
static constexpr std::uint32_t kChannels = 25;
static constexpr std::uint32_t kRate = 48000;
static constexpr std::size_t kDefaultBlockSamples = 131072;
AdmBwfWriter() = default;
~AdmBwfWriter();
AdmBwfWriter(const AdmBwfWriter&) = delete;
AdmBwfWriter& operator=(const AdmBwfWriter&) = delete;
Status open(const std::string& path, std::size_t block_samples = kDefaultBlockSamples);
Status write_objects16(const float* planar16);
Status finalize(const std::string& axml, const std::string& chna, const std::string& dbmd);
// Closes and removes a file that was never finalized (plan 28.3: abort must
void abort();
std::uint64_t frames() const { return frames_; }
bool open_ok() const { return file_ != nullptr; }
private:
Status write_chunk(const char id[4], const std::string& body);
Status flush();
std::FILE* file_ = nullptr;
std::string path_;
std::size_t block_samples_ = kDefaultBlockSamples;
std::size_t used_ = 0;
std::uint64_t frames_ = 0;
std::vector<float> buffer_;
std::string packed_;
bool finalized_ = false;
};
} // namespace joc::io
-1430
View File
File diff suppressed because it is too large Load Diff
-87
View File
@@ -1,87 +0,0 @@
#pragma once
#include <cstdint>
#include <memory>
#include <string>
#include <vector>
#include "foundation/status.h"
// Read-only subset of the HDF5 file format, sized for the SOFA files this
// project consumes: superblock v0, version 2 object headers, fractal-heap link
// and attribute storage, compact and contiguous datasets. The file is opened
// lazily: only the requested dataset's bytes are read into memory, everything
// else (superblock, object headers, heap blocks) is fetched on demand and the
// metadata that was parsed is cached by file address.
//
// Paths are HDF5 link paths ("Data.IR" is a single link name here, "Group/Set"
// walks two links); the empty path names the root group. Byte order is
// normalized on read, so callers never see the file's own endianness.
namespace joc::io {
enum class Hdf5Type {
Unknown,
Int8,
Int16,
Int32,
Int64,
UInt8,
UInt16,
UInt32,
UInt64,
Float32,
Float64,
String,
};
struct Hdf5TypeInfo {
Hdf5Type type = Hdf5Type::Unknown;
std::uint32_t size = 0; // bytes per element as stored in the file
bool big_endian = false;
bool is_signed = false;
};
struct Hdf5DatasetInfo {
std::vector<std::uint64_t> shape;
Hdf5TypeInfo type;
std::uint64_t element_count() const;
};
struct Hdf5Attribute {
Hdf5TypeInfo type;
std::vector<std::uint64_t> shape;
std::vector<std::uint8_t> raw; // C order, host byte order
std::string text; // decoded for fixed-length string attributes
};
class Hdf5File {
public:
Hdf5File();
~Hdf5File();
Hdf5File(Hdf5File&&) noexcept;
Hdf5File& operator=(Hdf5File&&) noexcept;
Hdf5File(const Hdf5File&) = delete;
Hdf5File& operator=(const Hdf5File&) = delete;
Status open(const std::string& path);
bool is_open() const;
// Names of the links of a group ("" is the root group).
Status links(const std::string& group_path, std::vector<std::string>* names) const;
bool has_dataset(const std::string& path) const;
Status dataset_info(const std::string& path, Hdf5DatasetInfo* out) const;
Status read_dataset_raw(const std::string& path, std::vector<std::uint8_t>* out) const;
Status read_dataset_double(const std::string& path, std::vector<double>* out) const;
Status attribute_names(const std::string& object_path, std::vector<std::string>* names) const;
Status attribute(const std::string& object_path, const std::string& name, Hdf5Attribute* out) const;
Status attribute_text(const std::string& object_path, const std::string& name, std::string* out) const;
private:
struct Impl;
std::unique_ptr<Impl> impl_;
};
} // namespace joc::io
-276
View File
@@ -1,276 +0,0 @@
#include "io/inflate.h"
#include <cstring>
namespace joc::io {
namespace {
class LsbBitReader {
public:
LsbBitReader(const std::uint8_t* data, std::size_t size) : data_(data), size_(size) {}
bool ok() const { return ok_; }
std::size_t byte_position() const { return position_ >> 3; }
std::uint32_t bits(unsigned count) {
std::uint32_t value = 0;
for (unsigned i = 0; i < count; ++i) {
if ((position_ >> 3) >= size_) {
ok_ = false;
return value;
}
const std::uint32_t bit = (data_[position_ >> 3] >> (position_ & 7u)) & 1u;
value |= bit << i;
++position_;
}
return value;
}
void align_to_byte() { position_ = (position_ + 7u) & ~static_cast<std::size_t>(7u); }
void skip_bytes(std::size_t count) { position_ += count * 8u; }
private:
const std::uint8_t* data_;
std::size_t size_;
std::size_t position_ = 0;
bool ok_ = true;
};
struct Huffman {
std::uint16_t counts[16] = {};
std::uint16_t symbols[288] = {};
int max_length = 0;
bool build(const std::uint8_t* lengths, int count) {
for (int i = 0; i < 16; ++i) {
counts[i] = 0;
}
for (int i = 0; i < count; ++i) {
counts[lengths[i]]++;
}
counts[0] = 0;
std::uint16_t offsets[16] = {};
std::uint16_t total = 0;
for (int length = 1; length < 16; ++length) {
offsets[length] = total;
total = static_cast<std::uint16_t>(total + counts[length]);
}
if (total == 0) {
return false;
}
for (int symbol = 0; symbol < count; ++symbol) {
const std::uint8_t length = lengths[symbol];
if (length != 0) {
symbols[offsets[length]++] = static_cast<std::uint16_t>(symbol);
}
}
max_length = 15;
while (max_length > 0 && counts[max_length] == 0) {
--max_length;
}
return max_length != 0;
}
int decode(LsbBitReader* reader) const {
int code = 0;
int first = 0;
int index = 0;
for (int length = 1; length <= max_length; ++length) {
code |= static_cast<int>(reader->bits(1));
if (!reader->ok()) {
return -1;
}
const int count = counts[length];
if (code - first < count) {
return symbols[index + (code - first)];
}
index += count;
first = (first + count) << 1;
code <<= 1;
}
return -1;
}
};
constexpr std::uint16_t kLengthBase[29] = {3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 15, 17, 19,
23, 27, 31, 35, 43, 51, 59, 67, 83, 99, 115, 131, 163,
195, 227, 258};
constexpr std::uint8_t kLengthExtra[29] = {0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2,
2, 3, 3, 3, 3, 4, 4, 4, 4, 5, 5, 5, 5, 0};
constexpr std::uint16_t kDistanceBase[30] = {1, 2, 3, 4, 5, 7, 9, 13,
17, 25, 33, 49, 65, 97, 129, 193,
257, 385, 513, 769, 1025, 1537, 2049, 3073,
4097, 6145, 8193, 12289, 16385, 24577};
constexpr std::uint8_t kDistanceExtra[30] = {0, 0, 0, 0, 1, 1, 2, 2, 3, 3, 4, 4, 5, 5, 6,
6, 7, 7, 8, 8, 9, 9, 10, 10, 11, 11, 12, 12, 13, 13};
constexpr std::uint8_t kCodeLengthOrder[19] = {16, 17, 18, 0, 8, 7, 9, 6, 10, 5, 11, 4, 12, 3, 13, 2,
14, 1, 15};
bool inflate_block_data(LsbBitReader* reader, const Huffman& literal, const Huffman& distance,
std::vector<std::uint8_t>* out) {
for (;;) {
const int symbol = literal.decode(reader);
if (symbol < 0) {
return false;
}
if (symbol < 256) {
out->push_back(static_cast<std::uint8_t>(symbol));
continue;
}
if (symbol == 256) {
return true;
}
const int length_index = symbol - 257;
if (length_index >= 29) {
return false;
}
const std::uint32_t length =
kLengthBase[length_index] + reader->bits(kLengthExtra[length_index]);
const int distance_symbol = distance.decode(reader);
if (distance_symbol < 0 || distance_symbol >= 30) {
return false;
}
const std::uint32_t distance_value =
kDistanceBase[distance_symbol] + reader->bits(kDistanceExtra[distance_symbol]);
if (!reader->ok() || distance_value == 0 || distance_value > out->size()) {
return false;
}
const std::size_t start = out->size() - distance_value;
for (std::uint32_t i = 0; i < length; ++i) {
out->push_back((*out)[start + i]);
}
}
}
bool inflate_fixed(LsbBitReader* reader, std::vector<std::uint8_t>* out) {
std::uint8_t lengths[288];
for (int i = 0; i < 144; ++i) { lengths[i] = 8; }
for (int i = 144; i < 256; ++i) { lengths[i] = 9; }
for (int i = 256; i < 280; ++i) { lengths[i] = 7; }
for (int i = 280; i < 288; ++i) { lengths[i] = 8; }
Huffman literal;
if (!literal.build(lengths, 288)) {
return false;
}
std::uint8_t distance_lengths[30];
for (int i = 0; i < 30; ++i) { distance_lengths[i] = 5; }
Huffman distance;
if (!distance.build(distance_lengths, 30)) {
return false;
}
return inflate_block_data(reader, literal, distance, out);
}
bool inflate_dynamic(LsbBitReader* reader, std::vector<std::uint8_t>* out) {
const int literal_count = static_cast<int>(reader->bits(5)) + 257;
const int distance_count = static_cast<int>(reader->bits(5)) + 1;
const int code_length_count = static_cast<int>(reader->bits(4)) + 4;
if (!reader->ok() || literal_count > 286 || distance_count > 30) {
return false;
}
std::uint8_t code_lengths[19] = {};
for (int i = 0; i < code_length_count; ++i) {
code_lengths[kCodeLengthOrder[i]] = static_cast<std::uint8_t>(reader->bits(3));
}
if (!reader->ok()) {
return false;
}
Huffman code_length_tree;
if (!code_length_tree.build(code_lengths, 19)) {
return false;
}
std::uint8_t lengths[288 + 30] = {};
const int total = literal_count + distance_count;
int index = 0;
while (index < total) {
const int symbol = code_length_tree.decode(reader);
if (symbol < 0) {
return false;
}
if (symbol < 16) {
lengths[index++] = static_cast<std::uint8_t>(symbol);
continue;
}
int repeat = 0;
std::uint8_t value = 0;
if (symbol == 16) {
if (index == 0) {
return false;
}
value = lengths[index - 1];
repeat = 3 + static_cast<int>(reader->bits(2));
} else if (symbol == 17) {
repeat = 3 + static_cast<int>(reader->bits(3));
} else {
repeat = 11 + static_cast<int>(reader->bits(7));
}
if (!reader->ok() || index + repeat > total) {
return false;
}
for (int i = 0; i < repeat; ++i) {
lengths[index++] = value;
}
}
Huffman literal;
if (!literal.build(lengths, literal_count)) {
return false;
}
Huffman distance;
if (!distance.build(lengths + literal_count, distance_count)) {
return false;
}
return inflate_block_data(reader, literal, distance, out);
}
} // namespace
bool inflate_raw(const std::uint8_t* data, std::size_t size, std::vector<std::uint8_t>* out) {
if (data == nullptr || out == nullptr) {
return false;
}
out->clear();
LsbBitReader reader(data, size);
for (;;) {
const std::uint32_t final_block = reader.bits(1);
const std::uint32_t type = reader.bits(2);
if (!reader.ok()) {
return false;
}
if (type == 0) {
reader.align_to_byte();
const std::size_t position = reader.byte_position();
if (position + 4 > size) {
return false;
}
const std::uint16_t length = static_cast<std::uint16_t>(data[position] | (data[position + 1] << 8));
const std::uint16_t complement =
static_cast<std::uint16_t>(data[position + 2] | (data[position + 3] << 8));
if (static_cast<std::uint16_t>(length ^ 0xFFFFu) != complement) {
return false;
}
if (position + 4 + length > size) {
return false;
}
out->insert(out->end(), data + position + 4, data + position + 4 + length);
reader.skip_bytes(4u + length);
} else if (type == 1) {
if (!inflate_fixed(&reader, out)) {
return false;
}
} else if (type == 2) {
if (!inflate_dynamic(&reader, out)) {
return false;
}
} else {
return false;
}
if (final_block != 0u) {
break;
}
}
return true;
}
} // namespace joc::io
-12
View File
@@ -1,12 +0,0 @@
#pragma once
#include <cstddef>
#include <cstdint>
#include <vector>
namespace joc::io {
bool inflate_raw(const std::uint8_t* data, std::size_t size, std::vector<std::uint8_t>* out);
} // namespace joc::io
-420
View File
@@ -1,420 +0,0 @@
#include "io/npy.h"
#include <cstring>
namespace joc::io {
namespace {
std::uint16_t read_u16(const std::uint8_t* p) { return static_cast<std::uint16_t>(p[0] | (p[1] << 8)); }
std::uint32_t read_u32(const std::uint8_t* p) {
return static_cast<std::uint32_t>(p[0]) | (static_cast<std::uint32_t>(p[1]) << 8) |
(static_cast<std::uint32_t>(p[2]) << 16) | (static_cast<std::uint32_t>(p[3]) << 24);
}
NpyType classify(const std::string& descr) {
if (descr == "<f8" || descr == "=f8" || descr == "|f8") { return NpyType::Float64; }
if (descr == "<f4" || descr == "=f4") { return NpyType::Float32; }
if (descr == "<i8" || descr == "=i8") { return NpyType::Int64; }
if (descr == "<i4" || descr == "=i4") { return NpyType::Int32; }
if (descr == "<i2" || descr == "=i2") { return NpyType::Int16; }
if (descr == "|u1" || descr == "<u1") { return NpyType::UInt8; }
if (descr == "<c16" || descr == "=c16") { return NpyType::Complex128; }
if (descr.size() > 2 && descr[0] == '<' && descr[1] == 'U') {
return NpyType::Unicode;
}
if (descr.size() > 2 && descr[0] == '=' && descr[1] == 'U') {
return NpyType::Unicode;
}
return NpyType::Unknown;
}
std::size_t unicode_length(const std::string& descr) {
std::size_t index = 0;
while (index < descr.size() && (descr[index] == '<' || descr[index] == '=')) {
++index;
}
if (index >= descr.size() || descr[index] != 'U') {
return 0;
}
++index;
std::size_t value = 0;
bool any = false;
while (index < descr.size() && descr[index] >= '0' && descr[index] <= '9') {
value = value * 10 + static_cast<std::size_t>(descr[index] - '0');
++index;
any = true;
}
return any ? value : 0;
}
bool is_big_endian(const std::string& descr) { return !descr.empty() && descr[0] == '>'; }
bool header_value(const std::string& header, const std::string& key, std::string* out) {
const std::string needle = "'" + key + "'";
const std::size_t position = header.find(needle);
if (position == std::string::npos) {
return false;
}
const std::size_t colon = header.find(':', position + needle.size());
if (colon == std::string::npos) {
return false;
}
std::size_t start = colon + 1;
while (start < header.size() && (header[start] == ' ' || header[start] == '\t')) {
++start;
}
*out = header.substr(start);
return true;
}
} // namespace
std::size_t NpyArray::element_count() const {
std::size_t count = 1;
for (const std::int64_t dimension : shape) {
count *= static_cast<std::size_t>(dimension < 0 ? 0 : dimension);
}
return count;
}
std::size_t NpyArray::element_size() const {
switch (type) {
case NpyType::Float64: return 8;
case NpyType::Float32: return 4;
case NpyType::Int64: return 8;
case NpyType::Int32: return 4;
case NpyType::Int16: return 2;
case NpyType::UInt8: return 1;
case NpyType::Complex128: return 16;
case NpyType::Unicode: return item_bytes;
default: return 0;
}
}
bool parse_npy(const std::uint8_t* data, std::size_t size, NpyArray* out, std::string* error) {
if (data == nullptr || out == nullptr) {
return false;
}
const std::uint8_t magic[6] = {0x93u, 'N', 'U', 'M', 'P', 'Y'};
if (size < 10u || std::memcmp(data, magic, 6) != 0) {
if (error != nullptr) { *error = "not a .npy image"; }
return false;
}
const std::uint8_t major = data[6];
std::size_t header_length = 0;
std::size_t header_offset = 0;
if (major == 1u) {
header_length = read_u16(data + 8);
header_offset = 10;
} else if (major == 2u || major == 3u) {
if (size < 12u) {
if (error != nullptr) { *error = "truncated .npy v2 header"; }
return false;
}
header_length = read_u32(data + 8);
header_offset = 12;
} else {
if (error != nullptr) { *error = "unsupported .npy version " + std::to_string(major); }
return false;
}
if (header_offset + header_length > size) {
if (error != nullptr) { *error = "truncated .npy header"; }
return false;
}
const std::string header(reinterpret_cast<const char*>(data + header_offset), header_length);
out->descr.clear();
std::string value;
if (!header_value(header, "descr", &value)) {
if (error != nullptr) { *error = ".npy header without descr"; }
return false;
}
const std::size_t first_quote = value.find('\'');
const std::size_t second_quote =
first_quote == std::string::npos ? std::string::npos : value.find('\'', first_quote + 1);
if (first_quote == std::string::npos || second_quote == std::string::npos) {
if (error != nullptr) { *error = ".npy descr is not a quoted string"; }
return false;
}
out->descr = value.substr(first_quote + 1, second_quote - first_quote - 1);
out->type = classify(out->descr);
if (out->type == NpyType::Unknown) {
if (error != nullptr) { *error = "unsupported .npy dtype " + out->descr; }
return false;
}
out->item_bytes = 0;
if (out->type == NpyType::Unicode) {
const std::size_t length = unicode_length(out->descr);
if (length == 0) {
if (error != nullptr) { *error = "malformed unicode .npy dtype " + out->descr; }
return false;
}
out->item_bytes = length * 4u;
}
out->fortran_order = header.find("'fortran_order': True") != std::string::npos;
if (!header_value(header, "shape", &value)) {
if (error != nullptr) { *error = ".npy header without shape"; }
return false;
}
out->shape.clear();
for (std::size_t i = 0; i < value.size(); ++i) {
if (value[i] >= '0' && value[i] <= '9') {
long long dimension = 0;
while (i < value.size() && value[i] >= '0' && value[i] <= '9') {
dimension = dimension * 10 + (value[i] - '0');
++i;
}
out->shape.push_back(dimension);
} else if (value[i] == ')') {
break;
}
}
const std::size_t expected = out->element_count() * out->element_size();
if (header_offset + header_length + expected > size) {
if (error != nullptr) {
*error = ".npy payload truncated (need " + std::to_string(expected) + " bytes)";
}
return false;
}
out->data = data + header_offset + header_length;
out->data_bytes = expected;
return true;
}
bool npy_shape_is(const NpyArray& array, const std::vector<std::int64_t>& expected) {
return array.shape == expected;
}
namespace {
template <typename T>
void load_le(const std::uint8_t* source, std::size_t count, bool swap, std::vector<T>* out) {
out->resize(count);
std::memcpy(out->data(), source, count * sizeof(T));
if (swap) {
std::uint8_t* bytes = reinterpret_cast<std::uint8_t*>(out->data());
for (std::size_t i = 0; i < count; ++i) {
for (std::size_t b = 0; b < sizeof(T) / 2; ++b) {
const std::uint8_t temporary = bytes[i * sizeof(T) + b];
bytes[i * sizeof(T) + b] = bytes[i * sizeof(T) + sizeof(T) - 1 - b];
bytes[i * sizeof(T) + sizeof(T) - 1 - b] = temporary;
}
}
}
}
} // namespace
bool npy_to_double(const NpyArray& array, std::vector<double>* out, std::string* error) {
const bool swap = is_big_endian(array.descr);
const std::size_t count = array.element_count();
switch (array.type) {
case NpyType::Float64:
load_le(array.data, count, swap, out);
return true;
case NpyType::Complex128:
load_le(array.data, count * 2u, swap, out);
return true;
case NpyType::Float32: {
std::vector<float> values;
load_le(array.data, count, swap, &values);
out->resize(count);
for (std::size_t i = 0; i < count; ++i) {
(*out)[i] = static_cast<double>(values[i]);
}
return true;
}
case NpyType::Int64: {
std::vector<std::int64_t> values;
load_le(array.data, count, swap, &values);
out->resize(count);
for (std::size_t i = 0; i < count; ++i) {
(*out)[i] = static_cast<double>(values[i]);
}
return true;
}
case NpyType::Int32: {
std::vector<std::int32_t> values;
load_le(array.data, count, swap, &values);
out->resize(count);
for (std::size_t i = 0; i < count; ++i) {
(*out)[i] = static_cast<double>(values[i]);
}
return true;
}
case NpyType::Int16: {
std::vector<std::int16_t> values;
load_le(array.data, count, swap, &values);
out->resize(count);
for (std::size_t i = 0; i < count; ++i) {
(*out)[i] = static_cast<double>(values[i]);
}
return true;
}
case NpyType::UInt8: {
out->resize(count);
for (std::size_t i = 0; i < count; ++i) {
(*out)[i] = static_cast<double>(array.data[i]);
}
return true;
}
default:
if (error != nullptr) { *error = "cannot convert " + array.descr + " to double"; }
return false;
}
}
bool npy_to_int16(const NpyArray& array, std::vector<std::int16_t>* out, std::string* error) {
const bool swap = is_big_endian(array.descr);
const std::size_t count = array.element_count();
switch (array.type) {
case NpyType::Int16:
load_le(array.data, count, swap, out);
return true;
case NpyType::Int32: {
std::vector<std::int32_t> values;
load_le(array.data, count, swap, &values);
out->resize(count);
for (std::size_t i = 0; i < count; ++i) {
(*out)[i] = static_cast<std::int16_t>(values[i]);
}
return true;
}
case NpyType::Int64: {
std::vector<std::int64_t> values;
load_le(array.data, count, swap, &values);
out->resize(count);
for (std::size_t i = 0; i < count; ++i) {
(*out)[i] = static_cast<std::int16_t>(values[i]);
}
return true;
}
default:
if (error != nullptr) { *error = "cannot convert " + array.descr + " to int16"; }
return false;
}
}
bool npy_to_int32(const NpyArray& array, std::vector<std::int32_t>* out, std::string* error) {
const bool swap = is_big_endian(array.descr);
const std::size_t count = array.element_count();
switch (array.type) {
case NpyType::Int32:
load_le(array.data, count, swap, out);
return true;
case NpyType::Int64: {
std::vector<std::int64_t> values;
load_le(array.data, count, swap, &values);
out->resize(count);
for (std::size_t i = 0; i < count; ++i) {
(*out)[i] = static_cast<std::int32_t>(values[i]);
}
return true;
}
case NpyType::Int16: {
std::vector<std::int16_t> values;
load_le(array.data, count, swap, &values);
out->resize(count);
for (std::size_t i = 0; i < count; ++i) {
(*out)[i] = static_cast<std::int32_t>(values[i]);
}
return true;
}
default:
if (error != nullptr) { *error = "cannot convert " + array.descr + " to int32"; }
return false;
}
}
bool npy_to_uint8(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error) {
if (array.type != NpyType::UInt8) {
if (error != nullptr) { *error = "cannot convert " + array.descr + " to uint8"; }
return false;
}
out->assign(array.data, array.data + array.element_count());
return true;
}
bool npy_unicode_to_utf8(const NpyArray& array, std::string* out, std::string* error) {
if (array.type != NpyType::Unicode) {
if (error != nullptr) { *error = "not a unicode .npy member: " + array.descr; }
return false;
}
if (array.shape.size() != 0) {
if (error != nullptr) { *error = "unicode .npy member must be a scalar"; }
return false;
}
out->clear();
const std::size_t count = array.item_bytes / 4u;
for (std::size_t i = 0; i < count; ++i) {
const std::uint8_t* p = array.data + i * 4u;
const std::uint32_t code = static_cast<std::uint32_t>(p[0]) | (static_cast<std::uint32_t>(p[1]) << 8) |
(static_cast<std::uint32_t>(p[2]) << 16) |
(static_cast<std::uint32_t>(p[3]) << 24);
if (code == 0u) {
break;
}
if (code < 0x80u) {
out->push_back(static_cast<char>(code));
} else if (code < 0x800u) {
out->push_back(static_cast<char>(0xC0u | (code >> 6)));
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
} else if (code < 0x10000u) {
out->push_back(static_cast<char>(0xE0u | (code >> 12)));
out->push_back(static_cast<char>(0x80u | ((code >> 6) & 0x3Fu)));
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
} else {
out->push_back(static_cast<char>(0xF0u | (code >> 18)));
out->push_back(static_cast<char>(0x80u | ((code >> 12) & 0x3Fu)));
out->push_back(static_cast<char>(0x80u | ((code >> 6) & 0x3Fu)));
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
}
}
return true;
}
bool npy_to_c_order(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error) {
const std::size_t element = array.element_size();
if (element == 0) {
if (error != nullptr) { *error = "unsupported element size for " + array.descr; }
return false;
}
if (!array.fortran_order) {
out->assign(array.data, array.data + array.data_bytes);
return true;
}
const std::size_t dimensions = array.shape.size();
if (dimensions == 0) {
out->assign(array.data, array.data + element);
return true;
}
// Source (Fortran) strides in elements; destination is C order.
std::vector<std::size_t> source_stride(dimensions, 1);
std::size_t running = 1;
for (std::size_t d = 0; d < dimensions; ++d) {
source_stride[d] = running;
running *= static_cast<std::size_t>(array.shape[d]);
}
out->assign(array.data_bytes, 0);
std::vector<std::size_t> index(dimensions, 0);
const std::size_t total = array.element_count();
for (std::size_t linear = 0; linear < total; ++linear) {
std::size_t remainder = linear;
for (std::size_t d = dimensions; d-- > 0;) {
index[d] = remainder % static_cast<std::size_t>(array.shape[d]);
remainder /= static_cast<std::size_t>(array.shape[d]);
}
std::size_t source = 0;
for (std::size_t d = 0; d < dimensions; ++d) {
source += index[d] * source_stride[d];
}
std::memcpy(out->data() + linear * element, array.data + source * element, element);
}
return true;
}
} // namespace joc::io
-42
View File
@@ -1,42 +0,0 @@
#pragma once
#include <cstdint>
#include <string>
#include <vector>
namespace joc::io {
enum class NpyType { Unknown, Float64, Float32, Int64, Int32, Int16, UInt8, Complex128, Unicode };
struct NpyArray {
std::string descr;
NpyType type = NpyType::Unknown;
bool fortran_order = false;
std::vector<std::int64_t> shape;
const std::uint8_t* data = nullptr;
std::size_t data_bytes = 0;
std::size_t item_bytes = 0; // bytes per element as stored
std::size_t element_count() const;
std::size_t element_size() const; // bytes per element in the file
};
// Parses the header of one `.npy` image. `data` must outlive the NpyArray.
bool parse_npy(const std::uint8_t* data, std::size_t size, NpyArray* out, std::string* error);
bool npy_to_double(const NpyArray& array, std::vector<double>* out, std::string* error);
bool npy_to_int16(const NpyArray& array, std::vector<std::int16_t>* out, std::string* error);
bool npy_to_int32(const NpyArray& array, std::vector<std::int32_t>* out, std::string* error);
bool npy_to_uint8(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error);
bool npy_unicode_to_utf8(const NpyArray& array, std::string* out, std::string* error);
// Materializes the array in C order as raw element bytes. Fortran-order members
// hybrid synthesis table as [count][4] row-major while the shipped table stores it
// Fortran-order, so passing the file bytes straight through would transpose it.
bool npy_to_c_order(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error);
bool npy_shape_is(const NpyArray& array, const std::vector<std::int64_t>& expected);
} // namespace joc::io
-263
View File
@@ -1,263 +0,0 @@
#include "io/npy_writer.h"
#include <array>
#include <charconv>
#include <cmath>
#include <cstdio>
#include <cstring>
#include <string>
#include "foundation/fs_utf8.h"
#include "io/zip_reader.h"
namespace joc::io {
namespace {
constexpr std::size_t kNpyHeaderAlignment = 64;
void append_u16(std::vector<std::uint8_t>* out, std::uint16_t value) {
out->push_back(static_cast<std::uint8_t>(value & 0xFFu));
out->push_back(static_cast<std::uint8_t>((value >> 8) & 0xFFu));
}
void append_u32(std::vector<std::uint8_t>* out, std::uint32_t value) {
for (int index = 0; index < 4; ++index) {
out->push_back(static_cast<std::uint8_t>((value >> (8 * index)) & 0xFFu));
}
}
void append_bytes(std::vector<std::uint8_t>* out, const void* data, std::size_t size) {
const std::uint8_t* bytes = static_cast<const std::uint8_t*>(data);
out->insert(out->end(), bytes, bytes + size);
}
std::string shape_literal(const std::vector<std::uint64_t>& shape) {
if (shape.empty()) {
return "()";
}
std::string text = "(";
for (std::size_t index = 0; index < shape.size(); ++index) {
if (index != 0u) {
text += ", ";
}
text += std::to_string(shape[index]);
}
if (shape.size() == 1u) {
text += ",";
}
text += ")";
return text;
}
} // namespace
std::vector<std::uint8_t> npy_image(const std::string& descr,
const std::vector<std::uint64_t>& shape,
const std::vector<std::uint8_t>& data) {
std::string header = "{'descr': '" + descr + "', 'fortran_order': False, 'shape': " +
shape_literal(shape) + ", }";
// NumPy pads the header so that the payload starts on a 64-byte boundary.
const std::size_t preamble = 10u; // magic, version, two byte header length
std::size_t total = preamble + header.size() + 1u;
const std::size_t padding = (kNpyHeaderAlignment - (total % kNpyHeaderAlignment)) %
kNpyHeaderAlignment;
header.append(padding, ' ');
header.push_back('\n');
std::vector<std::uint8_t> out;
out.reserve(preamble + header.size() + data.size());
static const std::uint8_t kMagic[6] = {0x93u, 'N', 'U', 'M', 'P', 'Y'};
append_bytes(&out, kMagic, sizeof(kMagic));
out.push_back(1u); // major
out.push_back(0u); // minor
append_u16(&out, static_cast<std::uint16_t>(header.size()));
append_bytes(&out, header.data(), header.size());
append_bytes(&out, data.data(), data.size());
return out;
}
std::vector<std::uint8_t> zip_bytes(const std::vector<NpyMember>& members) {
std::vector<std::uint8_t> out;
struct Entry {
std::string name;
std::uint32_t crc = 0;
std::uint32_t size = 0;
std::uint32_t offset = 0;
};
std::vector<Entry> entries;
entries.reserve(members.size());
for (const NpyMember& member : members) {
const std::string name = member.name + ".npy";
const std::vector<std::uint8_t> payload = npy_image(member.descr, member.shape, member.data);
Entry entry;
entry.name = name;
entry.crc = crc32_of(payload.data(), payload.size());
entry.size = static_cast<std::uint32_t>(payload.size());
entry.offset = static_cast<std::uint32_t>(out.size());
entries.push_back(entry);
append_u32(&out, 0x04034B50u); // local file header
append_u16(&out, 20u); // version needed
append_u16(&out, 0u); // flags
append_u16(&out, 0u); // method: stored
append_u16(&out, 0u); // time
append_u16(&out, 0x2821u); // date: 2000-01-01, fixed for reproducibility
append_u32(&out, entry.crc);
append_u32(&out, entry.size);
append_u32(&out, entry.size);
append_u16(&out, static_cast<std::uint16_t>(name.size()));
append_u16(&out, 0u); // extra length
append_bytes(&out, name.data(), name.size());
append_bytes(&out, payload.data(), payload.size());
}
const std::uint32_t directory_offset = static_cast<std::uint32_t>(out.size());
for (const Entry& entry : entries) {
append_u32(&out, 0x02014B50u); // central directory header
append_u16(&out, 20u); // version made by
append_u16(&out, 20u); // version needed
append_u16(&out, 0u); // flags
append_u16(&out, 0u); // method: stored
append_u16(&out, 0u); // time
append_u16(&out, 0x2821u); // date
append_u32(&out, entry.crc);
append_u32(&out, entry.size);
append_u32(&out, entry.size);
append_u16(&out, static_cast<std::uint16_t>(entry.name.size()));
append_u16(&out, 0u); // extra
append_u16(&out, 0u); // comment
append_u16(&out, 0u); // disk
append_u16(&out, 0u); // internal attributes
append_u32(&out, 0u); // external attributes
append_u32(&out, entry.offset);
append_bytes(&out, entry.name.data(), entry.name.size());
}
const std::uint32_t directory_size = static_cast<std::uint32_t>(out.size()) - directory_offset;
append_u32(&out, 0x06054B50u); // end of central directory
append_u16(&out, 0u);
append_u16(&out, 0u);
append_u16(&out, static_cast<std::uint16_t>(entries.size()));
append_u16(&out, static_cast<std::uint16_t>(entries.size()));
append_u32(&out, directory_size);
append_u32(&out, directory_offset);
append_u16(&out, 0u);
return out;
}
bool write_zip(const std::string& path, const std::vector<NpyMember>& members,
std::string* error) {
const std::vector<std::uint8_t> bytes = zip_bytes(members);
std::FILE* stream = fs_utf8::fopen(path, "wb");
if (stream == nullptr) {
if (error != nullptr) {
*error = "cannot open " + path + " for writing";
}
return false;
}
const std::size_t written = std::fwrite(bytes.data(), 1, bytes.size(), stream);
const bool flushed = std::fclose(stream) == 0;
if (written != bytes.size() || !flushed) {
if (error != nullptr) {
*error = "short write to " + path;
}
return false;
}
return true;
}
std::vector<std::uint8_t> utf8_to_utf32le(const std::string& text) {
std::vector<std::uint8_t> out;
out.reserve(text.size() * 4u);
std::size_t index = 0;
while (index < text.size()) {
const std::uint8_t lead = static_cast<std::uint8_t>(text[index]);
std::uint32_t code = 0;
std::size_t extra = 0;
if (lead < 0x80u) {
code = lead;
} else if ((lead & 0xE0u) == 0xC0u) {
code = lead & 0x1Fu;
extra = 1;
} else if ((lead & 0xF0u) == 0xE0u) {
code = lead & 0x0Fu;
extra = 2;
} else if ((lead & 0xF8u) == 0xF0u) {
code = lead & 0x07u;
extra = 3;
} else {
code = 0xFFFDu; // invalid lead byte: substitute rather than fail
extra = 0;
}
++index;
for (std::size_t count = 0; count < extra && index < text.size(); ++count) {
code = (code << 6) | (static_cast<std::uint8_t>(text[index]) & 0x3Fu);
++index;
}
for (int byte = 0; byte < 4; ++byte) {
out.push_back(static_cast<std::uint8_t>((code >> (8 * byte)) & 0xFFu));
}
}
return out;
}
std::string python_float_repr(double value) {
if (std::isnan(value)) {
return "NaN";
}
if (std::isinf(value)) {
return value > 0.0 ? "Infinity" : "-Infinity";
}
// to_chars gives the shortest round-trip digits; Python's repr uses the same
// digits but its own notation, so the digits are re-laid-out here.
std::array<char, 64> buffer{};
const std::to_chars_result converted =
std::to_chars(buffer.data(), buffer.data() + buffer.size(), value);
std::string text(buffer.data(), converted.ptr);
const bool negative = !text.empty() && text[0] == '-';
const std::string body = negative ? text.substr(1) : text;
const std::size_t exponent_at = body.find_first_of("eE");
std::string digits = body;
int exponent = 0;
if (exponent_at != std::string::npos) {
digits = body.substr(0, exponent_at);
exponent = std::atoi(body.c_str() + exponent_at + 1);
}
const std::size_t point = digits.find('.');
std::string mantissa = digits;
if (point != std::string::npos) {
mantissa = digits.substr(0, point) + digits.substr(point + 1);
exponent += static_cast<int>(point) - 1;
} else {
exponent += static_cast<int>(digits.size()) - 1;
}
while (mantissa.size() > 1u && mantissa.back() == '0') {
mantissa.pop_back();
}
// Python switches to exponent notation below 1e-4 and at 1e16 and above.
std::string result;
if (exponent < -4 || exponent >= 16) {
result = mantissa.substr(0, 1);
if (mantissa.size() > 1u) {
result += "." + mantissa.substr(1);
}
char tail[16];
std::snprintf(tail, sizeof(tail), "e%+03d", exponent);
result += tail;
} else if (exponent >= 0) {
if (static_cast<std::size_t>(exponent) + 1u >= mantissa.size()) {
result = mantissa + std::string(static_cast<std::size_t>(exponent) + 1u - mantissa.size(), '0');
result += ".0";
} else {
result = mantissa.substr(0, static_cast<std::size_t>(exponent) + 1u) + "." +
mantissa.substr(static_cast<std::size_t>(exponent) + 1u);
}
} else {
result = "0." + std::string(static_cast<std::size_t>(-exponent - 1), '0') + mantissa;
}
return negative ? "-" + result : result;
}
} // namespace joc::io
-39
View File
@@ -1,39 +0,0 @@
#pragma once
#include <cstdint>
#include <string>
#include <vector>
namespace joc::io {
// NPY 1.0 images and a minimal ZIP container, used to write the compiled HRTF
// cache in exactly the layout the reader (and NumPy) expects. Only what the
// cache needs is implemented: little-endian C-order arrays and stored members.
struct NpyMember {
std::string name; // archive member name, without the .npy suffix
std::string descr; // NumPy dtype string, e.g. "<f8", "<c16", "<U123"
std::vector<std::uint64_t> shape;
std::vector<std::uint8_t> data; // C order payload in the dtype's byte order
};
// Serializes one array as an NPY 1.0 image (magic, header, 64-byte aligned).
std::vector<std::uint8_t> npy_image(const std::string& descr,
const std::vector<std::uint64_t>& shape,
const std::vector<std::uint8_t>& data);
// Writes a ZIP archive with stored (uncompressed) members. The upstream reader
// accepts stored members, and compression would need a deflate encoder.
bool write_zip(const std::string& path, const std::vector<NpyMember>& members,
std::string* error);
// Serializes the archive in memory (same layout as write_zip).
std::vector<std::uint8_t> zip_bytes(const std::vector<NpyMember>& members);
// UTF-8 text as the payload of a NumPy Unicode scalar string ('<U<n>').
std::vector<std::uint8_t> utf8_to_utf32le(const std::string& text);
// Python's repr() for a double: shortest round-trip digits with Python's
// exponent rules, which is what json.dumps emits for the cache metadata.
std::string python_float_repr(double value);
} // namespace joc::io
-186
View File
@@ -1,186 +0,0 @@
#include "io/process.h"
#include "foundation/fs_utf8.h"
#include <cstdio>
#include <filesystem>
#include <fstream>
#include <random>
#if defined(_WIN32)
#define WIN32_LEAN_AND_MEAN
#define NOMINMAX
#include <windows.h>
#else
#include <sys/wait.h>
#endif
namespace joc::io {
namespace {
std::string quote_argument(const std::string& argument) {
if (!argument.empty() && argument.find_first_of(" \t\"") == std::string::npos) {
return argument;
}
std::string quoted = "\"";
unsigned backslashes = 0;
for (const char c : argument) {
if (c == '\\') {
++backslashes;
continue;
}
if (c == '"') {
quoted.append(backslashes * 2 + 1, '\\');
quoted.push_back('"');
backslashes = 0;
continue;
}
quoted.append(backslashes, '\\');
backslashes = 0;
quoted.push_back(c);
}
quoted.append(backslashes * 2, '\\');
quoted.push_back('"');
return quoted;
}
std::string tail_of(const std::string& text, std::size_t limit) {
if (text.size() <= limit) {
return text;
}
return text.substr(text.size() - limit);
}
} // namespace
Status run_process(const std::vector<std::string>& argv, ProcessResult* out) {
if (out == nullptr || argv.empty()) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput, "empty command");
}
out->output.clear();
out->exit_code = 0;
const std::filesystem::path log_path =
std::filesystem::temp_directory_path() /
("joc_process_" + std::to_string(std::random_device{}()) + ".log");
auto read_log = [&]() {
#if defined(_WIN32)
return; // the Windows branch reads the handle it opened
#else
std::ifstream log = fs_utf8::open_input(fs_utf8::from_path(log_path));
if (log) {
std::string text((std::istreambuf_iterator<char>(log)),
std::istreambuf_iterator<char>());
out->output = tail_of(text, 4096);
}
#endif
};
#if defined(_WIN32)
std::string command;
for (std::size_t i = 0; i < argv.size(); ++i) {
if (i != 0) {
command.push_back(' ');
}
command += quote_argument(argv[i]);
}
auto widen = [](const std::string& text) {
if (text.empty()) {
return std::wstring();
}
const int size = MultiByteToWideChar(CP_UTF8, 0, text.c_str(),
static_cast<int>(text.size()), nullptr, 0);
std::wstring wide(static_cast<std::size_t>(size), L'\0');
MultiByteToWideChar(CP_UTF8, 0, text.c_str(), static_cast<int>(text.size()), wide.data(),
size);
return wide;
};
const std::wstring wide_command = widen(command);
const std::wstring wide_log = widen(fs_utf8::from_path(log_path));
SECURITY_ATTRIBUTES attributes{};
attributes.nLength = sizeof(attributes);
attributes.bInheritHandle = TRUE;
// DELETE access plus FILE_FLAG_DELETE_ON_CLOSE means the log disappears when
// the last handle goes away - including when this process is killed, which
// would otherwise leave joc_process_*.log litter in the temp directory.
HANDLE log_handle = CreateFileW(
wide_log.c_str(), GENERIC_READ | GENERIC_WRITE | DELETE,
FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, &attributes, CREATE_ALWAYS,
FILE_ATTRIBUTE_NORMAL | FILE_FLAG_DELETE_ON_CLOSE, nullptr);
if (log_handle == INVALID_HANDLE_VALUE) {
return Status::fail(JOC_ERR_IO, stage::kOutput, "cannot create the process log file");
}
// A delete-on-close file cannot be reopened by name (it is delete-pending), so
// the child's output is read back through the handle it wrote to.
auto read_log_handle = [&]() {
LARGE_INTEGER start{};
start.QuadPart = 0;
if (!SetFilePointerEx(log_handle, start, nullptr, FILE_BEGIN)) {
return;
}
std::string text;
char buffer[1024];
DWORD count = 0;
while (ReadFile(log_handle, buffer, sizeof(buffer), &count, nullptr) && count > 0) {
text.append(buffer, count);
}
out->output = tail_of(text, 4096);
};
STARTUPINFOW startup{};
startup.cb = sizeof(startup);
startup.dwFlags = STARTF_USESTDHANDLES;
startup.hStdOutput = log_handle;
startup.hStdError = log_handle;
startup.hStdInput = GetStdHandle(STD_INPUT_HANDLE);
PROCESS_INFORMATION process{};
std::vector<wchar_t> mutable_command(wide_command.begin(), wide_command.end());
mutable_command.push_back(L'\0');
const BOOL started = CreateProcessW(nullptr, mutable_command.data(), nullptr, nullptr, TRUE,
CREATE_NO_WINDOW, nullptr, nullptr, &startup, &process);
if (!started) {
CloseHandle(log_handle); // delete-on-close removes the file
return Status::fail(JOC_ERR_LIBRARY_MISSING, stage::kOutput,
"cannot start " + argv[0] + " (is it on PATH?)");
}
WaitForSingleObject(process.hProcess, INFINITE);
DWORD exit_code = 0;
GetExitCodeProcess(process.hProcess, &exit_code);
CloseHandle(process.hThread);
CloseHandle(process.hProcess);
// Read the log before the delete-on-close handle goes away.
read_log_handle();
CloseHandle(log_handle);
out->exit_code = static_cast<std::uint32_t>(exit_code);
#else
std::string command;
for (std::size_t i = 0; i < argv.size(); ++i) {
if (i != 0) {
command.push_back(' ');
}
command += quote_argument(argv[i]);
}
command += " > " + quote_argument(fs_utf8::from_path(log_path)) + " 2>&1";
const int status = std::system(command.c_str());
// system() reports a wait status, not the child's exit code.
out->exit_code = status == -1 ? 127u
: WIFEXITED(status) ? static_cast<std::uint32_t>(WEXITSTATUS(status))
: 128u;
read_log();
std::error_code ignored;
std::filesystem::remove(log_path, ignored);
#endif
if (out->exit_code != 0) {
return Status::fail(JOC_ERR_INPUT_FORMAT, stage::kOutput,
argv[0] + " failed with exit code " + std::to_string(out->exit_code) +
(out->output.empty() ? "" : ": " + tail_of(out->output, 400)));
}
return Status::success();
}
} // namespace joc::io
-19
View File
@@ -1,19 +0,0 @@
#pragma once
#include <cstdint>
#include <string>
#include <vector>
#include "foundation/status.h"
namespace joc::io {
struct ProcessResult {
std::uint32_t exit_code = 0;
std::string output;
};
Status run_process(const std::vector<std::string>& argv, ProcessResult* out);
} // namespace joc::io
-242
View File
@@ -1,242 +0,0 @@
#include "io/wav_writer.h"
#include "foundation/fs_utf8.h"
#include <cmath>
#include <cstring>
#include <filesystem>
#include <limits>
#include <vector>
namespace joc::io {
namespace {
constexpr std::uint16_t kWaveFormatPcm = 0x0001;
constexpr std::uint16_t kWaveFormatIeeeFloat = 0x0003;
constexpr std::uint16_t kWaveFormatExtensible = 0xFFFE;
constexpr std::uint8_t kPcmGuid[16] = {0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x10, 0x00,
0x80, 0x00, 0x00, 0xAA, 0x00, 0x38, 0x9B, 0x71};
constexpr std::uint8_t kFloatGuid[16] = {0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x10, 0x00,
0x80, 0x00, 0x00, 0xAA, 0x00, 0x38, 0x9B, 0x71};
void put_u16(std::string* out, std::uint16_t value) {
char buffer[2];
std::memcpy(buffer, &value, 2);
out->append(buffer, 2);
}
void put_u32(std::string* out, std::uint32_t value) {
char buffer[4];
std::memcpy(buffer, &value, 4);
out->append(buffer, 4);
}
void put_u64(std::string* out, std::uint64_t value) {
char buffer[8];
std::memcpy(buffer, &value, 8);
out->append(buffer, 8);
}
// Port of speaker_wav._fmt_chunk.
std::string fmt_chunk(std::uint32_t channels, std::uint32_t rate, SampleFormat format, WavInfo* info) {
std::uint16_t simple_tag = 0;
const std::uint8_t* guid = nullptr;
if (format == SampleFormat::Float32) {
info->bits_per_sample = 32;
info->bytes_per_sample = 4;
simple_tag = kWaveFormatIeeeFloat;
guid = kFloatGuid;
} else {
info->bits_per_sample = 24;
info->bytes_per_sample = 3;
simple_tag = kWaveFormatPcm;
guid = kPcmGuid;
}
const std::uint32_t block_align = channels * info->bytes_per_sample;
const std::uint32_t byte_rate = rate * block_align;
info->block_align = block_align;
std::string body;
if (channels <= 2) {
put_u16(&body, simple_tag);
put_u16(&body, static_cast<std::uint16_t>(channels));
put_u32(&body, rate);
put_u32(&body, byte_rate);
put_u16(&body, static_cast<std::uint16_t>(block_align));
put_u16(&body, static_cast<std::uint16_t>(info->bits_per_sample));
} else {
put_u16(&body, kWaveFormatExtensible);
put_u16(&body, static_cast<std::uint16_t>(channels));
put_u32(&body, rate);
put_u32(&body, byte_rate);
put_u16(&body, static_cast<std::uint16_t>(block_align));
put_u16(&body, static_cast<std::uint16_t>(info->bits_per_sample));
put_u16(&body, 22);
put_u16(&body, static_cast<std::uint16_t>(info->bits_per_sample));
put_u32(&body, 0);
body.append(reinterpret_cast<const char*>(guid), 16);
}
return body;
}
// int32 conversion identical to NumPy's float32 -> int32 cast after clipping.
std::int32_t to_int32(const float value) {
if (!std::isfinite(value)) {
return std::numeric_limits<std::int32_t>::min();
}
return static_cast<std::int32_t>(value);
}
} // namespace
void pack_int24(const float* interleaved, std::size_t frames, std::size_t channels,
std::string* out) {
const std::size_t count = frames * channels;
out->resize(count * 3);
char* target = out->data();
for (std::size_t i = 0; i < count; ++i) {
float value = interleaved[i];
if (value > 1.0f) {
value = 1.0f;
} else if (value < -1.0f) {
value = -1.0f;
}
const std::int32_t scaled = to_int32(value * 8388607.0f);
const std::uint32_t bits = static_cast<std::uint32_t>(scaled);
target[i * 3 + 0] = static_cast<char>(bits & 0xFFu);
target[i * 3 + 1] = static_cast<char>((bits >> 8) & 0xFFu);
target[i * 3 + 2] = static_cast<char>((bits >> 16) & 0xFFu);
}
}
WavWriter::~WavWriter() {
if (file_ != nullptr) {
std::fclose(file_);
file_ = nullptr;
}
}
Status WavWriter::open(const std::string& path, std::uint32_t channels, std::uint32_t rate,
SampleFormat format, std::uint64_t total_frames) {
if (channels == 0 || rate == 0) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput,
"WAV writer needs a positive channel count and rate");
}
path_ = path;
channels_ = channels;
total_frames_ = total_frames;
frames_written_ = 0;
finalized_ = false;
info_ = WavInfo{};
info_.format = format;
const std::string fmt = fmt_chunk(channels, rate, format, &info_);
const std::uint64_t data_size = total_frames * info_.block_align;
info_.data_bytes = data_size;
const std::uint64_t riff_file_size = 12u + 8u + fmt.size() + 8u + data_size;
const bool rf64 = (riff_file_size - 8u) > 0xFFFFFFFFull;
info_.rf64 = rf64;
std::string header;
if (rf64) {
const std::uint64_t file_size = 12u + 36u + 8u + fmt.size() + 8u + data_size;
header.append("RF64", 4);
put_u32(&header, 0xFFFFFFFFu);
header.append("WAVE", 4);
header.append("ds64", 4);
put_u32(&header, 28);
put_u64(&header, file_size - 8u);
put_u64(&header, data_size);
put_u64(&header, total_frames);
put_u32(&header, 0);
} else {
header.append("RIFF", 4);
put_u32(&header, static_cast<std::uint32_t>(riff_file_size - 8u));
header.append("WAVE", 4);
}
header.append("fmt ", 4);
put_u32(&header, static_cast<std::uint32_t>(fmt.size()));
header.append(fmt);
header.append("data", 4);
put_u32(&header, rf64 ? 0xFFFFFFFFu : static_cast<std::uint32_t>(data_size));
file_ = fs_utf8::fopen(path, "wb");
if (file_ == nullptr) {
// The reference creates the parent directory itself.
std::error_code ignored;
const std::filesystem::path parent = std::filesystem::path(path).parent_path();
if (!parent.empty()) {
std::filesystem::create_directories(parent, ignored);
}
file_ = fs_utf8::fopen(path, "wb");
}
if (file_ == nullptr) {
return Status::fail(JOC_ERR_OUTPUT_OPEN, stage::kOutput, "cannot open " + path);
}
if (std::fwrite(header.data(), 1, header.size(), file_) != header.size()) {
std::fclose(file_);
file_ = nullptr;
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "cannot write header to " + path);
}
return Status::success();
}
Status WavWriter::write(const double* interleaved, std::size_t frames) {
if (file_ == nullptr) {
return Status::fail(JOC_ERR_STATE, stage::kOutput, "WAV writer is not open");
}
if (frames == 0) {
return Status::success();
}
if (frames_written_ + frames > total_frames_) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput,
"WAV writer received more frames than the header declared (declared " +
std::to_string(total_frames_) + ", written " +
std::to_string(frames_written_) + ", requested " +
std::to_string(frames) + ")");
}
const std::size_t count = frames * channels_;
if (info_.format == SampleFormat::Float32) {
std::vector<float> converted(count);
for (std::size_t i = 0; i < count; ++i) {
converted[i] = static_cast<float>(interleaved[i]);
}
if (std::fwrite(converted.data(), sizeof(float), count, file_) != count) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "write failed for " + path_);
}
} else {
std::vector<float> converted(count);
for (std::size_t i = 0; i < count; ++i) {
converted[i] = static_cast<float>(interleaved[i]);
}
std::string packed;
pack_int24(converted.data(), frames, channels_, &packed);
if (std::fwrite(packed.data(), 1, packed.size(), file_) != packed.size()) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "write failed for " + path_);
}
}
frames_written_ += frames;
return Status::success();
}
Status WavWriter::finalize() {
if (file_ == nullptr) {
return Status::fail(JOC_ERR_STATE, stage::kOutput, "WAV writer is not open");
}
if (frames_written_ != total_frames_) {
std::fclose(file_);
file_ = nullptr;
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput,
"WAV writer wrote " + std::to_string(frames_written_) + " of " +
std::to_string(total_frames_) + " frames");
}
const int result = std::fclose(file_);
file_ = nullptr;
finalized_ = true;
if (result != 0) {
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "close failed for " + path_);
}
return Status::success();
}
} // namespace joc::io
-61
View File
@@ -1,61 +0,0 @@
// Port of src/speaker_wav.py.
#pragma once
#include <cstdint>
#include <cstdio>
#include <string>
#include "foundation/status.h"
#include "joc_core.h"
namespace joc::io {
enum class SampleFormat { Float32, Int24 };
struct WavInfo {
SampleFormat format = SampleFormat::Float32;
std::uint32_t bits_per_sample = 32;
std::uint32_t bytes_per_sample = 4;
std::uint32_t block_align = 0;
std::uint64_t data_bytes = 0;
bool rf64 = false;
};
// int24 packing shared by the WAV and ADM writers:
// trunc(clip(v, -1, 1) * 8388607.0f) with the low three bytes written LE.
// NaN follows NumPy's float->int cast (INT32_MIN) so that the C++ conversion is
// never undefined; the reference passes it through unguarded (plan TD-3.11).
void pack_int24(const float* interleaved, std::size_t frames, std::size_t channels,
std::string* out);
class WavWriter {
public:
WavWriter() = default;
~WavWriter();
WavWriter(const WavWriter&) = delete;
WavWriter& operator=(const WavWriter&) = delete;
// `total_frames` must be known up front: the header depends on it.
Status open(const std::string& path, std::uint32_t channels, std::uint32_t rate,
SampleFormat format, std::uint64_t total_frames);
Status write(const double* interleaved, std::size_t frames);
Status finalize();
const WavInfo& info() const { return info_; }
std::uint64_t frames_written() const { return frames_written_; }
private:
std::FILE* file_ = nullptr;
std::string path_;
WavInfo info_;
std::uint32_t channels_ = 0;
std::uint64_t total_frames_ = 0;
std::uint64_t frames_written_ = 0;
bool finalized_ = false;
};
} // namespace joc::io
-215
View File
@@ -1,215 +0,0 @@
#include "io/zip_reader.h"
#include "foundation/fs_utf8.h"
#include <cstdio>
#include <cstring>
#include "io/inflate.h"
namespace joc::io {
namespace {
constexpr std::uint32_t kLocalHeaderSignature = 0x04034b50u;
constexpr std::uint32_t kCentralHeaderSignature = 0x02014b50u;
constexpr std::uint32_t kEndOfCentralDirectory = 0x06054b50u;
std::uint16_t read_u16(const std::uint8_t* p) {
return static_cast<std::uint16_t>(p[0] | (p[1] << 8));
}
std::uint32_t read_u32(const std::uint8_t* p) {
return static_cast<std::uint32_t>(p[0]) | (static_cast<std::uint32_t>(p[1]) << 8) |
(static_cast<std::uint32_t>(p[2]) << 16) | (static_cast<std::uint32_t>(p[3]) << 24);
}
} // namespace
std::uint32_t crc32_of(const std::uint8_t* data, std::size_t size) {
static std::uint32_t table[256];
static bool ready = false;
if (!ready) {
for (std::uint32_t i = 0; i < 256; ++i) {
std::uint32_t value = i;
for (int bit = 0; bit < 8; ++bit) {
value = (value & 1u) ? (0xEDB88320u ^ (value >> 1)) : (value >> 1);
}
table[i] = value;
}
ready = true;
}
std::uint32_t crc = 0xFFFFFFFFu;
for (std::size_t i = 0; i < size; ++i) {
crc = table[(crc ^ data[i]) & 0xFFu] ^ (crc >> 8);
}
return crc ^ 0xFFFFFFFFu;
}
bool ZipArchive::open(const std::string& path, std::string* error) {
entries_.clear();
data_.clear();
std::FILE* file = fs_utf8::fopen(path, "rb");
if (file == nullptr) {
if (error != nullptr) {
*error = "cannot open " + path;
}
return false;
}
std::fseek(file, 0, SEEK_END);
const long long size = std::ftell(file);
std::fseek(file, 0, SEEK_SET);
if (size <= 0) {
std::fclose(file);
if (error != nullptr) {
*error = "empty file " + path;
}
return false;
}
data_.resize(static_cast<std::size_t>(size));
const std::size_t got = std::fread(data_.data(), 1, data_.size(), file);
std::fclose(file);
if (got != data_.size()) {
if (error != nullptr) {
*error = "short read on " + path;
}
return false;
}
std::size_t eocd = std::string::npos;
const std::size_t scan_start = data_.size() > 65557u ? data_.size() - 65557u : 0u;
for (std::size_t i = data_.size(); i-- > scan_start;) {
if (i + 4u <= data_.size() && read_u32(&data_[i]) == kEndOfCentralDirectory) {
eocd = i;
break;
}
if (i == 0) {
break;
}
}
if (eocd == std::string::npos || eocd + 22u > data_.size()) {
if (error != nullptr) {
*error = "not a zip archive (no end-of-central-directory)";
}
return false;
}
const std::uint16_t entry_count = read_u16(&data_[eocd + 10]);
const std::uint32_t directory_offset = read_u32(&data_[eocd + 16]);
if (directory_offset >= data_.size()) {
if (error != nullptr) {
*error = "central directory offset out of range";
}
return false;
}
std::size_t cursor = directory_offset;
for (std::uint16_t index = 0; index < entry_count; ++index) {
if (cursor + 46u > data_.size() || read_u32(&data_[cursor]) != kCentralHeaderSignature) {
if (error != nullptr) {
*error = "malformed central directory entry " + std::to_string(index);
}
return false;
}
ZipEntry entry;
entry.method = read_u16(&data_[cursor + 10]);
entry.crc32 = read_u32(&data_[cursor + 16]);
entry.compressed_size = read_u32(&data_[cursor + 20]);
entry.uncompressed_size = read_u32(&data_[cursor + 24]);
const std::uint16_t name_length = read_u16(&data_[cursor + 28]);
const std::uint16_t extra_length = read_u16(&data_[cursor + 30]);
const std::uint16_t comment_length = read_u16(&data_[cursor + 32]);
entry.local_header_offset = read_u32(&data_[cursor + 42]);
if (entry.compressed_size == 0xFFFFFFFFu || entry.uncompressed_size == 0xFFFFFFFFu ||
entry.local_header_offset == 0xFFFFFFFFu) {
if (error != nullptr) {
*error = "zip64 archives are not supported";
}
return false;
}
if (cursor + 46u + name_length > data_.size()) {
if (error != nullptr) {
*error = "member name out of range";
}
return false;
}
entry.name.assign(reinterpret_cast<const char*>(&data_[cursor + 46]), name_length);
entries_.push_back(std::move(entry));
cursor += 46u + name_length + extra_length + comment_length;
}
return true;
}
const ZipEntry* ZipArchive::find(const std::string& name) const {
for (const ZipEntry& entry : entries_) {
if (entry.name == name) {
return &entry;
}
}
return nullptr;
}
bool ZipArchive::extract(const ZipEntry& entry, std::vector<std::uint8_t>* out,
std::string* error) const {
if (out == nullptr) {
return false;
}
const std::size_t offset = entry.local_header_offset;
if (offset + 30u > data_.size() || read_u32(&data_[offset]) != kLocalHeaderSignature) {
if (error != nullptr) {
*error = "bad local header for " + entry.name;
}
return false;
}
const std::uint16_t name_length = read_u16(&data_[offset + 26]);
const std::uint16_t extra_length = read_u16(&data_[offset + 28]);
const std::size_t start = offset + 30u + name_length + extra_length;
if (start + entry.compressed_size > data_.size()) {
if (error != nullptr) {
*error = "member data out of range for " + entry.name;
}
return false;
}
if (entry.method == 0u) {
out->assign(data_.begin() + static_cast<std::ptrdiff_t>(start),
data_.begin() + static_cast<std::ptrdiff_t>(start + entry.compressed_size));
} else if (entry.method == 8u) {
if (!inflate_raw(&data_[start], entry.compressed_size, out)) {
if (error != nullptr) {
*error = "deflate error in " + entry.name;
}
return false;
}
} else {
if (error != nullptr) {
*error = "unsupported compression method " + std::to_string(entry.method) + " for " +
entry.name;
}
return false;
}
if (entry.uncompressed_size != 0u && out->size() != entry.uncompressed_size) {
if (error != nullptr) {
*error = "size mismatch for " + entry.name + " (" + std::to_string(out->size()) +
" vs " + std::to_string(entry.uncompressed_size) + ")";
}
return false;
}
if (entry.crc32 != 0u && crc32_of(out->data(), out->size()) != entry.crc32) {
if (error != nullptr) {
*error = "CRC mismatch for " + entry.name;
}
return false;
}
return true;
}
bool ZipArchive::read_member(const std::string& name, std::vector<std::uint8_t>* out,
std::string* error) const {
const ZipEntry* entry = find(name);
if (entry == nullptr) {
if (error != nullptr) {
*error = "member not found: " + name;
}
return false;
}
return extract(*entry, out, error);
}
} // namespace joc::io
-38
View File
@@ -1,38 +0,0 @@
#pragma once
#include <cstdint>
#include <string>
#include <vector>
namespace joc::io {
struct ZipEntry {
std::string name;
std::uint16_t method = 0;
std::uint32_t crc32 = 0;
std::uint32_t compressed_size = 0;
std::uint32_t uncompressed_size = 0;
std::uint32_t local_header_offset = 0;
};
class ZipArchive {
public:
bool open(const std::string& path, std::string* error);
const std::vector<ZipEntry>& entries() const { return entries_; }
const ZipEntry* find(const std::string& name) const;
bool extract(const ZipEntry& entry, std::vector<std::uint8_t>* out, std::string* error) const;
bool read_member(const std::string& name, std::vector<std::uint8_t>* out, std::string* error) const;
private:
std::vector<std::uint8_t> data_;
std::vector<ZipEntry> entries_;
};
std::uint32_t crc32_of(const std::uint8_t* data, std::size_t size);
} // namespace joc::io
-415
View File
@@ -1,415 +0,0 @@
#include "joc_bitstream/joc_parser.h"
#include <algorithm>
#include <cmath>
#include <cstring>
#include <string>
#include "foundation/bit_reader.h"
#include "joc_huffman_tables.h"
namespace joc::joc {
namespace {
struct NumChannelsEntry {
std::uint32_t config;
std::int16_t channels;
};
constexpr NumChannelsEntry kNumChannels[] = {
{0u, 5}, {1u, 7}, {2u, 7}, {3u, 5}, {4u, 7},
};
struct NumBandsEntry {
std::uint32_t index;
std::int16_t bands;
};
constexpr NumBandsEntry kNumBands[] = {
{0u, 1}, {1u, 3}, {2u, 5}, {3u, 7}, {4u, 9}, {5u, 12}, {6u, 15}, {7u, 23},
};
// Floored modulo: Python's % operator semantics, so that the ported
inline std::int64_t floored_mod(std::int64_t value, std::int64_t modulus) {
const std::int64_t remainder = value % modulus;
return remainder < 0 ? remainder + modulus : remainder;
}
enum class SymbolKind { Mtx, Idx, Vec };
struct Tree {
const int (*nodes)[2] = nullptr;
int count = 0;
};
Tree select_tree(std::uint32_t quant_idx, SymbolKind kind, int n_channels) {
Tree tree;
switch (kind) {
case SymbolKind::Idx:
if (n_channels == 5) {
tree.nodes = joc_huff_code_5ch_pos_index_sparse;
tree.count = static_cast<int>(sizeof(joc_huff_code_5ch_pos_index_sparse) /
sizeof(joc_huff_code_5ch_pos_index_sparse[0]));
} else {
tree.nodes = joc_huff_code_7ch_pos_index_sparse;
tree.count = static_cast<int>(sizeof(joc_huff_code_7ch_pos_index_sparse) /
sizeof(joc_huff_code_7ch_pos_index_sparse[0]));
}
break;
case SymbolKind::Vec:
if (quant_idx == 0u) {
tree.nodes = joc_huff_code_coarse_coeff_sparse;
tree.count = static_cast<int>(sizeof(joc_huff_code_coarse_coeff_sparse) /
sizeof(joc_huff_code_coarse_coeff_sparse[0]));
} else {
tree.nodes = joc_huff_code_fine_coeff_sparse;
tree.count = static_cast<int>(sizeof(joc_huff_code_fine_coeff_sparse) /
sizeof(joc_huff_code_fine_coeff_sparse[0]));
}
break;
case SymbolKind::Mtx:
default:
if (quant_idx == 0u) {
tree.nodes = joc_huff_code_coarse_generic;
tree.count = static_cast<int>(sizeof(joc_huff_code_coarse_generic) /
sizeof(joc_huff_code_coarse_generic[0]));
} else {
tree.nodes = joc_huff_code_fine_generic;
tree.count = static_cast<int>(sizeof(joc_huff_code_fine_generic) /
sizeof(joc_huff_code_fine_generic[0]));
}
break;
}
return tree;
}
// infinite loop (plan 40.4 BL-6).
bool huff_decode(const Tree& tree, bits::BitReader& reader, std::int16_t* out_value) {
int node = 0;
int steps = 0;
while (node >= 0) {
if (node >= tree.count || steps > tree.count) {
reader.fail(JOC_ERR_JOC_SYNTAX, "Huffman tree walk left the valid node range");
return false;
}
++steps;
const std::uint32_t bit = reader.read(1);
if (reader.failed()) {
return false;
}
node = tree.nodes[node][bit];
}
*out_value = static_cast<std::int16_t>(-node - 1);
return true;
}
Status syntax_fail(const std::string& message) {
return Status::fail(JOC_ERR_JOC_SYNTAX, stage::kJoc, message);
}
Status truncated_fail(const bits::BitReader& reader) {
if (reader.error() == JOC_ERR_JOC_SYNTAX) {
return syntax_fail(reader.error_message());
}
return Status::fail(JOC_ERR_BITSTREAM_TRUNCATED, stage::kJoc,
std::string("JOC bitstream truncated: ") + reader.error_message());
}
void reconstruct_dense(const ObjectSymbols& symbols, std::uint32_t dp, int n_channels, std::int64_t nquant,
std::int64_t offset,
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS]) {
for (int ch = 0; ch < n_channels; ++ch) {
q[dp][ch][0] =
floored_mod(offset + static_cast<std::int64_t>(symbols.mtx[dp][ch][0]), nquant);
for (int pb = 1; pb < symbols.n_bands; ++pb) {
q[dp][ch][pb] = floored_mod(
q[dp][ch][pb - 1] + static_cast<std::int64_t>(symbols.mtx[dp][ch][pb]), nquant);
}
}
}
// across parameter bands and is deliberately NOT reset when the active channel
Status reconstruct_sparse(
const ObjectSymbols& symbols, std::uint32_t dp, int n_channels, std::int64_t nquant, std::int64_t offset,
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS]) {
if (n_channels != 5 && n_channels != 7) {
return Status::fail(JOC_ERR_JOC_UNSUPPORTED_VARIANT, stage::kJoc,
"sparse JOC requires 5 or 7 core channels, got " +
std::to_string(n_channels));
}
const int initial_channel = symbols.idx[dp][0];
if (initial_channel < 0 || initial_channel >= n_channels) {
return syntax_fail("sparse JOC initial channel " + std::to_string(initial_channel) +
" out of range for " + std::to_string(n_channels) + " channels");
}
// Non-active entries take nquant/2, which dequantizes to exactly zero.
for (int ch = 0; ch < n_channels; ++ch) {
for (int pb = 0; pb < symbols.n_bands; ++pb) {
q[dp][ch][pb] = nquant / 2;
}
}
int active = initial_channel;
std::int64_t coefficient = offset;
for (int pb = 0; pb < symbols.n_bands; ++pb) {
if (pb != 0) {
active = static_cast<int>(
floored_mod(static_cast<std::int64_t>(active) + symbols.idx[dp][pb], n_channels));
}
coefficient = floored_mod(coefficient + static_cast<std::int64_t>(symbols.vec[dp][pb]), nquant);
q[dp][active][pb] = coefficient;
}
return Status::success();
}
} // namespace
std::int16_t num_channels_for_config(std::uint32_t dmx_config_idx) {
for (const NumChannelsEntry& entry : kNumChannels) {
if (entry.config == dmx_config_idx) {
return entry.channels;
}
}
return -1;
}
std::int16_t num_bands_for_index(std::uint32_t num_bands_idx) {
for (const NumBandsEntry& entry : kNumBands) {
if (entry.index == num_bands_idx) {
return entry.bands;
}
}
return -1;
}
Status parse_id14(const std::uint8_t* payload, std::size_t payload_size, joc_frame_params* out,
FrameSymbols* symbols) {
if (payload == nullptr || out == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kJoc, "null payload or output");
}
std::memset(out, 0, sizeof(*out));
out->struct_size = sizeof(joc_frame_params);
out->struct_version = JOC_FRAME_PARAMS_VERSION;
// capture is purely additive and never changes the parse result.
FrameSymbols local_symbols{};
FrameSymbols& capture = (symbols != nullptr) ? *symbols : local_symbols;
std::memset(&capture, 0, sizeof(capture));
bits::BitReader reader(payload, payload_size);
out->dmx_config_idx = static_cast<std::uint8_t>(reader.read(3));
out->num_objects_bits = static_cast<std::uint8_t>(reader.read(6));
out->ext_config_idx = static_cast<std::uint8_t>(reader.read(3));
const std::uint32_t n_objects = static_cast<std::uint32_t>(out->num_objects_bits) + 1u;
const std::int16_t n_channels = num_channels_for_config(out->dmx_config_idx);
if (n_channels < 0) {
return syntax_fail("unknown JOC downmix configuration " +
std::to_string(out->dmx_config_idx));
}
out->n_channels = static_cast<std::uint8_t>(n_channels);
if (n_objects > JOC_MAX_OBJECTS) {
// The reference implementation has no check here and fails later inside
// NumPy; the port reports it explicitly (plan 28.2).
return Status::fail(JOC_ERR_JOC_UNSUPPORTED_VARIANT, stage::kJoc,
"JOC frame declares " + std::to_string(n_objects) +
" objects, the ABI supports at most " +
std::to_string(static_cast<int>(JOC_MAX_OBJECTS)));
}
out->n_objects = static_cast<std::uint8_t>(n_objects);
out->clipgain_x_bits = static_cast<std::uint8_t>(reader.read(3));
out->clipgain_y_bits = static_cast<std::uint8_t>(reader.read(5));
out->seq_count = reader.read(10);
// clipgain = 1 + (y/32) * 2^(x-4). The reference multiplies by an exact
// exactly for the whole legal range.
out->clipgain = 1.0 + static_cast<double>(out->clipgain_y_bits) / 32.0 *
std::ldexp(1.0, static_cast<int>(out->clipgain_x_bits) - 4);
for (std::uint32_t obj = 0; obj < n_objects; ++obj) {
joc_object_params& info = out->objects[obj];
info.present = static_cast<std::uint8_t>(reader.read(1));
if (info.present == 0u) {
continue;
}
info.num_bands_idx = static_cast<std::uint8_t>(reader.read(3));
const std::int16_t bands = num_bands_for_index(info.num_bands_idx);
if (bands < 0) {
return syntax_fail("unknown JOC num_bands index " +
std::to_string(info.num_bands_idx));
}
info.n_bands = static_cast<std::uint8_t>(bands);
info.sparse = static_cast<std::uint8_t>(reader.read(1));
info.quant_idx = static_cast<std::uint8_t>(reader.read(1));
info.slope_idx = static_cast<std::uint8_t>(reader.read(1));
info.num_dpoints_bits = static_cast<std::uint8_t>(reader.read(1));
info.n_dpoints = static_cast<std::uint8_t>(info.num_dpoints_bits + 1u);
if (info.slope_idx == 1u) {
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
info.offset_ts[dp] = static_cast<std::uint8_t>(reader.read(5) + 1u);
}
}
}
if (reader.failed()) {
return truncated_fail(reader);
}
for (std::uint32_t obj = 0; obj < n_objects; ++obj) {
const joc_object_params& info = out->objects[obj];
if (info.present == 0u) {
continue;
}
ObjectSymbols* symbol = &capture.objects[obj];
symbol->present = 1;
symbol->sparse = info.sparse;
symbol->n_bands = info.n_bands;
symbol->n_dpoints = info.n_dpoints;
symbol->n_channels = out->n_channels;
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
if (info.sparse == 1u) {
const Tree idx_tree = select_tree(info.quant_idx, SymbolKind::Idx, n_channels);
const std::uint32_t first = reader.read(3);
if (reader.failed()) {
return truncated_fail(reader);
}
symbol->idx[dp][0] = static_cast<std::uint8_t>(first);
for (int pb = 1; pb < info.n_bands; ++pb) {
std::int16_t value = 0;
if (!huff_decode(idx_tree, reader, &value)) {
return truncated_fail(reader);
}
symbol->idx[dp][pb] = static_cast<std::uint8_t>(value);
}
const Tree vec_tree = select_tree(info.quant_idx, SymbolKind::Vec, n_channels);
for (int pb = 0; pb < info.n_bands; ++pb) {
std::int16_t value = 0;
if (!huff_decode(vec_tree, reader, &value)) {
return truncated_fail(reader);
}
symbol->vec[dp][pb] = value;
}
} else {
const Tree mtx_tree = select_tree(info.quant_idx, SymbolKind::Mtx, n_channels);
for (int ch = 0; ch < n_channels; ++ch) {
for (int pb = 0; pb < info.n_bands; ++pb) {
std::int16_t value = 0;
if (!huff_decode(mtx_tree, reader, &value)) {
return truncated_fail(reader);
}
symbol->mtx[dp][ch][pb] = value;
}
}
}
}
}
out->data_end_bits = static_cast<std::uint32_t>(reader.position());
out->trailing_bits = static_cast<std::uint32_t>(payload_size * 8u - reader.position());
const std::size_t tail_offset = reader.position() / 8u;
if (tail_offset < payload_size) {
const std::size_t tail_bytes = std::min<std::size_t>(8u, payload_size - tail_offset);
std::memcpy(out->tail_bytes, payload + tail_offset, tail_bytes);
}
for (std::uint32_t obj = 0; obj < n_objects; ++obj) {
const joc_object_params& info = out->objects[obj];
if (info.present == 0u) {
continue;
}
std::uint32_t mask_bit = 1u << obj;
out->present_mask |= mask_bit;
const std::int64_t nquant = (info.quant_idx == 0u) ? 96 : 192;
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS] = {};
ObjectSymbols* symbol = &capture.objects[obj];
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
if (info.sparse == 1u) {
const std::int64_t offset = (info.quant_idx == 0u) ? 50 : 100;
const Status status =
reconstruct_sparse(*symbol, dp, n_channels, nquant, offset, q);
if (!status.ok()) {
return status;
}
} else {
const std::int64_t offset = (info.quant_idx == 0u) ? 48 : 96;
reconstruct_dense(*symbol, dp, n_channels, nquant, offset, q);
}
}
// Operand order and types are kept identical to the reference so the
// result is bit-exact, not merely close.
const double nquant_half = static_cast<double>(nquant) / 2.0;
const double denominator = 4096.0 * static_cast<double>(1 + static_cast<int>(info.quant_idx));
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
for (int ch = 0; ch < n_channels; ++ch) {
for (int pb = 0; pb < info.n_bands; ++pb) {
const double value = static_cast<double>(q[dp][ch][pb]) - nquant_half;
out->objects[obj].dq[dp][ch][pb] = value * 820.0 / denominator;
}
}
}
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
for (int ch = 0; ch < n_channels; ++ch) {
for (int pb = 0; pb < info.n_bands; ++pb) {
symbol->q[dp][ch][pb] = q[dp][ch][pb];
}
}
}
}
capture.n_objects = out->n_objects;
capture.n_channels = out->n_channels;
return Status::success();
}
Status parse_eac3_frame(const std::uint8_t* frame, std::size_t frame_size, joc_frame_params* out,
emdf::Container* container, FrameSymbols* symbols) {
if (frame == nullptr || out == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kJoc, "null frame or output");
}
emdf::Container local;
const Status status = emdf::find_joc_emdf(frame, frame_size, &local);
if (!status.ok()) {
return status;
}
if (container != nullptr) {
*container = local;
}
const emdf::Payload* payload = local.find(emdf::kIdJoc);
if (payload == nullptr) {
return Status::fail(JOC_ERR_EMDF_TRANSPORT, stage::kEmdf,
"EMDF container has no ID14 (JOC) payload");
}
std::vector<std::uint8_t> bytes;
const Status extract = emdf::extract_payload_bytes(frame, frame_size, *payload, &bytes);
if (!extract.ok()) {
return extract;
}
return parse_id14(bytes.data(), bytes.size(), out, symbols);
}
Status check_id14_padding(const std::uint8_t* payload, std::size_t payload_size,
std::uint32_t* out_trailing_bits) {
joc_frame_params params;
FrameSymbols symbols;
const Status status = parse_id14(payload, payload_size, &params, &symbols);
if (!status.ok()) {
return status;
}
if (out_trailing_bits != nullptr) {
*out_trailing_bits = params.trailing_bits;
}
if (params.trailing_bits > 7u) {
return Status::fail(JOC_ERR_BITSTREAM_PADDING, stage::kJoc,
"more than 7 bits left after joc_data (" +
std::to_string(params.trailing_bits) + ")");
}
for (std::size_t bit = params.data_end_bits; bit < payload_size * 8u; ++bit) {
if (((payload[bit >> 3] >> (7u - (bit & 7u))) & 1u) != 0u) {
return Status::fail(JOC_ERR_BITSTREAM_PADDING, stage::kJoc,
"non-zero trailing padding bit at " + std::to_string(bit));
}
}
return Status::success();
}
} // namespace joc::joc
-46
View File
@@ -1,46 +0,0 @@
// Port of src/joc_decode.py.
#pragma once
#include <cstddef>
#include <cstdint>
#include <vector>
#include "joc_core.h"
#include "emdf/emdf_parser.h"
#include "foundation/status.h"
namespace joc::joc {
struct ObjectSymbols {
std::uint8_t present = 0;
std::uint8_t sparse = 0;
std::uint8_t n_bands = 0;
std::uint8_t n_dpoints = 0;
std::uint8_t n_channels = 0;
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS] = {};
std::int16_t mtx[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS] = {};
std::uint8_t idx[JOC_MAX_DPOINTS][JOC_MAX_PARAMETER_BANDS] = {};
std::int16_t vec[JOC_MAX_DPOINTS][JOC_MAX_PARAMETER_BANDS] = {};
};
struct FrameSymbols {
std::uint8_t n_objects = 0;
std::uint8_t n_channels = 0;
ObjectSymbols objects[JOC_MAX_OBJECTS];
};
Status parse_id14(const std::uint8_t* payload, std::size_t payload_size, joc_frame_params* out,
FrameSymbols* symbols);
Status parse_eac3_frame(const std::uint8_t* frame, std::size_t frame_size, joc_frame_params* out,
emdf::Container* container, FrameSymbols* symbols);
Status check_id14_padding(const std::uint8_t* payload, std::size_t payload_size,
std::uint32_t* out_trailing_bits);
std::int16_t num_channels_for_config(std::uint32_t dmx_config_idx);
std::int16_t num_bands_for_index(std::uint32_t num_bands_idx);
} // namespace joc::joc
-80
View File
@@ -1,80 +0,0 @@
#include "joc_core/objects16.h"
#include <cstdint>
#include <cstdio>
namespace joc::joc {
Status rebuild_objects16(ejoc_renderer_handle handle, const joc_frame_params& params,
const float* bed5_planar, const float* lfe, float gain,
std::vector<float>* out16, std::string* error) {
if (handle == nullptr || bed5_planar == nullptr || out16 == nullptr) {
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kDsp, "null argument");
}
if (params.n_channels != JOC_CORE_CHANNELS) {
if (error != nullptr) {
*error = "the reused JOC kernel requires 5 core channels, frame declares " +
std::to_string(static_cast<unsigned>(params.n_channels));
}
return Status::fail(JOC_ERR_JOC_UNSUPPORTED_VARIANT, stage::kDsp, *error);
}
std::vector<double> dq(static_cast<std::size_t>(JOC_MAX_OBJECTS) * JOC_MAX_DPOINTS *
JOC_CORE_CHANNELS * JOC_MAX_PARAMETER_BANDS,
0.0);
std::uint8_t n_bands[JOC_MAX_OBJECTS] = {};
std::uint8_t n_dpoints[JOC_MAX_OBJECTS] = {};
std::uint8_t slope_idx[JOC_MAX_OBJECTS] = {};
std::uint8_t offset_ts[JOC_MAX_OBJECTS * JOC_MAX_DPOINTS] = {};
for (unsigned obj = 0; obj < JOC_MAX_OBJECTS; ++obj) {
const joc_object_params& object = params.objects[obj];
if (object.present == 0) {
continue;
}
n_bands[obj] = object.n_bands;
n_dpoints[obj] = object.n_dpoints;
slope_idx[obj] = object.slope_idx;
for (unsigned dp = 0; dp < JOC_MAX_DPOINTS; ++dp) {
offset_ts[obj * JOC_MAX_DPOINTS + dp] = object.offset_ts[dp];
}
for (unsigned dp = 0; dp < object.n_dpoints; ++dp) {
for (unsigned ch = 0; ch < JOC_CORE_CHANNELS; ++ch) {
for (unsigned pb = 0; pb < object.n_bands; ++pb) {
const std::size_t index =
((static_cast<std::size_t>(obj) * JOC_MAX_DPOINTS + dp) * JOC_CORE_CHANNELS +
ch) *
JOC_MAX_PARAMETER_BANDS +
pb;
dq[index] = object.dq[dp][ch][pb];
}
}
}
}
out16->assign(static_cast<std::size_t>(JOC_OUTPUT_CHANNELS) * JOC_FRAME_SAMPLES, 0.0f);
// band-0 的 21-tap DC 补偿只在 downmix 配置 3/4 下启用,其余配置 band 0
// 走与其他 band 相同的处理。
const bool dc_filter = params.dmx_config_idx == 3 || params.dmx_config_idx == 4;
if (ejoc_renderer_set_dc_filter(handle, dc_filter ? 1u : 0u) != 0) {
if (error != nullptr) {
*error = "ejoc_renderer_set_dc_filter failed";
}
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kDsp, "ejoc_renderer_set_dc_filter failed");
}
const int result = ejoc_renderer_process(
handle, bed5_planar, lfe, params.present_mask, n_bands, n_dpoints, slope_idx, offset_ts,
dq.data(), params.clipgain, 0.0625f, gain, out16->data());
if (result != 0) {
const char* detail = ejoc_renderer_last_error(handle);
const std::string message =
"ejoc_renderer_process failed (" + std::to_string(result) + "): " +
(detail != nullptr ? detail : "unknown");
if (error != nullptr) {
*error = message;
}
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kDsp, message);
}
return Status::success();
}
} // namespace joc::joc

Some files were not shown because too many files have changed in this diff Show More