Compare commits
20 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 428772eb87 | |||
| 6bc2c28856 | |||
| 704ea0897b | |||
| c6859e6b02 | |||
| afef6d2c52 | |||
| ab7e815a4d | |||
| b634326b5d | |||
| cf50beafcb | |||
| c8676d7aa3 | |||
| b619dee523 | |||
| 7da3a5eb95 | |||
| 128286153e | |||
| 85105f21d4 | |||
| 329445ed25 | |||
| 887b3e317f | |||
| 35536b0bd4 | |||
| 9b41fedb14 | |||
| 65a08e544d | |||
| 7031b43284 | |||
| aa3a519f79 |
@@ -1,38 +0,0 @@
|
||||
name: ci
|
||||
|
||||
on:
|
||||
push:
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: ${{ matrix.os }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest, windows-latest]
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -S . -B build -DCMAKE_BUILD_TYPE=Release
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build --config Release --parallel
|
||||
|
||||
- name: Test
|
||||
run: ctest --test-dir build --output-on-failure -C Release
|
||||
|
||||
# Publish the install tree: the shared library and the CLI.
|
||||
- name: Stage
|
||||
run: cmake --install build --config Release --prefix stage
|
||||
|
||||
- name: Upload
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: joc-core-${{ runner.os }}
|
||||
path: |
|
||||
stage/bin/*
|
||||
stage/lib/*
|
||||
if-no-files-found: error
|
||||
@@ -0,0 +1,99 @@
|
||||
name: Native builds
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
tags:
|
||||
- "v*"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: ${{ matrix.asset }}
|
||||
runs-on: ${{ matrix.runner }}
|
||||
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- runner: windows-2022
|
||||
asset: windows-x64
|
||||
cmake_args: -A x64
|
||||
|
||||
- runner: ubuntu-22.04
|
||||
asset: linux-x64
|
||||
cmake_args: ""
|
||||
|
||||
- runner: macos-15-intel
|
||||
asset: macos-x64
|
||||
cmake_args: -DCMAKE_OSX_DEPLOYMENT_TARGET=12.0
|
||||
|
||||
- runner: macos-15
|
||||
asset: macos-arm64
|
||||
cmake_args: -DCMAKE_OSX_DEPLOYMENT_TARGET=12.0
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Configure
|
||||
run: >
|
||||
cmake
|
||||
-S native
|
||||
-B build/native
|
||||
-DCMAKE_BUILD_TYPE=Release
|
||||
-DCMAKE_INSTALL_PREFIX="${{ github.workspace }}/stage"
|
||||
${{ matrix.cmake_args }}
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build/native --config Release --parallel
|
||||
|
||||
- name: Install
|
||||
run: cmake --install build/native --config Release
|
||||
|
||||
- name: Package
|
||||
working-directory: stage
|
||||
run: >
|
||||
cmake -E tar
|
||||
cf "../JustOneCacophony-native-${{ matrix.asset }}.zip"
|
||||
--format=zip
|
||||
-- .
|
||||
|
||||
- name: Upload workflow artifact
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: JustOneCacophony-native-${{ matrix.asset }}
|
||||
path: JustOneCacophony-native-${{ matrix.asset }}.zip
|
||||
if-no-files-found: error
|
||||
retention-days: 14
|
||||
|
||||
release:
|
||||
name: Publish GitHub Release
|
||||
if: startsWith(github.ref, 'refs/tags/v')
|
||||
needs: build
|
||||
runs-on: ubuntu-24.04
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
steps:
|
||||
- name: Download native packages
|
||||
uses: actions/download-artifact@v5
|
||||
with:
|
||||
pattern: JustOneCacophony-native-*
|
||||
path: dist
|
||||
merge-multiple: true
|
||||
|
||||
- name: Create release
|
||||
run: >
|
||||
gh release create "$GITHUB_REF_NAME"
|
||||
dist/*.zip
|
||||
--verify-tag
|
||||
--generate-notes
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
+22
-11
@@ -1,12 +1,23 @@
|
||||
/build/
|
||||
/output/
|
||||
/out/
|
||||
/traces/
|
||||
/vectors/
|
||||
/testdata/
|
||||
/devtools/
|
||||
.vs/
|
||||
.vscode/
|
||||
CMakeUserPresets.json
|
||||
__pycache__/
|
||||
*.spool.f32
|
||||
*.py[cod]
|
||||
.pytest_cache/
|
||||
|
||||
.venv/
|
||||
venv/
|
||||
|
||||
build/
|
||||
output/
|
||||
tests/
|
||||
lib/
|
||||
metadata_cache/
|
||||
|
||||
*.metadata.json
|
||||
*.report.json
|
||||
*.variant-error.json
|
||||
*.objects16.f32le
|
||||
|
||||
HRTF/
|
||||
|
||||
*.sofa
|
||||
*.personalized_headphone
|
||||
*.jochrtf
|
||||
|
||||
-330
@@ -1,330 +0,0 @@
|
||||
cmake_minimum_required(VERSION 3.20)
|
||||
|
||||
project(joc_core VERSION 0.1.0 LANGUAGES C CXX)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Products
|
||||
# joc_core shared library: the execution core (E-AC-3/EMDF/JOC/OAMD
|
||||
# bitstream, DSP, speaker and binaural rendering, ADM-BWF/WAV
|
||||
# writing, file task and the streaming push/pull surface)
|
||||
# joc_cli command line frontend for file tasks
|
||||
#
|
||||
# Public headers: include/joc_core.h engine, telemetry and file task
|
||||
# include/joc_stream.h embedder-facing streaming surface
|
||||
#
|
||||
# Options
|
||||
# JOC_BUILD_TESTS unit tests (CTest), on by default
|
||||
# JOC_BUILD_DEVTOOLS in-tree verification tools, off: those sources live in
|
||||
# devtools/, which is not part of the repository
|
||||
# JOC_ENABLE_AVX2 build the runtime-dispatched AVX2 kernels, on by default
|
||||
# JOC_ENABLE_AVX512 build the runtime-dispatched AVX-512 kernels, on by default
|
||||
#
|
||||
# Only the MSVC toolchain is validated locally; other platforms are built by CI.
|
||||
# The floating-point flags of the original native library are preserved on
|
||||
# purpose: /fp:precise on MSVC, -fno-fast-math elsewhere, and a static CRT so
|
||||
# that no redistributable is required.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
option(JOC_BUILD_TESTS "Build the unit tests" ON)
|
||||
option(JOC_BUILD_DEVTOOLS "Build the in-tree verification tools (needs devtools/)" OFF)
|
||||
# The DSP kernels are dispatched at run time (see src/simd/simd.h): the
|
||||
# baseline units stay on the ISA every x86-64 CPU has, and these two options
|
||||
# decide whether the wider units are linked in at all. Both are on by default.
|
||||
# The instruction set actually executed is chosen from CPUID/XGETBV (or the
|
||||
# AArch64 baseline) when the library is first used, so a binary carrying the
|
||||
# AVX-512 unit still runs on a CPU without it, and JOC_SIMD=scalar|sse2|avx2|
|
||||
# avx512|neon pins one tier for verification.
|
||||
option(JOC_ENABLE_AVX2 "Build the runtime-dispatched AVX2 kernels" ON)
|
||||
option(JOC_ENABLE_AVX512 "Build the runtime-dispatched AVX-512 kernels" ON)
|
||||
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_CXX_STANDARD_REQUIRED ON)
|
||||
set(CMAKE_CXX_EXTENSIONS OFF)
|
||||
|
||||
if(NOT CMAKE_BUILD_TYPE AND NOT CMAKE_CONFIGURATION_TYPES)
|
||||
set(CMAKE_BUILD_TYPE Release CACHE STRING "Build type" FORCE)
|
||||
endif()
|
||||
|
||||
find_package(Threads REQUIRED)
|
||||
|
||||
# Ninja learns MSVC's header dependencies by parsing the compiler's /showIncludes
|
||||
# notes, and it only recognises the prefix it is told about. A localized MSVC
|
||||
# prints a translated prefix, and when CMake cannot detect it the notes are
|
||||
# silently dropped: the build then reuses stale objects after a header changes and
|
||||
# produces a binary that does not match its sources -- wrong, not just slow. This
|
||||
# only warns, because supplying the value is not always possible either: it has to
|
||||
# survive the cache code page to be usable, and a build driver can compensate more
|
||||
# reliably by dropping objects when a header is newer than they are.
|
||||
if(MSVC AND CMAKE_GENERATOR MATCHES "Ninja" AND NOT CMAKE_CL_SHOWINCLUDES_PREFIX)
|
||||
message(WARNING
|
||||
"No /showIncludes prefix was detected, so Ninja will not track header "
|
||||
"dependencies and a header change will not rebuild what includes it. "
|
||||
"Configure with -DCMAKE_CL_SHOWINCLUDES_PREFIX=<the text MSVC prints in "
|
||||
"front of each included file>, or make sure the compiler emits its "
|
||||
"messages in the language CMake probes for.")
|
||||
endif()
|
||||
|
||||
# Verbatim copies of the upstream native library; see THIRD_PARTY_NOTICES.md.
|
||||
# These files are never edited: they are the validated DSP kernels.
|
||||
set(JOC_REUSED_SOURCES
|
||||
src/joc_core/eac3joc_core.cpp
|
||||
src/speaker/speaker_renderer.cpp
|
||||
src/binaural/binaural_renderer.cpp
|
||||
)
|
||||
|
||||
set(JOC_INTERNAL_SOURCES
|
||||
src/adm/adm_metadata.cpp
|
||||
src/adm/adm_tracks.cpp
|
||||
src/binaural/binaural_runtime.cpp
|
||||
src/binaural/sofa_binaural_renderer.cpp
|
||||
src/eac3_transport/eac3_reader.cpp
|
||||
src/emdf/emdf_parser.cpp
|
||||
src/foundation/bit_reader.cpp
|
||||
src/foundation/fft.cpp
|
||||
src/foundation/fs_utf8.cpp
|
||||
src/foundation/mini_json.cpp
|
||||
src/foundation/sha256.cpp
|
||||
src/hrtf/jochrtf.cpp
|
||||
src/hrtf/public_filterbank.cpp
|
||||
src/hrtf/rosella_model.cpp
|
||||
src/hrtf/rosella_renderer.cpp
|
||||
src/hrtf/kernel_tables.cpp
|
||||
src/hrtf/sofa.cpp
|
||||
src/hrtf/sofa_cache.cpp
|
||||
src/hrtf/sofa_field.cpp
|
||||
src/io/adm_writer.cpp
|
||||
src/io/hdf5.cpp
|
||||
src/io/inflate.cpp
|
||||
src/io/npy.cpp
|
||||
src/io/npy_writer.cpp
|
||||
src/io/process.cpp
|
||||
src/io/wav_writer.cpp
|
||||
src/io/zip_reader.cpp
|
||||
src/joc_bitstream/joc_parser.cpp
|
||||
src/joc_core/objects16.cpp
|
||||
src/oamd/oamd_parser.cpp
|
||||
src/speaker/speaker_layout_lookup.cpp
|
||||
src/speaker/speaker_step.cpp
|
||||
src/stream/stream.cpp
|
||||
src/task/task.cpp
|
||||
src/telemetry/event_bus.cpp
|
||||
src/timeline/position_timeline.cpp
|
||||
)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Runtime-dispatched SIMD kernels (src/simd/simd.h explains the contract).
|
||||
#
|
||||
# One flat directory, the ISA in the file name (`kernels_intrin_<isa>.cpp`),
|
||||
# never in a subdirectory. MSVC has no function-level ISA attribute, so every
|
||||
# ISA lives in its own translation unit compiled with its own flag, and
|
||||
# dispatch.cpp -- built for the baseline ISA -- picks one when the library is
|
||||
# first used. Only a `kernels_intrin_*.cpp` unit ever gets a wider flag, so a
|
||||
# baseline unit cannot inherit one by accident.
|
||||
#
|
||||
# The x86-64 baseline is SSE2 and there is no SSE2 unit on purpose: a 128-bit
|
||||
# SSE2 register is the register a scalar double already occupies, so SSE2 cannot
|
||||
# widen double-precision arithmetic and hand-written SSE2 would only add moves.
|
||||
# AArch64 needs no probe either; ASIMD is architectural, and the NEON unit is
|
||||
# how a vector path gets selected there.
|
||||
# ---------------------------------------------------------------------------
|
||||
set(JOC_SIMD_SOURCES
|
||||
src/simd/cpu_probe.cpp
|
||||
src/simd/dispatch.cpp
|
||||
src/simd/kernels_scalar.cpp
|
||||
)
|
||||
set(JOC_SIMD_DEFINES "")
|
||||
|
||||
set(JOC_ARCH "")
|
||||
if(CMAKE_SIZEOF_VOID_P EQUAL 8)
|
||||
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(AMD64|amd64|x86_64|X86_64|EM64T)$")
|
||||
set(JOC_ARCH x86_64)
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(ARM64|arm64|aarch64|AARCH64)$")
|
||||
set(JOC_ARCH aarch64)
|
||||
endif()
|
||||
endif()
|
||||
if(NOT JOC_ARCH AND DEFINED CMAKE_CXX_COMPILER_ARCHITECTURE_ID)
|
||||
if(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "x64")
|
||||
set(JOC_ARCH x86_64)
|
||||
elseif(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "ARM64")
|
||||
set(JOC_ARCH aarch64)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(JOC_ARCH STREQUAL "x86_64")
|
||||
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_SSE2=1)
|
||||
if(JOC_ENABLE_AVX2)
|
||||
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_avx2.cpp)
|
||||
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_AVX2=1)
|
||||
if(MSVC)
|
||||
set_source_files_properties(src/simd/kernels_intrin_avx2.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "/arch:AVX2")
|
||||
else()
|
||||
set_source_files_properties(src/simd/kernels_intrin_avx2.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "-mavx2")
|
||||
endif()
|
||||
endif()
|
||||
if(JOC_ENABLE_AVX512)
|
||||
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_avx512.cpp)
|
||||
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_AVX512=1)
|
||||
if(MSVC)
|
||||
set_source_files_properties(src/simd/kernels_intrin_avx512.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "/arch:AVX512")
|
||||
else()
|
||||
set_source_files_properties(src/simd/kernels_intrin_avx512.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "-mavx512f")
|
||||
endif()
|
||||
endif()
|
||||
elseif(JOC_ARCH STREQUAL "aarch64")
|
||||
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_NEON=1)
|
||||
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_neon.cpp)
|
||||
endif()
|
||||
|
||||
add_library(joc_simd OBJECT ${JOC_SIMD_SOURCES})
|
||||
target_include_directories(joc_simd PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
target_compile_definitions(joc_simd PRIVATE ${JOC_SIMD_DEFINES})
|
||||
# The objects are linked into the shared library as well, so they must be
|
||||
# position independent even though an object library does not inherit that.
|
||||
set_target_properties(joc_simd PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
||||
|
||||
# Static form of the engine, used by the in-tree tools and tests so they can use
|
||||
# internal modules without exporting them from the shared library.
|
||||
if(JOC_BUILD_TESTS OR JOC_BUILD_DEVTOOLS)
|
||||
add_library(joc_core_impl STATIC ${JOC_INTERNAL_SOURCES} ${JOC_REUSED_SOURCES}
|
||||
$<TARGET_OBJECTS:joc_simd>)
|
||||
target_include_directories(joc_core_impl
|
||||
PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/include"
|
||||
PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src"
|
||||
)
|
||||
target_link_libraries(joc_core_impl PUBLIC Threads::Threads)
|
||||
endif()
|
||||
|
||||
add_library(joc_core SHARED
|
||||
src/api/joc_api.cpp
|
||||
src/api/joc_stream_api.cpp
|
||||
src/api/joc_task_api.cpp
|
||||
${JOC_INTERNAL_SOURCES}
|
||||
${JOC_REUSED_SOURCES}
|
||||
$<TARGET_OBJECTS:joc_simd>
|
||||
)
|
||||
target_include_directories(joc_core
|
||||
PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/include"
|
||||
PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src"
|
||||
)
|
||||
target_compile_definitions(joc_core PRIVATE JOC_BUILD_DLL)
|
||||
target_link_libraries(joc_core PRIVATE Threads::Threads)
|
||||
set_target_properties(joc_core PROPERTIES
|
||||
OUTPUT_NAME "joc_core"
|
||||
CXX_VISIBILITY_PRESET hidden
|
||||
VISIBILITY_INLINES_HIDDEN YES
|
||||
POSITION_INDEPENDENT_CODE YES
|
||||
)
|
||||
|
||||
# The frontend compiles the UTF-8 path shim itself: it is small, and the shared
|
||||
# library keeps its internals unexported.
|
||||
add_executable(joc_cli src/cli/joc_cli.cpp src/foundation/fs_utf8.cpp)
|
||||
target_link_libraries(joc_cli PRIVATE joc_core)
|
||||
target_include_directories(joc_cli PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
|
||||
# The installed CLI is in bin/ and the shared library in lib/, and ELF/Mach-O
|
||||
# strip the build rpath on install, so the relative lookup has to be recorded
|
||||
# here or `bin/joc_cli` cannot find `../lib/libjoc_core.*`.
|
||||
if(UNIX)
|
||||
if(APPLE)
|
||||
set_target_properties(joc_cli PROPERTIES INSTALL_RPATH "@loader_path/../lib")
|
||||
else()
|
||||
set_target_properties(joc_cli PROPERTIES INSTALL_RPATH "$ORIGIN/../lib")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(JOC_BUILD_TESTS)
|
||||
enable_testing()
|
||||
add_executable(joc_tests tests/test_core.cpp)
|
||||
target_link_libraries(joc_tests PRIVATE joc_core_impl)
|
||||
target_include_directories(joc_tests PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
add_test(NAME core COMMAND joc_tests)
|
||||
# The public headers must stay valid C and C++: these targets exist to prove it.
|
||||
add_library(joc_headers_c OBJECT tests/test_headers.c)
|
||||
target_include_directories(joc_headers_c PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/include")
|
||||
add_library(joc_headers_cpp OBJECT tests/test_headers.cpp)
|
||||
target_link_libraries(joc_headers_cpp PRIVATE joc_core)
|
||||
endif()
|
||||
|
||||
if(JOC_BUILD_DEVTOOLS)
|
||||
add_executable(joc_dump devtools/joc_dump/main.cpp)
|
||||
target_link_libraries(joc_dump PRIVATE joc_core joc_core_impl)
|
||||
target_include_directories(joc_dump PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
# SOFA reader probe: the C++ side of devtools/checks/check_sofa.py.
|
||||
add_executable(joc_sofa_field_probe devtools/sofa_field_probe/main.cpp)
|
||||
add_executable(joc_dictionary_probe devtools/sofa_field_probe/dictionary_main.cpp)
|
||||
target_link_libraries(joc_dictionary_probe PRIVATE joc_core_impl)
|
||||
target_include_directories(joc_dictionary_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
target_link_libraries(joc_sofa_field_probe PRIVATE joc_core_impl)
|
||||
target_include_directories(joc_sofa_field_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
# Rosella model probe: the C++ side of devtools/checks/check_rosella.py.
|
||||
add_executable(joc_rosella_probe devtools/rosella_probe/main.cpp)
|
||||
target_link_libraries(joc_rosella_probe PRIVATE joc_core_impl)
|
||||
target_include_directories(joc_rosella_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
add_executable(joc_sofa_probe devtools/sofa_probe/main.cpp)
|
||||
target_link_libraries(joc_sofa_probe PRIVATE joc_core_impl)
|
||||
target_include_directories(joc_sofa_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
# Probe for the streaming surface: the acceptance harness and the usage
|
||||
# example for joc_stream.h. Not shipped: a player integrates the library.
|
||||
add_executable(joc_stream_probe devtools/joc_stream/main.cpp)
|
||||
target_link_libraries(joc_stream_probe PRIVATE joc_core)
|
||||
endif()
|
||||
|
||||
set(JOC_MSVC_TARGETS joc_core joc_cli joc_simd)
|
||||
if(JOC_BUILD_TESTS OR JOC_BUILD_DEVTOOLS)
|
||||
list(APPEND JOC_MSVC_TARGETS joc_core_impl)
|
||||
endif()
|
||||
if(JOC_BUILD_TESTS)
|
||||
list(APPEND JOC_MSVC_TARGETS joc_tests joc_headers_c joc_headers_cpp)
|
||||
endif()
|
||||
if(JOC_BUILD_DEVTOOLS)
|
||||
list(APPEND JOC_MSVC_TARGETS joc_rosella_probe joc_dump joc_stream_probe joc_sofa_probe joc_sofa_field_probe joc_dictionary_probe)
|
||||
endif()
|
||||
|
||||
if(MSVC)
|
||||
set(JOC_MSVC_FLAGS /W4 /permissive- /Zc:__cplusplus /utf-8 /EHsc /GR- /fp:precise)
|
||||
# The writers use std::fopen for seekable, byte-exact output.
|
||||
set(JOC_MSVC_DEFINES _CRT_SECURE_NO_WARNINGS)
|
||||
# C4324 comes from the reused kernel's std::barrier member at /W4; it is
|
||||
# pre-existing behaviour of a verbatim file, so the warning is silenced
|
||||
# rather than the file edited.
|
||||
set_source_files_properties(src/joc_core/eac3joc_core.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "/wd4324")
|
||||
foreach(target IN LISTS JOC_MSVC_TARGETS)
|
||||
set_property(TARGET ${target} PROPERTY
|
||||
MSVC_RUNTIME_LIBRARY "MultiThreaded$<$<CONFIG:Debug>:Debug>")
|
||||
# No /arch here on purpose: the baseline units keep the architecture's
|
||||
# guaranteed ISA and only the src/simd `kernels_intrin_*` units carry a
|
||||
# wider one (their flags are set per source file above).
|
||||
target_compile_options(${target} PRIVATE ${JOC_MSVC_FLAGS}
|
||||
$<$<CONFIG:Release>:/O2>
|
||||
$<$<CONFIG:Release>:/Oi>)
|
||||
target_compile_definitions(${target} PRIVATE ${JOC_MSVC_DEFINES})
|
||||
endforeach()
|
||||
target_link_options(joc_core PRIVATE /INCREMENTAL:NO /OPT:REF /OPT:ICF)
|
||||
target_link_options(joc_cli PRIVATE /INCREMENTAL:NO)
|
||||
else()
|
||||
foreach(target IN LISTS JOC_MSVC_TARGETS)
|
||||
target_compile_options(${target} PRIVATE -Wall -Wextra -Wpedantic -fno-fast-math
|
||||
$<$<CONFIG:Release>:-O3>)
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
# Headers listed for IDE visibility.
|
||||
target_sources(joc_core PRIVATE
|
||||
include/joc_core.h
|
||||
include/joc_stream.h
|
||||
include/eac3joc_core.h
|
||||
src/joc_bitstream/joc_huffman_tables.h
|
||||
src/joc_core/qmf_tables.h
|
||||
src/speaker/speaker_layouts.h
|
||||
)
|
||||
|
||||
install(TARGETS joc_core joc_cli
|
||||
RUNTIME DESTINATION bin
|
||||
LIBRARY DESTINATION lib
|
||||
ARCHIVE DESTINATION lib
|
||||
)
|
||||
+288
-144
@@ -1,159 +1,303 @@
|
||||
# JustOneCacophony — C++ Core
|
||||
# JustOneCacophony — JOC
|
||||
|
||||
[中文](README.md) · [Mathematics](docs/math.en.md) · [Binaural rendering](docs/binaural.en.md) · [SIMD dispatch](docs/simd.en.md)
|
||||
[中文版](README.md)
|
||||
|
||||
The C++ implementation of JustOneCacophony: an execution core for E-AC-3 JOC
|
||||
bitstream parsing, object reconstruction and rendering. It extracts EMDF, ID14 JOC
|
||||
parameters and ID11 OAMD metadata from E-AC-3 syncframes, combines them with the
|
||||
core 5.1 PCM decoded by FFmpeg to rebuild the LFE and 15 object signals, and writes
|
||||
ADM BWF, a WAV for a chosen speaker layout, or a binaural WAV using a compiled HRTF
|
||||
directional field.
|
||||
> JustOneCacophony is an experimental/test implementation of E-AC-3 JOC for studying JOC parsing, reconstruction, rendering, and the associated mathematics.
|
||||
|
||||
This is research code, not a complete, standard-conformant or production JOC
|
||||
decoder. It covers the bitstream forms it implements and reports an explicit error
|
||||
on unknown variants instead of pretending everything is in harmony.
|
||||
The project can extract and parse EMDF, ID14 JOC parameters, and ID11 OAMD metadata from common E-AC-3 JOC streams. It combines those data with the core 5.1 PCM decoded by FFmpeg, reconstructs LFE plus 15 object channels, and writes ADM BWF, a WAV file for a selected speaker layout, or direct binaural stereo using a standard SOFA HRTF.
|
||||
|
||||
## Building
|
||||
This is research code, not a complete, standards-compliant, or production-grade JOC decoder. It covers only the stream forms currently implemented. Unknown variants fail explicitly—because when the math goes wrong, all that may remain is the cacophony.
|
||||
|
||||
```powershell
|
||||
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release
|
||||
cmake --build build
|
||||
ctest --test-dir build --output-on-failure
|
||||
```
|
||||
## Current features
|
||||
|
||||
Only MSVC (VS 2022, static CRT) is validated locally; Linux and macOS are built and
|
||||
unit-tested by `.github/workflows/ci.yml`. Floating-point behaviour is part of the
|
||||
byte-exact acceptance, so fast-math is never enabled: `/fp:precise` on MSVC,
|
||||
`-fno-fast-math` elsewhere.
|
||||
- Scan common contiguous EMDF containers in E-AC-3 sync frames.
|
||||
- Parse ID14 dense / sparse JOC parameters, Huffman data, differential matrices, and `joc_clipgain`.
|
||||
- Parse ID11 OAMD position updates and build object trajectories.
|
||||
- Reconstruct LFE plus 15 object channels through analysis QMF, parameter interpolation, the object matrix, and inverse QMF.
|
||||
- Write a 25-channel ADM BWF: a 10-channel 7.1.2 bed (silent except for LFE) plus 15 objects.
|
||||
- Render directly to `2.0`, `3.1`, `5.1`, `7.1`, `5.1.2`, `5.1.4`, `7.1.2`, `7.1.4`, `9.1.4`, or `9.1.6`.
|
||||
- Run public SOFA binaural rendering directly from `pcm16 + ID11/OAMD`, without a temporary ADM BWF.
|
||||
- Keep the binaural DSP in float64/complex128, including 961-sample latency compensation, cross-frame state, and the room tail.
|
||||
- Use a shared float32/PCM24 WAV writer and explicit PCM24 clipping policy for direct outputs.
|
||||
- Use the NumPy backend or an optional C++20 core through `ctypes`; `auto` falls back to Python when the native library is unavailable.
|
||||
- Read or write metadata sidecars and produce metadata, timing, and output reports.
|
||||
|
||||
Windows Release builds target AVX2 by default (`JOC_ENABLE_AVX2`, ON, see
|
||||
`CMakeLists.txt`). That switch is itself part of the byte-exact acceptance — the
|
||||
SHA-256 of every rendered output is identical — and buys 12.4% on the SOFA binaural
|
||||
kernel and 1.8% on Rosella. The price is a runtime requirement: such a `joc_core.dll`
|
||||
executes AVX2 instructions and dies on an illegal instruction on pre-2013 x86. There
|
||||
is no runtime dispatch, so a binary is one or the other:
|
||||
|
||||
```powershell
|
||||
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release -DJOC_ENABLE_AVX2=OFF
|
||||
```
|
||||
|
||||
gives a baseline-ISA (SSE2) build that runs on any x86-64. Non-MSVC builds never
|
||||
receive the flag.
|
||||
|
||||
### Paths and encoding
|
||||
|
||||
Every path inside the library is **UTF-8**, converted only at the OS boundary
|
||||
(`src/foundation/fs_utf8.*`): on Windows through `std::filesystem::path` (UTF-16
|
||||
inside) into `_wfopen`/`CreateProcessW`, and as plain bytes elsewhere. Command line
|
||||
arguments are re-parsed from `GetCommandLineW` + `CommandLineToArgvW` on Windows and
|
||||
the console code page is set to UTF-8, so non-ASCII paths (Japanese, Chinese, ...)
|
||||
work for the input, the ffmpeg child process and the output files alike; a
|
||||
non-ASCII path regression case runs in `ctest`.
|
||||
|
||||
## Artifacts
|
||||
|
||||
| Artifact | Purpose |
|
||||
|---|---|
|
||||
| `joc_core.dll` | The engine: bitstream parsing, JOC/OAMD, DSP, speaker and binaural rendering, ADM BWF/WAV writing, file task, streaming surface |
|
||||
| `joc_cli.exe` | Command line frontend for file tasks |
|
||||
| `include/joc_core.h` | Engine, telemetry and file-task interface (pure C) |
|
||||
| `include/joc_stream.h` | Embedder-facing streaming push/pull interface (pure C, self-contained) |
|
||||
|
||||
## Command line
|
||||
|
||||
The arguments match the reference Python CLI exactly: the input is positional,
|
||||
**ADM BWF is the default output**, `--speaker-layout` or `--binaural` selects the
|
||||
other two modes, and without `-o` the result lands in `output/`.
|
||||
|
||||
```powershell
|
||||
# Default: ADM BWF (inherently 24-bit, so there is no format option)
|
||||
joc_cli "07. Gold Forever (2021 Master).m4a"
|
||||
# -> output/07. Gold Forever (2021 Master).adm.wav
|
||||
|
||||
joc_cli input.m4a -o out/adm.wav # explicit output
|
||||
|
||||
# Speaker layout
|
||||
joc_cli input.m4a --speaker-layout 5.1 # -> output/<name>.5.1.wav
|
||||
joc_cli input.m4a --speaker-layout 7.1.4 --speaker-output out/714.wav --speaker-format int24
|
||||
|
||||
# Binaural (HRTF defaults to <exe>/HRTF/binaural.sofa, then <exe>/HRTF/binaural.personalized_headphone)
|
||||
joc_cli input.m4a --binaural
|
||||
joc_cli input.m4a --binaural --sofa-hrtf HRTF/other.sofa # another SOFA
|
||||
joc_cli input.m4a --binaural --personalized-headphone # Rosella personalisation
|
||||
# -> output/<name>.binaural.wav
|
||||
|
||||
# Other common switches
|
||||
joc_cli input.m4a --duration 30 --gain-db -3 --trajectory-mode dense64
|
||||
joc_cli input.eac3 --metadata-only --print-metadata summary # parse and print metadata only
|
||||
```
|
||||
|
||||
`--speaker-format` / `--binaural-format` default to `float32`; an `int24` request that
|
||||
would clip follows `--clip-action` (default `ask`; a non-interactive terminal must
|
||||
pass `continue`, `float32` or `abort`). `--duration` is in **seconds**, and
|
||||
`--object-delay-samples`, `--speaker-metadata-offset`, `--binaural-tail-seconds` and
|
||||
`--binaural-tail-threshold` (1e-8, the binaural tail trim) keep the reference
|
||||
defaults. A run always writes `<output>.report.json` (`--report-json` overrides it).
|
||||
|
||||
**Differences from the reference:** `--sofa-hrtf`, `--personalized-headphone`,
|
||||
`--backend python` and the metadata sidecars (`--metadata-dir`, `--metadata-cache`,
|
||||
`--metadata-backend sidecar`) are unavailable in this build and fail immediately
|
||||
with an explanation instead of being ignored. This build adds `--bed` (pre-decoded
|
||||
6-channel float32 PCM, which skips ffmpeg decoding), `--kernels`, `--work-dir`,
|
||||
`--report-json`, `--dry-run` and `--quiet`.
|
||||
|
||||
## Library integration
|
||||
|
||||
The engine and the file task are exposed by `joc_core.h`: `joc_task_validate` /
|
||||
`joc_task_execute` run a file task and report state events through a callback, and
|
||||
`joc_task_result` carries frame counts, peak, byte count and SHA-256.
|
||||
|
||||
Players and decoder components use `joc_stream.h`: the caller pushes E-AC-3 bytes
|
||||
and the matching core PCM (or already-rebuilt objects16) at its own pace and pulls
|
||||
rendered PCM. Any chunking is allowed, and the result is byte-identical to the file
|
||||
task.
|
||||
|
||||
```c
|
||||
joc_stream_config config = {0};
|
||||
config.struct_size = sizeof(config);
|
||||
config.input = JOC_STREAM_IN_EAC3; /* or JOC_STREAM_IN_PCM_OBJECTS16 */
|
||||
config.output = JOC_STREAM_OUT_SPEAKER; /* or BINAURAL / PCM_OBJECTS16 */
|
||||
config.speaker_layout_name = "5.1";
|
||||
joc_stream* stream = NULL;
|
||||
joc_stream_create(&config, &stream);
|
||||
/* loop: joc_stream_push(...) / joc_stream_pull(...) */
|
||||
joc_stream_flush(stream);
|
||||
joc_stream_destroy(stream);
|
||||
```
|
||||
|
||||
Contract: state is instance-private, so streams coexist; push and pull on one
|
||||
instance must come from the same thread; rendering is stateful, so **this version
|
||||
offers no seek** - repositioning means decoding from the start of the stream. The
|
||||
kernel latency is 961 samples for speaker/binaural output and `joc_stream_flush`
|
||||
drains the binaural room tail.
|
||||
|
||||
## Layout
|
||||
## Processing flow
|
||||
|
||||
```text
|
||||
include/ public C ABI: joc_core.h (engine/file task), joc_stream.h (streaming),
|
||||
eac3joc_core.h (upstream ABI)
|
||||
src/ implementation: eac3_transport, emdf, joc_bitstream, joc_core, oamd,
|
||||
timeline, speaker, binaural, hrtf, adm, io, telemetry, task, stream,
|
||||
api, cli, simd
|
||||
tests/ unit tests (CTest, self-contained, no external data)
|
||||
docs/ mathematics, the binaural rendering flow and SIMD dispatch
|
||||
M4A / E-AC-3
|
||||
├─ FFmpeg extracts E-AC-3 and decodes the core 5.1 PCM
|
||||
├─ EMDF → ID14 JOC parameters → object matrix
|
||||
├─ core PCM → analysis QMF → parameter interpolation → inverse QMF
|
||||
├─ ID11 OAMD → object positions and timing
|
||||
└─ LFE + 15 objects
|
||||
├─ 25ch ADM BWF
|
||||
├─ speaker WAV for the selected layout
|
||||
└─ direct ID11 timeline + SOFA HRTF → binaural WAV
|
||||
```
|
||||
|
||||
Eight files under `src/` are byte-identical copies of the upstream JustOneCacophony
|
||||
native library and are never edited (see [THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md)).
|
||||
The Python and C++ backends follow the same mathematics for JOC object reconstruction and speaker rendering. The public SOFA binaural backend currently runs in Python; bitstream parsing, the OAMD timeline, and CLI behavior also remain in Python.
|
||||
|
||||
## Compatibility note
|
||||
## Requirements
|
||||
|
||||
`object_delay_samples` defaults to **1473**, preserving the behaviour of the existing
|
||||
implementation; it is a configurable field and changing it changes the OAMD/ADM time
|
||||
alignment. Upstream investigation suggests the value should be 0; this project keeps
|
||||
the current default to stay byte-identical.
|
||||
- Python 3.10+
|
||||
- NumPy 1.24+
|
||||
- h5py 3.8+
|
||||
- SciPy 1.10+
|
||||
- A standalone FFmpeg executable; `ffmpeg-python` is not required. FFmpeg is discovered through `PATH` by default or selected with `--ffmpeg`. On startup the decoder options are probed with `ffmpeg -h decoder=eac3`: a missing E-AC-3 decoder or `-drc_scale` is a hard error, while a missing `-target_level` only fails when `--eac3-target-level` is used
|
||||
- Optional: CMake and a C++20 toolchain to build the native core
|
||||
|
||||
## License
|
||||
Install the Python dependency in a project-specific environment:
|
||||
|
||||
MIT, see [LICENSE](LICENSE). Third-party provenance and patent boundaries are in
|
||||
[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md).
|
||||
```powershell
|
||||
python -m pip install -r requirements.txt
|
||||
```
|
||||
|
||||
If FFmpeg is not on `PATH`:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --ffmpeg C:\path\to\ffmpeg.exe
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Write a 25-channel ADM BWF by default:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a
|
||||
```
|
||||
|
||||
Select a backend or output path:
|
||||
|
||||
```powershell
|
||||
python main.py input.eac3 -o output.adm.wav --backend python
|
||||
python main.py input.m4a --backend native --native-threads 2
|
||||
python main.py input.m4a --native-library lib/eac3joc_core.dll
|
||||
```
|
||||
|
||||
Write a speaker-layout WAV directly:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --speaker-layout 2.0 --speaker-format float32
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24
|
||||
python main.py input.m4a --speaker-layout 7.1.2 --speaker-output output.7.1.2.wav
|
||||
```
|
||||
|
||||
Write binaural stereo directly (ordinary objects are Near/Mid/Far only; Mid is
|
||||
the default). The HRTF input accepts three sources:
|
||||
|
||||
```powershell
|
||||
# 1) SOFA (defaults to HRTF/binaural.sofa, or an explicit path)
|
||||
python main.py input.m4a --binaural
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa
|
||||
|
||||
# 2) Rosella .personalized_headphone (defaults to HRTF/binaural.personalized_headphone)
|
||||
python main.py input.m4a --binaural --personalized-headphone
|
||||
python main.py input.m4a --binaural --personalized-headphone C:\HRTF\subject.personalized_headphone
|
||||
|
||||
# 3) .jochrtf compiled cache
|
||||
python main.py input.m4a --binaural --compiled-hrtf-cache C:\HRTF\subject.jochrtf
|
||||
|
||||
# Common options
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
|
||||
--binaural-mode near
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
|
||||
--hrtf-cache-policy disk
|
||||
python main.py input.m4a --binaural --binaural-output output.binaural.wav
|
||||
```
|
||||
|
||||
With none of the three specified, resolution tries, in order:
|
||||
`HRTF/binaural.sofa`, the unique `.jochrtf` under `output/hrtf-cache`, then
|
||||
`HRTF/binaural.personalized_headphone`; if none exist, an error asks for an
|
||||
explicit path.
|
||||
|
||||
- `.sofa` is the portable source of truth; it can hold self-scanned or any
|
||||
generic HRTF data.
|
||||
- `.personalized_headphone` is a model produced by Dolby's official
|
||||
personalization scan; its JSON parsing is implemented by this project
|
||||
(`src/rosella_model.py`) and does not invoke any Dolby software.
|
||||
- `.jochrtf` is a project-internal cache compiled from SOFA; it is disposable,
|
||||
rebuildable, and written to `output/hrtf-cache` by default.
|
||||
|
||||
HRTF data lives under `HRTF/` (git-ignored): the default SOFA
|
||||
`HRTF/binaural.sofa` and the default model
|
||||
`HRTF/binaural.personalized_headphone`. Because the cache contains transformed
|
||||
HRTF data, its use and redistribution remain subject to the source dataset's
|
||||
terms. See [Binaural Rendering](docs/binaural.en.md) and
|
||||
[Third-party notices](THIRD_PARTY_NOTICES.md) for format boundaries, formulas,
|
||||
state, timing, and distribution considerations.
|
||||
|
||||
Speaker and binaural output share peak analysis, the WAV writer, and clipping policy. When PCM24 may clip in a non-interactive environment, select a policy explicitly:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action abort
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action float32
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action continue
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
|
||||
--binaural-format int24 --clip-action abort
|
||||
```
|
||||
|
||||
Metadata and diagnostics:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --print-metadata summary
|
||||
python main.py input.m4a --metadata-only --print-metadata frames
|
||||
python main.py input.m4a --metadata-cache metadata_cache
|
||||
python main.py input.m4a --metadata-dir metadata_cache
|
||||
```
|
||||
|
||||
### E-AC-3 decode-side dynamic range and level
|
||||
|
||||
By default FFmpeg applies the stream `dynrng` dynamic range compression when
|
||||
decoding E-AC-3 (`-drc_scale 1`). The core 5.1 PCM is the input of JOC object
|
||||
reconstruction, and `dynrng` is playback-time gain, so it is inherited linearly
|
||||
by every object and every output (ADM, speaker, binaural). This tool therefore
|
||||
decodes at **full dynamic range** by default:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a # default: -drc_scale 0, full range
|
||||
python main.py input.m4a --eac3-drc-scale 1 # reproduce consumer playback
|
||||
python main.py input.m4a --eac3-drc-scale 0.5 # apply half of it
|
||||
python main.py input.m4a --eac3-target-level -27 # dialnorm-referenced level
|
||||
```
|
||||
|
||||
- `--eac3-drc-scale` (`0`–`6`, default `0`) maps to FFmpeg `-drc_scale`: the gain
|
||||
of each E-AC-3 block is `dynrng factor ^ value`. `0` disables DRC, `1` is the
|
||||
author's intent, and `>1` is asymmetric (loud parts fully compressed, quiet
|
||||
parts enhanced).
|
||||
- `--eac3-target-level` (`-31`–`0`, default `0` = off) maps to FFmpeg
|
||||
`-target_level`: a static per-frame gain of about `target_level - dialnorm` dB,
|
||||
independent of and stackable with `--eac3-drc-scale`. dialnorm is a per-stream
|
||||
property (measured Apple Music Atmos streams are about `-18` to `-19` dB, so
|
||||
`-27` is roughly `8`–`9` dB of attenuation).
|
||||
- The level change is expected: compared with the FFmpeg default, measured
|
||||
tracks move by `0` to `-2.15` dB peak and `0` to `-1.69` dB RMS (direction
|
||||
depends on the stream `dynrng`), so `output_clip.peak` and the PCM24 clipping
|
||||
decision in `.report.json` change accordingly.
|
||||
- `--gain-db` is a static gain applied **after** reconstruction (float64 on the
|
||||
binaural path) and is not the same thing as decode-side DRC, which is
|
||||
block-varying; do not use `--gain-db` to cancel it.
|
||||
- The `ffmpeg` field of `.report.json` records the FFmpeg version and the decode
|
||||
options that were actually passed (`version`, `eac3_decode_options`).
|
||||
|
||||
### Binaural render mode
|
||||
|
||||
`--binaural-mode off|near|mid|far` selects the binaural render mode; the default
|
||||
is `mid`, and both outputs share this single option:
|
||||
|
||||
- **Direct binaural rendering** (`--binaural`): `off` is rejected (error);
|
||||
near/mid/far apply, defaulting to `mid`;
|
||||
- **ADM BWF**: the low 3 binaural-render-mode bits of the last 15 JOC object
|
||||
entries in DBMD segment 10 carry `off=0/near=1/far=2/mid=3`, leaving the first
|
||||
10 bed entries unchanged; the default is `mid`, and `off` explicitly disables
|
||||
the binaural metadata hint.
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --binaural-mode mid
|
||||
python main.py input.m4a --binaural-mode off # ADM BWF only: disable the DBMD hint
|
||||
```
|
||||
|
||||
**The default `mid` is a human-specified rendering hint**; it is not original
|
||||
binaural metadata extracted or recovered from the input E-AC-3 JOC bitstream,
|
||||
nor does it represent the original mix's per-object binaural settings. The hint
|
||||
does not change PCM, object trajectories, or direct speaker rendering. The
|
||||
adjacent `.report.json` records `binaural_mode` (the mode name) and
|
||||
`binaural_mode_value` (the ADM code; `null` for direct binaural output).
|
||||
|
||||
### OAMD time alignment
|
||||
|
||||
Object trajectories and direct speaker rendering both default to a metadata delay of `1473 samples`. This value describes the theoretical mapping between decoder-output PCM and OAMD updates. The speaker renderer retains its existing 32-sample control block, so the default update lands on effective block boundary `1472`:
|
||||
|
||||
```text
|
||||
align32(1473) = 1472
|
||||
```
|
||||
|
||||
Override the two paths with `--object-delay-samples` and `--speaker-metadata-offset`, respectively. The 1473-sample timing offset is distinct from the 640-value inverse-QMF filter/window state; 640 is a QMF state length, not a metadata delay.
|
||||
|
||||
The direct binaural path uses `--object-delay-samples`. Each ID11/OAMD event is
|
||||
placed on an absolute sample timeline from its frame start, outer-subpayload
|
||||
offset, and block offset, then shifted by that delay. Each 1536-sample input
|
||||
frame is processed as three consecutive 512-sample blocks; the interpolated
|
||||
position, direction, and profile are updated at each block's absolute starting
|
||||
sample.
|
||||
|
||||
### Binaural calculation
|
||||
|
||||
See [Binaural Rendering Mathematics](docs/binaural.en.md) for QMF, hybrid processing, direction fields, distance, ITD, room processing, the 512-sample parameter updates above, and 961-sample latency compensation.
|
||||
|
||||
For all options:
|
||||
|
||||
```powershell
|
||||
python main.py --help
|
||||
```
|
||||
|
||||
Without `-o`, output still goes to `output/` at the repository root. The directory move intentionally preserves this behavior.
|
||||
|
||||
## Native core
|
||||
|
||||
The repository does not include native binaries by default. Download a prebuilt runtime for the current platform from a project Release, or build one locally, then place the runtime library under `lib/` at the repository root; create the directory if it is absent. To build it yourself, run CMake from the repository root:
|
||||
|
||||
```powershell
|
||||
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
|
||||
cmake --build build/cmake --config Release
|
||||
cmake --install build/cmake --config Release
|
||||
```
|
||||
|
||||
The runtime lookup order is:
|
||||
|
||||
1. `--native-library`;
|
||||
2. `EAC3JOC_NATIVE_LIBRARY`;
|
||||
3. the standard platform library name under `lib/`.
|
||||
|
||||
See the [native-core notes](docs/native.en.md) for ABI, state, and precision details.
|
||||
|
||||
## Repository layout
|
||||
|
||||
```text
|
||||
JustOneCacophony/
|
||||
├─ main.py command-line entry point
|
||||
├─ src/ Python implementation modules
|
||||
├─ native/ C/C++ acceleration core, C ABI, and required table data
|
||||
├─ data/ Python runtime table data
|
||||
├─ lib/ native runtime drop-in directory (create as needed)
|
||||
├─ HRTF/ user HRTF data directory (create as needed, git-ignored)
|
||||
├─ output/ output directory (create as needed; the .jochrtf cache defaults to its hrtf-cache subdirectory)
|
||||
├─ docs/ math and native-core notes in both languages
|
||||
├─ requirements.txt Python dependency
|
||||
├─ README.md Chinese documentation
|
||||
└─ README.en.md English documentation
|
||||
```
|
||||
|
||||
## Mathematical implementation
|
||||
|
||||
The main documented stages are:
|
||||
|
||||
- dense JOC differential reconstruction and dequantization;
|
||||
- parameter-band mapping to 64 QMF subbands;
|
||||
- cross-frame parameter interpolation;
|
||||
- analysis/inverse QMF, surround delay, and FIR state;
|
||||
- the 1217-sample LFE delay;
|
||||
- OAMD Q15 coordinate conversion;
|
||||
- equal-power panning over target-layout regions;
|
||||
- layout-dependent position compensation and sample-wise gain ramps;
|
||||
- float32 and PCM24 output quantization;
|
||||
- SOFA canonical import, 64-QMF/77-hybrid projection, `36×2×77` fifth-order fields, exactly-once delay/phase, project early/late room behavior, and special LFE.
|
||||
|
||||
See the [mathematical notes](docs/math.en.md) for the equations used by the decoding and rendering process.
|
||||
|
||||
## Known limitations
|
||||
|
||||
- Only the common contiguous EMDF transport is covered. Fragmented transport across multiple audio-block skip fields is not covered.
|
||||
- The speaker and SOFA binaural paths currently cover ordinary point objects; extent, spread, diffuse, divergence, channel lock, and similar controls are outside the supported scope.
|
||||
- OAMD trim elements are boundary-checked and skipped; warp, balance, and trim parameters are not applied to raw object trajectories or speaker rendering.
|
||||
- Multi-data-point streams, uncommon band configurations, and unusual OAMD scheduling have less coverage than common 12-band, single-data-point material.
|
||||
- A speaker limiter is outside the current primary formula.
|
||||
- The SOFA importer currently supports the strict `SimpleFreeFieldHRIR` FIR subset; other SOFA conventions require explicit adapters.
|
||||
- The binaural runtime is fixed at 48 kHz, fifth order, and one measurement-radius shell at a time; the public binaural backend defaults to the native accelerator and falls back to Python when the native library is unavailable.
|
||||
- ADM output, native binaries, speaker layouts, and binaural models still need broader interoperability checks across platforms, players, and real material.
|
||||
|
||||
## Documentation
|
||||
|
||||
- [Mathematical notes](docs/math.en.md) · [中文](docs/math.md)
|
||||
- [Native-core notes](docs/native.en.md) · [中文](docs/native.md)
|
||||
- [Binaural rendering](docs/binaural.en.md) · [中文](docs/binaural.md)
|
||||
|
||||
@@ -1,138 +1,273 @@
|
||||
# JustOneCacophony — C++ Core
|
||||
# JustOneCacophony — JOC
|
||||
|
||||
[English](README.en.md) · [数学说明](docs/math.md) · [双耳渲染](docs/binaural.md) · [SIMD 派发](docs/simd.md)
|
||||
[English](README.en.md)
|
||||
|
||||
JustOneCacophony 的 C++ 实现:E-AC-3 JOC 码流解析、对象重建与渲染的执行内核。它从 E-AC-3
|
||||
同步帧中提取 EMDF、ID14 JOC 参数与 ID11 OAMD 元数据,结合 FFmpeg 解码出的核心 5.1 PCM
|
||||
重建 LFE 与 15 路对象 PCM,并输出 ADM BWF、指定扬声器布局的 WAV,或用编译好的 HRTF 方向场
|
||||
直接输出双耳 WAV。
|
||||
> JustOneCacophony 是一个 E-AC-3 JOC 的实验性 / 测试实现,用于研究 JOC 的解析、重建、渲染以及相关数学过程。
|
||||
|
||||
这是研究代码,不是完整、标准兼容或生产级的 JOC 解码器。它只覆盖已实现的码流形态,遇到未知
|
||||
变体时明确报错,而不是假装一切都很和谐。
|
||||
项目可以从常见 E-AC-3 JOC 码流中提取并解析 EMDF、ID14 JOC 参数和 ID11 OAMD 元数据,结合 FFmpeg 解码出的核心 5.1 PCM 重建 LFE 与 15 路对象 PCM,并输出 ADM BWF、指定扬声器布局的 WAV,或使用标准 SOFA HRTF 直接输出双耳 WAV。
|
||||
|
||||
## 构建
|
||||
这是研究代码,不是完整、标准兼容或生产级的 JOC 解码器。它只覆盖当前已实现的码流形态;遇到未知变体时会明确报错,而不是假装一切都很和谐——如果哪里算错了,它可能就真的只剩 cacophony 了。
|
||||
|
||||
```powershell
|
||||
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release
|
||||
cmake --build build
|
||||
ctest --test-dir build --output-on-failure
|
||||
```
|
||||
## 当前功能
|
||||
|
||||
Windows(MSVC / VS 2022,静态 CRT)、Linux 与 macOS 由 `.github/workflows/ci.yml` 同时构建并跑
|
||||
单元测试;本地只验证 MSVC。浮点行为是逐字节验收的一部分,因此不启用 fast-math:MSVC 用
|
||||
`/fp:precise`,其他编译器用 `-fno-fast-math`。
|
||||
- 扫描 E-AC-3 同步帧中的常见连续 EMDF 容器;
|
||||
- 解析 ID14 dense / sparse JOC 参数、Huffman 数据、差分矩阵与 `joc_clipgain`;
|
||||
- 解析 ID11 OAMD 位置更新并生成对象轨迹;
|
||||
- 通过 analysis QMF、参数插值、对象矩阵和 inverse QMF 重建 LFE + 15 路对象 PCM;
|
||||
- 输出 25 声道 ADM BWF:10 声道 7.1.2 bed(除 LFE 外静音)+ 15 个对象;
|
||||
- 直接渲染 `2.0`、`3.1`、`5.1`、`7.1`、`5.1.2`、`5.1.4`、`7.1.2`、`7.1.4`、`9.1.4`、`9.1.6`;
|
||||
- 从 `pcm16 + ID11/OAMD` 直接运行公开 SOFA 双耳渲染,不生成临时 ADM BWF;
|
||||
- 双耳 DSP 全程使用 float64/complex128,并保留 961-sample latency compensation、跨帧状态和 room 尾声;
|
||||
- 直接输出统一支持 float32 或 PCM24 WAV,并在 PCM24 削波前提供明确处理策略;
|
||||
- 使用 NumPy 后端,或通过 `ctypes` 调用可选的 C++20 原生核;`auto` 模式在原生库不可用时回退到 Python;
|
||||
- 读取或写入 metadata sidecar,并生成元数据、运行时间和输出摘要。
|
||||
|
||||
Windows Release 默认带 `/arch:AVX2`(`JOC_ENABLE_AVX2`,默认 ON,见 `CMakeLists.txt`)。这个
|
||||
开关是逐字节验收过的:所有渲染产物的 SHA-256 完全一致,换来 SOFA 双耳内核 12.4%、Rosella
|
||||
1.8% 的提升。代价是运行要求——这样的 `joc_core.dll` 会执行 AVX2 指令,在 2013 年以前的 x86 上
|
||||
直接非法指令退出。没有运行时派发,一个二进制只能二选一,所以:
|
||||
|
||||
```powershell
|
||||
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release -DJOC_ENABLE_AVX2=OFF
|
||||
```
|
||||
|
||||
得到基线 ISA(SSE2)、任何 x86-64 都能跑的产物。非 MSVC 构建永远不会带上这个开关。
|
||||
|
||||
### 路径与编码
|
||||
|
||||
库内部所有路径都是 **UTF-8**,只在系统边界转换(`src/foundation/fs_utf8.*`):Windows 上经
|
||||
`std::filesystem::path`(内部 UTF-16)落到 `_wfopen`/`CreateProcessW`,其他平台直接是字节。
|
||||
命令行参数在 Windows 上由 `GetCommandLineW` + `CommandLineToArgvW` 重新解析,控制台设为
|
||||
UTF-8,因此日文/中文等非 ASCII 路径(含 ffmpeg 子进程与输出文件)都能正常工作;单元测试里有
|
||||
一条非 ASCII 路径的回归用例守在 `ctest` 里。
|
||||
|
||||
## 产物
|
||||
|
||||
| 产物 | 说明 |
|
||||
|---|---|
|
||||
| `joc_core.dll` | 执行内核:码流解析、JOC/OAMD、DSP、扬声器与双耳渲染、ADM BWF/WAV 落盘、文件任务、流式接口 |
|
||||
| `joc_cli.exe` | 文件任务命令行前端 |
|
||||
| `include/joc_core.h` | 引擎、遥测与文件任务接口(纯 C) |
|
||||
| `include/joc_stream.h` | 面向嵌入者的流式 push/pull 接口(纯 C,仅包含它即可) |
|
||||
|
||||
## 命令行
|
||||
|
||||
参数与上游 Python CLI 完全一致:输入是位置参数,**默认输出 25 通道 ADM BWF**,用
|
||||
`--speaker-layout` 或 `--binaural` 切换到另外两种模式;未指定 `-o` 时产物落在 `output/`。
|
||||
|
||||
```powershell
|
||||
# 默认:ADM BWF(本身就是 24-bit,没有也不需要格式参数)
|
||||
joc_cli "07. Gold Forever (2021 Master).m4a"
|
||||
# -> output/07. Gold Forever (2021 Master).adm.wav
|
||||
|
||||
joc_cli input.m4a -o out/adm.wav # 指定输出
|
||||
|
||||
# 扬声器布局
|
||||
joc_cli input.m4a --speaker-layout 5.1 # -> output/<名称>.5.1.wav
|
||||
joc_cli input.m4a --speaker-layout 7.1.4 --speaker-output out/714.wav --speaker-format int24
|
||||
|
||||
# 双耳(HRTF 默认取 <exe>/HRTF/binaural.sofa,其次 <exe>/HRTF/binaural.personalized_headphone)
|
||||
joc_cli input.m4a --binaural
|
||||
joc_cli input.m4a --binaural --sofa-hrtf HRTF/other.sofa # 换一个 SOFA
|
||||
joc_cli input.m4a --binaural --personalized-headphone # Rosella 个性化模型
|
||||
# -> output/<名称>.binaural.wav
|
||||
|
||||
# 其它常用开关
|
||||
joc_cli input.m4a --duration 30 --gain-db -3 --trajectory-mode dense64
|
||||
joc_cli input.eac3 --metadata-only --print-metadata summary # 只解析并打印元数据
|
||||
```
|
||||
|
||||
`--speaker-format` / `--binaural-format` 默认 `float32`;`int24` 若会削波按 `--clip-action`
|
||||
处理(默认 `ask`,非交互终端下需显式给出 `continue`/`float32`/`abort`)。`--duration` 以**秒**
|
||||
为单位;`--object-delay-samples`、`--speaker-metadata-offset`、`--binaural-tail-seconds`、
|
||||
`--binaural-tail-threshold`(默认 1e-8,双耳尾音裁切阈值)等默认值与上游一致。命令总是写出
|
||||
`<输出>.report.json`(`--report-json` 可改路径)。
|
||||
|
||||
**与上游参数的差异**:`--sofa-hrtf`、`--personalized-headphone`、`--backend python` 以及
|
||||
metadata sidecar(`--metadata-dir`/`--metadata-cache`/`--metadata-backend sidecar`)在本构建中
|
||||
不可用,给出时立刻报错并说明原因,而不是静默忽略。本构建额外提供 `--bed`(已解码的 6 通道
|
||||
float32 PCM,给出后不调用 ffmpeg 解码)、`--kernels`(滤波器组表路径)、`--work-dir`、
|
||||
`--report-json`、`--dry-run`、`--quiet`。
|
||||
|
||||
## 库集成
|
||||
|
||||
引擎与文件任务使用 `joc_core.h`:`joc_task_validate` / `joc_task_execute` 跑一个文件任务并
|
||||
通过回调返回状态事件,`joc_task_result` 给出帧数、峰值、字节数与 SHA-256。
|
||||
|
||||
播放器或解码组件使用 `joc_stream.h`:调用方按自己的节奏推入 E-AC-3 字节与对应的核心 PCM
|
||||
(或已重建的 objects16),再拉取渲染后的 PCM;分块粒度任意,输出与文件任务逐字节一致。
|
||||
|
||||
```c
|
||||
joc_stream_config config = {0};
|
||||
config.struct_size = sizeof(config);
|
||||
config.input = JOC_STREAM_IN_EAC3; /* 或 JOC_STREAM_IN_PCM_OBJECTS16 */
|
||||
config.output = JOC_STREAM_OUT_SPEAKER; /* 或 BINAURAL / PCM_OBJECTS16 */
|
||||
config.speaker_layout_name = "5.1";
|
||||
joc_stream* stream = NULL;
|
||||
joc_stream_create(&config, &stream);
|
||||
/* 循环:joc_stream_push(...) / joc_stream_pull(...) */
|
||||
joc_stream_flush(stream);
|
||||
joc_stream_destroy(stream);
|
||||
```
|
||||
|
||||
契约:状态实例私有、可并存;同一实例的 push/pull 必须在同一线程;渲染是有状态的,
|
||||
因此**本版本不提供 seek**——定位需要从流起点重新解码。扬声器/双耳通路的内核延迟为
|
||||
961 样本,`joc_stream_flush` 负责排空双耳房间尾音。
|
||||
|
||||
## 目录
|
||||
## 处理流程
|
||||
|
||||
```text
|
||||
include/ 公共 C ABI:joc_core.h(引擎/文件任务)、joc_stream.h(流式)、eac3joc_core.h(上游 ABI)
|
||||
src/ 实现:eac3_transport、emdf、joc_bitstream、joc_core、oamd、timeline、speaker、
|
||||
binaural、hrtf、adm、io、telemetry、task、stream、api、cli、simd
|
||||
tests/ 单元测试(CTest,自足,不需要外部素材)
|
||||
docs/ 数学说明、双耳渲染流程与 SIMD 派发
|
||||
M4A / E-AC-3
|
||||
├─ FFmpeg 提取 E-AC-3 并解码核心 5.1 PCM
|
||||
├─ EMDF → ID14 JOC 参数 → 对象矩阵
|
||||
├─ 核心 PCM → analysis QMF → 参数插值 → inverse QMF
|
||||
├─ ID11 OAMD → 对象位置与时间轨迹
|
||||
└─ LFE + 15 objects
|
||||
├─ 25ch ADM BWF
|
||||
├─ 指定布局的扬声器 WAV
|
||||
└─ ID11 直接时间轴 + SOFA HRTF → 双耳 WAV
|
||||
```
|
||||
|
||||
`src/` 下 8 个文件是 JustOneCacophony 原生库的逐字节副本,永不修改(见
|
||||
[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md))。
|
||||
Python 与 C++ 后端在 JOC 对象重建、扬声器渲染和公开 SOFA 双耳渲染中使用同一组
|
||||
数学过程;native 双耳后端与 Python 参考实现逐值一致(差异 < 1e-9)。位流解析、
|
||||
OAMD 时间轴和命令行逻辑在 Python 中。
|
||||
|
||||
## 兼容性说明
|
||||
## 环境
|
||||
|
||||
`object_delay_samples` 默认 **1473**,与既有实现的行为保持一致;它是可配置字段,改动它会
|
||||
改变 OAMD/ADM 时间对齐。上游调查认为该值应为 0,本项目为保持逐字节等价暂不改默认值。
|
||||
- Python 3.10+
|
||||
- NumPy 1.24+
|
||||
- h5py 3.8+
|
||||
- SciPy 1.10+
|
||||
- 独立的 FFmpeg 可执行程序;不需要 `ffmpeg-python`。默认从 `PATH` 查找,也可通过 `--ffmpeg` 指定可执行文件路径。启动时会探测 `ffmpeg -h decoder=eac3`:缺 E-AC-3 解码器或 `-drc_scale` 直接报错,缺 `-target_level` 只在使用 `--eac3-target-level` 时报错
|
||||
- 可选:支持 C++20 的 CMake 工具链,用于自行构建原生核
|
||||
|
||||
## 许可
|
||||
建议在项目专用虚拟环境中安装依赖:
|
||||
|
||||
MIT,见 [LICENSE](LICENSE);第三方来源与专利边界见
|
||||
[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md)。
|
||||
```powershell
|
||||
python -m pip install -r requirements.txt
|
||||
```
|
||||
|
||||
如果 FFmpeg 不在 `PATH` 中:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --ffmpeg C:\path\to\ffmpeg.exe
|
||||
```
|
||||
|
||||
## 使用方法
|
||||
|
||||
默认输出 25 声道 ADM BWF:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a
|
||||
```
|
||||
|
||||
选择后端或输出路径:
|
||||
|
||||
```powershell
|
||||
python main.py input.eac3 -o output.adm.wav --backend python
|
||||
python main.py input.m4a --backend native --native-threads 2
|
||||
python main.py input.m4a --native-library lib/eac3joc_core.dll
|
||||
```
|
||||
|
||||
直接输出扬声器 WAV:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --speaker-layout 2.0 --speaker-format float32
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24
|
||||
python main.py input.m4a --speaker-layout 7.1.2 --speaker-output output.7.1.2.wav
|
||||
```
|
||||
|
||||
直接输出双耳渲染 WAV(普通对象仅 Near/Mid/Far,默认 Mid)。HRTF 输入支持三种来源:
|
||||
|
||||
```powershell
|
||||
# 1) SOFA(缺省取 HRTF/binaural.sofa,也可显式指定)
|
||||
python main.py input.m4a --binaural
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa
|
||||
|
||||
# 2) Rosella .personalized_headphone(缺省取 HRTF/binaural.personalized_headphone)
|
||||
python main.py input.m4a --binaural --personalized-headphone
|
||||
python main.py input.m4a --binaural --personalized-headphone C:\HRTF\subject.personalized_headphone
|
||||
|
||||
# 3) .jochrtf 编译缓存
|
||||
python main.py input.m4a --binaural --compiled-hrtf-cache C:\HRTF\subject.jochrtf
|
||||
|
||||
# 常用选项
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
|
||||
--binaural-mode near
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
|
||||
--hrtf-cache-policy disk
|
||||
python main.py input.m4a --binaural --binaural-output output.binaural.wav
|
||||
```
|
||||
|
||||
三者都不指定时的自动选择顺序:`HRTF/binaural.sofa` → `output/hrtf-cache` 下唯一的
|
||||
`.jochrtf` → `HRTF/binaural.personalized_headphone`;都没有则报错并提示显式指定。
|
||||
|
||||
- `.sofa` 是可移植的 source of truth;可以是自行扫描或任何来源的通用 HRTF 数据。
|
||||
- `.personalized_headphone` 是杜比官方软件个性化扫描得到的模型,其 JSON 解析由
|
||||
本项目自行实现(`src/rosella_model.py`),不调用杜比软件。
|
||||
- `.jochrtf` 是从 SOFA 编译出的项目内部 cache,可删除、可从 SOFA 重建,默认写在
|
||||
`output/hrtf-cache`。
|
||||
|
||||
HRTF 数据统一放在 `HRTF/`(git 忽略):默认 SOFA `HRTF/binaural.sofa`、默认模型
|
||||
`HRTF/binaural.personalized_headphone`。cache 含有源 HRTF 的变换数据,使用与再分发
|
||||
仍受源数据许可约束;格式边界、计算公式、状态、时间轴及发布注意事项见
|
||||
[双耳渲染](docs/binaural.md) 和 [第三方通知](THIRD_PARTY_NOTICES.md)。
|
||||
|
||||
扬声器和双耳输出共享峰值检查、writer 与削波策略。在非交互环境请求 PCM24 且可能削波时,需要显式选择处理方式:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action abort
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action float32
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action continue
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
|
||||
--binaural-format int24 --clip-action abort
|
||||
```
|
||||
|
||||
元数据与诊断:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --print-metadata summary
|
||||
python main.py input.m4a --metadata-only --print-metadata frames
|
||||
python main.py input.m4a --metadata-cache metadata_cache
|
||||
python main.py input.m4a --metadata-dir metadata_cache
|
||||
```
|
||||
|
||||
### E-AC-3 解码级动态范围与电平
|
||||
|
||||
FFmpeg 解码 E-AC-3 时默认施加码流 `dynrng` 动态范围压缩(`-drc_scale 1`)。核心 5.1 PCM 是 JOC 对象重建的输入,而 `dynrng` 属于回放期增益,会被线性继承到全部对象与成品(ADM/扬声器/双耳),因此本工具默认按**全动态范围**解码:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a # 默认:-drc_scale 0,全动态范围
|
||||
python main.py input.m4a --eac3-drc-scale 1 # 复现消费者回放(码流作者意图)
|
||||
python main.py input.m4a --eac3-drc-scale 0.5 # 施加一半
|
||||
python main.py input.m4a --eac3-target-level -27 # 按码流 dialnorm 归一化电平
|
||||
```
|
||||
|
||||
- `--eac3-drc-scale`(`0`~`6`,默认 `0`)对应 FFmpeg 的 `-drc_scale`:每个 E-AC-3 block 的增益为 `dynrng 因子 ^ 该值`。`0` 关闭 DRC;`1` 为码流作者意图;`>1` 非对称(响处全压、轻处增强)。
|
||||
- `--eac3-target-level`(`-31`~`0`,默认 `0` 不施加)对应 FFmpeg 的 `-target_level`:按每帧 dialnorm 施加静态增益,约 `target_level - dialnorm` dB,与 `--eac3-drc-scale` 相互独立、可叠加。dialnorm 是逐码流属性(实测 Apple Music Atmos 流约 `-18`~`-19` dB,故 `-27` 约等于衰减 `8`~`9` dB)。
|
||||
- 电平变化是预期的:与 FFmpeg 默认值相比,实测曲目峰值变化 `0`~`-2.15` dB、RMS `0`~`-1.69` dB(方向取决于码流 `dynrng`),`.report.json` 的 `output_clip.peak` 与 int24 削波判定会随之变化。
|
||||
- `--gain-db` 是**重建之后**的静态增益(双耳路径 float64),与解码级 DRC 不是一回事;解码级 DRC 是按 block 时变的,不要用 `--gain-db` 去抵消它。
|
||||
- `.report.json` 的 `ffmpeg` 字段记录 FFmpeg 版本与实际下发的解码选项(`version`、`eac3_decode_options`)。
|
||||
|
||||
### 双耳渲染模式
|
||||
|
||||
`--binaural-mode off|near|mid|far` 选择双耳渲染模式,默认 `mid`,两种输出共用这一个选项:
|
||||
|
||||
- **直接双耳渲染**(`--binaural`):`off` 不可用(报错),near/mid/far 生效,默认 `mid`;
|
||||
- **ADM BWF**:DBMD segment 10 中后 15 个 JOC 对象的 binaural render mode 写
|
||||
`off=0/near=1/far=2/mid=3`,前 10 个 bed 保持不变,默认 `mid`;`off` 用于显式
|
||||
关闭双耳元数据提示。
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --binaural-mode mid
|
||||
python main.py input.m4a --binaural-mode off # 仅 ADM BWF:关闭 DBMD 双耳提示
|
||||
```
|
||||
|
||||
**默认 `mid` 是本工具人为指定的渲染提示**,不是从输入 E-AC-3 JOC 码流中提取或
|
||||
还原的原始双耳元数据,也不代表原始混音中各对象的双耳设置。该提示不改变 PCM、
|
||||
对象轨迹或直接扬声器渲染。输出旁的 `.report.json` 用 `binaural_mode`(模式名)
|
||||
和 `binaural_mode_value`(ADM 编码值,直接双耳输出时为 `null`)记录。
|
||||
|
||||
### OAMD 时间对齐
|
||||
|
||||
对象轨迹和直接扬声器渲染的 metadata delay 默认均为 `1473 samples`。该值描述 decoder 输出 PCM 与 OAMD 更新之间的理论时间映射;扬声器 renderer 仍使用现有的 32-sample control block,因此默认更新的实际 block boundary 为 `1472`:
|
||||
|
||||
```text
|
||||
align32(1473) = 1472
|
||||
```
|
||||
|
||||
可分别用 `--object-delay-samples` 和 `--speaker-metadata-offset` 覆盖默认值。这里的 1473 不应与 inverse-QMF 的 640 项 filter/window state 混淆;后者是 QMF 状态长度,不是 metadata delay。
|
||||
|
||||
直接双耳路径使用 `--object-delay-samples`。每个 ID11/OAMD event 先按 frame start、
|
||||
outer subpayload offset 与 block offset 落到绝对 sample timeline,再加该 delay;每个
|
||||
1536-sample 输入帧按三个连续 512-sample block 处理,并在每块的绝对起始 sample
|
||||
查询插值后的位置、更新方向和 profile。
|
||||
|
||||
### 双耳计算
|
||||
|
||||
双耳路径的 QMF、hybrid、方向场、距离、ITD、room、上述 512-sample 参数更新和 961-sample 延迟补偿见[双耳渲染数学](docs/binaural.md)。
|
||||
|
||||
更多参数可查看:
|
||||
|
||||
```powershell
|
||||
python main.py --help
|
||||
```
|
||||
|
||||
未指定 `-o` 时,输出仍写入仓库根目录的 `output/`。这是文件移动后特意保持的原有行为。
|
||||
|
||||
## 原生核
|
||||
|
||||
仓库默认不附带原生二进制。可以从项目 Release 下载适合当前平台的预构建运行库,或自行构建,然后把运行库直接放入仓库根目录的 `lib/`;若该目录不存在,创建即可。自行构建时可从仓库根目录使用 CMake:
|
||||
|
||||
```powershell
|
||||
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
|
||||
cmake --build build/cmake --config Release
|
||||
cmake --install build/cmake --config Release
|
||||
```
|
||||
|
||||
运行时查找顺序为:
|
||||
|
||||
1. `--native-library`;
|
||||
2. `EAC3JOC_NATIVE_LIBRARY`;
|
||||
3. `lib/` 下当前平台的标准库文件名。
|
||||
|
||||
详细 ABI、状态与精度说明见[原生核说明](docs/native.md)。
|
||||
|
||||
## 目录结构
|
||||
|
||||
```text
|
||||
JustOneCacophony/
|
||||
├─ main.py 命令行启动入口
|
||||
├─ src/ Python 实现模块
|
||||
├─ native/ C/C++ 加速核、C ABI 与必要表数据
|
||||
├─ data/ Python 运行时表数据
|
||||
├─ lib/ 原生运行库投放目录(按需创建)
|
||||
├─ HRTF/ 用户 HRTF 数据目录(按需创建,git 忽略)
|
||||
├─ output/ 输出目录(按需创建;.jochrtf 缓存默认在其 hrtf-cache 子目录)
|
||||
├─ docs/ 数学与原生核文档(中英文)
|
||||
├─ requirements.txt Python 依赖
|
||||
├─ README.md 中文说明
|
||||
└─ README.en.md English documentation
|
||||
```
|
||||
|
||||
## 数学实现
|
||||
|
||||
核心过程包括:
|
||||
|
||||
- dense JOC 差分还原与去量化;
|
||||
- 参数带到 64 个 QMF 子带的映射;
|
||||
- 跨帧参数插值;
|
||||
- analysis / inverse QMF、环绕声道延迟与 FIR 状态;
|
||||
- LFE 1217-sample 延迟;
|
||||
- OAMD Q15 坐标转换;
|
||||
- 基于目标布局 region 的等功率声像;
|
||||
- 布局位置补偿与逐样本增益斜坡;
|
||||
- float32 与 PCM24 输出量化;
|
||||
- SOFA canonical importer、64-QMF/77-hybrid 投影、`36×2×77` 五阶方向 field、exactly-once delay/phase、项目 early/late room 与 special LFE。
|
||||
|
||||
解码与渲染过程使用的公式见[数学说明](docs/math.md)。
|
||||
|
||||
## 已知限制
|
||||
|
||||
- 当前只覆盖常见 continuous EMDF transport;跨多个 audio-block skip field 的碎片化 transport 尚未覆盖。
|
||||
- 扬声器与 SOFA 双耳路径当前只覆盖普通点对象;extent、spread、diffuse、divergence、channel lock 等对象控制不在支持范围内。
|
||||
- OAMD trim element 会按声明边界校验并跳过;warp、balance 和 trim 参数不应用于当前原始对象轨迹或扬声器渲染。
|
||||
- 多数据点、少见参数带配置和特殊 OAMD 调度的覆盖度低于常见 12-band、单数据点素材。
|
||||
- 扬声器 limiter 不属于当前实现的主公式。
|
||||
- SOFA importer 当前严格支持 `SimpleFreeFieldHRIR` FIR;其它 SOFA convention 需要显式 adapter。
|
||||
- 双耳 runtime 固定 48 kHz、五阶和一次选择一个 measurement-radius shell;公开双耳默认走 native 加速,原生库不可用时自动回退 Python。
|
||||
- ADM 输出、原生库、扬声器布局和双耳模型仍需在更多平台、播放器与真实素材上确认互操作性。
|
||||
|
||||
## 文档
|
||||
|
||||
- [数学说明](docs/math.md) · [English](docs/math.en.md)
|
||||
- [原生核说明](docs/native.md) · [English](docs/native.en.md)
|
||||
- [双耳渲染](docs/binaural.md) · [English](docs/binaural.en.md)
|
||||
|
||||
@@ -1,24 +1,26 @@
|
||||
# Third-party notices / 第三方通知
|
||||
|
||||
本文件记录 64-QMF / 77-hybrid 滤波器组表(`src/hrtf/public_filterbank.h`、
|
||||
`src/joc_core/qmf_tables.h`)与 JOC Huffman 表(`src/joc_bitstream/joc_huffman_tables.h`)
|
||||
本文件记录 `data/rosella_kernels.npz`(`src/public_filterbank.py` 使用的滤波器组表)
|
||||
的公开标准来源,以及 HRTF 数据与专利的边界说明。
|
||||
|
||||
## 公开标准来源
|
||||
|
||||
64-QMF → 77-hybrid 结构与 13-tap 低带 prototype 定义于
|
||||
[3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf)
|
||||
第 5.2.2 节(Table 1 的 $Q=8$/$Q=4$ 系数,delay 6):
|
||||
第 5.2.2 节(Table 1 的 $Q=8$/
|
||||
$Q=4$ 系数,delay 6):
|
||||
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr)$$
|
||||
|
||||
64-band QMF analysis 即 ISO/IEC 14496-3/AMD1:2003 第 4.B.18.2 节的 MPEG-4
|
||||
AAC/SBR 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap prototype 的多相重排:
|
||||
AAC/SBR 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap prototype 的
|
||||
多相重排:
|
||||
|
||||
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t}$$
|
||||
|
||||
QMF synthesis 表为 analysis 多相矩阵 $\mathbf{A}$ 的因果左逆
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$($\mathbf{P}$ 为 577-sample 延迟置换;
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$(
|
||||
$\mathbf{P}$ 为 577-sample 延迟置换;
|
||||
全链 $961 = 577 + 6\times64$),rank-4 分解存储:
|
||||
|
||||
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
|
||||
@@ -31,8 +33,8 @@ $$Y_p = \sum_{q\in C_p}\Bigl(\mathrm{Re}X_q + j\,s_q\,\mathrm{Im}X_q\Bigr),\qqua
|
||||
|
||||
## HRTF 数据与 `.jochrtf`
|
||||
|
||||
`.jochrtf` 含有特定源 SOFA/HRTF 数据集的变换系数与 delay;其使用、复制与再分发仍受源
|
||||
数据集许可约束,权限不明确时应作为私有 cache 保存。本仓库不分发任何 HRTF 数据集。
|
||||
`.jochrtf` 含有特定源 SOFA/HRTF 数据集的变换系数与 delay;其使用、复制与再分发
|
||||
仍受源数据集许可约束,权限不明确时应作为私有 cache 保存。
|
||||
|
||||
## 专利说明
|
||||
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
# Python runtime tables
|
||||
|
||||
[中文](README.md)
|
||||
|
||||
This directory contains static production tables. It does not contain user HRTFs.
|
||||
|
||||
`tables.npz` contains the JOC core decoding tables:
|
||||
|
||||
```text
|
||||
analysis_window float64[10,64]
|
||||
qmf5_window float64[640]
|
||||
joc_huff_code_coarse_generic int64[95,2]
|
||||
joc_huff_code_fine_generic int64[191,2]
|
||||
joc_huff_code_coarse_coeff_sparse int64[95,2]
|
||||
joc_huff_code_fine_coeff_sparse int64[191,2]
|
||||
joc_huff_code_5ch_pos_index_sparse int64[4,2]
|
||||
joc_huff_code_7ch_pos_index_sparse int64[6,2]
|
||||
```
|
||||
|
||||
`src/joc_qmf.py` loads the QMF tables, while `src/joc_decode.py` loads the JOC Huffman trees. Python does not read C/C++ headers under `native/`.
|
||||
|
||||
The corresponding native data are stored in `native/src/qmf_tables.h` and `native/src/joc_huffman_tables.h`. Changes on either side should update the other and be checked for value-by-value agreement.
|
||||
|
||||
## Binaural rendering tables
|
||||
|
||||
`rosella_kernels.npz` contains the fixed 64-QMF/77-hybrid tables used by the
|
||||
public SOFA binaural path:
|
||||
|
||||
```text
|
||||
format_version little-endian int32[1]
|
||||
qmf_analysis_coefficients float32[64,10]
|
||||
hybrid_analysis_low_kernel float32[3,2,13,16,2]
|
||||
hybrid_synthesis_indices int16[154,4]
|
||||
hybrid_synthesis_values float32[154]
|
||||
qmf_synthesis_basis float64[64,4,128]
|
||||
qmf_synthesis_taps float64[64,10,4]
|
||||
```
|
||||
|
||||
The float32 table values are promoted to float64 when loaded.
|
||||
`src/public_filterbank.py` verifies the archive and every array by SHA-256.
|
||||
Those hashes, the table version, and the 77 reference band-center values all
|
||||
participate in the `.jochrtf` cache key. The full analysis/synthesis latency is
|
||||
961 samples.
|
||||
|
||||
The packaged tables implement publicly standardized filter banks, computable
|
||||
from the following formulas.
|
||||
|
||||
The 64-QMF → 77-hybrid structure, the 13-tap low-band prototypes, and their
|
||||
half-bin complex modulation are defined in
|
||||
[3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf),
|
||||
Section 5.2.2 (Table 1 $Q=8$/$Q=4$ coefficients, delay 6):
|
||||
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\!\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr),\qquad n=0,\dots,12$$
|
||||
|
||||
The 64-band QMF analysis is the MPEG-4 AAC/SBR 64 complex QMF analysis bank of
|
||||
ISO/IEC 14496-3/AMD1:2003, subclause 4.B.18.2; the packaged $64\times10$ table
|
||||
is the polyphase reordering of the public 640-tap prototype $c_0,\dots,c_{639}$:
|
||||
|
||||
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t},\qquad r=0,\dots,63,\ t=0,\dots,9$$
|
||||
|
||||
The QMF synthesis table is the causal left inverse of the analysis polyphase
|
||||
matrix $\mathbf{A}$, i.e. the solution of $\mathbf{A}\,\mathbf{W}=\mathbf{P}$
|
||||
($\mathbf{P}$ is the 577-sample delay permutation; total latency
|
||||
$961 = 577 + 6\times64$), stored as a rank-4 factorization:
|
||||
|
||||
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
|
||||
|
||||
The hybrid synthesis table is the 77→64 recombination: identity for the high
|
||||
bands, $Y_{3+b}=X_{16+b}$, and for the low bands ($C_p$ is the $8+4+4$ child
|
||||
partition):
|
||||
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\operatorname{Re}X_q + j\,s_q\,\operatorname{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
|
||||
The same values also appear in other public implementations of these standards
|
||||
(for example FFmpeg's `aacps_tablegen.h` and `aacsbrdata.h`).
|
||||
|
||||
Public availability of a standard does not by itself grant permission to
|
||||
practice related patent claims.
|
||||
|
||||
SOFA is the user-visible source of truth. A `.jochrtf` file is a disposable JOC
|
||||
compiled HRTF cache that can be rebuilt from SOFA. The cache contains
|
||||
transformed source-HRTF data and remains subject to the source SOFA/HRTF
|
||||
dataset's licence and redistribution restrictions.
|
||||
@@ -0,0 +1,73 @@
|
||||
# Python 运行时表
|
||||
|
||||
[English](README.en.md)
|
||||
|
||||
本目录保存 Python 生产路径使用的静态表数据,不保存用户 HRTF。
|
||||
|
||||
`tables.npz` 保存 JOC 核心解码表:
|
||||
|
||||
```text
|
||||
analysis_window float64[10,64]
|
||||
qmf5_window float64[640]
|
||||
joc_huff_code_coarse_generic int64[95,2]
|
||||
joc_huff_code_fine_generic int64[191,2]
|
||||
joc_huff_code_coarse_coeff_sparse int64[95,2]
|
||||
joc_huff_code_fine_coeff_sparse int64[191,2]
|
||||
joc_huff_code_5ch_pos_index_sparse int64[4,2]
|
||||
joc_huff_code_7ch_pos_index_sparse int64[6,2]
|
||||
```
|
||||
|
||||
`src/joc_qmf.py` 读取 QMF 表,`src/joc_decode.py` 读取 JOC Huffman 树。Python 不读取 `native/` 下的 C/C++ 头文件。
|
||||
|
||||
原生侧对应数据分别位于 `native/src/qmf_tables.h` 与 `native/src/joc_huffman_tables.h`。修改任何一侧时,应同步更新另一侧并进行逐值一致性检查。
|
||||
|
||||
## 双耳渲染表
|
||||
|
||||
`rosella_kernels.npz` 保存公开 SOFA 双耳路径使用的 64-QMF/77-hybrid 固定表:
|
||||
|
||||
```text
|
||||
format_version little-endian int32[1]
|
||||
qmf_analysis_coefficients float32[64,10]
|
||||
hybrid_analysis_low_kernel float32[3,2,13,16,2]
|
||||
hybrid_synthesis_indices int16[154,4]
|
||||
hybrid_synthesis_values float32[154]
|
||||
qmf_synthesis_basis float64[64,4,128]
|
||||
qmf_synthesis_taps float64[64,10,4]
|
||||
```
|
||||
|
||||
float32 表值载入后提升为 float64。`src/public_filterbank.py` 在读取时校验 archive
|
||||
及每个数组的 SHA-256;这些 hash、table version 和 77 个 band-center 参考值共同进入
|
||||
`.jochrtf` cache key。analysis/synthesis 全链 latency 为 961 samples。
|
||||
|
||||
打包表实现的是公开标准化的滤波器组,各表可由如下公式计算。
|
||||
|
||||
64-QMF → 77-hybrid 结构、13-tap 低带 prototype 与半 bin 复调制定义于
|
||||
[3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf)
|
||||
第 5.2.2 节(Table 1 的 $Q=8$/$Q=4$ 系数,delay 6):
|
||||
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\!\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr),\qquad n=0,\dots,12$$
|
||||
|
||||
64-band QMF analysis 即 ISO/IEC 14496-3/AMD1:2003 第 4.B.18.2 节的 MPEG-4
|
||||
AAC/SBR 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap prototype
|
||||
$c_0,\dots,c_{639}$ 的多相重排:
|
||||
|
||||
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t},\qquad r=0,\dots,63,\ t=0,\dots,9$$
|
||||
|
||||
QMF synthesis 表为上述 analysis 多相矩阵 $\mathbf{A}$ 的因果左逆,即求解
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$($\mathbf{P}$ 为 577-sample 延迟置换;
|
||||
全链 $961 = 577 + 6\times64$),以 rank-4 分解形式存储:
|
||||
|
||||
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
|
||||
|
||||
hybrid synthesis 表为 77→64 重组:高频带恒等 $Y_{3+b}=X_{16+b}$;低频带
|
||||
($C_p$ 为 $8+4+4$ 子带划分):
|
||||
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\operatorname{Re}X_q + j\,s_q\,\operatorname{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
|
||||
相同数值可在 FFmpeg(`aacps_tablegen.h`、`aacsbrdata.h`)等公开实现中查到。
|
||||
|
||||
标准可公开获取不等于获准实施相关专利。
|
||||
|
||||
`.sofa` 是用户可见的 source of truth;`.jochrtf` 是可删除、可从 SOFA 重建的
|
||||
JOC compiled HRTF cache。cache 含有源 HRTF 的变换数据,仍受源 SOFA/HRTF
|
||||
数据集的许可与再分发限制约束。
|
||||
Binary file not shown.
Binary file not shown.
+59
-21
@@ -24,29 +24,61 @@ SOFA FIR
|
||||
|
||||
## Inputs
|
||||
|
||||
The binaural backend consumes a compiled directional field (a JOC compiled HRTF cache,
|
||||
`.jochrtf`) plus the shared filterbank tables. Both are read-only inputs: this library
|
||||
performs no parsing or conversion of measurement data formats.
|
||||
The CLI has three mutually exclusive HRTF input sources; with none given, a
|
||||
default rule resolves the input:
|
||||
|
||||
```powershell
|
||||
joc_cli input.m4a --binaural `
|
||||
--compiled-hrtf-cache path\to\subject.jochrtf `
|
||||
--kernels data\rosella_kernels.npz `
|
||||
--binaural-mode mid --binaural-tail-seconds 5.0
|
||||
# 1) SOFA: defaults to HRTF/binaural.sofa, or an explicit path
|
||||
python main.py input.m4a --binaural
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa
|
||||
|
||||
# 2) Rosella .personalized_headphone: defaults to HRTF/binaural.personalized_headphone
|
||||
python main.py input.m4a --binaural --personalized-headphone
|
||||
python main.py input.m4a --binaural --personalized-headphone C:\HRTF\subject.personalized_headphone
|
||||
|
||||
# 3) .jochrtf: explicitly load a compiled cache
|
||||
python main.py input.m4a --binaural `
|
||||
--compiled-hrtf-cache C:\HRTF\subject.jochrtf
|
||||
|
||||
# Optional: create/reuse a transparent disk cache for SOFA
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
|
||||
--hrtf-cache-policy disk
|
||||
```
|
||||
|
||||
The library exposes the same fields: `joc_task_config` for a file task,
|
||||
`joc_stream_config` for streaming, where `hrtf_path`, `kernels_path`,
|
||||
`binaural_mode` and `binaural_tail_seconds` configure the binaural path.
|
||||
The default order is `HRTF/binaural.sofa`, then the unique `.jochrtf` under
|
||||
`output/hrtf-cache`, then `HRTF/binaural.personalized_headphone`; if none of
|
||||
the three exist, an error asks for an explicit path. Multiple `.jochrtf` files
|
||||
under `output/hrtf-cache` are also an error requiring an explicit choice.
|
||||
|
||||
The `.jochrtf` file is an **input**, not a product of this library: compiling it from
|
||||
SOFA data or measurements belongs to the toolchain and is decoupled from this
|
||||
repository. Loading validates the member set, dtypes and shapes, C-contiguity,
|
||||
CRC-32 and a payload hash recomputed over the members (see `.jochrtf` below).
|
||||
The `.personalized_headphone` JSON parsing is implemented by this project
|
||||
(`src/rosella_model.py`) and does not invoke any Dolby software.
|
||||
|
||||
`binaural_mode` is `near`, `mid` or `far` and selects one of the backend's three
|
||||
preset parameter sets; `binaural_tail_seconds` sets the room-tail length drained on
|
||||
flush (default 5.0 s).
|
||||
`--hrtf-cache-policy` accepts `none`, `memory`, or `disk`. The default is
|
||||
`memory`; neither `none` nor `memory` creates a file. `disk` writes to
|
||||
`output/hrtf-cache` by default, or to `--hrtf-cache-dir`. `--hrtf-radius-m`
|
||||
selects the nearest measurement-radius shell.
|
||||
|
||||
The Python API also uses explicit factories:
|
||||
|
||||
```python
|
||||
from sofa_binaural_backend import SofaBinauralBackend
|
||||
|
||||
renderer = SofaBinauralBackend.from_sofa(
|
||||
"subject.sofa",
|
||||
source_count=16,
|
||||
default_profile="mid",
|
||||
cache_policy="memory",
|
||||
)
|
||||
|
||||
cached = SofaBinauralBackend.from_compiled_cache(
|
||||
"subject.jochrtf",
|
||||
source_count=16,
|
||||
default_profile="mid",
|
||||
)
|
||||
```
|
||||
|
||||
The factories never guess a format from an unknown suffix: SOFA and `.jochrtf`
|
||||
always use distinct loaders.
|
||||
|
||||
## Binaural render mode
|
||||
|
||||
@@ -222,10 +254,16 @@ late sends, the unitary FDN, the 120–180 Hz cosine-squared LFE low-pass, and r
|
||||
calibration are JOC project-defined behavior, not constants published by SOFA or
|
||||
Dolby.
|
||||
|
||||
The binaural renderer is implemented inside this library: the filterbank, the SH
|
||||
directional-field evaluation, the per-object early/direct histories and the shared
|
||||
FDN all run in `joc_core` (the `ejoc_sofa_binaural_*` kernel), consuming a compiled
|
||||
directional field and 512-sample metadata updates.
|
||||
The public SOFA binaural renderer defaults to the C++20 native core under
|
||||
`--backend auto/native` (`ejoc_sofa_binaural_*` in `lib/eac3joc_core.dll`): the
|
||||
filterbank, the SH direction-field evaluation, the per-object direct/early
|
||||
histories and the shared FDN all run natively, while Python only compiles the
|
||||
SOFA source and issues the per-512-sample metadata updates. When the native
|
||||
library is unavailable the renderer falls back to the Python/NumPy reference
|
||||
implementation; the two agree to better than 1e-9. `--backend python` forces
|
||||
the Python backend.
|
||||
`--backend` still selects native/Python JOC reconstruction and speaker rendering;
|
||||
native acceleration for the public binaural DSP is outside the current API.
|
||||
|
||||
## Technical references and rights boundary
|
||||
|
||||
|
||||
+50
-16
@@ -21,25 +21,57 @@ SOFA FIR
|
||||
|
||||
## 输入接口
|
||||
|
||||
双耳后端消费一个已编译的方向场(JOC compiled HRTF cache,`.jochrtf`)与共享滤波器组表;
|
||||
两者都是只读输入,本库不做任何测量数据格式的解析或转换:
|
||||
CLI 有三个互斥的 HRTF 输入来源;都不指定时按默认规则自动选择:
|
||||
|
||||
```powershell
|
||||
joc_cli input.m4a --binaural `
|
||||
--compiled-hrtf-cache path\to\subject.jochrtf `
|
||||
--kernels data\rosella_kernels.npz `
|
||||
--binaural-mode mid --binaural-tail-seconds 5.0
|
||||
# 1) SOFA:缺省取 HRTF/binaural.sofa,也可显式指定
|
||||
python main.py input.m4a --binaural
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa
|
||||
|
||||
# 2) Rosella .personalized_headphone:缺省取 HRTF/binaural.personalized_headphone
|
||||
python main.py input.m4a --binaural --personalized-headphone
|
||||
python main.py input.m4a --binaural --personalized-headphone C:\HRTF\subject.personalized_headphone
|
||||
|
||||
# 3) .jochrtf:显式读取预编译 cache
|
||||
python main.py input.m4a --binaural `
|
||||
--compiled-hrtf-cache C:\HRTF\subject.jochrtf
|
||||
|
||||
# 可选:SOFA 透明生成/复用磁盘 cache
|
||||
python main.py input.m4a --binaural --sofa-hrtf C:\HRTF\subject.sofa `
|
||||
--hrtf-cache-policy disk
|
||||
```
|
||||
|
||||
库接口使用同一组字段:文件任务用 `joc_task_config`,流式用 `joc_stream_config`,
|
||||
其中 `hrtf_path`、`kernels_path`、`binaural_mode`、`binaural_tail_seconds` 决定双耳通路。
|
||||
默认选择顺序:`HRTF/binaural.sofa` → `output/hrtf-cache` 下唯一的 `.jochrtf` →
|
||||
`HRTF/binaural.personalized_headphone`;三者都没有时报错并提示显式指定。
|
||||
`output/hrtf-cache` 下有多个 `.jochrtf` 时同样报错,要求显式选择。
|
||||
|
||||
`.jochrtf` 是**输入**而不是本库的产物:从 SOFA 或测量数据编译该缓存属于工具链的职责,
|
||||
与本仓库解耦。缓存加载时会校验成员集合、dtype 与形状、C 连续性、CRC-32 以及按成员重算的
|
||||
载荷哈希(见下文 `.jochrtf` 一节)。
|
||||
`.personalized_headphone` 的 JSON 解析由本项目自行实现(`src/rosella_model.py`),
|
||||
不调用任何杜比软件。
|
||||
|
||||
`binaural_mode` 取 `near`、`mid`、`far`,选择后端的三组预置参数;`binaural_tail_seconds`
|
||||
决定 flush 时排空的房间尾音长度(默认 5.0 s)。
|
||||
`--hrtf-cache-policy` 可取 `none`、`memory`、`disk`。默认是 `memory`;`none` 和
|
||||
`memory` 都不会创建磁盘文件。`disk` 默认写入 `output/hrtf-cache`,也可用
|
||||
`--hrtf-cache-dir` 指定。`--hrtf-radius-m` 选择距离目标最近的 measurement shell。
|
||||
|
||||
Python API 使用显式 factory:
|
||||
|
||||
```python
|
||||
from sofa_binaural_backend import SofaBinauralBackend
|
||||
|
||||
renderer = SofaBinauralBackend.from_sofa(
|
||||
"subject.sofa",
|
||||
source_count=16,
|
||||
default_profile="mid",
|
||||
cache_policy="memory",
|
||||
)
|
||||
|
||||
cached = SofaBinauralBackend.from_compiled_cache(
|
||||
"subject.jochrtf",
|
||||
source_count=16,
|
||||
default_profile="mid",
|
||||
)
|
||||
```
|
||||
|
||||
文件工厂不会按“未知后缀”猜格式:SOFA 和 `.jochrtf` 始终走不同 loader。
|
||||
|
||||
## 双耳渲染模式
|
||||
|
||||
@@ -192,9 +224,11 @@ Near/Mid/Far、equal-power direct level、六面 shoebox 一阶 image source、l
|
||||
unitary FDN、LFE 120–180 Hz cosine-squared 低通及 room calibration 都是 JOC
|
||||
项目定义行为,不是 SOFA 或 Dolby 公布常数。
|
||||
|
||||
双耳渲染在本库内实现:filterbank、SH 方向场求值、逐对象 early/direct 历史与共享 FDN
|
||||
全部执行于 `joc_core`(`ejoc_sofa_binaural_*` 内核),对外只消费编译好的方向场与
|
||||
512-sample 粒度的元数据更新。
|
||||
公开 SOFA 双耳渲染在 `--backend auto/native` 下默认走 C++20 原生核
|
||||
(`lib/eac3joc_core.dll` 的 `ejoc_sofa_binaural_*` 接口:filterbank、SH 方向场求值、
|
||||
逐对象 early/direct 历史与共享 FDN 全部在原生侧执行,Python 只做 SOFA 编译与每
|
||||
512-sample 的元数据更新);原生库不可用时自动回退 Python/NumPy 参考实现,两者
|
||||
逐值一致(差异 < 1e-9)。`--backend python` 强制使用 Python 后端。
|
||||
|
||||
## 技术引用与权利边界
|
||||
|
||||
|
||||
@@ -0,0 +1,178 @@
|
||||
# JustOneCacophony native-core notes
|
||||
|
||||
[中文](native.md) · [Back to README](../README.en.md)
|
||||
|
||||
## 1. Responsibility boundary
|
||||
|
||||
`native/` contains only the state-heavy, frequently called DSP and speaker-rendering kernels. High-level EMDF/JOC/OAMD parsing, error reporting, ADM assembly, and the CLI remain in Python.
|
||||
|
||||
Python calls a C ABI through the standard-library `ctypes` module. The native core does not use pybind11, Cython, FFTW, MKL, or OpenMP. It is an optional acceleration path and does not expand the set of supported stream variants.
|
||||
|
||||
Main files:
|
||||
|
||||
```text
|
||||
native/include/eac3joc_core.h C ABI
|
||||
native/src/eac3joc_core.cpp JOC/QMF object reconstruction
|
||||
native/src/speaker_renderer.cpp object-to-speaker rendering
|
||||
native/src/qmf_tables.h QMF tables
|
||||
native/src/speaker_layouts.h layout tables
|
||||
native/src/joc_huffman_tables.h JOC Huffman tables
|
||||
src/native_renderer.py JOC ctypes bridge
|
||||
src/speaker_native_renderer.py speaker ctypes bridge
|
||||
```
|
||||
|
||||
## 2. JOC rendering ABI
|
||||
|
||||
An opaque renderer owns all cross-frame state. Its main call is:
|
||||
|
||||
```c
|
||||
int ejoc_renderer_process(
|
||||
ejoc_renderer_handle handle,
|
||||
const float* bed5_planar, /* [5][1536] */
|
||||
const float* lfe, /* [1536] or NULL */
|
||||
uint32_t object_mask,
|
||||
const uint8_t* n_bands, /* [15] */
|
||||
const uint8_t* n_dpoints, /* [15] */
|
||||
const uint8_t* slope_idx, /* [15] */
|
||||
const uint8_t* offset_ts, /* [15][2] */
|
||||
const double* dq, /* [15][2][5][23] */
|
||||
double clipgain,
|
||||
float phase_new,
|
||||
float output_scale,
|
||||
float* output16_planar); /* [16][1536] */
|
||||
```
|
||||
|
||||
Python performs JOC Huffman decoding, differential reconstruction, and dequantization before the call, with the dense and sparse syntaxes sharing one entry point. The native core consumes the already dequantized `dq` in double precision, and both syntaxes have the same layout at that ABI.
|
||||
|
||||
Thread control is exposed as:
|
||||
|
||||
```c
|
||||
int ejoc_renderer_set_threads(ejoc_renderer_handle handle, uint32_t total_threads);
|
||||
uint32_t ejoc_renderer_thread_count(ejoc_renderer_handle handle);
|
||||
```
|
||||
|
||||
`total_threads` includes the calling thread. Frames must be submitted sequentially to one renderer instance; the instance may parallelize work across objects and analysis channels.
|
||||
|
||||
## 3. Cross-frame state
|
||||
|
||||
Each JOC renderer stores:
|
||||
|
||||
- analysis FIFO: `double[5][9][64]`;
|
||||
- L/R/C analysis delay: `float[3][10][64]`;
|
||||
- Ls/Rs QMF delay: `complex<double>[2][10][64]`;
|
||||
- Ls/Rs band-0 FIR history: `complex<double>[2][20]`;
|
||||
- previous matrix interpolation values: `double[15][5][64]`;
|
||||
- inverse-QMF state: `double[15][640]`;
|
||||
- LFE delay: `double[1217]`.
|
||||
|
||||
This state belongs to the renderer instance. Processing cannot be arbitrarily segmented or reordered without a corresponding state checkpoint.
|
||||
|
||||
## 4. FFT, QMF, and precision
|
||||
|
||||
The native core contains a fixed 64-point radix-2 complex FFT:
|
||||
|
||||
- analysis QMF uses a forward FFT followed by division by 64;
|
||||
- inverse QMF uses the fixed reorder, rotation, and 640-value active-window state;
|
||||
- no external FFT library is called.
|
||||
|
||||
The JOC path uses:
|
||||
|
||||
- float32 core-PCM input;
|
||||
- double matrices, complex QMF, FFT, FIR, and cross-frame state;
|
||||
- float32 phase and final gain;
|
||||
- float32 16-channel object output.
|
||||
|
||||
## 5. Speaker-rendering ABI
|
||||
|
||||
The same shared library exports object-to-speaker rendering:
|
||||
|
||||
```c
|
||||
uint32_t ejoc_speaker_layout_channel_count(uint32_t speaker_bitfield);
|
||||
|
||||
ejoc_speaker_renderer_handle
|
||||
ejoc_speaker_renderer_create(uint32_t speaker_bitfield);
|
||||
|
||||
int ejoc_speaker_renderer_process(
|
||||
ejoc_speaker_renderer_handle handle,
|
||||
const float* objects16_interleaved,
|
||||
uint32_t sample_count,
|
||||
uint32_t metadata_count,
|
||||
const uint32_t* metadata_offsets,
|
||||
const uint32_t* ramp_durations,
|
||||
const uint16_t* positions_q15,
|
||||
const uint8_t* region_indices,
|
||||
const uint8_t* height_enabled,
|
||||
const double* object_gains,
|
||||
double* output_interleaved);
|
||||
```
|
||||
|
||||
Input channel 0 is LFE and channels 1–15 are objects. Each metadata entry is an object-state snapshot. `sample_count` must be a multiple of 32; unfinished gain ramps remain in the handle and continue across calls.
|
||||
|
||||
The speaker path uses float32 object input, double coordinates/gains/accumulation, and interleaved double output. Quantization to float32 or PCM24 happens when the WAV is written.
|
||||
|
||||
Supported layouts:
|
||||
|
||||
```text
|
||||
2.0 3.1 5.1 7.1 5.1.2 5.1.4 7.1.2 7.1.4 9.1.4 9.1.6
|
||||
```
|
||||
|
||||
## 6. Binaural-rendering ABI
|
||||
|
||||
The shared library provides a 512-sample float64 binaural DSP interface:
|
||||
|
||||
```c
|
||||
ejoc_binaural_renderer_handle ejoc_binaural_renderer_create(void);
|
||||
int ejoc_binaural_renderer_configure_kernels(...);
|
||||
int ejoc_binaural_renderer_configure_room(...);
|
||||
int ejoc_binaural_renderer_process(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
const double* input16_interleaved, /* [512][16] */
|
||||
const double* gains_complex, /* [16][2][77][2] */
|
||||
const double* room_sends, /* [16] */
|
||||
double output_gain,
|
||||
double* output_stereo_interleaved); /* [512][2] */
|
||||
```
|
||||
|
||||
Python parses the model, evaluates the OAMD timeline, and supplies complex gains and room sends every 512 samples. The C++ handle owns QMF, hybrid, recursive-room, and QMF-synthesis state. Inputs, state, accumulation, and output are double/complex double.
|
||||
|
||||
## 7. Building
|
||||
|
||||
The CMake definition is `native/CMakeLists.txt`. Run from the repository root:
|
||||
|
||||
```powershell
|
||||
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
|
||||
cmake --build build/cmake --config Release
|
||||
cmake --install build/cmake --config Release
|
||||
```
|
||||
|
||||
Platform runtime names:
|
||||
|
||||
```text
|
||||
Windows lib/eac3joc_core.dll
|
||||
Linux lib/libeac3joc_core.so
|
||||
macOS lib/libeac3joc_core.dylib
|
||||
```
|
||||
|
||||
The MSVC configuration uses the static CRT. Other runtime dependencies depend on the platform and toolchain and should be checked independently before publishing a prebuilt library.
|
||||
|
||||
The repository does not include native binaries by default. A prebuilt Release runtime or a locally built runtime can be placed directly under `lib/`.
|
||||
|
||||
## 8. Runtime lookup and fallback
|
||||
|
||||
Lookup order:
|
||||
|
||||
1. explicit `--native-library`;
|
||||
2. `EAC3JOC_NATIVE_LIBRARY`;
|
||||
3. the standard platform filename under `lib/`.
|
||||
|
||||
`--backend auto` falls back to NumPy when loading fails, and `--backend python` skips native discovery. The current CLI also prints the failure and falls back for `--backend native`; this existing behavior should not be read as successful native execution.
|
||||
|
||||
## 9. Implementation boundaries
|
||||
|
||||
- The native layer accepts only dense-JOC data already parsed by Python.
|
||||
- The ABI fixes a 1536-sample JOC frame, at most 15 objects, at most 23 parameter bands, and at most 2 data points.
|
||||
- The shared library and Python bridge must report the same ABI version.
|
||||
- Only the ABI and data types are specified across platforms; bit-identical float64 results are not guaranteed.
|
||||
- Private table headers under `native/src/` serve the native side only. The current repository does not include the scripts that generated those headers.
|
||||
|
||||
See the [mathematical notes](math.en.md) for the related formulas.
|
||||
+191
@@ -0,0 +1,191 @@
|
||||
# JustOneCacophony 原生核说明
|
||||
|
||||
[English](native.en.md) · [返回 README](../README.md)
|
||||
|
||||
## 1. 职责边界
|
||||
|
||||
`native/` 只承载状态密集、调用频繁的 DSP 与扬声器渲染核。EMDF/JOC/OAMD 高层解析、错误报告、ADM 组装和 CLI 保留在 Python 中。
|
||||
|
||||
Python 通过标准库 `ctypes` 调用 C ABI;原生核不使用 pybind11、Cython、FFTW、MKL 或 OpenMP。它是可选加速路径,不扩大项目所支持的码流范围。
|
||||
|
||||
主要文件:
|
||||
|
||||
```text
|
||||
native/include/eac3joc_core.h C ABI
|
||||
native/src/eac3joc_core.cpp JOC/QMF 对象重建
|
||||
native/src/speaker_renderer.cpp 对象到扬声器渲染
|
||||
native/src/qmf_tables.h QMF 表
|
||||
native/src/speaker_layouts.h 布局表
|
||||
native/src/joc_huffman_tables.h JOC Huffman 表
|
||||
src/native_renderer.py JOC ctypes 桥
|
||||
src/speaker_native_renderer.py 扬声器 ctypes 桥
|
||||
```
|
||||
|
||||
## 2. JOC 渲染 ABI
|
||||
|
||||
一个 opaque renderer 保存所有跨帧状态。主要调用为:
|
||||
|
||||
```c
|
||||
int ejoc_renderer_process(
|
||||
ejoc_renderer_handle handle,
|
||||
const float* bed5_planar, /* [5][1536] */
|
||||
const float* lfe, /* [1536] or NULL */
|
||||
uint32_t object_mask,
|
||||
const uint8_t* n_bands, /* [15] */
|
||||
const uint8_t* n_dpoints, /* [15] */
|
||||
const uint8_t* slope_idx, /* [15] */
|
||||
const uint8_t* offset_ts, /* [15][2] */
|
||||
const double* dq, /* [15][2][5][23] */
|
||||
double clipgain,
|
||||
float phase_new,
|
||||
float output_scale,
|
||||
float* output16_planar); /* [16][1536] */
|
||||
```
|
||||
|
||||
JOC 的 Huffman 解码、差分还原和去量化先在 Python 中完成,dense 与 sparse 两条语法共用同一条入口。原生核心消费已去量化的 `dq`(double),两条语法在该 ABI 上布局一致。
|
||||
|
||||
线程接口为:
|
||||
|
||||
```c
|
||||
int ejoc_renderer_set_threads(ejoc_renderer_handle handle, uint32_t total_threads);
|
||||
uint32_t ejoc_renderer_thread_count(ejoc_renderer_handle handle);
|
||||
```
|
||||
|
||||
`total_threads` 包含调用线程。单个 renderer 实例必须顺序提交帧;实例内部可以按对象和 analysis channel 并行。
|
||||
|
||||
## 3. 跨帧状态
|
||||
|
||||
每个 JOC renderer 独立保存:
|
||||
|
||||
- analysis FIFO:`double[5][9][64]`;
|
||||
- L/R/C analysis delay:`float[3][10][64]`;
|
||||
- Ls/Rs QMF delay:`complex<double>[2][10][64]`;
|
||||
- Ls/Rs band-0 FIR history:`complex<double>[2][20]`;
|
||||
- 矩阵插值 previous:`double[15][5][64]`;
|
||||
- inverse-QMF state:`double[15][640]`;
|
||||
- LFE delay:`double[1217]`。
|
||||
|
||||
这些状态属于 renderer 实例,不能在无 checkpoint 的情况下任意分段或乱序处理。
|
||||
|
||||
## 4. FFT、QMF 与精度
|
||||
|
||||
原生核包含固定 64 点 radix-2 complex FFT:
|
||||
|
||||
- analysis QMF 使用 forward FFT 后除以 64;
|
||||
- inverse QMF 使用固定重排、旋转和 640 项有效窗状态;
|
||||
- 不调用外部 FFT 库。
|
||||
|
||||
JOC 路径的数值类型为:
|
||||
|
||||
- 核心 PCM 输入:float32;
|
||||
- 矩阵、复 QMF、FFT、FIR 和跨帧状态:double;
|
||||
- phase 与最终 gain:float32;
|
||||
- 16 声道对象输出:float32。
|
||||
|
||||
## 5. 扬声器渲染 ABI
|
||||
|
||||
同一个共享库还导出对象到扬声器布局的渲染接口:
|
||||
|
||||
```c
|
||||
uint32_t ejoc_speaker_layout_channel_count(uint32_t speaker_bitfield);
|
||||
|
||||
ejoc_speaker_renderer_handle
|
||||
ejoc_speaker_renderer_create(uint32_t speaker_bitfield);
|
||||
|
||||
int ejoc_speaker_renderer_process(
|
||||
ejoc_speaker_renderer_handle handle,
|
||||
const float* objects16_interleaved,
|
||||
uint32_t sample_count,
|
||||
uint32_t metadata_count,
|
||||
const uint32_t* metadata_offsets,
|
||||
const uint32_t* ramp_durations,
|
||||
const uint16_t* positions_q15,
|
||||
const uint8_t* region_indices,
|
||||
const uint8_t* height_enabled,
|
||||
const double* object_gains,
|
||||
double* output_interleaved);
|
||||
```
|
||||
|
||||
输入声道 0 为 LFE,1–15 为对象。每个 metadata entry 是一份对象状态快照。`sample_count` 必须是 32 的倍数;未完成的增益斜坡保存在 handle 中并跨调用继续。
|
||||
|
||||
扬声器路径使用 float32 对象输入、double 坐标/增益/累加与 interleaved double 输出;写 WAV 时才量化为 float32 或 PCM24。
|
||||
|
||||
支持的布局为:
|
||||
|
||||
```text
|
||||
2.0 3.1 5.1 7.1 5.1.2 5.1.4 7.1.2 7.1.4 9.1.4 9.1.6
|
||||
```
|
||||
|
||||
## 6. 双耳渲染 ABI
|
||||
|
||||
共享库提供 512-sample float64 双耳 DSP:
|
||||
|
||||
```c
|
||||
ejoc_binaural_renderer_handle ejoc_binaural_renderer_create(void);
|
||||
int ejoc_binaural_renderer_configure_kernels(...);
|
||||
int ejoc_binaural_renderer_configure_room(...);
|
||||
int ejoc_binaural_renderer_process(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
const double* input16_interleaved, /* [512][16] */
|
||||
const double* gains_complex, /* [16][2][77][2] */
|
||||
const double* room_sends, /* [16] */
|
||||
double output_gain,
|
||||
double* output_stereo_interleaved); /* [512][2] */
|
||||
```
|
||||
|
||||
Python 负责模型解析、OAMD 时间轴和每 512 samples 的 complex gains/room sends。C++ handle 保存 QMF、hybrid、递归 room 和 QMF synthesis 状态。全部输入、状态、乘加和输出均为 double/complex double。
|
||||
|
||||
## 6.1 公开 SOFA 双耳渲染 ABI
|
||||
|
||||
共享库同时提供完整的原生 SOFA 双耳渲染器(`ejoc_sofa_binaural_*`),它镜像
|
||||
Python `SofaBinauralBackend` 的全部数学:64-QMF/77-hybrid analysis/synthesis、
|
||||
五阶 ACN/N3D 实球谐方向场求值、whole-QMF-slot 逐对象 delay 历史、六面一阶
|
||||
image-source early reflections、共享 unitary FDN late room、LFE 120–180 Hz
|
||||
低通与 961-sample latency 语义。kernel 表、编译好的 HRTF 场与房间常数通过
|
||||
`configure_kernels/configure_field/configure_room` 一次上传;每 512-sample
|
||||
block 先 `set_source` 更新 16 个 source,再 `process` 输入 PCM;`process` 返回
|
||||
裁剪后的 stereo 样本数(首个 961 samples 被丢弃)。`finish` 以 64-sample 对齐的
|
||||
块排空尾音。Python 桥位于 `src/sofa_native_backend.py`,与 Python 参考实现逐值
|
||||
一致(差异 < 1e-9);原生库缺失时 `main.py` 自动回退 Python。
|
||||
|
||||
## 7. 构建
|
||||
|
||||
CMake 定义位于 `native/CMakeLists.txt`。从仓库根目录运行:
|
||||
|
||||
```powershell
|
||||
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
|
||||
cmake --build build/cmake --config Release
|
||||
cmake --install build/cmake --config Release
|
||||
```
|
||||
|
||||
平台运行库文件名:
|
||||
|
||||
```text
|
||||
Windows lib/eac3joc_core.dll
|
||||
Linux lib/libeac3joc_core.so
|
||||
macOS lib/libeac3joc_core.dylib
|
||||
```
|
||||
|
||||
MSVC 配置使用静态 CRT。其他运行时依赖由平台和工具链决定,发布预构建库前应对产物独立检查。
|
||||
|
||||
仓库默认不附带原生二进制。预构建的 Release 运行库或自行构建的运行库均可直接放入 `lib/`。
|
||||
|
||||
## 8. 运行时查找与回退
|
||||
|
||||
查找顺序为:
|
||||
|
||||
1. 显式 `--native-library`;
|
||||
2. `EAC3JOC_NATIVE_LIBRARY`;
|
||||
3. `lib/` 下当前平台的标准文件名。
|
||||
|
||||
`--backend auto` 在加载失败时回退到 NumPy;`--backend python` 跳过原生探测。`--backend native` 当前也会打印失败原因后回退,这是现有 CLI 行为,不应理解为原生库已成功使用。
|
||||
|
||||
## 9. 实现边界
|
||||
|
||||
- 原生层只接收 Python 已解析的 dense JOC 数据。
|
||||
- ABI 固定了 1536-sample JOC 帧、最多 15 个对象、最多 23 个参数带和最多 2 个数据点。
|
||||
- 共享库与 Python 桥需要 ABI version 一致。
|
||||
- 跨平台只约定 ABI 与数据类型,不保证 float64 结果逐位一致。
|
||||
- `native/src/` 中的私有表头只服务于原生侧;当前仓库不包含重新生成这些头文件的脚本。
|
||||
|
||||
相关公式见[数学说明](math.md)。
|
||||
-128
@@ -1,128 +0,0 @@
|
||||
# SIMD and runtime dispatch
|
||||
|
||||
[中文](simd.md) · [Back to README](../README.en.md)
|
||||
|
||||
The heaviest loops in the binaural path (QMF analysis and synthesis, the hybrid
|
||||
analysis low join, hybrid-domain path rendering, the ROOM FFT, the spherical
|
||||
harmonic alignment) each have a runtime-dispatched vector implementation: one
|
||||
binary carries several instruction-set variants, asks the CPU once at startup and
|
||||
runs the widest one. **The output is byte-identical either way** — that is a hard
|
||||
constraint, not a goal.
|
||||
|
||||
```text
|
||||
JOC_SIMD=auto|scalar|sse2|avx2|avx512|neon pin one tier (used for verification)
|
||||
JOC_SIMD_LOG=1 report the ISA each kernel actually got
|
||||
```
|
||||
|
||||
## Why the split has to happen per translation unit
|
||||
|
||||
MSVC has no function-level attribute like `__attribute__((target("avx2")))`: one
|
||||
`.cpp` file gets one `/arch`. So every ISA is its own translation unit with its own
|
||||
`/arch:AVX2` or `/arch:AVX512` (GCC/Clang: `-mavx2` / `-mavx512f`), and
|
||||
`dispatch.cpp` fills the function table at run time. The baseline units — the
|
||||
dispatcher itself, the CPU probe and the scalar reference — carry **no** `/arch` at
|
||||
all and stay on the SSE2 that x86-64 guarantees.
|
||||
|
||||
A trap from the history of this tree: `JOC_ENABLE_AVX2` used to be global, so
|
||||
turning it on put AVX2 instructions into the very code paths that exist for older
|
||||
CPUs. It now only selects whether the AVX2 unit is compiled in.
|
||||
|
||||
## Directory layout
|
||||
|
||||
One flat directory, **the instruction set in the file name and never in a
|
||||
subdirectory** — that is how FLAC does it (`lpc.c` sits next to
|
||||
`lpc_intrin_sse2.c`, `lpc_intrin_avx2.c` and `lpc_intrin_neon.c`, with the CPU
|
||||
probe in its own `cpu.c`).
|
||||
|
||||
```text
|
||||
src/simd/
|
||||
simd.h the contract: Isa / Kernel / Kernels / dimensions
|
||||
cpu_probe.{h,cpp} "can this machine run ISA X": CPUID+XGETBV / __builtin_cpu_supports / getauxval
|
||||
dispatch.cpp policy: JOC_SIMD, the fallback ladder, the table, the log
|
||||
kernels_scalar.cpp the reference (Isa::scalar; always built, always selectable)
|
||||
kernels_intrin_avx2.cpp /arch:AVX2 -mavx2
|
||||
kernels_intrin_avx512.cpp /arch:AVX512 -mavx512f
|
||||
kernels_intrin_neon.cpp AArch64 default
|
||||
```
|
||||
|
||||
The module lives at `src/simd/`, not `src/dsp/simd/`: `foundation/`, `hrtf/` and
|
||||
`binaural/` all call into these kernels, so it is a cross-cutting layer rather than
|
||||
a submodule of the DSP code (and `src/dsp/` held nothing else).
|
||||
|
||||
The header comment of `simd.h` carries the same module map; keep the two in sync
|
||||
when the layout changes.
|
||||
|
||||
## Three rules
|
||||
|
||||
1. **Only `kernels_intrin_*.cpp` gets a wider flag.** `CMakeLists.txt` names those
|
||||
files explicitly with `set_source_files_properties`; every other target stays on
|
||||
the architecture's guaranteed ISA. A unit that goes wide without matching that
|
||||
name fails `devtools/vec/isa_audit.ps1`.
|
||||
2. **No dynamic initialisation inside an ISA unit.** Those objects are linked into
|
||||
the same image as the baseline, so a global constructor would execute a wide
|
||||
instruction before the dispatcher has looked at the CPU. Constant tables are
|
||||
fine — they land in `.rdata`.
|
||||
3. **Bit-exactness comes from the lane assignment, not from the ISA.** A lane may
|
||||
only carry mutually independent outputs; the rounding sequence of a single
|
||||
output, the separation of multiply and add (never an FMA) and the summation
|
||||
order all stay exactly as `kernels_scalar.cpp` wrote them. Layout changes that
|
||||
only reorder stored doubles (rank-minor basis tables, term-ordered tap tables,
|
||||
stage-contiguous twiddle tables, band-major ROOM planes) are allowed.
|
||||
|
||||
## How the choice is made
|
||||
|
||||
`dispatch.cpp` parses `JOC_SIMD` first (forcing a tier this build or this machine
|
||||
does not have prints a diagnostic and falls back, rather than pretending and
|
||||
crashing), then walks `avx512 → avx2 → sse2 → neon` and picks, per kernel, the
|
||||
widest implementation that is both compiled into this binary **and** runnable
|
||||
here, falling back to the scalar reference. An AVX-512 unit therefore costs
|
||||
nothing on a CPU without AVX-512; it is simply never selected.
|
||||
|
||||
On x86 the probe requires CPUID *and* XGETBV to agree: CPUID says the silicon can
|
||||
do it, XCR0 says the OS saves the registers it needs. Either one alone is not
|
||||
enough — using AVX without OS state support corrupts other threads across a
|
||||
context switch. AArch64 needs no probe; ASIMD is the architectural baseline.
|
||||
|
||||
`sse2` is a selectable tier with **no unit of its own**, on purpose: a 128-bit SSE2
|
||||
register is the register a scalar double already occupies, SSE2 cannot widen
|
||||
double-precision arithmetic, and hand-written SSE2 would only add moves. The tier
|
||||
resolves to the baseline unit.
|
||||
|
||||
## Effect
|
||||
|
||||
30-second reference cases, one binary with only `JOC_SIMD` switched (DSP stage,
|
||||
`t_render_dsp`):
|
||||
|
||||
| Case | `scalar` | `auto` | DSP speed-up | End-to-end wall clock |
|
||||
|---|---|---|---|---|
|
||||
| Binaural Rosella | 1.256 s | **0.640 s** | **1.96×** | 1.690 → **0.941 s** |
|
||||
| Binaural SOFA | 1.560 s | **0.654 s** | **2.39×** | 1.859 → **0.849 s** |
|
||||
| Speaker 5.1 / 9.1.6 / ADM | — | — | 1.00× | no regression (these kernels are not on those paths) |
|
||||
|
||||
The vectorised loops themselves gain more: synthesis basis 5.89×, 13-tap low join
|
||||
4.50×, SOFA QMF synthesis 4.27×, 128-point FFT 2.24×. The whole pipeline stops
|
||||
short of 8× because a good part of the time goes to parameter setup, straight
|
||||
copies and file writing — none of which has independent work items — and because
|
||||
SOFA's 33 M sin/cos calls per sample cannot be vectorised under a byte-exactness
|
||||
contract.
|
||||
|
||||
## Verifying a change
|
||||
|
||||
```powershell
|
||||
$env:JOC_SIMD_LOG='1' # per-kernel ISA on this machine
|
||||
$env:JOC_SIMD='scalar' # force the reference: hashes must not move
|
||||
pwsh -NoProfile -File devtools\vec\isa_audit.ps1 # disassemble every .obj: 0 unguarded wide instructions
|
||||
pwsh -NoProfile -File devtools\vec\sha_matrix.ps1 # 5 tiers x 2 renders against the reference digests
|
||||
cmd /c devtools\vec\build_kernel_probe.bat # per-kernel byte digests (8 kernels)
|
||||
```
|
||||
|
||||
## Adding an ISA
|
||||
|
||||
1. Write `kernels_intrin_<isa>.cpp`, implementing the slots you have and leaving the
|
||||
rest `nullptr` — the dispatcher falls back per kernel (that is how
|
||||
`qmf_synthesis_basis` is handled in the AVX-512 unit).
|
||||
2. Add it to `JOC_SIMD_SOURCES` in `CMakeLists.txt` with `JOC_SIMD_HAVE_<ISA>=1` and
|
||||
its flag, and extend `Isa`, `isa_rank`, `isa_compiled`, `isa_supported` and the
|
||||
`JOC_SIMD` name table in `simd.h` / `dispatch.cpp`.
|
||||
3. Verify: the kernel digests must match the scalar unit byte for byte, and the
|
||||
reference renders must keep their SHA-256 on every tier.
|
||||
-107
@@ -1,107 +0,0 @@
|
||||
# SIMD 与运行时派发
|
||||
|
||||
[English](simd.en.md) · [返回 README](../README.md)
|
||||
|
||||
双耳通路里最重的那几段循环(QMF 分析/合成、混合分析的低频拼接、混合域路径渲染、
|
||||
ROOM 的 FFT、球谐对齐)都有一份运行时分派的向量实现:同一份二进制里装多套 ISA 代码,
|
||||
启动时问一次 CPU,然后选最宽的那套跑。**输出逐字节不变**——这是硬约束,不是目标。
|
||||
|
||||
```text
|
||||
JOC_SIMD=auto|scalar|sse2|avx2|avx512|neon 强制某一层(验收用)
|
||||
JOC_SIMD_LOG=1 打印每个 kernel 实际生效的 ISA
|
||||
```
|
||||
|
||||
## 为什么必须"按编译单元分 ISA"
|
||||
|
||||
MSVC 没有 `__attribute__((target("avx2")))` 这类函数级多版本能力,一个 .cpp 只能有
|
||||
一个 `/arch`。所以每个 ISA 一个编译单元,各自带自己的 `/arch:AVX2` / `/arch:AVX512`
|
||||
(GCC/Clang 是 `-mavx2` / `-mavx512f`),由 `dispatch.cpp` 在运行时填函数表。
|
||||
基线单元(含派发器本身、CPU 探测、标量参考实现)**不带任何 `/arch`**,它们只使用
|
||||
x86-64 架构保证的 SSE2。
|
||||
|
||||
历史坑:早先的 `JOC_ENABLE_AVX2` 是**全局**的,一旦打开,连"给老 CPU 用"的基线路径
|
||||
都带 AVX2 指令。现在这个选项只决定是否把 AVX2 单元编进二进制。
|
||||
|
||||
## 目录布局
|
||||
|
||||
一个扁平目录,**ISA 写在文件名里,不写进子目录**——这是 FLAC 的做法
|
||||
(`src/libFLAC/lpc.c` 旁边就是 `lpc_intrin_sse2.c` / `lpc_intrin_avx2.c` /
|
||||
`lpc_intrin_neon.c`,CPU 探测单独放在 `cpu.c`)。
|
||||
|
||||
```text
|
||||
src/simd/
|
||||
simd.h 唯一契约头:Isa / Kernel / Kernels / 维度常量
|
||||
cpu_probe.{h,cpp} "这台机器能不能跑 ISA X":CPUID+XGETBV / __builtin_cpu_supports / getauxval
|
||||
dispatch.cpp 策略:JOC_SIMD 解析、回退阶梯、填函数表、日志
|
||||
kernels_scalar.cpp 参考实现(Isa::scalar,永远编译、永远可选中)
|
||||
kernels_intrin_avx2.cpp /arch:AVX2 -mavx2
|
||||
kernels_intrin_avx512.cpp /arch:AVX512 -mavx512f
|
||||
kernels_intrin_neon.cpp AArch64 默认 -march=armv8-a+simd
|
||||
```
|
||||
|
||||
`simd.h` 的头注释里有一份同样的模块地图,改布局时两处一起改。
|
||||
|
||||
模块放在 `src/simd/` 而不是 `src/dsp/simd/`:这些 kernel 被 `foundation/`、`hrtf/`、
|
||||
`binaural/` 三个模块共用,是横切的一层,不是 DSP 的子模块(更何况 `src/dsp/` 里除了
|
||||
`simd/` 空无一物)。
|
||||
|
||||
## 三条规则
|
||||
|
||||
1. **只有 `kernels_intrin_*.cpp` 拿更宽的编译开关。** `CMakeLists.txt` 用
|
||||
`set_source_files_properties` 逐个点名,其余目标一律留在架构保证的 ISA 上。
|
||||
任何不属于这个命名却带了宽指令的单元都会被 `devtools/vec/isa_audit.ps1` 判失败。
|
||||
2. **ISA 单元里不许有动态初始化。** 它们和基线代码链进同一个镜像,全局构造函数会在
|
||||
派发器看 CPU 之前就跑宽指令。常量表没问题(落在 `.rdata`)。
|
||||
3. **逐位一致靠的是 lane 的划分,不是 ISA。** lane 里只能放**互相独立**的输出;单个输出
|
||||
的舍入序列、乘加分离(绝不用 FMA)、求和顺序都保持 `kernels_scalar.cpp` 原样。
|
||||
只改变 double **存放顺序**的布局改造(基函数表转秩小序、抽头表按项序、蝶形因子表
|
||||
按级连续化、ROOM 谱平面改频带主序)是允许的。
|
||||
|
||||
## 运行时怎么选
|
||||
|
||||
`dispatch.cpp` 先解析 `JOC_SIMD`(强制一个本机不支持的层会打印诊断并回退,而不是假装
|
||||
选中然后崩),再走阶梯 `avx512 → avx2 → sse2 → neon`,每个 kernel 单独挑"已编进本
|
||||
二进制 **且** 本机可跑"的最宽实现,挑不到就落到标量参考实现。所以 AVX-512 单元在
|
||||
不支持它的 CPU 上只是不被选中,不影响启动。
|
||||
|
||||
x86 的探测要 CPUID 与 XGETBV **同时**成立:CPUID 说明硅片有这个能力,XCR0 说明操作
|
||||
系统会保存对应寄存器状态,缺一个就不能用(否则上下文切换会踩坏别的线程)。AArch64
|
||||
不需要探测,ASIMD 是架构基线。
|
||||
|
||||
`sse2` 是一个有意保留的档位但**没有单独的单元**:128 位 SSE2 寄存器就是标量 double
|
||||
已经在用的寄存器,SSE2 加宽不了双精度运算,手写只会多出搬运指令,所以它选中的是基线
|
||||
单元。
|
||||
|
||||
## 效果
|
||||
|
||||
30 s 参考用例,同一二进制只切 `JOC_SIMD`(DSP 阶段 `t_render_dsp`):
|
||||
|
||||
| 用例 | `scalar` | `auto` | DSP 加速 | 端到端墙钟 |
|
||||
|---|---|---|---|---|
|
||||
| 双耳 Rosella | 1.256 s | **0.640 s** | **1.96×** | 1.690 → **0.941 s** |
|
||||
| 双耳 SOFA | 1.560 s | **0.654 s** | **2.39×** | 1.859 → **0.849 s** |
|
||||
| 扬声器 5.1 / 9.1.6 / ADM | — | — | 1.00× | 0 回归(不走这些 kernel) |
|
||||
|
||||
单看被向量化的循环,红利更大:合成基函数 5.89×、13 抽头低频拼接 4.50×、
|
||||
SOFA QMF 合成 4.27×、128 点 FFT 2.24×。整条流水线到不了 8×,是因为相当一部分时间在
|
||||
参数设置、直通拷贝、写盘这些没有独立工作项的代码上,以及 SOFA 每次采样的 33 M 次
|
||||
sin/cos 按逐位契约不能向量化。
|
||||
|
||||
## 验证
|
||||
|
||||
```powershell
|
||||
$env:JOC_SIMD_LOG='1' # 本机每个 kernel 实际选中的 ISA
|
||||
$env:JOC_SIMD='scalar' # 强制参考路径:哈希必须一动不动
|
||||
pwsh -NoProfile -File devtools\vec\isa_audit.ps1 # 反汇编全部 .obj:0 个无守卫的宽指令
|
||||
pwsh -NoProfile -File devtools\vec\sha_matrix.ps1 # 5 档 × 2 渲染,逐字节比对参考摘要
|
||||
cmd /c devtools\vec\build_kernel_probe.bat # kernel 级逐字节摘要(8 个 kernel)
|
||||
```
|
||||
|
||||
## 加一个新的 ISA
|
||||
|
||||
1. 写 `kernels_intrin_<isa>.cpp`,实现能实现的槽位,其余留 `nullptr`——派发器会逐
|
||||
kernel 回退(AVX-512 单元里的 `qmf_synthesis_basis` 就是这么处理的)。
|
||||
2. 在 `CMakeLists.txt` 里加进 `JOC_SIMD_SOURCES`、`JOC_SIMD_HAVE_<ISA>=1` 和它的编译
|
||||
开关,并在 `simd.h` / `dispatch.cpp` 里补上 `Isa`、`isa_rank`、`isa_compiled`、
|
||||
`isa_supported` 与 `JOC_SIMD` 名字表。
|
||||
3. 验证:kernel 摘要必须与标量逐字节相同,参考渲染在每个档位上的 SHA-256 都不能变。
|
||||
@@ -1,490 +0,0 @@
|
||||
/*
|
||||
* joc_core.h -- JustOneCacophony C++ Core, public ABI. Pure C.
|
||||
*
|
||||
* This header is the single authoritative definition of the Core's public
|
||||
* parameter/error surface. Frontends (thin Python CLI, joc_dump, foobar2000,
|
||||
* MPV, FFmpeg) only ever fill these POD structs and read these POD results;
|
||||
* no audio data, no internal DSP concept, and no Python type crosses this
|
||||
* boundary.
|
||||
*
|
||||
* Stability tiers
|
||||
* ---------------
|
||||
* [T1] host tier
|
||||
* joc_abi_version / joc_version_string / joc_build_info, joc_error,
|
||||
* joc_error_name / joc_error_stage, joc_last_error_detail.
|
||||
* Stable, versioned, safe for media hosts.
|
||||
*
|
||||
* [T2] bitstream / verification tier
|
||||
* joc_parse_eac3_frame, joc_parse_id14, joc_frame_params, joc_emdf_info.
|
||||
* These deliberately expose the dequantized JOC matrix coefficients so the
|
||||
* bitstream front-end can be validated bit-exactly and driven by tooling
|
||||
* (joc_dump) and A/B harnesses. Media hosts must NOT use this tier; they
|
||||
* use the task/stream API (added in later milestones).
|
||||
*
|
||||
* Conventions
|
||||
* -----------
|
||||
* * every struct's first two fields are struct_size / struct_version;
|
||||
* * all arrays are fixed size and POD, no pointers, no allocation;
|
||||
* * every function returns joc_error (JOC_OK == 0);
|
||||
* * joc_last_error_detail() returns a thread-local message that stays valid
|
||||
* until the next Core call on the same thread.
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stddef.h>
|
||||
|
||||
#define JOC_ABI_VERSION 3u
|
||||
#define JOC_FRAME_PARAMS_VERSION 1u
|
||||
#define JOC_EMDF_INFO_VERSION 1u
|
||||
#define JOC_TASK_CONFIG_VERSION 3u
|
||||
#define JOC_TASK_RESULT_VERSION 2u
|
||||
#define JOC_EVENT_VERSION 1u
|
||||
|
||||
/* JOC_STATIC: this branch exists for building the sources directly into an application, where nothing is imported or exported. */
|
||||
#if defined(JOC_STATIC)
|
||||
#define JOC_API
|
||||
#define JOC_CALL __cdecl
|
||||
#elif defined(_WIN32)
|
||||
#if defined(JOC_BUILD_DLL)
|
||||
#define JOC_API __declspec(dllexport)
|
||||
#else
|
||||
#define JOC_API __declspec(dllimport)
|
||||
#endif
|
||||
#define JOC_CALL __cdecl
|
||||
#else
|
||||
#define JOC_API __attribute__((visibility("default")))
|
||||
#define JOC_CALL
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* Fixed layout constants (single source of truth for every frontend) */
|
||||
/* ------------------------------------------------------------------ */
|
||||
enum {
|
||||
JOC_FRAME_SAMPLES = 1536, /* E-AC-3 frame = 24 * 64 */
|
||||
JOC_TIMESLOTS = 24,
|
||||
JOC_SUBBANDS = 64,
|
||||
JOC_CORE_CHANNELS = 5, /* L R C Ls Rs (JOC order) */
|
||||
JOC_MAX_CORE_CHANNELS = 7, /* dmx_config_idx 1/2/4 declare 7 */
|
||||
JOC_OUTPUT_CHANNELS = 16, /* ch0 = LFE, ch1..15 = objects */
|
||||
JOC_MAX_OBJECTS = 15,
|
||||
JOC_MAX_DPOINTS = 2,
|
||||
JOC_MAX_PARAMETER_BANDS = 23,
|
||||
JOC_MAX_EMDF_PAYLOADS = 16,
|
||||
JOC_SPEAKER_BLOCK_SAMPLES = 32,
|
||||
JOC_BINAURAL_BLOCK_SAMPLES = 512,
|
||||
JOC_BINAURAL_QMF_BANDS = 64,
|
||||
JOC_BINAURAL_HYBRID_BANDS = 77,
|
||||
JOC_QMF_HOP_SAMPLES = 64,
|
||||
JOC_BINAURAL_LATENCY_SAMPLES = 961,
|
||||
JOC_LFE_DELAY_SAMPLES = 1217
|
||||
};
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* [T1] library / error surface */
|
||||
/* ------------------------------------------------------------------ */
|
||||
typedef enum joc_error {
|
||||
JOC_OK = 0,
|
||||
JOC_ERR_INVALID_ARGUMENT,
|
||||
JOC_ERR_INVALID_CONFIG,
|
||||
JOC_ERR_OUT_OF_MEMORY,
|
||||
JOC_ERR_IO,
|
||||
JOC_ERR_UNSUPPORTED_PLATFORM,
|
||||
JOC_ERR_LIBRARY_MISSING,
|
||||
/* input / bitstream */
|
||||
JOC_ERR_INPUT_NOT_FOUND,
|
||||
JOC_ERR_INPUT_FORMAT,
|
||||
JOC_ERR_EAC3_SYNCFRAME,
|
||||
JOC_ERR_EMDF_TRANSPORT,
|
||||
JOC_ERR_EMDF_SYNTAX,
|
||||
JOC_ERR_JOC_SYNTAX,
|
||||
JOC_ERR_JOC_UNSUPPORTED_VARIANT,
|
||||
JOC_ERR_OAMD_SYNTAX,
|
||||
JOC_ERR_OAMD_UNSUPPORTED_VARIANT,
|
||||
JOC_ERR_BITSTREAM_TRUNCATED,
|
||||
JOC_ERR_BITSTREAM_PADDING,
|
||||
/* resources */
|
||||
JOC_ERR_HRTF_NOT_FOUND,
|
||||
JOC_ERR_HRTF_FORMAT,
|
||||
JOC_ERR_HRTF_VERSION,
|
||||
JOC_ERR_HRTF_HASH,
|
||||
JOC_ERR_HRTF_UNSUPPORTED_CONVENTION,
|
||||
/* rendering / output */
|
||||
JOC_ERR_LAYOUT_UNSUPPORTED,
|
||||
JOC_ERR_RENDER_FAILED,
|
||||
JOC_ERR_OUTPUT_OPEN,
|
||||
JOC_ERR_OUTPUT_WRITE,
|
||||
JOC_ERR_OUTPUT_CLIP_ABORT,
|
||||
JOC_ERR_ADM_VALIDATION,
|
||||
/* task / stream */
|
||||
JOC_ERR_CANCELLED,
|
||||
JOC_ERR_STATE,
|
||||
JOC_ERR_NOT_SUPPORTED,
|
||||
JOC_ERR_INTERNAL
|
||||
} joc_error;
|
||||
|
||||
JOC_API uint32_t JOC_CALL joc_abi_version(void);
|
||||
JOC_API const char* JOC_CALL joc_version_string(void);
|
||||
JOC_API const char* JOC_CALL joc_build_info(void);
|
||||
/* Struct sizes, so a binding can assert its layout matches the library instead of
|
||||
* assuming (mismatches are otherwise silent memory corruption). */
|
||||
JOC_API uint32_t JOC_CALL joc_event_size(void);
|
||||
JOC_API uint32_t JOC_CALL joc_task_config_size(void);
|
||||
JOC_API uint32_t JOC_CALL joc_task_result_size(void);
|
||||
JOC_API const char* JOC_CALL joc_error_name(joc_error code);
|
||||
JOC_API const char* JOC_CALL joc_error_stage(joc_error code);
|
||||
/* Thread-local structured detail for the most recent failing call. */
|
||||
JOC_API const char* JOC_CALL joc_last_error_detail(void);
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* [T2] bitstream / verification tier */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
/* One EMDF payload directory entry. bit_offset is the MSB-first bit
|
||||
* position of the payload's first byte inside the syncframe, exactly as the
|
||||
* EMDF container syntax defines it (payloads are not byte aligned in general). */
|
||||
typedef struct joc_emdf_payload_info {
|
||||
uint8_t id;
|
||||
uint8_t reserved[3];
|
||||
uint16_t sample_offset; /* EMDF outer smpoffst, 11 bits */
|
||||
uint16_t reserved2;
|
||||
uint32_t bit_offset;
|
||||
uint32_t size; /* payload bytes */
|
||||
} joc_emdf_payload_info;
|
||||
|
||||
typedef struct joc_emdf_info {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint32_t start_bit; /* container syncword bit position */
|
||||
uint32_t container_bytes; /* 4 + declared length */
|
||||
uint32_t payload_count;
|
||||
uint32_t reserved;
|
||||
joc_emdf_payload_info payloads[JOC_MAX_EMDF_PAYLOADS];
|
||||
} joc_emdf_info;
|
||||
|
||||
/* Per-object ID14/JOC descriptor plus its dequantized matrix.
|
||||
* dq[dp][ch][pb] is zero filled outside [0,n_dpoints) x [0,n_channels) x
|
||||
* [0,n_bands). Absent objects are entirely zero. */
|
||||
typedef struct joc_object_params {
|
||||
uint8_t present;
|
||||
uint8_t num_bands_idx;
|
||||
uint8_t n_bands;
|
||||
uint8_t sparse; /* 0 = dense (MTX), 1 = sparse (IDX+VEC) */
|
||||
uint8_t quant_idx; /* 0 = 96 levels, 1 = 192 levels */
|
||||
uint8_t slope_idx; /* 0 = interpolate, 1 = step at offset_ts */
|
||||
uint8_t num_dpoints_bits;
|
||||
uint8_t n_dpoints;
|
||||
uint8_t offset_ts[JOC_MAX_DPOINTS];
|
||||
uint8_t reserved[2];
|
||||
double dq[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS];
|
||||
} joc_object_params;
|
||||
|
||||
typedef struct joc_frame_params {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint8_t dmx_config_idx;
|
||||
uint8_t num_objects_bits;
|
||||
uint8_t ext_config_idx;
|
||||
uint8_t n_objects;
|
||||
uint8_t n_channels; /* 5 or 7 */
|
||||
uint8_t clipgain_x_bits;
|
||||
uint8_t clipgain_y_bits;
|
||||
uint8_t reserved;
|
||||
uint32_t seq_count; /* 10-bit JOC sequence counter, parsed only */
|
||||
uint32_t present_mask; /* bit i == object i present */
|
||||
uint32_t data_end_bits; /* bit position just after joc_data */
|
||||
uint32_t trailing_bits; /* bits left after joc_data (padding/ext) */
|
||||
uint8_t tail_bytes[8]; /* first up to 8 trailing bytes, for A/B */
|
||||
double clipgain; /* 1 + (y/32) * 2^(x-4), bit-exact */
|
||||
joc_object_params objects[JOC_MAX_OBJECTS];
|
||||
} joc_frame_params;
|
||||
|
||||
/* Parse an EMDF ID14 (JOC) payload. */
|
||||
JOC_API joc_error JOC_CALL joc_parse_id14(const uint8_t* payload, size_t payload_size,
|
||||
joc_frame_params* out_params);
|
||||
|
||||
/* Locate the contiguous JOC EMDF container inside one E-AC-3 syncframe and
|
||||
* parse its ID14 payload. out_emdf may be NULL. */
|
||||
JOC_API joc_error JOC_CALL joc_parse_eac3_frame(const uint8_t* frame, size_t frame_size,
|
||||
joc_frame_params* out_params,
|
||||
joc_emdf_info* out_emdf);
|
||||
|
||||
/* Copy one payload's bytes out of a syncframe (MSB-first bit extraction, which
|
||||
* equals a memcpy for byte-aligned containers). out_size receives the payload
|
||||
* byte count; pass out == NULL to query only the size. */
|
||||
JOC_API joc_error JOC_CALL joc_extract_payload(const uint8_t* frame, size_t frame_size,
|
||||
const joc_emdf_payload_info* payload,
|
||||
uint8_t* out, size_t out_capacity,
|
||||
size_t* out_size);
|
||||
|
||||
/* E-AC-3 syncframe traversal: given the offset of a frame start, report its
|
||||
* byte length so a host can walk a bare E-AC-3 stream without duplicating
|
||||
* frmsiz logic. Rejects resynchronisation (no silent recovery). */
|
||||
JOC_API joc_error JOC_CALL joc_eac3_frame_bytes(const uint8_t* data, size_t size,
|
||||
size_t offset, size_t* out_frame_bytes);
|
||||
|
||||
/* Strict trailing-bit check for the JOC payload (A/B robustness corpus).
|
||||
* Returns JOC_ERR_BITSTREAM_PADDING when more than 7 bits are left over or the
|
||||
* leftover bits are not zero. */
|
||||
JOC_API joc_error JOC_CALL joc_check_id14_padding(const uint8_t* payload, size_t payload_size,
|
||||
uint32_t* out_trailing_bits);
|
||||
|
||||
/* ================================================================== */
|
||||
/* [T1] host tier: task, telemetry and control */
|
||||
/* ================================================================== */
|
||||
|
||||
/* ---- 1. events ---------------------------------------------------- */
|
||||
/*
|
||||
* Events carry state only: frame/sample counters, stage, progress, statistics,
|
||||
* warnings, errors and paths. Audio never travels through an event; a fixed
|
||||
* size POD with no pointers keeps that enforceable (see the static assertion in
|
||||
* the implementation).
|
||||
*/
|
||||
typedef enum joc_event_type {
|
||||
JOC_EV_TASK_STARTED = 0x0001,
|
||||
JOC_EV_TASK_STATE_CHANGED = 0x0002,
|
||||
JOC_EV_TASK_COMPLETED = 0x0003,
|
||||
JOC_EV_TASK_FAILED = 0x0004,
|
||||
JOC_EV_TASK_CANCELLED = 0x0005,
|
||||
JOC_EV_INPUT_OPENED = 0x0101,
|
||||
JOC_EV_METADATA_INDEXED = 0x0103,
|
||||
JOC_EV_HRTF_LOADED = 0x0110,
|
||||
JOC_EV_RENDERER_INITIALIZED = 0x0120,
|
||||
JOC_EV_PROGRESS = 0x0201,
|
||||
JOC_EV_STAGE_CHANGED = 0x0202,
|
||||
JOC_EV_JOC_FRAME_STATS = 0x0301,
|
||||
JOC_EV_OAMD_STATS = 0x0302,
|
||||
JOC_EV_OUTPUT_STATS = 0x0303,
|
||||
JOC_EV_OUTPUT_OPENED = 0x0401,
|
||||
JOC_EV_OUTPUT_FORMAT_DECIDED = 0x0402,
|
||||
JOC_EV_OUTPUT_FINALIZED = 0x0403,
|
||||
JOC_EV_LOG = 0x0501,
|
||||
JOC_EV_WARNING = 0x0502,
|
||||
JOC_EV_ERROR = 0x0503
|
||||
} joc_event_type;
|
||||
|
||||
typedef enum joc_stage {
|
||||
JOC_STAGE_IDLE = 0,
|
||||
JOC_STAGE_INPUT = 1,
|
||||
JOC_STAGE_METADATA = 2,
|
||||
JOC_STAGE_DECODE = 3,
|
||||
JOC_STAGE_JOC = 4,
|
||||
JOC_STAGE_RENDER = 5,
|
||||
JOC_STAGE_OUTPUT = 6,
|
||||
JOC_STAGE_DONE = 7
|
||||
} joc_stage;
|
||||
|
||||
enum { JOC_LOG_TRACE = 0, JOC_LOG_DEBUG = 1, JOC_LOG_INFO = 2, JOC_LOG_NOTICE = 3,
|
||||
JOC_LOG_WARNING = 4, JOC_LOG_ERROR = 5, JOC_LOG_FATAL = 6 };
|
||||
|
||||
typedef struct joc_event {
|
||||
uint32_t struct_size;
|
||||
uint32_t type;
|
||||
uint64_t sequence;
|
||||
uint64_t timestamp_us;
|
||||
uint64_t current_frame;
|
||||
uint64_t total_frames;
|
||||
uint64_t current_sample;
|
||||
uint64_t total_samples;
|
||||
uint32_t stage;
|
||||
uint32_t backend;
|
||||
double progress; /* 0..1, -1 = unknown */
|
||||
double elapsed_seconds;
|
||||
double realtime_factor;
|
||||
uint64_t output_samples;
|
||||
uint64_t output_bytes;
|
||||
double output_duration_seconds;
|
||||
joc_error error_code;
|
||||
uint32_t log_level;
|
||||
char stage_name[32];
|
||||
char message[256];
|
||||
} joc_event;
|
||||
|
||||
typedef void (JOC_CALL *joc_event_fn)(void* user, const joc_event* event);
|
||||
|
||||
typedef struct joc_event_sink {
|
||||
uint32_t struct_size;
|
||||
joc_event_fn callback;
|
||||
void* user;
|
||||
uint32_t min_type; /* 0 = no filter */
|
||||
uint32_t max_type; /* 0 = no filter */
|
||||
} joc_event_sink;
|
||||
|
||||
/* ---- 2. cancellation ---------------------------------------------- */
|
||||
typedef struct joc_cancel_token joc_cancel_token;
|
||||
|
||||
JOC_API joc_cancel_token* JOC_CALL joc_cancel_token_create(void);
|
||||
JOC_API void JOC_CALL joc_cancel_token_request(joc_cancel_token* token);
|
||||
JOC_API int32_t JOC_CALL joc_cancel_token_is_requested(const joc_cancel_token* token);
|
||||
JOC_API void JOC_CALL joc_cancel_token_destroy(joc_cancel_token* token);
|
||||
|
||||
/* ---- 3. task configuration and results ---------------------------- */
|
||||
typedef enum joc_operation {
|
||||
JOC_OP_ADM_BWF = 0,
|
||||
JOC_OP_SPEAKER = 1,
|
||||
JOC_OP_BINAURAL = 2
|
||||
} joc_operation;
|
||||
|
||||
typedef enum joc_output_format {
|
||||
JOC_FORMAT_FLOAT32 = 0,
|
||||
JOC_FORMAT_PCM24 = 1
|
||||
} joc_output_format;
|
||||
|
||||
typedef enum joc_binaural_mode {
|
||||
JOC_BINAURAL_OFF = 0,
|
||||
JOC_BINAURAL_NEAR = 1,
|
||||
JOC_BINAURAL_FAR = 2,
|
||||
JOC_BINAURAL_MID = 3
|
||||
} joc_binaural_mode;
|
||||
|
||||
typedef enum joc_trajectory_mode {
|
||||
JOC_TRAJECTORY_COMPACT = 0,
|
||||
JOC_TRAJECTORY_DENSE64 = 1
|
||||
} joc_trajectory_mode;
|
||||
|
||||
/* What to do when an int24 WAV would clip (peak outside [-1, 1]). Mirrors the
|
||||
* reference CLI's --clip-action; ADM BWF output is always int24 and does not
|
||||
* consult this because no alternative format exists there. */
|
||||
typedef enum joc_clip_action {
|
||||
JOC_CLIP_ASK = 0, /* prompt on stdin; an error when stdin is not a terminal */
|
||||
JOC_CLIP_CONTINUE = 1, /* write int24, truncating out-of-range values */
|
||||
JOC_CLIP_FLOAT32 = 2, /* switch the output to float32 */
|
||||
JOC_CLIP_ABORT = 3 /* fail the task */
|
||||
} joc_clip_action;
|
||||
|
||||
/* Compiled-HRTF cache policy for a SOFA input; the .jochrtf itself is an
|
||||
* internal artifact of the compile step. */
|
||||
typedef enum joc_hrtf_cache_policy {
|
||||
JOC_HRTF_CACHE_NONE = 0, /* compile and discard */
|
||||
JOC_HRTF_CACHE_MEMORY = 1, /* compile and keep in this process (default) */
|
||||
JOC_HRTF_CACHE_DISK = 2 /* compile, reuse and write <cache_dir>/<name>.jochrtf */
|
||||
} joc_hrtf_cache_policy;
|
||||
|
||||
enum { JOC_TASK_F_SKIP_SHA256 = 1u, JOC_TASK_F_KEEP_INTERMEDIATE = 2u,
|
||||
JOC_TASK_F_QUIET = 4u, JOC_TASK_F_METADATA_ONLY = 8u };
|
||||
|
||||
typedef struct joc_task_config {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
|
||||
/* input */
|
||||
const char* input_path; /* .eac3/.ec3/.m4a/... */
|
||||
const char* ffmpeg_path; /* NULL = "ffmpeg" from PATH */
|
||||
const char* bed_path; /* NULL = decode the core PCM with ffmpeg */
|
||||
const char* work_dir; /* NULL = a temporary directory */
|
||||
double eac3_drc_scale; /* 0 = DRC off (reference default) */
|
||||
int32_t eac3_target_level; /* -31..0, 0 = not applied */
|
||||
|
||||
/* output */
|
||||
uint32_t operation;
|
||||
const char* output_path;
|
||||
uint32_t output_format; /* requested format for speaker/binaural */
|
||||
uint32_t flags;
|
||||
uint32_t clip_action; /* joc_clip_action; JOC_CLIP_ASK by default */
|
||||
|
||||
/* rendering */
|
||||
const char* speaker_layout_name;
|
||||
uint32_t speaker_metadata_offset; /* default 1473 */
|
||||
uint32_t binaural_mode;
|
||||
const char* hrtf_path; /* .jochrtf */
|
||||
const char* kernels_path; /* rosella_kernels.npz */
|
||||
double binaural_tail_seconds; /* default 5.0 */
|
||||
double binaural_tail_threshold; /* binaural only; 0 disables trimming */
|
||||
uint32_t binaural_chunk_frames; /* accepted for CLI parity; no effect */
|
||||
|
||||
/* ADM */
|
||||
uint32_t adm_binaural_mode; /* DBMD segment 10 encoding */
|
||||
uint32_t trajectory_mode;
|
||||
|
||||
/* generic */
|
||||
uint32_t object_delay_samples; /* default 1473 */
|
||||
double gain_db; /* default 0 */
|
||||
uint64_t duration_frames; /* 0 = the whole stream */
|
||||
uint32_t progress_interval_frames; /* default 1000 */
|
||||
uint32_t native_threads; /* 0 = automatic */
|
||||
|
||||
/* diagnostics */
|
||||
uint32_t print_metadata; /* 0 none, 1 summary, 2 per frame */
|
||||
const char* metadata_json_path; /* NULL = no JSON summary */
|
||||
|
||||
joc_cancel_token* cancel; /* optional */
|
||||
|
||||
/* Binaural HRTF input (config version 2): when hrtf_sofa_path is set the
|
||||
* library compiles it with hrtf_cache_policy / hrtf_cache_dir / hrtf_radius_m
|
||||
* and hrtf_path is unused. hrtf_path stays the advanced override that reads
|
||||
* a .jochrtf directly. */
|
||||
const char* hrtf_sofa_path; /* SOFA SimpleFreeFieldHRIR input */
|
||||
const char* personalized_headphone_path; /* Rosella .personalized_headphone input */
|
||||
const char* hrtf_cache_dir; /* disk policy directory */
|
||||
uint32_t hrtf_cache_policy; /* joc_hrtf_cache_policy */
|
||||
uint32_t reserved0;
|
||||
double hrtf_radius_m; /* SOFA measurement-radius shell */
|
||||
} joc_task_config;
|
||||
|
||||
typedef enum joc_task_status {
|
||||
JOC_TASK_OK = 0,
|
||||
JOC_TASK_FAILED = 1,
|
||||
JOC_TASK_CANCELLED = 2
|
||||
} joc_task_status;
|
||||
|
||||
typedef struct joc_task_result {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint32_t status;
|
||||
uint32_t error_code;
|
||||
char error_message[512];
|
||||
char error_stage[32];
|
||||
uint64_t input_frames;
|
||||
uint64_t output_samples;
|
||||
double duration_sec;
|
||||
uint32_t output_format_actual;
|
||||
double output_peak;
|
||||
uint64_t output_over_unity_values;
|
||||
uint64_t output_file_bytes;
|
||||
char output_sha256[65]; /* empty when skipped */
|
||||
uint64_t oamd_payloads;
|
||||
uint64_t oamd_transitions;
|
||||
double t_decode_bed;
|
||||
double t_render;
|
||||
double t_write;
|
||||
double t_total;
|
||||
double t_render_dsp; /* speaker/binaural DSP calls only; 0 for ADM */
|
||||
double t_write_file; /* disk writes including the finalize; >= t_write */
|
||||
} joc_task_result;
|
||||
|
||||
typedef struct joc_validation_issue {
|
||||
joc_error code;
|
||||
uint32_t severity; /* 0 = info, 1 = warning, 2 = error */
|
||||
char field[48];
|
||||
char message[256];
|
||||
} joc_validation_issue;
|
||||
|
||||
/* ---- 4. entry points ---------------------------------------------- */
|
||||
/* Fills `issues` (up to `capacity`) and reports how many were produced.
|
||||
* Returns JOC_OK when no *error*-severity issue was found. */
|
||||
JOC_API joc_error JOC_CALL joc_task_validate(const joc_task_config* config,
|
||||
joc_validation_issue* issues, uint32_t capacity,
|
||||
uint32_t* count);
|
||||
|
||||
/* Runs the whole task on the calling thread. `sink` may be NULL; `out` may be
|
||||
* NULL. On cancellation the partial output is removed and JOC_ERR_CANCELLED is
|
||||
* returned with out->status = JOC_TASK_CANCELLED. */
|
||||
JOC_API joc_error JOC_CALL joc_task_execute(const joc_task_config* config,
|
||||
const joc_event_sink* sink, joc_task_result* out);
|
||||
|
||||
/* Serialises the stable subset of the result as JSON (no environment fields -
|
||||
* those belong to the frontend). `needed` receives the required size including
|
||||
* the terminator; a NULL buffer queries only the size. */
|
||||
JOC_API joc_error JOC_CALL joc_task_result_to_json(const joc_task_result* result, char* buffer,
|
||||
size_t capacity, size_t* needed);
|
||||
|
||||
/* The streaming (push/pull) tier lives in "joc_stream.h" so an embedder can
|
||||
* include the narrow surface without the bitstream/verification tier. */
|
||||
|
||||
#ifdef __cplusplus
|
||||
} /* extern "C" */
|
||||
#endif
|
||||
@@ -1,136 +0,0 @@
|
||||
/*
|
||||
* joc_stream.h -- streaming (push/pull) interface of the JustOneCacophony core.
|
||||
*
|
||||
* This is the narrow, embedder-facing surface: a player or decoder component
|
||||
* includes only this header. It is the same shared library as joc_core.h, split
|
||||
* so an integrator never has to see the bitstream/verification tier.
|
||||
*
|
||||
* A stream is a stateful instance for hosts that cannot wait for a whole file:
|
||||
* the caller feeds E-AC-3 bytes and the core PCM of the same frames (or already
|
||||
* rebuilt objects16) and pulls rendered PCM as soon as it is available. It
|
||||
* mirrors the two library shapes a decoder library normally offers: this
|
||||
* push/pull form for host-owned I/O, and joc_task_execute() in joc_core.h for
|
||||
* library-owned file I/O.
|
||||
*
|
||||
* Contract:
|
||||
* - all state is instance-private, so several streams coexist;
|
||||
* - a stream is NOT thread safe: push and pull must come from one thread;
|
||||
* - rendering is stateful (matrix interpolation, gain ramps, room tail), so a
|
||||
* new position on the timeline requires decoding to continue from the start
|
||||
* of the stream; there is no seek in this version;
|
||||
* - the kernel latency is 961 samples for speaker/binaural output: the first
|
||||
* pull after two 512-sample blocks, and flush() drains the binaural tail.
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include "joc_core.h"
|
||||
|
||||
/* JOC_STATIC (building the sources directly into an application): joc_core.h above already installs the empty JOC_API. */
|
||||
#if defined(JOC_STATIC) && !defined(JOC_API)
|
||||
#define JOC_API
|
||||
#define JOC_CALL __cdecl
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
typedef struct joc_stream joc_stream;
|
||||
|
||||
typedef enum joc_stream_input {
|
||||
JOC_STREAM_IN_EAC3 = 0, /* bare E-AC-3 syncframes (the metadata stream) */
|
||||
JOC_STREAM_IN_PCM_OBJECTS16 = 1, /* 16-channel objects16, decoded by the host */
|
||||
JOC_STREAM_IN_CORE_PCM = 3 /* the 5.1 core PCM of the pushed E-AC-3 frames */
|
||||
} joc_stream_input;
|
||||
|
||||
typedef enum joc_stream_output {
|
||||
JOC_STREAM_OUT_PCM_OBJECTS16 = 0, /* planar [16][samples] float32 */
|
||||
JOC_STREAM_OUT_SPEAKER = 1, /* interleaved [samples][channels] f32 */
|
||||
JOC_STREAM_OUT_BINAURAL = 2 /* interleaved [samples][2] f32 */
|
||||
} joc_stream_output;
|
||||
|
||||
typedef struct joc_stream_config {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint32_t input;
|
||||
uint32_t output;
|
||||
const char* speaker_layout_name;
|
||||
uint32_t speaker_metadata_offset; /* default 1473 */
|
||||
uint32_t binaural_mode; /* near|mid|far; 0 means mid */
|
||||
const char* hrtf_path; /* binaural only */
|
||||
const char* kernels_path; /* binaural only */
|
||||
double binaural_tail_seconds; /* default 5.0 */
|
||||
uint32_t object_delay_samples; /* default 1473 */
|
||||
double gain_db; /* default 0 */
|
||||
uint32_t native_threads;
|
||||
uint32_t reserved;
|
||||
/* Binaural HRTF input, the same three shapes joc_task_config accepts: when
|
||||
* hrtf_sofa_path is set the library compiles it with hrtf_cache_policy /
|
||||
* hrtf_cache_dir / hrtf_radius_m and hrtf_path is unused; when
|
||||
* personalized_headphone_path is set the Rosella runtime renders instead.
|
||||
* hrtf_path stays the fallback/advanced input that reads a .jochrtf directly. */
|
||||
const char* hrtf_sofa_path; /* SOFA SimpleFreeFieldHRIR input */
|
||||
const char* personalized_headphone_path; /* Rosella .personalized_headphone input */
|
||||
const char* hrtf_cache_dir; /* disk cache directory for SOFA compilation */
|
||||
uint32_t hrtf_cache_policy; /* joc_hrtf_cache_policy: 0 none, 1 memory, 2 disk */
|
||||
double hrtf_radius_m; /* SOFA measurement-radius shell, default 1.0 */
|
||||
} joc_stream_config;
|
||||
|
||||
typedef struct joc_stream_buffer {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint32_t kind; /* which joc_stream_input/output this buffer carries */
|
||||
uint32_t channels;
|
||||
uint32_t sample_rate;
|
||||
uint32_t sample_count; /* in: capacity / out: produced (per channel) */
|
||||
uint32_t byte_count; /* in: capacity / out: consumed or produced bytes */
|
||||
uint32_t reserved;
|
||||
const uint8_t* bytes; /* EAC3 input */
|
||||
const float* pcm; /* PCM input */
|
||||
uint8_t* out_bytes; /* reserved for encoded outputs */
|
||||
float* out_pcm; /* PCM output */
|
||||
} joc_stream_buffer;
|
||||
|
||||
typedef struct joc_stream_status_info {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint64_t frames_in;
|
||||
uint64_t frames_out;
|
||||
uint64_t samples_in;
|
||||
uint64_t samples_out;
|
||||
uint64_t bytes_in;
|
||||
uint64_t buffered_samples; /* rendered but not yet pulled, per channel */
|
||||
uint64_t oamd_payloads;
|
||||
uint64_t oamd_transitions;
|
||||
uint32_t output_channels;
|
||||
uint32_t ended; /* 1 after flush() */
|
||||
} joc_stream_status_info;
|
||||
|
||||
JOC_API joc_error JOC_CALL joc_stream_create(const joc_stream_config* config, joc_stream** out);
|
||||
|
||||
/* Feeds one buffer. Kind selects the path: EAC3 bytes, the core PCM of those
|
||||
* frames, or objects16. Consumed counts are reported so a caller can resume
|
||||
* from a partial push. */
|
||||
JOC_API joc_error JOC_CALL joc_stream_push(joc_stream* stream, const joc_stream_buffer* input,
|
||||
uint32_t* consumed_samples, uint32_t* consumed_bytes);
|
||||
|
||||
/* Copies as many rendered samples as fit into `output` (interleaved, or planar
|
||||
* for PCM_OBJECTS16) and reports how many were produced; 0 means "push more". */
|
||||
JOC_API joc_error JOC_CALL joc_stream_pull(joc_stream* stream, joc_stream_buffer* output,
|
||||
uint32_t* produced_samples);
|
||||
|
||||
/* Marks the end of input and drains whatever the renderer still holds (the
|
||||
* binaural room tail); pull the remaining samples afterwards. */
|
||||
JOC_API joc_error JOC_CALL joc_stream_flush(joc_stream* stream);
|
||||
|
||||
/* Returns the instance to its initial state (kernel, ramps, timeline, room,
|
||||
* parser, counters) so the same stream can be reused for another pass. */
|
||||
JOC_API joc_error JOC_CALL joc_stream_reset(joc_stream* stream);
|
||||
|
||||
JOC_API joc_error JOC_CALL joc_stream_status(const joc_stream* stream,
|
||||
joc_stream_status_info* out);
|
||||
JOC_API joc_error JOC_CALL joc_stream_destroy(joc_stream* stream);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} /* extern "C" */
|
||||
#endif
|
||||
@@ -0,0 +1,998 @@
|
||||
"""JustOneCacophony 的 E-AC-3 JOC 命令行入口。"""
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
from pathlib import Path
|
||||
import platform
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
|
||||
PROJECT_DIR = Path(__file__).resolve().parent
|
||||
SOURCE_DIR = PROJECT_DIR / "src"
|
||||
if str(SOURCE_DIR) not in sys.path:
|
||||
sys.path.insert(0, str(SOURCE_DIR))
|
||||
|
||||
import numpy as np
|
||||
|
||||
import adm_assemble
|
||||
import adm_atmos
|
||||
from adm_validate import validate
|
||||
from metadata import DirectPayloadIndex, PayloadIndex, write_summary
|
||||
import oamd_tracks
|
||||
from renderer import JocRenderer
|
||||
from native_renderer import NativeBackendUnavailable, NativeJocRenderer
|
||||
from binaural_renderer import (
|
||||
DEFAULT_SOFA_HRTF,
|
||||
SofaBinauralRenderer,
|
||||
resolve_compiled_hrtf_cache,
|
||||
resolve_sofa_hrtf,
|
||||
)
|
||||
from rosella_binaural_renderer import (
|
||||
DEFAULT_PERSONALIZED_HEADPHONE,
|
||||
ROSSELLA_BLOCK_SAMPLES,
|
||||
ROSSELLA_LATENCY_SAMPLES,
|
||||
RosellaBinauralRenderer,
|
||||
resolve_personalized_headphone,
|
||||
)
|
||||
from sofa_hrtf_field import DEFAULT_HRTF_CACHE_DIR
|
||||
from speaker_backend import create_speaker_renderer
|
||||
from speaker_layouts import (SPEAKER_LAYOUT_CHOICES, get_speaker_layout,
|
||||
speaker_layout_display_name)
|
||||
from speaker_wav import BinauralPcmSpool, SpeakerPcmSpool, write_pcm_wav
|
||||
from variant_error import UnsupportedVariantError, write_variant_report
|
||||
|
||||
|
||||
RATE = 48000
|
||||
FRAME_SAMPLES = 1536
|
||||
DEFAULT_OUTPUT_DIR = PROJECT_DIR / "output"
|
||||
EAC3_DRC_SCALE_MAX = 6.0
|
||||
EAC3_TARGET_LEVEL_RANGE = (-31, 0)
|
||||
EAC3_DECODER_OPTION_RE = re.compile(r"(?m)^\s*-([A-Za-z0-9_]+)\s+<")
|
||||
|
||||
|
||||
def resolve_output(source, requested=None, speaker_layout=None, *, binaural=False):
|
||||
"""解析成品路径;未指定时使用项目内的 ``output`` 目录。"""
|
||||
source = Path(source)
|
||||
if requested is not None:
|
||||
target = Path(requested)
|
||||
elif speaker_layout is not None:
|
||||
target = DEFAULT_OUTPUT_DIR / f"{source.stem}.{speaker_layout}.wav"
|
||||
elif binaural:
|
||||
target = DEFAULT_OUTPUT_DIR / f"{source.stem}.binaural.wav"
|
||||
else:
|
||||
target = DEFAULT_OUTPUT_DIR / (source.stem + ".adm.wav")
|
||||
return target.expanduser().resolve()
|
||||
|
||||
|
||||
def _find_default_compiled_hrtf_cache():
|
||||
"""在默认 cache 目录寻找唯一的 .jochrtf;无文件返回 None,多个则报错。"""
|
||||
directory = DEFAULT_HRTF_CACHE_DIR
|
||||
if not directory.is_dir():
|
||||
return None
|
||||
candidates = sorted(directory.glob("*.jochrtf"))
|
||||
if not candidates:
|
||||
return None
|
||||
if len(candidates) > 1:
|
||||
listing = ", ".join(path.name for path in candidates[:8])
|
||||
raise ValueError(
|
||||
f"{directory} 下有多个 .jochrtf 缓存({listing}…),无法自动选择;"
|
||||
"请用 --compiled-hrtf-cache PATH 或 --sofa-hrtf PATH 显式指定")
|
||||
return candidates[0]
|
||||
|
||||
|
||||
def resolve_binaural_hrtf_input(args, *, required):
|
||||
"""解析 binaural 的 HRTF 输入。
|
||||
|
||||
无显式输入时按顺序回退:默认 HRTF/binaural.sofa → 默认 cache 目录下唯一的
|
||||
.jochrtf → 默认 HRTF/binaural.personalized_headphone → 报错。
|
||||
只校验路径,不做编译。
|
||||
"""
|
||||
sofa = args.sofa_hrtf
|
||||
compiled = args.compiled_hrtf_cache
|
||||
private = args.personalized_headphone
|
||||
cache_policy = args.hrtf_cache_policy
|
||||
cache_dir = args.hrtf_cache_dir
|
||||
radius = args.hrtf_radius_m
|
||||
|
||||
if compiled is not None and cache_policy is not None:
|
||||
raise ValueError("显式 .jochrtf 输入不能再指定 --hrtf-cache-policy")
|
||||
if compiled is not None and radius != 1.0:
|
||||
raise ValueError("显式 .jochrtf 输入不能再选择 SOFA radius shell")
|
||||
if private is not None and (cache_policy is not None or cache_dir is not None
|
||||
or radius != 1.0):
|
||||
raise ValueError(
|
||||
"Rosella 模型输入不能使用 "
|
||||
"--hrtf-cache-policy/--hrtf-cache-dir/--hrtf-radius-m")
|
||||
|
||||
if required and sofa is None and compiled is None and private is None:
|
||||
if DEFAULT_SOFA_HRTF.is_file():
|
||||
sofa = DEFAULT_SOFA_HRTF
|
||||
else:
|
||||
compiled = _find_default_compiled_hrtf_cache()
|
||||
if compiled is None and DEFAULT_PERSONALIZED_HEADPHONE.is_file():
|
||||
private = DEFAULT_PERSONALIZED_HEADPHONE
|
||||
|
||||
if sofa is None and compiled is None and private is None:
|
||||
if cache_policy is not None or cache_dir is not None or radius != 1.0:
|
||||
raise ValueError("HRTF cache/radius 选项需要 --sofa-hrtf")
|
||||
if required:
|
||||
raise ValueError(
|
||||
"--binaural 未找到 HRTF 输入:默认 "
|
||||
f"{DEFAULT_SOFA_HRTF}、{DEFAULT_PERSONALIZED_HEADPHONE} 与 "
|
||||
f"{DEFAULT_HRTF_CACHE_DIR} 下的 .jochrtf 缓存都不存在;请用 "
|
||||
"--sofa-hrtf PATH、--compiled-hrtf-cache PATH 或 "
|
||||
"--personalized-headphone PATH 指定")
|
||||
return None
|
||||
|
||||
if sofa is None and (cache_policy is not None or cache_dir is not None
|
||||
or radius != 1.0):
|
||||
raise ValueError("HRTF cache/radius 选项需要 --sofa-hrtf")
|
||||
effective_policy = "memory" if cache_policy is None else cache_policy
|
||||
if cache_dir is not None and (sofa is None or effective_policy != "disk"):
|
||||
raise ValueError("--hrtf-cache-dir 仅与 SOFA 的 disk cache policy 一起使用")
|
||||
if sofa is not None:
|
||||
return {
|
||||
"kind": "sofa",
|
||||
"path": resolve_sofa_hrtf(sofa),
|
||||
"cache_policy": effective_policy,
|
||||
"cache_dir": (DEFAULT_HRTF_CACHE_DIR if cache_dir is None else
|
||||
cache_dir.expanduser().resolve()),
|
||||
}
|
||||
if compiled is not None:
|
||||
return {
|
||||
"kind": "compiled_cache",
|
||||
"path": resolve_compiled_hrtf_cache(compiled),
|
||||
"cache_policy": None,
|
||||
"cache_dir": None,
|
||||
}
|
||||
if private is not None:
|
||||
return {
|
||||
"kind": "rosella",
|
||||
"path": resolve_personalized_headphone(private),
|
||||
"cache_policy": None,
|
||||
"cache_dir": None,
|
||||
}
|
||||
return None
|
||||
|
||||
|
||||
def executable(value, name):
|
||||
path = shutil.which(value) if value else None
|
||||
if path is None and value and Path(value).is_file():
|
||||
path = str(Path(value).resolve())
|
||||
if path is None:
|
||||
raise FileNotFoundError(f"找不到 {name}: {value!r}")
|
||||
return path
|
||||
|
||||
|
||||
def run(command, label):
|
||||
print(f"[{label}]", flush=True)
|
||||
result = subprocess.run(command, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE,
|
||||
text=True, encoding="utf-8", errors="replace")
|
||||
if result.returncode:
|
||||
tail = result.stderr[-4000:]
|
||||
raise RuntimeError(f"{label} 失败(exit {result.returncode})\n{tail}")
|
||||
|
||||
|
||||
def timed_call(timings, name, function, *args, **kwargs):
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
return function(*args, **kwargs)
|
||||
finally:
|
||||
timings[name] = time.perf_counter() - started
|
||||
|
||||
|
||||
def probe_eac3_decoder_options(ffmpeg):
|
||||
"""读取 ``ffmpeg -h decoder=eac3`` 暴露的 AVOption 名。"""
|
||||
result = subprocess.run(
|
||||
[ffmpeg, "-hide_banner", "-h", "decoder=eac3"],
|
||||
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||||
text=True, encoding="utf-8", errors="replace")
|
||||
options = frozenset(EAC3_DECODER_OPTION_RE.findall(result.stdout or ""))
|
||||
# decoder 名不存在时 ffmpeg 依然返回 0,因此以“解析不到任何选项”为失败。
|
||||
if not options:
|
||||
raise RuntimeError(
|
||||
"无法读取 FFmpeg 的 eac3 解码器选项(ffmpeg -h decoder=eac3);"
|
||||
"需要带 E-AC-3 解码器的构建")
|
||||
return options
|
||||
|
||||
|
||||
def ffmpeg_version(ffmpeg):
|
||||
"""FFmpeg 版本字符串;探测失败返回空串,不影响渲染。"""
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[ffmpeg, "-hide_banner", "-version"],
|
||||
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||||
text=True, encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
return ""
|
||||
lines = (result.stdout or "").splitlines()
|
||||
line = lines[0].strip() if lines else ""
|
||||
prefix = "ffmpeg version "
|
||||
return line[len(prefix):].strip() if line.startswith(prefix) else line
|
||||
|
||||
|
||||
def eac3_decode_options(drc_scale, target_level, available):
|
||||
"""构造 ``-i`` 之前的 E-AC-3 解码选项,返回 ``(argv, report 片段)``。"""
|
||||
if "drc_scale" not in available:
|
||||
raise RuntimeError(
|
||||
"FFmpeg 的 eac3 解码器缺少 -drc_scale,无法关闭码流 DRC")
|
||||
# -drc_scale 始终显式下发:0(全动态范围)不是 ffmpeg 的默认值。
|
||||
argv = ["-drc_scale", format(float(drc_scale), ".10g")]
|
||||
if target_level:
|
||||
if "target_level" not in available:
|
||||
raise RuntimeError(
|
||||
"FFmpeg 的 eac3 解码器不支持 -target_level;请升级 FFmpeg "
|
||||
"或去掉 --eac3-target-level")
|
||||
argv += ["-target_level", str(int(target_level))]
|
||||
applied = {"drc_scale": float(drc_scale), "target_level": int(target_level)}
|
||||
return argv, applied
|
||||
|
||||
|
||||
def extract_eac3(ffmpeg, source, target):
|
||||
if source.suffix.lower() in (".eac3", ".ec3"):
|
||||
return source
|
||||
run([ffmpeg, "-hide_banner", "-loglevel", "error", "-y", "-i", str(source),
|
||||
"-map", "0:a:0", "-vn", "-c:a", "copy", "-f", "eac3", str(target)],
|
||||
"FFmpeg 提取 E-AC-3")
|
||||
return target
|
||||
|
||||
|
||||
def decode_core(ffmpeg, eac3, target, duration_sec=None, *, options=()):
|
||||
# 5.1(side) 的 f32le 顺序为 FL FR FC LFE SL SR;JOC 使用其中 0,1,2,4,5。
|
||||
command = [ffmpeg, "-hide_banner", "-loglevel", "error", "-y", *options,
|
||||
"-i", str(eac3), "-map", "0:a:0", "-vn"]
|
||||
if duration_sec is not None:
|
||||
command.extend(["-t", f"{duration_sec:.9f}"])
|
||||
command.extend(["-ac", "6", "-ar", str(RATE),
|
||||
"-c:a", "pcm_f32le", "-f", "f32le", str(target)])
|
||||
run(command, "FFmpeg 解码核心 5.1 PCM")
|
||||
return target
|
||||
|
||||
|
||||
def sha256(path):
|
||||
digest = hashlib.sha256()
|
||||
with Path(path).open("rb") as fp:
|
||||
for block in iter(lambda: fp.read(16 << 20), b""):
|
||||
digest.update(block)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def choose_pcm_output_format(requested_format, clip_action, peak, clipped_values,
|
||||
*, input_func=input, interactive=None):
|
||||
"""Resolve int24 clipping interactively or through an explicit policy."""
|
||||
if requested_format != "int24" or clipped_values == 0:
|
||||
return requested_format
|
||||
print(
|
||||
f"[clip] int24 将发生削波:peak={peak:.9g},超出 [-1,1] 的样本值={clipped_values}",
|
||||
file=sys.stderr, flush=True)
|
||||
action = clip_action
|
||||
if action == "ask":
|
||||
if interactive is None:
|
||||
interactive = bool(getattr(sys.stdin, "isatty", lambda: False)())
|
||||
if not interactive:
|
||||
raise RuntimeError(
|
||||
"检测到 int24 削波,但当前不是交互终端;请使用 "
|
||||
"--clip-action continue、--clip-action float32 或 --clip-action abort")
|
||||
while True:
|
||||
answer = input_func(
|
||||
"继续写 int24 并截断 [i] / 改为 float32 [f,默认] / 取消 [a]:"
|
||||
).strip().lower()
|
||||
if answer in ("", "f", "float", "float32"):
|
||||
action = "float32"
|
||||
break
|
||||
if answer in ("i", "int", "int24", "c", "continue"):
|
||||
action = "continue"
|
||||
break
|
||||
if answer in ("a", "abort", "q", "quit", "n", "no"):
|
||||
action = "abort"
|
||||
break
|
||||
print("请输入 i、f 或 a。", file=sys.stderr, flush=True)
|
||||
if action == "continue":
|
||||
print("[clip] 将继续写 int24,超范围值会截断到 [-1,1]。", flush=True)
|
||||
return "int24"
|
||||
if action == "float32":
|
||||
print("[clip] 已切换为 float32 WAV,不执行截断。", flush=True)
|
||||
return "float32"
|
||||
if action == "abort":
|
||||
raise RuntimeError("用户因 int24 削波取消输出")
|
||||
raise ValueError(f"未知 clip action: {action}")
|
||||
|
||||
|
||||
# Backward-compatible public name used by existing tests and callers.
|
||||
choose_speaker_output_format = choose_pcm_output_format
|
||||
|
||||
|
||||
def resolve_metadata(args, eac3, temp_dir):
|
||||
if args.metadata_dir:
|
||||
directory = Path(args.metadata_dir).resolve()
|
||||
return PayloadIndex(directory), "sidecar", directory
|
||||
if args.metadata_backend == "sidecar":
|
||||
raise ValueError("metadata-backend=sidecar 时必须提供 --metadata-dir")
|
||||
cache_dir = (args.metadata_cache.expanduser().resolve()
|
||||
if args.metadata_cache else None)
|
||||
max_frames = (math.ceil(args.duration * RATE / FRAME_SAMPLES)
|
||||
if args.duration is not None else None)
|
||||
index = DirectPayloadIndex.from_eac3(
|
||||
eac3, max_frames=max_frames, cache_dir=cache_dir)
|
||||
return index, "python-emdf-memory", cache_dir
|
||||
|
||||
|
||||
def variant_call(output, source, function, *args, **kwargs):
|
||||
"""执行一个阶段;遇到未知变体时在目标文件旁写结构化报告。"""
|
||||
try:
|
||||
return function(*args, **kwargs)
|
||||
except UnsupportedVariantError as exc:
|
||||
report_path = Path(str(output) + ".variant-error.json")
|
||||
write_variant_report(report_path, exc, input_path=source, output_path=output)
|
||||
print(f"[VARIANT] {exc}", file=sys.stderr, flush=True)
|
||||
print(f"[VARIANT] 维修报告: {report_path}", file=sys.stderr, flush=True)
|
||||
raise
|
||||
|
||||
|
||||
def create_renderer(backend, gain, native_library=None, native_threads=None):
|
||||
"""选择整帧 DSP 后端;auto 优先使用 lib 中当前平台的原生构建。"""
|
||||
if backend in ("auto", "native"):
|
||||
try:
|
||||
decoder = NativeJocRenderer(
|
||||
output_scale=gain, library_path=native_library, threads=native_threads)
|
||||
info = {
|
||||
"name": "native",
|
||||
"library": str(decoder.library_path),
|
||||
"build": decoder.build_info,
|
||||
"threads": decoder.threads,
|
||||
}
|
||||
print(f"[backend] native: {info['build']} threads={info['threads']} "
|
||||
f"({info['library']})", flush=True)
|
||||
return decoder, info
|
||||
except (NativeBackendUnavailable, OSError) as exc:
|
||||
print(f"[backend] native unavailable, falling back to Python: {exc}", flush=True)
|
||||
decoder = JocRenderer(output_scale=gain)
|
||||
info = {"name": "python", "library": None, "build": None, "threads": None}
|
||||
print("[backend] python/numpy", flush=True)
|
||||
return decoder, info
|
||||
|
||||
|
||||
def render(index, bed_path, frame_count, raw_path, gain, progress_every,
|
||||
backend="auto", native_library=None, native_threads=None, frame_sink=None,
|
||||
speaker_renderer=None, speaker_sink=None, speaker_metadata_offset=1473,
|
||||
binaural_renderer=None, binaural_sink=None, binaural_metadata_offset=1473,
|
||||
raw_scale=1.0):
|
||||
values = np.memmap(bed_path, dtype=np.float32, mode="r")
|
||||
frame_width = FRAME_SAMPLES * 6
|
||||
if values.size % frame_width:
|
||||
raise ValueError(f"FFmpeg PCM 长度不是 1536×6 的整数倍: {values.size}")
|
||||
bed = values.reshape(-1, FRAME_SAMPLES, 6)
|
||||
if len(bed) < frame_count:
|
||||
raise ValueError(f"PCM 只有 {len(bed)} 帧,元数据需要 {frame_count} 帧")
|
||||
output = (np.memmap(raw_path, dtype=np.float32, mode="w+",
|
||||
shape=(frame_count, FRAME_SAMPLES, 16))
|
||||
if raw_path is not None else None)
|
||||
decoder, backend_info = create_renderer(backend, gain, native_library, native_threads)
|
||||
started = time.perf_counter()
|
||||
dsp_seconds = 0.0
|
||||
adm_stream_seconds = 0.0
|
||||
raw_write_seconds = 0.0
|
||||
speaker_render_seconds = 0.0
|
||||
speaker_write_seconds = 0.0
|
||||
binaural_render_seconds = 0.0
|
||||
binaural_write_seconds = 0.0
|
||||
elapsed = 0.0
|
||||
try:
|
||||
for frame_number, row in enumerate(index.rows[:frame_count]):
|
||||
bed6 = np.asarray(bed[frame_number], dtype=np.float32)
|
||||
subs = index.subpayloads(row)
|
||||
stage = time.perf_counter()
|
||||
pcm16, _ = decoder.render_subpayloads(
|
||||
subs, bed6[:, [0, 1, 2, 4, 5]].T, bed6[:, 3])
|
||||
dsp_seconds += time.perf_counter() - stage
|
||||
if output is not None:
|
||||
stage = time.perf_counter()
|
||||
output[frame_number] = np.multiply(
|
||||
pcm16.T, np.float32(raw_scale), dtype=np.float32)
|
||||
raw_write_seconds += time.perf_counter() - stage
|
||||
if frame_sink is not None:
|
||||
stage = time.perf_counter()
|
||||
frame_sink.write_frame(pcm16)
|
||||
adm_stream_seconds += time.perf_counter() - stage
|
||||
if speaker_renderer is not None:
|
||||
stage = time.perf_counter()
|
||||
speaker_pcm = speaker_renderer.render_frame(
|
||||
pcm16.T, subs.get(11), speaker_metadata_offset)
|
||||
speaker_render_seconds += time.perf_counter() - stage
|
||||
stage = time.perf_counter()
|
||||
speaker_sink.write_frame(speaker_pcm)
|
||||
speaker_write_seconds += time.perf_counter() - stage
|
||||
if binaural_renderer is not None:
|
||||
payload = subs.get(11)
|
||||
outer_offset = (
|
||||
index.subpayload_sample_offset(row, 11)
|
||||
if payload is not None and hasattr(index, "subpayload_sample_offset")
|
||||
else 0
|
||||
)
|
||||
stage = time.perf_counter()
|
||||
binaural_pcm = binaural_renderer.render_frame(
|
||||
pcm16.T, payload, binaural_metadata_offset,
|
||||
outer_sample_offset=outer_offset)
|
||||
binaural_render_seconds += time.perf_counter() - stage
|
||||
if len(binaural_pcm):
|
||||
stage = time.perf_counter()
|
||||
binaural_sink.write_frame(binaural_pcm)
|
||||
binaural_write_seconds += time.perf_counter() - stage
|
||||
done = frame_number + 1
|
||||
if done % progress_every == 0 or done == frame_count:
|
||||
elapsed = time.perf_counter() - started
|
||||
speed = done / max(elapsed, 1e-9)
|
||||
eta = (frame_count - done) / max(speed, 1e-9)
|
||||
print(f"[JOC:{backend_info['name']}] {done}/{frame_count} "
|
||||
f"{speed:.1f} frame/s ETA {eta:.1f}s", flush=True)
|
||||
if binaural_renderer is not None:
|
||||
stage = time.perf_counter()
|
||||
binaural_tail = binaural_renderer.finish()
|
||||
binaural_render_seconds += time.perf_counter() - stage
|
||||
if len(binaural_tail):
|
||||
stage = time.perf_counter()
|
||||
binaural_sink.write_frame(binaural_tail)
|
||||
binaural_write_seconds += time.perf_counter() - stage
|
||||
if output is not None:
|
||||
output.flush()
|
||||
elapsed = time.perf_counter() - started
|
||||
finally:
|
||||
close = getattr(decoder, "close", None)
|
||||
if close is not None:
|
||||
close()
|
||||
close = getattr(speaker_renderer, "close", None)
|
||||
if close is not None:
|
||||
close()
|
||||
close = getattr(binaural_renderer, "close", None)
|
||||
if close is not None:
|
||||
close()
|
||||
breakdown = {
|
||||
"pipeline_wall_seconds": elapsed,
|
||||
"dsp_and_joc_parse_seconds": dsp_seconds,
|
||||
"adm_stream_write_seconds": adm_stream_seconds,
|
||||
"raw_float_write_seconds": raw_write_seconds,
|
||||
"speaker_render_seconds": speaker_render_seconds,
|
||||
"speaker_spool_write_seconds": speaker_write_seconds,
|
||||
"binaural_render_seconds": binaural_render_seconds,
|
||||
"binaural_spool_write_seconds": binaural_write_seconds,
|
||||
}
|
||||
return dsp_seconds, backend_info, breakdown
|
||||
|
||||
|
||||
def build_parser():
|
||||
parser = argparse.ArgumentParser(
|
||||
description=("JustOneCacophony (JOC):E-AC-3 JOC → 25ch ADM BWF、"
|
||||
"扬声器 WAV 或公开 SOFA 双耳 WAV"))
|
||||
parser.add_argument("input", type=Path, help="输入 .m4a/.eac3/.ec3")
|
||||
parser.add_argument("-o", "--output", type=Path, help="输出文件;默认按模式和布局命名")
|
||||
parser.add_argument("--speaker-output", type=Path,
|
||||
help="扬声器 WAV 路径;仅与 --speaker-layout 一起使用")
|
||||
parser.add_argument("--binaural-output", type=Path,
|
||||
help="双耳 WAV 路径;仅与 --binaural 一起使用")
|
||||
direct_mode = parser.add_mutually_exclusive_group()
|
||||
direct_mode.add_argument("--speaker-layout", choices=SPEAKER_LAYOUT_CHOICES,
|
||||
help="直接扬声器渲染布局,例如 2.0、5.1、7.1.2")
|
||||
direct_mode.add_argument("--binaural", action="store_true",
|
||||
help="直接 SOFA 双耳渲染;不生成临时 ADM BWF")
|
||||
parser.add_argument("--speaker-format", choices=("float32", "int24"), default="float32",
|
||||
help="扬声器 WAV 格式,默认 float32")
|
||||
parser.add_argument("--binaural-format", choices=("float32", "int24"), default="float32",
|
||||
help="双耳 WAV 格式,默认 float32")
|
||||
parser.add_argument("--clip-action", choices=("ask", "continue", "float32", "abort"),
|
||||
default="ask",
|
||||
help="int24 削波处理:交互询问、继续截断、改 float32 或中止")
|
||||
parser.add_argument("--speaker-metadata-offset", type=int, default=1473,
|
||||
help="扬声器渲染 metadata 相对帧偏移,默认 1473 samples")
|
||||
parser.add_argument("--binaural-mode", choices=("off", "near", "mid", "far"),
|
||||
default="mid",
|
||||
help="双耳渲染模式,默认 mid(人为指定的渲染提示,非码流 "
|
||||
"原始元数据);直接双耳渲染与 ADM BWF 的 DBMD 提示共用。"
|
||||
"off 仅用于 ADM BWF:关闭 DBMD 双耳提示(编码 0)")
|
||||
hrtf_input = parser.add_mutually_exclusive_group()
|
||||
hrtf_input.add_argument(
|
||||
"--sofa-hrtf", type=Path,
|
||||
help="SimpleFreeFieldHRIR SOFA;缺省时依次尝试 HRTF/binaural.sofa、"
|
||||
"output/hrtf-cache 下唯一的 .jochrtf、"
|
||||
"HRTF/binaural.personalized_headphone,均无则报错")
|
||||
hrtf_input.add_argument(
|
||||
"--compiled-hrtf-cache", type=Path,
|
||||
help="高级入口:显式读取 JOC .jochrtf compiled cache")
|
||||
hrtf_input.add_argument(
|
||||
"--personalized-headphone", type=Path, nargs="?",
|
||||
const=DEFAULT_PERSONALIZED_HEADPHONE,
|
||||
help="Rosella .personalized_headphone 模型;不带路径时默认 "
|
||||
"HRTF/binaural.personalized_headphone")
|
||||
parser.add_argument(
|
||||
"--hrtf-cache-policy", choices=("none", "memory", "disk"), default=None,
|
||||
help="SOFA 编译缓存;默认 memory,disk 写入可删除的 .jochrtf")
|
||||
parser.add_argument(
|
||||
"--hrtf-cache-dir", type=Path,
|
||||
help="disk cache 目录;默认 output/hrtf-cache")
|
||||
parser.add_argument(
|
||||
"--hrtf-radius-m", type=float, default=1.0,
|
||||
help="选择最近的 SOFA measurement-radius shell,默认 1.0 m")
|
||||
parser.add_argument("--binaural-tail-seconds", type=float, default=5.0,
|
||||
help="双耳 room/filterbank flush 上限,默认 5 秒")
|
||||
parser.add_argument("--binaural-tail-threshold", type=float, default=1.0e-8,
|
||||
help="双耳尾声裁切阈值,默认 1e-8;主体至少保留原时长")
|
||||
parser.add_argument("--binaural-chunk-frames", type=int, default=64,
|
||||
help="双耳内部批处理 E-AC-3 帧数,默认 64")
|
||||
parser.add_argument("--gain-db", type=float, default=0.0,
|
||||
help="成品增益 dB,默认 0;双耳路径以 float64 应用")
|
||||
parser.add_argument("--duration", type=float, help="只处理开头指定秒数")
|
||||
parser.add_argument("--object-delay-samples", type=int, default=1473,
|
||||
help="对象 PCM/OAMD 时间补偿;ADM 与双耳默认 1473 samples")
|
||||
parser.add_argument("--trajectory-mode", choices=("compact", "dense64"), default="compact",
|
||||
help="ADM 对象轨迹表示;直接双耳路径不序列化 AXML")
|
||||
parser.add_argument("--ffmpeg", default=os.environ.get("FFMPEG", "ffmpeg"))
|
||||
parser.add_argument("--eac3-drc-scale", type=float, default=0.0,
|
||||
help="E-AC-3 解码器 -drc_scale:0=关闭码流 dynrng(全动态范围),"
|
||||
"1=码流作者意图,>1 非对称;默认 0")
|
||||
parser.add_argument("--eac3-target-level", type=int, default=0,
|
||||
help="E-AC-3 解码器 -target_level:按码流 dialnorm 归一化电平,"
|
||||
"增益约 target_level - dialnorm dB;0=不施加,默认 0")
|
||||
parser.add_argument("--backend", choices=("auto", "native", "python"), default="auto",
|
||||
help="JOC/扬声器 DSP 后端;SOFA 双耳 DSP 当前使用 Python")
|
||||
parser.add_argument("--native-library", type=Path,
|
||||
help="显式指定原生库;默认从单层 lib 目录选择当前平台文件")
|
||||
parser.add_argument("--native-threads", type=int,
|
||||
help="原生 DSP 总线程数;默认在 4 核以上使用 2,可用环境变量 EAC3JOC_NATIVE_THREADS 覆盖")
|
||||
metadata_source = parser.add_mutually_exclusive_group()
|
||||
metadata_source.add_argument("--metadata-dir", type=Path,
|
||||
help="含 frames.csv 和 emdf/ 或 payloads/ 的元数据 sidecar")
|
||||
metadata_source.add_argument("--metadata-cache", type=Path,
|
||||
help="把直接 EMDF 扫描或兼容桥结果持久保存到此目录")
|
||||
parser.add_argument("--metadata-backend", choices=("auto", "emdf", "sidecar"),
|
||||
default="auto", help="直接扫描连续 EMDF,或读取现有 sidecar")
|
||||
parser.add_argument("--print-metadata", choices=("none", "summary", "frames"), default="none",
|
||||
help="诊断元数据输出;默认 none,避免转换前重复完整解析")
|
||||
parser.add_argument("--metadata-json", type=Path, help="元数据汇总 JSON 路径")
|
||||
parser.add_argument("--metadata-only", action="store_true", help="解析/打印元数据后退出")
|
||||
parser.add_argument("--keep-raw", action="store_true", help="额外保留 16ch f32le 对象中间文件")
|
||||
parser.add_argument("--skip-sha256", action="store_true",
|
||||
help="跳过最终文件 SHA-256 全量复扫以缩短大文件处理时间")
|
||||
parser.add_argument("--progress-every", type=int, default=500)
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv=None):
|
||||
# Windows 控制台的活动代码页未必能表示日文文件名;保留信息并避免
|
||||
# UnicodeEncodeError 中断长任务。支持 UTF-8 的终端仍会原样显示。
|
||||
for stream in (sys.stdout, sys.stderr):
|
||||
if hasattr(stream, "reconfigure"):
|
||||
stream.reconfigure(encoding="utf-8", errors="backslashreplace")
|
||||
args = build_parser().parse_args(argv)
|
||||
source = args.input.expanduser().resolve()
|
||||
if not source.is_file():
|
||||
raise FileNotFoundError(source)
|
||||
speaker_mode = args.speaker_layout is not None
|
||||
binaural_mode = bool(args.binaural)
|
||||
binaural_render_mode = args.binaural_mode
|
||||
if binaural_mode and binaural_render_mode == "off":
|
||||
raise ValueError(
|
||||
"--binaural-mode off 仅用于 ADM BWF 输出(关闭 DBMD 双耳提示);"
|
||||
"直接双耳渲染请使用 near/mid/far")
|
||||
if args.speaker_output is not None and not speaker_mode:
|
||||
raise ValueError("--speaker-output 必须与 --speaker-layout 一起使用")
|
||||
if args.binaural_output is not None and not binaural_mode:
|
||||
raise ValueError("--binaural-output 必须与 --binaural 一起使用")
|
||||
specific_outputs = [value for value in (args.speaker_output, args.binaural_output)
|
||||
if value is not None]
|
||||
if args.output is not None and specific_outputs:
|
||||
raise ValueError("-o/--output 与 --speaker-output/--binaural-output 不能同时使用")
|
||||
if len(specific_outputs) > 1:
|
||||
raise ValueError("--speaker-output 与 --binaural-output 不能同时使用")
|
||||
if args.speaker_metadata_offset < 0:
|
||||
raise ValueError("speaker-metadata-offset 不能为负数")
|
||||
hrtf_options_used = any((
|
||||
args.sofa_hrtf is not None,
|
||||
args.compiled_hrtf_cache is not None,
|
||||
args.personalized_headphone is not None,
|
||||
args.hrtf_cache_policy is not None,
|
||||
args.hrtf_cache_dir is not None,
|
||||
args.hrtf_radius_m != 1.0,
|
||||
))
|
||||
if hrtf_options_used and not binaural_mode:
|
||||
raise ValueError("SOFA/HRTF 选项仅与 --binaural 一起使用")
|
||||
if (not math.isfinite(args.binaural_tail_seconds)
|
||||
or args.binaural_tail_seconds < 0):
|
||||
raise ValueError("binaural-tail-seconds 必须是非负有限值")
|
||||
if (not math.isfinite(args.binaural_tail_threshold)
|
||||
or args.binaural_tail_threshold < 0):
|
||||
raise ValueError("binaural-tail-threshold 必须是非负有限值")
|
||||
if args.binaural_chunk_frames <= 0:
|
||||
raise ValueError("binaural-chunk-frames 必须大于 0")
|
||||
if not math.isfinite(args.hrtf_radius_m) or args.hrtf_radius_m <= 0.0:
|
||||
raise ValueError("hrtf-radius-m 必须是正有限值")
|
||||
requested_output = (args.speaker_output if args.speaker_output is not None
|
||||
else args.binaural_output if args.binaural_output is not None
|
||||
else args.output)
|
||||
output = resolve_output(
|
||||
source, requested_output, args.speaker_layout if speaker_mode else None,
|
||||
binaural=binaural_mode)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
if args.duration is not None and args.duration <= 0:
|
||||
raise ValueError("duration 必须大于 0")
|
||||
if args.object_delay_samples < 0:
|
||||
raise ValueError("object-delay-samples 不能为负数")
|
||||
gain_float64 = 10.0 ** (args.gain_db / 20.0)
|
||||
gain = np.float32(gain_float64)
|
||||
if not math.isfinite(gain_float64) or not np.isfinite(gain):
|
||||
raise ValueError("gain-db 超出支持范围")
|
||||
if (not math.isfinite(args.eac3_drc_scale)
|
||||
or not 0.0 <= args.eac3_drc_scale <= EAC3_DRC_SCALE_MAX):
|
||||
raise ValueError(f"eac3-drc-scale 必须在 0..{EAC3_DRC_SCALE_MAX:g} 之间")
|
||||
if not (EAC3_TARGET_LEVEL_RANGE[0] <= args.eac3_target_level
|
||||
<= EAC3_TARGET_LEVEL_RANGE[1]):
|
||||
raise ValueError("eac3-target-level 必须在 -31..0 之间")
|
||||
binaural_hrtf_input = resolve_binaural_hrtf_input(
|
||||
args, required=binaural_mode and not args.metadata_only)
|
||||
ffmpeg = executable(args.ffmpeg, "FFmpeg")
|
||||
decode_options = ()
|
||||
decode_option_info = None
|
||||
if not args.metadata_only:
|
||||
available = probe_eac3_decoder_options(ffmpeg)
|
||||
decode_options, applied = eac3_decode_options(
|
||||
args.eac3_drc_scale, args.eac3_target_level, available)
|
||||
decode_option_info = {
|
||||
"version": ffmpeg_version(ffmpeg),
|
||||
"eac3_decode_options": applied,
|
||||
}
|
||||
print(f"[decode] ffmpeg {decode_option_info['version']} "
|
||||
f"drc_scale={args.eac3_drc_scale:g} "
|
||||
f"target_level={args.eac3_target_level}", flush=True)
|
||||
|
||||
total_started = time.perf_counter()
|
||||
timings = {}
|
||||
with tempfile.TemporaryDirectory(prefix="eac3joc-", dir=output.parent) as temporary:
|
||||
temp_dir = Path(temporary)
|
||||
eac3 = timed_call(timings, "extract_eac3", extract_eac3,
|
||||
ffmpeg, source, temp_dir / "input.eac3")
|
||||
index, metadata_backend, metadata_cache_dir = timed_call(
|
||||
timings, "resolve_metadata", variant_call,
|
||||
output, source, resolve_metadata, args, eac3, temp_dir)
|
||||
timings["load_metadata_index"] = 0.0
|
||||
frame_count = len(index)
|
||||
if args.duration is not None:
|
||||
frame_count = min(frame_count, math.ceil(args.duration * RATE / FRAME_SAMPLES))
|
||||
duration_sec = frame_count * FRAME_SAMPLES / RATE
|
||||
need_metadata_summary = (
|
||||
args.metadata_only or args.metadata_json is not None or args.print_metadata != "none")
|
||||
if need_metadata_summary:
|
||||
metadata_json = (args.metadata_json or Path(str(output) + ".metadata.json")).resolve()
|
||||
summary = timed_call(
|
||||
timings, "metadata_summary", variant_call,
|
||||
output, source, write_summary, index, metadata_json, limit=frame_count,
|
||||
print_frames=args.print_metadata == "frames")
|
||||
if args.print_metadata == "summary":
|
||||
print("[metadata] " + json.dumps(summary, ensure_ascii=False, separators=(",", ":")))
|
||||
print(f"[metadata] backend={metadata_backend} frames={frame_count} -> {metadata_json}")
|
||||
else:
|
||||
metadata_json = None
|
||||
timings["metadata_summary"] = 0.0
|
||||
print(f"[metadata] backend={metadata_backend} frames={frame_count} summary=skipped")
|
||||
if args.metadata_only:
|
||||
return 0
|
||||
|
||||
bed_path = timed_call(
|
||||
timings, "decode_core", decode_core,
|
||||
ffmpeg, eac3, temp_dir / "core51_f32le.raw", duration_sec,
|
||||
options=decode_options)
|
||||
raw_path = (output.with_name(output.name + ".objects16.f32le")
|
||||
if args.keep_raw else None)
|
||||
master = None
|
||||
speaker_backend_info = None
|
||||
speaker_wav_info = None
|
||||
speaker_clip_info = None
|
||||
speaker_actual_format = None
|
||||
binaural_backend_info = None
|
||||
binaural_hrtf_report = None
|
||||
binaural_wav_info = None
|
||||
binaural_clip_info = None
|
||||
binaural_actual_format = None
|
||||
if speaker_mode:
|
||||
timings["create_binaural_renderer"] = 0.0
|
||||
layout = get_speaker_layout(args.speaker_layout)
|
||||
speaker_name = speaker_layout_display_name(layout)
|
||||
speaker_decoder, speaker_backend_info = create_speaker_renderer(
|
||||
layout, backend=args.backend, native_library=args.native_library)
|
||||
fallback = speaker_backend_info.get("fallback_reason")
|
||||
if fallback:
|
||||
print(f"[speaker] native unavailable, falling back to Python: {fallback}",
|
||||
flush=True)
|
||||
print(f"[speaker] layout={speaker_name} backend={speaker_backend_info['name']} "
|
||||
f"channels={layout.channel_count}", flush=True)
|
||||
spool = SpeakerPcmSpool(
|
||||
temp_dir / "speaker_interleaved_f32.raw",
|
||||
frame_count * FRAME_SAMPLES, layout.channel_count)
|
||||
try:
|
||||
render_seconds, renderer_backend, render_breakdown = timed_call(
|
||||
timings, "render_and_stream", variant_call,
|
||||
output, source, render, index, bed_path, frame_count, raw_path, gain,
|
||||
max(1, args.progress_every), args.backend, args.native_library,
|
||||
args.native_threads, None, speaker_decoder, spool,
|
||||
args.speaker_metadata_offset)
|
||||
spool.finalize()
|
||||
speaker_actual_format = choose_pcm_output_format(
|
||||
args.speaker_format, args.clip_action, spool.peak,
|
||||
spool.clipped_values)
|
||||
speaker_wav_info = timed_call(
|
||||
timings, "write_speaker_wav", write_pcm_wav,
|
||||
output, spool.values, speaker_actual_format, rate=RATE)
|
||||
speaker_clip_info = {
|
||||
"peak": spool.peak,
|
||||
"over_unity_values": spool.clipped_values,
|
||||
"requested_format": args.speaker_format,
|
||||
"actual_format": speaker_actual_format,
|
||||
"clip_action": args.clip_action,
|
||||
}
|
||||
finally:
|
||||
spool.close()
|
||||
timings["build_adm_tracks"] = 0.0
|
||||
timings["finalize_adm"] = 0.0
|
||||
timings["validate_adm"] = 0.0
|
||||
info = (f"speaker layout={speaker_name}, format={speaker_actual_format}, "
|
||||
f"peak={speaker_clip_info['peak']:.9g}")
|
||||
elif binaural_mode:
|
||||
hrtf_source = binaural_hrtf_input
|
||||
common_options = {
|
||||
"mode": binaural_render_mode,
|
||||
"object_delay_samples": args.object_delay_samples,
|
||||
"tail_seconds": args.binaural_tail_seconds,
|
||||
"output_gain": gain_float64,
|
||||
"chunk_frames": args.binaural_chunk_frames,
|
||||
}
|
||||
if hrtf_source["kind"] == "sofa":
|
||||
binaural_decoder = None
|
||||
if args.backend in ("auto", "native"):
|
||||
try:
|
||||
from sofa_native_backend import create_native_sofa_renderer
|
||||
binaural_decoder = timed_call(
|
||||
timings, "create_binaural_renderer",
|
||||
create_native_sofa_renderer,
|
||||
hrtf_source["path"],
|
||||
cache_policy=hrtf_source["cache_policy"],
|
||||
cache_dir=hrtf_source["cache_dir"],
|
||||
shell_radius_m=args.hrtf_radius_m,
|
||||
**common_options)
|
||||
except (ImportError, OSError, RuntimeError, ValueError) as exc:
|
||||
print(
|
||||
f"[binaural] native SOFA backend unavailable "
|
||||
f"({exc.__class__.__name__}: {exc}); "
|
||||
f"falling back to Python", flush=True)
|
||||
binaural_decoder = None
|
||||
if binaural_decoder is None:
|
||||
binaural_decoder = timed_call(
|
||||
timings, "create_binaural_renderer",
|
||||
SofaBinauralRenderer.from_sofa,
|
||||
hrtf_source["path"],
|
||||
cache_policy=hrtf_source["cache_policy"],
|
||||
cache_dir=hrtf_source["cache_dir"],
|
||||
shell_radius_m=args.hrtf_radius_m,
|
||||
**common_options)
|
||||
elif hrtf_source["kind"] == "rosella":
|
||||
binaural_decoder = timed_call(
|
||||
timings, "create_binaural_renderer",
|
||||
RosellaBinauralRenderer,
|
||||
hrtf_source["path"],
|
||||
mode=binaural_render_mode,
|
||||
object_delay_samples=args.object_delay_samples,
|
||||
tail_seconds=args.binaural_tail_seconds,
|
||||
output_gain=gain_float64,
|
||||
chunk_frames=args.binaural_chunk_frames,
|
||||
backend=args.backend,
|
||||
native_library=args.native_library)
|
||||
else:
|
||||
binaural_decoder = None
|
||||
if args.backend in ("auto", "native"):
|
||||
try:
|
||||
from sofa_native_backend import (
|
||||
create_native_compiled_cache_renderer)
|
||||
binaural_decoder = timed_call(
|
||||
timings, "create_binaural_renderer",
|
||||
create_native_compiled_cache_renderer,
|
||||
hrtf_source["path"],
|
||||
**common_options)
|
||||
except (ImportError, OSError, RuntimeError, ValueError) as exc:
|
||||
print(
|
||||
f"[binaural] native SOFA backend unavailable "
|
||||
f"({exc.__class__.__name__}: {exc}); "
|
||||
f"falling back to Python", flush=True)
|
||||
binaural_decoder = None
|
||||
if binaural_decoder is None:
|
||||
binaural_decoder = timed_call(
|
||||
timings, "create_binaural_renderer",
|
||||
SofaBinauralRenderer.from_compiled_cache,
|
||||
hrtf_source["path"],
|
||||
**common_options)
|
||||
print(
|
||||
f"[binaural] mode={binaural_render_mode} "
|
||||
f"backend={binaural_decoder.dsp_backend} "
|
||||
f"precision=float64/complex128 "
|
||||
f"hrtf={hrtf_source['kind']}:{hrtf_source['path']}", flush=True)
|
||||
if hrtf_source["kind"] == "rosella":
|
||||
flush_samples = math.ceil(
|
||||
(args.binaural_tail_seconds * RATE
|
||||
+ ROSSELLA_LATENCY_SAMPLES + ROSSELLA_BLOCK_SAMPLES)
|
||||
/ ROSSELLA_BLOCK_SAMPLES) * ROSSELLA_BLOCK_SAMPLES
|
||||
spool_capacity = frame_count * FRAME_SAMPLES + flush_samples
|
||||
else:
|
||||
spool_capacity = (
|
||||
frame_count * FRAME_SAMPLES
|
||||
+ binaural_decoder.finish_capacity_samples)
|
||||
spool = BinauralPcmSpool(
|
||||
temp_dir / "binaural_interleaved_f64.raw",
|
||||
spool_capacity,
|
||||
tail_threshold=args.binaural_tail_threshold)
|
||||
try:
|
||||
render_seconds, renderer_backend, render_breakdown = timed_call(
|
||||
timings, "render_and_stream", variant_call,
|
||||
output, source, render, index, bed_path, frame_count, raw_path,
|
||||
np.float32(1.0), max(1, args.progress_every),
|
||||
backend=args.backend, native_library=args.native_library,
|
||||
native_threads=args.native_threads,
|
||||
binaural_renderer=binaural_decoder, binaural_sink=spool,
|
||||
binaural_metadata_offset=args.object_delay_samples, raw_scale=gain)
|
||||
spool.finalize(minimum_samples=frame_count * FRAME_SAMPLES)
|
||||
binaural_actual_format = choose_pcm_output_format(
|
||||
args.binaural_format, args.clip_action, spool.peak,
|
||||
spool.clipped_values)
|
||||
binaural_wav_info = timed_call(
|
||||
timings, "write_binaural_wav", write_pcm_wav,
|
||||
output, spool.values, binaural_actual_format, rate=RATE)
|
||||
binaural_clip_info = {
|
||||
"peak": spool.peak,
|
||||
"over_unity_values": spool.clipped_values,
|
||||
"requested_format": args.binaural_format,
|
||||
"actual_format": binaural_actual_format,
|
||||
"clip_action": args.clip_action,
|
||||
"tail_threshold": args.binaural_tail_threshold,
|
||||
"source_samples": frame_count * FRAME_SAMPLES,
|
||||
"kept_samples": spool.sample_count,
|
||||
}
|
||||
binaural_backend_info = binaural_decoder.backend_info
|
||||
if hrtf_source["kind"] == "rosella":
|
||||
binaural_hrtf_report = {
|
||||
"input_kind": "rosella",
|
||||
"input_path": str(binaural_decoder.model_path.resolve()),
|
||||
"model_coefficient_sha256": (
|
||||
binaural_decoder.model.coefficient_sha256),
|
||||
"cache_policy": None,
|
||||
}
|
||||
else:
|
||||
binaural_hrtf_report = {
|
||||
"input_kind": binaural_backend_info["hrtf_input_kind"],
|
||||
"input_path": binaural_backend_info["hrtf_input_path"],
|
||||
"source_sha256": (
|
||||
binaural_backend_info["field"]["source_sha256"]),
|
||||
"cache_policy": binaural_backend_info["cache_policy"],
|
||||
"cache_key": (
|
||||
binaural_backend_info["field"]["cache_key"]),
|
||||
"format_version": (
|
||||
binaural_backend_info["field"]["format_version"]),
|
||||
}
|
||||
finally:
|
||||
spool.close()
|
||||
timings["build_adm_tracks"] = 0.0
|
||||
timings["finalize_adm"] = 0.0
|
||||
timings["validate_adm"] = 0.0
|
||||
info = (f"binaural mode={binaural_render_mode}, "
|
||||
f"format={binaural_actual_format}, "
|
||||
f"peak={binaural_clip_info['peak']:.9g}, "
|
||||
f"samples={binaural_clip_info['kept_samples']}")
|
||||
else:
|
||||
timings["create_binaural_renderer"] = 0.0
|
||||
master = adm_assemble.StreamingMaster(
|
||||
output, duration_sec, rate=RATE,
|
||||
joc_binaural_mode=adm_atmos.JOC_BINAURAL_MODES[args.binaural_mode])
|
||||
try:
|
||||
render_seconds, renderer_backend, render_breakdown = timed_call(
|
||||
timings, "render_and_stream", variant_call,
|
||||
output, source, render, index, bed_path, frame_count, raw_path, gain,
|
||||
max(1, args.progress_every), args.backend, args.native_library,
|
||||
args.native_threads, master)
|
||||
tracks = timed_call(
|
||||
timings, "build_adm_tracks", variant_call,
|
||||
output, source, oamd_tracks.build_adm_tracks,
|
||||
index, index.rows[:frame_count], rate=RATE, frame_samples=FRAME_SAMPLES,
|
||||
object_delay_samples=args.object_delay_samples,
|
||||
trajectory_mode=args.trajectory_mode)
|
||||
timed_call(timings, "finalize_adm", master.finalize, tracks)
|
||||
except Exception:
|
||||
master.abort()
|
||||
raise
|
||||
errors, info = timed_call(timings, "validate_adm", validate, str(output))
|
||||
if errors:
|
||||
raise RuntimeError("ADM 校验失败: " + "; ".join(errors))
|
||||
# Windows 不允许删除仍被 NumPy memmap 持有的临时 core/raw;显式回收闭包。
|
||||
import gc
|
||||
gc.collect()
|
||||
|
||||
if args.skip_sha256:
|
||||
output_sha = None
|
||||
timings["sha256"] = 0.0
|
||||
else:
|
||||
output_sha = timed_call(timings, "sha256", sha256, output)
|
||||
total_seconds = time.perf_counter() - total_started
|
||||
mode_name = "speaker" if speaker_mode else "binaural" if binaural_mode else "adm"
|
||||
report = {
|
||||
"input": str(source),
|
||||
"output": str(output),
|
||||
"mode": mode_name,
|
||||
"metadata": str(metadata_json) if metadata_json is not None else None,
|
||||
"metadata_backend": metadata_backend,
|
||||
"metadata_cache": str(metadata_cache_dir) if metadata_cache_dir is not None else None,
|
||||
"frames": frame_count,
|
||||
"duration_sec": duration_sec,
|
||||
"gain_db": args.gain_db,
|
||||
"gain_float32": float(gain),
|
||||
"gain_float64": float(gain_float64),
|
||||
"object_delay_samples": (None if speaker_mode else args.object_delay_samples),
|
||||
"trajectory_mode": args.trajectory_mode if mode_name == "adm" else None,
|
||||
"binaural_mode_value": (
|
||||
adm_atmos.JOC_BINAURAL_MODES[args.binaural_mode]
|
||||
if mode_name == "adm" else None),
|
||||
"render_seconds": render_seconds,
|
||||
"render_breakdown": render_breakdown,
|
||||
"renderer_backend": renderer_backend,
|
||||
"speaker_renderer_backend": speaker_backend_info,
|
||||
"speaker_layout": args.speaker_layout if speaker_mode else None,
|
||||
"speaker_metadata_offset": args.speaker_metadata_offset if speaker_mode else None,
|
||||
"speaker_clip": speaker_clip_info,
|
||||
"speaker_wav": speaker_wav_info,
|
||||
"binaural_renderer_backend": binaural_backend_info,
|
||||
"binaural_mode": (
|
||||
args.binaural_mode if (binaural_mode or mode_name == "adm") else None),
|
||||
"binaural_hrtf": (
|
||||
binaural_hrtf_report if binaural_backend_info else None),
|
||||
"binaural_clip": binaural_clip_info,
|
||||
"binaural_wav": binaural_wav_info,
|
||||
"output_clip": speaker_clip_info if speaker_mode else binaural_clip_info,
|
||||
"output_wav": speaker_wav_info if speaker_mode else binaural_wav_info,
|
||||
"streaming_adm": mode_name == "adm",
|
||||
"kept_raw": str(raw_path) if raw_path is not None else None,
|
||||
"timings": timings,
|
||||
"total_seconds": total_seconds,
|
||||
"adm_validation": info if mode_name == "adm" else None,
|
||||
"adm_metadata": getattr(master, "metadata_info", None) if master is not None else None,
|
||||
"sha256": output_sha,
|
||||
"python": platform.python_version(),
|
||||
"numpy": np.__version__,
|
||||
"ffmpeg": decode_option_info,
|
||||
}
|
||||
report_path = Path(str(output) + ".report.json")
|
||||
report_path.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
print(f"[PASS] {output}")
|
||||
if report["sha256"] is None:
|
||||
print(f"[PASS] {info}; SHA-256 skipped")
|
||||
else:
|
||||
print(f"[PASS] {info}; SHA-256={report['sha256']}")
|
||||
if speaker_mode:
|
||||
print(f"[time] JOC-DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
|
||||
f"speaker={render_breakdown['speaker_render_seconds']:.2f}s "
|
||||
f"pipeline={render_breakdown['pipeline_wall_seconds']:.2f}s "
|
||||
f"total={report['total_seconds']:.2f}s")
|
||||
elif binaural_mode:
|
||||
print(f"[time] JOC-DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
|
||||
f"binaural={render_breakdown['binaural_render_seconds']:.2f}s "
|
||||
f"pipeline={render_breakdown['pipeline_wall_seconds']:.2f}s "
|
||||
f"total={report['total_seconds']:.2f}s")
|
||||
else:
|
||||
print(f"[time] DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
|
||||
f"render+ADM-stream={render_breakdown['pipeline_wall_seconds']:.2f}s "
|
||||
f"total={report['total_seconds']:.2f}s")
|
||||
print(f"[report] {report_path}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
raise SystemExit(main())
|
||||
except KeyboardInterrupt:
|
||||
raise SystemExit(130)
|
||||
@@ -0,0 +1,74 @@
|
||||
cmake_minimum_required(VERSION 3.20)
|
||||
|
||||
project(eac3joc_core VERSION 1.0.0 LANGUAGES CXX)
|
||||
|
||||
include(GNUInstallDirs)
|
||||
find_package(Threads REQUIRED)
|
||||
|
||||
add_library(eac3joc_core SHARED
|
||||
src/eac3joc_core.cpp
|
||||
src/speaker_renderer.cpp
|
||||
src/binaural_renderer.cpp
|
||||
src/sofa_binaural_renderer.cpp
|
||||
src/joc_huffman_tables.h
|
||||
src/qmf_tables.h
|
||||
src/speaker_layouts.h
|
||||
)
|
||||
|
||||
target_compile_features(eac3joc_core PRIVATE cxx_std_20)
|
||||
target_include_directories(eac3joc_core PRIVATE
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/include"
|
||||
)
|
||||
target_link_libraries(eac3joc_core PRIVATE Threads::Threads)
|
||||
|
||||
set_target_properties(eac3joc_core PROPERTIES
|
||||
OUTPUT_NAME "eac3joc_core"
|
||||
CXX_VISIBILITY_PRESET hidden
|
||||
VISIBILITY_INLINES_HIDDEN YES
|
||||
POSITION_INDEPENDENT_CODE YES
|
||||
)
|
||||
|
||||
if(MSVC)
|
||||
set_property(TARGET eac3joc_core PROPERTY
|
||||
MSVC_RUNTIME_LIBRARY "MultiThreaded$<$<CONFIG:Debug>:Debug>")
|
||||
target_compile_options(eac3joc_core PRIVATE
|
||||
/W4 /permissive- /Zc:__cplusplus /utf-8 /EHsc /GR- /fp:precise
|
||||
$<$<CONFIG:Release>:/O2>
|
||||
$<$<CONFIG:Release>:/Oi>
|
||||
$<$<CONFIG:Release>:/GL>
|
||||
)
|
||||
target_link_options(eac3joc_core PRIVATE
|
||||
$<$<CONFIG:Release>:/LTCG>
|
||||
/INCREMENTAL:NO /OPT:REF /OPT:ICF
|
||||
)
|
||||
else()
|
||||
target_compile_options(eac3joc_core PRIVATE
|
||||
-Wall -Wextra -Wpedantic -fno-fast-math
|
||||
$<$<CONFIG:Release>:-O3>
|
||||
)
|
||||
endif()
|
||||
|
||||
# Install directly into the chosen prefix. Recommended invocation from the
|
||||
# repository root:
|
||||
# cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release \
|
||||
# -DCMAKE_INSTALL_PREFIX=<repo>/lib
|
||||
# cmake --build build/cmake --config Release
|
||||
# cmake --install build/cmake --config Release
|
||||
#
|
||||
# Result names are supplied by the platform toolchain:
|
||||
# Windows: eac3joc_core.dll (+ eac3joc_core.lib import library)
|
||||
# Linux: libeac3joc_core.so
|
||||
# macOS: libeac3joc_core.dylib
|
||||
install(TARGETS eac3joc_core
|
||||
RUNTIME DESTINATION .
|
||||
LIBRARY DESTINATION .
|
||||
ARCHIVE DESTINATION .
|
||||
)
|
||||
|
||||
if(MSVC)
|
||||
install(FILES "$<TARGET_PDB_FILE:eac3joc_core>"
|
||||
DESTINATION .
|
||||
OPTIONAL
|
||||
)
|
||||
endif()
|
||||
|
||||
@@ -2,11 +2,7 @@
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
/* EJOC_STATIC: this branch exists for building the sources directly into an application, where nothing is imported or exported. */
|
||||
#if defined(EJOC_STATIC)
|
||||
#define EJOC_API
|
||||
#define EJOC_CALL __cdecl
|
||||
#elif defined(_WIN32)
|
||||
#if defined(_WIN32)
|
||||
#if defined(EJOC_BUILD_DLL)
|
||||
#define EJOC_API __declspec(dllexport)
|
||||
#else
|
||||
@@ -1,53 +1,3 @@
|
||||
// Derived from the upstream native/src/sofa_binaural_renderer.cpp (JustOneCacophony
|
||||
// @ 6bc2c2885666bb151bb66af93472199af9a99b81, sha256 81b485e4c71907672ab308ddc38228
|
||||
// 4acf29161cee93d7906785a4cce58941c7) and a drop-in replacement for it: the same
|
||||
// ejoc_sofa_binaural_* C ABI. The changes are table hoisting, result memoisation,
|
||||
// input validation and a dispatched SIMD cascade -- no arithmetic is reordered,
|
||||
// reassociated or contracted, so the output is bit-identical to the upstream
|
||||
// implementation (verified: identical FNV-1a over the exact output bit patterns at
|
||||
// 8/64/512/2000 blocks -- ef7ca498a6bb19b0 at 2000 -- and identical SHA-256 of the
|
||||
// joc_dump --binaural-out float64 stream for 1000/2000/4000 frames):
|
||||
//
|
||||
// The deviations from the upstream implementation, and why each one keeps the
|
||||
// output bit-identical:
|
||||
//
|
||||
// Hoisted invariants. QmfAnalysis::configure builds the two modulation tables
|
||||
// (cos/sin of -kPi*p/128 and of -3.0*(b+0.5)*kPi/128) once, instead of
|
||||
// recomputing 384 transcendental calls per slot and per channel (49,152 per
|
||||
// 512-sample block); RealSh::evaluate hoists the normalization (which depends
|
||||
// only on (degree, |order|)) and the six pmm_value(m, x) Legendre seeds out of
|
||||
// the 36-term loop. In both cases the stored values are std::cos/std::sin of
|
||||
// the identical double expressions, evaluated once, so they are the same bits.
|
||||
//
|
||||
// Memoised results. set_source memoises the seven paths it derives from a
|
||||
// source position. The wrapper calls it for every source on every 512-sample
|
||||
// block and the timeline holds a position constant between OAMD updates, so
|
||||
// most calls were rebuilding an identical result. A hit requires the exact bit
|
||||
// pattern of every input make_path reads to match the call that produced the
|
||||
// stored paths; the fade state machine and the late-send envelope are
|
||||
// unchanged, so the emitted samples are the same bits.
|
||||
//
|
||||
// Input validation. configure_room rejects zero FDN / allpass delay-line
|
||||
// lengths, which would otherwise reach an integer divide-by-zero from the
|
||||
// public C ABI, and render_paths wraps the history index with an explicit
|
||||
// conditional that is identical to `% slots` for every index in [0, 2*slots) --
|
||||
// which is every index the shipped room can form -- turning a negative index
|
||||
// (a delay longer than the 256-slot early history) into an in-range one instead
|
||||
// of reading out of bounds. Neither can change an accepted input.
|
||||
//
|
||||
// Dispatched SIMD cascade. Fft128::butterflies hands the 128-point cascade,
|
||||
// QmfSynthesis::process the basis application and HybridAnalysis::process the
|
||||
// 78-term low join to the runtime-dispatched kernels in src/simd. Only
|
||||
// mutually independent outputs share a vector lane and every lane keeps the
|
||||
// corresponding loop's own term order and its own two roundings: each butterfly
|
||||
// keeps its two, the synthesis keeps the j = 0..127 order with one multiply and
|
||||
// one add per term, and the hybrid join keeps its 32 independent accumulations.
|
||||
// The kernels read their tables contiguously, so the constructors materialise
|
||||
// them in the order the kernel wants (same doubles, same order, same table).
|
||||
//
|
||||
// Under JOC_SIMD=scalar and under every forced SIMD tier the output is still
|
||||
// bit-identical: the joc_dump --binaural-out stream and the rendered WAV keep the
|
||||
// SHA-256 they had before the change.
|
||||
#define EJOC_BUILD_DLL
|
||||
#include "eac3joc_core.h"
|
||||
|
||||
@@ -55,15 +5,11 @@
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cmath>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <new>
|
||||
#include <vector>
|
||||
|
||||
#include "simd/simd.h"
|
||||
|
||||
namespace ejoc::sofa_binaural {
|
||||
|
||||
constexpr double kPi = 3.141592653589793238462643383279502884;
|
||||
@@ -112,51 +58,36 @@ public:
|
||||
const double angle = -2.0 * kPi * k / kFftSize;
|
||||
twiddle_[k] = {std::cos(angle), std::sin(angle)};
|
||||
}
|
||||
// The bit-reversal permutation depends only on the index, so it is
|
||||
// derived once here instead of by a seven-iteration bit loop inside every
|
||||
// one of the 256 transforms a 512-sample block runs. Same indices, same
|
||||
// destinations, same store order.
|
||||
}
|
||||
|
||||
void forward(const Complex* input, Complex* output) const noexcept {
|
||||
// bit-reversal permutation for the 7-bit index (DIT)
|
||||
for (int index = 0; index < kFftSize; ++index) {
|
||||
unsigned int reversed = 0;
|
||||
for (int bit = 0; bit < 7; ++bit) {
|
||||
reversed = (reversed << 1)
|
||||
| ((static_cast<unsigned int>(index) >> bit) & 1u);
|
||||
}
|
||||
reverse_[index] = static_cast<int>(reversed);
|
||||
output[static_cast<int>(reversed)] = input[index];
|
||||
}
|
||||
// The dispatched butterfly kernel reads a stage's factors as one
|
||||
// contiguous run, so the per-stage stride `offset * step` is resolved once
|
||||
// here. The entries are copies of the same doubles in the same order.
|
||||
int cursor = 0;
|
||||
int stage = 0;
|
||||
for (int size = 2; size <= kFftSize; size <<= 1, ++stage) {
|
||||
stage_begin_[stage] = static_cast<std::size_t>(cursor);
|
||||
for (int size = 2; size <= kFftSize; size <<= 1) {
|
||||
const int half = size >> 1;
|
||||
const int step = kFftSize / size;
|
||||
for (int offset = 0; offset < size / 2; ++offset) {
|
||||
stage_twiddle_[cursor] = twiddle_[offset * step];
|
||||
++cursor;
|
||||
for (int base = 0; base < kFftSize; base += size) {
|
||||
for (int offset = 0; offset < half; ++offset) {
|
||||
const Complex w = twiddle_[offset * step];
|
||||
const Complex even = output[base + offset];
|
||||
const Complex odd = mul(output[base + offset + half], w);
|
||||
output[base + offset] = add(even, odd);
|
||||
output[base + offset + half] = {
|
||||
even.re - odd.re, even.im - odd.im};
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The transform is split so a caller that can produce its input already
|
||||
// bit-reversed does not have to stage it through a second array and permute it
|
||||
// afterwards. `permuted(index)` is the destination the permutation used, and
|
||||
// `butterflies` runs the stages in place on a permuted buffer.
|
||||
int permuted(int index) const noexcept { return reverse_[index]; }
|
||||
|
||||
void butterflies(Complex* data) const noexcept {
|
||||
joc::simd::fft_butterflies(reinterpret_cast<double*>(data), kFftSize,
|
||||
reinterpret_cast<const double*>(stage_twiddle_),
|
||||
stage_begin_);
|
||||
}
|
||||
|
||||
private:
|
||||
Complex twiddle_[kFftSize];
|
||||
int reverse_[kFftSize];
|
||||
// 1 + 2 + 4 + ... + 64 factors, one contiguous run per stage.
|
||||
Complex stage_twiddle_[kFftSize - 1];
|
||||
std::size_t stage_begin_[7];
|
||||
};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -166,16 +97,6 @@ class QmfAnalysis final {
|
||||
public:
|
||||
void configure(const double* coefficients) noexcept {
|
||||
std::memcpy(coeff_, coefficients, sizeof(coeff_));
|
||||
for (int p = 0; p < kQmf; ++p) {
|
||||
const double pre_angle = -kPi * p / kFftSize;
|
||||
pre_cos_[p] = std::cos(pre_angle);
|
||||
pre_sin_[p] = std::sin(pre_angle);
|
||||
}
|
||||
for (int b = 0; b < kQmf; ++b) {
|
||||
const double post_angle = -3.0 * (b + 0.5) * kPi / kFftSize;
|
||||
post_cos_[b] = std::cos(post_angle);
|
||||
post_sin_[b] = std::sin(post_angle);
|
||||
}
|
||||
}
|
||||
|
||||
void reset() noexcept {
|
||||
@@ -199,17 +120,11 @@ public:
|
||||
}
|
||||
for (int slot = 0; slot < slots; ++slot) {
|
||||
for (int channel = 0; channel < kChannels; ++channel) {
|
||||
Complex even_fft[kFftSize];
|
||||
Complex odd_fft[kFftSize];
|
||||
// The polyphase sums are written straight to the bit-reversed
|
||||
// positions the DIT permutation would have put them in, so the two
|
||||
// staging arrays and the permutation pass over them are gone. Each
|
||||
// destination is still written exactly once, with the same value, so
|
||||
// the permuted buffers hold the same bits as before, and only the zero-padded
|
||||
// upper half still needs clearing.
|
||||
for (int p = kQmf; p < kFftSize; ++p) {
|
||||
even_fft[fft_.permuted(p)] = {0.0, 0.0};
|
||||
odd_fft[fft_.permuted(p)] = {0.0, 0.0};
|
||||
Complex even[kFftSize];
|
||||
Complex odd[kFftSize];
|
||||
for (int p = 0; p < kFftSize; ++p) {
|
||||
even[p] = {0.0, 0.0};
|
||||
odd[p] = {0.0, 0.0};
|
||||
}
|
||||
// 64 polyphase positions; the 128-point FFT zero-pads the rest
|
||||
for (int p = 0; p < kQmf; ++p) {
|
||||
@@ -223,17 +138,20 @@ public:
|
||||
odd_re += joined_[9 + slot - (2 * lag + 1)][channel][p]
|
||||
* coeff_[p][2 * lag + 1];
|
||||
}
|
||||
even_fft[fft_.permuted(p)] = {even_re * pre_cos_[p],
|
||||
even_re * pre_sin_[p]};
|
||||
odd_fft[fft_.permuted(p)] = {odd_re * pre_cos_[p],
|
||||
odd_re * pre_sin_[p]};
|
||||
const double pre_angle = -kPi * p / kFftSize;
|
||||
even[p] = {even_re * std::cos(pre_angle),
|
||||
even_re * std::sin(pre_angle)};
|
||||
odd[p] = {odd_re * std::cos(pre_angle), odd_re * std::sin(pre_angle)};
|
||||
}
|
||||
fft_.butterflies(even_fft);
|
||||
fft_.butterflies(odd_fft);
|
||||
Complex even_fft[kFftSize];
|
||||
Complex odd_fft[kFftSize];
|
||||
fft_.forward(even, even_fft);
|
||||
fft_.forward(odd, odd_fft);
|
||||
Complex* band = output
|
||||
+ (static_cast<size_t>(slot) * kChannels + channel) * kQmf;
|
||||
for (int b = 0; b < kQmf; ++b) {
|
||||
const Complex post = {post_cos_[b], post_sin_[b]};
|
||||
const double post_angle = -3.0 * (b + 0.5) * kPi / kFftSize;
|
||||
const Complex post = {std::cos(post_angle), std::sin(post_angle)};
|
||||
const Complex even_post = {0.0, (b % 2 == 0) ? 1.0 : -1.0};
|
||||
// python: (odd_fft + even_fft * even_post) * post
|
||||
band[b] = mul(post, add(odd_fft[b],
|
||||
@@ -248,12 +166,6 @@ public:
|
||||
|
||||
private:
|
||||
double coeff_[kQmf][10];
|
||||
// The two modulation tables depend only on the loop index, so they are
|
||||
// built once with the identical expressions used in process().
|
||||
double pre_cos_[kQmf];
|
||||
double pre_sin_[kQmf];
|
||||
double post_cos_[kQmf];
|
||||
double post_sin_[kQmf];
|
||||
double history_[9][kChannels][64];
|
||||
double joined_[9 + kMaxBlockSamples / kHop][kChannels][64];
|
||||
Fft128 fft_;
|
||||
@@ -266,21 +178,6 @@ class HybridAnalysis final {
|
||||
public:
|
||||
void configure(const double* low_kernel) noexcept {
|
||||
std::memcpy(low_, low_kernel, sizeof(low_));
|
||||
// The dispatched join walks its terms in the order this class
|
||||
// accumulates them -- parent, then component, then lag -- and adds the 32
|
||||
// weights of one term at once, so the table is regrouped once here. Same
|
||||
// weights, same order.
|
||||
int cursor = 0;
|
||||
for (int parent = 0; parent < 3; ++parent) {
|
||||
for (int comp = 0; comp < 2; ++comp) {
|
||||
for (int lag = 0; lag < 13; ++lag) {
|
||||
for (int hb = 0; hb < 16; ++hb) {
|
||||
low_by_term_[cursor++] = low_[parent][comp][lag][hb][0];
|
||||
low_by_term_[cursor++] = low_[parent][comp][lag][hb][1];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void reset() noexcept {
|
||||
@@ -310,46 +207,24 @@ public:
|
||||
}
|
||||
}
|
||||
}
|
||||
// The low bands are a 78-term join into 32 outputs, and the 32 outputs
|
||||
// of a row are independent accumulations over the same terms -- that is what
|
||||
// the dispatched kernel puts in its lanes, keeping this loop's term order
|
||||
// (parent, component, lag) and its two roundings per term. Rows are staged
|
||||
// in blocks so the gathered values stay in the first-level cache.
|
||||
const std::size_t rows = static_cast<std::size_t>(slots) * kChannels;
|
||||
const std::size_t block = joc::simd::kHybridJoinBlock;
|
||||
const std::size_t terms = joc::simd::kHybridTerms;
|
||||
for (std::size_t first = 0u; first < rows; first += block) {
|
||||
const std::size_t count = std::min(block, rows - first);
|
||||
for (std::size_t index = 0u; index < count; ++index) {
|
||||
const std::size_t row = first + index;
|
||||
const int slot = static_cast<int>(row / kChannels);
|
||||
const int channel = static_cast<int>(row % kChannels);
|
||||
double* staged = low_values_ + index * terms;
|
||||
for (int parent = 0; parent < 3; ++parent) {
|
||||
for (int comp = 0; comp < 2; ++comp) {
|
||||
for (int lag = 0; lag < 13; ++lag) {
|
||||
staged[(static_cast<std::size_t>(parent) * 2u +
|
||||
static_cast<std::size_t>(comp)) * 13u +
|
||||
static_cast<std::size_t>(lag)] =
|
||||
low_joined_[12 + slot - lag][channel][parent][comp];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
joc::simd::hybrid_low_join(low_values_, low_by_term_, low_out_, count);
|
||||
for (std::size_t index = 0u; index < count; ++index) {
|
||||
Complex* out = output + (first + index) * kHybrid;
|
||||
const double* values = low_out_ + index * joc::simd::kHybridOutputs;
|
||||
for (int hb = 0; hb < 16; ++hb) {
|
||||
out[hb] = {values[static_cast<std::size_t>(hb) * 2u],
|
||||
values[static_cast<std::size_t>(hb) * 2u + 1u]};
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int slot = 0; slot < slots; ++slot) {
|
||||
for (int channel = 0; channel < kChannels; ++channel) {
|
||||
Complex* out = output
|
||||
+ (static_cast<size_t>(slot) * kChannels + channel) * kHybrid;
|
||||
for (int hb = 0; hb < 16; ++hb) {
|
||||
Complex value = {0.0, 0.0};
|
||||
for (int parent = 0; parent < 3; ++parent) {
|
||||
for (int comp = 0; comp < 2; ++comp) {
|
||||
for (int lag = 0; lag < 13; ++lag) {
|
||||
const double source =
|
||||
low_joined_[12 + slot - lag][channel][parent][comp];
|
||||
value.re += source * low_[parent][comp][lag][hb][0];
|
||||
value.im += source * low_[parent][comp][lag][hb][1];
|
||||
}
|
||||
}
|
||||
}
|
||||
out[hb] = value;
|
||||
}
|
||||
for (int b = 0; b < 61; ++b) {
|
||||
out[16 + b] = high_joined_[slot][channel][b];
|
||||
}
|
||||
@@ -366,9 +241,6 @@ public:
|
||||
|
||||
private:
|
||||
double low_[3][2][13][16][2];
|
||||
double low_by_term_[78 * 32]; // S3: the same weights in this class's term order
|
||||
double low_values_[joc::simd::kHybridJoinBlock * 78]; // scratch
|
||||
double low_out_[joc::simd::kHybridJoinBlock * 32]; // scratch
|
||||
double history_[12][kChannels][3][2];
|
||||
double low_joined_[12 + kMaxBlockSamples / kHop][kChannels][3][2];
|
||||
Complex high_history_[6][kChannels][61];
|
||||
@@ -437,18 +309,6 @@ public:
|
||||
void configure(const double* basis, const double* taps) noexcept {
|
||||
std::memcpy(basis_, basis, sizeof(basis_));
|
||||
std::memcpy(taps_, taps, sizeof(taps_));
|
||||
// The dispatched basis kernel reads the four ranks of one (band, tap)
|
||||
// as one vector, so the shipped [band][rank][tap] table is reordered once
|
||||
// here. Same doubles, different order.
|
||||
int cursor = 0;
|
||||
for (int b = 0; b < kQmf; ++b) {
|
||||
for (int j = 0; j < kFftSize; ++j) {
|
||||
for (int r = 0; r < kRank; ++r) {
|
||||
basis_by_tap_[cursor] = basis_[b][r][j];
|
||||
++cursor;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void reset() noexcept {
|
||||
@@ -461,27 +321,25 @@ public:
|
||||
std::memcpy(joined_[lag], history_[lag], sizeof(joined_[0]));
|
||||
}
|
||||
for (int slot = 0; slot < slots; ++slot) {
|
||||
// Both channels consume the same basis_[b][r][*] row. The
|
||||
// dispatched kernel does the same thing a vector at a time: the
|
||||
// four ranks of a band are four independent dot products over one
|
||||
// channel's 128 values, so they fill the lanes while each lane keeps
|
||||
// the original j = 0..127 order and its own two roundings.
|
||||
for (int channel = 0; channel < 2; ++channel) {
|
||||
const Complex* values = qmf
|
||||
+ (static_cast<size_t>(slot) * 2 + channel) * kQmf;
|
||||
double flat[kFftSize];
|
||||
for (int b = 0; b < kQmf; ++b) {
|
||||
flat_[slot][channel][2 * b] = values[b].re;
|
||||
flat_[slot][channel][2 * b + 1] = values[b].im;
|
||||
flat[2 * b] = values[b].re;
|
||||
flat[2 * b + 1] = values[b].im;
|
||||
}
|
||||
for (int b = 0; b < kQmf; ++b) {
|
||||
for (int r = 0; r < kRank; ++r) {
|
||||
double value = 0.0;
|
||||
for (int j = 0; j < kFftSize; ++j) {
|
||||
value += flat[j] * basis_[b][r][j];
|
||||
}
|
||||
joined_[9 + slot][channel][b][r] = value;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Every slot is handed over at once rather than one at a time: a band's
|
||||
// accumulation is 128 dependent adds, and the kernel hides that latency by
|
||||
// running several rows side by side, so it needs more than the two rows of
|
||||
// a single slot to work with.
|
||||
joc::simd::qmf_synthesis_basis(&flat_[0][0][0], basis_by_tap_,
|
||||
&joined_[9][0][0][0],
|
||||
static_cast<std::size_t>(slots) * 2u);
|
||||
for (int slot = 0; slot < slots; ++slot) {
|
||||
for (int channel = 0; channel < 2; ++channel) {
|
||||
for (int b = 0; b < kQmf; ++b) {
|
||||
@@ -504,9 +362,7 @@ public:
|
||||
|
||||
private:
|
||||
double basis_[kQmf][kRank][kFftSize];
|
||||
double basis_by_tap_[kQmf * kFftSize * kRank]; // S2: the same weights, transposed
|
||||
double taps_[kQmf][10][kRank];
|
||||
double flat_[kMaxBlockSamples / kHop][2][kFftSize]; // S2: staging for the kernel
|
||||
double history_[9][2][kQmf][kRank];
|
||||
double joined_[9 + kMaxBlockSamples / kHop][2][kQmf][kRank];
|
||||
};
|
||||
@@ -519,27 +375,15 @@ public:
|
||||
static void evaluate(const double* direction, double* basis) noexcept {
|
||||
const double azimuth = std::atan2(direction[1], direction[0]);
|
||||
const double x = std::max(-1.0, std::min(1.0, direction[2]));
|
||||
// Normalization depends only on (degree, |order|) and the Legendre
|
||||
// seed pmm_value(m, x) only on (m, x): 6 seeds, not 36 recomputations.
|
||||
double pmm[kOrder + 1];
|
||||
for (int m = 0; m <= kOrder; ++m) {
|
||||
pmm[m] = pmm_value(m, x);
|
||||
}
|
||||
double norm[kOrder + 1][kOrder + 1];
|
||||
for (int degree = 0; degree <= kOrder; ++degree) {
|
||||
for (int absolute = 0; absolute <= degree; ++absolute) {
|
||||
norm[degree][absolute] = std::sqrt(
|
||||
(2.0 * degree + 1.0) / (4.0 * kPi)
|
||||
* factorial_ratio(degree, absolute));
|
||||
}
|
||||
}
|
||||
int index = 0;
|
||||
for (int degree = 0; degree <= kOrder; ++degree) {
|
||||
for (int order = -degree; order <= degree; ++order) {
|
||||
const int absolute = std::abs(order);
|
||||
const double normalization = norm[degree][absolute];
|
||||
const double normalization = std::sqrt(
|
||||
(2.0 * degree + 1.0) / (4.0 * kPi)
|
||||
* factorial_ratio(degree, absolute));
|
||||
const double legendre = associated_legendre(
|
||||
degree, absolute, x, pmm[absolute]);
|
||||
degree, absolute, x, pmm_value(absolute, x));
|
||||
if (order > 0) {
|
||||
basis[index] = std::sqrt(2.0) * normalization * legendre
|
||||
* std::cos(order * azimuth);
|
||||
@@ -623,20 +467,6 @@ struct SourceState {
|
||||
double late_target;
|
||||
int64_t late_fade_position;
|
||||
int64_t late_fade_total;
|
||||
// The seven paths are a pure function of (position, profile, effective
|
||||
// gain, special_lfe) plus the configuration that only configure_* can change.
|
||||
// The wrapper calls set_source for every source on every 512-sample block and
|
||||
// the timeline holds a position constant between OAMD updates, so the same
|
||||
// paths were rebuilt from scratch over and over. The memo stores the last
|
||||
// result and the exact bit pattern of the arguments that produced it; a hit
|
||||
// reuses those bits instead of recomputing them. The fade state machine and
|
||||
// the late-send envelope below run identically either way.
|
||||
double memo_position[3] = {0.0, 0.0, 0.0};
|
||||
double memo_effective = 0.0;
|
||||
int memo_profile = 0;
|
||||
int memo_special_lfe = 0;
|
||||
int memo_valid = 0;
|
||||
std::vector<Path> memo_paths;
|
||||
};
|
||||
|
||||
constexpr double kDistanceM[3] = {1.00000465, 2.19327927, 6.40177584};
|
||||
@@ -710,18 +540,6 @@ public:
|
||||
|| !(damping >= 0.0 && damping < 1.0)) {
|
||||
return fail("invalid sofa binaural room configuration");
|
||||
}
|
||||
// A zero delay-line length makes the per-sample `% delay` an integer
|
||||
// divide by zero. Rejecting it here cannot change any accepted input.
|
||||
for (int line = 0; line < 4; ++line) {
|
||||
if (fdn_delays[line] == 0u) {
|
||||
return fail("invalid sofa binaural room configuration");
|
||||
}
|
||||
}
|
||||
for (int line = 0; line < 2; ++line) {
|
||||
if (allpass_delays[line] == 0u) {
|
||||
return fail("invalid sofa binaural room configuration");
|
||||
}
|
||||
}
|
||||
std::memcpy(dims_, dims, sizeof(dims_));
|
||||
std::memcpy(listener_, listener, sizeof(listener_));
|
||||
std::memcpy(wall_gain_, wall_gains, sizeof(wall_gain_));
|
||||
@@ -776,43 +594,23 @@ public:
|
||||
state.special_lfe = special_lfe != 0;
|
||||
|
||||
const double effective = enabled ? gain : 0.0;
|
||||
// Reuse the memoised paths when the arguments are bit-identical to the
|
||||
// ones that produced them. Everything make_path reads -- the direction and
|
||||
// distance derived from `position`, the profile, the effective gain, the LFE
|
||||
// flag, and the field/room tables -- is covered by this key or fixed by
|
||||
// configure_*, and the side effect ordinary_paths has on
|
||||
// maximum_early_delay_ is a maximum of the same values, so reusing the
|
||||
// stored bits is the same as recomputing them.
|
||||
const int memo_special_lfe = state.special_lfe ? 1 : 0;
|
||||
const bool memo_hit = state.memo_valid != 0 && state.memo_profile == state.profile
|
||||
&& state.memo_special_lfe == memo_special_lfe
|
||||
&& std::memcmp(state.memo_position, position, sizeof(state.memo_position)) == 0
|
||||
&& std::memcmp(&state.memo_effective, &effective, sizeof(effective)) == 0;
|
||||
std::vector<Path>& paths = state.memo_paths;
|
||||
std::vector<Path> paths;
|
||||
double late_send = 0.0;
|
||||
double direction[3];
|
||||
double radius;
|
||||
normalize_adm(position, direction, radius);
|
||||
const double distance =
|
||||
std::max(kMinimumDistance, radius * kDistanceM[state.profile]);
|
||||
if (!memo_hit) {
|
||||
paths.clear();
|
||||
if (state.special_lfe) {
|
||||
paths.push_back(lfe_path(effective));
|
||||
} else {
|
||||
paths = ordinary_paths(direction, distance, state.profile, effective);
|
||||
if (state.special_lfe) {
|
||||
paths.push_back(lfe_path(effective));
|
||||
} else {
|
||||
paths = ordinary_paths(direction, distance, state.profile, effective);
|
||||
if (enabled && enable_late_room_) {
|
||||
const double base = kLateSend[state.profile];
|
||||
const double radial = std::min(
|
||||
std::max(std::sqrt(std::max(radius, 0.0)), 0.25), 1.5);
|
||||
late_send = effective * kRoomCalibration * base * radial;
|
||||
}
|
||||
std::memcpy(state.memo_position, position, sizeof(state.memo_position));
|
||||
std::memcpy(&state.memo_effective, &effective, sizeof(effective));
|
||||
state.memo_profile = state.profile;
|
||||
state.memo_special_lfe = memo_special_lfe;
|
||||
state.memo_valid = 1;
|
||||
}
|
||||
if (!state.special_lfe && enabled && enable_late_room_) {
|
||||
const double base = kLateSend[state.profile];
|
||||
const double radial = std::min(
|
||||
std::max(std::sqrt(std::max(radius, 0.0)), 0.25), 1.5);
|
||||
late_send = effective * kRoomCalibration * base * radial;
|
||||
}
|
||||
set_late_target(source, late_send, fade != 0);
|
||||
|
||||
@@ -910,28 +708,21 @@ private:
|
||||
RealSh::evaluate(listener_direction, basis);
|
||||
Complex aligned[2][kHybrid];
|
||||
double delay[2];
|
||||
double delay_value[2] = {0.0, 0.0};
|
||||
for (int ear = 0; ear < 2; ++ear) {
|
||||
double delay_value = 0.0;
|
||||
for (int band = 0; band < kHybrid; ++band) {
|
||||
aligned[ear][band] = {0.0, 0.0};
|
||||
}
|
||||
}
|
||||
// The 36 spherical-harmonic terms are independent contributions to the
|
||||
// same 2 x 77 bands, so the band axis is what the dispatched accumulate
|
||||
// spreads across its lanes. Each band still adds `field * term` once per
|
||||
// term, in term order, with the same two roundings; only the delay sum --
|
||||
// which is a reduction -- stays scalar and keeps its own order.
|
||||
for (int term = 0; term < kTerms; ++term) {
|
||||
for (int ear = 0; ear < 2; ++ear) {
|
||||
delay_value[ear] += basis[term] * delay_coeff_[term][ear];
|
||||
for (int term = 0; term < kTerms; ++term) {
|
||||
delay_value += basis[term] * delay_coeff_[term][ear];
|
||||
for (int band = 0; band < kHybrid; ++band) {
|
||||
aligned[ear][band].re +=
|
||||
field_coeff_[term][ear][band].re * basis[term];
|
||||
aligned[ear][band].im +=
|
||||
field_coeff_[term][ear][band].im * basis[term];
|
||||
}
|
||||
}
|
||||
joc::simd::complex_axpy(
|
||||
reinterpret_cast<double*>(&aligned[0][0]),
|
||||
reinterpret_cast<const double*>(&field_coeff_[term][0][0]), basis[term],
|
||||
2u * kHybrid);
|
||||
}
|
||||
for (int ear = 0; ear < 2; ++ear) {
|
||||
delay[ear] = std::min(std::max(delay_value[ear], delay_bounds_[ear][0]),
|
||||
delay[ear] = std::min(std::max(delay_value, delay_bounds_[ear][0]),
|
||||
delay_bounds_[ear][1]);
|
||||
}
|
||||
Path path;
|
||||
@@ -1040,14 +831,8 @@ private:
|
||||
|
||||
void process_internal(const double* input, int slots, double output_gain) noexcept {
|
||||
const uint32_t sample_count = static_cast<uint32_t>(slots) * kHop;
|
||||
// These four planes are written in full before they are read, so they
|
||||
// are reused scratch buffers. `assign` keeps the capacity, so after the
|
||||
// first block each one is a fill with no allocation, and the fills that are
|
||||
// load-bearing (the mono and late accumulators, and the direct/early output
|
||||
// that is only added into) are preserved exactly.
|
||||
// 1) late send envelopes -> mono
|
||||
std::vector<double>& mono = mono_;
|
||||
mono.assign(sample_count, 0.0);
|
||||
std::vector<double> mono(sample_count, 0.0);
|
||||
for (int source = 0; source < kChannels; ++source) {
|
||||
SourceState& state = sources_[source];
|
||||
const int64_t total = state.late_fade_total;
|
||||
@@ -1089,15 +874,12 @@ private:
|
||||
}
|
||||
|
||||
// 2) FDN + 961-sample stereo delay
|
||||
std::vector<double>& late_pcm = late_pcm_;
|
||||
late_pcm.assign(static_cast<size_t>(sample_count) * 2, 0.0);
|
||||
std::vector<double> late_pcm(static_cast<size_t>(sample_count) * 2, 0.0);
|
||||
if (enable_late_room_) {
|
||||
diffused_.assign(mono.begin(), mono.end());
|
||||
std::vector<double> diffused = mono;
|
||||
for (int line = 0; line < 2; ++line) {
|
||||
allpass(line, diffused_, &allpass_work_);
|
||||
diffused_.swap(allpass_work_);
|
||||
diffused = allpass(line, diffused);
|
||||
}
|
||||
const std::vector<double>& diffused = diffused_;
|
||||
for (uint32_t s = 0; s < sample_count; ++s) {
|
||||
const double value = diffused[s];
|
||||
double delayed[4];
|
||||
@@ -1156,8 +938,8 @@ private:
|
||||
analysis_.process(input ? input : zeros_.data(), slots, qmf_work_.data());
|
||||
hybrid_analysis_.process(qmf_work_.data(), slots, hybrid_work_.data());
|
||||
// 4) per-object direct/early with crossfade
|
||||
std::vector<Complex>& direct_and_early = direct_early_;
|
||||
direct_and_early.assign(static_cast<size_t>(slots) * 2 * kHybrid, Complex{0.0, 0.0});
|
||||
std::vector<Complex> direct_and_early(
|
||||
static_cast<size_t>(slots) * 2 * kHybrid, Complex{0.0, 0.0});
|
||||
for (int slot = 0; slot < slots; ++slot) {
|
||||
const Complex* hybrid_slot =
|
||||
hybrid_work_.data() + static_cast<size_t>(slot) * kChannels * kHybrid;
|
||||
@@ -1214,32 +996,30 @@ private:
|
||||
Complex* out) const noexcept {
|
||||
for (const Path& path : paths) {
|
||||
const int slots = static_cast<int>(history_slots_);
|
||||
// The history index: position_ is in [0, slots), so for any delay_slots
|
||||
// <= slots the raw index lands in [0, 2*slots) and this conditional is
|
||||
// exactly `% slots`. A longer delay can make the raw index negative,
|
||||
// where C++ `%` would produce an out-of-bounds negative index; the
|
||||
// `raw < 0` arm wraps it into range instead.
|
||||
const int raw0 = static_cast<int>(position_) + slots - path.delay_slots[0];
|
||||
const int raw1 = static_cast<int>(position_) + slots - path.delay_slots[1];
|
||||
const int index0 = raw0 >= slots ? raw0 - slots : (raw0 < 0 ? raw0 + slots : raw0);
|
||||
const int index1 = raw1 >= slots ? raw1 - slots : (raw1 < 0 ? raw1 + slots : raw1);
|
||||
const int index0 = (static_cast<int>(position_) + slots
|
||||
- path.delay_slots[0]) % slots;
|
||||
const int index1 = (static_cast<int>(position_) + slots
|
||||
- path.delay_slots[1]) % slots;
|
||||
const Complex* history = history_[source].data();
|
||||
// The 77 bands of an ear are independent accumulations into
|
||||
// independent outputs, so they are what the dispatched kernel spreads
|
||||
// across its lanes; each lane keeps this loop's `h * t * scale` with its
|
||||
// own two roundings, and the ears read their own history rows.
|
||||
joc::simd::render_hybrid_path(
|
||||
reinterpret_cast<double*>(out),
|
||||
reinterpret_cast<const double*>(path.transfer),
|
||||
reinterpret_cast<const double*>(history + index0 * kHybrid),
|
||||
reinterpret_cast<const double*>(history + index1 * kHybrid), scale);
|
||||
for (int band = 0; band < kHybrid; ++band) {
|
||||
const Complex source0 = history[index0 * kHybrid + band];
|
||||
const Complex source1 = history[index1 * kHybrid + band];
|
||||
out[band].re += source0.re * path.transfer[0][band].re * scale
|
||||
- source0.im * path.transfer[0][band].im * scale;
|
||||
out[band].im += source0.re * path.transfer[0][band].im * scale
|
||||
+ source0.im * path.transfer[0][band].re * scale;
|
||||
out[kHybrid + band].re +=
|
||||
source1.re * path.transfer[1][band].re * scale
|
||||
- source1.im * path.transfer[1][band].im * scale;
|
||||
out[kHybrid + band].im +=
|
||||
source1.re * path.transfer[1][band].im * scale
|
||||
+ source1.im * path.transfer[1][band].re * scale;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void allpass(int line, const std::vector<double>& source,
|
||||
std::vector<double>* output) noexcept {
|
||||
// Every element is assigned below, so only the size has to be established.
|
||||
output->resize(source.size());
|
||||
std::vector<double> allpass(int line, const std::vector<double>& source) noexcept {
|
||||
std::vector<double> output(source.size(), 0.0);
|
||||
const double gain = allpass_gains_[line];
|
||||
const uint32_t delay = allpass_delays_[line];
|
||||
for (size_t i = 0; i < source.size(); ++i) {
|
||||
@@ -1248,8 +1028,9 @@ private:
|
||||
allpass_buffers_[line][allpass_positions_[line]] =
|
||||
source[i] + gain * result;
|
||||
allpass_positions_[line] = (allpass_positions_[line] + 1) % delay;
|
||||
(*output)[i] = result;
|
||||
output[i] = result;
|
||||
}
|
||||
return output;
|
||||
}
|
||||
|
||||
int fail(const char* message) noexcept {
|
||||
@@ -1268,13 +1049,6 @@ private:
|
||||
std::vector<Complex>(
|
||||
static_cast<size_t>(history_slots_) * kHybrid,
|
||||
Complex{0.0, 0.0}));
|
||||
// Re-assigning the source states is also what invalidates the
|
||||
// set_source path memo, because the memo fields are SourceState members
|
||||
// and a fresh SourceState starts with memo_valid == 0. Every configure_*
|
||||
// reaches reset_state() through reset(), so a reconfigured renderer can
|
||||
// never serve a path derived from the previous configuration. If this
|
||||
// function is ever changed to reset the fields in place, clear the memo
|
||||
// explicitly here instead.
|
||||
sources_.assign(kChannels, SourceState{});
|
||||
for (int source = 0; source < kChannels; ++source) {
|
||||
sources_[source].position[0] = 0.0;
|
||||
@@ -1360,12 +1134,6 @@ private:
|
||||
std::array<double, kMaxBlockSamples / kHop * 2 * 64> pcm_work_{};
|
||||
std::array<double, kMaxBlockSamples * 2> output_{};
|
||||
std::array<double, kMaxBlockSamples * kChannels> zeros_{};
|
||||
// Reused per-block scratch (sized once by the first block's `assign`).
|
||||
std::vector<double> mono_;
|
||||
std::vector<double> late_pcm_;
|
||||
std::vector<double> diffused_;
|
||||
std::vector<double> allpass_work_;
|
||||
std::vector<Complex> direct_early_;
|
||||
|
||||
char error_[256];
|
||||
bool enable_early_reflections_ = true;
|
||||
@@ -0,0 +1,3 @@
|
||||
numpy>=1.24
|
||||
scipy>=1.10
|
||||
h5py>=3.8
|
||||
@@ -1,333 +0,0 @@
|
||||
#include "adm/adm_metadata.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
|
||||
#include "foundation/py_num.h"
|
||||
|
||||
namespace joc::adm {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr const char* kBedNames[10] = {
|
||||
"RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter", "RoomCentricLFE",
|
||||
"RoomCentricLeftSideSurround", "RoomCentricRightSideSurround",
|
||||
"RoomCentricLeftRearSurround", "RoomCentricRightRearSurround",
|
||||
"RoomCentricLeftTopSurround", "RoomCentricRightTopSurround"};
|
||||
constexpr const char* kBedLabels[10] = {"RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss",
|
||||
"RC_Rss", "RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"};
|
||||
constexpr double kBedPos[10][3] = {{-1.0, 1.0, 0.0}, {1.0, 1.0, 0.0}, {0.0, 1.0, 0.0},
|
||||
{-1.0, 1.0, -1.0}, {-1.0, 0.0, 0.0}, {1.0, 0.0, 0.0},
|
||||
{-1.0, -1.0, 0.0}, {1.0, -1.0, 0.0}, {-1.0, 0.0, 1.0},
|
||||
{1.0, 0.0, 1.0}};
|
||||
|
||||
std::string hex4(std::uint32_t value) {
|
||||
char buffer[16];
|
||||
std::snprintf(buffer, sizeof(buffer), "%04x", value);
|
||||
return std::string(buffer);
|
||||
}
|
||||
|
||||
std::string hex8(std::uint32_t value) {
|
||||
char buffer[16];
|
||||
std::snprintf(buffer, sizeof(buffer), "%08x", value);
|
||||
return std::string(buffer);
|
||||
}
|
||||
|
||||
void put_u16(std::string* out, std::uint16_t value) {
|
||||
char buffer[2];
|
||||
std::memcpy(buffer, &value, 2);
|
||||
out->append(buffer, 2);
|
||||
}
|
||||
|
||||
void put_u32(std::string* out, std::uint32_t value) {
|
||||
char buffer[4];
|
||||
std::memcpy(buffer, &value, 4);
|
||||
out->append(buffer, 4);
|
||||
}
|
||||
|
||||
std::uint8_t checksum(const std::string& segment) {
|
||||
int sum = static_cast<int>(segment.size());
|
||||
for (const char raw : segment) {
|
||||
sum += static_cast<unsigned char>(raw);
|
||||
}
|
||||
return static_cast<std::uint8_t>((~sum + 1) & 0xFF);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::string ts(double seconds) {
|
||||
long long whole = static_cast<long long>(seconds);
|
||||
long long fraction = pynum::py_round((seconds - static_cast<double>(whole)) * 100000.0);
|
||||
if (fraction >= 100000) {
|
||||
whole += 1;
|
||||
fraction = 0;
|
||||
}
|
||||
char buffer[32];
|
||||
std::snprintf(buffer, sizeof(buffer), "%02lld:%02lld:%02lld.%05lld", whole / 3600,
|
||||
(whole % 3600) / 60, whole % 60, fraction);
|
||||
return std::string(buffer);
|
||||
}
|
||||
|
||||
bool binaural_mode_from_name(const char* name, BinauralMode* out) {
|
||||
if (name == nullptr || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (std::strcmp(name, "off") == 0) { *out = BinauralMode::Off; return true; }
|
||||
if (std::strcmp(name, "near") == 0) { *out = BinauralMode::Near; return true; }
|
||||
if (std::strcmp(name, "far") == 0) { *out = BinauralMode::Far; return true; }
|
||||
if (std::strcmp(name, "mid") == 0) { *out = BinauralMode::Mid; return true; }
|
||||
if (std::strcmp(name, "unspecified") == 0) { *out = BinauralMode::Unspecified; return true; }
|
||||
return false;
|
||||
}
|
||||
|
||||
std::string build_chna() {
|
||||
std::string out;
|
||||
put_u16(&out, static_cast<std::uint16_t>(kTrackCount));
|
||||
put_u16(&out, static_cast<std::uint16_t>(kTrackCount));
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
put_u16(&out, static_cast<std::uint16_t>(i + 1));
|
||||
out += "ATU_" + hex8(i + 1);
|
||||
out += "AT_0001" + hex4(0x1001 + i) + "_01";
|
||||
out += "AP_00011001";
|
||||
out.push_back('\0');
|
||||
}
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
put_u16(&out, static_cast<std::uint16_t>(i + 11));
|
||||
out += "ATU_" + hex8(i + 11);
|
||||
out += "AT_0003" + hex4(0x1001 + i) + "_01";
|
||||
out += "AP_0003" + hex4(0x1001 + i);
|
||||
out.push_back('\0');
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
Status build_dbmd(std::uint32_t object_count, BinauralMode mode, std::string* out) {
|
||||
if (out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput, "null output");
|
||||
}
|
||||
const std::uint32_t mode_value = static_cast<std::uint32_t>(mode);
|
||||
if (mode_value > 4u) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput,
|
||||
"invalid JOC binaural render mode");
|
||||
}
|
||||
out->clear();
|
||||
put_u32(out, 0x01000006u);
|
||||
|
||||
std::string segment7(96, '\0');
|
||||
segment7[1] = static_cast<char>(0x47);
|
||||
segment7[5] = static_cast<char>(0x60);
|
||||
segment7[8] = static_cast<char>(0x24);
|
||||
segment7[9] = static_cast<char>(0x24);
|
||||
out->push_back(7);
|
||||
put_u16(out, 96);
|
||||
out->append(segment7);
|
||||
out->push_back(static_cast<char>(checksum(segment7)));
|
||||
|
||||
std::string segment9(248, '\0');
|
||||
const std::string creator = "Created with EAC3JOC";
|
||||
const std::string renderer = "EAC3JOC Python Renderer";
|
||||
std::memcpy(&segment9[0], creator.data(), creator.size());
|
||||
std::memcpy(&segment9[32], renderer.data(), renderer.size());
|
||||
segment9[96] = 2;
|
||||
segment9[97] = 1;
|
||||
segment9[98] = 0;
|
||||
segment9[103] = 0x03;
|
||||
segment9[106] = 0x01;
|
||||
segment9[111] = 0x22;
|
||||
segment9[112] = static_cast<char>(0xFF);
|
||||
out->push_back(9);
|
||||
put_u16(out, 248);
|
||||
out->append(segment9);
|
||||
out->push_back(static_cast<char>(checksum(segment9)));
|
||||
|
||||
// The reference allocates the body zeroed and then fills only the trailing
|
||||
// `object_count` bytes with 0x84, so the template region stays zero.
|
||||
const std::size_t object_body = 5u + 262u + object_count;
|
||||
std::string segment10(object_body, '\0');
|
||||
const std::uint32_t sync = 0xF8726FBDu;
|
||||
std::memcpy(&segment10[0], &sync, 4);
|
||||
segment10[4] = static_cast<char>(object_count);
|
||||
for (std::size_t i = 5u + 262u; i < segment10.size(); ++i) {
|
||||
segment10[i] = static_cast<char>(0x84);
|
||||
}
|
||||
const std::size_t object_modes = 4u + 2u + 1u + 9u * 15u + object_count;
|
||||
for (std::uint32_t i = 10; i < std::min<std::uint32_t>(object_count, 10u + kObjectCount); ++i) {
|
||||
const std::size_t index = object_modes + i;
|
||||
if (index >= segment10.size()) {
|
||||
return Status::fail(JOC_ERR_INTERNAL, stage::kOutput, "dbmd object slot out of range");
|
||||
}
|
||||
segment10[index] = static_cast<char>((static_cast<unsigned char>(segment10[index]) & 0xF8u) |
|
||||
mode_value);
|
||||
}
|
||||
out->push_back(10);
|
||||
put_u16(out, static_cast<std::uint16_t>(segment10.size()));
|
||||
out->append(segment10);
|
||||
out->push_back(static_cast<char>(checksum(segment10)));
|
||||
out->append("\0\0", 2);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
std::string build_axml(const std::vector<Track>& tracks, double duration_sec, std::uint32_t rate) {
|
||||
const double scale = static_cast<double>(rate);
|
||||
std::string out;
|
||||
out.reserve(64u * 1024u);
|
||||
const std::string duration_ts = ts(duration_sec);
|
||||
|
||||
out += "<?xml version=\"1.0\" encoding=\"utf-8\"?>";
|
||||
out += "<ebuCoreMain xsi:schemaLocation=\"urn:ebu:metadata-schema:ebuCore_2016 ebucore.xsd\" "
|
||||
"lang=\"en\" xmlns:xsi=\"http://www.w3.org/2001/XMLSchema-instance\" "
|
||||
"xmlns=\"urn:ebu:metadata-schema:ebuCore_2016\">";
|
||||
out += "<coreMetadata><format><audioFormatExtended>";
|
||||
out += "<audioProgramme audioProgrammeID=\"APR_1001\" audioProgrammeName=\"EAC3JOC_Export\" "
|
||||
"start=\"" +
|
||||
ts(0.0) + "\" end=\"" + duration_ts + "\">";
|
||||
out += "<audioContentIDRef>ACO_1001</audioContentIDRef>";
|
||||
out += "<audioContentIDRef>ACO_1002</audioContentIDRef>";
|
||||
out += "</audioProgramme>";
|
||||
out += "<audioContent audioContentID=\"ACO_1001\" "
|
||||
"audioContentName=\"EAC3JOC_Master_Content\">";
|
||||
out += "<audioObjectIDRef>AO_1001</audioObjectIDRef>";
|
||||
out += "<dialogue mixedContentKind=\"0\">2</dialogue>";
|
||||
out += "</audioContent>";
|
||||
out += "<audioContent audioContentID=\"ACO_1002\" audioContentName=\"Objects\">";
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioObjectIDRef>AO_" + hex4(0x100b + i) + "</audioObjectIDRef>";
|
||||
}
|
||||
out += "<dialogue mixedContentKind=\"0\">2</dialogue>";
|
||||
out += "</audioContent>";
|
||||
out += "<audioObject audioObjectID=\"AO_1001\" audioObjectName=\"Bed\" start=\"" + ts(0.0) +
|
||||
"\" duration=\"" + duration_ts + "\">";
|
||||
out += "<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>";
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioTrackUIDRef>ATU_" + hex8(i + 1) + "</audioTrackUIDRef>";
|
||||
}
|
||||
out += "</audioObject>";
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioObject audioObjectID=\"AO_" + hex4(0x100b + i) +
|
||||
"\" audioObjectName=\"Audio Object " + std::to_string(i + 1) + "\" start=\"" +
|
||||
ts(0.0) + "\" duration=\"" + duration_ts + "\">";
|
||||
out += "<audioPackFormatIDRef>AP_0003" + hex4(0x1001 + i) + "</audioPackFormatIDRef>";
|
||||
out += "<audioTrackUIDRef>ATU_" + hex8(11 + i) + "</audioTrackUIDRef>";
|
||||
out += "</audioObject>";
|
||||
}
|
||||
out += "<audioPackFormat audioPackFormatID=\"AP_00011001\" "
|
||||
"audioPackFormatName=\"EAC3JOCBedPack\" typeDefinition=\"DirectSpeakers\" "
|
||||
"typeLabel=\"0001\">";
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioChannelFormatIDRef>AC_0001" + hex4(0x1001 + i) +
|
||||
"</audioChannelFormatIDRef>";
|
||||
}
|
||||
out += "</audioPackFormat>";
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioPackFormat audioPackFormatID=\"AP_0003" + hex4(0x1001 + i) +
|
||||
"\" audioPackFormatName=\"JOC_Object_" + std::to_string(i + 1) +
|
||||
"\" typeDefinition=\"Objects\" typeLabel=\"0003\">";
|
||||
out += "<audioChannelFormatIDRef>AC_0003" + hex4(0x1001 + i) +
|
||||
"</audioChannelFormatIDRef>";
|
||||
out += "</audioPackFormat>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioChannelFormat audioChannelFormatID=\"AC_0001" + hex4(0x1001 + i) +
|
||||
"\" audioChannelFormatName=\"" + kBedNames[i] +
|
||||
"\" typeDefinition=\"DirectSpeakers\" typeLabel=\"0001\">";
|
||||
out += "<audioBlockFormat audioBlockFormatID=\"AB_0001" + hex4(0x1001 + i) +
|
||||
"_00000001\">";
|
||||
out += "<cartesian>1</cartesian>";
|
||||
out += "<position coordinate=\"X\">" + pynum::format_fixed(kBedPos[i][0], 10) +
|
||||
"</position>";
|
||||
out += "<position coordinate=\"Y\">" + pynum::format_fixed(kBedPos[i][1], 10) +
|
||||
"</position>";
|
||||
if (kBedPos[i][2] != 0.0) {
|
||||
out += "<position coordinate=\"Z\">" + pynum::format_fixed(kBedPos[i][2], 10) +
|
||||
"</position>";
|
||||
}
|
||||
out += std::string("<speakerLabel>") + kBedLabels[i] + "</speakerLabel>";
|
||||
out += "</audioBlockFormat>";
|
||||
out += "</audioChannelFormat>";
|
||||
}
|
||||
for (std::size_t i = 0; i < tracks.size(); ++i) {
|
||||
const Track& track = tracks[i];
|
||||
out += "<audioChannelFormat audioChannelFormatID=\"AC_0003" +
|
||||
hex4(0x1001 + static_cast<std::uint32_t>(i)) + "\" audioChannelFormatName=\"" +
|
||||
track.name + "\" typeDefinition=\"Objects\" typeLabel=\"0003\">";
|
||||
for (std::size_t k = 0; k < track.blocks.size(); ++k) {
|
||||
const Keyframe& block = track.blocks[k];
|
||||
out += "<audioBlockFormat audioBlockFormatID=\"AB_0003" +
|
||||
hex4(0x1001 + static_cast<std::uint32_t>(i)) + "_" +
|
||||
hex8(static_cast<std::uint32_t>(k + 1)) + "\" rtime=\"" +
|
||||
ts(static_cast<double>(block.rtime_samples) / scale) + "\" duration=\"" +
|
||||
ts(static_cast<double>(block.duration_samples) / scale) + "\">";
|
||||
out += "<cartesian>1</cartesian>";
|
||||
out += "<position coordinate=\"X\">" + pynum::format_fixed(block.x, 10) + "</position>";
|
||||
out += "<position coordinate=\"Y\">" + pynum::format_fixed(block.y, 10) + "</position>";
|
||||
if (block.z != 0.0) {
|
||||
out += "<position coordinate=\"Z\">" + pynum::format_fixed(block.z, 10) +
|
||||
"</position>";
|
||||
}
|
||||
out += "<jumpPosition interpolationLength=\"" +
|
||||
pynum::format_fixed(static_cast<double>(block.interpolation_samples) / scale,
|
||||
5) +
|
||||
"\">1</jumpPosition>";
|
||||
out += "</audioBlockFormat>";
|
||||
}
|
||||
out += "</audioChannelFormat>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioTrackUID UID=\"ATU_" + hex8(i + 1) +
|
||||
"\" bitDepth=\"24\" sampleRate=\"48000\">";
|
||||
out += "<audioTrackFormatIDRef>AT_0001" + hex4(0x1001 + i) + "_01</audioTrackFormatIDRef>";
|
||||
out += "<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>";
|
||||
out += "</audioTrackUID>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioTrackUID UID=\"ATU_" + hex8(11 + i) +
|
||||
"\" bitDepth=\"24\" sampleRate=\"48000\">";
|
||||
out += "<audioTrackFormatIDRef>AT_0003" + hex4(0x1001 + i) + "_01</audioTrackFormatIDRef>";
|
||||
out += "<audioPackFormatIDRef>AP_0003" + hex4(0x1001 + i) + "</audioPackFormatIDRef>";
|
||||
out += "</audioTrackUID>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioTrackFormat audioTrackFormatID=\"AT_0001" + hex4(0x1001 + i) +
|
||||
"_01\" audioTrackFormatName=\"PCM_" + kBedNames[i] +
|
||||
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
|
||||
out += "<audioStreamFormatIDRef>AS_0001" + hex4(0x1001 + i) +
|
||||
"</audioStreamFormatIDRef>";
|
||||
out += "</audioTrackFormat>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioTrackFormat audioTrackFormatID=\"AT_0003" + hex4(0x1001 + i) +
|
||||
"_01\" audioTrackFormatName=\"PCM_JOC_Object_" + std::to_string(i + 1) +
|
||||
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
|
||||
out += "<audioStreamFormatIDRef>AS_0003" + hex4(0x1001 + i) +
|
||||
"</audioStreamFormatIDRef>";
|
||||
out += "</audioTrackFormat>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioStreamFormat audioStreamFormatID=\"AS_0001" + hex4(0x1001 + i) +
|
||||
"\" audioStreamFormatName=\"PCM_" + kBedNames[i] +
|
||||
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
|
||||
out += "<audioChannelFormatIDRef>AC_0001" + hex4(0x1001 + i) +
|
||||
"</audioChannelFormatIDRef>";
|
||||
out += "<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>";
|
||||
out += "<audioTrackFormatIDRef>AT_0001" + hex4(0x1001 + i) +
|
||||
"_01</audioTrackFormatIDRef>";
|
||||
out += "</audioStreamFormat>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioStreamFormat audioStreamFormatID=\"AS_0003" + hex4(0x1001 + i) +
|
||||
"\" audioStreamFormatName=\"PCM_JOC_Object_" + std::to_string(i + 1) +
|
||||
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
|
||||
out += "<audioChannelFormatIDRef>AC_0003" + hex4(0x1001 + i) +
|
||||
"</audioChannelFormatIDRef>";
|
||||
out += "<audioPackFormatIDRef>AP_0003" + hex4(0x1001 + i) + "</audioPackFormatIDRef>";
|
||||
out += "<audioTrackFormatIDRef>AT_0003" + hex4(0x1001 + i) +
|
||||
"_01</audioTrackFormatIDRef>";
|
||||
out += "</audioStreamFormat>";
|
||||
}
|
||||
out += "</audioFormatExtended></format></coreMetadata>";
|
||||
out += "</ebuCoreMain>";
|
||||
return out;
|
||||
}
|
||||
|
||||
} // namespace joc::adm
|
||||
@@ -1,43 +0,0 @@
|
||||
// Port of src/adm_atmos.py.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::adm {
|
||||
|
||||
inline constexpr int kObjectCount = 15;
|
||||
inline constexpr std::uint32_t kTrackCount = 25;
|
||||
|
||||
struct Keyframe {
|
||||
std::int64_t rtime_samples = 0;
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
std::int64_t duration_samples = 0;
|
||||
std::int64_t interpolation_samples = 0;
|
||||
};
|
||||
|
||||
struct Track {
|
||||
std::string name;
|
||||
std::vector<Keyframe> blocks;
|
||||
};
|
||||
|
||||
// HH:MM:SS.fffff with the reference's truncation + round-half-even carry.
|
||||
std::string ts(double seconds);
|
||||
|
||||
enum class BinauralMode : std::uint32_t { Off = 0, Near = 1, Far = 2, Mid = 3, Unspecified = 4 };
|
||||
|
||||
bool binaural_mode_from_name(const char* name, BinauralMode* out);
|
||||
|
||||
std::string build_axml(const std::vector<Track>& tracks, double duration_sec, std::uint32_t rate);
|
||||
|
||||
std::string build_chna();
|
||||
|
||||
Status build_dbmd(std::uint32_t object_count, BinauralMode mode, std::string* out);
|
||||
|
||||
} // namespace joc::adm
|
||||
@@ -1,281 +0,0 @@
|
||||
#include "adm/adm_tracks.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
|
||||
#include "foundation/geometry.h"
|
||||
|
||||
namespace joc::adm {
|
||||
|
||||
namespace {
|
||||
|
||||
struct Point {
|
||||
std::int64_t sample = 0;
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
std::int64_t interpolation_samples = 0;
|
||||
};
|
||||
|
||||
// otherwise (the reference drops silently).
|
||||
void append_point(std::vector<Point>* points, std::int64_t sample, double x, double y, double z,
|
||||
std::int64_t interpolation_samples) {
|
||||
if (!points->empty() && sample == points->back().sample) {
|
||||
points->back() = Point{sample, x, y, z, interpolation_samples};
|
||||
} else if (points->empty() || sample > points->back().sample) {
|
||||
points->push_back(Point{sample, x, y, z, interpolation_samples});
|
||||
}
|
||||
}
|
||||
|
||||
void lerp(double ax, double ay, double az, double bx, double by, double bz, double amount,
|
||||
double* x, double* y, double* z) {
|
||||
*x = ax + (bx - ax) * amount;
|
||||
*y = ay + (by - ay) * amount;
|
||||
*z = az + (bz - az) * amount;
|
||||
}
|
||||
|
||||
void points_to_blocks(const std::vector<Point>& points, std::int64_t total_samples,
|
||||
std::vector<Keyframe>* out) {
|
||||
for (std::size_t index = 0; index < points.size(); ++index) {
|
||||
const Point& point = points[index];
|
||||
const std::int64_t end =
|
||||
(index + 1 < points.size()) ? points[index + 1].sample : total_samples;
|
||||
const std::int64_t duration = std::max<std::int64_t>(0, end - point.sample);
|
||||
if (duration == 0) {
|
||||
continue;
|
||||
}
|
||||
Keyframe keyframe;
|
||||
keyframe.rtime_samples = point.sample;
|
||||
keyframe.x = point.x;
|
||||
keyframe.y = point.y;
|
||||
keyframe.z = point.z;
|
||||
keyframe.duration_samples = duration;
|
||||
keyframe.interpolation_samples = std::min(point.interpolation_samples, duration);
|
||||
out->push_back(keyframe);
|
||||
}
|
||||
}
|
||||
|
||||
Status non_monotonic(const char* name, const char* message, int object_index, std::int64_t sample,
|
||||
std::int64_t previous_sample) {
|
||||
return Status::fail(JOC_ERR_OAMD_UNSUPPORTED_VARIANT, stage::kOamd,
|
||||
std::string(name) + ": " + message + " (object " +
|
||||
std::to_string(object_index) + ", sample " + std::to_string(sample) +
|
||||
", previous " + std::to_string(previous_sample) + ")");
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Status expand_compact(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
|
||||
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
|
||||
int object_index, std::vector<Keyframe>* out) {
|
||||
out->clear();
|
||||
const double scale = static_cast<double>(rate);
|
||||
if (events.empty()) {
|
||||
Keyframe keyframe;
|
||||
keyframe.rtime_samples = 0;
|
||||
keyframe.duration_samples = total_samples;
|
||||
keyframe.interpolation_samples = 0;
|
||||
keyframe.x = 0.0;
|
||||
keyframe.y = 0.0;
|
||||
keyframe.z = 0.0;
|
||||
out->push_back(keyframe);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
std::vector<Point> points;
|
||||
double current[3] = {events[0].x, events[0].y, events[0].z};
|
||||
append_point(&points, 0, current[0], current[1], current[2], 0);
|
||||
|
||||
for (std::size_t index = 1; index < events.size(); ++index) {
|
||||
const OamdEvent& event = events[index];
|
||||
const std::int64_t event_start = event.sample + object_delay_samples;
|
||||
if (event_start >= total_samples) {
|
||||
break;
|
||||
}
|
||||
const std::int64_t effective_ramp =
|
||||
std::max<std::int64_t>(0, event.ramp_samples - update_quantum_samples);
|
||||
const std::int64_t block_start =
|
||||
event_start + (effective_ramp != 0 ? update_quantum_samples : 0);
|
||||
if (block_start >= total_samples) {
|
||||
break;
|
||||
}
|
||||
const std::int64_t ramp_end = block_start + effective_ramp;
|
||||
|
||||
if (block_start < points.back().sample) {
|
||||
return non_monotonic("non_monotonic_compact_position_updates",
|
||||
"compact object position update moved backwards", object_index,
|
||||
block_start, points.back().sample);
|
||||
}
|
||||
if (index + 1 < events.size()) {
|
||||
const std::int64_t next_event_start = events[index + 1].sample + object_delay_samples;
|
||||
const std::int64_t next_effective =
|
||||
std::max<std::int64_t>(0, events[index + 1].ramp_samples - update_quantum_samples);
|
||||
const std::int64_t next_block_start =
|
||||
next_event_start + (next_effective != 0 ? update_quantum_samples : 0);
|
||||
if (next_block_start < ramp_end) {
|
||||
return non_monotonic("overlapping_compact_position_ramps",
|
||||
"a new position update arrived before the previous compact "
|
||||
"ramp finished",
|
||||
object_index, block_start, ramp_end);
|
||||
}
|
||||
}
|
||||
|
||||
double target[3] = {event.x, event.y, event.z};
|
||||
std::int64_t interpolation = effective_ramp;
|
||||
const std::int64_t available = total_samples - block_start;
|
||||
if (effective_ramp > available) {
|
||||
const double amount =
|
||||
static_cast<double>(available) / static_cast<double>(effective_ramp);
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
lerp(current[0], current[1], current[2], event.x, event.y, event.z, amount, &x, &y, &z);
|
||||
target[0] = x;
|
||||
target[1] = y;
|
||||
target[2] = z;
|
||||
interpolation = available;
|
||||
}
|
||||
append_point(&points, block_start, target[0], target[1], target[2], interpolation);
|
||||
current[0] = event.x;
|
||||
current[1] = event.y;
|
||||
current[2] = event.z;
|
||||
}
|
||||
|
||||
points_to_blocks(points, total_samples, out);
|
||||
(void)scale;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status expand_dense64(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
|
||||
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
|
||||
int object_index, std::vector<Keyframe>* out) {
|
||||
out->clear();
|
||||
(void)rate;
|
||||
if (events.empty()) {
|
||||
Keyframe keyframe;
|
||||
keyframe.duration_samples = total_samples;
|
||||
out->push_back(keyframe);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
std::vector<Point> points;
|
||||
double current[3] = {events[0].x, events[0].y, events[0].z};
|
||||
append_point(&points, 0, current[0], current[1], current[2], 0);
|
||||
|
||||
for (std::size_t index = 1; index < events.size(); ++index) {
|
||||
const OamdEvent& event = events[index];
|
||||
const std::int64_t start = event.sample + object_delay_samples;
|
||||
if (start >= total_samples) {
|
||||
break;
|
||||
}
|
||||
if (start < points.back().sample) {
|
||||
return non_monotonic("non_monotonic_position_updates",
|
||||
"object position update moved backwards", object_index, start,
|
||||
points.back().sample);
|
||||
}
|
||||
if (start > points.back().sample) {
|
||||
append_point(&points, start, current[0], current[1], current[2], 0);
|
||||
}
|
||||
const std::int64_t effective_ramp =
|
||||
std::max<std::int64_t>(0, event.ramp_samples - update_quantum_samples);
|
||||
if (effective_ramp == 0) {
|
||||
append_point(&points, start, event.x, event.y, event.z, 0);
|
||||
current[0] = event.x;
|
||||
current[1] = event.y;
|
||||
current[2] = event.z;
|
||||
continue;
|
||||
}
|
||||
const std::int64_t steps =
|
||||
(effective_ramp + update_quantum_samples - 1) / update_quantum_samples;
|
||||
const std::int64_t end = start + steps * update_quantum_samples;
|
||||
if (index + 1 < events.size()) {
|
||||
const std::int64_t next_start = events[index + 1].sample + object_delay_samples;
|
||||
if (next_start < end) {
|
||||
return non_monotonic("overlapping_position_ramps",
|
||||
"a new position update arrived before the previous ramp "
|
||||
"finished",
|
||||
object_index, start, end);
|
||||
}
|
||||
}
|
||||
std::int64_t future = effective_ramp;
|
||||
std::int64_t elapsed = 0;
|
||||
double position[3] = {current[0], current[1], current[2]};
|
||||
while (future > 0) {
|
||||
const double amount =
|
||||
std::min(static_cast<double>(update_quantum_samples) / static_cast<double>(future),
|
||||
1.0);
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
lerp(position[0], position[1], position[2], event.x, event.y, event.z, amount, &x, &y,
|
||||
&z);
|
||||
position[0] = x;
|
||||
position[1] = y;
|
||||
position[2] = z;
|
||||
elapsed += update_quantum_samples;
|
||||
const std::int64_t sample = start + elapsed;
|
||||
if (sample >= total_samples) {
|
||||
break;
|
||||
}
|
||||
append_point(&points, sample, position[0], position[1], position[2],
|
||||
update_quantum_samples);
|
||||
future -= update_quantum_samples;
|
||||
}
|
||||
current[0] = event.x;
|
||||
current[1] = event.y;
|
||||
current[2] = event.z;
|
||||
}
|
||||
|
||||
points_to_blocks(points, total_samples, out);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
void TrajectoryBuilder::submit_frame(std::int64_t frame_index, const oamd::OamdUpdate* update,
|
||||
std::int64_t outer_offset) {
|
||||
std::int64_t event_sample = frame_index * 1536;
|
||||
std::int64_t ramp_samples = 0;
|
||||
if (update != nullptr) {
|
||||
state_.apply(*update);
|
||||
event_sample += outer_offset + static_cast<std::int64_t>(update->block_offset_samples);
|
||||
ramp_samples = static_cast<std::int64_t>(update->ramp_duration_samples);
|
||||
}
|
||||
for (int object = 1; object <= kObjectCount; ++object) {
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
geometry::q_to_adm_xyz(state_.q(object, 0), state_.q(object, 1), state_.q(object, 2), &x,
|
||||
&y, &z);
|
||||
const int slot = object - 1;
|
||||
if (!has_previous_[slot] || previous_[slot][0] != x || previous_[slot][1] != y ||
|
||||
previous_[slot][2] != z) {
|
||||
events_[slot].push_back(OamdEvent{event_sample, x, y, z, ramp_samples});
|
||||
previous_[slot][0] = x;
|
||||
previous_[slot][1] = y;
|
||||
previous_[slot][2] = z;
|
||||
has_previous_[slot] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Status TrajectoryBuilder::build(std::int64_t total_samples, TrajectoryMode mode,
|
||||
std::vector<Track>* out) const {
|
||||
out->clear();
|
||||
out->reserve(kObjectCount);
|
||||
for (int object = 1; object <= kObjectCount; ++object) {
|
||||
Track track;
|
||||
track.name = "JOC_Object_" + std::to_string(object);
|
||||
const Status status =
|
||||
(mode == TrajectoryMode::Compact)
|
||||
? expand_compact(events_[object - 1], total_samples, rate_, quantum_,
|
||||
object_delay_, object, &track.blocks)
|
||||
: expand_dense64(events_[object - 1], total_samples, rate_, quantum_,
|
||||
object_delay_, object, &track.blocks);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
out->push_back(std::move(track));
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::adm
|
||||
@@ -1,61 +0,0 @@
|
||||
// Port of src/oamd_tracks.py.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "adm/adm_metadata.h"
|
||||
#include "foundation/status.h"
|
||||
#include "oamd/oamd_parser.h"
|
||||
|
||||
namespace joc::adm {
|
||||
|
||||
struct OamdEvent {
|
||||
std::int64_t sample = 0;
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
std::int64_t ramp_samples = 0;
|
||||
};
|
||||
|
||||
enum class TrajectoryMode { Compact, Dense64 };
|
||||
|
||||
// Feeds the same per-frame OAMD state machine the reference's build_adm_tracks
|
||||
// runs, and records one event per object whenever its coordinates change.
|
||||
class TrajectoryBuilder {
|
||||
public:
|
||||
TrajectoryBuilder(std::uint32_t rate = 48000, std::int64_t update_quantum_samples = 64,
|
||||
std::int64_t object_delay_samples = 1473)
|
||||
: rate_(rate),
|
||||
quantum_(update_quantum_samples),
|
||||
object_delay_(object_delay_samples) {}
|
||||
|
||||
void submit_frame(std::int64_t frame_index, const oamd::OamdUpdate* update, std::int64_t outer_offset);
|
||||
|
||||
Status build(std::int64_t total_samples, TrajectoryMode mode, std::vector<Track>* out) const;
|
||||
|
||||
const std::vector<OamdEvent>& events(int object_index) const { return events_[object_index]; }
|
||||
std::uint32_t rate() const { return rate_; }
|
||||
std::int64_t object_delay_samples() const { return object_delay_; }
|
||||
|
||||
private:
|
||||
std::uint32_t rate_;
|
||||
std::int64_t quantum_;
|
||||
std::int64_t object_delay_;
|
||||
oamd::OamdState state_;
|
||||
std::vector<OamdEvent> events_[kObjectCount];
|
||||
bool has_previous_[kObjectCount] = {};
|
||||
double previous_[kObjectCount][3] = {};
|
||||
};
|
||||
|
||||
Status expand_compact(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
|
||||
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
|
||||
int object_index, std::vector<Keyframe>* out);
|
||||
|
||||
Status expand_dense64(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
|
||||
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
|
||||
int object_index, std::vector<Keyframe>* out);
|
||||
|
||||
} // namespace joc::adm
|
||||
@@ -0,0 +1,145 @@
|
||||
"""把 LFE、15 路对象 PCM 和对象轨迹组装为 ADM BWF。
|
||||
|
||||
固定输出契约:
|
||||
EAC3JOC 重放输出 = 16ch(ch0 = LFE + ch1-15 = 15 对象);
|
||||
最终 ADM BWF = 7.1.2 bed(L R C Ls Rs Lb Rb + LFE + Ltf Rtf = 10ch)
|
||||
—— 除 LFE 外全部静音;
|
||||
15 对象 = ch1-15 直接填充对象轨;轨迹 = OAMD(q1/q2/q3 → xyz)。
|
||||
"""
|
||||
import os
|
||||
|
||||
import numpy as np
|
||||
|
||||
import adm_atmos
|
||||
|
||||
|
||||
def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None,
|
||||
duration_sec=None, rate=48000, joc_binaural_mode=4):
|
||||
"""16ch f32 交织 raw → 25ch ADM BWF(空 7.1.2 bed + LFE + 15 对象)。
|
||||
|
||||
raw16: (n, 16) 交织(ch0 = LFE,ch1-15 = 对象)。
|
||||
scale: 1.0 = 默认 0 dB,不附加输出缩放。该参数与 joc_clipgain 无关;
|
||||
主命令行已在渲染阶段应用用户增益,因此这里传 1.0。
|
||||
kf_tracks: 可选轨迹关键帧(OAMD 输出,格式 [(obj_id, [(t, x, y, z), ...]), ...]);
|
||||
缺省 = 静止参考位置(adm_atmos 默认)。
|
||||
"""
|
||||
raw = np.memmap(raw16_path, dtype=np.float32, mode="r")
|
||||
n = len(raw) // 16
|
||||
raw = raw[:n * 16].reshape(-1, 16)
|
||||
if duration_sec is None:
|
||||
duration_sec = n / rate
|
||||
# 惰性视图:adm_atmos 按块读取,避免全片 25ch 在内存中展开。
|
||||
class BedView:
|
||||
shape = (n, 10)
|
||||
|
||||
def __getitem__(self, key):
|
||||
src = np.asarray(raw[key], dtype=np.float32)
|
||||
one = src.ndim == 1
|
||||
if one:
|
||||
src = src[None, :]
|
||||
out = np.zeros((len(src), 10), dtype=np.float32)
|
||||
out[:, 3] = np.multiply(src[:, 0], np.float32(scale), dtype=np.float32)
|
||||
return out[0] if one else out
|
||||
|
||||
class ObjView:
|
||||
shape = (n, 16)
|
||||
|
||||
def __getitem__(self, key):
|
||||
return np.multiply(np.asarray(raw[key], dtype=np.float32),
|
||||
np.float32(scale), dtype=np.float32)
|
||||
if kf_tracks is None:
|
||||
kf_tracks = []
|
||||
for oi in range(15):
|
||||
# 静止参考位置(q1=q2=q3=0 → 原点;实际坐标按 OAMD 输出填入)
|
||||
kf_tracks.append(("JOC_Object_%d" % (oi + 1),
|
||||
[(0.0, 0.0, 0.0, 0.0, max(duration_sec, 1e-6))]))
|
||||
adm_atmos.build_master(out_path, BedView(), ObjView(), kf_tracks,
|
||||
duration_sec, rate=rate,
|
||||
joc_binaural_mode=joc_binaural_mode)
|
||||
# 及时释放 Windows 文件句柄,允许 TemporaryDirectory 删除中间 raw。
|
||||
raw._mmap.close()
|
||||
return out_path
|
||||
|
||||
|
||||
class StreamingMaster:
|
||||
"""Incrementally write renderer frames into the final 25-channel ADM BWF.
|
||||
|
||||
This removes the default 16-channel float32 intermediate file. The mapping
|
||||
remains identical to :func:`assemble_from_raw`: bed channel 3 receives LFE,
|
||||
bed channels 0..2/4..9 are silent, and output objects 1..15 map to ADM
|
||||
channels 10..24.
|
||||
"""
|
||||
|
||||
def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072,
|
||||
joc_binaural_mode=4):
|
||||
if block_samples < 1536:
|
||||
raise ValueError("block_samples must be at least one E-AC-3 frame")
|
||||
self.out_path = os.fspath(out_path)
|
||||
self.duration_sec = float(duration_sec)
|
||||
self.rate = int(rate)
|
||||
self.joc_binaural_mode = joc_binaural_mode
|
||||
self._sink = adm_atmos.Sink25(self.out_path, 25, self.rate)
|
||||
self._buffer = np.empty((int(block_samples), 25), dtype=np.float32)
|
||||
self._used = 0
|
||||
self._finalized = False
|
||||
|
||||
def _flush(self):
|
||||
if self._used:
|
||||
self._sink.write_block(self._buffer[:self._used])
|
||||
self._used = 0
|
||||
|
||||
def write_frame(self, pcm16):
|
||||
pcm = np.asarray(pcm16, dtype=np.float32)
|
||||
if pcm.shape != (16, 1536):
|
||||
raise ValueError(f"renderer frame must be (16,1536), got {pcm.shape}")
|
||||
source = 0
|
||||
while source < 1536:
|
||||
available = len(self._buffer) - self._used
|
||||
count = min(available, 1536 - source)
|
||||
target = self._buffer[self._used:self._used + count]
|
||||
target.fill(0.0)
|
||||
target[:, 3] = pcm[0, source:source + count]
|
||||
target[:, 10:25] = pcm[1:16, source:source + count].T
|
||||
self._used += count
|
||||
source += count
|
||||
if self._used == len(self._buffer):
|
||||
self._flush()
|
||||
|
||||
def finalize(self, kf_tracks):
|
||||
if self._finalized:
|
||||
raise RuntimeError("StreamingMaster already finalized")
|
||||
self._flush()
|
||||
try:
|
||||
from . import adm_serializer
|
||||
except ImportError:
|
||||
import adm_serializer
|
||||
axml = adm_serializer.build_axml(kf_tracks, self.duration_sec)
|
||||
chna = adm_atmos.build_chna()
|
||||
dbmd = adm_atmos.build_dbmd(
|
||||
25, joc_binaural_mode=self.joc_binaural_mode)
|
||||
trajectory_blocks = sum(len(track[1]) for track in kf_tracks)
|
||||
self.metadata_info = {
|
||||
"axml_bytes": len(axml),
|
||||
"trajectory_blocks": trajectory_blocks,
|
||||
"chna_bytes": len(chna),
|
||||
"dbmd_bytes": len(dbmd),
|
||||
}
|
||||
self._sink.finalize(axml, chna, dbmd)
|
||||
self._finalized = True
|
||||
print(f"master25 -> {self.out_path} ({self.duration_sec:.2f}s, 25ch, "
|
||||
f"axml={len(axml)}B, chna={len(chna)}B, dbmd={len(dbmd)}B)")
|
||||
return self.out_path
|
||||
|
||||
def abort(self):
|
||||
if self._finalized:
|
||||
return
|
||||
sink = getattr(self, "_sink", None)
|
||||
fp = getattr(sink, "fp", None)
|
||||
if fp is not None and not fp.closed:
|
||||
fp.close()
|
||||
|
||||
def __del__(self):
|
||||
try:
|
||||
self.abort()
|
||||
except Exception:
|
||||
pass
|
||||
@@ -0,0 +1,300 @@
|
||||
"""生成 25 声道 RF64 ADM BWF 及其 axml、chna、dbmd 元数据。
|
||||
|
||||
输出由 10 声道 7.1.2 bed 和 15 路对象组成;RF64 尺寸字段在写入完成后回填。
|
||||
"""
|
||||
import operator
|
||||
import struct
|
||||
import numpy as np
|
||||
import xml.etree.ElementTree as ET
|
||||
|
||||
NS = "urn:ebu:metadata-schema:ebuCore_2016"
|
||||
XSI = "http://www.w3.org/2001/XMLSchema-instance"
|
||||
|
||||
BED_NAMES = ["RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter",
|
||||
"RoomCentricLFE", "RoomCentricLeftSideSurround",
|
||||
"RoomCentricRightSideSurround", "RoomCentricLeftRearSurround",
|
||||
"RoomCentricRightRearSurround", "RoomCentricLeftTopSurround",
|
||||
"RoomCentricRightTopSurround"]
|
||||
BED_LABELS = ["RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss", "RC_Rss",
|
||||
"RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"]
|
||||
BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0),
|
||||
(-1.0, 1.0, -1.0), (-1.0, 0.0, 0.0), (1.0, 0.0, 0.0),
|
||||
(-1.0, -1.0, 0.0), (1.0, -1.0, 0.0), (-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
|
||||
|
||||
N_OBJ = 15
|
||||
JOC_BINAURAL_MODES = {
|
||||
"off": 0,
|
||||
"near": 1,
|
||||
"far": 2,
|
||||
"mid": 3,
|
||||
"unspecified": 4,
|
||||
}
|
||||
JOC_BINAURAL_MODE_DEFAULT = "unspecified"
|
||||
|
||||
def q_to_adm_xyz(q1, q2, q3):
|
||||
posX = min(1.0, round(q1 * 62 / 32767.0) / 62.0)
|
||||
posY = min(1.0, round(q2 * 62 / 32767.0) / 62.0)
|
||||
posZ = round(q3 * 15 / 32767.0) / 15.0
|
||||
posZ = max(-1.0, min(1.0, posZ))
|
||||
return posX * 2 - 1, 1 - posY * 2, posZ
|
||||
|
||||
def ts(seconds):
|
||||
s = int(seconds)
|
||||
frac = int(round((seconds - s) * 100000))
|
||||
if frac >= 100000:
|
||||
s += 1; frac = 0
|
||||
return f"{s // 3600:02d}:{s % 3600 // 60:02d}:{s % 60:02d}.{frac:05d}"
|
||||
|
||||
def sub(parent, tag, attrib=None, text=None):
|
||||
e = ET.SubElement(parent, tag)
|
||||
if attrib:
|
||||
for k, v in attrib.items():
|
||||
e.set(k, v)
|
||||
if text is not None:
|
||||
e.text = text
|
||||
return e
|
||||
|
||||
def add_refs(parent, tag, ids):
|
||||
for i in ids:
|
||||
sub(parent, tag, text=i)
|
||||
|
||||
def obj_block(cf, bid, t, x, y, z, dur, interpolation=0.0):
|
||||
b = sub(cf, "audioBlockFormat", {
|
||||
"audioBlockFormatID": bid, "rtime": ts(t), "duration": ts(dur)})
|
||||
sub(b, "cartesian", text="1")
|
||||
for c, v in (("X", x), ("Y", y), ("Z", z)):
|
||||
if c == "Z" and v == 0:
|
||||
continue
|
||||
p = sub(b, "position", {"coordinate": c})
|
||||
p.text = f"{v:.10f}"
|
||||
sub(b, "jumpPosition", {"interpolationLength": f"{interpolation:.5f}"}, text="1")
|
||||
|
||||
def build_axml(obj_tracks, duration_sec):
|
||||
adm = ET.Element("ebuCoreMain", {
|
||||
"xmlns": NS, "xmlns:xsi": XSI,
|
||||
"xsi:schemaLocation": f"{NS} ebucore.xsd", "lang": "en"})
|
||||
core = sub(adm, "coreMetadata")
|
||||
fmt = sub(core, "format")
|
||||
af = sub(fmt, "audioFormatExtended")
|
||||
|
||||
prog = sub(af, "audioProgramme", {
|
||||
"audioProgrammeID": "APR_1001", "audioProgrammeName": "EAC3JOC_Export",
|
||||
"start": ts(0), "end": ts(duration_sec)})
|
||||
add_refs(prog, "audioContentIDRef", ("ACO_1001", "ACO_1002"))
|
||||
bc = sub(af, "audioContent", {"audioContentID": "ACO_1001",
|
||||
"audioContentName": "EAC3JOC_Master_Content"})
|
||||
add_refs(bc, "audioObjectIDRef", ["AO_1001"])
|
||||
sub(bc, "dialogue", {"mixedContentKind": "0"})
|
||||
oc = sub(af, "audioContent", {"audioContentID": "ACO_1002",
|
||||
"audioContentName": "Objects"})
|
||||
add_refs(oc, "audioObjectIDRef", ["AO_%04x" % (0x100b + i) for i in range(N_OBJ)])
|
||||
sub(oc, "dialogue", {"mixedContentKind": "0"})
|
||||
|
||||
bed_o = sub(af, "audioObject", {"audioObjectID": "AO_1001", "audioObjectName": "Bed",
|
||||
"start": ts(0), "duration": ts(duration_sec)})
|
||||
sub(bed_o, "audioPackFormatIDRef", text="AP_00011001")
|
||||
add_refs(bed_o, "audioTrackUIDRef", ["ATU_%08x" % (i + 1) for i in range(10)])
|
||||
for i in range(N_OBJ):
|
||||
o = sub(af, "audioObject", {"audioObjectID": "AO_%04x" % (0x100b + i),
|
||||
"audioObjectName": f"Audio Object {i+1}",
|
||||
"start": ts(0), "duration": ts(duration_sec)})
|
||||
sub(o, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
|
||||
add_refs(o, "audioTrackUIDRef", ["ATU_%08x" % (i + 11)])
|
||||
|
||||
bp = sub(af, "audioPackFormat", {"audioPackFormatID": "AP_00011001",
|
||||
"audioPackFormatName": "EAC3JOCBedPack",
|
||||
"typeDefinition": "DirectSpeakers", "typeLabel": "0001"})
|
||||
add_refs(bp, "audioChannelFormatIDRef", ["AC_0001%04x" % (0x1001 + i) for i in range(10)])
|
||||
for i in range(N_OBJ):
|
||||
pk = sub(af, "audioPackFormat", {"audioPackFormatID": "AP_0003%04x" % (0x1001 + i),
|
||||
"audioPackFormatName": f"JOC_Object_{i+1}",
|
||||
"typeDefinition": "Objects", "typeLabel": "0003"})
|
||||
add_refs(pk, "audioChannelFormatIDRef", ["AC_0003%04x" % (0x1001 + i)])
|
||||
|
||||
for i in range(10):
|
||||
cf = sub(af, "audioChannelFormat", {"audioChannelFormatID": "AC_0001%04x" % (0x1001 + i),
|
||||
"audioChannelFormatName": BED_NAMES[i],
|
||||
"typeDefinition": "DirectSpeakers", "typeLabel": "0001"})
|
||||
b = sub(cf, "audioBlockFormat", {"audioBlockFormatID": "AB_0001%04x_00000001" % (0x1001 + i)})
|
||||
sub(b, "cartesian", text="1")
|
||||
x, y, z = BED_POS[i]
|
||||
for c, v in (("X", x), ("Y", y), ("Z", z)):
|
||||
if c == "Z" and v == 0:
|
||||
continue
|
||||
p = sub(b, "position", {"coordinate": c})
|
||||
p.text = f"{v:.10f}"
|
||||
sub(b, "speakerLabel", text=BED_LABELS[i])
|
||||
|
||||
for i, (oname, kfs) in enumerate(obj_tracks):
|
||||
cf = sub(af, "audioChannelFormat", {"audioChannelFormatID": "AC_0003%04x" % (0x1001 + i),
|
||||
"audioChannelFormatName": oname,
|
||||
"typeDefinition": "Objects", "typeLabel": "0003"})
|
||||
for k, keyframe in enumerate(kfs):
|
||||
t, x, y, z, dur = keyframe[:5]
|
||||
interpolation = keyframe[5] if len(keyframe) > 5 else 0.0
|
||||
obj_block(cf, "AB_0003%04x_%08x" % (0x1001 + i, k + 1),
|
||||
t, x, y, z, dur, interpolation)
|
||||
|
||||
for i in range(10):
|
||||
t = sub(af, "audioTrackUID", {"UID": "ATU_%08x" % (i + 1),
|
||||
"bitDepth": "24", "sampleRate": "48000"})
|
||||
sub(t, "audioTrackFormatIDRef", text="AT_0001%04x_01" % (0x1001 + i))
|
||||
sub(t, "audioPackFormatIDRef", text="AP_00011001")
|
||||
for i in range(N_OBJ):
|
||||
t = sub(af, "audioTrackUID", {"UID": "ATU_%08x" % (i + 11),
|
||||
"bitDepth": "24", "sampleRate": "48000"})
|
||||
sub(t, "audioTrackFormatIDRef", text="AT_0003%04x_01" % (0x1001 + i))
|
||||
sub(t, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
|
||||
|
||||
for i in range(10):
|
||||
tf = sub(af, "audioTrackFormat", {"audioTrackFormatID": "AT_0001%04x_01" % (0x1001 + i),
|
||||
"audioTrackFormatName": "PCM_" + BED_NAMES[i],
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(tf, "audioStreamFormatIDRef", text="AS_0001%04x" % (0x1001 + i))
|
||||
for i in range(N_OBJ):
|
||||
tf = sub(af, "audioTrackFormat", {"audioTrackFormatID": "AT_0003%04x_01" % (0x1001 + i),
|
||||
"audioTrackFormatName": "PCM_JOC_Object_%d" % (i + 1),
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(tf, "audioStreamFormatIDRef", text="AS_0003%04x" % (0x1001 + i))
|
||||
|
||||
for i in range(10):
|
||||
sf = sub(af, "audioStreamFormat", {"audioStreamFormatID": "AS_0001%04x" % (0x1001 + i),
|
||||
"audioStreamFormatName": "PCM_" + BED_NAMES[i],
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(sf, "audioChannelFormatIDRef", text="AC_0001%04x" % (0x1001 + i))
|
||||
sub(sf, "audioPackFormatIDRef", text="AP_00011001")
|
||||
sub(sf, "audioTrackFormatIDRef", text="AT_0001%04x_01" % (0x1001 + i))
|
||||
for i in range(N_OBJ):
|
||||
sf = sub(af, "audioStreamFormat", {"audioStreamFormatID": "AS_0003%04x" % (0x1001 + i),
|
||||
"audioStreamFormatName": "PCM_JOC_Object_%d" % (i + 1),
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(sf, "audioChannelFormatIDRef", text="AC_0003%04x" % (0x1001 + i))
|
||||
sub(sf, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
|
||||
sub(sf, "audioTrackFormatIDRef", text="AT_0003%04x_01" % (0x1001 + i))
|
||||
|
||||
return ET.tostring(adm, encoding="utf-8", xml_declaration=True)
|
||||
|
||||
def build_chna():
|
||||
out = bytearray()
|
||||
out += struct.pack("<HH", 25, 25)
|
||||
for i in range(10):
|
||||
out += struct.pack("<H", i + 1)
|
||||
out += ("ATU_%08x" % (i + 1)).encode()
|
||||
out += ("AT_0001%04x_01" % (0x1001 + i)).encode()
|
||||
out += b"AP_00011001" + b"\x00"
|
||||
for i in range(N_OBJ):
|
||||
out += struct.pack("<H", i + 11)
|
||||
out += ("ATU_%08x" % (i + 11)).encode()
|
||||
out += ("AT_0003%04x_01" % (0x1001 + i)).encode()
|
||||
out += ("AP_0003%04x" % (0x1001 + i)).encode() + b"\x00"
|
||||
return bytes(out)
|
||||
|
||||
def _checksum(seg):
|
||||
s = len(seg)
|
||||
for b in seg:
|
||||
s += b
|
||||
return (~s + 1) & 0xFF
|
||||
|
||||
def build_dbmd(object_count=25, joc_binaural_mode=4):
|
||||
"""仅覆盖 segment 10 中 JOC object slots 10..24 的 mode 低 3 bit。"""
|
||||
mode = operator.index(joc_binaural_mode)
|
||||
if mode not in JOC_BINAURAL_MODES.values():
|
||||
raise ValueError(f"invalid JOC binaural render mode: {mode}")
|
||||
out = bytearray(struct.pack("<I", 0x01000006))
|
||||
dd = bytearray(96)
|
||||
dd[1] = 0x47
|
||||
dd[5] = 0x60
|
||||
dd[8] = 0x24; dd[9] = 0x24
|
||||
out.append(7); out += struct.pack("<H", 96); out += bytes(dd)
|
||||
out.append(_checksum(dd))
|
||||
at = bytearray(248)
|
||||
c0 = b"Created with EAC3JOC"; c1 = b"EAC3JOC Python Renderer"
|
||||
at[0:len(c0)] = c0
|
||||
at[32:32 + len(c1)] = c1
|
||||
at[96], at[97], at[98] = 2, 1, 0
|
||||
at[103] = 0x03
|
||||
at[106] = 0x01
|
||||
at[111] = 0x22; at[112] = 0xFF
|
||||
out.append(9); out += struct.pack("<H", 248); out += bytes(at)
|
||||
out.append(_checksum(at))
|
||||
ob = bytearray(5 + 262 + object_count)
|
||||
ob[0:4] = struct.pack("<I", 0xF8726FBD)
|
||||
ob[4] = object_count
|
||||
for i in range(5 + 262, len(ob)):
|
||||
ob[i] = 0x84
|
||||
# sync (4), count (2), reserved (1), nine 15-byte config trims,
|
||||
# then one trim-bypass byte per track before the headphone modes.
|
||||
# Preserve the existing template's bed fields and trailing bytes.
|
||||
object_modes = 4 + 2 + 1 + 9 * 15 + object_count
|
||||
for i in range(10, min(object_count, 10 + N_OBJ)):
|
||||
ob[object_modes + i] = (ob[object_modes + i] & 0xF8) | mode
|
||||
out.append(10); out += struct.pack("<H", len(ob)); out += bytes(ob)
|
||||
out.append(_checksum(ob))
|
||||
out += b"\x00\x00"
|
||||
return bytes(out)
|
||||
|
||||
class Sink25:
|
||||
"""RF64 ADM BWF writer.
|
||||
|
||||
Header layout is fixed so that sizes can be patched without rereading the
|
||||
file: RF64+size+WAVE (12) + ds64 chunk (8+28) + fmt chunk (8+16) + data
|
||||
chunk header (8). Sizes beyond 32 bits follow the RF64 convention: the
|
||||
chunk size field holds 0xFFFFFFFF and the true value lives in ds64.
|
||||
"""
|
||||
_DS64_BODY_OFFSET = 20
|
||||
_DATA_SIZE_OFFSET = 76
|
||||
|
||||
def __init__(self, path, channels, rate):
|
||||
self.ch = channels; self.rate = rate; self.frames = 0
|
||||
self.fp = open(path, "wb+")
|
||||
self.fp.write(b"RF64" + struct.pack("<I", 0xFFFFFFFF) + b"WAVE")
|
||||
self._chunk(b"ds64", b"\x00" * 28)
|
||||
self._chunk(b"fmt ", self._fmt())
|
||||
self._chunk(b"data", b"")
|
||||
def _chunk(self, cid, body):
|
||||
self.fp.write(cid + struct.pack("<I", len(body)) + body)
|
||||
if len(body) & 1:
|
||||
self.fp.write(b"\x00")
|
||||
def _fmt(self):
|
||||
return struct.pack("<HHIIHH", 1, self.ch, self.rate,
|
||||
self.rate * self.ch * 3, self.ch * 3, 24)
|
||||
def write_block(self, arr):
|
||||
arr = arr.reshape(-1, self.ch)
|
||||
i24 = (np.clip(arr, -1.0, 1.0) * 8388607.0).astype(np.int32)
|
||||
self.fp.write(i24.view(np.uint8).reshape(-1, 4)[:, :3].tobytes())
|
||||
self.frames += arr.shape[0]
|
||||
def finalize(self, axml_bytes, chna_bytes, dbmd_bytes):
|
||||
data_len = self.frames * self.ch * 3
|
||||
self._chunk(b"axml", axml_bytes)
|
||||
self._chunk(b"chna", chna_bytes)
|
||||
self._chunk(b"dbmd", dbmd_bytes)
|
||||
self.fp.seek(0, 2); total = self.fp.tell()
|
||||
# RF64: 超过 32-bit 的 chunk size 字段写 0xFFFFFFFF,真实大小回填 ds64。
|
||||
self.fp.seek(self._DATA_SIZE_OFFSET)
|
||||
self.fp.write(struct.pack(
|
||||
"<I", data_len if data_len <= 0xFFFFFFFF else 0xFFFFFFFF))
|
||||
self.fp.seek(self._DS64_BODY_OFFSET)
|
||||
self.fp.write(struct.pack("<QQQI", total - 8, data_len, self.frames, 0))
|
||||
self.fp.flush()
|
||||
self.fp.close()
|
||||
|
||||
def build_master(out_path, bed_mm, obj_mm, kf_tracks, duration_sec, rate=48000,
|
||||
block=480000, joc_binaural_mode=4):
|
||||
n = min(bed_mm.shape[0], obj_mm.shape[0])
|
||||
try:
|
||||
from . import adm_serializer
|
||||
except ImportError:
|
||||
import adm_serializer
|
||||
serial_axml = adm_serializer.build_axml
|
||||
axml = serial_axml(kf_tracks, duration_sec)
|
||||
chna = build_chna()
|
||||
dbmd = build_dbmd(25, joc_binaural_mode=joc_binaural_mode)
|
||||
sink = Sink25(out_path, 25, rate)
|
||||
for st in range(0, n, block):
|
||||
en = min(n, st + block)
|
||||
blk = np.hstack((np.asarray(bed_mm[st:en], dtype=np.float32),
|
||||
np.asarray(obj_mm[st:en, 1:16], dtype=np.float32)))
|
||||
sink.write_block(blk)
|
||||
sink.finalize(axml, chna, dbmd)
|
||||
print(f"master25 -> {out_path} ({duration_sec:.2f}s, 25ch, axml={len(axml)}B, "
|
||||
f"chna={len(chna)}B, dbmd={len(dbmd)}B)")
|
||||
@@ -0,0 +1,136 @@
|
||||
"""把 7.1.2 bed、15 个对象及其位置轨迹序列化为 ADM axml。
|
||||
|
||||
序列化结果采用固定元素顺序、属性顺序和十六进制 ADM 标识符,便于稳定输出和校验。
|
||||
"""
|
||||
BED_NAMES = ["RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter", "RoomCentricLFE",
|
||||
"RoomCentricLeftSideSurround", "RoomCentricRightSideSurround",
|
||||
"RoomCentricLeftRearSurround", "RoomCentricRightRearSurround",
|
||||
"RoomCentricLeftTopSurround", "RoomCentricRightTopSurround"]
|
||||
BED_LABELS = ["RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss", "RC_Rss",
|
||||
"RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"]
|
||||
BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0), (-1.0, 1.0, -1.0),
|
||||
(-1.0, 0.0, 0.0), (1.0, 0.0, 0.0), (-1.0, -1.0, 0.0), (1.0, -1.0, 0.0),
|
||||
(-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
|
||||
N_OBJ = 15
|
||||
|
||||
def ts(seconds):
|
||||
s = int(seconds)
|
||||
frac = int(round((seconds - s) * 100000))
|
||||
if frac >= 100000:
|
||||
s += 1; frac = 0
|
||||
return f"{s // 3600:02d}:{s % 3600 // 60:02d}:{s % 60:02d}.{frac:05d}"
|
||||
|
||||
def esc(v):
|
||||
return (str(v).replace("&", "&").replace("<", "<").replace(">", ">"))
|
||||
|
||||
def build_axml(obj_tracks, duration_sec):
|
||||
"""obj_tracks: [(name, [(rtime, x, y, z, dur), ...]) ×15]"""
|
||||
w = []
|
||||
a = w.append
|
||||
a('<?xml version="1.0" encoding="utf-8"?>')
|
||||
a('<ebuCoreMain xsi:schemaLocation="urn:ebu:metadata-schema:ebuCore_2016 ebucore.xsd" '
|
||||
'lang="en" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" '
|
||||
'xmlns="urn:ebu:metadata-schema:ebuCore_2016">')
|
||||
a('<coreMetadata><format><audioFormatExtended>')
|
||||
a(f'<audioProgramme audioProgrammeID="APR_1001" audioProgrammeName="EAC3JOC_Export" '
|
||||
f'start="{ts(0)}" end="{ts(duration_sec)}">')
|
||||
a('<audioContentIDRef>ACO_1001</audioContentIDRef>')
|
||||
a('<audioContentIDRef>ACO_1002</audioContentIDRef>')
|
||||
a('</audioProgramme>')
|
||||
a('<audioContent audioContentID="ACO_1001" audioContentName="EAC3JOC_Master_Content">')
|
||||
a('<audioObjectIDRef>AO_1001</audioObjectIDRef>')
|
||||
a('<dialogue mixedContentKind="0">2</dialogue>')
|
||||
a('</audioContent>')
|
||||
a('<audioContent audioContentID="ACO_1002" audioContentName="Objects">')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioObjectIDRef>AO_{0x100b + i:04x}</audioObjectIDRef>')
|
||||
a('<dialogue mixedContentKind="0">2</dialogue>')
|
||||
a('</audioContent>')
|
||||
a(f'<audioObject audioObjectID="AO_1001" audioObjectName="Bed" '
|
||||
f'start="{ts(0)}" duration="{ts(duration_sec)}">')
|
||||
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
|
||||
for i in range(10):
|
||||
a(f'<audioTrackUIDRef>ATU_{i + 1:08x}</audioTrackUIDRef>')
|
||||
a('</audioObject>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioObject audioObjectID="AO_{0x100b + i:04x}" audioObjectName="Audio Object {i+1}" '
|
||||
f'start="{ts(0)}" duration="{ts(duration_sec)}">')
|
||||
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
|
||||
a(f'<audioTrackUIDRef>ATU_{11 + i:08x}</audioTrackUIDRef>')
|
||||
a('</audioObject>')
|
||||
a('<audioPackFormat audioPackFormatID="AP_00011001" audioPackFormatName="EAC3JOCBedPack" '
|
||||
'typeDefinition="DirectSpeakers" typeLabel="0001">')
|
||||
for i in range(10):
|
||||
a(f'<audioChannelFormatIDRef>AC_0001{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a('</audioPackFormat>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioPackFormat audioPackFormatID="AP_0003{0x1001 + i:04x}" '
|
||||
f'audioPackFormatName="JOC_Object_{i+1}" typeDefinition="Objects" typeLabel="0003">')
|
||||
a(f'<audioChannelFormatIDRef>AC_0003{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a('</audioPackFormat>')
|
||||
for i in range(10):
|
||||
a(f'<audioChannelFormat audioChannelFormatID="AC_0001{0x1001 + i:04x}" '
|
||||
f'audioChannelFormatName="{BED_NAMES[i]}" typeDefinition="DirectSpeakers" typeLabel="0001">')
|
||||
a(f'<audioBlockFormat audioBlockFormatID="AB_0001{0x1001 + i:04x}_00000001">')
|
||||
a('<cartesian>1</cartesian>')
|
||||
x, y, z = BED_POS[i]
|
||||
a(f'<position coordinate="X">{x:.10f}</position>')
|
||||
a(f'<position coordinate="Y">{y:.10f}</position>')
|
||||
if z != 0:
|
||||
a(f'<position coordinate="Z">{z:.10f}</position>')
|
||||
a(f'<speakerLabel>{BED_LABELS[i]}</speakerLabel>')
|
||||
a('</audioBlockFormat>')
|
||||
a('</audioChannelFormat>')
|
||||
for i, (oname, kfs) in enumerate(obj_tracks):
|
||||
a(f'<audioChannelFormat audioChannelFormatID="AC_0003{0x1001 + i:04x}" '
|
||||
f'audioChannelFormatName="{oname}" typeDefinition="Objects" typeLabel="0003">')
|
||||
for k, keyframe in enumerate(kfs):
|
||||
t, x, y, z, dur = keyframe[:5]
|
||||
interpolation = keyframe[5] if len(keyframe) > 5 else 0.0
|
||||
a(f'<audioBlockFormat audioBlockFormatID="AB_0003{0x1001 + i:04x}_{k + 1:08x}" '
|
||||
f'rtime="{ts(t)}" duration="{ts(dur)}">')
|
||||
a('<cartesian>1</cartesian>')
|
||||
a(f'<position coordinate="X">{x:.10f}</position>')
|
||||
a(f'<position coordinate="Y">{y:.10f}</position>')
|
||||
if z != 0:
|
||||
a(f'<position coordinate="Z">{z:.10f}</position>')
|
||||
a(f'<jumpPosition interpolationLength="{interpolation:.5f}">1</jumpPosition>')
|
||||
a('</audioBlockFormat>')
|
||||
a('</audioChannelFormat>')
|
||||
for i in range(10):
|
||||
a(f'<audioTrackUID UID="ATU_{i + 1:08x}" bitDepth="24" sampleRate="48000">')
|
||||
a(f'<audioTrackFormatIDRef>AT_0001{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
|
||||
a('</audioTrackUID>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioTrackUID UID="ATU_{11 + i:08x}" bitDepth="24" sampleRate="48000">')
|
||||
a(f'<audioTrackFormatIDRef>AT_0003{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
|
||||
a('</audioTrackUID>')
|
||||
for i in range(10):
|
||||
a(f'<audioTrackFormat audioTrackFormatID="AT_0001{0x1001 + i:04x}_01" '
|
||||
f'audioTrackFormatName="PCM_{BED_NAMES[i]}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioStreamFormatIDRef>AS_0001{0x1001 + i:04x}</audioStreamFormatIDRef>')
|
||||
a('</audioTrackFormat>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioTrackFormat audioTrackFormatID="AT_0003{0x1001 + i:04x}_01" '
|
||||
f'audioTrackFormatName="PCM_JOC_Object_{i+1}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioStreamFormatIDRef>AS_0003{0x1001 + i:04x}</audioStreamFormatIDRef>')
|
||||
a('</audioTrackFormat>')
|
||||
for i in range(10):
|
||||
a(f'<audioStreamFormat audioStreamFormatID="AS_0001{0x1001 + i:04x}" '
|
||||
f'audioStreamFormatName="PCM_{BED_NAMES[i]}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioChannelFormatIDRef>AC_0001{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
|
||||
a(f'<audioTrackFormatIDRef>AT_0001{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a('</audioStreamFormat>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioStreamFormat audioStreamFormatID="AS_0003{0x1001 + i:04x}" '
|
||||
f'audioStreamFormatName="PCM_JOC_Object_{i+1}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioChannelFormatIDRef>AC_0003{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
|
||||
a(f'<audioTrackFormatIDRef>AT_0003{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a('</audioStreamFormat>')
|
||||
a('</audioFormatExtended></format></coreMetadata>')
|
||||
a('</ebuCoreMain>')
|
||||
return ''.join(w).encode('utf-8')
|
||||
@@ -0,0 +1,199 @@
|
||||
"""校验 ADM BWF 的 RF64、通道、axml、chna、dbmd 和对象引用结构。
|
||||
|
||||
用法:``python src/adm_validate.py <file.wav> [more.wav ...]``,全部通过时退出码为 0。
|
||||
"""
|
||||
import struct, sys, re, os
|
||||
|
||||
def fail(msgs, m): msgs.append(m)
|
||||
|
||||
def walk_chunks(path):
|
||||
chunks, ds64 = [], {}
|
||||
with open(path, "rb") as f:
|
||||
riff = f.read(4); f.read(4); wave = f.read(4)
|
||||
if riff not in (b"RIFF", b"RF64"):
|
||||
return None, None, f"File does not have a 'RIFF' or 'RF64' chunk"
|
||||
if wave != b"WAVE":
|
||||
return None, None, "File does not have a required 'WAVE' chunk"
|
||||
while True:
|
||||
off = f.tell()
|
||||
cid = f.read(4)
|
||||
if len(cid) < 4: break
|
||||
sz = struct.unpack("<I", f.read(4))[0]
|
||||
if cid == b"ds64":
|
||||
body = f.read(sz + (sz & 1))
|
||||
riff64, data64, sample64, _ = struct.unpack("<QQQI", body[:28])
|
||||
ds64 = dict(riff64=riff64, data64=data64, sample64=sample64)
|
||||
chunks.append(("ds64", off, sz)); continue
|
||||
chunks.append((cid.decode("latin1"), off, sz))
|
||||
eff = ds64.get("data64", sz) if (sz == 0xFFFFFFFF and cid == b"data") else sz
|
||||
f.seek(off + 8 + eff + (eff & 1))
|
||||
return chunks, ds64, None
|
||||
|
||||
def read_body(path, chunks, cid):
|
||||
for c, off, sz in chunks:
|
||||
if c == cid:
|
||||
with open(path, "rb") as f:
|
||||
f.seek(off + 8)
|
||||
return f.read(sz)
|
||||
return None
|
||||
|
||||
def parse_chna(body):
|
||||
n_track, n_uid = struct.unpack("<HH", body[:4])
|
||||
rows, p = [], 4
|
||||
while p + 40 <= len(body):
|
||||
trk = struct.unpack("<H", body[p:p+2])[0]
|
||||
uid = body[p+2:p+14].rstrip(b"\x00").decode()
|
||||
tf = body[p+14:p+28].rstrip(b"\x00").decode()
|
||||
pk = body[p+28:p+40].rstrip(b"\x00").decode()
|
||||
rows.append((trk, uid, tf, pk)); p += 40
|
||||
return n_track, n_uid, rows
|
||||
|
||||
def decode_channel_input(ao_id):
|
||||
"""将 ``AO_xxxx`` 的十六进制标识符解码为低 12 位通道输入号。"""
|
||||
m = re.fullmatch(r"AO_([0-9a-fA-F]+)", ao_id)
|
||||
if not m:
|
||||
return None
|
||||
v = int(m.group(1), 16)
|
||||
if v > 0x1FFF: # 超过 12 位通道域
|
||||
return None
|
||||
return v & 0x0FFF
|
||||
|
||||
def ts_sec(s):
|
||||
h, m, rest = s.split(":")
|
||||
return int(h) * 3600 + int(m) * 60 + float(rest)
|
||||
|
||||
def validate(path, axml_override=None, chna_override=None):
|
||||
msgs = []
|
||||
chunks, ds64, err = walk_chunks(path)
|
||||
if err:
|
||||
return [err]
|
||||
have = {c for c, _, _ in chunks}
|
||||
for need in ("fmt ", "data", "axml", "chna", "dbmd"):
|
||||
if need not in have:
|
||||
fail(msgs, f"File does not have a required '{need.strip()}' chunk")
|
||||
if msgs:
|
||||
return msgs
|
||||
fmt = read_body(path, chunks, "fmt ")
|
||||
f_tag, f_ch, f_rate, _, _, f_bits = struct.unpack("<HHIIHH", fmt[:16])
|
||||
chna_body = chna_override if chna_override is not None else read_body(path, chunks, "chna")
|
||||
n_track, n_uid, rows = parse_chna(chna_body)
|
||||
if f_ch != n_track:
|
||||
fail(msgs, f"Mismatched number of audio channels and chna entries "
|
||||
f"(fmt={f_ch} chna={n_track})")
|
||||
ax_raw = axml_override if axml_override is not None else read_body(path, chunks, "axml")
|
||||
ax = ax_raw.decode("utf-8")
|
||||
|
||||
# --- audioObjectID 十六进制通道输入号解码 ---
|
||||
objs = re.findall(r'audioObjectID="(AO_[0-9a-zA-Z]+)"', ax)
|
||||
bed_ch, obj_ch = [], []
|
||||
for ao in objs:
|
||||
cid = decode_channel_input(ao)
|
||||
if cid is None:
|
||||
fail(msgs, f"Invalid ADM BWF XML format: cannot decode channel "
|
||||
f"input ID from AudioObjectID '{ao}'")
|
||||
continue
|
||||
if ao == "AO_1001":
|
||||
bed_ch.append(cid)
|
||||
else:
|
||||
if cid <= 10:
|
||||
fail(msgs, f"Source channel index should be greater than 10 "
|
||||
f"for objects ('{ao}' -> {cid})")
|
||||
obj_ch.append(cid)
|
||||
|
||||
# UID 十六进制 → 必须与 chna 表一致
|
||||
uid_map = {uid: trk for trk, uid, tf, pk in rows}
|
||||
for uid in re.findall(r'UID="(ATU_[0-9a-zA-Z]+)"', ax):
|
||||
if uid not in uid_map:
|
||||
fail(msgs, f"'{uid}' is not referenced in 'chna' chunk UID table")
|
||||
continue
|
||||
m = re.fullmatch(r"ATU_([0-9a-fA-F]+)", uid)
|
||||
if m:
|
||||
v = int(m.group(1), 16)
|
||||
if v > 128:
|
||||
fail(msgs, f"Channel index out of range (UID {uid} -> {v})")
|
||||
|
||||
# 轨数一致性:axml audioTrackUID 数 == fmt 声道数
|
||||
n_tu = len(re.findall(r"<audioTrackUID ", ax))
|
||||
if n_tu != f_ch:
|
||||
fail(msgs, f"Number of channels declared in ADM ({n_tu}) does not "
|
||||
f"match 'fmt ' chunk ({f_ch})")
|
||||
|
||||
# sampleRate / bitDepth 一致
|
||||
for sr in set(re.findall(r'sampleRate="(\d+)"', ax)):
|
||||
if int(sr) != f_rate:
|
||||
fail(msgs, f"Mismatched track sample rate between ADM and WAV ({sr} vs {f_rate})")
|
||||
for bd in set(re.findall(r'bitDepth="(\d+)"', ax)):
|
||||
if int(bd) != f_bits:
|
||||
fail(msgs, f"Mismatched track bit depth between ADM and WAV ({bd} vs {f_bits})")
|
||||
|
||||
# audioProgramme 唯一性 / audioContent ≥1
|
||||
if ax.count("<audioProgramme ") != 1:
|
||||
fail(msgs, "ADM has more than one audioProgramme object -- there must be only one"
|
||||
if ax.count("<audioProgramme ") > 1
|
||||
else "ADM does not have a required audioProgramme object")
|
||||
if "<audioContent " not in ax:
|
||||
fail(msgs, "audioProgramme object does not have a required audioContent object")
|
||||
|
||||
# 每个 channelFormat ≥1 blockFormat + 对象块链连续性
|
||||
cfs = re.findall(r'<audioChannelFormat [^>]*typeLabel="0003".*?</audioChannelFormat>', ax, re.S)
|
||||
object_block_formats = 0
|
||||
for seg in cfs:
|
||||
cf_id = re.search(r'audioChannelFormatID="([^"]+)"', seg).group(1)
|
||||
block_xml = re.findall(r'<audioBlockFormat [^>]*rtime="[^"]+".*?</audioBlockFormat>',
|
||||
seg, re.S)
|
||||
object_block_formats += len(block_xml)
|
||||
blocks = []
|
||||
for block_index, block in enumerate(block_xml, 1):
|
||||
timing = re.search(r'rtime="([^"]+)" duration="([^"]+)"', block)
|
||||
if timing is None:
|
||||
continue
|
||||
rtime, duration = timing.groups()
|
||||
blocks.append((rtime, duration))
|
||||
jump = re.search(
|
||||
r'<jumpPosition interpolationLength="([^"]+)">1</jumpPosition>', block)
|
||||
if jump is not None and float(jump.group(1)) > ts_sec(duration) + 1e-8:
|
||||
fail(msgs, f"Interpolation length exceeds duration in block format "
|
||||
f"{block_index} of {cf_id}: {jump.group(1)} > {duration}")
|
||||
if not blocks:
|
||||
fail(msgs, f"AudioChannelFormat {cf_id} is missing audioBlockFormat sub-element")
|
||||
continue
|
||||
for i in range(len(blocks) - 1):
|
||||
end_i = ts_sec(blocks[i][0]) + ts_sec(blocks[i][1])
|
||||
nxt = ts_sec(blocks[i + 1][0])
|
||||
if abs(end_i - nxt) > 2e-5:
|
||||
fail(msgs, f"Time gap between block format {i+1} and {i+2} of {cf_id}: "
|
||||
f"{end_i:.5f} vs {nxt:.5f}")
|
||||
return msgs, dict(fmt_ch=f_ch, fmt_rate=f_rate, fmt_bits=f_bits,
|
||||
chna=n_track, objects=len(obj_ch), bed=len(bed_ch),
|
||||
trackUIDs=n_tu, axml_bytes=len(ax_raw),
|
||||
audioBlockFormats=ax.count("<audioBlockFormat "),
|
||||
objectBlockFormats=object_block_formats)
|
||||
|
||||
def main():
|
||||
import argparse
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("files", nargs="+")
|
||||
ap.add_argument("--axml-file", default=None, help="用该文件内容替换 wav 内 axml(对照实验)")
|
||||
ap.add_argument("--chna-file", default=None, help="用该文件内容替换 wav 内 chna(对照实验)")
|
||||
a = ap.parse_args()
|
||||
ax_o = open(a.axml_file, "rb").read() if a.axml_file else None
|
||||
ch_o = open(a.chna_file, "rb").read() if a.chna_file else None
|
||||
rc = 0
|
||||
for p in a.files:
|
||||
r = validate(p, axml_override=ax_o, chna_override=ch_o)
|
||||
name = os.path.basename(p)
|
||||
if isinstance(r, list):
|
||||
msgs, info = r, {}
|
||||
else:
|
||||
msgs, info = r
|
||||
if msgs:
|
||||
rc = 1
|
||||
print(f"[FAIL] {name}")
|
||||
for m in msgs:
|
||||
print(" -", m)
|
||||
else:
|
||||
print(f"[PASS] {name} {info}")
|
||||
return rc
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,249 +0,0 @@
|
||||
|
||||
#include "joc_core.h"
|
||||
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
#include "eac3_transport/eac3_reader.h"
|
||||
#include "emdf/emdf_parser.h"
|
||||
#include "foundation/status.h"
|
||||
#include "joc_bitstream/joc_parser.h"
|
||||
|
||||
namespace {
|
||||
|
||||
thread_local std::string g_detail;
|
||||
|
||||
joc_error finish(const joc::Status& status) {
|
||||
if (status.ok()) {
|
||||
g_detail.clear();
|
||||
return JOC_OK;
|
||||
}
|
||||
g_detail.assign(status.stage());
|
||||
g_detail.append(": ");
|
||||
g_detail.append(status.message());
|
||||
return status.code();
|
||||
}
|
||||
|
||||
joc_error arg_fail(const char* message) {
|
||||
return finish(joc::Status::fail(JOC_ERR_INVALID_ARGUMENT, "core", message));
|
||||
}
|
||||
|
||||
void fill_emdf_info(const joc::emdf::Container& container, joc_emdf_info* out) {
|
||||
std::memset(out, 0, sizeof(*out));
|
||||
out->struct_size = sizeof(joc_emdf_info);
|
||||
out->struct_version = JOC_EMDF_INFO_VERSION;
|
||||
out->start_bit = static_cast<std::uint32_t>(container.start_bit);
|
||||
out->container_bytes = static_cast<std::uint32_t>(container.raw_size);
|
||||
out->payload_count = static_cast<std::uint32_t>(container.payload_count);
|
||||
for (std::size_t i = 0; i < container.payload_count; ++i) {
|
||||
out->payloads[i].id = container.payloads[i].id;
|
||||
out->payloads[i].sample_offset = container.payloads[i].sample_offset;
|
||||
out->payloads[i].bit_offset = static_cast<std::uint32_t>(container.payloads[i].bit_offset);
|
||||
out->payloads[i].size = static_cast<std::uint32_t>(container.payloads[i].size);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
extern "C" {
|
||||
|
||||
std::uint32_t JOC_CALL joc_abi_version(void) { return JOC_ABI_VERSION; }
|
||||
|
||||
const char* JOC_CALL joc_version_string(void) { return "0.1.0-m1"; }
|
||||
|
||||
std::uint32_t JOC_CALL joc_event_size(void) { return static_cast<std::uint32_t>(sizeof(joc_event)); }
|
||||
std::uint32_t JOC_CALL joc_task_config_size(void) {
|
||||
return static_cast<std::uint32_t>(sizeof(joc_task_config));
|
||||
}
|
||||
std::uint32_t JOC_CALL joc_task_result_size(void) {
|
||||
return static_cast<std::uint32_t>(sizeof(joc_task_result));
|
||||
}
|
||||
|
||||
const char* JOC_CALL joc_build_info(void) {
|
||||
static const std::string info = [] {
|
||||
std::string text = "joc_core 0.1.0-m1 (";
|
||||
#if defined(_MSC_VER)
|
||||
text += "msvc " + std::to_string(_MSC_VER);
|
||||
#elif defined(__clang__)
|
||||
text += std::string("clang ") + __clang_version__;
|
||||
#elif defined(__GNUC__)
|
||||
text += "gcc " + std::to_string(__GNUC__) + "." + std::to_string(__GNUC_MINOR__);
|
||||
#else
|
||||
text += "unknown-compiler";
|
||||
#endif
|
||||
#if defined(_M_AMD64) || defined(__x86_64__)
|
||||
text += ", x64";
|
||||
#elif defined(_M_ARM64) || defined(__aarch64__)
|
||||
text += ", arm64";
|
||||
#elif defined(_M_IX86) || defined(__i386__)
|
||||
text += ", x86";
|
||||
#endif
|
||||
text += ", c++";
|
||||
text += std::to_string(static_cast<long long>(__cplusplus / 100 % 100));
|
||||
text += ")";
|
||||
return text;
|
||||
}();
|
||||
return info.c_str();
|
||||
}
|
||||
|
||||
const char* JOC_CALL joc_error_name(joc_error code) {
|
||||
switch (code) {
|
||||
case JOC_OK: return "JOC_OK";
|
||||
case JOC_ERR_INVALID_ARGUMENT: return "JOC_ERR_INVALID_ARGUMENT";
|
||||
case JOC_ERR_INVALID_CONFIG: return "JOC_ERR_INVALID_CONFIG";
|
||||
case JOC_ERR_OUT_OF_MEMORY: return "JOC_ERR_OUT_OF_MEMORY";
|
||||
case JOC_ERR_IO: return "JOC_ERR_IO";
|
||||
case JOC_ERR_UNSUPPORTED_PLATFORM: return "JOC_ERR_UNSUPPORTED_PLATFORM";
|
||||
case JOC_ERR_LIBRARY_MISSING: return "JOC_ERR_LIBRARY_MISSING";
|
||||
case JOC_ERR_INPUT_NOT_FOUND: return "JOC_ERR_INPUT_NOT_FOUND";
|
||||
case JOC_ERR_INPUT_FORMAT: return "JOC_ERR_INPUT_FORMAT";
|
||||
case JOC_ERR_EAC3_SYNCFRAME: return "JOC_ERR_EAC3_SYNCFRAME";
|
||||
case JOC_ERR_EMDF_TRANSPORT: return "JOC_ERR_EMDF_TRANSPORT";
|
||||
case JOC_ERR_EMDF_SYNTAX: return "JOC_ERR_EMDF_SYNTAX";
|
||||
case JOC_ERR_JOC_SYNTAX: return "JOC_ERR_JOC_SYNTAX";
|
||||
case JOC_ERR_JOC_UNSUPPORTED_VARIANT: return "JOC_ERR_JOC_UNSUPPORTED_VARIANT";
|
||||
case JOC_ERR_OAMD_SYNTAX: return "JOC_ERR_OAMD_SYNTAX";
|
||||
case JOC_ERR_OAMD_UNSUPPORTED_VARIANT: return "JOC_ERR_OAMD_UNSUPPORTED_VARIANT";
|
||||
case JOC_ERR_BITSTREAM_TRUNCATED: return "JOC_ERR_BITSTREAM_TRUNCATED";
|
||||
case JOC_ERR_BITSTREAM_PADDING: return "JOC_ERR_BITSTREAM_PADDING";
|
||||
case JOC_ERR_HRTF_NOT_FOUND: return "JOC_ERR_HRTF_NOT_FOUND";
|
||||
case JOC_ERR_HRTF_FORMAT: return "JOC_ERR_HRTF_FORMAT";
|
||||
case JOC_ERR_HRTF_VERSION: return "JOC_ERR_HRTF_VERSION";
|
||||
case JOC_ERR_HRTF_HASH: return "JOC_ERR_HRTF_HASH";
|
||||
case JOC_ERR_HRTF_UNSUPPORTED_CONVENTION: return "JOC_ERR_HRTF_UNSUPPORTED_CONVENTION";
|
||||
case JOC_ERR_LAYOUT_UNSUPPORTED: return "JOC_ERR_LAYOUT_UNSUPPORTED";
|
||||
case JOC_ERR_RENDER_FAILED: return "JOC_ERR_RENDER_FAILED";
|
||||
case JOC_ERR_OUTPUT_OPEN: return "JOC_ERR_OUTPUT_OPEN";
|
||||
case JOC_ERR_OUTPUT_WRITE: return "JOC_ERR_OUTPUT_WRITE";
|
||||
case JOC_ERR_OUTPUT_CLIP_ABORT: return "JOC_ERR_OUTPUT_CLIP_ABORT";
|
||||
case JOC_ERR_ADM_VALIDATION: return "JOC_ERR_ADM_VALIDATION";
|
||||
case JOC_ERR_CANCELLED: return "JOC_ERR_CANCELLED";
|
||||
case JOC_ERR_STATE: return "JOC_ERR_STATE";
|
||||
case JOC_ERR_NOT_SUPPORTED: return "JOC_ERR_NOT_SUPPORTED";
|
||||
case JOC_ERR_INTERNAL: return "JOC_ERR_INTERNAL";
|
||||
default: return "JOC_ERR_UNKNOWN";
|
||||
}
|
||||
}
|
||||
|
||||
const char* JOC_CALL joc_error_stage(joc_error code) {
|
||||
switch (code) {
|
||||
case JOC_ERR_EAC3_SYNCFRAME:
|
||||
case JOC_ERR_INPUT_NOT_FOUND:
|
||||
case JOC_ERR_INPUT_FORMAT:
|
||||
return "eac3_transport";
|
||||
case JOC_ERR_EMDF_TRANSPORT:
|
||||
case JOC_ERR_EMDF_SYNTAX:
|
||||
return "emdf";
|
||||
case JOC_ERR_JOC_SYNTAX:
|
||||
case JOC_ERR_JOC_UNSUPPORTED_VARIANT:
|
||||
return "joc";
|
||||
case JOC_ERR_OAMD_SYNTAX:
|
||||
case JOC_ERR_OAMD_UNSUPPORTED_VARIANT:
|
||||
return "oamd";
|
||||
case JOC_ERR_BITSTREAM_TRUNCATED:
|
||||
case JOC_ERR_BITSTREAM_PADDING:
|
||||
return "bitstream";
|
||||
case JOC_ERR_HRTF_NOT_FOUND:
|
||||
case JOC_ERR_HRTF_FORMAT:
|
||||
case JOC_ERR_HRTF_VERSION:
|
||||
case JOC_ERR_HRTF_HASH:
|
||||
case JOC_ERR_HRTF_UNSUPPORTED_CONVENTION:
|
||||
return "hrtf";
|
||||
case JOC_ERR_LAYOUT_UNSUPPORTED:
|
||||
case JOC_ERR_RENDER_FAILED:
|
||||
return "render";
|
||||
case JOC_ERR_OUTPUT_OPEN:
|
||||
case JOC_ERR_OUTPUT_WRITE:
|
||||
case JOC_ERR_OUTPUT_CLIP_ABORT:
|
||||
case JOC_ERR_ADM_VALIDATION:
|
||||
return "output";
|
||||
case JOC_ERR_CANCELLED:
|
||||
case JOC_ERR_STATE:
|
||||
case JOC_ERR_NOT_SUPPORTED:
|
||||
case JOC_ERR_INTERNAL:
|
||||
return "task";
|
||||
default:
|
||||
return "core";
|
||||
}
|
||||
}
|
||||
|
||||
const char* JOC_CALL joc_last_error_detail(void) { return g_detail.c_str(); }
|
||||
|
||||
joc_error JOC_CALL joc_parse_id14(const std::uint8_t* payload, std::size_t payload_size,
|
||||
joc_frame_params* out_params) {
|
||||
if (payload == nullptr || out_params == nullptr || payload_size == 0) {
|
||||
return arg_fail("joc_parse_id14 requires a non-empty payload and an output struct");
|
||||
}
|
||||
return finish(joc::joc::parse_id14(payload, payload_size, out_params, nullptr));
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_parse_eac3_frame(const std::uint8_t* frame, std::size_t frame_size,
|
||||
joc_frame_params* out_params, joc_emdf_info* out_emdf) {
|
||||
if (frame == nullptr || out_params == nullptr || frame_size == 0) {
|
||||
return arg_fail("joc_parse_eac3_frame requires a frame and an output struct");
|
||||
}
|
||||
joc::emdf::Container container;
|
||||
const joc::Status status =
|
||||
joc::joc::parse_eac3_frame(frame, frame_size, out_params, &container, nullptr);
|
||||
if (!status.ok()) {
|
||||
return finish(status);
|
||||
}
|
||||
if (out_emdf != nullptr) {
|
||||
fill_emdf_info(container, out_emdf);
|
||||
}
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_extract_payload(const std::uint8_t* frame, std::size_t frame_size,
|
||||
const joc_emdf_payload_info* payload, std::uint8_t* out,
|
||||
std::size_t out_capacity, std::size_t* out_size) {
|
||||
if (frame == nullptr || payload == nullptr || out_size == nullptr) {
|
||||
return arg_fail("joc_extract_payload requires frame, payload and out_size");
|
||||
}
|
||||
*out_size = payload->size;
|
||||
if (out == nullptr) {
|
||||
return JOC_OK;
|
||||
}
|
||||
if (out_capacity < payload->size) {
|
||||
return arg_fail("joc_extract_payload output buffer too small");
|
||||
}
|
||||
joc::emdf::Payload entry;
|
||||
entry.id = payload->id;
|
||||
entry.sample_offset = payload->sample_offset;
|
||||
entry.bit_offset = payload->bit_offset;
|
||||
entry.size = payload->size;
|
||||
std::vector<std::uint8_t> bytes;
|
||||
const joc::Status status = joc::emdf::extract_payload_bytes(frame, frame_size, entry, &bytes);
|
||||
if (!status.ok()) {
|
||||
return finish(status);
|
||||
}
|
||||
if (!bytes.empty()) {
|
||||
std::memcpy(out, bytes.data(), bytes.size());
|
||||
}
|
||||
*out_size = bytes.size();
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_eac3_frame_bytes(const std::uint8_t* data, std::size_t size, std::size_t offset,
|
||||
std::size_t* out_frame_bytes) {
|
||||
if (data == nullptr || out_frame_bytes == nullptr) {
|
||||
return arg_fail("joc_eac3_frame_bytes requires data and out_frame_bytes");
|
||||
}
|
||||
const joc_error code = joc::eac3::FrameReader::frame_bytes(data, size, offset, out_frame_bytes);
|
||||
if (code != JOC_OK) {
|
||||
return finish(joc::Status::fail(code, joc::stage::kEac3,
|
||||
"invalid or truncated E-AC-3 syncframe at byte " +
|
||||
std::to_string(offset)));
|
||||
}
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_check_id14_padding(const std::uint8_t* payload, std::size_t payload_size,
|
||||
std::uint32_t* out_trailing_bits) {
|
||||
if (payload == nullptr || payload_size == 0) {
|
||||
return arg_fail("joc_check_id14_padding requires a payload");
|
||||
}
|
||||
return finish(joc::joc::check_id14_padding(payload, payload_size, out_trailing_bits));
|
||||
}
|
||||
|
||||
} // extern "C"
|
||||
@@ -1,162 +0,0 @@
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "joc_core.h"
|
||||
#include "joc_stream.h"
|
||||
#include "stream/stream.h"
|
||||
|
||||
struct joc_stream {
|
||||
joc::stream::Stream instance;
|
||||
};
|
||||
|
||||
namespace {
|
||||
|
||||
joc_error finish_stream(const joc::Status& status) {
|
||||
return status.code();
|
||||
}
|
||||
|
||||
joc::stream::Config to_config(const joc_stream_config& config) {
|
||||
joc::stream::Config out;
|
||||
out.input = config.input;
|
||||
out.output = config.output;
|
||||
if (config.speaker_layout_name != nullptr) { out.layout = config.speaker_layout_name; }
|
||||
if (config.speaker_metadata_offset != 0u) {
|
||||
out.metadata_offset = config.speaker_metadata_offset;
|
||||
}
|
||||
out.binaural_mode = config.binaural_mode != 0u ? config.binaural_mode : JOC_BINAURAL_MID;
|
||||
if (config.hrtf_path != nullptr) { out.hrtf_path = config.hrtf_path; }
|
||||
if (config.kernels_path != nullptr) { out.kernels_path = config.kernels_path; }
|
||||
if (config.binaural_tail_seconds > 0.0) { out.tail_seconds = config.binaural_tail_seconds; }
|
||||
if (config.object_delay_samples != 0u) {
|
||||
out.object_delay_samples = config.object_delay_samples;
|
||||
}
|
||||
out.gain_db = config.gain_db;
|
||||
out.native_threads = config.native_threads;
|
||||
// The binaural HRTF inputs of joc_task_config, copied with the same defaults:
|
||||
// the policy is taken verbatim (0 is "none", a real choice, not "unset") and
|
||||
// the radius keeps its documented default of 1.0 when the field is not set.
|
||||
if (config.hrtf_sofa_path != nullptr) { out.hrtf_sofa_path = config.hrtf_sofa_path; }
|
||||
if (config.personalized_headphone_path != nullptr) {
|
||||
out.personalized_headphone_path = config.personalized_headphone_path;
|
||||
}
|
||||
if (config.hrtf_cache_dir != nullptr) { out.hrtf_cache_dir = config.hrtf_cache_dir; }
|
||||
out.hrtf_cache_policy = config.hrtf_cache_policy;
|
||||
if (config.hrtf_radius_m > 0.0) { out.hrtf_radius_m = config.hrtf_radius_m; }
|
||||
return out;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
extern "C" {
|
||||
|
||||
joc_error JOC_CALL joc_stream_create(const joc_stream_config* config, joc_stream** out) {
|
||||
if (config == nullptr || out == nullptr || config->struct_size != sizeof(joc_stream_config)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
auto* stream = new (std::nothrow) joc_stream();
|
||||
if (stream == nullptr) {
|
||||
return JOC_ERR_OUT_OF_MEMORY;
|
||||
}
|
||||
const joc::Status status = stream->instance.create(to_config(*config));
|
||||
if (!status.ok()) {
|
||||
delete stream;
|
||||
return finish_stream(status);
|
||||
}
|
||||
*out = stream;
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_push(joc_stream* stream, const joc_stream_buffer* input,
|
||||
std::uint32_t* consumed_samples, std::uint32_t* consumed_bytes) {
|
||||
if (stream == nullptr || input == nullptr ||
|
||||
input->struct_size != sizeof(joc_stream_buffer)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
const joc::Status status = [&] {
|
||||
if (input->kind == JOC_STREAM_IN_EAC3) {
|
||||
std::size_t consumed = 0;
|
||||
const joc::Status pushed =
|
||||
stream->instance.push_eac3(input->bytes, input->byte_count, &consumed);
|
||||
if (consumed_bytes != nullptr) {
|
||||
*consumed_bytes = static_cast<std::uint32_t>(consumed);
|
||||
}
|
||||
return pushed;
|
||||
}
|
||||
if (input->kind == JOC_STREAM_IN_PCM_OBJECTS16) {
|
||||
std::size_t consumed = 0;
|
||||
const joc::Status pushed =
|
||||
stream->instance.push_objects16(input->pcm, input->sample_count, &consumed);
|
||||
if (consumed_samples != nullptr) {
|
||||
*consumed_samples = static_cast<std::uint32_t>(consumed);
|
||||
}
|
||||
return pushed;
|
||||
}
|
||||
std::size_t consumed = 0;
|
||||
const joc::Status pushed =
|
||||
stream->instance.push_bed(input->pcm, input->sample_count, &consumed);
|
||||
if (consumed_samples != nullptr) {
|
||||
*consumed_samples = static_cast<std::uint32_t>(consumed);
|
||||
}
|
||||
return pushed; }();
|
||||
return finish_stream(status);
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_pull(joc_stream* stream, joc_stream_buffer* output,
|
||||
std::uint32_t* produced_samples) {
|
||||
if (stream == nullptr || output == nullptr || output->out_pcm == nullptr ||
|
||||
output->struct_size != sizeof(joc_stream_buffer)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
std::size_t produced = 0;
|
||||
const joc::Status status =
|
||||
stream->instance.pull(output->out_pcm, output->sample_count, &produced);
|
||||
if (!status.ok()) {
|
||||
return finish_stream(status);
|
||||
}
|
||||
if (produced_samples != nullptr) {
|
||||
*produced_samples = static_cast<std::uint32_t>(produced);
|
||||
}
|
||||
output->sample_count = static_cast<std::uint32_t>(produced);
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_flush(joc_stream* stream) {
|
||||
if (stream == nullptr) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
return finish_stream(stream->instance.flush());
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_reset(joc_stream* stream) {
|
||||
if (stream == nullptr) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
return finish_stream(stream->instance.reset());
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_status(const joc_stream* stream, joc_stream_status_info* out) {
|
||||
if (stream == nullptr || out == nullptr ||
|
||||
out->struct_size != sizeof(joc_stream_status_info)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
const joc::stream::Info& info = stream->instance.info();
|
||||
out->frames_in = info.frames_in;
|
||||
out->frames_out = info.frames_out;
|
||||
out->samples_in = info.samples_in;
|
||||
out->samples_out = info.samples_out;
|
||||
out->bytes_in = info.bytes_in;
|
||||
out->buffered_samples = stream->instance.buffered_samples();
|
||||
out->oamd_payloads = info.oamd_payloads;
|
||||
out->oamd_transitions = info.oamd_transitions;
|
||||
out->output_channels = info.output_channels;
|
||||
out->ended = info.ended;
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_destroy(joc_stream* stream) {
|
||||
delete stream;
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
} // extern "C"
|
||||
@@ -1,120 +0,0 @@
|
||||
|
||||
#include <atomic>
|
||||
#include <cstring>
|
||||
#include <new>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "joc_core.h"
|
||||
#include "task/task.h"
|
||||
|
||||
// task and whichever frontend wants to stop it (plan 29.1/29.2).
|
||||
struct joc_cancel_token {
|
||||
std::atomic<std::uint32_t> requested{0u};
|
||||
};
|
||||
|
||||
namespace {
|
||||
|
||||
thread_local std::string g_task_detail;
|
||||
|
||||
joc_error finish_task(const joc::Status& status) {
|
||||
if (status.ok()) {
|
||||
g_task_detail.clear();
|
||||
return JOC_OK;
|
||||
}
|
||||
g_task_detail.assign(status.stage());
|
||||
g_task_detail.append(": ");
|
||||
g_task_detail.append(status.message());
|
||||
return status.code();
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
extern "C" {
|
||||
|
||||
joc_cancel_token* JOC_CALL joc_cancel_token_create(void) {
|
||||
return new (std::nothrow) joc_cancel_token();
|
||||
}
|
||||
|
||||
void JOC_CALL joc_cancel_token_request(joc_cancel_token* token) {
|
||||
if (token != nullptr) {
|
||||
token->requested.store(1u, std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
std::int32_t JOC_CALL joc_cancel_token_is_requested(const joc_cancel_token* token) {
|
||||
return (token != nullptr && token->requested.load(std::memory_order_relaxed) != 0u) ? 1 : 0;
|
||||
}
|
||||
|
||||
void JOC_CALL joc_cancel_token_destroy(joc_cancel_token* token) { delete token; }
|
||||
|
||||
joc_error JOC_CALL joc_task_validate(const joc_task_config* config,
|
||||
joc_validation_issue* issues, std::uint32_t capacity,
|
||||
std::uint32_t* count) {
|
||||
if (config == nullptr) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
if (config->struct_size != sizeof(joc_task_config)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
std::vector<joc_validation_issue> found;
|
||||
std::uint32_t errors = 0;
|
||||
const joc::Status status = joc::task::validate(*config, &found, &errors);
|
||||
if (count != nullptr) {
|
||||
*count = static_cast<std::uint32_t>(found.size());
|
||||
}
|
||||
if (issues != nullptr) {
|
||||
for (std::uint32_t i = 0; i < capacity && i < found.size(); ++i) {
|
||||
issues[i] = found[i];
|
||||
}
|
||||
}
|
||||
return status.ok() ? JOC_OK : JOC_ERR_INVALID_CONFIG;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_task_execute(const joc_task_config* config, const joc_event_sink* sink,
|
||||
joc_task_result* out) {
|
||||
if (config == nullptr || config->struct_size != sizeof(joc_task_config)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
if (out != nullptr) {
|
||||
std::memset(out, 0, sizeof(*out));
|
||||
out->struct_size = sizeof(joc_task_result);
|
||||
out->struct_version = JOC_TASK_RESULT_VERSION;
|
||||
}
|
||||
const joc::Status status = joc::task::run(*config, sink, out);
|
||||
if (!status.ok() && out != nullptr && out->status == 0u) {
|
||||
out->status = JOC_TASK_FAILED;
|
||||
out->error_code = static_cast<std::uint32_t>(status.code());
|
||||
std::snprintf(out->error_stage, sizeof(out->error_stage), "%s", status.stage().c_str());
|
||||
std::snprintf(out->error_message, sizeof(out->error_message), "%s",
|
||||
status.message().c_str());
|
||||
}
|
||||
g_task_detail = status.ok() ? std::string() : status.message();
|
||||
return status.code();
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_task_result_to_json(const joc_task_result* result, char* buffer,
|
||||
std::size_t capacity, std::size_t* needed) {
|
||||
if (result == nullptr) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
std::string json;
|
||||
const joc::Status status = joc::task::result_to_json(*result, &json);
|
||||
if (!status.ok()) {
|
||||
return status.code();
|
||||
}
|
||||
if (needed != nullptr) {
|
||||
*needed = json.size() + 1u;
|
||||
}
|
||||
if (buffer == nullptr) {
|
||||
return JOC_OK;
|
||||
}
|
||||
if (capacity < json.size() + 1u) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
std::memcpy(buffer, json.c_str(), json.size() + 1u);
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
} // extern "C"
|
||||
@@ -1,251 +0,0 @@
|
||||
#include "binaural/binaural_runtime.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
namespace joc::binaural {
|
||||
|
||||
namespace {
|
||||
|
||||
double max_delay_bound(const hrtf::Field& field) {
|
||||
double maximum = 0.0;
|
||||
for (const double value : field.delay_bounds) {
|
||||
maximum = std::max(maximum, value);
|
||||
}
|
||||
return maximum;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool profile_from_name(const char* name, Profile* out) {
|
||||
if (name == nullptr || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (std::strcmp(name, "near") == 0) { *out = Profile::Near; return true; }
|
||||
if (std::strcmp(name, "mid") == 0) { *out = Profile::Mid; return true; }
|
||||
if (std::strcmp(name, "far") == 0) { *out = Profile::Far; return true; }
|
||||
return false;
|
||||
}
|
||||
|
||||
SofaBinauralRuntime::~SofaBinauralRuntime() {
|
||||
if (handle_ != nullptr) {
|
||||
ejoc_sofa_binaural_destroy(handle_);
|
||||
handle_ = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
Status SofaBinauralRuntime::open(const hrtf::Field& field, const hrtf::Kernels& kernels,
|
||||
Profile profile, const RoomConstants& room) {
|
||||
if (handle_ != nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime already open");
|
||||
}
|
||||
if (field.coefficients.size() != static_cast<std::size_t>(hrtf::kShTerms * hrtf::kEars *
|
||||
hrtf::kHybridBands * 2) ||
|
||||
field.band_centers_hz.size() != static_cast<std::size_t>(hrtf::kHybridBands)) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"compiled HRTF field has unexpected array sizes");
|
||||
}
|
||||
handle_ = ejoc_sofa_binaural_create();
|
||||
if (handle_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_OUT_OF_MEMORY, stage::kRender,
|
||||
"ejoc_sofa_binaural_create failed");
|
||||
}
|
||||
profile_ = profile;
|
||||
|
||||
if (ejoc_sofa_binaural_configure_kernels(
|
||||
handle_, kernels.qmf_analysis.data(), kernels.hybrid_low.data(),
|
||||
kernels.hybrid_indices.data(), kernels.hybrid_values.data(), kernels.hybrid_count,
|
||||
kernels.qmf_basis.data(), kernels.qmf_taps.data()) != 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
std::string("configure_kernels failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
if (ejoc_sofa_binaural_configure_field(handle_, field.coefficients.data(),
|
||||
field.delay_coefficients.data(),
|
||||
field.delay_bounds.data(), field.band_centers_hz.data(),
|
||||
field.measurement_radius_m) != 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
std::string("configure_field failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
if (ejoc_sofa_binaural_configure_room(
|
||||
handle_, room.dims, room.listener, room.walls, room.speed_of_sound, room.fdn_delays,
|
||||
room.fdn_feedback, room.damping, room.fdn_output_gain, room.allpass_delays,
|
||||
room.allpass_gains, room.enable_early_reflections, room.enable_late_room) != 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
|
||||
std::string("configure_room failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
|
||||
maximum_hrtf_delay_ =
|
||||
static_cast<std::int64_t>(std::ceil(max_delay_bound(field) - 1e-9));
|
||||
hrtf_history_slots_ = static_cast<std::uint32_t>(std::max<std::int64_t>(1, (maximum_hrtf_delay_ + 63) / 64));
|
||||
staging_.assign(kBlockSamples * kSourceCount, 0.0);
|
||||
block_output_.assign(kBlockSamples * 2u, 0.0);
|
||||
output_.clear();
|
||||
staged_ = 0;
|
||||
input_samples_ = 0;
|
||||
processed_samples_ = 0;
|
||||
blocks_processed_ = 0;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status SofaBinauralRuntime::process_block() {
|
||||
double positions[timeline::kTimelineObjects][3] = {};
|
||||
const Status queried = timeline_.positions_at(static_cast<std::int64_t>(processed_samples_),
|
||||
positions);
|
||||
if (!queried.ok()) {
|
||||
return queried;
|
||||
}
|
||||
// The reference adapter calls set_source without a `fade` argument, so the
|
||||
// backend default (fade enabled) applies - the per-object path crossfade is
|
||||
// part of the reference behaviour, not an optional extra.
|
||||
constexpr std::uint32_t kFade = 1u;
|
||||
if (ejoc_sofa_binaural_set_source(handle_, 0u, kLfePosition,
|
||||
static_cast<std::uint32_t>(profile_), 1.0, 1u, 1u,
|
||||
kFade) != 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
|
||||
std::string("set_source(LFE) failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
for (std::uint32_t source = 0; source < timeline::kTimelineObjects; ++source) {
|
||||
if (ejoc_sofa_binaural_set_source(handle_, source + 1u, positions[source],
|
||||
static_cast<std::uint32_t>(profile_), 1.0, 1u, 0u,
|
||||
kFade) != 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
|
||||
std::string("set_source(object ") + std::to_string(source + 1u) +
|
||||
") failed: " + (message != nullptr ? message : "unknown"));
|
||||
}
|
||||
}
|
||||
const int trimmed = ejoc_sofa_binaural_process(handle_, staging_.data(), kBlockSamples, 1.0,
|
||||
block_output_.data());
|
||||
if (trimmed < 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
|
||||
std::string("sofa process failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
if (trimmed > 0) {
|
||||
output_.insert(output_.end(), block_output_.begin(),
|
||||
block_output_.begin() + static_cast<std::ptrdiff_t>(trimmed) * 2);
|
||||
}
|
||||
staged_ = 0;
|
||||
processed_samples_ += kBlockSamples;
|
||||
++blocks_processed_;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status SofaBinauralRuntime::submit_frame(const float* objects16_planar,
|
||||
const oamd::OamdUpdate* update, std::int64_t frame_index,
|
||||
std::int64_t outer_sample_offset,
|
||||
std::int64_t object_delay_samples) {
|
||||
if (handle_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime is not open");
|
||||
}
|
||||
if (objects16_planar == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null frame");
|
||||
}
|
||||
// A frame without an ID11 payload submits nothing at all (the reference only
|
||||
if (update != nullptr) {
|
||||
const Status submitted = timeline_.submit_update(
|
||||
*update, frame_index * JOC_FRAME_SAMPLES, outer_sample_offset, object_delay_samples,
|
||||
static_cast<std::int64_t>(input_samples_));
|
||||
if (!submitted.ok()) {
|
||||
return submitted;
|
||||
}
|
||||
}
|
||||
|
||||
std::size_t offset = 0;
|
||||
while (offset < JOC_FRAME_SAMPLES) {
|
||||
const std::size_t room = kBlockSamples - staged_;
|
||||
const std::size_t count = std::min<std::size_t>(room, JOC_FRAME_SAMPLES - offset);
|
||||
for (std::size_t sample = 0; sample < count; ++sample) {
|
||||
double* row = staging_.data() + (staged_ + sample) * kSourceCount;
|
||||
for (std::size_t channel = 0; channel < kSourceCount; ++channel) {
|
||||
row[channel] = static_cast<double>(
|
||||
objects16_planar[channel * JOC_FRAME_SAMPLES + offset + sample]);
|
||||
}
|
||||
}
|
||||
staged_ += count;
|
||||
offset += count;
|
||||
if (staged_ == kBlockSamples) {
|
||||
const Status status = process_block();
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
}
|
||||
}
|
||||
input_samples_ += JOC_FRAME_SAMPLES;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
std::uint32_t SofaBinauralRuntime::finish_capacity(double tail_seconds) const {
|
||||
std::int64_t requested = tail_samples_;
|
||||
if (tail_seconds >= 0.0) {
|
||||
requested = static_cast<std::int64_t>(std::ceil(tail_seconds * 48000.0 - 1e-9));
|
||||
}
|
||||
const std::int64_t hrtf_bound = static_cast<std::int64_t>(hrtf_history_slots_) * 64;
|
||||
const std::int64_t early_bound = hrtf_bound + 2048 + 256 * 64;
|
||||
std::int64_t drain = std::max(requested, early_bound) + 961;
|
||||
drain = ((drain + 63) / 64) * 64;
|
||||
return static_cast<std::uint32_t>(drain);
|
||||
}
|
||||
|
||||
Status SofaBinauralRuntime::reset() {
|
||||
if (handle_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime is not open");
|
||||
}
|
||||
if (ejoc_sofa_binaural_reset(handle_) != 0) {
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender, "sofa reset failed");
|
||||
}
|
||||
timeline_ = timeline::OamdPositionTimeline();
|
||||
output_.clear();
|
||||
staged_ = 0;
|
||||
input_samples_ = 0;
|
||||
processed_samples_ = 0;
|
||||
blocks_processed_ = 0;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status SofaBinauralRuntime::finish(std::uint32_t flush_samples, std::vector<double>* out) {
|
||||
if (handle_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime is not open");
|
||||
}
|
||||
if (out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null output");
|
||||
}
|
||||
out->clear();
|
||||
if (flush_samples == 0u) {
|
||||
return Status::success();
|
||||
}
|
||||
std::vector<double> chunk(static_cast<std::size_t>(kBlockSamples) * 2u, 0.0);
|
||||
std::uint32_t produced_total = 0;
|
||||
std::uint32_t remaining = flush_samples;
|
||||
while (remaining > 0) {
|
||||
const std::uint32_t request = std::min<std::uint32_t>(remaining, kBlockSamples);
|
||||
const int produced = ejoc_sofa_binaural_finish(handle_, request, chunk.data(), request);
|
||||
if (produced < 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
|
||||
std::string("sofa finish failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
if (produced == 0) {
|
||||
break;
|
||||
}
|
||||
out->insert(out->end(), chunk.begin(),
|
||||
chunk.begin() + static_cast<std::ptrdiff_t>(produced) * 2);
|
||||
produced_total += static_cast<std::uint32_t>(produced);
|
||||
remaining -= std::min<std::uint32_t>(remaining, static_cast<std::uint32_t>(produced));
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::binaural
|
||||
@@ -1,96 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
#include "eac3joc_core.h"
|
||||
#include "foundation/status.h"
|
||||
#include "hrtf/jochrtf.h"
|
||||
#include "oamd/oamd_parser.h"
|
||||
#include "timeline/position_timeline.h"
|
||||
|
||||
namespace joc::binaural {
|
||||
|
||||
// ADM direction of the LFE source, as the reference passes it.
|
||||
inline constexpr double kLfePosition[3] = {0.0, 1.0, 0.0};
|
||||
inline constexpr int kSourceCount = 16;
|
||||
inline constexpr std::uint32_t kBlockSamples = 512;
|
||||
|
||||
// Room constants the reference's native bridge passes for the default shoebox.
|
||||
struct RoomConstants {
|
||||
double dims[3] = {18.0, 18.0, 14.0};
|
||||
double listener[3] = {9.0, 9.0, 7.0};
|
||||
double walls[6] = {0.62, 0.60, 0.58, 0.61, 0.52, 0.56};
|
||||
double speed_of_sound = 343.3;
|
||||
std::uint32_t fdn_delays[4] = {1427u, 1783u, 1973u, 2099u};
|
||||
double fdn_feedback[4] = {0.7853685923259284, 0.7394299865898056, 0.7160221718631921,
|
||||
0.7009092068085467};
|
||||
double damping = 0.32;
|
||||
double fdn_output_gain = 0.22;
|
||||
std::uint32_t allpass_delays[2] = {113u, 331u};
|
||||
double allpass_gains[2] = {0.63, 0.51};
|
||||
std::uint32_t enable_early_reflections = 1;
|
||||
std::uint32_t enable_late_room = 1;
|
||||
};
|
||||
|
||||
enum class Profile : std::uint32_t { Near = 0, Mid = 1, Far = 2 };
|
||||
|
||||
bool profile_from_name(const char* name, Profile* out);
|
||||
|
||||
class SofaBinauralRuntime {
|
||||
public:
|
||||
SofaBinauralRuntime() = default;
|
||||
~SofaBinauralRuntime();
|
||||
|
||||
SofaBinauralRuntime(const SofaBinauralRuntime&) = delete;
|
||||
SofaBinauralRuntime& operator=(const SofaBinauralRuntime&) = delete;
|
||||
|
||||
Status open(const hrtf::Field& field, const hrtf::Kernels& kernels, Profile profile,
|
||||
const RoomConstants& room = RoomConstants{});
|
||||
|
||||
Status submit_frame(const float* objects16_planar, const oamd::OamdUpdate* update,
|
||||
std::int64_t frame_index, std::int64_t outer_sample_offset,
|
||||
std::int64_t object_delay_samples);
|
||||
|
||||
// Resets the kernel, the timeline and the counters (plan 31.2).
|
||||
Status reset();
|
||||
|
||||
// Drains the room tail. `flush_samples` is the drain length; the reference
|
||||
Status finish(std::uint32_t flush_samples, std::vector<double>* out);
|
||||
|
||||
// Program output (input minus the 961-sample kernel latency), interleaved.
|
||||
const std::vector<double>& output() const { return output_; }
|
||||
|
||||
void take_output(std::vector<double>* out) {
|
||||
out->swap(output_);
|
||||
output_.clear();
|
||||
}
|
||||
|
||||
std::uint64_t input_samples() const { return input_samples_; }
|
||||
std::uint64_t blocks_processed() const { return blocks_processed_; }
|
||||
std::size_t staged_samples() const { return staged_; }
|
||||
const timeline::OamdPositionTimeline& timeline() const { return timeline_; }
|
||||
|
||||
// finish_output_capacity as the reference computes it (plan 21.4).
|
||||
std::uint32_t finish_capacity(double tail_seconds) const;
|
||||
|
||||
private:
|
||||
Status process_block();
|
||||
|
||||
ejoc_sofa_binaural_handle handle_ = nullptr;
|
||||
Profile profile_ = Profile::Mid;
|
||||
timeline::OamdPositionTimeline timeline_;
|
||||
std::vector<double> staging_;
|
||||
std::size_t staged_ = 0;
|
||||
std::vector<double> block_output_;
|
||||
std::vector<double> output_;
|
||||
std::uint64_t input_samples_ = 0;
|
||||
std::uint64_t processed_samples_ = 0;
|
||||
std::uint64_t blocks_processed_ = 0;
|
||||
std::int64_t maximum_hrtf_delay_ = 0;
|
||||
std::uint32_t hrtf_history_slots_ = 1;
|
||||
std::uint32_t tail_samples_ = 61200;
|
||||
};
|
||||
|
||||
} // namespace joc::binaural
|
||||
@@ -0,0 +1,184 @@
|
||||
"""Direct ID11/OAMD position scheduling for the binaural render path."""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
import numpy as np
|
||||
|
||||
from adm_atmos import q_to_adm_xyz
|
||||
from oamd_bits import JocFieldState, frame_update
|
||||
from oamd_tracks import align_metadata_sample
|
||||
from variant_error import UnsupportedVariantError
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PositionTransition:
|
||||
start_sample: int
|
||||
duration_samples: int
|
||||
origin: np.ndarray
|
||||
target: np.ndarray
|
||||
|
||||
@property
|
||||
def end_sample(self) -> int:
|
||||
return self.start_sample + self.duration_samples
|
||||
|
||||
|
||||
class _ObjectPositionTrack:
|
||||
def __init__(self):
|
||||
self.initial = np.zeros(3, dtype=np.float64)
|
||||
self.last_target = self.initial.copy()
|
||||
self.transitions: list[PositionTransition] = []
|
||||
self.cursor = 0
|
||||
self.last_query_sample = -1
|
||||
|
||||
def set_initial(self, position):
|
||||
target = np.asarray(position, dtype=np.float64)
|
||||
self.initial = target.copy()
|
||||
self.last_target = target.copy()
|
||||
|
||||
def append(self, start_sample: int, duration_samples: int, target,
|
||||
object_index: int):
|
||||
start = int(start_sample)
|
||||
duration = int(duration_samples)
|
||||
if start < 0 or duration < 0:
|
||||
raise ValueError("position transition timing must be non-negative")
|
||||
target = np.asarray(target, dtype=np.float64)
|
||||
if self.transitions:
|
||||
previous = self.transitions[-1]
|
||||
if start < previous.end_sample:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "overlapping_binaural_position_ramps",
|
||||
"同一对象的新位置更新在上一双耳 ramp 完成前到达",
|
||||
details={
|
||||
"object": object_index,
|
||||
"ramp_start_sample": previous.start_sample,
|
||||
"ramp_end_sample": previous.end_sample,
|
||||
"next_update_sample": start,
|
||||
})
|
||||
if start == previous.start_sample and previous.duration_samples == 0:
|
||||
self.transitions[-1] = PositionTransition(
|
||||
start, duration, previous.origin.copy(), target.copy())
|
||||
self.last_target = target.copy()
|
||||
return
|
||||
self.transitions.append(PositionTransition(
|
||||
start, duration, self.last_target.copy(), target.copy()))
|
||||
self.last_target = target.copy()
|
||||
|
||||
def position_at(self, sample: int) -> np.ndarray:
|
||||
sample = int(sample)
|
||||
if sample < self.last_query_sample:
|
||||
raise ValueError("binaural metadata positions must be queried monotonically")
|
||||
self.last_query_sample = sample
|
||||
while self.cursor < len(self.transitions):
|
||||
transition = self.transitions[self.cursor]
|
||||
if sample < transition.end_sample:
|
||||
break
|
||||
self.initial = transition.target.copy()
|
||||
self.cursor += 1
|
||||
if self.cursor >= len(self.transitions):
|
||||
return self.initial
|
||||
transition = self.transitions[self.cursor]
|
||||
if sample < transition.start_sample:
|
||||
return self.initial
|
||||
if transition.duration_samples == 0:
|
||||
return transition.target
|
||||
amount = (sample - transition.start_sample) / float(transition.duration_samples)
|
||||
return transition.origin + (transition.target - transition.origin) * amount
|
||||
|
||||
|
||||
class OamdPositionTimeline:
|
||||
"""Convert OAMD state updates into a sample-timed Cartesian trajectory."""
|
||||
|
||||
def __init__(self, object_count: int = 15):
|
||||
if object_count != 15:
|
||||
raise ValueError("JOC OAMD currently requires 15 object slots")
|
||||
self.object_count = int(object_count)
|
||||
self.state = JocFieldState()
|
||||
self.tracks = [_ObjectPositionTrack() for _ in range(self.object_count)]
|
||||
self.initialized = False
|
||||
self.previous_targets: list[tuple[float, float, float] | None] = [
|
||||
None] * self.object_count
|
||||
self.payload_count = 0
|
||||
self.transition_count = 0
|
||||
self.last_coded_event_sample = -1
|
||||
|
||||
def _targets(self) -> list[tuple[float, float, float]]:
|
||||
q = self.state.q
|
||||
return [
|
||||
q_to_adm_xyz(
|
||||
q[(object_index, "q1")],
|
||||
q[(object_index, "q2")],
|
||||
q[(object_index, "q3")],
|
||||
)
|
||||
for object_index in range(1, self.object_count + 1)
|
||||
]
|
||||
|
||||
def submit_update(self, update: dict, *, frame_start_sample: int,
|
||||
outer_sample_offset: int = 0,
|
||||
object_delay_samples: int = 1473,
|
||||
processed_sample: int = 0):
|
||||
"""Schedule one already-parsed :func:`oamd_bits.frame_update` result."""
|
||||
frame_start = int(frame_start_sample)
|
||||
outer_offset = int(outer_sample_offset)
|
||||
object_delay = int(object_delay_samples)
|
||||
if min(frame_start, outer_offset, object_delay) < 0:
|
||||
raise ValueError("OAMD frame, outer offset, and object delay must be non-negative")
|
||||
self.state.apply(update["values"])
|
||||
targets = self._targets()
|
||||
coded_event = (
|
||||
frame_start + outer_offset + int(update["block_offset_samples"]))
|
||||
if coded_event < self.last_coded_event_sample:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "non_monotonic_binaural_updates",
|
||||
"双耳 OAMD 更新时间倒退",
|
||||
details={
|
||||
"event_sample": coded_event,
|
||||
"previous_event_sample": self.last_coded_event_sample,
|
||||
})
|
||||
self.last_coded_event_sample = coded_event
|
||||
|
||||
if not self.initialized:
|
||||
if int(processed_sample) > 0:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "late_initial_binaural_state",
|
||||
"首个 OAMD 状态在双耳 PCM 已处理后才出现,无法回填 sample 0",
|
||||
details={
|
||||
"processed_sample": int(processed_sample),
|
||||
"first_event_sample": coded_event,
|
||||
})
|
||||
for index, target in enumerate(targets):
|
||||
self.tracks[index].set_initial(target)
|
||||
self.previous_targets[index] = target
|
||||
self.initialized = True
|
||||
self.payload_count += 1
|
||||
return
|
||||
|
||||
effective_ramp = max(0, int(update["ramp_duration_samples"]))
|
||||
transition_start = align_metadata_sample(coded_event + object_delay)
|
||||
for index, target in enumerate(targets):
|
||||
if self.previous_targets[index] == target:
|
||||
continue
|
||||
self.tracks[index].append(
|
||||
transition_start, effective_ramp, target, index + 1)
|
||||
self.previous_targets[index] = target
|
||||
self.transition_count += 1
|
||||
self.payload_count += 1
|
||||
|
||||
def submit_payload(self, payload, *, frame_start_sample: int,
|
||||
outer_sample_offset: int = 0,
|
||||
object_delay_samples: int = 1473,
|
||||
processed_sample: int = 0):
|
||||
update = frame_update(payload)
|
||||
self.submit_update(
|
||||
update,
|
||||
frame_start_sample=frame_start_sample,
|
||||
outer_sample_offset=outer_sample_offset,
|
||||
object_delay_samples=object_delay_samples,
|
||||
processed_sample=processed_sample,
|
||||
)
|
||||
return update
|
||||
|
||||
def positions_at(self, sample: int) -> np.ndarray:
|
||||
return np.stack(
|
||||
[track.position_at(sample) for track in self.tracks], axis=0
|
||||
).astype(np.float64, copy=False)
|
||||
@@ -0,0 +1,214 @@
|
||||
"""ctypes bridge for the native float64 binaural DSP."""
|
||||
from __future__ import annotations
|
||||
|
||||
import ctypes
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from native_renderer import ABI_VERSION, find_native_library
|
||||
from rosella_filterbank import DEFAULT_KERNEL_DATA, load_kernel_tables
|
||||
from rosella_model import RosellaModel
|
||||
|
||||
BLOCK_SAMPLES = 512
|
||||
INPUT_CHANNELS = 16
|
||||
OUTPUT_CHANNELS = 2
|
||||
HYBRID_BANDS = 77
|
||||
|
||||
|
||||
class NativeBinauralDsp:
|
||||
def __init__(self, model: RosellaModel, *, library_path=None,
|
||||
kernel_data: str | Path = DEFAULT_KERNEL_DATA):
|
||||
self.library_path = find_native_library(library_path)
|
||||
self._lib = ctypes.CDLL(str(self.library_path))
|
||||
self._bind()
|
||||
version = int(self._lib.ejoc_abi_version())
|
||||
if version != ABI_VERSION:
|
||||
raise RuntimeError(
|
||||
f"native ABI mismatch: expected {ABI_VERSION}, got {version}")
|
||||
self._handle = self._lib.ejoc_binaural_renderer_create()
|
||||
if not self._handle:
|
||||
raise RuntimeError("native binaural renderer creation failed")
|
||||
try:
|
||||
self._configure_kernels(kernel_data)
|
||||
self._configure_room(model)
|
||||
except Exception:
|
||||
self.close()
|
||||
raise
|
||||
|
||||
def _bind(self):
|
||||
void_p = ctypes.c_void_p
|
||||
f64_p = ctypes.POINTER(ctypes.c_double)
|
||||
i16_p = ctypes.POINTER(ctypes.c_int16)
|
||||
u32_p = ctypes.POINTER(ctypes.c_uint32)
|
||||
self._lib.ejoc_abi_version.argtypes = []
|
||||
self._lib.ejoc_abi_version.restype = ctypes.c_uint32
|
||||
self._lib.ejoc_binaural_renderer_create.argtypes = []
|
||||
self._lib.ejoc_binaural_renderer_create.restype = void_p
|
||||
self._lib.ejoc_binaural_renderer_destroy.argtypes = [void_p]
|
||||
self._lib.ejoc_binaural_renderer_destroy.restype = None
|
||||
self._lib.ejoc_binaural_renderer_reset.argtypes = [void_p]
|
||||
self._lib.ejoc_binaural_renderer_reset.restype = ctypes.c_int
|
||||
self._lib.ejoc_binaural_renderer_last_error.argtypes = [void_p]
|
||||
self._lib.ejoc_binaural_renderer_last_error.restype = ctypes.c_char_p
|
||||
self._lib.ejoc_binaural_renderer_configure_kernels.argtypes = [
|
||||
void_p, f64_p, f64_p, i16_p, f64_p, ctypes.c_uint32, f64_p, f64_p]
|
||||
self._lib.ejoc_binaural_renderer_configure_kernels.restype = ctypes.c_int
|
||||
self._lib.ejoc_binaural_renderer_configure_room.argtypes = [
|
||||
void_p, ctypes.c_uint32, ctypes.c_uint32, u32_p, f64_p,
|
||||
u32_p, f64_p, ctypes.c_uint32, f64_p, f64_p, f64_p,
|
||||
ctypes.c_uint32, u32_p, f64_p, f64_p]
|
||||
self._lib.ejoc_binaural_renderer_configure_room.restype = ctypes.c_int
|
||||
self._lib.ejoc_binaural_renderer_process.argtypes = [
|
||||
void_p, f64_p, f64_p, f64_p, ctypes.c_double, f64_p]
|
||||
self._lib.ejoc_binaural_renderer_process.restype = ctypes.c_int
|
||||
|
||||
def _raise(self, operation, status):
|
||||
message = self._lib.ejoc_binaural_renderer_last_error(self._handle)
|
||||
detail = (message or b"").decode("utf-8", "replace")
|
||||
raise RuntimeError(
|
||||
f"native binaural renderer {operation} failed ({status}): {detail}")
|
||||
|
||||
@staticmethod
|
||||
def _f64_pointer(values):
|
||||
return values.ctypes.data_as(ctypes.POINTER(ctypes.c_double))
|
||||
|
||||
def _configure_kernels(self, kernel_data):
|
||||
tables = load_kernel_tables(kernel_data)
|
||||
qmf_analysis = np.ascontiguousarray(
|
||||
tables["qmf_analysis_coefficients"], dtype=np.float64)
|
||||
hybrid_low = np.ascontiguousarray(
|
||||
tables["hybrid_analysis_low_kernel"], dtype=np.float64)
|
||||
hybrid_indices = np.ascontiguousarray(
|
||||
tables["hybrid_synthesis_indices"], dtype=np.int16)
|
||||
hybrid_values = np.ascontiguousarray(
|
||||
tables["hybrid_synthesis_values"], dtype=np.float64)
|
||||
qmf_basis = np.ascontiguousarray(
|
||||
tables["qmf_synthesis_basis"], dtype=np.float64)
|
||||
qmf_taps = np.ascontiguousarray(
|
||||
tables["qmf_synthesis_taps"], dtype=np.float64)
|
||||
status = self._lib.ejoc_binaural_renderer_configure_kernels(
|
||||
self._handle,
|
||||
self._f64_pointer(qmf_analysis),
|
||||
self._f64_pointer(hybrid_low),
|
||||
hybrid_indices.ctypes.data_as(ctypes.POINTER(ctypes.c_int16)),
|
||||
self._f64_pointer(hybrid_values),
|
||||
len(hybrid_values),
|
||||
self._f64_pointer(qmf_basis),
|
||||
self._f64_pointer(qmf_taps),
|
||||
)
|
||||
if status:
|
||||
self._raise("configure_kernels", status)
|
||||
|
||||
def _configure_room(self, model: RosellaModel):
|
||||
if float(model.table_a_scalar) >= 0.5:
|
||||
raise NotImplementedError("alternate table-A room mode")
|
||||
bands = min(64, model.table_a_dimension)
|
||||
allpass_delays = np.ascontiguousarray(
|
||||
model.table_a_option_ids, dtype=np.uint32)
|
||||
allpass_gains = np.ascontiguousarray(
|
||||
model.table_a_option_values, dtype=np.float64)
|
||||
fdn_delays = np.ascontiguousarray(
|
||||
model.table_a_four_integers, dtype=np.uint32)
|
||||
fdn_matrix = np.ascontiguousarray(
|
||||
np.asarray(model.table_a_vector16, dtype=np.float64).reshape(
|
||||
4, 4, order="F"))
|
||||
|
||||
filter8 = np.asarray(
|
||||
model.table_a_filter_8x64_padded, dtype=np.float64).reshape(20, 4, 2, 4)
|
||||
filter4 = np.asarray(
|
||||
model.table_a_filter_4x64_padded, dtype=np.float64).reshape(20, 4, 4)
|
||||
filter16 = np.asarray(
|
||||
model.table_a_filter_16x64_padded, dtype=np.float64).reshape(20, 4, 4, 4)
|
||||
feedback = np.empty((64, 4, 2), dtype=np.float64)
|
||||
output_taps = np.empty((64, 4), dtype=np.float64)
|
||||
output_matrix = np.empty((2, 64, 4, 2), dtype=np.float64)
|
||||
for band in range(64):
|
||||
group, lane = divmod(band, 4)
|
||||
feedback[band, :, 0] = filter8[group, :, 0, lane]
|
||||
feedback[band, :, 1] = filter8[group, :, 1, lane]
|
||||
output_taps[band] = filter4[group, :, lane]
|
||||
output_matrix[0, band, :, 0] = filter16[group, :, 0, lane]
|
||||
output_matrix[0, band, :, 1] = filter16[group, :, 1, lane]
|
||||
output_matrix[1, band, :, 0] = filter16[group, :, 2, lane]
|
||||
output_matrix[1, band, :, 1] = filter16[group, :, 3, lane]
|
||||
|
||||
extra_count = int(model.table_a_extra)
|
||||
extra_delays = np.ascontiguousarray(
|
||||
model.table_a_extra_indices, dtype=np.uint32)
|
||||
extra_fields = np.empty((extra_count, 64, 2), dtype=np.float64)
|
||||
extra_source = np.asarray(
|
||||
model.table_a_extra_fields_padded, dtype=np.float64).reshape(
|
||||
extra_count, 20, 2, 4)
|
||||
for extra in range(extra_count):
|
||||
for band in range(64):
|
||||
group, lane = divmod(band, 4)
|
||||
extra_fields[extra, band] = extra_source[extra, group, :, lane]
|
||||
extra_matrices = np.empty((extra_count, 4, 4), dtype=np.float64)
|
||||
for extra in range(extra_count):
|
||||
extra_matrices[extra] = np.asarray(
|
||||
model.table_a_extra_vectors[extra], dtype=np.float64).reshape(
|
||||
4, 4, order="F")
|
||||
|
||||
null_u32 = ctypes.POINTER(ctypes.c_uint32)()
|
||||
null_f64 = ctypes.POINTER(ctypes.c_double)()
|
||||
status = self._lib.ejoc_binaural_renderer_configure_room(
|
||||
self._handle,
|
||||
bands,
|
||||
len(allpass_delays),
|
||||
allpass_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)),
|
||||
self._f64_pointer(allpass_gains),
|
||||
fdn_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)),
|
||||
self._f64_pointer(fdn_matrix),
|
||||
int(model.table_a_integer),
|
||||
self._f64_pointer(feedback),
|
||||
self._f64_pointer(output_taps),
|
||||
self._f64_pointer(output_matrix),
|
||||
extra_count,
|
||||
(extra_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32))
|
||||
if extra_count else null_u32),
|
||||
self._f64_pointer(extra_fields) if extra_count else null_f64,
|
||||
self._f64_pointer(extra_matrices) if extra_count else null_f64,
|
||||
)
|
||||
if status:
|
||||
self._raise("configure_room", status)
|
||||
|
||||
def reset(self):
|
||||
if not self._handle:
|
||||
raise RuntimeError("native binaural renderer is closed")
|
||||
status = self._lib.ejoc_binaural_renderer_reset(self._handle)
|
||||
if status:
|
||||
self._raise("reset", status)
|
||||
|
||||
def process_block(self, pcm16, gains, room_sends, output_gain=1.0):
|
||||
if not self._handle:
|
||||
raise RuntimeError("native binaural renderer is closed")
|
||||
source = np.ascontiguousarray(pcm16, dtype=np.float64)
|
||||
gain_values = np.asarray(gains)
|
||||
sends = np.ascontiguousarray(room_sends, dtype=np.float64)
|
||||
if source.shape != (BLOCK_SAMPLES, INPUT_CHANNELS):
|
||||
raise ValueError(f"pcm16 block must be (512,16), got {source.shape}")
|
||||
if gain_values.shape != (INPUT_CHANNELS, OUTPUT_CHANNELS, HYBRID_BANDS):
|
||||
raise ValueError(f"gains must be (16,2,77), got {gain_values.shape}")
|
||||
direct = np.ascontiguousarray(
|
||||
gain_values, dtype=np.complex128).view(np.float64)
|
||||
if sends.shape != (INPUT_CHANNELS,):
|
||||
raise ValueError(f"room_sends must be (16,), got {sends.shape}")
|
||||
output = np.empty((BLOCK_SAMPLES, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
status = self._lib.ejoc_binaural_renderer_process(
|
||||
self._handle,
|
||||
self._f64_pointer(source),
|
||||
self._f64_pointer(direct),
|
||||
self._f64_pointer(sends),
|
||||
float(output_gain),
|
||||
self._f64_pointer(output),
|
||||
)
|
||||
if status:
|
||||
self._raise("process", status)
|
||||
return output
|
||||
|
||||
def close(self):
|
||||
handle = getattr(self, "_handle", None)
|
||||
if handle:
|
||||
self._lib.ejoc_binaural_renderer_destroy(handle)
|
||||
self._handle = None
|
||||
@@ -0,0 +1,272 @@
|
||||
"""JOC frame adapter for the public SOFA binaural backend."""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from binaural_metadata import OamdPositionTimeline
|
||||
from public_filterbank import ANALYSIS_SYNTHESIS_LATENCY_SAMPLES
|
||||
from sofa_binaural_backend import SofaBinauralBackend
|
||||
from sofa_hrtf_field import (
|
||||
DEFAULT_HRTF_CACHE_DIR,
|
||||
DEFAULT_PROJECTION_RIDGE,
|
||||
DEFAULT_SH_RIDGE,
|
||||
)
|
||||
|
||||
|
||||
SAMPLE_RATE = 48000
|
||||
FRAME_SAMPLES = 1536
|
||||
BINAURAL_BLOCK_SAMPLES = 512
|
||||
QMF_HOP_SAMPLES = 64
|
||||
BINAURAL_LATENCY_SAMPLES = ANALYSIS_SYNTHESIS_LATENCY_SAMPLES
|
||||
SOURCE_CHANNELS = 16
|
||||
OUTPUT_CHANNELS = 2
|
||||
PROJECT_DIR = Path(__file__).resolve().parent.parent
|
||||
DEFAULT_HRTF_DIR = PROJECT_DIR / "HRTF"
|
||||
DEFAULT_SOFA_HRTF = DEFAULT_HRTF_DIR / "binaural.sofa"
|
||||
|
||||
|
||||
def _resolve_hrtf_file(path: str | Path, suffix: str, label: str) -> Path:
|
||||
target = Path(path).expanduser().resolve()
|
||||
if target.suffix.lower() != suffix:
|
||||
raise ValueError(f"{label} must use the {suffix} extension: {target}")
|
||||
if not target.is_file():
|
||||
raise FileNotFoundError(f"{label} not found: {target}")
|
||||
return target
|
||||
|
||||
|
||||
def resolve_sofa_hrtf(path: str | Path) -> Path:
|
||||
"""Resolve an explicitly selected public SOFA source."""
|
||||
return _resolve_hrtf_file(path, ".sofa", "SOFA HRTF")
|
||||
|
||||
|
||||
def resolve_compiled_hrtf_cache(path: str | Path) -> Path:
|
||||
"""Resolve an explicitly selected JOC compiled HRTF cache."""
|
||||
return _resolve_hrtf_file(path, ".jochrtf", "compiled HRTF cache")
|
||||
|
||||
|
||||
class SofaBinauralRenderer:
|
||||
"""Render interleaved LFE plus fifteen JOC objects to stereo.
|
||||
|
||||
The adapter owns frame buffering and sample-timed OAMD updates. The
|
||||
backend owns the 64-QMF/77-hybrid state, the 961-sample latency policy,
|
||||
per-object direct/early state, and the shared late room.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self, backend, *,
|
||||
mode: str = "mid",
|
||||
object_delay_samples: int = 1473,
|
||||
tail_seconds: float = 5.0,
|
||||
chunk_frames: int = 64):
|
||||
required_interface = (
|
||||
"source_count", "default_profile", "set_source", "process",
|
||||
"finish", "finish_output_capacity", "info")
|
||||
missing = [name for name in required_interface if not hasattr(backend, name)]
|
||||
if missing:
|
||||
raise TypeError(
|
||||
f"backend must implement the binaural backend interface; "
|
||||
f"missing: {', '.join(missing)}")
|
||||
if backend.source_count != SOURCE_CHANNELS:
|
||||
raise ValueError(f"JOC binaural backend must have {SOURCE_CHANNELS} sources")
|
||||
if backend.default_profile != str(mode).lower():
|
||||
raise ValueError("backend default profile does not match renderer mode")
|
||||
if int(object_delay_samples) < 0:
|
||||
raise ValueError("object_delay_samples must be non-negative")
|
||||
if not math.isfinite(float(tail_seconds)) or float(tail_seconds) < 0.0:
|
||||
raise ValueError("tail_seconds must be finite and non-negative")
|
||||
if int(chunk_frames) <= 0:
|
||||
raise ValueError("chunk_frames must be positive")
|
||||
|
||||
self.backend = backend
|
||||
self.mode = str(mode).lower()
|
||||
self.object_delay_samples = int(object_delay_samples)
|
||||
self.tail_seconds = float(tail_seconds)
|
||||
self.chunk_frames = int(chunk_frames)
|
||||
self.chunk_samples = self.chunk_frames * FRAME_SAMPLES
|
||||
self.dsp_backend = getattr(backend, "dsp_backend", "python-sofa")
|
||||
self.timeline = OamdPositionTimeline(15)
|
||||
|
||||
self._input_buffer = np.empty(
|
||||
(self.chunk_samples, SOURCE_CHANNELS), dtype=np.float64)
|
||||
self._buffer_used = 0
|
||||
self.input_samples = 0
|
||||
self.processed_input_samples = 0
|
||||
self.output_samples = 0
|
||||
self.finished = False
|
||||
self.metadata_block_updates = 0
|
||||
|
||||
@classmethod
|
||||
def from_sofa(
|
||||
cls, sofa: str | Path, *,
|
||||
mode: str = "mid",
|
||||
cache_policy: str = "memory",
|
||||
cache_dir: str | Path | None = DEFAULT_HRTF_CACHE_DIR,
|
||||
shell_radius_m: float = 1.0,
|
||||
projection_ridge: float = DEFAULT_PROJECTION_RIDGE,
|
||||
sh_ridge: float = DEFAULT_SH_RIDGE,
|
||||
object_delay_samples: int = 1473,
|
||||
tail_seconds: float = 5.0,
|
||||
output_gain: float = 1.0,
|
||||
chunk_frames: int = 64) -> "SofaBinauralRenderer":
|
||||
source = resolve_sofa_hrtf(sofa)
|
||||
backend = SofaBinauralBackend.from_sofa(
|
||||
source,
|
||||
source_count=SOURCE_CHANNELS,
|
||||
default_profile=mode,
|
||||
output_gain=output_gain,
|
||||
cache_policy=cache_policy,
|
||||
cache_dir=cache_dir,
|
||||
shell_radius_m=shell_radius_m,
|
||||
projection_ridge=projection_ridge,
|
||||
sh_ridge=sh_ridge)
|
||||
return cls(
|
||||
backend,
|
||||
mode=mode,
|
||||
object_delay_samples=object_delay_samples,
|
||||
tail_seconds=tail_seconds,
|
||||
chunk_frames=chunk_frames)
|
||||
|
||||
@classmethod
|
||||
def from_compiled_cache(
|
||||
cls, cache: str | Path, *,
|
||||
mode: str = "mid",
|
||||
object_delay_samples: int = 1473,
|
||||
tail_seconds: float = 5.0,
|
||||
output_gain: float = 1.0,
|
||||
chunk_frames: int = 64) -> "SofaBinauralRenderer":
|
||||
source = resolve_compiled_hrtf_cache(cache)
|
||||
backend = SofaBinauralBackend.from_compiled_cache(
|
||||
source,
|
||||
source_count=SOURCE_CHANNELS,
|
||||
default_profile=mode,
|
||||
output_gain=output_gain)
|
||||
return cls(
|
||||
backend,
|
||||
mode=mode,
|
||||
object_delay_samples=object_delay_samples,
|
||||
tail_seconds=tail_seconds,
|
||||
chunk_frames=chunk_frames)
|
||||
|
||||
@property
|
||||
def finish_capacity_samples(self) -> int:
|
||||
return self.backend.finish_output_capacity(self.tail_seconds)
|
||||
|
||||
def _append_input(self, samples: np.ndarray) -> list[np.ndarray]:
|
||||
outputs = []
|
||||
source = np.asarray(samples, dtype=np.float64)
|
||||
position = 0
|
||||
while position < len(source):
|
||||
count = min(self.chunk_samples - self._buffer_used, len(source) - position)
|
||||
self._input_buffer[self._buffer_used:self._buffer_used + count] = (
|
||||
source[position:position + count])
|
||||
self._buffer_used += count
|
||||
position += count
|
||||
if self._buffer_used == self.chunk_samples:
|
||||
outputs.append(self._process_samples(self._input_buffer))
|
||||
self._buffer_used = 0
|
||||
return outputs
|
||||
|
||||
def render_frame(self, objects16, payload=None, metadata_offset=None,
|
||||
*, outer_sample_offset=0) -> np.ndarray:
|
||||
"""Submit one 1536-sample reconstructed frame and its ID11 payload."""
|
||||
if self.finished:
|
||||
raise RuntimeError("binaural renderer is already finished")
|
||||
source = np.asarray(objects16)
|
||||
if source.shape != (FRAME_SAMPLES, SOURCE_CHANNELS):
|
||||
raise ValueError(
|
||||
f"binaural frame must have shape ({FRAME_SAMPLES},{SOURCE_CHANNELS}), "
|
||||
f"got {source.shape}")
|
||||
frame_start = self.input_samples
|
||||
metadata_delay = (self.object_delay_samples if metadata_offset is None
|
||||
else int(metadata_offset))
|
||||
if metadata_delay < 0:
|
||||
raise ValueError("metadata_offset must be non-negative")
|
||||
if payload is not None:
|
||||
self.timeline.submit_payload(
|
||||
payload,
|
||||
frame_start_sample=frame_start,
|
||||
outer_sample_offset=int(outer_sample_offset),
|
||||
object_delay_samples=metadata_delay,
|
||||
processed_sample=self.processed_input_samples,
|
||||
)
|
||||
self.metadata_block_updates += 1
|
||||
self.input_samples += FRAME_SAMPLES
|
||||
chunks = self._append_input(source)
|
||||
if not chunks:
|
||||
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
return np.concatenate(chunks, axis=0) if len(chunks) > 1 else chunks[0]
|
||||
|
||||
def _set_block_parameters(self, sample: int) -> None:
|
||||
positions = self.timeline.positions_at(sample)
|
||||
self.backend.set_source(
|
||||
0, (0.0, 1.0, 0.0), profile=self.mode,
|
||||
special_lfe=True)
|
||||
for object_index in range(15):
|
||||
self.backend.set_source(
|
||||
object_index + 1,
|
||||
positions[object_index],
|
||||
profile=self.mode)
|
||||
|
||||
def _process_samples(self, source: np.ndarray) -> np.ndarray:
|
||||
values = np.asarray(source, dtype=np.float64)
|
||||
if values.ndim != 2 or values.shape[1] != SOURCE_CHANNELS:
|
||||
raise ValueError(f"expected [samples,{SOURCE_CHANNELS}], got {values.shape}")
|
||||
if len(values) % BINAURAL_BLOCK_SAMPLES:
|
||||
raise ValueError("binaural input must be divisible by 512 samples")
|
||||
outputs = []
|
||||
block_base = self.processed_input_samples
|
||||
for start in range(0, len(values), BINAURAL_BLOCK_SAMPLES):
|
||||
sample = block_base + start
|
||||
self._set_block_parameters(sample)
|
||||
outputs.append(self.backend.process(
|
||||
values[start:start + BINAURAL_BLOCK_SAMPLES]))
|
||||
self.processed_input_samples += len(values)
|
||||
nonempty = [value for value in outputs if len(value)]
|
||||
if not nonempty:
|
||||
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
output = np.concatenate(nonempty, axis=0)
|
||||
self.output_samples += len(output)
|
||||
return output
|
||||
|
||||
def finish(self) -> np.ndarray:
|
||||
"""Process pending source samples and drain early/late room state once."""
|
||||
if self.finished:
|
||||
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
outputs: list[np.ndarray] = []
|
||||
if self._buffer_used:
|
||||
outputs.append(self._process_samples(
|
||||
self._input_buffer[:self._buffer_used]))
|
||||
self._buffer_used = 0
|
||||
outputs.append(self.backend.finish(tail_seconds=self.tail_seconds))
|
||||
self.finished = True
|
||||
nonempty = [value for value in outputs if len(value)]
|
||||
if not nonempty:
|
||||
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
output = np.concatenate(nonempty, axis=0)
|
||||
self.output_samples += len(outputs[-1])
|
||||
return output
|
||||
|
||||
def close(self) -> None:
|
||||
self.finished = True
|
||||
|
||||
@property
|
||||
def backend_info(self) -> dict:
|
||||
info = self.backend.info()
|
||||
info.update({
|
||||
"adapter": "JOC 1536-frame / 512-sample metadata",
|
||||
"dsp_backend": self.dsp_backend,
|
||||
"mode": self.mode,
|
||||
"latency_compensated_samples": BINAURAL_LATENCY_SAMPLES,
|
||||
"object_delay_samples": self.object_delay_samples,
|
||||
"tail_seconds": self.tail_seconds,
|
||||
"metadata_payloads": self.timeline.payload_count,
|
||||
"metadata_position_transitions": self.timeline.transition_count,
|
||||
"input_samples": self.input_samples,
|
||||
"source_samples_processed": self.processed_input_samples,
|
||||
"output_samples_before_tail_trim": self.output_samples,
|
||||
"thread_safe": False,
|
||||
})
|
||||
return info
|
||||
@@ -1,795 +0,0 @@
|
||||
// joc_cli -- command line frontend, argument-compatible with the reference
|
||||
// Python CLI (main.py): the same positional input, the same mode selection
|
||||
// (ADM BWF by default, --speaker-layout or --binaural), the same option names,
|
||||
// choices and defaults, and the same default output naming under output/.
|
||||
//
|
||||
// Options that exist only because this build has no Python side or no Rosella
|
||||
// import chain (--backend python, --sofa-hrtf, --personalized-headphone,
|
||||
// metadata sidecars) fail with an explicit message instead of being ignored.
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <filesystem>
|
||||
#include <stdexcept>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/fs_utf8.h"
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace {
|
||||
|
||||
namespace fs = std::filesystem;
|
||||
namespace fs_utf8 = joc::fs_utf8;
|
||||
|
||||
constexpr double kRate = 48000.0;
|
||||
constexpr int kFrameSamples = 1536;
|
||||
|
||||
struct Options {
|
||||
std::string input;
|
||||
std::string output;
|
||||
std::string speaker_output;
|
||||
std::string binaural_output;
|
||||
std::string speaker_layout;
|
||||
bool binaural = false;
|
||||
std::string speaker_format = "float32";
|
||||
std::string binaural_format = "float32";
|
||||
std::string clip_action = "ask";
|
||||
int speaker_metadata_offset = 1473;
|
||||
std::string binaural_mode = "mid";
|
||||
std::string sofa_hrtf;
|
||||
std::string compiled_hrtf_cache;
|
||||
std::string personalized_headphone;
|
||||
bool personalized_headphone_used = false;
|
||||
std::string hrtf_cache_policy;
|
||||
std::string hrtf_cache_dir;
|
||||
double hrtf_radius_m = 1.0;
|
||||
double binaural_tail_seconds = 5.0;
|
||||
double binaural_tail_threshold = 1.0e-8;
|
||||
int binaural_chunk_frames = 64;
|
||||
double gain_db = 0.0;
|
||||
double duration = 0.0;
|
||||
bool duration_set = false;
|
||||
int object_delay_samples = 1473;
|
||||
std::string trajectory_mode = "compact";
|
||||
std::string ffmpeg;
|
||||
double eac3_drc_scale = 0.0;
|
||||
int eac3_target_level = 0;
|
||||
std::string backend = "auto";
|
||||
std::string native_library;
|
||||
int native_threads = 0;
|
||||
bool native_threads_set = false;
|
||||
std::string metadata_dir;
|
||||
std::string metadata_cache;
|
||||
std::string metadata_backend = "auto";
|
||||
std::string print_metadata = "none";
|
||||
std::string metadata_json;
|
||||
bool metadata_only = false;
|
||||
bool keep_raw = false;
|
||||
bool skip_sha256 = false;
|
||||
int progress_every = 1000;
|
||||
// C++-side additions (documented as such; the Python CLI has no equivalent).
|
||||
std::string bed;
|
||||
std::string kernels;
|
||||
std::string work_dir;
|
||||
std::string report_json;
|
||||
bool report_json_set = false;
|
||||
bool dry_run = false;
|
||||
bool quiet = false;
|
||||
bool help = false;
|
||||
};
|
||||
|
||||
const char* kLayoutChoices =
|
||||
"2.0 3.0 3.1 4.0 5.0 5.1 5.1.2 5.1.4 6.1 7.0 7.1 7.1.2 7.1.4 9.1.4 9.1.6 22.2";
|
||||
|
||||
void print_usage() {
|
||||
std::printf(
|
||||
"usage: joc_cli [options] input\n"
|
||||
"\n"
|
||||
"JustOneCacophony (JOC):E-AC-3 JOC → 25ch ADM BWF、扬声器 WAV 或双耳 WAV\n"
|
||||
"\n"
|
||||
"位置参数:\n"
|
||||
" input 输入 .m4a/.eac3/.ec3\n"
|
||||
"\n"
|
||||
"模式(默认输出 ADM BWF):\n"
|
||||
" --speaker-layout L 直接扬声器渲染布局,例如 2.0、5.1、7.1.2\n"
|
||||
" 可选值: %s\n"
|
||||
" --binaural 直接双耳渲染;不生成临时 ADM BWF\n"
|
||||
"\n"
|
||||
"输出:\n"
|
||||
" -o, --output PATH 输出文件;默认 output/<名称>.adm.wav、\n"
|
||||
" output/<名称>.<布局>.wav 或 output/<名称>.binaural.wav\n"
|
||||
" --speaker-output PATH 扬声器 WAV 路径;仅与 --speaker-layout 一起使用\n"
|
||||
" --binaural-output PATH 双耳 WAV 路径;仅与 --binaural 一起使用\n"
|
||||
" --speaker-format F 扬声器 WAV 格式 float32|int24,默认 float32\n"
|
||||
" --binaural-format F 双耳 WAV 格式 float32|int24,默认 float32\n"
|
||||
" --clip-action A int24 削波处理 ask|continue|float32|abort,默认 ask\n"
|
||||
"\n"
|
||||
"渲染:\n"
|
||||
" --speaker-metadata-offset N 扬声器渲染 metadata 相对帧偏移,默认 1473 samples\n"
|
||||
" --binaural-mode M 双耳渲染模式 off|near|mid|far,默认 mid;\n"
|
||||
" off 仅用于 ADM BWF(关闭 DBMD 双耳提示)\n"
|
||||
" --sofa-hrtf PATH SOFA SimpleFreeFieldHRIR 输入;.jochrtf 由本工具内部编译\n"
|
||||
" --personalized-headphone [PATH] Rosella 个性化模型,默认 "
|
||||
"HRTF/binaural.personalized_headphone\n"
|
||||
" --compiled-hrtf-cache PATH 直接读取 .jochrtf(高级用法,跳过 SOFA 编译)\n"
|
||||
" --hrtf-cache-policy P SOFA 编译缓存策略 none|memory|disk,默认 memory\n"
|
||||
" --hrtf-cache-dir DIR disk cache 目录,默认 <exe>/output/hrtf-cache\n"
|
||||
" --hrtf-radius-m R 选择最近的 SOFA measurement-radius shell,默认 1.0 m\n"
|
||||
" --binaural-tail-seconds S 双耳 room/filterbank flush 上限,默认 5 秒\n"
|
||||
" --binaural-tail-threshold T 双耳尾声裁切阈值,默认 1e-8;主体至少保留原时长\n"
|
||||
" --binaural-chunk-frames N 双耳内部批处理帧数,默认 64(本构建按 512 块渲染,\n"
|
||||
" 取值不影响输出)\n"
|
||||
" --gain-db X 成品增益 dB,默认 0;双耳路径以 float64 应用\n"
|
||||
" --duration S 只处理开头指定秒数\n"
|
||||
" --object-delay-samples N 对象 PCM/OAMD 时间补偿,默认 1473 samples\n"
|
||||
" --trajectory-mode M ADM 对象轨迹表示 compact|dense64,默认 compact\n"
|
||||
"\n"
|
||||
"输入与解码:\n"
|
||||
" --ffmpeg PATH ffmpeg 可执行文件,默认取 FFMPEG 环境变量或 PATH\n"
|
||||
" --eac3-drc-scale X E-AC-3 解码器 -drc_scale,0=关闭码流 dynrng,默认 0\n"
|
||||
" --eac3-target-level N E-AC-3 解码器 -target_level,0=不施加,默认 0\n"
|
||||
" --backend B JOC/扬声器 DSP 后端 auto|native;本构建无 python 后端\n"
|
||||
" --native-threads N 原生 DSP 总线程数;默认在 4 核以上使用 2\n"
|
||||
"\n"
|
||||
"诊断:\n"
|
||||
" --print-metadata M 诊断元数据输出 none|summary|frames,默认 none\n"
|
||||
" --metadata-json PATH 元数据汇总 JSON 路径\n"
|
||||
" --metadata-only 解析/打印元数据后退出\n"
|
||||
" --keep-raw 额外保留 16ch f32le 对象中间文件\n"
|
||||
" --skip-sha256 跳过最终文件 SHA-256 全量复扫\n"
|
||||
" --progress-every N 进度输出间隔,默认 1000 帧(渲染与收尾写盘同一节奏)\n"
|
||||
"\n"
|
||||
"本构建特有(Python 版没有对应参数):\n"
|
||||
" --bed PATH 已解码的 6 通道 float32 PCM;给出后不调用 ffmpeg 解码\n"
|
||||
" --kernels PATH 双耳滤波器组表 rosella_kernels.npz\n"
|
||||
" --work-dir DIR 临时目录\n"
|
||||
" --report-json PATH 结果 JSON 路径;默认 <输出>.report.json\n"
|
||||
" --dry-run 只校验配置\n"
|
||||
" --quiet 只输出警告与错误\n",
|
||||
kLayoutChoices);
|
||||
}
|
||||
|
||||
[[noreturn]] void fail(const std::string& message) { throw std::runtime_error(message); }
|
||||
|
||||
std::string require_value(const std::vector<std::string>& arguments, int* index) {
|
||||
if (static_cast<std::size_t>(*index) + 1u >= arguments.size()) {
|
||||
fail("argument " + arguments[static_cast<std::size_t>(*index)] +
|
||||
": expected one argument");
|
||||
}
|
||||
return arguments[static_cast<std::size_t>(++(*index))];
|
||||
}
|
||||
|
||||
double to_double(const std::string& text, const char* name) {
|
||||
try {
|
||||
std::size_t used = 0;
|
||||
const double value = std::stod(text, &used);
|
||||
if (used != text.size()) {
|
||||
fail(std::string(name) + ": invalid float value: " + text);
|
||||
}
|
||||
return value;
|
||||
} catch (const std::exception&) {
|
||||
fail(std::string(name) + ": invalid float value: " + text);
|
||||
}
|
||||
}
|
||||
|
||||
long long to_int(const std::string& text, const char* name) {
|
||||
try {
|
||||
std::size_t used = 0;
|
||||
const long long value = std::stoll(text, &used);
|
||||
if (used != text.size()) {
|
||||
fail(std::string(name) + ": invalid int value: " + text);
|
||||
}
|
||||
return value;
|
||||
} catch (const std::exception&) {
|
||||
fail(std::string(name) + ": invalid int value: " + text);
|
||||
}
|
||||
}
|
||||
|
||||
void check_choice(const std::string& value, const char* name,
|
||||
std::initializer_list<const char*> allowed) {
|
||||
for (const char* candidate : allowed) {
|
||||
if (value == candidate) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
std::string list;
|
||||
for (const char* candidate : allowed) {
|
||||
list += list.empty() ? candidate : (", " + std::string(candidate));
|
||||
}
|
||||
fail(std::string(name) + ": invalid choice: '" + value + "' (choose from " + list + ")");
|
||||
}
|
||||
|
||||
void parse_args(const std::vector<std::string>& arguments, Options* options) {
|
||||
std::vector<std::string> positional;
|
||||
const int argc = static_cast<int>(arguments.size());
|
||||
for (int index = 1; index < argc; ++index) {
|
||||
const std::string arg = arguments[static_cast<std::size_t>(index)];
|
||||
if (arg == "-h" || arg == "--help") { options->help = true; }
|
||||
else if (arg == "-o" || arg == "--output") { options->output = require_value(arguments, &index); }
|
||||
else if (arg == "--speaker-output") { options->speaker_output = require_value(arguments, &index); }
|
||||
else if (arg == "--binaural-output") { options->binaural_output = require_value(arguments, &index); }
|
||||
else if (arg == "--speaker-layout") { options->speaker_layout = require_value(arguments, &index); }
|
||||
else if (arg == "--binaural") { options->binaural = true; }
|
||||
else if (arg == "--speaker-format") { options->speaker_format = require_value(arguments, &index); }
|
||||
else if (arg == "--binaural-format") { options->binaural_format = require_value(arguments, &index); }
|
||||
else if (arg == "--clip-action") { options->clip_action = require_value(arguments, &index); }
|
||||
else if (arg == "--speaker-metadata-offset") {
|
||||
options->speaker_metadata_offset = static_cast<int>(
|
||||
to_int(require_value(arguments, &index), "--speaker-metadata-offset"));
|
||||
}
|
||||
else if (arg == "--binaural-mode") { options->binaural_mode = require_value(arguments, &index); }
|
||||
else if (arg == "--sofa-hrtf") { options->sofa_hrtf = require_value(arguments, &index); }
|
||||
else if (arg == "--compiled-hrtf-cache") { options->compiled_hrtf_cache = require_value(arguments, &index); }
|
||||
else if (arg == "--personalized-headphone") {
|
||||
options->personalized_headphone_used = true;
|
||||
// nargs="?": the path is optional, so the next token may be the input.
|
||||
// Without a path the executable-anchored default is resolved later.
|
||||
if (static_cast<std::size_t>(index) + 1u < arguments.size() &&
|
||||
arguments[static_cast<std::size_t>(index) + 1u][0] != '-') {
|
||||
options->personalized_headphone = require_value(arguments, &index);
|
||||
}
|
||||
}
|
||||
else if (arg == "--hrtf-cache-policy") { options->hrtf_cache_policy = require_value(arguments, &index); }
|
||||
else if (arg == "--hrtf-cache-dir") { options->hrtf_cache_dir = require_value(arguments, &index); }
|
||||
else if (arg == "--hrtf-radius-m") { options->hrtf_radius_m = to_double(require_value(arguments, &index), "--hrtf-radius-m"); }
|
||||
else if (arg == "--binaural-tail-seconds") { options->binaural_tail_seconds = to_double(require_value(arguments, &index), "--binaural-tail-seconds"); }
|
||||
else if (arg == "--binaural-tail-threshold") { options->binaural_tail_threshold = to_double(require_value(arguments, &index), "--binaural-tail-threshold"); }
|
||||
else if (arg == "--binaural-chunk-frames") { options->binaural_chunk_frames = static_cast<int>(to_int(require_value(arguments, &index), "--binaural-chunk-frames")); }
|
||||
else if (arg == "--gain-db") { options->gain_db = to_double(require_value(arguments, &index), "--gain-db"); }
|
||||
else if (arg == "--duration") { options->duration = to_double(require_value(arguments, &index), "--duration"); options->duration_set = true; }
|
||||
else if (arg == "--object-delay-samples") { options->object_delay_samples = static_cast<int>(to_int(require_value(arguments, &index), "--object-delay-samples")); }
|
||||
else if (arg == "--trajectory-mode") { options->trajectory_mode = require_value(arguments, &index); }
|
||||
else if (arg == "--ffmpeg") { options->ffmpeg = require_value(arguments, &index); }
|
||||
else if (arg == "--eac3-drc-scale") { options->eac3_drc_scale = to_double(require_value(arguments, &index), "--eac3-drc-scale"); }
|
||||
else if (arg == "--eac3-target-level") { options->eac3_target_level = static_cast<int>(to_int(require_value(arguments, &index), "--eac3-target-level")); }
|
||||
else if (arg == "--backend") { options->backend = require_value(arguments, &index); }
|
||||
else if (arg == "--native-library") { options->native_library = require_value(arguments, &index); }
|
||||
else if (arg == "--native-threads") { options->native_threads = static_cast<int>(to_int(require_value(arguments, &index), "--native-threads")); options->native_threads_set = true; }
|
||||
else if (arg == "--metadata-dir") { options->metadata_dir = require_value(arguments, &index); }
|
||||
else if (arg == "--metadata-cache") { options->metadata_cache = require_value(arguments, &index); }
|
||||
else if (arg == "--metadata-backend") { options->metadata_backend = require_value(arguments, &index); }
|
||||
else if (arg == "--print-metadata") { options->print_metadata = require_value(arguments, &index); }
|
||||
else if (arg == "--metadata-json") { options->metadata_json = require_value(arguments, &index); }
|
||||
else if (arg == "--metadata-only") { options->metadata_only = true; }
|
||||
else if (arg == "--keep-raw") { options->keep_raw = true; }
|
||||
else if (arg == "--skip-sha256") { options->skip_sha256 = true; }
|
||||
else if (arg == "--progress-every") { options->progress_every = static_cast<int>(to_int(require_value(arguments, &index), "--progress-every")); }
|
||||
else if (arg == "--bed") { options->bed = require_value(arguments, &index); }
|
||||
else if (arg == "--kernels") { options->kernels = require_value(arguments, &index); }
|
||||
else if (arg == "--work-dir") { options->work_dir = require_value(arguments, &index); }
|
||||
else if (arg == "--report-json") { options->report_json = require_value(arguments, &index); options->report_json_set = true; }
|
||||
else if (arg == "--dry-run") { options->dry_run = true; }
|
||||
else if (arg == "--quiet") { options->quiet = true; }
|
||||
else if (!arg.empty() && arg[0] == '-' && arg != "-") { fail("unrecognized argument: " + arg); }
|
||||
else { positional.push_back(arg); }
|
||||
}
|
||||
if (positional.size() > 1u) {
|
||||
fail("unrecognized extra arguments: " + positional[1] +
|
||||
(positional.size() > 2u ? " ..." : ""));
|
||||
}
|
||||
if (!positional.empty()) {
|
||||
options->input = positional.front();
|
||||
}
|
||||
}
|
||||
|
||||
// Mirrors the reference resolve_output(): <project>/output plus a mode-specific
|
||||
// name. The project directory is the executable's directory, as upstream uses
|
||||
// the script's directory, so the layout does not depend on the working directory.
|
||||
std::string resolve_output(const Options& options, const std::string& source,
|
||||
const std::string& executable_directory) {
|
||||
const std::string requested = !options.speaker_output.empty() ? options.speaker_output
|
||||
: !options.binaural_output.empty() ? options.binaural_output
|
||||
: options.output;
|
||||
if (!requested.empty()) {
|
||||
std::error_code error;
|
||||
const fs::path absolute = fs::absolute(fs_utf8::to_path(requested), error);
|
||||
return error ? requested : fs_utf8::from_path(absolute);
|
||||
}
|
||||
const fs::path directory = fs_utf8::to_path(executable_directory) / "output";
|
||||
const std::string stem = fs_utf8::from_path(fs_utf8::to_path(source).stem());
|
||||
if (!options.speaker_layout.empty()) {
|
||||
return fs_utf8::from_path(directory /
|
||||
fs_utf8::to_path(stem + "." + options.speaker_layout + ".wav"));
|
||||
}
|
||||
if (options.binaural) {
|
||||
return fs_utf8::from_path(directory / fs_utf8::to_path(stem + ".binaural.wav"));
|
||||
}
|
||||
return fs_utf8::from_path(directory / fs_utf8::to_path(stem + ".adm.wav"));
|
||||
}
|
||||
|
||||
// The project directory the reference anchors its defaults at: the directory of
|
||||
// the running executable, never the working directory.
|
||||
std::string executable_dir(const std::string& argv0) {
|
||||
const std::string own_path = fs_utf8::executable_path();
|
||||
if (!own_path.empty()) {
|
||||
const fs::path path = fs_utf8::to_path(own_path);
|
||||
if (path.has_parent_path()) {
|
||||
return fs_utf8::from_path(path.parent_path());
|
||||
}
|
||||
}
|
||||
if (argv0.empty()) {
|
||||
return ".";
|
||||
}
|
||||
std::error_code error;
|
||||
const fs::path path = fs::absolute(fs_utf8::to_path(argv0), error);
|
||||
if (error || path.empty()) {
|
||||
return ".";
|
||||
}
|
||||
return fs_utf8::from_path(path.parent_path());
|
||||
}
|
||||
|
||||
std::string find_kernels(const Options& options, const std::string& argv0) {
|
||||
(void)argv0;
|
||||
if (!options.kernels.empty() && !fs_utf8::exists(options.kernels)) {
|
||||
fail("--kernels 指向的文件不存在: " + options.kernels);
|
||||
}
|
||||
// Empty means the tables compiled into the library.
|
||||
return options.kernels;
|
||||
}
|
||||
|
||||
// Mirrors the reference binaural HRTF resolution (main.py:89-160): the SOFA file
|
||||
// is the user-facing input and the .jochrtf is only its compiled cache. Paths
|
||||
// are anchored at the executable directory, as the reference anchors them at the
|
||||
// project directory.
|
||||
struct HrtfInput {
|
||||
std::string sofa_path; // compile this
|
||||
std::string compiled_path; // or read this .jochrtf directly
|
||||
std::string cache_dir; // disk policy directory
|
||||
std::string personalized_path; // Rosella .personalized_headphone
|
||||
bool disk = false;
|
||||
};
|
||||
|
||||
std::string resolve_compiled_hrtf(const Options& options, const std::string& project_directory) {
|
||||
if (!options.compiled_hrtf_cache.empty()) {
|
||||
return options.compiled_hrtf_cache;
|
||||
}
|
||||
const std::string directory_utf8 =
|
||||
options.hrtf_cache_dir.empty()
|
||||
? fs_utf8::from_path(fs_utf8::to_path(project_directory) / "output" / "hrtf-cache")
|
||||
: options.hrtf_cache_dir;
|
||||
if (!fs_utf8::is_directory(directory_utf8)) {
|
||||
return std::string();
|
||||
}
|
||||
const fs::path directory = fs_utf8::to_path(directory_utf8);
|
||||
std::vector<fs::path> candidates;
|
||||
for (const fs::directory_entry& entry : fs::directory_iterator(directory)) {
|
||||
if (entry.is_regular_file() && entry.path().extension() == ".jochrtf") {
|
||||
candidates.push_back(entry.path());
|
||||
}
|
||||
}
|
||||
std::sort(candidates.begin(), candidates.end());
|
||||
if (candidates.size() > 1u) {
|
||||
fail(directory_utf8 +
|
||||
" 下有多个 .jochrtf 缓存,无法自动选择;请用 --sofa-hrtf PATH 或 "
|
||||
"--compiled-hrtf-cache PATH 显式指定");
|
||||
}
|
||||
return candidates.empty() ? std::string() : fs_utf8::from_path(candidates.front());
|
||||
}
|
||||
|
||||
HrtfInput resolve_hrtf_input(const Options& options, const std::string& project_directory) {
|
||||
HrtfInput input;
|
||||
const std::string default_sofa =
|
||||
fs_utf8::from_path(fs_utf8::to_path(project_directory) / "HRTF" / "binaural.sofa");
|
||||
const std::string default_private = fs_utf8::from_path(
|
||||
fs_utf8::to_path(project_directory) / "HRTF" / "binaural.personalized_headphone");
|
||||
const std::string default_cache_dir =
|
||||
fs_utf8::from_path(fs_utf8::to_path(project_directory) / "output" / "hrtf-cache");
|
||||
|
||||
if (!options.compiled_hrtf_cache.empty() && !options.hrtf_cache_policy.empty()) {
|
||||
fail("显式 .jochrtf 输入不能再指定 --hrtf-cache-policy");
|
||||
}
|
||||
if (!options.compiled_hrtf_cache.empty() && options.hrtf_radius_m != 1.0) {
|
||||
fail("显式 .jochrtf 输入不能再选择 SOFA radius shell");
|
||||
}
|
||||
if (options.personalized_headphone_used &&
|
||||
(!options.hrtf_cache_policy.empty() || !options.hrtf_cache_dir.empty() ||
|
||||
options.hrtf_radius_m != 1.0)) {
|
||||
fail("Rosella 模型输入不能使用 --hrtf-cache-policy/--hrtf-cache-dir/--hrtf-radius-m");
|
||||
}
|
||||
|
||||
std::string sofa = options.sofa_hrtf;
|
||||
std::string compiled = options.compiled_hrtf_cache;
|
||||
std::string personalized =
|
||||
options.personalized_headphone_used ? options.personalized_headphone : std::string();
|
||||
if (options.personalized_headphone_used && personalized.empty()) {
|
||||
// "--personalized-headphone" without a path means the project default.
|
||||
personalized = default_private;
|
||||
}
|
||||
if (sofa.empty() && compiled.empty() && personalized.empty()) {
|
||||
// The reference order: the SOFA file, then the unique compiled cache, then the
|
||||
// personalized model.
|
||||
if (fs_utf8::exists(default_sofa)) {
|
||||
sofa = default_sofa;
|
||||
} else {
|
||||
compiled = resolve_compiled_hrtf(options, project_directory);
|
||||
if (compiled.empty() && fs_utf8::exists(default_private)) {
|
||||
personalized = default_private;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!personalized.empty()) {
|
||||
if (!fs_utf8::exists(personalized)) {
|
||||
fail("双耳模型不存在: " + personalized);
|
||||
}
|
||||
input.personalized_path = personalized;
|
||||
return input;
|
||||
}
|
||||
if (sofa.empty() && compiled.empty()) {
|
||||
if (!options.hrtf_cache_policy.empty() || !options.hrtf_cache_dir.empty() ||
|
||||
options.hrtf_radius_m != 1.0) {
|
||||
fail("HRTF cache/radius 选项需要 --sofa-hrtf");
|
||||
}
|
||||
fail("--binaural 未找到 HRTF 输入:默认 " + default_sofa + "、" + default_private +
|
||||
" 或 " + default_cache_dir +
|
||||
" 下的 .jochrtf 都不存在,请用 --sofa-hrtf PATH、--personalized-headphone PATH "
|
||||
"或 --compiled-hrtf-cache PATH 指定");
|
||||
}
|
||||
if (sofa.empty() && (!options.hrtf_cache_policy.empty() || !options.hrtf_cache_dir.empty() ||
|
||||
options.hrtf_radius_m != 1.0)) {
|
||||
fail("HRTF cache/radius 选项需要 --sofa-hrtf");
|
||||
}
|
||||
if (sofa.empty()) {
|
||||
input.compiled_path = compiled;
|
||||
return input;
|
||||
}
|
||||
const std::string effective_policy =
|
||||
options.hrtf_cache_policy.empty() ? "memory" : options.hrtf_cache_policy;
|
||||
if (!options.hrtf_cache_dir.empty() && effective_policy != "disk") {
|
||||
fail("--hrtf-cache-dir 需要 SOFA 与 disk cache policy 一起使用");
|
||||
}
|
||||
input.sofa_path = sofa;
|
||||
input.disk = effective_policy == "disk";
|
||||
input.cache_dir = options.hrtf_cache_dir.empty() ? default_cache_dir : options.hrtf_cache_dir;
|
||||
if (!fs_utf8::exists(input.sofa_path)) {
|
||||
fail("SOFA HRTF 不存在: " + input.sofa_path);
|
||||
}
|
||||
return input;
|
||||
}
|
||||
|
||||
std::uint32_t binaural_mode_value(const std::string& name) {
|
||||
if (name == "off") { return JOC_BINAURAL_OFF; }
|
||||
if (name == "near") { return JOC_BINAURAL_NEAR; }
|
||||
if (name == "far") { return JOC_BINAURAL_FAR; }
|
||||
return JOC_BINAURAL_MID;
|
||||
}
|
||||
|
||||
std::uint32_t clip_action_value(const std::string& name) {
|
||||
if (name == "continue") { return JOC_CLIP_CONTINUE; }
|
||||
if (name == "float32") { return JOC_CLIP_FLOAT32; }
|
||||
if (name == "abort") { return JOC_CLIP_ABORT; }
|
||||
return JOC_CLIP_ASK;
|
||||
}
|
||||
|
||||
std::string format_eta(double seconds) {
|
||||
if (seconds < 0.0 || seconds > 86400.0) {
|
||||
return "--";
|
||||
}
|
||||
char buffer[64];
|
||||
std::snprintf(buffer, sizeof(buffer), "%.0fs", seconds);
|
||||
return buffer;
|
||||
}
|
||||
|
||||
void JOC_CALL on_event(void* user, const joc_event* event) {
|
||||
const Options* options = static_cast<const Options*>(user);
|
||||
if (event == nullptr) {
|
||||
return;
|
||||
}
|
||||
switch (event->type) {
|
||||
case JOC_EV_PROGRESS: {
|
||||
if (options->quiet) {
|
||||
return;
|
||||
}
|
||||
const double fraction = event->progress >= 0.0 ? event->progress : 0.0;
|
||||
const double remaining =
|
||||
fraction > 0.0 ? event->elapsed_seconds * (1.0 - fraction) / fraction : -1.0;
|
||||
std::printf("[%s] %llu/%llu %.1fx realtime ETA %s\n", event->stage_name,
|
||||
static_cast<unsigned long long>(event->current_frame),
|
||||
static_cast<unsigned long long>(event->total_frames),
|
||||
event->realtime_factor, format_eta(remaining).c_str());
|
||||
std::fflush(stdout);
|
||||
return;
|
||||
}
|
||||
case JOC_EV_LOG: {
|
||||
if (options->quiet || event->log_level < JOC_LOG_INFO || event->message[0] == '\0') {
|
||||
return;
|
||||
}
|
||||
std::printf("[%s] %s\n", event->stage_name, event->message);
|
||||
std::fflush(stdout);
|
||||
return;
|
||||
}
|
||||
case JOC_EV_WARNING:
|
||||
std::printf("[warning] %s\n", event->message);
|
||||
return;
|
||||
case JOC_EV_ERROR:
|
||||
std::fprintf(stderr, "[error] %s (%s)\n", event->message,
|
||||
joc_error_name(event->error_code));
|
||||
return;
|
||||
default:
|
||||
if (options->quiet || event->log_level < JOC_LOG_INFO || event->message[0] == '\0') {
|
||||
return;
|
||||
}
|
||||
std::printf("[%s] %s\n", event->stage_name, event->message);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
fs_utf8::configure_console();
|
||||
const std::vector<std::string> arguments = fs_utf8::command_line_arguments(argc, argv);
|
||||
Options options;
|
||||
try {
|
||||
parse_args(arguments, &options);
|
||||
} catch (const std::exception& error) {
|
||||
std::fprintf(stderr, "joc_cli: error: %s\n", error.what());
|
||||
return 2;
|
||||
}
|
||||
if (options.help) {
|
||||
print_usage();
|
||||
return 0;
|
||||
}
|
||||
if (options.input.empty()) {
|
||||
print_usage();
|
||||
return 2;
|
||||
}
|
||||
|
||||
try {
|
||||
check_choice(options.speaker_format, "--speaker-format", {"float32", "int24"});
|
||||
check_choice(options.binaural_format, "--binaural-format", {"float32", "int24"});
|
||||
check_choice(options.clip_action, "--clip-action",
|
||||
{"ask", "continue", "float32", "abort"});
|
||||
check_choice(options.binaural_mode, "--binaural-mode", {"off", "near", "mid", "far"});
|
||||
check_choice(options.trajectory_mode, "--trajectory-mode", {"compact", "dense64"});
|
||||
check_choice(options.backend, "--backend", {"auto", "native", "python"});
|
||||
check_choice(options.print_metadata, "--print-metadata", {"none", "summary", "frames"});
|
||||
check_choice(options.metadata_backend, "--metadata-backend", {"auto", "emdf", "sidecar"});
|
||||
if (!options.hrtf_cache_policy.empty()) {
|
||||
check_choice(options.hrtf_cache_policy, "--hrtf-cache-policy",
|
||||
{"none", "memory", "disk"});
|
||||
}
|
||||
|
||||
const bool speaker_mode = !options.speaker_layout.empty();
|
||||
const bool binaural_mode = options.binaural;
|
||||
if (speaker_mode && binaural_mode) {
|
||||
fail("argument --binaural: not allowed with argument --speaker-layout");
|
||||
}
|
||||
if (options.binaural_mode == "off" && (speaker_mode || binaural_mode)) {
|
||||
fail("--binaural-mode off 仅用于 ADM BWF 输出(关闭 DBMD 双耳提示);"
|
||||
"直接双耳渲染请使用 near/mid/far");
|
||||
}
|
||||
if (!options.speaker_output.empty() && !speaker_mode) {
|
||||
fail("--speaker-output 必须与 --speaker-layout 一起使用");
|
||||
}
|
||||
if (!options.binaural_output.empty() && !binaural_mode) {
|
||||
fail("--binaural-output 必须与 --binaural 一起使用");
|
||||
}
|
||||
const bool specific_output =
|
||||
!options.speaker_output.empty() || !options.binaural_output.empty();
|
||||
if (!options.output.empty() && specific_output) {
|
||||
fail("-o/--output 与 --speaker-output/--binaural-output 不能同时使用");
|
||||
}
|
||||
if (!options.speaker_output.empty() && !options.binaural_output.empty()) {
|
||||
fail("--speaker-output 与 --binaural-output 不能同时使用");
|
||||
}
|
||||
if (options.speaker_metadata_offset < 0) {
|
||||
fail("speaker-metadata-offset 不能为负数");
|
||||
}
|
||||
const bool hrtf_options_used =
|
||||
!options.sofa_hrtf.empty() || !options.compiled_hrtf_cache.empty() ||
|
||||
options.personalized_headphone_used || !options.hrtf_cache_policy.empty() ||
|
||||
!options.hrtf_cache_dir.empty() || options.hrtf_radius_m != 1.0;
|
||||
if (hrtf_options_used && !binaural_mode) {
|
||||
fail("SOFA/HRTF 选项仅与 --binaural 一起使用");
|
||||
}
|
||||
if (!std::isfinite(options.binaural_tail_seconds) ||
|
||||
options.binaural_tail_seconds < 0.0) {
|
||||
fail("binaural-tail-seconds 必须是非负有限值");
|
||||
}
|
||||
if (!std::isfinite(options.binaural_tail_threshold) ||
|
||||
options.binaural_tail_threshold < 0.0) {
|
||||
fail("binaural-tail-threshold 必须是非负有限值");
|
||||
}
|
||||
if (options.binaural_chunk_frames <= 0) {
|
||||
fail("binaural-chunk-frames 必须大于 0");
|
||||
}
|
||||
if (!std::isfinite(options.hrtf_radius_m) || options.hrtf_radius_m <= 0.0) {
|
||||
fail("hrtf-radius-m 必须是正有限值");
|
||||
}
|
||||
if (options.duration_set && options.duration <= 0.0) {
|
||||
fail("duration 必须大于 0");
|
||||
}
|
||||
if (options.object_delay_samples < 0) {
|
||||
fail("object-delay-samples 不能为负数");
|
||||
}
|
||||
if (options.native_threads_set && options.native_threads < 1) {
|
||||
fail("native-threads 必须大于 0");
|
||||
}
|
||||
if (!std::isfinite(options.gain_db) || std::abs(options.gain_db) > 200.0) {
|
||||
fail("gain-db 超出支持范围");
|
||||
}
|
||||
// Options this build cannot honour: fail loudly instead of ignoring them.
|
||||
if (options.backend == "python") {
|
||||
fail("--backend python 在本构建中不可用(已无 Python 后端);请使用 auto 或 native");
|
||||
}
|
||||
if (options.metadata_backend == "sidecar" || !options.metadata_dir.empty() ||
|
||||
!options.metadata_cache.empty()) {
|
||||
fail("metadata sidecar 在本构建中不可用(始终直接扫描 EMDF)");
|
||||
}
|
||||
if (!options.native_library.empty()) {
|
||||
std::fprintf(stderr, "[info] --native-library 在本构建中忽略(单一 joc_core.dll)\n");
|
||||
}
|
||||
if (!fs_utf8::exists(options.input)) {
|
||||
fail("输入文件不存在: " + options.input);
|
||||
}
|
||||
} catch (const std::exception& error) {
|
||||
std::fprintf(stderr, "joc_cli: error: %s\n", error.what());
|
||||
return 2;
|
||||
}
|
||||
|
||||
const bool speaker_mode = !options.speaker_layout.empty();
|
||||
const bool binaural_mode = options.binaural;
|
||||
const std::string project_directory =
|
||||
executable_dir(arguments.empty() ? std::string() : arguments.front());
|
||||
const std::string output_path = resolve_output(options, options.input, project_directory);
|
||||
std::error_code directory_error;
|
||||
fs::create_directories(fs_utf8::to_path(output_path).parent_path(), directory_error);
|
||||
|
||||
joc_task_config config{};
|
||||
config.struct_size = sizeof(config);
|
||||
config.struct_version = JOC_TASK_CONFIG_VERSION;
|
||||
config.input_path = options.input.c_str();
|
||||
config.output_path = output_path.c_str();
|
||||
config.ffmpeg_path = options.ffmpeg.empty() ? nullptr : options.ffmpeg.c_str();
|
||||
config.bed_path = options.bed.empty() ? nullptr : options.bed.c_str();
|
||||
config.work_dir = options.work_dir.empty() ? nullptr : options.work_dir.c_str();
|
||||
config.eac3_drc_scale = options.eac3_drc_scale;
|
||||
config.eac3_target_level = options.eac3_target_level;
|
||||
config.operation =
|
||||
binaural_mode ? JOC_OP_BINAURAL : speaker_mode ? JOC_OP_SPEAKER : JOC_OP_ADM_BWF;
|
||||
const std::string& requested_format =
|
||||
binaural_mode ? options.binaural_format : options.speaker_format;
|
||||
config.output_format =
|
||||
requested_format == "int24" ? JOC_FORMAT_PCM24 : JOC_FORMAT_FLOAT32;
|
||||
config.clip_action = clip_action_value(options.clip_action);
|
||||
config.speaker_layout_name = speaker_mode ? options.speaker_layout.c_str() : nullptr;
|
||||
config.speaker_metadata_offset = static_cast<std::uint32_t>(options.speaker_metadata_offset);
|
||||
config.binaural_mode = binaural_mode_value(options.binaural_mode);
|
||||
config.adm_binaural_mode = binaural_mode_value(options.binaural_mode);
|
||||
config.binaural_tail_seconds = options.binaural_tail_seconds;
|
||||
config.binaural_tail_threshold = options.binaural_tail_threshold;
|
||||
config.binaural_chunk_frames = static_cast<std::uint32_t>(options.binaural_chunk_frames);
|
||||
config.object_delay_samples = static_cast<std::uint32_t>(options.object_delay_samples);
|
||||
config.trajectory_mode =
|
||||
options.trajectory_mode == "dense64" ? JOC_TRAJECTORY_DENSE64 : JOC_TRAJECTORY_COMPACT;
|
||||
config.gain_db = options.gain_db;
|
||||
config.progress_interval_frames = static_cast<std::uint32_t>(options.progress_every);
|
||||
config.native_threads =
|
||||
options.native_threads_set ? static_cast<std::uint32_t>(options.native_threads) : 0u;
|
||||
config.print_metadata = options.print_metadata == "frames" ? 2u
|
||||
: options.print_metadata == "summary" ? 1u
|
||||
: 0u;
|
||||
config.metadata_json_path =
|
||||
options.metadata_json.empty() ? nullptr : options.metadata_json.c_str();
|
||||
config.duration_frames =
|
||||
options.duration_set
|
||||
? static_cast<std::uint64_t>(
|
||||
std::ceil(options.duration * kRate / static_cast<double>(kFrameSamples)))
|
||||
: 0u;
|
||||
config.flags = 0u;
|
||||
if (options.skip_sha256) { config.flags |= JOC_TASK_F_SKIP_SHA256; }
|
||||
if (options.keep_raw) { config.flags |= JOC_TASK_F_KEEP_INTERMEDIATE; }
|
||||
if (options.metadata_only) { config.flags |= JOC_TASK_F_METADATA_ONLY; }
|
||||
if (options.quiet) { config.flags |= JOC_TASK_F_QUIET; }
|
||||
|
||||
std::string hrtf_path;
|
||||
std::string hrtf_sofa_path;
|
||||
std::string hrtf_cache_dir;
|
||||
std::string personalized_path;
|
||||
std::string kernels_path;
|
||||
if (binaural_mode) {
|
||||
try {
|
||||
const HrtfInput input = resolve_hrtf_input(options, project_directory);
|
||||
hrtf_path = input.compiled_path;
|
||||
hrtf_sofa_path = input.sofa_path;
|
||||
hrtf_cache_dir = input.cache_dir;
|
||||
personalized_path = input.personalized_path;
|
||||
kernels_path = find_kernels(options, arguments.empty() ? std::string()
|
||||
: arguments.front());
|
||||
} catch (const std::exception& error) {
|
||||
std::fprintf(stderr, "joc_cli: error: %s\n", error.what());
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
config.hrtf_path = hrtf_path.empty() ? nullptr : hrtf_path.c_str();
|
||||
config.hrtf_sofa_path = hrtf_sofa_path.empty() ? nullptr : hrtf_sofa_path.c_str();
|
||||
config.hrtf_cache_dir = hrtf_cache_dir.empty() ? nullptr : hrtf_cache_dir.c_str();
|
||||
config.personalized_headphone_path =
|
||||
personalized_path.empty() ? nullptr : personalized_path.c_str();
|
||||
config.hrtf_cache_policy = options.hrtf_cache_policy == "disk" ? JOC_HRTF_CACHE_DISK
|
||||
: options.hrtf_cache_policy == "none" ? JOC_HRTF_CACHE_NONE
|
||||
: JOC_HRTF_CACHE_MEMORY;
|
||||
config.hrtf_radius_m = options.hrtf_radius_m;
|
||||
config.kernels_path = kernels_path.empty() ? nullptr : kernels_path.c_str();
|
||||
|
||||
joc_validation_issue issues[32];
|
||||
std::uint32_t issue_count = 0;
|
||||
const joc_error validated = joc_task_validate(&config, issues, 32u, &issue_count);
|
||||
for (std::uint32_t index = 0; index < std::min(issue_count, 32u); ++index) {
|
||||
if (issues[index].severity >= 2u) {
|
||||
std::fprintf(stderr, "[error] %s: %s\n", issues[index].field, issues[index].message);
|
||||
} else if (!options.quiet) {
|
||||
std::fprintf(stderr, "[warning] %s: %s\n", issues[index].field,
|
||||
issues[index].message);
|
||||
}
|
||||
}
|
||||
if (validated != JOC_OK) {
|
||||
std::fprintf(stderr, "joc_cli: error: configuration rejected (%u issue(s))\n", issue_count);
|
||||
return 2;
|
||||
}
|
||||
if (options.dry_run) {
|
||||
std::printf("configuration accepted (%u issue(s))\n", issue_count);
|
||||
return 0;
|
||||
}
|
||||
|
||||
joc_event_sink sink{};
|
||||
sink.struct_size = sizeof(sink);
|
||||
sink.callback = &on_event;
|
||||
sink.user = &options;
|
||||
|
||||
if (!options.quiet) {
|
||||
const char* mode_name = binaural_mode ? "binaural" : speaker_mode ? "speaker" : "adm";
|
||||
std::printf("[cli] %s -> %s (%s)\n", options.input.c_str(), output_path.c_str(),
|
||||
mode_name);
|
||||
std::fflush(stdout);
|
||||
}
|
||||
|
||||
joc_task_result result{};
|
||||
const joc_error status = joc_task_execute(&config, &sink, &result);
|
||||
|
||||
const std::string report_path =
|
||||
options.report_json_set ? options.report_json : (output_path + ".report.json");
|
||||
{
|
||||
std::size_t needed = 0;
|
||||
joc_task_result_to_json(&result, nullptr, 0u, &needed);
|
||||
std::vector<char> buffer(needed + 1u);
|
||||
if (joc_task_result_to_json(&result, buffer.data(), buffer.size(), &needed) == JOC_OK) {
|
||||
if (std::FILE* file = fs_utf8::fopen(report_path, "wb")) {
|
||||
std::fwrite(buffer.data(), 1, std::strlen(buffer.data()), file);
|
||||
std::fputc('\n', file);
|
||||
std::fclose(file);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::printf("\nresult: %s\n", status == JOC_OK ? "ok" : joc_error_name(status));
|
||||
std::printf(" output : %s\n", output_path.c_str());
|
||||
std::printf(" frames : %llu (%.2f s)\n",
|
||||
static_cast<unsigned long long>(result.input_frames), result.duration_sec);
|
||||
std::printf(" output samples: %llu\n",
|
||||
static_cast<unsigned long long>(result.output_samples));
|
||||
std::printf(" output bytes : %llu\n",
|
||||
static_cast<unsigned long long>(result.output_file_bytes));
|
||||
std::printf(" format : %s\n",
|
||||
result.output_format_actual == JOC_FORMAT_PCM24 ? "int24" : "float32");
|
||||
std::printf(" peak : %.9g (%llu sample(s) above full scale)\n", result.output_peak,
|
||||
static_cast<unsigned long long>(result.output_over_unity_values));
|
||||
std::printf(" sha256 : %s\n",
|
||||
result.output_sha256[0] != '\0' ? result.output_sha256 : "(skipped)");
|
||||
std::printf(" report : %s\n", report_path.c_str());
|
||||
// Each stage time is measured where that stage actually runs, and the three
|
||||
// stages now overlap (see the pipeline in src/task/task.cpp), so the stage
|
||||
// times deliberately do not add up to the wall-clock total.
|
||||
std::printf(" timings : decode %.2fs, joc %.2fs, dsp %.2fs, write %.2fs"
|
||||
" (stage times, concurrent), total %.2fs\n",
|
||||
result.t_decode_bed, result.t_render, result.t_render_dsp, result.t_write_file,
|
||||
result.t_total);
|
||||
if (status != JOC_OK) {
|
||||
std::fprintf(stderr, "joc_cli: error: %s: %s\n", result.error_stage, result.error_message);
|
||||
}
|
||||
return status == JOC_OK ? 0 : 1;
|
||||
}
|
||||
@@ -1,113 +0,0 @@
|
||||
#include "eac3_transport/eac3_reader.h"
|
||||
|
||||
#include <utility>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::eac3 {
|
||||
|
||||
namespace {
|
||||
constexpr std::size_t kHeaderBytes = 4;
|
||||
} // namespace
|
||||
|
||||
void FrameReader::push(const std::uint8_t* data, std::size_t size) {
|
||||
if (failed_ || data == nullptr || size == 0) {
|
||||
return;
|
||||
}
|
||||
if (consumed_ > 0) {
|
||||
compact();
|
||||
}
|
||||
buffer_.insert(buffer_.end(), data, data + size);
|
||||
}
|
||||
|
||||
void FrameReader::compact() {
|
||||
if (consumed_ == 0) {
|
||||
return;
|
||||
}
|
||||
buffer_.erase(buffer_.begin(), buffer_.begin() + static_cast<std::ptrdiff_t>(consumed_));
|
||||
base_offset_ += consumed_;
|
||||
consumed_ = 0;
|
||||
}
|
||||
|
||||
void FrameReader::fail(joc_error code, std::string message) {
|
||||
failed_ = true;
|
||||
error_ = code;
|
||||
message_ = std::move(message);
|
||||
}
|
||||
|
||||
FrameReader::Next FrameReader::next(Frame* out) {
|
||||
if (failed_) {
|
||||
return Next::Fail;
|
||||
}
|
||||
const std::size_t available = buffer_.size() - consumed_;
|
||||
if (available == 0) {
|
||||
return Next::End;
|
||||
}
|
||||
const std::uint8_t* p = buffer_.data() + consumed_;
|
||||
|
||||
// The reference implementation rejects a frame whose header does not fit,
|
||||
// rather than silently resynchronising on the next 0x0B77.
|
||||
if (available < kHeaderBytes) {
|
||||
if (finished_) {
|
||||
fail(JOC_ERR_EAC3_SYNCFRAME, "E-AC-3 syncframe header truncated at end of input");
|
||||
return Next::Fail;
|
||||
}
|
||||
return Next::End;
|
||||
}
|
||||
|
||||
const std::uint16_t syncword = static_cast<std::uint16_t>((static_cast<std::uint16_t>(p[0]) << 8) | p[1]);
|
||||
if (syncword != kSyncword) {
|
||||
fail(JOC_ERR_EAC3_SYNCFRAME, "invalid E-AC-3 syncword (silent resynchronisation is not allowed)");
|
||||
return Next::Fail;
|
||||
}
|
||||
|
||||
// frmsiz: 11 bits spread over the low 3 bits of byte 2 and all of byte 3,
|
||||
const std::size_t words =
|
||||
static_cast<std::size_t>(((p[2] & 0x07u) << 8) | p[3]) + 1u;
|
||||
const std::size_t frame_bytes = words * 2u;
|
||||
|
||||
if (frame_bytes > available) {
|
||||
if (!finished_) {
|
||||
return Next::End;
|
||||
}
|
||||
fail(JOC_ERR_BITSTREAM_TRUNCATED,
|
||||
"last E-AC-3 syncframe extends past end of input (declared " +
|
||||
std::to_string(frame_bytes) + " bytes, remaining " +
|
||||
std::to_string(available) + ")");
|
||||
return Next::Fail;
|
||||
}
|
||||
|
||||
if (out != nullptr) {
|
||||
out->data = p;
|
||||
out->size = frame_bytes;
|
||||
out->offset = base_offset_ + consumed_;
|
||||
}
|
||||
consumed_ += frame_bytes;
|
||||
stream_offset_ = base_offset_ + consumed_;
|
||||
++frames_emitted_;
|
||||
return Next::Ok;
|
||||
}
|
||||
|
||||
joc_error FrameReader::frame_bytes(const std::uint8_t* data, std::size_t size, std::size_t offset,
|
||||
std::size_t* out_frame_bytes) {
|
||||
if (data == nullptr || out_frame_bytes == nullptr) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
if (offset + kHeaderBytes > size) {
|
||||
return JOC_ERR_EAC3_SYNCFRAME;
|
||||
}
|
||||
if (static_cast<std::uint16_t>((static_cast<std::uint16_t>(data[offset]) << 8) | data[offset + 1]) !=
|
||||
kSyncword) {
|
||||
return JOC_ERR_EAC3_SYNCFRAME;
|
||||
}
|
||||
const std::size_t words =
|
||||
static_cast<std::size_t>(((data[offset + 2] & 0x07u) << 8) | data[offset + 3]) + 1u;
|
||||
const std::size_t frame_bytes = words * 2u;
|
||||
if (offset + frame_bytes > size) {
|
||||
return JOC_ERR_BITSTREAM_TRUNCATED;
|
||||
}
|
||||
*out_frame_bytes = frame_bytes;
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
} // namespace joc::eac3
|
||||
@@ -1,68 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc::eac3 {
|
||||
|
||||
struct Frame {
|
||||
const std::uint8_t* data = nullptr;
|
||||
std::size_t size = 0;
|
||||
std::size_t offset = 0; // byte offset of the frame start in the fed stream
|
||||
};
|
||||
|
||||
class FrameReader {
|
||||
public:
|
||||
enum class Next {
|
||||
Ok,
|
||||
End,
|
||||
Fail
|
||||
};
|
||||
|
||||
FrameReader() = default;
|
||||
FrameReader(const std::uint8_t* data, std::size_t size) {
|
||||
push(data, size);
|
||||
finish();
|
||||
}
|
||||
|
||||
// Appends bytes to the internal buffer (used in incremental mode).
|
||||
void push(const std::uint8_t* data, std::size_t size);
|
||||
|
||||
// Declares that no further bytes will arrive; a frame that is still
|
||||
void finish() { finished_ = true; }
|
||||
|
||||
Next next(Frame* out);
|
||||
|
||||
joc_error error() const { return error_; }
|
||||
const std::string& error_message() const { return message_; }
|
||||
|
||||
std::size_t frames_emitted() const { return frames_emitted_; }
|
||||
std::size_t stream_offset() const { return stream_offset_; }
|
||||
|
||||
// report its declared byte length.
|
||||
static joc_error frame_bytes(const std::uint8_t* data, std::size_t size, std::size_t offset,
|
||||
std::size_t* out_frame_bytes);
|
||||
|
||||
static constexpr std::uint16_t kSyncword = 0x0B77;
|
||||
|
||||
private:
|
||||
void compact();
|
||||
void fail(joc_error code, std::string message);
|
||||
|
||||
std::vector<std::uint8_t> buffer_;
|
||||
std::size_t consumed_ = 0; // bytes of buffer_ already turned into frames
|
||||
std::size_t base_offset_ = 0;
|
||||
bool finished_ = false;
|
||||
bool failed_ = false;
|
||||
joc_error error_ = JOC_OK;
|
||||
std::string message_;
|
||||
std::size_t frames_emitted_ = 0;
|
||||
std::size_t stream_offset_ = 0;
|
||||
};
|
||||
|
||||
} // namespace joc::eac3
|
||||
+307
@@ -0,0 +1,307 @@
|
||||
"""从常见 E-AC-3 同步帧直接提取连续 EMDF 容器。
|
||||
|
||||
扫描器检查八种全局位对齐,定位 ``0x5838`` 同步字,验证容器长度并解析各
|
||||
payload config,因此不要求 EMDF 在原始 E-AC-3 文件中按字节对齐。
|
||||
|
||||
本模块有意只覆盖“完整 EMDF 容器在一个同步帧中连续出现”的常见情形。不解析
|
||||
E-AC-3 mantissa,也不重组被音频数据隔开的多个 skip-field 碎片;遇到这种输入会
|
||||
明确报错,让上层决定是否使用兼容桥。
|
||||
"""
|
||||
from dataclasses import dataclass
|
||||
import csv
|
||||
import hashlib
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from variant_error import UnsupportedVariantError
|
||||
|
||||
|
||||
SYNCWORD = 0x5838
|
||||
REQUIRED_JOC_IDS = frozenset((11, 14))
|
||||
|
||||
|
||||
class EmdfError(ValueError):
|
||||
"""EMDF 或其 E-AC-3 传输结构不符合本实现支持的范围。"""
|
||||
|
||||
|
||||
class BitReader:
|
||||
"""MSB-first 位读取器;位置以源数据的绝对 bit offset 表示。"""
|
||||
|
||||
def __init__(self, data, position=0, limit=None):
|
||||
self.data = memoryview(data)
|
||||
self.position = int(position)
|
||||
self.limit = len(self.data) * 8 if limit is None else int(limit)
|
||||
|
||||
def read(self, count):
|
||||
count = int(count)
|
||||
if count < 0 or self.position + count > self.limit:
|
||||
raise EmdfError(f"位流越界 @bit{self.position}, need={count}, limit={self.limit}")
|
||||
value = 0
|
||||
while count:
|
||||
byte_pos = self.position >> 3
|
||||
removed_left = self.position & 7
|
||||
take = min(count, 8 - removed_left)
|
||||
shift = 8 - removed_left - take
|
||||
value = (value << take) | ((self.data[byte_pos] >> shift) & ((1 << take) - 1))
|
||||
self.position += take
|
||||
count -= take
|
||||
return value
|
||||
|
||||
def skip(self, count):
|
||||
self.read(count)
|
||||
|
||||
def read_bytes(self, count):
|
||||
return bytes(self.read(8) for _ in range(count))
|
||||
|
||||
|
||||
def variable_bits(reader, width, max_groups=8):
|
||||
"""读取 EMDF ``variable_bits(width)`` 变长整数。"""
|
||||
value = 0
|
||||
for _ in range(max_groups):
|
||||
value += reader.read(width)
|
||||
more = reader.read(1)
|
||||
if not more:
|
||||
return value
|
||||
value = (value + 1) << width
|
||||
raise EmdfError(f"variable_bits({width}) 延伸组过多")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EmdfContainer:
|
||||
start_bit: int
|
||||
raw: bytes
|
||||
payloads: dict
|
||||
sample_offsets: dict
|
||||
|
||||
|
||||
def _parse_at(data, start_bit):
|
||||
"""在已知 syncword 的 bit offset 解析一个 EMDF 容器。"""
|
||||
reader = BitReader(data, start_bit)
|
||||
if reader.read(16) != SYNCWORD:
|
||||
raise EmdfError(f"EMDF syncword 不匹配 @bit{start_bit}")
|
||||
length = reader.read(16)
|
||||
body_start = reader.position
|
||||
body_end = body_start + length * 8
|
||||
if body_end > reader.limit:
|
||||
raise EmdfError(f"EMDF 容器越界 @bit{start_bit}: length={length}")
|
||||
reader.limit = body_end
|
||||
|
||||
version = reader.read(2)
|
||||
if version == 3:
|
||||
version += variable_bits(reader, 2)
|
||||
key_id = reader.read(3)
|
||||
if key_id == 7:
|
||||
key_id += variable_bits(reader, 3)
|
||||
# TS 103 420 JOC 使用 version=0/key_id=0;严格限制也能排除音频中的伪 marker。
|
||||
if version != 0 or key_id != 0:
|
||||
raise EmdfError(f"不支持的 EMDF version/key_id: {version}/{key_id}")
|
||||
|
||||
payloads = {}
|
||||
sample_offsets = {}
|
||||
terminated = False
|
||||
while reader.position + 5 <= body_end:
|
||||
payload_id = reader.read(5)
|
||||
if payload_id == 0:
|
||||
terminated = True
|
||||
break
|
||||
if payload_id == 0x1F:
|
||||
payload_id += variable_bits(reader, 5)
|
||||
if payload_id in payloads:
|
||||
raise EmdfError(f"同一 EMDF 容器重复 payload id {payload_id}")
|
||||
|
||||
has_sample_offset = bool(reader.read(1))
|
||||
sample_offset = (reader.read(12) >> 1) if has_sample_offset else 0
|
||||
if reader.read(1):
|
||||
variable_bits(reader, 11) # duration
|
||||
if reader.read(1):
|
||||
variable_bits(reader, 2) # group id
|
||||
if reader.read(1):
|
||||
reader.skip(8) # codec data
|
||||
|
||||
if not reader.read(1): # discard_unknown_payload
|
||||
frame_aligned = False
|
||||
if not has_sample_offset:
|
||||
frame_aligned = bool(reader.read(1))
|
||||
if frame_aligned:
|
||||
reader.skip(2)
|
||||
if has_sample_offset or frame_aligned:
|
||||
reader.skip(7)
|
||||
|
||||
payload_size = variable_bits(reader, 8)
|
||||
if reader.position + payload_size * 8 > body_end:
|
||||
raise EmdfError(
|
||||
f"payload id {payload_id} 越界: size={payload_size}, @bit{reader.position}")
|
||||
payloads[payload_id] = reader.read_bytes(payload_size)
|
||||
sample_offsets[payload_id] = sample_offset
|
||||
|
||||
if not terminated:
|
||||
raise EmdfError("EMDF 容器缺少 payload id 0 终止符")
|
||||
total_bytes = 4 + length
|
||||
raw_reader = BitReader(data, start_bit, start_bit + total_bytes * 8)
|
||||
raw = raw_reader.read_bytes(total_bytes)
|
||||
return EmdfContainer(start_bit, raw, payloads, sample_offsets)
|
||||
|
||||
|
||||
def _marker_offsets(data):
|
||||
"""以 NumPy 批量检查八种位移,返回可能的 0x5838 bit offsets。"""
|
||||
source = np.frombuffer(data, dtype=np.uint8)
|
||||
if source.size < 4:
|
||||
return []
|
||||
offsets = []
|
||||
for shift in range(8):
|
||||
if shift == 0:
|
||||
aligned = source
|
||||
else:
|
||||
aligned = np.bitwise_or(
|
||||
np.left_shift(source[:-1].astype(np.uint16), shift) & 0xFF,
|
||||
np.right_shift(source[1:].astype(np.uint16), 8 - shift),
|
||||
).astype(np.uint8)
|
||||
hits = np.flatnonzero((aligned[:-1] == 0x58) & (aligned[1:] == 0x38))
|
||||
offsets.extend(int(hit) * 8 + shift for hit in hits)
|
||||
return sorted(offsets)
|
||||
|
||||
|
||||
def find_joc_emdf(frame):
|
||||
"""返回同步帧中唯一、顶层连续且包含 ID11/ID14 的 JOC EMDF 容器。
|
||||
|
||||
EMDF payload 是不透明字节串,其中可能自然出现另一个 ``0x5838``。若从这个
|
||||
内嵌 marker 开始的后续随机位恰好也能通过容器语法探测,它仍不是一个独立的
|
||||
transport 容器。因此,候选的起点一旦落在较早 JOC 容器的声明范围内,就只把
|
||||
它记作内嵌伪候选,不参与“多个容器”的判定。
|
||||
"""
|
||||
matches = []
|
||||
offsets = _marker_offsets(frame)
|
||||
parsed_candidates = []
|
||||
parse_errors = []
|
||||
for start_bit in offsets:
|
||||
try:
|
||||
container = _parse_at(frame, start_bit)
|
||||
except EmdfError as exc:
|
||||
if len(parse_errors) < 8:
|
||||
parse_errors.append({"start_bit": start_bit, "error": str(exc)})
|
||||
continue
|
||||
parsed_candidates.append({
|
||||
"start_bit": start_bit,
|
||||
"payload_ids": list(container.payloads),
|
||||
"payload_lengths": {str(k): len(v) for k, v in container.payloads.items()},
|
||||
})
|
||||
if REQUIRED_JOC_IDS.issubset(container.payloads):
|
||||
matches.append(container)
|
||||
if not matches:
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "no_contiguous_joc_container",
|
||||
"同步帧中未找到可连续解析且同时包含 ID11/ID14 的 EMDF 容器",
|
||||
details={
|
||||
"syncframe_bytes": len(frame),
|
||||
"marker_bit_offsets": offsets,
|
||||
"parsed_candidates": parsed_candidates,
|
||||
"candidate_parse_errors": parse_errors,
|
||||
"repair_hint": "检查 EMDF 是否跨多个 audio-block skip field 分片,或 payload config 是否变化",
|
||||
})
|
||||
top_level_matches = []
|
||||
nested_matches = []
|
||||
for container in sorted(matches, key=lambda item: item.start_bit):
|
||||
parent = next((candidate for candidate in top_level_matches
|
||||
if candidate.start_bit < container.start_bit <
|
||||
candidate.start_bit + len(candidate.raw) * 8), None)
|
||||
if parent is None:
|
||||
top_level_matches.append(container)
|
||||
else:
|
||||
nested_matches.append({
|
||||
"start_bit": container.start_bit,
|
||||
"end_bit": container.start_bit + len(container.raw) * 8,
|
||||
"parent_start_bit": parent.start_bit,
|
||||
"parent_end_bit": parent.start_bit + len(parent.raw) * 8,
|
||||
})
|
||||
if len(top_level_matches) != 1:
|
||||
starts = [item.start_bit for item in top_level_matches]
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "multiple_joc_containers",
|
||||
"同步帧中存在多个可用 JOC EMDF,当前无法自动选择",
|
||||
details={
|
||||
"syncframe_bytes": len(frame),
|
||||
"joc_container_start_bits": starts,
|
||||
"nested_joc_candidates": nested_matches,
|
||||
})
|
||||
return top_level_matches[0]
|
||||
|
||||
|
||||
def parse_container(data):
|
||||
"""解析从 syncword 开始、已经重新按字节对齐保存的 EMDF 容器。"""
|
||||
container = _parse_at(data, 0)
|
||||
if len(container.raw) != len(data):
|
||||
raise EmdfError(f"EMDF 文件尾有额外数据: parsed={len(container.raw)}, file={len(data)}")
|
||||
return container
|
||||
|
||||
|
||||
def iter_eac3_frames(data):
|
||||
"""按 E-AC-3 ``frmsiz`` 遍历同步帧,拒绝静默重同步。"""
|
||||
pos = 0
|
||||
while pos < len(data):
|
||||
if pos + 4 > len(data) or data[pos:pos + 2] != b"\x0b\x77":
|
||||
raise UnsupportedVariantError(
|
||||
"eac3_transport", "syncframe_header",
|
||||
"E-AC-3 同步帧头无效或出现了未处理的子流排列",
|
||||
details={
|
||||
"byte_offset": pos,
|
||||
"remaining_bytes": len(data) - pos,
|
||||
"next_16_bytes_hex": data[pos:pos + 16].hex(),
|
||||
})
|
||||
size = ((((data[pos + 2] & 7) << 8) | data[pos + 3]) + 1) * 2
|
||||
if pos + size > len(data):
|
||||
raise UnsupportedVariantError(
|
||||
"eac3_transport", "truncated_syncframe",
|
||||
"E-AC-3 末帧长度超过输入剩余数据",
|
||||
details={
|
||||
"byte_offset": pos,
|
||||
"declared_frame_bytes": size,
|
||||
"remaining_bytes": len(data) - pos,
|
||||
})
|
||||
yield data[pos:pos + size]
|
||||
pos += size
|
||||
|
||||
|
||||
def extract_index(eac3_path, output_dir, max_frames=None):
|
||||
"""将裸 E-AC-3 的连续 EMDF 保存为 ``frames.csv + emdf/``。"""
|
||||
output_dir = Path(output_dir)
|
||||
emdf_dir = output_dir / "emdf"
|
||||
emdf_dir.mkdir(parents=True, exist_ok=True)
|
||||
frames = iter_eac3_frames(Path(eac3_path).read_bytes())
|
||||
rows = []
|
||||
for frame_number, frame in enumerate(frames):
|
||||
if max_frames is not None and frame_number >= max_frames:
|
||||
break
|
||||
try:
|
||||
container = find_joc_emdf(frame)
|
||||
except UnsupportedVariantError as exc:
|
||||
exc.add_context(frame=frame_number, details={"syncframe_bytes": len(frame)})
|
||||
raise
|
||||
except EmdfError as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "container_syntax",
|
||||
"EMDF 容器语法无法解析",
|
||||
frame=frame_number,
|
||||
details={"syncframe_bytes": len(frame), "parser_error": str(exc)}) from exc
|
||||
digest = hashlib.sha256(container.raw).hexdigest()
|
||||
target = emdf_dir / f"{digest}.bin"
|
||||
if not target.is_file():
|
||||
target.write_bytes(container.raw)
|
||||
rows.append({
|
||||
"frame": frame_number,
|
||||
"emdf_hash": digest,
|
||||
"emdf_size": len(container.raw),
|
||||
"emdf_start_bit": container.start_bit,
|
||||
"payload_ids": ";".join(str(x) for x in container.payloads),
|
||||
"error": "",
|
||||
})
|
||||
if (frame_number + 1) % 1000 == 0:
|
||||
print(f"[metadata] {frame_number + 1} frames", flush=True)
|
||||
if not rows:
|
||||
raise EmdfError("E-AC-3 输入中没有可处理的同步帧")
|
||||
with (output_dir / "frames.csv").open("w", encoding="utf-8", newline="") as fp:
|
||||
fields = ("frame", "emdf_hash", "emdf_size", "emdf_start_bit", "payload_ids", "error")
|
||||
writer = csv.DictWriter(fp, fieldnames=fields)
|
||||
writer.writeheader()
|
||||
writer.writerows(rows)
|
||||
return output_dir
|
||||
@@ -1,305 +0,0 @@
|
||||
#include "emdf/emdf_parser.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <string>
|
||||
|
||||
#include "foundation/bit_reader.h"
|
||||
|
||||
namespace joc::emdf {
|
||||
|
||||
namespace {
|
||||
|
||||
Status syntax_fail(const std::string& message) {
|
||||
return Status::fail(JOC_ERR_EMDF_SYNTAX, stage::kEmdf, message);
|
||||
}
|
||||
|
||||
Status truncated_fail(const bits::BitReader& reader) {
|
||||
return Status::fail(JOC_ERR_BITSTREAM_TRUNCATED, stage::kEmdf,
|
||||
std::string("EMDF bitstream truncated: ") + reader.error_message());
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Status parse_at(const std::uint8_t* data, std::size_t size, std::size_t start_bit, Container* out) {
|
||||
if (data == nullptr || out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kEmdf, "null buffer or output");
|
||||
}
|
||||
if (start_bit + 16u > size * 8u) {
|
||||
return syntax_fail("EMDF syncword position beyond buffer");
|
||||
}
|
||||
|
||||
bits::BitReader reader;
|
||||
reader.reset(data, size, start_bit);
|
||||
|
||||
if (reader.read(16) != kSyncword) {
|
||||
return syntax_fail("EMDF syncword mismatch at bit " + std::to_string(start_bit));
|
||||
}
|
||||
const std::uint32_t length = reader.read(16);
|
||||
const std::size_t body_start = reader.position();
|
||||
const std::size_t body_end = body_start + static_cast<std::size_t>(length) * 8u;
|
||||
if (body_end > reader.limit()) {
|
||||
return syntax_fail("EMDF container length " + std::to_string(length) +
|
||||
" exceeds buffer at bit " + std::to_string(start_bit));
|
||||
}
|
||||
reader.set_limit_bits(body_end);
|
||||
|
||||
std::uint32_t version = reader.read(2);
|
||||
if (version == 3u) {
|
||||
std::uint32_t extra = 0;
|
||||
if (!bits::variable_bits(reader, 2, 8, &extra)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
version += extra;
|
||||
}
|
||||
std::uint32_t key_id = reader.read(3);
|
||||
if (key_id == 7u) {
|
||||
std::uint32_t extra = 0;
|
||||
if (!bits::variable_bits(reader, 3, 8, &extra)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
key_id += extra;
|
||||
}
|
||||
if (reader.failed()) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
// TS 103 420 JOC uses version 0 / key_id 0; the strict check also rejects
|
||||
// false 0x5838 markers that happen to sit inside audio data.
|
||||
if (version != 0u || key_id != 0u) {
|
||||
return syntax_fail("unsupported EMDF version/key_id " + std::to_string(version) + "/" +
|
||||
std::to_string(key_id));
|
||||
}
|
||||
|
||||
Container container;
|
||||
container.start_bit = start_bit;
|
||||
bool terminated = false;
|
||||
|
||||
while (reader.position() + 5u <= body_end) {
|
||||
std::uint32_t payload_id = reader.read(5);
|
||||
if (reader.failed()) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
if (payload_id == 0u) {
|
||||
terminated = true;
|
||||
break;
|
||||
}
|
||||
if (payload_id == 0x1Fu) {
|
||||
std::uint32_t extra = 0;
|
||||
if (!bits::variable_bits(reader, 5, 8, &extra)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
payload_id += extra;
|
||||
}
|
||||
for (std::size_t i = 0; i < container.payload_count; ++i) {
|
||||
if (container.payloads[i].id == static_cast<std::uint8_t>(payload_id)) {
|
||||
return syntax_fail("duplicate EMDF payload id " + std::to_string(payload_id));
|
||||
}
|
||||
}
|
||||
if (container.payload_count >= kMaxPayloads) {
|
||||
return syntax_fail("EMDF payload count exceeds " + std::to_string(kMaxPayloads));
|
||||
}
|
||||
|
||||
const std::uint32_t has_sample_offset = reader.read(1);
|
||||
std::uint16_t sample_offset = 0;
|
||||
if (has_sample_offset != 0u) {
|
||||
sample_offset = static_cast<std::uint16_t>(reader.read(12) >> 1);
|
||||
}
|
||||
if (reader.read(1) != 0u) {
|
||||
std::uint32_t ignored = 0;
|
||||
if (!bits::variable_bits(reader, 11, 8, &ignored)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
}
|
||||
if (reader.read(1) != 0u) {
|
||||
std::uint32_t ignored = 0;
|
||||
if (!bits::variable_bits(reader, 2, 8, &ignored)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
}
|
||||
if (reader.read(1) != 0u) {
|
||||
if (!reader.skip(8)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
}
|
||||
if (reader.read(1) == 0u) {
|
||||
bool frame_aligned = false;
|
||||
if (has_sample_offset == 0u) {
|
||||
frame_aligned = reader.read(1) != 0u;
|
||||
if (frame_aligned) {
|
||||
if (!reader.skip(2)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (has_sample_offset != 0u || frame_aligned) {
|
||||
if (!reader.skip(7)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (reader.failed()) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
|
||||
std::uint32_t payload_size = 0;
|
||||
if (!bits::variable_bits(reader, 8, 8, &payload_size)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
const std::size_t payload_bits = static_cast<std::size_t>(payload_size) * 8u;
|
||||
if (reader.position() + payload_bits > body_end) {
|
||||
return syntax_fail("EMDF payload id " + std::to_string(payload_id) +
|
||||
" extends past container body (size " + std::to_string(payload_size) +
|
||||
" at bit " + std::to_string(reader.position()) + ")");
|
||||
}
|
||||
|
||||
Payload& entry = container.payloads[container.payload_count++];
|
||||
entry.id = static_cast<std::uint8_t>(payload_id);
|
||||
entry.sample_offset = sample_offset;
|
||||
entry.bit_offset = reader.position();
|
||||
entry.size = payload_size;
|
||||
|
||||
if (!reader.skip(payload_bits)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
}
|
||||
|
||||
if (!terminated) {
|
||||
return syntax_fail("EMDF container has no payload id 0 terminator");
|
||||
}
|
||||
container.raw_size = 4u + static_cast<std::size_t>(length);
|
||||
*out = container;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
void marker_offsets(const std::uint8_t* data, std::size_t size, std::vector<std::size_t>* out) {
|
||||
out->clear();
|
||||
if (data == nullptr || size < 4u) {
|
||||
return;
|
||||
}
|
||||
// Eight global bit alignments. For shift != 0 the reference builds an
|
||||
// (n-1)-byte shifted view and only scans pairs inside it, which is what the
|
||||
// bounds below reproduce exactly.
|
||||
for (std::size_t shift = 0; shift < 8u; ++shift) {
|
||||
const std::size_t aligned_len = (shift == 0u) ? size : (size - 1u);
|
||||
auto aligned_byte = [&](std::size_t index) -> std::uint8_t {
|
||||
if (shift == 0u) {
|
||||
return data[index];
|
||||
}
|
||||
const std::uint16_t high = static_cast<std::uint16_t>(data[index]) << shift;
|
||||
const std::uint16_t low = static_cast<std::uint16_t>(data[index + 1u]) >> (8u - shift);
|
||||
return static_cast<std::uint8_t>((high | low) & 0xFFu);
|
||||
};
|
||||
if (aligned_len < 2u) {
|
||||
continue;
|
||||
}
|
||||
for (std::size_t i = 0; i + 1u < aligned_len; ++i) {
|
||||
if (aligned_byte(i) == 0x58u && aligned_byte(i + 1u) == 0x38u) {
|
||||
out->push_back(i * 8u + shift);
|
||||
}
|
||||
}
|
||||
}
|
||||
std::sort(out->begin(), out->end());
|
||||
}
|
||||
|
||||
Status find_joc_emdf(const std::uint8_t* data, std::size_t size, Container* out) {
|
||||
if (data == nullptr || out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kEmdf, "null buffer or output");
|
||||
}
|
||||
std::vector<std::size_t> offsets;
|
||||
marker_offsets(data, size, &offsets);
|
||||
|
||||
std::vector<Container> matches;
|
||||
std::size_t parse_errors = 0;
|
||||
std::string first_parse_error;
|
||||
for (const std::size_t start_bit : offsets) {
|
||||
Container candidate;
|
||||
const Status status = parse_at(data, size, start_bit, &candidate);
|
||||
if (!status.ok()) {
|
||||
++parse_errors;
|
||||
if (first_parse_error.empty()) {
|
||||
first_parse_error = "@bit" + std::to_string(start_bit) + ": " + status.message();
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (candidate.find(kIdOamd) != nullptr && candidate.find(kIdJoc) != nullptr) {
|
||||
matches.push_back(candidate);
|
||||
}
|
||||
}
|
||||
|
||||
if (matches.empty()) {
|
||||
// Classification stays at the transport level (identical to the reference
|
||||
// implementation, which raises emdf_transport here), but the underlying
|
||||
std::string message =
|
||||
"no contiguous EMDF container carrying ID11+ID14 in this syncframe (markers=" +
|
||||
std::to_string(offsets.size()) + ", parse_failures=" + std::to_string(parse_errors) +
|
||||
")";
|
||||
if (!first_parse_error.empty()) {
|
||||
message += "; first candidate error " + first_parse_error;
|
||||
}
|
||||
return Status::fail(JOC_ERR_EMDF_TRANSPORT, stage::kEmdf, message);
|
||||
}
|
||||
|
||||
std::sort(matches.begin(), matches.end(),
|
||||
[](const Container& a, const Container& b) { return a.start_bit < b.start_bit; });
|
||||
|
||||
// A payload may contain bytes that look like another 0x5838 container; a
|
||||
std::vector<Container> top_level;
|
||||
for (const Container& candidate : matches) {
|
||||
bool nested = false;
|
||||
for (const Container& parent : top_level) {
|
||||
if (parent.start_bit < candidate.start_bit &&
|
||||
candidate.start_bit < parent.start_bit + parent.raw_size * 8u) {
|
||||
nested = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!nested) {
|
||||
top_level.push_back(candidate);
|
||||
}
|
||||
}
|
||||
|
||||
if (top_level.size() != 1u) {
|
||||
return Status::fail(JOC_ERR_EMDF_TRANSPORT, stage::kEmdf,
|
||||
"multiple top-level JOC EMDF containers (" +
|
||||
std::to_string(top_level.size()) +
|
||||
"); automatic selection is not defined");
|
||||
}
|
||||
*out = top_level.front();
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
void extract_container_bytes(const std::uint8_t* data, std::size_t size, const Container& container,
|
||||
std::vector<std::uint8_t>* out) {
|
||||
out->assign(container.raw_size, 0u);
|
||||
if (out->empty()) {
|
||||
return;
|
||||
}
|
||||
bits::BitReader reader;
|
||||
reader.reset(data, size, container.start_bit);
|
||||
reader.read_bytes(out->data(), out->size());
|
||||
}
|
||||
|
||||
Status extract_payload_bytes(const std::uint8_t* data, std::size_t size, const Payload& payload,
|
||||
std::vector<std::uint8_t>* out) {
|
||||
if (data == nullptr || out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kEmdf, "null buffer or output");
|
||||
}
|
||||
out->assign(payload.size, 0u);
|
||||
if (out->empty()) {
|
||||
return Status::success();
|
||||
}
|
||||
bits::BitReader reader;
|
||||
reader.reset(data, size, payload.bit_offset);
|
||||
if (!reader.read_bytes(out->data(), out->size())) {
|
||||
return Status::fail(JOC_ERR_BITSTREAM_TRUNCATED, stage::kEmdf,
|
||||
"payload bytes extend past the syncframe");
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::emdf
|
||||
@@ -1,57 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
#include "joc_core.h"
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::emdf {
|
||||
|
||||
inline constexpr std::uint16_t kSyncword = 0x5838;
|
||||
inline constexpr std::uint8_t kIdOamd = 11;
|
||||
inline constexpr std::uint8_t kIdJoc = 14;
|
||||
inline constexpr std::size_t kMaxPayloads = JOC_MAX_EMDF_PAYLOADS;
|
||||
|
||||
struct Payload {
|
||||
std::uint8_t id = 0;
|
||||
std::uint16_t sample_offset = 0;
|
||||
std::size_t bit_offset = 0; // MSB-first bit position of the payload bytes
|
||||
std::size_t size = 0; // payload byte count
|
||||
};
|
||||
|
||||
struct Container {
|
||||
std::size_t start_bit = 0;
|
||||
std::size_t raw_size = 0;
|
||||
std::size_t payload_count = 0;
|
||||
Payload payloads[kMaxPayloads] = {};
|
||||
|
||||
const Payload* find(std::uint8_t id) const {
|
||||
for (std::size_t i = 0; i < payload_count; ++i) {
|
||||
if (payloads[i].id == id) {
|
||||
return &payloads[i];
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
};
|
||||
|
||||
Status parse_at(const std::uint8_t* data, std::size_t size, std::size_t start_bit, Container* out);
|
||||
|
||||
// All candidate 0x5838 bit offsets over the eight alignments, ascending.
|
||||
void marker_offsets(const std::uint8_t* data, std::size_t size, std::vector<std::size_t>* out);
|
||||
|
||||
Status find_joc_emdf(const std::uint8_t* data, std::size_t size, Container* out);
|
||||
|
||||
// Extract the container's bytes exactly as the bit reader sees them (identical
|
||||
// to a memcpy for byte-aligned containers).
|
||||
void extract_container_bytes(const std::uint8_t* data, std::size_t size, const Container& container,
|
||||
std::vector<std::uint8_t>* out);
|
||||
|
||||
// Extract one payload's bytes with the same MSB-first semantics.
|
||||
Status extract_payload_bytes(const std::uint8_t* data, std::size_t size, const Payload& payload,
|
||||
std::vector<std::uint8_t>* out);
|
||||
|
||||
} // namespace joc::emdf
|
||||
@@ -0,0 +1,76 @@
|
||||
"""解包 EVO MD-set evolution 载荷,返回各 payload ID 的字节数据和位偏移。"""
|
||||
_MARK = "1001001000000"
|
||||
|
||||
# 各子载荷字段: (id, 头部前缀, 同步标记, 后缀常量, 尺寸域位数, 是否有转义)
|
||||
# 头部前缀 = 5 位 id 的 MSB 二进制(id11 字段前另有 5 位容器前导 00000)
|
||||
# 尺寸域单位 = nibble(4 位)。id14 有转义:9 位值=0 → 再读 9 位 = 字节数。
|
||||
_LAYOUT = [
|
||||
(11, "0000001011", "010000000000000", 8, False),
|
||||
(14, "01110", "01000000000000", 9, True),
|
||||
(2, "00010", "000100", 7, False),
|
||||
(1, "00001", "1110000000000000000000000000", 4, False),
|
||||
(30, "11110", "1110000000000000000000000000", 4, False),
|
||||
]
|
||||
|
||||
|
||||
def _msb_bits(data: bytes):
|
||||
return [(x >> (7 - i)) & 1 for x in data for i in range(8)]
|
||||
|
||||
|
||||
def _val(bits, off, n):
|
||||
v = 0
|
||||
for b in bits[off:off + n]:
|
||||
v = (v << 1) | b
|
||||
return v
|
||||
|
||||
|
||||
class _LooseSkip(Exception):
|
||||
def __init__(self, ident):
|
||||
self.ident = ident
|
||||
|
||||
|
||||
def unpack_evolution(payload: bytes, loose=False):
|
||||
"""解包 evolution 载荷 → (subs, offsets)。subs 键为 id 整数。
|
||||
loose=True 时对每个 id 的 (前缀+标记+后缀) 全模式做位流重同步扫描
|
||||
(不同编码流的子载荷次序/内部常量可有合法差异,如 kanata 的 id11)。"""
|
||||
bits = _msb_bits(payload)
|
||||
pos = 0
|
||||
subs = {}
|
||||
offsets = {}
|
||||
for ident, pref, suff, sbits, escape in _LAYOUT:
|
||||
pat = pref + _MARK + suff
|
||||
if loose:
|
||||
hit = -1
|
||||
for i in range(pos, len(bits) - len(pat)):
|
||||
if ''.join(map(str, bits[i:i + len(pat)])) == pat:
|
||||
hit = i
|
||||
break
|
||||
if hit < 0:
|
||||
continue
|
||||
pos = hit + len(pat)
|
||||
else:
|
||||
for name, const, expect in (("前缀", bits[pos:pos + len(pref)], pref),
|
||||
("标记", bits[pos + len(pref):pos + len(pref) + len(_MARK)], _MARK),
|
||||
("后缀", bits[pos + len(pref) + len(_MARK):
|
||||
pos + len(pref) + len(_MARK) + len(suff)], suff)):
|
||||
got = ''.join(map(str, const))
|
||||
if got != expect:
|
||||
raise ValueError(f"id={ident}: {name}常量不匹配 @bit{pos} got={got} want={expect}")
|
||||
pos += len(pref) + len(_MARK) + len(suff)
|
||||
n_nib = _val(bits, pos, sbits)
|
||||
pos += sbits
|
||||
if escape and n_nib == 1:
|
||||
n_nib = 512 + _val(bits, pos, sbits)
|
||||
pos += sbits
|
||||
n_bits = n_nib * 4
|
||||
body = bits[pos:pos + n_bits]
|
||||
pos += n_bits
|
||||
b = bytearray(len(body) // 8)
|
||||
for i in range(0, len(body) // 8 * 8, 8):
|
||||
v = 0
|
||||
for x in body[i:i + 8]:
|
||||
v = (v << 1) | x
|
||||
b[i // 8] = v
|
||||
subs[ident] = bytes(b)
|
||||
offsets[ident] = pos
|
||||
return subs, offsets
|
||||
@@ -1,30 +0,0 @@
|
||||
#include "foundation/bit_reader.h"
|
||||
|
||||
namespace joc::bits {
|
||||
|
||||
bool variable_bits(BitReader& reader, unsigned width, unsigned max_groups, std::uint32_t* out_value) {
|
||||
std::uint32_t value = 0;
|
||||
for (unsigned group = 0; group < max_groups; ++group) {
|
||||
value += reader.read(width);
|
||||
if (reader.failed()) {
|
||||
return false;
|
||||
}
|
||||
const std::uint32_t more = reader.read(1);
|
||||
if (reader.failed()) {
|
||||
return false;
|
||||
}
|
||||
if (more == 0u) {
|
||||
if (out_value != nullptr) {
|
||||
*out_value = value;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
value = (value + 1u) << width;
|
||||
}
|
||||
// Same failure mode as the reference implementation: an extension chain
|
||||
// that never terminates is a syntax error, not a truncation.
|
||||
reader.fail(JOC_ERR_EMDF_SYNTAX, "variable_bits extension groups exceeded");
|
||||
return false;
|
||||
}
|
||||
|
||||
} // namespace joc::bits
|
||||
@@ -1,125 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc::bits {
|
||||
|
||||
class BitReader {
|
||||
public:
|
||||
BitReader() = default;
|
||||
BitReader(const std::uint8_t* data, std::size_t size) { reset(data, size); }
|
||||
|
||||
void reset(const std::uint8_t* data, std::size_t size, std::size_t start_bit = 0) {
|
||||
data_ = data;
|
||||
size_bits_ = size * 8u;
|
||||
pos_ = start_bit;
|
||||
limit_ = size_bits_;
|
||||
error_ = JOC_OK;
|
||||
message_ = "";
|
||||
}
|
||||
|
||||
void set_limit_bits(std::size_t limit_bits) {
|
||||
limit_ = limit_bits < size_bits_ ? limit_bits : size_bits_;
|
||||
}
|
||||
|
||||
std::size_t position() const { return pos_; }
|
||||
std::size_t limit() const { return limit_; }
|
||||
std::size_t remaining_bits() const { return pos_ <= limit_ ? limit_ - pos_ : 0; }
|
||||
const std::uint8_t* data() const { return data_; }
|
||||
|
||||
bool failed() const { return error_ != JOC_OK; }
|
||||
joc_error error() const { return error_; }
|
||||
const char* error_message() const { return message_; }
|
||||
|
||||
std::uint32_t read(unsigned count) {
|
||||
if (count == 0) {
|
||||
return 0;
|
||||
}
|
||||
if (!can_read(count)) {
|
||||
fail_truncated(count);
|
||||
return 0;
|
||||
}
|
||||
std::uint32_t value = 0;
|
||||
if ((pos_ & 7u) == 0u && count >= 8u) {
|
||||
while (count >= 8u) {
|
||||
value = (value << 8) | data_[pos_ >> 3];
|
||||
pos_ += 8u;
|
||||
count -= 8u;
|
||||
}
|
||||
}
|
||||
while (count-- > 0u) {
|
||||
const std::uint32_t bit = (data_[pos_ >> 3] >> (7u - (pos_ & 7u))) & 1u;
|
||||
value = (value << 1) | bit;
|
||||
++pos_;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
std::uint64_t read64(unsigned count) {
|
||||
if (count <= 32u) {
|
||||
return static_cast<std::uint64_t>(read(count));
|
||||
}
|
||||
const std::uint64_t high = static_cast<std::uint64_t>(read(count - 32u));
|
||||
const std::uint64_t low = static_cast<std::uint64_t>(read(32u));
|
||||
return (high << 32) | low;
|
||||
}
|
||||
|
||||
bool skip(std::size_t count) {
|
||||
if (!can_read(count)) {
|
||||
fail_truncated(count);
|
||||
return false;
|
||||
}
|
||||
pos_ += count;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool read_bytes(std::uint8_t* out, std::size_t count) {
|
||||
if (count == 0) {
|
||||
return true;
|
||||
}
|
||||
if (!can_read(count * 8u)) {
|
||||
fail_truncated(count * 8u);
|
||||
return false;
|
||||
}
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
out[i] = static_cast<std::uint8_t>(read(8u));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool can_read(std::size_t count) const {
|
||||
return !failed() && count <= limit_ && pos_ <= limit_ - count;
|
||||
}
|
||||
|
||||
// semantic check fails, so the reader never continues past it).
|
||||
void fail(joc_error code, const char* message) {
|
||||
if (!failed()) {
|
||||
error_ = code;
|
||||
message_ = message;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
void fail_truncated(std::size_t count) {
|
||||
fail(JOC_ERR_BITSTREAM_TRUNCATED, "bit read past end of buffer");
|
||||
last_request_ = count;
|
||||
}
|
||||
|
||||
const std::uint8_t* data_ = nullptr;
|
||||
std::size_t size_bits_ = 0;
|
||||
std::size_t pos_ = 0;
|
||||
std::size_t limit_ = 0;
|
||||
std::size_t last_request_ = 0;
|
||||
joc_error error_ = JOC_OK;
|
||||
const char* message_ = "";
|
||||
};
|
||||
|
||||
// followed by a continuation bit. Mirrors src/emdf.py:variable_bits().
|
||||
bool variable_bits(BitReader& reader, unsigned width, unsigned max_groups,
|
||||
std::uint32_t* out_value);
|
||||
|
||||
} // namespace joc::bits
|
||||
@@ -1,206 +0,0 @@
|
||||
#include "foundation/fft.h"
|
||||
|
||||
#include <cmath>
|
||||
|
||||
#include "simd/simd.h"
|
||||
|
||||
namespace joc::dsp {
|
||||
|
||||
// The dispatched kernels read and write the spectrum as interleaved doubles, and
|
||||
// an array of std::complex<double> is exactly that: two doubles per element, no
|
||||
// padding, no vtable.
|
||||
static_assert(sizeof(Complex) == 2u * sizeof(double), "complex layout");
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr double kPi = 3.14159265358979323846;
|
||||
|
||||
template <typename Container>
|
||||
void fft_in_place(Container* data, bool inverse) {
|
||||
const std::size_t count = data->size();
|
||||
if (count < 2u) {
|
||||
return;
|
||||
}
|
||||
for (std::size_t index = 1u, reversed = 0u; index < count; ++index) {
|
||||
std::size_t bit = count >> 1u;
|
||||
for (; (reversed & bit) != 0u; bit >>= 1u) {
|
||||
reversed ^= bit;
|
||||
}
|
||||
reversed ^= bit;
|
||||
if (index < reversed) {
|
||||
std::swap((*data)[index], (*data)[reversed]);
|
||||
}
|
||||
}
|
||||
for (std::size_t length = 2u; length <= count; length <<= 1u) {
|
||||
const double angle = (inverse ? 2.0 : -2.0) * kPi / static_cast<double>(length);
|
||||
const Complex step(std::cos(angle), std::sin(angle));
|
||||
for (std::size_t start = 0u; start < count; start += length) {
|
||||
Complex factor(1.0, 0.0);
|
||||
for (std::size_t offset = 0u; offset < length / 2u; ++offset) {
|
||||
const Complex even = (*data)[start + offset];
|
||||
const Complex odd = (*data)[start + offset + length / 2u] * factor;
|
||||
(*data)[start + offset] = even + odd;
|
||||
(*data)[start + offset + length / 2u] = even - odd;
|
||||
factor *= step;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (inverse) {
|
||||
for (Complex& value : *data) {
|
||||
value /= static_cast<double>(count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool is_power_of_two(std::size_t value) { return value != 0u && (value & (value - 1u)) == 0u; }
|
||||
|
||||
} // namespace
|
||||
|
||||
FftPlan::FftPlan(std::size_t size, bool inverse) : size_(size), inverse_(inverse) {
|
||||
reverse_.resize(size);
|
||||
for (std::size_t index = 1u, reversed = 0u; index < size; ++index) {
|
||||
std::size_t bit = size >> 1u;
|
||||
for (; (reversed & bit) != 0u; bit >>= 1u) {
|
||||
reversed ^= bit;
|
||||
}
|
||||
reversed ^= bit;
|
||||
reverse_[index] = reversed;
|
||||
}
|
||||
for (std::size_t length = 2u; length <= size; length <<= 1u) {
|
||||
const double angle = (inverse ? 2.0 : -2.0) * kPi / static_cast<double>(length);
|
||||
const Complex step(std::cos(angle), std::sin(angle));
|
||||
stage_begin_.push_back(twiddle_.size());
|
||||
Complex factor(1.0, 0.0);
|
||||
for (std::size_t offset = 0u; offset < length / 2u; ++offset) {
|
||||
twiddle_.push_back(factor);
|
||||
factor *= step;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Exactly the operations fft_in_place performs, in the same order, with the
|
||||
// twiddles read from the precomputed recurrence instead of being re-derived.
|
||||
template <typename Container>
|
||||
void FftPlan::apply(Container* data) const {
|
||||
const std::size_t count = data->size();
|
||||
if (count < 2u) {
|
||||
return;
|
||||
}
|
||||
const std::size_t* reverse = reverse_.data();
|
||||
for (std::size_t index = 1u; index < count; ++index) {
|
||||
const std::size_t reversed = reverse[index];
|
||||
if (index < reversed) {
|
||||
std::swap((*data)[index], (*data)[reversed]);
|
||||
}
|
||||
}
|
||||
// The cascade is dispatched for every power-of-two size the kernels can pack
|
||||
// whole groups into a vector (JOC_SIMD pins one tier for verification). A
|
||||
// kernel only ever puts independent butterflies in the same vector, so every
|
||||
// output keeps the operation sequence and the roundings written below; small
|
||||
// transforms -- and the caller's own table -- keep the portable loop.
|
||||
if (count >= simd::kMinVectorFftSize && (count & (count - 1u)) == 0u) {
|
||||
simd::fft_butterflies(reinterpret_cast<double*>(data->data()), count,
|
||||
reinterpret_cast<const double*>(twiddle_.data()),
|
||||
stage_begin_.data());
|
||||
} else {
|
||||
std::size_t stage = 0u;
|
||||
for (std::size_t length = 2u; length <= count; length <<= 1u, ++stage) {
|
||||
const Complex* table = twiddle_.data() + stage_begin_[stage];
|
||||
for (std::size_t start = 0u; start < count; start += length) {
|
||||
for (std::size_t offset = 0u; offset < length / 2u; ++offset) {
|
||||
const Complex even = (*data)[start + offset];
|
||||
const Complex odd = (*data)[start + offset + length / 2u] * table[offset];
|
||||
(*data)[start + offset] = even + odd;
|
||||
(*data)[start + offset + length / 2u] = even - odd;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (inverse_) {
|
||||
for (Complex& value : *data) {
|
||||
value /= static_cast<double>(count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void fft_radix2(std::vector<Complex>* data, const FftPlan& plan) { plan.apply(data); }
|
||||
|
||||
void fft_radix2(std::array<Complex, kQmfFftSize>* data, const FftPlan& plan) { plan.apply(data); }
|
||||
|
||||
void fft_radix2(std::vector<Complex>* data, bool inverse) { fft_in_place(data, inverse); }
|
||||
|
||||
void fft_radix2(std::array<Complex, kQmfFftSize>* data, bool inverse) {
|
||||
fft_in_place(data, inverse);
|
||||
}
|
||||
|
||||
void fft_any(const std::vector<Complex>& input, bool inverse, std::vector<Complex>* output) {
|
||||
const std::size_t count = input.size();
|
||||
if (is_power_of_two(count)) {
|
||||
*output = input;
|
||||
fft_radix2(output, inverse);
|
||||
return;
|
||||
}
|
||||
std::size_t size = 1u;
|
||||
while (size < 2u * count + 1u) {
|
||||
size <<= 1u;
|
||||
}
|
||||
const double sign = inverse ? 1.0 : -1.0;
|
||||
std::vector<Complex> left(size, Complex(0.0, 0.0));
|
||||
std::vector<Complex> right(size, Complex(0.0, 0.0));
|
||||
for (std::size_t index = 0u; index < count; ++index) {
|
||||
const std::size_t wrapped = (index * index) % (2u * count);
|
||||
const double angle = kPi * static_cast<double>(wrapped) / static_cast<double>(count);
|
||||
const Complex chirp(std::cos(angle), sign * std::sin(angle));
|
||||
left[index] = input[index] * chirp;
|
||||
right[index] = std::conj(chirp);
|
||||
if (index != 0u) {
|
||||
right[size - index] = std::conj(chirp);
|
||||
}
|
||||
}
|
||||
fft_radix2(&left, false);
|
||||
fft_radix2(&right, false);
|
||||
for (std::size_t index = 0u; index < size; ++index) {
|
||||
left[index] *= right[index];
|
||||
}
|
||||
fft_radix2(&left, true);
|
||||
output->resize(count);
|
||||
for (std::size_t index = 0u; index < count; ++index) {
|
||||
const std::size_t wrapped = (index * index) % (2u * count);
|
||||
const double angle = kPi * static_cast<double>(wrapped) / static_cast<double>(count);
|
||||
const Complex chirp(std::cos(angle), sign * std::sin(angle));
|
||||
(*output)[index] = left[index] * chirp;
|
||||
if (inverse) {
|
||||
(*output)[index] /= static_cast<double>(count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::size_t next_fast_len(std::size_t value) {
|
||||
if (value <= 6u) {
|
||||
return value;
|
||||
}
|
||||
std::size_t best = value;
|
||||
for (std::size_t power2 = 1u; power2 < value * 2u; power2 *= 2u) {
|
||||
for (std::size_t power3 = power2; power3 < value * 2u; power3 *= 3u) {
|
||||
std::size_t power5 = power3;
|
||||
while (power5 < value) {
|
||||
power5 *= 5u;
|
||||
}
|
||||
best = std::min(best, power5);
|
||||
if (power3 >= value) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
std::size_t next_power_of_two(std::size_t value) {
|
||||
std::size_t result = 1u;
|
||||
while (result < value) {
|
||||
result <<= 1u;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
} // namespace joc::dsp
|
||||
@@ -1,68 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <complex>
|
||||
#include <cstddef>
|
||||
#include <vector>
|
||||
|
||||
// Complex transforms shared by the HRTF and Rosella DSP cores. The convention is
|
||||
// NumPy's: the forward transform is unnormalised and the inverse scales by 1/N,
|
||||
// so a ported pipeline keeps the reference's arithmetic bit for bit.
|
||||
namespace joc::dsp {
|
||||
|
||||
using Complex = std::complex<double>;
|
||||
|
||||
inline constexpr std::size_t kQmfFftSize = 128;
|
||||
|
||||
// Precomputed radix-2 plan for one size and direction.
|
||||
//
|
||||
// The transform derives each butterfly's twiddle by multiplying the previous one
|
||||
// by the stage step, so the twiddle at offset k is `step` multiplied k times in
|
||||
// that order, independently of the group. Materialising that exact recurrence --
|
||||
// and the bit-reversal permutation -- removes one complex multiply and a
|
||||
// (length/2)-deep serial dependency from every stage's inner loop. The table
|
||||
// entries are the recurrence's own values, so the transform is bit-identical.
|
||||
//
|
||||
// The 128-point cascade is executed by the runtime-dispatched SIMD kernel
|
||||
// (src/simd/simd.h): it computes independent butterflies in parallel lanes,
|
||||
// which leaves both the table and every output's summation order untouched.
|
||||
class FftPlan {
|
||||
public:
|
||||
FftPlan(std::size_t size, bool inverse);
|
||||
|
||||
std::size_t size() const { return size_; }
|
||||
bool inverse() const { return inverse_; }
|
||||
|
||||
private:
|
||||
template <typename Container>
|
||||
void apply(Container* data) const;
|
||||
|
||||
friend void fft_radix2(std::vector<Complex>* data, const FftPlan& plan);
|
||||
friend void fft_radix2(std::array<Complex, kQmfFftSize>* data, const FftPlan& plan);
|
||||
|
||||
std::size_t size_ = 0;
|
||||
bool inverse_ = false;
|
||||
std::vector<std::size_t> reverse_; // bit-reversal permutation, [size]
|
||||
std::vector<std::size_t> stage_begin_; // twiddle offset of each stage
|
||||
std::vector<Complex> twiddle_; // per stage, length/2 entries, concatenated
|
||||
};
|
||||
|
||||
// In-place radix-2 transform; the size must be a power of two.
|
||||
void fft_radix2(std::vector<Complex>* data, bool inverse);
|
||||
void fft_radix2(std::array<Complex, kQmfFftSize>* data, bool inverse);
|
||||
|
||||
// Plan-driven forms: the plan carries the size and the direction, so a caller that
|
||||
// transforms the same length repeatedly builds it once.
|
||||
void fft_radix2(std::vector<Complex>* data, const FftPlan& plan);
|
||||
void fft_radix2(std::array<Complex, kQmfFftSize>* data, const FftPlan& plan);
|
||||
|
||||
// Exact-length transform: radix-2 when the size allows it, Bluestein otherwise.
|
||||
// scipy/numpy use a mixed-radix transform, which is the same transform.
|
||||
void fft_any(const std::vector<Complex>& input, bool inverse, std::vector<Complex>* output);
|
||||
|
||||
// scipy's next_fast_len: the smallest 5-smooth number that is not smaller.
|
||||
std::size_t next_fast_len(std::size_t value);
|
||||
|
||||
std::size_t next_power_of_two(std::size_t value);
|
||||
|
||||
} // namespace joc::dsp
|
||||
@@ -1,164 +0,0 @@
|
||||
#include "foundation/fs_utf8.h"
|
||||
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <vector>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#define NOMINMAX
|
||||
#include <windows.h>
|
||||
#include <shellapi.h>
|
||||
#include <fcntl.h>
|
||||
#include <io.h>
|
||||
#else
|
||||
#include <cstdlib>
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
namespace joc::fs_utf8 {
|
||||
|
||||
namespace fs = std::filesystem;
|
||||
|
||||
fs::path to_path(const std::string& utf8) {
|
||||
return fs::path(std::u8string(reinterpret_cast<const char8_t*>(utf8.data()), utf8.size()));
|
||||
}
|
||||
|
||||
std::string from_path(const fs::path& path) {
|
||||
const std::u8string text = path.u8string();
|
||||
return std::string(reinterpret_cast<const char*>(text.data()), text.size());
|
||||
}
|
||||
|
||||
std::FILE* fopen(const std::string& utf8_path, const char* mode) {
|
||||
#if defined(_WIN32)
|
||||
const std::wstring wide_mode(mode, mode + std::strlen(mode));
|
||||
return ::_wfopen(to_path(utf8_path).c_str(), wide_mode.c_str());
|
||||
#else
|
||||
return std::fopen(utf8_path.c_str(), mode);
|
||||
#endif
|
||||
}
|
||||
|
||||
std::FILE* fopen_spool(const std::string& utf8_path) {
|
||||
#if defined(_WIN32)
|
||||
// Delete-on-close handed to the CRT: if the process is killed the file goes with
|
||||
// it, which is what stops an aborted run from leaving hundreds of gigabytes.
|
||||
HANDLE handle = ::CreateFileW(
|
||||
to_path(utf8_path).c_str(), GENERIC_READ | GENERIC_WRITE | DELETE,
|
||||
FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, nullptr, CREATE_ALWAYS,
|
||||
FILE_ATTRIBUTE_TEMPORARY | FILE_FLAG_DELETE_ON_CLOSE, nullptr);
|
||||
if (handle == INVALID_HANDLE_VALUE) {
|
||||
return nullptr;
|
||||
}
|
||||
const int descriptor = ::_open_osfhandle(reinterpret_cast<std::intptr_t>(handle), 0);
|
||||
if (descriptor == -1) {
|
||||
::CloseHandle(handle);
|
||||
return nullptr;
|
||||
}
|
||||
return ::_fdopen(descriptor, "wb+");
|
||||
#else
|
||||
return std::fopen(utf8_path.c_str(), "wb+");
|
||||
#endif
|
||||
}
|
||||
|
||||
int remove(const std::string& utf8_path) {
|
||||
#if defined(_WIN32)
|
||||
return ::_wremove(to_path(utf8_path).c_str());
|
||||
#else
|
||||
return std::remove(utf8_path.c_str());
|
||||
#endif
|
||||
}
|
||||
|
||||
bool exists(const std::string& utf8_path) {
|
||||
std::error_code error;
|
||||
return fs::exists(to_path(utf8_path), error);
|
||||
}
|
||||
|
||||
bool is_directory(const std::string& utf8_path) {
|
||||
std::error_code error;
|
||||
return fs::is_directory(to_path(utf8_path), error);
|
||||
}
|
||||
|
||||
std::uintmax_t file_size(const std::string& utf8_path, std::error_code& error) {
|
||||
return fs::file_size(to_path(utf8_path), error);
|
||||
}
|
||||
|
||||
std::string temp_directory() {
|
||||
std::error_code error;
|
||||
const fs::path directory = fs::temp_directory_path(error);
|
||||
return error ? std::string(".") : from_path(directory);
|
||||
}
|
||||
|
||||
std::string executable_path() {
|
||||
#if defined(_WIN32)
|
||||
std::vector<wchar_t> buffer(MAX_PATH);
|
||||
while (true) {
|
||||
const DWORD written =
|
||||
::GetModuleFileNameW(nullptr, buffer.data(), static_cast<DWORD>(buffer.size()));
|
||||
if (written == 0) {
|
||||
return std::string();
|
||||
}
|
||||
if (written < buffer.size()) {
|
||||
return from_path(fs::path(std::wstring(buffer.data(), written)));
|
||||
}
|
||||
buffer.resize(buffer.size() * 2u);
|
||||
}
|
||||
#elif defined(__linux__)
|
||||
std::vector<char> buffer(4096u, '\0');
|
||||
const ssize_t written = ::readlink("/proc/self/exe", buffer.data(), buffer.size() - 1u);
|
||||
return written > 0 ? std::string(buffer.data(), static_cast<std::size_t>(written))
|
||||
: std::string();
|
||||
#else
|
||||
return std::string();
|
||||
#endif
|
||||
}
|
||||
|
||||
std::ifstream open_input(const std::string& utf8_path) {
|
||||
return std::ifstream(to_path(utf8_path), std::ios::binary);
|
||||
}
|
||||
|
||||
std::ofstream open_output(const std::string& utf8_path) {
|
||||
return std::ofstream(to_path(utf8_path), std::ios::binary);
|
||||
}
|
||||
|
||||
std::vector<std::string> command_line_arguments(int argc, char** argv) {
|
||||
#if defined(_WIN32)
|
||||
(void)argc;
|
||||
(void)argv;
|
||||
int count = 0;
|
||||
LPWSTR* wide = ::CommandLineToArgvW(::GetCommandLineW(), &count);
|
||||
std::vector<std::string> arguments;
|
||||
if (wide == nullptr) {
|
||||
return arguments;
|
||||
}
|
||||
arguments.reserve(static_cast<std::size_t>(count));
|
||||
for (int index = 0; index < count; ++index) {
|
||||
const std::wstring_view text(wide[index]);
|
||||
const int size = ::WideCharToMultiByte(CP_UTF8, 0, text.data(),
|
||||
static_cast<int>(text.size()), nullptr, 0, nullptr,
|
||||
nullptr);
|
||||
std::string utf8(static_cast<std::size_t>(size), '\0');
|
||||
if (size > 0) {
|
||||
::WideCharToMultiByte(CP_UTF8, 0, text.data(), static_cast<int>(text.size()),
|
||||
utf8.data(), size, nullptr, nullptr);
|
||||
}
|
||||
arguments.push_back(std::move(utf8));
|
||||
}
|
||||
::LocalFree(wide);
|
||||
return arguments;
|
||||
#else
|
||||
std::vector<std::string> arguments;
|
||||
arguments.reserve(static_cast<std::size_t>(argc));
|
||||
for (int index = 0; index < argc; ++index) {
|
||||
arguments.emplace_back(argv[index]);
|
||||
}
|
||||
return arguments;
|
||||
#endif
|
||||
}
|
||||
|
||||
void configure_console() {
|
||||
#if defined(_WIN32)
|
||||
::SetConsoleOutputCP(CP_UTF8);
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace joc::fs_utf8
|
||||
@@ -1,47 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <string>
|
||||
#include <system_error>
|
||||
#include <vector>
|
||||
|
||||
// Paths inside this library are always UTF-8, on every platform. std::filesystem
|
||||
// stores UTF-16 on Windows and bytes elsewhere, and the narrow CRT uses the ANSI
|
||||
// code page on Windows, so every path crosses into the OS through this shim: that
|
||||
// is what makes non-ASCII names (Japanese, Chinese, ...) work.
|
||||
namespace joc::fs_utf8 {
|
||||
|
||||
std::filesystem::path to_path(const std::string& utf8);
|
||||
std::string from_path(const std::filesystem::path& path);
|
||||
|
||||
// File handles and queries take a UTF-8 path: _wfopen on Windows, plain calls
|
||||
// elsewhere. Nothing else in the library may call the narrow CRT with a path.
|
||||
std::FILE* fopen(const std::string& utf8_path, const char* mode);
|
||||
// Temporary spool handle: on Windows the file is opened delete-on-close, so killing
|
||||
// the process removes it instead of leaving a multi-gigabyte leftover behind.
|
||||
std::FILE* fopen_spool(const std::string& utf8_path);
|
||||
int remove(const std::string& utf8_path);
|
||||
bool exists(const std::string& utf8_path);
|
||||
bool is_directory(const std::string& utf8_path);
|
||||
std::uintmax_t file_size(const std::string& utf8_path, std::error_code& error);
|
||||
std::string temp_directory();
|
||||
|
||||
// The running executable's own path, UTF-8, or empty when the platform cannot
|
||||
// report it. Defaults are anchored here so they never depend on the CWD.
|
||||
std::string executable_path();
|
||||
|
||||
// Streams: std::ifstream/ofstream accept a std::filesystem::path, which is the
|
||||
// portable way to open a UTF-8 path.
|
||||
std::ifstream open_input(const std::string& utf8_path);
|
||||
std::ofstream open_output(const std::string& utf8_path);
|
||||
|
||||
// Command line arguments as UTF-8. Windows hands the process UTF-16 and the
|
||||
// narrow CRT would convert it through the ANSI code page, so the wide command
|
||||
// line is re-parsed there; on POSIX argv is already bytes in the user's locale.
|
||||
std::vector<std::string> command_line_arguments(int argc, char** argv);
|
||||
void configure_console();
|
||||
|
||||
} // namespace joc::fs_utf8
|
||||
@@ -1,25 +0,0 @@
|
||||
// Port of the reference's adm_atmos.q_to_adm_xyz: OAMD Q15 coordinates to the ADM
|
||||
// cartesian triple. It lives in foundation because both the ADM writer and the
|
||||
// object position timeline need it, and the timeline must not depend on output.
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
#include "foundation/py_num.h"
|
||||
|
||||
namespace joc::geometry {
|
||||
|
||||
inline void q_to_adm_xyz(int q1, int q2, int q3, double* x, double* y, double* z) {
|
||||
const double posX = std::min(1.0, static_cast<double>(pynum::py_round(
|
||||
static_cast<double>(q1) * 62.0 / 32767.0)) / 62.0);
|
||||
const double posY = std::min(1.0, static_cast<double>(pynum::py_round(
|
||||
static_cast<double>(q2) * 62.0 / 32767.0)) / 62.0);
|
||||
double posZ = static_cast<double>(pynum::py_round(
|
||||
static_cast<double>(q3) * 15.0 / 32767.0)) / 15.0;
|
||||
posZ = std::max(-1.0, std::min(1.0, posZ));
|
||||
*x = posX * 2.0 - 1.0;
|
||||
*y = 1.0 - posY * 2.0;
|
||||
*z = posZ;
|
||||
}
|
||||
|
||||
} // namespace joc::geometry
|
||||
@@ -1,227 +0,0 @@
|
||||
#include "foundation/mini_json.h"
|
||||
|
||||
#include <cmath>
|
||||
#include <cstdlib>
|
||||
|
||||
namespace joc::json {
|
||||
|
||||
namespace {
|
||||
|
||||
void skip_space(const std::string& text, std::size_t* index) {
|
||||
while (*index < text.size() &&
|
||||
(text[*index] == ' ' || text[*index] == '\t' || text[*index] == '\n' ||
|
||||
text[*index] == '\r')) {
|
||||
++(*index);
|
||||
}
|
||||
}
|
||||
|
||||
bool read_string(const std::string& text, std::size_t* index, std::string* out) {
|
||||
if (*index >= text.size() || text[*index] != '"') {
|
||||
return false;
|
||||
}
|
||||
++(*index);
|
||||
out->clear();
|
||||
while (*index < text.size()) {
|
||||
const char c = text[*index];
|
||||
if (c == '\\') {
|
||||
if (*index + 1 >= text.size()) {
|
||||
return false;
|
||||
}
|
||||
const char escape = text[*index + 1];
|
||||
*index += 2;
|
||||
switch (escape) {
|
||||
case '"': out->push_back('"'); break;
|
||||
case '\\': out->push_back('\\'); break;
|
||||
case '/': out->push_back('/'); break;
|
||||
case 'b': out->push_back('\b'); break;
|
||||
case 'f': out->push_back('\f'); break;
|
||||
case 'n': out->push_back('\n'); break;
|
||||
case 'r': out->push_back('\r'); break;
|
||||
case 't': out->push_back('\t'); break;
|
||||
case 'u': {
|
||||
if (*index + 4 > text.size()) {
|
||||
return false;
|
||||
}
|
||||
unsigned code = 0;
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
const char digit = text[*index + static_cast<std::size_t>(i)];
|
||||
code <<= 4;
|
||||
if (digit >= '0' && digit <= '9') { code |= static_cast<unsigned>(digit - '0'); }
|
||||
else if (digit >= 'a' && digit <= 'f') { code |= static_cast<unsigned>(digit - 'a' + 10); }
|
||||
else if (digit >= 'A' && digit <= 'F') { code |= static_cast<unsigned>(digit - 'A' + 10); }
|
||||
else { return false; }
|
||||
}
|
||||
*index += 4;
|
||||
if (code < 0x80u) {
|
||||
out->push_back(static_cast<char>(code));
|
||||
} else if (code < 0x800u) {
|
||||
out->push_back(static_cast<char>(0xC0u | (code >> 6)));
|
||||
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
|
||||
} else {
|
||||
out->push_back(static_cast<char>(0xE0u | (code >> 12)));
|
||||
out->push_back(static_cast<char>(0x80u | ((code >> 6) & 0x3Fu)));
|
||||
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (c == '"') {
|
||||
++(*index);
|
||||
return true;
|
||||
}
|
||||
out->push_back(c);
|
||||
++(*index);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool read_compound(const std::string& text, std::size_t* index, std::string* out) {
|
||||
const char open = text[*index];
|
||||
const char close = open == '{' ? '}' : ']';
|
||||
int depth = 0;
|
||||
const std::size_t start = *index;
|
||||
while (*index < text.size()) {
|
||||
const char c = text[*index];
|
||||
if (c == '"') {
|
||||
std::string ignored;
|
||||
if (!read_string(text, index, &ignored)) {
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (c == open) {
|
||||
++depth;
|
||||
} else if (c == close) {
|
||||
--depth;
|
||||
if (depth == 0) {
|
||||
++(*index);
|
||||
*out = text.substr(start, *index - start);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
++(*index);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool parse_object(const std::string& text, std::vector<Member>* out, std::string* error) {
|
||||
out->clear();
|
||||
std::size_t index = 0;
|
||||
skip_space(text, &index);
|
||||
if (index >= text.size() || text[index] != '{') {
|
||||
if (error != nullptr) { *error = "metadata is not a JSON object"; }
|
||||
return false;
|
||||
}
|
||||
++index;
|
||||
for (;;) {
|
||||
skip_space(text, &index);
|
||||
if (index < text.size() && text[index] == '}') {
|
||||
++index;
|
||||
break;
|
||||
}
|
||||
if (index >= text.size() || text[index] == ',') {
|
||||
if (index >= text.size()) {
|
||||
if (error != nullptr) { *error = "unterminated JSON object"; }
|
||||
return false;
|
||||
}
|
||||
++index;
|
||||
continue;
|
||||
}
|
||||
Member member;
|
||||
if (!read_string(text, &index, &member.key)) {
|
||||
if (error != nullptr) { *error = "expected a JSON key"; }
|
||||
return false;
|
||||
}
|
||||
skip_space(text, &index);
|
||||
if (index >= text.size() || text[index] != ':') {
|
||||
if (error != nullptr) { *error = "expected ':' after JSON key " + member.key; }
|
||||
return false;
|
||||
}
|
||||
++index;
|
||||
skip_space(text, &index);
|
||||
if (index >= text.size()) {
|
||||
if (error != nullptr) { *error = "missing JSON value for " + member.key; }
|
||||
return false;
|
||||
}
|
||||
if (text[index] == '"') {
|
||||
member.is_string = true;
|
||||
if (!read_string(text, &index, &member.raw)) {
|
||||
if (error != nullptr) { *error = "bad JSON string for " + member.key; }
|
||||
return false;
|
||||
}
|
||||
} else if (text[index] == '{' || text[index] == '[') {
|
||||
if (!read_compound(text, &index, &member.raw)) {
|
||||
if (error != nullptr) { *error = "bad JSON container for " + member.key; }
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
const std::size_t start = index;
|
||||
while (index < text.size() && text[index] != ',' && text[index] != '}') {
|
||||
++index;
|
||||
}
|
||||
member.raw = text.substr(start, index - start);
|
||||
while (!member.raw.empty() &&
|
||||
(member.raw.back() == ' ' || member.raw.back() == '\n' ||
|
||||
member.raw.back() == '\r' || member.raw.back() == '\t')) {
|
||||
member.raw.pop_back();
|
||||
}
|
||||
}
|
||||
for (const Member& existing : *out) {
|
||||
if (existing.key == member.key) {
|
||||
if (error != nullptr) { *error = "duplicate JSON key " + member.key; }
|
||||
return false;
|
||||
}
|
||||
}
|
||||
out->push_back(std::move(member));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
const Member* find(const std::vector<Member>& members, const std::string& key) {
|
||||
for (const Member& member : members) {
|
||||
if (member.key == key) {
|
||||
return &member;
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
bool as_string(const Member& member, std::string* out) {
|
||||
if (!member.is_string || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
*out = member.raw;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool as_number(const Member& member, double* out) {
|
||||
if (member.is_string || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
char* end = nullptr;
|
||||
const double value = std::strtod(member.raw.c_str(), &end);
|
||||
if (end == member.raw.c_str() || !std::isfinite(value)) {
|
||||
return false;
|
||||
}
|
||||
*out = value;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool as_integer(const Member& member, long long* out) {
|
||||
double value = 0.0;
|
||||
if (!as_number(member, &value) || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (value != std::floor(value)) {
|
||||
return false;
|
||||
}
|
||||
*out = static_cast<long long>(value);
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace joc::json
|
||||
@@ -1,25 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::json {
|
||||
|
||||
struct Member {
|
||||
std::string key;
|
||||
std::string raw;
|
||||
bool is_string = false;
|
||||
};
|
||||
|
||||
// Parses a top-level JSON object. Rejects non-objects and duplicate keys.
|
||||
bool parse_object(const std::string& text, std::vector<Member>* out, std::string* error);
|
||||
|
||||
const Member* find(const std::vector<Member>& members, const std::string& key);
|
||||
|
||||
bool as_string(const Member& member, std::string* out);
|
||||
bool as_number(const Member& member, double* out);
|
||||
bool as_integer(const Member& member, long long* out);
|
||||
|
||||
} // namespace joc::json
|
||||
@@ -1,25 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cfenv>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <string>
|
||||
|
||||
namespace joc::pynum {
|
||||
|
||||
inline long long py_round(double value) {
|
||||
return static_cast<long long>(std::nearbyint(value));
|
||||
}
|
||||
|
||||
inline std::string format_fixed(double value, int decimals) {
|
||||
char buffer[64];
|
||||
std::snprintf(buffer, sizeof(buffer), "%.*f", decimals, value);
|
||||
return std::string(buffer);
|
||||
}
|
||||
|
||||
inline long long trunc_to_ll(double value) {
|
||||
return static_cast<long long>(value);
|
||||
}
|
||||
|
||||
} // namespace joc::pynum
|
||||
@@ -1,165 +0,0 @@
|
||||
#include "foundation/sha256.h"
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace joc::crypto {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr std::uint32_t kK[64] = {
|
||||
0x428a2f98u, 0x71374491u, 0xb5c0fbcfu, 0xe9b5dba5u, 0x3956c25bu, 0x59f111f1u, 0x923f82a4u,
|
||||
0xab1c5ed5u, 0xd807aa98u, 0x12835b01u, 0x243185beu, 0x550c7dc3u, 0x72be5d74u, 0x80deb1feu,
|
||||
0x9bdc06a7u, 0xc19bf174u, 0xe49b69c1u, 0xefbe4786u, 0x0fc19dc6u, 0x240ca1ccu, 0x2de92c6fu,
|
||||
0x4a7484aau, 0x5cb0a9dcu, 0x76f988dau, 0x983e5152u, 0xa831c66du, 0xb00327c8u, 0xbf597fc7u,
|
||||
0xc6e00bf3u, 0xd5a79147u, 0x06ca6351u, 0x14292967u, 0x27b70a85u, 0x2e1b2138u, 0x4d2c6dfcu,
|
||||
0x53380d13u, 0x650a7354u, 0x766a0abbu, 0x81c2c92eu, 0x92722c85u, 0xa2bfe8a1u, 0xa81a664bu,
|
||||
0xc24b8b70u, 0xc76c51a3u, 0xd192e819u, 0xd6990624u, 0xf40e3585u, 0x106aa070u, 0x19a4c116u,
|
||||
0x1e376c08u, 0x2748774cu, 0x34b0bcb5u, 0x391c0cb3u, 0x4ed8aa4au, 0x5b9cca4fu, 0x682e6ff3u,
|
||||
0x748f82eeu, 0x78a5636fu, 0x84c87814u, 0x8cc70208u, 0x90befffau, 0xa4506cebu, 0xbef9a3f7u,
|
||||
0xc67178f2u};
|
||||
|
||||
inline std::uint32_t rotr(std::uint32_t value, unsigned count) {
|
||||
return (value >> count) | (value << (32u - count));
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
void Sha256::reset() {
|
||||
state_[0] = 0x6a09e667u;
|
||||
state_[1] = 0xbb67ae85u;
|
||||
state_[2] = 0x3c6ef372u;
|
||||
state_[3] = 0xa54ff53au;
|
||||
state_[4] = 0x510e527fu;
|
||||
state_[5] = 0x9b05688cu;
|
||||
state_[6] = 0x1f83d9abu;
|
||||
state_[7] = 0x5be0cd19u;
|
||||
bit_count_ = 0;
|
||||
buffer_used_ = 0;
|
||||
std::memset(buffer_, 0, sizeof(buffer_));
|
||||
}
|
||||
|
||||
void Sha256::transform(const std::uint8_t block[64]) {
|
||||
std::uint32_t w[64];
|
||||
for (unsigned i = 0; i < 16; ++i) {
|
||||
w[i] = (static_cast<std::uint32_t>(block[i * 4]) << 24) |
|
||||
(static_cast<std::uint32_t>(block[i * 4 + 1]) << 16) |
|
||||
(static_cast<std::uint32_t>(block[i * 4 + 2]) << 8) |
|
||||
static_cast<std::uint32_t>(block[i * 4 + 3]);
|
||||
}
|
||||
for (unsigned i = 16; i < 64; ++i) {
|
||||
const std::uint32_t s0 = rotr(w[i - 15], 7) ^ rotr(w[i - 15], 18) ^ (w[i - 15] >> 3);
|
||||
const std::uint32_t s1 = rotr(w[i - 2], 17) ^ rotr(w[i - 2], 19) ^ (w[i - 2] >> 10);
|
||||
w[i] = w[i - 16] + s0 + w[i - 7] + s1;
|
||||
}
|
||||
std::uint32_t a = state_[0];
|
||||
std::uint32_t b = state_[1];
|
||||
std::uint32_t c = state_[2];
|
||||
std::uint32_t d = state_[3];
|
||||
std::uint32_t e = state_[4];
|
||||
std::uint32_t f = state_[5];
|
||||
std::uint32_t g = state_[6];
|
||||
std::uint32_t h = state_[7];
|
||||
for (unsigned i = 0; i < 64; ++i) {
|
||||
const std::uint32_t s1 = rotr(e, 6) ^ rotr(e, 11) ^ rotr(e, 25);
|
||||
const std::uint32_t ch = (e & f) ^ (~e & g);
|
||||
const std::uint32_t temp1 = h + s1 + ch + kK[i] + w[i];
|
||||
const std::uint32_t s0 = rotr(a, 2) ^ rotr(a, 13) ^ rotr(a, 22);
|
||||
const std::uint32_t maj = (a & b) ^ (a & c) ^ (b & c);
|
||||
const std::uint32_t temp2 = s0 + maj;
|
||||
h = g;
|
||||
g = f;
|
||||
f = e;
|
||||
e = d + temp1;
|
||||
d = c;
|
||||
c = b;
|
||||
b = a;
|
||||
a = temp1 + temp2;
|
||||
}
|
||||
state_[0] += a;
|
||||
state_[1] += b;
|
||||
state_[2] += c;
|
||||
state_[3] += d;
|
||||
state_[4] += e;
|
||||
state_[5] += f;
|
||||
state_[6] += g;
|
||||
state_[7] += h;
|
||||
}
|
||||
|
||||
void Sha256::update(const void* data, std::size_t size) {
|
||||
const std::uint8_t* bytes = static_cast<const std::uint8_t*>(data);
|
||||
bit_count_ += static_cast<std::uint64_t>(size) * 8u;
|
||||
while (size > 0) {
|
||||
const std::size_t space = 64u - buffer_used_;
|
||||
const std::size_t take = size < space ? size : space;
|
||||
std::memcpy(buffer_ + buffer_used_, bytes, take);
|
||||
buffer_used_ += take;
|
||||
bytes += take;
|
||||
size -= take;
|
||||
if (buffer_used_ == 64u) {
|
||||
transform(buffer_);
|
||||
buffer_used_ = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Sha256::finish(std::uint8_t out[32]) {
|
||||
const std::uint64_t total_bits = bit_count_;
|
||||
const std::uint8_t pad = 0x80u;
|
||||
update(&pad, 1);
|
||||
const std::uint8_t zero = 0x00u;
|
||||
while (buffer_used_ != 56u) {
|
||||
update(&zero, 1);
|
||||
}
|
||||
std::uint8_t length_bytes[8];
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
length_bytes[i] = static_cast<std::uint8_t>((total_bits >> (56u - i * 8u)) & 0xFFu);
|
||||
}
|
||||
std::memcpy(buffer_ + buffer_used_, length_bytes, 8);
|
||||
buffer_used_ += 8;
|
||||
transform(buffer_);
|
||||
buffer_used_ = 0;
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
out[i * 4 + 0] = static_cast<std::uint8_t>((state_[i] >> 24) & 0xFFu);
|
||||
out[i * 4 + 1] = static_cast<std::uint8_t>((state_[i] >> 16) & 0xFFu);
|
||||
out[i * 4 + 2] = static_cast<std::uint8_t>((state_[i] >> 8) & 0xFFu);
|
||||
out[i * 4 + 3] = static_cast<std::uint8_t>(state_[i] & 0xFFu);
|
||||
}
|
||||
}
|
||||
|
||||
std::string Sha256::finish_hex() {
|
||||
std::uint8_t digest[32];
|
||||
finish(digest);
|
||||
static const char* kHex = "0123456789abcdef";
|
||||
std::string text;
|
||||
text.resize(64);
|
||||
for (unsigned i = 0; i < 32; ++i) {
|
||||
text[i * 2] = kHex[(digest[i] >> 4) & 0x0Fu];
|
||||
text[i * 2 + 1] = kHex[digest[i] & 0x0Fu];
|
||||
}
|
||||
return text;
|
||||
}
|
||||
|
||||
std::string sha256_hex(const void* data, std::size_t size) {
|
||||
Sha256 hash;
|
||||
hash.update(data, size);
|
||||
return hash.finish_hex();
|
||||
}
|
||||
|
||||
bool sha256_hex_matches(const void* data, std::size_t size, const std::string& expected_hex) {
|
||||
if (expected_hex.size() != 64) {
|
||||
return false;
|
||||
}
|
||||
std::string actual = sha256_hex(data, size);
|
||||
for (std::size_t i = 0; i < 64; ++i) {
|
||||
char expected = expected_hex[i];
|
||||
if (expected >= 'A' && expected <= 'F') {
|
||||
expected = static_cast<char>(expected - 'A' + 'a');
|
||||
}
|
||||
if (actual[i] != expected) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace joc::crypto
|
||||
@@ -1,31 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
|
||||
namespace joc::crypto {
|
||||
|
||||
class Sha256 {
|
||||
public:
|
||||
Sha256() { reset(); }
|
||||
|
||||
void reset();
|
||||
void update(const void* data, std::size_t size);
|
||||
void finish(std::uint8_t out[32]);
|
||||
std::string finish_hex();
|
||||
|
||||
private:
|
||||
void transform(const std::uint8_t block[64]);
|
||||
|
||||
std::uint32_t state_[8] = {};
|
||||
std::uint64_t bit_count_ = 0;
|
||||
std::uint8_t buffer_[64] = {};
|
||||
std::size_t buffer_used_ = 0;
|
||||
};
|
||||
|
||||
std::string sha256_hex(const void* data, std::size_t size);
|
||||
bool sha256_hex_matches(const void* data, std::size_t size, const std::string& expected_hex);
|
||||
|
||||
} // namespace joc::crypto
|
||||
@@ -1,48 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <string>
|
||||
#include <utility>
|
||||
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc {
|
||||
|
||||
class Status {
|
||||
public:
|
||||
Status() = default;
|
||||
|
||||
static Status success() { return Status(); }
|
||||
|
||||
static Status fail(joc_error code, std::string stage, std::string message) {
|
||||
Status s;
|
||||
s.code_ = code;
|
||||
s.stage_ = std::move(stage);
|
||||
s.message_ = std::move(message);
|
||||
return s;
|
||||
}
|
||||
|
||||
bool ok() const { return code_ == JOC_OK; }
|
||||
joc_error code() const { return code_; }
|
||||
const std::string& stage() const { return stage_; }
|
||||
const std::string& message() const { return message_; }
|
||||
|
||||
private:
|
||||
joc_error code_ = JOC_OK;
|
||||
std::string stage_ = "none";
|
||||
std::string message_;
|
||||
};
|
||||
|
||||
// Stage names are kept as plain literals so that C++ and the Python frontend
|
||||
namespace stage {
|
||||
inline constexpr const char* kFoundation = "foundation";
|
||||
inline constexpr const char* kEac3 = "eac3_transport";
|
||||
inline constexpr const char* kEmdf = "emdf";
|
||||
inline constexpr const char* kJoc = "joc";
|
||||
inline constexpr const char* kOamd = "oamd";
|
||||
inline constexpr const char* kDsp = "dsp";
|
||||
inline constexpr const char* kRender = "render";
|
||||
inline constexpr const char* kOutput = "output";
|
||||
} // namespace stage
|
||||
|
||||
} // namespace joc
|
||||
@@ -1,401 +0,0 @@
|
||||
#include "hrtf/jochrtf.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
|
||||
#include "foundation/mini_json.h"
|
||||
#include "foundation/sha256.h"
|
||||
#include "io/npy.h"
|
||||
#include "io/zip_reader.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
namespace {
|
||||
|
||||
std::string to_upper(std::string text) {
|
||||
for (char& c : text) {
|
||||
if (c >= 'a' && c <= 'z') {
|
||||
c = static_cast<char>(c - 'a' + 'A');
|
||||
}
|
||||
}
|
||||
return text;
|
||||
}
|
||||
|
||||
bool is_sha256_hex(const std::string& text) {
|
||||
if (text.size() != 64) {
|
||||
return false;
|
||||
}
|
||||
for (const char c : text) {
|
||||
const bool digit = c >= '0' && c <= '9';
|
||||
const bool upper = c >= 'A' && c <= 'F';
|
||||
if (!digit && !upper) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// json.dumps(list(shape)) as the reference writes it, e.g. "[36, 2, 77]".
|
||||
std::string shape_json(const std::vector<std::int64_t>& shape) {
|
||||
std::string text = "[";
|
||||
for (std::size_t i = 0; i < shape.size(); ++i) {
|
||||
text += (i == 0 ? "" : ", ");
|
||||
text += std::to_string(shape[i]);
|
||||
}
|
||||
text += "]";
|
||||
return text;
|
||||
}
|
||||
|
||||
Status hrtf_fail(const std::string& message) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender, message);
|
||||
}
|
||||
|
||||
std::string payload_sha256(const std::vector<double>& centers,
|
||||
const std::vector<double>& coefficients,
|
||||
const std::vector<double>& delay_coefficients,
|
||||
const std::vector<double>& delay_bounds) {
|
||||
crypto::Sha256 hash;
|
||||
const char prefix[] = "JOC-HRTF-CACHE-PAYLOAD-V1";
|
||||
hash.update(prefix, sizeof(prefix) - 1);
|
||||
const std::uint8_t zero = 0;
|
||||
hash.update(&zero, 1);
|
||||
|
||||
struct Entry {
|
||||
const char* name;
|
||||
const char* dtype;
|
||||
const std::vector<double>* values;
|
||||
std::vector<std::int64_t> shape;
|
||||
};
|
||||
const Entry entries[4] = {
|
||||
{"band_center_frequencies_hz", "<f8", ¢ers, {kHybridBands}},
|
||||
{"coefficients", "<c16", &coefficients, {kShTerms, kEars, kHybridBands}},
|
||||
{"delay_coefficients", "<f8", &delay_coefficients, {kShTerms, kEars}},
|
||||
{"delay_bounds", "<f8", &delay_bounds, {2, 2}},
|
||||
};
|
||||
for (const Entry& entry : entries) {
|
||||
const std::string name(entry.name);
|
||||
const std::string dtype(entry.dtype);
|
||||
const std::string shape = shape_json(entry.shape);
|
||||
hash.update(name.data(), name.size());
|
||||
hash.update(&zero, 1);
|
||||
hash.update(dtype.data(), dtype.size());
|
||||
hash.update(&zero, 1);
|
||||
hash.update(shape.data(), shape.size());
|
||||
hash.update(&zero, 1);
|
||||
hash.update(entry.values->data(), entry.values->size() * sizeof(double));
|
||||
}
|
||||
return hash.finish_hex();
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Status load_jochrtf(const std::string& path, Field* out) {
|
||||
if (out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null field");
|
||||
}
|
||||
io::ZipArchive archive;
|
||||
std::string error;
|
||||
if (!archive.open(path, &error)) {
|
||||
return Status::fail(JOC_ERR_HRTF_NOT_FOUND, stage::kRender,
|
||||
"cannot read compiled HRTF " + path + ": " + error);
|
||||
}
|
||||
|
||||
// Member set must be exactly the five expected names.
|
||||
static const char* kMembers[5] = {"metadata_json.npy", "band_center_frequencies_hz.npy",
|
||||
"coefficients.npy", "delay_coefficients.npy",
|
||||
"delay_bounds.npy"};
|
||||
if (archive.entries().size() != 5u) {
|
||||
return hrtf_fail("compiled HRTF cache has an invalid member set (" +
|
||||
std::to_string(archive.entries().size()) + " members)");
|
||||
}
|
||||
for (const char* name : kMembers) {
|
||||
if (archive.find(name) == nullptr) {
|
||||
return hrtf_fail(std::string("compiled HRTF cache is missing ") + name);
|
||||
}
|
||||
}
|
||||
|
||||
auto read_member = [&](const char* name, std::vector<std::uint8_t>* raw,
|
||||
io::NpyArray* array) -> Status {
|
||||
if (!archive.read_member(name, raw, &error)) {
|
||||
return hrtf_fail(std::string("compiled HRTF member ") + name + ": " + error);
|
||||
}
|
||||
if (!io::parse_npy(raw->data(), raw->size(), array, &error)) {
|
||||
return hrtf_fail(std::string("compiled HRTF member ") + name + ": " + error);
|
||||
}
|
||||
if (array->fortran_order) {
|
||||
return hrtf_fail(std::string("compiled HRTF member must be C-contiguous: ") + name);
|
||||
}
|
||||
return Status::success();
|
||||
};
|
||||
|
||||
std::vector<std::uint8_t> raw;
|
||||
io::NpyArray array;
|
||||
|
||||
Status status = read_member("metadata_json.npy", &raw, &array);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::string metadata_text;
|
||||
if (!io::npy_unicode_to_utf8(array, &metadata_text, &error)) {
|
||||
return hrtf_fail("compiled HRTF metadata: " + error);
|
||||
}
|
||||
if (metadata_text.size() > 64u * 1024u) {
|
||||
return hrtf_fail("compiled HRTF metadata is too large");
|
||||
}
|
||||
|
||||
status = read_member("band_center_frequencies_hz.npy", &raw, &array);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (array.descr != "<f8" || !io::npy_shape_is(array, {kHybridBands})) {
|
||||
return hrtf_fail("band_center_frequencies_hz must be <f8(77,)");
|
||||
}
|
||||
std::vector<double> centers;
|
||||
io::npy_to_double(array, ¢ers, &error);
|
||||
|
||||
status = read_member("coefficients.npy", &raw, &array);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (array.descr != "<c16" || !io::npy_shape_is(array, {kShTerms, kEars, kHybridBands})) {
|
||||
return hrtf_fail("coefficients must be <c16(36, 2, 77)");
|
||||
}
|
||||
std::vector<double> coefficients;
|
||||
if (!io::npy_to_double(array, &coefficients, &error)) {
|
||||
return hrtf_fail("coefficients: " + error);
|
||||
}
|
||||
|
||||
status = read_member("delay_coefficients.npy", &raw, &array);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (array.descr != "<f8" || !io::npy_shape_is(array, {kShTerms, kEars})) {
|
||||
return hrtf_fail("delay_coefficients must be <f8(36, 2)");
|
||||
}
|
||||
std::vector<double> delay_coefficients;
|
||||
io::npy_to_double(array, &delay_coefficients, &error);
|
||||
|
||||
status = read_member("delay_bounds.npy", &raw, &array);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (array.descr != "<f8" || !io::npy_shape_is(array, {2, 2})) {
|
||||
return hrtf_fail("delay_bounds must be <f8(2, 2)");
|
||||
}
|
||||
std::vector<double> delay_bounds;
|
||||
io::npy_to_double(array, &delay_bounds, &error);
|
||||
|
||||
std::vector<json::Member> members;
|
||||
if (!json::parse_object(metadata_text, &members, &error)) {
|
||||
return hrtf_fail("compiled HRTF metadata: " + error);
|
||||
}
|
||||
auto require_string = [&](const char* key, std::string* value) -> Status {
|
||||
const json::Member* member = json::find(members, key);
|
||||
if (member == nullptr || !json::as_string(*member, value)) {
|
||||
return hrtf_fail(std::string("compiled HRTF metadata is missing ") + key);
|
||||
}
|
||||
return Status::success();
|
||||
};
|
||||
std::string magic;
|
||||
std::string schema;
|
||||
std::string source_sha256;
|
||||
std::string cache_key;
|
||||
std::string payload_hash;
|
||||
std::string delay_source;
|
||||
status = require_string("magic", &magic);
|
||||
if (!status.ok()) { return status; }
|
||||
status = require_string("cache_schema", &schema);
|
||||
if (!status.ok()) { return status; }
|
||||
status = require_string("source_sha256", &source_sha256);
|
||||
if (!status.ok()) { return status; }
|
||||
status = require_string("cache_key", &cache_key);
|
||||
if (!status.ok()) { return status; }
|
||||
status = require_string("payload_sha256", &payload_hash);
|
||||
if (!status.ok()) { return status; }
|
||||
status = require_string("delay_source", &delay_source);
|
||||
if (!status.ok()) { return status; }
|
||||
|
||||
if (magic != kMagic) {
|
||||
return hrtf_fail("compiled HRTF magic mismatch: " + magic);
|
||||
}
|
||||
if (schema != kCacheSchema) {
|
||||
return hrtf_fail("compiled HRTF cache schema mismatch: " + schema);
|
||||
}
|
||||
const json::Member* version_member = json::find(members, "format_version");
|
||||
long long version = -1;
|
||||
if (version_member == nullptr || !json::as_integer(*version_member, &version)) {
|
||||
return hrtf_fail("compiled HRTF metadata is missing format_version");
|
||||
}
|
||||
if (version != kFormatVersion) {
|
||||
return Status::fail(JOC_ERR_HRTF_VERSION, stage::kRender,
|
||||
"unsupported .jochrtf version " + std::to_string(version) +
|
||||
"; rebuild it from the source SOFA");
|
||||
}
|
||||
out->source_sha256 = to_upper(source_sha256);
|
||||
out->cache_key = to_upper(cache_key);
|
||||
if (!is_sha256_hex(out->source_sha256)) {
|
||||
return hrtf_fail("compiled HRTF source_sha256 is not a 64-digit digest");
|
||||
}
|
||||
if (!is_sha256_hex(out->cache_key)) {
|
||||
return hrtf_fail("compiled HRTF cache_key is not a 64-digit digest");
|
||||
}
|
||||
|
||||
const std::string expected = payload_sha256(centers, coefficients, delay_coefficients,
|
||||
delay_bounds);
|
||||
if (to_upper(payload_hash) != to_upper(expected)) {
|
||||
return Status::fail(JOC_ERR_HRTF_HASH, stage::kRender,
|
||||
"compiled HRTF payload hash mismatch");
|
||||
}
|
||||
out->payload_sha256 = to_upper(payload_hash);
|
||||
|
||||
const json::Member* radius_member = json::find(members, "measurement_radius_m");
|
||||
double radius = 0.0;
|
||||
if (radius_member == nullptr || !json::as_number(*radius_member, &radius) || radius <= 0.0) {
|
||||
return hrtf_fail("compiled HRTF measurement_radius_m must be a positive number");
|
||||
}
|
||||
out->measurement_radius_m = radius;
|
||||
const json::Member* order_member = json::find(members, "order");
|
||||
long long order = 0;
|
||||
if (order_member == nullptr || !json::as_integer(*order_member, &order) || order <= 0 ||
|
||||
order * order > kShTerms) {
|
||||
return hrtf_fail("compiled HRTF order is out of range");
|
||||
}
|
||||
out->order = order;
|
||||
|
||||
for (const double value : coefficients) {
|
||||
if (!std::isfinite(value)) {
|
||||
return hrtf_fail("compiled HRTF coefficients contain non-finite values");
|
||||
}
|
||||
}
|
||||
for (const double value : delay_coefficients) {
|
||||
if (!std::isfinite(value) || std::abs(value) > 48000.0 * 64.0) {
|
||||
return hrtf_fail("compiled HRTF delay coefficients are out of range");
|
||||
}
|
||||
}
|
||||
for (const double value : delay_bounds) {
|
||||
if (!std::isfinite(value)) {
|
||||
return hrtf_fail("compiled HRTF delay bounds contain non-finite values");
|
||||
}
|
||||
}
|
||||
if (delay_bounds.size() == 4u && delay_bounds[0] > delay_bounds[1]) {
|
||||
return hrtf_fail("compiled HRTF delay bounds are inverted");
|
||||
}
|
||||
|
||||
if (const json::Member* member = json::find(members, "compiler_version")) {
|
||||
json::as_string(*member, &out->compiler_version);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "phase_policy_version")) {
|
||||
json::as_string(*member, &out->phase_policy_version);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "sh_convention")) {
|
||||
json::as_string(*member, &out->sh_convention);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "source_display_name")) {
|
||||
json::as_string(*member, &out->source_display_name);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "projection_ridge")) {
|
||||
json::as_number(*member, &out->projection_ridge);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "spherical_harmonic_ridge")) {
|
||||
json::as_number(*member, &out->spherical_harmonic_ridge);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "fit_report")) {
|
||||
out->fit_report_json = member->raw;
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "filterbank")) {
|
||||
out->filterbank_json = member->raw;
|
||||
}
|
||||
out->delay_source = delay_source;
|
||||
out->metadata_json = metadata_text;
|
||||
out->coefficients = std::move(coefficients);
|
||||
out->delay_coefficients = std::move(delay_coefficients);
|
||||
out->delay_bounds = std::move(delay_bounds);
|
||||
out->band_centers_hz = std::move(centers);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status load_kernels(const std::string& npz_path, Kernels* out) {
|
||||
if (out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null kernels");
|
||||
}
|
||||
io::ZipArchive archive;
|
||||
std::string error;
|
||||
if (!archive.open(npz_path, &error)) {
|
||||
return Status::fail(JOC_ERR_HRTF_NOT_FOUND, stage::kRender,
|
||||
"cannot read kernel tables " + npz_path + ": " + error);
|
||||
}
|
||||
|
||||
struct Request {
|
||||
const char* member;
|
||||
const char* shape_text;
|
||||
std::vector<std::int64_t> shape;
|
||||
};
|
||||
const Request requests[6] = {
|
||||
{"qmf_analysis_coefficients.npy", "<f4", {64, 10}},
|
||||
{"hybrid_analysis_low_kernel.npy", "<f4", {3, 2, 13, 16, 2}},
|
||||
{"hybrid_synthesis_indices.npy", "<i2", {154, 4}},
|
||||
{"hybrid_synthesis_values.npy", "<f4", {154}},
|
||||
{"qmf_synthesis_basis.npy", "<f8", {64, 4, 128}},
|
||||
{"qmf_synthesis_taps.npy", "<f8", {64, 10, 4}},
|
||||
};
|
||||
|
||||
std::vector<std::uint8_t> raw;
|
||||
std::vector<std::uint8_t> ordered;
|
||||
for (const Request& request : requests) {
|
||||
const std::string name = request.member;
|
||||
if (!archive.read_member(name, &raw, &error)) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"kernel table member " + name + ": " + error);
|
||||
}
|
||||
io::NpyArray array;
|
||||
if (!io::parse_npy(raw.data(), raw.size(), &array, &error)) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"kernel table member " + name + ": " + error);
|
||||
}
|
||||
if (array.descr != request.shape_text || !io::npy_shape_is(array, request.shape)) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"kernel table member " + name + " has an unexpected dtype/shape");
|
||||
}
|
||||
// Logical C order: required because the reused kernel indexes the hybrid
|
||||
// synthesis table row-major while the shipped member is Fortran-order.
|
||||
if (!io::npy_to_c_order(array, &ordered, &error)) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"kernel table member " + name + ": " + error);
|
||||
}
|
||||
const std::size_t count = array.element_count();
|
||||
if (std::strcmp(request.member, "qmf_analysis_coefficients.npy") == 0) {
|
||||
std::vector<float> values(count);
|
||||
std::memcpy(values.data(), ordered.data(), count * sizeof(float));
|
||||
out->qmf_analysis.assign(values.begin(), values.end());
|
||||
} else if (std::strcmp(request.member, "hybrid_analysis_low_kernel.npy") == 0) {
|
||||
std::vector<float> values(count);
|
||||
std::memcpy(values.data(), ordered.data(), count * sizeof(float));
|
||||
out->hybrid_low.assign(values.begin(), values.end());
|
||||
} else if (std::strcmp(request.member, "hybrid_synthesis_indices.npy") == 0) {
|
||||
out->hybrid_indices.resize(count);
|
||||
std::memcpy(out->hybrid_indices.data(), ordered.data(), count * sizeof(std::int16_t));
|
||||
} else if (std::strcmp(request.member, "hybrid_synthesis_values.npy") == 0) {
|
||||
std::vector<float> values(count);
|
||||
std::memcpy(values.data(), ordered.data(), count * sizeof(float));
|
||||
out->hybrid_values.assign(values.begin(), values.end());
|
||||
} else if (std::strcmp(request.member, "qmf_synthesis_basis.npy") == 0) {
|
||||
std::memcpy(out->qmf_basis.empty() ? (out->qmf_basis.resize(count), out->qmf_basis.data())
|
||||
: out->qmf_basis.data(),
|
||||
ordered.data(), count * sizeof(double));
|
||||
out->qmf_basis.resize(count);
|
||||
} else {
|
||||
out->qmf_taps.resize(count);
|
||||
std::memcpy(out->qmf_taps.data(), ordered.data(), count * sizeof(double));
|
||||
}
|
||||
}
|
||||
out->hybrid_count = static_cast<std::uint32_t>(out->hybrid_values.size());
|
||||
if (out->hybrid_count == 0u) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"kernel tables contain no hybrid synthesis entries");
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -1,66 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
inline constexpr int kShTerms = 36;
|
||||
inline constexpr int kEars = 2;
|
||||
inline constexpr int kHybridBands = 77;
|
||||
inline constexpr int kFormatVersion = 1;
|
||||
inline constexpr const char* kMagic = "JOC-HRTF-CACHE";
|
||||
inline constexpr const char* kCacheSchema = "joc-compiled-hrtf-v1";
|
||||
|
||||
struct Field {
|
||||
std::vector<double> coefficients;
|
||||
std::vector<double> delay_coefficients;
|
||||
std::vector<double> delay_bounds;
|
||||
std::vector<double> band_centers_hz;
|
||||
double measurement_radius_m = 1.0;
|
||||
long long order = 5;
|
||||
std::string source_sha256;
|
||||
std::string cache_key;
|
||||
std::string payload_sha256;
|
||||
std::string delay_source;
|
||||
std::string compiler_version;
|
||||
std::string phase_policy_version;
|
||||
std::string sh_convention;
|
||||
std::string filterbank_json;
|
||||
std::string metadata_json;
|
||||
// Compile-side metadata, needed to write the cache back out unchanged.
|
||||
std::string source_display_name;
|
||||
std::string fit_report_json;
|
||||
double projection_ridge = 0.0;
|
||||
double spherical_harmonic_ridge = 0.0;
|
||||
};
|
||||
|
||||
Status load_jochrtf(const std::string& path, Field* out);
|
||||
|
||||
// Binaural filterbank kernels, as the reused kernel expects them (C order, the
|
||||
// exact dtypes of the ABI parameters).
|
||||
struct Kernels {
|
||||
std::vector<double> qmf_analysis;
|
||||
std::vector<double> hybrid_low;
|
||||
std::vector<std::int16_t> hybrid_indices;
|
||||
std::vector<double> hybrid_values;
|
||||
std::vector<double> qmf_basis;
|
||||
std::vector<double> qmf_taps;
|
||||
std::uint32_t hybrid_count = 0;
|
||||
};
|
||||
|
||||
// Loads a kernel-table archive. The file path is an override for verification;
|
||||
// the shipped tables are embedded (see builtin_kernels) so no data file is needed.
|
||||
// The Fortran-order index member is transposed into C order on purpose: the reused
|
||||
// kernel indexes the hybrid synthesis table row-major.
|
||||
Status load_kernels(const std::string& npz_path, Kernels* out);
|
||||
|
||||
// The public filterbank tables compiled into the library (identical values to the
|
||||
// archive the file loader accepts; the unit test checks their hashes).
|
||||
const Kernels& builtin_kernels();
|
||||
|
||||
} // namespace joc::hrtf
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,610 +0,0 @@
|
||||
#include "hrtf/public_filterbank.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cmath>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/fft.h"
|
||||
#include "simd/simd.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr double kPi = 3.14159265358979323846;
|
||||
constexpr std::size_t kQmfLength = dsp::kQmfFftSize;
|
||||
constexpr int kQmfTaps = 10;
|
||||
constexpr int kSynthesisRank = 4;
|
||||
constexpr int kSynthesisTaps = 10;
|
||||
|
||||
// ----------------------------------------------------------- filterbank -----
|
||||
|
||||
// One shared forward plan for the 128-point QMF transform. The analysis bank runs
|
||||
// it 2 * slots * channels times per chunk, so the twiddle recurrence is built once
|
||||
// instead of being re-derived inside every butterfly.
|
||||
const dsp::FftPlan& qmf_fft_plan() {
|
||||
static const dsp::FftPlan plan(dsp::kQmfFftSize, false);
|
||||
return plan;
|
||||
}
|
||||
|
||||
// Public 64-band complex QMF analysis (public_filterbank.QmfAnalysis).
|
||||
class QmfAnalysis {
|
||||
public:
|
||||
static_assert(static_cast<std::size_t>(kQmfBands) == simd::kQmfAnalysisBands,
|
||||
"the dispatched accumulate is written for this band count");
|
||||
QmfAnalysis(const Kernels& kernels, std::size_t channels)
|
||||
: channels_(channels), coefficients_(kernels.qmf_analysis) {
|
||||
history_.assign(9u * channels_ * kQmfBands, 0.0);
|
||||
// The polyphase MAC consumes one coefficient per band, so the shipped
|
||||
// [band][tap] layout makes its inner loop a stride-10 gather. Transposing
|
||||
// once here turns that into a contiguous AXPY. The coefficient values and
|
||||
// the accumulation order are untouched, so the sums are bit-identical.
|
||||
coefficients_by_lag_.resize(static_cast<std::size_t>(kQmfTaps) * kQmfBands);
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
for (int tap = 0; tap < kQmfTaps; ++tap) {
|
||||
coefficients_by_lag_[static_cast<std::size_t>(tap) * kQmfBands +
|
||||
static_cast<std::size_t>(band)] =
|
||||
coefficients_[static_cast<std::size_t>(band) * kQmfTaps +
|
||||
static_cast<std::size_t>(tap)];
|
||||
}
|
||||
}
|
||||
premultiply_.resize(kQmfBands);
|
||||
post_.resize(kQmfBands);
|
||||
even_post_.resize(kQmfBands);
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
const double phase = static_cast<double>(band);
|
||||
premultiply_[static_cast<std::size_t>(band)] =
|
||||
std::polar(1.0, -kPi * phase / 128.0);
|
||||
post_[static_cast<std::size_t>(band)] =
|
||||
std::polar(1.0, -3.0 * (phase + 0.5) * kPi / 128.0);
|
||||
even_post_[static_cast<std::size_t>(band)] =
|
||||
Complex(0.0, band % 2 == 0 ? 1.0 : -1.0);
|
||||
}
|
||||
}
|
||||
|
||||
void reset() { std::fill(history_.begin(), history_.end(), 0.0); }
|
||||
|
||||
// samples: [slots*64, channels]; output: [slots, channels, 64] complex.
|
||||
void process(const std::vector<double>& samples, std::size_t slots,
|
||||
std::vector<Complex>* output) {
|
||||
const std::size_t joined_slots = 9u + slots;
|
||||
const std::size_t history_size = 9u * channels_ * kQmfBands;
|
||||
const std::size_t joined_size = joined_slots * channels_ * kQmfBands;
|
||||
// The joined window is filled completely -- the history lands in its first
|
||||
// 9 * channels * 64 entries and the new samples in the rest -- so it is a
|
||||
// reusable scratch buffer rather than a fresh zero-filled allocation. The
|
||||
// history tail is taken by index instead of from end(), because the buffer may
|
||||
// be longer than the window this call uses.
|
||||
if (joined_.size() < joined_size) {
|
||||
joined_.resize(joined_size);
|
||||
}
|
||||
std::copy(history_.begin(), history_.end(), joined_.begin());
|
||||
std::copy(samples.begin(), samples.begin() + static_cast<std::ptrdiff_t>(slots * channels_ * kQmfBands),
|
||||
joined_.begin() + static_cast<std::ptrdiff_t>(history_size));
|
||||
|
||||
// The two polyphase accumulators are read before they are written, so their
|
||||
// zero fill is load-bearing and stays; only the per-call allocation goes.
|
||||
const std::size_t accumulator_size = slots * channels_ * kQmfBands;
|
||||
if (even_.size() < accumulator_size) {
|
||||
even_.resize(accumulator_size);
|
||||
}
|
||||
if (odd_.size() < accumulator_size) {
|
||||
odd_.resize(accumulator_size);
|
||||
}
|
||||
std::fill(even_.begin(), even_.begin() + static_cast<std::ptrdiff_t>(accumulator_size), 0.0);
|
||||
std::fill(odd_.begin(), odd_.begin() + static_cast<std::ptrdiff_t>(accumulator_size), 0.0);
|
||||
// The ten lags are ten accumulate passes over the same 64 bands with one
|
||||
// shared coefficient row; the bands are independent accumulations of a
|
||||
// single product each, so they are what the dispatched kernel puts in its
|
||||
// lanes, and every band keeps the caller's own multiply-then-add.
|
||||
//
|
||||
// Slots are processed in blocks, with the lag loop inside: one lag pass
|
||||
// touches every source row once, so running the ten passes over the whole
|
||||
// chunk re-reads the joined window ten times -- at 1536 slots that is
|
||||
// hundreds of megabytes per chunk and the loop ends up bound by memory, not
|
||||
// by arithmetic. A block's ten lag passes instead slide over a window of
|
||||
// (block + 9) rows that stays in the second-level cache. Lags still run in
|
||||
// ascending order inside a block, which is the order each output's sum is
|
||||
// formed in, so nothing about the arithmetic changes.
|
||||
constexpr std::size_t kSlotBlock = 32;
|
||||
for (std::size_t first = 0u; first < slots; first += kSlotBlock) {
|
||||
const std::size_t block = std::min(kSlotBlock, slots - first);
|
||||
for (int lag = 0; lag < kQmfTaps; ++lag) {
|
||||
std::vector<double>& target = (lag % 2 == 0) ? even_ : odd_;
|
||||
const double* row =
|
||||
coefficients_by_lag_.data() + static_cast<std::size_t>(lag) * kQmfBands;
|
||||
const std::size_t source_slot = 9u - static_cast<std::size_t>(lag) + first;
|
||||
simd::qmf_analysis_taps(
|
||||
target.data() + first * channels_ * kQmfBands,
|
||||
joined_.data() + source_slot * channels_ * kQmfBands, row,
|
||||
block * channels_);
|
||||
}
|
||||
}
|
||||
std::copy(joined_.begin() + static_cast<std::ptrdiff_t>(joined_size - history_size),
|
||||
joined_.begin() + static_cast<std::ptrdiff_t>(joined_size), history_.begin());
|
||||
|
||||
// Every output element is assigned below, so the size is all that has to be
|
||||
// established; a resize of an already correctly sized buffer touches nothing.
|
||||
output->resize(slots * channels_ * kQmfBands);
|
||||
std::array<Complex, dsp::kQmfFftSize> even_spectrum{};
|
||||
std::array<Complex, dsp::kQmfFftSize> odd_spectrum{};
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels_; ++channel) {
|
||||
const double* even_values = even_.data() + (slot * channels_ + channel) * kQmfBands;
|
||||
const double* odd_values = odd_.data() + (slot * channels_ + channel) * kQmfBands;
|
||||
transform(even_values, &even_spectrum);
|
||||
transform(odd_values, &odd_spectrum);
|
||||
Complex* destination =
|
||||
output->data() + (slot * channels_ + channel) * kQmfBands;
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
destination[band] = odd_spectrum[static_cast<std::size_t>(band)] +
|
||||
even_spectrum[static_cast<std::size_t>(band)] *
|
||||
even_post_[static_cast<std::size_t>(band)];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
void transform(const double* values, std::array<Complex, dsp::kQmfFftSize>* spectrum) {
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
(*spectrum)[static_cast<std::size_t>(band)] =
|
||||
Complex(values[band], 0.0) * premultiply_[static_cast<std::size_t>(band)];
|
||||
}
|
||||
for (int index = kQmfBands; index < dsp::kQmfFftSize; ++index) {
|
||||
(*spectrum)[static_cast<std::size_t>(index)] = Complex(0.0, 0.0);
|
||||
}
|
||||
dsp::fft_radix2(spectrum, qmf_fft_plan());
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
(*spectrum)[static_cast<std::size_t>(band)] *= post_[static_cast<std::size_t>(band)];
|
||||
}
|
||||
}
|
||||
|
||||
std::size_t channels_;
|
||||
std::vector<double> coefficients_; // [64][10]
|
||||
std::vector<double> coefficients_by_lag_; // [10][64], the same values transposed
|
||||
std::vector<double> history_; // [9][channels][64]
|
||||
std::vector<double> joined_; // scratch, [9 + slots][channels][64]
|
||||
std::vector<double> even_; // scratch, [slots][channels][64], zeroed per call
|
||||
std::vector<double> odd_; // scratch, [slots][channels][64], zeroed per call
|
||||
std::vector<Complex> premultiply_;
|
||||
std::vector<Complex> post_;
|
||||
std::vector<Complex> even_post_;
|
||||
};
|
||||
|
||||
// Sparse 64-QMF to 77-hybrid analysis (public_filterbank.HybridAnalysis).
|
||||
class HybridAnalysis {
|
||||
public:
|
||||
HybridAnalysis(const Kernels& kernels, std::size_t channels)
|
||||
: channels_(channels), low_kernel_(kernels.hybrid_low) {
|
||||
history_.assign(12u * channels_ * 3u * 2u, 0.0);
|
||||
high_history_.assign(6u * channels_ * 61u, Complex(0.0, 0.0));
|
||||
// The dispatched join walks one term at a time and adds its 32 weights to
|
||||
// 32 outputs, so the shipped [tap][band][component] table is regrouped to
|
||||
// the term order the caller accumulates in. Same weights, same order.
|
||||
const std::size_t outputs = simd::kHybridOutputs;
|
||||
low_by_term_.resize(simd::kHybridTerms * outputs);
|
||||
for (int lag = 0; lag < 13; ++lag) {
|
||||
for (int point = 0; point < 3; ++point) {
|
||||
for (int input = 0; input < 2; ++input) {
|
||||
const std::size_t term =
|
||||
(static_cast<std::size_t>(lag) * 3u + static_cast<std::size_t>(point)) * 2u +
|
||||
static_cast<std::size_t>(input);
|
||||
const std::size_t source = (static_cast<std::size_t>(point) * 2u +
|
||||
static_cast<std::size_t>(input)) * 13u +
|
||||
static_cast<std::size_t>(lag);
|
||||
for (std::size_t output = 0u; output < outputs; ++output) {
|
||||
low_by_term_[term * outputs + output] =
|
||||
low_kernel_[source * outputs + output];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
low_values_.resize(simd::kHybridJoinBlock * simd::kHybridTerms);
|
||||
low_out_.resize(simd::kHybridJoinBlock * outputs);
|
||||
}
|
||||
|
||||
void reset() {
|
||||
std::fill(history_.begin(), history_.end(), 0.0);
|
||||
std::fill(high_history_.begin(), high_history_.end(), Complex(0.0, 0.0));
|
||||
}
|
||||
|
||||
// qmf: [slots, channels, 64]; output: [slots, channels, 77] complex.
|
||||
void process(const std::vector<Complex>& qmf, std::size_t slots,
|
||||
std::vector<Complex>* output) {
|
||||
const std::size_t joined_slots = 12u + slots;
|
||||
const std::size_t history_size = 12u * channels_ * 6u;
|
||||
const std::size_t joined_size = joined_slots * channels_ * 3u * 2u;
|
||||
// Both the joined window and the pending high-band history are written in full
|
||||
// before they are read, so they are reused scratch buffers; the history tail is
|
||||
// taken by index because the buffer can be longer than this call's window.
|
||||
if (joined_.size() < joined_size) {
|
||||
joined_.resize(joined_size);
|
||||
}
|
||||
std::copy(history_.begin(), history_.end(), joined_.begin());
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels_; ++channel) {
|
||||
const Complex* source = qmf.data() + (slot * channels_ + channel) * kQmfBands;
|
||||
double* destination =
|
||||
joined_.data() + ((12u + slot) * channels_ + channel) * 6u;
|
||||
for (int band = 0; band < 3; ++band) {
|
||||
destination[static_cast<std::size_t>(band) * 2u] = source[band].real();
|
||||
destination[static_cast<std::size_t>(band) * 2u + 1u] = source[band].imag();
|
||||
}
|
||||
}
|
||||
}
|
||||
// The low bands are accumulated in a register block and written straight into
|
||||
// the output, and the high bands are written by the pass below; between them
|
||||
// every one of the 77 bands is assigned, so only the size has to be set.
|
||||
output->resize(slots * channels_ * kHybridBands);
|
||||
// The thirteen taps are summed in a per-output register block and the low
|
||||
// bands are written straight into the output. Keeping a separate low plane
|
||||
// and then copying it into the output re-streams tens of megabytes per chunk
|
||||
// for nothing, and only the first kHybridLow bands are ever touched. The
|
||||
// join itself is dispatched (see src/simd/simd.h): the 32 outputs of a
|
||||
// row are 32 independent accumulations over the same 78 terms, which is what
|
||||
// shares a vector. Every lane keeps the caller's term order -- lag, then
|
||||
// point, then input -- and its two roundings, and skips exactly the terms
|
||||
// this loop skips. Rows are staged in blocks so the gathered values do not
|
||||
// spill out of the first-level cache.
|
||||
const std::size_t hybrid_rows = slots * channels_;
|
||||
const std::size_t block = simd::kHybridJoinBlock;
|
||||
const std::size_t terms = simd::kHybridTerms;
|
||||
for (std::size_t first = 0u; first < hybrid_rows; first += block) {
|
||||
const std::size_t count = std::min(block, hybrid_rows - first);
|
||||
for (std::size_t index = 0u; index < count; ++index) {
|
||||
const std::size_t row = first + index;
|
||||
const std::size_t slot = row / channels_;
|
||||
const std::size_t channel = row % channels_;
|
||||
double* staged = low_values_.data() + index * terms;
|
||||
for (int lag = 0; lag < 13; ++lag) {
|
||||
const std::size_t source_slot = 12u - static_cast<std::size_t>(lag) + slot;
|
||||
const double* source =
|
||||
joined_.data() + (source_slot * channels_ + channel) * 6u;
|
||||
for (int point = 0; point < 3; ++point) {
|
||||
for (int input = 0; input < 2; ++input) {
|
||||
staged[(static_cast<std::size_t>(lag) * 3u +
|
||||
static_cast<std::size_t>(point)) * 2u +
|
||||
static_cast<std::size_t>(input)] =
|
||||
source[static_cast<std::size_t>(point) * 2u +
|
||||
static_cast<std::size_t>(input)];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
simd::hybrid_low_join(low_values_.data(), low_by_term_.data(),
|
||||
low_out_.data(), count);
|
||||
for (std::size_t index = 0u; index < count; ++index) {
|
||||
Complex* destination = output->data() + (first + index) * kHybridBands;
|
||||
const double* values = low_out_.data() + index * simd::kHybridOutputs;
|
||||
for (int band = 0; band < kHybridLow; ++band) {
|
||||
destination[band] = Complex(values[static_cast<std::size_t>(band) * 2u],
|
||||
values[static_cast<std::size_t>(band) * 2u + 1u]);
|
||||
}
|
||||
}
|
||||
}
|
||||
std::copy(joined_.begin() + static_cast<std::ptrdiff_t>(joined_size - history_size),
|
||||
joined_.begin() + static_cast<std::ptrdiff_t>(joined_size), history_.begin());
|
||||
|
||||
// The high bands pass through unchanged but delayed by the six slots of
|
||||
// history the reference concatenates in front of them. Only the last six
|
||||
// entries of that concatenation survive into high_history_, so a six-entry
|
||||
// register replaces the (6 + slots) plane and its full copy. Note the
|
||||
// output reads the concatenation at index `slot`, not `6 + slot`, so the
|
||||
// first six output slots come from the history: that offset is part of the
|
||||
// current output and is preserved verbatim.
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels_; ++channel) {
|
||||
Complex* destination = output->data() +
|
||||
(slot * channels_ + channel) * kHybridBands + kHybridLow;
|
||||
if (slot < 6u) {
|
||||
const Complex* source =
|
||||
high_history_.data() + (slot * channels_ + channel) * 61u;
|
||||
for (int band = 0; band < 61; ++band) {
|
||||
destination[band] = source[band];
|
||||
}
|
||||
} else {
|
||||
const Complex* source =
|
||||
qmf.data() + ((slot - 6u) * channels_ + channel) * kQmfBands;
|
||||
for (int band = 3; band < kQmfBands; ++band) {
|
||||
destination[static_cast<std::size_t>(band - 3)] = source[band];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Every entry of the pending high-band history is written here, so it is a
|
||||
// reusable scratch buffer; the copy into the live history is kept as it was.
|
||||
if (next_high_history_.size() < 6u * channels_ * 61u) {
|
||||
next_high_history_.resize(6u * channels_ * 61u);
|
||||
}
|
||||
for (std::size_t entry = 0u; entry < 6u; ++entry) {
|
||||
const std::size_t combined = slots + entry;
|
||||
for (std::size_t channel = 0u; channel < channels_; ++channel) {
|
||||
Complex* destination =
|
||||
next_high_history_.data() + (entry * channels_ + channel) * 61u;
|
||||
if (combined < 6u) {
|
||||
const Complex* source =
|
||||
high_history_.data() + (combined * channels_ + channel) * 61u;
|
||||
for (int band = 0; band < 61; ++band) {
|
||||
destination[band] = source[band];
|
||||
}
|
||||
} else {
|
||||
const Complex* source =
|
||||
qmf.data() + ((combined - 6u) * channels_ + channel) * kQmfBands;
|
||||
for (int band = 3; band < kQmfBands; ++band) {
|
||||
destination[static_cast<std::size_t>(band - 3)] = source[band];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
std::copy(next_high_history_.begin(), next_high_history_.end(), high_history_.begin());
|
||||
}
|
||||
|
||||
private:
|
||||
std::size_t channels_;
|
||||
std::vector<double> low_kernel_; // [3][2][13][16][2]
|
||||
std::vector<double> low_by_term_; // [78][32], the same weights in the caller's term order
|
||||
std::vector<double> low_values_; // scratch, [block][78]
|
||||
std::vector<double> low_out_; // scratch, [block][32]
|
||||
std::vector<double> history_; // [12][channels][3][2]
|
||||
std::vector<Complex> high_history_; // [6][channels][61]
|
||||
std::vector<double> joined_; // scratch, [12 + slots][channels][3][2]
|
||||
std::vector<Complex> next_high_history_; // scratch, [6][channels][61]
|
||||
};
|
||||
|
||||
// Instantaneous sparse 77-hybrid to 64-QMF synthesis map.
|
||||
class HybridSynthesis {
|
||||
public:
|
||||
explicit HybridSynthesis(const Kernels& kernels) {
|
||||
const std::size_t rows = kernels.hybrid_indices.size() / 4u;
|
||||
mapping_.reserve(rows);
|
||||
for (std::size_t index = 0u; index < rows; ++index) {
|
||||
Entry entry;
|
||||
for (int field = 0; field < 4; ++field) {
|
||||
entry.index[static_cast<std::size_t>(field)] =
|
||||
kernels.hybrid_indices[index * 4u + static_cast<std::size_t>(field)];
|
||||
}
|
||||
entry.gain = kernels.hybrid_values[index];
|
||||
mapping_.push_back(entry);
|
||||
}
|
||||
}
|
||||
|
||||
// hybrid: [slots, channels, 77]; output: [slots, channels, 64] complex.
|
||||
// The sparse map moves a real or imaginary part of one band into a real or
|
||||
// imaginary part of another, so the two components are accumulated apart.
|
||||
void process(const std::vector<Complex>& hybrid, std::size_t slots, std::size_t channels,
|
||||
std::vector<Complex>* output) const {
|
||||
const std::size_t rows = slots * channels;
|
||||
std::vector<double> real(rows * kQmfBands, 0.0);
|
||||
std::vector<double> imaginary(rows * kQmfBands, 0.0);
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels; ++channel) {
|
||||
const std::size_t row = slot * channels + channel;
|
||||
const Complex* source = hybrid.data() + row * kHybridBands;
|
||||
for (const Entry& entry : mapping_) {
|
||||
const double value = entry.index[1] == 0u ? source[entry.index[0]].real()
|
||||
: source[entry.index[0]].imag();
|
||||
if (value == 0.0) {
|
||||
continue;
|
||||
}
|
||||
double* destination =
|
||||
(entry.index[3] == 0u ? real.data() : imaginary.data()) + row * kQmfBands;
|
||||
destination[entry.index[2]] += value * entry.gain;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Every output element is assigned from the two accumulators below, so the
|
||||
// zero fill that `assign` performed was dead; only the size is needed.
|
||||
output->resize(rows * kQmfBands);
|
||||
for (std::size_t index = 0u; index < output->size(); ++index) {
|
||||
(*output)[index] = Complex(real[index], imaginary[index]);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
struct Entry {
|
||||
std::size_t index[4] = {0u, 0u, 0u, 0u};
|
||||
double gain = 0.0;
|
||||
};
|
||||
std::vector<Entry> mapping_;
|
||||
};
|
||||
|
||||
// Rank-4 64-band synthesis.
|
||||
class QmfSynthesis {
|
||||
public:
|
||||
QmfSynthesis(const Kernels& kernels, std::size_t channels)
|
||||
: channels_(channels), basis_(kernels.qmf_basis), taps_(kernels.qmf_taps) {
|
||||
history_.assign(9u * channels_ * kQmfBands * kSynthesisRank, 0.0);
|
||||
// The dispatched basis kernel reads the four ranks of one (band, tap) as
|
||||
// one vector, so the shipped [band][rank][tap] table is reordered once
|
||||
// here. The weights are the same doubles, only their order differs.
|
||||
const std::size_t bands = static_cast<std::size_t>(kQmfBands);
|
||||
const std::size_t ranks = static_cast<std::size_t>(kSynthesisRank);
|
||||
const std::size_t taps = dsp::kQmfFftSize;
|
||||
basis_by_tap_.resize(bands * taps * ranks);
|
||||
for (std::size_t band = 0u; band < bands; ++band) {
|
||||
for (std::size_t tap = 0u; tap < taps; ++tap) {
|
||||
for (std::size_t rank = 0u; rank < ranks; ++rank) {
|
||||
basis_by_tap_[(band * taps + tap) * ranks + rank] =
|
||||
basis_[(band * ranks + rank) * taps + tap];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void reset() { std::fill(history_.begin(), history_.end(), 0.0); }
|
||||
|
||||
// qmf: [slots, channels, 64]; output: [slots*64, channels] real.
|
||||
void process(const std::vector<Complex>& qmf, std::size_t slots, std::vector<double>* output) {
|
||||
const std::size_t rows = slots * channels_;
|
||||
// [row][band][component] staging for the basis application. Both staging
|
||||
// planes and the joined window are reusable scratch: every element of each is
|
||||
// written before it is read, so the buffers are sized once and kept instead of
|
||||
// being allocated and zero-filled on every call.
|
||||
const std::size_t flat_size = rows * dsp::kQmfFftSize;
|
||||
if (flat_.size() < flat_size) {
|
||||
flat_.resize(flat_size);
|
||||
}
|
||||
for (std::size_t row = 0u; row < rows; ++row) {
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
flat_[row * dsp::kQmfFftSize + static_cast<std::size_t>(band) * 2u] =
|
||||
qmf[row * kQmfBands + static_cast<std::size_t>(band)].real();
|
||||
flat_[row * dsp::kQmfFftSize + static_cast<std::size_t>(band) * 2u + 1u] =
|
||||
qmf[row * kQmfBands + static_cast<std::size_t>(band)].imag();
|
||||
}
|
||||
}
|
||||
// The sums are written straight into the joined window: the destination index
|
||||
// is known up front, the summation order is untouched, and the application
|
||||
// itself is dispatched -- the four ranks of a band are four independent dot
|
||||
// products over the same 128 values, so they share a vector while every lane
|
||||
// keeps the tap order and the two roundings of `sum +=`.
|
||||
const std::size_t history_size = 9u * channels_ * kQmfBands * kSynthesisRank;
|
||||
const std::size_t joined_size = history_size + rows * kQmfBands * kSynthesisRank;
|
||||
if (joined_.size() < joined_size) {
|
||||
joined_.resize(joined_size);
|
||||
}
|
||||
std::copy(history_.begin(), history_.end(), joined_.begin());
|
||||
simd::qmf_synthesis_basis(flat_.data(), basis_by_tap_.data(),
|
||||
joined_.data() + history_size, rows);
|
||||
std::copy(joined_.begin() + static_cast<std::ptrdiff_t>(joined_size - history_size),
|
||||
joined_.begin() + static_cast<std::ptrdiff_t>(joined_size), history_.begin());
|
||||
|
||||
output->assign(rows * kQmfBands, 0.0);
|
||||
for (int lag = 0; lag < kSynthesisTaps; ++lag) {
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
const std::size_t source_slot = 9u - static_cast<std::size_t>(lag) + slot;
|
||||
for (std::size_t channel = 0u; channel < channels_; ++channel) {
|
||||
const double* source =
|
||||
joined_.data() +
|
||||
(source_slot * channels_ + channel) * kQmfBands * kSynthesisRank;
|
||||
double* destination =
|
||||
output->data() + (slot * channels_ + channel) * kQmfBands;
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
double sum = 0.0;
|
||||
for (int rank = 0; rank < kSynthesisRank; ++rank) {
|
||||
sum += source[static_cast<std::size_t>(band) * kSynthesisRank +
|
||||
static_cast<std::size_t>(rank)] *
|
||||
taps_[(static_cast<std::size_t>(band) * kSynthesisTaps +
|
||||
static_cast<std::size_t>(lag)) * kSynthesisRank +
|
||||
static_cast<std::size_t>(rank)];
|
||||
}
|
||||
destination[band] += sum;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
std::size_t channels_;
|
||||
std::vector<double> basis_; // [64][4][128]
|
||||
std::vector<double> basis_by_tap_; // [64][128][4], the same weights transposed
|
||||
std::vector<double> taps_; // [64][10][4]
|
||||
std::vector<double> history_; // [9][channels][64][4]
|
||||
std::vector<double> flat_; // scratch, [rows][128], fully written per call
|
||||
std::vector<double> joined_; // scratch, [9 + slots][channels][64][4]
|
||||
};
|
||||
|
||||
// public_filterbank.PublicAnalysis77.process: [N, channels] -> [N/64, channels, 77].
|
||||
void analysis_77(const std::vector<double>& samples, std::size_t slots,
|
||||
std::size_t channels, QmfAnalysis& qmf,
|
||||
HybridAnalysis& hybrid_analysis, std::vector<Complex>* hybrid) {
|
||||
std::vector<double> hops(slots * channels * kQmfHop, 0.0);
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels; ++channel) {
|
||||
for (int index = 0; index < kQmfHop; ++index) {
|
||||
hops[(slot * channels + channel) * kQmfHop + static_cast<std::size_t>(index)] =
|
||||
samples[(slot * kQmfHop + static_cast<std::size_t>(index)) * channels + channel];
|
||||
}
|
||||
}
|
||||
}
|
||||
std::vector<Complex> qmf_bands;
|
||||
qmf.process(hops, slots, &qmf_bands);
|
||||
hybrid_analysis.process(qmf_bands, slots, hybrid);
|
||||
}
|
||||
|
||||
// public_filterbank.PublicSynthesis77.process: [slots, channels, 77] -> [slots*64, channels].
|
||||
void synthesis_77(const std::vector<Complex>& hybrid, std::size_t slots,
|
||||
std::size_t channels, const HybridSynthesis& synthesis,
|
||||
QmfSynthesis& qmf, std::vector<double>* time) {
|
||||
std::vector<Complex> qmf_bands;
|
||||
synthesis.process(hybrid, slots, channels, &qmf_bands);
|
||||
std::vector<double> samples;
|
||||
qmf.process(qmf_bands, slots, &samples);
|
||||
// The reference transposes (slots, channels, 64) to sample-major output.
|
||||
time->assign(samples.size(), 0.0);
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels; ++channel) {
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
(*time)[(slot * kQmfHop + static_cast<std::size_t>(band)) * channels + channel] =
|
||||
samples[(slot * channels + channel) * kQmfBands + static_cast<std::size_t>(band)];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
struct PublicFilterbank::Impl {
|
||||
Impl(const Kernels& kernels, std::size_t channels)
|
||||
: channels(channels), qmf(kernels, channels), hybrid_analysis(kernels, channels),
|
||||
hybrid_synthesis(kernels), qmf_synthesis(kernels, channels) {}
|
||||
|
||||
std::size_t channels;
|
||||
QmfAnalysis qmf;
|
||||
HybridAnalysis hybrid_analysis;
|
||||
HybridSynthesis hybrid_synthesis;
|
||||
QmfSynthesis qmf_synthesis;
|
||||
};
|
||||
|
||||
PublicFilterbank::PublicFilterbank(const Kernels& kernels, std::size_t channels)
|
||||
: impl_(std::make_unique<Impl>(kernels, channels)) {}
|
||||
|
||||
PublicFilterbank::~PublicFilterbank() = default;
|
||||
|
||||
void PublicFilterbank::reset() {
|
||||
impl_->qmf.reset();
|
||||
impl_->hybrid_analysis.reset();
|
||||
impl_->qmf_synthesis.reset();
|
||||
}
|
||||
|
||||
void PublicFilterbank::analyze_full_rate(const std::vector<double>& samples, std::size_t slots,
|
||||
std::vector<Complex>* hybrid) {
|
||||
analysis_77(samples, slots, impl_->channels, impl_->qmf, impl_->hybrid_analysis, hybrid);
|
||||
}
|
||||
|
||||
void PublicFilterbank::synthesize_full_rate(const std::vector<Complex>& hybrid, std::size_t slots,
|
||||
std::vector<double>* time) {
|
||||
synthesis_77(hybrid, slots, impl_->channels, impl_->hybrid_synthesis, impl_->qmf_synthesis,
|
||||
time);
|
||||
}
|
||||
|
||||
void PublicFilterbank::analyze_qmf(const std::vector<double>& hops, std::size_t slots,
|
||||
std::vector<Complex>* qmf) {
|
||||
impl_->qmf.process(hops, slots, qmf);
|
||||
}
|
||||
|
||||
void PublicFilterbank::analyze_hybrid(const std::vector<Complex>& qmf, std::size_t slots,
|
||||
std::vector<Complex>* hybrid) {
|
||||
impl_->hybrid_analysis.process(qmf, slots, hybrid);
|
||||
}
|
||||
|
||||
void PublicFilterbank::synthesize_hybrid(const std::vector<Complex>& hybrid, std::size_t slots,
|
||||
std::vector<Complex>* qmf) {
|
||||
impl_->hybrid_synthesis.process(hybrid, slots, impl_->channels, qmf);
|
||||
}
|
||||
|
||||
void PublicFilterbank::synthesize_qmf(const std::vector<Complex>& qmf, std::size_t slots,
|
||||
std::vector<double>* time) {
|
||||
impl_->qmf_synthesis.process(qmf, slots, time);
|
||||
}
|
||||
|
||||
} // namespace joc::hrtf
|
||||
|
||||
|
||||
@@ -1,55 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <complex>
|
||||
#include <cstddef>
|
||||
#include <memory>
|
||||
#include <vector>
|
||||
|
||||
#include "hrtf/jochrtf.h"
|
||||
|
||||
// Public 64-QMF / 77-hybrid filterbank, shared by the SOFA field compiler and the
|
||||
// Rosella renderer (upstream public_filterbank.py and rosella_filterbank.py are
|
||||
// the same bank). Everything is float64/complex128, as the reference computes it,
|
||||
// and the stateful half-steps are exposed because Rosella drives them directly.
|
||||
namespace joc::hrtf {
|
||||
|
||||
inline constexpr int kQmfBands = 64;
|
||||
inline constexpr int kQmfHop = 64;
|
||||
inline constexpr int kHybridLow = 16;
|
||||
inline constexpr int kHybridBandCount = 77;
|
||||
inline constexpr int kLatencySamples = 961;
|
||||
|
||||
using Complex = std::complex<double>;
|
||||
|
||||
class PublicFilterbank {
|
||||
public:
|
||||
PublicFilterbank(const Kernels& kernels, std::size_t channels);
|
||||
~PublicFilterbank();
|
||||
PublicFilterbank(const PublicFilterbank&) = delete;
|
||||
PublicFilterbank& operator=(const PublicFilterbank&) = delete;
|
||||
|
||||
void reset();
|
||||
|
||||
// Full-rate [slots*64, channels] -> hybrid [slots, channels, 77].
|
||||
void analyze_full_rate(const std::vector<double>& samples, std::size_t slots,
|
||||
std::vector<Complex>* hybrid);
|
||||
// Hybrid [slots, channels, 77] -> full-rate [slots*64, channels].
|
||||
void synthesize_full_rate(const std::vector<Complex>& hybrid, std::size_t slots,
|
||||
std::vector<double>* time);
|
||||
|
||||
// The stateful half-steps, in the order the reference runs them.
|
||||
void analyze_qmf(const std::vector<double>& hops, std::size_t slots,
|
||||
std::vector<Complex>* qmf);
|
||||
void analyze_hybrid(const std::vector<Complex>& qmf, std::size_t slots,
|
||||
std::vector<Complex>* hybrid);
|
||||
void synthesize_hybrid(const std::vector<Complex>& hybrid, std::size_t slots,
|
||||
std::vector<Complex>* qmf);
|
||||
void synthesize_qmf(const std::vector<Complex>& qmf, std::size_t slots,
|
||||
std::vector<double>* time);
|
||||
|
||||
private:
|
||||
struct Impl;
|
||||
std::unique_ptr<Impl> impl_;
|
||||
};
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -1,535 +0,0 @@
|
||||
#include "hrtf/rosella_model.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/fs_utf8.h"
|
||||
#include "foundation/mini_json.h"
|
||||
#include "foundation/sha256.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
namespace {
|
||||
|
||||
// The model's fixed-point lane scale: every stored value is a Q15 integer.
|
||||
constexpr float kQ15 = 1.0f / 32768.0f;
|
||||
|
||||
Status model_fail(joc_error code, const std::string& message) {
|
||||
return Status::fail(code, stage::kRender, message);
|
||||
}
|
||||
|
||||
float q15(std::int32_t value) { return static_cast<float>(value) * kQ15; }
|
||||
|
||||
float q15_exp(std::int32_t value, int exponent) {
|
||||
return q15(value) * static_cast<float>(std::ldexp(1.0, exponent));
|
||||
}
|
||||
|
||||
std::uint16_t low16(std::int32_t value) {
|
||||
return static_cast<std::uint16_t>(static_cast<std::uint32_t>(value) & 0xFFFFu);
|
||||
}
|
||||
|
||||
std::string trim(const std::string& text) {
|
||||
const std::size_t begin = text.find_first_not_of(" \t\r\n");
|
||||
const std::size_t end = text.find_last_not_of(" \t\r\n");
|
||||
return begin == std::string::npos ? std::string() : text.substr(begin, end - begin + 1u);
|
||||
}
|
||||
|
||||
// The lane array is read straight out of the JSON text: it is one flat list of
|
||||
// integers, and building a 15691-node DOM for it would only cost time.
|
||||
bool parse_int_array(const std::string& raw, std::vector<std::int32_t>* out, std::string* error) {
|
||||
out->clear();
|
||||
const char* cursor = raw.c_str();
|
||||
const char* end = cursor + raw.size();
|
||||
while (cursor < end && *cursor != '[') {
|
||||
++cursor;
|
||||
}
|
||||
if (cursor == end) {
|
||||
*error = "rosella_coefficients must be a JSON array";
|
||||
return false;
|
||||
}
|
||||
++cursor;
|
||||
while (cursor < end) {
|
||||
while (cursor < end && (*cursor == ' ' || *cursor == '\t' || *cursor == '\r' ||
|
||||
*cursor == '\n' || *cursor == ',')) {
|
||||
++cursor;
|
||||
}
|
||||
if (cursor >= end) {
|
||||
break;
|
||||
}
|
||||
if (*cursor == ']') {
|
||||
return true;
|
||||
}
|
||||
const bool negative = *cursor == '-';
|
||||
if (negative) {
|
||||
++cursor;
|
||||
}
|
||||
if (cursor >= end || *cursor < '0' || *cursor > '9') {
|
||||
*error = "rosella_coefficients contains a non-integer value";
|
||||
return false;
|
||||
}
|
||||
long long value = 0;
|
||||
while (cursor < end && *cursor >= '0' && *cursor <= '9') {
|
||||
value = value * 10 + (*cursor - '0');
|
||||
if (value > (1ll << 40)) {
|
||||
*error = "rosella_coefficients value is out of range";
|
||||
return false;
|
||||
}
|
||||
++cursor;
|
||||
}
|
||||
// A fractional part or an exponent means the value is not an exact integer.
|
||||
if (cursor < end && (*cursor == '.' || *cursor == 'e' || *cursor == 'E')) {
|
||||
*error = "rosella_coefficients contains a non-integer value";
|
||||
return false;
|
||||
}
|
||||
if (negative) {
|
||||
value = -value;
|
||||
}
|
||||
if (value < -(1ll << 31) || value > (1ll << 31) - 1) {
|
||||
*error = "rosella_coefficients value is outside signed int32";
|
||||
return false;
|
||||
}
|
||||
out->push_back(static_cast<std::int32_t>(value));
|
||||
}
|
||||
*error = "rosella_coefficients array is truncated";
|
||||
return false;
|
||||
}
|
||||
|
||||
struct RpHeader {
|
||||
std::uint16_t stored_checksum = 0;
|
||||
std::uint16_t computed_checksum = 0;
|
||||
bool checksum_valid = false;
|
||||
bool table_a_present = false;
|
||||
bool table_b_present = false;
|
||||
bool table_c_present = false;
|
||||
int table_a_dimension = 0;
|
||||
int table_a_option = 0;
|
||||
int table_a_extra = 0;
|
||||
int table_b_dimension = 0;
|
||||
int table_b_extra = 0;
|
||||
int table_b_groups = 0;
|
||||
int table_c_dimension = 0;
|
||||
std::size_t active_lanes = 0;
|
||||
};
|
||||
|
||||
Status inspect_rp(const std::vector<std::int32_t>& lanes, RpHeader* out) {
|
||||
if (lanes.size() < 5u) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella rp must contain whole int32 lanes");
|
||||
}
|
||||
if (low16(lanes[0]) != 0x7072u) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "bad Rosella rp magic");
|
||||
}
|
||||
out->stored_checksum = low16(lanes[1]);
|
||||
out->table_a_present = low16(lanes[2]) != 0u;
|
||||
out->table_b_present = low16(lanes[3]) != 0u;
|
||||
out->table_c_present = low16(lanes[4]) != 0u;
|
||||
std::size_t index = 5u;
|
||||
if (out->table_a_present) {
|
||||
out->table_a_dimension = low16(lanes[index]);
|
||||
out->table_a_option = low16(lanes[index + 1u]);
|
||||
out->table_a_extra = low16(lanes[index + 2u]);
|
||||
index += 5u;
|
||||
} else {
|
||||
out->table_a_dimension = 77;
|
||||
}
|
||||
if (out->table_b_present) {
|
||||
if (!out->table_a_present) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"Rosella rp table B cannot be present without table A");
|
||||
}
|
||||
out->table_b_dimension = low16(lanes[index]);
|
||||
out->table_b_extra = low16(lanes[index + 1u]);
|
||||
out->table_b_groups = low16(lanes[index + 2u]);
|
||||
index += 3u;
|
||||
}
|
||||
if (out->table_c_present) {
|
||||
out->table_c_dimension = low16(lanes[index]);
|
||||
index += 1u;
|
||||
}
|
||||
const long long payload_words =
|
||||
static_cast<long long>(index) - 2 +
|
||||
(out->table_b_present ? (out->table_b_dimension + 380 * out->table_b_groups +
|
||||
out->table_b_extra + 79)
|
||||
: 0) +
|
||||
(out->table_a_present ? (171 * out->table_a_extra + 79 +
|
||||
2 * (out->table_a_option + 14 * out->table_a_dimension))
|
||||
: 0) +
|
||||
11 + (out->table_c_present ? (314 * out->table_c_dimension + 1) : 0);
|
||||
if (payload_words < 0) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "malformed Rosella rp header");
|
||||
}
|
||||
out->active_lanes = static_cast<std::size_t>(2 + payload_words);
|
||||
if (lanes.size() < out->active_lanes) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella rp is truncated");
|
||||
}
|
||||
std::uint32_t computed = 0xA569u;
|
||||
for (std::size_t lane = 2u; lane < out->active_lanes; ++lane) {
|
||||
computed ^= low16(lanes[lane]);
|
||||
}
|
||||
out->computed_checksum = static_cast<std::uint16_t>(computed & 0xFFFFu);
|
||||
out->checksum_valid = out->computed_checksum == out->stored_checksum;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
// _unpack_field: the serialized 154-per-direction field lanes to the padded grid.
|
||||
void unpack_field(const std::int32_t* serialized, int directions, int exponent,
|
||||
std::vector<float>* padded) {
|
||||
padded->assign(static_cast<std::size_t>(160 * directions), 0.0f);
|
||||
const int stride8 = 8 * directions;
|
||||
const int stride2 = 2 * directions;
|
||||
for (int source = 0; source < 154 * directions; ++source) {
|
||||
const int group4 = (source % stride8) / stride2;
|
||||
const int destination = (group4 & 3) + 4 * (source % stride2 +
|
||||
2 * directions * (source / stride8 +
|
||||
(group4 >> 2)));
|
||||
(*padded)[static_cast<std::size_t>(destination)] =
|
||||
q15_exp(serialized[source], exponent);
|
||||
}
|
||||
}
|
||||
|
||||
// _unpack_table_a_grid: the serialized table-A rows to the padded lane grid.
|
||||
void unpack_table_a_grid(const std::int32_t* serialized, int dimension, int serialized_rows,
|
||||
int padded_rows, int lane_group, std::vector<float>* padded) {
|
||||
padded->assign(static_cast<std::size_t>(padded_rows) * static_cast<std::size_t>(dimension),
|
||||
0.0f);
|
||||
const int group_width = lane_group * 4;
|
||||
for (int source = 0; source < serialized_rows * dimension; ++source) {
|
||||
const int remainder = source % group_width;
|
||||
const int destination = (remainder / lane_group) +
|
||||
4 * (remainder % lane_group +
|
||||
group_width / 4 * (source / group_width));
|
||||
(*padded)[static_cast<std::size_t>(destination)] = q15(serialized[source]);
|
||||
}
|
||||
}
|
||||
|
||||
void unpack_table_a_extra(const std::int32_t* serialized, std::vector<float>* padded) {
|
||||
padded->assign(160u, 0.0f);
|
||||
for (int source = 0; source < 154; ++source) {
|
||||
const int remainder = source & 7;
|
||||
const int destination = (remainder >> 1) + 4 * ((source & 1) + 2 * (source >> 3));
|
||||
(*padded)[static_cast<std::size_t>(destination)] = q15(serialized[source]);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::string RosellaModel::summary() const {
|
||||
std::string name = capture.name.empty() ? std::string("unnamed") : capture.name;
|
||||
return "Rosella personalized_headphone '" + name + "' (" +
|
||||
(room_model.empty() ? std::string("unknown room") : room_model) + "), " +
|
||||
std::to_string(table_a_dimension) + " HQMF / 77 hybrid @ " +
|
||||
std::to_string(sample_rate) + " Hz";
|
||||
}
|
||||
|
||||
Status load_personalized_headphone(const std::string& path, RosellaModel* out) {
|
||||
if (out == nullptr) {
|
||||
return model_fail(JOC_ERR_INVALID_ARGUMENT, "null Rosella model destination");
|
||||
}
|
||||
if (!fs_utf8::exists(path)) {
|
||||
return model_fail(JOC_ERR_HRTF_NOT_FOUND, "personalized headphone model not found: " + path);
|
||||
}
|
||||
std::ifstream stream = fs_utf8::open_input(path);
|
||||
if (!stream.good()) {
|
||||
return model_fail(JOC_ERR_IO, "cannot open " + path);
|
||||
}
|
||||
std::string text((std::istreambuf_iterator<char>(stream)), std::istreambuf_iterator<char>());
|
||||
if (text.empty()) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "empty personalized headphone model: " + path);
|
||||
}
|
||||
// The checksum is taken over the coefficient lanes, exactly as upstream hashes
|
||||
// the int32 image of the array.
|
||||
const std::size_t first = text.find_first_not_of(" \t\r\n");
|
||||
if (first == std::string::npos || text[first] != '{') {
|
||||
return model_fail(JOC_ERR_NOT_SUPPORTED,
|
||||
"raw rp models are not supported; use a .personalized_headphone JSON");
|
||||
}
|
||||
std::vector<json::Member> root;
|
||||
std::string error;
|
||||
if (!json::parse_object(text, &root, &error)) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "invalid personalized headphone JSON: " + error);
|
||||
}
|
||||
const json::Member* personalized = json::find(root, "personalized_hrtf");
|
||||
if (personalized == nullptr) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "personalized_hrtf is missing");
|
||||
}
|
||||
std::vector<json::Member> inner;
|
||||
if (!json::parse_object(personalized->raw, &inner, &error)) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "invalid personalized_hrtf object: " + error);
|
||||
}
|
||||
const json::Member* virtualizer = json::find(inner, "virtualizer_parameters");
|
||||
if (virtualizer == nullptr) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "virtualizer_parameters is missing");
|
||||
}
|
||||
std::vector<json::Member> parameters;
|
||||
if (!json::parse_object(virtualizer->raw, ¶meters, &error)) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "invalid virtualizer_parameters: " + error);
|
||||
}
|
||||
const json::Member* coefficient_member = json::find(parameters, "rosella_coefficients");
|
||||
if (coefficient_member == nullptr) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "rosella_coefficients is missing");
|
||||
}
|
||||
RosellaModel model;
|
||||
model.source_path = path;
|
||||
if (const json::Member* member = json::find(parameters, "rosella_coefficients_version")) {
|
||||
json::as_string(*member, &model.coefficient_version);
|
||||
}
|
||||
if (const json::Member* member = json::find(parameters, "room_model")) {
|
||||
json::as_string(*member, &model.room_model);
|
||||
}
|
||||
if (const json::Member* capture = json::find(inner, "phrtf_capture_metadata")) {
|
||||
std::vector<json::Member> fields;
|
||||
if (json::parse_object(capture->raw, &fields, &error)) {
|
||||
const std::pair<const char*, std::string*> mapping[] = {
|
||||
{"capture_submission_date", &model.capture.capture_submission_date},
|
||||
{"capture_type", &model.capture.capture_type},
|
||||
{"label", &model.capture.label},
|
||||
{"name", &model.capture.name},
|
||||
{"phrtf_algorithm_version", &model.capture.algorithm_version},
|
||||
{"phrtf_creation_date", &model.capture.creation_date},
|
||||
{"uuid", &model.capture.uuid},
|
||||
{"version", &model.capture.version},
|
||||
};
|
||||
for (const auto& entry : mapping) {
|
||||
if (const json::Member* member = json::find(fields, entry.first)) {
|
||||
json::as_string(*member, entry.second);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
std::vector<std::int32_t> lanes;
|
||||
if (!parse_int_array(coefficient_member->raw, &lanes, &error)) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, error);
|
||||
}
|
||||
{
|
||||
crypto::Sha256 hash;
|
||||
hash.update(lanes.data(), lanes.size() * sizeof(std::int32_t));
|
||||
model.coefficient_sha256 = hash.finish_hex();
|
||||
}
|
||||
|
||||
RpHeader header;
|
||||
Status status = inspect_rp(lanes, &header);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (!header.checksum_valid || header.active_lanes != lanes.size()) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"invalid or non-active Rosella rp coefficient sequence");
|
||||
}
|
||||
if (!header.table_a_present || !header.table_b_present || header.table_c_present) {
|
||||
return model_fail(JOC_ERR_NOT_SUPPORTED,
|
||||
"the renderer requires table A+B and no table C");
|
||||
}
|
||||
if (header.table_a_dimension != 64 || header.table_a_option != 3) {
|
||||
return model_fail(JOC_ERR_NOT_SUPPORTED,
|
||||
"the renderer requires the observed 64-channel HQMF layout");
|
||||
}
|
||||
if (header.table_b_dimension != 20 || header.table_b_groups != 36) {
|
||||
return model_fail(JOC_ERR_NOT_SUPPORTED,
|
||||
"the renderer requires 20 hybrid groups and 36 direction terms");
|
||||
}
|
||||
|
||||
const std::int32_t* values = lanes.data();
|
||||
const std::size_t total = lanes.size();
|
||||
std::size_t position = 13u;
|
||||
const int extra = header.table_a_extra;
|
||||
model.table_a_dimension = header.table_a_dimension;
|
||||
model.table_a_option = header.table_a_option;
|
||||
model.table_a_extra = extra;
|
||||
model.table_a_header_field = low16(values[8]);
|
||||
model.table_a_header_25 = low16(values[9]);
|
||||
model.table_a_control = low16(values[position]);
|
||||
model.field_exponent = values[position];
|
||||
position += 1u;
|
||||
const int option_count = header.table_a_option;
|
||||
model.table_a_option_ids.resize(static_cast<std::size_t>(option_count));
|
||||
for (int index = 0; index < option_count; ++index) {
|
||||
model.table_a_option_ids[static_cast<std::size_t>(index)] =
|
||||
low16(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += static_cast<std::size_t>(option_count);
|
||||
model.table_a_option_values.resize(static_cast<std::size_t>(option_count));
|
||||
for (int index = 0; index < option_count; ++index) {
|
||||
model.table_a_option_values[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += static_cast<std::size_t>(option_count);
|
||||
model.table_a_scalar = q15(values[position]);
|
||||
position += 1u;
|
||||
const int dimension = header.table_a_dimension;
|
||||
unpack_table_a_grid(values + position, dimension, 16, 20, 16,
|
||||
&model.table_a_filter_16x64_padded);
|
||||
position += static_cast<std::size_t>(16 * dimension);
|
||||
for (int index = 0; index < 4; ++index) {
|
||||
model.table_a_four_integers[static_cast<std::size_t>(index)] =
|
||||
low16(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += 4u;
|
||||
model.table_a_integer = low16(values[position]);
|
||||
position += 1u;
|
||||
unpack_table_a_grid(values + position, dimension, 8, 10, 8,
|
||||
&model.table_a_filter_8x64_padded);
|
||||
position += static_cast<std::size_t>(8 * dimension);
|
||||
model.table_a_vector16.resize(16u);
|
||||
for (int index = 0; index < 16; ++index) {
|
||||
model.table_a_vector16[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += 16u;
|
||||
unpack_table_a_grid(values + position, dimension, 4, 5, 4,
|
||||
&model.table_a_filter_4x64_padded);
|
||||
position += static_cast<std::size_t>(4 * dimension);
|
||||
model.table_a_extra_indices.resize(static_cast<std::size_t>(extra));
|
||||
for (int index = 0; index < extra; ++index) {
|
||||
model.table_a_extra_indices[static_cast<std::size_t>(index)] =
|
||||
low16(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += static_cast<std::size_t>(extra);
|
||||
model.table_a_extra_fields_padded.assign(static_cast<std::size_t>(extra) * 160u, 0.0f);
|
||||
std::vector<float> unpacked;
|
||||
for (int index = 0; index < extra; ++index) {
|
||||
unpack_table_a_extra(values + position, &unpacked);
|
||||
std::copy(unpacked.begin(), unpacked.end(),
|
||||
model.table_a_extra_fields_padded.begin() + static_cast<std::ptrdiff_t>(index) * 160);
|
||||
position += 154u;
|
||||
}
|
||||
model.table_a_extra_vectors.assign(static_cast<std::size_t>(extra) * 16u, 0.0f);
|
||||
for (int index = 0; index < extra; ++index) {
|
||||
for (int lane = 0; lane < 16; ++lane) {
|
||||
model.table_a_extra_vectors[static_cast<std::size_t>(index) * 16u +
|
||||
static_cast<std::size_t>(lane)] =
|
||||
q15(values[position + static_cast<std::size_t>(lane)]);
|
||||
}
|
||||
position += 16u;
|
||||
}
|
||||
const std::size_t table_b_start = position;
|
||||
if (table_b_start != 13u + 1821u + static_cast<std::size_t>(171 * extra)) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella table-A parser lost its place");
|
||||
}
|
||||
|
||||
model.sample_rate = 2 * low16(values[position]);
|
||||
position += 1u;
|
||||
model.matrix_exponent = values[position];
|
||||
position += 1u;
|
||||
const std::size_t matrix_count = 36u * 36u;
|
||||
model.matrix_left.resize(matrix_count);
|
||||
model.matrix_right.resize(matrix_count);
|
||||
for (std::size_t index = 0; index < matrix_count; ++index) {
|
||||
model.matrix_left[index] = q15_exp(values[position + index], model.matrix_exponent);
|
||||
}
|
||||
position += matrix_count;
|
||||
for (std::size_t index = 0; index < matrix_count; ++index) {
|
||||
model.matrix_right[index] = q15_exp(values[position + index], model.matrix_exponent);
|
||||
}
|
||||
position += matrix_count;
|
||||
model.vector_left.resize(36u);
|
||||
model.vector_right.resize(36u);
|
||||
for (int index = 0; index < 36; ++index) {
|
||||
model.vector_left[static_cast<std::size_t>(index)] =
|
||||
q15_exp(values[position + static_cast<std::size_t>(index)], model.matrix_exponent);
|
||||
}
|
||||
position += 36u;
|
||||
for (int index = 0; index < 36; ++index) {
|
||||
model.vector_right[static_cast<std::size_t>(index)] =
|
||||
q15_exp(values[position + static_cast<std::size_t>(index)], model.matrix_exponent);
|
||||
}
|
||||
position += 36u;
|
||||
|
||||
const std::size_t serialized_count = 154u * 36u;
|
||||
unpack_field(values + position, 36, model.field_exponent, &model.field_left_padded);
|
||||
bool odd_zero = true;
|
||||
for (std::size_t index = 1u; index < serialized_count; index += 2u) {
|
||||
const float value = q15_exp(values[position + index], model.field_exponent);
|
||||
if (std::abs(value) > 1.0e-6f) {
|
||||
odd_zero = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
model.field_left_odd_serialized_zero = odd_zero;
|
||||
position += serialized_count;
|
||||
unpack_field(values + position, 36, model.field_exponent, &model.field_right_padded);
|
||||
position += serialized_count;
|
||||
|
||||
model.hybrid_flags.resize(20u);
|
||||
int active_hybrid = 0;
|
||||
for (int index = 0; index < 20; ++index) {
|
||||
model.hybrid_flags[static_cast<std::size_t>(index)] =
|
||||
low16(values[position + static_cast<std::size_t>(index)]);
|
||||
if (model.hybrid_flags[static_cast<std::size_t>(index)] == 1) {
|
||||
++active_hybrid;
|
||||
}
|
||||
}
|
||||
position += 20u;
|
||||
if (active_hybrid != header.table_b_extra) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "hybrid value count does not match the header");
|
||||
}
|
||||
model.hybrid_values.resize(static_cast<std::size_t>(active_hybrid));
|
||||
for (int index = 0; index < active_hybrid; ++index) {
|
||||
model.hybrid_values[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += static_cast<std::size_t>(active_hybrid);
|
||||
model.model_scalars.resize(5u);
|
||||
for (int index = 0; index < 5; ++index) {
|
||||
model.model_scalars[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += 5u;
|
||||
const std::size_t expected_tail =
|
||||
table_b_start + static_cast<std::size_t>(header.table_b_dimension +
|
||||
380 * header.table_b_groups +
|
||||
header.table_b_extra + 79);
|
||||
if (position != expected_tail) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella table-B parser lost its place");
|
||||
}
|
||||
|
||||
model.header_float_scalars[0] = q15(values[position]);
|
||||
model.header_float_scalars[1] = q15(values[position + 1u]) * 16.0f;
|
||||
model.header_integer_fields[0] = values[position + 2u];
|
||||
model.header_integer_fields[1] = low16(values[position + 3u]);
|
||||
position += 4u;
|
||||
for (int profile = 0; profile < 4; ++profile) {
|
||||
RosellaDistanceProfile parsed;
|
||||
for (int index = 0; index < 6; ++index) {
|
||||
parsed.bounds[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += 6u;
|
||||
parsed.distance_scale_m =
|
||||
q15_exp(values[position], values[position + 1u]);
|
||||
position += 2u;
|
||||
parsed.inverse_distance_per_m = q15(values[position]);
|
||||
parsed.axis_scales_internal[0] = q15(values[position + 1u]);
|
||||
parsed.axis_scales_internal[1] = q15(values[position + 2u]);
|
||||
parsed.axis_scales_internal[2] = q15(values[position + 3u]);
|
||||
parsed.minimum_normalized_radius = q15(values[position + 4u]);
|
||||
position += 5u;
|
||||
model.profiles[static_cast<std::size_t>(profile)] = parsed;
|
||||
}
|
||||
model.profile_tail.resize(8u);
|
||||
for (int index = 0; index < 8; ++index) {
|
||||
model.profile_tail[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += 8u;
|
||||
for (int index = 0; index < 3; ++index) {
|
||||
model.post_fields[static_cast<std::size_t>(index)] =
|
||||
values[position + static_cast<std::size_t>(index)];
|
||||
}
|
||||
position += 3u;
|
||||
if (position != total) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "unparsed Rosella coefficient lanes");
|
||||
}
|
||||
if (model.sample_rate != 48000) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"Rosella model sample rate must be 48000, got " +
|
||||
std::to_string(model.sample_rate));
|
||||
}
|
||||
*out = std::move(model);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -1,88 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
// Parser for the Dolby ".personalized_headphone" model (upstream rosella_model.py).
|
||||
// The file is JSON whose virtualizer_parameters carry the raw "rp" coefficient
|
||||
// lanes; everything the renderer needs is unpacked here, in the same float32
|
||||
// arithmetic the reference uses, because those values are part of the model.
|
||||
namespace joc::hrtf {
|
||||
|
||||
struct RosellaDistanceProfile {
|
||||
std::array<float, 6> bounds{};
|
||||
float distance_scale_m = 0.0f;
|
||||
float inverse_distance_per_m = 0.0f;
|
||||
std::array<float, 3> axis_scales_internal{};
|
||||
float minimum_normalized_radius = 0.0f;
|
||||
};
|
||||
|
||||
struct RosellaCaptureMetadata {
|
||||
std::string capture_submission_date;
|
||||
std::string capture_type;
|
||||
std::string label;
|
||||
std::string name;
|
||||
std::string algorithm_version;
|
||||
std::string creation_date;
|
||||
std::string uuid;
|
||||
std::string version;
|
||||
};
|
||||
|
||||
struct RosellaModel {
|
||||
std::string source_path;
|
||||
std::string coefficient_sha256;
|
||||
std::string coefficient_version;
|
||||
std::string room_model;
|
||||
RosellaCaptureMetadata capture;
|
||||
|
||||
int table_a_dimension = 0;
|
||||
int table_a_option = 0;
|
||||
int table_a_extra = 0;
|
||||
int table_a_header_field = 0;
|
||||
int table_a_header_25 = 0;
|
||||
int table_a_control = 0;
|
||||
std::vector<int> table_a_option_ids;
|
||||
std::vector<float> table_a_option_values;
|
||||
float table_a_scalar = 0.0f;
|
||||
std::vector<float> table_a_filter_16x64_padded;
|
||||
std::array<int, 4> table_a_four_integers{};
|
||||
int table_a_integer = 0;
|
||||
std::vector<float> table_a_filter_8x64_padded;
|
||||
std::vector<float> table_a_vector16;
|
||||
std::vector<float> table_a_filter_4x64_padded;
|
||||
std::vector<int> table_a_extra_indices;
|
||||
std::vector<float> table_a_extra_fields_padded;
|
||||
std::vector<float> table_a_extra_vectors;
|
||||
|
||||
int sample_rate = 0;
|
||||
int matrix_exponent = 0;
|
||||
int field_exponent = 0;
|
||||
std::vector<float> matrix_left;
|
||||
std::vector<float> matrix_right;
|
||||
std::vector<float> vector_left;
|
||||
std::vector<float> vector_right;
|
||||
std::vector<float> field_left_padded;
|
||||
std::vector<float> field_right_padded;
|
||||
bool field_left_odd_serialized_zero = false;
|
||||
std::vector<int> hybrid_flags;
|
||||
std::vector<float> hybrid_values;
|
||||
std::vector<float> model_scalars;
|
||||
std::array<float, 2> header_float_scalars{};
|
||||
std::array<int, 2> header_integer_fields{};
|
||||
std::array<RosellaDistanceProfile, 4> profiles{};
|
||||
std::vector<float> profile_tail;
|
||||
std::array<int, 3> post_fields{};
|
||||
|
||||
// One line for reports and logs: the capture name and room model are the
|
||||
// model's own strings, followed by the table layout and sample rate, e.g.
|
||||
// "Rosella personalized_headphone '<name>' (<room>), <N> HQMF / 77 hybrid @ <rate> Hz".
|
||||
std::string summary() const;
|
||||
};
|
||||
|
||||
Status load_personalized_headphone(const std::string& path, RosellaModel* out);
|
||||
|
||||
} // namespace joc::hrtf
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,64 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "hrtf/rosella_model.h"
|
||||
#include "oamd/oamd_parser.h"
|
||||
#include "timeline/position_timeline.h"
|
||||
|
||||
// Rosella ".personalized_headphone" binaural renderer (upstream rosella_core.py,
|
||||
// rosella_direct.py, rosella_room.py and rosella_binaural_renderer.py). It takes
|
||||
// the same pipeline slot as the SOFA runtime: sixteen object channels per frame in,
|
||||
// interleaved stereo out, with the OAMD timeline driving the per-block parameters.
|
||||
namespace joc::hrtf {
|
||||
|
||||
// rosella_direct.BINAURAL_PROFILE_NAMES.
|
||||
enum class RosellaProfile : std::int32_t { Near = 1, Far = 2, Mid = 3 };
|
||||
|
||||
struct RosellaRenderOptions {
|
||||
RosellaProfile profile = RosellaProfile::Mid;
|
||||
std::int64_t object_delay_samples = 1473;
|
||||
double tail_seconds = 5.0;
|
||||
double output_gain = 1.0;
|
||||
int chunk_frames = 64;
|
||||
int room_impulse_slots = 4096;
|
||||
};
|
||||
|
||||
class RosellaRuntime {
|
||||
public:
|
||||
RosellaRuntime();
|
||||
~RosellaRuntime();
|
||||
RosellaRuntime(const RosellaRuntime&) = delete;
|
||||
RosellaRuntime& operator=(const RosellaRuntime&) = delete;
|
||||
|
||||
Status open(const RosellaModel& model, const RosellaRenderOptions& options);
|
||||
|
||||
// objects16_planar is channel-major: channel * 1536 + sample.
|
||||
Status submit_frame(const float* objects16_planar, const oamd::OamdUpdate* update,
|
||||
std::int64_t frame_index, std::int64_t outer_sample_offset,
|
||||
std::int64_t object_delay_samples);
|
||||
|
||||
// Drains the flush tail: the pending partial chunk plus flush_samples of silence.
|
||||
Status finish(std::uint32_t flush_samples, std::vector<double>* out);
|
||||
std::uint32_t finish_capacity(double tail_seconds) const;
|
||||
|
||||
Status reset();
|
||||
|
||||
const std::vector<double>& output() const;
|
||||
void take_output(std::vector<double>* out);
|
||||
|
||||
std::uint64_t input_samples() const;
|
||||
std::uint64_t processed_input_samples() const;
|
||||
std::uint64_t metadata_block_updates() const;
|
||||
const timeline::OamdPositionTimeline& timeline() const;
|
||||
|
||||
private:
|
||||
struct Impl;
|
||||
std::unique_ptr<Impl> impl_;
|
||||
};
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -1,255 +0,0 @@
|
||||
#include "hrtf/sofa.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cctype>
|
||||
#include <cstdio>
|
||||
#include <fstream>
|
||||
#include <utility>
|
||||
|
||||
#include "foundation/fs_utf8.h"
|
||||
#include "foundation/sha256.h"
|
||||
#include "io/hdf5.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
namespace {
|
||||
|
||||
Status sofa_fail(joc_error code, const std::string& message) {
|
||||
return Status::fail(code, stage::kRender, message);
|
||||
}
|
||||
|
||||
std::string format_number(double value) {
|
||||
if (std::isfinite(value) && value == std::floor(value) && std::fabs(value) < 1.0e15) {
|
||||
return std::to_string(static_cast<long long>(value));
|
||||
}
|
||||
char buffer[32];
|
||||
std::snprintf(buffer, sizeof(buffer), "%.6g", value);
|
||||
return std::string(buffer);
|
||||
}
|
||||
|
||||
// Every array is checked against the element count the convention prescribes, so
|
||||
// a file whose shape disagrees with its metadata is rejected instead of silently
|
||||
// producing a shifted impulse response.
|
||||
Status read_doubles(const io::Hdf5File& file, const std::string& path, std::uint64_t expected,
|
||||
std::vector<double>* out) {
|
||||
if (!file.has_dataset(path)) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA file has no " + path + " dataset");
|
||||
}
|
||||
const Status status = file.read_dataset_double(path, out);
|
||||
if (!status.ok()) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA dataset " + path + ": " + status.message());
|
||||
}
|
||||
if (out->size() != expected) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"SOFA dataset " + path + " holds " + std::to_string(out->size()) +
|
||||
" values, expected " + std::to_string(expected));
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status read_text(const io::Hdf5File& file, const std::string& name, bool required,
|
||||
std::string* out) {
|
||||
io::Hdf5Attribute attribute;
|
||||
const Status status = file.attribute("", name, &attribute);
|
||||
if (!status.ok()) {
|
||||
if (required) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"SOFA file has no root attribute " + name + ": " + status.message());
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
*out = attribute.text;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
// SHA-256 of the whole file: the compiled-cache key is derived from it, so the
|
||||
// digest is taken over the exact bytes the parse consumed.
|
||||
std::string file_digest(const std::string& path) {
|
||||
std::ifstream stream = fs_utf8::open_input(path);
|
||||
if (!stream.good()) {
|
||||
return std::string();
|
||||
}
|
||||
crypto::Sha256 hash;
|
||||
std::vector<char> buffer(1u << 20);
|
||||
while (stream.good()) {
|
||||
stream.read(buffer.data(), static_cast<std::streamsize>(buffer.size()));
|
||||
const std::streamsize count = stream.gcount();
|
||||
if (count > 0) {
|
||||
hash.update(buffer.data(), static_cast<std::size_t>(count));
|
||||
}
|
||||
}
|
||||
return hash.finish_hex();
|
||||
}
|
||||
|
||||
// The coordinate declaration of one dataset, when the file carries it.
|
||||
void read_coordinates(const io::Hdf5File& file, const std::string& dataset,
|
||||
SofaCoordinate* out) {
|
||||
io::Hdf5Attribute attribute;
|
||||
if (file.attribute(dataset, "Type", &attribute).ok()) {
|
||||
out->type = attribute.text;
|
||||
}
|
||||
if (file.attribute(dataset, "Units", &attribute).ok()) {
|
||||
out->units = attribute.text;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::string SofaHrir::summary() const {
|
||||
return "SOFA " + sofa_conventions + ", " + std::to_string(ir_count) + " IRs x " +
|
||||
std::to_string(ir_length) + " taps @ " + format_number(sample_rate) + " Hz";
|
||||
}
|
||||
|
||||
Status load_sofa(const std::string& path, SofaHrir* out) {
|
||||
if (out == nullptr) {
|
||||
return sofa_fail(JOC_ERR_INVALID_ARGUMENT, "null SOFA destination");
|
||||
}
|
||||
if (!fs_utf8::exists(path)) {
|
||||
return sofa_fail(JOC_ERR_HRTF_NOT_FOUND, "SOFA file not found: " + path);
|
||||
}
|
||||
io::Hdf5File file;
|
||||
Status status = file.open(path);
|
||||
if (!status.ok()) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA file " + path + ": " + status.message());
|
||||
}
|
||||
|
||||
SofaHrir sofa;
|
||||
sofa.source_path = path;
|
||||
sofa.source_sha256 = file_digest(path);
|
||||
for (char& character : sofa.source_sha256) {
|
||||
character = static_cast<char>(std::toupper(static_cast<unsigned char>(character)));
|
||||
}
|
||||
status = read_text(file, "Conventions", true, &sofa.conventions);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "SOFAConventions", true, &sofa.sofa_conventions);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (sofa.conventions != "SOFA" || sofa.sofa_conventions != "SimpleFreeFieldHRIR") {
|
||||
return sofa_fail(JOC_ERR_HRTF_UNSUPPORTED_CONVENTION,
|
||||
"SOFA conventions " + sofa.conventions + "/" + sofa.sofa_conventions +
|
||||
" are not SimpleFreeFieldHRIR");
|
||||
}
|
||||
status = read_text(file, "SOFAConventionsVersion", false, &sofa.convention_version);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "Version", false, &sofa.version);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "DataType", false, &sofa.data_type);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "RoomType", false, &sofa.room_type);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "Title", false, &sofa.title);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "DatabaseName", false, &sofa.database_name);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "ListenerShortName", false, &sofa.listener_short_name);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "Comment", false, &sofa.comment);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
|
||||
io::Hdf5DatasetInfo info;
|
||||
status = file.dataset_info("Data.IR", &info);
|
||||
if (!status.ok()) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"SOFA file has no usable Data.IR dataset: " + status.message());
|
||||
}
|
||||
if (info.shape.size() != 3u || info.shape[1] != 2u || info.shape[0] == 0u ||
|
||||
info.shape[2] == 0u) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA Data.IR is not shaped (M, 2, N)");
|
||||
}
|
||||
if (info.shape[0] > 0xFFFFFFFFull || info.shape[2] > 0xFFFFFFFFull) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA Data.IR is larger than this reader accepts");
|
||||
}
|
||||
sofa.ir_count = static_cast<std::uint32_t>(info.shape[0]);
|
||||
sofa.ir_length = static_cast<std::uint32_t>(info.shape[2]);
|
||||
const std::uint64_t taps = static_cast<std::uint64_t>(sofa.ir_count) * 2u * sofa.ir_length;
|
||||
status = read_doubles(file, "Data.IR", taps, &sofa.ir);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
|
||||
std::vector<double> scalar;
|
||||
status = read_doubles(file, "Data.SamplingRate", 1u, &scalar);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
sofa.sample_rate = scalar[0];
|
||||
io::Hdf5Attribute attribute;
|
||||
if (file.attribute("Data.SamplingRate", "Units", &attribute).ok()) {
|
||||
sofa.sampling_rate_units = attribute.text;
|
||||
}
|
||||
|
||||
// Data.Delay is optional in the wild; absent means "no delay was measured".
|
||||
if (file.has_dataset("Data.Delay")) {
|
||||
std::vector<double> delay;
|
||||
status = read_doubles(file, "Data.Delay", 2u, &delay);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
sofa.delay[0] = delay[0];
|
||||
sofa.delay[1] = delay[1];
|
||||
}
|
||||
|
||||
const std::uint64_t measurements = sofa.ir_count;
|
||||
status = read_doubles(file, "SourcePosition", measurements * 3u, &sofa.source_position);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
read_coordinates(file, "SourcePosition", &sofa.source_position_coordinates);
|
||||
|
||||
std::vector<double> vector;
|
||||
status = read_doubles(file, "ListenerPosition", 3u, &vector);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::copy(vector.begin(), vector.end(), sofa.listener_position);
|
||||
read_coordinates(file, "ListenerPosition", &sofa.listener_position_coordinates);
|
||||
status = read_doubles(file, "ListenerView", 3u, &vector);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::copy(vector.begin(), vector.end(), sofa.listener_view);
|
||||
read_coordinates(file, "ListenerView", &sofa.listener_view_coordinates);
|
||||
status = read_doubles(file, "ListenerUp", 3u, &vector);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::copy(vector.begin(), vector.end(), sofa.listener_up);
|
||||
read_coordinates(file, "ListenerUp", &sofa.listener_up_coordinates);
|
||||
status = read_doubles(file, "EmitterPosition", 3u, &vector);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::copy(vector.begin(), vector.end(), sofa.emitter_position);
|
||||
read_coordinates(file, "EmitterPosition", &sofa.emitter_position_coordinates);
|
||||
status = read_doubles(file, "ReceiverPosition", 6u, &vector);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::copy(vector.begin(), vector.end(), sofa.receiver_position);
|
||||
read_coordinates(file, "ReceiverPosition", &sofa.receiver_position_coordinates);
|
||||
|
||||
*out = std::move(sofa);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -1,61 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
// Coordinate declaration of one SOFA variable: Type ("spherical"/"cartesian") and
|
||||
// Units. Empty when the file does not declare them (ListenerUp inherits).
|
||||
struct SofaCoordinate {
|
||||
std::string type;
|
||||
std::string units;
|
||||
};
|
||||
|
||||
// SOFA SimpleFreeFieldHRIR as this project consumes it: the impulse responses,
|
||||
// the measurement geometry and the metadata needed to report what was loaded.
|
||||
// All angles are degrees, all distances metres, exactly as the file stores them.
|
||||
struct SofaHrir {
|
||||
double sample_rate = 0.0;
|
||||
std::uint32_t ir_count = 0; // M: number of measurements
|
||||
std::uint32_t ir_length = 0; // N: taps per impulse response
|
||||
std::vector<double> ir; // C order [M][2][N]
|
||||
double delay[2] = {0.0, 0.0};
|
||||
std::vector<double> source_position; // M*3
|
||||
double listener_position[3] = {0.0, 0.0, 0.0};
|
||||
double listener_view[3] = {1.0, 0.0, 0.0};
|
||||
double listener_up[3] = {0.0, 0.0, 1.0};
|
||||
double emitter_position[3] = {0.0, 0.0, 0.0};
|
||||
double receiver_position[6] = {0.0, 0.0, 0.0, 0.0, 0.0, 0.0};
|
||||
std::string conventions;
|
||||
std::string sofa_conventions;
|
||||
std::string convention_version;
|
||||
std::string version;
|
||||
std::string data_type;
|
||||
std::string room_type;
|
||||
std::string title;
|
||||
std::string database_name;
|
||||
std::string listener_short_name;
|
||||
std::string comment;
|
||||
std::string sampling_rate_units;
|
||||
SofaCoordinate source_position_coordinates;
|
||||
SofaCoordinate listener_position_coordinates;
|
||||
SofaCoordinate listener_view_coordinates;
|
||||
SofaCoordinate listener_up_coordinates;
|
||||
SofaCoordinate emitter_position_coordinates;
|
||||
SofaCoordinate receiver_position_coordinates;
|
||||
// Identity of the file itself, needed for the compiled-cache key.
|
||||
std::string source_path;
|
||||
std::string source_sha256;
|
||||
|
||||
// One line for reports and logs:
|
||||
// "SOFA SimpleFreeFieldHRIR, <M> IRs x <N> taps @ <rate> Hz".
|
||||
std::string summary() const;
|
||||
};
|
||||
|
||||
Status load_sofa(const std::string& path, SofaHrir* out);
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -1,237 +0,0 @@
|
||||
#include "hrtf/sofa_cache.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <filesystem>
|
||||
#include <list>
|
||||
#include <mutex>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/fs_utf8.h"
|
||||
#include "hrtf/sofa.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
namespace {
|
||||
|
||||
namespace fs = std::filesystem;
|
||||
|
||||
// Small process-local cache: the reference keeps the last eight compiled fields.
|
||||
constexpr std::size_t kMemoryCacheEntries = 8;
|
||||
|
||||
struct MemoryEntry {
|
||||
std::string key;
|
||||
Field field;
|
||||
};
|
||||
|
||||
std::mutex& memory_mutex() {
|
||||
static std::mutex mutex;
|
||||
return mutex;
|
||||
}
|
||||
|
||||
std::list<MemoryEntry>& memory_cache() {
|
||||
static std::list<MemoryEntry> cache;
|
||||
return cache;
|
||||
}
|
||||
|
||||
bool memory_cache_get(const std::string& key, Field* out) {
|
||||
std::lock_guard<std::mutex> lock(memory_mutex());
|
||||
std::list<MemoryEntry>& cache = memory_cache();
|
||||
for (auto entry = cache.begin(); entry != cache.end(); ++entry) {
|
||||
if (entry->key == key) {
|
||||
*out = entry->field;
|
||||
cache.splice(cache.begin(), cache, entry);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void memory_cache_put(const std::string& key, const Field& field) {
|
||||
std::lock_guard<std::mutex> lock(memory_mutex());
|
||||
std::list<MemoryEntry>& cache = memory_cache();
|
||||
for (auto entry = cache.begin(); entry != cache.end(); ++entry) {
|
||||
if (entry->key == key) {
|
||||
entry->field = field;
|
||||
cache.splice(cache.begin(), cache, entry);
|
||||
return;
|
||||
}
|
||||
}
|
||||
cache.push_front(MemoryEntry{key, field});
|
||||
while (cache.size() > kMemoryCacheEntries) {
|
||||
cache.pop_back();
|
||||
}
|
||||
}
|
||||
|
||||
Status cache_fail(joc_error code, const std::string& message) {
|
||||
return Status::fail(code, stage::kRender, message);
|
||||
}
|
||||
|
||||
std::string upper(std::string text) {
|
||||
std::transform(text.begin(), text.end(), text.begin(), [](unsigned char value) {
|
||||
return static_cast<char>(std::toupper(value));
|
||||
});
|
||||
return text;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Status parse_cache_policy(const std::string& text, CachePolicy* out) {
|
||||
if (out == nullptr) {
|
||||
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "null cache policy");
|
||||
}
|
||||
if (text == "none") {
|
||||
*out = CachePolicy::None;
|
||||
return Status::success();
|
||||
}
|
||||
if (text == "memory") {
|
||||
*out = CachePolicy::Memory;
|
||||
return Status::success();
|
||||
}
|
||||
if (text == "disk") {
|
||||
*out = CachePolicy::Disk;
|
||||
return Status::success();
|
||||
}
|
||||
return cache_fail(JOC_ERR_INVALID_CONFIG, "cache_policy must be none, memory, or disk");
|
||||
}
|
||||
|
||||
Status validate_jochrtf(const std::string& path, const std::string& source_sha256,
|
||||
const std::string& cache_key, Field* out) {
|
||||
Field field;
|
||||
const Status status = load_jochrtf(path, &field);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (!source_sha256.empty() && field.source_sha256 != upper(source_sha256)) {
|
||||
return cache_fail(JOC_ERR_HRTF_HASH, "compiled HRTF source hash mismatch");
|
||||
}
|
||||
if (!cache_key.empty() && field.cache_key != upper(cache_key)) {
|
||||
return cache_fail(JOC_ERR_HRTF_HASH, "compiled HRTF configuration hash mismatch");
|
||||
}
|
||||
if (out != nullptr) {
|
||||
*out = std::move(field);
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status save_jochrtf_atomic(const Field& field, const std::string& path) {
|
||||
if (path.empty()) {
|
||||
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "empty compiled HRTF cache path");
|
||||
}
|
||||
const fs::path target = fs_utf8::to_path(path);
|
||||
std::error_code error;
|
||||
if (target.has_parent_path()) {
|
||||
fs::create_directories(target.parent_path(), error);
|
||||
if (error) {
|
||||
return cache_fail(JOC_ERR_OUTPUT_OPEN,
|
||||
"cannot create " + fs_utf8::from_path(target.parent_path()));
|
||||
}
|
||||
}
|
||||
const std::string temporary = path + ".tmp";
|
||||
Status status = write_jochrtf(field, temporary);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
// The rename is what makes a half-written cache impossible to observe.
|
||||
fs::rename(fs_utf8::to_path(temporary), target, error);
|
||||
if (error) {
|
||||
fs::remove(fs_utf8::to_path(temporary), error);
|
||||
return cache_fail(JOC_ERR_OUTPUT_WRITE, "cannot replace " + path);
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status load_or_compile_sofa_field(const SofaFieldRequest& request, Field* out,
|
||||
std::string* cache_path) {
|
||||
if (out == nullptr) {
|
||||
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "null compiled HRTF destination");
|
||||
}
|
||||
if (request.sofa_path.empty()) {
|
||||
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "no SOFA path for the compiled HRTF field");
|
||||
}
|
||||
if (request.policy == CachePolicy::Disk && request.cache_dir.empty()) {
|
||||
return cache_fail(JOC_ERR_INVALID_CONFIG, "the disk cache policy needs a cache directory");
|
||||
}
|
||||
SofaHrir sofa;
|
||||
Status status = load_sofa(request.sofa_path, &sofa);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
CanonicalHrtf canonical;
|
||||
status = canonicalize_sofa(sofa, &canonical);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
// The key depends on the shell that the radius selects, exactly as upstream.
|
||||
double actual_radius = request.options.shell_radius_m;
|
||||
(void)canonical_shell_indices(canonical, request.options.shell_radius_m, &actual_radius);
|
||||
const std::string key = compiled_hrtf_cache_key(
|
||||
canonical.source_sha256, canonical.sample_rate_hz, actual_radius, request.options.order,
|
||||
request.options.projection_ridge, request.options.sh_ridge);
|
||||
if (cache_path != nullptr) {
|
||||
cache_path->clear();
|
||||
}
|
||||
|
||||
std::string target;
|
||||
if (request.policy == CachePolicy::Disk) {
|
||||
target = request.cache_dir;
|
||||
if (!target.empty() && target.back() != '/' && target.back() != '\\') {
|
||||
target += "/";
|
||||
}
|
||||
target += cache_file_name(canonical.source_path.empty()
|
||||
? std::string()
|
||||
: canonical.source_path,
|
||||
key);
|
||||
if (fs_utf8::exists(target)) {
|
||||
Field cached_field;
|
||||
const Status cached =
|
||||
validate_jochrtf(target, canonical.source_sha256, key, &cached_field);
|
||||
if (cached.ok()) {
|
||||
memory_cache_put(key, cached_field);
|
||||
*out = std::move(cached_field);
|
||||
if (cache_path != nullptr) {
|
||||
*cache_path = target;
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
}
|
||||
}
|
||||
Field field;
|
||||
if (request.policy != CachePolicy::None && memory_cache_get(key, &field)) {
|
||||
// A memory hit still materialises the disk cache the caller asked for.
|
||||
if (request.policy == CachePolicy::Disk) {
|
||||
status = save_jochrtf_atomic(field, target);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (cache_path != nullptr) {
|
||||
*cache_path = target;
|
||||
}
|
||||
}
|
||||
*out = std::move(field);
|
||||
return Status::success();
|
||||
}
|
||||
status = compile_canonical_field(canonical, request.options, &field);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (field.cache_key != key) {
|
||||
return cache_fail(JOC_ERR_INTERNAL, "internal compiled HRTF cache-key mismatch");
|
||||
}
|
||||
if (request.policy == CachePolicy::Disk) {
|
||||
status = save_jochrtf_atomic(field, target);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (cache_path != nullptr) {
|
||||
*cache_path = target;
|
||||
}
|
||||
}
|
||||
if (request.policy != CachePolicy::None) {
|
||||
memory_cache_put(key, field);
|
||||
}
|
||||
*out = std::move(field);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -1,38 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <string>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "hrtf/jochrtf.h"
|
||||
#include "hrtf/sofa_field.h"
|
||||
|
||||
// Compiled-field cache: the .jochrtf is an internal artifact, so the caller only
|
||||
// names the SOFA file and the policy. "memory" keeps the compiled field in this
|
||||
// process, "disk" additionally reuses (and writes) <cache_dir>/<name>.<key>.jochrtf.
|
||||
namespace joc::hrtf {
|
||||
|
||||
enum class CachePolicy { None, Memory, Disk };
|
||||
|
||||
struct SofaFieldRequest {
|
||||
std::string sofa_path;
|
||||
CompileOptions options;
|
||||
CachePolicy policy = CachePolicy::Memory;
|
||||
std::string cache_dir; // required for the disk policy
|
||||
};
|
||||
|
||||
// Parses "none"/"memory"/"disk"; anything else is rejected.
|
||||
Status parse_cache_policy(const std::string& text, CachePolicy* out);
|
||||
|
||||
// Returns the compiled field, reusing a valid cache when the policy allows it.
|
||||
// `cache_path` (optional) receives the cache file that was read or written.
|
||||
Status load_or_compile_sofa_field(const SofaFieldRequest& request, Field* out,
|
||||
std::string* cache_path);
|
||||
|
||||
// Writes the field to `path` through a temporary file and an atomic rename.
|
||||
Status save_jochrtf_atomic(const Field& field, const std::string& path);
|
||||
|
||||
// Verifies that a cache file belongs to `source_sha256` and `cache_key`.
|
||||
Status validate_jochrtf(const std::string& path, const std::string& source_sha256,
|
||||
const std::string& cache_key, Field* out);
|
||||
|
||||
} // namespace joc::hrtf
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,103 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "hrtf/jochrtf.h"
|
||||
#include "hrtf/sofa.h"
|
||||
|
||||
// SOFA SimpleFreeFieldHRIR -> compiled directional field, ported from the
|
||||
// reference chain (sofa_canonical.py + sofa_hrtf_field.py + the public
|
||||
// filterbank): the measurement shell is selected, one delay representation is
|
||||
// separated, the FIRs are projected onto the 64-QMF/77-hybrid filterbank and the
|
||||
// result is fitted with fifth-order ACN/N3D real spherical harmonics.
|
||||
namespace joc::hrtf {
|
||||
|
||||
inline constexpr int kFieldOrder = 5;
|
||||
inline constexpr int kFieldTerms = 36;
|
||||
inline constexpr double kFieldSampleRateHz = 48000.0;
|
||||
inline constexpr double kDefaultShellRadiusM = 1.0;
|
||||
inline constexpr double kDefaultProjectionRidge = 1.0e-3;
|
||||
inline constexpr double kDefaultSphericalHarmonicRidge = 1.0e-5;
|
||||
inline constexpr const char* kCompilerVersion = "joc-sofa-compiler-v1";
|
||||
inline constexpr const char* kPhasePolicyVersion = "sofa-delay-exactly-once-v1";
|
||||
inline constexpr const char* kShConvention = "ACN/N3D real";
|
||||
inline constexpr const char* kFilterbankTableVersion = "joc-public-64qmf-77hybrid-v1";
|
||||
// SHA-256 of the standard filterbank archive the embedded tables came from. It
|
||||
// participates in the cache key, so it is part of the file-format contract.
|
||||
inline constexpr const char* kFilterbankArchiveSha256 =
|
||||
"C05BEF4D26E96ECBD4694E2572F05DA400255C777BA5047300B9D3B1F81081CD";
|
||||
|
||||
struct CompileOptions {
|
||||
double shell_radius_m = kDefaultShellRadiusM;
|
||||
int order = kFieldOrder;
|
||||
double projection_ridge = kDefaultProjectionRidge;
|
||||
double sh_ridge = kDefaultSphericalHarmonicRidge;
|
||||
};
|
||||
|
||||
// Canonical HRIR set: Data.IR and Data.Delay stay separate, the listener frame is
|
||||
// applied to the source positions and the ears are ordered left/right.
|
||||
struct CanonicalHrtf {
|
||||
std::string source_path;
|
||||
std::string source_sha256;
|
||||
std::string convention;
|
||||
std::string convention_version;
|
||||
std::string processing_label;
|
||||
double sample_rate_hz = 0.0;
|
||||
std::uint32_t measurements = 0;
|
||||
std::uint32_t taps = 0;
|
||||
int left_receiver_index = 0;
|
||||
int right_receiver_index = 1;
|
||||
std::vector<double> source_position_cartesian_m; // [M,3] listener-local
|
||||
std::vector<double> unit_directions; // [M,3]
|
||||
std::vector<double> measurement_radius_m; // [M]
|
||||
std::vector<double> hrir; // [M,2,N] canonical L/R
|
||||
std::vector<double> delay_samples; // [M,2], not applied
|
||||
};
|
||||
|
||||
// Port of load_simple_free_field_hrir(): strict SimpleFreeFieldHRIR import.
|
||||
Status canonicalize_sofa(const SofaHrir& sofa, CanonicalHrtf* out);
|
||||
|
||||
// Compiles the canonical set into the runtime field (port of SofaHrtfField.fit).
|
||||
Status compile_sofa_field(const SofaHrir& sofa, const CompileOptions& options, Field* out);
|
||||
|
||||
// The measurements on the shell nearest to radius_m; actual_radius_m receives the
|
||||
// mean radius of that shell (upstream CanonicalHrtf.shell_indices).
|
||||
std::vector<std::size_t> canonical_shell_indices(const CanonicalHrtf& canonical, double radius_m,
|
||||
double* actual_radius_m);
|
||||
|
||||
// Compiles an already canonicalized set (used by tests and the cache layer).
|
||||
Status compile_canonical_field(const CanonicalHrtf& canonical, const CompileOptions& options,
|
||||
Field* out);
|
||||
|
||||
// Configuration hash that names the cache file (upstream compiled_hrtf_cache_key).
|
||||
std::string compiled_hrtf_cache_key(const std::string& source_sha256, double sample_rate_hz,
|
||||
double shell_radius_m, int order, double projection_ridge,
|
||||
double sh_ridge);
|
||||
|
||||
// Payload hash over the four arrays (upstream _payload_sha256).
|
||||
std::string field_payload_sha256(const Field& field);
|
||||
|
||||
// "<stem>.<first 20 key digits>.jochrtf", the upstream cache file name.
|
||||
std::string cache_file_name(const std::string& display_name, const std::string& cache_key);
|
||||
|
||||
// Serializes the field as a .jochrtf cache the upstream loader also accepts.
|
||||
Status write_jochrtf(const Field& field, const std::string& path);
|
||||
|
||||
// The analysis/gain/synthesis dictionary the projection solves against (dev check).
|
||||
std::vector<double> hybrid_gain_synthesis_dictionary_for_check(std::size_t sample_count);
|
||||
|
||||
// Shell directions and their spherical Voronoi weights (dev check).
|
||||
void shell_directions_and_weights_for_check(const SofaHrir& sofa, double radius_m,
|
||||
std::vector<double>* directions,
|
||||
std::vector<double>* weights);
|
||||
|
||||
// PublicAnalysis77 on a unit impulse, interleaved complex (dev check).
|
||||
std::vector<double> analysis_impulse_for_check(std::size_t total_samples);
|
||||
|
||||
// The 77 hybrid-band centre frequencies at 48 kHz.
|
||||
const std::vector<double>& hybrid_band_center_frequencies_hz();
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -1,241 +0,0 @@
|
||||
#include "io/adm_writer.h"
|
||||
#include "foundation/fs_utf8.h"
|
||||
|
||||
#include <cstring>
|
||||
#include <filesystem>
|
||||
|
||||
#include "io/wav_writer.h" // pack_int24 (shared int24 quantisation)
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr long kDs64BodyOffset = 20;
|
||||
constexpr long kDataSizeOffset = 76;
|
||||
|
||||
void put_u16(std::string* out, std::uint16_t value) {
|
||||
char buffer[2];
|
||||
std::memcpy(buffer, &value, 2);
|
||||
out->append(buffer, 2);
|
||||
}
|
||||
|
||||
void put_u32(std::string* out, std::uint32_t value) {
|
||||
char buffer[4];
|
||||
std::memcpy(buffer, &value, 4);
|
||||
out->append(buffer, 4);
|
||||
}
|
||||
|
||||
void put_u64(std::string* out, std::uint64_t value) {
|
||||
char buffer[8];
|
||||
std::memcpy(buffer, &value, 8);
|
||||
out->append(buffer, 8);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
AdmBwfWriter::~AdmBwfWriter() { abort(); }
|
||||
|
||||
Status AdmBwfWriter::open(const std::string& path, std::size_t block_samples) {
|
||||
if (block_samples < JOC_FRAME_SAMPLES) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput,
|
||||
"ADM block size must hold at least one E-AC-3 frame");
|
||||
}
|
||||
path_ = path;
|
||||
block_samples_ = block_samples;
|
||||
used_ = 0;
|
||||
frames_ = 0;
|
||||
finalized_ = false;
|
||||
buffer_.assign(block_samples * kChannels, 0.0f);
|
||||
|
||||
file_ = fs_utf8::fopen(path, "wb+");
|
||||
if (file_ == nullptr) {
|
||||
std::error_code ignored;
|
||||
const std::filesystem::path parent = std::filesystem::path(path).parent_path();
|
||||
if (!parent.empty()) {
|
||||
std::filesystem::create_directories(parent, ignored);
|
||||
}
|
||||
file_ = fs_utf8::fopen(path, "wb+");
|
||||
}
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_OPEN, stage::kOutput, "cannot open " + path);
|
||||
}
|
||||
std::string header;
|
||||
header.append("RF64", 4);
|
||||
put_u32(&header, 0xFFFFFFFFu);
|
||||
header.append("WAVE", 4);
|
||||
if (std::fwrite(header.data(), 1, header.size(), file_) != header.size()) {
|
||||
abort();
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "cannot write " + path);
|
||||
}
|
||||
Status status = write_chunk("ds64", std::string(28, '\0'));
|
||||
if (!status.ok()) {
|
||||
abort();
|
||||
return status;
|
||||
}
|
||||
std::string fmt;
|
||||
put_u16(&fmt, 1);
|
||||
put_u16(&fmt, static_cast<std::uint16_t>(kChannels));
|
||||
put_u32(&fmt, kRate);
|
||||
put_u32(&fmt, kRate * kChannels * 3u);
|
||||
put_u16(&fmt, static_cast<std::uint16_t>(kChannels * 3u));
|
||||
put_u16(&fmt, 24);
|
||||
status = write_chunk("fmt ", fmt);
|
||||
if (!status.ok()) {
|
||||
abort();
|
||||
return status;
|
||||
}
|
||||
status = write_chunk("data", std::string());
|
||||
if (!status.ok()) {
|
||||
abort();
|
||||
return status;
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status AdmBwfWriter::write_chunk(const char id[4], const std::string& body) {
|
||||
std::string header;
|
||||
header.append(id, 4);
|
||||
put_u32(&header, static_cast<std::uint32_t>(body.size()));
|
||||
if (std::fwrite(header.data(), 1, header.size(), file_) != header.size()) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "chunk header write failed");
|
||||
}
|
||||
if (!body.empty() &&
|
||||
std::fwrite(body.data(), 1, body.size(), file_) != body.size()) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "chunk body write failed");
|
||||
}
|
||||
if ((body.size() & 1u) != 0u) {
|
||||
const char pad = '\0';
|
||||
if (std::fwrite(&pad, 1, 1, file_) != 1) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "chunk padding write failed");
|
||||
}
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status AdmBwfWriter::flush() {
|
||||
if (used_ == 0) {
|
||||
return Status::success();
|
||||
}
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer is not open");
|
||||
}
|
||||
packed_.clear();
|
||||
pack_int24(buffer_.data(), used_, kChannels, &packed_);
|
||||
if (std::fwrite(packed_.data(), 1, packed_.size(), file_) != packed_.size()) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "audio write failed for " + path_);
|
||||
}
|
||||
used_ = 0;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status AdmBwfWriter::write_objects16(const float* planar16) {
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer is not open");
|
||||
}
|
||||
if (planar16 == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput, "null frame");
|
||||
}
|
||||
std::size_t source = 0;
|
||||
while (source < JOC_FRAME_SAMPLES) {
|
||||
const std::size_t available = block_samples_ - used_;
|
||||
const std::size_t count =
|
||||
std::min(available, static_cast<std::size_t>(JOC_FRAME_SAMPLES) - source);
|
||||
float* target = buffer_.data() + used_ * kChannels;
|
||||
std::memset(target, 0, count * kChannels * sizeof(float));
|
||||
for (std::size_t sample = 0; sample < count; ++sample) {
|
||||
float* row = target + sample * kChannels;
|
||||
row[3] = planar16[0u * JOC_FRAME_SAMPLES + source + sample];
|
||||
for (std::size_t object = 0; object < 15u; ++object) {
|
||||
row[10u + object] =
|
||||
planar16[(object + 1u) * JOC_FRAME_SAMPLES + source + sample];
|
||||
}
|
||||
}
|
||||
used_ += count;
|
||||
source += count;
|
||||
if (used_ == block_samples_) {
|
||||
const Status status = flush();
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
}
|
||||
}
|
||||
frames_ += JOC_FRAME_SAMPLES;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status AdmBwfWriter::finalize(const std::string& axml, const std::string& chna,
|
||||
const std::string& dbmd) {
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer is not open");
|
||||
}
|
||||
if (finalized_) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer already finalized");
|
||||
}
|
||||
Status status = flush();
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = write_chunk("axml", axml);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = write_chunk("chna", chna);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = write_chunk("dbmd", dbmd);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
|
||||
if (std::fseek(file_, 0, SEEK_END) != 0) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "seek failed for " + path_);
|
||||
}
|
||||
const long long total = std::ftell(file_);
|
||||
if (total < 0) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "tell failed for " + path_);
|
||||
}
|
||||
const std::uint64_t data_len = frames_ * kChannels * 3u;
|
||||
const std::uint32_t data_field =
|
||||
data_len <= 0xFFFFFFFFull ? static_cast<std::uint32_t>(data_len) : 0xFFFFFFFFu;
|
||||
|
||||
if (std::fseek(file_, kDataSizeOffset, SEEK_SET) != 0) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "seek failed for " + path_);
|
||||
}
|
||||
char buffer[4];
|
||||
std::memcpy(buffer, &data_field, 4);
|
||||
if (std::fwrite(buffer, 1, 4, file_) != 4) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "data size patch failed");
|
||||
}
|
||||
|
||||
std::string ds64;
|
||||
put_u64(&ds64, static_cast<std::uint64_t>(total) - 8u);
|
||||
put_u64(&ds64, data_len);
|
||||
put_u64(&ds64, frames_);
|
||||
put_u32(&ds64, 0);
|
||||
if (std::fseek(file_, kDs64BodyOffset, SEEK_SET) != 0) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "seek failed for " + path_);
|
||||
}
|
||||
if (std::fwrite(ds64.data(), 1, ds64.size(), file_) != ds64.size()) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "ds64 patch failed");
|
||||
}
|
||||
finalized_ = true;
|
||||
if (std::fclose(file_) != 0) {
|
||||
file_ = nullptr;
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "close failed for " + path_);
|
||||
}
|
||||
file_ = nullptr;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
void AdmBwfWriter::abort() {
|
||||
if (file_ != nullptr) {
|
||||
std::fclose(file_);
|
||||
file_ = nullptr;
|
||||
}
|
||||
if (!finalized_ && !path_.empty()) {
|
||||
fs_utf8::remove(path_);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,52 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
class AdmBwfWriter {
|
||||
public:
|
||||
static constexpr std::uint32_t kChannels = 25;
|
||||
static constexpr std::uint32_t kRate = 48000;
|
||||
static constexpr std::size_t kDefaultBlockSamples = 131072;
|
||||
|
||||
AdmBwfWriter() = default;
|
||||
~AdmBwfWriter();
|
||||
|
||||
AdmBwfWriter(const AdmBwfWriter&) = delete;
|
||||
AdmBwfWriter& operator=(const AdmBwfWriter&) = delete;
|
||||
|
||||
Status open(const std::string& path, std::size_t block_samples = kDefaultBlockSamples);
|
||||
|
||||
Status write_objects16(const float* planar16);
|
||||
|
||||
Status finalize(const std::string& axml, const std::string& chna, const std::string& dbmd);
|
||||
|
||||
// Closes and removes a file that was never finalized (plan 28.3: abort must
|
||||
void abort();
|
||||
|
||||
std::uint64_t frames() const { return frames_; }
|
||||
bool open_ok() const { return file_ != nullptr; }
|
||||
|
||||
private:
|
||||
Status write_chunk(const char id[4], const std::string& body);
|
||||
Status flush();
|
||||
|
||||
std::FILE* file_ = nullptr;
|
||||
std::string path_;
|
||||
std::size_t block_samples_ = kDefaultBlockSamples;
|
||||
std::size_t used_ = 0;
|
||||
std::uint64_t frames_ = 0;
|
||||
std::vector<float> buffer_;
|
||||
std::string packed_;
|
||||
bool finalized_ = false;
|
||||
};
|
||||
|
||||
} // namespace joc::io
|
||||
-1430
File diff suppressed because it is too large
Load Diff
@@ -1,87 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
// Read-only subset of the HDF5 file format, sized for the SOFA files this
|
||||
// project consumes: superblock v0, version 2 object headers, fractal-heap link
|
||||
// and attribute storage, compact and contiguous datasets. The file is opened
|
||||
// lazily: only the requested dataset's bytes are read into memory, everything
|
||||
// else (superblock, object headers, heap blocks) is fetched on demand and the
|
||||
// metadata that was parsed is cached by file address.
|
||||
//
|
||||
// Paths are HDF5 link paths ("Data.IR" is a single link name here, "Group/Set"
|
||||
// walks two links); the empty path names the root group. Byte order is
|
||||
// normalized on read, so callers never see the file's own endianness.
|
||||
namespace joc::io {
|
||||
|
||||
enum class Hdf5Type {
|
||||
Unknown,
|
||||
Int8,
|
||||
Int16,
|
||||
Int32,
|
||||
Int64,
|
||||
UInt8,
|
||||
UInt16,
|
||||
UInt32,
|
||||
UInt64,
|
||||
Float32,
|
||||
Float64,
|
||||
String,
|
||||
};
|
||||
|
||||
struct Hdf5TypeInfo {
|
||||
Hdf5Type type = Hdf5Type::Unknown;
|
||||
std::uint32_t size = 0; // bytes per element as stored in the file
|
||||
bool big_endian = false;
|
||||
bool is_signed = false;
|
||||
};
|
||||
|
||||
struct Hdf5DatasetInfo {
|
||||
std::vector<std::uint64_t> shape;
|
||||
Hdf5TypeInfo type;
|
||||
|
||||
std::uint64_t element_count() const;
|
||||
};
|
||||
|
||||
struct Hdf5Attribute {
|
||||
Hdf5TypeInfo type;
|
||||
std::vector<std::uint64_t> shape;
|
||||
std::vector<std::uint8_t> raw; // C order, host byte order
|
||||
std::string text; // decoded for fixed-length string attributes
|
||||
};
|
||||
|
||||
class Hdf5File {
|
||||
public:
|
||||
Hdf5File();
|
||||
~Hdf5File();
|
||||
Hdf5File(Hdf5File&&) noexcept;
|
||||
Hdf5File& operator=(Hdf5File&&) noexcept;
|
||||
Hdf5File(const Hdf5File&) = delete;
|
||||
Hdf5File& operator=(const Hdf5File&) = delete;
|
||||
|
||||
Status open(const std::string& path);
|
||||
bool is_open() const;
|
||||
|
||||
// Names of the links of a group ("" is the root group).
|
||||
Status links(const std::string& group_path, std::vector<std::string>* names) const;
|
||||
|
||||
bool has_dataset(const std::string& path) const;
|
||||
Status dataset_info(const std::string& path, Hdf5DatasetInfo* out) const;
|
||||
Status read_dataset_raw(const std::string& path, std::vector<std::uint8_t>* out) const;
|
||||
Status read_dataset_double(const std::string& path, std::vector<double>* out) const;
|
||||
|
||||
Status attribute_names(const std::string& object_path, std::vector<std::string>* names) const;
|
||||
Status attribute(const std::string& object_path, const std::string& name, Hdf5Attribute* out) const;
|
||||
Status attribute_text(const std::string& object_path, const std::string& name, std::string* out) const;
|
||||
|
||||
private:
|
||||
struct Impl;
|
||||
std::unique_ptr<Impl> impl_;
|
||||
};
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,276 +0,0 @@
|
||||
#include "io/inflate.h"
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
class LsbBitReader {
|
||||
public:
|
||||
LsbBitReader(const std::uint8_t* data, std::size_t size) : data_(data), size_(size) {}
|
||||
|
||||
bool ok() const { return ok_; }
|
||||
std::size_t byte_position() const { return position_ >> 3; }
|
||||
|
||||
std::uint32_t bits(unsigned count) {
|
||||
std::uint32_t value = 0;
|
||||
for (unsigned i = 0; i < count; ++i) {
|
||||
if ((position_ >> 3) >= size_) {
|
||||
ok_ = false;
|
||||
return value;
|
||||
}
|
||||
const std::uint32_t bit = (data_[position_ >> 3] >> (position_ & 7u)) & 1u;
|
||||
value |= bit << i;
|
||||
++position_;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
void align_to_byte() { position_ = (position_ + 7u) & ~static_cast<std::size_t>(7u); }
|
||||
|
||||
void skip_bytes(std::size_t count) { position_ += count * 8u; }
|
||||
|
||||
private:
|
||||
const std::uint8_t* data_;
|
||||
std::size_t size_;
|
||||
std::size_t position_ = 0;
|
||||
bool ok_ = true;
|
||||
};
|
||||
|
||||
struct Huffman {
|
||||
std::uint16_t counts[16] = {};
|
||||
std::uint16_t symbols[288] = {};
|
||||
int max_length = 0;
|
||||
|
||||
bool build(const std::uint8_t* lengths, int count) {
|
||||
for (int i = 0; i < 16; ++i) {
|
||||
counts[i] = 0;
|
||||
}
|
||||
for (int i = 0; i < count; ++i) {
|
||||
counts[lengths[i]]++;
|
||||
}
|
||||
counts[0] = 0;
|
||||
std::uint16_t offsets[16] = {};
|
||||
std::uint16_t total = 0;
|
||||
for (int length = 1; length < 16; ++length) {
|
||||
offsets[length] = total;
|
||||
total = static_cast<std::uint16_t>(total + counts[length]);
|
||||
}
|
||||
if (total == 0) {
|
||||
return false;
|
||||
}
|
||||
for (int symbol = 0; symbol < count; ++symbol) {
|
||||
const std::uint8_t length = lengths[symbol];
|
||||
if (length != 0) {
|
||||
symbols[offsets[length]++] = static_cast<std::uint16_t>(symbol);
|
||||
}
|
||||
}
|
||||
max_length = 15;
|
||||
while (max_length > 0 && counts[max_length] == 0) {
|
||||
--max_length;
|
||||
}
|
||||
return max_length != 0;
|
||||
}
|
||||
|
||||
int decode(LsbBitReader* reader) const {
|
||||
int code = 0;
|
||||
int first = 0;
|
||||
int index = 0;
|
||||
for (int length = 1; length <= max_length; ++length) {
|
||||
code |= static_cast<int>(reader->bits(1));
|
||||
if (!reader->ok()) {
|
||||
return -1;
|
||||
}
|
||||
const int count = counts[length];
|
||||
if (code - first < count) {
|
||||
return symbols[index + (code - first)];
|
||||
}
|
||||
index += count;
|
||||
first = (first + count) << 1;
|
||||
code <<= 1;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
};
|
||||
|
||||
constexpr std::uint16_t kLengthBase[29] = {3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 15, 17, 19,
|
||||
23, 27, 31, 35, 43, 51, 59, 67, 83, 99, 115, 131, 163,
|
||||
195, 227, 258};
|
||||
constexpr std::uint8_t kLengthExtra[29] = {0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2,
|
||||
2, 3, 3, 3, 3, 4, 4, 4, 4, 5, 5, 5, 5, 0};
|
||||
constexpr std::uint16_t kDistanceBase[30] = {1, 2, 3, 4, 5, 7, 9, 13,
|
||||
17, 25, 33, 49, 65, 97, 129, 193,
|
||||
257, 385, 513, 769, 1025, 1537, 2049, 3073,
|
||||
4097, 6145, 8193, 12289, 16385, 24577};
|
||||
constexpr std::uint8_t kDistanceExtra[30] = {0, 0, 0, 0, 1, 1, 2, 2, 3, 3, 4, 4, 5, 5, 6,
|
||||
6, 7, 7, 8, 8, 9, 9, 10, 10, 11, 11, 12, 12, 13, 13};
|
||||
constexpr std::uint8_t kCodeLengthOrder[19] = {16, 17, 18, 0, 8, 7, 9, 6, 10, 5, 11, 4, 12, 3, 13, 2,
|
||||
14, 1, 15};
|
||||
|
||||
bool inflate_block_data(LsbBitReader* reader, const Huffman& literal, const Huffman& distance,
|
||||
std::vector<std::uint8_t>* out) {
|
||||
for (;;) {
|
||||
const int symbol = literal.decode(reader);
|
||||
if (symbol < 0) {
|
||||
return false;
|
||||
}
|
||||
if (symbol < 256) {
|
||||
out->push_back(static_cast<std::uint8_t>(symbol));
|
||||
continue;
|
||||
}
|
||||
if (symbol == 256) {
|
||||
return true;
|
||||
}
|
||||
const int length_index = symbol - 257;
|
||||
if (length_index >= 29) {
|
||||
return false;
|
||||
}
|
||||
const std::uint32_t length =
|
||||
kLengthBase[length_index] + reader->bits(kLengthExtra[length_index]);
|
||||
const int distance_symbol = distance.decode(reader);
|
||||
if (distance_symbol < 0 || distance_symbol >= 30) {
|
||||
return false;
|
||||
}
|
||||
const std::uint32_t distance_value =
|
||||
kDistanceBase[distance_symbol] + reader->bits(kDistanceExtra[distance_symbol]);
|
||||
if (!reader->ok() || distance_value == 0 || distance_value > out->size()) {
|
||||
return false;
|
||||
}
|
||||
const std::size_t start = out->size() - distance_value;
|
||||
for (std::uint32_t i = 0; i < length; ++i) {
|
||||
out->push_back((*out)[start + i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool inflate_fixed(LsbBitReader* reader, std::vector<std::uint8_t>* out) {
|
||||
std::uint8_t lengths[288];
|
||||
for (int i = 0; i < 144; ++i) { lengths[i] = 8; }
|
||||
for (int i = 144; i < 256; ++i) { lengths[i] = 9; }
|
||||
for (int i = 256; i < 280; ++i) { lengths[i] = 7; }
|
||||
for (int i = 280; i < 288; ++i) { lengths[i] = 8; }
|
||||
Huffman literal;
|
||||
if (!literal.build(lengths, 288)) {
|
||||
return false;
|
||||
}
|
||||
std::uint8_t distance_lengths[30];
|
||||
for (int i = 0; i < 30; ++i) { distance_lengths[i] = 5; }
|
||||
Huffman distance;
|
||||
if (!distance.build(distance_lengths, 30)) {
|
||||
return false;
|
||||
}
|
||||
return inflate_block_data(reader, literal, distance, out);
|
||||
}
|
||||
|
||||
bool inflate_dynamic(LsbBitReader* reader, std::vector<std::uint8_t>* out) {
|
||||
const int literal_count = static_cast<int>(reader->bits(5)) + 257;
|
||||
const int distance_count = static_cast<int>(reader->bits(5)) + 1;
|
||||
const int code_length_count = static_cast<int>(reader->bits(4)) + 4;
|
||||
if (!reader->ok() || literal_count > 286 || distance_count > 30) {
|
||||
return false;
|
||||
}
|
||||
std::uint8_t code_lengths[19] = {};
|
||||
for (int i = 0; i < code_length_count; ++i) {
|
||||
code_lengths[kCodeLengthOrder[i]] = static_cast<std::uint8_t>(reader->bits(3));
|
||||
}
|
||||
if (!reader->ok()) {
|
||||
return false;
|
||||
}
|
||||
Huffman code_length_tree;
|
||||
if (!code_length_tree.build(code_lengths, 19)) {
|
||||
return false;
|
||||
}
|
||||
std::uint8_t lengths[288 + 30] = {};
|
||||
const int total = literal_count + distance_count;
|
||||
int index = 0;
|
||||
while (index < total) {
|
||||
const int symbol = code_length_tree.decode(reader);
|
||||
if (symbol < 0) {
|
||||
return false;
|
||||
}
|
||||
if (symbol < 16) {
|
||||
lengths[index++] = static_cast<std::uint8_t>(symbol);
|
||||
continue;
|
||||
}
|
||||
int repeat = 0;
|
||||
std::uint8_t value = 0;
|
||||
if (symbol == 16) {
|
||||
if (index == 0) {
|
||||
return false;
|
||||
}
|
||||
value = lengths[index - 1];
|
||||
repeat = 3 + static_cast<int>(reader->bits(2));
|
||||
} else if (symbol == 17) {
|
||||
repeat = 3 + static_cast<int>(reader->bits(3));
|
||||
} else {
|
||||
repeat = 11 + static_cast<int>(reader->bits(7));
|
||||
}
|
||||
if (!reader->ok() || index + repeat > total) {
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < repeat; ++i) {
|
||||
lengths[index++] = value;
|
||||
}
|
||||
}
|
||||
Huffman literal;
|
||||
if (!literal.build(lengths, literal_count)) {
|
||||
return false;
|
||||
}
|
||||
Huffman distance;
|
||||
if (!distance.build(lengths + literal_count, distance_count)) {
|
||||
return false;
|
||||
}
|
||||
return inflate_block_data(reader, literal, distance, out);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool inflate_raw(const std::uint8_t* data, std::size_t size, std::vector<std::uint8_t>* out) {
|
||||
if (data == nullptr || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
out->clear();
|
||||
LsbBitReader reader(data, size);
|
||||
for (;;) {
|
||||
const std::uint32_t final_block = reader.bits(1);
|
||||
const std::uint32_t type = reader.bits(2);
|
||||
if (!reader.ok()) {
|
||||
return false;
|
||||
}
|
||||
if (type == 0) {
|
||||
reader.align_to_byte();
|
||||
const std::size_t position = reader.byte_position();
|
||||
if (position + 4 > size) {
|
||||
return false;
|
||||
}
|
||||
const std::uint16_t length = static_cast<std::uint16_t>(data[position] | (data[position + 1] << 8));
|
||||
const std::uint16_t complement =
|
||||
static_cast<std::uint16_t>(data[position + 2] | (data[position + 3] << 8));
|
||||
if (static_cast<std::uint16_t>(length ^ 0xFFFFu) != complement) {
|
||||
return false;
|
||||
}
|
||||
if (position + 4 + length > size) {
|
||||
return false;
|
||||
}
|
||||
out->insert(out->end(), data + position + 4, data + position + 4 + length);
|
||||
reader.skip_bytes(4u + length);
|
||||
} else if (type == 1) {
|
||||
if (!inflate_fixed(&reader, out)) {
|
||||
return false;
|
||||
}
|
||||
} else if (type == 2) {
|
||||
if (!inflate_dynamic(&reader, out)) {
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
if (final_block != 0u) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,12 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
bool inflate_raw(const std::uint8_t* data, std::size_t size, std::vector<std::uint8_t>* out);
|
||||
|
||||
} // namespace joc::io
|
||||
-420
@@ -1,420 +0,0 @@
|
||||
#include "io/npy.h"
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
std::uint16_t read_u16(const std::uint8_t* p) { return static_cast<std::uint16_t>(p[0] | (p[1] << 8)); }
|
||||
std::uint32_t read_u32(const std::uint8_t* p) {
|
||||
return static_cast<std::uint32_t>(p[0]) | (static_cast<std::uint32_t>(p[1]) << 8) |
|
||||
(static_cast<std::uint32_t>(p[2]) << 16) | (static_cast<std::uint32_t>(p[3]) << 24);
|
||||
}
|
||||
|
||||
NpyType classify(const std::string& descr) {
|
||||
if (descr == "<f8" || descr == "=f8" || descr == "|f8") { return NpyType::Float64; }
|
||||
if (descr == "<f4" || descr == "=f4") { return NpyType::Float32; }
|
||||
if (descr == "<i8" || descr == "=i8") { return NpyType::Int64; }
|
||||
if (descr == "<i4" || descr == "=i4") { return NpyType::Int32; }
|
||||
if (descr == "<i2" || descr == "=i2") { return NpyType::Int16; }
|
||||
if (descr == "|u1" || descr == "<u1") { return NpyType::UInt8; }
|
||||
if (descr == "<c16" || descr == "=c16") { return NpyType::Complex128; }
|
||||
if (descr.size() > 2 && descr[0] == '<' && descr[1] == 'U') {
|
||||
return NpyType::Unicode;
|
||||
}
|
||||
if (descr.size() > 2 && descr[0] == '=' && descr[1] == 'U') {
|
||||
return NpyType::Unicode;
|
||||
}
|
||||
return NpyType::Unknown;
|
||||
}
|
||||
|
||||
std::size_t unicode_length(const std::string& descr) {
|
||||
std::size_t index = 0;
|
||||
while (index < descr.size() && (descr[index] == '<' || descr[index] == '=')) {
|
||||
++index;
|
||||
}
|
||||
if (index >= descr.size() || descr[index] != 'U') {
|
||||
return 0;
|
||||
}
|
||||
++index;
|
||||
std::size_t value = 0;
|
||||
bool any = false;
|
||||
while (index < descr.size() && descr[index] >= '0' && descr[index] <= '9') {
|
||||
value = value * 10 + static_cast<std::size_t>(descr[index] - '0');
|
||||
++index;
|
||||
any = true;
|
||||
}
|
||||
return any ? value : 0;
|
||||
}
|
||||
|
||||
bool is_big_endian(const std::string& descr) { return !descr.empty() && descr[0] == '>'; }
|
||||
|
||||
bool header_value(const std::string& header, const std::string& key, std::string* out) {
|
||||
const std::string needle = "'" + key + "'";
|
||||
const std::size_t position = header.find(needle);
|
||||
if (position == std::string::npos) {
|
||||
return false;
|
||||
}
|
||||
const std::size_t colon = header.find(':', position + needle.size());
|
||||
if (colon == std::string::npos) {
|
||||
return false;
|
||||
}
|
||||
std::size_t start = colon + 1;
|
||||
while (start < header.size() && (header[start] == ' ' || header[start] == '\t')) {
|
||||
++start;
|
||||
}
|
||||
*out = header.substr(start);
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::size_t NpyArray::element_count() const {
|
||||
std::size_t count = 1;
|
||||
for (const std::int64_t dimension : shape) {
|
||||
count *= static_cast<std::size_t>(dimension < 0 ? 0 : dimension);
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
std::size_t NpyArray::element_size() const {
|
||||
switch (type) {
|
||||
case NpyType::Float64: return 8;
|
||||
case NpyType::Float32: return 4;
|
||||
case NpyType::Int64: return 8;
|
||||
case NpyType::Int32: return 4;
|
||||
case NpyType::Int16: return 2;
|
||||
case NpyType::UInt8: return 1;
|
||||
case NpyType::Complex128: return 16;
|
||||
case NpyType::Unicode: return item_bytes;
|
||||
default: return 0;
|
||||
}
|
||||
}
|
||||
|
||||
bool parse_npy(const std::uint8_t* data, std::size_t size, NpyArray* out, std::string* error) {
|
||||
if (data == nullptr || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
const std::uint8_t magic[6] = {0x93u, 'N', 'U', 'M', 'P', 'Y'};
|
||||
if (size < 10u || std::memcmp(data, magic, 6) != 0) {
|
||||
if (error != nullptr) { *error = "not a .npy image"; }
|
||||
return false;
|
||||
}
|
||||
const std::uint8_t major = data[6];
|
||||
std::size_t header_length = 0;
|
||||
std::size_t header_offset = 0;
|
||||
if (major == 1u) {
|
||||
header_length = read_u16(data + 8);
|
||||
header_offset = 10;
|
||||
} else if (major == 2u || major == 3u) {
|
||||
if (size < 12u) {
|
||||
if (error != nullptr) { *error = "truncated .npy v2 header"; }
|
||||
return false;
|
||||
}
|
||||
header_length = read_u32(data + 8);
|
||||
header_offset = 12;
|
||||
} else {
|
||||
if (error != nullptr) { *error = "unsupported .npy version " + std::to_string(major); }
|
||||
return false;
|
||||
}
|
||||
if (header_offset + header_length > size) {
|
||||
if (error != nullptr) { *error = "truncated .npy header"; }
|
||||
return false;
|
||||
}
|
||||
const std::string header(reinterpret_cast<const char*>(data + header_offset), header_length);
|
||||
|
||||
out->descr.clear();
|
||||
std::string value;
|
||||
if (!header_value(header, "descr", &value)) {
|
||||
if (error != nullptr) { *error = ".npy header without descr"; }
|
||||
return false;
|
||||
}
|
||||
const std::size_t first_quote = value.find('\'');
|
||||
const std::size_t second_quote =
|
||||
first_quote == std::string::npos ? std::string::npos : value.find('\'', first_quote + 1);
|
||||
if (first_quote == std::string::npos || second_quote == std::string::npos) {
|
||||
if (error != nullptr) { *error = ".npy descr is not a quoted string"; }
|
||||
return false;
|
||||
}
|
||||
out->descr = value.substr(first_quote + 1, second_quote - first_quote - 1);
|
||||
out->type = classify(out->descr);
|
||||
if (out->type == NpyType::Unknown) {
|
||||
if (error != nullptr) { *error = "unsupported .npy dtype " + out->descr; }
|
||||
return false;
|
||||
}
|
||||
out->item_bytes = 0;
|
||||
if (out->type == NpyType::Unicode) {
|
||||
const std::size_t length = unicode_length(out->descr);
|
||||
if (length == 0) {
|
||||
if (error != nullptr) { *error = "malformed unicode .npy dtype " + out->descr; }
|
||||
return false;
|
||||
}
|
||||
out->item_bytes = length * 4u;
|
||||
}
|
||||
|
||||
out->fortran_order = header.find("'fortran_order': True") != std::string::npos;
|
||||
|
||||
if (!header_value(header, "shape", &value)) {
|
||||
if (error != nullptr) { *error = ".npy header without shape"; }
|
||||
return false;
|
||||
}
|
||||
out->shape.clear();
|
||||
for (std::size_t i = 0; i < value.size(); ++i) {
|
||||
if (value[i] >= '0' && value[i] <= '9') {
|
||||
long long dimension = 0;
|
||||
while (i < value.size() && value[i] >= '0' && value[i] <= '9') {
|
||||
dimension = dimension * 10 + (value[i] - '0');
|
||||
++i;
|
||||
}
|
||||
out->shape.push_back(dimension);
|
||||
} else if (value[i] == ')') {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
const std::size_t expected = out->element_count() * out->element_size();
|
||||
if (header_offset + header_length + expected > size) {
|
||||
if (error != nullptr) {
|
||||
*error = ".npy payload truncated (need " + std::to_string(expected) + " bytes)";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
out->data = data + header_offset + header_length;
|
||||
out->data_bytes = expected;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool npy_shape_is(const NpyArray& array, const std::vector<std::int64_t>& expected) {
|
||||
return array.shape == expected;
|
||||
}
|
||||
|
||||
namespace {
|
||||
|
||||
template <typename T>
|
||||
void load_le(const std::uint8_t* source, std::size_t count, bool swap, std::vector<T>* out) {
|
||||
out->resize(count);
|
||||
std::memcpy(out->data(), source, count * sizeof(T));
|
||||
if (swap) {
|
||||
std::uint8_t* bytes = reinterpret_cast<std::uint8_t*>(out->data());
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
for (std::size_t b = 0; b < sizeof(T) / 2; ++b) {
|
||||
const std::uint8_t temporary = bytes[i * sizeof(T) + b];
|
||||
bytes[i * sizeof(T) + b] = bytes[i * sizeof(T) + sizeof(T) - 1 - b];
|
||||
bytes[i * sizeof(T) + sizeof(T) - 1 - b] = temporary;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool npy_to_double(const NpyArray& array, std::vector<double>* out, std::string* error) {
|
||||
const bool swap = is_big_endian(array.descr);
|
||||
const std::size_t count = array.element_count();
|
||||
switch (array.type) {
|
||||
case NpyType::Float64:
|
||||
load_le(array.data, count, swap, out);
|
||||
return true;
|
||||
case NpyType::Complex128:
|
||||
load_le(array.data, count * 2u, swap, out);
|
||||
return true;
|
||||
case NpyType::Float32: {
|
||||
std::vector<float> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<double>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::Int64: {
|
||||
std::vector<std::int64_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<double>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::Int32: {
|
||||
std::vector<std::int32_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<double>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::Int16: {
|
||||
std::vector<std::int16_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<double>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::UInt8: {
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<double>(array.data[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
default:
|
||||
if (error != nullptr) { *error = "cannot convert " + array.descr + " to double"; }
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool npy_to_int16(const NpyArray& array, std::vector<std::int16_t>* out, std::string* error) {
|
||||
const bool swap = is_big_endian(array.descr);
|
||||
const std::size_t count = array.element_count();
|
||||
switch (array.type) {
|
||||
case NpyType::Int16:
|
||||
load_le(array.data, count, swap, out);
|
||||
return true;
|
||||
case NpyType::Int32: {
|
||||
std::vector<std::int32_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<std::int16_t>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::Int64: {
|
||||
std::vector<std::int64_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<std::int16_t>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
default:
|
||||
if (error != nullptr) { *error = "cannot convert " + array.descr + " to int16"; }
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool npy_to_int32(const NpyArray& array, std::vector<std::int32_t>* out, std::string* error) {
|
||||
const bool swap = is_big_endian(array.descr);
|
||||
const std::size_t count = array.element_count();
|
||||
switch (array.type) {
|
||||
case NpyType::Int32:
|
||||
load_le(array.data, count, swap, out);
|
||||
return true;
|
||||
case NpyType::Int64: {
|
||||
std::vector<std::int64_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<std::int32_t>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::Int16: {
|
||||
std::vector<std::int16_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<std::int32_t>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
default:
|
||||
if (error != nullptr) { *error = "cannot convert " + array.descr + " to int32"; }
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool npy_to_uint8(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error) {
|
||||
if (array.type != NpyType::UInt8) {
|
||||
if (error != nullptr) { *error = "cannot convert " + array.descr + " to uint8"; }
|
||||
return false;
|
||||
}
|
||||
out->assign(array.data, array.data + array.element_count());
|
||||
return true;
|
||||
}
|
||||
|
||||
bool npy_unicode_to_utf8(const NpyArray& array, std::string* out, std::string* error) {
|
||||
if (array.type != NpyType::Unicode) {
|
||||
if (error != nullptr) { *error = "not a unicode .npy member: " + array.descr; }
|
||||
return false;
|
||||
}
|
||||
if (array.shape.size() != 0) {
|
||||
if (error != nullptr) { *error = "unicode .npy member must be a scalar"; }
|
||||
return false;
|
||||
}
|
||||
out->clear();
|
||||
const std::size_t count = array.item_bytes / 4u;
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
const std::uint8_t* p = array.data + i * 4u;
|
||||
const std::uint32_t code = static_cast<std::uint32_t>(p[0]) | (static_cast<std::uint32_t>(p[1]) << 8) |
|
||||
(static_cast<std::uint32_t>(p[2]) << 16) |
|
||||
(static_cast<std::uint32_t>(p[3]) << 24);
|
||||
if (code == 0u) {
|
||||
break;
|
||||
}
|
||||
if (code < 0x80u) {
|
||||
out->push_back(static_cast<char>(code));
|
||||
} else if (code < 0x800u) {
|
||||
out->push_back(static_cast<char>(0xC0u | (code >> 6)));
|
||||
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
|
||||
} else if (code < 0x10000u) {
|
||||
out->push_back(static_cast<char>(0xE0u | (code >> 12)));
|
||||
out->push_back(static_cast<char>(0x80u | ((code >> 6) & 0x3Fu)));
|
||||
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
|
||||
} else {
|
||||
out->push_back(static_cast<char>(0xF0u | (code >> 18)));
|
||||
out->push_back(static_cast<char>(0x80u | ((code >> 12) & 0x3Fu)));
|
||||
out->push_back(static_cast<char>(0x80u | ((code >> 6) & 0x3Fu)));
|
||||
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool npy_to_c_order(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error) {
|
||||
const std::size_t element = array.element_size();
|
||||
if (element == 0) {
|
||||
if (error != nullptr) { *error = "unsupported element size for " + array.descr; }
|
||||
return false;
|
||||
}
|
||||
if (!array.fortran_order) {
|
||||
out->assign(array.data, array.data + array.data_bytes);
|
||||
return true;
|
||||
}
|
||||
const std::size_t dimensions = array.shape.size();
|
||||
if (dimensions == 0) {
|
||||
out->assign(array.data, array.data + element);
|
||||
return true;
|
||||
}
|
||||
// Source (Fortran) strides in elements; destination is C order.
|
||||
std::vector<std::size_t> source_stride(dimensions, 1);
|
||||
std::size_t running = 1;
|
||||
for (std::size_t d = 0; d < dimensions; ++d) {
|
||||
source_stride[d] = running;
|
||||
running *= static_cast<std::size_t>(array.shape[d]);
|
||||
}
|
||||
out->assign(array.data_bytes, 0);
|
||||
std::vector<std::size_t> index(dimensions, 0);
|
||||
const std::size_t total = array.element_count();
|
||||
for (std::size_t linear = 0; linear < total; ++linear) {
|
||||
std::size_t remainder = linear;
|
||||
for (std::size_t d = dimensions; d-- > 0;) {
|
||||
index[d] = remainder % static_cast<std::size_t>(array.shape[d]);
|
||||
remainder /= static_cast<std::size_t>(array.shape[d]);
|
||||
}
|
||||
std::size_t source = 0;
|
||||
for (std::size_t d = 0; d < dimensions; ++d) {
|
||||
source += index[d] * source_stride[d];
|
||||
}
|
||||
std::memcpy(out->data() + linear * element, array.data + source * element, element);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,42 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
enum class NpyType { Unknown, Float64, Float32, Int64, Int32, Int16, UInt8, Complex128, Unicode };
|
||||
|
||||
struct NpyArray {
|
||||
std::string descr;
|
||||
NpyType type = NpyType::Unknown;
|
||||
bool fortran_order = false;
|
||||
std::vector<std::int64_t> shape;
|
||||
const std::uint8_t* data = nullptr;
|
||||
std::size_t data_bytes = 0;
|
||||
std::size_t item_bytes = 0; // bytes per element as stored
|
||||
|
||||
std::size_t element_count() const;
|
||||
std::size_t element_size() const; // bytes per element in the file
|
||||
};
|
||||
|
||||
// Parses the header of one `.npy` image. `data` must outlive the NpyArray.
|
||||
bool parse_npy(const std::uint8_t* data, std::size_t size, NpyArray* out, std::string* error);
|
||||
|
||||
bool npy_to_double(const NpyArray& array, std::vector<double>* out, std::string* error);
|
||||
bool npy_to_int16(const NpyArray& array, std::vector<std::int16_t>* out, std::string* error);
|
||||
bool npy_to_int32(const NpyArray& array, std::vector<std::int32_t>* out, std::string* error);
|
||||
bool npy_to_uint8(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error);
|
||||
|
||||
bool npy_unicode_to_utf8(const NpyArray& array, std::string* out, std::string* error);
|
||||
|
||||
// Materializes the array in C order as raw element bytes. Fortran-order members
|
||||
// hybrid synthesis table as [count][4] row-major while the shipped table stores it
|
||||
// Fortran-order, so passing the file bytes straight through would transpose it.
|
||||
bool npy_to_c_order(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error);
|
||||
|
||||
bool npy_shape_is(const NpyArray& array, const std::vector<std::int64_t>& expected);
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,263 +0,0 @@
|
||||
#include "io/npy_writer.h"
|
||||
|
||||
#include <array>
|
||||
#include <charconv>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
#include "foundation/fs_utf8.h"
|
||||
#include "io/zip_reader.h"
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr std::size_t kNpyHeaderAlignment = 64;
|
||||
|
||||
void append_u16(std::vector<std::uint8_t>* out, std::uint16_t value) {
|
||||
out->push_back(static_cast<std::uint8_t>(value & 0xFFu));
|
||||
out->push_back(static_cast<std::uint8_t>((value >> 8) & 0xFFu));
|
||||
}
|
||||
|
||||
void append_u32(std::vector<std::uint8_t>* out, std::uint32_t value) {
|
||||
for (int index = 0; index < 4; ++index) {
|
||||
out->push_back(static_cast<std::uint8_t>((value >> (8 * index)) & 0xFFu));
|
||||
}
|
||||
}
|
||||
|
||||
void append_bytes(std::vector<std::uint8_t>* out, const void* data, std::size_t size) {
|
||||
const std::uint8_t* bytes = static_cast<const std::uint8_t*>(data);
|
||||
out->insert(out->end(), bytes, bytes + size);
|
||||
}
|
||||
|
||||
std::string shape_literal(const std::vector<std::uint64_t>& shape) {
|
||||
if (shape.empty()) {
|
||||
return "()";
|
||||
}
|
||||
std::string text = "(";
|
||||
for (std::size_t index = 0; index < shape.size(); ++index) {
|
||||
if (index != 0u) {
|
||||
text += ", ";
|
||||
}
|
||||
text += std::to_string(shape[index]);
|
||||
}
|
||||
if (shape.size() == 1u) {
|
||||
text += ",";
|
||||
}
|
||||
text += ")";
|
||||
return text;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::vector<std::uint8_t> npy_image(const std::string& descr,
|
||||
const std::vector<std::uint64_t>& shape,
|
||||
const std::vector<std::uint8_t>& data) {
|
||||
std::string header = "{'descr': '" + descr + "', 'fortran_order': False, 'shape': " +
|
||||
shape_literal(shape) + ", }";
|
||||
// NumPy pads the header so that the payload starts on a 64-byte boundary.
|
||||
const std::size_t preamble = 10u; // magic, version, two byte header length
|
||||
std::size_t total = preamble + header.size() + 1u;
|
||||
const std::size_t padding = (kNpyHeaderAlignment - (total % kNpyHeaderAlignment)) %
|
||||
kNpyHeaderAlignment;
|
||||
header.append(padding, ' ');
|
||||
header.push_back('\n');
|
||||
|
||||
std::vector<std::uint8_t> out;
|
||||
out.reserve(preamble + header.size() + data.size());
|
||||
static const std::uint8_t kMagic[6] = {0x93u, 'N', 'U', 'M', 'P', 'Y'};
|
||||
append_bytes(&out, kMagic, sizeof(kMagic));
|
||||
out.push_back(1u); // major
|
||||
out.push_back(0u); // minor
|
||||
append_u16(&out, static_cast<std::uint16_t>(header.size()));
|
||||
append_bytes(&out, header.data(), header.size());
|
||||
append_bytes(&out, data.data(), data.size());
|
||||
return out;
|
||||
}
|
||||
|
||||
std::vector<std::uint8_t> zip_bytes(const std::vector<NpyMember>& members) {
|
||||
std::vector<std::uint8_t> out;
|
||||
struct Entry {
|
||||
std::string name;
|
||||
std::uint32_t crc = 0;
|
||||
std::uint32_t size = 0;
|
||||
std::uint32_t offset = 0;
|
||||
};
|
||||
std::vector<Entry> entries;
|
||||
entries.reserve(members.size());
|
||||
|
||||
for (const NpyMember& member : members) {
|
||||
const std::string name = member.name + ".npy";
|
||||
const std::vector<std::uint8_t> payload = npy_image(member.descr, member.shape, member.data);
|
||||
Entry entry;
|
||||
entry.name = name;
|
||||
entry.crc = crc32_of(payload.data(), payload.size());
|
||||
entry.size = static_cast<std::uint32_t>(payload.size());
|
||||
entry.offset = static_cast<std::uint32_t>(out.size());
|
||||
entries.push_back(entry);
|
||||
|
||||
append_u32(&out, 0x04034B50u); // local file header
|
||||
append_u16(&out, 20u); // version needed
|
||||
append_u16(&out, 0u); // flags
|
||||
append_u16(&out, 0u); // method: stored
|
||||
append_u16(&out, 0u); // time
|
||||
append_u16(&out, 0x2821u); // date: 2000-01-01, fixed for reproducibility
|
||||
append_u32(&out, entry.crc);
|
||||
append_u32(&out, entry.size);
|
||||
append_u32(&out, entry.size);
|
||||
append_u16(&out, static_cast<std::uint16_t>(name.size()));
|
||||
append_u16(&out, 0u); // extra length
|
||||
append_bytes(&out, name.data(), name.size());
|
||||
append_bytes(&out, payload.data(), payload.size());
|
||||
}
|
||||
|
||||
const std::uint32_t directory_offset = static_cast<std::uint32_t>(out.size());
|
||||
for (const Entry& entry : entries) {
|
||||
append_u32(&out, 0x02014B50u); // central directory header
|
||||
append_u16(&out, 20u); // version made by
|
||||
append_u16(&out, 20u); // version needed
|
||||
append_u16(&out, 0u); // flags
|
||||
append_u16(&out, 0u); // method: stored
|
||||
append_u16(&out, 0u); // time
|
||||
append_u16(&out, 0x2821u); // date
|
||||
append_u32(&out, entry.crc);
|
||||
append_u32(&out, entry.size);
|
||||
append_u32(&out, entry.size);
|
||||
append_u16(&out, static_cast<std::uint16_t>(entry.name.size()));
|
||||
append_u16(&out, 0u); // extra
|
||||
append_u16(&out, 0u); // comment
|
||||
append_u16(&out, 0u); // disk
|
||||
append_u16(&out, 0u); // internal attributes
|
||||
append_u32(&out, 0u); // external attributes
|
||||
append_u32(&out, entry.offset);
|
||||
append_bytes(&out, entry.name.data(), entry.name.size());
|
||||
}
|
||||
const std::uint32_t directory_size = static_cast<std::uint32_t>(out.size()) - directory_offset;
|
||||
|
||||
append_u32(&out, 0x06054B50u); // end of central directory
|
||||
append_u16(&out, 0u);
|
||||
append_u16(&out, 0u);
|
||||
append_u16(&out, static_cast<std::uint16_t>(entries.size()));
|
||||
append_u16(&out, static_cast<std::uint16_t>(entries.size()));
|
||||
append_u32(&out, directory_size);
|
||||
append_u32(&out, directory_offset);
|
||||
append_u16(&out, 0u);
|
||||
return out;
|
||||
}
|
||||
|
||||
bool write_zip(const std::string& path, const std::vector<NpyMember>& members,
|
||||
std::string* error) {
|
||||
const std::vector<std::uint8_t> bytes = zip_bytes(members);
|
||||
std::FILE* stream = fs_utf8::fopen(path, "wb");
|
||||
if (stream == nullptr) {
|
||||
if (error != nullptr) {
|
||||
*error = "cannot open " + path + " for writing";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
const std::size_t written = std::fwrite(bytes.data(), 1, bytes.size(), stream);
|
||||
const bool flushed = std::fclose(stream) == 0;
|
||||
if (written != bytes.size() || !flushed) {
|
||||
if (error != nullptr) {
|
||||
*error = "short write to " + path;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
std::vector<std::uint8_t> utf8_to_utf32le(const std::string& text) {
|
||||
std::vector<std::uint8_t> out;
|
||||
out.reserve(text.size() * 4u);
|
||||
std::size_t index = 0;
|
||||
while (index < text.size()) {
|
||||
const std::uint8_t lead = static_cast<std::uint8_t>(text[index]);
|
||||
std::uint32_t code = 0;
|
||||
std::size_t extra = 0;
|
||||
if (lead < 0x80u) {
|
||||
code = lead;
|
||||
} else if ((lead & 0xE0u) == 0xC0u) {
|
||||
code = lead & 0x1Fu;
|
||||
extra = 1;
|
||||
} else if ((lead & 0xF0u) == 0xE0u) {
|
||||
code = lead & 0x0Fu;
|
||||
extra = 2;
|
||||
} else if ((lead & 0xF8u) == 0xF0u) {
|
||||
code = lead & 0x07u;
|
||||
extra = 3;
|
||||
} else {
|
||||
code = 0xFFFDu; // invalid lead byte: substitute rather than fail
|
||||
extra = 0;
|
||||
}
|
||||
++index;
|
||||
for (std::size_t count = 0; count < extra && index < text.size(); ++count) {
|
||||
code = (code << 6) | (static_cast<std::uint8_t>(text[index]) & 0x3Fu);
|
||||
++index;
|
||||
}
|
||||
for (int byte = 0; byte < 4; ++byte) {
|
||||
out.push_back(static_cast<std::uint8_t>((code >> (8 * byte)) & 0xFFu));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
std::string python_float_repr(double value) {
|
||||
if (std::isnan(value)) {
|
||||
return "NaN";
|
||||
}
|
||||
if (std::isinf(value)) {
|
||||
return value > 0.0 ? "Infinity" : "-Infinity";
|
||||
}
|
||||
// to_chars gives the shortest round-trip digits; Python's repr uses the same
|
||||
// digits but its own notation, so the digits are re-laid-out here.
|
||||
std::array<char, 64> buffer{};
|
||||
const std::to_chars_result converted =
|
||||
std::to_chars(buffer.data(), buffer.data() + buffer.size(), value);
|
||||
std::string text(buffer.data(), converted.ptr);
|
||||
const bool negative = !text.empty() && text[0] == '-';
|
||||
const std::string body = negative ? text.substr(1) : text;
|
||||
const std::size_t exponent_at = body.find_first_of("eE");
|
||||
std::string digits = body;
|
||||
int exponent = 0;
|
||||
if (exponent_at != std::string::npos) {
|
||||
digits = body.substr(0, exponent_at);
|
||||
exponent = std::atoi(body.c_str() + exponent_at + 1);
|
||||
}
|
||||
const std::size_t point = digits.find('.');
|
||||
std::string mantissa = digits;
|
||||
if (point != std::string::npos) {
|
||||
mantissa = digits.substr(0, point) + digits.substr(point + 1);
|
||||
exponent += static_cast<int>(point) - 1;
|
||||
} else {
|
||||
exponent += static_cast<int>(digits.size()) - 1;
|
||||
}
|
||||
while (mantissa.size() > 1u && mantissa.back() == '0') {
|
||||
mantissa.pop_back();
|
||||
}
|
||||
// Python switches to exponent notation below 1e-4 and at 1e16 and above.
|
||||
std::string result;
|
||||
if (exponent < -4 || exponent >= 16) {
|
||||
result = mantissa.substr(0, 1);
|
||||
if (mantissa.size() > 1u) {
|
||||
result += "." + mantissa.substr(1);
|
||||
}
|
||||
char tail[16];
|
||||
std::snprintf(tail, sizeof(tail), "e%+03d", exponent);
|
||||
result += tail;
|
||||
} else if (exponent >= 0) {
|
||||
if (static_cast<std::size_t>(exponent) + 1u >= mantissa.size()) {
|
||||
result = mantissa + std::string(static_cast<std::size_t>(exponent) + 1u - mantissa.size(), '0');
|
||||
result += ".0";
|
||||
} else {
|
||||
result = mantissa.substr(0, static_cast<std::size_t>(exponent) + 1u) + "." +
|
||||
mantissa.substr(static_cast<std::size_t>(exponent) + 1u);
|
||||
}
|
||||
} else {
|
||||
result = "0." + std::string(static_cast<std::size_t>(-exponent - 1), '0') + mantissa;
|
||||
}
|
||||
return negative ? "-" + result : result;
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,39 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
// NPY 1.0 images and a minimal ZIP container, used to write the compiled HRTF
|
||||
// cache in exactly the layout the reader (and NumPy) expects. Only what the
|
||||
// cache needs is implemented: little-endian C-order arrays and stored members.
|
||||
struct NpyMember {
|
||||
std::string name; // archive member name, without the .npy suffix
|
||||
std::string descr; // NumPy dtype string, e.g. "<f8", "<c16", "<U123"
|
||||
std::vector<std::uint64_t> shape;
|
||||
std::vector<std::uint8_t> data; // C order payload in the dtype's byte order
|
||||
};
|
||||
|
||||
// Serializes one array as an NPY 1.0 image (magic, header, 64-byte aligned).
|
||||
std::vector<std::uint8_t> npy_image(const std::string& descr,
|
||||
const std::vector<std::uint64_t>& shape,
|
||||
const std::vector<std::uint8_t>& data);
|
||||
|
||||
// Writes a ZIP archive with stored (uncompressed) members. The upstream reader
|
||||
// accepts stored members, and compression would need a deflate encoder.
|
||||
bool write_zip(const std::string& path, const std::vector<NpyMember>& members,
|
||||
std::string* error);
|
||||
|
||||
// Serializes the archive in memory (same layout as write_zip).
|
||||
std::vector<std::uint8_t> zip_bytes(const std::vector<NpyMember>& members);
|
||||
|
||||
// UTF-8 text as the payload of a NumPy Unicode scalar string ('<U<n>').
|
||||
std::vector<std::uint8_t> utf8_to_utf32le(const std::string& text);
|
||||
|
||||
// Python's repr() for a double: shortest round-trip digits with Python's
|
||||
// exponent rules, which is what json.dumps emits for the cache metadata.
|
||||
std::string python_float_repr(double value);
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,186 +0,0 @@
|
||||
#include "io/process.h"
|
||||
#include "foundation/fs_utf8.h"
|
||||
|
||||
#include <cstdio>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <random>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#define NOMINMAX
|
||||
#include <windows.h>
|
||||
#else
|
||||
#include <sys/wait.h>
|
||||
#endif
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
std::string quote_argument(const std::string& argument) {
|
||||
if (!argument.empty() && argument.find_first_of(" \t\"") == std::string::npos) {
|
||||
return argument;
|
||||
}
|
||||
std::string quoted = "\"";
|
||||
unsigned backslashes = 0;
|
||||
for (const char c : argument) {
|
||||
if (c == '\\') {
|
||||
++backslashes;
|
||||
continue;
|
||||
}
|
||||
if (c == '"') {
|
||||
quoted.append(backslashes * 2 + 1, '\\');
|
||||
quoted.push_back('"');
|
||||
backslashes = 0;
|
||||
continue;
|
||||
}
|
||||
quoted.append(backslashes, '\\');
|
||||
backslashes = 0;
|
||||
quoted.push_back(c);
|
||||
}
|
||||
quoted.append(backslashes * 2, '\\');
|
||||
quoted.push_back('"');
|
||||
return quoted;
|
||||
}
|
||||
|
||||
std::string tail_of(const std::string& text, std::size_t limit) {
|
||||
if (text.size() <= limit) {
|
||||
return text;
|
||||
}
|
||||
return text.substr(text.size() - limit);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Status run_process(const std::vector<std::string>& argv, ProcessResult* out) {
|
||||
if (out == nullptr || argv.empty()) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput, "empty command");
|
||||
}
|
||||
out->output.clear();
|
||||
out->exit_code = 0;
|
||||
|
||||
const std::filesystem::path log_path =
|
||||
std::filesystem::temp_directory_path() /
|
||||
("joc_process_" + std::to_string(std::random_device{}()) + ".log");
|
||||
|
||||
auto read_log = [&]() {
|
||||
#if defined(_WIN32)
|
||||
return; // the Windows branch reads the handle it opened
|
||||
#else
|
||||
std::ifstream log = fs_utf8::open_input(fs_utf8::from_path(log_path));
|
||||
if (log) {
|
||||
std::string text((std::istreambuf_iterator<char>(log)),
|
||||
std::istreambuf_iterator<char>());
|
||||
out->output = tail_of(text, 4096);
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
#if defined(_WIN32)
|
||||
std::string command;
|
||||
for (std::size_t i = 0; i < argv.size(); ++i) {
|
||||
if (i != 0) {
|
||||
command.push_back(' ');
|
||||
}
|
||||
command += quote_argument(argv[i]);
|
||||
}
|
||||
auto widen = [](const std::string& text) {
|
||||
if (text.empty()) {
|
||||
return std::wstring();
|
||||
}
|
||||
const int size = MultiByteToWideChar(CP_UTF8, 0, text.c_str(),
|
||||
static_cast<int>(text.size()), nullptr, 0);
|
||||
std::wstring wide(static_cast<std::size_t>(size), L'\0');
|
||||
MultiByteToWideChar(CP_UTF8, 0, text.c_str(), static_cast<int>(text.size()), wide.data(),
|
||||
size);
|
||||
return wide;
|
||||
};
|
||||
const std::wstring wide_command = widen(command);
|
||||
const std::wstring wide_log = widen(fs_utf8::from_path(log_path));
|
||||
|
||||
SECURITY_ATTRIBUTES attributes{};
|
||||
attributes.nLength = sizeof(attributes);
|
||||
attributes.bInheritHandle = TRUE;
|
||||
// DELETE access plus FILE_FLAG_DELETE_ON_CLOSE means the log disappears when
|
||||
// the last handle goes away - including when this process is killed, which
|
||||
// would otherwise leave joc_process_*.log litter in the temp directory.
|
||||
HANDLE log_handle = CreateFileW(
|
||||
wide_log.c_str(), GENERIC_READ | GENERIC_WRITE | DELETE,
|
||||
FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, &attributes, CREATE_ALWAYS,
|
||||
FILE_ATTRIBUTE_NORMAL | FILE_FLAG_DELETE_ON_CLOSE, nullptr);
|
||||
if (log_handle == INVALID_HANDLE_VALUE) {
|
||||
return Status::fail(JOC_ERR_IO, stage::kOutput, "cannot create the process log file");
|
||||
}
|
||||
|
||||
// A delete-on-close file cannot be reopened by name (it is delete-pending), so
|
||||
// the child's output is read back through the handle it wrote to.
|
||||
auto read_log_handle = [&]() {
|
||||
LARGE_INTEGER start{};
|
||||
start.QuadPart = 0;
|
||||
if (!SetFilePointerEx(log_handle, start, nullptr, FILE_BEGIN)) {
|
||||
return;
|
||||
}
|
||||
std::string text;
|
||||
char buffer[1024];
|
||||
DWORD count = 0;
|
||||
while (ReadFile(log_handle, buffer, sizeof(buffer), &count, nullptr) && count > 0) {
|
||||
text.append(buffer, count);
|
||||
}
|
||||
out->output = tail_of(text, 4096);
|
||||
};
|
||||
|
||||
STARTUPINFOW startup{};
|
||||
startup.cb = sizeof(startup);
|
||||
startup.dwFlags = STARTF_USESTDHANDLES;
|
||||
startup.hStdOutput = log_handle;
|
||||
startup.hStdError = log_handle;
|
||||
startup.hStdInput = GetStdHandle(STD_INPUT_HANDLE);
|
||||
PROCESS_INFORMATION process{};
|
||||
|
||||
std::vector<wchar_t> mutable_command(wide_command.begin(), wide_command.end());
|
||||
mutable_command.push_back(L'\0');
|
||||
const BOOL started = CreateProcessW(nullptr, mutable_command.data(), nullptr, nullptr, TRUE,
|
||||
CREATE_NO_WINDOW, nullptr, nullptr, &startup, &process);
|
||||
if (!started) {
|
||||
CloseHandle(log_handle); // delete-on-close removes the file
|
||||
return Status::fail(JOC_ERR_LIBRARY_MISSING, stage::kOutput,
|
||||
"cannot start " + argv[0] + " (is it on PATH?)");
|
||||
}
|
||||
WaitForSingleObject(process.hProcess, INFINITE);
|
||||
DWORD exit_code = 0;
|
||||
GetExitCodeProcess(process.hProcess, &exit_code);
|
||||
CloseHandle(process.hThread);
|
||||
CloseHandle(process.hProcess);
|
||||
// Read the log before the delete-on-close handle goes away.
|
||||
read_log_handle();
|
||||
CloseHandle(log_handle);
|
||||
out->exit_code = static_cast<std::uint32_t>(exit_code);
|
||||
#else
|
||||
std::string command;
|
||||
for (std::size_t i = 0; i < argv.size(); ++i) {
|
||||
if (i != 0) {
|
||||
command.push_back(' ');
|
||||
}
|
||||
command += quote_argument(argv[i]);
|
||||
}
|
||||
command += " > " + quote_argument(fs_utf8::from_path(log_path)) + " 2>&1";
|
||||
const int status = std::system(command.c_str());
|
||||
// system() reports a wait status, not the child's exit code.
|
||||
out->exit_code = status == -1 ? 127u
|
||||
: WIFEXITED(status) ? static_cast<std::uint32_t>(WEXITSTATUS(status))
|
||||
: 128u;
|
||||
read_log();
|
||||
std::error_code ignored;
|
||||
std::filesystem::remove(log_path, ignored);
|
||||
#endif
|
||||
|
||||
if (out->exit_code != 0) {
|
||||
return Status::fail(JOC_ERR_INPUT_FORMAT, stage::kOutput,
|
||||
argv[0] + " failed with exit code " + std::to_string(out->exit_code) +
|
||||
(out->output.empty() ? "" : ": " + tail_of(out->output, 400)));
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,19 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
struct ProcessResult {
|
||||
std::uint32_t exit_code = 0;
|
||||
std::string output;
|
||||
};
|
||||
|
||||
Status run_process(const std::vector<std::string>& argv, ProcessResult* out);
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,242 +0,0 @@
|
||||
#include "io/wav_writer.h"
|
||||
#include "foundation/fs_utf8.h"
|
||||
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <filesystem>
|
||||
#include <limits>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr std::uint16_t kWaveFormatPcm = 0x0001;
|
||||
constexpr std::uint16_t kWaveFormatIeeeFloat = 0x0003;
|
||||
constexpr std::uint16_t kWaveFormatExtensible = 0xFFFE;
|
||||
constexpr std::uint8_t kPcmGuid[16] = {0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x10, 0x00,
|
||||
0x80, 0x00, 0x00, 0xAA, 0x00, 0x38, 0x9B, 0x71};
|
||||
constexpr std::uint8_t kFloatGuid[16] = {0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x10, 0x00,
|
||||
0x80, 0x00, 0x00, 0xAA, 0x00, 0x38, 0x9B, 0x71};
|
||||
|
||||
void put_u16(std::string* out, std::uint16_t value) {
|
||||
char buffer[2];
|
||||
std::memcpy(buffer, &value, 2);
|
||||
out->append(buffer, 2);
|
||||
}
|
||||
|
||||
void put_u32(std::string* out, std::uint32_t value) {
|
||||
char buffer[4];
|
||||
std::memcpy(buffer, &value, 4);
|
||||
out->append(buffer, 4);
|
||||
}
|
||||
|
||||
void put_u64(std::string* out, std::uint64_t value) {
|
||||
char buffer[8];
|
||||
std::memcpy(buffer, &value, 8);
|
||||
out->append(buffer, 8);
|
||||
}
|
||||
|
||||
// Port of speaker_wav._fmt_chunk.
|
||||
std::string fmt_chunk(std::uint32_t channels, std::uint32_t rate, SampleFormat format, WavInfo* info) {
|
||||
std::uint16_t simple_tag = 0;
|
||||
const std::uint8_t* guid = nullptr;
|
||||
if (format == SampleFormat::Float32) {
|
||||
info->bits_per_sample = 32;
|
||||
info->bytes_per_sample = 4;
|
||||
simple_tag = kWaveFormatIeeeFloat;
|
||||
guid = kFloatGuid;
|
||||
} else {
|
||||
info->bits_per_sample = 24;
|
||||
info->bytes_per_sample = 3;
|
||||
simple_tag = kWaveFormatPcm;
|
||||
guid = kPcmGuid;
|
||||
}
|
||||
const std::uint32_t block_align = channels * info->bytes_per_sample;
|
||||
const std::uint32_t byte_rate = rate * block_align;
|
||||
info->block_align = block_align;
|
||||
|
||||
std::string body;
|
||||
if (channels <= 2) {
|
||||
put_u16(&body, simple_tag);
|
||||
put_u16(&body, static_cast<std::uint16_t>(channels));
|
||||
put_u32(&body, rate);
|
||||
put_u32(&body, byte_rate);
|
||||
put_u16(&body, static_cast<std::uint16_t>(block_align));
|
||||
put_u16(&body, static_cast<std::uint16_t>(info->bits_per_sample));
|
||||
} else {
|
||||
put_u16(&body, kWaveFormatExtensible);
|
||||
put_u16(&body, static_cast<std::uint16_t>(channels));
|
||||
put_u32(&body, rate);
|
||||
put_u32(&body, byte_rate);
|
||||
put_u16(&body, static_cast<std::uint16_t>(block_align));
|
||||
put_u16(&body, static_cast<std::uint16_t>(info->bits_per_sample));
|
||||
put_u16(&body, 22);
|
||||
put_u16(&body, static_cast<std::uint16_t>(info->bits_per_sample));
|
||||
put_u32(&body, 0);
|
||||
body.append(reinterpret_cast<const char*>(guid), 16);
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
// int32 conversion identical to NumPy's float32 -> int32 cast after clipping.
|
||||
std::int32_t to_int32(const float value) {
|
||||
if (!std::isfinite(value)) {
|
||||
return std::numeric_limits<std::int32_t>::min();
|
||||
}
|
||||
return static_cast<std::int32_t>(value);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
void pack_int24(const float* interleaved, std::size_t frames, std::size_t channels,
|
||||
std::string* out) {
|
||||
const std::size_t count = frames * channels;
|
||||
out->resize(count * 3);
|
||||
char* target = out->data();
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
float value = interleaved[i];
|
||||
if (value > 1.0f) {
|
||||
value = 1.0f;
|
||||
} else if (value < -1.0f) {
|
||||
value = -1.0f;
|
||||
}
|
||||
const std::int32_t scaled = to_int32(value * 8388607.0f);
|
||||
const std::uint32_t bits = static_cast<std::uint32_t>(scaled);
|
||||
target[i * 3 + 0] = static_cast<char>(bits & 0xFFu);
|
||||
target[i * 3 + 1] = static_cast<char>((bits >> 8) & 0xFFu);
|
||||
target[i * 3 + 2] = static_cast<char>((bits >> 16) & 0xFFu);
|
||||
}
|
||||
}
|
||||
|
||||
WavWriter::~WavWriter() {
|
||||
if (file_ != nullptr) {
|
||||
std::fclose(file_);
|
||||
file_ = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
Status WavWriter::open(const std::string& path, std::uint32_t channels, std::uint32_t rate,
|
||||
SampleFormat format, std::uint64_t total_frames) {
|
||||
if (channels == 0 || rate == 0) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput,
|
||||
"WAV writer needs a positive channel count and rate");
|
||||
}
|
||||
path_ = path;
|
||||
channels_ = channels;
|
||||
total_frames_ = total_frames;
|
||||
frames_written_ = 0;
|
||||
finalized_ = false;
|
||||
info_ = WavInfo{};
|
||||
info_.format = format;
|
||||
|
||||
const std::string fmt = fmt_chunk(channels, rate, format, &info_);
|
||||
const std::uint64_t data_size = total_frames * info_.block_align;
|
||||
info_.data_bytes = data_size;
|
||||
const std::uint64_t riff_file_size = 12u + 8u + fmt.size() + 8u + data_size;
|
||||
const bool rf64 = (riff_file_size - 8u) > 0xFFFFFFFFull;
|
||||
info_.rf64 = rf64;
|
||||
|
||||
std::string header;
|
||||
if (rf64) {
|
||||
const std::uint64_t file_size = 12u + 36u + 8u + fmt.size() + 8u + data_size;
|
||||
header.append("RF64", 4);
|
||||
put_u32(&header, 0xFFFFFFFFu);
|
||||
header.append("WAVE", 4);
|
||||
header.append("ds64", 4);
|
||||
put_u32(&header, 28);
|
||||
put_u64(&header, file_size - 8u);
|
||||
put_u64(&header, data_size);
|
||||
put_u64(&header, total_frames);
|
||||
put_u32(&header, 0);
|
||||
} else {
|
||||
header.append("RIFF", 4);
|
||||
put_u32(&header, static_cast<std::uint32_t>(riff_file_size - 8u));
|
||||
header.append("WAVE", 4);
|
||||
}
|
||||
header.append("fmt ", 4);
|
||||
put_u32(&header, static_cast<std::uint32_t>(fmt.size()));
|
||||
header.append(fmt);
|
||||
header.append("data", 4);
|
||||
put_u32(&header, rf64 ? 0xFFFFFFFFu : static_cast<std::uint32_t>(data_size));
|
||||
|
||||
file_ = fs_utf8::fopen(path, "wb");
|
||||
if (file_ == nullptr) {
|
||||
// The reference creates the parent directory itself.
|
||||
std::error_code ignored;
|
||||
const std::filesystem::path parent = std::filesystem::path(path).parent_path();
|
||||
if (!parent.empty()) {
|
||||
std::filesystem::create_directories(parent, ignored);
|
||||
}
|
||||
file_ = fs_utf8::fopen(path, "wb");
|
||||
}
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_OPEN, stage::kOutput, "cannot open " + path);
|
||||
}
|
||||
if (std::fwrite(header.data(), 1, header.size(), file_) != header.size()) {
|
||||
std::fclose(file_);
|
||||
file_ = nullptr;
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "cannot write header to " + path);
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status WavWriter::write(const double* interleaved, std::size_t frames) {
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "WAV writer is not open");
|
||||
}
|
||||
if (frames == 0) {
|
||||
return Status::success();
|
||||
}
|
||||
if (frames_written_ + frames > total_frames_) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput,
|
||||
"WAV writer received more frames than the header declared (declared " +
|
||||
std::to_string(total_frames_) + ", written " +
|
||||
std::to_string(frames_written_) + ", requested " +
|
||||
std::to_string(frames) + ")");
|
||||
}
|
||||
const std::size_t count = frames * channels_;
|
||||
if (info_.format == SampleFormat::Float32) {
|
||||
std::vector<float> converted(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
converted[i] = static_cast<float>(interleaved[i]);
|
||||
}
|
||||
if (std::fwrite(converted.data(), sizeof(float), count, file_) != count) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "write failed for " + path_);
|
||||
}
|
||||
} else {
|
||||
std::vector<float> converted(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
converted[i] = static_cast<float>(interleaved[i]);
|
||||
}
|
||||
std::string packed;
|
||||
pack_int24(converted.data(), frames, channels_, &packed);
|
||||
if (std::fwrite(packed.data(), 1, packed.size(), file_) != packed.size()) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "write failed for " + path_);
|
||||
}
|
||||
}
|
||||
frames_written_ += frames;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status WavWriter::finalize() {
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "WAV writer is not open");
|
||||
}
|
||||
if (frames_written_ != total_frames_) {
|
||||
std::fclose(file_);
|
||||
file_ = nullptr;
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput,
|
||||
"WAV writer wrote " + std::to_string(frames_written_) + " of " +
|
||||
std::to_string(total_frames_) + " frames");
|
||||
}
|
||||
const int result = std::fclose(file_);
|
||||
file_ = nullptr;
|
||||
finalized_ = true;
|
||||
if (result != 0) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "close failed for " + path_);
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,61 +0,0 @@
|
||||
// Port of src/speaker_wav.py.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <string>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
enum class SampleFormat { Float32, Int24 };
|
||||
|
||||
struct WavInfo {
|
||||
SampleFormat format = SampleFormat::Float32;
|
||||
std::uint32_t bits_per_sample = 32;
|
||||
std::uint32_t bytes_per_sample = 4;
|
||||
std::uint32_t block_align = 0;
|
||||
std::uint64_t data_bytes = 0;
|
||||
bool rf64 = false;
|
||||
};
|
||||
|
||||
// int24 packing shared by the WAV and ADM writers:
|
||||
// trunc(clip(v, -1, 1) * 8388607.0f) with the low three bytes written LE.
|
||||
// NaN follows NumPy's float->int cast (INT32_MIN) so that the C++ conversion is
|
||||
// never undefined; the reference passes it through unguarded (plan TD-3.11).
|
||||
void pack_int24(const float* interleaved, std::size_t frames, std::size_t channels,
|
||||
std::string* out);
|
||||
|
||||
class WavWriter {
|
||||
public:
|
||||
WavWriter() = default;
|
||||
~WavWriter();
|
||||
|
||||
WavWriter(const WavWriter&) = delete;
|
||||
WavWriter& operator=(const WavWriter&) = delete;
|
||||
|
||||
// `total_frames` must be known up front: the header depends on it.
|
||||
Status open(const std::string& path, std::uint32_t channels, std::uint32_t rate,
|
||||
SampleFormat format, std::uint64_t total_frames);
|
||||
|
||||
Status write(const double* interleaved, std::size_t frames);
|
||||
|
||||
Status finalize();
|
||||
|
||||
const WavInfo& info() const { return info_; }
|
||||
std::uint64_t frames_written() const { return frames_written_; }
|
||||
|
||||
private:
|
||||
std::FILE* file_ = nullptr;
|
||||
std::string path_;
|
||||
WavInfo info_;
|
||||
std::uint32_t channels_ = 0;
|
||||
std::uint64_t total_frames_ = 0;
|
||||
std::uint64_t frames_written_ = 0;
|
||||
bool finalized_ = false;
|
||||
};
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,215 +0,0 @@
|
||||
#include "io/zip_reader.h"
|
||||
#include "foundation/fs_utf8.h"
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
|
||||
#include "io/inflate.h"
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr std::uint32_t kLocalHeaderSignature = 0x04034b50u;
|
||||
constexpr std::uint32_t kCentralHeaderSignature = 0x02014b50u;
|
||||
constexpr std::uint32_t kEndOfCentralDirectory = 0x06054b50u;
|
||||
|
||||
std::uint16_t read_u16(const std::uint8_t* p) {
|
||||
return static_cast<std::uint16_t>(p[0] | (p[1] << 8));
|
||||
}
|
||||
|
||||
std::uint32_t read_u32(const std::uint8_t* p) {
|
||||
return static_cast<std::uint32_t>(p[0]) | (static_cast<std::uint32_t>(p[1]) << 8) |
|
||||
(static_cast<std::uint32_t>(p[2]) << 16) | (static_cast<std::uint32_t>(p[3]) << 24);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::uint32_t crc32_of(const std::uint8_t* data, std::size_t size) {
|
||||
static std::uint32_t table[256];
|
||||
static bool ready = false;
|
||||
if (!ready) {
|
||||
for (std::uint32_t i = 0; i < 256; ++i) {
|
||||
std::uint32_t value = i;
|
||||
for (int bit = 0; bit < 8; ++bit) {
|
||||
value = (value & 1u) ? (0xEDB88320u ^ (value >> 1)) : (value >> 1);
|
||||
}
|
||||
table[i] = value;
|
||||
}
|
||||
ready = true;
|
||||
}
|
||||
std::uint32_t crc = 0xFFFFFFFFu;
|
||||
for (std::size_t i = 0; i < size; ++i) {
|
||||
crc = table[(crc ^ data[i]) & 0xFFu] ^ (crc >> 8);
|
||||
}
|
||||
return crc ^ 0xFFFFFFFFu;
|
||||
}
|
||||
|
||||
bool ZipArchive::open(const std::string& path, std::string* error) {
|
||||
entries_.clear();
|
||||
data_.clear();
|
||||
std::FILE* file = fs_utf8::fopen(path, "rb");
|
||||
if (file == nullptr) {
|
||||
if (error != nullptr) {
|
||||
*error = "cannot open " + path;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
std::fseek(file, 0, SEEK_END);
|
||||
const long long size = std::ftell(file);
|
||||
std::fseek(file, 0, SEEK_SET);
|
||||
if (size <= 0) {
|
||||
std::fclose(file);
|
||||
if (error != nullptr) {
|
||||
*error = "empty file " + path;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
data_.resize(static_cast<std::size_t>(size));
|
||||
const std::size_t got = std::fread(data_.data(), 1, data_.size(), file);
|
||||
std::fclose(file);
|
||||
if (got != data_.size()) {
|
||||
if (error != nullptr) {
|
||||
*error = "short read on " + path;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
std::size_t eocd = std::string::npos;
|
||||
const std::size_t scan_start = data_.size() > 65557u ? data_.size() - 65557u : 0u;
|
||||
for (std::size_t i = data_.size(); i-- > scan_start;) {
|
||||
if (i + 4u <= data_.size() && read_u32(&data_[i]) == kEndOfCentralDirectory) {
|
||||
eocd = i;
|
||||
break;
|
||||
}
|
||||
if (i == 0) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (eocd == std::string::npos || eocd + 22u > data_.size()) {
|
||||
if (error != nullptr) {
|
||||
*error = "not a zip archive (no end-of-central-directory)";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
const std::uint16_t entry_count = read_u16(&data_[eocd + 10]);
|
||||
const std::uint32_t directory_offset = read_u32(&data_[eocd + 16]);
|
||||
if (directory_offset >= data_.size()) {
|
||||
if (error != nullptr) {
|
||||
*error = "central directory offset out of range";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
std::size_t cursor = directory_offset;
|
||||
for (std::uint16_t index = 0; index < entry_count; ++index) {
|
||||
if (cursor + 46u > data_.size() || read_u32(&data_[cursor]) != kCentralHeaderSignature) {
|
||||
if (error != nullptr) {
|
||||
*error = "malformed central directory entry " + std::to_string(index);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
ZipEntry entry;
|
||||
entry.method = read_u16(&data_[cursor + 10]);
|
||||
entry.crc32 = read_u32(&data_[cursor + 16]);
|
||||
entry.compressed_size = read_u32(&data_[cursor + 20]);
|
||||
entry.uncompressed_size = read_u32(&data_[cursor + 24]);
|
||||
const std::uint16_t name_length = read_u16(&data_[cursor + 28]);
|
||||
const std::uint16_t extra_length = read_u16(&data_[cursor + 30]);
|
||||
const std::uint16_t comment_length = read_u16(&data_[cursor + 32]);
|
||||
entry.local_header_offset = read_u32(&data_[cursor + 42]);
|
||||
if (entry.compressed_size == 0xFFFFFFFFu || entry.uncompressed_size == 0xFFFFFFFFu ||
|
||||
entry.local_header_offset == 0xFFFFFFFFu) {
|
||||
if (error != nullptr) {
|
||||
*error = "zip64 archives are not supported";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (cursor + 46u + name_length > data_.size()) {
|
||||
if (error != nullptr) {
|
||||
*error = "member name out of range";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
entry.name.assign(reinterpret_cast<const char*>(&data_[cursor + 46]), name_length);
|
||||
entries_.push_back(std::move(entry));
|
||||
cursor += 46u + name_length + extra_length + comment_length;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
const ZipEntry* ZipArchive::find(const std::string& name) const {
|
||||
for (const ZipEntry& entry : entries_) {
|
||||
if (entry.name == name) {
|
||||
return &entry;
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
bool ZipArchive::extract(const ZipEntry& entry, std::vector<std::uint8_t>* out,
|
||||
std::string* error) const {
|
||||
if (out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
const std::size_t offset = entry.local_header_offset;
|
||||
if (offset + 30u > data_.size() || read_u32(&data_[offset]) != kLocalHeaderSignature) {
|
||||
if (error != nullptr) {
|
||||
*error = "bad local header for " + entry.name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
const std::uint16_t name_length = read_u16(&data_[offset + 26]);
|
||||
const std::uint16_t extra_length = read_u16(&data_[offset + 28]);
|
||||
const std::size_t start = offset + 30u + name_length + extra_length;
|
||||
if (start + entry.compressed_size > data_.size()) {
|
||||
if (error != nullptr) {
|
||||
*error = "member data out of range for " + entry.name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (entry.method == 0u) {
|
||||
out->assign(data_.begin() + static_cast<std::ptrdiff_t>(start),
|
||||
data_.begin() + static_cast<std::ptrdiff_t>(start + entry.compressed_size));
|
||||
} else if (entry.method == 8u) {
|
||||
if (!inflate_raw(&data_[start], entry.compressed_size, out)) {
|
||||
if (error != nullptr) {
|
||||
*error = "deflate error in " + entry.name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
if (error != nullptr) {
|
||||
*error = "unsupported compression method " + std::to_string(entry.method) + " for " +
|
||||
entry.name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (entry.uncompressed_size != 0u && out->size() != entry.uncompressed_size) {
|
||||
if (error != nullptr) {
|
||||
*error = "size mismatch for " + entry.name + " (" + std::to_string(out->size()) +
|
||||
" vs " + std::to_string(entry.uncompressed_size) + ")";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (entry.crc32 != 0u && crc32_of(out->data(), out->size()) != entry.crc32) {
|
||||
if (error != nullptr) {
|
||||
*error = "CRC mismatch for " + entry.name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ZipArchive::read_member(const std::string& name, std::vector<std::uint8_t>* out,
|
||||
std::string* error) const {
|
||||
const ZipEntry* entry = find(name);
|
||||
if (entry == nullptr) {
|
||||
if (error != nullptr) {
|
||||
*error = "member not found: " + name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return extract(*entry, out, error);
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,38 +0,0 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
struct ZipEntry {
|
||||
std::string name;
|
||||
std::uint16_t method = 0;
|
||||
std::uint32_t crc32 = 0;
|
||||
std::uint32_t compressed_size = 0;
|
||||
std::uint32_t uncompressed_size = 0;
|
||||
std::uint32_t local_header_offset = 0;
|
||||
};
|
||||
|
||||
class ZipArchive {
|
||||
public:
|
||||
bool open(const std::string& path, std::string* error);
|
||||
|
||||
const std::vector<ZipEntry>& entries() const { return entries_; }
|
||||
|
||||
const ZipEntry* find(const std::string& name) const;
|
||||
|
||||
bool extract(const ZipEntry& entry, std::vector<std::uint8_t>* out, std::string* error) const;
|
||||
|
||||
bool read_member(const std::string& name, std::vector<std::uint8_t>* out, std::string* error) const;
|
||||
|
||||
private:
|
||||
std::vector<std::uint8_t> data_;
|
||||
std::vector<ZipEntry> entries_;
|
||||
};
|
||||
|
||||
std::uint32_t crc32_of(const std::uint8_t* data, std::size_t size);
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -1,415 +0,0 @@
|
||||
#include "joc_bitstream/joc_parser.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
#include "foundation/bit_reader.h"
|
||||
#include "joc_huffman_tables.h"
|
||||
|
||||
namespace joc::joc {
|
||||
|
||||
namespace {
|
||||
|
||||
struct NumChannelsEntry {
|
||||
std::uint32_t config;
|
||||
std::int16_t channels;
|
||||
};
|
||||
constexpr NumChannelsEntry kNumChannels[] = {
|
||||
{0u, 5}, {1u, 7}, {2u, 7}, {3u, 5}, {4u, 7},
|
||||
};
|
||||
|
||||
struct NumBandsEntry {
|
||||
std::uint32_t index;
|
||||
std::int16_t bands;
|
||||
};
|
||||
constexpr NumBandsEntry kNumBands[] = {
|
||||
{0u, 1}, {1u, 3}, {2u, 5}, {3u, 7}, {4u, 9}, {5u, 12}, {6u, 15}, {7u, 23},
|
||||
};
|
||||
|
||||
// Floored modulo: Python's % operator semantics, so that the ported
|
||||
inline std::int64_t floored_mod(std::int64_t value, std::int64_t modulus) {
|
||||
const std::int64_t remainder = value % modulus;
|
||||
return remainder < 0 ? remainder + modulus : remainder;
|
||||
}
|
||||
|
||||
enum class SymbolKind { Mtx, Idx, Vec };
|
||||
|
||||
struct Tree {
|
||||
const int (*nodes)[2] = nullptr;
|
||||
int count = 0;
|
||||
};
|
||||
|
||||
Tree select_tree(std::uint32_t quant_idx, SymbolKind kind, int n_channels) {
|
||||
Tree tree;
|
||||
switch (kind) {
|
||||
case SymbolKind::Idx:
|
||||
if (n_channels == 5) {
|
||||
tree.nodes = joc_huff_code_5ch_pos_index_sparse;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_5ch_pos_index_sparse) /
|
||||
sizeof(joc_huff_code_5ch_pos_index_sparse[0]));
|
||||
} else {
|
||||
tree.nodes = joc_huff_code_7ch_pos_index_sparse;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_7ch_pos_index_sparse) /
|
||||
sizeof(joc_huff_code_7ch_pos_index_sparse[0]));
|
||||
}
|
||||
break;
|
||||
case SymbolKind::Vec:
|
||||
if (quant_idx == 0u) {
|
||||
tree.nodes = joc_huff_code_coarse_coeff_sparse;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_coarse_coeff_sparse) /
|
||||
sizeof(joc_huff_code_coarse_coeff_sparse[0]));
|
||||
} else {
|
||||
tree.nodes = joc_huff_code_fine_coeff_sparse;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_fine_coeff_sparse) /
|
||||
sizeof(joc_huff_code_fine_coeff_sparse[0]));
|
||||
}
|
||||
break;
|
||||
case SymbolKind::Mtx:
|
||||
default:
|
||||
if (quant_idx == 0u) {
|
||||
tree.nodes = joc_huff_code_coarse_generic;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_coarse_generic) /
|
||||
sizeof(joc_huff_code_coarse_generic[0]));
|
||||
} else {
|
||||
tree.nodes = joc_huff_code_fine_generic;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_fine_generic) /
|
||||
sizeof(joc_huff_code_fine_generic[0]));
|
||||
}
|
||||
break;
|
||||
}
|
||||
return tree;
|
||||
}
|
||||
|
||||
// infinite loop (plan 40.4 BL-6).
|
||||
bool huff_decode(const Tree& tree, bits::BitReader& reader, std::int16_t* out_value) {
|
||||
int node = 0;
|
||||
int steps = 0;
|
||||
while (node >= 0) {
|
||||
if (node >= tree.count || steps > tree.count) {
|
||||
reader.fail(JOC_ERR_JOC_SYNTAX, "Huffman tree walk left the valid node range");
|
||||
return false;
|
||||
}
|
||||
++steps;
|
||||
const std::uint32_t bit = reader.read(1);
|
||||
if (reader.failed()) {
|
||||
return false;
|
||||
}
|
||||
node = tree.nodes[node][bit];
|
||||
}
|
||||
*out_value = static_cast<std::int16_t>(-node - 1);
|
||||
return true;
|
||||
}
|
||||
|
||||
Status syntax_fail(const std::string& message) {
|
||||
return Status::fail(JOC_ERR_JOC_SYNTAX, stage::kJoc, message);
|
||||
}
|
||||
|
||||
Status truncated_fail(const bits::BitReader& reader) {
|
||||
if (reader.error() == JOC_ERR_JOC_SYNTAX) {
|
||||
return syntax_fail(reader.error_message());
|
||||
}
|
||||
return Status::fail(JOC_ERR_BITSTREAM_TRUNCATED, stage::kJoc,
|
||||
std::string("JOC bitstream truncated: ") + reader.error_message());
|
||||
}
|
||||
|
||||
void reconstruct_dense(const ObjectSymbols& symbols, std::uint32_t dp, int n_channels, std::int64_t nquant,
|
||||
std::int64_t offset,
|
||||
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS]) {
|
||||
for (int ch = 0; ch < n_channels; ++ch) {
|
||||
q[dp][ch][0] =
|
||||
floored_mod(offset + static_cast<std::int64_t>(symbols.mtx[dp][ch][0]), nquant);
|
||||
for (int pb = 1; pb < symbols.n_bands; ++pb) {
|
||||
q[dp][ch][pb] = floored_mod(
|
||||
q[dp][ch][pb - 1] + static_cast<std::int64_t>(symbols.mtx[dp][ch][pb]), nquant);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// across parameter bands and is deliberately NOT reset when the active channel
|
||||
Status reconstruct_sparse(
|
||||
const ObjectSymbols& symbols, std::uint32_t dp, int n_channels, std::int64_t nquant, std::int64_t offset,
|
||||
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS]) {
|
||||
if (n_channels != 5 && n_channels != 7) {
|
||||
return Status::fail(JOC_ERR_JOC_UNSUPPORTED_VARIANT, stage::kJoc,
|
||||
"sparse JOC requires 5 or 7 core channels, got " +
|
||||
std::to_string(n_channels));
|
||||
}
|
||||
const int initial_channel = symbols.idx[dp][0];
|
||||
if (initial_channel < 0 || initial_channel >= n_channels) {
|
||||
return syntax_fail("sparse JOC initial channel " + std::to_string(initial_channel) +
|
||||
" out of range for " + std::to_string(n_channels) + " channels");
|
||||
}
|
||||
// Non-active entries take nquant/2, which dequantizes to exactly zero.
|
||||
for (int ch = 0; ch < n_channels; ++ch) {
|
||||
for (int pb = 0; pb < symbols.n_bands; ++pb) {
|
||||
q[dp][ch][pb] = nquant / 2;
|
||||
}
|
||||
}
|
||||
int active = initial_channel;
|
||||
std::int64_t coefficient = offset;
|
||||
for (int pb = 0; pb < symbols.n_bands; ++pb) {
|
||||
if (pb != 0) {
|
||||
active = static_cast<int>(
|
||||
floored_mod(static_cast<std::int64_t>(active) + symbols.idx[dp][pb], n_channels));
|
||||
}
|
||||
coefficient = floored_mod(coefficient + static_cast<std::int64_t>(symbols.vec[dp][pb]), nquant);
|
||||
q[dp][active][pb] = coefficient;
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::int16_t num_channels_for_config(std::uint32_t dmx_config_idx) {
|
||||
for (const NumChannelsEntry& entry : kNumChannels) {
|
||||
if (entry.config == dmx_config_idx) {
|
||||
return entry.channels;
|
||||
}
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
std::int16_t num_bands_for_index(std::uint32_t num_bands_idx) {
|
||||
for (const NumBandsEntry& entry : kNumBands) {
|
||||
if (entry.index == num_bands_idx) {
|
||||
return entry.bands;
|
||||
}
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
Status parse_id14(const std::uint8_t* payload, std::size_t payload_size, joc_frame_params* out,
|
||||
FrameSymbols* symbols) {
|
||||
if (payload == nullptr || out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kJoc, "null payload or output");
|
||||
}
|
||||
std::memset(out, 0, sizeof(*out));
|
||||
out->struct_size = sizeof(joc_frame_params);
|
||||
out->struct_version = JOC_FRAME_PARAMS_VERSION;
|
||||
// capture is purely additive and never changes the parse result.
|
||||
FrameSymbols local_symbols{};
|
||||
FrameSymbols& capture = (symbols != nullptr) ? *symbols : local_symbols;
|
||||
std::memset(&capture, 0, sizeof(capture));
|
||||
|
||||
bits::BitReader reader(payload, payload_size);
|
||||
|
||||
out->dmx_config_idx = static_cast<std::uint8_t>(reader.read(3));
|
||||
out->num_objects_bits = static_cast<std::uint8_t>(reader.read(6));
|
||||
out->ext_config_idx = static_cast<std::uint8_t>(reader.read(3));
|
||||
|
||||
const std::uint32_t n_objects = static_cast<std::uint32_t>(out->num_objects_bits) + 1u;
|
||||
const std::int16_t n_channels = num_channels_for_config(out->dmx_config_idx);
|
||||
if (n_channels < 0) {
|
||||
return syntax_fail("unknown JOC downmix configuration " +
|
||||
std::to_string(out->dmx_config_idx));
|
||||
}
|
||||
out->n_channels = static_cast<std::uint8_t>(n_channels);
|
||||
if (n_objects > JOC_MAX_OBJECTS) {
|
||||
// The reference implementation has no check here and fails later inside
|
||||
// NumPy; the port reports it explicitly (plan 28.2).
|
||||
return Status::fail(JOC_ERR_JOC_UNSUPPORTED_VARIANT, stage::kJoc,
|
||||
"JOC frame declares " + std::to_string(n_objects) +
|
||||
" objects, the ABI supports at most " +
|
||||
std::to_string(static_cast<int>(JOC_MAX_OBJECTS)));
|
||||
}
|
||||
out->n_objects = static_cast<std::uint8_t>(n_objects);
|
||||
|
||||
out->clipgain_x_bits = static_cast<std::uint8_t>(reader.read(3));
|
||||
out->clipgain_y_bits = static_cast<std::uint8_t>(reader.read(5));
|
||||
out->seq_count = reader.read(10);
|
||||
// clipgain = 1 + (y/32) * 2^(x-4). The reference multiplies by an exact
|
||||
// exactly for the whole legal range.
|
||||
out->clipgain = 1.0 + static_cast<double>(out->clipgain_y_bits) / 32.0 *
|
||||
std::ldexp(1.0, static_cast<int>(out->clipgain_x_bits) - 4);
|
||||
|
||||
for (std::uint32_t obj = 0; obj < n_objects; ++obj) {
|
||||
joc_object_params& info = out->objects[obj];
|
||||
info.present = static_cast<std::uint8_t>(reader.read(1));
|
||||
if (info.present == 0u) {
|
||||
continue;
|
||||
}
|
||||
info.num_bands_idx = static_cast<std::uint8_t>(reader.read(3));
|
||||
const std::int16_t bands = num_bands_for_index(info.num_bands_idx);
|
||||
if (bands < 0) {
|
||||
return syntax_fail("unknown JOC num_bands index " +
|
||||
std::to_string(info.num_bands_idx));
|
||||
}
|
||||
info.n_bands = static_cast<std::uint8_t>(bands);
|
||||
info.sparse = static_cast<std::uint8_t>(reader.read(1));
|
||||
info.quant_idx = static_cast<std::uint8_t>(reader.read(1));
|
||||
info.slope_idx = static_cast<std::uint8_t>(reader.read(1));
|
||||
info.num_dpoints_bits = static_cast<std::uint8_t>(reader.read(1));
|
||||
info.n_dpoints = static_cast<std::uint8_t>(info.num_dpoints_bits + 1u);
|
||||
if (info.slope_idx == 1u) {
|
||||
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
|
||||
info.offset_ts[dp] = static_cast<std::uint8_t>(reader.read(5) + 1u);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (reader.failed()) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
|
||||
for (std::uint32_t obj = 0; obj < n_objects; ++obj) {
|
||||
const joc_object_params& info = out->objects[obj];
|
||||
if (info.present == 0u) {
|
||||
continue;
|
||||
}
|
||||
ObjectSymbols* symbol = &capture.objects[obj];
|
||||
symbol->present = 1;
|
||||
symbol->sparse = info.sparse;
|
||||
symbol->n_bands = info.n_bands;
|
||||
symbol->n_dpoints = info.n_dpoints;
|
||||
symbol->n_channels = out->n_channels;
|
||||
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
|
||||
if (info.sparse == 1u) {
|
||||
const Tree idx_tree = select_tree(info.quant_idx, SymbolKind::Idx, n_channels);
|
||||
const std::uint32_t first = reader.read(3);
|
||||
if (reader.failed()) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
symbol->idx[dp][0] = static_cast<std::uint8_t>(first);
|
||||
for (int pb = 1; pb < info.n_bands; ++pb) {
|
||||
std::int16_t value = 0;
|
||||
if (!huff_decode(idx_tree, reader, &value)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
symbol->idx[dp][pb] = static_cast<std::uint8_t>(value);
|
||||
}
|
||||
const Tree vec_tree = select_tree(info.quant_idx, SymbolKind::Vec, n_channels);
|
||||
for (int pb = 0; pb < info.n_bands; ++pb) {
|
||||
std::int16_t value = 0;
|
||||
if (!huff_decode(vec_tree, reader, &value)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
symbol->vec[dp][pb] = value;
|
||||
}
|
||||
} else {
|
||||
const Tree mtx_tree = select_tree(info.quant_idx, SymbolKind::Mtx, n_channels);
|
||||
for (int ch = 0; ch < n_channels; ++ch) {
|
||||
for (int pb = 0; pb < info.n_bands; ++pb) {
|
||||
std::int16_t value = 0;
|
||||
if (!huff_decode(mtx_tree, reader, &value)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
symbol->mtx[dp][ch][pb] = value;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
out->data_end_bits = static_cast<std::uint32_t>(reader.position());
|
||||
out->trailing_bits = static_cast<std::uint32_t>(payload_size * 8u - reader.position());
|
||||
const std::size_t tail_offset = reader.position() / 8u;
|
||||
if (tail_offset < payload_size) {
|
||||
const std::size_t tail_bytes = std::min<std::size_t>(8u, payload_size - tail_offset);
|
||||
std::memcpy(out->tail_bytes, payload + tail_offset, tail_bytes);
|
||||
}
|
||||
|
||||
for (std::uint32_t obj = 0; obj < n_objects; ++obj) {
|
||||
const joc_object_params& info = out->objects[obj];
|
||||
if (info.present == 0u) {
|
||||
continue;
|
||||
}
|
||||
std::uint32_t mask_bit = 1u << obj;
|
||||
out->present_mask |= mask_bit;
|
||||
|
||||
const std::int64_t nquant = (info.quant_idx == 0u) ? 96 : 192;
|
||||
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS] = {};
|
||||
ObjectSymbols* symbol = &capture.objects[obj];
|
||||
|
||||
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
|
||||
if (info.sparse == 1u) {
|
||||
const std::int64_t offset = (info.quant_idx == 0u) ? 50 : 100;
|
||||
const Status status =
|
||||
reconstruct_sparse(*symbol, dp, n_channels, nquant, offset, q);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
} else {
|
||||
const std::int64_t offset = (info.quant_idx == 0u) ? 48 : 96;
|
||||
reconstruct_dense(*symbol, dp, n_channels, nquant, offset, q);
|
||||
}
|
||||
}
|
||||
|
||||
// Operand order and types are kept identical to the reference so the
|
||||
// result is bit-exact, not merely close.
|
||||
const double nquant_half = static_cast<double>(nquant) / 2.0;
|
||||
const double denominator = 4096.0 * static_cast<double>(1 + static_cast<int>(info.quant_idx));
|
||||
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
|
||||
for (int ch = 0; ch < n_channels; ++ch) {
|
||||
for (int pb = 0; pb < info.n_bands; ++pb) {
|
||||
const double value = static_cast<double>(q[dp][ch][pb]) - nquant_half;
|
||||
out->objects[obj].dq[dp][ch][pb] = value * 820.0 / denominator;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
|
||||
for (int ch = 0; ch < n_channels; ++ch) {
|
||||
for (int pb = 0; pb < info.n_bands; ++pb) {
|
||||
symbol->q[dp][ch][pb] = q[dp][ch][pb];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
capture.n_objects = out->n_objects;
|
||||
capture.n_channels = out->n_channels;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status parse_eac3_frame(const std::uint8_t* frame, std::size_t frame_size, joc_frame_params* out,
|
||||
emdf::Container* container, FrameSymbols* symbols) {
|
||||
if (frame == nullptr || out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kJoc, "null frame or output");
|
||||
}
|
||||
emdf::Container local;
|
||||
const Status status = emdf::find_joc_emdf(frame, frame_size, &local);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (container != nullptr) {
|
||||
*container = local;
|
||||
}
|
||||
const emdf::Payload* payload = local.find(emdf::kIdJoc);
|
||||
if (payload == nullptr) {
|
||||
return Status::fail(JOC_ERR_EMDF_TRANSPORT, stage::kEmdf,
|
||||
"EMDF container has no ID14 (JOC) payload");
|
||||
}
|
||||
std::vector<std::uint8_t> bytes;
|
||||
const Status extract = emdf::extract_payload_bytes(frame, frame_size, *payload, &bytes);
|
||||
if (!extract.ok()) {
|
||||
return extract;
|
||||
}
|
||||
return parse_id14(bytes.data(), bytes.size(), out, symbols);
|
||||
}
|
||||
|
||||
Status check_id14_padding(const std::uint8_t* payload, std::size_t payload_size,
|
||||
std::uint32_t* out_trailing_bits) {
|
||||
joc_frame_params params;
|
||||
FrameSymbols symbols;
|
||||
const Status status = parse_id14(payload, payload_size, ¶ms, &symbols);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (out_trailing_bits != nullptr) {
|
||||
*out_trailing_bits = params.trailing_bits;
|
||||
}
|
||||
if (params.trailing_bits > 7u) {
|
||||
return Status::fail(JOC_ERR_BITSTREAM_PADDING, stage::kJoc,
|
||||
"more than 7 bits left after joc_data (" +
|
||||
std::to_string(params.trailing_bits) + ")");
|
||||
}
|
||||
for (std::size_t bit = params.data_end_bits; bit < payload_size * 8u; ++bit) {
|
||||
if (((payload[bit >> 3] >> (7u - (bit & 7u))) & 1u) != 0u) {
|
||||
return Status::fail(JOC_ERR_BITSTREAM_PADDING, stage::kJoc,
|
||||
"non-zero trailing padding bit at " + std::to_string(bit));
|
||||
}
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::joc
|
||||
@@ -1,46 +0,0 @@
|
||||
// Port of src/joc_decode.py.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
#include "joc_core.h"
|
||||
#include "emdf/emdf_parser.h"
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::joc {
|
||||
|
||||
struct ObjectSymbols {
|
||||
std::uint8_t present = 0;
|
||||
std::uint8_t sparse = 0;
|
||||
std::uint8_t n_bands = 0;
|
||||
std::uint8_t n_dpoints = 0;
|
||||
std::uint8_t n_channels = 0;
|
||||
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS] = {};
|
||||
std::int16_t mtx[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS] = {};
|
||||
std::uint8_t idx[JOC_MAX_DPOINTS][JOC_MAX_PARAMETER_BANDS] = {};
|
||||
std::int16_t vec[JOC_MAX_DPOINTS][JOC_MAX_PARAMETER_BANDS] = {};
|
||||
};
|
||||
|
||||
struct FrameSymbols {
|
||||
std::uint8_t n_objects = 0;
|
||||
std::uint8_t n_channels = 0;
|
||||
ObjectSymbols objects[JOC_MAX_OBJECTS];
|
||||
};
|
||||
|
||||
Status parse_id14(const std::uint8_t* payload, std::size_t payload_size, joc_frame_params* out,
|
||||
FrameSymbols* symbols);
|
||||
|
||||
Status parse_eac3_frame(const std::uint8_t* frame, std::size_t frame_size, joc_frame_params* out,
|
||||
emdf::Container* container, FrameSymbols* symbols);
|
||||
|
||||
Status check_id14_padding(const std::uint8_t* payload, std::size_t payload_size,
|
||||
std::uint32_t* out_trailing_bits);
|
||||
|
||||
std::int16_t num_channels_for_config(std::uint32_t dmx_config_idx);
|
||||
|
||||
std::int16_t num_bands_for_index(std::uint32_t num_bands_idx);
|
||||
|
||||
} // namespace joc::joc
|
||||
@@ -1,80 +0,0 @@
|
||||
#include "joc_core/objects16.h"
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
|
||||
namespace joc::joc {
|
||||
|
||||
Status rebuild_objects16(ejoc_renderer_handle handle, const joc_frame_params& params,
|
||||
const float* bed5_planar, const float* lfe, float gain,
|
||||
std::vector<float>* out16, std::string* error) {
|
||||
if (handle == nullptr || bed5_planar == nullptr || out16 == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kDsp, "null argument");
|
||||
}
|
||||
if (params.n_channels != JOC_CORE_CHANNELS) {
|
||||
if (error != nullptr) {
|
||||
*error = "the reused JOC kernel requires 5 core channels, frame declares " +
|
||||
std::to_string(static_cast<unsigned>(params.n_channels));
|
||||
}
|
||||
return Status::fail(JOC_ERR_JOC_UNSUPPORTED_VARIANT, stage::kDsp, *error);
|
||||
}
|
||||
|
||||
std::vector<double> dq(static_cast<std::size_t>(JOC_MAX_OBJECTS) * JOC_MAX_DPOINTS *
|
||||
JOC_CORE_CHANNELS * JOC_MAX_PARAMETER_BANDS,
|
||||
0.0);
|
||||
std::uint8_t n_bands[JOC_MAX_OBJECTS] = {};
|
||||
std::uint8_t n_dpoints[JOC_MAX_OBJECTS] = {};
|
||||
std::uint8_t slope_idx[JOC_MAX_OBJECTS] = {};
|
||||
std::uint8_t offset_ts[JOC_MAX_OBJECTS * JOC_MAX_DPOINTS] = {};
|
||||
for (unsigned obj = 0; obj < JOC_MAX_OBJECTS; ++obj) {
|
||||
const joc_object_params& object = params.objects[obj];
|
||||
if (object.present == 0) {
|
||||
continue;
|
||||
}
|
||||
n_bands[obj] = object.n_bands;
|
||||
n_dpoints[obj] = object.n_dpoints;
|
||||
slope_idx[obj] = object.slope_idx;
|
||||
for (unsigned dp = 0; dp < JOC_MAX_DPOINTS; ++dp) {
|
||||
offset_ts[obj * JOC_MAX_DPOINTS + dp] = object.offset_ts[dp];
|
||||
}
|
||||
for (unsigned dp = 0; dp < object.n_dpoints; ++dp) {
|
||||
for (unsigned ch = 0; ch < JOC_CORE_CHANNELS; ++ch) {
|
||||
for (unsigned pb = 0; pb < object.n_bands; ++pb) {
|
||||
const std::size_t index =
|
||||
((static_cast<std::size_t>(obj) * JOC_MAX_DPOINTS + dp) * JOC_CORE_CHANNELS +
|
||||
ch) *
|
||||
JOC_MAX_PARAMETER_BANDS +
|
||||
pb;
|
||||
dq[index] = object.dq[dp][ch][pb];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
out16->assign(static_cast<std::size_t>(JOC_OUTPUT_CHANNELS) * JOC_FRAME_SAMPLES, 0.0f);
|
||||
// band-0 的 21-tap DC 补偿只在 downmix 配置 3/4 下启用,其余配置 band 0
|
||||
// 走与其他 band 相同的处理。
|
||||
const bool dc_filter = params.dmx_config_idx == 3 || params.dmx_config_idx == 4;
|
||||
if (ejoc_renderer_set_dc_filter(handle, dc_filter ? 1u : 0u) != 0) {
|
||||
if (error != nullptr) {
|
||||
*error = "ejoc_renderer_set_dc_filter failed";
|
||||
}
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kDsp, "ejoc_renderer_set_dc_filter failed");
|
||||
}
|
||||
const int result = ejoc_renderer_process(
|
||||
handle, bed5_planar, lfe, params.present_mask, n_bands, n_dpoints, slope_idx, offset_ts,
|
||||
dq.data(), params.clipgain, 0.0625f, gain, out16->data());
|
||||
if (result != 0) {
|
||||
const char* detail = ejoc_renderer_last_error(handle);
|
||||
const std::string message =
|
||||
"ejoc_renderer_process failed (" + std::to_string(result) + "): " +
|
||||
(detail != nullptr ? detail : "unknown");
|
||||
if (error != nullptr) {
|
||||
*error = message;
|
||||
}
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kDsp, message);
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::joc
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user