Compare commits
6 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 80543e6bd3 | |||
| 697f130b66 | |||
| e46b52c3d9 | |||
| 48fe28d542 | |||
| 044a54c4e2 | |||
| 42d6da6302 |
@@ -0,0 +1,38 @@
|
||||
name: ci
|
||||
|
||||
on:
|
||||
push:
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: ${{ matrix.os }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest, windows-latest]
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -S . -B build -DCMAKE_BUILD_TYPE=Release
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build --config Release --parallel
|
||||
|
||||
- name: Test
|
||||
run: ctest --test-dir build --output-on-failure -C Release
|
||||
|
||||
# Publish the install tree: the shared library and the CLI.
|
||||
- name: Stage
|
||||
run: cmake --install build --config Release --prefix stage
|
||||
|
||||
- name: Upload
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: joc-core-${{ runner.os }}
|
||||
path: |
|
||||
stage/bin/*
|
||||
stage/lib/*
|
||||
if-no-files-found: error
|
||||
@@ -1,99 +0,0 @@
|
||||
name: Native builds
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
tags:
|
||||
- "v*"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: ${{ matrix.asset }}
|
||||
runs-on: ${{ matrix.runner }}
|
||||
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- runner: windows-2022
|
||||
asset: windows-x64
|
||||
cmake_args: -A x64
|
||||
|
||||
- runner: ubuntu-22.04
|
||||
asset: linux-x64
|
||||
cmake_args: ""
|
||||
|
||||
- runner: macos-15-intel
|
||||
asset: macos-x64
|
||||
cmake_args: -DCMAKE_OSX_DEPLOYMENT_TARGET=12.0
|
||||
|
||||
- runner: macos-15
|
||||
asset: macos-arm64
|
||||
cmake_args: -DCMAKE_OSX_DEPLOYMENT_TARGET=12.0
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Configure
|
||||
run: >
|
||||
cmake
|
||||
-S native
|
||||
-B build/native
|
||||
-DCMAKE_BUILD_TYPE=Release
|
||||
-DCMAKE_INSTALL_PREFIX="${{ github.workspace }}/stage"
|
||||
${{ matrix.cmake_args }}
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build/native --config Release --parallel
|
||||
|
||||
- name: Install
|
||||
run: cmake --install build/native --config Release
|
||||
|
||||
- name: Package
|
||||
working-directory: stage
|
||||
run: >
|
||||
cmake -E tar
|
||||
cf "../JustOneCacophony-native-${{ matrix.asset }}.zip"
|
||||
--format=zip
|
||||
-- .
|
||||
|
||||
- name: Upload workflow artifact
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: JustOneCacophony-native-${{ matrix.asset }}
|
||||
path: JustOneCacophony-native-${{ matrix.asset }}.zip
|
||||
if-no-files-found: error
|
||||
retention-days: 14
|
||||
|
||||
release:
|
||||
name: Publish GitHub Release
|
||||
if: startsWith(github.ref, 'refs/tags/v')
|
||||
needs: build
|
||||
runs-on: ubuntu-24.04
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
steps:
|
||||
- name: Download native packages
|
||||
uses: actions/download-artifact@v5
|
||||
with:
|
||||
pattern: JustOneCacophony-native-*
|
||||
path: dist
|
||||
merge-multiple: true
|
||||
|
||||
- name: Create release
|
||||
run: >
|
||||
gh release create "$GITHUB_REF_NAME"
|
||||
dist/*.zip
|
||||
--verify-tag
|
||||
--generate-notes
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
+11
-22
@@ -1,23 +1,12 @@
|
||||
/build/
|
||||
/output/
|
||||
/out/
|
||||
/traces/
|
||||
/vectors/
|
||||
/testdata/
|
||||
/devtools/
|
||||
.vs/
|
||||
.vscode/
|
||||
CMakeUserPresets.json
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
|
||||
.venv/
|
||||
venv/
|
||||
|
||||
build/
|
||||
output/
|
||||
tests/
|
||||
lib/
|
||||
metadata_cache/
|
||||
|
||||
*.metadata.json
|
||||
*.report.json
|
||||
*.variant-error.json
|
||||
*.objects16.f32le
|
||||
|
||||
HRTF/
|
||||
|
||||
# Keep the production binaural regression test while local research fixtures stay ignored.
|
||||
!tests/
|
||||
tests/*
|
||||
!tests/test_binaural_production.py
|
||||
*.spool.f32
|
||||
|
||||
+330
@@ -0,0 +1,330 @@
|
||||
cmake_minimum_required(VERSION 3.20)
|
||||
|
||||
project(joc_core VERSION 0.1.0 LANGUAGES C CXX)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Products
|
||||
# joc_core shared library: the execution core (E-AC-3/EMDF/JOC/OAMD
|
||||
# bitstream, DSP, speaker and binaural rendering, ADM-BWF/WAV
|
||||
# writing, file task and the streaming push/pull surface)
|
||||
# joc_cli command line frontend for file tasks
|
||||
#
|
||||
# Public headers: include/joc_core.h engine, telemetry and file task
|
||||
# include/joc_stream.h embedder-facing streaming surface
|
||||
#
|
||||
# Options
|
||||
# JOC_BUILD_TESTS unit tests (CTest), on by default
|
||||
# JOC_BUILD_DEVTOOLS in-tree verification tools, off: those sources live in
|
||||
# devtools/, which is not part of the repository
|
||||
# JOC_ENABLE_AVX2 build the runtime-dispatched AVX2 kernels, on by default
|
||||
# JOC_ENABLE_AVX512 build the runtime-dispatched AVX-512 kernels, on by default
|
||||
#
|
||||
# Only the MSVC toolchain is validated locally; other platforms are built by CI.
|
||||
# The floating-point flags of the original native library are preserved on
|
||||
# purpose: /fp:precise on MSVC, -fno-fast-math elsewhere, and a static CRT so
|
||||
# that no redistributable is required.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
option(JOC_BUILD_TESTS "Build the unit tests" ON)
|
||||
option(JOC_BUILD_DEVTOOLS "Build the in-tree verification tools (needs devtools/)" OFF)
|
||||
# The DSP kernels are dispatched at run time (see src/simd/simd.h): the
|
||||
# baseline units stay on the ISA every x86-64 CPU has, and these two options
|
||||
# decide whether the wider units are linked in at all. Both are on by default.
|
||||
# The instruction set actually executed is chosen from CPUID/XGETBV (or the
|
||||
# AArch64 baseline) when the library is first used, so a binary carrying the
|
||||
# AVX-512 unit still runs on a CPU without it, and JOC_SIMD=scalar|sse2|avx2|
|
||||
# avx512|neon pins one tier for verification.
|
||||
option(JOC_ENABLE_AVX2 "Build the runtime-dispatched AVX2 kernels" ON)
|
||||
option(JOC_ENABLE_AVX512 "Build the runtime-dispatched AVX-512 kernels" ON)
|
||||
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_CXX_STANDARD_REQUIRED ON)
|
||||
set(CMAKE_CXX_EXTENSIONS OFF)
|
||||
|
||||
if(NOT CMAKE_BUILD_TYPE AND NOT CMAKE_CONFIGURATION_TYPES)
|
||||
set(CMAKE_BUILD_TYPE Release CACHE STRING "Build type" FORCE)
|
||||
endif()
|
||||
|
||||
find_package(Threads REQUIRED)
|
||||
|
||||
# Ninja learns MSVC's header dependencies by parsing the compiler's /showIncludes
|
||||
# notes, and it only recognises the prefix it is told about. A localized MSVC
|
||||
# prints a translated prefix, and when CMake cannot detect it the notes are
|
||||
# silently dropped: the build then reuses stale objects after a header changes and
|
||||
# produces a binary that does not match its sources -- wrong, not just slow. This
|
||||
# only warns, because supplying the value is not always possible either: it has to
|
||||
# survive the cache code page to be usable, and a build driver can compensate more
|
||||
# reliably by dropping objects when a header is newer than they are.
|
||||
if(MSVC AND CMAKE_GENERATOR MATCHES "Ninja" AND NOT CMAKE_CL_SHOWINCLUDES_PREFIX)
|
||||
message(WARNING
|
||||
"No /showIncludes prefix was detected, so Ninja will not track header "
|
||||
"dependencies and a header change will not rebuild what includes it. "
|
||||
"Configure with -DCMAKE_CL_SHOWINCLUDES_PREFIX=<the text MSVC prints in "
|
||||
"front of each included file>, or make sure the compiler emits its "
|
||||
"messages in the language CMake probes for.")
|
||||
endif()
|
||||
|
||||
# Verbatim copies of the upstream native library; see THIRD_PARTY_NOTICES.md.
|
||||
# These files are never edited: they are the validated DSP kernels.
|
||||
set(JOC_REUSED_SOURCES
|
||||
src/joc_core/eac3joc_core.cpp
|
||||
src/speaker/speaker_renderer.cpp
|
||||
src/binaural/binaural_renderer.cpp
|
||||
)
|
||||
|
||||
set(JOC_INTERNAL_SOURCES
|
||||
src/adm/adm_metadata.cpp
|
||||
src/adm/adm_tracks.cpp
|
||||
src/binaural/binaural_runtime.cpp
|
||||
src/binaural/sofa_binaural_renderer.cpp
|
||||
src/eac3_transport/eac3_reader.cpp
|
||||
src/emdf/emdf_parser.cpp
|
||||
src/foundation/bit_reader.cpp
|
||||
src/foundation/fft.cpp
|
||||
src/foundation/fs_utf8.cpp
|
||||
src/foundation/mini_json.cpp
|
||||
src/foundation/sha256.cpp
|
||||
src/hrtf/jochrtf.cpp
|
||||
src/hrtf/public_filterbank.cpp
|
||||
src/hrtf/rosella_model.cpp
|
||||
src/hrtf/rosella_renderer.cpp
|
||||
src/hrtf/kernel_tables.cpp
|
||||
src/hrtf/sofa.cpp
|
||||
src/hrtf/sofa_cache.cpp
|
||||
src/hrtf/sofa_field.cpp
|
||||
src/io/adm_writer.cpp
|
||||
src/io/hdf5.cpp
|
||||
src/io/inflate.cpp
|
||||
src/io/npy.cpp
|
||||
src/io/npy_writer.cpp
|
||||
src/io/process.cpp
|
||||
src/io/wav_writer.cpp
|
||||
src/io/zip_reader.cpp
|
||||
src/joc_bitstream/joc_parser.cpp
|
||||
src/joc_core/objects16.cpp
|
||||
src/oamd/oamd_parser.cpp
|
||||
src/speaker/speaker_layout_lookup.cpp
|
||||
src/speaker/speaker_step.cpp
|
||||
src/stream/stream.cpp
|
||||
src/task/task.cpp
|
||||
src/telemetry/event_bus.cpp
|
||||
src/timeline/position_timeline.cpp
|
||||
)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Runtime-dispatched SIMD kernels (src/simd/simd.h explains the contract).
|
||||
#
|
||||
# One flat directory, the ISA in the file name (`kernels_intrin_<isa>.cpp`),
|
||||
# never in a subdirectory. MSVC has no function-level ISA attribute, so every
|
||||
# ISA lives in its own translation unit compiled with its own flag, and
|
||||
# dispatch.cpp -- built for the baseline ISA -- picks one when the library is
|
||||
# first used. Only a `kernels_intrin_*.cpp` unit ever gets a wider flag, so a
|
||||
# baseline unit cannot inherit one by accident.
|
||||
#
|
||||
# The x86-64 baseline is SSE2 and there is no SSE2 unit on purpose: a 128-bit
|
||||
# SSE2 register is the register a scalar double already occupies, so SSE2 cannot
|
||||
# widen double-precision arithmetic and hand-written SSE2 would only add moves.
|
||||
# AArch64 needs no probe either; ASIMD is architectural, and the NEON unit is
|
||||
# how a vector path gets selected there.
|
||||
# ---------------------------------------------------------------------------
|
||||
set(JOC_SIMD_SOURCES
|
||||
src/simd/cpu_probe.cpp
|
||||
src/simd/dispatch.cpp
|
||||
src/simd/kernels_scalar.cpp
|
||||
)
|
||||
set(JOC_SIMD_DEFINES "")
|
||||
|
||||
set(JOC_ARCH "")
|
||||
if(CMAKE_SIZEOF_VOID_P EQUAL 8)
|
||||
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(AMD64|amd64|x86_64|X86_64|EM64T)$")
|
||||
set(JOC_ARCH x86_64)
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(ARM64|arm64|aarch64|AARCH64)$")
|
||||
set(JOC_ARCH aarch64)
|
||||
endif()
|
||||
endif()
|
||||
if(NOT JOC_ARCH AND DEFINED CMAKE_CXX_COMPILER_ARCHITECTURE_ID)
|
||||
if(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "x64")
|
||||
set(JOC_ARCH x86_64)
|
||||
elseif(CMAKE_CXX_COMPILER_ARCHITECTURE_ID STREQUAL "ARM64")
|
||||
set(JOC_ARCH aarch64)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(JOC_ARCH STREQUAL "x86_64")
|
||||
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_SSE2=1)
|
||||
if(JOC_ENABLE_AVX2)
|
||||
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_avx2.cpp)
|
||||
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_AVX2=1)
|
||||
if(MSVC)
|
||||
set_source_files_properties(src/simd/kernels_intrin_avx2.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "/arch:AVX2")
|
||||
else()
|
||||
set_source_files_properties(src/simd/kernels_intrin_avx2.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "-mavx2")
|
||||
endif()
|
||||
endif()
|
||||
if(JOC_ENABLE_AVX512)
|
||||
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_avx512.cpp)
|
||||
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_AVX512=1)
|
||||
if(MSVC)
|
||||
set_source_files_properties(src/simd/kernels_intrin_avx512.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "/arch:AVX512")
|
||||
else()
|
||||
set_source_files_properties(src/simd/kernels_intrin_avx512.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "-mavx512f")
|
||||
endif()
|
||||
endif()
|
||||
elseif(JOC_ARCH STREQUAL "aarch64")
|
||||
list(APPEND JOC_SIMD_DEFINES JOC_SIMD_HAVE_NEON=1)
|
||||
list(APPEND JOC_SIMD_SOURCES src/simd/kernels_intrin_neon.cpp)
|
||||
endif()
|
||||
|
||||
add_library(joc_simd OBJECT ${JOC_SIMD_SOURCES})
|
||||
target_include_directories(joc_simd PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
target_compile_definitions(joc_simd PRIVATE ${JOC_SIMD_DEFINES})
|
||||
# The objects are linked into the shared library as well, so they must be
|
||||
# position independent even though an object library does not inherit that.
|
||||
set_target_properties(joc_simd PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
||||
|
||||
# Static form of the engine, used by the in-tree tools and tests so they can use
|
||||
# internal modules without exporting them from the shared library.
|
||||
if(JOC_BUILD_TESTS OR JOC_BUILD_DEVTOOLS)
|
||||
add_library(joc_core_impl STATIC ${JOC_INTERNAL_SOURCES} ${JOC_REUSED_SOURCES}
|
||||
$<TARGET_OBJECTS:joc_simd>)
|
||||
target_include_directories(joc_core_impl
|
||||
PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/include"
|
||||
PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src"
|
||||
)
|
||||
target_link_libraries(joc_core_impl PUBLIC Threads::Threads)
|
||||
endif()
|
||||
|
||||
add_library(joc_core SHARED
|
||||
src/api/joc_api.cpp
|
||||
src/api/joc_stream_api.cpp
|
||||
src/api/joc_task_api.cpp
|
||||
${JOC_INTERNAL_SOURCES}
|
||||
${JOC_REUSED_SOURCES}
|
||||
$<TARGET_OBJECTS:joc_simd>
|
||||
)
|
||||
target_include_directories(joc_core
|
||||
PUBLIC "${CMAKE_CURRENT_SOURCE_DIR}/include"
|
||||
PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src"
|
||||
)
|
||||
target_compile_definitions(joc_core PRIVATE JOC_BUILD_DLL)
|
||||
target_link_libraries(joc_core PRIVATE Threads::Threads)
|
||||
set_target_properties(joc_core PROPERTIES
|
||||
OUTPUT_NAME "joc_core"
|
||||
CXX_VISIBILITY_PRESET hidden
|
||||
VISIBILITY_INLINES_HIDDEN YES
|
||||
POSITION_INDEPENDENT_CODE YES
|
||||
)
|
||||
|
||||
# The frontend compiles the UTF-8 path shim itself: it is small, and the shared
|
||||
# library keeps its internals unexported.
|
||||
add_executable(joc_cli src/cli/joc_cli.cpp src/foundation/fs_utf8.cpp)
|
||||
target_link_libraries(joc_cli PRIVATE joc_core)
|
||||
target_include_directories(joc_cli PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
|
||||
# The installed CLI is in bin/ and the shared library in lib/, and ELF/Mach-O
|
||||
# strip the build rpath on install, so the relative lookup has to be recorded
|
||||
# here or `bin/joc_cli` cannot find `../lib/libjoc_core.*`.
|
||||
if(UNIX)
|
||||
if(APPLE)
|
||||
set_target_properties(joc_cli PROPERTIES INSTALL_RPATH "@loader_path/../lib")
|
||||
else()
|
||||
set_target_properties(joc_cli PROPERTIES INSTALL_RPATH "$ORIGIN/../lib")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(JOC_BUILD_TESTS)
|
||||
enable_testing()
|
||||
add_executable(joc_tests tests/test_core.cpp)
|
||||
target_link_libraries(joc_tests PRIVATE joc_core_impl)
|
||||
target_include_directories(joc_tests PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
add_test(NAME core COMMAND joc_tests)
|
||||
# The public headers must stay valid C and C++: these targets exist to prove it.
|
||||
add_library(joc_headers_c OBJECT tests/test_headers.c)
|
||||
target_include_directories(joc_headers_c PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/include")
|
||||
add_library(joc_headers_cpp OBJECT tests/test_headers.cpp)
|
||||
target_link_libraries(joc_headers_cpp PRIVATE joc_core)
|
||||
endif()
|
||||
|
||||
if(JOC_BUILD_DEVTOOLS)
|
||||
add_executable(joc_dump devtools/joc_dump/main.cpp)
|
||||
target_link_libraries(joc_dump PRIVATE joc_core joc_core_impl)
|
||||
target_include_directories(joc_dump PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
# SOFA reader probe: the C++ side of devtools/checks/check_sofa.py.
|
||||
add_executable(joc_sofa_field_probe devtools/sofa_field_probe/main.cpp)
|
||||
add_executable(joc_dictionary_probe devtools/sofa_field_probe/dictionary_main.cpp)
|
||||
target_link_libraries(joc_dictionary_probe PRIVATE joc_core_impl)
|
||||
target_include_directories(joc_dictionary_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
target_link_libraries(joc_sofa_field_probe PRIVATE joc_core_impl)
|
||||
target_include_directories(joc_sofa_field_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
# Rosella model probe: the C++ side of devtools/checks/check_rosella.py.
|
||||
add_executable(joc_rosella_probe devtools/rosella_probe/main.cpp)
|
||||
target_link_libraries(joc_rosella_probe PRIVATE joc_core_impl)
|
||||
target_include_directories(joc_rosella_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
add_executable(joc_sofa_probe devtools/sofa_probe/main.cpp)
|
||||
target_link_libraries(joc_sofa_probe PRIVATE joc_core_impl)
|
||||
target_include_directories(joc_sofa_probe PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/src")
|
||||
# Probe for the streaming surface: the acceptance harness and the usage
|
||||
# example for joc_stream.h. Not shipped: a player integrates the library.
|
||||
add_executable(joc_stream_probe devtools/joc_stream/main.cpp)
|
||||
target_link_libraries(joc_stream_probe PRIVATE joc_core)
|
||||
endif()
|
||||
|
||||
set(JOC_MSVC_TARGETS joc_core joc_cli joc_simd)
|
||||
if(JOC_BUILD_TESTS OR JOC_BUILD_DEVTOOLS)
|
||||
list(APPEND JOC_MSVC_TARGETS joc_core_impl)
|
||||
endif()
|
||||
if(JOC_BUILD_TESTS)
|
||||
list(APPEND JOC_MSVC_TARGETS joc_tests joc_headers_c joc_headers_cpp)
|
||||
endif()
|
||||
if(JOC_BUILD_DEVTOOLS)
|
||||
list(APPEND JOC_MSVC_TARGETS joc_rosella_probe joc_dump joc_stream_probe joc_sofa_probe joc_sofa_field_probe joc_dictionary_probe)
|
||||
endif()
|
||||
|
||||
if(MSVC)
|
||||
set(JOC_MSVC_FLAGS /W4 /permissive- /Zc:__cplusplus /utf-8 /EHsc /GR- /fp:precise)
|
||||
# The writers use std::fopen for seekable, byte-exact output.
|
||||
set(JOC_MSVC_DEFINES _CRT_SECURE_NO_WARNINGS)
|
||||
# C4324 comes from the reused kernel's std::barrier member at /W4; it is
|
||||
# pre-existing behaviour of a verbatim file, so the warning is silenced
|
||||
# rather than the file edited.
|
||||
set_source_files_properties(src/joc_core/eac3joc_core.cpp
|
||||
PROPERTIES COMPILE_OPTIONS "/wd4324")
|
||||
foreach(target IN LISTS JOC_MSVC_TARGETS)
|
||||
set_property(TARGET ${target} PROPERTY
|
||||
MSVC_RUNTIME_LIBRARY "MultiThreaded$<$<CONFIG:Debug>:Debug>")
|
||||
# No /arch here on purpose: the baseline units keep the architecture's
|
||||
# guaranteed ISA and only the src/simd `kernels_intrin_*` units carry a
|
||||
# wider one (their flags are set per source file above).
|
||||
target_compile_options(${target} PRIVATE ${JOC_MSVC_FLAGS}
|
||||
$<$<CONFIG:Release>:/O2>
|
||||
$<$<CONFIG:Release>:/Oi>)
|
||||
target_compile_definitions(${target} PRIVATE ${JOC_MSVC_DEFINES})
|
||||
endforeach()
|
||||
target_link_options(joc_core PRIVATE /INCREMENTAL:NO /OPT:REF /OPT:ICF)
|
||||
target_link_options(joc_cli PRIVATE /INCREMENTAL:NO)
|
||||
else()
|
||||
foreach(target IN LISTS JOC_MSVC_TARGETS)
|
||||
target_compile_options(${target} PRIVATE -Wall -Wextra -Wpedantic -fno-fast-math
|
||||
$<$<CONFIG:Release>:-O3>)
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
# Headers listed for IDE visibility.
|
||||
target_sources(joc_core PRIVATE
|
||||
include/joc_core.h
|
||||
include/joc_stream.h
|
||||
include/eac3joc_core.h
|
||||
src/joc_bitstream/joc_huffman_tables.h
|
||||
src/joc_core/qmf_tables.h
|
||||
src/speaker/speaker_layouts.h
|
||||
)
|
||||
|
||||
install(TARGETS joc_core joc_cli
|
||||
RUNTIME DESTINATION bin
|
||||
LIBRARY DESTINATION lib
|
||||
ARCHIVE DESTINATION lib
|
||||
)
|
||||
+144
-192
@@ -1,207 +1,159 @@
|
||||
# JustOneCacophony — JOC
|
||||
# JustOneCacophony — C++ Core
|
||||
|
||||
[中文版](README.md)
|
||||
[中文](README.md) · [Mathematics](docs/math.en.md) · [Binaural rendering](docs/binaural.en.md) · [SIMD dispatch](docs/simd.en.md)
|
||||
|
||||
> JustOneCacophony is an experimental/test implementation of E-AC-3 JOC for studying JOC parsing, reconstruction, rendering, and the associated mathematics.
|
||||
The C++ implementation of JustOneCacophony: an execution core for E-AC-3 JOC
|
||||
bitstream parsing, object reconstruction and rendering. It extracts EMDF, ID14 JOC
|
||||
parameters and ID11 OAMD metadata from E-AC-3 syncframes, combines them with the
|
||||
core 5.1 PCM decoded by FFmpeg to rebuild the LFE and 15 object signals, and writes
|
||||
ADM BWF, a WAV for a chosen speaker layout, or a binaural WAV using a compiled HRTF
|
||||
directional field.
|
||||
|
||||
The project can extract and parse EMDF, ID14 JOC parameters, and ID11 OAMD metadata from common E-AC-3 JOC streams. It combines those data with the core 5.1 PCM decoded by FFmpeg, reconstructs LFE plus 15 object channels, and writes ADM BWF, a WAV file for a selected speaker layout, or direct DLL-free Rosella binaural stereo.
|
||||
This is research code, not a complete, standard-conformant or production JOC
|
||||
decoder. It covers the bitstream forms it implements and reports an explicit error
|
||||
on unknown variants instead of pretending everything is in harmony.
|
||||
|
||||
This is research code, not a complete, standards-compliant, or production-grade JOC decoder. It covers only the stream forms currently implemented. Unknown variants fail explicitly—because when the math goes wrong, all that may remain is the cacophony.
|
||||
## Building
|
||||
|
||||
## Current features
|
||||
```powershell
|
||||
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release
|
||||
cmake --build build
|
||||
ctest --test-dir build --output-on-failure
|
||||
```
|
||||
|
||||
- Scan common contiguous EMDF containers in E-AC-3 sync frames.
|
||||
- Parse ID14 dense JOC parameters, Huffman data, differential matrices, and `joc_clipgain`.
|
||||
- Parse ID11 OAMD position updates and build object trajectories.
|
||||
- Reconstruct LFE plus 15 object channels through analysis QMF, parameter interpolation, the object matrix, and inverse QMF.
|
||||
- Write a 25-channel ADM BWF: a 10-channel 7.1.2 bed (silent except for LFE) plus 15 objects.
|
||||
- Render directly to `2.0`, `3.1`, `5.1`, `7.1`, `5.1.2`, `5.1.4`, `7.1.2`, `7.1.4`, `9.1.4`, or `9.1.6`.
|
||||
- Run DLL-free Rosella binaural rendering directly from `pcm16 + ID11/OAMD`, without a temporary ADM BWF.
|
||||
- Keep the binaural DSP in float64/complex128, including 961-sample latency compensation, cross-frame state, and the room tail.
|
||||
- Use a shared float32/PCM24 WAV writer and explicit PCM24 clipping policy for direct outputs.
|
||||
- Use the NumPy backend or an optional C++20 core through `ctypes`; `auto` falls back to Python when the native library is unavailable.
|
||||
- Read or write metadata sidecars and produce metadata, timing, and output reports.
|
||||
Only MSVC (VS 2022, static CRT) is validated locally; Linux and macOS are built and
|
||||
unit-tested by `.github/workflows/ci.yml`. Floating-point behaviour is part of the
|
||||
byte-exact acceptance, so fast-math is never enabled: `/fp:precise` on MSVC,
|
||||
`-fno-fast-math` elsewhere.
|
||||
|
||||
## Processing flow
|
||||
Windows Release builds target AVX2 by default (`JOC_ENABLE_AVX2`, ON, see
|
||||
`CMakeLists.txt`). That switch is itself part of the byte-exact acceptance — the
|
||||
SHA-256 of every rendered output is identical — and buys 12.4% on the SOFA binaural
|
||||
kernel and 1.8% on Rosella. The price is a runtime requirement: such a `joc_core.dll`
|
||||
executes AVX2 instructions and dies on an illegal instruction on pre-2013 x86. There
|
||||
is no runtime dispatch, so a binary is one or the other:
|
||||
|
||||
```powershell
|
||||
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release -DJOC_ENABLE_AVX2=OFF
|
||||
```
|
||||
|
||||
gives a baseline-ISA (SSE2) build that runs on any x86-64. Non-MSVC builds never
|
||||
receive the flag.
|
||||
|
||||
### Paths and encoding
|
||||
|
||||
Every path inside the library is **UTF-8**, converted only at the OS boundary
|
||||
(`src/foundation/fs_utf8.*`): on Windows through `std::filesystem::path` (UTF-16
|
||||
inside) into `_wfopen`/`CreateProcessW`, and as plain bytes elsewhere. Command line
|
||||
arguments are re-parsed from `GetCommandLineW` + `CommandLineToArgvW` on Windows and
|
||||
the console code page is set to UTF-8, so non-ASCII paths (Japanese, Chinese, ...)
|
||||
work for the input, the ffmpeg child process and the output files alike; a
|
||||
non-ASCII path regression case runs in `ctest`.
|
||||
|
||||
## Artifacts
|
||||
|
||||
| Artifact | Purpose |
|
||||
|---|---|
|
||||
| `joc_core.dll` | The engine: bitstream parsing, JOC/OAMD, DSP, speaker and binaural rendering, ADM BWF/WAV writing, file task, streaming surface |
|
||||
| `joc_cli.exe` | Command line frontend for file tasks |
|
||||
| `include/joc_core.h` | Engine, telemetry and file-task interface (pure C) |
|
||||
| `include/joc_stream.h` | Embedder-facing streaming push/pull interface (pure C, self-contained) |
|
||||
|
||||
## Command line
|
||||
|
||||
The arguments match the reference Python CLI exactly: the input is positional,
|
||||
**ADM BWF is the default output**, `--speaker-layout` or `--binaural` selects the
|
||||
other two modes, and without `-o` the result lands in `output/`.
|
||||
|
||||
```powershell
|
||||
# Default: ADM BWF (inherently 24-bit, so there is no format option)
|
||||
joc_cli "07. Gold Forever (2021 Master).m4a"
|
||||
# -> output/07. Gold Forever (2021 Master).adm.wav
|
||||
|
||||
joc_cli input.m4a -o out/adm.wav # explicit output
|
||||
|
||||
# Speaker layout
|
||||
joc_cli input.m4a --speaker-layout 5.1 # -> output/<name>.5.1.wav
|
||||
joc_cli input.m4a --speaker-layout 7.1.4 --speaker-output out/714.wav --speaker-format int24
|
||||
|
||||
# Binaural (HRTF defaults to <exe>/HRTF/binaural.sofa, then <exe>/HRTF/binaural.personalized_headphone)
|
||||
joc_cli input.m4a --binaural
|
||||
joc_cli input.m4a --binaural --sofa-hrtf HRTF/other.sofa # another SOFA
|
||||
joc_cli input.m4a --binaural --personalized-headphone # Rosella personalisation
|
||||
# -> output/<name>.binaural.wav
|
||||
|
||||
# Other common switches
|
||||
joc_cli input.m4a --duration 30 --gain-db -3 --trajectory-mode dense64
|
||||
joc_cli input.eac3 --metadata-only --print-metadata summary # parse and print metadata only
|
||||
```
|
||||
|
||||
`--speaker-format` / `--binaural-format` default to `float32`; an `int24` request that
|
||||
would clip follows `--clip-action` (default `ask`; a non-interactive terminal must
|
||||
pass `continue`, `float32` or `abort`). `--duration` is in **seconds**, and
|
||||
`--object-delay-samples`, `--speaker-metadata-offset`, `--binaural-tail-seconds` and
|
||||
`--binaural-tail-threshold` (1e-8, the binaural tail trim) keep the reference
|
||||
defaults. A run always writes `<output>.report.json` (`--report-json` overrides it).
|
||||
|
||||
**Differences from the reference:** `--sofa-hrtf`, `--personalized-headphone`,
|
||||
`--backend python` and the metadata sidecars (`--metadata-dir`, `--metadata-cache`,
|
||||
`--metadata-backend sidecar`) are unavailable in this build and fail immediately
|
||||
with an explanation instead of being ignored. This build adds `--bed` (pre-decoded
|
||||
6-channel float32 PCM, which skips ffmpeg decoding), `--kernels`, `--work-dir`,
|
||||
`--report-json`, `--dry-run` and `--quiet`.
|
||||
|
||||
## Library integration
|
||||
|
||||
The engine and the file task are exposed by `joc_core.h`: `joc_task_validate` /
|
||||
`joc_task_execute` run a file task and report state events through a callback, and
|
||||
`joc_task_result` carries frame counts, peak, byte count and SHA-256.
|
||||
|
||||
Players and decoder components use `joc_stream.h`: the caller pushes E-AC-3 bytes
|
||||
and the matching core PCM (or already-rebuilt objects16) at its own pace and pulls
|
||||
rendered PCM. Any chunking is allowed, and the result is byte-identical to the file
|
||||
task.
|
||||
|
||||
```c
|
||||
joc_stream_config config = {0};
|
||||
config.struct_size = sizeof(config);
|
||||
config.input = JOC_STREAM_IN_EAC3; /* or JOC_STREAM_IN_PCM_OBJECTS16 */
|
||||
config.output = JOC_STREAM_OUT_SPEAKER; /* or BINAURAL / PCM_OBJECTS16 */
|
||||
config.speaker_layout_name = "5.1";
|
||||
joc_stream* stream = NULL;
|
||||
joc_stream_create(&config, &stream);
|
||||
/* loop: joc_stream_push(...) / joc_stream_pull(...) */
|
||||
joc_stream_flush(stream);
|
||||
joc_stream_destroy(stream);
|
||||
```
|
||||
|
||||
Contract: state is instance-private, so streams coexist; push and pull on one
|
||||
instance must come from the same thread; rendering is stateful, so **this version
|
||||
offers no seek** - repositioning means decoding from the start of the stream. The
|
||||
kernel latency is 961 samples for speaker/binaural output and `joc_stream_flush`
|
||||
drains the binaural room tail.
|
||||
|
||||
## Layout
|
||||
|
||||
```text
|
||||
M4A / E-AC-3
|
||||
├─ FFmpeg extracts E-AC-3 and decodes the core 5.1 PCM
|
||||
├─ EMDF → ID14 JOC parameters → object matrix
|
||||
├─ core PCM → analysis QMF → parameter interpolation → inverse QMF
|
||||
├─ ID11 OAMD → object positions and timing
|
||||
└─ LFE + 15 objects
|
||||
├─ 25ch ADM BWF
|
||||
├─ speaker WAV for the selected layout
|
||||
└─ direct ID11 timeline → Rosella → binaural WAV
|
||||
include/ public C ABI: joc_core.h (engine/file task), joc_stream.h (streaming),
|
||||
eac3joc_core.h (upstream ABI)
|
||||
src/ implementation: eac3_transport, emdf, joc_bitstream, joc_core, oamd,
|
||||
timeline, speaker, binaural, hrtf, adm, io, telemetry, task, stream,
|
||||
api, cli, simd
|
||||
tests/ unit tests (CTest, self-contained, no external data)
|
||||
docs/ mathematics, the binaural rendering flow and SIMD dispatch
|
||||
```
|
||||
|
||||
The Python and C++ backends follow the same mathematics. The native core handles object reconstruction, speaker rendering, and binaural QMF/hybrid/room/synthesis; bitstream parsing, the OAMD timeline, model parsing, and CLI behavior remain in Python.
|
||||
Eight files under `src/` are byte-identical copies of the upstream JustOneCacophony
|
||||
native library and are never edited (see [THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md)).
|
||||
|
||||
## Requirements
|
||||
## Compatibility note
|
||||
|
||||
- Python 3.10+
|
||||
- NumPy 1.24+
|
||||
- A standalone FFmpeg executable; `ffmpeg-python` is not required. FFmpeg is discovered through `PATH` by default or selected with `--ffmpeg`
|
||||
- Optional: CMake and a C++20 toolchain to build the native core
|
||||
`object_delay_samples` defaults to **1473**, preserving the behaviour of the existing
|
||||
implementation; it is a configurable field and changing it changes the OAMD/ADM time
|
||||
alignment. Upstream investigation suggests the value should be 0; this project keeps
|
||||
the current default to stay byte-identical.
|
||||
|
||||
Install the Python dependency in a project-specific environment:
|
||||
## License
|
||||
|
||||
```powershell
|
||||
python -m pip install -r requirements.txt
|
||||
```
|
||||
|
||||
If FFmpeg is not on `PATH`:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --ffmpeg C:\path\to\ffmpeg.exe
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Write a 25-channel ADM BWF by default:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a
|
||||
```
|
||||
|
||||
Select a backend or output path:
|
||||
|
||||
```powershell
|
||||
python main.py input.eac3 -o output.adm.wav --backend python
|
||||
python main.py input.m4a --backend native --native-threads 2
|
||||
python main.py input.m4a --native-library lib/eac3joc_core.dll
|
||||
```
|
||||
|
||||
Write a speaker-layout WAV directly:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --speaker-layout 2.0 --speaker-format float32
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24
|
||||
python main.py input.m4a --speaker-layout 7.1.2 --speaker-output output.7.1.2.wav
|
||||
```
|
||||
|
||||
Write Rosella binaural stereo directly (ordinary objects are Near/Mid/Far only; Mid is the default):
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --binaural
|
||||
python main.py input.m4a --binaural --binaural-mode near
|
||||
python main.py input.m4a --binaural --binaural-mode far --binaural-format int24
|
||||
python main.py input.m4a --binaural --binaural-output output.binaural.wav `
|
||||
--personalized-headphone C:\HRTF\my.personalized_headphone
|
||||
```
|
||||
|
||||
The default model path is `HRTF/binaural.personalized_headphone`. An example HRTF file is available from:
|
||||
|
||||
https://professionalsupport.dolby.com/s/question/0D54u0000AAT85HCQT/the-state-of-personalized-binaural-rendering?language=en_US
|
||||
|
||||
See [Binaural Rendering Mathematics](docs/binaural.en.md) for the formulas, state, and timing model.
|
||||
|
||||
Speaker and binaural output share peak analysis, the WAV writer, and clipping policy. When PCM24 may clip in a non-interactive environment, select a policy explicitly:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action abort
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action float32
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action continue
|
||||
python main.py input.m4a --binaural --binaural-format int24 --clip-action abort
|
||||
```
|
||||
|
||||
Metadata and diagnostics:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --print-metadata summary
|
||||
python main.py input.m4a --metadata-only --print-metadata frames
|
||||
python main.py input.m4a --metadata-cache metadata_cache
|
||||
python main.py input.m4a --metadata-dir metadata_cache
|
||||
```
|
||||
|
||||
### Experimental binaural mode settings for JOC objects
|
||||
|
||||
The binaural mode written here is a user-selected, experimental rendering hint for downstream ADM renderers. It is **not original binaural metadata extracted or recovered from the input E-AC-3 JOC bitstream**, nor does it represent the original mix's per-object binaural settings. The selected mode is applied uniformly to all 15 JOC objects; the default `unspecified` is this tool's default, not a mode detected in the source file.
|
||||
|
||||
Use `--joc-binaural-mode off|near|far|mid|unspecified` to select a mode, encoded as `0|1|2|3|4` respectively. The default is `unspecified`:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --joc-binaural-mode mid
|
||||
```
|
||||
|
||||
This option only sets the low 3 binaural-render-mode bits of the last 15 JOC object entries in ADM BWF DBMD segment 10, leaving the first 10 bed entries unchanged. It does not change PCM, object trajectories, direct speaker rendering, or direct Rosella binaural rendering. The adjacent `.report.json` records the mode name and value in `joc_binaural_mode` and `joc_binaural_mode_value`; both are `null` for direct speaker or direct binaural output, where the option does not apply.
|
||||
|
||||
### Binaural calculation
|
||||
|
||||
See [Binaural Rendering Mathematics](docs/binaural.en.md) for QMF, hybrid processing, direction fields, distance, ITD, room processing, 512-sample parameter updates, and 961-sample latency compensation.
|
||||
|
||||
For all options:
|
||||
|
||||
```powershell
|
||||
python main.py --help
|
||||
```
|
||||
|
||||
Without `-o`, output still goes to `output/` at the repository root. The directory move intentionally preserves this behavior.
|
||||
|
||||
## Native core
|
||||
|
||||
The repository does not include native binaries by default. Download a prebuilt runtime for the current platform from a project Release, or build one locally, then place the runtime library under `lib/` at the repository root; create the directory if it is absent. To build it yourself, run CMake from the repository root:
|
||||
|
||||
```powershell
|
||||
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
|
||||
cmake --build build/cmake --config Release
|
||||
cmake --install build/cmake --config Release
|
||||
```
|
||||
|
||||
The runtime lookup order is:
|
||||
|
||||
1. `--native-library`;
|
||||
2. `EAC3JOC_NATIVE_LIBRARY`;
|
||||
3. the standard platform library name under `lib/`.
|
||||
|
||||
See the [native-core notes](docs/native.en.md) for ABI, state, and precision details.
|
||||
|
||||
## Repository layout
|
||||
|
||||
```text
|
||||
JustOneCacophony/
|
||||
├─ main.py command-line entry point
|
||||
├─ src/ Python implementation modules
|
||||
├─ native/ C/C++ acceleration core, C ABI, and required table data
|
||||
├─ data/ Python runtime table data
|
||||
├─ lib/ native runtime drop-in directory (create as needed)
|
||||
├─ docs/ math and native-core notes in both languages
|
||||
├─ requirements.txt Python dependency
|
||||
├─ README.md Chinese documentation
|
||||
└─ README.en.md English documentation
|
||||
```
|
||||
|
||||
## Mathematical implementation
|
||||
|
||||
The main documented stages are:
|
||||
|
||||
- dense JOC differential reconstruction and dequantization;
|
||||
- parameter-band mapping to 64 QMF subbands;
|
||||
- cross-frame parameter interpolation;
|
||||
- analysis/inverse QMF, surround delay, and FIR state;
|
||||
- the 1217-sample LFE delay;
|
||||
- OAMD Q15 coordinate conversion;
|
||||
- equal-power panning over target-layout regions;
|
||||
- layout-dependent position compensation and sample-wise gain ramps;
|
||||
- float32 and PCM24 output quantization;
|
||||
- Rosella 64-band QMF, 77-band hybrid processing, `77×36` direction fields, distance/ITD, room FIR, and special LFE.
|
||||
|
||||
See the [mathematical notes](docs/math.en.md) for the equations used by the decoding and rendering process.
|
||||
|
||||
## Known limitations
|
||||
|
||||
- Only the common contiguous EMDF transport is covered. Fragmented transport across multiple audio-block skip fields is not covered.
|
||||
- Dense JOC is the main path. The Sparse JOC branch should not be treated as supported.
|
||||
- The speaker and Rosella binaural paths currently cover ordinary point objects; extent, spread, diffuse, divergence, channel lock, and similar controls are outside the supported scope.
|
||||
- OAMD trim elements are boundary-checked and skipped; warp, balance, and trim parameters are not applied to raw object trajectories or speaker rendering.
|
||||
- Multi-data-point streams, uncommon band configurations, and unusual OAMD scheduling have less coverage than common 12-band, single-data-point material.
|
||||
- A speaker limiter is outside the current primary formula.
|
||||
- Rosella requires a user-supplied compatible `.personalized_headphone`; arbitrary SOFA data cannot become a valid Rosella rp through JSON rearrangement alone.
|
||||
- ADM output, native binaries, speaker layouts, and binaural models still need broader interoperability checks across platforms, players, and real material.
|
||||
|
||||
## Documentation
|
||||
|
||||
- [Mathematical notes](docs/math.en.md) · [中文](docs/math.md)
|
||||
- [Native-core notes](docs/native.en.md) · [中文](docs/native.md)
|
||||
- [DLL-free Rosella binaural](docs/binaural.en.md) · [中文](docs/binaural.md)
|
||||
MIT, see [LICENSE](LICENSE). Third-party provenance and patent boundaries are in
|
||||
[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md).
|
||||
|
||||
@@ -1,207 +1,138 @@
|
||||
# JustOneCacophony — JOC
|
||||
# JustOneCacophony — C++ Core
|
||||
|
||||
[English](README.en.md)
|
||||
[English](README.en.md) · [数学说明](docs/math.md) · [双耳渲染](docs/binaural.md) · [SIMD 派发](docs/simd.md)
|
||||
|
||||
> JustOneCacophony 是一个 E-AC-3 JOC 的实验性 / 测试实现,用于研究 JOC 的解析、重建、渲染以及相关数学过程。
|
||||
JustOneCacophony 的 C++ 实现:E-AC-3 JOC 码流解析、对象重建与渲染的执行内核。它从 E-AC-3
|
||||
同步帧中提取 EMDF、ID14 JOC 参数与 ID11 OAMD 元数据,结合 FFmpeg 解码出的核心 5.1 PCM
|
||||
重建 LFE 与 15 路对象 PCM,并输出 ADM BWF、指定扬声器布局的 WAV,或用编译好的 HRTF 方向场
|
||||
直接输出双耳 WAV。
|
||||
|
||||
项目可以从常见 E-AC-3 JOC 码流中提取并解析 EMDF、ID14 JOC 参数和 ID11 OAMD 元数据,结合 FFmpeg 解码出的核心 5.1 PCM 重建 LFE 与 15 路对象 PCM,并输出 ADM BWF、指定扬声器布局的 WAV,或直接输出 DLL-free Rosella 双耳 WAV。
|
||||
这是研究代码,不是完整、标准兼容或生产级的 JOC 解码器。它只覆盖已实现的码流形态,遇到未知
|
||||
变体时明确报错,而不是假装一切都很和谐。
|
||||
|
||||
这是研究代码,不是完整、标准兼容或生产级的 JOC 解码器。它只覆盖当前已实现的码流形态;遇到未知变体时会明确报错,而不是假装一切都很和谐——如果哪里算错了,它可能就真的只剩 cacophony 了。
|
||||
## 构建
|
||||
|
||||
## 当前功能
|
||||
```powershell
|
||||
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release
|
||||
cmake --build build
|
||||
ctest --test-dir build --output-on-failure
|
||||
```
|
||||
|
||||
- 扫描 E-AC-3 同步帧中的常见连续 EMDF 容器;
|
||||
- 解析 ID14 dense JOC 参数、Huffman 数据、差分矩阵与 `joc_clipgain`;
|
||||
- 解析 ID11 OAMD 位置更新并生成对象轨迹;
|
||||
- 通过 analysis QMF、参数插值、对象矩阵和 inverse QMF 重建 LFE + 15 路对象 PCM;
|
||||
- 输出 25 声道 ADM BWF:10 声道 7.1.2 bed(除 LFE 外静音)+ 15 个对象;
|
||||
- 直接渲染 `2.0`、`3.1`、`5.1`、`7.1`、`5.1.2`、`5.1.4`、`7.1.2`、`7.1.4`、`9.1.4`、`9.1.6`;
|
||||
- 从 `pcm16 + ID11/OAMD` 直接运行 DLL-free Rosella 双耳渲染,不生成临时 ADM BWF;
|
||||
- 双耳 DSP 全程使用 float64/complex128,并保留 961-sample latency compensation、跨帧状态和 room 尾声;
|
||||
- 直接输出统一支持 float32 或 PCM24 WAV,并在 PCM24 削波前提供明确处理策略;
|
||||
- 使用 NumPy 后端,或通过 `ctypes` 调用可选的 C++20 原生核;`auto` 模式在原生库不可用时回退到 Python;
|
||||
- 读取或写入 metadata sidecar,并生成元数据、运行时间和输出摘要。
|
||||
Windows(MSVC / VS 2022,静态 CRT)、Linux 与 macOS 由 `.github/workflows/ci.yml` 同时构建并跑
|
||||
单元测试;本地只验证 MSVC。浮点行为是逐字节验收的一部分,因此不启用 fast-math:MSVC 用
|
||||
`/fp:precise`,其他编译器用 `-fno-fast-math`。
|
||||
|
||||
## 处理流程
|
||||
Windows Release 默认带 `/arch:AVX2`(`JOC_ENABLE_AVX2`,默认 ON,见 `CMakeLists.txt`)。这个
|
||||
开关是逐字节验收过的:所有渲染产物的 SHA-256 完全一致,换来 SOFA 双耳内核 12.4%、Rosella
|
||||
1.8% 的提升。代价是运行要求——这样的 `joc_core.dll` 会执行 AVX2 指令,在 2013 年以前的 x86 上
|
||||
直接非法指令退出。没有运行时派发,一个二进制只能二选一,所以:
|
||||
|
||||
```powershell
|
||||
cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release -DJOC_ENABLE_AVX2=OFF
|
||||
```
|
||||
|
||||
得到基线 ISA(SSE2)、任何 x86-64 都能跑的产物。非 MSVC 构建永远不会带上这个开关。
|
||||
|
||||
### 路径与编码
|
||||
|
||||
库内部所有路径都是 **UTF-8**,只在系统边界转换(`src/foundation/fs_utf8.*`):Windows 上经
|
||||
`std::filesystem::path`(内部 UTF-16)落到 `_wfopen`/`CreateProcessW`,其他平台直接是字节。
|
||||
命令行参数在 Windows 上由 `GetCommandLineW` + `CommandLineToArgvW` 重新解析,控制台设为
|
||||
UTF-8,因此日文/中文等非 ASCII 路径(含 ffmpeg 子进程与输出文件)都能正常工作;单元测试里有
|
||||
一条非 ASCII 路径的回归用例守在 `ctest` 里。
|
||||
|
||||
## 产物
|
||||
|
||||
| 产物 | 说明 |
|
||||
|---|---|
|
||||
| `joc_core.dll` | 执行内核:码流解析、JOC/OAMD、DSP、扬声器与双耳渲染、ADM BWF/WAV 落盘、文件任务、流式接口 |
|
||||
| `joc_cli.exe` | 文件任务命令行前端 |
|
||||
| `include/joc_core.h` | 引擎、遥测与文件任务接口(纯 C) |
|
||||
| `include/joc_stream.h` | 面向嵌入者的流式 push/pull 接口(纯 C,仅包含它即可) |
|
||||
|
||||
## 命令行
|
||||
|
||||
参数与上游 Python CLI 完全一致:输入是位置参数,**默认输出 25 通道 ADM BWF**,用
|
||||
`--speaker-layout` 或 `--binaural` 切换到另外两种模式;未指定 `-o` 时产物落在 `output/`。
|
||||
|
||||
```powershell
|
||||
# 默认:ADM BWF(本身就是 24-bit,没有也不需要格式参数)
|
||||
joc_cli "07. Gold Forever (2021 Master).m4a"
|
||||
# -> output/07. Gold Forever (2021 Master).adm.wav
|
||||
|
||||
joc_cli input.m4a -o out/adm.wav # 指定输出
|
||||
|
||||
# 扬声器布局
|
||||
joc_cli input.m4a --speaker-layout 5.1 # -> output/<名称>.5.1.wav
|
||||
joc_cli input.m4a --speaker-layout 7.1.4 --speaker-output out/714.wav --speaker-format int24
|
||||
|
||||
# 双耳(HRTF 默认取 <exe>/HRTF/binaural.sofa,其次 <exe>/HRTF/binaural.personalized_headphone)
|
||||
joc_cli input.m4a --binaural
|
||||
joc_cli input.m4a --binaural --sofa-hrtf HRTF/other.sofa # 换一个 SOFA
|
||||
joc_cli input.m4a --binaural --personalized-headphone # Rosella 个性化模型
|
||||
# -> output/<名称>.binaural.wav
|
||||
|
||||
# 其它常用开关
|
||||
joc_cli input.m4a --duration 30 --gain-db -3 --trajectory-mode dense64
|
||||
joc_cli input.eac3 --metadata-only --print-metadata summary # 只解析并打印元数据
|
||||
```
|
||||
|
||||
`--speaker-format` / `--binaural-format` 默认 `float32`;`int24` 若会削波按 `--clip-action`
|
||||
处理(默认 `ask`,非交互终端下需显式给出 `continue`/`float32`/`abort`)。`--duration` 以**秒**
|
||||
为单位;`--object-delay-samples`、`--speaker-metadata-offset`、`--binaural-tail-seconds`、
|
||||
`--binaural-tail-threshold`(默认 1e-8,双耳尾音裁切阈值)等默认值与上游一致。命令总是写出
|
||||
`<输出>.report.json`(`--report-json` 可改路径)。
|
||||
|
||||
**与上游参数的差异**:`--sofa-hrtf`、`--personalized-headphone`、`--backend python` 以及
|
||||
metadata sidecar(`--metadata-dir`/`--metadata-cache`/`--metadata-backend sidecar`)在本构建中
|
||||
不可用,给出时立刻报错并说明原因,而不是静默忽略。本构建额外提供 `--bed`(已解码的 6 通道
|
||||
float32 PCM,给出后不调用 ffmpeg 解码)、`--kernels`(滤波器组表路径)、`--work-dir`、
|
||||
`--report-json`、`--dry-run`、`--quiet`。
|
||||
|
||||
## 库集成
|
||||
|
||||
引擎与文件任务使用 `joc_core.h`:`joc_task_validate` / `joc_task_execute` 跑一个文件任务并
|
||||
通过回调返回状态事件,`joc_task_result` 给出帧数、峰值、字节数与 SHA-256。
|
||||
|
||||
播放器或解码组件使用 `joc_stream.h`:调用方按自己的节奏推入 E-AC-3 字节与对应的核心 PCM
|
||||
(或已重建的 objects16),再拉取渲染后的 PCM;分块粒度任意,输出与文件任务逐字节一致。
|
||||
|
||||
```c
|
||||
joc_stream_config config = {0};
|
||||
config.struct_size = sizeof(config);
|
||||
config.input = JOC_STREAM_IN_EAC3; /* 或 JOC_STREAM_IN_PCM_OBJECTS16 */
|
||||
config.output = JOC_STREAM_OUT_SPEAKER; /* 或 BINAURAL / PCM_OBJECTS16 */
|
||||
config.speaker_layout_name = "5.1";
|
||||
joc_stream* stream = NULL;
|
||||
joc_stream_create(&config, &stream);
|
||||
/* 循环:joc_stream_push(...) / joc_stream_pull(...) */
|
||||
joc_stream_flush(stream);
|
||||
joc_stream_destroy(stream);
|
||||
```
|
||||
|
||||
契约:状态实例私有、可并存;同一实例的 push/pull 必须在同一线程;渲染是有状态的,
|
||||
因此**本版本不提供 seek**——定位需要从流起点重新解码。扬声器/双耳通路的内核延迟为
|
||||
961 样本,`joc_stream_flush` 负责排空双耳房间尾音。
|
||||
|
||||
## 目录
|
||||
|
||||
```text
|
||||
M4A / E-AC-3
|
||||
├─ FFmpeg 提取 E-AC-3 并解码核心 5.1 PCM
|
||||
├─ EMDF → ID14 JOC 参数 → 对象矩阵
|
||||
├─ 核心 PCM → analysis QMF → 参数插值 → inverse QMF
|
||||
├─ ID11 OAMD → 对象位置与时间轨迹
|
||||
└─ LFE + 15 objects
|
||||
├─ 25ch ADM BWF
|
||||
├─ 指定布局的扬声器 WAV
|
||||
└─ ID11 直接时间轴 → Rosella → 双耳 WAV
|
||||
include/ 公共 C ABI:joc_core.h(引擎/文件任务)、joc_stream.h(流式)、eac3joc_core.h(上游 ABI)
|
||||
src/ 实现:eac3_transport、emdf、joc_bitstream、joc_core、oamd、timeline、speaker、
|
||||
binaural、hrtf、adm、io、telemetry、task、stream、api、cli、simd
|
||||
tests/ 单元测试(CTest,自足,不需要外部素材)
|
||||
docs/ 数学说明、双耳渲染流程与 SIMD 派发
|
||||
```
|
||||
|
||||
Python 与 C++ 后端使用同一组数学过程。原生核处理对象重建、扬声器渲染以及双耳 QMF/hybrid/room/synthesis;位流解析、OAMD 时间轴、模型解析和命令行逻辑仍在 Python 中。
|
||||
`src/` 下 8 个文件是 JustOneCacophony 原生库的逐字节副本,永不修改(见
|
||||
[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md))。
|
||||
|
||||
## 环境
|
||||
## 兼容性说明
|
||||
|
||||
- Python 3.10+
|
||||
- NumPy 1.24+
|
||||
- 独立的 FFmpeg 可执行程序;不需要 `ffmpeg-python`。默认从 `PATH` 查找,也可通过 `--ffmpeg` 指定可执行文件路径
|
||||
- 可选:支持 C++20 的 CMake 工具链,用于自行构建原生核
|
||||
`object_delay_samples` 默认 **1473**,与既有实现的行为保持一致;它是可配置字段,改动它会
|
||||
改变 OAMD/ADM 时间对齐。上游调查认为该值应为 0,本项目为保持逐字节等价暂不改默认值。
|
||||
|
||||
建议在项目专用虚拟环境中安装依赖:
|
||||
## 许可
|
||||
|
||||
```powershell
|
||||
python -m pip install -r requirements.txt
|
||||
```
|
||||
|
||||
如果 FFmpeg 不在 `PATH` 中:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --ffmpeg C:\path\to\ffmpeg.exe
|
||||
```
|
||||
|
||||
## 使用方法
|
||||
|
||||
默认输出 25 声道 ADM BWF:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a
|
||||
```
|
||||
|
||||
选择后端或输出路径:
|
||||
|
||||
```powershell
|
||||
python main.py input.eac3 -o output.adm.wav --backend python
|
||||
python main.py input.m4a --backend native --native-threads 2
|
||||
python main.py input.m4a --native-library lib/eac3joc_core.dll
|
||||
```
|
||||
|
||||
直接输出扬声器 WAV:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --speaker-layout 2.0 --speaker-format float32
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24
|
||||
python main.py input.m4a --speaker-layout 7.1.2 --speaker-output output.7.1.2.wav
|
||||
```
|
||||
|
||||
直接输出 Rosella 双耳 WAV(普通对象仅 Near/Mid/Far,默认 Mid):
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --binaural
|
||||
python main.py input.m4a --binaural --binaural-mode near
|
||||
python main.py input.m4a --binaural --binaural-mode far --binaural-format int24
|
||||
python main.py input.m4a --binaural --binaural-output output.binaural.wav `
|
||||
--personalized-headphone C:\HRTF\my.personalized_headphone
|
||||
```
|
||||
|
||||
默认模型路径为 `HRTF/binaural.personalized_headphone`。示例 HRTF 文件见:
|
||||
|
||||
https://professionalsupport.dolby.com/s/question/0D54u0000AAT85HCQT/the-state-of-personalized-binaural-rendering?language=en_US
|
||||
|
||||
计算公式、状态和时间轴见[双耳渲染数学](docs/binaural.md)。
|
||||
|
||||
扬声器和双耳输出共享峰值检查、writer 与削波策略。在非交互环境请求 PCM24 且可能削波时,需要显式选择处理方式:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action abort
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action float32
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action continue
|
||||
python main.py input.m4a --binaural --binaural-format int24 --clip-action abort
|
||||
```
|
||||
|
||||
元数据与诊断:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --print-metadata summary
|
||||
python main.py input.m4a --metadata-only --print-metadata frames
|
||||
python main.py input.m4a --metadata-cache metadata_cache
|
||||
python main.py input.m4a --metadata-dir metadata_cache
|
||||
```
|
||||
|
||||
### 实验性 JOC 对象双耳模式设置
|
||||
|
||||
这里写入的双耳模式是用户手动指定、供下游 ADM 渲染器使用的实验性渲染提示,**不是从输入 E-AC-3 JOC 码流中提取或还原的原始双耳元数据**,也不代表原始混音中各对象的双耳设置。所选模式会统一应用到 15 个 JOC 对象;默认 `unspecified` 只是本工具的默认值,并非从源文件检测到的模式。
|
||||
|
||||
使用 `--joc-binaural-mode off|near|far|mid|unspecified` 选择模式,编码分别为 `0|1|2|3|4`,默认 `unspecified`:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --joc-binaural-mode mid
|
||||
```
|
||||
|
||||
此选项仅设置 ADM BWF 的 DBMD segment 10 中后 15 个 JOC 对象的 binaural render mode 低 3 bit;前 10 个 bed 保持不变。它不改变 PCM、对象轨迹、直接扬声器渲染或直接 Rosella 双耳渲染。输出旁的 `.report.json` 用 `joc_binaural_mode` 和 `joc_binaural_mode_value` 记录模式名称与数值;直接扬声器或直接双耳输出时两者为 `null`,表示不适用。
|
||||
|
||||
### 双耳计算
|
||||
|
||||
双耳路径的 QMF、hybrid、方向场、距离、ITD、room、512-sample 参数更新和 961-sample 延迟补偿见[双耳渲染数学](docs/binaural.md)。
|
||||
|
||||
更多参数可查看:
|
||||
|
||||
```powershell
|
||||
python main.py --help
|
||||
```
|
||||
|
||||
未指定 `-o` 时,输出仍写入仓库根目录的 `output/`。这是文件移动后特意保持的原有行为。
|
||||
|
||||
## 原生核
|
||||
|
||||
仓库默认不附带原生二进制。可以从项目 Release 下载适合当前平台的预构建运行库,或自行构建,然后把运行库直接放入仓库根目录的 `lib/`;若该目录不存在,创建即可。自行构建时可从仓库根目录使用 CMake:
|
||||
|
||||
```powershell
|
||||
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
|
||||
cmake --build build/cmake --config Release
|
||||
cmake --install build/cmake --config Release
|
||||
```
|
||||
|
||||
运行时查找顺序为:
|
||||
|
||||
1. `--native-library`;
|
||||
2. `EAC3JOC_NATIVE_LIBRARY`;
|
||||
3. `lib/` 下当前平台的标准库文件名。
|
||||
|
||||
详细 ABI、状态与精度说明见[原生核说明](docs/native.md)。
|
||||
|
||||
## 目录结构
|
||||
|
||||
```text
|
||||
JustOneCacophony/
|
||||
├─ main.py 命令行启动入口
|
||||
├─ src/ Python 实现模块
|
||||
├─ native/ C/C++ 加速核、C ABI 与必要表数据
|
||||
├─ data/ Python 运行时表数据
|
||||
├─ lib/ 原生运行库投放目录(按需创建)
|
||||
├─ docs/ 数学与原生核文档(中英文)
|
||||
├─ requirements.txt Python 依赖
|
||||
├─ README.md 中文说明
|
||||
└─ README.en.md English documentation
|
||||
```
|
||||
|
||||
## 数学实现
|
||||
|
||||
核心过程包括:
|
||||
|
||||
- dense JOC 差分还原与去量化;
|
||||
- 参数带到 64 个 QMF 子带的映射;
|
||||
- 跨帧参数插值;
|
||||
- analysis / inverse QMF、环绕声道延迟与 FIR 状态;
|
||||
- LFE 1217-sample 延迟;
|
||||
- OAMD Q15 坐标转换;
|
||||
- 基于目标布局 region 的等功率声像;
|
||||
- 布局位置补偿与逐样本增益斜坡;
|
||||
- float32 与 PCM24 输出量化;
|
||||
- Rosella 64-band QMF、77-band hybrid、`77×36` 方向 field、距离/ITD、room FIR 与 special LFE。
|
||||
|
||||
解码与渲染过程使用的公式见[数学说明](docs/math.md)。
|
||||
|
||||
## 已知限制
|
||||
|
||||
- 当前只覆盖常见 continuous EMDF transport;跨多个 audio-block skip field 的碎片化 transport 尚未覆盖。
|
||||
- Dense JOC 是当前主要路径;Sparse JOC 分支不应视为受支持能力。
|
||||
- 扬声器与 Rosella 双耳路径当前只覆盖普通点对象;extent、spread、diffuse、divergence、channel lock 等对象控制不在支持范围内。
|
||||
- OAMD trim element 会按声明边界校验并跳过;warp、balance 和 trim 参数不应用于当前原始对象轨迹或扬声器渲染。
|
||||
- 多数据点、少见参数带配置和特殊 OAMD 调度的覆盖度低于常见 12-band、单数据点素材。
|
||||
- 扬声器 limiter 不属于当前实现的主公式。
|
||||
- Rosella 路径需要用户提供兼容的 `.personalized_headphone`;任意 SOFA 不能仅靠 JSON 重排成为有效 Rosella rp。
|
||||
- ADM 输出、原生库、扬声器布局和双耳模型仍需在更多平台、播放器与真实素材上确认互操作性。
|
||||
|
||||
## 文档
|
||||
|
||||
- [数学说明](docs/math.md) · [English](docs/math.en.md)
|
||||
- [原生核说明](docs/native.md) · [English](docs/native.en.md)
|
||||
- [DLL-free Rosella 双耳](docs/binaural.md) · [English](docs/binaural.en.md)
|
||||
MIT,见 [LICENSE](LICENSE);第三方来源与专利边界见
|
||||
[THIRD_PARTY_NOTICES.md](THIRD_PARTY_NOTICES.md)。
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
# Third-party notices / 第三方通知
|
||||
|
||||
本文件记录 64-QMF / 77-hybrid 滤波器组表(`src/hrtf/public_filterbank.h`、
|
||||
`src/joc_core/qmf_tables.h`)与 JOC Huffman 表(`src/joc_bitstream/joc_huffman_tables.h`)
|
||||
的公开标准来源,以及 HRTF 数据与专利的边界说明。
|
||||
|
||||
## 公开标准来源
|
||||
|
||||
64-QMF → 77-hybrid 结构与 13-tap 低带 prototype 定义于
|
||||
[3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf)
|
||||
第 5.2.2 节(Table 1 的 $Q=8$/$Q=4$ 系数,delay 6):
|
||||
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr)$$
|
||||
|
||||
64-band QMF analysis 即 ISO/IEC 14496-3/AMD1:2003 第 4.B.18.2 节的 MPEG-4
|
||||
AAC/SBR 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap prototype 的多相重排:
|
||||
|
||||
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t}$$
|
||||
|
||||
QMF synthesis 表为 analysis 多相矩阵 $\mathbf{A}$ 的因果左逆
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$($\mathbf{P}$ 为 577-sample 延迟置换;
|
||||
全链 $961 = 577 + 6\times64$),rank-4 分解存储:
|
||||
|
||||
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
|
||||
|
||||
hybrid synthesis 表为 77→64 重组:高频带恒等 $Y_{3+b}=X_{16+b}$,低频带:
|
||||
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\mathrm{Re}X_q + j\,s_q\,\mathrm{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
|
||||
相同数值可在 FFmpeg(`aacps_tablegen.h`、`aacsbrdata.h`)等公开实现中查到。
|
||||
|
||||
## HRTF 数据与 `.jochrtf`
|
||||
|
||||
`.jochrtf` 含有特定源 SOFA/HRTF 数据集的变换系数与 delay;其使用、复制与再分发仍受源
|
||||
数据集许可约束,权限不明确时应作为私有 cache 保存。本仓库不分发任何 HRTF 数据集。
|
||||
|
||||
## 专利说明
|
||||
|
||||
标准可公开获取不等于获准实施相关专利。
|
||||
@@ -1,37 +0,0 @@
|
||||
# Python runtime tables
|
||||
|
||||
[中文](README.md)
|
||||
|
||||
This directory contains static production tables and the user-model directory.
|
||||
|
||||
`tables.npz` contains the JOC core decoding tables:
|
||||
|
||||
```text
|
||||
analysis_window float64[10,64]
|
||||
qmf5_window float64[640]
|
||||
joc_huff_code_coarse_generic int64[95,2]
|
||||
joc_huff_code_fine_generic int64[191,2]
|
||||
joc_huff_code_coarse_coeff_sparse int64[95,2]
|
||||
joc_huff_code_fine_coeff_sparse int64[191,2]
|
||||
joc_huff_code_5ch_pos_index_sparse int64[4,2]
|
||||
joc_huff_code_7ch_pos_index_sparse int64[6,2]
|
||||
```
|
||||
|
||||
`src/joc_qmf.py` loads the QMF tables, while `src/joc_decode.py` loads the JOC Huffman trees. Python does not read C/C++ headers under `native/`.
|
||||
|
||||
The corresponding native data are stored in `native/src/qmf_tables.h` and `native/src/joc_huffman_tables.h`. Changes on either side should update the other and be checked for value-by-value agreement.
|
||||
|
||||
## Binaural rendering tables
|
||||
|
||||
`rosella_kernels.npz` contains the fixed QMF/hybrid tables:
|
||||
|
||||
```text
|
||||
qmf_analysis_coefficients float32[64,10]
|
||||
hybrid_analysis_low_kernel float32[3,2,13,16,2]
|
||||
hybrid_synthesis_indices int16[154,4]
|
||||
hybrid_synthesis_values float32[154]
|
||||
qmf_synthesis_basis float64[64,4,128]
|
||||
qmf_synthesis_taps float64[64,10,4]
|
||||
```
|
||||
|
||||
The float32 table values are promoted to float64 when loaded.
|
||||
@@ -1,37 +0,0 @@
|
||||
# Python 运行时表
|
||||
|
||||
[English](README.en.md)
|
||||
|
||||
本目录保存 Python 生产路径使用的静态表数据与用户模型目录。
|
||||
|
||||
`tables.npz` 保存 JOC 核心解码表:
|
||||
|
||||
```text
|
||||
analysis_window float64[10,64]
|
||||
qmf5_window float64[640]
|
||||
joc_huff_code_coarse_generic int64[95,2]
|
||||
joc_huff_code_fine_generic int64[191,2]
|
||||
joc_huff_code_coarse_coeff_sparse int64[95,2]
|
||||
joc_huff_code_fine_coeff_sparse int64[191,2]
|
||||
joc_huff_code_5ch_pos_index_sparse int64[4,2]
|
||||
joc_huff_code_7ch_pos_index_sparse int64[6,2]
|
||||
```
|
||||
|
||||
`src/joc_qmf.py` 读取 QMF 表,`src/joc_decode.py` 读取 JOC Huffman 树。Python 不读取 `native/` 下的 C/C++ 头文件。
|
||||
|
||||
原生侧对应数据分别位于 `native/src/qmf_tables.h` 与 `native/src/joc_huffman_tables.h`。修改任何一侧时,应同步更新另一侧并进行逐值一致性检查。
|
||||
|
||||
## 双耳渲染表
|
||||
|
||||
`rosella_kernels.npz` 保存双耳 QMF/hybrid 固定表:
|
||||
|
||||
```text
|
||||
qmf_analysis_coefficients float32[64,10]
|
||||
hybrid_analysis_low_kernel float32[3,2,13,16,2]
|
||||
hybrid_synthesis_indices int16[154,4]
|
||||
hybrid_synthesis_values float32[154]
|
||||
qmf_synthesis_basis float64[64,4,128]
|
||||
qmf_synthesis_taps float64[64,10,4]
|
||||
```
|
||||
|
||||
float32 表值载入后提升为 float64。
|
||||
Binary file not shown.
Binary file not shown.
+238
-340
@@ -1,348 +1,246 @@
|
||||
# JustOneCacophony — Binaural Rendering Mathematics
|
||||
# Binaural rendering
|
||||
|
||||
[中文](binaural.md) · [Back to README](../README.en.md)
|
||||
|
||||
This document defines the `pcm16 + ID11/OAMD → stereo` calculation. The path begins after object reconstruction and does not pass through ADM BWF or AXML.
|
||||
|
||||
## 1. Signal path and notation
|
||||
JustOneCacophony's binaural backend supports three HRTF sources:
|
||||
`SimpleFreeFieldHRIR` SOFA, the Rosella `.personalized_headphone` model exported
|
||||
by Dolby's official personalization scan (its JSON parsing is implemented by
|
||||
this project and invokes no Dolby software), and the `.jochrtf` cache compiled
|
||||
from SOFA. SOFA is compiled into an in-memory directional field when the model
|
||||
is loaded. A `.jochrtf` file is only a disposable, reproducible JOC compiled
|
||||
HRTF cache; it is neither an interchange format nor a prerequisite for using
|
||||
SOFA.
|
||||
|
||||
```text
|
||||
LFE + 15 object PCM channels
|
||||
→ 64-band QMF analysis
|
||||
→ 77-band hybrid analysis
|
||||
→ per-object geometry, transfer functions, and room send
|
||||
→ direct accumulation + room network
|
||||
→ hybrid synthesis
|
||||
→ QMF synthesis
|
||||
→ 961-sample latency compensation
|
||||
→ stereo WAV
|
||||
SOFA FIR
|
||||
-> CanonicalHrtf
|
||||
-> 48 kHz / one radius shell / delay-phase policy
|
||||
-> 64-QMF / 77-hybrid projection
|
||||
-> fifth-order ACN/N3D real-SH field
|
||||
-> per-object direct + early reflections
|
||||
-> shared unitary-FDN late room
|
||||
-> float64 stereo
|
||||
```
|
||||
|
||||
| Symbol | Meaning |
|
||||
## Inputs
|
||||
|
||||
The binaural backend consumes a compiled directional field (a JOC compiled HRTF cache,
|
||||
`.jochrtf`) plus the shared filterbank tables. Both are read-only inputs: this library
|
||||
performs no parsing or conversion of measurement data formats.
|
||||
|
||||
```powershell
|
||||
joc_cli input.m4a --binaural `
|
||||
--compiled-hrtf-cache path\to\subject.jochrtf `
|
||||
--kernels data\rosella_kernels.npz `
|
||||
--binaural-mode mid --binaural-tail-seconds 5.0
|
||||
```
|
||||
|
||||
The library exposes the same fields: `joc_task_config` for a file task,
|
||||
`joc_stream_config` for streaming, where `hrtf_path`, `kernels_path`,
|
||||
`binaural_mode` and `binaural_tail_seconds` configure the binaural path.
|
||||
|
||||
The `.jochrtf` file is an **input**, not a product of this library: compiling it from
|
||||
SOFA data or measurements belongs to the toolchain and is decoupled from this
|
||||
repository. Loading validates the member set, dtypes and shapes, C-contiguity,
|
||||
CRC-32 and a payload hash recomputed over the members (see `.jochrtf` below).
|
||||
|
||||
`binaural_mode` is `near`, `mid` or `far` and selects one of the backend's three
|
||||
preset parameter sets; `binaural_tail_seconds` sets the room-tail length drained on
|
||||
flush (default 5.0 s).
|
||||
|
||||
## Binaural render mode
|
||||
|
||||
`--binaural-mode off|near|mid|far` (default `mid`) is a **human-specified
|
||||
rendering hint**, not original binaural metadata extracted or recovered from the
|
||||
input E-AC-3 JOC bitstream:
|
||||
|
||||
- Direct binaural rendering (`--binaural`): near/mid/far apply, default `mid`;
|
||||
`off` is an error;
|
||||
- ADM BWF: the low 3 binaural-render-mode bits of the last 15 JOC object entries
|
||||
in DBMD segment 10 carry `off=0/near=1/far=2/mid=3`, leaving the first 10 bed
|
||||
entries unchanged; the default is `mid`, and `off` explicitly disables the
|
||||
binaural metadata hint.
|
||||
|
||||
## Canonical SOFA contract
|
||||
|
||||
The strict importer currently accepts:
|
||||
|
||||
- `Conventions=SOFA`;
|
||||
- `SOFAConventions=SimpleFreeFieldHRIR`, version `0.4`, `1.0`, or `1.1`;
|
||||
- `DataType=FIR` and `Data.IR[M,2,N]`;
|
||||
- one positive finite `Data.SamplingRate` in hertz/Hz;
|
||||
- spherical or Cartesian `SourcePosition`;
|
||||
- singleton or per-measurement `ListenerPosition/View/Up`;
|
||||
- two receivers whose listener-local lateral geometry uniquely identifies L/R;
|
||||
- one zero-offset emitter;
|
||||
- causal `Data.Delay[I,2]` or `[M,2]`;
|
||||
- an explicitly free-field/anechoic `RoomType`.
|
||||
|
||||
Receiver order comes from geometry, never from the receiver array index. SOFA
|
||||
listener coordinates are $+X$ front,
|
||||
$+Y$ left,
|
||||
$+Z$ up; ADM coordinates are
|
||||
$+X$ right,
|
||||
$+Y$ front,
|
||||
$+Z$ up:
|
||||
|
||||
$$\bigl(x_{\mathrm{SOFA}},\ y_{\mathrm{SOFA}},\ z_{\mathrm{SOFA}}\bigr) = \bigl(y_{\mathrm{ADM}},\ -x_{\mathrm{ADM}},\ z_{\mathrm{ADM}}\bigr)$$
|
||||
|
||||
`CanonicalHrtf` keeps `Data.IR` and `Data.Delay` separate. Only a time-domain
|
||||
baseline calls `materialized_measurement()` to apply delay once; the runtime SH
|
||||
path never materializes and then restores the delay. Non-48-kHz HRIRs are
|
||||
normalized with float64 `scipy.signal.resample_poly`, and delay samples scale by
|
||||
the same ratio.
|
||||
|
||||
GeneralFIR, BRIR, TF, multiple emitters, ambiguous receivers, and non-free-field
|
||||
data require convention-specific adapters. They cannot enter the core importer
|
||||
through a reshape.
|
||||
|
||||
## Exactly-once delay and phase
|
||||
|
||||
The compiler recognizes three mutually exclusive representations:
|
||||
|
||||
1. Nonzero `Data.Delay` is external to `Data.IR`; the FIR is not de-rotated and
|
||||
runtime applies the delay once.
|
||||
2. With `Data.Delay=0` and an ordinary positive-onset HRIR, each ear's main peak
|
||||
supplies arrival time. Compilation separates it and runtime restores it once.
|
||||
The current threshold is a peak index greater than two samples.
|
||||
3. With `Data.Delay=0` and both FIRs at a shared sample-zero origin, no external
|
||||
delay is invented. The authored complex phase stays in the fifth-order field.
|
||||
|
||||
No path may add a second ear delay or phase-group delay.
|
||||
|
||||
## Public filterbank and directional field
|
||||
|
||||
The runtime is fixed at:
|
||||
|
||||
- 48 kHz;
|
||||
- a 64-sample QMF hop;
|
||||
- 64-QMF / 77 hybrid bands;
|
||||
- 961 samples of analysis/synthesis latency;
|
||||
- fifth order, 36 terms, ACN/N3D real spherical harmonics;
|
||||
- float64 PCM, delay, SH, and room state; complex128 band transfers and spectra.
|
||||
|
||||
Real and imaginary unit gains for every hybrid band pass through the same
|
||||
analysis/synthesis chain to form a 154-real-parameter impulse dictionary. The
|
||||
compiler does not sample 77 FFT bins. Defaults are `1e-3` projection ridge and
|
||||
`1e-5` SH ridge. Coincident directions are merged before a spherical-Voronoi
|
||||
weighted ridge fit.
|
||||
|
||||
The fixed resource is `data/rosella_kernels.npz`, which implements publicly
|
||||
standardized filter banks, computable from the following formulas.
|
||||
|
||||
The hybrid analysis kernels are defined in [3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf),
|
||||
Section 5.2.2 (Table 1 $Q=8$/
|
||||
$Q=4$ coefficients, delay 6):
|
||||
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr),\qquad n=0,\dots,12$$
|
||||
|
||||
The QMF analysis table is the MPEG-4 AAC/SBR 64 complex QMF bank of
|
||||
ISO/IEC 14496-3/AMD1:2003, subclause 4.B.18.2, stored as the polyphase
|
||||
reordering of the public 640-tap prototype $c_0,\dots,c_{639}$:
|
||||
|
||||
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t},\qquad r=0,\dots,63,\ t=0,\dots,9$$
|
||||
|
||||
The QMF synthesis table is the causal left inverse of the analysis polyphase
|
||||
matrix $\mathbf{A}$, i.e. the solution of
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$
|
||||
($\mathbf{P}$ is the 577-sample delay permutation; total latency
|
||||
$961 = 577 + 6\times64$), stored as a rank-4 factorization:
|
||||
|
||||
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
|
||||
|
||||
The hybrid synthesis table is the 77→64 recombination: identity for the high
|
||||
bands, $Y_{3+b}=X_{16+b}$, and for the low bands(
|
||||
$C_p$ is the
|
||||
$8+4+4$ child partition):
|
||||
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\mathrm{Re}X_q + j\,s_q\,\mathrm{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
|
||||
The loader verifies the archive and every array by SHA-256; the table version,
|
||||
all array hashes, and the 77 reference band-center values are part of the cache
|
||||
key. Public availability of a standard does not by itself grant permission to
|
||||
practice related patent claims. See
|
||||
[`data/README.en.md`](../data/README.en.md) and
|
||||
[`THIRD_PARTY_NOTICES.md`](../THIRD_PARTY_NOTICES.md) for the sources and the
|
||||
rights boundary.
|
||||
|
||||
## `.jochrtf`
|
||||
|
||||
A `.jochrtf` file is a pickle-free compressed NumPy archive with an exact member set:
|
||||
|
||||
| key | dtype / shape |
|
||||
|---|---|
|
||||
| $s=0\ldots15$ | input source; source 0 is LFE |
|
||||
| $e\in\{L,R\}$ | output ear |
|
||||
| $k=0\ldots63$ | QMF band |
|
||||
| $h=0\ldots76$ | hybrid band |
|
||||
| $j=0\ldots35$ | direction-basis term |
|
||||
| $m$ | 64-sample QMF slot |
|
||||
|
||||
A control block is
|
||||
|
||||
$$N_b=512=8\times64,$$
|
||||
|
||||
and an input frame is
|
||||
|
||||
$$N_f=1536=3N_b.$$
|
||||
|
||||
All filter and room state continues across frame boundaries.
|
||||
|
||||
## 2. QMF analysis
|
||||
|
||||
Let $a_{p,\ell}$ be the fixed 64×10 polyphase coefficients and $r_{s,\ell,p}[m]$ the current and previous nine phase vectors:
|
||||
|
||||
$$
|
||||
E_{s,p}[m]=\sum_{\ell\text{ even}}a_{p,\ell}r_{s,\ell,p}[m],
|
||||
$$
|
||||
|
||||
$$
|
||||
O_{s,p}[m]=\sum_{\ell\text{ odd}}a_{p,\ell}r_{s,\ell,p}[m].
|
||||
$$
|
||||
|
||||
Define
|
||||
|
||||
$$
|
||||
\mathcal Q(v)_k=
|
||||
\operatorname{FFT}_{128}
|
||||
\left([v[p]e^{-j\pi p/128}]_{p=0}^{63},0_{64}\right)_k
|
||||
e^{-j3\pi(k+1/2)/128}.
|
||||
$$
|
||||
|
||||
The complex QMF output is
|
||||
|
||||
$$
|
||||
X_{s,k}[m]=\mathcal Q(O_s)_k+j(-1)^k\mathcal Q(E_s)_k.
|
||||
$$
|
||||
|
||||
## 3. Hybrid analysis
|
||||
|
||||
The lowest three QMF bands are split into sixteen hybrid bands by a 13-slot FIR:
|
||||
|
||||
$$
|
||||
H_{s,h,o}[m]
|
||||
=
|
||||
\sum_{p=0}^{2}\sum_{i=0}^{1}\sum_{\ell=0}^{12}
|
||||
X_{s,p,i}[m-\ell]K_{p,i,\ell,h,o},
|
||||
\qquad h=0\ldots15.
|
||||
$$
|
||||
|
||||
The remaining bands are delayed QMF bands 3..63:
|
||||
|
||||
$$
|
||||
H_{s,16+q}[m]=X_{s,3+q}[m-6],
|
||||
\qquad q=0\ldots60.
|
||||
$$
|
||||
|
||||
## 4. OAMD coordinates and time
|
||||
|
||||
The Q15 object fields are restored to their discrete grids:
|
||||
|
||||
$$
|
||||
u_1=\min\left(1,\frac{\operatorname{round}(62q_1/32767)}{62}\right),$$
|
||||
|
||||
$$
|
||||
u_2=\min\left(1,\frac{\operatorname{round}(62q_2/32767)}{62}\right),$$
|
||||
|
||||
$$
|
||||
u_3=\operatorname{clip}\left(
|
||||
\frac{\operatorname{round}(15q_3/32767)}{15},-1,1\right),$$
|
||||
|
||||
$$
|
||||
(X,Y,Z)=(2u_1-1,\ 1-2u_2,\ u_3).
|
||||
$$
|
||||
|
||||
An update is coded at
|
||||
|
||||
$$
|
||||
n_{\mathrm{coded}}
|
||||
=n_{\mathrm{frame}}+n_{\mathrm{outer}}+n_{\mathrm{block}}.
|
||||
$$
|
||||
|
||||
The first valid state is the position at sample 0. Later updates add the object delay $D_o=1473$. For $R>64$:
|
||||
|
||||
$$
|
||||
n_{\mathrm{start}}=n_{\mathrm{coded}}+D_o+64,$$
|
||||
|
||||
$$R_{\mathrm{eff}}=R-64,$$
|
||||
|
||||
$$
|
||||
\mathbf p[n]=(1-\alpha)\mathbf p_0+\alpha\mathbf p_1,
|
||||
\qquad
|
||||
\alpha=\frac{n-n_{\mathrm{start}}}{R_{\mathrm{eff}}}.
|
||||
$$
|
||||
|
||||
The position is evaluated at each 512-sample block boundary.
|
||||
|
||||
## 5. Distance profile and direction
|
||||
|
||||
Each Near, Mid, or Far profile contains six bounds, distance scale $D$, inverse scale $D^{-1}$, three axis scales, and minimum radius $\rho_{\min}$.
|
||||
|
||||
After axis conversion and scale:
|
||||
|
||||
$$
|
||||
\mathbf s=(a_zq_f,a_xq_l,a_yq_v).
|
||||
$$
|
||||
|
||||
A single ray factor $\lambda\le1$ keeps the point inside the profile bounds:
|
||||
|
||||
$$
|
||||
\mathbf s'=\lambda\mathbf s.
|
||||
$$
|
||||
|
||||
Then
|
||||
|
||||
$$
|
||||
\rho=\|\mathbf s'\|_2,
|
||||
\quad
|
||||
\rho_c=\max(\rho,\rho_{\min}),
|
||||
\quad
|
||||
\alpha=\rho/\rho_c,
|
||||
$$
|
||||
|
||||
$$
|
||||
\mathbf d=\mathbf s'/\rho,
|
||||
\qquad
|
||||
R=D\rho.
|
||||
$$
|
||||
|
||||
## 6. Direction basis and ear paths
|
||||
|
||||
The direction is expanded into a fixed 36-term polynomial basis:
|
||||
|
||||
$$
|
||||
\mathbf b(\mathbf d)=
|
||||
[1,x,y,z,x^2-\tfrac13,xy,xz,y^2-\tfrac13,yz,\ldots]^T.
|
||||
$$
|
||||
|
||||
For ear offset $e$:
|
||||
|
||||
$$
|
||||
\epsilon=\frac{eD^{-1}}{\rho_c},
|
||||
$$
|
||||
|
||||
$$
|
||||
\mathbf d_{\mp}=
|
||||
\frac{(x,y\mp\epsilon,z)}{\|(x,y\mp\epsilon,z)\|_2}.
|
||||
$$
|
||||
|
||||
The normalized paths are
|
||||
|
||||
$$
|
||||
\ell_{\mp}=\rho_c\sqrt{x^2+(y\mp\epsilon)^2+z^2}.
|
||||
$$
|
||||
|
||||
A model direction vector may add a non-negative path correction:
|
||||
|
||||
$$
|
||||
\ell'_e=\ell_e+
|
||||
\max(\mathbf v_e^T\mathbf b_e,0)\,2cD^{-1}.
|
||||
$$
|
||||
|
||||
The interaural delay is
|
||||
|
||||
$$
|
||||
\tau=|\ell'_+-\ell'_-|D\frac{48000}{343.3}\alpha.
|
||||
$$
|
||||
|
||||
The longer path receives the hybrid phase
|
||||
|
||||
$$P_h=e^{j\omega_h\tau}.$$
|
||||
|
||||
## 7. Direction fields and direct gains
|
||||
|
||||
Each ear has a 77×36 complex field:
|
||||
|
||||
$$
|
||||
C_{e,h}(\mathbf d_e)=
|
||||
\sum_{j=0}^{35}F_{e,h,j}b_j(\mathbf d_e).
|
||||
$$
|
||||
|
||||
Path weights are
|
||||
|
||||
$$
|
||||
w_L=\frac{\ell_+}{\sqrt{\ell_-^2+\ell_+^2}},
|
||||
\qquad
|
||||
w_R=\frac{\ell_-}{\sqrt{\ell_-^2+\ell_+^2}}.
|
||||
$$
|
||||
|
||||
For effective distance $R_e=\rho s_dD$, Mid and Far use
|
||||
|
||||
$$
|
||||
g_c=\frac{1}{\sqrt{1+s_rR_e^2}},
|
||||
\qquad
|
||||
g_{\mathrm{room}}=R_eg_c.
|
||||
$$
|
||||
|
||||
Near uses $g_c=1$ and $g_{\mathrm{room}}=0$. With field term zero denoted by $C^{(0)}$:
|
||||
|
||||
$$
|
||||
G_{L,h}=g_c[C_{L,h}w_L\alpha+C_{L,h}^{(0)}c_L(1-\alpha)],
|
||||
$$
|
||||
|
||||
$$
|
||||
G_{R,h}=g_c[C_{R,h}w_R\alpha+C_{R,h}^{(0)}c_R(1-\alpha)].
|
||||
$$
|
||||
|
||||
## 8. LFE
|
||||
|
||||
LFE bypasses ordinary-object geometry:
|
||||
|
||||
$$
|
||||
G_{L,h}^{\mathrm{LFE}}=G_{R,h}^{\mathrm{LFE}}=
|
||||
\begin{cases}
|
||||
g_h,&0\le h<16,\\0,&16\le h<77.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
```text
|
||||
2.60290003, 1.80741799, 0.659342408, -0.0275855921,
|
||||
-0.105803289, -0.0699509233, 0.0749056414, -0.00919809937,
|
||||
0.00349014648,-0.0158600751,-0.000723021978,0.00188189559,
|
||||
-0.000421735429,0.0000329252762,0.0000317397971,0.000000580376991
|
||||
```
|
||||
|
||||
Its room send is zero.
|
||||
|
||||
## 9. Source accumulation and room network
|
||||
|
||||
Direct output and room input are
|
||||
|
||||
$$
|
||||
Y^{\mathrm{direct}}_{e,h}=
|
||||
\sum_{s=0}^{15}H_{s,h}G_{s,e,h},
|
||||
$$
|
||||
|
||||
$$
|
||||
U_h=\sum_{s=1}^{15}H_{s,h}g_{\mathrm{room},s}.
|
||||
$$
|
||||
|
||||
The room input is scaled by $0.70710677$. Each all-pass stage uses
|
||||
|
||||
$$r[n]=x[n]-ad[n],$$
|
||||
|
||||
$$y[n]=ar[n]+d[n].$$
|
||||
|
||||
For the four-branch delay network:
|
||||
|
||||
$$
|
||||
\mathbf b_h[m]=U_h[m]\mathbf1+M\mathbf d_h[m],
|
||||
$$
|
||||
|
||||
$$m_{h,i}[m]=f_{h,i}b_{h,i}[m].$$
|
||||
|
||||
The main tap, optional extra taps, and ear output matrices produce
|
||||
|
||||
$$
|
||||
Y^{\mathrm{room}}_{e,h}[m]=
|
||||
\sum_{i=0}^{3}O_{e,h,i}z_{h,i}[m].
|
||||
$$
|
||||
|
||||
The final hybrid signal is
|
||||
|
||||
$$Y_{e,h}=Y^{\mathrm{direct}}_{e,h}+Y^{\mathrm{room}}_{e,h}.$$
|
||||
|
||||
The Python backend uses a finite complex FIR/overlap-add realization. The C++ backend keeps the recursive room state directly.
|
||||
|
||||
## 10. Hybrid and QMF synthesis
|
||||
|
||||
Hybrid synthesis is a 154-entry sparse map. For an entry $(h,i,k,o,w)$:
|
||||
|
||||
$$Q_{e,k,o}[m]\mathrel{+}=Y_{e,h,i}[m]w.$$
|
||||
|
||||
The complex QMF vector is flattened to
|
||||
|
||||
$$
|
||||
\mathbf q_e=[\Re Q_{e,0},\Im Q_{e,0},\ldots,\Re Q_{e,63},\Im Q_{e,63}]^T.
|
||||
$$
|
||||
|
||||
Rank-four features and ten-slot synthesis are
|
||||
|
||||
$$f_{e,p,r}[m]=\mathbf b_{p,r}^T\mathbf q_e[m],$$
|
||||
|
||||
$$
|
||||
y_e[64m+p]=
|
||||
\sum_{\ell=0}^{9}\sum_{r=0}^{3}
|
||||
t_{p,\ell,r}f_{e,p,r}[m-\ell].
|
||||
$$
|
||||
|
||||
## 11. Latency, tail, and precision
|
||||
|
||||
The filterbank latency is 961 samples and is removed once at the beginning of the continuous stream. Zero input is then processed to release filterbank and room state. Tail trimming keeps the final sample satisfying
|
||||
|
||||
$$
|
||||
\max(|y_L[n]|,|y_R[n]|)>10^{-8},
|
||||
$$
|
||||
|
||||
while never shortening the output below the source PCM length.
|
||||
|
||||
All internal state, geometry, field products, room processing, source accumulation, and tail processing use `float64/complex128`. Conversion to float32 or PCM24 occurs only in the final writer.
|
||||
|
||||
## 12. Backends and model path
|
||||
|
||||
Python and C++ use the same fixed tables, parsed model parameters, 512-sample control timeline, direct gains, room sends, latency compensation, and tail policy.
|
||||
|
||||
The C++ backend owns QMF, hybrid, recursive room, and synthesis state. Python supplies parsed parameters and per-block gains.
|
||||
|
||||
The default model path is
|
||||
|
||||
```text
|
||||
HRTF/binaural.personalized_headphone
|
||||
```
|
||||
|
||||
Override it with `--personalized-headphone PATH`.
|
||||
|
||||
A SOFA FIR cannot be converted into this parameter model by array rearrangement alone. A conversion requires fitting the direction fields, ITD, distance profiles, ear geometry, and room parameters.
|
||||
|
||||
## 13. Scope
|
||||
|
||||
The current path covers fifteen point objects and one special LFE source. Extent, spread, diffuse, divergence, channel lock, and unsupported OAMD element variants are outside this model.
|
||||
| `metadata_json` | NumPy Unicode scalar containing JSON text (`dtype.kind == "U"`) |
|
||||
| `band_center_frequencies_hz` | little-endian `float64[77]` |
|
||||
| `coefficients` | little-endian `complex128[36,2,77]` |
|
||||
| `delay_coefficients` | little-endian `float64[36,2]` |
|
||||
| `delay_bounds` | little-endian `float64[2,2]` |
|
||||
|
||||
Metadata uses the `JOC-HRTF-CACHE` magic and records the schema, compiler and
|
||||
phase-policy versions, ACN/N3D convention, filterbank hashes, SOFA content
|
||||
SHA-256, sample rate, radius, order, both ridge values, payload hash, and fit
|
||||
report. Every setting that changes compilation participates in the cache key.
|
||||
Metadata never persists an absolute local `source_path`; it may keep a display
|
||||
name only.
|
||||
|
||||
Before constructing a field, the loader uses `allow_pickle=False` and validates
|
||||
ZIP members and expanded sizes, shapes, dtypes, byte order, contiguous layout,
|
||||
finite values, delay bounds, band centers, payload hash, and cache key. The
|
||||
writer uses a same-directory temporary file, `fsync`, a process-held OS file
|
||||
lock, and atomic `os.replace`. Its hidden `.lock` sidecar may remain and does not
|
||||
mean that a writer still owns the lock. Outdated, damaged, or mismatched
|
||||
caches cannot hit. SOFA input rebuilds an invalid cache; an explicitly selected
|
||||
cache reports the error.
|
||||
|
||||
Deleting a disk cache must not change the field or render produced from the same
|
||||
SOFA and compiler configuration.
|
||||
|
||||
A `.jochrtf` file contains directional-field coefficients and delay data
|
||||
transformed from the source HRIRs. Its reproducibility therefore does not make
|
||||
it licence-free. Creating a cache does not enlarge the rights granted by the
|
||||
source SOFA/HRTF dataset: use, copying, and redistribution remain subject to
|
||||
that dataset's terms. If those terms are unclear, keep `.jochrtf` as a private
|
||||
local cache and do not ship it with the program or another build artifact.
|
||||
`source_sha256` is only a content-integrity identifier, not proof of provenance
|
||||
or permission.
|
||||
|
||||
## JOC objects and room behavior
|
||||
|
||||
The production adapter retains the existing JOC schedule:
|
||||
|
||||
- `[1536,16]` input per frame;
|
||||
- channel 0 is special LFE and channels 1..15 are JOC objects;
|
||||
- ID11/OAMD positions use a sample-timed timeline;
|
||||
- source parameters update every 512 samples;
|
||||
- every object owns independent direct/early history while one late FDN is shared;
|
||||
- `finish()` drains early/late tails; output gain is explicit, with no implicit
|
||||
limiter or programme loudness normalization.
|
||||
|
||||
Near/Mid/Far, equal-power direct level, six first-order shoebox image sources,
|
||||
late sends, the unitary FDN, the 120–180 Hz cosine-squared LFE low-pass, and room
|
||||
calibration are JOC project-defined behavior, not constants published by SOFA or
|
||||
Dolby.
|
||||
|
||||
The binaural renderer is implemented inside this library: the filterbank, the SH
|
||||
directional-field evaluation, the per-object early/direct histories and the shared
|
||||
FDN all run in `joc_core` (the `ejoc_sofa_binaural_*` kernel), consuming a compiled
|
||||
directional field and 512-sample metadata updates.
|
||||
|
||||
## Technical references and rights boundary
|
||||
|
||||
- [SOFA SimpleFreeFieldHRIR convention](https://www.sofaconventions.org/mediawiki/index.php/SimpleFreeFieldHRIR)
|
||||
- [3GPP TS 26.405 / ETSI TS 126 405 (64-QMF/77-hybrid definition)](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf)
|
||||
- [Dolby binaural render-mode workflow](https://professionalsupport.dolby.com/s/article/What-is-Binaural-Render-Mode-and-how-do-the-settings-affect-my-mix)
|
||||
- [EP3090576A1](https://patents.google.com/patent/EP3090576A1/en), used only as
|
||||
architectural background for direct/early/late, subbands, and FDNs; it does
|
||||
not establish that any product uses a particular embodiment.
|
||||
|
||||
Public availability of a specification, source file, or patent document does
|
||||
not by itself authorize copying its contents, redistribution of derivatives,
|
||||
or practice of patent claims. These technical references grant no patent
|
||||
licence and make no non-infringement representation. Anyone preparing a release
|
||||
or product integration must assess the applicable data and software licences,
|
||||
patent permissions, and freedom to operate. See
|
||||
[`THIRD_PARTY_NOTICES.md`](../THIRD_PARTY_NOTICES.md) for the public-standard
|
||||
provenance and rights boundary.
|
||||
|
||||
+202
-527
@@ -1,536 +1,211 @@
|
||||
# JustOneCacophony — 双耳渲染数学
|
||||
# 双耳渲染
|
||||
|
||||
[English](binaural.en.md) · [返回 README](../README.md)
|
||||
|
||||
本文说明 `pcm16 + ID11/OAMD → stereo` 路径中的计算、状态和时间对齐。双耳渲染直接接在对象重建之后,不经过 ADM BWF 或 AXML。
|
||||
|
||||
## 1. 总体路径与记号
|
||||
JustOneCacophony 的双耳后端支持三种 HRTF 来源:`SimpleFreeFieldHRIR` SOFA、
|
||||
杜比官方软件个性化扫描导出的 Rosella `.personalized_headphone`(JSON 解析由本项目
|
||||
自行实现,不调用杜比软件),以及从 SOFA 编译出的 `.jochrtf` 缓存。SOFA 在模型加载
|
||||
时编译成内存方向场;`.jochrtf` 只是可删除、可重建的 JOC compiled HRTF cache,
|
||||
不是交换格式,也不是使用 SOFA 的前置步骤。
|
||||
|
||||
```text
|
||||
pcm16:LFE + 15 路对象 PCM
|
||||
→ 64-band QMF analysis
|
||||
→ 77-band hybrid analysis
|
||||
→ 逐对象方向、距离、双耳传递函数和 room send
|
||||
→ 对象累加 + room network
|
||||
→ hybrid synthesis
|
||||
→ QMF synthesis
|
||||
→ 961-sample 延迟补偿
|
||||
→ stereo WAV
|
||||
SOFA FIR
|
||||
-> CanonicalHrtf
|
||||
-> 48 kHz / 单 radius shell / delay-phase policy
|
||||
-> 64-QMF / 77-hybrid projection
|
||||
-> 五阶 ACN/N3D 实球谐场
|
||||
-> 逐对象 direct + early reflections
|
||||
-> shared unitary-FDN late room
|
||||
-> stereo float64
|
||||
```
|
||||
|
||||
主要记号:
|
||||
## 输入接口
|
||||
|
||||
| 符号 | 含义 |
|
||||
双耳后端消费一个已编译的方向场(JOC compiled HRTF cache,`.jochrtf`)与共享滤波器组表;
|
||||
两者都是只读输入,本库不做任何测量数据格式的解析或转换:
|
||||
|
||||
```powershell
|
||||
joc_cli input.m4a --binaural `
|
||||
--compiled-hrtf-cache path\to\subject.jochrtf `
|
||||
--kernels data\rosella_kernels.npz `
|
||||
--binaural-mode mid --binaural-tail-seconds 5.0
|
||||
```
|
||||
|
||||
库接口使用同一组字段:文件任务用 `joc_task_config`,流式用 `joc_stream_config`,
|
||||
其中 `hrtf_path`、`kernels_path`、`binaural_mode`、`binaural_tail_seconds` 决定双耳通路。
|
||||
|
||||
`.jochrtf` 是**输入**而不是本库的产物:从 SOFA 或测量数据编译该缓存属于工具链的职责,
|
||||
与本仓库解耦。缓存加载时会校验成员集合、dtype 与形状、C 连续性、CRC-32 以及按成员重算的
|
||||
载荷哈希(见下文 `.jochrtf` 一节)。
|
||||
|
||||
`binaural_mode` 取 `near`、`mid`、`far`,选择后端的三组预置参数;`binaural_tail_seconds`
|
||||
决定 flush 时排空的房间尾音长度(默认 5.0 s)。
|
||||
|
||||
## 双耳渲染模式
|
||||
|
||||
`--binaural-mode off|near|mid|far`(默认 `mid`)是**人为指定的渲染提示**,不是
|
||||
从输入 E-AC-3 JOC 码流提取或还原的原始双耳元数据:
|
||||
|
||||
- 直接双耳渲染(`--binaural`):near/mid/far 生效,默认 `mid`;`off` 报错;
|
||||
- ADM BWF:DBMD segment 10 中后 15 个 JOC 对象的 binaural render mode 写
|
||||
`off=0/near=1/far=2/mid=3`,前 10 个 bed 保持不变,默认 `mid`;`off` 用于显式
|
||||
关闭双耳元数据提示。
|
||||
|
||||
## Canonical SOFA 契约
|
||||
|
||||
当前 strict importer 接受:
|
||||
|
||||
- `Conventions=SOFA`;
|
||||
- `SOFAConventions=SimpleFreeFieldHRIR`,version `0.4`、`1.0` 或 `1.1`;
|
||||
- `DataType=FIR`,`Data.IR[M,2,N]`;
|
||||
- 单一正有限 `Data.SamplingRate`,单位为 hertz/Hz;
|
||||
- spherical 或 Cartesian `SourcePosition`;
|
||||
- 单值或 per-measurement 的 `ListenerPosition/View/Up`;
|
||||
- 两个能由 listener-local lateral 坐标唯一识别左右的 receiver;
|
||||
- 单一且零偏移的 emitter;
|
||||
- causal `Data.Delay[I,2]` 或 `[M,2]`;
|
||||
- 明确的 free-field/anechoic `RoomType`。
|
||||
|
||||
receiver 左右顺序由几何决定,不能假定 `Data.IR` 的 receiver index。SOFA listener
|
||||
坐标为 $+X$ front、
|
||||
$+Y$ left、
|
||||
$+Z$ up;ADM 坐标为
|
||||
$+X$ right、
|
||||
$+Y$ front、
|
||||
$+Z$ up,转换为:
|
||||
|
||||
$$\bigl(x_{\mathrm{SOFA}},\ y_{\mathrm{SOFA}},\ z_{\mathrm{SOFA}}\bigr) = \bigl(y_{\mathrm{ADM}},\ -x_{\mathrm{ADM}},\ z_{\mathrm{ADM}}\bigr)$$
|
||||
|
||||
`CanonicalHrtf` 将 `Data.IR` 与 `Data.Delay` 分开保存。只有时域 baseline 才调用
|
||||
`materialized_measurement()` 将 delay 应用一次;运行时 SH 路径不先 materialize。
|
||||
非 48 kHz HRIR 使用 float64 `scipy.signal.resample_poly` 规范化,delay samples 按
|
||||
相同比例缩放。
|
||||
|
||||
GeneralFIR、BRIR、TF、多 emitter、多义 receiver 或非 free-field 数据需要单独的
|
||||
convention adapter,不能只通过 reshape 进入核心 importer。
|
||||
|
||||
## Delay/phase:exactly once
|
||||
|
||||
编译器只允许三种互斥语义:
|
||||
|
||||
1. 非零 `Data.Delay` 是 `Data.IR` 外部 delay;FIR 不去旋,运行时应用一次。
|
||||
2. `Data.Delay=0` 且 HRIR 有普通正 onset:以每耳 main peak 分离 arrival,拟合后
|
||||
在运行时恢复一次;当前阈值为 peak index 大于 2 samples。
|
||||
3. `Data.Delay=0` 且双耳 FIR 共享 sample-0 起点:不发明外部 delay,原 complex
|
||||
phase 直接进入五阶场。
|
||||
|
||||
任何路径都不能再叠加第二套 ear delay 或 phase-group delay。
|
||||
|
||||
## 公开 filterbank 与方向场
|
||||
|
||||
运行时固定为:
|
||||
|
||||
- 48 kHz;
|
||||
- 64-sample QMF hop;
|
||||
- 64-QMF / 77-hybrid;
|
||||
- analysis/synthesis latency 961 samples;
|
||||
- 五阶、36 项、ACN/N3D real spherical harmonics;
|
||||
- PCM、delay、SH、room state 为 float64;频带传递和频域状态为 complex128。
|
||||
|
||||
每个 hybrid band 的 real/imaginary 单位增益都通过同一套 analysis/synthesis 链生成
|
||||
脉冲字典,共 154 个实参数;编译不是直接读取 77 个 FFT bin。默认 projection
|
||||
ridge 为 `1e-3`,SH ridge 为 `1e-5`。同方向 measurement 先合并,再用球面 Voronoi
|
||||
面积权重做 ridge fit。
|
||||
|
||||
固定表位于 `data/rosella_kernels.npz`,实现公开标准化的滤波器组,各表可由如下
|
||||
公式计算。
|
||||
|
||||
hybrid 分析核定义于 [3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf)
|
||||
第 5.2.2 节(Table 1 的 $Q=8$/
|
||||
$Q=4$ 系数,delay 6):
|
||||
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr),\qquad n=0,\dots,12$$
|
||||
|
||||
QMF analysis 表即 MPEG-4 AAC/SBR(ISO/IEC 14496-3/AMD1:2003 第 4.B.18.2 节)
|
||||
的 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap prototype
|
||||
$c_0,\dots,c_{639}$ 的多相重排:
|
||||
|
||||
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t},\qquad r=0,\dots,63,\ t=0,\dots,9$$
|
||||
|
||||
QMF synthesis 表为上述 analysis 多相矩阵 $\mathbf{A}$ 的因果左逆,即求解
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$(
|
||||
$\mathbf{P}$ 为 577-sample 延迟置换;
|
||||
全链 $961 = 577 + 6\times64$),以 rank-4 分解形式存储:
|
||||
|
||||
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
|
||||
|
||||
hybrid synthesis 表为 77→64 重组:高频带恒等 $Y_{3+b}=X_{16+b}$;低频带(
|
||||
$C_p$ 为
|
||||
$8+4+4$ 子带划分):
|
||||
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\mathrm{Re}X_q + j\,s_q\,\mathrm{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
|
||||
loader 校验 archive 和每个数组的 SHA-256;table version、所有数组 hash 与
|
||||
77 个 band-center 参考值都属于 cache key。标准可公开获取不等于获准实施相关
|
||||
专利;更多来源信息见 [`data/README.md`](../data/README.md) 与
|
||||
[`THIRD_PARTY_NOTICES.md`](../THIRD_PARTY_NOTICES.md)。
|
||||
|
||||
## `.jochrtf`
|
||||
|
||||
`.jochrtf` 是无 pickle 的压缩 NumPy archive,固定包含:
|
||||
|
||||
| key | dtype / shape |
|
||||
|---|---|
|
||||
| $s=0\ldots15$ | 输入源;$s=0$ 为 LFE,$s=1\ldots15$ 为对象 |
|
||||
| $e\in\{L,R\}$ | 左右输出耳 |
|
||||
| $p=0\ldots63$ | QMF phase / 时域 hop 内采样 |
|
||||
| $k=0\ldots63$ | QMF 子带 |
|
||||
| $h=0\ldots76$ | hybrid 子带 |
|
||||
| $j=0\ldots35$ | 方向 basis 项 |
|
||||
| $m$ | 64-sample QMF 时槽 |
|
||||
| $n$ | 时域采样位置 |
|
||||
|
||||
每个 QMF hop 为 64 samples,每个双耳控制块为
|
||||
|
||||
$$
|
||||
N_b=512=8\times64,
|
||||
$$
|
||||
|
||||
每个 E-AC-3/JOC 音频帧为
|
||||
|
||||
$$
|
||||
N_f=1536=3N_b.
|
||||
$$
|
||||
|
||||
## 2. 输入与控制块
|
||||
|
||||
输入矩阵为
|
||||
|
||||
$$
|
||||
x_s[n],\qquad s=0\ldots15.
|
||||
$$
|
||||
|
||||
`pcm16` 的通道约定为:
|
||||
|
||||
```text
|
||||
ch0 special LFE
|
||||
ch1..15 JOC 对象 1..15
|
||||
```
|
||||
|
||||
渲染器按连续采样流推进。QMF、hybrid、room 和 synthesis 状态不会在 1536-sample 帧边界清零。
|
||||
|
||||
## 3. 64-band QMF analysis
|
||||
|
||||
令 $a_{p,\ell}$ 为固定的 64×10 polyphase 系数,$r_{s,\ell,p}[m]$ 为当前和前 9 个 hop 的 phase 历史。奇偶 lag 分别累加:
|
||||
|
||||
$$
|
||||
E_{s,p}[m]
|
||||
=\sum_{\substack{\ell=0\\\ell\text{ even}}}^{9}
|
||||
a_{p,\ell}r_{s,\ell,p}[m],
|
||||
$$
|
||||
|
||||
$$
|
||||
O_{s,p}[m]
|
||||
=\sum_{\substack{\ell=0\\\ell\text{ odd}}}^{9}
|
||||
a_{p,\ell}r_{s,\ell,p}[m].
|
||||
$$
|
||||
|
||||
对任一 64-vector $v[p]$,定义调制变换
|
||||
|
||||
$$
|
||||
\mathcal Q(v)_k
|
||||
=
|
||||
\operatorname{FFT}_{128}
|
||||
\left(
|
||||
\left[v[p]e^{-j\pi p/128}\right]_{p=0}^{63},
|
||||
0_{64}
|
||||
\right)_k
|
||||
|
||||
e^{-j3\pi(k+1/2)/128}.
|
||||
$$
|
||||
|
||||
analysis 输出为
|
||||
|
||||
$$
|
||||
X_{s,k}[m]
|
||||
=
|
||||
\mathcal Q(O_s)_k
|
||||
+j(-1)^k\mathcal Q(E_s)_k.
|
||||
$$
|
||||
|
||||
所有历史、乘加和 FFT 结果使用 `float64/complex128`。
|
||||
|
||||
## 4. 77-band hybrid analysis
|
||||
|
||||
低 3 个 QMF 子带使用 13-slot FIR 拆分为 16 个 hybrid bands。把复数的实部和虚部分量记为 $i,o\in\{0,1\}$,固定核为 $K_{p,i,\ell,h,o}$:
|
||||
|
||||
$$
|
||||
H_{s,h,o}[m]
|
||||
=
|
||||
\sum_{p=0}^{2}
|
||||
\sum_{i=0}^{1}
|
||||
\sum_{\ell=0}^{12}
|
||||
X_{s,p,i}[m-\ell]K_{p,i,\ell,h,o},
|
||||
\qquad h=0\ldots15.
|
||||
$$
|
||||
|
||||
其余 61 个 hybrid bands 是 QMF 3..63 的 6-slot 延迟:
|
||||
|
||||
$$
|
||||
H_{s,16+q}[m]=X_{s,3+q}[m-6],
|
||||
\qquad q=0\ldots60.
|
||||
$$
|
||||
|
||||
因此 hybrid vector 的顺序为:
|
||||
|
||||
```text
|
||||
0..15 低 3 个 QMF bands 的细分
|
||||
16..76 延迟后的 QMF bands 3..63
|
||||
```
|
||||
|
||||
## 5. OAMD 坐标与时间轴
|
||||
|
||||
### 5.1 Q15 坐标到 Cartesian
|
||||
|
||||
对象状态中的 $q_1,q_2,q_3$ 先恢复到离散位置网格:
|
||||
|
||||
$$
|
||||
u_1=\min\left(1,\frac{\operatorname{round}(62q_1/32767)}{62}\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
u_2=\min\left(1,\frac{\operatorname{round}(62q_2/32767)}{62}\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
u_3=\operatorname{clip}\left(
|
||||
\frac{\operatorname{round}(15q_3/32767)}{15},-1,1\right).
|
||||
$$
|
||||
|
||||
ADM Cartesian 坐标为
|
||||
|
||||
$$
|
||||
(X,Y,Z)=(2u_1-1,\ 1-2u_2,\ u_3).
|
||||
$$
|
||||
|
||||
### 5.2 更新时间
|
||||
|
||||
一条位置更新的编码时刻为
|
||||
|
||||
$$
|
||||
n_{\mathrm{coded}}
|
||||
=n_{\mathrm{frame}}
|
||||
+n_{\mathrm{outer}}
|
||||
+n_{\mathrm{block}}.
|
||||
$$
|
||||
|
||||
首个有效状态作为 sample 0 的初始位置。后续更新加入对象 PCM 延迟 $D_o$,默认
|
||||
|
||||
$$
|
||||
D_o=1473.
|
||||
$$
|
||||
|
||||
若 ramp duration 为 $R>64$,连续运动为
|
||||
|
||||
$$
|
||||
n_{\mathrm{start}}=n_{\mathrm{coded}}+D_o+64,
|
||||
$$
|
||||
|
||||
$$
|
||||
R_{\mathrm{eff}}=R-64,
|
||||
$$
|
||||
|
||||
$$
|
||||
\mathbf p[n]
|
||||
=(1-\alpha)\mathbf p_0+\alpha\mathbf p_1,
|
||||
\qquad
|
||||
\alpha=\frac{n-n_{\mathrm{start}}}{R_{\mathrm{eff}}}.
|
||||
$$
|
||||
|
||||
当 $R\le64$ 时,目标位置在 $n_{\mathrm{coded}}+D_o$ 直接生效。
|
||||
|
||||
Rosella 参数在每个 512-sample block 起点求值,并用于该块的 8 个 hybrid slots。
|
||||
|
||||
## 6. 距离 profile 与方向
|
||||
|
||||
普通对象只使用 Near、Mid、Far 三个 profile。每个 profile 包含:
|
||||
|
||||
- 三轴负/正边界 $b_{x-},b_{x+},b_{y-},b_{y+},b_{z-},b_{z+}$;
|
||||
- 距离尺度 $D$ 和倒数尺度 $D^{-1}$;
|
||||
- 三轴内部尺度 $a_x,a_y,a_z$;
|
||||
- 最小归一化半径 $\rho_{\min}$。
|
||||
|
||||
Cartesian 坐标经过 Q15 metadata grid 后换成内部前、侧、上轴,乘以 profile 尺度:
|
||||
|
||||
$$
|
||||
\mathbf s=(a_zq_f,\ a_xq_l,\ a_yq_v).
|
||||
$$
|
||||
|
||||
若射线超出 profile 边界,则用单一比例 $\lambda\le1$ 缩放:
|
||||
|
||||
$$
|
||||
\mathbf s' = \lambda\mathbf s.
|
||||
$$
|
||||
|
||||
随后
|
||||
|
||||
$$
|
||||
\rho=\|\mathbf s'\|_2,
|
||||
\qquad
|
||||
\rho_c=\max(\rho,\rho_{\min}),
|
||||
\qquad
|
||||
\alpha=\frac{\rho}{\rho_c},
|
||||
$$
|
||||
|
||||
$$
|
||||
\mathbf d=
|
||||
\begin{cases}
|
||||
\mathbf s'/\rho,&\rho>0,\\
|
||||
(1,0,0),&\rho=0,
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
物理半径为
|
||||
|
||||
$$
|
||||
R=D\rho.
|
||||
$$
|
||||
|
||||
## 7. 36 项方向 basis
|
||||
|
||||
方向 $\mathbf d=(x,y,z)$ 被展开为 36 项实值多项式:
|
||||
|
||||
$$
|
||||
\mathbf b(\mathbf d)=
|
||||
[1,x,y,z,x^2-\tfrac13,xy,xz,y^2-\tfrac13,yz,\ldots]^T.
|
||||
$$
|
||||
|
||||
完整顺序由 `rosella_model.direction_basis()` 固定。最高次数为 5;所有 field 系数必须按该顺序点积,不能交换 basis 项。
|
||||
|
||||
逐耳 basis 会根据耳偏移重新归一化。令耳偏移标量为 $e$:
|
||||
|
||||
$$
|
||||
\epsilon=\frac{eD^{-1}}{\rho_c},
|
||||
$$
|
||||
|
||||
$$
|
||||
\mathbf d_-=
|
||||
\frac{(x,y-\epsilon,z)}{\|(x,y-\epsilon,z)\|_2},
|
||||
\qquad
|
||||
\mathbf d_+=
|
||||
\frac{(x,y+\epsilon,z)}{\|(x,y+\epsilon,z)\|_2}.
|
||||
$$
|
||||
|
||||
## 8. 逐耳路径与 ITD
|
||||
|
||||
归一化路径长度为
|
||||
|
||||
$$
|
||||
\ell_-=\rho_c\sqrt{x^2+(y-\epsilon)^2+z^2},
|
||||
$$
|
||||
|
||||
$$
|
||||
\ell_+=\rho_c\sqrt{x^2+(y+\epsilon)^2+z^2}.
|
||||
$$
|
||||
|
||||
模型允许通过 36-vector 对路径加入非负方向修正:
|
||||
|
||||
$$
|
||||
\ell'_e
|
||||
=
|
||||
\ell_e
|
||||
+
|
||||
\max(\mathbf v_e^T\mathbf b_e,0)\,2cD^{-1}.
|
||||
$$
|
||||
|
||||
耳间延迟为
|
||||
|
||||
$$
|
||||
\tau
|
||||
=|\ell'_+-\ell'_-|\,D\frac{f_s}{343.3}\alpha,
|
||||
\qquad f_s=48000.
|
||||
$$
|
||||
|
||||
路径较长的一耳应用 hybrid-band 相位:
|
||||
|
||||
$$
|
||||
P_h=e^{j\omega_h\tau},
|
||||
$$
|
||||
|
||||
其中 $\omega_h$ 由模型的 20 个 hybrid group 参数递推到 77 个 bands。
|
||||
|
||||
## 9. 方向 field 与直达增益
|
||||
|
||||
左右耳各有一个 77×36 complex field:
|
||||
|
||||
$$
|
||||
C_{e,h}(\mathbf d_e)
|
||||
=
|
||||
\sum_{j=0}^{35}F_{e,h,j}b_j(\mathbf d_e).
|
||||
$$
|
||||
|
||||
另一次耳路径计算给出左右权重:
|
||||
|
||||
$$
|
||||
w_L=\frac{\ell_+}{\sqrt{\ell_-^2+\ell_+^2}},
|
||||
\qquad
|
||||
w_R=\frac{\ell_-}{\sqrt{\ell_-^2+\ell_+^2}}.
|
||||
$$
|
||||
|
||||
有效距离为
|
||||
|
||||
$$
|
||||
R_e=\rho\,s_dD,
|
||||
$$
|
||||
|
||||
其中 $s_d$ 为模型距离标量。Mid/Far 的公共衰减和 room send 为
|
||||
|
||||
$$
|
||||
g_c=\frac{1}{\sqrt{1+s_rR_e^2}},
|
||||
$$
|
||||
|
||||
$$
|
||||
g_{\mathrm{room}}=R_eg_c.
|
||||
$$
|
||||
|
||||
Near 使用
|
||||
|
||||
$$
|
||||
g_c=1,
|
||||
\qquad
|
||||
g_{\mathrm{room}}=0.
|
||||
$$
|
||||
|
||||
令 $C_{e,h}^{(0)}$ 为 field 的第 0 个 basis 系数,中心保护项为 $1-\alpha$。普通直达传递函数可写成
|
||||
|
||||
$$
|
||||
G_{L,h}
|
||||
=g_c\left[C_{L,h}w_L\alpha+C_{L,h}^{(0)}c_L(1-\alpha)\right],
|
||||
$$
|
||||
|
||||
$$
|
||||
G_{R,h}
|
||||
=g_c\left[C_{R,h}w_R\alpha+C_{R,h}^{(0)}c_R(1-\alpha)\right].
|
||||
$$
|
||||
|
||||
$c_L,c_R$ 由模型的耳权重配置选择;路径较长的一耳再乘 $P_h$。
|
||||
|
||||
## 10. LFE 传递函数
|
||||
|
||||
LFE 不进入普通对象方向计算。其传递函数为
|
||||
|
||||
$$
|
||||
G_{L,h}^{\mathrm{LFE}}=G_{R,h}^{\mathrm{LFE}}=
|
||||
\begin{cases}
|
||||
g_h,&0\le h<16,\\
|
||||
0,&16\le h<77.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
前 16 个固定系数为
|
||||
|
||||
```text
|
||||
2.60290003, 1.80741799, 0.659342408, -0.0275855921,
|
||||
-0.105803289, -0.0699509233, 0.0749056414, -0.00919809937,
|
||||
0.00349014648,-0.0158600751,-0.000723021978,0.00188189559,
|
||||
-0.000421735429,0.0000329252762,0.0000317397971,0.000000580376991
|
||||
```
|
||||
|
||||
LFE 的 room send 恒为 0。
|
||||
|
||||
## 11. 对象累加与 room input
|
||||
|
||||
每个 hybrid slot 的直达输出为
|
||||
|
||||
$$
|
||||
Y^{\mathrm{direct}}_{e,h}
|
||||
=
|
||||
\sum_{s=0}^{15}H_{s,h}G_{s,e,h}.
|
||||
$$
|
||||
|
||||
room 输入为
|
||||
|
||||
$$
|
||||
U_h
|
||||
=
|
||||
\sum_{s=1}^{15}H_{s,h}g_{\mathrm{room},s}.
|
||||
$$
|
||||
|
||||
LFE 不进入该和式。
|
||||
|
||||
## 12. Room network
|
||||
|
||||
room 只处理前 64 个 hybrid bands。输入先乘
|
||||
|
||||
$$
|
||||
g_0=0.70710677.
|
||||
$$
|
||||
|
||||
对每级 all-pass,设延迟样本为 $d[n]$、系数为 $a$:
|
||||
|
||||
$$
|
||||
r[n]=x[n]-ad[n],
|
||||
$$
|
||||
|
||||
$$
|
||||
y[n]=ar[n]+d[n].
|
||||
$$
|
||||
|
||||
all-pass 输出复制到 4 个 FDN branches。设延迟输出为 $\mathbf d_h[m]$、4×4 混合矩阵为 $M$:
|
||||
|
||||
$$
|
||||
\mathbf b_h[m]
|
||||
=U_h[m]\mathbf 1+M\mathbf d_h[m].
|
||||
$$
|
||||
|
||||
每个 branch 使用复反馈系数 $f_{h,i}$:
|
||||
|
||||
$$
|
||||
m_{h,i}[m]=f_{h,i}b_{h,i}[m].
|
||||
$$
|
||||
|
||||
主 tap、可选额外 tap 和左右输出矩阵合成为
|
||||
|
||||
$$
|
||||
Y^{\mathrm{room}}_{e,h}[m]
|
||||
=
|
||||
\sum_{i=0}^{3}O_{e,h,i}z_{h,i}[m].
|
||||
$$
|
||||
|
||||
最终 hybrid 输出为
|
||||
|
||||
$$
|
||||
Y_{e,h}=Y^{\mathrm{direct}}_{e,h}+Y^{\mathrm{room}}_{e,h}.
|
||||
$$
|
||||
|
||||
Python 后端把该递归网络展开为有限 complex FIR 并使用 overlap-add;C++ 后端直接保持递归状态。两者均跨帧连续。
|
||||
|
||||
## 13. Hybrid synthesis
|
||||
|
||||
hybrid synthesis 是 154 项稀疏即时映射。令映射项为 $(h,i,k,o,w)$,其中 $i,o$ 表示实部或虚部,则
|
||||
|
||||
$$
|
||||
Q_{e,k,o}[m]
|
||||
\mathrel{+}=
|
||||
Y_{e,h,i}[m]w.
|
||||
$$
|
||||
|
||||
输出是每耳 64 个 complex QMF bands。
|
||||
|
||||
## 14. QMF synthesis
|
||||
|
||||
每耳 QMF vector 先展开为 128 项实向量
|
||||
|
||||
$$
|
||||
\mathbf q_e=[\Re Q_{e,0},\Im Q_{e,0},\ldots,\Re Q_{e,63},\Im Q_{e,63}]^T.
|
||||
$$
|
||||
|
||||
对每个 phase $p$ 和 rank $r=0\ldots3$:
|
||||
|
||||
$$
|
||||
f_{e,p,r}[m]
|
||||
=\mathbf b_{p,r}^T\mathbf q_e[m].
|
||||
$$
|
||||
|
||||
使用 10-slot taps 合成时域样本:
|
||||
|
||||
$$
|
||||
y_e[64m+p]
|
||||
=
|
||||
\sum_{\ell=0}^{9}
|
||||
\sum_{r=0}^{3}
|
||||
t_{p,\ell,r}f_{e,p,r}[m-\ell].
|
||||
$$
|
||||
|
||||
## 15. 延迟、尾声和输出
|
||||
|
||||
完整 filterbank 的固定延迟为
|
||||
|
||||
$$
|
||||
L=961\ \text{samples}.
|
||||
$$
|
||||
|
||||
只在连续流起点丢弃一次前 $L$ 个输出 samples。输入结束后继续送零,以释放 QMF、hybrid 和 room 状态。尾声裁切只作用于文件末端:
|
||||
|
||||
$$
|
||||
\max(|y_L[n]|,|y_R[n]|) > 10^{-8}
|
||||
$$
|
||||
|
||||
的最后一个 sample 被保留,同时输出长度不得短于源 PCM 长度。
|
||||
|
||||
所有内部状态、参数计算和对象累加使用 `float64/complex128`。最终 writer 才转换为 float32 或 PCM24。
|
||||
|
||||
## 16. Python 与 C++ 后端
|
||||
|
||||
两套后端共享:
|
||||
|
||||
- 同一份 QMF/hybrid 固定表;
|
||||
- 同一份模型解析结果;
|
||||
- 同一套 512-sample 参数更新时间轴;
|
||||
- 同一组逐对象 complex gains 和 room sends;
|
||||
- 同一 961-sample 延迟补偿与尾声策略。
|
||||
|
||||
C++ 后端以 512-sample block 为处理单位,内部持有 QMF、hybrid、room 和 synthesis 状态。Python 只负责模型解析、OAMD 时间轴和每块参数更新。
|
||||
|
||||
## 17. 模型文件
|
||||
|
||||
默认路径为
|
||||
|
||||
```text
|
||||
HRTF/binaural.personalized_headphone
|
||||
```
|
||||
|
||||
也可通过
|
||||
|
||||
```text
|
||||
--personalized-headphone PATH
|
||||
```
|
||||
|
||||
指定其它文件。
|
||||
|
||||
`.personalized_headphone` 中的 int32/Q15 参数在解析后提升为 float64。任意 SOFA FIR 不能只通过数组重排变成该参数模型;若要转换,需要拟合方向 fields、ITD、距离 profile、耳几何和 room 参数。
|
||||
|
||||
## 18. 适用范围
|
||||
|
||||
当前路径处理 15 个普通点对象和 1 路 special LFE。对象 extent、spread、diffuse、divergence、channel lock,以及未实现的 OAMD element 变体不在本公式范围内。
|
||||
| `metadata_json` | 含 JSON 文本的 NumPy Unicode scalar(`dtype.kind == "U"`) |
|
||||
| `band_center_frequencies_hz` | little-endian `float64[77]` |
|
||||
| `coefficients` | little-endian `complex128[36,2,77]` |
|
||||
| `delay_coefficients` | little-endian `float64[36,2]` |
|
||||
| `delay_bounds` | little-endian `float64[2,2]` |
|
||||
|
||||
metadata magic 固定为 `JOC-HRTF-CACHE`,并记录 schema/compiler/phase-policy、
|
||||
ACN/N3D、filterbank table hashes、SOFA content SHA-256、采样率、radius、order、
|
||||
两个 ridge、payload hash 和 fit report。cache key 覆盖所有会改变编译结果的字段。
|
||||
metadata 不保存本机绝对 `source_path`,仅可保存 source display name。
|
||||
|
||||
loader 使用 `allow_pickle=False`,并在构造对象前检查 ZIP 成员集、解压大小、shape、
|
||||
dtype、端序、连续布局、有限值、delay bounds、band centers、payload hash 和 cache
|
||||
key。writer 使用同目录临时文件、`fsync`、进程持有的 OS 文件锁和原子
|
||||
`os.replace`;对应的隐藏 `.lock` sidecar 可保留,但不代表仍有 writer 持锁。
|
||||
旧版本、损坏或配置不匹配的 cache 不能命中;从 SOFA 启动时会重建,显式 cache
|
||||
入口则直接报错。
|
||||
|
||||
删除磁盘 cache 后,从同一 SOFA 和同一编译配置得到的场与渲染结果不得改变。
|
||||
|
||||
`.jochrtf` 包含由源 HRIR 变换得到的方向场系数与 delay 数据,因此“可以重建”不表示
|
||||
它不受数据许可约束。生成 cache 不会扩大源 SOFA/HRTF 数据集授予的权利;cache 的
|
||||
使用、复制和再分发仍须遵守源数据集条款。不能确认条款时,应把 `.jochrtf` 作为本地
|
||||
私有 cache,不随程序或构建产物发布。`source_sha256` 只用于内容一致性校验,不是许可
|
||||
或来源证明。
|
||||
|
||||
## JOC 对象与房间
|
||||
|
||||
生产适配器继续使用现有 JOC 调度:
|
||||
|
||||
- 每帧输入 `[1536,16]`;
|
||||
- channel 0 是 special LFE,channel 1..15 是 JOC objects;
|
||||
- ID11/OAMD position 使用 sample-timed timeline;
|
||||
- 每 512 samples 更新方向/profile;
|
||||
- 每个对象拥有独立 direct/early history,late FDN 全局共享;
|
||||
- `finish()` 排空 early/late tail;输出增益显式应用,不隐含 limiter 或节目响度归一化。
|
||||
|
||||
Near/Mid/Far、equal-power direct level、六面 shoebox 一阶 image source、late send、
|
||||
unitary FDN、LFE 120–180 Hz cosine-squared 低通及 room calibration 都是 JOC
|
||||
项目定义行为,不是 SOFA 或 Dolby 公布常数。
|
||||
|
||||
双耳渲染在本库内实现:filterbank、SH 方向场求值、逐对象 early/direct 历史与共享 FDN
|
||||
全部执行于 `joc_core`(`ejoc_sofa_binaural_*` 内核),对外只消费编译好的方向场与
|
||||
512-sample 粒度的元数据更新。
|
||||
|
||||
## 技术引用与权利边界
|
||||
|
||||
- [SOFA SimpleFreeFieldHRIR convention](https://www.sofaconventions.org/mediawiki/index.php/SimpleFreeFieldHRIR)
|
||||
- [3GPP TS 26.405 / ETSI TS 126 405(64-QMF/77-hybrid 定义)](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf)
|
||||
- [Dolby binaural render mode workflow](https://professionalsupport.dolby.com/s/article/What-is-Binaural-Render-Mode-and-how-do-the-settings-affect-my-mix)
|
||||
- [EP3090576A1](https://patents.google.com/patent/EP3090576A1/en),仅作 direct/early/late、
|
||||
subband 与 FDN 架构背景,不证明某个产品使用特定实施例。
|
||||
|
||||
规范、源码或专利文献可公开获取,不等于获准复制其内容、再分发派生产物或实施其中的
|
||||
专利权利要求。本项目的技术引用本身不授予专利许可,也不作不侵权保证;准备发布或集成
|
||||
到产品的一方应自行审查适用的数据许可、软件许可、专利许可及 freedom-to-operate。
|
||||
公开标准来源与权利边界见
|
||||
[`THIRD_PARTY_NOTICES.md`](../THIRD_PARTY_NOTICES.md)。
|
||||
|
||||
+97
-78
@@ -4,7 +4,7 @@
|
||||
|
||||
This document covers only the signal model and formulas used in the JustOneCacophony research path: how JOC parameters combine with core PCM to reconstruct object signals, and how OAMD coordinates become speaker gains.
|
||||
|
||||
The formulas describe the dense-JOC and ordinary point-object paths studied by the project. They are not a complete definition of every E-AC-3 JOC variant.
|
||||
The formulas describe the JOC matrix parameters (both the dense and the sparse differential syntax) and the ordinary point-object paths studied by the project. They are not a complete definition of every E-AC-3 JOC variant.
|
||||
|
||||
## 1. Overall path and notation
|
||||
|
||||
@@ -52,9 +52,11 @@ $$
|
||||
N_f=1536=24\times64.
|
||||
$$
|
||||
|
||||
## 2. Dense-JOC matrix parameters
|
||||
## 2. JOC matrix parameters
|
||||
|
||||
### 2.1 Differential reconstruction
|
||||
For every object and data point, the quantized matrix `joc_mix_mtx_q` is defined on $N_q$ quantization levels. The `b_joc_sparse` flag selects one of two differential syntaxes: dense sends one MTX difference per core channel, while sparse sends one active channel plus one coefficient difference per parameter band.
|
||||
|
||||
### 2.1 Dense differential reconstruction
|
||||
|
||||
Let `quant_idx` be $q_i\in\{0,1\}$. The number of quantization levels is
|
||||
|
||||
@@ -75,38 +77,76 @@ $$
|
||||
For object $o$, data point $d$, core channel $c$, and parameter band $p$, the coded difference $\Delta_{o,d,c,p}$ reconstructs to
|
||||
|
||||
$$
|
||||
Q_{o,d,c,0}
|
||||
=
|
||||
Q_{o,d,c,0}=
|
||||
\left(O_q+\Delta_{o,d,c,0}\right)\bmod N_q,
|
||||
$$
|
||||
|
||||
$$
|
||||
Q_{o,d,c,p}
|
||||
=
|
||||
Q_{o,d,c,p}=
|
||||
\left(Q_{o,d,c,p-1}+\Delta_{o,d,c,p}\right)\bmod N_q,
|
||||
\qquad p>0.
|
||||
$$
|
||||
|
||||
### 2.2 Dequantization
|
||||
### 2.2 Sparse differential reconstruction
|
||||
|
||||
Let $I_{o,d,p}$ be the `joc_channel_idx` symbol (IDX), $V_{o,d,p}$ the `joc_vec` symbol (VEC), and $N_c\in\lbrace5,7\rbrace$ the number of core channels. Each parameter band has exactly one active channel:
|
||||
|
||||
$$
|
||||
A_{o,d,p}=
|
||||
\begin{cases}
|
||||
I_{o,d,0}, & p=0,\\
|
||||
\left(A_{o,d,p-1}+I_{o,d,p}\right)\bmod N_c, & p>0,
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
where $I_{o,d,0}$ is a 3-bit absolute channel index and every later IDX symbol is an increment relative to the previous **active channel**. The coefficient is a single accumulator running across parameter bands:
|
||||
|
||||
$$
|
||||
\kappa_{o,d,-1}=O^{(s)}_q,\qquad
|
||||
\kappa_{o,d,p}=
|
||||
\left(\kappa_{o,d,p-1}+V_{o,d,p}\right)\bmod N_q,
|
||||
$$
|
||||
|
||||
with a sparse starting point two quantization levels above the dense center offset:
|
||||
|
||||
$$
|
||||
O^{(s)}_q=
|
||||
\begin{cases}
|
||||
50, & q_i=0,\\
|
||||
100, & q_i=1.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
The accumulator is **not** reset when the active channel changes. The complete matrix is
|
||||
|
||||
$$
|
||||
Q_{o,d,c,p}=
|
||||
\begin{cases}
|
||||
\kappa_{o,d,p}, & c=A_{o,d,p},\\
|
||||
\dfrac{N_q}{2}, & c\neq A_{o,d,p}.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
Non-active entries take $N_q/2$, which dequantizes to exactly 0.
|
||||
|
||||
### 2.3 Dequantization
|
||||
|
||||
The dequantized matrix coefficient is
|
||||
|
||||
$$
|
||||
D_{o,d,c,p}
|
||||
=
|
||||
D_{o,d,c,p}=
|
||||
\left(Q_{o,d,c,p}-\frac{N_q}{2}\right)
|
||||
\frac{820}{4096(1+q_i)}.
|
||||
$$
|
||||
|
||||
The effective denominator is therefore 4096 in coarse mode and 8192 in fine mode.
|
||||
|
||||
### 2.3 JOC clipgain
|
||||
### 2.4 JOC clipgain
|
||||
|
||||
If the clipgain field consists of integer $x$ and mantissa $y$, then
|
||||
|
||||
$$
|
||||
G_{\mathrm{clip}}
|
||||
=
|
||||
G_{\mathrm{clip}}=
|
||||
1+\frac{y}{32}2^{x-4}.
|
||||
$$
|
||||
|
||||
@@ -152,8 +192,7 @@ $$
|
||||
$$
|
||||
|
||||
$$
|
||||
M_{o,c,b,t}
|
||||
=
|
||||
M_{o,c,b,t}=
|
||||
(1-\alpha_t)P_{o,c,b}
|
||||
+\alpha_tD_{o,c,p(b)}.
|
||||
$$
|
||||
@@ -181,9 +220,8 @@ $$
|
||||
Let $\mathcal A_b$ denote the 64-band analysis-QMF operator with polyphase history state. Then
|
||||
|
||||
$$
|
||||
X_{c,b,t}
|
||||
=
|
||||
\mathcal A_b\!\left(
|
||||
X_{c,b,t}=
|
||||
\mathcal A_b\left(
|
||||
\widetilde x_c[64t],\ldots,\widetilde x_c[64t+63];
|
||||
\mathbf s^{\mathrm A}_{c,t}
|
||||
\right).
|
||||
@@ -210,8 +248,7 @@ $$
|
||||
Band 0 of each surround channel additionally passes through a 21-tap complex FIR:
|
||||
|
||||
$$
|
||||
\widehat X_{c,0,t}
|
||||
=
|
||||
\widehat X_{c,0,t}=
|
||||
\sum_{k=0}^{20}h_kX_{c,0,t-k}.
|
||||
$$
|
||||
|
||||
@@ -222,8 +259,7 @@ These delays and filter histories are decoder state and cannot be reset independ
|
||||
For each object $o$, subband $b$, and slot $t$, the object's frequency-domain value is a linear combination of the five core channels:
|
||||
|
||||
$$
|
||||
Z_{o,b,t}
|
||||
=
|
||||
Z_{o,b,t}=
|
||||
\sum_{c=0}^{4}
|
||||
M_{o,c,b,t}\widehat X_{c,b,t}.
|
||||
$$
|
||||
@@ -238,21 +274,20 @@ Write the 64 complex subbands as 128 interleaved real values in `src`. For $k=0\
|
||||
|
||||
$$
|
||||
\begin{aligned}
|
||||
\operatorname{zone}[2k] &= \operatorname{src}[4k],\\
|
||||
\operatorname{zone}[2k+1] &= -\operatorname{src}[4k+1],\\
|
||||
\operatorname{zone}[126-2k] &= \operatorname{src}[4k+2],\\
|
||||
\operatorname{zone}[127-2k] &= \operatorname{src}[4k+3].
|
||||
\mathrm{zone}[2k] &= \mathrm{src}[4k],\\
|
||||
\mathrm{zone}[2k+1] &= -\mathrm{src}[4k+1],\\
|
||||
\mathrm{zone}[126-2k] &= \mathrm{src}[4k+2],\\
|
||||
\mathrm{zone}[127-2k] &= \mathrm{src}[4k+3].
|
||||
\end{aligned}
|
||||
$$
|
||||
|
||||
Treat `zone` as 64 complex values and apply an unnormalized 64-point FFT:
|
||||
|
||||
$$
|
||||
F_k
|
||||
=
|
||||
F_k=
|
||||
\sum_{n=0}^{63}
|
||||
\operatorname{zone}_n
|
||||
\exp\!\left(-j\frac{2\pi kn}{64}\right).
|
||||
\mathrm{zone}_n
|
||||
\exp\left(-j\frac{2\pi kn}{64}\right).
|
||||
$$
|
||||
|
||||
### 7.2 Modulation and synthesis
|
||||
@@ -260,8 +295,7 @@ $$
|
||||
Define the rotation coefficient
|
||||
|
||||
$$
|
||||
r_k
|
||||
=
|
||||
r_k=
|
||||
\frac12\left(
|
||||
\sin\frac{\pi k}{128}
|
||||
+j\cos\frac{\pi k}{128}
|
||||
@@ -277,9 +311,8 @@ $$
|
||||
Let $\mathcal S$ denote polyphase synthesis with a 640-value synthesis window and cross-slot state:
|
||||
|
||||
$$
|
||||
\mathbf y_{o,t}
|
||||
=
|
||||
\mathcal S\!\left(
|
||||
\mathbf y_{o,t}=
|
||||
\mathcal S\left(
|
||||
\mathbf R_{o,t},W,\mathbf s^{\mathrm S}_{o,t}
|
||||
\right).
|
||||
$$
|
||||
@@ -287,9 +320,8 @@ $$
|
||||
Object output is
|
||||
|
||||
$$
|
||||
y_o[64t+r]
|
||||
=
|
||||
\operatorname{clip}\!\left(
|
||||
y_o[64t+r]=
|
||||
\mathrm{clip}\left(
|
||||
16\,\mathbf y_{o,t}[r],-1,1
|
||||
\right)G_{\mathrm{clip}},
|
||||
$$
|
||||
@@ -301,9 +333,8 @@ where $r=0\ldots63$. Synthesis state must advance continuously by slot.
|
||||
LFE bypasses the object matrix and inverse QMF and uses a 1217-sample delay. After the input and output scale factors cancel:
|
||||
|
||||
$$
|
||||
y_{\mathrm{LFE}}[n]
|
||||
=
|
||||
\operatorname{clip}\!\left(
|
||||
y_{\mathrm{LFE}}[n]=
|
||||
\mathrm{clip}\left(
|
||||
x_{\mathrm{LFE,core}}[n-1217],-1,1
|
||||
\right).
|
||||
$$
|
||||
@@ -313,9 +344,8 @@ $$
|
||||
The lateral and longitudinal grids use $N=62$; the height grid uses $N=15$. The quantizer is
|
||||
|
||||
$$
|
||||
q_N(k)
|
||||
=
|
||||
\min\!\left(
|
||||
q_N(k)=
|
||||
\min\left(
|
||||
32767,
|
||||
\left\lfloor\frac{32768k}{N}+\frac12\right\rfloor
|
||||
\right).
|
||||
@@ -336,11 +366,11 @@ Their maximum runtime value is $32767/32768$, not exactly 1.
|
||||
For conversion to the ADM grid:
|
||||
|
||||
$$
|
||||
k_1=\operatorname{round}\!\left(\frac{62q_1}{32767}\right),
|
||||
k_1=\mathrm{round}\left(\frac{62q_1}{32767}\right),
|
||||
\quad
|
||||
k_2=\operatorname{round}\!\left(\frac{62q_2}{32767}\right),
|
||||
k_2=\mathrm{round}\left(\frac{62q_2}{32767}\right),
|
||||
\quad
|
||||
k_3=\operatorname{round}\!\left(\frac{15q_3}{32767}\right),
|
||||
k_3=\mathrm{round}\left(\frac{15q_3}{32767}\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
@@ -404,17 +434,15 @@ $$
|
||||
The two-dimensional point gain is
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{2D}}(u,v)
|
||||
=
|
||||
\mathbf G_{\mathrm{2D}}(u,v)=
|
||||
\mathbf h(u)\odot\mathbf v(v).
|
||||
$$
|
||||
|
||||
For 5.1-family layouts with one horizontal surround pair rather than separate side and rear pairs, the longitudinal coordinate is
|
||||
|
||||
$$
|
||||
v_{\mathrm{floor}}
|
||||
=
|
||||
\operatorname{clamp}(2v,0,1).
|
||||
v_{\mathrm{floor}}=
|
||||
\mathrm{clamp}(2v,0,1).
|
||||
$$
|
||||
|
||||
Other layouts use $v_{\mathrm{floor}}=v$.
|
||||
@@ -424,8 +452,7 @@ Other layouts use $v_{\mathrm{floor}}=v$.
|
||||
Three-dimensional layouts compute floor gain $\mathbf G_f$ and height gain $\mathbf G_h$ separately:
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{point}}(u,v,w)
|
||||
=
|
||||
\mathbf G_{\mathrm{point}}(u,v,w)=
|
||||
\cos\left(\frac\pi2w\right)\mathbf G_f
|
||||
+
|
||||
\sin\left(\frac\pi2w\right)\mathbf G_h.
|
||||
@@ -450,8 +477,7 @@ $$
|
||||
Maximum position compensation is
|
||||
|
||||
$$
|
||||
A_{\max}
|
||||
=
|
||||
A_{\max}=
|
||||
-\max\left(4.5-1.5H-3F,0\right)
|
||||
\quad\text{dB}.
|
||||
$$
|
||||
@@ -459,15 +485,15 @@ $$
|
||||
Longitudinal and height weights are
|
||||
|
||||
$$
|
||||
p_v=\operatorname{clamp}\left(\frac v{0.6},0,1\right),
|
||||
p_v=\mathrm{clamp}\left(\frac v{0.6},0,1\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
p_w=\operatorname{clamp}\left(\frac{w-0.2}{0.8},0,1\right),
|
||||
p_w=\mathrm{clamp}\left(\frac{w-0.2}{0.8},0,1\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
p=\operatorname{clamp}(p_v+p_w,0,1).
|
||||
p=\mathrm{clamp}(p_v+p_w,0,1).
|
||||
$$
|
||||
|
||||
The linear compensation gain is
|
||||
@@ -479,8 +505,7 @@ $$
|
||||
The object's target-gain vector is
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{target}}
|
||||
=
|
||||
\mathbf G_{\mathrm{target}}=
|
||||
G_{\mathrm{object}}
|
||||
G_{\mathrm{pos}}
|
||||
\mathbf G_{\mathrm{point}}.
|
||||
@@ -491,8 +516,7 @@ $$
|
||||
The coded position of an OAMD update is
|
||||
|
||||
$$
|
||||
s_{\mathrm{coded}}
|
||||
=
|
||||
s_{\mathrm{coded}}=
|
||||
s_{\mathrm{frame}}
|
||||
+s_{\mathrm{outer}}
|
||||
+s_{\mathrm{OAMD}}
|
||||
@@ -502,16 +526,15 @@ $$
|
||||
The theoretical update position on the decoder-output PCM timeline is
|
||||
|
||||
$$
|
||||
s_{\mathrm{theoretical}}
|
||||
=s_{\mathrm{coded}}+d_{\mathrm{decoder}},
|
||||
s_{\mathrm{theoretical}}=
|
||||
s_{\mathrm{coded}}+d_{\mathrm{decoder}},
|
||||
\qquad d_{\mathrm{decoder}}=1473.
|
||||
$$
|
||||
|
||||
The speaker renderer retains the existing processing-block length $B=32$, so the aligned update point is
|
||||
|
||||
$$
|
||||
\widehat s
|
||||
=
|
||||
\widehat s=
|
||||
B\left\lfloor
|
||||
\frac{s_{\mathrm{theoretical}}+B/2-1}{B}
|
||||
\right\rfloor.
|
||||
@@ -522,8 +545,7 @@ Thus, for frame-aligned updates, `align32(1473)=1472`. The 1473 value is the the
|
||||
For ramp duration $D$, the number of blocks is
|
||||
|
||||
$$
|
||||
K
|
||||
=
|
||||
K=
|
||||
\left\lfloor
|
||||
\frac{D+B/2-1}{B}
|
||||
\right\rfloor.
|
||||
@@ -550,8 +572,7 @@ If no new metadata update intervenes, this is equivalent to a sample-wise linear
|
||||
For target output channel $c$:
|
||||
|
||||
$$
|
||||
y_c[n]
|
||||
=
|
||||
y_c[n]=
|
||||
\delta_{c,\mathrm{LFE}}x_{\mathrm{LFE}}[n]
|
||||
+
|
||||
\sum_{o=1}^{15}x_o[n]g_{o,c}[n].
|
||||
@@ -560,8 +581,7 @@ $$
|
||||
Here
|
||||
|
||||
$$
|
||||
\delta_{c,\mathrm{LFE}}
|
||||
=
|
||||
\delta_{c,\mathrm{LFE}}=
|
||||
\begin{cases}
|
||||
1, & c\text{ is the target layout's LFE channel},\\
|
||||
0, & \text{otherwise}.
|
||||
@@ -573,16 +593,15 @@ A layout without LFE output does not mix input LFE into other channels. After ob
|
||||
For PCM24 output, quantization is
|
||||
|
||||
$$
|
||||
y_{24}[n]
|
||||
=
|
||||
\operatorname{trunc}\left(
|
||||
8388607\,\operatorname{clip}(y[n],-1,1)
|
||||
y_{24}[n]=
|
||||
\mathrm{trunc}\left(
|
||||
8388607\,\mathrm{clip}(y[n],-1,1)
|
||||
\right).
|
||||
$$
|
||||
|
||||
## 14. Scope of the formulas
|
||||
|
||||
- The JOC matrix section describes dense JOC; Sparse JOC uses a different sparse coefficient/index path.
|
||||
- The JOC matrix section covers both the dense MTX and the sparse IDX/VEC differential syntax.
|
||||
- The speaker-panning section describes ordinary point objects; extent, spread, divergence, and similar modes require additional models.
|
||||
- Multiple OAMD position blocks must be scheduled in time order.
|
||||
- A limiter is separate post-processing and is not included in the mixing equations above.
|
||||
|
||||
+97
-78
@@ -4,7 +4,7 @@
|
||||
|
||||
本文只说明 JustOneCacophony 研究路径中使用的信号模型和公式:JOC 参数如何与核心 PCM 结合并重建对象信号,以及 OAMD 坐标如何转换为扬声器增益。
|
||||
|
||||
这些公式描述项目当前研究的 dense JOC 与普通点对象路径,不代表对所有 E-AC-3 JOC 变体的完整定义。
|
||||
这些公式描述项目当前研究的 JOC 矩阵参数(dense 与 sparse 两条差分语法)与普通点对象路径,不代表对所有 E-AC-3 JOC 变体的完整定义。
|
||||
|
||||
## 1. 总体路径与记号
|
||||
|
||||
@@ -52,9 +52,11 @@ $$
|
||||
N_f=1536=24\times64.
|
||||
$$
|
||||
|
||||
## 2. Dense JOC 矩阵参数
|
||||
## 2. JOC 矩阵参数
|
||||
|
||||
### 2.1 差分还原
|
||||
每个对象、每个数据点的量化矩阵 `joc_mix_mtx_q` 都定义在 $N_q$ 个量化级上。标志位 `b_joc_sparse` 选择两条差分语法之一:dense 为每个核心声道各送一路 MTX 差分,sparse 每参数带只送一个 active 声道与一路系数差分。
|
||||
|
||||
### 2.1 Dense 差分还原
|
||||
|
||||
令 `quant_idx` 为 $q_i\in\{0,1\}$,量化级数为
|
||||
|
||||
@@ -75,38 +77,76 @@ $$
|
||||
对对象 $o$、数据点 $d$、核心声道 $c$ 和参数带 $p$,编码差分 $\Delta_{o,d,c,p}$ 还原为
|
||||
|
||||
$$
|
||||
Q_{o,d,c,0}
|
||||
=
|
||||
Q_{o,d,c,0}=
|
||||
\left(O_q+\Delta_{o,d,c,0}\right)\bmod N_q,
|
||||
$$
|
||||
|
||||
$$
|
||||
Q_{o,d,c,p}
|
||||
=
|
||||
Q_{o,d,c,p}=
|
||||
\left(Q_{o,d,c,p-1}+\Delta_{o,d,c,p}\right)\bmod N_q,
|
||||
\qquad p>0.
|
||||
$$
|
||||
|
||||
### 2.2 去量化
|
||||
### 2.2 Sparse 差分还原
|
||||
|
||||
令 $I_{o,d,p}$ 为 `joc_channel_idx` 符号(IDX), $V_{o,d,p}$ 为 `joc_vec` 符号(VEC), $N_c\in\lbrace5,7\rbrace$ 为核心声道数。每参数带只有一个 active 声道
|
||||
|
||||
$$
|
||||
A_{o,d,p}=
|
||||
\begin{cases}
|
||||
I_{o,d,0}, & p=0,\\
|
||||
\left(A_{o,d,p-1}+I_{o,d,p}\right)\bmod N_c, & p>0,
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
其中 $I_{o,d,0}$ 是 3 bit 绝对声道号,其余 IDX 符号是相对上一个 **active 声道**的增量。系数是一个跨参数带连续的单累加器
|
||||
|
||||
$$
|
||||
\kappa_{o,d,-1}=O^{(s)}_q,\qquad
|
||||
\kappa_{o,d,p}=
|
||||
\left(\kappa_{o,d,p-1}+V_{o,d,p}\right)\bmod N_q,
|
||||
$$
|
||||
|
||||
sparse 起点比 dense 的中心偏移高两个量化级:
|
||||
|
||||
$$
|
||||
O^{(s)}_q=
|
||||
\begin{cases}
|
||||
50, & q_i=0,\\
|
||||
100, & q_i=1.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
active 声道切换时累加器**不**重置。完整矩阵为
|
||||
|
||||
$$
|
||||
Q_{o,d,c,p}=
|
||||
\begin{cases}
|
||||
\kappa_{o,d,p}, & c=A_{o,d,p},\\
|
||||
\dfrac{N_q}{2}, & c\neq A_{o,d,p}.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
非 active 项取 $N_q/2$,即去量化后恰为 0。
|
||||
|
||||
### 2.3 去量化
|
||||
|
||||
矩阵系数的去量化值为
|
||||
|
||||
$$
|
||||
D_{o,d,c,p}
|
||||
=
|
||||
D_{o,d,c,p}=
|
||||
\left(Q_{o,d,c,p}-\frac{N_q}{2}\right)
|
||||
\frac{820}{4096(1+q_i)}.
|
||||
$$
|
||||
|
||||
因此 coarse 模式的有效分母为 4096,fine 模式为 8192。
|
||||
|
||||
### 2.3 JOC clipgain
|
||||
### 2.4 JOC clipgain
|
||||
|
||||
若 clipgain 字段由整数 $x$ 和尾数 $y$ 组成,则
|
||||
|
||||
$$
|
||||
G_{\mathrm{clip}}
|
||||
=
|
||||
G_{\mathrm{clip}}=
|
||||
1+\frac{y}{32}2^{x-4}.
|
||||
$$
|
||||
|
||||
@@ -152,8 +192,7 @@ $$
|
||||
$$
|
||||
|
||||
$$
|
||||
M_{o,c,b,t}
|
||||
=
|
||||
M_{o,c,b,t}=
|
||||
(1-\alpha_t)P_{o,c,b}
|
||||
+\alpha_tD_{o,c,p(b)}.
|
||||
$$
|
||||
@@ -181,9 +220,8 @@ $$
|
||||
令 $\mathcal A_b$ 表示带 polyphase 历史状态的 64-band analysis-QMF 算子,则
|
||||
|
||||
$$
|
||||
X_{c,b,t}
|
||||
=
|
||||
\mathcal A_b\!\left(
|
||||
X_{c,b,t}=
|
||||
\mathcal A_b\left(
|
||||
\widetilde x_c[64t],\ldots,\widetilde x_c[64t+63];
|
||||
\mathbf s^{\mathrm A}_{c,t}
|
||||
\right).
|
||||
@@ -210,8 +248,7 @@ $$
|
||||
环绕声道的 band 0 还经过 21-tap 复 FIR:
|
||||
|
||||
$$
|
||||
\widehat X_{c,0,t}
|
||||
=
|
||||
\widehat X_{c,0,t}=
|
||||
\sum_{k=0}^{20}h_kX_{c,0,t-k}.
|
||||
$$
|
||||
|
||||
@@ -222,8 +259,7 @@ $$
|
||||
对每个对象 $o$、子带 $b$ 和时槽 $t$,对象频域值为五个核心声道的线性组合:
|
||||
|
||||
$$
|
||||
Z_{o,b,t}
|
||||
=
|
||||
Z_{o,b,t}=
|
||||
\sum_{c=0}^{4}
|
||||
M_{o,c,b,t}\widehat X_{c,b,t}.
|
||||
$$
|
||||
@@ -238,21 +274,20 @@ analysis 输入的 $1/16$ 缩放会在 inverse QMF 输出端由 $\times16$ 抵
|
||||
|
||||
$$
|
||||
\begin{aligned}
|
||||
\operatorname{zone}[2k] &= \operatorname{src}[4k],\\
|
||||
\operatorname{zone}[2k+1] &= -\operatorname{src}[4k+1],\\
|
||||
\operatorname{zone}[126-2k] &= \operatorname{src}[4k+2],\\
|
||||
\operatorname{zone}[127-2k] &= \operatorname{src}[4k+3].
|
||||
\mathrm{zone}[2k] &= \mathrm{src}[4k],\\
|
||||
\mathrm{zone}[2k+1] &= -\mathrm{src}[4k+1],\\
|
||||
\mathrm{zone}[126-2k] &= \mathrm{src}[4k+2],\\
|
||||
\mathrm{zone}[127-2k] &= \mathrm{src}[4k+3].
|
||||
\end{aligned}
|
||||
$$
|
||||
|
||||
把 `zone` 重新视为 64 个复数后执行未归一化 64 点 FFT:
|
||||
|
||||
$$
|
||||
F_k
|
||||
=
|
||||
F_k=
|
||||
\sum_{n=0}^{63}
|
||||
\operatorname{zone}_n
|
||||
\exp\!\left(-j\frac{2\pi kn}{64}\right).
|
||||
\mathrm{zone}_n
|
||||
\exp\left(-j\frac{2\pi kn}{64}\right).
|
||||
$$
|
||||
|
||||
### 7.2 调制与合成
|
||||
@@ -260,8 +295,7 @@ $$
|
||||
定义旋转系数
|
||||
|
||||
$$
|
||||
r_k
|
||||
=
|
||||
r_k=
|
||||
\frac12\left(
|
||||
\sin\frac{\pi k}{128}
|
||||
+j\cos\frac{\pi k}{128}
|
||||
@@ -277,9 +311,8 @@ $$
|
||||
令 $\mathcal S$ 表示带 640 项 synthesis window 和跨时槽状态的 polyphase 合成算子:
|
||||
|
||||
$$
|
||||
\mathbf y_{o,t}
|
||||
=
|
||||
\mathcal S\!\left(
|
||||
\mathbf y_{o,t}=
|
||||
\mathcal S\left(
|
||||
\mathbf R_{o,t},W,\mathbf s^{\mathrm S}_{o,t}
|
||||
\right).
|
||||
$$
|
||||
@@ -287,9 +320,8 @@ $$
|
||||
对象输出为
|
||||
|
||||
$$
|
||||
y_o[64t+r]
|
||||
=
|
||||
\operatorname{clip}\!\left(
|
||||
y_o[64t+r]=
|
||||
\mathrm{clip}\left(
|
||||
16\,\mathbf y_{o,t}[r],-1,1
|
||||
\right)G_{\mathrm{clip}},
|
||||
$$
|
||||
@@ -301,9 +333,8 @@ $$
|
||||
LFE 不经过对象矩阵或 inverse QMF,而是使用 1217-sample 延迟。输入与输出端的比例因子抵消后:
|
||||
|
||||
$$
|
||||
y_{\mathrm{LFE}}[n]
|
||||
=
|
||||
\operatorname{clip}\!\left(
|
||||
y_{\mathrm{LFE}}[n]=
|
||||
\mathrm{clip}\left(
|
||||
x_{\mathrm{LFE,core}}[n-1217],-1,1
|
||||
\right).
|
||||
$$
|
||||
@@ -313,9 +344,8 @@ $$
|
||||
横向和纵向网格使用 $N=62$,高度网格使用 $N=15$。量化函数为
|
||||
|
||||
$$
|
||||
q_N(k)
|
||||
=
|
||||
\min\!\left(
|
||||
q_N(k)=
|
||||
\min\left(
|
||||
32767,
|
||||
\left\lfloor\frac{32768k}{N}+\frac12\right\rfloor
|
||||
\right).
|
||||
@@ -336,11 +366,11 @@ $$
|
||||
转换为 ADM 网格时:
|
||||
|
||||
$$
|
||||
k_1=\operatorname{round}\!\left(\frac{62q_1}{32767}\right),
|
||||
k_1=\mathrm{round}\left(\frac{62q_1}{32767}\right),
|
||||
\quad
|
||||
k_2=\operatorname{round}\!\left(\frac{62q_2}{32767}\right),
|
||||
k_2=\mathrm{round}\left(\frac{62q_2}{32767}\right),
|
||||
\quad
|
||||
k_3=\operatorname{round}\!\left(\frac{15q_3}{32767}\right),
|
||||
k_3=\mathrm{round}\left(\frac{15q_3}{32767}\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
@@ -404,17 +434,15 @@ $$
|
||||
二维点增益为
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{2D}}(u,v)
|
||||
=
|
||||
\mathbf G_{\mathrm{2D}}(u,v)=
|
||||
\mathbf h(u)\odot\mathbf v(v).
|
||||
$$
|
||||
|
||||
对于只有一对水平环绕、没有独立 side/rear 两对的 5.1 系列布局,纵向坐标使用
|
||||
|
||||
$$
|
||||
v_{\mathrm{floor}}
|
||||
=
|
||||
\operatorname{clamp}(2v,0,1).
|
||||
v_{\mathrm{floor}}=
|
||||
\mathrm{clamp}(2v,0,1).
|
||||
$$
|
||||
|
||||
其他布局使用 $v_{\mathrm{floor}}=v$。
|
||||
@@ -424,8 +452,7 @@ $$
|
||||
三维布局分别计算地面层增益 $\mathbf G_f$ 和高度层增益 $\mathbf G_h$:
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{point}}(u,v,w)
|
||||
=
|
||||
\mathbf G_{\mathrm{point}}(u,v,w)=
|
||||
\cos\left(\frac\pi2w\right)\mathbf G_f
|
||||
+
|
||||
\sin\left(\frac\pi2w\right)\mathbf G_h.
|
||||
@@ -450,8 +477,7 @@ $$
|
||||
最大位置补偿为
|
||||
|
||||
$$
|
||||
A_{\max}
|
||||
=
|
||||
A_{\max}=
|
||||
-\max\left(4.5-1.5H-3F,0\right)
|
||||
\quad\text{dB}.
|
||||
$$
|
||||
@@ -459,15 +485,15 @@ $$
|
||||
前后与高度位置权重为
|
||||
|
||||
$$
|
||||
p_v=\operatorname{clamp}\left(\frac v{0.6},0,1\right),
|
||||
p_v=\mathrm{clamp}\left(\frac v{0.6},0,1\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
p_w=\operatorname{clamp}\left(\frac{w-0.2}{0.8},0,1\right),
|
||||
p_w=\mathrm{clamp}\left(\frac{w-0.2}{0.8},0,1\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
p=\operatorname{clamp}(p_v+p_w,0,1).
|
||||
p=\mathrm{clamp}(p_v+p_w,0,1).
|
||||
$$
|
||||
|
||||
线性补偿增益为
|
||||
@@ -479,8 +505,7 @@ $$
|
||||
对象的目标增益向量为
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{target}}
|
||||
=
|
||||
\mathbf G_{\mathrm{target}}=
|
||||
G_{\mathrm{object}}
|
||||
G_{\mathrm{pos}}
|
||||
\mathbf G_{\mathrm{point}}.
|
||||
@@ -491,8 +516,7 @@ $$
|
||||
OAMD 更新的编码位置为
|
||||
|
||||
$$
|
||||
s_{\mathrm{coded}}
|
||||
=
|
||||
s_{\mathrm{coded}}=
|
||||
s_{\mathrm{frame}}
|
||||
+s_{\mathrm{outer}}
|
||||
+s_{\mathrm{OAMD}}
|
||||
@@ -502,16 +526,15 @@ $$
|
||||
decoder 输出 PCM timeline 上的理论更新位置为
|
||||
|
||||
$$
|
||||
s_{\mathrm{theoretical}}
|
||||
=s_{\mathrm{coded}}+d_{\mathrm{decoder}},
|
||||
s_{\mathrm{theoretical}}=
|
||||
s_{\mathrm{coded}}+d_{\mathrm{decoder}},
|
||||
\qquad d_{\mathrm{decoder}}=1473.
|
||||
$$
|
||||
|
||||
扬声器 renderer 保留现有的处理块长度 $B=32$,更新点对齐为
|
||||
|
||||
$$
|
||||
\widehat s
|
||||
=
|
||||
\widehat s=
|
||||
B\left\lfloor
|
||||
\frac{s_{\mathrm{theoretical}}+B/2-1}{B}
|
||||
\right\rfloor.
|
||||
@@ -522,8 +545,7 @@ $$
|
||||
给定 ramp duration $D$,block 数为
|
||||
|
||||
$$
|
||||
K
|
||||
=
|
||||
K=
|
||||
\left\lfloor
|
||||
\frac{D+B/2-1}{B}
|
||||
\right\rfloor.
|
||||
@@ -550,8 +572,7 @@ $$
|
||||
对目标输出声道 $c$:
|
||||
|
||||
$$
|
||||
y_c[n]
|
||||
=
|
||||
y_c[n]=
|
||||
\delta_{c,\mathrm{LFE}}x_{\mathrm{LFE}}[n]
|
||||
+
|
||||
\sum_{o=1}^{15}x_o[n]g_{o,c}[n].
|
||||
@@ -560,8 +581,7 @@ $$
|
||||
其中
|
||||
|
||||
$$
|
||||
\delta_{c,\mathrm{LFE}}
|
||||
=
|
||||
\delta_{c,\mathrm{LFE}}=
|
||||
\begin{cases}
|
||||
1, & c\text{ 为目标布局的 LFE},\\
|
||||
0, & \text{其他声道}.
|
||||
@@ -573,16 +593,15 @@ $$
|
||||
若输出 PCM24,量化关系为
|
||||
|
||||
$$
|
||||
y_{24}[n]
|
||||
=
|
||||
\operatorname{trunc}\left(
|
||||
8388607\,\operatorname{clip}(y[n],-1,1)
|
||||
y_{24}[n]=
|
||||
\mathrm{trunc}\left(
|
||||
8388607\,\mathrm{clip}(y[n],-1,1)
|
||||
\right).
|
||||
$$
|
||||
|
||||
## 14. 公式适用范围
|
||||
|
||||
- JOC 矩阵部分描述 dense JOC;Sparse JOC 使用不同的稀疏系数/索引路径。
|
||||
- JOC 矩阵部分同时描述 dense MTX 与 sparse IDX/VEC 两条差分语法。
|
||||
- 扬声器声像部分描述普通点对象;extent、spread、divergence 等模式需要额外模型。
|
||||
- 多个 OAMD position block 必须按其时间顺序调度。
|
||||
- limiter 属于独立后处理,不包含在上述混音公式中。
|
||||
|
||||
@@ -1,178 +0,0 @@
|
||||
# JustOneCacophony native-core notes
|
||||
|
||||
[中文](native.md) · [Back to README](../README.en.md)
|
||||
|
||||
## 1. Responsibility boundary
|
||||
|
||||
`native/` contains only the state-heavy, frequently called DSP and speaker-rendering kernels. High-level EMDF/JOC/OAMD parsing, error reporting, ADM assembly, and the CLI remain in Python.
|
||||
|
||||
Python calls a C ABI through the standard-library `ctypes` module. The native core does not use pybind11, Cython, FFTW, MKL, or OpenMP. It is an optional acceleration path and does not expand the set of supported stream variants.
|
||||
|
||||
Main files:
|
||||
|
||||
```text
|
||||
native/include/eac3joc_core.h C ABI
|
||||
native/src/eac3joc_core.cpp JOC/QMF object reconstruction
|
||||
native/src/speaker_renderer.cpp object-to-speaker rendering
|
||||
native/src/qmf_tables.h QMF tables
|
||||
native/src/speaker_layouts.h layout tables
|
||||
native/src/joc_huffman_tables.h JOC Huffman tables
|
||||
src/native_renderer.py JOC ctypes bridge
|
||||
src/speaker_native_renderer.py speaker ctypes bridge
|
||||
```
|
||||
|
||||
## 2. JOC rendering ABI
|
||||
|
||||
An opaque renderer owns all cross-frame state. Its main call is:
|
||||
|
||||
```c
|
||||
int ejoc_renderer_process(
|
||||
ejoc_renderer_handle handle,
|
||||
const float* bed5_planar, /* [5][1536] */
|
||||
const float* lfe, /* [1536] or NULL */
|
||||
uint32_t object_mask,
|
||||
const uint8_t* n_bands, /* [15] */
|
||||
const uint8_t* n_dpoints, /* [15] */
|
||||
const uint8_t* slope_idx, /* [15] */
|
||||
const uint8_t* offset_ts, /* [15][2] */
|
||||
const double* dq, /* [15][2][5][23] */
|
||||
double clipgain,
|
||||
float phase_new,
|
||||
float output_scale,
|
||||
float* output16_planar); /* [16][1536] */
|
||||
```
|
||||
|
||||
Python performs dense-JOC Huffman decoding, differential reconstruction, and dequantization before the call. Sparse JOC is not silently passed to the dense native path.
|
||||
|
||||
Thread control is exposed as:
|
||||
|
||||
```c
|
||||
int ejoc_renderer_set_threads(ejoc_renderer_handle handle, uint32_t total_threads);
|
||||
uint32_t ejoc_renderer_thread_count(ejoc_renderer_handle handle);
|
||||
```
|
||||
|
||||
`total_threads` includes the calling thread. Frames must be submitted sequentially to one renderer instance; the instance may parallelize work across objects and analysis channels.
|
||||
|
||||
## 3. Cross-frame state
|
||||
|
||||
Each JOC renderer stores:
|
||||
|
||||
- analysis FIFO: `double[5][9][64]`;
|
||||
- L/R/C analysis delay: `float[3][10][64]`;
|
||||
- Ls/Rs QMF delay: `complex<double>[2][10][64]`;
|
||||
- Ls/Rs band-0 FIR history: `complex<double>[2][20]`;
|
||||
- previous matrix interpolation values: `double[15][5][64]`;
|
||||
- inverse-QMF state: `double[15][640]`;
|
||||
- LFE delay: `double[1217]`.
|
||||
|
||||
This state belongs to the renderer instance. Processing cannot be arbitrarily segmented or reordered without a corresponding state checkpoint.
|
||||
|
||||
## 4. FFT, QMF, and precision
|
||||
|
||||
The native core contains a fixed 64-point radix-2 complex FFT:
|
||||
|
||||
- analysis QMF uses a forward FFT followed by division by 64;
|
||||
- inverse QMF uses the fixed reorder, rotation, and 640-value active-window state;
|
||||
- no external FFT library is called.
|
||||
|
||||
The JOC path uses:
|
||||
|
||||
- float32 core-PCM input;
|
||||
- double matrices, complex QMF, FFT, FIR, and cross-frame state;
|
||||
- float32 phase and final gain;
|
||||
- float32 16-channel object output.
|
||||
|
||||
## 5. Speaker-rendering ABI
|
||||
|
||||
The same shared library exports object-to-speaker rendering:
|
||||
|
||||
```c
|
||||
uint32_t ejoc_speaker_layout_channel_count(uint32_t speaker_bitfield);
|
||||
|
||||
ejoc_speaker_renderer_handle
|
||||
ejoc_speaker_renderer_create(uint32_t speaker_bitfield);
|
||||
|
||||
int ejoc_speaker_renderer_process(
|
||||
ejoc_speaker_renderer_handle handle,
|
||||
const float* objects16_interleaved,
|
||||
uint32_t sample_count,
|
||||
uint32_t metadata_count,
|
||||
const uint32_t* metadata_offsets,
|
||||
const uint32_t* ramp_durations,
|
||||
const uint16_t* positions_q15,
|
||||
const uint8_t* region_indices,
|
||||
const uint8_t* height_enabled,
|
||||
const double* object_gains,
|
||||
double* output_interleaved);
|
||||
```
|
||||
|
||||
Input channel 0 is LFE and channels 1–15 are objects. Each metadata entry is an object-state snapshot. `sample_count` must be a multiple of 32; unfinished gain ramps remain in the handle and continue across calls.
|
||||
|
||||
The speaker path uses float32 object input, double coordinates/gains/accumulation, and interleaved double output. Quantization to float32 or PCM24 happens when the WAV is written.
|
||||
|
||||
Supported layouts:
|
||||
|
||||
```text
|
||||
2.0 3.1 5.1 7.1 5.1.2 5.1.4 7.1.2 7.1.4 9.1.4 9.1.6
|
||||
```
|
||||
|
||||
## 6. Binaural-rendering ABI
|
||||
|
||||
The shared library provides a 512-sample float64 binaural DSP interface:
|
||||
|
||||
```c
|
||||
ejoc_binaural_renderer_handle ejoc_binaural_renderer_create(void);
|
||||
int ejoc_binaural_renderer_configure_kernels(...);
|
||||
int ejoc_binaural_renderer_configure_room(...);
|
||||
int ejoc_binaural_renderer_process(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
const double* input16_interleaved, /* [512][16] */
|
||||
const double* gains_complex, /* [16][2][77][2] */
|
||||
const double* room_sends, /* [16] */
|
||||
double output_gain,
|
||||
double* output_stereo_interleaved); /* [512][2] */
|
||||
```
|
||||
|
||||
Python parses the model, evaluates the OAMD timeline, and supplies complex gains and room sends every 512 samples. The C++ handle owns QMF, hybrid, recursive-room, and QMF-synthesis state. Inputs, state, accumulation, and output are double/complex double.
|
||||
|
||||
## 7. Building
|
||||
|
||||
The CMake definition is `native/CMakeLists.txt`. Run from the repository root:
|
||||
|
||||
```powershell
|
||||
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
|
||||
cmake --build build/cmake --config Release
|
||||
cmake --install build/cmake --config Release
|
||||
```
|
||||
|
||||
Platform runtime names:
|
||||
|
||||
```text
|
||||
Windows lib/eac3joc_core.dll
|
||||
Linux lib/libeac3joc_core.so
|
||||
macOS lib/libeac3joc_core.dylib
|
||||
```
|
||||
|
||||
The MSVC configuration uses the static CRT. Other runtime dependencies depend on the platform and toolchain and should be checked independently before publishing a prebuilt library.
|
||||
|
||||
The repository does not include native binaries by default. A prebuilt Release runtime or a locally built runtime can be placed directly under `lib/`.
|
||||
|
||||
## 8. Runtime lookup and fallback
|
||||
|
||||
Lookup order:
|
||||
|
||||
1. explicit `--native-library`;
|
||||
2. `EAC3JOC_NATIVE_LIBRARY`;
|
||||
3. the standard platform filename under `lib/`.
|
||||
|
||||
`--backend auto` falls back to NumPy when loading fails, and `--backend python` skips native discovery. The current CLI also prints the failure and falls back for `--backend native`; this existing behavior should not be read as successful native execution.
|
||||
|
||||
## 9. Implementation boundaries
|
||||
|
||||
- The native layer accepts only dense-JOC data already parsed by Python.
|
||||
- The ABI fixes a 1536-sample JOC frame, at most 15 objects, at most 23 parameter bands, and at most 2 data points.
|
||||
- The shared library and Python bridge must report the same ABI version.
|
||||
- Only the ABI and data types are specified across platforms; bit-identical float64 results are not guaranteed.
|
||||
- Private table headers under `native/src/` serve the native side only. The current repository does not include the scripts that generated those headers.
|
||||
|
||||
See the [mathematical notes](math.en.md) for the related formulas.
|
||||
-178
@@ -1,178 +0,0 @@
|
||||
# JustOneCacophony 原生核说明
|
||||
|
||||
[English](native.en.md) · [返回 README](../README.md)
|
||||
|
||||
## 1. 职责边界
|
||||
|
||||
`native/` 只承载状态密集、调用频繁的 DSP 与扬声器渲染核。EMDF/JOC/OAMD 高层解析、错误报告、ADM 组装和 CLI 保留在 Python 中。
|
||||
|
||||
Python 通过标准库 `ctypes` 调用 C ABI;原生核不使用 pybind11、Cython、FFTW、MKL 或 OpenMP。它是可选加速路径,不扩大项目所支持的码流范围。
|
||||
|
||||
主要文件:
|
||||
|
||||
```text
|
||||
native/include/eac3joc_core.h C ABI
|
||||
native/src/eac3joc_core.cpp JOC/QMF 对象重建
|
||||
native/src/speaker_renderer.cpp 对象到扬声器渲染
|
||||
native/src/qmf_tables.h QMF 表
|
||||
native/src/speaker_layouts.h 布局表
|
||||
native/src/joc_huffman_tables.h JOC Huffman 表
|
||||
src/native_renderer.py JOC ctypes 桥
|
||||
src/speaker_native_renderer.py 扬声器 ctypes 桥
|
||||
```
|
||||
|
||||
## 2. JOC 渲染 ABI
|
||||
|
||||
一个 opaque renderer 保存所有跨帧状态。主要调用为:
|
||||
|
||||
```c
|
||||
int ejoc_renderer_process(
|
||||
ejoc_renderer_handle handle,
|
||||
const float* bed5_planar, /* [5][1536] */
|
||||
const float* lfe, /* [1536] or NULL */
|
||||
uint32_t object_mask,
|
||||
const uint8_t* n_bands, /* [15] */
|
||||
const uint8_t* n_dpoints, /* [15] */
|
||||
const uint8_t* slope_idx, /* [15] */
|
||||
const uint8_t* offset_ts, /* [15][2] */
|
||||
const double* dq, /* [15][2][5][23] */
|
||||
double clipgain,
|
||||
float phase_new,
|
||||
float output_scale,
|
||||
float* output16_planar); /* [16][1536] */
|
||||
```
|
||||
|
||||
Dense JOC 的 Huffman 解码、差分还原和去量化先在 Python 中完成。Sparse JOC 不会被静默送入 dense 原生路径。
|
||||
|
||||
线程接口为:
|
||||
|
||||
```c
|
||||
int ejoc_renderer_set_threads(ejoc_renderer_handle handle, uint32_t total_threads);
|
||||
uint32_t ejoc_renderer_thread_count(ejoc_renderer_handle handle);
|
||||
```
|
||||
|
||||
`total_threads` 包含调用线程。单个 renderer 实例必须顺序提交帧;实例内部可以按对象和 analysis channel 并行。
|
||||
|
||||
## 3. 跨帧状态
|
||||
|
||||
每个 JOC renderer 独立保存:
|
||||
|
||||
- analysis FIFO:`double[5][9][64]`;
|
||||
- L/R/C analysis delay:`float[3][10][64]`;
|
||||
- Ls/Rs QMF delay:`complex<double>[2][10][64]`;
|
||||
- Ls/Rs band-0 FIR history:`complex<double>[2][20]`;
|
||||
- 矩阵插值 previous:`double[15][5][64]`;
|
||||
- inverse-QMF state:`double[15][640]`;
|
||||
- LFE delay:`double[1217]`。
|
||||
|
||||
这些状态属于 renderer 实例,不能在无 checkpoint 的情况下任意分段或乱序处理。
|
||||
|
||||
## 4. FFT、QMF 与精度
|
||||
|
||||
原生核包含固定 64 点 radix-2 complex FFT:
|
||||
|
||||
- analysis QMF 使用 forward FFT 后除以 64;
|
||||
- inverse QMF 使用固定重排、旋转和 640 项有效窗状态;
|
||||
- 不调用外部 FFT 库。
|
||||
|
||||
JOC 路径的数值类型为:
|
||||
|
||||
- 核心 PCM 输入:float32;
|
||||
- 矩阵、复 QMF、FFT、FIR 和跨帧状态:double;
|
||||
- phase 与最终 gain:float32;
|
||||
- 16 声道对象输出:float32。
|
||||
|
||||
## 5. 扬声器渲染 ABI
|
||||
|
||||
同一个共享库还导出对象到扬声器布局的渲染接口:
|
||||
|
||||
```c
|
||||
uint32_t ejoc_speaker_layout_channel_count(uint32_t speaker_bitfield);
|
||||
|
||||
ejoc_speaker_renderer_handle
|
||||
ejoc_speaker_renderer_create(uint32_t speaker_bitfield);
|
||||
|
||||
int ejoc_speaker_renderer_process(
|
||||
ejoc_speaker_renderer_handle handle,
|
||||
const float* objects16_interleaved,
|
||||
uint32_t sample_count,
|
||||
uint32_t metadata_count,
|
||||
const uint32_t* metadata_offsets,
|
||||
const uint32_t* ramp_durations,
|
||||
const uint16_t* positions_q15,
|
||||
const uint8_t* region_indices,
|
||||
const uint8_t* height_enabled,
|
||||
const double* object_gains,
|
||||
double* output_interleaved);
|
||||
```
|
||||
|
||||
输入声道 0 为 LFE,1–15 为对象。每个 metadata entry 是一份对象状态快照。`sample_count` 必须是 32 的倍数;未完成的增益斜坡保存在 handle 中并跨调用继续。
|
||||
|
||||
扬声器路径使用 float32 对象输入、double 坐标/增益/累加与 interleaved double 输出;写 WAV 时才量化为 float32 或 PCM24。
|
||||
|
||||
支持的布局为:
|
||||
|
||||
```text
|
||||
2.0 3.1 5.1 7.1 5.1.2 5.1.4 7.1.2 7.1.4 9.1.4 9.1.6
|
||||
```
|
||||
|
||||
## 6. 双耳渲染 ABI
|
||||
|
||||
共享库提供 512-sample float64 双耳 DSP:
|
||||
|
||||
```c
|
||||
ejoc_binaural_renderer_handle ejoc_binaural_renderer_create(void);
|
||||
int ejoc_binaural_renderer_configure_kernels(...);
|
||||
int ejoc_binaural_renderer_configure_room(...);
|
||||
int ejoc_binaural_renderer_process(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
const double* input16_interleaved, /* [512][16] */
|
||||
const double* gains_complex, /* [16][2][77][2] */
|
||||
const double* room_sends, /* [16] */
|
||||
double output_gain,
|
||||
double* output_stereo_interleaved); /* [512][2] */
|
||||
```
|
||||
|
||||
Python 负责模型解析、OAMD 时间轴和每 512 samples 的 complex gains/room sends。C++ handle 保存 QMF、hybrid、递归 room 和 QMF synthesis 状态。全部输入、状态、乘加和输出均为 double/complex double。
|
||||
|
||||
## 7. 构建
|
||||
|
||||
CMake 定义位于 `native/CMakeLists.txt`。从仓库根目录运行:
|
||||
|
||||
```powershell
|
||||
cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX="$PWD/lib"
|
||||
cmake --build build/cmake --config Release
|
||||
cmake --install build/cmake --config Release
|
||||
```
|
||||
|
||||
平台运行库文件名:
|
||||
|
||||
```text
|
||||
Windows lib/eac3joc_core.dll
|
||||
Linux lib/libeac3joc_core.so
|
||||
macOS lib/libeac3joc_core.dylib
|
||||
```
|
||||
|
||||
MSVC 配置使用静态 CRT。其他运行时依赖由平台和工具链决定,发布预构建库前应对产物独立检查。
|
||||
|
||||
仓库默认不附带原生二进制。预构建的 Release 运行库或自行构建的运行库均可直接放入 `lib/`。
|
||||
|
||||
## 8. 运行时查找与回退
|
||||
|
||||
查找顺序为:
|
||||
|
||||
1. 显式 `--native-library`;
|
||||
2. `EAC3JOC_NATIVE_LIBRARY`;
|
||||
3. `lib/` 下当前平台的标准文件名。
|
||||
|
||||
`--backend auto` 在加载失败时回退到 NumPy;`--backend python` 跳过原生探测。`--backend native` 当前也会打印失败原因后回退,这是现有 CLI 行为,不应理解为原生库已成功使用。
|
||||
|
||||
## 9. 实现边界
|
||||
|
||||
- 原生层只接收 Python 已解析的 dense JOC 数据。
|
||||
- ABI 固定了 1536-sample JOC 帧、最多 15 个对象、最多 23 个参数带和最多 2 个数据点。
|
||||
- 共享库与 Python 桥需要 ABI version 一致。
|
||||
- 跨平台只约定 ABI 与数据类型,不保证 float64 结果逐位一致。
|
||||
- `native/src/` 中的私有表头只服务于原生侧;当前仓库不包含重新生成这些头文件的脚本。
|
||||
|
||||
相关公式见[数学说明](math.md)。
|
||||
+128
@@ -0,0 +1,128 @@
|
||||
# SIMD and runtime dispatch
|
||||
|
||||
[中文](simd.md) · [Back to README](../README.en.md)
|
||||
|
||||
The heaviest loops in the binaural path (QMF analysis and synthesis, the hybrid
|
||||
analysis low join, hybrid-domain path rendering, the ROOM FFT, the spherical
|
||||
harmonic alignment) each have a runtime-dispatched vector implementation: one
|
||||
binary carries several instruction-set variants, asks the CPU once at startup and
|
||||
runs the widest one. **The output is byte-identical either way** — that is a hard
|
||||
constraint, not a goal.
|
||||
|
||||
```text
|
||||
JOC_SIMD=auto|scalar|sse2|avx2|avx512|neon pin one tier (used for verification)
|
||||
JOC_SIMD_LOG=1 report the ISA each kernel actually got
|
||||
```
|
||||
|
||||
## Why the split has to happen per translation unit
|
||||
|
||||
MSVC has no function-level attribute like `__attribute__((target("avx2")))`: one
|
||||
`.cpp` file gets one `/arch`. So every ISA is its own translation unit with its own
|
||||
`/arch:AVX2` or `/arch:AVX512` (GCC/Clang: `-mavx2` / `-mavx512f`), and
|
||||
`dispatch.cpp` fills the function table at run time. The baseline units — the
|
||||
dispatcher itself, the CPU probe and the scalar reference — carry **no** `/arch` at
|
||||
all and stay on the SSE2 that x86-64 guarantees.
|
||||
|
||||
A trap from the history of this tree: `JOC_ENABLE_AVX2` used to be global, so
|
||||
turning it on put AVX2 instructions into the very code paths that exist for older
|
||||
CPUs. It now only selects whether the AVX2 unit is compiled in.
|
||||
|
||||
## Directory layout
|
||||
|
||||
One flat directory, **the instruction set in the file name and never in a
|
||||
subdirectory** — that is how FLAC does it (`lpc.c` sits next to
|
||||
`lpc_intrin_sse2.c`, `lpc_intrin_avx2.c` and `lpc_intrin_neon.c`, with the CPU
|
||||
probe in its own `cpu.c`).
|
||||
|
||||
```text
|
||||
src/simd/
|
||||
simd.h the contract: Isa / Kernel / Kernels / dimensions
|
||||
cpu_probe.{h,cpp} "can this machine run ISA X": CPUID+XGETBV / __builtin_cpu_supports / getauxval
|
||||
dispatch.cpp policy: JOC_SIMD, the fallback ladder, the table, the log
|
||||
kernels_scalar.cpp the reference (Isa::scalar; always built, always selectable)
|
||||
kernels_intrin_avx2.cpp /arch:AVX2 -mavx2
|
||||
kernels_intrin_avx512.cpp /arch:AVX512 -mavx512f
|
||||
kernels_intrin_neon.cpp AArch64 default
|
||||
```
|
||||
|
||||
The module lives at `src/simd/`, not `src/dsp/simd/`: `foundation/`, `hrtf/` and
|
||||
`binaural/` all call into these kernels, so it is a cross-cutting layer rather than
|
||||
a submodule of the DSP code (and `src/dsp/` held nothing else).
|
||||
|
||||
The header comment of `simd.h` carries the same module map; keep the two in sync
|
||||
when the layout changes.
|
||||
|
||||
## Three rules
|
||||
|
||||
1. **Only `kernels_intrin_*.cpp` gets a wider flag.** `CMakeLists.txt` names those
|
||||
files explicitly with `set_source_files_properties`; every other target stays on
|
||||
the architecture's guaranteed ISA. A unit that goes wide without matching that
|
||||
name fails `devtools/vec/isa_audit.ps1`.
|
||||
2. **No dynamic initialisation inside an ISA unit.** Those objects are linked into
|
||||
the same image as the baseline, so a global constructor would execute a wide
|
||||
instruction before the dispatcher has looked at the CPU. Constant tables are
|
||||
fine — they land in `.rdata`.
|
||||
3. **Bit-exactness comes from the lane assignment, not from the ISA.** A lane may
|
||||
only carry mutually independent outputs; the rounding sequence of a single
|
||||
output, the separation of multiply and add (never an FMA) and the summation
|
||||
order all stay exactly as `kernels_scalar.cpp` wrote them. Layout changes that
|
||||
only reorder stored doubles (rank-minor basis tables, term-ordered tap tables,
|
||||
stage-contiguous twiddle tables, band-major ROOM planes) are allowed.
|
||||
|
||||
## How the choice is made
|
||||
|
||||
`dispatch.cpp` parses `JOC_SIMD` first (forcing a tier this build or this machine
|
||||
does not have prints a diagnostic and falls back, rather than pretending and
|
||||
crashing), then walks `avx512 → avx2 → sse2 → neon` and picks, per kernel, the
|
||||
widest implementation that is both compiled into this binary **and** runnable
|
||||
here, falling back to the scalar reference. An AVX-512 unit therefore costs
|
||||
nothing on a CPU without AVX-512; it is simply never selected.
|
||||
|
||||
On x86 the probe requires CPUID *and* XGETBV to agree: CPUID says the silicon can
|
||||
do it, XCR0 says the OS saves the registers it needs. Either one alone is not
|
||||
enough — using AVX without OS state support corrupts other threads across a
|
||||
context switch. AArch64 needs no probe; ASIMD is the architectural baseline.
|
||||
|
||||
`sse2` is a selectable tier with **no unit of its own**, on purpose: a 128-bit SSE2
|
||||
register is the register a scalar double already occupies, SSE2 cannot widen
|
||||
double-precision arithmetic, and hand-written SSE2 would only add moves. The tier
|
||||
resolves to the baseline unit.
|
||||
|
||||
## Effect
|
||||
|
||||
30-second reference cases, one binary with only `JOC_SIMD` switched (DSP stage,
|
||||
`t_render_dsp`):
|
||||
|
||||
| Case | `scalar` | `auto` | DSP speed-up | End-to-end wall clock |
|
||||
|---|---|---|---|---|
|
||||
| Binaural Rosella | 1.256 s | **0.640 s** | **1.96×** | 1.690 → **0.941 s** |
|
||||
| Binaural SOFA | 1.560 s | **0.654 s** | **2.39×** | 1.859 → **0.849 s** |
|
||||
| Speaker 5.1 / 9.1.6 / ADM | — | — | 1.00× | no regression (these kernels are not on those paths) |
|
||||
|
||||
The vectorised loops themselves gain more: synthesis basis 5.89×, 13-tap low join
|
||||
4.50×, SOFA QMF synthesis 4.27×, 128-point FFT 2.24×. The whole pipeline stops
|
||||
short of 8× because a good part of the time goes to parameter setup, straight
|
||||
copies and file writing — none of which has independent work items — and because
|
||||
SOFA's 33 M sin/cos calls per sample cannot be vectorised under a byte-exactness
|
||||
contract.
|
||||
|
||||
## Verifying a change
|
||||
|
||||
```powershell
|
||||
$env:JOC_SIMD_LOG='1' # per-kernel ISA on this machine
|
||||
$env:JOC_SIMD='scalar' # force the reference: hashes must not move
|
||||
pwsh -NoProfile -File devtools\vec\isa_audit.ps1 # disassemble every .obj: 0 unguarded wide instructions
|
||||
pwsh -NoProfile -File devtools\vec\sha_matrix.ps1 # 5 tiers x 2 renders against the reference digests
|
||||
cmd /c devtools\vec\build_kernel_probe.bat # per-kernel byte digests (8 kernels)
|
||||
```
|
||||
|
||||
## Adding an ISA
|
||||
|
||||
1. Write `kernels_intrin_<isa>.cpp`, implementing the slots you have and leaving the
|
||||
rest `nullptr` — the dispatcher falls back per kernel (that is how
|
||||
`qmf_synthesis_basis` is handled in the AVX-512 unit).
|
||||
2. Add it to `JOC_SIMD_SOURCES` in `CMakeLists.txt` with `JOC_SIMD_HAVE_<ISA>=1` and
|
||||
its flag, and extend `Isa`, `isa_rank`, `isa_compiled`, `isa_supported` and the
|
||||
`JOC_SIMD` name table in `simd.h` / `dispatch.cpp`.
|
||||
3. Verify: the kernel digests must match the scalar unit byte for byte, and the
|
||||
reference renders must keep their SHA-256 on every tier.
|
||||
+107
@@ -0,0 +1,107 @@
|
||||
# SIMD 与运行时派发
|
||||
|
||||
[English](simd.en.md) · [返回 README](../README.md)
|
||||
|
||||
双耳通路里最重的那几段循环(QMF 分析/合成、混合分析的低频拼接、混合域路径渲染、
|
||||
ROOM 的 FFT、球谐对齐)都有一份运行时分派的向量实现:同一份二进制里装多套 ISA 代码,
|
||||
启动时问一次 CPU,然后选最宽的那套跑。**输出逐字节不变**——这是硬约束,不是目标。
|
||||
|
||||
```text
|
||||
JOC_SIMD=auto|scalar|sse2|avx2|avx512|neon 强制某一层(验收用)
|
||||
JOC_SIMD_LOG=1 打印每个 kernel 实际生效的 ISA
|
||||
```
|
||||
|
||||
## 为什么必须"按编译单元分 ISA"
|
||||
|
||||
MSVC 没有 `__attribute__((target("avx2")))` 这类函数级多版本能力,一个 .cpp 只能有
|
||||
一个 `/arch`。所以每个 ISA 一个编译单元,各自带自己的 `/arch:AVX2` / `/arch:AVX512`
|
||||
(GCC/Clang 是 `-mavx2` / `-mavx512f`),由 `dispatch.cpp` 在运行时填函数表。
|
||||
基线单元(含派发器本身、CPU 探测、标量参考实现)**不带任何 `/arch`**,它们只使用
|
||||
x86-64 架构保证的 SSE2。
|
||||
|
||||
历史坑:早先的 `JOC_ENABLE_AVX2` 是**全局**的,一旦打开,连"给老 CPU 用"的基线路径
|
||||
都带 AVX2 指令。现在这个选项只决定是否把 AVX2 单元编进二进制。
|
||||
|
||||
## 目录布局
|
||||
|
||||
一个扁平目录,**ISA 写在文件名里,不写进子目录**——这是 FLAC 的做法
|
||||
(`src/libFLAC/lpc.c` 旁边就是 `lpc_intrin_sse2.c` / `lpc_intrin_avx2.c` /
|
||||
`lpc_intrin_neon.c`,CPU 探测单独放在 `cpu.c`)。
|
||||
|
||||
```text
|
||||
src/simd/
|
||||
simd.h 唯一契约头:Isa / Kernel / Kernels / 维度常量
|
||||
cpu_probe.{h,cpp} "这台机器能不能跑 ISA X":CPUID+XGETBV / __builtin_cpu_supports / getauxval
|
||||
dispatch.cpp 策略:JOC_SIMD 解析、回退阶梯、填函数表、日志
|
||||
kernels_scalar.cpp 参考实现(Isa::scalar,永远编译、永远可选中)
|
||||
kernels_intrin_avx2.cpp /arch:AVX2 -mavx2
|
||||
kernels_intrin_avx512.cpp /arch:AVX512 -mavx512f
|
||||
kernels_intrin_neon.cpp AArch64 默认 -march=armv8-a+simd
|
||||
```
|
||||
|
||||
`simd.h` 的头注释里有一份同样的模块地图,改布局时两处一起改。
|
||||
|
||||
模块放在 `src/simd/` 而不是 `src/dsp/simd/`:这些 kernel 被 `foundation/`、`hrtf/`、
|
||||
`binaural/` 三个模块共用,是横切的一层,不是 DSP 的子模块(更何况 `src/dsp/` 里除了
|
||||
`simd/` 空无一物)。
|
||||
|
||||
## 三条规则
|
||||
|
||||
1. **只有 `kernels_intrin_*.cpp` 拿更宽的编译开关。** `CMakeLists.txt` 用
|
||||
`set_source_files_properties` 逐个点名,其余目标一律留在架构保证的 ISA 上。
|
||||
任何不属于这个命名却带了宽指令的单元都会被 `devtools/vec/isa_audit.ps1` 判失败。
|
||||
2. **ISA 单元里不许有动态初始化。** 它们和基线代码链进同一个镜像,全局构造函数会在
|
||||
派发器看 CPU 之前就跑宽指令。常量表没问题(落在 `.rdata`)。
|
||||
3. **逐位一致靠的是 lane 的划分,不是 ISA。** lane 里只能放**互相独立**的输出;单个输出
|
||||
的舍入序列、乘加分离(绝不用 FMA)、求和顺序都保持 `kernels_scalar.cpp` 原样。
|
||||
只改变 double **存放顺序**的布局改造(基函数表转秩小序、抽头表按项序、蝶形因子表
|
||||
按级连续化、ROOM 谱平面改频带主序)是允许的。
|
||||
|
||||
## 运行时怎么选
|
||||
|
||||
`dispatch.cpp` 先解析 `JOC_SIMD`(强制一个本机不支持的层会打印诊断并回退,而不是假装
|
||||
选中然后崩),再走阶梯 `avx512 → avx2 → sse2 → neon`,每个 kernel 单独挑"已编进本
|
||||
二进制 **且** 本机可跑"的最宽实现,挑不到就落到标量参考实现。所以 AVX-512 单元在
|
||||
不支持它的 CPU 上只是不被选中,不影响启动。
|
||||
|
||||
x86 的探测要 CPUID 与 XGETBV **同时**成立:CPUID 说明硅片有这个能力,XCR0 说明操作
|
||||
系统会保存对应寄存器状态,缺一个就不能用(否则上下文切换会踩坏别的线程)。AArch64
|
||||
不需要探测,ASIMD 是架构基线。
|
||||
|
||||
`sse2` 是一个有意保留的档位但**没有单独的单元**:128 位 SSE2 寄存器就是标量 double
|
||||
已经在用的寄存器,SSE2 加宽不了双精度运算,手写只会多出搬运指令,所以它选中的是基线
|
||||
单元。
|
||||
|
||||
## 效果
|
||||
|
||||
30 s 参考用例,同一二进制只切 `JOC_SIMD`(DSP 阶段 `t_render_dsp`):
|
||||
|
||||
| 用例 | `scalar` | `auto` | DSP 加速 | 端到端墙钟 |
|
||||
|---|---|---|---|---|
|
||||
| 双耳 Rosella | 1.256 s | **0.640 s** | **1.96×** | 1.690 → **0.941 s** |
|
||||
| 双耳 SOFA | 1.560 s | **0.654 s** | **2.39×** | 1.859 → **0.849 s** |
|
||||
| 扬声器 5.1 / 9.1.6 / ADM | — | — | 1.00× | 0 回归(不走这些 kernel) |
|
||||
|
||||
单看被向量化的循环,红利更大:合成基函数 5.89×、13 抽头低频拼接 4.50×、
|
||||
SOFA QMF 合成 4.27×、128 点 FFT 2.24×。整条流水线到不了 8×,是因为相当一部分时间在
|
||||
参数设置、直通拷贝、写盘这些没有独立工作项的代码上,以及 SOFA 每次采样的 33 M 次
|
||||
sin/cos 按逐位契约不能向量化。
|
||||
|
||||
## 验证
|
||||
|
||||
```powershell
|
||||
$env:JOC_SIMD_LOG='1' # 本机每个 kernel 实际选中的 ISA
|
||||
$env:JOC_SIMD='scalar' # 强制参考路径:哈希必须一动不动
|
||||
pwsh -NoProfile -File devtools\vec\isa_audit.ps1 # 反汇编全部 .obj:0 个无守卫的宽指令
|
||||
pwsh -NoProfile -File devtools\vec\sha_matrix.ps1 # 5 档 × 2 渲染,逐字节比对参考摘要
|
||||
cmd /c devtools\vec\build_kernel_probe.bat # kernel 级逐字节摘要(8 个 kernel)
|
||||
```
|
||||
|
||||
## 加一个新的 ISA
|
||||
|
||||
1. 写 `kernels_intrin_<isa>.cpp`,实现能实现的槽位,其余留 `nullptr`——派发器会逐
|
||||
kernel 回退(AVX-512 单元里的 `qmf_synthesis_basis` 就是这么处理的)。
|
||||
2. 在 `CMakeLists.txt` 里加进 `JOC_SIMD_SOURCES`、`JOC_SIMD_HAVE_<ISA>=1` 和它的编译
|
||||
开关,并在 `simd.h` / `dispatch.cpp` 里补上 `Isa`、`isa_rank`、`isa_compiled`、
|
||||
`isa_supported` 与 `JOC_SIMD` 名字表。
|
||||
3. 验证:kernel 摘要必须与标量逐字节相同,参考渲染在每个档位上的 SHA-256 都不能变。
|
||||
@@ -2,7 +2,11 @@
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#if defined(_WIN32)
|
||||
/* EJOC_STATIC: this branch exists for building the sources directly into an application, where nothing is imported or exported. */
|
||||
#if defined(EJOC_STATIC)
|
||||
#define EJOC_API
|
||||
#define EJOC_CALL __cdecl
|
||||
#elif defined(_WIN32)
|
||||
#if defined(EJOC_BUILD_DLL)
|
||||
#define EJOC_API __declspec(dllexport)
|
||||
#else
|
||||
@@ -53,8 +57,9 @@ Fixed array layouts used by ejoc_renderer_process():
|
||||
output16 [16][1536]
|
||||
|
||||
Only objects selected by object_mask are read from the descriptor arrays.
|
||||
Sparse JOC must be rejected by the caller; this ABI accepts already dequantized
|
||||
dense matrix coefficients.
|
||||
dq carries already dequantized matrix coefficients in double precision; the
|
||||
caller performs the JOC bitstream differential decoding for both dense and
|
||||
sparse objects, so this ABI is identical for both syntaxes.
|
||||
*/
|
||||
|
||||
EJOC_API uint32_t EJOC_CALL ejoc_abi_version(void);
|
||||
@@ -63,6 +68,15 @@ EJOC_API ejoc_renderer_handle EJOC_CALL ejoc_renderer_create(void);
|
||||
EJOC_API void EJOC_CALL ejoc_renderer_destroy(ejoc_renderer_handle handle);
|
||||
EJOC_API int EJOC_CALL ejoc_renderer_reset(ejoc_renderer_handle handle);
|
||||
EJOC_API int EJOC_CALL ejoc_renderer_set_threads(ejoc_renderer_handle handle, uint32_t total_threads);
|
||||
/*
|
||||
Enables or disables the Ls/Rs band-0 21-tap DC compensation. The caller derives
|
||||
it from the JOC downmix configuration: only configurations 3 and 4 enable the
|
||||
filter. When disabled, band 0 keeps the common per-band processing (surround
|
||||
delay plus -j rotation) instead of being overwritten by the FIR. The delay line
|
||||
and DC history advance either way, so the flag may change between frames.
|
||||
Defaults to enabled when never called.
|
||||
*/
|
||||
EJOC_API int EJOC_CALL ejoc_renderer_set_dc_filter(ejoc_renderer_handle handle, uint32_t enabled);
|
||||
EJOC_API uint32_t EJOC_CALL ejoc_renderer_thread_count(ejoc_renderer_handle handle);
|
||||
EJOC_API const char* EJOC_CALL ejoc_renderer_last_error(ejoc_renderer_handle handle);
|
||||
|
||||
@@ -158,6 +172,77 @@ EJOC_API int EJOC_CALL ejoc_binaural_renderer_process(
|
||||
double output_gain,
|
||||
double* output_stereo_interleaved);
|
||||
|
||||
/*
|
||||
Native SOFA binaural renderer.
|
||||
|
||||
The handle owns the complete runtime: 64-QMF/77-hybrid analysis and synthesis,
|
||||
fifth-order ACN/N3D real spherical-harmonic direction-field evaluation,
|
||||
per-object whole-QMF-slot delay histories, six first-order image-source early
|
||||
reflections, the shared unitary-FDN late room, the 120-180 Hz LFE low-pass and
|
||||
the 961-sample latency compensation. The caller configures the filterbank
|
||||
tables, the compiled HRTF field and the room constants once, then per 512-sample
|
||||
block updates every source with ejoc_sofa_binaural_set_source() and calls
|
||||
ejoc_sofa_binaural_process(). Process returns the number of trimmed stereo
|
||||
samples written; the first 961 processed samples across calls are discarded.
|
||||
*/
|
||||
typedef void* ejoc_sofa_binaural_handle;
|
||||
|
||||
EJOC_API ejoc_sofa_binaural_handle EJOC_CALL ejoc_sofa_binaural_create(void);
|
||||
EJOC_API void EJOC_CALL ejoc_sofa_binaural_destroy(ejoc_sofa_binaural_handle handle);
|
||||
EJOC_API int EJOC_CALL ejoc_sofa_binaural_reset(ejoc_sofa_binaural_handle handle);
|
||||
EJOC_API const char* EJOC_CALL ejoc_sofa_binaural_last_error(
|
||||
ejoc_sofa_binaural_handle handle);
|
||||
EJOC_API int EJOC_CALL ejoc_sofa_binaural_configure_kernels(
|
||||
ejoc_sofa_binaural_handle handle,
|
||||
const double* qmf_analysis,
|
||||
const double* hybrid_low,
|
||||
const int16_t* hybrid_indices,
|
||||
const double* hybrid_values,
|
||||
uint32_t hybrid_count,
|
||||
const double* qmf_basis,
|
||||
const double* qmf_taps);
|
||||
EJOC_API int EJOC_CALL ejoc_sofa_binaural_configure_field(
|
||||
ejoc_sofa_binaural_handle handle,
|
||||
const double* coefficients,
|
||||
const double* delay_coefficients,
|
||||
const double* delay_bounds,
|
||||
const double* band_centers,
|
||||
double measurement_radius_m);
|
||||
EJOC_API int EJOC_CALL ejoc_sofa_binaural_configure_room(
|
||||
ejoc_sofa_binaural_handle handle,
|
||||
const double* room_dims,
|
||||
const double* listener_pos,
|
||||
const double* wall_gains,
|
||||
double speed_of_sound,
|
||||
const uint32_t* fdn_delays,
|
||||
const double* fdn_feedback,
|
||||
double damping,
|
||||
double fdn_output_gain,
|
||||
const uint32_t* allpass_delays,
|
||||
const double* allpass_gains,
|
||||
uint32_t enable_early_reflections,
|
||||
uint32_t enable_late_room);
|
||||
EJOC_API int EJOC_CALL ejoc_sofa_binaural_set_source(
|
||||
ejoc_sofa_binaural_handle handle,
|
||||
uint32_t source,
|
||||
const double* position_adm,
|
||||
uint32_t profile,
|
||||
double gain,
|
||||
uint32_t enabled,
|
||||
uint32_t special_lfe,
|
||||
uint32_t fade);
|
||||
EJOC_API int EJOC_CALL ejoc_sofa_binaural_process(
|
||||
ejoc_sofa_binaural_handle handle,
|
||||
const double* input16_interleaved,
|
||||
uint32_t sample_count,
|
||||
double output_gain,
|
||||
double* output_stereo_interleaved);
|
||||
EJOC_API int EJOC_CALL ejoc_sofa_binaural_finish(
|
||||
ejoc_sofa_binaural_handle handle,
|
||||
uint32_t flush_samples,
|
||||
double* output_stereo_interleaved,
|
||||
uint32_t capacity);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,490 @@
|
||||
/*
|
||||
* joc_core.h -- JustOneCacophony C++ Core, public ABI. Pure C.
|
||||
*
|
||||
* This header is the single authoritative definition of the Core's public
|
||||
* parameter/error surface. Frontends (thin Python CLI, joc_dump, foobar2000,
|
||||
* MPV, FFmpeg) only ever fill these POD structs and read these POD results;
|
||||
* no audio data, no internal DSP concept, and no Python type crosses this
|
||||
* boundary.
|
||||
*
|
||||
* Stability tiers
|
||||
* ---------------
|
||||
* [T1] host tier
|
||||
* joc_abi_version / joc_version_string / joc_build_info, joc_error,
|
||||
* joc_error_name / joc_error_stage, joc_last_error_detail.
|
||||
* Stable, versioned, safe for media hosts.
|
||||
*
|
||||
* [T2] bitstream / verification tier
|
||||
* joc_parse_eac3_frame, joc_parse_id14, joc_frame_params, joc_emdf_info.
|
||||
* These deliberately expose the dequantized JOC matrix coefficients so the
|
||||
* bitstream front-end can be validated bit-exactly and driven by tooling
|
||||
* (joc_dump) and A/B harnesses. Media hosts must NOT use this tier; they
|
||||
* use the task/stream API (added in later milestones).
|
||||
*
|
||||
* Conventions
|
||||
* -----------
|
||||
* * every struct's first two fields are struct_size / struct_version;
|
||||
* * all arrays are fixed size and POD, no pointers, no allocation;
|
||||
* * every function returns joc_error (JOC_OK == 0);
|
||||
* * joc_last_error_detail() returns a thread-local message that stays valid
|
||||
* until the next Core call on the same thread.
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stddef.h>
|
||||
|
||||
#define JOC_ABI_VERSION 3u
|
||||
#define JOC_FRAME_PARAMS_VERSION 1u
|
||||
#define JOC_EMDF_INFO_VERSION 1u
|
||||
#define JOC_TASK_CONFIG_VERSION 3u
|
||||
#define JOC_TASK_RESULT_VERSION 2u
|
||||
#define JOC_EVENT_VERSION 1u
|
||||
|
||||
/* JOC_STATIC: this branch exists for building the sources directly into an application, where nothing is imported or exported. */
|
||||
#if defined(JOC_STATIC)
|
||||
#define JOC_API
|
||||
#define JOC_CALL __cdecl
|
||||
#elif defined(_WIN32)
|
||||
#if defined(JOC_BUILD_DLL)
|
||||
#define JOC_API __declspec(dllexport)
|
||||
#else
|
||||
#define JOC_API __declspec(dllimport)
|
||||
#endif
|
||||
#define JOC_CALL __cdecl
|
||||
#else
|
||||
#define JOC_API __attribute__((visibility("default")))
|
||||
#define JOC_CALL
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* Fixed layout constants (single source of truth for every frontend) */
|
||||
/* ------------------------------------------------------------------ */
|
||||
enum {
|
||||
JOC_FRAME_SAMPLES = 1536, /* E-AC-3 frame = 24 * 64 */
|
||||
JOC_TIMESLOTS = 24,
|
||||
JOC_SUBBANDS = 64,
|
||||
JOC_CORE_CHANNELS = 5, /* L R C Ls Rs (JOC order) */
|
||||
JOC_MAX_CORE_CHANNELS = 7, /* dmx_config_idx 1/2/4 declare 7 */
|
||||
JOC_OUTPUT_CHANNELS = 16, /* ch0 = LFE, ch1..15 = objects */
|
||||
JOC_MAX_OBJECTS = 15,
|
||||
JOC_MAX_DPOINTS = 2,
|
||||
JOC_MAX_PARAMETER_BANDS = 23,
|
||||
JOC_MAX_EMDF_PAYLOADS = 16,
|
||||
JOC_SPEAKER_BLOCK_SAMPLES = 32,
|
||||
JOC_BINAURAL_BLOCK_SAMPLES = 512,
|
||||
JOC_BINAURAL_QMF_BANDS = 64,
|
||||
JOC_BINAURAL_HYBRID_BANDS = 77,
|
||||
JOC_QMF_HOP_SAMPLES = 64,
|
||||
JOC_BINAURAL_LATENCY_SAMPLES = 961,
|
||||
JOC_LFE_DELAY_SAMPLES = 1217
|
||||
};
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* [T1] library / error surface */
|
||||
/* ------------------------------------------------------------------ */
|
||||
typedef enum joc_error {
|
||||
JOC_OK = 0,
|
||||
JOC_ERR_INVALID_ARGUMENT,
|
||||
JOC_ERR_INVALID_CONFIG,
|
||||
JOC_ERR_OUT_OF_MEMORY,
|
||||
JOC_ERR_IO,
|
||||
JOC_ERR_UNSUPPORTED_PLATFORM,
|
||||
JOC_ERR_LIBRARY_MISSING,
|
||||
/* input / bitstream */
|
||||
JOC_ERR_INPUT_NOT_FOUND,
|
||||
JOC_ERR_INPUT_FORMAT,
|
||||
JOC_ERR_EAC3_SYNCFRAME,
|
||||
JOC_ERR_EMDF_TRANSPORT,
|
||||
JOC_ERR_EMDF_SYNTAX,
|
||||
JOC_ERR_JOC_SYNTAX,
|
||||
JOC_ERR_JOC_UNSUPPORTED_VARIANT,
|
||||
JOC_ERR_OAMD_SYNTAX,
|
||||
JOC_ERR_OAMD_UNSUPPORTED_VARIANT,
|
||||
JOC_ERR_BITSTREAM_TRUNCATED,
|
||||
JOC_ERR_BITSTREAM_PADDING,
|
||||
/* resources */
|
||||
JOC_ERR_HRTF_NOT_FOUND,
|
||||
JOC_ERR_HRTF_FORMAT,
|
||||
JOC_ERR_HRTF_VERSION,
|
||||
JOC_ERR_HRTF_HASH,
|
||||
JOC_ERR_HRTF_UNSUPPORTED_CONVENTION,
|
||||
/* rendering / output */
|
||||
JOC_ERR_LAYOUT_UNSUPPORTED,
|
||||
JOC_ERR_RENDER_FAILED,
|
||||
JOC_ERR_OUTPUT_OPEN,
|
||||
JOC_ERR_OUTPUT_WRITE,
|
||||
JOC_ERR_OUTPUT_CLIP_ABORT,
|
||||
JOC_ERR_ADM_VALIDATION,
|
||||
/* task / stream */
|
||||
JOC_ERR_CANCELLED,
|
||||
JOC_ERR_STATE,
|
||||
JOC_ERR_NOT_SUPPORTED,
|
||||
JOC_ERR_INTERNAL
|
||||
} joc_error;
|
||||
|
||||
JOC_API uint32_t JOC_CALL joc_abi_version(void);
|
||||
JOC_API const char* JOC_CALL joc_version_string(void);
|
||||
JOC_API const char* JOC_CALL joc_build_info(void);
|
||||
/* Struct sizes, so a binding can assert its layout matches the library instead of
|
||||
* assuming (mismatches are otherwise silent memory corruption). */
|
||||
JOC_API uint32_t JOC_CALL joc_event_size(void);
|
||||
JOC_API uint32_t JOC_CALL joc_task_config_size(void);
|
||||
JOC_API uint32_t JOC_CALL joc_task_result_size(void);
|
||||
JOC_API const char* JOC_CALL joc_error_name(joc_error code);
|
||||
JOC_API const char* JOC_CALL joc_error_stage(joc_error code);
|
||||
/* Thread-local structured detail for the most recent failing call. */
|
||||
JOC_API const char* JOC_CALL joc_last_error_detail(void);
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* [T2] bitstream / verification tier */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
/* One EMDF payload directory entry. bit_offset is the MSB-first bit
|
||||
* position of the payload's first byte inside the syncframe, exactly as the
|
||||
* EMDF container syntax defines it (payloads are not byte aligned in general). */
|
||||
typedef struct joc_emdf_payload_info {
|
||||
uint8_t id;
|
||||
uint8_t reserved[3];
|
||||
uint16_t sample_offset; /* EMDF outer smpoffst, 11 bits */
|
||||
uint16_t reserved2;
|
||||
uint32_t bit_offset;
|
||||
uint32_t size; /* payload bytes */
|
||||
} joc_emdf_payload_info;
|
||||
|
||||
typedef struct joc_emdf_info {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint32_t start_bit; /* container syncword bit position */
|
||||
uint32_t container_bytes; /* 4 + declared length */
|
||||
uint32_t payload_count;
|
||||
uint32_t reserved;
|
||||
joc_emdf_payload_info payloads[JOC_MAX_EMDF_PAYLOADS];
|
||||
} joc_emdf_info;
|
||||
|
||||
/* Per-object ID14/JOC descriptor plus its dequantized matrix.
|
||||
* dq[dp][ch][pb] is zero filled outside [0,n_dpoints) x [0,n_channels) x
|
||||
* [0,n_bands). Absent objects are entirely zero. */
|
||||
typedef struct joc_object_params {
|
||||
uint8_t present;
|
||||
uint8_t num_bands_idx;
|
||||
uint8_t n_bands;
|
||||
uint8_t sparse; /* 0 = dense (MTX), 1 = sparse (IDX+VEC) */
|
||||
uint8_t quant_idx; /* 0 = 96 levels, 1 = 192 levels */
|
||||
uint8_t slope_idx; /* 0 = interpolate, 1 = step at offset_ts */
|
||||
uint8_t num_dpoints_bits;
|
||||
uint8_t n_dpoints;
|
||||
uint8_t offset_ts[JOC_MAX_DPOINTS];
|
||||
uint8_t reserved[2];
|
||||
double dq[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS];
|
||||
} joc_object_params;
|
||||
|
||||
typedef struct joc_frame_params {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint8_t dmx_config_idx;
|
||||
uint8_t num_objects_bits;
|
||||
uint8_t ext_config_idx;
|
||||
uint8_t n_objects;
|
||||
uint8_t n_channels; /* 5 or 7 */
|
||||
uint8_t clipgain_x_bits;
|
||||
uint8_t clipgain_y_bits;
|
||||
uint8_t reserved;
|
||||
uint32_t seq_count; /* 10-bit JOC sequence counter, parsed only */
|
||||
uint32_t present_mask; /* bit i == object i present */
|
||||
uint32_t data_end_bits; /* bit position just after joc_data */
|
||||
uint32_t trailing_bits; /* bits left after joc_data (padding/ext) */
|
||||
uint8_t tail_bytes[8]; /* first up to 8 trailing bytes, for A/B */
|
||||
double clipgain; /* 1 + (y/32) * 2^(x-4), bit-exact */
|
||||
joc_object_params objects[JOC_MAX_OBJECTS];
|
||||
} joc_frame_params;
|
||||
|
||||
/* Parse an EMDF ID14 (JOC) payload. */
|
||||
JOC_API joc_error JOC_CALL joc_parse_id14(const uint8_t* payload, size_t payload_size,
|
||||
joc_frame_params* out_params);
|
||||
|
||||
/* Locate the contiguous JOC EMDF container inside one E-AC-3 syncframe and
|
||||
* parse its ID14 payload. out_emdf may be NULL. */
|
||||
JOC_API joc_error JOC_CALL joc_parse_eac3_frame(const uint8_t* frame, size_t frame_size,
|
||||
joc_frame_params* out_params,
|
||||
joc_emdf_info* out_emdf);
|
||||
|
||||
/* Copy one payload's bytes out of a syncframe (MSB-first bit extraction, which
|
||||
* equals a memcpy for byte-aligned containers). out_size receives the payload
|
||||
* byte count; pass out == NULL to query only the size. */
|
||||
JOC_API joc_error JOC_CALL joc_extract_payload(const uint8_t* frame, size_t frame_size,
|
||||
const joc_emdf_payload_info* payload,
|
||||
uint8_t* out, size_t out_capacity,
|
||||
size_t* out_size);
|
||||
|
||||
/* E-AC-3 syncframe traversal: given the offset of a frame start, report its
|
||||
* byte length so a host can walk a bare E-AC-3 stream without duplicating
|
||||
* frmsiz logic. Rejects resynchronisation (no silent recovery). */
|
||||
JOC_API joc_error JOC_CALL joc_eac3_frame_bytes(const uint8_t* data, size_t size,
|
||||
size_t offset, size_t* out_frame_bytes);
|
||||
|
||||
/* Strict trailing-bit check for the JOC payload (A/B robustness corpus).
|
||||
* Returns JOC_ERR_BITSTREAM_PADDING when more than 7 bits are left over or the
|
||||
* leftover bits are not zero. */
|
||||
JOC_API joc_error JOC_CALL joc_check_id14_padding(const uint8_t* payload, size_t payload_size,
|
||||
uint32_t* out_trailing_bits);
|
||||
|
||||
/* ================================================================== */
|
||||
/* [T1] host tier: task, telemetry and control */
|
||||
/* ================================================================== */
|
||||
|
||||
/* ---- 1. events ---------------------------------------------------- */
|
||||
/*
|
||||
* Events carry state only: frame/sample counters, stage, progress, statistics,
|
||||
* warnings, errors and paths. Audio never travels through an event; a fixed
|
||||
* size POD with no pointers keeps that enforceable (see the static assertion in
|
||||
* the implementation).
|
||||
*/
|
||||
typedef enum joc_event_type {
|
||||
JOC_EV_TASK_STARTED = 0x0001,
|
||||
JOC_EV_TASK_STATE_CHANGED = 0x0002,
|
||||
JOC_EV_TASK_COMPLETED = 0x0003,
|
||||
JOC_EV_TASK_FAILED = 0x0004,
|
||||
JOC_EV_TASK_CANCELLED = 0x0005,
|
||||
JOC_EV_INPUT_OPENED = 0x0101,
|
||||
JOC_EV_METADATA_INDEXED = 0x0103,
|
||||
JOC_EV_HRTF_LOADED = 0x0110,
|
||||
JOC_EV_RENDERER_INITIALIZED = 0x0120,
|
||||
JOC_EV_PROGRESS = 0x0201,
|
||||
JOC_EV_STAGE_CHANGED = 0x0202,
|
||||
JOC_EV_JOC_FRAME_STATS = 0x0301,
|
||||
JOC_EV_OAMD_STATS = 0x0302,
|
||||
JOC_EV_OUTPUT_STATS = 0x0303,
|
||||
JOC_EV_OUTPUT_OPENED = 0x0401,
|
||||
JOC_EV_OUTPUT_FORMAT_DECIDED = 0x0402,
|
||||
JOC_EV_OUTPUT_FINALIZED = 0x0403,
|
||||
JOC_EV_LOG = 0x0501,
|
||||
JOC_EV_WARNING = 0x0502,
|
||||
JOC_EV_ERROR = 0x0503
|
||||
} joc_event_type;
|
||||
|
||||
typedef enum joc_stage {
|
||||
JOC_STAGE_IDLE = 0,
|
||||
JOC_STAGE_INPUT = 1,
|
||||
JOC_STAGE_METADATA = 2,
|
||||
JOC_STAGE_DECODE = 3,
|
||||
JOC_STAGE_JOC = 4,
|
||||
JOC_STAGE_RENDER = 5,
|
||||
JOC_STAGE_OUTPUT = 6,
|
||||
JOC_STAGE_DONE = 7
|
||||
} joc_stage;
|
||||
|
||||
enum { JOC_LOG_TRACE = 0, JOC_LOG_DEBUG = 1, JOC_LOG_INFO = 2, JOC_LOG_NOTICE = 3,
|
||||
JOC_LOG_WARNING = 4, JOC_LOG_ERROR = 5, JOC_LOG_FATAL = 6 };
|
||||
|
||||
typedef struct joc_event {
|
||||
uint32_t struct_size;
|
||||
uint32_t type;
|
||||
uint64_t sequence;
|
||||
uint64_t timestamp_us;
|
||||
uint64_t current_frame;
|
||||
uint64_t total_frames;
|
||||
uint64_t current_sample;
|
||||
uint64_t total_samples;
|
||||
uint32_t stage;
|
||||
uint32_t backend;
|
||||
double progress; /* 0..1, -1 = unknown */
|
||||
double elapsed_seconds;
|
||||
double realtime_factor;
|
||||
uint64_t output_samples;
|
||||
uint64_t output_bytes;
|
||||
double output_duration_seconds;
|
||||
joc_error error_code;
|
||||
uint32_t log_level;
|
||||
char stage_name[32];
|
||||
char message[256];
|
||||
} joc_event;
|
||||
|
||||
typedef void (JOC_CALL *joc_event_fn)(void* user, const joc_event* event);
|
||||
|
||||
typedef struct joc_event_sink {
|
||||
uint32_t struct_size;
|
||||
joc_event_fn callback;
|
||||
void* user;
|
||||
uint32_t min_type; /* 0 = no filter */
|
||||
uint32_t max_type; /* 0 = no filter */
|
||||
} joc_event_sink;
|
||||
|
||||
/* ---- 2. cancellation ---------------------------------------------- */
|
||||
typedef struct joc_cancel_token joc_cancel_token;
|
||||
|
||||
JOC_API joc_cancel_token* JOC_CALL joc_cancel_token_create(void);
|
||||
JOC_API void JOC_CALL joc_cancel_token_request(joc_cancel_token* token);
|
||||
JOC_API int32_t JOC_CALL joc_cancel_token_is_requested(const joc_cancel_token* token);
|
||||
JOC_API void JOC_CALL joc_cancel_token_destroy(joc_cancel_token* token);
|
||||
|
||||
/* ---- 3. task configuration and results ---------------------------- */
|
||||
typedef enum joc_operation {
|
||||
JOC_OP_ADM_BWF = 0,
|
||||
JOC_OP_SPEAKER = 1,
|
||||
JOC_OP_BINAURAL = 2
|
||||
} joc_operation;
|
||||
|
||||
typedef enum joc_output_format {
|
||||
JOC_FORMAT_FLOAT32 = 0,
|
||||
JOC_FORMAT_PCM24 = 1
|
||||
} joc_output_format;
|
||||
|
||||
typedef enum joc_binaural_mode {
|
||||
JOC_BINAURAL_OFF = 0,
|
||||
JOC_BINAURAL_NEAR = 1,
|
||||
JOC_BINAURAL_FAR = 2,
|
||||
JOC_BINAURAL_MID = 3
|
||||
} joc_binaural_mode;
|
||||
|
||||
typedef enum joc_trajectory_mode {
|
||||
JOC_TRAJECTORY_COMPACT = 0,
|
||||
JOC_TRAJECTORY_DENSE64 = 1
|
||||
} joc_trajectory_mode;
|
||||
|
||||
/* What to do when an int24 WAV would clip (peak outside [-1, 1]). Mirrors the
|
||||
* reference CLI's --clip-action; ADM BWF output is always int24 and does not
|
||||
* consult this because no alternative format exists there. */
|
||||
typedef enum joc_clip_action {
|
||||
JOC_CLIP_ASK = 0, /* prompt on stdin; an error when stdin is not a terminal */
|
||||
JOC_CLIP_CONTINUE = 1, /* write int24, truncating out-of-range values */
|
||||
JOC_CLIP_FLOAT32 = 2, /* switch the output to float32 */
|
||||
JOC_CLIP_ABORT = 3 /* fail the task */
|
||||
} joc_clip_action;
|
||||
|
||||
/* Compiled-HRTF cache policy for a SOFA input; the .jochrtf itself is an
|
||||
* internal artifact of the compile step. */
|
||||
typedef enum joc_hrtf_cache_policy {
|
||||
JOC_HRTF_CACHE_NONE = 0, /* compile and discard */
|
||||
JOC_HRTF_CACHE_MEMORY = 1, /* compile and keep in this process (default) */
|
||||
JOC_HRTF_CACHE_DISK = 2 /* compile, reuse and write <cache_dir>/<name>.jochrtf */
|
||||
} joc_hrtf_cache_policy;
|
||||
|
||||
enum { JOC_TASK_F_SKIP_SHA256 = 1u, JOC_TASK_F_KEEP_INTERMEDIATE = 2u,
|
||||
JOC_TASK_F_QUIET = 4u, JOC_TASK_F_METADATA_ONLY = 8u };
|
||||
|
||||
typedef struct joc_task_config {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
|
||||
/* input */
|
||||
const char* input_path; /* .eac3/.ec3/.m4a/... */
|
||||
const char* ffmpeg_path; /* NULL = "ffmpeg" from PATH */
|
||||
const char* bed_path; /* NULL = decode the core PCM with ffmpeg */
|
||||
const char* work_dir; /* NULL = a temporary directory */
|
||||
double eac3_drc_scale; /* 0 = DRC off (reference default) */
|
||||
int32_t eac3_target_level; /* -31..0, 0 = not applied */
|
||||
|
||||
/* output */
|
||||
uint32_t operation;
|
||||
const char* output_path;
|
||||
uint32_t output_format; /* requested format for speaker/binaural */
|
||||
uint32_t flags;
|
||||
uint32_t clip_action; /* joc_clip_action; JOC_CLIP_ASK by default */
|
||||
|
||||
/* rendering */
|
||||
const char* speaker_layout_name;
|
||||
uint32_t speaker_metadata_offset; /* default 1473 */
|
||||
uint32_t binaural_mode;
|
||||
const char* hrtf_path; /* .jochrtf */
|
||||
const char* kernels_path; /* rosella_kernels.npz */
|
||||
double binaural_tail_seconds; /* default 5.0 */
|
||||
double binaural_tail_threshold; /* binaural only; 0 disables trimming */
|
||||
uint32_t binaural_chunk_frames; /* accepted for CLI parity; no effect */
|
||||
|
||||
/* ADM */
|
||||
uint32_t adm_binaural_mode; /* DBMD segment 10 encoding */
|
||||
uint32_t trajectory_mode;
|
||||
|
||||
/* generic */
|
||||
uint32_t object_delay_samples; /* default 1473 */
|
||||
double gain_db; /* default 0 */
|
||||
uint64_t duration_frames; /* 0 = the whole stream */
|
||||
uint32_t progress_interval_frames; /* default 1000 */
|
||||
uint32_t native_threads; /* 0 = automatic */
|
||||
|
||||
/* diagnostics */
|
||||
uint32_t print_metadata; /* 0 none, 1 summary, 2 per frame */
|
||||
const char* metadata_json_path; /* NULL = no JSON summary */
|
||||
|
||||
joc_cancel_token* cancel; /* optional */
|
||||
|
||||
/* Binaural HRTF input (config version 2): when hrtf_sofa_path is set the
|
||||
* library compiles it with hrtf_cache_policy / hrtf_cache_dir / hrtf_radius_m
|
||||
* and hrtf_path is unused. hrtf_path stays the advanced override that reads
|
||||
* a .jochrtf directly. */
|
||||
const char* hrtf_sofa_path; /* SOFA SimpleFreeFieldHRIR input */
|
||||
const char* personalized_headphone_path; /* Rosella .personalized_headphone input */
|
||||
const char* hrtf_cache_dir; /* disk policy directory */
|
||||
uint32_t hrtf_cache_policy; /* joc_hrtf_cache_policy */
|
||||
uint32_t reserved0;
|
||||
double hrtf_radius_m; /* SOFA measurement-radius shell */
|
||||
} joc_task_config;
|
||||
|
||||
typedef enum joc_task_status {
|
||||
JOC_TASK_OK = 0,
|
||||
JOC_TASK_FAILED = 1,
|
||||
JOC_TASK_CANCELLED = 2
|
||||
} joc_task_status;
|
||||
|
||||
typedef struct joc_task_result {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint32_t status;
|
||||
uint32_t error_code;
|
||||
char error_message[512];
|
||||
char error_stage[32];
|
||||
uint64_t input_frames;
|
||||
uint64_t output_samples;
|
||||
double duration_sec;
|
||||
uint32_t output_format_actual;
|
||||
double output_peak;
|
||||
uint64_t output_over_unity_values;
|
||||
uint64_t output_file_bytes;
|
||||
char output_sha256[65]; /* empty when skipped */
|
||||
uint64_t oamd_payloads;
|
||||
uint64_t oamd_transitions;
|
||||
double t_decode_bed;
|
||||
double t_render;
|
||||
double t_write;
|
||||
double t_total;
|
||||
double t_render_dsp; /* speaker/binaural DSP calls only; 0 for ADM */
|
||||
double t_write_file; /* disk writes including the finalize; >= t_write */
|
||||
} joc_task_result;
|
||||
|
||||
typedef struct joc_validation_issue {
|
||||
joc_error code;
|
||||
uint32_t severity; /* 0 = info, 1 = warning, 2 = error */
|
||||
char field[48];
|
||||
char message[256];
|
||||
} joc_validation_issue;
|
||||
|
||||
/* ---- 4. entry points ---------------------------------------------- */
|
||||
/* Fills `issues` (up to `capacity`) and reports how many were produced.
|
||||
* Returns JOC_OK when no *error*-severity issue was found. */
|
||||
JOC_API joc_error JOC_CALL joc_task_validate(const joc_task_config* config,
|
||||
joc_validation_issue* issues, uint32_t capacity,
|
||||
uint32_t* count);
|
||||
|
||||
/* Runs the whole task on the calling thread. `sink` may be NULL; `out` may be
|
||||
* NULL. On cancellation the partial output is removed and JOC_ERR_CANCELLED is
|
||||
* returned with out->status = JOC_TASK_CANCELLED. */
|
||||
JOC_API joc_error JOC_CALL joc_task_execute(const joc_task_config* config,
|
||||
const joc_event_sink* sink, joc_task_result* out);
|
||||
|
||||
/* Serialises the stable subset of the result as JSON (no environment fields -
|
||||
* those belong to the frontend). `needed` receives the required size including
|
||||
* the terminator; a NULL buffer queries only the size. */
|
||||
JOC_API joc_error JOC_CALL joc_task_result_to_json(const joc_task_result* result, char* buffer,
|
||||
size_t capacity, size_t* needed);
|
||||
|
||||
/* The streaming (push/pull) tier lives in "joc_stream.h" so an embedder can
|
||||
* include the narrow surface without the bitstream/verification tier. */
|
||||
|
||||
#ifdef __cplusplus
|
||||
} /* extern "C" */
|
||||
#endif
|
||||
@@ -0,0 +1,136 @@
|
||||
/*
|
||||
* joc_stream.h -- streaming (push/pull) interface of the JustOneCacophony core.
|
||||
*
|
||||
* This is the narrow, embedder-facing surface: a player or decoder component
|
||||
* includes only this header. It is the same shared library as joc_core.h, split
|
||||
* so an integrator never has to see the bitstream/verification tier.
|
||||
*
|
||||
* A stream is a stateful instance for hosts that cannot wait for a whole file:
|
||||
* the caller feeds E-AC-3 bytes and the core PCM of the same frames (or already
|
||||
* rebuilt objects16) and pulls rendered PCM as soon as it is available. It
|
||||
* mirrors the two library shapes a decoder library normally offers: this
|
||||
* push/pull form for host-owned I/O, and joc_task_execute() in joc_core.h for
|
||||
* library-owned file I/O.
|
||||
*
|
||||
* Contract:
|
||||
* - all state is instance-private, so several streams coexist;
|
||||
* - a stream is NOT thread safe: push and pull must come from one thread;
|
||||
* - rendering is stateful (matrix interpolation, gain ramps, room tail), so a
|
||||
* new position on the timeline requires decoding to continue from the start
|
||||
* of the stream; there is no seek in this version;
|
||||
* - the kernel latency is 961 samples for speaker/binaural output: the first
|
||||
* pull after two 512-sample blocks, and flush() drains the binaural tail.
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include "joc_core.h"
|
||||
|
||||
/* JOC_STATIC (building the sources directly into an application): joc_core.h above already installs the empty JOC_API. */
|
||||
#if defined(JOC_STATIC) && !defined(JOC_API)
|
||||
#define JOC_API
|
||||
#define JOC_CALL __cdecl
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
typedef struct joc_stream joc_stream;
|
||||
|
||||
typedef enum joc_stream_input {
|
||||
JOC_STREAM_IN_EAC3 = 0, /* bare E-AC-3 syncframes (the metadata stream) */
|
||||
JOC_STREAM_IN_PCM_OBJECTS16 = 1, /* 16-channel objects16, decoded by the host */
|
||||
JOC_STREAM_IN_CORE_PCM = 3 /* the 5.1 core PCM of the pushed E-AC-3 frames */
|
||||
} joc_stream_input;
|
||||
|
||||
typedef enum joc_stream_output {
|
||||
JOC_STREAM_OUT_PCM_OBJECTS16 = 0, /* planar [16][samples] float32 */
|
||||
JOC_STREAM_OUT_SPEAKER = 1, /* interleaved [samples][channels] f32 */
|
||||
JOC_STREAM_OUT_BINAURAL = 2 /* interleaved [samples][2] f32 */
|
||||
} joc_stream_output;
|
||||
|
||||
typedef struct joc_stream_config {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint32_t input;
|
||||
uint32_t output;
|
||||
const char* speaker_layout_name;
|
||||
uint32_t speaker_metadata_offset; /* default 1473 */
|
||||
uint32_t binaural_mode; /* near|mid|far; 0 means mid */
|
||||
const char* hrtf_path; /* binaural only */
|
||||
const char* kernels_path; /* binaural only */
|
||||
double binaural_tail_seconds; /* default 5.0 */
|
||||
uint32_t object_delay_samples; /* default 1473 */
|
||||
double gain_db; /* default 0 */
|
||||
uint32_t native_threads;
|
||||
uint32_t reserved;
|
||||
/* Binaural HRTF input, the same three shapes joc_task_config accepts: when
|
||||
* hrtf_sofa_path is set the library compiles it with hrtf_cache_policy /
|
||||
* hrtf_cache_dir / hrtf_radius_m and hrtf_path is unused; when
|
||||
* personalized_headphone_path is set the Rosella runtime renders instead.
|
||||
* hrtf_path stays the fallback/advanced input that reads a .jochrtf directly. */
|
||||
const char* hrtf_sofa_path; /* SOFA SimpleFreeFieldHRIR input */
|
||||
const char* personalized_headphone_path; /* Rosella .personalized_headphone input */
|
||||
const char* hrtf_cache_dir; /* disk cache directory for SOFA compilation */
|
||||
uint32_t hrtf_cache_policy; /* joc_hrtf_cache_policy: 0 none, 1 memory, 2 disk */
|
||||
double hrtf_radius_m; /* SOFA measurement-radius shell, default 1.0 */
|
||||
} joc_stream_config;
|
||||
|
||||
typedef struct joc_stream_buffer {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint32_t kind; /* which joc_stream_input/output this buffer carries */
|
||||
uint32_t channels;
|
||||
uint32_t sample_rate;
|
||||
uint32_t sample_count; /* in: capacity / out: produced (per channel) */
|
||||
uint32_t byte_count; /* in: capacity / out: consumed or produced bytes */
|
||||
uint32_t reserved;
|
||||
const uint8_t* bytes; /* EAC3 input */
|
||||
const float* pcm; /* PCM input */
|
||||
uint8_t* out_bytes; /* reserved for encoded outputs */
|
||||
float* out_pcm; /* PCM output */
|
||||
} joc_stream_buffer;
|
||||
|
||||
typedef struct joc_stream_status_info {
|
||||
uint32_t struct_size;
|
||||
uint32_t struct_version;
|
||||
uint64_t frames_in;
|
||||
uint64_t frames_out;
|
||||
uint64_t samples_in;
|
||||
uint64_t samples_out;
|
||||
uint64_t bytes_in;
|
||||
uint64_t buffered_samples; /* rendered but not yet pulled, per channel */
|
||||
uint64_t oamd_payloads;
|
||||
uint64_t oamd_transitions;
|
||||
uint32_t output_channels;
|
||||
uint32_t ended; /* 1 after flush() */
|
||||
} joc_stream_status_info;
|
||||
|
||||
JOC_API joc_error JOC_CALL joc_stream_create(const joc_stream_config* config, joc_stream** out);
|
||||
|
||||
/* Feeds one buffer. Kind selects the path: EAC3 bytes, the core PCM of those
|
||||
* frames, or objects16. Consumed counts are reported so a caller can resume
|
||||
* from a partial push. */
|
||||
JOC_API joc_error JOC_CALL joc_stream_push(joc_stream* stream, const joc_stream_buffer* input,
|
||||
uint32_t* consumed_samples, uint32_t* consumed_bytes);
|
||||
|
||||
/* Copies as many rendered samples as fit into `output` (interleaved, or planar
|
||||
* for PCM_OBJECTS16) and reports how many were produced; 0 means "push more". */
|
||||
JOC_API joc_error JOC_CALL joc_stream_pull(joc_stream* stream, joc_stream_buffer* output,
|
||||
uint32_t* produced_samples);
|
||||
|
||||
/* Marks the end of input and drains whatever the renderer still holds (the
|
||||
* binaural room tail); pull the remaining samples afterwards. */
|
||||
JOC_API joc_error JOC_CALL joc_stream_flush(joc_stream* stream);
|
||||
|
||||
/* Returns the instance to its initial state (kernel, ramps, timeline, room,
|
||||
* parser, counters) so the same stream can be reused for another pass. */
|
||||
JOC_API joc_error JOC_CALL joc_stream_reset(joc_stream* stream);
|
||||
|
||||
JOC_API joc_error JOC_CALL joc_stream_status(const joc_stream* stream,
|
||||
joc_stream_status_info* out);
|
||||
JOC_API joc_error JOC_CALL joc_stream_destroy(joc_stream* stream);
|
||||
|
||||
#ifdef __cplusplus
|
||||
} /* extern "C" */
|
||||
#endif
|
||||
@@ -1,695 +0,0 @@
|
||||
"""JustOneCacophony 的 E-AC-3 JOC 命令行入口。"""
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
from pathlib import Path
|
||||
import platform
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
|
||||
PROJECT_DIR = Path(__file__).resolve().parent
|
||||
SOURCE_DIR = PROJECT_DIR / "src"
|
||||
if str(SOURCE_DIR) not in sys.path:
|
||||
sys.path.insert(0, str(SOURCE_DIR))
|
||||
|
||||
import numpy as np
|
||||
|
||||
import adm_assemble
|
||||
import adm_atmos
|
||||
from adm_validate import validate
|
||||
from metadata import DirectPayloadIndex, PayloadIndex, write_summary
|
||||
import oamd_tracks
|
||||
from renderer import JocRenderer
|
||||
from native_renderer import NativeBackendUnavailable, NativeJocRenderer
|
||||
from binaural_renderer import (
|
||||
ROSSELLA_BLOCK_SAMPLES,
|
||||
ROSSELLA_LATENCY_SAMPLES,
|
||||
RosellaBinauralRenderer,
|
||||
resolve_personalized_headphone,
|
||||
)
|
||||
from speaker_backend import create_speaker_renderer
|
||||
from speaker_layouts import (SPEAKER_LAYOUT_CHOICES, get_speaker_layout,
|
||||
speaker_layout_display_name)
|
||||
from speaker_wav import BinauralPcmSpool, SpeakerPcmSpool, write_pcm_wav
|
||||
from variant_error import UnsupportedVariantError, write_variant_report
|
||||
|
||||
|
||||
RATE = 48000
|
||||
FRAME_SAMPLES = 1536
|
||||
DEFAULT_OUTPUT_DIR = PROJECT_DIR / "output"
|
||||
|
||||
|
||||
def resolve_output(source, requested=None, speaker_layout=None, *, binaural=False):
|
||||
"""解析成品路径;未指定时使用项目内的 ``output`` 目录。"""
|
||||
source = Path(source)
|
||||
if requested is not None:
|
||||
target = Path(requested)
|
||||
elif speaker_layout is not None:
|
||||
target = DEFAULT_OUTPUT_DIR / f"{source.stem}.{speaker_layout}.wav"
|
||||
elif binaural:
|
||||
target = DEFAULT_OUTPUT_DIR / f"{source.stem}.binaural.wav"
|
||||
else:
|
||||
target = DEFAULT_OUTPUT_DIR / (source.stem + ".adm.wav")
|
||||
return target.expanduser().resolve()
|
||||
|
||||
|
||||
def executable(value, name):
|
||||
path = shutil.which(value) if value else None
|
||||
if path is None and value and Path(value).is_file():
|
||||
path = str(Path(value).resolve())
|
||||
if path is None:
|
||||
raise FileNotFoundError(f"找不到 {name}: {value!r}")
|
||||
return path
|
||||
|
||||
|
||||
def run(command, label):
|
||||
print(f"[{label}]", flush=True)
|
||||
result = subprocess.run(command, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE,
|
||||
text=True, encoding="utf-8", errors="replace")
|
||||
if result.returncode:
|
||||
tail = result.stderr[-4000:]
|
||||
raise RuntimeError(f"{label} 失败(exit {result.returncode})\n{tail}")
|
||||
|
||||
|
||||
def timed_call(timings, name, function, *args, **kwargs):
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
return function(*args, **kwargs)
|
||||
finally:
|
||||
timings[name] = time.perf_counter() - started
|
||||
|
||||
|
||||
def extract_eac3(ffmpeg, source, target):
|
||||
if source.suffix.lower() in (".eac3", ".ec3"):
|
||||
return source
|
||||
run([ffmpeg, "-hide_banner", "-loglevel", "error", "-y", "-i", str(source),
|
||||
"-map", "0:a:0", "-vn", "-c:a", "copy", "-f", "eac3", str(target)],
|
||||
"FFmpeg 提取 E-AC-3")
|
||||
return target
|
||||
|
||||
|
||||
def decode_core(ffmpeg, eac3, target, duration_sec=None):
|
||||
# 5.1(side) 的 f32le 顺序为 FL FR FC LFE SL SR;JOC 使用其中 0,1,2,4,5。
|
||||
command = [ffmpeg, "-hide_banner", "-loglevel", "error", "-y", "-i", str(eac3),
|
||||
"-map", "0:a:0", "-vn"]
|
||||
if duration_sec is not None:
|
||||
command.extend(["-t", f"{duration_sec:.9f}"])
|
||||
command.extend(["-ac", "6", "-ar", str(RATE),
|
||||
"-c:a", "pcm_f32le", "-f", "f32le", str(target)])
|
||||
run(command, "FFmpeg 解码核心 5.1 PCM")
|
||||
return target
|
||||
|
||||
|
||||
def sha256(path):
|
||||
digest = hashlib.sha256()
|
||||
with Path(path).open("rb") as fp:
|
||||
for block in iter(lambda: fp.read(16 << 20), b""):
|
||||
digest.update(block)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def choose_pcm_output_format(requested_format, clip_action, peak, clipped_values,
|
||||
*, input_func=input, interactive=None):
|
||||
"""Resolve int24 clipping interactively or through an explicit policy."""
|
||||
if requested_format != "int24" or clipped_values == 0:
|
||||
return requested_format
|
||||
print(
|
||||
f"[clip] int24 将发生削波:peak={peak:.9g},超出 [-1,1] 的样本值={clipped_values}",
|
||||
file=sys.stderr, flush=True)
|
||||
action = clip_action
|
||||
if action == "ask":
|
||||
if interactive is None:
|
||||
interactive = bool(getattr(sys.stdin, "isatty", lambda: False)())
|
||||
if not interactive:
|
||||
raise RuntimeError(
|
||||
"检测到 int24 削波,但当前不是交互终端;请使用 "
|
||||
"--clip-action continue、--clip-action float32 或 --clip-action abort")
|
||||
while True:
|
||||
answer = input_func(
|
||||
"继续写 int24 并截断 [i] / 改为 float32 [f,默认] / 取消 [a]:"
|
||||
).strip().lower()
|
||||
if answer in ("", "f", "float", "float32"):
|
||||
action = "float32"
|
||||
break
|
||||
if answer in ("i", "int", "int24", "c", "continue"):
|
||||
action = "continue"
|
||||
break
|
||||
if answer in ("a", "abort", "q", "quit", "n", "no"):
|
||||
action = "abort"
|
||||
break
|
||||
print("请输入 i、f 或 a。", file=sys.stderr, flush=True)
|
||||
if action == "continue":
|
||||
print("[clip] 将继续写 int24,超范围值会截断到 [-1,1]。", flush=True)
|
||||
return "int24"
|
||||
if action == "float32":
|
||||
print("[clip] 已切换为 float32 WAV,不执行截断。", flush=True)
|
||||
return "float32"
|
||||
if action == "abort":
|
||||
raise RuntimeError("用户因 int24 削波取消输出")
|
||||
raise ValueError(f"未知 clip action: {action}")
|
||||
|
||||
|
||||
# Backward-compatible public name used by existing tests and callers.
|
||||
choose_speaker_output_format = choose_pcm_output_format
|
||||
|
||||
|
||||
def resolve_metadata(args, eac3, temp_dir):
|
||||
if args.metadata_dir:
|
||||
directory = Path(args.metadata_dir).resolve()
|
||||
return PayloadIndex(directory), "sidecar", directory
|
||||
if args.metadata_backend == "sidecar":
|
||||
raise ValueError("metadata-backend=sidecar 时必须提供 --metadata-dir")
|
||||
cache_dir = (args.metadata_cache.expanduser().resolve()
|
||||
if args.metadata_cache else None)
|
||||
max_frames = (math.ceil(args.duration * RATE / FRAME_SAMPLES)
|
||||
if args.duration is not None else None)
|
||||
index = DirectPayloadIndex.from_eac3(
|
||||
eac3, max_frames=max_frames, cache_dir=cache_dir)
|
||||
return index, "python-emdf-memory", cache_dir
|
||||
|
||||
|
||||
def variant_call(output, source, function, *args, **kwargs):
|
||||
"""执行一个阶段;遇到未知变体时在目标文件旁写结构化报告。"""
|
||||
try:
|
||||
return function(*args, **kwargs)
|
||||
except UnsupportedVariantError as exc:
|
||||
report_path = Path(str(output) + ".variant-error.json")
|
||||
write_variant_report(report_path, exc, input_path=source, output_path=output)
|
||||
print(f"[VARIANT] {exc}", file=sys.stderr, flush=True)
|
||||
print(f"[VARIANT] 维修报告: {report_path}", file=sys.stderr, flush=True)
|
||||
raise
|
||||
|
||||
|
||||
def create_renderer(backend, gain, native_library=None, native_threads=None):
|
||||
"""选择整帧 DSP 后端;auto 优先使用 lib 中当前平台的原生构建。"""
|
||||
if backend in ("auto", "native"):
|
||||
try:
|
||||
decoder = NativeJocRenderer(
|
||||
output_scale=gain, library_path=native_library, threads=native_threads)
|
||||
info = {
|
||||
"name": "native",
|
||||
"library": str(decoder.library_path),
|
||||
"build": decoder.build_info,
|
||||
"threads": decoder.threads,
|
||||
}
|
||||
print(f"[backend] native: {info['build']} threads={info['threads']} "
|
||||
f"({info['library']})", flush=True)
|
||||
return decoder, info
|
||||
except (NativeBackendUnavailable, OSError) as exc:
|
||||
print(f"[backend] native unavailable, falling back to Python: {exc}", flush=True)
|
||||
decoder = JocRenderer(output_scale=gain)
|
||||
info = {"name": "python", "library": None, "build": None, "threads": None}
|
||||
print("[backend] python/numpy", flush=True)
|
||||
return decoder, info
|
||||
|
||||
|
||||
def render(index, bed_path, frame_count, raw_path, gain, progress_every,
|
||||
backend="auto", native_library=None, native_threads=None, frame_sink=None,
|
||||
speaker_renderer=None, speaker_sink=None, speaker_metadata_offset=1473,
|
||||
binaural_renderer=None, binaural_sink=None, binaural_metadata_offset=1473,
|
||||
raw_scale=1.0):
|
||||
values = np.memmap(bed_path, dtype=np.float32, mode="r")
|
||||
frame_width = FRAME_SAMPLES * 6
|
||||
if values.size % frame_width:
|
||||
raise ValueError(f"FFmpeg PCM 长度不是 1536×6 的整数倍: {values.size}")
|
||||
bed = values.reshape(-1, FRAME_SAMPLES, 6)
|
||||
if len(bed) < frame_count:
|
||||
raise ValueError(f"PCM 只有 {len(bed)} 帧,元数据需要 {frame_count} 帧")
|
||||
output = (np.memmap(raw_path, dtype=np.float32, mode="w+",
|
||||
shape=(frame_count, FRAME_SAMPLES, 16))
|
||||
if raw_path is not None else None)
|
||||
decoder, backend_info = create_renderer(backend, gain, native_library, native_threads)
|
||||
started = time.perf_counter()
|
||||
dsp_seconds = 0.0
|
||||
adm_stream_seconds = 0.0
|
||||
raw_write_seconds = 0.0
|
||||
speaker_render_seconds = 0.0
|
||||
speaker_write_seconds = 0.0
|
||||
binaural_render_seconds = 0.0
|
||||
binaural_write_seconds = 0.0
|
||||
elapsed = 0.0
|
||||
try:
|
||||
for frame_number, row in enumerate(index.rows[:frame_count]):
|
||||
bed6 = np.asarray(bed[frame_number], dtype=np.float32)
|
||||
subs = index.subpayloads(row)
|
||||
stage = time.perf_counter()
|
||||
pcm16, _ = decoder.render_subpayloads(
|
||||
subs, bed6[:, [0, 1, 2, 4, 5]].T, bed6[:, 3])
|
||||
dsp_seconds += time.perf_counter() - stage
|
||||
if output is not None:
|
||||
stage = time.perf_counter()
|
||||
output[frame_number] = np.multiply(
|
||||
pcm16.T, np.float32(raw_scale), dtype=np.float32)
|
||||
raw_write_seconds += time.perf_counter() - stage
|
||||
if frame_sink is not None:
|
||||
stage = time.perf_counter()
|
||||
frame_sink.write_frame(pcm16)
|
||||
adm_stream_seconds += time.perf_counter() - stage
|
||||
if speaker_renderer is not None:
|
||||
stage = time.perf_counter()
|
||||
speaker_pcm = speaker_renderer.render_frame(
|
||||
pcm16.T, subs.get(11), speaker_metadata_offset)
|
||||
speaker_render_seconds += time.perf_counter() - stage
|
||||
stage = time.perf_counter()
|
||||
speaker_sink.write_frame(speaker_pcm)
|
||||
speaker_write_seconds += time.perf_counter() - stage
|
||||
if binaural_renderer is not None:
|
||||
payload = subs.get(11)
|
||||
outer_offset = (
|
||||
index.subpayload_sample_offset(row, 11)
|
||||
if payload is not None and hasattr(index, "subpayload_sample_offset")
|
||||
else 0
|
||||
)
|
||||
stage = time.perf_counter()
|
||||
binaural_pcm = binaural_renderer.render_frame(
|
||||
pcm16.T, payload, binaural_metadata_offset,
|
||||
outer_sample_offset=outer_offset)
|
||||
binaural_render_seconds += time.perf_counter() - stage
|
||||
if len(binaural_pcm):
|
||||
stage = time.perf_counter()
|
||||
binaural_sink.write_frame(binaural_pcm)
|
||||
binaural_write_seconds += time.perf_counter() - stage
|
||||
done = frame_number + 1
|
||||
if done % progress_every == 0 or done == frame_count:
|
||||
elapsed = time.perf_counter() - started
|
||||
speed = done / max(elapsed, 1e-9)
|
||||
eta = (frame_count - done) / max(speed, 1e-9)
|
||||
print(f"[JOC:{backend_info['name']}] {done}/{frame_count} "
|
||||
f"{speed:.1f} frame/s ETA {eta:.1f}s", flush=True)
|
||||
if binaural_renderer is not None:
|
||||
stage = time.perf_counter()
|
||||
binaural_tail = binaural_renderer.finish()
|
||||
binaural_render_seconds += time.perf_counter() - stage
|
||||
if len(binaural_tail):
|
||||
stage = time.perf_counter()
|
||||
binaural_sink.write_frame(binaural_tail)
|
||||
binaural_write_seconds += time.perf_counter() - stage
|
||||
if output is not None:
|
||||
output.flush()
|
||||
elapsed = time.perf_counter() - started
|
||||
finally:
|
||||
close = getattr(decoder, "close", None)
|
||||
if close is not None:
|
||||
close()
|
||||
close = getattr(speaker_renderer, "close", None)
|
||||
if close is not None:
|
||||
close()
|
||||
close = getattr(binaural_renderer, "close", None)
|
||||
if close is not None:
|
||||
close()
|
||||
breakdown = {
|
||||
"pipeline_wall_seconds": elapsed,
|
||||
"dsp_and_joc_parse_seconds": dsp_seconds,
|
||||
"adm_stream_write_seconds": adm_stream_seconds,
|
||||
"raw_float_write_seconds": raw_write_seconds,
|
||||
"speaker_render_seconds": speaker_render_seconds,
|
||||
"speaker_spool_write_seconds": speaker_write_seconds,
|
||||
"binaural_render_seconds": binaural_render_seconds,
|
||||
"binaural_spool_write_seconds": binaural_write_seconds,
|
||||
}
|
||||
return dsp_seconds, backend_info, breakdown
|
||||
|
||||
|
||||
def build_parser():
|
||||
parser = argparse.ArgumentParser(
|
||||
description=("JustOneCacophony (JOC):E-AC-3 JOC → 25ch ADM BWF、"
|
||||
"扬声器 WAV 或 DLL-free Rosella 双耳 WAV"))
|
||||
parser.add_argument("input", type=Path, help="输入 .m4a/.eac3/.ec3")
|
||||
parser.add_argument("-o", "--output", type=Path, help="输出文件;默认按模式和布局命名")
|
||||
parser.add_argument("--speaker-output", type=Path,
|
||||
help="扬声器 WAV 路径;仅与 --speaker-layout 一起使用")
|
||||
parser.add_argument("--binaural-output", type=Path,
|
||||
help="双耳 WAV 路径;仅与 --binaural 一起使用")
|
||||
direct_mode = parser.add_mutually_exclusive_group()
|
||||
direct_mode.add_argument("--speaker-layout", choices=SPEAKER_LAYOUT_CHOICES,
|
||||
help="直接扬声器渲染布局,例如 2.0、5.1、7.1.2")
|
||||
direct_mode.add_argument("--binaural", action="store_true",
|
||||
help="直接 DLL-free Rosella 双耳渲染;不生成临时 ADM BWF")
|
||||
parser.add_argument("--speaker-format", choices=("float32", "int24"), default="float32",
|
||||
help="扬声器 WAV 格式,默认 float32")
|
||||
parser.add_argument("--binaural-format", choices=("float32", "int24"), default="float32",
|
||||
help="双耳 WAV 格式,默认 float32")
|
||||
parser.add_argument("--clip-action", choices=("ask", "continue", "float32", "abort"),
|
||||
default="ask",
|
||||
help="int24 削波处理:交互询问、继续截断、改 float32 或中止")
|
||||
parser.add_argument("--speaker-metadata-offset", type=int, default=1473,
|
||||
help="扬声器渲染 metadata 相对帧偏移,默认 1473 samples")
|
||||
parser.add_argument("--binaural-mode", choices=("near", "mid", "far"), default="mid",
|
||||
help="普通对象 Rosella 距离模式,默认 mid;LFE 始终走 special 低通")
|
||||
parser.add_argument("--personalized-headphone", "--binaural-hrtf", dest="personalized_headphone",
|
||||
type=Path, help="覆盖 HRTF/binaural.personalized_headphone")
|
||||
parser.add_argument("--binaural-tail-seconds", type=float, default=5.0,
|
||||
help="双耳 room/filterbank flush 上限,默认 5 秒")
|
||||
parser.add_argument("--binaural-tail-threshold", type=float, default=1.0e-8,
|
||||
help="双耳尾声裁切阈值,默认 1e-8;主体至少保留原时长")
|
||||
parser.add_argument("--binaural-chunk-frames", type=int, default=64,
|
||||
help="双耳内部批处理 E-AC-3 帧数,默认 64")
|
||||
parser.add_argument("--gain-db", type=float, default=0.0,
|
||||
help="成品增益 dB,默认 0;双耳路径以 float64 应用")
|
||||
parser.add_argument("--duration", type=float, help="只处理开头指定秒数")
|
||||
parser.add_argument("--object-delay-samples", type=int, default=1473,
|
||||
help="对象 PCM/OAMD 时间补偿;ADM 与双耳默认 1473 samples")
|
||||
parser.add_argument(
|
||||
"--joc-binaural-mode", choices=tuple(adm_atmos.JOC_BINAURAL_MODES),
|
||||
default=adm_atmos.JOC_BINAURAL_MODE_DEFAULT,
|
||||
help="ADM DBMD JOC 模式:off/near/far/mid/unspecified;仅影响 ADM BWF")
|
||||
parser.add_argument("--trajectory-mode", choices=("compact", "dense64"), default="compact",
|
||||
help="ADM 对象轨迹表示;直接双耳路径不序列化 AXML")
|
||||
parser.add_argument("--ffmpeg", default=os.environ.get("FFMPEG", "ffmpeg"))
|
||||
parser.add_argument("--backend", choices=("auto", "native", "python"), default="auto",
|
||||
help="DSP 后端;auto 优先 C++,不可用时回退 Python")
|
||||
parser.add_argument("--native-library", type=Path,
|
||||
help="显式指定原生库;默认从单层 lib 目录选择当前平台文件")
|
||||
parser.add_argument("--native-threads", type=int,
|
||||
help="原生 DSP 总线程数;默认在 4 核以上使用 2,可用环境变量 EAC3JOC_NATIVE_THREADS 覆盖")
|
||||
metadata_source = parser.add_mutually_exclusive_group()
|
||||
metadata_source.add_argument("--metadata-dir", type=Path,
|
||||
help="含 frames.csv 和 emdf/ 或 payloads/ 的元数据 sidecar")
|
||||
metadata_source.add_argument("--metadata-cache", type=Path,
|
||||
help="把直接 EMDF 扫描或兼容桥结果持久保存到此目录")
|
||||
parser.add_argument("--metadata-backend", choices=("auto", "emdf", "sidecar"),
|
||||
default="auto", help="直接扫描连续 EMDF,或读取现有 sidecar")
|
||||
parser.add_argument("--print-metadata", choices=("none", "summary", "frames"), default="none",
|
||||
help="诊断元数据输出;默认 none,避免转换前重复完整解析")
|
||||
parser.add_argument("--metadata-json", type=Path, help="元数据汇总 JSON 路径")
|
||||
parser.add_argument("--metadata-only", action="store_true", help="解析/打印元数据后退出")
|
||||
parser.add_argument("--keep-raw", action="store_true", help="额外保留 16ch f32le 对象中间文件")
|
||||
parser.add_argument("--skip-sha256", action="store_true",
|
||||
help="跳过最终文件 SHA-256 全量复扫以缩短大文件处理时间")
|
||||
parser.add_argument("--progress-every", type=int, default=500)
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv=None):
|
||||
# Windows 控制台的活动代码页未必能表示日文文件名;保留信息并避免
|
||||
# UnicodeEncodeError 中断长任务。支持 UTF-8 的终端仍会原样显示。
|
||||
for stream in (sys.stdout, sys.stderr):
|
||||
if hasattr(stream, "reconfigure"):
|
||||
stream.reconfigure(encoding="utf-8", errors="backslashreplace")
|
||||
args = build_parser().parse_args(argv)
|
||||
source = args.input.expanduser().resolve()
|
||||
if not source.is_file():
|
||||
raise FileNotFoundError(source)
|
||||
speaker_mode = args.speaker_layout is not None
|
||||
binaural_mode = bool(args.binaural)
|
||||
if args.speaker_output is not None and not speaker_mode:
|
||||
raise ValueError("--speaker-output 必须与 --speaker-layout 一起使用")
|
||||
if args.binaural_output is not None and not binaural_mode:
|
||||
raise ValueError("--binaural-output 必须与 --binaural 一起使用")
|
||||
specific_outputs = [value for value in (args.speaker_output, args.binaural_output)
|
||||
if value is not None]
|
||||
if args.output is not None and specific_outputs:
|
||||
raise ValueError("-o/--output 与 --speaker-output/--binaural-output 不能同时使用")
|
||||
if len(specific_outputs) > 1:
|
||||
raise ValueError("--speaker-output 与 --binaural-output 不能同时使用")
|
||||
if args.speaker_metadata_offset < 0:
|
||||
raise ValueError("speaker-metadata-offset 不能为负数")
|
||||
if args.personalized_headphone is not None and not binaural_mode:
|
||||
raise ValueError("--personalized-headphone 仅与 --binaural 一起使用")
|
||||
if args.binaural_tail_seconds < 0:
|
||||
raise ValueError("binaural-tail-seconds 不能为负数")
|
||||
if args.binaural_tail_threshold < 0:
|
||||
raise ValueError("binaural-tail-threshold 不能为负数")
|
||||
if args.binaural_chunk_frames <= 0:
|
||||
raise ValueError("binaural-chunk-frames 必须大于 0")
|
||||
requested_output = (args.speaker_output if args.speaker_output is not None
|
||||
else args.binaural_output if args.binaural_output is not None
|
||||
else args.output)
|
||||
output = resolve_output(
|
||||
source, requested_output, args.speaker_layout if speaker_mode else None,
|
||||
binaural=binaural_mode)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
if args.duration is not None and args.duration <= 0:
|
||||
raise ValueError("duration 必须大于 0")
|
||||
if args.object_delay_samples < 0:
|
||||
raise ValueError("object-delay-samples 不能为负数")
|
||||
gain_float64 = 10.0 ** (args.gain_db / 20.0)
|
||||
gain = np.float32(gain_float64)
|
||||
if not math.isfinite(gain_float64) or not np.isfinite(gain):
|
||||
raise ValueError("gain-db 超出支持范围")
|
||||
binaural_model_path = (
|
||||
resolve_personalized_headphone(args.personalized_headphone)
|
||||
if binaural_mode and not args.metadata_only else None)
|
||||
ffmpeg = executable(args.ffmpeg, "FFmpeg")
|
||||
|
||||
total_started = time.perf_counter()
|
||||
timings = {}
|
||||
with tempfile.TemporaryDirectory(prefix="eac3joc-", dir=output.parent) as temporary:
|
||||
temp_dir = Path(temporary)
|
||||
eac3 = timed_call(timings, "extract_eac3", extract_eac3,
|
||||
ffmpeg, source, temp_dir / "input.eac3")
|
||||
index, metadata_backend, metadata_cache_dir = timed_call(
|
||||
timings, "resolve_metadata", variant_call,
|
||||
output, source, resolve_metadata, args, eac3, temp_dir)
|
||||
timings["load_metadata_index"] = 0.0
|
||||
frame_count = len(index)
|
||||
if args.duration is not None:
|
||||
frame_count = min(frame_count, math.ceil(args.duration * RATE / FRAME_SAMPLES))
|
||||
duration_sec = frame_count * FRAME_SAMPLES / RATE
|
||||
need_metadata_summary = (
|
||||
args.metadata_only or args.metadata_json is not None or args.print_metadata != "none")
|
||||
if need_metadata_summary:
|
||||
metadata_json = (args.metadata_json or Path(str(output) + ".metadata.json")).resolve()
|
||||
summary = timed_call(
|
||||
timings, "metadata_summary", variant_call,
|
||||
output, source, write_summary, index, metadata_json, limit=frame_count,
|
||||
print_frames=args.print_metadata == "frames")
|
||||
if args.print_metadata == "summary":
|
||||
print("[metadata] " + json.dumps(summary, ensure_ascii=False, separators=(",", ":")))
|
||||
print(f"[metadata] backend={metadata_backend} frames={frame_count} -> {metadata_json}")
|
||||
else:
|
||||
metadata_json = None
|
||||
timings["metadata_summary"] = 0.0
|
||||
print(f"[metadata] backend={metadata_backend} frames={frame_count} summary=skipped")
|
||||
if args.metadata_only:
|
||||
return 0
|
||||
|
||||
bed_path = timed_call(
|
||||
timings, "decode_core", decode_core,
|
||||
ffmpeg, eac3, temp_dir / "core51_f32le.raw", duration_sec)
|
||||
raw_path = (output.with_name(output.name + ".objects16.f32le")
|
||||
if args.keep_raw else None)
|
||||
master = None
|
||||
speaker_backend_info = None
|
||||
speaker_wav_info = None
|
||||
speaker_clip_info = None
|
||||
speaker_actual_format = None
|
||||
binaural_backend_info = None
|
||||
binaural_wav_info = None
|
||||
binaural_clip_info = None
|
||||
binaural_actual_format = None
|
||||
if speaker_mode:
|
||||
timings["create_binaural_renderer"] = 0.0
|
||||
layout = get_speaker_layout(args.speaker_layout)
|
||||
speaker_name = speaker_layout_display_name(layout)
|
||||
speaker_decoder, speaker_backend_info = create_speaker_renderer(
|
||||
layout, backend=args.backend, native_library=args.native_library)
|
||||
fallback = speaker_backend_info.get("fallback_reason")
|
||||
if fallback:
|
||||
print(f"[speaker] native unavailable, falling back to Python: {fallback}",
|
||||
flush=True)
|
||||
print(f"[speaker] layout={speaker_name} backend={speaker_backend_info['name']} "
|
||||
f"channels={layout.channel_count}", flush=True)
|
||||
spool = SpeakerPcmSpool(
|
||||
temp_dir / "speaker_interleaved_f32.raw",
|
||||
frame_count * FRAME_SAMPLES, layout.channel_count)
|
||||
try:
|
||||
render_seconds, renderer_backend, render_breakdown = timed_call(
|
||||
timings, "render_and_stream", variant_call,
|
||||
output, source, render, index, bed_path, frame_count, raw_path, gain,
|
||||
max(1, args.progress_every), args.backend, args.native_library,
|
||||
args.native_threads, None, speaker_decoder, spool,
|
||||
args.speaker_metadata_offset)
|
||||
spool.finalize()
|
||||
speaker_actual_format = choose_pcm_output_format(
|
||||
args.speaker_format, args.clip_action, spool.peak,
|
||||
spool.clipped_values)
|
||||
speaker_wav_info = timed_call(
|
||||
timings, "write_speaker_wav", write_pcm_wav,
|
||||
output, spool.values, speaker_actual_format, rate=RATE)
|
||||
speaker_clip_info = {
|
||||
"peak": spool.peak,
|
||||
"over_unity_values": spool.clipped_values,
|
||||
"requested_format": args.speaker_format,
|
||||
"actual_format": speaker_actual_format,
|
||||
"clip_action": args.clip_action,
|
||||
}
|
||||
finally:
|
||||
spool.close()
|
||||
timings["build_adm_tracks"] = 0.0
|
||||
timings["finalize_adm"] = 0.0
|
||||
timings["validate_adm"] = 0.0
|
||||
info = (f"speaker layout={speaker_name}, format={speaker_actual_format}, "
|
||||
f"peak={speaker_clip_info['peak']:.9g}")
|
||||
elif binaural_mode:
|
||||
model_path = binaural_model_path
|
||||
binaural_decoder = timed_call(
|
||||
timings, "create_binaural_renderer", RosellaBinauralRenderer,
|
||||
model_path, mode=args.binaural_mode,
|
||||
object_delay_samples=args.object_delay_samples,
|
||||
tail_seconds=args.binaural_tail_seconds,
|
||||
output_gain=gain_float64,
|
||||
chunk_frames=args.binaural_chunk_frames,
|
||||
backend=args.backend, native_library=args.native_library)
|
||||
print(
|
||||
f"[binaural] mode={args.binaural_mode} "
|
||||
f"backend={binaural_decoder.dsp_backend} "
|
||||
f"precision=float64/complex128 model={model_path}", flush=True)
|
||||
flush_samples = math.ceil(
|
||||
(args.binaural_tail_seconds * RATE
|
||||
+ ROSSELLA_LATENCY_SAMPLES + ROSSELLA_BLOCK_SAMPLES)
|
||||
/ ROSSELLA_BLOCK_SAMPLES) * ROSSELLA_BLOCK_SAMPLES
|
||||
spool = BinauralPcmSpool(
|
||||
temp_dir / "binaural_interleaved_f64.raw",
|
||||
frame_count * FRAME_SAMPLES + flush_samples,
|
||||
tail_threshold=args.binaural_tail_threshold)
|
||||
try:
|
||||
render_seconds, renderer_backend, render_breakdown = timed_call(
|
||||
timings, "render_and_stream", variant_call,
|
||||
output, source, render, index, bed_path, frame_count, raw_path,
|
||||
np.float32(1.0), max(1, args.progress_every),
|
||||
backend=args.backend, native_library=args.native_library,
|
||||
native_threads=args.native_threads,
|
||||
binaural_renderer=binaural_decoder, binaural_sink=spool,
|
||||
binaural_metadata_offset=args.object_delay_samples, raw_scale=gain)
|
||||
spool.finalize(minimum_samples=frame_count * FRAME_SAMPLES)
|
||||
binaural_actual_format = choose_pcm_output_format(
|
||||
args.binaural_format, args.clip_action, spool.peak,
|
||||
spool.clipped_values)
|
||||
binaural_wav_info = timed_call(
|
||||
timings, "write_binaural_wav", write_pcm_wav,
|
||||
output, spool.values, binaural_actual_format, rate=RATE)
|
||||
binaural_clip_info = {
|
||||
"peak": spool.peak,
|
||||
"over_unity_values": spool.clipped_values,
|
||||
"requested_format": args.binaural_format,
|
||||
"actual_format": binaural_actual_format,
|
||||
"clip_action": args.clip_action,
|
||||
"tail_threshold": args.binaural_tail_threshold,
|
||||
"source_samples": frame_count * FRAME_SAMPLES,
|
||||
"kept_samples": spool.sample_count,
|
||||
}
|
||||
binaural_backend_info = binaural_decoder.backend_info
|
||||
finally:
|
||||
spool.close()
|
||||
timings["build_adm_tracks"] = 0.0
|
||||
timings["finalize_adm"] = 0.0
|
||||
timings["validate_adm"] = 0.0
|
||||
info = (f"binaural mode={args.binaural_mode}, "
|
||||
f"format={binaural_actual_format}, "
|
||||
f"peak={binaural_clip_info['peak']:.9g}, "
|
||||
f"samples={binaural_clip_info['kept_samples']}")
|
||||
else:
|
||||
timings["create_binaural_renderer"] = 0.0
|
||||
master = adm_assemble.StreamingMaster(
|
||||
output, duration_sec, rate=RATE,
|
||||
joc_binaural_mode=adm_atmos.JOC_BINAURAL_MODES[args.joc_binaural_mode])
|
||||
try:
|
||||
render_seconds, renderer_backend, render_breakdown = timed_call(
|
||||
timings, "render_and_stream", variant_call,
|
||||
output, source, render, index, bed_path, frame_count, raw_path, gain,
|
||||
max(1, args.progress_every), args.backend, args.native_library,
|
||||
args.native_threads, master)
|
||||
tracks = timed_call(
|
||||
timings, "build_adm_tracks", variant_call,
|
||||
output, source, oamd_tracks.build_adm_tracks,
|
||||
index, index.rows[:frame_count], rate=RATE, frame_samples=FRAME_SAMPLES,
|
||||
object_delay_samples=args.object_delay_samples,
|
||||
trajectory_mode=args.trajectory_mode)
|
||||
timed_call(timings, "finalize_adm", master.finalize, tracks)
|
||||
except Exception:
|
||||
master.abort()
|
||||
raise
|
||||
errors, info = timed_call(timings, "validate_adm", validate, str(output))
|
||||
if errors:
|
||||
raise RuntimeError("ADM 校验失败: " + "; ".join(errors))
|
||||
# Windows 不允许删除仍被 NumPy memmap 持有的临时 core/raw;显式回收闭包。
|
||||
import gc
|
||||
gc.collect()
|
||||
|
||||
if args.skip_sha256:
|
||||
output_sha = None
|
||||
timings["sha256"] = 0.0
|
||||
else:
|
||||
output_sha = timed_call(timings, "sha256", sha256, output)
|
||||
total_seconds = time.perf_counter() - total_started
|
||||
mode_name = "speaker" if speaker_mode else "binaural" if binaural_mode else "adm"
|
||||
report = {
|
||||
"input": str(source),
|
||||
"output": str(output),
|
||||
"mode": mode_name,
|
||||
"metadata": str(metadata_json) if metadata_json is not None else None,
|
||||
"metadata_backend": metadata_backend,
|
||||
"metadata_cache": str(metadata_cache_dir) if metadata_cache_dir is not None else None,
|
||||
"frames": frame_count,
|
||||
"duration_sec": duration_sec,
|
||||
"gain_db": args.gain_db,
|
||||
"gain_float32": float(gain),
|
||||
"gain_float64": float(gain_float64),
|
||||
"object_delay_samples": (None if speaker_mode else args.object_delay_samples),
|
||||
"trajectory_mode": args.trajectory_mode if mode_name == "adm" else None,
|
||||
"joc_binaural_mode": args.joc_binaural_mode if mode_name == "adm" else None,
|
||||
"joc_binaural_mode_value": (
|
||||
adm_atmos.JOC_BINAURAL_MODES[args.joc_binaural_mode]
|
||||
if mode_name == "adm" else None),
|
||||
"render_seconds": render_seconds,
|
||||
"render_breakdown": render_breakdown,
|
||||
"renderer_backend": renderer_backend,
|
||||
"speaker_renderer_backend": speaker_backend_info,
|
||||
"speaker_layout": args.speaker_layout if speaker_mode else None,
|
||||
"speaker_metadata_offset": args.speaker_metadata_offset if speaker_mode else None,
|
||||
"speaker_clip": speaker_clip_info,
|
||||
"speaker_wav": speaker_wav_info,
|
||||
"binaural_renderer_backend": binaural_backend_info,
|
||||
"binaural_mode": args.binaural_mode if binaural_mode else None,
|
||||
"personalized_headphone": (
|
||||
binaural_backend_info["model"] if binaural_backend_info else None),
|
||||
"binaural_clip": binaural_clip_info,
|
||||
"binaural_wav": binaural_wav_info,
|
||||
"output_clip": speaker_clip_info if speaker_mode else binaural_clip_info,
|
||||
"output_wav": speaker_wav_info if speaker_mode else binaural_wav_info,
|
||||
"streaming_adm": mode_name == "adm",
|
||||
"kept_raw": str(raw_path) if raw_path is not None else None,
|
||||
"timings": timings,
|
||||
"total_seconds": total_seconds,
|
||||
"adm_validation": info if mode_name == "adm" else None,
|
||||
"adm_metadata": getattr(master, "metadata_info", None) if master is not None else None,
|
||||
"sha256": output_sha,
|
||||
"python": platform.python_version(),
|
||||
"numpy": np.__version__,
|
||||
}
|
||||
report_path = Path(str(output) + ".report.json")
|
||||
report_path.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
print(f"[PASS] {output}")
|
||||
if report["sha256"] is None:
|
||||
print(f"[PASS] {info}; SHA-256 skipped")
|
||||
else:
|
||||
print(f"[PASS] {info}; SHA-256={report['sha256']}")
|
||||
if speaker_mode:
|
||||
print(f"[time] JOC-DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
|
||||
f"speaker={render_breakdown['speaker_render_seconds']:.2f}s "
|
||||
f"pipeline={render_breakdown['pipeline_wall_seconds']:.2f}s "
|
||||
f"total={report['total_seconds']:.2f}s")
|
||||
elif binaural_mode:
|
||||
print(f"[time] JOC-DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
|
||||
f"binaural={render_breakdown['binaural_render_seconds']:.2f}s "
|
||||
f"pipeline={render_breakdown['pipeline_wall_seconds']:.2f}s "
|
||||
f"total={report['total_seconds']:.2f}s")
|
||||
else:
|
||||
print(f"[time] DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
|
||||
f"render+ADM-stream={render_breakdown['pipeline_wall_seconds']:.2f}s "
|
||||
f"total={report['total_seconds']:.2f}s")
|
||||
print(f"[report] {report_path}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
raise SystemExit(main())
|
||||
except KeyboardInterrupt:
|
||||
raise SystemExit(130)
|
||||
@@ -1,73 +0,0 @@
|
||||
cmake_minimum_required(VERSION 3.20)
|
||||
|
||||
project(eac3joc_core VERSION 1.0.0 LANGUAGES CXX)
|
||||
|
||||
include(GNUInstallDirs)
|
||||
find_package(Threads REQUIRED)
|
||||
|
||||
add_library(eac3joc_core SHARED
|
||||
src/eac3joc_core.cpp
|
||||
src/speaker_renderer.cpp
|
||||
src/binaural_renderer.cpp
|
||||
src/joc_huffman_tables.h
|
||||
src/qmf_tables.h
|
||||
src/speaker_layouts.h
|
||||
)
|
||||
|
||||
target_compile_features(eac3joc_core PRIVATE cxx_std_20)
|
||||
target_include_directories(eac3joc_core PRIVATE
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/include"
|
||||
)
|
||||
target_link_libraries(eac3joc_core PRIVATE Threads::Threads)
|
||||
|
||||
set_target_properties(eac3joc_core PROPERTIES
|
||||
OUTPUT_NAME "eac3joc_core"
|
||||
CXX_VISIBILITY_PRESET hidden
|
||||
VISIBILITY_INLINES_HIDDEN YES
|
||||
POSITION_INDEPENDENT_CODE YES
|
||||
)
|
||||
|
||||
if(MSVC)
|
||||
set_property(TARGET eac3joc_core PROPERTY
|
||||
MSVC_RUNTIME_LIBRARY "MultiThreaded$<$<CONFIG:Debug>:Debug>")
|
||||
target_compile_options(eac3joc_core PRIVATE
|
||||
/W4 /permissive- /Zc:__cplusplus /utf-8 /EHsc /GR- /fp:precise
|
||||
$<$<CONFIG:Release>:/O2>
|
||||
$<$<CONFIG:Release>:/Oi>
|
||||
$<$<CONFIG:Release>:/GL>
|
||||
)
|
||||
target_link_options(eac3joc_core PRIVATE
|
||||
$<$<CONFIG:Release>:/LTCG>
|
||||
/INCREMENTAL:NO /OPT:REF /OPT:ICF
|
||||
)
|
||||
else()
|
||||
target_compile_options(eac3joc_core PRIVATE
|
||||
-Wall -Wextra -Wpedantic -fno-fast-math
|
||||
$<$<CONFIG:Release>:-O3>
|
||||
)
|
||||
endif()
|
||||
|
||||
# Install directly into the chosen prefix. Recommended invocation from the
|
||||
# repository root:
|
||||
# cmake -S native -B build/cmake -DCMAKE_BUILD_TYPE=Release \
|
||||
# -DCMAKE_INSTALL_PREFIX=<repo>/lib
|
||||
# cmake --build build/cmake --config Release
|
||||
# cmake --install build/cmake --config Release
|
||||
#
|
||||
# Result names are supplied by the platform toolchain:
|
||||
# Windows: eac3joc_core.dll (+ eac3joc_core.lib import library)
|
||||
# Linux: libeac3joc_core.so
|
||||
# macOS: libeac3joc_core.dylib
|
||||
install(TARGETS eac3joc_core
|
||||
RUNTIME DESTINATION .
|
||||
LIBRARY DESTINATION .
|
||||
ARCHIVE DESTINATION .
|
||||
)
|
||||
|
||||
if(MSVC)
|
||||
install(FILES "$<TARGET_PDB_FILE:eac3joc_core>"
|
||||
DESTINATION .
|
||||
OPTIONAL
|
||||
)
|
||||
endif()
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
numpy>=1.24
|
||||
@@ -0,0 +1,333 @@
|
||||
#include "adm/adm_metadata.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
|
||||
#include "foundation/py_num.h"
|
||||
|
||||
namespace joc::adm {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr const char* kBedNames[10] = {
|
||||
"RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter", "RoomCentricLFE",
|
||||
"RoomCentricLeftSideSurround", "RoomCentricRightSideSurround",
|
||||
"RoomCentricLeftRearSurround", "RoomCentricRightRearSurround",
|
||||
"RoomCentricLeftTopSurround", "RoomCentricRightTopSurround"};
|
||||
constexpr const char* kBedLabels[10] = {"RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss",
|
||||
"RC_Rss", "RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"};
|
||||
constexpr double kBedPos[10][3] = {{-1.0, 1.0, 0.0}, {1.0, 1.0, 0.0}, {0.0, 1.0, 0.0},
|
||||
{-1.0, 1.0, -1.0}, {-1.0, 0.0, 0.0}, {1.0, 0.0, 0.0},
|
||||
{-1.0, -1.0, 0.0}, {1.0, -1.0, 0.0}, {-1.0, 0.0, 1.0},
|
||||
{1.0, 0.0, 1.0}};
|
||||
|
||||
std::string hex4(std::uint32_t value) {
|
||||
char buffer[16];
|
||||
std::snprintf(buffer, sizeof(buffer), "%04x", value);
|
||||
return std::string(buffer);
|
||||
}
|
||||
|
||||
std::string hex8(std::uint32_t value) {
|
||||
char buffer[16];
|
||||
std::snprintf(buffer, sizeof(buffer), "%08x", value);
|
||||
return std::string(buffer);
|
||||
}
|
||||
|
||||
void put_u16(std::string* out, std::uint16_t value) {
|
||||
char buffer[2];
|
||||
std::memcpy(buffer, &value, 2);
|
||||
out->append(buffer, 2);
|
||||
}
|
||||
|
||||
void put_u32(std::string* out, std::uint32_t value) {
|
||||
char buffer[4];
|
||||
std::memcpy(buffer, &value, 4);
|
||||
out->append(buffer, 4);
|
||||
}
|
||||
|
||||
std::uint8_t checksum(const std::string& segment) {
|
||||
int sum = static_cast<int>(segment.size());
|
||||
for (const char raw : segment) {
|
||||
sum += static_cast<unsigned char>(raw);
|
||||
}
|
||||
return static_cast<std::uint8_t>((~sum + 1) & 0xFF);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::string ts(double seconds) {
|
||||
long long whole = static_cast<long long>(seconds);
|
||||
long long fraction = pynum::py_round((seconds - static_cast<double>(whole)) * 100000.0);
|
||||
if (fraction >= 100000) {
|
||||
whole += 1;
|
||||
fraction = 0;
|
||||
}
|
||||
char buffer[32];
|
||||
std::snprintf(buffer, sizeof(buffer), "%02lld:%02lld:%02lld.%05lld", whole / 3600,
|
||||
(whole % 3600) / 60, whole % 60, fraction);
|
||||
return std::string(buffer);
|
||||
}
|
||||
|
||||
bool binaural_mode_from_name(const char* name, BinauralMode* out) {
|
||||
if (name == nullptr || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (std::strcmp(name, "off") == 0) { *out = BinauralMode::Off; return true; }
|
||||
if (std::strcmp(name, "near") == 0) { *out = BinauralMode::Near; return true; }
|
||||
if (std::strcmp(name, "far") == 0) { *out = BinauralMode::Far; return true; }
|
||||
if (std::strcmp(name, "mid") == 0) { *out = BinauralMode::Mid; return true; }
|
||||
if (std::strcmp(name, "unspecified") == 0) { *out = BinauralMode::Unspecified; return true; }
|
||||
return false;
|
||||
}
|
||||
|
||||
std::string build_chna() {
|
||||
std::string out;
|
||||
put_u16(&out, static_cast<std::uint16_t>(kTrackCount));
|
||||
put_u16(&out, static_cast<std::uint16_t>(kTrackCount));
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
put_u16(&out, static_cast<std::uint16_t>(i + 1));
|
||||
out += "ATU_" + hex8(i + 1);
|
||||
out += "AT_0001" + hex4(0x1001 + i) + "_01";
|
||||
out += "AP_00011001";
|
||||
out.push_back('\0');
|
||||
}
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
put_u16(&out, static_cast<std::uint16_t>(i + 11));
|
||||
out += "ATU_" + hex8(i + 11);
|
||||
out += "AT_0003" + hex4(0x1001 + i) + "_01";
|
||||
out += "AP_0003" + hex4(0x1001 + i);
|
||||
out.push_back('\0');
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
Status build_dbmd(std::uint32_t object_count, BinauralMode mode, std::string* out) {
|
||||
if (out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput, "null output");
|
||||
}
|
||||
const std::uint32_t mode_value = static_cast<std::uint32_t>(mode);
|
||||
if (mode_value > 4u) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput,
|
||||
"invalid JOC binaural render mode");
|
||||
}
|
||||
out->clear();
|
||||
put_u32(out, 0x01000006u);
|
||||
|
||||
std::string segment7(96, '\0');
|
||||
segment7[1] = static_cast<char>(0x47);
|
||||
segment7[5] = static_cast<char>(0x60);
|
||||
segment7[8] = static_cast<char>(0x24);
|
||||
segment7[9] = static_cast<char>(0x24);
|
||||
out->push_back(7);
|
||||
put_u16(out, 96);
|
||||
out->append(segment7);
|
||||
out->push_back(static_cast<char>(checksum(segment7)));
|
||||
|
||||
std::string segment9(248, '\0');
|
||||
const std::string creator = "Created with EAC3JOC";
|
||||
const std::string renderer = "EAC3JOC Python Renderer";
|
||||
std::memcpy(&segment9[0], creator.data(), creator.size());
|
||||
std::memcpy(&segment9[32], renderer.data(), renderer.size());
|
||||
segment9[96] = 2;
|
||||
segment9[97] = 1;
|
||||
segment9[98] = 0;
|
||||
segment9[103] = 0x03;
|
||||
segment9[106] = 0x01;
|
||||
segment9[111] = 0x22;
|
||||
segment9[112] = static_cast<char>(0xFF);
|
||||
out->push_back(9);
|
||||
put_u16(out, 248);
|
||||
out->append(segment9);
|
||||
out->push_back(static_cast<char>(checksum(segment9)));
|
||||
|
||||
// The reference allocates the body zeroed and then fills only the trailing
|
||||
// `object_count` bytes with 0x84, so the template region stays zero.
|
||||
const std::size_t object_body = 5u + 262u + object_count;
|
||||
std::string segment10(object_body, '\0');
|
||||
const std::uint32_t sync = 0xF8726FBDu;
|
||||
std::memcpy(&segment10[0], &sync, 4);
|
||||
segment10[4] = static_cast<char>(object_count);
|
||||
for (std::size_t i = 5u + 262u; i < segment10.size(); ++i) {
|
||||
segment10[i] = static_cast<char>(0x84);
|
||||
}
|
||||
const std::size_t object_modes = 4u + 2u + 1u + 9u * 15u + object_count;
|
||||
for (std::uint32_t i = 10; i < std::min<std::uint32_t>(object_count, 10u + kObjectCount); ++i) {
|
||||
const std::size_t index = object_modes + i;
|
||||
if (index >= segment10.size()) {
|
||||
return Status::fail(JOC_ERR_INTERNAL, stage::kOutput, "dbmd object slot out of range");
|
||||
}
|
||||
segment10[index] = static_cast<char>((static_cast<unsigned char>(segment10[index]) & 0xF8u) |
|
||||
mode_value);
|
||||
}
|
||||
out->push_back(10);
|
||||
put_u16(out, static_cast<std::uint16_t>(segment10.size()));
|
||||
out->append(segment10);
|
||||
out->push_back(static_cast<char>(checksum(segment10)));
|
||||
out->append("\0\0", 2);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
std::string build_axml(const std::vector<Track>& tracks, double duration_sec, std::uint32_t rate) {
|
||||
const double scale = static_cast<double>(rate);
|
||||
std::string out;
|
||||
out.reserve(64u * 1024u);
|
||||
const std::string duration_ts = ts(duration_sec);
|
||||
|
||||
out += "<?xml version=\"1.0\" encoding=\"utf-8\"?>";
|
||||
out += "<ebuCoreMain xsi:schemaLocation=\"urn:ebu:metadata-schema:ebuCore_2016 ebucore.xsd\" "
|
||||
"lang=\"en\" xmlns:xsi=\"http://www.w3.org/2001/XMLSchema-instance\" "
|
||||
"xmlns=\"urn:ebu:metadata-schema:ebuCore_2016\">";
|
||||
out += "<coreMetadata><format><audioFormatExtended>";
|
||||
out += "<audioProgramme audioProgrammeID=\"APR_1001\" audioProgrammeName=\"EAC3JOC_Export\" "
|
||||
"start=\"" +
|
||||
ts(0.0) + "\" end=\"" + duration_ts + "\">";
|
||||
out += "<audioContentIDRef>ACO_1001</audioContentIDRef>";
|
||||
out += "<audioContentIDRef>ACO_1002</audioContentIDRef>";
|
||||
out += "</audioProgramme>";
|
||||
out += "<audioContent audioContentID=\"ACO_1001\" "
|
||||
"audioContentName=\"EAC3JOC_Master_Content\">";
|
||||
out += "<audioObjectIDRef>AO_1001</audioObjectIDRef>";
|
||||
out += "<dialogue mixedContentKind=\"0\">2</dialogue>";
|
||||
out += "</audioContent>";
|
||||
out += "<audioContent audioContentID=\"ACO_1002\" audioContentName=\"Objects\">";
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioObjectIDRef>AO_" + hex4(0x100b + i) + "</audioObjectIDRef>";
|
||||
}
|
||||
out += "<dialogue mixedContentKind=\"0\">2</dialogue>";
|
||||
out += "</audioContent>";
|
||||
out += "<audioObject audioObjectID=\"AO_1001\" audioObjectName=\"Bed\" start=\"" + ts(0.0) +
|
||||
"\" duration=\"" + duration_ts + "\">";
|
||||
out += "<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>";
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioTrackUIDRef>ATU_" + hex8(i + 1) + "</audioTrackUIDRef>";
|
||||
}
|
||||
out += "</audioObject>";
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioObject audioObjectID=\"AO_" + hex4(0x100b + i) +
|
||||
"\" audioObjectName=\"Audio Object " + std::to_string(i + 1) + "\" start=\"" +
|
||||
ts(0.0) + "\" duration=\"" + duration_ts + "\">";
|
||||
out += "<audioPackFormatIDRef>AP_0003" + hex4(0x1001 + i) + "</audioPackFormatIDRef>";
|
||||
out += "<audioTrackUIDRef>ATU_" + hex8(11 + i) + "</audioTrackUIDRef>";
|
||||
out += "</audioObject>";
|
||||
}
|
||||
out += "<audioPackFormat audioPackFormatID=\"AP_00011001\" "
|
||||
"audioPackFormatName=\"EAC3JOCBedPack\" typeDefinition=\"DirectSpeakers\" "
|
||||
"typeLabel=\"0001\">";
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioChannelFormatIDRef>AC_0001" + hex4(0x1001 + i) +
|
||||
"</audioChannelFormatIDRef>";
|
||||
}
|
||||
out += "</audioPackFormat>";
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioPackFormat audioPackFormatID=\"AP_0003" + hex4(0x1001 + i) +
|
||||
"\" audioPackFormatName=\"JOC_Object_" + std::to_string(i + 1) +
|
||||
"\" typeDefinition=\"Objects\" typeLabel=\"0003\">";
|
||||
out += "<audioChannelFormatIDRef>AC_0003" + hex4(0x1001 + i) +
|
||||
"</audioChannelFormatIDRef>";
|
||||
out += "</audioPackFormat>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioChannelFormat audioChannelFormatID=\"AC_0001" + hex4(0x1001 + i) +
|
||||
"\" audioChannelFormatName=\"" + kBedNames[i] +
|
||||
"\" typeDefinition=\"DirectSpeakers\" typeLabel=\"0001\">";
|
||||
out += "<audioBlockFormat audioBlockFormatID=\"AB_0001" + hex4(0x1001 + i) +
|
||||
"_00000001\">";
|
||||
out += "<cartesian>1</cartesian>";
|
||||
out += "<position coordinate=\"X\">" + pynum::format_fixed(kBedPos[i][0], 10) +
|
||||
"</position>";
|
||||
out += "<position coordinate=\"Y\">" + pynum::format_fixed(kBedPos[i][1], 10) +
|
||||
"</position>";
|
||||
if (kBedPos[i][2] != 0.0) {
|
||||
out += "<position coordinate=\"Z\">" + pynum::format_fixed(kBedPos[i][2], 10) +
|
||||
"</position>";
|
||||
}
|
||||
out += std::string("<speakerLabel>") + kBedLabels[i] + "</speakerLabel>";
|
||||
out += "</audioBlockFormat>";
|
||||
out += "</audioChannelFormat>";
|
||||
}
|
||||
for (std::size_t i = 0; i < tracks.size(); ++i) {
|
||||
const Track& track = tracks[i];
|
||||
out += "<audioChannelFormat audioChannelFormatID=\"AC_0003" +
|
||||
hex4(0x1001 + static_cast<std::uint32_t>(i)) + "\" audioChannelFormatName=\"" +
|
||||
track.name + "\" typeDefinition=\"Objects\" typeLabel=\"0003\">";
|
||||
for (std::size_t k = 0; k < track.blocks.size(); ++k) {
|
||||
const Keyframe& block = track.blocks[k];
|
||||
out += "<audioBlockFormat audioBlockFormatID=\"AB_0003" +
|
||||
hex4(0x1001 + static_cast<std::uint32_t>(i)) + "_" +
|
||||
hex8(static_cast<std::uint32_t>(k + 1)) + "\" rtime=\"" +
|
||||
ts(static_cast<double>(block.rtime_samples) / scale) + "\" duration=\"" +
|
||||
ts(static_cast<double>(block.duration_samples) / scale) + "\">";
|
||||
out += "<cartesian>1</cartesian>";
|
||||
out += "<position coordinate=\"X\">" + pynum::format_fixed(block.x, 10) + "</position>";
|
||||
out += "<position coordinate=\"Y\">" + pynum::format_fixed(block.y, 10) + "</position>";
|
||||
if (block.z != 0.0) {
|
||||
out += "<position coordinate=\"Z\">" + pynum::format_fixed(block.z, 10) +
|
||||
"</position>";
|
||||
}
|
||||
out += "<jumpPosition interpolationLength=\"" +
|
||||
pynum::format_fixed(static_cast<double>(block.interpolation_samples) / scale,
|
||||
5) +
|
||||
"\">1</jumpPosition>";
|
||||
out += "</audioBlockFormat>";
|
||||
}
|
||||
out += "</audioChannelFormat>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioTrackUID UID=\"ATU_" + hex8(i + 1) +
|
||||
"\" bitDepth=\"24\" sampleRate=\"48000\">";
|
||||
out += "<audioTrackFormatIDRef>AT_0001" + hex4(0x1001 + i) + "_01</audioTrackFormatIDRef>";
|
||||
out += "<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>";
|
||||
out += "</audioTrackUID>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioTrackUID UID=\"ATU_" + hex8(11 + i) +
|
||||
"\" bitDepth=\"24\" sampleRate=\"48000\">";
|
||||
out += "<audioTrackFormatIDRef>AT_0003" + hex4(0x1001 + i) + "_01</audioTrackFormatIDRef>";
|
||||
out += "<audioPackFormatIDRef>AP_0003" + hex4(0x1001 + i) + "</audioPackFormatIDRef>";
|
||||
out += "</audioTrackUID>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioTrackFormat audioTrackFormatID=\"AT_0001" + hex4(0x1001 + i) +
|
||||
"_01\" audioTrackFormatName=\"PCM_" + kBedNames[i] +
|
||||
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
|
||||
out += "<audioStreamFormatIDRef>AS_0001" + hex4(0x1001 + i) +
|
||||
"</audioStreamFormatIDRef>";
|
||||
out += "</audioTrackFormat>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioTrackFormat audioTrackFormatID=\"AT_0003" + hex4(0x1001 + i) +
|
||||
"_01\" audioTrackFormatName=\"PCM_JOC_Object_" + std::to_string(i + 1) +
|
||||
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
|
||||
out += "<audioStreamFormatIDRef>AS_0003" + hex4(0x1001 + i) +
|
||||
"</audioStreamFormatIDRef>";
|
||||
out += "</audioTrackFormat>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < 10; ++i) {
|
||||
out += "<audioStreamFormat audioStreamFormatID=\"AS_0001" + hex4(0x1001 + i) +
|
||||
"\" audioStreamFormatName=\"PCM_" + kBedNames[i] +
|
||||
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
|
||||
out += "<audioChannelFormatIDRef>AC_0001" + hex4(0x1001 + i) +
|
||||
"</audioChannelFormatIDRef>";
|
||||
out += "<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>";
|
||||
out += "<audioTrackFormatIDRef>AT_0001" + hex4(0x1001 + i) +
|
||||
"_01</audioTrackFormatIDRef>";
|
||||
out += "</audioStreamFormat>";
|
||||
}
|
||||
for (std::uint32_t i = 0; i < kObjectCount; ++i) {
|
||||
out += "<audioStreamFormat audioStreamFormatID=\"AS_0003" + hex4(0x1001 + i) +
|
||||
"\" audioStreamFormatName=\"PCM_JOC_Object_" + std::to_string(i + 1) +
|
||||
"\" formatDefinition=\"PCM\" formatLabel=\"0001\">";
|
||||
out += "<audioChannelFormatIDRef>AC_0003" + hex4(0x1001 + i) +
|
||||
"</audioChannelFormatIDRef>";
|
||||
out += "<audioPackFormatIDRef>AP_0003" + hex4(0x1001 + i) + "</audioPackFormatIDRef>";
|
||||
out += "<audioTrackFormatIDRef>AT_0003" + hex4(0x1001 + i) +
|
||||
"_01</audioTrackFormatIDRef>";
|
||||
out += "</audioStreamFormat>";
|
||||
}
|
||||
out += "</audioFormatExtended></format></coreMetadata>";
|
||||
out += "</ebuCoreMain>";
|
||||
return out;
|
||||
}
|
||||
|
||||
} // namespace joc::adm
|
||||
@@ -0,0 +1,43 @@
|
||||
// Port of src/adm_atmos.py.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::adm {
|
||||
|
||||
inline constexpr int kObjectCount = 15;
|
||||
inline constexpr std::uint32_t kTrackCount = 25;
|
||||
|
||||
struct Keyframe {
|
||||
std::int64_t rtime_samples = 0;
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
std::int64_t duration_samples = 0;
|
||||
std::int64_t interpolation_samples = 0;
|
||||
};
|
||||
|
||||
struct Track {
|
||||
std::string name;
|
||||
std::vector<Keyframe> blocks;
|
||||
};
|
||||
|
||||
// HH:MM:SS.fffff with the reference's truncation + round-half-even carry.
|
||||
std::string ts(double seconds);
|
||||
|
||||
enum class BinauralMode : std::uint32_t { Off = 0, Near = 1, Far = 2, Mid = 3, Unspecified = 4 };
|
||||
|
||||
bool binaural_mode_from_name(const char* name, BinauralMode* out);
|
||||
|
||||
std::string build_axml(const std::vector<Track>& tracks, double duration_sec, std::uint32_t rate);
|
||||
|
||||
std::string build_chna();
|
||||
|
||||
Status build_dbmd(std::uint32_t object_count, BinauralMode mode, std::string* out);
|
||||
|
||||
} // namespace joc::adm
|
||||
@@ -0,0 +1,281 @@
|
||||
#include "adm/adm_tracks.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
|
||||
#include "foundation/geometry.h"
|
||||
|
||||
namespace joc::adm {
|
||||
|
||||
namespace {
|
||||
|
||||
struct Point {
|
||||
std::int64_t sample = 0;
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
std::int64_t interpolation_samples = 0;
|
||||
};
|
||||
|
||||
// otherwise (the reference drops silently).
|
||||
void append_point(std::vector<Point>* points, std::int64_t sample, double x, double y, double z,
|
||||
std::int64_t interpolation_samples) {
|
||||
if (!points->empty() && sample == points->back().sample) {
|
||||
points->back() = Point{sample, x, y, z, interpolation_samples};
|
||||
} else if (points->empty() || sample > points->back().sample) {
|
||||
points->push_back(Point{sample, x, y, z, interpolation_samples});
|
||||
}
|
||||
}
|
||||
|
||||
void lerp(double ax, double ay, double az, double bx, double by, double bz, double amount,
|
||||
double* x, double* y, double* z) {
|
||||
*x = ax + (bx - ax) * amount;
|
||||
*y = ay + (by - ay) * amount;
|
||||
*z = az + (bz - az) * amount;
|
||||
}
|
||||
|
||||
void points_to_blocks(const std::vector<Point>& points, std::int64_t total_samples,
|
||||
std::vector<Keyframe>* out) {
|
||||
for (std::size_t index = 0; index < points.size(); ++index) {
|
||||
const Point& point = points[index];
|
||||
const std::int64_t end =
|
||||
(index + 1 < points.size()) ? points[index + 1].sample : total_samples;
|
||||
const std::int64_t duration = std::max<std::int64_t>(0, end - point.sample);
|
||||
if (duration == 0) {
|
||||
continue;
|
||||
}
|
||||
Keyframe keyframe;
|
||||
keyframe.rtime_samples = point.sample;
|
||||
keyframe.x = point.x;
|
||||
keyframe.y = point.y;
|
||||
keyframe.z = point.z;
|
||||
keyframe.duration_samples = duration;
|
||||
keyframe.interpolation_samples = std::min(point.interpolation_samples, duration);
|
||||
out->push_back(keyframe);
|
||||
}
|
||||
}
|
||||
|
||||
Status non_monotonic(const char* name, const char* message, int object_index, std::int64_t sample,
|
||||
std::int64_t previous_sample) {
|
||||
return Status::fail(JOC_ERR_OAMD_UNSUPPORTED_VARIANT, stage::kOamd,
|
||||
std::string(name) + ": " + message + " (object " +
|
||||
std::to_string(object_index) + ", sample " + std::to_string(sample) +
|
||||
", previous " + std::to_string(previous_sample) + ")");
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Status expand_compact(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
|
||||
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
|
||||
int object_index, std::vector<Keyframe>* out) {
|
||||
out->clear();
|
||||
const double scale = static_cast<double>(rate);
|
||||
if (events.empty()) {
|
||||
Keyframe keyframe;
|
||||
keyframe.rtime_samples = 0;
|
||||
keyframe.duration_samples = total_samples;
|
||||
keyframe.interpolation_samples = 0;
|
||||
keyframe.x = 0.0;
|
||||
keyframe.y = 0.0;
|
||||
keyframe.z = 0.0;
|
||||
out->push_back(keyframe);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
std::vector<Point> points;
|
||||
double current[3] = {events[0].x, events[0].y, events[0].z};
|
||||
append_point(&points, 0, current[0], current[1], current[2], 0);
|
||||
|
||||
for (std::size_t index = 1; index < events.size(); ++index) {
|
||||
const OamdEvent& event = events[index];
|
||||
const std::int64_t event_start = event.sample + object_delay_samples;
|
||||
if (event_start >= total_samples) {
|
||||
break;
|
||||
}
|
||||
const std::int64_t effective_ramp =
|
||||
std::max<std::int64_t>(0, event.ramp_samples - update_quantum_samples);
|
||||
const std::int64_t block_start =
|
||||
event_start + (effective_ramp != 0 ? update_quantum_samples : 0);
|
||||
if (block_start >= total_samples) {
|
||||
break;
|
||||
}
|
||||
const std::int64_t ramp_end = block_start + effective_ramp;
|
||||
|
||||
if (block_start < points.back().sample) {
|
||||
return non_monotonic("non_monotonic_compact_position_updates",
|
||||
"compact object position update moved backwards", object_index,
|
||||
block_start, points.back().sample);
|
||||
}
|
||||
if (index + 1 < events.size()) {
|
||||
const std::int64_t next_event_start = events[index + 1].sample + object_delay_samples;
|
||||
const std::int64_t next_effective =
|
||||
std::max<std::int64_t>(0, events[index + 1].ramp_samples - update_quantum_samples);
|
||||
const std::int64_t next_block_start =
|
||||
next_event_start + (next_effective != 0 ? update_quantum_samples : 0);
|
||||
if (next_block_start < ramp_end) {
|
||||
return non_monotonic("overlapping_compact_position_ramps",
|
||||
"a new position update arrived before the previous compact "
|
||||
"ramp finished",
|
||||
object_index, block_start, ramp_end);
|
||||
}
|
||||
}
|
||||
|
||||
double target[3] = {event.x, event.y, event.z};
|
||||
std::int64_t interpolation = effective_ramp;
|
||||
const std::int64_t available = total_samples - block_start;
|
||||
if (effective_ramp > available) {
|
||||
const double amount =
|
||||
static_cast<double>(available) / static_cast<double>(effective_ramp);
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
lerp(current[0], current[1], current[2], event.x, event.y, event.z, amount, &x, &y, &z);
|
||||
target[0] = x;
|
||||
target[1] = y;
|
||||
target[2] = z;
|
||||
interpolation = available;
|
||||
}
|
||||
append_point(&points, block_start, target[0], target[1], target[2], interpolation);
|
||||
current[0] = event.x;
|
||||
current[1] = event.y;
|
||||
current[2] = event.z;
|
||||
}
|
||||
|
||||
points_to_blocks(points, total_samples, out);
|
||||
(void)scale;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status expand_dense64(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
|
||||
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
|
||||
int object_index, std::vector<Keyframe>* out) {
|
||||
out->clear();
|
||||
(void)rate;
|
||||
if (events.empty()) {
|
||||
Keyframe keyframe;
|
||||
keyframe.duration_samples = total_samples;
|
||||
out->push_back(keyframe);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
std::vector<Point> points;
|
||||
double current[3] = {events[0].x, events[0].y, events[0].z};
|
||||
append_point(&points, 0, current[0], current[1], current[2], 0);
|
||||
|
||||
for (std::size_t index = 1; index < events.size(); ++index) {
|
||||
const OamdEvent& event = events[index];
|
||||
const std::int64_t start = event.sample + object_delay_samples;
|
||||
if (start >= total_samples) {
|
||||
break;
|
||||
}
|
||||
if (start < points.back().sample) {
|
||||
return non_monotonic("non_monotonic_position_updates",
|
||||
"object position update moved backwards", object_index, start,
|
||||
points.back().sample);
|
||||
}
|
||||
if (start > points.back().sample) {
|
||||
append_point(&points, start, current[0], current[1], current[2], 0);
|
||||
}
|
||||
const std::int64_t effective_ramp =
|
||||
std::max<std::int64_t>(0, event.ramp_samples - update_quantum_samples);
|
||||
if (effective_ramp == 0) {
|
||||
append_point(&points, start, event.x, event.y, event.z, 0);
|
||||
current[0] = event.x;
|
||||
current[1] = event.y;
|
||||
current[2] = event.z;
|
||||
continue;
|
||||
}
|
||||
const std::int64_t steps =
|
||||
(effective_ramp + update_quantum_samples - 1) / update_quantum_samples;
|
||||
const std::int64_t end = start + steps * update_quantum_samples;
|
||||
if (index + 1 < events.size()) {
|
||||
const std::int64_t next_start = events[index + 1].sample + object_delay_samples;
|
||||
if (next_start < end) {
|
||||
return non_monotonic("overlapping_position_ramps",
|
||||
"a new position update arrived before the previous ramp "
|
||||
"finished",
|
||||
object_index, start, end);
|
||||
}
|
||||
}
|
||||
std::int64_t future = effective_ramp;
|
||||
std::int64_t elapsed = 0;
|
||||
double position[3] = {current[0], current[1], current[2]};
|
||||
while (future > 0) {
|
||||
const double amount =
|
||||
std::min(static_cast<double>(update_quantum_samples) / static_cast<double>(future),
|
||||
1.0);
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
lerp(position[0], position[1], position[2], event.x, event.y, event.z, amount, &x, &y,
|
||||
&z);
|
||||
position[0] = x;
|
||||
position[1] = y;
|
||||
position[2] = z;
|
||||
elapsed += update_quantum_samples;
|
||||
const std::int64_t sample = start + elapsed;
|
||||
if (sample >= total_samples) {
|
||||
break;
|
||||
}
|
||||
append_point(&points, sample, position[0], position[1], position[2],
|
||||
update_quantum_samples);
|
||||
future -= update_quantum_samples;
|
||||
}
|
||||
current[0] = event.x;
|
||||
current[1] = event.y;
|
||||
current[2] = event.z;
|
||||
}
|
||||
|
||||
points_to_blocks(points, total_samples, out);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
void TrajectoryBuilder::submit_frame(std::int64_t frame_index, const oamd::OamdUpdate* update,
|
||||
std::int64_t outer_offset) {
|
||||
std::int64_t event_sample = frame_index * 1536;
|
||||
std::int64_t ramp_samples = 0;
|
||||
if (update != nullptr) {
|
||||
state_.apply(*update);
|
||||
event_sample += outer_offset + static_cast<std::int64_t>(update->block_offset_samples);
|
||||
ramp_samples = static_cast<std::int64_t>(update->ramp_duration_samples);
|
||||
}
|
||||
for (int object = 1; object <= kObjectCount; ++object) {
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
geometry::q_to_adm_xyz(state_.q(object, 0), state_.q(object, 1), state_.q(object, 2), &x,
|
||||
&y, &z);
|
||||
const int slot = object - 1;
|
||||
if (!has_previous_[slot] || previous_[slot][0] != x || previous_[slot][1] != y ||
|
||||
previous_[slot][2] != z) {
|
||||
events_[slot].push_back(OamdEvent{event_sample, x, y, z, ramp_samples});
|
||||
previous_[slot][0] = x;
|
||||
previous_[slot][1] = y;
|
||||
previous_[slot][2] = z;
|
||||
has_previous_[slot] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Status TrajectoryBuilder::build(std::int64_t total_samples, TrajectoryMode mode,
|
||||
std::vector<Track>* out) const {
|
||||
out->clear();
|
||||
out->reserve(kObjectCount);
|
||||
for (int object = 1; object <= kObjectCount; ++object) {
|
||||
Track track;
|
||||
track.name = "JOC_Object_" + std::to_string(object);
|
||||
const Status status =
|
||||
(mode == TrajectoryMode::Compact)
|
||||
? expand_compact(events_[object - 1], total_samples, rate_, quantum_,
|
||||
object_delay_, object, &track.blocks)
|
||||
: expand_dense64(events_[object - 1], total_samples, rate_, quantum_,
|
||||
object_delay_, object, &track.blocks);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
out->push_back(std::move(track));
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::adm
|
||||
@@ -0,0 +1,61 @@
|
||||
// Port of src/oamd_tracks.py.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "adm/adm_metadata.h"
|
||||
#include "foundation/status.h"
|
||||
#include "oamd/oamd_parser.h"
|
||||
|
||||
namespace joc::adm {
|
||||
|
||||
struct OamdEvent {
|
||||
std::int64_t sample = 0;
|
||||
double x = 0.0;
|
||||
double y = 0.0;
|
||||
double z = 0.0;
|
||||
std::int64_t ramp_samples = 0;
|
||||
};
|
||||
|
||||
enum class TrajectoryMode { Compact, Dense64 };
|
||||
|
||||
// Feeds the same per-frame OAMD state machine the reference's build_adm_tracks
|
||||
// runs, and records one event per object whenever its coordinates change.
|
||||
class TrajectoryBuilder {
|
||||
public:
|
||||
TrajectoryBuilder(std::uint32_t rate = 48000, std::int64_t update_quantum_samples = 64,
|
||||
std::int64_t object_delay_samples = 1473)
|
||||
: rate_(rate),
|
||||
quantum_(update_quantum_samples),
|
||||
object_delay_(object_delay_samples) {}
|
||||
|
||||
void submit_frame(std::int64_t frame_index, const oamd::OamdUpdate* update, std::int64_t outer_offset);
|
||||
|
||||
Status build(std::int64_t total_samples, TrajectoryMode mode, std::vector<Track>* out) const;
|
||||
|
||||
const std::vector<OamdEvent>& events(int object_index) const { return events_[object_index]; }
|
||||
std::uint32_t rate() const { return rate_; }
|
||||
std::int64_t object_delay_samples() const { return object_delay_; }
|
||||
|
||||
private:
|
||||
std::uint32_t rate_;
|
||||
std::int64_t quantum_;
|
||||
std::int64_t object_delay_;
|
||||
oamd::OamdState state_;
|
||||
std::vector<OamdEvent> events_[kObjectCount];
|
||||
bool has_previous_[kObjectCount] = {};
|
||||
double previous_[kObjectCount][3] = {};
|
||||
};
|
||||
|
||||
Status expand_compact(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
|
||||
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
|
||||
int object_index, std::vector<Keyframe>* out);
|
||||
|
||||
Status expand_dense64(const std::vector<OamdEvent>& events, std::int64_t total_samples, std::uint32_t rate,
|
||||
std::int64_t update_quantum_samples, std::int64_t object_delay_samples,
|
||||
int object_index, std::vector<Keyframe>* out);
|
||||
|
||||
} // namespace joc::adm
|
||||
@@ -1,145 +0,0 @@
|
||||
"""把 LFE、15 路对象 PCM 和对象轨迹组装为 ADM BWF。
|
||||
|
||||
固定输出契约:
|
||||
EAC3JOC 重放输出 = 16ch(ch0 = LFE + ch1-15 = 15 对象);
|
||||
最终 ADM BWF = 7.1.2 bed(L R C Ls Rs Lb Rb + LFE + Ltf Rtf = 10ch)
|
||||
—— 除 LFE 外全部静音;
|
||||
15 对象 = ch1-15 直接填充对象轨;轨迹 = OAMD(q1/q2/q3 → xyz)。
|
||||
"""
|
||||
import os
|
||||
|
||||
import numpy as np
|
||||
|
||||
import adm_atmos
|
||||
|
||||
|
||||
def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None,
|
||||
duration_sec=None, rate=48000, joc_binaural_mode=4):
|
||||
"""16ch f32 交织 raw → 25ch ADM BWF(空 7.1.2 bed + LFE + 15 对象)。
|
||||
|
||||
raw16: (n, 16) 交织(ch0 = LFE,ch1-15 = 对象)。
|
||||
scale: 1.0 = 默认 0 dB,不附加输出缩放。该参数与 joc_clipgain 无关;
|
||||
主命令行已在渲染阶段应用用户增益,因此这里传 1.0。
|
||||
kf_tracks: 可选轨迹关键帧(OAMD 输出,格式 [(obj_id, [(t, x, y, z), ...]), ...]);
|
||||
缺省 = 静止参考位置(adm_atmos 默认)。
|
||||
"""
|
||||
raw = np.memmap(raw16_path, dtype=np.float32, mode="r")
|
||||
n = len(raw) // 16
|
||||
raw = raw[:n * 16].reshape(-1, 16)
|
||||
if duration_sec is None:
|
||||
duration_sec = n / rate
|
||||
# 惰性视图:adm_atmos 按块读取,避免全片 25ch 在内存中展开。
|
||||
class BedView:
|
||||
shape = (n, 10)
|
||||
|
||||
def __getitem__(self, key):
|
||||
src = np.asarray(raw[key], dtype=np.float32)
|
||||
one = src.ndim == 1
|
||||
if one:
|
||||
src = src[None, :]
|
||||
out = np.zeros((len(src), 10), dtype=np.float32)
|
||||
out[:, 3] = np.multiply(src[:, 0], np.float32(scale), dtype=np.float32)
|
||||
return out[0] if one else out
|
||||
|
||||
class ObjView:
|
||||
shape = (n, 16)
|
||||
|
||||
def __getitem__(self, key):
|
||||
return np.multiply(np.asarray(raw[key], dtype=np.float32),
|
||||
np.float32(scale), dtype=np.float32)
|
||||
if kf_tracks is None:
|
||||
kf_tracks = []
|
||||
for oi in range(15):
|
||||
# 静止参考位置(q1=q2=q3=0 → 原点;实际坐标按 OAMD 输出填入)
|
||||
kf_tracks.append(("JOC_Object_%d" % (oi + 1),
|
||||
[(0.0, 0.0, 0.0, 0.0, max(duration_sec, 1e-6))]))
|
||||
adm_atmos.build_master(out_path, BedView(), ObjView(), kf_tracks,
|
||||
duration_sec, rate=rate,
|
||||
joc_binaural_mode=joc_binaural_mode)
|
||||
# 及时释放 Windows 文件句柄,允许 TemporaryDirectory 删除中间 raw。
|
||||
raw._mmap.close()
|
||||
return out_path
|
||||
|
||||
|
||||
class StreamingMaster:
|
||||
"""Incrementally write renderer frames into the final 25-channel ADM BWF.
|
||||
|
||||
This removes the default 16-channel float32 intermediate file. The mapping
|
||||
remains identical to :func:`assemble_from_raw`: bed channel 3 receives LFE,
|
||||
bed channels 0..2/4..9 are silent, and output objects 1..15 map to ADM
|
||||
channels 10..24.
|
||||
"""
|
||||
|
||||
def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072,
|
||||
joc_binaural_mode=4):
|
||||
if block_samples < 1536:
|
||||
raise ValueError("block_samples must be at least one E-AC-3 frame")
|
||||
self.out_path = os.fspath(out_path)
|
||||
self.duration_sec = float(duration_sec)
|
||||
self.rate = int(rate)
|
||||
self.joc_binaural_mode = joc_binaural_mode
|
||||
self._sink = adm_atmos.Sink25(self.out_path, 25, self.rate)
|
||||
self._buffer = np.empty((int(block_samples), 25), dtype=np.float32)
|
||||
self._used = 0
|
||||
self._finalized = False
|
||||
|
||||
def _flush(self):
|
||||
if self._used:
|
||||
self._sink.write_block(self._buffer[:self._used])
|
||||
self._used = 0
|
||||
|
||||
def write_frame(self, pcm16):
|
||||
pcm = np.asarray(pcm16, dtype=np.float32)
|
||||
if pcm.shape != (16, 1536):
|
||||
raise ValueError(f"renderer frame must be (16,1536), got {pcm.shape}")
|
||||
source = 0
|
||||
while source < 1536:
|
||||
available = len(self._buffer) - self._used
|
||||
count = min(available, 1536 - source)
|
||||
target = self._buffer[self._used:self._used + count]
|
||||
target.fill(0.0)
|
||||
target[:, 3] = pcm[0, source:source + count]
|
||||
target[:, 10:25] = pcm[1:16, source:source + count].T
|
||||
self._used += count
|
||||
source += count
|
||||
if self._used == len(self._buffer):
|
||||
self._flush()
|
||||
|
||||
def finalize(self, kf_tracks):
|
||||
if self._finalized:
|
||||
raise RuntimeError("StreamingMaster already finalized")
|
||||
self._flush()
|
||||
try:
|
||||
from . import adm_serializer
|
||||
except ImportError:
|
||||
import adm_serializer
|
||||
axml = adm_serializer.build_axml(kf_tracks, self.duration_sec)
|
||||
chna = adm_atmos.build_chna()
|
||||
dbmd = adm_atmos.build_dbmd(
|
||||
25, joc_binaural_mode=self.joc_binaural_mode)
|
||||
trajectory_blocks = sum(len(track[1]) for track in kf_tracks)
|
||||
self.metadata_info = {
|
||||
"axml_bytes": len(axml),
|
||||
"trajectory_blocks": trajectory_blocks,
|
||||
"chna_bytes": len(chna),
|
||||
"dbmd_bytes": len(dbmd),
|
||||
}
|
||||
self._sink.finalize(axml, chna, dbmd)
|
||||
self._finalized = True
|
||||
print(f"master25 -> {self.out_path} ({self.duration_sec:.2f}s, 25ch, "
|
||||
f"axml={len(axml)}B, chna={len(chna)}B, dbmd={len(dbmd)}B)")
|
||||
return self.out_path
|
||||
|
||||
def abort(self):
|
||||
if self._finalized:
|
||||
return
|
||||
sink = getattr(self, "_sink", None)
|
||||
fp = getattr(sink, "fp", None)
|
||||
if fp is not None and not fp.closed:
|
||||
fp.close()
|
||||
|
||||
def __del__(self):
|
||||
try:
|
||||
self.abort()
|
||||
except Exception:
|
||||
pass
|
||||
@@ -1,292 +0,0 @@
|
||||
"""生成 25 声道 RF64 ADM BWF 及其 axml、chna、dbmd 元数据。
|
||||
|
||||
输出由 10 声道 7.1.2 bed 和 15 路对象组成;RF64 尺寸字段在写入完成后回填。
|
||||
"""
|
||||
import operator
|
||||
import struct
|
||||
import numpy as np
|
||||
import xml.etree.ElementTree as ET
|
||||
|
||||
NS = "urn:ebu:metadata-schema:ebuCore_2016"
|
||||
XSI = "http://www.w3.org/2001/XMLSchema-instance"
|
||||
|
||||
BED_NAMES = ["RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter",
|
||||
"RoomCentricLFE", "RoomCentricLeftSideSurround",
|
||||
"RoomCentricRightSideSurround", "RoomCentricLeftRearSurround",
|
||||
"RoomCentricRightRearSurround", "RoomCentricLeftTopSurround",
|
||||
"RoomCentricRightTopSurround"]
|
||||
BED_LABELS = ["RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss", "RC_Rss",
|
||||
"RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"]
|
||||
BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0),
|
||||
(-1.0, 1.0, -1.0), (-1.0, 0.0, 0.0), (1.0, 0.0, 0.0),
|
||||
(-1.0, -1.0, 0.0), (1.0, -1.0, 0.0), (-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
|
||||
|
||||
N_OBJ = 15
|
||||
JOC_BINAURAL_MODES = {
|
||||
"off": 0,
|
||||
"near": 1,
|
||||
"far": 2,
|
||||
"mid": 3,
|
||||
"unspecified": 4,
|
||||
}
|
||||
JOC_BINAURAL_MODE_DEFAULT = "unspecified"
|
||||
|
||||
def q_to_adm_xyz(q1, q2, q3):
|
||||
posX = min(1.0, round(q1 * 62 / 32767.0) / 62.0)
|
||||
posY = min(1.0, round(q2 * 62 / 32767.0) / 62.0)
|
||||
posZ = round(q3 * 15 / 32767.0) / 15.0
|
||||
posZ = max(-1.0, min(1.0, posZ))
|
||||
return posX * 2 - 1, 1 - posY * 2, posZ
|
||||
|
||||
def ts(seconds):
|
||||
s = int(seconds)
|
||||
frac = int(round((seconds - s) * 100000))
|
||||
if frac >= 100000:
|
||||
s += 1; frac = 0
|
||||
return f"{s // 3600:02d}:{s % 3600 // 60:02d}:{s % 60:02d}.{frac:05d}"
|
||||
|
||||
def sub(parent, tag, attrib=None, text=None):
|
||||
e = ET.SubElement(parent, tag)
|
||||
if attrib:
|
||||
for k, v in attrib.items():
|
||||
e.set(k, v)
|
||||
if text is not None:
|
||||
e.text = text
|
||||
return e
|
||||
|
||||
def add_refs(parent, tag, ids):
|
||||
for i in ids:
|
||||
sub(parent, tag, text=i)
|
||||
|
||||
def obj_block(cf, bid, t, x, y, z, dur, interpolation=0.0):
|
||||
b = sub(cf, "audioBlockFormat", {
|
||||
"audioBlockFormatID": bid, "rtime": ts(t), "duration": ts(dur)})
|
||||
sub(b, "cartesian", text="1")
|
||||
for c, v in (("X", x), ("Y", y), ("Z", z)):
|
||||
if c == "Z" and v == 0:
|
||||
continue
|
||||
p = sub(b, "position", {"coordinate": c})
|
||||
p.text = f"{v:.10f}"
|
||||
sub(b, "jumpPosition", {"interpolationLength": f"{interpolation:.5f}"}, text="1")
|
||||
|
||||
def build_axml(obj_tracks, duration_sec):
|
||||
adm = ET.Element("ebuCoreMain", {
|
||||
"xmlns": NS, "xmlns:xsi": XSI,
|
||||
"xsi:schemaLocation": f"{NS} ebucore.xsd", "lang": "en"})
|
||||
core = sub(adm, "coreMetadata")
|
||||
fmt = sub(core, "format")
|
||||
af = sub(fmt, "audioFormatExtended")
|
||||
|
||||
prog = sub(af, "audioProgramme", {
|
||||
"audioProgrammeID": "APR_1001", "audioProgrammeName": "EAC3JOC_Export",
|
||||
"start": ts(0), "end": ts(duration_sec)})
|
||||
add_refs(prog, "audioContentIDRef", ("ACO_1001", "ACO_1002"))
|
||||
bc = sub(af, "audioContent", {"audioContentID": "ACO_1001",
|
||||
"audioContentName": "EAC3JOC_Master_Content"})
|
||||
add_refs(bc, "audioObjectIDRef", ["AO_1001"])
|
||||
sub(bc, "dialogue", {"mixedContentKind": "0"})
|
||||
oc = sub(af, "audioContent", {"audioContentID": "ACO_1002",
|
||||
"audioContentName": "Objects"})
|
||||
add_refs(oc, "audioObjectIDRef", ["AO_%04x" % (0x100b + i) for i in range(N_OBJ)])
|
||||
sub(oc, "dialogue", {"mixedContentKind": "0"})
|
||||
|
||||
bed_o = sub(af, "audioObject", {"audioObjectID": "AO_1001", "audioObjectName": "Bed",
|
||||
"start": ts(0), "duration": ts(duration_sec)})
|
||||
sub(bed_o, "audioPackFormatIDRef", text="AP_00011001")
|
||||
add_refs(bed_o, "audioTrackUIDRef", ["ATU_%08x" % (i + 1) for i in range(10)])
|
||||
for i in range(N_OBJ):
|
||||
o = sub(af, "audioObject", {"audioObjectID": "AO_%04x" % (0x100b + i),
|
||||
"audioObjectName": f"Audio Object {i+1}",
|
||||
"start": ts(0), "duration": ts(duration_sec)})
|
||||
sub(o, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
|
||||
add_refs(o, "audioTrackUIDRef", ["ATU_%08x" % (i + 11)])
|
||||
|
||||
bp = sub(af, "audioPackFormat", {"audioPackFormatID": "AP_00011001",
|
||||
"audioPackFormatName": "EAC3JOCBedPack",
|
||||
"typeDefinition": "DirectSpeakers", "typeLabel": "0001"})
|
||||
add_refs(bp, "audioChannelFormatIDRef", ["AC_0001%04x" % (0x1001 + i) for i in range(10)])
|
||||
for i in range(N_OBJ):
|
||||
pk = sub(af, "audioPackFormat", {"audioPackFormatID": "AP_0003%04x" % (0x1001 + i),
|
||||
"audioPackFormatName": f"JOC_Object_{i+1}",
|
||||
"typeDefinition": "Objects", "typeLabel": "0003"})
|
||||
add_refs(pk, "audioChannelFormatIDRef", ["AC_0003%04x" % (0x1001 + i)])
|
||||
|
||||
for i in range(10):
|
||||
cf = sub(af, "audioChannelFormat", {"audioChannelFormatID": "AC_0001%04x" % (0x1001 + i),
|
||||
"audioChannelFormatName": BED_NAMES[i],
|
||||
"typeDefinition": "DirectSpeakers", "typeLabel": "0001"})
|
||||
b = sub(cf, "audioBlockFormat", {"audioBlockFormatID": "AB_0001%04x_00000001" % (0x1001 + i)})
|
||||
sub(b, "cartesian", text="1")
|
||||
x, y, z = BED_POS[i]
|
||||
for c, v in (("X", x), ("Y", y), ("Z", z)):
|
||||
if c == "Z" and v == 0:
|
||||
continue
|
||||
p = sub(b, "position", {"coordinate": c})
|
||||
p.text = f"{v:.10f}"
|
||||
sub(b, "speakerLabel", text=BED_LABELS[i])
|
||||
|
||||
for i, (oname, kfs) in enumerate(obj_tracks):
|
||||
cf = sub(af, "audioChannelFormat", {"audioChannelFormatID": "AC_0003%04x" % (0x1001 + i),
|
||||
"audioChannelFormatName": oname,
|
||||
"typeDefinition": "Objects", "typeLabel": "0003"})
|
||||
for k, keyframe in enumerate(kfs):
|
||||
t, x, y, z, dur = keyframe[:5]
|
||||
interpolation = keyframe[5] if len(keyframe) > 5 else 0.0
|
||||
obj_block(cf, "AB_0003%04x_%08x" % (0x1001 + i, k + 1),
|
||||
t, x, y, z, dur, interpolation)
|
||||
|
||||
for i in range(10):
|
||||
t = sub(af, "audioTrackUID", {"UID": "ATU_%08x" % (i + 1),
|
||||
"bitDepth": "24", "sampleRate": "48000"})
|
||||
sub(t, "audioTrackFormatIDRef", text="AT_0001%04x_01" % (0x1001 + i))
|
||||
sub(t, "audioPackFormatIDRef", text="AP_00011001")
|
||||
for i in range(N_OBJ):
|
||||
t = sub(af, "audioTrackUID", {"UID": "ATU_%08x" % (i + 11),
|
||||
"bitDepth": "24", "sampleRate": "48000"})
|
||||
sub(t, "audioTrackFormatIDRef", text="AT_0003%04x_01" % (0x1001 + i))
|
||||
sub(t, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
|
||||
|
||||
for i in range(10):
|
||||
tf = sub(af, "audioTrackFormat", {"audioTrackFormatID": "AT_0001%04x_01" % (0x1001 + i),
|
||||
"audioTrackFormatName": "PCM_" + BED_NAMES[i],
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(tf, "audioStreamFormatIDRef", text="AS_0001%04x" % (0x1001 + i))
|
||||
for i in range(N_OBJ):
|
||||
tf = sub(af, "audioTrackFormat", {"audioTrackFormatID": "AT_0003%04x_01" % (0x1001 + i),
|
||||
"audioTrackFormatName": "PCM_JOC_Object_%d" % (i + 1),
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(tf, "audioStreamFormatIDRef", text="AS_0003%04x" % (0x1001 + i))
|
||||
|
||||
for i in range(10):
|
||||
sf = sub(af, "audioStreamFormat", {"audioStreamFormatID": "AS_0001%04x" % (0x1001 + i),
|
||||
"audioStreamFormatName": "PCM_" + BED_NAMES[i],
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(sf, "audioChannelFormatIDRef", text="AC_0001%04x" % (0x1001 + i))
|
||||
sub(sf, "audioPackFormatIDRef", text="AP_00011001")
|
||||
sub(sf, "audioTrackFormatIDRef", text="AT_0001%04x_01" % (0x1001 + i))
|
||||
for i in range(N_OBJ):
|
||||
sf = sub(af, "audioStreamFormat", {"audioStreamFormatID": "AS_0003%04x" % (0x1001 + i),
|
||||
"audioStreamFormatName": "PCM_JOC_Object_%d" % (i + 1),
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(sf, "audioChannelFormatIDRef", text="AC_0003%04x" % (0x1001 + i))
|
||||
sub(sf, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
|
||||
sub(sf, "audioTrackFormatIDRef", text="AT_0003%04x_01" % (0x1001 + i))
|
||||
|
||||
return ET.tostring(adm, encoding="utf-8", xml_declaration=True)
|
||||
|
||||
def build_chna():
|
||||
out = bytearray()
|
||||
out += struct.pack("<HH", 25, 25)
|
||||
for i in range(10):
|
||||
out += struct.pack("<H", i + 1)
|
||||
out += ("ATU_%08x" % (i + 1)).encode()
|
||||
out += ("AT_0001%04x_01" % (0x1001 + i)).encode()
|
||||
out += b"AP_00011001" + b"\x00"
|
||||
for i in range(N_OBJ):
|
||||
out += struct.pack("<H", i + 11)
|
||||
out += ("ATU_%08x" % (i + 11)).encode()
|
||||
out += ("AT_0003%04x_01" % (0x1001 + i)).encode()
|
||||
out += ("AP_0003%04x" % (0x1001 + i)).encode() + b"\x00"
|
||||
return bytes(out)
|
||||
|
||||
def _checksum(seg):
|
||||
s = len(seg)
|
||||
for b in seg:
|
||||
s += b
|
||||
return (~s + 1) & 0xFF
|
||||
|
||||
def build_dbmd(object_count=25, joc_binaural_mode=4):
|
||||
"""仅覆盖 segment 10 中 JOC object slots 10..24 的 mode 低 3 bit。"""
|
||||
mode = operator.index(joc_binaural_mode)
|
||||
if mode not in JOC_BINAURAL_MODES.values():
|
||||
raise ValueError(f"invalid JOC binaural render mode: {mode}")
|
||||
out = bytearray(struct.pack("<I", 0x01000006))
|
||||
dd = bytearray(96)
|
||||
dd[1] = 0x47
|
||||
dd[5] = 0x60
|
||||
dd[8] = 0x24; dd[9] = 0x24
|
||||
out.append(7); out += struct.pack("<H", 96); out += bytes(dd)
|
||||
out.append(_checksum(dd))
|
||||
at = bytearray(248)
|
||||
c0 = b"Created with EAC3JOC"; c1 = b"EAC3JOC Python Renderer"
|
||||
at[0:len(c0)] = c0
|
||||
at[32:32 + len(c1)] = c1
|
||||
at[96], at[97], at[98] = 2, 1, 0
|
||||
at[103] = 0x03
|
||||
at[106] = 0x01
|
||||
at[111] = 0x22; at[112] = 0xFF
|
||||
out.append(9); out += struct.pack("<H", 248); out += bytes(at)
|
||||
out.append(_checksum(at))
|
||||
ob = bytearray(5 + 262 + object_count)
|
||||
ob[0:4] = struct.pack("<I", 0xF8726FBD)
|
||||
ob[4] = object_count
|
||||
for i in range(5 + 262, len(ob)):
|
||||
ob[i] = 0x84
|
||||
# sync (4), count (2), reserved (1), nine 15-byte config trims,
|
||||
# then one trim-bypass byte per track before the headphone modes.
|
||||
# Preserve the existing template's bed fields and trailing bytes.
|
||||
object_modes = 4 + 2 + 1 + 9 * 15 + object_count
|
||||
for i in range(10, min(object_count, 10 + N_OBJ)):
|
||||
ob[object_modes + i] = (ob[object_modes + i] & 0xF8) | mode
|
||||
out.append(10); out += struct.pack("<H", len(ob)); out += bytes(ob)
|
||||
out.append(_checksum(ob))
|
||||
out += b"\x00\x00"
|
||||
return bytes(out)
|
||||
|
||||
class Sink25:
|
||||
def __init__(self, path, channels, rate):
|
||||
self.ch = channels; self.rate = rate; self.frames = 0
|
||||
self.fp = open(path, "wb+")
|
||||
self.fp.write(b"RF64" + struct.pack("<I", 0xFFFFFFFF) + b"WAVE")
|
||||
self._chunk(b"ds64", b"\x00" * 64)
|
||||
self._chunk(b"fmt ", self._fmt())
|
||||
self._chunk(b"data", b"")
|
||||
def _chunk(self, cid, body):
|
||||
self.fp.write(cid + struct.pack("<I", len(body)) + body)
|
||||
if len(body) & 1:
|
||||
self.fp.write(b"\x00")
|
||||
def _fmt(self):
|
||||
return struct.pack("<HHIIHH", 1, self.ch, self.rate,
|
||||
self.rate * self.ch * 3, self.ch * 3, 24)
|
||||
def write_block(self, arr):
|
||||
arr = arr.reshape(-1, self.ch)
|
||||
i24 = (np.clip(arr, -1.0, 1.0) * 8388607.0).astype(np.int32)
|
||||
self.fp.write(i24.view(np.uint8).reshape(-1, 4)[:, :3].tobytes())
|
||||
self.frames += arr.shape[0]
|
||||
def finalize(self, axml_bytes, chna_bytes, dbmd_bytes):
|
||||
data_len = self.frames * self.ch * 3
|
||||
self._chunk(b"axml", axml_bytes)
|
||||
self._chunk(b"chna", chna_bytes)
|
||||
self._chunk(b"dbmd", dbmd_bytes)
|
||||
self.fp.seek(0, 2); total = self.fp.tell()
|
||||
self.fp.seek(0); head = self.fp.read()
|
||||
m = head.find(b"data")
|
||||
if m >= 0:
|
||||
self.fp.seek(m + 4); self.fp.write(struct.pack("<I", data_len))
|
||||
m = head.find(b"ds64")
|
||||
if m >= 0:
|
||||
self.fp.seek(m + 8)
|
||||
self.fp.write(struct.pack("<QQQI", total - 8, data_len, self.frames, 0))
|
||||
self.fp.flush()
|
||||
self.fp.close()
|
||||
|
||||
def build_master(out_path, bed_mm, obj_mm, kf_tracks, duration_sec, rate=48000,
|
||||
block=480000, joc_binaural_mode=4):
|
||||
n = min(bed_mm.shape[0], obj_mm.shape[0])
|
||||
try:
|
||||
from . import adm_serializer
|
||||
except ImportError:
|
||||
import adm_serializer
|
||||
serial_axml = adm_serializer.build_axml
|
||||
axml = serial_axml(kf_tracks, duration_sec)
|
||||
chna = build_chna()
|
||||
dbmd = build_dbmd(25, joc_binaural_mode=joc_binaural_mode)
|
||||
sink = Sink25(out_path, 25, rate)
|
||||
for st in range(0, n, block):
|
||||
en = min(n, st + block)
|
||||
blk = np.hstack((np.asarray(bed_mm[st:en], dtype=np.float32),
|
||||
np.asarray(obj_mm[st:en, 1:16], dtype=np.float32)))
|
||||
sink.write_block(blk)
|
||||
sink.finalize(axml, chna, dbmd)
|
||||
print(f"master25 -> {out_path} ({duration_sec:.2f}s, 25ch, axml={len(axml)}B, "
|
||||
f"chna={len(chna)}B, dbmd={len(dbmd)}B)")
|
||||
@@ -1,136 +0,0 @@
|
||||
"""把 7.1.2 bed、15 个对象及其位置轨迹序列化为 ADM axml。
|
||||
|
||||
序列化结果采用固定元素顺序、属性顺序和十六进制 ADM 标识符,便于稳定输出和校验。
|
||||
"""
|
||||
BED_NAMES = ["RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter", "RoomCentricLFE",
|
||||
"RoomCentricLeftSideSurround", "RoomCentricRightSideSurround",
|
||||
"RoomCentricLeftRearSurround", "RoomCentricRightRearSurround",
|
||||
"RoomCentricLeftTopSurround", "RoomCentricRightTopSurround"]
|
||||
BED_LABELS = ["RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss", "RC_Rss",
|
||||
"RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"]
|
||||
BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0), (-1.0, 1.0, -1.0),
|
||||
(-1.0, 0.0, 0.0), (1.0, 0.0, 0.0), (-1.0, -1.0, 0.0), (1.0, -1.0, 0.0),
|
||||
(-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
|
||||
N_OBJ = 15
|
||||
|
||||
def ts(seconds):
|
||||
s = int(seconds)
|
||||
frac = int(round((seconds - s) * 100000))
|
||||
if frac >= 100000:
|
||||
s += 1; frac = 0
|
||||
return f"{s // 3600:02d}:{s % 3600 // 60:02d}:{s % 60:02d}.{frac:05d}"
|
||||
|
||||
def esc(v):
|
||||
return (str(v).replace("&", "&").replace("<", "<").replace(">", ">"))
|
||||
|
||||
def build_axml(obj_tracks, duration_sec):
|
||||
"""obj_tracks: [(name, [(rtime, x, y, z, dur), ...]) ×15]"""
|
||||
w = []
|
||||
a = w.append
|
||||
a('<?xml version="1.0" encoding="utf-8"?>')
|
||||
a('<ebuCoreMain xsi:schemaLocation="urn:ebu:metadata-schema:ebuCore_2016 ebucore.xsd" '
|
||||
'lang="en" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" '
|
||||
'xmlns="urn:ebu:metadata-schema:ebuCore_2016">')
|
||||
a('<coreMetadata><format><audioFormatExtended>')
|
||||
a(f'<audioProgramme audioProgrammeID="APR_1001" audioProgrammeName="EAC3JOC_Export" '
|
||||
f'start="{ts(0)}" end="{ts(duration_sec)}">')
|
||||
a('<audioContentIDRef>ACO_1001</audioContentIDRef>')
|
||||
a('<audioContentIDRef>ACO_1002</audioContentIDRef>')
|
||||
a('</audioProgramme>')
|
||||
a('<audioContent audioContentID="ACO_1001" audioContentName="EAC3JOC_Master_Content">')
|
||||
a('<audioObjectIDRef>AO_1001</audioObjectIDRef>')
|
||||
a('<dialogue mixedContentKind="0">2</dialogue>')
|
||||
a('</audioContent>')
|
||||
a('<audioContent audioContentID="ACO_1002" audioContentName="Objects">')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioObjectIDRef>AO_{0x100b + i:04x}</audioObjectIDRef>')
|
||||
a('<dialogue mixedContentKind="0">2</dialogue>')
|
||||
a('</audioContent>')
|
||||
a(f'<audioObject audioObjectID="AO_1001" audioObjectName="Bed" '
|
||||
f'start="{ts(0)}" duration="{ts(duration_sec)}">')
|
||||
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
|
||||
for i in range(10):
|
||||
a(f'<audioTrackUIDRef>ATU_{i + 1:08x}</audioTrackUIDRef>')
|
||||
a('</audioObject>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioObject audioObjectID="AO_{0x100b + i:04x}" audioObjectName="Audio Object {i+1}" '
|
||||
f'start="{ts(0)}" duration="{ts(duration_sec)}">')
|
||||
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
|
||||
a(f'<audioTrackUIDRef>ATU_{11 + i:08x}</audioTrackUIDRef>')
|
||||
a('</audioObject>')
|
||||
a('<audioPackFormat audioPackFormatID="AP_00011001" audioPackFormatName="EAC3JOCBedPack" '
|
||||
'typeDefinition="DirectSpeakers" typeLabel="0001">')
|
||||
for i in range(10):
|
||||
a(f'<audioChannelFormatIDRef>AC_0001{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a('</audioPackFormat>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioPackFormat audioPackFormatID="AP_0003{0x1001 + i:04x}" '
|
||||
f'audioPackFormatName="JOC_Object_{i+1}" typeDefinition="Objects" typeLabel="0003">')
|
||||
a(f'<audioChannelFormatIDRef>AC_0003{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a('</audioPackFormat>')
|
||||
for i in range(10):
|
||||
a(f'<audioChannelFormat audioChannelFormatID="AC_0001{0x1001 + i:04x}" '
|
||||
f'audioChannelFormatName="{BED_NAMES[i]}" typeDefinition="DirectSpeakers" typeLabel="0001">')
|
||||
a(f'<audioBlockFormat audioBlockFormatID="AB_0001{0x1001 + i:04x}_00000001">')
|
||||
a('<cartesian>1</cartesian>')
|
||||
x, y, z = BED_POS[i]
|
||||
a(f'<position coordinate="X">{x:.10f}</position>')
|
||||
a(f'<position coordinate="Y">{y:.10f}</position>')
|
||||
if z != 0:
|
||||
a(f'<position coordinate="Z">{z:.10f}</position>')
|
||||
a(f'<speakerLabel>{BED_LABELS[i]}</speakerLabel>')
|
||||
a('</audioBlockFormat>')
|
||||
a('</audioChannelFormat>')
|
||||
for i, (oname, kfs) in enumerate(obj_tracks):
|
||||
a(f'<audioChannelFormat audioChannelFormatID="AC_0003{0x1001 + i:04x}" '
|
||||
f'audioChannelFormatName="{oname}" typeDefinition="Objects" typeLabel="0003">')
|
||||
for k, keyframe in enumerate(kfs):
|
||||
t, x, y, z, dur = keyframe[:5]
|
||||
interpolation = keyframe[5] if len(keyframe) > 5 else 0.0
|
||||
a(f'<audioBlockFormat audioBlockFormatID="AB_0003{0x1001 + i:04x}_{k + 1:08x}" '
|
||||
f'rtime="{ts(t)}" duration="{ts(dur)}">')
|
||||
a('<cartesian>1</cartesian>')
|
||||
a(f'<position coordinate="X">{x:.10f}</position>')
|
||||
a(f'<position coordinate="Y">{y:.10f}</position>')
|
||||
if z != 0:
|
||||
a(f'<position coordinate="Z">{z:.10f}</position>')
|
||||
a(f'<jumpPosition interpolationLength="{interpolation:.5f}">1</jumpPosition>')
|
||||
a('</audioBlockFormat>')
|
||||
a('</audioChannelFormat>')
|
||||
for i in range(10):
|
||||
a(f'<audioTrackUID UID="ATU_{i + 1:08x}" bitDepth="24" sampleRate="48000">')
|
||||
a(f'<audioTrackFormatIDRef>AT_0001{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
|
||||
a('</audioTrackUID>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioTrackUID UID="ATU_{11 + i:08x}" bitDepth="24" sampleRate="48000">')
|
||||
a(f'<audioTrackFormatIDRef>AT_0003{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
|
||||
a('</audioTrackUID>')
|
||||
for i in range(10):
|
||||
a(f'<audioTrackFormat audioTrackFormatID="AT_0001{0x1001 + i:04x}_01" '
|
||||
f'audioTrackFormatName="PCM_{BED_NAMES[i]}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioStreamFormatIDRef>AS_0001{0x1001 + i:04x}</audioStreamFormatIDRef>')
|
||||
a('</audioTrackFormat>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioTrackFormat audioTrackFormatID="AT_0003{0x1001 + i:04x}_01" '
|
||||
f'audioTrackFormatName="PCM_JOC_Object_{i+1}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioStreamFormatIDRef>AS_0003{0x1001 + i:04x}</audioStreamFormatIDRef>')
|
||||
a('</audioTrackFormat>')
|
||||
for i in range(10):
|
||||
a(f'<audioStreamFormat audioStreamFormatID="AS_0001{0x1001 + i:04x}" '
|
||||
f'audioStreamFormatName="PCM_{BED_NAMES[i]}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioChannelFormatIDRef>AC_0001{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
|
||||
a(f'<audioTrackFormatIDRef>AT_0001{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a('</audioStreamFormat>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioStreamFormat audioStreamFormatID="AS_0003{0x1001 + i:04x}" '
|
||||
f'audioStreamFormatName="PCM_JOC_Object_{i+1}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioChannelFormatIDRef>AC_0003{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
|
||||
a(f'<audioTrackFormatIDRef>AT_0003{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a('</audioStreamFormat>')
|
||||
a('</audioFormatExtended></format></coreMetadata>')
|
||||
a('</ebuCoreMain>')
|
||||
return ''.join(w).encode('utf-8')
|
||||
@@ -1,199 +0,0 @@
|
||||
"""校验 ADM BWF 的 RF64、通道、axml、chna、dbmd 和对象引用结构。
|
||||
|
||||
用法:``python src/adm_validate.py <file.wav> [more.wav ...]``,全部通过时退出码为 0。
|
||||
"""
|
||||
import struct, sys, re, os
|
||||
|
||||
def fail(msgs, m): msgs.append(m)
|
||||
|
||||
def walk_chunks(path):
|
||||
chunks, ds64 = [], {}
|
||||
with open(path, "rb") as f:
|
||||
riff = f.read(4); f.read(4); wave = f.read(4)
|
||||
if riff not in (b"RIFF", b"RF64"):
|
||||
return None, None, f"File does not have a 'RIFF' or 'RF64' chunk"
|
||||
if wave != b"WAVE":
|
||||
return None, None, "File does not have a required 'WAVE' chunk"
|
||||
while True:
|
||||
off = f.tell()
|
||||
cid = f.read(4)
|
||||
if len(cid) < 4: break
|
||||
sz = struct.unpack("<I", f.read(4))[0]
|
||||
if cid == b"ds64":
|
||||
body = f.read(sz + (sz & 1))
|
||||
riff64, data64, sample64, _ = struct.unpack("<QQQI", body[:28])
|
||||
ds64 = dict(riff64=riff64, data64=data64, sample64=sample64)
|
||||
chunks.append(("ds64", off, sz)); continue
|
||||
chunks.append((cid.decode("latin1"), off, sz))
|
||||
eff = ds64.get("data64", sz) if (sz == 0xFFFFFFFF and cid == b"data") else sz
|
||||
f.seek(off + 8 + eff + (eff & 1))
|
||||
return chunks, ds64, None
|
||||
|
||||
def read_body(path, chunks, cid):
|
||||
for c, off, sz in chunks:
|
||||
if c == cid:
|
||||
with open(path, "rb") as f:
|
||||
f.seek(off + 8)
|
||||
return f.read(sz)
|
||||
return None
|
||||
|
||||
def parse_chna(body):
|
||||
n_track, n_uid = struct.unpack("<HH", body[:4])
|
||||
rows, p = [], 4
|
||||
while p + 40 <= len(body):
|
||||
trk = struct.unpack("<H", body[p:p+2])[0]
|
||||
uid = body[p+2:p+14].rstrip(b"\x00").decode()
|
||||
tf = body[p+14:p+28].rstrip(b"\x00").decode()
|
||||
pk = body[p+28:p+40].rstrip(b"\x00").decode()
|
||||
rows.append((trk, uid, tf, pk)); p += 40
|
||||
return n_track, n_uid, rows
|
||||
|
||||
def decode_channel_input(ao_id):
|
||||
"""将 ``AO_xxxx`` 的十六进制标识符解码为低 12 位通道输入号。"""
|
||||
m = re.fullmatch(r"AO_([0-9a-fA-F]+)", ao_id)
|
||||
if not m:
|
||||
return None
|
||||
v = int(m.group(1), 16)
|
||||
if v > 0x1FFF: # 超过 12 位通道域
|
||||
return None
|
||||
return v & 0x0FFF
|
||||
|
||||
def ts_sec(s):
|
||||
h, m, rest = s.split(":")
|
||||
return int(h) * 3600 + int(m) * 60 + float(rest)
|
||||
|
||||
def validate(path, axml_override=None, chna_override=None):
|
||||
msgs = []
|
||||
chunks, ds64, err = walk_chunks(path)
|
||||
if err:
|
||||
return [err]
|
||||
have = {c for c, _, _ in chunks}
|
||||
for need in ("fmt ", "data", "axml", "chna", "dbmd"):
|
||||
if need not in have:
|
||||
fail(msgs, f"File does not have a required '{need.strip()}' chunk")
|
||||
if msgs:
|
||||
return msgs
|
||||
fmt = read_body(path, chunks, "fmt ")
|
||||
f_tag, f_ch, f_rate, _, _, f_bits = struct.unpack("<HHIIHH", fmt[:16])
|
||||
chna_body = chna_override if chna_override is not None else read_body(path, chunks, "chna")
|
||||
n_track, n_uid, rows = parse_chna(chna_body)
|
||||
if f_ch != n_track:
|
||||
fail(msgs, f"Mismatched number of audio channels and chna entries "
|
||||
f"(fmt={f_ch} chna={n_track})")
|
||||
ax_raw = axml_override if axml_override is not None else read_body(path, chunks, "axml")
|
||||
ax = ax_raw.decode("utf-8")
|
||||
|
||||
# --- audioObjectID 十六进制通道输入号解码 ---
|
||||
objs = re.findall(r'audioObjectID="(AO_[0-9a-zA-Z]+)"', ax)
|
||||
bed_ch, obj_ch = [], []
|
||||
for ao in objs:
|
||||
cid = decode_channel_input(ao)
|
||||
if cid is None:
|
||||
fail(msgs, f"Invalid ADM BWF XML format: cannot decode channel "
|
||||
f"input ID from AudioObjectID '{ao}'")
|
||||
continue
|
||||
if ao == "AO_1001":
|
||||
bed_ch.append(cid)
|
||||
else:
|
||||
if cid <= 10:
|
||||
fail(msgs, f"Source channel index should be greater than 10 "
|
||||
f"for objects ('{ao}' -> {cid})")
|
||||
obj_ch.append(cid)
|
||||
|
||||
# UID 十六进制 → 必须与 chna 表一致
|
||||
uid_map = {uid: trk for trk, uid, tf, pk in rows}
|
||||
for uid in re.findall(r'UID="(ATU_[0-9a-zA-Z]+)"', ax):
|
||||
if uid not in uid_map:
|
||||
fail(msgs, f"'{uid}' is not referenced in 'chna' chunk UID table")
|
||||
continue
|
||||
m = re.fullmatch(r"ATU_([0-9a-fA-F]+)", uid)
|
||||
if m:
|
||||
v = int(m.group(1), 16)
|
||||
if v > 128:
|
||||
fail(msgs, f"Channel index out of range (UID {uid} -> {v})")
|
||||
|
||||
# 轨数一致性:axml audioTrackUID 数 == fmt 声道数
|
||||
n_tu = len(re.findall(r"<audioTrackUID ", ax))
|
||||
if n_tu != f_ch:
|
||||
fail(msgs, f"Number of channels declared in ADM ({n_tu}) does not "
|
||||
f"match 'fmt ' chunk ({f_ch})")
|
||||
|
||||
# sampleRate / bitDepth 一致
|
||||
for sr in set(re.findall(r'sampleRate="(\d+)"', ax)):
|
||||
if int(sr) != f_rate:
|
||||
fail(msgs, f"Mismatched track sample rate between ADM and WAV ({sr} vs {f_rate})")
|
||||
for bd in set(re.findall(r'bitDepth="(\d+)"', ax)):
|
||||
if int(bd) != f_bits:
|
||||
fail(msgs, f"Mismatched track bit depth between ADM and WAV ({bd} vs {f_bits})")
|
||||
|
||||
# audioProgramme 唯一性 / audioContent ≥1
|
||||
if ax.count("<audioProgramme ") != 1:
|
||||
fail(msgs, "ADM has more than one audioProgramme object -- there must be only one"
|
||||
if ax.count("<audioProgramme ") > 1
|
||||
else "ADM does not have a required audioProgramme object")
|
||||
if "<audioContent " not in ax:
|
||||
fail(msgs, "audioProgramme object does not have a required audioContent object")
|
||||
|
||||
# 每个 channelFormat ≥1 blockFormat + 对象块链连续性
|
||||
cfs = re.findall(r'<audioChannelFormat [^>]*typeLabel="0003".*?</audioChannelFormat>', ax, re.S)
|
||||
object_block_formats = 0
|
||||
for seg in cfs:
|
||||
cf_id = re.search(r'audioChannelFormatID="([^"]+)"', seg).group(1)
|
||||
block_xml = re.findall(r'<audioBlockFormat [^>]*rtime="[^"]+".*?</audioBlockFormat>',
|
||||
seg, re.S)
|
||||
object_block_formats += len(block_xml)
|
||||
blocks = []
|
||||
for block_index, block in enumerate(block_xml, 1):
|
||||
timing = re.search(r'rtime="([^"]+)" duration="([^"]+)"', block)
|
||||
if timing is None:
|
||||
continue
|
||||
rtime, duration = timing.groups()
|
||||
blocks.append((rtime, duration))
|
||||
jump = re.search(
|
||||
r'<jumpPosition interpolationLength="([^"]+)">1</jumpPosition>', block)
|
||||
if jump is not None and float(jump.group(1)) > ts_sec(duration) + 1e-8:
|
||||
fail(msgs, f"Interpolation length exceeds duration in block format "
|
||||
f"{block_index} of {cf_id}: {jump.group(1)} > {duration}")
|
||||
if not blocks:
|
||||
fail(msgs, f"AudioChannelFormat {cf_id} is missing audioBlockFormat sub-element")
|
||||
continue
|
||||
for i in range(len(blocks) - 1):
|
||||
end_i = ts_sec(blocks[i][0]) + ts_sec(blocks[i][1])
|
||||
nxt = ts_sec(blocks[i + 1][0])
|
||||
if abs(end_i - nxt) > 2e-5:
|
||||
fail(msgs, f"Time gap between block format {i+1} and {i+2} of {cf_id}: "
|
||||
f"{end_i:.5f} vs {nxt:.5f}")
|
||||
return msgs, dict(fmt_ch=f_ch, fmt_rate=f_rate, fmt_bits=f_bits,
|
||||
chna=n_track, objects=len(obj_ch), bed=len(bed_ch),
|
||||
trackUIDs=n_tu, axml_bytes=len(ax_raw),
|
||||
audioBlockFormats=ax.count("<audioBlockFormat "),
|
||||
objectBlockFormats=object_block_formats)
|
||||
|
||||
def main():
|
||||
import argparse
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("files", nargs="+")
|
||||
ap.add_argument("--axml-file", default=None, help="用该文件内容替换 wav 内 axml(对照实验)")
|
||||
ap.add_argument("--chna-file", default=None, help="用该文件内容替换 wav 内 chna(对照实验)")
|
||||
a = ap.parse_args()
|
||||
ax_o = open(a.axml_file, "rb").read() if a.axml_file else None
|
||||
ch_o = open(a.chna_file, "rb").read() if a.chna_file else None
|
||||
rc = 0
|
||||
for p in a.files:
|
||||
r = validate(p, axml_override=ax_o, chna_override=ch_o)
|
||||
name = os.path.basename(p)
|
||||
if isinstance(r, list):
|
||||
msgs, info = r, {}
|
||||
else:
|
||||
msgs, info = r
|
||||
if msgs:
|
||||
rc = 1
|
||||
print(f"[FAIL] {name}")
|
||||
for m in msgs:
|
||||
print(" -", m)
|
||||
else:
|
||||
print(f"[PASS] {name} {info}")
|
||||
return rc
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,249 @@
|
||||
|
||||
#include "joc_core.h"
|
||||
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
#include "eac3_transport/eac3_reader.h"
|
||||
#include "emdf/emdf_parser.h"
|
||||
#include "foundation/status.h"
|
||||
#include "joc_bitstream/joc_parser.h"
|
||||
|
||||
namespace {
|
||||
|
||||
thread_local std::string g_detail;
|
||||
|
||||
joc_error finish(const joc::Status& status) {
|
||||
if (status.ok()) {
|
||||
g_detail.clear();
|
||||
return JOC_OK;
|
||||
}
|
||||
g_detail.assign(status.stage());
|
||||
g_detail.append(": ");
|
||||
g_detail.append(status.message());
|
||||
return status.code();
|
||||
}
|
||||
|
||||
joc_error arg_fail(const char* message) {
|
||||
return finish(joc::Status::fail(JOC_ERR_INVALID_ARGUMENT, "core", message));
|
||||
}
|
||||
|
||||
void fill_emdf_info(const joc::emdf::Container& container, joc_emdf_info* out) {
|
||||
std::memset(out, 0, sizeof(*out));
|
||||
out->struct_size = sizeof(joc_emdf_info);
|
||||
out->struct_version = JOC_EMDF_INFO_VERSION;
|
||||
out->start_bit = static_cast<std::uint32_t>(container.start_bit);
|
||||
out->container_bytes = static_cast<std::uint32_t>(container.raw_size);
|
||||
out->payload_count = static_cast<std::uint32_t>(container.payload_count);
|
||||
for (std::size_t i = 0; i < container.payload_count; ++i) {
|
||||
out->payloads[i].id = container.payloads[i].id;
|
||||
out->payloads[i].sample_offset = container.payloads[i].sample_offset;
|
||||
out->payloads[i].bit_offset = static_cast<std::uint32_t>(container.payloads[i].bit_offset);
|
||||
out->payloads[i].size = static_cast<std::uint32_t>(container.payloads[i].size);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
extern "C" {
|
||||
|
||||
std::uint32_t JOC_CALL joc_abi_version(void) { return JOC_ABI_VERSION; }
|
||||
|
||||
const char* JOC_CALL joc_version_string(void) { return "0.1.0-m1"; }
|
||||
|
||||
std::uint32_t JOC_CALL joc_event_size(void) { return static_cast<std::uint32_t>(sizeof(joc_event)); }
|
||||
std::uint32_t JOC_CALL joc_task_config_size(void) {
|
||||
return static_cast<std::uint32_t>(sizeof(joc_task_config));
|
||||
}
|
||||
std::uint32_t JOC_CALL joc_task_result_size(void) {
|
||||
return static_cast<std::uint32_t>(sizeof(joc_task_result));
|
||||
}
|
||||
|
||||
const char* JOC_CALL joc_build_info(void) {
|
||||
static const std::string info = [] {
|
||||
std::string text = "joc_core 0.1.0-m1 (";
|
||||
#if defined(_MSC_VER)
|
||||
text += "msvc " + std::to_string(_MSC_VER);
|
||||
#elif defined(__clang__)
|
||||
text += std::string("clang ") + __clang_version__;
|
||||
#elif defined(__GNUC__)
|
||||
text += "gcc " + std::to_string(__GNUC__) + "." + std::to_string(__GNUC_MINOR__);
|
||||
#else
|
||||
text += "unknown-compiler";
|
||||
#endif
|
||||
#if defined(_M_AMD64) || defined(__x86_64__)
|
||||
text += ", x64";
|
||||
#elif defined(_M_ARM64) || defined(__aarch64__)
|
||||
text += ", arm64";
|
||||
#elif defined(_M_IX86) || defined(__i386__)
|
||||
text += ", x86";
|
||||
#endif
|
||||
text += ", c++";
|
||||
text += std::to_string(static_cast<long long>(__cplusplus / 100 % 100));
|
||||
text += ")";
|
||||
return text;
|
||||
}();
|
||||
return info.c_str();
|
||||
}
|
||||
|
||||
const char* JOC_CALL joc_error_name(joc_error code) {
|
||||
switch (code) {
|
||||
case JOC_OK: return "JOC_OK";
|
||||
case JOC_ERR_INVALID_ARGUMENT: return "JOC_ERR_INVALID_ARGUMENT";
|
||||
case JOC_ERR_INVALID_CONFIG: return "JOC_ERR_INVALID_CONFIG";
|
||||
case JOC_ERR_OUT_OF_MEMORY: return "JOC_ERR_OUT_OF_MEMORY";
|
||||
case JOC_ERR_IO: return "JOC_ERR_IO";
|
||||
case JOC_ERR_UNSUPPORTED_PLATFORM: return "JOC_ERR_UNSUPPORTED_PLATFORM";
|
||||
case JOC_ERR_LIBRARY_MISSING: return "JOC_ERR_LIBRARY_MISSING";
|
||||
case JOC_ERR_INPUT_NOT_FOUND: return "JOC_ERR_INPUT_NOT_FOUND";
|
||||
case JOC_ERR_INPUT_FORMAT: return "JOC_ERR_INPUT_FORMAT";
|
||||
case JOC_ERR_EAC3_SYNCFRAME: return "JOC_ERR_EAC3_SYNCFRAME";
|
||||
case JOC_ERR_EMDF_TRANSPORT: return "JOC_ERR_EMDF_TRANSPORT";
|
||||
case JOC_ERR_EMDF_SYNTAX: return "JOC_ERR_EMDF_SYNTAX";
|
||||
case JOC_ERR_JOC_SYNTAX: return "JOC_ERR_JOC_SYNTAX";
|
||||
case JOC_ERR_JOC_UNSUPPORTED_VARIANT: return "JOC_ERR_JOC_UNSUPPORTED_VARIANT";
|
||||
case JOC_ERR_OAMD_SYNTAX: return "JOC_ERR_OAMD_SYNTAX";
|
||||
case JOC_ERR_OAMD_UNSUPPORTED_VARIANT: return "JOC_ERR_OAMD_UNSUPPORTED_VARIANT";
|
||||
case JOC_ERR_BITSTREAM_TRUNCATED: return "JOC_ERR_BITSTREAM_TRUNCATED";
|
||||
case JOC_ERR_BITSTREAM_PADDING: return "JOC_ERR_BITSTREAM_PADDING";
|
||||
case JOC_ERR_HRTF_NOT_FOUND: return "JOC_ERR_HRTF_NOT_FOUND";
|
||||
case JOC_ERR_HRTF_FORMAT: return "JOC_ERR_HRTF_FORMAT";
|
||||
case JOC_ERR_HRTF_VERSION: return "JOC_ERR_HRTF_VERSION";
|
||||
case JOC_ERR_HRTF_HASH: return "JOC_ERR_HRTF_HASH";
|
||||
case JOC_ERR_HRTF_UNSUPPORTED_CONVENTION: return "JOC_ERR_HRTF_UNSUPPORTED_CONVENTION";
|
||||
case JOC_ERR_LAYOUT_UNSUPPORTED: return "JOC_ERR_LAYOUT_UNSUPPORTED";
|
||||
case JOC_ERR_RENDER_FAILED: return "JOC_ERR_RENDER_FAILED";
|
||||
case JOC_ERR_OUTPUT_OPEN: return "JOC_ERR_OUTPUT_OPEN";
|
||||
case JOC_ERR_OUTPUT_WRITE: return "JOC_ERR_OUTPUT_WRITE";
|
||||
case JOC_ERR_OUTPUT_CLIP_ABORT: return "JOC_ERR_OUTPUT_CLIP_ABORT";
|
||||
case JOC_ERR_ADM_VALIDATION: return "JOC_ERR_ADM_VALIDATION";
|
||||
case JOC_ERR_CANCELLED: return "JOC_ERR_CANCELLED";
|
||||
case JOC_ERR_STATE: return "JOC_ERR_STATE";
|
||||
case JOC_ERR_NOT_SUPPORTED: return "JOC_ERR_NOT_SUPPORTED";
|
||||
case JOC_ERR_INTERNAL: return "JOC_ERR_INTERNAL";
|
||||
default: return "JOC_ERR_UNKNOWN";
|
||||
}
|
||||
}
|
||||
|
||||
const char* JOC_CALL joc_error_stage(joc_error code) {
|
||||
switch (code) {
|
||||
case JOC_ERR_EAC3_SYNCFRAME:
|
||||
case JOC_ERR_INPUT_NOT_FOUND:
|
||||
case JOC_ERR_INPUT_FORMAT:
|
||||
return "eac3_transport";
|
||||
case JOC_ERR_EMDF_TRANSPORT:
|
||||
case JOC_ERR_EMDF_SYNTAX:
|
||||
return "emdf";
|
||||
case JOC_ERR_JOC_SYNTAX:
|
||||
case JOC_ERR_JOC_UNSUPPORTED_VARIANT:
|
||||
return "joc";
|
||||
case JOC_ERR_OAMD_SYNTAX:
|
||||
case JOC_ERR_OAMD_UNSUPPORTED_VARIANT:
|
||||
return "oamd";
|
||||
case JOC_ERR_BITSTREAM_TRUNCATED:
|
||||
case JOC_ERR_BITSTREAM_PADDING:
|
||||
return "bitstream";
|
||||
case JOC_ERR_HRTF_NOT_FOUND:
|
||||
case JOC_ERR_HRTF_FORMAT:
|
||||
case JOC_ERR_HRTF_VERSION:
|
||||
case JOC_ERR_HRTF_HASH:
|
||||
case JOC_ERR_HRTF_UNSUPPORTED_CONVENTION:
|
||||
return "hrtf";
|
||||
case JOC_ERR_LAYOUT_UNSUPPORTED:
|
||||
case JOC_ERR_RENDER_FAILED:
|
||||
return "render";
|
||||
case JOC_ERR_OUTPUT_OPEN:
|
||||
case JOC_ERR_OUTPUT_WRITE:
|
||||
case JOC_ERR_OUTPUT_CLIP_ABORT:
|
||||
case JOC_ERR_ADM_VALIDATION:
|
||||
return "output";
|
||||
case JOC_ERR_CANCELLED:
|
||||
case JOC_ERR_STATE:
|
||||
case JOC_ERR_NOT_SUPPORTED:
|
||||
case JOC_ERR_INTERNAL:
|
||||
return "task";
|
||||
default:
|
||||
return "core";
|
||||
}
|
||||
}
|
||||
|
||||
const char* JOC_CALL joc_last_error_detail(void) { return g_detail.c_str(); }
|
||||
|
||||
joc_error JOC_CALL joc_parse_id14(const std::uint8_t* payload, std::size_t payload_size,
|
||||
joc_frame_params* out_params) {
|
||||
if (payload == nullptr || out_params == nullptr || payload_size == 0) {
|
||||
return arg_fail("joc_parse_id14 requires a non-empty payload and an output struct");
|
||||
}
|
||||
return finish(joc::joc::parse_id14(payload, payload_size, out_params, nullptr));
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_parse_eac3_frame(const std::uint8_t* frame, std::size_t frame_size,
|
||||
joc_frame_params* out_params, joc_emdf_info* out_emdf) {
|
||||
if (frame == nullptr || out_params == nullptr || frame_size == 0) {
|
||||
return arg_fail("joc_parse_eac3_frame requires a frame and an output struct");
|
||||
}
|
||||
joc::emdf::Container container;
|
||||
const joc::Status status =
|
||||
joc::joc::parse_eac3_frame(frame, frame_size, out_params, &container, nullptr);
|
||||
if (!status.ok()) {
|
||||
return finish(status);
|
||||
}
|
||||
if (out_emdf != nullptr) {
|
||||
fill_emdf_info(container, out_emdf);
|
||||
}
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_extract_payload(const std::uint8_t* frame, std::size_t frame_size,
|
||||
const joc_emdf_payload_info* payload, std::uint8_t* out,
|
||||
std::size_t out_capacity, std::size_t* out_size) {
|
||||
if (frame == nullptr || payload == nullptr || out_size == nullptr) {
|
||||
return arg_fail("joc_extract_payload requires frame, payload and out_size");
|
||||
}
|
||||
*out_size = payload->size;
|
||||
if (out == nullptr) {
|
||||
return JOC_OK;
|
||||
}
|
||||
if (out_capacity < payload->size) {
|
||||
return arg_fail("joc_extract_payload output buffer too small");
|
||||
}
|
||||
joc::emdf::Payload entry;
|
||||
entry.id = payload->id;
|
||||
entry.sample_offset = payload->sample_offset;
|
||||
entry.bit_offset = payload->bit_offset;
|
||||
entry.size = payload->size;
|
||||
std::vector<std::uint8_t> bytes;
|
||||
const joc::Status status = joc::emdf::extract_payload_bytes(frame, frame_size, entry, &bytes);
|
||||
if (!status.ok()) {
|
||||
return finish(status);
|
||||
}
|
||||
if (!bytes.empty()) {
|
||||
std::memcpy(out, bytes.data(), bytes.size());
|
||||
}
|
||||
*out_size = bytes.size();
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_eac3_frame_bytes(const std::uint8_t* data, std::size_t size, std::size_t offset,
|
||||
std::size_t* out_frame_bytes) {
|
||||
if (data == nullptr || out_frame_bytes == nullptr) {
|
||||
return arg_fail("joc_eac3_frame_bytes requires data and out_frame_bytes");
|
||||
}
|
||||
const joc_error code = joc::eac3::FrameReader::frame_bytes(data, size, offset, out_frame_bytes);
|
||||
if (code != JOC_OK) {
|
||||
return finish(joc::Status::fail(code, joc::stage::kEac3,
|
||||
"invalid or truncated E-AC-3 syncframe at byte " +
|
||||
std::to_string(offset)));
|
||||
}
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_check_id14_padding(const std::uint8_t* payload, std::size_t payload_size,
|
||||
std::uint32_t* out_trailing_bits) {
|
||||
if (payload == nullptr || payload_size == 0) {
|
||||
return arg_fail("joc_check_id14_padding requires a payload");
|
||||
}
|
||||
return finish(joc::joc::check_id14_padding(payload, payload_size, out_trailing_bits));
|
||||
}
|
||||
|
||||
} // extern "C"
|
||||
@@ -0,0 +1,162 @@
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "joc_core.h"
|
||||
#include "joc_stream.h"
|
||||
#include "stream/stream.h"
|
||||
|
||||
struct joc_stream {
|
||||
joc::stream::Stream instance;
|
||||
};
|
||||
|
||||
namespace {
|
||||
|
||||
joc_error finish_stream(const joc::Status& status) {
|
||||
return status.code();
|
||||
}
|
||||
|
||||
joc::stream::Config to_config(const joc_stream_config& config) {
|
||||
joc::stream::Config out;
|
||||
out.input = config.input;
|
||||
out.output = config.output;
|
||||
if (config.speaker_layout_name != nullptr) { out.layout = config.speaker_layout_name; }
|
||||
if (config.speaker_metadata_offset != 0u) {
|
||||
out.metadata_offset = config.speaker_metadata_offset;
|
||||
}
|
||||
out.binaural_mode = config.binaural_mode != 0u ? config.binaural_mode : JOC_BINAURAL_MID;
|
||||
if (config.hrtf_path != nullptr) { out.hrtf_path = config.hrtf_path; }
|
||||
if (config.kernels_path != nullptr) { out.kernels_path = config.kernels_path; }
|
||||
if (config.binaural_tail_seconds > 0.0) { out.tail_seconds = config.binaural_tail_seconds; }
|
||||
if (config.object_delay_samples != 0u) {
|
||||
out.object_delay_samples = config.object_delay_samples;
|
||||
}
|
||||
out.gain_db = config.gain_db;
|
||||
out.native_threads = config.native_threads;
|
||||
// The binaural HRTF inputs of joc_task_config, copied with the same defaults:
|
||||
// the policy is taken verbatim (0 is "none", a real choice, not "unset") and
|
||||
// the radius keeps its documented default of 1.0 when the field is not set.
|
||||
if (config.hrtf_sofa_path != nullptr) { out.hrtf_sofa_path = config.hrtf_sofa_path; }
|
||||
if (config.personalized_headphone_path != nullptr) {
|
||||
out.personalized_headphone_path = config.personalized_headphone_path;
|
||||
}
|
||||
if (config.hrtf_cache_dir != nullptr) { out.hrtf_cache_dir = config.hrtf_cache_dir; }
|
||||
out.hrtf_cache_policy = config.hrtf_cache_policy;
|
||||
if (config.hrtf_radius_m > 0.0) { out.hrtf_radius_m = config.hrtf_radius_m; }
|
||||
return out;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
extern "C" {
|
||||
|
||||
joc_error JOC_CALL joc_stream_create(const joc_stream_config* config, joc_stream** out) {
|
||||
if (config == nullptr || out == nullptr || config->struct_size != sizeof(joc_stream_config)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
auto* stream = new (std::nothrow) joc_stream();
|
||||
if (stream == nullptr) {
|
||||
return JOC_ERR_OUT_OF_MEMORY;
|
||||
}
|
||||
const joc::Status status = stream->instance.create(to_config(*config));
|
||||
if (!status.ok()) {
|
||||
delete stream;
|
||||
return finish_stream(status);
|
||||
}
|
||||
*out = stream;
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_push(joc_stream* stream, const joc_stream_buffer* input,
|
||||
std::uint32_t* consumed_samples, std::uint32_t* consumed_bytes) {
|
||||
if (stream == nullptr || input == nullptr ||
|
||||
input->struct_size != sizeof(joc_stream_buffer)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
const joc::Status status = [&] {
|
||||
if (input->kind == JOC_STREAM_IN_EAC3) {
|
||||
std::size_t consumed = 0;
|
||||
const joc::Status pushed =
|
||||
stream->instance.push_eac3(input->bytes, input->byte_count, &consumed);
|
||||
if (consumed_bytes != nullptr) {
|
||||
*consumed_bytes = static_cast<std::uint32_t>(consumed);
|
||||
}
|
||||
return pushed;
|
||||
}
|
||||
if (input->kind == JOC_STREAM_IN_PCM_OBJECTS16) {
|
||||
std::size_t consumed = 0;
|
||||
const joc::Status pushed =
|
||||
stream->instance.push_objects16(input->pcm, input->sample_count, &consumed);
|
||||
if (consumed_samples != nullptr) {
|
||||
*consumed_samples = static_cast<std::uint32_t>(consumed);
|
||||
}
|
||||
return pushed;
|
||||
}
|
||||
std::size_t consumed = 0;
|
||||
const joc::Status pushed =
|
||||
stream->instance.push_bed(input->pcm, input->sample_count, &consumed);
|
||||
if (consumed_samples != nullptr) {
|
||||
*consumed_samples = static_cast<std::uint32_t>(consumed);
|
||||
}
|
||||
return pushed; }();
|
||||
return finish_stream(status);
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_pull(joc_stream* stream, joc_stream_buffer* output,
|
||||
std::uint32_t* produced_samples) {
|
||||
if (stream == nullptr || output == nullptr || output->out_pcm == nullptr ||
|
||||
output->struct_size != sizeof(joc_stream_buffer)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
std::size_t produced = 0;
|
||||
const joc::Status status =
|
||||
stream->instance.pull(output->out_pcm, output->sample_count, &produced);
|
||||
if (!status.ok()) {
|
||||
return finish_stream(status);
|
||||
}
|
||||
if (produced_samples != nullptr) {
|
||||
*produced_samples = static_cast<std::uint32_t>(produced);
|
||||
}
|
||||
output->sample_count = static_cast<std::uint32_t>(produced);
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_flush(joc_stream* stream) {
|
||||
if (stream == nullptr) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
return finish_stream(stream->instance.flush());
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_reset(joc_stream* stream) {
|
||||
if (stream == nullptr) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
return finish_stream(stream->instance.reset());
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_status(const joc_stream* stream, joc_stream_status_info* out) {
|
||||
if (stream == nullptr || out == nullptr ||
|
||||
out->struct_size != sizeof(joc_stream_status_info)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
const joc::stream::Info& info = stream->instance.info();
|
||||
out->frames_in = info.frames_in;
|
||||
out->frames_out = info.frames_out;
|
||||
out->samples_in = info.samples_in;
|
||||
out->samples_out = info.samples_out;
|
||||
out->bytes_in = info.bytes_in;
|
||||
out->buffered_samples = stream->instance.buffered_samples();
|
||||
out->oamd_payloads = info.oamd_payloads;
|
||||
out->oamd_transitions = info.oamd_transitions;
|
||||
out->output_channels = info.output_channels;
|
||||
out->ended = info.ended;
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_stream_destroy(joc_stream* stream) {
|
||||
delete stream;
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
} // extern "C"
|
||||
@@ -0,0 +1,120 @@
|
||||
|
||||
#include <atomic>
|
||||
#include <cstring>
|
||||
#include <new>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "joc_core.h"
|
||||
#include "task/task.h"
|
||||
|
||||
// task and whichever frontend wants to stop it (plan 29.1/29.2).
|
||||
struct joc_cancel_token {
|
||||
std::atomic<std::uint32_t> requested{0u};
|
||||
};
|
||||
|
||||
namespace {
|
||||
|
||||
thread_local std::string g_task_detail;
|
||||
|
||||
joc_error finish_task(const joc::Status& status) {
|
||||
if (status.ok()) {
|
||||
g_task_detail.clear();
|
||||
return JOC_OK;
|
||||
}
|
||||
g_task_detail.assign(status.stage());
|
||||
g_task_detail.append(": ");
|
||||
g_task_detail.append(status.message());
|
||||
return status.code();
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
extern "C" {
|
||||
|
||||
joc_cancel_token* JOC_CALL joc_cancel_token_create(void) {
|
||||
return new (std::nothrow) joc_cancel_token();
|
||||
}
|
||||
|
||||
void JOC_CALL joc_cancel_token_request(joc_cancel_token* token) {
|
||||
if (token != nullptr) {
|
||||
token->requested.store(1u, std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
std::int32_t JOC_CALL joc_cancel_token_is_requested(const joc_cancel_token* token) {
|
||||
return (token != nullptr && token->requested.load(std::memory_order_relaxed) != 0u) ? 1 : 0;
|
||||
}
|
||||
|
||||
void JOC_CALL joc_cancel_token_destroy(joc_cancel_token* token) { delete token; }
|
||||
|
||||
joc_error JOC_CALL joc_task_validate(const joc_task_config* config,
|
||||
joc_validation_issue* issues, std::uint32_t capacity,
|
||||
std::uint32_t* count) {
|
||||
if (config == nullptr) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
if (config->struct_size != sizeof(joc_task_config)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
std::vector<joc_validation_issue> found;
|
||||
std::uint32_t errors = 0;
|
||||
const joc::Status status = joc::task::validate(*config, &found, &errors);
|
||||
if (count != nullptr) {
|
||||
*count = static_cast<std::uint32_t>(found.size());
|
||||
}
|
||||
if (issues != nullptr) {
|
||||
for (std::uint32_t i = 0; i < capacity && i < found.size(); ++i) {
|
||||
issues[i] = found[i];
|
||||
}
|
||||
}
|
||||
return status.ok() ? JOC_OK : JOC_ERR_INVALID_CONFIG;
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_task_execute(const joc_task_config* config, const joc_event_sink* sink,
|
||||
joc_task_result* out) {
|
||||
if (config == nullptr || config->struct_size != sizeof(joc_task_config)) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
if (out != nullptr) {
|
||||
std::memset(out, 0, sizeof(*out));
|
||||
out->struct_size = sizeof(joc_task_result);
|
||||
out->struct_version = JOC_TASK_RESULT_VERSION;
|
||||
}
|
||||
const joc::Status status = joc::task::run(*config, sink, out);
|
||||
if (!status.ok() && out != nullptr && out->status == 0u) {
|
||||
out->status = JOC_TASK_FAILED;
|
||||
out->error_code = static_cast<std::uint32_t>(status.code());
|
||||
std::snprintf(out->error_stage, sizeof(out->error_stage), "%s", status.stage().c_str());
|
||||
std::snprintf(out->error_message, sizeof(out->error_message), "%s",
|
||||
status.message().c_str());
|
||||
}
|
||||
g_task_detail = status.ok() ? std::string() : status.message();
|
||||
return status.code();
|
||||
}
|
||||
|
||||
joc_error JOC_CALL joc_task_result_to_json(const joc_task_result* result, char* buffer,
|
||||
std::size_t capacity, std::size_t* needed) {
|
||||
if (result == nullptr) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
std::string json;
|
||||
const joc::Status status = joc::task::result_to_json(*result, &json);
|
||||
if (!status.ok()) {
|
||||
return status.code();
|
||||
}
|
||||
if (needed != nullptr) {
|
||||
*needed = json.size() + 1u;
|
||||
}
|
||||
if (buffer == nullptr) {
|
||||
return JOC_OK;
|
||||
}
|
||||
if (capacity < json.size() + 1u) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
std::memcpy(buffer, json.c_str(), json.size() + 1u);
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
} // extern "C"
|
||||
@@ -0,0 +1,251 @@
|
||||
#include "binaural/binaural_runtime.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
namespace joc::binaural {
|
||||
|
||||
namespace {
|
||||
|
||||
double max_delay_bound(const hrtf::Field& field) {
|
||||
double maximum = 0.0;
|
||||
for (const double value : field.delay_bounds) {
|
||||
maximum = std::max(maximum, value);
|
||||
}
|
||||
return maximum;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool profile_from_name(const char* name, Profile* out) {
|
||||
if (name == nullptr || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (std::strcmp(name, "near") == 0) { *out = Profile::Near; return true; }
|
||||
if (std::strcmp(name, "mid") == 0) { *out = Profile::Mid; return true; }
|
||||
if (std::strcmp(name, "far") == 0) { *out = Profile::Far; return true; }
|
||||
return false;
|
||||
}
|
||||
|
||||
SofaBinauralRuntime::~SofaBinauralRuntime() {
|
||||
if (handle_ != nullptr) {
|
||||
ejoc_sofa_binaural_destroy(handle_);
|
||||
handle_ = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
Status SofaBinauralRuntime::open(const hrtf::Field& field, const hrtf::Kernels& kernels,
|
||||
Profile profile, const RoomConstants& room) {
|
||||
if (handle_ != nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime already open");
|
||||
}
|
||||
if (field.coefficients.size() != static_cast<std::size_t>(hrtf::kShTerms * hrtf::kEars *
|
||||
hrtf::kHybridBands * 2) ||
|
||||
field.band_centers_hz.size() != static_cast<std::size_t>(hrtf::kHybridBands)) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"compiled HRTF field has unexpected array sizes");
|
||||
}
|
||||
handle_ = ejoc_sofa_binaural_create();
|
||||
if (handle_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_OUT_OF_MEMORY, stage::kRender,
|
||||
"ejoc_sofa_binaural_create failed");
|
||||
}
|
||||
profile_ = profile;
|
||||
|
||||
if (ejoc_sofa_binaural_configure_kernels(
|
||||
handle_, kernels.qmf_analysis.data(), kernels.hybrid_low.data(),
|
||||
kernels.hybrid_indices.data(), kernels.hybrid_values.data(), kernels.hybrid_count,
|
||||
kernels.qmf_basis.data(), kernels.qmf_taps.data()) != 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
std::string("configure_kernels failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
if (ejoc_sofa_binaural_configure_field(handle_, field.coefficients.data(),
|
||||
field.delay_coefficients.data(),
|
||||
field.delay_bounds.data(), field.band_centers_hz.data(),
|
||||
field.measurement_radius_m) != 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
std::string("configure_field failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
if (ejoc_sofa_binaural_configure_room(
|
||||
handle_, room.dims, room.listener, room.walls, room.speed_of_sound, room.fdn_delays,
|
||||
room.fdn_feedback, room.damping, room.fdn_output_gain, room.allpass_delays,
|
||||
room.allpass_gains, room.enable_early_reflections, room.enable_late_room) != 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
|
||||
std::string("configure_room failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
|
||||
maximum_hrtf_delay_ =
|
||||
static_cast<std::int64_t>(std::ceil(max_delay_bound(field) - 1e-9));
|
||||
hrtf_history_slots_ = static_cast<std::uint32_t>(std::max<std::int64_t>(1, (maximum_hrtf_delay_ + 63) / 64));
|
||||
staging_.assign(kBlockSamples * kSourceCount, 0.0);
|
||||
block_output_.assign(kBlockSamples * 2u, 0.0);
|
||||
output_.clear();
|
||||
staged_ = 0;
|
||||
input_samples_ = 0;
|
||||
processed_samples_ = 0;
|
||||
blocks_processed_ = 0;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status SofaBinauralRuntime::process_block() {
|
||||
double positions[timeline::kTimelineObjects][3] = {};
|
||||
const Status queried = timeline_.positions_at(static_cast<std::int64_t>(processed_samples_),
|
||||
positions);
|
||||
if (!queried.ok()) {
|
||||
return queried;
|
||||
}
|
||||
// The reference adapter calls set_source without a `fade` argument, so the
|
||||
// backend default (fade enabled) applies - the per-object path crossfade is
|
||||
// part of the reference behaviour, not an optional extra.
|
||||
constexpr std::uint32_t kFade = 1u;
|
||||
if (ejoc_sofa_binaural_set_source(handle_, 0u, kLfePosition,
|
||||
static_cast<std::uint32_t>(profile_), 1.0, 1u, 1u,
|
||||
kFade) != 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
|
||||
std::string("set_source(LFE) failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
for (std::uint32_t source = 0; source < timeline::kTimelineObjects; ++source) {
|
||||
if (ejoc_sofa_binaural_set_source(handle_, source + 1u, positions[source],
|
||||
static_cast<std::uint32_t>(profile_), 1.0, 1u, 0u,
|
||||
kFade) != 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
|
||||
std::string("set_source(object ") + std::to_string(source + 1u) +
|
||||
") failed: " + (message != nullptr ? message : "unknown"));
|
||||
}
|
||||
}
|
||||
const int trimmed = ejoc_sofa_binaural_process(handle_, staging_.data(), kBlockSamples, 1.0,
|
||||
block_output_.data());
|
||||
if (trimmed < 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
|
||||
std::string("sofa process failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
if (trimmed > 0) {
|
||||
output_.insert(output_.end(), block_output_.begin(),
|
||||
block_output_.begin() + static_cast<std::ptrdiff_t>(trimmed) * 2);
|
||||
}
|
||||
staged_ = 0;
|
||||
processed_samples_ += kBlockSamples;
|
||||
++blocks_processed_;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status SofaBinauralRuntime::submit_frame(const float* objects16_planar,
|
||||
const oamd::OamdUpdate* update, std::int64_t frame_index,
|
||||
std::int64_t outer_sample_offset,
|
||||
std::int64_t object_delay_samples) {
|
||||
if (handle_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime is not open");
|
||||
}
|
||||
if (objects16_planar == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null frame");
|
||||
}
|
||||
// A frame without an ID11 payload submits nothing at all (the reference only
|
||||
if (update != nullptr) {
|
||||
const Status submitted = timeline_.submit_update(
|
||||
*update, frame_index * JOC_FRAME_SAMPLES, outer_sample_offset, object_delay_samples,
|
||||
static_cast<std::int64_t>(input_samples_));
|
||||
if (!submitted.ok()) {
|
||||
return submitted;
|
||||
}
|
||||
}
|
||||
|
||||
std::size_t offset = 0;
|
||||
while (offset < JOC_FRAME_SAMPLES) {
|
||||
const std::size_t room = kBlockSamples - staged_;
|
||||
const std::size_t count = std::min<std::size_t>(room, JOC_FRAME_SAMPLES - offset);
|
||||
for (std::size_t sample = 0; sample < count; ++sample) {
|
||||
double* row = staging_.data() + (staged_ + sample) * kSourceCount;
|
||||
for (std::size_t channel = 0; channel < kSourceCount; ++channel) {
|
||||
row[channel] = static_cast<double>(
|
||||
objects16_planar[channel * JOC_FRAME_SAMPLES + offset + sample]);
|
||||
}
|
||||
}
|
||||
staged_ += count;
|
||||
offset += count;
|
||||
if (staged_ == kBlockSamples) {
|
||||
const Status status = process_block();
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
}
|
||||
}
|
||||
input_samples_ += JOC_FRAME_SAMPLES;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
std::uint32_t SofaBinauralRuntime::finish_capacity(double tail_seconds) const {
|
||||
std::int64_t requested = tail_samples_;
|
||||
if (tail_seconds >= 0.0) {
|
||||
requested = static_cast<std::int64_t>(std::ceil(tail_seconds * 48000.0 - 1e-9));
|
||||
}
|
||||
const std::int64_t hrtf_bound = static_cast<std::int64_t>(hrtf_history_slots_) * 64;
|
||||
const std::int64_t early_bound = hrtf_bound + 2048 + 256 * 64;
|
||||
std::int64_t drain = std::max(requested, early_bound) + 961;
|
||||
drain = ((drain + 63) / 64) * 64;
|
||||
return static_cast<std::uint32_t>(drain);
|
||||
}
|
||||
|
||||
Status SofaBinauralRuntime::reset() {
|
||||
if (handle_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime is not open");
|
||||
}
|
||||
if (ejoc_sofa_binaural_reset(handle_) != 0) {
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender, "sofa reset failed");
|
||||
}
|
||||
timeline_ = timeline::OamdPositionTimeline();
|
||||
output_.clear();
|
||||
staged_ = 0;
|
||||
input_samples_ = 0;
|
||||
processed_samples_ = 0;
|
||||
blocks_processed_ = 0;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status SofaBinauralRuntime::finish(std::uint32_t flush_samples, std::vector<double>* out) {
|
||||
if (handle_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kRender, "binaural runtime is not open");
|
||||
}
|
||||
if (out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null output");
|
||||
}
|
||||
out->clear();
|
||||
if (flush_samples == 0u) {
|
||||
return Status::success();
|
||||
}
|
||||
std::vector<double> chunk(static_cast<std::size_t>(kBlockSamples) * 2u, 0.0);
|
||||
std::uint32_t produced_total = 0;
|
||||
std::uint32_t remaining = flush_samples;
|
||||
while (remaining > 0) {
|
||||
const std::uint32_t request = std::min<std::uint32_t>(remaining, kBlockSamples);
|
||||
const int produced = ejoc_sofa_binaural_finish(handle_, request, chunk.data(), request);
|
||||
if (produced < 0) {
|
||||
const char* message = ejoc_sofa_binaural_last_error(handle_);
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kRender,
|
||||
std::string("sofa finish failed: ") +
|
||||
(message != nullptr ? message : "unknown"));
|
||||
}
|
||||
if (produced == 0) {
|
||||
break;
|
||||
}
|
||||
out->insert(out->end(), chunk.begin(),
|
||||
chunk.begin() + static_cast<std::ptrdiff_t>(produced) * 2);
|
||||
produced_total += static_cast<std::uint32_t>(produced);
|
||||
remaining -= std::min<std::uint32_t>(remaining, static_cast<std::uint32_t>(produced));
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::binaural
|
||||
@@ -0,0 +1,96 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
#include "eac3joc_core.h"
|
||||
#include "foundation/status.h"
|
||||
#include "hrtf/jochrtf.h"
|
||||
#include "oamd/oamd_parser.h"
|
||||
#include "timeline/position_timeline.h"
|
||||
|
||||
namespace joc::binaural {
|
||||
|
||||
// ADM direction of the LFE source, as the reference passes it.
|
||||
inline constexpr double kLfePosition[3] = {0.0, 1.0, 0.0};
|
||||
inline constexpr int kSourceCount = 16;
|
||||
inline constexpr std::uint32_t kBlockSamples = 512;
|
||||
|
||||
// Room constants the reference's native bridge passes for the default shoebox.
|
||||
struct RoomConstants {
|
||||
double dims[3] = {18.0, 18.0, 14.0};
|
||||
double listener[3] = {9.0, 9.0, 7.0};
|
||||
double walls[6] = {0.62, 0.60, 0.58, 0.61, 0.52, 0.56};
|
||||
double speed_of_sound = 343.3;
|
||||
std::uint32_t fdn_delays[4] = {1427u, 1783u, 1973u, 2099u};
|
||||
double fdn_feedback[4] = {0.7853685923259284, 0.7394299865898056, 0.7160221718631921,
|
||||
0.7009092068085467};
|
||||
double damping = 0.32;
|
||||
double fdn_output_gain = 0.22;
|
||||
std::uint32_t allpass_delays[2] = {113u, 331u};
|
||||
double allpass_gains[2] = {0.63, 0.51};
|
||||
std::uint32_t enable_early_reflections = 1;
|
||||
std::uint32_t enable_late_room = 1;
|
||||
};
|
||||
|
||||
enum class Profile : std::uint32_t { Near = 0, Mid = 1, Far = 2 };
|
||||
|
||||
bool profile_from_name(const char* name, Profile* out);
|
||||
|
||||
class SofaBinauralRuntime {
|
||||
public:
|
||||
SofaBinauralRuntime() = default;
|
||||
~SofaBinauralRuntime();
|
||||
|
||||
SofaBinauralRuntime(const SofaBinauralRuntime&) = delete;
|
||||
SofaBinauralRuntime& operator=(const SofaBinauralRuntime&) = delete;
|
||||
|
||||
Status open(const hrtf::Field& field, const hrtf::Kernels& kernels, Profile profile,
|
||||
const RoomConstants& room = RoomConstants{});
|
||||
|
||||
Status submit_frame(const float* objects16_planar, const oamd::OamdUpdate* update,
|
||||
std::int64_t frame_index, std::int64_t outer_sample_offset,
|
||||
std::int64_t object_delay_samples);
|
||||
|
||||
// Resets the kernel, the timeline and the counters (plan 31.2).
|
||||
Status reset();
|
||||
|
||||
// Drains the room tail. `flush_samples` is the drain length; the reference
|
||||
Status finish(std::uint32_t flush_samples, std::vector<double>* out);
|
||||
|
||||
// Program output (input minus the 961-sample kernel latency), interleaved.
|
||||
const std::vector<double>& output() const { return output_; }
|
||||
|
||||
void take_output(std::vector<double>* out) {
|
||||
out->swap(output_);
|
||||
output_.clear();
|
||||
}
|
||||
|
||||
std::uint64_t input_samples() const { return input_samples_; }
|
||||
std::uint64_t blocks_processed() const { return blocks_processed_; }
|
||||
std::size_t staged_samples() const { return staged_; }
|
||||
const timeline::OamdPositionTimeline& timeline() const { return timeline_; }
|
||||
|
||||
// finish_output_capacity as the reference computes it (plan 21.4).
|
||||
std::uint32_t finish_capacity(double tail_seconds) const;
|
||||
|
||||
private:
|
||||
Status process_block();
|
||||
|
||||
ejoc_sofa_binaural_handle handle_ = nullptr;
|
||||
Profile profile_ = Profile::Mid;
|
||||
timeline::OamdPositionTimeline timeline_;
|
||||
std::vector<double> staging_;
|
||||
std::size_t staged_ = 0;
|
||||
std::vector<double> block_output_;
|
||||
std::vector<double> output_;
|
||||
std::uint64_t input_samples_ = 0;
|
||||
std::uint64_t processed_samples_ = 0;
|
||||
std::uint64_t blocks_processed_ = 0;
|
||||
std::int64_t maximum_hrtf_delay_ = 0;
|
||||
std::uint32_t hrtf_history_slots_ = 1;
|
||||
std::uint32_t tail_samples_ = 61200;
|
||||
};
|
||||
|
||||
} // namespace joc::binaural
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,188 +0,0 @@
|
||||
"""Direct ID11/OAMD position scheduling for the Rosella binaural path."""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
import numpy as np
|
||||
|
||||
from adm_atmos import q_to_adm_xyz
|
||||
from oamd_bits import JocFieldState, frame_update
|
||||
from variant_error import UnsupportedVariantError
|
||||
|
||||
OAMD_UPDATE_QUANTUM_SAMPLES = 64
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PositionTransition:
|
||||
start_sample: int
|
||||
duration_samples: int
|
||||
origin: np.ndarray
|
||||
target: np.ndarray
|
||||
|
||||
@property
|
||||
def end_sample(self) -> int:
|
||||
return self.start_sample + self.duration_samples
|
||||
|
||||
|
||||
class _ObjectPositionTrack:
|
||||
def __init__(self):
|
||||
self.initial = np.zeros(3, dtype=np.float64)
|
||||
self.last_target = self.initial.copy()
|
||||
self.transitions: list[PositionTransition] = []
|
||||
self.cursor = 0
|
||||
self.last_query_sample = -1
|
||||
|
||||
def set_initial(self, position):
|
||||
target = np.asarray(position, dtype=np.float64)
|
||||
self.initial = target.copy()
|
||||
self.last_target = target.copy()
|
||||
|
||||
def append(self, start_sample: int, duration_samples: int, target,
|
||||
object_index: int):
|
||||
start = int(start_sample)
|
||||
duration = int(duration_samples)
|
||||
if start < 0 or duration < 0:
|
||||
raise ValueError("position transition timing must be non-negative")
|
||||
target = np.asarray(target, dtype=np.float64)
|
||||
if self.transitions:
|
||||
previous = self.transitions[-1]
|
||||
if start < previous.end_sample:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "overlapping_binaural_position_ramps",
|
||||
"同一对象的新位置更新在上一双耳 ramp 完成前到达",
|
||||
details={
|
||||
"object": object_index,
|
||||
"ramp_start_sample": previous.start_sample,
|
||||
"ramp_end_sample": previous.end_sample,
|
||||
"next_update_sample": start,
|
||||
})
|
||||
if start == previous.start_sample and previous.duration_samples == 0:
|
||||
self.transitions[-1] = PositionTransition(
|
||||
start, duration, previous.origin.copy(), target.copy())
|
||||
self.last_target = target.copy()
|
||||
return
|
||||
self.transitions.append(PositionTransition(
|
||||
start, duration, self.last_target.copy(), target.copy()))
|
||||
self.last_target = target.copy()
|
||||
|
||||
def position_at(self, sample: int) -> np.ndarray:
|
||||
sample = int(sample)
|
||||
if sample < self.last_query_sample:
|
||||
raise ValueError("binaural metadata positions must be queried monotonically")
|
||||
self.last_query_sample = sample
|
||||
while self.cursor < len(self.transitions):
|
||||
transition = self.transitions[self.cursor]
|
||||
if sample < transition.end_sample:
|
||||
break
|
||||
self.initial = transition.target.copy()
|
||||
self.cursor += 1
|
||||
if self.cursor >= len(self.transitions):
|
||||
return self.initial
|
||||
transition = self.transitions[self.cursor]
|
||||
if sample < transition.start_sample:
|
||||
return self.initial
|
||||
if transition.duration_samples == 0:
|
||||
return transition.target
|
||||
amount = (sample - transition.start_sample) / float(transition.duration_samples)
|
||||
return transition.origin + (transition.target - transition.origin) * amount
|
||||
|
||||
|
||||
class OamdPositionTimeline:
|
||||
"""Convert OAMD state updates into a sample-timed Cartesian trajectory."""
|
||||
|
||||
def __init__(self, object_count: int = 15):
|
||||
if object_count != 15:
|
||||
raise ValueError("JOC OAMD currently requires 15 object slots")
|
||||
self.object_count = int(object_count)
|
||||
self.state = JocFieldState()
|
||||
self.tracks = [_ObjectPositionTrack() for _ in range(self.object_count)]
|
||||
self.initialized = False
|
||||
self.previous_targets: list[tuple[float, float, float] | None] = [
|
||||
None] * self.object_count
|
||||
self.payload_count = 0
|
||||
self.transition_count = 0
|
||||
self.last_coded_event_sample = -1
|
||||
|
||||
def _targets(self) -> list[tuple[float, float, float]]:
|
||||
q = self.state.q
|
||||
return [
|
||||
q_to_adm_xyz(
|
||||
q[(object_index, "q1")],
|
||||
q[(object_index, "q2")],
|
||||
q[(object_index, "q3")],
|
||||
)
|
||||
for object_index in range(1, self.object_count + 1)
|
||||
]
|
||||
|
||||
def submit_update(self, update: dict, *, frame_start_sample: int,
|
||||
outer_sample_offset: int = 0,
|
||||
object_delay_samples: int = 1473,
|
||||
processed_sample: int = 0):
|
||||
"""Schedule one already-parsed :func:`oamd_bits.frame_update` result."""
|
||||
frame_start = int(frame_start_sample)
|
||||
outer_offset = int(outer_sample_offset)
|
||||
object_delay = int(object_delay_samples)
|
||||
if min(frame_start, outer_offset, object_delay) < 0:
|
||||
raise ValueError("OAMD frame, outer offset, and object delay must be non-negative")
|
||||
self.state.apply(update["values"])
|
||||
targets = self._targets()
|
||||
coded_event = (
|
||||
frame_start + outer_offset + int(update["block_offset_samples"]))
|
||||
if coded_event < self.last_coded_event_sample:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "non_monotonic_binaural_updates",
|
||||
"双耳 OAMD 更新时间倒退",
|
||||
details={
|
||||
"event_sample": coded_event,
|
||||
"previous_event_sample": self.last_coded_event_sample,
|
||||
})
|
||||
self.last_coded_event_sample = coded_event
|
||||
|
||||
if not self.initialized:
|
||||
if int(processed_sample) > 0:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "late_initial_binaural_state",
|
||||
"首个 OAMD 状态在双耳 PCM 已处理后才出现,无法回填 sample 0",
|
||||
details={
|
||||
"processed_sample": int(processed_sample),
|
||||
"first_event_sample": coded_event,
|
||||
})
|
||||
for index, target in enumerate(targets):
|
||||
self.tracks[index].set_initial(target)
|
||||
self.previous_targets[index] = target
|
||||
self.initialized = True
|
||||
self.payload_count += 1
|
||||
return
|
||||
|
||||
ramp_duration = int(update["ramp_duration_samples"])
|
||||
effective_ramp = max(0, ramp_duration - OAMD_UPDATE_QUANTUM_SAMPLES)
|
||||
transition_start = coded_event + object_delay
|
||||
if effective_ramp:
|
||||
transition_start += OAMD_UPDATE_QUANTUM_SAMPLES
|
||||
for index, target in enumerate(targets):
|
||||
if self.previous_targets[index] == target:
|
||||
continue
|
||||
self.tracks[index].append(
|
||||
transition_start, effective_ramp, target, index + 1)
|
||||
self.previous_targets[index] = target
|
||||
self.transition_count += 1
|
||||
self.payload_count += 1
|
||||
|
||||
def submit_payload(self, payload, *, frame_start_sample: int,
|
||||
outer_sample_offset: int = 0,
|
||||
object_delay_samples: int = 1473,
|
||||
processed_sample: int = 0):
|
||||
update = frame_update(payload)
|
||||
self.submit_update(
|
||||
update,
|
||||
frame_start_sample=frame_start_sample,
|
||||
outer_sample_offset=outer_sample_offset,
|
||||
object_delay_samples=object_delay_samples,
|
||||
processed_sample=processed_sample,
|
||||
)
|
||||
return update
|
||||
|
||||
def positions_at(self, sample: int) -> np.ndarray:
|
||||
return np.stack(
|
||||
[track.position_at(sample) for track in self.tracks], axis=0
|
||||
).astype(np.float64, copy=False)
|
||||
@@ -1,214 +0,0 @@
|
||||
"""ctypes bridge for the native float64 binaural DSP."""
|
||||
from __future__ import annotations
|
||||
|
||||
import ctypes
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from native_renderer import ABI_VERSION, find_native_library
|
||||
from rosella_filterbank import DEFAULT_KERNEL_DATA, load_kernel_tables
|
||||
from rosella_model import RosellaModel
|
||||
|
||||
BLOCK_SAMPLES = 512
|
||||
INPUT_CHANNELS = 16
|
||||
OUTPUT_CHANNELS = 2
|
||||
HYBRID_BANDS = 77
|
||||
|
||||
|
||||
class NativeBinauralDsp:
|
||||
def __init__(self, model: RosellaModel, *, library_path=None,
|
||||
kernel_data: str | Path = DEFAULT_KERNEL_DATA):
|
||||
self.library_path = find_native_library(library_path)
|
||||
self._lib = ctypes.CDLL(str(self.library_path))
|
||||
self._bind()
|
||||
version = int(self._lib.ejoc_abi_version())
|
||||
if version != ABI_VERSION:
|
||||
raise RuntimeError(
|
||||
f"native ABI mismatch: expected {ABI_VERSION}, got {version}")
|
||||
self._handle = self._lib.ejoc_binaural_renderer_create()
|
||||
if not self._handle:
|
||||
raise RuntimeError("native binaural renderer creation failed")
|
||||
try:
|
||||
self._configure_kernels(kernel_data)
|
||||
self._configure_room(model)
|
||||
except Exception:
|
||||
self.close()
|
||||
raise
|
||||
|
||||
def _bind(self):
|
||||
void_p = ctypes.c_void_p
|
||||
f64_p = ctypes.POINTER(ctypes.c_double)
|
||||
i16_p = ctypes.POINTER(ctypes.c_int16)
|
||||
u32_p = ctypes.POINTER(ctypes.c_uint32)
|
||||
self._lib.ejoc_abi_version.argtypes = []
|
||||
self._lib.ejoc_abi_version.restype = ctypes.c_uint32
|
||||
self._lib.ejoc_binaural_renderer_create.argtypes = []
|
||||
self._lib.ejoc_binaural_renderer_create.restype = void_p
|
||||
self._lib.ejoc_binaural_renderer_destroy.argtypes = [void_p]
|
||||
self._lib.ejoc_binaural_renderer_destroy.restype = None
|
||||
self._lib.ejoc_binaural_renderer_reset.argtypes = [void_p]
|
||||
self._lib.ejoc_binaural_renderer_reset.restype = ctypes.c_int
|
||||
self._lib.ejoc_binaural_renderer_last_error.argtypes = [void_p]
|
||||
self._lib.ejoc_binaural_renderer_last_error.restype = ctypes.c_char_p
|
||||
self._lib.ejoc_binaural_renderer_configure_kernels.argtypes = [
|
||||
void_p, f64_p, f64_p, i16_p, f64_p, ctypes.c_uint32, f64_p, f64_p]
|
||||
self._lib.ejoc_binaural_renderer_configure_kernels.restype = ctypes.c_int
|
||||
self._lib.ejoc_binaural_renderer_configure_room.argtypes = [
|
||||
void_p, ctypes.c_uint32, ctypes.c_uint32, u32_p, f64_p,
|
||||
u32_p, f64_p, ctypes.c_uint32, f64_p, f64_p, f64_p,
|
||||
ctypes.c_uint32, u32_p, f64_p, f64_p]
|
||||
self._lib.ejoc_binaural_renderer_configure_room.restype = ctypes.c_int
|
||||
self._lib.ejoc_binaural_renderer_process.argtypes = [
|
||||
void_p, f64_p, f64_p, f64_p, ctypes.c_double, f64_p]
|
||||
self._lib.ejoc_binaural_renderer_process.restype = ctypes.c_int
|
||||
|
||||
def _raise(self, operation, status):
|
||||
message = self._lib.ejoc_binaural_renderer_last_error(self._handle)
|
||||
detail = (message or b"").decode("utf-8", "replace")
|
||||
raise RuntimeError(
|
||||
f"native binaural renderer {operation} failed ({status}): {detail}")
|
||||
|
||||
@staticmethod
|
||||
def _f64_pointer(values):
|
||||
return values.ctypes.data_as(ctypes.POINTER(ctypes.c_double))
|
||||
|
||||
def _configure_kernels(self, kernel_data):
|
||||
tables = load_kernel_tables(kernel_data)
|
||||
qmf_analysis = np.ascontiguousarray(
|
||||
tables["qmf_analysis_coefficients"], dtype=np.float64)
|
||||
hybrid_low = np.ascontiguousarray(
|
||||
tables["hybrid_analysis_low_kernel"], dtype=np.float64)
|
||||
hybrid_indices = np.ascontiguousarray(
|
||||
tables["hybrid_synthesis_indices"], dtype=np.int16)
|
||||
hybrid_values = np.ascontiguousarray(
|
||||
tables["hybrid_synthesis_values"], dtype=np.float64)
|
||||
qmf_basis = np.ascontiguousarray(
|
||||
tables["qmf_synthesis_basis"], dtype=np.float64)
|
||||
qmf_taps = np.ascontiguousarray(
|
||||
tables["qmf_synthesis_taps"], dtype=np.float64)
|
||||
status = self._lib.ejoc_binaural_renderer_configure_kernels(
|
||||
self._handle,
|
||||
self._f64_pointer(qmf_analysis),
|
||||
self._f64_pointer(hybrid_low),
|
||||
hybrid_indices.ctypes.data_as(ctypes.POINTER(ctypes.c_int16)),
|
||||
self._f64_pointer(hybrid_values),
|
||||
len(hybrid_values),
|
||||
self._f64_pointer(qmf_basis),
|
||||
self._f64_pointer(qmf_taps),
|
||||
)
|
||||
if status:
|
||||
self._raise("configure_kernels", status)
|
||||
|
||||
def _configure_room(self, model: RosellaModel):
|
||||
if float(model.table_a_scalar) >= 0.5:
|
||||
raise NotImplementedError("alternate table-A room mode")
|
||||
bands = min(64, model.table_a_dimension)
|
||||
allpass_delays = np.ascontiguousarray(
|
||||
model.table_a_option_ids, dtype=np.uint32)
|
||||
allpass_gains = np.ascontiguousarray(
|
||||
model.table_a_option_values, dtype=np.float64)
|
||||
fdn_delays = np.ascontiguousarray(
|
||||
model.table_a_four_integers, dtype=np.uint32)
|
||||
fdn_matrix = np.ascontiguousarray(
|
||||
np.asarray(model.table_a_vector16, dtype=np.float64).reshape(
|
||||
4, 4, order="F"))
|
||||
|
||||
filter8 = np.asarray(
|
||||
model.table_a_filter_8x64_padded, dtype=np.float64).reshape(20, 4, 2, 4)
|
||||
filter4 = np.asarray(
|
||||
model.table_a_filter_4x64_padded, dtype=np.float64).reshape(20, 4, 4)
|
||||
filter16 = np.asarray(
|
||||
model.table_a_filter_16x64_padded, dtype=np.float64).reshape(20, 4, 4, 4)
|
||||
feedback = np.empty((64, 4, 2), dtype=np.float64)
|
||||
output_taps = np.empty((64, 4), dtype=np.float64)
|
||||
output_matrix = np.empty((2, 64, 4, 2), dtype=np.float64)
|
||||
for band in range(64):
|
||||
group, lane = divmod(band, 4)
|
||||
feedback[band, :, 0] = filter8[group, :, 0, lane]
|
||||
feedback[band, :, 1] = filter8[group, :, 1, lane]
|
||||
output_taps[band] = filter4[group, :, lane]
|
||||
output_matrix[0, band, :, 0] = filter16[group, :, 0, lane]
|
||||
output_matrix[0, band, :, 1] = filter16[group, :, 1, lane]
|
||||
output_matrix[1, band, :, 0] = filter16[group, :, 2, lane]
|
||||
output_matrix[1, band, :, 1] = filter16[group, :, 3, lane]
|
||||
|
||||
extra_count = int(model.table_a_extra)
|
||||
extra_delays = np.ascontiguousarray(
|
||||
model.table_a_extra_indices, dtype=np.uint32)
|
||||
extra_fields = np.empty((extra_count, 64, 2), dtype=np.float64)
|
||||
extra_source = np.asarray(
|
||||
model.table_a_extra_fields_padded, dtype=np.float64).reshape(
|
||||
extra_count, 20, 2, 4)
|
||||
for extra in range(extra_count):
|
||||
for band in range(64):
|
||||
group, lane = divmod(band, 4)
|
||||
extra_fields[extra, band] = extra_source[extra, group, :, lane]
|
||||
extra_matrices = np.empty((extra_count, 4, 4), dtype=np.float64)
|
||||
for extra in range(extra_count):
|
||||
extra_matrices[extra] = np.asarray(
|
||||
model.table_a_extra_vectors[extra], dtype=np.float64).reshape(
|
||||
4, 4, order="F")
|
||||
|
||||
null_u32 = ctypes.POINTER(ctypes.c_uint32)()
|
||||
null_f64 = ctypes.POINTER(ctypes.c_double)()
|
||||
status = self._lib.ejoc_binaural_renderer_configure_room(
|
||||
self._handle,
|
||||
bands,
|
||||
len(allpass_delays),
|
||||
allpass_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)),
|
||||
self._f64_pointer(allpass_gains),
|
||||
fdn_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)),
|
||||
self._f64_pointer(fdn_matrix),
|
||||
int(model.table_a_integer),
|
||||
self._f64_pointer(feedback),
|
||||
self._f64_pointer(output_taps),
|
||||
self._f64_pointer(output_matrix),
|
||||
extra_count,
|
||||
(extra_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32))
|
||||
if extra_count else null_u32),
|
||||
self._f64_pointer(extra_fields) if extra_count else null_f64,
|
||||
self._f64_pointer(extra_matrices) if extra_count else null_f64,
|
||||
)
|
||||
if status:
|
||||
self._raise("configure_room", status)
|
||||
|
||||
def reset(self):
|
||||
if not self._handle:
|
||||
raise RuntimeError("native binaural renderer is closed")
|
||||
status = self._lib.ejoc_binaural_renderer_reset(self._handle)
|
||||
if status:
|
||||
self._raise("reset", status)
|
||||
|
||||
def process_block(self, pcm16, gains, room_sends, output_gain=1.0):
|
||||
if not self._handle:
|
||||
raise RuntimeError("native binaural renderer is closed")
|
||||
source = np.ascontiguousarray(pcm16, dtype=np.float64)
|
||||
gain_values = np.asarray(gains)
|
||||
sends = np.ascontiguousarray(room_sends, dtype=np.float64)
|
||||
if source.shape != (BLOCK_SAMPLES, INPUT_CHANNELS):
|
||||
raise ValueError(f"pcm16 block must be (512,16), got {source.shape}")
|
||||
if gain_values.shape != (INPUT_CHANNELS, OUTPUT_CHANNELS, HYBRID_BANDS):
|
||||
raise ValueError(f"gains must be (16,2,77), got {gain_values.shape}")
|
||||
direct = np.ascontiguousarray(
|
||||
gain_values, dtype=np.complex128).view(np.float64)
|
||||
if sends.shape != (INPUT_CHANNELS,):
|
||||
raise ValueError(f"room_sends must be (16,), got {sends.shape}")
|
||||
output = np.empty((BLOCK_SAMPLES, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
status = self._lib.ejoc_binaural_renderer_process(
|
||||
self._handle,
|
||||
self._f64_pointer(source),
|
||||
self._f64_pointer(direct),
|
||||
self._f64_pointer(sends),
|
||||
float(output_gain),
|
||||
self._f64_pointer(output),
|
||||
)
|
||||
if status:
|
||||
self._raise("process", status)
|
||||
return output
|
||||
|
||||
def close(self):
|
||||
handle = getattr(self, "_handle", None)
|
||||
if handle:
|
||||
self._lib.ejoc_binaural_renderer_destroy(handle)
|
||||
self._handle = None
|
||||
@@ -1,301 +0,0 @@
|
||||
"""Stateful binaural renderer for reconstructed JOC objects."""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import math
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from binaural_metadata import OamdPositionTimeline
|
||||
from binaural_native_renderer import NativeBinauralDsp
|
||||
from rosella_core import RosellaRenderer
|
||||
from rosella_direct import BINAURAL_PROFILE_NAMES
|
||||
from rosella_filterbank import (
|
||||
DEFAULT_KERNEL_DATA,
|
||||
HybridAnalysis,
|
||||
HybridSynthesis,
|
||||
QmfAnalysis,
|
||||
QmfSynthesis,
|
||||
)
|
||||
from rosella_model import RosellaModel, load_personalized_headphone
|
||||
|
||||
SAMPLE_RATE = 48000
|
||||
FRAME_SAMPLES = 1536
|
||||
ROSSELLA_BLOCK_SAMPLES = 512
|
||||
QMF_HOP_SAMPLES = 64
|
||||
ROSSELLA_LATENCY_SAMPLES = 961
|
||||
SOURCE_CHANNELS = 16
|
||||
OUTPUT_CHANNELS = 2
|
||||
PROJECT_DIR = Path(__file__).resolve().parent.parent
|
||||
DEFAULT_PERSONALIZED_HEADPHONE = (
|
||||
PROJECT_DIR / "HRTF" / "binaural.personalized_headphone")
|
||||
|
||||
|
||||
def _sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as stream:
|
||||
for block in iter(lambda: stream.read(1 << 20), b""):
|
||||
digest.update(block)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def resolve_personalized_headphone(path: str | Path | None = None) -> Path:
|
||||
target = (DEFAULT_PERSONALIZED_HEADPHONE if path is None
|
||||
else Path(path).expanduser().resolve())
|
||||
if not target.is_file():
|
||||
raise FileNotFoundError(
|
||||
f"未找到双耳模型:{target}\n"
|
||||
"请将兼容模型保存为 HRTF/binaural.personalized_headphone,"
|
||||
"或使用 --personalized-headphone PATH 指定文件。"
|
||||
)
|
||||
return target
|
||||
|
||||
|
||||
class RosellaBinauralRenderer:
|
||||
"""Render interleaved LFE plus fifteen objects to stereo."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
personalized_headphone: str | Path | RosellaModel,
|
||||
*,
|
||||
mode: str = "mid",
|
||||
kernel_data: str | Path = DEFAULT_KERNEL_DATA,
|
||||
object_delay_samples: int = 1473,
|
||||
tail_seconds: float = 5.0,
|
||||
output_gain: float = 1.0,
|
||||
chunk_frames: int = 64,
|
||||
room_impulse_slots: int = 4096,
|
||||
backend: str = "python",
|
||||
native_library=None):
|
||||
if mode not in BINAURAL_PROFILE_NAMES:
|
||||
raise ValueError("binaural mode must be near, mid, or far")
|
||||
if int(object_delay_samples) < 0:
|
||||
raise ValueError("object_delay_samples must be non-negative")
|
||||
if float(tail_seconds) < 0.0:
|
||||
raise ValueError("tail_seconds must be non-negative")
|
||||
if int(chunk_frames) <= 0:
|
||||
raise ValueError("chunk_frames must be positive")
|
||||
if not math.isfinite(float(output_gain)):
|
||||
raise ValueError("output_gain must be finite")
|
||||
if backend not in ("auto", "native", "python"):
|
||||
raise ValueError("backend must be auto, native, or python")
|
||||
|
||||
if isinstance(personalized_headphone, RosellaModel):
|
||||
self.model = personalized_headphone
|
||||
self.model_path = Path(self.model.source_path)
|
||||
else:
|
||||
self.model_path = resolve_personalized_headphone(personalized_headphone)
|
||||
self.model = load_personalized_headphone(self.model_path)
|
||||
if self.model.sample_rate != SAMPLE_RATE:
|
||||
raise ValueError(
|
||||
f"Rosella model sample rate must be {SAMPLE_RATE}, got {self.model.sample_rate}")
|
||||
|
||||
self.mode = mode
|
||||
self.profile_index = BINAURAL_PROFILE_NAMES[mode]
|
||||
self.kernel_data = Path(kernel_data).expanduser().resolve()
|
||||
self.kernel_data_sha256 = _sha256_file(self.kernel_data)
|
||||
self.object_delay_samples = int(object_delay_samples)
|
||||
self.tail_seconds = float(tail_seconds)
|
||||
self.output_gain = np.float64(output_gain)
|
||||
self.chunk_frames = int(chunk_frames)
|
||||
self.chunk_samples = self.chunk_frames * FRAME_SAMPLES
|
||||
|
||||
self.native_dsp = None
|
||||
self.backend_fallback = None
|
||||
if backend in ("auto", "native"):
|
||||
try:
|
||||
self.native_dsp = NativeBinauralDsp(
|
||||
self.model, library_path=native_library,
|
||||
kernel_data=self.kernel_data)
|
||||
except (AttributeError, OSError, RuntimeError) as exc:
|
||||
if backend == "native":
|
||||
raise RuntimeError(f"native binaural backend unavailable: {exc}") from exc
|
||||
self.backend_fallback = str(exc)
|
||||
if self.native_dsp is not None:
|
||||
self.dsp_backend = "native"
|
||||
self.qmf_analysis = None
|
||||
self.hybrid_analysis = None
|
||||
self.hybrid_synthesis = None
|
||||
self.qmf_synthesis = None
|
||||
self.core = RosellaRenderer(
|
||||
self.model, SOURCE_CHANNELS, create_room=False)
|
||||
else:
|
||||
self.dsp_backend = "python"
|
||||
self.qmf_analysis = QmfAnalysis(SOURCE_CHANNELS, self.kernel_data)
|
||||
self.hybrid_analysis = HybridAnalysis(SOURCE_CHANNELS, self.kernel_data)
|
||||
self.core = RosellaRenderer(
|
||||
self.model, SOURCE_CHANNELS,
|
||||
room_impulse_slots=room_impulse_slots)
|
||||
self.hybrid_synthesis = HybridSynthesis(OUTPUT_CHANNELS, self.kernel_data)
|
||||
self.qmf_synthesis = QmfSynthesis(OUTPUT_CHANNELS, self.kernel_data)
|
||||
self.timeline = OamdPositionTimeline(15)
|
||||
|
||||
self._input_buffer = np.empty(
|
||||
(self.chunk_samples, SOURCE_CHANNELS), dtype=np.float64)
|
||||
self._buffer_used = 0
|
||||
self.input_samples = 0
|
||||
self.processed_input_samples = 0
|
||||
self.raw_output_samples = 0
|
||||
self.output_samples = 0
|
||||
self.finished = False
|
||||
self.metadata_block_updates = 0
|
||||
|
||||
def _append_input(self, samples: np.ndarray) -> list[np.ndarray]:
|
||||
outputs = []
|
||||
source = np.asarray(samples, dtype=np.float64)
|
||||
position = 0
|
||||
while position < len(source):
|
||||
count = min(self.chunk_samples - self._buffer_used,
|
||||
len(source) - position)
|
||||
self._input_buffer[self._buffer_used:self._buffer_used + count] = (
|
||||
source[position:position + count])
|
||||
self._buffer_used += count
|
||||
position += count
|
||||
if self._buffer_used == self.chunk_samples:
|
||||
outputs.append(self._process_samples(self._input_buffer))
|
||||
self._buffer_used = 0
|
||||
return outputs
|
||||
|
||||
def render_frame(self, objects16, payload=None, metadata_offset=None,
|
||||
*, outer_sample_offset=0) -> np.ndarray:
|
||||
"""Submit one 1536-sample reconstructed frame and its ID11 payload."""
|
||||
if self.finished:
|
||||
raise RuntimeError("binaural renderer is already finished")
|
||||
source = np.asarray(objects16)
|
||||
if source.shape != (FRAME_SAMPLES, SOURCE_CHANNELS):
|
||||
raise ValueError(
|
||||
f"binaural frame must have shape ({FRAME_SAMPLES},{SOURCE_CHANNELS}), "
|
||||
f"got {source.shape}")
|
||||
frame_start = self.input_samples
|
||||
metadata_delay = (self.object_delay_samples if metadata_offset is None
|
||||
else int(metadata_offset))
|
||||
if metadata_delay < 0:
|
||||
raise ValueError("metadata_offset must be non-negative")
|
||||
if payload is not None:
|
||||
self.timeline.submit_payload(
|
||||
payload,
|
||||
frame_start_sample=frame_start,
|
||||
outer_sample_offset=int(outer_sample_offset),
|
||||
object_delay_samples=metadata_delay,
|
||||
processed_sample=self.processed_input_samples,
|
||||
)
|
||||
self.metadata_block_updates += 1
|
||||
self.input_samples += FRAME_SAMPLES
|
||||
chunks = self._append_input(source)
|
||||
if not chunks:
|
||||
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
return np.concatenate(chunks, axis=0) if len(chunks) > 1 else chunks[0]
|
||||
|
||||
def _set_block_parameters(self, sample: int):
|
||||
positions = self.timeline.positions_at(sample)
|
||||
self.core.set_source(0, (0.0, 1.0, 0.0), special_lfe=True)
|
||||
for object_index in range(15):
|
||||
self.core.set_source(
|
||||
object_index + 1, positions[object_index], self.profile_index)
|
||||
|
||||
def _process_samples(self, source: np.ndarray) -> np.ndarray:
|
||||
values = np.asarray(source, dtype=np.float64)
|
||||
if values.ndim != 2 or values.shape[1] != SOURCE_CHANNELS:
|
||||
raise ValueError(f"expected [samples,{SOURCE_CHANNELS}], got {values.shape}")
|
||||
if len(values) % ROSSELLA_BLOCK_SAMPLES:
|
||||
raise ValueError("binaural input must be divisible by 512 samples")
|
||||
blocks = len(values) // ROSSELLA_BLOCK_SAMPLES
|
||||
block_base = self.processed_input_samples
|
||||
|
||||
if self.native_dsp is not None:
|
||||
stereo = np.empty((len(values), OUTPUT_CHANNELS), dtype=np.float64)
|
||||
for block in range(blocks):
|
||||
sample = block_base + block * ROSSELLA_BLOCK_SAMPLES
|
||||
self._set_block_parameters(sample)
|
||||
start = block * ROSSELLA_BLOCK_SAMPLES
|
||||
stop = start + ROSSELLA_BLOCK_SAMPLES
|
||||
stereo[start:stop] = self.native_dsp.process_block(
|
||||
values[start:stop], self.core.gains, self.core.room_sends,
|
||||
self.output_gain)
|
||||
else:
|
||||
hops = values.reshape(
|
||||
blocks, ROSSELLA_BLOCK_SAMPLES // QMF_HOP_SAMPLES,
|
||||
QMF_HOP_SAMPLES, SOURCE_CHANNELS,
|
||||
).transpose(0, 1, 3, 2).reshape(
|
||||
blocks * (ROSSELLA_BLOCK_SAMPLES // QMF_HOP_SAMPLES),
|
||||
SOURCE_CHANNELS, QMF_HOP_SAMPLES)
|
||||
hybrid = self.hybrid_analysis.process_chunk(
|
||||
self.qmf_analysis.process_chunk(hops))
|
||||
direct = np.empty((blocks * 8, OUTPUT_CHANNELS, 77), dtype=np.complex128)
|
||||
room_send = np.empty((blocks * 8, 77), dtype=np.complex128)
|
||||
for block in range(blocks):
|
||||
sample = block_base + block * ROSSELLA_BLOCK_SAMPLES
|
||||
self._set_block_parameters(sample)
|
||||
start = block * 8
|
||||
stop = start + 8
|
||||
direct[start:stop], room_send[start:stop] = (
|
||||
self.core.direct_and_send_static(hybrid[start:stop]))
|
||||
rendered = direct + self.core.room.process_chunk(room_send)
|
||||
time_bands = self.qmf_synthesis.process_chunk(
|
||||
self.hybrid_synthesis.process_chunk(rendered))
|
||||
stereo = time_bands.transpose(0, 2, 1).reshape(
|
||||
blocks * ROSSELLA_BLOCK_SAMPLES, OUTPUT_CHANNELS)
|
||||
stereo *= self.output_gain
|
||||
|
||||
skip = max(0, min(
|
||||
len(stereo), ROSSELLA_LATENCY_SAMPLES - self.raw_output_samples))
|
||||
self.raw_output_samples += len(stereo)
|
||||
self.processed_input_samples += len(values)
|
||||
output = stereo[skip:]
|
||||
self.output_samples += len(output)
|
||||
return output
|
||||
|
||||
def finish(self) -> np.ndarray:
|
||||
"""Process pending source samples and preserve the configured room tail."""
|
||||
if self.finished:
|
||||
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
outputs: list[np.ndarray] = []
|
||||
if self._buffer_used:
|
||||
outputs.append(self._process_samples(
|
||||
self._input_buffer[:self._buffer_used]))
|
||||
self._buffer_used = 0
|
||||
flush_samples = math.ceil(
|
||||
(self.tail_seconds * SAMPLE_RATE
|
||||
+ ROSSELLA_LATENCY_SAMPLES + ROSSELLA_BLOCK_SAMPLES)
|
||||
/ ROSSELLA_BLOCK_SAMPLES) * ROSSELLA_BLOCK_SAMPLES
|
||||
while flush_samples:
|
||||
count = min(flush_samples, self.chunk_samples)
|
||||
zero = np.zeros((count, SOURCE_CHANNELS), dtype=np.float64)
|
||||
outputs.append(self._process_samples(zero))
|
||||
flush_samples -= count
|
||||
self.finished = True
|
||||
nonempty = [value for value in outputs if len(value)]
|
||||
if not nonempty:
|
||||
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
return np.concatenate(nonempty, axis=0)
|
||||
|
||||
def close(self):
|
||||
if self.native_dsp is not None:
|
||||
self.native_dsp.close()
|
||||
self.finished = True
|
||||
|
||||
@property
|
||||
def backend_info(self) -> dict:
|
||||
return {
|
||||
"name": self.dsp_backend,
|
||||
"precision": "float64/complex128",
|
||||
"fallback_reason": self.backend_fallback,
|
||||
"library": (str(self.native_dsp.library_path)
|
||||
if self.native_dsp is not None else None),
|
||||
"model": str(self.model_path.resolve()),
|
||||
"model_coefficients": int(len(self.model.coefficients)),
|
||||
"model_coefficient_sha256": self.model.coefficient_sha256,
|
||||
"model_version": self.model.coefficient_version,
|
||||
"kernel_data": str(self.kernel_data),
|
||||
"kernel_data_sha256": self.kernel_data_sha256,
|
||||
"mode": self.mode,
|
||||
"latency_compensated_samples": ROSSELLA_LATENCY_SAMPLES,
|
||||
"object_delay_samples": self.object_delay_samples,
|
||||
"tail_seconds": self.tail_seconds,
|
||||
"metadata_payloads": self.timeline.payload_count,
|
||||
"metadata_position_transitions": self.timeline.transition_count,
|
||||
"input_samples": self.input_samples,
|
||||
"processed_samples_including_flush": self.processed_input_samples,
|
||||
"output_samples_before_tail_trim": self.output_samples,
|
||||
}
|
||||
@@ -0,0 +1,795 @@
|
||||
// joc_cli -- command line frontend, argument-compatible with the reference
|
||||
// Python CLI (main.py): the same positional input, the same mode selection
|
||||
// (ADM BWF by default, --speaker-layout or --binaural), the same option names,
|
||||
// choices and defaults, and the same default output naming under output/.
|
||||
//
|
||||
// Options that exist only because this build has no Python side or no Rosella
|
||||
// import chain (--backend python, --sofa-hrtf, --personalized-headphone,
|
||||
// metadata sidecars) fail with an explicit message instead of being ignored.
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <filesystem>
|
||||
#include <stdexcept>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/fs_utf8.h"
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace {
|
||||
|
||||
namespace fs = std::filesystem;
|
||||
namespace fs_utf8 = joc::fs_utf8;
|
||||
|
||||
constexpr double kRate = 48000.0;
|
||||
constexpr int kFrameSamples = 1536;
|
||||
|
||||
struct Options {
|
||||
std::string input;
|
||||
std::string output;
|
||||
std::string speaker_output;
|
||||
std::string binaural_output;
|
||||
std::string speaker_layout;
|
||||
bool binaural = false;
|
||||
std::string speaker_format = "float32";
|
||||
std::string binaural_format = "float32";
|
||||
std::string clip_action = "ask";
|
||||
int speaker_metadata_offset = 1473;
|
||||
std::string binaural_mode = "mid";
|
||||
std::string sofa_hrtf;
|
||||
std::string compiled_hrtf_cache;
|
||||
std::string personalized_headphone;
|
||||
bool personalized_headphone_used = false;
|
||||
std::string hrtf_cache_policy;
|
||||
std::string hrtf_cache_dir;
|
||||
double hrtf_radius_m = 1.0;
|
||||
double binaural_tail_seconds = 5.0;
|
||||
double binaural_tail_threshold = 1.0e-8;
|
||||
int binaural_chunk_frames = 64;
|
||||
double gain_db = 0.0;
|
||||
double duration = 0.0;
|
||||
bool duration_set = false;
|
||||
int object_delay_samples = 1473;
|
||||
std::string trajectory_mode = "compact";
|
||||
std::string ffmpeg;
|
||||
double eac3_drc_scale = 0.0;
|
||||
int eac3_target_level = 0;
|
||||
std::string backend = "auto";
|
||||
std::string native_library;
|
||||
int native_threads = 0;
|
||||
bool native_threads_set = false;
|
||||
std::string metadata_dir;
|
||||
std::string metadata_cache;
|
||||
std::string metadata_backend = "auto";
|
||||
std::string print_metadata = "none";
|
||||
std::string metadata_json;
|
||||
bool metadata_only = false;
|
||||
bool keep_raw = false;
|
||||
bool skip_sha256 = false;
|
||||
int progress_every = 1000;
|
||||
// C++-side additions (documented as such; the Python CLI has no equivalent).
|
||||
std::string bed;
|
||||
std::string kernels;
|
||||
std::string work_dir;
|
||||
std::string report_json;
|
||||
bool report_json_set = false;
|
||||
bool dry_run = false;
|
||||
bool quiet = false;
|
||||
bool help = false;
|
||||
};
|
||||
|
||||
const char* kLayoutChoices =
|
||||
"2.0 3.0 3.1 4.0 5.0 5.1 5.1.2 5.1.4 6.1 7.0 7.1 7.1.2 7.1.4 9.1.4 9.1.6 22.2";
|
||||
|
||||
void print_usage() {
|
||||
std::printf(
|
||||
"usage: joc_cli [options] input\n"
|
||||
"\n"
|
||||
"JustOneCacophony (JOC):E-AC-3 JOC → 25ch ADM BWF、扬声器 WAV 或双耳 WAV\n"
|
||||
"\n"
|
||||
"位置参数:\n"
|
||||
" input 输入 .m4a/.eac3/.ec3\n"
|
||||
"\n"
|
||||
"模式(默认输出 ADM BWF):\n"
|
||||
" --speaker-layout L 直接扬声器渲染布局,例如 2.0、5.1、7.1.2\n"
|
||||
" 可选值: %s\n"
|
||||
" --binaural 直接双耳渲染;不生成临时 ADM BWF\n"
|
||||
"\n"
|
||||
"输出:\n"
|
||||
" -o, --output PATH 输出文件;默认 output/<名称>.adm.wav、\n"
|
||||
" output/<名称>.<布局>.wav 或 output/<名称>.binaural.wav\n"
|
||||
" --speaker-output PATH 扬声器 WAV 路径;仅与 --speaker-layout 一起使用\n"
|
||||
" --binaural-output PATH 双耳 WAV 路径;仅与 --binaural 一起使用\n"
|
||||
" --speaker-format F 扬声器 WAV 格式 float32|int24,默认 float32\n"
|
||||
" --binaural-format F 双耳 WAV 格式 float32|int24,默认 float32\n"
|
||||
" --clip-action A int24 削波处理 ask|continue|float32|abort,默认 ask\n"
|
||||
"\n"
|
||||
"渲染:\n"
|
||||
" --speaker-metadata-offset N 扬声器渲染 metadata 相对帧偏移,默认 1473 samples\n"
|
||||
" --binaural-mode M 双耳渲染模式 off|near|mid|far,默认 mid;\n"
|
||||
" off 仅用于 ADM BWF(关闭 DBMD 双耳提示)\n"
|
||||
" --sofa-hrtf PATH SOFA SimpleFreeFieldHRIR 输入;.jochrtf 由本工具内部编译\n"
|
||||
" --personalized-headphone [PATH] Rosella 个性化模型,默认 "
|
||||
"HRTF/binaural.personalized_headphone\n"
|
||||
" --compiled-hrtf-cache PATH 直接读取 .jochrtf(高级用法,跳过 SOFA 编译)\n"
|
||||
" --hrtf-cache-policy P SOFA 编译缓存策略 none|memory|disk,默认 memory\n"
|
||||
" --hrtf-cache-dir DIR disk cache 目录,默认 <exe>/output/hrtf-cache\n"
|
||||
" --hrtf-radius-m R 选择最近的 SOFA measurement-radius shell,默认 1.0 m\n"
|
||||
" --binaural-tail-seconds S 双耳 room/filterbank flush 上限,默认 5 秒\n"
|
||||
" --binaural-tail-threshold T 双耳尾声裁切阈值,默认 1e-8;主体至少保留原时长\n"
|
||||
" --binaural-chunk-frames N 双耳内部批处理帧数,默认 64(本构建按 512 块渲染,\n"
|
||||
" 取值不影响输出)\n"
|
||||
" --gain-db X 成品增益 dB,默认 0;双耳路径以 float64 应用\n"
|
||||
" --duration S 只处理开头指定秒数\n"
|
||||
" --object-delay-samples N 对象 PCM/OAMD 时间补偿,默认 1473 samples\n"
|
||||
" --trajectory-mode M ADM 对象轨迹表示 compact|dense64,默认 compact\n"
|
||||
"\n"
|
||||
"输入与解码:\n"
|
||||
" --ffmpeg PATH ffmpeg 可执行文件,默认取 FFMPEG 环境变量或 PATH\n"
|
||||
" --eac3-drc-scale X E-AC-3 解码器 -drc_scale,0=关闭码流 dynrng,默认 0\n"
|
||||
" --eac3-target-level N E-AC-3 解码器 -target_level,0=不施加,默认 0\n"
|
||||
" --backend B JOC/扬声器 DSP 后端 auto|native;本构建无 python 后端\n"
|
||||
" --native-threads N 原生 DSP 总线程数;默认在 4 核以上使用 2\n"
|
||||
"\n"
|
||||
"诊断:\n"
|
||||
" --print-metadata M 诊断元数据输出 none|summary|frames,默认 none\n"
|
||||
" --metadata-json PATH 元数据汇总 JSON 路径\n"
|
||||
" --metadata-only 解析/打印元数据后退出\n"
|
||||
" --keep-raw 额外保留 16ch f32le 对象中间文件\n"
|
||||
" --skip-sha256 跳过最终文件 SHA-256 全量复扫\n"
|
||||
" --progress-every N 进度输出间隔,默认 1000 帧(渲染与收尾写盘同一节奏)\n"
|
||||
"\n"
|
||||
"本构建特有(Python 版没有对应参数):\n"
|
||||
" --bed PATH 已解码的 6 通道 float32 PCM;给出后不调用 ffmpeg 解码\n"
|
||||
" --kernels PATH 双耳滤波器组表 rosella_kernels.npz\n"
|
||||
" --work-dir DIR 临时目录\n"
|
||||
" --report-json PATH 结果 JSON 路径;默认 <输出>.report.json\n"
|
||||
" --dry-run 只校验配置\n"
|
||||
" --quiet 只输出警告与错误\n",
|
||||
kLayoutChoices);
|
||||
}
|
||||
|
||||
[[noreturn]] void fail(const std::string& message) { throw std::runtime_error(message); }
|
||||
|
||||
std::string require_value(const std::vector<std::string>& arguments, int* index) {
|
||||
if (static_cast<std::size_t>(*index) + 1u >= arguments.size()) {
|
||||
fail("argument " + arguments[static_cast<std::size_t>(*index)] +
|
||||
": expected one argument");
|
||||
}
|
||||
return arguments[static_cast<std::size_t>(++(*index))];
|
||||
}
|
||||
|
||||
double to_double(const std::string& text, const char* name) {
|
||||
try {
|
||||
std::size_t used = 0;
|
||||
const double value = std::stod(text, &used);
|
||||
if (used != text.size()) {
|
||||
fail(std::string(name) + ": invalid float value: " + text);
|
||||
}
|
||||
return value;
|
||||
} catch (const std::exception&) {
|
||||
fail(std::string(name) + ": invalid float value: " + text);
|
||||
}
|
||||
}
|
||||
|
||||
long long to_int(const std::string& text, const char* name) {
|
||||
try {
|
||||
std::size_t used = 0;
|
||||
const long long value = std::stoll(text, &used);
|
||||
if (used != text.size()) {
|
||||
fail(std::string(name) + ": invalid int value: " + text);
|
||||
}
|
||||
return value;
|
||||
} catch (const std::exception&) {
|
||||
fail(std::string(name) + ": invalid int value: " + text);
|
||||
}
|
||||
}
|
||||
|
||||
void check_choice(const std::string& value, const char* name,
|
||||
std::initializer_list<const char*> allowed) {
|
||||
for (const char* candidate : allowed) {
|
||||
if (value == candidate) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
std::string list;
|
||||
for (const char* candidate : allowed) {
|
||||
list += list.empty() ? candidate : (", " + std::string(candidate));
|
||||
}
|
||||
fail(std::string(name) + ": invalid choice: '" + value + "' (choose from " + list + ")");
|
||||
}
|
||||
|
||||
void parse_args(const std::vector<std::string>& arguments, Options* options) {
|
||||
std::vector<std::string> positional;
|
||||
const int argc = static_cast<int>(arguments.size());
|
||||
for (int index = 1; index < argc; ++index) {
|
||||
const std::string arg = arguments[static_cast<std::size_t>(index)];
|
||||
if (arg == "-h" || arg == "--help") { options->help = true; }
|
||||
else if (arg == "-o" || arg == "--output") { options->output = require_value(arguments, &index); }
|
||||
else if (arg == "--speaker-output") { options->speaker_output = require_value(arguments, &index); }
|
||||
else if (arg == "--binaural-output") { options->binaural_output = require_value(arguments, &index); }
|
||||
else if (arg == "--speaker-layout") { options->speaker_layout = require_value(arguments, &index); }
|
||||
else if (arg == "--binaural") { options->binaural = true; }
|
||||
else if (arg == "--speaker-format") { options->speaker_format = require_value(arguments, &index); }
|
||||
else if (arg == "--binaural-format") { options->binaural_format = require_value(arguments, &index); }
|
||||
else if (arg == "--clip-action") { options->clip_action = require_value(arguments, &index); }
|
||||
else if (arg == "--speaker-metadata-offset") {
|
||||
options->speaker_metadata_offset = static_cast<int>(
|
||||
to_int(require_value(arguments, &index), "--speaker-metadata-offset"));
|
||||
}
|
||||
else if (arg == "--binaural-mode") { options->binaural_mode = require_value(arguments, &index); }
|
||||
else if (arg == "--sofa-hrtf") { options->sofa_hrtf = require_value(arguments, &index); }
|
||||
else if (arg == "--compiled-hrtf-cache") { options->compiled_hrtf_cache = require_value(arguments, &index); }
|
||||
else if (arg == "--personalized-headphone") {
|
||||
options->personalized_headphone_used = true;
|
||||
// nargs="?": the path is optional, so the next token may be the input.
|
||||
// Without a path the executable-anchored default is resolved later.
|
||||
if (static_cast<std::size_t>(index) + 1u < arguments.size() &&
|
||||
arguments[static_cast<std::size_t>(index) + 1u][0] != '-') {
|
||||
options->personalized_headphone = require_value(arguments, &index);
|
||||
}
|
||||
}
|
||||
else if (arg == "--hrtf-cache-policy") { options->hrtf_cache_policy = require_value(arguments, &index); }
|
||||
else if (arg == "--hrtf-cache-dir") { options->hrtf_cache_dir = require_value(arguments, &index); }
|
||||
else if (arg == "--hrtf-radius-m") { options->hrtf_radius_m = to_double(require_value(arguments, &index), "--hrtf-radius-m"); }
|
||||
else if (arg == "--binaural-tail-seconds") { options->binaural_tail_seconds = to_double(require_value(arguments, &index), "--binaural-tail-seconds"); }
|
||||
else if (arg == "--binaural-tail-threshold") { options->binaural_tail_threshold = to_double(require_value(arguments, &index), "--binaural-tail-threshold"); }
|
||||
else if (arg == "--binaural-chunk-frames") { options->binaural_chunk_frames = static_cast<int>(to_int(require_value(arguments, &index), "--binaural-chunk-frames")); }
|
||||
else if (arg == "--gain-db") { options->gain_db = to_double(require_value(arguments, &index), "--gain-db"); }
|
||||
else if (arg == "--duration") { options->duration = to_double(require_value(arguments, &index), "--duration"); options->duration_set = true; }
|
||||
else if (arg == "--object-delay-samples") { options->object_delay_samples = static_cast<int>(to_int(require_value(arguments, &index), "--object-delay-samples")); }
|
||||
else if (arg == "--trajectory-mode") { options->trajectory_mode = require_value(arguments, &index); }
|
||||
else if (arg == "--ffmpeg") { options->ffmpeg = require_value(arguments, &index); }
|
||||
else if (arg == "--eac3-drc-scale") { options->eac3_drc_scale = to_double(require_value(arguments, &index), "--eac3-drc-scale"); }
|
||||
else if (arg == "--eac3-target-level") { options->eac3_target_level = static_cast<int>(to_int(require_value(arguments, &index), "--eac3-target-level")); }
|
||||
else if (arg == "--backend") { options->backend = require_value(arguments, &index); }
|
||||
else if (arg == "--native-library") { options->native_library = require_value(arguments, &index); }
|
||||
else if (arg == "--native-threads") { options->native_threads = static_cast<int>(to_int(require_value(arguments, &index), "--native-threads")); options->native_threads_set = true; }
|
||||
else if (arg == "--metadata-dir") { options->metadata_dir = require_value(arguments, &index); }
|
||||
else if (arg == "--metadata-cache") { options->metadata_cache = require_value(arguments, &index); }
|
||||
else if (arg == "--metadata-backend") { options->metadata_backend = require_value(arguments, &index); }
|
||||
else if (arg == "--print-metadata") { options->print_metadata = require_value(arguments, &index); }
|
||||
else if (arg == "--metadata-json") { options->metadata_json = require_value(arguments, &index); }
|
||||
else if (arg == "--metadata-only") { options->metadata_only = true; }
|
||||
else if (arg == "--keep-raw") { options->keep_raw = true; }
|
||||
else if (arg == "--skip-sha256") { options->skip_sha256 = true; }
|
||||
else if (arg == "--progress-every") { options->progress_every = static_cast<int>(to_int(require_value(arguments, &index), "--progress-every")); }
|
||||
else if (arg == "--bed") { options->bed = require_value(arguments, &index); }
|
||||
else if (arg == "--kernels") { options->kernels = require_value(arguments, &index); }
|
||||
else if (arg == "--work-dir") { options->work_dir = require_value(arguments, &index); }
|
||||
else if (arg == "--report-json") { options->report_json = require_value(arguments, &index); options->report_json_set = true; }
|
||||
else if (arg == "--dry-run") { options->dry_run = true; }
|
||||
else if (arg == "--quiet") { options->quiet = true; }
|
||||
else if (!arg.empty() && arg[0] == '-' && arg != "-") { fail("unrecognized argument: " + arg); }
|
||||
else { positional.push_back(arg); }
|
||||
}
|
||||
if (positional.size() > 1u) {
|
||||
fail("unrecognized extra arguments: " + positional[1] +
|
||||
(positional.size() > 2u ? " ..." : ""));
|
||||
}
|
||||
if (!positional.empty()) {
|
||||
options->input = positional.front();
|
||||
}
|
||||
}
|
||||
|
||||
// Mirrors the reference resolve_output(): <project>/output plus a mode-specific
|
||||
// name. The project directory is the executable's directory, as upstream uses
|
||||
// the script's directory, so the layout does not depend on the working directory.
|
||||
std::string resolve_output(const Options& options, const std::string& source,
|
||||
const std::string& executable_directory) {
|
||||
const std::string requested = !options.speaker_output.empty() ? options.speaker_output
|
||||
: !options.binaural_output.empty() ? options.binaural_output
|
||||
: options.output;
|
||||
if (!requested.empty()) {
|
||||
std::error_code error;
|
||||
const fs::path absolute = fs::absolute(fs_utf8::to_path(requested), error);
|
||||
return error ? requested : fs_utf8::from_path(absolute);
|
||||
}
|
||||
const fs::path directory = fs_utf8::to_path(executable_directory) / "output";
|
||||
const std::string stem = fs_utf8::from_path(fs_utf8::to_path(source).stem());
|
||||
if (!options.speaker_layout.empty()) {
|
||||
return fs_utf8::from_path(directory /
|
||||
fs_utf8::to_path(stem + "." + options.speaker_layout + ".wav"));
|
||||
}
|
||||
if (options.binaural) {
|
||||
return fs_utf8::from_path(directory / fs_utf8::to_path(stem + ".binaural.wav"));
|
||||
}
|
||||
return fs_utf8::from_path(directory / fs_utf8::to_path(stem + ".adm.wav"));
|
||||
}
|
||||
|
||||
// The project directory the reference anchors its defaults at: the directory of
|
||||
// the running executable, never the working directory.
|
||||
std::string executable_dir(const std::string& argv0) {
|
||||
const std::string own_path = fs_utf8::executable_path();
|
||||
if (!own_path.empty()) {
|
||||
const fs::path path = fs_utf8::to_path(own_path);
|
||||
if (path.has_parent_path()) {
|
||||
return fs_utf8::from_path(path.parent_path());
|
||||
}
|
||||
}
|
||||
if (argv0.empty()) {
|
||||
return ".";
|
||||
}
|
||||
std::error_code error;
|
||||
const fs::path path = fs::absolute(fs_utf8::to_path(argv0), error);
|
||||
if (error || path.empty()) {
|
||||
return ".";
|
||||
}
|
||||
return fs_utf8::from_path(path.parent_path());
|
||||
}
|
||||
|
||||
std::string find_kernels(const Options& options, const std::string& argv0) {
|
||||
(void)argv0;
|
||||
if (!options.kernels.empty() && !fs_utf8::exists(options.kernels)) {
|
||||
fail("--kernels 指向的文件不存在: " + options.kernels);
|
||||
}
|
||||
// Empty means the tables compiled into the library.
|
||||
return options.kernels;
|
||||
}
|
||||
|
||||
// Mirrors the reference binaural HRTF resolution (main.py:89-160): the SOFA file
|
||||
// is the user-facing input and the .jochrtf is only its compiled cache. Paths
|
||||
// are anchored at the executable directory, as the reference anchors them at the
|
||||
// project directory.
|
||||
struct HrtfInput {
|
||||
std::string sofa_path; // compile this
|
||||
std::string compiled_path; // or read this .jochrtf directly
|
||||
std::string cache_dir; // disk policy directory
|
||||
std::string personalized_path; // Rosella .personalized_headphone
|
||||
bool disk = false;
|
||||
};
|
||||
|
||||
std::string resolve_compiled_hrtf(const Options& options, const std::string& project_directory) {
|
||||
if (!options.compiled_hrtf_cache.empty()) {
|
||||
return options.compiled_hrtf_cache;
|
||||
}
|
||||
const std::string directory_utf8 =
|
||||
options.hrtf_cache_dir.empty()
|
||||
? fs_utf8::from_path(fs_utf8::to_path(project_directory) / "output" / "hrtf-cache")
|
||||
: options.hrtf_cache_dir;
|
||||
if (!fs_utf8::is_directory(directory_utf8)) {
|
||||
return std::string();
|
||||
}
|
||||
const fs::path directory = fs_utf8::to_path(directory_utf8);
|
||||
std::vector<fs::path> candidates;
|
||||
for (const fs::directory_entry& entry : fs::directory_iterator(directory)) {
|
||||
if (entry.is_regular_file() && entry.path().extension() == ".jochrtf") {
|
||||
candidates.push_back(entry.path());
|
||||
}
|
||||
}
|
||||
std::sort(candidates.begin(), candidates.end());
|
||||
if (candidates.size() > 1u) {
|
||||
fail(directory_utf8 +
|
||||
" 下有多个 .jochrtf 缓存,无法自动选择;请用 --sofa-hrtf PATH 或 "
|
||||
"--compiled-hrtf-cache PATH 显式指定");
|
||||
}
|
||||
return candidates.empty() ? std::string() : fs_utf8::from_path(candidates.front());
|
||||
}
|
||||
|
||||
HrtfInput resolve_hrtf_input(const Options& options, const std::string& project_directory) {
|
||||
HrtfInput input;
|
||||
const std::string default_sofa =
|
||||
fs_utf8::from_path(fs_utf8::to_path(project_directory) / "HRTF" / "binaural.sofa");
|
||||
const std::string default_private = fs_utf8::from_path(
|
||||
fs_utf8::to_path(project_directory) / "HRTF" / "binaural.personalized_headphone");
|
||||
const std::string default_cache_dir =
|
||||
fs_utf8::from_path(fs_utf8::to_path(project_directory) / "output" / "hrtf-cache");
|
||||
|
||||
if (!options.compiled_hrtf_cache.empty() && !options.hrtf_cache_policy.empty()) {
|
||||
fail("显式 .jochrtf 输入不能再指定 --hrtf-cache-policy");
|
||||
}
|
||||
if (!options.compiled_hrtf_cache.empty() && options.hrtf_radius_m != 1.0) {
|
||||
fail("显式 .jochrtf 输入不能再选择 SOFA radius shell");
|
||||
}
|
||||
if (options.personalized_headphone_used &&
|
||||
(!options.hrtf_cache_policy.empty() || !options.hrtf_cache_dir.empty() ||
|
||||
options.hrtf_radius_m != 1.0)) {
|
||||
fail("Rosella 模型输入不能使用 --hrtf-cache-policy/--hrtf-cache-dir/--hrtf-radius-m");
|
||||
}
|
||||
|
||||
std::string sofa = options.sofa_hrtf;
|
||||
std::string compiled = options.compiled_hrtf_cache;
|
||||
std::string personalized =
|
||||
options.personalized_headphone_used ? options.personalized_headphone : std::string();
|
||||
if (options.personalized_headphone_used && personalized.empty()) {
|
||||
// "--personalized-headphone" without a path means the project default.
|
||||
personalized = default_private;
|
||||
}
|
||||
if (sofa.empty() && compiled.empty() && personalized.empty()) {
|
||||
// The reference order: the SOFA file, then the unique compiled cache, then the
|
||||
// personalized model.
|
||||
if (fs_utf8::exists(default_sofa)) {
|
||||
sofa = default_sofa;
|
||||
} else {
|
||||
compiled = resolve_compiled_hrtf(options, project_directory);
|
||||
if (compiled.empty() && fs_utf8::exists(default_private)) {
|
||||
personalized = default_private;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!personalized.empty()) {
|
||||
if (!fs_utf8::exists(personalized)) {
|
||||
fail("双耳模型不存在: " + personalized);
|
||||
}
|
||||
input.personalized_path = personalized;
|
||||
return input;
|
||||
}
|
||||
if (sofa.empty() && compiled.empty()) {
|
||||
if (!options.hrtf_cache_policy.empty() || !options.hrtf_cache_dir.empty() ||
|
||||
options.hrtf_radius_m != 1.0) {
|
||||
fail("HRTF cache/radius 选项需要 --sofa-hrtf");
|
||||
}
|
||||
fail("--binaural 未找到 HRTF 输入:默认 " + default_sofa + "、" + default_private +
|
||||
" 或 " + default_cache_dir +
|
||||
" 下的 .jochrtf 都不存在,请用 --sofa-hrtf PATH、--personalized-headphone PATH "
|
||||
"或 --compiled-hrtf-cache PATH 指定");
|
||||
}
|
||||
if (sofa.empty() && (!options.hrtf_cache_policy.empty() || !options.hrtf_cache_dir.empty() ||
|
||||
options.hrtf_radius_m != 1.0)) {
|
||||
fail("HRTF cache/radius 选项需要 --sofa-hrtf");
|
||||
}
|
||||
if (sofa.empty()) {
|
||||
input.compiled_path = compiled;
|
||||
return input;
|
||||
}
|
||||
const std::string effective_policy =
|
||||
options.hrtf_cache_policy.empty() ? "memory" : options.hrtf_cache_policy;
|
||||
if (!options.hrtf_cache_dir.empty() && effective_policy != "disk") {
|
||||
fail("--hrtf-cache-dir 需要 SOFA 与 disk cache policy 一起使用");
|
||||
}
|
||||
input.sofa_path = sofa;
|
||||
input.disk = effective_policy == "disk";
|
||||
input.cache_dir = options.hrtf_cache_dir.empty() ? default_cache_dir : options.hrtf_cache_dir;
|
||||
if (!fs_utf8::exists(input.sofa_path)) {
|
||||
fail("SOFA HRTF 不存在: " + input.sofa_path);
|
||||
}
|
||||
return input;
|
||||
}
|
||||
|
||||
std::uint32_t binaural_mode_value(const std::string& name) {
|
||||
if (name == "off") { return JOC_BINAURAL_OFF; }
|
||||
if (name == "near") { return JOC_BINAURAL_NEAR; }
|
||||
if (name == "far") { return JOC_BINAURAL_FAR; }
|
||||
return JOC_BINAURAL_MID;
|
||||
}
|
||||
|
||||
std::uint32_t clip_action_value(const std::string& name) {
|
||||
if (name == "continue") { return JOC_CLIP_CONTINUE; }
|
||||
if (name == "float32") { return JOC_CLIP_FLOAT32; }
|
||||
if (name == "abort") { return JOC_CLIP_ABORT; }
|
||||
return JOC_CLIP_ASK;
|
||||
}
|
||||
|
||||
std::string format_eta(double seconds) {
|
||||
if (seconds < 0.0 || seconds > 86400.0) {
|
||||
return "--";
|
||||
}
|
||||
char buffer[64];
|
||||
std::snprintf(buffer, sizeof(buffer), "%.0fs", seconds);
|
||||
return buffer;
|
||||
}
|
||||
|
||||
void JOC_CALL on_event(void* user, const joc_event* event) {
|
||||
const Options* options = static_cast<const Options*>(user);
|
||||
if (event == nullptr) {
|
||||
return;
|
||||
}
|
||||
switch (event->type) {
|
||||
case JOC_EV_PROGRESS: {
|
||||
if (options->quiet) {
|
||||
return;
|
||||
}
|
||||
const double fraction = event->progress >= 0.0 ? event->progress : 0.0;
|
||||
const double remaining =
|
||||
fraction > 0.0 ? event->elapsed_seconds * (1.0 - fraction) / fraction : -1.0;
|
||||
std::printf("[%s] %llu/%llu %.1fx realtime ETA %s\n", event->stage_name,
|
||||
static_cast<unsigned long long>(event->current_frame),
|
||||
static_cast<unsigned long long>(event->total_frames),
|
||||
event->realtime_factor, format_eta(remaining).c_str());
|
||||
std::fflush(stdout);
|
||||
return;
|
||||
}
|
||||
case JOC_EV_LOG: {
|
||||
if (options->quiet || event->log_level < JOC_LOG_INFO || event->message[0] == '\0') {
|
||||
return;
|
||||
}
|
||||
std::printf("[%s] %s\n", event->stage_name, event->message);
|
||||
std::fflush(stdout);
|
||||
return;
|
||||
}
|
||||
case JOC_EV_WARNING:
|
||||
std::printf("[warning] %s\n", event->message);
|
||||
return;
|
||||
case JOC_EV_ERROR:
|
||||
std::fprintf(stderr, "[error] %s (%s)\n", event->message,
|
||||
joc_error_name(event->error_code));
|
||||
return;
|
||||
default:
|
||||
if (options->quiet || event->log_level < JOC_LOG_INFO || event->message[0] == '\0') {
|
||||
return;
|
||||
}
|
||||
std::printf("[%s] %s\n", event->stage_name, event->message);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
fs_utf8::configure_console();
|
||||
const std::vector<std::string> arguments = fs_utf8::command_line_arguments(argc, argv);
|
||||
Options options;
|
||||
try {
|
||||
parse_args(arguments, &options);
|
||||
} catch (const std::exception& error) {
|
||||
std::fprintf(stderr, "joc_cli: error: %s\n", error.what());
|
||||
return 2;
|
||||
}
|
||||
if (options.help) {
|
||||
print_usage();
|
||||
return 0;
|
||||
}
|
||||
if (options.input.empty()) {
|
||||
print_usage();
|
||||
return 2;
|
||||
}
|
||||
|
||||
try {
|
||||
check_choice(options.speaker_format, "--speaker-format", {"float32", "int24"});
|
||||
check_choice(options.binaural_format, "--binaural-format", {"float32", "int24"});
|
||||
check_choice(options.clip_action, "--clip-action",
|
||||
{"ask", "continue", "float32", "abort"});
|
||||
check_choice(options.binaural_mode, "--binaural-mode", {"off", "near", "mid", "far"});
|
||||
check_choice(options.trajectory_mode, "--trajectory-mode", {"compact", "dense64"});
|
||||
check_choice(options.backend, "--backend", {"auto", "native", "python"});
|
||||
check_choice(options.print_metadata, "--print-metadata", {"none", "summary", "frames"});
|
||||
check_choice(options.metadata_backend, "--metadata-backend", {"auto", "emdf", "sidecar"});
|
||||
if (!options.hrtf_cache_policy.empty()) {
|
||||
check_choice(options.hrtf_cache_policy, "--hrtf-cache-policy",
|
||||
{"none", "memory", "disk"});
|
||||
}
|
||||
|
||||
const bool speaker_mode = !options.speaker_layout.empty();
|
||||
const bool binaural_mode = options.binaural;
|
||||
if (speaker_mode && binaural_mode) {
|
||||
fail("argument --binaural: not allowed with argument --speaker-layout");
|
||||
}
|
||||
if (options.binaural_mode == "off" && (speaker_mode || binaural_mode)) {
|
||||
fail("--binaural-mode off 仅用于 ADM BWF 输出(关闭 DBMD 双耳提示);"
|
||||
"直接双耳渲染请使用 near/mid/far");
|
||||
}
|
||||
if (!options.speaker_output.empty() && !speaker_mode) {
|
||||
fail("--speaker-output 必须与 --speaker-layout 一起使用");
|
||||
}
|
||||
if (!options.binaural_output.empty() && !binaural_mode) {
|
||||
fail("--binaural-output 必须与 --binaural 一起使用");
|
||||
}
|
||||
const bool specific_output =
|
||||
!options.speaker_output.empty() || !options.binaural_output.empty();
|
||||
if (!options.output.empty() && specific_output) {
|
||||
fail("-o/--output 与 --speaker-output/--binaural-output 不能同时使用");
|
||||
}
|
||||
if (!options.speaker_output.empty() && !options.binaural_output.empty()) {
|
||||
fail("--speaker-output 与 --binaural-output 不能同时使用");
|
||||
}
|
||||
if (options.speaker_metadata_offset < 0) {
|
||||
fail("speaker-metadata-offset 不能为负数");
|
||||
}
|
||||
const bool hrtf_options_used =
|
||||
!options.sofa_hrtf.empty() || !options.compiled_hrtf_cache.empty() ||
|
||||
options.personalized_headphone_used || !options.hrtf_cache_policy.empty() ||
|
||||
!options.hrtf_cache_dir.empty() || options.hrtf_radius_m != 1.0;
|
||||
if (hrtf_options_used && !binaural_mode) {
|
||||
fail("SOFA/HRTF 选项仅与 --binaural 一起使用");
|
||||
}
|
||||
if (!std::isfinite(options.binaural_tail_seconds) ||
|
||||
options.binaural_tail_seconds < 0.0) {
|
||||
fail("binaural-tail-seconds 必须是非负有限值");
|
||||
}
|
||||
if (!std::isfinite(options.binaural_tail_threshold) ||
|
||||
options.binaural_tail_threshold < 0.0) {
|
||||
fail("binaural-tail-threshold 必须是非负有限值");
|
||||
}
|
||||
if (options.binaural_chunk_frames <= 0) {
|
||||
fail("binaural-chunk-frames 必须大于 0");
|
||||
}
|
||||
if (!std::isfinite(options.hrtf_radius_m) || options.hrtf_radius_m <= 0.0) {
|
||||
fail("hrtf-radius-m 必须是正有限值");
|
||||
}
|
||||
if (options.duration_set && options.duration <= 0.0) {
|
||||
fail("duration 必须大于 0");
|
||||
}
|
||||
if (options.object_delay_samples < 0) {
|
||||
fail("object-delay-samples 不能为负数");
|
||||
}
|
||||
if (options.native_threads_set && options.native_threads < 1) {
|
||||
fail("native-threads 必须大于 0");
|
||||
}
|
||||
if (!std::isfinite(options.gain_db) || std::abs(options.gain_db) > 200.0) {
|
||||
fail("gain-db 超出支持范围");
|
||||
}
|
||||
// Options this build cannot honour: fail loudly instead of ignoring them.
|
||||
if (options.backend == "python") {
|
||||
fail("--backend python 在本构建中不可用(已无 Python 后端);请使用 auto 或 native");
|
||||
}
|
||||
if (options.metadata_backend == "sidecar" || !options.metadata_dir.empty() ||
|
||||
!options.metadata_cache.empty()) {
|
||||
fail("metadata sidecar 在本构建中不可用(始终直接扫描 EMDF)");
|
||||
}
|
||||
if (!options.native_library.empty()) {
|
||||
std::fprintf(stderr, "[info] --native-library 在本构建中忽略(单一 joc_core.dll)\n");
|
||||
}
|
||||
if (!fs_utf8::exists(options.input)) {
|
||||
fail("输入文件不存在: " + options.input);
|
||||
}
|
||||
} catch (const std::exception& error) {
|
||||
std::fprintf(stderr, "joc_cli: error: %s\n", error.what());
|
||||
return 2;
|
||||
}
|
||||
|
||||
const bool speaker_mode = !options.speaker_layout.empty();
|
||||
const bool binaural_mode = options.binaural;
|
||||
const std::string project_directory =
|
||||
executable_dir(arguments.empty() ? std::string() : arguments.front());
|
||||
const std::string output_path = resolve_output(options, options.input, project_directory);
|
||||
std::error_code directory_error;
|
||||
fs::create_directories(fs_utf8::to_path(output_path).parent_path(), directory_error);
|
||||
|
||||
joc_task_config config{};
|
||||
config.struct_size = sizeof(config);
|
||||
config.struct_version = JOC_TASK_CONFIG_VERSION;
|
||||
config.input_path = options.input.c_str();
|
||||
config.output_path = output_path.c_str();
|
||||
config.ffmpeg_path = options.ffmpeg.empty() ? nullptr : options.ffmpeg.c_str();
|
||||
config.bed_path = options.bed.empty() ? nullptr : options.bed.c_str();
|
||||
config.work_dir = options.work_dir.empty() ? nullptr : options.work_dir.c_str();
|
||||
config.eac3_drc_scale = options.eac3_drc_scale;
|
||||
config.eac3_target_level = options.eac3_target_level;
|
||||
config.operation =
|
||||
binaural_mode ? JOC_OP_BINAURAL : speaker_mode ? JOC_OP_SPEAKER : JOC_OP_ADM_BWF;
|
||||
const std::string& requested_format =
|
||||
binaural_mode ? options.binaural_format : options.speaker_format;
|
||||
config.output_format =
|
||||
requested_format == "int24" ? JOC_FORMAT_PCM24 : JOC_FORMAT_FLOAT32;
|
||||
config.clip_action = clip_action_value(options.clip_action);
|
||||
config.speaker_layout_name = speaker_mode ? options.speaker_layout.c_str() : nullptr;
|
||||
config.speaker_metadata_offset = static_cast<std::uint32_t>(options.speaker_metadata_offset);
|
||||
config.binaural_mode = binaural_mode_value(options.binaural_mode);
|
||||
config.adm_binaural_mode = binaural_mode_value(options.binaural_mode);
|
||||
config.binaural_tail_seconds = options.binaural_tail_seconds;
|
||||
config.binaural_tail_threshold = options.binaural_tail_threshold;
|
||||
config.binaural_chunk_frames = static_cast<std::uint32_t>(options.binaural_chunk_frames);
|
||||
config.object_delay_samples = static_cast<std::uint32_t>(options.object_delay_samples);
|
||||
config.trajectory_mode =
|
||||
options.trajectory_mode == "dense64" ? JOC_TRAJECTORY_DENSE64 : JOC_TRAJECTORY_COMPACT;
|
||||
config.gain_db = options.gain_db;
|
||||
config.progress_interval_frames = static_cast<std::uint32_t>(options.progress_every);
|
||||
config.native_threads =
|
||||
options.native_threads_set ? static_cast<std::uint32_t>(options.native_threads) : 0u;
|
||||
config.print_metadata = options.print_metadata == "frames" ? 2u
|
||||
: options.print_metadata == "summary" ? 1u
|
||||
: 0u;
|
||||
config.metadata_json_path =
|
||||
options.metadata_json.empty() ? nullptr : options.metadata_json.c_str();
|
||||
config.duration_frames =
|
||||
options.duration_set
|
||||
? static_cast<std::uint64_t>(
|
||||
std::ceil(options.duration * kRate / static_cast<double>(kFrameSamples)))
|
||||
: 0u;
|
||||
config.flags = 0u;
|
||||
if (options.skip_sha256) { config.flags |= JOC_TASK_F_SKIP_SHA256; }
|
||||
if (options.keep_raw) { config.flags |= JOC_TASK_F_KEEP_INTERMEDIATE; }
|
||||
if (options.metadata_only) { config.flags |= JOC_TASK_F_METADATA_ONLY; }
|
||||
if (options.quiet) { config.flags |= JOC_TASK_F_QUIET; }
|
||||
|
||||
std::string hrtf_path;
|
||||
std::string hrtf_sofa_path;
|
||||
std::string hrtf_cache_dir;
|
||||
std::string personalized_path;
|
||||
std::string kernels_path;
|
||||
if (binaural_mode) {
|
||||
try {
|
||||
const HrtfInput input = resolve_hrtf_input(options, project_directory);
|
||||
hrtf_path = input.compiled_path;
|
||||
hrtf_sofa_path = input.sofa_path;
|
||||
hrtf_cache_dir = input.cache_dir;
|
||||
personalized_path = input.personalized_path;
|
||||
kernels_path = find_kernels(options, arguments.empty() ? std::string()
|
||||
: arguments.front());
|
||||
} catch (const std::exception& error) {
|
||||
std::fprintf(stderr, "joc_cli: error: %s\n", error.what());
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
config.hrtf_path = hrtf_path.empty() ? nullptr : hrtf_path.c_str();
|
||||
config.hrtf_sofa_path = hrtf_sofa_path.empty() ? nullptr : hrtf_sofa_path.c_str();
|
||||
config.hrtf_cache_dir = hrtf_cache_dir.empty() ? nullptr : hrtf_cache_dir.c_str();
|
||||
config.personalized_headphone_path =
|
||||
personalized_path.empty() ? nullptr : personalized_path.c_str();
|
||||
config.hrtf_cache_policy = options.hrtf_cache_policy == "disk" ? JOC_HRTF_CACHE_DISK
|
||||
: options.hrtf_cache_policy == "none" ? JOC_HRTF_CACHE_NONE
|
||||
: JOC_HRTF_CACHE_MEMORY;
|
||||
config.hrtf_radius_m = options.hrtf_radius_m;
|
||||
config.kernels_path = kernels_path.empty() ? nullptr : kernels_path.c_str();
|
||||
|
||||
joc_validation_issue issues[32];
|
||||
std::uint32_t issue_count = 0;
|
||||
const joc_error validated = joc_task_validate(&config, issues, 32u, &issue_count);
|
||||
for (std::uint32_t index = 0; index < std::min(issue_count, 32u); ++index) {
|
||||
if (issues[index].severity >= 2u) {
|
||||
std::fprintf(stderr, "[error] %s: %s\n", issues[index].field, issues[index].message);
|
||||
} else if (!options.quiet) {
|
||||
std::fprintf(stderr, "[warning] %s: %s\n", issues[index].field,
|
||||
issues[index].message);
|
||||
}
|
||||
}
|
||||
if (validated != JOC_OK) {
|
||||
std::fprintf(stderr, "joc_cli: error: configuration rejected (%u issue(s))\n", issue_count);
|
||||
return 2;
|
||||
}
|
||||
if (options.dry_run) {
|
||||
std::printf("configuration accepted (%u issue(s))\n", issue_count);
|
||||
return 0;
|
||||
}
|
||||
|
||||
joc_event_sink sink{};
|
||||
sink.struct_size = sizeof(sink);
|
||||
sink.callback = &on_event;
|
||||
sink.user = &options;
|
||||
|
||||
if (!options.quiet) {
|
||||
const char* mode_name = binaural_mode ? "binaural" : speaker_mode ? "speaker" : "adm";
|
||||
std::printf("[cli] %s -> %s (%s)\n", options.input.c_str(), output_path.c_str(),
|
||||
mode_name);
|
||||
std::fflush(stdout);
|
||||
}
|
||||
|
||||
joc_task_result result{};
|
||||
const joc_error status = joc_task_execute(&config, &sink, &result);
|
||||
|
||||
const std::string report_path =
|
||||
options.report_json_set ? options.report_json : (output_path + ".report.json");
|
||||
{
|
||||
std::size_t needed = 0;
|
||||
joc_task_result_to_json(&result, nullptr, 0u, &needed);
|
||||
std::vector<char> buffer(needed + 1u);
|
||||
if (joc_task_result_to_json(&result, buffer.data(), buffer.size(), &needed) == JOC_OK) {
|
||||
if (std::FILE* file = fs_utf8::fopen(report_path, "wb")) {
|
||||
std::fwrite(buffer.data(), 1, std::strlen(buffer.data()), file);
|
||||
std::fputc('\n', file);
|
||||
std::fclose(file);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::printf("\nresult: %s\n", status == JOC_OK ? "ok" : joc_error_name(status));
|
||||
std::printf(" output : %s\n", output_path.c_str());
|
||||
std::printf(" frames : %llu (%.2f s)\n",
|
||||
static_cast<unsigned long long>(result.input_frames), result.duration_sec);
|
||||
std::printf(" output samples: %llu\n",
|
||||
static_cast<unsigned long long>(result.output_samples));
|
||||
std::printf(" output bytes : %llu\n",
|
||||
static_cast<unsigned long long>(result.output_file_bytes));
|
||||
std::printf(" format : %s\n",
|
||||
result.output_format_actual == JOC_FORMAT_PCM24 ? "int24" : "float32");
|
||||
std::printf(" peak : %.9g (%llu sample(s) above full scale)\n", result.output_peak,
|
||||
static_cast<unsigned long long>(result.output_over_unity_values));
|
||||
std::printf(" sha256 : %s\n",
|
||||
result.output_sha256[0] != '\0' ? result.output_sha256 : "(skipped)");
|
||||
std::printf(" report : %s\n", report_path.c_str());
|
||||
// Each stage time is measured where that stage actually runs, and the three
|
||||
// stages now overlap (see the pipeline in src/task/task.cpp), so the stage
|
||||
// times deliberately do not add up to the wall-clock total.
|
||||
std::printf(" timings : decode %.2fs, joc %.2fs, dsp %.2fs, write %.2fs"
|
||||
" (stage times, concurrent), total %.2fs\n",
|
||||
result.t_decode_bed, result.t_render, result.t_render_dsp, result.t_write_file,
|
||||
result.t_total);
|
||||
if (status != JOC_OK) {
|
||||
std::fprintf(stderr, "joc_cli: error: %s: %s\n", result.error_stage, result.error_message);
|
||||
}
|
||||
return status == JOC_OK ? 0 : 1;
|
||||
}
|
||||
@@ -0,0 +1,113 @@
|
||||
#include "eac3_transport/eac3_reader.h"
|
||||
|
||||
#include <utility>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::eac3 {
|
||||
|
||||
namespace {
|
||||
constexpr std::size_t kHeaderBytes = 4;
|
||||
} // namespace
|
||||
|
||||
void FrameReader::push(const std::uint8_t* data, std::size_t size) {
|
||||
if (failed_ || data == nullptr || size == 0) {
|
||||
return;
|
||||
}
|
||||
if (consumed_ > 0) {
|
||||
compact();
|
||||
}
|
||||
buffer_.insert(buffer_.end(), data, data + size);
|
||||
}
|
||||
|
||||
void FrameReader::compact() {
|
||||
if (consumed_ == 0) {
|
||||
return;
|
||||
}
|
||||
buffer_.erase(buffer_.begin(), buffer_.begin() + static_cast<std::ptrdiff_t>(consumed_));
|
||||
base_offset_ += consumed_;
|
||||
consumed_ = 0;
|
||||
}
|
||||
|
||||
void FrameReader::fail(joc_error code, std::string message) {
|
||||
failed_ = true;
|
||||
error_ = code;
|
||||
message_ = std::move(message);
|
||||
}
|
||||
|
||||
FrameReader::Next FrameReader::next(Frame* out) {
|
||||
if (failed_) {
|
||||
return Next::Fail;
|
||||
}
|
||||
const std::size_t available = buffer_.size() - consumed_;
|
||||
if (available == 0) {
|
||||
return Next::End;
|
||||
}
|
||||
const std::uint8_t* p = buffer_.data() + consumed_;
|
||||
|
||||
// The reference implementation rejects a frame whose header does not fit,
|
||||
// rather than silently resynchronising on the next 0x0B77.
|
||||
if (available < kHeaderBytes) {
|
||||
if (finished_) {
|
||||
fail(JOC_ERR_EAC3_SYNCFRAME, "E-AC-3 syncframe header truncated at end of input");
|
||||
return Next::Fail;
|
||||
}
|
||||
return Next::End;
|
||||
}
|
||||
|
||||
const std::uint16_t syncword = static_cast<std::uint16_t>((static_cast<std::uint16_t>(p[0]) << 8) | p[1]);
|
||||
if (syncword != kSyncword) {
|
||||
fail(JOC_ERR_EAC3_SYNCFRAME, "invalid E-AC-3 syncword (silent resynchronisation is not allowed)");
|
||||
return Next::Fail;
|
||||
}
|
||||
|
||||
// frmsiz: 11 bits spread over the low 3 bits of byte 2 and all of byte 3,
|
||||
const std::size_t words =
|
||||
static_cast<std::size_t>(((p[2] & 0x07u) << 8) | p[3]) + 1u;
|
||||
const std::size_t frame_bytes = words * 2u;
|
||||
|
||||
if (frame_bytes > available) {
|
||||
if (!finished_) {
|
||||
return Next::End;
|
||||
}
|
||||
fail(JOC_ERR_BITSTREAM_TRUNCATED,
|
||||
"last E-AC-3 syncframe extends past end of input (declared " +
|
||||
std::to_string(frame_bytes) + " bytes, remaining " +
|
||||
std::to_string(available) + ")");
|
||||
return Next::Fail;
|
||||
}
|
||||
|
||||
if (out != nullptr) {
|
||||
out->data = p;
|
||||
out->size = frame_bytes;
|
||||
out->offset = base_offset_ + consumed_;
|
||||
}
|
||||
consumed_ += frame_bytes;
|
||||
stream_offset_ = base_offset_ + consumed_;
|
||||
++frames_emitted_;
|
||||
return Next::Ok;
|
||||
}
|
||||
|
||||
joc_error FrameReader::frame_bytes(const std::uint8_t* data, std::size_t size, std::size_t offset,
|
||||
std::size_t* out_frame_bytes) {
|
||||
if (data == nullptr || out_frame_bytes == nullptr) {
|
||||
return JOC_ERR_INVALID_ARGUMENT;
|
||||
}
|
||||
if (offset + kHeaderBytes > size) {
|
||||
return JOC_ERR_EAC3_SYNCFRAME;
|
||||
}
|
||||
if (static_cast<std::uint16_t>((static_cast<std::uint16_t>(data[offset]) << 8) | data[offset + 1]) !=
|
||||
kSyncword) {
|
||||
return JOC_ERR_EAC3_SYNCFRAME;
|
||||
}
|
||||
const std::size_t words =
|
||||
static_cast<std::size_t>(((data[offset + 2] & 0x07u) << 8) | data[offset + 3]) + 1u;
|
||||
const std::size_t frame_bytes = words * 2u;
|
||||
if (offset + frame_bytes > size) {
|
||||
return JOC_ERR_BITSTREAM_TRUNCATED;
|
||||
}
|
||||
*out_frame_bytes = frame_bytes;
|
||||
return JOC_OK;
|
||||
}
|
||||
|
||||
} // namespace joc::eac3
|
||||
@@ -0,0 +1,68 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc::eac3 {
|
||||
|
||||
struct Frame {
|
||||
const std::uint8_t* data = nullptr;
|
||||
std::size_t size = 0;
|
||||
std::size_t offset = 0; // byte offset of the frame start in the fed stream
|
||||
};
|
||||
|
||||
class FrameReader {
|
||||
public:
|
||||
enum class Next {
|
||||
Ok,
|
||||
End,
|
||||
Fail
|
||||
};
|
||||
|
||||
FrameReader() = default;
|
||||
FrameReader(const std::uint8_t* data, std::size_t size) {
|
||||
push(data, size);
|
||||
finish();
|
||||
}
|
||||
|
||||
// Appends bytes to the internal buffer (used in incremental mode).
|
||||
void push(const std::uint8_t* data, std::size_t size);
|
||||
|
||||
// Declares that no further bytes will arrive; a frame that is still
|
||||
void finish() { finished_ = true; }
|
||||
|
||||
Next next(Frame* out);
|
||||
|
||||
joc_error error() const { return error_; }
|
||||
const std::string& error_message() const { return message_; }
|
||||
|
||||
std::size_t frames_emitted() const { return frames_emitted_; }
|
||||
std::size_t stream_offset() const { return stream_offset_; }
|
||||
|
||||
// report its declared byte length.
|
||||
static joc_error frame_bytes(const std::uint8_t* data, std::size_t size, std::size_t offset,
|
||||
std::size_t* out_frame_bytes);
|
||||
|
||||
static constexpr std::uint16_t kSyncword = 0x0B77;
|
||||
|
||||
private:
|
||||
void compact();
|
||||
void fail(joc_error code, std::string message);
|
||||
|
||||
std::vector<std::uint8_t> buffer_;
|
||||
std::size_t consumed_ = 0; // bytes of buffer_ already turned into frames
|
||||
std::size_t base_offset_ = 0;
|
||||
bool finished_ = false;
|
||||
bool failed_ = false;
|
||||
joc_error error_ = JOC_OK;
|
||||
std::string message_;
|
||||
std::size_t frames_emitted_ = 0;
|
||||
std::size_t stream_offset_ = 0;
|
||||
};
|
||||
|
||||
} // namespace joc::eac3
|
||||
-307
@@ -1,307 +0,0 @@
|
||||
"""从常见 E-AC-3 同步帧直接提取连续 EMDF 容器。
|
||||
|
||||
扫描器检查八种全局位对齐,定位 ``0x5838`` 同步字,验证容器长度并解析各
|
||||
payload config,因此不要求 EMDF 在原始 E-AC-3 文件中按字节对齐。
|
||||
|
||||
本模块有意只覆盖“完整 EMDF 容器在一个同步帧中连续出现”的常见情形。不解析
|
||||
E-AC-3 mantissa,也不重组被音频数据隔开的多个 skip-field 碎片;遇到这种输入会
|
||||
明确报错,让上层决定是否使用兼容桥。
|
||||
"""
|
||||
from dataclasses import dataclass
|
||||
import csv
|
||||
import hashlib
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from variant_error import UnsupportedVariantError
|
||||
|
||||
|
||||
SYNCWORD = 0x5838
|
||||
REQUIRED_JOC_IDS = frozenset((11, 14))
|
||||
|
||||
|
||||
class EmdfError(ValueError):
|
||||
"""EMDF 或其 E-AC-3 传输结构不符合本实现支持的范围。"""
|
||||
|
||||
|
||||
class BitReader:
|
||||
"""MSB-first 位读取器;位置以源数据的绝对 bit offset 表示。"""
|
||||
|
||||
def __init__(self, data, position=0, limit=None):
|
||||
self.data = memoryview(data)
|
||||
self.position = int(position)
|
||||
self.limit = len(self.data) * 8 if limit is None else int(limit)
|
||||
|
||||
def read(self, count):
|
||||
count = int(count)
|
||||
if count < 0 or self.position + count > self.limit:
|
||||
raise EmdfError(f"位流越界 @bit{self.position}, need={count}, limit={self.limit}")
|
||||
value = 0
|
||||
while count:
|
||||
byte_pos = self.position >> 3
|
||||
removed_left = self.position & 7
|
||||
take = min(count, 8 - removed_left)
|
||||
shift = 8 - removed_left - take
|
||||
value = (value << take) | ((self.data[byte_pos] >> shift) & ((1 << take) - 1))
|
||||
self.position += take
|
||||
count -= take
|
||||
return value
|
||||
|
||||
def skip(self, count):
|
||||
self.read(count)
|
||||
|
||||
def read_bytes(self, count):
|
||||
return bytes(self.read(8) for _ in range(count))
|
||||
|
||||
|
||||
def variable_bits(reader, width, max_groups=8):
|
||||
"""读取 EMDF ``variable_bits(width)`` 变长整数。"""
|
||||
value = 0
|
||||
for _ in range(max_groups):
|
||||
value += reader.read(width)
|
||||
more = reader.read(1)
|
||||
if not more:
|
||||
return value
|
||||
value = (value + 1) << width
|
||||
raise EmdfError(f"variable_bits({width}) 延伸组过多")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EmdfContainer:
|
||||
start_bit: int
|
||||
raw: bytes
|
||||
payloads: dict
|
||||
sample_offsets: dict
|
||||
|
||||
|
||||
def _parse_at(data, start_bit):
|
||||
"""在已知 syncword 的 bit offset 解析一个 EMDF 容器。"""
|
||||
reader = BitReader(data, start_bit)
|
||||
if reader.read(16) != SYNCWORD:
|
||||
raise EmdfError(f"EMDF syncword 不匹配 @bit{start_bit}")
|
||||
length = reader.read(16)
|
||||
body_start = reader.position
|
||||
body_end = body_start + length * 8
|
||||
if body_end > reader.limit:
|
||||
raise EmdfError(f"EMDF 容器越界 @bit{start_bit}: length={length}")
|
||||
reader.limit = body_end
|
||||
|
||||
version = reader.read(2)
|
||||
if version == 3:
|
||||
version += variable_bits(reader, 2)
|
||||
key_id = reader.read(3)
|
||||
if key_id == 7:
|
||||
key_id += variable_bits(reader, 3)
|
||||
# TS 103 420 JOC 使用 version=0/key_id=0;严格限制也能排除音频中的伪 marker。
|
||||
if version != 0 or key_id != 0:
|
||||
raise EmdfError(f"不支持的 EMDF version/key_id: {version}/{key_id}")
|
||||
|
||||
payloads = {}
|
||||
sample_offsets = {}
|
||||
terminated = False
|
||||
while reader.position + 5 <= body_end:
|
||||
payload_id = reader.read(5)
|
||||
if payload_id == 0:
|
||||
terminated = True
|
||||
break
|
||||
if payload_id == 0x1F:
|
||||
payload_id += variable_bits(reader, 5)
|
||||
if payload_id in payloads:
|
||||
raise EmdfError(f"同一 EMDF 容器重复 payload id {payload_id}")
|
||||
|
||||
has_sample_offset = bool(reader.read(1))
|
||||
sample_offset = (reader.read(12) >> 1) if has_sample_offset else 0
|
||||
if reader.read(1):
|
||||
variable_bits(reader, 11) # duration
|
||||
if reader.read(1):
|
||||
variable_bits(reader, 2) # group id
|
||||
if reader.read(1):
|
||||
reader.skip(8) # codec data
|
||||
|
||||
if not reader.read(1): # discard_unknown_payload
|
||||
frame_aligned = False
|
||||
if not has_sample_offset:
|
||||
frame_aligned = bool(reader.read(1))
|
||||
if frame_aligned:
|
||||
reader.skip(2)
|
||||
if has_sample_offset or frame_aligned:
|
||||
reader.skip(7)
|
||||
|
||||
payload_size = variable_bits(reader, 8)
|
||||
if reader.position + payload_size * 8 > body_end:
|
||||
raise EmdfError(
|
||||
f"payload id {payload_id} 越界: size={payload_size}, @bit{reader.position}")
|
||||
payloads[payload_id] = reader.read_bytes(payload_size)
|
||||
sample_offsets[payload_id] = sample_offset
|
||||
|
||||
if not terminated:
|
||||
raise EmdfError("EMDF 容器缺少 payload id 0 终止符")
|
||||
total_bytes = 4 + length
|
||||
raw_reader = BitReader(data, start_bit, start_bit + total_bytes * 8)
|
||||
raw = raw_reader.read_bytes(total_bytes)
|
||||
return EmdfContainer(start_bit, raw, payloads, sample_offsets)
|
||||
|
||||
|
||||
def _marker_offsets(data):
|
||||
"""以 NumPy 批量检查八种位移,返回可能的 0x5838 bit offsets。"""
|
||||
source = np.frombuffer(data, dtype=np.uint8)
|
||||
if source.size < 4:
|
||||
return []
|
||||
offsets = []
|
||||
for shift in range(8):
|
||||
if shift == 0:
|
||||
aligned = source
|
||||
else:
|
||||
aligned = np.bitwise_or(
|
||||
np.left_shift(source[:-1].astype(np.uint16), shift) & 0xFF,
|
||||
np.right_shift(source[1:].astype(np.uint16), 8 - shift),
|
||||
).astype(np.uint8)
|
||||
hits = np.flatnonzero((aligned[:-1] == 0x58) & (aligned[1:] == 0x38))
|
||||
offsets.extend(int(hit) * 8 + shift for hit in hits)
|
||||
return sorted(offsets)
|
||||
|
||||
|
||||
def find_joc_emdf(frame):
|
||||
"""返回同步帧中唯一、顶层连续且包含 ID11/ID14 的 JOC EMDF 容器。
|
||||
|
||||
EMDF payload 是不透明字节串,其中可能自然出现另一个 ``0x5838``。若从这个
|
||||
内嵌 marker 开始的后续随机位恰好也能通过容器语法探测,它仍不是一个独立的
|
||||
transport 容器。因此,候选的起点一旦落在较早 JOC 容器的声明范围内,就只把
|
||||
它记作内嵌伪候选,不参与“多个容器”的判定。
|
||||
"""
|
||||
matches = []
|
||||
offsets = _marker_offsets(frame)
|
||||
parsed_candidates = []
|
||||
parse_errors = []
|
||||
for start_bit in offsets:
|
||||
try:
|
||||
container = _parse_at(frame, start_bit)
|
||||
except EmdfError as exc:
|
||||
if len(parse_errors) < 8:
|
||||
parse_errors.append({"start_bit": start_bit, "error": str(exc)})
|
||||
continue
|
||||
parsed_candidates.append({
|
||||
"start_bit": start_bit,
|
||||
"payload_ids": list(container.payloads),
|
||||
"payload_lengths": {str(k): len(v) for k, v in container.payloads.items()},
|
||||
})
|
||||
if REQUIRED_JOC_IDS.issubset(container.payloads):
|
||||
matches.append(container)
|
||||
if not matches:
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "no_contiguous_joc_container",
|
||||
"同步帧中未找到可连续解析且同时包含 ID11/ID14 的 EMDF 容器",
|
||||
details={
|
||||
"syncframe_bytes": len(frame),
|
||||
"marker_bit_offsets": offsets,
|
||||
"parsed_candidates": parsed_candidates,
|
||||
"candidate_parse_errors": parse_errors,
|
||||
"repair_hint": "检查 EMDF 是否跨多个 audio-block skip field 分片,或 payload config 是否变化",
|
||||
})
|
||||
top_level_matches = []
|
||||
nested_matches = []
|
||||
for container in sorted(matches, key=lambda item: item.start_bit):
|
||||
parent = next((candidate for candidate in top_level_matches
|
||||
if candidate.start_bit < container.start_bit <
|
||||
candidate.start_bit + len(candidate.raw) * 8), None)
|
||||
if parent is None:
|
||||
top_level_matches.append(container)
|
||||
else:
|
||||
nested_matches.append({
|
||||
"start_bit": container.start_bit,
|
||||
"end_bit": container.start_bit + len(container.raw) * 8,
|
||||
"parent_start_bit": parent.start_bit,
|
||||
"parent_end_bit": parent.start_bit + len(parent.raw) * 8,
|
||||
})
|
||||
if len(top_level_matches) != 1:
|
||||
starts = [item.start_bit for item in top_level_matches]
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "multiple_joc_containers",
|
||||
"同步帧中存在多个可用 JOC EMDF,当前无法自动选择",
|
||||
details={
|
||||
"syncframe_bytes": len(frame),
|
||||
"joc_container_start_bits": starts,
|
||||
"nested_joc_candidates": nested_matches,
|
||||
})
|
||||
return top_level_matches[0]
|
||||
|
||||
|
||||
def parse_container(data):
|
||||
"""解析从 syncword 开始、已经重新按字节对齐保存的 EMDF 容器。"""
|
||||
container = _parse_at(data, 0)
|
||||
if len(container.raw) != len(data):
|
||||
raise EmdfError(f"EMDF 文件尾有额外数据: parsed={len(container.raw)}, file={len(data)}")
|
||||
return container
|
||||
|
||||
|
||||
def iter_eac3_frames(data):
|
||||
"""按 E-AC-3 ``frmsiz`` 遍历同步帧,拒绝静默重同步。"""
|
||||
pos = 0
|
||||
while pos < len(data):
|
||||
if pos + 4 > len(data) or data[pos:pos + 2] != b"\x0b\x77":
|
||||
raise UnsupportedVariantError(
|
||||
"eac3_transport", "syncframe_header",
|
||||
"E-AC-3 同步帧头无效或出现了未处理的子流排列",
|
||||
details={
|
||||
"byte_offset": pos,
|
||||
"remaining_bytes": len(data) - pos,
|
||||
"next_16_bytes_hex": data[pos:pos + 16].hex(),
|
||||
})
|
||||
size = ((((data[pos + 2] & 7) << 8) | data[pos + 3]) + 1) * 2
|
||||
if pos + size > len(data):
|
||||
raise UnsupportedVariantError(
|
||||
"eac3_transport", "truncated_syncframe",
|
||||
"E-AC-3 末帧长度超过输入剩余数据",
|
||||
details={
|
||||
"byte_offset": pos,
|
||||
"declared_frame_bytes": size,
|
||||
"remaining_bytes": len(data) - pos,
|
||||
})
|
||||
yield data[pos:pos + size]
|
||||
pos += size
|
||||
|
||||
|
||||
def extract_index(eac3_path, output_dir, max_frames=None):
|
||||
"""将裸 E-AC-3 的连续 EMDF 保存为 ``frames.csv + emdf/``。"""
|
||||
output_dir = Path(output_dir)
|
||||
emdf_dir = output_dir / "emdf"
|
||||
emdf_dir.mkdir(parents=True, exist_ok=True)
|
||||
frames = iter_eac3_frames(Path(eac3_path).read_bytes())
|
||||
rows = []
|
||||
for frame_number, frame in enumerate(frames):
|
||||
if max_frames is not None and frame_number >= max_frames:
|
||||
break
|
||||
try:
|
||||
container = find_joc_emdf(frame)
|
||||
except UnsupportedVariantError as exc:
|
||||
exc.add_context(frame=frame_number, details={"syncframe_bytes": len(frame)})
|
||||
raise
|
||||
except EmdfError as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "container_syntax",
|
||||
"EMDF 容器语法无法解析",
|
||||
frame=frame_number,
|
||||
details={"syncframe_bytes": len(frame), "parser_error": str(exc)}) from exc
|
||||
digest = hashlib.sha256(container.raw).hexdigest()
|
||||
target = emdf_dir / f"{digest}.bin"
|
||||
if not target.is_file():
|
||||
target.write_bytes(container.raw)
|
||||
rows.append({
|
||||
"frame": frame_number,
|
||||
"emdf_hash": digest,
|
||||
"emdf_size": len(container.raw),
|
||||
"emdf_start_bit": container.start_bit,
|
||||
"payload_ids": ";".join(str(x) for x in container.payloads),
|
||||
"error": "",
|
||||
})
|
||||
if (frame_number + 1) % 1000 == 0:
|
||||
print(f"[metadata] {frame_number + 1} frames", flush=True)
|
||||
if not rows:
|
||||
raise EmdfError("E-AC-3 输入中没有可处理的同步帧")
|
||||
with (output_dir / "frames.csv").open("w", encoding="utf-8", newline="") as fp:
|
||||
fields = ("frame", "emdf_hash", "emdf_size", "emdf_start_bit", "payload_ids", "error")
|
||||
writer = csv.DictWriter(fp, fieldnames=fields)
|
||||
writer.writeheader()
|
||||
writer.writerows(rows)
|
||||
return output_dir
|
||||
@@ -0,0 +1,305 @@
|
||||
#include "emdf/emdf_parser.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <string>
|
||||
|
||||
#include "foundation/bit_reader.h"
|
||||
|
||||
namespace joc::emdf {
|
||||
|
||||
namespace {
|
||||
|
||||
Status syntax_fail(const std::string& message) {
|
||||
return Status::fail(JOC_ERR_EMDF_SYNTAX, stage::kEmdf, message);
|
||||
}
|
||||
|
||||
Status truncated_fail(const bits::BitReader& reader) {
|
||||
return Status::fail(JOC_ERR_BITSTREAM_TRUNCATED, stage::kEmdf,
|
||||
std::string("EMDF bitstream truncated: ") + reader.error_message());
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Status parse_at(const std::uint8_t* data, std::size_t size, std::size_t start_bit, Container* out) {
|
||||
if (data == nullptr || out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kEmdf, "null buffer or output");
|
||||
}
|
||||
if (start_bit + 16u > size * 8u) {
|
||||
return syntax_fail("EMDF syncword position beyond buffer");
|
||||
}
|
||||
|
||||
bits::BitReader reader;
|
||||
reader.reset(data, size, start_bit);
|
||||
|
||||
if (reader.read(16) != kSyncword) {
|
||||
return syntax_fail("EMDF syncword mismatch at bit " + std::to_string(start_bit));
|
||||
}
|
||||
const std::uint32_t length = reader.read(16);
|
||||
const std::size_t body_start = reader.position();
|
||||
const std::size_t body_end = body_start + static_cast<std::size_t>(length) * 8u;
|
||||
if (body_end > reader.limit()) {
|
||||
return syntax_fail("EMDF container length " + std::to_string(length) +
|
||||
" exceeds buffer at bit " + std::to_string(start_bit));
|
||||
}
|
||||
reader.set_limit_bits(body_end);
|
||||
|
||||
std::uint32_t version = reader.read(2);
|
||||
if (version == 3u) {
|
||||
std::uint32_t extra = 0;
|
||||
if (!bits::variable_bits(reader, 2, 8, &extra)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
version += extra;
|
||||
}
|
||||
std::uint32_t key_id = reader.read(3);
|
||||
if (key_id == 7u) {
|
||||
std::uint32_t extra = 0;
|
||||
if (!bits::variable_bits(reader, 3, 8, &extra)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
key_id += extra;
|
||||
}
|
||||
if (reader.failed()) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
// TS 103 420 JOC uses version 0 / key_id 0; the strict check also rejects
|
||||
// false 0x5838 markers that happen to sit inside audio data.
|
||||
if (version != 0u || key_id != 0u) {
|
||||
return syntax_fail("unsupported EMDF version/key_id " + std::to_string(version) + "/" +
|
||||
std::to_string(key_id));
|
||||
}
|
||||
|
||||
Container container;
|
||||
container.start_bit = start_bit;
|
||||
bool terminated = false;
|
||||
|
||||
while (reader.position() + 5u <= body_end) {
|
||||
std::uint32_t payload_id = reader.read(5);
|
||||
if (reader.failed()) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
if (payload_id == 0u) {
|
||||
terminated = true;
|
||||
break;
|
||||
}
|
||||
if (payload_id == 0x1Fu) {
|
||||
std::uint32_t extra = 0;
|
||||
if (!bits::variable_bits(reader, 5, 8, &extra)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
payload_id += extra;
|
||||
}
|
||||
for (std::size_t i = 0; i < container.payload_count; ++i) {
|
||||
if (container.payloads[i].id == static_cast<std::uint8_t>(payload_id)) {
|
||||
return syntax_fail("duplicate EMDF payload id " + std::to_string(payload_id));
|
||||
}
|
||||
}
|
||||
if (container.payload_count >= kMaxPayloads) {
|
||||
return syntax_fail("EMDF payload count exceeds " + std::to_string(kMaxPayloads));
|
||||
}
|
||||
|
||||
const std::uint32_t has_sample_offset = reader.read(1);
|
||||
std::uint16_t sample_offset = 0;
|
||||
if (has_sample_offset != 0u) {
|
||||
sample_offset = static_cast<std::uint16_t>(reader.read(12) >> 1);
|
||||
}
|
||||
if (reader.read(1) != 0u) {
|
||||
std::uint32_t ignored = 0;
|
||||
if (!bits::variable_bits(reader, 11, 8, &ignored)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
}
|
||||
if (reader.read(1) != 0u) {
|
||||
std::uint32_t ignored = 0;
|
||||
if (!bits::variable_bits(reader, 2, 8, &ignored)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
}
|
||||
if (reader.read(1) != 0u) {
|
||||
if (!reader.skip(8)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
}
|
||||
if (reader.read(1) == 0u) {
|
||||
bool frame_aligned = false;
|
||||
if (has_sample_offset == 0u) {
|
||||
frame_aligned = reader.read(1) != 0u;
|
||||
if (frame_aligned) {
|
||||
if (!reader.skip(2)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (has_sample_offset != 0u || frame_aligned) {
|
||||
if (!reader.skip(7)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (reader.failed()) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
|
||||
std::uint32_t payload_size = 0;
|
||||
if (!bits::variable_bits(reader, 8, 8, &payload_size)) {
|
||||
return reader.error() == JOC_ERR_BITSTREAM_TRUNCATED ? truncated_fail(reader)
|
||||
: syntax_fail(reader.error_message());
|
||||
}
|
||||
const std::size_t payload_bits = static_cast<std::size_t>(payload_size) * 8u;
|
||||
if (reader.position() + payload_bits > body_end) {
|
||||
return syntax_fail("EMDF payload id " + std::to_string(payload_id) +
|
||||
" extends past container body (size " + std::to_string(payload_size) +
|
||||
" at bit " + std::to_string(reader.position()) + ")");
|
||||
}
|
||||
|
||||
Payload& entry = container.payloads[container.payload_count++];
|
||||
entry.id = static_cast<std::uint8_t>(payload_id);
|
||||
entry.sample_offset = sample_offset;
|
||||
entry.bit_offset = reader.position();
|
||||
entry.size = payload_size;
|
||||
|
||||
if (!reader.skip(payload_bits)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
}
|
||||
|
||||
if (!terminated) {
|
||||
return syntax_fail("EMDF container has no payload id 0 terminator");
|
||||
}
|
||||
container.raw_size = 4u + static_cast<std::size_t>(length);
|
||||
*out = container;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
void marker_offsets(const std::uint8_t* data, std::size_t size, std::vector<std::size_t>* out) {
|
||||
out->clear();
|
||||
if (data == nullptr || size < 4u) {
|
||||
return;
|
||||
}
|
||||
// Eight global bit alignments. For shift != 0 the reference builds an
|
||||
// (n-1)-byte shifted view and only scans pairs inside it, which is what the
|
||||
// bounds below reproduce exactly.
|
||||
for (std::size_t shift = 0; shift < 8u; ++shift) {
|
||||
const std::size_t aligned_len = (shift == 0u) ? size : (size - 1u);
|
||||
auto aligned_byte = [&](std::size_t index) -> std::uint8_t {
|
||||
if (shift == 0u) {
|
||||
return data[index];
|
||||
}
|
||||
const std::uint16_t high = static_cast<std::uint16_t>(data[index]) << shift;
|
||||
const std::uint16_t low = static_cast<std::uint16_t>(data[index + 1u]) >> (8u - shift);
|
||||
return static_cast<std::uint8_t>((high | low) & 0xFFu);
|
||||
};
|
||||
if (aligned_len < 2u) {
|
||||
continue;
|
||||
}
|
||||
for (std::size_t i = 0; i + 1u < aligned_len; ++i) {
|
||||
if (aligned_byte(i) == 0x58u && aligned_byte(i + 1u) == 0x38u) {
|
||||
out->push_back(i * 8u + shift);
|
||||
}
|
||||
}
|
||||
}
|
||||
std::sort(out->begin(), out->end());
|
||||
}
|
||||
|
||||
Status find_joc_emdf(const std::uint8_t* data, std::size_t size, Container* out) {
|
||||
if (data == nullptr || out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kEmdf, "null buffer or output");
|
||||
}
|
||||
std::vector<std::size_t> offsets;
|
||||
marker_offsets(data, size, &offsets);
|
||||
|
||||
std::vector<Container> matches;
|
||||
std::size_t parse_errors = 0;
|
||||
std::string first_parse_error;
|
||||
for (const std::size_t start_bit : offsets) {
|
||||
Container candidate;
|
||||
const Status status = parse_at(data, size, start_bit, &candidate);
|
||||
if (!status.ok()) {
|
||||
++parse_errors;
|
||||
if (first_parse_error.empty()) {
|
||||
first_parse_error = "@bit" + std::to_string(start_bit) + ": " + status.message();
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (candidate.find(kIdOamd) != nullptr && candidate.find(kIdJoc) != nullptr) {
|
||||
matches.push_back(candidate);
|
||||
}
|
||||
}
|
||||
|
||||
if (matches.empty()) {
|
||||
// Classification stays at the transport level (identical to the reference
|
||||
// implementation, which raises emdf_transport here), but the underlying
|
||||
std::string message =
|
||||
"no contiguous EMDF container carrying ID11+ID14 in this syncframe (markers=" +
|
||||
std::to_string(offsets.size()) + ", parse_failures=" + std::to_string(parse_errors) +
|
||||
")";
|
||||
if (!first_parse_error.empty()) {
|
||||
message += "; first candidate error " + first_parse_error;
|
||||
}
|
||||
return Status::fail(JOC_ERR_EMDF_TRANSPORT, stage::kEmdf, message);
|
||||
}
|
||||
|
||||
std::sort(matches.begin(), matches.end(),
|
||||
[](const Container& a, const Container& b) { return a.start_bit < b.start_bit; });
|
||||
|
||||
// A payload may contain bytes that look like another 0x5838 container; a
|
||||
std::vector<Container> top_level;
|
||||
for (const Container& candidate : matches) {
|
||||
bool nested = false;
|
||||
for (const Container& parent : top_level) {
|
||||
if (parent.start_bit < candidate.start_bit &&
|
||||
candidate.start_bit < parent.start_bit + parent.raw_size * 8u) {
|
||||
nested = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!nested) {
|
||||
top_level.push_back(candidate);
|
||||
}
|
||||
}
|
||||
|
||||
if (top_level.size() != 1u) {
|
||||
return Status::fail(JOC_ERR_EMDF_TRANSPORT, stage::kEmdf,
|
||||
"multiple top-level JOC EMDF containers (" +
|
||||
std::to_string(top_level.size()) +
|
||||
"); automatic selection is not defined");
|
||||
}
|
||||
*out = top_level.front();
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
void extract_container_bytes(const std::uint8_t* data, std::size_t size, const Container& container,
|
||||
std::vector<std::uint8_t>* out) {
|
||||
out->assign(container.raw_size, 0u);
|
||||
if (out->empty()) {
|
||||
return;
|
||||
}
|
||||
bits::BitReader reader;
|
||||
reader.reset(data, size, container.start_bit);
|
||||
reader.read_bytes(out->data(), out->size());
|
||||
}
|
||||
|
||||
Status extract_payload_bytes(const std::uint8_t* data, std::size_t size, const Payload& payload,
|
||||
std::vector<std::uint8_t>* out) {
|
||||
if (data == nullptr || out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kEmdf, "null buffer or output");
|
||||
}
|
||||
out->assign(payload.size, 0u);
|
||||
if (out->empty()) {
|
||||
return Status::success();
|
||||
}
|
||||
bits::BitReader reader;
|
||||
reader.reset(data, size, payload.bit_offset);
|
||||
if (!reader.read_bytes(out->data(), out->size())) {
|
||||
return Status::fail(JOC_ERR_BITSTREAM_TRUNCATED, stage::kEmdf,
|
||||
"payload bytes extend past the syncframe");
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::emdf
|
||||
@@ -0,0 +1,57 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
#include "joc_core.h"
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::emdf {
|
||||
|
||||
inline constexpr std::uint16_t kSyncword = 0x5838;
|
||||
inline constexpr std::uint8_t kIdOamd = 11;
|
||||
inline constexpr std::uint8_t kIdJoc = 14;
|
||||
inline constexpr std::size_t kMaxPayloads = JOC_MAX_EMDF_PAYLOADS;
|
||||
|
||||
struct Payload {
|
||||
std::uint8_t id = 0;
|
||||
std::uint16_t sample_offset = 0;
|
||||
std::size_t bit_offset = 0; // MSB-first bit position of the payload bytes
|
||||
std::size_t size = 0; // payload byte count
|
||||
};
|
||||
|
||||
struct Container {
|
||||
std::size_t start_bit = 0;
|
||||
std::size_t raw_size = 0;
|
||||
std::size_t payload_count = 0;
|
||||
Payload payloads[kMaxPayloads] = {};
|
||||
|
||||
const Payload* find(std::uint8_t id) const {
|
||||
for (std::size_t i = 0; i < payload_count; ++i) {
|
||||
if (payloads[i].id == id) {
|
||||
return &payloads[i];
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
};
|
||||
|
||||
Status parse_at(const std::uint8_t* data, std::size_t size, std::size_t start_bit, Container* out);
|
||||
|
||||
// All candidate 0x5838 bit offsets over the eight alignments, ascending.
|
||||
void marker_offsets(const std::uint8_t* data, std::size_t size, std::vector<std::size_t>* out);
|
||||
|
||||
Status find_joc_emdf(const std::uint8_t* data, std::size_t size, Container* out);
|
||||
|
||||
// Extract the container's bytes exactly as the bit reader sees them (identical
|
||||
// to a memcpy for byte-aligned containers).
|
||||
void extract_container_bytes(const std::uint8_t* data, std::size_t size, const Container& container,
|
||||
std::vector<std::uint8_t>* out);
|
||||
|
||||
// Extract one payload's bytes with the same MSB-first semantics.
|
||||
Status extract_payload_bytes(const std::uint8_t* data, std::size_t size, const Payload& payload,
|
||||
std::vector<std::uint8_t>* out);
|
||||
|
||||
} // namespace joc::emdf
|
||||
@@ -1,76 +0,0 @@
|
||||
"""解包 EVO MD-set evolution 载荷,返回各 payload ID 的字节数据和位偏移。"""
|
||||
_MARK = "1001001000000"
|
||||
|
||||
# 各子载荷字段: (id, 头部前缀, 同步标记, 后缀常量, 尺寸域位数, 是否有转义)
|
||||
# 头部前缀 = 5 位 id 的 MSB 二进制(id11 字段前另有 5 位容器前导 00000)
|
||||
# 尺寸域单位 = nibble(4 位)。id14 有转义:9 位值=0 → 再读 9 位 = 字节数。
|
||||
_LAYOUT = [
|
||||
(11, "0000001011", "010000000000000", 8, False),
|
||||
(14, "01110", "01000000000000", 9, True),
|
||||
(2, "00010", "000100", 7, False),
|
||||
(1, "00001", "1110000000000000000000000000", 4, False),
|
||||
(30, "11110", "1110000000000000000000000000", 4, False),
|
||||
]
|
||||
|
||||
|
||||
def _msb_bits(data: bytes):
|
||||
return [(x >> (7 - i)) & 1 for x in data for i in range(8)]
|
||||
|
||||
|
||||
def _val(bits, off, n):
|
||||
v = 0
|
||||
for b in bits[off:off + n]:
|
||||
v = (v << 1) | b
|
||||
return v
|
||||
|
||||
|
||||
class _LooseSkip(Exception):
|
||||
def __init__(self, ident):
|
||||
self.ident = ident
|
||||
|
||||
|
||||
def unpack_evolution(payload: bytes, loose=False):
|
||||
"""解包 evolution 载荷 → (subs, offsets)。subs 键为 id 整数。
|
||||
loose=True 时对每个 id 的 (前缀+标记+后缀) 全模式做位流重同步扫描
|
||||
(不同编码流的子载荷次序/内部常量可有合法差异,如 kanata 的 id11)。"""
|
||||
bits = _msb_bits(payload)
|
||||
pos = 0
|
||||
subs = {}
|
||||
offsets = {}
|
||||
for ident, pref, suff, sbits, escape in _LAYOUT:
|
||||
pat = pref + _MARK + suff
|
||||
if loose:
|
||||
hit = -1
|
||||
for i in range(pos, len(bits) - len(pat)):
|
||||
if ''.join(map(str, bits[i:i + len(pat)])) == pat:
|
||||
hit = i
|
||||
break
|
||||
if hit < 0:
|
||||
continue
|
||||
pos = hit + len(pat)
|
||||
else:
|
||||
for name, const, expect in (("前缀", bits[pos:pos + len(pref)], pref),
|
||||
("标记", bits[pos + len(pref):pos + len(pref) + len(_MARK)], _MARK),
|
||||
("后缀", bits[pos + len(pref) + len(_MARK):
|
||||
pos + len(pref) + len(_MARK) + len(suff)], suff)):
|
||||
got = ''.join(map(str, const))
|
||||
if got != expect:
|
||||
raise ValueError(f"id={ident}: {name}常量不匹配 @bit{pos} got={got} want={expect}")
|
||||
pos += len(pref) + len(_MARK) + len(suff)
|
||||
n_nib = _val(bits, pos, sbits)
|
||||
pos += sbits
|
||||
if escape and n_nib == 1:
|
||||
n_nib = 512 + _val(bits, pos, sbits)
|
||||
pos += sbits
|
||||
n_bits = n_nib * 4
|
||||
body = bits[pos:pos + n_bits]
|
||||
pos += n_bits
|
||||
b = bytearray(len(body) // 8)
|
||||
for i in range(0, len(body) // 8 * 8, 8):
|
||||
v = 0
|
||||
for x in body[i:i + 8]:
|
||||
v = (v << 1) | x
|
||||
b[i // 8] = v
|
||||
subs[ident] = bytes(b)
|
||||
offsets[ident] = pos
|
||||
return subs, offsets
|
||||
@@ -0,0 +1,30 @@
|
||||
#include "foundation/bit_reader.h"
|
||||
|
||||
namespace joc::bits {
|
||||
|
||||
bool variable_bits(BitReader& reader, unsigned width, unsigned max_groups, std::uint32_t* out_value) {
|
||||
std::uint32_t value = 0;
|
||||
for (unsigned group = 0; group < max_groups; ++group) {
|
||||
value += reader.read(width);
|
||||
if (reader.failed()) {
|
||||
return false;
|
||||
}
|
||||
const std::uint32_t more = reader.read(1);
|
||||
if (reader.failed()) {
|
||||
return false;
|
||||
}
|
||||
if (more == 0u) {
|
||||
if (out_value != nullptr) {
|
||||
*out_value = value;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
value = (value + 1u) << width;
|
||||
}
|
||||
// Same failure mode as the reference implementation: an extension chain
|
||||
// that never terminates is a syntax error, not a truncation.
|
||||
reader.fail(JOC_ERR_EMDF_SYNTAX, "variable_bits extension groups exceeded");
|
||||
return false;
|
||||
}
|
||||
|
||||
} // namespace joc::bits
|
||||
@@ -0,0 +1,125 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc::bits {
|
||||
|
||||
class BitReader {
|
||||
public:
|
||||
BitReader() = default;
|
||||
BitReader(const std::uint8_t* data, std::size_t size) { reset(data, size); }
|
||||
|
||||
void reset(const std::uint8_t* data, std::size_t size, std::size_t start_bit = 0) {
|
||||
data_ = data;
|
||||
size_bits_ = size * 8u;
|
||||
pos_ = start_bit;
|
||||
limit_ = size_bits_;
|
||||
error_ = JOC_OK;
|
||||
message_ = "";
|
||||
}
|
||||
|
||||
void set_limit_bits(std::size_t limit_bits) {
|
||||
limit_ = limit_bits < size_bits_ ? limit_bits : size_bits_;
|
||||
}
|
||||
|
||||
std::size_t position() const { return pos_; }
|
||||
std::size_t limit() const { return limit_; }
|
||||
std::size_t remaining_bits() const { return pos_ <= limit_ ? limit_ - pos_ : 0; }
|
||||
const std::uint8_t* data() const { return data_; }
|
||||
|
||||
bool failed() const { return error_ != JOC_OK; }
|
||||
joc_error error() const { return error_; }
|
||||
const char* error_message() const { return message_; }
|
||||
|
||||
std::uint32_t read(unsigned count) {
|
||||
if (count == 0) {
|
||||
return 0;
|
||||
}
|
||||
if (!can_read(count)) {
|
||||
fail_truncated(count);
|
||||
return 0;
|
||||
}
|
||||
std::uint32_t value = 0;
|
||||
if ((pos_ & 7u) == 0u && count >= 8u) {
|
||||
while (count >= 8u) {
|
||||
value = (value << 8) | data_[pos_ >> 3];
|
||||
pos_ += 8u;
|
||||
count -= 8u;
|
||||
}
|
||||
}
|
||||
while (count-- > 0u) {
|
||||
const std::uint32_t bit = (data_[pos_ >> 3] >> (7u - (pos_ & 7u))) & 1u;
|
||||
value = (value << 1) | bit;
|
||||
++pos_;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
std::uint64_t read64(unsigned count) {
|
||||
if (count <= 32u) {
|
||||
return static_cast<std::uint64_t>(read(count));
|
||||
}
|
||||
const std::uint64_t high = static_cast<std::uint64_t>(read(count - 32u));
|
||||
const std::uint64_t low = static_cast<std::uint64_t>(read(32u));
|
||||
return (high << 32) | low;
|
||||
}
|
||||
|
||||
bool skip(std::size_t count) {
|
||||
if (!can_read(count)) {
|
||||
fail_truncated(count);
|
||||
return false;
|
||||
}
|
||||
pos_ += count;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool read_bytes(std::uint8_t* out, std::size_t count) {
|
||||
if (count == 0) {
|
||||
return true;
|
||||
}
|
||||
if (!can_read(count * 8u)) {
|
||||
fail_truncated(count * 8u);
|
||||
return false;
|
||||
}
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
out[i] = static_cast<std::uint8_t>(read(8u));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool can_read(std::size_t count) const {
|
||||
return !failed() && count <= limit_ && pos_ <= limit_ - count;
|
||||
}
|
||||
|
||||
// semantic check fails, so the reader never continues past it).
|
||||
void fail(joc_error code, const char* message) {
|
||||
if (!failed()) {
|
||||
error_ = code;
|
||||
message_ = message;
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
void fail_truncated(std::size_t count) {
|
||||
fail(JOC_ERR_BITSTREAM_TRUNCATED, "bit read past end of buffer");
|
||||
last_request_ = count;
|
||||
}
|
||||
|
||||
const std::uint8_t* data_ = nullptr;
|
||||
std::size_t size_bits_ = 0;
|
||||
std::size_t pos_ = 0;
|
||||
std::size_t limit_ = 0;
|
||||
std::size_t last_request_ = 0;
|
||||
joc_error error_ = JOC_OK;
|
||||
const char* message_ = "";
|
||||
};
|
||||
|
||||
// followed by a continuation bit. Mirrors src/emdf.py:variable_bits().
|
||||
bool variable_bits(BitReader& reader, unsigned width, unsigned max_groups,
|
||||
std::uint32_t* out_value);
|
||||
|
||||
} // namespace joc::bits
|
||||
@@ -0,0 +1,206 @@
|
||||
#include "foundation/fft.h"
|
||||
|
||||
#include <cmath>
|
||||
|
||||
#include "simd/simd.h"
|
||||
|
||||
namespace joc::dsp {
|
||||
|
||||
// The dispatched kernels read and write the spectrum as interleaved doubles, and
|
||||
// an array of std::complex<double> is exactly that: two doubles per element, no
|
||||
// padding, no vtable.
|
||||
static_assert(sizeof(Complex) == 2u * sizeof(double), "complex layout");
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr double kPi = 3.14159265358979323846;
|
||||
|
||||
template <typename Container>
|
||||
void fft_in_place(Container* data, bool inverse) {
|
||||
const std::size_t count = data->size();
|
||||
if (count < 2u) {
|
||||
return;
|
||||
}
|
||||
for (std::size_t index = 1u, reversed = 0u; index < count; ++index) {
|
||||
std::size_t bit = count >> 1u;
|
||||
for (; (reversed & bit) != 0u; bit >>= 1u) {
|
||||
reversed ^= bit;
|
||||
}
|
||||
reversed ^= bit;
|
||||
if (index < reversed) {
|
||||
std::swap((*data)[index], (*data)[reversed]);
|
||||
}
|
||||
}
|
||||
for (std::size_t length = 2u; length <= count; length <<= 1u) {
|
||||
const double angle = (inverse ? 2.0 : -2.0) * kPi / static_cast<double>(length);
|
||||
const Complex step(std::cos(angle), std::sin(angle));
|
||||
for (std::size_t start = 0u; start < count; start += length) {
|
||||
Complex factor(1.0, 0.0);
|
||||
for (std::size_t offset = 0u; offset < length / 2u; ++offset) {
|
||||
const Complex even = (*data)[start + offset];
|
||||
const Complex odd = (*data)[start + offset + length / 2u] * factor;
|
||||
(*data)[start + offset] = even + odd;
|
||||
(*data)[start + offset + length / 2u] = even - odd;
|
||||
factor *= step;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (inverse) {
|
||||
for (Complex& value : *data) {
|
||||
value /= static_cast<double>(count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool is_power_of_two(std::size_t value) { return value != 0u && (value & (value - 1u)) == 0u; }
|
||||
|
||||
} // namespace
|
||||
|
||||
FftPlan::FftPlan(std::size_t size, bool inverse) : size_(size), inverse_(inverse) {
|
||||
reverse_.resize(size);
|
||||
for (std::size_t index = 1u, reversed = 0u; index < size; ++index) {
|
||||
std::size_t bit = size >> 1u;
|
||||
for (; (reversed & bit) != 0u; bit >>= 1u) {
|
||||
reversed ^= bit;
|
||||
}
|
||||
reversed ^= bit;
|
||||
reverse_[index] = reversed;
|
||||
}
|
||||
for (std::size_t length = 2u; length <= size; length <<= 1u) {
|
||||
const double angle = (inverse ? 2.0 : -2.0) * kPi / static_cast<double>(length);
|
||||
const Complex step(std::cos(angle), std::sin(angle));
|
||||
stage_begin_.push_back(twiddle_.size());
|
||||
Complex factor(1.0, 0.0);
|
||||
for (std::size_t offset = 0u; offset < length / 2u; ++offset) {
|
||||
twiddle_.push_back(factor);
|
||||
factor *= step;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Exactly the operations fft_in_place performs, in the same order, with the
|
||||
// twiddles read from the precomputed recurrence instead of being re-derived.
|
||||
template <typename Container>
|
||||
void FftPlan::apply(Container* data) const {
|
||||
const std::size_t count = data->size();
|
||||
if (count < 2u) {
|
||||
return;
|
||||
}
|
||||
const std::size_t* reverse = reverse_.data();
|
||||
for (std::size_t index = 1u; index < count; ++index) {
|
||||
const std::size_t reversed = reverse[index];
|
||||
if (index < reversed) {
|
||||
std::swap((*data)[index], (*data)[reversed]);
|
||||
}
|
||||
}
|
||||
// The cascade is dispatched for every power-of-two size the kernels can pack
|
||||
// whole groups into a vector (JOC_SIMD pins one tier for verification). A
|
||||
// kernel only ever puts independent butterflies in the same vector, so every
|
||||
// output keeps the operation sequence and the roundings written below; small
|
||||
// transforms -- and the caller's own table -- keep the portable loop.
|
||||
if (count >= simd::kMinVectorFftSize && (count & (count - 1u)) == 0u) {
|
||||
simd::fft_butterflies(reinterpret_cast<double*>(data->data()), count,
|
||||
reinterpret_cast<const double*>(twiddle_.data()),
|
||||
stage_begin_.data());
|
||||
} else {
|
||||
std::size_t stage = 0u;
|
||||
for (std::size_t length = 2u; length <= count; length <<= 1u, ++stage) {
|
||||
const Complex* table = twiddle_.data() + stage_begin_[stage];
|
||||
for (std::size_t start = 0u; start < count; start += length) {
|
||||
for (std::size_t offset = 0u; offset < length / 2u; ++offset) {
|
||||
const Complex even = (*data)[start + offset];
|
||||
const Complex odd = (*data)[start + offset + length / 2u] * table[offset];
|
||||
(*data)[start + offset] = even + odd;
|
||||
(*data)[start + offset + length / 2u] = even - odd;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (inverse_) {
|
||||
for (Complex& value : *data) {
|
||||
value /= static_cast<double>(count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void fft_radix2(std::vector<Complex>* data, const FftPlan& plan) { plan.apply(data); }
|
||||
|
||||
void fft_radix2(std::array<Complex, kQmfFftSize>* data, const FftPlan& plan) { plan.apply(data); }
|
||||
|
||||
void fft_radix2(std::vector<Complex>* data, bool inverse) { fft_in_place(data, inverse); }
|
||||
|
||||
void fft_radix2(std::array<Complex, kQmfFftSize>* data, bool inverse) {
|
||||
fft_in_place(data, inverse);
|
||||
}
|
||||
|
||||
void fft_any(const std::vector<Complex>& input, bool inverse, std::vector<Complex>* output) {
|
||||
const std::size_t count = input.size();
|
||||
if (is_power_of_two(count)) {
|
||||
*output = input;
|
||||
fft_radix2(output, inverse);
|
||||
return;
|
||||
}
|
||||
std::size_t size = 1u;
|
||||
while (size < 2u * count + 1u) {
|
||||
size <<= 1u;
|
||||
}
|
||||
const double sign = inverse ? 1.0 : -1.0;
|
||||
std::vector<Complex> left(size, Complex(0.0, 0.0));
|
||||
std::vector<Complex> right(size, Complex(0.0, 0.0));
|
||||
for (std::size_t index = 0u; index < count; ++index) {
|
||||
const std::size_t wrapped = (index * index) % (2u * count);
|
||||
const double angle = kPi * static_cast<double>(wrapped) / static_cast<double>(count);
|
||||
const Complex chirp(std::cos(angle), sign * std::sin(angle));
|
||||
left[index] = input[index] * chirp;
|
||||
right[index] = std::conj(chirp);
|
||||
if (index != 0u) {
|
||||
right[size - index] = std::conj(chirp);
|
||||
}
|
||||
}
|
||||
fft_radix2(&left, false);
|
||||
fft_radix2(&right, false);
|
||||
for (std::size_t index = 0u; index < size; ++index) {
|
||||
left[index] *= right[index];
|
||||
}
|
||||
fft_radix2(&left, true);
|
||||
output->resize(count);
|
||||
for (std::size_t index = 0u; index < count; ++index) {
|
||||
const std::size_t wrapped = (index * index) % (2u * count);
|
||||
const double angle = kPi * static_cast<double>(wrapped) / static_cast<double>(count);
|
||||
const Complex chirp(std::cos(angle), sign * std::sin(angle));
|
||||
(*output)[index] = left[index] * chirp;
|
||||
if (inverse) {
|
||||
(*output)[index] /= static_cast<double>(count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::size_t next_fast_len(std::size_t value) {
|
||||
if (value <= 6u) {
|
||||
return value;
|
||||
}
|
||||
std::size_t best = value;
|
||||
for (std::size_t power2 = 1u; power2 < value * 2u; power2 *= 2u) {
|
||||
for (std::size_t power3 = power2; power3 < value * 2u; power3 *= 3u) {
|
||||
std::size_t power5 = power3;
|
||||
while (power5 < value) {
|
||||
power5 *= 5u;
|
||||
}
|
||||
best = std::min(best, power5);
|
||||
if (power3 >= value) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
std::size_t next_power_of_two(std::size_t value) {
|
||||
std::size_t result = 1u;
|
||||
while (result < value) {
|
||||
result <<= 1u;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
} // namespace joc::dsp
|
||||
@@ -0,0 +1,68 @@
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <complex>
|
||||
#include <cstddef>
|
||||
#include <vector>
|
||||
|
||||
// Complex transforms shared by the HRTF and Rosella DSP cores. The convention is
|
||||
// NumPy's: the forward transform is unnormalised and the inverse scales by 1/N,
|
||||
// so a ported pipeline keeps the reference's arithmetic bit for bit.
|
||||
namespace joc::dsp {
|
||||
|
||||
using Complex = std::complex<double>;
|
||||
|
||||
inline constexpr std::size_t kQmfFftSize = 128;
|
||||
|
||||
// Precomputed radix-2 plan for one size and direction.
|
||||
//
|
||||
// The transform derives each butterfly's twiddle by multiplying the previous one
|
||||
// by the stage step, so the twiddle at offset k is `step` multiplied k times in
|
||||
// that order, independently of the group. Materialising that exact recurrence --
|
||||
// and the bit-reversal permutation -- removes one complex multiply and a
|
||||
// (length/2)-deep serial dependency from every stage's inner loop. The table
|
||||
// entries are the recurrence's own values, so the transform is bit-identical.
|
||||
//
|
||||
// The 128-point cascade is executed by the runtime-dispatched SIMD kernel
|
||||
// (src/simd/simd.h): it computes independent butterflies in parallel lanes,
|
||||
// which leaves both the table and every output's summation order untouched.
|
||||
class FftPlan {
|
||||
public:
|
||||
FftPlan(std::size_t size, bool inverse);
|
||||
|
||||
std::size_t size() const { return size_; }
|
||||
bool inverse() const { return inverse_; }
|
||||
|
||||
private:
|
||||
template <typename Container>
|
||||
void apply(Container* data) const;
|
||||
|
||||
friend void fft_radix2(std::vector<Complex>* data, const FftPlan& plan);
|
||||
friend void fft_radix2(std::array<Complex, kQmfFftSize>* data, const FftPlan& plan);
|
||||
|
||||
std::size_t size_ = 0;
|
||||
bool inverse_ = false;
|
||||
std::vector<std::size_t> reverse_; // bit-reversal permutation, [size]
|
||||
std::vector<std::size_t> stage_begin_; // twiddle offset of each stage
|
||||
std::vector<Complex> twiddle_; // per stage, length/2 entries, concatenated
|
||||
};
|
||||
|
||||
// In-place radix-2 transform; the size must be a power of two.
|
||||
void fft_radix2(std::vector<Complex>* data, bool inverse);
|
||||
void fft_radix2(std::array<Complex, kQmfFftSize>* data, bool inverse);
|
||||
|
||||
// Plan-driven forms: the plan carries the size and the direction, so a caller that
|
||||
// transforms the same length repeatedly builds it once.
|
||||
void fft_radix2(std::vector<Complex>* data, const FftPlan& plan);
|
||||
void fft_radix2(std::array<Complex, kQmfFftSize>* data, const FftPlan& plan);
|
||||
|
||||
// Exact-length transform: radix-2 when the size allows it, Bluestein otherwise.
|
||||
// scipy/numpy use a mixed-radix transform, which is the same transform.
|
||||
void fft_any(const std::vector<Complex>& input, bool inverse, std::vector<Complex>* output);
|
||||
|
||||
// scipy's next_fast_len: the smallest 5-smooth number that is not smaller.
|
||||
std::size_t next_fast_len(std::size_t value);
|
||||
|
||||
std::size_t next_power_of_two(std::size_t value);
|
||||
|
||||
} // namespace joc::dsp
|
||||
@@ -0,0 +1,164 @@
|
||||
#include "foundation/fs_utf8.h"
|
||||
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <vector>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#define NOMINMAX
|
||||
#include <windows.h>
|
||||
#include <shellapi.h>
|
||||
#include <fcntl.h>
|
||||
#include <io.h>
|
||||
#else
|
||||
#include <cstdlib>
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
namespace joc::fs_utf8 {
|
||||
|
||||
namespace fs = std::filesystem;
|
||||
|
||||
fs::path to_path(const std::string& utf8) {
|
||||
return fs::path(std::u8string(reinterpret_cast<const char8_t*>(utf8.data()), utf8.size()));
|
||||
}
|
||||
|
||||
std::string from_path(const fs::path& path) {
|
||||
const std::u8string text = path.u8string();
|
||||
return std::string(reinterpret_cast<const char*>(text.data()), text.size());
|
||||
}
|
||||
|
||||
std::FILE* fopen(const std::string& utf8_path, const char* mode) {
|
||||
#if defined(_WIN32)
|
||||
const std::wstring wide_mode(mode, mode + std::strlen(mode));
|
||||
return ::_wfopen(to_path(utf8_path).c_str(), wide_mode.c_str());
|
||||
#else
|
||||
return std::fopen(utf8_path.c_str(), mode);
|
||||
#endif
|
||||
}
|
||||
|
||||
std::FILE* fopen_spool(const std::string& utf8_path) {
|
||||
#if defined(_WIN32)
|
||||
// Delete-on-close handed to the CRT: if the process is killed the file goes with
|
||||
// it, which is what stops an aborted run from leaving hundreds of gigabytes.
|
||||
HANDLE handle = ::CreateFileW(
|
||||
to_path(utf8_path).c_str(), GENERIC_READ | GENERIC_WRITE | DELETE,
|
||||
FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, nullptr, CREATE_ALWAYS,
|
||||
FILE_ATTRIBUTE_TEMPORARY | FILE_FLAG_DELETE_ON_CLOSE, nullptr);
|
||||
if (handle == INVALID_HANDLE_VALUE) {
|
||||
return nullptr;
|
||||
}
|
||||
const int descriptor = ::_open_osfhandle(reinterpret_cast<std::intptr_t>(handle), 0);
|
||||
if (descriptor == -1) {
|
||||
::CloseHandle(handle);
|
||||
return nullptr;
|
||||
}
|
||||
return ::_fdopen(descriptor, "wb+");
|
||||
#else
|
||||
return std::fopen(utf8_path.c_str(), "wb+");
|
||||
#endif
|
||||
}
|
||||
|
||||
int remove(const std::string& utf8_path) {
|
||||
#if defined(_WIN32)
|
||||
return ::_wremove(to_path(utf8_path).c_str());
|
||||
#else
|
||||
return std::remove(utf8_path.c_str());
|
||||
#endif
|
||||
}
|
||||
|
||||
bool exists(const std::string& utf8_path) {
|
||||
std::error_code error;
|
||||
return fs::exists(to_path(utf8_path), error);
|
||||
}
|
||||
|
||||
bool is_directory(const std::string& utf8_path) {
|
||||
std::error_code error;
|
||||
return fs::is_directory(to_path(utf8_path), error);
|
||||
}
|
||||
|
||||
std::uintmax_t file_size(const std::string& utf8_path, std::error_code& error) {
|
||||
return fs::file_size(to_path(utf8_path), error);
|
||||
}
|
||||
|
||||
std::string temp_directory() {
|
||||
std::error_code error;
|
||||
const fs::path directory = fs::temp_directory_path(error);
|
||||
return error ? std::string(".") : from_path(directory);
|
||||
}
|
||||
|
||||
std::string executable_path() {
|
||||
#if defined(_WIN32)
|
||||
std::vector<wchar_t> buffer(MAX_PATH);
|
||||
while (true) {
|
||||
const DWORD written =
|
||||
::GetModuleFileNameW(nullptr, buffer.data(), static_cast<DWORD>(buffer.size()));
|
||||
if (written == 0) {
|
||||
return std::string();
|
||||
}
|
||||
if (written < buffer.size()) {
|
||||
return from_path(fs::path(std::wstring(buffer.data(), written)));
|
||||
}
|
||||
buffer.resize(buffer.size() * 2u);
|
||||
}
|
||||
#elif defined(__linux__)
|
||||
std::vector<char> buffer(4096u, '\0');
|
||||
const ssize_t written = ::readlink("/proc/self/exe", buffer.data(), buffer.size() - 1u);
|
||||
return written > 0 ? std::string(buffer.data(), static_cast<std::size_t>(written))
|
||||
: std::string();
|
||||
#else
|
||||
return std::string();
|
||||
#endif
|
||||
}
|
||||
|
||||
std::ifstream open_input(const std::string& utf8_path) {
|
||||
return std::ifstream(to_path(utf8_path), std::ios::binary);
|
||||
}
|
||||
|
||||
std::ofstream open_output(const std::string& utf8_path) {
|
||||
return std::ofstream(to_path(utf8_path), std::ios::binary);
|
||||
}
|
||||
|
||||
std::vector<std::string> command_line_arguments(int argc, char** argv) {
|
||||
#if defined(_WIN32)
|
||||
(void)argc;
|
||||
(void)argv;
|
||||
int count = 0;
|
||||
LPWSTR* wide = ::CommandLineToArgvW(::GetCommandLineW(), &count);
|
||||
std::vector<std::string> arguments;
|
||||
if (wide == nullptr) {
|
||||
return arguments;
|
||||
}
|
||||
arguments.reserve(static_cast<std::size_t>(count));
|
||||
for (int index = 0; index < count; ++index) {
|
||||
const std::wstring_view text(wide[index]);
|
||||
const int size = ::WideCharToMultiByte(CP_UTF8, 0, text.data(),
|
||||
static_cast<int>(text.size()), nullptr, 0, nullptr,
|
||||
nullptr);
|
||||
std::string utf8(static_cast<std::size_t>(size), '\0');
|
||||
if (size > 0) {
|
||||
::WideCharToMultiByte(CP_UTF8, 0, text.data(), static_cast<int>(text.size()),
|
||||
utf8.data(), size, nullptr, nullptr);
|
||||
}
|
||||
arguments.push_back(std::move(utf8));
|
||||
}
|
||||
::LocalFree(wide);
|
||||
return arguments;
|
||||
#else
|
||||
std::vector<std::string> arguments;
|
||||
arguments.reserve(static_cast<std::size_t>(argc));
|
||||
for (int index = 0; index < argc; ++index) {
|
||||
arguments.emplace_back(argv[index]);
|
||||
}
|
||||
return arguments;
|
||||
#endif
|
||||
}
|
||||
|
||||
void configure_console() {
|
||||
#if defined(_WIN32)
|
||||
::SetConsoleOutputCP(CP_UTF8);
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace joc::fs_utf8
|
||||
@@ -0,0 +1,47 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <string>
|
||||
#include <system_error>
|
||||
#include <vector>
|
||||
|
||||
// Paths inside this library are always UTF-8, on every platform. std::filesystem
|
||||
// stores UTF-16 on Windows and bytes elsewhere, and the narrow CRT uses the ANSI
|
||||
// code page on Windows, so every path crosses into the OS through this shim: that
|
||||
// is what makes non-ASCII names (Japanese, Chinese, ...) work.
|
||||
namespace joc::fs_utf8 {
|
||||
|
||||
std::filesystem::path to_path(const std::string& utf8);
|
||||
std::string from_path(const std::filesystem::path& path);
|
||||
|
||||
// File handles and queries take a UTF-8 path: _wfopen on Windows, plain calls
|
||||
// elsewhere. Nothing else in the library may call the narrow CRT with a path.
|
||||
std::FILE* fopen(const std::string& utf8_path, const char* mode);
|
||||
// Temporary spool handle: on Windows the file is opened delete-on-close, so killing
|
||||
// the process removes it instead of leaving a multi-gigabyte leftover behind.
|
||||
std::FILE* fopen_spool(const std::string& utf8_path);
|
||||
int remove(const std::string& utf8_path);
|
||||
bool exists(const std::string& utf8_path);
|
||||
bool is_directory(const std::string& utf8_path);
|
||||
std::uintmax_t file_size(const std::string& utf8_path, std::error_code& error);
|
||||
std::string temp_directory();
|
||||
|
||||
// The running executable's own path, UTF-8, or empty when the platform cannot
|
||||
// report it. Defaults are anchored here so they never depend on the CWD.
|
||||
std::string executable_path();
|
||||
|
||||
// Streams: std::ifstream/ofstream accept a std::filesystem::path, which is the
|
||||
// portable way to open a UTF-8 path.
|
||||
std::ifstream open_input(const std::string& utf8_path);
|
||||
std::ofstream open_output(const std::string& utf8_path);
|
||||
|
||||
// Command line arguments as UTF-8. Windows hands the process UTF-16 and the
|
||||
// narrow CRT would convert it through the ANSI code page, so the wide command
|
||||
// line is re-parsed there; on POSIX argv is already bytes in the user's locale.
|
||||
std::vector<std::string> command_line_arguments(int argc, char** argv);
|
||||
void configure_console();
|
||||
|
||||
} // namespace joc::fs_utf8
|
||||
@@ -0,0 +1,25 @@
|
||||
// Port of the reference's adm_atmos.q_to_adm_xyz: OAMD Q15 coordinates to the ADM
|
||||
// cartesian triple. It lives in foundation because both the ADM writer and the
|
||||
// object position timeline need it, and the timeline must not depend on output.
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
#include "foundation/py_num.h"
|
||||
|
||||
namespace joc::geometry {
|
||||
|
||||
inline void q_to_adm_xyz(int q1, int q2, int q3, double* x, double* y, double* z) {
|
||||
const double posX = std::min(1.0, static_cast<double>(pynum::py_round(
|
||||
static_cast<double>(q1) * 62.0 / 32767.0)) / 62.0);
|
||||
const double posY = std::min(1.0, static_cast<double>(pynum::py_round(
|
||||
static_cast<double>(q2) * 62.0 / 32767.0)) / 62.0);
|
||||
double posZ = static_cast<double>(pynum::py_round(
|
||||
static_cast<double>(q3) * 15.0 / 32767.0)) / 15.0;
|
||||
posZ = std::max(-1.0, std::min(1.0, posZ));
|
||||
*x = posX * 2.0 - 1.0;
|
||||
*y = 1.0 - posY * 2.0;
|
||||
*z = posZ;
|
||||
}
|
||||
|
||||
} // namespace joc::geometry
|
||||
@@ -0,0 +1,227 @@
|
||||
#include "foundation/mini_json.h"
|
||||
|
||||
#include <cmath>
|
||||
#include <cstdlib>
|
||||
|
||||
namespace joc::json {
|
||||
|
||||
namespace {
|
||||
|
||||
void skip_space(const std::string& text, std::size_t* index) {
|
||||
while (*index < text.size() &&
|
||||
(text[*index] == ' ' || text[*index] == '\t' || text[*index] == '\n' ||
|
||||
text[*index] == '\r')) {
|
||||
++(*index);
|
||||
}
|
||||
}
|
||||
|
||||
bool read_string(const std::string& text, std::size_t* index, std::string* out) {
|
||||
if (*index >= text.size() || text[*index] != '"') {
|
||||
return false;
|
||||
}
|
||||
++(*index);
|
||||
out->clear();
|
||||
while (*index < text.size()) {
|
||||
const char c = text[*index];
|
||||
if (c == '\\') {
|
||||
if (*index + 1 >= text.size()) {
|
||||
return false;
|
||||
}
|
||||
const char escape = text[*index + 1];
|
||||
*index += 2;
|
||||
switch (escape) {
|
||||
case '"': out->push_back('"'); break;
|
||||
case '\\': out->push_back('\\'); break;
|
||||
case '/': out->push_back('/'); break;
|
||||
case 'b': out->push_back('\b'); break;
|
||||
case 'f': out->push_back('\f'); break;
|
||||
case 'n': out->push_back('\n'); break;
|
||||
case 'r': out->push_back('\r'); break;
|
||||
case 't': out->push_back('\t'); break;
|
||||
case 'u': {
|
||||
if (*index + 4 > text.size()) {
|
||||
return false;
|
||||
}
|
||||
unsigned code = 0;
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
const char digit = text[*index + static_cast<std::size_t>(i)];
|
||||
code <<= 4;
|
||||
if (digit >= '0' && digit <= '9') { code |= static_cast<unsigned>(digit - '0'); }
|
||||
else if (digit >= 'a' && digit <= 'f') { code |= static_cast<unsigned>(digit - 'a' + 10); }
|
||||
else if (digit >= 'A' && digit <= 'F') { code |= static_cast<unsigned>(digit - 'A' + 10); }
|
||||
else { return false; }
|
||||
}
|
||||
*index += 4;
|
||||
if (code < 0x80u) {
|
||||
out->push_back(static_cast<char>(code));
|
||||
} else if (code < 0x800u) {
|
||||
out->push_back(static_cast<char>(0xC0u | (code >> 6)));
|
||||
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
|
||||
} else {
|
||||
out->push_back(static_cast<char>(0xE0u | (code >> 12)));
|
||||
out->push_back(static_cast<char>(0x80u | ((code >> 6) & 0x3Fu)));
|
||||
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (c == '"') {
|
||||
++(*index);
|
||||
return true;
|
||||
}
|
||||
out->push_back(c);
|
||||
++(*index);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool read_compound(const std::string& text, std::size_t* index, std::string* out) {
|
||||
const char open = text[*index];
|
||||
const char close = open == '{' ? '}' : ']';
|
||||
int depth = 0;
|
||||
const std::size_t start = *index;
|
||||
while (*index < text.size()) {
|
||||
const char c = text[*index];
|
||||
if (c == '"') {
|
||||
std::string ignored;
|
||||
if (!read_string(text, index, &ignored)) {
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (c == open) {
|
||||
++depth;
|
||||
} else if (c == close) {
|
||||
--depth;
|
||||
if (depth == 0) {
|
||||
++(*index);
|
||||
*out = text.substr(start, *index - start);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
++(*index);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool parse_object(const std::string& text, std::vector<Member>* out, std::string* error) {
|
||||
out->clear();
|
||||
std::size_t index = 0;
|
||||
skip_space(text, &index);
|
||||
if (index >= text.size() || text[index] != '{') {
|
||||
if (error != nullptr) { *error = "metadata is not a JSON object"; }
|
||||
return false;
|
||||
}
|
||||
++index;
|
||||
for (;;) {
|
||||
skip_space(text, &index);
|
||||
if (index < text.size() && text[index] == '}') {
|
||||
++index;
|
||||
break;
|
||||
}
|
||||
if (index >= text.size() || text[index] == ',') {
|
||||
if (index >= text.size()) {
|
||||
if (error != nullptr) { *error = "unterminated JSON object"; }
|
||||
return false;
|
||||
}
|
||||
++index;
|
||||
continue;
|
||||
}
|
||||
Member member;
|
||||
if (!read_string(text, &index, &member.key)) {
|
||||
if (error != nullptr) { *error = "expected a JSON key"; }
|
||||
return false;
|
||||
}
|
||||
skip_space(text, &index);
|
||||
if (index >= text.size() || text[index] != ':') {
|
||||
if (error != nullptr) { *error = "expected ':' after JSON key " + member.key; }
|
||||
return false;
|
||||
}
|
||||
++index;
|
||||
skip_space(text, &index);
|
||||
if (index >= text.size()) {
|
||||
if (error != nullptr) { *error = "missing JSON value for " + member.key; }
|
||||
return false;
|
||||
}
|
||||
if (text[index] == '"') {
|
||||
member.is_string = true;
|
||||
if (!read_string(text, &index, &member.raw)) {
|
||||
if (error != nullptr) { *error = "bad JSON string for " + member.key; }
|
||||
return false;
|
||||
}
|
||||
} else if (text[index] == '{' || text[index] == '[') {
|
||||
if (!read_compound(text, &index, &member.raw)) {
|
||||
if (error != nullptr) { *error = "bad JSON container for " + member.key; }
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
const std::size_t start = index;
|
||||
while (index < text.size() && text[index] != ',' && text[index] != '}') {
|
||||
++index;
|
||||
}
|
||||
member.raw = text.substr(start, index - start);
|
||||
while (!member.raw.empty() &&
|
||||
(member.raw.back() == ' ' || member.raw.back() == '\n' ||
|
||||
member.raw.back() == '\r' || member.raw.back() == '\t')) {
|
||||
member.raw.pop_back();
|
||||
}
|
||||
}
|
||||
for (const Member& existing : *out) {
|
||||
if (existing.key == member.key) {
|
||||
if (error != nullptr) { *error = "duplicate JSON key " + member.key; }
|
||||
return false;
|
||||
}
|
||||
}
|
||||
out->push_back(std::move(member));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
const Member* find(const std::vector<Member>& members, const std::string& key) {
|
||||
for (const Member& member : members) {
|
||||
if (member.key == key) {
|
||||
return &member;
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
bool as_string(const Member& member, std::string* out) {
|
||||
if (!member.is_string || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
*out = member.raw;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool as_number(const Member& member, double* out) {
|
||||
if (member.is_string || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
char* end = nullptr;
|
||||
const double value = std::strtod(member.raw.c_str(), &end);
|
||||
if (end == member.raw.c_str() || !std::isfinite(value)) {
|
||||
return false;
|
||||
}
|
||||
*out = value;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool as_integer(const Member& member, long long* out) {
|
||||
double value = 0.0;
|
||||
if (!as_number(member, &value) || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
if (value != std::floor(value)) {
|
||||
return false;
|
||||
}
|
||||
*out = static_cast<long long>(value);
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace joc::json
|
||||
@@ -0,0 +1,25 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::json {
|
||||
|
||||
struct Member {
|
||||
std::string key;
|
||||
std::string raw;
|
||||
bool is_string = false;
|
||||
};
|
||||
|
||||
// Parses a top-level JSON object. Rejects non-objects and duplicate keys.
|
||||
bool parse_object(const std::string& text, std::vector<Member>* out, std::string* error);
|
||||
|
||||
const Member* find(const std::vector<Member>& members, const std::string& key);
|
||||
|
||||
bool as_string(const Member& member, std::string* out);
|
||||
bool as_number(const Member& member, double* out);
|
||||
bool as_integer(const Member& member, long long* out);
|
||||
|
||||
} // namespace joc::json
|
||||
@@ -0,0 +1,25 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cfenv>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <string>
|
||||
|
||||
namespace joc::pynum {
|
||||
|
||||
inline long long py_round(double value) {
|
||||
return static_cast<long long>(std::nearbyint(value));
|
||||
}
|
||||
|
||||
inline std::string format_fixed(double value, int decimals) {
|
||||
char buffer[64];
|
||||
std::snprintf(buffer, sizeof(buffer), "%.*f", decimals, value);
|
||||
return std::string(buffer);
|
||||
}
|
||||
|
||||
inline long long trunc_to_ll(double value) {
|
||||
return static_cast<long long>(value);
|
||||
}
|
||||
|
||||
} // namespace joc::pynum
|
||||
@@ -0,0 +1,165 @@
|
||||
#include "foundation/sha256.h"
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace joc::crypto {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr std::uint32_t kK[64] = {
|
||||
0x428a2f98u, 0x71374491u, 0xb5c0fbcfu, 0xe9b5dba5u, 0x3956c25bu, 0x59f111f1u, 0x923f82a4u,
|
||||
0xab1c5ed5u, 0xd807aa98u, 0x12835b01u, 0x243185beu, 0x550c7dc3u, 0x72be5d74u, 0x80deb1feu,
|
||||
0x9bdc06a7u, 0xc19bf174u, 0xe49b69c1u, 0xefbe4786u, 0x0fc19dc6u, 0x240ca1ccu, 0x2de92c6fu,
|
||||
0x4a7484aau, 0x5cb0a9dcu, 0x76f988dau, 0x983e5152u, 0xa831c66du, 0xb00327c8u, 0xbf597fc7u,
|
||||
0xc6e00bf3u, 0xd5a79147u, 0x06ca6351u, 0x14292967u, 0x27b70a85u, 0x2e1b2138u, 0x4d2c6dfcu,
|
||||
0x53380d13u, 0x650a7354u, 0x766a0abbu, 0x81c2c92eu, 0x92722c85u, 0xa2bfe8a1u, 0xa81a664bu,
|
||||
0xc24b8b70u, 0xc76c51a3u, 0xd192e819u, 0xd6990624u, 0xf40e3585u, 0x106aa070u, 0x19a4c116u,
|
||||
0x1e376c08u, 0x2748774cu, 0x34b0bcb5u, 0x391c0cb3u, 0x4ed8aa4au, 0x5b9cca4fu, 0x682e6ff3u,
|
||||
0x748f82eeu, 0x78a5636fu, 0x84c87814u, 0x8cc70208u, 0x90befffau, 0xa4506cebu, 0xbef9a3f7u,
|
||||
0xc67178f2u};
|
||||
|
||||
inline std::uint32_t rotr(std::uint32_t value, unsigned count) {
|
||||
return (value >> count) | (value << (32u - count));
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
void Sha256::reset() {
|
||||
state_[0] = 0x6a09e667u;
|
||||
state_[1] = 0xbb67ae85u;
|
||||
state_[2] = 0x3c6ef372u;
|
||||
state_[3] = 0xa54ff53au;
|
||||
state_[4] = 0x510e527fu;
|
||||
state_[5] = 0x9b05688cu;
|
||||
state_[6] = 0x1f83d9abu;
|
||||
state_[7] = 0x5be0cd19u;
|
||||
bit_count_ = 0;
|
||||
buffer_used_ = 0;
|
||||
std::memset(buffer_, 0, sizeof(buffer_));
|
||||
}
|
||||
|
||||
void Sha256::transform(const std::uint8_t block[64]) {
|
||||
std::uint32_t w[64];
|
||||
for (unsigned i = 0; i < 16; ++i) {
|
||||
w[i] = (static_cast<std::uint32_t>(block[i * 4]) << 24) |
|
||||
(static_cast<std::uint32_t>(block[i * 4 + 1]) << 16) |
|
||||
(static_cast<std::uint32_t>(block[i * 4 + 2]) << 8) |
|
||||
static_cast<std::uint32_t>(block[i * 4 + 3]);
|
||||
}
|
||||
for (unsigned i = 16; i < 64; ++i) {
|
||||
const std::uint32_t s0 = rotr(w[i - 15], 7) ^ rotr(w[i - 15], 18) ^ (w[i - 15] >> 3);
|
||||
const std::uint32_t s1 = rotr(w[i - 2], 17) ^ rotr(w[i - 2], 19) ^ (w[i - 2] >> 10);
|
||||
w[i] = w[i - 16] + s0 + w[i - 7] + s1;
|
||||
}
|
||||
std::uint32_t a = state_[0];
|
||||
std::uint32_t b = state_[1];
|
||||
std::uint32_t c = state_[2];
|
||||
std::uint32_t d = state_[3];
|
||||
std::uint32_t e = state_[4];
|
||||
std::uint32_t f = state_[5];
|
||||
std::uint32_t g = state_[6];
|
||||
std::uint32_t h = state_[7];
|
||||
for (unsigned i = 0; i < 64; ++i) {
|
||||
const std::uint32_t s1 = rotr(e, 6) ^ rotr(e, 11) ^ rotr(e, 25);
|
||||
const std::uint32_t ch = (e & f) ^ (~e & g);
|
||||
const std::uint32_t temp1 = h + s1 + ch + kK[i] + w[i];
|
||||
const std::uint32_t s0 = rotr(a, 2) ^ rotr(a, 13) ^ rotr(a, 22);
|
||||
const std::uint32_t maj = (a & b) ^ (a & c) ^ (b & c);
|
||||
const std::uint32_t temp2 = s0 + maj;
|
||||
h = g;
|
||||
g = f;
|
||||
f = e;
|
||||
e = d + temp1;
|
||||
d = c;
|
||||
c = b;
|
||||
b = a;
|
||||
a = temp1 + temp2;
|
||||
}
|
||||
state_[0] += a;
|
||||
state_[1] += b;
|
||||
state_[2] += c;
|
||||
state_[3] += d;
|
||||
state_[4] += e;
|
||||
state_[5] += f;
|
||||
state_[6] += g;
|
||||
state_[7] += h;
|
||||
}
|
||||
|
||||
void Sha256::update(const void* data, std::size_t size) {
|
||||
const std::uint8_t* bytes = static_cast<const std::uint8_t*>(data);
|
||||
bit_count_ += static_cast<std::uint64_t>(size) * 8u;
|
||||
while (size > 0) {
|
||||
const std::size_t space = 64u - buffer_used_;
|
||||
const std::size_t take = size < space ? size : space;
|
||||
std::memcpy(buffer_ + buffer_used_, bytes, take);
|
||||
buffer_used_ += take;
|
||||
bytes += take;
|
||||
size -= take;
|
||||
if (buffer_used_ == 64u) {
|
||||
transform(buffer_);
|
||||
buffer_used_ = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Sha256::finish(std::uint8_t out[32]) {
|
||||
const std::uint64_t total_bits = bit_count_;
|
||||
const std::uint8_t pad = 0x80u;
|
||||
update(&pad, 1);
|
||||
const std::uint8_t zero = 0x00u;
|
||||
while (buffer_used_ != 56u) {
|
||||
update(&zero, 1);
|
||||
}
|
||||
std::uint8_t length_bytes[8];
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
length_bytes[i] = static_cast<std::uint8_t>((total_bits >> (56u - i * 8u)) & 0xFFu);
|
||||
}
|
||||
std::memcpy(buffer_ + buffer_used_, length_bytes, 8);
|
||||
buffer_used_ += 8;
|
||||
transform(buffer_);
|
||||
buffer_used_ = 0;
|
||||
for (unsigned i = 0; i < 8; ++i) {
|
||||
out[i * 4 + 0] = static_cast<std::uint8_t>((state_[i] >> 24) & 0xFFu);
|
||||
out[i * 4 + 1] = static_cast<std::uint8_t>((state_[i] >> 16) & 0xFFu);
|
||||
out[i * 4 + 2] = static_cast<std::uint8_t>((state_[i] >> 8) & 0xFFu);
|
||||
out[i * 4 + 3] = static_cast<std::uint8_t>(state_[i] & 0xFFu);
|
||||
}
|
||||
}
|
||||
|
||||
std::string Sha256::finish_hex() {
|
||||
std::uint8_t digest[32];
|
||||
finish(digest);
|
||||
static const char* kHex = "0123456789abcdef";
|
||||
std::string text;
|
||||
text.resize(64);
|
||||
for (unsigned i = 0; i < 32; ++i) {
|
||||
text[i * 2] = kHex[(digest[i] >> 4) & 0x0Fu];
|
||||
text[i * 2 + 1] = kHex[digest[i] & 0x0Fu];
|
||||
}
|
||||
return text;
|
||||
}
|
||||
|
||||
std::string sha256_hex(const void* data, std::size_t size) {
|
||||
Sha256 hash;
|
||||
hash.update(data, size);
|
||||
return hash.finish_hex();
|
||||
}
|
||||
|
||||
bool sha256_hex_matches(const void* data, std::size_t size, const std::string& expected_hex) {
|
||||
if (expected_hex.size() != 64) {
|
||||
return false;
|
||||
}
|
||||
std::string actual = sha256_hex(data, size);
|
||||
for (std::size_t i = 0; i < 64; ++i) {
|
||||
char expected = expected_hex[i];
|
||||
if (expected >= 'A' && expected <= 'F') {
|
||||
expected = static_cast<char>(expected - 'A' + 'a');
|
||||
}
|
||||
if (actual[i] != expected) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace joc::crypto
|
||||
@@ -0,0 +1,31 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
|
||||
namespace joc::crypto {
|
||||
|
||||
class Sha256 {
|
||||
public:
|
||||
Sha256() { reset(); }
|
||||
|
||||
void reset();
|
||||
void update(const void* data, std::size_t size);
|
||||
void finish(std::uint8_t out[32]);
|
||||
std::string finish_hex();
|
||||
|
||||
private:
|
||||
void transform(const std::uint8_t block[64]);
|
||||
|
||||
std::uint32_t state_[8] = {};
|
||||
std::uint64_t bit_count_ = 0;
|
||||
std::uint8_t buffer_[64] = {};
|
||||
std::size_t buffer_used_ = 0;
|
||||
};
|
||||
|
||||
std::string sha256_hex(const void* data, std::size_t size);
|
||||
bool sha256_hex_matches(const void* data, std::size_t size, const std::string& expected_hex);
|
||||
|
||||
} // namespace joc::crypto
|
||||
@@ -0,0 +1,48 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <string>
|
||||
#include <utility>
|
||||
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc {
|
||||
|
||||
class Status {
|
||||
public:
|
||||
Status() = default;
|
||||
|
||||
static Status success() { return Status(); }
|
||||
|
||||
static Status fail(joc_error code, std::string stage, std::string message) {
|
||||
Status s;
|
||||
s.code_ = code;
|
||||
s.stage_ = std::move(stage);
|
||||
s.message_ = std::move(message);
|
||||
return s;
|
||||
}
|
||||
|
||||
bool ok() const { return code_ == JOC_OK; }
|
||||
joc_error code() const { return code_; }
|
||||
const std::string& stage() const { return stage_; }
|
||||
const std::string& message() const { return message_; }
|
||||
|
||||
private:
|
||||
joc_error code_ = JOC_OK;
|
||||
std::string stage_ = "none";
|
||||
std::string message_;
|
||||
};
|
||||
|
||||
// Stage names are kept as plain literals so that C++ and the Python frontend
|
||||
namespace stage {
|
||||
inline constexpr const char* kFoundation = "foundation";
|
||||
inline constexpr const char* kEac3 = "eac3_transport";
|
||||
inline constexpr const char* kEmdf = "emdf";
|
||||
inline constexpr const char* kJoc = "joc";
|
||||
inline constexpr const char* kOamd = "oamd";
|
||||
inline constexpr const char* kDsp = "dsp";
|
||||
inline constexpr const char* kRender = "render";
|
||||
inline constexpr const char* kOutput = "output";
|
||||
} // namespace stage
|
||||
|
||||
} // namespace joc
|
||||
@@ -0,0 +1,401 @@
|
||||
#include "hrtf/jochrtf.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
|
||||
#include "foundation/mini_json.h"
|
||||
#include "foundation/sha256.h"
|
||||
#include "io/npy.h"
|
||||
#include "io/zip_reader.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
namespace {
|
||||
|
||||
std::string to_upper(std::string text) {
|
||||
for (char& c : text) {
|
||||
if (c >= 'a' && c <= 'z') {
|
||||
c = static_cast<char>(c - 'a' + 'A');
|
||||
}
|
||||
}
|
||||
return text;
|
||||
}
|
||||
|
||||
bool is_sha256_hex(const std::string& text) {
|
||||
if (text.size() != 64) {
|
||||
return false;
|
||||
}
|
||||
for (const char c : text) {
|
||||
const bool digit = c >= '0' && c <= '9';
|
||||
const bool upper = c >= 'A' && c <= 'F';
|
||||
if (!digit && !upper) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// json.dumps(list(shape)) as the reference writes it, e.g. "[36, 2, 77]".
|
||||
std::string shape_json(const std::vector<std::int64_t>& shape) {
|
||||
std::string text = "[";
|
||||
for (std::size_t i = 0; i < shape.size(); ++i) {
|
||||
text += (i == 0 ? "" : ", ");
|
||||
text += std::to_string(shape[i]);
|
||||
}
|
||||
text += "]";
|
||||
return text;
|
||||
}
|
||||
|
||||
Status hrtf_fail(const std::string& message) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender, message);
|
||||
}
|
||||
|
||||
std::string payload_sha256(const std::vector<double>& centers,
|
||||
const std::vector<double>& coefficients,
|
||||
const std::vector<double>& delay_coefficients,
|
||||
const std::vector<double>& delay_bounds) {
|
||||
crypto::Sha256 hash;
|
||||
const char prefix[] = "JOC-HRTF-CACHE-PAYLOAD-V1";
|
||||
hash.update(prefix, sizeof(prefix) - 1);
|
||||
const std::uint8_t zero = 0;
|
||||
hash.update(&zero, 1);
|
||||
|
||||
struct Entry {
|
||||
const char* name;
|
||||
const char* dtype;
|
||||
const std::vector<double>* values;
|
||||
std::vector<std::int64_t> shape;
|
||||
};
|
||||
const Entry entries[4] = {
|
||||
{"band_center_frequencies_hz", "<f8", ¢ers, {kHybridBands}},
|
||||
{"coefficients", "<c16", &coefficients, {kShTerms, kEars, kHybridBands}},
|
||||
{"delay_coefficients", "<f8", &delay_coefficients, {kShTerms, kEars}},
|
||||
{"delay_bounds", "<f8", &delay_bounds, {2, 2}},
|
||||
};
|
||||
for (const Entry& entry : entries) {
|
||||
const std::string name(entry.name);
|
||||
const std::string dtype(entry.dtype);
|
||||
const std::string shape = shape_json(entry.shape);
|
||||
hash.update(name.data(), name.size());
|
||||
hash.update(&zero, 1);
|
||||
hash.update(dtype.data(), dtype.size());
|
||||
hash.update(&zero, 1);
|
||||
hash.update(shape.data(), shape.size());
|
||||
hash.update(&zero, 1);
|
||||
hash.update(entry.values->data(), entry.values->size() * sizeof(double));
|
||||
}
|
||||
return hash.finish_hex();
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Status load_jochrtf(const std::string& path, Field* out) {
|
||||
if (out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null field");
|
||||
}
|
||||
io::ZipArchive archive;
|
||||
std::string error;
|
||||
if (!archive.open(path, &error)) {
|
||||
return Status::fail(JOC_ERR_HRTF_NOT_FOUND, stage::kRender,
|
||||
"cannot read compiled HRTF " + path + ": " + error);
|
||||
}
|
||||
|
||||
// Member set must be exactly the five expected names.
|
||||
static const char* kMembers[5] = {"metadata_json.npy", "band_center_frequencies_hz.npy",
|
||||
"coefficients.npy", "delay_coefficients.npy",
|
||||
"delay_bounds.npy"};
|
||||
if (archive.entries().size() != 5u) {
|
||||
return hrtf_fail("compiled HRTF cache has an invalid member set (" +
|
||||
std::to_string(archive.entries().size()) + " members)");
|
||||
}
|
||||
for (const char* name : kMembers) {
|
||||
if (archive.find(name) == nullptr) {
|
||||
return hrtf_fail(std::string("compiled HRTF cache is missing ") + name);
|
||||
}
|
||||
}
|
||||
|
||||
auto read_member = [&](const char* name, std::vector<std::uint8_t>* raw,
|
||||
io::NpyArray* array) -> Status {
|
||||
if (!archive.read_member(name, raw, &error)) {
|
||||
return hrtf_fail(std::string("compiled HRTF member ") + name + ": " + error);
|
||||
}
|
||||
if (!io::parse_npy(raw->data(), raw->size(), array, &error)) {
|
||||
return hrtf_fail(std::string("compiled HRTF member ") + name + ": " + error);
|
||||
}
|
||||
if (array->fortran_order) {
|
||||
return hrtf_fail(std::string("compiled HRTF member must be C-contiguous: ") + name);
|
||||
}
|
||||
return Status::success();
|
||||
};
|
||||
|
||||
std::vector<std::uint8_t> raw;
|
||||
io::NpyArray array;
|
||||
|
||||
Status status = read_member("metadata_json.npy", &raw, &array);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::string metadata_text;
|
||||
if (!io::npy_unicode_to_utf8(array, &metadata_text, &error)) {
|
||||
return hrtf_fail("compiled HRTF metadata: " + error);
|
||||
}
|
||||
if (metadata_text.size() > 64u * 1024u) {
|
||||
return hrtf_fail("compiled HRTF metadata is too large");
|
||||
}
|
||||
|
||||
status = read_member("band_center_frequencies_hz.npy", &raw, &array);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (array.descr != "<f8" || !io::npy_shape_is(array, {kHybridBands})) {
|
||||
return hrtf_fail("band_center_frequencies_hz must be <f8(77,)");
|
||||
}
|
||||
std::vector<double> centers;
|
||||
io::npy_to_double(array, ¢ers, &error);
|
||||
|
||||
status = read_member("coefficients.npy", &raw, &array);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (array.descr != "<c16" || !io::npy_shape_is(array, {kShTerms, kEars, kHybridBands})) {
|
||||
return hrtf_fail("coefficients must be <c16(36, 2, 77)");
|
||||
}
|
||||
std::vector<double> coefficients;
|
||||
if (!io::npy_to_double(array, &coefficients, &error)) {
|
||||
return hrtf_fail("coefficients: " + error);
|
||||
}
|
||||
|
||||
status = read_member("delay_coefficients.npy", &raw, &array);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (array.descr != "<f8" || !io::npy_shape_is(array, {kShTerms, kEars})) {
|
||||
return hrtf_fail("delay_coefficients must be <f8(36, 2)");
|
||||
}
|
||||
std::vector<double> delay_coefficients;
|
||||
io::npy_to_double(array, &delay_coefficients, &error);
|
||||
|
||||
status = read_member("delay_bounds.npy", &raw, &array);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (array.descr != "<f8" || !io::npy_shape_is(array, {2, 2})) {
|
||||
return hrtf_fail("delay_bounds must be <f8(2, 2)");
|
||||
}
|
||||
std::vector<double> delay_bounds;
|
||||
io::npy_to_double(array, &delay_bounds, &error);
|
||||
|
||||
std::vector<json::Member> members;
|
||||
if (!json::parse_object(metadata_text, &members, &error)) {
|
||||
return hrtf_fail("compiled HRTF metadata: " + error);
|
||||
}
|
||||
auto require_string = [&](const char* key, std::string* value) -> Status {
|
||||
const json::Member* member = json::find(members, key);
|
||||
if (member == nullptr || !json::as_string(*member, value)) {
|
||||
return hrtf_fail(std::string("compiled HRTF metadata is missing ") + key);
|
||||
}
|
||||
return Status::success();
|
||||
};
|
||||
std::string magic;
|
||||
std::string schema;
|
||||
std::string source_sha256;
|
||||
std::string cache_key;
|
||||
std::string payload_hash;
|
||||
std::string delay_source;
|
||||
status = require_string("magic", &magic);
|
||||
if (!status.ok()) { return status; }
|
||||
status = require_string("cache_schema", &schema);
|
||||
if (!status.ok()) { return status; }
|
||||
status = require_string("source_sha256", &source_sha256);
|
||||
if (!status.ok()) { return status; }
|
||||
status = require_string("cache_key", &cache_key);
|
||||
if (!status.ok()) { return status; }
|
||||
status = require_string("payload_sha256", &payload_hash);
|
||||
if (!status.ok()) { return status; }
|
||||
status = require_string("delay_source", &delay_source);
|
||||
if (!status.ok()) { return status; }
|
||||
|
||||
if (magic != kMagic) {
|
||||
return hrtf_fail("compiled HRTF magic mismatch: " + magic);
|
||||
}
|
||||
if (schema != kCacheSchema) {
|
||||
return hrtf_fail("compiled HRTF cache schema mismatch: " + schema);
|
||||
}
|
||||
const json::Member* version_member = json::find(members, "format_version");
|
||||
long long version = -1;
|
||||
if (version_member == nullptr || !json::as_integer(*version_member, &version)) {
|
||||
return hrtf_fail("compiled HRTF metadata is missing format_version");
|
||||
}
|
||||
if (version != kFormatVersion) {
|
||||
return Status::fail(JOC_ERR_HRTF_VERSION, stage::kRender,
|
||||
"unsupported .jochrtf version " + std::to_string(version) +
|
||||
"; rebuild it from the source SOFA");
|
||||
}
|
||||
out->source_sha256 = to_upper(source_sha256);
|
||||
out->cache_key = to_upper(cache_key);
|
||||
if (!is_sha256_hex(out->source_sha256)) {
|
||||
return hrtf_fail("compiled HRTF source_sha256 is not a 64-digit digest");
|
||||
}
|
||||
if (!is_sha256_hex(out->cache_key)) {
|
||||
return hrtf_fail("compiled HRTF cache_key is not a 64-digit digest");
|
||||
}
|
||||
|
||||
const std::string expected = payload_sha256(centers, coefficients, delay_coefficients,
|
||||
delay_bounds);
|
||||
if (to_upper(payload_hash) != to_upper(expected)) {
|
||||
return Status::fail(JOC_ERR_HRTF_HASH, stage::kRender,
|
||||
"compiled HRTF payload hash mismatch");
|
||||
}
|
||||
out->payload_sha256 = to_upper(payload_hash);
|
||||
|
||||
const json::Member* radius_member = json::find(members, "measurement_radius_m");
|
||||
double radius = 0.0;
|
||||
if (radius_member == nullptr || !json::as_number(*radius_member, &radius) || radius <= 0.0) {
|
||||
return hrtf_fail("compiled HRTF measurement_radius_m must be a positive number");
|
||||
}
|
||||
out->measurement_radius_m = radius;
|
||||
const json::Member* order_member = json::find(members, "order");
|
||||
long long order = 0;
|
||||
if (order_member == nullptr || !json::as_integer(*order_member, &order) || order <= 0 ||
|
||||
order * order > kShTerms) {
|
||||
return hrtf_fail("compiled HRTF order is out of range");
|
||||
}
|
||||
out->order = order;
|
||||
|
||||
for (const double value : coefficients) {
|
||||
if (!std::isfinite(value)) {
|
||||
return hrtf_fail("compiled HRTF coefficients contain non-finite values");
|
||||
}
|
||||
}
|
||||
for (const double value : delay_coefficients) {
|
||||
if (!std::isfinite(value) || std::abs(value) > 48000.0 * 64.0) {
|
||||
return hrtf_fail("compiled HRTF delay coefficients are out of range");
|
||||
}
|
||||
}
|
||||
for (const double value : delay_bounds) {
|
||||
if (!std::isfinite(value)) {
|
||||
return hrtf_fail("compiled HRTF delay bounds contain non-finite values");
|
||||
}
|
||||
}
|
||||
if (delay_bounds.size() == 4u && delay_bounds[0] > delay_bounds[1]) {
|
||||
return hrtf_fail("compiled HRTF delay bounds are inverted");
|
||||
}
|
||||
|
||||
if (const json::Member* member = json::find(members, "compiler_version")) {
|
||||
json::as_string(*member, &out->compiler_version);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "phase_policy_version")) {
|
||||
json::as_string(*member, &out->phase_policy_version);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "sh_convention")) {
|
||||
json::as_string(*member, &out->sh_convention);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "source_display_name")) {
|
||||
json::as_string(*member, &out->source_display_name);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "projection_ridge")) {
|
||||
json::as_number(*member, &out->projection_ridge);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "spherical_harmonic_ridge")) {
|
||||
json::as_number(*member, &out->spherical_harmonic_ridge);
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "fit_report")) {
|
||||
out->fit_report_json = member->raw;
|
||||
}
|
||||
if (const json::Member* member = json::find(members, "filterbank")) {
|
||||
out->filterbank_json = member->raw;
|
||||
}
|
||||
out->delay_source = delay_source;
|
||||
out->metadata_json = metadata_text;
|
||||
out->coefficients = std::move(coefficients);
|
||||
out->delay_coefficients = std::move(delay_coefficients);
|
||||
out->delay_bounds = std::move(delay_bounds);
|
||||
out->band_centers_hz = std::move(centers);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status load_kernels(const std::string& npz_path, Kernels* out) {
|
||||
if (out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kRender, "null kernels");
|
||||
}
|
||||
io::ZipArchive archive;
|
||||
std::string error;
|
||||
if (!archive.open(npz_path, &error)) {
|
||||
return Status::fail(JOC_ERR_HRTF_NOT_FOUND, stage::kRender,
|
||||
"cannot read kernel tables " + npz_path + ": " + error);
|
||||
}
|
||||
|
||||
struct Request {
|
||||
const char* member;
|
||||
const char* shape_text;
|
||||
std::vector<std::int64_t> shape;
|
||||
};
|
||||
const Request requests[6] = {
|
||||
{"qmf_analysis_coefficients.npy", "<f4", {64, 10}},
|
||||
{"hybrid_analysis_low_kernel.npy", "<f4", {3, 2, 13, 16, 2}},
|
||||
{"hybrid_synthesis_indices.npy", "<i2", {154, 4}},
|
||||
{"hybrid_synthesis_values.npy", "<f4", {154}},
|
||||
{"qmf_synthesis_basis.npy", "<f8", {64, 4, 128}},
|
||||
{"qmf_synthesis_taps.npy", "<f8", {64, 10, 4}},
|
||||
};
|
||||
|
||||
std::vector<std::uint8_t> raw;
|
||||
std::vector<std::uint8_t> ordered;
|
||||
for (const Request& request : requests) {
|
||||
const std::string name = request.member;
|
||||
if (!archive.read_member(name, &raw, &error)) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"kernel table member " + name + ": " + error);
|
||||
}
|
||||
io::NpyArray array;
|
||||
if (!io::parse_npy(raw.data(), raw.size(), &array, &error)) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"kernel table member " + name + ": " + error);
|
||||
}
|
||||
if (array.descr != request.shape_text || !io::npy_shape_is(array, request.shape)) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"kernel table member " + name + " has an unexpected dtype/shape");
|
||||
}
|
||||
// Logical C order: required because the reused kernel indexes the hybrid
|
||||
// synthesis table row-major while the shipped member is Fortran-order.
|
||||
if (!io::npy_to_c_order(array, &ordered, &error)) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"kernel table member " + name + ": " + error);
|
||||
}
|
||||
const std::size_t count = array.element_count();
|
||||
if (std::strcmp(request.member, "qmf_analysis_coefficients.npy") == 0) {
|
||||
std::vector<float> values(count);
|
||||
std::memcpy(values.data(), ordered.data(), count * sizeof(float));
|
||||
out->qmf_analysis.assign(values.begin(), values.end());
|
||||
} else if (std::strcmp(request.member, "hybrid_analysis_low_kernel.npy") == 0) {
|
||||
std::vector<float> values(count);
|
||||
std::memcpy(values.data(), ordered.data(), count * sizeof(float));
|
||||
out->hybrid_low.assign(values.begin(), values.end());
|
||||
} else if (std::strcmp(request.member, "hybrid_synthesis_indices.npy") == 0) {
|
||||
out->hybrid_indices.resize(count);
|
||||
std::memcpy(out->hybrid_indices.data(), ordered.data(), count * sizeof(std::int16_t));
|
||||
} else if (std::strcmp(request.member, "hybrid_synthesis_values.npy") == 0) {
|
||||
std::vector<float> values(count);
|
||||
std::memcpy(values.data(), ordered.data(), count * sizeof(float));
|
||||
out->hybrid_values.assign(values.begin(), values.end());
|
||||
} else if (std::strcmp(request.member, "qmf_synthesis_basis.npy") == 0) {
|
||||
std::memcpy(out->qmf_basis.empty() ? (out->qmf_basis.resize(count), out->qmf_basis.data())
|
||||
: out->qmf_basis.data(),
|
||||
ordered.data(), count * sizeof(double));
|
||||
out->qmf_basis.resize(count);
|
||||
} else {
|
||||
out->qmf_taps.resize(count);
|
||||
std::memcpy(out->qmf_taps.data(), ordered.data(), count * sizeof(double));
|
||||
}
|
||||
}
|
||||
out->hybrid_count = static_cast<std::uint32_t>(out->hybrid_values.size());
|
||||
if (out->hybrid_count == 0u) {
|
||||
return Status::fail(JOC_ERR_HRTF_FORMAT, stage::kRender,
|
||||
"kernel tables contain no hybrid synthesis entries");
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -0,0 +1,66 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
inline constexpr int kShTerms = 36;
|
||||
inline constexpr int kEars = 2;
|
||||
inline constexpr int kHybridBands = 77;
|
||||
inline constexpr int kFormatVersion = 1;
|
||||
inline constexpr const char* kMagic = "JOC-HRTF-CACHE";
|
||||
inline constexpr const char* kCacheSchema = "joc-compiled-hrtf-v1";
|
||||
|
||||
struct Field {
|
||||
std::vector<double> coefficients;
|
||||
std::vector<double> delay_coefficients;
|
||||
std::vector<double> delay_bounds;
|
||||
std::vector<double> band_centers_hz;
|
||||
double measurement_radius_m = 1.0;
|
||||
long long order = 5;
|
||||
std::string source_sha256;
|
||||
std::string cache_key;
|
||||
std::string payload_sha256;
|
||||
std::string delay_source;
|
||||
std::string compiler_version;
|
||||
std::string phase_policy_version;
|
||||
std::string sh_convention;
|
||||
std::string filterbank_json;
|
||||
std::string metadata_json;
|
||||
// Compile-side metadata, needed to write the cache back out unchanged.
|
||||
std::string source_display_name;
|
||||
std::string fit_report_json;
|
||||
double projection_ridge = 0.0;
|
||||
double spherical_harmonic_ridge = 0.0;
|
||||
};
|
||||
|
||||
Status load_jochrtf(const std::string& path, Field* out);
|
||||
|
||||
// Binaural filterbank kernels, as the reused kernel expects them (C order, the
|
||||
// exact dtypes of the ABI parameters).
|
||||
struct Kernels {
|
||||
std::vector<double> qmf_analysis;
|
||||
std::vector<double> hybrid_low;
|
||||
std::vector<std::int16_t> hybrid_indices;
|
||||
std::vector<double> hybrid_values;
|
||||
std::vector<double> qmf_basis;
|
||||
std::vector<double> qmf_taps;
|
||||
std::uint32_t hybrid_count = 0;
|
||||
};
|
||||
|
||||
// Loads a kernel-table archive. The file path is an override for verification;
|
||||
// the shipped tables are embedded (see builtin_kernels) so no data file is needed.
|
||||
// The Fortran-order index member is transposed into C order on purpose: the reused
|
||||
// kernel indexes the hybrid synthesis table row-major.
|
||||
Status load_kernels(const std::string& npz_path, Kernels* out);
|
||||
|
||||
// The public filterbank tables compiled into the library (identical values to the
|
||||
// archive the file loader accepts; the unit test checks their hashes).
|
||||
const Kernels& builtin_kernels();
|
||||
|
||||
} // namespace joc::hrtf
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,610 @@
|
||||
#include "hrtf/public_filterbank.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cmath>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/fft.h"
|
||||
#include "simd/simd.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr double kPi = 3.14159265358979323846;
|
||||
constexpr std::size_t kQmfLength = dsp::kQmfFftSize;
|
||||
constexpr int kQmfTaps = 10;
|
||||
constexpr int kSynthesisRank = 4;
|
||||
constexpr int kSynthesisTaps = 10;
|
||||
|
||||
// ----------------------------------------------------------- filterbank -----
|
||||
|
||||
// One shared forward plan for the 128-point QMF transform. The analysis bank runs
|
||||
// it 2 * slots * channels times per chunk, so the twiddle recurrence is built once
|
||||
// instead of being re-derived inside every butterfly.
|
||||
const dsp::FftPlan& qmf_fft_plan() {
|
||||
static const dsp::FftPlan plan(dsp::kQmfFftSize, false);
|
||||
return plan;
|
||||
}
|
||||
|
||||
// Public 64-band complex QMF analysis (public_filterbank.QmfAnalysis).
|
||||
class QmfAnalysis {
|
||||
public:
|
||||
static_assert(static_cast<std::size_t>(kQmfBands) == simd::kQmfAnalysisBands,
|
||||
"the dispatched accumulate is written for this band count");
|
||||
QmfAnalysis(const Kernels& kernels, std::size_t channels)
|
||||
: channels_(channels), coefficients_(kernels.qmf_analysis) {
|
||||
history_.assign(9u * channels_ * kQmfBands, 0.0);
|
||||
// The polyphase MAC consumes one coefficient per band, so the shipped
|
||||
// [band][tap] layout makes its inner loop a stride-10 gather. Transposing
|
||||
// once here turns that into a contiguous AXPY. The coefficient values and
|
||||
// the accumulation order are untouched, so the sums are bit-identical.
|
||||
coefficients_by_lag_.resize(static_cast<std::size_t>(kQmfTaps) * kQmfBands);
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
for (int tap = 0; tap < kQmfTaps; ++tap) {
|
||||
coefficients_by_lag_[static_cast<std::size_t>(tap) * kQmfBands +
|
||||
static_cast<std::size_t>(band)] =
|
||||
coefficients_[static_cast<std::size_t>(band) * kQmfTaps +
|
||||
static_cast<std::size_t>(tap)];
|
||||
}
|
||||
}
|
||||
premultiply_.resize(kQmfBands);
|
||||
post_.resize(kQmfBands);
|
||||
even_post_.resize(kQmfBands);
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
const double phase = static_cast<double>(band);
|
||||
premultiply_[static_cast<std::size_t>(band)] =
|
||||
std::polar(1.0, -kPi * phase / 128.0);
|
||||
post_[static_cast<std::size_t>(band)] =
|
||||
std::polar(1.0, -3.0 * (phase + 0.5) * kPi / 128.0);
|
||||
even_post_[static_cast<std::size_t>(band)] =
|
||||
Complex(0.0, band % 2 == 0 ? 1.0 : -1.0);
|
||||
}
|
||||
}
|
||||
|
||||
void reset() { std::fill(history_.begin(), history_.end(), 0.0); }
|
||||
|
||||
// samples: [slots*64, channels]; output: [slots, channels, 64] complex.
|
||||
void process(const std::vector<double>& samples, std::size_t slots,
|
||||
std::vector<Complex>* output) {
|
||||
const std::size_t joined_slots = 9u + slots;
|
||||
const std::size_t history_size = 9u * channels_ * kQmfBands;
|
||||
const std::size_t joined_size = joined_slots * channels_ * kQmfBands;
|
||||
// The joined window is filled completely -- the history lands in its first
|
||||
// 9 * channels * 64 entries and the new samples in the rest -- so it is a
|
||||
// reusable scratch buffer rather than a fresh zero-filled allocation. The
|
||||
// history tail is taken by index instead of from end(), because the buffer may
|
||||
// be longer than the window this call uses.
|
||||
if (joined_.size() < joined_size) {
|
||||
joined_.resize(joined_size);
|
||||
}
|
||||
std::copy(history_.begin(), history_.end(), joined_.begin());
|
||||
std::copy(samples.begin(), samples.begin() + static_cast<std::ptrdiff_t>(slots * channels_ * kQmfBands),
|
||||
joined_.begin() + static_cast<std::ptrdiff_t>(history_size));
|
||||
|
||||
// The two polyphase accumulators are read before they are written, so their
|
||||
// zero fill is load-bearing and stays; only the per-call allocation goes.
|
||||
const std::size_t accumulator_size = slots * channels_ * kQmfBands;
|
||||
if (even_.size() < accumulator_size) {
|
||||
even_.resize(accumulator_size);
|
||||
}
|
||||
if (odd_.size() < accumulator_size) {
|
||||
odd_.resize(accumulator_size);
|
||||
}
|
||||
std::fill(even_.begin(), even_.begin() + static_cast<std::ptrdiff_t>(accumulator_size), 0.0);
|
||||
std::fill(odd_.begin(), odd_.begin() + static_cast<std::ptrdiff_t>(accumulator_size), 0.0);
|
||||
// The ten lags are ten accumulate passes over the same 64 bands with one
|
||||
// shared coefficient row; the bands are independent accumulations of a
|
||||
// single product each, so they are what the dispatched kernel puts in its
|
||||
// lanes, and every band keeps the caller's own multiply-then-add.
|
||||
//
|
||||
// Slots are processed in blocks, with the lag loop inside: one lag pass
|
||||
// touches every source row once, so running the ten passes over the whole
|
||||
// chunk re-reads the joined window ten times -- at 1536 slots that is
|
||||
// hundreds of megabytes per chunk and the loop ends up bound by memory, not
|
||||
// by arithmetic. A block's ten lag passes instead slide over a window of
|
||||
// (block + 9) rows that stays in the second-level cache. Lags still run in
|
||||
// ascending order inside a block, which is the order each output's sum is
|
||||
// formed in, so nothing about the arithmetic changes.
|
||||
constexpr std::size_t kSlotBlock = 32;
|
||||
for (std::size_t first = 0u; first < slots; first += kSlotBlock) {
|
||||
const std::size_t block = std::min(kSlotBlock, slots - first);
|
||||
for (int lag = 0; lag < kQmfTaps; ++lag) {
|
||||
std::vector<double>& target = (lag % 2 == 0) ? even_ : odd_;
|
||||
const double* row =
|
||||
coefficients_by_lag_.data() + static_cast<std::size_t>(lag) * kQmfBands;
|
||||
const std::size_t source_slot = 9u - static_cast<std::size_t>(lag) + first;
|
||||
simd::qmf_analysis_taps(
|
||||
target.data() + first * channels_ * kQmfBands,
|
||||
joined_.data() + source_slot * channels_ * kQmfBands, row,
|
||||
block * channels_);
|
||||
}
|
||||
}
|
||||
std::copy(joined_.begin() + static_cast<std::ptrdiff_t>(joined_size - history_size),
|
||||
joined_.begin() + static_cast<std::ptrdiff_t>(joined_size), history_.begin());
|
||||
|
||||
// Every output element is assigned below, so the size is all that has to be
|
||||
// established; a resize of an already correctly sized buffer touches nothing.
|
||||
output->resize(slots * channels_ * kQmfBands);
|
||||
std::array<Complex, dsp::kQmfFftSize> even_spectrum{};
|
||||
std::array<Complex, dsp::kQmfFftSize> odd_spectrum{};
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels_; ++channel) {
|
||||
const double* even_values = even_.data() + (slot * channels_ + channel) * kQmfBands;
|
||||
const double* odd_values = odd_.data() + (slot * channels_ + channel) * kQmfBands;
|
||||
transform(even_values, &even_spectrum);
|
||||
transform(odd_values, &odd_spectrum);
|
||||
Complex* destination =
|
||||
output->data() + (slot * channels_ + channel) * kQmfBands;
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
destination[band] = odd_spectrum[static_cast<std::size_t>(band)] +
|
||||
even_spectrum[static_cast<std::size_t>(band)] *
|
||||
even_post_[static_cast<std::size_t>(band)];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
void transform(const double* values, std::array<Complex, dsp::kQmfFftSize>* spectrum) {
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
(*spectrum)[static_cast<std::size_t>(band)] =
|
||||
Complex(values[band], 0.0) * premultiply_[static_cast<std::size_t>(band)];
|
||||
}
|
||||
for (int index = kQmfBands; index < dsp::kQmfFftSize; ++index) {
|
||||
(*spectrum)[static_cast<std::size_t>(index)] = Complex(0.0, 0.0);
|
||||
}
|
||||
dsp::fft_radix2(spectrum, qmf_fft_plan());
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
(*spectrum)[static_cast<std::size_t>(band)] *= post_[static_cast<std::size_t>(band)];
|
||||
}
|
||||
}
|
||||
|
||||
std::size_t channels_;
|
||||
std::vector<double> coefficients_; // [64][10]
|
||||
std::vector<double> coefficients_by_lag_; // [10][64], the same values transposed
|
||||
std::vector<double> history_; // [9][channels][64]
|
||||
std::vector<double> joined_; // scratch, [9 + slots][channels][64]
|
||||
std::vector<double> even_; // scratch, [slots][channels][64], zeroed per call
|
||||
std::vector<double> odd_; // scratch, [slots][channels][64], zeroed per call
|
||||
std::vector<Complex> premultiply_;
|
||||
std::vector<Complex> post_;
|
||||
std::vector<Complex> even_post_;
|
||||
};
|
||||
|
||||
// Sparse 64-QMF to 77-hybrid analysis (public_filterbank.HybridAnalysis).
|
||||
class HybridAnalysis {
|
||||
public:
|
||||
HybridAnalysis(const Kernels& kernels, std::size_t channels)
|
||||
: channels_(channels), low_kernel_(kernels.hybrid_low) {
|
||||
history_.assign(12u * channels_ * 3u * 2u, 0.0);
|
||||
high_history_.assign(6u * channels_ * 61u, Complex(0.0, 0.0));
|
||||
// The dispatched join walks one term at a time and adds its 32 weights to
|
||||
// 32 outputs, so the shipped [tap][band][component] table is regrouped to
|
||||
// the term order the caller accumulates in. Same weights, same order.
|
||||
const std::size_t outputs = simd::kHybridOutputs;
|
||||
low_by_term_.resize(simd::kHybridTerms * outputs);
|
||||
for (int lag = 0; lag < 13; ++lag) {
|
||||
for (int point = 0; point < 3; ++point) {
|
||||
for (int input = 0; input < 2; ++input) {
|
||||
const std::size_t term =
|
||||
(static_cast<std::size_t>(lag) * 3u + static_cast<std::size_t>(point)) * 2u +
|
||||
static_cast<std::size_t>(input);
|
||||
const std::size_t source = (static_cast<std::size_t>(point) * 2u +
|
||||
static_cast<std::size_t>(input)) * 13u +
|
||||
static_cast<std::size_t>(lag);
|
||||
for (std::size_t output = 0u; output < outputs; ++output) {
|
||||
low_by_term_[term * outputs + output] =
|
||||
low_kernel_[source * outputs + output];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
low_values_.resize(simd::kHybridJoinBlock * simd::kHybridTerms);
|
||||
low_out_.resize(simd::kHybridJoinBlock * outputs);
|
||||
}
|
||||
|
||||
void reset() {
|
||||
std::fill(history_.begin(), history_.end(), 0.0);
|
||||
std::fill(high_history_.begin(), high_history_.end(), Complex(0.0, 0.0));
|
||||
}
|
||||
|
||||
// qmf: [slots, channels, 64]; output: [slots, channels, 77] complex.
|
||||
void process(const std::vector<Complex>& qmf, std::size_t slots,
|
||||
std::vector<Complex>* output) {
|
||||
const std::size_t joined_slots = 12u + slots;
|
||||
const std::size_t history_size = 12u * channels_ * 6u;
|
||||
const std::size_t joined_size = joined_slots * channels_ * 3u * 2u;
|
||||
// Both the joined window and the pending high-band history are written in full
|
||||
// before they are read, so they are reused scratch buffers; the history tail is
|
||||
// taken by index because the buffer can be longer than this call's window.
|
||||
if (joined_.size() < joined_size) {
|
||||
joined_.resize(joined_size);
|
||||
}
|
||||
std::copy(history_.begin(), history_.end(), joined_.begin());
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels_; ++channel) {
|
||||
const Complex* source = qmf.data() + (slot * channels_ + channel) * kQmfBands;
|
||||
double* destination =
|
||||
joined_.data() + ((12u + slot) * channels_ + channel) * 6u;
|
||||
for (int band = 0; band < 3; ++band) {
|
||||
destination[static_cast<std::size_t>(band) * 2u] = source[band].real();
|
||||
destination[static_cast<std::size_t>(band) * 2u + 1u] = source[band].imag();
|
||||
}
|
||||
}
|
||||
}
|
||||
// The low bands are accumulated in a register block and written straight into
|
||||
// the output, and the high bands are written by the pass below; between them
|
||||
// every one of the 77 bands is assigned, so only the size has to be set.
|
||||
output->resize(slots * channels_ * kHybridBands);
|
||||
// The thirteen taps are summed in a per-output register block and the low
|
||||
// bands are written straight into the output. Keeping a separate low plane
|
||||
// and then copying it into the output re-streams tens of megabytes per chunk
|
||||
// for nothing, and only the first kHybridLow bands are ever touched. The
|
||||
// join itself is dispatched (see src/simd/simd.h): the 32 outputs of a
|
||||
// row are 32 independent accumulations over the same 78 terms, which is what
|
||||
// shares a vector. Every lane keeps the caller's term order -- lag, then
|
||||
// point, then input -- and its two roundings, and skips exactly the terms
|
||||
// this loop skips. Rows are staged in blocks so the gathered values do not
|
||||
// spill out of the first-level cache.
|
||||
const std::size_t hybrid_rows = slots * channels_;
|
||||
const std::size_t block = simd::kHybridJoinBlock;
|
||||
const std::size_t terms = simd::kHybridTerms;
|
||||
for (std::size_t first = 0u; first < hybrid_rows; first += block) {
|
||||
const std::size_t count = std::min(block, hybrid_rows - first);
|
||||
for (std::size_t index = 0u; index < count; ++index) {
|
||||
const std::size_t row = first + index;
|
||||
const std::size_t slot = row / channels_;
|
||||
const std::size_t channel = row % channels_;
|
||||
double* staged = low_values_.data() + index * terms;
|
||||
for (int lag = 0; lag < 13; ++lag) {
|
||||
const std::size_t source_slot = 12u - static_cast<std::size_t>(lag) + slot;
|
||||
const double* source =
|
||||
joined_.data() + (source_slot * channels_ + channel) * 6u;
|
||||
for (int point = 0; point < 3; ++point) {
|
||||
for (int input = 0; input < 2; ++input) {
|
||||
staged[(static_cast<std::size_t>(lag) * 3u +
|
||||
static_cast<std::size_t>(point)) * 2u +
|
||||
static_cast<std::size_t>(input)] =
|
||||
source[static_cast<std::size_t>(point) * 2u +
|
||||
static_cast<std::size_t>(input)];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
simd::hybrid_low_join(low_values_.data(), low_by_term_.data(),
|
||||
low_out_.data(), count);
|
||||
for (std::size_t index = 0u; index < count; ++index) {
|
||||
Complex* destination = output->data() + (first + index) * kHybridBands;
|
||||
const double* values = low_out_.data() + index * simd::kHybridOutputs;
|
||||
for (int band = 0; band < kHybridLow; ++band) {
|
||||
destination[band] = Complex(values[static_cast<std::size_t>(band) * 2u],
|
||||
values[static_cast<std::size_t>(band) * 2u + 1u]);
|
||||
}
|
||||
}
|
||||
}
|
||||
std::copy(joined_.begin() + static_cast<std::ptrdiff_t>(joined_size - history_size),
|
||||
joined_.begin() + static_cast<std::ptrdiff_t>(joined_size), history_.begin());
|
||||
|
||||
// The high bands pass through unchanged but delayed by the six slots of
|
||||
// history the reference concatenates in front of them. Only the last six
|
||||
// entries of that concatenation survive into high_history_, so a six-entry
|
||||
// register replaces the (6 + slots) plane and its full copy. Note the
|
||||
// output reads the concatenation at index `slot`, not `6 + slot`, so the
|
||||
// first six output slots come from the history: that offset is part of the
|
||||
// current output and is preserved verbatim.
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels_; ++channel) {
|
||||
Complex* destination = output->data() +
|
||||
(slot * channels_ + channel) * kHybridBands + kHybridLow;
|
||||
if (slot < 6u) {
|
||||
const Complex* source =
|
||||
high_history_.data() + (slot * channels_ + channel) * 61u;
|
||||
for (int band = 0; band < 61; ++band) {
|
||||
destination[band] = source[band];
|
||||
}
|
||||
} else {
|
||||
const Complex* source =
|
||||
qmf.data() + ((slot - 6u) * channels_ + channel) * kQmfBands;
|
||||
for (int band = 3; band < kQmfBands; ++band) {
|
||||
destination[static_cast<std::size_t>(band - 3)] = source[band];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Every entry of the pending high-band history is written here, so it is a
|
||||
// reusable scratch buffer; the copy into the live history is kept as it was.
|
||||
if (next_high_history_.size() < 6u * channels_ * 61u) {
|
||||
next_high_history_.resize(6u * channels_ * 61u);
|
||||
}
|
||||
for (std::size_t entry = 0u; entry < 6u; ++entry) {
|
||||
const std::size_t combined = slots + entry;
|
||||
for (std::size_t channel = 0u; channel < channels_; ++channel) {
|
||||
Complex* destination =
|
||||
next_high_history_.data() + (entry * channels_ + channel) * 61u;
|
||||
if (combined < 6u) {
|
||||
const Complex* source =
|
||||
high_history_.data() + (combined * channels_ + channel) * 61u;
|
||||
for (int band = 0; band < 61; ++band) {
|
||||
destination[band] = source[band];
|
||||
}
|
||||
} else {
|
||||
const Complex* source =
|
||||
qmf.data() + ((combined - 6u) * channels_ + channel) * kQmfBands;
|
||||
for (int band = 3; band < kQmfBands; ++band) {
|
||||
destination[static_cast<std::size_t>(band - 3)] = source[band];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
std::copy(next_high_history_.begin(), next_high_history_.end(), high_history_.begin());
|
||||
}
|
||||
|
||||
private:
|
||||
std::size_t channels_;
|
||||
std::vector<double> low_kernel_; // [3][2][13][16][2]
|
||||
std::vector<double> low_by_term_; // [78][32], the same weights in the caller's term order
|
||||
std::vector<double> low_values_; // scratch, [block][78]
|
||||
std::vector<double> low_out_; // scratch, [block][32]
|
||||
std::vector<double> history_; // [12][channels][3][2]
|
||||
std::vector<Complex> high_history_; // [6][channels][61]
|
||||
std::vector<double> joined_; // scratch, [12 + slots][channels][3][2]
|
||||
std::vector<Complex> next_high_history_; // scratch, [6][channels][61]
|
||||
};
|
||||
|
||||
// Instantaneous sparse 77-hybrid to 64-QMF synthesis map.
|
||||
class HybridSynthesis {
|
||||
public:
|
||||
explicit HybridSynthesis(const Kernels& kernels) {
|
||||
const std::size_t rows = kernels.hybrid_indices.size() / 4u;
|
||||
mapping_.reserve(rows);
|
||||
for (std::size_t index = 0u; index < rows; ++index) {
|
||||
Entry entry;
|
||||
for (int field = 0; field < 4; ++field) {
|
||||
entry.index[static_cast<std::size_t>(field)] =
|
||||
kernels.hybrid_indices[index * 4u + static_cast<std::size_t>(field)];
|
||||
}
|
||||
entry.gain = kernels.hybrid_values[index];
|
||||
mapping_.push_back(entry);
|
||||
}
|
||||
}
|
||||
|
||||
// hybrid: [slots, channels, 77]; output: [slots, channels, 64] complex.
|
||||
// The sparse map moves a real or imaginary part of one band into a real or
|
||||
// imaginary part of another, so the two components are accumulated apart.
|
||||
void process(const std::vector<Complex>& hybrid, std::size_t slots, std::size_t channels,
|
||||
std::vector<Complex>* output) const {
|
||||
const std::size_t rows = slots * channels;
|
||||
std::vector<double> real(rows * kQmfBands, 0.0);
|
||||
std::vector<double> imaginary(rows * kQmfBands, 0.0);
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels; ++channel) {
|
||||
const std::size_t row = slot * channels + channel;
|
||||
const Complex* source = hybrid.data() + row * kHybridBands;
|
||||
for (const Entry& entry : mapping_) {
|
||||
const double value = entry.index[1] == 0u ? source[entry.index[0]].real()
|
||||
: source[entry.index[0]].imag();
|
||||
if (value == 0.0) {
|
||||
continue;
|
||||
}
|
||||
double* destination =
|
||||
(entry.index[3] == 0u ? real.data() : imaginary.data()) + row * kQmfBands;
|
||||
destination[entry.index[2]] += value * entry.gain;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Every output element is assigned from the two accumulators below, so the
|
||||
// zero fill that `assign` performed was dead; only the size is needed.
|
||||
output->resize(rows * kQmfBands);
|
||||
for (std::size_t index = 0u; index < output->size(); ++index) {
|
||||
(*output)[index] = Complex(real[index], imaginary[index]);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
struct Entry {
|
||||
std::size_t index[4] = {0u, 0u, 0u, 0u};
|
||||
double gain = 0.0;
|
||||
};
|
||||
std::vector<Entry> mapping_;
|
||||
};
|
||||
|
||||
// Rank-4 64-band synthesis.
|
||||
class QmfSynthesis {
|
||||
public:
|
||||
QmfSynthesis(const Kernels& kernels, std::size_t channels)
|
||||
: channels_(channels), basis_(kernels.qmf_basis), taps_(kernels.qmf_taps) {
|
||||
history_.assign(9u * channels_ * kQmfBands * kSynthesisRank, 0.0);
|
||||
// The dispatched basis kernel reads the four ranks of one (band, tap) as
|
||||
// one vector, so the shipped [band][rank][tap] table is reordered once
|
||||
// here. The weights are the same doubles, only their order differs.
|
||||
const std::size_t bands = static_cast<std::size_t>(kQmfBands);
|
||||
const std::size_t ranks = static_cast<std::size_t>(kSynthesisRank);
|
||||
const std::size_t taps = dsp::kQmfFftSize;
|
||||
basis_by_tap_.resize(bands * taps * ranks);
|
||||
for (std::size_t band = 0u; band < bands; ++band) {
|
||||
for (std::size_t tap = 0u; tap < taps; ++tap) {
|
||||
for (std::size_t rank = 0u; rank < ranks; ++rank) {
|
||||
basis_by_tap_[(band * taps + tap) * ranks + rank] =
|
||||
basis_[(band * ranks + rank) * taps + tap];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void reset() { std::fill(history_.begin(), history_.end(), 0.0); }
|
||||
|
||||
// qmf: [slots, channels, 64]; output: [slots*64, channels] real.
|
||||
void process(const std::vector<Complex>& qmf, std::size_t slots, std::vector<double>* output) {
|
||||
const std::size_t rows = slots * channels_;
|
||||
// [row][band][component] staging for the basis application. Both staging
|
||||
// planes and the joined window are reusable scratch: every element of each is
|
||||
// written before it is read, so the buffers are sized once and kept instead of
|
||||
// being allocated and zero-filled on every call.
|
||||
const std::size_t flat_size = rows * dsp::kQmfFftSize;
|
||||
if (flat_.size() < flat_size) {
|
||||
flat_.resize(flat_size);
|
||||
}
|
||||
for (std::size_t row = 0u; row < rows; ++row) {
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
flat_[row * dsp::kQmfFftSize + static_cast<std::size_t>(band) * 2u] =
|
||||
qmf[row * kQmfBands + static_cast<std::size_t>(band)].real();
|
||||
flat_[row * dsp::kQmfFftSize + static_cast<std::size_t>(band) * 2u + 1u] =
|
||||
qmf[row * kQmfBands + static_cast<std::size_t>(band)].imag();
|
||||
}
|
||||
}
|
||||
// The sums are written straight into the joined window: the destination index
|
||||
// is known up front, the summation order is untouched, and the application
|
||||
// itself is dispatched -- the four ranks of a band are four independent dot
|
||||
// products over the same 128 values, so they share a vector while every lane
|
||||
// keeps the tap order and the two roundings of `sum +=`.
|
||||
const std::size_t history_size = 9u * channels_ * kQmfBands * kSynthesisRank;
|
||||
const std::size_t joined_size = history_size + rows * kQmfBands * kSynthesisRank;
|
||||
if (joined_.size() < joined_size) {
|
||||
joined_.resize(joined_size);
|
||||
}
|
||||
std::copy(history_.begin(), history_.end(), joined_.begin());
|
||||
simd::qmf_synthesis_basis(flat_.data(), basis_by_tap_.data(),
|
||||
joined_.data() + history_size, rows);
|
||||
std::copy(joined_.begin() + static_cast<std::ptrdiff_t>(joined_size - history_size),
|
||||
joined_.begin() + static_cast<std::ptrdiff_t>(joined_size), history_.begin());
|
||||
|
||||
output->assign(rows * kQmfBands, 0.0);
|
||||
for (int lag = 0; lag < kSynthesisTaps; ++lag) {
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
const std::size_t source_slot = 9u - static_cast<std::size_t>(lag) + slot;
|
||||
for (std::size_t channel = 0u; channel < channels_; ++channel) {
|
||||
const double* source =
|
||||
joined_.data() +
|
||||
(source_slot * channels_ + channel) * kQmfBands * kSynthesisRank;
|
||||
double* destination =
|
||||
output->data() + (slot * channels_ + channel) * kQmfBands;
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
double sum = 0.0;
|
||||
for (int rank = 0; rank < kSynthesisRank; ++rank) {
|
||||
sum += source[static_cast<std::size_t>(band) * kSynthesisRank +
|
||||
static_cast<std::size_t>(rank)] *
|
||||
taps_[(static_cast<std::size_t>(band) * kSynthesisTaps +
|
||||
static_cast<std::size_t>(lag)) * kSynthesisRank +
|
||||
static_cast<std::size_t>(rank)];
|
||||
}
|
||||
destination[band] += sum;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
std::size_t channels_;
|
||||
std::vector<double> basis_; // [64][4][128]
|
||||
std::vector<double> basis_by_tap_; // [64][128][4], the same weights transposed
|
||||
std::vector<double> taps_; // [64][10][4]
|
||||
std::vector<double> history_; // [9][channels][64][4]
|
||||
std::vector<double> flat_; // scratch, [rows][128], fully written per call
|
||||
std::vector<double> joined_; // scratch, [9 + slots][channels][64][4]
|
||||
};
|
||||
|
||||
// public_filterbank.PublicAnalysis77.process: [N, channels] -> [N/64, channels, 77].
|
||||
void analysis_77(const std::vector<double>& samples, std::size_t slots,
|
||||
std::size_t channels, QmfAnalysis& qmf,
|
||||
HybridAnalysis& hybrid_analysis, std::vector<Complex>* hybrid) {
|
||||
std::vector<double> hops(slots * channels * kQmfHop, 0.0);
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels; ++channel) {
|
||||
for (int index = 0; index < kQmfHop; ++index) {
|
||||
hops[(slot * channels + channel) * kQmfHop + static_cast<std::size_t>(index)] =
|
||||
samples[(slot * kQmfHop + static_cast<std::size_t>(index)) * channels + channel];
|
||||
}
|
||||
}
|
||||
}
|
||||
std::vector<Complex> qmf_bands;
|
||||
qmf.process(hops, slots, &qmf_bands);
|
||||
hybrid_analysis.process(qmf_bands, slots, hybrid);
|
||||
}
|
||||
|
||||
// public_filterbank.PublicSynthesis77.process: [slots, channels, 77] -> [slots*64, channels].
|
||||
void synthesis_77(const std::vector<Complex>& hybrid, std::size_t slots,
|
||||
std::size_t channels, const HybridSynthesis& synthesis,
|
||||
QmfSynthesis& qmf, std::vector<double>* time) {
|
||||
std::vector<Complex> qmf_bands;
|
||||
synthesis.process(hybrid, slots, channels, &qmf_bands);
|
||||
std::vector<double> samples;
|
||||
qmf.process(qmf_bands, slots, &samples);
|
||||
// The reference transposes (slots, channels, 64) to sample-major output.
|
||||
time->assign(samples.size(), 0.0);
|
||||
for (std::size_t slot = 0u; slot < slots; ++slot) {
|
||||
for (std::size_t channel = 0u; channel < channels; ++channel) {
|
||||
for (int band = 0; band < kQmfBands; ++band) {
|
||||
(*time)[(slot * kQmfHop + static_cast<std::size_t>(band)) * channels + channel] =
|
||||
samples[(slot * channels + channel) * kQmfBands + static_cast<std::size_t>(band)];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
struct PublicFilterbank::Impl {
|
||||
Impl(const Kernels& kernels, std::size_t channels)
|
||||
: channels(channels), qmf(kernels, channels), hybrid_analysis(kernels, channels),
|
||||
hybrid_synthesis(kernels), qmf_synthesis(kernels, channels) {}
|
||||
|
||||
std::size_t channels;
|
||||
QmfAnalysis qmf;
|
||||
HybridAnalysis hybrid_analysis;
|
||||
HybridSynthesis hybrid_synthesis;
|
||||
QmfSynthesis qmf_synthesis;
|
||||
};
|
||||
|
||||
PublicFilterbank::PublicFilterbank(const Kernels& kernels, std::size_t channels)
|
||||
: impl_(std::make_unique<Impl>(kernels, channels)) {}
|
||||
|
||||
PublicFilterbank::~PublicFilterbank() = default;
|
||||
|
||||
void PublicFilterbank::reset() {
|
||||
impl_->qmf.reset();
|
||||
impl_->hybrid_analysis.reset();
|
||||
impl_->qmf_synthesis.reset();
|
||||
}
|
||||
|
||||
void PublicFilterbank::analyze_full_rate(const std::vector<double>& samples, std::size_t slots,
|
||||
std::vector<Complex>* hybrid) {
|
||||
analysis_77(samples, slots, impl_->channels, impl_->qmf, impl_->hybrid_analysis, hybrid);
|
||||
}
|
||||
|
||||
void PublicFilterbank::synthesize_full_rate(const std::vector<Complex>& hybrid, std::size_t slots,
|
||||
std::vector<double>* time) {
|
||||
synthesis_77(hybrid, slots, impl_->channels, impl_->hybrid_synthesis, impl_->qmf_synthesis,
|
||||
time);
|
||||
}
|
||||
|
||||
void PublicFilterbank::analyze_qmf(const std::vector<double>& hops, std::size_t slots,
|
||||
std::vector<Complex>* qmf) {
|
||||
impl_->qmf.process(hops, slots, qmf);
|
||||
}
|
||||
|
||||
void PublicFilterbank::analyze_hybrid(const std::vector<Complex>& qmf, std::size_t slots,
|
||||
std::vector<Complex>* hybrid) {
|
||||
impl_->hybrid_analysis.process(qmf, slots, hybrid);
|
||||
}
|
||||
|
||||
void PublicFilterbank::synthesize_hybrid(const std::vector<Complex>& hybrid, std::size_t slots,
|
||||
std::vector<Complex>* qmf) {
|
||||
impl_->hybrid_synthesis.process(hybrid, slots, impl_->channels, qmf);
|
||||
}
|
||||
|
||||
void PublicFilterbank::synthesize_qmf(const std::vector<Complex>& qmf, std::size_t slots,
|
||||
std::vector<double>* time) {
|
||||
impl_->qmf_synthesis.process(qmf, slots, time);
|
||||
}
|
||||
|
||||
} // namespace joc::hrtf
|
||||
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
#pragma once
|
||||
|
||||
#include <complex>
|
||||
#include <cstddef>
|
||||
#include <memory>
|
||||
#include <vector>
|
||||
|
||||
#include "hrtf/jochrtf.h"
|
||||
|
||||
// Public 64-QMF / 77-hybrid filterbank, shared by the SOFA field compiler and the
|
||||
// Rosella renderer (upstream public_filterbank.py and rosella_filterbank.py are
|
||||
// the same bank). Everything is float64/complex128, as the reference computes it,
|
||||
// and the stateful half-steps are exposed because Rosella drives them directly.
|
||||
namespace joc::hrtf {
|
||||
|
||||
inline constexpr int kQmfBands = 64;
|
||||
inline constexpr int kQmfHop = 64;
|
||||
inline constexpr int kHybridLow = 16;
|
||||
inline constexpr int kHybridBandCount = 77;
|
||||
inline constexpr int kLatencySamples = 961;
|
||||
|
||||
using Complex = std::complex<double>;
|
||||
|
||||
class PublicFilterbank {
|
||||
public:
|
||||
PublicFilterbank(const Kernels& kernels, std::size_t channels);
|
||||
~PublicFilterbank();
|
||||
PublicFilterbank(const PublicFilterbank&) = delete;
|
||||
PublicFilterbank& operator=(const PublicFilterbank&) = delete;
|
||||
|
||||
void reset();
|
||||
|
||||
// Full-rate [slots*64, channels] -> hybrid [slots, channels, 77].
|
||||
void analyze_full_rate(const std::vector<double>& samples, std::size_t slots,
|
||||
std::vector<Complex>* hybrid);
|
||||
// Hybrid [slots, channels, 77] -> full-rate [slots*64, channels].
|
||||
void synthesize_full_rate(const std::vector<Complex>& hybrid, std::size_t slots,
|
||||
std::vector<double>* time);
|
||||
|
||||
// The stateful half-steps, in the order the reference runs them.
|
||||
void analyze_qmf(const std::vector<double>& hops, std::size_t slots,
|
||||
std::vector<Complex>* qmf);
|
||||
void analyze_hybrid(const std::vector<Complex>& qmf, std::size_t slots,
|
||||
std::vector<Complex>* hybrid);
|
||||
void synthesize_hybrid(const std::vector<Complex>& hybrid, std::size_t slots,
|
||||
std::vector<Complex>* qmf);
|
||||
void synthesize_qmf(const std::vector<Complex>& qmf, std::size_t slots,
|
||||
std::vector<double>* time);
|
||||
|
||||
private:
|
||||
struct Impl;
|
||||
std::unique_ptr<Impl> impl_;
|
||||
};
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -0,0 +1,535 @@
|
||||
#include "hrtf/rosella_model.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/fs_utf8.h"
|
||||
#include "foundation/mini_json.h"
|
||||
#include "foundation/sha256.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
namespace {
|
||||
|
||||
// The model's fixed-point lane scale: every stored value is a Q15 integer.
|
||||
constexpr float kQ15 = 1.0f / 32768.0f;
|
||||
|
||||
Status model_fail(joc_error code, const std::string& message) {
|
||||
return Status::fail(code, stage::kRender, message);
|
||||
}
|
||||
|
||||
float q15(std::int32_t value) { return static_cast<float>(value) * kQ15; }
|
||||
|
||||
float q15_exp(std::int32_t value, int exponent) {
|
||||
return q15(value) * static_cast<float>(std::ldexp(1.0, exponent));
|
||||
}
|
||||
|
||||
std::uint16_t low16(std::int32_t value) {
|
||||
return static_cast<std::uint16_t>(static_cast<std::uint32_t>(value) & 0xFFFFu);
|
||||
}
|
||||
|
||||
std::string trim(const std::string& text) {
|
||||
const std::size_t begin = text.find_first_not_of(" \t\r\n");
|
||||
const std::size_t end = text.find_last_not_of(" \t\r\n");
|
||||
return begin == std::string::npos ? std::string() : text.substr(begin, end - begin + 1u);
|
||||
}
|
||||
|
||||
// The lane array is read straight out of the JSON text: it is one flat list of
|
||||
// integers, and building a 15691-node DOM for it would only cost time.
|
||||
bool parse_int_array(const std::string& raw, std::vector<std::int32_t>* out, std::string* error) {
|
||||
out->clear();
|
||||
const char* cursor = raw.c_str();
|
||||
const char* end = cursor + raw.size();
|
||||
while (cursor < end && *cursor != '[') {
|
||||
++cursor;
|
||||
}
|
||||
if (cursor == end) {
|
||||
*error = "rosella_coefficients must be a JSON array";
|
||||
return false;
|
||||
}
|
||||
++cursor;
|
||||
while (cursor < end) {
|
||||
while (cursor < end && (*cursor == ' ' || *cursor == '\t' || *cursor == '\r' ||
|
||||
*cursor == '\n' || *cursor == ',')) {
|
||||
++cursor;
|
||||
}
|
||||
if (cursor >= end) {
|
||||
break;
|
||||
}
|
||||
if (*cursor == ']') {
|
||||
return true;
|
||||
}
|
||||
const bool negative = *cursor == '-';
|
||||
if (negative) {
|
||||
++cursor;
|
||||
}
|
||||
if (cursor >= end || *cursor < '0' || *cursor > '9') {
|
||||
*error = "rosella_coefficients contains a non-integer value";
|
||||
return false;
|
||||
}
|
||||
long long value = 0;
|
||||
while (cursor < end && *cursor >= '0' && *cursor <= '9') {
|
||||
value = value * 10 + (*cursor - '0');
|
||||
if (value > (1ll << 40)) {
|
||||
*error = "rosella_coefficients value is out of range";
|
||||
return false;
|
||||
}
|
||||
++cursor;
|
||||
}
|
||||
// A fractional part or an exponent means the value is not an exact integer.
|
||||
if (cursor < end && (*cursor == '.' || *cursor == 'e' || *cursor == 'E')) {
|
||||
*error = "rosella_coefficients contains a non-integer value";
|
||||
return false;
|
||||
}
|
||||
if (negative) {
|
||||
value = -value;
|
||||
}
|
||||
if (value < -(1ll << 31) || value > (1ll << 31) - 1) {
|
||||
*error = "rosella_coefficients value is outside signed int32";
|
||||
return false;
|
||||
}
|
||||
out->push_back(static_cast<std::int32_t>(value));
|
||||
}
|
||||
*error = "rosella_coefficients array is truncated";
|
||||
return false;
|
||||
}
|
||||
|
||||
struct RpHeader {
|
||||
std::uint16_t stored_checksum = 0;
|
||||
std::uint16_t computed_checksum = 0;
|
||||
bool checksum_valid = false;
|
||||
bool table_a_present = false;
|
||||
bool table_b_present = false;
|
||||
bool table_c_present = false;
|
||||
int table_a_dimension = 0;
|
||||
int table_a_option = 0;
|
||||
int table_a_extra = 0;
|
||||
int table_b_dimension = 0;
|
||||
int table_b_extra = 0;
|
||||
int table_b_groups = 0;
|
||||
int table_c_dimension = 0;
|
||||
std::size_t active_lanes = 0;
|
||||
};
|
||||
|
||||
Status inspect_rp(const std::vector<std::int32_t>& lanes, RpHeader* out) {
|
||||
if (lanes.size() < 5u) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella rp must contain whole int32 lanes");
|
||||
}
|
||||
if (low16(lanes[0]) != 0x7072u) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "bad Rosella rp magic");
|
||||
}
|
||||
out->stored_checksum = low16(lanes[1]);
|
||||
out->table_a_present = low16(lanes[2]) != 0u;
|
||||
out->table_b_present = low16(lanes[3]) != 0u;
|
||||
out->table_c_present = low16(lanes[4]) != 0u;
|
||||
std::size_t index = 5u;
|
||||
if (out->table_a_present) {
|
||||
out->table_a_dimension = low16(lanes[index]);
|
||||
out->table_a_option = low16(lanes[index + 1u]);
|
||||
out->table_a_extra = low16(lanes[index + 2u]);
|
||||
index += 5u;
|
||||
} else {
|
||||
out->table_a_dimension = 77;
|
||||
}
|
||||
if (out->table_b_present) {
|
||||
if (!out->table_a_present) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"Rosella rp table B cannot be present without table A");
|
||||
}
|
||||
out->table_b_dimension = low16(lanes[index]);
|
||||
out->table_b_extra = low16(lanes[index + 1u]);
|
||||
out->table_b_groups = low16(lanes[index + 2u]);
|
||||
index += 3u;
|
||||
}
|
||||
if (out->table_c_present) {
|
||||
out->table_c_dimension = low16(lanes[index]);
|
||||
index += 1u;
|
||||
}
|
||||
const long long payload_words =
|
||||
static_cast<long long>(index) - 2 +
|
||||
(out->table_b_present ? (out->table_b_dimension + 380 * out->table_b_groups +
|
||||
out->table_b_extra + 79)
|
||||
: 0) +
|
||||
(out->table_a_present ? (171 * out->table_a_extra + 79 +
|
||||
2 * (out->table_a_option + 14 * out->table_a_dimension))
|
||||
: 0) +
|
||||
11 + (out->table_c_present ? (314 * out->table_c_dimension + 1) : 0);
|
||||
if (payload_words < 0) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "malformed Rosella rp header");
|
||||
}
|
||||
out->active_lanes = static_cast<std::size_t>(2 + payload_words);
|
||||
if (lanes.size() < out->active_lanes) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella rp is truncated");
|
||||
}
|
||||
std::uint32_t computed = 0xA569u;
|
||||
for (std::size_t lane = 2u; lane < out->active_lanes; ++lane) {
|
||||
computed ^= low16(lanes[lane]);
|
||||
}
|
||||
out->computed_checksum = static_cast<std::uint16_t>(computed & 0xFFFFu);
|
||||
out->checksum_valid = out->computed_checksum == out->stored_checksum;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
// _unpack_field: the serialized 154-per-direction field lanes to the padded grid.
|
||||
void unpack_field(const std::int32_t* serialized, int directions, int exponent,
|
||||
std::vector<float>* padded) {
|
||||
padded->assign(static_cast<std::size_t>(160 * directions), 0.0f);
|
||||
const int stride8 = 8 * directions;
|
||||
const int stride2 = 2 * directions;
|
||||
for (int source = 0; source < 154 * directions; ++source) {
|
||||
const int group4 = (source % stride8) / stride2;
|
||||
const int destination = (group4 & 3) + 4 * (source % stride2 +
|
||||
2 * directions * (source / stride8 +
|
||||
(group4 >> 2)));
|
||||
(*padded)[static_cast<std::size_t>(destination)] =
|
||||
q15_exp(serialized[source], exponent);
|
||||
}
|
||||
}
|
||||
|
||||
// _unpack_table_a_grid: the serialized table-A rows to the padded lane grid.
|
||||
void unpack_table_a_grid(const std::int32_t* serialized, int dimension, int serialized_rows,
|
||||
int padded_rows, int lane_group, std::vector<float>* padded) {
|
||||
padded->assign(static_cast<std::size_t>(padded_rows) * static_cast<std::size_t>(dimension),
|
||||
0.0f);
|
||||
const int group_width = lane_group * 4;
|
||||
for (int source = 0; source < serialized_rows * dimension; ++source) {
|
||||
const int remainder = source % group_width;
|
||||
const int destination = (remainder / lane_group) +
|
||||
4 * (remainder % lane_group +
|
||||
group_width / 4 * (source / group_width));
|
||||
(*padded)[static_cast<std::size_t>(destination)] = q15(serialized[source]);
|
||||
}
|
||||
}
|
||||
|
||||
void unpack_table_a_extra(const std::int32_t* serialized, std::vector<float>* padded) {
|
||||
padded->assign(160u, 0.0f);
|
||||
for (int source = 0; source < 154; ++source) {
|
||||
const int remainder = source & 7;
|
||||
const int destination = (remainder >> 1) + 4 * ((source & 1) + 2 * (source >> 3));
|
||||
(*padded)[static_cast<std::size_t>(destination)] = q15(serialized[source]);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::string RosellaModel::summary() const {
|
||||
std::string name = capture.name.empty() ? std::string("unnamed") : capture.name;
|
||||
return "Rosella personalized_headphone '" + name + "' (" +
|
||||
(room_model.empty() ? std::string("unknown room") : room_model) + "), " +
|
||||
std::to_string(table_a_dimension) + " HQMF / 77 hybrid @ " +
|
||||
std::to_string(sample_rate) + " Hz";
|
||||
}
|
||||
|
||||
Status load_personalized_headphone(const std::string& path, RosellaModel* out) {
|
||||
if (out == nullptr) {
|
||||
return model_fail(JOC_ERR_INVALID_ARGUMENT, "null Rosella model destination");
|
||||
}
|
||||
if (!fs_utf8::exists(path)) {
|
||||
return model_fail(JOC_ERR_HRTF_NOT_FOUND, "personalized headphone model not found: " + path);
|
||||
}
|
||||
std::ifstream stream = fs_utf8::open_input(path);
|
||||
if (!stream.good()) {
|
||||
return model_fail(JOC_ERR_IO, "cannot open " + path);
|
||||
}
|
||||
std::string text((std::istreambuf_iterator<char>(stream)), std::istreambuf_iterator<char>());
|
||||
if (text.empty()) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "empty personalized headphone model: " + path);
|
||||
}
|
||||
// The checksum is taken over the coefficient lanes, exactly as upstream hashes
|
||||
// the int32 image of the array.
|
||||
const std::size_t first = text.find_first_not_of(" \t\r\n");
|
||||
if (first == std::string::npos || text[first] != '{') {
|
||||
return model_fail(JOC_ERR_NOT_SUPPORTED,
|
||||
"raw rp models are not supported; use a .personalized_headphone JSON");
|
||||
}
|
||||
std::vector<json::Member> root;
|
||||
std::string error;
|
||||
if (!json::parse_object(text, &root, &error)) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "invalid personalized headphone JSON: " + error);
|
||||
}
|
||||
const json::Member* personalized = json::find(root, "personalized_hrtf");
|
||||
if (personalized == nullptr) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "personalized_hrtf is missing");
|
||||
}
|
||||
std::vector<json::Member> inner;
|
||||
if (!json::parse_object(personalized->raw, &inner, &error)) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "invalid personalized_hrtf object: " + error);
|
||||
}
|
||||
const json::Member* virtualizer = json::find(inner, "virtualizer_parameters");
|
||||
if (virtualizer == nullptr) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "virtualizer_parameters is missing");
|
||||
}
|
||||
std::vector<json::Member> parameters;
|
||||
if (!json::parse_object(virtualizer->raw, ¶meters, &error)) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "invalid virtualizer_parameters: " + error);
|
||||
}
|
||||
const json::Member* coefficient_member = json::find(parameters, "rosella_coefficients");
|
||||
if (coefficient_member == nullptr) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "rosella_coefficients is missing");
|
||||
}
|
||||
RosellaModel model;
|
||||
model.source_path = path;
|
||||
if (const json::Member* member = json::find(parameters, "rosella_coefficients_version")) {
|
||||
json::as_string(*member, &model.coefficient_version);
|
||||
}
|
||||
if (const json::Member* member = json::find(parameters, "room_model")) {
|
||||
json::as_string(*member, &model.room_model);
|
||||
}
|
||||
if (const json::Member* capture = json::find(inner, "phrtf_capture_metadata")) {
|
||||
std::vector<json::Member> fields;
|
||||
if (json::parse_object(capture->raw, &fields, &error)) {
|
||||
const std::pair<const char*, std::string*> mapping[] = {
|
||||
{"capture_submission_date", &model.capture.capture_submission_date},
|
||||
{"capture_type", &model.capture.capture_type},
|
||||
{"label", &model.capture.label},
|
||||
{"name", &model.capture.name},
|
||||
{"phrtf_algorithm_version", &model.capture.algorithm_version},
|
||||
{"phrtf_creation_date", &model.capture.creation_date},
|
||||
{"uuid", &model.capture.uuid},
|
||||
{"version", &model.capture.version},
|
||||
};
|
||||
for (const auto& entry : mapping) {
|
||||
if (const json::Member* member = json::find(fields, entry.first)) {
|
||||
json::as_string(*member, entry.second);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
std::vector<std::int32_t> lanes;
|
||||
if (!parse_int_array(coefficient_member->raw, &lanes, &error)) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, error);
|
||||
}
|
||||
{
|
||||
crypto::Sha256 hash;
|
||||
hash.update(lanes.data(), lanes.size() * sizeof(std::int32_t));
|
||||
model.coefficient_sha256 = hash.finish_hex();
|
||||
}
|
||||
|
||||
RpHeader header;
|
||||
Status status = inspect_rp(lanes, &header);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (!header.checksum_valid || header.active_lanes != lanes.size()) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"invalid or non-active Rosella rp coefficient sequence");
|
||||
}
|
||||
if (!header.table_a_present || !header.table_b_present || header.table_c_present) {
|
||||
return model_fail(JOC_ERR_NOT_SUPPORTED,
|
||||
"the renderer requires table A+B and no table C");
|
||||
}
|
||||
if (header.table_a_dimension != 64 || header.table_a_option != 3) {
|
||||
return model_fail(JOC_ERR_NOT_SUPPORTED,
|
||||
"the renderer requires the observed 64-channel HQMF layout");
|
||||
}
|
||||
if (header.table_b_dimension != 20 || header.table_b_groups != 36) {
|
||||
return model_fail(JOC_ERR_NOT_SUPPORTED,
|
||||
"the renderer requires 20 hybrid groups and 36 direction terms");
|
||||
}
|
||||
|
||||
const std::int32_t* values = lanes.data();
|
||||
const std::size_t total = lanes.size();
|
||||
std::size_t position = 13u;
|
||||
const int extra = header.table_a_extra;
|
||||
model.table_a_dimension = header.table_a_dimension;
|
||||
model.table_a_option = header.table_a_option;
|
||||
model.table_a_extra = extra;
|
||||
model.table_a_header_field = low16(values[8]);
|
||||
model.table_a_header_25 = low16(values[9]);
|
||||
model.table_a_control = low16(values[position]);
|
||||
model.field_exponent = values[position];
|
||||
position += 1u;
|
||||
const int option_count = header.table_a_option;
|
||||
model.table_a_option_ids.resize(static_cast<std::size_t>(option_count));
|
||||
for (int index = 0; index < option_count; ++index) {
|
||||
model.table_a_option_ids[static_cast<std::size_t>(index)] =
|
||||
low16(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += static_cast<std::size_t>(option_count);
|
||||
model.table_a_option_values.resize(static_cast<std::size_t>(option_count));
|
||||
for (int index = 0; index < option_count; ++index) {
|
||||
model.table_a_option_values[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += static_cast<std::size_t>(option_count);
|
||||
model.table_a_scalar = q15(values[position]);
|
||||
position += 1u;
|
||||
const int dimension = header.table_a_dimension;
|
||||
unpack_table_a_grid(values + position, dimension, 16, 20, 16,
|
||||
&model.table_a_filter_16x64_padded);
|
||||
position += static_cast<std::size_t>(16 * dimension);
|
||||
for (int index = 0; index < 4; ++index) {
|
||||
model.table_a_four_integers[static_cast<std::size_t>(index)] =
|
||||
low16(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += 4u;
|
||||
model.table_a_integer = low16(values[position]);
|
||||
position += 1u;
|
||||
unpack_table_a_grid(values + position, dimension, 8, 10, 8,
|
||||
&model.table_a_filter_8x64_padded);
|
||||
position += static_cast<std::size_t>(8 * dimension);
|
||||
model.table_a_vector16.resize(16u);
|
||||
for (int index = 0; index < 16; ++index) {
|
||||
model.table_a_vector16[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += 16u;
|
||||
unpack_table_a_grid(values + position, dimension, 4, 5, 4,
|
||||
&model.table_a_filter_4x64_padded);
|
||||
position += static_cast<std::size_t>(4 * dimension);
|
||||
model.table_a_extra_indices.resize(static_cast<std::size_t>(extra));
|
||||
for (int index = 0; index < extra; ++index) {
|
||||
model.table_a_extra_indices[static_cast<std::size_t>(index)] =
|
||||
low16(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += static_cast<std::size_t>(extra);
|
||||
model.table_a_extra_fields_padded.assign(static_cast<std::size_t>(extra) * 160u, 0.0f);
|
||||
std::vector<float> unpacked;
|
||||
for (int index = 0; index < extra; ++index) {
|
||||
unpack_table_a_extra(values + position, &unpacked);
|
||||
std::copy(unpacked.begin(), unpacked.end(),
|
||||
model.table_a_extra_fields_padded.begin() + static_cast<std::ptrdiff_t>(index) * 160);
|
||||
position += 154u;
|
||||
}
|
||||
model.table_a_extra_vectors.assign(static_cast<std::size_t>(extra) * 16u, 0.0f);
|
||||
for (int index = 0; index < extra; ++index) {
|
||||
for (int lane = 0; lane < 16; ++lane) {
|
||||
model.table_a_extra_vectors[static_cast<std::size_t>(index) * 16u +
|
||||
static_cast<std::size_t>(lane)] =
|
||||
q15(values[position + static_cast<std::size_t>(lane)]);
|
||||
}
|
||||
position += 16u;
|
||||
}
|
||||
const std::size_t table_b_start = position;
|
||||
if (table_b_start != 13u + 1821u + static_cast<std::size_t>(171 * extra)) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella table-A parser lost its place");
|
||||
}
|
||||
|
||||
model.sample_rate = 2 * low16(values[position]);
|
||||
position += 1u;
|
||||
model.matrix_exponent = values[position];
|
||||
position += 1u;
|
||||
const std::size_t matrix_count = 36u * 36u;
|
||||
model.matrix_left.resize(matrix_count);
|
||||
model.matrix_right.resize(matrix_count);
|
||||
for (std::size_t index = 0; index < matrix_count; ++index) {
|
||||
model.matrix_left[index] = q15_exp(values[position + index], model.matrix_exponent);
|
||||
}
|
||||
position += matrix_count;
|
||||
for (std::size_t index = 0; index < matrix_count; ++index) {
|
||||
model.matrix_right[index] = q15_exp(values[position + index], model.matrix_exponent);
|
||||
}
|
||||
position += matrix_count;
|
||||
model.vector_left.resize(36u);
|
||||
model.vector_right.resize(36u);
|
||||
for (int index = 0; index < 36; ++index) {
|
||||
model.vector_left[static_cast<std::size_t>(index)] =
|
||||
q15_exp(values[position + static_cast<std::size_t>(index)], model.matrix_exponent);
|
||||
}
|
||||
position += 36u;
|
||||
for (int index = 0; index < 36; ++index) {
|
||||
model.vector_right[static_cast<std::size_t>(index)] =
|
||||
q15_exp(values[position + static_cast<std::size_t>(index)], model.matrix_exponent);
|
||||
}
|
||||
position += 36u;
|
||||
|
||||
const std::size_t serialized_count = 154u * 36u;
|
||||
unpack_field(values + position, 36, model.field_exponent, &model.field_left_padded);
|
||||
bool odd_zero = true;
|
||||
for (std::size_t index = 1u; index < serialized_count; index += 2u) {
|
||||
const float value = q15_exp(values[position + index], model.field_exponent);
|
||||
if (std::abs(value) > 1.0e-6f) {
|
||||
odd_zero = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
model.field_left_odd_serialized_zero = odd_zero;
|
||||
position += serialized_count;
|
||||
unpack_field(values + position, 36, model.field_exponent, &model.field_right_padded);
|
||||
position += serialized_count;
|
||||
|
||||
model.hybrid_flags.resize(20u);
|
||||
int active_hybrid = 0;
|
||||
for (int index = 0; index < 20; ++index) {
|
||||
model.hybrid_flags[static_cast<std::size_t>(index)] =
|
||||
low16(values[position + static_cast<std::size_t>(index)]);
|
||||
if (model.hybrid_flags[static_cast<std::size_t>(index)] == 1) {
|
||||
++active_hybrid;
|
||||
}
|
||||
}
|
||||
position += 20u;
|
||||
if (active_hybrid != header.table_b_extra) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "hybrid value count does not match the header");
|
||||
}
|
||||
model.hybrid_values.resize(static_cast<std::size_t>(active_hybrid));
|
||||
for (int index = 0; index < active_hybrid; ++index) {
|
||||
model.hybrid_values[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += static_cast<std::size_t>(active_hybrid);
|
||||
model.model_scalars.resize(5u);
|
||||
for (int index = 0; index < 5; ++index) {
|
||||
model.model_scalars[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += 5u;
|
||||
const std::size_t expected_tail =
|
||||
table_b_start + static_cast<std::size_t>(header.table_b_dimension +
|
||||
380 * header.table_b_groups +
|
||||
header.table_b_extra + 79);
|
||||
if (position != expected_tail) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "Rosella table-B parser lost its place");
|
||||
}
|
||||
|
||||
model.header_float_scalars[0] = q15(values[position]);
|
||||
model.header_float_scalars[1] = q15(values[position + 1u]) * 16.0f;
|
||||
model.header_integer_fields[0] = values[position + 2u];
|
||||
model.header_integer_fields[1] = low16(values[position + 3u]);
|
||||
position += 4u;
|
||||
for (int profile = 0; profile < 4; ++profile) {
|
||||
RosellaDistanceProfile parsed;
|
||||
for (int index = 0; index < 6; ++index) {
|
||||
parsed.bounds[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += 6u;
|
||||
parsed.distance_scale_m =
|
||||
q15_exp(values[position], values[position + 1u]);
|
||||
position += 2u;
|
||||
parsed.inverse_distance_per_m = q15(values[position]);
|
||||
parsed.axis_scales_internal[0] = q15(values[position + 1u]);
|
||||
parsed.axis_scales_internal[1] = q15(values[position + 2u]);
|
||||
parsed.axis_scales_internal[2] = q15(values[position + 3u]);
|
||||
parsed.minimum_normalized_radius = q15(values[position + 4u]);
|
||||
position += 5u;
|
||||
model.profiles[static_cast<std::size_t>(profile)] = parsed;
|
||||
}
|
||||
model.profile_tail.resize(8u);
|
||||
for (int index = 0; index < 8; ++index) {
|
||||
model.profile_tail[static_cast<std::size_t>(index)] =
|
||||
q15(values[position + static_cast<std::size_t>(index)]);
|
||||
}
|
||||
position += 8u;
|
||||
for (int index = 0; index < 3; ++index) {
|
||||
model.post_fields[static_cast<std::size_t>(index)] =
|
||||
values[position + static_cast<std::size_t>(index)];
|
||||
}
|
||||
position += 3u;
|
||||
if (position != total) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT, "unparsed Rosella coefficient lanes");
|
||||
}
|
||||
if (model.sample_rate != 48000) {
|
||||
return model_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"Rosella model sample rate must be 48000, got " +
|
||||
std::to_string(model.sample_rate));
|
||||
}
|
||||
*out = std::move(model);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -0,0 +1,88 @@
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
// Parser for the Dolby ".personalized_headphone" model (upstream rosella_model.py).
|
||||
// The file is JSON whose virtualizer_parameters carry the raw "rp" coefficient
|
||||
// lanes; everything the renderer needs is unpacked here, in the same float32
|
||||
// arithmetic the reference uses, because those values are part of the model.
|
||||
namespace joc::hrtf {
|
||||
|
||||
struct RosellaDistanceProfile {
|
||||
std::array<float, 6> bounds{};
|
||||
float distance_scale_m = 0.0f;
|
||||
float inverse_distance_per_m = 0.0f;
|
||||
std::array<float, 3> axis_scales_internal{};
|
||||
float minimum_normalized_radius = 0.0f;
|
||||
};
|
||||
|
||||
struct RosellaCaptureMetadata {
|
||||
std::string capture_submission_date;
|
||||
std::string capture_type;
|
||||
std::string label;
|
||||
std::string name;
|
||||
std::string algorithm_version;
|
||||
std::string creation_date;
|
||||
std::string uuid;
|
||||
std::string version;
|
||||
};
|
||||
|
||||
struct RosellaModel {
|
||||
std::string source_path;
|
||||
std::string coefficient_sha256;
|
||||
std::string coefficient_version;
|
||||
std::string room_model;
|
||||
RosellaCaptureMetadata capture;
|
||||
|
||||
int table_a_dimension = 0;
|
||||
int table_a_option = 0;
|
||||
int table_a_extra = 0;
|
||||
int table_a_header_field = 0;
|
||||
int table_a_header_25 = 0;
|
||||
int table_a_control = 0;
|
||||
std::vector<int> table_a_option_ids;
|
||||
std::vector<float> table_a_option_values;
|
||||
float table_a_scalar = 0.0f;
|
||||
std::vector<float> table_a_filter_16x64_padded;
|
||||
std::array<int, 4> table_a_four_integers{};
|
||||
int table_a_integer = 0;
|
||||
std::vector<float> table_a_filter_8x64_padded;
|
||||
std::vector<float> table_a_vector16;
|
||||
std::vector<float> table_a_filter_4x64_padded;
|
||||
std::vector<int> table_a_extra_indices;
|
||||
std::vector<float> table_a_extra_fields_padded;
|
||||
std::vector<float> table_a_extra_vectors;
|
||||
|
||||
int sample_rate = 0;
|
||||
int matrix_exponent = 0;
|
||||
int field_exponent = 0;
|
||||
std::vector<float> matrix_left;
|
||||
std::vector<float> matrix_right;
|
||||
std::vector<float> vector_left;
|
||||
std::vector<float> vector_right;
|
||||
std::vector<float> field_left_padded;
|
||||
std::vector<float> field_right_padded;
|
||||
bool field_left_odd_serialized_zero = false;
|
||||
std::vector<int> hybrid_flags;
|
||||
std::vector<float> hybrid_values;
|
||||
std::vector<float> model_scalars;
|
||||
std::array<float, 2> header_float_scalars{};
|
||||
std::array<int, 2> header_integer_fields{};
|
||||
std::array<RosellaDistanceProfile, 4> profiles{};
|
||||
std::vector<float> profile_tail;
|
||||
std::array<int, 3> post_fields{};
|
||||
|
||||
// One line for reports and logs: the capture name and room model are the
|
||||
// model's own strings, followed by the table layout and sample rate, e.g.
|
||||
// "Rosella personalized_headphone '<name>' (<room>), <N> HQMF / 77 hybrid @ <rate> Hz".
|
||||
std::string summary() const;
|
||||
};
|
||||
|
||||
Status load_personalized_headphone(const std::string& path, RosellaModel* out);
|
||||
|
||||
} // namespace joc::hrtf
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,64 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "hrtf/rosella_model.h"
|
||||
#include "oamd/oamd_parser.h"
|
||||
#include "timeline/position_timeline.h"
|
||||
|
||||
// Rosella ".personalized_headphone" binaural renderer (upstream rosella_core.py,
|
||||
// rosella_direct.py, rosella_room.py and rosella_binaural_renderer.py). It takes
|
||||
// the same pipeline slot as the SOFA runtime: sixteen object channels per frame in,
|
||||
// interleaved stereo out, with the OAMD timeline driving the per-block parameters.
|
||||
namespace joc::hrtf {
|
||||
|
||||
// rosella_direct.BINAURAL_PROFILE_NAMES.
|
||||
enum class RosellaProfile : std::int32_t { Near = 1, Far = 2, Mid = 3 };
|
||||
|
||||
struct RosellaRenderOptions {
|
||||
RosellaProfile profile = RosellaProfile::Mid;
|
||||
std::int64_t object_delay_samples = 1473;
|
||||
double tail_seconds = 5.0;
|
||||
double output_gain = 1.0;
|
||||
int chunk_frames = 64;
|
||||
int room_impulse_slots = 4096;
|
||||
};
|
||||
|
||||
class RosellaRuntime {
|
||||
public:
|
||||
RosellaRuntime();
|
||||
~RosellaRuntime();
|
||||
RosellaRuntime(const RosellaRuntime&) = delete;
|
||||
RosellaRuntime& operator=(const RosellaRuntime&) = delete;
|
||||
|
||||
Status open(const RosellaModel& model, const RosellaRenderOptions& options);
|
||||
|
||||
// objects16_planar is channel-major: channel * 1536 + sample.
|
||||
Status submit_frame(const float* objects16_planar, const oamd::OamdUpdate* update,
|
||||
std::int64_t frame_index, std::int64_t outer_sample_offset,
|
||||
std::int64_t object_delay_samples);
|
||||
|
||||
// Drains the flush tail: the pending partial chunk plus flush_samples of silence.
|
||||
Status finish(std::uint32_t flush_samples, std::vector<double>* out);
|
||||
std::uint32_t finish_capacity(double tail_seconds) const;
|
||||
|
||||
Status reset();
|
||||
|
||||
const std::vector<double>& output() const;
|
||||
void take_output(std::vector<double>* out);
|
||||
|
||||
std::uint64_t input_samples() const;
|
||||
std::uint64_t processed_input_samples() const;
|
||||
std::uint64_t metadata_block_updates() const;
|
||||
const timeline::OamdPositionTimeline& timeline() const;
|
||||
|
||||
private:
|
||||
struct Impl;
|
||||
std::unique_ptr<Impl> impl_;
|
||||
};
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -0,0 +1,255 @@
|
||||
#include "hrtf/sofa.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cctype>
|
||||
#include <cstdio>
|
||||
#include <fstream>
|
||||
#include <utility>
|
||||
|
||||
#include "foundation/fs_utf8.h"
|
||||
#include "foundation/sha256.h"
|
||||
#include "io/hdf5.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
namespace {
|
||||
|
||||
Status sofa_fail(joc_error code, const std::string& message) {
|
||||
return Status::fail(code, stage::kRender, message);
|
||||
}
|
||||
|
||||
std::string format_number(double value) {
|
||||
if (std::isfinite(value) && value == std::floor(value) && std::fabs(value) < 1.0e15) {
|
||||
return std::to_string(static_cast<long long>(value));
|
||||
}
|
||||
char buffer[32];
|
||||
std::snprintf(buffer, sizeof(buffer), "%.6g", value);
|
||||
return std::string(buffer);
|
||||
}
|
||||
|
||||
// Every array is checked against the element count the convention prescribes, so
|
||||
// a file whose shape disagrees with its metadata is rejected instead of silently
|
||||
// producing a shifted impulse response.
|
||||
Status read_doubles(const io::Hdf5File& file, const std::string& path, std::uint64_t expected,
|
||||
std::vector<double>* out) {
|
||||
if (!file.has_dataset(path)) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA file has no " + path + " dataset");
|
||||
}
|
||||
const Status status = file.read_dataset_double(path, out);
|
||||
if (!status.ok()) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA dataset " + path + ": " + status.message());
|
||||
}
|
||||
if (out->size() != expected) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"SOFA dataset " + path + " holds " + std::to_string(out->size()) +
|
||||
" values, expected " + std::to_string(expected));
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status read_text(const io::Hdf5File& file, const std::string& name, bool required,
|
||||
std::string* out) {
|
||||
io::Hdf5Attribute attribute;
|
||||
const Status status = file.attribute("", name, &attribute);
|
||||
if (!status.ok()) {
|
||||
if (required) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"SOFA file has no root attribute " + name + ": " + status.message());
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
*out = attribute.text;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
// SHA-256 of the whole file: the compiled-cache key is derived from it, so the
|
||||
// digest is taken over the exact bytes the parse consumed.
|
||||
std::string file_digest(const std::string& path) {
|
||||
std::ifstream stream = fs_utf8::open_input(path);
|
||||
if (!stream.good()) {
|
||||
return std::string();
|
||||
}
|
||||
crypto::Sha256 hash;
|
||||
std::vector<char> buffer(1u << 20);
|
||||
while (stream.good()) {
|
||||
stream.read(buffer.data(), static_cast<std::streamsize>(buffer.size()));
|
||||
const std::streamsize count = stream.gcount();
|
||||
if (count > 0) {
|
||||
hash.update(buffer.data(), static_cast<std::size_t>(count));
|
||||
}
|
||||
}
|
||||
return hash.finish_hex();
|
||||
}
|
||||
|
||||
// The coordinate declaration of one dataset, when the file carries it.
|
||||
void read_coordinates(const io::Hdf5File& file, const std::string& dataset,
|
||||
SofaCoordinate* out) {
|
||||
io::Hdf5Attribute attribute;
|
||||
if (file.attribute(dataset, "Type", &attribute).ok()) {
|
||||
out->type = attribute.text;
|
||||
}
|
||||
if (file.attribute(dataset, "Units", &attribute).ok()) {
|
||||
out->units = attribute.text;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::string SofaHrir::summary() const {
|
||||
return "SOFA " + sofa_conventions + ", " + std::to_string(ir_count) + " IRs x " +
|
||||
std::to_string(ir_length) + " taps @ " + format_number(sample_rate) + " Hz";
|
||||
}
|
||||
|
||||
Status load_sofa(const std::string& path, SofaHrir* out) {
|
||||
if (out == nullptr) {
|
||||
return sofa_fail(JOC_ERR_INVALID_ARGUMENT, "null SOFA destination");
|
||||
}
|
||||
if (!fs_utf8::exists(path)) {
|
||||
return sofa_fail(JOC_ERR_HRTF_NOT_FOUND, "SOFA file not found: " + path);
|
||||
}
|
||||
io::Hdf5File file;
|
||||
Status status = file.open(path);
|
||||
if (!status.ok()) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA file " + path + ": " + status.message());
|
||||
}
|
||||
|
||||
SofaHrir sofa;
|
||||
sofa.source_path = path;
|
||||
sofa.source_sha256 = file_digest(path);
|
||||
for (char& character : sofa.source_sha256) {
|
||||
character = static_cast<char>(std::toupper(static_cast<unsigned char>(character)));
|
||||
}
|
||||
status = read_text(file, "Conventions", true, &sofa.conventions);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "SOFAConventions", true, &sofa.sofa_conventions);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (sofa.conventions != "SOFA" || sofa.sofa_conventions != "SimpleFreeFieldHRIR") {
|
||||
return sofa_fail(JOC_ERR_HRTF_UNSUPPORTED_CONVENTION,
|
||||
"SOFA conventions " + sofa.conventions + "/" + sofa.sofa_conventions +
|
||||
" are not SimpleFreeFieldHRIR");
|
||||
}
|
||||
status = read_text(file, "SOFAConventionsVersion", false, &sofa.convention_version);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "Version", false, &sofa.version);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "DataType", false, &sofa.data_type);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "RoomType", false, &sofa.room_type);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "Title", false, &sofa.title);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "DatabaseName", false, &sofa.database_name);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "ListenerShortName", false, &sofa.listener_short_name);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = read_text(file, "Comment", false, &sofa.comment);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
|
||||
io::Hdf5DatasetInfo info;
|
||||
status = file.dataset_info("Data.IR", &info);
|
||||
if (!status.ok()) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT,
|
||||
"SOFA file has no usable Data.IR dataset: " + status.message());
|
||||
}
|
||||
if (info.shape.size() != 3u || info.shape[1] != 2u || info.shape[0] == 0u ||
|
||||
info.shape[2] == 0u) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA Data.IR is not shaped (M, 2, N)");
|
||||
}
|
||||
if (info.shape[0] > 0xFFFFFFFFull || info.shape[2] > 0xFFFFFFFFull) {
|
||||
return sofa_fail(JOC_ERR_HRTF_FORMAT, "SOFA Data.IR is larger than this reader accepts");
|
||||
}
|
||||
sofa.ir_count = static_cast<std::uint32_t>(info.shape[0]);
|
||||
sofa.ir_length = static_cast<std::uint32_t>(info.shape[2]);
|
||||
const std::uint64_t taps = static_cast<std::uint64_t>(sofa.ir_count) * 2u * sofa.ir_length;
|
||||
status = read_doubles(file, "Data.IR", taps, &sofa.ir);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
|
||||
std::vector<double> scalar;
|
||||
status = read_doubles(file, "Data.SamplingRate", 1u, &scalar);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
sofa.sample_rate = scalar[0];
|
||||
io::Hdf5Attribute attribute;
|
||||
if (file.attribute("Data.SamplingRate", "Units", &attribute).ok()) {
|
||||
sofa.sampling_rate_units = attribute.text;
|
||||
}
|
||||
|
||||
// Data.Delay is optional in the wild; absent means "no delay was measured".
|
||||
if (file.has_dataset("Data.Delay")) {
|
||||
std::vector<double> delay;
|
||||
status = read_doubles(file, "Data.Delay", 2u, &delay);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
sofa.delay[0] = delay[0];
|
||||
sofa.delay[1] = delay[1];
|
||||
}
|
||||
|
||||
const std::uint64_t measurements = sofa.ir_count;
|
||||
status = read_doubles(file, "SourcePosition", measurements * 3u, &sofa.source_position);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
read_coordinates(file, "SourcePosition", &sofa.source_position_coordinates);
|
||||
|
||||
std::vector<double> vector;
|
||||
status = read_doubles(file, "ListenerPosition", 3u, &vector);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::copy(vector.begin(), vector.end(), sofa.listener_position);
|
||||
read_coordinates(file, "ListenerPosition", &sofa.listener_position_coordinates);
|
||||
status = read_doubles(file, "ListenerView", 3u, &vector);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::copy(vector.begin(), vector.end(), sofa.listener_view);
|
||||
read_coordinates(file, "ListenerView", &sofa.listener_view_coordinates);
|
||||
status = read_doubles(file, "ListenerUp", 3u, &vector);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::copy(vector.begin(), vector.end(), sofa.listener_up);
|
||||
read_coordinates(file, "ListenerUp", &sofa.listener_up_coordinates);
|
||||
status = read_doubles(file, "EmitterPosition", 3u, &vector);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::copy(vector.begin(), vector.end(), sofa.emitter_position);
|
||||
read_coordinates(file, "EmitterPosition", &sofa.emitter_position_coordinates);
|
||||
status = read_doubles(file, "ReceiverPosition", 6u, &vector);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
std::copy(vector.begin(), vector.end(), sofa.receiver_position);
|
||||
read_coordinates(file, "ReceiverPosition", &sofa.receiver_position_coordinates);
|
||||
|
||||
*out = std::move(sofa);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -0,0 +1,61 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
// Coordinate declaration of one SOFA variable: Type ("spherical"/"cartesian") and
|
||||
// Units. Empty when the file does not declare them (ListenerUp inherits).
|
||||
struct SofaCoordinate {
|
||||
std::string type;
|
||||
std::string units;
|
||||
};
|
||||
|
||||
// SOFA SimpleFreeFieldHRIR as this project consumes it: the impulse responses,
|
||||
// the measurement geometry and the metadata needed to report what was loaded.
|
||||
// All angles are degrees, all distances metres, exactly as the file stores them.
|
||||
struct SofaHrir {
|
||||
double sample_rate = 0.0;
|
||||
std::uint32_t ir_count = 0; // M: number of measurements
|
||||
std::uint32_t ir_length = 0; // N: taps per impulse response
|
||||
std::vector<double> ir; // C order [M][2][N]
|
||||
double delay[2] = {0.0, 0.0};
|
||||
std::vector<double> source_position; // M*3
|
||||
double listener_position[3] = {0.0, 0.0, 0.0};
|
||||
double listener_view[3] = {1.0, 0.0, 0.0};
|
||||
double listener_up[3] = {0.0, 0.0, 1.0};
|
||||
double emitter_position[3] = {0.0, 0.0, 0.0};
|
||||
double receiver_position[6] = {0.0, 0.0, 0.0, 0.0, 0.0, 0.0};
|
||||
std::string conventions;
|
||||
std::string sofa_conventions;
|
||||
std::string convention_version;
|
||||
std::string version;
|
||||
std::string data_type;
|
||||
std::string room_type;
|
||||
std::string title;
|
||||
std::string database_name;
|
||||
std::string listener_short_name;
|
||||
std::string comment;
|
||||
std::string sampling_rate_units;
|
||||
SofaCoordinate source_position_coordinates;
|
||||
SofaCoordinate listener_position_coordinates;
|
||||
SofaCoordinate listener_view_coordinates;
|
||||
SofaCoordinate listener_up_coordinates;
|
||||
SofaCoordinate emitter_position_coordinates;
|
||||
SofaCoordinate receiver_position_coordinates;
|
||||
// Identity of the file itself, needed for the compiled-cache key.
|
||||
std::string source_path;
|
||||
std::string source_sha256;
|
||||
|
||||
// One line for reports and logs:
|
||||
// "SOFA SimpleFreeFieldHRIR, <M> IRs x <N> taps @ <rate> Hz".
|
||||
std::string summary() const;
|
||||
};
|
||||
|
||||
Status load_sofa(const std::string& path, SofaHrir* out);
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -0,0 +1,237 @@
|
||||
#include "hrtf/sofa_cache.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <filesystem>
|
||||
#include <list>
|
||||
#include <mutex>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/fs_utf8.h"
|
||||
#include "hrtf/sofa.h"
|
||||
|
||||
namespace joc::hrtf {
|
||||
|
||||
namespace {
|
||||
|
||||
namespace fs = std::filesystem;
|
||||
|
||||
// Small process-local cache: the reference keeps the last eight compiled fields.
|
||||
constexpr std::size_t kMemoryCacheEntries = 8;
|
||||
|
||||
struct MemoryEntry {
|
||||
std::string key;
|
||||
Field field;
|
||||
};
|
||||
|
||||
std::mutex& memory_mutex() {
|
||||
static std::mutex mutex;
|
||||
return mutex;
|
||||
}
|
||||
|
||||
std::list<MemoryEntry>& memory_cache() {
|
||||
static std::list<MemoryEntry> cache;
|
||||
return cache;
|
||||
}
|
||||
|
||||
bool memory_cache_get(const std::string& key, Field* out) {
|
||||
std::lock_guard<std::mutex> lock(memory_mutex());
|
||||
std::list<MemoryEntry>& cache = memory_cache();
|
||||
for (auto entry = cache.begin(); entry != cache.end(); ++entry) {
|
||||
if (entry->key == key) {
|
||||
*out = entry->field;
|
||||
cache.splice(cache.begin(), cache, entry);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void memory_cache_put(const std::string& key, const Field& field) {
|
||||
std::lock_guard<std::mutex> lock(memory_mutex());
|
||||
std::list<MemoryEntry>& cache = memory_cache();
|
||||
for (auto entry = cache.begin(); entry != cache.end(); ++entry) {
|
||||
if (entry->key == key) {
|
||||
entry->field = field;
|
||||
cache.splice(cache.begin(), cache, entry);
|
||||
return;
|
||||
}
|
||||
}
|
||||
cache.push_front(MemoryEntry{key, field});
|
||||
while (cache.size() > kMemoryCacheEntries) {
|
||||
cache.pop_back();
|
||||
}
|
||||
}
|
||||
|
||||
Status cache_fail(joc_error code, const std::string& message) {
|
||||
return Status::fail(code, stage::kRender, message);
|
||||
}
|
||||
|
||||
std::string upper(std::string text) {
|
||||
std::transform(text.begin(), text.end(), text.begin(), [](unsigned char value) {
|
||||
return static_cast<char>(std::toupper(value));
|
||||
});
|
||||
return text;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Status parse_cache_policy(const std::string& text, CachePolicy* out) {
|
||||
if (out == nullptr) {
|
||||
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "null cache policy");
|
||||
}
|
||||
if (text == "none") {
|
||||
*out = CachePolicy::None;
|
||||
return Status::success();
|
||||
}
|
||||
if (text == "memory") {
|
||||
*out = CachePolicy::Memory;
|
||||
return Status::success();
|
||||
}
|
||||
if (text == "disk") {
|
||||
*out = CachePolicy::Disk;
|
||||
return Status::success();
|
||||
}
|
||||
return cache_fail(JOC_ERR_INVALID_CONFIG, "cache_policy must be none, memory, or disk");
|
||||
}
|
||||
|
||||
Status validate_jochrtf(const std::string& path, const std::string& source_sha256,
|
||||
const std::string& cache_key, Field* out) {
|
||||
Field field;
|
||||
const Status status = load_jochrtf(path, &field);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (!source_sha256.empty() && field.source_sha256 != upper(source_sha256)) {
|
||||
return cache_fail(JOC_ERR_HRTF_HASH, "compiled HRTF source hash mismatch");
|
||||
}
|
||||
if (!cache_key.empty() && field.cache_key != upper(cache_key)) {
|
||||
return cache_fail(JOC_ERR_HRTF_HASH, "compiled HRTF configuration hash mismatch");
|
||||
}
|
||||
if (out != nullptr) {
|
||||
*out = std::move(field);
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status save_jochrtf_atomic(const Field& field, const std::string& path) {
|
||||
if (path.empty()) {
|
||||
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "empty compiled HRTF cache path");
|
||||
}
|
||||
const fs::path target = fs_utf8::to_path(path);
|
||||
std::error_code error;
|
||||
if (target.has_parent_path()) {
|
||||
fs::create_directories(target.parent_path(), error);
|
||||
if (error) {
|
||||
return cache_fail(JOC_ERR_OUTPUT_OPEN,
|
||||
"cannot create " + fs_utf8::from_path(target.parent_path()));
|
||||
}
|
||||
}
|
||||
const std::string temporary = path + ".tmp";
|
||||
Status status = write_jochrtf(field, temporary);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
// The rename is what makes a half-written cache impossible to observe.
|
||||
fs::rename(fs_utf8::to_path(temporary), target, error);
|
||||
if (error) {
|
||||
fs::remove(fs_utf8::to_path(temporary), error);
|
||||
return cache_fail(JOC_ERR_OUTPUT_WRITE, "cannot replace " + path);
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status load_or_compile_sofa_field(const SofaFieldRequest& request, Field* out,
|
||||
std::string* cache_path) {
|
||||
if (out == nullptr) {
|
||||
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "null compiled HRTF destination");
|
||||
}
|
||||
if (request.sofa_path.empty()) {
|
||||
return cache_fail(JOC_ERR_INVALID_ARGUMENT, "no SOFA path for the compiled HRTF field");
|
||||
}
|
||||
if (request.policy == CachePolicy::Disk && request.cache_dir.empty()) {
|
||||
return cache_fail(JOC_ERR_INVALID_CONFIG, "the disk cache policy needs a cache directory");
|
||||
}
|
||||
SofaHrir sofa;
|
||||
Status status = load_sofa(request.sofa_path, &sofa);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
CanonicalHrtf canonical;
|
||||
status = canonicalize_sofa(sofa, &canonical);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
// The key depends on the shell that the radius selects, exactly as upstream.
|
||||
double actual_radius = request.options.shell_radius_m;
|
||||
(void)canonical_shell_indices(canonical, request.options.shell_radius_m, &actual_radius);
|
||||
const std::string key = compiled_hrtf_cache_key(
|
||||
canonical.source_sha256, canonical.sample_rate_hz, actual_radius, request.options.order,
|
||||
request.options.projection_ridge, request.options.sh_ridge);
|
||||
if (cache_path != nullptr) {
|
||||
cache_path->clear();
|
||||
}
|
||||
|
||||
std::string target;
|
||||
if (request.policy == CachePolicy::Disk) {
|
||||
target = request.cache_dir;
|
||||
if (!target.empty() && target.back() != '/' && target.back() != '\\') {
|
||||
target += "/";
|
||||
}
|
||||
target += cache_file_name(canonical.source_path.empty()
|
||||
? std::string()
|
||||
: canonical.source_path,
|
||||
key);
|
||||
if (fs_utf8::exists(target)) {
|
||||
Field cached_field;
|
||||
const Status cached =
|
||||
validate_jochrtf(target, canonical.source_sha256, key, &cached_field);
|
||||
if (cached.ok()) {
|
||||
memory_cache_put(key, cached_field);
|
||||
*out = std::move(cached_field);
|
||||
if (cache_path != nullptr) {
|
||||
*cache_path = target;
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
}
|
||||
}
|
||||
Field field;
|
||||
if (request.policy != CachePolicy::None && memory_cache_get(key, &field)) {
|
||||
// A memory hit still materialises the disk cache the caller asked for.
|
||||
if (request.policy == CachePolicy::Disk) {
|
||||
status = save_jochrtf_atomic(field, target);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (cache_path != nullptr) {
|
||||
*cache_path = target;
|
||||
}
|
||||
}
|
||||
*out = std::move(field);
|
||||
return Status::success();
|
||||
}
|
||||
status = compile_canonical_field(canonical, request.options, &field);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (field.cache_key != key) {
|
||||
return cache_fail(JOC_ERR_INTERNAL, "internal compiled HRTF cache-key mismatch");
|
||||
}
|
||||
if (request.policy == CachePolicy::Disk) {
|
||||
status = save_jochrtf_atomic(field, target);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (cache_path != nullptr) {
|
||||
*cache_path = target;
|
||||
}
|
||||
}
|
||||
if (request.policy != CachePolicy::None) {
|
||||
memory_cache_put(key, field);
|
||||
}
|
||||
*out = std::move(field);
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -0,0 +1,38 @@
|
||||
#pragma once
|
||||
|
||||
#include <string>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "hrtf/jochrtf.h"
|
||||
#include "hrtf/sofa_field.h"
|
||||
|
||||
// Compiled-field cache: the .jochrtf is an internal artifact, so the caller only
|
||||
// names the SOFA file and the policy. "memory" keeps the compiled field in this
|
||||
// process, "disk" additionally reuses (and writes) <cache_dir>/<name>.<key>.jochrtf.
|
||||
namespace joc::hrtf {
|
||||
|
||||
enum class CachePolicy { None, Memory, Disk };
|
||||
|
||||
struct SofaFieldRequest {
|
||||
std::string sofa_path;
|
||||
CompileOptions options;
|
||||
CachePolicy policy = CachePolicy::Memory;
|
||||
std::string cache_dir; // required for the disk policy
|
||||
};
|
||||
|
||||
// Parses "none"/"memory"/"disk"; anything else is rejected.
|
||||
Status parse_cache_policy(const std::string& text, CachePolicy* out);
|
||||
|
||||
// Returns the compiled field, reusing a valid cache when the policy allows it.
|
||||
// `cache_path` (optional) receives the cache file that was read or written.
|
||||
Status load_or_compile_sofa_field(const SofaFieldRequest& request, Field* out,
|
||||
std::string* cache_path);
|
||||
|
||||
// Writes the field to `path` through a temporary file and an atomic rename.
|
||||
Status save_jochrtf_atomic(const Field& field, const std::string& path);
|
||||
|
||||
// Verifies that a cache file belongs to `source_sha256` and `cache_key`.
|
||||
Status validate_jochrtf(const std::string& path, const std::string& source_sha256,
|
||||
const std::string& cache_key, Field* out);
|
||||
|
||||
} // namespace joc::hrtf
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,103 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "hrtf/jochrtf.h"
|
||||
#include "hrtf/sofa.h"
|
||||
|
||||
// SOFA SimpleFreeFieldHRIR -> compiled directional field, ported from the
|
||||
// reference chain (sofa_canonical.py + sofa_hrtf_field.py + the public
|
||||
// filterbank): the measurement shell is selected, one delay representation is
|
||||
// separated, the FIRs are projected onto the 64-QMF/77-hybrid filterbank and the
|
||||
// result is fitted with fifth-order ACN/N3D real spherical harmonics.
|
||||
namespace joc::hrtf {
|
||||
|
||||
inline constexpr int kFieldOrder = 5;
|
||||
inline constexpr int kFieldTerms = 36;
|
||||
inline constexpr double kFieldSampleRateHz = 48000.0;
|
||||
inline constexpr double kDefaultShellRadiusM = 1.0;
|
||||
inline constexpr double kDefaultProjectionRidge = 1.0e-3;
|
||||
inline constexpr double kDefaultSphericalHarmonicRidge = 1.0e-5;
|
||||
inline constexpr const char* kCompilerVersion = "joc-sofa-compiler-v1";
|
||||
inline constexpr const char* kPhasePolicyVersion = "sofa-delay-exactly-once-v1";
|
||||
inline constexpr const char* kShConvention = "ACN/N3D real";
|
||||
inline constexpr const char* kFilterbankTableVersion = "joc-public-64qmf-77hybrid-v1";
|
||||
// SHA-256 of the standard filterbank archive the embedded tables came from. It
|
||||
// participates in the cache key, so it is part of the file-format contract.
|
||||
inline constexpr const char* kFilterbankArchiveSha256 =
|
||||
"C05BEF4D26E96ECBD4694E2572F05DA400255C777BA5047300B9D3B1F81081CD";
|
||||
|
||||
struct CompileOptions {
|
||||
double shell_radius_m = kDefaultShellRadiusM;
|
||||
int order = kFieldOrder;
|
||||
double projection_ridge = kDefaultProjectionRidge;
|
||||
double sh_ridge = kDefaultSphericalHarmonicRidge;
|
||||
};
|
||||
|
||||
// Canonical HRIR set: Data.IR and Data.Delay stay separate, the listener frame is
|
||||
// applied to the source positions and the ears are ordered left/right.
|
||||
struct CanonicalHrtf {
|
||||
std::string source_path;
|
||||
std::string source_sha256;
|
||||
std::string convention;
|
||||
std::string convention_version;
|
||||
std::string processing_label;
|
||||
double sample_rate_hz = 0.0;
|
||||
std::uint32_t measurements = 0;
|
||||
std::uint32_t taps = 0;
|
||||
int left_receiver_index = 0;
|
||||
int right_receiver_index = 1;
|
||||
std::vector<double> source_position_cartesian_m; // [M,3] listener-local
|
||||
std::vector<double> unit_directions; // [M,3]
|
||||
std::vector<double> measurement_radius_m; // [M]
|
||||
std::vector<double> hrir; // [M,2,N] canonical L/R
|
||||
std::vector<double> delay_samples; // [M,2], not applied
|
||||
};
|
||||
|
||||
// Port of load_simple_free_field_hrir(): strict SimpleFreeFieldHRIR import.
|
||||
Status canonicalize_sofa(const SofaHrir& sofa, CanonicalHrtf* out);
|
||||
|
||||
// Compiles the canonical set into the runtime field (port of SofaHrtfField.fit).
|
||||
Status compile_sofa_field(const SofaHrir& sofa, const CompileOptions& options, Field* out);
|
||||
|
||||
// The measurements on the shell nearest to radius_m; actual_radius_m receives the
|
||||
// mean radius of that shell (upstream CanonicalHrtf.shell_indices).
|
||||
std::vector<std::size_t> canonical_shell_indices(const CanonicalHrtf& canonical, double radius_m,
|
||||
double* actual_radius_m);
|
||||
|
||||
// Compiles an already canonicalized set (used by tests and the cache layer).
|
||||
Status compile_canonical_field(const CanonicalHrtf& canonical, const CompileOptions& options,
|
||||
Field* out);
|
||||
|
||||
// Configuration hash that names the cache file (upstream compiled_hrtf_cache_key).
|
||||
std::string compiled_hrtf_cache_key(const std::string& source_sha256, double sample_rate_hz,
|
||||
double shell_radius_m, int order, double projection_ridge,
|
||||
double sh_ridge);
|
||||
|
||||
// Payload hash over the four arrays (upstream _payload_sha256).
|
||||
std::string field_payload_sha256(const Field& field);
|
||||
|
||||
// "<stem>.<first 20 key digits>.jochrtf", the upstream cache file name.
|
||||
std::string cache_file_name(const std::string& display_name, const std::string& cache_key);
|
||||
|
||||
// Serializes the field as a .jochrtf cache the upstream loader also accepts.
|
||||
Status write_jochrtf(const Field& field, const std::string& path);
|
||||
|
||||
// The analysis/gain/synthesis dictionary the projection solves against (dev check).
|
||||
std::vector<double> hybrid_gain_synthesis_dictionary_for_check(std::size_t sample_count);
|
||||
|
||||
// Shell directions and their spherical Voronoi weights (dev check).
|
||||
void shell_directions_and_weights_for_check(const SofaHrir& sofa, double radius_m,
|
||||
std::vector<double>* directions,
|
||||
std::vector<double>* weights);
|
||||
|
||||
// PublicAnalysis77 on a unit impulse, interleaved complex (dev check).
|
||||
std::vector<double> analysis_impulse_for_check(std::size_t total_samples);
|
||||
|
||||
// The 77 hybrid-band centre frequencies at 48 kHz.
|
||||
const std::vector<double>& hybrid_band_center_frequencies_hz();
|
||||
|
||||
} // namespace joc::hrtf
|
||||
@@ -0,0 +1,241 @@
|
||||
#include "io/adm_writer.h"
|
||||
#include "foundation/fs_utf8.h"
|
||||
|
||||
#include <cstring>
|
||||
#include <filesystem>
|
||||
|
||||
#include "io/wav_writer.h" // pack_int24 (shared int24 quantisation)
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr long kDs64BodyOffset = 20;
|
||||
constexpr long kDataSizeOffset = 76;
|
||||
|
||||
void put_u16(std::string* out, std::uint16_t value) {
|
||||
char buffer[2];
|
||||
std::memcpy(buffer, &value, 2);
|
||||
out->append(buffer, 2);
|
||||
}
|
||||
|
||||
void put_u32(std::string* out, std::uint32_t value) {
|
||||
char buffer[4];
|
||||
std::memcpy(buffer, &value, 4);
|
||||
out->append(buffer, 4);
|
||||
}
|
||||
|
||||
void put_u64(std::string* out, std::uint64_t value) {
|
||||
char buffer[8];
|
||||
std::memcpy(buffer, &value, 8);
|
||||
out->append(buffer, 8);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
AdmBwfWriter::~AdmBwfWriter() { abort(); }
|
||||
|
||||
Status AdmBwfWriter::open(const std::string& path, std::size_t block_samples) {
|
||||
if (block_samples < JOC_FRAME_SAMPLES) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput,
|
||||
"ADM block size must hold at least one E-AC-3 frame");
|
||||
}
|
||||
path_ = path;
|
||||
block_samples_ = block_samples;
|
||||
used_ = 0;
|
||||
frames_ = 0;
|
||||
finalized_ = false;
|
||||
buffer_.assign(block_samples * kChannels, 0.0f);
|
||||
|
||||
file_ = fs_utf8::fopen(path, "wb+");
|
||||
if (file_ == nullptr) {
|
||||
std::error_code ignored;
|
||||
const std::filesystem::path parent = std::filesystem::path(path).parent_path();
|
||||
if (!parent.empty()) {
|
||||
std::filesystem::create_directories(parent, ignored);
|
||||
}
|
||||
file_ = fs_utf8::fopen(path, "wb+");
|
||||
}
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_OPEN, stage::kOutput, "cannot open " + path);
|
||||
}
|
||||
std::string header;
|
||||
header.append("RF64", 4);
|
||||
put_u32(&header, 0xFFFFFFFFu);
|
||||
header.append("WAVE", 4);
|
||||
if (std::fwrite(header.data(), 1, header.size(), file_) != header.size()) {
|
||||
abort();
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "cannot write " + path);
|
||||
}
|
||||
Status status = write_chunk("ds64", std::string(28, '\0'));
|
||||
if (!status.ok()) {
|
||||
abort();
|
||||
return status;
|
||||
}
|
||||
std::string fmt;
|
||||
put_u16(&fmt, 1);
|
||||
put_u16(&fmt, static_cast<std::uint16_t>(kChannels));
|
||||
put_u32(&fmt, kRate);
|
||||
put_u32(&fmt, kRate * kChannels * 3u);
|
||||
put_u16(&fmt, static_cast<std::uint16_t>(kChannels * 3u));
|
||||
put_u16(&fmt, 24);
|
||||
status = write_chunk("fmt ", fmt);
|
||||
if (!status.ok()) {
|
||||
abort();
|
||||
return status;
|
||||
}
|
||||
status = write_chunk("data", std::string());
|
||||
if (!status.ok()) {
|
||||
abort();
|
||||
return status;
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status AdmBwfWriter::write_chunk(const char id[4], const std::string& body) {
|
||||
std::string header;
|
||||
header.append(id, 4);
|
||||
put_u32(&header, static_cast<std::uint32_t>(body.size()));
|
||||
if (std::fwrite(header.data(), 1, header.size(), file_) != header.size()) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "chunk header write failed");
|
||||
}
|
||||
if (!body.empty() &&
|
||||
std::fwrite(body.data(), 1, body.size(), file_) != body.size()) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "chunk body write failed");
|
||||
}
|
||||
if ((body.size() & 1u) != 0u) {
|
||||
const char pad = '\0';
|
||||
if (std::fwrite(&pad, 1, 1, file_) != 1) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "chunk padding write failed");
|
||||
}
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status AdmBwfWriter::flush() {
|
||||
if (used_ == 0) {
|
||||
return Status::success();
|
||||
}
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer is not open");
|
||||
}
|
||||
packed_.clear();
|
||||
pack_int24(buffer_.data(), used_, kChannels, &packed_);
|
||||
if (std::fwrite(packed_.data(), 1, packed_.size(), file_) != packed_.size()) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "audio write failed for " + path_);
|
||||
}
|
||||
used_ = 0;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status AdmBwfWriter::write_objects16(const float* planar16) {
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer is not open");
|
||||
}
|
||||
if (planar16 == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput, "null frame");
|
||||
}
|
||||
std::size_t source = 0;
|
||||
while (source < JOC_FRAME_SAMPLES) {
|
||||
const std::size_t available = block_samples_ - used_;
|
||||
const std::size_t count =
|
||||
std::min(available, static_cast<std::size_t>(JOC_FRAME_SAMPLES) - source);
|
||||
float* target = buffer_.data() + used_ * kChannels;
|
||||
std::memset(target, 0, count * kChannels * sizeof(float));
|
||||
for (std::size_t sample = 0; sample < count; ++sample) {
|
||||
float* row = target + sample * kChannels;
|
||||
row[3] = planar16[0u * JOC_FRAME_SAMPLES + source + sample];
|
||||
for (std::size_t object = 0; object < 15u; ++object) {
|
||||
row[10u + object] =
|
||||
planar16[(object + 1u) * JOC_FRAME_SAMPLES + source + sample];
|
||||
}
|
||||
}
|
||||
used_ += count;
|
||||
source += count;
|
||||
if (used_ == block_samples_) {
|
||||
const Status status = flush();
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
}
|
||||
}
|
||||
frames_ += JOC_FRAME_SAMPLES;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status AdmBwfWriter::finalize(const std::string& axml, const std::string& chna,
|
||||
const std::string& dbmd) {
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer is not open");
|
||||
}
|
||||
if (finalized_) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "ADM writer already finalized");
|
||||
}
|
||||
Status status = flush();
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = write_chunk("axml", axml);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = write_chunk("chna", chna);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
status = write_chunk("dbmd", dbmd);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
|
||||
if (std::fseek(file_, 0, SEEK_END) != 0) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "seek failed for " + path_);
|
||||
}
|
||||
const long long total = std::ftell(file_);
|
||||
if (total < 0) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "tell failed for " + path_);
|
||||
}
|
||||
const std::uint64_t data_len = frames_ * kChannels * 3u;
|
||||
const std::uint32_t data_field =
|
||||
data_len <= 0xFFFFFFFFull ? static_cast<std::uint32_t>(data_len) : 0xFFFFFFFFu;
|
||||
|
||||
if (std::fseek(file_, kDataSizeOffset, SEEK_SET) != 0) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "seek failed for " + path_);
|
||||
}
|
||||
char buffer[4];
|
||||
std::memcpy(buffer, &data_field, 4);
|
||||
if (std::fwrite(buffer, 1, 4, file_) != 4) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "data size patch failed");
|
||||
}
|
||||
|
||||
std::string ds64;
|
||||
put_u64(&ds64, static_cast<std::uint64_t>(total) - 8u);
|
||||
put_u64(&ds64, data_len);
|
||||
put_u64(&ds64, frames_);
|
||||
put_u32(&ds64, 0);
|
||||
if (std::fseek(file_, kDs64BodyOffset, SEEK_SET) != 0) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "seek failed for " + path_);
|
||||
}
|
||||
if (std::fwrite(ds64.data(), 1, ds64.size(), file_) != ds64.size()) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "ds64 patch failed");
|
||||
}
|
||||
finalized_ = true;
|
||||
if (std::fclose(file_) != 0) {
|
||||
file_ = nullptr;
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "close failed for " + path_);
|
||||
}
|
||||
file_ = nullptr;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
void AdmBwfWriter::abort() {
|
||||
if (file_ != nullptr) {
|
||||
std::fclose(file_);
|
||||
file_ = nullptr;
|
||||
}
|
||||
if (!finalized_ && !path_.empty()) {
|
||||
fs_utf8::remove(path_);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,52 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
class AdmBwfWriter {
|
||||
public:
|
||||
static constexpr std::uint32_t kChannels = 25;
|
||||
static constexpr std::uint32_t kRate = 48000;
|
||||
static constexpr std::size_t kDefaultBlockSamples = 131072;
|
||||
|
||||
AdmBwfWriter() = default;
|
||||
~AdmBwfWriter();
|
||||
|
||||
AdmBwfWriter(const AdmBwfWriter&) = delete;
|
||||
AdmBwfWriter& operator=(const AdmBwfWriter&) = delete;
|
||||
|
||||
Status open(const std::string& path, std::size_t block_samples = kDefaultBlockSamples);
|
||||
|
||||
Status write_objects16(const float* planar16);
|
||||
|
||||
Status finalize(const std::string& axml, const std::string& chna, const std::string& dbmd);
|
||||
|
||||
// Closes and removes a file that was never finalized (plan 28.3: abort must
|
||||
void abort();
|
||||
|
||||
std::uint64_t frames() const { return frames_; }
|
||||
bool open_ok() const { return file_ != nullptr; }
|
||||
|
||||
private:
|
||||
Status write_chunk(const char id[4], const std::string& body);
|
||||
Status flush();
|
||||
|
||||
std::FILE* file_ = nullptr;
|
||||
std::string path_;
|
||||
std::size_t block_samples_ = kDefaultBlockSamples;
|
||||
std::size_t used_ = 0;
|
||||
std::uint64_t frames_ = 0;
|
||||
std::vector<float> buffer_;
|
||||
std::string packed_;
|
||||
bool finalized_ = false;
|
||||
};
|
||||
|
||||
} // namespace joc::io
|
||||
+1430
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,87 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
// Read-only subset of the HDF5 file format, sized for the SOFA files this
|
||||
// project consumes: superblock v0, version 2 object headers, fractal-heap link
|
||||
// and attribute storage, compact and contiguous datasets. The file is opened
|
||||
// lazily: only the requested dataset's bytes are read into memory, everything
|
||||
// else (superblock, object headers, heap blocks) is fetched on demand and the
|
||||
// metadata that was parsed is cached by file address.
|
||||
//
|
||||
// Paths are HDF5 link paths ("Data.IR" is a single link name here, "Group/Set"
|
||||
// walks two links); the empty path names the root group. Byte order is
|
||||
// normalized on read, so callers never see the file's own endianness.
|
||||
namespace joc::io {
|
||||
|
||||
enum class Hdf5Type {
|
||||
Unknown,
|
||||
Int8,
|
||||
Int16,
|
||||
Int32,
|
||||
Int64,
|
||||
UInt8,
|
||||
UInt16,
|
||||
UInt32,
|
||||
UInt64,
|
||||
Float32,
|
||||
Float64,
|
||||
String,
|
||||
};
|
||||
|
||||
struct Hdf5TypeInfo {
|
||||
Hdf5Type type = Hdf5Type::Unknown;
|
||||
std::uint32_t size = 0; // bytes per element as stored in the file
|
||||
bool big_endian = false;
|
||||
bool is_signed = false;
|
||||
};
|
||||
|
||||
struct Hdf5DatasetInfo {
|
||||
std::vector<std::uint64_t> shape;
|
||||
Hdf5TypeInfo type;
|
||||
|
||||
std::uint64_t element_count() const;
|
||||
};
|
||||
|
||||
struct Hdf5Attribute {
|
||||
Hdf5TypeInfo type;
|
||||
std::vector<std::uint64_t> shape;
|
||||
std::vector<std::uint8_t> raw; // C order, host byte order
|
||||
std::string text; // decoded for fixed-length string attributes
|
||||
};
|
||||
|
||||
class Hdf5File {
|
||||
public:
|
||||
Hdf5File();
|
||||
~Hdf5File();
|
||||
Hdf5File(Hdf5File&&) noexcept;
|
||||
Hdf5File& operator=(Hdf5File&&) noexcept;
|
||||
Hdf5File(const Hdf5File&) = delete;
|
||||
Hdf5File& operator=(const Hdf5File&) = delete;
|
||||
|
||||
Status open(const std::string& path);
|
||||
bool is_open() const;
|
||||
|
||||
// Names of the links of a group ("" is the root group).
|
||||
Status links(const std::string& group_path, std::vector<std::string>* names) const;
|
||||
|
||||
bool has_dataset(const std::string& path) const;
|
||||
Status dataset_info(const std::string& path, Hdf5DatasetInfo* out) const;
|
||||
Status read_dataset_raw(const std::string& path, std::vector<std::uint8_t>* out) const;
|
||||
Status read_dataset_double(const std::string& path, std::vector<double>* out) const;
|
||||
|
||||
Status attribute_names(const std::string& object_path, std::vector<std::string>* names) const;
|
||||
Status attribute(const std::string& object_path, const std::string& name, Hdf5Attribute* out) const;
|
||||
Status attribute_text(const std::string& object_path, const std::string& name, std::string* out) const;
|
||||
|
||||
private:
|
||||
struct Impl;
|
||||
std::unique_ptr<Impl> impl_;
|
||||
};
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,276 @@
|
||||
#include "io/inflate.h"
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
class LsbBitReader {
|
||||
public:
|
||||
LsbBitReader(const std::uint8_t* data, std::size_t size) : data_(data), size_(size) {}
|
||||
|
||||
bool ok() const { return ok_; }
|
||||
std::size_t byte_position() const { return position_ >> 3; }
|
||||
|
||||
std::uint32_t bits(unsigned count) {
|
||||
std::uint32_t value = 0;
|
||||
for (unsigned i = 0; i < count; ++i) {
|
||||
if ((position_ >> 3) >= size_) {
|
||||
ok_ = false;
|
||||
return value;
|
||||
}
|
||||
const std::uint32_t bit = (data_[position_ >> 3] >> (position_ & 7u)) & 1u;
|
||||
value |= bit << i;
|
||||
++position_;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
void align_to_byte() { position_ = (position_ + 7u) & ~static_cast<std::size_t>(7u); }
|
||||
|
||||
void skip_bytes(std::size_t count) { position_ += count * 8u; }
|
||||
|
||||
private:
|
||||
const std::uint8_t* data_;
|
||||
std::size_t size_;
|
||||
std::size_t position_ = 0;
|
||||
bool ok_ = true;
|
||||
};
|
||||
|
||||
struct Huffman {
|
||||
std::uint16_t counts[16] = {};
|
||||
std::uint16_t symbols[288] = {};
|
||||
int max_length = 0;
|
||||
|
||||
bool build(const std::uint8_t* lengths, int count) {
|
||||
for (int i = 0; i < 16; ++i) {
|
||||
counts[i] = 0;
|
||||
}
|
||||
for (int i = 0; i < count; ++i) {
|
||||
counts[lengths[i]]++;
|
||||
}
|
||||
counts[0] = 0;
|
||||
std::uint16_t offsets[16] = {};
|
||||
std::uint16_t total = 0;
|
||||
for (int length = 1; length < 16; ++length) {
|
||||
offsets[length] = total;
|
||||
total = static_cast<std::uint16_t>(total + counts[length]);
|
||||
}
|
||||
if (total == 0) {
|
||||
return false;
|
||||
}
|
||||
for (int symbol = 0; symbol < count; ++symbol) {
|
||||
const std::uint8_t length = lengths[symbol];
|
||||
if (length != 0) {
|
||||
symbols[offsets[length]++] = static_cast<std::uint16_t>(symbol);
|
||||
}
|
||||
}
|
||||
max_length = 15;
|
||||
while (max_length > 0 && counts[max_length] == 0) {
|
||||
--max_length;
|
||||
}
|
||||
return max_length != 0;
|
||||
}
|
||||
|
||||
int decode(LsbBitReader* reader) const {
|
||||
int code = 0;
|
||||
int first = 0;
|
||||
int index = 0;
|
||||
for (int length = 1; length <= max_length; ++length) {
|
||||
code |= static_cast<int>(reader->bits(1));
|
||||
if (!reader->ok()) {
|
||||
return -1;
|
||||
}
|
||||
const int count = counts[length];
|
||||
if (code - first < count) {
|
||||
return symbols[index + (code - first)];
|
||||
}
|
||||
index += count;
|
||||
first = (first + count) << 1;
|
||||
code <<= 1;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
};
|
||||
|
||||
constexpr std::uint16_t kLengthBase[29] = {3, 4, 5, 6, 7, 8, 9, 10, 11, 13, 15, 17, 19,
|
||||
23, 27, 31, 35, 43, 51, 59, 67, 83, 99, 115, 131, 163,
|
||||
195, 227, 258};
|
||||
constexpr std::uint8_t kLengthExtra[29] = {0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2,
|
||||
2, 3, 3, 3, 3, 4, 4, 4, 4, 5, 5, 5, 5, 0};
|
||||
constexpr std::uint16_t kDistanceBase[30] = {1, 2, 3, 4, 5, 7, 9, 13,
|
||||
17, 25, 33, 49, 65, 97, 129, 193,
|
||||
257, 385, 513, 769, 1025, 1537, 2049, 3073,
|
||||
4097, 6145, 8193, 12289, 16385, 24577};
|
||||
constexpr std::uint8_t kDistanceExtra[30] = {0, 0, 0, 0, 1, 1, 2, 2, 3, 3, 4, 4, 5, 5, 6,
|
||||
6, 7, 7, 8, 8, 9, 9, 10, 10, 11, 11, 12, 12, 13, 13};
|
||||
constexpr std::uint8_t kCodeLengthOrder[19] = {16, 17, 18, 0, 8, 7, 9, 6, 10, 5, 11, 4, 12, 3, 13, 2,
|
||||
14, 1, 15};
|
||||
|
||||
bool inflate_block_data(LsbBitReader* reader, const Huffman& literal, const Huffman& distance,
|
||||
std::vector<std::uint8_t>* out) {
|
||||
for (;;) {
|
||||
const int symbol = literal.decode(reader);
|
||||
if (symbol < 0) {
|
||||
return false;
|
||||
}
|
||||
if (symbol < 256) {
|
||||
out->push_back(static_cast<std::uint8_t>(symbol));
|
||||
continue;
|
||||
}
|
||||
if (symbol == 256) {
|
||||
return true;
|
||||
}
|
||||
const int length_index = symbol - 257;
|
||||
if (length_index >= 29) {
|
||||
return false;
|
||||
}
|
||||
const std::uint32_t length =
|
||||
kLengthBase[length_index] + reader->bits(kLengthExtra[length_index]);
|
||||
const int distance_symbol = distance.decode(reader);
|
||||
if (distance_symbol < 0 || distance_symbol >= 30) {
|
||||
return false;
|
||||
}
|
||||
const std::uint32_t distance_value =
|
||||
kDistanceBase[distance_symbol] + reader->bits(kDistanceExtra[distance_symbol]);
|
||||
if (!reader->ok() || distance_value == 0 || distance_value > out->size()) {
|
||||
return false;
|
||||
}
|
||||
const std::size_t start = out->size() - distance_value;
|
||||
for (std::uint32_t i = 0; i < length; ++i) {
|
||||
out->push_back((*out)[start + i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool inflate_fixed(LsbBitReader* reader, std::vector<std::uint8_t>* out) {
|
||||
std::uint8_t lengths[288];
|
||||
for (int i = 0; i < 144; ++i) { lengths[i] = 8; }
|
||||
for (int i = 144; i < 256; ++i) { lengths[i] = 9; }
|
||||
for (int i = 256; i < 280; ++i) { lengths[i] = 7; }
|
||||
for (int i = 280; i < 288; ++i) { lengths[i] = 8; }
|
||||
Huffman literal;
|
||||
if (!literal.build(lengths, 288)) {
|
||||
return false;
|
||||
}
|
||||
std::uint8_t distance_lengths[30];
|
||||
for (int i = 0; i < 30; ++i) { distance_lengths[i] = 5; }
|
||||
Huffman distance;
|
||||
if (!distance.build(distance_lengths, 30)) {
|
||||
return false;
|
||||
}
|
||||
return inflate_block_data(reader, literal, distance, out);
|
||||
}
|
||||
|
||||
bool inflate_dynamic(LsbBitReader* reader, std::vector<std::uint8_t>* out) {
|
||||
const int literal_count = static_cast<int>(reader->bits(5)) + 257;
|
||||
const int distance_count = static_cast<int>(reader->bits(5)) + 1;
|
||||
const int code_length_count = static_cast<int>(reader->bits(4)) + 4;
|
||||
if (!reader->ok() || literal_count > 286 || distance_count > 30) {
|
||||
return false;
|
||||
}
|
||||
std::uint8_t code_lengths[19] = {};
|
||||
for (int i = 0; i < code_length_count; ++i) {
|
||||
code_lengths[kCodeLengthOrder[i]] = static_cast<std::uint8_t>(reader->bits(3));
|
||||
}
|
||||
if (!reader->ok()) {
|
||||
return false;
|
||||
}
|
||||
Huffman code_length_tree;
|
||||
if (!code_length_tree.build(code_lengths, 19)) {
|
||||
return false;
|
||||
}
|
||||
std::uint8_t lengths[288 + 30] = {};
|
||||
const int total = literal_count + distance_count;
|
||||
int index = 0;
|
||||
while (index < total) {
|
||||
const int symbol = code_length_tree.decode(reader);
|
||||
if (symbol < 0) {
|
||||
return false;
|
||||
}
|
||||
if (symbol < 16) {
|
||||
lengths[index++] = static_cast<std::uint8_t>(symbol);
|
||||
continue;
|
||||
}
|
||||
int repeat = 0;
|
||||
std::uint8_t value = 0;
|
||||
if (symbol == 16) {
|
||||
if (index == 0) {
|
||||
return false;
|
||||
}
|
||||
value = lengths[index - 1];
|
||||
repeat = 3 + static_cast<int>(reader->bits(2));
|
||||
} else if (symbol == 17) {
|
||||
repeat = 3 + static_cast<int>(reader->bits(3));
|
||||
} else {
|
||||
repeat = 11 + static_cast<int>(reader->bits(7));
|
||||
}
|
||||
if (!reader->ok() || index + repeat > total) {
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < repeat; ++i) {
|
||||
lengths[index++] = value;
|
||||
}
|
||||
}
|
||||
Huffman literal;
|
||||
if (!literal.build(lengths, literal_count)) {
|
||||
return false;
|
||||
}
|
||||
Huffman distance;
|
||||
if (!distance.build(lengths + literal_count, distance_count)) {
|
||||
return false;
|
||||
}
|
||||
return inflate_block_data(reader, literal, distance, out);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool inflate_raw(const std::uint8_t* data, std::size_t size, std::vector<std::uint8_t>* out) {
|
||||
if (data == nullptr || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
out->clear();
|
||||
LsbBitReader reader(data, size);
|
||||
for (;;) {
|
||||
const std::uint32_t final_block = reader.bits(1);
|
||||
const std::uint32_t type = reader.bits(2);
|
||||
if (!reader.ok()) {
|
||||
return false;
|
||||
}
|
||||
if (type == 0) {
|
||||
reader.align_to_byte();
|
||||
const std::size_t position = reader.byte_position();
|
||||
if (position + 4 > size) {
|
||||
return false;
|
||||
}
|
||||
const std::uint16_t length = static_cast<std::uint16_t>(data[position] | (data[position + 1] << 8));
|
||||
const std::uint16_t complement =
|
||||
static_cast<std::uint16_t>(data[position + 2] | (data[position + 3] << 8));
|
||||
if (static_cast<std::uint16_t>(length ^ 0xFFFFu) != complement) {
|
||||
return false;
|
||||
}
|
||||
if (position + 4 + length > size) {
|
||||
return false;
|
||||
}
|
||||
out->insert(out->end(), data + position + 4, data + position + 4 + length);
|
||||
reader.skip_bytes(4u + length);
|
||||
} else if (type == 1) {
|
||||
if (!inflate_fixed(&reader, out)) {
|
||||
return false;
|
||||
}
|
||||
} else if (type == 2) {
|
||||
if (!inflate_dynamic(&reader, out)) {
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
if (final_block != 0u) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,12 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
bool inflate_raw(const std::uint8_t* data, std::size_t size, std::vector<std::uint8_t>* out);
|
||||
|
||||
} // namespace joc::io
|
||||
+420
@@ -0,0 +1,420 @@
|
||||
#include "io/npy.h"
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
std::uint16_t read_u16(const std::uint8_t* p) { return static_cast<std::uint16_t>(p[0] | (p[1] << 8)); }
|
||||
std::uint32_t read_u32(const std::uint8_t* p) {
|
||||
return static_cast<std::uint32_t>(p[0]) | (static_cast<std::uint32_t>(p[1]) << 8) |
|
||||
(static_cast<std::uint32_t>(p[2]) << 16) | (static_cast<std::uint32_t>(p[3]) << 24);
|
||||
}
|
||||
|
||||
NpyType classify(const std::string& descr) {
|
||||
if (descr == "<f8" || descr == "=f8" || descr == "|f8") { return NpyType::Float64; }
|
||||
if (descr == "<f4" || descr == "=f4") { return NpyType::Float32; }
|
||||
if (descr == "<i8" || descr == "=i8") { return NpyType::Int64; }
|
||||
if (descr == "<i4" || descr == "=i4") { return NpyType::Int32; }
|
||||
if (descr == "<i2" || descr == "=i2") { return NpyType::Int16; }
|
||||
if (descr == "|u1" || descr == "<u1") { return NpyType::UInt8; }
|
||||
if (descr == "<c16" || descr == "=c16") { return NpyType::Complex128; }
|
||||
if (descr.size() > 2 && descr[0] == '<' && descr[1] == 'U') {
|
||||
return NpyType::Unicode;
|
||||
}
|
||||
if (descr.size() > 2 && descr[0] == '=' && descr[1] == 'U') {
|
||||
return NpyType::Unicode;
|
||||
}
|
||||
return NpyType::Unknown;
|
||||
}
|
||||
|
||||
std::size_t unicode_length(const std::string& descr) {
|
||||
std::size_t index = 0;
|
||||
while (index < descr.size() && (descr[index] == '<' || descr[index] == '=')) {
|
||||
++index;
|
||||
}
|
||||
if (index >= descr.size() || descr[index] != 'U') {
|
||||
return 0;
|
||||
}
|
||||
++index;
|
||||
std::size_t value = 0;
|
||||
bool any = false;
|
||||
while (index < descr.size() && descr[index] >= '0' && descr[index] <= '9') {
|
||||
value = value * 10 + static_cast<std::size_t>(descr[index] - '0');
|
||||
++index;
|
||||
any = true;
|
||||
}
|
||||
return any ? value : 0;
|
||||
}
|
||||
|
||||
bool is_big_endian(const std::string& descr) { return !descr.empty() && descr[0] == '>'; }
|
||||
|
||||
bool header_value(const std::string& header, const std::string& key, std::string* out) {
|
||||
const std::string needle = "'" + key + "'";
|
||||
const std::size_t position = header.find(needle);
|
||||
if (position == std::string::npos) {
|
||||
return false;
|
||||
}
|
||||
const std::size_t colon = header.find(':', position + needle.size());
|
||||
if (colon == std::string::npos) {
|
||||
return false;
|
||||
}
|
||||
std::size_t start = colon + 1;
|
||||
while (start < header.size() && (header[start] == ' ' || header[start] == '\t')) {
|
||||
++start;
|
||||
}
|
||||
*out = header.substr(start);
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::size_t NpyArray::element_count() const {
|
||||
std::size_t count = 1;
|
||||
for (const std::int64_t dimension : shape) {
|
||||
count *= static_cast<std::size_t>(dimension < 0 ? 0 : dimension);
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
std::size_t NpyArray::element_size() const {
|
||||
switch (type) {
|
||||
case NpyType::Float64: return 8;
|
||||
case NpyType::Float32: return 4;
|
||||
case NpyType::Int64: return 8;
|
||||
case NpyType::Int32: return 4;
|
||||
case NpyType::Int16: return 2;
|
||||
case NpyType::UInt8: return 1;
|
||||
case NpyType::Complex128: return 16;
|
||||
case NpyType::Unicode: return item_bytes;
|
||||
default: return 0;
|
||||
}
|
||||
}
|
||||
|
||||
bool parse_npy(const std::uint8_t* data, std::size_t size, NpyArray* out, std::string* error) {
|
||||
if (data == nullptr || out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
const std::uint8_t magic[6] = {0x93u, 'N', 'U', 'M', 'P', 'Y'};
|
||||
if (size < 10u || std::memcmp(data, magic, 6) != 0) {
|
||||
if (error != nullptr) { *error = "not a .npy image"; }
|
||||
return false;
|
||||
}
|
||||
const std::uint8_t major = data[6];
|
||||
std::size_t header_length = 0;
|
||||
std::size_t header_offset = 0;
|
||||
if (major == 1u) {
|
||||
header_length = read_u16(data + 8);
|
||||
header_offset = 10;
|
||||
} else if (major == 2u || major == 3u) {
|
||||
if (size < 12u) {
|
||||
if (error != nullptr) { *error = "truncated .npy v2 header"; }
|
||||
return false;
|
||||
}
|
||||
header_length = read_u32(data + 8);
|
||||
header_offset = 12;
|
||||
} else {
|
||||
if (error != nullptr) { *error = "unsupported .npy version " + std::to_string(major); }
|
||||
return false;
|
||||
}
|
||||
if (header_offset + header_length > size) {
|
||||
if (error != nullptr) { *error = "truncated .npy header"; }
|
||||
return false;
|
||||
}
|
||||
const std::string header(reinterpret_cast<const char*>(data + header_offset), header_length);
|
||||
|
||||
out->descr.clear();
|
||||
std::string value;
|
||||
if (!header_value(header, "descr", &value)) {
|
||||
if (error != nullptr) { *error = ".npy header without descr"; }
|
||||
return false;
|
||||
}
|
||||
const std::size_t first_quote = value.find('\'');
|
||||
const std::size_t second_quote =
|
||||
first_quote == std::string::npos ? std::string::npos : value.find('\'', first_quote + 1);
|
||||
if (first_quote == std::string::npos || second_quote == std::string::npos) {
|
||||
if (error != nullptr) { *error = ".npy descr is not a quoted string"; }
|
||||
return false;
|
||||
}
|
||||
out->descr = value.substr(first_quote + 1, second_quote - first_quote - 1);
|
||||
out->type = classify(out->descr);
|
||||
if (out->type == NpyType::Unknown) {
|
||||
if (error != nullptr) { *error = "unsupported .npy dtype " + out->descr; }
|
||||
return false;
|
||||
}
|
||||
out->item_bytes = 0;
|
||||
if (out->type == NpyType::Unicode) {
|
||||
const std::size_t length = unicode_length(out->descr);
|
||||
if (length == 0) {
|
||||
if (error != nullptr) { *error = "malformed unicode .npy dtype " + out->descr; }
|
||||
return false;
|
||||
}
|
||||
out->item_bytes = length * 4u;
|
||||
}
|
||||
|
||||
out->fortran_order = header.find("'fortran_order': True") != std::string::npos;
|
||||
|
||||
if (!header_value(header, "shape", &value)) {
|
||||
if (error != nullptr) { *error = ".npy header without shape"; }
|
||||
return false;
|
||||
}
|
||||
out->shape.clear();
|
||||
for (std::size_t i = 0; i < value.size(); ++i) {
|
||||
if (value[i] >= '0' && value[i] <= '9') {
|
||||
long long dimension = 0;
|
||||
while (i < value.size() && value[i] >= '0' && value[i] <= '9') {
|
||||
dimension = dimension * 10 + (value[i] - '0');
|
||||
++i;
|
||||
}
|
||||
out->shape.push_back(dimension);
|
||||
} else if (value[i] == ')') {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
const std::size_t expected = out->element_count() * out->element_size();
|
||||
if (header_offset + header_length + expected > size) {
|
||||
if (error != nullptr) {
|
||||
*error = ".npy payload truncated (need " + std::to_string(expected) + " bytes)";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
out->data = data + header_offset + header_length;
|
||||
out->data_bytes = expected;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool npy_shape_is(const NpyArray& array, const std::vector<std::int64_t>& expected) {
|
||||
return array.shape == expected;
|
||||
}
|
||||
|
||||
namespace {
|
||||
|
||||
template <typename T>
|
||||
void load_le(const std::uint8_t* source, std::size_t count, bool swap, std::vector<T>* out) {
|
||||
out->resize(count);
|
||||
std::memcpy(out->data(), source, count * sizeof(T));
|
||||
if (swap) {
|
||||
std::uint8_t* bytes = reinterpret_cast<std::uint8_t*>(out->data());
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
for (std::size_t b = 0; b < sizeof(T) / 2; ++b) {
|
||||
const std::uint8_t temporary = bytes[i * sizeof(T) + b];
|
||||
bytes[i * sizeof(T) + b] = bytes[i * sizeof(T) + sizeof(T) - 1 - b];
|
||||
bytes[i * sizeof(T) + sizeof(T) - 1 - b] = temporary;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool npy_to_double(const NpyArray& array, std::vector<double>* out, std::string* error) {
|
||||
const bool swap = is_big_endian(array.descr);
|
||||
const std::size_t count = array.element_count();
|
||||
switch (array.type) {
|
||||
case NpyType::Float64:
|
||||
load_le(array.data, count, swap, out);
|
||||
return true;
|
||||
case NpyType::Complex128:
|
||||
load_le(array.data, count * 2u, swap, out);
|
||||
return true;
|
||||
case NpyType::Float32: {
|
||||
std::vector<float> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<double>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::Int64: {
|
||||
std::vector<std::int64_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<double>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::Int32: {
|
||||
std::vector<std::int32_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<double>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::Int16: {
|
||||
std::vector<std::int16_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<double>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::UInt8: {
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<double>(array.data[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
default:
|
||||
if (error != nullptr) { *error = "cannot convert " + array.descr + " to double"; }
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool npy_to_int16(const NpyArray& array, std::vector<std::int16_t>* out, std::string* error) {
|
||||
const bool swap = is_big_endian(array.descr);
|
||||
const std::size_t count = array.element_count();
|
||||
switch (array.type) {
|
||||
case NpyType::Int16:
|
||||
load_le(array.data, count, swap, out);
|
||||
return true;
|
||||
case NpyType::Int32: {
|
||||
std::vector<std::int32_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<std::int16_t>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::Int64: {
|
||||
std::vector<std::int64_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<std::int16_t>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
default:
|
||||
if (error != nullptr) { *error = "cannot convert " + array.descr + " to int16"; }
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool npy_to_int32(const NpyArray& array, std::vector<std::int32_t>* out, std::string* error) {
|
||||
const bool swap = is_big_endian(array.descr);
|
||||
const std::size_t count = array.element_count();
|
||||
switch (array.type) {
|
||||
case NpyType::Int32:
|
||||
load_le(array.data, count, swap, out);
|
||||
return true;
|
||||
case NpyType::Int64: {
|
||||
std::vector<std::int64_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<std::int32_t>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
case NpyType::Int16: {
|
||||
std::vector<std::int16_t> values;
|
||||
load_le(array.data, count, swap, &values);
|
||||
out->resize(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
(*out)[i] = static_cast<std::int32_t>(values[i]);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
default:
|
||||
if (error != nullptr) { *error = "cannot convert " + array.descr + " to int32"; }
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool npy_to_uint8(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error) {
|
||||
if (array.type != NpyType::UInt8) {
|
||||
if (error != nullptr) { *error = "cannot convert " + array.descr + " to uint8"; }
|
||||
return false;
|
||||
}
|
||||
out->assign(array.data, array.data + array.element_count());
|
||||
return true;
|
||||
}
|
||||
|
||||
bool npy_unicode_to_utf8(const NpyArray& array, std::string* out, std::string* error) {
|
||||
if (array.type != NpyType::Unicode) {
|
||||
if (error != nullptr) { *error = "not a unicode .npy member: " + array.descr; }
|
||||
return false;
|
||||
}
|
||||
if (array.shape.size() != 0) {
|
||||
if (error != nullptr) { *error = "unicode .npy member must be a scalar"; }
|
||||
return false;
|
||||
}
|
||||
out->clear();
|
||||
const std::size_t count = array.item_bytes / 4u;
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
const std::uint8_t* p = array.data + i * 4u;
|
||||
const std::uint32_t code = static_cast<std::uint32_t>(p[0]) | (static_cast<std::uint32_t>(p[1]) << 8) |
|
||||
(static_cast<std::uint32_t>(p[2]) << 16) |
|
||||
(static_cast<std::uint32_t>(p[3]) << 24);
|
||||
if (code == 0u) {
|
||||
break;
|
||||
}
|
||||
if (code < 0x80u) {
|
||||
out->push_back(static_cast<char>(code));
|
||||
} else if (code < 0x800u) {
|
||||
out->push_back(static_cast<char>(0xC0u | (code >> 6)));
|
||||
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
|
||||
} else if (code < 0x10000u) {
|
||||
out->push_back(static_cast<char>(0xE0u | (code >> 12)));
|
||||
out->push_back(static_cast<char>(0x80u | ((code >> 6) & 0x3Fu)));
|
||||
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
|
||||
} else {
|
||||
out->push_back(static_cast<char>(0xF0u | (code >> 18)));
|
||||
out->push_back(static_cast<char>(0x80u | ((code >> 12) & 0x3Fu)));
|
||||
out->push_back(static_cast<char>(0x80u | ((code >> 6) & 0x3Fu)));
|
||||
out->push_back(static_cast<char>(0x80u | (code & 0x3Fu)));
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool npy_to_c_order(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error) {
|
||||
const std::size_t element = array.element_size();
|
||||
if (element == 0) {
|
||||
if (error != nullptr) { *error = "unsupported element size for " + array.descr; }
|
||||
return false;
|
||||
}
|
||||
if (!array.fortran_order) {
|
||||
out->assign(array.data, array.data + array.data_bytes);
|
||||
return true;
|
||||
}
|
||||
const std::size_t dimensions = array.shape.size();
|
||||
if (dimensions == 0) {
|
||||
out->assign(array.data, array.data + element);
|
||||
return true;
|
||||
}
|
||||
// Source (Fortran) strides in elements; destination is C order.
|
||||
std::vector<std::size_t> source_stride(dimensions, 1);
|
||||
std::size_t running = 1;
|
||||
for (std::size_t d = 0; d < dimensions; ++d) {
|
||||
source_stride[d] = running;
|
||||
running *= static_cast<std::size_t>(array.shape[d]);
|
||||
}
|
||||
out->assign(array.data_bytes, 0);
|
||||
std::vector<std::size_t> index(dimensions, 0);
|
||||
const std::size_t total = array.element_count();
|
||||
for (std::size_t linear = 0; linear < total; ++linear) {
|
||||
std::size_t remainder = linear;
|
||||
for (std::size_t d = dimensions; d-- > 0;) {
|
||||
index[d] = remainder % static_cast<std::size_t>(array.shape[d]);
|
||||
remainder /= static_cast<std::size_t>(array.shape[d]);
|
||||
}
|
||||
std::size_t source = 0;
|
||||
for (std::size_t d = 0; d < dimensions; ++d) {
|
||||
source += index[d] * source_stride[d];
|
||||
}
|
||||
std::memcpy(out->data() + linear * element, array.data + source * element, element);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,42 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
enum class NpyType { Unknown, Float64, Float32, Int64, Int32, Int16, UInt8, Complex128, Unicode };
|
||||
|
||||
struct NpyArray {
|
||||
std::string descr;
|
||||
NpyType type = NpyType::Unknown;
|
||||
bool fortran_order = false;
|
||||
std::vector<std::int64_t> shape;
|
||||
const std::uint8_t* data = nullptr;
|
||||
std::size_t data_bytes = 0;
|
||||
std::size_t item_bytes = 0; // bytes per element as stored
|
||||
|
||||
std::size_t element_count() const;
|
||||
std::size_t element_size() const; // bytes per element in the file
|
||||
};
|
||||
|
||||
// Parses the header of one `.npy` image. `data` must outlive the NpyArray.
|
||||
bool parse_npy(const std::uint8_t* data, std::size_t size, NpyArray* out, std::string* error);
|
||||
|
||||
bool npy_to_double(const NpyArray& array, std::vector<double>* out, std::string* error);
|
||||
bool npy_to_int16(const NpyArray& array, std::vector<std::int16_t>* out, std::string* error);
|
||||
bool npy_to_int32(const NpyArray& array, std::vector<std::int32_t>* out, std::string* error);
|
||||
bool npy_to_uint8(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error);
|
||||
|
||||
bool npy_unicode_to_utf8(const NpyArray& array, std::string* out, std::string* error);
|
||||
|
||||
// Materializes the array in C order as raw element bytes. Fortran-order members
|
||||
// hybrid synthesis table as [count][4] row-major while the shipped table stores it
|
||||
// Fortran-order, so passing the file bytes straight through would transpose it.
|
||||
bool npy_to_c_order(const NpyArray& array, std::vector<std::uint8_t>* out, std::string* error);
|
||||
|
||||
bool npy_shape_is(const NpyArray& array, const std::vector<std::int64_t>& expected);
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,263 @@
|
||||
#include "io/npy_writer.h"
|
||||
|
||||
#include <array>
|
||||
#include <charconv>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
#include "foundation/fs_utf8.h"
|
||||
#include "io/zip_reader.h"
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr std::size_t kNpyHeaderAlignment = 64;
|
||||
|
||||
void append_u16(std::vector<std::uint8_t>* out, std::uint16_t value) {
|
||||
out->push_back(static_cast<std::uint8_t>(value & 0xFFu));
|
||||
out->push_back(static_cast<std::uint8_t>((value >> 8) & 0xFFu));
|
||||
}
|
||||
|
||||
void append_u32(std::vector<std::uint8_t>* out, std::uint32_t value) {
|
||||
for (int index = 0; index < 4; ++index) {
|
||||
out->push_back(static_cast<std::uint8_t>((value >> (8 * index)) & 0xFFu));
|
||||
}
|
||||
}
|
||||
|
||||
void append_bytes(std::vector<std::uint8_t>* out, const void* data, std::size_t size) {
|
||||
const std::uint8_t* bytes = static_cast<const std::uint8_t*>(data);
|
||||
out->insert(out->end(), bytes, bytes + size);
|
||||
}
|
||||
|
||||
std::string shape_literal(const std::vector<std::uint64_t>& shape) {
|
||||
if (shape.empty()) {
|
||||
return "()";
|
||||
}
|
||||
std::string text = "(";
|
||||
for (std::size_t index = 0; index < shape.size(); ++index) {
|
||||
if (index != 0u) {
|
||||
text += ", ";
|
||||
}
|
||||
text += std::to_string(shape[index]);
|
||||
}
|
||||
if (shape.size() == 1u) {
|
||||
text += ",";
|
||||
}
|
||||
text += ")";
|
||||
return text;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::vector<std::uint8_t> npy_image(const std::string& descr,
|
||||
const std::vector<std::uint64_t>& shape,
|
||||
const std::vector<std::uint8_t>& data) {
|
||||
std::string header = "{'descr': '" + descr + "', 'fortran_order': False, 'shape': " +
|
||||
shape_literal(shape) + ", }";
|
||||
// NumPy pads the header so that the payload starts on a 64-byte boundary.
|
||||
const std::size_t preamble = 10u; // magic, version, two byte header length
|
||||
std::size_t total = preamble + header.size() + 1u;
|
||||
const std::size_t padding = (kNpyHeaderAlignment - (total % kNpyHeaderAlignment)) %
|
||||
kNpyHeaderAlignment;
|
||||
header.append(padding, ' ');
|
||||
header.push_back('\n');
|
||||
|
||||
std::vector<std::uint8_t> out;
|
||||
out.reserve(preamble + header.size() + data.size());
|
||||
static const std::uint8_t kMagic[6] = {0x93u, 'N', 'U', 'M', 'P', 'Y'};
|
||||
append_bytes(&out, kMagic, sizeof(kMagic));
|
||||
out.push_back(1u); // major
|
||||
out.push_back(0u); // minor
|
||||
append_u16(&out, static_cast<std::uint16_t>(header.size()));
|
||||
append_bytes(&out, header.data(), header.size());
|
||||
append_bytes(&out, data.data(), data.size());
|
||||
return out;
|
||||
}
|
||||
|
||||
std::vector<std::uint8_t> zip_bytes(const std::vector<NpyMember>& members) {
|
||||
std::vector<std::uint8_t> out;
|
||||
struct Entry {
|
||||
std::string name;
|
||||
std::uint32_t crc = 0;
|
||||
std::uint32_t size = 0;
|
||||
std::uint32_t offset = 0;
|
||||
};
|
||||
std::vector<Entry> entries;
|
||||
entries.reserve(members.size());
|
||||
|
||||
for (const NpyMember& member : members) {
|
||||
const std::string name = member.name + ".npy";
|
||||
const std::vector<std::uint8_t> payload = npy_image(member.descr, member.shape, member.data);
|
||||
Entry entry;
|
||||
entry.name = name;
|
||||
entry.crc = crc32_of(payload.data(), payload.size());
|
||||
entry.size = static_cast<std::uint32_t>(payload.size());
|
||||
entry.offset = static_cast<std::uint32_t>(out.size());
|
||||
entries.push_back(entry);
|
||||
|
||||
append_u32(&out, 0x04034B50u); // local file header
|
||||
append_u16(&out, 20u); // version needed
|
||||
append_u16(&out, 0u); // flags
|
||||
append_u16(&out, 0u); // method: stored
|
||||
append_u16(&out, 0u); // time
|
||||
append_u16(&out, 0x2821u); // date: 2000-01-01, fixed for reproducibility
|
||||
append_u32(&out, entry.crc);
|
||||
append_u32(&out, entry.size);
|
||||
append_u32(&out, entry.size);
|
||||
append_u16(&out, static_cast<std::uint16_t>(name.size()));
|
||||
append_u16(&out, 0u); // extra length
|
||||
append_bytes(&out, name.data(), name.size());
|
||||
append_bytes(&out, payload.data(), payload.size());
|
||||
}
|
||||
|
||||
const std::uint32_t directory_offset = static_cast<std::uint32_t>(out.size());
|
||||
for (const Entry& entry : entries) {
|
||||
append_u32(&out, 0x02014B50u); // central directory header
|
||||
append_u16(&out, 20u); // version made by
|
||||
append_u16(&out, 20u); // version needed
|
||||
append_u16(&out, 0u); // flags
|
||||
append_u16(&out, 0u); // method: stored
|
||||
append_u16(&out, 0u); // time
|
||||
append_u16(&out, 0x2821u); // date
|
||||
append_u32(&out, entry.crc);
|
||||
append_u32(&out, entry.size);
|
||||
append_u32(&out, entry.size);
|
||||
append_u16(&out, static_cast<std::uint16_t>(entry.name.size()));
|
||||
append_u16(&out, 0u); // extra
|
||||
append_u16(&out, 0u); // comment
|
||||
append_u16(&out, 0u); // disk
|
||||
append_u16(&out, 0u); // internal attributes
|
||||
append_u32(&out, 0u); // external attributes
|
||||
append_u32(&out, entry.offset);
|
||||
append_bytes(&out, entry.name.data(), entry.name.size());
|
||||
}
|
||||
const std::uint32_t directory_size = static_cast<std::uint32_t>(out.size()) - directory_offset;
|
||||
|
||||
append_u32(&out, 0x06054B50u); // end of central directory
|
||||
append_u16(&out, 0u);
|
||||
append_u16(&out, 0u);
|
||||
append_u16(&out, static_cast<std::uint16_t>(entries.size()));
|
||||
append_u16(&out, static_cast<std::uint16_t>(entries.size()));
|
||||
append_u32(&out, directory_size);
|
||||
append_u32(&out, directory_offset);
|
||||
append_u16(&out, 0u);
|
||||
return out;
|
||||
}
|
||||
|
||||
bool write_zip(const std::string& path, const std::vector<NpyMember>& members,
|
||||
std::string* error) {
|
||||
const std::vector<std::uint8_t> bytes = zip_bytes(members);
|
||||
std::FILE* stream = fs_utf8::fopen(path, "wb");
|
||||
if (stream == nullptr) {
|
||||
if (error != nullptr) {
|
||||
*error = "cannot open " + path + " for writing";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
const std::size_t written = std::fwrite(bytes.data(), 1, bytes.size(), stream);
|
||||
const bool flushed = std::fclose(stream) == 0;
|
||||
if (written != bytes.size() || !flushed) {
|
||||
if (error != nullptr) {
|
||||
*error = "short write to " + path;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
std::vector<std::uint8_t> utf8_to_utf32le(const std::string& text) {
|
||||
std::vector<std::uint8_t> out;
|
||||
out.reserve(text.size() * 4u);
|
||||
std::size_t index = 0;
|
||||
while (index < text.size()) {
|
||||
const std::uint8_t lead = static_cast<std::uint8_t>(text[index]);
|
||||
std::uint32_t code = 0;
|
||||
std::size_t extra = 0;
|
||||
if (lead < 0x80u) {
|
||||
code = lead;
|
||||
} else if ((lead & 0xE0u) == 0xC0u) {
|
||||
code = lead & 0x1Fu;
|
||||
extra = 1;
|
||||
} else if ((lead & 0xF0u) == 0xE0u) {
|
||||
code = lead & 0x0Fu;
|
||||
extra = 2;
|
||||
} else if ((lead & 0xF8u) == 0xF0u) {
|
||||
code = lead & 0x07u;
|
||||
extra = 3;
|
||||
} else {
|
||||
code = 0xFFFDu; // invalid lead byte: substitute rather than fail
|
||||
extra = 0;
|
||||
}
|
||||
++index;
|
||||
for (std::size_t count = 0; count < extra && index < text.size(); ++count) {
|
||||
code = (code << 6) | (static_cast<std::uint8_t>(text[index]) & 0x3Fu);
|
||||
++index;
|
||||
}
|
||||
for (int byte = 0; byte < 4; ++byte) {
|
||||
out.push_back(static_cast<std::uint8_t>((code >> (8 * byte)) & 0xFFu));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
std::string python_float_repr(double value) {
|
||||
if (std::isnan(value)) {
|
||||
return "NaN";
|
||||
}
|
||||
if (std::isinf(value)) {
|
||||
return value > 0.0 ? "Infinity" : "-Infinity";
|
||||
}
|
||||
// to_chars gives the shortest round-trip digits; Python's repr uses the same
|
||||
// digits but its own notation, so the digits are re-laid-out here.
|
||||
std::array<char, 64> buffer{};
|
||||
const std::to_chars_result converted =
|
||||
std::to_chars(buffer.data(), buffer.data() + buffer.size(), value);
|
||||
std::string text(buffer.data(), converted.ptr);
|
||||
const bool negative = !text.empty() && text[0] == '-';
|
||||
const std::string body = negative ? text.substr(1) : text;
|
||||
const std::size_t exponent_at = body.find_first_of("eE");
|
||||
std::string digits = body;
|
||||
int exponent = 0;
|
||||
if (exponent_at != std::string::npos) {
|
||||
digits = body.substr(0, exponent_at);
|
||||
exponent = std::atoi(body.c_str() + exponent_at + 1);
|
||||
}
|
||||
const std::size_t point = digits.find('.');
|
||||
std::string mantissa = digits;
|
||||
if (point != std::string::npos) {
|
||||
mantissa = digits.substr(0, point) + digits.substr(point + 1);
|
||||
exponent += static_cast<int>(point) - 1;
|
||||
} else {
|
||||
exponent += static_cast<int>(digits.size()) - 1;
|
||||
}
|
||||
while (mantissa.size() > 1u && mantissa.back() == '0') {
|
||||
mantissa.pop_back();
|
||||
}
|
||||
// Python switches to exponent notation below 1e-4 and at 1e16 and above.
|
||||
std::string result;
|
||||
if (exponent < -4 || exponent >= 16) {
|
||||
result = mantissa.substr(0, 1);
|
||||
if (mantissa.size() > 1u) {
|
||||
result += "." + mantissa.substr(1);
|
||||
}
|
||||
char tail[16];
|
||||
std::snprintf(tail, sizeof(tail), "e%+03d", exponent);
|
||||
result += tail;
|
||||
} else if (exponent >= 0) {
|
||||
if (static_cast<std::size_t>(exponent) + 1u >= mantissa.size()) {
|
||||
result = mantissa + std::string(static_cast<std::size_t>(exponent) + 1u - mantissa.size(), '0');
|
||||
result += ".0";
|
||||
} else {
|
||||
result = mantissa.substr(0, static_cast<std::size_t>(exponent) + 1u) + "." +
|
||||
mantissa.substr(static_cast<std::size_t>(exponent) + 1u);
|
||||
}
|
||||
} else {
|
||||
result = "0." + std::string(static_cast<std::size_t>(-exponent - 1), '0') + mantissa;
|
||||
}
|
||||
return negative ? "-" + result : result;
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,39 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
// NPY 1.0 images and a minimal ZIP container, used to write the compiled HRTF
|
||||
// cache in exactly the layout the reader (and NumPy) expects. Only what the
|
||||
// cache needs is implemented: little-endian C-order arrays and stored members.
|
||||
struct NpyMember {
|
||||
std::string name; // archive member name, without the .npy suffix
|
||||
std::string descr; // NumPy dtype string, e.g. "<f8", "<c16", "<U123"
|
||||
std::vector<std::uint64_t> shape;
|
||||
std::vector<std::uint8_t> data; // C order payload in the dtype's byte order
|
||||
};
|
||||
|
||||
// Serializes one array as an NPY 1.0 image (magic, header, 64-byte aligned).
|
||||
std::vector<std::uint8_t> npy_image(const std::string& descr,
|
||||
const std::vector<std::uint64_t>& shape,
|
||||
const std::vector<std::uint8_t>& data);
|
||||
|
||||
// Writes a ZIP archive with stored (uncompressed) members. The upstream reader
|
||||
// accepts stored members, and compression would need a deflate encoder.
|
||||
bool write_zip(const std::string& path, const std::vector<NpyMember>& members,
|
||||
std::string* error);
|
||||
|
||||
// Serializes the archive in memory (same layout as write_zip).
|
||||
std::vector<std::uint8_t> zip_bytes(const std::vector<NpyMember>& members);
|
||||
|
||||
// UTF-8 text as the payload of a NumPy Unicode scalar string ('<U<n>').
|
||||
std::vector<std::uint8_t> utf8_to_utf32le(const std::string& text);
|
||||
|
||||
// Python's repr() for a double: shortest round-trip digits with Python's
|
||||
// exponent rules, which is what json.dumps emits for the cache metadata.
|
||||
std::string python_float_repr(double value);
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,186 @@
|
||||
#include "io/process.h"
|
||||
#include "foundation/fs_utf8.h"
|
||||
|
||||
#include <cstdio>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <random>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#define NOMINMAX
|
||||
#include <windows.h>
|
||||
#else
|
||||
#include <sys/wait.h>
|
||||
#endif
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
std::string quote_argument(const std::string& argument) {
|
||||
if (!argument.empty() && argument.find_first_of(" \t\"") == std::string::npos) {
|
||||
return argument;
|
||||
}
|
||||
std::string quoted = "\"";
|
||||
unsigned backslashes = 0;
|
||||
for (const char c : argument) {
|
||||
if (c == '\\') {
|
||||
++backslashes;
|
||||
continue;
|
||||
}
|
||||
if (c == '"') {
|
||||
quoted.append(backslashes * 2 + 1, '\\');
|
||||
quoted.push_back('"');
|
||||
backslashes = 0;
|
||||
continue;
|
||||
}
|
||||
quoted.append(backslashes, '\\');
|
||||
backslashes = 0;
|
||||
quoted.push_back(c);
|
||||
}
|
||||
quoted.append(backslashes * 2, '\\');
|
||||
quoted.push_back('"');
|
||||
return quoted;
|
||||
}
|
||||
|
||||
std::string tail_of(const std::string& text, std::size_t limit) {
|
||||
if (text.size() <= limit) {
|
||||
return text;
|
||||
}
|
||||
return text.substr(text.size() - limit);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
Status run_process(const std::vector<std::string>& argv, ProcessResult* out) {
|
||||
if (out == nullptr || argv.empty()) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput, "empty command");
|
||||
}
|
||||
out->output.clear();
|
||||
out->exit_code = 0;
|
||||
|
||||
const std::filesystem::path log_path =
|
||||
std::filesystem::temp_directory_path() /
|
||||
("joc_process_" + std::to_string(std::random_device{}()) + ".log");
|
||||
|
||||
auto read_log = [&]() {
|
||||
#if defined(_WIN32)
|
||||
return; // the Windows branch reads the handle it opened
|
||||
#else
|
||||
std::ifstream log = fs_utf8::open_input(fs_utf8::from_path(log_path));
|
||||
if (log) {
|
||||
std::string text((std::istreambuf_iterator<char>(log)),
|
||||
std::istreambuf_iterator<char>());
|
||||
out->output = tail_of(text, 4096);
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
#if defined(_WIN32)
|
||||
std::string command;
|
||||
for (std::size_t i = 0; i < argv.size(); ++i) {
|
||||
if (i != 0) {
|
||||
command.push_back(' ');
|
||||
}
|
||||
command += quote_argument(argv[i]);
|
||||
}
|
||||
auto widen = [](const std::string& text) {
|
||||
if (text.empty()) {
|
||||
return std::wstring();
|
||||
}
|
||||
const int size = MultiByteToWideChar(CP_UTF8, 0, text.c_str(),
|
||||
static_cast<int>(text.size()), nullptr, 0);
|
||||
std::wstring wide(static_cast<std::size_t>(size), L'\0');
|
||||
MultiByteToWideChar(CP_UTF8, 0, text.c_str(), static_cast<int>(text.size()), wide.data(),
|
||||
size);
|
||||
return wide;
|
||||
};
|
||||
const std::wstring wide_command = widen(command);
|
||||
const std::wstring wide_log = widen(fs_utf8::from_path(log_path));
|
||||
|
||||
SECURITY_ATTRIBUTES attributes{};
|
||||
attributes.nLength = sizeof(attributes);
|
||||
attributes.bInheritHandle = TRUE;
|
||||
// DELETE access plus FILE_FLAG_DELETE_ON_CLOSE means the log disappears when
|
||||
// the last handle goes away - including when this process is killed, which
|
||||
// would otherwise leave joc_process_*.log litter in the temp directory.
|
||||
HANDLE log_handle = CreateFileW(
|
||||
wide_log.c_str(), GENERIC_READ | GENERIC_WRITE | DELETE,
|
||||
FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, &attributes, CREATE_ALWAYS,
|
||||
FILE_ATTRIBUTE_NORMAL | FILE_FLAG_DELETE_ON_CLOSE, nullptr);
|
||||
if (log_handle == INVALID_HANDLE_VALUE) {
|
||||
return Status::fail(JOC_ERR_IO, stage::kOutput, "cannot create the process log file");
|
||||
}
|
||||
|
||||
// A delete-on-close file cannot be reopened by name (it is delete-pending), so
|
||||
// the child's output is read back through the handle it wrote to.
|
||||
auto read_log_handle = [&]() {
|
||||
LARGE_INTEGER start{};
|
||||
start.QuadPart = 0;
|
||||
if (!SetFilePointerEx(log_handle, start, nullptr, FILE_BEGIN)) {
|
||||
return;
|
||||
}
|
||||
std::string text;
|
||||
char buffer[1024];
|
||||
DWORD count = 0;
|
||||
while (ReadFile(log_handle, buffer, sizeof(buffer), &count, nullptr) && count > 0) {
|
||||
text.append(buffer, count);
|
||||
}
|
||||
out->output = tail_of(text, 4096);
|
||||
};
|
||||
|
||||
STARTUPINFOW startup{};
|
||||
startup.cb = sizeof(startup);
|
||||
startup.dwFlags = STARTF_USESTDHANDLES;
|
||||
startup.hStdOutput = log_handle;
|
||||
startup.hStdError = log_handle;
|
||||
startup.hStdInput = GetStdHandle(STD_INPUT_HANDLE);
|
||||
PROCESS_INFORMATION process{};
|
||||
|
||||
std::vector<wchar_t> mutable_command(wide_command.begin(), wide_command.end());
|
||||
mutable_command.push_back(L'\0');
|
||||
const BOOL started = CreateProcessW(nullptr, mutable_command.data(), nullptr, nullptr, TRUE,
|
||||
CREATE_NO_WINDOW, nullptr, nullptr, &startup, &process);
|
||||
if (!started) {
|
||||
CloseHandle(log_handle); // delete-on-close removes the file
|
||||
return Status::fail(JOC_ERR_LIBRARY_MISSING, stage::kOutput,
|
||||
"cannot start " + argv[0] + " (is it on PATH?)");
|
||||
}
|
||||
WaitForSingleObject(process.hProcess, INFINITE);
|
||||
DWORD exit_code = 0;
|
||||
GetExitCodeProcess(process.hProcess, &exit_code);
|
||||
CloseHandle(process.hThread);
|
||||
CloseHandle(process.hProcess);
|
||||
// Read the log before the delete-on-close handle goes away.
|
||||
read_log_handle();
|
||||
CloseHandle(log_handle);
|
||||
out->exit_code = static_cast<std::uint32_t>(exit_code);
|
||||
#else
|
||||
std::string command;
|
||||
for (std::size_t i = 0; i < argv.size(); ++i) {
|
||||
if (i != 0) {
|
||||
command.push_back(' ');
|
||||
}
|
||||
command += quote_argument(argv[i]);
|
||||
}
|
||||
command += " > " + quote_argument(fs_utf8::from_path(log_path)) + " 2>&1";
|
||||
const int status = std::system(command.c_str());
|
||||
// system() reports a wait status, not the child's exit code.
|
||||
out->exit_code = status == -1 ? 127u
|
||||
: WIFEXITED(status) ? static_cast<std::uint32_t>(WEXITSTATUS(status))
|
||||
: 128u;
|
||||
read_log();
|
||||
std::error_code ignored;
|
||||
std::filesystem::remove(log_path, ignored);
|
||||
#endif
|
||||
|
||||
if (out->exit_code != 0) {
|
||||
return Status::fail(JOC_ERR_INPUT_FORMAT, stage::kOutput,
|
||||
argv[0] + " failed with exit code " + std::to_string(out->exit_code) +
|
||||
(out->output.empty() ? "" : ": " + tail_of(out->output, 400)));
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,19 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
struct ProcessResult {
|
||||
std::uint32_t exit_code = 0;
|
||||
std::string output;
|
||||
};
|
||||
|
||||
Status run_process(const std::vector<std::string>& argv, ProcessResult* out);
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,242 @@
|
||||
#include "io/wav_writer.h"
|
||||
#include "foundation/fs_utf8.h"
|
||||
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <filesystem>
|
||||
#include <limits>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr std::uint16_t kWaveFormatPcm = 0x0001;
|
||||
constexpr std::uint16_t kWaveFormatIeeeFloat = 0x0003;
|
||||
constexpr std::uint16_t kWaveFormatExtensible = 0xFFFE;
|
||||
constexpr std::uint8_t kPcmGuid[16] = {0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x10, 0x00,
|
||||
0x80, 0x00, 0x00, 0xAA, 0x00, 0x38, 0x9B, 0x71};
|
||||
constexpr std::uint8_t kFloatGuid[16] = {0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x10, 0x00,
|
||||
0x80, 0x00, 0x00, 0xAA, 0x00, 0x38, 0x9B, 0x71};
|
||||
|
||||
void put_u16(std::string* out, std::uint16_t value) {
|
||||
char buffer[2];
|
||||
std::memcpy(buffer, &value, 2);
|
||||
out->append(buffer, 2);
|
||||
}
|
||||
|
||||
void put_u32(std::string* out, std::uint32_t value) {
|
||||
char buffer[4];
|
||||
std::memcpy(buffer, &value, 4);
|
||||
out->append(buffer, 4);
|
||||
}
|
||||
|
||||
void put_u64(std::string* out, std::uint64_t value) {
|
||||
char buffer[8];
|
||||
std::memcpy(buffer, &value, 8);
|
||||
out->append(buffer, 8);
|
||||
}
|
||||
|
||||
// Port of speaker_wav._fmt_chunk.
|
||||
std::string fmt_chunk(std::uint32_t channels, std::uint32_t rate, SampleFormat format, WavInfo* info) {
|
||||
std::uint16_t simple_tag = 0;
|
||||
const std::uint8_t* guid = nullptr;
|
||||
if (format == SampleFormat::Float32) {
|
||||
info->bits_per_sample = 32;
|
||||
info->bytes_per_sample = 4;
|
||||
simple_tag = kWaveFormatIeeeFloat;
|
||||
guid = kFloatGuid;
|
||||
} else {
|
||||
info->bits_per_sample = 24;
|
||||
info->bytes_per_sample = 3;
|
||||
simple_tag = kWaveFormatPcm;
|
||||
guid = kPcmGuid;
|
||||
}
|
||||
const std::uint32_t block_align = channels * info->bytes_per_sample;
|
||||
const std::uint32_t byte_rate = rate * block_align;
|
||||
info->block_align = block_align;
|
||||
|
||||
std::string body;
|
||||
if (channels <= 2) {
|
||||
put_u16(&body, simple_tag);
|
||||
put_u16(&body, static_cast<std::uint16_t>(channels));
|
||||
put_u32(&body, rate);
|
||||
put_u32(&body, byte_rate);
|
||||
put_u16(&body, static_cast<std::uint16_t>(block_align));
|
||||
put_u16(&body, static_cast<std::uint16_t>(info->bits_per_sample));
|
||||
} else {
|
||||
put_u16(&body, kWaveFormatExtensible);
|
||||
put_u16(&body, static_cast<std::uint16_t>(channels));
|
||||
put_u32(&body, rate);
|
||||
put_u32(&body, byte_rate);
|
||||
put_u16(&body, static_cast<std::uint16_t>(block_align));
|
||||
put_u16(&body, static_cast<std::uint16_t>(info->bits_per_sample));
|
||||
put_u16(&body, 22);
|
||||
put_u16(&body, static_cast<std::uint16_t>(info->bits_per_sample));
|
||||
put_u32(&body, 0);
|
||||
body.append(reinterpret_cast<const char*>(guid), 16);
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
// int32 conversion identical to NumPy's float32 -> int32 cast after clipping.
|
||||
std::int32_t to_int32(const float value) {
|
||||
if (!std::isfinite(value)) {
|
||||
return std::numeric_limits<std::int32_t>::min();
|
||||
}
|
||||
return static_cast<std::int32_t>(value);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
void pack_int24(const float* interleaved, std::size_t frames, std::size_t channels,
|
||||
std::string* out) {
|
||||
const std::size_t count = frames * channels;
|
||||
out->resize(count * 3);
|
||||
char* target = out->data();
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
float value = interleaved[i];
|
||||
if (value > 1.0f) {
|
||||
value = 1.0f;
|
||||
} else if (value < -1.0f) {
|
||||
value = -1.0f;
|
||||
}
|
||||
const std::int32_t scaled = to_int32(value * 8388607.0f);
|
||||
const std::uint32_t bits = static_cast<std::uint32_t>(scaled);
|
||||
target[i * 3 + 0] = static_cast<char>(bits & 0xFFu);
|
||||
target[i * 3 + 1] = static_cast<char>((bits >> 8) & 0xFFu);
|
||||
target[i * 3 + 2] = static_cast<char>((bits >> 16) & 0xFFu);
|
||||
}
|
||||
}
|
||||
|
||||
WavWriter::~WavWriter() {
|
||||
if (file_ != nullptr) {
|
||||
std::fclose(file_);
|
||||
file_ = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
Status WavWriter::open(const std::string& path, std::uint32_t channels, std::uint32_t rate,
|
||||
SampleFormat format, std::uint64_t total_frames) {
|
||||
if (channels == 0 || rate == 0) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kOutput,
|
||||
"WAV writer needs a positive channel count and rate");
|
||||
}
|
||||
path_ = path;
|
||||
channels_ = channels;
|
||||
total_frames_ = total_frames;
|
||||
frames_written_ = 0;
|
||||
finalized_ = false;
|
||||
info_ = WavInfo{};
|
||||
info_.format = format;
|
||||
|
||||
const std::string fmt = fmt_chunk(channels, rate, format, &info_);
|
||||
const std::uint64_t data_size = total_frames * info_.block_align;
|
||||
info_.data_bytes = data_size;
|
||||
const std::uint64_t riff_file_size = 12u + 8u + fmt.size() + 8u + data_size;
|
||||
const bool rf64 = (riff_file_size - 8u) > 0xFFFFFFFFull;
|
||||
info_.rf64 = rf64;
|
||||
|
||||
std::string header;
|
||||
if (rf64) {
|
||||
const std::uint64_t file_size = 12u + 36u + 8u + fmt.size() + 8u + data_size;
|
||||
header.append("RF64", 4);
|
||||
put_u32(&header, 0xFFFFFFFFu);
|
||||
header.append("WAVE", 4);
|
||||
header.append("ds64", 4);
|
||||
put_u32(&header, 28);
|
||||
put_u64(&header, file_size - 8u);
|
||||
put_u64(&header, data_size);
|
||||
put_u64(&header, total_frames);
|
||||
put_u32(&header, 0);
|
||||
} else {
|
||||
header.append("RIFF", 4);
|
||||
put_u32(&header, static_cast<std::uint32_t>(riff_file_size - 8u));
|
||||
header.append("WAVE", 4);
|
||||
}
|
||||
header.append("fmt ", 4);
|
||||
put_u32(&header, static_cast<std::uint32_t>(fmt.size()));
|
||||
header.append(fmt);
|
||||
header.append("data", 4);
|
||||
put_u32(&header, rf64 ? 0xFFFFFFFFu : static_cast<std::uint32_t>(data_size));
|
||||
|
||||
file_ = fs_utf8::fopen(path, "wb");
|
||||
if (file_ == nullptr) {
|
||||
// The reference creates the parent directory itself.
|
||||
std::error_code ignored;
|
||||
const std::filesystem::path parent = std::filesystem::path(path).parent_path();
|
||||
if (!parent.empty()) {
|
||||
std::filesystem::create_directories(parent, ignored);
|
||||
}
|
||||
file_ = fs_utf8::fopen(path, "wb");
|
||||
}
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_OPEN, stage::kOutput, "cannot open " + path);
|
||||
}
|
||||
if (std::fwrite(header.data(), 1, header.size(), file_) != header.size()) {
|
||||
std::fclose(file_);
|
||||
file_ = nullptr;
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "cannot write header to " + path);
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status WavWriter::write(const double* interleaved, std::size_t frames) {
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "WAV writer is not open");
|
||||
}
|
||||
if (frames == 0) {
|
||||
return Status::success();
|
||||
}
|
||||
if (frames_written_ + frames > total_frames_) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput,
|
||||
"WAV writer received more frames than the header declared (declared " +
|
||||
std::to_string(total_frames_) + ", written " +
|
||||
std::to_string(frames_written_) + ", requested " +
|
||||
std::to_string(frames) + ")");
|
||||
}
|
||||
const std::size_t count = frames * channels_;
|
||||
if (info_.format == SampleFormat::Float32) {
|
||||
std::vector<float> converted(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
converted[i] = static_cast<float>(interleaved[i]);
|
||||
}
|
||||
if (std::fwrite(converted.data(), sizeof(float), count, file_) != count) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "write failed for " + path_);
|
||||
}
|
||||
} else {
|
||||
std::vector<float> converted(count);
|
||||
for (std::size_t i = 0; i < count; ++i) {
|
||||
converted[i] = static_cast<float>(interleaved[i]);
|
||||
}
|
||||
std::string packed;
|
||||
pack_int24(converted.data(), frames, channels_, &packed);
|
||||
if (std::fwrite(packed.data(), 1, packed.size(), file_) != packed.size()) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "write failed for " + path_);
|
||||
}
|
||||
}
|
||||
frames_written_ += frames;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status WavWriter::finalize() {
|
||||
if (file_ == nullptr) {
|
||||
return Status::fail(JOC_ERR_STATE, stage::kOutput, "WAV writer is not open");
|
||||
}
|
||||
if (frames_written_ != total_frames_) {
|
||||
std::fclose(file_);
|
||||
file_ = nullptr;
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput,
|
||||
"WAV writer wrote " + std::to_string(frames_written_) + " of " +
|
||||
std::to_string(total_frames_) + " frames");
|
||||
}
|
||||
const int result = std::fclose(file_);
|
||||
file_ = nullptr;
|
||||
finalized_ = true;
|
||||
if (result != 0) {
|
||||
return Status::fail(JOC_ERR_OUTPUT_WRITE, stage::kOutput, "close failed for " + path_);
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,61 @@
|
||||
// Port of src/speaker_wav.py.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <string>
|
||||
|
||||
#include "foundation/status.h"
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
enum class SampleFormat { Float32, Int24 };
|
||||
|
||||
struct WavInfo {
|
||||
SampleFormat format = SampleFormat::Float32;
|
||||
std::uint32_t bits_per_sample = 32;
|
||||
std::uint32_t bytes_per_sample = 4;
|
||||
std::uint32_t block_align = 0;
|
||||
std::uint64_t data_bytes = 0;
|
||||
bool rf64 = false;
|
||||
};
|
||||
|
||||
// int24 packing shared by the WAV and ADM writers:
|
||||
// trunc(clip(v, -1, 1) * 8388607.0f) with the low three bytes written LE.
|
||||
// NaN follows NumPy's float->int cast (INT32_MIN) so that the C++ conversion is
|
||||
// never undefined; the reference passes it through unguarded (plan TD-3.11).
|
||||
void pack_int24(const float* interleaved, std::size_t frames, std::size_t channels,
|
||||
std::string* out);
|
||||
|
||||
class WavWriter {
|
||||
public:
|
||||
WavWriter() = default;
|
||||
~WavWriter();
|
||||
|
||||
WavWriter(const WavWriter&) = delete;
|
||||
WavWriter& operator=(const WavWriter&) = delete;
|
||||
|
||||
// `total_frames` must be known up front: the header depends on it.
|
||||
Status open(const std::string& path, std::uint32_t channels, std::uint32_t rate,
|
||||
SampleFormat format, std::uint64_t total_frames);
|
||||
|
||||
Status write(const double* interleaved, std::size_t frames);
|
||||
|
||||
Status finalize();
|
||||
|
||||
const WavInfo& info() const { return info_; }
|
||||
std::uint64_t frames_written() const { return frames_written_; }
|
||||
|
||||
private:
|
||||
std::FILE* file_ = nullptr;
|
||||
std::string path_;
|
||||
WavInfo info_;
|
||||
std::uint32_t channels_ = 0;
|
||||
std::uint64_t total_frames_ = 0;
|
||||
std::uint64_t frames_written_ = 0;
|
||||
bool finalized_ = false;
|
||||
};
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,215 @@
|
||||
#include "io/zip_reader.h"
|
||||
#include "foundation/fs_utf8.h"
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
|
||||
#include "io/inflate.h"
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
namespace {
|
||||
|
||||
constexpr std::uint32_t kLocalHeaderSignature = 0x04034b50u;
|
||||
constexpr std::uint32_t kCentralHeaderSignature = 0x02014b50u;
|
||||
constexpr std::uint32_t kEndOfCentralDirectory = 0x06054b50u;
|
||||
|
||||
std::uint16_t read_u16(const std::uint8_t* p) {
|
||||
return static_cast<std::uint16_t>(p[0] | (p[1] << 8));
|
||||
}
|
||||
|
||||
std::uint32_t read_u32(const std::uint8_t* p) {
|
||||
return static_cast<std::uint32_t>(p[0]) | (static_cast<std::uint32_t>(p[1]) << 8) |
|
||||
(static_cast<std::uint32_t>(p[2]) << 16) | (static_cast<std::uint32_t>(p[3]) << 24);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::uint32_t crc32_of(const std::uint8_t* data, std::size_t size) {
|
||||
static std::uint32_t table[256];
|
||||
static bool ready = false;
|
||||
if (!ready) {
|
||||
for (std::uint32_t i = 0; i < 256; ++i) {
|
||||
std::uint32_t value = i;
|
||||
for (int bit = 0; bit < 8; ++bit) {
|
||||
value = (value & 1u) ? (0xEDB88320u ^ (value >> 1)) : (value >> 1);
|
||||
}
|
||||
table[i] = value;
|
||||
}
|
||||
ready = true;
|
||||
}
|
||||
std::uint32_t crc = 0xFFFFFFFFu;
|
||||
for (std::size_t i = 0; i < size; ++i) {
|
||||
crc = table[(crc ^ data[i]) & 0xFFu] ^ (crc >> 8);
|
||||
}
|
||||
return crc ^ 0xFFFFFFFFu;
|
||||
}
|
||||
|
||||
bool ZipArchive::open(const std::string& path, std::string* error) {
|
||||
entries_.clear();
|
||||
data_.clear();
|
||||
std::FILE* file = fs_utf8::fopen(path, "rb");
|
||||
if (file == nullptr) {
|
||||
if (error != nullptr) {
|
||||
*error = "cannot open " + path;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
std::fseek(file, 0, SEEK_END);
|
||||
const long long size = std::ftell(file);
|
||||
std::fseek(file, 0, SEEK_SET);
|
||||
if (size <= 0) {
|
||||
std::fclose(file);
|
||||
if (error != nullptr) {
|
||||
*error = "empty file " + path;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
data_.resize(static_cast<std::size_t>(size));
|
||||
const std::size_t got = std::fread(data_.data(), 1, data_.size(), file);
|
||||
std::fclose(file);
|
||||
if (got != data_.size()) {
|
||||
if (error != nullptr) {
|
||||
*error = "short read on " + path;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
std::size_t eocd = std::string::npos;
|
||||
const std::size_t scan_start = data_.size() > 65557u ? data_.size() - 65557u : 0u;
|
||||
for (std::size_t i = data_.size(); i-- > scan_start;) {
|
||||
if (i + 4u <= data_.size() && read_u32(&data_[i]) == kEndOfCentralDirectory) {
|
||||
eocd = i;
|
||||
break;
|
||||
}
|
||||
if (i == 0) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (eocd == std::string::npos || eocd + 22u > data_.size()) {
|
||||
if (error != nullptr) {
|
||||
*error = "not a zip archive (no end-of-central-directory)";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
const std::uint16_t entry_count = read_u16(&data_[eocd + 10]);
|
||||
const std::uint32_t directory_offset = read_u32(&data_[eocd + 16]);
|
||||
if (directory_offset >= data_.size()) {
|
||||
if (error != nullptr) {
|
||||
*error = "central directory offset out of range";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
std::size_t cursor = directory_offset;
|
||||
for (std::uint16_t index = 0; index < entry_count; ++index) {
|
||||
if (cursor + 46u > data_.size() || read_u32(&data_[cursor]) != kCentralHeaderSignature) {
|
||||
if (error != nullptr) {
|
||||
*error = "malformed central directory entry " + std::to_string(index);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
ZipEntry entry;
|
||||
entry.method = read_u16(&data_[cursor + 10]);
|
||||
entry.crc32 = read_u32(&data_[cursor + 16]);
|
||||
entry.compressed_size = read_u32(&data_[cursor + 20]);
|
||||
entry.uncompressed_size = read_u32(&data_[cursor + 24]);
|
||||
const std::uint16_t name_length = read_u16(&data_[cursor + 28]);
|
||||
const std::uint16_t extra_length = read_u16(&data_[cursor + 30]);
|
||||
const std::uint16_t comment_length = read_u16(&data_[cursor + 32]);
|
||||
entry.local_header_offset = read_u32(&data_[cursor + 42]);
|
||||
if (entry.compressed_size == 0xFFFFFFFFu || entry.uncompressed_size == 0xFFFFFFFFu ||
|
||||
entry.local_header_offset == 0xFFFFFFFFu) {
|
||||
if (error != nullptr) {
|
||||
*error = "zip64 archives are not supported";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (cursor + 46u + name_length > data_.size()) {
|
||||
if (error != nullptr) {
|
||||
*error = "member name out of range";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
entry.name.assign(reinterpret_cast<const char*>(&data_[cursor + 46]), name_length);
|
||||
entries_.push_back(std::move(entry));
|
||||
cursor += 46u + name_length + extra_length + comment_length;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
const ZipEntry* ZipArchive::find(const std::string& name) const {
|
||||
for (const ZipEntry& entry : entries_) {
|
||||
if (entry.name == name) {
|
||||
return &entry;
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
bool ZipArchive::extract(const ZipEntry& entry, std::vector<std::uint8_t>* out,
|
||||
std::string* error) const {
|
||||
if (out == nullptr) {
|
||||
return false;
|
||||
}
|
||||
const std::size_t offset = entry.local_header_offset;
|
||||
if (offset + 30u > data_.size() || read_u32(&data_[offset]) != kLocalHeaderSignature) {
|
||||
if (error != nullptr) {
|
||||
*error = "bad local header for " + entry.name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
const std::uint16_t name_length = read_u16(&data_[offset + 26]);
|
||||
const std::uint16_t extra_length = read_u16(&data_[offset + 28]);
|
||||
const std::size_t start = offset + 30u + name_length + extra_length;
|
||||
if (start + entry.compressed_size > data_.size()) {
|
||||
if (error != nullptr) {
|
||||
*error = "member data out of range for " + entry.name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (entry.method == 0u) {
|
||||
out->assign(data_.begin() + static_cast<std::ptrdiff_t>(start),
|
||||
data_.begin() + static_cast<std::ptrdiff_t>(start + entry.compressed_size));
|
||||
} else if (entry.method == 8u) {
|
||||
if (!inflate_raw(&data_[start], entry.compressed_size, out)) {
|
||||
if (error != nullptr) {
|
||||
*error = "deflate error in " + entry.name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
if (error != nullptr) {
|
||||
*error = "unsupported compression method " + std::to_string(entry.method) + " for " +
|
||||
entry.name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (entry.uncompressed_size != 0u && out->size() != entry.uncompressed_size) {
|
||||
if (error != nullptr) {
|
||||
*error = "size mismatch for " + entry.name + " (" + std::to_string(out->size()) +
|
||||
" vs " + std::to_string(entry.uncompressed_size) + ")";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (entry.crc32 != 0u && crc32_of(out->data(), out->size()) != entry.crc32) {
|
||||
if (error != nullptr) {
|
||||
*error = "CRC mismatch for " + entry.name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ZipArchive::read_member(const std::string& name, std::vector<std::uint8_t>* out,
|
||||
std::string* error) const {
|
||||
const ZipEntry* entry = find(name);
|
||||
if (entry == nullptr) {
|
||||
if (error != nullptr) {
|
||||
*error = "member not found: " + name;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return extract(*entry, out, error);
|
||||
}
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,38 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
namespace joc::io {
|
||||
|
||||
struct ZipEntry {
|
||||
std::string name;
|
||||
std::uint16_t method = 0;
|
||||
std::uint32_t crc32 = 0;
|
||||
std::uint32_t compressed_size = 0;
|
||||
std::uint32_t uncompressed_size = 0;
|
||||
std::uint32_t local_header_offset = 0;
|
||||
};
|
||||
|
||||
class ZipArchive {
|
||||
public:
|
||||
bool open(const std::string& path, std::string* error);
|
||||
|
||||
const std::vector<ZipEntry>& entries() const { return entries_; }
|
||||
|
||||
const ZipEntry* find(const std::string& name) const;
|
||||
|
||||
bool extract(const ZipEntry& entry, std::vector<std::uint8_t>* out, std::string* error) const;
|
||||
|
||||
bool read_member(const std::string& name, std::vector<std::uint8_t>* out, std::string* error) const;
|
||||
|
||||
private:
|
||||
std::vector<std::uint8_t> data_;
|
||||
std::vector<ZipEntry> entries_;
|
||||
};
|
||||
|
||||
std::uint32_t crc32_of(const std::uint8_t* data, std::size_t size);
|
||||
|
||||
} // namespace joc::io
|
||||
@@ -0,0 +1,415 @@
|
||||
#include "joc_bitstream/joc_parser.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
#include "foundation/bit_reader.h"
|
||||
#include "joc_huffman_tables.h"
|
||||
|
||||
namespace joc::joc {
|
||||
|
||||
namespace {
|
||||
|
||||
struct NumChannelsEntry {
|
||||
std::uint32_t config;
|
||||
std::int16_t channels;
|
||||
};
|
||||
constexpr NumChannelsEntry kNumChannels[] = {
|
||||
{0u, 5}, {1u, 7}, {2u, 7}, {3u, 5}, {4u, 7},
|
||||
};
|
||||
|
||||
struct NumBandsEntry {
|
||||
std::uint32_t index;
|
||||
std::int16_t bands;
|
||||
};
|
||||
constexpr NumBandsEntry kNumBands[] = {
|
||||
{0u, 1}, {1u, 3}, {2u, 5}, {3u, 7}, {4u, 9}, {5u, 12}, {6u, 15}, {7u, 23},
|
||||
};
|
||||
|
||||
// Floored modulo: Python's % operator semantics, so that the ported
|
||||
inline std::int64_t floored_mod(std::int64_t value, std::int64_t modulus) {
|
||||
const std::int64_t remainder = value % modulus;
|
||||
return remainder < 0 ? remainder + modulus : remainder;
|
||||
}
|
||||
|
||||
enum class SymbolKind { Mtx, Idx, Vec };
|
||||
|
||||
struct Tree {
|
||||
const int (*nodes)[2] = nullptr;
|
||||
int count = 0;
|
||||
};
|
||||
|
||||
Tree select_tree(std::uint32_t quant_idx, SymbolKind kind, int n_channels) {
|
||||
Tree tree;
|
||||
switch (kind) {
|
||||
case SymbolKind::Idx:
|
||||
if (n_channels == 5) {
|
||||
tree.nodes = joc_huff_code_5ch_pos_index_sparse;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_5ch_pos_index_sparse) /
|
||||
sizeof(joc_huff_code_5ch_pos_index_sparse[0]));
|
||||
} else {
|
||||
tree.nodes = joc_huff_code_7ch_pos_index_sparse;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_7ch_pos_index_sparse) /
|
||||
sizeof(joc_huff_code_7ch_pos_index_sparse[0]));
|
||||
}
|
||||
break;
|
||||
case SymbolKind::Vec:
|
||||
if (quant_idx == 0u) {
|
||||
tree.nodes = joc_huff_code_coarse_coeff_sparse;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_coarse_coeff_sparse) /
|
||||
sizeof(joc_huff_code_coarse_coeff_sparse[0]));
|
||||
} else {
|
||||
tree.nodes = joc_huff_code_fine_coeff_sparse;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_fine_coeff_sparse) /
|
||||
sizeof(joc_huff_code_fine_coeff_sparse[0]));
|
||||
}
|
||||
break;
|
||||
case SymbolKind::Mtx:
|
||||
default:
|
||||
if (quant_idx == 0u) {
|
||||
tree.nodes = joc_huff_code_coarse_generic;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_coarse_generic) /
|
||||
sizeof(joc_huff_code_coarse_generic[0]));
|
||||
} else {
|
||||
tree.nodes = joc_huff_code_fine_generic;
|
||||
tree.count = static_cast<int>(sizeof(joc_huff_code_fine_generic) /
|
||||
sizeof(joc_huff_code_fine_generic[0]));
|
||||
}
|
||||
break;
|
||||
}
|
||||
return tree;
|
||||
}
|
||||
|
||||
// infinite loop (plan 40.4 BL-6).
|
||||
bool huff_decode(const Tree& tree, bits::BitReader& reader, std::int16_t* out_value) {
|
||||
int node = 0;
|
||||
int steps = 0;
|
||||
while (node >= 0) {
|
||||
if (node >= tree.count || steps > tree.count) {
|
||||
reader.fail(JOC_ERR_JOC_SYNTAX, "Huffman tree walk left the valid node range");
|
||||
return false;
|
||||
}
|
||||
++steps;
|
||||
const std::uint32_t bit = reader.read(1);
|
||||
if (reader.failed()) {
|
||||
return false;
|
||||
}
|
||||
node = tree.nodes[node][bit];
|
||||
}
|
||||
*out_value = static_cast<std::int16_t>(-node - 1);
|
||||
return true;
|
||||
}
|
||||
|
||||
Status syntax_fail(const std::string& message) {
|
||||
return Status::fail(JOC_ERR_JOC_SYNTAX, stage::kJoc, message);
|
||||
}
|
||||
|
||||
Status truncated_fail(const bits::BitReader& reader) {
|
||||
if (reader.error() == JOC_ERR_JOC_SYNTAX) {
|
||||
return syntax_fail(reader.error_message());
|
||||
}
|
||||
return Status::fail(JOC_ERR_BITSTREAM_TRUNCATED, stage::kJoc,
|
||||
std::string("JOC bitstream truncated: ") + reader.error_message());
|
||||
}
|
||||
|
||||
void reconstruct_dense(const ObjectSymbols& symbols, std::uint32_t dp, int n_channels, std::int64_t nquant,
|
||||
std::int64_t offset,
|
||||
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS]) {
|
||||
for (int ch = 0; ch < n_channels; ++ch) {
|
||||
q[dp][ch][0] =
|
||||
floored_mod(offset + static_cast<std::int64_t>(symbols.mtx[dp][ch][0]), nquant);
|
||||
for (int pb = 1; pb < symbols.n_bands; ++pb) {
|
||||
q[dp][ch][pb] = floored_mod(
|
||||
q[dp][ch][pb - 1] + static_cast<std::int64_t>(symbols.mtx[dp][ch][pb]), nquant);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// across parameter bands and is deliberately NOT reset when the active channel
|
||||
Status reconstruct_sparse(
|
||||
const ObjectSymbols& symbols, std::uint32_t dp, int n_channels, std::int64_t nquant, std::int64_t offset,
|
||||
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS]) {
|
||||
if (n_channels != 5 && n_channels != 7) {
|
||||
return Status::fail(JOC_ERR_JOC_UNSUPPORTED_VARIANT, stage::kJoc,
|
||||
"sparse JOC requires 5 or 7 core channels, got " +
|
||||
std::to_string(n_channels));
|
||||
}
|
||||
const int initial_channel = symbols.idx[dp][0];
|
||||
if (initial_channel < 0 || initial_channel >= n_channels) {
|
||||
return syntax_fail("sparse JOC initial channel " + std::to_string(initial_channel) +
|
||||
" out of range for " + std::to_string(n_channels) + " channels");
|
||||
}
|
||||
// Non-active entries take nquant/2, which dequantizes to exactly zero.
|
||||
for (int ch = 0; ch < n_channels; ++ch) {
|
||||
for (int pb = 0; pb < symbols.n_bands; ++pb) {
|
||||
q[dp][ch][pb] = nquant / 2;
|
||||
}
|
||||
}
|
||||
int active = initial_channel;
|
||||
std::int64_t coefficient = offset;
|
||||
for (int pb = 0; pb < symbols.n_bands; ++pb) {
|
||||
if (pb != 0) {
|
||||
active = static_cast<int>(
|
||||
floored_mod(static_cast<std::int64_t>(active) + symbols.idx[dp][pb], n_channels));
|
||||
}
|
||||
coefficient = floored_mod(coefficient + static_cast<std::int64_t>(symbols.vec[dp][pb]), nquant);
|
||||
q[dp][active][pb] = coefficient;
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
std::int16_t num_channels_for_config(std::uint32_t dmx_config_idx) {
|
||||
for (const NumChannelsEntry& entry : kNumChannels) {
|
||||
if (entry.config == dmx_config_idx) {
|
||||
return entry.channels;
|
||||
}
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
std::int16_t num_bands_for_index(std::uint32_t num_bands_idx) {
|
||||
for (const NumBandsEntry& entry : kNumBands) {
|
||||
if (entry.index == num_bands_idx) {
|
||||
return entry.bands;
|
||||
}
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
Status parse_id14(const std::uint8_t* payload, std::size_t payload_size, joc_frame_params* out,
|
||||
FrameSymbols* symbols) {
|
||||
if (payload == nullptr || out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kJoc, "null payload or output");
|
||||
}
|
||||
std::memset(out, 0, sizeof(*out));
|
||||
out->struct_size = sizeof(joc_frame_params);
|
||||
out->struct_version = JOC_FRAME_PARAMS_VERSION;
|
||||
// capture is purely additive and never changes the parse result.
|
||||
FrameSymbols local_symbols{};
|
||||
FrameSymbols& capture = (symbols != nullptr) ? *symbols : local_symbols;
|
||||
std::memset(&capture, 0, sizeof(capture));
|
||||
|
||||
bits::BitReader reader(payload, payload_size);
|
||||
|
||||
out->dmx_config_idx = static_cast<std::uint8_t>(reader.read(3));
|
||||
out->num_objects_bits = static_cast<std::uint8_t>(reader.read(6));
|
||||
out->ext_config_idx = static_cast<std::uint8_t>(reader.read(3));
|
||||
|
||||
const std::uint32_t n_objects = static_cast<std::uint32_t>(out->num_objects_bits) + 1u;
|
||||
const std::int16_t n_channels = num_channels_for_config(out->dmx_config_idx);
|
||||
if (n_channels < 0) {
|
||||
return syntax_fail("unknown JOC downmix configuration " +
|
||||
std::to_string(out->dmx_config_idx));
|
||||
}
|
||||
out->n_channels = static_cast<std::uint8_t>(n_channels);
|
||||
if (n_objects > JOC_MAX_OBJECTS) {
|
||||
// The reference implementation has no check here and fails later inside
|
||||
// NumPy; the port reports it explicitly (plan 28.2).
|
||||
return Status::fail(JOC_ERR_JOC_UNSUPPORTED_VARIANT, stage::kJoc,
|
||||
"JOC frame declares " + std::to_string(n_objects) +
|
||||
" objects, the ABI supports at most " +
|
||||
std::to_string(static_cast<int>(JOC_MAX_OBJECTS)));
|
||||
}
|
||||
out->n_objects = static_cast<std::uint8_t>(n_objects);
|
||||
|
||||
out->clipgain_x_bits = static_cast<std::uint8_t>(reader.read(3));
|
||||
out->clipgain_y_bits = static_cast<std::uint8_t>(reader.read(5));
|
||||
out->seq_count = reader.read(10);
|
||||
// clipgain = 1 + (y/32) * 2^(x-4). The reference multiplies by an exact
|
||||
// exactly for the whole legal range.
|
||||
out->clipgain = 1.0 + static_cast<double>(out->clipgain_y_bits) / 32.0 *
|
||||
std::ldexp(1.0, static_cast<int>(out->clipgain_x_bits) - 4);
|
||||
|
||||
for (std::uint32_t obj = 0; obj < n_objects; ++obj) {
|
||||
joc_object_params& info = out->objects[obj];
|
||||
info.present = static_cast<std::uint8_t>(reader.read(1));
|
||||
if (info.present == 0u) {
|
||||
continue;
|
||||
}
|
||||
info.num_bands_idx = static_cast<std::uint8_t>(reader.read(3));
|
||||
const std::int16_t bands = num_bands_for_index(info.num_bands_idx);
|
||||
if (bands < 0) {
|
||||
return syntax_fail("unknown JOC num_bands index " +
|
||||
std::to_string(info.num_bands_idx));
|
||||
}
|
||||
info.n_bands = static_cast<std::uint8_t>(bands);
|
||||
info.sparse = static_cast<std::uint8_t>(reader.read(1));
|
||||
info.quant_idx = static_cast<std::uint8_t>(reader.read(1));
|
||||
info.slope_idx = static_cast<std::uint8_t>(reader.read(1));
|
||||
info.num_dpoints_bits = static_cast<std::uint8_t>(reader.read(1));
|
||||
info.n_dpoints = static_cast<std::uint8_t>(info.num_dpoints_bits + 1u);
|
||||
if (info.slope_idx == 1u) {
|
||||
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
|
||||
info.offset_ts[dp] = static_cast<std::uint8_t>(reader.read(5) + 1u);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (reader.failed()) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
|
||||
for (std::uint32_t obj = 0; obj < n_objects; ++obj) {
|
||||
const joc_object_params& info = out->objects[obj];
|
||||
if (info.present == 0u) {
|
||||
continue;
|
||||
}
|
||||
ObjectSymbols* symbol = &capture.objects[obj];
|
||||
symbol->present = 1;
|
||||
symbol->sparse = info.sparse;
|
||||
symbol->n_bands = info.n_bands;
|
||||
symbol->n_dpoints = info.n_dpoints;
|
||||
symbol->n_channels = out->n_channels;
|
||||
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
|
||||
if (info.sparse == 1u) {
|
||||
const Tree idx_tree = select_tree(info.quant_idx, SymbolKind::Idx, n_channels);
|
||||
const std::uint32_t first = reader.read(3);
|
||||
if (reader.failed()) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
symbol->idx[dp][0] = static_cast<std::uint8_t>(first);
|
||||
for (int pb = 1; pb < info.n_bands; ++pb) {
|
||||
std::int16_t value = 0;
|
||||
if (!huff_decode(idx_tree, reader, &value)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
symbol->idx[dp][pb] = static_cast<std::uint8_t>(value);
|
||||
}
|
||||
const Tree vec_tree = select_tree(info.quant_idx, SymbolKind::Vec, n_channels);
|
||||
for (int pb = 0; pb < info.n_bands; ++pb) {
|
||||
std::int16_t value = 0;
|
||||
if (!huff_decode(vec_tree, reader, &value)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
symbol->vec[dp][pb] = value;
|
||||
}
|
||||
} else {
|
||||
const Tree mtx_tree = select_tree(info.quant_idx, SymbolKind::Mtx, n_channels);
|
||||
for (int ch = 0; ch < n_channels; ++ch) {
|
||||
for (int pb = 0; pb < info.n_bands; ++pb) {
|
||||
std::int16_t value = 0;
|
||||
if (!huff_decode(mtx_tree, reader, &value)) {
|
||||
return truncated_fail(reader);
|
||||
}
|
||||
symbol->mtx[dp][ch][pb] = value;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
out->data_end_bits = static_cast<std::uint32_t>(reader.position());
|
||||
out->trailing_bits = static_cast<std::uint32_t>(payload_size * 8u - reader.position());
|
||||
const std::size_t tail_offset = reader.position() / 8u;
|
||||
if (tail_offset < payload_size) {
|
||||
const std::size_t tail_bytes = std::min<std::size_t>(8u, payload_size - tail_offset);
|
||||
std::memcpy(out->tail_bytes, payload + tail_offset, tail_bytes);
|
||||
}
|
||||
|
||||
for (std::uint32_t obj = 0; obj < n_objects; ++obj) {
|
||||
const joc_object_params& info = out->objects[obj];
|
||||
if (info.present == 0u) {
|
||||
continue;
|
||||
}
|
||||
std::uint32_t mask_bit = 1u << obj;
|
||||
out->present_mask |= mask_bit;
|
||||
|
||||
const std::int64_t nquant = (info.quant_idx == 0u) ? 96 : 192;
|
||||
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS] = {};
|
||||
ObjectSymbols* symbol = &capture.objects[obj];
|
||||
|
||||
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
|
||||
if (info.sparse == 1u) {
|
||||
const std::int64_t offset = (info.quant_idx == 0u) ? 50 : 100;
|
||||
const Status status =
|
||||
reconstruct_sparse(*symbol, dp, n_channels, nquant, offset, q);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
} else {
|
||||
const std::int64_t offset = (info.quant_idx == 0u) ? 48 : 96;
|
||||
reconstruct_dense(*symbol, dp, n_channels, nquant, offset, q);
|
||||
}
|
||||
}
|
||||
|
||||
// Operand order and types are kept identical to the reference so the
|
||||
// result is bit-exact, not merely close.
|
||||
const double nquant_half = static_cast<double>(nquant) / 2.0;
|
||||
const double denominator = 4096.0 * static_cast<double>(1 + static_cast<int>(info.quant_idx));
|
||||
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
|
||||
for (int ch = 0; ch < n_channels; ++ch) {
|
||||
for (int pb = 0; pb < info.n_bands; ++pb) {
|
||||
const double value = static_cast<double>(q[dp][ch][pb]) - nquant_half;
|
||||
out->objects[obj].dq[dp][ch][pb] = value * 820.0 / denominator;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (std::uint32_t dp = 0; dp < info.n_dpoints; ++dp) {
|
||||
for (int ch = 0; ch < n_channels; ++ch) {
|
||||
for (int pb = 0; pb < info.n_bands; ++pb) {
|
||||
symbol->q[dp][ch][pb] = q[dp][ch][pb];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
capture.n_objects = out->n_objects;
|
||||
capture.n_channels = out->n_channels;
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
Status parse_eac3_frame(const std::uint8_t* frame, std::size_t frame_size, joc_frame_params* out,
|
||||
emdf::Container* container, FrameSymbols* symbols) {
|
||||
if (frame == nullptr || out == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kJoc, "null frame or output");
|
||||
}
|
||||
emdf::Container local;
|
||||
const Status status = emdf::find_joc_emdf(frame, frame_size, &local);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (container != nullptr) {
|
||||
*container = local;
|
||||
}
|
||||
const emdf::Payload* payload = local.find(emdf::kIdJoc);
|
||||
if (payload == nullptr) {
|
||||
return Status::fail(JOC_ERR_EMDF_TRANSPORT, stage::kEmdf,
|
||||
"EMDF container has no ID14 (JOC) payload");
|
||||
}
|
||||
std::vector<std::uint8_t> bytes;
|
||||
const Status extract = emdf::extract_payload_bytes(frame, frame_size, *payload, &bytes);
|
||||
if (!extract.ok()) {
|
||||
return extract;
|
||||
}
|
||||
return parse_id14(bytes.data(), bytes.size(), out, symbols);
|
||||
}
|
||||
|
||||
Status check_id14_padding(const std::uint8_t* payload, std::size_t payload_size,
|
||||
std::uint32_t* out_trailing_bits) {
|
||||
joc_frame_params params;
|
||||
FrameSymbols symbols;
|
||||
const Status status = parse_id14(payload, payload_size, ¶ms, &symbols);
|
||||
if (!status.ok()) {
|
||||
return status;
|
||||
}
|
||||
if (out_trailing_bits != nullptr) {
|
||||
*out_trailing_bits = params.trailing_bits;
|
||||
}
|
||||
if (params.trailing_bits > 7u) {
|
||||
return Status::fail(JOC_ERR_BITSTREAM_PADDING, stage::kJoc,
|
||||
"more than 7 bits left after joc_data (" +
|
||||
std::to_string(params.trailing_bits) + ")");
|
||||
}
|
||||
for (std::size_t bit = params.data_end_bits; bit < payload_size * 8u; ++bit) {
|
||||
if (((payload[bit >> 3] >> (7u - (bit & 7u))) & 1u) != 0u) {
|
||||
return Status::fail(JOC_ERR_BITSTREAM_PADDING, stage::kJoc,
|
||||
"non-zero trailing padding bit at " + std::to_string(bit));
|
||||
}
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::joc
|
||||
@@ -0,0 +1,46 @@
|
||||
// Port of src/joc_decode.py.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
#include "joc_core.h"
|
||||
#include "emdf/emdf_parser.h"
|
||||
#include "foundation/status.h"
|
||||
|
||||
namespace joc::joc {
|
||||
|
||||
struct ObjectSymbols {
|
||||
std::uint8_t present = 0;
|
||||
std::uint8_t sparse = 0;
|
||||
std::uint8_t n_bands = 0;
|
||||
std::uint8_t n_dpoints = 0;
|
||||
std::uint8_t n_channels = 0;
|
||||
std::int64_t q[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS] = {};
|
||||
std::int16_t mtx[JOC_MAX_DPOINTS][JOC_MAX_CORE_CHANNELS][JOC_MAX_PARAMETER_BANDS] = {};
|
||||
std::uint8_t idx[JOC_MAX_DPOINTS][JOC_MAX_PARAMETER_BANDS] = {};
|
||||
std::int16_t vec[JOC_MAX_DPOINTS][JOC_MAX_PARAMETER_BANDS] = {};
|
||||
};
|
||||
|
||||
struct FrameSymbols {
|
||||
std::uint8_t n_objects = 0;
|
||||
std::uint8_t n_channels = 0;
|
||||
ObjectSymbols objects[JOC_MAX_OBJECTS];
|
||||
};
|
||||
|
||||
Status parse_id14(const std::uint8_t* payload, std::size_t payload_size, joc_frame_params* out,
|
||||
FrameSymbols* symbols);
|
||||
|
||||
Status parse_eac3_frame(const std::uint8_t* frame, std::size_t frame_size, joc_frame_params* out,
|
||||
emdf::Container* container, FrameSymbols* symbols);
|
||||
|
||||
Status check_id14_padding(const std::uint8_t* payload, std::size_t payload_size,
|
||||
std::uint32_t* out_trailing_bits);
|
||||
|
||||
std::int16_t num_channels_for_config(std::uint32_t dmx_config_idx);
|
||||
|
||||
std::int16_t num_bands_for_index(std::uint32_t num_bands_idx);
|
||||
|
||||
} // namespace joc::joc
|
||||
@@ -125,6 +125,10 @@ public:
|
||||
return error_[0] ? error_ : "";
|
||||
}
|
||||
|
||||
void set_dc_filter(const bool enabled) noexcept {
|
||||
dc_filter_enabled_ = enabled;
|
||||
}
|
||||
|
||||
int process(
|
||||
const float* bed5,
|
||||
const float* lfe,
|
||||
@@ -305,16 +309,18 @@ private:
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
dc_buffer[20 + i] = current[i][0];
|
||||
}
|
||||
for (int slot = 0; slot < 4; ++slot) {
|
||||
Complex sum{0.0, 0.0};
|
||||
for (int tap = 0; tap < 21; ++tap) {
|
||||
const Complex sample = dc_buffer[slot + tap];
|
||||
const double cr = kDcB[tap];
|
||||
const double ci = kDcA[tap];
|
||||
sum.re += sample.re * cr - sample.im * ci;
|
||||
sum.im += sample.re * ci + sample.im * cr;
|
||||
if (dc_filter_enabled_) {
|
||||
for (int slot = 0; slot < 4; ++slot) {
|
||||
Complex sum{0.0, 0.0};
|
||||
for (int tap = 0; tap < 21; ++tap) {
|
||||
const Complex sample = dc_buffer[slot + tap];
|
||||
const double cr = kDcB[tap];
|
||||
const double ci = kDcA[tap];
|
||||
sum.re += sample.re * cr - sample.im * ci;
|
||||
sum.im += sample.re * ci + sample.im * cr;
|
||||
}
|
||||
x_[channel][0][group + slot] = {2.0 * sum.re, 2.0 * sum.im};
|
||||
}
|
||||
x_[channel][0][group + slot] = {2.0 * sum.re, 2.0 * sum.im};
|
||||
}
|
||||
for (int i = 0; i < 20; ++i) {
|
||||
surround_history_[surround][i] = dc_buffer[i + 4];
|
||||
@@ -641,6 +647,8 @@ private:
|
||||
float analysis_phase_;
|
||||
alignas(64) Complex surround_delay_[2][10][64];
|
||||
alignas(64) Complex surround_history_[2][20];
|
||||
// band-0 的 21-tap DC 补偿开关;仅 downmix 配置 3/4 由调用方置位。
|
||||
bool dc_filter_enabled_ = true;
|
||||
alignas(64) double lfe_delay_[kLfeDelay];
|
||||
alignas(64) double matrix_previous_[15][5][64];
|
||||
alignas(64) double synthesis_state_[15][640];
|
||||
@@ -696,6 +704,14 @@ int EJOC_CALL ejoc_renderer_set_threads(ejoc_renderer_handle handle, uint32_t to
|
||||
return static_cast<ejoc::Renderer*>(handle)->set_threads(total_threads);
|
||||
}
|
||||
|
||||
int EJOC_CALL ejoc_renderer_set_dc_filter(ejoc_renderer_handle handle, uint32_t enabled) {
|
||||
if (!handle) {
|
||||
return -1;
|
||||
}
|
||||
static_cast<ejoc::Renderer*>(handle)->set_dc_filter(enabled != 0);
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint32_t EJOC_CALL ejoc_renderer_thread_count(ejoc_renderer_handle handle) {
|
||||
if (!handle) {
|
||||
return 0;
|
||||
@@ -0,0 +1,80 @@
|
||||
#include "joc_core/objects16.h"
|
||||
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
|
||||
namespace joc::joc {
|
||||
|
||||
Status rebuild_objects16(ejoc_renderer_handle handle, const joc_frame_params& params,
|
||||
const float* bed5_planar, const float* lfe, float gain,
|
||||
std::vector<float>* out16, std::string* error) {
|
||||
if (handle == nullptr || bed5_planar == nullptr || out16 == nullptr) {
|
||||
return Status::fail(JOC_ERR_INVALID_ARGUMENT, stage::kDsp, "null argument");
|
||||
}
|
||||
if (params.n_channels != JOC_CORE_CHANNELS) {
|
||||
if (error != nullptr) {
|
||||
*error = "the reused JOC kernel requires 5 core channels, frame declares " +
|
||||
std::to_string(static_cast<unsigned>(params.n_channels));
|
||||
}
|
||||
return Status::fail(JOC_ERR_JOC_UNSUPPORTED_VARIANT, stage::kDsp, *error);
|
||||
}
|
||||
|
||||
std::vector<double> dq(static_cast<std::size_t>(JOC_MAX_OBJECTS) * JOC_MAX_DPOINTS *
|
||||
JOC_CORE_CHANNELS * JOC_MAX_PARAMETER_BANDS,
|
||||
0.0);
|
||||
std::uint8_t n_bands[JOC_MAX_OBJECTS] = {};
|
||||
std::uint8_t n_dpoints[JOC_MAX_OBJECTS] = {};
|
||||
std::uint8_t slope_idx[JOC_MAX_OBJECTS] = {};
|
||||
std::uint8_t offset_ts[JOC_MAX_OBJECTS * JOC_MAX_DPOINTS] = {};
|
||||
for (unsigned obj = 0; obj < JOC_MAX_OBJECTS; ++obj) {
|
||||
const joc_object_params& object = params.objects[obj];
|
||||
if (object.present == 0) {
|
||||
continue;
|
||||
}
|
||||
n_bands[obj] = object.n_bands;
|
||||
n_dpoints[obj] = object.n_dpoints;
|
||||
slope_idx[obj] = object.slope_idx;
|
||||
for (unsigned dp = 0; dp < JOC_MAX_DPOINTS; ++dp) {
|
||||
offset_ts[obj * JOC_MAX_DPOINTS + dp] = object.offset_ts[dp];
|
||||
}
|
||||
for (unsigned dp = 0; dp < object.n_dpoints; ++dp) {
|
||||
for (unsigned ch = 0; ch < JOC_CORE_CHANNELS; ++ch) {
|
||||
for (unsigned pb = 0; pb < object.n_bands; ++pb) {
|
||||
const std::size_t index =
|
||||
((static_cast<std::size_t>(obj) * JOC_MAX_DPOINTS + dp) * JOC_CORE_CHANNELS +
|
||||
ch) *
|
||||
JOC_MAX_PARAMETER_BANDS +
|
||||
pb;
|
||||
dq[index] = object.dq[dp][ch][pb];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
out16->assign(static_cast<std::size_t>(JOC_OUTPUT_CHANNELS) * JOC_FRAME_SAMPLES, 0.0f);
|
||||
// band-0 的 21-tap DC 补偿只在 downmix 配置 3/4 下启用,其余配置 band 0
|
||||
// 走与其他 band 相同的处理。
|
||||
const bool dc_filter = params.dmx_config_idx == 3 || params.dmx_config_idx == 4;
|
||||
if (ejoc_renderer_set_dc_filter(handle, dc_filter ? 1u : 0u) != 0) {
|
||||
if (error != nullptr) {
|
||||
*error = "ejoc_renderer_set_dc_filter failed";
|
||||
}
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kDsp, "ejoc_renderer_set_dc_filter failed");
|
||||
}
|
||||
const int result = ejoc_renderer_process(
|
||||
handle, bed5_planar, lfe, params.present_mask, n_bands, n_dpoints, slope_idx, offset_ts,
|
||||
dq.data(), params.clipgain, 0.0625f, gain, out16->data());
|
||||
if (result != 0) {
|
||||
const char* detail = ejoc_renderer_last_error(handle);
|
||||
const std::string message =
|
||||
"ejoc_renderer_process failed (" + std::to_string(result) + "): " +
|
||||
(detail != nullptr ? detail : "unknown");
|
||||
if (error != nullptr) {
|
||||
*error = message;
|
||||
}
|
||||
return Status::fail(JOC_ERR_RENDER_FAILED, stage::kDsp, message);
|
||||
}
|
||||
return Status::success();
|
||||
}
|
||||
|
||||
} // namespace joc::joc
|
||||
@@ -0,0 +1,18 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "eac3joc_core.h"
|
||||
#include "foundation/status.h"
|
||||
#include "joc_core.h"
|
||||
|
||||
namespace joc::joc {
|
||||
|
||||
// [16][1536] planar float32. The kernel requires exactly 5 core channels.
|
||||
Status rebuild_objects16(ejoc_renderer_handle handle, const joc_frame_params& params,
|
||||
const float* bed5_planar, const float* lfe, float gain,
|
||||
std::vector<float>* out16, std::string* error);
|
||||
|
||||
} // namespace joc::joc
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user