Compare commits
6 Commits
v1.0.0
...
22a16ab60d
| Author | SHA1 | Date | |
|---|---|---|---|
| 22a16ab60d | |||
| 329445ed25 | |||
| 887b3e317f | |||
| 35536b0bd4 | |||
| 9b41fedb14 | |||
| 65a08e544d |
@@ -6,6 +6,8 @@ venv/
|
||||
|
||||
build/
|
||||
output/
|
||||
tests/
|
||||
lib/
|
||||
metadata_cache/
|
||||
|
||||
*.metadata.json
|
||||
@@ -13,3 +15,9 @@ metadata_cache/
|
||||
*.variant-error.json
|
||||
*.objects16.f32le
|
||||
|
||||
HRTF/
|
||||
|
||||
# Keep the production binaural regression test while local research fixtures stay ignored.
|
||||
!tests/
|
||||
tests/*
|
||||
!tests/test_binaural_production.py
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 TheM14
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
+50
-10
@@ -4,9 +4,9 @@
|
||||
|
||||
> JustOneCacophony is an experimental/test implementation of E-AC-3 JOC for studying JOC parsing, reconstruction, rendering, and the associated mathematics.
|
||||
|
||||
The project can extract and parse EMDF, ID14 JOC parameters, and ID11 OAMD metadata from common E-AC-3 JOC streams. It combines those data with the core 5.1 PCM decoded by FFmpeg, reconstructs LFE plus 15 object channels, and writes either ADM BWF or a WAV file for a selected speaker layout.
|
||||
The project can extract and parse EMDF, ID14 JOC parameters, and ID11 OAMD metadata from common E-AC-3 JOC streams. It combines those data with the core 5.1 PCM decoded by FFmpeg, reconstructs LFE plus 15 object channels, and writes ADM BWF, a WAV file for a selected speaker layout, or direct DLL-free Rosella binaural stereo.
|
||||
|
||||
This is research code, not a complete, standards-compliant, or production-grade Dolby JOC decoder. It covers only the stream forms currently implemented. Unknown variants fail explicitly—because when the math goes wrong, all that may remain is the cacophony.
|
||||
This is research code, not a complete, standards-compliant, or production-grade JOC decoder. It covers only the stream forms currently implemented. Unknown variants fail explicitly—because when the math goes wrong, all that may remain is the cacophony.
|
||||
|
||||
## Current features
|
||||
|
||||
@@ -16,7 +16,9 @@ This is research code, not a complete, standards-compliant, or production-grade
|
||||
- Reconstruct LFE plus 15 object channels through analysis QMF, parameter interpolation, the object matrix, and inverse QMF.
|
||||
- Write a 25-channel ADM BWF: a 10-channel 7.1.2 bed (silent except for LFE) plus 15 objects.
|
||||
- Render directly to `2.0`, `3.1`, `5.1`, `7.1`, `5.1.2`, `5.1.4`, `7.1.2`, `7.1.4`, `9.1.4`, or `9.1.6`.
|
||||
- Write float32 or PCM24 WAV and require an explicit policy when PCM24 would clip.
|
||||
- Run DLL-free Rosella binaural rendering directly from `pcm16 + ID11/OAMD`, without a temporary ADM BWF.
|
||||
- Keep the binaural DSP in float64/complex128, including 961-sample latency compensation, cross-frame state, and the room tail.
|
||||
- Use a shared float32/PCM24 WAV writer and explicit PCM24 clipping policy for direct outputs.
|
||||
- Use the NumPy backend or an optional C++20 core through `ctypes`; `auto` falls back to Python when the native library is unavailable.
|
||||
- Read or write metadata sidecars and produce metadata, timing, and output reports.
|
||||
|
||||
@@ -30,10 +32,11 @@ M4A / E-AC-3
|
||||
├─ ID11 OAMD → object positions and timing
|
||||
└─ LFE + 15 objects
|
||||
├─ 25ch ADM BWF
|
||||
└─ speaker WAV for the selected layout
|
||||
├─ speaker WAV for the selected layout
|
||||
└─ direct ID11 timeline → Rosella → binaural WAV
|
||||
```
|
||||
|
||||
The Python and C++ backends follow the same documented mathematics. The native core handles the state-heavy DSP and speaker rendering; high-level bitstream parsing, ADM assembly, and CLI behavior remain in Python.
|
||||
The Python and C++ backends follow the same mathematics. The native core handles object reconstruction, speaker rendering, and binaural QMF/hybrid/room/synthesis; bitstream parsing, the OAMD timeline, model parsing, and CLI behavior remain in Python.
|
||||
|
||||
## Requirements
|
||||
|
||||
@@ -78,12 +81,29 @@ python main.py input.m4a --speaker-layout 5.1 --speaker-format int24
|
||||
python main.py input.m4a --speaker-layout 7.1.2 --speaker-output output.7.1.2.wav
|
||||
```
|
||||
|
||||
When PCM24 may clip in a non-interactive environment, select a policy explicitly:
|
||||
Write Rosella binaural stereo directly (ordinary objects are Near/Mid/Far only; Mid is the default):
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --binaural
|
||||
python main.py input.m4a --binaural --binaural-mode near
|
||||
python main.py input.m4a --binaural --binaural-mode far --binaural-format int24
|
||||
python main.py input.m4a --binaural --binaural-output output.binaural.wav `
|
||||
--personalized-headphone C:\HRTF\my.personalized_headphone
|
||||
```
|
||||
|
||||
The default model path is `HRTF/binaural.personalized_headphone`. An example HRTF file is available from:
|
||||
|
||||
https://professionalsupport.dolby.com/s/question/0D54u0000AAT85HCQT/the-state-of-personalized-binaural-rendering?language=en_US
|
||||
|
||||
See [Binaural Rendering Mathematics](docs/binaural.en.md) for the formulas, state, and timing model.
|
||||
|
||||
Speaker and binaural output share peak analysis, the WAV writer, and clipping policy. When PCM24 may clip in a non-interactive environment, select a policy explicitly:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action abort
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action float32
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action continue
|
||||
python main.py input.m4a --binaural --binaural-format int24 --clip-action abort
|
||||
```
|
||||
|
||||
Metadata and diagnostics:
|
||||
@@ -95,6 +115,22 @@ python main.py input.m4a --metadata-cache metadata_cache
|
||||
python main.py input.m4a --metadata-dir metadata_cache
|
||||
```
|
||||
|
||||
### Experimental binaural mode settings for JOC objects
|
||||
|
||||
The binaural mode written here is a user-selected, experimental rendering hint for downstream ADM renderers. It is **not original binaural metadata extracted or recovered from the input E-AC-3 JOC bitstream**, nor does it represent the original mix's per-object binaural settings. The selected mode is applied uniformly to all 15 JOC objects; the default `unspecified` is this tool's default, not a mode detected in the source file.
|
||||
|
||||
Use `--joc-binaural-mode off|near|far|mid|unspecified` to select a mode, encoded as `0|1|2|3|4` respectively. The default is `unspecified`:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --joc-binaural-mode mid
|
||||
```
|
||||
|
||||
This option only sets the low 3 binaural-render-mode bits of the last 15 JOC object entries in ADM BWF DBMD segment 10, leaving the first 10 bed entries unchanged. It does not change PCM, object trajectories, direct speaker rendering, or direct Rosella binaural rendering. The adjacent `.report.json` records the mode name and value in `joc_binaural_mode` and `joc_binaural_mode_value`; both are `null` for direct speaker or direct binaural output, where the option does not apply.
|
||||
|
||||
### Binaural calculation
|
||||
|
||||
See [Binaural Rendering Mathematics](docs/binaural.en.md) for QMF, hybrid processing, direction fields, distance, ITD, room processing, 512-sample parameter updates, and 961-sample latency compensation.
|
||||
|
||||
For all options:
|
||||
|
||||
```powershell
|
||||
@@ -128,7 +164,7 @@ JustOneCacophony/
|
||||
├─ main.py command-line entry point
|
||||
├─ src/ Python implementation modules
|
||||
├─ native/ C/C++ acceleration core, C ABI, and required table data
|
||||
├─ data/ runtime table data for Python
|
||||
├─ data/ Python runtime table data
|
||||
├─ lib/ native runtime drop-in directory (create as needed)
|
||||
├─ docs/ math and native-core notes in both languages
|
||||
├─ requirements.txt Python dependency
|
||||
@@ -148,7 +184,8 @@ The main documented stages are:
|
||||
- OAMD Q15 coordinate conversion;
|
||||
- equal-power panning over target-layout regions;
|
||||
- layout-dependent position compensation and sample-wise gain ramps;
|
||||
- float32 and PCM24 output quantization.
|
||||
- float32 and PCM24 output quantization;
|
||||
- Rosella 64-band QMF, 77-band hybrid processing, `77×36` direction fields, distance/ITD, room FIR, and special LFE.
|
||||
|
||||
See the [mathematical notes](docs/math.en.md) for the equations used by the decoding and rendering process.
|
||||
|
||||
@@ -156,12 +193,15 @@ See the [mathematical notes](docs/math.en.md) for the equations used by the deco
|
||||
|
||||
- Only the common contiguous EMDF transport is covered. Fragmented transport across multiple audio-block skip fields is not covered.
|
||||
- Dense JOC is the main path. The Sparse JOC branch should not be treated as supported.
|
||||
- The speaker path currently covers ordinary point objects; extent, spread, divergence, and similar modes are outside the supported scope.
|
||||
- The speaker and Rosella binaural paths currently cover ordinary point objects; extent, spread, diffuse, divergence, channel lock, and similar controls are outside the supported scope.
|
||||
- OAMD trim elements are boundary-checked and skipped; warp, balance, and trim parameters are not applied to raw object trajectories or speaker rendering.
|
||||
- Multi-data-point streams, uncommon band configurations, and unusual OAMD scheduling have less coverage than common 12-band, single-data-point material.
|
||||
- A speaker limiter is outside the current primary formula.
|
||||
- ADM output, native binaries, and speaker layouts still need broader interoperability checks across platforms, players, and real material.
|
||||
- Rosella requires a user-supplied compatible `.personalized_headphone`; arbitrary SOFA data cannot become a valid Rosella rp through JSON rearrangement alone.
|
||||
- ADM output, native binaries, speaker layouts, and binaural models still need broader interoperability checks across platforms, players, and real material.
|
||||
|
||||
## Documentation
|
||||
|
||||
- [Mathematical notes](docs/math.en.md) · [中文](docs/math.md)
|
||||
- [Native-core notes](docs/native.en.md) · [中文](docs/native.md)
|
||||
- [DLL-free Rosella binaural](docs/binaural.en.md) · [中文](docs/binaural.md)
|
||||
|
||||
@@ -4,9 +4,9 @@
|
||||
|
||||
> JustOneCacophony 是一个 E-AC-3 JOC 的实验性 / 测试实现,用于研究 JOC 的解析、重建、渲染以及相关数学过程。
|
||||
|
||||
项目可以从常见 E-AC-3 JOC 码流中提取并解析 EMDF、ID14 JOC 参数和 ID11 OAMD 元数据,结合 FFmpeg 解码出的核心 5.1 PCM 重建 LFE 与 15 路对象 PCM,并输出 ADM BWF 或指定扬声器布局的 WAV。
|
||||
项目可以从常见 E-AC-3 JOC 码流中提取并解析 EMDF、ID14 JOC 参数和 ID11 OAMD 元数据,结合 FFmpeg 解码出的核心 5.1 PCM 重建 LFE 与 15 路对象 PCM,并输出 ADM BWF、指定扬声器布局的 WAV,或直接输出 DLL-free Rosella 双耳 WAV。
|
||||
|
||||
这是研究代码,不是完整、标准兼容或生产级的 Dolby JOC 解码器。它只覆盖当前已实现的码流形态;遇到未知变体时会明确报错,而不是假装一切都很和谐——如果哪里算错了,它可能就真的只剩 cacophony 了。
|
||||
这是研究代码,不是完整、标准兼容或生产级的 JOC 解码器。它只覆盖当前已实现的码流形态;遇到未知变体时会明确报错,而不是假装一切都很和谐——如果哪里算错了,它可能就真的只剩 cacophony 了。
|
||||
|
||||
## 当前功能
|
||||
|
||||
@@ -16,7 +16,9 @@
|
||||
- 通过 analysis QMF、参数插值、对象矩阵和 inverse QMF 重建 LFE + 15 路对象 PCM;
|
||||
- 输出 25 声道 ADM BWF:10 声道 7.1.2 bed(除 LFE 外静音)+ 15 个对象;
|
||||
- 直接渲染 `2.0`、`3.1`、`5.1`、`7.1`、`5.1.2`、`5.1.4`、`7.1.2`、`7.1.4`、`9.1.4`、`9.1.6`;
|
||||
- 输出 float32 或 PCM24 WAV,并在 PCM24 削波前提供明确处理策略;
|
||||
- 从 `pcm16 + ID11/OAMD` 直接运行 DLL-free Rosella 双耳渲染,不生成临时 ADM BWF;
|
||||
- 双耳 DSP 全程使用 float64/complex128,并保留 961-sample latency compensation、跨帧状态和 room 尾声;
|
||||
- 直接输出统一支持 float32 或 PCM24 WAV,并在 PCM24 削波前提供明确处理策略;
|
||||
- 使用 NumPy 后端,或通过 `ctypes` 调用可选的 C++20 原生核;`auto` 模式在原生库不可用时回退到 Python;
|
||||
- 读取或写入 metadata sidecar,并生成元数据、运行时间和输出摘要。
|
||||
|
||||
@@ -30,10 +32,11 @@ M4A / E-AC-3
|
||||
├─ ID11 OAMD → 对象位置与时间轨迹
|
||||
└─ LFE + 15 objects
|
||||
├─ 25ch ADM BWF
|
||||
└─ 指定布局的扬声器 WAV
|
||||
├─ 指定布局的扬声器 WAV
|
||||
└─ ID11 直接时间轴 → Rosella → 双耳 WAV
|
||||
```
|
||||
|
||||
Python 与 C++ 后端使用同一组已记录的数学过程。原生核只处理状态密集的 DSP 和扬声器渲染,高层位流解析、ADM 组装与命令行逻辑仍在 Python 中。
|
||||
Python 与 C++ 后端使用同一组数学过程。原生核处理对象重建、扬声器渲染以及双耳 QMF/hybrid/room/synthesis;位流解析、OAMD 时间轴、模型解析和命令行逻辑仍在 Python 中。
|
||||
|
||||
## 环境
|
||||
|
||||
@@ -78,12 +81,29 @@ python main.py input.m4a --speaker-layout 5.1 --speaker-format int24
|
||||
python main.py input.m4a --speaker-layout 7.1.2 --speaker-output output.7.1.2.wav
|
||||
```
|
||||
|
||||
在非交互环境请求 PCM24 且可能削波时,需要显式选择处理方式:
|
||||
直接输出 Rosella 双耳 WAV(普通对象仅 Near/Mid/Far,默认 Mid):
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --binaural
|
||||
python main.py input.m4a --binaural --binaural-mode near
|
||||
python main.py input.m4a --binaural --binaural-mode far --binaural-format int24
|
||||
python main.py input.m4a --binaural --binaural-output output.binaural.wav `
|
||||
--personalized-headphone C:\HRTF\my.personalized_headphone
|
||||
```
|
||||
|
||||
默认模型路径为 `HRTF/binaural.personalized_headphone`。示例 HRTF 文件见:
|
||||
|
||||
https://professionalsupport.dolby.com/s/question/0D54u0000AAT85HCQT/the-state-of-personalized-binaural-rendering?language=en_US
|
||||
|
||||
计算公式、状态和时间轴见[双耳渲染数学](docs/binaural.md)。
|
||||
|
||||
扬声器和双耳输出共享峰值检查、writer 与削波策略。在非交互环境请求 PCM24 且可能削波时,需要显式选择处理方式:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action abort
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action float32
|
||||
python main.py input.m4a --speaker-layout 5.1 --speaker-format int24 --clip-action continue
|
||||
python main.py input.m4a --binaural --binaural-format int24 --clip-action abort
|
||||
```
|
||||
|
||||
元数据与诊断:
|
||||
@@ -95,6 +115,22 @@ python main.py input.m4a --metadata-cache metadata_cache
|
||||
python main.py input.m4a --metadata-dir metadata_cache
|
||||
```
|
||||
|
||||
### 实验性 JOC 对象双耳模式设置
|
||||
|
||||
这里写入的双耳模式是用户手动指定、供下游 ADM 渲染器使用的实验性渲染提示,**不是从输入 E-AC-3 JOC 码流中提取或还原的原始双耳元数据**,也不代表原始混音中各对象的双耳设置。所选模式会统一应用到 15 个 JOC 对象;默认 `unspecified` 只是本工具的默认值,并非从源文件检测到的模式。
|
||||
|
||||
使用 `--joc-binaural-mode off|near|far|mid|unspecified` 选择模式,编码分别为 `0|1|2|3|4`,默认 `unspecified`:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a --joc-binaural-mode mid
|
||||
```
|
||||
|
||||
此选项仅设置 ADM BWF 的 DBMD segment 10 中后 15 个 JOC 对象的 binaural render mode 低 3 bit;前 10 个 bed 保持不变。它不改变 PCM、对象轨迹、直接扬声器渲染或直接 Rosella 双耳渲染。输出旁的 `.report.json` 用 `joc_binaural_mode` 和 `joc_binaural_mode_value` 记录模式名称与数值;直接扬声器或直接双耳输出时两者为 `null`,表示不适用。
|
||||
|
||||
### 双耳计算
|
||||
|
||||
双耳路径的 QMF、hybrid、方向场、距离、ITD、room、512-sample 参数更新和 961-sample 延迟补偿见[双耳渲染数学](docs/binaural.md)。
|
||||
|
||||
更多参数可查看:
|
||||
|
||||
```powershell
|
||||
@@ -148,7 +184,8 @@ JustOneCacophony/
|
||||
- OAMD Q15 坐标转换;
|
||||
- 基于目标布局 region 的等功率声像;
|
||||
- 布局位置补偿与逐样本增益斜坡;
|
||||
- float32 与 PCM24 输出量化。
|
||||
- float32 与 PCM24 输出量化;
|
||||
- Rosella 64-band QMF、77-band hybrid、`77×36` 方向 field、距离/ITD、room FIR 与 special LFE。
|
||||
|
||||
解码与渲染过程使用的公式见[数学说明](docs/math.md)。
|
||||
|
||||
@@ -156,12 +193,15 @@ JustOneCacophony/
|
||||
|
||||
- 当前只覆盖常见 continuous EMDF transport;跨多个 audio-block skip field 的碎片化 transport 尚未覆盖。
|
||||
- Dense JOC 是当前主要路径;Sparse JOC 分支不应视为受支持能力。
|
||||
- 扬声器路径当前只覆盖普通点对象;extent、spread、divergence 等对象模式不在支持范围内。
|
||||
- 扬声器与 Rosella 双耳路径当前只覆盖普通点对象;extent、spread、diffuse、divergence、channel lock 等对象控制不在支持范围内。
|
||||
- OAMD trim element 会按声明边界校验并跳过;warp、balance 和 trim 参数不应用于当前原始对象轨迹或扬声器渲染。
|
||||
- 多数据点、少见参数带配置和特殊 OAMD 调度的覆盖度低于常见 12-band、单数据点素材。
|
||||
- 扬声器 limiter 不属于当前实现的主公式。
|
||||
- ADM 输出、原生库和扬声器布局仍需在更多平台、播放器与真实素材上确认互操作性。
|
||||
- Rosella 路径需要用户提供兼容的 `.personalized_headphone`;任意 SOFA 不能仅靠 JSON 重排成为有效 Rosella rp。
|
||||
- ADM 输出、原生库、扬声器布局和双耳模型仍需在更多平台、播放器与真实素材上确认互操作性。
|
||||
|
||||
## 文档
|
||||
|
||||
- [数学说明](docs/math.md) · [English](docs/math.en.md)
|
||||
- [原生核说明](docs/native.md) · [English](docs/native.en.md)
|
||||
- [DLL-free Rosella 双耳](docs/binaural.md) · [English](docs/binaural.en.md)
|
||||
|
||||
+18
-1
@@ -2,7 +2,9 @@
|
||||
|
||||
[中文](README.md)
|
||||
|
||||
`tables.npz` contains the static table data used by the Python path:
|
||||
This directory contains static production tables and the user-model directory.
|
||||
|
||||
`tables.npz` contains the JOC core decoding tables:
|
||||
|
||||
```text
|
||||
analysis_window float64[10,64]
|
||||
@@ -18,3 +20,18 @@ joc_huff_code_7ch_pos_index_sparse int64[6,2]
|
||||
`src/joc_qmf.py` loads the QMF tables, while `src/joc_decode.py` loads the JOC Huffman trees. Python does not read C/C++ headers under `native/`.
|
||||
|
||||
The corresponding native data are stored in `native/src/qmf_tables.h` and `native/src/joc_huffman_tables.h`. Changes on either side should update the other and be checked for value-by-value agreement.
|
||||
|
||||
## Binaural rendering tables
|
||||
|
||||
`rosella_kernels.npz` contains the fixed QMF/hybrid tables:
|
||||
|
||||
```text
|
||||
qmf_analysis_coefficients float32[64,10]
|
||||
hybrid_analysis_low_kernel float32[3,2,13,16,2]
|
||||
hybrid_synthesis_indices int16[154,4]
|
||||
hybrid_synthesis_values float32[154]
|
||||
qmf_synthesis_basis float64[64,4,128]
|
||||
qmf_synthesis_taps float64[64,10,4]
|
||||
```
|
||||
|
||||
The float32 table values are promoted to float64 when loaded.
|
||||
|
||||
+18
-1
@@ -2,7 +2,9 @@
|
||||
|
||||
[English](README.en.md)
|
||||
|
||||
`tables.npz` 集中保存 Python 路径使用的静态表数据:
|
||||
本目录保存 Python 生产路径使用的静态表数据与用户模型目录。
|
||||
|
||||
`tables.npz` 保存 JOC 核心解码表:
|
||||
|
||||
```text
|
||||
analysis_window float64[10,64]
|
||||
@@ -18,3 +20,18 @@ joc_huff_code_7ch_pos_index_sparse int64[6,2]
|
||||
`src/joc_qmf.py` 读取 QMF 表,`src/joc_decode.py` 读取 JOC Huffman 树。Python 不读取 `native/` 下的 C/C++ 头文件。
|
||||
|
||||
原生侧对应数据分别位于 `native/src/qmf_tables.h` 与 `native/src/joc_huffman_tables.h`。修改任何一侧时,应同步更新另一侧并进行逐值一致性检查。
|
||||
|
||||
## 双耳渲染表
|
||||
|
||||
`rosella_kernels.npz` 保存双耳 QMF/hybrid 固定表:
|
||||
|
||||
```text
|
||||
qmf_analysis_coefficients float32[64,10]
|
||||
hybrid_analysis_low_kernel float32[3,2,13,16,2]
|
||||
hybrid_synthesis_indices int16[154,4]
|
||||
hybrid_synthesis_values float32[154]
|
||||
qmf_synthesis_basis float64[64,4,128]
|
||||
qmf_synthesis_taps float64[64,10,4]
|
||||
```
|
||||
|
||||
float32 表值载入后提升为 float64。
|
||||
|
||||
Binary file not shown.
@@ -0,0 +1,348 @@
|
||||
# JustOneCacophony — Binaural Rendering Mathematics
|
||||
|
||||
[中文](binaural.md) · [Back to README](../README.en.md)
|
||||
|
||||
This document defines the `pcm16 + ID11/OAMD → stereo` calculation. The path begins after object reconstruction and does not pass through ADM BWF or AXML.
|
||||
|
||||
## 1. Signal path and notation
|
||||
|
||||
```text
|
||||
LFE + 15 object PCM channels
|
||||
→ 64-band QMF analysis
|
||||
→ 77-band hybrid analysis
|
||||
→ per-object geometry, transfer functions, and room send
|
||||
→ direct accumulation + room network
|
||||
→ hybrid synthesis
|
||||
→ QMF synthesis
|
||||
→ 961-sample latency compensation
|
||||
→ stereo WAV
|
||||
```
|
||||
|
||||
| Symbol | Meaning |
|
||||
|---|---|
|
||||
| $s=0\ldots15$ | input source; source 0 is LFE |
|
||||
| $e\in\{L,R\}$ | output ear |
|
||||
| $k=0\ldots63$ | QMF band |
|
||||
| $h=0\ldots76$ | hybrid band |
|
||||
| $j=0\ldots35$ | direction-basis term |
|
||||
| $m$ | 64-sample QMF slot |
|
||||
|
||||
A control block is
|
||||
|
||||
$$N_b=512=8\times64,$$
|
||||
|
||||
and an input frame is
|
||||
|
||||
$$N_f=1536=3N_b.$$
|
||||
|
||||
All filter and room state continues across frame boundaries.
|
||||
|
||||
## 2. QMF analysis
|
||||
|
||||
Let $a_{p,\ell}$ be the fixed 64×10 polyphase coefficients and $r_{s,\ell,p}[m]$ the current and previous nine phase vectors:
|
||||
|
||||
$$
|
||||
E_{s,p}[m]=\sum_{\ell\text{ even}}a_{p,\ell}r_{s,\ell,p}[m],
|
||||
$$
|
||||
|
||||
$$
|
||||
O_{s,p}[m]=\sum_{\ell\text{ odd}}a_{p,\ell}r_{s,\ell,p}[m].
|
||||
$$
|
||||
|
||||
Define
|
||||
|
||||
$$
|
||||
\mathcal Q(v)_k=
|
||||
\operatorname{FFT}_{128}
|
||||
\left([v[p]e^{-j\pi p/128}]_{p=0}^{63},0_{64}\right)_k
|
||||
e^{-j3\pi(k+1/2)/128}.
|
||||
$$
|
||||
|
||||
The complex QMF output is
|
||||
|
||||
$$
|
||||
X_{s,k}[m]=\mathcal Q(O_s)_k+j(-1)^k\mathcal Q(E_s)_k.
|
||||
$$
|
||||
|
||||
## 3. Hybrid analysis
|
||||
|
||||
The lowest three QMF bands are split into sixteen hybrid bands by a 13-slot FIR:
|
||||
|
||||
$$
|
||||
H_{s,h,o}[m]
|
||||
=
|
||||
\sum_{p=0}^{2}\sum_{i=0}^{1}\sum_{\ell=0}^{12}
|
||||
X_{s,p,i}[m-\ell]K_{p,i,\ell,h,o},
|
||||
\qquad h=0\ldots15.
|
||||
$$
|
||||
|
||||
The remaining bands are delayed QMF bands 3..63:
|
||||
|
||||
$$
|
||||
H_{s,16+q}[m]=X_{s,3+q}[m-6],
|
||||
\qquad q=0\ldots60.
|
||||
$$
|
||||
|
||||
## 4. OAMD coordinates and time
|
||||
|
||||
The Q15 object fields are restored to their discrete grids:
|
||||
|
||||
$$
|
||||
u_1=\min\left(1,\frac{\operatorname{round}(62q_1/32767)}{62}\right),$$
|
||||
|
||||
$$
|
||||
u_2=\min\left(1,\frac{\operatorname{round}(62q_2/32767)}{62}\right),$$
|
||||
|
||||
$$
|
||||
u_3=\operatorname{clip}\left(
|
||||
\frac{\operatorname{round}(15q_3/32767)}{15},-1,1\right),$$
|
||||
|
||||
$$
|
||||
(X,Y,Z)=(2u_1-1,\ 1-2u_2,\ u_3).
|
||||
$$
|
||||
|
||||
An update is coded at
|
||||
|
||||
$$
|
||||
n_{\mathrm{coded}}
|
||||
=n_{\mathrm{frame}}+n_{\mathrm{outer}}+n_{\mathrm{block}}.
|
||||
$$
|
||||
|
||||
The first valid state is the position at sample 0. Later updates add the object delay $D_o=1473$. For $R>64$:
|
||||
|
||||
$$
|
||||
n_{\mathrm{start}}=n_{\mathrm{coded}}+D_o+64,$$
|
||||
|
||||
$$R_{\mathrm{eff}}=R-64,$$
|
||||
|
||||
$$
|
||||
\mathbf p[n]=(1-\alpha)\mathbf p_0+\alpha\mathbf p_1,
|
||||
\qquad
|
||||
\alpha=\frac{n-n_{\mathrm{start}}}{R_{\mathrm{eff}}}.
|
||||
$$
|
||||
|
||||
The position is evaluated at each 512-sample block boundary.
|
||||
|
||||
## 5. Distance profile and direction
|
||||
|
||||
Each Near, Mid, or Far profile contains six bounds, distance scale $D$, inverse scale $D^{-1}$, three axis scales, and minimum radius $\rho_{\min}$.
|
||||
|
||||
After axis conversion and scale:
|
||||
|
||||
$$
|
||||
\mathbf s=(a_zq_f,a_xq_l,a_yq_v).
|
||||
$$
|
||||
|
||||
A single ray factor $\lambda\le1$ keeps the point inside the profile bounds:
|
||||
|
||||
$$
|
||||
\mathbf s'=\lambda\mathbf s.
|
||||
$$
|
||||
|
||||
Then
|
||||
|
||||
$$
|
||||
\rho=\|\mathbf s'\|_2,
|
||||
\quad
|
||||
\rho_c=\max(\rho,\rho_{\min}),
|
||||
\quad
|
||||
\alpha=\rho/\rho_c,
|
||||
$$
|
||||
|
||||
$$
|
||||
\mathbf d=\mathbf s'/\rho,
|
||||
\qquad
|
||||
R=D\rho.
|
||||
$$
|
||||
|
||||
## 6. Direction basis and ear paths
|
||||
|
||||
The direction is expanded into a fixed 36-term polynomial basis:
|
||||
|
||||
$$
|
||||
\mathbf b(\mathbf d)=
|
||||
[1,x,y,z,x^2-\tfrac13,xy,xz,y^2-\tfrac13,yz,\ldots]^T.
|
||||
$$
|
||||
|
||||
For ear offset $e$:
|
||||
|
||||
$$
|
||||
\epsilon=\frac{eD^{-1}}{\rho_c},
|
||||
$$
|
||||
|
||||
$$
|
||||
\mathbf d_{\mp}=
|
||||
\frac{(x,y\mp\epsilon,z)}{\|(x,y\mp\epsilon,z)\|_2}.
|
||||
$$
|
||||
|
||||
The normalized paths are
|
||||
|
||||
$$
|
||||
\ell_{\mp}=\rho_c\sqrt{x^2+(y\mp\epsilon)^2+z^2}.
|
||||
$$
|
||||
|
||||
A model direction vector may add a non-negative path correction:
|
||||
|
||||
$$
|
||||
\ell'_e=\ell_e+
|
||||
\max(\mathbf v_e^T\mathbf b_e,0)\,2cD^{-1}.
|
||||
$$
|
||||
|
||||
The interaural delay is
|
||||
|
||||
$$
|
||||
\tau=|\ell'_+-\ell'_-|D\frac{48000}{343.3}\alpha.
|
||||
$$
|
||||
|
||||
The longer path receives the hybrid phase
|
||||
|
||||
$$P_h=e^{j\omega_h\tau}.$$
|
||||
|
||||
## 7. Direction fields and direct gains
|
||||
|
||||
Each ear has a 77×36 complex field:
|
||||
|
||||
$$
|
||||
C_{e,h}(\mathbf d_e)=
|
||||
\sum_{j=0}^{35}F_{e,h,j}b_j(\mathbf d_e).
|
||||
$$
|
||||
|
||||
Path weights are
|
||||
|
||||
$$
|
||||
w_L=\frac{\ell_+}{\sqrt{\ell_-^2+\ell_+^2}},
|
||||
\qquad
|
||||
w_R=\frac{\ell_-}{\sqrt{\ell_-^2+\ell_+^2}}.
|
||||
$$
|
||||
|
||||
For effective distance $R_e=\rho s_dD$, Mid and Far use
|
||||
|
||||
$$
|
||||
g_c=\frac{1}{\sqrt{1+s_rR_e^2}},
|
||||
\qquad
|
||||
g_{\mathrm{room}}=R_eg_c.
|
||||
$$
|
||||
|
||||
Near uses $g_c=1$ and $g_{\mathrm{room}}=0$. With field term zero denoted by $C^{(0)}$:
|
||||
|
||||
$$
|
||||
G_{L,h}=g_c[C_{L,h}w_L\alpha+C_{L,h}^{(0)}c_L(1-\alpha)],
|
||||
$$
|
||||
|
||||
$$
|
||||
G_{R,h}=g_c[C_{R,h}w_R\alpha+C_{R,h}^{(0)}c_R(1-\alpha)].
|
||||
$$
|
||||
|
||||
## 8. LFE
|
||||
|
||||
LFE bypasses ordinary-object geometry:
|
||||
|
||||
$$
|
||||
G_{L,h}^{\mathrm{LFE}}=G_{R,h}^{\mathrm{LFE}}=
|
||||
\begin{cases}
|
||||
g_h,&0\le h<16,\\0,&16\le h<77.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
```text
|
||||
2.60290003, 1.80741799, 0.659342408, -0.0275855921,
|
||||
-0.105803289, -0.0699509233, 0.0749056414, -0.00919809937,
|
||||
0.00349014648,-0.0158600751,-0.000723021978,0.00188189559,
|
||||
-0.000421735429,0.0000329252762,0.0000317397971,0.000000580376991
|
||||
```
|
||||
|
||||
Its room send is zero.
|
||||
|
||||
## 9. Source accumulation and room network
|
||||
|
||||
Direct output and room input are
|
||||
|
||||
$$
|
||||
Y^{\mathrm{direct}}_{e,h}=
|
||||
\sum_{s=0}^{15}H_{s,h}G_{s,e,h},
|
||||
$$
|
||||
|
||||
$$
|
||||
U_h=\sum_{s=1}^{15}H_{s,h}g_{\mathrm{room},s}.
|
||||
$$
|
||||
|
||||
The room input is scaled by $0.70710677$. Each all-pass stage uses
|
||||
|
||||
$$r[n]=x[n]-ad[n],$$
|
||||
|
||||
$$y[n]=ar[n]+d[n].$$
|
||||
|
||||
For the four-branch delay network:
|
||||
|
||||
$$
|
||||
\mathbf b_h[m]=U_h[m]\mathbf1+M\mathbf d_h[m],
|
||||
$$
|
||||
|
||||
$$m_{h,i}[m]=f_{h,i}b_{h,i}[m].$$
|
||||
|
||||
The main tap, optional extra taps, and ear output matrices produce
|
||||
|
||||
$$
|
||||
Y^{\mathrm{room}}_{e,h}[m]=
|
||||
\sum_{i=0}^{3}O_{e,h,i}z_{h,i}[m].
|
||||
$$
|
||||
|
||||
The final hybrid signal is
|
||||
|
||||
$$Y_{e,h}=Y^{\mathrm{direct}}_{e,h}+Y^{\mathrm{room}}_{e,h}.$$
|
||||
|
||||
The Python backend uses a finite complex FIR/overlap-add realization. The C++ backend keeps the recursive room state directly.
|
||||
|
||||
## 10. Hybrid and QMF synthesis
|
||||
|
||||
Hybrid synthesis is a 154-entry sparse map. For an entry $(h,i,k,o,w)$:
|
||||
|
||||
$$Q_{e,k,o}[m]\mathrel{+}=Y_{e,h,i}[m]w.$$
|
||||
|
||||
The complex QMF vector is flattened to
|
||||
|
||||
$$
|
||||
\mathbf q_e=[\Re Q_{e,0},\Im Q_{e,0},\ldots,\Re Q_{e,63},\Im Q_{e,63}]^T.
|
||||
$$
|
||||
|
||||
Rank-four features and ten-slot synthesis are
|
||||
|
||||
$$f_{e,p,r}[m]=\mathbf b_{p,r}^T\mathbf q_e[m],$$
|
||||
|
||||
$$
|
||||
y_e[64m+p]=
|
||||
\sum_{\ell=0}^{9}\sum_{r=0}^{3}
|
||||
t_{p,\ell,r}f_{e,p,r}[m-\ell].
|
||||
$$
|
||||
|
||||
## 11. Latency, tail, and precision
|
||||
|
||||
The filterbank latency is 961 samples and is removed once at the beginning of the continuous stream. Zero input is then processed to release filterbank and room state. Tail trimming keeps the final sample satisfying
|
||||
|
||||
$$
|
||||
\max(|y_L[n]|,|y_R[n]|)>10^{-8},
|
||||
$$
|
||||
|
||||
while never shortening the output below the source PCM length.
|
||||
|
||||
All internal state, geometry, field products, room processing, source accumulation, and tail processing use `float64/complex128`. Conversion to float32 or PCM24 occurs only in the final writer.
|
||||
|
||||
## 12. Backends and model path
|
||||
|
||||
Python and C++ use the same fixed tables, parsed model parameters, 512-sample control timeline, direct gains, room sends, latency compensation, and tail policy.
|
||||
|
||||
The C++ backend owns QMF, hybrid, recursive room, and synthesis state. Python supplies parsed parameters and per-block gains.
|
||||
|
||||
The default model path is
|
||||
|
||||
```text
|
||||
HRTF/binaural.personalized_headphone
|
||||
```
|
||||
|
||||
Override it with `--personalized-headphone PATH`.
|
||||
|
||||
A SOFA FIR cannot be converted into this parameter model by array rearrangement alone. A conversion requires fitting the direction fields, ITD, distance profiles, ear geometry, and room parameters.
|
||||
|
||||
## 13. Scope
|
||||
|
||||
The current path covers fifteen point objects and one special LFE source. Extent, spread, diffuse, divergence, channel lock, and unsupported OAMD element variants are outside this model.
|
||||
@@ -0,0 +1,536 @@
|
||||
# JustOneCacophony — 双耳渲染数学
|
||||
|
||||
[English](binaural.en.md) · [返回 README](../README.md)
|
||||
|
||||
本文说明 `pcm16 + ID11/OAMD → stereo` 路径中的计算、状态和时间对齐。双耳渲染直接接在对象重建之后,不经过 ADM BWF 或 AXML。
|
||||
|
||||
## 1. 总体路径与记号
|
||||
|
||||
```text
|
||||
pcm16:LFE + 15 路对象 PCM
|
||||
→ 64-band QMF analysis
|
||||
→ 77-band hybrid analysis
|
||||
→ 逐对象方向、距离、双耳传递函数和 room send
|
||||
→ 对象累加 + room network
|
||||
→ hybrid synthesis
|
||||
→ QMF synthesis
|
||||
→ 961-sample 延迟补偿
|
||||
→ stereo WAV
|
||||
```
|
||||
|
||||
主要记号:
|
||||
|
||||
| 符号 | 含义 |
|
||||
|---|---|
|
||||
| $s=0\ldots15$ | 输入源;$s=0$ 为 LFE,$s=1\ldots15$ 为对象 |
|
||||
| $e\in\{L,R\}$ | 左右输出耳 |
|
||||
| $p=0\ldots63$ | QMF phase / 时域 hop 内采样 |
|
||||
| $k=0\ldots63$ | QMF 子带 |
|
||||
| $h=0\ldots76$ | hybrid 子带 |
|
||||
| $j=0\ldots35$ | 方向 basis 项 |
|
||||
| $m$ | 64-sample QMF 时槽 |
|
||||
| $n$ | 时域采样位置 |
|
||||
|
||||
每个 QMF hop 为 64 samples,每个双耳控制块为
|
||||
|
||||
$$
|
||||
N_b=512=8\times64,
|
||||
$$
|
||||
|
||||
每个 E-AC-3/JOC 音频帧为
|
||||
|
||||
$$
|
||||
N_f=1536=3N_b.
|
||||
$$
|
||||
|
||||
## 2. 输入与控制块
|
||||
|
||||
输入矩阵为
|
||||
|
||||
$$
|
||||
x_s[n],\qquad s=0\ldots15.
|
||||
$$
|
||||
|
||||
`pcm16` 的通道约定为:
|
||||
|
||||
```text
|
||||
ch0 special LFE
|
||||
ch1..15 JOC 对象 1..15
|
||||
```
|
||||
|
||||
渲染器按连续采样流推进。QMF、hybrid、room 和 synthesis 状态不会在 1536-sample 帧边界清零。
|
||||
|
||||
## 3. 64-band QMF analysis
|
||||
|
||||
令 $a_{p,\ell}$ 为固定的 64×10 polyphase 系数,$r_{s,\ell,p}[m]$ 为当前和前 9 个 hop 的 phase 历史。奇偶 lag 分别累加:
|
||||
|
||||
$$
|
||||
E_{s,p}[m]
|
||||
=\sum_{\substack{\ell=0\\\ell\text{ even}}}^{9}
|
||||
a_{p,\ell}r_{s,\ell,p}[m],
|
||||
$$
|
||||
|
||||
$$
|
||||
O_{s,p}[m]
|
||||
=\sum_{\substack{\ell=0\\\ell\text{ odd}}}^{9}
|
||||
a_{p,\ell}r_{s,\ell,p}[m].
|
||||
$$
|
||||
|
||||
对任一 64-vector $v[p]$,定义调制变换
|
||||
|
||||
$$
|
||||
\mathcal Q(v)_k
|
||||
=
|
||||
\operatorname{FFT}_{128}
|
||||
\left(
|
||||
\left[v[p]e^{-j\pi p/128}\right]_{p=0}^{63},
|
||||
0_{64}
|
||||
\right)_k
|
||||
|
||||
e^{-j3\pi(k+1/2)/128}.
|
||||
$$
|
||||
|
||||
analysis 输出为
|
||||
|
||||
$$
|
||||
X_{s,k}[m]
|
||||
=
|
||||
\mathcal Q(O_s)_k
|
||||
+j(-1)^k\mathcal Q(E_s)_k.
|
||||
$$
|
||||
|
||||
所有历史、乘加和 FFT 结果使用 `float64/complex128`。
|
||||
|
||||
## 4. 77-band hybrid analysis
|
||||
|
||||
低 3 个 QMF 子带使用 13-slot FIR 拆分为 16 个 hybrid bands。把复数的实部和虚部分量记为 $i,o\in\{0,1\}$,固定核为 $K_{p,i,\ell,h,o}$:
|
||||
|
||||
$$
|
||||
H_{s,h,o}[m]
|
||||
=
|
||||
\sum_{p=0}^{2}
|
||||
\sum_{i=0}^{1}
|
||||
\sum_{\ell=0}^{12}
|
||||
X_{s,p,i}[m-\ell]K_{p,i,\ell,h,o},
|
||||
\qquad h=0\ldots15.
|
||||
$$
|
||||
|
||||
其余 61 个 hybrid bands 是 QMF 3..63 的 6-slot 延迟:
|
||||
|
||||
$$
|
||||
H_{s,16+q}[m]=X_{s,3+q}[m-6],
|
||||
\qquad q=0\ldots60.
|
||||
$$
|
||||
|
||||
因此 hybrid vector 的顺序为:
|
||||
|
||||
```text
|
||||
0..15 低 3 个 QMF bands 的细分
|
||||
16..76 延迟后的 QMF bands 3..63
|
||||
```
|
||||
|
||||
## 5. OAMD 坐标与时间轴
|
||||
|
||||
### 5.1 Q15 坐标到 Cartesian
|
||||
|
||||
对象状态中的 $q_1,q_2,q_3$ 先恢复到离散位置网格:
|
||||
|
||||
$$
|
||||
u_1=\min\left(1,\frac{\operatorname{round}(62q_1/32767)}{62}\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
u_2=\min\left(1,\frac{\operatorname{round}(62q_2/32767)}{62}\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
u_3=\operatorname{clip}\left(
|
||||
\frac{\operatorname{round}(15q_3/32767)}{15},-1,1\right).
|
||||
$$
|
||||
|
||||
ADM Cartesian 坐标为
|
||||
|
||||
$$
|
||||
(X,Y,Z)=(2u_1-1,\ 1-2u_2,\ u_3).
|
||||
$$
|
||||
|
||||
### 5.2 更新时间
|
||||
|
||||
一条位置更新的编码时刻为
|
||||
|
||||
$$
|
||||
n_{\mathrm{coded}}
|
||||
=n_{\mathrm{frame}}
|
||||
+n_{\mathrm{outer}}
|
||||
+n_{\mathrm{block}}.
|
||||
$$
|
||||
|
||||
首个有效状态作为 sample 0 的初始位置。后续更新加入对象 PCM 延迟 $D_o$,默认
|
||||
|
||||
$$
|
||||
D_o=1473.
|
||||
$$
|
||||
|
||||
若 ramp duration 为 $R>64$,连续运动为
|
||||
|
||||
$$
|
||||
n_{\mathrm{start}}=n_{\mathrm{coded}}+D_o+64,
|
||||
$$
|
||||
|
||||
$$
|
||||
R_{\mathrm{eff}}=R-64,
|
||||
$$
|
||||
|
||||
$$
|
||||
\mathbf p[n]
|
||||
=(1-\alpha)\mathbf p_0+\alpha\mathbf p_1,
|
||||
\qquad
|
||||
\alpha=\frac{n-n_{\mathrm{start}}}{R_{\mathrm{eff}}}.
|
||||
$$
|
||||
|
||||
当 $R\le64$ 时,目标位置在 $n_{\mathrm{coded}}+D_o$ 直接生效。
|
||||
|
||||
Rosella 参数在每个 512-sample block 起点求值,并用于该块的 8 个 hybrid slots。
|
||||
|
||||
## 6. 距离 profile 与方向
|
||||
|
||||
普通对象只使用 Near、Mid、Far 三个 profile。每个 profile 包含:
|
||||
|
||||
- 三轴负/正边界 $b_{x-},b_{x+},b_{y-},b_{y+},b_{z-},b_{z+}$;
|
||||
- 距离尺度 $D$ 和倒数尺度 $D^{-1}$;
|
||||
- 三轴内部尺度 $a_x,a_y,a_z$;
|
||||
- 最小归一化半径 $\rho_{\min}$。
|
||||
|
||||
Cartesian 坐标经过 Q15 metadata grid 后换成内部前、侧、上轴,乘以 profile 尺度:
|
||||
|
||||
$$
|
||||
\mathbf s=(a_zq_f,\ a_xq_l,\ a_yq_v).
|
||||
$$
|
||||
|
||||
若射线超出 profile 边界,则用单一比例 $\lambda\le1$ 缩放:
|
||||
|
||||
$$
|
||||
\mathbf s' = \lambda\mathbf s.
|
||||
$$
|
||||
|
||||
随后
|
||||
|
||||
$$
|
||||
\rho=\|\mathbf s'\|_2,
|
||||
\qquad
|
||||
\rho_c=\max(\rho,\rho_{\min}),
|
||||
\qquad
|
||||
\alpha=\frac{\rho}{\rho_c},
|
||||
$$
|
||||
|
||||
$$
|
||||
\mathbf d=
|
||||
\begin{cases}
|
||||
\mathbf s'/\rho,&\rho>0,\\
|
||||
(1,0,0),&\rho=0,
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
物理半径为
|
||||
|
||||
$$
|
||||
R=D\rho.
|
||||
$$
|
||||
|
||||
## 7. 36 项方向 basis
|
||||
|
||||
方向 $\mathbf d=(x,y,z)$ 被展开为 36 项实值多项式:
|
||||
|
||||
$$
|
||||
\mathbf b(\mathbf d)=
|
||||
[1,x,y,z,x^2-\tfrac13,xy,xz,y^2-\tfrac13,yz,\ldots]^T.
|
||||
$$
|
||||
|
||||
完整顺序由 `rosella_model.direction_basis()` 固定。最高次数为 5;所有 field 系数必须按该顺序点积,不能交换 basis 项。
|
||||
|
||||
逐耳 basis 会根据耳偏移重新归一化。令耳偏移标量为 $e$:
|
||||
|
||||
$$
|
||||
\epsilon=\frac{eD^{-1}}{\rho_c},
|
||||
$$
|
||||
|
||||
$$
|
||||
\mathbf d_-=
|
||||
\frac{(x,y-\epsilon,z)}{\|(x,y-\epsilon,z)\|_2},
|
||||
\qquad
|
||||
\mathbf d_+=
|
||||
\frac{(x,y+\epsilon,z)}{\|(x,y+\epsilon,z)\|_2}.
|
||||
$$
|
||||
|
||||
## 8. 逐耳路径与 ITD
|
||||
|
||||
归一化路径长度为
|
||||
|
||||
$$
|
||||
\ell_-=\rho_c\sqrt{x^2+(y-\epsilon)^2+z^2},
|
||||
$$
|
||||
|
||||
$$
|
||||
\ell_+=\rho_c\sqrt{x^2+(y+\epsilon)^2+z^2}.
|
||||
$$
|
||||
|
||||
模型允许通过 36-vector 对路径加入非负方向修正:
|
||||
|
||||
$$
|
||||
\ell'_e
|
||||
=
|
||||
\ell_e
|
||||
+
|
||||
\max(\mathbf v_e^T\mathbf b_e,0)\,2cD^{-1}.
|
||||
$$
|
||||
|
||||
耳间延迟为
|
||||
|
||||
$$
|
||||
\tau
|
||||
=|\ell'_+-\ell'_-|\,D\frac{f_s}{343.3}\alpha,
|
||||
\qquad f_s=48000.
|
||||
$$
|
||||
|
||||
路径较长的一耳应用 hybrid-band 相位:
|
||||
|
||||
$$
|
||||
P_h=e^{j\omega_h\tau},
|
||||
$$
|
||||
|
||||
其中 $\omega_h$ 由模型的 20 个 hybrid group 参数递推到 77 个 bands。
|
||||
|
||||
## 9. 方向 field 与直达增益
|
||||
|
||||
左右耳各有一个 77×36 complex field:
|
||||
|
||||
$$
|
||||
C_{e,h}(\mathbf d_e)
|
||||
=
|
||||
\sum_{j=0}^{35}F_{e,h,j}b_j(\mathbf d_e).
|
||||
$$
|
||||
|
||||
另一次耳路径计算给出左右权重:
|
||||
|
||||
$$
|
||||
w_L=\frac{\ell_+}{\sqrt{\ell_-^2+\ell_+^2}},
|
||||
\qquad
|
||||
w_R=\frac{\ell_-}{\sqrt{\ell_-^2+\ell_+^2}}.
|
||||
$$
|
||||
|
||||
有效距离为
|
||||
|
||||
$$
|
||||
R_e=\rho\,s_dD,
|
||||
$$
|
||||
|
||||
其中 $s_d$ 为模型距离标量。Mid/Far 的公共衰减和 room send 为
|
||||
|
||||
$$
|
||||
g_c=\frac{1}{\sqrt{1+s_rR_e^2}},
|
||||
$$
|
||||
|
||||
$$
|
||||
g_{\mathrm{room}}=R_eg_c.
|
||||
$$
|
||||
|
||||
Near 使用
|
||||
|
||||
$$
|
||||
g_c=1,
|
||||
\qquad
|
||||
g_{\mathrm{room}}=0.
|
||||
$$
|
||||
|
||||
令 $C_{e,h}^{(0)}$ 为 field 的第 0 个 basis 系数,中心保护项为 $1-\alpha$。普通直达传递函数可写成
|
||||
|
||||
$$
|
||||
G_{L,h}
|
||||
=g_c\left[C_{L,h}w_L\alpha+C_{L,h}^{(0)}c_L(1-\alpha)\right],
|
||||
$$
|
||||
|
||||
$$
|
||||
G_{R,h}
|
||||
=g_c\left[C_{R,h}w_R\alpha+C_{R,h}^{(0)}c_R(1-\alpha)\right].
|
||||
$$
|
||||
|
||||
$c_L,c_R$ 由模型的耳权重配置选择;路径较长的一耳再乘 $P_h$。
|
||||
|
||||
## 10. LFE 传递函数
|
||||
|
||||
LFE 不进入普通对象方向计算。其传递函数为
|
||||
|
||||
$$
|
||||
G_{L,h}^{\mathrm{LFE}}=G_{R,h}^{\mathrm{LFE}}=
|
||||
\begin{cases}
|
||||
g_h,&0\le h<16,\\
|
||||
0,&16\le h<77.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
前 16 个固定系数为
|
||||
|
||||
```text
|
||||
2.60290003, 1.80741799, 0.659342408, -0.0275855921,
|
||||
-0.105803289, -0.0699509233, 0.0749056414, -0.00919809937,
|
||||
0.00349014648,-0.0158600751,-0.000723021978,0.00188189559,
|
||||
-0.000421735429,0.0000329252762,0.0000317397971,0.000000580376991
|
||||
```
|
||||
|
||||
LFE 的 room send 恒为 0。
|
||||
|
||||
## 11. 对象累加与 room input
|
||||
|
||||
每个 hybrid slot 的直达输出为
|
||||
|
||||
$$
|
||||
Y^{\mathrm{direct}}_{e,h}
|
||||
=
|
||||
\sum_{s=0}^{15}H_{s,h}G_{s,e,h}.
|
||||
$$
|
||||
|
||||
room 输入为
|
||||
|
||||
$$
|
||||
U_h
|
||||
=
|
||||
\sum_{s=1}^{15}H_{s,h}g_{\mathrm{room},s}.
|
||||
$$
|
||||
|
||||
LFE 不进入该和式。
|
||||
|
||||
## 12. Room network
|
||||
|
||||
room 只处理前 64 个 hybrid bands。输入先乘
|
||||
|
||||
$$
|
||||
g_0=0.70710677.
|
||||
$$
|
||||
|
||||
对每级 all-pass,设延迟样本为 $d[n]$、系数为 $a$:
|
||||
|
||||
$$
|
||||
r[n]=x[n]-ad[n],
|
||||
$$
|
||||
|
||||
$$
|
||||
y[n]=ar[n]+d[n].
|
||||
$$
|
||||
|
||||
all-pass 输出复制到 4 个 FDN branches。设延迟输出为 $\mathbf d_h[m]$、4×4 混合矩阵为 $M$:
|
||||
|
||||
$$
|
||||
\mathbf b_h[m]
|
||||
=U_h[m]\mathbf 1+M\mathbf d_h[m].
|
||||
$$
|
||||
|
||||
每个 branch 使用复反馈系数 $f_{h,i}$:
|
||||
|
||||
$$
|
||||
m_{h,i}[m]=f_{h,i}b_{h,i}[m].
|
||||
$$
|
||||
|
||||
主 tap、可选额外 tap 和左右输出矩阵合成为
|
||||
|
||||
$$
|
||||
Y^{\mathrm{room}}_{e,h}[m]
|
||||
=
|
||||
\sum_{i=0}^{3}O_{e,h,i}z_{h,i}[m].
|
||||
$$
|
||||
|
||||
最终 hybrid 输出为
|
||||
|
||||
$$
|
||||
Y_{e,h}=Y^{\mathrm{direct}}_{e,h}+Y^{\mathrm{room}}_{e,h}.
|
||||
$$
|
||||
|
||||
Python 后端把该递归网络展开为有限 complex FIR 并使用 overlap-add;C++ 后端直接保持递归状态。两者均跨帧连续。
|
||||
|
||||
## 13. Hybrid synthesis
|
||||
|
||||
hybrid synthesis 是 154 项稀疏即时映射。令映射项为 $(h,i,k,o,w)$,其中 $i,o$ 表示实部或虚部,则
|
||||
|
||||
$$
|
||||
Q_{e,k,o}[m]
|
||||
\mathrel{+}=
|
||||
Y_{e,h,i}[m]w.
|
||||
$$
|
||||
|
||||
输出是每耳 64 个 complex QMF bands。
|
||||
|
||||
## 14. QMF synthesis
|
||||
|
||||
每耳 QMF vector 先展开为 128 项实向量
|
||||
|
||||
$$
|
||||
\mathbf q_e=[\Re Q_{e,0},\Im Q_{e,0},\ldots,\Re Q_{e,63},\Im Q_{e,63}]^T.
|
||||
$$
|
||||
|
||||
对每个 phase $p$ 和 rank $r=0\ldots3$:
|
||||
|
||||
$$
|
||||
f_{e,p,r}[m]
|
||||
=\mathbf b_{p,r}^T\mathbf q_e[m].
|
||||
$$
|
||||
|
||||
使用 10-slot taps 合成时域样本:
|
||||
|
||||
$$
|
||||
y_e[64m+p]
|
||||
=
|
||||
\sum_{\ell=0}^{9}
|
||||
\sum_{r=0}^{3}
|
||||
t_{p,\ell,r}f_{e,p,r}[m-\ell].
|
||||
$$
|
||||
|
||||
## 15. 延迟、尾声和输出
|
||||
|
||||
完整 filterbank 的固定延迟为
|
||||
|
||||
$$
|
||||
L=961\ \text{samples}.
|
||||
$$
|
||||
|
||||
只在连续流起点丢弃一次前 $L$ 个输出 samples。输入结束后继续送零,以释放 QMF、hybrid 和 room 状态。尾声裁切只作用于文件末端:
|
||||
|
||||
$$
|
||||
\max(|y_L[n]|,|y_R[n]|) > 10^{-8}
|
||||
$$
|
||||
|
||||
的最后一个 sample 被保留,同时输出长度不得短于源 PCM 长度。
|
||||
|
||||
所有内部状态、参数计算和对象累加使用 `float64/complex128`。最终 writer 才转换为 float32 或 PCM24。
|
||||
|
||||
## 16. Python 与 C++ 后端
|
||||
|
||||
两套后端共享:
|
||||
|
||||
- 同一份 QMF/hybrid 固定表;
|
||||
- 同一份模型解析结果;
|
||||
- 同一套 512-sample 参数更新时间轴;
|
||||
- 同一组逐对象 complex gains 和 room sends;
|
||||
- 同一 961-sample 延迟补偿与尾声策略。
|
||||
|
||||
C++ 后端以 512-sample block 为处理单位,内部持有 QMF、hybrid、room 和 synthesis 状态。Python 只负责模型解析、OAMD 时间轴和每块参数更新。
|
||||
|
||||
## 17. 模型文件
|
||||
|
||||
默认路径为
|
||||
|
||||
```text
|
||||
HRTF/binaural.personalized_headphone
|
||||
```
|
||||
|
||||
也可通过
|
||||
|
||||
```text
|
||||
--personalized-headphone PATH
|
||||
```
|
||||
|
||||
指定其它文件。
|
||||
|
||||
`.personalized_headphone` 中的 int32/Q15 参数在解析后提升为 float64。任意 SOFA FIR 不能只通过数组重排变成该参数模型;若要转换,需要拟合方向 fields、ITD、距离 profile、耳几何和 room 参数。
|
||||
|
||||
## 18. 适用范围
|
||||
|
||||
当前路径处理 15 个普通点对象和 1 路 special LFE。对象 extent、spread、diffuse、divergence、channel lock,以及未实现的 OAMD element 变体不在本公式范围内。
|
||||
+12
-2
@@ -499,16 +499,26 @@ s_{\mathrm{frame}}
|
||||
+32f_{\mathrm{block}}.
|
||||
$$
|
||||
|
||||
For processing-block length $B=32$, the aligned update point is
|
||||
The theoretical update position on the decoder-output PCM timeline is
|
||||
|
||||
$$
|
||||
s_{\mathrm{theoretical}}
|
||||
=s_{\mathrm{coded}}+d_{\mathrm{decoder}},
|
||||
\qquad d_{\mathrm{decoder}}=1473.
|
||||
$$
|
||||
|
||||
The speaker renderer retains the existing processing-block length $B=32$, so the aligned update point is
|
||||
|
||||
$$
|
||||
\widehat s
|
||||
=
|
||||
B\left\lfloor
|
||||
\frac{s_{\mathrm{coded}}+B/2-1}{B}
|
||||
\frac{s_{\mathrm{theoretical}}+B/2-1}{B}
|
||||
\right\rfloor.
|
||||
$$
|
||||
|
||||
Thus, for frame-aligned updates, `align32(1473)=1472`. The 1473 value is the theoretical decoder delay at the metadata interface; 1472 is its effective boundary in the current 32-sample control block. The 640-value inverse-QMF window/state is not part of this metadata-timing formula.
|
||||
|
||||
For ramp duration $D$, the number of blocks is
|
||||
|
||||
$$
|
||||
|
||||
+12
-2
@@ -499,16 +499,26 @@ s_{\mathrm{frame}}
|
||||
+32f_{\mathrm{block}}.
|
||||
$$
|
||||
|
||||
对处理块长度 $B=32$,更新点对齐为
|
||||
decoder 输出 PCM timeline 上的理论更新位置为
|
||||
|
||||
$$
|
||||
s_{\mathrm{theoretical}}
|
||||
=s_{\mathrm{coded}}+d_{\mathrm{decoder}},
|
||||
\qquad d_{\mathrm{decoder}}=1473.
|
||||
$$
|
||||
|
||||
扬声器 renderer 保留现有的处理块长度 $B=32$,更新点对齐为
|
||||
|
||||
$$
|
||||
\widehat s
|
||||
=
|
||||
B\left\lfloor
|
||||
\frac{s_{\mathrm{coded}}+B/2-1}{B}
|
||||
\frac{s_{\mathrm{theoretical}}+B/2-1}{B}
|
||||
\right\rfloor.
|
||||
$$
|
||||
|
||||
因此,对 frame-aligned 更新有 `align32(1473)=1472`。1473 是 metadata interface 的理论 decoder delay;1472 是当前 32-sample control block 中的有效边界。inverse-QMF 使用的 640 项 window/state 不属于这条 metadata timing 公式。
|
||||
|
||||
给定 ramp duration $D$,block 数为
|
||||
|
||||
$$
|
||||
|
||||
+22
-3
@@ -116,7 +116,26 @@ Supported layouts:
|
||||
2.0 3.1 5.1 7.1 5.1.2 5.1.4 7.1.2 7.1.4 9.1.4 9.1.6
|
||||
```
|
||||
|
||||
## 6. Building
|
||||
## 6. Binaural-rendering ABI
|
||||
|
||||
The shared library provides a 512-sample float64 binaural DSP interface:
|
||||
|
||||
```c
|
||||
ejoc_binaural_renderer_handle ejoc_binaural_renderer_create(void);
|
||||
int ejoc_binaural_renderer_configure_kernels(...);
|
||||
int ejoc_binaural_renderer_configure_room(...);
|
||||
int ejoc_binaural_renderer_process(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
const double* input16_interleaved, /* [512][16] */
|
||||
const double* gains_complex, /* [16][2][77][2] */
|
||||
const double* room_sends, /* [16] */
|
||||
double output_gain,
|
||||
double* output_stereo_interleaved); /* [512][2] */
|
||||
```
|
||||
|
||||
Python parses the model, evaluates the OAMD timeline, and supplies complex gains and room sends every 512 samples. The C++ handle owns QMF, hybrid, recursive-room, and QMF-synthesis state. Inputs, state, accumulation, and output are double/complex double.
|
||||
|
||||
## 7. Building
|
||||
|
||||
The CMake definition is `native/CMakeLists.txt`. Run from the repository root:
|
||||
|
||||
@@ -138,7 +157,7 @@ The MSVC configuration uses the static CRT. Other runtime dependencies depend on
|
||||
|
||||
The repository does not include native binaries by default. A prebuilt Release runtime or a locally built runtime can be placed directly under `lib/`.
|
||||
|
||||
## 7. Runtime lookup and fallback
|
||||
## 8. Runtime lookup and fallback
|
||||
|
||||
Lookup order:
|
||||
|
||||
@@ -148,7 +167,7 @@ Lookup order:
|
||||
|
||||
`--backend auto` falls back to NumPy when loading fails, and `--backend python` skips native discovery. The current CLI also prints the failure and falls back for `--backend native`; this existing behavior should not be read as successful native execution.
|
||||
|
||||
## 8. Implementation boundaries
|
||||
## 9. Implementation boundaries
|
||||
|
||||
- The native layer accepts only dense-JOC data already parsed by Python.
|
||||
- The ABI fixes a 1536-sample JOC frame, at most 15 objects, at most 23 parameter bands, and at most 2 data points.
|
||||
|
||||
+22
-3
@@ -116,7 +116,26 @@ int ejoc_speaker_renderer_process(
|
||||
2.0 3.1 5.1 7.1 5.1.2 5.1.4 7.1.2 7.1.4 9.1.4 9.1.6
|
||||
```
|
||||
|
||||
## 6. 构建
|
||||
## 6. 双耳渲染 ABI
|
||||
|
||||
共享库提供 512-sample float64 双耳 DSP:
|
||||
|
||||
```c
|
||||
ejoc_binaural_renderer_handle ejoc_binaural_renderer_create(void);
|
||||
int ejoc_binaural_renderer_configure_kernels(...);
|
||||
int ejoc_binaural_renderer_configure_room(...);
|
||||
int ejoc_binaural_renderer_process(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
const double* input16_interleaved, /* [512][16] */
|
||||
const double* gains_complex, /* [16][2][77][2] */
|
||||
const double* room_sends, /* [16] */
|
||||
double output_gain,
|
||||
double* output_stereo_interleaved); /* [512][2] */
|
||||
```
|
||||
|
||||
Python 负责模型解析、OAMD 时间轴和每 512 samples 的 complex gains/room sends。C++ handle 保存 QMF、hybrid、递归 room 和 QMF synthesis 状态。全部输入、状态、乘加和输出均为 double/complex double。
|
||||
|
||||
## 7. 构建
|
||||
|
||||
CMake 定义位于 `native/CMakeLists.txt`。从仓库根目录运行:
|
||||
|
||||
@@ -138,7 +157,7 @@ MSVC 配置使用静态 CRT。其他运行时依赖由平台和工具链决定
|
||||
|
||||
仓库默认不附带原生二进制。预构建的 Release 运行库或自行构建的运行库均可直接放入 `lib/`。
|
||||
|
||||
## 7. 运行时查找与回退
|
||||
## 8. 运行时查找与回退
|
||||
|
||||
查找顺序为:
|
||||
|
||||
@@ -148,7 +167,7 @@ MSVC 配置使用静态 CRT。其他运行时依赖由平台和工具链决定
|
||||
|
||||
`--backend auto` 在加载失败时回退到 NumPy;`--backend python` 跳过原生探测。`--backend native` 当前也会打印失败原因后回退,这是现有 CLI 行为,不应理解为原生库已成功使用。
|
||||
|
||||
## 8. 实现边界
|
||||
## 9. 实现边界
|
||||
|
||||
- 原生层只接收 Python 已解析的 dense JOC 数据。
|
||||
- ABI 固定了 1536-sample JOC 帧、最多 15 个对象、最多 23 个参数带和最多 2 个数据点。
|
||||
|
||||
@@ -20,15 +20,22 @@ if str(SOURCE_DIR) not in sys.path:
|
||||
import numpy as np
|
||||
|
||||
import adm_assemble
|
||||
import adm_atmos
|
||||
from adm_validate import validate
|
||||
from metadata import DirectPayloadIndex, PayloadIndex, write_summary
|
||||
import oamd_tracks
|
||||
from renderer import JocRenderer
|
||||
from native_renderer import NativeBackendUnavailable, NativeJocRenderer
|
||||
from binaural_renderer import (
|
||||
ROSSELLA_BLOCK_SAMPLES,
|
||||
ROSSELLA_LATENCY_SAMPLES,
|
||||
RosellaBinauralRenderer,
|
||||
resolve_personalized_headphone,
|
||||
)
|
||||
from speaker_backend import create_speaker_renderer
|
||||
from speaker_layouts import (SPEAKER_LAYOUT_CHOICES, get_speaker_layout,
|
||||
speaker_layout_display_name)
|
||||
from speaker_wav import SpeakerPcmSpool, write_speaker_wav
|
||||
from speaker_wav import BinauralPcmSpool, SpeakerPcmSpool, write_pcm_wav
|
||||
from variant_error import UnsupportedVariantError, write_variant_report
|
||||
|
||||
|
||||
@@ -37,13 +44,15 @@ FRAME_SAMPLES = 1536
|
||||
DEFAULT_OUTPUT_DIR = PROJECT_DIR / "output"
|
||||
|
||||
|
||||
def resolve_output(source, requested=None, speaker_layout=None):
|
||||
def resolve_output(source, requested=None, speaker_layout=None, *, binaural=False):
|
||||
"""解析成品路径;未指定时使用项目内的 ``output`` 目录。"""
|
||||
source = Path(source)
|
||||
if requested is not None:
|
||||
target = Path(requested)
|
||||
elif speaker_layout is not None:
|
||||
target = DEFAULT_OUTPUT_DIR / f"{source.stem}.{speaker_layout}.wav"
|
||||
elif binaural:
|
||||
target = DEFAULT_OUTPUT_DIR / f"{source.stem}.binaural.wav"
|
||||
else:
|
||||
target = DEFAULT_OUTPUT_DIR / (source.stem + ".adm.wav")
|
||||
return target.expanduser().resolve()
|
||||
@@ -104,8 +113,8 @@ def sha256(path):
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def choose_speaker_output_format(requested_format, clip_action, peak, clipped_values,
|
||||
*, input_func=input, interactive=None):
|
||||
def choose_pcm_output_format(requested_format, clip_action, peak, clipped_values,
|
||||
*, input_func=input, interactive=None):
|
||||
"""Resolve int24 clipping interactively or through an explicit policy."""
|
||||
if requested_format != "int24" or clipped_values == 0:
|
||||
return requested_format
|
||||
@@ -145,6 +154,10 @@ def choose_speaker_output_format(requested_format, clip_action, peak, clipped_va
|
||||
raise ValueError(f"未知 clip action: {action}")
|
||||
|
||||
|
||||
# Backward-compatible public name used by existing tests and callers.
|
||||
choose_speaker_output_format = choose_pcm_output_format
|
||||
|
||||
|
||||
def resolve_metadata(args, eac3, temp_dir):
|
||||
if args.metadata_dir:
|
||||
directory = Path(args.metadata_dir).resolve()
|
||||
@@ -197,7 +210,9 @@ def create_renderer(backend, gain, native_library=None, native_threads=None):
|
||||
|
||||
def render(index, bed_path, frame_count, raw_path, gain, progress_every,
|
||||
backend="auto", native_library=None, native_threads=None, frame_sink=None,
|
||||
speaker_renderer=None, speaker_sink=None, speaker_metadata_offset=1473):
|
||||
speaker_renderer=None, speaker_sink=None, speaker_metadata_offset=1473,
|
||||
binaural_renderer=None, binaural_sink=None, binaural_metadata_offset=1473,
|
||||
raw_scale=1.0):
|
||||
values = np.memmap(bed_path, dtype=np.float32, mode="r")
|
||||
frame_width = FRAME_SAMPLES * 6
|
||||
if values.size % frame_width:
|
||||
@@ -215,6 +230,9 @@ def render(index, bed_path, frame_count, raw_path, gain, progress_every,
|
||||
raw_write_seconds = 0.0
|
||||
speaker_render_seconds = 0.0
|
||||
speaker_write_seconds = 0.0
|
||||
binaural_render_seconds = 0.0
|
||||
binaural_write_seconds = 0.0
|
||||
elapsed = 0.0
|
||||
try:
|
||||
for frame_number, row in enumerate(index.rows[:frame_count]):
|
||||
bed6 = np.asarray(bed[frame_number], dtype=np.float32)
|
||||
@@ -225,7 +243,8 @@ def render(index, bed_path, frame_count, raw_path, gain, progress_every,
|
||||
dsp_seconds += time.perf_counter() - stage
|
||||
if output is not None:
|
||||
stage = time.perf_counter()
|
||||
output[frame_number] = pcm16.T
|
||||
output[frame_number] = np.multiply(
|
||||
pcm16.T, np.float32(raw_scale), dtype=np.float32)
|
||||
raw_write_seconds += time.perf_counter() - stage
|
||||
if frame_sink is not None:
|
||||
stage = time.perf_counter()
|
||||
@@ -239,6 +258,22 @@ def render(index, bed_path, frame_count, raw_path, gain, progress_every,
|
||||
stage = time.perf_counter()
|
||||
speaker_sink.write_frame(speaker_pcm)
|
||||
speaker_write_seconds += time.perf_counter() - stage
|
||||
if binaural_renderer is not None:
|
||||
payload = subs.get(11)
|
||||
outer_offset = (
|
||||
index.subpayload_sample_offset(row, 11)
|
||||
if payload is not None and hasattr(index, "subpayload_sample_offset")
|
||||
else 0
|
||||
)
|
||||
stage = time.perf_counter()
|
||||
binaural_pcm = binaural_renderer.render_frame(
|
||||
pcm16.T, payload, binaural_metadata_offset,
|
||||
outer_sample_offset=outer_offset)
|
||||
binaural_render_seconds += time.perf_counter() - stage
|
||||
if len(binaural_pcm):
|
||||
stage = time.perf_counter()
|
||||
binaural_sink.write_frame(binaural_pcm)
|
||||
binaural_write_seconds += time.perf_counter() - stage
|
||||
done = frame_number + 1
|
||||
if done % progress_every == 0 or done == frame_count:
|
||||
elapsed = time.perf_counter() - started
|
||||
@@ -246,6 +281,14 @@ def render(index, bed_path, frame_count, raw_path, gain, progress_every,
|
||||
eta = (frame_count - done) / max(speed, 1e-9)
|
||||
print(f"[JOC:{backend_info['name']}] {done}/{frame_count} "
|
||||
f"{speed:.1f} frame/s ETA {eta:.1f}s", flush=True)
|
||||
if binaural_renderer is not None:
|
||||
stage = time.perf_counter()
|
||||
binaural_tail = binaural_renderer.finish()
|
||||
binaural_render_seconds += time.perf_counter() - stage
|
||||
if len(binaural_tail):
|
||||
stage = time.perf_counter()
|
||||
binaural_sink.write_frame(binaural_tail)
|
||||
binaural_write_seconds += time.perf_counter() - stage
|
||||
if output is not None:
|
||||
output.flush()
|
||||
elapsed = time.perf_counter() - started
|
||||
@@ -256,6 +299,9 @@ def render(index, bed_path, frame_count, raw_path, gain, progress_every,
|
||||
close = getattr(speaker_renderer, "close", None)
|
||||
if close is not None:
|
||||
close()
|
||||
close = getattr(binaural_renderer, "close", None)
|
||||
if close is not None:
|
||||
close()
|
||||
breakdown = {
|
||||
"pipeline_wall_seconds": elapsed,
|
||||
"dsp_and_joc_parse_seconds": dsp_seconds,
|
||||
@@ -263,33 +309,57 @@ def render(index, bed_path, frame_count, raw_path, gain, progress_every,
|
||||
"raw_float_write_seconds": raw_write_seconds,
|
||||
"speaker_render_seconds": speaker_render_seconds,
|
||||
"speaker_spool_write_seconds": speaker_write_seconds,
|
||||
"binaural_render_seconds": binaural_render_seconds,
|
||||
"binaural_spool_write_seconds": binaural_write_seconds,
|
||||
}
|
||||
return dsp_seconds, backend_info, breakdown
|
||||
|
||||
|
||||
def build_parser():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="JustOneCacophony (JOC):E-AC-3 JOC → 25ch ADM BWF 或扬声器 WAV")
|
||||
description=("JustOneCacophony (JOC):E-AC-3 JOC → 25ch ADM BWF、"
|
||||
"扬声器 WAV 或 DLL-free Rosella 双耳 WAV"))
|
||||
parser.add_argument("input", type=Path, help="输入 .m4a/.eac3/.ec3")
|
||||
parser.add_argument("-o", "--output", type=Path, help="输出文件;默认按模式和布局命名")
|
||||
parser.add_argument("--speaker-output", type=Path,
|
||||
help="扬声器 WAV 路径;仅与 --speaker-layout 一起使用")
|
||||
parser.add_argument("--speaker-layout", choices=SPEAKER_LAYOUT_CHOICES,
|
||||
help="直接扬声器渲染布局,例如 2.0、5.1、7.1.2")
|
||||
parser.add_argument("--binaural-output", type=Path,
|
||||
help="双耳 WAV 路径;仅与 --binaural 一起使用")
|
||||
direct_mode = parser.add_mutually_exclusive_group()
|
||||
direct_mode.add_argument("--speaker-layout", choices=SPEAKER_LAYOUT_CHOICES,
|
||||
help="直接扬声器渲染布局,例如 2.0、5.1、7.1.2")
|
||||
direct_mode.add_argument("--binaural", action="store_true",
|
||||
help="直接 DLL-free Rosella 双耳渲染;不生成临时 ADM BWF")
|
||||
parser.add_argument("--speaker-format", choices=("float32", "int24"), default="float32",
|
||||
help="扬声器 WAV 格式,默认 float32")
|
||||
parser.add_argument("--binaural-format", choices=("float32", "int24"), default="float32",
|
||||
help="双耳 WAV 格式,默认 float32")
|
||||
parser.add_argument("--clip-action", choices=("ask", "continue", "float32", "abort"),
|
||||
default="ask",
|
||||
help="int24 削波处理:交互询问、继续截断、改 float32 或中止")
|
||||
parser.add_argument("--speaker-metadata-offset", type=int, default=1473,
|
||||
help="扬声器渲染 metadata 相对帧偏移,默认 1473 samples")
|
||||
parser.add_argument("--binaural-mode", choices=("near", "mid", "far"), default="mid",
|
||||
help="普通对象 Rosella 距离模式,默认 mid;LFE 始终走 special 低通")
|
||||
parser.add_argument("--personalized-headphone", "--binaural-hrtf", dest="personalized_headphone",
|
||||
type=Path, help="覆盖 HRTF/binaural.personalized_headphone")
|
||||
parser.add_argument("--binaural-tail-seconds", type=float, default=5.0,
|
||||
help="双耳 room/filterbank flush 上限,默认 5 秒")
|
||||
parser.add_argument("--binaural-tail-threshold", type=float, default=1.0e-8,
|
||||
help="双耳尾声裁切阈值,默认 1e-8;主体至少保留原时长")
|
||||
parser.add_argument("--binaural-chunk-frames", type=int, default=64,
|
||||
help="双耳内部批处理 E-AC-3 帧数,默认 64")
|
||||
parser.add_argument("--gain-db", type=float, default=0.0,
|
||||
help="成品增益 dB,默认 0(float32 系数 1.0)")
|
||||
help="成品增益 dB,默认 0;双耳路径以 float64 应用")
|
||||
parser.add_argument("--duration", type=float, help="只处理开头指定秒数")
|
||||
parser.add_argument("--object-delay-samples", type=int, default=640,
|
||||
help="可选的对象 PCM/OAMD 时间补偿,默认 640 samples")
|
||||
parser.add_argument("--object-delay-samples", type=int, default=1473,
|
||||
help="对象 PCM/OAMD 时间补偿;ADM 与双耳默认 1473 samples")
|
||||
parser.add_argument(
|
||||
"--joc-binaural-mode", choices=tuple(adm_atmos.JOC_BINAURAL_MODES),
|
||||
default=adm_atmos.JOC_BINAURAL_MODE_DEFAULT,
|
||||
help="ADM DBMD JOC 模式:off/near/far/mid/unspecified;仅影响 ADM BWF")
|
||||
parser.add_argument("--trajectory-mode", choices=("compact", "dense64"), default="compact",
|
||||
help="对象轨迹表示;compact 用长线性插值压缩 AXML,dense64 保留逐 64-sample 块")
|
||||
help="ADM 对象轨迹表示;直接双耳路径不序列化 AXML")
|
||||
parser.add_argument("--ffmpeg", default=os.environ.get("FFMPEG", "ffmpeg"))
|
||||
parser.add_argument("--backend", choices=("auto", "native", "python"), default="auto",
|
||||
help="DSP 后端;auto 优先 C++,不可用时回退 Python")
|
||||
@@ -326,24 +396,45 @@ def main(argv=None):
|
||||
if not source.is_file():
|
||||
raise FileNotFoundError(source)
|
||||
speaker_mode = args.speaker_layout is not None
|
||||
binaural_mode = bool(args.binaural)
|
||||
if args.speaker_output is not None and not speaker_mode:
|
||||
raise ValueError("--speaker-output 必须与 --speaker-layout 一起使用")
|
||||
if args.output is not None and args.speaker_output is not None:
|
||||
raise ValueError("-o/--output 与 --speaker-output 不能同时使用")
|
||||
if args.binaural_output is not None and not binaural_mode:
|
||||
raise ValueError("--binaural-output 必须与 --binaural 一起使用")
|
||||
specific_outputs = [value for value in (args.speaker_output, args.binaural_output)
|
||||
if value is not None]
|
||||
if args.output is not None and specific_outputs:
|
||||
raise ValueError("-o/--output 与 --speaker-output/--binaural-output 不能同时使用")
|
||||
if len(specific_outputs) > 1:
|
||||
raise ValueError("--speaker-output 与 --binaural-output 不能同时使用")
|
||||
if args.speaker_metadata_offset < 0:
|
||||
raise ValueError("speaker-metadata-offset 不能为负数")
|
||||
if args.personalized_headphone is not None and not binaural_mode:
|
||||
raise ValueError("--personalized-headphone 仅与 --binaural 一起使用")
|
||||
if args.binaural_tail_seconds < 0:
|
||||
raise ValueError("binaural-tail-seconds 不能为负数")
|
||||
if args.binaural_tail_threshold < 0:
|
||||
raise ValueError("binaural-tail-threshold 不能为负数")
|
||||
if args.binaural_chunk_frames <= 0:
|
||||
raise ValueError("binaural-chunk-frames 必须大于 0")
|
||||
requested_output = (args.speaker_output if args.speaker_output is not None
|
||||
else args.binaural_output if args.binaural_output is not None
|
||||
else args.output)
|
||||
output = resolve_output(
|
||||
source, requested_output, args.speaker_layout if speaker_mode else None)
|
||||
source, requested_output, args.speaker_layout if speaker_mode else None,
|
||||
binaural=binaural_mode)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
if args.duration is not None and args.duration <= 0:
|
||||
raise ValueError("duration 必须大于 0")
|
||||
if args.object_delay_samples < 0:
|
||||
raise ValueError("object-delay-samples 不能为负数")
|
||||
gain = np.float32(10.0 ** (args.gain_db / 20.0))
|
||||
if not np.isfinite(gain):
|
||||
raise ValueError("gain-db 超出 float32 范围")
|
||||
gain_float64 = 10.0 ** (args.gain_db / 20.0)
|
||||
gain = np.float32(gain_float64)
|
||||
if not math.isfinite(gain_float64) or not np.isfinite(gain):
|
||||
raise ValueError("gain-db 超出支持范围")
|
||||
binaural_model_path = (
|
||||
resolve_personalized_headphone(args.personalized_headphone)
|
||||
if binaural_mode and not args.metadata_only else None)
|
||||
ffmpeg = executable(args.ffmpeg, "FFmpeg")
|
||||
|
||||
total_started = time.perf_counter()
|
||||
@@ -388,7 +479,12 @@ def main(argv=None):
|
||||
speaker_wav_info = None
|
||||
speaker_clip_info = None
|
||||
speaker_actual_format = None
|
||||
binaural_backend_info = None
|
||||
binaural_wav_info = None
|
||||
binaural_clip_info = None
|
||||
binaural_actual_format = None
|
||||
if speaker_mode:
|
||||
timings["create_binaural_renderer"] = 0.0
|
||||
layout = get_speaker_layout(args.speaker_layout)
|
||||
speaker_name = speaker_layout_display_name(layout)
|
||||
speaker_decoder, speaker_backend_info = create_speaker_renderer(
|
||||
@@ -410,11 +506,11 @@ def main(argv=None):
|
||||
args.native_threads, None, speaker_decoder, spool,
|
||||
args.speaker_metadata_offset)
|
||||
spool.finalize()
|
||||
speaker_actual_format = choose_speaker_output_format(
|
||||
speaker_actual_format = choose_pcm_output_format(
|
||||
args.speaker_format, args.clip_action, spool.peak,
|
||||
spool.clipped_values)
|
||||
speaker_wav_info = timed_call(
|
||||
timings, "write_speaker_wav", write_speaker_wav,
|
||||
timings, "write_speaker_wav", write_pcm_wav,
|
||||
output, spool.values, speaker_actual_format, rate=RATE)
|
||||
speaker_clip_info = {
|
||||
"peak": spool.peak,
|
||||
@@ -430,8 +526,69 @@ def main(argv=None):
|
||||
timings["validate_adm"] = 0.0
|
||||
info = (f"speaker layout={speaker_name}, format={speaker_actual_format}, "
|
||||
f"peak={speaker_clip_info['peak']:.9g}")
|
||||
elif binaural_mode:
|
||||
model_path = binaural_model_path
|
||||
binaural_decoder = timed_call(
|
||||
timings, "create_binaural_renderer", RosellaBinauralRenderer,
|
||||
model_path, mode=args.binaural_mode,
|
||||
object_delay_samples=args.object_delay_samples,
|
||||
tail_seconds=args.binaural_tail_seconds,
|
||||
output_gain=gain_float64,
|
||||
chunk_frames=args.binaural_chunk_frames,
|
||||
backend=args.backend, native_library=args.native_library)
|
||||
print(
|
||||
f"[binaural] mode={args.binaural_mode} "
|
||||
f"backend={binaural_decoder.dsp_backend} "
|
||||
f"precision=float64/complex128 model={model_path}", flush=True)
|
||||
flush_samples = math.ceil(
|
||||
(args.binaural_tail_seconds * RATE
|
||||
+ ROSSELLA_LATENCY_SAMPLES + ROSSELLA_BLOCK_SAMPLES)
|
||||
/ ROSSELLA_BLOCK_SAMPLES) * ROSSELLA_BLOCK_SAMPLES
|
||||
spool = BinauralPcmSpool(
|
||||
temp_dir / "binaural_interleaved_f64.raw",
|
||||
frame_count * FRAME_SAMPLES + flush_samples,
|
||||
tail_threshold=args.binaural_tail_threshold)
|
||||
try:
|
||||
render_seconds, renderer_backend, render_breakdown = timed_call(
|
||||
timings, "render_and_stream", variant_call,
|
||||
output, source, render, index, bed_path, frame_count, raw_path,
|
||||
np.float32(1.0), max(1, args.progress_every),
|
||||
backend=args.backend, native_library=args.native_library,
|
||||
native_threads=args.native_threads,
|
||||
binaural_renderer=binaural_decoder, binaural_sink=spool,
|
||||
binaural_metadata_offset=args.object_delay_samples, raw_scale=gain)
|
||||
spool.finalize(minimum_samples=frame_count * FRAME_SAMPLES)
|
||||
binaural_actual_format = choose_pcm_output_format(
|
||||
args.binaural_format, args.clip_action, spool.peak,
|
||||
spool.clipped_values)
|
||||
binaural_wav_info = timed_call(
|
||||
timings, "write_binaural_wav", write_pcm_wav,
|
||||
output, spool.values, binaural_actual_format, rate=RATE)
|
||||
binaural_clip_info = {
|
||||
"peak": spool.peak,
|
||||
"over_unity_values": spool.clipped_values,
|
||||
"requested_format": args.binaural_format,
|
||||
"actual_format": binaural_actual_format,
|
||||
"clip_action": args.clip_action,
|
||||
"tail_threshold": args.binaural_tail_threshold,
|
||||
"source_samples": frame_count * FRAME_SAMPLES,
|
||||
"kept_samples": spool.sample_count,
|
||||
}
|
||||
binaural_backend_info = binaural_decoder.backend_info
|
||||
finally:
|
||||
spool.close()
|
||||
timings["build_adm_tracks"] = 0.0
|
||||
timings["finalize_adm"] = 0.0
|
||||
timings["validate_adm"] = 0.0
|
||||
info = (f"binaural mode={args.binaural_mode}, "
|
||||
f"format={binaural_actual_format}, "
|
||||
f"peak={binaural_clip_info['peak']:.9g}, "
|
||||
f"samples={binaural_clip_info['kept_samples']}")
|
||||
else:
|
||||
master = adm_assemble.StreamingMaster(output, duration_sec, rate=RATE)
|
||||
timings["create_binaural_renderer"] = 0.0
|
||||
master = adm_assemble.StreamingMaster(
|
||||
output, duration_sec, rate=RATE,
|
||||
joc_binaural_mode=adm_atmos.JOC_BINAURAL_MODES[args.joc_binaural_mode])
|
||||
try:
|
||||
render_seconds, renderer_backend, render_breakdown = timed_call(
|
||||
timings, "render_and_stream", variant_call,
|
||||
@@ -461,10 +618,11 @@ def main(argv=None):
|
||||
else:
|
||||
output_sha = timed_call(timings, "sha256", sha256, output)
|
||||
total_seconds = time.perf_counter() - total_started
|
||||
mode_name = "speaker" if speaker_mode else "binaural" if binaural_mode else "adm"
|
||||
report = {
|
||||
"input": str(source),
|
||||
"output": str(output),
|
||||
"mode": "speaker" if speaker_mode else "adm",
|
||||
"mode": mode_name,
|
||||
"metadata": str(metadata_json) if metadata_json is not None else None,
|
||||
"metadata_backend": metadata_backend,
|
||||
"metadata_cache": str(metadata_cache_dir) if metadata_cache_dir is not None else None,
|
||||
@@ -472,8 +630,13 @@ def main(argv=None):
|
||||
"duration_sec": duration_sec,
|
||||
"gain_db": args.gain_db,
|
||||
"gain_float32": float(gain),
|
||||
"object_delay_samples": None if speaker_mode else args.object_delay_samples,
|
||||
"trajectory_mode": None if speaker_mode else args.trajectory_mode,
|
||||
"gain_float64": float(gain_float64),
|
||||
"object_delay_samples": (None if speaker_mode else args.object_delay_samples),
|
||||
"trajectory_mode": args.trajectory_mode if mode_name == "adm" else None,
|
||||
"joc_binaural_mode": args.joc_binaural_mode if mode_name == "adm" else None,
|
||||
"joc_binaural_mode_value": (
|
||||
adm_atmos.JOC_BINAURAL_MODES[args.joc_binaural_mode]
|
||||
if mode_name == "adm" else None),
|
||||
"render_seconds": render_seconds,
|
||||
"render_breakdown": render_breakdown,
|
||||
"renderer_backend": renderer_backend,
|
||||
@@ -482,11 +645,19 @@ def main(argv=None):
|
||||
"speaker_metadata_offset": args.speaker_metadata_offset if speaker_mode else None,
|
||||
"speaker_clip": speaker_clip_info,
|
||||
"speaker_wav": speaker_wav_info,
|
||||
"streaming_adm": not speaker_mode,
|
||||
"binaural_renderer_backend": binaural_backend_info,
|
||||
"binaural_mode": args.binaural_mode if binaural_mode else None,
|
||||
"personalized_headphone": (
|
||||
binaural_backend_info["model"] if binaural_backend_info else None),
|
||||
"binaural_clip": binaural_clip_info,
|
||||
"binaural_wav": binaural_wav_info,
|
||||
"output_clip": speaker_clip_info if speaker_mode else binaural_clip_info,
|
||||
"output_wav": speaker_wav_info if speaker_mode else binaural_wav_info,
|
||||
"streaming_adm": mode_name == "adm",
|
||||
"kept_raw": str(raw_path) if raw_path is not None else None,
|
||||
"timings": timings,
|
||||
"total_seconds": total_seconds,
|
||||
"adm_validation": None if speaker_mode else info,
|
||||
"adm_validation": info if mode_name == "adm" else None,
|
||||
"adm_metadata": getattr(master, "metadata_info", None) if master is not None else None,
|
||||
"sha256": output_sha,
|
||||
"python": platform.python_version(),
|
||||
@@ -504,6 +675,11 @@ def main(argv=None):
|
||||
f"speaker={render_breakdown['speaker_render_seconds']:.2f}s "
|
||||
f"pipeline={render_breakdown['pipeline_wall_seconds']:.2f}s "
|
||||
f"total={report['total_seconds']:.2f}s")
|
||||
elif binaural_mode:
|
||||
print(f"[time] JOC-DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
|
||||
f"binaural={render_breakdown['binaural_render_seconds']:.2f}s "
|
||||
f"pipeline={render_breakdown['pipeline_wall_seconds']:.2f}s "
|
||||
f"total={report['total_seconds']:.2f}s")
|
||||
else:
|
||||
print(f"[time] DSP={render_seconds:.2f}s ({renderer_backend['name']}) "
|
||||
f"render+ADM-stream={render_breakdown['pipeline_wall_seconds']:.2f}s "
|
||||
|
||||
@@ -8,6 +8,7 @@ find_package(Threads REQUIRED)
|
||||
add_library(eac3joc_core SHARED
|
||||
src/eac3joc_core.cpp
|
||||
src/speaker_renderer.cpp
|
||||
src/binaural_renderer.cpp
|
||||
src/joc_huffman_tables.h
|
||||
src/qmf_tables.h
|
||||
src/speaker_layouts.h
|
||||
|
||||
@@ -29,11 +29,17 @@ enum {
|
||||
EJOC_MAX_DPOINTS = 2,
|
||||
EJOC_MAX_PARAMETER_BANDS = 23,
|
||||
EJOC_SPEAKER_BLOCK_SAMPLES = 32,
|
||||
EJOC_SPEAKER_COORDINATES = 3
|
||||
EJOC_SPEAKER_COORDINATES = 3,
|
||||
EJOC_BINAURAL_BLOCK_SAMPLES = 512,
|
||||
EJOC_BINAURAL_INPUT_CHANNELS = 16,
|
||||
EJOC_BINAURAL_OUTPUT_CHANNELS = 2,
|
||||
EJOC_BINAURAL_QMF_BANDS = 64,
|
||||
EJOC_BINAURAL_HYBRID_BANDS = 77
|
||||
};
|
||||
|
||||
typedef void* ejoc_renderer_handle;
|
||||
typedef void* ejoc_speaker_renderer_handle;
|
||||
typedef void* ejoc_binaural_renderer_handle;
|
||||
|
||||
/*
|
||||
Fixed array layouts used by ejoc_renderer_process():
|
||||
@@ -114,6 +120,44 @@ EJOC_API int EJOC_CALL ejoc_speaker_renderer_process(
|
||||
const double* object_gains,
|
||||
double* output_interleaved);
|
||||
|
||||
EJOC_API ejoc_binaural_renderer_handle EJOC_CALL ejoc_binaural_renderer_create(void);
|
||||
EJOC_API void EJOC_CALL ejoc_binaural_renderer_destroy(ejoc_binaural_renderer_handle handle);
|
||||
EJOC_API int EJOC_CALL ejoc_binaural_renderer_reset(ejoc_binaural_renderer_handle handle);
|
||||
EJOC_API const char* EJOC_CALL ejoc_binaural_renderer_last_error(
|
||||
ejoc_binaural_renderer_handle handle);
|
||||
EJOC_API int EJOC_CALL ejoc_binaural_renderer_configure_kernels(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
const double* qmf_analysis,
|
||||
const double* hybrid_low,
|
||||
const int16_t* hybrid_indices,
|
||||
const double* hybrid_values,
|
||||
uint32_t hybrid_count,
|
||||
const double* qmf_basis,
|
||||
const double* qmf_taps);
|
||||
EJOC_API int EJOC_CALL ejoc_binaural_renderer_configure_room(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
uint32_t bands,
|
||||
uint32_t allpass_count,
|
||||
const uint32_t* allpass_delays,
|
||||
const double* allpass_gains,
|
||||
const uint32_t* fdn_delays,
|
||||
const double* fdn_matrix,
|
||||
uint32_t output_tap_delay,
|
||||
const double* feedback_complex,
|
||||
const double* output_taps,
|
||||
const double* output_complex,
|
||||
uint32_t extra_count,
|
||||
const uint32_t* extra_delays,
|
||||
const double* extra_fields_complex,
|
||||
const double* extra_matrices);
|
||||
EJOC_API int EJOC_CALL ejoc_binaural_renderer_process(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
const double* input16_interleaved,
|
||||
const double* gains_complex,
|
||||
const double* room_sends,
|
||||
double output_gain,
|
||||
double* output_stereo_interleaved);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,664 @@
|
||||
#define EJOC_BUILD_DLL
|
||||
#include "eac3joc_core.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cmath>
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <new>
|
||||
#include <vector>
|
||||
|
||||
namespace ejoc::binaural {
|
||||
|
||||
struct Complex {
|
||||
double re;
|
||||
double im;
|
||||
};
|
||||
|
||||
inline Complex add(Complex a, Complex b) noexcept {
|
||||
return {a.re + b.re, a.im + b.im};
|
||||
}
|
||||
|
||||
inline Complex mul(Complex a, Complex b) noexcept {
|
||||
return {a.re * b.re - a.im * b.im, a.re * b.im + a.im * b.re};
|
||||
}
|
||||
|
||||
inline Complex scale(Complex value, double gain) noexcept {
|
||||
return {value.re * gain, value.im * gain};
|
||||
}
|
||||
|
||||
constexpr double kPi = 3.141592653589793238462643383279502884;
|
||||
constexpr int kChannels = EJOC_BINAURAL_INPUT_CHANNELS;
|
||||
constexpr int kEars = EJOC_BINAURAL_OUTPUT_CHANNELS;
|
||||
constexpr int kBlock = EJOC_BINAURAL_BLOCK_SAMPLES;
|
||||
constexpr int kSlots = kBlock / 64;
|
||||
constexpr int kQmf = EJOC_BINAURAL_QMF_BANDS;
|
||||
constexpr int kHybrid = EJOC_BINAURAL_HYBRID_BANDS;
|
||||
constexpr int kRank = 4;
|
||||
|
||||
class Renderer final {
|
||||
public:
|
||||
Renderer() noexcept {
|
||||
initialize_fft();
|
||||
reset();
|
||||
}
|
||||
|
||||
int configure_kernels(
|
||||
const double* qmf_analysis,
|
||||
const double* hybrid_low,
|
||||
const int16_t* hybrid_indices,
|
||||
const double* hybrid_values,
|
||||
uint32_t hybrid_count,
|
||||
const double* qmf_basis,
|
||||
const double* qmf_taps) noexcept {
|
||||
if (!qmf_analysis || !hybrid_low || !hybrid_indices || !hybrid_values ||
|
||||
!qmf_basis || !qmf_taps || hybrid_count == 0) {
|
||||
return fail("invalid binaural kernel configuration");
|
||||
}
|
||||
std::memcpy(qmf_analysis_.data(), qmf_analysis,
|
||||
qmf_analysis_.size() * sizeof(double));
|
||||
hybrid_low_.assign(hybrid_low, hybrid_low + 3 * 2 * 13 * 16 * 2);
|
||||
hybrid_indices_.assign(hybrid_indices, hybrid_indices + hybrid_count * 4);
|
||||
hybrid_values_.assign(hybrid_values, hybrid_values + hybrid_count);
|
||||
std::memcpy(qmf_basis_.data(), qmf_basis,
|
||||
qmf_basis_.size() * sizeof(double));
|
||||
std::memcpy(qmf_taps_.data(), qmf_taps,
|
||||
qmf_taps_.size() * sizeof(double));
|
||||
kernels_ready_ = true;
|
||||
reset();
|
||||
return 0;
|
||||
}
|
||||
|
||||
int configure_room(
|
||||
uint32_t bands,
|
||||
uint32_t allpass_count,
|
||||
const uint32_t* allpass_delays,
|
||||
const double* allpass_gains,
|
||||
const uint32_t* fdn_delays,
|
||||
const double* fdn_matrix,
|
||||
uint32_t output_tap_delay,
|
||||
const double* feedback_complex,
|
||||
const double* output_taps,
|
||||
const double* output_complex,
|
||||
uint32_t extra_count,
|
||||
const uint32_t* extra_delays,
|
||||
const double* extra_fields_complex,
|
||||
const double* extra_matrices) noexcept {
|
||||
if (bands != 64 || !fdn_delays || !fdn_matrix || !feedback_complex ||
|
||||
!output_taps || !output_complex ||
|
||||
(allpass_count && (!allpass_delays || !allpass_gains)) ||
|
||||
(extra_count && (!extra_delays || !extra_fields_complex || !extra_matrices))) {
|
||||
return fail("invalid binaural room configuration");
|
||||
}
|
||||
room_bands_ = bands;
|
||||
if (allpass_count) {
|
||||
allpass_delays_.assign(allpass_delays, allpass_delays + allpass_count);
|
||||
allpass_gains_.assign(allpass_gains, allpass_gains + allpass_count);
|
||||
} else {
|
||||
allpass_delays_.clear();
|
||||
allpass_gains_.clear();
|
||||
}
|
||||
allpass_offsets_.resize(allpass_count);
|
||||
allpass_positions_.assign(allpass_count, 0);
|
||||
size_t allpass_size = 0;
|
||||
for (uint32_t index = 0; index < allpass_count; ++index) {
|
||||
if (allpass_delays_[index] == 0) {
|
||||
return fail("binaural allpass delay must be positive");
|
||||
}
|
||||
allpass_offsets_[index] = allpass_size;
|
||||
allpass_size += static_cast<size_t>(allpass_delays_[index]) * bands;
|
||||
}
|
||||
allpass_memory_.assign(allpass_size, {});
|
||||
|
||||
room_capacity_ = 0;
|
||||
for (int branch = 0; branch < 4; ++branch) {
|
||||
fdn_delays_[branch] = fdn_delays[branch];
|
||||
room_capacity_ = std::max(room_capacity_, fdn_delays_[branch]);
|
||||
}
|
||||
if (room_capacity_ == 0) {
|
||||
return fail("binaural room delay must be positive");
|
||||
}
|
||||
std::copy(fdn_matrix, fdn_matrix + 16, fdn_matrix_.begin());
|
||||
output_tap_delay_ = output_tap_delay;
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
for (int branch = 0; branch < 4; ++branch) {
|
||||
const size_t complex_index = (static_cast<size_t>(band) * 4 + branch) * 2;
|
||||
feedback_[band][branch] = {
|
||||
feedback_complex[complex_index], feedback_complex[complex_index + 1]};
|
||||
output_taps_[band][branch] = output_taps[band * 4 + branch];
|
||||
for (int ear = 0; ear < 2; ++ear) {
|
||||
const size_t output_index =
|
||||
((static_cast<size_t>(ear) * 64 + band) * 4 + branch) * 2;
|
||||
output_matrix_[ear][band][branch] = {
|
||||
output_complex[output_index], output_complex[output_index + 1]};
|
||||
}
|
||||
}
|
||||
}
|
||||
room_memory_.assign(static_cast<size_t>(room_capacity_) * 64 * 4, {});
|
||||
if (extra_count) {
|
||||
extra_delays_.assign(extra_delays, extra_delays + extra_count);
|
||||
} else {
|
||||
extra_delays_.clear();
|
||||
}
|
||||
extra_fields_.resize(static_cast<size_t>(extra_count) * 64);
|
||||
extra_matrices_.resize(static_cast<size_t>(extra_count) * 16);
|
||||
for (uint32_t extra = 0; extra < extra_count; ++extra) {
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
const size_t source = (static_cast<size_t>(extra) * 64 + band) * 2;
|
||||
extra_fields_[static_cast<size_t>(extra) * 64 + band] = {
|
||||
extra_fields_complex[source], extra_fields_complex[source + 1]};
|
||||
}
|
||||
std::copy(extra_matrices + static_cast<size_t>(extra) * 16,
|
||||
extra_matrices + static_cast<size_t>(extra + 1) * 16,
|
||||
extra_matrices_.begin() + static_cast<size_t>(extra) * 16);
|
||||
}
|
||||
room_ready_ = true;
|
||||
reset();
|
||||
return 0;
|
||||
}
|
||||
|
||||
int reset() noexcept {
|
||||
qmf_history_.fill(0.0);
|
||||
hybrid_low_history_.fill({});
|
||||
hybrid_high_history_.fill({});
|
||||
synthesis_history_.fill(0.0);
|
||||
std::fill(allpass_memory_.begin(), allpass_memory_.end(), Complex{});
|
||||
std::fill(allpass_positions_.begin(), allpass_positions_.end(), 0u);
|
||||
std::fill(room_memory_.begin(), room_memory_.end(), Complex{});
|
||||
room_position_ = 0;
|
||||
error_[0] = '\0';
|
||||
return 0;
|
||||
}
|
||||
|
||||
const char* error() const noexcept {
|
||||
return error_[0] ? error_ : "";
|
||||
}
|
||||
|
||||
int process(
|
||||
const double* input,
|
||||
const double* gains,
|
||||
const double* room_sends,
|
||||
double output_gain,
|
||||
double* output) noexcept {
|
||||
if (!kernels_ready_ || !room_ready_) {
|
||||
return fail("binaural renderer is not configured");
|
||||
}
|
||||
if (!input || !gains || !room_sends || !output || !std::isfinite(output_gain)) {
|
||||
return fail("invalid binaural process arguments");
|
||||
}
|
||||
for (int slot = 0; slot < kSlots; ++slot) {
|
||||
std::array<Complex, kChannels * kQmf> qmf{};
|
||||
std::array<Complex, kChannels * kHybrid> hybrid{};
|
||||
analyze_qmf(input + static_cast<size_t>(slot) * 64 * kChannels, qmf);
|
||||
analyze_hybrid(qmf, hybrid);
|
||||
|
||||
std::array<Complex, kEars * kHybrid> rendered{};
|
||||
std::array<Complex, kHybrid> room_input{};
|
||||
for (int source = kChannels - 1; source >= 0; --source) {
|
||||
for (int band = 0; band < kHybrid; ++band) {
|
||||
const Complex value = hybrid[source * kHybrid + band];
|
||||
room_input[band] = add(room_input[band], scale(value, room_sends[source]));
|
||||
for (int ear = 0; ear < kEars; ++ear) {
|
||||
const size_t gain_index =
|
||||
(((static_cast<size_t>(source) * kEars + ear) * kHybrid + band) * 2);
|
||||
const Complex gain{gains[gain_index], gains[gain_index + 1]};
|
||||
rendered[ear * kHybrid + band] = add(
|
||||
rendered[ear * kHybrid + band], mul(value, gain));
|
||||
}
|
||||
}
|
||||
}
|
||||
const auto room = process_room(room_input);
|
||||
for (size_t index = 0; index < rendered.size(); ++index) {
|
||||
rendered[index] = add(rendered[index], room[index]);
|
||||
}
|
||||
|
||||
std::array<Complex, kEars * kQmf> qmf_output{};
|
||||
synthesize_hybrid(rendered, qmf_output);
|
||||
for (int ear = 0; ear < kEars; ++ear) {
|
||||
std::array<double, 64> samples{};
|
||||
synthesize_qmf(qmf_output.data() + ear * kQmf, ear, samples);
|
||||
for (int sample = 0; sample < 64; ++sample) {
|
||||
output[(static_cast<size_t>(slot) * 64 + sample) * 2 + ear] =
|
||||
samples[sample] * output_gain;
|
||||
}
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
private:
|
||||
int fail(const char* message) noexcept {
|
||||
std::snprintf(error_, sizeof(error_), "%s", message);
|
||||
return -1;
|
||||
}
|
||||
|
||||
void initialize_fft() noexcept {
|
||||
for (int index = 0; index < 128; ++index) {
|
||||
int value = index;
|
||||
int reversed = 0;
|
||||
for (int bit = 0; bit < 7; ++bit) {
|
||||
reversed = (reversed << 1) | (value & 1);
|
||||
value >>= 1;
|
||||
}
|
||||
bit_reverse_[index] = static_cast<uint8_t>(reversed);
|
||||
}
|
||||
for (int phase = 0; phase < 64; ++phase) {
|
||||
const double angle = -kPi * static_cast<double>(phase) / 128.0;
|
||||
premod_[phase] = {std::cos(angle), std::sin(angle)};
|
||||
const double post_angle =
|
||||
-3.0 * (static_cast<double>(phase) + 0.5) * kPi / 128.0;
|
||||
post_[phase] = {std::cos(post_angle), std::sin(post_angle)};
|
||||
even_post_[phase] = {0.0, (phase & 1) ? -1.0 : 1.0};
|
||||
}
|
||||
}
|
||||
|
||||
void fft128(std::array<Complex, 128>& values) const noexcept {
|
||||
for (int index = 0; index < 128; ++index) {
|
||||
const int reversed = bit_reverse_[index];
|
||||
if (reversed > index) {
|
||||
std::swap(values[index], values[reversed]);
|
||||
}
|
||||
}
|
||||
for (int length = 2; length <= 128; length <<= 1) {
|
||||
const double angle = -2.0 * kPi / static_cast<double>(length);
|
||||
const Complex step{std::cos(angle), std::sin(angle)};
|
||||
for (int start = 0; start < 128; start += length) {
|
||||
Complex rotation{1.0, 0.0};
|
||||
for (int offset = 0; offset < length / 2; ++offset) {
|
||||
const Complex even = values[start + offset];
|
||||
const Complex odd = mul(values[start + offset + length / 2], rotation);
|
||||
values[start + offset] = {even.re + odd.re, even.im + odd.im};
|
||||
values[start + offset + length / 2] = {
|
||||
even.re - odd.re, even.im - odd.im};
|
||||
rotation = mul(rotation, step);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void qmf_transform(const std::array<double, 64>& source,
|
||||
std::array<Complex, 64>& target) const noexcept {
|
||||
std::array<Complex, 128> work{};
|
||||
for (int phase = 0; phase < 64; ++phase) {
|
||||
work[phase] = scale(premod_[phase], source[phase]);
|
||||
}
|
||||
fft128(work);
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
target[band] = mul(work[band], post_[band]);
|
||||
}
|
||||
}
|
||||
|
||||
void analyze_qmf(const double* input,
|
||||
std::array<Complex, kChannels * kQmf>& output) noexcept {
|
||||
for (int channel = 0; channel < kChannels; ++channel) {
|
||||
for (int lag = 9; lag > 0; --lag) {
|
||||
for (int phase = 0; phase < 64; ++phase) {
|
||||
qmf_history_[qmf_history_index(lag, channel, phase)] =
|
||||
qmf_history_[qmf_history_index(lag - 1, channel, phase)];
|
||||
}
|
||||
}
|
||||
for (int phase = 0; phase < 64; ++phase) {
|
||||
qmf_history_[qmf_history_index(0, channel, phase)] =
|
||||
input[phase * kChannels + channel];
|
||||
}
|
||||
std::array<double, 64> even{};
|
||||
std::array<double, 64> odd{};
|
||||
for (int phase = 0; phase < 64; ++phase) {
|
||||
for (int lag = 0; lag < 10; ++lag) {
|
||||
const double value =
|
||||
qmf_history_[qmf_history_index(lag, channel, phase)] *
|
||||
qmf_analysis_[phase * 10 + lag];
|
||||
(lag & 1 ? odd[phase] : even[phase]) += value;
|
||||
}
|
||||
}
|
||||
std::array<Complex, 64> even_fft{};
|
||||
std::array<Complex, 64> odd_fft{};
|
||||
qmf_transform(even, even_fft);
|
||||
qmf_transform(odd, odd_fft);
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
output[channel * 64 + band] = add(
|
||||
odd_fft[band], mul(even_fft[band], even_post_[band]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void analyze_hybrid(
|
||||
const std::array<Complex, kChannels * kQmf>& qmf,
|
||||
std::array<Complex, kChannels * kHybrid>& output) noexcept {
|
||||
for (int channel = 0; channel < kChannels; ++channel) {
|
||||
for (int lag = 12; lag > 0; --lag) {
|
||||
for (int band = 0; band < 3; ++band) {
|
||||
hybrid_low_history_[hybrid_low_history_index(lag, channel, band)] =
|
||||
hybrid_low_history_[hybrid_low_history_index(lag - 1, channel, band)];
|
||||
}
|
||||
}
|
||||
for (int band = 0; band < 3; ++band) {
|
||||
hybrid_low_history_[hybrid_low_history_index(0, channel, band)] =
|
||||
qmf[channel * 64 + band];
|
||||
}
|
||||
for (int output_band = 0; output_band < 16; ++output_band) {
|
||||
Complex value{};
|
||||
for (int lag = 0; lag < 13; ++lag) {
|
||||
for (int input_band = 0; input_band < 3; ++input_band) {
|
||||
const Complex source = hybrid_low_history_[
|
||||
hybrid_low_history_index(lag, channel, input_band)];
|
||||
const double components[2]{source.re, source.im};
|
||||
for (int input_component = 0; input_component < 2; ++input_component) {
|
||||
value.re += components[input_component] * hybrid_low_[
|
||||
hybrid_low_kernel_index(input_band, input_component, lag,
|
||||
output_band, 0)];
|
||||
value.im += components[input_component] * hybrid_low_[
|
||||
hybrid_low_kernel_index(input_band, input_component, lag,
|
||||
output_band, 1)];
|
||||
}
|
||||
}
|
||||
}
|
||||
output[channel * kHybrid + output_band] = value;
|
||||
}
|
||||
for (int band = 0; band < 61; ++band) {
|
||||
output[channel * kHybrid + 16 + band] =
|
||||
hybrid_high_history_[hybrid_high_history_index(0, channel, band)];
|
||||
for (int delay = 0; delay < 5; ++delay) {
|
||||
hybrid_high_history_[hybrid_high_history_index(delay, channel, band)] =
|
||||
hybrid_high_history_[hybrid_high_history_index(delay + 1, channel, band)];
|
||||
}
|
||||
hybrid_high_history_[hybrid_high_history_index(5, channel, band)] =
|
||||
qmf[channel * 64 + 3 + band];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::array<Complex, kEars * kHybrid> process_room(
|
||||
const std::array<Complex, kHybrid>& input) noexcept {
|
||||
std::array<Complex, 64> filtered{};
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
filtered[band] = scale(input[band], 0.70710677);
|
||||
}
|
||||
for (size_t stage = 0; stage < allpass_delays_.size(); ++stage) {
|
||||
const uint32_t position = allpass_positions_[stage];
|
||||
const double gain = allpass_gains_[stage];
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
Complex& memory = allpass_memory_[
|
||||
allpass_offsets_[stage] + static_cast<size_t>(position) * 64 + band];
|
||||
const Complex residual = add(filtered[band], scale(memory, -gain));
|
||||
filtered[band] = add(scale(residual, gain), memory);
|
||||
memory = residual;
|
||||
}
|
||||
allpass_positions_[stage] = (position + 1) % allpass_delays_[stage];
|
||||
}
|
||||
|
||||
std::array<Complex, 64 * 4> branches{};
|
||||
std::array<Complex, 64 * 4> taps{};
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
for (int branch = 0; branch < 4; ++branch) {
|
||||
Complex value = filtered[band];
|
||||
for (int source = 0; source < 4; ++source) {
|
||||
const uint32_t position =
|
||||
(room_position_ + room_capacity_ - fdn_delays_[source]) % room_capacity_;
|
||||
value = add(value, scale(room_memory_[
|
||||
room_memory_index(position, band, source)],
|
||||
fdn_matrix_[branch * 4 + source]));
|
||||
}
|
||||
branches[band * 4 + branch] = value;
|
||||
const uint32_t tap_position =
|
||||
(room_position_ + room_capacity_ -
|
||||
(output_tap_delay_ % room_capacity_)) % room_capacity_;
|
||||
taps[band * 4 + branch] =
|
||||
room_memory_[room_memory_index(tap_position, band, branch)];
|
||||
}
|
||||
}
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
for (int branch = 0; branch < 4; ++branch) {
|
||||
room_memory_[room_memory_index(room_position_, band, branch)] =
|
||||
mul(branches[band * 4 + branch], feedback_[band][branch]);
|
||||
}
|
||||
}
|
||||
room_position_ = (room_position_ + 1) % room_capacity_;
|
||||
|
||||
std::array<Complex, 64 * 4> extra{};
|
||||
for (size_t index = 0; index < extra_delays_.size(); ++index) {
|
||||
const uint32_t position =
|
||||
(room_position_ + room_capacity_ -
|
||||
((extra_delays_[index] + 1) % room_capacity_)) % room_capacity_;
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
for (int target = 0; target < 4; ++target) {
|
||||
Complex mixed{};
|
||||
for (int source = 0; source < 4; ++source) {
|
||||
mixed = add(mixed, scale(room_memory_[
|
||||
room_memory_index(position, band, source)],
|
||||
extra_matrices_[index * 16 + target * 4 + source]));
|
||||
}
|
||||
extra[band * 4 + target] = add(
|
||||
extra[band * 4 + target],
|
||||
mul(mixed, extra_fields_[index * 64 + band]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::array<Complex, kEars * kHybrid> output{};
|
||||
for (int ear = 0; ear < 2; ++ear) {
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
Complex value{};
|
||||
for (int branch = 0; branch < 4; ++branch) {
|
||||
const Complex signal = add(
|
||||
scale(taps[band * 4 + branch], output_taps_[band][branch]),
|
||||
extra[band * 4 + branch]);
|
||||
value = add(value, mul(
|
||||
signal, output_matrix_[ear][band][branch]));
|
||||
}
|
||||
output[ear * kHybrid + band] = value;
|
||||
}
|
||||
}
|
||||
return output;
|
||||
}
|
||||
|
||||
void synthesize_hybrid(
|
||||
const std::array<Complex, kEars * kHybrid>& input,
|
||||
std::array<Complex, kEars * kQmf>& output) const noexcept {
|
||||
for (size_t mapping = 0; mapping < hybrid_values_.size(); ++mapping) {
|
||||
const int16_t* index = hybrid_indices_.data() + mapping * 4;
|
||||
const int input_band = index[0];
|
||||
const int input_component = index[1];
|
||||
const int output_band = index[2];
|
||||
const int output_component = index[3];
|
||||
const double gain = hybrid_values_[mapping];
|
||||
for (int ear = 0; ear < 2; ++ear) {
|
||||
const Complex source = input[ear * kHybrid + input_band];
|
||||
Complex& target = output[ear * kQmf + output_band];
|
||||
const double component = input_component == 0 ? source.re : source.im;
|
||||
(output_component == 0 ? target.re : target.im) += component * gain;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void synthesize_qmf(const Complex* input, int ear,
|
||||
std::array<double, 64>& output) noexcept {
|
||||
std::array<double, 64 * kRank> features{};
|
||||
std::array<double, 128> flat{};
|
||||
for (int band = 0; band < 64; ++band) {
|
||||
flat[band * 2] = input[band].re;
|
||||
flat[band * 2 + 1] = input[band].im;
|
||||
}
|
||||
for (int phase = 0; phase < 64; ++phase) {
|
||||
for (int rank = 0; rank < kRank; ++rank) {
|
||||
double value = 0.0;
|
||||
const size_t base = (static_cast<size_t>(phase) * kRank + rank) * 128;
|
||||
for (int component = 0; component < 128; ++component) {
|
||||
value += flat[component] * qmf_basis_[base + component];
|
||||
}
|
||||
features[phase * kRank + rank] = value;
|
||||
}
|
||||
}
|
||||
for (int phase = 0; phase < 64; ++phase) {
|
||||
double value = 0.0;
|
||||
for (int lag = 0; lag < 10; ++lag) {
|
||||
for (int rank = 0; rank < kRank; ++rank) {
|
||||
const double feature = lag == 0
|
||||
? features[phase * kRank + rank]
|
||||
: synthesis_history_[synthesis_history_index(
|
||||
ear, lag - 1, phase, rank)];
|
||||
value += feature * qmf_taps_[
|
||||
((static_cast<size_t>(phase) * 10 + lag) * kRank + rank)];
|
||||
}
|
||||
}
|
||||
output[phase] = value;
|
||||
}
|
||||
for (int lag = 8; lag > 0; --lag) {
|
||||
for (int phase = 0; phase < 64; ++phase) {
|
||||
for (int rank = 0; rank < kRank; ++rank) {
|
||||
synthesis_history_[synthesis_history_index(ear, lag, phase, rank)] =
|
||||
synthesis_history_[synthesis_history_index(
|
||||
ear, lag - 1, phase, rank)];
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int phase = 0; phase < 64; ++phase) {
|
||||
for (int rank = 0; rank < kRank; ++rank) {
|
||||
synthesis_history_[synthesis_history_index(ear, 0, phase, rank)] =
|
||||
features[phase * kRank + rank];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static size_t qmf_history_index(int lag, int channel, int phase) noexcept {
|
||||
return (static_cast<size_t>(lag) * kChannels + channel) * 64 + phase;
|
||||
}
|
||||
|
||||
static size_t hybrid_low_history_index(int lag, int channel, int band) noexcept {
|
||||
return (static_cast<size_t>(lag) * kChannels + channel) * 3 + band;
|
||||
}
|
||||
|
||||
static size_t hybrid_high_history_index(int delay, int channel, int band) noexcept {
|
||||
return (static_cast<size_t>(delay) * kChannels + channel) * 61 + band;
|
||||
}
|
||||
|
||||
static size_t hybrid_low_kernel_index(
|
||||
int input_band, int input_component, int lag,
|
||||
int output_band, int output_component) noexcept {
|
||||
return (((static_cast<size_t>(input_band) * 2 + input_component) * 13 + lag) *
|
||||
16 + output_band) * 2 + output_component;
|
||||
}
|
||||
|
||||
size_t room_memory_index(uint32_t position, int band, int branch) const noexcept {
|
||||
return (static_cast<size_t>(position) * 64 + band) * 4 + branch;
|
||||
}
|
||||
|
||||
static size_t synthesis_history_index(
|
||||
int ear, int lag, int phase, int rank) noexcept {
|
||||
return (((static_cast<size_t>(ear) * 9 + lag) * 64 + phase) * kRank + rank);
|
||||
}
|
||||
|
||||
bool kernels_ready_ = false;
|
||||
bool room_ready_ = false;
|
||||
std::array<double, 64 * 10> qmf_analysis_{};
|
||||
std::vector<double> hybrid_low_;
|
||||
std::vector<int16_t> hybrid_indices_;
|
||||
std::vector<double> hybrid_values_;
|
||||
std::array<double, 64 * kRank * 128> qmf_basis_{};
|
||||
std::array<double, 64 * 10 * kRank> qmf_taps_{};
|
||||
|
||||
std::array<double, 10 * kChannels * 64> qmf_history_{};
|
||||
std::array<Complex, 13 * kChannels * 3> hybrid_low_history_{};
|
||||
std::array<Complex, 6 * kChannels * 61> hybrid_high_history_{};
|
||||
std::array<double, kEars * 9 * 64 * kRank> synthesis_history_{};
|
||||
|
||||
uint32_t room_bands_ = 0;
|
||||
std::vector<uint32_t> allpass_delays_;
|
||||
std::vector<double> allpass_gains_;
|
||||
std::vector<size_t> allpass_offsets_;
|
||||
std::vector<uint32_t> allpass_positions_;
|
||||
std::vector<Complex> allpass_memory_;
|
||||
std::array<uint32_t, 4> fdn_delays_{};
|
||||
std::array<double, 16> fdn_matrix_{};
|
||||
uint32_t room_capacity_ = 0;
|
||||
uint32_t output_tap_delay_ = 0;
|
||||
std::array<std::array<Complex, 4>, 64> feedback_{};
|
||||
std::array<std::array<double, 4>, 64> output_taps_{};
|
||||
std::array<std::array<std::array<Complex, 4>, 64>, 2> output_matrix_{};
|
||||
std::vector<Complex> room_memory_;
|
||||
uint32_t room_position_ = 0;
|
||||
std::vector<uint32_t> extra_delays_;
|
||||
std::vector<Complex> extra_fields_;
|
||||
std::vector<double> extra_matrices_;
|
||||
|
||||
std::array<uint8_t, 128> bit_reverse_{};
|
||||
std::array<Complex, 64> premod_{};
|
||||
std::array<Complex, 64> post_{};
|
||||
std::array<Complex, 64> even_post_{};
|
||||
char error_[256]{};
|
||||
};
|
||||
|
||||
} // namespace ejoc::binaural
|
||||
|
||||
extern "C" {
|
||||
|
||||
ejoc_binaural_renderer_handle EJOC_CALL ejoc_binaural_renderer_create(void) {
|
||||
return new (std::nothrow) ejoc::binaural::Renderer();
|
||||
}
|
||||
|
||||
void EJOC_CALL ejoc_binaural_renderer_destroy(ejoc_binaural_renderer_handle handle) {
|
||||
delete static_cast<ejoc::binaural::Renderer*>(handle);
|
||||
}
|
||||
|
||||
int EJOC_CALL ejoc_binaural_renderer_reset(ejoc_binaural_renderer_handle handle) {
|
||||
return handle ? static_cast<ejoc::binaural::Renderer*>(handle)->reset() : -1;
|
||||
}
|
||||
|
||||
const char* EJOC_CALL ejoc_binaural_renderer_last_error(
|
||||
ejoc_binaural_renderer_handle handle) {
|
||||
return handle ? static_cast<ejoc::binaural::Renderer*>(handle)->error()
|
||||
: "null binaural renderer handle";
|
||||
}
|
||||
|
||||
int EJOC_CALL ejoc_binaural_renderer_configure_kernels(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
const double* qmf_analysis,
|
||||
const double* hybrid_low,
|
||||
const int16_t* hybrid_indices,
|
||||
const double* hybrid_values,
|
||||
uint32_t hybrid_count,
|
||||
const double* qmf_basis,
|
||||
const double* qmf_taps) {
|
||||
return handle ? static_cast<ejoc::binaural::Renderer*>(handle)->configure_kernels(
|
||||
qmf_analysis, hybrid_low, hybrid_indices, hybrid_values,
|
||||
hybrid_count, qmf_basis, qmf_taps) : -1;
|
||||
}
|
||||
|
||||
int EJOC_CALL ejoc_binaural_renderer_configure_room(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
uint32_t bands,
|
||||
uint32_t allpass_count,
|
||||
const uint32_t* allpass_delays,
|
||||
const double* allpass_gains,
|
||||
const uint32_t* fdn_delays,
|
||||
const double* fdn_matrix,
|
||||
uint32_t output_tap_delay,
|
||||
const double* feedback_complex,
|
||||
const double* output_taps,
|
||||
const double* output_complex,
|
||||
uint32_t extra_count,
|
||||
const uint32_t* extra_delays,
|
||||
const double* extra_fields_complex,
|
||||
const double* extra_matrices) {
|
||||
return handle ? static_cast<ejoc::binaural::Renderer*>(handle)->configure_room(
|
||||
bands, allpass_count, allpass_delays, allpass_gains,
|
||||
fdn_delays, fdn_matrix, output_tap_delay,
|
||||
feedback_complex, output_taps, output_complex,
|
||||
extra_count, extra_delays, extra_fields_complex, extra_matrices) : -1;
|
||||
}
|
||||
|
||||
int EJOC_CALL ejoc_binaural_renderer_process(
|
||||
ejoc_binaural_renderer_handle handle,
|
||||
const double* input16_interleaved,
|
||||
const double* gains_complex,
|
||||
const double* room_sends,
|
||||
double output_gain,
|
||||
double* output_stereo_interleaved) {
|
||||
return handle ? static_cast<ejoc::binaural::Renderer*>(handle)->process(
|
||||
input16_interleaved, gains_complex, room_sends,
|
||||
output_gain, output_stereo_interleaved) : -1;
|
||||
}
|
||||
|
||||
} // extern "C"
|
||||
@@ -664,13 +664,13 @@ uint32_t EJOC_CALL ejoc_abi_version(void) {
|
||||
|
||||
const char* EJOC_CALL ejoc_build_info(void) {
|
||||
#if defined(_MSC_VER)
|
||||
return "eac3joc-core abi=1 compiler=MSVC fft=fixed64 speaker=double crt=static-by-build";
|
||||
return "eac3joc-core abi=1 compiler=MSVC fft=fixed64 speaker=double binaural=double crt=static-by-build";
|
||||
#elif defined(__clang__)
|
||||
return "eac3joc-core abi=1 compiler=Clang fft=fixed64 speaker=double";
|
||||
return "eac3joc-core abi=1 compiler=Clang fft=fixed64 speaker=double binaural=double";
|
||||
#elif defined(__GNUC__)
|
||||
return "eac3joc-core abi=1 compiler=GCC fft=fixed64 speaker=double";
|
||||
return "eac3joc-core abi=1 compiler=GCC fft=fixed64 speaker=double binaural=double";
|
||||
#else
|
||||
return "eac3joc-core abi=1 compiler=unknown fft=fixed64 speaker=double";
|
||||
return "eac3joc-core abi=1 compiler=unknown fft=fixed64 speaker=double binaural=double";
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
+8
-4
@@ -14,7 +14,7 @@ import adm_atmos
|
||||
|
||||
|
||||
def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None,
|
||||
duration_sec=None, rate=48000):
|
||||
duration_sec=None, rate=48000, joc_binaural_mode=4):
|
||||
"""16ch f32 交织 raw → 25ch ADM BWF(空 7.1.2 bed + LFE + 15 对象)。
|
||||
|
||||
raw16: (n, 16) 交织(ch0 = LFE,ch1-15 = 对象)。
|
||||
@@ -54,7 +54,8 @@ def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None,
|
||||
kf_tracks.append(("JOC_Object_%d" % (oi + 1),
|
||||
[(0.0, 0.0, 0.0, 0.0, max(duration_sec, 1e-6))]))
|
||||
adm_atmos.build_master(out_path, BedView(), ObjView(), kf_tracks,
|
||||
duration_sec, rate=rate)
|
||||
duration_sec, rate=rate,
|
||||
joc_binaural_mode=joc_binaural_mode)
|
||||
# 及时释放 Windows 文件句柄,允许 TemporaryDirectory 删除中间 raw。
|
||||
raw._mmap.close()
|
||||
return out_path
|
||||
@@ -69,12 +70,14 @@ class StreamingMaster:
|
||||
channels 10..24.
|
||||
"""
|
||||
|
||||
def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072):
|
||||
def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072,
|
||||
joc_binaural_mode=4):
|
||||
if block_samples < 1536:
|
||||
raise ValueError("block_samples must be at least one E-AC-3 frame")
|
||||
self.out_path = os.fspath(out_path)
|
||||
self.duration_sec = float(duration_sec)
|
||||
self.rate = int(rate)
|
||||
self.joc_binaural_mode = joc_binaural_mode
|
||||
self._sink = adm_atmos.Sink25(self.out_path, 25, self.rate)
|
||||
self._buffer = np.empty((int(block_samples), 25), dtype=np.float32)
|
||||
self._used = 0
|
||||
@@ -112,7 +115,8 @@ class StreamingMaster:
|
||||
import adm_serializer
|
||||
axml = adm_serializer.build_axml(kf_tracks, self.duration_sec)
|
||||
chna = adm_atmos.build_chna()
|
||||
dbmd = adm_atmos.build_dbmd(25)
|
||||
dbmd = adm_atmos.build_dbmd(
|
||||
25, joc_binaural_mode=self.joc_binaural_mode)
|
||||
trajectory_blocks = sum(len(track[1]) for track in kf_tracks)
|
||||
self.metadata_info = {
|
||||
"axml_bytes": len(axml),
|
||||
|
||||
+22
-3
@@ -2,6 +2,7 @@
|
||||
|
||||
输出由 10 声道 7.1.2 bed 和 15 路对象组成;RF64 尺寸字段在写入完成后回填。
|
||||
"""
|
||||
import operator
|
||||
import struct
|
||||
import numpy as np
|
||||
import xml.etree.ElementTree as ET
|
||||
@@ -21,6 +22,14 @@ BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0),
|
||||
(-1.0, -1.0, 0.0), (1.0, -1.0, 0.0), (-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
|
||||
|
||||
N_OBJ = 15
|
||||
JOC_BINAURAL_MODES = {
|
||||
"off": 0,
|
||||
"near": 1,
|
||||
"far": 2,
|
||||
"mid": 3,
|
||||
"unspecified": 4,
|
||||
}
|
||||
JOC_BINAURAL_MODE_DEFAULT = "unspecified"
|
||||
|
||||
def q_to_adm_xyz(q1, q2, q3):
|
||||
posX = min(1.0, round(q1 * 62 / 32767.0) / 62.0)
|
||||
@@ -186,7 +195,11 @@ def _checksum(seg):
|
||||
s += b
|
||||
return (~s + 1) & 0xFF
|
||||
|
||||
def build_dbmd(object_count=25):
|
||||
def build_dbmd(object_count=25, joc_binaural_mode=4):
|
||||
"""仅覆盖 segment 10 中 JOC object slots 10..24 的 mode 低 3 bit。"""
|
||||
mode = operator.index(joc_binaural_mode)
|
||||
if mode not in JOC_BINAURAL_MODES.values():
|
||||
raise ValueError(f"invalid JOC binaural render mode: {mode}")
|
||||
out = bytearray(struct.pack("<I", 0x01000006))
|
||||
dd = bytearray(96)
|
||||
dd[1] = 0x47
|
||||
@@ -209,6 +222,12 @@ def build_dbmd(object_count=25):
|
||||
ob[4] = object_count
|
||||
for i in range(5 + 262, len(ob)):
|
||||
ob[i] = 0x84
|
||||
# sync (4), count (2), reserved (1), nine 15-byte config trims,
|
||||
# then one trim-bypass byte per track before the headphone modes.
|
||||
# Preserve the existing template's bed fields and trailing bytes.
|
||||
object_modes = 4 + 2 + 1 + 9 * 15 + object_count
|
||||
for i in range(10, min(object_count, 10 + N_OBJ)):
|
||||
ob[object_modes + i] = (ob[object_modes + i] & 0xF8) | mode
|
||||
out.append(10); out += struct.pack("<H", len(ob)); out += bytes(ob)
|
||||
out.append(_checksum(ob))
|
||||
out += b"\x00\x00"
|
||||
@@ -252,7 +271,7 @@ class Sink25:
|
||||
self.fp.close()
|
||||
|
||||
def build_master(out_path, bed_mm, obj_mm, kf_tracks, duration_sec, rate=48000,
|
||||
block=480000):
|
||||
block=480000, joc_binaural_mode=4):
|
||||
n = min(bed_mm.shape[0], obj_mm.shape[0])
|
||||
try:
|
||||
from . import adm_serializer
|
||||
@@ -261,7 +280,7 @@ def build_master(out_path, bed_mm, obj_mm, kf_tracks, duration_sec, rate=48000,
|
||||
serial_axml = adm_serializer.build_axml
|
||||
axml = serial_axml(kf_tracks, duration_sec)
|
||||
chna = build_chna()
|
||||
dbmd = build_dbmd(25)
|
||||
dbmd = build_dbmd(25, joc_binaural_mode=joc_binaural_mode)
|
||||
sink = Sink25(out_path, 25, rate)
|
||||
for st in range(0, n, block):
|
||||
en = min(n, st + block)
|
||||
|
||||
@@ -0,0 +1,188 @@
|
||||
"""Direct ID11/OAMD position scheduling for the Rosella binaural path."""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
import numpy as np
|
||||
|
||||
from adm_atmos import q_to_adm_xyz
|
||||
from oamd_bits import JocFieldState, frame_update
|
||||
from variant_error import UnsupportedVariantError
|
||||
|
||||
OAMD_UPDATE_QUANTUM_SAMPLES = 64
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PositionTransition:
|
||||
start_sample: int
|
||||
duration_samples: int
|
||||
origin: np.ndarray
|
||||
target: np.ndarray
|
||||
|
||||
@property
|
||||
def end_sample(self) -> int:
|
||||
return self.start_sample + self.duration_samples
|
||||
|
||||
|
||||
class _ObjectPositionTrack:
|
||||
def __init__(self):
|
||||
self.initial = np.zeros(3, dtype=np.float64)
|
||||
self.last_target = self.initial.copy()
|
||||
self.transitions: list[PositionTransition] = []
|
||||
self.cursor = 0
|
||||
self.last_query_sample = -1
|
||||
|
||||
def set_initial(self, position):
|
||||
target = np.asarray(position, dtype=np.float64)
|
||||
self.initial = target.copy()
|
||||
self.last_target = target.copy()
|
||||
|
||||
def append(self, start_sample: int, duration_samples: int, target,
|
||||
object_index: int):
|
||||
start = int(start_sample)
|
||||
duration = int(duration_samples)
|
||||
if start < 0 or duration < 0:
|
||||
raise ValueError("position transition timing must be non-negative")
|
||||
target = np.asarray(target, dtype=np.float64)
|
||||
if self.transitions:
|
||||
previous = self.transitions[-1]
|
||||
if start < previous.end_sample:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "overlapping_binaural_position_ramps",
|
||||
"同一对象的新位置更新在上一双耳 ramp 完成前到达",
|
||||
details={
|
||||
"object": object_index,
|
||||
"ramp_start_sample": previous.start_sample,
|
||||
"ramp_end_sample": previous.end_sample,
|
||||
"next_update_sample": start,
|
||||
})
|
||||
if start == previous.start_sample and previous.duration_samples == 0:
|
||||
self.transitions[-1] = PositionTransition(
|
||||
start, duration, previous.origin.copy(), target.copy())
|
||||
self.last_target = target.copy()
|
||||
return
|
||||
self.transitions.append(PositionTransition(
|
||||
start, duration, self.last_target.copy(), target.copy()))
|
||||
self.last_target = target.copy()
|
||||
|
||||
def position_at(self, sample: int) -> np.ndarray:
|
||||
sample = int(sample)
|
||||
if sample < self.last_query_sample:
|
||||
raise ValueError("binaural metadata positions must be queried monotonically")
|
||||
self.last_query_sample = sample
|
||||
while self.cursor < len(self.transitions):
|
||||
transition = self.transitions[self.cursor]
|
||||
if sample < transition.end_sample:
|
||||
break
|
||||
self.initial = transition.target.copy()
|
||||
self.cursor += 1
|
||||
if self.cursor >= len(self.transitions):
|
||||
return self.initial
|
||||
transition = self.transitions[self.cursor]
|
||||
if sample < transition.start_sample:
|
||||
return self.initial
|
||||
if transition.duration_samples == 0:
|
||||
return transition.target
|
||||
amount = (sample - transition.start_sample) / float(transition.duration_samples)
|
||||
return transition.origin + (transition.target - transition.origin) * amount
|
||||
|
||||
|
||||
class OamdPositionTimeline:
|
||||
"""Convert OAMD state updates into a sample-timed Cartesian trajectory."""
|
||||
|
||||
def __init__(self, object_count: int = 15):
|
||||
if object_count != 15:
|
||||
raise ValueError("JOC OAMD currently requires 15 object slots")
|
||||
self.object_count = int(object_count)
|
||||
self.state = JocFieldState()
|
||||
self.tracks = [_ObjectPositionTrack() for _ in range(self.object_count)]
|
||||
self.initialized = False
|
||||
self.previous_targets: list[tuple[float, float, float] | None] = [
|
||||
None] * self.object_count
|
||||
self.payload_count = 0
|
||||
self.transition_count = 0
|
||||
self.last_coded_event_sample = -1
|
||||
|
||||
def _targets(self) -> list[tuple[float, float, float]]:
|
||||
q = self.state.q
|
||||
return [
|
||||
q_to_adm_xyz(
|
||||
q[(object_index, "q1")],
|
||||
q[(object_index, "q2")],
|
||||
q[(object_index, "q3")],
|
||||
)
|
||||
for object_index in range(1, self.object_count + 1)
|
||||
]
|
||||
|
||||
def submit_update(self, update: dict, *, frame_start_sample: int,
|
||||
outer_sample_offset: int = 0,
|
||||
object_delay_samples: int = 1473,
|
||||
processed_sample: int = 0):
|
||||
"""Schedule one already-parsed :func:`oamd_bits.frame_update` result."""
|
||||
frame_start = int(frame_start_sample)
|
||||
outer_offset = int(outer_sample_offset)
|
||||
object_delay = int(object_delay_samples)
|
||||
if min(frame_start, outer_offset, object_delay) < 0:
|
||||
raise ValueError("OAMD frame, outer offset, and object delay must be non-negative")
|
||||
self.state.apply(update["values"])
|
||||
targets = self._targets()
|
||||
coded_event = (
|
||||
frame_start + outer_offset + int(update["block_offset_samples"]))
|
||||
if coded_event < self.last_coded_event_sample:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "non_monotonic_binaural_updates",
|
||||
"双耳 OAMD 更新时间倒退",
|
||||
details={
|
||||
"event_sample": coded_event,
|
||||
"previous_event_sample": self.last_coded_event_sample,
|
||||
})
|
||||
self.last_coded_event_sample = coded_event
|
||||
|
||||
if not self.initialized:
|
||||
if int(processed_sample) > 0:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "late_initial_binaural_state",
|
||||
"首个 OAMD 状态在双耳 PCM 已处理后才出现,无法回填 sample 0",
|
||||
details={
|
||||
"processed_sample": int(processed_sample),
|
||||
"first_event_sample": coded_event,
|
||||
})
|
||||
for index, target in enumerate(targets):
|
||||
self.tracks[index].set_initial(target)
|
||||
self.previous_targets[index] = target
|
||||
self.initialized = True
|
||||
self.payload_count += 1
|
||||
return
|
||||
|
||||
ramp_duration = int(update["ramp_duration_samples"])
|
||||
effective_ramp = max(0, ramp_duration - OAMD_UPDATE_QUANTUM_SAMPLES)
|
||||
transition_start = coded_event + object_delay
|
||||
if effective_ramp:
|
||||
transition_start += OAMD_UPDATE_QUANTUM_SAMPLES
|
||||
for index, target in enumerate(targets):
|
||||
if self.previous_targets[index] == target:
|
||||
continue
|
||||
self.tracks[index].append(
|
||||
transition_start, effective_ramp, target, index + 1)
|
||||
self.previous_targets[index] = target
|
||||
self.transition_count += 1
|
||||
self.payload_count += 1
|
||||
|
||||
def submit_payload(self, payload, *, frame_start_sample: int,
|
||||
outer_sample_offset: int = 0,
|
||||
object_delay_samples: int = 1473,
|
||||
processed_sample: int = 0):
|
||||
update = frame_update(payload)
|
||||
self.submit_update(
|
||||
update,
|
||||
frame_start_sample=frame_start_sample,
|
||||
outer_sample_offset=outer_sample_offset,
|
||||
object_delay_samples=object_delay_samples,
|
||||
processed_sample=processed_sample,
|
||||
)
|
||||
return update
|
||||
|
||||
def positions_at(self, sample: int) -> np.ndarray:
|
||||
return np.stack(
|
||||
[track.position_at(sample) for track in self.tracks], axis=0
|
||||
).astype(np.float64, copy=False)
|
||||
@@ -0,0 +1,214 @@
|
||||
"""ctypes bridge for the native float64 binaural DSP."""
|
||||
from __future__ import annotations
|
||||
|
||||
import ctypes
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from native_renderer import ABI_VERSION, find_native_library
|
||||
from rosella_filterbank import DEFAULT_KERNEL_DATA, load_kernel_tables
|
||||
from rosella_model import RosellaModel
|
||||
|
||||
BLOCK_SAMPLES = 512
|
||||
INPUT_CHANNELS = 16
|
||||
OUTPUT_CHANNELS = 2
|
||||
HYBRID_BANDS = 77
|
||||
|
||||
|
||||
class NativeBinauralDsp:
|
||||
def __init__(self, model: RosellaModel, *, library_path=None,
|
||||
kernel_data: str | Path = DEFAULT_KERNEL_DATA):
|
||||
self.library_path = find_native_library(library_path)
|
||||
self._lib = ctypes.CDLL(str(self.library_path))
|
||||
self._bind()
|
||||
version = int(self._lib.ejoc_abi_version())
|
||||
if version != ABI_VERSION:
|
||||
raise RuntimeError(
|
||||
f"native ABI mismatch: expected {ABI_VERSION}, got {version}")
|
||||
self._handle = self._lib.ejoc_binaural_renderer_create()
|
||||
if not self._handle:
|
||||
raise RuntimeError("native binaural renderer creation failed")
|
||||
try:
|
||||
self._configure_kernels(kernel_data)
|
||||
self._configure_room(model)
|
||||
except Exception:
|
||||
self.close()
|
||||
raise
|
||||
|
||||
def _bind(self):
|
||||
void_p = ctypes.c_void_p
|
||||
f64_p = ctypes.POINTER(ctypes.c_double)
|
||||
i16_p = ctypes.POINTER(ctypes.c_int16)
|
||||
u32_p = ctypes.POINTER(ctypes.c_uint32)
|
||||
self._lib.ejoc_abi_version.argtypes = []
|
||||
self._lib.ejoc_abi_version.restype = ctypes.c_uint32
|
||||
self._lib.ejoc_binaural_renderer_create.argtypes = []
|
||||
self._lib.ejoc_binaural_renderer_create.restype = void_p
|
||||
self._lib.ejoc_binaural_renderer_destroy.argtypes = [void_p]
|
||||
self._lib.ejoc_binaural_renderer_destroy.restype = None
|
||||
self._lib.ejoc_binaural_renderer_reset.argtypes = [void_p]
|
||||
self._lib.ejoc_binaural_renderer_reset.restype = ctypes.c_int
|
||||
self._lib.ejoc_binaural_renderer_last_error.argtypes = [void_p]
|
||||
self._lib.ejoc_binaural_renderer_last_error.restype = ctypes.c_char_p
|
||||
self._lib.ejoc_binaural_renderer_configure_kernels.argtypes = [
|
||||
void_p, f64_p, f64_p, i16_p, f64_p, ctypes.c_uint32, f64_p, f64_p]
|
||||
self._lib.ejoc_binaural_renderer_configure_kernels.restype = ctypes.c_int
|
||||
self._lib.ejoc_binaural_renderer_configure_room.argtypes = [
|
||||
void_p, ctypes.c_uint32, ctypes.c_uint32, u32_p, f64_p,
|
||||
u32_p, f64_p, ctypes.c_uint32, f64_p, f64_p, f64_p,
|
||||
ctypes.c_uint32, u32_p, f64_p, f64_p]
|
||||
self._lib.ejoc_binaural_renderer_configure_room.restype = ctypes.c_int
|
||||
self._lib.ejoc_binaural_renderer_process.argtypes = [
|
||||
void_p, f64_p, f64_p, f64_p, ctypes.c_double, f64_p]
|
||||
self._lib.ejoc_binaural_renderer_process.restype = ctypes.c_int
|
||||
|
||||
def _raise(self, operation, status):
|
||||
message = self._lib.ejoc_binaural_renderer_last_error(self._handle)
|
||||
detail = (message or b"").decode("utf-8", "replace")
|
||||
raise RuntimeError(
|
||||
f"native binaural renderer {operation} failed ({status}): {detail}")
|
||||
|
||||
@staticmethod
|
||||
def _f64_pointer(values):
|
||||
return values.ctypes.data_as(ctypes.POINTER(ctypes.c_double))
|
||||
|
||||
def _configure_kernels(self, kernel_data):
|
||||
tables = load_kernel_tables(kernel_data)
|
||||
qmf_analysis = np.ascontiguousarray(
|
||||
tables["qmf_analysis_coefficients"], dtype=np.float64)
|
||||
hybrid_low = np.ascontiguousarray(
|
||||
tables["hybrid_analysis_low_kernel"], dtype=np.float64)
|
||||
hybrid_indices = np.ascontiguousarray(
|
||||
tables["hybrid_synthesis_indices"], dtype=np.int16)
|
||||
hybrid_values = np.ascontiguousarray(
|
||||
tables["hybrid_synthesis_values"], dtype=np.float64)
|
||||
qmf_basis = np.ascontiguousarray(
|
||||
tables["qmf_synthesis_basis"], dtype=np.float64)
|
||||
qmf_taps = np.ascontiguousarray(
|
||||
tables["qmf_synthesis_taps"], dtype=np.float64)
|
||||
status = self._lib.ejoc_binaural_renderer_configure_kernels(
|
||||
self._handle,
|
||||
self._f64_pointer(qmf_analysis),
|
||||
self._f64_pointer(hybrid_low),
|
||||
hybrid_indices.ctypes.data_as(ctypes.POINTER(ctypes.c_int16)),
|
||||
self._f64_pointer(hybrid_values),
|
||||
len(hybrid_values),
|
||||
self._f64_pointer(qmf_basis),
|
||||
self._f64_pointer(qmf_taps),
|
||||
)
|
||||
if status:
|
||||
self._raise("configure_kernels", status)
|
||||
|
||||
def _configure_room(self, model: RosellaModel):
|
||||
if float(model.table_a_scalar) >= 0.5:
|
||||
raise NotImplementedError("alternate table-A room mode")
|
||||
bands = min(64, model.table_a_dimension)
|
||||
allpass_delays = np.ascontiguousarray(
|
||||
model.table_a_option_ids, dtype=np.uint32)
|
||||
allpass_gains = np.ascontiguousarray(
|
||||
model.table_a_option_values, dtype=np.float64)
|
||||
fdn_delays = np.ascontiguousarray(
|
||||
model.table_a_four_integers, dtype=np.uint32)
|
||||
fdn_matrix = np.ascontiguousarray(
|
||||
np.asarray(model.table_a_vector16, dtype=np.float64).reshape(
|
||||
4, 4, order="F"))
|
||||
|
||||
filter8 = np.asarray(
|
||||
model.table_a_filter_8x64_padded, dtype=np.float64).reshape(20, 4, 2, 4)
|
||||
filter4 = np.asarray(
|
||||
model.table_a_filter_4x64_padded, dtype=np.float64).reshape(20, 4, 4)
|
||||
filter16 = np.asarray(
|
||||
model.table_a_filter_16x64_padded, dtype=np.float64).reshape(20, 4, 4, 4)
|
||||
feedback = np.empty((64, 4, 2), dtype=np.float64)
|
||||
output_taps = np.empty((64, 4), dtype=np.float64)
|
||||
output_matrix = np.empty((2, 64, 4, 2), dtype=np.float64)
|
||||
for band in range(64):
|
||||
group, lane = divmod(band, 4)
|
||||
feedback[band, :, 0] = filter8[group, :, 0, lane]
|
||||
feedback[band, :, 1] = filter8[group, :, 1, lane]
|
||||
output_taps[band] = filter4[group, :, lane]
|
||||
output_matrix[0, band, :, 0] = filter16[group, :, 0, lane]
|
||||
output_matrix[0, band, :, 1] = filter16[group, :, 1, lane]
|
||||
output_matrix[1, band, :, 0] = filter16[group, :, 2, lane]
|
||||
output_matrix[1, band, :, 1] = filter16[group, :, 3, lane]
|
||||
|
||||
extra_count = int(model.table_a_extra)
|
||||
extra_delays = np.ascontiguousarray(
|
||||
model.table_a_extra_indices, dtype=np.uint32)
|
||||
extra_fields = np.empty((extra_count, 64, 2), dtype=np.float64)
|
||||
extra_source = np.asarray(
|
||||
model.table_a_extra_fields_padded, dtype=np.float64).reshape(
|
||||
extra_count, 20, 2, 4)
|
||||
for extra in range(extra_count):
|
||||
for band in range(64):
|
||||
group, lane = divmod(band, 4)
|
||||
extra_fields[extra, band] = extra_source[extra, group, :, lane]
|
||||
extra_matrices = np.empty((extra_count, 4, 4), dtype=np.float64)
|
||||
for extra in range(extra_count):
|
||||
extra_matrices[extra] = np.asarray(
|
||||
model.table_a_extra_vectors[extra], dtype=np.float64).reshape(
|
||||
4, 4, order="F")
|
||||
|
||||
null_u32 = ctypes.POINTER(ctypes.c_uint32)()
|
||||
null_f64 = ctypes.POINTER(ctypes.c_double)()
|
||||
status = self._lib.ejoc_binaural_renderer_configure_room(
|
||||
self._handle,
|
||||
bands,
|
||||
len(allpass_delays),
|
||||
allpass_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)),
|
||||
self._f64_pointer(allpass_gains),
|
||||
fdn_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)),
|
||||
self._f64_pointer(fdn_matrix),
|
||||
int(model.table_a_integer),
|
||||
self._f64_pointer(feedback),
|
||||
self._f64_pointer(output_taps),
|
||||
self._f64_pointer(output_matrix),
|
||||
extra_count,
|
||||
(extra_delays.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32))
|
||||
if extra_count else null_u32),
|
||||
self._f64_pointer(extra_fields) if extra_count else null_f64,
|
||||
self._f64_pointer(extra_matrices) if extra_count else null_f64,
|
||||
)
|
||||
if status:
|
||||
self._raise("configure_room", status)
|
||||
|
||||
def reset(self):
|
||||
if not self._handle:
|
||||
raise RuntimeError("native binaural renderer is closed")
|
||||
status = self._lib.ejoc_binaural_renderer_reset(self._handle)
|
||||
if status:
|
||||
self._raise("reset", status)
|
||||
|
||||
def process_block(self, pcm16, gains, room_sends, output_gain=1.0):
|
||||
if not self._handle:
|
||||
raise RuntimeError("native binaural renderer is closed")
|
||||
source = np.ascontiguousarray(pcm16, dtype=np.float64)
|
||||
gain_values = np.asarray(gains)
|
||||
sends = np.ascontiguousarray(room_sends, dtype=np.float64)
|
||||
if source.shape != (BLOCK_SAMPLES, INPUT_CHANNELS):
|
||||
raise ValueError(f"pcm16 block must be (512,16), got {source.shape}")
|
||||
if gain_values.shape != (INPUT_CHANNELS, OUTPUT_CHANNELS, HYBRID_BANDS):
|
||||
raise ValueError(f"gains must be (16,2,77), got {gain_values.shape}")
|
||||
direct = np.ascontiguousarray(
|
||||
gain_values, dtype=np.complex128).view(np.float64)
|
||||
if sends.shape != (INPUT_CHANNELS,):
|
||||
raise ValueError(f"room_sends must be (16,), got {sends.shape}")
|
||||
output = np.empty((BLOCK_SAMPLES, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
status = self._lib.ejoc_binaural_renderer_process(
|
||||
self._handle,
|
||||
self._f64_pointer(source),
|
||||
self._f64_pointer(direct),
|
||||
self._f64_pointer(sends),
|
||||
float(output_gain),
|
||||
self._f64_pointer(output),
|
||||
)
|
||||
if status:
|
||||
self._raise("process", status)
|
||||
return output
|
||||
|
||||
def close(self):
|
||||
handle = getattr(self, "_handle", None)
|
||||
if handle:
|
||||
self._lib.ejoc_binaural_renderer_destroy(handle)
|
||||
self._handle = None
|
||||
@@ -0,0 +1,301 @@
|
||||
"""Stateful binaural renderer for reconstructed JOC objects."""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import math
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from binaural_metadata import OamdPositionTimeline
|
||||
from binaural_native_renderer import NativeBinauralDsp
|
||||
from rosella_core import RosellaRenderer
|
||||
from rosella_direct import BINAURAL_PROFILE_NAMES
|
||||
from rosella_filterbank import (
|
||||
DEFAULT_KERNEL_DATA,
|
||||
HybridAnalysis,
|
||||
HybridSynthesis,
|
||||
QmfAnalysis,
|
||||
QmfSynthesis,
|
||||
)
|
||||
from rosella_model import RosellaModel, load_personalized_headphone
|
||||
|
||||
SAMPLE_RATE = 48000
|
||||
FRAME_SAMPLES = 1536
|
||||
ROSSELLA_BLOCK_SAMPLES = 512
|
||||
QMF_HOP_SAMPLES = 64
|
||||
ROSSELLA_LATENCY_SAMPLES = 961
|
||||
SOURCE_CHANNELS = 16
|
||||
OUTPUT_CHANNELS = 2
|
||||
PROJECT_DIR = Path(__file__).resolve().parent.parent
|
||||
DEFAULT_PERSONALIZED_HEADPHONE = (
|
||||
PROJECT_DIR / "HRTF" / "binaural.personalized_headphone")
|
||||
|
||||
|
||||
def _sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as stream:
|
||||
for block in iter(lambda: stream.read(1 << 20), b""):
|
||||
digest.update(block)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def resolve_personalized_headphone(path: str | Path | None = None) -> Path:
|
||||
target = (DEFAULT_PERSONALIZED_HEADPHONE if path is None
|
||||
else Path(path).expanduser().resolve())
|
||||
if not target.is_file():
|
||||
raise FileNotFoundError(
|
||||
f"未找到双耳模型:{target}\n"
|
||||
"请将兼容模型保存为 HRTF/binaural.personalized_headphone,"
|
||||
"或使用 --personalized-headphone PATH 指定文件。"
|
||||
)
|
||||
return target
|
||||
|
||||
|
||||
class RosellaBinauralRenderer:
|
||||
"""Render interleaved LFE plus fifteen objects to stereo."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
personalized_headphone: str | Path | RosellaModel,
|
||||
*,
|
||||
mode: str = "mid",
|
||||
kernel_data: str | Path = DEFAULT_KERNEL_DATA,
|
||||
object_delay_samples: int = 1473,
|
||||
tail_seconds: float = 5.0,
|
||||
output_gain: float = 1.0,
|
||||
chunk_frames: int = 64,
|
||||
room_impulse_slots: int = 4096,
|
||||
backend: str = "python",
|
||||
native_library=None):
|
||||
if mode not in BINAURAL_PROFILE_NAMES:
|
||||
raise ValueError("binaural mode must be near, mid, or far")
|
||||
if int(object_delay_samples) < 0:
|
||||
raise ValueError("object_delay_samples must be non-negative")
|
||||
if float(tail_seconds) < 0.0:
|
||||
raise ValueError("tail_seconds must be non-negative")
|
||||
if int(chunk_frames) <= 0:
|
||||
raise ValueError("chunk_frames must be positive")
|
||||
if not math.isfinite(float(output_gain)):
|
||||
raise ValueError("output_gain must be finite")
|
||||
if backend not in ("auto", "native", "python"):
|
||||
raise ValueError("backend must be auto, native, or python")
|
||||
|
||||
if isinstance(personalized_headphone, RosellaModel):
|
||||
self.model = personalized_headphone
|
||||
self.model_path = Path(self.model.source_path)
|
||||
else:
|
||||
self.model_path = resolve_personalized_headphone(personalized_headphone)
|
||||
self.model = load_personalized_headphone(self.model_path)
|
||||
if self.model.sample_rate != SAMPLE_RATE:
|
||||
raise ValueError(
|
||||
f"Rosella model sample rate must be {SAMPLE_RATE}, got {self.model.sample_rate}")
|
||||
|
||||
self.mode = mode
|
||||
self.profile_index = BINAURAL_PROFILE_NAMES[mode]
|
||||
self.kernel_data = Path(kernel_data).expanduser().resolve()
|
||||
self.kernel_data_sha256 = _sha256_file(self.kernel_data)
|
||||
self.object_delay_samples = int(object_delay_samples)
|
||||
self.tail_seconds = float(tail_seconds)
|
||||
self.output_gain = np.float64(output_gain)
|
||||
self.chunk_frames = int(chunk_frames)
|
||||
self.chunk_samples = self.chunk_frames * FRAME_SAMPLES
|
||||
|
||||
self.native_dsp = None
|
||||
self.backend_fallback = None
|
||||
if backend in ("auto", "native"):
|
||||
try:
|
||||
self.native_dsp = NativeBinauralDsp(
|
||||
self.model, library_path=native_library,
|
||||
kernel_data=self.kernel_data)
|
||||
except (AttributeError, OSError, RuntimeError) as exc:
|
||||
if backend == "native":
|
||||
raise RuntimeError(f"native binaural backend unavailable: {exc}") from exc
|
||||
self.backend_fallback = str(exc)
|
||||
if self.native_dsp is not None:
|
||||
self.dsp_backend = "native"
|
||||
self.qmf_analysis = None
|
||||
self.hybrid_analysis = None
|
||||
self.hybrid_synthesis = None
|
||||
self.qmf_synthesis = None
|
||||
self.core = RosellaRenderer(
|
||||
self.model, SOURCE_CHANNELS, create_room=False)
|
||||
else:
|
||||
self.dsp_backend = "python"
|
||||
self.qmf_analysis = QmfAnalysis(SOURCE_CHANNELS, self.kernel_data)
|
||||
self.hybrid_analysis = HybridAnalysis(SOURCE_CHANNELS, self.kernel_data)
|
||||
self.core = RosellaRenderer(
|
||||
self.model, SOURCE_CHANNELS,
|
||||
room_impulse_slots=room_impulse_slots)
|
||||
self.hybrid_synthesis = HybridSynthesis(OUTPUT_CHANNELS, self.kernel_data)
|
||||
self.qmf_synthesis = QmfSynthesis(OUTPUT_CHANNELS, self.kernel_data)
|
||||
self.timeline = OamdPositionTimeline(15)
|
||||
|
||||
self._input_buffer = np.empty(
|
||||
(self.chunk_samples, SOURCE_CHANNELS), dtype=np.float64)
|
||||
self._buffer_used = 0
|
||||
self.input_samples = 0
|
||||
self.processed_input_samples = 0
|
||||
self.raw_output_samples = 0
|
||||
self.output_samples = 0
|
||||
self.finished = False
|
||||
self.metadata_block_updates = 0
|
||||
|
||||
def _append_input(self, samples: np.ndarray) -> list[np.ndarray]:
|
||||
outputs = []
|
||||
source = np.asarray(samples, dtype=np.float64)
|
||||
position = 0
|
||||
while position < len(source):
|
||||
count = min(self.chunk_samples - self._buffer_used,
|
||||
len(source) - position)
|
||||
self._input_buffer[self._buffer_used:self._buffer_used + count] = (
|
||||
source[position:position + count])
|
||||
self._buffer_used += count
|
||||
position += count
|
||||
if self._buffer_used == self.chunk_samples:
|
||||
outputs.append(self._process_samples(self._input_buffer))
|
||||
self._buffer_used = 0
|
||||
return outputs
|
||||
|
||||
def render_frame(self, objects16, payload=None, metadata_offset=None,
|
||||
*, outer_sample_offset=0) -> np.ndarray:
|
||||
"""Submit one 1536-sample reconstructed frame and its ID11 payload."""
|
||||
if self.finished:
|
||||
raise RuntimeError("binaural renderer is already finished")
|
||||
source = np.asarray(objects16)
|
||||
if source.shape != (FRAME_SAMPLES, SOURCE_CHANNELS):
|
||||
raise ValueError(
|
||||
f"binaural frame must have shape ({FRAME_SAMPLES},{SOURCE_CHANNELS}), "
|
||||
f"got {source.shape}")
|
||||
frame_start = self.input_samples
|
||||
metadata_delay = (self.object_delay_samples if metadata_offset is None
|
||||
else int(metadata_offset))
|
||||
if metadata_delay < 0:
|
||||
raise ValueError("metadata_offset must be non-negative")
|
||||
if payload is not None:
|
||||
self.timeline.submit_payload(
|
||||
payload,
|
||||
frame_start_sample=frame_start,
|
||||
outer_sample_offset=int(outer_sample_offset),
|
||||
object_delay_samples=metadata_delay,
|
||||
processed_sample=self.processed_input_samples,
|
||||
)
|
||||
self.metadata_block_updates += 1
|
||||
self.input_samples += FRAME_SAMPLES
|
||||
chunks = self._append_input(source)
|
||||
if not chunks:
|
||||
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
return np.concatenate(chunks, axis=0) if len(chunks) > 1 else chunks[0]
|
||||
|
||||
def _set_block_parameters(self, sample: int):
|
||||
positions = self.timeline.positions_at(sample)
|
||||
self.core.set_source(0, (0.0, 1.0, 0.0), special_lfe=True)
|
||||
for object_index in range(15):
|
||||
self.core.set_source(
|
||||
object_index + 1, positions[object_index], self.profile_index)
|
||||
|
||||
def _process_samples(self, source: np.ndarray) -> np.ndarray:
|
||||
values = np.asarray(source, dtype=np.float64)
|
||||
if values.ndim != 2 or values.shape[1] != SOURCE_CHANNELS:
|
||||
raise ValueError(f"expected [samples,{SOURCE_CHANNELS}], got {values.shape}")
|
||||
if len(values) % ROSSELLA_BLOCK_SAMPLES:
|
||||
raise ValueError("binaural input must be divisible by 512 samples")
|
||||
blocks = len(values) // ROSSELLA_BLOCK_SAMPLES
|
||||
block_base = self.processed_input_samples
|
||||
|
||||
if self.native_dsp is not None:
|
||||
stereo = np.empty((len(values), OUTPUT_CHANNELS), dtype=np.float64)
|
||||
for block in range(blocks):
|
||||
sample = block_base + block * ROSSELLA_BLOCK_SAMPLES
|
||||
self._set_block_parameters(sample)
|
||||
start = block * ROSSELLA_BLOCK_SAMPLES
|
||||
stop = start + ROSSELLA_BLOCK_SAMPLES
|
||||
stereo[start:stop] = self.native_dsp.process_block(
|
||||
values[start:stop], self.core.gains, self.core.room_sends,
|
||||
self.output_gain)
|
||||
else:
|
||||
hops = values.reshape(
|
||||
blocks, ROSSELLA_BLOCK_SAMPLES // QMF_HOP_SAMPLES,
|
||||
QMF_HOP_SAMPLES, SOURCE_CHANNELS,
|
||||
).transpose(0, 1, 3, 2).reshape(
|
||||
blocks * (ROSSELLA_BLOCK_SAMPLES // QMF_HOP_SAMPLES),
|
||||
SOURCE_CHANNELS, QMF_HOP_SAMPLES)
|
||||
hybrid = self.hybrid_analysis.process_chunk(
|
||||
self.qmf_analysis.process_chunk(hops))
|
||||
direct = np.empty((blocks * 8, OUTPUT_CHANNELS, 77), dtype=np.complex128)
|
||||
room_send = np.empty((blocks * 8, 77), dtype=np.complex128)
|
||||
for block in range(blocks):
|
||||
sample = block_base + block * ROSSELLA_BLOCK_SAMPLES
|
||||
self._set_block_parameters(sample)
|
||||
start = block * 8
|
||||
stop = start + 8
|
||||
direct[start:stop], room_send[start:stop] = (
|
||||
self.core.direct_and_send_static(hybrid[start:stop]))
|
||||
rendered = direct + self.core.room.process_chunk(room_send)
|
||||
time_bands = self.qmf_synthesis.process_chunk(
|
||||
self.hybrid_synthesis.process_chunk(rendered))
|
||||
stereo = time_bands.transpose(0, 2, 1).reshape(
|
||||
blocks * ROSSELLA_BLOCK_SAMPLES, OUTPUT_CHANNELS)
|
||||
stereo *= self.output_gain
|
||||
|
||||
skip = max(0, min(
|
||||
len(stereo), ROSSELLA_LATENCY_SAMPLES - self.raw_output_samples))
|
||||
self.raw_output_samples += len(stereo)
|
||||
self.processed_input_samples += len(values)
|
||||
output = stereo[skip:]
|
||||
self.output_samples += len(output)
|
||||
return output
|
||||
|
||||
def finish(self) -> np.ndarray:
|
||||
"""Process pending source samples and preserve the configured room tail."""
|
||||
if self.finished:
|
||||
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
outputs: list[np.ndarray] = []
|
||||
if self._buffer_used:
|
||||
outputs.append(self._process_samples(
|
||||
self._input_buffer[:self._buffer_used]))
|
||||
self._buffer_used = 0
|
||||
flush_samples = math.ceil(
|
||||
(self.tail_seconds * SAMPLE_RATE
|
||||
+ ROSSELLA_LATENCY_SAMPLES + ROSSELLA_BLOCK_SAMPLES)
|
||||
/ ROSSELLA_BLOCK_SAMPLES) * ROSSELLA_BLOCK_SAMPLES
|
||||
while flush_samples:
|
||||
count = min(flush_samples, self.chunk_samples)
|
||||
zero = np.zeros((count, SOURCE_CHANNELS), dtype=np.float64)
|
||||
outputs.append(self._process_samples(zero))
|
||||
flush_samples -= count
|
||||
self.finished = True
|
||||
nonempty = [value for value in outputs if len(value)]
|
||||
if not nonempty:
|
||||
return np.empty((0, OUTPUT_CHANNELS), dtype=np.float64)
|
||||
return np.concatenate(nonempty, axis=0)
|
||||
|
||||
def close(self):
|
||||
if self.native_dsp is not None:
|
||||
self.native_dsp.close()
|
||||
self.finished = True
|
||||
|
||||
@property
|
||||
def backend_info(self) -> dict:
|
||||
return {
|
||||
"name": self.dsp_backend,
|
||||
"precision": "float64/complex128",
|
||||
"fallback_reason": self.backend_fallback,
|
||||
"library": (str(self.native_dsp.library_path)
|
||||
if self.native_dsp is not None else None),
|
||||
"model": str(self.model_path.resolve()),
|
||||
"model_coefficients": int(len(self.model.coefficients)),
|
||||
"model_coefficient_sha256": self.model.coefficient_sha256,
|
||||
"model_version": self.model.coefficient_version,
|
||||
"kernel_data": str(self.kernel_data),
|
||||
"kernel_data_sha256": self.kernel_data_sha256,
|
||||
"mode": self.mode,
|
||||
"latency_compensated_samples": ROSSELLA_LATENCY_SAMPLES,
|
||||
"object_delay_samples": self.object_delay_samples,
|
||||
"tail_seconds": self.tail_seconds,
|
||||
"metadata_payloads": self.timeline.payload_count,
|
||||
"metadata_position_transitions": self.timeline.transition_count,
|
||||
"input_samples": self.input_samples,
|
||||
"processed_samples_including_flush": self.processed_input_samples,
|
||||
"output_samples_before_tail_trim": self.output_samples,
|
||||
}
|
||||
+30
-5
@@ -163,7 +163,13 @@ def _marker_offsets(data):
|
||||
|
||||
|
||||
def find_joc_emdf(frame):
|
||||
"""返回同步帧中唯一、连续且包含 ID11/ID14 的 JOC EMDF 容器。"""
|
||||
"""返回同步帧中唯一、顶层连续且包含 ID11/ID14 的 JOC EMDF 容器。
|
||||
|
||||
EMDF payload 是不透明字节串,其中可能自然出现另一个 ``0x5838``。若从这个
|
||||
内嵌 marker 开始的后续随机位恰好也能通过容器语法探测,它仍不是一个独立的
|
||||
transport 容器。因此,候选的起点一旦落在较早 JOC 容器的声明范围内,就只把
|
||||
它记作内嵌伪候选,不参与“多个容器”的判定。
|
||||
"""
|
||||
matches = []
|
||||
offsets = _marker_offsets(frame)
|
||||
parsed_candidates = []
|
||||
@@ -193,13 +199,32 @@ def find_joc_emdf(frame):
|
||||
"candidate_parse_errors": parse_errors,
|
||||
"repair_hint": "检查 EMDF 是否跨多个 audio-block skip field 分片,或 payload config 是否变化",
|
||||
})
|
||||
if len(matches) != 1:
|
||||
starts = [item.start_bit for item in matches]
|
||||
top_level_matches = []
|
||||
nested_matches = []
|
||||
for container in sorted(matches, key=lambda item: item.start_bit):
|
||||
parent = next((candidate for candidate in top_level_matches
|
||||
if candidate.start_bit < container.start_bit <
|
||||
candidate.start_bit + len(candidate.raw) * 8), None)
|
||||
if parent is None:
|
||||
top_level_matches.append(container)
|
||||
else:
|
||||
nested_matches.append({
|
||||
"start_bit": container.start_bit,
|
||||
"end_bit": container.start_bit + len(container.raw) * 8,
|
||||
"parent_start_bit": parent.start_bit,
|
||||
"parent_end_bit": parent.start_bit + len(parent.raw) * 8,
|
||||
})
|
||||
if len(top_level_matches) != 1:
|
||||
starts = [item.start_bit for item in top_level_matches]
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "multiple_joc_containers",
|
||||
"同步帧中存在多个可用 JOC EMDF,当前无法自动选择",
|
||||
details={"syncframe_bytes": len(frame), "joc_container_start_bits": starts})
|
||||
return matches[0]
|
||||
details={
|
||||
"syncframe_bytes": len(frame),
|
||||
"joc_container_start_bits": starts,
|
||||
"nested_joc_candidates": nested_matches,
|
||||
})
|
||||
return top_level_matches[0]
|
||||
|
||||
|
||||
def parse_container(data):
|
||||
|
||||
+265
-76
@@ -15,6 +15,10 @@ RAMP_DURATION_INDEX = (
|
||||
32, 64, 128, 256, 320, 480, 1000, 1001,
|
||||
1024, 1600, 1601, 1602, 1920, 2000, 2002, 2048,
|
||||
)
|
||||
OBJECT_ELEMENT_ID = 1
|
||||
TRIM_ELEMENT_ID = 2
|
||||
EXTENDED_OBJECT_ELEMENT_ID = 5
|
||||
POSITION_WINDOW_END_BIT = 112 + 31 * (15 - 3) + 24
|
||||
|
||||
|
||||
def q_of(k, n):
|
||||
@@ -24,36 +28,58 @@ def q_of(k, n):
|
||||
|
||||
def _payload_bits(bits_one):
|
||||
if isinstance(bits_one, (bytes, bytearray, memoryview)):
|
||||
src = np.frombuffer(bits_one, dtype=np.uint8)
|
||||
raw_payload = bytes(bits_one)
|
||||
bits = np.unpackbits(
|
||||
np.frombuffer(raw_payload, dtype=np.uint8), bitorder="big")
|
||||
else:
|
||||
src = np.asarray(bits_one, dtype=np.uint8)
|
||||
if src.ndim == 1 and src.shape[0] in (536, 552):
|
||||
bits = src
|
||||
elif src.size in (67, 69):
|
||||
bits = np.unpackbits(src.reshape(-1), bitorder="big")
|
||||
else:
|
||||
raw = (bytes(bits_one) if isinstance(bits_one, (bytes, bytearray, memoryview))
|
||||
else np.asarray(bits_one, dtype=np.uint8).tobytes())
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", f"payload_length_{len(raw)}B",
|
||||
f"发现未覆盖的 OAMD 载荷长度 {len(raw)}B",
|
||||
details={
|
||||
"supported_payload_bytes": [67, 69],
|
||||
"payload": bytes_descriptor(raw),
|
||||
"repair_hint": "检查 OAMD header、element 数量及可选字段造成的位偏移变化",
|
||||
})
|
||||
raw_payload = np.packbits(bits, bitorder="big").tobytes()
|
||||
src = np.asarray(bits_one)
|
||||
if src.ndim != 1:
|
||||
raw = np.asarray(bits_one, dtype=np.uint8).tobytes()
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "payload_shape",
|
||||
"OAMD 载荷必须是一维 byte 或 bit 序列",
|
||||
details={"shape": list(src.shape), "payload": bytes_descriptor(raw)})
|
||||
is_bit_vector = bool(src.size) and bool(np.all((src == 0) | (src == 1)))
|
||||
if is_bit_vector:
|
||||
if src.size % 8:
|
||||
raw = np.packbits(src.astype(np.uint8), bitorder="big").tobytes()
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "payload_bit_alignment",
|
||||
"OAMD bit 载荷没有按整字节对齐",
|
||||
details={
|
||||
"payload_bits": int(src.size),
|
||||
"payload": bytes_descriptor(raw),
|
||||
})
|
||||
bits = src.astype(np.uint8, copy=False)
|
||||
raw_payload = np.packbits(bits, bitorder="big").tobytes()
|
||||
else:
|
||||
try:
|
||||
values = src.astype(np.int64, copy=False)
|
||||
except (TypeError, ValueError, OverflowError) as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "payload_type",
|
||||
"OAMD 载荷不能转换为 byte 序列",
|
||||
details={"dtype": str(src.dtype), "parser_error": str(exc)}) from exc
|
||||
if np.any(values < 0) or np.any(values > 255):
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "payload_byte_range",
|
||||
"OAMD byte 载荷包含 0..255 之外的值",
|
||||
details={"dtype": str(src.dtype), "payload_values": int(src.size)})
|
||||
raw_payload = values.astype(np.uint8).tobytes()
|
||||
bits = np.unpackbits(
|
||||
np.frombuffer(raw_payload, dtype=np.uint8), bitorder="big")
|
||||
return bits, raw_payload
|
||||
|
||||
|
||||
class _BitReader:
|
||||
def __init__(self, bits):
|
||||
def __init__(self, bits, position=0, limit=None):
|
||||
self.bits = bits
|
||||
self.position = 0
|
||||
self.position = int(position)
|
||||
self.limit = len(bits) if limit is None else int(limit)
|
||||
|
||||
def read(self, count):
|
||||
end = self.position + count
|
||||
if end > len(self.bits):
|
||||
if end > self.limit:
|
||||
raise ValueError(f"OAMD 位流越界: bit={self.position}, need={count}")
|
||||
value = 0
|
||||
for bit in self.bits[self.position:end]:
|
||||
@@ -75,64 +101,216 @@ def _variable_bits(reader, width, max_groups=5):
|
||||
raise ValueError(f"OAMD variable_bits({width}) 延伸组过多")
|
||||
|
||||
|
||||
def _update_timing(bits, alternate_object_present, element_count, raw_payload):
|
||||
"""按 OAMD element/MDUpdateInfo 读取位置块的开始偏移和 ramp 时长。"""
|
||||
reader = _BitReader(bits)
|
||||
reader.position = 14
|
||||
for _ in range(element_count):
|
||||
element_index = reader.read(4)
|
||||
element_length = _variable_bits(reader, 4)
|
||||
element_end = reader.position + element_length + 1
|
||||
reader.skip(5 if alternate_object_present else 1)
|
||||
if element_index == 1:
|
||||
offset_code = reader.read(2)
|
||||
if offset_code == 0:
|
||||
sample_offset = 0
|
||||
elif offset_code == 1:
|
||||
sample_offset = SAMPLE_OFFSET_INDEX[reader.read(2)]
|
||||
elif offset_code == 2:
|
||||
sample_offset = reader.read(5)
|
||||
else:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "md_sample_offset_mode",
|
||||
"OAMD 使用了当前未覆盖的 MD sample-offset 模式",
|
||||
details={"payload": bytes_descriptor(raw_payload)})
|
||||
def _element_details(elements):
|
||||
return [{
|
||||
"ordinal": element["ordinal"],
|
||||
"element_id": element["element_id"],
|
||||
"size_bytes": element["size_bytes"],
|
||||
"header_start_bit": element["header_start_bit"],
|
||||
"body_start_bit": element["body_start_bit"],
|
||||
"body_end_bit": element["body_end_bit"],
|
||||
"discard_unknown": element["discard_unknown"],
|
||||
} for element in elements]
|
||||
|
||||
block_count = reader.read(3) + 1
|
||||
blocks = []
|
||||
for _block in range(block_count):
|
||||
block_offset_factor = reader.read(6)
|
||||
block_offset = sample_offset + block_offset_factor * 32
|
||||
ramp_code = reader.read(2)
|
||||
if ramp_code == 3:
|
||||
if reader.read(1):
|
||||
ramp_duration = RAMP_DURATION_INDEX[reader.read(4)]
|
||||
else:
|
||||
ramp_duration = reader.read(11)
|
||||
|
||||
def _parse_elements(bits, header, header_end, raw_payload):
|
||||
"""建立 OAMD element 目录;element size 表示其后 body 的字节数。"""
|
||||
reader = _BitReader(bits, header_end)
|
||||
elements = []
|
||||
for ordinal in range(header["element_count"]):
|
||||
header_start = reader.position
|
||||
try:
|
||||
element_id = reader.read(4)
|
||||
size_bytes = _variable_bits(reader, 4) + 1
|
||||
except ValueError as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "element_header_truncated",
|
||||
"OAMD element header 不完整",
|
||||
details={
|
||||
"element_ordinal": ordinal,
|
||||
"header_start_bit": header_start,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"parser_error": str(exc),
|
||||
}) from exc
|
||||
|
||||
body_start = reader.position
|
||||
body_end = body_start + size_bytes * 8
|
||||
if body_end > len(bits):
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "element_bounds",
|
||||
"OAMD element 声明长度超过 payload 边界",
|
||||
details={
|
||||
"element_ordinal": ordinal,
|
||||
"element_id": element_id,
|
||||
"size_bytes": size_bytes,
|
||||
"body_start_bit": body_start,
|
||||
"body_end_bit": body_end,
|
||||
"payload_bits": len(bits),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
|
||||
control_bits = 5 if header["alternate_object_present"] else 1
|
||||
if body_start + control_bits > body_end:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "element_control_bounds",
|
||||
"OAMD element 太短,无法容纳控制字段",
|
||||
details={
|
||||
"element_ordinal": ordinal,
|
||||
"element_id": element_id,
|
||||
"size_bytes": size_bytes,
|
||||
"control_bits": control_bits,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
control = _BitReader(bits, body_start, body_end)
|
||||
alternate_data_id = (control.read(4)
|
||||
if header["alternate_object_present"] else None)
|
||||
discard_unknown = bool(control.read(1))
|
||||
elements.append({
|
||||
"ordinal": ordinal,
|
||||
"element_id": element_id,
|
||||
"size_bytes": size_bytes,
|
||||
"header_start_bit": header_start,
|
||||
"body_start_bit": body_start,
|
||||
"data_start_bit": control.position,
|
||||
"body_end_bit": body_end,
|
||||
"alternate_data_id": alternate_data_id,
|
||||
"discard_unknown": discard_unknown,
|
||||
})
|
||||
reader.position = body_end
|
||||
|
||||
padding = bits[reader.position:]
|
||||
if len(padding) > 7:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "trailing_payload_data",
|
||||
"OAMD element 结束后仍有超过一个字节的未声明数据",
|
||||
details={
|
||||
"elements": _element_details(elements),
|
||||
"trailing_bits": len(padding),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
if np.any(padding):
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "nonzero_padding",
|
||||
"OAMD payload 尾部 padding 含非零位",
|
||||
details={
|
||||
"elements": _element_details(elements),
|
||||
"padding_start_bit": reader.position,
|
||||
"padding_bits": "".join(str(int(bit)) for bit in padding),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
|
||||
object_elements = [element for element in elements
|
||||
if element["element_id"] == OBJECT_ELEMENT_ID]
|
||||
if len(object_elements) != 1:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "object_element_count",
|
||||
"OAMD 必须包含且只能包含一个 object element",
|
||||
details={
|
||||
"object_element_count": len(object_elements),
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
object_element = object_elements[0]
|
||||
if object_element["ordinal"] != 0 or object_element["header_start_bit"] != header_end:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "object_element_order",
|
||||
"object element 不在当前固定位置窗口支持的首个 element 位置",
|
||||
details={
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "以 object element 的实际位置为基准重新定位对象窗口",
|
||||
})
|
||||
|
||||
for element in elements:
|
||||
element_id = element["element_id"]
|
||||
if element_id in (OBJECT_ELEMENT_ID, TRIM_ELEMENT_ID):
|
||||
continue
|
||||
if element_id == EXTENDED_OBJECT_ELEMENT_ID:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "extended_object_element",
|
||||
"OAMD 含可能改变坐标语义的 extended object element",
|
||||
details={
|
||||
"element": _element_details([element])[0],
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "解析 divergence/extended-precision position 后再应用轨迹",
|
||||
})
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", f"unsupported_element_{element_id}",
|
||||
f"OAMD 含当前未覆盖的 element id {element_id}",
|
||||
details={
|
||||
"element": _element_details([element])[0],
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
return object_element, elements
|
||||
|
||||
|
||||
def _update_timing(bits, object_element, raw_payload):
|
||||
"""从已定位的 object element 读取位置块开始偏移和 ramp 时长。"""
|
||||
reader = _BitReader(
|
||||
bits, object_element["data_start_bit"], object_element["body_end_bit"])
|
||||
try:
|
||||
offset_code = reader.read(2)
|
||||
if offset_code == 0:
|
||||
sample_offset = 0
|
||||
elif offset_code == 1:
|
||||
sample_offset = SAMPLE_OFFSET_INDEX[reader.read(2)]
|
||||
elif offset_code == 2:
|
||||
sample_offset = reader.read(5)
|
||||
else:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "md_sample_offset_mode",
|
||||
"OAMD 使用了当前未覆盖的 MD sample-offset 模式",
|
||||
details={"payload": bytes_descriptor(raw_payload)})
|
||||
|
||||
block_count = reader.read(3) + 1
|
||||
blocks = []
|
||||
for _block in range(block_count):
|
||||
block_offset_factor = reader.read(6)
|
||||
block_offset = sample_offset + block_offset_factor * 32
|
||||
ramp_code = reader.read(2)
|
||||
if ramp_code == 3:
|
||||
if reader.read(1):
|
||||
ramp_duration = RAMP_DURATION_INDEX[reader.read(4)]
|
||||
else:
|
||||
ramp_duration = RAMP_DURATIONS[ramp_code]
|
||||
blocks.append((block_offset, ramp_duration))
|
||||
if len(blocks) != 1:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "multiple_position_blocks",
|
||||
"OAMD 一帧含多个对象位置更新块,固定位置窗口不能安全套用",
|
||||
details={
|
||||
"block_count": len(blocks),
|
||||
"blocks": blocks,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "按 ObjectInfoBlock 顺序逐块解析坐标,再生成分段 ADM ramp",
|
||||
})
|
||||
return blocks[0]
|
||||
reader.position = element_end
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "missing_object_element",
|
||||
"OAMD 中没有 object element (element_index=1)",
|
||||
details={"payload": bytes_descriptor(raw_payload)})
|
||||
ramp_duration = reader.read(11)
|
||||
else:
|
||||
ramp_duration = RAMP_DURATIONS[ramp_code]
|
||||
blocks.append((block_offset, ramp_duration))
|
||||
except UnsupportedVariantError:
|
||||
raise
|
||||
except ValueError as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "object_element_syntax",
|
||||
"OAMD object element 的 timing 字段越界或不完整",
|
||||
details={
|
||||
"element": _element_details([object_element])[0],
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"parser_error": str(exc),
|
||||
}) from exc
|
||||
|
||||
if len(blocks) != 1:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "multiple_position_blocks",
|
||||
"OAMD 一帧含多个对象位置更新块,固定位置窗口不能安全套用",
|
||||
details={
|
||||
"block_count": len(blocks),
|
||||
"blocks": blocks,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "按 ObjectInfoBlock 顺序逐块解析坐标,再生成分段 ADM ramp",
|
||||
})
|
||||
return blocks[0]
|
||||
|
||||
|
||||
def frame_update(bits_one):
|
||||
"""单帧 OAMD → 位置字段增量及其 sample offset/ramp duration。"""
|
||||
bits, raw_payload = _payload_bits(bits_one)
|
||||
if len(bits) < 14:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "header_truncated",
|
||||
"OAMD payload 不足以容纳受支持的 header",
|
||||
details={"payload_bits": len(bits), "payload": bytes_descriptor(raw_payload)})
|
||||
header = {
|
||||
"version": int((bits[0] << 1) | bits[1]),
|
||||
"objects_minus_one": int(sum(int(bits[2 + i]) << (4 - i) for i in range(5))),
|
||||
@@ -144,15 +322,25 @@ def frame_update(bits_one):
|
||||
if raw_payload[:2] != b"\x1f\x88":
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", f"header_signature_{raw_payload[:2].hex()}",
|
||||
"OAMD 长度已知,但 header 与当前位置字段布局不一致",
|
||||
"OAMD header 与当前位置字段布局不一致",
|
||||
details={
|
||||
"supported_header_prefix_hex": "1f88",
|
||||
"header_probe": header,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "按新 header 的 program assignment 和 element 布局重新定位对象位置字段",
|
||||
})
|
||||
block_offset, ramp_duration = _update_timing(
|
||||
bits, bool(header["alternate_object_present"]), header["element_count"], raw_payload)
|
||||
object_element, elements = _parse_elements(bits, header, 14, raw_payload)
|
||||
if object_element["body_end_bit"] < POSITION_WINDOW_END_BIT:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "object_element_too_short",
|
||||
"OAMD object element 无法容纳当前固定位置窗口",
|
||||
details={
|
||||
"element": _element_details([object_element])[0],
|
||||
"required_position_end_bit": POSITION_WINDOW_END_BIT,
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
block_offset, ramp_duration = _update_timing(bits, object_element, raw_payload)
|
||||
weights = 1 << np.arange(7, -1, -1)
|
||||
out = {}
|
||||
for obj in range(16):
|
||||
@@ -163,7 +351,7 @@ def frame_update(bits_one):
|
||||
if obj and (wq1 >> 6 != 3 or wq2 & 2 != 2 or wq3 & 0x1F != 1):
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "position_layout_signature",
|
||||
"OAMD 长度和 header 已知,但对象位置字段标记或位偏移发生变化",
|
||||
"OAMD 对象位置字段标记或位偏移发生变化",
|
||||
details={
|
||||
"object_slot": obj,
|
||||
"position_start_bit": start,
|
||||
@@ -171,6 +359,7 @@ def frame_update(bits_one):
|
||||
"q2_window_hex": f"{wq2:02x}",
|
||||
"q3_window_hex": f"{wq3:02x}",
|
||||
"header_probe": header,
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "解析 OAMD element 可选字段并更新每个对象的位置窗口偏移",
|
||||
})
|
||||
|
||||
+2
-2
@@ -179,7 +179,7 @@ def _expand_events(events, total_samples, rate, update_quantum_samples,
|
||||
raise ValueError(f"未知 trajectory_mode: {trajectory_mode}")
|
||||
|
||||
def build_adm_tracks(index, frames=None, rate=48000, frame_samples=1536,
|
||||
update_quantum_samples=64, object_delay_samples=640,
|
||||
update_quantum_samples=64, object_delay_samples=1473,
|
||||
trajectory_mode="compact"):
|
||||
"""从统一 metadata index 构造 15 条 ADM 轨迹。
|
||||
|
||||
@@ -187,7 +187,7 @@ def build_adm_tracks(index, frames=None, rate=48000, frame_samples=1536,
|
||||
OAMD 的内外层 sample offset、block offset 和 ramp 均保留。
|
||||
``trajectory_mode="compact"`` 用一个长 ADM interpolation block 表示每条
|
||||
线性 ramp;``dense64`` 保留逐 64-sample 展开作为兼容回退。
|
||||
``object_delay_samples`` 将位置更新与对象逆 QMF 的输出时刻对齐。
|
||||
``object_delay_samples`` 将位置更新与对象 PCM 的 decoder 输出时刻对齐。
|
||||
slot1..15 与对象 PCM ch1..15 一一对应。
|
||||
"""
|
||||
frames = index.rows if frames is None else frames
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
"""Stateful float64/complex128 Rosella hybrid-band renderer."""
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
|
||||
from rosella_direct import (
|
||||
PROFILE_MID,
|
||||
direct_and_room_send,
|
||||
special_lfe_direct,
|
||||
)
|
||||
from rosella_model import RosellaModel
|
||||
from rosella_room import RosellaRoomFir
|
||||
|
||||
|
||||
class RosellaRenderer:
|
||||
"""Hold per-source direct parameters and the cross-block room state."""
|
||||
|
||||
def __init__(self, model: RosellaModel, source_count: int,
|
||||
room_impulse_slots: int = 4096, *, create_room: bool = True):
|
||||
if source_count <= 0:
|
||||
raise ValueError("source_count must be positive")
|
||||
self.model = model
|
||||
self.source_count = int(source_count)
|
||||
self.room = (RosellaRoomFir(model, impulse_slots=room_impulse_slots)
|
||||
if create_room else None)
|
||||
self.positions = np.zeros((self.source_count, 3), dtype=np.float64)
|
||||
self.positions[:, 1] = 1.0
|
||||
self.profiles = np.full(self.source_count, PROFILE_MID, dtype=np.int32)
|
||||
self.special_lfe = np.zeros(self.source_count, dtype=bool)
|
||||
self.gains = np.empty(
|
||||
(self.source_count, 2, 77), dtype=np.complex128)
|
||||
self.room_sends = np.empty(self.source_count, dtype=np.float64)
|
||||
self._parameter_keys = [None] * self.source_count
|
||||
for source in range(self.source_count):
|
||||
self.set_source(source, self.positions[source], PROFILE_MID)
|
||||
|
||||
def reset(self):
|
||||
if self.room is not None:
|
||||
self.room.reset()
|
||||
|
||||
def set_source(self, source: int, position, profile: int = PROFILE_MID,
|
||||
*, special_lfe: bool = False):
|
||||
source = int(source)
|
||||
if not 0 <= source < self.source_count:
|
||||
raise IndexError(source)
|
||||
coordinates = np.asarray(position, dtype=np.float64)
|
||||
if coordinates.shape != (3,) or not np.all(np.isfinite(coordinates)):
|
||||
raise ValueError(f"source position must be three finite values, got {position!r}")
|
||||
effective_profile = 0 if special_lfe else int(profile)
|
||||
key = ((bool(special_lfe), effective_profile)
|
||||
+ tuple(float(value) for value in coordinates))
|
||||
if self._parameter_keys[source] == key:
|
||||
return
|
||||
self.positions[source] = coordinates
|
||||
self.profiles[source] = effective_profile
|
||||
self.special_lfe[source] = bool(special_lfe)
|
||||
parameters = (special_lfe_direct() if special_lfe else
|
||||
direct_and_room_send(self.model, coordinates, effective_profile))
|
||||
self.gains[source] = parameters.gains
|
||||
self.room_sends[source] = parameters.room_send
|
||||
self._parameter_keys[source] = key
|
||||
|
||||
def direct_and_send_static(self, sources):
|
||||
"""Mix one static-parameter slot chunk without advancing room state."""
|
||||
values = np.asarray(sources, dtype=np.complex128)
|
||||
if values.ndim != 3 or values.shape[1:] != (self.source_count, 77):
|
||||
raise ValueError(
|
||||
f"expected [slots,{self.source_count},77], got {values.shape}")
|
||||
direct = np.zeros((values.shape[0], 2, 77), dtype=np.complex128)
|
||||
room_send = np.zeros((values.shape[0], 77), dtype=np.complex128)
|
||||
for source in range(self.source_count - 1, -1, -1):
|
||||
direct += values[:, source, None, :] * self.gains[source][None, :, :]
|
||||
room_send += values[:, source, :] * self.room_sends[source]
|
||||
return direct, room_send
|
||||
|
||||
def process_static_chunk(self, sources) -> np.ndarray:
|
||||
direct, room_send = self.direct_and_send_static(sources)
|
||||
if self.room is None:
|
||||
raise RuntimeError("room renderer is not configured")
|
||||
direct += self.room.process_chunk(room_send)
|
||||
return direct
|
||||
@@ -0,0 +1,302 @@
|
||||
"""Float64 Rosella direction, distance, HRTF, and room-send calculations."""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
|
||||
import numpy as np
|
||||
|
||||
from rosella_model import RosellaModel, direction_basis
|
||||
|
||||
PROFILE_NEAR = 1
|
||||
PROFILE_FAR = 2
|
||||
PROFILE_MID = 3
|
||||
BINAURAL_PROFILE_NAMES = {
|
||||
"near": PROFILE_NEAR,
|
||||
"far": PROFILE_FAR,
|
||||
"mid": PROFILE_MID,
|
||||
}
|
||||
|
||||
_SPECIAL_LFE_LOW_16 = np.asarray([
|
||||
0x402695EA, 0x3FE75979, 0x3F28CAAA, 0xBCE1FB2E,
|
||||
0xBDD8AF65, 0xBD8F426E, 0x3D996821, 0xBC16B3A0,
|
||||
0x3B64BAF1, 0xBC81ECFD, 0xBA3D892F, 0x3AF6A9F0,
|
||||
0xB9DD1C5F, 0x380A193F, 0x38052059, 0x351BCB34,
|
||||
], dtype=np.uint32).view(np.float32).astype(np.float64)
|
||||
_CENTRE_EQUAL = 0.9998489618301392
|
||||
_CENTRE_ALTERNATE = 0.7070000171661377
|
||||
|
||||
_FIELD_CACHE: dict[int, tuple[np.ndarray, np.ndarray]] = {}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DirectResult:
|
||||
gains: np.ndarray # complex128 [ear=2, hybrid_band=77]
|
||||
room_send: np.float64
|
||||
physical_radius_m: np.float64
|
||||
normalized_radius: np.float64
|
||||
clamped_radius: np.float64
|
||||
delay_samples: np.float64
|
||||
delayed_ear: int | None
|
||||
|
||||
|
||||
def special_lfe_direct() -> DirectResult:
|
||||
"""Return the fixed 16-band low-pass used by a special/LFE source."""
|
||||
mono = np.zeros(77, dtype=np.complex128)
|
||||
mono[:16] = _SPECIAL_LFE_LOW_16
|
||||
return DirectResult(
|
||||
gains=np.repeat(mono[None, :], 2, axis=0),
|
||||
room_send=np.float64(0.0),
|
||||
physical_radius_m=np.float64(0.0),
|
||||
normalized_radius=np.float64(0.0),
|
||||
clamped_radius=np.float64(0.0),
|
||||
delay_samples=np.float64(0.0),
|
||||
delayed_ear=None,
|
||||
)
|
||||
|
||||
|
||||
def _round_away_from_zero(value: float) -> int:
|
||||
return math.floor(value + 0.5) if value >= 0.0 else math.ceil(value - 0.5)
|
||||
|
||||
|
||||
def _q15_position(position) -> np.ndarray:
|
||||
"""Quantize ADM Cartesian coordinates to the Rosella metadata grid.
|
||||
|
||||
Quantization is metadata decoding. The returned integer lanes are promoted
|
||||
to float64 before any geometry is evaluated.
|
||||
"""
|
||||
x, y, z = (float(value) for value in position)
|
||||
encoded = (
|
||||
min(max((x + 1.0) * 0.5, 0.0), 1.0),
|
||||
min(max((1.0 - y) * 0.5, 0.0), 1.0),
|
||||
min(max(z, -1.0), 1.0),
|
||||
)
|
||||
return np.asarray([
|
||||
min(_round_away_from_zero(value * 32768.0), 32767)
|
||||
for value in encoded
|
||||
], dtype=np.int32)
|
||||
|
||||
|
||||
def _profile_geometry(model: RosellaModel, position, profile_index: int):
|
||||
if profile_index not in (PROFILE_NEAR, PROFILE_FAR, PROFILE_MID):
|
||||
raise ValueError("binaural object profile must be near, mid, or far")
|
||||
profile = model.profiles[profile_index]
|
||||
encoded = _q15_position(position)
|
||||
q_front = 1.0 - 2.0 * float(encoded[1]) / 32768.0
|
||||
q_x = 2.0 * float(encoded[0]) / 32768.0 - 1.0
|
||||
q_vertical = float(encoded[2]) / 32768.0
|
||||
|
||||
if int(model.header_integer_fields[0]) != 0:
|
||||
if q_x == 0.0 and q_front == 0.0:
|
||||
mapped_front = 0.0
|
||||
mapped_lateral = 0.0
|
||||
mapped_vertical = q_vertical
|
||||
else:
|
||||
horizontal_max = max(abs(q_x), abs(q_front))
|
||||
horizontal_norm = ((q_x / horizontal_max) ** 2
|
||||
+ (q_front / horizontal_max) ** 2)
|
||||
if q_vertical == 0.0:
|
||||
vertical_norm = 1.0
|
||||
else:
|
||||
smaller = min(abs(q_vertical), horizontal_max)
|
||||
larger = max(abs(q_vertical), horizontal_max)
|
||||
vertical_norm = 1.0 + (smaller / larger) ** 2
|
||||
horizontal_factor = 1.0 / math.sqrt(horizontal_norm * vertical_norm)
|
||||
vertical_factor = 1.0 / math.sqrt(vertical_norm)
|
||||
mapped_front = q_front * horizontal_factor
|
||||
mapped_lateral = -q_x * horizontal_factor
|
||||
mapped_vertical = q_vertical * vertical_factor
|
||||
else:
|
||||
mapped_front = q_front
|
||||
mapped_lateral = -q_x
|
||||
mapped_vertical = q_vertical
|
||||
|
||||
scales = np.asarray(profile.axis_scales_internal, dtype=np.float64)
|
||||
scaled = np.asarray([
|
||||
mapped_front * scales[2],
|
||||
mapped_lateral * scales[0],
|
||||
mapped_vertical * scales[1],
|
||||
], dtype=np.float64)
|
||||
bounds = np.asarray(profile.bounds, dtype=np.float64)
|
||||
ray = 1.0
|
||||
for axis in range(3):
|
||||
value = scaled[axis]
|
||||
lower, upper = bounds[axis * 2:axis * 2 + 2]
|
||||
if value < lower:
|
||||
ray = min(ray, lower / value)
|
||||
elif value > upper:
|
||||
ray = min(ray, upper / value)
|
||||
if ray < 1.0:
|
||||
scaled *= ray
|
||||
|
||||
radius = float(np.linalg.norm(scaled))
|
||||
clamped = max(radius, float(profile.minimum_normalized_radius))
|
||||
alpha = radius / clamped
|
||||
direction = (scaled / radius if radius > 1.0e-30
|
||||
else np.asarray([1.0, 0.0, 0.0], dtype=np.float64))
|
||||
return profile, direction, radius, clamped, alpha
|
||||
|
||||
|
||||
def _logical_field(padded: np.ndarray) -> np.ndarray:
|
||||
result = np.empty((77, 36, 2), dtype=np.float64)
|
||||
source = np.asarray(padded, dtype=np.float64)
|
||||
for band in range(77):
|
||||
block, lane = divmod(band, 4)
|
||||
for term in range(36):
|
||||
for component in range(2):
|
||||
result[band, term, component] = source[
|
||||
lane + 4 * (term * 2 + component + 72 * block)]
|
||||
return result
|
||||
|
||||
|
||||
def _model_fields(model: RosellaModel) -> tuple[np.ndarray, np.ndarray]:
|
||||
key = id(model)
|
||||
fields = _FIELD_CACHE.get(key)
|
||||
if fields is None:
|
||||
fields = (_logical_field(model.field_left_padded),
|
||||
_logical_field(model.field_right_padded))
|
||||
_FIELD_CACHE[key] = fields
|
||||
return fields
|
||||
|
||||
|
||||
def _ear_geometry(model: RosellaModel, profile, direction, clamped: float,
|
||||
offset: float, correction: float):
|
||||
x, y, z = (float(value) for value in direction)
|
||||
inverse_distance = float(profile.inverse_distance_per_m)
|
||||
ear = float(offset) * inverse_distance / clamped
|
||||
y_minus = y - ear
|
||||
y_plus = y + ear
|
||||
common = x * x + z * z
|
||||
length_minus = math.sqrt(y_minus * y_minus + common)
|
||||
length_plus = math.sqrt(y_plus * y_plus + common)
|
||||
basis_minus = direction_basis(
|
||||
x / length_minus, y_minus / length_minus, z / length_minus,
|
||||
dtype=np.float64)
|
||||
basis_plus = direction_basis(
|
||||
x / length_plus, y_plus / length_plus, z / length_plus,
|
||||
dtype=np.float64)
|
||||
path_minus = length_minus * clamped
|
||||
path_plus = length_plus * clamped
|
||||
if correction != 0.0:
|
||||
multiplier = 2.0 * float(correction) * inverse_distance
|
||||
path_minus += max(float(np.dot(
|
||||
np.asarray(model.vector_left, dtype=np.float64), basis_minus)), 0.0) * multiplier
|
||||
path_plus += max(float(np.dot(
|
||||
np.asarray(model.vector_right, dtype=np.float64), basis_plus)), 0.0) * multiplier
|
||||
return basis_minus, basis_plus, path_minus, path_plus
|
||||
|
||||
|
||||
def _phase_groups(model: RosellaModel, delay_samples: float) -> np.ndarray:
|
||||
result = np.ones(77, dtype=np.complex128)
|
||||
current = 1.0 + 0.0j
|
||||
step = 1.0 + 0.0j
|
||||
value_index = 0
|
||||
for band, flag in enumerate(model.hybrid_flags):
|
||||
if flag != 2:
|
||||
if flag == 1:
|
||||
angle = float(model.hybrid_values[value_index]) * delay_samples
|
||||
value_index += 1
|
||||
step = complex(math.cos(angle), math.sin(angle))
|
||||
current *= step
|
||||
result[band] = current
|
||||
return result
|
||||
|
||||
|
||||
def direct_and_room_send(model: RosellaModel, position,
|
||||
profile_index: int) -> DirectResult:
|
||||
"""Evaluate one ordinary source using float64/complex128 throughout."""
|
||||
profile, direction, radius, clamped, alpha = _profile_geometry(
|
||||
model, position, profile_index)
|
||||
|
||||
_, _, path_minus, path_plus = _ear_geometry(
|
||||
model, profile, direction, clamped,
|
||||
float(model.model_scalars[1]), float(model.model_scalars[2]))
|
||||
delay = (abs(path_plus - path_minus)
|
||||
* float(profile.distance_scale_m)
|
||||
* (float(model.sample_rate) / 343.3) * alpha)
|
||||
delayed_ear = 0 if path_minus > path_plus else (
|
||||
1 if path_plus > path_minus else None)
|
||||
|
||||
_, _, weight_minus_path, weight_plus_path = _ear_geometry(
|
||||
model, profile, direction, clamped,
|
||||
float(model.model_scalars[3]), float(model.model_scalars[4]))
|
||||
weight_norm = math.sqrt(
|
||||
weight_minus_path * weight_minus_path
|
||||
+ weight_plus_path * weight_plus_path)
|
||||
weight_left = weight_plus_path / weight_norm
|
||||
weight_right = weight_minus_path / weight_norm
|
||||
|
||||
final_offset = float(model.model_scalars[0])
|
||||
if final_offset == 0.0:
|
||||
basis_minus = direction_basis(*direction, dtype=np.float64)
|
||||
basis_plus = basis_minus.copy()
|
||||
else:
|
||||
x, y, z = (float(value) for value in direction)
|
||||
ear = final_offset * float(profile.inverse_distance_per_m) / clamped
|
||||
y_minus = y - ear
|
||||
y_plus = y + ear
|
||||
common_length = x * x + z * z
|
||||
length_minus = math.sqrt(y_minus * y_minus + common_length)
|
||||
length_plus = math.sqrt(y_plus * y_plus + common_length)
|
||||
basis_minus = direction_basis(
|
||||
x / length_minus, y_minus / length_minus, z / length_minus,
|
||||
dtype=np.float64)
|
||||
basis_plus = direction_basis(
|
||||
x / length_plus, y_plus / length_plus, z / length_plus,
|
||||
dtype=np.float64)
|
||||
|
||||
field_left, field_right = _model_fields(model)
|
||||
left_components = np.einsum(
|
||||
"bjc,j->bc", field_left, basis_minus,
|
||||
dtype=np.float64, optimize=False)
|
||||
right_components = np.einsum(
|
||||
"bjc,j->bc", field_right, basis_plus,
|
||||
dtype=np.float64, optimize=False)
|
||||
left = left_components[:, 0] + 1j * left_components[:, 1]
|
||||
right = right_components[:, 0] + 1j * right_components[:, 1]
|
||||
if delayed_ear is not None:
|
||||
phase = _phase_groups(model, delay)
|
||||
if delayed_ear == 0:
|
||||
left *= phase
|
||||
else:
|
||||
right *= phase
|
||||
|
||||
effective_radius = (radius * float(model.header_float_scalars[0])
|
||||
* float(profile.distance_scale_m))
|
||||
if profile_index in (PROFILE_FAR, PROFILE_MID):
|
||||
common = 1.0 / math.sqrt(
|
||||
1.0 + float(model.header_float_scalars[1])
|
||||
* effective_radius * effective_radius)
|
||||
room_send = effective_radius * common
|
||||
else:
|
||||
common = 1.0
|
||||
room_send = 0.0
|
||||
|
||||
left_term0 = field_left[:, 0, 0] + 1j * field_left[:, 0, 1]
|
||||
right_term0 = field_right[:, 0, 0] + 1j * field_right[:, 0, 1]
|
||||
weights_are_default_equal = (
|
||||
float(model.model_scalars[3]) == 0.0
|
||||
and float(model.model_scalars[4]) == 0.0)
|
||||
if weights_are_default_equal:
|
||||
centre_left = weight_left * (1.0 - alpha) * _CENTRE_EQUAL
|
||||
centre_right = centre_left
|
||||
right_direction_weight = weight_left
|
||||
else:
|
||||
centre_left = (1.0 - alpha) * _CENTRE_ALTERNATE
|
||||
centre_right = centre_left
|
||||
right_direction_weight = weight_right
|
||||
|
||||
gains = np.empty((2, 77), dtype=np.complex128)
|
||||
gains[0] = common * (
|
||||
left * (weight_left * alpha) + left_term0 * centre_left)
|
||||
gains[1] = common * (
|
||||
right * (right_direction_weight * alpha) + right_term0 * centre_right)
|
||||
return DirectResult(
|
||||
gains=gains,
|
||||
room_send=np.float64(room_send),
|
||||
physical_radius_m=np.float64(float(profile.distance_scale_m) * radius),
|
||||
normalized_radius=np.float64(radius),
|
||||
clamped_radius=np.float64(clamped),
|
||||
delay_samples=np.float64(delay),
|
||||
delayed_ear=delayed_ear,
|
||||
)
|
||||
@@ -0,0 +1,190 @@
|
||||
"""Float64/complex128 Rosella QMF and hybrid filterbanks."""
|
||||
from __future__ import annotations
|
||||
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
PROJECT_DIR = Path(__file__).resolve().parent.parent
|
||||
DEFAULT_KERNEL_DATA = PROJECT_DIR / "data" / "rosella_kernels.npz"
|
||||
|
||||
|
||||
@lru_cache(maxsize=4)
|
||||
def _load_tables(path_string: str) -> dict[str, np.ndarray]:
|
||||
path = Path(path_string)
|
||||
if not path.is_file():
|
||||
raise FileNotFoundError(f"Rosella kernel data not found: {path}")
|
||||
with np.load(path, allow_pickle=False) as archive:
|
||||
version = archive["format_version"]
|
||||
if version.shape != (1,) or int(version[0]) != 1:
|
||||
raise ValueError(f"unsupported Rosella kernel data version in {path}")
|
||||
return {name: archive[name].copy() for name in archive.files}
|
||||
|
||||
|
||||
def load_kernel_tables(path: str | Path = DEFAULT_KERNEL_DATA) -> dict[str, np.ndarray]:
|
||||
"""Load and cache the compact, production Rosella kernel tables."""
|
||||
return _load_tables(str(Path(path).expanduser().resolve()))
|
||||
|
||||
|
||||
class QmfAnalysis:
|
||||
"""Batchable 64-band analysis with float64 state and complex128 FFTs."""
|
||||
|
||||
def __init__(self, channels: int, kernel_data: str | Path = DEFAULT_KERNEL_DATA):
|
||||
if channels <= 0:
|
||||
raise ValueError("channels must be positive")
|
||||
tables = load_kernel_tables(kernel_data)
|
||||
self.coefficients = np.asarray(
|
||||
tables["qmf_analysis_coefficients"], dtype=np.float64)
|
||||
if self.coefficients.shape != (64, 10):
|
||||
raise ValueError("invalid qmf_analysis_coefficients shape")
|
||||
self.channels = int(channels)
|
||||
self.history = np.zeros((9, self.channels, 64), dtype=np.float64)
|
||||
phase = np.arange(64, dtype=np.float64)
|
||||
self.premod = np.exp(-1j * np.pi * phase / 128.0).astype(np.complex128)
|
||||
self.post = np.exp(
|
||||
-1j * 3.0 * (np.arange(64, dtype=np.float64) + 0.5) * np.pi / 128.0
|
||||
).astype(np.complex128)
|
||||
self.even_post = (
|
||||
1j * ((-1.0) ** np.arange(64, dtype=np.float64))
|
||||
).astype(np.complex128)
|
||||
|
||||
def reset(self):
|
||||
self.history.fill(0.0)
|
||||
|
||||
def process_chunk(self, hops) -> np.ndarray:
|
||||
values = np.asarray(hops, dtype=np.float64)
|
||||
if values.ndim != 3 or values.shape[1:] != (self.channels, 64):
|
||||
raise ValueError(f"expected [slots,{self.channels},64], got {values.shape}")
|
||||
count = values.shape[0]
|
||||
joined = np.concatenate((self.history, values), axis=0)
|
||||
even = np.zeros_like(values)
|
||||
odd = np.zeros_like(values)
|
||||
for lag in range(10):
|
||||
source = joined[9 - lag:9 - lag + count]
|
||||
target = even if lag % 2 == 0 else odd
|
||||
target += source * self.coefficients[:, lag][None, None, :]
|
||||
self.history[:] = joined[-9:]
|
||||
|
||||
def transform(block):
|
||||
prepared = block.astype(np.complex128, copy=False) * self.premod
|
||||
transformed = np.fft.fft(prepared, n=128, axis=-1)[..., :64]
|
||||
return transformed * self.post
|
||||
|
||||
return np.asarray(transform(odd) + transform(even) * self.even_post,
|
||||
dtype=np.complex128)
|
||||
|
||||
|
||||
class HybridAnalysis:
|
||||
"""Sparse 64-QMF to 77-hybrid analysis in float64/complex128."""
|
||||
|
||||
def __init__(self, channels: int, kernel_data: str | Path = DEFAULT_KERNEL_DATA):
|
||||
if channels <= 0:
|
||||
raise ValueError("channels must be positive")
|
||||
tables = load_kernel_tables(kernel_data)
|
||||
self.low_kernel = np.asarray(
|
||||
tables["hybrid_analysis_low_kernel"], dtype=np.float64)
|
||||
if self.low_kernel.shape != (3, 2, 13, 16, 2):
|
||||
raise ValueError("invalid hybrid_analysis_low_kernel shape")
|
||||
self.channels = int(channels)
|
||||
self.history = np.zeros((12, self.channels, 3, 2), dtype=np.float64)
|
||||
self.high_history = np.zeros(
|
||||
(6, self.channels, 61), dtype=np.complex128)
|
||||
|
||||
def reset(self):
|
||||
self.history.fill(0.0)
|
||||
self.high_history.fill(0.0)
|
||||
|
||||
def process_chunk(self, qmf) -> np.ndarray:
|
||||
values = np.asarray(qmf, dtype=np.complex128)
|
||||
if values.ndim != 3 or values.shape[1:] != (self.channels, 64):
|
||||
raise ValueError(f"expected [slots,{self.channels},64], got {values.shape}")
|
||||
count = values.shape[0]
|
||||
low = np.stack((values[:, :, :3].real, values[:, :, :3].imag), axis=-1)
|
||||
joined = np.concatenate((self.history, low), axis=0)
|
||||
output = np.zeros((count, self.channels, 77, 2), dtype=np.float64)
|
||||
for lag in range(13):
|
||||
source = joined[12 - lag:12 - lag + count]
|
||||
output[:, :, :16] += np.einsum(
|
||||
"tcpi,pibo->tcbo", source, self.low_kernel[:, :, lag],
|
||||
dtype=np.float64, optimize=False)
|
||||
self.history[:] = joined[-12:]
|
||||
|
||||
high_joined = np.concatenate((self.high_history, values[:, :, 3:]), axis=0)
|
||||
high = high_joined[:count]
|
||||
output[:, :, 16:, 0] = high.real
|
||||
output[:, :, 16:, 1] = high.imag
|
||||
self.high_history[:] = high_joined[-6:]
|
||||
return np.asarray(output[..., 0] + 1j * output[..., 1], dtype=np.complex128)
|
||||
|
||||
|
||||
class HybridSynthesis:
|
||||
"""Instantaneous sparse 77-hybrid to 64-QMF synthesis map."""
|
||||
|
||||
def __init__(self, channels: int, kernel_data: str | Path = DEFAULT_KERNEL_DATA):
|
||||
if channels <= 0:
|
||||
raise ValueError("channels must be positive")
|
||||
tables = load_kernel_tables(kernel_data)
|
||||
indices = np.asarray(tables["hybrid_synthesis_indices"], dtype=np.int64)
|
||||
values = np.asarray(tables["hybrid_synthesis_values"], dtype=np.float64)
|
||||
if indices.ndim != 2 or indices.shape[1] != 4 or len(indices) != len(values):
|
||||
raise ValueError("invalid hybrid synthesis sparse table")
|
||||
self.mapping = [
|
||||
(int(index[0]), int(index[1]), int(index[2]), int(index[3]), float(value))
|
||||
for index, value in zip(indices, values)
|
||||
]
|
||||
self.channels = int(channels)
|
||||
|
||||
def reset(self):
|
||||
return None
|
||||
|
||||
def process_chunk(self, hybrid) -> np.ndarray:
|
||||
values = np.asarray(hybrid, dtype=np.complex128)
|
||||
if values.ndim != 3 or values.shape[1:] != (self.channels, 77):
|
||||
raise ValueError(f"expected [slots,{self.channels},77], got {values.shape}")
|
||||
source = np.stack((values.real, values.imag), axis=-1)
|
||||
output = np.zeros((values.shape[0], self.channels, 64, 2), dtype=np.float64)
|
||||
for input_band, input_component, output_band, output_component, gain in self.mapping:
|
||||
output[:, :, output_band, output_component] += (
|
||||
source[:, :, input_band, input_component] * gain)
|
||||
return np.asarray(output[..., 0] + 1j * output[..., 1], dtype=np.complex128)
|
||||
|
||||
|
||||
class QmfSynthesis:
|
||||
"""Rank-4 64-band synthesis with float64 state and accumulation."""
|
||||
|
||||
def __init__(self, channels: int, kernel_data: str | Path = DEFAULT_KERNEL_DATA):
|
||||
if channels <= 0:
|
||||
raise ValueError("channels must be positive")
|
||||
tables = load_kernel_tables(kernel_data)
|
||||
self.basis = np.asarray(tables["qmf_synthesis_basis"], dtype=np.float64)
|
||||
self.taps = np.asarray(tables["qmf_synthesis_taps"], dtype=np.float64)
|
||||
if self.basis.shape != (64, 4, 128) or self.taps.shape != (64, 10, 4):
|
||||
raise ValueError("invalid QMF synthesis factorization")
|
||||
self.channels = int(channels)
|
||||
self.rank = 4
|
||||
self.history = np.zeros(
|
||||
(9, self.channels, 64, self.rank), dtype=np.float64)
|
||||
|
||||
def reset(self):
|
||||
self.history.fill(0.0)
|
||||
|
||||
def process_chunk(self, qmf) -> np.ndarray:
|
||||
values = np.asarray(qmf, dtype=np.complex128)
|
||||
if values.ndim != 3 or values.shape[1:] != (self.channels, 64):
|
||||
raise ValueError(f"expected [slots,{self.channels},64], got {values.shape}")
|
||||
count = values.shape[0]
|
||||
flat = np.stack((values.real, values.imag), axis=-1).reshape(
|
||||
count * self.channels, 128)
|
||||
modulation = self.basis.reshape(64 * self.rank, 128)
|
||||
features = (flat @ modulation.T).reshape(
|
||||
count, self.channels, 64, self.rank)
|
||||
joined = np.concatenate((self.history, features), axis=0)
|
||||
output = np.zeros((count, self.channels, 64), dtype=np.float64)
|
||||
for lag in range(10):
|
||||
output += np.sum(
|
||||
joined[9 - lag:9 - lag + count]
|
||||
* self.taps[:, lag, :][None, None, :, :],
|
||||
axis=-1, dtype=np.float64)
|
||||
self.history[:] = joined[-9:]
|
||||
return output
|
||||
@@ -0,0 +1,517 @@
|
||||
"""Parser for ``.personalized_headphone`` and raw ``rp`` models."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
from numbers import Real
|
||||
import struct
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
Q15 = np.float32(1.0 / 32768.0)
|
||||
|
||||
|
||||
def _f32(value) -> np.float32:
|
||||
return np.float32(value)
|
||||
|
||||
|
||||
def _q15(value: int) -> np.float32:
|
||||
return _f32(_f32(value) * Q15)
|
||||
|
||||
|
||||
def _q15_exp(value: int, exponent: int) -> np.float32:
|
||||
return _f32(_q15(value) * _f32(np.ldexp(1.0, exponent)))
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DistanceProfile:
|
||||
bounds: np.ndarray
|
||||
distance_scale_m: np.float32
|
||||
inverse_distance_per_m: np.float32
|
||||
axis_scales_internal: np.ndarray
|
||||
minimum_normalized_radius: np.float32
|
||||
|
||||
@property
|
||||
def floats(self) -> np.ndarray:
|
||||
return np.concatenate((
|
||||
self.bounds,
|
||||
np.asarray([self.distance_scale_m,
|
||||
self.inverse_distance_per_m], dtype=np.float32),
|
||||
self.axis_scales_internal,
|
||||
np.asarray([self.minimum_normalized_radius], dtype=np.float32),
|
||||
))
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RosellaModel:
|
||||
source_path: str
|
||||
coefficients: np.ndarray
|
||||
coefficient_sha256: str
|
||||
coefficient_version: str | None
|
||||
room_model: str | None
|
||||
table_a_dimension: int
|
||||
table_a_option: int
|
||||
table_a_extra: int
|
||||
table_a_header_field: int
|
||||
table_a_header_25: int
|
||||
table_a_control: int
|
||||
table_a_option_ids: np.ndarray
|
||||
table_a_option_values: np.ndarray
|
||||
table_a_scalar: np.float32
|
||||
table_a_filter_16x64_padded: np.ndarray
|
||||
table_a_four_integers: np.ndarray
|
||||
table_a_integer: int
|
||||
table_a_filter_8x64_padded: np.ndarray
|
||||
table_a_vector16: np.ndarray
|
||||
table_a_filter_4x64_padded: np.ndarray
|
||||
table_a_extra_indices: np.ndarray
|
||||
table_a_extra_fields_padded: np.ndarray
|
||||
table_a_extra_vectors: np.ndarray
|
||||
sample_rate: int
|
||||
matrix_exponent: int
|
||||
field_exponent: int
|
||||
matrix_left: np.ndarray
|
||||
matrix_right: np.ndarray
|
||||
vector_left: np.ndarray
|
||||
vector_right: np.ndarray
|
||||
field_left_padded: np.ndarray
|
||||
field_right_padded: np.ndarray
|
||||
field_left_odd_serialized_zero: bool
|
||||
hybrid_flags: np.ndarray
|
||||
hybrid_values: np.ndarray
|
||||
model_scalars: np.ndarray
|
||||
header_float_scalars: np.ndarray
|
||||
header_integer_fields: np.ndarray
|
||||
profiles: tuple[DistanceProfile, ...]
|
||||
profile_tail: np.ndarray
|
||||
post_fields: np.ndarray
|
||||
table_a_main_serialized: np.ndarray
|
||||
|
||||
|
||||
def _lane(data: bytes, index: int) -> int:
|
||||
if (index + 1) * 4 > len(data):
|
||||
raise ValueError(f"Rosella rp truncated before int32 lane {index}")
|
||||
return struct.unpack_from("<I", data, index * 4)[0]
|
||||
|
||||
|
||||
def inspect_rp(data: bytes) -> dict:
|
||||
"""Return the active-lane layout and checksum status for one raw rp image."""
|
||||
if len(data) < 20 or len(data) % 4:
|
||||
raise ValueError("Rosella rp must contain whole little-endian int32 lanes")
|
||||
if _lane(data, 0) != 0x7072:
|
||||
raise ValueError(f"bad Rosella rp magic: 0x{_lane(data, 0):08X}")
|
||||
|
||||
low16 = lambda value: value & 0xFFFF
|
||||
checksum = low16(_lane(data, 1))
|
||||
table_a_present = low16(_lane(data, 2))
|
||||
table_b_present = low16(_lane(data, 3))
|
||||
table_c_present = low16(_lane(data, 4))
|
||||
index = 5
|
||||
if table_a_present:
|
||||
table_a_dimension = low16(_lane(data, index))
|
||||
table_a_option = low16(_lane(data, index + 1))
|
||||
table_a_extra = low16(_lane(data, index + 2))
|
||||
index += 5
|
||||
else:
|
||||
table_a_dimension, table_a_option, table_a_extra = 77, 0, 0
|
||||
if table_b_present:
|
||||
if not table_a_present:
|
||||
raise ValueError("Rosella rp table B cannot be present without table A")
|
||||
table_b_dimension = low16(_lane(data, index))
|
||||
table_b_extra = low16(_lane(data, index + 1))
|
||||
table_b_groups = low16(_lane(data, index + 2))
|
||||
index += 3
|
||||
else:
|
||||
table_b_dimension = table_b_extra = table_b_groups = 0
|
||||
if table_c_present:
|
||||
table_c_dimension = low16(_lane(data, index))
|
||||
index += 1
|
||||
else:
|
||||
table_c_dimension = 0
|
||||
|
||||
payload_words = (
|
||||
(index - 2)
|
||||
+ table_b_present * (
|
||||
table_b_dimension + 380 * table_b_groups + table_b_extra + 79)
|
||||
+ table_a_present * (
|
||||
171 * table_a_extra + 79 + 2 * (table_a_option + 14 * table_a_dimension))
|
||||
+ 11
|
||||
+ table_c_present * (314 * table_c_dimension + 1)
|
||||
)
|
||||
total_lanes = 2 + payload_words
|
||||
if len(data) < total_lanes * 4:
|
||||
raise ValueError(
|
||||
f"Rosella rp truncated: need {total_lanes * 4} bytes, have {len(data)}")
|
||||
computed = 0xA569
|
||||
for lane_index in range(2, total_lanes):
|
||||
computed ^= low16(_lane(data, lane_index))
|
||||
computed &= 0xFFFF
|
||||
return {
|
||||
"stored_checksum": checksum,
|
||||
"computed_checksum": computed,
|
||||
"checksum_valid": computed == checksum,
|
||||
"table_a_present": table_a_present,
|
||||
"table_b_present": table_b_present,
|
||||
"table_c_present": table_c_present,
|
||||
"table_a_dimension": table_a_dimension,
|
||||
"table_a_option": table_a_option,
|
||||
"table_a_extra": table_a_extra,
|
||||
"table_b_dimension": table_b_dimension,
|
||||
"table_b_extra": table_b_extra,
|
||||
"table_b_groups": table_b_groups,
|
||||
"table_c_dimension": table_c_dimension,
|
||||
"active_int32_lanes": total_lanes,
|
||||
}
|
||||
|
||||
|
||||
def _json_int32(values) -> np.ndarray:
|
||||
if not isinstance(values, list):
|
||||
raise ValueError("rosella_coefficients must be a JSON array")
|
||||
result = np.empty(len(values), dtype=np.int32)
|
||||
for index, value in enumerate(values):
|
||||
if isinstance(value, bool) or not isinstance(value, Real):
|
||||
raise ValueError(f"rosella_coefficients[{index}] is not a number")
|
||||
numeric = float(value)
|
||||
if not math.isfinite(numeric) or numeric != math.trunc(numeric):
|
||||
raise ValueError(
|
||||
f"rosella_coefficients[{index}] is not an exact integer: {value!r}")
|
||||
integer = int(value)
|
||||
if integer < -(1 << 31) or integer > (1 << 31) - 1:
|
||||
raise ValueError(
|
||||
f"rosella_coefficients[{index}] is outside signed int32: {integer}")
|
||||
result[index] = integer
|
||||
return result
|
||||
|
||||
|
||||
def _load_coefficients(path: Path) -> tuple[np.ndarray, str | None, str | None]:
|
||||
if not path.is_file():
|
||||
raise FileNotFoundError(path)
|
||||
source = path.read_bytes()
|
||||
stripped = source.lstrip()
|
||||
if stripped.startswith(b"{"):
|
||||
try:
|
||||
document = json.loads(source.decode("utf-8"))
|
||||
virtualizer = document["personalized_hrtf"]["virtualizer_parameters"]
|
||||
coefficients = _json_int32(virtualizer["rosella_coefficients"])
|
||||
except (UnicodeDecodeError, json.JSONDecodeError, KeyError, TypeError) as exc:
|
||||
raise ValueError(f"invalid personalized_headphone JSON: {exc}") from exc
|
||||
version = virtualizer.get("rosella_coefficients_version")
|
||||
room = virtualizer.get("room_model")
|
||||
else:
|
||||
if len(source) % 4:
|
||||
raise ValueError("raw rp payload must contain complete int32 lanes")
|
||||
coefficients = np.frombuffer(source, dtype="<i4").copy()
|
||||
version = room = None
|
||||
return coefficients, version, room
|
||||
|
||||
|
||||
def _unpack_field(serialized: np.ndarray, directions: int,
|
||||
exponent: int) -> np.ndarray:
|
||||
expected = 154 * directions
|
||||
if serialized.size != expected:
|
||||
raise ValueError(f"expected {expected} field lanes, got {serialized.size}")
|
||||
padded = np.zeros(160 * directions, dtype=np.float32)
|
||||
stride8 = 8 * directions
|
||||
stride2 = 2 * directions
|
||||
for source_index, value in enumerate(serialized):
|
||||
group4 = (source_index % stride8) // stride2
|
||||
destination = ((group4 & 3) + 4 * (
|
||||
source_index % stride2 +
|
||||
2 * directions * (source_index // stride8 + (group4 >> 2))))
|
||||
padded[destination] = _q15_exp(int(value), exponent)
|
||||
return padded
|
||||
|
||||
|
||||
def _unpack_table_a_grid(serialized: np.ndarray, dimension: int,
|
||||
serialized_rows: int, padded_rows: int,
|
||||
lane_group: int) -> np.ndarray:
|
||||
if serialized.size != serialized_rows * dimension:
|
||||
raise ValueError("unexpected table-A grid size")
|
||||
padded = np.zeros(padded_rows * dimension, dtype=np.float32)
|
||||
group_width = lane_group * 4
|
||||
for source_index, value in enumerate(serialized):
|
||||
remainder = source_index % group_width
|
||||
destination = ((remainder // lane_group) + 4 * (
|
||||
remainder % lane_group +
|
||||
group_width // 4 * (source_index // group_width)))
|
||||
padded[destination] = _q15(int(value))
|
||||
return padded
|
||||
|
||||
|
||||
def _unpack_table_a_extra(serialized: np.ndarray) -> np.ndarray:
|
||||
if serialized.size != 154:
|
||||
raise ValueError("table-A extra field must contain 154 serialized values")
|
||||
padded = np.zeros(160, dtype=np.float32)
|
||||
for source_index, value in enumerate(serialized):
|
||||
remainder = source_index & 7
|
||||
destination = ((remainder >> 1) + 4 * (
|
||||
(source_index & 1) + 2 * (source_index >> 3)))
|
||||
padded[destination] = _q15(int(value))
|
||||
return padded
|
||||
|
||||
|
||||
def _parse_profile(values: np.ndarray, position: int) -> tuple[DistanceProfile, int]:
|
||||
bounds = np.asarray([_q15(int(value)) for value in values[position:position + 6]],
|
||||
dtype=np.float32)
|
||||
position += 6
|
||||
distance = _q15_exp(int(values[position]), int(values[position + 1]))
|
||||
position += 2
|
||||
remaining = np.asarray(
|
||||
[_q15(int(value)) for value in values[position:position + 5]],
|
||||
dtype=np.float32)
|
||||
position += 5
|
||||
return DistanceProfile(
|
||||
bounds=bounds,
|
||||
distance_scale_m=distance,
|
||||
inverse_distance_per_m=remaining[0],
|
||||
axis_scales_internal=remaining[1:4],
|
||||
minimum_normalized_radius=remaining[4],
|
||||
), position
|
||||
|
||||
|
||||
def load_personalized_headphone(path: str | Path) -> RosellaModel:
|
||||
path = Path(path).resolve()
|
||||
coefficients, version, room = _load_coefficients(path)
|
||||
raw = coefficients.astype("<i4", copy=False).tobytes()
|
||||
header = inspect_rp(raw)
|
||||
if not header["checksum_valid"] or header["active_int32_lanes"] != coefficients.size:
|
||||
raise ValueError("invalid or non-active Rosella rp coefficient sequence")
|
||||
if not header["table_a_present"] or not header["table_b_present"] or header["table_c_present"]:
|
||||
raise NotImplementedError("current local renderer requires table A+B and no table C")
|
||||
if header["table_a_dimension"] != 64 or header["table_a_option"] != 3:
|
||||
raise NotImplementedError("current local renderer requires the observed 64-channel HQMF layout")
|
||||
if header["table_b_dimension"] != 20 or header["table_b_groups"] != 36:
|
||||
raise NotImplementedError("current local renderer requires 20 hybrid groups and 36 direction terms")
|
||||
|
||||
values = coefficients
|
||||
extra = header["table_a_extra"]
|
||||
table_a_main_start = 13
|
||||
position = table_a_main_start
|
||||
table_a_control = int(values[position]) & 0xFFFF
|
||||
field_exponent = int(values[position])
|
||||
position += 1
|
||||
option_count = header["table_a_option"]
|
||||
option_ids = (values[position:position + option_count].astype(np.int64) &
|
||||
0xFFFF).astype(np.int32)
|
||||
position += option_count
|
||||
option_values = np.asarray(
|
||||
[_q15(int(value)) for value in values[position:position + option_count]],
|
||||
dtype=np.float32)
|
||||
position += option_count
|
||||
table_a_scalar = _q15(int(values[position]))
|
||||
position += 1
|
||||
dimension = header["table_a_dimension"]
|
||||
table_a_filter_16x64 = _unpack_table_a_grid(
|
||||
values[position:position + 16 * dimension], dimension, 16, 20, 16)
|
||||
position += 16 * dimension
|
||||
table_a_four_integers = (values[position:position + 4].astype(np.int64) &
|
||||
0xFFFF).astype(np.int32)
|
||||
position += 4
|
||||
table_a_integer = int(values[position]) & 0xFFFF
|
||||
position += 1
|
||||
table_a_filter_8x64 = _unpack_table_a_grid(
|
||||
values[position:position + 8 * dimension], dimension, 8, 10, 8)
|
||||
position += 8 * dimension
|
||||
table_a_vector16 = np.asarray(
|
||||
[_q15(int(value)) for value in values[position:position + 16]],
|
||||
dtype=np.float32)
|
||||
position += 16
|
||||
table_a_filter_4x64 = _unpack_table_a_grid(
|
||||
values[position:position + 4 * dimension], dimension, 4, 5, 4)
|
||||
position += 4 * dimension
|
||||
extra_indices = (values[position:position + extra].astype(np.int64) &
|
||||
0xFFFF).astype(np.int32)
|
||||
position += extra
|
||||
extra_fields = np.empty((extra, 160), dtype=np.float32)
|
||||
for index in range(extra):
|
||||
extra_fields[index] = _unpack_table_a_extra(values[position:position + 154])
|
||||
position += 154
|
||||
extra_vectors = np.empty((extra, 16), dtype=np.float32)
|
||||
for index in range(extra):
|
||||
extra_vectors[index] = np.asarray(
|
||||
[_q15(int(value)) for value in values[position:position + 16]],
|
||||
dtype=np.float32)
|
||||
position += 16
|
||||
table_b_start = position
|
||||
expected_table_b_start = table_a_main_start + 1821 + 171 * extra
|
||||
if table_b_start != expected_table_b_start:
|
||||
raise AssertionError(
|
||||
f"table-A parser ended at {table_b_start}, expected {expected_table_b_start}")
|
||||
table_a_main = values[table_a_main_start:table_b_start].copy()
|
||||
|
||||
sample_rate = 2 * (int(values[position]) & 0xFFFF)
|
||||
position += 1
|
||||
matrix_exponent = int(values[position])
|
||||
position += 1
|
||||
matrix_count = 36 * 36
|
||||
scale_matrix = lambda block: np.asarray(
|
||||
[_q15_exp(int(value), matrix_exponent) for value in block],
|
||||
dtype=np.float32).reshape(36, 36)
|
||||
matrix_left = scale_matrix(values[position:position + matrix_count])
|
||||
position += matrix_count
|
||||
matrix_right = scale_matrix(values[position:position + matrix_count])
|
||||
position += matrix_count
|
||||
vector_left = np.asarray(
|
||||
[_q15_exp(int(value), matrix_exponent)
|
||||
for value in values[position:position + 36]], dtype=np.float32)
|
||||
position += 36
|
||||
vector_right = np.asarray(
|
||||
[_q15_exp(int(value), matrix_exponent)
|
||||
for value in values[position:position + 36]], dtype=np.float32)
|
||||
position += 36
|
||||
|
||||
serialized_count = 154 * 36
|
||||
field_left_serialized = values[position:position + serialized_count]
|
||||
field_left = _unpack_field(field_left_serialized, 36, field_exponent)
|
||||
field_left_odd_zero = not np.any(
|
||||
np.abs(np.asarray([_q15_exp(int(value), field_exponent)
|
||||
for value in field_left_serialized[1::2]],
|
||||
dtype=np.float32)) > np.float32(1e-6))
|
||||
position += serialized_count
|
||||
field_right = _unpack_field(
|
||||
values[position:position + serialized_count], 36, field_exponent)
|
||||
position += serialized_count
|
||||
|
||||
hybrid_flags = (values[position:position + 20].astype(np.int64) & 0xFFFF).astype(np.int32)
|
||||
position += 20
|
||||
active_hybrid_values = int(np.count_nonzero(hybrid_flags == 1))
|
||||
if active_hybrid_values != header["table_b_extra"]:
|
||||
raise ValueError(
|
||||
f"hybrid value count {active_hybrid_values} != header {header['table_b_extra']}")
|
||||
hybrid_values = np.asarray(
|
||||
[_q15(int(value)) for value in values[position:position + active_hybrid_values]],
|
||||
dtype=np.float32)
|
||||
position += active_hybrid_values
|
||||
model_scalars = np.asarray(
|
||||
[_q15(int(value)) for value in values[position:position + 5]],
|
||||
dtype=np.float32)
|
||||
position += 5
|
||||
|
||||
expected_table_a_tail = table_b_start + (
|
||||
header["table_b_dimension"] +
|
||||
380 * header["table_b_groups"] +
|
||||
header["table_b_extra"] + 79)
|
||||
if position != expected_table_a_tail:
|
||||
raise AssertionError(f"table-B parser ended at {position}, expected {expected_table_a_tail}")
|
||||
|
||||
header_float_scalars = np.asarray([
|
||||
_q15(int(values[position])),
|
||||
_f32(_q15(int(values[position + 1])) * _f32(16.0)),
|
||||
], dtype=np.float32)
|
||||
header_integer_fields = np.asarray([
|
||||
int(values[position + 2]),
|
||||
int(values[position + 3]) & 0xFFFF,
|
||||
], dtype=np.int32)
|
||||
position += 4
|
||||
profiles = []
|
||||
for _ in range(4):
|
||||
profile, position = _parse_profile(values, position)
|
||||
profiles.append(profile)
|
||||
profile_tail = np.asarray(
|
||||
[_q15(int(value)) for value in values[position:position + 8]],
|
||||
dtype=np.float32)
|
||||
position += 8
|
||||
post_fields = values[position:position + 3].astype(np.int32, copy=True)
|
||||
position += 3
|
||||
if position != values.size:
|
||||
raise AssertionError(f"unparsed coefficient lanes: {values.size - position}")
|
||||
|
||||
return RosellaModel(
|
||||
source_path=str(path),
|
||||
coefficients=coefficients,
|
||||
coefficient_sha256=hashlib.sha256(raw).hexdigest(),
|
||||
coefficient_version=version,
|
||||
room_model=room,
|
||||
table_a_dimension=header["table_a_dimension"],
|
||||
table_a_option=header["table_a_option"],
|
||||
table_a_extra=extra,
|
||||
table_a_header_field=int(values[8]) & 0xFFFF,
|
||||
table_a_header_25=int(values[9]) & 0xFFFF,
|
||||
table_a_control=table_a_control,
|
||||
table_a_option_ids=option_ids,
|
||||
table_a_option_values=option_values,
|
||||
table_a_scalar=table_a_scalar,
|
||||
table_a_filter_16x64_padded=table_a_filter_16x64,
|
||||
table_a_four_integers=table_a_four_integers,
|
||||
table_a_integer=table_a_integer,
|
||||
table_a_filter_8x64_padded=table_a_filter_8x64,
|
||||
table_a_vector16=table_a_vector16,
|
||||
table_a_filter_4x64_padded=table_a_filter_4x64,
|
||||
table_a_extra_indices=extra_indices,
|
||||
table_a_extra_fields_padded=extra_fields,
|
||||
table_a_extra_vectors=extra_vectors,
|
||||
sample_rate=sample_rate,
|
||||
matrix_exponent=matrix_exponent,
|
||||
field_exponent=field_exponent,
|
||||
matrix_left=matrix_left,
|
||||
matrix_right=matrix_right,
|
||||
vector_left=vector_left,
|
||||
vector_right=vector_right,
|
||||
field_left_padded=field_left,
|
||||
field_right_padded=field_right,
|
||||
field_left_odd_serialized_zero=field_left_odd_zero,
|
||||
hybrid_flags=hybrid_flags,
|
||||
hybrid_values=hybrid_values,
|
||||
model_scalars=model_scalars,
|
||||
header_float_scalars=header_float_scalars,
|
||||
header_integer_fields=header_integer_fields,
|
||||
profiles=tuple(profiles),
|
||||
profile_tail=profile_tail,
|
||||
post_fields=post_fields,
|
||||
table_a_main_serialized=table_a_main,
|
||||
)
|
||||
|
||||
|
||||
def direction_basis(x: float, y: float, z: float,
|
||||
dtype=np.float64) -> np.ndarray:
|
||||
"""Return the observed 36-term Rosella direction basis."""
|
||||
f = dtype
|
||||
x, y, z = f(x), f(y), f(z)
|
||||
out = np.empty(36, dtype=dtype)
|
||||
yz = f(y * z)
|
||||
x2 = f(x * x)
|
||||
y2 = f(y * y)
|
||||
x2m02 = f(x2 - f(0.2))
|
||||
xy = f(x * y)
|
||||
out[0:4] = (f(1.0), x, y, z)
|
||||
out[4] = f(x2 - f(1.0 / 3.0))
|
||||
out[5] = xy
|
||||
out[6] = f(x * z)
|
||||
out[7] = f(y2 - f(1.0 / 3.0))
|
||||
out[8] = yz
|
||||
out[9] = f(f(x2 - f(0.6)) * x)
|
||||
out[10] = f(x2m02 * y)
|
||||
out[11] = f(x2m02 * z)
|
||||
out[12] = f(f(y2 - f(0.2)) * x)
|
||||
out[13] = f(yz * x)
|
||||
out[14] = f(f(y2 - f(0.6)) * y)
|
||||
out[15] = f(f(y2 - f(0.2)) * z)
|
||||
out[16] = f(f(x2 * x2) - f(0.2))
|
||||
out[17] = f(xy * x2)
|
||||
out[18] = f(f(x * z) * x2)
|
||||
out[19] = f(f(y2 * x2) - f(1.0 / 15.0))
|
||||
out[20] = f(yz * x2)
|
||||
out[21] = f(x * y2 * y)
|
||||
out[22] = f(x * y2 * z)
|
||||
out[23] = f(f(y2 * y2) - f(0.2))
|
||||
x4 = f(x2 * x2)
|
||||
x2y2 = f(y2 * x2)
|
||||
y4 = f(y2 * y2)
|
||||
out[24] = f(yz * y2)
|
||||
out[25] = f(f(x4 - f(3.0 / 7.0)) * x)
|
||||
out[26] = f(f(x4 - f(3.0 / 35.0)) * y)
|
||||
out[27] = f(f(x4 - f(3.0 / 35.0)) * z)
|
||||
out[28] = f(f(x2y2 - f(3.0 / 35.0)) * x)
|
||||
out[29] = f(f(x2 * z) * xy)
|
||||
out[30] = f(f(x2y2 - f(3.0 / 35.0)) * y)
|
||||
out[31] = f(f(x2y2 - f(1.0 / 35.0)) * z)
|
||||
out[32] = f(f(y4 - f(3.0 / 35.0)) * x)
|
||||
out[33] = f(f(y2 * z) * xy)
|
||||
out[34] = f(f(y4 - f(3.0 / 7.0)) * y)
|
||||
out[35] = f(f(y4 - f(3.0 / 35.0)) * z)
|
||||
return out
|
||||
@@ -0,0 +1,198 @@
|
||||
"""Float64 Rosella table-A room model and overlap-add realization."""
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
|
||||
from rosella_model import RosellaModel
|
||||
|
||||
|
||||
class _RosellaRoomState:
|
||||
"""Recursive table-A state used to generate the stable FIR realization."""
|
||||
|
||||
def __init__(self, model: RosellaModel):
|
||||
self.model = model
|
||||
self.bands = min(64, model.table_a_dimension)
|
||||
self.delays = model.table_a_four_integers.astype(np.int32)
|
||||
self.capacity = int(np.max(self.delays))
|
||||
self.matrix = np.asarray(model.table_a_vector16, dtype=np.float64).reshape(
|
||||
4, 4, order="F")
|
||||
f8 = np.asarray(model.table_a_filter_8x64_padded, dtype=np.float64).reshape(
|
||||
20, 4, 2, 4)
|
||||
f4 = np.asarray(model.table_a_filter_4x64_padded, dtype=np.float64).reshape(
|
||||
20, 4, 4)
|
||||
f16 = np.asarray(model.table_a_filter_16x64_padded, dtype=np.float64).reshape(
|
||||
20, 4, 4, 4)
|
||||
self.feedback_real = np.empty((self.bands, 4), dtype=np.float64)
|
||||
self.feedback_imag = np.empty_like(self.feedback_real)
|
||||
self.output_tap = np.empty_like(self.feedback_real)
|
||||
self.left_real = np.empty_like(self.feedback_real)
|
||||
self.left_imag = np.empty_like(self.feedback_real)
|
||||
self.right_real = np.empty_like(self.feedback_real)
|
||||
self.right_imag = np.empty_like(self.feedback_real)
|
||||
for band in range(self.bands):
|
||||
group, lane = divmod(band, 4)
|
||||
self.feedback_real[band] = f8[group, :, 0, lane]
|
||||
self.feedback_imag[band] = f8[group, :, 1, lane]
|
||||
self.output_tap[band] = f4[group, :, lane]
|
||||
self.left_real[band] = f16[group, :, 0, lane]
|
||||
self.left_imag[band] = f16[group, :, 1, lane]
|
||||
self.right_real[band] = f16[group, :, 2, lane]
|
||||
self.right_imag[band] = f16[group, :, 3, lane]
|
||||
|
||||
self.allpass_gain = np.asarray(model.table_a_option_values, dtype=np.float64)
|
||||
self.allpass_delay = model.table_a_option_ids.astype(np.int32)
|
||||
self.allpass_real = [
|
||||
np.zeros((int(delay), self.bands), dtype=np.float64)
|
||||
for delay in self.allpass_delay
|
||||
]
|
||||
self.allpass_imag = [np.zeros_like(value) for value in self.allpass_real]
|
||||
self.allpass_position = np.zeros(len(self.allpass_real), dtype=np.int32)
|
||||
self.memory_real = np.zeros(
|
||||
(self.capacity, self.bands, 4), dtype=np.float64)
|
||||
self.memory_imag = np.zeros_like(self.memory_real)
|
||||
self.position = 0
|
||||
self.extra_fields = np.asarray(
|
||||
model.table_a_extra_fields_padded, dtype=np.float64).reshape(-1, 20, 2, 4)
|
||||
self.extra_matrices = [
|
||||
np.asarray(value, dtype=np.float64).reshape(4, 4, order="F")
|
||||
for value in model.table_a_extra_vectors
|
||||
]
|
||||
|
||||
def reset(self):
|
||||
for value in self.allpass_real + self.allpass_imag:
|
||||
value.fill(0.0)
|
||||
self.allpass_position.fill(0)
|
||||
self.memory_real.fill(0.0)
|
||||
self.memory_imag.fill(0.0)
|
||||
self.position = 0
|
||||
|
||||
def process_slot(self, room_send) -> np.ndarray:
|
||||
values = np.asarray(room_send, dtype=np.complex128)
|
||||
input_real = values[:self.bands].real * 0.70710677
|
||||
input_imag = values[:self.bands].imag * 0.70710677
|
||||
if float(self.model.table_a_scalar) >= 0.5:
|
||||
raise NotImplementedError("alternate Rosella table-A room mode")
|
||||
|
||||
for index, gain in enumerate(self.allpass_gain):
|
||||
position = int(self.allpass_position[index])
|
||||
previous_real = self.allpass_real[index][position].copy()
|
||||
previous_imag = self.allpass_imag[index][position].copy()
|
||||
residual_real = input_real - previous_real * gain
|
||||
residual_imag = input_imag - previous_imag * gain
|
||||
input_real = residual_real * gain + previous_real
|
||||
input_imag = residual_imag * gain + previous_imag
|
||||
self.allpass_real[index][position] = residual_real
|
||||
self.allpass_imag[index][position] = residual_imag
|
||||
self.allpass_position[index] = (
|
||||
position + 1) % len(self.allpass_real[index])
|
||||
|
||||
branch_real = np.repeat(input_real[:, None], 4, axis=1)
|
||||
branch_imag = np.repeat(input_imag[:, None], 4, axis=1)
|
||||
delayed_real = np.empty_like(branch_real)
|
||||
delayed_imag = np.empty_like(branch_imag)
|
||||
for branch, delay in enumerate(self.delays):
|
||||
delayed_real[:, branch] = self.memory_real[
|
||||
(self.position - int(delay)) % self.capacity, :, branch]
|
||||
delayed_imag[:, branch] = self.memory_imag[
|
||||
(self.position - int(delay)) % self.capacity, :, branch]
|
||||
branch_real += np.einsum(
|
||||
"bj,ij->bi", delayed_real, self.matrix,
|
||||
dtype=np.float64, optimize=False)
|
||||
branch_imag += np.einsum(
|
||||
"bj,ij->bi", delayed_imag, self.matrix,
|
||||
dtype=np.float64, optimize=False)
|
||||
|
||||
tap_index = (self.position - self.model.table_a_integer) % self.capacity
|
||||
tap_real = self.memory_real[tap_index].copy()
|
||||
tap_imag = self.memory_imag[tap_index].copy()
|
||||
next_real = branch_real * self.feedback_real - branch_imag * self.feedback_imag
|
||||
next_imag = branch_imag * self.feedback_real + branch_real * self.feedback_imag
|
||||
self.memory_real[self.position] = next_real
|
||||
self.memory_imag[self.position] = next_imag
|
||||
self.position = (self.position + 1) % self.capacity
|
||||
|
||||
extra_real = np.zeros_like(branch_real)
|
||||
extra_imag = np.zeros_like(branch_imag)
|
||||
for index, delay in enumerate(self.model.table_a_extra_indices):
|
||||
source_real = self.memory_real[
|
||||
(self.position - (int(delay) + 1)) % self.capacity]
|
||||
source_imag = self.memory_imag[
|
||||
(self.position - (int(delay) + 1)) % self.capacity]
|
||||
matrix = self.extra_matrices[index]
|
||||
mixed_real = np.einsum(
|
||||
"bj,ij->bi", source_real, matrix,
|
||||
dtype=np.float64, optimize=False)
|
||||
mixed_imag = np.einsum(
|
||||
"bj,ij->bi", source_imag, matrix,
|
||||
dtype=np.float64, optimize=False)
|
||||
coefficient_real = np.empty(self.bands, dtype=np.float64)
|
||||
coefficient_imag = np.empty(self.bands, dtype=np.float64)
|
||||
for band in range(self.bands):
|
||||
group, lane = divmod(band, 4)
|
||||
coefficient_real[band] = self.extra_fields[index, group, 0, lane]
|
||||
coefficient_imag[band] = self.extra_fields[index, group, 1, lane]
|
||||
extra_real += (mixed_real * coefficient_real[:, None]
|
||||
- mixed_imag * coefficient_imag[:, None])
|
||||
extra_imag += (mixed_imag * coefficient_real[:, None]
|
||||
+ mixed_real * coefficient_imag[:, None])
|
||||
|
||||
output_real = tap_real * self.output_tap + extra_real
|
||||
output_imag = tap_imag * self.output_tap + extra_imag
|
||||
left = np.sum(
|
||||
self.left_real * output_real - self.left_imag * output_imag,
|
||||
axis=1, dtype=np.float64)
|
||||
left_imag = np.sum(
|
||||
self.left_imag * output_real + self.left_real * output_imag,
|
||||
axis=1, dtype=np.float64)
|
||||
right = np.sum(
|
||||
self.right_real * output_real - self.right_imag * output_imag,
|
||||
axis=1, dtype=np.float64)
|
||||
right_imag = np.sum(
|
||||
self.right_imag * output_real + self.right_real * output_imag,
|
||||
axis=1, dtype=np.float64)
|
||||
result = np.zeros((2, 77), dtype=np.complex128)
|
||||
result[0, :self.bands] = left + 1j * left_imag
|
||||
result[1, :self.bands] = right + 1j * right_imag
|
||||
return result
|
||||
|
||||
|
||||
class RosellaRoomFir:
|
||||
"""Complex128 overlap-add room FIR generated locally from table-A."""
|
||||
|
||||
def __init__(self, model: RosellaModel, impulse_slots: int = 4096):
|
||||
if impulse_slots <= 0:
|
||||
raise ValueError("impulse_slots must be positive")
|
||||
reference = _RosellaRoomState(model)
|
||||
self.length = int(impulse_slots)
|
||||
self.kernel = np.empty((self.length, 2, 64), dtype=np.complex128)
|
||||
for slot in range(self.length):
|
||||
impulse = np.zeros(77, dtype=np.complex128)
|
||||
if slot == 0:
|
||||
impulse[:64] = 1.0
|
||||
self.kernel[slot] = reference.process_slot(impulse)[:, :64]
|
||||
self.tail = np.zeros((self.length - 1, 2, 64), dtype=np.complex128)
|
||||
self._fft_cache: dict[int, np.ndarray] = {}
|
||||
|
||||
def reset(self):
|
||||
self.tail.fill(0.0)
|
||||
|
||||
def process_chunk(self, room_send) -> np.ndarray:
|
||||
values = np.asarray(room_send, dtype=np.complex128)
|
||||
if values.ndim != 2 or values.shape[1] != 77:
|
||||
raise ValueError("room_send must have shape [slots,77]")
|
||||
count = len(values)
|
||||
if count == 0:
|
||||
return np.zeros((0, 2, 77), dtype=np.complex128)
|
||||
needed = count + self.length - 1
|
||||
fft_size = 1 << (needed - 1).bit_length()
|
||||
kernel_fft = self._fft_cache.get(fft_size)
|
||||
if kernel_fft is None:
|
||||
kernel_fft = np.fft.fft(self.kernel, fft_size, axis=0)
|
||||
self._fft_cache[fft_size] = kernel_fft
|
||||
input_fft = np.fft.fft(values[:, :64], fft_size, axis=0)
|
||||
block = np.fft.ifft(input_fft[:, None, :] * kernel_fft, axis=0)[:needed]
|
||||
block[:len(self.tail)] += self.tail
|
||||
result = np.zeros((count, 2, 77), dtype=np.complex128)
|
||||
result[:, :, :64] = block[:count]
|
||||
self.tail = block[count:count + self.length - 1].copy()
|
||||
return result
|
||||
+101
-29
@@ -1,4 +1,4 @@
|
||||
"""Streaming spool and WAV writer for direct speaker-layout output."""
|
||||
"""Shared PCM spool, peak analysis, and WAV writer for direct outputs."""
|
||||
from __future__ import annotations
|
||||
|
||||
import struct
|
||||
@@ -13,48 +13,114 @@ _PCM_GUID = bytes.fromhex("0100000000001000800000aa00389b71")
|
||||
_FLOAT_GUID = bytes.fromhex("0300000000001000800000aa00389b71")
|
||||
|
||||
|
||||
class SpeakerPcmSpool:
|
||||
"""Temporary interleaved float32 store with float64 peak analysis."""
|
||||
class PcmSpool:
|
||||
"""Temporary interleaved PCM store with float64 peak and clipping analysis.
|
||||
|
||||
def __init__(self, path, sample_count, channel_count):
|
||||
``storage_dtype`` controls only the temporary representation. Speaker
|
||||
output keeps its historical float32 spool, while binaural uses float64 so
|
||||
precision is reduced only by the selected final WAV format.
|
||||
"""
|
||||
|
||||
def __init__(self, path, sample_capacity, channel_count, *,
|
||||
expected_samples=None, storage_dtype="<f4",
|
||||
tail_threshold=None):
|
||||
self.path = Path(path)
|
||||
self.sample_count = int(sample_count)
|
||||
self.sample_capacity = int(sample_capacity)
|
||||
self.sample_count = int(
|
||||
self.sample_capacity if expected_samples is None else expected_samples)
|
||||
self.expected_samples = (
|
||||
None if expected_samples is None else int(expected_samples))
|
||||
self.channel_count = int(channel_count)
|
||||
self.storage_dtype = np.dtype(storage_dtype)
|
||||
self.tail_threshold = (
|
||||
None if tail_threshold is None else float(tail_threshold))
|
||||
if self.sample_capacity < 0 or self.channel_count <= 0:
|
||||
raise ValueError("invalid PCM spool dimensions")
|
||||
if self.expected_samples is not None and not (
|
||||
0 <= self.expected_samples <= self.sample_capacity):
|
||||
raise ValueError("expected_samples exceeds sample_capacity")
|
||||
if self.tail_threshold is not None and self.tail_threshold < 0.0:
|
||||
raise ValueError("tail_threshold must be non-negative")
|
||||
self.position = 0
|
||||
self.peak = 0.0
|
||||
self.clipped_values = 0
|
||||
self.values = np.memmap(
|
||||
self.path, dtype="<f4", mode="w+",
|
||||
shape=(self.sample_count, self.channel_count),
|
||||
self.last_above_threshold = -1
|
||||
self.kept_samples = None
|
||||
self._values = np.memmap(
|
||||
self.path, dtype=self.storage_dtype, mode="w+",
|
||||
shape=(self.sample_capacity, self.channel_count),
|
||||
)
|
||||
|
||||
@property
|
||||
def values(self):
|
||||
if self._values is None:
|
||||
raise RuntimeError("PCM spool is closed")
|
||||
length = (self.position if self.kept_samples is None
|
||||
else self.kept_samples)
|
||||
return self._values[:length]
|
||||
|
||||
def write_frame(self, pcm):
|
||||
values = np.asarray(pcm, dtype=np.float64)
|
||||
if values.ndim != 2 or values.shape[1] != self.channel_count:
|
||||
raise ValueError(
|
||||
f"speaker frame must have shape [samples,{self.channel_count}], got {values.shape}")
|
||||
if self.position + len(values) > self.sample_count:
|
||||
raise ValueError("speaker spool received more samples than allocated")
|
||||
f"PCM frame must have shape [samples,{self.channel_count}], got {values.shape}")
|
||||
if self.position + len(values) > self.sample_capacity:
|
||||
raise ValueError("PCM spool received more samples than allocated")
|
||||
if not np.all(np.isfinite(values)):
|
||||
raise ValueError("speaker renderer produced NaN or infinity")
|
||||
raise ValueError("renderer produced NaN or infinity")
|
||||
absolute = np.abs(values)
|
||||
if absolute.size:
|
||||
self.peak = max(self.peak, float(np.max(absolute)))
|
||||
self.clipped_values += int(np.count_nonzero(absolute > 1.0))
|
||||
self.values[self.position:self.position + len(values)] = values.astype(np.float32)
|
||||
if self.tail_threshold is not None:
|
||||
per_sample = np.max(absolute, axis=1)
|
||||
above = np.flatnonzero(per_sample > self.tail_threshold)
|
||||
if above.size:
|
||||
self.last_above_threshold = self.position + int(above[-1])
|
||||
self._values[self.position:self.position + len(values)] = values.astype(
|
||||
self.storage_dtype, copy=False)
|
||||
self.position += len(values)
|
||||
|
||||
def finalize(self):
|
||||
if self.position != self.sample_count:
|
||||
def finalize(self, *, minimum_samples=0):
|
||||
if (self.expected_samples is not None
|
||||
and self.position != self.expected_samples):
|
||||
raise ValueError(
|
||||
f"speaker spool has {self.position} samples, expected {self.sample_count}")
|
||||
self.values.flush()
|
||||
f"PCM spool has {self.position} samples, expected {self.expected_samples}")
|
||||
keep = self.position
|
||||
if self.tail_threshold is not None:
|
||||
keep = min(
|
||||
self.position,
|
||||
max(int(minimum_samples), self.last_above_threshold + 1),
|
||||
)
|
||||
self.kept_samples = keep
|
||||
self.sample_count = keep
|
||||
self._values.flush()
|
||||
return self
|
||||
|
||||
def close(self):
|
||||
values = self.values
|
||||
self.values = None
|
||||
del values
|
||||
values = self._values
|
||||
self._values = None
|
||||
if values is not None:
|
||||
del values
|
||||
|
||||
|
||||
class SpeakerPcmSpool(PcmSpool):
|
||||
"""Backward-compatible fixed-length float32 speaker spool."""
|
||||
|
||||
def __init__(self, path, sample_count, channel_count):
|
||||
super().__init__(
|
||||
path, sample_count, channel_count,
|
||||
expected_samples=sample_count, storage_dtype="<f4")
|
||||
|
||||
|
||||
class BinauralPcmSpool(PcmSpool):
|
||||
"""Float64 variable-tail spool for the Rosella binaural renderer."""
|
||||
|
||||
def __init__(self, path, sample_capacity, *, tail_threshold=1.0e-8):
|
||||
super().__init__(
|
||||
path, sample_capacity, 2,
|
||||
expected_samples=None, storage_dtype="<f8",
|
||||
tail_threshold=tail_threshold)
|
||||
|
||||
|
||||
def _fmt_chunk(channel_count, rate, sample_format):
|
||||
@@ -69,7 +135,7 @@ def _fmt_chunk(channel_count, rate, sample_format):
|
||||
simple_tag = WAVE_FORMAT_PCM
|
||||
guid = _PCM_GUID
|
||||
else:
|
||||
raise ValueError(f"unsupported speaker WAV format: {sample_format}")
|
||||
raise ValueError(f"unsupported WAV format: {sample_format}")
|
||||
block_align = channel_count * bytes_per_sample
|
||||
byte_rate = rate * block_align
|
||||
if channel_count <= 2:
|
||||
@@ -94,13 +160,13 @@ def _write_header(stream, channel_count, sample_count, rate, sample_format):
|
||||
riff_file_size = 12 + 8 + len(fmt) + 8 + data_size
|
||||
use_rf64 = riff_file_size - 8 > 0xFFFFFFFF
|
||||
if use_rf64:
|
||||
# RF64 + ds64 + fmt + data.
|
||||
file_size = 12 + 36 + 8 + len(fmt) + 8 + data_size
|
||||
stream.write(b"RF64")
|
||||
stream.write(struct.pack("<I", 0xFFFFFFFF))
|
||||
stream.write(b"WAVE")
|
||||
stream.write(b"ds64")
|
||||
stream.write(struct.pack("<IQQQI", 28, file_size - 8, data_size, sample_count, 0))
|
||||
stream.write(struct.pack(
|
||||
"<IQQQI", 28, file_size - 8, data_size, sample_count, 0))
|
||||
else:
|
||||
stream.write(b"RIFF")
|
||||
stream.write(struct.pack("<I", riff_file_size - 8))
|
||||
@@ -121,7 +187,8 @@ def _write_header(stream, channel_count, sample_count, rate, sample_format):
|
||||
|
||||
|
||||
def _pack_int24(values):
|
||||
scaled = (np.clip(values, -1.0, 1.0) * np.float32(8388607.0)).astype(np.int32)
|
||||
source = np.asarray(values)
|
||||
scaled = (np.clip(source, -1.0, 1.0) * np.float32(8388607.0)).astype(np.int32)
|
||||
unsigned = scaled.reshape(-1).view(np.uint32)
|
||||
packed = np.empty((unsigned.size, 3), dtype=np.uint8)
|
||||
packed[:, 0] = unsigned & 0xFF
|
||||
@@ -130,21 +197,22 @@ def _pack_int24(values):
|
||||
return packed.tobytes()
|
||||
|
||||
|
||||
def write_speaker_wav(path, pcm, sample_format, *, rate=48000, chunk_samples=262144):
|
||||
"""Write an interleaved float32 array/memmap as float32 or PCM24 WAV."""
|
||||
def write_pcm_wav(path, pcm, sample_format, *, rate=48000,
|
||||
chunk_samples=262144):
|
||||
"""Write an interleaved array/memmap as float32 or PCM24 WAV."""
|
||||
target = Path(path)
|
||||
values = np.asarray(pcm)
|
||||
if values.ndim != 2:
|
||||
raise ValueError(f"speaker PCM must be 2D, got {values.shape}")
|
||||
raise ValueError(f"PCM must be 2D, got {values.shape}")
|
||||
sample_count, channel_count = values.shape
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
with target.open("wb") as stream:
|
||||
info = _write_header(
|
||||
stream, channel_count, sample_count, int(rate), sample_format)
|
||||
for start in range(0, sample_count, int(chunk_samples)):
|
||||
block = np.asarray(values[start:start + chunk_samples], dtype="<f4")
|
||||
block = np.asarray(values[start:start + chunk_samples])
|
||||
if sample_format == "float32":
|
||||
stream.write(block.tobytes(order="C"))
|
||||
stream.write(block.astype("<f4", copy=False).tobytes(order="C"))
|
||||
else:
|
||||
stream.write(_pack_int24(block))
|
||||
info.update({
|
||||
@@ -155,3 +223,7 @@ def write_speaker_wav(path, pcm, sample_format, *, rate=48000, chunk_samples=262
|
||||
"file_bytes": target.stat().st_size,
|
||||
})
|
||||
return info
|
||||
|
||||
|
||||
# Existing imports remain valid.
|
||||
write_speaker_wav = write_pcm_wav
|
||||
|
||||
@@ -0,0 +1,243 @@
|
||||
import importlib.util
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
SRC = ROOT / "src"
|
||||
if str(SRC) not in sys.path:
|
||||
sys.path.insert(0, str(SRC))
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
import main
|
||||
from adm_atmos import q_to_adm_xyz
|
||||
from binaural_metadata import OamdPositionTimeline
|
||||
from binaural_renderer import (
|
||||
DEFAULT_PERSONALIZED_HEADPHONE,
|
||||
RosellaBinauralRenderer,
|
||||
resolve_personalized_headphone,
|
||||
)
|
||||
from oamd_bits import q_of
|
||||
from rosella_direct import BINAURAL_PROFILE_NAMES, direct_and_room_send, special_lfe_direct
|
||||
from rosella_filterbank import HybridAnalysis, HybridSynthesis, QmfAnalysis, QmfSynthesis
|
||||
from rosella_model import load_personalized_headphone
|
||||
from speaker_wav import BinauralPcmSpool, write_pcm_wav
|
||||
|
||||
_MODEL_CANDIDATES = list(
|
||||
(ROOT / "tests" / "binauraltests" / "evidence").glob(
|
||||
"*/test.personalized_headphone"))
|
||||
TEST_MODEL = _MODEL_CANDIDATES[0] if _MODEL_CANDIDATES else Path()
|
||||
|
||||
|
||||
class BinauralProductionTest(unittest.TestCase):
|
||||
def test_cli_exposes_only_near_mid_far_and_defaults_mid_float32(self):
|
||||
parser = main.build_parser()
|
||||
args = parser.parse_args(["input.eac3", "--binaural"])
|
||||
self.assertEqual(args.binaural_mode, "mid")
|
||||
self.assertEqual(args.binaural_format, "float32")
|
||||
action = next(a for a in parser._actions if a.dest == "binaural_mode")
|
||||
self.assertEqual(tuple(action.choices), ("near", "mid", "far"))
|
||||
self.assertNotIn("off", BINAURAL_PROFILE_NAMES)
|
||||
|
||||
def test_default_model_path_and_missing_model_message(self):
|
||||
self.assertEqual(
|
||||
DEFAULT_PERSONALIZED_HEADPHONE,
|
||||
ROOT / "HRTF" / "binaural.personalized_headphone")
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
missing = Path(directory) / "missing.personalized_headphone"
|
||||
with self.assertRaisesRegex(FileNotFoundError, "未找到双耳模型"):
|
||||
resolve_personalized_headphone(missing)
|
||||
|
||||
def test_runtime_filterbanks_are_double_precision(self):
|
||||
qmf = QmfAnalysis(1)
|
||||
hybrid = HybridAnalysis(1)
|
||||
hybrid_synthesis = HybridSynthesis(1)
|
||||
qmf_synthesis = QmfSynthesis(1)
|
||||
self.assertEqual(qmf.coefficients.dtype, np.float64)
|
||||
self.assertEqual(qmf.history.dtype, np.float64)
|
||||
self.assertEqual(hybrid.low_kernel.dtype, np.float64)
|
||||
self.assertEqual(hybrid.high_history.dtype, np.complex128)
|
||||
self.assertEqual(qmf_synthesis.basis.dtype, np.float64)
|
||||
source = np.zeros((1, 1, 64), dtype=np.float64)
|
||||
q = qmf.process_chunk(source)
|
||||
h = hybrid.process_chunk(q)
|
||||
self.assertEqual(q.dtype, np.complex128)
|
||||
self.assertEqual(h.dtype, np.complex128)
|
||||
back = hybrid_synthesis.process_chunk(h)
|
||||
time = qmf_synthesis.process_chunk(back)
|
||||
self.assertEqual(back.dtype, np.complex128)
|
||||
self.assertEqual(time.dtype, np.float64)
|
||||
|
||||
@unittest.skipUnless(
|
||||
(ROOT / "tests" / "binauraltests" / "hqmf.py").is_file(),
|
||||
"research filterbank reference not present")
|
||||
def test_compact_kernel_archive_matches_research_float64_filterbanks(self):
|
||||
reference_root = ROOT / "tests" / "binauraltests"
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"binaural_reference_hqmf", reference_root / "hqmf.py")
|
||||
reference = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(reference)
|
||||
evidence = reference_root / "evidence"
|
||||
source = np.fromfile(
|
||||
evidence / "hqmf_kernel" / "hqmf_validation_input.f32le",
|
||||
dtype="<f4").astype(np.float64).reshape(-1, 1, 64)
|
||||
production_qmf = QmfAnalysis(1).process_chunk(source)
|
||||
reference_qmf = reference.QmfAnalysisFast(
|
||||
evidence / "hqmf_kernel" / "hqmf_analysis_manifest.json",
|
||||
1, dtype=np.float64).process_chunk(source)
|
||||
np.testing.assert_array_equal(production_qmf, reference_qmf)
|
||||
|
||||
production_hybrid = HybridAnalysis(1).process_chunk(production_qmf)
|
||||
reference_hybrid = reference.HybridAnalysis(
|
||||
evidence / "hybrid_kernel" / "hybrid_analysis_manifest.json",
|
||||
1, dtype=np.float64).process_chunk(reference_qmf)
|
||||
np.testing.assert_array_equal(production_hybrid, reference_hybrid)
|
||||
|
||||
production_qmf_back = HybridSynthesis(1).process_chunk(production_hybrid)
|
||||
reference_qmf_back = reference.HybridSynthesis(
|
||||
evidence / "hybrid_synthesis_kernel" / "hybrid_synthesis_manifest.json",
|
||||
1, dtype=np.float64).process_chunk(reference_hybrid)
|
||||
np.testing.assert_array_equal(production_qmf_back, reference_qmf_back)
|
||||
|
||||
production_time = QmfSynthesis(1).process_chunk(production_qmf_back)
|
||||
reference_time = reference.QmfSynthesisFast(
|
||||
evidence / "qmf_synthesis_kernel" / "qmf_synthesis_manifest.json",
|
||||
1, dtype=np.float64).process_chunk(reference_qmf_back)
|
||||
np.testing.assert_array_equal(production_time, reference_time)
|
||||
|
||||
def test_special_lfe_is_fixed_16_band_complex128_without_room_send(self):
|
||||
result = special_lfe_direct()
|
||||
self.assertEqual(result.gains.dtype, np.complex128)
|
||||
np.testing.assert_array_equal(result.gains[0], result.gains[1])
|
||||
self.assertTrue(np.any(result.gains[:, :16] != 0))
|
||||
np.testing.assert_array_equal(result.gains[:, 16:], 0)
|
||||
self.assertEqual(float(result.room_send), 0.0)
|
||||
|
||||
def test_oamd_timing_retains_outer_block_delay_and_ramp(self):
|
||||
timeline = OamdPositionTimeline()
|
||||
initial = {
|
||||
"values": {
|
||||
(1, "q1"): q_of(0, 62),
|
||||
(1, "q2"): q_of(0, 62),
|
||||
(1, "q3"): q_of(0, 15),
|
||||
},
|
||||
"block_offset_samples": 0,
|
||||
"ramp_duration_samples": 1536,
|
||||
}
|
||||
timeline.submit_update(initial, frame_start_sample=0)
|
||||
old = np.asarray(q_to_adm_xyz(q_of(0, 62), q_of(0, 62), q_of(0, 15)))
|
||||
np.testing.assert_allclose(timeline.positions_at(0)[0], old)
|
||||
|
||||
target_q1 = q_of(62, 62)
|
||||
changed = {
|
||||
"values": {(1, "q1"): target_q1},
|
||||
"block_offset_samples": 32,
|
||||
"ramp_duration_samples": 1536,
|
||||
}
|
||||
timeline.submit_update(
|
||||
changed,
|
||||
frame_start_sample=1536,
|
||||
outer_sample_offset=16,
|
||||
object_delay_samples=1473,
|
||||
)
|
||||
start = 1536 + 16 + 32 + 1473 + 64
|
||||
duration = 1536 - 64
|
||||
target = np.asarray(q_to_adm_xyz(target_q1, q_of(0, 62), q_of(0, 15)))
|
||||
np.testing.assert_allclose(timeline.positions_at(start - 1)[0], old)
|
||||
np.testing.assert_allclose(timeline.positions_at(start)[0], old)
|
||||
np.testing.assert_allclose(
|
||||
timeline.positions_at(start + duration // 2)[0],
|
||||
old + (target - old) * 0.5,
|
||||
)
|
||||
np.testing.assert_allclose(timeline.positions_at(start + duration)[0], target)
|
||||
|
||||
@unittest.skipUnless(TEST_MODEL.is_file(), "test model not present")
|
||||
def test_model_parser_and_direct_path_promote_to_double(self):
|
||||
model = load_personalized_headphone(TEST_MODEL)
|
||||
self.assertEqual(model.sample_rate, 48000)
|
||||
self.assertEqual(len(model.coefficients), 16033)
|
||||
self.assertEqual(
|
||||
model.coefficient_sha256,
|
||||
"2d4b40c27925ec8827556585d90c371384458d31bb84b20f94a946145558ea27",
|
||||
)
|
||||
result = direct_and_room_send(model, (0.0, 1.0, 0.0), 3)
|
||||
self.assertEqual(result.gains.dtype, np.complex128)
|
||||
with self.assertRaisesRegex(ValueError, "near, mid, or far"):
|
||||
direct_and_room_send(model, (0.0, 1.0, 0.0), 0)
|
||||
|
||||
@unittest.skipUnless(TEST_MODEL.is_file(), "test model not present")
|
||||
def test_renderer_keeps_float64_state_and_compensates_961_samples(self):
|
||||
renderer = RosellaBinauralRenderer(
|
||||
TEST_MODEL,
|
||||
chunk_frames=1,
|
||||
tail_seconds=0,
|
||||
room_impulse_slots=32,
|
||||
)
|
||||
source = np.zeros((1536, 16), dtype=np.float32)
|
||||
source[0, 1] = 1.0
|
||||
first = renderer.render_frame(source)
|
||||
tail = renderer.finish()
|
||||
self.assertEqual(first.dtype, np.float64)
|
||||
self.assertEqual(tail.dtype, np.float64)
|
||||
self.assertEqual(len(first), 1536 - 961)
|
||||
self.assertEqual(renderer.qmf_analysis.history.dtype, np.float64)
|
||||
self.assertEqual(renderer.hybrid_analysis.high_history.dtype, np.complex128)
|
||||
self.assertEqual(renderer.core.gains.dtype, np.complex128)
|
||||
self.assertEqual(renderer.core.room.tail.dtype, np.complex128)
|
||||
|
||||
@unittest.skipUnless(TEST_MODEL.is_file(), "test model not present")
|
||||
def test_native_backend_matches_python_backend(self):
|
||||
source = np.zeros((1536, 16), dtype=np.float32)
|
||||
source[0, 0] = 0.1
|
||||
source[64, 1] = -0.2
|
||||
python_renderer = RosellaBinauralRenderer(
|
||||
TEST_MODEL, backend="python", chunk_frames=1,
|
||||
tail_seconds=0, room_impulse_slots=64)
|
||||
try:
|
||||
native_renderer = RosellaBinauralRenderer(
|
||||
TEST_MODEL, backend="native",
|
||||
native_library=ROOT / "lib" / "eac3joc_core.dll",
|
||||
chunk_frames=1, tail_seconds=0)
|
||||
except (OSError, RuntimeError):
|
||||
python_renderer.close()
|
||||
self.skipTest("native binaural backend not built")
|
||||
try:
|
||||
self.assertEqual(native_renderer.dsp_backend, "native")
|
||||
python_output = np.concatenate((
|
||||
python_renderer.render_frame(source), python_renderer.finish()))
|
||||
native_output = np.concatenate((
|
||||
native_renderer.render_frame(source), native_renderer.finish()))
|
||||
np.testing.assert_allclose(
|
||||
native_output, python_output, rtol=0.0, atol=2.0e-14)
|
||||
finally:
|
||||
python_renderer.close()
|
||||
native_renderer.close()
|
||||
|
||||
def test_binaural_spool_preserves_float64_then_trims_tail(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
raw = Path(directory) / "binaural.raw"
|
||||
wav = Path(directory) / "binaural.wav"
|
||||
spool = BinauralPcmSpool(raw, 8, tail_threshold=1.0e-8)
|
||||
values = np.asarray([
|
||||
[0.0, 0.0],
|
||||
[0.25, -0.25],
|
||||
[1.0 + 2.0 ** -40, 0.0],
|
||||
[1.0e-9, 0.0],
|
||||
], dtype=np.float64)
|
||||
spool.write_frame(values)
|
||||
spool.finalize(minimum_samples=2)
|
||||
self.assertEqual(spool.values.dtype, np.float64)
|
||||
self.assertEqual(spool.sample_count, 3)
|
||||
self.assertEqual(spool.clipped_values, 1)
|
||||
info = write_pcm_wav(wav, spool.values, "float32")
|
||||
self.assertEqual(info["sample_count"], 3)
|
||||
self.assertEqual(info["channel_count"], 2)
|
||||
spool.close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user