Compare commits
8 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| afef6d2c52 | |||
| ab7e815a4d | |||
| b634326b5d | |||
| cf50beafcb | |||
| c8676d7aa3 | |||
| b619dee523 | |||
| 7da3a5eb95 | |||
| 128286153e |
@@ -1,99 +1,99 @@
|
||||
name: Native builds
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
tags:
|
||||
- "v*"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: ${{ matrix.asset }}
|
||||
runs-on: ${{ matrix.runner }}
|
||||
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- runner: windows-2022
|
||||
asset: windows-x64
|
||||
cmake_args: -A x64
|
||||
|
||||
- runner: ubuntu-22.04
|
||||
asset: linux-x64
|
||||
cmake_args: ""
|
||||
|
||||
- runner: macos-15-intel
|
||||
asset: macos-x64
|
||||
cmake_args: -DCMAKE_OSX_DEPLOYMENT_TARGET=12.0
|
||||
|
||||
- runner: macos-15
|
||||
asset: macos-arm64
|
||||
cmake_args: -DCMAKE_OSX_DEPLOYMENT_TARGET=12.0
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Configure
|
||||
run: >
|
||||
cmake
|
||||
-S native
|
||||
-B build/native
|
||||
-DCMAKE_BUILD_TYPE=Release
|
||||
-DCMAKE_INSTALL_PREFIX="${{ github.workspace }}/stage"
|
||||
${{ matrix.cmake_args }}
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build/native --config Release --parallel
|
||||
|
||||
- name: Install
|
||||
run: cmake --install build/native --config Release
|
||||
|
||||
- name: Package
|
||||
working-directory: stage
|
||||
run: >
|
||||
cmake -E tar
|
||||
cf "../JustOneCacophony-native-${{ matrix.asset }}.zip"
|
||||
--format=zip
|
||||
-- .
|
||||
|
||||
- name: Upload workflow artifact
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: JustOneCacophony-native-${{ matrix.asset }}
|
||||
path: JustOneCacophony-native-${{ matrix.asset }}.zip
|
||||
if-no-files-found: error
|
||||
retention-days: 14
|
||||
|
||||
release:
|
||||
name: Publish GitHub Release
|
||||
if: startsWith(github.ref, 'refs/tags/v')
|
||||
needs: build
|
||||
runs-on: ubuntu-24.04
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
steps:
|
||||
- name: Download native packages
|
||||
uses: actions/download-artifact@v5
|
||||
with:
|
||||
pattern: JustOneCacophony-native-*
|
||||
path: dist
|
||||
merge-multiple: true
|
||||
|
||||
- name: Create release
|
||||
run: >
|
||||
gh release create "$GITHUB_REF_NAME"
|
||||
dist/*.zip
|
||||
--verify-tag
|
||||
--generate-notes
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
name: Native builds
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
tags:
|
||||
- "v*"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: ${{ matrix.asset }}
|
||||
runs-on: ${{ matrix.runner }}
|
||||
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- runner: windows-2022
|
||||
asset: windows-x64
|
||||
cmake_args: -A x64
|
||||
|
||||
- runner: ubuntu-22.04
|
||||
asset: linux-x64
|
||||
cmake_args: ""
|
||||
|
||||
- runner: macos-15-intel
|
||||
asset: macos-x64
|
||||
cmake_args: -DCMAKE_OSX_DEPLOYMENT_TARGET=12.0
|
||||
|
||||
- runner: macos-15
|
||||
asset: macos-arm64
|
||||
cmake_args: -DCMAKE_OSX_DEPLOYMENT_TARGET=12.0
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Configure
|
||||
run: >
|
||||
cmake
|
||||
-S native
|
||||
-B build/native
|
||||
-DCMAKE_BUILD_TYPE=Release
|
||||
-DCMAKE_INSTALL_PREFIX="${{ github.workspace }}/stage"
|
||||
${{ matrix.cmake_args }}
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build/native --config Release --parallel
|
||||
|
||||
- name: Install
|
||||
run: cmake --install build/native --config Release
|
||||
|
||||
- name: Package
|
||||
working-directory: stage
|
||||
run: >
|
||||
cmake -E tar
|
||||
cf "../JustOneCacophony-native-${{ matrix.asset }}.zip"
|
||||
--format=zip
|
||||
-- .
|
||||
|
||||
- name: Upload workflow artifact
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: JustOneCacophony-native-${{ matrix.asset }}
|
||||
path: JustOneCacophony-native-${{ matrix.asset }}.zip
|
||||
if-no-files-found: error
|
||||
retention-days: 14
|
||||
|
||||
release:
|
||||
name: Publish GitHub Release
|
||||
if: startsWith(github.ref, 'refs/tags/v')
|
||||
needs: build
|
||||
runs-on: ubuntu-24.04
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
steps:
|
||||
- name: Download native packages
|
||||
uses: actions/download-artifact@v5
|
||||
with:
|
||||
pattern: JustOneCacophony-native-*
|
||||
path: dist
|
||||
merge-multiple: true
|
||||
|
||||
- name: Create release
|
||||
run: >
|
||||
gh release create "$GITHUB_REF_NAME"
|
||||
dist/*.zip
|
||||
--verify-tag
|
||||
--generate-notes
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
|
||||
@@ -18,11 +18,6 @@ metadata_cache/
|
||||
|
||||
HRTF/
|
||||
|
||||
# User HRTF data and compiled caches are never committed:
|
||||
# SOFA/measurement data (conventionally under HRTF/), Dolby personalization
|
||||
# scan models, and rebuilt-from-SOFA .jochrtf caches.
|
||||
*.sofa
|
||||
*.personalized_headphone
|
||||
*.jochrtf
|
||||
|
||||
# 测试与研究内容一律不入库(tests/ 全部忽略,无白名单)。
|
||||
|
||||
+35
-2
@@ -44,7 +44,7 @@ The Python and C++ backends follow the same mathematics for JOC object reconstru
|
||||
- NumPy 1.24+
|
||||
- h5py 3.8+
|
||||
- SciPy 1.10+
|
||||
- A standalone FFmpeg executable; `ffmpeg-python` is not required. FFmpeg is discovered through `PATH` by default or selected with `--ffmpeg`
|
||||
- A standalone FFmpeg executable; `ffmpeg-python` is not required. FFmpeg is discovered through `PATH` by default or selected with `--ffmpeg`. On startup the decoder options are probed with `ffmpeg -h decoder=eac3`: a missing E-AC-3 decoder or `-drc_scale` is a hard error, while a missing `-target_level` only fails when `--eac3-target-level` is used
|
||||
- Optional: CMake and a C++20 toolchain to build the native core
|
||||
|
||||
Install the Python dependency in a project-specific environment:
|
||||
@@ -146,6 +146,40 @@ python main.py input.m4a --metadata-cache metadata_cache
|
||||
python main.py input.m4a --metadata-dir metadata_cache
|
||||
```
|
||||
|
||||
### E-AC-3 decode-side dynamic range and level
|
||||
|
||||
By default FFmpeg applies the stream `dynrng` dynamic range compression when
|
||||
decoding E-AC-3 (`-drc_scale 1`). The core 5.1 PCM is the input of JOC object
|
||||
reconstruction, and `dynrng` is playback-time gain, so it is inherited linearly
|
||||
by every object and every output (ADM, speaker, binaural). This tool therefore
|
||||
decodes at **full dynamic range** by default:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a # default: -drc_scale 0, full range
|
||||
python main.py input.m4a --eac3-drc-scale 1 # reproduce consumer playback
|
||||
python main.py input.m4a --eac3-drc-scale 0.5 # apply half of it
|
||||
python main.py input.m4a --eac3-target-level -27 # dialnorm-referenced level
|
||||
```
|
||||
|
||||
- `--eac3-drc-scale` (`0`–`6`, default `0`) maps to FFmpeg `-drc_scale`: the gain
|
||||
of each E-AC-3 block is `dynrng factor ^ value`. `0` disables DRC, `1` is the
|
||||
author's intent, and `>1` is asymmetric (loud parts fully compressed, quiet
|
||||
parts enhanced).
|
||||
- `--eac3-target-level` (`-31`–`0`, default `0` = off) maps to FFmpeg
|
||||
`-target_level`: a static per-frame gain of about `target_level - dialnorm` dB,
|
||||
independent of and stackable with `--eac3-drc-scale`. dialnorm is a per-stream
|
||||
property (measured Apple Music Atmos streams are about `-18` to `-19` dB, so
|
||||
`-27` is roughly `8`–`9` dB of attenuation).
|
||||
- The level change is expected: compared with the FFmpeg default, measured
|
||||
tracks move by `0` to `-2.15` dB peak and `0` to `-1.69` dB RMS (direction
|
||||
depends on the stream `dynrng`), so `output_clip.peak` and the PCM24 clipping
|
||||
decision in `.report.json` change accordingly.
|
||||
- `--gain-db` is a static gain applied **after** reconstruction (float64 on the
|
||||
binaural path) and is not the same thing as decode-side DRC, which is
|
||||
block-varying; do not use `--gain-db` to cancel it.
|
||||
- The `ffmpeg` field of `.report.json` records the FFmpeg version and the decode
|
||||
options that were actually passed (`version`, `eac3_decode_options`).
|
||||
|
||||
### Binaural render mode
|
||||
|
||||
`--binaural-mode off|near|mid|far` selects the binaural render mode; the default
|
||||
@@ -254,7 +288,6 @@ See the [mathematical notes](docs/math.en.md) for the equations used by the deco
|
||||
## Known limitations
|
||||
|
||||
- Only the common contiguous EMDF transport is covered. Fragmented transport across multiple audio-block skip fields is not covered.
|
||||
- Dense JOC is the main path. The Sparse JOC branch should not be treated as supported.
|
||||
- The speaker and SOFA binaural paths currently cover ordinary point objects; extent, spread, diffuse, divergence, channel lock, and similar controls are outside the supported scope.
|
||||
- OAMD trim elements are boundary-checked and skipped; warp, balance, and trim parameters are not applied to raw object trajectories or speaker rendering.
|
||||
- Multi-data-point streams, uncommon band configurations, and unusual OAMD scheduling have less coverage than common 12-band, single-data-point material.
|
||||
|
||||
@@ -46,7 +46,7 @@ OAMD 时间轴和命令行逻辑在 Python 中。
|
||||
- NumPy 1.24+
|
||||
- h5py 3.8+
|
||||
- SciPy 1.10+
|
||||
- 独立的 FFmpeg 可执行程序;不需要 `ffmpeg-python`。默认从 `PATH` 查找,也可通过 `--ffmpeg` 指定可执行文件路径
|
||||
- 独立的 FFmpeg 可执行程序;不需要 `ffmpeg-python`。默认从 `PATH` 查找,也可通过 `--ffmpeg` 指定可执行文件路径。启动时会探测 `ffmpeg -h decoder=eac3`:缺 E-AC-3 解码器或 `-drc_scale` 直接报错,缺 `-target_level` 只在使用 `--eac3-target-level` 时报错
|
||||
- 可选:支持 C++20 的 CMake 工具链,用于自行构建原生核
|
||||
|
||||
建议在项目专用虚拟环境中安装依赖:
|
||||
@@ -140,6 +140,23 @@ python main.py input.m4a --metadata-cache metadata_cache
|
||||
python main.py input.m4a --metadata-dir metadata_cache
|
||||
```
|
||||
|
||||
### E-AC-3 解码级动态范围与电平
|
||||
|
||||
FFmpeg 解码 E-AC-3 时默认施加码流 `dynrng` 动态范围压缩(`-drc_scale 1`)。核心 5.1 PCM 是 JOC 对象重建的输入,而 `dynrng` 属于回放期增益,会被线性继承到全部对象与成品(ADM/扬声器/双耳),因此本工具默认按**全动态范围**解码:
|
||||
|
||||
```powershell
|
||||
python main.py input.m4a # 默认:-drc_scale 0,全动态范围
|
||||
python main.py input.m4a --eac3-drc-scale 1 # 复现消费者回放(码流作者意图)
|
||||
python main.py input.m4a --eac3-drc-scale 0.5 # 施加一半
|
||||
python main.py input.m4a --eac3-target-level -27 # 按码流 dialnorm 归一化电平
|
||||
```
|
||||
|
||||
- `--eac3-drc-scale`(`0`~`6`,默认 `0`)对应 FFmpeg 的 `-drc_scale`:每个 E-AC-3 block 的增益为 `dynrng 因子 ^ 该值`。`0` 关闭 DRC;`1` 为码流作者意图;`>1` 非对称(响处全压、轻处增强)。
|
||||
- `--eac3-target-level`(`-31`~`0`,默认 `0` 不施加)对应 FFmpeg 的 `-target_level`:按每帧 dialnorm 施加静态增益,约 `target_level - dialnorm` dB,与 `--eac3-drc-scale` 相互独立、可叠加。dialnorm 是逐码流属性(实测 Apple Music Atmos 流约 `-18`~`-19` dB,故 `-27` 约等于衰减 `8`~`9` dB)。
|
||||
- 电平变化是预期的:与 FFmpeg 默认值相比,实测曲目峰值变化 `0`~`-2.15` dB、RMS `0`~`-1.69` dB(方向取决于码流 `dynrng`),`.report.json` 的 `output_clip.peak` 与 int24 削波判定会随之变化。
|
||||
- `--gain-db` 是**重建之后**的静态增益(双耳路径 float64),与解码级 DRC 不是一回事;解码级 DRC 是按 block 时变的,不要用 `--gain-db` 去抵消它。
|
||||
- `.report.json` 的 `ffmpeg` 字段记录 FFmpeg 版本与实际下发的解码选项(`version`、`eac3_decode_options`)。
|
||||
|
||||
### 双耳渲染模式
|
||||
|
||||
`--binaural-mode off|near|mid|far` 选择双耳渲染模式,默认 `mid`,两种输出共用这一个选项:
|
||||
@@ -241,7 +258,6 @@ JustOneCacophony/
|
||||
## 已知限制
|
||||
|
||||
- 当前只覆盖常见 continuous EMDF transport;跨多个 audio-block skip field 的碎片化 transport 尚未覆盖。
|
||||
- Dense JOC 是当前主要路径;Sparse JOC 分支不应视为受支持能力。
|
||||
- 扬声器与 SOFA 双耳路径当前只覆盖普通点对象;extent、spread、diffuse、divergence、channel lock 等对象控制不在支持范围内。
|
||||
- OAMD trim element 会按声明边界校验并跳过;warp、balance 和 trim 参数不应用于当前原始对象轨迹或扬声器渲染。
|
||||
- 多数据点、少见参数带配置和特殊 OAMD 调度的覆盖度低于常见 12-band、单数据点素材。
|
||||
|
||||
@@ -7,9 +7,10 @@
|
||||
|
||||
64-QMF → 77-hybrid 结构与 13-tap 低带 prototype 定义于
|
||||
[3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf)
|
||||
第 5.2.2 节(Table 1 的 $Q=8$/$Q=4$ 系数,delay 6):
|
||||
第 5.2.2 节(Table 1 的 $Q=8$/
|
||||
$Q=4$ 系数,delay 6):
|
||||
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\!\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr)$$
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr)$$
|
||||
|
||||
64-band QMF analysis 即 ISO/IEC 14496-3/AMD1:2003 第 4.B.18.2 节的 MPEG-4
|
||||
AAC/SBR 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap prototype 的
|
||||
@@ -18,14 +19,15 @@ AAC/SBR 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap protot
|
||||
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t}$$
|
||||
|
||||
QMF synthesis 表为 analysis 多相矩阵 $\mathbf{A}$ 的因果左逆
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$($\mathbf{P}$ 为 577-sample 延迟置换;
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$(
|
||||
$\mathbf{P}$ 为 577-sample 延迟置换;
|
||||
全链 $961 = 577 + 6\times64$),rank-4 分解存储:
|
||||
|
||||
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
|
||||
|
||||
hybrid synthesis 表为 77→64 重组:高频带恒等 $Y_{3+b}=X_{16+b}$,低频带:
|
||||
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\operatorname{Re}X_q + j\,s_q\,\operatorname{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\mathrm{Re}X_q + j\,s_q\,\mathrm{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
|
||||
相同数值可在 FFmpeg(`aacps_tablegen.h`、`aacsbrdata.h`)等公开实现中查到。
|
||||
|
||||
|
||||
+15
-8
@@ -109,8 +109,12 @@ The strict importer currently accepts:
|
||||
- an explicitly free-field/anechoic `RoomType`.
|
||||
|
||||
Receiver order comes from geometry, never from the receiver array index. SOFA
|
||||
listener coordinates are $+X$ front, $+Y$ left, $+Z$ up; ADM coordinates are
|
||||
$+X$ right, $+Y$ front, $+Z$ up:
|
||||
listener coordinates are $+X$ front,
|
||||
$+Y$ left,
|
||||
$+Z$ up; ADM coordinates are
|
||||
$+X$ right,
|
||||
$+Y$ front,
|
||||
$+Z$ up:
|
||||
|
||||
$$\bigl(x_{\mathrm{SOFA}},\ y_{\mathrm{SOFA}},\ z_{\mathrm{SOFA}}\bigr) = \bigl(y_{\mathrm{ADM}},\ -x_{\mathrm{ADM}},\ z_{\mathrm{ADM}}\bigr)$$
|
||||
|
||||
@@ -159,9 +163,10 @@ The fixed resource is `data/rosella_kernels.npz`, which implements publicly
|
||||
standardized filter banks, computable from the following formulas.
|
||||
|
||||
The hybrid analysis kernels are defined in [3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf),
|
||||
Section 5.2.2 (Table 1 $Q=8$/$Q=4$ coefficients, delay 6):
|
||||
Section 5.2.2 (Table 1 $Q=8$/
|
||||
$Q=4$ coefficients, delay 6):
|
||||
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\!\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr),\qquad n=0,\dots,12$$
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr),\qquad n=0,\dots,12$$
|
||||
|
||||
The QMF analysis table is the MPEG-4 AAC/SBR 64 complex QMF bank of
|
||||
ISO/IEC 14496-3/AMD1:2003, subclause 4.B.18.2, stored as the polyphase
|
||||
@@ -170,17 +175,19 @@ reordering of the public 640-tap prototype $c_0,\dots,c_{639}$:
|
||||
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t},\qquad r=0,\dots,63,\ t=0,\dots,9$$
|
||||
|
||||
The QMF synthesis table is the causal left inverse of the analysis polyphase
|
||||
matrix $\mathbf{A}$, i.e. the solution of $\mathbf{A}\,\mathbf{W}=\mathbf{P}$
|
||||
matrix $\mathbf{A}$, i.e. the solution of
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$
|
||||
($\mathbf{P}$ is the 577-sample delay permutation; total latency
|
||||
$961 = 577 + 6\times64$), stored as a rank-4 factorization:
|
||||
|
||||
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
|
||||
|
||||
The hybrid synthesis table is the 77→64 recombination: identity for the high
|
||||
bands, $Y_{3+b}=X_{16+b}$, and for the low bands ($C_p$ is the $8+4+4$ child
|
||||
partition):
|
||||
bands, $Y_{3+b}=X_{16+b}$, and for the low bands(
|
||||
$C_p$ is the
|
||||
$8+4+4$ child partition):
|
||||
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\operatorname{Re}X_q + j\,s_q\,\operatorname{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\mathrm{Re}X_q + j\,s_q\,\mathrm{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
|
||||
The loader verifies the archive and every array by SHA-256; the table version,
|
||||
all array hashes, and the 77 reference band-center values are part of the cache
|
||||
|
||||
+14
-7
@@ -99,7 +99,11 @@ cached = SofaBinauralBackend.from_compiled_cache(
|
||||
- 明确的 free-field/anechoic `RoomType`。
|
||||
|
||||
receiver 左右顺序由几何决定,不能假定 `Data.IR` 的 receiver index。SOFA listener
|
||||
坐标为 $+X$ front、$+Y$ left、$+Z$ up;ADM 坐标为 $+X$ right、$+Y$ front、
|
||||
坐标为 $+X$ front、
|
||||
$+Y$ left、
|
||||
$+Z$ up;ADM 坐标为
|
||||
$+X$ right、
|
||||
$+Y$ front、
|
||||
$+Z$ up,转换为:
|
||||
|
||||
$$\bigl(x_{\mathrm{SOFA}},\ y_{\mathrm{SOFA}},\ z_{\mathrm{SOFA}}\bigr) = \bigl(y_{\mathrm{ADM}},\ -x_{\mathrm{ADM}},\ z_{\mathrm{ADM}}\bigr)$$
|
||||
@@ -144,9 +148,10 @@ ridge 为 `1e-3`,SH ridge 为 `1e-5`。同方向 measurement 先合并,再
|
||||
公式计算。
|
||||
|
||||
hybrid 分析核定义于 [3GPP TS 26.405 / ETSI TS 126 405](https://www.etsi.org/deliver/etsi_ts/126400_126499/126405/06.00.00_60/ts_126405v060000p.pdf)
|
||||
第 5.2.2 节(Table 1 的 $Q=8$/$Q=4$ 系数,delay 6):
|
||||
第 5.2.2 节(Table 1 的 $Q=8$/
|
||||
$Q=4$ 系数,delay 6):
|
||||
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\!\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr),\qquad n=0,\dots,12$$
|
||||
$$G_q^p[n] = g^p[n]\cdot\exp\Bigl(j\,\frac{2\pi}{Q^p}\bigl(q+\tfrac12\bigr)(n-6)\Bigr),\qquad n=0,\dots,12$$
|
||||
|
||||
QMF analysis 表即 MPEG-4 AAC/SBR(ISO/IEC 14496-3/AMD1:2003 第 4.B.18.2 节)
|
||||
的 64 complex QMF bank;打包的 $64\times10$ 表是公开 640-tap prototype
|
||||
@@ -155,15 +160,17 @@ $c_0,\dots,c_{639}$ 的多相重排:
|
||||
$$A_{r,t} = \frac{(-1)^t}{128}\,c_{63-r+64t},\qquad r=0,\dots,63,\ t=0,\dots,9$$
|
||||
|
||||
QMF synthesis 表为上述 analysis 多相矩阵 $\mathbf{A}$ 的因果左逆,即求解
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$($\mathbf{P}$ 为 577-sample 延迟置换;
|
||||
$\mathbf{A}\,\mathbf{W}=\mathbf{P}$(
|
||||
$\mathbf{P}$ 为 577-sample 延迟置换;
|
||||
全链 $961 = 577 + 6\times64$),以 rank-4 分解形式存储:
|
||||
|
||||
$$W_{b,l} = \sum_{r=1}^{4} t_{b,l,r}\,\mathbf{b}_{b,r}^{\top}$$
|
||||
|
||||
hybrid synthesis 表为 77→64 重组:高频带恒等 $Y_{3+b}=X_{16+b}$;低频带
|
||||
($C_p$ 为 $8+4+4$ 子带划分):
|
||||
hybrid synthesis 表为 77→64 重组:高频带恒等 $Y_{3+b}=X_{16+b}$;低频带(
|
||||
$C_p$ 为
|
||||
$8+4+4$ 子带划分):
|
||||
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\operatorname{Re}X_q + j\,s_q\,\operatorname{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
$$Y_p = \sum_{q\in C_p}\Bigl(\mathrm{Re}X_q + j\,s_q\,\mathrm{Im}X_q\Bigr),\qquad s_q\in\{\pm1\}$$
|
||||
|
||||
loader 校验 archive 和每个数组的 SHA-256;table version、所有数组 hash 与
|
||||
77 个 band-center 参考值都属于 cache key。标准可公开获取不等于获准实施相关
|
||||
|
||||
+97
-78
@@ -4,7 +4,7 @@
|
||||
|
||||
This document covers only the signal model and formulas used in the JustOneCacophony research path: how JOC parameters combine with core PCM to reconstruct object signals, and how OAMD coordinates become speaker gains.
|
||||
|
||||
The formulas describe the dense-JOC and ordinary point-object paths studied by the project. They are not a complete definition of every E-AC-3 JOC variant.
|
||||
The formulas describe the JOC matrix parameters (both the dense and the sparse differential syntax) and the ordinary point-object paths studied by the project. They are not a complete definition of every E-AC-3 JOC variant.
|
||||
|
||||
## 1. Overall path and notation
|
||||
|
||||
@@ -52,9 +52,11 @@ $$
|
||||
N_f=1536=24\times64.
|
||||
$$
|
||||
|
||||
## 2. Dense-JOC matrix parameters
|
||||
## 2. JOC matrix parameters
|
||||
|
||||
### 2.1 Differential reconstruction
|
||||
For every object and data point, the quantized matrix `joc_mix_mtx_q` is defined on $N_q$ quantization levels. The `b_joc_sparse` flag selects one of two differential syntaxes: dense sends one MTX difference per core channel, while sparse sends one active channel plus one coefficient difference per parameter band.
|
||||
|
||||
### 2.1 Dense differential reconstruction
|
||||
|
||||
Let `quant_idx` be $q_i\in\{0,1\}$. The number of quantization levels is
|
||||
|
||||
@@ -75,38 +77,76 @@ $$
|
||||
For object $o$, data point $d$, core channel $c$, and parameter band $p$, the coded difference $\Delta_{o,d,c,p}$ reconstructs to
|
||||
|
||||
$$
|
||||
Q_{o,d,c,0}
|
||||
=
|
||||
Q_{o,d,c,0}=
|
||||
\left(O_q+\Delta_{o,d,c,0}\right)\bmod N_q,
|
||||
$$
|
||||
|
||||
$$
|
||||
Q_{o,d,c,p}
|
||||
=
|
||||
Q_{o,d,c,p}=
|
||||
\left(Q_{o,d,c,p-1}+\Delta_{o,d,c,p}\right)\bmod N_q,
|
||||
\qquad p>0.
|
||||
$$
|
||||
|
||||
### 2.2 Dequantization
|
||||
### 2.2 Sparse differential reconstruction
|
||||
|
||||
Let $I_{o,d,p}$ be the `joc_channel_idx` symbol (IDX), $V_{o,d,p}$ the `joc_vec` symbol (VEC), and $N_c\in\{5,7\}$ the number of core channels. Each parameter band has exactly one active channel:
|
||||
|
||||
$$
|
||||
A_{o,d,p}=
|
||||
\begin{cases}
|
||||
I_{o,d,0}, & p=0,\\[2pt]
|
||||
\left(A_{o,d,p-1}+I_{o,d,p}\right)\bmod N_c, & p>0,
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
where $I_{o,d,0}$ is a 3-bit absolute channel index and every later IDX symbol is an increment relative to the previous **active channel**. The coefficient is a single accumulator running across parameter bands:
|
||||
|
||||
$$
|
||||
\kappa_{o,d,-1}=O^{(s)}_q,\qquad
|
||||
\kappa_{o,d,p}=
|
||||
\left(\kappa_{o,d,p-1}+V_{o,d,p}\right)\bmod N_q,
|
||||
$$
|
||||
|
||||
with a sparse starting point two quantization levels above the dense center offset:
|
||||
|
||||
$$
|
||||
O^{(s)}_q=
|
||||
\begin{cases}
|
||||
50, & q_i=0,\\
|
||||
100, & q_i=1.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
The accumulator is **not** reset when the active channel changes. The complete matrix is
|
||||
|
||||
$$
|
||||
Q_{o,d,c,p}=
|
||||
\begin{cases}
|
||||
\kappa_{o,d,p}, & c=A_{o,d,p},\\[2pt]
|
||||
\dfrac{N_q}{2}, & c\neq A_{o,d,p}.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
Non-active entries take $N_q/2$, which dequantizes to exactly 0.
|
||||
|
||||
### 2.3 Dequantization
|
||||
|
||||
The dequantized matrix coefficient is
|
||||
|
||||
$$
|
||||
D_{o,d,c,p}
|
||||
=
|
||||
D_{o,d,c,p}=
|
||||
\left(Q_{o,d,c,p}-\frac{N_q}{2}\right)
|
||||
\frac{820}{4096(1+q_i)}.
|
||||
$$
|
||||
|
||||
The effective denominator is therefore 4096 in coarse mode and 8192 in fine mode.
|
||||
|
||||
### 2.3 JOC clipgain
|
||||
### 2.4 JOC clipgain
|
||||
|
||||
If the clipgain field consists of integer $x$ and mantissa $y$, then
|
||||
|
||||
$$
|
||||
G_{\mathrm{clip}}
|
||||
=
|
||||
G_{\mathrm{clip}}=
|
||||
1+\frac{y}{32}2^{x-4}.
|
||||
$$
|
||||
|
||||
@@ -152,8 +192,7 @@ $$
|
||||
$$
|
||||
|
||||
$$
|
||||
M_{o,c,b,t}
|
||||
=
|
||||
M_{o,c,b,t}=
|
||||
(1-\alpha_t)P_{o,c,b}
|
||||
+\alpha_tD_{o,c,p(b)}.
|
||||
$$
|
||||
@@ -181,9 +220,8 @@ $$
|
||||
Let $\mathcal A_b$ denote the 64-band analysis-QMF operator with polyphase history state. Then
|
||||
|
||||
$$
|
||||
X_{c,b,t}
|
||||
=
|
||||
\mathcal A_b\!\left(
|
||||
X_{c,b,t}=
|
||||
\mathcal A_b\left(
|
||||
\widetilde x_c[64t],\ldots,\widetilde x_c[64t+63];
|
||||
\mathbf s^{\mathrm A}_{c,t}
|
||||
\right).
|
||||
@@ -210,8 +248,7 @@ $$
|
||||
Band 0 of each surround channel additionally passes through a 21-tap complex FIR:
|
||||
|
||||
$$
|
||||
\widehat X_{c,0,t}
|
||||
=
|
||||
\widehat X_{c,0,t}=
|
||||
\sum_{k=0}^{20}h_kX_{c,0,t-k}.
|
||||
$$
|
||||
|
||||
@@ -222,8 +259,7 @@ These delays and filter histories are decoder state and cannot be reset independ
|
||||
For each object $o$, subband $b$, and slot $t$, the object's frequency-domain value is a linear combination of the five core channels:
|
||||
|
||||
$$
|
||||
Z_{o,b,t}
|
||||
=
|
||||
Z_{o,b,t}=
|
||||
\sum_{c=0}^{4}
|
||||
M_{o,c,b,t}\widehat X_{c,b,t}.
|
||||
$$
|
||||
@@ -238,21 +274,20 @@ Write the 64 complex subbands as 128 interleaved real values in `src`. For $k=0\
|
||||
|
||||
$$
|
||||
\begin{aligned}
|
||||
\operatorname{zone}[2k] &= \operatorname{src}[4k],\\
|
||||
\operatorname{zone}[2k+1] &= -\operatorname{src}[4k+1],\\
|
||||
\operatorname{zone}[126-2k] &= \operatorname{src}[4k+2],\\
|
||||
\operatorname{zone}[127-2k] &= \operatorname{src}[4k+3].
|
||||
\mathrm{zone}[2k] &= \mathrm{src}[4k],\\
|
||||
\mathrm{zone}[2k+1] &= -\mathrm{src}[4k+1],\\
|
||||
\mathrm{zone}[126-2k] &= \mathrm{src}[4k+2],\\
|
||||
\mathrm{zone}[127-2k] &= \mathrm{src}[4k+3].
|
||||
\end{aligned}
|
||||
$$
|
||||
|
||||
Treat `zone` as 64 complex values and apply an unnormalized 64-point FFT:
|
||||
|
||||
$$
|
||||
F_k
|
||||
=
|
||||
F_k=
|
||||
\sum_{n=0}^{63}
|
||||
\operatorname{zone}_n
|
||||
\exp\!\left(-j\frac{2\pi kn}{64}\right).
|
||||
\mathrm{zone}_n
|
||||
\exp\left(-j\frac{2\pi kn}{64}\right).
|
||||
$$
|
||||
|
||||
### 7.2 Modulation and synthesis
|
||||
@@ -260,8 +295,7 @@ $$
|
||||
Define the rotation coefficient
|
||||
|
||||
$$
|
||||
r_k
|
||||
=
|
||||
r_k=
|
||||
\frac12\left(
|
||||
\sin\frac{\pi k}{128}
|
||||
+j\cos\frac{\pi k}{128}
|
||||
@@ -277,9 +311,8 @@ $$
|
||||
Let $\mathcal S$ denote polyphase synthesis with a 640-value synthesis window and cross-slot state:
|
||||
|
||||
$$
|
||||
\mathbf y_{o,t}
|
||||
=
|
||||
\mathcal S\!\left(
|
||||
\mathbf y_{o,t}=
|
||||
\mathcal S\left(
|
||||
\mathbf R_{o,t},W,\mathbf s^{\mathrm S}_{o,t}
|
||||
\right).
|
||||
$$
|
||||
@@ -287,9 +320,8 @@ $$
|
||||
Object output is
|
||||
|
||||
$$
|
||||
y_o[64t+r]
|
||||
=
|
||||
\operatorname{clip}\!\left(
|
||||
y_o[64t+r]=
|
||||
\mathrm{clip}\left(
|
||||
16\,\mathbf y_{o,t}[r],-1,1
|
||||
\right)G_{\mathrm{clip}},
|
||||
$$
|
||||
@@ -301,9 +333,8 @@ where $r=0\ldots63$. Synthesis state must advance continuously by slot.
|
||||
LFE bypasses the object matrix and inverse QMF and uses a 1217-sample delay. After the input and output scale factors cancel:
|
||||
|
||||
$$
|
||||
y_{\mathrm{LFE}}[n]
|
||||
=
|
||||
\operatorname{clip}\!\left(
|
||||
y_{\mathrm{LFE}}[n]=
|
||||
\mathrm{clip}\left(
|
||||
x_{\mathrm{LFE,core}}[n-1217],-1,1
|
||||
\right).
|
||||
$$
|
||||
@@ -313,9 +344,8 @@ $$
|
||||
The lateral and longitudinal grids use $N=62$; the height grid uses $N=15$. The quantizer is
|
||||
|
||||
$$
|
||||
q_N(k)
|
||||
=
|
||||
\min\!\left(
|
||||
q_N(k)=
|
||||
\min\left(
|
||||
32767,
|
||||
\left\lfloor\frac{32768k}{N}+\frac12\right\rfloor
|
||||
\right).
|
||||
@@ -336,11 +366,11 @@ Their maximum runtime value is $32767/32768$, not exactly 1.
|
||||
For conversion to the ADM grid:
|
||||
|
||||
$$
|
||||
k_1=\operatorname{round}\!\left(\frac{62q_1}{32767}\right),
|
||||
k_1=\mathrm{round}\left(\frac{62q_1}{32767}\right),
|
||||
\quad
|
||||
k_2=\operatorname{round}\!\left(\frac{62q_2}{32767}\right),
|
||||
k_2=\mathrm{round}\left(\frac{62q_2}{32767}\right),
|
||||
\quad
|
||||
k_3=\operatorname{round}\!\left(\frac{15q_3}{32767}\right),
|
||||
k_3=\mathrm{round}\left(\frac{15q_3}{32767}\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
@@ -404,17 +434,15 @@ $$
|
||||
The two-dimensional point gain is
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{2D}}(u,v)
|
||||
=
|
||||
\mathbf G_{\mathrm{2D}}(u,v)=
|
||||
\mathbf h(u)\odot\mathbf v(v).
|
||||
$$
|
||||
|
||||
For 5.1-family layouts with one horizontal surround pair rather than separate side and rear pairs, the longitudinal coordinate is
|
||||
|
||||
$$
|
||||
v_{\mathrm{floor}}
|
||||
=
|
||||
\operatorname{clamp}(2v,0,1).
|
||||
v_{\mathrm{floor}}=
|
||||
\mathrm{clamp}(2v,0,1).
|
||||
$$
|
||||
|
||||
Other layouts use $v_{\mathrm{floor}}=v$.
|
||||
@@ -424,8 +452,7 @@ Other layouts use $v_{\mathrm{floor}}=v$.
|
||||
Three-dimensional layouts compute floor gain $\mathbf G_f$ and height gain $\mathbf G_h$ separately:
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{point}}(u,v,w)
|
||||
=
|
||||
\mathbf G_{\mathrm{point}}(u,v,w)=
|
||||
\cos\left(\frac\pi2w\right)\mathbf G_f
|
||||
+
|
||||
\sin\left(\frac\pi2w\right)\mathbf G_h.
|
||||
@@ -450,8 +477,7 @@ $$
|
||||
Maximum position compensation is
|
||||
|
||||
$$
|
||||
A_{\max}
|
||||
=
|
||||
A_{\max}=
|
||||
-\max\left(4.5-1.5H-3F,0\right)
|
||||
\quad\text{dB}.
|
||||
$$
|
||||
@@ -459,15 +485,15 @@ $$
|
||||
Longitudinal and height weights are
|
||||
|
||||
$$
|
||||
p_v=\operatorname{clamp}\left(\frac v{0.6},0,1\right),
|
||||
p_v=\mathrm{clamp}\left(\frac v{0.6},0,1\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
p_w=\operatorname{clamp}\left(\frac{w-0.2}{0.8},0,1\right),
|
||||
p_w=\mathrm{clamp}\left(\frac{w-0.2}{0.8},0,1\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
p=\operatorname{clamp}(p_v+p_w,0,1).
|
||||
p=\mathrm{clamp}(p_v+p_w,0,1).
|
||||
$$
|
||||
|
||||
The linear compensation gain is
|
||||
@@ -479,8 +505,7 @@ $$
|
||||
The object's target-gain vector is
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{target}}
|
||||
=
|
||||
\mathbf G_{\mathrm{target}}=
|
||||
G_{\mathrm{object}}
|
||||
G_{\mathrm{pos}}
|
||||
\mathbf G_{\mathrm{point}}.
|
||||
@@ -491,8 +516,7 @@ $$
|
||||
The coded position of an OAMD update is
|
||||
|
||||
$$
|
||||
s_{\mathrm{coded}}
|
||||
=
|
||||
s_{\mathrm{coded}}=
|
||||
s_{\mathrm{frame}}
|
||||
+s_{\mathrm{outer}}
|
||||
+s_{\mathrm{OAMD}}
|
||||
@@ -502,16 +526,15 @@ $$
|
||||
The theoretical update position on the decoder-output PCM timeline is
|
||||
|
||||
$$
|
||||
s_{\mathrm{theoretical}}
|
||||
=s_{\mathrm{coded}}+d_{\mathrm{decoder}},
|
||||
s_{\mathrm{theoretical}}=
|
||||
s_{\mathrm{coded}}+d_{\mathrm{decoder}},
|
||||
\qquad d_{\mathrm{decoder}}=1473.
|
||||
$$
|
||||
|
||||
The speaker renderer retains the existing processing-block length $B=32$, so the aligned update point is
|
||||
|
||||
$$
|
||||
\widehat s
|
||||
=
|
||||
\widehat s=
|
||||
B\left\lfloor
|
||||
\frac{s_{\mathrm{theoretical}}+B/2-1}{B}
|
||||
\right\rfloor.
|
||||
@@ -522,8 +545,7 @@ Thus, for frame-aligned updates, `align32(1473)=1472`. The 1473 value is the the
|
||||
For ramp duration $D$, the number of blocks is
|
||||
|
||||
$$
|
||||
K
|
||||
=
|
||||
K=
|
||||
\left\lfloor
|
||||
\frac{D+B/2-1}{B}
|
||||
\right\rfloor.
|
||||
@@ -550,8 +572,7 @@ If no new metadata update intervenes, this is equivalent to a sample-wise linear
|
||||
For target output channel $c$:
|
||||
|
||||
$$
|
||||
y_c[n]
|
||||
=
|
||||
y_c[n]=
|
||||
\delta_{c,\mathrm{LFE}}x_{\mathrm{LFE}}[n]
|
||||
+
|
||||
\sum_{o=1}^{15}x_o[n]g_{o,c}[n].
|
||||
@@ -560,8 +581,7 @@ $$
|
||||
Here
|
||||
|
||||
$$
|
||||
\delta_{c,\mathrm{LFE}}
|
||||
=
|
||||
\delta_{c,\mathrm{LFE}}=
|
||||
\begin{cases}
|
||||
1, & c\text{ is the target layout's LFE channel},\\
|
||||
0, & \text{otherwise}.
|
||||
@@ -573,16 +593,15 @@ A layout without LFE output does not mix input LFE into other channels. After ob
|
||||
For PCM24 output, quantization is
|
||||
|
||||
$$
|
||||
y_{24}[n]
|
||||
=
|
||||
\operatorname{trunc}\left(
|
||||
8388607\,\operatorname{clip}(y[n],-1,1)
|
||||
y_{24}[n]=
|
||||
\mathrm{trunc}\left(
|
||||
8388607\,\mathrm{clip}(y[n],-1,1)
|
||||
\right).
|
||||
$$
|
||||
|
||||
## 14. Scope of the formulas
|
||||
|
||||
- The JOC matrix section describes dense JOC; Sparse JOC uses a different sparse coefficient/index path.
|
||||
- The JOC matrix section covers both the dense MTX and the sparse IDX/VEC differential syntax.
|
||||
- The speaker-panning section describes ordinary point objects; extent, spread, divergence, and similar modes require additional models.
|
||||
- Multiple OAMD position blocks must be scheduled in time order.
|
||||
- A limiter is separate post-processing and is not included in the mixing equations above.
|
||||
|
||||
+97
-78
@@ -4,7 +4,7 @@
|
||||
|
||||
本文只说明 JustOneCacophony 研究路径中使用的信号模型和公式:JOC 参数如何与核心 PCM 结合并重建对象信号,以及 OAMD 坐标如何转换为扬声器增益。
|
||||
|
||||
这些公式描述项目当前研究的 dense JOC 与普通点对象路径,不代表对所有 E-AC-3 JOC 变体的完整定义。
|
||||
这些公式描述项目当前研究的 JOC 矩阵参数(dense 与 sparse 两条差分语法)与普通点对象路径,不代表对所有 E-AC-3 JOC 变体的完整定义。
|
||||
|
||||
## 1. 总体路径与记号
|
||||
|
||||
@@ -52,9 +52,11 @@ $$
|
||||
N_f=1536=24\times64.
|
||||
$$
|
||||
|
||||
## 2. Dense JOC 矩阵参数
|
||||
## 2. JOC 矩阵参数
|
||||
|
||||
### 2.1 差分还原
|
||||
每个对象、每个数据点的量化矩阵 `joc_mix_mtx_q` 都定义在 $N_q$ 个量化级上。标志位 `b_joc_sparse` 选择两条差分语法之一:dense 为每个核心声道各送一路 MTX 差分,sparse 每参数带只送一个 active 声道与一路系数差分。
|
||||
|
||||
### 2.1 Dense 差分还原
|
||||
|
||||
令 `quant_idx` 为 $q_i\in\{0,1\}$,量化级数为
|
||||
|
||||
@@ -75,38 +77,76 @@ $$
|
||||
对对象 $o$、数据点 $d$、核心声道 $c$ 和参数带 $p$,编码差分 $\Delta_{o,d,c,p}$ 还原为
|
||||
|
||||
$$
|
||||
Q_{o,d,c,0}
|
||||
=
|
||||
Q_{o,d,c,0}=
|
||||
\left(O_q+\Delta_{o,d,c,0}\right)\bmod N_q,
|
||||
$$
|
||||
|
||||
$$
|
||||
Q_{o,d,c,p}
|
||||
=
|
||||
Q_{o,d,c,p}=
|
||||
\left(Q_{o,d,c,p-1}+\Delta_{o,d,c,p}\right)\bmod N_q,
|
||||
\qquad p>0.
|
||||
$$
|
||||
|
||||
### 2.2 去量化
|
||||
### 2.2 Sparse 差分还原
|
||||
|
||||
令 $I_{o,d,p}$ 为 `joc_channel_idx` 符号(IDX),$V_{o,d,p}$ 为 `joc_vec` 符号(VEC),$N_c\in\{5,7\}$ 为核心声道数。每参数带只有一个 active 声道
|
||||
|
||||
$$
|
||||
A_{o,d,p}=
|
||||
\begin{cases}
|
||||
I_{o,d,0}, & p=0,\\[2pt]
|
||||
\left(A_{o,d,p-1}+I_{o,d,p}\right)\bmod N_c, & p>0,
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
其中 $I_{o,d,0}$ 是 3 bit 绝对声道号,其余 IDX 符号是相对上一个 **active 声道**的增量。系数是一个跨参数带连续的单累加器
|
||||
|
||||
$$
|
||||
\kappa_{o,d,-1}=O^{(s)}_q,\qquad
|
||||
\kappa_{o,d,p}=
|
||||
\left(\kappa_{o,d,p-1}+V_{o,d,p}\right)\bmod N_q,
|
||||
$$
|
||||
|
||||
sparse 起点比 dense 的中心偏移高两个量化级:
|
||||
|
||||
$$
|
||||
O^{(s)}_q=
|
||||
\begin{cases}
|
||||
50, & q_i=0,\\
|
||||
100, & q_i=1.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
active 声道切换时累加器**不**重置。完整矩阵为
|
||||
|
||||
$$
|
||||
Q_{o,d,c,p}=
|
||||
\begin{cases}
|
||||
\kappa_{o,d,p}, & c=A_{o,d,p},\\[2pt]
|
||||
\dfrac{N_q}{2}, & c\neq A_{o,d,p}.
|
||||
\end{cases}
|
||||
$$
|
||||
|
||||
非 active 项取 $N_q/2$,即去量化后恰为 0。
|
||||
|
||||
### 2.3 去量化
|
||||
|
||||
矩阵系数的去量化值为
|
||||
|
||||
$$
|
||||
D_{o,d,c,p}
|
||||
=
|
||||
D_{o,d,c,p}=
|
||||
\left(Q_{o,d,c,p}-\frac{N_q}{2}\right)
|
||||
\frac{820}{4096(1+q_i)}.
|
||||
$$
|
||||
|
||||
因此 coarse 模式的有效分母为 4096,fine 模式为 8192。
|
||||
|
||||
### 2.3 JOC clipgain
|
||||
### 2.4 JOC clipgain
|
||||
|
||||
若 clipgain 字段由整数 $x$ 和尾数 $y$ 组成,则
|
||||
|
||||
$$
|
||||
G_{\mathrm{clip}}
|
||||
=
|
||||
G_{\mathrm{clip}}=
|
||||
1+\frac{y}{32}2^{x-4}.
|
||||
$$
|
||||
|
||||
@@ -152,8 +192,7 @@ $$
|
||||
$$
|
||||
|
||||
$$
|
||||
M_{o,c,b,t}
|
||||
=
|
||||
M_{o,c,b,t}=
|
||||
(1-\alpha_t)P_{o,c,b}
|
||||
+\alpha_tD_{o,c,p(b)}.
|
||||
$$
|
||||
@@ -181,9 +220,8 @@ $$
|
||||
令 $\mathcal A_b$ 表示带 polyphase 历史状态的 64-band analysis-QMF 算子,则
|
||||
|
||||
$$
|
||||
X_{c,b,t}
|
||||
=
|
||||
\mathcal A_b\!\left(
|
||||
X_{c,b,t}=
|
||||
\mathcal A_b\left(
|
||||
\widetilde x_c[64t],\ldots,\widetilde x_c[64t+63];
|
||||
\mathbf s^{\mathrm A}_{c,t}
|
||||
\right).
|
||||
@@ -210,8 +248,7 @@ $$
|
||||
环绕声道的 band 0 还经过 21-tap 复 FIR:
|
||||
|
||||
$$
|
||||
\widehat X_{c,0,t}
|
||||
=
|
||||
\widehat X_{c,0,t}=
|
||||
\sum_{k=0}^{20}h_kX_{c,0,t-k}.
|
||||
$$
|
||||
|
||||
@@ -222,8 +259,7 @@ $$
|
||||
对每个对象 $o$、子带 $b$ 和时槽 $t$,对象频域值为五个核心声道的线性组合:
|
||||
|
||||
$$
|
||||
Z_{o,b,t}
|
||||
=
|
||||
Z_{o,b,t}=
|
||||
\sum_{c=0}^{4}
|
||||
M_{o,c,b,t}\widehat X_{c,b,t}.
|
||||
$$
|
||||
@@ -238,21 +274,20 @@ analysis 输入的 $1/16$ 缩放会在 inverse QMF 输出端由 $\times16$ 抵
|
||||
|
||||
$$
|
||||
\begin{aligned}
|
||||
\operatorname{zone}[2k] &= \operatorname{src}[4k],\\
|
||||
\operatorname{zone}[2k+1] &= -\operatorname{src}[4k+1],\\
|
||||
\operatorname{zone}[126-2k] &= \operatorname{src}[4k+2],\\
|
||||
\operatorname{zone}[127-2k] &= \operatorname{src}[4k+3].
|
||||
\mathrm{zone}[2k] &= \mathrm{src}[4k],\\
|
||||
\mathrm{zone}[2k+1] &= -\mathrm{src}[4k+1],\\
|
||||
\mathrm{zone}[126-2k] &= \mathrm{src}[4k+2],\\
|
||||
\mathrm{zone}[127-2k] &= \mathrm{src}[4k+3].
|
||||
\end{aligned}
|
||||
$$
|
||||
|
||||
把 `zone` 重新视为 64 个复数后执行未归一化 64 点 FFT:
|
||||
|
||||
$$
|
||||
F_k
|
||||
=
|
||||
F_k=
|
||||
\sum_{n=0}^{63}
|
||||
\operatorname{zone}_n
|
||||
\exp\!\left(-j\frac{2\pi kn}{64}\right).
|
||||
\mathrm{zone}_n
|
||||
\exp\left(-j\frac{2\pi kn}{64}\right).
|
||||
$$
|
||||
|
||||
### 7.2 调制与合成
|
||||
@@ -260,8 +295,7 @@ $$
|
||||
定义旋转系数
|
||||
|
||||
$$
|
||||
r_k
|
||||
=
|
||||
r_k=
|
||||
\frac12\left(
|
||||
\sin\frac{\pi k}{128}
|
||||
+j\cos\frac{\pi k}{128}
|
||||
@@ -277,9 +311,8 @@ $$
|
||||
令 $\mathcal S$ 表示带 640 项 synthesis window 和跨时槽状态的 polyphase 合成算子:
|
||||
|
||||
$$
|
||||
\mathbf y_{o,t}
|
||||
=
|
||||
\mathcal S\!\left(
|
||||
\mathbf y_{o,t}=
|
||||
\mathcal S\left(
|
||||
\mathbf R_{o,t},W,\mathbf s^{\mathrm S}_{o,t}
|
||||
\right).
|
||||
$$
|
||||
@@ -287,9 +320,8 @@ $$
|
||||
对象输出为
|
||||
|
||||
$$
|
||||
y_o[64t+r]
|
||||
=
|
||||
\operatorname{clip}\!\left(
|
||||
y_o[64t+r]=
|
||||
\mathrm{clip}\left(
|
||||
16\,\mathbf y_{o,t}[r],-1,1
|
||||
\right)G_{\mathrm{clip}},
|
||||
$$
|
||||
@@ -301,9 +333,8 @@ $$
|
||||
LFE 不经过对象矩阵或 inverse QMF,而是使用 1217-sample 延迟。输入与输出端的比例因子抵消后:
|
||||
|
||||
$$
|
||||
y_{\mathrm{LFE}}[n]
|
||||
=
|
||||
\operatorname{clip}\!\left(
|
||||
y_{\mathrm{LFE}}[n]=
|
||||
\mathrm{clip}\left(
|
||||
x_{\mathrm{LFE,core}}[n-1217],-1,1
|
||||
\right).
|
||||
$$
|
||||
@@ -313,9 +344,8 @@ $$
|
||||
横向和纵向网格使用 $N=62$,高度网格使用 $N=15$。量化函数为
|
||||
|
||||
$$
|
||||
q_N(k)
|
||||
=
|
||||
\min\!\left(
|
||||
q_N(k)=
|
||||
\min\left(
|
||||
32767,
|
||||
\left\lfloor\frac{32768k}{N}+\frac12\right\rfloor
|
||||
\right).
|
||||
@@ -336,11 +366,11 @@ $$
|
||||
转换为 ADM 网格时:
|
||||
|
||||
$$
|
||||
k_1=\operatorname{round}\!\left(\frac{62q_1}{32767}\right),
|
||||
k_1=\mathrm{round}\left(\frac{62q_1}{32767}\right),
|
||||
\quad
|
||||
k_2=\operatorname{round}\!\left(\frac{62q_2}{32767}\right),
|
||||
k_2=\mathrm{round}\left(\frac{62q_2}{32767}\right),
|
||||
\quad
|
||||
k_3=\operatorname{round}\!\left(\frac{15q_3}{32767}\right),
|
||||
k_3=\mathrm{round}\left(\frac{15q_3}{32767}\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
@@ -404,17 +434,15 @@ $$
|
||||
二维点增益为
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{2D}}(u,v)
|
||||
=
|
||||
\mathbf G_{\mathrm{2D}}(u,v)=
|
||||
\mathbf h(u)\odot\mathbf v(v).
|
||||
$$
|
||||
|
||||
对于只有一对水平环绕、没有独立 side/rear 两对的 5.1 系列布局,纵向坐标使用
|
||||
|
||||
$$
|
||||
v_{\mathrm{floor}}
|
||||
=
|
||||
\operatorname{clamp}(2v,0,1).
|
||||
v_{\mathrm{floor}}=
|
||||
\mathrm{clamp}(2v,0,1).
|
||||
$$
|
||||
|
||||
其他布局使用 $v_{\mathrm{floor}}=v$。
|
||||
@@ -424,8 +452,7 @@ $$
|
||||
三维布局分别计算地面层增益 $\mathbf G_f$ 和高度层增益 $\mathbf G_h$:
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{point}}(u,v,w)
|
||||
=
|
||||
\mathbf G_{\mathrm{point}}(u,v,w)=
|
||||
\cos\left(\frac\pi2w\right)\mathbf G_f
|
||||
+
|
||||
\sin\left(\frac\pi2w\right)\mathbf G_h.
|
||||
@@ -450,8 +477,7 @@ $$
|
||||
最大位置补偿为
|
||||
|
||||
$$
|
||||
A_{\max}
|
||||
=
|
||||
A_{\max}=
|
||||
-\max\left(4.5-1.5H-3F,0\right)
|
||||
\quad\text{dB}.
|
||||
$$
|
||||
@@ -459,15 +485,15 @@ $$
|
||||
前后与高度位置权重为
|
||||
|
||||
$$
|
||||
p_v=\operatorname{clamp}\left(\frac v{0.6},0,1\right),
|
||||
p_v=\mathrm{clamp}\left(\frac v{0.6},0,1\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
p_w=\operatorname{clamp}\left(\frac{w-0.2}{0.8},0,1\right),
|
||||
p_w=\mathrm{clamp}\left(\frac{w-0.2}{0.8},0,1\right),
|
||||
$$
|
||||
|
||||
$$
|
||||
p=\operatorname{clamp}(p_v+p_w,0,1).
|
||||
p=\mathrm{clamp}(p_v+p_w,0,1).
|
||||
$$
|
||||
|
||||
线性补偿增益为
|
||||
@@ -479,8 +505,7 @@ $$
|
||||
对象的目标增益向量为
|
||||
|
||||
$$
|
||||
\mathbf G_{\mathrm{target}}
|
||||
=
|
||||
\mathbf G_{\mathrm{target}}=
|
||||
G_{\mathrm{object}}
|
||||
G_{\mathrm{pos}}
|
||||
\mathbf G_{\mathrm{point}}.
|
||||
@@ -491,8 +516,7 @@ $$
|
||||
OAMD 更新的编码位置为
|
||||
|
||||
$$
|
||||
s_{\mathrm{coded}}
|
||||
=
|
||||
s_{\mathrm{coded}}=
|
||||
s_{\mathrm{frame}}
|
||||
+s_{\mathrm{outer}}
|
||||
+s_{\mathrm{OAMD}}
|
||||
@@ -502,16 +526,15 @@ $$
|
||||
decoder 输出 PCM timeline 上的理论更新位置为
|
||||
|
||||
$$
|
||||
s_{\mathrm{theoretical}}
|
||||
=s_{\mathrm{coded}}+d_{\mathrm{decoder}},
|
||||
s_{\mathrm{theoretical}}=
|
||||
s_{\mathrm{coded}}+d_{\mathrm{decoder}},
|
||||
\qquad d_{\mathrm{decoder}}=1473.
|
||||
$$
|
||||
|
||||
扬声器 renderer 保留现有的处理块长度 $B=32$,更新点对齐为
|
||||
|
||||
$$
|
||||
\widehat s
|
||||
=
|
||||
\widehat s=
|
||||
B\left\lfloor
|
||||
\frac{s_{\mathrm{theoretical}}+B/2-1}{B}
|
||||
\right\rfloor.
|
||||
@@ -522,8 +545,7 @@ $$
|
||||
给定 ramp duration $D$,block 数为
|
||||
|
||||
$$
|
||||
K
|
||||
=
|
||||
K=
|
||||
\left\lfloor
|
||||
\frac{D+B/2-1}{B}
|
||||
\right\rfloor.
|
||||
@@ -550,8 +572,7 @@ $$
|
||||
对目标输出声道 $c$:
|
||||
|
||||
$$
|
||||
y_c[n]
|
||||
=
|
||||
y_c[n]=
|
||||
\delta_{c,\mathrm{LFE}}x_{\mathrm{LFE}}[n]
|
||||
+
|
||||
\sum_{o=1}^{15}x_o[n]g_{o,c}[n].
|
||||
@@ -560,8 +581,7 @@ $$
|
||||
其中
|
||||
|
||||
$$
|
||||
\delta_{c,\mathrm{LFE}}
|
||||
=
|
||||
\delta_{c,\mathrm{LFE}}=
|
||||
\begin{cases}
|
||||
1, & c\text{ 为目标布局的 LFE},\\
|
||||
0, & \text{其他声道}.
|
||||
@@ -573,16 +593,15 @@ $$
|
||||
若输出 PCM24,量化关系为
|
||||
|
||||
$$
|
||||
y_{24}[n]
|
||||
=
|
||||
\operatorname{trunc}\left(
|
||||
8388607\,\operatorname{clip}(y[n],-1,1)
|
||||
y_{24}[n]=
|
||||
\mathrm{trunc}\left(
|
||||
8388607\,\mathrm{clip}(y[n],-1,1)
|
||||
\right).
|
||||
$$
|
||||
|
||||
## 14. 公式适用范围
|
||||
|
||||
- JOC 矩阵部分描述 dense JOC;Sparse JOC 使用不同的稀疏系数/索引路径。
|
||||
- JOC 矩阵部分同时描述 dense MTX 与 sparse IDX/VEC 两条差分语法。
|
||||
- 扬声器声像部分描述普通点对象;extent、spread、divergence 等模式需要额外模型。
|
||||
- 多个 OAMD position block 必须按其时间顺序调度。
|
||||
- limiter 属于独立后处理,不包含在上述混音公式中。
|
||||
|
||||
+1
-1
@@ -42,7 +42,7 @@ int ejoc_renderer_process(
|
||||
float* output16_planar); /* [16][1536] */
|
||||
```
|
||||
|
||||
Python performs dense-JOC Huffman decoding, differential reconstruction, and dequantization before the call. Sparse JOC is not silently passed to the dense native path.
|
||||
Python performs JOC Huffman decoding, differential reconstruction, and dequantization before the call, with the dense and sparse syntaxes sharing one entry point. The native core consumes the already dequantized `dq` in double precision, and both syntaxes have the same layout at that ABI.
|
||||
|
||||
Thread control is exposed as:
|
||||
|
||||
|
||||
+1
-1
@@ -42,7 +42,7 @@ int ejoc_renderer_process(
|
||||
float* output16_planar); /* [16][1536] */
|
||||
```
|
||||
|
||||
Dense JOC 的 Huffman 解码、差分还原和去量化先在 Python 中完成。Sparse JOC 不会被静默送入 dense 原生路径。
|
||||
JOC 的 Huffman 解码、差分还原和去量化先在 Python 中完成,dense 与 sparse 两条语法共用同一条入口。原生核心消费已去量化的 `dq`(double),两条语法在该 ABI 上布局一致。
|
||||
|
||||
线程接口为:
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ import math
|
||||
import os
|
||||
from pathlib import Path
|
||||
import platform
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
@@ -50,6 +51,9 @@ from variant_error import UnsupportedVariantError, write_variant_report
|
||||
RATE = 48000
|
||||
FRAME_SAMPLES = 1536
|
||||
DEFAULT_OUTPUT_DIR = PROJECT_DIR / "output"
|
||||
EAC3_DRC_SCALE_MAX = 6.0
|
||||
EAC3_TARGET_LEVEL_RANGE = (-31, 0)
|
||||
EAC3_DECODER_OPTION_RE = re.compile(r"(?m)^\s*-([A-Za-z0-9_]+)\s+<")
|
||||
|
||||
|
||||
def resolve_output(source, requested=None, speaker_layout=None, *, binaural=False):
|
||||
@@ -183,6 +187,53 @@ def timed_call(timings, name, function, *args, **kwargs):
|
||||
timings[name] = time.perf_counter() - started
|
||||
|
||||
|
||||
def probe_eac3_decoder_options(ffmpeg):
|
||||
"""读取 ``ffmpeg -h decoder=eac3`` 暴露的 AVOption 名。"""
|
||||
result = subprocess.run(
|
||||
[ffmpeg, "-hide_banner", "-h", "decoder=eac3"],
|
||||
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||||
text=True, encoding="utf-8", errors="replace")
|
||||
options = frozenset(EAC3_DECODER_OPTION_RE.findall(result.stdout or ""))
|
||||
# decoder 名不存在时 ffmpeg 依然返回 0,因此以“解析不到任何选项”为失败。
|
||||
if not options:
|
||||
raise RuntimeError(
|
||||
"无法读取 FFmpeg 的 eac3 解码器选项(ffmpeg -h decoder=eac3);"
|
||||
"需要带 E-AC-3 解码器的构建")
|
||||
return options
|
||||
|
||||
|
||||
def ffmpeg_version(ffmpeg):
|
||||
"""FFmpeg 版本字符串;探测失败返回空串,不影响渲染。"""
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[ffmpeg, "-hide_banner", "-version"],
|
||||
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||||
text=True, encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
return ""
|
||||
lines = (result.stdout or "").splitlines()
|
||||
line = lines[0].strip() if lines else ""
|
||||
prefix = "ffmpeg version "
|
||||
return line[len(prefix):].strip() if line.startswith(prefix) else line
|
||||
|
||||
|
||||
def eac3_decode_options(drc_scale, target_level, available):
|
||||
"""构造 ``-i`` 之前的 E-AC-3 解码选项,返回 ``(argv, report 片段)``。"""
|
||||
if "drc_scale" not in available:
|
||||
raise RuntimeError(
|
||||
"FFmpeg 的 eac3 解码器缺少 -drc_scale,无法关闭码流 DRC")
|
||||
# -drc_scale 始终显式下发:0(全动态范围)不是 ffmpeg 的默认值。
|
||||
argv = ["-drc_scale", format(float(drc_scale), ".10g")]
|
||||
if target_level:
|
||||
if "target_level" not in available:
|
||||
raise RuntimeError(
|
||||
"FFmpeg 的 eac3 解码器不支持 -target_level;请升级 FFmpeg "
|
||||
"或去掉 --eac3-target-level")
|
||||
argv += ["-target_level", str(int(target_level))]
|
||||
applied = {"drc_scale": float(drc_scale), "target_level": int(target_level)}
|
||||
return argv, applied
|
||||
|
||||
|
||||
def extract_eac3(ffmpeg, source, target):
|
||||
if source.suffix.lower() in (".eac3", ".ec3"):
|
||||
return source
|
||||
@@ -192,10 +243,10 @@ def extract_eac3(ffmpeg, source, target):
|
||||
return target
|
||||
|
||||
|
||||
def decode_core(ffmpeg, eac3, target, duration_sec=None):
|
||||
def decode_core(ffmpeg, eac3, target, duration_sec=None, *, options=()):
|
||||
# 5.1(side) 的 f32le 顺序为 FL FR FC LFE SL SR;JOC 使用其中 0,1,2,4,5。
|
||||
command = [ffmpeg, "-hide_banner", "-loglevel", "error", "-y", "-i", str(eac3),
|
||||
"-map", "0:a:0", "-vn"]
|
||||
command = [ffmpeg, "-hide_banner", "-loglevel", "error", "-y", *options,
|
||||
"-i", str(eac3), "-map", "0:a:0", "-vn"]
|
||||
if duration_sec is not None:
|
||||
command.extend(["-t", f"{duration_sec:.9f}"])
|
||||
command.extend(["-ac", "6", "-ar", str(RATE),
|
||||
@@ -480,6 +531,12 @@ def build_parser():
|
||||
parser.add_argument("--trajectory-mode", choices=("compact", "dense64"), default="compact",
|
||||
help="ADM 对象轨迹表示;直接双耳路径不序列化 AXML")
|
||||
parser.add_argument("--ffmpeg", default=os.environ.get("FFMPEG", "ffmpeg"))
|
||||
parser.add_argument("--eac3-drc-scale", type=float, default=0.0,
|
||||
help="E-AC-3 解码器 -drc_scale:0=关闭码流 dynrng(全动态范围),"
|
||||
"1=码流作者意图,>1 非对称;默认 0")
|
||||
parser.add_argument("--eac3-target-level", type=int, default=0,
|
||||
help="E-AC-3 解码器 -target_level:按码流 dialnorm 归一化电平,"
|
||||
"增益约 target_level - dialnorm dB;0=不施加,默认 0")
|
||||
parser.add_argument("--backend", choices=("auto", "native", "python"), default="auto",
|
||||
help="JOC/扬声器 DSP 后端;SOFA 双耳 DSP 当前使用 Python")
|
||||
parser.add_argument("--native-library", type=Path,
|
||||
@@ -568,9 +625,28 @@ def main(argv=None):
|
||||
gain = np.float32(gain_float64)
|
||||
if not math.isfinite(gain_float64) or not np.isfinite(gain):
|
||||
raise ValueError("gain-db 超出支持范围")
|
||||
if (not math.isfinite(args.eac3_drc_scale)
|
||||
or not 0.0 <= args.eac3_drc_scale <= EAC3_DRC_SCALE_MAX):
|
||||
raise ValueError(f"eac3-drc-scale 必须在 0..{EAC3_DRC_SCALE_MAX:g} 之间")
|
||||
if not (EAC3_TARGET_LEVEL_RANGE[0] <= args.eac3_target_level
|
||||
<= EAC3_TARGET_LEVEL_RANGE[1]):
|
||||
raise ValueError("eac3-target-level 必须在 -31..0 之间")
|
||||
binaural_hrtf_input = resolve_binaural_hrtf_input(
|
||||
args, required=binaural_mode and not args.metadata_only)
|
||||
ffmpeg = executable(args.ffmpeg, "FFmpeg")
|
||||
decode_options = ()
|
||||
decode_option_info = None
|
||||
if not args.metadata_only:
|
||||
available = probe_eac3_decoder_options(ffmpeg)
|
||||
decode_options, applied = eac3_decode_options(
|
||||
args.eac3_drc_scale, args.eac3_target_level, available)
|
||||
decode_option_info = {
|
||||
"version": ffmpeg_version(ffmpeg),
|
||||
"eac3_decode_options": applied,
|
||||
}
|
||||
print(f"[decode] ffmpeg {decode_option_info['version']} "
|
||||
f"drc_scale={args.eac3_drc_scale:g} "
|
||||
f"target_level={args.eac3_target_level}", flush=True)
|
||||
|
||||
total_started = time.perf_counter()
|
||||
timings = {}
|
||||
@@ -606,7 +682,8 @@ def main(argv=None):
|
||||
|
||||
bed_path = timed_call(
|
||||
timings, "decode_core", decode_core,
|
||||
ffmpeg, eac3, temp_dir / "core51_f32le.raw", duration_sec)
|
||||
ffmpeg, eac3, temp_dir / "core51_f32le.raw", duration_sec,
|
||||
options=decode_options)
|
||||
raw_path = (output.with_name(output.name + ".objects16.f32le")
|
||||
if args.keep_raw else None)
|
||||
master = None
|
||||
@@ -887,6 +964,7 @@ def main(argv=None):
|
||||
"sha256": output_sha,
|
||||
"python": platform.python_version(),
|
||||
"numpy": np.__version__,
|
||||
"ffmpeg": decode_option_info,
|
||||
}
|
||||
report_path = Path(str(output) + ".report.json")
|
||||
report_path.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
|
||||
@@ -53,8 +53,9 @@ Fixed array layouts used by ejoc_renderer_process():
|
||||
output16 [16][1536]
|
||||
|
||||
Only objects selected by object_mask are read from the descriptor arrays.
|
||||
Sparse JOC must be rejected by the caller; this ABI accepts already dequantized
|
||||
dense matrix coefficients.
|
||||
dq carries already dequantized matrix coefficients in double precision; the
|
||||
caller performs the JOC bitstream differential decoding for both dense and
|
||||
sparse objects, so this ABI is identical for both syntaxes.
|
||||
*/
|
||||
|
||||
EJOC_API uint32_t EJOC_CALL ejoc_abi_version(void);
|
||||
|
||||
+96
-44
@@ -2,11 +2,11 @@
|
||||
|
||||
范围:
|
||||
- EMDF ID14 的 joc_header、joc_info 和 Huffman joc_data;
|
||||
- 差分解码得到 joc_mix_mtx_q;
|
||||
- 差分还原得到 joc_mix_mtx_q(dense MTX 与 sparse IDX/VEC 两条语法);
|
||||
- 去量化得到 joc_mix_mtx_dq;
|
||||
- 位流自洽验证(joc_data 后剩余 = padding_bits 0..7 + 可能 joc_ext_data)
|
||||
后续的时间插值、QMF/时域重建和 ``joc_clipgain`` 位于 ``renderer.py``。
|
||||
Sparse 分支仍缺少实际样本验证。
|
||||
公式与符号定义见 ``docs/math.md`` 第 2 节。
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
@@ -23,6 +23,7 @@ _HUFF_NAMES = (
|
||||
"joc_huff_code_7ch_pos_index_sparse",
|
||||
)
|
||||
|
||||
|
||||
def _load_huff_tables():
|
||||
with np.load(_TABLES_PATH) as tables:
|
||||
return {
|
||||
@@ -30,18 +31,24 @@ def _load_huff_tables():
|
||||
for name in _HUFF_NAMES
|
||||
}
|
||||
|
||||
|
||||
H = _load_huff_tables()
|
||||
|
||||
JOC_NUM_CHANNELS = {0: 5, 1: 7, 2: 7, 3: 5, 4: 7} # Table 33
|
||||
JOC_NUM_BANDS = {0: 1, 1: 3, 2: 5, 3: 7, 4: 9, 5: 12, 6: 15, 7: 23} # Table 35
|
||||
_PUBLIC_TABLE39_EXCERPT_UNUSED = { # 仅保留作表格差异说明;渲染映射在 joc_qmf.py。
|
||||
23: [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22],
|
||||
}
|
||||
JOC_NUM_QUANT = {0: 96, 1: 192} # Table 51
|
||||
# dense 的量化零点就是 nquant/2;sparse 的递推起点比它高 2 个量化步。
|
||||
JOC_DENSE_OFFSET = {0: 48, 1: 96}
|
||||
JOC_SPARSE_OFFSET = {0: 50, 1: 100}
|
||||
|
||||
|
||||
class BR:
|
||||
"""MSB-first 位读取器;位置以载荷内的 bit offset 表示。"""
|
||||
|
||||
def __init__(self, data, pos=0):
|
||||
self.d = data
|
||||
self.p = pos
|
||||
|
||||
def bits(self, n):
|
||||
v = 0
|
||||
for _ in range(n):
|
||||
@@ -53,18 +60,19 @@ class BR:
|
||||
def huff_decode(tree, br):
|
||||
node = 0
|
||||
while node >= 0:
|
||||
b = br.bits(1)
|
||||
node = tree[node][b]
|
||||
node = tree[node][br.bits(1)]
|
||||
return -node - 1
|
||||
|
||||
|
||||
def get_huff_code(mode, typ, nch):
|
||||
if typ == "IDX":
|
||||
return H["joc_huff_code_5ch_pos_index_sparse" if nch == 5 else "joc_huff_code_7ch_pos_index_sparse"]
|
||||
return H["joc_huff_code_5ch_pos_index_sparse" if nch == 5
|
||||
else "joc_huff_code_7ch_pos_index_sparse"]
|
||||
if typ == "VEC":
|
||||
return H["joc_huff_code_coarse_coeff_sparse" if mode == 0 else "joc_huff_code_fine_coeff_sparse"]
|
||||
# MTX
|
||||
return H["joc_huff_code_coarse_generic" if mode == 0 else "joc_huff_code_fine_generic"]
|
||||
return H["joc_huff_code_coarse_coeff_sparse" if mode == 0
|
||||
else "joc_huff_code_fine_coeff_sparse"]
|
||||
return H["joc_huff_code_coarse_generic" if mode == 0
|
||||
else "joc_huff_code_fine_generic"] # MTX
|
||||
|
||||
|
||||
def parse_joc(payload):
|
||||
@@ -76,6 +84,8 @@ def parse_joc(payload):
|
||||
out["ext_config_idx"] = br.bits(3)
|
||||
n_objects = out["num_objects_bits"] + 1
|
||||
n_channels = JOC_NUM_CHANNELS.get(out["dmx_config_idx"])
|
||||
if n_channels is None:
|
||||
raise ValueError(f"未知的 JOC downmix 配置 {out['dmx_config_idx']}")
|
||||
out["n_objects"], out["n_channels"] = n_objects, n_channels
|
||||
out["clipgain_x_bits"] = br.bits(3)
|
||||
out["clipgain_y_bits"] = br.bits(5)
|
||||
@@ -98,78 +108,120 @@ def parse_joc(payload):
|
||||
o["offset_ts"] = [br.bits(5) + 1 for _ in range(o["n_dpoints"])]
|
||||
objs.append(o)
|
||||
out["objs"] = objs
|
||||
# joc_data(Huffman)
|
||||
# joc_data(Huffman):dense 逐声道逐带读 MTX;sparse 读 IDX 后读 VEC。
|
||||
for obj, o in enumerate(objs):
|
||||
if not o["present"]:
|
||||
continue
|
||||
nquant = 96 if o["quant_idx"] == 0 else 192
|
||||
o["channel_idx"] = []
|
||||
o["vec"] = []
|
||||
o["mtx"] = []
|
||||
for dp in range(o["n_dpoints"]):
|
||||
if o["sparse"] == 1:
|
||||
# Sparse JOC 使用 VEC/IDX Huffman 树;此分支尚无真实码流验证。
|
||||
ci0 = br.bits(3)
|
||||
tree = get_huff_code(n_channels, "IDX", n_channels)
|
||||
ci = [ci0] + [huff_decode(tree, br) for _ in range(o["n_bands"] - 1)]
|
||||
o["channel_idx"].append(ci)
|
||||
tree = get_huff_code(o["quant_idx"], "IDX", n_channels)
|
||||
idx = [br.bits(3)]
|
||||
idx += [huff_decode(tree, br) for _ in range(o["n_bands"] - 1)]
|
||||
tree = get_huff_code(o["quant_idx"], "VEC", n_channels)
|
||||
vec = [huff_decode(tree, br) for _ in range(o["n_bands"])]
|
||||
o["vec"].append(vec)
|
||||
o["channel_idx"].append(idx)
|
||||
o["vec"].append([huff_decode(tree, br) for _ in range(o["n_bands"])])
|
||||
o["mtx"].append(None)
|
||||
else:
|
||||
tree = get_huff_code(o["quant_idx"], "MTX", n_channels)
|
||||
mtx = [[huff_decode(tree, br) for _ in range(o["n_bands"])]
|
||||
for _ in range(n_channels)]
|
||||
o["mtx"].append(mtx)
|
||||
o["channel_idx"].append(None)
|
||||
o["vec"].append(None)
|
||||
out["data_end_bits"] = br.p
|
||||
out["remaining_bits"] = len(payload) * 8 - br.p
|
||||
out["tail_bytes"] = payload[br.p // 8:]
|
||||
return out
|
||||
|
||||
|
||||
def reconstruct_dense(o, dp, n_ch):
|
||||
"""Dense 差分还原 → 量化矩阵 ``[ch][pb]``。
|
||||
|
||||
每个核心声道各自从 ``nquant/2``(去量化 0)出发,沿参数带累加 MTX 符号。
|
||||
"""
|
||||
nquant = JOC_NUM_QUANT[o["quant_idx"]]
|
||||
offset = JOC_DENSE_OFFSET[o["quant_idx"]]
|
||||
mtx = o["mtx"][dp]
|
||||
if mtx is None:
|
||||
raise ValueError("Dense JOC 对象缺少 MTX 符号")
|
||||
q = np.zeros((n_ch, o["n_bands"]), dtype=np.int64)
|
||||
for ch in range(n_ch):
|
||||
q[ch][0] = (offset + mtx[ch][0]) % nquant
|
||||
for pb in range(1, o["n_bands"]):
|
||||
q[ch][pb] = (q[ch][pb - 1] + mtx[ch][pb]) % nquant
|
||||
return q
|
||||
|
||||
|
||||
def reconstruct_sparse(o, dp, n_ch):
|
||||
"""Sparse 差分还原 → 量化矩阵 ``[ch][pb]``。
|
||||
|
||||
每个参数带只有一个 active 声道:
|
||||
- ``active[0]`` 是 3 bit 绝对声道号,其后由 IDX 符号累加得到,
|
||||
因此递推锚点是**已重建**的 active 声道,而不是编码器送的符号本身;
|
||||
- 系数是一个跨参数带连续的单累加器(active 声道切换时**不**重置),
|
||||
起点为 sparse offset,增量为 VEC 符号;
|
||||
- 非 active 项取 ``nquant/2``,即去量化后的 0。
|
||||
|
||||
IDX 符号取值恒为 ``0..n_ch-1``(Huffman 叶数即声道数),
|
||||
故 ``(active + idx) % n_ch`` 与单次条件减等价。
|
||||
"""
|
||||
if n_ch not in (5, 7):
|
||||
raise ValueError(f"Sparse JOC 需要 5 或 7 个核心声道,实际 {n_ch}")
|
||||
nquant = JOC_NUM_QUANT[o["quant_idx"]]
|
||||
offset = JOC_SPARSE_OFFSET[o["quant_idx"]]
|
||||
n_bands = o["n_bands"]
|
||||
idx = o["channel_idx"][dp]
|
||||
vec = o["vec"][dp]
|
||||
if idx is None or vec is None:
|
||||
raise ValueError("Sparse JOC 对象缺少 channel_idx/vec 符号")
|
||||
if len(idx) != n_bands or len(vec) != n_bands:
|
||||
raise ValueError(
|
||||
f"Sparse JOC 维度不符: bands={n_bands}, idx={len(idx)}, vec={len(vec)}")
|
||||
if not 0 <= idx[0] < n_ch:
|
||||
raise ValueError(f"Sparse JOC 初始声道 {idx[0]} 超出 {n_ch} 个声道")
|
||||
|
||||
q = np.full((n_ch, n_bands), nquant // 2, dtype=np.int64)
|
||||
active = idx[0]
|
||||
coefficient = offset
|
||||
for pb in range(n_bands):
|
||||
if pb:
|
||||
active = (active + idx[pb]) % n_ch
|
||||
coefficient = (coefficient + vec[pb]) % nquant
|
||||
q[active][pb] = coefficient
|
||||
return q
|
||||
|
||||
|
||||
def diff_decode(out):
|
||||
"""6.6.2:差分解码 → joc_mix_mtx_q[obj][dp][ch][pb]。"""
|
||||
"""差分还原 → joc_mix_mtx_q[obj][dp][ch][pb]。"""
|
||||
mix_q = {}
|
||||
n_ch = out["n_channels"]
|
||||
for obj, o in enumerate(out["objs"]):
|
||||
if not o["present"]:
|
||||
continue
|
||||
nquant = 96 if o["quant_idx"] == 0 else 192
|
||||
q = np.zeros((o["n_dpoints"], n_ch, o["n_bands"]), dtype=np.int64)
|
||||
for dp in range(o["n_dpoints"]):
|
||||
if o["sparse"] == 1:
|
||||
# Sparse 差分路径尚无真实码流验证。
|
||||
offset = 50 if o["quant_idx"] == 0 else 100
|
||||
ci = o["channel_idx"][dp]
|
||||
vec = o["vec"][dp]
|
||||
for pb in range(o["n_bands"]):
|
||||
ci_mod = ci[0] if pb == 0 else (ci[pb - 1] + ci[pb]) % n_ch
|
||||
for ch in range(n_ch):
|
||||
if ch == ci_mod:
|
||||
if pb == 0:
|
||||
q[dp][ch][pb] = (offset + vec[pb]) % nquant
|
||||
else:
|
||||
q[dp][ch][pb] = (q[dp][ch][pb - 1] + vec[pb]) % nquant
|
||||
else:
|
||||
q[dp][ch][pb] = offset
|
||||
q[dp] = reconstruct_sparse(o, dp, n_ch)
|
||||
else:
|
||||
offset = 48 if o["quant_idx"] == 0 else 96
|
||||
mtx = o["mtx"][dp]
|
||||
for ch in range(n_ch):
|
||||
q[dp][ch][0] = (offset + mtx[ch][0]) % nquant
|
||||
for pb in range(1, o["n_bands"]):
|
||||
q[dp][ch][pb] = (q[dp][ch][pb - 1] + mtx[ch][pb]) % nquant
|
||||
q[dp] = reconstruct_dense(o, dp, n_ch)
|
||||
mix_q[obj] = q
|
||||
return mix_q
|
||||
|
||||
|
||||
def dequantize(out, mix_q):
|
||||
"""6.6.4:去量化 → joc_mix_mtx_dq。"""
|
||||
"""去量化 → joc_mix_mtx_dq。
|
||||
|
||||
Sparse 的非 active 项在 ``joc_mix_mtx_q`` 中取 ``nquant/2``,
|
||||
因此与 dense 共用同一条去量化公式即得到 0。
|
||||
"""
|
||||
mix_dq = {}
|
||||
for obj, o in enumerate(out["objs"]):
|
||||
if not o["present"]:
|
||||
continue
|
||||
nquant = 96 if o["quant_idx"] == 0 else 192
|
||||
nquant = JOC_NUM_QUANT[o["quant_idx"]]
|
||||
q = mix_q[obj]
|
||||
dq = (q.astype(np.float64) - nquant / 2) * 820 / (4096 * (1 + o["quant_idx"]))
|
||||
mix_dq[obj] = dq
|
||||
|
||||
@@ -246,20 +246,6 @@ def inspect(index, limit=None, print_frames=False):
|
||||
"parser_error": str(exc),
|
||||
"repair_hint": "检查 JOC header、对象数、参数带、Huffman 或扩展字段",
|
||||
}) from exc
|
||||
sparse = [i for i, obj in enumerate(parsed["objs"]) if obj["present"] and obj["sparse"]]
|
||||
if sparse:
|
||||
raise UnsupportedVariantError(
|
||||
"joc", "sparse_joc",
|
||||
"发现尚未验证的 Sparse JOC 帧",
|
||||
frame=frame_number,
|
||||
details={
|
||||
"sparse_objects": sparse,
|
||||
"downmix_config": parsed["dmx_config_idx"],
|
||||
"extension_config": parsed["ext_config_idx"],
|
||||
"objects": parsed["n_objects"],
|
||||
"payload": bytes_descriptor(subs[14]),
|
||||
"repair_hint": "需要 Sparse JOC 实际样本及对应输出建立回归后再启用",
|
||||
})
|
||||
if parsed["n_channels"] != 5 or parsed["n_objects"] > 15:
|
||||
raise UnsupportedVariantError(
|
||||
"joc", "unsupported_configuration",
|
||||
|
||||
@@ -208,8 +208,6 @@ class NativeJocRenderer:
|
||||
for object_index, info in enumerate(out["objs"]):
|
||||
if not info["present"]:
|
||||
continue
|
||||
if info["sparse"]:
|
||||
raise ValueError("native core does not accept unvalidated Sparse JOC")
|
||||
bands = int(info["n_bands"])
|
||||
points = int(info["n_dpoints"])
|
||||
if bands > MAX_BANDS or points > MAX_DPOINTS:
|
||||
|
||||
+422
-235
@@ -1,12 +1,14 @@
|
||||
"""OAMD 位载荷 → 16 个对象槽的 q1/q2/q3 增量状态。
|
||||
|
||||
槽 0 是 bed/LFE;槽 1..15 对应输出 ch1..15 的对象元数据。
|
||||
依据 ETSI TS 103 420 V1.2.1 clause 5。槽 0 为 bed/LFE,槽 1..15 为输出
|
||||
ch1..15;bed/ISF/未激活对象没有位置字段,保持上一帧位置。element 目录按声明长度
|
||||
驱动,未知 element 按边界跳过;个别编码器的 ``oa_element_size`` 比实际内容短时以
|
||||
结构解析为准,差异记入 ``diagnostics``,不算变体错误。
|
||||
"""
|
||||
import numpy as np
|
||||
|
||||
from variant_error import UnsupportedVariantError, bytes_descriptor
|
||||
|
||||
Q1_OFF = 192
|
||||
N_Q12 = 62
|
||||
N_Q3 = 15
|
||||
SAMPLE_OFFSET_INDEX = (8, 16, 18, 24)
|
||||
@@ -16,9 +18,20 @@ RAMP_DURATION_INDEX = (
|
||||
1024, 1600, 1601, 1602, 1920, 2000, 2002, 2048,
|
||||
)
|
||||
OBJECT_ELEMENT_ID = 1
|
||||
TRIM_ELEMENT_ID = 2
|
||||
EXTENDED_OBJECT_ELEMENT_ID = 5
|
||||
POSITION_WINDOW_END_BIT = 112 + 31 * (15 - 3) + 24
|
||||
# 5.6.0.11 Table 11b:ISF 类型 → 对象数。
|
||||
ISF_OBJECT_COUNTS = (4, 8, 10, 14, 15, 30)
|
||||
# 5.6.1.1.4 Table 12:10 bit 标准 bed 掩码按位对应的声道(LSB = RC_L/RC_R)。
|
||||
BED_CHANNEL_LABELS = (
|
||||
"RC_L/RC_R", "RC_C", "RC_LFE", "RC_LS/RC_RS", "RC_LB/RC_RB",
|
||||
"RC_TFL/RC_TFR", "RC_TSL/RC_TSR", "RC_TBL/RC_TBR", "RC_LW/RC_RW", "RC_LFE2",
|
||||
)
|
||||
# 5.6.1.1.5 Table 13:17 bit 非标准 bed 掩码按位对应的声道标签。
|
||||
NONSTD_BED_LABELS = (
|
||||
"RC_LFE2", "RC_RW", "RC_LW", "RC_TBR", "RC_TBL", "RC_TSR", "RC_TSL",
|
||||
"RC_TFR", "RC_TFL", "RC_RB", "RC_LB", "RC_RS", "RC_LS", "RC_LFE", "RC_C",
|
||||
"RC_R", "RC_L",
|
||||
)
|
||||
MAX_SLOTS = 16
|
||||
|
||||
|
||||
def q_of(k, n):
|
||||
@@ -91,14 +104,23 @@ class _BitReader:
|
||||
self.read(count)
|
||||
|
||||
|
||||
def _variable_bits(reader, width, max_groups=5):
|
||||
value = 0
|
||||
for _ in range(max_groups + 1):
|
||||
value += reader.read(width)
|
||||
if not reader.read(1):
|
||||
return value
|
||||
value = (value + 1) << width
|
||||
raise ValueError(f"OAMD variable_bits({width}) 延伸组过多")
|
||||
def _variable_bits_max(reader, width, max_groups):
|
||||
"""5.5.1 ``variable_bits_max(n, max_num_groups)``。"""
|
||||
value = reader.read(width)
|
||||
more = reader.read(1)
|
||||
num_group = 1
|
||||
if max_groups > num_group:
|
||||
if more:
|
||||
value = (value + 1) << width
|
||||
while more:
|
||||
value += reader.read(width)
|
||||
more = reader.read(1)
|
||||
if num_group >= max_groups:
|
||||
break
|
||||
if more:
|
||||
value = (value + 1) << width
|
||||
num_group += 1
|
||||
return value
|
||||
|
||||
|
||||
def _element_details(elements):
|
||||
@@ -109,166 +131,185 @@ def _element_details(elements):
|
||||
"header_start_bit": element["header_start_bit"],
|
||||
"body_start_bit": element["body_start_bit"],
|
||||
"body_end_bit": element["body_end_bit"],
|
||||
"parsed_end_bit": element.get("parsed_end_bit"),
|
||||
"discard_unknown": element["discard_unknown"],
|
||||
"alternate_data_id": element["alternate_data_id"],
|
||||
} for element in elements]
|
||||
|
||||
|
||||
def _parse_elements(bits, header, header_end, raw_payload):
|
||||
"""建立 OAMD element 目录;element size 表示其后 body 的字节数。"""
|
||||
reader = _BitReader(bits, header_end)
|
||||
elements = []
|
||||
for ordinal in range(header["element_count"]):
|
||||
header_start = reader.position
|
||||
try:
|
||||
element_id = reader.read(4)
|
||||
size_bytes = _variable_bits(reader, 4) + 1
|
||||
except ValueError as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "element_header_truncated",
|
||||
"OAMD element header 不完整",
|
||||
details={
|
||||
"element_ordinal": ordinal,
|
||||
"header_start_bit": header_start,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"parser_error": str(exc),
|
||||
}) from exc
|
||||
def _syntax_error(variant, message, raw_payload, exc=None, **details):
|
||||
payload = {"payload": bytes_descriptor(raw_payload)}
|
||||
if exc is not None:
|
||||
payload["parser_error"] = str(exc)
|
||||
payload.update(details)
|
||||
return UnsupportedVariantError("oamd", variant, message, details=payload)
|
||||
|
||||
body_start = reader.position
|
||||
body_end = body_start + size_bytes * 8
|
||||
if body_end > len(bits):
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "element_bounds",
|
||||
"OAMD element 声明长度超过 payload 边界",
|
||||
details={
|
||||
"element_ordinal": ordinal,
|
||||
"element_id": element_id,
|
||||
"size_bytes": size_bytes,
|
||||
"body_start_bit": body_start,
|
||||
"body_end_bit": body_end,
|
||||
"payload_bits": len(bits),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
|
||||
def _parse_program_assignment(reader, raw_payload):
|
||||
"""5.5.3 ``program_assignment()``:bed / ISF / dynamic 对象数量。"""
|
||||
program = {
|
||||
"dynamic_object_only": bool(reader.read(1)),
|
||||
"lfe_present": False,
|
||||
"content_description": None,
|
||||
"bed_assignments": [],
|
||||
"num_bed_objects": 0,
|
||||
"isf_idx": None,
|
||||
"num_isf_objects": 0,
|
||||
"num_dynamic_objects": None,
|
||||
}
|
||||
if program["dynamic_object_only"]:
|
||||
program["lfe_present"] = bool(reader.read(1))
|
||||
# 5.6.4.8:对象顺序为 bed → ISF → dynamic,LFE 属于 bed,排在最前。
|
||||
program["num_bed_objects"] = 1 if program["lfe_present"] else 0
|
||||
return program
|
||||
|
||||
mask = reader.read(4)
|
||||
program["content_description"] = {
|
||||
"reserved": bool(mask & 0x8),
|
||||
"dynamic": bool(mask & 0x4),
|
||||
"isf": bool(mask & 0x2),
|
||||
"bed": bool(mask & 0x1),
|
||||
}
|
||||
if mask & 0x1:
|
||||
program["b_bed_chan_distribute"] = bool(reader.read(1))
|
||||
num_instances = (reader.read(3) + 2) if reader.read(1) else 1
|
||||
for instance in range(num_instances):
|
||||
if reader.read(1): # b_lfe_only
|
||||
program["bed_assignments"].append({
|
||||
"instance": instance, "lfe_only": True, "channels": ["RC_LFE"],
|
||||
})
|
||||
|
||||
control_bits = 5 if header["alternate_object_present"] else 1
|
||||
if body_start + control_bits > body_end:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "element_control_bounds",
|
||||
"OAMD element 太短,无法容纳控制字段",
|
||||
details={
|
||||
"element_ordinal": ordinal,
|
||||
"element_id": element_id,
|
||||
"size_bytes": size_bytes,
|
||||
"control_bits": control_bits,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
continue
|
||||
if reader.read(1): # b_standard_chan_assign
|
||||
bed_mask = reader.read(10)
|
||||
channels = []
|
||||
for bit in range(10):
|
||||
if (bed_mask >> bit) & 1:
|
||||
channels.extend(BED_CHANNEL_LABELS[bit].split("/"))
|
||||
program["bed_assignments"].append({
|
||||
"instance": instance, "lfe_only": False, "standard": True,
|
||||
"mask": bed_mask, "channels": channels,
|
||||
})
|
||||
control = _BitReader(bits, body_start, body_end)
|
||||
alternate_data_id = (control.read(4)
|
||||
if header["alternate_object_present"] else None)
|
||||
discard_unknown = bool(control.read(1))
|
||||
elements.append({
|
||||
"ordinal": ordinal,
|
||||
"element_id": element_id,
|
||||
"size_bytes": size_bytes,
|
||||
"header_start_bit": header_start,
|
||||
"body_start_bit": body_start,
|
||||
"data_start_bit": control.position,
|
||||
"body_end_bit": body_end,
|
||||
"alternate_data_id": alternate_data_id,
|
||||
"discard_unknown": discard_unknown,
|
||||
})
|
||||
reader.position = body_end
|
||||
|
||||
padding = bits[reader.position:]
|
||||
if len(padding) > 7:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "trailing_payload_data",
|
||||
"OAMD element 结束后仍有超过一个字节的未声明数据",
|
||||
details={
|
||||
"elements": _element_details(elements),
|
||||
"trailing_bits": len(padding),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
if np.any(padding):
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "nonzero_padding",
|
||||
"OAMD payload 尾部 padding 含非零位",
|
||||
details={
|
||||
"elements": _element_details(elements),
|
||||
"padding_start_bit": reader.position,
|
||||
"padding_bits": "".join(str(int(bit)) for bit in padding),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
|
||||
object_elements = [element for element in elements
|
||||
if element["element_id"] == OBJECT_ELEMENT_ID]
|
||||
if len(object_elements) != 1:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "object_element_count",
|
||||
"OAMD 必须包含且只能包含一个 object element",
|
||||
details={
|
||||
"object_element_count": len(object_elements),
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
object_element = object_elements[0]
|
||||
if object_element["ordinal"] != 0 or object_element["header_start_bit"] != header_end:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "object_element_order",
|
||||
"object element 不在当前固定位置窗口支持的首个 element 位置",
|
||||
details={
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "以 object element 的实际位置为基准重新定位对象窗口",
|
||||
})
|
||||
|
||||
for element in elements:
|
||||
element_id = element["element_id"]
|
||||
if element_id in (OBJECT_ELEMENT_ID, TRIM_ELEMENT_ID):
|
||||
continue
|
||||
if element_id == EXTENDED_OBJECT_ELEMENT_ID:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "extended_object_element",
|
||||
"OAMD 含可能改变坐标语义的 extended object element",
|
||||
details={
|
||||
"element": _element_details([element])[0],
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "解析 divergence/extended-precision position 后再应用轨迹",
|
||||
else:
|
||||
bed_mask = reader.read(17)
|
||||
channels = [NONSTD_BED_LABELS[bit]
|
||||
for bit in range(17) if (bed_mask >> bit) & 1]
|
||||
program["bed_assignments"].append({
|
||||
"instance": instance, "lfe_only": False, "standard": False,
|
||||
"mask": bed_mask, "channels": channels,
|
||||
})
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", f"unsupported_element_{element_id}",
|
||||
f"OAMD 含当前未覆盖的 element id {element_id}",
|
||||
details={
|
||||
"element": _element_details([element])[0],
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
return object_element, elements
|
||||
program["num_bed_objects"] = sum(
|
||||
len(assignment["channels"]) for assignment in program["bed_assignments"])
|
||||
if mask & 0x2:
|
||||
isf_idx = reader.read(3)
|
||||
program["isf_idx"] = isf_idx
|
||||
program["num_isf_objects"] = (ISF_OBJECT_COUNTS[isf_idx]
|
||||
if isf_idx < len(ISF_OBJECT_COUNTS) else 0)
|
||||
if mask & 0x4:
|
||||
count = reader.read(5)
|
||||
if count == 0x1F:
|
||||
count += reader.read(7)
|
||||
program["num_dynamic_objects"] = count + 1
|
||||
if mask & 0x8:
|
||||
# 5.6.4.1:reserved_data_size = reserved_data_size_bits + 1(字节)。
|
||||
reader.skip((reader.read(4) + 1) * 8)
|
||||
return program
|
||||
|
||||
|
||||
def _update_timing(bits, object_element, raw_payload):
|
||||
"""从已定位的 object element 读取位置块开始偏移和 ramp 时长。"""
|
||||
reader = _BitReader(
|
||||
bits, object_element["data_start_bit"], object_element["body_end_bit"])
|
||||
def _parse_object_info_block(reader, object_index, in_bed_or_isf, raw_payload):
|
||||
"""5.5.9 ``object_info_block(0)``:一个对象的属性更新。"""
|
||||
info = {
|
||||
"object": object_index,
|
||||
"in_bed_or_isf": bool(in_bed_or_isf),
|
||||
"not_active": bool(reader.read(1)),
|
||||
}
|
||||
# blk == 0 且对象激活时 basic info 恒为 full update(status 0b01)。
|
||||
basic_status = 0 if info["not_active"] else 1
|
||||
info["basic_status"] = basic_status
|
||||
if basic_status in (1, 3):
|
||||
if basic_status == 1:
|
||||
# 5.5.10:object_basic_info[] = {true, true};
|
||||
# 5.6.4.12:array[1] = object_gain_idx,array[0] = b_default_object_priority。
|
||||
gain_present = priority_present = True
|
||||
else:
|
||||
flags = reader.read(2)
|
||||
gain_present = bool(flags & 1)
|
||||
priority_present = bool(flags & 2)
|
||||
if gain_present:
|
||||
gain_idx = reader.read(2)
|
||||
info["gain_idx"] = gain_idx
|
||||
if gain_idx == 2:
|
||||
info["gain_bits"] = reader.read(6)
|
||||
if priority_present:
|
||||
default_priority = bool(reader.read(1))
|
||||
info["default_priority"] = default_priority
|
||||
if not default_priority:
|
||||
info["priority_bits"] = reader.read(5)
|
||||
render_status = 0 if (info["not_active"] or info["in_bed_or_isf"]) else 1
|
||||
info["render_status"] = render_status
|
||||
if render_status in (1, 3):
|
||||
if render_status == 1:
|
||||
# 5.5.11:obj_render_info[] = {true, true, true, true}。
|
||||
position_present = zone_present = size_present = screen_present = True
|
||||
else:
|
||||
flags = reader.read(4)
|
||||
position_present = bool(flags & 0x1)
|
||||
zone_present = bool(flags & 0x2)
|
||||
size_present = bool(flags & 0x4)
|
||||
screen_present = bool(flags & 0x8)
|
||||
if position_present:
|
||||
# blk == 0 时 b_differential_position_specified 恒为 FALSE。
|
||||
x = reader.read(6)
|
||||
y = reader.read(6)
|
||||
z_sign = reader.read(1)
|
||||
z = reader.read(4)
|
||||
info["position"] = (x, y, z if z_sign else -z)
|
||||
if reader.read(1): # b_object_distance_specified
|
||||
if not reader.read(1): # b_object_at_infinity
|
||||
reader.read(4) # distance_factor_idx
|
||||
if zone_present:
|
||||
info["zone_constraints_idx"] = reader.read(3)
|
||||
info["enable_elevation"] = bool(reader.read(1))
|
||||
if size_present:
|
||||
size_idx = reader.read(2)
|
||||
info["object_size_idx"] = size_idx
|
||||
if size_idx == 1:
|
||||
reader.read(5)
|
||||
elif size_idx == 2:
|
||||
reader.read(15)
|
||||
if screen_present:
|
||||
if reader.read(1): # b_object_use_screen_ref
|
||||
reader.read(3) # screen_factor_bits
|
||||
reader.read(2) # depth_factor_idx
|
||||
info["snap"] = bool(reader.read(1))
|
||||
if reader.read(1): # b_additional_table_data_exists
|
||||
info["additional_table_bytes"] = reader.read(4) + 1
|
||||
reader.skip(info["additional_table_bytes"] * 8)
|
||||
return info
|
||||
|
||||
|
||||
def _parse_object_element(reader, element, raw_payload):
|
||||
"""5.5.5 ``object_element()``:时间信息 + 每个对象一条 object_info_block。"""
|
||||
object_count = element["object_count"]
|
||||
try:
|
||||
offset_code = reader.read(2)
|
||||
if offset_code == 0:
|
||||
sample_offset_code = reader.read(2)
|
||||
if sample_offset_code == 0:
|
||||
sample_offset = 0
|
||||
elif offset_code == 1:
|
||||
elif sample_offset_code == 1:
|
||||
sample_offset = SAMPLE_OFFSET_INDEX[reader.read(2)]
|
||||
elif offset_code == 2:
|
||||
elif sample_offset_code == 2:
|
||||
sample_offset = reader.read(5)
|
||||
else:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "md_sample_offset_mode",
|
||||
"OAMD 使用了当前未覆盖的 MD sample-offset 模式",
|
||||
details={"payload": bytes_descriptor(raw_payload)})
|
||||
|
||||
details={
|
||||
"sample_offset_code": sample_offset_code,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
block_count = reader.read(3) + 1
|
||||
blocks = []
|
||||
for _block in range(block_count):
|
||||
block_offset_factor = reader.read(6)
|
||||
block_offset = sample_offset + block_offset_factor * 32
|
||||
ramp_code = reader.read(2)
|
||||
if ramp_code == 3:
|
||||
if reader.read(1):
|
||||
@@ -277,107 +318,253 @@ def _update_timing(bits, object_element, raw_payload):
|
||||
ramp_duration = reader.read(11)
|
||||
else:
|
||||
ramp_duration = RAMP_DURATIONS[ramp_code]
|
||||
blocks.append((block_offset, ramp_duration))
|
||||
blocks.append({
|
||||
"block_offset_samples": sample_offset + block_offset_factor * 32,
|
||||
"ramp_duration_samples": ramp_duration,
|
||||
})
|
||||
reserved_data_not_present = bool(reader.read(1))
|
||||
if not reserved_data_not_present:
|
||||
reader.read(5)
|
||||
objects = [
|
||||
_parse_object_info_block(
|
||||
reader, index, index < element["bed_isf_objects"], raw_payload)
|
||||
for index in range(object_count)
|
||||
]
|
||||
except UnsupportedVariantError:
|
||||
raise
|
||||
except ValueError as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "object_element_syntax",
|
||||
"OAMD object element 的 timing 字段越界或不完整",
|
||||
details={
|
||||
"element": _element_details([object_element])[0],
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"parser_error": str(exc),
|
||||
}) from exc
|
||||
raise _syntax_error(
|
||||
"object_element_syntax",
|
||||
"OAMD object element 字段越界或不完整",
|
||||
raw_payload, exc,
|
||||
element=_element_details([element])[0],
|
||||
object_count=object_count,
|
||||
) from exc
|
||||
return {
|
||||
"sample_offset": sample_offset,
|
||||
"blocks": blocks,
|
||||
"reserved_data_not_present": reserved_data_not_present,
|
||||
"objects": objects,
|
||||
}
|
||||
|
||||
|
||||
def _parse_elements(reader, alternate_present, object_count, program, raw_payload):
|
||||
"""5.5.4 ``oa_element_md()`` 目录;对象 element 按结构解析,其余按声明长度跳过。"""
|
||||
try:
|
||||
element_count = reader.read(4)
|
||||
if element_count == 0xF:
|
||||
element_count += reader.read(5)
|
||||
except ValueError as exc:
|
||||
raise _syntax_error(
|
||||
"element_count_truncated", "OAMD element 数量字段不完整",
|
||||
raw_payload, exc) from exc
|
||||
if element_count == 0:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "missing_object_element",
|
||||
"OAMD 没有声明任何 element",
|
||||
details={"payload": bytes_descriptor(raw_payload)})
|
||||
|
||||
bed_isf_objects = program["num_bed_objects"] + program["num_isf_objects"]
|
||||
elements = []
|
||||
object_element = None
|
||||
for ordinal in range(element_count):
|
||||
header_start = reader.position
|
||||
try:
|
||||
element_id = reader.read(4)
|
||||
size_bytes = _variable_bits_max(reader, 4, 4) + 1
|
||||
except ValueError as exc:
|
||||
raise _syntax_error(
|
||||
"element_header_truncated", "OAMD element header 不完整",
|
||||
raw_payload, exc,
|
||||
element_ordinal=ordinal, header_start_bit=header_start) from exc
|
||||
region_start = reader.position
|
||||
region_end = region_start + size_bytes * 8
|
||||
if region_end > reader.limit:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "element_bounds",
|
||||
"OAMD element 声明长度超过 payload 边界",
|
||||
details={
|
||||
"element_ordinal": ordinal,
|
||||
"element_id": element_id,
|
||||
"size_bytes": size_bytes,
|
||||
"body_start_bit": region_start,
|
||||
"body_end_bit": region_end,
|
||||
"payload_bits": reader.limit,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
try:
|
||||
alternate_data_id = (reader.read(4)
|
||||
if alternate_present else None)
|
||||
discard_unknown = bool(reader.read(1))
|
||||
except ValueError as exc:
|
||||
raise _syntax_error(
|
||||
"element_control_bounds", "OAMD element 太短,无法容纳控制字段",
|
||||
raw_payload, exc,
|
||||
element_ordinal=ordinal, element_id=element_id,
|
||||
size_bytes=size_bytes) from exc
|
||||
element = {
|
||||
"ordinal": ordinal,
|
||||
"element_id": element_id,
|
||||
"size_bytes": size_bytes,
|
||||
"header_start_bit": header_start,
|
||||
"body_start_bit": region_start,
|
||||
"body_end_bit": region_end,
|
||||
"data_start_bit": reader.position,
|
||||
"alternate_data_id": alternate_data_id,
|
||||
"discard_unknown": discard_unknown,
|
||||
"object_count": object_count,
|
||||
"bed_isf_objects": bed_isf_objects,
|
||||
}
|
||||
if element_id == OBJECT_ELEMENT_ID:
|
||||
if object_element is not None:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "multiple_object_elements",
|
||||
"OAMD 含多个 object element",
|
||||
details={
|
||||
"elements": _element_details(elements + [element]),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
object_element = _parse_object_element(
|
||||
reader, element, raw_payload)
|
||||
element["parsed_end_bit"] = reader.position
|
||||
element["object_element"] = object_element
|
||||
# 个别编码器声明的 oa_element_size 比实际内容短;以结构解析结果为准。
|
||||
reader.position = max(region_end, reader.position)
|
||||
else:
|
||||
# trim / extended / 未知 element:按声明长度整体跳过(5.5.4)。
|
||||
element["parsed_end_bit"] = region_end
|
||||
reader.position = region_end
|
||||
elements.append(element)
|
||||
|
||||
if object_element is None:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "missing_object_element",
|
||||
"OAMD 缺少 object element",
|
||||
details={
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
padding = reader.bits[reader.position:]
|
||||
if np.any(padding):
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "nonzero_padding",
|
||||
"OAMD payload 尾部 padding 含非零位",
|
||||
details={
|
||||
"elements": _element_details(elements),
|
||||
"padding_start_bit": reader.position,
|
||||
"padding_bits": "".join(str(int(bit)) for bit in padding[:64]),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
return object_element, elements
|
||||
|
||||
|
||||
def frame_update(bits_one):
|
||||
"""单帧 OAMD → 位置字段增量及其 sample offset/ramp duration。"""
|
||||
bits, raw_payload = _payload_bits(bits_one)
|
||||
try:
|
||||
version = bits[0] << 1 | bits[1]
|
||||
except IndexError:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "header_truncated",
|
||||
"OAMD payload 不足以容纳 header",
|
||||
details={"payload_bits": len(bits),
|
||||
"payload": bytes_descriptor(raw_payload)}) from None
|
||||
reader = _BitReader(bits, 2)
|
||||
try:
|
||||
if version == 3:
|
||||
version += reader.read(3)
|
||||
object_count_bits = reader.read(5)
|
||||
if object_count_bits == 0x1F:
|
||||
object_count_bits += reader.read(7)
|
||||
object_count = object_count_bits + 1
|
||||
program = _parse_program_assignment(reader, raw_payload)
|
||||
alternate_present = bool(reader.read(1))
|
||||
object_element, elements = _parse_elements(
|
||||
reader, alternate_present, object_count, program, raw_payload)
|
||||
except UnsupportedVariantError:
|
||||
raise
|
||||
except ValueError as exc:
|
||||
raise _syntax_error(
|
||||
"header_truncated", "OAMD header 字段越界或不完整",
|
||||
raw_payload, exc, payload_bits=len(bits)) from exc
|
||||
|
||||
if version != 0:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "oamd_version",
|
||||
"OAMD 使用了当前未覆盖的 syntax version",
|
||||
details={
|
||||
"version": version,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
if object_count > MAX_SLOTS:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "object_count",
|
||||
"OAMD 对象数超过当前 16 槽对象模型",
|
||||
details={
|
||||
"object_count": object_count,
|
||||
"max_slots": MAX_SLOTS,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
blocks = object_element["blocks"]
|
||||
if len(blocks) != 1:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "multiple_position_blocks",
|
||||
"OAMD 一帧含多个对象位置更新块,固定位置窗口不能安全套用",
|
||||
"OAMD 一帧含多个对象位置更新块,单次 frame_update 无法表达",
|
||||
details={
|
||||
"block_count": len(blocks),
|
||||
"blocks": blocks,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "按 ObjectInfoBlock 顺序逐块解析坐标,再生成分段 ADM ramp",
|
||||
})
|
||||
return blocks[0]
|
||||
|
||||
# 对外契约:始终给出 16 个槽;本帧没有覆盖到的槽保持上一帧位置。
|
||||
out = {(slot, field): None for slot in range(MAX_SLOTS)
|
||||
for field in ("q1", "q2", "q3")}
|
||||
for info in object_element["objects"]:
|
||||
slot = info["object"]
|
||||
position = info.get("position")
|
||||
if position is None:
|
||||
# bed/ISF 对象与未激活对象没有位置字段:保持上一帧位置。
|
||||
out[(slot, "q1")] = None
|
||||
out[(slot, "q2")] = None
|
||||
out[(slot, "q3")] = None
|
||||
continue
|
||||
x, y, z = position
|
||||
out[(slot, "q1")] = q_of(x, N_Q12) if 0 <= x <= N_Q12 else None
|
||||
out[(slot, "q2")] = q_of(y, N_Q12) if 0 <= y <= N_Q12 else None
|
||||
# 5.6.1.1.10/5.6.1.1.11:pos3D_Z 带符号;当前 16 槽模型只表达非负高度。
|
||||
out[(slot, "q3")] = q_of(z, N_Q3) if 0 <= z <= N_Q3 else (
|
||||
0 if z < 0 else None)
|
||||
|
||||
def frame_update(bits_one):
|
||||
"""单帧 OAMD → 位置字段增量及其 sample offset/ramp duration。"""
|
||||
bits, raw_payload = _payload_bits(bits_one)
|
||||
if len(bits) < 14:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "header_truncated",
|
||||
"OAMD payload 不足以容纳受支持的 header",
|
||||
details={"payload_bits": len(bits), "payload": bytes_descriptor(raw_payload)})
|
||||
header = {
|
||||
"version": int((bits[0] << 1) | bits[1]),
|
||||
"objects_minus_one": int(sum(int(bits[2 + i]) << (4 - i) for i in range(5))),
|
||||
"dynamic_object_only": int(bits[7]),
|
||||
"lfe_present": int(bits[8]),
|
||||
"alternate_object_present": int(bits[9]),
|
||||
"element_count": int(sum(int(bits[10 + i]) << (3 - i) for i in range(4))),
|
||||
}
|
||||
if raw_payload[:2] != b"\x1f\x88":
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", f"header_signature_{raw_payload[:2].hex()}",
|
||||
"OAMD header 与当前位置字段布局不一致",
|
||||
details={
|
||||
"supported_header_prefix_hex": "1f88",
|
||||
"header_probe": header,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "按新 header 的 program assignment 和 element 布局重新定位对象位置字段",
|
||||
})
|
||||
object_element, elements = _parse_elements(bits, header, 14, raw_payload)
|
||||
if object_element["body_end_bit"] < POSITION_WINDOW_END_BIT:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "object_element_too_short",
|
||||
"OAMD object element 无法容纳当前固定位置窗口",
|
||||
details={
|
||||
"element": _element_details([object_element])[0],
|
||||
"required_position_end_bit": POSITION_WINDOW_END_BIT,
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
})
|
||||
block_offset, ramp_duration = _update_timing(bits, object_element, raw_payload)
|
||||
weights = 1 << np.arange(7, -1, -1)
|
||||
out = {}
|
||||
for obj in range(16):
|
||||
start = 112 + 31 * (obj - 3)
|
||||
wq1 = int((bits[start:start + 8] * weights).sum())
|
||||
wq2 = int((bits[start + 8:start + 16] * weights).sum())
|
||||
wq3 = int((bits[start + 16:start + 24] * weights).sum())
|
||||
if obj and (wq1 >> 6 != 3 or wq2 & 2 != 2 or wq3 & 0x1F != 1):
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "position_layout_signature",
|
||||
"OAMD 对象位置字段标记或位偏移发生变化",
|
||||
details={
|
||||
"object_slot": obj,
|
||||
"position_start_bit": start,
|
||||
"q1_window_hex": f"{wq1:02x}",
|
||||
"q2_window_hex": f"{wq2:02x}",
|
||||
"q3_window_hex": f"{wq3:02x}",
|
||||
"header_probe": header,
|
||||
"elements": _element_details(elements),
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "解析 OAMD element 可选字段并更新每个对象的位置窗口偏移",
|
||||
})
|
||||
k1 = wq1 - Q1_OFF
|
||||
if obj == 0:
|
||||
out[(0, "q1")] = None
|
||||
out[(0, "q2")] = None
|
||||
else:
|
||||
out[(obj, "q1")] = q_of(k1, N_Q12) if 0 <= k1 <= N_Q12 else None
|
||||
k2 = wq2 >> 2
|
||||
if obj != 0:
|
||||
out[(obj, "q2")] = q_of(k2, N_Q12) if 0 <= k2 <= N_Q12 else None
|
||||
k3 = (wq2 & 1) * 8 + (wq3 >> 5)
|
||||
out[(obj, "q3")] = q_of(k3, N_Q3) if 0 <= k3 <= N_Q3 else None
|
||||
size_mismatch = [
|
||||
{
|
||||
"element_ordinal": element["ordinal"],
|
||||
"element_id": element["element_id"],
|
||||
"declared_size_bytes": element["size_bytes"],
|
||||
"declared_end_bit": element["body_end_bit"],
|
||||
"parsed_end_bit": element["parsed_end_bit"],
|
||||
}
|
||||
for element in elements
|
||||
if element["parsed_end_bit"] > element["body_end_bit"]
|
||||
]
|
||||
return {
|
||||
"values": out,
|
||||
"block_offset_samples": block_offset,
|
||||
"ramp_duration_samples": ramp_duration,
|
||||
"block_offset_samples": blocks[0]["block_offset_samples"],
|
||||
"ramp_duration_samples": blocks[0]["ramp_duration_samples"],
|
||||
"object_count": object_count,
|
||||
"program": {
|
||||
"dynamic_object_only": program["dynamic_object_only"],
|
||||
"lfe_present": program["lfe_present"],
|
||||
"content_description": program["content_description"],
|
||||
"num_bed_objects": program["num_bed_objects"],
|
||||
"num_isf_objects": program["num_isf_objects"],
|
||||
"num_dynamic_objects": program["num_dynamic_objects"],
|
||||
"bed_channels": [list(assignment["channels"])
|
||||
for assignment in program["bed_assignments"]],
|
||||
},
|
||||
"elements": _element_details(elements),
|
||||
"objects": object_element["objects"],
|
||||
"size_mismatch": size_mismatch,
|
||||
}
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user