From 79bb8b08ad1036b5ffec212ef8d9c1d45e1fbcfe Mon Sep 17 00:00:00 2001 From: TheM14 Date: Thu, 3 Sep 2026 02:59:45 +0800 Subject: [PATCH] Add experimental JOC binaural mode controls --- README.en.md | 12 ++++++++++++ README.md | 12 ++++++++++++ main.py | 13 ++++++++++++- src/adm_assemble.py | 12 ++++++++---- src/adm_atmos.py | 25 ++++++++++++++++++++++--- 5 files changed, 66 insertions(+), 8 deletions(-) diff --git a/README.en.md b/README.en.md index 04238aa..091d50f 100644 --- a/README.en.md +++ b/README.en.md @@ -95,6 +95,18 @@ python main.py input.m4a --metadata-cache metadata_cache python main.py input.m4a --metadata-dir metadata_cache ``` +### Experimental binaural mode settings for JOC objects + +The binaural mode written here is a user-selected, experimental rendering hint for downstream ADM renderers. It is **not original binaural metadata extracted or recovered from the input E-AC-3 JOC bitstream**, nor does it represent the original mix's per-object binaural settings. The selected mode is applied uniformly to all 15 JOC objects; the default `unspecified` is this tool's default, not a mode detected in the source file. + +Use `--joc-binaural-mode off|near|far|mid|unspecified` to select a mode, encoded as `0|1|2|3|4` respectively. The default is `unspecified`: + +```powershell +python main.py input.m4a --joc-binaural-mode mid +``` + +This option only sets the low 3 binaural-render-mode bits of the last 15 JOC object entries in ADM BWF DBMD segment 10, leaving the first 10 bed entries unchanged. It does not change PCM, object trajectories, or direct speaker rendering, and does not itself produce binaural stereo audio. The adjacent `.report.json` records the mode name and value in `joc_binaural_mode` and `joc_binaural_mode_value`; both are `null` for direct speaker output, where the option does not apply. + ### OAMD time alignment Object trajectories and direct speaker rendering both default to a metadata delay of `1473 samples`. This value describes the theoretical mapping between decoder-output PCM and OAMD updates. The speaker renderer retains its existing 32-sample control block, so the default update lands on effective block boundary `1472`: diff --git a/README.md b/README.md index 9400a63..1a506db 100644 --- a/README.md +++ b/README.md @@ -95,6 +95,18 @@ python main.py input.m4a --metadata-cache metadata_cache python main.py input.m4a --metadata-dir metadata_cache ``` +### 实验性 JOC 对象双耳模式设置 + +这里写入的双耳模式是用户手动指定、供下游 ADM 渲染器使用的实验性渲染提示,**不是从输入 E-AC-3 JOC 码流中提取或还原的原始双耳元数据**,也不代表原始混音中各对象的双耳设置。所选模式会统一应用到 15 个 JOC 对象;默认 `unspecified` 只是本工具的默认值,并非从源文件检测到的模式。 + +使用 `--joc-binaural-mode off|near|far|mid|unspecified` 选择模式,编码分别为 `0|1|2|3|4`,默认 `unspecified`: + +```powershell +python main.py input.m4a --joc-binaural-mode mid +``` + +此选项仅设置 ADM BWF 的 DBMD segment 10 中后 15 个 JOC 对象的 binaural render mode 低 3 bit;前 10 个 bed 保持不变。它不改变 PCM、对象轨迹或直接扬声器渲染,也不直接生成双耳立体声音频。输出旁的 `.report.json` 用 `joc_binaural_mode` 和 `joc_binaural_mode_value` 记录模式名称与数值;直接扬声器输出时两者为 `null`,表示不适用。 + ### OAMD 时间对齐 对象轨迹和直接扬声器渲染的 metadata delay 默认均为 `1473 samples`。该值描述 decoder 输出 PCM 与 OAMD 更新之间的理论时间映射;扬声器 renderer 仍使用现有的 32-sample control block,因此默认更新的实际 block boundary 为 `1472`: diff --git a/main.py b/main.py index 2e55e10..c88a253 100644 --- a/main.py +++ b/main.py @@ -20,6 +20,7 @@ if str(SOURCE_DIR) not in sys.path: import numpy as np import adm_assemble +import adm_atmos from adm_validate import validate from metadata import DirectPayloadIndex, PayloadIndex, write_summary import oamd_tracks @@ -288,6 +289,11 @@ def build_parser(): parser.add_argument("--duration", type=float, help="只处理开头指定秒数") parser.add_argument("--object-delay-samples", type=int, default=1473, help="可选的对象 PCM/OAMD 时间补偿,默认 1473 samples") + parser.add_argument( + "--joc-binaural-mode", choices=tuple(adm_atmos.JOC_BINAURAL_MODES), + default=adm_atmos.JOC_BINAURAL_MODE_DEFAULT, + help="实验性 ADM DBMD JOC 对象双耳模式:off=0、near=1、far=2、mid=3、" + "unspecified=4(默认);不改变 PCM 或直接扬声器渲染") parser.add_argument("--trajectory-mode", choices=("compact", "dense64"), default="compact", help="对象轨迹表示;compact 用长线性插值压缩 AXML,dense64 保留逐 64-sample 块") parser.add_argument("--ffmpeg", default=os.environ.get("FFMPEG", "ffmpeg")) @@ -431,7 +437,9 @@ def main(argv=None): info = (f"speaker layout={speaker_name}, format={speaker_actual_format}, " f"peak={speaker_clip_info['peak']:.9g}") else: - master = adm_assemble.StreamingMaster(output, duration_sec, rate=RATE) + master = adm_assemble.StreamingMaster( + output, duration_sec, rate=RATE, + joc_binaural_mode=adm_atmos.JOC_BINAURAL_MODES[args.joc_binaural_mode]) try: render_seconds, renderer_backend, render_breakdown = timed_call( timings, "render_and_stream", variant_call, @@ -474,6 +482,9 @@ def main(argv=None): "gain_float32": float(gain), "object_delay_samples": None if speaker_mode else args.object_delay_samples, "trajectory_mode": None if speaker_mode else args.trajectory_mode, + "joc_binaural_mode": None if speaker_mode else args.joc_binaural_mode, + "joc_binaural_mode_value": (None if speaker_mode else + adm_atmos.JOC_BINAURAL_MODES[args.joc_binaural_mode]), "render_seconds": render_seconds, "render_breakdown": render_breakdown, "renderer_backend": renderer_backend, diff --git a/src/adm_assemble.py b/src/adm_assemble.py index fbd8e38..3752769 100644 --- a/src/adm_assemble.py +++ b/src/adm_assemble.py @@ -14,7 +14,7 @@ import adm_atmos def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None, - duration_sec=None, rate=48000): + duration_sec=None, rate=48000, joc_binaural_mode=4): """16ch f32 交织 raw → 25ch ADM BWF(空 7.1.2 bed + LFE + 15 对象)。 raw16: (n, 16) 交织(ch0 = LFE,ch1-15 = 对象)。 @@ -54,7 +54,8 @@ def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None, kf_tracks.append(("JOC_Object_%d" % (oi + 1), [(0.0, 0.0, 0.0, 0.0, max(duration_sec, 1e-6))])) adm_atmos.build_master(out_path, BedView(), ObjView(), kf_tracks, - duration_sec, rate=rate) + duration_sec, rate=rate, + joc_binaural_mode=joc_binaural_mode) # 及时释放 Windows 文件句柄,允许 TemporaryDirectory 删除中间 raw。 raw._mmap.close() return out_path @@ -69,12 +70,14 @@ class StreamingMaster: channels 10..24. """ - def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072): + def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072, + joc_binaural_mode=4): if block_samples < 1536: raise ValueError("block_samples must be at least one E-AC-3 frame") self.out_path = os.fspath(out_path) self.duration_sec = float(duration_sec) self.rate = int(rate) + self.joc_binaural_mode = joc_binaural_mode self._sink = adm_atmos.Sink25(self.out_path, 25, self.rate) self._buffer = np.empty((int(block_samples), 25), dtype=np.float32) self._used = 0 @@ -112,7 +115,8 @@ class StreamingMaster: import adm_serializer axml = adm_serializer.build_axml(kf_tracks, self.duration_sec) chna = adm_atmos.build_chna() - dbmd = adm_atmos.build_dbmd(25) + dbmd = adm_atmos.build_dbmd( + 25, joc_binaural_mode=self.joc_binaural_mode) trajectory_blocks = sum(len(track[1]) for track in kf_tracks) self.metadata_info = { "axml_bytes": len(axml), diff --git a/src/adm_atmos.py b/src/adm_atmos.py index 9d52997..ea282ed 100644 --- a/src/adm_atmos.py +++ b/src/adm_atmos.py @@ -2,6 +2,7 @@ 输出由 10 声道 7.1.2 bed 和 15 路对象组成;RF64 尺寸字段在写入完成后回填。 """ +import operator import struct import numpy as np import xml.etree.ElementTree as ET @@ -21,6 +22,14 @@ BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0), (-1.0, -1.0, 0.0), (1.0, -1.0, 0.0), (-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)] N_OBJ = 15 +JOC_BINAURAL_MODES = { + "off": 0, + "near": 1, + "far": 2, + "mid": 3, + "unspecified": 4, +} +JOC_BINAURAL_MODE_DEFAULT = "unspecified" def q_to_adm_xyz(q1, q2, q3): posX = min(1.0, round(q1 * 62 / 32767.0) / 62.0) @@ -186,7 +195,11 @@ def _checksum(seg): s += b return (~s + 1) & 0xFF -def build_dbmd(object_count=25): +def build_dbmd(object_count=25, joc_binaural_mode=4): + """仅覆盖 segment 10 中 JOC object slots 10..24 的 mode 低 3 bit。""" + mode = operator.index(joc_binaural_mode) + if mode not in JOC_BINAURAL_MODES.values(): + raise ValueError(f"invalid JOC binaural render mode: {mode}") out = bytearray(struct.pack("