Add experimental JOC binaural mode controls

This commit is contained in:
2026-09-03 02:59:44 +08:00
parent 887b3e317f
commit 329445ed25
5 changed files with 66 additions and 8 deletions
+12
View File
@@ -95,6 +95,18 @@ python main.py input.m4a --metadata-cache metadata_cache
python main.py input.m4a --metadata-dir metadata_cache
```
### Experimental binaural mode settings for JOC objects
The binaural mode written here is a user-selected, experimental rendering hint for downstream ADM renderers. It is **not original binaural metadata extracted or recovered from the input E-AC-3 JOC bitstream**, nor does it represent the original mix's per-object binaural settings. The selected mode is applied uniformly to all 15 JOC objects; the default `unspecified` is this tool's default, not a mode detected in the source file.
Use `--joc-binaural-mode off|near|far|mid|unspecified` to select a mode, encoded as `0|1|2|3|4` respectively. The default is `unspecified`:
```powershell
python main.py input.m4a --joc-binaural-mode mid
```
This option only sets the low 3 binaural-render-mode bits of the last 15 JOC object entries in ADM BWF DBMD segment 10, leaving the first 10 bed entries unchanged. It does not change PCM, object trajectories, or direct speaker rendering, and does not itself produce binaural stereo audio. The adjacent `.report.json` records the mode name and value in `joc_binaural_mode` and `joc_binaural_mode_value`; both are `null` for direct speaker output, where the option does not apply.
### OAMD time alignment
Object trajectories and direct speaker rendering both default to a metadata delay of `1473 samples`. This value describes the theoretical mapping between decoder-output PCM and OAMD updates. The speaker renderer retains its existing 32-sample control block, so the default update lands on effective block boundary `1472`:
+12
View File
@@ -95,6 +95,18 @@ python main.py input.m4a --metadata-cache metadata_cache
python main.py input.m4a --metadata-dir metadata_cache
```
### 实验性 JOC 对象双耳模式设置
这里写入的双耳模式是用户手动指定、供下游 ADM 渲染器使用的实验性渲染提示,**不是从输入 E-AC-3 JOC 码流中提取或还原的原始双耳元数据**,也不代表原始混音中各对象的双耳设置。所选模式会统一应用到 15 个 JOC 对象;默认 `unspecified` 只是本工具的默认值,并非从源文件检测到的模式。
使用 `--joc-binaural-mode off|near|far|mid|unspecified` 选择模式,编码分别为 `0|1|2|3|4`,默认 `unspecified`:
```powershell
python main.py input.m4a --joc-binaural-mode mid
```
此选项仅设置 ADM BWF 的 DBMD segment 10 中后 15 个 JOC 对象的 binaural render mode 低 3 bit;前 10 个 bed 保持不变。它不改变 PCM、对象轨迹或直接扬声器渲染,也不直接生成双耳立体声音频。输出旁的 `.report.json` 用 `joc_binaural_mode` 和 `joc_binaural_mode_value` 记录模式名称与数值;直接扬声器输出时两者为 `null`,表示不适用。
### OAMD 时间对齐
对象轨迹和直接扬声器渲染的 metadata delay 默认均为 `1473 samples`。该值描述 decoder 输出 PCM 与 OAMD 更新之间的理论时间映射;扬声器 renderer 仍使用现有的 32-sample control block,因此默认更新的实际 block boundary 为 `1472`:
+12 -1
View File
@@ -20,6 +20,7 @@ if str(SOURCE_DIR) not in sys.path:
import numpy as np
import adm_assemble
import adm_atmos
from adm_validate import validate
from metadata import DirectPayloadIndex, PayloadIndex, write_summary
import oamd_tracks
@@ -288,6 +289,11 @@ def build_parser():
parser.add_argument("--duration", type=float, help="只处理开头指定秒数")
parser.add_argument("--object-delay-samples", type=int, default=1473,
help="可选的对象 PCM/OAMD 时间补偿,默认 1473 samples")
parser.add_argument(
"--joc-binaural-mode", choices=tuple(adm_atmos.JOC_BINAURAL_MODES),
default=adm_atmos.JOC_BINAURAL_MODE_DEFAULT,
help="实验性 ADM DBMD JOC 对象双耳模式:off=0、near=1、far=2、mid=3、"
"unspecified=4(默认);不改变 PCM 或直接扬声器渲染")
parser.add_argument("--trajectory-mode", choices=("compact", "dense64"), default="compact",
help="对象轨迹表示;compact 用长线性插值压缩 AXML,dense64 保留逐 64-sample 块")
parser.add_argument("--ffmpeg", default=os.environ.get("FFMPEG", "ffmpeg"))
@@ -431,7 +437,9 @@ def main(argv=None):
info = (f"speaker layout={speaker_name}, format={speaker_actual_format}, "
f"peak={speaker_clip_info['peak']:.9g}")
else:
master = adm_assemble.StreamingMaster(output, duration_sec, rate=RATE)
master = adm_assemble.StreamingMaster(
output, duration_sec, rate=RATE,
joc_binaural_mode=adm_atmos.JOC_BINAURAL_MODES[args.joc_binaural_mode])
try:
render_seconds, renderer_backend, render_breakdown = timed_call(
timings, "render_and_stream", variant_call,
@@ -474,6 +482,9 @@ def main(argv=None):
"gain_float32": float(gain),
"object_delay_samples": None if speaker_mode else args.object_delay_samples,
"trajectory_mode": None if speaker_mode else args.trajectory_mode,
"joc_binaural_mode": None if speaker_mode else args.joc_binaural_mode,
"joc_binaural_mode_value": (None if speaker_mode else
adm_atmos.JOC_BINAURAL_MODES[args.joc_binaural_mode]),
"render_seconds": render_seconds,
"render_breakdown": render_breakdown,
"renderer_backend": renderer_backend,
+8 -4
View File
@@ -14,7 +14,7 @@ import adm_atmos
def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None,
duration_sec=None, rate=48000):
duration_sec=None, rate=48000, joc_binaural_mode=4):
"""16ch f32 交织 raw → 25ch ADM BWF(空 7.1.2 bed + LFE + 15 对象)。
raw16: (n, 16) 交织(ch0 = LFE,ch1-15 = 对象)。
@@ -54,7 +54,8 @@ def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None,
kf_tracks.append(("JOC_Object_%d" % (oi + 1),
[(0.0, 0.0, 0.0, 0.0, max(duration_sec, 1e-6))]))
adm_atmos.build_master(out_path, BedView(), ObjView(), kf_tracks,
duration_sec, rate=rate)
duration_sec, rate=rate,
joc_binaural_mode=joc_binaural_mode)
# 及时释放 Windows 文件句柄,允许 TemporaryDirectory 删除中间 raw。
raw._mmap.close()
return out_path
@@ -69,12 +70,14 @@ class StreamingMaster:
channels 10..24.
"""
def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072):
def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072,
joc_binaural_mode=4):
if block_samples < 1536:
raise ValueError("block_samples must be at least one E-AC-3 frame")
self.out_path = os.fspath(out_path)
self.duration_sec = float(duration_sec)
self.rate = int(rate)
self.joc_binaural_mode = joc_binaural_mode
self._sink = adm_atmos.Sink25(self.out_path, 25, self.rate)
self._buffer = np.empty((int(block_samples), 25), dtype=np.float32)
self._used = 0
@@ -112,7 +115,8 @@ class StreamingMaster:
import adm_serializer
axml = adm_serializer.build_axml(kf_tracks, self.duration_sec)
chna = adm_atmos.build_chna()
dbmd = adm_atmos.build_dbmd(25)
dbmd = adm_atmos.build_dbmd(
25, joc_binaural_mode=self.joc_binaural_mode)
trajectory_blocks = sum(len(track[1]) for track in kf_tracks)
self.metadata_info = {
"axml_bytes": len(axml),
+22 -3
View File
@@ -2,6 +2,7 @@
输出由 10 声道 7.1.2 bed 和 15 路对象组成;RF64 尺寸字段在写入完成后回填。
"""
import operator
import struct
import numpy as np
import xml.etree.ElementTree as ET
@@ -21,6 +22,14 @@ BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0),
(-1.0, -1.0, 0.0), (1.0, -1.0, 0.0), (-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
N_OBJ = 15
JOC_BINAURAL_MODES = {
"off": 0,
"near": 1,
"far": 2,
"mid": 3,
"unspecified": 4,
}
JOC_BINAURAL_MODE_DEFAULT = "unspecified"
def q_to_adm_xyz(q1, q2, q3):
posX = min(1.0, round(q1 * 62 / 32767.0) / 62.0)
@@ -186,7 +195,11 @@ def _checksum(seg):
s += b
return (~s + 1) & 0xFF
def build_dbmd(object_count=25):
def build_dbmd(object_count=25, joc_binaural_mode=4):
"""仅覆盖 segment 10 中 JOC object slots 10..24 的 mode 低 3 bit。"""
mode = operator.index(joc_binaural_mode)
if mode not in JOC_BINAURAL_MODES.values():
raise ValueError(f"invalid JOC binaural render mode: {mode}")
out = bytearray(struct.pack("<I", 0x01000006))
dd = bytearray(96)
dd[1] = 0x47
@@ -209,6 +222,12 @@ def build_dbmd(object_count=25):
ob[4] = object_count
for i in range(5 + 262, len(ob)):
ob[i] = 0x84
# sync (4), count (2), reserved (1), nine 15-byte config trims,
# then one trim-bypass byte per track before the headphone modes.
# Preserve the existing template's bed fields and trailing bytes.
object_modes = 4 + 2 + 1 + 9 * 15 + object_count
for i in range(10, min(object_count, 10 + N_OBJ)):
ob[object_modes + i] = (ob[object_modes + i] & 0xF8) | mode
out.append(10); out += struct.pack("<H", len(ob)); out += bytes(ob)
out.append(_checksum(ob))
out += b"\x00\x00"
@@ -252,7 +271,7 @@ class Sink25:
self.fp.close()
def build_master(out_path, bed_mm, obj_mm, kf_tracks, duration_sec, rate=48000,
block=480000):
block=480000, joc_binaural_mode=4):
n = min(bed_mm.shape[0], obj_mm.shape[0])
try:
from . import adm_serializer
@@ -261,7 +280,7 @@ def build_master(out_path, bed_mm, obj_mm, kf_tracks, duration_sec, rate=48000,
serial_axml = adm_serializer.build_axml
axml = serial_axml(kf_tracks, duration_sec)
chna = build_chna()
dbmd = build_dbmd(25)
dbmd = build_dbmd(25, joc_binaural_mode=joc_binaural_mode)
sink = Sink25(out_path, 25, rate)
for st in range(0, n, block):
en = min(n, st + block)