From 91ddfc682344064c6f63e2dcb50196b8b3985bd1 Mon Sep 17 00:00:00 2001 From: TheM14 Date: Thu, 10 Sep 2026 21:39:46 +0800 Subject: [PATCH] Decode E-AC-3 at full dynamic range by default and add -drc-scale/target-level options. --- README.en.md | 36 +++++++++++++++++++- README.md | 19 ++++++++++- main.py | 86 +++++++++++++++++++++++++++++++++++++++++++++--- src/oamd_bits.py | 25 +++----------- 4 files changed, 139 insertions(+), 27 deletions(-) diff --git a/README.en.md b/README.en.md index 3adfebe..afd854f 100644 --- a/README.en.md +++ b/README.en.md @@ -44,7 +44,7 @@ The Python and C++ backends follow the same mathematics for JOC object reconstru - NumPy 1.24+ - h5py 3.8+ - SciPy 1.10+ -- A standalone FFmpeg executable; `ffmpeg-python` is not required. FFmpeg is discovered through `PATH` by default or selected with `--ffmpeg` +- A standalone FFmpeg executable; `ffmpeg-python` is not required. FFmpeg is discovered through `PATH` by default or selected with `--ffmpeg`. On startup the decoder options are probed with `ffmpeg -h decoder=eac3`: a missing E-AC-3 decoder or `-drc_scale` is a hard error, while a missing `-target_level` only fails when `--eac3-target-level` is used - Optional: CMake and a C++20 toolchain to build the native core Install the Python dependency in a project-specific environment: @@ -146,6 +146,40 @@ python main.py input.m4a --metadata-cache metadata_cache python main.py input.m4a --metadata-dir metadata_cache ``` +### E-AC-3 decode-side dynamic range and level + +By default FFmpeg applies the stream `dynrng` dynamic range compression when +decoding E-AC-3 (`-drc_scale 1`). The core 5.1 PCM is the input of JOC object +reconstruction, and `dynrng` is playback-time gain, so it is inherited linearly +by every object and every output (ADM, speaker, binaural). This tool therefore +decodes at **full dynamic range** by default: + +```powershell +python main.py input.m4a # default: -drc_scale 0, full range +python main.py input.m4a --eac3-drc-scale 1 # reproduce consumer playback +python main.py input.m4a --eac3-drc-scale 0.5 # apply half of it +python main.py input.m4a --eac3-target-level -27 # dialnorm-referenced level +``` + +- `--eac3-drc-scale` (`0`–`6`, default `0`) maps to FFmpeg `-drc_scale`: the gain + of each E-AC-3 block is `dynrng factor ^ value`. `0` disables DRC, `1` is the + author's intent, and `>1` is asymmetric (loud parts fully compressed, quiet + parts enhanced). +- `--eac3-target-level` (`-31`–`0`, default `0` = off) maps to FFmpeg + `-target_level`: a static per-frame gain of about `target_level - dialnorm` dB, + independent of and stackable with `--eac3-drc-scale`. dialnorm is a per-stream + property (measured Apple Music Atmos streams are about `-18` to `-19` dB, so + `-27` is roughly `8`–`9` dB of attenuation). +- The level change is expected: compared with the FFmpeg default, measured + tracks move by `0` to `-2.15` dB peak and `0` to `-1.69` dB RMS (direction + depends on the stream `dynrng`), so `output_clip.peak` and the PCM24 clipping + decision in `.report.json` change accordingly. +- `--gain-db` is a static gain applied **after** reconstruction (float64 on the + binaural path) and is not the same thing as decode-side DRC, which is + block-varying; do not use `--gain-db` to cancel it. +- The `ffmpeg` field of `.report.json` records the FFmpeg version and the decode + options that were actually passed (`version`, `eac3_decode_options`). + ### Binaural render mode `--binaural-mode off|near|mid|far` selects the binaural render mode; the default diff --git a/README.md b/README.md index cc6b240..69316fe 100644 --- a/README.md +++ b/README.md @@ -46,7 +46,7 @@ OAMD 时间轴和命令行逻辑在 Python 中。 - NumPy 1.24+ - h5py 3.8+ - SciPy 1.10+ -- 独立的 FFmpeg 可执行程序;不需要 `ffmpeg-python`。默认从 `PATH` 查找,也可通过 `--ffmpeg` 指定可执行文件路径 +- 独立的 FFmpeg 可执行程序;不需要 `ffmpeg-python`。默认从 `PATH` 查找,也可通过 `--ffmpeg` 指定可执行文件路径。启动时会探测 `ffmpeg -h decoder=eac3`:缺 E-AC-3 解码器或 `-drc_scale` 直接报错,缺 `-target_level` 只在使用 `--eac3-target-level` 时报错 - 可选:支持 C++20 的 CMake 工具链,用于自行构建原生核 建议在项目专用虚拟环境中安装依赖: @@ -140,6 +140,23 @@ python main.py input.m4a --metadata-cache metadata_cache python main.py input.m4a --metadata-dir metadata_cache ``` +### E-AC-3 解码级动态范围与电平 + +FFmpeg 解码 E-AC-3 时默认施加码流 `dynrng` 动态范围压缩(`-drc_scale 1`)。核心 5.1 PCM 是 JOC 对象重建的输入,而 `dynrng` 属于回放期增益,会被线性继承到全部对象与成品(ADM/扬声器/双耳),因此本工具默认按**全动态范围**解码: + +```powershell +python main.py input.m4a # 默认:-drc_scale 0,全动态范围 +python main.py input.m4a --eac3-drc-scale 1 # 复现消费者回放(码流作者意图) +python main.py input.m4a --eac3-drc-scale 0.5 # 施加一半 +python main.py input.m4a --eac3-target-level -27 # 按码流 dialnorm 归一化电平 +``` + +- `--eac3-drc-scale`(`0`~`6`,默认 `0`)对应 FFmpeg 的 `-drc_scale`:每个 E-AC-3 block 的增益为 `dynrng 因子 ^ 该值`。`0` 关闭 DRC;`1` 为码流作者意图;`>1` 非对称(响处全压、轻处增强)。 +- `--eac3-target-level`(`-31`~`0`,默认 `0` 不施加)对应 FFmpeg 的 `-target_level`:按每帧 dialnorm 施加静态增益,约 `target_level - dialnorm` dB,与 `--eac3-drc-scale` 相互独立、可叠加。dialnorm 是逐码流属性(实测 Apple Music Atmos 流约 `-18`~`-19` dB,故 `-27` 约等于衰减 `8`~`9` dB)。 +- 电平变化是预期的:与 FFmpeg 默认值相比,实测曲目峰值变化 `0`~`-2.15` dB、RMS `0`~`-1.69` dB(方向取决于码流 `dynrng`),`.report.json` 的 `output_clip.peak` 与 int24 削波判定会随之变化。 +- `--gain-db` 是**重建之后**的静态增益(双耳路径 float64),与解码级 DRC 不是一回事;解码级 DRC 是按 block 时变的,不要用 `--gain-db` 去抵消它。 +- `.report.json` 的 `ffmpeg` 字段记录 FFmpeg 版本与实际下发的解码选项(`version`、`eac3_decode_options`)。 + ### 双耳渲染模式 `--binaural-mode off|near|mid|far` 选择双耳渲染模式,默认 `mid`,两种输出共用这一个选项: diff --git a/main.py b/main.py index f5868e3..f207043 100644 --- a/main.py +++ b/main.py @@ -6,6 +6,7 @@ import math import os from pathlib import Path import platform +import re import shutil import subprocess import sys @@ -50,6 +51,9 @@ from variant_error import UnsupportedVariantError, write_variant_report RATE = 48000 FRAME_SAMPLES = 1536 DEFAULT_OUTPUT_DIR = PROJECT_DIR / "output" +EAC3_DRC_SCALE_MAX = 6.0 +EAC3_TARGET_LEVEL_RANGE = (-31, 0) +EAC3_DECODER_OPTION_RE = re.compile(r"(?m)^\s*-([A-Za-z0-9_]+)\s+<") def resolve_output(source, requested=None, speaker_layout=None, *, binaural=False): @@ -183,6 +187,53 @@ def timed_call(timings, name, function, *args, **kwargs): timings[name] = time.perf_counter() - started +def probe_eac3_decoder_options(ffmpeg): + """读取 ``ffmpeg -h decoder=eac3`` 暴露的 AVOption 名。""" + result = subprocess.run( + [ffmpeg, "-hide_banner", "-h", "decoder=eac3"], + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + text=True, encoding="utf-8", errors="replace") + options = frozenset(EAC3_DECODER_OPTION_RE.findall(result.stdout or "")) + # decoder 名不存在时 ffmpeg 依然返回 0,因此以“解析不到任何选项”为失败。 + if not options: + raise RuntimeError( + "无法读取 FFmpeg 的 eac3 解码器选项(ffmpeg -h decoder=eac3);" + "需要带 E-AC-3 解码器的构建") + return options + + +def ffmpeg_version(ffmpeg): + """FFmpeg 版本字符串;探测失败返回空串,不影响渲染。""" + try: + result = subprocess.run( + [ffmpeg, "-hide_banner", "-version"], + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + text=True, encoding="utf-8", errors="replace") + except OSError: + return "" + lines = (result.stdout or "").splitlines() + line = lines[0].strip() if lines else "" + prefix = "ffmpeg version " + return line[len(prefix):].strip() if line.startswith(prefix) else line + + +def eac3_decode_options(drc_scale, target_level, available): + """构造 ``-i`` 之前的 E-AC-3 解码选项,返回 ``(argv, report 片段)``。""" + if "drc_scale" not in available: + raise RuntimeError( + "FFmpeg 的 eac3 解码器缺少 -drc_scale,无法关闭码流 DRC") + # -drc_scale 始终显式下发:0(全动态范围)不是 ffmpeg 的默认值。 + argv = ["-drc_scale", format(float(drc_scale), ".10g")] + if target_level: + if "target_level" not in available: + raise RuntimeError( + "FFmpeg 的 eac3 解码器不支持 -target_level;请升级 FFmpeg " + "或去掉 --eac3-target-level") + argv += ["-target_level", str(int(target_level))] + applied = {"drc_scale": float(drc_scale), "target_level": int(target_level)} + return argv, applied + + def extract_eac3(ffmpeg, source, target): if source.suffix.lower() in (".eac3", ".ec3"): return source @@ -192,10 +243,10 @@ def extract_eac3(ffmpeg, source, target): return target -def decode_core(ffmpeg, eac3, target, duration_sec=None): +def decode_core(ffmpeg, eac3, target, duration_sec=None, *, options=()): # 5.1(side) 的 f32le 顺序为 FL FR FC LFE SL SR;JOC 使用其中 0,1,2,4,5。 - command = [ffmpeg, "-hide_banner", "-loglevel", "error", "-y", "-i", str(eac3), - "-map", "0:a:0", "-vn"] + command = [ffmpeg, "-hide_banner", "-loglevel", "error", "-y", *options, + "-i", str(eac3), "-map", "0:a:0", "-vn"] if duration_sec is not None: command.extend(["-t", f"{duration_sec:.9f}"]) command.extend(["-ac", "6", "-ar", str(RATE), @@ -480,6 +531,12 @@ def build_parser(): parser.add_argument("--trajectory-mode", choices=("compact", "dense64"), default="compact", help="ADM 对象轨迹表示;直接双耳路径不序列化 AXML") parser.add_argument("--ffmpeg", default=os.environ.get("FFMPEG", "ffmpeg")) + parser.add_argument("--eac3-drc-scale", type=float, default=0.0, + help="E-AC-3 解码器 -drc_scale:0=关闭码流 dynrng(全动态范围)," + "1=码流作者意图,>1 非对称;默认 0") + parser.add_argument("--eac3-target-level", type=int, default=0, + help="E-AC-3 解码器 -target_level:按码流 dialnorm 归一化电平," + "增益约 target_level - dialnorm dB;0=不施加,默认 0") parser.add_argument("--backend", choices=("auto", "native", "python"), default="auto", help="JOC/扬声器 DSP 后端;SOFA 双耳 DSP 当前使用 Python") parser.add_argument("--native-library", type=Path, @@ -568,9 +625,28 @@ def main(argv=None): gain = np.float32(gain_float64) if not math.isfinite(gain_float64) or not np.isfinite(gain): raise ValueError("gain-db 超出支持范围") + if (not math.isfinite(args.eac3_drc_scale) + or not 0.0 <= args.eac3_drc_scale <= EAC3_DRC_SCALE_MAX): + raise ValueError(f"eac3-drc-scale 必须在 0..{EAC3_DRC_SCALE_MAX:g} 之间") + if not (EAC3_TARGET_LEVEL_RANGE[0] <= args.eac3_target_level + <= EAC3_TARGET_LEVEL_RANGE[1]): + raise ValueError("eac3-target-level 必须在 -31..0 之间") binaural_hrtf_input = resolve_binaural_hrtf_input( args, required=binaural_mode and not args.metadata_only) ffmpeg = executable(args.ffmpeg, "FFmpeg") + decode_options = () + decode_option_info = None + if not args.metadata_only: + available = probe_eac3_decoder_options(ffmpeg) + decode_options, applied = eac3_decode_options( + args.eac3_drc_scale, args.eac3_target_level, available) + decode_option_info = { + "version": ffmpeg_version(ffmpeg), + "eac3_decode_options": applied, + } + print(f"[decode] ffmpeg {decode_option_info['version']} " + f"drc_scale={args.eac3_drc_scale:g} " + f"target_level={args.eac3_target_level}", flush=True) total_started = time.perf_counter() timings = {} @@ -606,7 +682,8 @@ def main(argv=None): bed_path = timed_call( timings, "decode_core", decode_core, - ffmpeg, eac3, temp_dir / "core51_f32le.raw", duration_sec) + ffmpeg, eac3, temp_dir / "core51_f32le.raw", duration_sec, + options=decode_options) raw_path = (output.with_name(output.name + ".objects16.f32le") if args.keep_raw else None) master = None @@ -887,6 +964,7 @@ def main(argv=None): "sha256": output_sha, "python": platform.python_version(), "numpy": np.__version__, + "ffmpeg": decode_option_info, } report_path = Path(str(output) + ".report.json") report_path.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8") diff --git a/src/oamd_bits.py b/src/oamd_bits.py index 1dcae27..011fea5 100644 --- a/src/oamd_bits.py +++ b/src/oamd_bits.py @@ -1,26 +1,9 @@ """OAMD 位载荷 → 16 个对象槽的 q1/q2/q3 增量状态。 -解析依据 ETSI TS 103 420 V1.2.1(Backwards-compatible object audio carriage -using Enhanced AC-3)clause 5: - -* 5.5.2 ``object_audio_metadata_payload()``:版本、对象数、program assignment、 - element 目录; -* 5.5.3 ``program_assignment()``:bed / ISF / dynamic 三类对象及其数量; -* 5.5.4 ``oa_element_md()``:element id、字节长度、alternate data id; -* 5.5.5/5.5.6/5.5.7 object_element/md_update_info/block_update_info: - ``start_sample = sample_offset + 32 * block_offset_factor``; -* 5.5.9/5.5.10/5.5.11 object_info_block/object_basic_info/object_render_info: - 逐对象的位置字段;bed 与 ISF 对象不携带 render info; -* 5.6.1.1.8~5.6.1.1.11 pos3D_X/Y/Z:横向/纵向 62 格、高度 15 格 + 符号位。 - -槽 0 是 bed/LFE;槽 1..15 对应输出 ch1..15 的对象元数据。bed/ISF 对象与 -``b_object_not_active`` 对象没有位置字段(5.5.9),保持上一帧位置。 - -element 目录由声明长度驱动,因此 alternate_object_data_present、任意对象数、 -多 element(trim/extended/未知 id 按声明边界跳过)都能解析。个别编码器写出的 -``oa_element_size`` 比实际内容短(例如 Dolby 测试信号 -``Audio_ID_..._special_6ch_..._ddp_joc.mp4`` 的 ID11 少 2 字节),此时以结构解析 -出的实际位置为准,并把差异放进 ``diagnostics``,不当作变体错误。 +依据 ETSI TS 103 420 V1.2.1 clause 5。槽 0 为 bed/LFE,槽 1..15 为输出 +ch1..15;bed/ISF/未激活对象没有位置字段,保持上一帧位置。element 目录按声明长度 +驱动,未知 element 按边界跳过;个别编码器的 ``oa_element_size`` 比实际内容短时以 +结构解析为准,差异记入 ``diagnostics``,不算变体错误。 """ import numpy as np