Initial public release of JustOneCacophony

This commit is contained in:
TheM14
2026-09-01 16:31:38 +08:00
commit 6536782a0f
38 changed files with 9073 additions and 0 deletions
+141
View File
@@ -0,0 +1,141 @@
"""把 LFE、15 路对象 PCM 和对象轨迹组装为 ADM BWF。
固定输出契约:
EAC3JOC 重放输出 = 16ch(ch0 = LFE + ch1-15 = 15 对象);
最终 ADM BWF = 7.1.2 bed(L R C Ls Rs Lb Rb + LFE + Ltf Rtf = 10ch)
—— 除 LFE 外全部静音;
15 对象 = ch1-15 直接填充对象轨;轨迹 = OAMD(q1/q2/q3 → xyz)。
"""
import os
import numpy as np
import adm_atmos
def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None,
duration_sec=None, rate=48000):
"""16ch f32 交织 raw → 25ch ADM BWF(空 7.1.2 bed + LFE + 15 对象)。
raw16: (n, 16) 交织(ch0 = LFE,ch1-15 = 对象)。
scale: 1.0 = 默认 0 dB,不附加输出缩放。该参数与 joc_clipgain 无关;
主命令行已在渲染阶段应用用户增益,因此这里传 1.0。
kf_tracks: 可选轨迹关键帧(OAMD 输出,格式 [(obj_id, [(t, x, y, z), ...]), ...]);
缺省 = 静止参考位置(adm_atmos 默认)。
"""
raw = np.memmap(raw16_path, dtype=np.float32, mode="r")
n = len(raw) // 16
raw = raw[:n * 16].reshape(-1, 16)
if duration_sec is None:
duration_sec = n / rate
# 惰性视图:adm_atmos 按块读取,避免全片 25ch 在内存中展开。
class BedView:
shape = (n, 10)
def __getitem__(self, key):
src = np.asarray(raw[key], dtype=np.float32)
one = src.ndim == 1
if one:
src = src[None, :]
out = np.zeros((len(src), 10), dtype=np.float32)
out[:, 3] = np.multiply(src[:, 0], np.float32(scale), dtype=np.float32)
return out[0] if one else out
class ObjView:
shape = (n, 16)
def __getitem__(self, key):
return np.multiply(np.asarray(raw[key], dtype=np.float32),
np.float32(scale), dtype=np.float32)
if kf_tracks is None:
kf_tracks = []
for oi in range(15):
# 静止参考位置(q1=q2=q3=0 → 原点;实际坐标按 OAMD 输出填入)
kf_tracks.append(("JOC_Object_%d" % (oi + 1),
[(0.0, 0.0, 0.0, 0.0, max(duration_sec, 1e-6))]))
adm_atmos.build_master(out_path, BedView(), ObjView(), kf_tracks,
duration_sec, rate=rate)
# 及时释放 Windows 文件句柄,允许 TemporaryDirectory 删除中间 raw。
raw._mmap.close()
return out_path
class StreamingMaster:
"""Incrementally write renderer frames into the final 25-channel ADM BWF.
This removes the default 16-channel float32 intermediate file. The mapping
remains identical to :func:`assemble_from_raw`: bed channel 3 receives LFE,
bed channels 0..2/4..9 are silent, and output objects 1..15 map to ADM
channels 10..24.
"""
def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072):
if block_samples < 1536:
raise ValueError("block_samples must be at least one E-AC-3 frame")
self.out_path = os.fspath(out_path)
self.duration_sec = float(duration_sec)
self.rate = int(rate)
self._sink = adm_atmos.Sink25(self.out_path, 25, self.rate)
self._buffer = np.empty((int(block_samples), 25), dtype=np.float32)
self._used = 0
self._finalized = False
def _flush(self):
if self._used:
self._sink.write_block(self._buffer[:self._used])
self._used = 0
def write_frame(self, pcm16):
pcm = np.asarray(pcm16, dtype=np.float32)
if pcm.shape != (16, 1536):
raise ValueError(f"renderer frame must be (16,1536), got {pcm.shape}")
source = 0
while source < 1536:
available = len(self._buffer) - self._used
count = min(available, 1536 - source)
target = self._buffer[self._used:self._used + count]
target.fill(0.0)
target[:, 3] = pcm[0, source:source + count]
target[:, 10:25] = pcm[1:16, source:source + count].T
self._used += count
source += count
if self._used == len(self._buffer):
self._flush()
def finalize(self, kf_tracks):
if self._finalized:
raise RuntimeError("StreamingMaster already finalized")
self._flush()
try:
from . import adm_serializer
except ImportError:
import adm_serializer
axml = adm_serializer.build_axml(kf_tracks, self.duration_sec)
chna = adm_atmos.build_chna()
dbmd = adm_atmos.build_dbmd(25)
trajectory_blocks = sum(len(track[1]) for track in kf_tracks)
self.metadata_info = {
"axml_bytes": len(axml),
"trajectory_blocks": trajectory_blocks,
"chna_bytes": len(chna),
"dbmd_bytes": len(dbmd),
}
self._sink.finalize(axml, chna, dbmd)
self._finalized = True
print(f"master25 -> {self.out_path} ({self.duration_sec:.2f}s, 25ch, "
f"axml={len(axml)}B, chna={len(chna)}B, dbmd={len(dbmd)}B)")
return self.out_path
def abort(self):
if self._finalized:
return
sink = getattr(self, "_sink", None)
fp = getattr(sink, "fp", None)
if fp is not None and not fp.closed:
fp.close()
def __del__(self):
try:
self.abort()
except Exception:
pass
+273
View File
@@ -0,0 +1,273 @@
"""生成 25 声道 RF64 ADM BWF 及其 axml、chna、dbmd 元数据。
输出由 10 声道 7.1.2 bed 和 15 路对象组成;RF64 尺寸字段在写入完成后回填。
"""
import struct
import numpy as np
import xml.etree.ElementTree as ET
NS = "urn:ebu:metadata-schema:ebuCore_2016"
XSI = "http://www.w3.org/2001/XMLSchema-instance"
BED_NAMES = ["RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter",
"RoomCentricLFE", "RoomCentricLeftSideSurround",
"RoomCentricRightSideSurround", "RoomCentricLeftRearSurround",
"RoomCentricRightRearSurround", "RoomCentricLeftTopSurround",
"RoomCentricRightTopSurround"]
BED_LABELS = ["RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss", "RC_Rss",
"RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"]
BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0),
(-1.0, 1.0, -1.0), (-1.0, 0.0, 0.0), (1.0, 0.0, 0.0),
(-1.0, -1.0, 0.0), (1.0, -1.0, 0.0), (-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
N_OBJ = 15
def q_to_adm_xyz(q1, q2, q3):
posX = min(1.0, round(q1 * 62 / 32767.0) / 62.0)
posY = min(1.0, round(q2 * 62 / 32767.0) / 62.0)
posZ = round(q3 * 15 / 32767.0) / 15.0
posZ = max(-1.0, min(1.0, posZ))
return posX * 2 - 1, 1 - posY * 2, posZ
def ts(seconds):
s = int(seconds)
frac = int(round((seconds - s) * 100000))
if frac >= 100000:
s += 1; frac = 0
return f"{s // 3600:02d}:{s % 3600 // 60:02d}:{s % 60:02d}.{frac:05d}"
def sub(parent, tag, attrib=None, text=None):
e = ET.SubElement(parent, tag)
if attrib:
for k, v in attrib.items():
e.set(k, v)
if text is not None:
e.text = text
return e
def add_refs(parent, tag, ids):
for i in ids:
sub(parent, tag, text=i)
def obj_block(cf, bid, t, x, y, z, dur, interpolation=0.0):
b = sub(cf, "audioBlockFormat", {
"audioBlockFormatID": bid, "rtime": ts(t), "duration": ts(dur)})
sub(b, "cartesian", text="1")
for c, v in (("X", x), ("Y", y), ("Z", z)):
if c == "Z" and v == 0:
continue
p = sub(b, "position", {"coordinate": c})
p.text = f"{v:.10f}"
sub(b, "jumpPosition", {"interpolationLength": f"{interpolation:.5f}"}, text="1")
def build_axml(obj_tracks, duration_sec):
adm = ET.Element("ebuCoreMain", {
"xmlns": NS, "xmlns:xsi": XSI,
"xsi:schemaLocation": f"{NS} ebucore.xsd", "lang": "en"})
core = sub(adm, "coreMetadata")
fmt = sub(core, "format")
af = sub(fmt, "audioFormatExtended")
prog = sub(af, "audioProgramme", {
"audioProgrammeID": "APR_1001", "audioProgrammeName": "EAC3JOC_Export",
"start": ts(0), "end": ts(duration_sec)})
add_refs(prog, "audioContentIDRef", ("ACO_1001", "ACO_1002"))
bc = sub(af, "audioContent", {"audioContentID": "ACO_1001",
"audioContentName": "EAC3JOC_Master_Content"})
add_refs(bc, "audioObjectIDRef", ["AO_1001"])
sub(bc, "dialogue", {"mixedContentKind": "0"})
oc = sub(af, "audioContent", {"audioContentID": "ACO_1002",
"audioContentName": "Objects"})
add_refs(oc, "audioObjectIDRef", ["AO_%04x" % (0x100b + i) for i in range(N_OBJ)])
sub(oc, "dialogue", {"mixedContentKind": "0"})
bed_o = sub(af, "audioObject", {"audioObjectID": "AO_1001", "audioObjectName": "Bed",
"start": ts(0), "duration": ts(duration_sec)})
sub(bed_o, "audioPackFormatIDRef", text="AP_00011001")
add_refs(bed_o, "audioTrackUIDRef", ["ATU_%08x" % (i + 1) for i in range(10)])
for i in range(N_OBJ):
o = sub(af, "audioObject", {"audioObjectID": "AO_%04x" % (0x100b + i),
"audioObjectName": f"Audio Object {i+1}",
"start": ts(0), "duration": ts(duration_sec)})
sub(o, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
add_refs(o, "audioTrackUIDRef", ["ATU_%08x" % (i + 11)])
bp = sub(af, "audioPackFormat", {"audioPackFormatID": "AP_00011001",
"audioPackFormatName": "EAC3JOCBedPack",
"typeDefinition": "DirectSpeakers", "typeLabel": "0001"})
add_refs(bp, "audioChannelFormatIDRef", ["AC_0001%04x" % (0x1001 + i) for i in range(10)])
for i in range(N_OBJ):
pk = sub(af, "audioPackFormat", {"audioPackFormatID": "AP_0003%04x" % (0x1001 + i),
"audioPackFormatName": f"JOC_Object_{i+1}",
"typeDefinition": "Objects", "typeLabel": "0003"})
add_refs(pk, "audioChannelFormatIDRef", ["AC_0003%04x" % (0x1001 + i)])
for i in range(10):
cf = sub(af, "audioChannelFormat", {"audioChannelFormatID": "AC_0001%04x" % (0x1001 + i),
"audioChannelFormatName": BED_NAMES[i],
"typeDefinition": "DirectSpeakers", "typeLabel": "0001"})
b = sub(cf, "audioBlockFormat", {"audioBlockFormatID": "AB_0001%04x_00000001" % (0x1001 + i)})
sub(b, "cartesian", text="1")
x, y, z = BED_POS[i]
for c, v in (("X", x), ("Y", y), ("Z", z)):
if c == "Z" and v == 0:
continue
p = sub(b, "position", {"coordinate": c})
p.text = f"{v:.10f}"
sub(b, "speakerLabel", text=BED_LABELS[i])
for i, (oname, kfs) in enumerate(obj_tracks):
cf = sub(af, "audioChannelFormat", {"audioChannelFormatID": "AC_0003%04x" % (0x1001 + i),
"audioChannelFormatName": oname,
"typeDefinition": "Objects", "typeLabel": "0003"})
for k, keyframe in enumerate(kfs):
t, x, y, z, dur = keyframe[:5]
interpolation = keyframe[5] if len(keyframe) > 5 else 0.0
obj_block(cf, "AB_0003%04x_%08x" % (0x1001 + i, k + 1),
t, x, y, z, dur, interpolation)
for i in range(10):
t = sub(af, "audioTrackUID", {"UID": "ATU_%08x" % (i + 1),
"bitDepth": "24", "sampleRate": "48000"})
sub(t, "audioTrackFormatIDRef", text="AT_0001%04x_01" % (0x1001 + i))
sub(t, "audioPackFormatIDRef", text="AP_00011001")
for i in range(N_OBJ):
t = sub(af, "audioTrackUID", {"UID": "ATU_%08x" % (i + 11),
"bitDepth": "24", "sampleRate": "48000"})
sub(t, "audioTrackFormatIDRef", text="AT_0003%04x_01" % (0x1001 + i))
sub(t, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
for i in range(10):
tf = sub(af, "audioTrackFormat", {"audioTrackFormatID": "AT_0001%04x_01" % (0x1001 + i),
"audioTrackFormatName": "PCM_" + BED_NAMES[i],
"formatDefinition": "PCM", "formatLabel": "0001"})
sub(tf, "audioStreamFormatIDRef", text="AS_0001%04x" % (0x1001 + i))
for i in range(N_OBJ):
tf = sub(af, "audioTrackFormat", {"audioTrackFormatID": "AT_0003%04x_01" % (0x1001 + i),
"audioTrackFormatName": "PCM_JOC_Object_%d" % (i + 1),
"formatDefinition": "PCM", "formatLabel": "0001"})
sub(tf, "audioStreamFormatIDRef", text="AS_0003%04x" % (0x1001 + i))
for i in range(10):
sf = sub(af, "audioStreamFormat", {"audioStreamFormatID": "AS_0001%04x" % (0x1001 + i),
"audioStreamFormatName": "PCM_" + BED_NAMES[i],
"formatDefinition": "PCM", "formatLabel": "0001"})
sub(sf, "audioChannelFormatIDRef", text="AC_0001%04x" % (0x1001 + i))
sub(sf, "audioPackFormatIDRef", text="AP_00011001")
sub(sf, "audioTrackFormatIDRef", text="AT_0001%04x_01" % (0x1001 + i))
for i in range(N_OBJ):
sf = sub(af, "audioStreamFormat", {"audioStreamFormatID": "AS_0003%04x" % (0x1001 + i),
"audioStreamFormatName": "PCM_JOC_Object_%d" % (i + 1),
"formatDefinition": "PCM", "formatLabel": "0001"})
sub(sf, "audioChannelFormatIDRef", text="AC_0003%04x" % (0x1001 + i))
sub(sf, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
sub(sf, "audioTrackFormatIDRef", text="AT_0003%04x_01" % (0x1001 + i))
return ET.tostring(adm, encoding="utf-8", xml_declaration=True)
def build_chna():
out = bytearray()
out += struct.pack("<HH", 25, 25)
for i in range(10):
out += struct.pack("<H", i + 1)
out += ("ATU_%08x" % (i + 1)).encode()
out += ("AT_0001%04x_01" % (0x1001 + i)).encode()
out += b"AP_00011001" + b"\x00"
for i in range(N_OBJ):
out += struct.pack("<H", i + 11)
out += ("ATU_%08x" % (i + 11)).encode()
out += ("AT_0003%04x_01" % (0x1001 + i)).encode()
out += ("AP_0003%04x" % (0x1001 + i)).encode() + b"\x00"
return bytes(out)
def _checksum(seg):
s = len(seg)
for b in seg:
s += b
return (~s + 1) & 0xFF
def build_dbmd(object_count=25):
out = bytearray(struct.pack("<I", 0x01000006))
dd = bytearray(96)
dd[1] = 0x47
dd[5] = 0x60
dd[8] = 0x24; dd[9] = 0x24
out.append(7); out += struct.pack("<H", 96); out += bytes(dd)
out.append(_checksum(dd))
at = bytearray(248)
c0 = b"Created with EAC3JOC"; c1 = b"EAC3JOC Python Renderer"
at[0:len(c0)] = c0
at[32:32 + len(c1)] = c1
at[96], at[97], at[98] = 2, 1, 0
at[103] = 0x03
at[106] = 0x01
at[111] = 0x22; at[112] = 0xFF
out.append(9); out += struct.pack("<H", 248); out += bytes(at)
out.append(_checksum(at))
ob = bytearray(5 + 262 + object_count)
ob[0:4] = struct.pack("<I", 0xF8726FBD)
ob[4] = object_count
for i in range(5 + 262, len(ob)):
ob[i] = 0x84
out.append(10); out += struct.pack("<H", len(ob)); out += bytes(ob)
out.append(_checksum(ob))
out += b"\x00\x00"
return bytes(out)
class Sink25:
def __init__(self, path, channels, rate):
self.ch = channels; self.rate = rate; self.frames = 0
self.fp = open(path, "wb+")
self.fp.write(b"RF64" + struct.pack("<I", 0xFFFFFFFF) + b"WAVE")
self._chunk(b"ds64", b"\x00" * 64)
self._chunk(b"fmt ", self._fmt())
self._chunk(b"data", b"")
def _chunk(self, cid, body):
self.fp.write(cid + struct.pack("<I", len(body)) + body)
if len(body) & 1:
self.fp.write(b"\x00")
def _fmt(self):
return struct.pack("<HHIIHH", 1, self.ch, self.rate,
self.rate * self.ch * 3, self.ch * 3, 24)
def write_block(self, arr):
arr = arr.reshape(-1, self.ch)
i24 = (np.clip(arr, -1.0, 1.0) * 8388607.0).astype(np.int32)
self.fp.write(i24.view(np.uint8).reshape(-1, 4)[:, :3].tobytes())
self.frames += arr.shape[0]
def finalize(self, axml_bytes, chna_bytes, dbmd_bytes):
data_len = self.frames * self.ch * 3
self._chunk(b"axml", axml_bytes)
self._chunk(b"chna", chna_bytes)
self._chunk(b"dbmd", dbmd_bytes)
self.fp.seek(0, 2); total = self.fp.tell()
self.fp.seek(0); head = self.fp.read()
m = head.find(b"data")
if m >= 0:
self.fp.seek(m + 4); self.fp.write(struct.pack("<I", data_len))
m = head.find(b"ds64")
if m >= 0:
self.fp.seek(m + 8)
self.fp.write(struct.pack("<QQQI", total - 8, data_len, self.frames, 0))
self.fp.flush()
self.fp.close()
def build_master(out_path, bed_mm, obj_mm, kf_tracks, duration_sec, rate=48000,
block=480000):
n = min(bed_mm.shape[0], obj_mm.shape[0])
try:
from . import adm_serializer
except ImportError:
import adm_serializer
serial_axml = adm_serializer.build_axml
axml = serial_axml(kf_tracks, duration_sec)
chna = build_chna()
dbmd = build_dbmd(25)
sink = Sink25(out_path, 25, rate)
for st in range(0, n, block):
en = min(n, st + block)
blk = np.hstack((np.asarray(bed_mm[st:en], dtype=np.float32),
np.asarray(obj_mm[st:en, 1:16], dtype=np.float32)))
sink.write_block(blk)
sink.finalize(axml, chna, dbmd)
print(f"master25 -> {out_path} ({duration_sec:.2f}s, 25ch, axml={len(axml)}B, "
f"chna={len(chna)}B, dbmd={len(dbmd)}B)")
+136
View File
@@ -0,0 +1,136 @@
"""把 7.1.2 bed、15 个对象及其位置轨迹序列化为 ADM axml。
序列化结果采用固定元素顺序、属性顺序和十六进制 ADM 标识符,便于稳定输出和校验。
"""
BED_NAMES = ["RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter", "RoomCentricLFE",
"RoomCentricLeftSideSurround", "RoomCentricRightSideSurround",
"RoomCentricLeftRearSurround", "RoomCentricRightRearSurround",
"RoomCentricLeftTopSurround", "RoomCentricRightTopSurround"]
BED_LABELS = ["RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss", "RC_Rss",
"RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"]
BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0), (-1.0, 1.0, -1.0),
(-1.0, 0.0, 0.0), (1.0, 0.0, 0.0), (-1.0, -1.0, 0.0), (1.0, -1.0, 0.0),
(-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
N_OBJ = 15
def ts(seconds):
s = int(seconds)
frac = int(round((seconds - s) * 100000))
if frac >= 100000:
s += 1; frac = 0
return f"{s // 3600:02d}:{s % 3600 // 60:02d}:{s % 60:02d}.{frac:05d}"
def esc(v):
return (str(v).replace("&", "&amp;").replace("<", "&lt;").replace(">", "&gt;"))
def build_axml(obj_tracks, duration_sec):
"""obj_tracks: [(name, [(rtime, x, y, z, dur), ...]) ×15]"""
w = []
a = w.append
a('<?xml version="1.0" encoding="utf-8"?>')
a('<ebuCoreMain xsi:schemaLocation="urn:ebu:metadata-schema:ebuCore_2016 ebucore.xsd" '
'lang="en" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" '
'xmlns="urn:ebu:metadata-schema:ebuCore_2016">')
a('<coreMetadata><format><audioFormatExtended>')
a(f'<audioProgramme audioProgrammeID="APR_1001" audioProgrammeName="EAC3JOC_Export" '
f'start="{ts(0)}" end="{ts(duration_sec)}">')
a('<audioContentIDRef>ACO_1001</audioContentIDRef>')
a('<audioContentIDRef>ACO_1002</audioContentIDRef>')
a('</audioProgramme>')
a('<audioContent audioContentID="ACO_1001" audioContentName="EAC3JOC_Master_Content">')
a('<audioObjectIDRef>AO_1001</audioObjectIDRef>')
a('<dialogue mixedContentKind="0">2</dialogue>')
a('</audioContent>')
a('<audioContent audioContentID="ACO_1002" audioContentName="Objects">')
for i in range(N_OBJ):
a(f'<audioObjectIDRef>AO_{0x100b + i:04x}</audioObjectIDRef>')
a('<dialogue mixedContentKind="0">2</dialogue>')
a('</audioContent>')
a(f'<audioObject audioObjectID="AO_1001" audioObjectName="Bed" '
f'start="{ts(0)}" duration="{ts(duration_sec)}">')
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
for i in range(10):
a(f'<audioTrackUIDRef>ATU_{i + 1:08x}</audioTrackUIDRef>')
a('</audioObject>')
for i in range(N_OBJ):
a(f'<audioObject audioObjectID="AO_{0x100b + i:04x}" audioObjectName="Audio Object {i+1}" '
f'start="{ts(0)}" duration="{ts(duration_sec)}">')
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
a(f'<audioTrackUIDRef>ATU_{11 + i:08x}</audioTrackUIDRef>')
a('</audioObject>')
a('<audioPackFormat audioPackFormatID="AP_00011001" audioPackFormatName="EAC3JOCBedPack" '
'typeDefinition="DirectSpeakers" typeLabel="0001">')
for i in range(10):
a(f'<audioChannelFormatIDRef>AC_0001{0x1001 + i:04x}</audioChannelFormatIDRef>')
a('</audioPackFormat>')
for i in range(N_OBJ):
a(f'<audioPackFormat audioPackFormatID="AP_0003{0x1001 + i:04x}" '
f'audioPackFormatName="JOC_Object_{i+1}" typeDefinition="Objects" typeLabel="0003">')
a(f'<audioChannelFormatIDRef>AC_0003{0x1001 + i:04x}</audioChannelFormatIDRef>')
a('</audioPackFormat>')
for i in range(10):
a(f'<audioChannelFormat audioChannelFormatID="AC_0001{0x1001 + i:04x}" '
f'audioChannelFormatName="{BED_NAMES[i]}" typeDefinition="DirectSpeakers" typeLabel="0001">')
a(f'<audioBlockFormat audioBlockFormatID="AB_0001{0x1001 + i:04x}_00000001">')
a('<cartesian>1</cartesian>')
x, y, z = BED_POS[i]
a(f'<position coordinate="X">{x:.10f}</position>')
a(f'<position coordinate="Y">{y:.10f}</position>')
if z != 0:
a(f'<position coordinate="Z">{z:.10f}</position>')
a(f'<speakerLabel>{BED_LABELS[i]}</speakerLabel>')
a('</audioBlockFormat>')
a('</audioChannelFormat>')
for i, (oname, kfs) in enumerate(obj_tracks):
a(f'<audioChannelFormat audioChannelFormatID="AC_0003{0x1001 + i:04x}" '
f'audioChannelFormatName="{oname}" typeDefinition="Objects" typeLabel="0003">')
for k, keyframe in enumerate(kfs):
t, x, y, z, dur = keyframe[:5]
interpolation = keyframe[5] if len(keyframe) > 5 else 0.0
a(f'<audioBlockFormat audioBlockFormatID="AB_0003{0x1001 + i:04x}_{k + 1:08x}" '
f'rtime="{ts(t)}" duration="{ts(dur)}">')
a('<cartesian>1</cartesian>')
a(f'<position coordinate="X">{x:.10f}</position>')
a(f'<position coordinate="Y">{y:.10f}</position>')
if z != 0:
a(f'<position coordinate="Z">{z:.10f}</position>')
a(f'<jumpPosition interpolationLength="{interpolation:.5f}">1</jumpPosition>')
a('</audioBlockFormat>')
a('</audioChannelFormat>')
for i in range(10):
a(f'<audioTrackUID UID="ATU_{i + 1:08x}" bitDepth="24" sampleRate="48000">')
a(f'<audioTrackFormatIDRef>AT_0001{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
a('</audioTrackUID>')
for i in range(N_OBJ):
a(f'<audioTrackUID UID="ATU_{11 + i:08x}" bitDepth="24" sampleRate="48000">')
a(f'<audioTrackFormatIDRef>AT_0003{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
a('</audioTrackUID>')
for i in range(10):
a(f'<audioTrackFormat audioTrackFormatID="AT_0001{0x1001 + i:04x}_01" '
f'audioTrackFormatName="PCM_{BED_NAMES[i]}" formatDefinition="PCM" formatLabel="0001">')
a(f'<audioStreamFormatIDRef>AS_0001{0x1001 + i:04x}</audioStreamFormatIDRef>')
a('</audioTrackFormat>')
for i in range(N_OBJ):
a(f'<audioTrackFormat audioTrackFormatID="AT_0003{0x1001 + i:04x}_01" '
f'audioTrackFormatName="PCM_JOC_Object_{i+1}" formatDefinition="PCM" formatLabel="0001">')
a(f'<audioStreamFormatIDRef>AS_0003{0x1001 + i:04x}</audioStreamFormatIDRef>')
a('</audioTrackFormat>')
for i in range(10):
a(f'<audioStreamFormat audioStreamFormatID="AS_0001{0x1001 + i:04x}" '
f'audioStreamFormatName="PCM_{BED_NAMES[i]}" formatDefinition="PCM" formatLabel="0001">')
a(f'<audioChannelFormatIDRef>AC_0001{0x1001 + i:04x}</audioChannelFormatIDRef>')
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
a(f'<audioTrackFormatIDRef>AT_0001{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
a('</audioStreamFormat>')
for i in range(N_OBJ):
a(f'<audioStreamFormat audioStreamFormatID="AS_0003{0x1001 + i:04x}" '
f'audioStreamFormatName="PCM_JOC_Object_{i+1}" formatDefinition="PCM" formatLabel="0001">')
a(f'<audioChannelFormatIDRef>AC_0003{0x1001 + i:04x}</audioChannelFormatIDRef>')
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
a(f'<audioTrackFormatIDRef>AT_0003{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
a('</audioStreamFormat>')
a('</audioFormatExtended></format></coreMetadata>')
a('</ebuCoreMain>')
return ''.join(w).encode('utf-8')
+199
View File
@@ -0,0 +1,199 @@
"""校验 ADM BWF 的 RF64、通道、axml、chna、dbmd 和对象引用结构。
用法:``python src/adm_validate.py <file.wav> [more.wav ...]``,全部通过时退出码为 0。
"""
import struct, sys, re, os
def fail(msgs, m): msgs.append(m)
def walk_chunks(path):
chunks, ds64 = [], {}
with open(path, "rb") as f:
riff = f.read(4); f.read(4); wave = f.read(4)
if riff not in (b"RIFF", b"RF64"):
return None, None, f"File does not have a 'RIFF' or 'RF64' chunk"
if wave != b"WAVE":
return None, None, "File does not have a required 'WAVE' chunk"
while True:
off = f.tell()
cid = f.read(4)
if len(cid) < 4: break
sz = struct.unpack("<I", f.read(4))[0]
if cid == b"ds64":
body = f.read(sz + (sz & 1))
riff64, data64, sample64, _ = struct.unpack("<QQQI", body[:28])
ds64 = dict(riff64=riff64, data64=data64, sample64=sample64)
chunks.append(("ds64", off, sz)); continue
chunks.append((cid.decode("latin1"), off, sz))
eff = ds64.get("data64", sz) if (sz == 0xFFFFFFFF and cid == b"data") else sz
f.seek(off + 8 + eff + (eff & 1))
return chunks, ds64, None
def read_body(path, chunks, cid):
for c, off, sz in chunks:
if c == cid:
with open(path, "rb") as f:
f.seek(off + 8)
return f.read(sz)
return None
def parse_chna(body):
n_track, n_uid = struct.unpack("<HH", body[:4])
rows, p = [], 4
while p + 40 <= len(body):
trk = struct.unpack("<H", body[p:p+2])[0]
uid = body[p+2:p+14].rstrip(b"\x00").decode()
tf = body[p+14:p+28].rstrip(b"\x00").decode()
pk = body[p+28:p+40].rstrip(b"\x00").decode()
rows.append((trk, uid, tf, pk)); p += 40
return n_track, n_uid, rows
def decode_channel_input(ao_id):
"""将 ``AO_xxxx`` 的十六进制标识符解码为低 12 位通道输入号。"""
m = re.fullmatch(r"AO_([0-9a-fA-F]+)", ao_id)
if not m:
return None
v = int(m.group(1), 16)
if v > 0x1FFF: # 超过 12 位通道域
return None
return v & 0x0FFF
def ts_sec(s):
h, m, rest = s.split(":")
return int(h) * 3600 + int(m) * 60 + float(rest)
def validate(path, axml_override=None, chna_override=None):
msgs = []
chunks, ds64, err = walk_chunks(path)
if err:
return [err]
have = {c for c, _, _ in chunks}
for need in ("fmt ", "data", "axml", "chna", "dbmd"):
if need not in have:
fail(msgs, f"File does not have a required '{need.strip()}' chunk")
if msgs:
return msgs
fmt = read_body(path, chunks, "fmt ")
f_tag, f_ch, f_rate, _, _, f_bits = struct.unpack("<HHIIHH", fmt[:16])
chna_body = chna_override if chna_override is not None else read_body(path, chunks, "chna")
n_track, n_uid, rows = parse_chna(chna_body)
if f_ch != n_track:
fail(msgs, f"Mismatched number of audio channels and chna entries "
f"(fmt={f_ch} chna={n_track})")
ax_raw = axml_override if axml_override is not None else read_body(path, chunks, "axml")
ax = ax_raw.decode("utf-8")
# --- audioObjectID 十六进制通道输入号解码 ---
objs = re.findall(r'audioObjectID="(AO_[0-9a-zA-Z]+)"', ax)
bed_ch, obj_ch = [], []
for ao in objs:
cid = decode_channel_input(ao)
if cid is None:
fail(msgs, f"Invalid ADM BWF XML format: cannot decode channel "
f"input ID from AudioObjectID '{ao}'")
continue
if ao == "AO_1001":
bed_ch.append(cid)
else:
if cid <= 10:
fail(msgs, f"Source channel index should be greater than 10 "
f"for objects ('{ao}' -> {cid})")
obj_ch.append(cid)
# UID 十六进制 → 必须与 chna 表一致
uid_map = {uid: trk for trk, uid, tf, pk in rows}
for uid in re.findall(r'UID="(ATU_[0-9a-zA-Z]+)"', ax):
if uid not in uid_map:
fail(msgs, f"'{uid}' is not referenced in 'chna' chunk UID table")
continue
m = re.fullmatch(r"ATU_([0-9a-fA-F]+)", uid)
if m:
v = int(m.group(1), 16)
if v > 128:
fail(msgs, f"Channel index out of range (UID {uid} -> {v})")
# 轨数一致性:axml audioTrackUID 数 == fmt 声道数
n_tu = len(re.findall(r"<audioTrackUID ", ax))
if n_tu != f_ch:
fail(msgs, f"Number of channels declared in ADM ({n_tu}) does not "
f"match 'fmt ' chunk ({f_ch})")
# sampleRate / bitDepth 一致
for sr in set(re.findall(r'sampleRate="(\d+)"', ax)):
if int(sr) != f_rate:
fail(msgs, f"Mismatched track sample rate between ADM and WAV ({sr} vs {f_rate})")
for bd in set(re.findall(r'bitDepth="(\d+)"', ax)):
if int(bd) != f_bits:
fail(msgs, f"Mismatched track bit depth between ADM and WAV ({bd} vs {f_bits})")
# audioProgramme 唯一性 / audioContent ≥1
if ax.count("<audioProgramme ") != 1:
fail(msgs, "ADM has more than one audioProgramme object -- there must be only one"
if ax.count("<audioProgramme ") > 1
else "ADM does not have a required audioProgramme object")
if "<audioContent " not in ax:
fail(msgs, "audioProgramme object does not have a required audioContent object")
# 每个 channelFormat ≥1 blockFormat + 对象块链连续性
cfs = re.findall(r'<audioChannelFormat [^>]*typeLabel="0003".*?</audioChannelFormat>', ax, re.S)
object_block_formats = 0
for seg in cfs:
cf_id = re.search(r'audioChannelFormatID="([^"]+)"', seg).group(1)
block_xml = re.findall(r'<audioBlockFormat [^>]*rtime="[^"]+".*?</audioBlockFormat>',
seg, re.S)
object_block_formats += len(block_xml)
blocks = []
for block_index, block in enumerate(block_xml, 1):
timing = re.search(r'rtime="([^"]+)" duration="([^"]+)"', block)
if timing is None:
continue
rtime, duration = timing.groups()
blocks.append((rtime, duration))
jump = re.search(
r'<jumpPosition interpolationLength="([^"]+)">1</jumpPosition>', block)
if jump is not None and float(jump.group(1)) > ts_sec(duration) + 1e-8:
fail(msgs, f"Interpolation length exceeds duration in block format "
f"{block_index} of {cf_id}: {jump.group(1)} > {duration}")
if not blocks:
fail(msgs, f"AudioChannelFormat {cf_id} is missing audioBlockFormat sub-element")
continue
for i in range(len(blocks) - 1):
end_i = ts_sec(blocks[i][0]) + ts_sec(blocks[i][1])
nxt = ts_sec(blocks[i + 1][0])
if abs(end_i - nxt) > 2e-5:
fail(msgs, f"Time gap between block format {i+1} and {i+2} of {cf_id}: "
f"{end_i:.5f} vs {nxt:.5f}")
return msgs, dict(fmt_ch=f_ch, fmt_rate=f_rate, fmt_bits=f_bits,
chna=n_track, objects=len(obj_ch), bed=len(bed_ch),
trackUIDs=n_tu, axml_bytes=len(ax_raw),
audioBlockFormats=ax.count("<audioBlockFormat "),
objectBlockFormats=object_block_formats)
def main():
import argparse
ap = argparse.ArgumentParser()
ap.add_argument("files", nargs="+")
ap.add_argument("--axml-file", default=None, help="用该文件内容替换 wav 内 axml(对照实验)")
ap.add_argument("--chna-file", default=None, help="用该文件内容替换 wav 内 chna(对照实验)")
a = ap.parse_args()
ax_o = open(a.axml_file, "rb").read() if a.axml_file else None
ch_o = open(a.chna_file, "rb").read() if a.chna_file else None
rc = 0
for p in a.files:
r = validate(p, axml_override=ax_o, chna_override=ch_o)
name = os.path.basename(p)
if isinstance(r, list):
msgs, info = r, {}
else:
msgs, info = r
if msgs:
rc = 1
print(f"[FAIL] {name}")
for m in msgs:
print(" -", m)
else:
print(f"[PASS] {name} {info}")
return rc
if __name__ == "__main__":
sys.exit(main())
+282
View File
@@ -0,0 +1,282 @@
"""从常见 E-AC-3 同步帧直接提取连续 EMDF 容器。
扫描器检查八种全局位对齐,定位 ``0x5838`` 同步字,验证容器长度并解析各
payload config,因此不要求 EMDF 在原始 E-AC-3 文件中按字节对齐。
本模块有意只覆盖“完整 EMDF 容器在一个同步帧中连续出现”的常见情形。不解析
E-AC-3 mantissa,也不重组被音频数据隔开的多个 skip-field 碎片;遇到这种输入会
明确报错,让上层决定是否使用兼容桥。
"""
from dataclasses import dataclass
import csv
import hashlib
from pathlib import Path
import numpy as np
from variant_error import UnsupportedVariantError
SYNCWORD = 0x5838
REQUIRED_JOC_IDS = frozenset((11, 14))
class EmdfError(ValueError):
"""EMDF 或其 E-AC-3 传输结构不符合本实现支持的范围。"""
class BitReader:
"""MSB-first 位读取器;位置以源数据的绝对 bit offset 表示。"""
def __init__(self, data, position=0, limit=None):
self.data = memoryview(data)
self.position = int(position)
self.limit = len(self.data) * 8 if limit is None else int(limit)
def read(self, count):
count = int(count)
if count < 0 or self.position + count > self.limit:
raise EmdfError(f"位流越界 @bit{self.position}, need={count}, limit={self.limit}")
value = 0
while count:
byte_pos = self.position >> 3
removed_left = self.position & 7
take = min(count, 8 - removed_left)
shift = 8 - removed_left - take
value = (value << take) | ((self.data[byte_pos] >> shift) & ((1 << take) - 1))
self.position += take
count -= take
return value
def skip(self, count):
self.read(count)
def read_bytes(self, count):
return bytes(self.read(8) for _ in range(count))
def variable_bits(reader, width, max_groups=8):
"""读取 EMDF ``variable_bits(width)`` 变长整数。"""
value = 0
for _ in range(max_groups):
value += reader.read(width)
more = reader.read(1)
if not more:
return value
value = (value + 1) << width
raise EmdfError(f"variable_bits({width}) 延伸组过多")
@dataclass(frozen=True)
class EmdfContainer:
start_bit: int
raw: bytes
payloads: dict
sample_offsets: dict
def _parse_at(data, start_bit):
"""在已知 syncword 的 bit offset 解析一个 EMDF 容器。"""
reader = BitReader(data, start_bit)
if reader.read(16) != SYNCWORD:
raise EmdfError(f"EMDF syncword 不匹配 @bit{start_bit}")
length = reader.read(16)
body_start = reader.position
body_end = body_start + length * 8
if body_end > reader.limit:
raise EmdfError(f"EMDF 容器越界 @bit{start_bit}: length={length}")
reader.limit = body_end
version = reader.read(2)
if version == 3:
version += variable_bits(reader, 2)
key_id = reader.read(3)
if key_id == 7:
key_id += variable_bits(reader, 3)
# TS 103 420 JOC 使用 version=0/key_id=0;严格限制也能排除音频中的伪 marker。
if version != 0 or key_id != 0:
raise EmdfError(f"不支持的 EMDF version/key_id: {version}/{key_id}")
payloads = {}
sample_offsets = {}
terminated = False
while reader.position + 5 <= body_end:
payload_id = reader.read(5)
if payload_id == 0:
terminated = True
break
if payload_id == 0x1F:
payload_id += variable_bits(reader, 5)
if payload_id in payloads:
raise EmdfError(f"同一 EMDF 容器重复 payload id {payload_id}")
has_sample_offset = bool(reader.read(1))
sample_offset = (reader.read(12) >> 1) if has_sample_offset else 0
if reader.read(1):
variable_bits(reader, 11) # duration
if reader.read(1):
variable_bits(reader, 2) # group id
if reader.read(1):
reader.skip(8) # codec data
if not reader.read(1): # discard_unknown_payload
frame_aligned = False
if not has_sample_offset:
frame_aligned = bool(reader.read(1))
if frame_aligned:
reader.skip(2)
if has_sample_offset or frame_aligned:
reader.skip(7)
payload_size = variable_bits(reader, 8)
if reader.position + payload_size * 8 > body_end:
raise EmdfError(
f"payload id {payload_id} 越界: size={payload_size}, @bit{reader.position}")
payloads[payload_id] = reader.read_bytes(payload_size)
sample_offsets[payload_id] = sample_offset
if not terminated:
raise EmdfError("EMDF 容器缺少 payload id 0 终止符")
total_bytes = 4 + length
raw_reader = BitReader(data, start_bit, start_bit + total_bytes * 8)
raw = raw_reader.read_bytes(total_bytes)
return EmdfContainer(start_bit, raw, payloads, sample_offsets)
def _marker_offsets(data):
"""以 NumPy 批量检查八种位移,返回可能的 0x5838 bit offsets。"""
source = np.frombuffer(data, dtype=np.uint8)
if source.size < 4:
return []
offsets = []
for shift in range(8):
if shift == 0:
aligned = source
else:
aligned = np.bitwise_or(
np.left_shift(source[:-1].astype(np.uint16), shift) & 0xFF,
np.right_shift(source[1:].astype(np.uint16), 8 - shift),
).astype(np.uint8)
hits = np.flatnonzero((aligned[:-1] == 0x58) & (aligned[1:] == 0x38))
offsets.extend(int(hit) * 8 + shift for hit in hits)
return sorted(offsets)
def find_joc_emdf(frame):
"""返回同步帧中唯一、连续且包含 ID11/ID14 的 JOC EMDF 容器。"""
matches = []
offsets = _marker_offsets(frame)
parsed_candidates = []
parse_errors = []
for start_bit in offsets:
try:
container = _parse_at(frame, start_bit)
except EmdfError as exc:
if len(parse_errors) < 8:
parse_errors.append({"start_bit": start_bit, "error": str(exc)})
continue
parsed_candidates.append({
"start_bit": start_bit,
"payload_ids": list(container.payloads),
"payload_lengths": {str(k): len(v) for k, v in container.payloads.items()},
})
if REQUIRED_JOC_IDS.issubset(container.payloads):
matches.append(container)
if not matches:
raise UnsupportedVariantError(
"emdf_transport", "no_contiguous_joc_container",
"同步帧中未找到可连续解析且同时包含 ID11/ID14 的 EMDF 容器",
details={
"syncframe_bytes": len(frame),
"marker_bit_offsets": offsets,
"parsed_candidates": parsed_candidates,
"candidate_parse_errors": parse_errors,
"repair_hint": "检查 EMDF 是否跨多个 audio-block skip field 分片,或 payload config 是否变化",
})
if len(matches) != 1:
starts = [item.start_bit for item in matches]
raise UnsupportedVariantError(
"emdf_transport", "multiple_joc_containers",
"同步帧中存在多个可用 JOC EMDF,当前无法自动选择",
details={"syncframe_bytes": len(frame), "joc_container_start_bits": starts})
return matches[0]
def parse_container(data):
"""解析从 syncword 开始、已经重新按字节对齐保存的 EMDF 容器。"""
container = _parse_at(data, 0)
if len(container.raw) != len(data):
raise EmdfError(f"EMDF 文件尾有额外数据: parsed={len(container.raw)}, file={len(data)}")
return container
def iter_eac3_frames(data):
"""按 E-AC-3 ``frmsiz`` 遍历同步帧,拒绝静默重同步。"""
pos = 0
while pos < len(data):
if pos + 4 > len(data) or data[pos:pos + 2] != b"\x0b\x77":
raise UnsupportedVariantError(
"eac3_transport", "syncframe_header",
"E-AC-3 同步帧头无效或出现了未处理的子流排列",
details={
"byte_offset": pos,
"remaining_bytes": len(data) - pos,
"next_16_bytes_hex": data[pos:pos + 16].hex(),
})
size = ((((data[pos + 2] & 7) << 8) | data[pos + 3]) + 1) * 2
if pos + size > len(data):
raise UnsupportedVariantError(
"eac3_transport", "truncated_syncframe",
"E-AC-3 末帧长度超过输入剩余数据",
details={
"byte_offset": pos,
"declared_frame_bytes": size,
"remaining_bytes": len(data) - pos,
})
yield data[pos:pos + size]
pos += size
def extract_index(eac3_path, output_dir, max_frames=None):
"""将裸 E-AC-3 的连续 EMDF 保存为 ``frames.csv + emdf/``。"""
output_dir = Path(output_dir)
emdf_dir = output_dir / "emdf"
emdf_dir.mkdir(parents=True, exist_ok=True)
frames = iter_eac3_frames(Path(eac3_path).read_bytes())
rows = []
for frame_number, frame in enumerate(frames):
if max_frames is not None and frame_number >= max_frames:
break
try:
container = find_joc_emdf(frame)
except UnsupportedVariantError as exc:
exc.add_context(frame=frame_number, details={"syncframe_bytes": len(frame)})
raise
except EmdfError as exc:
raise UnsupportedVariantError(
"emdf_transport", "container_syntax",
"EMDF 容器语法无法解析",
frame=frame_number,
details={"syncframe_bytes": len(frame), "parser_error": str(exc)}) from exc
digest = hashlib.sha256(container.raw).hexdigest()
target = emdf_dir / f"{digest}.bin"
if not target.is_file():
target.write_bytes(container.raw)
rows.append({
"frame": frame_number,
"emdf_hash": digest,
"emdf_size": len(container.raw),
"emdf_start_bit": container.start_bit,
"payload_ids": ";".join(str(x) for x in container.payloads),
"error": "",
})
if (frame_number + 1) % 1000 == 0:
print(f"[metadata] {frame_number + 1} frames", flush=True)
if not rows:
raise EmdfError("E-AC-3 输入中没有可处理的同步帧")
with (output_dir / "frames.csv").open("w", encoding="utf-8", newline="") as fp:
fields = ("frame", "emdf_hash", "emdf_size", "emdf_start_bit", "payload_ids", "error")
writer = csv.DictWriter(fp, fieldnames=fields)
writer.writeheader()
writer.writerows(rows)
return output_dir
+76
View File
@@ -0,0 +1,76 @@
"""解包 EVO MD-set evolution 载荷,返回各 payload ID 的字节数据和位偏移。"""
_MARK = "1001001000000"
# 各子载荷字段: (id, 头部前缀, 同步标记, 后缀常量, 尺寸域位数, 是否有转义)
# 头部前缀 = 5 位 id 的 MSB 二进制(id11 字段前另有 5 位容器前导 00000)
# 尺寸域单位 = nibble(4 位)。id14 有转义:9 位值=0 → 再读 9 位 = 字节数。
_LAYOUT = [
(11, "0000001011", "010000000000000", 8, False),
(14, "01110", "01000000000000", 9, True),
(2, "00010", "000100", 7, False),
(1, "00001", "1110000000000000000000000000", 4, False),
(30, "11110", "1110000000000000000000000000", 4, False),
]
def _msb_bits(data: bytes):
return [(x >> (7 - i)) & 1 for x in data for i in range(8)]
def _val(bits, off, n):
v = 0
for b in bits[off:off + n]:
v = (v << 1) | b
return v
class _LooseSkip(Exception):
def __init__(self, ident):
self.ident = ident
def unpack_evolution(payload: bytes, loose=False):
"""解包 evolution 载荷 → (subs, offsets)。subs 键为 id 整数。
loose=True 时对每个 id 的 (前缀+标记+后缀) 全模式做位流重同步扫描
(不同编码流的子载荷次序/内部常量可有合法差异,如 kanata 的 id11)。"""
bits = _msb_bits(payload)
pos = 0
subs = {}
offsets = {}
for ident, pref, suff, sbits, escape in _LAYOUT:
pat = pref + _MARK + suff
if loose:
hit = -1
for i in range(pos, len(bits) - len(pat)):
if ''.join(map(str, bits[i:i + len(pat)])) == pat:
hit = i
break
if hit < 0:
continue
pos = hit + len(pat)
else:
for name, const, expect in (("前缀", bits[pos:pos + len(pref)], pref),
("标记", bits[pos + len(pref):pos + len(pref) + len(_MARK)], _MARK),
("后缀", bits[pos + len(pref) + len(_MARK):
pos + len(pref) + len(_MARK) + len(suff)], suff)):
got = ''.join(map(str, const))
if got != expect:
raise ValueError(f"id={ident}: {name}常量不匹配 @bit{pos} got={got} want={expect}")
pos += len(pref) + len(_MARK) + len(suff)
n_nib = _val(bits, pos, sbits)
pos += sbits
if escape and n_nib == 1:
n_nib = 512 + _val(bits, pos, sbits)
pos += sbits
n_bits = n_nib * 4
body = bits[pos:pos + n_bits]
pos += n_bits
b = bytearray(len(body) // 8)
for i in range(0, len(body) // 8 * 8, 8):
v = 0
for x in body[i:i + 8]:
v = (v << 1) | x
b[i // 8] = v
subs[ident] = bytes(b)
offsets[ident] = pos
return subs, offsets
+176
View File
@@ -0,0 +1,176 @@
"""解析 JOC 位流并生成对象混合矩阵。
范围:
- EMDF ID14 的 joc_header、joc_info 和 Huffman joc_data;
- 差分解码得到 joc_mix_mtx_q;
- 去量化得到 joc_mix_mtx_dq;
- 位流自洽验证(joc_data 后剩余 = padding_bits 0..7 + 可能 joc_ext_data)
后续的时间插值、QMF/时域重建和 ``joc_clipgain`` 位于 ``renderer.py``。
Sparse 分支仍缺少实际样本验证。
"""
from pathlib import Path
import numpy as np
# 格式:节点数组 [left, right];正 = 内部节点索引,负 = 叶(值 = -node-1)
_TABLES_PATH = Path(__file__).resolve().parent.parent / "data" / "tables.npz"
_HUFF_NAMES = (
"joc_huff_code_coarse_generic",
"joc_huff_code_fine_generic",
"joc_huff_code_coarse_coeff_sparse",
"joc_huff_code_fine_coeff_sparse",
"joc_huff_code_5ch_pos_index_sparse",
"joc_huff_code_7ch_pos_index_sparse",
)
def _load_huff_tables():
with np.load(_TABLES_PATH) as tables:
return {
name: np.asarray(tables[name], dtype=np.int64).tolist()
for name in _HUFF_NAMES
}
H = _load_huff_tables()
JOC_NUM_CHANNELS = {0: 5, 1: 7, 2: 7, 3: 5, 4: 7} # Table 33
JOC_NUM_BANDS = {0: 1, 1: 3, 2: 5, 3: 7, 4: 9, 5: 12, 6: 15, 7: 23} # Table 35
_PUBLIC_TABLE39_EXCERPT_UNUSED = { # 仅保留作表格差异说明;渲染映射在 joc_qmf.py。
23: [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22],
}
class BR:
def __init__(self, data, pos=0):
self.d = data
self.p = pos
def bits(self, n):
v = 0
for _ in range(n):
v = (v << 1) | ((self.d[self.p >> 3] >> (7 - (self.p & 7))) & 1)
self.p += 1
return v
def huff_decode(tree, br):
node = 0
while node >= 0:
b = br.bits(1)
node = tree[node][b]
return -node - 1
def get_huff_code(mode, typ, nch):
if typ == "IDX":
return H["joc_huff_code_5ch_pos_index_sparse" if nch == 5 else "joc_huff_code_7ch_pos_index_sparse"]
if typ == "VEC":
return H["joc_huff_code_coarse_coeff_sparse" if mode == 0 else "joc_huff_code_fine_coeff_sparse"]
# MTX
return H["joc_huff_code_coarse_generic" if mode == 0 else "joc_huff_code_fine_generic"]
def parse_joc(payload):
"""解析 id14 载荷(joc() 位流)。返回字段 dict + 解析后剩余位数。"""
br = BR(payload)
out = {}
out["dmx_config_idx"] = br.bits(3)
out["num_objects_bits"] = br.bits(6)
out["ext_config_idx"] = br.bits(3)
n_objects = out["num_objects_bits"] + 1
n_channels = JOC_NUM_CHANNELS.get(out["dmx_config_idx"])
out["n_objects"], out["n_channels"] = n_objects, n_channels
out["clipgain_x_bits"] = br.bits(3)
out["clipgain_y_bits"] = br.bits(5)
out["seq_count_bits"] = br.bits(10)
# clipgain = 1 + (y/32)·2^(x−4),值域为 [1, 8.75]。
out["clipgain"] = 1 + out["clipgain_y_bits"] / 32.0 * 2 ** (out["clipgain_x_bits"] - 4)
objs = []
for obj in range(n_objects):
o = {}
o["present"] = br.bits(1)
if o["present"]:
o["num_bands_idx"] = br.bits(3)
o["n_bands"] = JOC_NUM_BANDS[o["num_bands_idx"]]
o["sparse"] = br.bits(1)
o["quant_idx"] = br.bits(1)
o["slope_idx"] = br.bits(1)
o["num_dpoints_bits"] = br.bits(1)
o["n_dpoints"] = o["num_dpoints_bits"] + 1
if o["slope_idx"] == 1:
o["offset_ts"] = [br.bits(5) + 1 for _ in range(o["n_dpoints"])]
objs.append(o)
out["objs"] = objs
# joc_data(Huffman)
for obj, o in enumerate(objs):
if not o["present"]:
continue
nquant = 96 if o["quant_idx"] == 0 else 192
o["channel_idx"] = []
o["vec"] = []
o["mtx"] = []
for dp in range(o["n_dpoints"]):
if o["sparse"] == 1:
# Sparse JOC 使用 VEC/IDX Huffman 树;此分支尚无真实码流验证。
ci0 = br.bits(3)
tree = get_huff_code(n_channels, "IDX", n_channels)
ci = [ci0] + [huff_decode(tree, br) for _ in range(o["n_bands"] - 1)]
o["channel_idx"].append(ci)
tree = get_huff_code(o["quant_idx"], "VEC", n_channels)
vec = [huff_decode(tree, br) for _ in range(o["n_bands"])]
o["vec"].append(vec)
else:
tree = get_huff_code(o["quant_idx"], "MTX", n_channels)
mtx = [[huff_decode(tree, br) for _ in range(o["n_bands"])]
for _ in range(n_channels)]
o["mtx"].append(mtx)
out["data_end_bits"] = br.p
out["remaining_bits"] = len(payload) * 8 - br.p
out["tail_bytes"] = payload[br.p // 8:]
return out
def diff_decode(out):
"""6.6.2:差分解码 → joc_mix_mtx_q[obj][dp][ch][pb]。"""
mix_q = {}
n_ch = out["n_channels"]
for obj, o in enumerate(out["objs"]):
if not o["present"]:
continue
nquant = 96 if o["quant_idx"] == 0 else 192
q = np.zeros((o["n_dpoints"], n_ch, o["n_bands"]), dtype=np.int64)
for dp in range(o["n_dpoints"]):
if o["sparse"] == 1:
# Sparse 差分路径尚无真实码流验证。
offset = 50 if o["quant_idx"] == 0 else 100
ci = o["channel_idx"][dp]
vec = o["vec"][dp]
for pb in range(o["n_bands"]):
ci_mod = ci[0] if pb == 0 else (ci[pb - 1] + ci[pb]) % n_ch
for ch in range(n_ch):
if ch == ci_mod:
if pb == 0:
q[dp][ch][pb] = (offset + vec[pb]) % nquant
else:
q[dp][ch][pb] = (q[dp][ch][pb - 1] + vec[pb]) % nquant
else:
q[dp][ch][pb] = offset
else:
offset = 48 if o["quant_idx"] == 0 else 96
mtx = o["mtx"][dp]
for ch in range(n_ch):
q[dp][ch][0] = (offset + mtx[ch][0]) % nquant
for pb in range(1, o["n_bands"]):
q[dp][ch][pb] = (q[dp][ch][pb - 1] + mtx[ch][pb]) % nquant
mix_q[obj] = q
return mix_q
def dequantize(out, mix_q):
"""6.6.4:去量化 → joc_mix_mtx_dq。"""
mix_dq = {}
for obj, o in enumerate(out["objs"]):
if not o["present"]:
continue
nquant = 96 if o["quant_idx"] == 0 else 192
q = mix_q[obj]
dq = (q.astype(np.float64) - nquant / 2) * 820 / (4096 * (1 + o["quant_idx"]))
mix_dq[obj] = dq
return mix_dq
+151
View File
@@ -0,0 +1,151 @@
"""实现 JOC QMF、参数带映射和矩阵时间插值。
静态表保存在 ``data`` 中;NumPy 批量函数保持各通道、各对象的状态彼此独立。
"""
from pathlib import Path
import numpy as np
N = 64
_TABLES = np.load(Path(__file__).resolve().parent.parent / "data" / "tables.npz")
ANALYSIS_WINDOW = np.asarray(_TABLES["analysis_window"], dtype=np.float64)
QMF5_WINDOW = np.asarray(_TABLES["qmf5_window"], dtype=np.float64)
# 子带到参数带的映射表。
_TABLE39 = {
23: "0 1 2 3 4 5 6 7 8 9 10 11 12 12 13 13 14 14 15 15 16 16 16 17 17 17 18 18 18 18 19 19 19 19 19 20 20 20 20 20 20 21 21 21 21 21 21 21 22 22 22 22 22 22 22 22 22 22 22 22 22 22 22 22",
15: "0 1 2 3 4 5 6 7 8 9 9 10 10 11 11 11 12 12 12 12 13 13 13 13 13 13 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14",
12: "0 1 2 3 4 4 5 5 6 6 6 7 7 7 8 8 8 8 9 9 9 9 9 10 10 10 10 10 10 10 10 10 10 10 10 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11",
9: "0 1 2 3 3 3 4 5 5 6 6 6 7 7 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8",
7: "0 1 2 2 3 3 3 3 4 4 4 4 4 4 5 5 5 5 5 5 5 5 5 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6",
5: "0 1 1 2 2 2 2 2 2 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4",
3: "0 0 0 1 1 1 1 1 1 1 1 1 1 1 1 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2",
1: "0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0",
}
_PB_MAP = {k: np.fromstring(v, dtype=np.int64, sep=" ") for k, v in _TABLE39.items()}
def sb_to_pb_table(n_bands):
try:
return _PB_MAP[n_bands]
except KeyError as exc:
raise ValueError(f"不支持的 JOC 参数带数: {n_bands}") from exc
def interp_matrix(obj_info, dq, prev, n_ts=24):
"""按数据点和跨帧状态插值,返回 ``[channel,subband,timeslot]``。"""
dq_sb = np.asarray(dq, dtype=np.float64)[:, :, sb_to_pb_table(dq.shape[2])]
previous = np.asarray(prev, dtype=np.float64)
n_dp = obj_info["n_dpoints"]
slope = obj_info["slope_idx"]
if slope == 0:
if n_dp == 1:
alpha = (np.arange(n_ts, dtype=np.float64) + 1.0) / n_ts
return previous[:, :, None] * (1.0 - alpha) + dq_sb[0, :, :, None] * alpha
half = n_ts // 2
a0 = (np.arange(half, dtype=np.float64) + 1.0) / half
a1 = (np.arange(n_ts - half, dtype=np.float64) + 1.0) / (n_ts - half)
first = previous[:, :, None] * (1.0 - a0) + dq_sb[0, :, :, None] * a0
second = dq_sb[0, :, :, None] * (1.0 - a1) + dq_sb[1, :, :, None] * a1
return np.concatenate((first, second), axis=2)
ts = np.arange(n_ts)
offsets = obj_info.get("offset_ts", [])
if n_dp == 1:
return np.where(ts[None, None, :] < offsets[0], previous[:, :, None], dq_sb[0, :, :, None])
out = np.where(ts[None, None, :] < offsets[0], previous[:, :, None], dq_sb[0, :, :, None])
return np.where(ts[None, None, :] < offsets[1], out, dq_sb[1, :, :, None])
def qmf_analysis_step(fifo, ring):
"""分析 QMF 的单时隙入口;批量入口见 :func:`qmf_analysis_frame`。"""
f = np.asarray(fifo, dtype=np.float64)
r = np.asarray(ring, dtype=np.float64)
single = f.ndim == 2
if single:
f, r = f[None, ...], r[None, ...]
x, new = qmf_analysis_frame(f, r[:, None, :])
interleaved = np.empty((len(f), 128), dtype=np.float64)
interleaved[:, 0::2] = x[:, :, 0].real
interleaved[:, 1::2] = x[:, :, 0].imag
return (interleaved[0], new[0]) if single else (interleaved, new)
def qmf_analysis_frame(fifo, rings):
"""批量完成整帧时隙的分析窗、旋转和 64 点 FFT。"""
f = np.asarray(fifo, dtype=np.float64)
r = np.asarray(rings, dtype=np.float64)
if f.ndim != 3 or f.shape[1:] != (9, 64) or r.ndim != 3 or r.shape[0] != f.shape[0] or r.shape[2] != 64:
raise ValueError(f"analysis QMF shape 错误: fifo={f.shape}, rings={r.shape}")
slots = r.shape[1]
seq = np.concatenate((f[:, ::-1, :], r), axis=1)
windows = np.lib.stride_tricks.sliding_window_view(seq, 9, axis=1)
history = windows[:, :slots].transpose(0, 1, 3, 2)[:, :, ::-1, :]
t = ANALYSIS_WINDOW
v36 = np.sum(history[:, :, 0::2, :] * t[[8, 6, 4, 2, 0]][None, None], axis=2)
v40 = np.sum(history[:, :, 1::2, :] * t[[7, 5, 3, 1]][None, None], axis=2) + r * t[9]
pairs = np.stack((v40, v36), axis=-1)
kernel = pairs.reshape(len(f), slots, 16, 4, 2)[:, :, ::-1, ::-1, :].reshape(len(f), slots, 128)
k = np.arange(64, dtype=np.float64)
a = 0.5 * np.sin(np.pi * k / 128.0)
b = 0.5 * np.cos(np.pi * k / 128.0)
re, im = kernel[:, :, 0::2], kernel[:, :, 1::2]
freq = np.fft.fft((im * a - re * b) + 1j * (im * b + re * a), axis=2) / 64.0
src = np.empty((len(f), slots, 128), dtype=np.float64)
src[:, :, 0::2], src[:, :, 1::2] = freq.real, freq.imag
out = np.empty_like(src).reshape(len(f), slots, 32, 4)
out[:, :, :, 0] = src[:, :, 0:64:2]
out[:, :, :, 1] = -src[:, :, 1:64:2]
out[:, :, :, 2] = src[:, :, 126:62:-2]
out[:, :, :, 3] = src[:, :, 127:63:-2]
flat = out.reshape(len(f), slots, 128)
complex_qmf = (flat[:, :, 0::2] + 1j * flat[:, :, 1::2]).transpose(0, 2, 1)
combined = np.concatenate((f[:, ::-1, :], r), axis=1)
new_fifo = combined[:, -9:, :][:, ::-1, :].copy()
return complex_qmf, new_fifo
_SURROUND_DC_A = np.array([
-.0006242550443857908, -.0019234686624258757, -.0042654648423194885,
-.008168308064341545, -.014327201060950756, -.023759860545396805,
-.03757232800126076, -.05577569454908371, -.07568276673555374,
-.09172472357749939, -.5979374051094055, -.09172472357749939,
-.07568276673555374, -.05577569454908371, -.03757232800126076,
-.023759860545396805, -.014327201060950756, -.008168308064341545,
-.0042654648423194885, -.0019234686624258757, -.0006242550443857908,
], dtype=np.float64)
_SURROUND_DC_B = np.array([
.0013996040215715766, .003839150769636035, .007512642536312342,
.012419373728334904, .018367428332567215, .0249701626598835,
.03167900815606117, .03785000368952751, .04283412545919418,
.04607561603188515, .047200120985507965, .04607561603188515,
.04283412545919418, .03785000368952751, .03167900815606117,
.0249701626598835, .018367428332567215, .012419373728334904,
.007512642536312342, .003839150769636035, .0013996040215715766,
], dtype=np.float64)
_SURROUND_DC_C = _SURROUND_DC_B + 1j * _SURROUND_DC_A
def surround_post_frame(x, delay, dc_hist):
"""处理 Ls/Rs 的 10 槽延迟、-j 旋转和 band-0 FIR。"""
src = np.asarray(x, dtype=np.complex128)
qdelay = np.asarray(delay, dtype=np.complex128).copy()
hist = np.asarray(dc_hist, dtype=np.complex128).copy()
if src.shape != (2, 64, 24):
raise ValueError(f"surround QMF shape 错误: {src.shape}")
out = np.empty_like(src)
for group in range(0, 24, 4):
current = src[:, :, group:group + 4].transpose(0, 2, 1)
queued = np.concatenate((qdelay, current), axis=1)
block = -1j * queued[:, :4, :]
qdelay = queued[:, 4:, :]
dc_buf = np.concatenate((hist, current[:, :, 0]), axis=1)
windows = np.lib.stride_tricks.sliding_window_view(dc_buf, 21, axis=1)
block[:, :, 0] = 2.0 * np.sum(windows * _SURROUND_DC_C[None, None, :], axis=2)
hist = dc_buf[:, 4:]
out[:, :, group:group + 4] = block.transpose(0, 2, 1)
return out, qdelay, hist
+324
View File
@@ -0,0 +1,324 @@
"""Evolution sidecar 的 JOC/OAMD 纯 Python 解析、校验和可读输出。"""
from collections import Counter
import csv
import hashlib
import json
from pathlib import Path
from adm_atmos import q_to_adm_xyz
from evo_unpack import unpack_evolution
from emdf import EmdfError, find_joc_emdf, iter_eac3_frames, parse_container
from joc_decode import parse_joc
from oamd_bits import JocFieldState, frame_update_values
from variant_error import UnsupportedVariantError, bytes_descriptor
class DirectPayloadIndex:
"""One-pass in-memory view of contiguous EMDF containers in an E-AC-3 file.
The optional cache directory keeps the existing ``frames.csv + emdf/``
contract, but normal processing reuses the containers already parsed during
the scan instead of reading and parsing thousands of small files again.
"""
def __init__(self, rows, subpayloads, sample_offsets, directory=None):
self.rows = rows
self._subpayloads = subpayloads
self._sample_offsets = sample_offsets
self.directory = Path(directory) if directory is not None else None
@classmethod
def from_eac3(cls, eac3_path, max_frames=None, cache_dir=None):
cache_dir = Path(cache_dir) if cache_dir is not None else None
emdf_dir = None
if cache_dir is not None:
emdf_dir = cache_dir / "emdf"
emdf_dir.mkdir(parents=True, exist_ok=True)
rows = []
payloads = []
offsets = []
frames = iter_eac3_frames(Path(eac3_path).read_bytes())
for frame_number, frame in enumerate(frames):
if max_frames is not None and frame_number >= max_frames:
break
try:
container = find_joc_emdf(frame)
except UnsupportedVariantError as exc:
exc.add_context(frame=frame_number, details={"syncframe_bytes": len(frame)})
raise
except EmdfError as exc:
raise UnsupportedVariantError(
"emdf_transport", "container_syntax",
"EMDF 容器语法无法解析",
frame=frame_number,
details={"syncframe_bytes": len(frame), "parser_error": str(exc)}) from exc
digest = hashlib.sha256(container.raw).hexdigest()
if emdf_dir is not None:
target = emdf_dir / f"{digest}.bin"
if not target.is_file():
target.write_bytes(container.raw)
rows.append({
"frame": frame_number,
"emdf_hash": digest,
"emdf_size": len(container.raw),
"emdf_start_bit": container.start_bit,
"payload_ids": ";".join(str(x) for x in container.payloads),
"error": "",
})
payloads.append(container.payloads)
offsets.append(container.sample_offsets)
if (frame_number + 1) % 1000 == 0:
print(f"[metadata] {frame_number + 1} frames", flush=True)
if not rows:
raise EmdfError("E-AC-3 输入中没有可处理的同步帧")
if cache_dir is not None:
with (cache_dir / "frames.csv").open("w", encoding="utf-8", newline="") as fp:
fields = ("frame", "emdf_hash", "emdf_size", "emdf_start_bit", "payload_ids", "error")
writer = csv.DictWriter(fp, fieldnames=fields)
writer.writeheader()
writer.writerows(rows)
return cls(rows, payloads, offsets, cache_dir)
def subpayloads(self, row):
return self._subpayloads[int(row["frame"])]
def subpayload_sample_offset(self, row, payload_id):
return int(self._sample_offsets[int(row["frame"])].get(payload_id, 0))
def __len__(self):
return len(self.rows)
class PayloadIndex:
"""旧 evolution sidecar 或直接 EMDF sidecar 的严格顺序视图。"""
def __init__(self, directory):
self.directory = Path(directory)
csv_path = self.directory / "frames.csv"
self.payload_dir = self.directory / "payloads"
self.emdf_dir = self.directory / "emdf"
if not csv_path.is_file() or not (self.payload_dir.is_dir() or self.emdf_dir.is_dir()):
raise FileNotFoundError(
f"元数据目录需要 frames.csv 和 payloads/ 或 emdf/: {self.directory}")
with csv_path.open(encoding="utf-8", newline="") as fp:
self.rows = list(csv.DictReader(fp))
if not self.rows:
raise ValueError("frames.csv 为空")
self._sub_cache = {}
self._sample_offset_cache = {}
def payload(self, row):
payload_hash = row.get("payload_hash", "")
if not payload_hash:
raise ValueError(f"frame {row.get('frame', '?')} 缺少 evolution payload")
path = self.payload_dir / f"{payload_hash}.bin"
if not path.is_file():
raise FileNotFoundError(path)
return path.read_bytes()
def subpayloads(self, row):
"""返回 ``{payload_id: bytes}``,屏蔽两种 sidecar 容器的差异。"""
emdf_hash = row.get("emdf_hash", "")
if emdf_hash:
key = ("emdf", emdf_hash)
if key not in self._sub_cache:
path = self.emdf_dir / f"{emdf_hash}.bin"
if not path.is_file():
raise FileNotFoundError(path)
container = parse_container(path.read_bytes())
self._sub_cache[key] = container.payloads
self._sample_offset_cache[key] = container.sample_offsets
return self._sub_cache[key]
payload_hash = row.get("payload_hash", "")
key = ("evolution", payload_hash)
if key not in self._sub_cache:
self._sub_cache[key], _ = unpack_evolution(self.payload(row), loose=True)
self._sample_offset_cache[key] = {}
return self._sub_cache[key]
def subpayload_sample_offset(self, row, payload_id):
"""返回 EMDF payload 的外层 sample offset;旧 sidecar 没有该字段时为 0。"""
self.subpayloads(row)
emdf_hash = row.get("emdf_hash", "")
key = (("emdf", emdf_hash) if emdf_hash else
("evolution", row.get("payload_hash", "")))
return int(self._sample_offset_cache.get(key, {}).get(payload_id, 0))
def __len__(self):
return len(self.rows)
def _frame_record(frame, parsed, state):
present = [i for i, obj in enumerate(parsed["objs"]) if obj["present"]]
sparse = [i for i in present if parsed["objs"][i]["sparse"]]
bands = Counter(parsed["objs"][i]["n_bands"] for i in present)
positions = []
for obj in range(1, 16):
q1, q2, q3 = (state.q[(obj, key)] for key in ("q1", "q2", "q3"))
x, y, z = q_to_adm_xyz(q1, q2, q3)
positions.append([round(float(x), 7), round(float(y), 7), round(float(z), 7)])
return {
"frame": frame,
"sequence": parsed["seq_count_bits"],
"downmix_config": parsed["dmx_config_idx"],
"extension_config": parsed["ext_config_idx"],
"objects_declared": parsed["n_objects"],
"objects_present": present,
"sparse_objects": sparse,
"parameter_bands": {str(k): v for k, v in sorted(bands.items())},
"clipgain": parsed["clipgain"],
"oamd_xyz": positions,
}
def inspect(index, limit=None, print_frames=False):
"""逐帧完整解析 ID11/ID14,并返回可 JSON 序列化的汇总。"""
rows = index.rows if limit is None else index.rows[:limit]
oamd = JocFieldState()
frame_records = []
configs = Counter()
active = Counter()
band_totals = Counter()
sparse_counts = Counter()
clipgains = []
for seq, row in enumerate(rows):
frame_number = int(row.get("frame", seq))
try:
subs = index.subpayloads(row)
except UnsupportedVariantError as exc:
exc.add_context(frame=frame_number)
raise
except Exception as exc:
raise UnsupportedVariantError(
"metadata_container", "sidecar_or_emdf_parse",
"元数据容器无法解析",
frame=frame_number,
details={"exception_type": type(exc).__name__, "parser_error": str(exc)}) from exc
if 11 not in subs or 14 not in subs:
raise UnsupportedVariantError(
"emdf_payloads", "missing_required_payload",
"EMDF 缺少 ID11/OAMD 或 ID14/JOC",
frame=frame_number,
details={
"payload_ids": list(subs),
"payload_lengths": {str(k): len(v) for k, v in subs.items()},
})
try:
oamd.apply(frame_update_values(subs[11]))
except UnsupportedVariantError as exc:
exc.add_context(frame=frame_number, details={"payload_id": 11})
raise
except Exception as exc:
raise UnsupportedVariantError(
"oamd", "payload_syntax",
"OAMD 字段解析失败",
frame=frame_number,
details={
"payload_id": 11,
"payload": bytes_descriptor(subs[11]),
"exception_type": type(exc).__name__,
"parser_error": str(exc),
}) from exc
try:
parsed = parse_joc(subs[14])
except UnsupportedVariantError as exc:
exc.add_context(frame=frame_number, details={"payload_id": 14})
raise
except Exception as exc:
payload = subs[14]
header = {}
if len(payload) >= 4:
value = int.from_bytes(payload[:4], "big")
header = {
"downmix_config": (value >> 29) & 7,
"objects_minus_one": (value >> 23) & 63,
"extension_config": (value >> 20) & 7,
}
raise UnsupportedVariantError(
"joc", "payload_syntax",
"JOC ID14 解析失败",
frame=frame_number,
details={
"payload_id": 14,
"header_probe": header,
"payload": bytes_descriptor(payload),
"exception_type": type(exc).__name__,
"parser_error": str(exc),
"repair_hint": "检查 JOC header、对象数、参数带、Huffman 或扩展字段",
}) from exc
sparse = [i for i, obj in enumerate(parsed["objs"]) if obj["present"] and obj["sparse"]]
if sparse:
raise UnsupportedVariantError(
"joc", "sparse_joc",
"发现尚未验证的 Sparse JOC 帧",
frame=frame_number,
details={
"sparse_objects": sparse,
"downmix_config": parsed["dmx_config_idx"],
"extension_config": parsed["ext_config_idx"],
"objects": parsed["n_objects"],
"payload": bytes_descriptor(subs[14]),
"repair_hint": "需要 Sparse JOC 实际样本及对应输出建立回归后再启用",
})
if parsed["n_channels"] != 5 or parsed["n_objects"] > 15:
raise UnsupportedVariantError(
"joc", "unsupported_configuration",
"JOC 核心通道数或对象数超出当前渲染器范围",
frame=frame_number,
details={
"downmix_config": parsed["dmx_config_idx"],
"core_channels": parsed["n_channels"],
"objects": parsed["n_objects"],
"extension_config": parsed["ext_config_idx"],
"payload": bytes_descriptor(subs[14]),
})
rec = _frame_record(frame_number, parsed, oamd)
configs[(parsed["dmx_config_idx"], parsed["ext_config_idx"],
parsed["n_channels"], parsed["n_objects"])] += 1
active[len(rec["objects_present"])] += 1
band_totals.update({int(k): v for k, v in rec["parameter_bands"].items()})
sparse_counts[len(rec["sparse_objects"])] += 1
clipgains.append(float(parsed["clipgain"]))
if print_frames:
print(json.dumps(rec, ensure_ascii=False, separators=(",", ":")))
frame_records.append(rec)
summary = {
"frames": len(rows),
"frame_samples": 1536,
"sample_rate": 48000,
"duration_sec": len(rows) * 1536 / 48000,
"configurations": [
{"downmix_config": k[0], "extension_config": k[1], "core_channels": k[2],
"objects": k[3], "frames": count}
for k, count in sorted(configs.items())
],
"active_object_count_histogram": {str(k): v for k, v in sorted(active.items())},
"parameter_band_totals": {str(k): v for k, v in sorted(band_totals.items())},
"sparse_object_count_histogram": {str(k): v for k, v in sorted(sparse_counts.items())},
"clipgain_min": min(clipgains),
"clipgain_max": max(clipgains),
"first_frame": frame_records[0],
"last_frame": frame_records[-1],
}
emdf_rows = [row for row in rows if row.get("emdf_start_bit", "") != ""]
if emdf_rows:
starts = [int(row["emdf_start_bit"]) for row in emdf_rows]
id_sets = Counter(row.get("payload_ids", "") for row in emdf_rows)
summary["emdf_transport"] = {
"continuous_containers": len(emdf_rows),
"start_bit_min": min(starts),
"start_bit_max": max(starts),
"bit_alignment_histogram": {
str(k): v for k, v in sorted(Counter(x & 7 for x in starts).items())
},
"payload_id_order_histogram": dict(sorted(id_sets.items())),
}
return summary
def write_summary(index, output, limit=None, print_frames=False):
summary = inspect(index, limit=limit, print_frames=print_frames)
output = Path(output)
output.write_text(json.dumps(summary, ensure_ascii=False, indent=2), encoding="utf-8")
return summary
+276
View File
@@ -0,0 +1,276 @@
"""ctypes bridge for the dependency-free MSVC C++ JOC DSP core.
Metadata parsing intentionally stays in Python. One C call consumes a complete
1536-sample frame, so Python is not involved in the hot 24-timeslot x 15-object
DSP loops.
"""
from __future__ import annotations
import ctypes
import os
from pathlib import Path
import sys
import numpy as np
from evo_unpack import unpack_evolution
from joc_decode import dequantize, diff_decode, parse_joc
FRAME_SAMPLES = 1536
MAX_OBJECTS = 15
MAX_DPOINTS = 2
CORE_CHANNELS = 5
MAX_BANDS = 23
ABI_VERSION = 1
class NativeBackendUnavailable(RuntimeError):
pass
def native_library_filename():
if sys.platform == "win32":
return "eac3joc_core.dll"
if sys.platform == "darwin":
return "libeac3joc_core.dylib"
if sys.platform.startswith("linux"):
return "libeac3joc_core.so"
raise NativeBackendUnavailable(f"unsupported native platform: {sys.platform}")
def _candidate_libraries():
override = os.environ.get("EAC3JOC_NATIVE_LIBRARY")
if override:
yield Path(override).expanduser()
root = Path(__file__).resolve().parent.parent
yield root / "lib" / native_library_filename()
def find_native_library(explicit=None):
if explicit is not None:
path = Path(explicit).expanduser().resolve()
if not path.is_file():
raise NativeBackendUnavailable(f"native library not found: {path}")
return path
checked = []
for item in _candidate_libraries():
path = item.resolve()
checked.append(str(path))
if path.is_file():
return path
raise NativeBackendUnavailable("native library not found; checked: " + "; ".join(checked))
def default_native_threads():
override = os.environ.get("EAC3JOC_NATIVE_THREADS")
if override is not None:
value = int(override)
if value < 1:
raise ValueError("EAC3JOC_NATIVE_THREADS must be at least 1")
return min(value, MAX_OBJECTS)
return 2 if (os.cpu_count() or 1) >= 4 else 1
def _load_library(path):
lib = ctypes.CDLL(str(path))
float_p = ctypes.POINTER(ctypes.c_float)
u8_p = ctypes.POINTER(ctypes.c_uint8)
double_p = ctypes.POINTER(ctypes.c_double)
lib.ejoc_abi_version.argtypes = []
lib.ejoc_abi_version.restype = ctypes.c_uint32
lib.ejoc_build_info.argtypes = []
lib.ejoc_build_info.restype = ctypes.c_char_p
lib.ejoc_renderer_create.argtypes = []
lib.ejoc_renderer_create.restype = ctypes.c_void_p
lib.ejoc_renderer_destroy.argtypes = [ctypes.c_void_p]
lib.ejoc_renderer_destroy.restype = None
lib.ejoc_renderer_reset.argtypes = [ctypes.c_void_p]
lib.ejoc_renderer_reset.restype = ctypes.c_int
lib.ejoc_renderer_set_threads.argtypes = [ctypes.c_void_p, ctypes.c_uint32]
lib.ejoc_renderer_set_threads.restype = ctypes.c_int
lib.ejoc_renderer_thread_count.argtypes = [ctypes.c_void_p]
lib.ejoc_renderer_thread_count.restype = ctypes.c_uint32
lib.ejoc_renderer_last_error.argtypes = [ctypes.c_void_p]
lib.ejoc_renderer_last_error.restype = ctypes.c_char_p
lib.ejoc_renderer_process.argtypes = [
ctypes.c_void_p,
float_p,
float_p,
ctypes.c_uint32,
u8_p,
u8_p,
u8_p,
u8_p,
double_p,
ctypes.c_double,
ctypes.c_float,
ctypes.c_float,
float_p,
]
lib.ejoc_renderer_process.restype = ctypes.c_int
abi = int(lib.ejoc_abi_version())
if abi != ABI_VERSION:
raise NativeBackendUnavailable(f"native ABI mismatch: library={abi}, Python={ABI_VERSION}")
return lib
class NativeJocRenderer:
"""Stateful whole-frame native DSP renderer with the Python renderer API shape."""
def __init__(self, output_scale=1.0, library_path=None, threads=None):
self.output_scale = np.float32(output_scale)
if not np.isfinite(self.output_scale):
raise ValueError("output_scale must be finite")
self.library_path = find_native_library(library_path)
self._lib = _load_library(self.library_path)
self._handle = self._lib.ejoc_renderer_create()
if not self._handle:
raise MemoryError("ejoc_renderer_create failed")
requested_threads = default_native_threads() if threads is None else int(threads)
if requested_threads < 1:
raise ValueError("threads must be at least 1")
result = self._lib.ejoc_renderer_set_threads(self._handle, requested_threads)
if result:
self._raise_native("set_threads", result)
self.threads = int(self._lib.ejoc_renderer_thread_count(self._handle))
self._n_bands = np.zeros(MAX_OBJECTS, dtype=np.uint8)
self._n_dpoints = np.zeros(MAX_OBJECTS, dtype=np.uint8)
self._slope_idx = np.zeros(MAX_OBJECTS, dtype=np.uint8)
self._offset_ts = np.zeros((MAX_OBJECTS, MAX_DPOINTS), dtype=np.uint8)
self._dq = np.zeros(
(MAX_OBJECTS, MAX_DPOINTS, CORE_CHANNELS, MAX_BANDS),
dtype=np.float64,
)
self._output = np.zeros((16, FRAME_SAMPLES), dtype=np.float32)
@property
def build_info(self):
value = self._lib.ejoc_build_info()
return value.decode("utf-8", "replace") if value else ""
@staticmethod
def decode_payload(payload_bytes):
subs, _ = unpack_evolution(payload_bytes, loose=True)
return NativeJocRenderer.decode_subpayloads(subs)
@staticmethod
def decode_subpayloads(subs):
if 14 not in subs:
raise ValueError("EMDF missing ID14/JOC")
out = parse_joc(subs[14])
mix_q = diff_decode(out)
mix_dq = dequantize(out, mix_q)
return out, mix_q, mix_dq
def reset(self):
self._require_open()
result = self._lib.ejoc_renderer_reset(self._handle)
if result:
self._raise_native("reset", result)
def close(self):
handle = getattr(self, "_handle", None)
if handle:
self._lib.ejoc_renderer_destroy(handle)
self._handle = None
def __enter__(self):
return self
def __exit__(self, exc_type, exc, tb):
self.close()
def __del__(self):
try:
self.close()
except Exception:
pass
def _require_open(self):
if not self._handle:
raise RuntimeError("native renderer is closed")
def _raise_native(self, operation, code):
raw = self._lib.ejoc_renderer_last_error(self._handle)
detail = raw.decode("utf-8", "replace") if raw else "unknown native error"
raise RuntimeError(f"native {operation} failed ({code}): {detail}")
def _pack_frame(self, out, mix_dq):
if out["n_channels"] != CORE_CHANNELS:
raise ValueError(f"native core requires 5 JOC channels, got {out['n_channels']}")
if out["n_objects"] > MAX_OBJECTS:
raise ValueError(f"native core supports at most 15 objects, got {out['n_objects']}")
self._n_bands.fill(0)
self._n_dpoints.fill(0)
self._slope_idx.fill(0)
self._offset_ts.fill(0)
mask = 0
for object_index, info in enumerate(out["objs"]):
if not info["present"]:
continue
if info["sparse"]:
raise ValueError("native core does not accept unvalidated Sparse JOC")
bands = int(info["n_bands"])
points = int(info["n_dpoints"])
if bands > MAX_BANDS or points > MAX_DPOINTS:
raise ValueError(f"native descriptor out of range: bands={bands}, points={points}")
values = np.asarray(mix_dq[object_index], dtype=np.float64)
expected = (points, CORE_CHANNELS, bands)
if values.shape != expected:
raise ValueError(f"object {object_index} dq shape {values.shape}, expected {expected}")
mask |= 1 << object_index
self._n_bands[object_index] = bands
self._n_dpoints[object_index] = points
self._slope_idx[object_index] = int(info["slope_idx"])
offsets = info.get("offset_ts", ())
self._offset_ts[object_index, :len(offsets)] = offsets
self._dq[object_index, :points, :, :bands] = values
return mask
def render_frame(self, payload_bytes, bed5_pcm, lfe_pcm=None):
subs, _ = unpack_evolution(payload_bytes, loose=True)
return self.render_subpayloads(subs, bed5_pcm, lfe_pcm)
def render_subpayloads(self, subs, bed5_pcm, lfe_pcm=None):
self._require_open()
out, _, mix_dq = self.decode_subpayloads(subs)
object_mask = self._pack_frame(out, mix_dq)
bed5 = np.ascontiguousarray(bed5_pcm, dtype=np.float32)
if bed5.shape != (CORE_CHANNELS, FRAME_SAMPLES):
raise ValueError(f"core PCM shape must be (5,1536), got {bed5.shape}")
if lfe_pcm is None:
lfe = None
lfe_ptr = ctypes.POINTER(ctypes.c_float)()
else:
lfe = np.ascontiguousarray(lfe_pcm, dtype=np.float32)
if lfe.shape != (FRAME_SAMPLES,):
raise ValueError(f"LFE shape must be (1536,), got {lfe.shape}")
lfe_ptr = lfe.ctypes.data_as(ctypes.POINTER(ctypes.c_float))
result = self._lib.ejoc_renderer_process(
self._handle,
bed5.ctypes.data_as(ctypes.POINTER(ctypes.c_float)),
lfe_ptr,
object_mask,
self._n_bands.ctypes.data_as(ctypes.POINTER(ctypes.c_uint8)),
self._n_dpoints.ctypes.data_as(ctypes.POINTER(ctypes.c_uint8)),
self._slope_idx.ctypes.data_as(ctypes.POINTER(ctypes.c_uint8)),
self._offset_ts.ctypes.data_as(ctypes.POINTER(ctypes.c_uint8)),
self._dq.ctypes.data_as(ctypes.POINTER(ctypes.c_double)),
float(out["clipgain"]),
ctypes.c_float(0.0625),
ctypes.c_float(self.output_scale),
self._output.ctypes.data_as(ctypes.POINTER(ctypes.c_float)),
)
if result:
self._raise_native("process", result)
return self._output, None
def native_available(library_path=None):
try:
path = find_native_library(library_path)
lib = _load_library(path)
return True, str(path), (lib.ejoc_build_info() or b"").decode("utf-8", "replace")
except Exception as exc:
return False, None, str(exc)
+216
View File
@@ -0,0 +1,216 @@
"""OAMD 位载荷 → 16 个对象槽的 q1/q2/q3 增量状态。
槽 0 是 bed/LFE;槽 1..15 对应输出 ch1..15 的对象元数据。
"""
import numpy as np
from variant_error import UnsupportedVariantError, bytes_descriptor
Q1_OFF = 192
N_Q12 = 62
N_Q3 = 15
SAMPLE_OFFSET_INDEX = (8, 16, 18, 24)
RAMP_DURATIONS = (0, 512, 1536)
RAMP_DURATION_INDEX = (
32, 64, 128, 256, 320, 480, 1000, 1001,
1024, 1600, 1601, 1602, 1920, 2000, 2002, 2048,
)
def q_of(k, n):
q = int(np.floor(32768.0 * k / n + 0.5))
return min(32767, q)
def _payload_bits(bits_one):
if isinstance(bits_one, (bytes, bytearray, memoryview)):
src = np.frombuffer(bits_one, dtype=np.uint8)
else:
src = np.asarray(bits_one, dtype=np.uint8)
if src.ndim == 1 and src.shape[0] in (536, 552):
bits = src
elif src.size in (67, 69):
bits = np.unpackbits(src.reshape(-1), bitorder="big")
else:
raw = (bytes(bits_one) if isinstance(bits_one, (bytes, bytearray, memoryview))
else np.asarray(bits_one, dtype=np.uint8).tobytes())
raise UnsupportedVariantError(
"oamd", f"payload_length_{len(raw)}B",
f"发现未覆盖的 OAMD 载荷长度 {len(raw)}B",
details={
"supported_payload_bytes": [67, 69],
"payload": bytes_descriptor(raw),
"repair_hint": "检查 OAMD header、element 数量及可选字段造成的位偏移变化",
})
raw_payload = np.packbits(bits, bitorder="big").tobytes()
return bits, raw_payload
class _BitReader:
def __init__(self, bits):
self.bits = bits
self.position = 0
def read(self, count):
end = self.position + count
if end > len(self.bits):
raise ValueError(f"OAMD 位流越界: bit={self.position}, need={count}")
value = 0
for bit in self.bits[self.position:end]:
value = (value << 1) | int(bit)
self.position = end
return value
def skip(self, count):
self.read(count)
def _variable_bits(reader, width, max_groups=5):
value = 0
for _ in range(max_groups + 1):
value += reader.read(width)
if not reader.read(1):
return value
value = (value + 1) << width
raise ValueError(f"OAMD variable_bits({width}) 延伸组过多")
def _update_timing(bits, alternate_object_present, element_count, raw_payload):
"""按 OAMD element/MDUpdateInfo 读取位置块的开始偏移和 ramp 时长。"""
reader = _BitReader(bits)
reader.position = 14
for _ in range(element_count):
element_index = reader.read(4)
element_length = _variable_bits(reader, 4)
element_end = reader.position + element_length + 1
reader.skip(5 if alternate_object_present else 1)
if element_index == 1:
offset_code = reader.read(2)
if offset_code == 0:
sample_offset = 0
elif offset_code == 1:
sample_offset = SAMPLE_OFFSET_INDEX[reader.read(2)]
elif offset_code == 2:
sample_offset = reader.read(5)
else:
raise UnsupportedVariantError(
"oamd", "md_sample_offset_mode",
"OAMD 使用了当前未覆盖的 MD sample-offset 模式",
details={"payload": bytes_descriptor(raw_payload)})
block_count = reader.read(3) + 1
blocks = []
for _block in range(block_count):
block_offset_factor = reader.read(6)
block_offset = sample_offset + block_offset_factor * 32
ramp_code = reader.read(2)
if ramp_code == 3:
if reader.read(1):
ramp_duration = RAMP_DURATION_INDEX[reader.read(4)]
else:
ramp_duration = reader.read(11)
else:
ramp_duration = RAMP_DURATIONS[ramp_code]
blocks.append((block_offset, ramp_duration))
if len(blocks) != 1:
raise UnsupportedVariantError(
"oamd", "multiple_position_blocks",
"OAMD 一帧含多个对象位置更新块,固定位置窗口不能安全套用",
details={
"block_count": len(blocks),
"blocks": blocks,
"payload": bytes_descriptor(raw_payload),
"repair_hint": "按 ObjectInfoBlock 顺序逐块解析坐标,再生成分段 ADM ramp",
})
return blocks[0]
reader.position = element_end
raise UnsupportedVariantError(
"oamd", "missing_object_element",
"OAMD 中没有 object element (element_index=1)",
details={"payload": bytes_descriptor(raw_payload)})
def frame_update(bits_one):
"""单帧 OAMD → 位置字段增量及其 sample offset/ramp duration。"""
bits, raw_payload = _payload_bits(bits_one)
header = {
"version": int((bits[0] << 1) | bits[1]),
"objects_minus_one": int(sum(int(bits[2 + i]) << (4 - i) for i in range(5))),
"dynamic_object_only": int(bits[7]),
"lfe_present": int(bits[8]),
"alternate_object_present": int(bits[9]),
"element_count": int(sum(int(bits[10 + i]) << (3 - i) for i in range(4))),
}
if raw_payload[:2] != b"\x1f\x88":
raise UnsupportedVariantError(
"oamd", f"header_signature_{raw_payload[:2].hex()}",
"OAMD 长度已知,但 header 与当前位置字段布局不一致",
details={
"supported_header_prefix_hex": "1f88",
"header_probe": header,
"payload": bytes_descriptor(raw_payload),
"repair_hint": "按新 header 的 program assignment 和 element 布局重新定位对象位置字段",
})
block_offset, ramp_duration = _update_timing(
bits, bool(header["alternate_object_present"]), header["element_count"], raw_payload)
weights = 1 << np.arange(7, -1, -1)
out = {}
for obj in range(16):
start = 112 + 31 * (obj - 3)
wq1 = int((bits[start:start + 8] * weights).sum())
wq2 = int((bits[start + 8:start + 16] * weights).sum())
wq3 = int((bits[start + 16:start + 24] * weights).sum())
if obj and (wq1 >> 6 != 3 or wq2 & 2 != 2 or wq3 & 0x1F != 1):
raise UnsupportedVariantError(
"oamd", "position_layout_signature",
"OAMD 长度和 header 已知,但对象位置字段标记或位偏移发生变化",
details={
"object_slot": obj,
"position_start_bit": start,
"q1_window_hex": f"{wq1:02x}",
"q2_window_hex": f"{wq2:02x}",
"q3_window_hex": f"{wq3:02x}",
"header_probe": header,
"payload": bytes_descriptor(raw_payload),
"repair_hint": "解析 OAMD element 可选字段并更新每个对象的位置窗口偏移",
})
k1 = wq1 - Q1_OFF
if obj == 0:
out[(0, "q1")] = None
out[(0, "q2")] = None
else:
out[(obj, "q1")] = q_of(k1, N_Q12) if 0 <= k1 <= N_Q12 else None
k2 = wq2 >> 2
if obj != 0:
out[(obj, "q2")] = q_of(k2, N_Q12) if 0 <= k2 <= N_Q12 else None
k3 = (wq2 & 1) * 8 + (wq3 >> 5)
out[(obj, "q3")] = q_of(k3, N_Q3) if 0 <= k3 <= N_Q3 else None
return {
"values": out,
"block_offset_samples": block_offset,
"ramp_duration_samples": ramp_duration,
}
def frame_update_values(bits_one):
"""兼容接口:只返回 ``{(slot, field): q|None}``。"""
return frame_update(bits_one)["values"]
class JocFieldState:
"""未更新/非法窗保持旧值;slot0 q1/q2 的 DLL 初值为中心 16384。"""
def __init__(self):
self.q = {(obj, field): 0 for obj in range(16)
for field in ("q1", "q2", "q3")}
self.q[(0, "q1")] = 16384
self.q[(0, "q2")] = 16384
def apply(self, updates):
for key, value in updates.items():
if value is not None:
self.q[key] = value
return self
def snapshot(self):
return dict(self.q)
+227
View File
@@ -0,0 +1,227 @@
"""Evolution id11/OAMD 帧序列 → ADM 15 对象关键帧。"""
import math
from adm_atmos import q_to_adm_xyz
from oamd_bits import JocFieldState, frame_update
from variant_error import UnsupportedVariantError
def _lerp_xyz(start, target, amount):
return tuple(a + (b - a) * amount for a, b in zip(start, target))
def _append_point(points, sample, xyz, interpolation_samples):
item = (int(sample), *map(float, xyz), int(interpolation_samples))
if points and item[0] == points[-1][0]:
points[-1] = item
elif not points or item[0] > points[-1][0]:
points.append(item)
def _expand_events_dense64(events, total_samples, rate, update_quantum_samples,
object_delay_samples, object_index):
if not events:
return [(0.0, 0.0, 0.0, 0.0, total_samples / float(rate), 0.0)]
# 初始位置从成品 sample 0 起有效;合成延迟只作用于后续位置变化。
current = events[0][1]
points = []
_append_point(points, 0, current, 0)
for event_index, (coded_start, target, ramp_samples) in enumerate(events[1:], 1):
start = coded_start + object_delay_samples
if start >= total_samples:
break
if start < points[-1][0]:
raise UnsupportedVariantError(
"oamd", "non_monotonic_position_updates",
"对象位置更新时间倒退,无法生成连续 ADM 轨迹",
details={"object": object_index, "sample": start,
"previous_sample": points[-1][0]})
# 在运动起点保留上一位置,避免下游把长时间静止段直接连到首个中间点。
if start > points[-1][0]:
_append_point(points, start, current, 0)
effective_ramp = max(0, int(ramp_samples) - update_quantum_samples)
if effective_ramp == 0:
_append_point(points, start, target, 0)
current = target
continue
end = start + math.ceil(effective_ramp / update_quantum_samples) * update_quantum_samples
if event_index + 1 < len(events):
next_start = events[event_index + 1][0] + object_delay_samples
if next_start < end:
raise UnsupportedVariantError(
"oamd", "overlapping_position_ramps",
"同一对象的新位置更新在上一 ramp 完成前到达",
details={
"object": object_index,
"ramp_start_sample": start,
"ramp_end_sample": end,
"next_update_sample": next_start,
"repair_hint": "按 64-sample 状态机截断旧 ramp,再从当前插值位置启动新 ramp",
})
# 逐位置更新节拍复现状态机。1536-sample ramp 在首次 64-sample
# 更新后剩余 1472 samples,因此共有 23 个中间/终点坐标。
future = effective_ramp
elapsed = 0
position = current
while future > 0:
amount = min(update_quantum_samples / float(future), 1.0)
position = _lerp_xyz(position, target, amount)
elapsed += update_quantum_samples
sample = start + elapsed
if sample >= total_samples:
break
_append_point(points, sample, position, update_quantum_samples)
future -= update_quantum_samples
current = target
blocks = []
for point_index, (sample, x, y, z, interpolation_samples) in enumerate(points):
end = points[point_index + 1][0] if point_index + 1 < len(points) else total_samples
duration_samples = max(0, end - sample)
if duration_samples == 0:
continue
interpolation_samples = min(interpolation_samples, duration_samples)
blocks.append((sample / float(rate), x, y, z,
duration_samples / float(rate),
interpolation_samples / float(rate)))
return blocks
def _compact_events(events, total_samples, rate, update_quantum_samples,
object_delay_samples, object_index):
"""Represent each linear OAMD ramp with one ADM interpolation block.
The existing dense64 representation keeps the old position at ``start``,
writes its first interpolated target at ``start + quantum``, and lets ADM
interpolate that block over one quantum. Consequently, the interpreted
motion begins at ``start + quantum`` and reaches the final target at
``start + ramp_duration``. This compact form preserves that timing with one
target block whose interpolationLength is ``ramp_duration - quantum``.
"""
if not events:
return [(0.0, 0.0, 0.0, 0.0, total_samples / float(rate), 0.0)]
points = []
current = events[0][1]
_append_point(points, 0, current, 0)
for event_index, (coded_start, target, ramp_samples) in enumerate(events[1:], 1):
event_start = coded_start + object_delay_samples
if event_start >= total_samples:
break
effective_ramp = max(0, int(ramp_samples) - update_quantum_samples)
block_start = event_start + (update_quantum_samples if effective_ramp else 0)
if block_start >= total_samples:
break
ramp_end = block_start + effective_ramp
if block_start < points[-1][0]:
raise UnsupportedVariantError(
"oamd", "non_monotonic_compact_position_updates",
"紧凑对象位置更新时间倒退",
details={"object": object_index, "sample": block_start,
"previous_sample": points[-1][0]})
if event_index + 1 < len(events):
next_coded_start, _, next_ramp_samples = events[event_index + 1]
next_event_start = next_coded_start + object_delay_samples
next_effective = max(0, int(next_ramp_samples) - update_quantum_samples)
next_block_start = next_event_start + (update_quantum_samples if next_effective else 0)
if next_block_start < ramp_end:
raise UnsupportedVariantError(
"oamd", "overlapping_compact_position_ramps",
"同一对象的新位置更新在上一紧凑 ramp 完成前到达",
details={
"object": object_index,
"ramp_start_sample": block_start,
"ramp_end_sample": ramp_end,
"next_update_sample": next_block_start,
"repair_hint": "对此变体使用 --trajectory-mode dense64 并检查 OAMD 调度",
})
block_target = target
block_interpolation = effective_ramp
available = total_samples - block_start
if effective_ramp > available:
block_target = _lerp_xyz(current, target, available / float(effective_ramp))
block_interpolation = available
_append_point(points, block_start, block_target, block_interpolation)
current = target
blocks = []
for point_index, (sample, x, y, z, interpolation_samples) in enumerate(points):
end = points[point_index + 1][0] if point_index + 1 < len(points) else total_samples
duration_samples = max(0, end - sample)
if duration_samples == 0:
continue
interpolation_samples = min(interpolation_samples, duration_samples)
blocks.append((sample / float(rate), x, y, z,
duration_samples / float(rate),
interpolation_samples / float(rate)))
return blocks
def _expand_events(events, total_samples, rate, update_quantum_samples,
object_delay_samples, object_index, trajectory_mode):
if trajectory_mode == "compact":
return _compact_events(events, total_samples, rate, update_quantum_samples,
object_delay_samples, object_index)
if trajectory_mode == "dense64":
return _expand_events_dense64(events, total_samples, rate, update_quantum_samples,
object_delay_samples, object_index)
raise ValueError(f"未知 trajectory_mode: {trajectory_mode}")
def build_adm_tracks(index, frames=None, rate=48000, frame_samples=1536,
update_quantum_samples=64, object_delay_samples=640,
trajectory_mode="compact"):
"""从统一 metadata index 构造 15 条 ADM 轨迹。
返回 ``[(name, [(rtime,x,y,z,duration,interpolation), ...]), ...]``。
OAMD 的内外层 sample offset、block offset 和 ramp 均保留。
``trajectory_mode="compact"`` 用一个长 ADM interpolation block 表示每条
线性 ramp;``dense64`` 保留逐 64-sample 展开作为兼容回退。
``object_delay_samples`` 将位置更新与对象逆 QMF 的输出时刻对齐。
slot1..15 与对象 PCM ch1..15 一一对应。
"""
frames = index.rows if frames is None else frames
state = JocFieldState()
events = [[] for _ in range(15)]
previous = [None] * 15
for seq, row in enumerate(frames):
subs = index.subpayloads(row)
timing = None
if 11 in subs:
timing = frame_update(subs[11])
state.apply(timing["values"])
event_sample = seq * frame_samples
ramp_samples = 0
if timing is not None:
outer_offset = (index.subpayload_sample_offset(row, 11)
if hasattr(index, "subpayload_sample_offset") else 0)
event_sample += outer_offset + timing["block_offset_samples"]
ramp_samples = timing["ramp_duration_samples"]
q = state.q
for obj in range(1, 16):
xyz = q_to_adm_xyz(q[(obj, "q1")], q[(obj, "q2")], q[(obj, "q3")])
if previous[obj - 1] != xyz:
events[obj - 1].append((event_sample, xyz, ramp_samples))
previous[obj - 1] = xyz
total_samples = len(frames) * frame_samples
return [
(f"JOC_Object_{obj}",
_expand_events(events[obj - 1], total_samples, rate,
update_quantum_samples, object_delay_samples, obj,
trajectory_mode))
for obj in range(1, 16)
]
+259
View File
@@ -0,0 +1,259 @@
"""把 E-AC-3 核心 PCM 和 JOC 元数据重建为 LFE 加 15 路对象 PCM。
管线(每帧):
核心 5.1 PCM → QMF 分析 x → 填充器 z = Σ m·x(每对象)
→ 对象时域组装(前步、64 点 FFT、旋转、合成窗)→ ×16 + clamp
→ 对象波形 × joc_clipgain
+ LFE 专用 1217-sample 环形延迟
→ 16ch(ch0 = LFE, ch1-15 = 15 对象)→ ×output_scale
output_scale:
默认 1.0(0 dB)。用户选择的 dB 在命令行转换为 float32 系数;该系数
不进入 JOC 数学本体,也不是 joc_clipgain。
"""
import numpy as np
from joc_decode import parse_joc, diff_decode, dequantize
from joc_qmf import (N, QMF5_WINDOW, qmf_analysis_frame,
surround_post_frame, interp_matrix)
from evo_unpack import unpack_evolution
CORE_CHANNELS = [0, 1, 2, 4, 5] # L R C Ls Rs(EAC3 5.1 核心顺序)
class JocRenderer:
"""E-AC-3 JOC 帧渲染器。状态跨帧保持(FIFO/插值 prev/合成状态)。"""
def __init__(self, output_scale=1.0):
# 成品增益明确按 float32 运算;默认值为 1.0。
self.output_scale = np.float32(output_scale)
self._prev = {} # 每对象插值 prev(m 的帧末 sb 值)
self._synthesis_state = np.zeros((15, 640), dtype=np.float64)
# 分析 QMF:五路各自保留 9×64 FIFO;L/R/C 另有 10-timeslot 延迟。
self._analysis_fifo = np.zeros((5, 9, 64), dtype=np.float64)
self._analysis_delay = np.zeros((3, 10, 64), dtype=np.float32)
self._analysis_phase = np.float32(0.0625)
self._surround_qmf_delay = np.zeros((2, 10, 64), dtype=np.complex128)
self._surround_dc_hist = np.zeros((2, 20), dtype=np.complex128)
# ch0/LFE 使用循环历史,读指针恒落后写指针 1217 个采样。
# phase=1/16 与输出端 *16 抵消,净效果是 LFE 延迟。
self._lfe_delay = np.zeros(1217, dtype=np.float64)
self._last_x = None # 仅供逐槽诊断读取,不参与状态推进
self._rot = None # 每带旋转表
@staticmethod
def decode_payload(payload_bytes):
"""id14 载荷 → (parse 结果, mix_q, mix_dq)。"""
subs, _ = unpack_evolution(payload_bytes, loose=True)
out = parse_joc(subs[14])
mix_q = diff_decode(out)
mix_dq = dequantize(out, mix_q)
return out, mix_q, mix_dq
@staticmethod
def decode_subpayloads(subs):
"""已解析 EMDF 子载荷 → (parse 结果, mix_q, mix_dq)。"""
if 14 not in subs:
raise ValueError("EMDF 缺少 ID14/JOC")
out = parse_joc(subs[14])
mix_q = diff_decode(out)
mix_dq = dequantize(out, mix_q)
return out, mix_q, mix_dq
def qmf_x(self, bed5, phase_new=0.0625):
"""核心 5ch PCM → 对象矩阵使用的复数 QMF ``x``。
先以 float32 对当前帧应用 phase;phase 变化时仅前 256 个样本从旧值
线性过渡。缩放后的 L/R/C 延迟 10 槽,Ls/Rs 不延迟,再进入分析 QMF。
"""
pcm = np.asarray(bed5, dtype=np.float32)
if pcm.shape != (5, 1536):
raise ValueError(f"核心 5ch 帧应为 (5,1536),实际 {pcm.shape}")
new_phase = np.float32(phase_new)
old_phase = self._analysis_phase
gains = np.full(1536, new_phase, dtype=np.float32)
if old_phase != new_phase:
step = np.float32((new_phase - old_phase) / np.float32(256.0))
gains[:256] = old_phase + np.arange(256, dtype=np.float32) * step
scaled = np.multiply(pcm, gains[None, :], dtype=np.float32).reshape(5, 24, 64)
blocks = np.empty_like(scaled)
for ch in range(3):
delayed = np.concatenate((self._analysis_delay[ch], scaled[ch]), axis=0)
blocks[ch] = delayed[:24]
self._analysis_delay[ch] = delayed[24:]
blocks[3:] = scaled[3:]
x, self._analysis_fifo = qmf_analysis_frame(self._analysis_fifo, blocks)
self._analysis_phase = new_phase
x[3:], self._surround_qmf_delay, self._surround_dc_hist = surround_post_frame(
x[3:], self._surround_qmf_delay, self._surround_dc_hist)
return x
def object_z(self, out, mix_dq, x):
"""计算 ``z[obj] = Σ_ch m[ch,sb,ts]·x[ch,sb,ts]``。"""
z_all = {}
new_prev = {}
for obj, o in enumerate(out["objs"]):
if not o["present"]:
continue
dq = mix_dq[obj]
prev = self._prev.get(obj)
if prev is None:
prev = np.zeros((out["n_channels"], N), dtype=np.float64)
m = interp_matrix(o, dq, prev) # [ch][sb][ts](内部按 o["n_bands"] 映射)
z = np.sum(x * m, axis=0) # [sb][ts](复)
z_all[obj] = z
new_prev[obj] = m[:, :, -1]
self._prev.update(new_prev)
return z_all
def _rot_table_86840(self):
"""生成 ``θ=πk/128`` 的 ``0.5·(sin θ, cos θ)`` 旋转表。"""
if self._rot is None:
k = np.arange(64)
theta = np.pi * k / 128.0
self._rot = np.empty(128, dtype=np.float64)
self._rot[0::2] = 0.5 * np.sin(theta) # sin 分量
self._rot[1::2] = 0.5 * np.cos(theta) # cos 分量
return self._rot
@staticmethod
def _front_step(src):
"""重排 128 个交织复数分量:
outA[2k] = src[4k]、outA[2k+1] = −src[4k+1](区 [0:64]);
outB[126−2k] = src[4k+2]、outB[127−2k] = src[4k+3](区 [64:128] 倒序)。"""
src = np.asarray(src, dtype=np.float64)
zone = np.empty_like(src)
k = np.arange(32)
zone[..., 2 * k] = src[..., 4 * k]
zone[..., 2 * k + 1] = -src[..., 4 * k + 1]
zone[..., 126 - 2 * k] = src[..., 4 * k + 2]
zone[..., 127 - 2 * k] = src[..., 4 * k + 3]
return zone
@staticmethod
def _h880_vec(a2):
"""对 128 个 re/im 交织值执行标准 64 点复数 FFT。"""
a2 = np.asarray(a2, dtype=np.float64)
zc = a2[..., 0::2] + 1j * a2[..., 1::2]
f = np.fft.fft(zc, axis=-1)
a1 = np.empty_like(a2)
a1[..., 0::2] = f.real
a1[..., 1::2] = f.imag
return a1
@staticmethod
def _h0b0_vec(a2, a3):
"""NumPy 向量化的 ``out = 2·复乘(a2, a3)``,re/im 交织。"""
a2 = np.asarray(a2, dtype=np.float64)
a3 = np.asarray(a3, dtype=np.float64)
shape = a2.shape[:-1] + (16, 4)
v11 = a2[..., 1::2].reshape(shape)
v12 = a2[..., 0::2].reshape(shape)
v13 = a3[..., 0::2].reshape(a3.shape[:-1] + (16, 4))
v14 = a3[..., 1::2].reshape(a3.shape[:-1] + (16, 4))
v15 = (v11 * v14 - v12 * v13) * 2.0
v16 = (v12 * v14 + v11 * v13) * 2.0
out = np.empty_like(a2)
out[..., 0::2] = v16.reshape(a2.shape[:-1] + (64,))
out[..., 1::2] = v15.reshape(a2.shape[:-1] + (64,))
return out
@staticmethod
def _qmf5_vec(state, win_flat, rot):
"""推进合成窗状态并返回 64 个时域样本。"""
state = np.asarray(state, dtype=np.float64)
rot = np.asarray(rot, dtype=np.float64)
single = state.ndim == 1
if single:
state = state[None, :]
if rot.ndim == 1:
rot = np.broadcast_to(rot, (len(state), len(rot)))
rot2 = rot.reshape(len(state), 16, 8)
v22 = rot2[..., 0:8:2]
v19 = rot2[..., 1:8:2]
S = state[:, :576].reshape(len(state), 16, 9, 4)
w10 = win_flat[:640].reshape(10, 64)
idx = np.arange(16)[:, None] * 4 + np.arange(4)
w0 = w10[0, idx][None, ...]
w_all = w10[1:10, idx].transpose(1, 0, 2)[None, ...]
out64 = (2.0 * (w0 * v22 + S[:, :, 0])).reshape(len(state), 64)
# 输出使用窗行 0,状态第 0 行从窗行 1 开始推进。
S[:, :, 0] = w_all[:, :, 0] * v19 + S[:, :, 1]
for k in range(7):
alt = v22 if k % 2 == 0 else v19
S[:, :, k + 1] = w_all[:, :, k + 1] * alt + S[:, :, k + 2]
S[:, :, 8] = w_all[:, :, 8] * v19
return out64[0] if single else out64
def dll_synth_objects(self, z_all):
"""批量执行对象逆 QMF,返回对象 PCM 字典。
每个对象保持独立合成状态,64 点 FFT 和窗核在对象维批量计算。
"""
object_ids = sorted(z_all)
if not object_ids:
return {}
z = np.stack([z_all[obj] for obj in object_ids], axis=0)
state = self._synthesis_state[object_ids].copy()
rot868 = self._rot_table_86840()
pcm = np.zeros((len(object_ids), 1536), dtype=np.float64)
for tsg in range(6):
ring = np.zeros((len(object_ids), 512), dtype=np.float64)
chunk = z[:, :, tsg * 4:tsg * 4 + 4].transpose(0, 2, 1)
slots = ring.reshape(len(object_ids), 4, 128)
slots[..., 0::2] = chunk.real
slots[..., 1::2] = chunk.imag
for it in range(4):
zone = self._front_step(ring[:, 128 * it:128 * it + 128])
zone = self._h0b0_vec(self._h880_vec(zone), rot868)
ring[:, 64 * it:64 * it + 64] = self._qmf5_vec(state, QMF5_WINDOW, zone)
pcm[:, tsg * 256:(tsg + 1) * 256] = np.clip(16.0 * ring[:, :256], -1.0, 1.0)
self._synthesis_state[object_ids] = state
return {obj: pcm[i] for i, obj in enumerate(object_ids)}
def dll_synth_object(self, obj, z):
"""单对象兼容入口;整帧渲染使用对象维批量实现。"""
return self.dll_synth_objects({obj: z})[obj]
@staticmethod
def _win_flat():
"""返回逆 QMF 使用的 640 项有效合成窗表。"""
return QMF5_WINDOW
def decode_lfe(self, lfe_pcm):
"""将核心 ch3/LFE 经过跨帧 1217-sample 延迟后输出到 ch0。"""
current = np.asarray(lfe_pcm, dtype=np.float64)
if current.shape != (1536,):
raise ValueError(f"LFE 帧应为 (1536,),实际 {current.shape}")
delayed = np.concatenate([self._lfe_delay, current])
out = np.clip(delayed[:1536], -1.0, 1.0)
self._lfe_delay = delayed[-1217:].copy()
return out
def render_frame(self, payload_bytes, bed5_pcm, lfe_pcm=None):
"""payload_bytes = evolution 载荷(含 id14 JOC);bed5_pcm = 5×1536 核心 PCM。
lfe_pcm = 1536 核心 LFE;提供时生成 ch0,省略时 ch0 静音
(保留给只验证对象/z 链的诊断脚本)。
返回 (pcm16, z_all):pcm16 = 16×1536(ch0 LFE + 15 对象)f32,
已用 FP32 系数乘 output_scale。"""
subs, _ = unpack_evolution(payload_bytes, loose=True)
return self.render_subpayloads(subs, bed5_pcm, lfe_pcm)
def render_subpayloads(self, subs, bed5_pcm, lfe_pcm=None):
"""以已拆出的 EMDF payload 字典渲染一帧,避免绑定 transport 容器。"""
out, mix_q, mix_dq = self.decode_subpayloads(subs)
x = self.qmf_x(bed5_pcm)
self._last_x = x.copy()
z_all = self.object_z(out, mix_dq, x)
# joc_clipgain 在对象逆 QMF 后应用,并与用户 output_scale 分离。
objs_pcm = self.dll_synth_objects(z_all)
pcm16 = np.zeros((16, 1536), dtype=np.float64)
pcm16[0] = self.decode_lfe(lfe_pcm) if lfe_pcm is not None else 0.0
for obj, p in objs_pcm.items():
pcm16[obj + 1] = p * out["clipgain"]
pcm16_f32 = np.asarray(pcm16, dtype=np.float32)
return np.multiply(pcm16_f32, self.output_scale, dtype=np.float32), z_all
+71
View File
@@ -0,0 +1,71 @@
"""Unified auto/native/python entry point for float64 speaker rendering."""
from __future__ import annotations
import time
import numpy as np
from speaker_layouts import SpeakerLayout, get_speaker_layout
from speaker_native_renderer import NativeSpeakerRenderer
from speaker_renderer import PythonSpeakerRenderer, payload_events_from_index, render_objects16
FRAME_SAMPLES = 1536
def create_speaker_renderer(layout: str | SpeakerLayout, *, backend="auto", native_library=None):
"""Create a stateful speaker renderer and return ``(renderer, info)``."""
if backend not in ("auto", "native", "python"):
raise ValueError(f"unknown speaker backend: {backend}")
target = get_speaker_layout(layout) if isinstance(layout, str) else layout
fallback_reason = None
if backend in ("auto", "native"):
try:
renderer = NativeSpeakerRenderer(target, native_library)
return renderer, {
"name": "native",
"layout": target.name,
"library": str(renderer.library_path),
"fallback_reason": None,
}
except (OSError, RuntimeError) as exc:
fallback_reason = str(exc)
return PythonSpeakerRenderer(target), {
"name": "python",
"layout": target.name,
"library": None,
"fallback_reason": fallback_reason,
}
def render_speaker_layout(objects16, index, layout: str | SpeakerLayout, *,
backend="auto", native_library=None,
metadata_offset=1473):
"""Render a complete indexed object stream and return ``(pcm64, info)``."""
target = get_speaker_layout(layout) if isinstance(layout, str) else layout
source = np.asarray(objects16)
if source.ndim != 2 or source.shape[1] != 16:
raise ValueError(f"objects16 must have shape [samples,16], got {source.shape}")
if len(source) % FRAME_SAMPLES:
raise ValueError("objects16 sample count must be divisible by 1536")
frames = len(source) // FRAME_SAMPLES
if frames > len(index.rows):
raise ValueError(f"metadata has {len(index.rows)} frames, PCM needs {frames}")
renderer, info = create_speaker_renderer(
target, backend=backend, native_library=native_library)
output = np.empty((len(source), target.channel_count), dtype=np.float64)
started = time.perf_counter()
try:
for frame in range(frames):
start = frame * FRAME_SAMPLES
subs = index.subpayloads(index.rows[frame])
output[start:start + FRAME_SAMPLES] = renderer.render_frame(
source[start:start + FRAME_SAMPLES], subs.get(11), metadata_offset)
finally:
close = getattr(renderer, "close", None)
if close is not None:
close()
info = dict(info)
info["seconds"] = time.perf_counter() - started
return output, info
+764
View File
@@ -0,0 +1,764 @@
"""Standard speaker-layout geometry used by the spatial renderer.
Coordinates are unsigned Q15 room coordinates. This is geometry/topology data,
not a sampled gain table.
"""
from __future__ import annotations
from dataclasses import dataclass
@dataclass(frozen=True)
class RegionGeometry:
coordinates_q15: tuple[tuple[int, int, int], ...]
speaker_ids: tuple[int, ...]
axis0_groups: tuple[tuple[int, ...], ...]
axis1_groups: tuple[tuple[int, ...], ...]
mode: int
@dataclass(frozen=True)
class SpeakerLayout:
name: str
out_ch_config: int
speaker_bitfield: int
channels: tuple[str, ...]
internal_to_standard: tuple[int, ...]
regions: tuple[RegionGeometry, ...]
@property
def channel_count(self) -> int:
return len(self.channels)
_RAW_LAYOUTS = {'20': {'out_ch_config': 0,
'speaker_bitfield': 1,
'channels': ('L', 'R'),
'internal_to_standard': (0, 1),
'regions': ({'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
'speaker_ids': (0, 1),
'axis0_groups': ((0, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
'speaker_ids': (0, 1),
'axis0_groups': ((0, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
'speaker_ids': (0, 1),
'axis0_groups': ((0, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
'speaker_ids': (0, 1),
'axis0_groups': ((0, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
'speaker_ids': (0, 1),
'axis0_groups': ((0, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
'speaker_ids': (0, 1),
'axis0_groups': ((0, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
'speaker_ids': (0, 1),
'axis0_groups': ((0, 1),),
'axis1_groups': (),
'mode': 1})},
'31': {'out_ch_config': 3,
'speaker_bitfield': 7,
'channels': ('L', 'R', 'C', 'LFE'),
'internal_to_standard': (0, 1, 2, 3),
'regions': ({'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((16384, 0, 0),),
'speaker_ids': (2,),
'axis0_groups': ((0,),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1})},
'51': {'out_ch_config': 7,
'speaker_bitfield': 15,
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs'),
'internal_to_standard': (0, 1, 2, 3, 4, 5),
'regions': ({'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
'speaker_ids': (0, 1, 2, 4, 5),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': (),
'mode': 2},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
'speaker_ids': (0, 1, 2, 4, 5),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': (),
'mode': 2},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
'speaker_ids': (0, 1, 2, 4, 5),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': (),
'mode': 2},
{'coordinates_q15': ((16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
'speaker_ids': (2, 4, 5),
'axis0_groups': ((0,), (1, 2)),
'axis1_groups': (),
'mode': 2},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 32767, 0), (32767, 32767, 0)),
'speaker_ids': (4, 5),
'axis0_groups': ((0, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1})},
'71': {'out_ch_config': 11,
'speaker_bitfield': 31,
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Lrs', 'Rrs'),
'internal_to_standard': (0, 1, 2, 3, 6, 7, 4, 5),
'regions': ({'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 16384, 0),
(32767, 16384, 0),
(0, 32767, 0),
(32767, 32767, 0)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7),
'axis0_groups': ((0, 2, 1), (3, 4), (5, 6)),
'axis1_groups': (),
'mode': 2},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0), (0, 16384, 0), (32767, 16384, 0)),
'speaker_ids': (0, 1, 2, 4, 5),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': (),
'mode': 2},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
'speaker_ids': (0, 1, 2, 6, 7),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': (),
'mode': 2},
{'coordinates_q15': ((16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
'speaker_ids': (2, 6, 7),
'axis0_groups': ((0,), (1, 2)),
'axis1_groups': (),
'mode': 2},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1},
{'coordinates_q15': ((0, 16384, 0), (32767, 16384, 0), (0, 32767, 0), (32767, 32767, 0)),
'speaker_ids': (4, 5, 6, 7),
'axis0_groups': ((0, 1), (2, 3)),
'axis1_groups': (),
'mode': 2},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1})},
'512': {'out_ch_config': 13,
'speaker_bitfield': 1039,
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Ltm', 'Rtm'),
'internal_to_standard': (0, 1, 2, 3, 4, 5, 6, 7),
'regions': ({'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6),),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6),),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6),),
'mode': 3},
{'coordinates_q15': ((16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (2, 4, 5, 6, 7),
'axis0_groups': ((0,), (1, 2)),
'axis1_groups': ((3, 4),),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (0, 1, 2, 6, 7),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': ((3, 4),),
'mode': 3},
{'coordinates_q15': ((0, 32767, 0),
(32767, 32767, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (4, 5, 6, 7),
'axis0_groups': ((0, 1),),
'axis1_groups': ((2, 3),),
'mode': 3},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1})},
'514': {'out_ch_config': 14,
'speaker_bitfield': 2575,
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Ltf', 'Rtf', 'Ltr', 'Rtr'),
'internal_to_standard': (0, 1, 2, 3, 4, 5, 6, 7, 8, 9),
'regions': ({'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6), (7, 8)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6), (7, 8)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6), (7, 8)),
'mode': 3},
{'coordinates_q15': ((16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (2, 4, 5, 6, 7, 8, 9),
'axis0_groups': ((0,), (1, 2)),
'axis1_groups': ((3, 4), (5, 6)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 6, 7, 8, 9),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': ((3, 4), (5, 6)),
'mode': 3},
{'coordinates_q15': ((0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (4, 5, 6, 7, 8, 9),
'axis0_groups': ((0, 1),),
'axis1_groups': ((2, 3), (4, 5)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(7928, 7928, 32767),
(24840, 7928, 32767)),
'speaker_ids': (0, 1, 2, 6, 7),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': ((3, 4),),
'mode': 3})},
'712': {'out_ch_config': 15,
'speaker_bitfield': 1055,
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Lrs', 'Rrs', 'Ltm', 'Rtm'),
'internal_to_standard': (0, 1, 2, 3, 6, 7, 4, 5, 8, 9),
'regions': ({'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 16384, 0),
(32767, 16384, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9),
'axis0_groups': ((0, 2, 1), (3, 4), (5, 6)),
'axis1_groups': ((7, 8),),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 16384, 0),
(32767, 16384, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 8, 9),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6),),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (0, 1, 2, 6, 7, 8, 9),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6),),
'mode': 3},
{'coordinates_q15': ((16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (2, 6, 7, 8, 9),
'axis0_groups': ((0,), (1, 2)),
'axis1_groups': ((3, 4),),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (0, 1, 2, 8, 9),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': ((3, 4),),
'mode': 3},
{'coordinates_q15': ((0, 16384, 0),
(32767, 16384, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 16384, 32767),
(24840, 16384, 32767)),
'speaker_ids': (4, 5, 6, 7, 8, 9),
'axis0_groups': ((0, 1), (2, 3)),
'axis1_groups': ((4, 5),),
'mode': 3},
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
'speaker_ids': (0, 1, 2),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': (),
'mode': 1})},
'714': {'out_ch_config': 16,
'speaker_bitfield': 2591,
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Lrs', 'Rrs', 'Ltf', 'Rtf', 'Ltr', 'Rtr'),
'internal_to_standard': (0, 1, 2, 3, 6, 7, 4, 5, 8, 9, 10, 11),
'regions': ({'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 16384, 0),
(32767, 16384, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11),
'axis0_groups': ((0, 2, 1), (3, 4), (5, 6)),
'axis1_groups': ((7, 8), (9, 10)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 16384, 0),
(32767, 16384, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 8, 9, 10, 11),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6), (7, 8)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 6, 7, 8, 9, 10, 11),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6), (7, 8)),
'mode': 3},
{'coordinates_q15': ((16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (2, 6, 7, 8, 9, 10, 11),
'axis0_groups': ((0,), (1, 2)),
'axis1_groups': ((3, 4), (5, 6)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 8, 9, 10, 11),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': ((3, 4), (5, 6)),
'mode': 3},
{'coordinates_q15': ((0, 16384, 0),
(32767, 16384, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (4, 5, 6, 7, 8, 9, 10, 11),
'axis0_groups': ((0, 1), (2, 3)),
'axis1_groups': ((4, 5), (6, 7)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(7928, 7928, 32767),
(24840, 7928, 32767)),
'speaker_ids': (0, 1, 2, 8, 9),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': ((3, 4),),
'mode': 3})},
'914': {'out_ch_config': 19,
'speaker_bitfield': 2719,
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Lrs', 'Rrs', 'Lw', 'Rw', 'Ltf', 'Rtf', 'Ltr', 'Rtr'),
'internal_to_standard': (0, 1, 2, 3, 6, 7, 4, 5, 10, 11, 12, 13, 8, 9),
'regions': ({'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 16384, 0),
(32767, 16384, 0),
(0, 32767, 0),
(32767, 32767, 0),
(0, 5285, 0),
(32767, 5285, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13),
'axis0_groups': ((0, 2, 1), (7, 8), (3, 4), (5, 6)),
'axis1_groups': ((9, 10), (11, 12)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 16384, 0),
(32767, 16384, 0),
(0, 5285, 0),
(32767, 5285, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 8, 9, 10, 11, 12, 13),
'axis0_groups': ((0, 2, 1), (5, 6), (3, 4)),
'axis1_groups': ((7, 8), (9, 10)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 6, 7, 10, 11, 12, 13),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6), (7, 8)),
'mode': 3},
{'coordinates_q15': ((16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (2, 6, 7, 10, 11, 12, 13),
'axis0_groups': ((0,), (1, 2)),
'axis1_groups': ((3, 4), (5, 6)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 10, 11, 12, 13),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': ((3, 4), (5, 6)),
'mode': 3},
{'coordinates_q15': ((0, 16384, 0),
(32767, 16384, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (4, 5, 6, 7, 10, 11, 12, 13),
'axis0_groups': ((0, 1), (2, 3)),
'axis1_groups': ((4, 5), (6, 7)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 5285, 0),
(32767, 5285, 0),
(7928, 7928, 32767),
(24840, 7928, 32767)),
'speaker_ids': (0, 1, 2, 8, 9, 10, 11),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6),),
'mode': 3})},
'916': {'out_ch_config': 20,
'speaker_bitfield': 3743,
'channels': ('L',
'R',
'C',
'LFE',
'Ls',
'Rs',
'Lrs',
'Rrs',
'Lw',
'Rw',
'Ltf',
'Rtf',
'Ltm',
'Rtm',
'Ltr',
'Rtr'),
'internal_to_standard': (0, 1, 2, 3, 6, 7, 4, 5, 10, 11, 14, 15, 12, 13, 8, 9),
'regions': ({'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 16384, 0),
(32767, 16384, 0),
(0, 32767, 0),
(32767, 32767, 0),
(0, 5285, 0),
(32767, 5285, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 16384, 32767),
(24840, 16384, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15),
'axis0_groups': ((0, 2, 1), (7, 8), (3, 4), (5, 6)),
'axis1_groups': ((9, 10), (11, 12), (13, 14)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 16384, 0),
(32767, 16384, 0),
(0, 5285, 0),
(32767, 5285, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 16384, 32767),
(24840, 16384, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 4, 5, 8, 9, 10, 11, 12, 13, 14, 15),
'axis0_groups': ((0, 2, 1), (5, 6), (3, 4)),
'axis1_groups': ((7, 8), (9, 10), (11, 12)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 16384, 32767),
(24840, 16384, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 6, 7, 10, 11, 12, 13, 14, 15),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6), (7, 8), (9, 10)),
'mode': 3},
{'coordinates_q15': ((16384, 0, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 16384, 32767),
(24840, 16384, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (2, 6, 7, 10, 11, 12, 13, 14, 15),
'axis0_groups': ((0,), (1, 2)),
'axis1_groups': ((3, 4), (5, 6), (7, 8)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 16384, 32767),
(24840, 16384, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (0, 1, 2, 10, 11, 12, 13, 14, 15),
'axis0_groups': ((0, 2, 1),),
'axis1_groups': ((3, 4), (5, 6), (7, 8)),
'mode': 3},
{'coordinates_q15': ((0, 16384, 0),
(32767, 16384, 0),
(0, 32767, 0),
(32767, 32767, 0),
(7928, 7928, 32767),
(24840, 7928, 32767),
(7928, 16384, 32767),
(24840, 16384, 32767),
(7928, 24840, 32767),
(24840, 24840, 32767)),
'speaker_ids': (4, 5, 6, 7, 10, 11, 12, 13, 14, 15),
'axis0_groups': ((0, 1), (2, 3)),
'axis1_groups': ((4, 5), (6, 7), (8, 9)),
'mode': 3},
{'coordinates_q15': ((0, 0, 0),
(32767, 0, 0),
(16384, 0, 0),
(0, 5285, 0),
(32767, 5285, 0),
(7928, 7928, 32767),
(24840, 7928, 32767)),
'speaker_ids': (0, 1, 2, 8, 9, 10, 11),
'axis0_groups': ((0, 2, 1), (3, 4)),
'axis1_groups': ((5, 6),),
'mode': 3})}}
def _make_layout(name: str, item: dict) -> SpeakerLayout:
return SpeakerLayout(
name=name,
out_ch_config=item["out_ch_config"],
speaker_bitfield=item["speaker_bitfield"],
channels=item["channels"],
internal_to_standard=item["internal_to_standard"],
regions=tuple(RegionGeometry(**region) for region in item["regions"]),
)
SPEAKER_LAYOUTS = {name: _make_layout(name, item) for name, item in _RAW_LAYOUTS.items()}
SPEAKER_LAYOUT_ALIASES = {
"2.0": "20",
"3.1": "31",
"5.1": "51",
"7.1": "71",
"5.1.2": "512",
"5.1.4": "514",
"7.1.2": "712",
"7.1.4": "714",
"9.1.4": "914",
"9.1.6": "916",
}
SPEAKER_LAYOUT_CHOICES = tuple(SPEAKER_LAYOUT_ALIASES)
SPEAKER_LAYOUT_DISPLAY_NAMES = {value: key for key, value in SPEAKER_LAYOUT_ALIASES.items()}
def get_speaker_layout(name: str) -> SpeakerLayout:
key = SPEAKER_LAYOUT_ALIASES.get(name, name)
try:
return SPEAKER_LAYOUTS[key]
except KeyError as exc:
raise ValueError(f"unsupported speaker layout: {name}") from exc
def speaker_layout_display_name(layout: str | SpeakerLayout) -> str:
target = get_speaker_layout(layout) if isinstance(layout, str) else layout
return SPEAKER_LAYOUT_DISPLAY_NAMES[target.name]
+154
View File
@@ -0,0 +1,154 @@
"""ctypes bridge for the float64 native object-to-speaker renderer."""
from __future__ import annotations
import ctypes
from pathlib import Path
import numpy as np
from native_renderer import ABI_VERSION, find_native_library
from oamd_bits import JocFieldState, frame_update
from speaker_layouts import SpeakerLayout, get_speaker_layout
FRAME_SAMPLES = 1536
INPUT_CHANNELS = 16
OBJECTS = 15
COORDINATES = 3
class NativeSpeakerRenderer:
def __init__(self, layout: str | SpeakerLayout, library_path=None):
self.layout = get_speaker_layout(layout) if isinstance(layout, str) else layout
self.library_path = find_native_library(library_path)
self._lib = ctypes.CDLL(str(self.library_path))
self._bind()
version = int(self._lib.ejoc_abi_version())
if version != ABI_VERSION:
raise RuntimeError(f"native ABI mismatch: expected {ABI_VERSION}, got {version}")
channels = int(self._lib.ejoc_speaker_layout_channel_count(self.layout.speaker_bitfield))
if channels != self.layout.channel_count:
raise RuntimeError(
f"native layout channel count mismatch: expected {self.layout.channel_count}, got {channels}"
)
self._handle = self._lib.ejoc_speaker_renderer_create(self.layout.speaker_bitfield)
if not self._handle:
raise RuntimeError(f"native speaker renderer rejected mask 0x{self.layout.speaker_bitfield:X}")
self._state = JocFieldState()
def _bind(self):
void_p = ctypes.c_void_p
self._lib.ejoc_abi_version.argtypes = []
self._lib.ejoc_abi_version.restype = ctypes.c_uint32
self._lib.ejoc_speaker_layout_channel_count.argtypes = [ctypes.c_uint32]
self._lib.ejoc_speaker_layout_channel_count.restype = ctypes.c_uint32
self._lib.ejoc_speaker_renderer_create.argtypes = [ctypes.c_uint32]
self._lib.ejoc_speaker_renderer_create.restype = void_p
self._lib.ejoc_speaker_renderer_destroy.argtypes = [void_p]
self._lib.ejoc_speaker_renderer_destroy.restype = None
self._lib.ejoc_speaker_renderer_reset.argtypes = [void_p]
self._lib.ejoc_speaker_renderer_reset.restype = ctypes.c_int
self._lib.ejoc_speaker_renderer_last_error.argtypes = [void_p]
self._lib.ejoc_speaker_renderer_last_error.restype = ctypes.c_char_p
self._lib.ejoc_speaker_renderer_process.argtypes = [
void_p,
ctypes.POINTER(ctypes.c_float),
ctypes.c_uint32,
ctypes.c_uint32,
ctypes.POINTER(ctypes.c_uint32),
ctypes.POINTER(ctypes.c_uint32),
ctypes.POINTER(ctypes.c_uint16),
ctypes.POINTER(ctypes.c_uint8),
ctypes.POINTER(ctypes.c_uint8),
ctypes.POINTER(ctypes.c_double),
ctypes.POINTER(ctypes.c_double),
]
self._lib.ejoc_speaker_renderer_process.restype = ctypes.c_int
def _raise(self, operation, status):
message = self._lib.ejoc_speaker_renderer_last_error(self._handle)
detail = (message or b"").decode("utf-8", "replace")
raise RuntimeError(f"native speaker renderer {operation} failed ({status}): {detail}")
def reset(self):
if not self._handle:
raise RuntimeError("native speaker renderer is closed")
status = self._lib.ejoc_speaker_renderer_reset(self._handle)
if status:
self._raise("reset", status)
self._state = JocFieldState()
def close(self):
if self._handle:
self._lib.ejoc_speaker_renderer_destroy(self._handle)
self._handle = None
def __enter__(self):
return self
def __exit__(self, exc_type, exc, tb):
self.close()
def __del__(self):
try:
self.close()
except Exception:
pass
def _metadata_arrays(self, events):
events = list(events)
count = len(events)
if not count:
return count, None, None, None
offsets = np.empty(count, dtype=np.uint32)
ramps = np.empty(count, dtype=np.uint32)
positions = np.empty((count, OBJECTS, COORDINATES), dtype=np.uint16)
for event_index, (outer_offset, payload) in enumerate(events):
update = frame_update(payload)
self._state.apply(update["values"])
offsets[event_index] = int(outer_offset) + int(update["block_offset_samples"])
ramps[event_index] = int(update["ramp_duration_samples"])
for object_index in range(1, OBJECTS + 1):
positions[event_index, object_index - 1] = (
self._state.q[(object_index, "q1")],
self._state.q[(object_index, "q2")],
self._state.q[(object_index, "q3")],
)
return count, offsets, ramps, positions
def process(self, objects16, events=()):
if not self._handle:
raise RuntimeError("native speaker renderer is closed")
source = np.ascontiguousarray(objects16, dtype=np.float32)
if source.ndim != 2 or source.shape[1] != INPUT_CHANNELS:
raise ValueError(f"objects16 must have shape [samples,16], got {source.shape}")
if len(source) % 32:
raise ValueError("sample count must be a multiple of 32")
count, offsets, ramps, positions = self._metadata_arrays(events)
output = np.empty((len(source), self.layout.channel_count), dtype=np.float64)
null_u32 = ctypes.POINTER(ctypes.c_uint32)()
null_u16 = ctypes.POINTER(ctypes.c_uint16)()
null_u8 = ctypes.POINTER(ctypes.c_uint8)()
null_f64 = ctypes.POINTER(ctypes.c_double)()
status = self._lib.ejoc_speaker_renderer_process(
self._handle,
source.ctypes.data_as(ctypes.POINTER(ctypes.c_float)),
len(source),
count,
offsets.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)) if count else null_u32,
ramps.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)) if count else null_u32,
positions.ctypes.data_as(ctypes.POINTER(ctypes.c_uint16)) if count else null_u16,
null_u8,
null_u8,
null_f64,
output.ctypes.data_as(ctypes.POINTER(ctypes.c_double)),
)
if status:
self._raise("process", status)
return output
def render_frame(self, objects16, payload=None, metadata_offset=1473):
source = np.asarray(objects16)
if source.shape != (FRAME_SAMPLES, INPUT_CHANNELS):
raise ValueError(f"frame must have shape (1536,16), got {source.shape}")
events = () if payload is None else ((int(metadata_offset), bytes(payload)),)
return self.process(source, events)
+311
View File
@@ -0,0 +1,311 @@
"""High-precision object-to-speaker renderer.
All spatial calculations, gain ramps, and object accumulation use float64. Input
PCM may be float32, but precision is reduced only when the caller explicitly
writes a lower-precision output format.
"""
from __future__ import annotations
from collections import defaultdict
import math
from typing import Iterable
import numpy as np
from oamd_bits import JocFieldState, frame_update
from speaker_layouts import RegionGeometry, SpeakerLayout, get_speaker_layout
Q15_SCALE = 32768.0
Q15_MAX = 32767.0 / Q15_SCALE
GAIN_SNAP_THRESHOLD = 1.0e-4
def expand_speaker_bitfield(compact_mask: int, center_height: bool = False) -> int:
"""Expand the compact target-layout bitfield into individual speakers."""
expansions = (
0x00000003, 0x00000004, 0x00000008, 0x00000030,
0x000000C0, 0x00000100, 0x00000600, 0x00001800,
0x00006000, 0x00018000, 0x00060000, 0x00180000,
0x00600000, 0x01800000, 0x06000000, 0x18000000,
0x60000000, 0x080000000, 0x600000000, 0x800000000,
0x1000000000, 0x2000000000,
)
expanded = 0
for bit, value in enumerate(expansions):
if int(compact_mask) & (1 << bit):
expanded |= value
if center_height:
expanded |= 0x4000000000
return expanded
def layout_attenuation_db(compact_mask: int) -> float:
"""Compute the layout-dependent maximum positional compensation in dB."""
expanded = expand_speaker_bitfield(compact_mask)
height_channels = 2 * sum((expanded >> bit) & 1 for bit in (13, 15, 17, 19, 21))
floor_channels = ((expanded >> 8) & 1) + 2 * sum(
(expanded >> bit) & 1 for bit in (31, 4, 6, 11, 25, 27, 29, 33)
)
height_factor = min(height_channels / 4.0, 1.0)
floor_factor = min(floor_channels / 4.0, 1.0)
return -max(4.5 - 1.5 * height_factor - 3.0 * floor_factor, 0.0)
def floor_y_exponent(compact_mask: int, metadata_scaling_enabled: bool = True) -> int:
if not metadata_scaling_enabled:
return 0
low_mask = expand_speaker_bitfield(compact_mask) & 0xFFFFFFFF
return int(bool(low_mask & 0x130) and not bool(low_mask & 0x18C0))
def position_gain(compact_mask: int, v: float, w: float) -> float:
"""Return the high-precision position-dependent layout compensation."""
y_term = min(max(float(v) / 0.6, 0.0), 1.0)
z_term = min(max((float(w) - 0.2) / 0.8, 0.0), 1.0)
amount = min(max(y_term + z_term, 0.0), 1.0)
return math.pow(10.0, layout_attenuation_db(compact_mask) * amount / 20.0)
def equal_power_pair(position: float) -> tuple[float, float]:
angle = math.pi * 0.5 * float(position)
return math.cos(angle), math.sin(angle)
def _coordinates(region: RegionGeometry) -> np.ndarray:
return np.asarray(region.coordinates_q15, dtype=np.float64) / Q15_SCALE
def _normalized(value: float, lower: float, upper: float) -> float:
return (float(value) - float(lower)) / (float(upper) - float(lower))
def _axis0_gains(region: RegionGeometry, coordinates: np.ndarray, value: float) -> np.ndarray:
result = np.zeros(len(region.speaker_ids), dtype=np.float64)
position = float(value)
for group in region.axis0_groups:
if not group:
continue
first, last = group[0], group[-1]
if position <= coordinates[first, 0]:
result[first] = 1.0
continue
if position >= coordinates[last, 0]:
result[last] = 1.0
continue
for lower_index, upper_index in zip(group, group[1:]):
lower = coordinates[lower_index, 0]
upper = coordinates[upper_index, 0]
if position > lower and position <= upper:
result[lower_index], result[upper_index] = equal_power_pair(
_normalized(position, lower, upper)
)
break
return result
def _axis1_gains(region: RegionGeometry, coordinates: np.ndarray, groups,
value: float) -> np.ndarray:
result = np.zeros(len(region.speaker_ids), dtype=np.float64)
if not groups:
return result
position = float(value)
first_value = coordinates[groups[0][0], 1]
last_value = coordinates[groups[-1][0], 1]
if position <= first_value:
result[list(groups[0])] = 1.0
return result
if position > last_value:
result[list(groups[-1])] = 1.0
return result
for lower_group, upper_group in zip(groups, groups[1:]):
lower = coordinates[lower_group[0], 1]
upper = coordinates[upper_group[0], 1]
if position >= lower and position <= upper:
lower_gain, upper_gain = equal_power_pair(_normalized(position, lower, upper))
result[list(lower_group)] = lower_gain
result[list(upper_group)] = upper_gain
break
return result
def _plane_gains(region: RegionGeometry, coordinates: np.ndarray, groups,
u: float, v: float, mode: int) -> np.ndarray:
gains = _axis0_gains(
RegionGeometry(region.coordinates_q15, region.speaker_ids, tuple(groups), (), region.mode),
coordinates,
u,
)
if mode >= 2:
gains *= _axis1_gains(region, coordinates, groups, v)
return gains
def render_point_gains(layout: str | SpeakerLayout, u: float, v: float, w: float,
*, region_index: int = 0, enable_height: bool = True,
object_gain: float = 1.0, standard_order: bool = True) -> np.ndarray:
"""Render one point object to a target speaker layout using float64."""
target = get_speaker_layout(layout) if isinstance(layout, str) else layout
region = target.regions[int(region_index)]
coordinates = _coordinates(region)
floor_v = min(max(math.ldexp(float(v), floor_y_exponent(target.speaker_bitfield)), 0.0), 1.0)
floor = _plane_gains(region, coordinates, region.axis0_groups, u, floor_v, region.mode)
point_gains = floor
if region.mode == 3:
top = _plane_gains(region, coordinates, region.axis1_groups, u, float(v), 3)
height = min(max(float(w) if enable_height else 0.0, 0.0), Q15_MAX)
if height >= Q15_MAX:
point_gains = top
elif height > 0.0:
floor_weight, height_weight = equal_power_pair(height)
point_gains = floor * floor_weight + top * height_weight
internal = np.zeros(target.channel_count, dtype=np.float64)
gain = position_gain(target.speaker_bitfield, v, w) * float(object_gain)
for point_gain, speaker_id in zip(point_gains, region.speaker_ids):
internal[speaker_id] = point_gain * gain
if standard_order:
return internal[list(target.internal_to_standard)]
return internal
def align_metadata_sample(sample: int, block_size: int = 32) -> int:
return block_size * ((int(sample) + block_size // 2 - 1) // block_size)
def ramp_block_count(duration_samples: int, block_size: int = 32,
rate_scale: int = 1) -> int:
return (int(duration_samples) * int(rate_scale) + block_size // 2 - 1) // block_size
def _all_object_targets(layout: SpeakerLayout, state: JocFieldState) -> np.ndarray:
result = np.zeros((15, layout.channel_count), dtype=np.float64)
for object_index in range(1, 16):
result[object_index - 1] = render_point_gains(
layout,
state.q[(object_index, "q1")] / Q15_SCALE,
state.q[(object_index, "q2")] / Q15_SCALE,
state.q[(object_index, "q3")] / Q15_SCALE,
standard_order=False,
)
return result
class PythonSpeakerRenderer:
"""Stateful float64 speaker renderer with the same frame API as native."""
def __init__(self, layout: str | SpeakerLayout, block_size: int = 32):
self.layout = get_speaker_layout(layout) if isinstance(layout, str) else layout
self.block_size = int(block_size)
if self.block_size <= 0:
raise ValueError("block_size must be positive")
self.reset()
def reset(self):
channels = self.layout.channel_count
self._state = JocFieldState()
self._current = np.zeros((15, channels), dtype=np.float64)
self._target = np.zeros_like(self._current)
self._step = np.zeros_like(self._current)
self._remaining = np.zeros((15, channels), dtype=np.int32)
self._window = np.arange(self.block_size, dtype=np.float64) / float(self.block_size)
def _apply_payload(self, payload):
update = frame_update(payload)
self._state.apply(update["values"])
new_target = _all_object_targets(self.layout, self._state)
blocks = ramp_block_count(update["ramp_duration_samples"], self.block_size)
difference = new_target - self._current
significant = np.abs(difference) >= GAIN_SNAP_THRESHOLD
self._target[...] = new_target
self._current[~significant] = new_target[~significant]
self._step[~significant] = 0.0
self._remaining[~significant] = 0
if blocks:
self._step[significant] = difference[significant] / float(blocks)
self._remaining[significant] = blocks
else:
self._current[significant] = new_target[significant]
self._step[significant] = 0.0
self._remaining[significant] = 0
def process(self, objects16: np.ndarray, events=(), *, standard_order: bool = True,
output: np.ndarray | None = None) -> np.ndarray:
source = np.asarray(objects16)
if source.ndim != 2 or source.shape[1] != 16:
raise ValueError(f"objects16 must have shape [samples,16], got {source.shape}")
if len(source) % self.block_size:
raise ValueError(f"sample count must be divisible by {self.block_size}")
total_samples = len(source)
channels = self.layout.channel_count
if output is None:
output = np.zeros((total_samples, channels), dtype=np.float64)
elif output.shape != (total_samples, channels) or output.dtype != np.float64:
raise ValueError((output.shape, output.dtype))
else:
output[...] = 0.0
if "LFE" in self.layout.channels:
output[:, 3] = source[:, 0].astype(np.float64, copy=False)
events_by_block: dict[int, list[bytes]] = defaultdict(list)
for sample_offset, payload in events:
if not 0 <= int(sample_offset) <= total_samples:
raise ValueError(f"metadata offset outside process call: {sample_offset}")
block = align_metadata_sample(sample_offset, self.block_size) // self.block_size
events_by_block[block].append(bytes(payload))
total_blocks = total_samples // self.block_size
for block in range(total_blocks):
for payload in events_by_block.get(block, ()):
self._apply_payload(payload)
start = block * self.block_size
stop = start + self.block_size
out_block = output[start:stop]
for object_index in range(15):
active = self._remaining[object_index] > 0
gains = np.broadcast_to(self._target[object_index],
(self.block_size, channels)).copy()
if np.any(active):
gains[:, active] = (
self._current[object_index, active][None, :]
+ self._window[:, None] * self._step[object_index, active][None, :]
)
out_block += (
source[start:stop, object_index + 1, None].astype(np.float64, copy=False)
* gains
)
if np.any(active):
self._current[object_index, active] += self._step[object_index, active]
self._remaining[object_index, active] -= 1
finished = self._remaining[object_index] == 0
self._current[object_index, finished] = self._target[object_index, finished]
for payload in events_by_block.get(total_blocks, ()):
self._apply_payload(payload)
if standard_order:
return output[:, list(self.layout.internal_to_standard)]
return output
def render_frame(self, objects16, payload=None, metadata_offset=1473):
source = np.asarray(objects16)
if source.shape != (1536, 16):
raise ValueError(f"frame must have shape (1536,16), got {source.shape}")
events = () if payload is None else ((int(metadata_offset), bytes(payload)),)
return self.process(source, events)
def render_objects16(objects16: np.ndarray, payload_events: Iterable[tuple[int, bytes]],
layout: str | SpeakerLayout, *, block_size: int = 32,
standard_order: bool = True, output: np.ndarray | None = None) -> np.ndarray:
"""Render a complete interleaved LFE + 15 object stream."""
renderer = PythonSpeakerRenderer(layout, block_size=block_size)
return renderer.process(objects16, payload_events, standard_order=standard_order, output=output)
def payload_events_from_index(index, frame_count: int, *, frame_samples: int = 1536,
payload_id: int = 11, outer_offset_samples: int = 1473):
for frame, row in enumerate(index.rows[:frame_count]):
subpayloads = index.subpayloads(row)
if payload_id in subpayloads:
timing = frame_update(subpayloads[payload_id])
yield (
frame * frame_samples + outer_offset_samples + timing["block_offset_samples"],
subpayloads[payload_id],
)
+157
View File
@@ -0,0 +1,157 @@
"""Streaming spool and WAV writer for direct speaker-layout output."""
from __future__ import annotations
import struct
from pathlib import Path
import numpy as np
WAVE_FORMAT_PCM = 0x0001
WAVE_FORMAT_IEEE_FLOAT = 0x0003
WAVE_FORMAT_EXTENSIBLE = 0xFFFE
_PCM_GUID = bytes.fromhex("0100000000001000800000aa00389b71")
_FLOAT_GUID = bytes.fromhex("0300000000001000800000aa00389b71")
class SpeakerPcmSpool:
"""Temporary interleaved float32 store with float64 peak analysis."""
def __init__(self, path, sample_count, channel_count):
self.path = Path(path)
self.sample_count = int(sample_count)
self.channel_count = int(channel_count)
self.position = 0
self.peak = 0.0
self.clipped_values = 0
self.values = np.memmap(
self.path, dtype="<f4", mode="w+",
shape=(self.sample_count, self.channel_count),
)
def write_frame(self, pcm):
values = np.asarray(pcm, dtype=np.float64)
if values.ndim != 2 or values.shape[1] != self.channel_count:
raise ValueError(
f"speaker frame must have shape [samples,{self.channel_count}], got {values.shape}")
if self.position + len(values) > self.sample_count:
raise ValueError("speaker spool received more samples than allocated")
if not np.all(np.isfinite(values)):
raise ValueError("speaker renderer produced NaN or infinity")
absolute = np.abs(values)
if absolute.size:
self.peak = max(self.peak, float(np.max(absolute)))
self.clipped_values += int(np.count_nonzero(absolute > 1.0))
self.values[self.position:self.position + len(values)] = values.astype(np.float32)
self.position += len(values)
def finalize(self):
if self.position != self.sample_count:
raise ValueError(
f"speaker spool has {self.position} samples, expected {self.sample_count}")
self.values.flush()
return self
def close(self):
values = self.values
self.values = None
del values
def _fmt_chunk(channel_count, rate, sample_format):
if sample_format == "float32":
bits = 32
bytes_per_sample = 4
simple_tag = WAVE_FORMAT_IEEE_FLOAT
guid = _FLOAT_GUID
elif sample_format == "int24":
bits = 24
bytes_per_sample = 3
simple_tag = WAVE_FORMAT_PCM
guid = _PCM_GUID
else:
raise ValueError(f"unsupported speaker WAV format: {sample_format}")
block_align = channel_count * bytes_per_sample
byte_rate = rate * block_align
if channel_count <= 2:
body = struct.pack(
"<HHIIHH", simple_tag, channel_count, rate,
byte_rate, block_align, bits)
else:
body = (
struct.pack(
"<HHIIHHH", WAVE_FORMAT_EXTENSIBLE, channel_count, rate,
byte_rate, block_align, bits, 22)
+ struct.pack("<HI", bits, 0)
+ guid
)
return body, bits, bytes_per_sample, block_align
def _write_header(stream, channel_count, sample_count, rate, sample_format):
fmt, bits, bytes_per_sample, block_align = _fmt_chunk(
channel_count, rate, sample_format)
data_size = sample_count * block_align
riff_file_size = 12 + 8 + len(fmt) + 8 + data_size
use_rf64 = riff_file_size - 8 > 0xFFFFFFFF
if use_rf64:
# RF64 + ds64 + fmt + data.
file_size = 12 + 36 + 8 + len(fmt) + 8 + data_size
stream.write(b"RF64")
stream.write(struct.pack("<I", 0xFFFFFFFF))
stream.write(b"WAVE")
stream.write(b"ds64")
stream.write(struct.pack("<IQQQI", 28, file_size - 8, data_size, sample_count, 0))
else:
stream.write(b"RIFF")
stream.write(struct.pack("<I", riff_file_size - 8))
stream.write(b"WAVE")
stream.write(b"fmt ")
stream.write(struct.pack("<I", len(fmt)))
stream.write(fmt)
stream.write(b"data")
stream.write(struct.pack("<I", 0xFFFFFFFF if use_rf64 else data_size))
return {
"format": sample_format,
"bits_per_sample": bits,
"bytes_per_sample": bytes_per_sample,
"block_align": block_align,
"data_bytes": data_size,
"rf64": use_rf64,
}
def _pack_int24(values):
scaled = (np.clip(values, -1.0, 1.0) * np.float32(8388607.0)).astype(np.int32)
unsigned = scaled.reshape(-1).view(np.uint32)
packed = np.empty((unsigned.size, 3), dtype=np.uint8)
packed[:, 0] = unsigned & 0xFF
packed[:, 1] = (unsigned >> 8) & 0xFF
packed[:, 2] = (unsigned >> 16) & 0xFF
return packed.tobytes()
def write_speaker_wav(path, pcm, sample_format, *, rate=48000, chunk_samples=262144):
"""Write an interleaved float32 array/memmap as float32 or PCM24 WAV."""
target = Path(path)
values = np.asarray(pcm)
if values.ndim != 2:
raise ValueError(f"speaker PCM must be 2D, got {values.shape}")
sample_count, channel_count = values.shape
target.parent.mkdir(parents=True, exist_ok=True)
with target.open("wb") as stream:
info = _write_header(
stream, channel_count, sample_count, int(rate), sample_format)
for start in range(0, sample_count, int(chunk_samples)):
block = np.asarray(values[start:start + chunk_samples], dtype="<f4")
if sample_format == "float32":
stream.write(block.tobytes(order="C"))
else:
stream.write(_pack_int24(block))
info.update({
"path": str(target.resolve()),
"sample_rate": int(rate),
"sample_count": int(sample_count),
"channel_count": int(channel_count),
"file_bytes": target.stat().st_size,
})
return info
+65
View File
@@ -0,0 +1,65 @@
"""为未覆盖的 E-AC-3 JOC/EMDF 变体生成结构化错误和维修报告。"""
import hashlib
import json
from pathlib import Path
def bytes_descriptor(data):
"""返回足以识别载荷、但不会复制整段载荷的摘要。"""
raw = bytes(data)
return {
"bytes": len(raw),
"sha256": hashlib.sha256(raw).hexdigest(),
"prefix_hex": raw[:32].hex(),
"suffix_hex": raw[-16:].hex() if raw else "",
}
class UnsupportedVariantError(ValueError):
"""表示输入结构有效或疑似有效,但当前解析器没有覆盖该变体。"""
def __init__(self, stage, variant, message, *, frame=None, details=None):
self.stage = str(stage)
self.variant = str(variant)
self.message = str(message)
self.frame = frame
self.details = dict(details or {})
super().__init__(self.__str__())
def add_context(self, *, frame=None, details=None):
if self.frame is None and frame is not None:
self.frame = int(frame)
if details:
for key, value in details.items():
self.details.setdefault(key, value)
self.args = (self.__str__(),)
return self
def to_dict(self):
return {
"report_schema": 1,
"error": "unsupported_variant",
"stage": self.stage,
"variant": self.variant,
"frame": self.frame,
"message": self.message,
"details": self.details,
}
def __str__(self):
where = f" frame={self.frame}" if self.frame is not None else ""
detail = json.dumps(self.details, ensure_ascii=False, separators=(",", ":"))
return f"[{self.stage}/{self.variant}]{where} {self.message}; details={detail}"
def write_variant_report(path, error, *, input_path=None, output_path=None):
"""将变体错误写成可交给后续维修工具的 JSON。"""
report = error.to_dict()
if input_path is not None:
report["input"] = str(Path(input_path))
if output_path is not None:
report["requested_output"] = str(Path(output_path))
target = Path(path)
target.parent.mkdir(parents=True, exist_ok=True)
target.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
return target