Initial public release of JustOneCacophony
This commit is contained in:
@@ -0,0 +1,141 @@
|
||||
"""把 LFE、15 路对象 PCM 和对象轨迹组装为 ADM BWF。
|
||||
|
||||
固定输出契约:
|
||||
EAC3JOC 重放输出 = 16ch(ch0 = LFE + ch1-15 = 15 对象);
|
||||
最终 ADM BWF = 7.1.2 bed(L R C Ls Rs Lb Rb + LFE + Ltf Rtf = 10ch)
|
||||
—— 除 LFE 外全部静音;
|
||||
15 对象 = ch1-15 直接填充对象轨;轨迹 = OAMD(q1/q2/q3 → xyz)。
|
||||
"""
|
||||
import os
|
||||
|
||||
import numpy as np
|
||||
|
||||
import adm_atmos
|
||||
|
||||
|
||||
def assemble_from_raw(raw16_path, out_path, scale=1.0, kf_tracks=None,
|
||||
duration_sec=None, rate=48000):
|
||||
"""16ch f32 交织 raw → 25ch ADM BWF(空 7.1.2 bed + LFE + 15 对象)。
|
||||
|
||||
raw16: (n, 16) 交织(ch0 = LFE,ch1-15 = 对象)。
|
||||
scale: 1.0 = 默认 0 dB,不附加输出缩放。该参数与 joc_clipgain 无关;
|
||||
主命令行已在渲染阶段应用用户增益,因此这里传 1.0。
|
||||
kf_tracks: 可选轨迹关键帧(OAMD 输出,格式 [(obj_id, [(t, x, y, z), ...]), ...]);
|
||||
缺省 = 静止参考位置(adm_atmos 默认)。
|
||||
"""
|
||||
raw = np.memmap(raw16_path, dtype=np.float32, mode="r")
|
||||
n = len(raw) // 16
|
||||
raw = raw[:n * 16].reshape(-1, 16)
|
||||
if duration_sec is None:
|
||||
duration_sec = n / rate
|
||||
# 惰性视图:adm_atmos 按块读取,避免全片 25ch 在内存中展开。
|
||||
class BedView:
|
||||
shape = (n, 10)
|
||||
|
||||
def __getitem__(self, key):
|
||||
src = np.asarray(raw[key], dtype=np.float32)
|
||||
one = src.ndim == 1
|
||||
if one:
|
||||
src = src[None, :]
|
||||
out = np.zeros((len(src), 10), dtype=np.float32)
|
||||
out[:, 3] = np.multiply(src[:, 0], np.float32(scale), dtype=np.float32)
|
||||
return out[0] if one else out
|
||||
|
||||
class ObjView:
|
||||
shape = (n, 16)
|
||||
|
||||
def __getitem__(self, key):
|
||||
return np.multiply(np.asarray(raw[key], dtype=np.float32),
|
||||
np.float32(scale), dtype=np.float32)
|
||||
if kf_tracks is None:
|
||||
kf_tracks = []
|
||||
for oi in range(15):
|
||||
# 静止参考位置(q1=q2=q3=0 → 原点;实际坐标按 OAMD 输出填入)
|
||||
kf_tracks.append(("JOC_Object_%d" % (oi + 1),
|
||||
[(0.0, 0.0, 0.0, 0.0, max(duration_sec, 1e-6))]))
|
||||
adm_atmos.build_master(out_path, BedView(), ObjView(), kf_tracks,
|
||||
duration_sec, rate=rate)
|
||||
# 及时释放 Windows 文件句柄,允许 TemporaryDirectory 删除中间 raw。
|
||||
raw._mmap.close()
|
||||
return out_path
|
||||
|
||||
|
||||
class StreamingMaster:
|
||||
"""Incrementally write renderer frames into the final 25-channel ADM BWF.
|
||||
|
||||
This removes the default 16-channel float32 intermediate file. The mapping
|
||||
remains identical to :func:`assemble_from_raw`: bed channel 3 receives LFE,
|
||||
bed channels 0..2/4..9 are silent, and output objects 1..15 map to ADM
|
||||
channels 10..24.
|
||||
"""
|
||||
|
||||
def __init__(self, out_path, duration_sec, rate=48000, block_samples=131072):
|
||||
if block_samples < 1536:
|
||||
raise ValueError("block_samples must be at least one E-AC-3 frame")
|
||||
self.out_path = os.fspath(out_path)
|
||||
self.duration_sec = float(duration_sec)
|
||||
self.rate = int(rate)
|
||||
self._sink = adm_atmos.Sink25(self.out_path, 25, self.rate)
|
||||
self._buffer = np.empty((int(block_samples), 25), dtype=np.float32)
|
||||
self._used = 0
|
||||
self._finalized = False
|
||||
|
||||
def _flush(self):
|
||||
if self._used:
|
||||
self._sink.write_block(self._buffer[:self._used])
|
||||
self._used = 0
|
||||
|
||||
def write_frame(self, pcm16):
|
||||
pcm = np.asarray(pcm16, dtype=np.float32)
|
||||
if pcm.shape != (16, 1536):
|
||||
raise ValueError(f"renderer frame must be (16,1536), got {pcm.shape}")
|
||||
source = 0
|
||||
while source < 1536:
|
||||
available = len(self._buffer) - self._used
|
||||
count = min(available, 1536 - source)
|
||||
target = self._buffer[self._used:self._used + count]
|
||||
target.fill(0.0)
|
||||
target[:, 3] = pcm[0, source:source + count]
|
||||
target[:, 10:25] = pcm[1:16, source:source + count].T
|
||||
self._used += count
|
||||
source += count
|
||||
if self._used == len(self._buffer):
|
||||
self._flush()
|
||||
|
||||
def finalize(self, kf_tracks):
|
||||
if self._finalized:
|
||||
raise RuntimeError("StreamingMaster already finalized")
|
||||
self._flush()
|
||||
try:
|
||||
from . import adm_serializer
|
||||
except ImportError:
|
||||
import adm_serializer
|
||||
axml = adm_serializer.build_axml(kf_tracks, self.duration_sec)
|
||||
chna = adm_atmos.build_chna()
|
||||
dbmd = adm_atmos.build_dbmd(25)
|
||||
trajectory_blocks = sum(len(track[1]) for track in kf_tracks)
|
||||
self.metadata_info = {
|
||||
"axml_bytes": len(axml),
|
||||
"trajectory_blocks": trajectory_blocks,
|
||||
"chna_bytes": len(chna),
|
||||
"dbmd_bytes": len(dbmd),
|
||||
}
|
||||
self._sink.finalize(axml, chna, dbmd)
|
||||
self._finalized = True
|
||||
print(f"master25 -> {self.out_path} ({self.duration_sec:.2f}s, 25ch, "
|
||||
f"axml={len(axml)}B, chna={len(chna)}B, dbmd={len(dbmd)}B)")
|
||||
return self.out_path
|
||||
|
||||
def abort(self):
|
||||
if self._finalized:
|
||||
return
|
||||
sink = getattr(self, "_sink", None)
|
||||
fp = getattr(sink, "fp", None)
|
||||
if fp is not None and not fp.closed:
|
||||
fp.close()
|
||||
|
||||
def __del__(self):
|
||||
try:
|
||||
self.abort()
|
||||
except Exception:
|
||||
pass
|
||||
@@ -0,0 +1,273 @@
|
||||
"""生成 25 声道 RF64 ADM BWF 及其 axml、chna、dbmd 元数据。
|
||||
|
||||
输出由 10 声道 7.1.2 bed 和 15 路对象组成;RF64 尺寸字段在写入完成后回填。
|
||||
"""
|
||||
import struct
|
||||
import numpy as np
|
||||
import xml.etree.ElementTree as ET
|
||||
|
||||
NS = "urn:ebu:metadata-schema:ebuCore_2016"
|
||||
XSI = "http://www.w3.org/2001/XMLSchema-instance"
|
||||
|
||||
BED_NAMES = ["RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter",
|
||||
"RoomCentricLFE", "RoomCentricLeftSideSurround",
|
||||
"RoomCentricRightSideSurround", "RoomCentricLeftRearSurround",
|
||||
"RoomCentricRightRearSurround", "RoomCentricLeftTopSurround",
|
||||
"RoomCentricRightTopSurround"]
|
||||
BED_LABELS = ["RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss", "RC_Rss",
|
||||
"RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"]
|
||||
BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0),
|
||||
(-1.0, 1.0, -1.0), (-1.0, 0.0, 0.0), (1.0, 0.0, 0.0),
|
||||
(-1.0, -1.0, 0.0), (1.0, -1.0, 0.0), (-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
|
||||
|
||||
N_OBJ = 15
|
||||
|
||||
def q_to_adm_xyz(q1, q2, q3):
|
||||
posX = min(1.0, round(q1 * 62 / 32767.0) / 62.0)
|
||||
posY = min(1.0, round(q2 * 62 / 32767.0) / 62.0)
|
||||
posZ = round(q3 * 15 / 32767.0) / 15.0
|
||||
posZ = max(-1.0, min(1.0, posZ))
|
||||
return posX * 2 - 1, 1 - posY * 2, posZ
|
||||
|
||||
def ts(seconds):
|
||||
s = int(seconds)
|
||||
frac = int(round((seconds - s) * 100000))
|
||||
if frac >= 100000:
|
||||
s += 1; frac = 0
|
||||
return f"{s // 3600:02d}:{s % 3600 // 60:02d}:{s % 60:02d}.{frac:05d}"
|
||||
|
||||
def sub(parent, tag, attrib=None, text=None):
|
||||
e = ET.SubElement(parent, tag)
|
||||
if attrib:
|
||||
for k, v in attrib.items():
|
||||
e.set(k, v)
|
||||
if text is not None:
|
||||
e.text = text
|
||||
return e
|
||||
|
||||
def add_refs(parent, tag, ids):
|
||||
for i in ids:
|
||||
sub(parent, tag, text=i)
|
||||
|
||||
def obj_block(cf, bid, t, x, y, z, dur, interpolation=0.0):
|
||||
b = sub(cf, "audioBlockFormat", {
|
||||
"audioBlockFormatID": bid, "rtime": ts(t), "duration": ts(dur)})
|
||||
sub(b, "cartesian", text="1")
|
||||
for c, v in (("X", x), ("Y", y), ("Z", z)):
|
||||
if c == "Z" and v == 0:
|
||||
continue
|
||||
p = sub(b, "position", {"coordinate": c})
|
||||
p.text = f"{v:.10f}"
|
||||
sub(b, "jumpPosition", {"interpolationLength": f"{interpolation:.5f}"}, text="1")
|
||||
|
||||
def build_axml(obj_tracks, duration_sec):
|
||||
adm = ET.Element("ebuCoreMain", {
|
||||
"xmlns": NS, "xmlns:xsi": XSI,
|
||||
"xsi:schemaLocation": f"{NS} ebucore.xsd", "lang": "en"})
|
||||
core = sub(adm, "coreMetadata")
|
||||
fmt = sub(core, "format")
|
||||
af = sub(fmt, "audioFormatExtended")
|
||||
|
||||
prog = sub(af, "audioProgramme", {
|
||||
"audioProgrammeID": "APR_1001", "audioProgrammeName": "EAC3JOC_Export",
|
||||
"start": ts(0), "end": ts(duration_sec)})
|
||||
add_refs(prog, "audioContentIDRef", ("ACO_1001", "ACO_1002"))
|
||||
bc = sub(af, "audioContent", {"audioContentID": "ACO_1001",
|
||||
"audioContentName": "EAC3JOC_Master_Content"})
|
||||
add_refs(bc, "audioObjectIDRef", ["AO_1001"])
|
||||
sub(bc, "dialogue", {"mixedContentKind": "0"})
|
||||
oc = sub(af, "audioContent", {"audioContentID": "ACO_1002",
|
||||
"audioContentName": "Objects"})
|
||||
add_refs(oc, "audioObjectIDRef", ["AO_%04x" % (0x100b + i) for i in range(N_OBJ)])
|
||||
sub(oc, "dialogue", {"mixedContentKind": "0"})
|
||||
|
||||
bed_o = sub(af, "audioObject", {"audioObjectID": "AO_1001", "audioObjectName": "Bed",
|
||||
"start": ts(0), "duration": ts(duration_sec)})
|
||||
sub(bed_o, "audioPackFormatIDRef", text="AP_00011001")
|
||||
add_refs(bed_o, "audioTrackUIDRef", ["ATU_%08x" % (i + 1) for i in range(10)])
|
||||
for i in range(N_OBJ):
|
||||
o = sub(af, "audioObject", {"audioObjectID": "AO_%04x" % (0x100b + i),
|
||||
"audioObjectName": f"Audio Object {i+1}",
|
||||
"start": ts(0), "duration": ts(duration_sec)})
|
||||
sub(o, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
|
||||
add_refs(o, "audioTrackUIDRef", ["ATU_%08x" % (i + 11)])
|
||||
|
||||
bp = sub(af, "audioPackFormat", {"audioPackFormatID": "AP_00011001",
|
||||
"audioPackFormatName": "EAC3JOCBedPack",
|
||||
"typeDefinition": "DirectSpeakers", "typeLabel": "0001"})
|
||||
add_refs(bp, "audioChannelFormatIDRef", ["AC_0001%04x" % (0x1001 + i) for i in range(10)])
|
||||
for i in range(N_OBJ):
|
||||
pk = sub(af, "audioPackFormat", {"audioPackFormatID": "AP_0003%04x" % (0x1001 + i),
|
||||
"audioPackFormatName": f"JOC_Object_{i+1}",
|
||||
"typeDefinition": "Objects", "typeLabel": "0003"})
|
||||
add_refs(pk, "audioChannelFormatIDRef", ["AC_0003%04x" % (0x1001 + i)])
|
||||
|
||||
for i in range(10):
|
||||
cf = sub(af, "audioChannelFormat", {"audioChannelFormatID": "AC_0001%04x" % (0x1001 + i),
|
||||
"audioChannelFormatName": BED_NAMES[i],
|
||||
"typeDefinition": "DirectSpeakers", "typeLabel": "0001"})
|
||||
b = sub(cf, "audioBlockFormat", {"audioBlockFormatID": "AB_0001%04x_00000001" % (0x1001 + i)})
|
||||
sub(b, "cartesian", text="1")
|
||||
x, y, z = BED_POS[i]
|
||||
for c, v in (("X", x), ("Y", y), ("Z", z)):
|
||||
if c == "Z" and v == 0:
|
||||
continue
|
||||
p = sub(b, "position", {"coordinate": c})
|
||||
p.text = f"{v:.10f}"
|
||||
sub(b, "speakerLabel", text=BED_LABELS[i])
|
||||
|
||||
for i, (oname, kfs) in enumerate(obj_tracks):
|
||||
cf = sub(af, "audioChannelFormat", {"audioChannelFormatID": "AC_0003%04x" % (0x1001 + i),
|
||||
"audioChannelFormatName": oname,
|
||||
"typeDefinition": "Objects", "typeLabel": "0003"})
|
||||
for k, keyframe in enumerate(kfs):
|
||||
t, x, y, z, dur = keyframe[:5]
|
||||
interpolation = keyframe[5] if len(keyframe) > 5 else 0.0
|
||||
obj_block(cf, "AB_0003%04x_%08x" % (0x1001 + i, k + 1),
|
||||
t, x, y, z, dur, interpolation)
|
||||
|
||||
for i in range(10):
|
||||
t = sub(af, "audioTrackUID", {"UID": "ATU_%08x" % (i + 1),
|
||||
"bitDepth": "24", "sampleRate": "48000"})
|
||||
sub(t, "audioTrackFormatIDRef", text="AT_0001%04x_01" % (0x1001 + i))
|
||||
sub(t, "audioPackFormatIDRef", text="AP_00011001")
|
||||
for i in range(N_OBJ):
|
||||
t = sub(af, "audioTrackUID", {"UID": "ATU_%08x" % (i + 11),
|
||||
"bitDepth": "24", "sampleRate": "48000"})
|
||||
sub(t, "audioTrackFormatIDRef", text="AT_0003%04x_01" % (0x1001 + i))
|
||||
sub(t, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
|
||||
|
||||
for i in range(10):
|
||||
tf = sub(af, "audioTrackFormat", {"audioTrackFormatID": "AT_0001%04x_01" % (0x1001 + i),
|
||||
"audioTrackFormatName": "PCM_" + BED_NAMES[i],
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(tf, "audioStreamFormatIDRef", text="AS_0001%04x" % (0x1001 + i))
|
||||
for i in range(N_OBJ):
|
||||
tf = sub(af, "audioTrackFormat", {"audioTrackFormatID": "AT_0003%04x_01" % (0x1001 + i),
|
||||
"audioTrackFormatName": "PCM_JOC_Object_%d" % (i + 1),
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(tf, "audioStreamFormatIDRef", text="AS_0003%04x" % (0x1001 + i))
|
||||
|
||||
for i in range(10):
|
||||
sf = sub(af, "audioStreamFormat", {"audioStreamFormatID": "AS_0001%04x" % (0x1001 + i),
|
||||
"audioStreamFormatName": "PCM_" + BED_NAMES[i],
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(sf, "audioChannelFormatIDRef", text="AC_0001%04x" % (0x1001 + i))
|
||||
sub(sf, "audioPackFormatIDRef", text="AP_00011001")
|
||||
sub(sf, "audioTrackFormatIDRef", text="AT_0001%04x_01" % (0x1001 + i))
|
||||
for i in range(N_OBJ):
|
||||
sf = sub(af, "audioStreamFormat", {"audioStreamFormatID": "AS_0003%04x" % (0x1001 + i),
|
||||
"audioStreamFormatName": "PCM_JOC_Object_%d" % (i + 1),
|
||||
"formatDefinition": "PCM", "formatLabel": "0001"})
|
||||
sub(sf, "audioChannelFormatIDRef", text="AC_0003%04x" % (0x1001 + i))
|
||||
sub(sf, "audioPackFormatIDRef", text="AP_0003%04x" % (0x1001 + i))
|
||||
sub(sf, "audioTrackFormatIDRef", text="AT_0003%04x_01" % (0x1001 + i))
|
||||
|
||||
return ET.tostring(adm, encoding="utf-8", xml_declaration=True)
|
||||
|
||||
def build_chna():
|
||||
out = bytearray()
|
||||
out += struct.pack("<HH", 25, 25)
|
||||
for i in range(10):
|
||||
out += struct.pack("<H", i + 1)
|
||||
out += ("ATU_%08x" % (i + 1)).encode()
|
||||
out += ("AT_0001%04x_01" % (0x1001 + i)).encode()
|
||||
out += b"AP_00011001" + b"\x00"
|
||||
for i in range(N_OBJ):
|
||||
out += struct.pack("<H", i + 11)
|
||||
out += ("ATU_%08x" % (i + 11)).encode()
|
||||
out += ("AT_0003%04x_01" % (0x1001 + i)).encode()
|
||||
out += ("AP_0003%04x" % (0x1001 + i)).encode() + b"\x00"
|
||||
return bytes(out)
|
||||
|
||||
def _checksum(seg):
|
||||
s = len(seg)
|
||||
for b in seg:
|
||||
s += b
|
||||
return (~s + 1) & 0xFF
|
||||
|
||||
def build_dbmd(object_count=25):
|
||||
out = bytearray(struct.pack("<I", 0x01000006))
|
||||
dd = bytearray(96)
|
||||
dd[1] = 0x47
|
||||
dd[5] = 0x60
|
||||
dd[8] = 0x24; dd[9] = 0x24
|
||||
out.append(7); out += struct.pack("<H", 96); out += bytes(dd)
|
||||
out.append(_checksum(dd))
|
||||
at = bytearray(248)
|
||||
c0 = b"Created with EAC3JOC"; c1 = b"EAC3JOC Python Renderer"
|
||||
at[0:len(c0)] = c0
|
||||
at[32:32 + len(c1)] = c1
|
||||
at[96], at[97], at[98] = 2, 1, 0
|
||||
at[103] = 0x03
|
||||
at[106] = 0x01
|
||||
at[111] = 0x22; at[112] = 0xFF
|
||||
out.append(9); out += struct.pack("<H", 248); out += bytes(at)
|
||||
out.append(_checksum(at))
|
||||
ob = bytearray(5 + 262 + object_count)
|
||||
ob[0:4] = struct.pack("<I", 0xF8726FBD)
|
||||
ob[4] = object_count
|
||||
for i in range(5 + 262, len(ob)):
|
||||
ob[i] = 0x84
|
||||
out.append(10); out += struct.pack("<H", len(ob)); out += bytes(ob)
|
||||
out.append(_checksum(ob))
|
||||
out += b"\x00\x00"
|
||||
return bytes(out)
|
||||
|
||||
class Sink25:
|
||||
def __init__(self, path, channels, rate):
|
||||
self.ch = channels; self.rate = rate; self.frames = 0
|
||||
self.fp = open(path, "wb+")
|
||||
self.fp.write(b"RF64" + struct.pack("<I", 0xFFFFFFFF) + b"WAVE")
|
||||
self._chunk(b"ds64", b"\x00" * 64)
|
||||
self._chunk(b"fmt ", self._fmt())
|
||||
self._chunk(b"data", b"")
|
||||
def _chunk(self, cid, body):
|
||||
self.fp.write(cid + struct.pack("<I", len(body)) + body)
|
||||
if len(body) & 1:
|
||||
self.fp.write(b"\x00")
|
||||
def _fmt(self):
|
||||
return struct.pack("<HHIIHH", 1, self.ch, self.rate,
|
||||
self.rate * self.ch * 3, self.ch * 3, 24)
|
||||
def write_block(self, arr):
|
||||
arr = arr.reshape(-1, self.ch)
|
||||
i24 = (np.clip(arr, -1.0, 1.0) * 8388607.0).astype(np.int32)
|
||||
self.fp.write(i24.view(np.uint8).reshape(-1, 4)[:, :3].tobytes())
|
||||
self.frames += arr.shape[0]
|
||||
def finalize(self, axml_bytes, chna_bytes, dbmd_bytes):
|
||||
data_len = self.frames * self.ch * 3
|
||||
self._chunk(b"axml", axml_bytes)
|
||||
self._chunk(b"chna", chna_bytes)
|
||||
self._chunk(b"dbmd", dbmd_bytes)
|
||||
self.fp.seek(0, 2); total = self.fp.tell()
|
||||
self.fp.seek(0); head = self.fp.read()
|
||||
m = head.find(b"data")
|
||||
if m >= 0:
|
||||
self.fp.seek(m + 4); self.fp.write(struct.pack("<I", data_len))
|
||||
m = head.find(b"ds64")
|
||||
if m >= 0:
|
||||
self.fp.seek(m + 8)
|
||||
self.fp.write(struct.pack("<QQQI", total - 8, data_len, self.frames, 0))
|
||||
self.fp.flush()
|
||||
self.fp.close()
|
||||
|
||||
def build_master(out_path, bed_mm, obj_mm, kf_tracks, duration_sec, rate=48000,
|
||||
block=480000):
|
||||
n = min(bed_mm.shape[0], obj_mm.shape[0])
|
||||
try:
|
||||
from . import adm_serializer
|
||||
except ImportError:
|
||||
import adm_serializer
|
||||
serial_axml = adm_serializer.build_axml
|
||||
axml = serial_axml(kf_tracks, duration_sec)
|
||||
chna = build_chna()
|
||||
dbmd = build_dbmd(25)
|
||||
sink = Sink25(out_path, 25, rate)
|
||||
for st in range(0, n, block):
|
||||
en = min(n, st + block)
|
||||
blk = np.hstack((np.asarray(bed_mm[st:en], dtype=np.float32),
|
||||
np.asarray(obj_mm[st:en, 1:16], dtype=np.float32)))
|
||||
sink.write_block(blk)
|
||||
sink.finalize(axml, chna, dbmd)
|
||||
print(f"master25 -> {out_path} ({duration_sec:.2f}s, 25ch, axml={len(axml)}B, "
|
||||
f"chna={len(chna)}B, dbmd={len(dbmd)}B)")
|
||||
@@ -0,0 +1,136 @@
|
||||
"""把 7.1.2 bed、15 个对象及其位置轨迹序列化为 ADM axml。
|
||||
|
||||
序列化结果采用固定元素顺序、属性顺序和十六进制 ADM 标识符,便于稳定输出和校验。
|
||||
"""
|
||||
BED_NAMES = ["RoomCentricLeft", "RoomCentricRight", "RoomCentricCenter", "RoomCentricLFE",
|
||||
"RoomCentricLeftSideSurround", "RoomCentricRightSideSurround",
|
||||
"RoomCentricLeftRearSurround", "RoomCentricRightRearSurround",
|
||||
"RoomCentricLeftTopSurround", "RoomCentricRightTopSurround"]
|
||||
BED_LABELS = ["RC_L", "RC_R", "RC_C", "RC_LFE", "RC_Lss", "RC_Rss",
|
||||
"RC_Lrs", "RC_Rrs", "RC_Lts", "RC_Rts"]
|
||||
BED_POS = [(-1.0, 1.0, 0.0), (1.0, 1.0, 0.0), (0.0, 1.0, 0.0), (-1.0, 1.0, -1.0),
|
||||
(-1.0, 0.0, 0.0), (1.0, 0.0, 0.0), (-1.0, -1.0, 0.0), (1.0, -1.0, 0.0),
|
||||
(-1.0, 0.0, 1.0), (1.0, 0.0, 1.0)]
|
||||
N_OBJ = 15
|
||||
|
||||
def ts(seconds):
|
||||
s = int(seconds)
|
||||
frac = int(round((seconds - s) * 100000))
|
||||
if frac >= 100000:
|
||||
s += 1; frac = 0
|
||||
return f"{s // 3600:02d}:{s % 3600 // 60:02d}:{s % 60:02d}.{frac:05d}"
|
||||
|
||||
def esc(v):
|
||||
return (str(v).replace("&", "&").replace("<", "<").replace(">", ">"))
|
||||
|
||||
def build_axml(obj_tracks, duration_sec):
|
||||
"""obj_tracks: [(name, [(rtime, x, y, z, dur), ...]) ×15]"""
|
||||
w = []
|
||||
a = w.append
|
||||
a('<?xml version="1.0" encoding="utf-8"?>')
|
||||
a('<ebuCoreMain xsi:schemaLocation="urn:ebu:metadata-schema:ebuCore_2016 ebucore.xsd" '
|
||||
'lang="en" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" '
|
||||
'xmlns="urn:ebu:metadata-schema:ebuCore_2016">')
|
||||
a('<coreMetadata><format><audioFormatExtended>')
|
||||
a(f'<audioProgramme audioProgrammeID="APR_1001" audioProgrammeName="EAC3JOC_Export" '
|
||||
f'start="{ts(0)}" end="{ts(duration_sec)}">')
|
||||
a('<audioContentIDRef>ACO_1001</audioContentIDRef>')
|
||||
a('<audioContentIDRef>ACO_1002</audioContentIDRef>')
|
||||
a('</audioProgramme>')
|
||||
a('<audioContent audioContentID="ACO_1001" audioContentName="EAC3JOC_Master_Content">')
|
||||
a('<audioObjectIDRef>AO_1001</audioObjectIDRef>')
|
||||
a('<dialogue mixedContentKind="0">2</dialogue>')
|
||||
a('</audioContent>')
|
||||
a('<audioContent audioContentID="ACO_1002" audioContentName="Objects">')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioObjectIDRef>AO_{0x100b + i:04x}</audioObjectIDRef>')
|
||||
a('<dialogue mixedContentKind="0">2</dialogue>')
|
||||
a('</audioContent>')
|
||||
a(f'<audioObject audioObjectID="AO_1001" audioObjectName="Bed" '
|
||||
f'start="{ts(0)}" duration="{ts(duration_sec)}">')
|
||||
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
|
||||
for i in range(10):
|
||||
a(f'<audioTrackUIDRef>ATU_{i + 1:08x}</audioTrackUIDRef>')
|
||||
a('</audioObject>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioObject audioObjectID="AO_{0x100b + i:04x}" audioObjectName="Audio Object {i+1}" '
|
||||
f'start="{ts(0)}" duration="{ts(duration_sec)}">')
|
||||
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
|
||||
a(f'<audioTrackUIDRef>ATU_{11 + i:08x}</audioTrackUIDRef>')
|
||||
a('</audioObject>')
|
||||
a('<audioPackFormat audioPackFormatID="AP_00011001" audioPackFormatName="EAC3JOCBedPack" '
|
||||
'typeDefinition="DirectSpeakers" typeLabel="0001">')
|
||||
for i in range(10):
|
||||
a(f'<audioChannelFormatIDRef>AC_0001{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a('</audioPackFormat>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioPackFormat audioPackFormatID="AP_0003{0x1001 + i:04x}" '
|
||||
f'audioPackFormatName="JOC_Object_{i+1}" typeDefinition="Objects" typeLabel="0003">')
|
||||
a(f'<audioChannelFormatIDRef>AC_0003{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a('</audioPackFormat>')
|
||||
for i in range(10):
|
||||
a(f'<audioChannelFormat audioChannelFormatID="AC_0001{0x1001 + i:04x}" '
|
||||
f'audioChannelFormatName="{BED_NAMES[i]}" typeDefinition="DirectSpeakers" typeLabel="0001">')
|
||||
a(f'<audioBlockFormat audioBlockFormatID="AB_0001{0x1001 + i:04x}_00000001">')
|
||||
a('<cartesian>1</cartesian>')
|
||||
x, y, z = BED_POS[i]
|
||||
a(f'<position coordinate="X">{x:.10f}</position>')
|
||||
a(f'<position coordinate="Y">{y:.10f}</position>')
|
||||
if z != 0:
|
||||
a(f'<position coordinate="Z">{z:.10f}</position>')
|
||||
a(f'<speakerLabel>{BED_LABELS[i]}</speakerLabel>')
|
||||
a('</audioBlockFormat>')
|
||||
a('</audioChannelFormat>')
|
||||
for i, (oname, kfs) in enumerate(obj_tracks):
|
||||
a(f'<audioChannelFormat audioChannelFormatID="AC_0003{0x1001 + i:04x}" '
|
||||
f'audioChannelFormatName="{oname}" typeDefinition="Objects" typeLabel="0003">')
|
||||
for k, keyframe in enumerate(kfs):
|
||||
t, x, y, z, dur = keyframe[:5]
|
||||
interpolation = keyframe[5] if len(keyframe) > 5 else 0.0
|
||||
a(f'<audioBlockFormat audioBlockFormatID="AB_0003{0x1001 + i:04x}_{k + 1:08x}" '
|
||||
f'rtime="{ts(t)}" duration="{ts(dur)}">')
|
||||
a('<cartesian>1</cartesian>')
|
||||
a(f'<position coordinate="X">{x:.10f}</position>')
|
||||
a(f'<position coordinate="Y">{y:.10f}</position>')
|
||||
if z != 0:
|
||||
a(f'<position coordinate="Z">{z:.10f}</position>')
|
||||
a(f'<jumpPosition interpolationLength="{interpolation:.5f}">1</jumpPosition>')
|
||||
a('</audioBlockFormat>')
|
||||
a('</audioChannelFormat>')
|
||||
for i in range(10):
|
||||
a(f'<audioTrackUID UID="ATU_{i + 1:08x}" bitDepth="24" sampleRate="48000">')
|
||||
a(f'<audioTrackFormatIDRef>AT_0001{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
|
||||
a('</audioTrackUID>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioTrackUID UID="ATU_{11 + i:08x}" bitDepth="24" sampleRate="48000">')
|
||||
a(f'<audioTrackFormatIDRef>AT_0003{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
|
||||
a('</audioTrackUID>')
|
||||
for i in range(10):
|
||||
a(f'<audioTrackFormat audioTrackFormatID="AT_0001{0x1001 + i:04x}_01" '
|
||||
f'audioTrackFormatName="PCM_{BED_NAMES[i]}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioStreamFormatIDRef>AS_0001{0x1001 + i:04x}</audioStreamFormatIDRef>')
|
||||
a('</audioTrackFormat>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioTrackFormat audioTrackFormatID="AT_0003{0x1001 + i:04x}_01" '
|
||||
f'audioTrackFormatName="PCM_JOC_Object_{i+1}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioStreamFormatIDRef>AS_0003{0x1001 + i:04x}</audioStreamFormatIDRef>')
|
||||
a('</audioTrackFormat>')
|
||||
for i in range(10):
|
||||
a(f'<audioStreamFormat audioStreamFormatID="AS_0001{0x1001 + i:04x}" '
|
||||
f'audioStreamFormatName="PCM_{BED_NAMES[i]}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioChannelFormatIDRef>AC_0001{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a('<audioPackFormatIDRef>AP_00011001</audioPackFormatIDRef>')
|
||||
a(f'<audioTrackFormatIDRef>AT_0001{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a('</audioStreamFormat>')
|
||||
for i in range(N_OBJ):
|
||||
a(f'<audioStreamFormat audioStreamFormatID="AS_0003{0x1001 + i:04x}" '
|
||||
f'audioStreamFormatName="PCM_JOC_Object_{i+1}" formatDefinition="PCM" formatLabel="0001">')
|
||||
a(f'<audioChannelFormatIDRef>AC_0003{0x1001 + i:04x}</audioChannelFormatIDRef>')
|
||||
a(f'<audioPackFormatIDRef>AP_0003{0x1001 + i:04x}</audioPackFormatIDRef>')
|
||||
a(f'<audioTrackFormatIDRef>AT_0003{0x1001 + i:04x}_01</audioTrackFormatIDRef>')
|
||||
a('</audioStreamFormat>')
|
||||
a('</audioFormatExtended></format></coreMetadata>')
|
||||
a('</ebuCoreMain>')
|
||||
return ''.join(w).encode('utf-8')
|
||||
@@ -0,0 +1,199 @@
|
||||
"""校验 ADM BWF 的 RF64、通道、axml、chna、dbmd 和对象引用结构。
|
||||
|
||||
用法:``python src/adm_validate.py <file.wav> [more.wav ...]``,全部通过时退出码为 0。
|
||||
"""
|
||||
import struct, sys, re, os
|
||||
|
||||
def fail(msgs, m): msgs.append(m)
|
||||
|
||||
def walk_chunks(path):
|
||||
chunks, ds64 = [], {}
|
||||
with open(path, "rb") as f:
|
||||
riff = f.read(4); f.read(4); wave = f.read(4)
|
||||
if riff not in (b"RIFF", b"RF64"):
|
||||
return None, None, f"File does not have a 'RIFF' or 'RF64' chunk"
|
||||
if wave != b"WAVE":
|
||||
return None, None, "File does not have a required 'WAVE' chunk"
|
||||
while True:
|
||||
off = f.tell()
|
||||
cid = f.read(4)
|
||||
if len(cid) < 4: break
|
||||
sz = struct.unpack("<I", f.read(4))[0]
|
||||
if cid == b"ds64":
|
||||
body = f.read(sz + (sz & 1))
|
||||
riff64, data64, sample64, _ = struct.unpack("<QQQI", body[:28])
|
||||
ds64 = dict(riff64=riff64, data64=data64, sample64=sample64)
|
||||
chunks.append(("ds64", off, sz)); continue
|
||||
chunks.append((cid.decode("latin1"), off, sz))
|
||||
eff = ds64.get("data64", sz) if (sz == 0xFFFFFFFF and cid == b"data") else sz
|
||||
f.seek(off + 8 + eff + (eff & 1))
|
||||
return chunks, ds64, None
|
||||
|
||||
def read_body(path, chunks, cid):
|
||||
for c, off, sz in chunks:
|
||||
if c == cid:
|
||||
with open(path, "rb") as f:
|
||||
f.seek(off + 8)
|
||||
return f.read(sz)
|
||||
return None
|
||||
|
||||
def parse_chna(body):
|
||||
n_track, n_uid = struct.unpack("<HH", body[:4])
|
||||
rows, p = [], 4
|
||||
while p + 40 <= len(body):
|
||||
trk = struct.unpack("<H", body[p:p+2])[0]
|
||||
uid = body[p+2:p+14].rstrip(b"\x00").decode()
|
||||
tf = body[p+14:p+28].rstrip(b"\x00").decode()
|
||||
pk = body[p+28:p+40].rstrip(b"\x00").decode()
|
||||
rows.append((trk, uid, tf, pk)); p += 40
|
||||
return n_track, n_uid, rows
|
||||
|
||||
def decode_channel_input(ao_id):
|
||||
"""将 ``AO_xxxx`` 的十六进制标识符解码为低 12 位通道输入号。"""
|
||||
m = re.fullmatch(r"AO_([0-9a-fA-F]+)", ao_id)
|
||||
if not m:
|
||||
return None
|
||||
v = int(m.group(1), 16)
|
||||
if v > 0x1FFF: # 超过 12 位通道域
|
||||
return None
|
||||
return v & 0x0FFF
|
||||
|
||||
def ts_sec(s):
|
||||
h, m, rest = s.split(":")
|
||||
return int(h) * 3600 + int(m) * 60 + float(rest)
|
||||
|
||||
def validate(path, axml_override=None, chna_override=None):
|
||||
msgs = []
|
||||
chunks, ds64, err = walk_chunks(path)
|
||||
if err:
|
||||
return [err]
|
||||
have = {c for c, _, _ in chunks}
|
||||
for need in ("fmt ", "data", "axml", "chna", "dbmd"):
|
||||
if need not in have:
|
||||
fail(msgs, f"File does not have a required '{need.strip()}' chunk")
|
||||
if msgs:
|
||||
return msgs
|
||||
fmt = read_body(path, chunks, "fmt ")
|
||||
f_tag, f_ch, f_rate, _, _, f_bits = struct.unpack("<HHIIHH", fmt[:16])
|
||||
chna_body = chna_override if chna_override is not None else read_body(path, chunks, "chna")
|
||||
n_track, n_uid, rows = parse_chna(chna_body)
|
||||
if f_ch != n_track:
|
||||
fail(msgs, f"Mismatched number of audio channels and chna entries "
|
||||
f"(fmt={f_ch} chna={n_track})")
|
||||
ax_raw = axml_override if axml_override is not None else read_body(path, chunks, "axml")
|
||||
ax = ax_raw.decode("utf-8")
|
||||
|
||||
# --- audioObjectID 十六进制通道输入号解码 ---
|
||||
objs = re.findall(r'audioObjectID="(AO_[0-9a-zA-Z]+)"', ax)
|
||||
bed_ch, obj_ch = [], []
|
||||
for ao in objs:
|
||||
cid = decode_channel_input(ao)
|
||||
if cid is None:
|
||||
fail(msgs, f"Invalid ADM BWF XML format: cannot decode channel "
|
||||
f"input ID from AudioObjectID '{ao}'")
|
||||
continue
|
||||
if ao == "AO_1001":
|
||||
bed_ch.append(cid)
|
||||
else:
|
||||
if cid <= 10:
|
||||
fail(msgs, f"Source channel index should be greater than 10 "
|
||||
f"for objects ('{ao}' -> {cid})")
|
||||
obj_ch.append(cid)
|
||||
|
||||
# UID 十六进制 → 必须与 chna 表一致
|
||||
uid_map = {uid: trk for trk, uid, tf, pk in rows}
|
||||
for uid in re.findall(r'UID="(ATU_[0-9a-zA-Z]+)"', ax):
|
||||
if uid not in uid_map:
|
||||
fail(msgs, f"'{uid}' is not referenced in 'chna' chunk UID table")
|
||||
continue
|
||||
m = re.fullmatch(r"ATU_([0-9a-fA-F]+)", uid)
|
||||
if m:
|
||||
v = int(m.group(1), 16)
|
||||
if v > 128:
|
||||
fail(msgs, f"Channel index out of range (UID {uid} -> {v})")
|
||||
|
||||
# 轨数一致性:axml audioTrackUID 数 == fmt 声道数
|
||||
n_tu = len(re.findall(r"<audioTrackUID ", ax))
|
||||
if n_tu != f_ch:
|
||||
fail(msgs, f"Number of channels declared in ADM ({n_tu}) does not "
|
||||
f"match 'fmt ' chunk ({f_ch})")
|
||||
|
||||
# sampleRate / bitDepth 一致
|
||||
for sr in set(re.findall(r'sampleRate="(\d+)"', ax)):
|
||||
if int(sr) != f_rate:
|
||||
fail(msgs, f"Mismatched track sample rate between ADM and WAV ({sr} vs {f_rate})")
|
||||
for bd in set(re.findall(r'bitDepth="(\d+)"', ax)):
|
||||
if int(bd) != f_bits:
|
||||
fail(msgs, f"Mismatched track bit depth between ADM and WAV ({bd} vs {f_bits})")
|
||||
|
||||
# audioProgramme 唯一性 / audioContent ≥1
|
||||
if ax.count("<audioProgramme ") != 1:
|
||||
fail(msgs, "ADM has more than one audioProgramme object -- there must be only one"
|
||||
if ax.count("<audioProgramme ") > 1
|
||||
else "ADM does not have a required audioProgramme object")
|
||||
if "<audioContent " not in ax:
|
||||
fail(msgs, "audioProgramme object does not have a required audioContent object")
|
||||
|
||||
# 每个 channelFormat ≥1 blockFormat + 对象块链连续性
|
||||
cfs = re.findall(r'<audioChannelFormat [^>]*typeLabel="0003".*?</audioChannelFormat>', ax, re.S)
|
||||
object_block_formats = 0
|
||||
for seg in cfs:
|
||||
cf_id = re.search(r'audioChannelFormatID="([^"]+)"', seg).group(1)
|
||||
block_xml = re.findall(r'<audioBlockFormat [^>]*rtime="[^"]+".*?</audioBlockFormat>',
|
||||
seg, re.S)
|
||||
object_block_formats += len(block_xml)
|
||||
blocks = []
|
||||
for block_index, block in enumerate(block_xml, 1):
|
||||
timing = re.search(r'rtime="([^"]+)" duration="([^"]+)"', block)
|
||||
if timing is None:
|
||||
continue
|
||||
rtime, duration = timing.groups()
|
||||
blocks.append((rtime, duration))
|
||||
jump = re.search(
|
||||
r'<jumpPosition interpolationLength="([^"]+)">1</jumpPosition>', block)
|
||||
if jump is not None and float(jump.group(1)) > ts_sec(duration) + 1e-8:
|
||||
fail(msgs, f"Interpolation length exceeds duration in block format "
|
||||
f"{block_index} of {cf_id}: {jump.group(1)} > {duration}")
|
||||
if not blocks:
|
||||
fail(msgs, f"AudioChannelFormat {cf_id} is missing audioBlockFormat sub-element")
|
||||
continue
|
||||
for i in range(len(blocks) - 1):
|
||||
end_i = ts_sec(blocks[i][0]) + ts_sec(blocks[i][1])
|
||||
nxt = ts_sec(blocks[i + 1][0])
|
||||
if abs(end_i - nxt) > 2e-5:
|
||||
fail(msgs, f"Time gap between block format {i+1} and {i+2} of {cf_id}: "
|
||||
f"{end_i:.5f} vs {nxt:.5f}")
|
||||
return msgs, dict(fmt_ch=f_ch, fmt_rate=f_rate, fmt_bits=f_bits,
|
||||
chna=n_track, objects=len(obj_ch), bed=len(bed_ch),
|
||||
trackUIDs=n_tu, axml_bytes=len(ax_raw),
|
||||
audioBlockFormats=ax.count("<audioBlockFormat "),
|
||||
objectBlockFormats=object_block_formats)
|
||||
|
||||
def main():
|
||||
import argparse
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("files", nargs="+")
|
||||
ap.add_argument("--axml-file", default=None, help="用该文件内容替换 wav 内 axml(对照实验)")
|
||||
ap.add_argument("--chna-file", default=None, help="用该文件内容替换 wav 内 chna(对照实验)")
|
||||
a = ap.parse_args()
|
||||
ax_o = open(a.axml_file, "rb").read() if a.axml_file else None
|
||||
ch_o = open(a.chna_file, "rb").read() if a.chna_file else None
|
||||
rc = 0
|
||||
for p in a.files:
|
||||
r = validate(p, axml_override=ax_o, chna_override=ch_o)
|
||||
name = os.path.basename(p)
|
||||
if isinstance(r, list):
|
||||
msgs, info = r, {}
|
||||
else:
|
||||
msgs, info = r
|
||||
if msgs:
|
||||
rc = 1
|
||||
print(f"[FAIL] {name}")
|
||||
for m in msgs:
|
||||
print(" -", m)
|
||||
else:
|
||||
print(f"[PASS] {name} {info}")
|
||||
return rc
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
+282
@@ -0,0 +1,282 @@
|
||||
"""从常见 E-AC-3 同步帧直接提取连续 EMDF 容器。
|
||||
|
||||
扫描器检查八种全局位对齐,定位 ``0x5838`` 同步字,验证容器长度并解析各
|
||||
payload config,因此不要求 EMDF 在原始 E-AC-3 文件中按字节对齐。
|
||||
|
||||
本模块有意只覆盖“完整 EMDF 容器在一个同步帧中连续出现”的常见情形。不解析
|
||||
E-AC-3 mantissa,也不重组被音频数据隔开的多个 skip-field 碎片;遇到这种输入会
|
||||
明确报错,让上层决定是否使用兼容桥。
|
||||
"""
|
||||
from dataclasses import dataclass
|
||||
import csv
|
||||
import hashlib
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from variant_error import UnsupportedVariantError
|
||||
|
||||
|
||||
SYNCWORD = 0x5838
|
||||
REQUIRED_JOC_IDS = frozenset((11, 14))
|
||||
|
||||
|
||||
class EmdfError(ValueError):
|
||||
"""EMDF 或其 E-AC-3 传输结构不符合本实现支持的范围。"""
|
||||
|
||||
|
||||
class BitReader:
|
||||
"""MSB-first 位读取器;位置以源数据的绝对 bit offset 表示。"""
|
||||
|
||||
def __init__(self, data, position=0, limit=None):
|
||||
self.data = memoryview(data)
|
||||
self.position = int(position)
|
||||
self.limit = len(self.data) * 8 if limit is None else int(limit)
|
||||
|
||||
def read(self, count):
|
||||
count = int(count)
|
||||
if count < 0 or self.position + count > self.limit:
|
||||
raise EmdfError(f"位流越界 @bit{self.position}, need={count}, limit={self.limit}")
|
||||
value = 0
|
||||
while count:
|
||||
byte_pos = self.position >> 3
|
||||
removed_left = self.position & 7
|
||||
take = min(count, 8 - removed_left)
|
||||
shift = 8 - removed_left - take
|
||||
value = (value << take) | ((self.data[byte_pos] >> shift) & ((1 << take) - 1))
|
||||
self.position += take
|
||||
count -= take
|
||||
return value
|
||||
|
||||
def skip(self, count):
|
||||
self.read(count)
|
||||
|
||||
def read_bytes(self, count):
|
||||
return bytes(self.read(8) for _ in range(count))
|
||||
|
||||
|
||||
def variable_bits(reader, width, max_groups=8):
|
||||
"""读取 EMDF ``variable_bits(width)`` 变长整数。"""
|
||||
value = 0
|
||||
for _ in range(max_groups):
|
||||
value += reader.read(width)
|
||||
more = reader.read(1)
|
||||
if not more:
|
||||
return value
|
||||
value = (value + 1) << width
|
||||
raise EmdfError(f"variable_bits({width}) 延伸组过多")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EmdfContainer:
|
||||
start_bit: int
|
||||
raw: bytes
|
||||
payloads: dict
|
||||
sample_offsets: dict
|
||||
|
||||
|
||||
def _parse_at(data, start_bit):
|
||||
"""在已知 syncword 的 bit offset 解析一个 EMDF 容器。"""
|
||||
reader = BitReader(data, start_bit)
|
||||
if reader.read(16) != SYNCWORD:
|
||||
raise EmdfError(f"EMDF syncword 不匹配 @bit{start_bit}")
|
||||
length = reader.read(16)
|
||||
body_start = reader.position
|
||||
body_end = body_start + length * 8
|
||||
if body_end > reader.limit:
|
||||
raise EmdfError(f"EMDF 容器越界 @bit{start_bit}: length={length}")
|
||||
reader.limit = body_end
|
||||
|
||||
version = reader.read(2)
|
||||
if version == 3:
|
||||
version += variable_bits(reader, 2)
|
||||
key_id = reader.read(3)
|
||||
if key_id == 7:
|
||||
key_id += variable_bits(reader, 3)
|
||||
# TS 103 420 JOC 使用 version=0/key_id=0;严格限制也能排除音频中的伪 marker。
|
||||
if version != 0 or key_id != 0:
|
||||
raise EmdfError(f"不支持的 EMDF version/key_id: {version}/{key_id}")
|
||||
|
||||
payloads = {}
|
||||
sample_offsets = {}
|
||||
terminated = False
|
||||
while reader.position + 5 <= body_end:
|
||||
payload_id = reader.read(5)
|
||||
if payload_id == 0:
|
||||
terminated = True
|
||||
break
|
||||
if payload_id == 0x1F:
|
||||
payload_id += variable_bits(reader, 5)
|
||||
if payload_id in payloads:
|
||||
raise EmdfError(f"同一 EMDF 容器重复 payload id {payload_id}")
|
||||
|
||||
has_sample_offset = bool(reader.read(1))
|
||||
sample_offset = (reader.read(12) >> 1) if has_sample_offset else 0
|
||||
if reader.read(1):
|
||||
variable_bits(reader, 11) # duration
|
||||
if reader.read(1):
|
||||
variable_bits(reader, 2) # group id
|
||||
if reader.read(1):
|
||||
reader.skip(8) # codec data
|
||||
|
||||
if not reader.read(1): # discard_unknown_payload
|
||||
frame_aligned = False
|
||||
if not has_sample_offset:
|
||||
frame_aligned = bool(reader.read(1))
|
||||
if frame_aligned:
|
||||
reader.skip(2)
|
||||
if has_sample_offset or frame_aligned:
|
||||
reader.skip(7)
|
||||
|
||||
payload_size = variable_bits(reader, 8)
|
||||
if reader.position + payload_size * 8 > body_end:
|
||||
raise EmdfError(
|
||||
f"payload id {payload_id} 越界: size={payload_size}, @bit{reader.position}")
|
||||
payloads[payload_id] = reader.read_bytes(payload_size)
|
||||
sample_offsets[payload_id] = sample_offset
|
||||
|
||||
if not terminated:
|
||||
raise EmdfError("EMDF 容器缺少 payload id 0 终止符")
|
||||
total_bytes = 4 + length
|
||||
raw_reader = BitReader(data, start_bit, start_bit + total_bytes * 8)
|
||||
raw = raw_reader.read_bytes(total_bytes)
|
||||
return EmdfContainer(start_bit, raw, payloads, sample_offsets)
|
||||
|
||||
|
||||
def _marker_offsets(data):
|
||||
"""以 NumPy 批量检查八种位移,返回可能的 0x5838 bit offsets。"""
|
||||
source = np.frombuffer(data, dtype=np.uint8)
|
||||
if source.size < 4:
|
||||
return []
|
||||
offsets = []
|
||||
for shift in range(8):
|
||||
if shift == 0:
|
||||
aligned = source
|
||||
else:
|
||||
aligned = np.bitwise_or(
|
||||
np.left_shift(source[:-1].astype(np.uint16), shift) & 0xFF,
|
||||
np.right_shift(source[1:].astype(np.uint16), 8 - shift),
|
||||
).astype(np.uint8)
|
||||
hits = np.flatnonzero((aligned[:-1] == 0x58) & (aligned[1:] == 0x38))
|
||||
offsets.extend(int(hit) * 8 + shift for hit in hits)
|
||||
return sorted(offsets)
|
||||
|
||||
|
||||
def find_joc_emdf(frame):
|
||||
"""返回同步帧中唯一、连续且包含 ID11/ID14 的 JOC EMDF 容器。"""
|
||||
matches = []
|
||||
offsets = _marker_offsets(frame)
|
||||
parsed_candidates = []
|
||||
parse_errors = []
|
||||
for start_bit in offsets:
|
||||
try:
|
||||
container = _parse_at(frame, start_bit)
|
||||
except EmdfError as exc:
|
||||
if len(parse_errors) < 8:
|
||||
parse_errors.append({"start_bit": start_bit, "error": str(exc)})
|
||||
continue
|
||||
parsed_candidates.append({
|
||||
"start_bit": start_bit,
|
||||
"payload_ids": list(container.payloads),
|
||||
"payload_lengths": {str(k): len(v) for k, v in container.payloads.items()},
|
||||
})
|
||||
if REQUIRED_JOC_IDS.issubset(container.payloads):
|
||||
matches.append(container)
|
||||
if not matches:
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "no_contiguous_joc_container",
|
||||
"同步帧中未找到可连续解析且同时包含 ID11/ID14 的 EMDF 容器",
|
||||
details={
|
||||
"syncframe_bytes": len(frame),
|
||||
"marker_bit_offsets": offsets,
|
||||
"parsed_candidates": parsed_candidates,
|
||||
"candidate_parse_errors": parse_errors,
|
||||
"repair_hint": "检查 EMDF 是否跨多个 audio-block skip field 分片,或 payload config 是否变化",
|
||||
})
|
||||
if len(matches) != 1:
|
||||
starts = [item.start_bit for item in matches]
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "multiple_joc_containers",
|
||||
"同步帧中存在多个可用 JOC EMDF,当前无法自动选择",
|
||||
details={"syncframe_bytes": len(frame), "joc_container_start_bits": starts})
|
||||
return matches[0]
|
||||
|
||||
|
||||
def parse_container(data):
|
||||
"""解析从 syncword 开始、已经重新按字节对齐保存的 EMDF 容器。"""
|
||||
container = _parse_at(data, 0)
|
||||
if len(container.raw) != len(data):
|
||||
raise EmdfError(f"EMDF 文件尾有额外数据: parsed={len(container.raw)}, file={len(data)}")
|
||||
return container
|
||||
|
||||
|
||||
def iter_eac3_frames(data):
|
||||
"""按 E-AC-3 ``frmsiz`` 遍历同步帧,拒绝静默重同步。"""
|
||||
pos = 0
|
||||
while pos < len(data):
|
||||
if pos + 4 > len(data) or data[pos:pos + 2] != b"\x0b\x77":
|
||||
raise UnsupportedVariantError(
|
||||
"eac3_transport", "syncframe_header",
|
||||
"E-AC-3 同步帧头无效或出现了未处理的子流排列",
|
||||
details={
|
||||
"byte_offset": pos,
|
||||
"remaining_bytes": len(data) - pos,
|
||||
"next_16_bytes_hex": data[pos:pos + 16].hex(),
|
||||
})
|
||||
size = ((((data[pos + 2] & 7) << 8) | data[pos + 3]) + 1) * 2
|
||||
if pos + size > len(data):
|
||||
raise UnsupportedVariantError(
|
||||
"eac3_transport", "truncated_syncframe",
|
||||
"E-AC-3 末帧长度超过输入剩余数据",
|
||||
details={
|
||||
"byte_offset": pos,
|
||||
"declared_frame_bytes": size,
|
||||
"remaining_bytes": len(data) - pos,
|
||||
})
|
||||
yield data[pos:pos + size]
|
||||
pos += size
|
||||
|
||||
|
||||
def extract_index(eac3_path, output_dir, max_frames=None):
|
||||
"""将裸 E-AC-3 的连续 EMDF 保存为 ``frames.csv + emdf/``。"""
|
||||
output_dir = Path(output_dir)
|
||||
emdf_dir = output_dir / "emdf"
|
||||
emdf_dir.mkdir(parents=True, exist_ok=True)
|
||||
frames = iter_eac3_frames(Path(eac3_path).read_bytes())
|
||||
rows = []
|
||||
for frame_number, frame in enumerate(frames):
|
||||
if max_frames is not None and frame_number >= max_frames:
|
||||
break
|
||||
try:
|
||||
container = find_joc_emdf(frame)
|
||||
except UnsupportedVariantError as exc:
|
||||
exc.add_context(frame=frame_number, details={"syncframe_bytes": len(frame)})
|
||||
raise
|
||||
except EmdfError as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "container_syntax",
|
||||
"EMDF 容器语法无法解析",
|
||||
frame=frame_number,
|
||||
details={"syncframe_bytes": len(frame), "parser_error": str(exc)}) from exc
|
||||
digest = hashlib.sha256(container.raw).hexdigest()
|
||||
target = emdf_dir / f"{digest}.bin"
|
||||
if not target.is_file():
|
||||
target.write_bytes(container.raw)
|
||||
rows.append({
|
||||
"frame": frame_number,
|
||||
"emdf_hash": digest,
|
||||
"emdf_size": len(container.raw),
|
||||
"emdf_start_bit": container.start_bit,
|
||||
"payload_ids": ";".join(str(x) for x in container.payloads),
|
||||
"error": "",
|
||||
})
|
||||
if (frame_number + 1) % 1000 == 0:
|
||||
print(f"[metadata] {frame_number + 1} frames", flush=True)
|
||||
if not rows:
|
||||
raise EmdfError("E-AC-3 输入中没有可处理的同步帧")
|
||||
with (output_dir / "frames.csv").open("w", encoding="utf-8", newline="") as fp:
|
||||
fields = ("frame", "emdf_hash", "emdf_size", "emdf_start_bit", "payload_ids", "error")
|
||||
writer = csv.DictWriter(fp, fieldnames=fields)
|
||||
writer.writeheader()
|
||||
writer.writerows(rows)
|
||||
return output_dir
|
||||
@@ -0,0 +1,76 @@
|
||||
"""解包 EVO MD-set evolution 载荷,返回各 payload ID 的字节数据和位偏移。"""
|
||||
_MARK = "1001001000000"
|
||||
|
||||
# 各子载荷字段: (id, 头部前缀, 同步标记, 后缀常量, 尺寸域位数, 是否有转义)
|
||||
# 头部前缀 = 5 位 id 的 MSB 二进制(id11 字段前另有 5 位容器前导 00000)
|
||||
# 尺寸域单位 = nibble(4 位)。id14 有转义:9 位值=0 → 再读 9 位 = 字节数。
|
||||
_LAYOUT = [
|
||||
(11, "0000001011", "010000000000000", 8, False),
|
||||
(14, "01110", "01000000000000", 9, True),
|
||||
(2, "00010", "000100", 7, False),
|
||||
(1, "00001", "1110000000000000000000000000", 4, False),
|
||||
(30, "11110", "1110000000000000000000000000", 4, False),
|
||||
]
|
||||
|
||||
|
||||
def _msb_bits(data: bytes):
|
||||
return [(x >> (7 - i)) & 1 for x in data for i in range(8)]
|
||||
|
||||
|
||||
def _val(bits, off, n):
|
||||
v = 0
|
||||
for b in bits[off:off + n]:
|
||||
v = (v << 1) | b
|
||||
return v
|
||||
|
||||
|
||||
class _LooseSkip(Exception):
|
||||
def __init__(self, ident):
|
||||
self.ident = ident
|
||||
|
||||
|
||||
def unpack_evolution(payload: bytes, loose=False):
|
||||
"""解包 evolution 载荷 → (subs, offsets)。subs 键为 id 整数。
|
||||
loose=True 时对每个 id 的 (前缀+标记+后缀) 全模式做位流重同步扫描
|
||||
(不同编码流的子载荷次序/内部常量可有合法差异,如 kanata 的 id11)。"""
|
||||
bits = _msb_bits(payload)
|
||||
pos = 0
|
||||
subs = {}
|
||||
offsets = {}
|
||||
for ident, pref, suff, sbits, escape in _LAYOUT:
|
||||
pat = pref + _MARK + suff
|
||||
if loose:
|
||||
hit = -1
|
||||
for i in range(pos, len(bits) - len(pat)):
|
||||
if ''.join(map(str, bits[i:i + len(pat)])) == pat:
|
||||
hit = i
|
||||
break
|
||||
if hit < 0:
|
||||
continue
|
||||
pos = hit + len(pat)
|
||||
else:
|
||||
for name, const, expect in (("前缀", bits[pos:pos + len(pref)], pref),
|
||||
("标记", bits[pos + len(pref):pos + len(pref) + len(_MARK)], _MARK),
|
||||
("后缀", bits[pos + len(pref) + len(_MARK):
|
||||
pos + len(pref) + len(_MARK) + len(suff)], suff)):
|
||||
got = ''.join(map(str, const))
|
||||
if got != expect:
|
||||
raise ValueError(f"id={ident}: {name}常量不匹配 @bit{pos} got={got} want={expect}")
|
||||
pos += len(pref) + len(_MARK) + len(suff)
|
||||
n_nib = _val(bits, pos, sbits)
|
||||
pos += sbits
|
||||
if escape and n_nib == 1:
|
||||
n_nib = 512 + _val(bits, pos, sbits)
|
||||
pos += sbits
|
||||
n_bits = n_nib * 4
|
||||
body = bits[pos:pos + n_bits]
|
||||
pos += n_bits
|
||||
b = bytearray(len(body) // 8)
|
||||
for i in range(0, len(body) // 8 * 8, 8):
|
||||
v = 0
|
||||
for x in body[i:i + 8]:
|
||||
v = (v << 1) | x
|
||||
b[i // 8] = v
|
||||
subs[ident] = bytes(b)
|
||||
offsets[ident] = pos
|
||||
return subs, offsets
|
||||
@@ -0,0 +1,176 @@
|
||||
"""解析 JOC 位流并生成对象混合矩阵。
|
||||
|
||||
范围:
|
||||
- EMDF ID14 的 joc_header、joc_info 和 Huffman joc_data;
|
||||
- 差分解码得到 joc_mix_mtx_q;
|
||||
- 去量化得到 joc_mix_mtx_dq;
|
||||
- 位流自洽验证(joc_data 后剩余 = padding_bits 0..7 + 可能 joc_ext_data)
|
||||
后续的时间插值、QMF/时域重建和 ``joc_clipgain`` 位于 ``renderer.py``。
|
||||
Sparse 分支仍缺少实际样本验证。
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
# 格式:节点数组 [left, right];正 = 内部节点索引,负 = 叶(值 = -node-1)
|
||||
_TABLES_PATH = Path(__file__).resolve().parent.parent / "data" / "tables.npz"
|
||||
_HUFF_NAMES = (
|
||||
"joc_huff_code_coarse_generic",
|
||||
"joc_huff_code_fine_generic",
|
||||
"joc_huff_code_coarse_coeff_sparse",
|
||||
"joc_huff_code_fine_coeff_sparse",
|
||||
"joc_huff_code_5ch_pos_index_sparse",
|
||||
"joc_huff_code_7ch_pos_index_sparse",
|
||||
)
|
||||
|
||||
def _load_huff_tables():
|
||||
with np.load(_TABLES_PATH) as tables:
|
||||
return {
|
||||
name: np.asarray(tables[name], dtype=np.int64).tolist()
|
||||
for name in _HUFF_NAMES
|
||||
}
|
||||
|
||||
H = _load_huff_tables()
|
||||
|
||||
JOC_NUM_CHANNELS = {0: 5, 1: 7, 2: 7, 3: 5, 4: 7} # Table 33
|
||||
JOC_NUM_BANDS = {0: 1, 1: 3, 2: 5, 3: 7, 4: 9, 5: 12, 6: 15, 7: 23} # Table 35
|
||||
_PUBLIC_TABLE39_EXCERPT_UNUSED = { # 仅保留作表格差异说明;渲染映射在 joc_qmf.py。
|
||||
23: [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22,22],
|
||||
}
|
||||
|
||||
class BR:
|
||||
def __init__(self, data, pos=0):
|
||||
self.d = data
|
||||
self.p = pos
|
||||
def bits(self, n):
|
||||
v = 0
|
||||
for _ in range(n):
|
||||
v = (v << 1) | ((self.d[self.p >> 3] >> (7 - (self.p & 7))) & 1)
|
||||
self.p += 1
|
||||
return v
|
||||
|
||||
|
||||
def huff_decode(tree, br):
|
||||
node = 0
|
||||
while node >= 0:
|
||||
b = br.bits(1)
|
||||
node = tree[node][b]
|
||||
return -node - 1
|
||||
|
||||
|
||||
def get_huff_code(mode, typ, nch):
|
||||
if typ == "IDX":
|
||||
return H["joc_huff_code_5ch_pos_index_sparse" if nch == 5 else "joc_huff_code_7ch_pos_index_sparse"]
|
||||
if typ == "VEC":
|
||||
return H["joc_huff_code_coarse_coeff_sparse" if mode == 0 else "joc_huff_code_fine_coeff_sparse"]
|
||||
# MTX
|
||||
return H["joc_huff_code_coarse_generic" if mode == 0 else "joc_huff_code_fine_generic"]
|
||||
|
||||
|
||||
def parse_joc(payload):
|
||||
"""解析 id14 载荷(joc() 位流)。返回字段 dict + 解析后剩余位数。"""
|
||||
br = BR(payload)
|
||||
out = {}
|
||||
out["dmx_config_idx"] = br.bits(3)
|
||||
out["num_objects_bits"] = br.bits(6)
|
||||
out["ext_config_idx"] = br.bits(3)
|
||||
n_objects = out["num_objects_bits"] + 1
|
||||
n_channels = JOC_NUM_CHANNELS.get(out["dmx_config_idx"])
|
||||
out["n_objects"], out["n_channels"] = n_objects, n_channels
|
||||
out["clipgain_x_bits"] = br.bits(3)
|
||||
out["clipgain_y_bits"] = br.bits(5)
|
||||
out["seq_count_bits"] = br.bits(10)
|
||||
# clipgain = 1 + (y/32)·2^(x−4),值域为 [1, 8.75]。
|
||||
out["clipgain"] = 1 + out["clipgain_y_bits"] / 32.0 * 2 ** (out["clipgain_x_bits"] - 4)
|
||||
objs = []
|
||||
for obj in range(n_objects):
|
||||
o = {}
|
||||
o["present"] = br.bits(1)
|
||||
if o["present"]:
|
||||
o["num_bands_idx"] = br.bits(3)
|
||||
o["n_bands"] = JOC_NUM_BANDS[o["num_bands_idx"]]
|
||||
o["sparse"] = br.bits(1)
|
||||
o["quant_idx"] = br.bits(1)
|
||||
o["slope_idx"] = br.bits(1)
|
||||
o["num_dpoints_bits"] = br.bits(1)
|
||||
o["n_dpoints"] = o["num_dpoints_bits"] + 1
|
||||
if o["slope_idx"] == 1:
|
||||
o["offset_ts"] = [br.bits(5) + 1 for _ in range(o["n_dpoints"])]
|
||||
objs.append(o)
|
||||
out["objs"] = objs
|
||||
# joc_data(Huffman)
|
||||
for obj, o in enumerate(objs):
|
||||
if not o["present"]:
|
||||
continue
|
||||
nquant = 96 if o["quant_idx"] == 0 else 192
|
||||
o["channel_idx"] = []
|
||||
o["vec"] = []
|
||||
o["mtx"] = []
|
||||
for dp in range(o["n_dpoints"]):
|
||||
if o["sparse"] == 1:
|
||||
# Sparse JOC 使用 VEC/IDX Huffman 树;此分支尚无真实码流验证。
|
||||
ci0 = br.bits(3)
|
||||
tree = get_huff_code(n_channels, "IDX", n_channels)
|
||||
ci = [ci0] + [huff_decode(tree, br) for _ in range(o["n_bands"] - 1)]
|
||||
o["channel_idx"].append(ci)
|
||||
tree = get_huff_code(o["quant_idx"], "VEC", n_channels)
|
||||
vec = [huff_decode(tree, br) for _ in range(o["n_bands"])]
|
||||
o["vec"].append(vec)
|
||||
else:
|
||||
tree = get_huff_code(o["quant_idx"], "MTX", n_channels)
|
||||
mtx = [[huff_decode(tree, br) for _ in range(o["n_bands"])]
|
||||
for _ in range(n_channels)]
|
||||
o["mtx"].append(mtx)
|
||||
out["data_end_bits"] = br.p
|
||||
out["remaining_bits"] = len(payload) * 8 - br.p
|
||||
out["tail_bytes"] = payload[br.p // 8:]
|
||||
return out
|
||||
|
||||
|
||||
def diff_decode(out):
|
||||
"""6.6.2:差分解码 → joc_mix_mtx_q[obj][dp][ch][pb]。"""
|
||||
mix_q = {}
|
||||
n_ch = out["n_channels"]
|
||||
for obj, o in enumerate(out["objs"]):
|
||||
if not o["present"]:
|
||||
continue
|
||||
nquant = 96 if o["quant_idx"] == 0 else 192
|
||||
q = np.zeros((o["n_dpoints"], n_ch, o["n_bands"]), dtype=np.int64)
|
||||
for dp in range(o["n_dpoints"]):
|
||||
if o["sparse"] == 1:
|
||||
# Sparse 差分路径尚无真实码流验证。
|
||||
offset = 50 if o["quant_idx"] == 0 else 100
|
||||
ci = o["channel_idx"][dp]
|
||||
vec = o["vec"][dp]
|
||||
for pb in range(o["n_bands"]):
|
||||
ci_mod = ci[0] if pb == 0 else (ci[pb - 1] + ci[pb]) % n_ch
|
||||
for ch in range(n_ch):
|
||||
if ch == ci_mod:
|
||||
if pb == 0:
|
||||
q[dp][ch][pb] = (offset + vec[pb]) % nquant
|
||||
else:
|
||||
q[dp][ch][pb] = (q[dp][ch][pb - 1] + vec[pb]) % nquant
|
||||
else:
|
||||
q[dp][ch][pb] = offset
|
||||
else:
|
||||
offset = 48 if o["quant_idx"] == 0 else 96
|
||||
mtx = o["mtx"][dp]
|
||||
for ch in range(n_ch):
|
||||
q[dp][ch][0] = (offset + mtx[ch][0]) % nquant
|
||||
for pb in range(1, o["n_bands"]):
|
||||
q[dp][ch][pb] = (q[dp][ch][pb - 1] + mtx[ch][pb]) % nquant
|
||||
mix_q[obj] = q
|
||||
return mix_q
|
||||
|
||||
|
||||
def dequantize(out, mix_q):
|
||||
"""6.6.4:去量化 → joc_mix_mtx_dq。"""
|
||||
mix_dq = {}
|
||||
for obj, o in enumerate(out["objs"]):
|
||||
if not o["present"]:
|
||||
continue
|
||||
nquant = 96 if o["quant_idx"] == 0 else 192
|
||||
q = mix_q[obj]
|
||||
dq = (q.astype(np.float64) - nquant / 2) * 820 / (4096 * (1 + o["quant_idx"]))
|
||||
mix_dq[obj] = dq
|
||||
return mix_dq
|
||||
+151
@@ -0,0 +1,151 @@
|
||||
"""实现 JOC QMF、参数带映射和矩阵时间插值。
|
||||
|
||||
静态表保存在 ``data`` 中;NumPy 批量函数保持各通道、各对象的状态彼此独立。
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
N = 64
|
||||
_TABLES = np.load(Path(__file__).resolve().parent.parent / "data" / "tables.npz")
|
||||
ANALYSIS_WINDOW = np.asarray(_TABLES["analysis_window"], dtype=np.float64)
|
||||
QMF5_WINDOW = np.asarray(_TABLES["qmf5_window"], dtype=np.float64)
|
||||
|
||||
|
||||
# 子带到参数带的映射表。
|
||||
_TABLE39 = {
|
||||
23: "0 1 2 3 4 5 6 7 8 9 10 11 12 12 13 13 14 14 15 15 16 16 16 17 17 17 18 18 18 18 19 19 19 19 19 20 20 20 20 20 20 21 21 21 21 21 21 21 22 22 22 22 22 22 22 22 22 22 22 22 22 22 22 22",
|
||||
15: "0 1 2 3 4 5 6 7 8 9 9 10 10 11 11 11 12 12 12 12 13 13 13 13 13 13 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14 14",
|
||||
12: "0 1 2 3 4 4 5 5 6 6 6 7 7 7 8 8 8 8 9 9 9 9 9 10 10 10 10 10 10 10 10 10 10 10 10 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11 11",
|
||||
9: "0 1 2 3 3 3 4 5 5 6 6 6 7 7 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8 8",
|
||||
7: "0 1 2 2 3 3 3 3 4 4 4 4 4 4 5 5 5 5 5 5 5 5 5 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6 6",
|
||||
5: "0 1 1 2 2 2 2 2 2 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 3 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4 4",
|
||||
3: "0 0 0 1 1 1 1 1 1 1 1 1 1 1 1 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2 2",
|
||||
1: "0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0",
|
||||
}
|
||||
_PB_MAP = {k: np.fromstring(v, dtype=np.int64, sep=" ") for k, v in _TABLE39.items()}
|
||||
|
||||
|
||||
def sb_to_pb_table(n_bands):
|
||||
try:
|
||||
return _PB_MAP[n_bands]
|
||||
except KeyError as exc:
|
||||
raise ValueError(f"不支持的 JOC 参数带数: {n_bands}") from exc
|
||||
|
||||
|
||||
def interp_matrix(obj_info, dq, prev, n_ts=24):
|
||||
"""按数据点和跨帧状态插值,返回 ``[channel,subband,timeslot]``。"""
|
||||
dq_sb = np.asarray(dq, dtype=np.float64)[:, :, sb_to_pb_table(dq.shape[2])]
|
||||
previous = np.asarray(prev, dtype=np.float64)
|
||||
n_dp = obj_info["n_dpoints"]
|
||||
slope = obj_info["slope_idx"]
|
||||
if slope == 0:
|
||||
if n_dp == 1:
|
||||
alpha = (np.arange(n_ts, dtype=np.float64) + 1.0) / n_ts
|
||||
return previous[:, :, None] * (1.0 - alpha) + dq_sb[0, :, :, None] * alpha
|
||||
half = n_ts // 2
|
||||
a0 = (np.arange(half, dtype=np.float64) + 1.0) / half
|
||||
a1 = (np.arange(n_ts - half, dtype=np.float64) + 1.0) / (n_ts - half)
|
||||
first = previous[:, :, None] * (1.0 - a0) + dq_sb[0, :, :, None] * a0
|
||||
second = dq_sb[0, :, :, None] * (1.0 - a1) + dq_sb[1, :, :, None] * a1
|
||||
return np.concatenate((first, second), axis=2)
|
||||
|
||||
ts = np.arange(n_ts)
|
||||
offsets = obj_info.get("offset_ts", [])
|
||||
if n_dp == 1:
|
||||
return np.where(ts[None, None, :] < offsets[0], previous[:, :, None], dq_sb[0, :, :, None])
|
||||
out = np.where(ts[None, None, :] < offsets[0], previous[:, :, None], dq_sb[0, :, :, None])
|
||||
return np.where(ts[None, None, :] < offsets[1], out, dq_sb[1, :, :, None])
|
||||
|
||||
|
||||
def qmf_analysis_step(fifo, ring):
|
||||
"""分析 QMF 的单时隙入口;批量入口见 :func:`qmf_analysis_frame`。"""
|
||||
f = np.asarray(fifo, dtype=np.float64)
|
||||
r = np.asarray(ring, dtype=np.float64)
|
||||
single = f.ndim == 2
|
||||
if single:
|
||||
f, r = f[None, ...], r[None, ...]
|
||||
x, new = qmf_analysis_frame(f, r[:, None, :])
|
||||
interleaved = np.empty((len(f), 128), dtype=np.float64)
|
||||
interleaved[:, 0::2] = x[:, :, 0].real
|
||||
interleaved[:, 1::2] = x[:, :, 0].imag
|
||||
return (interleaved[0], new[0]) if single else (interleaved, new)
|
||||
|
||||
|
||||
def qmf_analysis_frame(fifo, rings):
|
||||
"""批量完成整帧时隙的分析窗、旋转和 64 点 FFT。"""
|
||||
f = np.asarray(fifo, dtype=np.float64)
|
||||
r = np.asarray(rings, dtype=np.float64)
|
||||
if f.ndim != 3 or f.shape[1:] != (9, 64) or r.ndim != 3 or r.shape[0] != f.shape[0] or r.shape[2] != 64:
|
||||
raise ValueError(f"analysis QMF shape 错误: fifo={f.shape}, rings={r.shape}")
|
||||
slots = r.shape[1]
|
||||
seq = np.concatenate((f[:, ::-1, :], r), axis=1)
|
||||
windows = np.lib.stride_tricks.sliding_window_view(seq, 9, axis=1)
|
||||
history = windows[:, :slots].transpose(0, 1, 3, 2)[:, :, ::-1, :]
|
||||
|
||||
t = ANALYSIS_WINDOW
|
||||
v36 = np.sum(history[:, :, 0::2, :] * t[[8, 6, 4, 2, 0]][None, None], axis=2)
|
||||
v40 = np.sum(history[:, :, 1::2, :] * t[[7, 5, 3, 1]][None, None], axis=2) + r * t[9]
|
||||
pairs = np.stack((v40, v36), axis=-1)
|
||||
kernel = pairs.reshape(len(f), slots, 16, 4, 2)[:, :, ::-1, ::-1, :].reshape(len(f), slots, 128)
|
||||
|
||||
k = np.arange(64, dtype=np.float64)
|
||||
a = 0.5 * np.sin(np.pi * k / 128.0)
|
||||
b = 0.5 * np.cos(np.pi * k / 128.0)
|
||||
re, im = kernel[:, :, 0::2], kernel[:, :, 1::2]
|
||||
freq = np.fft.fft((im * a - re * b) + 1j * (im * b + re * a), axis=2) / 64.0
|
||||
src = np.empty((len(f), slots, 128), dtype=np.float64)
|
||||
src[:, :, 0::2], src[:, :, 1::2] = freq.real, freq.imag
|
||||
out = np.empty_like(src).reshape(len(f), slots, 32, 4)
|
||||
out[:, :, :, 0] = src[:, :, 0:64:2]
|
||||
out[:, :, :, 1] = -src[:, :, 1:64:2]
|
||||
out[:, :, :, 2] = src[:, :, 126:62:-2]
|
||||
out[:, :, :, 3] = src[:, :, 127:63:-2]
|
||||
flat = out.reshape(len(f), slots, 128)
|
||||
complex_qmf = (flat[:, :, 0::2] + 1j * flat[:, :, 1::2]).transpose(0, 2, 1)
|
||||
combined = np.concatenate((f[:, ::-1, :], r), axis=1)
|
||||
new_fifo = combined[:, -9:, :][:, ::-1, :].copy()
|
||||
return complex_qmf, new_fifo
|
||||
|
||||
|
||||
_SURROUND_DC_A = np.array([
|
||||
-.0006242550443857908, -.0019234686624258757, -.0042654648423194885,
|
||||
-.008168308064341545, -.014327201060950756, -.023759860545396805,
|
||||
-.03757232800126076, -.05577569454908371, -.07568276673555374,
|
||||
-.09172472357749939, -.5979374051094055, -.09172472357749939,
|
||||
-.07568276673555374, -.05577569454908371, -.03757232800126076,
|
||||
-.023759860545396805, -.014327201060950756, -.008168308064341545,
|
||||
-.0042654648423194885, -.0019234686624258757, -.0006242550443857908,
|
||||
], dtype=np.float64)
|
||||
_SURROUND_DC_B = np.array([
|
||||
.0013996040215715766, .003839150769636035, .007512642536312342,
|
||||
.012419373728334904, .018367428332567215, .0249701626598835,
|
||||
.03167900815606117, .03785000368952751, .04283412545919418,
|
||||
.04607561603188515, .047200120985507965, .04607561603188515,
|
||||
.04283412545919418, .03785000368952751, .03167900815606117,
|
||||
.0249701626598835, .018367428332567215, .012419373728334904,
|
||||
.007512642536312342, .003839150769636035, .0013996040215715766,
|
||||
], dtype=np.float64)
|
||||
_SURROUND_DC_C = _SURROUND_DC_B + 1j * _SURROUND_DC_A
|
||||
|
||||
|
||||
def surround_post_frame(x, delay, dc_hist):
|
||||
"""处理 Ls/Rs 的 10 槽延迟、-j 旋转和 band-0 FIR。"""
|
||||
src = np.asarray(x, dtype=np.complex128)
|
||||
qdelay = np.asarray(delay, dtype=np.complex128).copy()
|
||||
hist = np.asarray(dc_hist, dtype=np.complex128).copy()
|
||||
if src.shape != (2, 64, 24):
|
||||
raise ValueError(f"surround QMF shape 错误: {src.shape}")
|
||||
out = np.empty_like(src)
|
||||
for group in range(0, 24, 4):
|
||||
current = src[:, :, group:group + 4].transpose(0, 2, 1)
|
||||
queued = np.concatenate((qdelay, current), axis=1)
|
||||
block = -1j * queued[:, :4, :]
|
||||
qdelay = queued[:, 4:, :]
|
||||
dc_buf = np.concatenate((hist, current[:, :, 0]), axis=1)
|
||||
windows = np.lib.stride_tricks.sliding_window_view(dc_buf, 21, axis=1)
|
||||
block[:, :, 0] = 2.0 * np.sum(windows * _SURROUND_DC_C[None, None, :], axis=2)
|
||||
hist = dc_buf[:, 4:]
|
||||
out[:, :, group:group + 4] = block.transpose(0, 2, 1)
|
||||
return out, qdelay, hist
|
||||
+324
@@ -0,0 +1,324 @@
|
||||
"""Evolution sidecar 的 JOC/OAMD 纯 Python 解析、校验和可读输出。"""
|
||||
from collections import Counter
|
||||
import csv
|
||||
import hashlib
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from adm_atmos import q_to_adm_xyz
|
||||
from evo_unpack import unpack_evolution
|
||||
from emdf import EmdfError, find_joc_emdf, iter_eac3_frames, parse_container
|
||||
from joc_decode import parse_joc
|
||||
from oamd_bits import JocFieldState, frame_update_values
|
||||
from variant_error import UnsupportedVariantError, bytes_descriptor
|
||||
|
||||
|
||||
class DirectPayloadIndex:
|
||||
"""One-pass in-memory view of contiguous EMDF containers in an E-AC-3 file.
|
||||
|
||||
The optional cache directory keeps the existing ``frames.csv + emdf/``
|
||||
contract, but normal processing reuses the containers already parsed during
|
||||
the scan instead of reading and parsing thousands of small files again.
|
||||
"""
|
||||
|
||||
def __init__(self, rows, subpayloads, sample_offsets, directory=None):
|
||||
self.rows = rows
|
||||
self._subpayloads = subpayloads
|
||||
self._sample_offsets = sample_offsets
|
||||
self.directory = Path(directory) if directory is not None else None
|
||||
|
||||
@classmethod
|
||||
def from_eac3(cls, eac3_path, max_frames=None, cache_dir=None):
|
||||
cache_dir = Path(cache_dir) if cache_dir is not None else None
|
||||
emdf_dir = None
|
||||
if cache_dir is not None:
|
||||
emdf_dir = cache_dir / "emdf"
|
||||
emdf_dir.mkdir(parents=True, exist_ok=True)
|
||||
rows = []
|
||||
payloads = []
|
||||
offsets = []
|
||||
frames = iter_eac3_frames(Path(eac3_path).read_bytes())
|
||||
for frame_number, frame in enumerate(frames):
|
||||
if max_frames is not None and frame_number >= max_frames:
|
||||
break
|
||||
try:
|
||||
container = find_joc_emdf(frame)
|
||||
except UnsupportedVariantError as exc:
|
||||
exc.add_context(frame=frame_number, details={"syncframe_bytes": len(frame)})
|
||||
raise
|
||||
except EmdfError as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_transport", "container_syntax",
|
||||
"EMDF 容器语法无法解析",
|
||||
frame=frame_number,
|
||||
details={"syncframe_bytes": len(frame), "parser_error": str(exc)}) from exc
|
||||
digest = hashlib.sha256(container.raw).hexdigest()
|
||||
if emdf_dir is not None:
|
||||
target = emdf_dir / f"{digest}.bin"
|
||||
if not target.is_file():
|
||||
target.write_bytes(container.raw)
|
||||
rows.append({
|
||||
"frame": frame_number,
|
||||
"emdf_hash": digest,
|
||||
"emdf_size": len(container.raw),
|
||||
"emdf_start_bit": container.start_bit,
|
||||
"payload_ids": ";".join(str(x) for x in container.payloads),
|
||||
"error": "",
|
||||
})
|
||||
payloads.append(container.payloads)
|
||||
offsets.append(container.sample_offsets)
|
||||
if (frame_number + 1) % 1000 == 0:
|
||||
print(f"[metadata] {frame_number + 1} frames", flush=True)
|
||||
if not rows:
|
||||
raise EmdfError("E-AC-3 输入中没有可处理的同步帧")
|
||||
if cache_dir is not None:
|
||||
with (cache_dir / "frames.csv").open("w", encoding="utf-8", newline="") as fp:
|
||||
fields = ("frame", "emdf_hash", "emdf_size", "emdf_start_bit", "payload_ids", "error")
|
||||
writer = csv.DictWriter(fp, fieldnames=fields)
|
||||
writer.writeheader()
|
||||
writer.writerows(rows)
|
||||
return cls(rows, payloads, offsets, cache_dir)
|
||||
|
||||
def subpayloads(self, row):
|
||||
return self._subpayloads[int(row["frame"])]
|
||||
|
||||
def subpayload_sample_offset(self, row, payload_id):
|
||||
return int(self._sample_offsets[int(row["frame"])].get(payload_id, 0))
|
||||
|
||||
def __len__(self):
|
||||
return len(self.rows)
|
||||
|
||||
|
||||
class PayloadIndex:
|
||||
"""旧 evolution sidecar 或直接 EMDF sidecar 的严格顺序视图。"""
|
||||
|
||||
def __init__(self, directory):
|
||||
self.directory = Path(directory)
|
||||
csv_path = self.directory / "frames.csv"
|
||||
self.payload_dir = self.directory / "payloads"
|
||||
self.emdf_dir = self.directory / "emdf"
|
||||
if not csv_path.is_file() or not (self.payload_dir.is_dir() or self.emdf_dir.is_dir()):
|
||||
raise FileNotFoundError(
|
||||
f"元数据目录需要 frames.csv 和 payloads/ 或 emdf/: {self.directory}")
|
||||
with csv_path.open(encoding="utf-8", newline="") as fp:
|
||||
self.rows = list(csv.DictReader(fp))
|
||||
if not self.rows:
|
||||
raise ValueError("frames.csv 为空")
|
||||
self._sub_cache = {}
|
||||
self._sample_offset_cache = {}
|
||||
|
||||
def payload(self, row):
|
||||
payload_hash = row.get("payload_hash", "")
|
||||
if not payload_hash:
|
||||
raise ValueError(f"frame {row.get('frame', '?')} 缺少 evolution payload")
|
||||
path = self.payload_dir / f"{payload_hash}.bin"
|
||||
if not path.is_file():
|
||||
raise FileNotFoundError(path)
|
||||
return path.read_bytes()
|
||||
|
||||
def subpayloads(self, row):
|
||||
"""返回 ``{payload_id: bytes}``,屏蔽两种 sidecar 容器的差异。"""
|
||||
emdf_hash = row.get("emdf_hash", "")
|
||||
if emdf_hash:
|
||||
key = ("emdf", emdf_hash)
|
||||
if key not in self._sub_cache:
|
||||
path = self.emdf_dir / f"{emdf_hash}.bin"
|
||||
if not path.is_file():
|
||||
raise FileNotFoundError(path)
|
||||
container = parse_container(path.read_bytes())
|
||||
self._sub_cache[key] = container.payloads
|
||||
self._sample_offset_cache[key] = container.sample_offsets
|
||||
return self._sub_cache[key]
|
||||
payload_hash = row.get("payload_hash", "")
|
||||
key = ("evolution", payload_hash)
|
||||
if key not in self._sub_cache:
|
||||
self._sub_cache[key], _ = unpack_evolution(self.payload(row), loose=True)
|
||||
self._sample_offset_cache[key] = {}
|
||||
return self._sub_cache[key]
|
||||
|
||||
def subpayload_sample_offset(self, row, payload_id):
|
||||
"""返回 EMDF payload 的外层 sample offset;旧 sidecar 没有该字段时为 0。"""
|
||||
self.subpayloads(row)
|
||||
emdf_hash = row.get("emdf_hash", "")
|
||||
key = (("emdf", emdf_hash) if emdf_hash else
|
||||
("evolution", row.get("payload_hash", "")))
|
||||
return int(self._sample_offset_cache.get(key, {}).get(payload_id, 0))
|
||||
|
||||
def __len__(self):
|
||||
return len(self.rows)
|
||||
|
||||
|
||||
def _frame_record(frame, parsed, state):
|
||||
present = [i for i, obj in enumerate(parsed["objs"]) if obj["present"]]
|
||||
sparse = [i for i in present if parsed["objs"][i]["sparse"]]
|
||||
bands = Counter(parsed["objs"][i]["n_bands"] for i in present)
|
||||
positions = []
|
||||
for obj in range(1, 16):
|
||||
q1, q2, q3 = (state.q[(obj, key)] for key in ("q1", "q2", "q3"))
|
||||
x, y, z = q_to_adm_xyz(q1, q2, q3)
|
||||
positions.append([round(float(x), 7), round(float(y), 7), round(float(z), 7)])
|
||||
return {
|
||||
"frame": frame,
|
||||
"sequence": parsed["seq_count_bits"],
|
||||
"downmix_config": parsed["dmx_config_idx"],
|
||||
"extension_config": parsed["ext_config_idx"],
|
||||
"objects_declared": parsed["n_objects"],
|
||||
"objects_present": present,
|
||||
"sparse_objects": sparse,
|
||||
"parameter_bands": {str(k): v for k, v in sorted(bands.items())},
|
||||
"clipgain": parsed["clipgain"],
|
||||
"oamd_xyz": positions,
|
||||
}
|
||||
|
||||
|
||||
def inspect(index, limit=None, print_frames=False):
|
||||
"""逐帧完整解析 ID11/ID14,并返回可 JSON 序列化的汇总。"""
|
||||
rows = index.rows if limit is None else index.rows[:limit]
|
||||
oamd = JocFieldState()
|
||||
frame_records = []
|
||||
configs = Counter()
|
||||
active = Counter()
|
||||
band_totals = Counter()
|
||||
sparse_counts = Counter()
|
||||
clipgains = []
|
||||
for seq, row in enumerate(rows):
|
||||
frame_number = int(row.get("frame", seq))
|
||||
try:
|
||||
subs = index.subpayloads(row)
|
||||
except UnsupportedVariantError as exc:
|
||||
exc.add_context(frame=frame_number)
|
||||
raise
|
||||
except Exception as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"metadata_container", "sidecar_or_emdf_parse",
|
||||
"元数据容器无法解析",
|
||||
frame=frame_number,
|
||||
details={"exception_type": type(exc).__name__, "parser_error": str(exc)}) from exc
|
||||
if 11 not in subs or 14 not in subs:
|
||||
raise UnsupportedVariantError(
|
||||
"emdf_payloads", "missing_required_payload",
|
||||
"EMDF 缺少 ID11/OAMD 或 ID14/JOC",
|
||||
frame=frame_number,
|
||||
details={
|
||||
"payload_ids": list(subs),
|
||||
"payload_lengths": {str(k): len(v) for k, v in subs.items()},
|
||||
})
|
||||
try:
|
||||
oamd.apply(frame_update_values(subs[11]))
|
||||
except UnsupportedVariantError as exc:
|
||||
exc.add_context(frame=frame_number, details={"payload_id": 11})
|
||||
raise
|
||||
except Exception as exc:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "payload_syntax",
|
||||
"OAMD 字段解析失败",
|
||||
frame=frame_number,
|
||||
details={
|
||||
"payload_id": 11,
|
||||
"payload": bytes_descriptor(subs[11]),
|
||||
"exception_type": type(exc).__name__,
|
||||
"parser_error": str(exc),
|
||||
}) from exc
|
||||
try:
|
||||
parsed = parse_joc(subs[14])
|
||||
except UnsupportedVariantError as exc:
|
||||
exc.add_context(frame=frame_number, details={"payload_id": 14})
|
||||
raise
|
||||
except Exception as exc:
|
||||
payload = subs[14]
|
||||
header = {}
|
||||
if len(payload) >= 4:
|
||||
value = int.from_bytes(payload[:4], "big")
|
||||
header = {
|
||||
"downmix_config": (value >> 29) & 7,
|
||||
"objects_minus_one": (value >> 23) & 63,
|
||||
"extension_config": (value >> 20) & 7,
|
||||
}
|
||||
raise UnsupportedVariantError(
|
||||
"joc", "payload_syntax",
|
||||
"JOC ID14 解析失败",
|
||||
frame=frame_number,
|
||||
details={
|
||||
"payload_id": 14,
|
||||
"header_probe": header,
|
||||
"payload": bytes_descriptor(payload),
|
||||
"exception_type": type(exc).__name__,
|
||||
"parser_error": str(exc),
|
||||
"repair_hint": "检查 JOC header、对象数、参数带、Huffman 或扩展字段",
|
||||
}) from exc
|
||||
sparse = [i for i, obj in enumerate(parsed["objs"]) if obj["present"] and obj["sparse"]]
|
||||
if sparse:
|
||||
raise UnsupportedVariantError(
|
||||
"joc", "sparse_joc",
|
||||
"发现尚未验证的 Sparse JOC 帧",
|
||||
frame=frame_number,
|
||||
details={
|
||||
"sparse_objects": sparse,
|
||||
"downmix_config": parsed["dmx_config_idx"],
|
||||
"extension_config": parsed["ext_config_idx"],
|
||||
"objects": parsed["n_objects"],
|
||||
"payload": bytes_descriptor(subs[14]),
|
||||
"repair_hint": "需要 Sparse JOC 实际样本及对应输出建立回归后再启用",
|
||||
})
|
||||
if parsed["n_channels"] != 5 or parsed["n_objects"] > 15:
|
||||
raise UnsupportedVariantError(
|
||||
"joc", "unsupported_configuration",
|
||||
"JOC 核心通道数或对象数超出当前渲染器范围",
|
||||
frame=frame_number,
|
||||
details={
|
||||
"downmix_config": parsed["dmx_config_idx"],
|
||||
"core_channels": parsed["n_channels"],
|
||||
"objects": parsed["n_objects"],
|
||||
"extension_config": parsed["ext_config_idx"],
|
||||
"payload": bytes_descriptor(subs[14]),
|
||||
})
|
||||
rec = _frame_record(frame_number, parsed, oamd)
|
||||
configs[(parsed["dmx_config_idx"], parsed["ext_config_idx"],
|
||||
parsed["n_channels"], parsed["n_objects"])] += 1
|
||||
active[len(rec["objects_present"])] += 1
|
||||
band_totals.update({int(k): v for k, v in rec["parameter_bands"].items()})
|
||||
sparse_counts[len(rec["sparse_objects"])] += 1
|
||||
clipgains.append(float(parsed["clipgain"]))
|
||||
if print_frames:
|
||||
print(json.dumps(rec, ensure_ascii=False, separators=(",", ":")))
|
||||
frame_records.append(rec)
|
||||
|
||||
summary = {
|
||||
"frames": len(rows),
|
||||
"frame_samples": 1536,
|
||||
"sample_rate": 48000,
|
||||
"duration_sec": len(rows) * 1536 / 48000,
|
||||
"configurations": [
|
||||
{"downmix_config": k[0], "extension_config": k[1], "core_channels": k[2],
|
||||
"objects": k[3], "frames": count}
|
||||
for k, count in sorted(configs.items())
|
||||
],
|
||||
"active_object_count_histogram": {str(k): v for k, v in sorted(active.items())},
|
||||
"parameter_band_totals": {str(k): v for k, v in sorted(band_totals.items())},
|
||||
"sparse_object_count_histogram": {str(k): v for k, v in sorted(sparse_counts.items())},
|
||||
"clipgain_min": min(clipgains),
|
||||
"clipgain_max": max(clipgains),
|
||||
"first_frame": frame_records[0],
|
||||
"last_frame": frame_records[-1],
|
||||
}
|
||||
emdf_rows = [row for row in rows if row.get("emdf_start_bit", "") != ""]
|
||||
if emdf_rows:
|
||||
starts = [int(row["emdf_start_bit"]) for row in emdf_rows]
|
||||
id_sets = Counter(row.get("payload_ids", "") for row in emdf_rows)
|
||||
summary["emdf_transport"] = {
|
||||
"continuous_containers": len(emdf_rows),
|
||||
"start_bit_min": min(starts),
|
||||
"start_bit_max": max(starts),
|
||||
"bit_alignment_histogram": {
|
||||
str(k): v for k, v in sorted(Counter(x & 7 for x in starts).items())
|
||||
},
|
||||
"payload_id_order_histogram": dict(sorted(id_sets.items())),
|
||||
}
|
||||
return summary
|
||||
|
||||
|
||||
def write_summary(index, output, limit=None, print_frames=False):
|
||||
summary = inspect(index, limit=limit, print_frames=print_frames)
|
||||
output = Path(output)
|
||||
output.write_text(json.dumps(summary, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
return summary
|
||||
@@ -0,0 +1,276 @@
|
||||
"""ctypes bridge for the dependency-free MSVC C++ JOC DSP core.
|
||||
|
||||
Metadata parsing intentionally stays in Python. One C call consumes a complete
|
||||
1536-sample frame, so Python is not involved in the hot 24-timeslot x 15-object
|
||||
DSP loops.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import ctypes
|
||||
import os
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
import numpy as np
|
||||
|
||||
from evo_unpack import unpack_evolution
|
||||
from joc_decode import dequantize, diff_decode, parse_joc
|
||||
|
||||
FRAME_SAMPLES = 1536
|
||||
MAX_OBJECTS = 15
|
||||
MAX_DPOINTS = 2
|
||||
CORE_CHANNELS = 5
|
||||
MAX_BANDS = 23
|
||||
ABI_VERSION = 1
|
||||
|
||||
|
||||
class NativeBackendUnavailable(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
def native_library_filename():
|
||||
if sys.platform == "win32":
|
||||
return "eac3joc_core.dll"
|
||||
if sys.platform == "darwin":
|
||||
return "libeac3joc_core.dylib"
|
||||
if sys.platform.startswith("linux"):
|
||||
return "libeac3joc_core.so"
|
||||
raise NativeBackendUnavailable(f"unsupported native platform: {sys.platform}")
|
||||
|
||||
|
||||
def _candidate_libraries():
|
||||
override = os.environ.get("EAC3JOC_NATIVE_LIBRARY")
|
||||
if override:
|
||||
yield Path(override).expanduser()
|
||||
root = Path(__file__).resolve().parent.parent
|
||||
yield root / "lib" / native_library_filename()
|
||||
|
||||
|
||||
def find_native_library(explicit=None):
|
||||
if explicit is not None:
|
||||
path = Path(explicit).expanduser().resolve()
|
||||
if not path.is_file():
|
||||
raise NativeBackendUnavailable(f"native library not found: {path}")
|
||||
return path
|
||||
checked = []
|
||||
for item in _candidate_libraries():
|
||||
path = item.resolve()
|
||||
checked.append(str(path))
|
||||
if path.is_file():
|
||||
return path
|
||||
raise NativeBackendUnavailable("native library not found; checked: " + "; ".join(checked))
|
||||
|
||||
|
||||
def default_native_threads():
|
||||
override = os.environ.get("EAC3JOC_NATIVE_THREADS")
|
||||
if override is not None:
|
||||
value = int(override)
|
||||
if value < 1:
|
||||
raise ValueError("EAC3JOC_NATIVE_THREADS must be at least 1")
|
||||
return min(value, MAX_OBJECTS)
|
||||
return 2 if (os.cpu_count() or 1) >= 4 else 1
|
||||
|
||||
def _load_library(path):
|
||||
lib = ctypes.CDLL(str(path))
|
||||
float_p = ctypes.POINTER(ctypes.c_float)
|
||||
u8_p = ctypes.POINTER(ctypes.c_uint8)
|
||||
double_p = ctypes.POINTER(ctypes.c_double)
|
||||
|
||||
lib.ejoc_abi_version.argtypes = []
|
||||
lib.ejoc_abi_version.restype = ctypes.c_uint32
|
||||
lib.ejoc_build_info.argtypes = []
|
||||
lib.ejoc_build_info.restype = ctypes.c_char_p
|
||||
lib.ejoc_renderer_create.argtypes = []
|
||||
lib.ejoc_renderer_create.restype = ctypes.c_void_p
|
||||
lib.ejoc_renderer_destroy.argtypes = [ctypes.c_void_p]
|
||||
lib.ejoc_renderer_destroy.restype = None
|
||||
lib.ejoc_renderer_reset.argtypes = [ctypes.c_void_p]
|
||||
lib.ejoc_renderer_reset.restype = ctypes.c_int
|
||||
lib.ejoc_renderer_set_threads.argtypes = [ctypes.c_void_p, ctypes.c_uint32]
|
||||
lib.ejoc_renderer_set_threads.restype = ctypes.c_int
|
||||
lib.ejoc_renderer_thread_count.argtypes = [ctypes.c_void_p]
|
||||
lib.ejoc_renderer_thread_count.restype = ctypes.c_uint32
|
||||
lib.ejoc_renderer_last_error.argtypes = [ctypes.c_void_p]
|
||||
lib.ejoc_renderer_last_error.restype = ctypes.c_char_p
|
||||
lib.ejoc_renderer_process.argtypes = [
|
||||
ctypes.c_void_p,
|
||||
float_p,
|
||||
float_p,
|
||||
ctypes.c_uint32,
|
||||
u8_p,
|
||||
u8_p,
|
||||
u8_p,
|
||||
u8_p,
|
||||
double_p,
|
||||
ctypes.c_double,
|
||||
ctypes.c_float,
|
||||
ctypes.c_float,
|
||||
float_p,
|
||||
]
|
||||
lib.ejoc_renderer_process.restype = ctypes.c_int
|
||||
abi = int(lib.ejoc_abi_version())
|
||||
if abi != ABI_VERSION:
|
||||
raise NativeBackendUnavailable(f"native ABI mismatch: library={abi}, Python={ABI_VERSION}")
|
||||
return lib
|
||||
|
||||
|
||||
class NativeJocRenderer:
|
||||
"""Stateful whole-frame native DSP renderer with the Python renderer API shape."""
|
||||
|
||||
def __init__(self, output_scale=1.0, library_path=None, threads=None):
|
||||
self.output_scale = np.float32(output_scale)
|
||||
if not np.isfinite(self.output_scale):
|
||||
raise ValueError("output_scale must be finite")
|
||||
self.library_path = find_native_library(library_path)
|
||||
self._lib = _load_library(self.library_path)
|
||||
self._handle = self._lib.ejoc_renderer_create()
|
||||
if not self._handle:
|
||||
raise MemoryError("ejoc_renderer_create failed")
|
||||
requested_threads = default_native_threads() if threads is None else int(threads)
|
||||
if requested_threads < 1:
|
||||
raise ValueError("threads must be at least 1")
|
||||
result = self._lib.ejoc_renderer_set_threads(self._handle, requested_threads)
|
||||
if result:
|
||||
self._raise_native("set_threads", result)
|
||||
self.threads = int(self._lib.ejoc_renderer_thread_count(self._handle))
|
||||
self._n_bands = np.zeros(MAX_OBJECTS, dtype=np.uint8)
|
||||
self._n_dpoints = np.zeros(MAX_OBJECTS, dtype=np.uint8)
|
||||
self._slope_idx = np.zeros(MAX_OBJECTS, dtype=np.uint8)
|
||||
self._offset_ts = np.zeros((MAX_OBJECTS, MAX_DPOINTS), dtype=np.uint8)
|
||||
self._dq = np.zeros(
|
||||
(MAX_OBJECTS, MAX_DPOINTS, CORE_CHANNELS, MAX_BANDS),
|
||||
dtype=np.float64,
|
||||
)
|
||||
self._output = np.zeros((16, FRAME_SAMPLES), dtype=np.float32)
|
||||
|
||||
@property
|
||||
def build_info(self):
|
||||
value = self._lib.ejoc_build_info()
|
||||
return value.decode("utf-8", "replace") if value else ""
|
||||
|
||||
@staticmethod
|
||||
def decode_payload(payload_bytes):
|
||||
subs, _ = unpack_evolution(payload_bytes, loose=True)
|
||||
return NativeJocRenderer.decode_subpayloads(subs)
|
||||
|
||||
@staticmethod
|
||||
def decode_subpayloads(subs):
|
||||
if 14 not in subs:
|
||||
raise ValueError("EMDF missing ID14/JOC")
|
||||
out = parse_joc(subs[14])
|
||||
mix_q = diff_decode(out)
|
||||
mix_dq = dequantize(out, mix_q)
|
||||
return out, mix_q, mix_dq
|
||||
|
||||
def reset(self):
|
||||
self._require_open()
|
||||
result = self._lib.ejoc_renderer_reset(self._handle)
|
||||
if result:
|
||||
self._raise_native("reset", result)
|
||||
|
||||
def close(self):
|
||||
handle = getattr(self, "_handle", None)
|
||||
if handle:
|
||||
self._lib.ejoc_renderer_destroy(handle)
|
||||
self._handle = None
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc, tb):
|
||||
self.close()
|
||||
|
||||
def __del__(self):
|
||||
try:
|
||||
self.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _require_open(self):
|
||||
if not self._handle:
|
||||
raise RuntimeError("native renderer is closed")
|
||||
|
||||
def _raise_native(self, operation, code):
|
||||
raw = self._lib.ejoc_renderer_last_error(self._handle)
|
||||
detail = raw.decode("utf-8", "replace") if raw else "unknown native error"
|
||||
raise RuntimeError(f"native {operation} failed ({code}): {detail}")
|
||||
|
||||
def _pack_frame(self, out, mix_dq):
|
||||
if out["n_channels"] != CORE_CHANNELS:
|
||||
raise ValueError(f"native core requires 5 JOC channels, got {out['n_channels']}")
|
||||
if out["n_objects"] > MAX_OBJECTS:
|
||||
raise ValueError(f"native core supports at most 15 objects, got {out['n_objects']}")
|
||||
self._n_bands.fill(0)
|
||||
self._n_dpoints.fill(0)
|
||||
self._slope_idx.fill(0)
|
||||
self._offset_ts.fill(0)
|
||||
mask = 0
|
||||
for object_index, info in enumerate(out["objs"]):
|
||||
if not info["present"]:
|
||||
continue
|
||||
if info["sparse"]:
|
||||
raise ValueError("native core does not accept unvalidated Sparse JOC")
|
||||
bands = int(info["n_bands"])
|
||||
points = int(info["n_dpoints"])
|
||||
if bands > MAX_BANDS or points > MAX_DPOINTS:
|
||||
raise ValueError(f"native descriptor out of range: bands={bands}, points={points}")
|
||||
values = np.asarray(mix_dq[object_index], dtype=np.float64)
|
||||
expected = (points, CORE_CHANNELS, bands)
|
||||
if values.shape != expected:
|
||||
raise ValueError(f"object {object_index} dq shape {values.shape}, expected {expected}")
|
||||
mask |= 1 << object_index
|
||||
self._n_bands[object_index] = bands
|
||||
self._n_dpoints[object_index] = points
|
||||
self._slope_idx[object_index] = int(info["slope_idx"])
|
||||
offsets = info.get("offset_ts", ())
|
||||
self._offset_ts[object_index, :len(offsets)] = offsets
|
||||
self._dq[object_index, :points, :, :bands] = values
|
||||
return mask
|
||||
|
||||
def render_frame(self, payload_bytes, bed5_pcm, lfe_pcm=None):
|
||||
subs, _ = unpack_evolution(payload_bytes, loose=True)
|
||||
return self.render_subpayloads(subs, bed5_pcm, lfe_pcm)
|
||||
|
||||
def render_subpayloads(self, subs, bed5_pcm, lfe_pcm=None):
|
||||
self._require_open()
|
||||
out, _, mix_dq = self.decode_subpayloads(subs)
|
||||
object_mask = self._pack_frame(out, mix_dq)
|
||||
bed5 = np.ascontiguousarray(bed5_pcm, dtype=np.float32)
|
||||
if bed5.shape != (CORE_CHANNELS, FRAME_SAMPLES):
|
||||
raise ValueError(f"core PCM shape must be (5,1536), got {bed5.shape}")
|
||||
if lfe_pcm is None:
|
||||
lfe = None
|
||||
lfe_ptr = ctypes.POINTER(ctypes.c_float)()
|
||||
else:
|
||||
lfe = np.ascontiguousarray(lfe_pcm, dtype=np.float32)
|
||||
if lfe.shape != (FRAME_SAMPLES,):
|
||||
raise ValueError(f"LFE shape must be (1536,), got {lfe.shape}")
|
||||
lfe_ptr = lfe.ctypes.data_as(ctypes.POINTER(ctypes.c_float))
|
||||
|
||||
result = self._lib.ejoc_renderer_process(
|
||||
self._handle,
|
||||
bed5.ctypes.data_as(ctypes.POINTER(ctypes.c_float)),
|
||||
lfe_ptr,
|
||||
object_mask,
|
||||
self._n_bands.ctypes.data_as(ctypes.POINTER(ctypes.c_uint8)),
|
||||
self._n_dpoints.ctypes.data_as(ctypes.POINTER(ctypes.c_uint8)),
|
||||
self._slope_idx.ctypes.data_as(ctypes.POINTER(ctypes.c_uint8)),
|
||||
self._offset_ts.ctypes.data_as(ctypes.POINTER(ctypes.c_uint8)),
|
||||
self._dq.ctypes.data_as(ctypes.POINTER(ctypes.c_double)),
|
||||
float(out["clipgain"]),
|
||||
ctypes.c_float(0.0625),
|
||||
ctypes.c_float(self.output_scale),
|
||||
self._output.ctypes.data_as(ctypes.POINTER(ctypes.c_float)),
|
||||
)
|
||||
if result:
|
||||
self._raise_native("process", result)
|
||||
return self._output, None
|
||||
|
||||
|
||||
def native_available(library_path=None):
|
||||
try:
|
||||
path = find_native_library(library_path)
|
||||
lib = _load_library(path)
|
||||
return True, str(path), (lib.ejoc_build_info() or b"").decode("utf-8", "replace")
|
||||
except Exception as exc:
|
||||
return False, None, str(exc)
|
||||
@@ -0,0 +1,216 @@
|
||||
"""OAMD 位载荷 → 16 个对象槽的 q1/q2/q3 增量状态。
|
||||
|
||||
槽 0 是 bed/LFE;槽 1..15 对应输出 ch1..15 的对象元数据。
|
||||
"""
|
||||
import numpy as np
|
||||
|
||||
from variant_error import UnsupportedVariantError, bytes_descriptor
|
||||
|
||||
Q1_OFF = 192
|
||||
N_Q12 = 62
|
||||
N_Q3 = 15
|
||||
SAMPLE_OFFSET_INDEX = (8, 16, 18, 24)
|
||||
RAMP_DURATIONS = (0, 512, 1536)
|
||||
RAMP_DURATION_INDEX = (
|
||||
32, 64, 128, 256, 320, 480, 1000, 1001,
|
||||
1024, 1600, 1601, 1602, 1920, 2000, 2002, 2048,
|
||||
)
|
||||
|
||||
|
||||
def q_of(k, n):
|
||||
q = int(np.floor(32768.0 * k / n + 0.5))
|
||||
return min(32767, q)
|
||||
|
||||
|
||||
def _payload_bits(bits_one):
|
||||
if isinstance(bits_one, (bytes, bytearray, memoryview)):
|
||||
src = np.frombuffer(bits_one, dtype=np.uint8)
|
||||
else:
|
||||
src = np.asarray(bits_one, dtype=np.uint8)
|
||||
if src.ndim == 1 and src.shape[0] in (536, 552):
|
||||
bits = src
|
||||
elif src.size in (67, 69):
|
||||
bits = np.unpackbits(src.reshape(-1), bitorder="big")
|
||||
else:
|
||||
raw = (bytes(bits_one) if isinstance(bits_one, (bytes, bytearray, memoryview))
|
||||
else np.asarray(bits_one, dtype=np.uint8).tobytes())
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", f"payload_length_{len(raw)}B",
|
||||
f"发现未覆盖的 OAMD 载荷长度 {len(raw)}B",
|
||||
details={
|
||||
"supported_payload_bytes": [67, 69],
|
||||
"payload": bytes_descriptor(raw),
|
||||
"repair_hint": "检查 OAMD header、element 数量及可选字段造成的位偏移变化",
|
||||
})
|
||||
raw_payload = np.packbits(bits, bitorder="big").tobytes()
|
||||
return bits, raw_payload
|
||||
|
||||
|
||||
class _BitReader:
|
||||
def __init__(self, bits):
|
||||
self.bits = bits
|
||||
self.position = 0
|
||||
|
||||
def read(self, count):
|
||||
end = self.position + count
|
||||
if end > len(self.bits):
|
||||
raise ValueError(f"OAMD 位流越界: bit={self.position}, need={count}")
|
||||
value = 0
|
||||
for bit in self.bits[self.position:end]:
|
||||
value = (value << 1) | int(bit)
|
||||
self.position = end
|
||||
return value
|
||||
|
||||
def skip(self, count):
|
||||
self.read(count)
|
||||
|
||||
|
||||
def _variable_bits(reader, width, max_groups=5):
|
||||
value = 0
|
||||
for _ in range(max_groups + 1):
|
||||
value += reader.read(width)
|
||||
if not reader.read(1):
|
||||
return value
|
||||
value = (value + 1) << width
|
||||
raise ValueError(f"OAMD variable_bits({width}) 延伸组过多")
|
||||
|
||||
|
||||
def _update_timing(bits, alternate_object_present, element_count, raw_payload):
|
||||
"""按 OAMD element/MDUpdateInfo 读取位置块的开始偏移和 ramp 时长。"""
|
||||
reader = _BitReader(bits)
|
||||
reader.position = 14
|
||||
for _ in range(element_count):
|
||||
element_index = reader.read(4)
|
||||
element_length = _variable_bits(reader, 4)
|
||||
element_end = reader.position + element_length + 1
|
||||
reader.skip(5 if alternate_object_present else 1)
|
||||
if element_index == 1:
|
||||
offset_code = reader.read(2)
|
||||
if offset_code == 0:
|
||||
sample_offset = 0
|
||||
elif offset_code == 1:
|
||||
sample_offset = SAMPLE_OFFSET_INDEX[reader.read(2)]
|
||||
elif offset_code == 2:
|
||||
sample_offset = reader.read(5)
|
||||
else:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "md_sample_offset_mode",
|
||||
"OAMD 使用了当前未覆盖的 MD sample-offset 模式",
|
||||
details={"payload": bytes_descriptor(raw_payload)})
|
||||
|
||||
block_count = reader.read(3) + 1
|
||||
blocks = []
|
||||
for _block in range(block_count):
|
||||
block_offset_factor = reader.read(6)
|
||||
block_offset = sample_offset + block_offset_factor * 32
|
||||
ramp_code = reader.read(2)
|
||||
if ramp_code == 3:
|
||||
if reader.read(1):
|
||||
ramp_duration = RAMP_DURATION_INDEX[reader.read(4)]
|
||||
else:
|
||||
ramp_duration = reader.read(11)
|
||||
else:
|
||||
ramp_duration = RAMP_DURATIONS[ramp_code]
|
||||
blocks.append((block_offset, ramp_duration))
|
||||
if len(blocks) != 1:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "multiple_position_blocks",
|
||||
"OAMD 一帧含多个对象位置更新块,固定位置窗口不能安全套用",
|
||||
details={
|
||||
"block_count": len(blocks),
|
||||
"blocks": blocks,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "按 ObjectInfoBlock 顺序逐块解析坐标,再生成分段 ADM ramp",
|
||||
})
|
||||
return blocks[0]
|
||||
reader.position = element_end
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "missing_object_element",
|
||||
"OAMD 中没有 object element (element_index=1)",
|
||||
details={"payload": bytes_descriptor(raw_payload)})
|
||||
|
||||
|
||||
def frame_update(bits_one):
|
||||
"""单帧 OAMD → 位置字段增量及其 sample offset/ramp duration。"""
|
||||
bits, raw_payload = _payload_bits(bits_one)
|
||||
header = {
|
||||
"version": int((bits[0] << 1) | bits[1]),
|
||||
"objects_minus_one": int(sum(int(bits[2 + i]) << (4 - i) for i in range(5))),
|
||||
"dynamic_object_only": int(bits[7]),
|
||||
"lfe_present": int(bits[8]),
|
||||
"alternate_object_present": int(bits[9]),
|
||||
"element_count": int(sum(int(bits[10 + i]) << (3 - i) for i in range(4))),
|
||||
}
|
||||
if raw_payload[:2] != b"\x1f\x88":
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", f"header_signature_{raw_payload[:2].hex()}",
|
||||
"OAMD 长度已知,但 header 与当前位置字段布局不一致",
|
||||
details={
|
||||
"supported_header_prefix_hex": "1f88",
|
||||
"header_probe": header,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "按新 header 的 program assignment 和 element 布局重新定位对象位置字段",
|
||||
})
|
||||
block_offset, ramp_duration = _update_timing(
|
||||
bits, bool(header["alternate_object_present"]), header["element_count"], raw_payload)
|
||||
weights = 1 << np.arange(7, -1, -1)
|
||||
out = {}
|
||||
for obj in range(16):
|
||||
start = 112 + 31 * (obj - 3)
|
||||
wq1 = int((bits[start:start + 8] * weights).sum())
|
||||
wq2 = int((bits[start + 8:start + 16] * weights).sum())
|
||||
wq3 = int((bits[start + 16:start + 24] * weights).sum())
|
||||
if obj and (wq1 >> 6 != 3 or wq2 & 2 != 2 or wq3 & 0x1F != 1):
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "position_layout_signature",
|
||||
"OAMD 长度和 header 已知,但对象位置字段标记或位偏移发生变化",
|
||||
details={
|
||||
"object_slot": obj,
|
||||
"position_start_bit": start,
|
||||
"q1_window_hex": f"{wq1:02x}",
|
||||
"q2_window_hex": f"{wq2:02x}",
|
||||
"q3_window_hex": f"{wq3:02x}",
|
||||
"header_probe": header,
|
||||
"payload": bytes_descriptor(raw_payload),
|
||||
"repair_hint": "解析 OAMD element 可选字段并更新每个对象的位置窗口偏移",
|
||||
})
|
||||
k1 = wq1 - Q1_OFF
|
||||
if obj == 0:
|
||||
out[(0, "q1")] = None
|
||||
out[(0, "q2")] = None
|
||||
else:
|
||||
out[(obj, "q1")] = q_of(k1, N_Q12) if 0 <= k1 <= N_Q12 else None
|
||||
k2 = wq2 >> 2
|
||||
if obj != 0:
|
||||
out[(obj, "q2")] = q_of(k2, N_Q12) if 0 <= k2 <= N_Q12 else None
|
||||
k3 = (wq2 & 1) * 8 + (wq3 >> 5)
|
||||
out[(obj, "q3")] = q_of(k3, N_Q3) if 0 <= k3 <= N_Q3 else None
|
||||
return {
|
||||
"values": out,
|
||||
"block_offset_samples": block_offset,
|
||||
"ramp_duration_samples": ramp_duration,
|
||||
}
|
||||
|
||||
|
||||
def frame_update_values(bits_one):
|
||||
"""兼容接口:只返回 ``{(slot, field): q|None}``。"""
|
||||
return frame_update(bits_one)["values"]
|
||||
|
||||
|
||||
class JocFieldState:
|
||||
"""未更新/非法窗保持旧值;slot0 q1/q2 的 DLL 初值为中心 16384。"""
|
||||
|
||||
def __init__(self):
|
||||
self.q = {(obj, field): 0 for obj in range(16)
|
||||
for field in ("q1", "q2", "q3")}
|
||||
self.q[(0, "q1")] = 16384
|
||||
self.q[(0, "q2")] = 16384
|
||||
|
||||
def apply(self, updates):
|
||||
for key, value in updates.items():
|
||||
if value is not None:
|
||||
self.q[key] = value
|
||||
return self
|
||||
|
||||
def snapshot(self):
|
||||
return dict(self.q)
|
||||
@@ -0,0 +1,227 @@
|
||||
"""Evolution id11/OAMD 帧序列 → ADM 15 对象关键帧。"""
|
||||
import math
|
||||
|
||||
from adm_atmos import q_to_adm_xyz
|
||||
from oamd_bits import JocFieldState, frame_update
|
||||
from variant_error import UnsupportedVariantError
|
||||
|
||||
|
||||
def _lerp_xyz(start, target, amount):
|
||||
return tuple(a + (b - a) * amount for a, b in zip(start, target))
|
||||
|
||||
|
||||
def _append_point(points, sample, xyz, interpolation_samples):
|
||||
item = (int(sample), *map(float, xyz), int(interpolation_samples))
|
||||
if points and item[0] == points[-1][0]:
|
||||
points[-1] = item
|
||||
elif not points or item[0] > points[-1][0]:
|
||||
points.append(item)
|
||||
|
||||
|
||||
def _expand_events_dense64(events, total_samples, rate, update_quantum_samples,
|
||||
object_delay_samples, object_index):
|
||||
if not events:
|
||||
return [(0.0, 0.0, 0.0, 0.0, total_samples / float(rate), 0.0)]
|
||||
|
||||
# 初始位置从成品 sample 0 起有效;合成延迟只作用于后续位置变化。
|
||||
current = events[0][1]
|
||||
points = []
|
||||
_append_point(points, 0, current, 0)
|
||||
|
||||
for event_index, (coded_start, target, ramp_samples) in enumerate(events[1:], 1):
|
||||
start = coded_start + object_delay_samples
|
||||
if start >= total_samples:
|
||||
break
|
||||
if start < points[-1][0]:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "non_monotonic_position_updates",
|
||||
"对象位置更新时间倒退,无法生成连续 ADM 轨迹",
|
||||
details={"object": object_index, "sample": start,
|
||||
"previous_sample": points[-1][0]})
|
||||
|
||||
# 在运动起点保留上一位置,避免下游把长时间静止段直接连到首个中间点。
|
||||
if start > points[-1][0]:
|
||||
_append_point(points, start, current, 0)
|
||||
|
||||
effective_ramp = max(0, int(ramp_samples) - update_quantum_samples)
|
||||
if effective_ramp == 0:
|
||||
_append_point(points, start, target, 0)
|
||||
current = target
|
||||
continue
|
||||
|
||||
end = start + math.ceil(effective_ramp / update_quantum_samples) * update_quantum_samples
|
||||
if event_index + 1 < len(events):
|
||||
next_start = events[event_index + 1][0] + object_delay_samples
|
||||
if next_start < end:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "overlapping_position_ramps",
|
||||
"同一对象的新位置更新在上一 ramp 完成前到达",
|
||||
details={
|
||||
"object": object_index,
|
||||
"ramp_start_sample": start,
|
||||
"ramp_end_sample": end,
|
||||
"next_update_sample": next_start,
|
||||
"repair_hint": "按 64-sample 状态机截断旧 ramp,再从当前插值位置启动新 ramp",
|
||||
})
|
||||
|
||||
# 逐位置更新节拍复现状态机。1536-sample ramp 在首次 64-sample
|
||||
# 更新后剩余 1472 samples,因此共有 23 个中间/终点坐标。
|
||||
future = effective_ramp
|
||||
elapsed = 0
|
||||
position = current
|
||||
while future > 0:
|
||||
amount = min(update_quantum_samples / float(future), 1.0)
|
||||
position = _lerp_xyz(position, target, amount)
|
||||
elapsed += update_quantum_samples
|
||||
sample = start + elapsed
|
||||
if sample >= total_samples:
|
||||
break
|
||||
_append_point(points, sample, position, update_quantum_samples)
|
||||
future -= update_quantum_samples
|
||||
current = target
|
||||
|
||||
blocks = []
|
||||
for point_index, (sample, x, y, z, interpolation_samples) in enumerate(points):
|
||||
end = points[point_index + 1][0] if point_index + 1 < len(points) else total_samples
|
||||
duration_samples = max(0, end - sample)
|
||||
if duration_samples == 0:
|
||||
continue
|
||||
interpolation_samples = min(interpolation_samples, duration_samples)
|
||||
blocks.append((sample / float(rate), x, y, z,
|
||||
duration_samples / float(rate),
|
||||
interpolation_samples / float(rate)))
|
||||
return blocks
|
||||
|
||||
|
||||
|
||||
def _compact_events(events, total_samples, rate, update_quantum_samples,
|
||||
object_delay_samples, object_index):
|
||||
"""Represent each linear OAMD ramp with one ADM interpolation block.
|
||||
|
||||
The existing dense64 representation keeps the old position at ``start``,
|
||||
writes its first interpolated target at ``start + quantum``, and lets ADM
|
||||
interpolate that block over one quantum. Consequently, the interpreted
|
||||
motion begins at ``start + quantum`` and reaches the final target at
|
||||
``start + ramp_duration``. This compact form preserves that timing with one
|
||||
target block whose interpolationLength is ``ramp_duration - quantum``.
|
||||
"""
|
||||
if not events:
|
||||
return [(0.0, 0.0, 0.0, 0.0, total_samples / float(rate), 0.0)]
|
||||
|
||||
points = []
|
||||
current = events[0][1]
|
||||
_append_point(points, 0, current, 0)
|
||||
|
||||
for event_index, (coded_start, target, ramp_samples) in enumerate(events[1:], 1):
|
||||
event_start = coded_start + object_delay_samples
|
||||
if event_start >= total_samples:
|
||||
break
|
||||
effective_ramp = max(0, int(ramp_samples) - update_quantum_samples)
|
||||
block_start = event_start + (update_quantum_samples if effective_ramp else 0)
|
||||
if block_start >= total_samples:
|
||||
break
|
||||
ramp_end = block_start + effective_ramp
|
||||
|
||||
if block_start < points[-1][0]:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "non_monotonic_compact_position_updates",
|
||||
"紧凑对象位置更新时间倒退",
|
||||
details={"object": object_index, "sample": block_start,
|
||||
"previous_sample": points[-1][0]})
|
||||
|
||||
if event_index + 1 < len(events):
|
||||
next_coded_start, _, next_ramp_samples = events[event_index + 1]
|
||||
next_event_start = next_coded_start + object_delay_samples
|
||||
next_effective = max(0, int(next_ramp_samples) - update_quantum_samples)
|
||||
next_block_start = next_event_start + (update_quantum_samples if next_effective else 0)
|
||||
if next_block_start < ramp_end:
|
||||
raise UnsupportedVariantError(
|
||||
"oamd", "overlapping_compact_position_ramps",
|
||||
"同一对象的新位置更新在上一紧凑 ramp 完成前到达",
|
||||
details={
|
||||
"object": object_index,
|
||||
"ramp_start_sample": block_start,
|
||||
"ramp_end_sample": ramp_end,
|
||||
"next_update_sample": next_block_start,
|
||||
"repair_hint": "对此变体使用 --trajectory-mode dense64 并检查 OAMD 调度",
|
||||
})
|
||||
|
||||
block_target = target
|
||||
block_interpolation = effective_ramp
|
||||
available = total_samples - block_start
|
||||
if effective_ramp > available:
|
||||
block_target = _lerp_xyz(current, target, available / float(effective_ramp))
|
||||
block_interpolation = available
|
||||
_append_point(points, block_start, block_target, block_interpolation)
|
||||
current = target
|
||||
|
||||
blocks = []
|
||||
for point_index, (sample, x, y, z, interpolation_samples) in enumerate(points):
|
||||
end = points[point_index + 1][0] if point_index + 1 < len(points) else total_samples
|
||||
duration_samples = max(0, end - sample)
|
||||
if duration_samples == 0:
|
||||
continue
|
||||
interpolation_samples = min(interpolation_samples, duration_samples)
|
||||
blocks.append((sample / float(rate), x, y, z,
|
||||
duration_samples / float(rate),
|
||||
interpolation_samples / float(rate)))
|
||||
return blocks
|
||||
|
||||
|
||||
def _expand_events(events, total_samples, rate, update_quantum_samples,
|
||||
object_delay_samples, object_index, trajectory_mode):
|
||||
if trajectory_mode == "compact":
|
||||
return _compact_events(events, total_samples, rate, update_quantum_samples,
|
||||
object_delay_samples, object_index)
|
||||
if trajectory_mode == "dense64":
|
||||
return _expand_events_dense64(events, total_samples, rate, update_quantum_samples,
|
||||
object_delay_samples, object_index)
|
||||
raise ValueError(f"未知 trajectory_mode: {trajectory_mode}")
|
||||
|
||||
def build_adm_tracks(index, frames=None, rate=48000, frame_samples=1536,
|
||||
update_quantum_samples=64, object_delay_samples=640,
|
||||
trajectory_mode="compact"):
|
||||
"""从统一 metadata index 构造 15 条 ADM 轨迹。
|
||||
|
||||
返回 ``[(name, [(rtime,x,y,z,duration,interpolation), ...]), ...]``。
|
||||
OAMD 的内外层 sample offset、block offset 和 ramp 均保留。
|
||||
``trajectory_mode="compact"`` 用一个长 ADM interpolation block 表示每条
|
||||
线性 ramp;``dense64`` 保留逐 64-sample 展开作为兼容回退。
|
||||
``object_delay_samples`` 将位置更新与对象逆 QMF 的输出时刻对齐。
|
||||
slot1..15 与对象 PCM ch1..15 一一对应。
|
||||
"""
|
||||
frames = index.rows if frames is None else frames
|
||||
state = JocFieldState()
|
||||
events = [[] for _ in range(15)]
|
||||
previous = [None] * 15
|
||||
|
||||
for seq, row in enumerate(frames):
|
||||
subs = index.subpayloads(row)
|
||||
timing = None
|
||||
if 11 in subs:
|
||||
timing = frame_update(subs[11])
|
||||
state.apply(timing["values"])
|
||||
|
||||
event_sample = seq * frame_samples
|
||||
ramp_samples = 0
|
||||
if timing is not None:
|
||||
outer_offset = (index.subpayload_sample_offset(row, 11)
|
||||
if hasattr(index, "subpayload_sample_offset") else 0)
|
||||
event_sample += outer_offset + timing["block_offset_samples"]
|
||||
ramp_samples = timing["ramp_duration_samples"]
|
||||
|
||||
q = state.q
|
||||
for obj in range(1, 16):
|
||||
xyz = q_to_adm_xyz(q[(obj, "q1")], q[(obj, "q2")], q[(obj, "q3")])
|
||||
if previous[obj - 1] != xyz:
|
||||
events[obj - 1].append((event_sample, xyz, ramp_samples))
|
||||
previous[obj - 1] = xyz
|
||||
|
||||
total_samples = len(frames) * frame_samples
|
||||
return [
|
||||
(f"JOC_Object_{obj}",
|
||||
_expand_events(events[obj - 1], total_samples, rate,
|
||||
update_quantum_samples, object_delay_samples, obj,
|
||||
trajectory_mode))
|
||||
for obj in range(1, 16)
|
||||
]
|
||||
+259
@@ -0,0 +1,259 @@
|
||||
"""把 E-AC-3 核心 PCM 和 JOC 元数据重建为 LFE 加 15 路对象 PCM。
|
||||
|
||||
管线(每帧):
|
||||
核心 5.1 PCM → QMF 分析 x → 填充器 z = Σ m·x(每对象)
|
||||
→ 对象时域组装(前步、64 点 FFT、旋转、合成窗)→ ×16 + clamp
|
||||
→ 对象波形 × joc_clipgain
|
||||
+ LFE 专用 1217-sample 环形延迟
|
||||
→ 16ch(ch0 = LFE, ch1-15 = 15 对象)→ ×output_scale
|
||||
|
||||
output_scale:
|
||||
默认 1.0(0 dB)。用户选择的 dB 在命令行转换为 float32 系数;该系数
|
||||
不进入 JOC 数学本体,也不是 joc_clipgain。
|
||||
"""
|
||||
import numpy as np
|
||||
|
||||
from joc_decode import parse_joc, diff_decode, dequantize
|
||||
from joc_qmf import (N, QMF5_WINDOW, qmf_analysis_frame,
|
||||
surround_post_frame, interp_matrix)
|
||||
from evo_unpack import unpack_evolution
|
||||
|
||||
CORE_CHANNELS = [0, 1, 2, 4, 5] # L R C Ls Rs(EAC3 5.1 核心顺序)
|
||||
|
||||
|
||||
class JocRenderer:
|
||||
"""E-AC-3 JOC 帧渲染器。状态跨帧保持(FIFO/插值 prev/合成状态)。"""
|
||||
|
||||
def __init__(self, output_scale=1.0):
|
||||
# 成品增益明确按 float32 运算;默认值为 1.0。
|
||||
self.output_scale = np.float32(output_scale)
|
||||
self._prev = {} # 每对象插值 prev(m 的帧末 sb 值)
|
||||
self._synthesis_state = np.zeros((15, 640), dtype=np.float64)
|
||||
# 分析 QMF:五路各自保留 9×64 FIFO;L/R/C 另有 10-timeslot 延迟。
|
||||
self._analysis_fifo = np.zeros((5, 9, 64), dtype=np.float64)
|
||||
self._analysis_delay = np.zeros((3, 10, 64), dtype=np.float32)
|
||||
self._analysis_phase = np.float32(0.0625)
|
||||
self._surround_qmf_delay = np.zeros((2, 10, 64), dtype=np.complex128)
|
||||
self._surround_dc_hist = np.zeros((2, 20), dtype=np.complex128)
|
||||
# ch0/LFE 使用循环历史,读指针恒落后写指针 1217 个采样。
|
||||
# phase=1/16 与输出端 *16 抵消,净效果是 LFE 延迟。
|
||||
self._lfe_delay = np.zeros(1217, dtype=np.float64)
|
||||
self._last_x = None # 仅供逐槽诊断读取,不参与状态推进
|
||||
self._rot = None # 每带旋转表
|
||||
|
||||
@staticmethod
|
||||
def decode_payload(payload_bytes):
|
||||
"""id14 载荷 → (parse 结果, mix_q, mix_dq)。"""
|
||||
subs, _ = unpack_evolution(payload_bytes, loose=True)
|
||||
out = parse_joc(subs[14])
|
||||
mix_q = diff_decode(out)
|
||||
mix_dq = dequantize(out, mix_q)
|
||||
return out, mix_q, mix_dq
|
||||
|
||||
@staticmethod
|
||||
def decode_subpayloads(subs):
|
||||
"""已解析 EMDF 子载荷 → (parse 结果, mix_q, mix_dq)。"""
|
||||
if 14 not in subs:
|
||||
raise ValueError("EMDF 缺少 ID14/JOC")
|
||||
out = parse_joc(subs[14])
|
||||
mix_q = diff_decode(out)
|
||||
mix_dq = dequantize(out, mix_q)
|
||||
return out, mix_q, mix_dq
|
||||
|
||||
def qmf_x(self, bed5, phase_new=0.0625):
|
||||
"""核心 5ch PCM → 对象矩阵使用的复数 QMF ``x``。
|
||||
|
||||
先以 float32 对当前帧应用 phase;phase 变化时仅前 256 个样本从旧值
|
||||
线性过渡。缩放后的 L/R/C 延迟 10 槽,Ls/Rs 不延迟,再进入分析 QMF。
|
||||
"""
|
||||
pcm = np.asarray(bed5, dtype=np.float32)
|
||||
if pcm.shape != (5, 1536):
|
||||
raise ValueError(f"核心 5ch 帧应为 (5,1536),实际 {pcm.shape}")
|
||||
new_phase = np.float32(phase_new)
|
||||
old_phase = self._analysis_phase
|
||||
gains = np.full(1536, new_phase, dtype=np.float32)
|
||||
if old_phase != new_phase:
|
||||
step = np.float32((new_phase - old_phase) / np.float32(256.0))
|
||||
gains[:256] = old_phase + np.arange(256, dtype=np.float32) * step
|
||||
scaled = np.multiply(pcm, gains[None, :], dtype=np.float32).reshape(5, 24, 64)
|
||||
|
||||
blocks = np.empty_like(scaled)
|
||||
for ch in range(3):
|
||||
delayed = np.concatenate((self._analysis_delay[ch], scaled[ch]), axis=0)
|
||||
blocks[ch] = delayed[:24]
|
||||
self._analysis_delay[ch] = delayed[24:]
|
||||
blocks[3:] = scaled[3:]
|
||||
|
||||
x, self._analysis_fifo = qmf_analysis_frame(self._analysis_fifo, blocks)
|
||||
self._analysis_phase = new_phase
|
||||
x[3:], self._surround_qmf_delay, self._surround_dc_hist = surround_post_frame(
|
||||
x[3:], self._surround_qmf_delay, self._surround_dc_hist)
|
||||
return x
|
||||
|
||||
def object_z(self, out, mix_dq, x):
|
||||
"""计算 ``z[obj] = Σ_ch m[ch,sb,ts]·x[ch,sb,ts]``。"""
|
||||
z_all = {}
|
||||
new_prev = {}
|
||||
for obj, o in enumerate(out["objs"]):
|
||||
if not o["present"]:
|
||||
continue
|
||||
dq = mix_dq[obj]
|
||||
prev = self._prev.get(obj)
|
||||
if prev is None:
|
||||
prev = np.zeros((out["n_channels"], N), dtype=np.float64)
|
||||
m = interp_matrix(o, dq, prev) # [ch][sb][ts](内部按 o["n_bands"] 映射)
|
||||
z = np.sum(x * m, axis=0) # [sb][ts](复)
|
||||
z_all[obj] = z
|
||||
new_prev[obj] = m[:, :, -1]
|
||||
self._prev.update(new_prev)
|
||||
return z_all
|
||||
|
||||
def _rot_table_86840(self):
|
||||
"""生成 ``θ=πk/128`` 的 ``0.5·(sin θ, cos θ)`` 旋转表。"""
|
||||
if self._rot is None:
|
||||
k = np.arange(64)
|
||||
theta = np.pi * k / 128.0
|
||||
self._rot = np.empty(128, dtype=np.float64)
|
||||
self._rot[0::2] = 0.5 * np.sin(theta) # sin 分量
|
||||
self._rot[1::2] = 0.5 * np.cos(theta) # cos 分量
|
||||
return self._rot
|
||||
|
||||
@staticmethod
|
||||
def _front_step(src):
|
||||
"""重排 128 个交织复数分量:
|
||||
outA[2k] = src[4k]、outA[2k+1] = −src[4k+1](区 [0:64]);
|
||||
outB[126−2k] = src[4k+2]、outB[127−2k] = src[4k+3](区 [64:128] 倒序)。"""
|
||||
src = np.asarray(src, dtype=np.float64)
|
||||
zone = np.empty_like(src)
|
||||
k = np.arange(32)
|
||||
zone[..., 2 * k] = src[..., 4 * k]
|
||||
zone[..., 2 * k + 1] = -src[..., 4 * k + 1]
|
||||
zone[..., 126 - 2 * k] = src[..., 4 * k + 2]
|
||||
zone[..., 127 - 2 * k] = src[..., 4 * k + 3]
|
||||
return zone
|
||||
|
||||
@staticmethod
|
||||
def _h880_vec(a2):
|
||||
"""对 128 个 re/im 交织值执行标准 64 点复数 FFT。"""
|
||||
a2 = np.asarray(a2, dtype=np.float64)
|
||||
zc = a2[..., 0::2] + 1j * a2[..., 1::2]
|
||||
f = np.fft.fft(zc, axis=-1)
|
||||
a1 = np.empty_like(a2)
|
||||
a1[..., 0::2] = f.real
|
||||
a1[..., 1::2] = f.imag
|
||||
return a1
|
||||
|
||||
@staticmethod
|
||||
def _h0b0_vec(a2, a3):
|
||||
"""NumPy 向量化的 ``out = 2·复乘(a2, a3)``,re/im 交织。"""
|
||||
a2 = np.asarray(a2, dtype=np.float64)
|
||||
a3 = np.asarray(a3, dtype=np.float64)
|
||||
shape = a2.shape[:-1] + (16, 4)
|
||||
v11 = a2[..., 1::2].reshape(shape)
|
||||
v12 = a2[..., 0::2].reshape(shape)
|
||||
v13 = a3[..., 0::2].reshape(a3.shape[:-1] + (16, 4))
|
||||
v14 = a3[..., 1::2].reshape(a3.shape[:-1] + (16, 4))
|
||||
v15 = (v11 * v14 - v12 * v13) * 2.0
|
||||
v16 = (v12 * v14 + v11 * v13) * 2.0
|
||||
out = np.empty_like(a2)
|
||||
out[..., 0::2] = v16.reshape(a2.shape[:-1] + (64,))
|
||||
out[..., 1::2] = v15.reshape(a2.shape[:-1] + (64,))
|
||||
return out
|
||||
|
||||
@staticmethod
|
||||
def _qmf5_vec(state, win_flat, rot):
|
||||
"""推进合成窗状态并返回 64 个时域样本。"""
|
||||
state = np.asarray(state, dtype=np.float64)
|
||||
rot = np.asarray(rot, dtype=np.float64)
|
||||
single = state.ndim == 1
|
||||
if single:
|
||||
state = state[None, :]
|
||||
if rot.ndim == 1:
|
||||
rot = np.broadcast_to(rot, (len(state), len(rot)))
|
||||
rot2 = rot.reshape(len(state), 16, 8)
|
||||
v22 = rot2[..., 0:8:2]
|
||||
v19 = rot2[..., 1:8:2]
|
||||
S = state[:, :576].reshape(len(state), 16, 9, 4)
|
||||
w10 = win_flat[:640].reshape(10, 64)
|
||||
idx = np.arange(16)[:, None] * 4 + np.arange(4)
|
||||
w0 = w10[0, idx][None, ...]
|
||||
w_all = w10[1:10, idx].transpose(1, 0, 2)[None, ...]
|
||||
out64 = (2.0 * (w0 * v22 + S[:, :, 0])).reshape(len(state), 64)
|
||||
# 输出使用窗行 0,状态第 0 行从窗行 1 开始推进。
|
||||
S[:, :, 0] = w_all[:, :, 0] * v19 + S[:, :, 1]
|
||||
for k in range(7):
|
||||
alt = v22 if k % 2 == 0 else v19
|
||||
S[:, :, k + 1] = w_all[:, :, k + 1] * alt + S[:, :, k + 2]
|
||||
S[:, :, 8] = w_all[:, :, 8] * v19
|
||||
return out64[0] if single else out64
|
||||
|
||||
def dll_synth_objects(self, z_all):
|
||||
"""批量执行对象逆 QMF,返回对象 PCM 字典。
|
||||
|
||||
每个对象保持独立合成状态,64 点 FFT 和窗核在对象维批量计算。
|
||||
"""
|
||||
object_ids = sorted(z_all)
|
||||
if not object_ids:
|
||||
return {}
|
||||
z = np.stack([z_all[obj] for obj in object_ids], axis=0)
|
||||
state = self._synthesis_state[object_ids].copy()
|
||||
rot868 = self._rot_table_86840()
|
||||
pcm = np.zeros((len(object_ids), 1536), dtype=np.float64)
|
||||
for tsg in range(6):
|
||||
ring = np.zeros((len(object_ids), 512), dtype=np.float64)
|
||||
chunk = z[:, :, tsg * 4:tsg * 4 + 4].transpose(0, 2, 1)
|
||||
slots = ring.reshape(len(object_ids), 4, 128)
|
||||
slots[..., 0::2] = chunk.real
|
||||
slots[..., 1::2] = chunk.imag
|
||||
for it in range(4):
|
||||
zone = self._front_step(ring[:, 128 * it:128 * it + 128])
|
||||
zone = self._h0b0_vec(self._h880_vec(zone), rot868)
|
||||
ring[:, 64 * it:64 * it + 64] = self._qmf5_vec(state, QMF5_WINDOW, zone)
|
||||
pcm[:, tsg * 256:(tsg + 1) * 256] = np.clip(16.0 * ring[:, :256], -1.0, 1.0)
|
||||
self._synthesis_state[object_ids] = state
|
||||
return {obj: pcm[i] for i, obj in enumerate(object_ids)}
|
||||
|
||||
def dll_synth_object(self, obj, z):
|
||||
"""单对象兼容入口;整帧渲染使用对象维批量实现。"""
|
||||
return self.dll_synth_objects({obj: z})[obj]
|
||||
|
||||
@staticmethod
|
||||
def _win_flat():
|
||||
"""返回逆 QMF 使用的 640 项有效合成窗表。"""
|
||||
return QMF5_WINDOW
|
||||
|
||||
def decode_lfe(self, lfe_pcm):
|
||||
"""将核心 ch3/LFE 经过跨帧 1217-sample 延迟后输出到 ch0。"""
|
||||
current = np.asarray(lfe_pcm, dtype=np.float64)
|
||||
if current.shape != (1536,):
|
||||
raise ValueError(f"LFE 帧应为 (1536,),实际 {current.shape}")
|
||||
delayed = np.concatenate([self._lfe_delay, current])
|
||||
out = np.clip(delayed[:1536], -1.0, 1.0)
|
||||
self._lfe_delay = delayed[-1217:].copy()
|
||||
return out
|
||||
|
||||
def render_frame(self, payload_bytes, bed5_pcm, lfe_pcm=None):
|
||||
"""payload_bytes = evolution 载荷(含 id14 JOC);bed5_pcm = 5×1536 核心 PCM。
|
||||
lfe_pcm = 1536 核心 LFE;提供时生成 ch0,省略时 ch0 静音
|
||||
(保留给只验证对象/z 链的诊断脚本)。
|
||||
返回 (pcm16, z_all):pcm16 = 16×1536(ch0 LFE + 15 对象)f32,
|
||||
已用 FP32 系数乘 output_scale。"""
|
||||
subs, _ = unpack_evolution(payload_bytes, loose=True)
|
||||
return self.render_subpayloads(subs, bed5_pcm, lfe_pcm)
|
||||
|
||||
def render_subpayloads(self, subs, bed5_pcm, lfe_pcm=None):
|
||||
"""以已拆出的 EMDF payload 字典渲染一帧,避免绑定 transport 容器。"""
|
||||
out, mix_q, mix_dq = self.decode_subpayloads(subs)
|
||||
x = self.qmf_x(bed5_pcm)
|
||||
self._last_x = x.copy()
|
||||
z_all = self.object_z(out, mix_dq, x)
|
||||
# joc_clipgain 在对象逆 QMF 后应用,并与用户 output_scale 分离。
|
||||
objs_pcm = self.dll_synth_objects(z_all)
|
||||
pcm16 = np.zeros((16, 1536), dtype=np.float64)
|
||||
pcm16[0] = self.decode_lfe(lfe_pcm) if lfe_pcm is not None else 0.0
|
||||
for obj, p in objs_pcm.items():
|
||||
pcm16[obj + 1] = p * out["clipgain"]
|
||||
pcm16_f32 = np.asarray(pcm16, dtype=np.float32)
|
||||
return np.multiply(pcm16_f32, self.output_scale, dtype=np.float32), z_all
|
||||
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Unified auto/native/python entry point for float64 speaker rendering."""
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
|
||||
from speaker_layouts import SpeakerLayout, get_speaker_layout
|
||||
from speaker_native_renderer import NativeSpeakerRenderer
|
||||
from speaker_renderer import PythonSpeakerRenderer, payload_events_from_index, render_objects16
|
||||
|
||||
FRAME_SAMPLES = 1536
|
||||
|
||||
|
||||
def create_speaker_renderer(layout: str | SpeakerLayout, *, backend="auto", native_library=None):
|
||||
"""Create a stateful speaker renderer and return ``(renderer, info)``."""
|
||||
if backend not in ("auto", "native", "python"):
|
||||
raise ValueError(f"unknown speaker backend: {backend}")
|
||||
target = get_speaker_layout(layout) if isinstance(layout, str) else layout
|
||||
fallback_reason = None
|
||||
if backend in ("auto", "native"):
|
||||
try:
|
||||
renderer = NativeSpeakerRenderer(target, native_library)
|
||||
return renderer, {
|
||||
"name": "native",
|
||||
"layout": target.name,
|
||||
"library": str(renderer.library_path),
|
||||
"fallback_reason": None,
|
||||
}
|
||||
except (OSError, RuntimeError) as exc:
|
||||
fallback_reason = str(exc)
|
||||
return PythonSpeakerRenderer(target), {
|
||||
"name": "python",
|
||||
"layout": target.name,
|
||||
"library": None,
|
||||
"fallback_reason": fallback_reason,
|
||||
}
|
||||
|
||||
|
||||
def render_speaker_layout(objects16, index, layout: str | SpeakerLayout, *,
|
||||
backend="auto", native_library=None,
|
||||
metadata_offset=1473):
|
||||
"""Render a complete indexed object stream and return ``(pcm64, info)``."""
|
||||
target = get_speaker_layout(layout) if isinstance(layout, str) else layout
|
||||
source = np.asarray(objects16)
|
||||
if source.ndim != 2 or source.shape[1] != 16:
|
||||
raise ValueError(f"objects16 must have shape [samples,16], got {source.shape}")
|
||||
if len(source) % FRAME_SAMPLES:
|
||||
raise ValueError("objects16 sample count must be divisible by 1536")
|
||||
frames = len(source) // FRAME_SAMPLES
|
||||
if frames > len(index.rows):
|
||||
raise ValueError(f"metadata has {len(index.rows)} frames, PCM needs {frames}")
|
||||
|
||||
renderer, info = create_speaker_renderer(
|
||||
target, backend=backend, native_library=native_library)
|
||||
output = np.empty((len(source), target.channel_count), dtype=np.float64)
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
for frame in range(frames):
|
||||
start = frame * FRAME_SAMPLES
|
||||
subs = index.subpayloads(index.rows[frame])
|
||||
output[start:start + FRAME_SAMPLES] = renderer.render_frame(
|
||||
source[start:start + FRAME_SAMPLES], subs.get(11), metadata_offset)
|
||||
finally:
|
||||
close = getattr(renderer, "close", None)
|
||||
if close is not None:
|
||||
close()
|
||||
info = dict(info)
|
||||
info["seconds"] = time.perf_counter() - started
|
||||
return output, info
|
||||
|
||||
@@ -0,0 +1,764 @@
|
||||
"""Standard speaker-layout geometry used by the spatial renderer.
|
||||
|
||||
Coordinates are unsigned Q15 room coordinates. This is geometry/topology data,
|
||||
not a sampled gain table.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RegionGeometry:
|
||||
coordinates_q15: tuple[tuple[int, int, int], ...]
|
||||
speaker_ids: tuple[int, ...]
|
||||
axis0_groups: tuple[tuple[int, ...], ...]
|
||||
axis1_groups: tuple[tuple[int, ...], ...]
|
||||
mode: int
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SpeakerLayout:
|
||||
name: str
|
||||
out_ch_config: int
|
||||
speaker_bitfield: int
|
||||
channels: tuple[str, ...]
|
||||
internal_to_standard: tuple[int, ...]
|
||||
regions: tuple[RegionGeometry, ...]
|
||||
|
||||
@property
|
||||
def channel_count(self) -> int:
|
||||
return len(self.channels)
|
||||
|
||||
|
||||
_RAW_LAYOUTS = {'20': {'out_ch_config': 0,
|
||||
'speaker_bitfield': 1,
|
||||
'channels': ('L', 'R'),
|
||||
'internal_to_standard': (0, 1),
|
||||
'regions': ({'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
|
||||
'speaker_ids': (0, 1),
|
||||
'axis0_groups': ((0, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
|
||||
'speaker_ids': (0, 1),
|
||||
'axis0_groups': ((0, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
|
||||
'speaker_ids': (0, 1),
|
||||
'axis0_groups': ((0, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
|
||||
'speaker_ids': (0, 1),
|
||||
'axis0_groups': ((0, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
|
||||
'speaker_ids': (0, 1),
|
||||
'axis0_groups': ((0, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
|
||||
'speaker_ids': (0, 1),
|
||||
'axis0_groups': ((0, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0)),
|
||||
'speaker_ids': (0, 1),
|
||||
'axis0_groups': ((0, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1})},
|
||||
'31': {'out_ch_config': 3,
|
||||
'speaker_bitfield': 7,
|
||||
'channels': ('L', 'R', 'C', 'LFE'),
|
||||
'internal_to_standard': (0, 1, 2, 3),
|
||||
'regions': ({'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((16384, 0, 0),),
|
||||
'speaker_ids': (2,),
|
||||
'axis0_groups': ((0,),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1})},
|
||||
'51': {'out_ch_config': 7,
|
||||
'speaker_bitfield': 15,
|
||||
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs'),
|
||||
'internal_to_standard': (0, 1, 2, 3, 4, 5),
|
||||
'regions': ({'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': (),
|
||||
'mode': 2},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': (),
|
||||
'mode': 2},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': (),
|
||||
'mode': 2},
|
||||
{'coordinates_q15': ((16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
|
||||
'speaker_ids': (2, 4, 5),
|
||||
'axis0_groups': ((0,), (1, 2)),
|
||||
'axis1_groups': (),
|
||||
'mode': 2},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 32767, 0), (32767, 32767, 0)),
|
||||
'speaker_ids': (4, 5),
|
||||
'axis0_groups': ((0, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1})},
|
||||
'71': {'out_ch_config': 11,
|
||||
'speaker_bitfield': 31,
|
||||
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Lrs', 'Rrs'),
|
||||
'internal_to_standard': (0, 1, 2, 3, 6, 7, 4, 5),
|
||||
'regions': ({'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4), (5, 6)),
|
||||
'axis1_groups': (),
|
||||
'mode': 2},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0), (0, 16384, 0), (32767, 16384, 0)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': (),
|
||||
'mode': 2},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
|
||||
'speaker_ids': (0, 1, 2, 6, 7),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': (),
|
||||
'mode': 2},
|
||||
{'coordinates_q15': ((16384, 0, 0), (0, 32767, 0), (32767, 32767, 0)),
|
||||
'speaker_ids': (2, 6, 7),
|
||||
'axis0_groups': ((0,), (1, 2)),
|
||||
'axis1_groups': (),
|
||||
'mode': 2},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1},
|
||||
{'coordinates_q15': ((0, 16384, 0), (32767, 16384, 0), (0, 32767, 0), (32767, 32767, 0)),
|
||||
'speaker_ids': (4, 5, 6, 7),
|
||||
'axis0_groups': ((0, 1), (2, 3)),
|
||||
'axis1_groups': (),
|
||||
'mode': 2},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1})},
|
||||
'512': {'out_ch_config': 13,
|
||||
'speaker_bitfield': 1039,
|
||||
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Ltm', 'Rtm'),
|
||||
'internal_to_standard': (0, 1, 2, 3, 4, 5, 6, 7),
|
||||
'regions': ({'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (2, 4, 5, 6, 7),
|
||||
'axis0_groups': ((0,), (1, 2)),
|
||||
'axis1_groups': ((3, 4),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 6, 7),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': ((3, 4),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (4, 5, 6, 7),
|
||||
'axis0_groups': ((0, 1),),
|
||||
'axis1_groups': ((2, 3),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1})},
|
||||
'514': {'out_ch_config': 14,
|
||||
'speaker_bitfield': 2575,
|
||||
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Ltf', 'Rtf', 'Ltr', 'Rtr'),
|
||||
'internal_to_standard': (0, 1, 2, 3, 4, 5, 6, 7, 8, 9),
|
||||
'regions': ({'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6), (7, 8)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6), (7, 8)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6), (7, 8)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (2, 4, 5, 6, 7, 8, 9),
|
||||
'axis0_groups': ((0,), (1, 2)),
|
||||
'axis1_groups': ((3, 4), (5, 6)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 6, 7, 8, 9),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': ((3, 4), (5, 6)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (4, 5, 6, 7, 8, 9),
|
||||
'axis0_groups': ((0, 1),),
|
||||
'axis1_groups': ((2, 3), (4, 5)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 6, 7),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': ((3, 4),),
|
||||
'mode': 3})},
|
||||
'712': {'out_ch_config': 15,
|
||||
'speaker_bitfield': 1055,
|
||||
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Lrs', 'Rrs', 'Ltm', 'Rtm'),
|
||||
'internal_to_standard': (0, 1, 2, 3, 6, 7, 4, 5, 8, 9),
|
||||
'regions': ({'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4), (5, 6)),
|
||||
'axis1_groups': ((7, 8),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 8, 9),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 6, 7, 8, 9),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (2, 6, 7, 8, 9),
|
||||
'axis0_groups': ((0,), (1, 2)),
|
||||
'axis1_groups': ((3, 4),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 8, 9),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': ((3, 4),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767)),
|
||||
'speaker_ids': (4, 5, 6, 7, 8, 9),
|
||||
'axis0_groups': ((0, 1), (2, 3)),
|
||||
'axis1_groups': ((4, 5),),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0), (32767, 0, 0), (16384, 0, 0)),
|
||||
'speaker_ids': (0, 1, 2),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': (),
|
||||
'mode': 1})},
|
||||
'714': {'out_ch_config': 16,
|
||||
'speaker_bitfield': 2591,
|
||||
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Lrs', 'Rrs', 'Ltf', 'Rtf', 'Ltr', 'Rtr'),
|
||||
'internal_to_standard': (0, 1, 2, 3, 6, 7, 4, 5, 8, 9, 10, 11),
|
||||
'regions': ({'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4), (5, 6)),
|
||||
'axis1_groups': ((7, 8), (9, 10)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 8, 9, 10, 11),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6), (7, 8)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 6, 7, 8, 9, 10, 11),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6), (7, 8)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (2, 6, 7, 8, 9, 10, 11),
|
||||
'axis0_groups': ((0,), (1, 2)),
|
||||
'axis1_groups': ((3, 4), (5, 6)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 8, 9, 10, 11),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': ((3, 4), (5, 6)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (4, 5, 6, 7, 8, 9, 10, 11),
|
||||
'axis0_groups': ((0, 1), (2, 3)),
|
||||
'axis1_groups': ((4, 5), (6, 7)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 8, 9),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': ((3, 4),),
|
||||
'mode': 3})},
|
||||
'914': {'out_ch_config': 19,
|
||||
'speaker_bitfield': 2719,
|
||||
'channels': ('L', 'R', 'C', 'LFE', 'Ls', 'Rs', 'Lrs', 'Rrs', 'Lw', 'Rw', 'Ltf', 'Rtf', 'Ltr', 'Rtr'),
|
||||
'internal_to_standard': (0, 1, 2, 3, 6, 7, 4, 5, 10, 11, 12, 13, 8, 9),
|
||||
'regions': ({'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(0, 5285, 0),
|
||||
(32767, 5285, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13),
|
||||
'axis0_groups': ((0, 2, 1), (7, 8), (3, 4), (5, 6)),
|
||||
'axis1_groups': ((9, 10), (11, 12)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 5285, 0),
|
||||
(32767, 5285, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 8, 9, 10, 11, 12, 13),
|
||||
'axis0_groups': ((0, 2, 1), (5, 6), (3, 4)),
|
||||
'axis1_groups': ((7, 8), (9, 10)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 6, 7, 10, 11, 12, 13),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6), (7, 8)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (2, 6, 7, 10, 11, 12, 13),
|
||||
'axis0_groups': ((0,), (1, 2)),
|
||||
'axis1_groups': ((3, 4), (5, 6)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 10, 11, 12, 13),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': ((3, 4), (5, 6)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (4, 5, 6, 7, 10, 11, 12, 13),
|
||||
'axis0_groups': ((0, 1), (2, 3)),
|
||||
'axis1_groups': ((4, 5), (6, 7)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 5285, 0),
|
||||
(32767, 5285, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 8, 9, 10, 11),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6),),
|
||||
'mode': 3})},
|
||||
'916': {'out_ch_config': 20,
|
||||
'speaker_bitfield': 3743,
|
||||
'channels': ('L',
|
||||
'R',
|
||||
'C',
|
||||
'LFE',
|
||||
'Ls',
|
||||
'Rs',
|
||||
'Lrs',
|
||||
'Rrs',
|
||||
'Lw',
|
||||
'Rw',
|
||||
'Ltf',
|
||||
'Rtf',
|
||||
'Ltm',
|
||||
'Rtm',
|
||||
'Ltr',
|
||||
'Rtr'),
|
||||
'internal_to_standard': (0, 1, 2, 3, 6, 7, 4, 5, 10, 11, 14, 15, 12, 13, 8, 9),
|
||||
'regions': ({'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(0, 5285, 0),
|
||||
(32767, 5285, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15),
|
||||
'axis0_groups': ((0, 2, 1), (7, 8), (3, 4), (5, 6)),
|
||||
'axis1_groups': ((9, 10), (11, 12), (13, 14)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 5285, 0),
|
||||
(32767, 5285, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 4, 5, 8, 9, 10, 11, 12, 13, 14, 15),
|
||||
'axis0_groups': ((0, 2, 1), (5, 6), (3, 4)),
|
||||
'axis1_groups': ((7, 8), (9, 10), (11, 12)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 6, 7, 10, 11, 12, 13, 14, 15),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6), (7, 8), (9, 10)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((16384, 0, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (2, 6, 7, 10, 11, 12, 13, 14, 15),
|
||||
'axis0_groups': ((0,), (1, 2)),
|
||||
'axis1_groups': ((3, 4), (5, 6), (7, 8)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 10, 11, 12, 13, 14, 15),
|
||||
'axis0_groups': ((0, 2, 1),),
|
||||
'axis1_groups': ((3, 4), (5, 6), (7, 8)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 16384, 0),
|
||||
(32767, 16384, 0),
|
||||
(0, 32767, 0),
|
||||
(32767, 32767, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767),
|
||||
(7928, 16384, 32767),
|
||||
(24840, 16384, 32767),
|
||||
(7928, 24840, 32767),
|
||||
(24840, 24840, 32767)),
|
||||
'speaker_ids': (4, 5, 6, 7, 10, 11, 12, 13, 14, 15),
|
||||
'axis0_groups': ((0, 1), (2, 3)),
|
||||
'axis1_groups': ((4, 5), (6, 7), (8, 9)),
|
||||
'mode': 3},
|
||||
{'coordinates_q15': ((0, 0, 0),
|
||||
(32767, 0, 0),
|
||||
(16384, 0, 0),
|
||||
(0, 5285, 0),
|
||||
(32767, 5285, 0),
|
||||
(7928, 7928, 32767),
|
||||
(24840, 7928, 32767)),
|
||||
'speaker_ids': (0, 1, 2, 8, 9, 10, 11),
|
||||
'axis0_groups': ((0, 2, 1), (3, 4)),
|
||||
'axis1_groups': ((5, 6),),
|
||||
'mode': 3})}}
|
||||
|
||||
|
||||
def _make_layout(name: str, item: dict) -> SpeakerLayout:
|
||||
return SpeakerLayout(
|
||||
name=name,
|
||||
out_ch_config=item["out_ch_config"],
|
||||
speaker_bitfield=item["speaker_bitfield"],
|
||||
channels=item["channels"],
|
||||
internal_to_standard=item["internal_to_standard"],
|
||||
regions=tuple(RegionGeometry(**region) for region in item["regions"]),
|
||||
)
|
||||
|
||||
|
||||
SPEAKER_LAYOUTS = {name: _make_layout(name, item) for name, item in _RAW_LAYOUTS.items()}
|
||||
|
||||
SPEAKER_LAYOUT_ALIASES = {
|
||||
"2.0": "20",
|
||||
"3.1": "31",
|
||||
"5.1": "51",
|
||||
"7.1": "71",
|
||||
"5.1.2": "512",
|
||||
"5.1.4": "514",
|
||||
"7.1.2": "712",
|
||||
"7.1.4": "714",
|
||||
"9.1.4": "914",
|
||||
"9.1.6": "916",
|
||||
}
|
||||
SPEAKER_LAYOUT_CHOICES = tuple(SPEAKER_LAYOUT_ALIASES)
|
||||
SPEAKER_LAYOUT_DISPLAY_NAMES = {value: key for key, value in SPEAKER_LAYOUT_ALIASES.items()}
|
||||
|
||||
|
||||
def get_speaker_layout(name: str) -> SpeakerLayout:
|
||||
key = SPEAKER_LAYOUT_ALIASES.get(name, name)
|
||||
try:
|
||||
return SPEAKER_LAYOUTS[key]
|
||||
except KeyError as exc:
|
||||
raise ValueError(f"unsupported speaker layout: {name}") from exc
|
||||
|
||||
|
||||
def speaker_layout_display_name(layout: str | SpeakerLayout) -> str:
|
||||
target = get_speaker_layout(layout) if isinstance(layout, str) else layout
|
||||
return SPEAKER_LAYOUT_DISPLAY_NAMES[target.name]
|
||||
@@ -0,0 +1,154 @@
|
||||
"""ctypes bridge for the float64 native object-to-speaker renderer."""
|
||||
from __future__ import annotations
|
||||
|
||||
import ctypes
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from native_renderer import ABI_VERSION, find_native_library
|
||||
from oamd_bits import JocFieldState, frame_update
|
||||
from speaker_layouts import SpeakerLayout, get_speaker_layout
|
||||
|
||||
FRAME_SAMPLES = 1536
|
||||
INPUT_CHANNELS = 16
|
||||
OBJECTS = 15
|
||||
COORDINATES = 3
|
||||
|
||||
|
||||
class NativeSpeakerRenderer:
|
||||
def __init__(self, layout: str | SpeakerLayout, library_path=None):
|
||||
self.layout = get_speaker_layout(layout) if isinstance(layout, str) else layout
|
||||
self.library_path = find_native_library(library_path)
|
||||
self._lib = ctypes.CDLL(str(self.library_path))
|
||||
self._bind()
|
||||
version = int(self._lib.ejoc_abi_version())
|
||||
if version != ABI_VERSION:
|
||||
raise RuntimeError(f"native ABI mismatch: expected {ABI_VERSION}, got {version}")
|
||||
channels = int(self._lib.ejoc_speaker_layout_channel_count(self.layout.speaker_bitfield))
|
||||
if channels != self.layout.channel_count:
|
||||
raise RuntimeError(
|
||||
f"native layout channel count mismatch: expected {self.layout.channel_count}, got {channels}"
|
||||
)
|
||||
self._handle = self._lib.ejoc_speaker_renderer_create(self.layout.speaker_bitfield)
|
||||
if not self._handle:
|
||||
raise RuntimeError(f"native speaker renderer rejected mask 0x{self.layout.speaker_bitfield:X}")
|
||||
self._state = JocFieldState()
|
||||
|
||||
def _bind(self):
|
||||
void_p = ctypes.c_void_p
|
||||
self._lib.ejoc_abi_version.argtypes = []
|
||||
self._lib.ejoc_abi_version.restype = ctypes.c_uint32
|
||||
self._lib.ejoc_speaker_layout_channel_count.argtypes = [ctypes.c_uint32]
|
||||
self._lib.ejoc_speaker_layout_channel_count.restype = ctypes.c_uint32
|
||||
self._lib.ejoc_speaker_renderer_create.argtypes = [ctypes.c_uint32]
|
||||
self._lib.ejoc_speaker_renderer_create.restype = void_p
|
||||
self._lib.ejoc_speaker_renderer_destroy.argtypes = [void_p]
|
||||
self._lib.ejoc_speaker_renderer_destroy.restype = None
|
||||
self._lib.ejoc_speaker_renderer_reset.argtypes = [void_p]
|
||||
self._lib.ejoc_speaker_renderer_reset.restype = ctypes.c_int
|
||||
self._lib.ejoc_speaker_renderer_last_error.argtypes = [void_p]
|
||||
self._lib.ejoc_speaker_renderer_last_error.restype = ctypes.c_char_p
|
||||
self._lib.ejoc_speaker_renderer_process.argtypes = [
|
||||
void_p,
|
||||
ctypes.POINTER(ctypes.c_float),
|
||||
ctypes.c_uint32,
|
||||
ctypes.c_uint32,
|
||||
ctypes.POINTER(ctypes.c_uint32),
|
||||
ctypes.POINTER(ctypes.c_uint32),
|
||||
ctypes.POINTER(ctypes.c_uint16),
|
||||
ctypes.POINTER(ctypes.c_uint8),
|
||||
ctypes.POINTER(ctypes.c_uint8),
|
||||
ctypes.POINTER(ctypes.c_double),
|
||||
ctypes.POINTER(ctypes.c_double),
|
||||
]
|
||||
self._lib.ejoc_speaker_renderer_process.restype = ctypes.c_int
|
||||
|
||||
def _raise(self, operation, status):
|
||||
message = self._lib.ejoc_speaker_renderer_last_error(self._handle)
|
||||
detail = (message or b"").decode("utf-8", "replace")
|
||||
raise RuntimeError(f"native speaker renderer {operation} failed ({status}): {detail}")
|
||||
|
||||
def reset(self):
|
||||
if not self._handle:
|
||||
raise RuntimeError("native speaker renderer is closed")
|
||||
status = self._lib.ejoc_speaker_renderer_reset(self._handle)
|
||||
if status:
|
||||
self._raise("reset", status)
|
||||
self._state = JocFieldState()
|
||||
|
||||
def close(self):
|
||||
if self._handle:
|
||||
self._lib.ejoc_speaker_renderer_destroy(self._handle)
|
||||
self._handle = None
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc, tb):
|
||||
self.close()
|
||||
|
||||
def __del__(self):
|
||||
try:
|
||||
self.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _metadata_arrays(self, events):
|
||||
events = list(events)
|
||||
count = len(events)
|
||||
if not count:
|
||||
return count, None, None, None
|
||||
offsets = np.empty(count, dtype=np.uint32)
|
||||
ramps = np.empty(count, dtype=np.uint32)
|
||||
positions = np.empty((count, OBJECTS, COORDINATES), dtype=np.uint16)
|
||||
for event_index, (outer_offset, payload) in enumerate(events):
|
||||
update = frame_update(payload)
|
||||
self._state.apply(update["values"])
|
||||
offsets[event_index] = int(outer_offset) + int(update["block_offset_samples"])
|
||||
ramps[event_index] = int(update["ramp_duration_samples"])
|
||||
for object_index in range(1, OBJECTS + 1):
|
||||
positions[event_index, object_index - 1] = (
|
||||
self._state.q[(object_index, "q1")],
|
||||
self._state.q[(object_index, "q2")],
|
||||
self._state.q[(object_index, "q3")],
|
||||
)
|
||||
return count, offsets, ramps, positions
|
||||
|
||||
def process(self, objects16, events=()):
|
||||
if not self._handle:
|
||||
raise RuntimeError("native speaker renderer is closed")
|
||||
source = np.ascontiguousarray(objects16, dtype=np.float32)
|
||||
if source.ndim != 2 or source.shape[1] != INPUT_CHANNELS:
|
||||
raise ValueError(f"objects16 must have shape [samples,16], got {source.shape}")
|
||||
if len(source) % 32:
|
||||
raise ValueError("sample count must be a multiple of 32")
|
||||
count, offsets, ramps, positions = self._metadata_arrays(events)
|
||||
output = np.empty((len(source), self.layout.channel_count), dtype=np.float64)
|
||||
null_u32 = ctypes.POINTER(ctypes.c_uint32)()
|
||||
null_u16 = ctypes.POINTER(ctypes.c_uint16)()
|
||||
null_u8 = ctypes.POINTER(ctypes.c_uint8)()
|
||||
null_f64 = ctypes.POINTER(ctypes.c_double)()
|
||||
status = self._lib.ejoc_speaker_renderer_process(
|
||||
self._handle,
|
||||
source.ctypes.data_as(ctypes.POINTER(ctypes.c_float)),
|
||||
len(source),
|
||||
count,
|
||||
offsets.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)) if count else null_u32,
|
||||
ramps.ctypes.data_as(ctypes.POINTER(ctypes.c_uint32)) if count else null_u32,
|
||||
positions.ctypes.data_as(ctypes.POINTER(ctypes.c_uint16)) if count else null_u16,
|
||||
null_u8,
|
||||
null_u8,
|
||||
null_f64,
|
||||
output.ctypes.data_as(ctypes.POINTER(ctypes.c_double)),
|
||||
)
|
||||
if status:
|
||||
self._raise("process", status)
|
||||
return output
|
||||
|
||||
def render_frame(self, objects16, payload=None, metadata_offset=1473):
|
||||
source = np.asarray(objects16)
|
||||
if source.shape != (FRAME_SAMPLES, INPUT_CHANNELS):
|
||||
raise ValueError(f"frame must have shape (1536,16), got {source.shape}")
|
||||
events = () if payload is None else ((int(metadata_offset), bytes(payload)),)
|
||||
return self.process(source, events)
|
||||
@@ -0,0 +1,311 @@
|
||||
"""High-precision object-to-speaker renderer.
|
||||
|
||||
All spatial calculations, gain ramps, and object accumulation use float64. Input
|
||||
PCM may be float32, but precision is reduced only when the caller explicitly
|
||||
writes a lower-precision output format.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
import math
|
||||
from typing import Iterable
|
||||
|
||||
import numpy as np
|
||||
|
||||
from oamd_bits import JocFieldState, frame_update
|
||||
from speaker_layouts import RegionGeometry, SpeakerLayout, get_speaker_layout
|
||||
|
||||
Q15_SCALE = 32768.0
|
||||
Q15_MAX = 32767.0 / Q15_SCALE
|
||||
GAIN_SNAP_THRESHOLD = 1.0e-4
|
||||
|
||||
|
||||
def expand_speaker_bitfield(compact_mask: int, center_height: bool = False) -> int:
|
||||
"""Expand the compact target-layout bitfield into individual speakers."""
|
||||
expansions = (
|
||||
0x00000003, 0x00000004, 0x00000008, 0x00000030,
|
||||
0x000000C0, 0x00000100, 0x00000600, 0x00001800,
|
||||
0x00006000, 0x00018000, 0x00060000, 0x00180000,
|
||||
0x00600000, 0x01800000, 0x06000000, 0x18000000,
|
||||
0x60000000, 0x080000000, 0x600000000, 0x800000000,
|
||||
0x1000000000, 0x2000000000,
|
||||
)
|
||||
expanded = 0
|
||||
for bit, value in enumerate(expansions):
|
||||
if int(compact_mask) & (1 << bit):
|
||||
expanded |= value
|
||||
if center_height:
|
||||
expanded |= 0x4000000000
|
||||
return expanded
|
||||
|
||||
|
||||
def layout_attenuation_db(compact_mask: int) -> float:
|
||||
"""Compute the layout-dependent maximum positional compensation in dB."""
|
||||
expanded = expand_speaker_bitfield(compact_mask)
|
||||
height_channels = 2 * sum((expanded >> bit) & 1 for bit in (13, 15, 17, 19, 21))
|
||||
floor_channels = ((expanded >> 8) & 1) + 2 * sum(
|
||||
(expanded >> bit) & 1 for bit in (31, 4, 6, 11, 25, 27, 29, 33)
|
||||
)
|
||||
height_factor = min(height_channels / 4.0, 1.0)
|
||||
floor_factor = min(floor_channels / 4.0, 1.0)
|
||||
return -max(4.5 - 1.5 * height_factor - 3.0 * floor_factor, 0.0)
|
||||
|
||||
|
||||
def floor_y_exponent(compact_mask: int, metadata_scaling_enabled: bool = True) -> int:
|
||||
if not metadata_scaling_enabled:
|
||||
return 0
|
||||
low_mask = expand_speaker_bitfield(compact_mask) & 0xFFFFFFFF
|
||||
return int(bool(low_mask & 0x130) and not bool(low_mask & 0x18C0))
|
||||
|
||||
|
||||
def position_gain(compact_mask: int, v: float, w: float) -> float:
|
||||
"""Return the high-precision position-dependent layout compensation."""
|
||||
y_term = min(max(float(v) / 0.6, 0.0), 1.0)
|
||||
z_term = min(max((float(w) - 0.2) / 0.8, 0.0), 1.0)
|
||||
amount = min(max(y_term + z_term, 0.0), 1.0)
|
||||
return math.pow(10.0, layout_attenuation_db(compact_mask) * amount / 20.0)
|
||||
|
||||
|
||||
def equal_power_pair(position: float) -> tuple[float, float]:
|
||||
angle = math.pi * 0.5 * float(position)
|
||||
return math.cos(angle), math.sin(angle)
|
||||
|
||||
|
||||
def _coordinates(region: RegionGeometry) -> np.ndarray:
|
||||
return np.asarray(region.coordinates_q15, dtype=np.float64) / Q15_SCALE
|
||||
|
||||
|
||||
def _normalized(value: float, lower: float, upper: float) -> float:
|
||||
return (float(value) - float(lower)) / (float(upper) - float(lower))
|
||||
|
||||
|
||||
def _axis0_gains(region: RegionGeometry, coordinates: np.ndarray, value: float) -> np.ndarray:
|
||||
result = np.zeros(len(region.speaker_ids), dtype=np.float64)
|
||||
position = float(value)
|
||||
for group in region.axis0_groups:
|
||||
if not group:
|
||||
continue
|
||||
first, last = group[0], group[-1]
|
||||
if position <= coordinates[first, 0]:
|
||||
result[first] = 1.0
|
||||
continue
|
||||
if position >= coordinates[last, 0]:
|
||||
result[last] = 1.0
|
||||
continue
|
||||
for lower_index, upper_index in zip(group, group[1:]):
|
||||
lower = coordinates[lower_index, 0]
|
||||
upper = coordinates[upper_index, 0]
|
||||
if position > lower and position <= upper:
|
||||
result[lower_index], result[upper_index] = equal_power_pair(
|
||||
_normalized(position, lower, upper)
|
||||
)
|
||||
break
|
||||
return result
|
||||
|
||||
|
||||
def _axis1_gains(region: RegionGeometry, coordinates: np.ndarray, groups,
|
||||
value: float) -> np.ndarray:
|
||||
result = np.zeros(len(region.speaker_ids), dtype=np.float64)
|
||||
if not groups:
|
||||
return result
|
||||
position = float(value)
|
||||
first_value = coordinates[groups[0][0], 1]
|
||||
last_value = coordinates[groups[-1][0], 1]
|
||||
if position <= first_value:
|
||||
result[list(groups[0])] = 1.0
|
||||
return result
|
||||
if position > last_value:
|
||||
result[list(groups[-1])] = 1.0
|
||||
return result
|
||||
for lower_group, upper_group in zip(groups, groups[1:]):
|
||||
lower = coordinates[lower_group[0], 1]
|
||||
upper = coordinates[upper_group[0], 1]
|
||||
if position >= lower and position <= upper:
|
||||
lower_gain, upper_gain = equal_power_pair(_normalized(position, lower, upper))
|
||||
result[list(lower_group)] = lower_gain
|
||||
result[list(upper_group)] = upper_gain
|
||||
break
|
||||
return result
|
||||
|
||||
|
||||
def _plane_gains(region: RegionGeometry, coordinates: np.ndarray, groups,
|
||||
u: float, v: float, mode: int) -> np.ndarray:
|
||||
gains = _axis0_gains(
|
||||
RegionGeometry(region.coordinates_q15, region.speaker_ids, tuple(groups), (), region.mode),
|
||||
coordinates,
|
||||
u,
|
||||
)
|
||||
if mode >= 2:
|
||||
gains *= _axis1_gains(region, coordinates, groups, v)
|
||||
return gains
|
||||
|
||||
|
||||
def render_point_gains(layout: str | SpeakerLayout, u: float, v: float, w: float,
|
||||
*, region_index: int = 0, enable_height: bool = True,
|
||||
object_gain: float = 1.0, standard_order: bool = True) -> np.ndarray:
|
||||
"""Render one point object to a target speaker layout using float64."""
|
||||
target = get_speaker_layout(layout) if isinstance(layout, str) else layout
|
||||
region = target.regions[int(region_index)]
|
||||
coordinates = _coordinates(region)
|
||||
floor_v = min(max(math.ldexp(float(v), floor_y_exponent(target.speaker_bitfield)), 0.0), 1.0)
|
||||
floor = _plane_gains(region, coordinates, region.axis0_groups, u, floor_v, region.mode)
|
||||
point_gains = floor
|
||||
if region.mode == 3:
|
||||
top = _plane_gains(region, coordinates, region.axis1_groups, u, float(v), 3)
|
||||
height = min(max(float(w) if enable_height else 0.0, 0.0), Q15_MAX)
|
||||
if height >= Q15_MAX:
|
||||
point_gains = top
|
||||
elif height > 0.0:
|
||||
floor_weight, height_weight = equal_power_pair(height)
|
||||
point_gains = floor * floor_weight + top * height_weight
|
||||
internal = np.zeros(target.channel_count, dtype=np.float64)
|
||||
gain = position_gain(target.speaker_bitfield, v, w) * float(object_gain)
|
||||
for point_gain, speaker_id in zip(point_gains, region.speaker_ids):
|
||||
internal[speaker_id] = point_gain * gain
|
||||
if standard_order:
|
||||
return internal[list(target.internal_to_standard)]
|
||||
return internal
|
||||
|
||||
|
||||
def align_metadata_sample(sample: int, block_size: int = 32) -> int:
|
||||
return block_size * ((int(sample) + block_size // 2 - 1) // block_size)
|
||||
|
||||
|
||||
def ramp_block_count(duration_samples: int, block_size: int = 32,
|
||||
rate_scale: int = 1) -> int:
|
||||
return (int(duration_samples) * int(rate_scale) + block_size // 2 - 1) // block_size
|
||||
|
||||
|
||||
def _all_object_targets(layout: SpeakerLayout, state: JocFieldState) -> np.ndarray:
|
||||
result = np.zeros((15, layout.channel_count), dtype=np.float64)
|
||||
for object_index in range(1, 16):
|
||||
result[object_index - 1] = render_point_gains(
|
||||
layout,
|
||||
state.q[(object_index, "q1")] / Q15_SCALE,
|
||||
state.q[(object_index, "q2")] / Q15_SCALE,
|
||||
state.q[(object_index, "q3")] / Q15_SCALE,
|
||||
standard_order=False,
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
class PythonSpeakerRenderer:
|
||||
"""Stateful float64 speaker renderer with the same frame API as native."""
|
||||
|
||||
def __init__(self, layout: str | SpeakerLayout, block_size: int = 32):
|
||||
self.layout = get_speaker_layout(layout) if isinstance(layout, str) else layout
|
||||
self.block_size = int(block_size)
|
||||
if self.block_size <= 0:
|
||||
raise ValueError("block_size must be positive")
|
||||
self.reset()
|
||||
|
||||
def reset(self):
|
||||
channels = self.layout.channel_count
|
||||
self._state = JocFieldState()
|
||||
self._current = np.zeros((15, channels), dtype=np.float64)
|
||||
self._target = np.zeros_like(self._current)
|
||||
self._step = np.zeros_like(self._current)
|
||||
self._remaining = np.zeros((15, channels), dtype=np.int32)
|
||||
self._window = np.arange(self.block_size, dtype=np.float64) / float(self.block_size)
|
||||
|
||||
def _apply_payload(self, payload):
|
||||
update = frame_update(payload)
|
||||
self._state.apply(update["values"])
|
||||
new_target = _all_object_targets(self.layout, self._state)
|
||||
blocks = ramp_block_count(update["ramp_duration_samples"], self.block_size)
|
||||
difference = new_target - self._current
|
||||
significant = np.abs(difference) >= GAIN_SNAP_THRESHOLD
|
||||
self._target[...] = new_target
|
||||
self._current[~significant] = new_target[~significant]
|
||||
self._step[~significant] = 0.0
|
||||
self._remaining[~significant] = 0
|
||||
if blocks:
|
||||
self._step[significant] = difference[significant] / float(blocks)
|
||||
self._remaining[significant] = blocks
|
||||
else:
|
||||
self._current[significant] = new_target[significant]
|
||||
self._step[significant] = 0.0
|
||||
self._remaining[significant] = 0
|
||||
|
||||
def process(self, objects16: np.ndarray, events=(), *, standard_order: bool = True,
|
||||
output: np.ndarray | None = None) -> np.ndarray:
|
||||
source = np.asarray(objects16)
|
||||
if source.ndim != 2 or source.shape[1] != 16:
|
||||
raise ValueError(f"objects16 must have shape [samples,16], got {source.shape}")
|
||||
if len(source) % self.block_size:
|
||||
raise ValueError(f"sample count must be divisible by {self.block_size}")
|
||||
total_samples = len(source)
|
||||
channels = self.layout.channel_count
|
||||
if output is None:
|
||||
output = np.zeros((total_samples, channels), dtype=np.float64)
|
||||
elif output.shape != (total_samples, channels) or output.dtype != np.float64:
|
||||
raise ValueError((output.shape, output.dtype))
|
||||
else:
|
||||
output[...] = 0.0
|
||||
if "LFE" in self.layout.channels:
|
||||
output[:, 3] = source[:, 0].astype(np.float64, copy=False)
|
||||
|
||||
events_by_block: dict[int, list[bytes]] = defaultdict(list)
|
||||
for sample_offset, payload in events:
|
||||
if not 0 <= int(sample_offset) <= total_samples:
|
||||
raise ValueError(f"metadata offset outside process call: {sample_offset}")
|
||||
block = align_metadata_sample(sample_offset, self.block_size) // self.block_size
|
||||
events_by_block[block].append(bytes(payload))
|
||||
|
||||
total_blocks = total_samples // self.block_size
|
||||
for block in range(total_blocks):
|
||||
for payload in events_by_block.get(block, ()):
|
||||
self._apply_payload(payload)
|
||||
start = block * self.block_size
|
||||
stop = start + self.block_size
|
||||
out_block = output[start:stop]
|
||||
for object_index in range(15):
|
||||
active = self._remaining[object_index] > 0
|
||||
gains = np.broadcast_to(self._target[object_index],
|
||||
(self.block_size, channels)).copy()
|
||||
if np.any(active):
|
||||
gains[:, active] = (
|
||||
self._current[object_index, active][None, :]
|
||||
+ self._window[:, None] * self._step[object_index, active][None, :]
|
||||
)
|
||||
out_block += (
|
||||
source[start:stop, object_index + 1, None].astype(np.float64, copy=False)
|
||||
* gains
|
||||
)
|
||||
if np.any(active):
|
||||
self._current[object_index, active] += self._step[object_index, active]
|
||||
self._remaining[object_index, active] -= 1
|
||||
finished = self._remaining[object_index] == 0
|
||||
self._current[object_index, finished] = self._target[object_index, finished]
|
||||
|
||||
for payload in events_by_block.get(total_blocks, ()):
|
||||
self._apply_payload(payload)
|
||||
if standard_order:
|
||||
return output[:, list(self.layout.internal_to_standard)]
|
||||
return output
|
||||
|
||||
def render_frame(self, objects16, payload=None, metadata_offset=1473):
|
||||
source = np.asarray(objects16)
|
||||
if source.shape != (1536, 16):
|
||||
raise ValueError(f"frame must have shape (1536,16), got {source.shape}")
|
||||
events = () if payload is None else ((int(metadata_offset), bytes(payload)),)
|
||||
return self.process(source, events)
|
||||
|
||||
|
||||
def render_objects16(objects16: np.ndarray, payload_events: Iterable[tuple[int, bytes]],
|
||||
layout: str | SpeakerLayout, *, block_size: int = 32,
|
||||
standard_order: bool = True, output: np.ndarray | None = None) -> np.ndarray:
|
||||
"""Render a complete interleaved LFE + 15 object stream."""
|
||||
renderer = PythonSpeakerRenderer(layout, block_size=block_size)
|
||||
return renderer.process(objects16, payload_events, standard_order=standard_order, output=output)
|
||||
|
||||
def payload_events_from_index(index, frame_count: int, *, frame_samples: int = 1536,
|
||||
payload_id: int = 11, outer_offset_samples: int = 1473):
|
||||
for frame, row in enumerate(index.rows[:frame_count]):
|
||||
subpayloads = index.subpayloads(row)
|
||||
if payload_id in subpayloads:
|
||||
timing = frame_update(subpayloads[payload_id])
|
||||
yield (
|
||||
frame * frame_samples + outer_offset_samples + timing["block_offset_samples"],
|
||||
subpayloads[payload_id],
|
||||
)
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Streaming spool and WAV writer for direct speaker-layout output."""
|
||||
from __future__ import annotations
|
||||
|
||||
import struct
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
WAVE_FORMAT_PCM = 0x0001
|
||||
WAVE_FORMAT_IEEE_FLOAT = 0x0003
|
||||
WAVE_FORMAT_EXTENSIBLE = 0xFFFE
|
||||
_PCM_GUID = bytes.fromhex("0100000000001000800000aa00389b71")
|
||||
_FLOAT_GUID = bytes.fromhex("0300000000001000800000aa00389b71")
|
||||
|
||||
|
||||
class SpeakerPcmSpool:
|
||||
"""Temporary interleaved float32 store with float64 peak analysis."""
|
||||
|
||||
def __init__(self, path, sample_count, channel_count):
|
||||
self.path = Path(path)
|
||||
self.sample_count = int(sample_count)
|
||||
self.channel_count = int(channel_count)
|
||||
self.position = 0
|
||||
self.peak = 0.0
|
||||
self.clipped_values = 0
|
||||
self.values = np.memmap(
|
||||
self.path, dtype="<f4", mode="w+",
|
||||
shape=(self.sample_count, self.channel_count),
|
||||
)
|
||||
|
||||
def write_frame(self, pcm):
|
||||
values = np.asarray(pcm, dtype=np.float64)
|
||||
if values.ndim != 2 or values.shape[1] != self.channel_count:
|
||||
raise ValueError(
|
||||
f"speaker frame must have shape [samples,{self.channel_count}], got {values.shape}")
|
||||
if self.position + len(values) > self.sample_count:
|
||||
raise ValueError("speaker spool received more samples than allocated")
|
||||
if not np.all(np.isfinite(values)):
|
||||
raise ValueError("speaker renderer produced NaN or infinity")
|
||||
absolute = np.abs(values)
|
||||
if absolute.size:
|
||||
self.peak = max(self.peak, float(np.max(absolute)))
|
||||
self.clipped_values += int(np.count_nonzero(absolute > 1.0))
|
||||
self.values[self.position:self.position + len(values)] = values.astype(np.float32)
|
||||
self.position += len(values)
|
||||
|
||||
def finalize(self):
|
||||
if self.position != self.sample_count:
|
||||
raise ValueError(
|
||||
f"speaker spool has {self.position} samples, expected {self.sample_count}")
|
||||
self.values.flush()
|
||||
return self
|
||||
|
||||
def close(self):
|
||||
values = self.values
|
||||
self.values = None
|
||||
del values
|
||||
|
||||
|
||||
def _fmt_chunk(channel_count, rate, sample_format):
|
||||
if sample_format == "float32":
|
||||
bits = 32
|
||||
bytes_per_sample = 4
|
||||
simple_tag = WAVE_FORMAT_IEEE_FLOAT
|
||||
guid = _FLOAT_GUID
|
||||
elif sample_format == "int24":
|
||||
bits = 24
|
||||
bytes_per_sample = 3
|
||||
simple_tag = WAVE_FORMAT_PCM
|
||||
guid = _PCM_GUID
|
||||
else:
|
||||
raise ValueError(f"unsupported speaker WAV format: {sample_format}")
|
||||
block_align = channel_count * bytes_per_sample
|
||||
byte_rate = rate * block_align
|
||||
if channel_count <= 2:
|
||||
body = struct.pack(
|
||||
"<HHIIHH", simple_tag, channel_count, rate,
|
||||
byte_rate, block_align, bits)
|
||||
else:
|
||||
body = (
|
||||
struct.pack(
|
||||
"<HHIIHHH", WAVE_FORMAT_EXTENSIBLE, channel_count, rate,
|
||||
byte_rate, block_align, bits, 22)
|
||||
+ struct.pack("<HI", bits, 0)
|
||||
+ guid
|
||||
)
|
||||
return body, bits, bytes_per_sample, block_align
|
||||
|
||||
|
||||
def _write_header(stream, channel_count, sample_count, rate, sample_format):
|
||||
fmt, bits, bytes_per_sample, block_align = _fmt_chunk(
|
||||
channel_count, rate, sample_format)
|
||||
data_size = sample_count * block_align
|
||||
riff_file_size = 12 + 8 + len(fmt) + 8 + data_size
|
||||
use_rf64 = riff_file_size - 8 > 0xFFFFFFFF
|
||||
if use_rf64:
|
||||
# RF64 + ds64 + fmt + data.
|
||||
file_size = 12 + 36 + 8 + len(fmt) + 8 + data_size
|
||||
stream.write(b"RF64")
|
||||
stream.write(struct.pack("<I", 0xFFFFFFFF))
|
||||
stream.write(b"WAVE")
|
||||
stream.write(b"ds64")
|
||||
stream.write(struct.pack("<IQQQI", 28, file_size - 8, data_size, sample_count, 0))
|
||||
else:
|
||||
stream.write(b"RIFF")
|
||||
stream.write(struct.pack("<I", riff_file_size - 8))
|
||||
stream.write(b"WAVE")
|
||||
stream.write(b"fmt ")
|
||||
stream.write(struct.pack("<I", len(fmt)))
|
||||
stream.write(fmt)
|
||||
stream.write(b"data")
|
||||
stream.write(struct.pack("<I", 0xFFFFFFFF if use_rf64 else data_size))
|
||||
return {
|
||||
"format": sample_format,
|
||||
"bits_per_sample": bits,
|
||||
"bytes_per_sample": bytes_per_sample,
|
||||
"block_align": block_align,
|
||||
"data_bytes": data_size,
|
||||
"rf64": use_rf64,
|
||||
}
|
||||
|
||||
|
||||
def _pack_int24(values):
|
||||
scaled = (np.clip(values, -1.0, 1.0) * np.float32(8388607.0)).astype(np.int32)
|
||||
unsigned = scaled.reshape(-1).view(np.uint32)
|
||||
packed = np.empty((unsigned.size, 3), dtype=np.uint8)
|
||||
packed[:, 0] = unsigned & 0xFF
|
||||
packed[:, 1] = (unsigned >> 8) & 0xFF
|
||||
packed[:, 2] = (unsigned >> 16) & 0xFF
|
||||
return packed.tobytes()
|
||||
|
||||
|
||||
def write_speaker_wav(path, pcm, sample_format, *, rate=48000, chunk_samples=262144):
|
||||
"""Write an interleaved float32 array/memmap as float32 or PCM24 WAV."""
|
||||
target = Path(path)
|
||||
values = np.asarray(pcm)
|
||||
if values.ndim != 2:
|
||||
raise ValueError(f"speaker PCM must be 2D, got {values.shape}")
|
||||
sample_count, channel_count = values.shape
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
with target.open("wb") as stream:
|
||||
info = _write_header(
|
||||
stream, channel_count, sample_count, int(rate), sample_format)
|
||||
for start in range(0, sample_count, int(chunk_samples)):
|
||||
block = np.asarray(values[start:start + chunk_samples], dtype="<f4")
|
||||
if sample_format == "float32":
|
||||
stream.write(block.tobytes(order="C"))
|
||||
else:
|
||||
stream.write(_pack_int24(block))
|
||||
info.update({
|
||||
"path": str(target.resolve()),
|
||||
"sample_rate": int(rate),
|
||||
"sample_count": int(sample_count),
|
||||
"channel_count": int(channel_count),
|
||||
"file_bytes": target.stat().st_size,
|
||||
})
|
||||
return info
|
||||
@@ -0,0 +1,65 @@
|
||||
"""为未覆盖的 E-AC-3 JOC/EMDF 变体生成结构化错误和维修报告。"""
|
||||
import hashlib
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def bytes_descriptor(data):
|
||||
"""返回足以识别载荷、但不会复制整段载荷的摘要。"""
|
||||
raw = bytes(data)
|
||||
return {
|
||||
"bytes": len(raw),
|
||||
"sha256": hashlib.sha256(raw).hexdigest(),
|
||||
"prefix_hex": raw[:32].hex(),
|
||||
"suffix_hex": raw[-16:].hex() if raw else "",
|
||||
}
|
||||
|
||||
|
||||
class UnsupportedVariantError(ValueError):
|
||||
"""表示输入结构有效或疑似有效,但当前解析器没有覆盖该变体。"""
|
||||
|
||||
def __init__(self, stage, variant, message, *, frame=None, details=None):
|
||||
self.stage = str(stage)
|
||||
self.variant = str(variant)
|
||||
self.message = str(message)
|
||||
self.frame = frame
|
||||
self.details = dict(details or {})
|
||||
super().__init__(self.__str__())
|
||||
|
||||
def add_context(self, *, frame=None, details=None):
|
||||
if self.frame is None and frame is not None:
|
||||
self.frame = int(frame)
|
||||
if details:
|
||||
for key, value in details.items():
|
||||
self.details.setdefault(key, value)
|
||||
self.args = (self.__str__(),)
|
||||
return self
|
||||
|
||||
def to_dict(self):
|
||||
return {
|
||||
"report_schema": 1,
|
||||
"error": "unsupported_variant",
|
||||
"stage": self.stage,
|
||||
"variant": self.variant,
|
||||
"frame": self.frame,
|
||||
"message": self.message,
|
||||
"details": self.details,
|
||||
}
|
||||
|
||||
def __str__(self):
|
||||
where = f" frame={self.frame}" if self.frame is not None else ""
|
||||
detail = json.dumps(self.details, ensure_ascii=False, separators=(",", ":"))
|
||||
return f"[{self.stage}/{self.variant}]{where} {self.message}; details={detail}"
|
||||
|
||||
|
||||
def write_variant_report(path, error, *, input_path=None, output_path=None):
|
||||
"""将变体错误写成可交给后续维修工具的 JSON。"""
|
||||
report = error.to_dict()
|
||||
if input_path is not None:
|
||||
report["input"] = str(Path(input_path))
|
||||
if output_path is not None:
|
||||
report["requested_output"] = str(Path(output_path))
|
||||
target = Path(path)
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
target.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
return target
|
||||
Reference in New Issue
Block a user