Files

939 lines
40 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""Phase 5 verify — TASKS_DAW_A.md §39-47 (chuẩn DAW checklist).
5.3 Golden live=export: realtime SHM loop vs offline --render-fx cùng chain.
5.4 Order: chain [eq, vst3-delay, limiter] giữ thứ tự (contract) + impulse
qua adelay (VST delay áp giữa chain, không gom cuối, không bỏ).
5.5 Fader: master chain active + fader -6dB → đỉnh giảm 2x, chain vẫn áp.
5.6 PDC: slot latency_samples=1024 + audio bị delay 1024 → bù track → xuyên
pha hết (sum = 2×B); control không PDC → lệch pha.
5.7 Smoke: /capabilities + render master imager+maximizer+vst3.
Skip nếu thiếu fx_vst_bridge.exe (native_bridge/build/Release) — ngoại trừ
các test contract thuần (không spawn bridge).
"""
import io
import json
import math
import os
import subprocess
import sys
import tempfile
import time
import wave
import numpy as np
import pytest
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, REPO)
FX_BRIDGE = os.path.join(REPO, "native_bridge", "build", "Release", "fx_vst_bridge.exe")
DAW_BRIDGE = os.path.join(REPO, "native_bridge", "build", "Release", "daw_vst_bridge.exe")
ADELAY = os.path.join(REPO, "native_bridge", "build", "VST3", "Release", "adelay.vst3")
SR = 44100
REAL_SF2 = os.path.join(REPO, "app", "storage", "soundfonts", "518e850f-a5d3-4790-b1f9-0c90c203c524.sf2")
@pytest.fixture(scope="module")
def fx_bridge():
if not os.path.isfile(FX_BRIDGE):
pytest.skip("fx_vst_bridge.exe chưa build (native_bridge/build/Release)")
return FX_BRIDGE
@pytest.fixture(scope="module")
def adelay():
if not os.path.isfile(os.path.join(ADELAY, "Contents", "x86_64-win", "adelay.vst3")):
pytest.skip("adelay.vst3 sample chưa build (native_bridge/build/VST3/Release)")
return ADELAY
def _write_wav(path, data, sr):
pcm = (np.clip(data, -1.0, 1.0).T * 32767).astype(np.int16)
with wave.open(path, "wb") as w:
w.setnchannels(2)
w.setsampwidth(2)
w.setframerate(sr)
w.writeframes(pcm.tobytes())
def _read_wav(path):
import soundfile as sf
data, sr = sf.read(path, dtype="float32", always_2d=True)
return data.T, sr
def _render_bridge(bridge, wav_in, sr, chain, block=256):
job = {"sample_rate": sr, "block_size": block, "fx_chain": chain}
with tempfile.TemporaryDirectory() as td:
job_path = os.path.join(td, "job.json")
out_path = os.path.join(td, "out.wav")
with io.open(job_path, "w", encoding="utf-8") as f:
json.dump(job, f)
r = subprocess.run([bridge, "--render-fx", job_path, "--in", wav_in,
"--out", out_path],
capture_output=True, text=True, timeout=180)
assert r.returncode == 0, f"bridge rc={r.returncode}: {r.stdout[-500:]} {r.stderr[-500:]}"
assert os.path.isfile(out_path), f"no out wav: {r.stdout[-500:]} {r.stderr[-500:]}"
return _read_wav(out_path)[0]
def _snr_db(ref, got):
ref = ref.astype(np.float64)
got = got.astype(np.float64)
n = min(ref.shape[1], got.shape[1])
if n == 0:
return -float("inf")
ref, got = ref[:, :n], got[:, :n]
denom = float(np.sum((ref - got) ** 2))
if denom < 1e-12:
return float("inf")
return 10.0 * math.log10(float(np.sum(ref ** 2)) / denom)
def _find_delay(live, ref, max_lag=2048):
a = live[0].astype(np.float64)
b = ref[0].astype(np.float64)
n = min(a.shape[0], b.shape[0])
a, b = a[:n], b[:n]
corr = np.correlate(a, b, mode="same")
lag = int(np.argmax(np.abs(corr))) - n // 2
if abs(lag) > max_lag:
return 0
return lag
def _midi_track():
return {
"id": "t1",
"type": "MIDI",
"instrument_id": "__no_such_plugin__",
"instrument_source": "vsti",
"fx_type": None,
"volume_db": 0.0,
"pan": 0.0,
"mute": False,
"items": [{
"type": "MIDI_ITEM",
"start_bar": 0.0,
"duration_bars": 1.0,
"clip_start_offset_bars": 0.0,
"source_data": {"notes": [
{"pitch": 60, "velocity": 0.8, "start_beat": 0.0, "duration_beats": 1.0}
]},
}],
}
def _side_rms(path):
import soundfile as sf
data, _ = sf.read(path, dtype="float32", always_2d=True)
side = data[:, 0] - data[:, 1]
return float(np.sqrt(np.mean(side ** 2)))
# ── 5.3 Golden live=export ────────────────────────────────────────────────
def test_53_live_equals_export(fx_bridge, monkeypatch):
"""Realtime SHM loop vs offline --render-fx, cùng chain builtin, block 256
→ SNR ≥ 30 dB (cùng engine C++, live = export)."""
monkeypatch.setenv("SF_FX_BRIDGE_PATH", fx_bridge)
from app.core import fx_realtime
chain = [
{"type": "builtin", "id": "eq",
"params": {"g1": 3.0, "g2": -2.0, "g3": 1.0, "g4": 0.5}},
{"type": "builtin", "id": "compressor",
"params": {"threshold": -18.0, "ratio": 3.0, "makeup": 2.0}},
{"type": "builtin", "id": "limiter", "params": {"ceiling": -3.0}},
]
n = 172 * 256 # bội của block 256
t = np.arange(n) / SR
rng = np.random.default_rng(7)
sig = (0.2 * np.sin(2 * np.pi * 220 * t) + 0.1 * np.sin(2 * np.pi * 1100 * t)
+ 0.02 * rng.standard_normal(n))
audio = np.stack([sig, 0.9 * sig]).astype(np.float32)
with tempfile.TemporaryDirectory() as td:
wav_in = os.path.join(td, "in.wav")
_write_wav(wav_in, audio, SR)
offline = _render_bridge(fx_bridge, wav_in, SR, chain, block=256)
sess = fx_realtime.start_session(chain, SR)
try:
out = bytearray()
nb = n // 256
for i in range(nb):
blk = audio[:, i * 256:(i + 1) * 256]
inter = np.empty(512, dtype=np.float32)
inter[0::2] = blk[0]
inter[1::2] = blk[1]
sess.write_input(inter.tobytes())
time.sleep(0.006)
for b in sess.read_output():
out += b
time.sleep(0.05)
for b in sess.read_output():
out += b
finally:
sess.close()
arr = np.frombuffer(bytes(out), dtype=np.float32)
assert arr.size > 0, "realtime loop không ra output"
live = arr.reshape(-1, 2).T
assert live.shape[1] >= n // 2, f"live quá ngắn: {live.shape}"
lag = _find_delay(live, offline)
if lag >= 0:
got = live[:, lag:lag + offline.shape[1]]
else:
got = live[:, :offline.shape[1] + lag]
n_cmp = min(offline.shape[1], got.shape[1]) - 8
assert n_cmp > 1000, "đoạn so sánh quá ngắn"
snr = _snr_db(offline[:, :n_cmp], got[:, :n_cmp])
assert snr >= 30.0, f"live≠export SNR {snr:.1f} dB (lag {lag})"
# ── 5.4 Order ─────────────────────────────────────────────────────────────
def test_54_chain_order_contract():
"""Chain [eq, vst3-delay, limiter] giữ NGUYÊN thứ tự qua merge offline +
live JSON — không gom VST cuối (regression cho "không gom cuối")."""
from app.core import fx_realtime
from app.core.render_engine import _merge_track_fx_chain
track = {
"id": "t1",
"fx_chain": [
{"type": "builtin", "id": "eq", "params": {"g1": 3.0}},
{"type": "vst3", "path": ADELAY, "active": True},
{"type": "builtin", "id": "limiter", "params": {"ceiling": -3.0}},
],
"vst_fx_chain": [],
"vst_fx_bypass": False,
}
merged = _merge_track_fx_chain(track)
assert [s.get("id") or s.get("type") for s in merged] == ["eq", "vst3", "limiter"]
s = object.__new__(fx_realtime.FxRealtimeSession)
s.chain = list(merged)
s._lock = None
out = s._chain_json_for_bridge()
assert [o["type"] for o in out] == ["builtin", "vst3", "builtin"]
assert out[1]["path"] == ADELAY
def test_54_vst3_delay_in_middle_audio(fx_bridge, adelay):
"""Impulse qua [eq, adelay(default 1s), limiter] → peak tại ≈44100, im lặng
trước đó; SNR vs reference Python (eq → shift 44100 → limiter) ≥ 18 dB —
(EQ Python vs C++ impulse response lệch ~22 dB; check order thật là vị trí
peak + pre-silence) — VST delay áp giữa chain (sau EQ, trước limiter)."""
from app.core.render_engine import _apply_builtin_fx_chain
chain = [
{"type": "builtin", "id": "eq",
"params": {"g1": 3.0, "g2": -2.0, "g3": 1.0, "g4": 0.5}},
{"type": "vst3", "path": adelay},
{"type": "builtin", "id": "limiter", "params": {"ceiling": -3.0}},
]
n = int(SR * 1.5)
audio = np.zeros((2, n), dtype=np.float32)
audio[0, 0] = 1.0
audio[1, 0] = 1.0
with tempfile.TemporaryDirectory() as td:
wav_in = os.path.join(td, "in.wav")
_write_wav(wav_in, audio, SR)
got = _render_bridge(fx_bridge, wav_in, SR, chain, block=256)
idx = int(np.unravel_index(np.argmax(np.abs(got)), got.shape)[1])
assert abs(idx - SR) < 200, f"peak tại {idx} — mong đợi ~{SR} (delay 1s)"
assert float(np.max(np.abs(got[:, :int(SR * 0.9)]))) == 0.0, "có tín hiệu trước delay"
ref = _apply_builtin_fx_chain(
audio.copy(),
[{"type": "eq", "params": {"g1": 3.0, "g2": -2.0, "g3": 1.0, "g4": 0.5}}], SR)
ref = np.roll(ref, SR, axis=1)
ref = _apply_builtin_fx_chain(ref, [{"type": "limiter", "params": {"ceiling": -3.0}}], SR)
n_cmp = min(ref.shape[1], got.shape[1]) - 8
snr = _snr_db(ref[:, :n_cmp], got[:, :n_cmp])
assert snr >= 18.0, f"SNR {snr:.1f} dB"
# ── 5.5 Fader sau master chain ────────────────────────────────────────────
def test_55_fader_after_master_chain(monkeypatch, tmp_path, fx_bridge):
"""Master chain active (gain +6dB) + fader -6dB → đỉnh giảm 2x, chain vẫn
áp (đỉnh fader-6 ≈ base, đỉnh fader0 ≈ 2×base) — fader SAU chain."""
monkeypatch.setenv("SF_FX_BRIDGE_PATH", fx_bridge)
if not os.path.isfile(DAW_BRIDGE):
pytest.skip("daw_vst_bridge.exe chưa build")
monkeypatch.setenv("SF_BRIDGE_PATH", DAW_BRIDGE)
from app.api.v1 import plugins as pl
monkeypatch.setattr(pl.settings, "RENDER_ENGINE", "bridge")
from app.core.render_engine import PythonRenderEngine
engine = PythonRenderEngine()
def project(master_vol, chain):
track = _midi_track()
track["volume_db"] = -20.0 # tránh normalize (>1.0)
return {
"metadata": {"bpm": 120.0, "time_signature_numerator": 4},
"main_session": {
"length_bars": 1.0,
"tracks": [track],
"master": {
"fx_chain": chain,
"bypass": False,
"volume_db": master_vol,
},
},
"section_store": {},
}
GAIN = [{"type": "builtin", "id": "gain", "params": {"db": 6.0}}]
def peak(name):
chain = {"base.wav": [], "0.wav": GAIN, "6.wav": GAIN}[name]
vol = {"base.wav": 0.0, "0.wav": 0.0, "6.wav": -6.0}[name]
out = str(tmp_path / name)
engine.render_project(project(vol, chain), out)
data, _ = _read_wav(out)
return float(np.max(np.abs(data)))
base = peak("base.wav")
p0 = peak("0.wav")
p6 = peak("6.wav")
assert p0 > 1.7 * base, f"chain +6dB không áp: {p0:.4f} vs base {base:.4f}"
assert 0.4 < p6 / p0 < 0.6, f"fader -6dB phải ~1/2 đỉnh: {p6:.4f}/{p0:.4f}"
assert 0.8 < p6 / base < 1.2, f"fader-6 phải ≈ base: {p6:.4f} vs {base:.4f}"
# ── 5.8 Master L/R volume + LINK + MUTE ─────────────────────────────────
def test_58_master_lr_link_mute(monkeypatch, tmp_path, fx_bridge):
"""Master strip L/R: volumeL_db/volumeR_db tách kênh, link_lr → R theo L,
muted → zero output. Fallback volume_db (project cũ) vẫn chạy."""
monkeypatch.setenv("SF_FX_BRIDGE_PATH", fx_bridge)
if not os.path.isfile(DAW_BRIDGE):
pytest.skip("daw_vst_bridge.exe chưa build")
monkeypatch.setenv("SF_BRIDGE_PATH", DAW_BRIDGE)
from app.api.v1 import plugins as pl
monkeypatch.setattr(pl.settings, "RENDER_ENGINE", "bridge")
from app.core.render_engine import PythonRenderEngine
engine = PythonRenderEngine()
def project(master):
track = _midi_track()
track["volume_db"] = -20.0 # tránh normalize (>1.0)
return {
"metadata": {"bpm": 120.0, "time_signature_numerator": 4},
"main_session": {
"length_bars": 1.0,
"tracks": [track],
"master": dict({"fx_chain": [], "bypass": False}, **master),
},
"section_store": {},
}
def render(name, master):
out = str(tmp_path / name)
engine.render_project(project(master), out)
data, _ = _read_wav(out)
return data
base = render("base.wav", {"volumeL_db": 0.0, "volumeR_db": 0.0, "link_lr": True})
baseL = float(np.max(np.abs(base[0])))
baseR = float(np.max(np.abs(base[1])))
assert baseL > 0
lr = render("lr.wav", {"volumeL_db": -6.0, "volumeR_db": 0.0, "link_lr": False})
assert 0.4 < float(np.max(np.abs(lr[0]))) / baseL < 0.6, f"L -6dB phải ~1/2 đỉnh: {float(np.max(np.abs(lr[0]))):.4f}/{baseL:.4f}"
assert abs(float(np.max(np.abs(lr[1]))) - baseR) / baseR < 0.15, "R phải giữ nguyên khi link_lr=false"
link = render("link.wav", {"volumeL_db": -6.0, "volumeR_db": -3.0, "link_lr": True})
ratio = float(np.max(np.abs(link[1]))) / float(np.max(np.abs(link[0])))
assert 0.9 < ratio < 1.1, f"link_lr: R phải theo L ({ratio:.3f})"
muted = render("muted.wav", {"volumeL_db": 0.0, "volumeR_db": 0.0,
"link_lr": True, "muted": True})
assert float(np.max(np.abs(muted))) < 1e-6, "MUTE master phải zero output"
legacy = render("legacy.wav", {"volume_db": -6.0})
assert 0.4 < float(np.max(np.abs(legacy[0]))) / baseL < 0.6 and 0.4 < float(np.max(np.abs(legacy[1]))) / baseR < 0.6, "fallback volume_db (project cũ) phải áp cả 2 kênh"
# ── 5.6 PDC ───────────────────────────────────────────────────────────────
def test_56_pdc_aligns_tracks(monkeypatch, tmp_path, fx_bridge):
"""Plugin latency 1024 (slot latency_samples) + audio thật bị delay 1024 →
PDC bù track ngắn → sum = 2×B (xuyên pha hết). Control latency=0 → sum
lệch pha (SNR thấp)."""
monkeypatch.setenv("SF_FX_BRIDGE_PATH", fx_bridge)
from app.api.v1 import plugins as pl
monkeypatch.setattr(pl.settings, "RENDER_ENGINE", "bridge")
from app.core.render_engine import PythonRenderEngine
engine = PythonRenderEngine()
n = SR * 2 # 1 bar @120bpm 4/4 = 88200
t = np.arange(n) / SR
w = (0.2 * np.sin(2 * np.pi * 220 * t) + 0.1 * np.sin(2 * np.pi * 1100 * t)).astype(np.float32)
ws = np.zeros_like(w)
ws[1024:] = w[:n - 1024]
fa = str(tmp_path / "a.wav")
fb = str(tmp_path / "b.wav")
_write_wav(fa, np.stack([w, w]), SR)
_write_wav(fb, np.stack([ws, ws]), SR)
def project(lat):
def track(path, chain):
return {
"id": path,
"type": "AUDIO",
"volume_db": 0.0,
"pan": 0.0,
"fx_chain": chain,
"vst_fx_chain": [],
"vst_fx_bypass": False,
"items": [{
"type": "AUDIO_ITEM",
"start_bar": 0.0,
"duration_bars": 1.0,
"clip_start_offset_bars": 0.0,
"source_data": {"audio_file_url": path},
}],
}
chain_b = []
if lat:
# bypass passthrough — chỉ khai latency cho PDC (spec §3 field)
chain_b = [{"type": "vst3", "path": "", "bypass": True,
"active": True, "latency_samples": lat}]
return {
"metadata": {"bpm": 120.0, "time_signature_numerator": 4},
"main_session": {
"length_bars": 1.0,
"tracks": [track(fa, []), track(fb, chain_b)],
},
"section_store": {},
}
out_pdc = str(tmp_path / "pdc.wav")
out_nopdc = str(tmp_path / "nopdc.wav")
engine.render_project(project(1024), out_pdc)
engine.render_project(project(0), out_nopdc)
d_pdc, _ = _read_wav(out_pdc)
d_nopdc, _ = _read_wav(out_nopdc)
expected = np.stack([ws, ws]) * 2.0 # PDC delay track A → cùng pha B
n_cmp = n - 16
snr_pdc = _snr_db(expected[:, :n_cmp], d_pdc[:, :n_cmp])
snr_nopdc = _snr_db(expected[:, :n_cmp], d_nopdc[:, :n_cmp])
assert snr_pdc >= 30.0, f"PDC sum lệch: SNR {snr_pdc:.1f} dB"
assert snr_nopdc < 15.0, f"control không lệch pha: SNR {snr_nopdc:.1f} dB"
# ── 5.7 Smoke ─────────────────────────────────────────────────────────────
def test_57_capabilities_smoke():
"""Server /capabilities phản hồi OK (smoke API)."""
from fastapi.testclient import TestClient
from app.main import app
r = TestClient(app).get("/api/v1/system/capabilities")
assert r.status_code == 200, r.text[:300]
assert isinstance(r.json(), dict)
def test_57_master_chain_imager_maximizer_vst3(monkeypatch, tmp_path, fx_bridge, adelay):
"""Render master imager+maximizer+vst3 (adelay) qua bridge — không crash,
non-silent; imager vẫn tác dụng khi chain có vst3."""
monkeypatch.setenv("SF_FX_BRIDGE_PATH", fx_bridge)
if not os.path.isfile(DAW_BRIDGE):
pytest.skip("daw_vst_bridge.exe chưa build")
monkeypatch.setenv("SF_BRIDGE_PATH", DAW_BRIDGE)
from app.api.v1 import plugins as pl
monkeypatch.setattr(pl.settings, "RENDER_ENGINE", "bridge")
from app.core.render_engine import PythonRenderEngine
engine = PythonRenderEngine()
def render(chain, name):
track = _midi_track()
track["volume_db"] = -20.0
track["pan"] = 0.6
proj = {
"metadata": {"bpm": 120.0, "time_signature_numerator": 4},
"main_session": {"length_bars": 1.0, "tracks": [track]},
"section_store": {},
}
if chain is not None:
proj["mastering_settings"] = {
"chain": chain,
"w1": 0, "w2": 0, "w3": 0, "w4": 0,
"maxGain": 0.0, "maxSoftClip": False, "maxUpward": False,
"ceiling": -1.0,
}
out = str(tmp_path / name)
engine.render_project(proj, out)
return out
base = _side_rms(render(None, "base.wav"))
s0 = _side_rms(render([
{"id": "imager", "type": "imager", "name": "Imager", "active": True},
{"type": "vst3", "path": adelay},
{"id": "maximizer", "type": "maximizer", "name": "Max", "active": True},
], "im0.wav"))
s2 = _side_rms(render([
{"id": "imager", "type": "imager", "name": "Imager", "active": True,
"params": {"w1": 200, "w2": 200, "w3": 200, "w4": 200}},
{"type": "vst3", "path": adelay},
{"id": "maximizer", "type": "maximizer", "name": "Max", "active": True},
], "im200.wav"))
assert base > 0.001, "pan tạo side content"
assert s0 < 0.05 * base, f"imager w=0 không giết side: {s0:.5f} vs {base:.5f}"
assert s2 > 1.5 * base, f"imager w=200 không tăng side: {s2:.5f} vs {base:.5f}"
def test_59_limiter_brickwall_oversample(fx_bridge):
"""D2: LimiterFx mode brickwall — lookahead + oversample 4x + hard clip.
Impulse 0dB qua ceiling -1dB: peak <= ceiling + 0.005, length giu nguyen;
soft mode giu tanh (peak ~1.0); low-level qua brickwall ~ unity."""
n = 4096
imp = np.zeros(n, dtype=np.float32)
imp[512] = 1.0
audio = np.stack([imp, imp * 0.8]).astype(np.float32)
ceil_lin = 10.0 ** (-1.0 / 20.0)
brick = [{"type": "builtin", "id": "limiter",
"params": {"ceiling": -1.0, "mode": 1, "lookahead_ms": 2.0}}]
brick_str = [{"type": "builtin", "id": "limiter",
"params": {"ceiling": -1.0, "mode": "brickwall", "lookahead_ms": 3.0}}]
soft = [{"type": "builtin", "id": "limiter",
"params": {"ceiling": -1.0}}]
with tempfile.TemporaryDirectory() as td:
wav_in = os.path.join(td, "in.wav")
_write_wav(wav_in, audio, SR)
out_brick = _render_bridge(fx_bridge, wav_in, SR, brick, block=512)
out_brick_str = _render_bridge(fx_bridge, wav_in, SR, brick_str, block=512)
out_soft = _render_bridge(fx_bridge, wav_in, SR, soft, block=512)
assert out_brick.shape == audio.shape, "brickwall length giu nguyen"
pk = float(np.max(np.abs(out_brick)))
assert pk <= ceil_lin + 0.005, f"brickwall peak {pk:.4f} > ceiling {ceil_lin:.4f}"
assert pk >= 0.70, f"brickwall khong giam dinh? peak {pk:.4f}"
pk_str = float(np.max(np.abs(out_brick_str)))
assert pk_str <= ceil_lin + 0.005, f"mode string peak {pk_str:.4f}"
pk_soft = float(np.max(np.abs(out_soft)))
assert pk_soft > 0.95, f"soft peak {pk_soft:.4f} (tanh normalized ~1.0)"
assert pk_soft <= 1.0005, f"soft peak {pk_soft:.4f} vuot 1.0"
imp2 = np.zeros(n, dtype=np.float32)
imp2[2048] = 0.1
audio2 = np.stack([imp2, imp2]).astype(np.float32)
with tempfile.TemporaryDirectory() as td:
wav_in2 = os.path.join(td, "in2.wav")
_write_wav(wav_in2, audio2, SR)
out_low = _render_bridge(fx_bridge, wav_in2, SR, brick, block=512)
pk_low = float(np.max(np.abs(out_low)))
assert 0.05 <= pk_low <= 0.15, f"low-level peak {pk_low:.4f} lech unity"
def test_61_maximizer_oversample(fx_bridge):
"""Gap 4: MaximizerFx oversample 4x — impulse delay 8 mẫu (2×FIR33 group
delay 16@4x / 4) chứng minh oversample active; sine -20dB boost 0 trong
suốt (unity); 5kHz soft_clip 100: harmonic 3 (15k) in-band giữ > -40dBFS
còn alias spur 9.1k (fold cua harmonic 7 = 35k) bi FIR33 anti-alias chặn
< -50dBFS; hot sine bám ceiling -1dB."""
def _bin_db(x, f):
x0 = x[0].astype(np.float64)
n = len(x0)
win = np.hanning(n)
X = np.fft.rfft((x0 - np.mean(x0)) * win)
freq = np.fft.rfftfreq(n, 1.0 / SR)
i = int(np.argmin(np.abs(freq - f)))
amp = np.abs(X[i]) * 2.0 / np.sum(win)
return 20.0 * np.log10(amp + 1e-12)
mx = [{"type": "builtin", "id": "maximizer",
"params": {"boost_db": 0.0, "soft_clip": 0.0, "upward": 0.0,
"ceiling_db": -1.0}}]
n = 8192
t = np.arange(n) / SR
# 1) impulse → delay 8 (oversample FIR chain active), gain ~1
imp = np.zeros(n, dtype=np.float32)
imp[512] = 1.0
audio = np.stack([imp, imp]).astype(np.float32)
with tempfile.TemporaryDirectory() as td:
wav_in = os.path.join(td, "in.wav")
_write_wav(wav_in, audio, SR)
out = _render_bridge(fx_bridge, wav_in, SR, mx, block=512)
argmax = int(np.argmax(np.abs(out[0])))
assert argmax == 512 + 8, f"impulse delay {argmax - 512} (expect 8)"
pk = float(np.max(np.abs(out[0])))
assert 0.8 <= pk <= 1.2, f"impulse peak {pk:.4f} (expect ~1.0)"
# 2) trong suốt: -20dBFS 1kHz qua maximizer gain 0 → unity
sine = (0.1 * np.sin(2 * np.pi * 1000.0 * t)).astype(np.float32)
audio = np.stack([sine, sine])
with tempfile.TemporaryDirectory() as td:
wav_in = os.path.join(td, "in.wav")
_write_wav(wav_in, audio, SR)
out2 = _render_bridge(fx_bridge, wav_in, SR, mx, block=512)
pk2 = float(np.max(np.abs(out2[0])))
assert 0.09 <= pk2 <= 0.11, f"transparent peak {pk2:.4f} (expect 0.1)"
# 3) soft_clip 100: 15k in-band giữ, alias 9.1k bị chặn
mx_hot = [{"type": "builtin", "id": "maximizer",
"params": {"boost_db": 0.0, "soft_clip": 100.0, "upward": 0.0,
"ceiling_db": -1.0}}]
sine5 = (0.7 * np.sin(2 * np.pi * 5000.0 * t)).astype(np.float32)
audio = np.stack([sine5, sine5])
with tempfile.TemporaryDirectory() as td:
wav_in = os.path.join(td, "in.wav")
_write_wav(wav_in, audio, SR)
out3 = _render_bridge(fx_bridge, wav_in, SR, mx_hot, block=512)
assert out3.shape == audio.shape, "length giu nguyen"
h3 = _bin_db(out3, 15000.0)
alias = _bin_db(out3, 9100.0)
print(f"maximizer: h3@15k={h3:.1f}dBFS alias@9.1k={alias:.1f}dBFS")
assert h3 > -40.0, f"3rd harmonic 15k {h3:.1f}dBFS qua thap"
assert alias < -50.0, f"alias spur 9.1k {alias:.1f}dBFS (oversample phai chan fold 35k)"
# 4) hot sine bám ceiling -1dB
ceil_lin = 10.0 ** (-1.0 / 20.0)
sine_hot = (0.9 * np.sin(2 * np.pi * 1000.0 * t)).astype(np.float32)
audio = np.stack([sine_hot, sine_hot])
mx_boost = [{"type": "builtin", "id": "maximizer",
"params": {"boost_db": 18.0, "soft_clip": 100.0, "upward": 0.0,
"ceiling_db": -1.0}}]
with tempfile.TemporaryDirectory() as td:
wav_in = os.path.join(td, "in.wav")
_write_wav(wav_in, audio, SR)
out4 = _render_bridge(fx_bridge, wav_in, SR, mx_boost, block=512)
pk4 = float(np.max(np.abs(out4[0])))
assert pk4 <= ceil_lin + 0.005, f"ceiling peak {pk4:.4f} > {ceil_lin:.4f}"
assert pk4 >= 0.7, f"ceiling peak {pk4:.4f} qua thap"
# ── 5.C Realtime VST3 setParam live (PLAN_MASTER_ENHANCE C) ───────────────
def test_60_vst3_setparam_live(fx_bridge, adelay, monkeypatch):
"""set_param runtime cho VST3: chain [vst3-delay, builtin limiter], đổi
param 'delay' giữa stream 1.0 → 0.25 → echo dịch ~1s → ~0.25s NGAY trên
session hiện tại (restart → plugin default 1s → echo vẫn ở 1s)."""
monkeypatch.setenv("SF_FX_BRIDGE_PATH", fx_bridge)
from app.core import fx_realtime
chain = [
{"type": "vst3", "path": adelay},
{"type": "builtin", "id": "limiter", "params": {"ceiling": -3.0}},
]
sess = fx_realtime.start_session(chain, SR)
try:
out = bytearray()
nblk = 280 # 1.24s — đủ phủ vùng echo cũ (1s) để chứng minh không còn
for i in range(nblk):
blk = np.zeros((2, 256), dtype=np.float32)
if i == 103: # impulse2 tại 0.6s (26368)
blk[0, 0] = 1.0
blk[1, 0] = 1.0
if i == 86: # 0.5s — setParam VST3 'delay' 1.0 -> 0.25 (normalized)
sess.set_param(0, "delay", 0.25)
inter = np.empty(512, dtype=np.float32)
inter[0::2] = blk[0]
inter[1::2] = blk[1]
sess.write_input(inter.tobytes())
time.sleep(0.006)
for b in sess.read_output():
out += b
time.sleep(0.05)
for b in sess.read_output():
out += b
finally:
sess.close()
arr = np.frombuffer(bytes(out), dtype=np.float32)
assert arr.size > 0, "realtime loop không ra output"
got = arr.reshape(-1, 2).T
def _peak(lo, hi):
if lo >= got.shape[1]:
return 0.0
return float(np.max(np.abs(got[:, lo:hi])))
e2 = 103 * 256 + int(0.25 * SR) # echo delay 0.25s ≈ 37393
assert _peak(e2 - 400, e2 + 400) > 0.5, f"echo delay 0.25s không thấy tại {e2} (peak {_peak(e2-400, e2+400):.3f})"
e_old = 103 * 256 + SR # echo delay cũ 1s ≈ 70468 (restart/no-apply)
assert _peak(e_old - 400, e_old + 400) < 0.3, "param không áp live — echo vẫn ở delay 1s"
assert _peak(0, e2 - 700) < 0.3, "có tín hiệu lạ trước echo 0.25s"
# ── 5.9 Multiband compressor + De-Esser (PLAN_MASTER_ENHANCE B.10) ──────
def test_62_multiband_deesser(fx_bridge):
"""Multiband: GR on active band + unity transparent (LR4 flat).
DeEsser: GR at 8kHz (sibilance), 500Hz untouched."""
def peak_db(x):
# steady-state: nua cuoi (bo transient khoi dau cua HP/crossover filters)
m = float(np.max(np.abs(x[:, x.shape[1] // 2:])))
return 20.0 * math.log10(max(m, 1e-12))
def render(chain, freq, amp):
t = np.arange(SR * 2) / SR
sine = (amp * np.sin(2 * np.pi * freq * t)).astype(np.float32)
audio = np.stack([sine, sine])
with tempfile.TemporaryDirectory() as td:
wav_in = os.path.join(td, "in.wav")
_write_wav(wav_in, audio, SR)
return _render_bridge(fx_bridge, wav_in, SR, chain, block=512)
# 1) multiband: low band comp 100Hz (thr -20 ratio 10) -> GR
mb_lo = [{"type": "builtin", "id": "multiband", "params": {"lf_cross": 200, "hf_cross": 4000,
"low_thr": -20, "low_ratio": 10, "low_makeup": 0, "mid_thr": -80, "mid_ratio": 1,
"mid_makeup": 0, "high_thr": -80, "high_ratio": 1, "high_makeup": 0}}]
out = render(mb_lo, 100.0, 0.5)
g = peak_db(out)
assert g < -12.0, f"low comp GR weak: {g:.1f}dBFS (expect <-12)"
# 2) multiband: unity transparent +-0.5dB (LR4 crossover flat)
mb_unity = [{"type": "builtin", "id": "multiband", "params": {"lf_cross": 200, "hf_cross": 4000,
"low_thr": -10, "low_ratio": 1, "low_makeup": 0, "mid_thr": -10, "mid_ratio": 1,
"mid_makeup": 0, "high_thr": -10, "high_ratio": 1, "high_makeup": 0}}]
for freq in (100.0, 1000.0, 8000.0):
out = render(mb_unity, freq, 0.5)
g = peak_db(out)
assert -6.5 <= g <= -5.5, f"multiband unity {freq}Hz: {g:.2f}dBFS (expect ~-6.02)"
# 3) deesser: 8kHz GR (thr -30 ratio 100 freq 6000)
de = [{"type": "builtin", "id": "deesser", "params": {"threshold": -30, "ratio": 100, "freq": 6000, "makeup": 0}}]
out = render(de, 8000.0, 0.7)
g = peak_db(out)
assert g < -15.0, f"deesser 8kHz GR weak: {g:.1f}dBFS (expect <-15)"
# 4) deesser: 500Hz unchanged +-0.5dB
out = render(de, 500.0, 0.5)
g = peak_db(out)
assert -6.5 <= g <= -5.5, f"deesser 500Hz changed: {g:.2f}dBFS (expect ~-6.02)"
# ── 5.63 Gap 8: PDC realtime — WS latency push ───────────────────────────
def test_63_fxrt_ws_latency_push(monkeypatch):
"""Gap 8: WS /ws/fx-realtime/{id} day text frame {cmd:'latency', total,
latencies} khi chain bao latency (SHM ring drain trong output pump) —
client bu delay dry-bypass path. Fake session bao {0:1024,1:0} → total 1024."""
from fastapi.testclient import TestClient
from app.core import fx_realtime
from app.main import app as main_app
class _FakeSess:
block_size = 256
sample_rate = 44100
def alive(self):
return True
def read_output(self):
return []
def get_latencies(self):
return {0: 1024, 1: 0}
def set_param(self, *a):
pass
def report_latency(self, *a):
pass
def write_input(self, b):
pass
monkeypatch.setattr(fx_realtime, "get_session", lambda sid: _FakeSess())
client = TestClient(main_app)
with client.websocket_connect("/api/v1/plugins/ws/fx-realtime/fake") as ws:
d = ws.receive_json()
assert d.get("cmd") == "latency", d
assert d.get("total") == 1024, d
assert d.get("latencies", {}).get("0") == 1024, d
# ── 5.10 Gap 12: SoundFont thật qua native bridge (thay pyfluidsynth) ─────
def test_64_soundfont_render_via_bridge(monkeypatch, tmp_path):
"""Gap 12: pyfluidsynth thiếu FluidSynth lib → native bridge render SF2 thật
(bridge nhúng FluidSynth C API, NativeInstrumentEngine.cpp).
(1) render_soundfont_offline trực tiếp: non-silent, đúng duration, program
khác → audio khác (program_select áp).
(2) render_project với HAS_PYFLUIDSYNTH=False: track MIDI soundfont ra âm
thật — không fallback oscillator synth."""
if not os.path.isfile(DAW_BRIDGE):
pytest.skip("daw_vst_bridge.exe chưa build (native_bridge/build/Release)")
if not os.path.isfile(REAL_SF2):
pytest.skip("không có SF2 mẫu trong app/storage/soundfonts")
monkeypatch.setenv("SF_BRIDGE_PATH", DAW_BRIDGE)
from app.core import native_render
notes = [{"pitch": 60, "velocity": 0.8, "start_beat": 0.0, "duration_beats": 2.0}]
out0, dur0 = native_render.render_soundfont_offline(REAL_SF2, notes, 120.0, SR, bank=0, program=0)
try:
d0, sr0 = _read_wav(out0)
assert sr0 == SR, f"sr {sr0}"
assert d0.shape[1] >= 44000, f"duration ngắn: {d0.shape[1]} (2 beats @120 = 44100)"
assert float(np.max(np.abs(d0))) > 0.01, "SF2 render im lặng"
finally:
os.remove(out0)
out40, _ = native_render.render_soundfont_offline(REAL_SF2, notes, 120.0, SR, bank=0, program=40)
try:
d40, _ = _read_wav(out40)
n = min(d0.shape[1], d40.shape[1])
assert float(np.max(np.abs(d40[:, :n] - d0[:, :n]))) > 1e-3, "program_select không áp — program 0 == 40"
finally:
os.remove(out40)
# (2) engine-level: bắt buộc branch bridge (pyfluidsynth vắng mặt)
import app.core.vst_engine as ve
import app.core.render_engine as re_mod
monkeypatch.setattr(ve, "HAS_PYFLUIDSYNTH", False)
monkeypatch.setattr(re_mod, "HAS_PYFLUIDSYNTH", False)
monkeypatch.setattr(re_mod, "_find_sf2_path", lambda sf_id: REAL_SF2)
from app.api.v1 import plugins as pl
monkeypatch.setattr(pl.settings, "RENDER_ENGINE", "bridge")
from app.core.render_engine import PythonRenderEngine
engine = PythonRenderEngine()
track = _midi_track()
track["instrument_id"] = ""
track["instrument_source"] = "soundfont"
track["soundfont_id"] = "518e850f-a5d3-4790-b1f9-0c90c203c524"
track["volume_db"] = -12.0
project = {
"metadata": {"bpm": 120.0, "time_signature_numerator": 4},
"main_session": {
"length_bars": 1.0,
"tracks": [track],
"master": {"fx_chain": [], "bypass": False, "volume_db": 0.0},
},
"section_store": {},
}
out = str(tmp_path / "sf2_engine.wav")
engine.render_project(project, out)
data, _ = _read_wav(out)
assert float(np.max(np.abs(data))) > 0.01, "engine render soundfont im lặng — fallback oscillator?"
# ── 5.11 Gap 7: master volume automation lane ─────────────────────────────
def test_65_master_volume_automation(monkeypatch, tmp_path, fx_bridge):
"""Gap 7: master.automation.volume_db [{time,db}] — envelope áp sau fader,
trước normalize: trước node 1s giữ nguyên, sau node 1s gain ~ -12dB (0.25)."""
monkeypatch.setenv("SF_FX_BRIDGE_PATH", fx_bridge)
if not os.path.isfile(DAW_BRIDGE):
pytest.skip("daw_vst_bridge.exe chưa build")
monkeypatch.setenv("SF_BRIDGE_PATH", DAW_BRIDGE)
from app.api.v1 import plugins as pl
monkeypatch.setattr(pl.settings, "RENDER_ENGINE", "bridge")
from app.core.render_engine import PythonRenderEngine
engine = PythonRenderEngine()
def project(auto_nodes):
track = _midi_track()
track["items"][0]["source_data"]["notes"][0]["duration_beats"] = 4.0 # phủ cả 2s
track["volume_db"] = -12.0 # tránh normalize (>1.0)
p = {
"metadata": {"bpm": 120.0, "time_signature_numerator": 4},
"main_session": {
"length_bars": 1.0,
"tracks": [track],
"master": {"fx_chain": [], "bypass": False, "volume_db": 0.0},
},
"section_store": {},
}
if auto_nodes is not None:
p["main_session"]["master"]["automation"] = {"volume_db": auto_nodes}
return p
def rms(path, start_s, dur_s):
data, sr = _read_wav(path)
i0, i1 = int(start_s * sr), int((start_s + dur_s) * sr)
return float(np.sqrt(np.mean(data[:, i0:i1] ** 2)))
out_base = str(tmp_path / "auto_base.wav")
out_env = str(tmp_path / "auto_env.wav")
engine.render_project(project(None), out_base)
# Hai đoạn phẳng + chuyển nhanh 0dB→-12dB quanh t=1s (interp dB tuyến tính
# giữa node liền kề; đoạn phẳng không bị ảnh hưởng bởi ramp lân cận).
engine.render_project(project([
{"time": 0.0, "db": 0.0}, {"time": 0.9, "db": 0.0},
{"time": 1.0, "db": -12.0}, {"time": 2.0, "db": -12.0},
]), out_env)
base0 = rms(out_base, 0.1, 0.6)
env0 = rms(out_env, 0.1, 0.6)
env1 = rms(out_env, 1.15, 0.6)
assert env0 == pytest.approx(base0, rel=0.05), f"đoạn 0dB phải giữ nguyên: {env0} vs {base0}"
ratio = env1 / max(env0, 1e-9)
assert 0.20 < ratio < 0.30, f"envelope -12dB → gain ~0.25: {ratio}"
def test_66_bwf_metadata(monkeypatch, tmp_path):
"""Gap 13: render_project(metadata=...) chen bext + LIST/INFO vao WAV —
title/artist/isrc doc lai duoc, PCM bytes khong doi."""
monkeypatch.delenv("SF_BRIDGE_PATH", raising=False)
from app.core.render_engine import PythonRenderEngine
engine = PythonRenderEngine()
def project():
track = _midi_track()
track["volume_db"] = -12.0
return {
"metadata": {"bpm": 120.0, "time_signature_numerator": 4},
"main_session": {
"length_bars": 1.0,
"tracks": [track],
"master": {"fx_chain": [], "bypass": False, "volume_db": 0.0},
},
"section_store": {},
}
def chunks(path):
with open(path, "rb") as f:
data = f.read()
assert data[:4] == b"RIFF" and data[8:12] == b"WAVE"
out = {}
off = 12
while off + 8 <= len(data):
cid = data[off:off + 4]
size = int.from_bytes(data[off + 4:off + 8], "little")
body = data[off + 8:off + 8 + size]
if cid == b"data":
out[cid] = body
break
out[cid] = body
off += 8 + size + (size % 2)
return out
out_plain = str(tmp_path / "plain.wav")
out_meta = str(tmp_path / "meta.wav")
engine.render_project(project(), out_plain)
engine.render_project(project(), out_meta, metadata={
"title": "Test Song",
"artist": "Penguin",
"isrc": "US-ABC-26-00001",
"comment": "Gap 13 check",
"description": "BWF description",
"originator": "SonicForgeStudio",
"originator_reference": "proj-123",
})
# Khong metadata → khong bext/LIST (backward compatible)
cp = chunks(out_plain)
assert b"bext" not in cp and b"LIST" not in cp
cm = chunks(out_meta)
assert b"bext" in cm and b"LIST" in cm
# bext: description (0:256) + originator (256:288) + originator_reference (288:320)
bext = cm[b"bext"]
assert bext[0:256].split(b"\x00")[0] == b"BWF description"
assert bext[256:288].split(b"\x00")[0] == b"SonicForgeStudio"
assert bext[288:320].split(b"\x00")[0] == b"proj-123"
# INFO: INAM/IART/ISRC subchunks
info = cm[b"LIST"]
assert info[:4] == b"INFO"
subs = {}
off = 4
while off + 8 <= len(info):
sid = info[off:off + 4]
size = int.from_bytes(info[off + 4:off + 8], "little")
subs[sid] = info[off + 8:off + 8 + size].rstrip(b"\x00")
off += 8 + size + (size % 2)
assert subs[b"INAM"] == b"Test Song"
assert subs[b"IART"] == b"Penguin"
assert subs[b"ISRC"] == b"US-ABC-26-00001"
# PCM giong nhau (metadata khong lam doi audio bits)
d_plain, sr_plain = _read_wav(out_plain)
d_meta, sr_meta = _read_wav(out_meta)
assert sr_meta == sr_plain
assert np.allclose(d_meta, d_plain, atol=0.0, rtol=0.0)