Files
SonicForgeStudio/app/core/render_engine.py
admin 2c96a223af feat: CC11 expression automation (spec midi_note_cc11_expression)
- MIDINote schema + expression_curve (position_ratio 0..1, value 0..127)
- Piano roll CC lane mode Expression (CC11): draw/move/del anchors
- Offline: _expression_events grid 64 samples -> job.events -> RenderJob.cpp sample-accurate CC
- Realtime: scheduleNoteExpressionCC11 grid 16ms via bridge pushEvent / SonicSF fallback, epoch-guarded, cleared on stop
- tests/test_cc11_expression.py 6/6 (unit interp/grid/clamp + integration energy follows curve)
- rebuild daw_vst_bridge (narrowing fix) + sync binaries/dist, babel rebuild + sync 5 copies
2026-08-26 11:14:56 +07:00

1112 lines
56 KiB
Python

import os, logging, math
import numpy as np
import soundfile as sf
# scipy.signal import LAZY (chi dung trong ham) — giam thoi gian khoi dong
# engine (khong nap scipy+OpenBLAS ~70MB luc boot)
from app.config import settings
from app.core.vst_engine import (
render_midi_events_to_audio,
HAS_PYFLUIDSYNTH,
RENDER_LOCK,
)
logger = logging.getLogger(__name__)
UPLOAD_SF_DIR = os.path.join(settings.STORAGE_DIR, "soundfonts")
SYSTEM_SF_DIR = "/opt/daw_engine/soundfonts"
SYS_SOUNDFONTS = [
("GeneralUser_GS.sf2", "GeneralUser GS"),
("SGM_v2.01.sf2", "SGM v2.01"),
("SGM-V2.01.sf2", "SGM v2.01"),
]
# GM font dự phòng khi track dùng sf_generaluser_gs nhưng máy không có font đó
# (Windows không ship SF2; SYSTEM_SF_DIR là path Linux). Mirror policy frontend
# realtime (app.jsx pickFallbackSoundfont: sgm → generaluser → tgsf21x → timgm
# → font đầu tiên) — offline render phải ra tiếng GM gần đúng, không rơi xuống
# oscillator synth.
GM_SF_FALLBACK_STEMS = ["sgm", "generaluser", "tgsf21x", "timgm", "fluid", "chorium", "musescore"]
def _find_sf2_path(sf_id: str) -> str:
clean_id = sf_id.replace("sf_", "") if sf_id.startswith("sf_") else sf_id
clean_lower = clean_id.lower()
sf_lower = sf_id.lower()
for base_dir in [UPLOAD_SF_DIR, SYSTEM_SF_DIR]:
if not os.path.isdir(base_dir):
continue
for fname in os.listdir(base_dir):
fbase, fext = os.path.splitext(fname)
if fext.lower() in (".sf2", ".sf3") and (fbase.lower() == clean_lower or fbase.lower() == sf_lower):
return os.path.join(base_dir, fname)
# Thư mục user thêm qua Plugin Manager (plugin_dirs — Add Directory):
# soundfont trong thư mục user phải render được (Windows thường dùng cách này)
try:
from app.core.vst_engine import _load_user_plugin_dirs
for base_dir in _load_user_plugin_dirs():
if not os.path.isdir(base_dir):
continue
for root, dirs, files in os.walk(base_dir):
for fname in files:
fbase, fext = os.path.splitext(fname)
if fext.lower() in (".sf2", ".sf3") and (fbase.lower() == clean_lower or fbase.lower() == sf_lower):
return os.path.join(root, fname)
dirs[:] = [] # không walk sâu
except Exception:
pass
static_dir = os.path.join(settings.APP_DIR, "static", "soundfonts")
if os.path.isdir(static_dir):
for fname in os.listdir(static_dir):
fbase, fext = os.path.splitext(fname)
if fext.lower() in (".sf2", ".sf3") and (fbase.lower() == clean_lower or fbase.lower() == sf_lower):
return os.path.join(static_dir, fname)
return ""
def _find_default_sf2() -> str:
for sf_name, _ in SYS_SOUNDFONTS:
for base_dir in [SYSTEM_SF_DIR, UPLOAD_SF_DIR]:
p = os.path.join(base_dir, sf_name)
if os.path.exists(p):
return p
# Windows: scan plugin dirs user — ưu tiên font GM quen thuộc (sgm/
# generaluser/tgsf21x...), bất kỳ SF2 khác làm fallback cuối (vẫn hơn
# oscillator). Walk sâu: soundfont có thể nằm trong subdir (vd
# Downloads\Collections soundfont\<subdir>\foo.sf2).
try:
from app.core.vst_engine import _load_user_plugin_dirs
found = []
for base_dir in _load_user_plugin_dirs():
if not os.path.isdir(base_dir):
continue
for root, dirs, files in os.walk(base_dir):
for fname in files:
_fb, fext = os.path.splitext(fname)
if fext.lower() not in (".sf2", ".sf3"):
continue
fname_l = fname.lower()
score = 0
for i, stem in enumerate(GM_SF_FALLBACK_STEMS):
if stem in fname_l:
score = 100 - i
break
if score == 0 and ("gm" in fname_l or "general" in fname_l):
score = 50
found.append((score, os.path.join(root, fname)))
if found:
found.sort(key=lambda x: -x[0])
return found[0][1]
except Exception:
pass
return ""
def _render_sf2_pyfluidsynth(sf_path: str, sample_rate: int, bpm: float, midi_channel: int,
bank: int, program: int, midi_events) -> np.ndarray:
# Fallback pyfluidsynth (RENDER_ENGINE != bridge, hoặc bridge fail).
# Bắt buộc synth.gain=1.0: libfluidsynth mặc định 0.2 → âm nhỏ ~5x.
import fluidsynth as _fs
_settings = _fs.new_fluid_settings()
_fs.fluid_settings_setnum(_settings, b'synth.sample-rate', float(sample_rate))
_fs.fluid_settings_setnum(_settings, b'synth.gain', 1.0)
_fl = _fs.new_fluid_synth(_settings)
try:
_fid = _fs.fluid_synth_sfload(_fl, sf_path.encode("utf-8"), 1)
_fs.fluid_synth_program_select(_fl, midi_channel, _fid, bank, program)
beat_sec = 60.0 / bpm
total_sec = 0
for ev in midi_events:
end_sec = (ev.get("start_beat", 0) + ev.get("duration_beats", 1)) * beat_sec
if end_sec > total_sec:
total_sec = end_sec
sf_total_samples = int((total_sec + 1.0) * sample_rate)
midi_data = np.zeros((2, sf_total_samples), dtype=np.float32)
_cursor = 0
for ev in sorted(midi_events, key=lambda e: e.get("start_beat", 0)):
note = ev.get("note", 60)
velocity = ev.get("velocity", 100)
start_beat = ev.get("start_beat", 0.0)
dur_beats = ev.get("duration_beats", 1.0)
start_sec = start_beat * beat_sec
dur_sec = dur_beats * beat_sec
start_s = int(start_sec * sample_rate)
dur_s = int(dur_sec * sample_rate)
# Advance synth time by rendering silence
if start_s > _cursor:
gap = start_s - _cursor
_fs.fluid_synth_write_s16_stereo(_fl, gap)
_cursor = start_s
# Start note
_fs.fluid_synth_noteon(_fl, midi_channel, note, min(velocity, 127))
block_s16 = _fs.fluid_synth_write_s16_stereo(_fl, dur_s)
_fs.fluid_synth_noteoff(_fl, midi_channel, note)
block = block_s16.astype(np.float32).reshape(-1, 2).T / 32768.0
end_s = min(_cursor + block.shape[1], sf_total_samples)
actual = end_s - _cursor
if actual > 0 and block.shape[1] > 0:
midi_data[:, _cursor:end_s] += block[:, :actual]
_cursor = end_s
return midi_data
finally:
_fs.delete_fluid_synth(_fl)
def _apply_master_volume_automation(buf, sr, nodes):
"""Gap 7: master volume automation lane (master.automation.volume_db).
nodes: [{time: giây, db}] — envelope tuyến tính dB, clamp -30..+3 dB như
track automation (sub_tab_dsp). Áp TRÊN master_buffer (2, N) — trả copy."""
nodes = [n for n in (nodes or []) if n.get("time") is not None and n.get("db") is not None]
if not nodes:
return buf
nodes = sorted(nodes, key=lambda x: float(x["time"]))
n_samples = buf.shape[1]
times = np.array([float(n["time"]) for n in nodes])
dbs = np.clip(np.array([float(n["db"]) for n in nodes]), -30.0, 3.0)
sample_times = np.arange(n_samples) / sr
interp = np.interp(sample_times, times, dbs, left=dbs[0], right=dbs[-1])
return buf * (10.0 ** (interp / 20.0))
# ── Builtin custom FX (spec fx_chain_architecture §2 Step 2: Custom FX) ─────
# Track FX chain slots: eq / eqpro / compressor / limiter / exciter / rebalance
# (khớp tham số WebAudio track FX modules + EQ Pro bands) + legacy chorus/reverb
# (fx_type cũ). Chạy numpy/scipy thuần — thay pedalboard GPL; lazy import scipy
# chỉ khi thực sự có FX (không tăng boot time).
def _rbj_peaking(f0, gain_db, q, sr):
A = 10 ** (gain_db / 40.0)
w0 = 2 * math.pi * f0 / sr
alpha = math.sin(w0) / (2 * q)
cw = math.cos(w0)
a0 = 1 + alpha / A
b0 = 1 + alpha * A
b1 = -2 * cw
b2 = 1 - alpha * A
a1 = -2 * cw
a2 = 1 - alpha / A
return [b0 / a0, b1 / a0, b2 / a0], [1.0, a1 / a0, a2 / a0]
def _rbj_shelf(f0, gain_db, q, sr, low):
A = 10 ** (gain_db / 40.0)
w0 = 2 * math.pi * f0 / sr
alpha = math.sin(w0) / (2 * q)
cw = math.cos(w0)
sA = 2 * math.sqrt(A) * alpha
if low:
b0 = A * ((A + 1) - (A - 1) * cw + sA)
b1 = 2 * A * ((A - 1) - (A + 1) * cw)
b2 = A * ((A + 1) - (A - 1) * cw - sA)
a0 = (A + 1) + (A - 1) * cw + sA
a1 = -2 * ((A - 1) + (A + 1) * cw)
a2 = (A + 1) + (A - 1) * cw - sA
else:
b0 = A * ((A + 1) + (A - 1) * cw + sA)
b1 = -2 * A * ((A - 1) + (A + 1) * cw)
b2 = A * ((A + 1) + (A - 1) * cw - sA)
a0 = (A + 1) - (A - 1) * cw + sA
a1 = 2 * ((A - 1) - (A + 1) * cw)
a2 = (A + 1) - (A - 1) * cw - sA
return [b0 / a0, b1 / a0, b2 / a0], [1.0, a1 / a0, a2 / a0]
def _rbj_highpass(f0, q, sr):
w0 = 2 * math.pi * f0 / sr
alpha = math.sin(w0) / (2 * q)
cw = math.cos(w0)
a0 = 1 + alpha
b0 = (1 + cw) / 2
b1 = -(1 + cw)
b2 = (1 + cw) / 2
a1 = -2 * cw
a2 = 1 - alpha
return [b0 / a0, b1 / a0, b2 / a0], [1.0, a1 / a0, a2 / a0]
def _apply_biquad(buf, b, a):
from scipy.signal import lfilter
out = np.empty_like(buf)
for ch in range(buf.shape[0]):
out[ch] = lfilter(b, a, buf[ch])
return out
def _apply_eq4(buf, params, sr):
params = params or {}
gains = [float(params.get(k, 0) or 0) for k in ("g1", "g2", "g3", "g4")]
if not any(gains):
return buf
# WebAudio track 'eq': lowshelf 100Hz, peaking 800Hz Q0.7, peaking 3200Hz
# Q1.2, highshelf 10kHz — cùng thứ tự/đáp ứng.
for (f0, gain, q, low) in ((100, gains[0], 0.707, True),
(800, gains[1], 0.7, False),
(3200, gains[2], 1.2, False),
(10000, gains[3], 0.707, False)):
if gain == 0:
continue
b, a = (_rbj_shelf(f0, gain, q, sr, True) if low
else _rbj_peaking(f0, gain, q, sr))
buf = _apply_biquad(buf, b, a)
return buf
def _apply_eqpro(buf, params, sr):
params = params or {}
bands = params.get("bands") or []
if not bands:
return buf
amount = float(params.get("amount", 100) or 100) / 100.0
for b in bands:
if b.get("active") is False:
continue
gain = (float(b.get("gain", 0) or 0)) * amount
if gain == 0:
continue
f0 = float(b.get("freq", 1000))
q = float(b.get("q", 1.0) or 1.0)
typ = b.get("type", "peaking")
if typ in ("lowshelf", "highshelf"):
bq, aq = _rbj_shelf(f0, gain, q, sr, typ == "lowshelf")
elif typ == "highpass":
bq, aq = _rbj_highpass(f0, q, sr)
else: # peaking / lowpass / notch / bandpass → peaking (biquad gần đúng)
bq, aq = _rbj_peaking(f0, gain, q, sr)
buf = _apply_biquad(buf, bq, aq)
return buf
def _apply_compressor(buf, params, sr):
params = params or {}
threshold_db = float(params.get("threshold", -16))
ratio = max(1.0, float(params.get("ratio", 3) or 3))
makeup_db = float(params.get("makeup", 0) or 0)
makeup = 10 ** (makeup_db / 20.0)
block = 256
out = np.empty_like(buf)
rel = math.exp(-1.0 / (sr * 0.25)) # release 250ms, khớp WebAudio default
for ch in range(buf.shape[0]):
x = buf[ch]
n = x.shape[0]
nblocks = (n + block - 1) // block
gr = np.ones(n, dtype=np.float32)
env = 0.0
for bi in range(nblocks):
seg = x[bi * block:(bi + 1) * block]
peak = float(np.max(np.abs(seg))) if seg.size else 0.0
env = max(peak, env * rel) # release smoothing giữa block
if env > 1e-9:
db = 20 * math.log10(env)
over = db - threshold_db
if over > 0:
gain_db = -over * (1.0 - 1.0 / ratio)
gr[bi * block:(bi + 1) * block] = (10 ** (gain_db / 20.0)) * makeup
else:
gr[bi * block:(bi + 1) * block] = makeup
out[ch] = x * gr
return out
def _apply_limiter(buf, params, sr):
params = params or {}
ceiling_db = min(0.0, float(params.get("ceiling", -1.0)))
th = 10 ** (ceiling_db / 20.0)
k = 1.0 / max(0.02, th)
tanh_k = math.tanh(k)
# Brickwall tanh soft-clip — mirror WebAudio limiter (tránh NaN của
# DynamicsCompressor Chromium; offline dùng cùng curve). WaveShaper
# curve chỉ định nghĩa trên [-1,1] → clamp đầu vào như WebAudio.
x = np.clip(buf, -1.0, 1.0)
return np.tanh(x * k) / tanh_k
def _apply_mix_limiter(buf, sr):
"""Mixer summing cap (spec §2): smoothed brickwall -3 dBFS mirror C++
realtime (NativeInstrumentEngine.cpp renderAll) — ceiling 0.7071, attack
0.1 (~10 samples), release 0.0006, gain chung 2 kênh. Áp cho MASTER CHAIN
INPUT offline (realtime đã cap sau khi sum; offline trước đây thiếu).
ponytail: loop Python O(n) — thay numba/vectorized khi render >10 phút."""
if buf.size == 0:
return buf
k_ceil = 0.7071
k_attack = 0.1
k_release = 0.0006
eps = 1e-12
m_list = np.max(np.abs(buf), axis=0).tolist()
n = buf.shape[1]
g = 1.0
gains = np.empty(n, dtype=np.float32)
for i in range(n):
t = k_ceil / (m_list[i] + eps)
if t > 1.0:
t = 1.0
if t < g:
g += (t - g) * k_attack
else:
g += (1.0 - g) * k_release
gains[i] = g
return buf * gains
def _apply_exciter(buf, params, sr):
params = params or {}
drive = float(params.get("drive", 40) or 40)
b, a = _rbj_highpass(2000.0, 0.7, sr)
hp = _apply_biquad(buf, b, a)
wet = (drive / 100.0) * 0.6
return buf + np.tanh(hp * 3.0) * wet
def _apply_rebalance(buf, params, sr):
params = params or {}
mid = 10 ** (float(params.get("mid", 0) or 0) / 20.0)
side = 10 ** (float(params.get("side", 0) or 0) / 20.0)
a = (mid + side) / 2.0
b = (mid - side) / 2.0
L = buf[0]
R = buf[1]
return np.stack([a * L + b * R, b * L + a * R]).astype(np.float32)
def _apply_builtin_fx_chain(buf, chain, sr):
"""Serial pipeline qua các builtin custom FX slots (spec §2 Step 2: Slot 1..n
in-place). Slot inactive/bypass → bỏ qua. vst3 slots = xử lý ngoài
(bridge) → không thêm DSP ở đây."""
for slot in chain or []:
if isinstance(slot, str):
slot = {"type": slot}
if not isinstance(slot, dict):
continue
if slot.get("active") is False or slot.get("bypass"):
continue
typ = slot.get("type") or ""
params = slot.get("params") or {}
try:
if typ == "eq":
buf = _apply_eq4(buf, params, sr)
elif typ == "eqpro":
buf = _apply_eqpro(buf, params, sr)
elif typ == "compressor":
buf = _apply_compressor(buf, params, sr)
elif typ == "limiter":
buf = _apply_limiter(buf, params, sr)
elif typ == "exciter":
buf = _apply_exciter(buf, params, sr)
elif typ == "rebalance":
buf = _apply_rebalance(buf, params, sr)
except Exception as e:
logger.warning("[RenderEngine] Builtin FX slot %s failed: %s", typ, e)
return buf
def _apply_legacy_chorus(buf, sr):
total_samples = buf.shape[1]
lfo = 0.020 + 0.005 * np.sin(2 * np.pi * 1.5 * np.arange(total_samples) / sr)
dry = buf * 0.6
wet = np.zeros_like(buf)
for ch in range(2):
indices = np.arange(total_samples) - (lfo * sr)
indices = np.clip(indices, 0, total_samples - 1).astype(np.int32)
wet[ch, :] = buf[ch, indices]
return dry + wet * 0.5
def _apply_legacy_reverb(buf, sr):
total_samples = buf.shape[1]
len_ir = int(sr * 2.0)
t_ir = np.arange(len_ir) / sr
decay = np.exp(-t_ir / 0.5)
ir_l = (np.random.rand(len_ir) * 2 - 1) * decay
ir_r = (np.random.rand(len_ir) * 2 - 1) * decay
dry = buf * 0.6
wet = np.zeros_like(buf)
for ch in range(2):
ir = ir_l if ch == 0 else ir_r
from scipy.signal import convolve
conv = convolve(buf[ch, :], ir, mode='full')[:total_samples]
wet[ch, :] = conv
return dry + wet * 0.4
def _track_latency_samples(track):
"""PDC (spec §3): tổng latency (samples) của track từ chain slots. Plugin
khai qua field `latency_samples` (bridge hiện chưa báo → 0); khi có
getLatencySamples() native, slot điền field này và engine tự align."""
total = 0
for chain_key in ("fx_chain", "vst_fx_chain"):
for slot in (track.get(chain_key) or []):
if isinstance(slot, dict) and slot.get("latency_samples"):
total += int(slot["latency_samples"])
return total
def _merge_track_fx_chain(track):
"""Chain hợp nhất track cho bridge (Phase 3.1): builtin fx_chain + vst3
vst_fx_chain, giữ NGUYÊN thứ tự UI. vst_fx_bypass → bỏ hẳn slot vst3
(builtin vẫn chạy)."""
merged = []
for slot in (track.get("fx_chain") or []):
if isinstance(slot, str):
slot = {"type": slot}
if isinstance(slot, dict):
merged.append(slot)
if not track.get("vst_fx_bypass"):
seen = {s.get("path") for s in merged if s.get("path")}
for slot in (track.get("vst_fx_chain") or []):
if isinstance(slot, dict) and slot.get("path") in seen:
continue # đã có trong fx_chain (migration) — không nhân đôi
merged.append(slot)
return merged
# Param mapping master chain entry → C++ BuiltinFx (Phase 3.2). UI chain entry
# chỉ lưu params cho eqpro; imager/maximizer/eq/... giữ knob ở flat mastering
# state (maxGain, w1..w4, ...). Offline fill từ flat state để khớp live
# (MASTER_MODULE_IO WebAudio đọc cùng state đó).
_BUILTIN_PARAM_MAP = {
"eq": [("g1", "eqLowGain"), ("g2", "eqMid1Gain"),
("g3", "eqMid2Gain"), ("g4", "eqHighGain")],
"compressor": [("threshold", "compThreshold"), ("ratio", "compRatio"),
("makeup", "compMakeup")],
"limiter": [("ceiling", "limThreshold"), ("mode", "limMode"), ("lookahead_ms", "limLookahead")],
"exciter": [("drive", "excDrive")],
"rebalance": [("mid", "rebalMid"), ("side", "rebalSide")],
"imager": [("w1", "w1"), ("w2", "w2"), ("w3", "w3"), ("w4", "w4")],
"maximizer": [("boost_db", "maxGain"), ("soft_clip", "maxSoftClip"),
("upward", "maxUpward"), ("ceiling_db", "ceiling")],
}
def _master_chain_entry(ms, entry):
"""Entry master chain cho bridge: params entry (eqpro) + fill từ flat
mastering state (knob imager/maximizer/eq/...)."""
typ = entry.get("type")
if typ not in _BUILTIN_PARAM_MAP:
return entry
params = dict(entry.get("params") or {})
for cpp_key, js_key in _BUILTIN_PARAM_MAP[typ]:
v = ms.get(js_key)
if v is not None and cpp_key not in params:
params[cpp_key] = v
out = dict(entry)
out["params"] = params
return out
def _probe_track_latency(track, sample_rate):
"""Latency thật (samples) của track từ bridge --render-fx (Phase 3.3):
chạy chain qua probe silence ngắn, parse FX_LATENCIES. 0 nếu không đo
được. Chỉ chạy khi track có slot vst3 (builtin DSP latency = 0)."""
if settings.RENDER_ENGINE != "bridge":
return 0
merged = _merge_track_fx_chain(track)
if not any(isinstance(s, dict) and s.get("type") == "vst3" for s in merged):
return 0
from app.core.native_render import render_fx_chain
os.makedirs(settings.PROCESSED_DIR, exist_ok=True)
probe_in = os.path.join(
settings.PROCESSED_DIR,
f"latprobe_in_{os.getpid()}_{np.random.randint(100000)}.wav")
probe_out = probe_in.replace("latprobe_in_", "latprobe_out_")
try:
sf.write(probe_in, np.zeros((2, 64), dtype=np.float32).T,
sample_rate, subtype="FLOAT")
_, lats = render_fx_chain(input_wav=probe_in, fx_chain=merged,
sample_rate=sample_rate, out_path=probe_out)
return int(sum(lats)) if lats else 0
except Exception as e:
logger.warning("[RenderEngine] PDC probe failed: %s", e)
return 0
finally:
for f in (probe_in, probe_out):
if os.path.exists(f):
try:
os.remove(f)
except Exception:
pass
def _quantize_pcm(audio, bits, dither):
"""Quantize float [-1,1] to PCM grid (bits=16|24), optional TPDF dither +
1st-order noise shaping. Tra ve int array de sf.write ghi EXACT bits
(libsndfile float->PCM khong round dung 32768-grid - kiem chung D3)."""
N = 2 ** (bits - 1)
x = audio.astype(np.float64)
if dither and bits < 32:
lsb = 1.0 / N
rng = np.random.default_rng()
d = (rng.random(x.shape) + rng.random(x.shape) - 1.0) * lsb # TPDF [-lsb, lsb]
out = np.empty_like(x)
err = np.zeros(x.shape[0])
for i in range(x.shape[1]):
s = x[:, i] + d[:, i] + err
q = np.rint(s * N)
out[:, i] = q
err = q / N - s # 1st-order noise shaping feedback
q = np.clip(out, -N, N - 1)
else:
q = np.clip(np.rint(x * N), -N, N - 1)
if bits == 16:
return q.astype(np.int16)
# PCM_24: libsndfile luu top-24-bit cua int32 -> <<8
return (q.astype(np.int64) << 8).astype(np.int32)
class PythonRenderEngine:
def __init__(self, sample_rate=44100):
self.sample_rate = sample_rate
def bars_to_samples(self, bars: float, bpm: float, time_sig_num: int) -> int:
seconds_per_beat = 60.0 / max(20.0, bpm)
seconds_per_bar = seconds_per_beat * time_sig_num
return int(bars * seconds_per_bar * self.sample_rate)
def resolve_file_path(self, url_or_id: str) -> str:
if not url_or_id:
return ""
base = os.path.basename(url_or_id)
# Check uploads directory
p_uploads = os.path.join(settings.UPLOADS_DIR, base)
if os.path.exists(p_uploads):
return p_uploads
# Check processed directory
p_processed = os.path.join(settings.PROCESSED_DIR, base)
if os.path.exists(p_processed):
return p_processed
# Check general storage directory
p_storage = os.path.join(settings.STORAGE_DIR, base)
if os.path.exists(p_storage):
return p_storage
# Direct check
if os.path.exists(url_or_id):
return url_or_id
return url_or_id
def render_session_container(self, session: dict, section_store: dict, bpm: float, time_sig_num: int, total_samples: int, _cache: dict = None) -> np.ndarray:
session_buffer = np.zeros((2, total_samples), dtype=np.float32)
# Solo semantics: when any track is soloed, only soloed tracks sound.
tracks = session.get("tracks", [])
solo_ids = {t.get("id") for t in tracks if t.get("solo")}
# PDC pre-scan (spec §3): latency mỗi track từ chain slots; track ngắn
# hơn bị delay khi mix để mọi track sum sample-accurate phase-aligned.
# Phase 3.3: slot chưa khai latency_samples → probe bridge --render-fx
# (FX_LATENCIES) đo latency thật của VST3 (getLatencySamples).
_track_lat = {}
_max_lat = 0
for _t in tracks:
if solo_ids and _t.get("id") not in solo_ids:
continue
_lat = _track_latency_samples(_t)
if not _lat:
_lat = _probe_track_latency(_t, self.sample_rate)
_track_lat[_t.get("id")] = _lat
_max_lat = max(_max_lat, _lat)
_channel_counter = 0
for track in tracks:
if solo_ids and track.get("id") not in solo_ids:
continue
track_type = track.get("type", "AUDIO")
track_buffer = np.zeros((2, total_samples), dtype=np.float32)
# Parse synth_engine struct (Task C) — fall back to flat fields
se = track.get("synth_engine", {}) or {}
instrument_id = se.get("plugin_id") or track.get("instrument_id", "") or track.get("instrument", "")
instrument_source = se.get("type") or track.get("instrument_source", "soundfont")
soundfont_bank = se.get("soundfont_bank") if se.get("soundfont_bank") is not None else track.get("soundfont_bank", 0)
soundfont_program = se.get("soundfont_program") if se.get("soundfont_program") is not None else track.get("soundfont_program", 0)
soundfont_id = se.get("soundfont_id") or track.get("soundfont_id", "")
instrument_program = track.get("instrument_program")
# GM-only track (instrumentProgram, khong synth_engine/soundfont_id):
# preview choi qua WASM default font -> export phai dung default SF2
# + program do, khong roi xuong oscillator synth (TIMBRE bug).
if (not soundfont_id and not instrument_id
and instrument_program is not None and instrument_program != ""):
soundfont_id = "_default_gm"
soundfont_bank = 0
soundfont_program = int(instrument_program)
is_percussion = track.get("is_percussion", False) or (soundfont_bank == 128)
midi_channel = 9 if is_percussion else (_channel_counter % 9)
if not is_percussion:
_channel_counter += 1
for item in track.get("items", []):
start_sample = self.bars_to_samples(item["start_bar"], bpm, time_sig_num)
dur_samples = self.bars_to_samples(item["duration_bars"], bpm, time_sig_num)
offset_sample = self.bars_to_samples(item["clip_start_offset_bars"], bpm, time_sig_num)
item_type = item.get("type")
if item_type == "AUDIO_ITEM":
source_data = item.get("source_data", {})
audio_url = source_data.get("audio_file_url", "")
resolved_path = self.resolve_file_path(audio_url)
if resolved_path and os.path.exists(resolved_path):
try:
audio_data, sr = sf.read(resolved_path, dtype='float32')
if sr != self.sample_rate:
# Proper resampling: previously a silent no-op that
# played 48kHz audio at the wrong speed/pitch.
from scipy.signal import resample_poly
g = math.gcd(sr, self.sample_rate)
audio_data = resample_poly(
audio_data,
up=self.sample_rate // g,
down=sr // g,
axis=-1,
window=("kaiser", 16), # stopband >= 130 dB (lossless-audio-compliance)
)
sr = self.sample_rate
# Handle channel mapping (Mono/Stereo)
if len(audio_data.shape) == 1:
audio_data = np.vstack([audio_data, audio_data])
else:
audio_data = audio_data.T # Shape: (channels, samples)
# Trim source offset & duration
src_len = audio_data.shape[1]
if offset_sample < src_len:
actual_dur = min(dur_samples, src_len - offset_sample)
sliced_audio = audio_data[:, offset_sample : offset_sample + actual_dur]
# Apply gain
gain_val = source_data.get("gain", 1.0)
sliced_audio = sliced_audio * gain_val
# Write to track buffer with boundaries
write_end = min(start_sample + sliced_audio.shape[1], total_samples)
actual_len = write_end - start_sample
if actual_len > 0:
track_buffer[:, start_sample:write_end] += sliced_audio[:, :actual_len]
except Exception as e:
logger.warning("[RenderEngine] Error reading audio file %s: %s", resolved_path, e)
elif item_type == "MIDI_ITEM":
source_data = item.get("source_data", {})
notes = source_data.get("notes", [])
# Convert to midi events required by vst_engine
midi_events = []
for note in notes:
note_start_bar = note["start_beat"] / time_sig_num
# Filter notes within the non-destructive visible window
offset_bar = item["clip_start_offset_bars"]
dur_bar = item["duration_bars"]
if note_start_bar >= offset_bar and note_start_bar < (offset_bar + dur_bar):
rel_bar_in_item = note_start_bar - offset_bar
target_global_bar = item["start_bar"] + rel_bar_in_item
midi_events.append({
"note": note["pitch"],
"start_beat": target_global_bar * time_sig_num,
"duration_beats": note["duration_beats"],
"velocity": int(note.get("velocity", 0.8) * 127),
"expression_curve": note.get("expression_curve")
})
if midi_events:
try:
if instrument_source == "pianobook":
# Phase 4 (§3.4): .dspreset không hỗ trợ nữa (pedalboard
# GPL-3.0 gỡ; spike load .dspreset fail). Chặn + fallback synth.
dspreset_path = track.get("dspreset_path", "")
logger.warning(
"[RenderEngine] Pianobook track bỏ qua: .dspreset không còn hỗ trợ. "
"Convert sang .vstpreset (DecentSampler → Save preset) rồi gán lại track."
)
synth_buffer = render_midi_events_to_audio(
midi_events=midi_events, sr=self.sample_rate, bpm=bpm, instrument='synth'
)
elif instrument_id and not instrument_id.startswith("sf_"):
# VSTi track: native_bridge --render (SF_RENDER_ENGINE=bridge).
# Giá trị engine khác → fallback synth (pedalboard GPL-3.0 đã gỡ).
if settings.RENDER_ENGINE == "bridge":
try:
from app.core import native_render
se2 = track.get("synth_engine", {}) or {}
bridge_notes = [{
"pitch": ev["note"],
"velocity": float(ev["velocity"]) / 127.0,
"start_beat": ev["start_beat"],
"duration_beats": ev["duration_beats"],
"expression_curve": ev.get("expression_curve"),
} for ev in midi_events]
# preset_data rỗng → fallback preset
# bridge đang giữ (bridge_state.json) —
# export tạo bridge mới init patch ≠ âm
# realtime user đang nghe.
preset_b64 = (se2.get("preset_data")
or track.get("preset_data"))
if not preset_b64:
# BUG v6: channel phải là midi_channel
# (round-robin melodic — giống
# unifiedMidiRouter.allocateChannel),
# KHÔNG phải track.id → trước đây lấy
# preset của channel stale (sai loại
# nhạc cụ khi realtime chơi channel khác).
preset_b64 = native_render.bridge_state_preset_b64(
channel=midi_channel,
plugin_path=(se2.get("plugin_path")
or track.get("plugin_path")),
)
out_path, _dur = native_render.render_offline(
instrument_id=instrument_id,
notes=bridge_notes,
bpm=bpm,
sample_rate=self.sample_rate,
preset_id=se2.get("preset_id") or track.get("preset_id"),
preset_path=se2.get("preset_path") or track.get("preset_path"),
preset_data_b64=preset_b64,
soundfont_bank=soundfont_bank,
soundfont_program=soundfont_program,
plugin_path=se2.get("plugin_path") or track.get("plugin_path"),
)
synth_buffer, _ = sf.read(out_path, dtype="float32")
if synth_buffer.ndim == 1:
synth_buffer = np.vstack([synth_buffer, synth_buffer])
else:
synth_buffer = synth_buffer.T
try:
os.remove(out_path)
except Exception:
pass
except Exception as e:
logger.warning("[RenderEngine] native_bridge render failed: %s", e)
synth_buffer = render_midi_events_to_audio(
midi_events=midi_events, sr=self.sample_rate, bpm=bpm, instrument='synth'
)
else:
synth_buffer = render_midi_events_to_audio(
midi_events=midi_events, sr=self.sample_rate, bpm=bpm, instrument='synth'
)
elif soundfont_id or (instrument_id and instrument_id.startswith("sf_")):
sf_path = _find_sf2_path(soundfont_id or instrument_id)
# 4-level fallback: selected SF → default SF → native bridge → oscillator synth.
# Gap 12: pyfluidsynth thiếu FluidSynth lib → bridge render SF2 thật thay vì oscillator.
if not sf_path or not os.path.exists(sf_path):
logger.warning(f"[RenderEngine] SoundFont not found for {soundfont_id or instrument_id}, trying default")
sf_path = _find_default_sf2()
if sf_path and os.path.exists(sf_path) and settings.RENDER_ENGINE == "bridge":
# Bridge-first: native bridge nhúng FluidSynth C API render SF2/SF3 thật.
# Frozen engine HAS_PYFLUIDSYNTH=True (libfluidsynth-3.dll trong _internal)
# nhưng pyfluidsynth mặc định synth.gain=0.2 → âm nhỏ; bridge có makeup gain.
try:
from app.core import native_render
out_path, _dur = native_render.render_soundfont_offline(
sf_path=sf_path,
notes=[{
"pitch": ev["note"],
"velocity": float(ev["velocity"]) / 127.0,
"start_beat": ev["start_beat"],
"duration_beats": ev["duration_beats"],
"expression_curve": ev.get("expression_curve"),
} for ev in midi_events],
bpm=bpm,
sample_rate=self.sample_rate,
bank=soundfont_bank,
program=soundfont_program,
)
synth_buffer, _ = sf.read(out_path, dtype="float32")
if synth_buffer.ndim == 1:
synth_buffer = np.vstack([synth_buffer, synth_buffer])
else:
synth_buffer = synth_buffer.T
try:
os.remove(out_path)
except Exception:
pass
except Exception as e:
logger.warning("[RenderEngine] native soundfont render failed: %s", e)
if HAS_PYFLUIDSYNTH:
synth_buffer = _render_sf2_pyfluidsynth(
sf_path, self.sample_rate, bpm, midi_channel,
soundfont_bank, soundfont_program, midi_events,
)
else:
synth_buffer = render_midi_events_to_audio(
midi_events=midi_events, sr=self.sample_rate, bpm=bpm, instrument='synth'
)
elif sf_path and os.path.exists(sf_path) and HAS_PYFLUIDSYNTH:
# Fallback khi RENDER_ENGINE != bridge: pyfluidsynth với synth.gain=1.0.
synth_buffer = _render_sf2_pyfluidsynth(
sf_path, self.sample_rate, bpm, midi_channel,
soundfont_bank, soundfont_program, midi_events,
)
else:
synth_buffer = render_midi_events_to_audio(
midi_events=midi_events, sr=self.sample_rate, bpm=bpm, instrument='synth'
)
else:
synth_buffer = render_midi_events_to_audio(
midi_events=midi_events, sr=self.sample_rate, bpm=bpm, instrument='synth'
)
actual_len = min(synth_buffer.shape[1], total_samples)
track_buffer[:, :actual_len] += synth_buffer[:, :actual_len]
except Exception as e:
logger.warning("[RenderEngine] Error rendering MIDI: %s", e)
elif item_type == "SECTION_ITEM":
source_data = item.get("source_data", {})
sec_id = source_data.get("referenced_section_id", "")
if sec_id and sec_id in section_store:
# Render nested section recursively, cached per section id
# so repeated section instances don't re-render every time.
cache = _cache if _cache is not None else {}
if sec_id in cache:
sec_buffer = cache[sec_id]
else:
sec_buffer = self.render_session_container(
session=section_store[sec_id],
section_store=section_store,
bpm=bpm,
time_sig_num=time_sig_num,
total_samples=total_samples,
_cache=cache,
)
cache[sec_id] = sec_buffer
# Apply non-destructive crop/slicing on section buffer
if offset_sample < total_samples:
actual_dur = min(dur_samples, total_samples - offset_sample)
sliced_sec = sec_buffer[:, offset_sample : offset_sample + actual_dur]
# Write to track buffer
write_end = min(start_sample + sliced_sec.shape[1], total_samples)
actual_len = write_end - start_sample
if actual_len > 0:
track_buffer[:, start_sample:write_end] += sliced_sec[:, :actual_len]
# Track volume / pan / mute (spec §2 Step 3: summing gains)
vol_db = track.get("volume_db", 0.0)
pan = track.get("pan", 0.0)
mute = track.get("mute", False)
if mute:
continue
# ── Track FX chain (spec §2 Step 2): serial custom FX → VST FX ──
# Phase 3.1: RENDER_ENGINE=bridge → chain HỢP NHẤT (builtin
# fx_chain + vst3 vst_fx_chain xen kẽ đúng thứ tự UI) qua 1 lần
# native_bridge --render-fx. Mỗi slot sửa buffer in-place đúng
# Buffer_out = Plugin_n(...Plugin_1(Buffer_in)). Legacy
# chorus/reverb (fx_type cũ) chỉ có Python DSP → áp trước bridge;
# `ponytail:` track có cả fx_chain + fx_type sẽ chạy legacy trước
# chain (track cũ không có chain hiện đại — order không quan trọng).
fx_type = track.get("fx_type")
if track.get("fx_active", True):
merged_fx = _merge_track_fx_chain(track)
if merged_fx and settings.RENDER_ENGINE == "bridge":
try:
from app.core.native_render import render_fx_chain
tmp_in = os.path.join(
settings.PROCESSED_DIR,
f"trackfx_in_{os.getpid()}_{np.random.randint(100000)}.wav")
tmp_out = tmp_in.replace("trackfx_in_", "trackfx_out_")
try:
if fx_type == "chorus":
track_buffer = _apply_legacy_chorus(track_buffer, self.sample_rate)
elif fx_type == "reverb":
track_buffer = _apply_legacy_reverb(track_buffer, self.sample_rate)
sf.write(tmp_in, track_buffer.T, self.sample_rate, subtype="FLOAT")
render_fx_chain(input_wav=tmp_in, fx_chain=merged_fx,
sample_rate=self.sample_rate, out_path=tmp_out)
data, _ = sf.read(tmp_out, dtype="float32", always_2d=True)
if data.shape[1] >= 2:
track_buffer = data.T[:2, :]
else:
track_buffer = np.repeat(data.T, 2, axis=0)
finally:
for f in (tmp_in, tmp_out):
if os.path.exists(f):
try:
os.remove(f)
except Exception:
pass
except Exception as e:
logger.warning("[RenderEngine] Track FX chain skipped: %s", e)
else:
# Fallback Python DSP (RENDER_ENGINE≠bridge) — giữ luồng cũ:
# builtin fx_chain → legacy fx_type. VST3 cần bridge.
try:
track_buffer = _apply_builtin_fx_chain(
track_buffer, track.get("fx_chain"), self.sample_rate)
except Exception as e:
logger.warning("[RenderEngine] Builtin FX chain failed: %s", e)
if fx_type == "chorus":
try:
track_buffer = _apply_legacy_chorus(track_buffer, self.sample_rate)
except Exception as e:
logger.warning("[RenderEngine] Fallback Chorus failed: %s", e)
elif fx_type == "reverb":
try:
track_buffer = _apply_legacy_reverb(track_buffer, self.sample_rate)
except Exception as e:
logger.warning("[RenderEngine] Fallback Reverb failed: %s", e)
if (track.get("vst_fx_chain") and not track.get("vst_fx_bypass")
and settings.RENDER_ENGINE != "bridge"):
logger.warning("[RenderEngine] Track VST FX cần RENDER_ENGINE=bridge — bỏ qua")
# Process track volume (spec §2 Step 3: Gain_i scaling)
gain_linear = 10 ** (vol_db / 20.0)
processed_track = track_buffer * gain_linear
# Apply Track Pan (spec §2 Step 3: constant-power pan law)
if pan != 0.0:
theta = ((np.clip(pan, -1.0, 1.0) + 1.0) / 2.0) * (np.pi / 2.0)
processed_track[0, :] *= np.cos(theta)
processed_track[1, :] *= np.sin(theta)
# PDC (spec §3): delay track ngắn hơn (max_lat - own_lat) để mọi
# track sum sample-accurate. latency 0 hiện tại → no-op.
delay = _max_lat - _track_lat.get(track.get("id"), 0)
if delay > 0:
shifted = np.zeros_like(processed_track)
shifted[:, delay:] = processed_track[:, :total_samples - delay]
processed_track = shifted
# Mix track to session (spec §2 Step 3: linear summing)
session_buffer += processed_track
return session_buffer
def render_project(self, project_json: dict, output_filepath: str,
bit_depth: int = 32, out_sample_rate: int = 0,
dither: bool = False, normalize: bool = False,
normalize_target: str = "-1dbtp",
metadata: dict = None):
# Gap 13: BWF/INFO metadata — tham số phụ, không đổi hành vi cũ khi None.
with RENDER_LOCK:
bpm = project_json["metadata"]["bpm"]
time_sig_num = project_json["metadata"].get("time_signature_numerator", 4)
main_session = project_json["main_session"]
section_store = project_json.get("section_store", {})
# Compute total project samples
total_bars = main_session.get("length_bars", 16.0)
total_samples = self.bars_to_samples(total_bars, bpm, time_sig_num)
# Render main session
master_buffer = self.render_session_container(
session=main_session,
section_store=section_store,
bpm=bpm,
time_sig_num=time_sig_num,
total_samples=total_samples,
_cache={},
)
# Mixer summing cap (spec §2): -3 dBFS brickwall mirror C++ realtime —
# áp cho MASTER CHAIN INPUT, TRƯỚC master FX chain/volume. Offline
# trước đây thiếu cap này (chỉ anti-clip >1.0 sau master chain).
master_buffer = _apply_mix_limiter(master_buffer, self.sample_rate)
# Masterbus FX chain (VST3 FX / builtin) — render qua native_bridge
# --render-fx, rồi volume_db master; backward compatible (không có
# master.fx_chain → giữ nguyên luồng cũ).
# Phase 3.2: ưu tiên mastering_settings.chain (chain hợp nhất UI —
# builtin imager/maximizer + vst3 xen kẽ, params qua entry + flat
# mastering state) → bỏ bất đối xứng offline thiếu imager/maximizer;
# fallback master.fx_chain (project cũ, vst3-only).
master = main_session.get("master") or {}
_ms = project_json.get("mastering_settings") or {}
_ms_chain = _ms.get("chain") if isinstance(_ms.get("chain"), list) else None
if _ms_chain:
fx_chain = [_master_chain_entry(_ms, s) for s in _ms_chain
if isinstance(s, dict)]
else:
fx_chain = master.get("fx_chain") or []
if fx_chain and not master.get("bypass"):
if settings.RENDER_ENGINE == "bridge":
try:
from app.core.native_render import render_fx_chain
tmp_in = os.path.join(settings.PROCESSED_DIR,
f"master_in_{os.getpid()}_{np.random.randint(100000)}.wav")
tmp_out = tmp_in.replace("master_in_", "master_out_")
try:
sf.write(tmp_in, master_buffer.T, self.sample_rate, subtype="FLOAT")
render_fx_chain(input_wav=tmp_in, fx_chain=fx_chain,
sample_rate=self.sample_rate, out_path=tmp_out)
data, _ = sf.read(tmp_out, dtype="float32", always_2d=True)
if data.shape[1] >= 2:
master_buffer = data.T[:2, :]
else:
master_buffer = np.repeat(data.T, 2, axis=0)
finally:
for f in (tmp_in, tmp_out):
if os.path.exists(f):
try:
os.remove(f)
except Exception:
pass
except Exception as e:
logger.warning("[RenderEngine] Masterbus FX skipped: %s", e)
else:
logger.warning("[RenderEngine] Masterbus FX cần RENDER_ENGINE=bridge — bỏ qua")
# Master bus volume (spec §2 Step 4: master gain stage after chain).
# Per-channel L/R faders: volumeL_db/volumeR_db (link_lr -> R theo L);
# MUTE zeroes output; fallback volume_db cho project cu.
if master.get("muted") or master.get("mute"):
master_buffer *= 0.0
else:
vol_l = master.get("volumeL_db", master.get("volume_db", 0.0))
if master.get("link_lr"):
vol_r = vol_l
else:
vol_r = master.get("volumeR_db", vol_l)
if master_buffer.shape[0] >= 2:
if vol_l:
master_buffer[0] *= 10 ** (vol_l / 20.0)
if vol_r:
master_buffer[1] *= 10 ** (vol_r / 20.0)
else:
if vol_l:
master_buffer *= 10 ** (vol_l / 20.0)
# Gap 7: master volume automation — envelope áp SAU fader, TRƯỚC
# normalize (fader cuối; automation = lớp âm lượng trên fader).
_auto = master.get("automation") or {}
_auto_vol = _auto.get("volume_db") or []
if _auto_vol:
master_buffer = _apply_master_volume_automation(
master_buffer, self.sample_rate, _auto_vol)
# Normalization to prevent clipping
max_peak = np.max(np.abs(master_buffer))
if max_peak > 1.0:
master_buffer /= max_peak
# ── D3 export stage: resample / normalize / dither / PCM ──
out_sr = out_sample_rate or self.sample_rate
if out_sr != self.sample_rate:
from scipy.signal import resample_poly
g = math.gcd(int(out_sr), int(self.sample_rate))
master_buffer = resample_poly(
master_buffer, int(out_sr) // g, int(self.sample_rate) // g, axis=1)
if normalize:
from app.core import loudness
if normalize_target == "-14lufs":
# Spec: master ra -14 LUFS (scale toàn cục) VÀ hard-peak
# <= -1.0 dBTP (chặn peak sau scale, deterministic — peak
# cap thắng khi nguồn crest cao, LUFS hạ xuống dưới -14).
# ponytail: lookahead limiter khi cần giữ -14 LUFS với
# nguồn crest cao (master thật hiếm gặp).
_l = loudness.integrated_lufs(master_buffer, out_sr)
if np.isfinite(_l):
master_buffer *= 10 ** ((-14.0 - _l) / 20.0)
_tp = loudness.true_peak_db(master_buffer, out_sr)
if np.isfinite(_tp) and _tp > -1.0:
master_buffer *= 10 ** ((-1.0 - _tp) / 20.0)
else: # "-1dbtp"
_tp = loudness.true_peak_db(master_buffer, out_sr)
if np.isfinite(_tp):
master_buffer *= 10 ** ((-1.0 - _tp) / 20.0)
# Write final output file (D3: PCM_16/PCM_24 + dither, PCM_16 mac dinh —
# 32-bit float WAV bi Windows Media Player va nhieu app khac tu choi).
# OGG: Vorbis float lossy — bỏ dither/PCM/BWF (chunk WAV-only).
ext = os.path.splitext(output_filepath)[1].lower()
if ext == ".ogg":
sf.write(output_filepath, master_buffer.T, out_sr,
format="OGG", subtype="VORBIS")
elif bit_depth == 16:
sf.write(output_filepath, _quantize_pcm(master_buffer, 16, dither).T,
out_sr, subtype="PCM_16")
elif bit_depth == 24:
sf.write(output_filepath, _quantize_pcm(master_buffer, 24, dither).T,
out_sr, subtype="PCM_24")
else:
sf.write(output_filepath, _quantize_pcm(master_buffer, 16, dither).T,
out_sr, subtype="PCM_16")
# Gap 13: BWF/INFO chunks (bext + LIST/INFO title/artist/ISRC) —
# chen sau khi ghi PCM, loi metadata khong lam hong file audio.
if metadata and ext != ".ogg":
try:
from app.core.wav_bwf import patch_bwf_metadata
patch_bwf_metadata(output_filepath, metadata)
except Exception as e:
logger.warning("[RenderEngine] BWF metadata skipped: %s", e)
return output_filepath