fix(engine): giam jitter bridge timer va fade-out khi transport stall

- RealtimeFxLoop: timeBeginPeriod(1) + sleep_until thay Sleep(1) -> loopMax 32.9->24.8ms
- plugins.py _pump_output: stall-fade 8 block gain 1->0 khi gap > 5.5*block_sec (~29ms),
  burst send, fade-in 0->1 khi resume; poll dung asyncio.to_thread(time.sleep) vi asyncio.sleep
  bi cap 15.6ms tren Win11 (IOCP) du timeBeginPeriod
- fx_realtime.py: stats dem publish/drop/stale/read cho diagnostics
- rebuild bridge binary
This commit is contained in:
2026-08-24 10:40:21 +07:00
parent 708e7de32b
commit 21130858d6
5 changed files with 135 additions and 14 deletions
+70 -13
View File
@@ -1093,14 +1093,40 @@ async def ws_fx_realtime(websocket: WebSocket, session_id: str):
# nếu chỉ drain trong vòng receive_bytes, output thừa kẹt lại trong SHM ring
# (client ngừng gửi → server block ở receive → âm tắc).
async def _pump_output():
# Pace mỗi block theo tốc độ âm thật (256/sr ~ 5.8ms): bridge sản xuất
# output theo burst khi nhận input dồn; bắn toàn bộ cùng lúc → worklet
# overflow drop block cũ nhất → âm ngắt quãng.
# Gửi ASAP (không pace theo realtime): bridge sản xuất burst theo batch
# 4 → gửi hết ngay → CLIENT tích lũy buffer (worklet FILL) hấp thụ
# jitter bridge/WS. Pace đúng realtime trước đây làm client queue ≈1 →
# bất kỳ jitter (bridge process VST biến thiên, GC, WS) → queue cạn →
# zero-fill crackle. Worklet cap 32 block chặn overflow khi bridge
# nhanh hơn.
# StALL-FADE (fix crackle nặng khi load VST FX): transport stall (browser
# main-thread block GUI/rAF/React) làm bridge ngừng sản xuất output →
# worklet hết buffer (FILL=24 ~128ms) → HOLD 2 block → CẮT cứng xuống
# silence → click. Khi nhận ra output ngừng > ngưỡng → gửi fade-out:
# lặp block real cuối với gain 1→0 qua 8 block (~42ms, pace realtime).
# Worklet hết buffer nhưng nội dung đã về ~0 → hold/silence không click.
# Output trở lại → fade-in 8 block đầu 0→1 (hết click resume).
import struct as _struct
block_sec = (sess.block_size or 256) / max(int(sess.sample_rate or 44100), 1)
# Gap 8 (PDC realtime): drain latency chain tu SHM ring (bridge bao qua
# REPORT_LATENCY per-slot) -> push text frame khi tong doi. Client dung
# de bu delay dry-bypass path (app.jsx fxRtApplyPdc).
# Ngưỡng stall: > batch period (4*block_sec) + jitter; 5.5*block_sec ≈ 29.3ms
# (batch gap thường ~21.3ms + jitter bridge ~2ms sau fix timer 1ms + wakeup
# pump hiếm ~8ms). 6x (32ms) bỏ sót case queue worklet thấp lúc recovery;
# 5x false-fade khi gap 29ms không phải stall thật. 5.5x dung hòa.
_stall_s = 5.5 * block_sec
_fade_blocks = 8
_fade_fmt = '<%df' % ((sess.block_size or 256) * 2)
def _gain(blk, g):
if g >= 1.0: return blk
if g <= 0.0: return b'\x00' * len(blk)
vals = _struct.unpack(_fade_fmt, blk)
return _struct.pack(_fade_fmt, *[v * g for v in vals])
_last_lat = None
_last_blk = None
_last_out = 0.0
_had_out = False
_fading = False
_fade_i = 0
_fade_in = 0
try:
_loop = asyncio.get_event_loop()
while True:
@@ -1119,16 +1145,47 @@ async def ws_fx_realtime(websocket: WebSocket, session_id: str):
pass
blks = sess.read_output()
if not blks:
await asyncio.sleep(0.002)
_now = _loop.time()
_gap = _now - _last_out if _had_out else 0.0
if (_had_out and not _fading and _fade_in <= 0 and
_last_blk is not None and _fade_i == 0 and
_gap > _stall_s):
_fading = True
_fade_i = 0
if _fading and _last_blk is not None and _fade_i < _fade_blocks:
# Gửi BURST toàn bộ fade-out (không pace realtime): worklet
# queue (FILL 24) hấp thụ 8 block; fade frames xếp sau buffer
# real → worklet phát dốc 1→0 khi hết buffer → silence ~0.
# Burst cũng phủ case queue worklet THẤP lúc recovery (fade
# đến trước khi queue cạn, thay vì chậm 42ms như pace thật).
while _fade_i < _fade_blocks:
# gain (8-i-1)/7: block cuối đúng 0.0 → hold/silence ~0
await websocket.send_bytes(_gain(
_last_blk,
(_fade_blocks - 1 - _fade_i) / (_fade_blocks - 1)))
_fade_i += 1
_last_out = _loop.time()
_fading = False # đã về ~0 → worklet tự silence
continue
# to_thread(time.sleep): asyncio.sleep trên Windows bị cap
# 15.6ms (GQCS/IOCP, Win11 cap nền) dù timeBeginPeriod;
# time.sleep dùng high-res waitable timer -> poll đúng 2ms.
await asyncio.to_thread(_time.sleep, 0.002)
continue
# Gửi ASAP (không pace theo realtime): bridge sản xuất burst
# theo batch 4 → gửi hết ngay → CLIENT tích lũy buffer
# (worklet FILL) hấp thụ jitter bridge/WS. Pace đúng realtime
# trước đây làm client queue ≈1 → bất kỳ jitter (bridge process
# VST biến thiên, GC, WS) → queue cạn → zero-fill crackle.
# Worklet cap 32 block chặn overflow khi bridge nhanh hơn.
for _blk in blks:
if _fading or _fade_i >= _fade_blocks:
# resume (giữa fade hoặc sau fade xong) → reset + fade-in
_fading = False
_fade_i = 0
_fade_in = _fade_blocks
if _fade_in > 0:
# gain (8-i-1)/7 dốc 0→1: block cuối full
_blk = _gain(_blk, 1.0 - (_fade_in - 1) / _fade_blocks)
_fade_in -= 1
await websocket.send_bytes(_blk)
_last_blk = _blk
_last_out = _loop.time()
_had_out = True
except Exception:
pass
pump_task = asyncio.create_task(_pump_output())
+10
View File
@@ -128,6 +128,9 @@ class FxRealtimeSession:
# Phase 2.8: C++ consume qua SHM control ring.
self.pending_params = {} # {slot_idx: {key: value}}
self.latencies = {} # {slot_idx: samples}
# Diagnostics (FXRT crackle hunt): drop/publish counters.
self.stats = {'publish': 0, 'drop_batch': 0, 'drop_blocks': 0,
'stale_flush': 0, 'read_blocks': 0, 'read_empty': 0}
self._spawn()
# ── process ─────────────────────────────────────────────────────────────
@@ -266,13 +269,16 @@ class FxRealtimeSession:
self._in_pending = []
self._in_pending_t0 = None
avail = h.in_slots - (h.in_write - h.in_read)
self.stats['publish'] += len(frames)
if avail <= 0:
h.in_read += len(frames) # ring full → drop cả batch
self.stats['drop_batch'] += len(frames)
return
n = min(len(frames), avail)
drop = len(frames) - n
if drop > 0:
h.in_read += drop # drop frame cũ nhất
self.stats['drop_blocks'] += drop
for i in range(n):
L, R = frames[i + drop]
slot = (h.in_write + i) & (h.in_slots - 1)
@@ -287,6 +293,7 @@ class FxRealtimeSession:
burst) → không kẹt latency. Gọi từ _pump_output mỗi vòng."""
with self._lock:
if self._in_pending and self._in_pending_t0 is not None and time.time() - self._in_pending_t0 > timeout:
self.stats['stale_flush'] += 1
self._flush_input_locked()
def read_output(self):
@@ -303,6 +310,9 @@ class FxRealtimeSession:
inter[1::2] = self._np[br:br + self.block_size]
out.append(inter.tobytes())
h.out_read += 1
self.stats['read_blocks'] += len(out)
if not out:
self.stats['read_empty'] += 1
return out
def heartbeat_age(self):
+55 -1
View File
@@ -16,6 +16,7 @@
#define NOMINMAX
#endif
#include <windows.h>
#include <mmsystem.h>
#endif
#include <algorithm>
@@ -156,6 +157,12 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName,
closeShm(v);
return 2;
}
// Windows timer resolution 1ms — bỏ granularity Sleep 15.6ms cho toàn
// process (idle sleep, chain wait, heartbeat). timeEndPeriod đối ứng ở
// cuối. (winmm đã link sẵn — CMakeLists fx_vst_bridge.)
#ifdef _WIN32
timeBeginPeriod(1);
#endif
// Chờ chain ĐẦU TIÊN được worker build xong (gen>=1) trước khi READY.
// setChain() là ASYNC — worker thread load VST3 + preset mất 1-4s; nếu set
// READY ngay, engine mở WS và audio chảy qua chain RỖNG (passthrough), rồi
@@ -197,6 +204,14 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName,
uint32_t outMask = ipc->h.outSlots - 1;
uint64_t processed = 0;
uint64_t lastGen = 0; // chain generation đã báo latency
// ---- FXRT_PERF instrumentation (tạm, đo crackle) ----
uint64_t perfIter = 0, perfProc = 0, perfIdle = 0;
double perfProcSum = 0.0, perfProcMax = 0.0, perfProcMin = 1e9;
double perfLoopSum = 0.0, perfLoopMax = 0.0;
uint32_t perfTakeSum = 0;
auto perfT0 = std::chrono::steady_clock::now();
auto perfRunStart = perfT0;
uint64_t perfLastOut = 0;
while (ipc->h.running) {
if (parentPid && !parentAlive(parentPid)) {
std::cerr << "[RealtimeFxLoop] parent gone — exiting" << std::endl;
@@ -250,7 +265,16 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName,
<< " slot(s)" << std::endl;
}
uint32_t avail = ipc->h.inWrite - ipc->h.inRead;
if (avail == 0) { sleepMs(1); continue; }
if (avail == 0) {
++perfIdle;
// sleep_until: pacing chính xác theo steady_clock (Sleep(1) bị
// granularity 15.6ms khi không có timeBeginPeriod → jitter loop
// tới 32ms, phá headroom FILL của worklet). timeBeginPeriod(1)
// ở đầu loop + sleep_until giữ loopMax ~2ms.
std::this_thread::sleep_until(std::chrono::steady_clock::now() +
std::chrono::milliseconds(1));
continue;
}
// Batch: tiêu thụ tối đa FXRT_IN_SLOTS block, xử lí 1 call. Input về
// đều → take=1 (latency thấp nhất); dồn burst → take>1 gom lại.
const uint32_t take = std::min<uint32_t>(avail, FXRT_IN_SLOTS);
@@ -262,7 +286,13 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName,
ipc->h.inRead++; // consume
off += n;
}
const auto perfP0 = std::chrono::steady_clock::now();
chain.process(L, R, off); // SEH-guarded; chain rỗng = passthrough
const auto perfP1 = std::chrono::steady_clock::now();
const double perfProcMs = std::chrono::duration<double, std::milli>(perfP1 - perfP0).count();
++perfProc; perfProcSum += perfProcMs; perfTakeSum += take;
if (perfProcMs > perfProcMax) perfProcMax = perfProcMs;
if (perfProcMs < perfProcMin) perfProcMin = perfProcMs;
for (uint32_t i = 0, o = 0; i < take; ++i, o += n) {
const uint32_t oslot = ipc->h.outWrite & outMask;
std::memcpy(ipc->outL[oslot], L + o, n * sizeof(float));
@@ -271,10 +301,34 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName,
ipc->h.outWrite++; // publish
++processed;
}
const auto perfIterT1 = std::chrono::steady_clock::now();
const double perfLoopMs = std::chrono::duration<double, std::milli>(perfIterT1 - perfT0).count();
perfT0 = perfIterT1; ++perfIter;
perfLoopSum += perfLoopMs; if (perfLoopMs > perfLoopMax) perfLoopMax = perfLoopMs;
if (perfIter % 100 == 0) {
const double runSec = std::chrono::duration<double>(std::chrono::steady_clock::now() - perfRunStart).count();
std::cerr << "[FxRTPerf] iter=" << perfIter
<< " procN=" << perfProc
<< " procAvg=" << (perfProc ? perfProcSum / perfProc : 0.0) << "ms"
<< " procMax=" << perfProcMax << "ms"
<< " procMin=" << (perfProc ? perfProcMin : 0.0) << "ms"
<< " loopAvg=" << (perfIter ? perfLoopSum / perfIter : 0.0) << "ms"
<< " loopMax=" << perfLoopMax << "ms"
<< " idle=" << perfIdle
<< " takeAvg=" << (perfProc ? (double)perfTakeSum / perfProc : 0.0)
<< " procBlocks=" << processed
<< " rate=" << (runSec > 0 ? processed / runSec : 0.0) << "blk/s"
<< " outDepth=" << (ipc->h.outWrite - ipc->h.outRead)
<< std::endl;
perfProc = 0; perfProcSum = 0.0; perfTakeSum = 0; perfIdle = 0;
}
}
ipc->h.state = FXRT_STATE_STARTING; // đã dừng
hb.join();
closeShm(v);
#ifdef _WIN32
timeEndPeriod(1);
#endif
std::cerr << "[RealtimeFxLoop] exit processed=" << processed << std::endl;
return 0;
}
Binary file not shown.