diff --git a/app/api/v1/plugins.py b/app/api/v1/plugins.py index 4bf5c56..2ffae96 100644 --- a/app/api/v1/plugins.py +++ b/app/api/v1/plugins.py @@ -1093,14 +1093,40 @@ async def ws_fx_realtime(websocket: WebSocket, session_id: str): # nếu chỉ drain trong vòng receive_bytes, output thừa kẹt lại trong SHM ring # (client ngừng gửi → server block ở receive → âm tắc). async def _pump_output(): - # Pace mỗi block theo tốc độ âm thật (256/sr ~ 5.8ms): bridge sản xuất - # output theo burst khi nhận input dồn; bắn toàn bộ cùng lúc → worklet - # overflow drop block cũ nhất → âm ngắt quãng. + # Gửi ASAP (không pace theo realtime): bridge sản xuất burst theo batch + # 4 → gửi hết ngay → CLIENT tích lũy buffer (worklet FILL) hấp thụ + # jitter bridge/WS. Pace đúng realtime trước đây làm client queue ≈1 → + # bất kỳ jitter (bridge process VST biến thiên, GC, WS) → queue cạn → + # zero-fill crackle. Worklet cap 32 block chặn overflow khi bridge + # nhanh hơn. + # StALL-FADE (fix crackle nặng khi load VST FX): transport stall (browser + # main-thread block GUI/rAF/React) làm bridge ngừng sản xuất output → + # worklet hết buffer (FILL=24 ~128ms) → HOLD 2 block → CẮT cứng xuống + # silence → click. Khi nhận ra output ngừng > ngưỡng → gửi fade-out: + # lặp block real cuối với gain 1→0 qua 8 block (~42ms, pace realtime). + # Worklet hết buffer nhưng nội dung đã về ~0 → hold/silence không click. + # Output trở lại → fade-in 8 block đầu 0→1 (hết click resume). + import struct as _struct block_sec = (sess.block_size or 256) / max(int(sess.sample_rate or 44100), 1) - # Gap 8 (PDC realtime): drain latency chain tu SHM ring (bridge bao qua - # REPORT_LATENCY per-slot) -> push text frame khi tong doi. Client dung - # de bu delay dry-bypass path (app.jsx fxRtApplyPdc). + # Ngưỡng stall: > batch period (4*block_sec) + jitter; 5.5*block_sec ≈ 29.3ms + # (batch gap thường ~21.3ms + jitter bridge ~2ms sau fix timer 1ms + wakeup + # pump hiếm ~8ms). 6x (32ms) bỏ sót case queue worklet thấp lúc recovery; + # 5x false-fade khi gap 29ms không phải stall thật. 5.5x dung hòa. + _stall_s = 5.5 * block_sec + _fade_blocks = 8 + _fade_fmt = '<%df' % ((sess.block_size or 256) * 2) + def _gain(blk, g): + if g >= 1.0: return blk + if g <= 0.0: return b'\x00' * len(blk) + vals = _struct.unpack(_fade_fmt, blk) + return _struct.pack(_fade_fmt, *[v * g for v in vals]) _last_lat = None + _last_blk = None + _last_out = 0.0 + _had_out = False + _fading = False + _fade_i = 0 + _fade_in = 0 try: _loop = asyncio.get_event_loop() while True: @@ -1119,16 +1145,47 @@ async def ws_fx_realtime(websocket: WebSocket, session_id: str): pass blks = sess.read_output() if not blks: - await asyncio.sleep(0.002) + _now = _loop.time() + _gap = _now - _last_out if _had_out else 0.0 + if (_had_out and not _fading and _fade_in <= 0 and + _last_blk is not None and _fade_i == 0 and + _gap > _stall_s): + _fading = True + _fade_i = 0 + if _fading and _last_blk is not None and _fade_i < _fade_blocks: + # Gửi BURST toàn bộ fade-out (không pace realtime): worklet + # queue (FILL 24) hấp thụ 8 block; fade frames xếp sau buffer + # real → worklet phát dốc 1→0 khi hết buffer → silence ~0. + # Burst cũng phủ case queue worklet THẤP lúc recovery (fade + # đến trước khi queue cạn, thay vì chậm 42ms như pace thật). + while _fade_i < _fade_blocks: + # gain (8-i-1)/7: block cuối đúng 0.0 → hold/silence ~0 + await websocket.send_bytes(_gain( + _last_blk, + (_fade_blocks - 1 - _fade_i) / (_fade_blocks - 1))) + _fade_i += 1 + _last_out = _loop.time() + _fading = False # đã về ~0 → worklet tự silence + continue + # to_thread(time.sleep): asyncio.sleep trên Windows bị cap + # 15.6ms (GQCS/IOCP, Win11 cap nền) dù timeBeginPeriod; + # time.sleep dùng high-res waitable timer -> poll đúng 2ms. + await asyncio.to_thread(_time.sleep, 0.002) continue - # Gửi ASAP (không pace theo realtime): bridge sản xuất burst - # theo batch 4 → gửi hết ngay → CLIENT tích lũy buffer - # (worklet FILL) hấp thụ jitter bridge/WS. Pace đúng realtime - # trước đây làm client queue ≈1 → bất kỳ jitter (bridge process - # VST biến thiên, GC, WS) → queue cạn → zero-fill crackle. - # Worklet cap 32 block chặn overflow khi bridge nhanh hơn. for _blk in blks: + if _fading or _fade_i >= _fade_blocks: + # resume (giữa fade hoặc sau fade xong) → reset + fade-in + _fading = False + _fade_i = 0 + _fade_in = _fade_blocks + if _fade_in > 0: + # gain (8-i-1)/7 dốc 0→1: block cuối full + _blk = _gain(_blk, 1.0 - (_fade_in - 1) / _fade_blocks) + _fade_in -= 1 await websocket.send_bytes(_blk) + _last_blk = _blk + _last_out = _loop.time() + _had_out = True except Exception: pass pump_task = asyncio.create_task(_pump_output()) diff --git a/app/core/fx_realtime.py b/app/core/fx_realtime.py index af28825..1fe0fdd 100644 --- a/app/core/fx_realtime.py +++ b/app/core/fx_realtime.py @@ -128,6 +128,9 @@ class FxRealtimeSession: # Phase 2.8: C++ consume qua SHM control ring. self.pending_params = {} # {slot_idx: {key: value}} self.latencies = {} # {slot_idx: samples} + # Diagnostics (FXRT crackle hunt): drop/publish counters. + self.stats = {'publish': 0, 'drop_batch': 0, 'drop_blocks': 0, + 'stale_flush': 0, 'read_blocks': 0, 'read_empty': 0} self._spawn() # ── process ───────────────────────────────────────────────────────────── @@ -266,13 +269,16 @@ class FxRealtimeSession: self._in_pending = [] self._in_pending_t0 = None avail = h.in_slots - (h.in_write - h.in_read) + self.stats['publish'] += len(frames) if avail <= 0: h.in_read += len(frames) # ring full → drop cả batch + self.stats['drop_batch'] += len(frames) return n = min(len(frames), avail) drop = len(frames) - n if drop > 0: h.in_read += drop # drop frame cũ nhất + self.stats['drop_blocks'] += drop for i in range(n): L, R = frames[i + drop] slot = (h.in_write + i) & (h.in_slots - 1) @@ -287,6 +293,7 @@ class FxRealtimeSession: burst) → không kẹt latency. Gọi từ _pump_output mỗi vòng.""" with self._lock: if self._in_pending and self._in_pending_t0 is not None and time.time() - self._in_pending_t0 > timeout: + self.stats['stale_flush'] += 1 self._flush_input_locked() def read_output(self): @@ -303,6 +310,9 @@ class FxRealtimeSession: inter[1::2] = self._np[br:br + self.block_size] out.append(inter.tobytes()) h.out_read += 1 + self.stats['read_blocks'] += len(out) + if not out: + self.stats['read_empty'] += 1 return out def heartbeat_age(self): diff --git a/native_bridge/src/RealtimeFxLoop.cpp b/native_bridge/src/RealtimeFxLoop.cpp index 4d3b040..aaf9c4c 100644 --- a/native_bridge/src/RealtimeFxLoop.cpp +++ b/native_bridge/src/RealtimeFxLoop.cpp @@ -16,6 +16,7 @@ #define NOMINMAX #endif #include +#include #endif #include @@ -156,6 +157,12 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName, closeShm(v); return 2; } + // Windows timer resolution 1ms — bỏ granularity Sleep 15.6ms cho toàn + // process (idle sleep, chain wait, heartbeat). timeEndPeriod đối ứng ở + // cuối. (winmm đã link sẵn — CMakeLists fx_vst_bridge.) +#ifdef _WIN32 + timeBeginPeriod(1); +#endif // Chờ chain ĐẦU TIÊN được worker build xong (gen>=1) trước khi READY. // setChain() là ASYNC — worker thread load VST3 + preset mất 1-4s; nếu set // READY ngay, engine mở WS và audio chảy qua chain RỖNG (passthrough), rồi @@ -197,6 +204,14 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName, uint32_t outMask = ipc->h.outSlots - 1; uint64_t processed = 0; uint64_t lastGen = 0; // chain generation đã báo latency + // ---- FXRT_PERF instrumentation (tạm, đo crackle) ---- + uint64_t perfIter = 0, perfProc = 0, perfIdle = 0; + double perfProcSum = 0.0, perfProcMax = 0.0, perfProcMin = 1e9; + double perfLoopSum = 0.0, perfLoopMax = 0.0; + uint32_t perfTakeSum = 0; + auto perfT0 = std::chrono::steady_clock::now(); + auto perfRunStart = perfT0; + uint64_t perfLastOut = 0; while (ipc->h.running) { if (parentPid && !parentAlive(parentPid)) { std::cerr << "[RealtimeFxLoop] parent gone — exiting" << std::endl; @@ -250,7 +265,16 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName, << " slot(s)" << std::endl; } uint32_t avail = ipc->h.inWrite - ipc->h.inRead; - if (avail == 0) { sleepMs(1); continue; } + if (avail == 0) { + ++perfIdle; + // sleep_until: pacing chính xác theo steady_clock (Sleep(1) bị + // granularity 15.6ms khi không có timeBeginPeriod → jitter loop + // tới 32ms, phá headroom FILL của worklet). timeBeginPeriod(1) + // ở đầu loop + sleep_until giữ loopMax ~2ms. + std::this_thread::sleep_until(std::chrono::steady_clock::now() + + std::chrono::milliseconds(1)); + continue; + } // Batch: tiêu thụ tối đa FXRT_IN_SLOTS block, xử lí 1 call. Input về // đều → take=1 (latency thấp nhất); dồn burst → take>1 gom lại. const uint32_t take = std::min(avail, FXRT_IN_SLOTS); @@ -262,7 +286,13 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName, ipc->h.inRead++; // consume off += n; } + const auto perfP0 = std::chrono::steady_clock::now(); chain.process(L, R, off); // SEH-guarded; chain rỗng = passthrough + const auto perfP1 = std::chrono::steady_clock::now(); + const double perfProcMs = std::chrono::duration(perfP1 - perfP0).count(); + ++perfProc; perfProcSum += perfProcMs; perfTakeSum += take; + if (perfProcMs > perfProcMax) perfProcMax = perfProcMs; + if (perfProcMs < perfProcMin) perfProcMin = perfProcMs; for (uint32_t i = 0, o = 0; i < take; ++i, o += n) { const uint32_t oslot = ipc->h.outWrite & outMask; std::memcpy(ipc->outL[oslot], L + o, n * sizeof(float)); @@ -271,10 +301,34 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName, ipc->h.outWrite++; // publish ++processed; } + const auto perfIterT1 = std::chrono::steady_clock::now(); + const double perfLoopMs = std::chrono::duration(perfIterT1 - perfT0).count(); + perfT0 = perfIterT1; ++perfIter; + perfLoopSum += perfLoopMs; if (perfLoopMs > perfLoopMax) perfLoopMax = perfLoopMs; + if (perfIter % 100 == 0) { + const double runSec = std::chrono::duration(std::chrono::steady_clock::now() - perfRunStart).count(); + std::cerr << "[FxRTPerf] iter=" << perfIter + << " procN=" << perfProc + << " procAvg=" << (perfProc ? perfProcSum / perfProc : 0.0) << "ms" + << " procMax=" << perfProcMax << "ms" + << " procMin=" << (perfProc ? perfProcMin : 0.0) << "ms" + << " loopAvg=" << (perfIter ? perfLoopSum / perfIter : 0.0) << "ms" + << " loopMax=" << perfLoopMax << "ms" + << " idle=" << perfIdle + << " takeAvg=" << (perfProc ? (double)perfTakeSum / perfProc : 0.0) + << " procBlocks=" << processed + << " rate=" << (runSec > 0 ? processed / runSec : 0.0) << "blk/s" + << " outDepth=" << (ipc->h.outWrite - ipc->h.outRead) + << std::endl; + perfProc = 0; perfProcSum = 0.0; perfTakeSum = 0; perfIdle = 0; + } } ipc->h.state = FXRT_STATE_STARTING; // đã dừng hb.join(); closeShm(v); +#ifdef _WIN32 + timeEndPeriod(1); +#endif std::cerr << "[RealtimeFxLoop] exit processed=" << processed << std::endl; return 0; } diff --git a/src-tauri/binaries/fx_vst_bridge-x86_64-pc-windows-msvc.exe b/src-tauri/binaries/fx_vst_bridge-x86_64-pc-windows-msvc.exe index cf5c430..650d675 100644 Binary files a/src-tauri/binaries/fx_vst_bridge-x86_64-pc-windows-msvc.exe and b/src-tauri/binaries/fx_vst_bridge-x86_64-pc-windows-msvc.exe differ diff --git a/src-tauri/binaries/fx_vst_bridge.exe b/src-tauri/binaries/fx_vst_bridge.exe index cf5c430..650d675 100644 Binary files a/src-tauri/binaries/fx_vst_bridge.exe and b/src-tauri/binaries/fx_vst_bridge.exe differ