fix(fx-rt): burst 4-block input batch qua bridge — hết lag/giật/crackle master FX

- worklet gom 1024 mẫu (4 block) vào 1 frame interleaved, main thread tách 4 WS frame 512 floats
- bridge take=min(avail,4) xử lí 1 call process cho burst → overhead VST3 amortize 4 lần
- engine _pump_output pace theo deadline thay vì sleep cố định (outRead 131→187.5/s)
- FX_RT_ROUNDTRIP_SAMPLES 512→1792 (RTT batching ~7 block)
This commit is contained in:
2026-08-23 11:44:54 +07:00
parent bec82a20bc
commit 0acc3ee7ee
7 changed files with 108 additions and 52 deletions
+29 -14
View File
@@ -127,8 +127,14 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName,
std::cerr << "[RealtimeFxLoop] no fx_chain in job (empty chain = passthrough)" << std::endl;
}
// Batching (fix lag/giật/crackle): gom tối đa FXRT_IN_SLOTS block input
// (256 mẫu) → 1 call chain.process(L,R,batchN). Overhead VST3 (memcpy,
// lock param, setup processData, memset output) amortize 4 lần → engine
// theo kịp 187.5 block/s. setChain với batchN để plugin buffer đủ cho
// process n lớn (không overflow khi xử lí burst).
const uint32_t batchN = block * FXRT_IN_SLOTS;
RealtimeFxChain chain;
chain.setChain(chainJson, sampleRate, block);
chain.setChain(chainJson, sampleRate, (int32_t)batchN);
ShmView* v = openShm(shmName, sizeof(FxRealtimeIPC));
if (!v) {
@@ -163,6 +169,7 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName,
});
const uint32_t n = block;
float L[FXRT_BLOCK * FXRT_IN_SLOTS], R[FXRT_BLOCK * FXRT_IN_SLOTS];
uint32_t inMask = ipc->h.inSlots - 1;
uint32_t outMask = ipc->h.outSlots - 1;
uint64_t processed = 0;
@@ -201,20 +208,28 @@ int run_realtime_fx_loop(const std::string& jobPath, const std::string& shmName,
std::cerr << "[RealtimeFxLoop] latency reported: " << lats.size()
<< " slot(s)" << std::endl;
}
const uint32_t avail = ipc->h.inWrite - ipc->h.inRead;
uint32_t avail = ipc->h.inWrite - ipc->h.inRead;
if (avail == 0) { sleepMs(1); continue; }
const uint32_t slot = ipc->h.inRead & inMask;
float L[FXRT_BLOCK], R[FXRT_BLOCK];
std::memcpy(L, ipc->inL[slot], n * sizeof(float));
std::memcpy(R, ipc->inR[slot], n * sizeof(float));
ipc->h.inRead++; // consume
chain.process(L, R, n); // SEH-guarded; chain rỗng = passthrough
const uint32_t oslot = ipc->h.outWrite & outMask;
std::memcpy(ipc->outL[oslot], L, n * sizeof(float));
std::memcpy(ipc->outR[oslot], R, n * sizeof(float));
MemoryBarrier(); // dữ liệu trước index (đúng thứ tự trên ARM64)
ipc->h.outWrite++; // publish
++processed;
// Batch: tiêu thụ tối đa FXRT_IN_SLOTS block, xử lí 1 call. Input về
// đều → take=1 (latency thấp nhất); dồn burst → take>1 gom lại.
const uint32_t take = std::min<uint32_t>(avail, FXRT_IN_SLOTS);
uint32_t off = 0;
for (uint32_t i = 0; i < take; ++i) {
const uint32_t slot = ipc->h.inRead & inMask;
std::memcpy(L + off, ipc->inL[slot], n * sizeof(float));
std::memcpy(R + off, ipc->inR[slot], n * sizeof(float));
ipc->h.inRead++; // consume
off += n;
}
chain.process(L, R, off); // SEH-guarded; chain rỗng = passthrough
for (uint32_t i = 0, o = 0; i < take; ++i, o += n) {
const uint32_t oslot = ipc->h.outWrite & outMask;
std::memcpy(ipc->outL[oslot], L + o, n * sizeof(float));
std::memcpy(ipc->outR[oslot], R + o, n * sizeof(float));
MemoryBarrier(); // dữ liệu trước index (đúng thứ tự trên ARM64)
ipc->h.outWrite++; // publish
++processed;
}
}
ipc->h.state = FXRT_STATE_STARTING; // đã dừng
hb.join();