Normalize Linux audio pipeline to lossless Float32 (skill: lossless-audio-compliance)

- native bridge WAV writers: 16-bit PCM truncation -> IEEE Float32 (fmt 3, bit-exact)
- vendor sheredom_json.h (vst3sdk 3.8.1 lacks moduleinfo/json.h)
- Dockerfile: drop nonexistent sfizz tag v1.2.3
- Python sf.write -> subtype=FLOAT everywhere
- resample_poly -> kaiser(16) window (>=130dB stopband)
- audio_editor._write_wav_lossless(): FLOAT passthrough, TPDF dither only for
  8/16/24-bit final exports
- app.jsx WAV encoders: Float32 default, TPDF dither for integer depths, fix 24-bit
- tests: lossless compliance gate (null test < -140dBFS, float32 format) 9/9 pass
- TASKS_WINDOWS_AUDIO_NORMALIZE.md: Windows-side remaining tasks
This commit is contained in:
2026-08-18 11:50:26 +07:00
parent c382afe2a6
commit a2f93138be
18 changed files with 3842 additions and 72 deletions
+17 -10
View File
@@ -5,6 +5,20 @@ import soundfile as sf
from pydub import AudioSegment
from app.core.dsp_utils import find_nearest_zero_crossing_file, apply_micro_fade
def _write_wav_lossless(output_path, y, sample_rate, bit_depth):
"""Lossless WAV export (lossless-audio-compliance skill):
- 32-bit -> IEEE Float32, bit-exact (no dither, format 3)
- 8/16/24-bit -> single-stage TPDF dither ONLY at final quantization"""
if bit_depth == 32:
sf.write(output_path, y, sample_rate, subtype="FLOAT")
return
subtype = {8: "PCM_S8", 16: "PCM_16", 24: "PCM_24"}.get(bit_depth, "PCM_24")
rng = np.random.default_rng()
y = y + (rng.random(y.shape) - rng.random(y.shape)) # TPDF, [-1,1)
y = np.clip(y, -1.0, 1.0) # clamp after dither — no wrap on quantization
sf.write(output_path, y, sample_rate, subtype=subtype)
def edit_audio_file(config: dict, input_path: str, output_path: str):
"""
Applies editing commands on the audio file based on config:
@@ -192,12 +206,8 @@ def mix_multitrack_session(tracks_meta: list, output_path: str, sample_rate: int
master_mix.export(temp_wav, format="wav")
y, sr_read = sf.read(temp_wav)
# Xác định subtype mã hóa bit-depth
subtype_map = {8: "PCM_S8", 16: "PCM_16", 24: "PCM_24"}
selected_subtype = subtype_map.get(bit_depth, "PCM_16")
# Ghi tệp WAV chất lượng cao
sf.write(output_path, y, sample_rate, subtype=selected_subtype)
# Ghi tệp WAV chất lượng cao (float32 / TPDF-dithered)
_write_wav_lossless(output_path, y, sample_rate, bit_depth)
finally:
# Dọn dẹp tệp tạm (luôn thực hiện)
if os.path.exists(temp_wav):
@@ -254,10 +264,7 @@ def export_audio(input_path: str, output_path: str, format: str = "wav",
sound.export(temp_wav, format="wav")
y, sr_read = sf.read(temp_wav)
subtype_map = {8: "PCM_S8", 16: "PCM_16", 24: "PCM_24"}
selected_subtype = subtype_map.get(bit_depth, "PCM_16")
sf.write(output_path, y, sample_rate, subtype=selected_subtype)
_write_wav_lossless(output_path, y, sample_rate, bit_depth)
finally:
if os.path.exists(temp_wav):
os.remove(temp_wav)