import numpy as np import librosa from pydub import AudioSegment def find_zero_crossing(y: np.ndarray, sr: int, target_time: float, window_seconds: float = 0.04) -> float: """ Tìm điểm zero-crossing gần nhất với mốc thời gian đích (giây) để tránh click/pop. Args: y: Mảng biên độ âm thanh (1D numpy array, Mono) sr: Tần số lấy mẫu (Sample Rate) target_time: Vị trí mong muốn cắt (giây) window_seconds: Cửa sổ quét (mặc định 40ms) Returns: float: Thời gian của điểm zero-crossing gần nhất (giây) """ if target_time is None or target_time < 0: return target_time target_sample = int(target_time * sr) window_samples = int(window_seconds * sr) # Xác định giới hạn vùng quét an toàn start_idx = max(0, target_sample - window_samples) end_idx = min(len(y) - 2, target_sample + window_samples) if start_idx >= end_idx: return target_time # Lấy phân khúc sóng âm trong cửa sổ quét y_window = y[start_idx:end_idx] # Tìm các điểm đổi dấu: y[i] * y[i+1] <= 0 # Sử dụng np.sign và np.diff để tìm điểm đổi dấu nhanh chóng signs = np.sign(y_window) # Bất kỳ vị trí nào diff != 0 nghĩa là có sự đổi dấu (đi qua điểm 0) zero_crossings = np.where(np.diff(signs) != 0)[0] if len(zero_crossings) == 0: return target_time # Không tìm thấy, trả về vị trí gốc # Chuyển chỉ số vùng quét về chỉ số mảng tuyệt đối absolute_crossings = zero_crossings + start_idx # Tìm điểm gần với target_sample nhất distances = np.abs(absolute_crossings - target_sample) closest_sample_idx = absolute_crossings[np.argmin(distances)] # Trả về thời gian tương ứng (giây) return float(closest_sample_idx / sr) def find_nearest_zero_crossing_file(file_path: str, target_time_sec: float, search_window_sec: float = 0.04) -> float: """ Tìm điểm zero-crossing từ file âm thanh. Wrapper cho hàm find_zero_crossing để tương thích với code cũ. """ try: # Load mono audio for zero crossing analysis y, sr = librosa.load(file_path, sr=None, mono=True) return find_zero_crossing(y, sr, target_time_sec, search_window_sec) except Exception as e: print(f"Error finding zero crossing: {e}") return target_time_sec def find_nearest_zero_crossing(y: np.ndarray, sr: int, target_time_sec: float, search_window_sec: float = 0.04) -> float: """ Tương thích với code cũ. Gọi đến hàm find_zero_crossing mới. """ return find_zero_crossing(y, sr, target_time_sec, search_window_sec) def apply_micro_fade(segment: AudioSegment, fade_duration_ms: int = 50) -> AudioSegment: """ Áp dụng micro-fades (fade-in và fade-out) để triệt tiêu click/pop. """ if len(segment) > fade_duration_ms * 2: return segment.fade_in(fade_duration_ms).fade_out(fade_duration_ms) elif len(segment) > fade_duration_ms: return segment.fade_in(fade_duration_ms // 2).fade_out(fade_duration_ms // 2) return segment def generate_peak_waveform(file_path: str, num_peaks: int = 800) -> dict: """ Tạo dữ liệu peak waveform cho hiển thị đồ thị sóng âm trên Frontend. Dùng để vẽ waveform đồng bộ với Client (thay thế Web Audio API decodeAudioData). Args: file_path: Đường dẫn tệp âm thanh num_peaks: Số lượng điểm peak trả về (tương ứng pixel width trên UI) Returns: dict: {"peaks": [...], "duration": float, "sample_rate": int} """ # Load mono audio y, sr = librosa.load(file_path, sr=None, mono=True) total_samples = len(y) duration = float(total_samples) / sr if total_samples == 0: return { "peaks": [], "duration": 0.0, "sample_rate": sr } # Tính kích thước mỗi chunk samples_per_peak = max(1, total_samples // num_peaks) peaks = [] for i in range(0, total_samples, samples_per_peak): chunk = y[i:i + samples_per_peak] if len(chunk) > 0: # Peak = giá trị tuyệt đối lớn nhất trong chunk peak_val = float(np.max(np.abs(chunk))) peaks.append(round(peak_val, 6)) # Giới hạn đúng số lượng peaks yêu cầu if len(peaks) > num_peaks: peaks = peaks[:num_peaks] return { "peaks": peaks, "duration": round(duration, 4), "sample_rate": sr } def generate_rms_waveform(file_path: str, num_points: int = 800) -> dict: """ Tạo dữ liệu RMS waveform (mượt hơn peak waveform). Args: file_path: Đường dẫn tệp âm thanh num_points: Số lượng điểm RMS trả về Returns: dict: {"rms": [...], "duration": float, "sample_rate": int} """ y, sr = librosa.load(file_path, sr=None, mono=True) total_samples = len(y) duration = float(total_samples) / sr if total_samples == 0: return { "rms": [], "duration": 0.0, "sample_rate": sr } samples_per_point = max(1, total_samples // num_points) rms_values = [] for i in range(0, total_samples, samples_per_point): chunk = y[i:i + samples_per_point] if len(chunk) > 0: rms_val = float(np.sqrt(np.mean(chunk ** 2))) rms_values.append(round(rms_val, 6)) if len(rms_values) > num_points: rms_values = rms_values[:num_points] return { "rms": rms_values, "duration": round(duration, 4), "sample_rate": sr }