""" Audio processing service for data augmentation and manipulation. """ import io import logging import subprocess import tempfile from pathlib import Path import numpy as np import librosa import soundfile as sf import scipy.signal from fastapi import UploadFile from app.schemas import AudioAugmentationOptions, AudioConvertOptions logger = logging.getLogger(__name__) # soundfile/libsndfile writes these natively — no subprocess needed. _NATIVE_FORMATS = {"wav": "WAV", "flac": "FLAC"} # Everything else (mp3, ogg) goes through ffmpeg, same subprocess pattern # already used elsewhere in this codebase for audio decode (see # feature_extractor.py / vocal_analyzer.py's _ffmpeg_decode). _FFMPEG_CODEC = {"mp3": "libmp3lame", "ogg": "libvorbis"} def process_audio( file_bytes: bytes, options: AudioAugmentationOptions, mix_with_bytes: bytes | None = None, ) -> io.BytesIO: """ Process audio file with requested augmentation options. Returns processed audio as BytesIO (WAV format). mix_with_bytes: a second audio file to mix in when options.mix_audio is set. Silently ignored (mix skipped) if mix_audio is off or no second file was provided — the route layer decides whether that's an error. """ try: # Load audio from bytes # librosa.load expects a file path or file-like object y, sr = librosa.load(io.BytesIO(file_bytes), sr=None) # 0. Mix Audio — blend in a second track, resampled to match and # looped/trimmed to the primary track's length so levels stay sane # regardless of which clip is longer. if options.mix_audio and mix_with_bytes: y2, sr2 = librosa.load(io.BytesIO(mix_with_bytes), sr=sr) if len(y2) < len(y): repeats = int(np.ceil(len(y) / max(len(y2), 1))) y2 = np.tile(y2, repeats) y2 = y2[: len(y)] y = librosa.util.normalize(y * 0.6 + y2 * 0.6) logger.info("Applied mix_audio") # 1. Trim Silence if options.trim_silence: y, _ = librosa.effects.trim(y, top_db=20) logger.info("Applied trim_silence") # 2. Pitch Shift (Randomly between -2 and +2 semitones if enabled) if options.pitch_shift: n_steps = np.random.uniform(-2, 2) y = librosa.effects.pitch_shift(y, sr=sr, n_steps=n_steps) logger.info(f"Applied pitch_shift: {n_steps:.2f}") # 3. Speed Change (Randomly between 0.9x and 1.1x) if options.speed_change: rate = np.random.uniform(0.9, 1.1) y = librosa.effects.time_stretch(y, rate=rate) logger.info(f"Applied speed_change: {rate:.2f}") # 4. Add Noise if options.add_noise: noise_amp = 0.005 * np.max(np.abs(y)) y = y + noise_amp * np.random.normal(size=len(y)) logger.info("Applied add_noise") # 5. Bass Boost (Simple Low-Shelf Filter) if options.bass_boost: # Create a simple low-shelf filter emphasizing < 200Hz # This is a basic implementation using scipy sos = scipy.signal.butter(10, 200, 'lp', fs=sr, output='sos') y_boosted = scipy.signal.sosfilt(sos, y) # Mix original with boosted low-end y = y + (y_boosted * 0.5) # Normalize to prevent clipping y = librosa.util.normalize(y) logger.info("Applied bass_boost") # Export to BytesIO as WAV out_buffer = io.BytesIO() sf.write(out_buffer, y, sr, format='WAV') out_buffer.seek(0) return out_buffer except Exception as e: logger.error(f"Error processing audio: {str(e)}", exc_info=True) raise ValueError(f"Audio processing failed: {str(e)}") def convert_audio_format(file_bytes: bytes, options: AudioConvertOptions) -> io.BytesIO: """ Convert an audio file to the requested target format. Returns the converted audio as BytesIO. """ target = options.target_format if target in _NATIVE_FORMATS: try: y, sr = librosa.load(io.BytesIO(file_bytes), sr=None) out_buffer = io.BytesIO() sf.write(out_buffer, y, sr, format=_NATIVE_FORMATS[target]) out_buffer.seek(0) return out_buffer except Exception as e: logger.error(f"Error converting audio to {target}: {str(e)}", exc_info=True) raise ValueError(f"Audio conversion failed: {str(e)}") if target in _FFMPEG_CODEC: with tempfile.NamedTemporaryFile(suffix=f".{target}", delete=False) as tmp: tmp_path = tmp.name try: result = subprocess.run( [ "ffmpeg", "-y", "-i", "pipe:0", "-c:a", _FFMPEG_CODEC[target], "-b:a", f"{options.bitrate_kbps}k", tmp_path, ], input=file_bytes, capture_output=True, timeout=60, ) if result.returncode != 0: logger.error(f"ffmpeg conversion to {target} failed: {result.stderr.decode(errors='replace')[:300]}") raise ValueError(f"Audio conversion to {target} failed") with open(tmp_path, "rb") as f: return io.BytesIO(f.read()) finally: Path(tmp_path).unlink(missing_ok=True) raise ValueError(f"Unsupported target format: {target}")