ai-singing-studio / demucs_wrapper.py
krishnabalaji's picture
Upload 7 files
c8441af verified
Raw History Blame Contribute Delete
2.8 kB
"""
demucs_wrapper.py
Vocal separation using Demucs htdemucs model.
"""
import os
import subprocess
import shutil
from pathlib import Path
def separate_vocals(input_path: str, output_dir: str) -> dict:
"""
Separate vocals and instrumental from a song using Demucs.
Args:
input_path: Path to input audio file (MP3/WAV)
output_dir: Directory to save separated tracks
Returns:
dict with keys 'vocals' and 'instrumental' pointing to output paths
"""
input_path = Path(input_path)
output_dir = Path(output_dir)
output_dir.mkdir(parents=True, exist_ok=True)
if not input_path.exists():
raise FileNotFoundError(f"Input file not found: {input_path}")
# Run Demucs via subprocess
cmd = [
"python", "-m", "demucs",
"--two-stems", "vocals", # Only separate vocals vs rest
"-n", "htdemucs", # Use htdemucs model
"--out", str(output_dir),
str(input_path)
]
try:
result = subprocess.run(
cmd,
capture_output=True,
text=True,
timeout=600 # 10 min timeout for long songs
)
if result.returncode != 0:
raise RuntimeError(f"Demucs failed:\n{result.stderr}")
except subprocess.TimeoutExpired:
raise RuntimeError("Demucs timed out. Try a shorter audio clip.")
except FileNotFoundError:
raise RuntimeError("Demucs not found. Install with: pip install demucs")
# Demucs output structure: output_dir/htdemucs/<song_name>/vocals.wav + no_vocals.wav
song_name = input_path.stem
demucs_out = output_dir / "htdemucs" / song_name
vocals_src = demucs_out / "vocals.wav"
instrumental_src = demucs_out / "no_vocals.wav"
if not vocals_src.exists():
raise FileNotFoundError(f"Demucs did not produce vocals.wav at {vocals_src}")
if not instrumental_src.exists():
raise FileNotFoundError(f"Demucs did not produce no_vocals.wav at {instrumental_src}")
# Copy to flat output directory for easier access
vocals_dst = output_dir / "vocals.wav"
instrumental_dst = output_dir / "instrumental.wav"
shutil.copy2(vocals_src, vocals_dst)
shutil.copy2(instrumental_src, instrumental_dst)
return {
"vocals": str(vocals_dst),
"instrumental": str(instrumental_dst)
}
def get_audio_duration(path: str) -> float:
"""Get audio duration in seconds using ffprobe."""
cmd = [
"ffprobe", "-v", "quiet",
"-show_entries", "format=duration",
"-of", "default=noprint_wrappers=1:nokey=1",
str(path)
]
try:
result = subprocess.run(cmd, capture_output=True, text=True)
return float(result.stdout.strip())
except Exception:
return 0.0