feat(02-03): add waveform extraction module and HTTP endpoint

- Create lightsync/audio/waveform.py with extract_peaks() (AUD-05)
- MP3 via ffmpeg pipe, WAV/FLAC/OGG via soundfile+numpy
- Add GET /api/audio/waveform endpoint with asyncio.to_thread
- peaks param clamped 100-5000, returns {peaks, count, file}
This commit is contained in:
Claude
2026-04-06 13:11:38 +00:00
parent 3de7f43b3f
commit da474d978d
2 changed files with 84 additions and 0 deletions

View File

@@ -3,6 +3,7 @@ import asyncio
from pathlib import Path from pathlib import Path
from fastapi import APIRouter, HTTPException, Request from fastapi import APIRouter, HTTPException, Request
from pydantic import BaseModel from pydantic import BaseModel
from lightsync.audio.waveform import extract_peaks
router = APIRouter() router = APIRouter()
@@ -45,3 +46,28 @@ async def get_audio_state(request: Request):
if engine is None: if engine is None:
raise HTTPException(status_code=503, detail="Audio engine not ready") raise HTTPException(status_code=503, detail="Audio engine not ready")
return engine.get_state() return engine.get_state()
@router.get("/waveform")
async def get_waveform(request: Request, peaks: int = 1000):
"""Get waveform peak data for the currently loaded audio file (AUD-05).
Query params:
peaks: Number of peak samples (default 1000, max 5000)
"""
engine = request.app.state.engine
if engine is None:
raise HTTPException(status_code=503, detail="Audio engine not ready")
state = engine.get_state()
if not state["loaded"] or not state["file"]:
raise HTTPException(status_code=400, detail="No audio file loaded")
num_peaks = min(max(peaks, 100), 5000)
try:
data = await asyncio.to_thread(extract_peaks, state["file"], num_peaks)
except Exception as e:
raise HTTPException(status_code=500, detail=f"Waveform extraction failed: {e}")
return {"peaks": data, "count": len(data), "file": state["file"]}

View File

@@ -0,0 +1,58 @@
"""Waveform peak extraction for timeline display (AUD-05)."""
import subprocess
from pathlib import Path
import numpy as np
import soundfile as sf
def extract_peaks(path: str, num_peaks: int = 1000) -> list[float]:
"""Extract downsampled peak amplitudes from audio file.
Returns a list of floats (0.0-1.0) representing peak amplitude per chunk.
Handles MP3 via ffmpeg subprocess (libsndfile doesn't support MP3).
Args:
path: Absolute path to audio file
num_peaks: Number of peak samples to return (default 1000)
Returns:
List of float peak values, length <= num_peaks
"""
p = Path(path)
if p.suffix.lower() == ".mp3":
return _extract_peaks_mp3(path, num_peaks)
return _extract_peaks_soundfile(path, num_peaks)
def _extract_peaks_soundfile(path: str, num_peaks: int) -> list[float]:
"""Extract peaks using soundfile (WAV, FLAC, OGG)."""
data, _ = sf.read(path, always_2d=True)
mono = np.mean(data, axis=1)
return _downsample_peaks(mono, num_peaks)
def _extract_peaks_mp3(path: str, num_peaks: int) -> list[float]:
"""Extract peaks from MP3 via ffmpeg pipe to float32 PCM."""
cmd = [
"ffmpeg", "-i", path,
"-f", "f32le", "-ar", "44100", "-ac", "1",
"pipe:1", "-loglevel", "quiet",
]
result = subprocess.run(cmd, capture_output=True, timeout=60)
if result.returncode != 0:
raise RuntimeError(f"ffmpeg failed for {path}: exit code {result.returncode}")
mono = np.frombuffer(result.stdout, dtype=np.float32)
return _downsample_peaks(mono, num_peaks)
def _downsample_peaks(mono: np.ndarray, num_peaks: int) -> list[float]:
"""Downsample mono audio to peak amplitude array."""
if len(mono) == 0:
return []
chunk_size = max(1, len(mono) // num_peaks)
peaks = []
for i in range(0, len(mono), chunk_size):
chunk = mono[i : i + chunk_size]
peaks.append(float(np.max(np.abs(chunk))))
return peaks[:num_peaks]