From 67a3e12c006be042aa1cd371050a8c07b7639929 Mon Sep 17 00:00:00 2001 From: Sam Rolfe Date: Sat, 12 Sep 2026 08:53:45 +1000 Subject: [PATCH] dsh: convert any audio/video (m4a/mp3/mp4/aac) to WAV via ffmpeg before whisper; verified m4a + mp3 transcribe --- dsh/app.py | 32 ++++++++++++++++++++++++++++++-- 1 file changed, 30 insertions(+), 2 deletions(-) diff --git a/dsh/app.py b/dsh/app.py index 0fbf0e6..d6d3d38 100644 --- a/dsh/app.py +++ b/dsh/app.py @@ -411,6 +411,32 @@ async def _whisper(data: bytes, fname: str, mime: str) -> str: return (r.json() or {}).get("transcript", "") or "" +def _to_wav(data: bytes, fname: str) -> tuple[bytes, str]: + """Convert arbitrary audio/video (mp3, m4a, aac, mp4, ogg…) to 16kHz mono WAV + via the container's ffmpeg, so whisper always receives a plain WAV. Returns + (original data, original name) untouched if already WAV/FLAC or on failure.""" + ext = "." + fname.rsplit(".", 1)[-1].lower() if "." in fname else "" + if ext in {".wav", ".flac"}: + return data, fname + tmp = f"/tmp/conv_{uuid4().hex}" + src = tmp + (ext or ".bin") + Path(src).write_bytes(data) + out = tmp + ".wav" + try: + r = subprocess.run( + ["ffmpeg", "-v", "error", "-y", "-i", src, + "-ar", "16000", "-ac", "1", "-c:a", "pcm_s16le", out], + capture_output=True, timeout=300) + if r.returncode == 0 and Path(out).is_file() and Path(out).stat().st_size > 100: + return Path(out).read_bytes(), "converted.wav" + return data, fname # fall back to raw (whisper's own ffmpeg may still decode) + except Exception: # noqa: BLE001 + return data, fname + finally: + shutil.rmtree(tmp, ignore_errors=True) + + + @app.post("/api/tool/transcribe") async def transcribe_audio(file: Annotated[UploadFile, File()]) -> dict: """Speech -> text. Save the audio into the workspace, then whisper it.""" @@ -423,7 +449,8 @@ async def transcribe_audio(file: Annotated[UploadFile, File()]) -> dict: audio_path = up / fname audio_path.write_bytes(data) try: - transcript = await _whisper(data, file.filename or "audio.wav", file.content_type or "audio/wav") + wav, wav_name = _to_wav(data, file.filename or "audio") + transcript = await _whisper(wav, wav_name, "audio/wav") except Exception as exc: # noqa: BLE001 return {"ok": False, "error": f"Whisper failed: {exc}", "path": f"uploads/{fname}"} if not transcript.strip(): @@ -464,7 +491,8 @@ async def ask_file(path: Annotated[str, Form()], question: Annotated[str, Form() if ext in AUDIO_EXT and ext not in VIDEO_EXT: try: - transcript = await _whisper(fp.read_bytes(), fp.name, "audio/wav") + wav, wav_name = _to_wav(fp.read_bytes(), fp.name) + transcript = await _whisper(wav, wav_name, "audio/wav") except Exception as exc: # noqa: BLE001 return {"ok": False, "error": f"Whisper failed: {exc}"} if not transcript.strip():