186 lines
6.3 KiB
Python
186 lines
6.3 KiB
Python
from fastapi import APIRouter, HTTPException
|
|
from pydantic import BaseModel
|
|
from fastapi.responses import FileResponse
|
|
import os
|
|
from pathlib import Path
|
|
import uuid
|
|
from typing import List
|
|
import asyncio
|
|
import subprocess
|
|
import shutil
|
|
import uuid
|
|
# moviepy is imported lazily inside the conversion function to avoid
|
|
# failing module import at application startup when the package is missing.
|
|
|
|
router = APIRouter(prefix="/convert", tags=["convert"])
|
|
|
|
|
|
@router.get("/list-audio", response_model=List[str])
|
|
async def list_audio_files():
|
|
base_dir = Path(__file__).resolve().parent.parent.parent / "data" / "audio"
|
|
if not base_dir.exists():
|
|
return []
|
|
files = [f.name for f in base_dir.iterdir() if f.is_file()]
|
|
return files
|
|
|
|
|
|
class DownloadAudioRequest(BaseModel):
|
|
filename: str
|
|
|
|
|
|
@router.post("/download-audio")
|
|
async def download_audio_file(data: DownloadAudioRequest):
|
|
base_dir = Path(__file__).resolve().parent.parent.parent / "data" / "audio"
|
|
src = (base_dir / data.filename).resolve()
|
|
try:
|
|
src.relative_to(base_dir.parent)
|
|
except Exception:
|
|
raise HTTPException(status_code=400, detail="Invalid filename")
|
|
if not src.exists():
|
|
raise HTTPException(status_code=404, detail="File not found")
|
|
return FileResponse(str(src), filename=src.name, media_type="audio/mpeg")
|
|
|
|
|
|
# New: convert a video file (located in ./data) to mp3 using bundled imageio-ffmpeg
|
|
class VideoToMp3Request(BaseModel):
|
|
filename: str
|
|
|
|
|
|
def _run_moviepy(src: Path, dest: Path):
|
|
from moviepy import VideoFileClip
|
|
|
|
clip = VideoFileClip(str(src))
|
|
try:
|
|
if clip.audio is None:
|
|
raise RuntimeError("Source file has no audio stream")
|
|
clip.audio.write_audiofile(str(dest), codec="libmp3lame", bitrate="192k")
|
|
finally:
|
|
clip.close()
|
|
|
|
|
|
@router.post("/video-to-mp3")
|
|
async def video_to_mp3(req: VideoToMp3Request):
|
|
# ensure MoviePy is available before doing work, return clear 503 if not
|
|
try:
|
|
import moviepy # noqa: F401
|
|
except Exception:
|
|
raise HTTPException(status_code=503, detail="moviepy is not installed in the runtime. Rebuild image or install moviepy/imageio-ffmpeg")
|
|
|
|
data_dir = Path(__file__).resolve().parent.parent.parent / "data"
|
|
src = (data_dir / req.filename).resolve()
|
|
try:
|
|
src.relative_to(data_dir)
|
|
except Exception:
|
|
raise HTTPException(status_code=400, detail="Invalid filename")
|
|
if not src.exists() or not src.is_file():
|
|
raise HTTPException(status_code=404, detail="Source file not found")
|
|
|
|
out_dir = data_dir / "audio"
|
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
dest = out_dir / f"{src.stem}.mp3"
|
|
|
|
try:
|
|
# run blocking MoviePy conversion in threadpool to avoid blocking event loop
|
|
await asyncio.get_running_loop().run_in_executor(None, _run_moviepy, src, dest)
|
|
except Exception as e:
|
|
raise HTTPException(status_code=500, detail=f"conversion failed: {e}")
|
|
|
|
return FileResponse(str(dest), filename=dest.name, media_type="audio/mpeg")
|
|
|
|
|
|
# Convert an existing MP3 (in data/audio) to WAV and run transcription
|
|
class Mp3ToTextRequest(BaseModel):
|
|
filename: str
|
|
model_path: str
|
|
|
|
|
|
def _run_ffmpeg(src: Path, dest: Path):
|
|
# produce mono 16k WAV suitable for Vosk
|
|
cmd = [
|
|
"ffmpeg",
|
|
"-y",
|
|
"-i",
|
|
str(src),
|
|
"-ac",
|
|
"1",
|
|
"-ar",
|
|
"16000",
|
|
"-vn",
|
|
"-f",
|
|
"wav",
|
|
str(dest),
|
|
]
|
|
subprocess.run(cmd, check=True)
|
|
|
|
|
|
@router.post("/mp3-to-text")
|
|
async def mp3_to_text(req: Mp3ToTextRequest):
|
|
logger = __import__("logging").getLogger("convert.mp3-to-text")
|
|
try:
|
|
if shutil.which("ffmpeg") is None:
|
|
raise HTTPException(status_code=503, detail="ffmpeg is not available in the runtime. Install ffmpeg in the environment/container.")
|
|
except Exception:
|
|
raise HTTPException(status_code=503, detail="ffmpeg is not available in the runtime. Install ffmpeg in the environment/container.")
|
|
|
|
data_dir = Path(__file__).resolve().parent.parent.parent / "data"
|
|
audio_dir = data_dir / "audio"
|
|
src = (audio_dir / req.filename).resolve()
|
|
try:
|
|
src.relative_to(audio_dir)
|
|
except Exception:
|
|
raise HTTPException(status_code=400, detail="Invalid filename")
|
|
if not src.exists() or not src.is_file():
|
|
raise HTTPException(status_code=404, detail="Source file not found")
|
|
|
|
audio_dir.mkdir(parents=True, exist_ok=True)
|
|
dest = audio_dir / f"{src.stem}_{uuid.uuid4().hex}.wav"
|
|
|
|
try:
|
|
logger.info("Starting ffmpeg conversion %s -> %s", src.name, dest.name)
|
|
await asyncio.get_running_loop().run_in_executor(None, _run_ffmpeg, src, dest)
|
|
logger.info("ffmpeg conversion finished")
|
|
except subprocess.CalledProcessError as e:
|
|
raise HTTPException(status_code=500, detail=f"ffmpeg conversion failed: {e}")
|
|
except Exception as e:
|
|
raise HTTPException(status_code=500, detail=f"conversion failed: {e}")
|
|
|
|
# Use class-based transcriber
|
|
try:
|
|
from api.lib.audio_processing import VoskModelManager
|
|
except Exception as e:
|
|
try:
|
|
dest.unlink()
|
|
except Exception:
|
|
pass
|
|
raise HTTPException(status_code=500, detail=f"Internal import error: {e}")
|
|
|
|
try:
|
|
logger.info("Starting transcription stage for %s", dest.name)
|
|
# Resolve model path (relative to project root or absolute) and ensure it's inside ./models
|
|
project_root = Path(__file__).resolve().parent.parent.parent
|
|
models_dir = project_root / "models"
|
|
model_path = Path(req.model_path)
|
|
if not model_path.is_absolute():
|
|
model_path = project_root / req.model_path
|
|
model_path = model_path.resolve()
|
|
try:
|
|
model_path.relative_to(models_dir)
|
|
except Exception:
|
|
raise HTTPException(status_code=400, detail="model_path must point inside the project's models directory")
|
|
if not model_path.exists():
|
|
raise HTTPException(status_code=400, detail=f"Model path not found: {model_path}")
|
|
|
|
mgr = VoskModelManager(model_path)
|
|
text = mgr.transcribe_wav(dest)
|
|
logger.info("Transcription stage finished for %s", dest.name)
|
|
finally:
|
|
try:
|
|
dest.unlink()
|
|
except Exception:
|
|
pass
|
|
|
|
return {"text": text}
|
|
|
|
|
|
# Note: this endpoint uses the bundled ffmpeg provided by `imageio-ffmpeg`.
|