Files
LLM-infa/api/route/pipeline.py
jze9 bf854f2d6e Разделы, фильтры и просмотр текстов в UI; опция «не хранить видео»
- таблица sections (дерево через parent_id), у видео section_id
- API: CRUD /sections, фильтры /videos (поиск, раздел с потомками, метод),
  GET /videos/{id}/text, PATCH раздела, DELETE вместе с файлами
- pipeline: keep_video=false удаляет mp4 после расшифровки, section_id
- UI: панель разделов с вложенностью и счётчиками, поиск, фильтр по методу,
  диалоги полного текста/выжимки, привязка к разделу, удаление,
  переключатель «Сохранять видео на диске»
- compose: bind-mount ./web для правок без пересборки

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-14 12:07:55 +05:00

212 lines
6.9 KiB
Python

"""End-to-end pipeline: download → subtitles-or-transcribe → summarize → persist."""
from __future__ import annotations
import asyncio
import logging
import uuid as _uuid
from pathlib import Path
from fastapi import APIRouter, HTTPException
from pydantic import BaseModel
from api.config import settings
from api.db.video import create_video
from api.lib.llm import LLMConfig, from_settings as llm_from_settings
from api.lib.summarize import summarize
from api.lib.transcribe import transcribe_via_ffmpeg
from api.lib.youtube import download_video_with_subs, vtt_to_text
router = APIRouter(prefix="/pipeline", tags=["pipeline"])
logger = logging.getLogger("pipeline")
_PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
_VIDEO_DIR = _PROJECT_ROOT / "data" / "videos"
_TEXT_DIR = _PROJECT_ROOT / "data" / "text"
_MODELS_DIR = _PROJECT_ROOT / "models"
class LLMOverride(BaseModel):
base_url: str
api_key: str = ""
model: str
timeout: int = 120
max_tokens: int = 1000
temperature: float = 0.3
extra_headers: dict[str, str] = {}
system_prompt: str | None = None
user_template: str | None = None
class ProcessRequest(BaseModel):
url: str
model_path: str | None = None # path inside models/, used when subtitles are unavailable
summary_max_sentences: int = 15
keep_video: bool = True # False: файл видео удаляется после расшифровки
section_id: int | None = None
# If supplied, this LLM is used for the summary; otherwise env-configured LLM;
# otherwise the built-in extractive fallback.
llm: LLMOverride | None = None
class ProcessResponse(BaseModel):
uuid: str
title: str
source_url: str
video_path: str
text_full_path: str
text_summary_path: str
transcription_method: str
summary_preview: str
def _resolve_model_path(req_model_path: str | None) -> Path:
raw = req_model_path or settings.DEFAULT_VOSK_MODEL
p = Path(raw)
if not p.is_absolute():
p = _PROJECT_ROOT / raw
p = p.resolve()
try:
p.relative_to(_MODELS_DIR.resolve())
except ValueError:
raise HTTPException(
status_code=400,
detail="model_path must point inside the project's models directory",
)
if not p.exists():
raise HTTPException(status_code=400, detail=f"Model not found: {p}")
return p
@router.post("/process", response_model=ProcessResponse)
async def process_video(req: ProcessRequest):
video_uuid = _uuid.uuid4().hex
_VIDEO_DIR.mkdir(parents=True, exist_ok=True)
_TEXT_DIR.mkdir(parents=True, exist_ok=True)
loop = asyncio.get_running_loop()
logger.info("[%s] downloading %s", video_uuid, req.url)
try:
info = await loop.run_in_executor(
None,
download_video_with_subs,
req.url,
video_uuid,
_VIDEO_DIR,
settings.subtitle_langs_list,
)
except Exception as e:
raise HTTPException(status_code=500, detail=f"Download failed: {e}")
title: str = info["title"]
video_path: Path = info["video_path"]
sub_path: Path | None = info["subtitle_path"]
if not video_path.exists():
raise HTTPException(status_code=500, detail="Video file missing after download")
full_text = ""
method = ""
if sub_path is not None and sub_path.exists():
try:
content = sub_path.read_text(encoding="utf-8", errors="ignore")
full_text = vtt_to_text(content)
if full_text.strip():
method = f"subtitles:{info.get('subtitle_kind') or 'auto'}:{info.get('subtitle_lang') or 'unknown'}"
logger.info("[%s] using subtitles (%s)", video_uuid, method)
except Exception as e:
logger.warning("[%s] subtitle parse failed: %s", video_uuid, e)
full_text = ""
if not full_text.strip():
logger.info("[%s] no usable subtitles, falling back to Vosk", video_uuid)
model_path = _resolve_model_path(req.model_path)
try:
full_text = await loop.run_in_executor(
None, transcribe_via_ffmpeg, video_path, model_path
)
method = f"vosk:{model_path.name}"
except Exception as e:
raise HTTPException(status_code=500, detail=f"Transcription failed: {e}")
if not full_text.strip():
raise HTTPException(
status_code=500, detail="No subtitles and transcription returned empty text"
)
text_full_path = _TEXT_DIR / f"{video_uuid}.txt"
text_summary_path = _TEXT_DIR / f"{video_uuid}_summary.txt"
text_full_path.write_text(full_text, encoding="utf-8")
if req.llm is not None:
llm_cfg = LLMConfig(
base_url=req.llm.base_url,
api_key=req.llm.api_key,
model=req.llm.model,
timeout=req.llm.timeout,
max_tokens=req.llm.max_tokens,
temperature=req.llm.temperature,
extra_headers=req.llm.extra_headers,
)
sys_p = req.llm.system_prompt
user_t = req.llm.user_template
else:
llm_cfg = llm_from_settings()
sys_p = None
user_t = None
try:
summary = await loop.run_in_executor(
None,
lambda: summarize(
full_text,
llm_cfg=llm_cfg,
system_prompt=sys_p,
user_template=user_t,
max_sentences=req.summary_max_sentences,
),
)
except Exception as e:
raise HTTPException(status_code=500, detail=f"Summarization failed: {e}")
text_summary_path.write_text(summary, encoding="utf-8")
if req.keep_video:
rel_video = str(video_path.relative_to(_PROJECT_ROOT))
else:
# пользователь просил не хранить видео: текст уже извлечён, файлы не нужны
video_path.unlink(missing_ok=True)
for p in _VIDEO_DIR.glob(f"{video_uuid}*.vtt"):
p.unlink(missing_ok=True)
rel_video = ""
rel_full = str(text_full_path.relative_to(_PROJECT_ROOT))
rel_summary = str(text_summary_path.relative_to(_PROJECT_ROOT))
try:
await create_video(
uuid=video_uuid,
source_url=req.url,
title=title,
video_path=rel_video,
text_full_path=rel_full,
text_summary_path=rel_summary,
transcription_method=method,
section_id=req.section_id,
)
except Exception as e:
raise HTTPException(status_code=500, detail=f"DB insert failed: {e}")
logger.info("[%s] done (method=%s)", video_uuid, method)
return ProcessResponse(
uuid=video_uuid,
title=title,
source_url=req.url,
video_path=rel_video,
text_full_path=rel_full,
text_summary_path=rel_summary,
transcription_method=method,
summary_preview=summary[:300],
)