Files
LLM-infa/api/lib/youtube.py
jze9 bf854f2d6e Разделы, фильтры и просмотр текстов в UI; опция «не хранить видео»
- таблица sections (дерево через parent_id), у видео section_id
- API: CRUD /sections, фильтры /videos (поиск, раздел с потомками, метод),
  GET /videos/{id}/text, PATCH раздела, DELETE вместе с файлами
- pipeline: keep_video=false удаляет mp4 после расшифровки, section_id
- UI: панель разделов с вложенностью и счётчиками, поиск, фильтр по методу,
  диалоги полного текста/выжимки, привязка к разделу, удаление,
  переключатель «Сохранять видео на диске»
- compose: bind-mount ./web для правок без пересборки

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-14 12:07:55 +05:00

183 lines
5.6 KiB
Python

"""yt-dlp wrapper: download a YouTube video together with available subtitles."""
from __future__ import annotations
import logging
import re
from pathlib import Path
from typing import Optional
import yt_dlp
logger = logging.getLogger("youtube")
_VIDEO_EXTS = {".mp4", ".mkv", ".webm", ".m4a", ".mp3", ".mov"}
def download_video_with_subs(
url: str,
video_id: str,
video_dir: Path,
langs: Optional[list[str]] = None,
) -> dict:
"""Download a single video, also requesting subtitles for the given languages.
Returns:
{
"title": str,
"video_path": Path,
"subtitle_path": Path | None,
"subtitle_lang": str | None,
"subtitle_kind": "manual" | "auto" | None,
}
"""
if langs is None:
langs = ["ru", "en"]
video_dir.mkdir(parents=True, exist_ok=True)
outtmpl = str(video_dir / f"{video_id}.%(ext)s")
base_opts = {
"outtmpl": outtmpl,
"format": "bv*+ba/b",
"merge_output_format": "mp4",
"quiet": True,
"no_warnings": True,
}
sub_opts = {
**base_opts,
"writesubtitles": True,
"writeautomaticsub": True,
"subtitleslangs": langs,
"subtitlesformat": "vtt",
}
# YouTube often 429s the subtitle endpoints; a failed subtitle download must not
# kill the pipeline — retry without subs and let the Vosk fallback transcribe.
try:
with yt_dlp.YoutubeDL(sub_opts) as ydl:
info = ydl.extract_info(url, download=True)
except yt_dlp.utils.DownloadError as e:
logger.warning("Download with subtitles failed (%s); retrying without subs", e)
with yt_dlp.YoutubeDL(base_opts) as ydl:
info = ydl.extract_info(url, download=True)
title = info.get("title") or video_id
# Locate the merged video file
video_path = video_dir / f"{video_id}.mp4"
if not video_path.exists():
for p in video_dir.glob(f"{video_id}.*"):
if p.suffix.lower() in _VIDEO_EXTS:
video_path = p
break
sub_path, sub_lang, sub_kind = _find_subtitle(video_dir, video_id, langs, info)
return {
"title": title,
"video_path": video_path,
"subtitle_path": sub_path,
"subtitle_lang": sub_lang,
"subtitle_kind": sub_kind,
}
def _find_subtitle(
video_dir: Path,
video_id: str,
langs: list[str],
info: dict,
) -> tuple[Optional[Path], Optional[str], Optional[str]]:
"""Pick the best subtitle file among downloaded ones, preferring manual subs."""
requested = info.get("requested_subtitles") or {}
manual_keys = set(info.get("subtitles", {}).keys())
# Build a list of (lang, kind, path) candidates from yt-dlp's reported requested_subtitles
candidates: list[tuple[str, str, Path]] = []
for lang_code, sub_info in requested.items():
filepath = sub_info.get("filepath")
if not filepath:
continue
p = Path(filepath)
if not p.exists():
continue
kind = "manual" if lang_code in manual_keys else "auto"
candidates.append((lang_code, kind, p))
# Fallback: glob the directory
if not candidates:
for p in video_dir.glob(f"{video_id}*.vtt"):
m = re.match(rf"^{re.escape(video_id)}\.([A-Za-z0-9_\-]+)\.vtt$", p.name)
lang_code = m.group(1) if m else "unknown"
candidates.append((lang_code, "auto", p))
if not candidates:
return None, None, None
def score(c: tuple[str, str, Path]) -> tuple[int, int]:
lang, kind, _ = c
try:
lang_rank = next(
i for i, l in enumerate(langs) if lang == l or lang.startswith(l + "-")
)
except StopIteration:
lang_rank = len(langs)
kind_rank = 0 if kind == "manual" else 1
return (lang_rank, kind_rank)
candidates.sort(key=score)
lang, kind, path = candidates[0]
return path, lang, kind
def vtt_to_text(vtt_content: str) -> str:
"""Convert a VTT subtitle file to a deduplicated plain-text string.
YouTube auto-captions ship as rolling cues — each new cue extends the previous.
Strategy: per cue block, keep only the last text line, then drop consecutive
duplicates and identical sentence fragments.
"""
blocks = re.split(r"\r?\n\s*\r?\n", vtt_content)
out: list[str] = []
last = ""
for block in blocks:
block = block.strip()
if not block:
continue
text_lines: list[str] = []
for raw in block.splitlines():
line = raw.strip()
if not line:
continue
if line.startswith("WEBVTT"):
continue
if line.startswith(("NOTE", "STYLE", "Kind:", "Language:", "Region:")):
continue
if "-->" in line:
continue
if re.match(r"^\d+$", line):
continue
line = re.sub(r"<[^>]+>", "", line)
line = re.sub(r"\s+", " ", line).strip()
if line:
text_lines.append(line)
if not text_lines:
continue
candidate = text_lines[-1]
if candidate == last:
continue
# Avoid the rolling-caption case where a cue strictly extends the previous one
if last and candidate.startswith(last):
out[-1] = candidate
last = candidate
continue
if last and last.startswith(candidate):
continue
out.append(candidate)
last = candidate
return " ".join(out)