Рабочий пайплайн: YouTube -> транскрипция -> выжимка -> Postgres

- новый api/: /pipeline/process (yt-dlp, субтитры или Vosk, LLM/экстрактивная
  суммаризация), /videos, /llm; старый код перенесён в api_legacy/
- web/: Flet UI (flet 0.28.3 + flet-web, порт 8550)
- Dockerfile.api: ffmpeg слоем из mwader/static-ffmpeg, pip с кэш-маунтом
- requirements/pyproject: убраны moviepy, imageio-ffmpeg, битый asyncio
- README с инструкцией запуска и загрузкой Vosk-модели

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
jze9
2026-07-14 11:12:46 +05:00
parent 564c9c2d2f
commit 2697e01714
51 changed files with 2313 additions and 284 deletions

0
api/lib/__init__.py Normal file
View File

View File

@@ -1,134 +0,0 @@
from pathlib import Path
import os
import sys
import threading
import logging
import wave
import json
import subprocess
import shutil
try:
from vosk import Model, KaldiRecognizer
except Exception:
Model = None
KaldiRecognizer = None
class FFmpegConverter:
"""Utility to convert audio files using ffmpeg."""
@staticmethod
def convert_mp3_to_wav(src: Path, dest: Path) -> None:
cmd = [
"ffmpeg",
"-y",
"-i",
str(src),
"-ac",
"1",
"-ar",
"16000",
"-vn",
"-f",
"wav",
str(dest),
]
subprocess.run(cmd, check=True)
class VoskModelManager:
"""Loads a Vosk model and provides transcription methods.
Loading captures model stderr into Python logs to show progress.
"""
def __init__(self, model_path: Path):
self.logger = logging.getLogger("vosk.model_manager")
if Model is None or KaldiRecognizer is None:
raise RuntimeError("vosk is not installed in runtime")
self.model = self._load_model_with_progress(model_path)
def _load_model_with_progress(self, path: Path):
logger = logging.getLogger("vosk.loader")
r_fd, w_fd = os.pipe()
saved_stderr_fd = os.dup(sys.stderr.fileno())
os.dup2(w_fd, sys.stderr.fileno())
def _reader(fd):
with os.fdopen(fd, "rb") as fh:
for raw in iter(fh.readline, b""):
try:
logger.info(raw.decode(errors="ignore").rstrip())
except Exception:
pass
reader_thread = threading.Thread(target=_reader, args=(r_fd,), daemon=True)
reader_thread.start()
try:
model = Model(str(path))
finally:
try:
os.dup2(saved_stderr_fd, sys.stderr.fileno())
except Exception:
pass
try:
os.close(saved_stderr_fd)
except Exception:
pass
reader_thread.join(timeout=2)
return model
def transcribe_wav(self, wav_path: Path) -> str:
if not wav_path.exists():
raise FileNotFoundError(wav_path)
try:
with wave.open(str(wav_path), "rb") as wf:
if wf.getnchannels() != 1:
raise ValueError("Audio must be mono (1 channel)")
recognizer = KaldiRecognizer(self.model, wf.getframerate())
try:
recognizer.SetWords(True)
except Exception:
# older vosk bindings may not expose SetWords; ignore if absent
pass
while True:
data = wf.readframes(4000)
if not data:
break
recognizer.AcceptWaveform(data)
result = json.loads(recognizer.FinalResult())
return result.get("text", "")
except wave.Error as e:
raise ValueError(f"Invalid WAV file: {e}")
def transcribe_wav_with_words(self, wav_path: Path):
"""Transcribe WAV and return dict with 'text' and 'words' (word-level timestamps).
Returns: { 'text': str, 'words': [{'word': str, 'start': float, 'end': float}, ...] }
"""
if not wav_path.exists():
raise FileNotFoundError(wav_path)
try:
with wave.open(str(wav_path), "rb") as wf:
if wf.getnchannels() != 1:
raise ValueError("Audio must be mono (1 channel)")
recognizer = KaldiRecognizer(self.model, wf.getframerate())
try:
recognizer.SetWords(True)
except Exception:
pass
while True:
data = wf.readframes(4000)
if not data:
break
recognizer.AcceptWaveform(data)
res = json.loads(recognizer.FinalResult())
words = res.get("result", [])
# words are dictionaries with word, start, end
text = res.get("text", "")
return {"text": text, "words": words}
except wave.Error as e:
raise ValueError(f"Invalid WAV file: {e}")

View File

@@ -1,69 +0,0 @@
from pathlib import Path
from typing import List, Dict
def diarize_with_resemblyzer(wav_path: Path, window_s: float = 1.5, hop_s: float = 0.75, distance_threshold: float = 0.6) -> List[Dict]:
"""Lightweight embedding-based diarization using resemblyzer + sklearn.
Returns list of segments: {'start': float, 'end': float, 'speaker': 'spk_N'}
"""
try:
import numpy as np
import librosa
from resemblyzer import VoiceEncoder
from sklearn.cluster import AgglomerativeClustering
except Exception as e:
raise RuntimeError(f"Missing dependency for embedding diarization: {e}")
sr = 16000
wav, sr_loaded = librosa.load(str(wav_path), sr=sr)
n = wav.shape[0]
win = int(window_s * sr)
hop = int(hop_s * sr)
if n <= win:
# short file: embed whole
encoder = VoiceEncoder()
emb = encoder.embed_utterance(wav)
labels = [0]
centers = [ (0.0 + n / sr) / 2.0 ]
windows = [(0.0, n / sr)]
else:
encoder = VoiceEncoder()
embeddings = []
centers = []
windows = []
for start in range(0, n - win + 1, hop):
chunk = wav[start:start+win]
try:
e = encoder.embed_utterance(chunk)
except Exception:
# fallback: mean pooling
e = np.mean(chunk)
embeddings.append(e)
t0 = start / sr
t1 = (start + win) / sr
centers.append((t0 + t1) / 2.0)
windows.append((t0, t1))
X = np.vstack(embeddings)
# Agglomerative clustering with distance threshold
model = AgglomerativeClustering(n_clusters=None, distance_threshold=distance_threshold, affinity='euclidean', linkage='average')
labels = model.fit_predict(X)
# merge consecutive windows with same label into segments
segs = []
if len(labels) == 0:
return segs
cur_label = labels[0]
cur_start = windows[0][0]
cur_end = windows[0][1]
for i, lab in enumerate(labels[1:], start=1):
if lab == cur_label:
cur_end = windows[i][1]
else:
segs.append({"start": cur_start, "end": cur_end, "speaker": f"spk_{int(cur_label)+1}"})
cur_label = lab
cur_start = windows[i][0]
cur_end = windows[i][1]
segs.append({"start": cur_start, "end": cur_end, "speaker": f"spk_{int(cur_label)+1}"})
return segs

164
api/lib/llm.py Normal file
View File

@@ -0,0 +1,164 @@
"""Generic OpenAI-compatible chat-completions client.
The LLM is *not* deployed in this container — point this client at any external
endpoint that speaks the OpenAI Chat Completions wire format:
- https://api.openai.com/v1
- https://openrouter.ai/api/v1
- https://api.anthropic.com (via OpenAI-compat proxies)
- http://host.docker.internal:11434/v1 (local Ollama)
- http://vllm-host:8000/v1 (vLLM / llama.cpp server / LM Studio)
Configure once via environment (LLM_BASE_URL, LLM_API_KEY, LLM_MODEL),
or override per-request through /pipeline/process by passing an `llm` block.
"""
from __future__ import annotations
import json
import logging
import urllib.error
import urllib.request
from typing import Optional
from pydantic import BaseModel, Field
from api.config import settings
logger = logging.getLogger("llm")
class LLMConfig(BaseModel):
"""Wire-level config for an OpenAI-compatible chat-completions endpoint."""
base_url: str = Field(..., description="Base URL ending in /v1 (or equivalent)")
api_key: str = ""
# `model` is optional so the same config can be used for /v1/models discovery
# before the user has picked one.
model: str = ""
timeout: int = 120
max_tokens: int = 1000
temperature: float = 0.3
extra_headers: dict[str, str] = Field(default_factory=dict)
class LLMError(RuntimeError):
pass
def from_settings() -> Optional[LLMConfig]:
"""Build LLMConfig from environment variables, or return None if not configured."""
if not settings.LLM_BASE_URL or not settings.LLM_MODEL:
return None
return LLMConfig(
base_url=settings.LLM_BASE_URL,
api_key=settings.LLM_API_KEY,
model=settings.LLM_MODEL,
timeout=settings.LLM_TIMEOUT,
max_tokens=settings.LLM_MAX_TOKENS,
temperature=settings.LLM_TEMPERATURE,
)
def chat_complete(cfg: LLMConfig, messages: list[dict]) -> str:
"""Synchronous Chat Completions call. Returns assistant message content."""
url = cfg.base_url.rstrip("/") + "/chat/completions"
payload: dict = {
"model": cfg.model,
"messages": messages,
"max_tokens": cfg.max_tokens,
"temperature": cfg.temperature,
}
headers = {"Content-Type": "application/json", "Accept": "application/json"}
if cfg.api_key:
headers["Authorization"] = f"Bearer {cfg.api_key}"
headers.update(cfg.extra_headers or {})
body = json.dumps(payload).encode("utf-8")
req = urllib.request.Request(url, data=body, headers=headers, method="POST")
try:
with urllib.request.urlopen(req, timeout=cfg.timeout) as resp:
raw = resp.read()
except urllib.error.HTTPError as e:
try:
err_body = e.read().decode("utf-8", errors="ignore")
except Exception:
err_body = ""
raise LLMError(f"LLM HTTP {e.code} {e.reason}: {err_body[:500]}")
except Exception as e:
raise LLMError(f"LLM request failed: {e}")
try:
data = json.loads(raw.decode("utf-8"))
choices = data.get("choices") or []
if not choices:
raise LLMError(f"LLM returned no choices: {data}")
msg = choices[0].get("message") or {}
content = msg.get("content")
if content is None:
raise LLMError(f"LLM choice has no content: {choices[0]}")
return content.strip()
except LLMError:
raise
except Exception as e:
raise LLMError(f"Failed to parse LLM response: {e}")
def ping(cfg: LLMConfig) -> str:
"""Tiny round-trip to verify connectivity. Returns the assistant reply text."""
return chat_complete(
cfg,
[
{"role": "system", "content": "You are a helpful assistant."},
{"role": "user", "content": "Reply with the single word OK."},
],
)
def list_models(cfg: LLMConfig) -> list[dict]:
"""GET {base_url}/models. Returns a normalized list of {id, name, owned_by?}.
Compatible with OpenAI, OpenRouter, Ollama OpenAI-compat, vLLM, LM Studio.
"""
url = cfg.base_url.rstrip("/") + "/models"
headers = {"Accept": "application/json"}
if cfg.api_key:
headers["Authorization"] = f"Bearer {cfg.api_key}"
headers.update(cfg.extra_headers or {})
req = urllib.request.Request(url, headers=headers, method="GET")
try:
with urllib.request.urlopen(req, timeout=cfg.timeout) as resp:
raw = resp.read()
except urllib.error.HTTPError as e:
try:
err_body = e.read().decode("utf-8", errors="ignore")
except Exception:
err_body = ""
raise LLMError(f"LLM HTTP {e.code} {e.reason}: {err_body[:500]}")
except Exception as e:
raise LLMError(f"List models failed: {e}")
try:
data = json.loads(raw.decode("utf-8"))
except Exception as e:
raise LLMError(f"Failed to parse models response: {e}")
items = data.get("data") or data.get("models") or data.get("results") or []
out: list[dict] = []
for item in items:
if isinstance(item, str):
out.append({"id": item, "name": item})
continue
if not isinstance(item, dict):
continue
mid = item.get("id") or item.get("name") or item.get("model")
if not mid:
continue
entry = {"id": mid, "name": item.get("name") or mid}
owned = item.get("owned_by") or item.get("organization")
if owned:
entry["owned_by"] = owned
out.append(entry)
out.sort(key=lambda x: x["id"].lower())
return out

221
api/lib/summarize.py Normal file
View File

@@ -0,0 +1,221 @@
"""Summarization with two backends:
* `summarize_llm` — uses any external OpenAI-compatible API (configured via
`LLMConfig`). Long transcripts are map-reduced: per-chunk summaries are
concatenated and re-summarized.
* `summarize_extractive` — pure-Python TF-IDF-ish fallback (RU + EN aware).
`summarize(text, llm_cfg=...)` picks the LLM path when a config is supplied,
otherwise falls back to extractive. The pipeline calls this entry point.
"""
from __future__ import annotations
import logging
import re
from collections import Counter
from typing import Optional
from api.lib.llm import LLMConfig, LLMError, chat_complete
logger = logging.getLogger("summarize")
_RU_STOPWORDS = {
"и", "в", "во", "не", "на", "я", "что", "тот", "быть", "с", "со", "а", "весь",
"как", "это", "но", "он", "она", "оно", "мы", "вы", "они", "к", "у", "из", "за",
"по", "до", "от", "для", "же", "или", "бы", "если", "так", "там", "тут", "когда",
"где", "кто", "какой", "этот", "эта", "эти", "мой", "твой", "наш", "ваш",
"свой", "его", "её", "их", "был", "была", "было", "были", "есть", "нет", "да",
"ну", "вот", "ещё", "еще", "уже", "только", "очень", "может", "можно", "надо",
"будет", "будут", "всё", "все", "этого", "этом", "этой", "тоже", "также", "чтобы",
"кого", "чего", "чем", "тем", "том", "нем", "ней", "себя", "себе", "мне", "тебе",
"нам", "вам", "нас", "вас", "про", "без", "над", "под", "через", "при", "об",
"о", "обо", "ли", "вон", "сюда", "туда", "оттуда", "потом", "теперь", "тогда",
"сейчас", "иногда", "всегда", "никогда", "просто", "именно", "ведь", "значит",
"хотя", "что-то", "кто-то",
}
_EN_STOPWORDS = {
"the", "a", "an", "and", "or", "but", "is", "are", "was", "were", "be", "been",
"being", "have", "has", "had", "do", "does", "did", "will", "would", "should",
"could", "may", "might", "must", "shall", "can", "i", "you", "he", "she", "it",
"we", "they", "them", "my", "your", "his", "her", "its", "our", "their", "this",
"that", "these", "those", "in", "on", "at", "by", "for", "with", "about", "as",
"of", "to", "from", "into", "so", "not", "no", "yes", "if", "then", "than",
"when", "where", "why", "how", "what", "which", "who", "whom", "whose", "me",
"us", "him", "just", "only", "also", "too", "very", "there", "here", "out",
"up", "down", "over", "under", "between", "through", "again", "more", "most",
"some", "any", "all", "each", "every", "such", "while", "because", "until",
"during", "above", "below", "now", "ever", "never",
}
_STOPWORDS = _RU_STOPWORDS | _EN_STOPWORDS
_WORD_RE = re.compile(r"[\w']+", re.UNICODE)
_SENT_SPLIT_RE = re.compile(r"(?<=[.!?…])\s+(?=[А-ЯA-Z\"«„])")
DEFAULT_SYSTEM_PROMPT = (
"Ты ассистент, который делает краткую и точную выжимку текста. "
"Сохраняй ключевые факты, имена, цифры, причинно-следственные связи. "
"Если оригинал на русском — пиши по-русски, иначе — на языке оригинала. "
"Не добавляй информацию, которой нет в исходном тексте."
)
DEFAULT_USER_TEMPLATE = (
"Сделай связную краткую выжимку следующего текста. "
"Выдели основные тезисы и логику изложения. Объём — 512 предложений.\n\n"
"---\n{text}\n---"
)
DEFAULT_REDUCE_TEMPLATE = (
"Ниже — последовательность кратких выжимок частей одного длинного текста. "
"Объедини их в единую связную выжимку, сохранив главные факты и логику.\n\n"
"---\n{text}\n---"
)
def _split_sentences(text: str) -> list[str]:
text = re.sub(r"\s+", " ", text).strip()
if not text:
return []
parts = _SENT_SPLIT_RE.split(text)
if len(parts) <= 1:
parts = re.split(r"(?<=[.!?…])\s+", text)
return [p.strip() for p in parts if p.strip()]
def _chunk_by_chars(text: str, max_chars: int) -> list[str]:
if len(text) <= max_chars:
return [text]
chunks: list[str] = []
cur: list[str] = []
cur_len = 0
for s in _split_sentences(text):
if cur_len + len(s) + 1 > max_chars and cur:
chunks.append(" ".join(cur))
cur, cur_len = [], 0
cur.append(s)
cur_len += len(s) + 1
if cur:
chunks.append(" ".join(cur))
return chunks
def summarize_extractive(
text: str,
*,
max_sentences: int = 15,
min_sentences: int = 3,
ratio: float = 0.15,
) -> str:
"""Frequency-based extractive summary. Picks top sentences by mean word weight."""
text = (text or "").strip()
if not text:
return ""
sentences = _split_sentences(text)
if not sentences:
return ""
target = max(min_sentences, min(max_sentences, int(len(sentences) * ratio)))
target = max(1, min(target, len(sentences)))
if len(sentences) <= target:
return " ".join(sentences)
words = _WORD_RE.findall(text.lower())
freq = Counter(w for w in words if w not in _STOPWORDS and len(w) > 2)
if not freq:
return " ".join(sentences[:target])
max_f = max(freq.values())
norm = {w: c / max_f for w, c in freq.items()}
scored: list[tuple[int, float]] = []
for i, s in enumerate(sentences):
sw = _WORD_RE.findall(s.lower())
if not sw:
continue
score = sum(norm.get(w, 0.0) for w in sw) / len(sw)
scored.append((i, score))
if not scored:
return " ".join(sentences[:target])
top = sorted(scored, key=lambda x: -x[1])[:target]
top.sort(key=lambda x: x[0])
return " ".join(sentences[i] for i, _ in top)
def summarize_llm(
text: str,
cfg: LLMConfig,
*,
system_prompt: Optional[str] = None,
user_template: Optional[str] = None,
reduce_template: Optional[str] = None,
chunk_chars: int = 10000,
) -> str:
"""Summarize via an external OpenAI-compatible API. Map-reduce for long texts."""
text = (text or "").strip()
if not text:
return ""
sys_p = system_prompt or DEFAULT_SYSTEM_PROMPT
user_t = user_template or DEFAULT_USER_TEMPLATE
reduce_t = reduce_template or DEFAULT_REDUCE_TEMPLATE
def _one(chunk: str, template: str) -> str:
return chat_complete(
cfg,
[
{"role": "system", "content": sys_p},
{"role": "user", "content": template.format(text=chunk)},
],
)
chunks = _chunk_by_chars(text, chunk_chars)
logger.info("LLM summarize: %d chunk(s), total chars=%d", len(chunks), len(text))
if len(chunks) == 1:
return _one(chunks[0], user_t)
partials: list[str] = []
for i, ch in enumerate(chunks, start=1):
logger.info("LLM summarize chunk %d/%d", i, len(chunks))
partials.append(_one(ch, user_t))
combined = "\n\n".join(partials)
if len(combined) > chunk_chars:
# If even concatenated partials are too long, recurse: summarize partials in batches.
return summarize_llm(
combined,
cfg,
system_prompt=sys_p,
user_template=reduce_t,
reduce_template=reduce_t,
chunk_chars=chunk_chars,
)
logger.info("LLM reduce step over %d partials", len(partials))
return _one(combined, reduce_t)
def summarize(
text: str,
*,
llm_cfg: Optional[LLMConfig] = None,
system_prompt: Optional[str] = None,
user_template: Optional[str] = None,
max_sentences: int = 15,
) -> str:
"""Primary entry point. Uses LLM if configured, else extractive fallback."""
if llm_cfg is not None:
try:
return summarize_llm(
text,
llm_cfg,
system_prompt=system_prompt,
user_template=user_template,
)
except LLMError as e:
logger.warning("LLM summarization failed, falling back to extractive: %s", e)
return summarize_extractive(text, max_sentences=max_sentences)

119
api/lib/transcribe.py Normal file
View File

@@ -0,0 +1,119 @@
"""Vosk-based fallback transcription: stream PCM from ffmpeg into a Vosk recognizer."""
from __future__ import annotations
import json
import logging
import os
import shutil
import subprocess
import sys
import threading
from pathlib import Path
try:
from vosk import KaldiRecognizer, Model
except Exception:
Model = None
KaldiRecognizer = None
logger = logging.getLogger("transcribe")
class VoskModelManager:
"""Loads a Vosk model. Captures the model's stderr to Python logs while loading."""
def __init__(self, model_path: Path):
if Model is None or KaldiRecognizer is None:
raise RuntimeError("vosk is not installed in runtime")
self.model = self._load_with_stderr_capture(model_path)
@staticmethod
def _load_with_stderr_capture(path: Path):
log = logging.getLogger("vosk.loader")
r_fd, w_fd = os.pipe()
saved_stderr = os.dup(sys.stderr.fileno())
os.dup2(w_fd, sys.stderr.fileno())
os.close(w_fd)
def reader(fd: int) -> None:
with os.fdopen(fd, "rb") as fh:
for raw in iter(fh.readline, b""):
log.info(raw.decode(errors="ignore").rstrip())
t = threading.Thread(target=reader, args=(r_fd,), daemon=True)
t.start()
try:
return Model(str(path))
finally:
try:
os.dup2(saved_stderr, sys.stderr.fileno())
finally:
os.close(saved_stderr)
t.join(timeout=2)
def transcribe_via_ffmpeg(media_path: Path, model_path: Path, sample_rate: int = 16000) -> str:
"""Decode the source media to PCM s16le mono via ffmpeg and feed it to Vosk in chunks."""
if Model is None or KaldiRecognizer is None:
raise RuntimeError("vosk is not installed in runtime")
if shutil.which("ffmpeg") is None:
raise RuntimeError("ffmpeg is not available in PATH")
if not media_path.exists():
raise FileNotFoundError(media_path)
mgr = VoskModelManager(model_path)
recognizer = KaldiRecognizer(mgr.model, sample_rate)
try:
recognizer.SetWords(True)
except Exception:
pass
cmd = [
"ffmpeg",
"-hide_banner",
"-loglevel",
"error",
"-i",
str(media_path),
"-f",
"s16le",
"-acodec",
"pcm_s16le",
"-ac",
"1",
"-ar",
str(sample_rate),
"-vn",
"-",
]
logger.info("Spawning ffmpeg streaming decode for %s", media_path.name)
proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
if proc.stdout is None:
proc.kill()
raise RuntimeError("ffmpeg did not provide stdout")
try:
bytes_read = 0
last_logged_mb = 0
while True:
chunk = proc.stdout.read(4000)
if not chunk:
break
recognizer.AcceptWaveform(chunk)
bytes_read += len(chunk)
mb = bytes_read // (1024 * 1024)
if mb >= last_logged_mb + 5:
last_logged_mb = mb
logger.info("Transcribed %d MB of audio so far", mb)
result = json.loads(recognizer.FinalResult())
text = result.get("text", "")
logger.info("Transcription finished, length=%d", len(text))
return text
finally:
try:
proc.kill()
except Exception:
pass

View File

@@ -1,236 +0,0 @@
from pathlib import Path
import wave
import json
import math
from typing import List, Dict
def _rms(frames: bytes, sample_width: int) -> float:
# compute RMS of raw frames (little-endian signed samples)
if sample_width == 2:
import struct
count = len(frames) // 2
if count == 0:
return 0.0
fmt = "<%dh" % count
vals = struct.unpack(fmt, frames)
s = 0
for v in vals:
s += v * v
return math.sqrt(s / count)
else:
return 0.0
def diarize_by_silence(wav_path: Path, win_ms: int = 30, silence_thresh: float = 500.0, min_silence_ms: int = 400) -> List[Dict]:
"""Naive diarization by splitting on long silence.
Returns list of segments: {'start': float, 'end': float, 'speaker': str}
Speakers are assigned alternately: Speaker 1, Speaker 2, ...
"""
segs = []
with wave.open(str(wav_path), "rb") as wf:
sr = wf.getframerate()
sw = wf.getsampwidth()
n_channels = wf.getnchannels()
# we expect mono
frames_per_win = int(sr * win_ms / 1000)
min_silence_frames = int(sr * min_silence_ms / 1000)
total_frames = wf.getnframes()
pos = 0
silent_acc = 0
current_start = 0
speaker_idx = 1
while pos < total_frames:
to_read = min(frames_per_win, total_frames - pos)
raw = wf.readframes(to_read)
pos += to_read
level = _rms(raw, sw)
if level < silence_thresh:
silent_acc += to_read
else:
# if there was long silence before speech, start a new segment
if silent_acc >= min_silence_frames and pos / sr > current_start:
end_time = (pos - silent_acc) / sr
segs.append({"start": current_start, "end": end_time, "speaker": f"Speaker {speaker_idx}"})
speaker_idx = (speaker_idx % 2) + 1
current_start = end_time
silent_acc = 0
# finish last segment
if current_start < total_frames / sr:
segs.append({"start": current_start, "end": total_frames / sr, "speaker": f"Speaker {speaker_idx}"})
return segs
def assign_words_to_speakers(words: List[Dict], segments: List[Dict]) -> List[Dict]:
"""Assign word dicts (with start/end) to speaker segments. Returns list of speaker blocks with text."""
result = []
seg_idx = 0
current_block = {"speaker": segments[0]["speaker"] if segments else "Speaker 1", "start": None, "end": None, "words": []}
for w in words:
t = w.get("start", 0)
# move seg_idx until this word falls into segment
while seg_idx + 1 < len(segments) and t >= segments[seg_idx]["end"]:
# flush current
if current_block["words"]:
current_block["end"] = segments[seg_idx]["end"]
result.append(current_block)
seg_idx += 1
current_block = {"speaker": segments[seg_idx]["speaker"], "start": None, "end": None, "words": []}
# append word
if current_block["start"] is None:
current_block["start"] = w.get("start")
current_block["words"].append(w)
if current_block["words"]:
# set end
current_block["end"] = current_block["words"][-1].get("end")
result.append(current_block)
return result
def chapter_by_wordcount(words: List[Dict], speaker_blocks: List[Dict] = None, words_per_chapter: int = 100, min_words_on_speaker_change: int = 30) -> List[Dict]:
"""Chaptering that prefers boundaries at speaker changes.
- If `speaker_blocks` provided, will try to split a chapter when speaker changes
and the current chapter has at least `min_words_on_speaker_change` words.
- Also split when `words_per_chapter` is reached.
Returns list of chapters: {'start': float, 'end': float, 'words': [...]}.
"""
chapters = []
chapter = {"start": None, "end": None, "words": []}
count = 0
def speaker_for_time(t: float):
if not speaker_blocks or t is None:
return None
for b in speaker_blocks:
bstart = b.get("start")
bend = b.get("end")
if bstart is None or bend is None:
continue
try:
if bstart <= t <= bend:
return b.get("speaker")
except Exception:
continue
return None
prev_speaker = None
for w in words:
wstart = w.get("start")
if chapter["start"] is None:
chapter["start"] = wstart
chapter["words"].append(w)
chapter["end"] = w.get("end")
count += 1
# check speaker change boundary
cur_speaker = speaker_for_time(wstart)
if prev_speaker is None:
prev_speaker = cur_speaker
if cur_speaker is not None and prev_speaker is not None and cur_speaker != prev_speaker:
if count >= min_words_on_speaker_change:
# close chapter at previous word
chapters.append(chapter)
chapter = {"start": None, "end": None, "words": []}
count = 0
prev_speaker = cur_speaker
continue
else:
# don't split if too short to make a chapter
prev_speaker = cur_speaker
# check fixed-size boundary
if count >= words_per_chapter:
chapters.append(chapter)
chapter = {"start": None, "end": None, "words": []}
count = 0
prev_speaker = None
if chapter["words"]:
chapters.append(chapter)
return chapters
def save_transcript(text_dir: Path, base_name: str, speaker_blocks: List[Dict], chapters: List[Dict]):
text_dir.mkdir(parents=True, exist_ok=True)
out_json = text_dir / f"{base_name}.json"
out_txt = text_dir / f"{base_name}.txt"
# Automatic host labeling: speaker with most words -> 'Ведущий', others -> 'Участник N'
try:
# Prefer selecting host by total spoken duration (better for long monologues).
durations = []
for b in speaker_blocks:
dur = 0.0
for w in b.get("words", []):
s = w.get("start")
e = w.get("end")
if s is None or e is None:
continue
try:
dur += float(e) - float(s)
except Exception:
continue
durations.append(dur)
if any(durations):
host_idx = int(max(range(len(durations)), key=lambda i: durations[i]))
else:
# fallback to word counts if timestamps missing
counts = [sum(1 for w in b.get("words", [])) for b in speaker_blocks]
host_idx = int(max(range(len(counts)), key=lambda i: counts[i])) if counts else 0
new_blocks = []
participant_idx = 1
for i, b in enumerate(speaker_blocks):
b_copy = dict(b)
if i == host_idx:
b_copy["speaker"] = "Ведущий"
else:
b_copy["speaker"] = f"Участник {participant_idx}"
participant_idx += 1
new_blocks.append(b_copy)
speaker_blocks = new_blocks
except Exception:
# fall back to provided labels on error
pass
data = {"speakers": speaker_blocks, "chapters": chapters}
with open(out_json, "w", encoding="utf-8") as fh:
json.dump(data, fh, ensure_ascii=False, indent=2)
# plain text with speaker labels and chapter headings
with open(out_txt, "w", encoding="utf-8") as fh:
for i, ch in enumerate(chapters, start=1):
fh.write(f"=== Глава {i} ===\n")
# collect speaker text overlapping this chapter by word membership
# build a simple view: iterate speaker blocks and print their words that fall into chapter
for b in speaker_blocks:
# collect words in this chapter that belong to this speaker block
b_words = []
for w in ch["words"]:
# a word belongs to speaker block if its start is within block start/end (if available)
wstart = w.get("start")
if wstart is None:
continue
bstart = b.get("start")
bend = b.get("end")
if bstart is None or bend is None:
continue
if bstart <= wstart <= bend:
b_words.append(w["word"])
if b_words:
fh.write(f"{b['speaker']}: ")
fh.write(" ".join(b_words) + "\n")
fh.write("\n")
return out_json, out_txt

172
api/lib/youtube.py Normal file
View File

@@ -0,0 +1,172 @@
"""yt-dlp wrapper: download a YouTube video together with available subtitles."""
from __future__ import annotations
import logging
import re
from pathlib import Path
from typing import Optional
import yt_dlp
logger = logging.getLogger("youtube")
_VIDEO_EXTS = {".mp4", ".mkv", ".webm", ".m4a", ".mp3", ".mov"}
def download_video_with_subs(
url: str,
video_id: str,
video_dir: Path,
langs: Optional[list[str]] = None,
) -> dict:
"""Download a single video, also requesting subtitles for the given languages.
Returns:
{
"title": str,
"video_path": Path,
"subtitle_path": Path | None,
"subtitle_lang": str | None,
"subtitle_kind": "manual" | "auto" | None,
}
"""
if langs is None:
langs = ["ru", "en"]
video_dir.mkdir(parents=True, exist_ok=True)
outtmpl = str(video_dir / f"{video_id}.%(ext)s")
ydl_opts = {
"outtmpl": outtmpl,
"format": "bv*+ba/b",
"merge_output_format": "mp4",
"writesubtitles": True,
"writeautomaticsub": True,
"subtitleslangs": langs,
"subtitlesformat": "vtt",
"quiet": True,
"no_warnings": True,
}
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
info = ydl.extract_info(url, download=True)
title = info.get("title") or video_id
# Locate the merged video file
video_path = video_dir / f"{video_id}.mp4"
if not video_path.exists():
for p in video_dir.glob(f"{video_id}.*"):
if p.suffix.lower() in _VIDEO_EXTS:
video_path = p
break
sub_path, sub_lang, sub_kind = _find_subtitle(video_dir, video_id, langs, info)
return {
"title": title,
"video_path": video_path,
"subtitle_path": sub_path,
"subtitle_lang": sub_lang,
"subtitle_kind": sub_kind,
}
def _find_subtitle(
video_dir: Path,
video_id: str,
langs: list[str],
info: dict,
) -> tuple[Optional[Path], Optional[str], Optional[str]]:
"""Pick the best subtitle file among downloaded ones, preferring manual subs."""
requested = info.get("requested_subtitles") or {}
manual_keys = set(info.get("subtitles", {}).keys())
# Build a list of (lang, kind, path) candidates from yt-dlp's reported requested_subtitles
candidates: list[tuple[str, str, Path]] = []
for lang_code, sub_info in requested.items():
filepath = sub_info.get("filepath")
if not filepath:
continue
p = Path(filepath)
if not p.exists():
continue
kind = "manual" if lang_code in manual_keys else "auto"
candidates.append((lang_code, kind, p))
# Fallback: glob the directory
if not candidates:
for p in video_dir.glob(f"{video_id}*.vtt"):
m = re.match(rf"^{re.escape(video_id)}\.([A-Za-z0-9_\-]+)\.vtt$", p.name)
lang_code = m.group(1) if m else "unknown"
candidates.append((lang_code, "auto", p))
if not candidates:
return None, None, None
def score(c: tuple[str, str, Path]) -> tuple[int, int]:
lang, kind, _ = c
try:
lang_rank = next(
i for i, l in enumerate(langs) if lang == l or lang.startswith(l + "-")
)
except StopIteration:
lang_rank = len(langs)
kind_rank = 0 if kind == "manual" else 1
return (lang_rank, kind_rank)
candidates.sort(key=score)
lang, kind, path = candidates[0]
return path, lang, kind
def vtt_to_text(vtt_content: str) -> str:
"""Convert a VTT subtitle file to a deduplicated plain-text string.
YouTube auto-captions ship as rolling cues — each new cue extends the previous.
Strategy: per cue block, keep only the last text line, then drop consecutive
duplicates and identical sentence fragments.
"""
blocks = re.split(r"\r?\n\s*\r?\n", vtt_content)
out: list[str] = []
last = ""
for block in blocks:
block = block.strip()
if not block:
continue
text_lines: list[str] = []
for raw in block.splitlines():
line = raw.strip()
if not line:
continue
if line.startswith("WEBVTT"):
continue
if line.startswith(("NOTE", "STYLE", "Kind:", "Language:", "Region:")):
continue
if "-->" in line:
continue
if re.match(r"^\d+$", line):
continue
line = re.sub(r"<[^>]+>", "", line)
line = re.sub(r"\s+", " ", line).strip()
if line:
text_lines.append(line)
if not text_lines:
continue
candidate = text_lines[-1]
if candidate == last:
continue
# Avoid the rolling-caption case where a cue strictly extends the previous one
if last and candidate.startswith(last):
out[-1] = candidate
last = candidate
continue
if last and last.startswith(candidate):
continue
out.append(candidate)
last = candidate
return " ".join(out)