Files
LLM-infa/api_legacy/lib/transcript_utils.py
jze9 2697e01714 Рабочий пайплайн: YouTube -> транскрипция -> выжимка -> Postgres
- новый api/: /pipeline/process (yt-dlp, субтитры или Vosk, LLM/экстрактивная
  суммаризация), /videos, /llm; старый код перенесён в api_legacy/
- web/: Flet UI (flet 0.28.3 + flet-web, порт 8550)
- Dockerfile.api: ffmpeg слоем из mwader/static-ffmpeg, pip с кэш-маунтом
- requirements/pyproject: убраны moviepy, imageio-ffmpeg, битый asyncio
- README с инструкцией запуска и загрузкой Vosk-модели

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-14 11:12:46 +05:00

237 lines
8.7 KiB
Python

from pathlib import Path
import wave
import json
import math
from typing import List, Dict
def _rms(frames: bytes, sample_width: int) -> float:
# compute RMS of raw frames (little-endian signed samples)
if sample_width == 2:
import struct
count = len(frames) // 2
if count == 0:
return 0.0
fmt = "<%dh" % count
vals = struct.unpack(fmt, frames)
s = 0
for v in vals:
s += v * v
return math.sqrt(s / count)
else:
return 0.0
def diarize_by_silence(wav_path: Path, win_ms: int = 30, silence_thresh: float = 500.0, min_silence_ms: int = 400) -> List[Dict]:
"""Naive diarization by splitting on long silence.
Returns list of segments: {'start': float, 'end': float, 'speaker': str}
Speakers are assigned alternately: Speaker 1, Speaker 2, ...
"""
segs = []
with wave.open(str(wav_path), "rb") as wf:
sr = wf.getframerate()
sw = wf.getsampwidth()
n_channels = wf.getnchannels()
# we expect mono
frames_per_win = int(sr * win_ms / 1000)
min_silence_frames = int(sr * min_silence_ms / 1000)
total_frames = wf.getnframes()
pos = 0
silent_acc = 0
current_start = 0
speaker_idx = 1
while pos < total_frames:
to_read = min(frames_per_win, total_frames - pos)
raw = wf.readframes(to_read)
pos += to_read
level = _rms(raw, sw)
if level < silence_thresh:
silent_acc += to_read
else:
# if there was long silence before speech, start a new segment
if silent_acc >= min_silence_frames and pos / sr > current_start:
end_time = (pos - silent_acc) / sr
segs.append({"start": current_start, "end": end_time, "speaker": f"Speaker {speaker_idx}"})
speaker_idx = (speaker_idx % 2) + 1
current_start = end_time
silent_acc = 0
# finish last segment
if current_start < total_frames / sr:
segs.append({"start": current_start, "end": total_frames / sr, "speaker": f"Speaker {speaker_idx}"})
return segs
def assign_words_to_speakers(words: List[Dict], segments: List[Dict]) -> List[Dict]:
"""Assign word dicts (with start/end) to speaker segments. Returns list of speaker blocks with text."""
result = []
seg_idx = 0
current_block = {"speaker": segments[0]["speaker"] if segments else "Speaker 1", "start": None, "end": None, "words": []}
for w in words:
t = w.get("start", 0)
# move seg_idx until this word falls into segment
while seg_idx + 1 < len(segments) and t >= segments[seg_idx]["end"]:
# flush current
if current_block["words"]:
current_block["end"] = segments[seg_idx]["end"]
result.append(current_block)
seg_idx += 1
current_block = {"speaker": segments[seg_idx]["speaker"], "start": None, "end": None, "words": []}
# append word
if current_block["start"] is None:
current_block["start"] = w.get("start")
current_block["words"].append(w)
if current_block["words"]:
# set end
current_block["end"] = current_block["words"][-1].get("end")
result.append(current_block)
return result
def chapter_by_wordcount(words: List[Dict], speaker_blocks: List[Dict] = None, words_per_chapter: int = 100, min_words_on_speaker_change: int = 30) -> List[Dict]:
"""Chaptering that prefers boundaries at speaker changes.
- If `speaker_blocks` provided, will try to split a chapter when speaker changes
and the current chapter has at least `min_words_on_speaker_change` words.
- Also split when `words_per_chapter` is reached.
Returns list of chapters: {'start': float, 'end': float, 'words': [...]}.
"""
chapters = []
chapter = {"start": None, "end": None, "words": []}
count = 0
def speaker_for_time(t: float):
if not speaker_blocks or t is None:
return None
for b in speaker_blocks:
bstart = b.get("start")
bend = b.get("end")
if bstart is None or bend is None:
continue
try:
if bstart <= t <= bend:
return b.get("speaker")
except Exception:
continue
return None
prev_speaker = None
for w in words:
wstart = w.get("start")
if chapter["start"] is None:
chapter["start"] = wstart
chapter["words"].append(w)
chapter["end"] = w.get("end")
count += 1
# check speaker change boundary
cur_speaker = speaker_for_time(wstart)
if prev_speaker is None:
prev_speaker = cur_speaker
if cur_speaker is not None and prev_speaker is not None and cur_speaker != prev_speaker:
if count >= min_words_on_speaker_change:
# close chapter at previous word
chapters.append(chapter)
chapter = {"start": None, "end": None, "words": []}
count = 0
prev_speaker = cur_speaker
continue
else:
# don't split if too short to make a chapter
prev_speaker = cur_speaker
# check fixed-size boundary
if count >= words_per_chapter:
chapters.append(chapter)
chapter = {"start": None, "end": None, "words": []}
count = 0
prev_speaker = None
if chapter["words"]:
chapters.append(chapter)
return chapters
def save_transcript(text_dir: Path, base_name: str, speaker_blocks: List[Dict], chapters: List[Dict]):
text_dir.mkdir(parents=True, exist_ok=True)
out_json = text_dir / f"{base_name}.json"
out_txt = text_dir / f"{base_name}.txt"
# Automatic host labeling: speaker with most words -> 'Ведущий', others -> 'Участник N'
try:
# Prefer selecting host by total spoken duration (better for long monologues).
durations = []
for b in speaker_blocks:
dur = 0.0
for w in b.get("words", []):
s = w.get("start")
e = w.get("end")
if s is None or e is None:
continue
try:
dur += float(e) - float(s)
except Exception:
continue
durations.append(dur)
if any(durations):
host_idx = int(max(range(len(durations)), key=lambda i: durations[i]))
else:
# fallback to word counts if timestamps missing
counts = [sum(1 for w in b.get("words", [])) for b in speaker_blocks]
host_idx = int(max(range(len(counts)), key=lambda i: counts[i])) if counts else 0
new_blocks = []
participant_idx = 1
for i, b in enumerate(speaker_blocks):
b_copy = dict(b)
if i == host_idx:
b_copy["speaker"] = "Ведущий"
else:
b_copy["speaker"] = f"Участник {participant_idx}"
participant_idx += 1
new_blocks.append(b_copy)
speaker_blocks = new_blocks
except Exception:
# fall back to provided labels on error
pass
data = {"speakers": speaker_blocks, "chapters": chapters}
with open(out_json, "w", encoding="utf-8") as fh:
json.dump(data, fh, ensure_ascii=False, indent=2)
# plain text with speaker labels and chapter headings
with open(out_txt, "w", encoding="utf-8") as fh:
for i, ch in enumerate(chapters, start=1):
fh.write(f"=== Глава {i} ===\n")
# collect speaker text overlapping this chapter by word membership
# build a simple view: iterate speaker blocks and print their words that fall into chapter
for b in speaker_blocks:
# collect words in this chapter that belong to this speaker block
b_words = []
for w in ch["words"]:
# a word belongs to speaker block if its start is within block start/end (if available)
wstart = w.get("start")
if wstart is None:
continue
bstart = b.get("start")
bend = b.get("end")
if bstart is None or bend is None:
continue
if bstart <= wstart <= bend:
b_words.append(w["word"])
if b_words:
fh.write(f"{b['speaker']}: ")
fh.write(" ".join(b_words) + "\n")
fh.write("\n")
return out_json, out_txt