Рабочий пайплайн: YouTube -> транскрипция -> выжимка -> Postgres
- новый api/: /pipeline/process (yt-dlp, субтитры или Vosk, LLM/экстрактивная суммаризация), /videos, /llm; старый код перенесён в api_legacy/ - web/: Flet UI (flet 0.28.3 + flet-web, порт 8550) - Dockerfile.api: ffmpeg слоем из mwader/static-ffmpeg, pip с кэш-маунтом - requirements/pyproject: убраны moviepy, imageio-ffmpeg, битый asyncio - README с инструкцией запуска и загрузкой Vosk-модели Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
134
api_legacy/lib/audio_processing.py
Normal file
134
api_legacy/lib/audio_processing.py
Normal file
@@ -0,0 +1,134 @@
|
||||
from pathlib import Path
|
||||
import os
|
||||
import sys
|
||||
import threading
|
||||
import logging
|
||||
import wave
|
||||
import json
|
||||
import subprocess
|
||||
import shutil
|
||||
|
||||
try:
|
||||
from vosk import Model, KaldiRecognizer
|
||||
except Exception:
|
||||
Model = None
|
||||
KaldiRecognizer = None
|
||||
|
||||
|
||||
class FFmpegConverter:
|
||||
"""Utility to convert audio files using ffmpeg."""
|
||||
|
||||
@staticmethod
|
||||
def convert_mp3_to_wav(src: Path, dest: Path) -> None:
|
||||
cmd = [
|
||||
"ffmpeg",
|
||||
"-y",
|
||||
"-i",
|
||||
str(src),
|
||||
"-ac",
|
||||
"1",
|
||||
"-ar",
|
||||
"16000",
|
||||
"-vn",
|
||||
"-f",
|
||||
"wav",
|
||||
str(dest),
|
||||
]
|
||||
subprocess.run(cmd, check=True)
|
||||
|
||||
|
||||
class VoskModelManager:
|
||||
"""Loads a Vosk model and provides transcription methods.
|
||||
|
||||
Loading captures model stderr into Python logs to show progress.
|
||||
"""
|
||||
|
||||
def __init__(self, model_path: Path):
|
||||
self.logger = logging.getLogger("vosk.model_manager")
|
||||
if Model is None or KaldiRecognizer is None:
|
||||
raise RuntimeError("vosk is not installed in runtime")
|
||||
self.model = self._load_model_with_progress(model_path)
|
||||
|
||||
def _load_model_with_progress(self, path: Path):
|
||||
logger = logging.getLogger("vosk.loader")
|
||||
r_fd, w_fd = os.pipe()
|
||||
saved_stderr_fd = os.dup(sys.stderr.fileno())
|
||||
os.dup2(w_fd, sys.stderr.fileno())
|
||||
|
||||
def _reader(fd):
|
||||
with os.fdopen(fd, "rb") as fh:
|
||||
for raw in iter(fh.readline, b""):
|
||||
try:
|
||||
logger.info(raw.decode(errors="ignore").rstrip())
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
reader_thread = threading.Thread(target=_reader, args=(r_fd,), daemon=True)
|
||||
reader_thread.start()
|
||||
|
||||
try:
|
||||
model = Model(str(path))
|
||||
finally:
|
||||
try:
|
||||
os.dup2(saved_stderr_fd, sys.stderr.fileno())
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
os.close(saved_stderr_fd)
|
||||
except Exception:
|
||||
pass
|
||||
reader_thread.join(timeout=2)
|
||||
|
||||
return model
|
||||
|
||||
def transcribe_wav(self, wav_path: Path) -> str:
|
||||
if not wav_path.exists():
|
||||
raise FileNotFoundError(wav_path)
|
||||
try:
|
||||
with wave.open(str(wav_path), "rb") as wf:
|
||||
if wf.getnchannels() != 1:
|
||||
raise ValueError("Audio must be mono (1 channel)")
|
||||
recognizer = KaldiRecognizer(self.model, wf.getframerate())
|
||||
try:
|
||||
recognizer.SetWords(True)
|
||||
except Exception:
|
||||
# older vosk bindings may not expose SetWords; ignore if absent
|
||||
pass
|
||||
while True:
|
||||
data = wf.readframes(4000)
|
||||
if not data:
|
||||
break
|
||||
recognizer.AcceptWaveform(data)
|
||||
result = json.loads(recognizer.FinalResult())
|
||||
return result.get("text", "")
|
||||
except wave.Error as e:
|
||||
raise ValueError(f"Invalid WAV file: {e}")
|
||||
|
||||
def transcribe_wav_with_words(self, wav_path: Path):
|
||||
"""Transcribe WAV and return dict with 'text' and 'words' (word-level timestamps).
|
||||
|
||||
Returns: { 'text': str, 'words': [{'word': str, 'start': float, 'end': float}, ...] }
|
||||
"""
|
||||
if not wav_path.exists():
|
||||
raise FileNotFoundError(wav_path)
|
||||
try:
|
||||
with wave.open(str(wav_path), "rb") as wf:
|
||||
if wf.getnchannels() != 1:
|
||||
raise ValueError("Audio must be mono (1 channel)")
|
||||
recognizer = KaldiRecognizer(self.model, wf.getframerate())
|
||||
try:
|
||||
recognizer.SetWords(True)
|
||||
except Exception:
|
||||
pass
|
||||
while True:
|
||||
data = wf.readframes(4000)
|
||||
if not data:
|
||||
break
|
||||
recognizer.AcceptWaveform(data)
|
||||
res = json.loads(recognizer.FinalResult())
|
||||
words = res.get("result", [])
|
||||
# words are dictionaries with word, start, end
|
||||
text = res.get("text", "")
|
||||
return {"text": text, "words": words}
|
||||
except wave.Error as e:
|
||||
raise ValueError(f"Invalid WAV file: {e}")
|
||||
69
api_legacy/lib/diarize_embeddings.py
Normal file
69
api_legacy/lib/diarize_embeddings.py
Normal file
@@ -0,0 +1,69 @@
|
||||
from pathlib import Path
|
||||
from typing import List, Dict
|
||||
|
||||
def diarize_with_resemblyzer(wav_path: Path, window_s: float = 1.5, hop_s: float = 0.75, distance_threshold: float = 0.6) -> List[Dict]:
|
||||
"""Lightweight embedding-based diarization using resemblyzer + sklearn.
|
||||
|
||||
Returns list of segments: {'start': float, 'end': float, 'speaker': 'spk_N'}
|
||||
"""
|
||||
try:
|
||||
import numpy as np
|
||||
import librosa
|
||||
from resemblyzer import VoiceEncoder
|
||||
from sklearn.cluster import AgglomerativeClustering
|
||||
except Exception as e:
|
||||
raise RuntimeError(f"Missing dependency for embedding diarization: {e}")
|
||||
|
||||
sr = 16000
|
||||
wav, sr_loaded = librosa.load(str(wav_path), sr=sr)
|
||||
n = wav.shape[0]
|
||||
win = int(window_s * sr)
|
||||
hop = int(hop_s * sr)
|
||||
if n <= win:
|
||||
# short file: embed whole
|
||||
encoder = VoiceEncoder()
|
||||
emb = encoder.embed_utterance(wav)
|
||||
labels = [0]
|
||||
centers = [ (0.0 + n / sr) / 2.0 ]
|
||||
windows = [(0.0, n / sr)]
|
||||
else:
|
||||
encoder = VoiceEncoder()
|
||||
embeddings = []
|
||||
centers = []
|
||||
windows = []
|
||||
for start in range(0, n - win + 1, hop):
|
||||
chunk = wav[start:start+win]
|
||||
try:
|
||||
e = encoder.embed_utterance(chunk)
|
||||
except Exception:
|
||||
# fallback: mean pooling
|
||||
e = np.mean(chunk)
|
||||
embeddings.append(e)
|
||||
t0 = start / sr
|
||||
t1 = (start + win) / sr
|
||||
centers.append((t0 + t1) / 2.0)
|
||||
windows.append((t0, t1))
|
||||
|
||||
X = np.vstack(embeddings)
|
||||
# Agglomerative clustering with distance threshold
|
||||
model = AgglomerativeClustering(n_clusters=None, distance_threshold=distance_threshold, affinity='euclidean', linkage='average')
|
||||
labels = model.fit_predict(X)
|
||||
|
||||
# merge consecutive windows with same label into segments
|
||||
segs = []
|
||||
if len(labels) == 0:
|
||||
return segs
|
||||
cur_label = labels[0]
|
||||
cur_start = windows[0][0]
|
||||
cur_end = windows[0][1]
|
||||
for i, lab in enumerate(labels[1:], start=1):
|
||||
if lab == cur_label:
|
||||
cur_end = windows[i][1]
|
||||
else:
|
||||
segs.append({"start": cur_start, "end": cur_end, "speaker": f"spk_{int(cur_label)+1}"})
|
||||
cur_label = lab
|
||||
cur_start = windows[i][0]
|
||||
cur_end = windows[i][1]
|
||||
segs.append({"start": cur_start, "end": cur_end, "speaker": f"spk_{int(cur_label)+1}"})
|
||||
|
||||
return segs
|
||||
236
api_legacy/lib/transcript_utils.py
Normal file
236
api_legacy/lib/transcript_utils.py
Normal file
@@ -0,0 +1,236 @@
|
||||
from pathlib import Path
|
||||
import wave
|
||||
import json
|
||||
import math
|
||||
from typing import List, Dict
|
||||
|
||||
|
||||
def _rms(frames: bytes, sample_width: int) -> float:
|
||||
# compute RMS of raw frames (little-endian signed samples)
|
||||
if sample_width == 2:
|
||||
import struct
|
||||
|
||||
count = len(frames) // 2
|
||||
if count == 0:
|
||||
return 0.0
|
||||
fmt = "<%dh" % count
|
||||
vals = struct.unpack(fmt, frames)
|
||||
s = 0
|
||||
for v in vals:
|
||||
s += v * v
|
||||
return math.sqrt(s / count)
|
||||
else:
|
||||
return 0.0
|
||||
|
||||
|
||||
def diarize_by_silence(wav_path: Path, win_ms: int = 30, silence_thresh: float = 500.0, min_silence_ms: int = 400) -> List[Dict]:
|
||||
"""Naive diarization by splitting on long silence.
|
||||
|
||||
Returns list of segments: {'start': float, 'end': float, 'speaker': str}
|
||||
Speakers are assigned alternately: Speaker 1, Speaker 2, ...
|
||||
"""
|
||||
segs = []
|
||||
with wave.open(str(wav_path), "rb") as wf:
|
||||
sr = wf.getframerate()
|
||||
sw = wf.getsampwidth()
|
||||
n_channels = wf.getnchannels()
|
||||
# we expect mono
|
||||
frames_per_win = int(sr * win_ms / 1000)
|
||||
min_silence_frames = int(sr * min_silence_ms / 1000)
|
||||
total_frames = wf.getnframes()
|
||||
pos = 0
|
||||
silent_acc = 0
|
||||
current_start = 0
|
||||
speaker_idx = 1
|
||||
|
||||
while pos < total_frames:
|
||||
to_read = min(frames_per_win, total_frames - pos)
|
||||
raw = wf.readframes(to_read)
|
||||
pos += to_read
|
||||
level = _rms(raw, sw)
|
||||
if level < silence_thresh:
|
||||
silent_acc += to_read
|
||||
else:
|
||||
# if there was long silence before speech, start a new segment
|
||||
if silent_acc >= min_silence_frames and pos / sr > current_start:
|
||||
end_time = (pos - silent_acc) / sr
|
||||
segs.append({"start": current_start, "end": end_time, "speaker": f"Speaker {speaker_idx}"})
|
||||
speaker_idx = (speaker_idx % 2) + 1
|
||||
current_start = end_time
|
||||
silent_acc = 0
|
||||
|
||||
# finish last segment
|
||||
if current_start < total_frames / sr:
|
||||
segs.append({"start": current_start, "end": total_frames / sr, "speaker": f"Speaker {speaker_idx}"})
|
||||
|
||||
return segs
|
||||
|
||||
|
||||
def assign_words_to_speakers(words: List[Dict], segments: List[Dict]) -> List[Dict]:
|
||||
"""Assign word dicts (with start/end) to speaker segments. Returns list of speaker blocks with text."""
|
||||
result = []
|
||||
seg_idx = 0
|
||||
current_block = {"speaker": segments[0]["speaker"] if segments else "Speaker 1", "start": None, "end": None, "words": []}
|
||||
|
||||
for w in words:
|
||||
t = w.get("start", 0)
|
||||
# move seg_idx until this word falls into segment
|
||||
while seg_idx + 1 < len(segments) and t >= segments[seg_idx]["end"]:
|
||||
# flush current
|
||||
if current_block["words"]:
|
||||
current_block["end"] = segments[seg_idx]["end"]
|
||||
result.append(current_block)
|
||||
seg_idx += 1
|
||||
current_block = {"speaker": segments[seg_idx]["speaker"], "start": None, "end": None, "words": []}
|
||||
|
||||
# append word
|
||||
if current_block["start"] is None:
|
||||
current_block["start"] = w.get("start")
|
||||
current_block["words"].append(w)
|
||||
|
||||
if current_block["words"]:
|
||||
# set end
|
||||
current_block["end"] = current_block["words"][-1].get("end")
|
||||
result.append(current_block)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def chapter_by_wordcount(words: List[Dict], speaker_blocks: List[Dict] = None, words_per_chapter: int = 100, min_words_on_speaker_change: int = 30) -> List[Dict]:
|
||||
"""Chaptering that prefers boundaries at speaker changes.
|
||||
|
||||
- If `speaker_blocks` provided, will try to split a chapter when speaker changes
|
||||
and the current chapter has at least `min_words_on_speaker_change` words.
|
||||
- Also split when `words_per_chapter` is reached.
|
||||
Returns list of chapters: {'start': float, 'end': float, 'words': [...]}.
|
||||
"""
|
||||
chapters = []
|
||||
chapter = {"start": None, "end": None, "words": []}
|
||||
count = 0
|
||||
|
||||
def speaker_for_time(t: float):
|
||||
if not speaker_blocks or t is None:
|
||||
return None
|
||||
for b in speaker_blocks:
|
||||
bstart = b.get("start")
|
||||
bend = b.get("end")
|
||||
if bstart is None or bend is None:
|
||||
continue
|
||||
try:
|
||||
if bstart <= t <= bend:
|
||||
return b.get("speaker")
|
||||
except Exception:
|
||||
continue
|
||||
return None
|
||||
|
||||
prev_speaker = None
|
||||
for w in words:
|
||||
wstart = w.get("start")
|
||||
if chapter["start"] is None:
|
||||
chapter["start"] = wstart
|
||||
chapter["words"].append(w)
|
||||
chapter["end"] = w.get("end")
|
||||
count += 1
|
||||
|
||||
# check speaker change boundary
|
||||
cur_speaker = speaker_for_time(wstart)
|
||||
if prev_speaker is None:
|
||||
prev_speaker = cur_speaker
|
||||
|
||||
if cur_speaker is not None and prev_speaker is not None and cur_speaker != prev_speaker:
|
||||
if count >= min_words_on_speaker_change:
|
||||
# close chapter at previous word
|
||||
chapters.append(chapter)
|
||||
chapter = {"start": None, "end": None, "words": []}
|
||||
count = 0
|
||||
prev_speaker = cur_speaker
|
||||
continue
|
||||
else:
|
||||
# don't split if too short to make a chapter
|
||||
prev_speaker = cur_speaker
|
||||
|
||||
# check fixed-size boundary
|
||||
if count >= words_per_chapter:
|
||||
chapters.append(chapter)
|
||||
chapter = {"start": None, "end": None, "words": []}
|
||||
count = 0
|
||||
prev_speaker = None
|
||||
|
||||
if chapter["words"]:
|
||||
chapters.append(chapter)
|
||||
return chapters
|
||||
|
||||
|
||||
def save_transcript(text_dir: Path, base_name: str, speaker_blocks: List[Dict], chapters: List[Dict]):
|
||||
text_dir.mkdir(parents=True, exist_ok=True)
|
||||
out_json = text_dir / f"{base_name}.json"
|
||||
out_txt = text_dir / f"{base_name}.txt"
|
||||
# Automatic host labeling: speaker with most words -> 'Ведущий', others -> 'Участник N'
|
||||
try:
|
||||
# Prefer selecting host by total spoken duration (better for long monologues).
|
||||
durations = []
|
||||
for b in speaker_blocks:
|
||||
dur = 0.0
|
||||
for w in b.get("words", []):
|
||||
s = w.get("start")
|
||||
e = w.get("end")
|
||||
if s is None or e is None:
|
||||
continue
|
||||
try:
|
||||
dur += float(e) - float(s)
|
||||
except Exception:
|
||||
continue
|
||||
durations.append(dur)
|
||||
|
||||
if any(durations):
|
||||
host_idx = int(max(range(len(durations)), key=lambda i: durations[i]))
|
||||
else:
|
||||
# fallback to word counts if timestamps missing
|
||||
counts = [sum(1 for w in b.get("words", [])) for b in speaker_blocks]
|
||||
host_idx = int(max(range(len(counts)), key=lambda i: counts[i])) if counts else 0
|
||||
|
||||
new_blocks = []
|
||||
participant_idx = 1
|
||||
for i, b in enumerate(speaker_blocks):
|
||||
b_copy = dict(b)
|
||||
if i == host_idx:
|
||||
b_copy["speaker"] = "Ведущий"
|
||||
else:
|
||||
b_copy["speaker"] = f"Участник {participant_idx}"
|
||||
participant_idx += 1
|
||||
new_blocks.append(b_copy)
|
||||
speaker_blocks = new_blocks
|
||||
except Exception:
|
||||
# fall back to provided labels on error
|
||||
pass
|
||||
|
||||
data = {"speakers": speaker_blocks, "chapters": chapters}
|
||||
with open(out_json, "w", encoding="utf-8") as fh:
|
||||
json.dump(data, fh, ensure_ascii=False, indent=2)
|
||||
|
||||
# plain text with speaker labels and chapter headings
|
||||
with open(out_txt, "w", encoding="utf-8") as fh:
|
||||
for i, ch in enumerate(chapters, start=1):
|
||||
fh.write(f"=== Глава {i} ===\n")
|
||||
# collect speaker text overlapping this chapter by word membership
|
||||
# build a simple view: iterate speaker blocks and print their words that fall into chapter
|
||||
for b in speaker_blocks:
|
||||
# collect words in this chapter that belong to this speaker block
|
||||
b_words = []
|
||||
for w in ch["words"]:
|
||||
# a word belongs to speaker block if its start is within block start/end (if available)
|
||||
wstart = w.get("start")
|
||||
if wstart is None:
|
||||
continue
|
||||
bstart = b.get("start")
|
||||
bend = b.get("end")
|
||||
if bstart is None or bend is None:
|
||||
continue
|
||||
if bstart <= wstart <= bend:
|
||||
b_words.append(w["word"])
|
||||
if b_words:
|
||||
fh.write(f"{b['speaker']}: ")
|
||||
fh.write(" ".join(b_words) + "\n")
|
||||
fh.write("\n")
|
||||
|
||||
return out_json, out_txt
|
||||
Reference in New Issue
Block a user