from pathlib import Path import wave import json import math from typing import List, Dict def _rms(frames: bytes, sample_width: int) -> float: # compute RMS of raw frames (little-endian signed samples) if sample_width == 2: import struct count = len(frames) // 2 if count == 0: return 0.0 fmt = "<%dh" % count vals = struct.unpack(fmt, frames) s = 0 for v in vals: s += v * v return math.sqrt(s / count) else: return 0.0 def diarize_by_silence(wav_path: Path, win_ms: int = 30, silence_thresh: float = 500.0, min_silence_ms: int = 400) -> List[Dict]: """Naive diarization by splitting on long silence. Returns list of segments: {'start': float, 'end': float, 'speaker': str} Speakers are assigned alternately: Speaker 1, Speaker 2, ... """ segs = [] with wave.open(str(wav_path), "rb") as wf: sr = wf.getframerate() sw = wf.getsampwidth() n_channels = wf.getnchannels() # we expect mono frames_per_win = int(sr * win_ms / 1000) min_silence_frames = int(sr * min_silence_ms / 1000) total_frames = wf.getnframes() pos = 0 silent_acc = 0 current_start = 0 speaker_idx = 1 while pos < total_frames: to_read = min(frames_per_win, total_frames - pos) raw = wf.readframes(to_read) pos += to_read level = _rms(raw, sw) if level < silence_thresh: silent_acc += to_read else: # if there was long silence before speech, start a new segment if silent_acc >= min_silence_frames and pos / sr > current_start: end_time = (pos - silent_acc) / sr segs.append({"start": current_start, "end": end_time, "speaker": f"Speaker {speaker_idx}"}) speaker_idx = (speaker_idx % 2) + 1 current_start = end_time silent_acc = 0 # finish last segment if current_start < total_frames / sr: segs.append({"start": current_start, "end": total_frames / sr, "speaker": f"Speaker {speaker_idx}"}) return segs def assign_words_to_speakers(words: List[Dict], segments: List[Dict]) -> List[Dict]: """Assign word dicts (with start/end) to speaker segments. Returns list of speaker blocks with text.""" result = [] seg_idx = 0 current_block = {"speaker": segments[0]["speaker"] if segments else "Speaker 1", "start": None, "end": None, "words": []} for w in words: t = w.get("start", 0) # move seg_idx until this word falls into segment while seg_idx + 1 < len(segments) and t >= segments[seg_idx]["end"]: # flush current if current_block["words"]: current_block["end"] = segments[seg_idx]["end"] result.append(current_block) seg_idx += 1 current_block = {"speaker": segments[seg_idx]["speaker"], "start": None, "end": None, "words": []} # append word if current_block["start"] is None: current_block["start"] = w.get("start") current_block["words"].append(w) if current_block["words"]: # set end current_block["end"] = current_block["words"][-1].get("end") result.append(current_block) return result def chapter_by_wordcount(words: List[Dict], speaker_blocks: List[Dict] = None, words_per_chapter: int = 100, min_words_on_speaker_change: int = 30) -> List[Dict]: """Chaptering that prefers boundaries at speaker changes. - If `speaker_blocks` provided, will try to split a chapter when speaker changes and the current chapter has at least `min_words_on_speaker_change` words. - Also split when `words_per_chapter` is reached. Returns list of chapters: {'start': float, 'end': float, 'words': [...]}. """ chapters = [] chapter = {"start": None, "end": None, "words": []} count = 0 def speaker_for_time(t: float): if not speaker_blocks or t is None: return None for b in speaker_blocks: bstart = b.get("start") bend = b.get("end") if bstart is None or bend is None: continue try: if bstart <= t <= bend: return b.get("speaker") except Exception: continue return None prev_speaker = None for w in words: wstart = w.get("start") if chapter["start"] is None: chapter["start"] = wstart chapter["words"].append(w) chapter["end"] = w.get("end") count += 1 # check speaker change boundary cur_speaker = speaker_for_time(wstart) if prev_speaker is None: prev_speaker = cur_speaker if cur_speaker is not None and prev_speaker is not None and cur_speaker != prev_speaker: if count >= min_words_on_speaker_change: # close chapter at previous word chapters.append(chapter) chapter = {"start": None, "end": None, "words": []} count = 0 prev_speaker = cur_speaker continue else: # don't split if too short to make a chapter prev_speaker = cur_speaker # check fixed-size boundary if count >= words_per_chapter: chapters.append(chapter) chapter = {"start": None, "end": None, "words": []} count = 0 prev_speaker = None if chapter["words"]: chapters.append(chapter) return chapters def save_transcript(text_dir: Path, base_name: str, speaker_blocks: List[Dict], chapters: List[Dict]): text_dir.mkdir(parents=True, exist_ok=True) out_json = text_dir / f"{base_name}.json" out_txt = text_dir / f"{base_name}.txt" # Automatic host labeling: speaker with most words -> 'Ведущий', others -> 'Участник N' try: # Prefer selecting host by total spoken duration (better for long monologues). durations = [] for b in speaker_blocks: dur = 0.0 for w in b.get("words", []): s = w.get("start") e = w.get("end") if s is None or e is None: continue try: dur += float(e) - float(s) except Exception: continue durations.append(dur) if any(durations): host_idx = int(max(range(len(durations)), key=lambda i: durations[i])) else: # fallback to word counts if timestamps missing counts = [sum(1 for w in b.get("words", [])) for b in speaker_blocks] host_idx = int(max(range(len(counts)), key=lambda i: counts[i])) if counts else 0 new_blocks = [] participant_idx = 1 for i, b in enumerate(speaker_blocks): b_copy = dict(b) if i == host_idx: b_copy["speaker"] = "Ведущий" else: b_copy["speaker"] = f"Участник {participant_idx}" participant_idx += 1 new_blocks.append(b_copy) speaker_blocks = new_blocks except Exception: # fall back to provided labels on error pass data = {"speakers": speaker_blocks, "chapters": chapters} with open(out_json, "w", encoding="utf-8") as fh: json.dump(data, fh, ensure_ascii=False, indent=2) # plain text with speaker labels and chapter headings with open(out_txt, "w", encoding="utf-8") as fh: for i, ch in enumerate(chapters, start=1): fh.write(f"=== Глава {i} ===\n") # collect speaker text overlapping this chapter by word membership # build a simple view: iterate speaker blocks and print their words that fall into chapter for b in speaker_blocks: # collect words in this chapter that belong to this speaker block b_words = [] for w in ch["words"]: # a word belongs to speaker block if its start is within block start/end (if available) wstart = w.get("start") if wstart is None: continue bstart = b.get("start") bend = b.get("end") if bstart is None or bend is None: continue if bstart <= wstart <= bend: b_words.append(w["word"]) if b_words: fh.write(f"{b['speaker']}: ") fh.write(" ".join(b_words) + "\n") fh.write("\n") return out_json, out_txt