Files
talkscore-asr/build/check_gaps.py
T
Vladimir BryzgalovandClaude Opus 5 56cf7e5b0e Второй проход по промежуткам, которые детектор счёл тишиной
Диаризация задавала не только «кто говорит», но и границы того, что вообще
попадает в распознавание: фразы вне её сегментов пропадали молча. Второй
проход их возвращает с пометкой recovered=true и speaker=0.

Замер на восьми записях: 131 слово из 17 004 (0,8 %), 3-19 с на запись.
Возвращается и настоящий диалог (вопрос о цене, даты, документы), и радио
на фоне, поэтому проход отключается настройкой recover_gaps.

Скрипты render_transcript.py и check_gaps.py - для сверки расшифровки
с записью и поиска потерянных фраз.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-16 22:07:02 +05:00

116 lines
4.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Проверяет, была ли речь в пропусках расшифровки.
Распознаётся только то, что детектор речи отметил как речь. Пропуск в
расшифровке может означать и настоящую тишину, и потерянную фразу - на глаз
это не различить. Здесь громкость внутри пропуска сравнивается с громкостью
распознанных реплик той же записи: если она сопоставима, речь была, и
детектор её не увидел.
uv run python build/check_gaps.py результат.json запись.mp3
"""
import argparse
import json
import subprocess
import sys
import tempfile
import wave
from pathlib import Path
import numpy as np
SAMPLE_RATE = 16000
GAP_MIN_SEC = 3.0
# Насколько тише распознанной речи должен быть фрагмент, чтобы считать его
# тишиной. 12 дБ - это примерно вчетверо тише по амплитуде.
SILENCE_MARGIN_DB = 12.0
def to_wav(src: Path, dst: Path) -> None:
subprocess.run(
["ffmpeg", "-nostdin", "-y", "-i", str(src), "-ac", "1",
"-ar", str(SAMPLE_RATE), "-vn", str(dst)],
check=True, capture_output=True)
def read_wav(path: Path) -> np.ndarray:
with wave.open(str(path), "rb") as handle:
raw = handle.readframes(handle.getnframes())
return np.frombuffer(raw, dtype=np.int16).astype(np.float32) / 32768.0
def level_db(samples: np.ndarray) -> float:
if len(samples) == 0:
return -120.0
return 20 * float(np.log10(float(np.sqrt((samples ** 2).mean())) + 1e-9))
def slice_at(samples: np.ndarray, start: float, end: float) -> np.ndarray:
return samples[int(start * SAMPLE_RATE):int(end * SAMPLE_RATE)]
def timecode(seconds: float) -> str:
return f"{int(seconds) // 60:02d}:{int(seconds) % 60:02d}"
def main() -> int:
ap = argparse.ArgumentParser()
ap.add_argument("result", type=Path)
ap.add_argument("audio", type=Path)
args = ap.parse_args()
data = json.loads(args.result.read_text(encoding="utf-8"))
turns = data.get("turns", [])
if not turns:
print("в результате нет реплик")
return 1
with tempfile.TemporaryDirectory() as tmp:
wav = Path(tmp) / "audio.wav"
to_wav(args.audio, wav)
samples = read_wav(wav)
duration = len(samples) / SAMPLE_RATE
speech = float(np.median([level_db(slice_at(samples, t["start"], t["end"]))
for t in turns]))
threshold = speech - SILENCE_MARGIN_DB
print(f"Запись: {args.audio.name}, {timecode(duration)}")
print(f"Медианная громкость распознанной речи: {speech:.1f} дБ")
print(f"Порог: тише {threshold:.1f} дБ считаем тишиной\n")
gaps = []
previous_end = 0.0
for turn in turns:
if turn["start"] - previous_end >= GAP_MIN_SEC:
gaps.append((previous_end, turn["start"]))
previous_end = turn["end"]
if duration - previous_end >= GAP_MIN_SEC:
gaps.append((previous_end, duration))
suspicious = 0.0
quiet = 0.0
print(f"{'пропуск':<18} {'длина':>7} {'громкость':>10} вердикт")
for start, end in gaps:
loud = level_db(slice_at(samples, start, end))
span = end - start
if loud >= threshold:
suspicious += span
verdict = "ПОХОЖЕ НА РЕЧЬ"
else:
quiet += span
verdict = "тишина"
print(f"{timecode(start)}-{timecode(end):<11} {span:>6.0f}с "
f"{loud:>9.1f}дБ {verdict}")
total = suspicious + quiet
print(f"\nВсего в пропусках: {total:.0f} с")
print(f" тишина: {quiet:.0f} с ({quiet / total * 100 if total else 0:.0f}%)")
print(f" похоже на речь: {suspicious:.0f} с "
f"({suspicious / total * 100 if total else 0:.0f}%) "
f"- {suspicious / duration * 100:.0f}% всей записи")
return 0
if __name__ == "__main__":
sys.exit(main())