Files
talkscore-asr/tests/test_text.py
T
Vladimir BryzgalovandClaude Opus 5 9dc67bda5c talkscore-asr 0.1.0: сервис транскрибации и диаризации
Локальный FastAPI-сервис поверх GigaAM v3 и sherpa-onnx: приём аудио,
очередь задач, разделение по говорящим, постобработка терминов.

Доставка на Windows - ZIP со встроенным Python, без установки чего-либо.
Обновление кода при запуске тянется из релизов Gitea: меняется только
папка app, десятки килобайт вместо всего пакета.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-15 21:41:04 +05:00

77 lines
4.0 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Тесты постобработки текста: словарь замен и типографика."""
import pytest
from app.text import apply_replacements, load_replacements, normalize_typography
class TestReplacements:
def test_replaces_known_term(self):
rules = {"тэг менеджер": "Tag Manager"}
assert apply_replacements("открываем тэг менеджер", rules) == "открываем Tag Manager"
def test_replacement_is_case_insensitive_on_match(self):
rules = {"тэг менеджер": "Tag Manager"}
assert apply_replacements("Открываем Тэг Менеджер тут", rules) == "Открываем Tag Manager тут"
def test_replacement_keeps_surrounding_punctuation(self):
rules = {"джетур": "G-Tour"}
assert apply_replacements("бренд джетур, смотрите", rules) == "бренд G-Tour, смотрите"
def test_does_not_replace_inside_word(self):
"""«лид» не должен ломать «лидер»."""
rules = {"лед": "лид"}
assert apply_replacements("лидер рынка", rules) == "лидер рынка"
def test_multiword_replacement(self):
rules = {"гугл так менеджер": "Google Tag Manager"}
assert apply_replacements("это гугл так менеджер", rules) == "это Google Tag Manager"
def test_empty_rules_returns_original(self):
assert apply_replacements("текст без изменений", {}) == "текст без изменений"
def test_longest_rule_wins(self):
"""Более длинное правило применяется раньше короткого."""
rules = {"так менеджер": "Tag Manager", "гугл так менеджер": "Google Tag Manager"}
assert apply_replacements("гугл так менеджер", rules) == "Google Tag Manager"
class TestLoadReplacements:
def test_parses_simple_file(self, tmp_path):
f = tmp_path / "r.txt"
f.write_text("джетур = G-Tour\nтэг менеджер = Tag Manager\n", encoding="utf-8")
rules = load_replacements(f)
assert rules == {"джетур": "G-Tour", "тэг менеджер": "Tag Manager"}
def test_skips_comments_and_blanks(self, tmp_path):
f = tmp_path / "r.txt"
f.write_text("# комментарий\n\nджетур = G-Tour\n\n# ещё\n", encoding="utf-8")
assert load_replacements(f) == {"джетур": "G-Tour"}
def test_missing_file_returns_empty(self, tmp_path):
assert load_replacements(tmp_path / "нет.txt") == {}
def test_ignores_line_without_separator(self, tmp_path):
f = tmp_path / "r.txt"
f.write_text("мусор без разделителя\nджетур = G-Tour\n", encoding="utf-8")
assert load_replacements(f) == {"джетур": "G-Tour"}
class TestTypography:
@pytest.mark.parametrize("dash", ["—", "–", "‒", "―", "−"])
def test_replaces_all_dash_variants_with_hyphen(self, dash):
assert normalize_typography(f"слово {dash} слово") == "слово - слово"
def test_glues_dash_without_spaces(self):
"""GigaAM пишет «слово—слово» без пробелов."""
assert normalize_typography("это—тестирование") == "это - тестирование"
def test_adds_space_after_opening_quote(self):
"""GigaAM пишет «слово«цитата» без пробела перед кавычкой."""
assert normalize_typography("спрашивает:«По метрике") == "спрашивает: «По метрике"
def test_collapses_multiple_spaces(self):
assert normalize_typography("много пробелов") == "много пробелов"
def test_keeps_hyphen_in_compound_word(self):
assert normalize_typography("look-alike и из-за") == "look-alike и из-за"