Обвязка озвучки: снятие маркеров разметки для TTS и проверка целостности синтеза

This commit is contained in:
Илья Поляков
2026-08-03 10:48:04 +03:00
parent b92a0a39c0
commit 93c40c76bf
4 changed files with 373 additions and 1 deletions
+100
View File
@@ -0,0 +1,100 @@
#!/usr/bin/env python3
"""Самопроверка обвязки озвучки: python3 test_audiobook.py (нужен ffmpeg)"""
import json
import subprocess
import sys
import tempfile
from pathlib import Path
import audiobook as a
def mp3(path, seconds):
"""Тишина заданной длительности — реальный mp3 для ffprobe."""
subprocess.run(
["ffmpeg", "-v", "error", "-y", "-f", "lavfi",
"-i", "anullsrc=r=22050:cl=mono", "-t", str(seconds),
"-b:a", "32k", str(path)],
check=True)
def write_chapter(d, num, paragraphs):
(d / ("chapter_%03d_translated.json" % num)).write_text(
json.dumps({"number": num, "title": "Глава ⟦b⟧%d⟦/b⟧" % num,
"paragraphs": paragraphs}, ensure_ascii=False),
encoding="utf-8")
def test_prep(tmp: Path):
src, out = tmp / "translations", tmp / "tts"
src.mkdir()
write_chapter(src, 0, ["Он сказал ⟦i⟧привет⟧⟦/i⟧ громко.",
"Обычный абзац.", " "])
ns = type("ns", (), {"translations": src, "output": out})
a.cmd_prep(ns)
data = json.loads((out / "chapter_000_translated.json").read_text(encoding="utf-8"))
assert "⟦" not in json.dumps(data, ensure_ascii=False), "маркеры должны исчезнуть"
assert data["title"] == "Глава 0"
assert len(data["paragraphs"]) == 2, "пустой абзац должен отброситься"
assert " " not in data["paragraphs"][1], "двойные пробелы должны схлопнуться"
def test_verify_detects_gap(tmp: Path):
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
# две главы по 4 абзаца одинаковой длины
para = "х" * 100
for n in (0, 1):
write_chapter(tts, n, [para] * 4)
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
# у главы 1 намеренно пропущен последний абзац
count = 4 if n == 0 else 3
for i in range(count):
mp3(temp / ("chapter_%03d_para_%04d.mp3" % (n, i)), 6)
ns = type("ns", (), {"translations": tts, "audiobook": ab})
assert a.cmd_verify(ns) is False, "пропущенный фрагмент должен завалить проверку"
def test_verify_clean(tmp: Path):
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
para = "х" * 100
for n in (0, 1):
write_chapter(tts, n, [para] * 3)
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
for i in range(3):
mp3(temp / ("chapter_%03d_para_%04d.mp3" % (n, i)), 6)
ns = type("ns", (), {"translations": tts, "audiobook": ab})
assert a.cmd_verify(ns) is True, "целая книга должна проходить проверку"
def test_verify_detects_empty_file(tmp: Path):
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
para = "х" * 100
write_chapter(tts, 0, [para] * 3)
mp3(temp / "chapter_000_intro.mp3", 1)
mp3(temp / "chapter_000_para_0000.mp3", 6)
mp3(temp / "chapter_000_para_0001.mp3", 6)
(temp / "chapter_000_para_0002.mp3").touch() # нулевой размер
ns = type("ns", (), {"translations": tts, "audiobook": ab})
assert a.cmd_verify(ns) is False, "пустой файл должен завалить проверку"
if __name__ == "__main__":
if subprocess.run(["which", "ffmpeg"], capture_output=True).returncode:
sys.exit("нужен ffmpeg")
for fn in (test_prep, test_verify_detects_gap, test_verify_clean,
test_verify_detects_empty_file):
with tempfile.TemporaryDirectory() as d:
print("---", fn.__name__)
fn(Path(d))
print("OK")