Files
pdf2epub/scripts/test_audiobook.py
T

183 lines
7.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Самопроверка обвязки озвучки: python3 test_audiobook.py (нужен ffmpeg)"""
import json
import subprocess
import sys
import tempfile
from pathlib import Path
import audiobook as a
def mp3(path, seconds):
"""Тишина заданной длительности — реальный mp3 для ffprobe."""
subprocess.run(
["ffmpeg", "-v", "error", "-y", "-f", "lavfi",
"-i", "anullsrc=r=22050:cl=mono", "-t", str(seconds),
"-b:a", "32k", str(path)],
check=True)
def write_chapter(d, num, paragraphs):
(d / ("chapter_%03d_translated.json" % num)).write_text(
json.dumps({"number": num, "title": "Глава ⟦b⟧%d⟦/b⟧" % num,
"paragraphs": paragraphs}, ensure_ascii=False),
encoding="utf-8")
def test_prep(tmp: Path):
src, out = tmp / "translations", tmp / "tts"
src.mkdir()
write_chapter(src, 0, ["Он сказал ⟦i⟧привет⟧⟦/i⟧ громко.",
"Обычный абзац.", " "])
ns = type("ns", (), {"translations": src, "output": out})
a.cmd_prep(ns)
data = json.loads((out / "chapter_000_translated.json").read_text(encoding="utf-8"))
assert "⟦" not in json.dumps(data, ensure_ascii=False), "маркеры должны исчезнуть"
assert data["title"] == "Глава 0"
assert len(data["paragraphs"]) == 2, "пустой абзац должен отброситься"
assert " " not in data["paragraphs"][1], "двойные пробелы должны схлопнуться"
def test_verify_detects_gap(tmp: Path):
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
# две главы по 4 абзаца одинаковой длины
para = "х" * 100
for n in (0, 1):
write_chapter(tts, n, [para] * 4)
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
# у главы 1 намеренно пропущена последняя группа
count = 2 if n == 0 else 1
for i in range(count):
mp3(temp / ("chapter_%03d_group_%03d.mp3" % (n, i)), 9)
ns = type("ns", (), {"translations": tts, "audiobook": ab, "group_size": 3})
assert a.cmd_verify(ns) is False, "пропущенный фрагмент должен завалить проверку"
def test_verify_clean(tmp: Path):
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
para = "х" * 100
for n in (0, 1):
write_chapter(tts, n, [para] * 3)
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
mp3(temp / ("chapter_%03d_group_000.mp3" % n), 18)
ns = type("ns", (), {"translations": tts, "audiobook": ab, "group_size": 3})
assert a.cmd_verify(ns) is True, "целая книга должна проходить проверку"
def test_verify_detects_empty_file(tmp: Path):
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
para = "х" * 100
write_chapter(tts, 0, [para] * 9)
mp3(temp / "chapter_000_intro.mp3", 1)
mp3(temp / "chapter_000_group_000.mp3", 18)
mp3(temp / "chapter_000_group_001.mp3", 18)
(temp / "chapter_000_group_002.mp3").touch() # нулевой размер
ns = type("ns", (), {"translations": tts, "audiobook": ab, "group_size": 3})
assert a.cmd_verify(ns) is False, "пустой файл должен завалить проверку"
def build_book(tmp: Path, chapters=2, paras=3, group=3, secs=2):
"""Готовый синтез: главы + фрагменты, сгруппированные как в чужом скрипте."""
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
for n in range(chapters):
write_chapter(tts, n, ["х" * 100] * paras)
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
for i in range(-(-paras // group)):
mp3(temp / ("chapter_%03d_group_%03d.mp3" % (n, i)), secs)
return tts, ab
def split_ns(tts, ab, out, **kw):
fields = {"translations": tts, "audiobook": ab, "output": out,
"album": "Книга", "author": "Автор", "gap": 0.3,
"force": False, "group_size": 3}
fields.update(kw)
return type("ns", (), fields)
def test_split_makes_chapter_tracks(tmp: Path):
tts, ab = build_book(tmp, chapters=2, paras=3)
out = tmp / "tracks"
assert a.cmd_split(split_ns(tts, ab, out)) is True
tracks = sorted(out.glob("*.mp3"))
assert len(tracks) == 2, "по треку на главу"
assert tracks[0].name.startswith("000 - "), tracks[0].name
assert not (out / ".tmp").exists(), "временный каталог должен убираться"
# 1 вводный + 1 группа + 1 пауза = 1 + 2 + 0.3 ≈ 3.3 с
d = a.duration(tracks[0])
assert 2.8 < d < 4.0, d
# теги проставлены
r = subprocess.run(
["ffprobe", "-v", "error", "-show_entries",
"format_tags=album,artist,track,title", "-of", "default=nw=1",
str(tracks[0])], capture_output=True, text=True)
assert "Книга" in r.stdout and "Автор" in r.stdout, r.stdout
assert "track=1/2" in r.stdout, r.stdout
def test_split_skips_incomplete_chapter(tmp: Path):
tts, ab = build_book(tmp, chapters=2, paras=3)
(ab / "temp_audio" / "chapter_001_group_000.mp3").unlink()
out = tmp / "tracks"
assert a.cmd_split(split_ns(tts, ab, out)) is False, "неполная глава должна пропускаться"
assert len(list(out.glob("*.mp3"))) == 1, "целая глава всё равно нарезается"
out2 = tmp / "forced"
assert a.cmd_split(split_ns(tts, ab, out2, force=True)) is True
assert len(list(out2.glob("*.mp3"))) == 2, "--force режет обе"
def test_split_survives_broken_fragment(tmp: Path):
"""Регресс: пустой фрагмент обрывал concat молча, ffmpeg возвращал 0."""
tts, ab = build_book(tmp, chapters=1, paras=9, secs=4)
broken = ab / "temp_audio" / "chapter_000_group_001.mp3"
broken.write_bytes(b"") # ровно тот случай, что дал edge-tts
out = tmp / "tracks"
assert a.cmd_split(split_ns(tts, ab, out)) is False, \
"глава с битым фрагментом не должна выдаваться как готовая"
assert not list(out.glob("*.mp3")), "обрезанный трек не должен остаться на диске"
# с --force собирается из уцелевших, но без тихого обрыва посередине
out2 = tmp / "forced"
assert a.cmd_split(split_ns(tts, ab, out2, force=True, gap=0)) is True
d = a.duration(next(out2.glob("*.mp3")))
assert 8.5 < d < 9.5, d # вводный 1 с + две уцелевшие группы по 4 с
def test_split_without_gap(tmp: Path):
tts, ab = build_book(tmp, chapters=1, paras=6)
out = tmp / "tracks"
assert a.cmd_split(split_ns(tts, ab, out, gap=0)) is True
d = a.duration(next(out.glob("*.mp3")))
assert 4.5 < d < 5.6, d # 1 вводный + 2 группы по 2 с, без пауз
if __name__ == "__main__":
if subprocess.run(["which", "ffmpeg"], capture_output=True).returncode:
sys.exit("нужен ffmpeg")
for fn in (test_prep, test_verify_detects_gap, test_verify_clean,
test_verify_detects_empty_file, test_split_makes_chapter_tracks,
test_split_skips_incomplete_chapter,
test_split_survives_broken_fragment, test_split_without_gap):
with tempfile.TemporaryDirectory() as d:
print("---", fn.__name__)
fn(Path(d))
print("OK")