Files
pdf2epub/scripts/test_audiobook.py
T

165 lines
6.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Самопроверка обвязки озвучки: python3 test_audiobook.py (нужен ffmpeg)"""
import json
import subprocess
import sys
import tempfile
from pathlib import Path
import audiobook as a
def mp3(path, seconds):
"""Тишина заданной длительности — реальный mp3 для ffprobe."""
subprocess.run(
["ffmpeg", "-v", "error", "-y", "-f", "lavfi",
"-i", "anullsrc=r=22050:cl=mono", "-t", str(seconds),
"-b:a", "32k", str(path)],
check=True)
def write_chapter(d, num, paragraphs):
(d / ("chapter_%03d_translated.json" % num)).write_text(
json.dumps({"number": num, "title": "Глава ⟦b⟧%d⟦/b⟧" % num,
"paragraphs": paragraphs}, ensure_ascii=False),
encoding="utf-8")
def test_prep(tmp: Path):
src, out = tmp / "translations", tmp / "tts"
src.mkdir()
write_chapter(src, 0, ["Он сказал ⟦i⟧привет⟧⟦/i⟧ громко.",
"Обычный абзац.", " "])
ns = type("ns", (), {"translations": src, "output": out})
a.cmd_prep(ns)
data = json.loads((out / "chapter_000_translated.json").read_text(encoding="utf-8"))
assert "⟦" not in json.dumps(data, ensure_ascii=False), "маркеры должны исчезнуть"
assert data["title"] == "Глава 0"
assert len(data["paragraphs"]) == 2, "пустой абзац должен отброситься"
assert " " not in data["paragraphs"][1], "двойные пробелы должны схлопнуться"
def test_verify_detects_gap(tmp: Path):
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
# две главы по 4 абзаца одинаковой длины
para = "х" * 100
for n in (0, 1):
write_chapter(tts, n, [para] * 4)
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
# у главы 1 намеренно пропущен последний абзац
count = 4 if n == 0 else 3
for i in range(count):
mp3(temp / ("chapter_%03d_para_%04d.mp3" % (n, i)), 6)
ns = type("ns", (), {"translations": tts, "audiobook": ab})
assert a.cmd_verify(ns) is False, "пропущенный фрагмент должен завалить проверку"
def test_verify_clean(tmp: Path):
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
para = "х" * 100
for n in (0, 1):
write_chapter(tts, n, [para] * 3)
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
for i in range(3):
mp3(temp / ("chapter_%03d_para_%04d.mp3" % (n, i)), 6)
ns = type("ns", (), {"translations": tts, "audiobook": ab})
assert a.cmd_verify(ns) is True, "целая книга должна проходить проверку"
def test_verify_detects_empty_file(tmp: Path):
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
para = "х" * 100
write_chapter(tts, 0, [para] * 3)
mp3(temp / "chapter_000_intro.mp3", 1)
mp3(temp / "chapter_000_para_0000.mp3", 6)
mp3(temp / "chapter_000_para_0001.mp3", 6)
(temp / "chapter_000_para_0002.mp3").touch() # нулевой размер
ns = type("ns", (), {"translations": tts, "audiobook": ab})
assert a.cmd_verify(ns) is False, "пустой файл должен завалить проверку"
def build_book(tmp: Path, chapters=2, paras=3):
"""Готовый синтез: подготовленные главы + фрагменты в temp_audio/."""
tts, ab = tmp / "tts", tmp / "audiobook"
tts.mkdir()
temp = ab / "temp_audio"
temp.mkdir(parents=True)
for n in range(chapters):
write_chapter(tts, n, ["х" * 100] * paras)
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
for i in range(paras):
mp3(temp / ("chapter_%03d_para_%04d.mp3" % (n, i)), 2)
return tts, ab
def split_ns(tts, ab, out, **kw):
fields = {"translations": tts, "audiobook": ab, "output": out,
"album": "Книга", "author": "Автор", "gap": 0.3, "force": False}
fields.update(kw)
return type("ns", (), fields)
def test_split_makes_chapter_tracks(tmp: Path):
tts, ab = build_book(tmp, chapters=2, paras=3)
out = tmp / "tracks"
assert a.cmd_split(split_ns(tts, ab, out)) is True
tracks = sorted(out.glob("*.mp3"))
assert len(tracks) == 2, "по треку на главу"
assert tracks[0].name.startswith("000 - "), tracks[0].name
assert not (out / ".tmp").exists(), "временный каталог должен убираться"
# длительность: 1 вводный + 3 абзаца + 3 паузы = 1 + 6 + 0.9 ≈ 7.9 с
d = a.duration(tracks[0])
assert 7.0 < d < 9.0, d
# теги проставлены
r = subprocess.run(
["ffprobe", "-v", "error", "-show_entries",
"format_tags=album,artist,track,title", "-of", "default=nw=1",
str(tracks[0])], capture_output=True, text=True)
assert "Книга" in r.stdout and "Автор" in r.stdout, r.stdout
assert "track=1/2" in r.stdout, r.stdout
def test_split_skips_incomplete_chapter(tmp: Path):
tts, ab = build_book(tmp, chapters=2, paras=3)
(ab / "temp_audio" / "chapter_001_para_0002.mp3").unlink()
out = tmp / "tracks"
assert a.cmd_split(split_ns(tts, ab, out)) is False, "неполная глава должна пропускаться"
assert len(list(out.glob("*.mp3"))) == 1, "целая глава всё равно нарезается"
out2 = tmp / "forced"
assert a.cmd_split(split_ns(tts, ab, out2, force=True)) is True
assert len(list(out2.glob("*.mp3"))) == 2, "--force режет обе"
def test_split_without_gap(tmp: Path):
tts, ab = build_book(tmp, chapters=1, paras=2)
out = tmp / "tracks"
assert a.cmd_split(split_ns(tts, ab, out, gap=0)) is True
d = a.duration(next(out.glob("*.mp3")))
assert 4.5 < d < 5.6, d # 1 + 2 + 2 без пауз
if __name__ == "__main__":
if subprocess.run(["which", "ffmpeg"], capture_output=True).returncode:
sys.exit("нужен ffmpeg")
for fn in (test_prep, test_verify_detects_gap, test_verify_clean,
test_verify_detects_empty_file, test_split_makes_chapter_tracks,
test_split_skips_incomplete_chapter, test_split_without_gap):
with tempfile.TemporaryDirectory() as d:
print("---", fn.__name__)
fn(Path(d))
print("OK")