From 93c40c76bf40c8268f0a490ae5e23bdb7f2e2144 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=D0=98=D0=BB=D1=8C=D1=8F=20=D0=9F=D0=BE=D0=BB=D1=8F=D0=BA?= =?UTF-8?q?=D0=BE=D0=B2?= Date: Mon, 3 Aug 2026 10:48:04 +0300 Subject: [PATCH] =?UTF-8?q?=D0=9E=D0=B1=D0=B2=D1=8F=D0=B7=D0=BA=D0=B0=20?= =?UTF-8?q?=D0=BE=D0=B7=D0=B2=D1=83=D1=87=D0=BA=D0=B8:=20=D1=81=D0=BD?= =?UTF-8?q?=D1=8F=D1=82=D0=B8=D0=B5=20=D0=BC=D0=B0=D1=80=D0=BA=D0=B5=D1=80?= =?UTF-8?q?=D0=BE=D0=B2=20=D1=80=D0=B0=D0=B7=D0=BC=D0=B5=D1=82=D0=BA=D0=B8?= =?UTF-8?q?=20=D0=B4=D0=BB=D1=8F=20TTS=20=D0=B8=20=D0=BF=D1=80=D0=BE=D0=B2?= =?UTF-8?q?=D0=B5=D1=80=D0=BA=D0=B0=20=D1=86=D0=B5=D0=BB=D0=BE=D1=81=D1=82?= =?UTF-8?q?=D0=BD=D0=BE=D1=81=D1=82=D0=B8=20=D1=81=D0=B8=D0=BD=D1=82=D0=B5?= =?UTF-8?q?=D0=B7=D0=B0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- README.md | 35 +++++++ SKILL.md | 45 ++++++++- scripts/audiobook.py | 194 ++++++++++++++++++++++++++++++++++++++ scripts/test_audiobook.py | 100 ++++++++++++++++++++ 4 files changed, 373 insertions(+), 1 deletion(-) create mode 100644 scripts/audiobook.py create mode 100644 scripts/test_audiobook.py diff --git a/README.md b/README.md index 0503869..1fd2206 100644 --- a/README.md +++ b/README.md @@ -172,6 +172,41 @@ rm -rf out/tr/progress out/tr/context out/tr/translations cd scripts && python3 test_translate.py ``` +## Озвучка (опционально) + +Синтез тоже делает `book_translator` (`05_create_audiobook.py`, движок +edge-tts от Microsoft — бесплатный, нужен интернет). В `scripts/audiobook.py` +две обвязки; чужой репозиторий по-прежнему не трогаем. + +```bash +pip install edge-tts pydub # ffmpeg нужен отдельно, из пакетов системы + +# 1. снять маркеры разметки — иначе TTS проговорит ⟦i⟧ +python3 scripts/audiobook.py prep out/tr/translations out/tr/translations_tts + +# 2. синтез (в фоне; на 12 часов звучания уходит 1.5–3 часа) +cd out/tr && python3 ~/Projects/book_translator/05_create_audiobook.py \ + --translations-dir translations_tts --voice dmitry --rate '+0%' + +# 3. проверка ДО того, как скрипт уберёт temp_audio/ +python3 scripts/audiobook.py verify out/tr/translations_tts out/tr/audiobook +``` + +`verify` находит главы с недостающими фрагментами, битые и нулевые mp3 и главы, +которые короче расчётной длительности. Норму «секунд на знак» он берёт как +медиану по всем главам, поэтому не зависит от выбранного голоса и скорости. +Код возврата ненулевой, если что-то не так. + +Повторный запуск синтеза добирает пропущенное: сторонний скрипт пропускает +фрагмент, если файл существует **и непустой**. Битые файлы, названные `verify`, +надо удалить руками — их он проверять не станет. + +Ограничения чужой стадии: на выходе один слитный `audiobook_complete.mp3` без +разбивки по главам и без меток — для плеера неудобно, разбивка здесь не +реализована. Для технической книги стоит сначала прогнать фонетику +(`07_extract_terms.py` + `08_generate_phonetics.py`), иначе русский голос +исковеркает все английские термины. + ## Проверка результата ```bash diff --git a/SKILL.md b/SKILL.md index e7423e5..277e425 100644 --- a/SKILL.md +++ b/SKILL.md @@ -1,6 +1,6 @@ --- name: pdf-to-kindle -description: Convert a text-layer PDF (usually one generated by calibre from an ebook) into EPUB/AZW3 for Kindle while preserving italics, bold, chapter headings, sidebars/journal blocks, images and cross-page paragraphs. Use when the user asks to read a PDF book on Kindle or another e-reader, to convert a PDF to EPUB/AZW3/MOBI, or complains that a converted book lost its italics, merged its chapters, or reflows badly on the device. Not for scanned PDFs (no text layer) — those need OCR first. Also covers translating such a book into another language before packing it, delegated to the external book_translator project. +description: Convert a text-layer PDF (usually one generated by calibre from an ebook) into EPUB/AZW3 for Kindle while preserving italics, bold, chapter headings, sidebars/journal blocks, images and cross-page paragraphs. Use when the user asks to read a PDF book on Kindle or another e-reader, to convert a PDF to EPUB/AZW3/MOBI, or complains that a converted book lost its italics, merged its chapters, or reflows badly on the device. Not for scanned PDFs (no text layer) — those need OCR first. Also covers translating such a book into another language before packing it, delegated to the external book_translator project, and generating a TTS audiobook from it with an integrity check for dropped fragments. --- # PDF to Kindle @@ -153,6 +153,49 @@ of each block round-trips, and blocks are reassembled by index. fallback, language detection, and a full split/rebuild against a faked translator response — no network, no API key. Run it after touching the bridge. +## Optional stage: audiobook + +Also delegated to `book_translator` (`05_create_audiobook.py`, Microsoft +edge-tts — free, needs internet). Two wrappers live in `scripts/audiobook.py`; +the external repo is still never edited. + +1. **Strip markup markers first.** The translated JSON still holds the + `⟦i⟧` markers from the translation stage — TTS would read them aloud: + + `python scripts/audiobook.py prep /translations /translations_tts` + + Point the external script at the *stripped* copy; the original keeps its + italics for the EPUB. +2. **Synthesize:** `05_create_audiobook.py --translations-dir <…>/translations_tts + --voice dmitry --rate '+0%'`. Fragments are **one per paragraph** (the + `--paragraphs-per-group` flag is not used by the loop), named + `chapter_NNN_intro.mp3` / `chapter_NNN_para_NNNN.mp3` under + `audiobook/temp_audio/`. +3. **Verify before the temp files are deleted** — `cleanup_temp_files()` wipes + `temp_audio/`, and after that only the merged file can be checked: + + `python scripts/audiobook.py verify /translations_tts /audiobook` + + It flags chapters missing fragments, zero-byte/undecodable mp3s, and + chapters whose duration falls short of what their character count predicts. + The seconds-per-character baseline is the median across chapters, so it + self-calibrates to whatever voice and `--rate` were used. Exit code is + non-zero when anything is wrong. +4. **Re-running fills gaps cheaply** — the external script skips any fragment + file that already exists *and is non-empty*, so delete the bad ones `verify` + named and run it again. It never re-checks that an existing file is sane, + which is exactly why step 3 exists. + +Known limits of the external stage: it merges everything into a single +`audiobook_complete.mp3` with no per-chapter files and no chapter marks — +poor for players; splitting is not implemented here. The phonetics stage +(`07_extract_terms.py` + `08_generate_phonetics.py`) is worth running first for +a technical book, or the Russian voice will mangle every English term. + +`scripts/test_audiobook.py` builds real silent mp3s with ffmpeg and checks that +`verify` catches a missing fragment, an empty file, and passes a clean book. +Needs ffmpeg; no network. + ## What the script does - style from span font names (`-It`, `Italic`, `Bold`, `Semibold`) → ``/``; diff --git a/scripts/audiobook.py b/scripts/audiobook.py new file mode 100644 index 0000000..2fd8dca --- /dev/null +++ b/scripts/audiobook.py @@ -0,0 +1,194 @@ +#!/usr/bin/env python3 +"""Обвязка вокруг стадии озвучки book_translator (репозиторий не модифицируется). + +prep — копия translations/ без разметочных маркеров, пригодная для TTS. +verify — проверка целостности: не потерялись ли фрагменты при синтезе. + +Зачем verify: edge-tts бесплатный и на большом объёме отказывает; сторонний +скрипт пропускает уже существующий файл, проверяя только «размер больше нуля», +и молча склеивает то, что получилось. Дыры в звуке обнаруживаются лишь при +прослушивании — эта проверка ловит их сразу после генерации. +""" +import argparse +import json +import re +import statistics +import subprocess +import sys +from pathlib import Path + +MARK = re.compile(r"⟦/?[ib]⟧") +FRAG = re.compile(r"^chapter_(\d{3})_(intro|para_\d{4})\.mp3$") +# ниже этой доли от расчётной длительности глава считается обрезанной +SHORT = 0.75 +LONG = 1.6 + + +def clean(text): + return re.sub(r"\s{2,}", " ", MARK.sub("", text)).strip() + + +def cmd_prep(args): + src = args.translations + files = sorted(src.glob("chapter_*_translated*.json")) + if not files: + sys.exit("в %s нет chapter_*_translated*.json" % src) + args.output.mkdir(parents=True, exist_ok=True) + stripped = empty = 0 + for f in files: + data = json.loads(f.read_text(encoding="utf-8")) + paragraphs = [] + for p in data.get("paragraphs", []): + stripped += len(MARK.findall(p)) + c = clean(p) + if c: + paragraphs.append(c) + else: + empty += 1 + data["paragraphs"] = paragraphs + if "title" in data: + data["title"] = clean(data["title"]) + (args.output / f.name).write_text( + json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8") + print("глав: %d, снято маркеров: %d, пустых абзацев отброшено: %d" + % (len(files), stripped, empty)) + print("каталог для озвучки: %s" % args.output) + print("запускать: 05_create_audiobook.py --translations-dir %s" % args.output) + + +def duration(path): + """Длительность mp3 в секундах; None если файл битый или пустой.""" + if path.stat().st_size == 0: + return None + r = subprocess.run( + ["ffprobe", "-v", "error", "-show_entries", "format=duration", + "-of", "default=nw=1:nk=1", str(path)], + capture_output=True, text=True) + try: + return float(r.stdout.strip()) + except ValueError: + return None + + +def expected_fragments(translations): + """{номер главы: число абзацев} по подготовленным для TTS файлам.""" + out = {} + for f in sorted(translations.glob("chapter_*_translated*.json")): + m = re.search(r"chapter_(\d+)", f.name) + data = json.loads(f.read_text(encoding="utf-8")) + paragraphs = [p for p in data.get("paragraphs", []) if p.strip()] + out[int(m.group(1))] = { + "paragraphs": len(paragraphs), + "chars": sum(len(p) for p in paragraphs), + } + return out + + +def cmd_verify(args): + expect = expected_fragments(args.translations) + if not expect: + sys.exit("в %s нет подготовленных глав" % args.translations) + temp = args.audiobook / "temp_audio" + total_chars = sum(v["chars"] for v in expect.values()) + + if not temp.is_dir(): + print("temp_audio/ уже убран — доступна только проверка итогового файла") + return check_final(args.audiobook, total_chars, None) + + found, bad = {}, [] + for mp3 in temp.glob("*.mp3"): + m = FRAG.match(mp3.name) + if not m: + continue + d = duration(mp3) + if d is None or d <= 0: + bad.append(mp3.name) + continue + found.setdefault(int(m.group(1)), []).append(d) + + # секунд на знак — калибруем по самим главам, чтобы не зависеть от + # голоса и скорости речи + rates = [sum(found[n]) / expect[n]["chars"] + for n in found if n in expect and expect[n]["chars"]] + rate = statistics.median(rates) if rates else 0.0 + + problems = 0 + for n in sorted(expect): + want = expect[n]["paragraphs"] + 1 # +1 вводный фрагмент + got = len(found.get(n, [])) + secs = sum(found.get(n, [])) + pred = expect[n]["chars"] * rate + note = [] + if got < want: + note.append("не хватает %d фрагментов" % (want - got)) + if pred and secs < pred * SHORT: + note.append("короче расчётной на %.0f%%" % (100 * (1 - secs / pred))) + if pred and secs > pred * LONG: + note.append("длиннее расчётной в %.1f раза" % (secs / pred)) + if note: + problems += 1 + print("глава %03d: %s (%d/%d фрагментов, %.1f мин)" + % (n, "; ".join(note), got, want, secs / 60)) + + total = sum(sum(v) for v in found.values()) + print("\nфрагментов: %d из %d, звучание %.1f ч, темп %.1f знака/с" + % (sum(len(v) for v in found.values()), + sum(v["paragraphs"] + 1 for v in expect.values()), + total / 3600, (1 / rate) if rate else 0)) + if bad: + print("битых или пустых файлов: %d (удали их и перезапусти синтез — " + "сторонний скрипт пропускает только непустые)" % len(bad)) + for name in bad[:10]: + print(" ", name) + if problems: + print("глав с замечаниями: %d — склеивать рано" % problems) + else: + print("пропусков не найдено") + + ok_final = check_final(args.audiobook, total_chars, rate) + return not problems and not bad and ok_final + + +def check_final(audiobook, total_chars, rate): + final = audiobook / "audiobook_complete.mp3" + if not final.exists(): + print("итоговый audiobook_complete.mp3 ещё не собран") + return True + d = duration(final) + if d is None: + print("итоговый файл битый") + return False + mb = final.stat().st_size / 1024 / 1024 + print("итоговый файл: %.1f ч, %.0f МБ" % (d / 3600, mb)) + if rate: + pred = total_chars * rate + if d < pred * SHORT: + print("ВНИМАНИЕ: итог короче суммы глав на %.0f%% — потеряно при склейке" + % (100 * (1 - d / pred))) + return False + return True + + +def main(): + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + sub = ap.add_subparsers(dest="cmd", required=True) + + p = sub.add_parser("prep", help="снять маркеры разметки для TTS") + p.add_argument("translations", type=Path, help="каталог translations/ после перевода") + p.add_argument("output", type=Path, help="куда положить копию для озвучки") + p.set_defaults(func=cmd_prep) + + v = sub.add_parser("verify", help="проверить целостность синтеза") + v.add_argument("translations", type=Path, help="каталог, поданный в озвучку") + v.add_argument("audiobook", type=Path, help="каталог audiobook/") + v.set_defaults(func=cmd_verify) + + args = ap.parse_args() + result = args.func(args) + if result is False: + sys.exit(1) + + +if __name__ == "__main__": + main() diff --git a/scripts/test_audiobook.py b/scripts/test_audiobook.py new file mode 100644 index 0000000..272b6ce --- /dev/null +++ b/scripts/test_audiobook.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +"""Самопроверка обвязки озвучки: python3 test_audiobook.py (нужен ffmpeg)""" +import json +import subprocess +import sys +import tempfile +from pathlib import Path + +import audiobook as a + + +def mp3(path, seconds): + """Тишина заданной длительности — реальный mp3 для ffprobe.""" + subprocess.run( + ["ffmpeg", "-v", "error", "-y", "-f", "lavfi", + "-i", "anullsrc=r=22050:cl=mono", "-t", str(seconds), + "-b:a", "32k", str(path)], + check=True) + + +def write_chapter(d, num, paragraphs): + (d / ("chapter_%03d_translated.json" % num)).write_text( + json.dumps({"number": num, "title": "Глава ⟦b⟧%d⟦/b⟧" % num, + "paragraphs": paragraphs}, ensure_ascii=False), + encoding="utf-8") + + +def test_prep(tmp: Path): + src, out = tmp / "translations", tmp / "tts" + src.mkdir() + write_chapter(src, 0, ["Он сказал ⟦i⟧привет⟧⟦/i⟧ громко.", + "Обычный абзац.", " "]) + ns = type("ns", (), {"translations": src, "output": out}) + a.cmd_prep(ns) + data = json.loads((out / "chapter_000_translated.json").read_text(encoding="utf-8")) + assert "⟦" not in json.dumps(data, ensure_ascii=False), "маркеры должны исчезнуть" + assert data["title"] == "Глава 0" + assert len(data["paragraphs"]) == 2, "пустой абзац должен отброситься" + assert " " not in data["paragraphs"][1], "двойные пробелы должны схлопнуться" + + +def test_verify_detects_gap(tmp: Path): + tts, ab = tmp / "tts", tmp / "audiobook" + tts.mkdir() + temp = ab / "temp_audio" + temp.mkdir(parents=True) + + # две главы по 4 абзаца одинаковой длины + para = "х" * 100 + for n in (0, 1): + write_chapter(tts, n, [para] * 4) + mp3(temp / ("chapter_%03d_intro.mp3" % n), 1) + # у главы 1 намеренно пропущен последний абзац + count = 4 if n == 0 else 3 + for i in range(count): + mp3(temp / ("chapter_%03d_para_%04d.mp3" % (n, i)), 6) + + ns = type("ns", (), {"translations": tts, "audiobook": ab}) + assert a.cmd_verify(ns) is False, "пропущенный фрагмент должен завалить проверку" + + +def test_verify_clean(tmp: Path): + tts, ab = tmp / "tts", tmp / "audiobook" + tts.mkdir() + temp = ab / "temp_audio" + temp.mkdir(parents=True) + para = "х" * 100 + for n in (0, 1): + write_chapter(tts, n, [para] * 3) + mp3(temp / ("chapter_%03d_intro.mp3" % n), 1) + for i in range(3): + mp3(temp / ("chapter_%03d_para_%04d.mp3" % (n, i)), 6) + ns = type("ns", (), {"translations": tts, "audiobook": ab}) + assert a.cmd_verify(ns) is True, "целая книга должна проходить проверку" + + +def test_verify_detects_empty_file(tmp: Path): + tts, ab = tmp / "tts", tmp / "audiobook" + tts.mkdir() + temp = ab / "temp_audio" + temp.mkdir(parents=True) + para = "х" * 100 + write_chapter(tts, 0, [para] * 3) + mp3(temp / "chapter_000_intro.mp3", 1) + mp3(temp / "chapter_000_para_0000.mp3", 6) + mp3(temp / "chapter_000_para_0001.mp3", 6) + (temp / "chapter_000_para_0002.mp3").touch() # нулевой размер + ns = type("ns", (), {"translations": tts, "audiobook": ab}) + assert a.cmd_verify(ns) is False, "пустой файл должен завалить проверку" + + +if __name__ == "__main__": + if subprocess.run(["which", "ffmpeg"], capture_output=True).returncode: + sys.exit("нужен ffmpeg") + for fn in (test_prep, test_verify_detects_gap, test_verify_clean, + test_verify_detects_empty_file): + with tempfile.TemporaryDirectory() as d: + print("---", fn.__name__) + fn(Path(d)) + print("OK")