From 9c7658df5eb8444dd6522744358c03ac9a2227dc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=D0=98=D0=BB=D1=8C=D1=8F=20=D0=9F=D0=BE=D0=BB=D1=8F=D0=BA?= =?UTF-8?q?=D0=BE=D0=B2?= Date: Mon, 3 Aug 2026 07:59:12 +0300 Subject: [PATCH] =?UTF-8?q?=D0=A1=D1=82=D0=B0=D0=B4=D0=B8=D1=8F=20=D0=BF?= =?UTF-8?q?=D0=B5=D1=80=D0=B5=D0=B2=D0=BE=D0=B4=D0=B0:=20=D0=BC=D0=BE?= =?UTF-8?q?=D1=81=D1=82=20=D0=BA=20book=5Ftranslator=20=D0=B1=D0=B5=D0=B7?= =?UTF-8?q?=20=D0=BC=D0=BE=D0=B4=D0=B8=D1=84=D0=B8=D0=BA=D0=B0=D1=86=D0=B8?= =?UTF-8?q?=D0=B8=20=D1=81=D1=82=D0=BE=D1=80=D0=BE=D0=BD=D0=BD=D0=B5=D0=B3?= =?UTF-8?q?=D0=BE=20=D1=80=D0=B5=D0=BF=D0=BE=D0=B7=D0=B8=D1=82=D0=BE=D1=80?= =?UTF-8?q?=D0=B8=D1=8F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- README.md | 50 ++++++++++ SKILL.md | 49 +++++++++- scripts/test_translate.py | 82 ++++++++++++++++ scripts/translate.py | 191 ++++++++++++++++++++++++++++++++++++++ 4 files changed, 371 insertions(+), 1 deletion(-) create mode 100644 scripts/test_translate.py create mode 100644 scripts/translate.py diff --git a/README.md b/README.md index d5f5f44..f915356 100644 --- a/README.md +++ b/README.md @@ -98,6 +98,56 @@ ebook-convert "Название.epub" "Название.azw3" - `.epub` — через Send to Kindle (веб или почта), Amazon конвертирует на своей стороне; - `.azw3` — по USB прямо в `documents/` на устройстве. +## Перевод (опционально) + +Перевод делает сторонний проект +[`vetermanve/book_translator`](https://github.com/vetermanve/book_translator) +(DeepSeek или локальная Ollama). **Репозиторий не модифицируется** — он +вызывается через свой штатный CLI и форматы файлов. `scripts/translate.py` — +мост: раскладывает наш XHTML в его входной JSON, запускает +`03_translate_parallel.py --all`, собирает результат обратно. + +Подготовка (один раз): + +```bash +git clone https://github.com/vetermanve/book_translator.git ~/Projects/book_translator +~/.venvs/pdf2epub/bin/pip install openai python-dotenv # openai нет в их requirements.txt +export DEEPSEEK_API_KEY=... # либо .env в рабочем каталоге, либо USE_OLLAMA +``` + +Сначала проверка, нужен ли перевод вообще (язык, объём; ничего не тратит): + +```bash +~/.venvs/pdf2epub/bin/python scripts/translate.py out/book.html out/book.ru.html \ + --repo ~/Projects/book_translator --workdir out/tr --check-only +``` + +Если книга уже на целевом языке, скрипт откажется переводить (`--force` +переопределяет). Сам перевод: + +```bash +~/.venvs/pdf2epub/bin/python scripts/translate.py out/book.html out/book.ru.html \ + --repo ~/Projects/book_translator --workdir out/tr --workers 12 +``` + +Прерванный перевод возобновляется: сторонний скрипт помнит готовые главы. +Дальше `out/book.ru.html` упаковывается тем же `ebook-convert` с +`--language=ru`. + +Курсив и полужирный переживают внешний переводчик в виде маркеров +`⟦i⟧…⟦/i⟧`; на обратном пути баланс маркеров проверяется по каждому абзацу. +Картинки, заголовки и порядок блоков вообще не покидают наш скрипт — наружу +уходит только текст, сборка идёт по индексам. В отчёте после перевода: +«без перевода» — главы, вернувшиеся с другим числом абзацев (оставлен +оригинал), «разметка потеряна» — абзацы, где модель испортила маркеры и +курсив снят. Оба числа должны быть близки к нулю. + +Самопроверка моста без сети и ключей: + +```bash +cd scripts && python3 test_translate.py +``` + ## Проверка результата ```bash diff --git a/SKILL.md b/SKILL.md index 09aec4e..b6d6fcd 100644 --- a/SKILL.md +++ b/SKILL.md @@ -1,6 +1,6 @@ --- name: pdf-to-kindle -description: Convert a text-layer PDF (usually one generated by calibre from an ebook) into EPUB/AZW3 for Kindle while preserving italics, bold, chapter headings, sidebars/journal blocks, images and cross-page paragraphs. Use when the user asks to read a PDF book on Kindle or another e-reader, to convert a PDF to EPUB/AZW3/MOBI, or complains that a converted book lost its italics, merged its chapters, or reflows badly on the device. Not for scanned PDFs (no text layer) — those need OCR first. +description: Convert a text-layer PDF (usually one generated by calibre from an ebook) into EPUB/AZW3 for Kindle while preserving italics, bold, chapter headings, sidebars/journal blocks, images and cross-page paragraphs. Use when the user asks to read a PDF book on Kindle or another e-reader, to convert a PDF to EPUB/AZW3/MOBI, or complains that a converted book lost its italics, merged its chapters, or reflows badly on the device. Not for scanned PDFs (no text layer) — those need OCR first. Also covers translating such a book into another language before packing it, delegated to the external book_translator project. --- # PDF to Kindle @@ -92,6 +92,53 @@ EOF After packing, confirm the EPUB: `ebook-meta` for metadata, and the `.ncx` navPoint count for the TOC size. +## Optional stage: translation + +Only when the user asks for a translated book. It slots between step 6 and +step 7 — translate the XHTML, then pack the translated file with calibre. + +Translation is delegated to an external project, +[`vetermanve/book_translator`](https://github.com/vetermanve/book_translator) +(DeepSeek or a local Ollama model). **Never modify that repository** — it is +driven through its documented CLI and file formats only. `scripts/translate.py` +is the bridge: it writes the repo's input format, shells out to +`03_translate_parallel.py --all`, and reads the repo's output format back. + +1. **Check the book actually needs translating.** The script does this first + and refuses to burn API credits on a book already in the target language: + + `python scripts/translate.py book.html out.html --repo --workdir --check-only` + + It reports block/chapter/character counts and the detected source language. + `--force` overrides the refusal. +2. **Set up the external repo once** (a plain clone; never edit it), and its + credentials — `DEEPSEEK_API_KEY` in the environment or a `.env` in the + working directory, or `USE_OLLAMA` for a local model. Its `requirements.txt` + omits `openai`, which `deepseek_translator.py` imports — install it too. +3. **Translate:** + + `python scripts/translate.py book.html book.ru.html --repo --workdir --workers 12` + + Resumable: the external repo tracks completed chapters and skips them on a + re-run, so an interrupted run costs nothing to restart. +4. **Read the reported counts.** "без перевода" above zero means a chapter came + back with a different paragraph count and kept its original text; "разметка + потеряна" counts paragraphs where the model mangled the inline-tag markers + and the italics were dropped rather than corrupted. Both are expected to be + near zero — a large number means the translator misbehaved, not that the + bridge is broken. +5. Pack `book.ru.html` with `--language=ru` and translated `--title`/`--authors`. + +How formatting survives a translator that only speaks plain text: inline `` +and `` become `⟦i⟧…⟦/i⟧` markers before the text leaves, and are restored +after. Every paragraph's markers are balance-checked on the way back. Images, +`

`/`

` structure, and block order never leave this side — only the text +of each block round-trips, and blocks are reassembled by index. + +`scripts/test_translate.py` covers the marker round-trip, the broken-marker +fallback, language detection, and a full split/rebuild against a faked +translator response — no network, no API key. Run it after touching the bridge. + ## What the script does - style from span font names (`-It`, `Italic`, `Bold`, `Semibold`) → ``/``; diff --git a/scripts/test_translate.py b/scripts/test_translate.py new file mode 100644 index 0000000..11df63f --- /dev/null +++ b/scripts/test_translate.py @@ -0,0 +1,82 @@ +#!/usr/bin/env python3 +"""Самопроверка моста без сети: python3 test_translate.py""" +import json +from pathlib import Path + +import translate as t + + +def test_marks_roundtrip(): + src = 'She thinks this is bad & says “no”' + marked = t.to_marks(src) + assert "" not in marked and "⟦i⟧" in marked + assert "&" not in marked, "сущности должны раскрываться для переводчика" + body, ok = t.from_marks(marked) + assert ok + assert "this is bad" in body + assert "&" in body, "спецсимволы должны экранироваться обратно" + + +def test_broken_marks_drop_tags(): + body, ok = t.from_marks("текст ⟦i⟧без закрытия") + assert not ok and "⟦" not in body and "" not in body + + +def test_language_detection(): + assert t.detect_language("the quick brown fox " * 40)[0] == "en" + assert t.detect_language("быстрая бурая лиса " * 40)[0] == "ru" + assert t.detect_language("short")[0] == "unknown" + + +def test_split_and_rebuild(tmp: Path): + src = tmp / "book.html" + src.write_text("\n".join([ + "", + "

CHAPTER 1

", + "

He said hello loudly.

", + '

', + "

CHAPTER 2

", + "

Second chapter text.

", + "", + ]), encoding="utf-8") + + blocks = t.parse_blocks(src) + assert len(blocks) == 4, "картинка не должна попадать в перевод" + chapters = t.split_chapters(blocks) + assert [len(c) for c in chapters] == [2, 2] + + extracted = tmp / "extracted" + t.write_input(chapters, extracted) + ch0 = json.loads((extracted / "chapter_000.json").read_text(encoding="utf-8")) + assert ch0["title"] == "CHAPTER 1" + assert ch0["paragraphs"][1] == "He said ⟦i⟧hello⟦/i⟧ loudly." + + # подделываем ответ book_translator: тот же формат, тот же порядок + trans = tmp / "translations" + trans.mkdir() + for n, ch in enumerate(chapters): + paragraphs = [json.loads((extracted / ("chapter_%03d.json" % n)) + .read_text(encoding="utf-8"))["paragraphs"][i] + .replace("He said", "Он сказал").replace("hello", "привет") + for i in range(len(ch))] + (trans / ("chapter_%03d_translated.json" % n)).write_text( + json.dumps({"number": n, "paragraphs": paragraphs}, ensure_ascii=False), + encoding="utf-8") + + out = tmp / "book.ru.html" + assert t.rebuild(src, chapters, tmp, out) + result = out.read_text(encoding="utf-8") + assert "

Он сказал привет loudly.

" in result + assert '' in result, "картинка должна уцелеть" + assert result.count("

") == 2 + + +if __name__ == "__main__": + import tempfile + + test_marks_roundtrip() + test_broken_marks_drop_tags() + test_language_detection() + with tempfile.TemporaryDirectory() as d: + test_split_and_rebuild(Path(d)) + print("OK") diff --git a/scripts/translate.py b/scripts/translate.py new file mode 100644 index 0000000..9a89318 --- /dev/null +++ b/scripts/translate.py @@ -0,0 +1,191 @@ +#!/usr/bin/env python3 +"""Перевод XHTML от pdf2html.py через сторонний book_translator (репозиторий не +модифицируется — используется его штатный CLI и формат файлов). + +Схема: наш XHTML -> extracted/chapter_NNN.json + metadata.json -> +03_translate_parallel.py --all -> translations/chapter_NNN_translated.json -> +наш XHTML с сохранённой разметкой. + +Курсив/полужирный переживают внешний переводчик в виде маркеров ⟦i⟧…⟦/i⟧: +модель их обычно не трогает. Каждый абзац проверяется на баланс маркеров; при +поломке абзац сохраняется переведённым, но без внутренней разметки. +""" +import argparse +import html +import json +import re +import subprocess +import sys +from pathlib import Path + +BLOCK = re.compile(r"^<(p|h1|h2)([^>]*)>(.*)$") +TAG = re.compile(r"") +MARK = re.compile(r"⟦(/?)([ib])⟧") +# картинки и пустые блоки не переводим +SKIP = re.compile(r" 0.5 else "en"), share + + +def to_marks(s): + """Инлайновые теги -> маркеры, HTML-сущности -> обычные символы.""" + return html.unescape(TAG.sub(lambda m: "⟦%s⟧" % m.group(0)[1:-1], s)) + + +def from_marks(s): + """Маркеры -> теги. Возвращает (html, ok): ok=False если разметка разъехалась.""" + escaped = html.escape(s, quote=False) + depth = {"i": 0, "b": 0} + ok = True + for slash, tag in MARK.findall(escaped): + depth[tag] += -1 if slash else 1 + if depth[tag] < 0: + ok = False + if any(v != 0 for v in depth.values()): + ok = False + if not ok: + return MARK.sub("", escaped), False + return MARK.sub(lambda m: "<%s%s>" % (m.group(1), m.group(2)), escaped), True + + +def parse_blocks(path): + """[(строка_исходника, tag, attrs, inner)] — только переводимые блоки.""" + out = [] + for i, line in enumerate(path.read_text(encoding="utf-8").splitlines()): + m = BLOCK.match(line.strip()) + if m and not SKIP.search(m.group(3)) and m.group(3).strip(): + out.append((i, m.group(1), m.group(2), m.group(3))) + return out + + +def split_chapters(blocks): + """Разбивка по h1 — их переводчик держит контекст в пределах главы.""" + chapters, cur = [], [] + for b in blocks: + if b[1] == "h1" and cur: + chapters.append(cur) + cur = [] + cur.append(b) + if cur: + chapters.append(cur) + return chapters + + +def write_input(chapters, extracted): + extracted.mkdir(parents=True, exist_ok=True) + index = [] + for n, ch in enumerate(chapters): + paragraphs = [to_marks(b[3]) for b in ch] + title = re.sub("<[^>]+>", "", ch[0][3])[:120] if ch[0][1] == "h1" else "Часть %d" % n + words = sum(len(p.split()) for p in paragraphs) + data = { + "number": n, + "title": title, + "paragraphs": paragraphs, + "paragraph_count": len(paragraphs), + "word_count": words, + "char_count": sum(len(p) for p in paragraphs), + "source_format": "xhtml", + "extraction_method": "pdf2html-bridge", + } + (extracted / ("chapter_%03d.json" % n)).write_text( + json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8") + index.append({"number": n, "title": title, + "paragraph_count": len(paragraphs), "word_count": words}) + (extracted / "metadata.json").write_text(json.dumps( + {"total_chapters": len(chapters), + "total_words": sum(c["word_count"] for c in index), + "chapters": index, "source_format": "xhtml"}, + ensure_ascii=False, indent=2), encoding="utf-8") + return index + + +def run_translator(repo, workdir, extracted, workers): + script = repo / "03_translate_parallel.py" + if not script.exists(): + sys.exit("не найден %s — укажи --repo на клон book_translator" % script) + cmd = [sys.executable, str(script), "--all", + "--workers", str(workers), "--extracted-dir", str(extracted)] + print("запуск:", " ".join(cmd), "(cwd=%s)" % workdir) + r = subprocess.run(cmd, cwd=workdir) + if r.returncode != 0: + sys.exit("book_translator завершился с кодом %d" % r.returncode) + + +def rebuild(src, chapters, workdir, out): + lines = src.read_text(encoding="utf-8").splitlines() + broken = missing = translated = 0 + for n, ch in enumerate(chapters): + f = workdir / "translations" / ("chapter_%03d_translated.json" % n) + if not f.exists(): + missing += len(ch) + continue + paragraphs = json.loads(f.read_text(encoding="utf-8"))["paragraphs"] + if len(paragraphs) != len(ch): + print("глава %d: %d абзацев вместо %d, оставлен оригинал" + % (n, len(paragraphs), len(ch))) + missing += len(ch) + continue + for (idx, tag, attrs, _), text in zip(ch, paragraphs): + body, ok = from_marks(text) + broken += not ok + translated += 1 + lines[idx] = "<%s%s>%s" % (tag, attrs, body, tag) + out.write_text("\n".join(lines) + "\n", encoding="utf-8") + print("переведено блоков: %d, без перевода: %d, разметка потеряна в %d" + % (translated, missing, broken)) + return missing == 0 + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("source", type=Path, help="XHTML от pdf2html.py") + ap.add_argument("output", type=Path, help="куда писать переведённый XHTML") + ap.add_argument("--repo", type=Path, required=True, + help="клон vetermanve/book_translator (не модифицируется)") + ap.add_argument("--workdir", type=Path, required=True, + help="рабочий каталог: extracted/ и translations/") + ap.add_argument("--workers", type=int, default=12) + ap.add_argument("--target", default="ru", help="целевой язык (по умолчанию ru)") + ap.add_argument("--force", action="store_true", + help="переводить, даже если книга уже на целевом языке") + ap.add_argument("--check-only", action="store_true", + help="только проверить язык и объём, ничего не переводить") + args = ap.parse_args() + + blocks = parse_blocks(args.source) + if not blocks: + sys.exit("в %s нет блоков

/

/

— это не вывод pdf2html.py" % args.source) + plain = " ".join(re.sub("<[^>]+>", "", b[3]) for b in blocks) + lang, share = detect_language(plain) + chapters = split_chapters(blocks) + print("блоков: %d, глав: %d, знаков: %d" % (len(blocks), len(chapters), len(plain))) + print("язык источника: %s (кириллица %.0f%%)" % (lang, share * 100)) + + if lang == args.target and not args.force: + print("книга уже на языке %s — перевод не требуется (--force чтобы всё равно)" + % args.target) + return + if lang == "unknown": + print("язык определить не удалось — продолжаю") + if args.check_only: + return + + args.workdir.mkdir(parents=True, exist_ok=True) + extracted = (args.workdir / "extracted").resolve() + write_input(chapters, extracted) + run_translator(args.repo.resolve(), args.workdir.resolve(), extracted, args.workers) + rebuild(args.source, chapters, args.workdir, args.output) + + +if __name__ == "__main__": + main()