From db9fe5e0fa87d1527cbaae659d0c0f0a0d632cc8 Mon Sep 17 00:00:00 2001 From: chesirecatt Date: Thu, 20 Aug 2026 17:58:09 +0300 Subject: [PATCH] =?UTF-8?q?EPUB=20=D0=BA=D0=B0=D0=BA=20=D0=B2=D1=82=D0=BE?= =?UTF-8?q?=D1=80=D0=BE=D0=B9=20=D0=B2=D1=85=D0=BE=D0=B4=20=D0=B2=20=D0=BA?= =?UTF-8?q?=D0=BE=D0=BD=D0=B2=D0=B5=D0=B9=D0=B5=D1=80;=20=D0=BB=D0=B8?= =?UTF-8?q?=D1=81=D1=82=D0=B8=D0=BD=D0=B3=D0=B8=20=D0=BA=D0=BE=D0=B4=D0=B0?= =?UTF-8?q?=20=D0=BD=D0=B5=20=D0=BF=D0=B5=D1=80=D0=B5=D0=B2=D0=BE=D0=B4?= =?UTF-8?q?=D1=8F=D1=82=D1=81=D1=8F=20=D0=B8=20=D0=BD=D0=B5=20=D0=BB=D0=BE?= =?UTF-8?q?=D0=BC=D0=B0=D1=8E=D1=82=20=D0=BF=D1=80=D0=B8=D1=91=D0=BC=D0=BA?= =?UTF-8?q?=D1=83?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - epub2html.py: EPUB -> тот же XHTML, что pdf2html.py, только стандартная библиотека. Уровень глав определяется по книге, а не берётся из

: в EPUB там обычно название и части. - pdf2html.py: листинг узнаётся по флагу моноширинного шрифта PyMuPDF, переносы и отступы внутри
 сохраняются, де-дефисация к коду не
  применяется.
- translate.py: 
 исключён из проверок языка (английский код утягивал
  долю кириллицы ниже порога приёмки), рабочий каталог привязан к книге,
  устаревшие главы прошлого прогона чистятся, инлайновый  переживает
  переводчика.
- bookhtml.py: общий каркас документа и CSS для обоих входов.
- Тесты: test_epub2html.py на собранном в памяти EPUB, четыре новых случая
  в test_translate.py.
---
 SKILL.md                  |  50 ++++++-
 scripts/bookhtml.py       |  44 +++++++
 scripts/epub2html.py      | 267 ++++++++++++++++++++++++++++++++++++++
 scripts/pdf2html.py       |  66 +++++-----
 scripts/test_epub2html.py | 119 +++++++++++++++++
 scripts/test_translate.py |  71 +++++++++-
 scripts/translate.py      |  40 +++++-
 7 files changed, 614 insertions(+), 43 deletions(-)
 create mode 100644 scripts/bookhtml.py
 create mode 100644 scripts/epub2html.py
 create mode 100644 scripts/test_epub2html.py

diff --git a/SKILL.md b/SKILL.md
index 6e838da..19fac81 100644
--- a/SKILL.md
+++ b/SKILL.md
@@ -15,6 +15,30 @@ tag, and 2817 body paragraphs became `

`. properties, and emits semantic XHTML. calibre is then used only as the XHTML→EPUB packer. +## Two entry points + +`scripts/pdf2html.py` for PDF, `scripts/epub2html.py` for EPUB. Both emit the +same normalized XHTML — one block per line, `
` for code listings — and
+everything downstream (translation, packing, audiobook) is identical. Shared
+document skeleton and CSS live in `scripts/bookhtml.py`.
+
+**Do not route an EPUB through the PDF path.** Converting EPUB→PDF→XHTML throws
+away the semantic markup that is already there and re-derives it from font
+sizes. `epub2html.py` needs no font profiling and no threshold tuning: it reads
+the spine, drops the nav document, and maps the book's own headings.
+
+Steps 1, 3 and 4 below are PDF-only. For EPUB start at step 2 (only calibre is
+needed, not PyMuPDF) and convert with:
+
+`python scripts/epub2html.py book.epub out/book.html`
+
+It reports which heading level turned out to be the chapter level. In most
+EPUBs `

` is the book title and the parts, while chapters are `

` — the +script picks the deepest level that still gives a sane chapter count, because +the translation bridge splits the book on `

`. Measured on Stroustrup's PPP +3rd edition: 7 `

` against 49 `

`, chapters correctly detected as `h2`, +56 chapters out of 656 pages. + ## Workflow 1. **Verify a text layer exists.** `pdfinfo` and `pdffonts` on the file. No @@ -35,7 +59,9 @@ XHTML→EPUB packer. Read off: the body font (largest character count), the italic variant, the heading sizes, and any secondary family used for sidebars, journal entries, chat logs or slides. -4. **Tune the thresholds** in `block_kind()` and `style()` to that output. The +4. **Tune the thresholds** in `block_kind()` and `style()` to that output. Code + listings need no tuning — a monospaced span is recognized by the PyMuPDF font + flag (`flags & 8`), not by font name, and becomes a `
` block. The
    defaults target a 15pt/letter calibre layout: `>= 28` or a display font is a
    title, `>= 24` a chapter/part `h1`, `>= 17` an `h2` subtitle, a secondary
    family is `p.note`. `INDENT_X` (default 88) is the x-coordinate that
@@ -45,7 +71,10 @@ XHTML→EPUB packer.
 
    `python scripts/pdf2html.py book.pdf out/book.html`
 
-6. **Verify before packing** (see Verification). Fix thresholds and re-run until
+6. **Verify before packing** (see Verification). Check the `pre:` counter against
+   the real number of listings in the book — a technical book reporting `pre: 0`
+   means the mono flag never fired and every listing is about to be reflowed as
+   prose. Fix thresholds and re-run until
    the counts are sane. Cheap to iterate; do not skip to packing.
 7. **Pack with calibre**, always passing explicit TOC XPaths — without them
    calibre applies its own heuristics and re-breaks the chapters:
@@ -97,6 +126,23 @@ navPoint count for the TOC size.
 Only when the user asks for a translated book. It slots between step 6 and
 step 7 — translate the XHTML, then pack the translated file with calibre.
 
+**Code listings are never translated.** The bridge only recognizes `

`, `

` +and `

`; a `
` block is not parsed, and anything unparsed is carried into
+the result untouched. Nothing extra is needed to protect code — but this also
+means a listing that was misclassified as a paragraph upstream *will* be
+translated, which is the real reason step 6 checks the `pre:` counter.
+
+For the same reason `
` is cut out of both language checks. A book that is
+40% listings translates correctly and would otherwise fail acceptance, because
+the English code drags the Cyrillic share below the threshold.
+
+**One workdir per book.** Chapter files are named `chapter_NNN.json` for every
+book, and the external repo skips chapters it has marked done — a reused workdir
+silently stitches one book's translation onto another's text. The bridge writes
+`bridge_source.json` into the workdir on first run and refuses to start if the
+directory belongs to a different book. Re-running the same book is unaffected;
+that is the resume path.
+
 Translation is delegated to an external project,
 [`vetermanve/book_translator`](https://github.com/vetermanve/book_translator)
 (DeepSeek or a local Ollama model). **Never modify that repository** — it is
diff --git a/scripts/bookhtml.py b/scripts/bookhtml.py
new file mode 100644
index 0000000..a178367
--- /dev/null
+++ b/scripts/bookhtml.py
@@ -0,0 +1,44 @@
+#!/usr/bin/env python3
+"""Общая сборка XHTML для pdf2html.py и epub2html.py.
+
+Формат намеренно жёсткий: один блок — одна строка (кроме 
, где переносы
+значимы). На этом держится мост translate.py — он правит блоки по номеру
+строки, а всё, чего не узнал, доносит до результата нетронутым.
+"""
+import html
+
+CSS = """
+body { margin: 0 1em; }
+h1 { text-align: center; margin: 2em 0 0.2em; page-break-before: always; }
+h1.title { page-break-before: avoid; }
+h2 { text-align: center; font-style: italic; font-weight: normal;
+     font-size: 1.1em; margin: 0.2em 0 1.5em; }
+p { text-indent: 1.2em; margin: 0; text-align: justify; }
+p.note { text-indent: 0; margin: 1em 2em; font-size: 0.9em;
+         font-family: sans-serif; }
+p.li { text-indent: 0; margin: 0.3em 0 0.3em 1.5em; text-align: left; }
+p.row { text-indent: 0; margin: 0.3em 0; text-align: left;
+        font-size: 0.9em; }
+p.figure { text-indent: 0; text-align: center; margin: 1em 0; }
+img { max-width: 100%; }
+pre { font-family: monospace; font-size: 0.75em; margin: 1em 0;
+      white-space: pre-wrap; word-wrap: break-word; text-align: left;
+      text-indent: 0; }
+code { font-family: monospace; font-size: 0.9em; }
+"""
+
+
+def document(title, parts):
+    """parts: [(tag, cls, inner_html)]; tag 'figure' — псевдотег для картинки."""
+    buf = ['',
+           '',
+           "%s" % html.escape(title),
+           "" % CSS]
+    for tag, cls, txt in parts:
+        if tag == "figure":
+            buf.append('

%s

' % txt) + else: + c = ' class="%s"' % cls if cls else "" + buf.append("<%s%s>%s" % (tag, c, txt, tag)) + buf.append("") + return "\n".join(buf) diff --git a/scripts/epub2html.py b/scripts/epub2html.py new file mode 100644 index 0000000..37ccef4 --- /dev/null +++ b/scripts/epub2html.py @@ -0,0 +1,267 @@ +#!/usr/bin/env python3 +"""EPUB -> тот же XHTML, что выдаёт pdf2html.py (один блок = одна строка). + +EPUB уже размечен семантически, поэтому здесь не разведка шрифтов, а +нормализация: привести чужую вёрстку к формату, который понимает мост +translate.py, и вычистить всё, что переводчику видеть не нужно. + +Листинги остаются в
 и не переводятся вовсе: BLOCK-регулярка моста их не
+узнаёт, а всё неузнанное доходит до результата нетронутым.
+
+Только стандартная библиотека — ни calibre, ни lxml здесь не нужны.
+"""
+import argparse
+import collections
+import html
+import posixpath
+import re
+import sys
+import zipfile
+from html.parser import HTMLParser
+from pathlib import Path
+from xml.etree import ElementTree
+
+import bookhtml
+
+CONTAINER = "META-INF/container.xml"
+OPF_NS = "{http://www.idpf.org/2007/opf}"
+CONT_NS = "{urn:oasis:names:tc:opendocument:xmlns:container}"
+
+# tag источника -> (tag результата, css-класс)
+BLOCK_MAP = {
+    "h1": ("h1", None), "h2": ("h2", None), "h3": ("h3", None),
+    "h4": ("h4", None), "h5": ("h5", None), "h6": ("h6", None),
+    "p": ("p", None), "pre": ("pre", None), "li": ("p", "li"),
+    "tr": ("p", "row"), "figcaption": ("p", "note"),
+    "dt": ("p", "note"), "dd": ("p", "note"),
+}
+INLINE_MAP = {"i": "i", "em": "i", "cite": "i", "b": "b", "strong": "b",
+              "code": "code", "kbd": "code", "samp": "code", "tt": "code",
+              "var": "code"}
+DROP = {"script", "style", "head", "title", "nav"}
+IMG_EXT = {"image/jpeg": ".jpg", "image/png": ".png", "image/gif": ".gif",
+           "image/svg+xml": ".svg", "image/webp": ".webp"}
+
+
+class Reader(HTMLParser):
+    """Собирает блоки из одного XHTML-документа книги."""
+
+    def __init__(self, on_image):
+        super().__init__(convert_charrefs=True)
+        self.on_image = on_image
+        self.blocks = []
+        self.cur = None          # (tag, cls, [куски])
+        self.open_inline = []    # незакрытые инлайновые теги текущего блока
+        self.drop = 0
+
+    # -- служебное ---------------------------------------------------------
+    def flush(self):
+        if not self.cur:
+            return
+        tag, cls, parts = self.cur
+        self.cur = None
+        for t in reversed(self.open_inline):
+            parts.append("" % t)   # чужая вёрстка бывает несбалансированной
+        self.open_inline = []
+        txt = "".join(parts)
+        if tag == "pre":
+            txt = txt.strip("\n").rstrip()
+        else:
+            txt = re.sub(r"\s+", " ", txt).strip()
+        txt = re.sub(r"<(i|b|code)>(\s*)", r"\2", txt)  # пустая разметка
+        if txt.strip():
+            self.blocks.append((tag, cls, txt))
+
+    def start(self, tag, cls):
+        self.flush()
+        self.cur = (tag, cls, [])
+
+    # -- HTMLParser --------------------------------------------------------
+    def handle_starttag(self, tag, attrs):
+        if tag in DROP:
+            self.drop += 1
+            return
+        if self.drop:
+            return
+        a = dict(attrs)
+        if tag in ("img", "image"):
+            src = a.get("src") or a.get("{http://www.w3.org/1999/xlink}href") \
+                or a.get("xlink:href") or a.get("href")
+            name = self.on_image(src) if src else None
+            if name:
+                self.flush()
+                self.blocks.append(("figure", None,
+                                    '' % name))
+            return
+        if tag == "br":
+            if self.cur:
+                self.cur[2].append("\n" if self.cur[0] == "pre" else " ")
+            return
+        if tag in BLOCK_MAP:
+            self.start(*BLOCK_MAP[tag])
+            return
+        if tag in ("td", "th") and self.cur and self.cur[0] == "p" \
+                and self.cur[1] == "row" and self.cur[2]:
+            self.cur[2].append(" | ")
+            return
+        if tag in INLINE_MAP and self.cur and self.cur[0] != "pre":
+            # Внутри листинга инлайн не нужен: книги размечают код сплошным
+            # , и на e-ink это страница жирного текста.
+            out = INLINE_MAP[tag]
+            self.open_inline.append(out)
+            self.cur[2].append("<%s>" % out)
+
+    def handle_endtag(self, tag):
+        if tag in DROP:
+            self.drop = max(0, self.drop - 1)
+            return
+        if self.drop:
+            return
+        if tag in BLOCK_MAP:
+            self.flush()
+            return
+        if tag in INLINE_MAP and self.cur and self.cur[0] != "pre":
+            out = INLINE_MAP[tag]
+            if out in self.open_inline:
+                self.open_inline.remove(out)
+                self.cur[2].append("" % out)
+
+    def handle_data(self, data):
+        if self.drop:
+            return
+        if not self.cur:
+            if not data.strip():
+                return
+            self.start("p", None)   # текст вне блочного тега не теряем
+        self.cur[2].append(html.escape(data, quote=False))
+
+    def close(self):
+        super().close()
+        self.flush()
+
+
+def spine_documents(zf):
+    """[(путь в архиве, media-type)] в порядке чтения + путь к OPF."""
+    root = ElementTree.fromstring(zf.read(CONTAINER))
+    opf_path = root.find(".//%srootfile" % CONT_NS).get("full-path")
+    opf = ElementTree.fromstring(zf.read(opf_path))
+    base = posixpath.dirname(opf_path)
+    items, cover = {}, None
+    for it in opf.iter("%sitem" % OPF_NS):
+        href = posixpath.normpath(posixpath.join(base, it.get("href")))
+        props = it.get("properties") or ""
+        items[it.get("id")] = (href, it.get("media-type"), props)
+        if "cover-image" in props:
+            cover = href
+    if cover is None:  # EPUB 2: обложка объявляется через 
+        meta = opf.find(".//%smeta[@name='cover']" % OPF_NS)
+        if meta is not None and meta.get("content") in items:
+            cover = items[meta.get("content")][0]
+    docs = []
+    for ref in opf.iter("%sitemref" % OPF_NS):
+        item = items.get(ref.get("idref"))
+        if not item:
+            continue
+        href, mtype, props = item
+        if "nav" in props or re.search(r"(toc|nav|content)s?\.x?html?$", href, re.I):
+            continue
+        docs.append(href)
+    return docs, items, cover
+
+
+# Больше этого числа глав дробить незачем: переводчик держит контекст в пределах
+# главы, слишком мелкая нарезка его обесценивает.
+MAX_CHAPTERS = 150
+
+
+def chapter_level(parts):
+    """Каким уровнем заголовка в этой книге размечены главы.
+
+    Мост режет книгу на главы по 

, а в EPUB

обычно занят названием + книги и частями: у Страуструпа их 7 на 656 страниц, тогда как главы — это +

(49 штук). Берём самый глубокий уровень, при котором число глав ещё + остаётся вменяемым. + """ + counts = collections.Counter(t for t, _, _ in parts if re.fullmatch(r"h[1-6]", t)) + best, total = 1, 0 + for lvl in range(1, 7): + total += counts.get("h%d" % lvl, 0) + if total > MAX_CHAPTERS: + break + if total: + best = lvl + return best + + +def convert(src, out): + imgdir = out.parent / "images" + imgdir.mkdir(parents=True, exist_ok=True) + parts, seen, counter = [], {}, [0] + + with zipfile.ZipFile(src) as zf: + names = set(zf.namelist()) + docs, items, cover = spine_documents(zf) + + def extract(path, stem=None): + if path in seen: + return seen[path] + if path not in names: + return None + ext = posixpath.splitext(path)[1] or ".img" + if stem: + name = stem + ext + else: + counter[0] += 1 + name = "img%03d%s" % (counter[0], ext) + (imgdir / name).write_bytes(zf.read(path)) + seen[path] = name + return name + + if cover: + extract(cover, stem="cover") + + for doc in docs: + if doc not in names: + print("нет в архиве, пропущен: %s" % doc, file=sys.stderr) + continue + here = posixpath.dirname(doc) + + def on_image(src_attr, here=here): + target = posixpath.normpath(posixpath.join(here, src_attr.split("#")[0])) + return extract(target) + + r = Reader(on_image) + r.feed(zf.read(doc).decode("utf-8", "replace")) + r.close() + parts.extend(r.blocks) + + title = out.stem + lvl = chapter_level(parts) + parts = [(("h1" if int(t[1]) <= lvl else "h2", c, x) + if re.fullmatch(r"h[1-6]", t) else (t, c, x)) + for t, c, x in parts] + h1 = sum(1 for p in parts if p[0] == "h1") + print("главы размечены h%d, глав получилось: %d" % (lvl, h1)) + if h1 < 3: + print("ВНИМАНИЕ: заголовков-глав всего %d — переводчик будет держать " + "контекст крупными кусками, проверь результат внимательнее" % h1) + + out.write_text(bookhtml.document(title, parts), encoding="utf-8") + print("blocks: %d, images: %d" % (len(parts), len(seen))) + print("h1: %d p: %d pre: %d" + % (h1, sum(1 for p in parts if p[0] == "p"), + sum(1 for p in parts if p[0] == "pre"))) + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("source", type=Path, help="исходный .epub") + ap.add_argument("output", type=Path, help="куда писать XHTML") + args = ap.parse_args() + if not args.source.exists(): + sys.exit("нет файла %s" % args.source) + convert(args.source, args.output) + + +if __name__ == "__main__": + main() diff --git a/scripts/pdf2html.py b/scripts/pdf2html.py index 850eefa..f1d1417 100644 --- a/scripts/pdf2html.py +++ b/scripts/pdf2html.py @@ -7,6 +7,11 @@ from pathlib import Path import pymupdf +import bookhtml + +if len(sys.argv) < 3 or sys.argv[1] in ("-h", "--help"): + sys.exit("usage: pdf2html.py book.pdf out.html | pdf2html.py --fonts book.pdf") + if sys.argv[1] == "--fonts": # разведка: какие шрифты/кегли в PDF import collections @@ -29,6 +34,7 @@ IMGDIR.mkdir(parents=True, exist_ok=True) # Пороги подобраны под вёрстку 15pt/letter. Для другой книги сначала # посмотреть реальные шрифты и кегли: pdf2html.py --fonts file.pdf INDENT_X = 88 # x0 первой строки: больше — абзац с красной строки +MONO = 8 # бит моноширинного шрифта в span["flags"] (PyMuPDF) def style(font): @@ -36,10 +42,20 @@ def style(font): return ("bold" in f or "semibold" in f, "-it" in f or "italic" in f) -def span_html(s): +def is_mono(s): + """Моноширинный шрифт = листинг кода. Признак берётся из флагов PyMuPDF, а + не из имени шрифта: имена у каждого издательства свои, флаг одинаковый.""" + return bool(s["flags"] & MONO) + + +def span_html(s, plain=False): t = html.escape(s["text"]) if not t: return "" + if plain: # внутри
 курсив и полужирный только мешают
+        return t
+    if is_mono(s):
+        return "%s" % t
     bold, ital = style(s["font"])
     if ital:
         t = "%s" % t
@@ -59,6 +75,8 @@ def block_kind(b):
         return ("h1", "title")
     if sz >= 24:
         return ("h1", None)
+    if all(is_mono(sp) for sp in spans):
+        return ("pre", None)
     if sz >= 17:
         return ("h2", None)
     if f.startswith("MyriadPro"):
@@ -66,7 +84,13 @@ def block_kind(b):
     return ("p", None)
 
 
-def block_text(b):
+def block_text(b, pre=False):
+    if pre:
+        # Перенос строки в коде значим, висящий дефис — это минус, а не перенос
+        # слова. Ни склейки строк, ни де-дефисации здесь быть не должно.
+        rows = ["".join(span_html(s, plain=True) for s in l["spans"])
+                for l in b["lines"]]
+        return "\n".join(rows).rstrip()
     out = []
     for i, l in enumerate(b["lines"]):
         line = "".join(span_html(s) for s in l["spans"])
@@ -82,6 +106,7 @@ def block_text(b):
     txt = "".join(out)
     txt = re.sub(r"(\s*)", r"\1", txt)
     txt = re.sub(r"(\s*)", r"\1", txt)
+    txt = re.sub(r"(?<=\S) {2,}(?=\S)", " ", txt)
     return txt.strip()
 
 
@@ -112,7 +137,7 @@ for pno, page in enumerate(doc):
         if not kind:
             continue
         tag, cls = kind
-        txt = block_text(b)
+        txt = block_text(b, pre=(tag == "pre"))
         if not txt:
             continue
         indented = b["lines"][0]["bbox"][0] >= INDENT_X
@@ -134,36 +159,13 @@ for pno, page in enumerate(doc):
 if open_para:
     parts.append(open_para)
 
-CSS = """
-body { margin: 0 1em; }
-h1 { text-align: center; margin: 2em 0 0.2em; page-break-before: always; }
-h1.title { page-break-before: avoid; }
-h2 { text-align: center; font-style: italic; font-weight: normal;
-     font-size: 1.1em; margin: 0.2em 0 1.5em; }
-p { text-indent: 1.2em; margin: 0; text-align: justify; }
-p.note { text-indent: 0; margin: 1em 2em; font-size: 0.9em;
-         font-family: sans-serif; }
-p.figure { text-indent: 0; text-align: center; margin: 1em 0; }
-img { max-width: 100%; }
-"""
-
-buf = ['',
-       '',
-       "%s" % html.escape(doc.metadata.get("title") or SRC.stem),
-       "" % CSS]
-for tag, cls, txt in parts:
-    if tag == "figure":
-        buf.append('

%s

' % txt) - else: - c = ' class="%s"' % cls if cls else "" - buf.append("<%s%s>%s" % (tag, c, txt, tag)) -buf.append("") -doc_html = "\n".join(buf) -# merge tag runs split at page/line boundaries, collapse doubled spaces -doc_html = re.sub(r"(\s*)<\1>", r"\2", doc_html) -doc_html = re.sub(r"(?<=\S) {2,}(?=\S)", " ", doc_html) +doc_html = bookhtml.document(doc.metadata.get("title") or SRC.stem, parts) +# merge tag runs split at page/line boundaries. Пробелы здесь уже не трогаем: +# внутри
 они значимы, а в прозе схлопнуты в block_text.
+doc_html = re.sub(r"(\s*)<\1>", r"\2", doc_html)
 OUT.write_text(doc_html, encoding="utf-8")
 
 print("blocks:", len(parts), "images:", img_n)
 print("h1:", sum(1 for p in parts if p[0] == "h1"),
-      "p:", sum(1 for p in parts if p[0] == "p"))
+      "p:", sum(1 for p in parts if p[0] == "p"),
+      "pre:", sum(1 for p in parts if p[0] == "pre"))
diff --git a/scripts/test_epub2html.py b/scripts/test_epub2html.py
new file mode 100644
index 0000000..2a5c49f
--- /dev/null
+++ b/scripts/test_epub2html.py
@@ -0,0 +1,119 @@
+#!/usr/bin/env python3
+"""Самопроверка конвертера EPUB без сети: python3 test_epub2html.py"""
+import tempfile
+import zipfile
+from pathlib import Path
+
+import epub2html
+
+CONTAINER = """
+
+  
+"""
+
+OPF = """
+
+  
+  
+    
+    
+    
+    
+    
+    
+  
+  
+    
+    
+  
+"""
+
+CH1 = """
+skip me
+
+

Chapter One

+

The lazy fox calls fetch_data() twice a day.

+
def main():
+    x = 1  -  2
+    return x
+
  • first item
  • second item
+
NameValue
alpha1
+

Broken markup that never closes.

+

figure

+ +""" + +CH2 = """ + +

Chapter Two

Second chapter body text.

+""" + +CH3 = """ + +

Chapter Three

Loose text outside any block tag.
+""" + +NAV = """ + +""" + +PNG = bytes.fromhex("89504e470d0a1a0a") # достаточно как содержимое файла + + +def build_epub(path): + with zipfile.ZipFile(path, "w") as z: + z.writestr("mimetype", "application/epub+zip") + z.writestr("META-INF/container.xml", CONTAINER) + z.writestr("OEBPS/content.opf", OPF) + z.writestr("OEBPS/ch1.xhtml", CH1) + z.writestr("OEBPS/ch2.xhtml", CH2) + z.writestr("OEBPS/ch3.xhtml", CH3) + z.writestr("OEBPS/nav.xhtml", NAV) + z.writestr("OEBPS/img/fig.png", PNG) + z.writestr("OEBPS/img/cover.png", PNG) + + +def test_convert(tmp: Path): + src = tmp / "book.epub" + build_epub(src) + out = tmp / "book.html" + epub2html.convert(src, out) + text = out.read_text(encoding="utf-8") + lines = text.splitlines() + + # Листинг: переносы строк, отступы и минусы не тронуты + pre = text[text.index("
"):text.index("
")] + assert "def main():\n x = 1 - 2\n return x" in pre, pre + assert "" not in pre, "внутри листинга инлайновая разметка не нужна" + + # Проза: инлайн переведён в наш набор тегов, сущности раскрыты + para = next(l for l in lines if "fox" in l) + assert "lazy" in para and "fetch_data()" in para, para + assert " " not in para and " " not in para, para + + assert '

first item

' in lines + assert '

Name | Value

' in lines + assert '

alpha | 1

' in lines + assert '' in text + assert (out.parent / "images" / "cover.png").exists(), "обложка не извлечена" + + # Несбалансированная чужая разметка закрывается на границе блока + broken = next(l for l in lines if "never closes" in l) + assert broken.count("") == broken.count("") == 1, broken + + assert "noise" not in text, "