From 1bacd38292de45f0f2875567b6eef08c293b7d80 Mon Sep 17 00:00:00 2001 From: chesirecatt Date: Thu, 20 Aug 2026 19:11:51 +0300 Subject: [PATCH] =?UTF-8?q?=D0=92=D0=BE=D1=81=D1=81=D1=82=D0=B0=D0=BD?= =?UTF-8?q?=D0=BE=D0=B2=D0=BB=D0=B5=D0=BD=D0=B8=D0=B5=20=D1=81=D1=82=D1=80?= =?UTF-8?q?=D1=83=D0=BA=D1=82=D1=83=D1=80=D1=8B=20EPUB=20=D0=BF=D0=BE=20CS?= =?UTF-8?q?S=20=D0=B4=D0=BB=D1=8F=20=D0=B4=D0=B0=D0=BC=D0=BF=D0=BE=D0=B2?= =?UTF-8?q?=20=D0=B8=D0=B7=20PDF?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Дамп, собранный конвертером из PDF, теряет всю семантику: ни заголовков, ни листингов, только абзацы с обфусцированными классами. Такие книги уезжали в перевод одним куском, а код — вместе с текстом. Теперь при нулевом числе заголовков или листингов разметка восстанавливается из таблицы стилей книги: три самых крупных кегля становятся уровнями заголовков, моноширинная гарнитура — листингом, соседние строки листинга склеиваются в один блок. Книга со своей разметкой в этот путь не попадает. Проверено на двух настоящих книгах: дамп APoSD — 1833 плоских абзаца стали 29 главами и 58 листингами; Страуструп с собственной разметкой не изменился. --- SKILL.md | 9 +++ scripts/epub2html.py | 115 ++++++++++++++++++++++++++++++++------ scripts/test_epub2html.py | 75 ++++++++++++++++++++++++- 3 files changed, 181 insertions(+), 18 deletions(-) diff --git a/SKILL.md b/SKILL.md index 19fac81..a68aa47 100644 --- a/SKILL.md +++ b/SKILL.md @@ -32,6 +32,15 @@ needed, not PyMuPDF) and convert with: `python scripts/epub2html.py book.epub out/book.html` +**Dumps converted from PDF carry no semantics at all** — no headings, no `
`,
+just `

` with obfuscated names. The script detects this +(zero headings or zero listings) and rebuilds the structure from the book's own +stylesheet: the three largest font sizes become heading levels, a typewriter or +monospace family becomes `

`, and adjacent listing paragraphs merge back into
+one block. A book that carries its own markup never enters this path. Measured on
+the dokumen.pub dump of Ousterhout's APoSD: 1833 flat paragraphs became 29
+chapters and 58 listings.
+
 It reports which heading level turned out to be the chapter level. In most
 EPUBs `

` is the book title and the parts, while chapters are `

` — the script picks the deepest level that still gives a sane chapter count, because diff --git a/scripts/epub2html.py b/scripts/epub2html.py index 37ccef4..1dc3a41 100644 --- a/scripts/epub2html.py +++ b/scripts/epub2html.py @@ -39,16 +39,57 @@ INLINE_MAP = {"i": "i", "em": "i", "cite": "i", "b": "b", "strong": "b", "code": "code", "kbd": "code", "samp": "code", "tt": "code", "var": "code"} DROP = {"script", "style", "head", "title", "nav"} +# Дампы, собранные конвертером из PDF, теряют всю семантику: ни заголовков, ни +#
, только 

с обфусцированными именами. Разметку +# приходится восстанавливать из таблицы стилей — по кеглю и по гарнитуре. +MONO_FONT = re.compile(r"mono|courier|consol|typewriter|menlo|inconsolata|" + r"liberationmono|andalemono|lucidasanstype", re.I) +CSS_RULE = re.compile(r"\.([\w-]+)\s*\{([^}]*)\}") IMG_EXT = {"image/jpeg": ".jpg", "image/png": ".png", "image/gif": ".gif", "image/svg+xml": ".svg", "image/webp": ".webp"} +def css_styles(zf): + """class -> (кегль в em, моноширинный ли) из всех таблиц стилей книги.""" + out = {} + for name in zf.namelist(): + if not name.lower().endswith(".css"): + continue + for cls, body in CSS_RULE.findall(zf.read(name).decode("utf-8", "replace")): + size = out.get(cls, (0.0, False))[0] + m = re.search(r"font-size:\s*([\d.]+)\s*(em|rem|pt|px|%)", body) + if m: + v = float(m.group(1)) + size = {"em": v, "rem": v, "pt": v / 12, "px": v / 16, + "%": v / 100}[m.group(2)] + fam = re.search(r"font-family:\s*([^;]+)", body) + mono = bool(fam and MONO_FONT.search(fam.group(1))) + was = out.get(cls, (0.0, False)) + out[cls] = (size or was[0], mono or was[1]) + return out + + +def derive_from_css(styles): + """(класс -> уровень заголовка, множество моноширинных классов). + + Заголовки — три самых крупных кегля выше основного текста. Больше трёх + уровней брать нельзя: дальше начинаются подписи и колонтитулы. + """ + sizes = sorted({s for s, _ in styles.values() if s > 1.05}, reverse=True)[:3] + level = {s: i + 1 for i, s in enumerate(sizes)} + head = {c: level[s] for c, (s, _) in styles.items() if s in level} + mono = {c for c, (_, m) in styles.items() if m} + return head, mono + + class Reader(HTMLParser): """Собирает блоки из одного XHTML-документа книги.""" - def __init__(self, on_image): + def __init__(self, on_image, head_cls=None, mono_cls=None): super().__init__(convert_charrefs=True) self.on_image = on_image + self.head_cls = head_cls or {} + self.mono_cls = mono_cls or set() self.blocks = [] self.cur = None # (tag, cls, [куски]) self.open_inline = [] # незакрытые инлайновые теги текущего блока @@ -69,8 +110,13 @@ class Reader(HTMLParser): else: txt = re.sub(r"\s+", " ", txt).strip() txt = re.sub(r"<(i|b|code)>(\s*)", r"\2", txt) # пустая разметка - if txt.strip(): - self.blocks.append((tag, cls, txt)) + if not txt.strip(): + return + if tag == "pre" and self.blocks and self.blocks[-1][0] == "pre": + prev = self.blocks[-1] + self.blocks[-1] = (prev[0], prev[1], prev[2] + "\n" + txt) + return + self.blocks.append((tag, cls, txt)) def start(self, tag, cls): self.flush() @@ -98,7 +144,16 @@ class Reader(HTMLParser): self.cur[2].append("\n" if self.cur[0] == "pre" else " ") return if tag in BLOCK_MAP: - self.start(*BLOCK_MAP[tag]) + out_tag, out_cls = BLOCK_MAP[tag] + if out_tag in ("p", "pre"): + for c in (a.get("class") or "").split(): + if c in self.head_cls: + out_tag, out_cls = "h%d" % self.head_cls[c], None + break + if c in self.mono_cls: + out_tag, out_cls = "pre", None + break + self.start(out_tag, out_cls) return if tag in ("td", "th") and self.cur and self.cur[0] == "p" \ and self.cur[1] == "row" and self.cur[2]: @@ -220,26 +275,54 @@ def convert(src, out): if cover: extract(cover, stem="cover") - for doc in docs: - if doc not in names: - print("нет в архиве, пропущен: %s" % doc, file=sys.stderr) - continue - here = posixpath.dirname(doc) + def parse_all(head_cls=None, mono_cls=None): + got = [] + for doc in docs: + if doc not in names: + print("нет в архиве, пропущен: %s" % doc, file=sys.stderr) + continue + here = posixpath.dirname(doc) - def on_image(src_attr, here=here): - target = posixpath.normpath(posixpath.join(here, src_attr.split("#")[0])) - return extract(target) + def on_image(src_attr, here=here): + target = posixpath.normpath( + posixpath.join(here, src_attr.split("#")[0])) + return extract(target) - r = Reader(on_image) - r.feed(zf.read(doc).decode("utf-8", "replace")) - r.close() - parts.extend(r.blocks) + r = Reader(on_image, head_cls, mono_cls) + r.feed(zf.read(doc).decode("utf-8", "replace")) + r.close() + got.extend(r.blocks) + return got + + parts = parse_all() + has_head = any(re.fullmatch(r"h[1-6]", t) for t, _, _ in parts) + has_pre = any(t == "pre" for t, _, _ in parts) + if not (has_head and has_pre): + head_cls, mono_cls = derive_from_css(css_styles(zf)) + if (not has_head and head_cls) or (not has_pre and mono_cls): + print("семантики в книге нет (заголовки: %s, листинги: %s) — " + "восстанавливаю по CSS" % (has_head, has_pre)) + seen.clear() + counter[0] = 0 + parts = parse_all(head_cls if not has_head else None, + mono_cls if not has_pre else None) title = out.stem lvl = chapter_level(parts) parts = [(("h1" if int(t[1]) <= lvl else "h2", c, x) if re.fullmatch(r"h[1-6]", t) else (t, c, x)) for t, c, x in parts] + # «Chapter 1» и «Introduction» приезжают отдельными заголовками: в дампе это + # две строки. Иначе каждая вторая глава состоит из одного блока. + merged = [] + for part in parts: + if (part[0] == "h1" and merged and merged[-1][0] == "h1" + and len(merged[-1][2]) < 60): + merged[-1] = ("h1", None, merged[-1][2] + ". " + part[2]) + continue + merged.append(part) + parts = merged + h1 = sum(1 for p in parts if p[0] == "h1") print("главы размечены h%d, глав получилось: %d" % (lvl, h1)) if h1 < 3: diff --git a/scripts/test_epub2html.py b/scripts/test_epub2html.py index 2a5c49f..640ac8e 100644 --- a/scripts/test_epub2html.py +++ b/scripts/test_epub2html.py @@ -113,7 +113,78 @@ def test_convert(tmp: Path): assert text.count("

") == 3 and "

" not in text +# Дамп, собранный конвертером из PDF: ни одного заголовка, ни одного
,
+# только 

с обфусцированными классами и таблица стилей. +FLAT_CSS = """ +.cls_head { font-size: 1.66667em; font-family: Reg; } +.cls_body { font-size: 1em; font-family: Reg; } +.cls_code { font-size: 0.9em; font-family: lucidasanstypewriter; } +""" + +FLAT_OPF = """ + + + + + + + + + +""" + + +def flat_doc(n): + return """ + +

Chapter %d

+

Title Of Chapter %d

+

Body text of the chapter, long enough to be a paragraph.

+

def f(x):

+

return x + 1

+

Closing paragraph of the chapter goes here.

+""" % (n, n) + + +def test_flat_dump_recovered(tmp: Path): + src = tmp / "flat.epub" + with zipfile.ZipFile(src, "w") as z: + z.writestr("mimetype", "application/epub+zip") + z.writestr("META-INF/container.xml", CONTAINER) + z.writestr("OEBPS/content.opf", FLAT_OPF) + z.writestr("OEBPS/style.css", FLAT_CSS) + for i in (1, 2, 3): + z.writestr("OEBPS/p%d.xhtml" % i, flat_doc(i)) + out = tmp / "flat.html" + epub2html.convert(src, out) + text = out.read_text(encoding="utf-8") + + # заголовки восстановлены по кеглю и склеены с номером главы + assert "

Chapter 2. Title Of Chapter 2

" in text, text + assert text.count("

") == 3 + + # листинг восстановлен по гарнитуре, соседние строки склеены в один блок + assert "
def f(x):\n    return x + 1
" in text, text + assert text.count("
") == 3
+
+    # обычный текст остался абзацем
+    assert "

Body text of the chapter" in text + + +def test_real_markup_wins_over_css(tmp: Path): + """Книга со своей разметкой не должна уезжать в запасной путь.""" + src = tmp / "book.epub" + build_epub(src) + out = tmp / "book.html" + epub2html.convert(src, out) + text = out.read_text(encoding="utf-8") + assert text.count("

") == 3, "заголовки книги должны остаться её собственными" + assert "def main():" in text + + if __name__ == "__main__": - with tempfile.TemporaryDirectory() as d: - test_convert(Path(d)) + for case in (test_convert, test_flat_dump_recovered, + test_real_markup_wins_over_css): + with tempfile.TemporaryDirectory() as d: + case(Path(d)) print("OK")