Восстановление структуры EPUB по CSS для дампов из PDF

Дамп, собранный конвертером из PDF, теряет всю семантику: ни заголовков, ни
листингов, только абзацы с обфусцированными классами. Такие книги уезжали в
перевод одним куском, а код — вместе с текстом.

Теперь при нулевом числе заголовков или листингов разметка восстанавливается из
таблицы стилей книги: три самых крупных кегля становятся уровнями заголовков,
моноширинная гарнитура — листингом, соседние строки листинга склеиваются в один
блок. Книга со своей разметкой в этот путь не попадает.

Проверено на двух настоящих книгах: дамп APoSD — 1833 плоских абзаца стали 29
главами и 58 листингами; Страуструп с собственной разметкой не изменился.
This commit is contained in:
chesirecatt
2026-08-20 19:11:51 +03:00
parent db9fe5e0fa
commit 1bacd38292
3 changed files with 181 additions and 18 deletions
+9
View File
@@ -32,6 +32,15 @@ needed, not PyMuPDF) and convert with:
`python scripts/epub2html.py book.epub out/book.html` `python scripts/epub2html.py book.epub out/book.html`
**Dumps converted from PDF carry no semantics at all** — no headings, no `<pre>`,
just `<p class="class_s1e2">` with obfuscated names. The script detects this
(zero headings or zero listings) and rebuilds the structure from the book's own
stylesheet: the three largest font sizes become heading levels, a typewriter or
monospace family becomes `<pre>`, and adjacent listing paragraphs merge back into
one block. A book that carries its own markup never enters this path. Measured on
the dokumen.pub dump of Ousterhout's APoSD: 1833 flat paragraphs became 29
chapters and 58 listings.
It reports which heading level turned out to be the chapter level. In most It reports which heading level turned out to be the chapter level. In most
EPUBs `<h1>` is the book title and the parts, while chapters are `<h2>` — the EPUBs `<h1>` is the book title and the parts, while chapters are `<h2>` — the
script picks the deepest level that still gives a sane chapter count, because script picks the deepest level that still gives a sane chapter count, because
+89 -6
View File
@@ -39,16 +39,57 @@ INLINE_MAP = {"i": "i", "em": "i", "cite": "i", "b": "b", "strong": "b",
"code": "code", "kbd": "code", "samp": "code", "tt": "code", "code": "code", "kbd": "code", "samp": "code", "tt": "code",
"var": "code"} "var": "code"}
DROP = {"script", "style", "head", "title", "nav"} DROP = {"script", "style", "head", "title", "nav"}
# Дампы, собранные конвертером из PDF, теряют всю семантику: ни заголовков, ни
# <pre>, только <p class="class_s1e2"> с обфусцированными именами. Разметку
# приходится восстанавливать из таблицы стилей — по кеглю и по гарнитуре.
MONO_FONT = re.compile(r"mono|courier|consol|typewriter|menlo|inconsolata|"
r"liberationmono|andalemono|lucidasanstype", re.I)
CSS_RULE = re.compile(r"\.([\w-]+)\s*\{([^}]*)\}")
IMG_EXT = {"image/jpeg": ".jpg", "image/png": ".png", "image/gif": ".gif", IMG_EXT = {"image/jpeg": ".jpg", "image/png": ".png", "image/gif": ".gif",
"image/svg+xml": ".svg", "image/webp": ".webp"} "image/svg+xml": ".svg", "image/webp": ".webp"}
def css_styles(zf):
"""class -> (кегль в em, моноширинный ли) из всех таблиц стилей книги."""
out = {}
for name in zf.namelist():
if not name.lower().endswith(".css"):
continue
for cls, body in CSS_RULE.findall(zf.read(name).decode("utf-8", "replace")):
size = out.get(cls, (0.0, False))[0]
m = re.search(r"font-size:\s*([\d.]+)\s*(em|rem|pt|px|%)", body)
if m:
v = float(m.group(1))
size = {"em": v, "rem": v, "pt": v / 12, "px": v / 16,
"%": v / 100}[m.group(2)]
fam = re.search(r"font-family:\s*([^;]+)", body)
mono = bool(fam and MONO_FONT.search(fam.group(1)))
was = out.get(cls, (0.0, False))
out[cls] = (size or was[0], mono or was[1])
return out
def derive_from_css(styles):
"""(класс -> уровень заголовка, множество моноширинных классов).
Заголовки — три самых крупных кегля выше основного текста. Больше трёх
уровней брать нельзя: дальше начинаются подписи и колонтитулы.
"""
sizes = sorted({s for s, _ in styles.values() if s > 1.05}, reverse=True)[:3]
level = {s: i + 1 for i, s in enumerate(sizes)}
head = {c: level[s] for c, (s, _) in styles.items() if s in level}
mono = {c for c, (_, m) in styles.items() if m}
return head, mono
class Reader(HTMLParser): class Reader(HTMLParser):
"""Собирает блоки из одного XHTML-документа книги.""" """Собирает блоки из одного XHTML-документа книги."""
def __init__(self, on_image): def __init__(self, on_image, head_cls=None, mono_cls=None):
super().__init__(convert_charrefs=True) super().__init__(convert_charrefs=True)
self.on_image = on_image self.on_image = on_image
self.head_cls = head_cls or {}
self.mono_cls = mono_cls or set()
self.blocks = [] self.blocks = []
self.cur = None # (tag, cls, [куски]) self.cur = None # (tag, cls, [куски])
self.open_inline = [] # незакрытые инлайновые теги текущего блока self.open_inline = [] # незакрытые инлайновые теги текущего блока
@@ -69,7 +110,12 @@ class Reader(HTMLParser):
else: else:
txt = re.sub(r"\s+", " ", txt).strip() txt = re.sub(r"\s+", " ", txt).strip()
txt = re.sub(r"<(i|b|code)>(\s*)</\1>", r"\2", txt) # пустая разметка txt = re.sub(r"<(i|b|code)>(\s*)</\1>", r"\2", txt) # пустая разметка
if txt.strip(): if not txt.strip():
return
if tag == "pre" and self.blocks and self.blocks[-1][0] == "pre":
prev = self.blocks[-1]
self.blocks[-1] = (prev[0], prev[1], prev[2] + "\n" + txt)
return
self.blocks.append((tag, cls, txt)) self.blocks.append((tag, cls, txt))
def start(self, tag, cls): def start(self, tag, cls):
@@ -98,7 +144,16 @@ class Reader(HTMLParser):
self.cur[2].append("\n" if self.cur[0] == "pre" else " ") self.cur[2].append("\n" if self.cur[0] == "pre" else " ")
return return
if tag in BLOCK_MAP: if tag in BLOCK_MAP:
self.start(*BLOCK_MAP[tag]) out_tag, out_cls = BLOCK_MAP[tag]
if out_tag in ("p", "pre"):
for c in (a.get("class") or "").split():
if c in self.head_cls:
out_tag, out_cls = "h%d" % self.head_cls[c], None
break
if c in self.mono_cls:
out_tag, out_cls = "pre", None
break
self.start(out_tag, out_cls)
return return
if tag in ("td", "th") and self.cur and self.cur[0] == "p" \ if tag in ("td", "th") and self.cur and self.cur[0] == "p" \
and self.cur[1] == "row" and self.cur[2]: and self.cur[1] == "row" and self.cur[2]:
@@ -220,6 +275,8 @@ def convert(src, out):
if cover: if cover:
extract(cover, stem="cover") extract(cover, stem="cover")
def parse_all(head_cls=None, mono_cls=None):
got = []
for doc in docs: for doc in docs:
if doc not in names: if doc not in names:
print("нет в архиве, пропущен: %s" % doc, file=sys.stderr) print("нет в архиве, пропущен: %s" % doc, file=sys.stderr)
@@ -227,19 +284,45 @@ def convert(src, out):
here = posixpath.dirname(doc) here = posixpath.dirname(doc)
def on_image(src_attr, here=here): def on_image(src_attr, here=here):
target = posixpath.normpath(posixpath.join(here, src_attr.split("#")[0])) target = posixpath.normpath(
posixpath.join(here, src_attr.split("#")[0]))
return extract(target) return extract(target)
r = Reader(on_image) r = Reader(on_image, head_cls, mono_cls)
r.feed(zf.read(doc).decode("utf-8", "replace")) r.feed(zf.read(doc).decode("utf-8", "replace"))
r.close() r.close()
parts.extend(r.blocks) got.extend(r.blocks)
return got
parts = parse_all()
has_head = any(re.fullmatch(r"h[1-6]", t) for t, _, _ in parts)
has_pre = any(t == "pre" for t, _, _ in parts)
if not (has_head and has_pre):
head_cls, mono_cls = derive_from_css(css_styles(zf))
if (not has_head and head_cls) or (not has_pre and mono_cls):
print("семантики в книге нет (заголовки: %s, листинги: %s) — "
"восстанавливаю по CSS" % (has_head, has_pre))
seen.clear()
counter[0] = 0
parts = parse_all(head_cls if not has_head else None,
mono_cls if not has_pre else None)
title = out.stem title = out.stem
lvl = chapter_level(parts) lvl = chapter_level(parts)
parts = [(("h1" if int(t[1]) <= lvl else "h2", c, x) parts = [(("h1" if int(t[1]) <= lvl else "h2", c, x)
if re.fullmatch(r"h[1-6]", t) else (t, c, x)) if re.fullmatch(r"h[1-6]", t) else (t, c, x))
for t, c, x in parts] for t, c, x in parts]
# «Chapter 1» и «Introduction» приезжают отдельными заголовками: в дампе это
# две строки. Иначе каждая вторая глава состоит из одного блока.
merged = []
for part in parts:
if (part[0] == "h1" and merged and merged[-1][0] == "h1"
and len(merged[-1][2]) < 60):
merged[-1] = ("h1", None, merged[-1][2] + ". " + part[2])
continue
merged.append(part)
parts = merged
h1 = sum(1 for p in parts if p[0] == "h1") h1 = sum(1 for p in parts if p[0] == "h1")
print("главы размечены h%d, глав получилось: %d" % (lvl, h1)) print("главы размечены h%d, глав получилось: %d" % (lvl, h1))
if h1 < 3: if h1 < 3:
+72 -1
View File
@@ -113,7 +113,78 @@ def test_convert(tmp: Path):
assert text.count("<h1>") == 3 and "<h2>" not in text assert text.count("<h1>") == 3 and "<h2>" not in text
# Дамп, собранный конвертером из PDF: ни одного заголовка, ни одного <pre>,
# только <p> с обфусцированными классами и таблица стилей.
FLAT_CSS = """
.cls_head { font-size: 1.66667em; font-family: Reg; }
.cls_body { font-size: 1em; font-family: Reg; }
.cls_code { font-size: 0.9em; font-family: lucidasanstypewriter; }
"""
FLAT_OPF = """<?xml version="1.0"?>
<package xmlns="http://www.idpf.org/2007/opf" version="2.0">
<metadata/>
<manifest>
<item id="s" href="style.css" media-type="text/css"/>
<item id="a" href="p1.xhtml" media-type="application/xhtml+xml"/>
<item id="b" href="p2.xhtml" media-type="application/xhtml+xml"/>
<item id="c" href="p3.xhtml" media-type="application/xhtml+xml"/>
</manifest>
<spine><itemref idref="a"/><itemref idref="b"/><itemref idref="c"/></spine>
</package>"""
def flat_doc(n):
return """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<p class="cls_head">Chapter %d</p>
<p class="cls_head">Title Of Chapter %d</p>
<p class="cls_body">Body text of the chapter, long enough to be a paragraph.</p>
<p class="cls_code">def f(x):</p>
<p class="cls_code"> return x + 1</p>
<p class="cls_body">Closing paragraph of the chapter goes here.</p>
</body></html>""" % (n, n)
def test_flat_dump_recovered(tmp: Path):
src = tmp / "flat.epub"
with zipfile.ZipFile(src, "w") as z:
z.writestr("mimetype", "application/epub+zip")
z.writestr("META-INF/container.xml", CONTAINER)
z.writestr("OEBPS/content.opf", FLAT_OPF)
z.writestr("OEBPS/style.css", FLAT_CSS)
for i in (1, 2, 3):
z.writestr("OEBPS/p%d.xhtml" % i, flat_doc(i))
out = tmp / "flat.html"
epub2html.convert(src, out)
text = out.read_text(encoding="utf-8")
# заголовки восстановлены по кеглю и склеены с номером главы
assert "<h1>Chapter 2. Title Of Chapter 2</h1>" in text, text
assert text.count("<h1>") == 3
# листинг восстановлен по гарнитуре, соседние строки склеены в один блок
assert "<pre>def f(x):\n return x + 1</pre>" in text, text
assert text.count("<pre>") == 3
# обычный текст остался абзацем
assert "<p>Body text of the chapter" in text
def test_real_markup_wins_over_css(tmp: Path):
"""Книга со своей разметкой не должна уезжать в запасной путь."""
src = tmp / "book.epub"
build_epub(src)
out = tmp / "book.html"
epub2html.convert(src, out)
text = out.read_text(encoding="utf-8")
assert text.count("<h1>") == 3, "заголовки книги должны остаться её собственными"
assert "def main():" in text
if __name__ == "__main__": if __name__ == "__main__":
for case in (test_convert, test_flat_dump_recovered,
test_real_markup_wins_over_css):
with tempfile.TemporaryDirectory() as d: with tempfile.TemporaryDirectory() as d:
test_convert(Path(d)) case(Path(d))
print("OK") print("OK")