Восстановление структуры EPUB по CSS для дампов из PDF

Дамп, собранный конвертером из PDF, теряет всю семантику: ни заголовков, ни
листингов, только абзацы с обфусцированными классами. Такие книги уезжали в
перевод одним куском, а код — вместе с текстом.

Теперь при нулевом числе заголовков или листингов разметка восстанавливается из
таблицы стилей книги: три самых крупных кегля становятся уровнями заголовков,
моноширинная гарнитура — листингом, соседние строки листинга склеиваются в один
блок. Книга со своей разметкой в этот путь не попадает.

Проверено на двух настоящих книгах: дамп APoSD — 1833 плоских абзаца стали 29
главами и 58 листингами; Страуструп с собственной разметкой не изменился.
This commit is contained in:
chesirecatt
2026-08-20 19:11:51 +03:00
parent db9fe5e0fa
commit 1bacd38292
3 changed files with 181 additions and 18 deletions
+99 -16
View File
@@ -39,16 +39,57 @@ INLINE_MAP = {"i": "i", "em": "i", "cite": "i", "b": "b", "strong": "b",
"code": "code", "kbd": "code", "samp": "code", "tt": "code",
"var": "code"}
DROP = {"script", "style", "head", "title", "nav"}
# Дампы, собранные конвертером из PDF, теряют всю семантику: ни заголовков, ни
# <pre>, только <p class="class_s1e2"> с обфусцированными именами. Разметку
# приходится восстанавливать из таблицы стилей — по кеглю и по гарнитуре.
MONO_FONT = re.compile(r"mono|courier|consol|typewriter|menlo|inconsolata|"
r"liberationmono|andalemono|lucidasanstype", re.I)
CSS_RULE = re.compile(r"\.([\w-]+)\s*\{([^}]*)\}")
IMG_EXT = {"image/jpeg": ".jpg", "image/png": ".png", "image/gif": ".gif",
"image/svg+xml": ".svg", "image/webp": ".webp"}
def css_styles(zf):
"""class -> (кегль в em, моноширинный ли) из всех таблиц стилей книги."""
out = {}
for name in zf.namelist():
if not name.lower().endswith(".css"):
continue
for cls, body in CSS_RULE.findall(zf.read(name).decode("utf-8", "replace")):
size = out.get(cls, (0.0, False))[0]
m = re.search(r"font-size:\s*([\d.]+)\s*(em|rem|pt|px|%)", body)
if m:
v = float(m.group(1))
size = {"em": v, "rem": v, "pt": v / 12, "px": v / 16,
"%": v / 100}[m.group(2)]
fam = re.search(r"font-family:\s*([^;]+)", body)
mono = bool(fam and MONO_FONT.search(fam.group(1)))
was = out.get(cls, (0.0, False))
out[cls] = (size or was[0], mono or was[1])
return out
def derive_from_css(styles):
"""(класс -> уровень заголовка, множество моноширинных классов).
Заголовки — три самых крупных кегля выше основного текста. Больше трёх
уровней брать нельзя: дальше начинаются подписи и колонтитулы.
"""
sizes = sorted({s for s, _ in styles.values() if s > 1.05}, reverse=True)[:3]
level = {s: i + 1 for i, s in enumerate(sizes)}
head = {c: level[s] for c, (s, _) in styles.items() if s in level}
mono = {c for c, (_, m) in styles.items() if m}
return head, mono
class Reader(HTMLParser):
"""Собирает блоки из одного XHTML-документа книги."""
def __init__(self, on_image):
def __init__(self, on_image, head_cls=None, mono_cls=None):
super().__init__(convert_charrefs=True)
self.on_image = on_image
self.head_cls = head_cls or {}
self.mono_cls = mono_cls or set()
self.blocks = []
self.cur = None # (tag, cls, [куски])
self.open_inline = [] # незакрытые инлайновые теги текущего блока
@@ -69,8 +110,13 @@ class Reader(HTMLParser):
else:
txt = re.sub(r"\s+", " ", txt).strip()
txt = re.sub(r"<(i|b|code)>(\s*)</\1>", r"\2", txt) # пустая разметка
if txt.strip():
self.blocks.append((tag, cls, txt))
if not txt.strip():
return
if tag == "pre" and self.blocks and self.blocks[-1][0] == "pre":
prev = self.blocks[-1]
self.blocks[-1] = (prev[0], prev[1], prev[2] + "\n" + txt)
return
self.blocks.append((tag, cls, txt))
def start(self, tag, cls):
self.flush()
@@ -98,7 +144,16 @@ class Reader(HTMLParser):
self.cur[2].append("\n" if self.cur[0] == "pre" else " ")
return
if tag in BLOCK_MAP:
self.start(*BLOCK_MAP[tag])
out_tag, out_cls = BLOCK_MAP[tag]
if out_tag in ("p", "pre"):
for c in (a.get("class") or "").split():
if c in self.head_cls:
out_tag, out_cls = "h%d" % self.head_cls[c], None
break
if c in self.mono_cls:
out_tag, out_cls = "pre", None
break
self.start(out_tag, out_cls)
return
if tag in ("td", "th") and self.cur and self.cur[0] == "p" \
and self.cur[1] == "row" and self.cur[2]:
@@ -220,26 +275,54 @@ def convert(src, out):
if cover:
extract(cover, stem="cover")
for doc in docs:
if doc not in names:
print("нет в архиве, пропущен: %s" % doc, file=sys.stderr)
continue
here = posixpath.dirname(doc)
def parse_all(head_cls=None, mono_cls=None):
got = []
for doc in docs:
if doc not in names:
print("нет в архиве, пропущен: %s" % doc, file=sys.stderr)
continue
here = posixpath.dirname(doc)
def on_image(src_attr, here=here):
target = posixpath.normpath(posixpath.join(here, src_attr.split("#")[0]))
return extract(target)
def on_image(src_attr, here=here):
target = posixpath.normpath(
posixpath.join(here, src_attr.split("#")[0]))
return extract(target)
r = Reader(on_image)
r.feed(zf.read(doc).decode("utf-8", "replace"))
r.close()
parts.extend(r.blocks)
r = Reader(on_image, head_cls, mono_cls)
r.feed(zf.read(doc).decode("utf-8", "replace"))
r.close()
got.extend(r.blocks)
return got
parts = parse_all()
has_head = any(re.fullmatch(r"h[1-6]", t) for t, _, _ in parts)
has_pre = any(t == "pre" for t, _, _ in parts)
if not (has_head and has_pre):
head_cls, mono_cls = derive_from_css(css_styles(zf))
if (not has_head and head_cls) or (not has_pre and mono_cls):
print("семантики в книге нет (заголовки: %s, листинги: %s) — "
"восстанавливаю по CSS" % (has_head, has_pre))
seen.clear()
counter[0] = 0
parts = parse_all(head_cls if not has_head else None,
mono_cls if not has_pre else None)
title = out.stem
lvl = chapter_level(parts)
parts = [(("h1" if int(t[1]) <= lvl else "h2", c, x)
if re.fullmatch(r"h[1-6]", t) else (t, c, x))
for t, c, x in parts]
# «Chapter 1» и «Introduction» приезжают отдельными заголовками: в дампе это
# две строки. Иначе каждая вторая глава состоит из одного блока.
merged = []
for part in parts:
if (part[0] == "h1" and merged and merged[-1][0] == "h1"
and len(merged[-1][2]) < 60):
merged[-1] = ("h1", None, merged[-1][2] + ". " + part[2])
continue
merged.append(part)
parts = merged
h1 = sum(1 for p in parts if p[0] == "h1")
print("главы размечены h%d, глав получилось: %d" % (lvl, h1))
if h1 < 3:
+73 -2
View File
@@ -113,7 +113,78 @@ def test_convert(tmp: Path):
assert text.count("<h1>") == 3 and "<h2>" not in text
# Дамп, собранный конвертером из PDF: ни одного заголовка, ни одного <pre>,
# только <p> с обфусцированными классами и таблица стилей.
FLAT_CSS = """
.cls_head { font-size: 1.66667em; font-family: Reg; }
.cls_body { font-size: 1em; font-family: Reg; }
.cls_code { font-size: 0.9em; font-family: lucidasanstypewriter; }
"""
FLAT_OPF = """<?xml version="1.0"?>
<package xmlns="http://www.idpf.org/2007/opf" version="2.0">
<metadata/>
<manifest>
<item id="s" href="style.css" media-type="text/css"/>
<item id="a" href="p1.xhtml" media-type="application/xhtml+xml"/>
<item id="b" href="p2.xhtml" media-type="application/xhtml+xml"/>
<item id="c" href="p3.xhtml" media-type="application/xhtml+xml"/>
</manifest>
<spine><itemref idref="a"/><itemref idref="b"/><itemref idref="c"/></spine>
</package>"""
def flat_doc(n):
return """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<p class="cls_head">Chapter %d</p>
<p class="cls_head">Title Of Chapter %d</p>
<p class="cls_body">Body text of the chapter, long enough to be a paragraph.</p>
<p class="cls_code">def f(x):</p>
<p class="cls_code"> return x + 1</p>
<p class="cls_body">Closing paragraph of the chapter goes here.</p>
</body></html>""" % (n, n)
def test_flat_dump_recovered(tmp: Path):
src = tmp / "flat.epub"
with zipfile.ZipFile(src, "w") as z:
z.writestr("mimetype", "application/epub+zip")
z.writestr("META-INF/container.xml", CONTAINER)
z.writestr("OEBPS/content.opf", FLAT_OPF)
z.writestr("OEBPS/style.css", FLAT_CSS)
for i in (1, 2, 3):
z.writestr("OEBPS/p%d.xhtml" % i, flat_doc(i))
out = tmp / "flat.html"
epub2html.convert(src, out)
text = out.read_text(encoding="utf-8")
# заголовки восстановлены по кеглю и склеены с номером главы
assert "<h1>Chapter 2. Title Of Chapter 2</h1>" in text, text
assert text.count("<h1>") == 3
# листинг восстановлен по гарнитуре, соседние строки склеены в один блок
assert "<pre>def f(x):\n return x + 1</pre>" in text, text
assert text.count("<pre>") == 3
# обычный текст остался абзацем
assert "<p>Body text of the chapter" in text
def test_real_markup_wins_over_css(tmp: Path):
"""Книга со своей разметкой не должна уезжать в запасной путь."""
src = tmp / "book.epub"
build_epub(src)
out = tmp / "book.html"
epub2html.convert(src, out)
text = out.read_text(encoding="utf-8")
assert text.count("<h1>") == 3, "заголовки книги должны остаться её собственными"
assert "def main():" in text
if __name__ == "__main__":
with tempfile.TemporaryDirectory() as d:
test_convert(Path(d))
for case in (test_convert, test_flat_dump_recovered,
test_real_markup_wins_over_css):
with tempfile.TemporaryDirectory() as d:
case(Path(d))
print("OK")