Восстановление структуры EPUB по CSS для дампов из PDF
Дамп, собранный конвертером из PDF, теряет всю семантику: ни заголовков, ни листингов, только абзацы с обфусцированными классами. Такие книги уезжали в перевод одним куском, а код — вместе с текстом. Теперь при нулевом числе заголовков или листингов разметка восстанавливается из таблицы стилей книги: три самых крупных кегля становятся уровнями заголовков, моноширинная гарнитура — листингом, соседние строки листинга склеиваются в один блок. Книга со своей разметкой в этот путь не попадает. Проверено на двух настоящих книгах: дамп APoSD — 1833 плоских абзаца стали 29 главами и 58 листингами; Страуструп с собственной разметкой не изменился.
This commit is contained in:
@@ -32,6 +32,15 @@ needed, not PyMuPDF) and convert with:
|
|||||||
|
|
||||||
`python scripts/epub2html.py book.epub out/book.html`
|
`python scripts/epub2html.py book.epub out/book.html`
|
||||||
|
|
||||||
|
**Dumps converted from PDF carry no semantics at all** — no headings, no `<pre>`,
|
||||||
|
just `<p class="class_s1e2">` with obfuscated names. The script detects this
|
||||||
|
(zero headings or zero listings) and rebuilds the structure from the book's own
|
||||||
|
stylesheet: the three largest font sizes become heading levels, a typewriter or
|
||||||
|
monospace family becomes `<pre>`, and adjacent listing paragraphs merge back into
|
||||||
|
one block. A book that carries its own markup never enters this path. Measured on
|
||||||
|
the dokumen.pub dump of Ousterhout's APoSD: 1833 flat paragraphs became 29
|
||||||
|
chapters and 58 listings.
|
||||||
|
|
||||||
It reports which heading level turned out to be the chapter level. In most
|
It reports which heading level turned out to be the chapter level. In most
|
||||||
EPUBs `<h1>` is the book title and the parts, while chapters are `<h2>` — the
|
EPUBs `<h1>` is the book title and the parts, while chapters are `<h2>` — the
|
||||||
script picks the deepest level that still gives a sane chapter count, because
|
script picks the deepest level that still gives a sane chapter count, because
|
||||||
|
|||||||
+99
-16
@@ -39,16 +39,57 @@ INLINE_MAP = {"i": "i", "em": "i", "cite": "i", "b": "b", "strong": "b",
|
|||||||
"code": "code", "kbd": "code", "samp": "code", "tt": "code",
|
"code": "code", "kbd": "code", "samp": "code", "tt": "code",
|
||||||
"var": "code"}
|
"var": "code"}
|
||||||
DROP = {"script", "style", "head", "title", "nav"}
|
DROP = {"script", "style", "head", "title", "nav"}
|
||||||
|
# Дампы, собранные конвертером из PDF, теряют всю семантику: ни заголовков, ни
|
||||||
|
# <pre>, только <p class="class_s1e2"> с обфусцированными именами. Разметку
|
||||||
|
# приходится восстанавливать из таблицы стилей — по кеглю и по гарнитуре.
|
||||||
|
MONO_FONT = re.compile(r"mono|courier|consol|typewriter|menlo|inconsolata|"
|
||||||
|
r"liberationmono|andalemono|lucidasanstype", re.I)
|
||||||
|
CSS_RULE = re.compile(r"\.([\w-]+)\s*\{([^}]*)\}")
|
||||||
IMG_EXT = {"image/jpeg": ".jpg", "image/png": ".png", "image/gif": ".gif",
|
IMG_EXT = {"image/jpeg": ".jpg", "image/png": ".png", "image/gif": ".gif",
|
||||||
"image/svg+xml": ".svg", "image/webp": ".webp"}
|
"image/svg+xml": ".svg", "image/webp": ".webp"}
|
||||||
|
|
||||||
|
|
||||||
|
def css_styles(zf):
|
||||||
|
"""class -> (кегль в em, моноширинный ли) из всех таблиц стилей книги."""
|
||||||
|
out = {}
|
||||||
|
for name in zf.namelist():
|
||||||
|
if not name.lower().endswith(".css"):
|
||||||
|
continue
|
||||||
|
for cls, body in CSS_RULE.findall(zf.read(name).decode("utf-8", "replace")):
|
||||||
|
size = out.get(cls, (0.0, False))[0]
|
||||||
|
m = re.search(r"font-size:\s*([\d.]+)\s*(em|rem|pt|px|%)", body)
|
||||||
|
if m:
|
||||||
|
v = float(m.group(1))
|
||||||
|
size = {"em": v, "rem": v, "pt": v / 12, "px": v / 16,
|
||||||
|
"%": v / 100}[m.group(2)]
|
||||||
|
fam = re.search(r"font-family:\s*([^;]+)", body)
|
||||||
|
mono = bool(fam and MONO_FONT.search(fam.group(1)))
|
||||||
|
was = out.get(cls, (0.0, False))
|
||||||
|
out[cls] = (size or was[0], mono or was[1])
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def derive_from_css(styles):
|
||||||
|
"""(класс -> уровень заголовка, множество моноширинных классов).
|
||||||
|
|
||||||
|
Заголовки — три самых крупных кегля выше основного текста. Больше трёх
|
||||||
|
уровней брать нельзя: дальше начинаются подписи и колонтитулы.
|
||||||
|
"""
|
||||||
|
sizes = sorted({s for s, _ in styles.values() if s > 1.05}, reverse=True)[:3]
|
||||||
|
level = {s: i + 1 for i, s in enumerate(sizes)}
|
||||||
|
head = {c: level[s] for c, (s, _) in styles.items() if s in level}
|
||||||
|
mono = {c for c, (_, m) in styles.items() if m}
|
||||||
|
return head, mono
|
||||||
|
|
||||||
|
|
||||||
class Reader(HTMLParser):
|
class Reader(HTMLParser):
|
||||||
"""Собирает блоки из одного XHTML-документа книги."""
|
"""Собирает блоки из одного XHTML-документа книги."""
|
||||||
|
|
||||||
def __init__(self, on_image):
|
def __init__(self, on_image, head_cls=None, mono_cls=None):
|
||||||
super().__init__(convert_charrefs=True)
|
super().__init__(convert_charrefs=True)
|
||||||
self.on_image = on_image
|
self.on_image = on_image
|
||||||
|
self.head_cls = head_cls or {}
|
||||||
|
self.mono_cls = mono_cls or set()
|
||||||
self.blocks = []
|
self.blocks = []
|
||||||
self.cur = None # (tag, cls, [куски])
|
self.cur = None # (tag, cls, [куски])
|
||||||
self.open_inline = [] # незакрытые инлайновые теги текущего блока
|
self.open_inline = [] # незакрытые инлайновые теги текущего блока
|
||||||
@@ -69,8 +110,13 @@ class Reader(HTMLParser):
|
|||||||
else:
|
else:
|
||||||
txt = re.sub(r"\s+", " ", txt).strip()
|
txt = re.sub(r"\s+", " ", txt).strip()
|
||||||
txt = re.sub(r"<(i|b|code)>(\s*)</\1>", r"\2", txt) # пустая разметка
|
txt = re.sub(r"<(i|b|code)>(\s*)</\1>", r"\2", txt) # пустая разметка
|
||||||
if txt.strip():
|
if not txt.strip():
|
||||||
self.blocks.append((tag, cls, txt))
|
return
|
||||||
|
if tag == "pre" and self.blocks and self.blocks[-1][0] == "pre":
|
||||||
|
prev = self.blocks[-1]
|
||||||
|
self.blocks[-1] = (prev[0], prev[1], prev[2] + "\n" + txt)
|
||||||
|
return
|
||||||
|
self.blocks.append((tag, cls, txt))
|
||||||
|
|
||||||
def start(self, tag, cls):
|
def start(self, tag, cls):
|
||||||
self.flush()
|
self.flush()
|
||||||
@@ -98,7 +144,16 @@ class Reader(HTMLParser):
|
|||||||
self.cur[2].append("\n" if self.cur[0] == "pre" else " ")
|
self.cur[2].append("\n" if self.cur[0] == "pre" else " ")
|
||||||
return
|
return
|
||||||
if tag in BLOCK_MAP:
|
if tag in BLOCK_MAP:
|
||||||
self.start(*BLOCK_MAP[tag])
|
out_tag, out_cls = BLOCK_MAP[tag]
|
||||||
|
if out_tag in ("p", "pre"):
|
||||||
|
for c in (a.get("class") or "").split():
|
||||||
|
if c in self.head_cls:
|
||||||
|
out_tag, out_cls = "h%d" % self.head_cls[c], None
|
||||||
|
break
|
||||||
|
if c in self.mono_cls:
|
||||||
|
out_tag, out_cls = "pre", None
|
||||||
|
break
|
||||||
|
self.start(out_tag, out_cls)
|
||||||
return
|
return
|
||||||
if tag in ("td", "th") and self.cur and self.cur[0] == "p" \
|
if tag in ("td", "th") and self.cur and self.cur[0] == "p" \
|
||||||
and self.cur[1] == "row" and self.cur[2]:
|
and self.cur[1] == "row" and self.cur[2]:
|
||||||
@@ -220,26 +275,54 @@ def convert(src, out):
|
|||||||
if cover:
|
if cover:
|
||||||
extract(cover, stem="cover")
|
extract(cover, stem="cover")
|
||||||
|
|
||||||
for doc in docs:
|
def parse_all(head_cls=None, mono_cls=None):
|
||||||
if doc not in names:
|
got = []
|
||||||
print("нет в архиве, пропущен: %s" % doc, file=sys.stderr)
|
for doc in docs:
|
||||||
continue
|
if doc not in names:
|
||||||
here = posixpath.dirname(doc)
|
print("нет в архиве, пропущен: %s" % doc, file=sys.stderr)
|
||||||
|
continue
|
||||||
|
here = posixpath.dirname(doc)
|
||||||
|
|
||||||
def on_image(src_attr, here=here):
|
def on_image(src_attr, here=here):
|
||||||
target = posixpath.normpath(posixpath.join(here, src_attr.split("#")[0]))
|
target = posixpath.normpath(
|
||||||
return extract(target)
|
posixpath.join(here, src_attr.split("#")[0]))
|
||||||
|
return extract(target)
|
||||||
|
|
||||||
r = Reader(on_image)
|
r = Reader(on_image, head_cls, mono_cls)
|
||||||
r.feed(zf.read(doc).decode("utf-8", "replace"))
|
r.feed(zf.read(doc).decode("utf-8", "replace"))
|
||||||
r.close()
|
r.close()
|
||||||
parts.extend(r.blocks)
|
got.extend(r.blocks)
|
||||||
|
return got
|
||||||
|
|
||||||
|
parts = parse_all()
|
||||||
|
has_head = any(re.fullmatch(r"h[1-6]", t) for t, _, _ in parts)
|
||||||
|
has_pre = any(t == "pre" for t, _, _ in parts)
|
||||||
|
if not (has_head and has_pre):
|
||||||
|
head_cls, mono_cls = derive_from_css(css_styles(zf))
|
||||||
|
if (not has_head and head_cls) or (not has_pre and mono_cls):
|
||||||
|
print("семантики в книге нет (заголовки: %s, листинги: %s) — "
|
||||||
|
"восстанавливаю по CSS" % (has_head, has_pre))
|
||||||
|
seen.clear()
|
||||||
|
counter[0] = 0
|
||||||
|
parts = parse_all(head_cls if not has_head else None,
|
||||||
|
mono_cls if not has_pre else None)
|
||||||
|
|
||||||
title = out.stem
|
title = out.stem
|
||||||
lvl = chapter_level(parts)
|
lvl = chapter_level(parts)
|
||||||
parts = [(("h1" if int(t[1]) <= lvl else "h2", c, x)
|
parts = [(("h1" if int(t[1]) <= lvl else "h2", c, x)
|
||||||
if re.fullmatch(r"h[1-6]", t) else (t, c, x))
|
if re.fullmatch(r"h[1-6]", t) else (t, c, x))
|
||||||
for t, c, x in parts]
|
for t, c, x in parts]
|
||||||
|
# «Chapter 1» и «Introduction» приезжают отдельными заголовками: в дампе это
|
||||||
|
# две строки. Иначе каждая вторая глава состоит из одного блока.
|
||||||
|
merged = []
|
||||||
|
for part in parts:
|
||||||
|
if (part[0] == "h1" and merged and merged[-1][0] == "h1"
|
||||||
|
and len(merged[-1][2]) < 60):
|
||||||
|
merged[-1] = ("h1", None, merged[-1][2] + ". " + part[2])
|
||||||
|
continue
|
||||||
|
merged.append(part)
|
||||||
|
parts = merged
|
||||||
|
|
||||||
h1 = sum(1 for p in parts if p[0] == "h1")
|
h1 = sum(1 for p in parts if p[0] == "h1")
|
||||||
print("главы размечены h%d, глав получилось: %d" % (lvl, h1))
|
print("главы размечены h%d, глав получилось: %d" % (lvl, h1))
|
||||||
if h1 < 3:
|
if h1 < 3:
|
||||||
|
|||||||
@@ -113,7 +113,78 @@ def test_convert(tmp: Path):
|
|||||||
assert text.count("<h1>") == 3 and "<h2>" not in text
|
assert text.count("<h1>") == 3 and "<h2>" not in text
|
||||||
|
|
||||||
|
|
||||||
|
# Дамп, собранный конвертером из PDF: ни одного заголовка, ни одного <pre>,
|
||||||
|
# только <p> с обфусцированными классами и таблица стилей.
|
||||||
|
FLAT_CSS = """
|
||||||
|
.cls_head { font-size: 1.66667em; font-family: Reg; }
|
||||||
|
.cls_body { font-size: 1em; font-family: Reg; }
|
||||||
|
.cls_code { font-size: 0.9em; font-family: lucidasanstypewriter; }
|
||||||
|
"""
|
||||||
|
|
||||||
|
FLAT_OPF = """<?xml version="1.0"?>
|
||||||
|
<package xmlns="http://www.idpf.org/2007/opf" version="2.0">
|
||||||
|
<metadata/>
|
||||||
|
<manifest>
|
||||||
|
<item id="s" href="style.css" media-type="text/css"/>
|
||||||
|
<item id="a" href="p1.xhtml" media-type="application/xhtml+xml"/>
|
||||||
|
<item id="b" href="p2.xhtml" media-type="application/xhtml+xml"/>
|
||||||
|
<item id="c" href="p3.xhtml" media-type="application/xhtml+xml"/>
|
||||||
|
</manifest>
|
||||||
|
<spine><itemref idref="a"/><itemref idref="b"/><itemref idref="c"/></spine>
|
||||||
|
</package>"""
|
||||||
|
|
||||||
|
|
||||||
|
def flat_doc(n):
|
||||||
|
return """<?xml version="1.0" encoding="utf-8"?>
|
||||||
|
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||||||
|
<p class="cls_head">Chapter %d</p>
|
||||||
|
<p class="cls_head">Title Of Chapter %d</p>
|
||||||
|
<p class="cls_body">Body text of the chapter, long enough to be a paragraph.</p>
|
||||||
|
<p class="cls_code">def f(x):</p>
|
||||||
|
<p class="cls_code"> return x + 1</p>
|
||||||
|
<p class="cls_body">Closing paragraph of the chapter goes here.</p>
|
||||||
|
</body></html>""" % (n, n)
|
||||||
|
|
||||||
|
|
||||||
|
def test_flat_dump_recovered(tmp: Path):
|
||||||
|
src = tmp / "flat.epub"
|
||||||
|
with zipfile.ZipFile(src, "w") as z:
|
||||||
|
z.writestr("mimetype", "application/epub+zip")
|
||||||
|
z.writestr("META-INF/container.xml", CONTAINER)
|
||||||
|
z.writestr("OEBPS/content.opf", FLAT_OPF)
|
||||||
|
z.writestr("OEBPS/style.css", FLAT_CSS)
|
||||||
|
for i in (1, 2, 3):
|
||||||
|
z.writestr("OEBPS/p%d.xhtml" % i, flat_doc(i))
|
||||||
|
out = tmp / "flat.html"
|
||||||
|
epub2html.convert(src, out)
|
||||||
|
text = out.read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
# заголовки восстановлены по кеглю и склеены с номером главы
|
||||||
|
assert "<h1>Chapter 2. Title Of Chapter 2</h1>" in text, text
|
||||||
|
assert text.count("<h1>") == 3
|
||||||
|
|
||||||
|
# листинг восстановлен по гарнитуре, соседние строки склеены в один блок
|
||||||
|
assert "<pre>def f(x):\n return x + 1</pre>" in text, text
|
||||||
|
assert text.count("<pre>") == 3
|
||||||
|
|
||||||
|
# обычный текст остался абзацем
|
||||||
|
assert "<p>Body text of the chapter" in text
|
||||||
|
|
||||||
|
|
||||||
|
def test_real_markup_wins_over_css(tmp: Path):
|
||||||
|
"""Книга со своей разметкой не должна уезжать в запасной путь."""
|
||||||
|
src = tmp / "book.epub"
|
||||||
|
build_epub(src)
|
||||||
|
out = tmp / "book.html"
|
||||||
|
epub2html.convert(src, out)
|
||||||
|
text = out.read_text(encoding="utf-8")
|
||||||
|
assert text.count("<h1>") == 3, "заголовки книги должны остаться её собственными"
|
||||||
|
assert "def main():" in text
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
with tempfile.TemporaryDirectory() as d:
|
for case in (test_convert, test_flat_dump_recovered,
|
||||||
test_convert(Path(d))
|
test_real_markup_wins_over_css):
|
||||||
|
with tempfile.TemporaryDirectory() as d:
|
||||||
|
case(Path(d))
|
||||||
print("OK")
|
print("OK")
|
||||||
|
|||||||
Reference in New Issue
Block a user