Восстановление структуры EPUB по CSS для дампов из PDF
Дамп, собранный конвертером из PDF, теряет всю семантику: ни заголовков, ни листингов, только абзацы с обфусцированными классами. Такие книги уезжали в перевод одним куском, а код — вместе с текстом. Теперь при нулевом числе заголовков или листингов разметка восстанавливается из таблицы стилей книги: три самых крупных кегля становятся уровнями заголовков, моноширинная гарнитура — листингом, соседние строки листинга склеиваются в один блок. Книга со своей разметкой в этот путь не попадает. Проверено на двух настоящих книгах: дамп APoSD — 1833 плоских абзаца стали 29 главами и 58 листингами; Страуструп с собственной разметкой не изменился.
This commit is contained in:
+99
-16
@@ -39,16 +39,57 @@ INLINE_MAP = {"i": "i", "em": "i", "cite": "i", "b": "b", "strong": "b",
|
||||
"code": "code", "kbd": "code", "samp": "code", "tt": "code",
|
||||
"var": "code"}
|
||||
DROP = {"script", "style", "head", "title", "nav"}
|
||||
# Дампы, собранные конвертером из PDF, теряют всю семантику: ни заголовков, ни
|
||||
# <pre>, только <p class="class_s1e2"> с обфусцированными именами. Разметку
|
||||
# приходится восстанавливать из таблицы стилей — по кеглю и по гарнитуре.
|
||||
MONO_FONT = re.compile(r"mono|courier|consol|typewriter|menlo|inconsolata|"
|
||||
r"liberationmono|andalemono|lucidasanstype", re.I)
|
||||
CSS_RULE = re.compile(r"\.([\w-]+)\s*\{([^}]*)\}")
|
||||
IMG_EXT = {"image/jpeg": ".jpg", "image/png": ".png", "image/gif": ".gif",
|
||||
"image/svg+xml": ".svg", "image/webp": ".webp"}
|
||||
|
||||
|
||||
def css_styles(zf):
|
||||
"""class -> (кегль в em, моноширинный ли) из всех таблиц стилей книги."""
|
||||
out = {}
|
||||
for name in zf.namelist():
|
||||
if not name.lower().endswith(".css"):
|
||||
continue
|
||||
for cls, body in CSS_RULE.findall(zf.read(name).decode("utf-8", "replace")):
|
||||
size = out.get(cls, (0.0, False))[0]
|
||||
m = re.search(r"font-size:\s*([\d.]+)\s*(em|rem|pt|px|%)", body)
|
||||
if m:
|
||||
v = float(m.group(1))
|
||||
size = {"em": v, "rem": v, "pt": v / 12, "px": v / 16,
|
||||
"%": v / 100}[m.group(2)]
|
||||
fam = re.search(r"font-family:\s*([^;]+)", body)
|
||||
mono = bool(fam and MONO_FONT.search(fam.group(1)))
|
||||
was = out.get(cls, (0.0, False))
|
||||
out[cls] = (size or was[0], mono or was[1])
|
||||
return out
|
||||
|
||||
|
||||
def derive_from_css(styles):
|
||||
"""(класс -> уровень заголовка, множество моноширинных классов).
|
||||
|
||||
Заголовки — три самых крупных кегля выше основного текста. Больше трёх
|
||||
уровней брать нельзя: дальше начинаются подписи и колонтитулы.
|
||||
"""
|
||||
sizes = sorted({s for s, _ in styles.values() if s > 1.05}, reverse=True)[:3]
|
||||
level = {s: i + 1 for i, s in enumerate(sizes)}
|
||||
head = {c: level[s] for c, (s, _) in styles.items() if s in level}
|
||||
mono = {c for c, (_, m) in styles.items() if m}
|
||||
return head, mono
|
||||
|
||||
|
||||
class Reader(HTMLParser):
|
||||
"""Собирает блоки из одного XHTML-документа книги."""
|
||||
|
||||
def __init__(self, on_image):
|
||||
def __init__(self, on_image, head_cls=None, mono_cls=None):
|
||||
super().__init__(convert_charrefs=True)
|
||||
self.on_image = on_image
|
||||
self.head_cls = head_cls or {}
|
||||
self.mono_cls = mono_cls or set()
|
||||
self.blocks = []
|
||||
self.cur = None # (tag, cls, [куски])
|
||||
self.open_inline = [] # незакрытые инлайновые теги текущего блока
|
||||
@@ -69,8 +110,13 @@ class Reader(HTMLParser):
|
||||
else:
|
||||
txt = re.sub(r"\s+", " ", txt).strip()
|
||||
txt = re.sub(r"<(i|b|code)>(\s*)</\1>", r"\2", txt) # пустая разметка
|
||||
if txt.strip():
|
||||
self.blocks.append((tag, cls, txt))
|
||||
if not txt.strip():
|
||||
return
|
||||
if tag == "pre" and self.blocks and self.blocks[-1][0] == "pre":
|
||||
prev = self.blocks[-1]
|
||||
self.blocks[-1] = (prev[0], prev[1], prev[2] + "\n" + txt)
|
||||
return
|
||||
self.blocks.append((tag, cls, txt))
|
||||
|
||||
def start(self, tag, cls):
|
||||
self.flush()
|
||||
@@ -98,7 +144,16 @@ class Reader(HTMLParser):
|
||||
self.cur[2].append("\n" if self.cur[0] == "pre" else " ")
|
||||
return
|
||||
if tag in BLOCK_MAP:
|
||||
self.start(*BLOCK_MAP[tag])
|
||||
out_tag, out_cls = BLOCK_MAP[tag]
|
||||
if out_tag in ("p", "pre"):
|
||||
for c in (a.get("class") or "").split():
|
||||
if c in self.head_cls:
|
||||
out_tag, out_cls = "h%d" % self.head_cls[c], None
|
||||
break
|
||||
if c in self.mono_cls:
|
||||
out_tag, out_cls = "pre", None
|
||||
break
|
||||
self.start(out_tag, out_cls)
|
||||
return
|
||||
if tag in ("td", "th") and self.cur and self.cur[0] == "p" \
|
||||
and self.cur[1] == "row" and self.cur[2]:
|
||||
@@ -220,26 +275,54 @@ def convert(src, out):
|
||||
if cover:
|
||||
extract(cover, stem="cover")
|
||||
|
||||
for doc in docs:
|
||||
if doc not in names:
|
||||
print("нет в архиве, пропущен: %s" % doc, file=sys.stderr)
|
||||
continue
|
||||
here = posixpath.dirname(doc)
|
||||
def parse_all(head_cls=None, mono_cls=None):
|
||||
got = []
|
||||
for doc in docs:
|
||||
if doc not in names:
|
||||
print("нет в архиве, пропущен: %s" % doc, file=sys.stderr)
|
||||
continue
|
||||
here = posixpath.dirname(doc)
|
||||
|
||||
def on_image(src_attr, here=here):
|
||||
target = posixpath.normpath(posixpath.join(here, src_attr.split("#")[0]))
|
||||
return extract(target)
|
||||
def on_image(src_attr, here=here):
|
||||
target = posixpath.normpath(
|
||||
posixpath.join(here, src_attr.split("#")[0]))
|
||||
return extract(target)
|
||||
|
||||
r = Reader(on_image)
|
||||
r.feed(zf.read(doc).decode("utf-8", "replace"))
|
||||
r.close()
|
||||
parts.extend(r.blocks)
|
||||
r = Reader(on_image, head_cls, mono_cls)
|
||||
r.feed(zf.read(doc).decode("utf-8", "replace"))
|
||||
r.close()
|
||||
got.extend(r.blocks)
|
||||
return got
|
||||
|
||||
parts = parse_all()
|
||||
has_head = any(re.fullmatch(r"h[1-6]", t) for t, _, _ in parts)
|
||||
has_pre = any(t == "pre" for t, _, _ in parts)
|
||||
if not (has_head and has_pre):
|
||||
head_cls, mono_cls = derive_from_css(css_styles(zf))
|
||||
if (not has_head and head_cls) or (not has_pre and mono_cls):
|
||||
print("семантики в книге нет (заголовки: %s, листинги: %s) — "
|
||||
"восстанавливаю по CSS" % (has_head, has_pre))
|
||||
seen.clear()
|
||||
counter[0] = 0
|
||||
parts = parse_all(head_cls if not has_head else None,
|
||||
mono_cls if not has_pre else None)
|
||||
|
||||
title = out.stem
|
||||
lvl = chapter_level(parts)
|
||||
parts = [(("h1" if int(t[1]) <= lvl else "h2", c, x)
|
||||
if re.fullmatch(r"h[1-6]", t) else (t, c, x))
|
||||
for t, c, x in parts]
|
||||
# «Chapter 1» и «Introduction» приезжают отдельными заголовками: в дампе это
|
||||
# две строки. Иначе каждая вторая глава состоит из одного блока.
|
||||
merged = []
|
||||
for part in parts:
|
||||
if (part[0] == "h1" and merged and merged[-1][0] == "h1"
|
||||
and len(merged[-1][2]) < 60):
|
||||
merged[-1] = ("h1", None, merged[-1][2] + ". " + part[2])
|
||||
continue
|
||||
merged.append(part)
|
||||
parts = merged
|
||||
|
||||
h1 = sum(1 for p in parts if p[0] == "h1")
|
||||
print("главы размечены h%d, глав получилось: %d" % (lvl, h1))
|
||||
if h1 < 3:
|
||||
|
||||
@@ -113,7 +113,78 @@ def test_convert(tmp: Path):
|
||||
assert text.count("<h1>") == 3 and "<h2>" not in text
|
||||
|
||||
|
||||
# Дамп, собранный конвертером из PDF: ни одного заголовка, ни одного <pre>,
|
||||
# только <p> с обфусцированными классами и таблица стилей.
|
||||
FLAT_CSS = """
|
||||
.cls_head { font-size: 1.66667em; font-family: Reg; }
|
||||
.cls_body { font-size: 1em; font-family: Reg; }
|
||||
.cls_code { font-size: 0.9em; font-family: lucidasanstypewriter; }
|
||||
"""
|
||||
|
||||
FLAT_OPF = """<?xml version="1.0"?>
|
||||
<package xmlns="http://www.idpf.org/2007/opf" version="2.0">
|
||||
<metadata/>
|
||||
<manifest>
|
||||
<item id="s" href="style.css" media-type="text/css"/>
|
||||
<item id="a" href="p1.xhtml" media-type="application/xhtml+xml"/>
|
||||
<item id="b" href="p2.xhtml" media-type="application/xhtml+xml"/>
|
||||
<item id="c" href="p3.xhtml" media-type="application/xhtml+xml"/>
|
||||
</manifest>
|
||||
<spine><itemref idref="a"/><itemref idref="b"/><itemref idref="c"/></spine>
|
||||
</package>"""
|
||||
|
||||
|
||||
def flat_doc(n):
|
||||
return """<?xml version="1.0" encoding="utf-8"?>
|
||||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||||
<p class="cls_head">Chapter %d</p>
|
||||
<p class="cls_head">Title Of Chapter %d</p>
|
||||
<p class="cls_body">Body text of the chapter, long enough to be a paragraph.</p>
|
||||
<p class="cls_code">def f(x):</p>
|
||||
<p class="cls_code"> return x + 1</p>
|
||||
<p class="cls_body">Closing paragraph of the chapter goes here.</p>
|
||||
</body></html>""" % (n, n)
|
||||
|
||||
|
||||
def test_flat_dump_recovered(tmp: Path):
|
||||
src = tmp / "flat.epub"
|
||||
with zipfile.ZipFile(src, "w") as z:
|
||||
z.writestr("mimetype", "application/epub+zip")
|
||||
z.writestr("META-INF/container.xml", CONTAINER)
|
||||
z.writestr("OEBPS/content.opf", FLAT_OPF)
|
||||
z.writestr("OEBPS/style.css", FLAT_CSS)
|
||||
for i in (1, 2, 3):
|
||||
z.writestr("OEBPS/p%d.xhtml" % i, flat_doc(i))
|
||||
out = tmp / "flat.html"
|
||||
epub2html.convert(src, out)
|
||||
text = out.read_text(encoding="utf-8")
|
||||
|
||||
# заголовки восстановлены по кеглю и склеены с номером главы
|
||||
assert "<h1>Chapter 2. Title Of Chapter 2</h1>" in text, text
|
||||
assert text.count("<h1>") == 3
|
||||
|
||||
# листинг восстановлен по гарнитуре, соседние строки склеены в один блок
|
||||
assert "<pre>def f(x):\n return x + 1</pre>" in text, text
|
||||
assert text.count("<pre>") == 3
|
||||
|
||||
# обычный текст остался абзацем
|
||||
assert "<p>Body text of the chapter" in text
|
||||
|
||||
|
||||
def test_real_markup_wins_over_css(tmp: Path):
|
||||
"""Книга со своей разметкой не должна уезжать в запасной путь."""
|
||||
src = tmp / "book.epub"
|
||||
build_epub(src)
|
||||
out = tmp / "book.html"
|
||||
epub2html.convert(src, out)
|
||||
text = out.read_text(encoding="utf-8")
|
||||
assert text.count("<h1>") == 3, "заголовки книги должны остаться её собственными"
|
||||
assert "def main():" in text
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
test_convert(Path(d))
|
||||
for case in (test_convert, test_flat_dump_recovered,
|
||||
test_real_markup_wins_over_css):
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
case(Path(d))
|
||||
print("OK")
|
||||
|
||||
Reference in New Issue
Block a user