Files
pdf2epub/scripts/test_epub2html.py
T
chesirecatt 1bacd38292 Восстановление структуры EPUB по CSS для дампов из PDF
Дамп, собранный конвертером из PDF, теряет всю семантику: ни заголовков, ни
листингов, только абзацы с обфусцированными классами. Такие книги уезжали в
перевод одним куском, а код — вместе с текстом.

Теперь при нулевом числе заголовков или листингов разметка восстанавливается из
таблицы стилей книги: три самых крупных кегля становятся уровнями заголовков,
моноширинная гарнитура — листингом, соседние строки листинга склеиваются в один
блок. Книга со своей разметкой в этот путь не попадает.

Проверено на двух настоящих книгах: дамп APoSD — 1833 плоских абзаца стали 29
главами и 58 листингами; Страуструп с собственной разметкой не изменился.
2026-08-20 19:11:51 +03:00

191 lines
7.8 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Самопроверка конвертера EPUB без сети: python3 test_epub2html.py"""
import tempfile
import zipfile
from pathlib import Path
import epub2html
CONTAINER = """<?xml version="1.0"?>
<container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
<rootfiles><rootfile full-path="OEBPS/content.opf"
media-type="application/oebps-package+xml"/></rootfiles>
</container>"""
OPF = """<?xml version="1.0"?>
<package xmlns="http://www.idpf.org/2007/opf" version="3.0">
<metadata/>
<manifest>
<item id="c1" href="ch1.xhtml" media-type="application/xhtml+xml"/>
<item id="c2" href="ch2.xhtml" media-type="application/xhtml+xml"/>
<item id="c3" href="ch3.xhtml" media-type="application/xhtml+xml"/>
<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml"
properties="nav"/>
<item id="pic" href="img/fig.png" media-type="image/png"/>
<item id="cov" href="img/cover.png" media-type="image/png"
properties="cover-image"/>
</manifest>
<spine>
<itemref idref="nav"/><itemref idref="c1"/>
<itemref idref="c2"/><itemref idref="c3"/>
</spine>
</package>"""
CH1 = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><head><title>skip me</title></head>
<body>
<h2>Chapter One</h2>
<p>The <em>lazy</em> fox calls <code>fetch_data()</code> twice&nbsp;a day.</p>
<pre>def main():
x = 1 - 2
return x</pre>
<ul><li>first item</li><li>second item</li></ul>
<table><tr><th>Name</th><th>Value</th></tr><tr><td>alpha</td><td>1</td></tr></table>
<p>Broken <b>markup that never closes.</p>
<p><img src="img/fig.png" alt="figure"/></p>
<script>var noise = 1;</script>
</body></html>"""
CH2 = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<h2>Chapter Two</h2><p>Second chapter body text.</p>
</body></html>"""
CH3 = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<h2>Chapter Three</h2><div>Loose text outside any block tag.</div>
</body></html>"""
NAV = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<nav><ol><li>Chapter One</li></ol></nav></body></html>"""
PNG = bytes.fromhex("89504e470d0a1a0a") # достаточно как содержимое файла
def build_epub(path):
with zipfile.ZipFile(path, "w") as z:
z.writestr("mimetype", "application/epub+zip")
z.writestr("META-INF/container.xml", CONTAINER)
z.writestr("OEBPS/content.opf", OPF)
z.writestr("OEBPS/ch1.xhtml", CH1)
z.writestr("OEBPS/ch2.xhtml", CH2)
z.writestr("OEBPS/ch3.xhtml", CH3)
z.writestr("OEBPS/nav.xhtml", NAV)
z.writestr("OEBPS/img/fig.png", PNG)
z.writestr("OEBPS/img/cover.png", PNG)
def test_convert(tmp: Path):
src = tmp / "book.epub"
build_epub(src)
out = tmp / "book.html"
epub2html.convert(src, out)
text = out.read_text(encoding="utf-8")
lines = text.splitlines()
# Листинг: переносы строк, отступы и минусы не тронуты
pre = text[text.index("<pre>"):text.index("</pre>")]
assert "def main():\n x = 1 - 2\n return x" in pre, pre
assert "<code>" not in pre, "внутри листинга инлайновая разметка не нужна"
# Проза: инлайн переведён в наш набор тегов, сущности раскрыты
para = next(l for l in lines if "fox" in l)
assert "<i>lazy</i>" in para and "<code>fetch_data()</code>" in para, para
assert "&nbsp;" not in para and " " not in para, para
assert '<p class="li">first item</p>' in lines
assert '<p class="row">Name | Value</p>' in lines
assert '<p class="row">alpha | 1</p>' in lines
assert '<img src="images/img001.png"/>' in text
assert (out.parent / "images" / "cover.png").exists(), "обложка не извлечена"
# Несбалансированная чужая разметка закрывается на границе блока
broken = next(l for l in lines if "never closes" in l)
assert broken.count("<b>") == broken.count("</b>") == 1, broken
assert "noise" not in text, "<script> не должен попадать в книгу"
assert "skip me" not in text, "<title> документа — не текст книги"
assert "Loose text outside any block tag." in text, "текст вне блока потерян"
# nav-документ выкинут, h2 подняты до h1 — иначе книга уедет одной главой
assert text.count("Chapter One") == 1, "оглавление попало в текст"
assert text.count("<h1>") == 3 and "<h2>" not in text
# Дамп, собранный конвертером из PDF: ни одного заголовка, ни одного <pre>,
# только <p> с обфусцированными классами и таблица стилей.
FLAT_CSS = """
.cls_head { font-size: 1.66667em; font-family: Reg; }
.cls_body { font-size: 1em; font-family: Reg; }
.cls_code { font-size: 0.9em; font-family: lucidasanstypewriter; }
"""
FLAT_OPF = """<?xml version="1.0"?>
<package xmlns="http://www.idpf.org/2007/opf" version="2.0">
<metadata/>
<manifest>
<item id="s" href="style.css" media-type="text/css"/>
<item id="a" href="p1.xhtml" media-type="application/xhtml+xml"/>
<item id="b" href="p2.xhtml" media-type="application/xhtml+xml"/>
<item id="c" href="p3.xhtml" media-type="application/xhtml+xml"/>
</manifest>
<spine><itemref idref="a"/><itemref idref="b"/><itemref idref="c"/></spine>
</package>"""
def flat_doc(n):
return """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<p class="cls_head">Chapter %d</p>
<p class="cls_head">Title Of Chapter %d</p>
<p class="cls_body">Body text of the chapter, long enough to be a paragraph.</p>
<p class="cls_code">def f(x):</p>
<p class="cls_code"> return x + 1</p>
<p class="cls_body">Closing paragraph of the chapter goes here.</p>
</body></html>""" % (n, n)
def test_flat_dump_recovered(tmp: Path):
src = tmp / "flat.epub"
with zipfile.ZipFile(src, "w") as z:
z.writestr("mimetype", "application/epub+zip")
z.writestr("META-INF/container.xml", CONTAINER)
z.writestr("OEBPS/content.opf", FLAT_OPF)
z.writestr("OEBPS/style.css", FLAT_CSS)
for i in (1, 2, 3):
z.writestr("OEBPS/p%d.xhtml" % i, flat_doc(i))
out = tmp / "flat.html"
epub2html.convert(src, out)
text = out.read_text(encoding="utf-8")
# заголовки восстановлены по кеглю и склеены с номером главы
assert "<h1>Chapter 2. Title Of Chapter 2</h1>" in text, text
assert text.count("<h1>") == 3
# листинг восстановлен по гарнитуре, соседние строки склеены в один блок
assert "<pre>def f(x):\n return x + 1</pre>" in text, text
assert text.count("<pre>") == 3
# обычный текст остался абзацем
assert "<p>Body text of the chapter" in text
def test_real_markup_wins_over_css(tmp: Path):
"""Книга со своей разметкой не должна уезжать в запасной путь."""
src = tmp / "book.epub"
build_epub(src)
out = tmp / "book.html"
epub2html.convert(src, out)
text = out.read_text(encoding="utf-8")
assert text.count("<h1>") == 3, "заголовки книги должны остаться её собственными"
assert "def main():" in text
if __name__ == "__main__":
for case in (test_convert, test_flat_dump_recovered,
test_real_markup_wins_over_css):
with tempfile.TemporaryDirectory() as d:
case(Path(d))
print("OK")