1bacd38292
Дамп, собранный конвертером из PDF, теряет всю семантику: ни заголовков, ни листингов, только абзацы с обфусцированными классами. Такие книги уезжали в перевод одним куском, а код — вместе с текстом. Теперь при нулевом числе заголовков или листингов разметка восстанавливается из таблицы стилей книги: три самых крупных кегля становятся уровнями заголовков, моноширинная гарнитура — листингом, соседние строки листинга склеиваются в один блок. Книга со своей разметкой в этот путь не попадает. Проверено на двух настоящих книгах: дамп APoSD — 1833 плоских абзаца стали 29 главами и 58 листингами; Страуструп с собственной разметкой не изменился.
191 lines
7.8 KiB
Python
191 lines
7.8 KiB
Python
#!/usr/bin/env python3
|
||
"""Самопроверка конвертера EPUB без сети: python3 test_epub2html.py"""
|
||
import tempfile
|
||
import zipfile
|
||
from pathlib import Path
|
||
|
||
import epub2html
|
||
|
||
CONTAINER = """<?xml version="1.0"?>
|
||
<container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
|
||
<rootfiles><rootfile full-path="OEBPS/content.opf"
|
||
media-type="application/oebps-package+xml"/></rootfiles>
|
||
</container>"""
|
||
|
||
OPF = """<?xml version="1.0"?>
|
||
<package xmlns="http://www.idpf.org/2007/opf" version="3.0">
|
||
<metadata/>
|
||
<manifest>
|
||
<item id="c1" href="ch1.xhtml" media-type="application/xhtml+xml"/>
|
||
<item id="c2" href="ch2.xhtml" media-type="application/xhtml+xml"/>
|
||
<item id="c3" href="ch3.xhtml" media-type="application/xhtml+xml"/>
|
||
<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml"
|
||
properties="nav"/>
|
||
<item id="pic" href="img/fig.png" media-type="image/png"/>
|
||
<item id="cov" href="img/cover.png" media-type="image/png"
|
||
properties="cover-image"/>
|
||
</manifest>
|
||
<spine>
|
||
<itemref idref="nav"/><itemref idref="c1"/>
|
||
<itemref idref="c2"/><itemref idref="c3"/>
|
||
</spine>
|
||
</package>"""
|
||
|
||
CH1 = """<?xml version="1.0" encoding="utf-8"?>
|
||
<html xmlns="http://www.w3.org/1999/xhtml"><head><title>skip me</title></head>
|
||
<body>
|
||
<h2>Chapter One</h2>
|
||
<p>The <em>lazy</em> fox calls <code>fetch_data()</code> twice a day.</p>
|
||
<pre>def main():
|
||
x = 1 - 2
|
||
return x</pre>
|
||
<ul><li>first item</li><li>second item</li></ul>
|
||
<table><tr><th>Name</th><th>Value</th></tr><tr><td>alpha</td><td>1</td></tr></table>
|
||
<p>Broken <b>markup that never closes.</p>
|
||
<p><img src="img/fig.png" alt="figure"/></p>
|
||
<script>var noise = 1;</script>
|
||
</body></html>"""
|
||
|
||
CH2 = """<?xml version="1.0" encoding="utf-8"?>
|
||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||
<h2>Chapter Two</h2><p>Second chapter body text.</p>
|
||
</body></html>"""
|
||
|
||
CH3 = """<?xml version="1.0" encoding="utf-8"?>
|
||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||
<h2>Chapter Three</h2><div>Loose text outside any block tag.</div>
|
||
</body></html>"""
|
||
|
||
NAV = """<?xml version="1.0" encoding="utf-8"?>
|
||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||
<nav><ol><li>Chapter One</li></ol></nav></body></html>"""
|
||
|
||
PNG = bytes.fromhex("89504e470d0a1a0a") # достаточно как содержимое файла
|
||
|
||
|
||
def build_epub(path):
|
||
with zipfile.ZipFile(path, "w") as z:
|
||
z.writestr("mimetype", "application/epub+zip")
|
||
z.writestr("META-INF/container.xml", CONTAINER)
|
||
z.writestr("OEBPS/content.opf", OPF)
|
||
z.writestr("OEBPS/ch1.xhtml", CH1)
|
||
z.writestr("OEBPS/ch2.xhtml", CH2)
|
||
z.writestr("OEBPS/ch3.xhtml", CH3)
|
||
z.writestr("OEBPS/nav.xhtml", NAV)
|
||
z.writestr("OEBPS/img/fig.png", PNG)
|
||
z.writestr("OEBPS/img/cover.png", PNG)
|
||
|
||
|
||
def test_convert(tmp: Path):
|
||
src = tmp / "book.epub"
|
||
build_epub(src)
|
||
out = tmp / "book.html"
|
||
epub2html.convert(src, out)
|
||
text = out.read_text(encoding="utf-8")
|
||
lines = text.splitlines()
|
||
|
||
# Листинг: переносы строк, отступы и минусы не тронуты
|
||
pre = text[text.index("<pre>"):text.index("</pre>")]
|
||
assert "def main():\n x = 1 - 2\n return x" in pre, pre
|
||
assert "<code>" not in pre, "внутри листинга инлайновая разметка не нужна"
|
||
|
||
# Проза: инлайн переведён в наш набор тегов, сущности раскрыты
|
||
para = next(l for l in lines if "fox" in l)
|
||
assert "<i>lazy</i>" in para and "<code>fetch_data()</code>" in para, para
|
||
assert " " not in para and " " not in para, para
|
||
|
||
assert '<p class="li">first item</p>' in lines
|
||
assert '<p class="row">Name | Value</p>' in lines
|
||
assert '<p class="row">alpha | 1</p>' in lines
|
||
assert '<img src="images/img001.png"/>' in text
|
||
assert (out.parent / "images" / "cover.png").exists(), "обложка не извлечена"
|
||
|
||
# Несбалансированная чужая разметка закрывается на границе блока
|
||
broken = next(l for l in lines if "never closes" in l)
|
||
assert broken.count("<b>") == broken.count("</b>") == 1, broken
|
||
|
||
assert "noise" not in text, "<script> не должен попадать в книгу"
|
||
assert "skip me" not in text, "<title> документа — не текст книги"
|
||
assert "Loose text outside any block tag." in text, "текст вне блока потерян"
|
||
|
||
# nav-документ выкинут, h2 подняты до h1 — иначе книга уедет одной главой
|
||
assert text.count("Chapter One") == 1, "оглавление попало в текст"
|
||
assert text.count("<h1>") == 3 and "<h2>" not in text
|
||
|
||
|
||
# Дамп, собранный конвертером из PDF: ни одного заголовка, ни одного <pre>,
|
||
# только <p> с обфусцированными классами и таблица стилей.
|
||
FLAT_CSS = """
|
||
.cls_head { font-size: 1.66667em; font-family: Reg; }
|
||
.cls_body { font-size: 1em; font-family: Reg; }
|
||
.cls_code { font-size: 0.9em; font-family: lucidasanstypewriter; }
|
||
"""
|
||
|
||
FLAT_OPF = """<?xml version="1.0"?>
|
||
<package xmlns="http://www.idpf.org/2007/opf" version="2.0">
|
||
<metadata/>
|
||
<manifest>
|
||
<item id="s" href="style.css" media-type="text/css"/>
|
||
<item id="a" href="p1.xhtml" media-type="application/xhtml+xml"/>
|
||
<item id="b" href="p2.xhtml" media-type="application/xhtml+xml"/>
|
||
<item id="c" href="p3.xhtml" media-type="application/xhtml+xml"/>
|
||
</manifest>
|
||
<spine><itemref idref="a"/><itemref idref="b"/><itemref idref="c"/></spine>
|
||
</package>"""
|
||
|
||
|
||
def flat_doc(n):
|
||
return """<?xml version="1.0" encoding="utf-8"?>
|
||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||
<p class="cls_head">Chapter %d</p>
|
||
<p class="cls_head">Title Of Chapter %d</p>
|
||
<p class="cls_body">Body text of the chapter, long enough to be a paragraph.</p>
|
||
<p class="cls_code">def f(x):</p>
|
||
<p class="cls_code"> return x + 1</p>
|
||
<p class="cls_body">Closing paragraph of the chapter goes here.</p>
|
||
</body></html>""" % (n, n)
|
||
|
||
|
||
def test_flat_dump_recovered(tmp: Path):
|
||
src = tmp / "flat.epub"
|
||
with zipfile.ZipFile(src, "w") as z:
|
||
z.writestr("mimetype", "application/epub+zip")
|
||
z.writestr("META-INF/container.xml", CONTAINER)
|
||
z.writestr("OEBPS/content.opf", FLAT_OPF)
|
||
z.writestr("OEBPS/style.css", FLAT_CSS)
|
||
for i in (1, 2, 3):
|
||
z.writestr("OEBPS/p%d.xhtml" % i, flat_doc(i))
|
||
out = tmp / "flat.html"
|
||
epub2html.convert(src, out)
|
||
text = out.read_text(encoding="utf-8")
|
||
|
||
# заголовки восстановлены по кеглю и склеены с номером главы
|
||
assert "<h1>Chapter 2. Title Of Chapter 2</h1>" in text, text
|
||
assert text.count("<h1>") == 3
|
||
|
||
# листинг восстановлен по гарнитуре, соседние строки склеены в один блок
|
||
assert "<pre>def f(x):\n return x + 1</pre>" in text, text
|
||
assert text.count("<pre>") == 3
|
||
|
||
# обычный текст остался абзацем
|
||
assert "<p>Body text of the chapter" in text
|
||
|
||
|
||
def test_real_markup_wins_over_css(tmp: Path):
|
||
"""Книга со своей разметкой не должна уезжать в запасной путь."""
|
||
src = tmp / "book.epub"
|
||
build_epub(src)
|
||
out = tmp / "book.html"
|
||
epub2html.convert(src, out)
|
||
text = out.read_text(encoding="utf-8")
|
||
assert text.count("<h1>") == 3, "заголовки книги должны остаться её собственными"
|
||
assert "def main():" in text
|
||
|
||
|
||
if __name__ == "__main__":
|
||
for case in (test_convert, test_flat_dump_recovered,
|
||
test_real_markup_wins_over_css):
|
||
with tempfile.TemporaryDirectory() as d:
|
||
case(Path(d))
|
||
print("OK")
|