EPUB как второй вход в конвейер; листинги кода не переводятся и не ломают приёмку
- epub2html.py: EPUB -> тот же XHTML, что pdf2html.py, только стандартная библиотека. Уровень глав определяется по книге, а не берётся из <h1>: в EPUB там обычно название и части. - pdf2html.py: листинг узнаётся по флагу моноширинного шрифта PyMuPDF, переносы и отступы внутри <pre> сохраняются, де-дефисация к коду не применяется. - translate.py: <pre> исключён из проверок языка (английский код утягивал долю кириллицы ниже порога приёмки), рабочий каталог привязан к книге, устаревшие главы прошлого прогона чистятся, инлайновый <code> переживает переводчика. - bookhtml.py: общий каркас документа и CSS для обоих входов. - Тесты: test_epub2html.py на собранном в памяти EPUB, четыре новых случая в test_translate.py.
This commit is contained in:
@@ -0,0 +1,119 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Самопроверка конвертера EPUB без сети: python3 test_epub2html.py"""
|
||||
import tempfile
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
import epub2html
|
||||
|
||||
CONTAINER = """<?xml version="1.0"?>
|
||||
<container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
|
||||
<rootfiles><rootfile full-path="OEBPS/content.opf"
|
||||
media-type="application/oebps-package+xml"/></rootfiles>
|
||||
</container>"""
|
||||
|
||||
OPF = """<?xml version="1.0"?>
|
||||
<package xmlns="http://www.idpf.org/2007/opf" version="3.0">
|
||||
<metadata/>
|
||||
<manifest>
|
||||
<item id="c1" href="ch1.xhtml" media-type="application/xhtml+xml"/>
|
||||
<item id="c2" href="ch2.xhtml" media-type="application/xhtml+xml"/>
|
||||
<item id="c3" href="ch3.xhtml" media-type="application/xhtml+xml"/>
|
||||
<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml"
|
||||
properties="nav"/>
|
||||
<item id="pic" href="img/fig.png" media-type="image/png"/>
|
||||
<item id="cov" href="img/cover.png" media-type="image/png"
|
||||
properties="cover-image"/>
|
||||
</manifest>
|
||||
<spine>
|
||||
<itemref idref="nav"/><itemref idref="c1"/>
|
||||
<itemref idref="c2"/><itemref idref="c3"/>
|
||||
</spine>
|
||||
</package>"""
|
||||
|
||||
CH1 = """<?xml version="1.0" encoding="utf-8"?>
|
||||
<html xmlns="http://www.w3.org/1999/xhtml"><head><title>skip me</title></head>
|
||||
<body>
|
||||
<h2>Chapter One</h2>
|
||||
<p>The <em>lazy</em> fox calls <code>fetch_data()</code> twice a day.</p>
|
||||
<pre>def main():
|
||||
x = 1 - 2
|
||||
return x</pre>
|
||||
<ul><li>first item</li><li>second item</li></ul>
|
||||
<table><tr><th>Name</th><th>Value</th></tr><tr><td>alpha</td><td>1</td></tr></table>
|
||||
<p>Broken <b>markup that never closes.</p>
|
||||
<p><img src="img/fig.png" alt="figure"/></p>
|
||||
<script>var noise = 1;</script>
|
||||
</body></html>"""
|
||||
|
||||
CH2 = """<?xml version="1.0" encoding="utf-8"?>
|
||||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||||
<h2>Chapter Two</h2><p>Second chapter body text.</p>
|
||||
</body></html>"""
|
||||
|
||||
CH3 = """<?xml version="1.0" encoding="utf-8"?>
|
||||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||||
<h2>Chapter Three</h2><div>Loose text outside any block tag.</div>
|
||||
</body></html>"""
|
||||
|
||||
NAV = """<?xml version="1.0" encoding="utf-8"?>
|
||||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||||
<nav><ol><li>Chapter One</li></ol></nav></body></html>"""
|
||||
|
||||
PNG = bytes.fromhex("89504e470d0a1a0a") # достаточно как содержимое файла
|
||||
|
||||
|
||||
def build_epub(path):
|
||||
with zipfile.ZipFile(path, "w") as z:
|
||||
z.writestr("mimetype", "application/epub+zip")
|
||||
z.writestr("META-INF/container.xml", CONTAINER)
|
||||
z.writestr("OEBPS/content.opf", OPF)
|
||||
z.writestr("OEBPS/ch1.xhtml", CH1)
|
||||
z.writestr("OEBPS/ch2.xhtml", CH2)
|
||||
z.writestr("OEBPS/ch3.xhtml", CH3)
|
||||
z.writestr("OEBPS/nav.xhtml", NAV)
|
||||
z.writestr("OEBPS/img/fig.png", PNG)
|
||||
z.writestr("OEBPS/img/cover.png", PNG)
|
||||
|
||||
|
||||
def test_convert(tmp: Path):
|
||||
src = tmp / "book.epub"
|
||||
build_epub(src)
|
||||
out = tmp / "book.html"
|
||||
epub2html.convert(src, out)
|
||||
text = out.read_text(encoding="utf-8")
|
||||
lines = text.splitlines()
|
||||
|
||||
# Листинг: переносы строк, отступы и минусы не тронуты
|
||||
pre = text[text.index("<pre>"):text.index("</pre>")]
|
||||
assert "def main():\n x = 1 - 2\n return x" in pre, pre
|
||||
assert "<code>" not in pre, "внутри листинга инлайновая разметка не нужна"
|
||||
|
||||
# Проза: инлайн переведён в наш набор тегов, сущности раскрыты
|
||||
para = next(l for l in lines if "fox" in l)
|
||||
assert "<i>lazy</i>" in para and "<code>fetch_data()</code>" in para, para
|
||||
assert " " not in para and " " not in para, para
|
||||
|
||||
assert '<p class="li">first item</p>' in lines
|
||||
assert '<p class="row">Name | Value</p>' in lines
|
||||
assert '<p class="row">alpha | 1</p>' in lines
|
||||
assert '<img src="images/img001.png"/>' in text
|
||||
assert (out.parent / "images" / "cover.png").exists(), "обложка не извлечена"
|
||||
|
||||
# Несбалансированная чужая разметка закрывается на границе блока
|
||||
broken = next(l for l in lines if "never closes" in l)
|
||||
assert broken.count("<b>") == broken.count("</b>") == 1, broken
|
||||
|
||||
assert "noise" not in text, "<script> не должен попадать в книгу"
|
||||
assert "skip me" not in text, "<title> документа — не текст книги"
|
||||
assert "Loose text outside any block tag." in text, "текст вне блока потерян"
|
||||
|
||||
# nav-документ выкинут, h2 подняты до h1 — иначе книга уедет одной главой
|
||||
assert text.count("Chapter One") == 1, "оглавление попало в текст"
|
||||
assert text.count("<h1>") == 3 and "<h2>" not in text
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
test_convert(Path(d))
|
||||
print("OK")
|
||||
Reference in New Issue
Block a user