db9fe5e0fa
- epub2html.py: EPUB -> тот же XHTML, что pdf2html.py, только стандартная библиотека. Уровень глав определяется по книге, а не берётся из <h1>: в EPUB там обычно название и части. - pdf2html.py: листинг узнаётся по флагу моноширинного шрифта PyMuPDF, переносы и отступы внутри <pre> сохраняются, де-дефисация к коду не применяется. - translate.py: <pre> исключён из проверок языка (английский код утягивал долю кириллицы ниже порога приёмки), рабочий каталог привязан к книге, устаревшие главы прошлого прогона чистятся, инлайновый <code> переживает переводчика. - bookhtml.py: общий каркас документа и CSS для обоих входов. - Тесты: test_epub2html.py на собранном в памяти EPUB, четыре новых случая в test_translate.py.
120 lines
4.8 KiB
Python
120 lines
4.8 KiB
Python
#!/usr/bin/env python3
|
||
"""Самопроверка конвертера EPUB без сети: python3 test_epub2html.py"""
|
||
import tempfile
|
||
import zipfile
|
||
from pathlib import Path
|
||
|
||
import epub2html
|
||
|
||
CONTAINER = """<?xml version="1.0"?>
|
||
<container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
|
||
<rootfiles><rootfile full-path="OEBPS/content.opf"
|
||
media-type="application/oebps-package+xml"/></rootfiles>
|
||
</container>"""
|
||
|
||
OPF = """<?xml version="1.0"?>
|
||
<package xmlns="http://www.idpf.org/2007/opf" version="3.0">
|
||
<metadata/>
|
||
<manifest>
|
||
<item id="c1" href="ch1.xhtml" media-type="application/xhtml+xml"/>
|
||
<item id="c2" href="ch2.xhtml" media-type="application/xhtml+xml"/>
|
||
<item id="c3" href="ch3.xhtml" media-type="application/xhtml+xml"/>
|
||
<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml"
|
||
properties="nav"/>
|
||
<item id="pic" href="img/fig.png" media-type="image/png"/>
|
||
<item id="cov" href="img/cover.png" media-type="image/png"
|
||
properties="cover-image"/>
|
||
</manifest>
|
||
<spine>
|
||
<itemref idref="nav"/><itemref idref="c1"/>
|
||
<itemref idref="c2"/><itemref idref="c3"/>
|
||
</spine>
|
||
</package>"""
|
||
|
||
CH1 = """<?xml version="1.0" encoding="utf-8"?>
|
||
<html xmlns="http://www.w3.org/1999/xhtml"><head><title>skip me</title></head>
|
||
<body>
|
||
<h2>Chapter One</h2>
|
||
<p>The <em>lazy</em> fox calls <code>fetch_data()</code> twice a day.</p>
|
||
<pre>def main():
|
||
x = 1 - 2
|
||
return x</pre>
|
||
<ul><li>first item</li><li>second item</li></ul>
|
||
<table><tr><th>Name</th><th>Value</th></tr><tr><td>alpha</td><td>1</td></tr></table>
|
||
<p>Broken <b>markup that never closes.</p>
|
||
<p><img src="img/fig.png" alt="figure"/></p>
|
||
<script>var noise = 1;</script>
|
||
</body></html>"""
|
||
|
||
CH2 = """<?xml version="1.0" encoding="utf-8"?>
|
||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||
<h2>Chapter Two</h2><p>Second chapter body text.</p>
|
||
</body></html>"""
|
||
|
||
CH3 = """<?xml version="1.0" encoding="utf-8"?>
|
||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||
<h2>Chapter Three</h2><div>Loose text outside any block tag.</div>
|
||
</body></html>"""
|
||
|
||
NAV = """<?xml version="1.0" encoding="utf-8"?>
|
||
<html xmlns="http://www.w3.org/1999/xhtml"><body>
|
||
<nav><ol><li>Chapter One</li></ol></nav></body></html>"""
|
||
|
||
PNG = bytes.fromhex("89504e470d0a1a0a") # достаточно как содержимое файла
|
||
|
||
|
||
def build_epub(path):
|
||
with zipfile.ZipFile(path, "w") as z:
|
||
z.writestr("mimetype", "application/epub+zip")
|
||
z.writestr("META-INF/container.xml", CONTAINER)
|
||
z.writestr("OEBPS/content.opf", OPF)
|
||
z.writestr("OEBPS/ch1.xhtml", CH1)
|
||
z.writestr("OEBPS/ch2.xhtml", CH2)
|
||
z.writestr("OEBPS/ch3.xhtml", CH3)
|
||
z.writestr("OEBPS/nav.xhtml", NAV)
|
||
z.writestr("OEBPS/img/fig.png", PNG)
|
||
z.writestr("OEBPS/img/cover.png", PNG)
|
||
|
||
|
||
def test_convert(tmp: Path):
|
||
src = tmp / "book.epub"
|
||
build_epub(src)
|
||
out = tmp / "book.html"
|
||
epub2html.convert(src, out)
|
||
text = out.read_text(encoding="utf-8")
|
||
lines = text.splitlines()
|
||
|
||
# Листинг: переносы строк, отступы и минусы не тронуты
|
||
pre = text[text.index("<pre>"):text.index("</pre>")]
|
||
assert "def main():\n x = 1 - 2\n return x" in pre, pre
|
||
assert "<code>" not in pre, "внутри листинга инлайновая разметка не нужна"
|
||
|
||
# Проза: инлайн переведён в наш набор тегов, сущности раскрыты
|
||
para = next(l for l in lines if "fox" in l)
|
||
assert "<i>lazy</i>" in para and "<code>fetch_data()</code>" in para, para
|
||
assert " " not in para and " " not in para, para
|
||
|
||
assert '<p class="li">first item</p>' in lines
|
||
assert '<p class="row">Name | Value</p>' in lines
|
||
assert '<p class="row">alpha | 1</p>' in lines
|
||
assert '<img src="images/img001.png"/>' in text
|
||
assert (out.parent / "images" / "cover.png").exists(), "обложка не извлечена"
|
||
|
||
# Несбалансированная чужая разметка закрывается на границе блока
|
||
broken = next(l for l in lines if "never closes" in l)
|
||
assert broken.count("<b>") == broken.count("</b>") == 1, broken
|
||
|
||
assert "noise" not in text, "<script> не должен попадать в книгу"
|
||
assert "skip me" not in text, "<title> документа — не текст книги"
|
||
assert "Loose text outside any block tag." in text, "текст вне блока потерян"
|
||
|
||
# nav-документ выкинут, h2 подняты до h1 — иначе книга уедет одной главой
|
||
assert text.count("Chapter One") == 1, "оглавление попало в текст"
|
||
assert text.count("<h1>") == 3 and "<h2>" not in text
|
||
|
||
|
||
if __name__ == "__main__":
|
||
with tempfile.TemporaryDirectory() as d:
|
||
test_convert(Path(d))
|
||
print("OK")
|