EPUB как второй вход в конвейер; листинги кода не переводятся и не ломают приёмку

- epub2html.py: EPUB -> тот же XHTML, что pdf2html.py, только стандартная
  библиотека. Уровень глав определяется по книге, а не берётся из <h1>:
  в EPUB там обычно название и части.
- pdf2html.py: листинг узнаётся по флагу моноширинного шрифта PyMuPDF,
  переносы и отступы внутри <pre> сохраняются, де-дефисация к коду не
  применяется.
- translate.py: <pre> исключён из проверок языка (английский код утягивал
  долю кириллицы ниже порога приёмки), рабочий каталог привязан к книге,
  устаревшие главы прошлого прогона чистятся, инлайновый <code> переживает
  переводчика.
- bookhtml.py: общий каркас документа и CSS для обоих входов.
- Тесты: test_epub2html.py на собранном в памяти EPUB, четыре новых случая
  в test_translate.py.
This commit is contained in:
chesirecatt
2026-08-20 17:58:09 +03:00
parent 6c81673b4f
commit db9fe5e0fa
7 changed files with 614 additions and 43 deletions
+119
View File
@@ -0,0 +1,119 @@
#!/usr/bin/env python3
"""Самопроверка конвертера EPUB без сети: python3 test_epub2html.py"""
import tempfile
import zipfile
from pathlib import Path
import epub2html
CONTAINER = """<?xml version="1.0"?>
<container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
<rootfiles><rootfile full-path="OEBPS/content.opf"
media-type="application/oebps-package+xml"/></rootfiles>
</container>"""
OPF = """<?xml version="1.0"?>
<package xmlns="http://www.idpf.org/2007/opf" version="3.0">
<metadata/>
<manifest>
<item id="c1" href="ch1.xhtml" media-type="application/xhtml+xml"/>
<item id="c2" href="ch2.xhtml" media-type="application/xhtml+xml"/>
<item id="c3" href="ch3.xhtml" media-type="application/xhtml+xml"/>
<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml"
properties="nav"/>
<item id="pic" href="img/fig.png" media-type="image/png"/>
<item id="cov" href="img/cover.png" media-type="image/png"
properties="cover-image"/>
</manifest>
<spine>
<itemref idref="nav"/><itemref idref="c1"/>
<itemref idref="c2"/><itemref idref="c3"/>
</spine>
</package>"""
CH1 = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><head><title>skip me</title></head>
<body>
<h2>Chapter One</h2>
<p>The <em>lazy</em> fox calls <code>fetch_data()</code> twice&nbsp;a day.</p>
<pre>def main():
x = 1 - 2
return x</pre>
<ul><li>first item</li><li>second item</li></ul>
<table><tr><th>Name</th><th>Value</th></tr><tr><td>alpha</td><td>1</td></tr></table>
<p>Broken <b>markup that never closes.</p>
<p><img src="img/fig.png" alt="figure"/></p>
<script>var noise = 1;</script>
</body></html>"""
CH2 = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<h2>Chapter Two</h2><p>Second chapter body text.</p>
</body></html>"""
CH3 = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<h2>Chapter Three</h2><div>Loose text outside any block tag.</div>
</body></html>"""
NAV = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<nav><ol><li>Chapter One</li></ol></nav></body></html>"""
PNG = bytes.fromhex("89504e470d0a1a0a") # достаточно как содержимое файла
def build_epub(path):
with zipfile.ZipFile(path, "w") as z:
z.writestr("mimetype", "application/epub+zip")
z.writestr("META-INF/container.xml", CONTAINER)
z.writestr("OEBPS/content.opf", OPF)
z.writestr("OEBPS/ch1.xhtml", CH1)
z.writestr("OEBPS/ch2.xhtml", CH2)
z.writestr("OEBPS/ch3.xhtml", CH3)
z.writestr("OEBPS/nav.xhtml", NAV)
z.writestr("OEBPS/img/fig.png", PNG)
z.writestr("OEBPS/img/cover.png", PNG)
def test_convert(tmp: Path):
src = tmp / "book.epub"
build_epub(src)
out = tmp / "book.html"
epub2html.convert(src, out)
text = out.read_text(encoding="utf-8")
lines = text.splitlines()
# Листинг: переносы строк, отступы и минусы не тронуты
pre = text[text.index("<pre>"):text.index("</pre>")]
assert "def main():\n x = 1 - 2\n return x" in pre, pre
assert "<code>" not in pre, "внутри листинга инлайновая разметка не нужна"
# Проза: инлайн переведён в наш набор тегов, сущности раскрыты
para = next(l for l in lines if "fox" in l)
assert "<i>lazy</i>" in para and "<code>fetch_data()</code>" in para, para
assert "&nbsp;" not in para and " " not in para, para
assert '<p class="li">first item</p>' in lines
assert '<p class="row">Name | Value</p>' in lines
assert '<p class="row">alpha | 1</p>' in lines
assert '<img src="images/img001.png"/>' in text
assert (out.parent / "images" / "cover.png").exists(), "обложка не извлечена"
# Несбалансированная чужая разметка закрывается на границе блока
broken = next(l for l in lines if "never closes" in l)
assert broken.count("<b>") == broken.count("</b>") == 1, broken
assert "noise" not in text, "<script> не должен попадать в книгу"
assert "skip me" not in text, "<title> документа — не текст книги"
assert "Loose text outside any block tag." in text, "текст вне блока потерян"
# nav-документ выкинут, h2 подняты до h1 — иначе книга уедет одной главой
assert text.count("Chapter One") == 1, "оглавление попало в текст"
assert text.count("<h1>") == 3 and "<h2>" not in text
if __name__ == "__main__":
with tempfile.TemporaryDirectory() as d:
test_convert(Path(d))
print("OK")