Files
pdf2epub/scripts/test_epub2html.py
T
chesirecatt db9fe5e0fa EPUB как второй вход в конвейер; листинги кода не переводятся и не ломают приёмку
- epub2html.py: EPUB -> тот же XHTML, что pdf2html.py, только стандартная
  библиотека. Уровень глав определяется по книге, а не берётся из <h1>:
  в EPUB там обычно название и части.
- pdf2html.py: листинг узнаётся по флагу моноширинного шрифта PyMuPDF,
  переносы и отступы внутри <pre> сохраняются, де-дефисация к коду не
  применяется.
- translate.py: <pre> исключён из проверок языка (английский код утягивал
  долю кириллицы ниже порога приёмки), рабочий каталог привязан к книге,
  устаревшие главы прошлого прогона чистятся, инлайновый <code> переживает
  переводчика.
- bookhtml.py: общий каркас документа и CSS для обоих входов.
- Тесты: test_epub2html.py на собранном в памяти EPUB, четыре новых случая
  в test_translate.py.
2026-08-20 17:58:09 +03:00

120 lines
4.8 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Самопроверка конвертера EPUB без сети: python3 test_epub2html.py"""
import tempfile
import zipfile
from pathlib import Path
import epub2html
CONTAINER = """<?xml version="1.0"?>
<container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
<rootfiles><rootfile full-path="OEBPS/content.opf"
media-type="application/oebps-package+xml"/></rootfiles>
</container>"""
OPF = """<?xml version="1.0"?>
<package xmlns="http://www.idpf.org/2007/opf" version="3.0">
<metadata/>
<manifest>
<item id="c1" href="ch1.xhtml" media-type="application/xhtml+xml"/>
<item id="c2" href="ch2.xhtml" media-type="application/xhtml+xml"/>
<item id="c3" href="ch3.xhtml" media-type="application/xhtml+xml"/>
<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml"
properties="nav"/>
<item id="pic" href="img/fig.png" media-type="image/png"/>
<item id="cov" href="img/cover.png" media-type="image/png"
properties="cover-image"/>
</manifest>
<spine>
<itemref idref="nav"/><itemref idref="c1"/>
<itemref idref="c2"/><itemref idref="c3"/>
</spine>
</package>"""
CH1 = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><head><title>skip me</title></head>
<body>
<h2>Chapter One</h2>
<p>The <em>lazy</em> fox calls <code>fetch_data()</code> twice&nbsp;a day.</p>
<pre>def main():
x = 1 - 2
return x</pre>
<ul><li>first item</li><li>second item</li></ul>
<table><tr><th>Name</th><th>Value</th></tr><tr><td>alpha</td><td>1</td></tr></table>
<p>Broken <b>markup that never closes.</p>
<p><img src="img/fig.png" alt="figure"/></p>
<script>var noise = 1;</script>
</body></html>"""
CH2 = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<h2>Chapter Two</h2><p>Second chapter body text.</p>
</body></html>"""
CH3 = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<h2>Chapter Three</h2><div>Loose text outside any block tag.</div>
</body></html>"""
NAV = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml"><body>
<nav><ol><li>Chapter One</li></ol></nav></body></html>"""
PNG = bytes.fromhex("89504e470d0a1a0a") # достаточно как содержимое файла
def build_epub(path):
with zipfile.ZipFile(path, "w") as z:
z.writestr("mimetype", "application/epub+zip")
z.writestr("META-INF/container.xml", CONTAINER)
z.writestr("OEBPS/content.opf", OPF)
z.writestr("OEBPS/ch1.xhtml", CH1)
z.writestr("OEBPS/ch2.xhtml", CH2)
z.writestr("OEBPS/ch3.xhtml", CH3)
z.writestr("OEBPS/nav.xhtml", NAV)
z.writestr("OEBPS/img/fig.png", PNG)
z.writestr("OEBPS/img/cover.png", PNG)
def test_convert(tmp: Path):
src = tmp / "book.epub"
build_epub(src)
out = tmp / "book.html"
epub2html.convert(src, out)
text = out.read_text(encoding="utf-8")
lines = text.splitlines()
# Листинг: переносы строк, отступы и минусы не тронуты
pre = text[text.index("<pre>"):text.index("</pre>")]
assert "def main():\n x = 1 - 2\n return x" in pre, pre
assert "<code>" not in pre, "внутри листинга инлайновая разметка не нужна"
# Проза: инлайн переведён в наш набор тегов, сущности раскрыты
para = next(l for l in lines if "fox" in l)
assert "<i>lazy</i>" in para and "<code>fetch_data()</code>" in para, para
assert "&nbsp;" not in para and " " not in para, para
assert '<p class="li">first item</p>' in lines
assert '<p class="row">Name | Value</p>' in lines
assert '<p class="row">alpha | 1</p>' in lines
assert '<img src="images/img001.png"/>' in text
assert (out.parent / "images" / "cover.png").exists(), "обложка не извлечена"
# Несбалансированная чужая разметка закрывается на границе блока
broken = next(l for l in lines if "never closes" in l)
assert broken.count("<b>") == broken.count("</b>") == 1, broken
assert "noise" not in text, "<script> не должен попадать в книгу"
assert "skip me" not in text, "<title> документа — не текст книги"
assert "Loose text outside any block tag." in text, "текст вне блока потерян"
# nav-документ выкинут, h2 подняты до h1 — иначе книга уедет одной главой
assert text.count("Chapter One") == 1, "оглавление попало в текст"
assert text.count("<h1>") == 3 and "<h2>" not in text
if __name__ == "__main__":
with tempfile.TemporaryDirectory() as d:
test_convert(Path(d))
print("OK")