From f19410d9ce23c6293e08e6e55ddd8c12683dbdc5 Mon Sep 17 00:00:00 2001 From: chesirecatt Date: Fri, 21 Aug 2026 19:45:37 +0300 Subject: [PATCH] =?UTF-8?q?=D0=92=D1=85=D0=BE=D0=B4=20fb2=20=D0=B8=20?= =?UTF-8?q?=D0=BC=D0=B5=D1=82=D0=B0=D0=B4=D0=B0=D0=BD=D0=BD=D1=8B=D0=B5=20?= =?UTF-8?q?=D1=86=D0=B8=D0=BA=D0=BB=D0=B0=20=D0=B2=20EPUB?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit fb2html.py — третий вход в конвейер. FB2 в домашней библиотеке основной формат (704 тысячи файлов), и без него конвейер до них не дотягивался. Разметка fb2 семантическая, поэтому задача та же, что у epub2html.py: свести чужие теги к нашим. Уровень заголовка берётся из вложенности section. pack_epub.py: --series и --series-index. Пишутся в двух видах — calibre-совместимый meta понимают почти все читалки, belongs-to-collection требует EPUB 3. Без них читалка не выстраивает книги цикла по порядку. --- scripts/fb2html.py | 163 +++++++++++++++++++++++++++++++++++++++++++ scripts/pack_epub.py | 27 ++++++- 2 files changed, 187 insertions(+), 3 deletions(-) create mode 100644 scripts/fb2html.py diff --git a/scripts/fb2html.py b/scripts/fb2html.py new file mode 100644 index 0000000..0c678d0 --- /dev/null +++ b/scripts/fb2html.py @@ -0,0 +1,163 @@ +#!/usr/bin/env python3 +"""FB2 -> тот же XHTML, что выдают pdf2html.py и epub2html.py. + +FB2 в этой библиотеке основной формат (704 тысячи файлов), и без этого входа +конвейер до них не дотягивается. Разметка у fb2 уже семантическая, поэтому +задача та же, что у epub2html.py: привести чужие теги к нашим. + +Только стандартная библиотека. +""" +import argparse +import html +import re +import sys +import zipfile +from html.parser import HTMLParser +from pathlib import Path + +import bookhtml + +BLOCK = {"p": ("p", None), "v": ("p", "li"), "subtitle": ("h2", None), + "text-author": ("p", "note"), "th": ("p", "row"), "td": ("p", "row")} +INLINE = {"emphasis": "i", "strong": "b", "code": "code", "sub": "i", "sup": "i"} +DROP = {"description", "binary", "stylesheet"} + + +class Reader(HTMLParser): + def __init__(self): + super().__init__(convert_charrefs=True) + self.blocks, self.cur, self.open_i, self.drop = [], None, [], 0 + self.depth = 0 # вложенность section: даёт уровень заголовка + self.in_title = False + + def flush(self): + if not self.cur: + return + tag, cls, parts = self.cur + self.cur = None + for t in reversed(self.open_i): + parts.append("" % t) + self.open_i = [] + txt = re.sub(r"\s+", " ", "".join(parts)).strip() + if txt: + self.blocks.append((tag, cls, txt)) + + def handle_starttag(self, tag, attrs): + if tag in DROP: + self.drop += 1 + return + if self.drop: + return + if tag == "section": + self.depth += 1 + return + if tag == "title": + self.flush() + self.in_title = True + return + if tag == "empty-line": + self.flush() + return + if tag == "image": + return # картинки fb2 лежат в base64, пропускаем + if self.in_title and tag == "p": + # заголовок раздела: уровень по вложенности section + self.flush() + self.cur = ("h1" if self.depth <= 2 else "h2", None, []) + return + if tag in BLOCK: + self.flush() + self.cur = (BLOCK[tag][0], BLOCK[tag][1], []) + return + if tag in INLINE and self.cur: + out = INLINE[tag] + self.open_i.append(out) + self.cur[2].append("<%s>" % out) + + def handle_endtag(self, tag): + if tag in DROP: + self.drop = max(0, self.drop - 1) + return + if self.drop: + return + if tag == "section": + self.flush() + self.depth = max(0, self.depth - 1) + return + if tag == "title": + self.flush() + self.in_title = False + return + if tag in BLOCK: + self.flush() + return + if tag in INLINE and self.cur: + out = INLINE[tag] + if out in self.open_i: + self.open_i.remove(out) + self.cur[2].append("" % out) + + def handle_data(self, data): + if self.drop or not self.cur: + return + self.cur[2].append(html.escape(data, quote=False)) + + def close(self): + super().close() + self.flush() + + +def read_fb2(path): + p = Path(path) + if p.suffix.lower() == ".zip" or zipfile.is_zipfile(p): + with zipfile.ZipFile(p) as z: + name = next(n for n in z.namelist() if n.lower().endswith(".fb2")) + raw = z.read(name) + else: + raw = p.read_bytes() + enc = "utf-8" + m = re.search(rb'encoding="([\w-]+)"', raw[:200]) + if m: + enc = m.group(1).decode("ascii", "ignore") + return raw.decode(enc, "replace") + + +def meta(text): + """(автор, название) из description — для метаданных EPUB.""" + def tag(name): + m = re.search(r"<%s>(.*?)" % (name, name), text, re.S) + return re.sub(r"<[^>]+>", " ", m.group(1)).strip() if m else "" + ti = re.search(r"(.*?)", text, re.S) + block = ti.group(1) if ti else text + first = re.search(r"(.*?)", block, re.S) + last = re.search(r"(.*?)", block, re.S) + author = " ".join(re.sub(r"<[^>]+>", "", x.group(1)).strip() + for x in (first, last) if x).strip() + book = re.search(r"(.*?)", block, re.S) + return author, (re.sub(r"<[^>]+>", "", book.group(1)).strip() if book else "") + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("source", type=Path) + ap.add_argument("output", type=Path) + args = ap.parse_args() + text = read_fb2(args.source) + body = re.search(r"]*>(.*)", text, re.S) + r = Reader() + r.feed(body.group(1) if body else text) + r.close() + if not r.blocks: + sys.exit("в %s не нашлось текста" % args.source) + author, title = meta(text) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(bookhtml.document(title or args.output.stem, r.blocks), + encoding="utf-8") + print("blocks: %d, h1: %d, p: %d" % (len(r.blocks), + sum(1 for t, _, _ in r.blocks if t == "h1"), + sum(1 for t, _, _ in r.blocks if t == "p"))) + print("метаданные: %s — %s" % (author or "?", title or "?")) + + +if __name__ == "__main__": + main() diff --git a/scripts/pack_epub.py b/scripts/pack_epub.py index ad6c6d6..c6295e4 100644 --- a/scripts/pack_epub.py +++ b/scripts/pack_epub.py @@ -48,6 +48,23 @@ def split_chapters(lines): return [(t, b) for t, b in chapters if any(x.strip() for x in b)] +def series_meta(meta): + """Цикл и номер в нём. Пишем в двух видах: calibre-совместимый meta читают + почти все читалки, belongs-to-collection — требование EPUB 3.""" + name = meta.get("series") + if not name: + return "" + idx = meta.get("series_index") or "" + out = ['\n ' % html.escape(name)] + if idx: + out.append('' % html.escape(str(idx))) + out.append('%s' % html.escape(name)) + out.append('series') + if idx: + out.append('%s' % html.escape(str(idx))) + return "\n ".join(out) + + def build(src, out, meta): text = src.read_text(encoding="utf-8") css = re.search(r"", text, re.S) @@ -107,12 +124,12 @@ def build(src, out, meta): %s %s %s - %s + %s%s %s %s """ % (html.escape(meta["id"]), html.escape(meta["title"]), - html.escape(meta["author"]), meta["lang"], + html.escape(meta["author"]), meta["lang"], series_meta(meta), "\n ".join(items), "\n ".join(spine))) nav = "\n".join( @@ -136,9 +153,13 @@ def main(): ap.add_argument("--author", default="") ap.add_argument("--lang", default="ru") ap.add_argument("--cover", default="cover.jpg", help="имя файла в images/") + ap.add_argument("--series", default="", help="название цикла") + ap.add_argument("--series-index", default="", help="номер в цикле") args = ap.parse_args() meta = {"title": args.title, "author": args.author, "lang": args.lang, - "cover": args.cover, "id": "urn:uuid:" + args.output.stem.replace(" ", "-")} + "cover": args.cover, "series": args.series, + "series_index": args.series_index, + "id": "urn:uuid:" + args.output.stem.replace(" ", "-")} n, imgs = build(args.source, args.output, meta) size = args.output.stat().st_size / 2**20 print("%s: глав %d, картинок %d, %.1f МБ" % (args.output.name, n, imgs, size))