#!/usr/bin/env python3 """FB2 -> тот же XHTML, что выдают pdf2html.py и epub2html.py. FB2 в этой библиотеке основной формат (704 тысячи файлов), и без этого входа конвейер до них не дотягивается. Разметка у fb2 уже семантическая, поэтому задача та же, что у epub2html.py: привести чужие теги к нашим. Только стандартная библиотека. """ import argparse import html import re import sys import zipfile from html.parser import HTMLParser from pathlib import Path import bookhtml BLOCK = {"p": ("p", None), "v": ("p", "li"), "subtitle": ("h2", None), "text-author": ("p", "note"), "th": ("p", "row"), "td": ("p", "row")} INLINE = {"emphasis": "i", "strong": "b", "code": "code", "sub": "i", "sup": "i"} DROP = {"description", "binary", "stylesheet"} class Reader(HTMLParser): def __init__(self): super().__init__(convert_charrefs=True) self.blocks, self.cur, self.open_i, self.drop = [], None, [], 0 self.depth = 0 # вложенность section: даёт уровень заголовка self.in_title = False def flush(self): if not self.cur: return tag, cls, parts = self.cur self.cur = None for t in reversed(self.open_i): parts.append("" % t) self.open_i = [] txt = re.sub(r"\s+", " ", "".join(parts)).strip() if txt: self.blocks.append((tag, cls, txt)) def handle_starttag(self, tag, attrs): if tag in DROP: self.drop += 1 return if self.drop: return if tag == "section": self.depth += 1 return if tag == "title": self.flush() self.in_title = True return if tag == "empty-line": self.flush() return if tag == "image": return # картинки fb2 лежат в base64, пропускаем if self.in_title and tag == "p": # заголовок раздела: уровень по вложенности section self.flush() self.cur = ("h1" if self.depth <= 2 else "h2", None, []) return if tag in BLOCK: self.flush() self.cur = (BLOCK[tag][0], BLOCK[tag][1], []) return if tag in INLINE and self.cur: out = INLINE[tag] self.open_i.append(out) self.cur[2].append("<%s>" % out) def handle_endtag(self, tag): if tag in DROP: self.drop = max(0, self.drop - 1) return if self.drop: return if tag == "section": self.flush() self.depth = max(0, self.depth - 1) return if tag == "title": self.flush() self.in_title = False return if tag in BLOCK: self.flush() return if tag in INLINE and self.cur: out = INLINE[tag] if out in self.open_i: self.open_i.remove(out) self.cur[2].append("" % out) def handle_data(self, data): if self.drop or not self.cur: return self.cur[2].append(html.escape(data, quote=False)) def close(self): super().close() self.flush() def read_fb2(path): p = Path(path) if p.suffix.lower() == ".zip" or zipfile.is_zipfile(p): with zipfile.ZipFile(p) as z: name = next(n for n in z.namelist() if n.lower().endswith(".fb2")) raw = z.read(name) else: raw = p.read_bytes() enc = "utf-8" m = re.search(rb'encoding="([\w-]+)"', raw[:200]) if m: enc = m.group(1).decode("ascii", "ignore") return raw.decode(enc, "replace") def meta(text): """(автор, название) из description — для метаданных EPUB.""" def tag(name): m = re.search(r"<%s>(.*?)" % (name, name), text, re.S) return re.sub(r"<[^>]+>", " ", m.group(1)).strip() if m else "" ti = re.search(r"(.*?)", text, re.S) block = ti.group(1) if ti else text first = re.search(r"(.*?)", block, re.S) last = re.search(r"(.*?)", block, re.S) author = " ".join(re.sub(r"<[^>]+>", "", x.group(1)).strip() for x in (first, last) if x).strip() book = re.search(r"(.*?)", block, re.S) return author, (re.sub(r"<[^>]+>", "", book.group(1)).strip() if book else "") def main(): ap = argparse.ArgumentParser(description=__doc__) ap.add_argument("source", type=Path) ap.add_argument("output", type=Path) args = ap.parse_args() text = read_fb2(args.source) body = re.search(r"]*>(.*)", text, re.S) r = Reader() r.feed(body.group(1) if body else text) r.close() if not r.blocks: sys.exit("в %s не нашлось текста" % args.source) author, title = meta(text) args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(bookhtml.document(title or args.output.stem, r.blocks), encoding="utf-8") print("blocks: %d, h1: %d, p: %d" % (len(r.blocks), sum(1 for t, _, _ in r.blocks if t == "h1"), sum(1 for t, _, _ in r.blocks if t == "p"))) print("метаданные: %s — %s" % (author or "?", title or "?")) if __name__ == "__main__": main()