From 43d6cfc4561ba2caceb44098694078bb94a1f030 Mon Sep 17 00:00:00 2001 From: chesirecatt Date: Fri, 21 Aug 2026 18:41:45 +0300 Subject: [PATCH] =?UTF-8?q?glossary.py:=20=D0=B4=D0=B2=D1=83=D1=8F=D0=B7?= =?UTF-8?q?=D1=8B=D1=87=D0=BD=D1=8B=D0=B9=20=D0=B3=D0=BB=D0=BE=D1=81=D1=81?= =?UTF-8?q?=D0=B0=D1=80=D0=B8=D0=B9=20=D0=B8=D0=B7=20=D0=BF=D0=B0=D1=80?= =?UTF-8?q?=D0=B0=D0=BB=D0=BB=D0=B5=D0=BB=D1=8C=D0=BD=D1=8B=D1=85=20=D0=B8?= =?UTF-8?q?=D0=B7=D0=B4=D0=B0=D0=BD=D0=B8=D0=B9=20=D1=86=D0=B8=D0=BA=D0=BB?= =?UTF-8?q?=D0=B0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- SKILL.md | 27 +++++ scripts/glossary.py | 219 ++++++++++++++++++++++++++++++++++++++ scripts/pack_epub.py | 9 +- scripts/test_pack_epub.py | 20 +++- 4 files changed, 271 insertions(+), 4 deletions(-) create mode 100644 scripts/glossary.py diff --git a/SKILL.md b/SKILL.md index d800440..ec39beb 100644 --- a/SKILL.md +++ b/SKILL.md @@ -130,6 +130,33 @@ EOF After packing, confirm the EPUB: `ebook-meta` for metadata, and the `.ncx` navPoint count for the TOC size. +## Glossary from existing translations + +For a book in a series that already has published translations, `scripts/glossary.py` +mines a bilingual glossary so the machine translation does not invent new spellings +for names the reader already knows. Feed it pairs of editions of the *same* volume: + +``` +python scripts/glossary.py --en vol12.fb2 --ru vol12.ru.fb2 \ + --en vol15.epub --ru vol15.ru.fb2 --score 0.6 +``` + +Two signals, and both are needed. Position: paragraph indices do not line up +(Russian editions split dialogue, giving 2–3× more paragraphs), so offsets are +measured as a **share of characters**, where the texts track each other closely. +Transliteration: a proper name in Russian is nearly always a transliteration, so +the Cyrillic candidate is romanized and compared to the English term — this is +what turns the output from noise into a usable list. + +Two mirrored filters remove the rest of the junk: a candidate whose head word +also appears lowercase in the same text is a sentence-initial common word, not a +name — applied on both sides. Measured on four Dresden Files volumes: 71 pairs, +of which two were wrong. + +⚠️ Concept terms (`White Council` → `Белый Совет`, `Spire` → `Копьё`) do **not** +come out of the transliteration path and the positional one alone is too noisy +for them. Extract those by hand and verify by grepping the existing translation. + ## Optional stage: translation Only when the user asks for a translated book. It slots between step 6 and diff --git a/scripts/glossary.py b/scripts/glossary.py new file mode 100644 index 0000000..ab66d4b --- /dev/null +++ b/scripts/glossary.py @@ -0,0 +1,219 @@ +#!/usr/bin/env python3 +"""Двуязычный глоссарий из параллельных изданий одной книги. + +Задача: машинный перевод очередного тома цикла не должен расходиться с уже +изданными переводами в именах и реалиях. Частотный список даёт только имена; +пары вида White Council → Белый Совет так не получить. + +Метод — выравнивание по относительной позиции в тексте. Абзацы английского и +русского изданий не совпадают ни числом, ни границами, но идут в одном порядке, +поэтому термин, встречающийся в английском тексте на 12%, 34% и 78% длины, +в переводе окажется примерно там же. Кандидат, чьи позиции совпали с позициями +термина лучше, чем с текстом вообще, и есть перевод. + +Работает с fb2, epub и голым текстом. Только стандартная библиотека. +""" +import argparse +import collections +import difflib +import html +import re +import sys +import zipfile +from pathlib import Path + +# Слова, с которых начинается предложение, — не имена собственные. +RU_STOP = {"Когда", "Если", "Что", "Как", "Это", "Она", "Они", "Мы", "Вы", "Так", + "Потом", "Затем", "Даже", "Его", "Её", "Их", "Может", "Все", "Теперь", + "Нет", "Да", "Меня", "Мне", "Тогда", "Только", "После", "Пока", "Там", + "Здесь", "Однако", "Впрочем", "Просто", "Один", "Вот", "Или", "Почему", + "Мой", "Моя", "Кто", "Конечно", "Хорошо", "Ага", "Думаю", "Возможно", + "Глава", "Никто", "Ничего", "Что-то", "Кажется", "Значит", "Сейчас"} +EN_STOP = {"The", "A", "An", "And", "But", "I", "It", "He", "She", "They", "We", + "You", "This", "That", "There", "Then", "When", "If", "So", "My", "His", + "Her", "Chapter", "What", "Why", "How", "No", "Yes", "Not", "For", "Of", + "In", "On", "At", "To", "With", "As", "Was", "Were", "Had", "Have", + "Would", "Could", "Should", "One", "All", "Just", "Like", "Now", "Well", + "Okay", "Oh", "Maybe", "Something", "Nothing", "Someone"} + + +def read_text(path): + """Текст книги из fb2, epub или txt — разметка выкидывается.""" + p = Path(path) + if p.suffix.lower() == ".epub": + with zipfile.ZipFile(p) as z: + parts = [z.read(n).decode("utf-8", "replace") + for n in z.namelist() if n.lower().endswith((".xhtml", ".html"))] + raw = "\n".join(parts) + elif p.suffix.lower() in (".fb2", ".xml"): + raw = p.read_bytes().decode("utf-8", "replace") + else: + raw = p.read_text(encoding="utf-8", errors="replace") + raw = re.sub(r"<(script|style|head)[^>]*>.*?", " ", raw, flags=re.S | re.I) + raw = re.sub(r"

||", "\n", raw, flags=re.I) + return html.unescape(re.sub(r"<[^>]+>", " ", raw)) + + +def paragraphs(text): + return [p.strip() for p in text.split("\n") if len(p.strip()) > 40] + + +def candidates(paras, pattern, stop, min_count): + """{термин: [относительные позиции вхождений]} + + Позиция считается по символам, а не по номеру абзаца: русские издания + разбивают диалоги построчно, и абзацев там втрое больше — по индексу + тексты расходятся, по доле знаков идут почти вровень. + """ + pos = collections.defaultdict(list) + total = sum(len(p) for p in paras) or 1 + acc = 0 + for p in paras: + rel = acc / total + acc += len(p) + for m in set(pattern.findall(p)): + head = m.split()[0] + if head in stop or len(m) < 3: + continue + pos[m].append(rel) + return {t: v for t, v in pos.items() if len(v) >= min_count} + + +# Кириллица -> латиница для сравнения имён. Имена собственные в переводе почти +# всегда транслитерация, и это признак куда надёжнее позиционного совпадения. +LAT = {"а":"a","б":"b","в":"v","г":"g","д":"d","е":"e","ё":"e","ж":"j","з":"z", + "и":"i","й":"i","к":"k","л":"l","м":"m","н":"n","о":"o","п":"p","р":"r", + "с":"s","т":"t","у":"u","ф":"f","х":"h","ц":"c","ч":"c","ш":"s","щ":"s", + "ъ":"","ы":"i","ь":"","э":"e","ю":"u","я":"a"} + + +def translit(s): + return "".join(LAT.get(c, c) for c in s.lower()) + + +def name_similarity(en, ru): + """Похожесть после огрубления: латиница обеих сторон без удвоений и гласных + на конце. Nicodemus/Никодимус дают почти единицу, Harry/Молли — ноль.""" + a = re.sub(r"[^a-z]", "", en.lower()) + b = re.sub(r"[^a-z]", "", translit(ru)) + a = re.sub(r"(.)\1+", r"\1", a) + b = re.sub(r"(.)\1+", r"\1", b) + if not a or not b: + return 0.0 + return difflib.SequenceMatcher(None, a, b[:len(a) + 3]).ratio() + + +def overlap(a, b, tol): + """Доля вхождений a, у которых нашлось вхождение b поблизости.""" + if not a or not b: + return 0.0 + b = sorted(b) + hit = 0 + for x in a: + lo, hi = 0, len(b) - 1 + best = 1.0 + while lo <= hi: + mid = (lo + hi) // 2 + best = min(best, abs(b[mid] - x)) + if b[mid] < x: + lo = mid + 1 + else: + hi = mid - 1 + if best <= tol: + hit += 1 + return hit / len(a) + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--en", action="append", required=True, help="английское издание") + ap.add_argument("--ru", action="append", required=True, help="русское издание того же тома") + ap.add_argument("--min-count", type=int, default=4) + ap.add_argument("--tol", type=float, default=0.02, help="допуск по относительной позиции") + ap.add_argument("--score", type=float, default=0.6, help="порог уверенности") + ap.add_argument("--top", type=int, default=80) + args = ap.parse_args() + if len(args.en) != len(args.ru): + sys.exit("нужно одинаковое число --en и --ru: это должны быть пары изданий") + + EN = re.compile(r"\b[A-Z][a-z]+(?:\s+(?:of\s+|the\s+)?[A-Z][a-z]+){0,2}") + lower_seen = collections.Counter() + ru_lower = collections.Counter() + RU = re.compile(r"\b[А-ЯЁ][а-яё]+(?:\s+[А-ЯЁ][а-яё]+){0,2}") + en_pos, ru_pos = collections.defaultdict(list), collections.defaultdict(list) + for i, (fe, fr) in enumerate(zip(args.en, args.ru)): + pe, pr = paragraphs(read_text(fe)), paragraphs(read_text(fr)) + # Настоящее имя собственное почти не встречается со строчной буквы: + # так отсеиваются After, Once, More и прочие начала предложений. + for w in re.findall(r"\b[a-z]{3,}\b", " ".join(pe)): + lower_seen[w] += 1 + for w in re.findall(r"\b[а-яё]{3,}\b", " ".join(pr)): + ru_lower[w] += 1 + print("пара %d: %d абзацев EN, %d RU" % (i + 1, len(pe), len(pr)), file=sys.stderr) + for t, v in candidates(pe, EN, EN_STOP, 1).items(): + en_pos[t] += [(i, x) for x in v] + for t, v in candidates(pr, RU, RU_STOP, 1).items(): + ru_pos[t] += [(i, x) for x in v] + + en_pos = {t: v for t, v in en_pos.items() + if len(v) >= args.min_count + and lower_seen[t.split()[0].lower()] < max(3, len(v) * 0.2)} + # Зеркальный отсев: «Надеюсь», «Зачем», «Парень» — обычные слова, они + # встречаются со строчной буквы и именами собственными быть не могут. + ru_pos = {t: v for t, v in ru_pos.items() + if len(v) >= args.min_count + and ru_lower[t.split()[0].lower()] < max(3, len(v) * 0.2)} + print("кандидатов: %d EN, %d RU" % (len(en_pos), len(ru_pos)), file=sys.stderr) + + by_book_ru = collections.defaultdict(dict) + for t, v in ru_pos.items(): + for b, x in v: + by_book_ru[b].setdefault(t, []).append(x) + + out = [] + for term, occ in sorted(en_pos.items(), key=lambda kv: -len(kv[1])): + mine = collections.defaultdict(list) + for b, x in occ: + mine[b].append(x) + best, best_score = None, 0.0 + for cand, cocc in ru_pos.items(): + # Частота кандидата должна быть сопоставима: «Гарри» встречается на + # каждой странице и по одностороннему совпадению побеждает всех. + if not (0.25 <= len(cocc) / len(occ) <= 4.0): + continue + fwd = back = tot_f = tot_b = 0.0 + for b, xs in mine.items(): + cand_pos = by_book_ru[b].get(cand) + if not cand_pos: + continue + fwd += overlap(xs, cand_pos, args.tol) * len(xs) + back += overlap(cand_pos, xs, args.tol) * len(cand_pos) + tot_f += len(xs); tot_b += len(cand_pos) + if tot_f < len(occ) * 0.5 or not tot_b: + continue + # Симметричная мера: термин должен находить перевод, а перевод — + # термин. Иначе частотное слово выигрывает у настоящего соответствия. + score = (fwd / tot_f) * (back / tot_b) + # Транслитерация — сильный самостоятельный довод: если написание + # совпадает, позиционного подтверждения нужно куда меньше. + sim = name_similarity(term, cand) + if sim >= 0.72: + score = max(score, sim) + 0.25 + elif sim < 0.3 and " " not in term: + score *= 0.35 # разные имена, позиции совпали случайно + if score > best_score: + best, best_score = cand, score + if best and best_score >= args.score: + out.append((len(occ), best_score, term, best)) + + print("# частота | уверенность | оригинал -> перевод") + seen = set() + for n, sc, en, ru in out[:args.top]: + if (en, ru) in seen: + continue + seen.add((en, ru)) + print("%5d %.2f %-30s -> %s" % (n, sc, en, ru)) + + +if __name__ == "__main__": + main() diff --git a/scripts/pack_epub.py b/scripts/pack_epub.py index b0e93dc..ad6c6d6 100644 --- a/scripts/pack_epub.py +++ b/scripts/pack_epub.py @@ -82,8 +82,11 @@ def build(src, out, meta): CHAPTER % (NS, html.escape(title), chunk)) for img in used: z.write(imgdir / img, "OEBPS/images/" + img) - if meta["cover"] and (imgdir / meta["cover"]).exists(): - z.write(imgdir / meta["cover"], "OEBPS/images/" + meta["cover"]) + # Обложка бывает и среди картинок текста — второй раз её класть нельзя, + # zip примет дубликат имени, а читалки на такой архив ругаются. + cover = meta["cover"] if meta["cover"] not in used else "" + if cover and (imgdir / cover).exists(): + z.write(imgdir / cover, "OEBPS/images/" + cover) items = ['', ''] @@ -94,7 +97,7 @@ def build(src, out, meta): spine.append('' % i) mime = {".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg", ".gif": "image/gif", ".svg": "image/svg+xml"} - for i, img in enumerate(used + ([meta["cover"]] if meta["cover"] else [])): + for i, img in enumerate(used + ([cover] if cover else [])): items.append('' % (i, img, mime.get(Path(img).suffix.lower(), "image/jpeg"), ' properties="cover-image"' if img == meta["cover"] else "")) diff --git a/scripts/test_pack_epub.py b/scripts/test_pack_epub.py index 270ba93..3702847 100644 --- a/scripts/test_pack_epub.py +++ b/scripts/test_pack_epub.py @@ -57,6 +57,23 @@ def test_pack(tmp: Path): assert "Текст второй главы" not in ch1, "главы не должны склеиваться" +def test_cover_used_in_text_is_not_duplicated(tmp: Path): + """Обложка, встречающаяся и в тексте, не должна попадать в архив дважды.""" + src = tmp / "book.html" + src.write_text("

Глава

\n" + '

\n' + "

Текст главы.

", encoding="utf-8") + (tmp / "images").mkdir() + (tmp / "images" / "cover.jpg").write_bytes(b"\xff\xd8\xff") + out = tmp / "book.epub" + pack_epub.build(src, out, {"title": "t", "author": "a", "lang": "ru", + "cover": "cover.jpg", "id": "i"}) + names = zipfile.ZipFile(out).namelist() + assert names.count("OEBPS/images/cover.jpg") == 1, names + opf = zipfile.ZipFile(out).read("OEBPS/content.opf").decode() + assert opf.count('href="images/cover.jpg"') == 1, opf + + def test_no_content(tmp: Path): src = tmp / "empty.html" src.write_text("\n", encoding="utf-8") @@ -69,7 +86,8 @@ def test_no_content(tmp: Path): if __name__ == "__main__": - for case in (test_pack, test_no_content): + for case in (test_pack, test_cover_used_in_text_is_not_duplicated, + test_no_content): with tempfile.TemporaryDirectory() as d: case(Path(d)) print("OK")