diff --git a/SKILL.md b/SKILL.md
index d800440..ec39beb 100644
--- a/SKILL.md
+++ b/SKILL.md
@@ -130,6 +130,33 @@ EOF
After packing, confirm the EPUB: `ebook-meta` for metadata, and the `.ncx`
navPoint count for the TOC size.
+## Glossary from existing translations
+
+For a book in a series that already has published translations, `scripts/glossary.py`
+mines a bilingual glossary so the machine translation does not invent new spellings
+for names the reader already knows. Feed it pairs of editions of the *same* volume:
+
+```
+python scripts/glossary.py --en vol12.fb2 --ru vol12.ru.fb2 \
+ --en vol15.epub --ru vol15.ru.fb2 --score 0.6
+```
+
+Two signals, and both are needed. Position: paragraph indices do not line up
+(Russian editions split dialogue, giving 2–3× more paragraphs), so offsets are
+measured as a **share of characters**, where the texts track each other closely.
+Transliteration: a proper name in Russian is nearly always a transliteration, so
+the Cyrillic candidate is romanized and compared to the English term — this is
+what turns the output from noise into a usable list.
+
+Two mirrored filters remove the rest of the junk: a candidate whose head word
+also appears lowercase in the same text is a sentence-initial common word, not a
+name — applied on both sides. Measured on four Dresden Files volumes: 71 pairs,
+of which two were wrong.
+
+⚠️ Concept terms (`White Council` → `Белый Совет`, `Spire` → `Копьё`) do **not**
+come out of the transliteration path and the positional one alone is too noisy
+for them. Extract those by hand and verify by grepping the existing translation.
+
## Optional stage: translation
Only when the user asks for a translated book. It slots between step 6 and
diff --git a/scripts/glossary.py b/scripts/glossary.py
new file mode 100644
index 0000000..ab66d4b
--- /dev/null
+++ b/scripts/glossary.py
@@ -0,0 +1,219 @@
+#!/usr/bin/env python3
+"""Двуязычный глоссарий из параллельных изданий одной книги.
+
+Задача: машинный перевод очередного тома цикла не должен расходиться с уже
+изданными переводами в именах и реалиях. Частотный список даёт только имена;
+пары вида White Council → Белый Совет так не получить.
+
+Метод — выравнивание по относительной позиции в тексте. Абзацы английского и
+русского изданий не совпадают ни числом, ни границами, но идут в одном порядке,
+поэтому термин, встречающийся в английском тексте на 12%, 34% и 78% длины,
+в переводе окажется примерно там же. Кандидат, чьи позиции совпали с позициями
+термина лучше, чем с текстом вообще, и есть перевод.
+
+Работает с fb2, epub и голым текстом. Только стандартная библиотека.
+"""
+import argparse
+import collections
+import difflib
+import html
+import re
+import sys
+import zipfile
+from pathlib import Path
+
+# Слова, с которых начинается предложение, — не имена собственные.
+RU_STOP = {"Когда", "Если", "Что", "Как", "Это", "Она", "Они", "Мы", "Вы", "Так",
+ "Потом", "Затем", "Даже", "Его", "Её", "Их", "Может", "Все", "Теперь",
+ "Нет", "Да", "Меня", "Мне", "Тогда", "Только", "После", "Пока", "Там",
+ "Здесь", "Однако", "Впрочем", "Просто", "Один", "Вот", "Или", "Почему",
+ "Мой", "Моя", "Кто", "Конечно", "Хорошо", "Ага", "Думаю", "Возможно",
+ "Глава", "Никто", "Ничего", "Что-то", "Кажется", "Значит", "Сейчас"}
+EN_STOP = {"The", "A", "An", "And", "But", "I", "It", "He", "She", "They", "We",
+ "You", "This", "That", "There", "Then", "When", "If", "So", "My", "His",
+ "Her", "Chapter", "What", "Why", "How", "No", "Yes", "Not", "For", "Of",
+ "In", "On", "At", "To", "With", "As", "Was", "Were", "Had", "Have",
+ "Would", "Could", "Should", "One", "All", "Just", "Like", "Now", "Well",
+ "Okay", "Oh", "Maybe", "Something", "Nothing", "Someone"}
+
+
+def read_text(path):
+ """Текст книги из fb2, epub или txt — разметка выкидывается."""
+ p = Path(path)
+ if p.suffix.lower() == ".epub":
+ with zipfile.ZipFile(p) as z:
+ parts = [z.read(n).decode("utf-8", "replace")
+ for n in z.namelist() if n.lower().endswith((".xhtml", ".html"))]
+ raw = "\n".join(parts)
+ elif p.suffix.lower() in (".fb2", ".xml"):
+ raw = p.read_bytes().decode("utf-8", "replace")
+ else:
+ raw = p.read_text(encoding="utf-8", errors="replace")
+ raw = re.sub(r"<(script|style|head)[^>]*>.*?\1>", " ", raw, flags=re.S | re.I)
+ raw = re.sub(r"
||
", "\n", raw, flags=re.I)
+ return html.unescape(re.sub(r"<[^>]+>", " ", raw))
+
+
+def paragraphs(text):
+ return [p.strip() for p in text.split("\n") if len(p.strip()) > 40]
+
+
+def candidates(paras, pattern, stop, min_count):
+ """{термин: [относительные позиции вхождений]}
+
+ Позиция считается по символам, а не по номеру абзаца: русские издания
+ разбивают диалоги построчно, и абзацев там втрое больше — по индексу
+ тексты расходятся, по доле знаков идут почти вровень.
+ """
+ pos = collections.defaultdict(list)
+ total = sum(len(p) for p in paras) or 1
+ acc = 0
+ for p in paras:
+ rel = acc / total
+ acc += len(p)
+ for m in set(pattern.findall(p)):
+ head = m.split()[0]
+ if head in stop or len(m) < 3:
+ continue
+ pos[m].append(rel)
+ return {t: v for t, v in pos.items() if len(v) >= min_count}
+
+
+# Кириллица -> латиница для сравнения имён. Имена собственные в переводе почти
+# всегда транслитерация, и это признак куда надёжнее позиционного совпадения.
+LAT = {"а":"a","б":"b","в":"v","г":"g","д":"d","е":"e","ё":"e","ж":"j","з":"z",
+ "и":"i","й":"i","к":"k","л":"l","м":"m","н":"n","о":"o","п":"p","р":"r",
+ "с":"s","т":"t","у":"u","ф":"f","х":"h","ц":"c","ч":"c","ш":"s","щ":"s",
+ "ъ":"","ы":"i","ь":"","э":"e","ю":"u","я":"a"}
+
+
+def translit(s):
+ return "".join(LAT.get(c, c) for c in s.lower())
+
+
+def name_similarity(en, ru):
+ """Похожесть после огрубления: латиница обеих сторон без удвоений и гласных
+ на конце. Nicodemus/Никодимус дают почти единицу, Harry/Молли — ноль."""
+ a = re.sub(r"[^a-z]", "", en.lower())
+ b = re.sub(r"[^a-z]", "", translit(ru))
+ a = re.sub(r"(.)\1+", r"\1", a)
+ b = re.sub(r"(.)\1+", r"\1", b)
+ if not a or not b:
+ return 0.0
+ return difflib.SequenceMatcher(None, a, b[:len(a) + 3]).ratio()
+
+
+def overlap(a, b, tol):
+ """Доля вхождений a, у которых нашлось вхождение b поблизости."""
+ if not a or not b:
+ return 0.0
+ b = sorted(b)
+ hit = 0
+ for x in a:
+ lo, hi = 0, len(b) - 1
+ best = 1.0
+ while lo <= hi:
+ mid = (lo + hi) // 2
+ best = min(best, abs(b[mid] - x))
+ if b[mid] < x:
+ lo = mid + 1
+ else:
+ hi = mid - 1
+ if best <= tol:
+ hit += 1
+ return hit / len(a)
+
+
+def main():
+ ap = argparse.ArgumentParser(description=__doc__)
+ ap.add_argument("--en", action="append", required=True, help="английское издание")
+ ap.add_argument("--ru", action="append", required=True, help="русское издание того же тома")
+ ap.add_argument("--min-count", type=int, default=4)
+ ap.add_argument("--tol", type=float, default=0.02, help="допуск по относительной позиции")
+ ap.add_argument("--score", type=float, default=0.6, help="порог уверенности")
+ ap.add_argument("--top", type=int, default=80)
+ args = ap.parse_args()
+ if len(args.en) != len(args.ru):
+ sys.exit("нужно одинаковое число --en и --ru: это должны быть пары изданий")
+
+ EN = re.compile(r"\b[A-Z][a-z]+(?:\s+(?:of\s+|the\s+)?[A-Z][a-z]+){0,2}")
+ lower_seen = collections.Counter()
+ ru_lower = collections.Counter()
+ RU = re.compile(r"\b[А-ЯЁ][а-яё]+(?:\s+[А-ЯЁ][а-яё]+){0,2}")
+ en_pos, ru_pos = collections.defaultdict(list), collections.defaultdict(list)
+ for i, (fe, fr) in enumerate(zip(args.en, args.ru)):
+ pe, pr = paragraphs(read_text(fe)), paragraphs(read_text(fr))
+ # Настоящее имя собственное почти не встречается со строчной буквы:
+ # так отсеиваются After, Once, More и прочие начала предложений.
+ for w in re.findall(r"\b[a-z]{3,}\b", " ".join(pe)):
+ lower_seen[w] += 1
+ for w in re.findall(r"\b[а-яё]{3,}\b", " ".join(pr)):
+ ru_lower[w] += 1
+ print("пара %d: %d абзацев EN, %d RU" % (i + 1, len(pe), len(pr)), file=sys.stderr)
+ for t, v in candidates(pe, EN, EN_STOP, 1).items():
+ en_pos[t] += [(i, x) for x in v]
+ for t, v in candidates(pr, RU, RU_STOP, 1).items():
+ ru_pos[t] += [(i, x) for x in v]
+
+ en_pos = {t: v for t, v in en_pos.items()
+ if len(v) >= args.min_count
+ and lower_seen[t.split()[0].lower()] < max(3, len(v) * 0.2)}
+ # Зеркальный отсев: «Надеюсь», «Зачем», «Парень» — обычные слова, они
+ # встречаются со строчной буквы и именами собственными быть не могут.
+ ru_pos = {t: v for t, v in ru_pos.items()
+ if len(v) >= args.min_count
+ and ru_lower[t.split()[0].lower()] < max(3, len(v) * 0.2)}
+ print("кандидатов: %d EN, %d RU" % (len(en_pos), len(ru_pos)), file=sys.stderr)
+
+ by_book_ru = collections.defaultdict(dict)
+ for t, v in ru_pos.items():
+ for b, x in v:
+ by_book_ru[b].setdefault(t, []).append(x)
+
+ out = []
+ for term, occ in sorted(en_pos.items(), key=lambda kv: -len(kv[1])):
+ mine = collections.defaultdict(list)
+ for b, x in occ:
+ mine[b].append(x)
+ best, best_score = None, 0.0
+ for cand, cocc in ru_pos.items():
+ # Частота кандидата должна быть сопоставима: «Гарри» встречается на
+ # каждой странице и по одностороннему совпадению побеждает всех.
+ if not (0.25 <= len(cocc) / len(occ) <= 4.0):
+ continue
+ fwd = back = tot_f = tot_b = 0.0
+ for b, xs in mine.items():
+ cand_pos = by_book_ru[b].get(cand)
+ if not cand_pos:
+ continue
+ fwd += overlap(xs, cand_pos, args.tol) * len(xs)
+ back += overlap(cand_pos, xs, args.tol) * len(cand_pos)
+ tot_f += len(xs); tot_b += len(cand_pos)
+ if tot_f < len(occ) * 0.5 or not tot_b:
+ continue
+ # Симметричная мера: термин должен находить перевод, а перевод —
+ # термин. Иначе частотное слово выигрывает у настоящего соответствия.
+ score = (fwd / tot_f) * (back / tot_b)
+ # Транслитерация — сильный самостоятельный довод: если написание
+ # совпадает, позиционного подтверждения нужно куда меньше.
+ sim = name_similarity(term, cand)
+ if sim >= 0.72:
+ score = max(score, sim) + 0.25
+ elif sim < 0.3 and " " not in term:
+ score *= 0.35 # разные имена, позиции совпали случайно
+ if score > best_score:
+ best, best_score = cand, score
+ if best and best_score >= args.score:
+ out.append((len(occ), best_score, term, best))
+
+ print("# частота | уверенность | оригинал -> перевод")
+ seen = set()
+ for n, sc, en, ru in out[:args.top]:
+ if (en, ru) in seen:
+ continue
+ seen.add((en, ru))
+ print("%5d %.2f %-30s -> %s" % (n, sc, en, ru))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/pack_epub.py b/scripts/pack_epub.py
index b0e93dc..ad6c6d6 100644
--- a/scripts/pack_epub.py
+++ b/scripts/pack_epub.py
@@ -82,8 +82,11 @@ def build(src, out, meta):
CHAPTER % (NS, html.escape(title), chunk))
for img in used:
z.write(imgdir / img, "OEBPS/images/" + img)
- if meta["cover"] and (imgdir / meta["cover"]).exists():
- z.write(imgdir / meta["cover"], "OEBPS/images/" + meta["cover"])
+ # Обложка бывает и среди картинок текста — второй раз её класть нельзя,
+ # zip примет дубликат имени, а читалки на такой архив ругаются.
+ cover = meta["cover"] if meta["cover"] not in used else ""
+ if cover and (imgdir / cover).exists():
+ z.write(imgdir / cover, "OEBPS/images/" + cover)
items = [' ',
' ']
@@ -94,7 +97,7 @@ def build(src, out, meta):
spine.append('' % i)
mime = {".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg",
".gif": "image/gif", ".svg": "image/svg+xml"}
- for i, img in enumerate(used + ([meta["cover"]] if meta["cover"] else [])):
+ for i, img in enumerate(used + ([cover] if cover else [])):
items.append(' '
% (i, img, mime.get(Path(img).suffix.lower(), "image/jpeg"),
' properties="cover-image"' if img == meta["cover"] else ""))
diff --git a/scripts/test_pack_epub.py b/scripts/test_pack_epub.py
index 270ba93..3702847 100644
--- a/scripts/test_pack_epub.py
+++ b/scripts/test_pack_epub.py
@@ -57,6 +57,23 @@ def test_pack(tmp: Path):
assert "Текст второй главы" not in ch1, "главы не должны склеиваться"
+def test_cover_used_in_text_is_not_duplicated(tmp: Path):
+ """Обложка, встречающаяся и в тексте, не должна попадать в архив дважды."""
+ src = tmp / "book.html"
+ src.write_text("Глава
\n"
+ '
\n'
+ "Текст главы.
", encoding="utf-8")
+ (tmp / "images").mkdir()
+ (tmp / "images" / "cover.jpg").write_bytes(b"\xff\xd8\xff")
+ out = tmp / "book.epub"
+ pack_epub.build(src, out, {"title": "t", "author": "a", "lang": "ru",
+ "cover": "cover.jpg", "id": "i"})
+ names = zipfile.ZipFile(out).namelist()
+ assert names.count("OEBPS/images/cover.jpg") == 1, names
+ opf = zipfile.ZipFile(out).read("OEBPS/content.opf").decode()
+ assert opf.count('href="images/cover.jpg"') == 1, opf
+
+
def test_no_content(tmp: Path):
src = tmp / "empty.html"
src.write_text("\n", encoding="utf-8")
@@ -69,7 +86,8 @@ def test_no_content(tmp: Path):
if __name__ == "__main__":
- for case in (test_pack, test_no_content):
+ for case in (test_pack, test_cover_used_in_text_is_not_duplicated,
+ test_no_content):
with tempfile.TemporaryDirectory() as d:
case(Path(d))
print("OK")