From 7ed6e500ca37bc6d4d75b389fb3fae4ca476bfa6 Mon Sep 17 00:00:00 2001 From: chesirecatt Date: Fri, 21 Aug 2026 17:36:10 +0300 Subject: [PATCH] =?UTF-8?q?=D0=9F=D0=B5=D1=80=D0=B5=D0=B2=D0=BE=D0=B4=20?= =?UTF-8?q?=D0=BD=D0=B5=20=D0=B2=D0=BE=D0=B7=D0=BE=D0=B1=D0=BD=D0=BE=D0=B2?= =?UTF-8?q?=D0=BB=D1=8F=D0=B5=D0=BC:=20--all=20=D0=BF=D0=B5=D1=80=D0=B5?= =?UTF-8?q?=D0=B2=D0=BE=D0=B4=D0=B8=D1=82=20=D0=B2=D1=81=D0=B5=20=D0=B3?= =?UTF-8?q?=D0=BB=D0=B0=D0=B2=D1=8B=20=D0=B7=D0=B0=D0=BD=D0=BE=D0=B2=D0=BE?= =?UTF-8?q?,=20=D0=BC=D0=B5=D1=85=D0=B0=D0=BD=D0=B8=D0=B7=D0=BC=20progress?= =?UTF-8?q?=20=D0=BD=D0=B5=20=D1=80=D0=B0=D0=B1=D0=BE=D1=82=D0=B0=D0=B5?= =?UTF-8?q?=D1=82?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- SKILL.md | 12 ++++++++++-- scripts/test_translate.py | 28 ++++++++++++++++++++++++++++ scripts/translate.py | 21 ++++++++++++++++----- 3 files changed, 54 insertions(+), 7 deletions(-) diff --git a/SKILL.md b/SKILL.md index a68aa47..852756a 100644 --- a/SKILL.md +++ b/SKILL.md @@ -180,8 +180,16 @@ is the bridge: it writes the repo's input format, shells out to `python scripts/translate.py book.html book.ru.html --repo --workdir --workers 12` - Resumable: the external repo tracks completed chapters and skips them on a - re-run, so an interrupted run costs nothing to restart. + ⚠️ **Not resumable, despite what the external repo claims.** Its + `progress/translation_progress.json` stays `{"chapters": {}}` and `--all` + re-translates every chapter, including ones already sitting in + `translations/`. Measured 2026-08-21 on Stroustrup: a re-run to repair 4 + failed chapters re-did all 29. Budget a full book on every restart, and + prefer getting one clean run over patching a partial one. + + The one thing that *does* skip work: deleting a chapter from `extracted/` + before the run. It is then never sent, and `rebuild` keeps the original + English for it — the right treatment for an index. 4. **Read the reported counts.** "без перевода" above zero means a chapter came back with a different paragraph count and kept its original text; "разметка потеряна" counts paragraphs where the model mangled the inline-tag markers diff --git a/scripts/test_translate.py b/scripts/test_translate.py index eaeff61..7e3d6d5 100644 --- a/scripts/test_translate.py +++ b/scripts/test_translate.py @@ -105,6 +105,33 @@ def test_code_listings_are_left_alone(tmp: Path): assert t.verify_output(src, "ru"), "код не должен заваливать приёмку" +def test_junk_translation_falls_back_to_original(tmp: Path): + """Модель иногда возвращает на длинный абзац огрызок «, v», а число абзацев + при этом сходится — прежние ворота такое пропускали.""" + src = tmp / "book.html" + long_en = ("A vector is simply a sequence of elements that you can access by " + "an index, and it is the workhorse of the standard library. ") * 2 + src.write_text("\n".join([ + "

Chapter

", + "

%s

" % long_en, + "

Second paragraph of the very same chapter, also reasonably long.

", + ]), encoding="utf-8") + chapters = t.split_chapters(t.parse_blocks(src)) + trans = tmp / "translations" + trans.mkdir() + (trans / "chapter_000_translated.json").write_text(json.dumps( + {"number": 0, "paragraphs": ["Глава", ", v", + "Второй абзац той же самой главы, тоже достаточно длинный."]}, + ensure_ascii=False), encoding="utf-8") + + out = tmp / "out.html" + t.rebuild(src, chapters, tmp, out) + result = out.read_text(encoding="utf-8") + assert ", v" not in result, "огрызок не должен попадать в книгу" + assert "A vector is simply a sequence" in result, "вместо огрызка нужен оригинал" + assert "Второй абзац" in result, "нормальный перевод должен остаться" + + def test_workdir_belongs_to_one_book(tmp: Path): """Имена chapter_NNN.json у всех книг одинаковы: чужой workdir склеит перевод одной книги с текстом другой.""" @@ -155,6 +182,7 @@ if __name__ == "__main__": test_broken_marks_drop_tags() test_language_detection() for case in (test_split_and_rebuild, test_code_listings_are_left_alone, + test_junk_translation_falls_back_to_original, test_workdir_belongs_to_one_book, test_stale_chapters_removed, test_verify_output_catches_untranslated): with tempfile.TemporaryDirectory() as d: diff --git a/scripts/translate.py b/scripts/translate.py index c8b8436..4f56878 100644 --- a/scripts/translate.py +++ b/scripts/translate.py @@ -148,9 +148,15 @@ def run_translator(repo, workdir, extracted, workers): sys.exit("book_translator завершился с кодом %d" % r.returncode) +# Доля от длины оригинала, ниже которой перевод считается мусором. Совпадение +# числа абзацев ничего не гарантирует: модель иногда возвращает на длинный абзац +# огрызок вида «, v», и прежние ворота такое пропускали. +MIN_LEN_SHARE = 0.25 + + def rebuild(src, chapters, workdir, out): lines = src.read_text(encoding="utf-8").splitlines() - broken = missing = translated = 0 + broken = missing = translated = junk = 0 for n, ch in enumerate(chapters): f = workdir / "translations" / ("chapter_%03d_translated.json" % n) if not f.exists(): @@ -162,15 +168,20 @@ def rebuild(src, chapters, workdir, out): % (n, len(paragraphs), len(ch))) missing += len(ch) continue - for (idx, tag, attrs, _), text in zip(ch, paragraphs): + for (idx, tag, attrs, original), text in zip(ch, paragraphs): + plain_src = re.sub("<[^>]+>", "", original) + if len(plain_src) > 120 and len(text) < max(20, len(plain_src) * MIN_LEN_SHARE): + junk += 1 + lines[idx] = "<%s%s>%s" % (tag, attrs, original, tag) + continue # оставляем оригинал: английский абзац лучше огрызка body, ok = from_marks(text) broken += not ok translated += 1 lines[idx] = "<%s%s>%s" % (tag, attrs, body, tag) out.write_text("\n".join(lines) + "\n", encoding="utf-8") - print("переведено блоков: %d, без перевода: %d, разметка потеряна в %d" - % (translated, missing, broken)) - return missing == 0 + print("переведено блоков: %d, без перевода: %d, разметка потеряна в %d, " + "огрызков заменено оригиналом: %d" % (translated, missing, broken, junk)) + return missing == 0 and junk * 200 <= translated def verify_output(out, target):