From 14c7fad763cd3dfab3effa80b4e8ee37794b4537 Mon Sep 17 00:00:00 2001 From: chesirecatt Date: Fri, 21 Aug 2026 17:47:48 +0300 Subject: [PATCH] =?UTF-8?q?=D0=A1=D1=87=D1=91=D1=82=D1=87=D0=B8=D0=BA=20?= =?UTF-8?q?=D0=B0=D0=B1=D0=B7=D0=B0=D1=86=D0=B5=D0=B2,=20=D0=B2=D0=B5?= =?UTF-8?q?=D1=80=D0=BD=D1=83=D0=B2=D1=88=D0=B8=D1=85=D1=81=D1=8F=20=D0=BD?= =?UTF-8?q?=D0=B0=20=D1=8F=D0=B7=D1=8B=D0=BA=D0=B5=20=D0=BE=D1=80=D0=B8?= =?UTF-8?q?=D0=B3=D0=B8=D0=BD=D0=B0=D0=BB=D0=B0:=20=D1=83=D0=BF=D1=80?= =?UTF-8?q?=D0=B0=D0=B6=D0=BD=D0=B5=D0=BD=D0=B8=D1=8F=20=D0=B2=D0=B8=D0=B4?= =?UTF-8?q?=D0=B0=20[2]=20=D0=BC=D0=BE=D0=B4=D0=B5=D0=BB=D1=8C=20=D0=BD?= =?UTF-8?q?=D0=B5=20=D0=BF=D0=B5=D1=80=D0=B5=D0=B2=D0=BE=D0=B4=D0=B8=D1=82?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- SKILL.md | 7 +++++++ scripts/test_translate.py | 27 +++++++++++++++++++++++++++ scripts/translate.py | 11 +++++++++-- 3 files changed, 43 insertions(+), 2 deletions(-) diff --git a/SKILL.md b/SKILL.md index 852756a..d800440 100644 --- a/SKILL.md +++ b/SKILL.md @@ -190,6 +190,13 @@ is the bridge: it writes the repo's input format, shells out to The one thing that *does* skip work: deleting a chapter from `extracted/` before the run. It is then never sent, and `rebuild` keeps the original English for it — the right treatment for an index. +**Numbered exercise items come back untranslated.** Measured on Stroustrup +2026-08-21: 182 paragraphs of 10151 (1.8%) shaped `[2] Expanding on what you +have learned…` were echoed back in English. The document-wide Cyrillic check +cannot see this — 1.8% drowns in it — so `rebuild` counts them separately as +"осталось на языке оригинала". If the count is high, add an explicit line to the +prompt that numbered items are prose and must be translated too. + 4. **Read the reported counts.** "без перевода" above zero means a chapter came back with a different paragraph count and kept its original text; "разметка потеряна" counts paragraphs where the model mangled the inline-tag markers diff --git a/scripts/test_translate.py b/scripts/test_translate.py index 7e3d6d5..4698e35 100644 --- a/scripts/test_translate.py +++ b/scripts/test_translate.py @@ -132,6 +132,32 @@ def test_junk_translation_falls_back_to_original(tmp: Path): assert "Второй абзац" in result, "нормальный перевод должен остаться" +def test_untouched_paragraphs_are_counted(tmp: Path): + """Абзац, вернувшийся по-английски, доля кириллицы по документу не ловит: + у Страуструпа так осталось 182 упражнения из 10151 блока.""" + src = tmp / "book.html" + en = ("Expanding on what you have learned, write a program that lists the " + "instructions for a computer to find the upstairs bedroom. ") * 2 + ru_src = ("Этот абзац достаточно длинный, чтобы попасть под проверку языка " + "и быть переведённым как положено. ") * 2 + src.write_text("\n".join(["

Chapter

", "

%s

" % en, "

%s

" % en]), + encoding="utf-8") + chapters = t.split_chapters(t.parse_blocks(src)) + trans = tmp / "translations" + trans.mkdir() + (trans / "chapter_000_translated.json").write_text(json.dumps( + {"number": 0, "paragraphs": ["Глава", en, ru_src]}, ensure_ascii=False), + encoding="utf-8") + + out = tmp / "out.html" + import io, contextlib + buf = io.StringIO() + with contextlib.redirect_stdout(buf): + t.rebuild(src, chapters, tmp, out) + report = buf.getvalue() + assert "осталось на языке оригинала: 1" in report, report + + def test_workdir_belongs_to_one_book(tmp: Path): """Имена chapter_NNN.json у всех книг одинаковы: чужой workdir склеит перевод одной книги с текстом другой.""" @@ -183,6 +209,7 @@ if __name__ == "__main__": test_language_detection() for case in (test_split_and_rebuild, test_code_listings_are_left_alone, test_junk_translation_falls_back_to_original, + test_untouched_paragraphs_are_counted, test_workdir_belongs_to_one_book, test_stale_chapters_removed, test_verify_output_catches_untranslated): with tempfile.TemporaryDirectory() as d: diff --git a/scripts/translate.py b/scripts/translate.py index 4f56878..f4e5187 100644 --- a/scripts/translate.py +++ b/scripts/translate.py @@ -156,7 +156,7 @@ MIN_LEN_SHARE = 0.25 def rebuild(src, chapters, workdir, out): lines = src.read_text(encoding="utf-8").splitlines() - broken = missing = translated = junk = 0 + broken = missing = translated = junk = untouched = 0 for n, ch in enumerate(chapters): f = workdir / "translations" / ("chapter_%03d_translated.json" % n) if not f.exists(): @@ -174,13 +174,20 @@ def rebuild(src, chapters, workdir, out): junk += 1 lines[idx] = "<%s%s>%s" % (tag, attrs, original, tag) continue # оставляем оригинал: английский абзац лучше огрызка + # Абзац, вернувшийся на языке оригинала. Доля кириллицы по всему + # документу такое не ловит: полтора процента в ней тонут, а на + # странице это заметный кусок английского текста. + if len(plain_src) > 150 and detect_language(re.sub("<[^>]+>", "", text))[0] \ + == detect_language(plain_src)[0] != "unknown": + untouched += 1 body, ok = from_marks(text) broken += not ok translated += 1 lines[idx] = "<%s%s>%s" % (tag, attrs, body, tag) out.write_text("\n".join(lines) + "\n", encoding="utf-8") print("переведено блоков: %d, без перевода: %d, разметка потеряна в %d, " - "огрызков заменено оригиналом: %d" % (translated, missing, broken, junk)) + "огрызков заменено оригиналом: %d, осталось на языке оригинала: %d" + % (translated, missing, broken, junk, untouched)) return missing == 0 and junk * 200 <= translated