diff --git a/SKILL.md b/SKILL.md index 852756a..d800440 100644 --- a/SKILL.md +++ b/SKILL.md @@ -190,6 +190,13 @@ is the bridge: it writes the repo's input format, shells out to The one thing that *does* skip work: deleting a chapter from `extracted/` before the run. It is then never sent, and `rebuild` keeps the original English for it — the right treatment for an index. +**Numbered exercise items come back untranslated.** Measured on Stroustrup +2026-08-21: 182 paragraphs of 10151 (1.8%) shaped `[2] Expanding on what you +have learned…` were echoed back in English. The document-wide Cyrillic check +cannot see this — 1.8% drowns in it — so `rebuild` counts them separately as +"осталось на языке оригинала". If the count is high, add an explicit line to the +prompt that numbered items are prose and must be translated too. + 4. **Read the reported counts.** "без перевода" above zero means a chapter came back with a different paragraph count and kept its original text; "разметка потеряна" counts paragraphs where the model mangled the inline-tag markers diff --git a/scripts/test_translate.py b/scripts/test_translate.py index 7e3d6d5..4698e35 100644 --- a/scripts/test_translate.py +++ b/scripts/test_translate.py @@ -132,6 +132,32 @@ def test_junk_translation_falls_back_to_original(tmp: Path): assert "Второй абзац" in result, "нормальный перевод должен остаться" +def test_untouched_paragraphs_are_counted(tmp: Path): + """Абзац, вернувшийся по-английски, доля кириллицы по документу не ловит: + у Страуструпа так осталось 182 упражнения из 10151 блока.""" + src = tmp / "book.html" + en = ("Expanding on what you have learned, write a program that lists the " + "instructions for a computer to find the upstairs bedroom. ") * 2 + ru_src = ("Этот абзац достаточно длинный, чтобы попасть под проверку языка " + "и быть переведённым как положено. ") * 2 + src.write_text("\n".join(["
%s
" % en, "%s
" % en]), + encoding="utf-8") + chapters = t.split_chapters(t.parse_blocks(src)) + trans = tmp / "translations" + trans.mkdir() + (trans / "chapter_000_translated.json").write_text(json.dumps( + {"number": 0, "paragraphs": ["Глава", en, ru_src]}, ensure_ascii=False), + encoding="utf-8") + + out = tmp / "out.html" + import io, contextlib + buf = io.StringIO() + with contextlib.redirect_stdout(buf): + t.rebuild(src, chapters, tmp, out) + report = buf.getvalue() + assert "осталось на языке оригинала: 1" in report, report + + def test_workdir_belongs_to_one_book(tmp: Path): """Имена chapter_NNN.json у всех книг одинаковы: чужой workdir склеит перевод одной книги с текстом другой.""" @@ -183,6 +209,7 @@ if __name__ == "__main__": test_language_detection() for case in (test_split_and_rebuild, test_code_listings_are_left_alone, test_junk_translation_falls_back_to_original, + test_untouched_paragraphs_are_counted, test_workdir_belongs_to_one_book, test_stale_chapters_removed, test_verify_output_catches_untranslated): with tempfile.TemporaryDirectory() as d: diff --git a/scripts/translate.py b/scripts/translate.py index 4f56878..f4e5187 100644 --- a/scripts/translate.py +++ b/scripts/translate.py @@ -156,7 +156,7 @@ MIN_LEN_SHARE = 0.25 def rebuild(src, chapters, workdir, out): lines = src.read_text(encoding="utf-8").splitlines() - broken = missing = translated = junk = 0 + broken = missing = translated = junk = untouched = 0 for n, ch in enumerate(chapters): f = workdir / "translations" / ("chapter_%03d_translated.json" % n) if not f.exists(): @@ -174,13 +174,20 @@ def rebuild(src, chapters, workdir, out): junk += 1 lines[idx] = "<%s%s>%s%s>" % (tag, attrs, original, tag) continue # оставляем оригинал: английский абзац лучше огрызка + # Абзац, вернувшийся на языке оригинала. Доля кириллицы по всему + # документу такое не ловит: полтора процента в ней тонут, а на + # странице это заметный кусок английского текста. + if len(plain_src) > 150 and detect_language(re.sub("<[^>]+>", "", text))[0] \ + == detect_language(plain_src)[0] != "unknown": + untouched += 1 body, ok = from_marks(text) broken += not ok translated += 1 lines[idx] = "<%s%s>%s%s>" % (tag, attrs, body, tag) out.write_text("\n".join(lines) + "\n", encoding="utf-8") print("переведено блоков: %d, без перевода: %d, разметка потеряна в %d, " - "огрызков заменено оригиналом: %d" % (translated, missing, broken, junk)) + "огрызков заменено оригиналом: %d, осталось на языке оригинала: %d" + % (translated, missing, broken, junk, untouched)) return missing == 0 and junk * 200 <= translated