Счётчик абзацев, вернувшихся на языке оригинала: упражнения вида [2] модель не переводит

This commit is contained in:
chesirecatt
2026-08-21 17:47:48 +03:00
parent 7ed6e500ca
commit 14c7fad763
3 changed files with 43 additions and 2 deletions
+7
View File
@@ -190,6 +190,13 @@ is the bridge: it writes the repo's input format, shells out to
The one thing that *does* skip work: deleting a chapter from `extracted/` The one thing that *does* skip work: deleting a chapter from `extracted/`
before the run. It is then never sent, and `rebuild` keeps the original before the run. It is then never sent, and `rebuild` keeps the original
English for it — the right treatment for an index. English for it — the right treatment for an index.
**Numbered exercise items come back untranslated.** Measured on Stroustrup
2026-08-21: 182 paragraphs of 10151 (1.8%) shaped `[2] Expanding on what you
have learned…` were echoed back in English. The document-wide Cyrillic check
cannot see this — 1.8% drowns in it — so `rebuild` counts them separately as
"осталось на языке оригинала". If the count is high, add an explicit line to the
prompt that numbered items are prose and must be translated too.
4. **Read the reported counts.** "без перевода" above zero means a chapter came 4. **Read the reported counts.** "без перевода" above zero means a chapter came
back with a different paragraph count and kept its original text; "разметка back with a different paragraph count and kept its original text; "разметка
потеряна" counts paragraphs where the model mangled the inline-tag markers потеряна" counts paragraphs where the model mangled the inline-tag markers
+27
View File
@@ -132,6 +132,32 @@ def test_junk_translation_falls_back_to_original(tmp: Path):
assert "Второй абзац" in result, "нормальный перевод должен остаться" assert "Второй абзац" in result, "нормальный перевод должен остаться"
def test_untouched_paragraphs_are_counted(tmp: Path):
"""Абзац, вернувшийся по-английски, доля кириллицы по документу не ловит:
у Страуструпа так осталось 182 упражнения из 10151 блока."""
src = tmp / "book.html"
en = ("Expanding on what you have learned, write a program that lists the "
"instructions for a computer to find the upstairs bedroom. ") * 2
ru_src = ("Этот абзац достаточно длинный, чтобы попасть под проверку языка "
"и быть переведённым как положено. ") * 2
src.write_text("\n".join(["<h1>Chapter</h1>", "<p>%s</p>" % en, "<p>%s</p>" % en]),
encoding="utf-8")
chapters = t.split_chapters(t.parse_blocks(src))
trans = tmp / "translations"
trans.mkdir()
(trans / "chapter_000_translated.json").write_text(json.dumps(
{"number": 0, "paragraphs": ["Глава", en, ru_src]}, ensure_ascii=False),
encoding="utf-8")
out = tmp / "out.html"
import io, contextlib
buf = io.StringIO()
with contextlib.redirect_stdout(buf):
t.rebuild(src, chapters, tmp, out)
report = buf.getvalue()
assert "осталось на языке оригинала: 1" in report, report
def test_workdir_belongs_to_one_book(tmp: Path): def test_workdir_belongs_to_one_book(tmp: Path):
"""Имена chapter_NNN.json у всех книг одинаковы: чужой workdir склеит """Имена chapter_NNN.json у всех книг одинаковы: чужой workdir склеит
перевод одной книги с текстом другой.""" перевод одной книги с текстом другой."""
@@ -183,6 +209,7 @@ if __name__ == "__main__":
test_language_detection() test_language_detection()
for case in (test_split_and_rebuild, test_code_listings_are_left_alone, for case in (test_split_and_rebuild, test_code_listings_are_left_alone,
test_junk_translation_falls_back_to_original, test_junk_translation_falls_back_to_original,
test_untouched_paragraphs_are_counted,
test_workdir_belongs_to_one_book, test_stale_chapters_removed, test_workdir_belongs_to_one_book, test_stale_chapters_removed,
test_verify_output_catches_untranslated): test_verify_output_catches_untranslated):
with tempfile.TemporaryDirectory() as d: with tempfile.TemporaryDirectory() as d:
+9 -2
View File
@@ -156,7 +156,7 @@ MIN_LEN_SHARE = 0.25
def rebuild(src, chapters, workdir, out): def rebuild(src, chapters, workdir, out):
lines = src.read_text(encoding="utf-8").splitlines() lines = src.read_text(encoding="utf-8").splitlines()
broken = missing = translated = junk = 0 broken = missing = translated = junk = untouched = 0
for n, ch in enumerate(chapters): for n, ch in enumerate(chapters):
f = workdir / "translations" / ("chapter_%03d_translated.json" % n) f = workdir / "translations" / ("chapter_%03d_translated.json" % n)
if not f.exists(): if not f.exists():
@@ -174,13 +174,20 @@ def rebuild(src, chapters, workdir, out):
junk += 1 junk += 1
lines[idx] = "<%s%s>%s</%s>" % (tag, attrs, original, tag) lines[idx] = "<%s%s>%s</%s>" % (tag, attrs, original, tag)
continue # оставляем оригинал: английский абзац лучше огрызка continue # оставляем оригинал: английский абзац лучше огрызка
# Абзац, вернувшийся на языке оригинала. Доля кириллицы по всему
# документу такое не ловит: полтора процента в ней тонут, а на
# странице это заметный кусок английского текста.
if len(plain_src) > 150 and detect_language(re.sub("<[^>]+>", "", text))[0] \
== detect_language(plain_src)[0] != "unknown":
untouched += 1
body, ok = from_marks(text) body, ok = from_marks(text)
broken += not ok broken += not ok
translated += 1 translated += 1
lines[idx] = "<%s%s>%s</%s>" % (tag, attrs, body, tag) lines[idx] = "<%s%s>%s</%s>" % (tag, attrs, body, tag)
out.write_text("\n".join(lines) + "\n", encoding="utf-8") out.write_text("\n".join(lines) + "\n", encoding="utf-8")
print("переведено блоков: %d, без перевода: %d, разметка потеряна в %d, " print("переведено блоков: %d, без перевода: %d, разметка потеряна в %d, "
"огрызков заменено оригиналом: %d" % (translated, missing, broken, junk)) "огрызков заменено оригиналом: %d, осталось на языке оригинала: %d"
% (translated, missing, broken, junk, untouched))
return missing == 0 and junk * 200 <= translated return missing == 0 and junk * 200 <= translated