Счётчик абзацев, вернувшихся на языке оригинала: упражнения вида [2] модель не переводит
This commit is contained in:
@@ -190,6 +190,13 @@ is the bridge: it writes the repo's input format, shells out to
|
|||||||
The one thing that *does* skip work: deleting a chapter from `extracted/`
|
The one thing that *does* skip work: deleting a chapter from `extracted/`
|
||||||
before the run. It is then never sent, and `rebuild` keeps the original
|
before the run. It is then never sent, and `rebuild` keeps the original
|
||||||
English for it — the right treatment for an index.
|
English for it — the right treatment for an index.
|
||||||
|
**Numbered exercise items come back untranslated.** Measured on Stroustrup
|
||||||
|
2026-08-21: 182 paragraphs of 10151 (1.8%) shaped `[2] Expanding on what you
|
||||||
|
have learned…` were echoed back in English. The document-wide Cyrillic check
|
||||||
|
cannot see this — 1.8% drowns in it — so `rebuild` counts them separately as
|
||||||
|
"осталось на языке оригинала". If the count is high, add an explicit line to the
|
||||||
|
prompt that numbered items are prose and must be translated too.
|
||||||
|
|
||||||
4. **Read the reported counts.** "без перевода" above zero means a chapter came
|
4. **Read the reported counts.** "без перевода" above zero means a chapter came
|
||||||
back with a different paragraph count and kept its original text; "разметка
|
back with a different paragraph count and kept its original text; "разметка
|
||||||
потеряна" counts paragraphs where the model mangled the inline-tag markers
|
потеряна" counts paragraphs where the model mangled the inline-tag markers
|
||||||
|
|||||||
@@ -132,6 +132,32 @@ def test_junk_translation_falls_back_to_original(tmp: Path):
|
|||||||
assert "Второй абзац" in result, "нормальный перевод должен остаться"
|
assert "Второй абзац" in result, "нормальный перевод должен остаться"
|
||||||
|
|
||||||
|
|
||||||
|
def test_untouched_paragraphs_are_counted(tmp: Path):
|
||||||
|
"""Абзац, вернувшийся по-английски, доля кириллицы по документу не ловит:
|
||||||
|
у Страуструпа так осталось 182 упражнения из 10151 блока."""
|
||||||
|
src = tmp / "book.html"
|
||||||
|
en = ("Expanding on what you have learned, write a program that lists the "
|
||||||
|
"instructions for a computer to find the upstairs bedroom. ") * 2
|
||||||
|
ru_src = ("Этот абзац достаточно длинный, чтобы попасть под проверку языка "
|
||||||
|
"и быть переведённым как положено. ") * 2
|
||||||
|
src.write_text("\n".join(["<h1>Chapter</h1>", "<p>%s</p>" % en, "<p>%s</p>" % en]),
|
||||||
|
encoding="utf-8")
|
||||||
|
chapters = t.split_chapters(t.parse_blocks(src))
|
||||||
|
trans = tmp / "translations"
|
||||||
|
trans.mkdir()
|
||||||
|
(trans / "chapter_000_translated.json").write_text(json.dumps(
|
||||||
|
{"number": 0, "paragraphs": ["Глава", en, ru_src]}, ensure_ascii=False),
|
||||||
|
encoding="utf-8")
|
||||||
|
|
||||||
|
out = tmp / "out.html"
|
||||||
|
import io, contextlib
|
||||||
|
buf = io.StringIO()
|
||||||
|
with contextlib.redirect_stdout(buf):
|
||||||
|
t.rebuild(src, chapters, tmp, out)
|
||||||
|
report = buf.getvalue()
|
||||||
|
assert "осталось на языке оригинала: 1" in report, report
|
||||||
|
|
||||||
|
|
||||||
def test_workdir_belongs_to_one_book(tmp: Path):
|
def test_workdir_belongs_to_one_book(tmp: Path):
|
||||||
"""Имена chapter_NNN.json у всех книг одинаковы: чужой workdir склеит
|
"""Имена chapter_NNN.json у всех книг одинаковы: чужой workdir склеит
|
||||||
перевод одной книги с текстом другой."""
|
перевод одной книги с текстом другой."""
|
||||||
@@ -183,6 +209,7 @@ if __name__ == "__main__":
|
|||||||
test_language_detection()
|
test_language_detection()
|
||||||
for case in (test_split_and_rebuild, test_code_listings_are_left_alone,
|
for case in (test_split_and_rebuild, test_code_listings_are_left_alone,
|
||||||
test_junk_translation_falls_back_to_original,
|
test_junk_translation_falls_back_to_original,
|
||||||
|
test_untouched_paragraphs_are_counted,
|
||||||
test_workdir_belongs_to_one_book, test_stale_chapters_removed,
|
test_workdir_belongs_to_one_book, test_stale_chapters_removed,
|
||||||
test_verify_output_catches_untranslated):
|
test_verify_output_catches_untranslated):
|
||||||
with tempfile.TemporaryDirectory() as d:
|
with tempfile.TemporaryDirectory() as d:
|
||||||
|
|||||||
@@ -156,7 +156,7 @@ MIN_LEN_SHARE = 0.25
|
|||||||
|
|
||||||
def rebuild(src, chapters, workdir, out):
|
def rebuild(src, chapters, workdir, out):
|
||||||
lines = src.read_text(encoding="utf-8").splitlines()
|
lines = src.read_text(encoding="utf-8").splitlines()
|
||||||
broken = missing = translated = junk = 0
|
broken = missing = translated = junk = untouched = 0
|
||||||
for n, ch in enumerate(chapters):
|
for n, ch in enumerate(chapters):
|
||||||
f = workdir / "translations" / ("chapter_%03d_translated.json" % n)
|
f = workdir / "translations" / ("chapter_%03d_translated.json" % n)
|
||||||
if not f.exists():
|
if not f.exists():
|
||||||
@@ -174,13 +174,20 @@ def rebuild(src, chapters, workdir, out):
|
|||||||
junk += 1
|
junk += 1
|
||||||
lines[idx] = "<%s%s>%s</%s>" % (tag, attrs, original, tag)
|
lines[idx] = "<%s%s>%s</%s>" % (tag, attrs, original, tag)
|
||||||
continue # оставляем оригинал: английский абзац лучше огрызка
|
continue # оставляем оригинал: английский абзац лучше огрызка
|
||||||
|
# Абзац, вернувшийся на языке оригинала. Доля кириллицы по всему
|
||||||
|
# документу такое не ловит: полтора процента в ней тонут, а на
|
||||||
|
# странице это заметный кусок английского текста.
|
||||||
|
if len(plain_src) > 150 and detect_language(re.sub("<[^>]+>", "", text))[0] \
|
||||||
|
== detect_language(plain_src)[0] != "unknown":
|
||||||
|
untouched += 1
|
||||||
body, ok = from_marks(text)
|
body, ok = from_marks(text)
|
||||||
broken += not ok
|
broken += not ok
|
||||||
translated += 1
|
translated += 1
|
||||||
lines[idx] = "<%s%s>%s</%s>" % (tag, attrs, body, tag)
|
lines[idx] = "<%s%s>%s</%s>" % (tag, attrs, body, tag)
|
||||||
out.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
out.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||||
print("переведено блоков: %d, без перевода: %d, разметка потеряна в %d, "
|
print("переведено блоков: %d, без перевода: %d, разметка потеряна в %d, "
|
||||||
"огрызков заменено оригиналом: %d" % (translated, missing, broken, junk))
|
"огрызков заменено оригиналом: %d, осталось на языке оригинала: %d"
|
||||||
|
% (translated, missing, broken, junk, untouched))
|
||||||
return missing == 0 and junk * 200 <= translated
|
return missing == 0 and junk * 200 <= translated
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user