Озвучка: фрагменты режутся группами (не по абзацу), защита от тихого обрыва склейки, досинтез потерянных фрагментов
This commit is contained in:
+167
-15
@@ -15,10 +15,14 @@ import re
|
|||||||
import statistics
|
import statistics
|
||||||
import subprocess
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
|
import time
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
MARK = re.compile(r"⟦/?[ib]⟧")
|
MARK = re.compile(r"⟦/?[ib]⟧")
|
||||||
FRAG = re.compile(r"^chapter_(\d{3})_(intro|para_\d{4})\.mp3$")
|
# сторонний скрипт режет абзацы группами и пишет chapter_NNN_group_NNN.mp3;
|
||||||
|
# ветка с _para_ в нём есть, но не используется — принимаем обе
|
||||||
|
FRAG = re.compile(r"^chapter_(\d{3})_(intro|group_\d+|para_\d+)\.mp3$")
|
||||||
|
GROUP_SIZE = 3 # значение --paragraphs-per-group по умолчанию
|
||||||
# ниже этой доли от расчётной длительности глава считается обрезанной
|
# ниже этой доли от расчётной длительности глава считается обрезанной
|
||||||
SHORT = 0.75
|
SHORT = 0.75
|
||||||
LONG = 1.6
|
LONG = 1.6
|
||||||
@@ -70,22 +74,27 @@ def duration(path):
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def expected_fragments(translations):
|
def expected_fragments(translations, group_size=GROUP_SIZE):
|
||||||
"""{номер главы: число абзацев} по подготовленным для TTS файлам."""
|
"""{номер главы: сколько фрагментов ждать и сколько в главе знаков}.
|
||||||
|
|
||||||
|
Абзацы склеиваются группами по group_size, плюс один вводный фрагмент.
|
||||||
|
"""
|
||||||
out = {}
|
out = {}
|
||||||
for f in sorted(translations.glob("chapter_*_translated*.json")):
|
for f in sorted(translations.glob("chapter_*_translated*.json")):
|
||||||
m = re.search(r"chapter_(\d+)", f.name)
|
m = re.search(r"chapter_(\d+)", f.name)
|
||||||
data = json.loads(f.read_text(encoding="utf-8"))
|
data = json.loads(f.read_text(encoding="utf-8"))
|
||||||
paragraphs = [p for p in data.get("paragraphs", []) if p.strip()]
|
paragraphs = [p for p in data.get("paragraphs", []) if p.strip()]
|
||||||
|
groups = -(-len(paragraphs) // group_size) if paragraphs else 0
|
||||||
out[int(m.group(1))] = {
|
out[int(m.group(1))] = {
|
||||||
"paragraphs": len(paragraphs),
|
"paragraphs": len(paragraphs),
|
||||||
|
"fragments": groups + 1, # +1 вводный
|
||||||
"chars": sum(len(p) for p in paragraphs),
|
"chars": sum(len(p) for p in paragraphs),
|
||||||
}
|
}
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
def cmd_verify(args):
|
def cmd_verify(args):
|
||||||
expect = expected_fragments(args.translations)
|
expect = expected_fragments(args.translations, args.group_size)
|
||||||
if not expect:
|
if not expect:
|
||||||
sys.exit("в %s нет подготовленных глав" % args.translations)
|
sys.exit("в %s нет подготовленных глав" % args.translations)
|
||||||
temp = args.audiobook / "temp_audio"
|
temp = args.audiobook / "temp_audio"
|
||||||
@@ -114,7 +123,7 @@ def cmd_verify(args):
|
|||||||
|
|
||||||
problems = 0
|
problems = 0
|
||||||
for n in sorted(expect):
|
for n in sorted(expect):
|
||||||
want = expect[n]["paragraphs"] + 1 # +1 вводный фрагмент
|
want = expect[n]["fragments"]
|
||||||
got = len(found.get(n, []))
|
got = len(found.get(n, []))
|
||||||
secs = sum(found.get(n, []))
|
secs = sum(found.get(n, []))
|
||||||
pred = expect[n]["chars"] * rate
|
pred = expect[n]["chars"] * rate
|
||||||
@@ -133,7 +142,7 @@ def cmd_verify(args):
|
|||||||
total = sum(sum(v) for v in found.values())
|
total = sum(sum(v) for v in found.values())
|
||||||
print("\nфрагментов: %d из %d, звучание %.1f ч, темп %.1f знака/с"
|
print("\nфрагментов: %d из %d, звучание %.1f ч, темп %.1f знака/с"
|
||||||
% (sum(len(v) for v in found.values()),
|
% (sum(len(v) for v in found.values()),
|
||||||
sum(v["paragraphs"] + 1 for v in expect.values()),
|
sum(v["fragments"] for v in expect.values()),
|
||||||
total / 3600, (1 / rate) if rate else 0))
|
total / 3600, (1 / rate) if rate else 0))
|
||||||
if bad:
|
if bad:
|
||||||
print("битых или пустых файлов: %d (удали их и перезапусти синтез — "
|
print("битых или пустых файлов: %d (удали их и перезапусти синтез — "
|
||||||
@@ -169,11 +178,124 @@ def check_final(audiobook, total_chars, rate):
|
|||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def group_texts(translations, group_size):
|
||||||
|
"""{(глава, номер группы): текст} — ровно так, как их режет чужой скрипт."""
|
||||||
|
out = {}
|
||||||
|
for f in sorted(translations.glob("chapter_*_translated*.json")):
|
||||||
|
n = int(re.search(r"chapter_(\d+)", f.name).group(1))
|
||||||
|
data = json.loads(f.read_text(encoding="utf-8"))
|
||||||
|
paras = [p for p in data.get("paragraphs", [])
|
||||||
|
if p and not p.startswith("[IMAGE_")]
|
||||||
|
for i in range(0, len(paras), group_size):
|
||||||
|
out[(n, i // group_size)] = "\n\n".join(paras[i:i + group_size])
|
||||||
|
out[(n, "intro")] = "Глава %d. %s." % (n, data.get("title", ""))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def speech_prep(repo, voice, rate, volume, outdir):
|
||||||
|
"""Их же подготовка текста (фонетические замены), чтобы добранный фрагмент
|
||||||
|
звучал так же, как соседние. Без репозитория — текст как есть."""
|
||||||
|
if repo is None:
|
||||||
|
return lambda s: s
|
||||||
|
sys.path.insert(0, str(repo))
|
||||||
|
try:
|
||||||
|
import importlib
|
||||||
|
mod = importlib.import_module("05_create_audiobook".lstrip("0") or "x")
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
import importlib.util
|
||||||
|
spec = importlib.util.spec_from_file_location(
|
||||||
|
"ab_creator", repo / "05_create_audiobook.py")
|
||||||
|
mod = importlib.util.module_from_spec(spec)
|
||||||
|
spec.loader.exec_module(mod)
|
||||||
|
except Exception as e:
|
||||||
|
print("не удалось подключить подготовку текста (%s) — беру текст как есть"
|
||||||
|
% type(e).__name__)
|
||||||
|
return lambda s: s
|
||||||
|
try:
|
||||||
|
c = mod.AudiobookCreator(output_dir=str(outdir))
|
||||||
|
c.selected_voice, c.rate, c.volume = voice, rate, volume
|
||||||
|
return c.prepare_text_for_speech
|
||||||
|
except Exception as e:
|
||||||
|
print("не удалось создать AudiobookCreator (%s) — беру текст как есть"
|
||||||
|
% type(e).__name__)
|
||||||
|
return lambda s: s
|
||||||
|
|
||||||
|
|
||||||
|
def cmd_repair(args):
|
||||||
|
"""Досинтез потерянных фрагментов — последовательно, без параллели,
|
||||||
|
которая и приводит к отказам edge-tts."""
|
||||||
|
try:
|
||||||
|
import asyncio
|
||||||
|
import edge_tts
|
||||||
|
except ImportError:
|
||||||
|
sys.exit("нужен edge-tts: pip install edge-tts")
|
||||||
|
|
||||||
|
temp = args.audiobook / "temp_audio"
|
||||||
|
if not temp.is_dir():
|
||||||
|
sys.exit("нет %s" % temp)
|
||||||
|
expect = expected_fragments(args.translations, args.group_size)
|
||||||
|
texts = group_texts(args.translations, args.group_size)
|
||||||
|
prep = speech_prep(args.repo, args.voice, args.rate, args.volume, args.audiobook)
|
||||||
|
|
||||||
|
missing = []
|
||||||
|
for n in sorted(expect):
|
||||||
|
intro = temp / ("chapter_%03d_intro.mp3" % n)
|
||||||
|
if not duration(intro):
|
||||||
|
missing.append((intro, texts.get((n, "intro"), "")))
|
||||||
|
for g in range(expect[n]["fragments"] - 1):
|
||||||
|
f = temp / ("chapter_%03d_group_%03d.mp3" % (n, g))
|
||||||
|
if not duration(f):
|
||||||
|
missing.append((f, texts.get((n, g), "")))
|
||||||
|
|
||||||
|
if not missing:
|
||||||
|
print("добирать нечего — все фрагменты на месте")
|
||||||
|
return True
|
||||||
|
|
||||||
|
print("к досинтезу: %d фрагментов" % len(missing))
|
||||||
|
fixed = 0
|
||||||
|
for path, text in missing:
|
||||||
|
text = prep(text)
|
||||||
|
if not text.strip():
|
||||||
|
print(" %s: пустой текст, пропуск" % path.name)
|
||||||
|
continue
|
||||||
|
ok = False
|
||||||
|
for attempt in range(3):
|
||||||
|
try:
|
||||||
|
path.unlink(missing_ok=True)
|
||||||
|
asyncio.run(edge_tts.Communicate(
|
||||||
|
text, args.voice, rate=args.rate, volume=args.volume
|
||||||
|
).save(str(path)))
|
||||||
|
if duration(path):
|
||||||
|
ok = True
|
||||||
|
break
|
||||||
|
except Exception as e:
|
||||||
|
print(" %s: попытка %d — %s" % (path.name, attempt + 1, type(e).__name__))
|
||||||
|
time.sleep(2 * (attempt + 1))
|
||||||
|
if ok:
|
||||||
|
fixed += 1
|
||||||
|
print(" %s: готово (%.1f с)" % (path.name, duration(path)))
|
||||||
|
else:
|
||||||
|
path.unlink(missing_ok=True)
|
||||||
|
print(" %s: НЕ УДАЛОСЬ" % path.name)
|
||||||
|
print("досинтезировано %d из %d" % (fixed, len(missing)))
|
||||||
|
return fixed == len(missing)
|
||||||
|
|
||||||
|
|
||||||
def chapter_fragments(temp, num):
|
def chapter_fragments(temp, num):
|
||||||
"""Фрагменты главы в порядке воспроизведения: вводный, затем абзацы."""
|
"""Фрагменты главы в порядке воспроизведения: вводный, затем группы абзацев.
|
||||||
|
|
||||||
|
Пустые и нечитаемые файлы отбрасываются: concat-демультиплексор на таком
|
||||||
|
файле обрывает склейку молча, с нулевым кодом возврата.
|
||||||
|
"""
|
||||||
intro = temp / ("chapter_%03d_intro.mp3" % num)
|
intro = temp / ("chapter_%03d_intro.mp3" % num)
|
||||||
paras = sorted(temp.glob("chapter_%03d_para_*.mp3" % num))
|
body = sorted(temp.glob("chapter_%03d_group_*.mp3" % num)) \
|
||||||
return ([intro] if intro.exists() else []) + paras
|
or sorted(temp.glob("chapter_%03d_para_*.mp3" % num))
|
||||||
|
files = ([intro] if intro.exists() else []) + body
|
||||||
|
good, bad = [], []
|
||||||
|
for f in files:
|
||||||
|
(good if duration(f) else bad).append(f)
|
||||||
|
return good, bad
|
||||||
|
|
||||||
|
|
||||||
def audio_params(sample):
|
def audio_params(sample):
|
||||||
@@ -234,7 +356,7 @@ def cmd_split(args):
|
|||||||
if not temp.is_dir():
|
if not temp.is_dir():
|
||||||
sys.exit("нет %s — нарезать не из чего. Каталог удаляется методом "
|
sys.exit("нет %s — нарезать не из чего. Каталог удаляется методом "
|
||||||
"cleanup_temp_files() стороннего скрипта: режь до уборки." % temp)
|
"cleanup_temp_files() стороннего скрипта: режь до уборки." % temp)
|
||||||
expect = expected_fragments(args.translations)
|
expect = expected_fragments(args.translations, args.group_size)
|
||||||
titles = {}
|
titles = {}
|
||||||
for f in sorted(args.translations.glob("chapter_*_translated*.json")):
|
for f in sorted(args.translations.glob("chapter_*_translated*.json")):
|
||||||
n = int(re.search(r"chapter_(\d+)", f.name).group(1))
|
n = int(re.search(r"chapter_(\d+)", f.name).group(1))
|
||||||
@@ -255,8 +377,11 @@ def cmd_split(args):
|
|||||||
|
|
||||||
made = skipped = 0
|
made = skipped = 0
|
||||||
for n in sorted(expect):
|
for n in sorted(expect):
|
||||||
frags = chapter_fragments(temp, n)
|
frags, bad = chapter_fragments(temp, n)
|
||||||
want = expect[n]["paragraphs"] + 1
|
want = expect[n]["fragments"]
|
||||||
|
if bad:
|
||||||
|
print("глава %03d: %d битых фрагментов отброшено (%s)"
|
||||||
|
% (n, len(bad), ", ".join(f.name for f in bad[:3])))
|
||||||
if len(frags) < want and not args.force:
|
if len(frags) < want and not args.force:
|
||||||
print("глава %03d: %d из %d фрагментов — пропущена (--force чтобы всё равно)"
|
print("глава %03d: %d из %d фрагментов — пропущена (--force чтобы всё равно)"
|
||||||
% (n, len(frags), want))
|
% (n, len(frags), want))
|
||||||
@@ -273,10 +398,22 @@ def cmd_split(args):
|
|||||||
meta = {"title": title, "track": "%d/%d" % (n + 1, len(expect)),
|
meta = {"title": title, "track": "%d/%d" % (n + 1, len(expect)),
|
||||||
"album": args.album, "artist": args.author,
|
"album": args.album, "artist": args.author,
|
||||||
"album_artist": args.author, "genre": "Audiobook"}
|
"album_artist": args.author, "genre": "Audiobook"}
|
||||||
if concat(seq, out, work, meta):
|
if not concat(seq, out, work, meta):
|
||||||
made += 1
|
|
||||||
else:
|
|
||||||
print("глава %03d: склейка не удалась" % n)
|
print("глава %03d: склейка не удалась" % n)
|
||||||
|
skipped += 1
|
||||||
|
continue
|
||||||
|
# ffmpeg может оборвать concat молча и вернуть 0 — сверяем результат
|
||||||
|
# с суммой входов
|
||||||
|
want_secs = sum(duration(f) or 0 for f in seq)
|
||||||
|
got_secs = duration(out) or 0
|
||||||
|
if want_secs and got_secs < want_secs * 0.98:
|
||||||
|
print("глава %03d: трек короче суммы фрагментов (%.1f мин против %.1f) "
|
||||||
|
"— склейка оборвалась, файл удалён"
|
||||||
|
% (n, got_secs / 60, want_secs / 60))
|
||||||
|
out.unlink(missing_ok=True)
|
||||||
|
skipped += 1
|
||||||
|
continue
|
||||||
|
made += 1
|
||||||
|
|
||||||
for f in work.glob("*"):
|
for f in work.glob("*"):
|
||||||
f.unlink()
|
f.unlink()
|
||||||
@@ -303,6 +440,8 @@ def main():
|
|||||||
v = sub.add_parser("verify", help="проверить целостность синтеза")
|
v = sub.add_parser("verify", help="проверить целостность синтеза")
|
||||||
v.add_argument("translations", type=Path, help="каталог, поданный в озвучку")
|
v.add_argument("translations", type=Path, help="каталог, поданный в озвучку")
|
||||||
v.add_argument("audiobook", type=Path, help="каталог audiobook/")
|
v.add_argument("audiobook", type=Path, help="каталог audiobook/")
|
||||||
|
v.add_argument("--group-size", type=int, default=GROUP_SIZE,
|
||||||
|
help="сколько абзацев в одном фрагменте (--paragraphs-per-group)")
|
||||||
v.set_defaults(func=cmd_verify)
|
v.set_defaults(func=cmd_verify)
|
||||||
|
|
||||||
s = sub.add_parser("split", help="нарезать по главам вместо одного файла")
|
s = sub.add_parser("split", help="нарезать по главам вместо одного файла")
|
||||||
@@ -315,8 +454,21 @@ def main():
|
|||||||
help="пауза между абзацами, секунд (0 — без пауз)")
|
help="пауза между абзацами, секунд (0 — без пауз)")
|
||||||
s.add_argument("--force", action="store_true",
|
s.add_argument("--force", action="store_true",
|
||||||
help="резать даже главы с недостающими фрагментами")
|
help="резать даже главы с недостающими фрагментами")
|
||||||
|
s.add_argument("--group-size", type=int, default=GROUP_SIZE,
|
||||||
|
help="сколько абзацев в одном фрагменте (--paragraphs-per-group)")
|
||||||
s.set_defaults(func=cmd_split)
|
s.set_defaults(func=cmd_split)
|
||||||
|
|
||||||
|
r = sub.add_parser("repair", help="досинтезировать потерянные фрагменты")
|
||||||
|
r.add_argument("translations", type=Path, help="каталог, поданный в озвучку")
|
||||||
|
r.add_argument("audiobook", type=Path, help="каталог audiobook/ с temp_audio/")
|
||||||
|
r.add_argument("--voice", default="ru-RU-DmitryNeural")
|
||||||
|
r.add_argument("--rate", default="+0%")
|
||||||
|
r.add_argument("--volume", default="+0%")
|
||||||
|
r.add_argument("--repo", type=Path, default=None,
|
||||||
|
help="клон book_translator — чтобы фонетика совпала с соседями")
|
||||||
|
r.add_argument("--group-size", type=int, default=GROUP_SIZE)
|
||||||
|
r.set_defaults(func=cmd_repair)
|
||||||
|
|
||||||
args = ap.parse_args()
|
args = ap.parse_args()
|
||||||
result = args.func(args)
|
result = args.func(args)
|
||||||
if result is False:
|
if result is False:
|
||||||
|
|||||||
+41
-23
@@ -50,12 +50,12 @@ def test_verify_detects_gap(tmp: Path):
|
|||||||
for n in (0, 1):
|
for n in (0, 1):
|
||||||
write_chapter(tts, n, [para] * 4)
|
write_chapter(tts, n, [para] * 4)
|
||||||
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
|
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
|
||||||
# у главы 1 намеренно пропущен последний абзац
|
# у главы 1 намеренно пропущена последняя группа
|
||||||
count = 4 if n == 0 else 3
|
count = 2 if n == 0 else 1
|
||||||
for i in range(count):
|
for i in range(count):
|
||||||
mp3(temp / ("chapter_%03d_para_%04d.mp3" % (n, i)), 6)
|
mp3(temp / ("chapter_%03d_group_%03d.mp3" % (n, i)), 9)
|
||||||
|
|
||||||
ns = type("ns", (), {"translations": tts, "audiobook": ab})
|
ns = type("ns", (), {"translations": tts, "audiobook": ab, "group_size": 3})
|
||||||
assert a.cmd_verify(ns) is False, "пропущенный фрагмент должен завалить проверку"
|
assert a.cmd_verify(ns) is False, "пропущенный фрагмент должен завалить проверку"
|
||||||
|
|
||||||
|
|
||||||
@@ -68,9 +68,8 @@ def test_verify_clean(tmp: Path):
|
|||||||
for n in (0, 1):
|
for n in (0, 1):
|
||||||
write_chapter(tts, n, [para] * 3)
|
write_chapter(tts, n, [para] * 3)
|
||||||
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
|
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
|
||||||
for i in range(3):
|
mp3(temp / ("chapter_%03d_group_000.mp3" % n), 18)
|
||||||
mp3(temp / ("chapter_%03d_para_%04d.mp3" % (n, i)), 6)
|
ns = type("ns", (), {"translations": tts, "audiobook": ab, "group_size": 3})
|
||||||
ns = type("ns", (), {"translations": tts, "audiobook": ab})
|
|
||||||
assert a.cmd_verify(ns) is True, "целая книга должна проходить проверку"
|
assert a.cmd_verify(ns) is True, "целая книга должна проходить проверку"
|
||||||
|
|
||||||
|
|
||||||
@@ -80,17 +79,17 @@ def test_verify_detects_empty_file(tmp: Path):
|
|||||||
temp = ab / "temp_audio"
|
temp = ab / "temp_audio"
|
||||||
temp.mkdir(parents=True)
|
temp.mkdir(parents=True)
|
||||||
para = "х" * 100
|
para = "х" * 100
|
||||||
write_chapter(tts, 0, [para] * 3)
|
write_chapter(tts, 0, [para] * 9)
|
||||||
mp3(temp / "chapter_000_intro.mp3", 1)
|
mp3(temp / "chapter_000_intro.mp3", 1)
|
||||||
mp3(temp / "chapter_000_para_0000.mp3", 6)
|
mp3(temp / "chapter_000_group_000.mp3", 18)
|
||||||
mp3(temp / "chapter_000_para_0001.mp3", 6)
|
mp3(temp / "chapter_000_group_001.mp3", 18)
|
||||||
(temp / "chapter_000_para_0002.mp3").touch() # нулевой размер
|
(temp / "chapter_000_group_002.mp3").touch() # нулевой размер
|
||||||
ns = type("ns", (), {"translations": tts, "audiobook": ab})
|
ns = type("ns", (), {"translations": tts, "audiobook": ab, "group_size": 3})
|
||||||
assert a.cmd_verify(ns) is False, "пустой файл должен завалить проверку"
|
assert a.cmd_verify(ns) is False, "пустой файл должен завалить проверку"
|
||||||
|
|
||||||
|
|
||||||
def build_book(tmp: Path, chapters=2, paras=3):
|
def build_book(tmp: Path, chapters=2, paras=3, group=3, secs=2):
|
||||||
"""Готовый синтез: подготовленные главы + фрагменты в temp_audio/."""
|
"""Готовый синтез: главы + фрагменты, сгруппированные как в чужом скрипте."""
|
||||||
tts, ab = tmp / "tts", tmp / "audiobook"
|
tts, ab = tmp / "tts", tmp / "audiobook"
|
||||||
tts.mkdir()
|
tts.mkdir()
|
||||||
temp = ab / "temp_audio"
|
temp = ab / "temp_audio"
|
||||||
@@ -98,14 +97,15 @@ def build_book(tmp: Path, chapters=2, paras=3):
|
|||||||
for n in range(chapters):
|
for n in range(chapters):
|
||||||
write_chapter(tts, n, ["х" * 100] * paras)
|
write_chapter(tts, n, ["х" * 100] * paras)
|
||||||
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
|
mp3(temp / ("chapter_%03d_intro.mp3" % n), 1)
|
||||||
for i in range(paras):
|
for i in range(-(-paras // group)):
|
||||||
mp3(temp / ("chapter_%03d_para_%04d.mp3" % (n, i)), 2)
|
mp3(temp / ("chapter_%03d_group_%03d.mp3" % (n, i)), secs)
|
||||||
return tts, ab
|
return tts, ab
|
||||||
|
|
||||||
|
|
||||||
def split_ns(tts, ab, out, **kw):
|
def split_ns(tts, ab, out, **kw):
|
||||||
fields = {"translations": tts, "audiobook": ab, "output": out,
|
fields = {"translations": tts, "audiobook": ab, "output": out,
|
||||||
"album": "Книга", "author": "Автор", "gap": 0.3, "force": False}
|
"album": "Книга", "author": "Автор", "gap": 0.3,
|
||||||
|
"force": False, "group_size": 3}
|
||||||
fields.update(kw)
|
fields.update(kw)
|
||||||
return type("ns", (), fields)
|
return type("ns", (), fields)
|
||||||
|
|
||||||
@@ -119,9 +119,9 @@ def test_split_makes_chapter_tracks(tmp: Path):
|
|||||||
assert tracks[0].name.startswith("000 - "), tracks[0].name
|
assert tracks[0].name.startswith("000 - "), tracks[0].name
|
||||||
assert not (out / ".tmp").exists(), "временный каталог должен убираться"
|
assert not (out / ".tmp").exists(), "временный каталог должен убираться"
|
||||||
|
|
||||||
# длительность: 1 вводный + 3 абзаца + 3 паузы = 1 + 6 + 0.9 ≈ 7.9 с
|
# 1 вводный + 1 группа + 1 пауза = 1 + 2 + 0.3 ≈ 3.3 с
|
||||||
d = a.duration(tracks[0])
|
d = a.duration(tracks[0])
|
||||||
assert 7.0 < d < 9.0, d
|
assert 2.8 < d < 4.0, d
|
||||||
|
|
||||||
# теги проставлены
|
# теги проставлены
|
||||||
r = subprocess.run(
|
r = subprocess.run(
|
||||||
@@ -134,7 +134,7 @@ def test_split_makes_chapter_tracks(tmp: Path):
|
|||||||
|
|
||||||
def test_split_skips_incomplete_chapter(tmp: Path):
|
def test_split_skips_incomplete_chapter(tmp: Path):
|
||||||
tts, ab = build_book(tmp, chapters=2, paras=3)
|
tts, ab = build_book(tmp, chapters=2, paras=3)
|
||||||
(ab / "temp_audio" / "chapter_001_para_0002.mp3").unlink()
|
(ab / "temp_audio" / "chapter_001_group_000.mp3").unlink()
|
||||||
out = tmp / "tracks"
|
out = tmp / "tracks"
|
||||||
assert a.cmd_split(split_ns(tts, ab, out)) is False, "неполная глава должна пропускаться"
|
assert a.cmd_split(split_ns(tts, ab, out)) is False, "неполная глава должна пропускаться"
|
||||||
assert len(list(out.glob("*.mp3"))) == 1, "целая глава всё равно нарезается"
|
assert len(list(out.glob("*.mp3"))) == 1, "целая глава всё равно нарезается"
|
||||||
@@ -144,12 +144,29 @@ def test_split_skips_incomplete_chapter(tmp: Path):
|
|||||||
assert len(list(out2.glob("*.mp3"))) == 2, "--force режет обе"
|
assert len(list(out2.glob("*.mp3"))) == 2, "--force режет обе"
|
||||||
|
|
||||||
|
|
||||||
|
def test_split_survives_broken_fragment(tmp: Path):
|
||||||
|
"""Регресс: пустой фрагмент обрывал concat молча, ffmpeg возвращал 0."""
|
||||||
|
tts, ab = build_book(tmp, chapters=1, paras=9, secs=4)
|
||||||
|
broken = ab / "temp_audio" / "chapter_000_group_001.mp3"
|
||||||
|
broken.write_bytes(b"") # ровно тот случай, что дал edge-tts
|
||||||
|
out = tmp / "tracks"
|
||||||
|
assert a.cmd_split(split_ns(tts, ab, out)) is False, \
|
||||||
|
"глава с битым фрагментом не должна выдаваться как готовая"
|
||||||
|
assert not list(out.glob("*.mp3")), "обрезанный трек не должен остаться на диске"
|
||||||
|
|
||||||
|
# с --force собирается из уцелевших, но без тихого обрыва посередине
|
||||||
|
out2 = tmp / "forced"
|
||||||
|
assert a.cmd_split(split_ns(tts, ab, out2, force=True, gap=0)) is True
|
||||||
|
d = a.duration(next(out2.glob("*.mp3")))
|
||||||
|
assert 8.5 < d < 9.5, d # вводный 1 с + две уцелевшие группы по 4 с
|
||||||
|
|
||||||
|
|
||||||
def test_split_without_gap(tmp: Path):
|
def test_split_without_gap(tmp: Path):
|
||||||
tts, ab = build_book(tmp, chapters=1, paras=2)
|
tts, ab = build_book(tmp, chapters=1, paras=6)
|
||||||
out = tmp / "tracks"
|
out = tmp / "tracks"
|
||||||
assert a.cmd_split(split_ns(tts, ab, out, gap=0)) is True
|
assert a.cmd_split(split_ns(tts, ab, out, gap=0)) is True
|
||||||
d = a.duration(next(out.glob("*.mp3")))
|
d = a.duration(next(out.glob("*.mp3")))
|
||||||
assert 4.5 < d < 5.6, d # 1 + 2 + 2 без пауз
|
assert 4.5 < d < 5.6, d # 1 вводный + 2 группы по 2 с, без пауз
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
@@ -157,7 +174,8 @@ if __name__ == "__main__":
|
|||||||
sys.exit("нужен ffmpeg")
|
sys.exit("нужен ffmpeg")
|
||||||
for fn in (test_prep, test_verify_detects_gap, test_verify_clean,
|
for fn in (test_prep, test_verify_detects_gap, test_verify_clean,
|
||||||
test_verify_detects_empty_file, test_split_makes_chapter_tracks,
|
test_verify_detects_empty_file, test_split_makes_chapter_tracks,
|
||||||
test_split_skips_incomplete_chapter, test_split_without_gap):
|
test_split_skips_incomplete_chapter,
|
||||||
|
test_split_survives_broken_fragment, test_split_without_gap):
|
||||||
with tempfile.TemporaryDirectory() as d:
|
with tempfile.TemporaryDirectory() as d:
|
||||||
print("---", fn.__name__)
|
print("---", fn.__name__)
|
||||||
fn(Path(d))
|
fn(Path(d))
|
||||||
|
|||||||
Reference in New Issue
Block a user