Files
pdf2epub/scripts/audiobook.py
T

195 lines
8.0 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Обвязка вокруг стадии озвучки book_translator (репозиторий не модифицируется).
prep — копия translations/ без разметочных маркеров, пригодная для TTS.
verify — проверка целостности: не потерялись ли фрагменты при синтезе.
Зачем verify: edge-tts бесплатный и на большом объёме отказывает; сторонний
скрипт пропускает уже существующий файл, проверяя только «размер больше нуля»,
и молча склеивает то, что получилось. Дыры в звуке обнаруживаются лишь при
прослушивании — эта проверка ловит их сразу после генерации.
"""
import argparse
import json
import re
import statistics
import subprocess
import sys
from pathlib import Path
MARK = re.compile(r"⟦/?[ib]⟧")
FRAG = re.compile(r"^chapter_(\d{3})_(intro|para_\d{4})\.mp3$")
# ниже этой доли от расчётной длительности глава считается обрезанной
SHORT = 0.75
LONG = 1.6
def clean(text):
return re.sub(r"\s{2,}", " ", MARK.sub("", text)).strip()
def cmd_prep(args):
src = args.translations
files = sorted(src.glob("chapter_*_translated*.json"))
if not files:
sys.exit("в %s нет chapter_*_translated*.json" % src)
args.output.mkdir(parents=True, exist_ok=True)
stripped = empty = 0
for f in files:
data = json.loads(f.read_text(encoding="utf-8"))
paragraphs = []
for p in data.get("paragraphs", []):
stripped += len(MARK.findall(p))
c = clean(p)
if c:
paragraphs.append(c)
else:
empty += 1
data["paragraphs"] = paragraphs
if "title" in data:
data["title"] = clean(data["title"])
(args.output / f.name).write_text(
json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
print("глав: %d, снято маркеров: %d, пустых абзацев отброшено: %d"
% (len(files), stripped, empty))
print("каталог для озвучки: %s" % args.output)
print("запускать: 05_create_audiobook.py --translations-dir %s" % args.output)
def duration(path):
"""Длительность mp3 в секундах; None если файл битый или пустой."""
if path.stat().st_size == 0:
return None
r = subprocess.run(
["ffprobe", "-v", "error", "-show_entries", "format=duration",
"-of", "default=nw=1:nk=1", str(path)],
capture_output=True, text=True)
try:
return float(r.stdout.strip())
except ValueError:
return None
def expected_fragments(translations):
"""{номер главы: число абзацев} по подготовленным для TTS файлам."""
out = {}
for f in sorted(translations.glob("chapter_*_translated*.json")):
m = re.search(r"chapter_(\d+)", f.name)
data = json.loads(f.read_text(encoding="utf-8"))
paragraphs = [p for p in data.get("paragraphs", []) if p.strip()]
out[int(m.group(1))] = {
"paragraphs": len(paragraphs),
"chars": sum(len(p) for p in paragraphs),
}
return out
def cmd_verify(args):
expect = expected_fragments(args.translations)
if not expect:
sys.exit("в %s нет подготовленных глав" % args.translations)
temp = args.audiobook / "temp_audio"
total_chars = sum(v["chars"] for v in expect.values())
if not temp.is_dir():
print("temp_audio/ уже убран — доступна только проверка итогового файла")
return check_final(args.audiobook, total_chars, None)
found, bad = {}, []
for mp3 in temp.glob("*.mp3"):
m = FRAG.match(mp3.name)
if not m:
continue
d = duration(mp3)
if d is None or d <= 0:
bad.append(mp3.name)
continue
found.setdefault(int(m.group(1)), []).append(d)
# секунд на знак — калибруем по самим главам, чтобы не зависеть от
# голоса и скорости речи
rates = [sum(found[n]) / expect[n]["chars"]
for n in found if n in expect and expect[n]["chars"]]
rate = statistics.median(rates) if rates else 0.0
problems = 0
for n in sorted(expect):
want = expect[n]["paragraphs"] + 1 # +1 вводный фрагмент
got = len(found.get(n, []))
secs = sum(found.get(n, []))
pred = expect[n]["chars"] * rate
note = []
if got < want:
note.append("не хватает %d фрагментов" % (want - got))
if pred and secs < pred * SHORT:
note.append("короче расчётной на %.0f%%" % (100 * (1 - secs / pred)))
if pred and secs > pred * LONG:
note.append("длиннее расчётной в %.1f раза" % (secs / pred))
if note:
problems += 1
print("глава %03d: %s (%d/%d фрагментов, %.1f мин)"
% (n, "; ".join(note), got, want, secs / 60))
total = sum(sum(v) for v in found.values())
print("\nфрагментов: %d из %d, звучание %.1f ч, темп %.1f знака/с"
% (sum(len(v) for v in found.values()),
sum(v["paragraphs"] + 1 for v in expect.values()),
total / 3600, (1 / rate) if rate else 0))
if bad:
print("битых или пустых файлов: %d (удали их и перезапусти синтез — "
"сторонний скрипт пропускает только непустые)" % len(bad))
for name in bad[:10]:
print(" ", name)
if problems:
print("глав с замечаниями: %d — склеивать рано" % problems)
else:
print("пропусков не найдено")
ok_final = check_final(args.audiobook, total_chars, rate)
return not problems and not bad and ok_final
def check_final(audiobook, total_chars, rate):
final = audiobook / "audiobook_complete.mp3"
if not final.exists():
print("итоговый audiobook_complete.mp3 ещё не собран")
return True
d = duration(final)
if d is None:
print("итоговый файл битый")
return False
mb = final.stat().st_size / 1024 / 1024
print("итоговый файл: %.1f ч, %.0f МБ" % (d / 3600, mb))
if rate:
pred = total_chars * rate
if d < pred * SHORT:
print("ВНИМАНИЕ: итог короче суммы глав на %.0f%% — потеряно при склейке"
% (100 * (1 - d / pred)))
return False
return True
def main():
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
sub = ap.add_subparsers(dest="cmd", required=True)
p = sub.add_parser("prep", help="снять маркеры разметки для TTS")
p.add_argument("translations", type=Path, help="каталог translations/ после перевода")
p.add_argument("output", type=Path, help="куда положить копию для озвучки")
p.set_defaults(func=cmd_prep)
v = sub.add_parser("verify", help="проверить целостность синтеза")
v.add_argument("translations", type=Path, help="каталог, поданный в озвучку")
v.add_argument("audiobook", type=Path, help="каталог audiobook/")
v.set_defaults(func=cmd_verify)
args = ap.parse_args()
result = args.func(args)
if result is False:
sys.exit(1)
if __name__ == "__main__":
main()