Files
pdf2epub/scripts/audiobook.py
T

480 lines
21 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Обвязка вокруг стадии озвучки book_translator (репозиторий не модифицируется).
prep — копия translations/ без разметочных маркеров, пригодная для TTS.
verify — проверка целостности: не потерялись ли фрагменты при синтезе.
Зачем verify: edge-tts бесплатный и на большом объёме отказывает; сторонний
скрипт пропускает уже существующий файл, проверяя только «размер больше нуля»,
и молча склеивает то, что получилось. Дыры в звуке обнаруживаются лишь при
прослушивании — эта проверка ловит их сразу после генерации.
"""
import argparse
import json
import re
import statistics
import subprocess
import sys
import time
from pathlib import Path
MARK = re.compile(r"⟦/?[ib]⟧")
# сторонний скрипт режет абзацы группами и пишет chapter_NNN_group_NNN.mp3;
# ветка с _para_ в нём есть, но не используется — принимаем обе
FRAG = re.compile(r"^chapter_(\d{3})_(intro|group_\d+|para_\d+)\.mp3$")
GROUP_SIZE = 3 # значение --paragraphs-per-group по умолчанию
# ниже этой доли от расчётной длительности глава считается обрезанной
SHORT = 0.75
LONG = 1.6
def clean(text):
return re.sub(r"\s{2,}", " ", MARK.sub("", text)).strip()
def cmd_prep(args):
src = args.translations
files = sorted(src.glob("chapter_*_translated*.json"))
if not files:
sys.exit("в %s нет chapter_*_translated*.json" % src)
args.output.mkdir(parents=True, exist_ok=True)
stripped = empty = 0
for f in files:
data = json.loads(f.read_text(encoding="utf-8"))
paragraphs = []
for p in data.get("paragraphs", []):
stripped += len(MARK.findall(p))
c = clean(p)
if c:
paragraphs.append(c)
else:
empty += 1
data["paragraphs"] = paragraphs
if "title" in data:
data["title"] = clean(data["title"])
(args.output / f.name).write_text(
json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
print("глав: %d, снято маркеров: %d, пустых абзацев отброшено: %d"
% (len(files), stripped, empty))
print("каталог для озвучки: %s" % args.output)
print("запускать: 05_create_audiobook.py --translations-dir %s" % args.output)
def duration(path):
"""Длительность mp3 в секундах; None если файл битый или пустой."""
if path.stat().st_size == 0:
return None
r = subprocess.run(
["ffprobe", "-v", "error", "-show_entries", "format=duration",
"-of", "default=nw=1:nk=1", str(path)],
capture_output=True, text=True)
try:
return float(r.stdout.strip())
except ValueError:
return None
def expected_fragments(translations, group_size=GROUP_SIZE):
"""{номер главы: сколько фрагментов ждать и сколько в главе знаков}.
Абзацы склеиваются группами по group_size, плюс один вводный фрагмент.
"""
out = {}
for f in sorted(translations.glob("chapter_*_translated*.json")):
m = re.search(r"chapter_(\d+)", f.name)
data = json.loads(f.read_text(encoding="utf-8"))
paragraphs = [p for p in data.get("paragraphs", []) if p.strip()]
groups = -(-len(paragraphs) // group_size) if paragraphs else 0
out[int(m.group(1))] = {
"paragraphs": len(paragraphs),
"fragments": groups + 1, # +1 вводный
"chars": sum(len(p) for p in paragraphs),
}
return out
def cmd_verify(args):
expect = expected_fragments(args.translations, args.group_size)
if not expect:
sys.exit("в %s нет подготовленных глав" % args.translations)
temp = args.audiobook / "temp_audio"
total_chars = sum(v["chars"] for v in expect.values())
if not temp.is_dir():
print("temp_audio/ уже убран — доступна только проверка итогового файла")
return check_final(args.audiobook, total_chars, None)
found, bad = {}, []
for mp3 in temp.glob("*.mp3"):
m = FRAG.match(mp3.name)
if not m:
continue
d = duration(mp3)
if d is None or d <= 0:
bad.append(mp3.name)
continue
found.setdefault(int(m.group(1)), []).append(d)
# секунд на знак — калибруем по самим главам, чтобы не зависеть от
# голоса и скорости речи
rates = [sum(found[n]) / expect[n]["chars"]
for n in found if n in expect and expect[n]["chars"]]
rate = statistics.median(rates) if rates else 0.0
problems = 0
for n in sorted(expect):
want = expect[n]["fragments"]
got = len(found.get(n, []))
secs = sum(found.get(n, []))
pred = expect[n]["chars"] * rate
note = []
if got < want:
note.append("не хватает %d фрагментов" % (want - got))
if pred and secs < pred * SHORT:
note.append("короче расчётной на %.0f%%" % (100 * (1 - secs / pred)))
if pred and secs > pred * LONG:
note.append("длиннее расчётной в %.1f раза" % (secs / pred))
if note:
problems += 1
print("глава %03d: %s (%d/%d фрагментов, %.1f мин)"
% (n, "; ".join(note), got, want, secs / 60))
total = sum(sum(v) for v in found.values())
print("\nфрагментов: %d из %d, звучание %.1f ч, темп %.1f знака/с"
% (sum(len(v) for v in found.values()),
sum(v["fragments"] for v in expect.values()),
total / 3600, (1 / rate) if rate else 0))
if bad:
print("битых или пустых файлов: %d (удали их и перезапусти синтез — "
"сторонний скрипт пропускает только непустые)" % len(bad))
for name in bad[:10]:
print(" ", name)
if problems:
print("глав с замечаниями: %d — склеивать рано" % problems)
else:
print("пропусков не найдено")
ok_final = check_final(args.audiobook, total_chars, rate)
return not problems and not bad and ok_final
def check_final(audiobook, total_chars, rate):
final = audiobook / "audiobook_complete.mp3"
if not final.exists():
print("итоговый audiobook_complete.mp3 ещё не собран")
return True
d = duration(final)
if d is None:
print("итоговый файл битый")
return False
mb = final.stat().st_size / 1024 / 1024
print("итоговый файл: %.1f ч, %.0f МБ" % (d / 3600, mb))
if rate:
pred = total_chars * rate
if d < pred * SHORT:
print("ВНИМАНИЕ: итог короче суммы глав на %.0f%% — потеряно при склейке"
% (100 * (1 - d / pred)))
return False
return True
def group_texts(translations, group_size):
"""{(глава, номер группы): текст} — ровно так, как их режет чужой скрипт."""
out = {}
for f in sorted(translations.glob("chapter_*_translated*.json")):
n = int(re.search(r"chapter_(\d+)", f.name).group(1))
data = json.loads(f.read_text(encoding="utf-8"))
paras = [p for p in data.get("paragraphs", [])
if p and not p.startswith("[IMAGE_")]
for i in range(0, len(paras), group_size):
out[(n, i // group_size)] = "\n\n".join(paras[i:i + group_size])
out[(n, "intro")] = "Глава %d. %s." % (n, data.get("title", ""))
return out
def speech_prep(repo, voice, rate, volume, outdir):
"""Их же подготовка текста (фонетические замены), чтобы добранный фрагмент
звучал так же, как соседние. Без репозитория — текст как есть."""
if repo is None:
return lambda s: s
sys.path.insert(0, str(repo))
try:
import importlib
mod = importlib.import_module("05_create_audiobook".lstrip("0") or "x")
except Exception:
try:
import importlib.util
spec = importlib.util.spec_from_file_location(
"ab_creator", repo / "05_create_audiobook.py")
mod = importlib.util.module_from_spec(spec)
spec.loader.exec_module(mod)
except Exception as e:
print("не удалось подключить подготовку текста (%s) — беру текст как есть"
% type(e).__name__)
return lambda s: s
try:
c = mod.AudiobookCreator(output_dir=str(outdir))
c.selected_voice, c.rate, c.volume = voice, rate, volume
return c.prepare_text_for_speech
except Exception as e:
print("не удалось создать AudiobookCreator (%s) — беру текст как есть"
% type(e).__name__)
return lambda s: s
def cmd_repair(args):
"""Досинтез потерянных фрагментов — последовательно, без параллели,
которая и приводит к отказам edge-tts."""
try:
import asyncio
import edge_tts
except ImportError:
sys.exit("нужен edge-tts: pip install edge-tts")
temp = args.audiobook / "temp_audio"
if not temp.is_dir():
sys.exit("нет %s" % temp)
expect = expected_fragments(args.translations, args.group_size)
texts = group_texts(args.translations, args.group_size)
prep = speech_prep(args.repo, args.voice, args.rate, args.volume, args.audiobook)
missing = []
for n in sorted(expect):
intro = temp / ("chapter_%03d_intro.mp3" % n)
if not duration(intro):
missing.append((intro, texts.get((n, "intro"), "")))
for g in range(expect[n]["fragments"] - 1):
f = temp / ("chapter_%03d_group_%03d.mp3" % (n, g))
if not duration(f):
missing.append((f, texts.get((n, g), "")))
if not missing:
print("добирать нечего — все фрагменты на месте")
return True
print("к досинтезу: %d фрагментов" % len(missing))
fixed = 0
for path, text in missing:
text = prep(text)
if not text.strip():
print(" %s: пустой текст, пропуск" % path.name)
continue
ok = False
for attempt in range(3):
try:
path.unlink(missing_ok=True)
asyncio.run(edge_tts.Communicate(
text, args.voice, rate=args.rate, volume=args.volume
).save(str(path)))
if duration(path):
ok = True
break
except Exception as e:
print(" %s: попытка %d — %s" % (path.name, attempt + 1, type(e).__name__))
time.sleep(2 * (attempt + 1))
if ok:
fixed += 1
print(" %s: готово (%.1f с)" % (path.name, duration(path)))
else:
path.unlink(missing_ok=True)
print(" %s: НЕ УДАЛОСЬ" % path.name)
print("досинтезировано %d из %d" % (fixed, len(missing)))
return fixed == len(missing)
def chapter_fragments(temp, num):
"""Фрагменты главы в порядке воспроизведения: вводный, затем группы абзацев.
Пустые и нечитаемые файлы отбрасываются: concat-демультиплексор на таком
файле обрывает склейку молча, с нулевым кодом возврата.
"""
intro = temp / ("chapter_%03d_intro.mp3" % num)
body = sorted(temp.glob("chapter_%03d_group_*.mp3" % num)) \
or sorted(temp.glob("chapter_%03d_para_*.mp3" % num))
files = ([intro] if intro.exists() else []) + body
good, bad = [], []
for f in files:
(good if duration(f) else bad).append(f)
return good, bad
def audio_params(sample):
"""Параметры кодирования первого фрагмента — чтобы тишина совпала и
склейка прошла без перекодирования."""
r = subprocess.run(
["ffprobe", "-v", "error", "-select_streams", "a:0",
"-show_entries", "stream=sample_rate,channels,bit_rate",
"-of", "default=nw=1", str(sample)],
capture_output=True, text=True)
p = dict(l.split("=", 1) for l in r.stdout.strip().splitlines() if "=" in l)
return (p.get("sample_rate", "24000"), p.get("channels", "1"),
p.get("bit_rate", "48000"))
def make_silence(path, seconds, params):
rate, channels, bitrate = params
subprocess.run(
["ffmpeg", "-v", "error", "-y", "-f", "lavfi",
"-i", "anullsrc=r=%s:cl=%s" % (rate, "mono" if channels == "1" else "stereo"),
"-t", str(seconds), "-b:a", bitrate, str(path)],
check=True)
def concat(files, out, tmp, meta):
"""Склейка через concat-демультиплексор. Сначала без перекодирования;
если mp3-потоки не сошлись — пересжатие."""
listing = tmp / "concat.txt"
listing.write_text(
"".join("file '%s'\n" % str(f.resolve()).replace("'", r"'\''") for f in files),
encoding="utf-8")
tags = []
for k, v in meta.items():
tags += ["-metadata", "%s=%s" % (k, v)]
base = ["ffmpeg", "-v", "error", "-y", "-f", "concat", "-safe", "0",
"-i", str(listing)]
r = subprocess.run(base + ["-c", "copy"] + tags + [str(out)],
capture_output=True, text=True)
if r.returncode == 0 and out.exists() and out.stat().st_size:
return True
r = subprocess.run(base + ["-c:a", "libmp3lame"] + tags + [str(out)],
capture_output=True, text=True)
if r.returncode:
print(" ffmpeg:", r.stderr.strip()[:200])
return False
return True
def safe_name(s, limit=70):
s = re.sub(r"[/\\\x00-\x1f]", " ", s)
s = re.sub(r"\s+", " ", s).strip(" .")
return s[:limit] or "Без названия"
def cmd_split(args):
"""Треки по главам вместо одного слитного файла."""
temp = args.audiobook / "temp_audio"
if not temp.is_dir():
sys.exit("нет %s — нарезать не из чего. Каталог удаляется методом "
"cleanup_temp_files() стороннего скрипта: режь до уборки." % temp)
expect = expected_fragments(args.translations, args.group_size)
titles = {}
for f in sorted(args.translations.glob("chapter_*_translated*.json")):
n = int(re.search(r"chapter_(\d+)", f.name).group(1))
titles[n] = json.loads(f.read_text(encoding="utf-8")).get("title", "")
args.output.mkdir(parents=True, exist_ok=True)
work = args.output / ".tmp"
work.mkdir(exist_ok=True)
first = next(iter(sorted(temp.glob("*.mp3"))), None)
if first is None:
sys.exit("в %s нет mp3" % temp)
params = audio_params(first)
silence = None
if args.gap > 0:
silence = work / "silence.mp3"
make_silence(silence, args.gap, params)
made = skipped = 0
for n in sorted(expect):
frags, bad = chapter_fragments(temp, n)
want = expect[n]["fragments"]
if bad:
print("глава %03d: %d битых фрагментов отброшено (%s)"
% (n, len(bad), ", ".join(f.name for f in bad[:3])))
if len(frags) < want and not args.force:
print("глава %03d: %d из %d фрагментов — пропущена (--force чтобы всё равно)"
% (n, len(frags), want))
skipped += 1
continue
seq = frags
if silence:
seq = []
for f in frags:
seq += [f, silence]
seq = seq[:-1]
title = safe_name(titles.get(n) or ("Глава %d" % n))
out = args.output / ("%03d - %s.mp3" % (n, title))
meta = {"title": title, "track": "%d/%d" % (n + 1, len(expect)),
"album": args.album, "artist": args.author,
"album_artist": args.author, "genre": "Audiobook"}
if not concat(seq, out, work, meta):
print("глава %03d: склейка не удалась" % n)
skipped += 1
continue
# ffmpeg может оборвать concat молча и вернуть 0 — сверяем результат
# с суммой входов
want_secs = sum(duration(f) or 0 for f in seq)
got_secs = duration(out) or 0
if want_secs and got_secs < want_secs * 0.98:
print("глава %03d: трек короче суммы фрагментов (%.1f мин против %.1f) "
"— склейка оборвалась, файл удалён"
% (n, got_secs / 60, want_secs / 60))
out.unlink(missing_ok=True)
skipped += 1
continue
made += 1
for f in work.glob("*"):
f.unlink()
work.rmdir()
total = sum(duration(f) or 0 for f in args.output.glob("*.mp3"))
size = sum(f.stat().st_size for f in args.output.glob("*.mp3")) / 1024 / 1024
print("\nтреков: %d, пропущено глав: %d, звучание %.1f ч, %.0f МБ"
% (made, skipped, total / 3600, size))
print("каталог: %s" % args.output)
return skipped == 0 and made > 0
def main():
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
sub = ap.add_subparsers(dest="cmd", required=True)
p = sub.add_parser("prep", help="снять маркеры разметки для TTS")
p.add_argument("translations", type=Path, help="каталог translations/ после перевода")
p.add_argument("output", type=Path, help="куда положить копию для озвучки")
p.set_defaults(func=cmd_prep)
v = sub.add_parser("verify", help="проверить целостность синтеза")
v.add_argument("translations", type=Path, help="каталог, поданный в озвучку")
v.add_argument("audiobook", type=Path, help="каталог audiobook/")
v.add_argument("--group-size", type=int, default=GROUP_SIZE,
help="сколько абзацев в одном фрагменте (--paragraphs-per-group)")
v.set_defaults(func=cmd_verify)
s = sub.add_parser("split", help="нарезать по главам вместо одного файла")
s.add_argument("translations", type=Path, help="каталог, поданный в озвучку")
s.add_argument("audiobook", type=Path, help="каталог audiobook/ с temp_audio/")
s.add_argument("output", type=Path, help="куда сложить треки")
s.add_argument("--album", default="Аудиокнига", help="название книги для тегов")
s.add_argument("--author", default="", help="автор для тегов")
s.add_argument("--gap", type=float, default=0.3,
help="пауза между абзацами, секунд (0 — без пауз)")
s.add_argument("--force", action="store_true",
help="резать даже главы с недостающими фрагментами")
s.add_argument("--group-size", type=int, default=GROUP_SIZE,
help="сколько абзацев в одном фрагменте (--paragraphs-per-group)")
s.set_defaults(func=cmd_split)
r = sub.add_parser("repair", help="досинтезировать потерянные фрагменты")
r.add_argument("translations", type=Path, help="каталог, поданный в озвучку")
r.add_argument("audiobook", type=Path, help="каталог audiobook/ с temp_audio/")
r.add_argument("--voice", default="ru-RU-DmitryNeural")
r.add_argument("--rate", default="+0%")
r.add_argument("--volume", default="+0%")
r.add_argument("--repo", type=Path, default=None,
help="клон book_translator — чтобы фонетика совпала с соседями")
r.add_argument("--group-size", type=int, default=GROUP_SIZE)
r.set_defaults(func=cmd_repair)
args = ap.parse_args()
result = args.func(args)
if result is False:
sys.exit(1)
if __name__ == "__main__":
main()