Files
pdf2epub/scripts/silero_render.py
T

151 lines
6.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Озвучка книги локальной моделью Silero — треки по главам, без сети.
Зачем не edge-tts: тот бесплатен, но на объёме молча теряет фрагменты
(на одной главе выпадало от 1 до 7 из 49). Silero считает локально,
детерминированно и без пропусков.
ВАЖНО про железо: PyTorch считает нейросеть инструкциями AVX2. На процессоре
без них (проверять `lscpu | grep avx2`) скорость падает примерно в тридцать
раз — на Celeron N5095 вышло 3.4x реального времени против 97x на Ryzen 5800H.
Гипертрединг вредит: 8 потоков быстрее 16.
Модель: https://models.silero.ai/models/tts/ru/v5_5_ru.pt (145 МБ)
Нужны: torch (CPU-сборка), numpy, ffmpeg.
"""
import argparse
import html
import json
import pathlib
import re
import subprocess
import sys
import time
import wave
import torch
sys.path.insert(0, str(pathlib.Path(__file__).parent))
import tts_normalize as norm
SR = 48000
def ssml(text):
"""Абзац как <p> из <s>: модель сама строит интонацию конца фразы."""
sents = [s.strip() for s in re.split(r'(?<=[.!?…])\s+', text) if s.strip()]
sents = [s for s in sents if re.search(r'\w', s)]
if not sents:
return None
return '<speak><p>%s</p></speak>' % "".join(
"<s>%s</s>" % html.escape(s, quote=False) for s in sents)
def safe(s):
return re.sub(r'\s+', ' ', re.sub(r'[/\\\x00-\x1f]', ' ', s)).strip(' .')[:70]
def main():
ap = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("model", type=pathlib.Path, help="v5_5_ru.pt")
ap.add_argument("texts", type=pathlib.Path,
help="каталог с chapter_*_translated*.json (вывод audiobook.py prep)")
ap.add_argument("out", type=pathlib.Path, help="куда складывать треки")
ap.add_argument("--album", required=True, help="название книги для тегов")
ap.add_argument("--author", required=True, help="автор для тегов")
ap.add_argument("--speaker", default="eugene",
help="aidar, baya, kseniya, eugene, xenia")
ap.add_argument("--tempo", type=float, default=1.0,
help="растяжение времени: 0.87 = на 13%% медленнее, высота сохраняется")
ap.add_argument("--threads", type=int, default=8)
ap.add_argument("--skip", type=int, nargs="*", default=[],
help="номера разделов, которые не озвучивать (титулы, библиография)")
args = ap.parse_args()
torch.set_num_threads(args.threads)
model = torch.package.PackageImporter(str(args.model)).load_pickle("tts_models", "model")
model.to(torch.device('cpu'))
args.out.mkdir(parents=True, exist_ok=True)
kw = dict(speaker=args.speaker, sample_rate=SR, put_accent=True, put_yo=True,
put_stress_homo=True, put_yo_homo=True)
skip = set(args.skip)
todo = []
for f in sorted(args.texts.glob('chapter_*_translated*.json')):
n = int(re.search(r'chapter_(\d+)', f.name).group(1))
if n in skip:
continue
j = json.loads(f.read_text(encoding="utf-8"))
title = norm.chapter_title(n, j.get('title', '')) or ('Раздел %03d' % n)
todo.append((n, title, j['paragraphs']))
if not todo:
sys.exit("в %s нет глав" % args.texts)
print("глав к озвучке: %d (пропущено %d)" % (len(todo), len(skip)), flush=True)
started = time.time()
for idx, (n, title, paras) in enumerate(todo, 1):
dest = args.out / ('%03d - %s.mp3' % (n, safe(title)))
if dest.exists() and dest.stat().st_size:
print(" %03d %-24s уже готово" % (n, title), flush=True)
continue
t0, chunks, failed = time.time(), [], 0
intro = ssml(title + '.')
if intro:
chunks += [model.apply_tts(ssml_text=intro, **kw), torch.zeros(int(SR * 0.6))]
for p in paras:
src = norm.normalize(p)
if not src.strip():
continue
s = ssml(src)
try:
a = model.apply_tts(ssml_text=s, **kw) if s else None
except Exception:
try: # парсер SSML спотыкается на редких абзацах
a = model.apply_tts(text=src, **kw)
except Exception:
failed += 1
continue
if a is not None:
chunks += [a, torch.zeros(int(SR * 0.30))]
if not chunks:
print(" %03d %-24s пусто, пропуск" % (n, title), flush=True)
continue
full = torch.cat(chunks)
wav = args.out / ('.render_%03d.wav' % n)
with wave.open(str(wav), 'wb') as w:
w.setnchannels(1); w.setsampwidth(2); w.setframerate(SR)
w.writeframes((full.clamp(-1, 1) * 32767).to(torch.int16).numpy().tobytes())
cmd = ['ffmpeg', '-v', 'error', '-y', '-i', str(wav)]
if abs(args.tempo - 1.0) > 1e-6:
cmd += ['-filter:a', 'atempo=%.3f' % args.tempo]
cmd += ['-b:a', '64k', '-ac', '1',
'-metadata', 'title=%s' % title,
'-metadata', 'album=%s' % args.album,
'-metadata', 'artist=%s' % args.author,
'-metadata', 'album_artist=%s' % args.author,
'-metadata', 'track=%d/%d' % (idx, len(todo)),
'-metadata', 'genre=Audiobook', str(dest)]
subprocess.run(cmd, check=True)
wav.unlink()
raw, el = len(full) / SR, time.time() - t0
print(" %03d %-24s %5.1f мин (синтез %4.1f мин, %.1fx)%s"
% (n, title, raw / args.tempo / 60, el / 60, raw / el,
' СБОЕВ: %d' % failed if failed else ''), flush=True)
files = sorted(args.out.glob('*.mp3'))
total = sum(float(subprocess.run(
['ffprobe', '-v', 'error', '-show_entries', 'format=duration',
'-of', 'default=nw=1:nk=1', str(f)],
capture_output=True, text=True).stdout or 0) for f in files)
print("\nготово: %d треков, %.1f ч, %.0f МБ, заняло %.1f ч"
% (len(files), total / 3600,
sum(f.stat().st_size for f in files) / 1024 / 1024,
(time.time() - started) / 3600))
print("каталог:", args.out)
if __name__ == '__main__':
main()