Generates Russian audiobooks from ASFR stories using Silero TTS v5. Narrator voice (eugene) for prose, robotic voice for system errors. - tts/tts_silas.py: generator script with stress accentor + robotic effect - tts/README.md: documentation - .gitignore: exclude output WAV files
185 lines
5.8 KiB
Python
185 lines
5.8 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
TTS generator for ASFR stories with dual-voice effect.
|
|
|
|
Uses Silero TTS v5 (Russian) with the 'aidar' voice for narration,
|
|
applies a robotic effect (pitch shift + ring modulation + bitcrush)
|
|
to system error messages (lines starting with ОШИБКА:/ПРЕДУПРЕЖДЕНИЕ:).
|
|
|
|
Usage:
|
|
uv run tts/tts_silas.py # generate Silas audiobook
|
|
uv run tts/tts_silas.py <file.md> # generate for any story
|
|
|
|
Output:
|
|
tts/output/<name>_full.wav — full audiobook
|
|
tts/output/<name>_demo.wav — first 4 segments
|
|
|
|
Requirements (installed via uv):
|
|
torch, torchaudio, silero, silero-stress, soundfile, librosa
|
|
"""
|
|
|
|
import re
|
|
import sys
|
|
import torch
|
|
import soundfile as sf
|
|
import numpy as np
|
|
from pathlib import Path
|
|
|
|
# === Config ===
|
|
MODEL_ID = "v5_ru"
|
|
SPEAKER = "eugene" # male, more expressive intonation
|
|
USE_STRESS = True # silero-stress for correct word emphasis
|
|
SAMPLE_RATE = 48000
|
|
MAX_CHARS = 450 # silero works best with short segments
|
|
|
|
# === Load model ===
|
|
def load_model():
|
|
device = torch.device("cpu")
|
|
model, _ = torch.hub.load(
|
|
repo_or_dir="snakers4/silero-models",
|
|
model="silero_tts",
|
|
language="ru",
|
|
speaker=MODEL_ID,
|
|
)
|
|
model.to(device)
|
|
return model
|
|
|
|
def apply_stress(text):
|
|
"""Add stress marks for correct pronunciation using silero-stress."""
|
|
try:
|
|
from silero_stress import StressAccentor
|
|
accentor = StressAccentor()
|
|
return accentor.apply_stress(text)
|
|
except Exception:
|
|
return text
|
|
|
|
# === Robotic effect ===
|
|
def make_robotic(audio, sr, pitch_shift=-4, ring_freq=60, bit_depth=200):
|
|
"""Transform normal voice into robotic/cyborg voice."""
|
|
import librosa
|
|
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch_shift)
|
|
t = np.linspace(0, len(audio) / sr, len(audio))
|
|
ring = np.sin(2 * np.pi * ring_freq * t)
|
|
audio = audio * (0.75 + 0.25 * ring)
|
|
audio = np.round(audio * bit_depth) / bit_depth
|
|
audio = audio / np.max(np.abs(audio)) * 0.9
|
|
return audio
|
|
|
|
# === Split text into segments ===
|
|
def parse_segments(text):
|
|
"""Split text into (type, text) pairs. System errors get 'robot' type."""
|
|
text = re.sub(r"^---\n.*?---\n", "", text, flags=re.DOTALL)
|
|
text = text.strip()
|
|
|
|
segments = []
|
|
paragraphs = text.split("\n\n")
|
|
|
|
for para in paragraphs:
|
|
para = para.strip()
|
|
if not para:
|
|
continue
|
|
lines = para.split("\n")
|
|
is_system = all(
|
|
line.strip().startswith(("ПРЕДУПРЕЖДЕНИЕ:", "ОШИБКА:"))
|
|
for line in lines if line.strip()
|
|
)
|
|
if is_system:
|
|
clean = "\n".join(line.strip().lstrip("—").strip() for line in lines if line.strip())
|
|
segments.append(("robot", clean))
|
|
else:
|
|
segments.append(("normal", para))
|
|
|
|
return segments
|
|
|
|
# === Split long text for silero ===
|
|
def split_long(text, max_chars=MAX_CHARS):
|
|
parts = []
|
|
remaining = text
|
|
while remaining:
|
|
if len(remaining) <= max_chars:
|
|
parts.append(remaining)
|
|
break
|
|
cut = remaining.rfind(". ", 200, max_chars)
|
|
if cut == -1:
|
|
cut = remaining.rfind(" ", 300, max_chars)
|
|
if cut == -1:
|
|
cut = max_chars
|
|
parts.append(remaining[:cut+1])
|
|
remaining = remaining[cut+1:].lstrip()
|
|
return parts
|
|
|
|
# === Main generation ===
|
|
def generate(input_path, output_dir):
|
|
input_path = Path(input_path)
|
|
output_dir = Path(output_dir)
|
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
name = input_path.stem
|
|
|
|
print(f"Loading Silero TTS v5 (voice: {SPEAKER})...")
|
|
model = load_model()
|
|
|
|
text = input_path.read_text(encoding="utf-8")
|
|
segments = parse_segments(text)
|
|
|
|
print(f"Found {len(segments)} segments:")
|
|
robot_count = sum(1 for t, _ in segments if t == "robot")
|
|
print(f" Normal: {len(segments) - robot_count}, Robot: {robot_count}")
|
|
|
|
all_audio = []
|
|
demo_audio = []
|
|
demo_count = 0
|
|
|
|
for i, (stype, stext) in enumerate(segments):
|
|
print(f" [{i+1}/{len(segments)}] {stype:6s} {stext[:50]}...")
|
|
|
|
parts = split_long(stext)
|
|
segment_audio = []
|
|
for part in parts:
|
|
stressed = apply_stress(part) if USE_STRESS else part
|
|
audio = model.apply_tts(text=stressed, speaker=SPEAKER, sample_rate=SAMPLE_RATE)
|
|
audio_np = audio.cpu().numpy() if isinstance(audio, torch.Tensor) else np.array(audio)
|
|
segment_audio.append(audio_np)
|
|
|
|
combined = np.concatenate(segment_audio)
|
|
|
|
if stype == "robot":
|
|
try:
|
|
import librosa
|
|
combined = make_robotic(combined, SAMPLE_RATE)
|
|
except ImportError:
|
|
pass
|
|
|
|
pause = np.zeros(int(SAMPLE_RATE * 0.3))
|
|
all_audio.append(combined)
|
|
all_audio.append(pause)
|
|
|
|
if demo_count < 4:
|
|
demo_audio.append(combined)
|
|
demo_audio.append(np.zeros(int(SAMPLE_RATE * 0.3)))
|
|
demo_count += 1
|
|
|
|
# Save full
|
|
final = np.concatenate(all_audio)
|
|
final = final / np.max(np.abs(final)) * 0.9
|
|
full_path = output_dir / f"{name}_full.wav"
|
|
sf.write(str(full_path), final, SAMPLE_RATE)
|
|
dur = len(final) / SAMPLE_RATE
|
|
print(f"\n✅ {full_path} ({dur:.0f}s = {dur/60:.1f}min)")
|
|
|
|
# Save demo
|
|
demo = np.concatenate(demo_audio)
|
|
demo = demo / np.max(np.abs(demo)) * 0.9
|
|
demo_path = output_dir / f"{name}_demo.wav"
|
|
sf.write(str(demo_path), demo, SAMPLE_RATE)
|
|
demo_dur = len(demo) / SAMPLE_RATE
|
|
print(f"✅ {demo_path} ({demo_dur:.0f}s demo)")
|
|
|
|
if __name__ == "__main__":
|
|
if len(sys.argv) > 1:
|
|
input_file = sys.argv[1]
|
|
else:
|
|
input_file = "Silas.md"
|
|
|
|
generate(input_file, "tts/output")
|