Files
asfr/tts/tts_silas.py
T

185 lines
5.8 KiB
Python
Raw Normal View History

#!/usr/bin/env python3
"""
TTS generator for ASFR stories with dual-voice effect.
Uses Silero TTS v5 (Russian) with the 'aidar' voice for narration,
applies a robotic effect (pitch shift + ring modulation + bitcrush)
to system error messages (lines starting with ОШИБКА:/ПРЕДУПРЕЖДЕНИЕ:).
Usage:
uv run tts/tts_silas.py # generate Silas audiobook
uv run tts/tts_silas.py <file.md> # generate for any story
Output:
tts/output/<name>_full.wav — full audiobook
tts/output/<name>_demo.wav — first 4 segments
Requirements (installed via uv):
torch, torchaudio, silero, silero-stress, soundfile, librosa
"""
import re
import sys
import torch
import soundfile as sf
import numpy as np
from pathlib import Path
# === Config ===
MODEL_ID = "v5_ru"
SPEAKER = "eugene" # male, more expressive intonation
USE_STRESS = True # silero-stress for correct word emphasis
SAMPLE_RATE = 48000
MAX_CHARS = 450 # silero works best with short segments
# === Load model ===
def load_model():
device = torch.device("cpu")
model, _ = torch.hub.load(
repo_or_dir="snakers4/silero-models",
model="silero_tts",
language="ru",
speaker=MODEL_ID,
)
model.to(device)
return model
def apply_stress(text):
"""Add stress marks for correct pronunciation using silero-stress."""
try:
from silero_stress import StressAccentor
accentor = StressAccentor()
return accentor.apply_stress(text)
except Exception:
return text
# === Robotic effect ===
def make_robotic(audio, sr, pitch_shift=-4, ring_freq=60, bit_depth=200):
"""Transform normal voice into robotic/cyborg voice."""
import librosa
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch_shift)
t = np.linspace(0, len(audio) / sr, len(audio))
ring = np.sin(2 * np.pi * ring_freq * t)
audio = audio * (0.75 + 0.25 * ring)
audio = np.round(audio * bit_depth) / bit_depth
audio = audio / np.max(np.abs(audio)) * 0.9
return audio
# === Split text into segments ===
def parse_segments(text):
"""Split text into (type, text) pairs. System errors get 'robot' type."""
text = re.sub(r"^---\n.*?---\n", "", text, flags=re.DOTALL)
text = text.strip()
segments = []
paragraphs = text.split("\n\n")
for para in paragraphs:
para = para.strip()
if not para:
continue
lines = para.split("\n")
is_system = all(
line.strip().startswith(("ПРЕДУПРЕЖДЕНИЕ:", "ОШИБКА:"))
for line in lines if line.strip()
)
if is_system:
clean = "\n".join(line.strip().lstrip("—").strip() for line in lines if line.strip())
segments.append(("robot", clean))
else:
segments.append(("normal", para))
return segments
# === Split long text for silero ===
def split_long(text, max_chars=MAX_CHARS):
parts = []
remaining = text
while remaining:
if len(remaining) <= max_chars:
parts.append(remaining)
break
cut = remaining.rfind(". ", 200, max_chars)
if cut == -1:
cut = remaining.rfind(" ", 300, max_chars)
if cut == -1:
cut = max_chars
parts.append(remaining[:cut+1])
remaining = remaining[cut+1:].lstrip()
return parts
# === Main generation ===
def generate(input_path, output_dir):
input_path = Path(input_path)
output_dir = Path(output_dir)
output_dir.mkdir(parents=True, exist_ok=True)
name = input_path.stem
print(f"Loading Silero TTS v5 (voice: {SPEAKER})...")
model = load_model()
text = input_path.read_text(encoding="utf-8")
segments = parse_segments(text)
print(f"Found {len(segments)} segments:")
robot_count = sum(1 for t, _ in segments if t == "robot")
print(f" Normal: {len(segments) - robot_count}, Robot: {robot_count}")
all_audio = []
demo_audio = []
demo_count = 0
for i, (stype, stext) in enumerate(segments):
print(f" [{i+1}/{len(segments)}] {stype:6s} {stext[:50]}...")
parts = split_long(stext)
segment_audio = []
for part in parts:
stressed = apply_stress(part) if USE_STRESS else part
audio = model.apply_tts(text=stressed, speaker=SPEAKER, sample_rate=SAMPLE_RATE)
audio_np = audio.cpu().numpy() if isinstance(audio, torch.Tensor) else np.array(audio)
segment_audio.append(audio_np)
combined = np.concatenate(segment_audio)
if stype == "robot":
try:
import librosa
combined = make_robotic(combined, SAMPLE_RATE)
except ImportError:
pass
pause = np.zeros(int(SAMPLE_RATE * 0.3))
all_audio.append(combined)
all_audio.append(pause)
if demo_count < 4:
demo_audio.append(combined)
demo_audio.append(np.zeros(int(SAMPLE_RATE * 0.3)))
demo_count += 1
# Save full
final = np.concatenate(all_audio)
final = final / np.max(np.abs(final)) * 0.9
full_path = output_dir / f"{name}_full.wav"
sf.write(str(full_path), final, SAMPLE_RATE)
dur = len(final) / SAMPLE_RATE
print(f"\n✅ {full_path} ({dur:.0f}s = {dur/60:.1f}min)")
# Save demo
demo = np.concatenate(demo_audio)
demo = demo / np.max(np.abs(demo)) * 0.9
demo_path = output_dir / f"{name}_demo.wav"
sf.write(str(demo_path), demo, SAMPLE_RATE)
demo_dur = len(demo) / SAMPLE_RATE
print(f"✅ {demo_path} ({demo_dur:.0f}s demo)")
if __name__ == "__main__":
if len(sys.argv) > 1:
input_file = sys.argv[1]
else:
input_file = "Silas.md"
generate(input_file, "tts/output")