Add TTS audiobook generator with dual-voice effect
Generates Russian audiobooks from ASFR stories using Silero TTS v5. Narrator voice (eugene) for prose, robotic voice for system errors. - tts/tts_silas.py: generator script with stress accentor + robotic effect - tts/README.md: documentation - .gitignore: exclude output WAV files
This commit is contained in:
@@ -0,0 +1,184 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
TTS generator for ASFR stories with dual-voice effect.
|
||||
|
||||
Uses Silero TTS v5 (Russian) with the 'aidar' voice for narration,
|
||||
applies a robotic effect (pitch shift + ring modulation + bitcrush)
|
||||
to system error messages (lines starting with ОШИБКА:/ПРЕДУПРЕЖДЕНИЕ:).
|
||||
|
||||
Usage:
|
||||
uv run tts/tts_silas.py # generate Silas audiobook
|
||||
uv run tts/tts_silas.py <file.md> # generate for any story
|
||||
|
||||
Output:
|
||||
tts/output/<name>_full.wav — full audiobook
|
||||
tts/output/<name>_demo.wav — first 4 segments
|
||||
|
||||
Requirements (installed via uv):
|
||||
torch, torchaudio, silero, silero-stress, soundfile, librosa
|
||||
"""
|
||||
|
||||
import re
|
||||
import sys
|
||||
import torch
|
||||
import soundfile as sf
|
||||
import numpy as np
|
||||
from pathlib import Path
|
||||
|
||||
# === Config ===
|
||||
MODEL_ID = "v5_ru"
|
||||
SPEAKER = "eugene" # male, more expressive intonation
|
||||
USE_STRESS = True # silero-stress for correct word emphasis
|
||||
SAMPLE_RATE = 48000
|
||||
MAX_CHARS = 450 # silero works best with short segments
|
||||
|
||||
# === Load model ===
|
||||
def load_model():
|
||||
device = torch.device("cpu")
|
||||
model, _ = torch.hub.load(
|
||||
repo_or_dir="snakers4/silero-models",
|
||||
model="silero_tts",
|
||||
language="ru",
|
||||
speaker=MODEL_ID,
|
||||
)
|
||||
model.to(device)
|
||||
return model
|
||||
|
||||
def apply_stress(text):
|
||||
"""Add stress marks for correct pronunciation using silero-stress."""
|
||||
try:
|
||||
from silero_stress import StressAccentor
|
||||
accentor = StressAccentor()
|
||||
return accentor.apply_stress(text)
|
||||
except Exception:
|
||||
return text
|
||||
|
||||
# === Robotic effect ===
|
||||
def make_robotic(audio, sr, pitch_shift=-4, ring_freq=60, bit_depth=200):
|
||||
"""Transform normal voice into robotic/cyborg voice."""
|
||||
import librosa
|
||||
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch_shift)
|
||||
t = np.linspace(0, len(audio) / sr, len(audio))
|
||||
ring = np.sin(2 * np.pi * ring_freq * t)
|
||||
audio = audio * (0.75 + 0.25 * ring)
|
||||
audio = np.round(audio * bit_depth) / bit_depth
|
||||
audio = audio / np.max(np.abs(audio)) * 0.9
|
||||
return audio
|
||||
|
||||
# === Split text into segments ===
|
||||
def parse_segments(text):
|
||||
"""Split text into (type, text) pairs. System errors get 'robot' type."""
|
||||
text = re.sub(r"^---\n.*?---\n", "", text, flags=re.DOTALL)
|
||||
text = text.strip()
|
||||
|
||||
segments = []
|
||||
paragraphs = text.split("\n\n")
|
||||
|
||||
for para in paragraphs:
|
||||
para = para.strip()
|
||||
if not para:
|
||||
continue
|
||||
lines = para.split("\n")
|
||||
is_system = all(
|
||||
line.strip().startswith(("ПРЕДУПРЕЖДЕНИЕ:", "ОШИБКА:"))
|
||||
for line in lines if line.strip()
|
||||
)
|
||||
if is_system:
|
||||
clean = "\n".join(line.strip().lstrip("—").strip() for line in lines if line.strip())
|
||||
segments.append(("robot", clean))
|
||||
else:
|
||||
segments.append(("normal", para))
|
||||
|
||||
return segments
|
||||
|
||||
# === Split long text for silero ===
|
||||
def split_long(text, max_chars=MAX_CHARS):
|
||||
parts = []
|
||||
remaining = text
|
||||
while remaining:
|
||||
if len(remaining) <= max_chars:
|
||||
parts.append(remaining)
|
||||
break
|
||||
cut = remaining.rfind(". ", 200, max_chars)
|
||||
if cut == -1:
|
||||
cut = remaining.rfind(" ", 300, max_chars)
|
||||
if cut == -1:
|
||||
cut = max_chars
|
||||
parts.append(remaining[:cut+1])
|
||||
remaining = remaining[cut+1:].lstrip()
|
||||
return parts
|
||||
|
||||
# === Main generation ===
|
||||
def generate(input_path, output_dir):
|
||||
input_path = Path(input_path)
|
||||
output_dir = Path(output_dir)
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
name = input_path.stem
|
||||
|
||||
print(f"Loading Silero TTS v5 (voice: {SPEAKER})...")
|
||||
model = load_model()
|
||||
|
||||
text = input_path.read_text(encoding="utf-8")
|
||||
segments = parse_segments(text)
|
||||
|
||||
print(f"Found {len(segments)} segments:")
|
||||
robot_count = sum(1 for t, _ in segments if t == "robot")
|
||||
print(f" Normal: {len(segments) - robot_count}, Robot: {robot_count}")
|
||||
|
||||
all_audio = []
|
||||
demo_audio = []
|
||||
demo_count = 0
|
||||
|
||||
for i, (stype, stext) in enumerate(segments):
|
||||
print(f" [{i+1}/{len(segments)}] {stype:6s} {stext[:50]}...")
|
||||
|
||||
parts = split_long(stext)
|
||||
segment_audio = []
|
||||
for part in parts:
|
||||
stressed = apply_stress(part) if USE_STRESS else part
|
||||
audio = model.apply_tts(text=stressed, speaker=SPEAKER, sample_rate=SAMPLE_RATE)
|
||||
audio_np = audio.cpu().numpy() if isinstance(audio, torch.Tensor) else np.array(audio)
|
||||
segment_audio.append(audio_np)
|
||||
|
||||
combined = np.concatenate(segment_audio)
|
||||
|
||||
if stype == "robot":
|
||||
try:
|
||||
import librosa
|
||||
combined = make_robotic(combined, SAMPLE_RATE)
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
pause = np.zeros(int(SAMPLE_RATE * 0.3))
|
||||
all_audio.append(combined)
|
||||
all_audio.append(pause)
|
||||
|
||||
if demo_count < 4:
|
||||
demo_audio.append(combined)
|
||||
demo_audio.append(np.zeros(int(SAMPLE_RATE * 0.3)))
|
||||
demo_count += 1
|
||||
|
||||
# Save full
|
||||
final = np.concatenate(all_audio)
|
||||
final = final / np.max(np.abs(final)) * 0.9
|
||||
full_path = output_dir / f"{name}_full.wav"
|
||||
sf.write(str(full_path), final, SAMPLE_RATE)
|
||||
dur = len(final) / SAMPLE_RATE
|
||||
print(f"\n✅ {full_path} ({dur:.0f}s = {dur/60:.1f}min)")
|
||||
|
||||
# Save demo
|
||||
demo = np.concatenate(demo_audio)
|
||||
demo = demo / np.max(np.abs(demo)) * 0.9
|
||||
demo_path = output_dir / f"{name}_demo.wav"
|
||||
sf.write(str(demo_path), demo, SAMPLE_RATE)
|
||||
demo_dur = len(demo) / SAMPLE_RATE
|
||||
print(f"✅ {demo_path} ({demo_dur:.0f}s demo)")
|
||||
|
||||
if __name__ == "__main__":
|
||||
if len(sys.argv) > 1:
|
||||
input_file = sys.argv[1]
|
||||
else:
|
||||
input_file = "Silas.md"
|
||||
|
||||
generate(input_file, "tts/output")
|
||||
Reference in New Issue
Block a user