Add TTS audiobook generator with dual-voice effect

Generates Russian audiobooks from ASFR stories using Silero TTS v5.
Narrator voice (eugene) for prose, robotic voice for system errors.

- tts/tts_silas.py: generator script with stress accentor + robotic effect
- tts/README.md: documentation
- .gitignore: exclude output WAV files
This commit is contained in:
Jason Eximoelle
2026-10-08 09:51:41 +10:00
parent dec5929c8d
commit e53cbdcc48
3 changed files with 238 additions and 0 deletions
+1
View File
@@ -0,0 +1 @@
tts/output/
+53
View File
@@ -0,0 +1,53 @@
# TTS — Audio Generation for ASFR Stories
## What
Generates audiobooks from story files using **Silero TTS v5** (Russian). Two voices:
- **Narrator** — male voice (`eugene`) for the story prose
- **Cyborg voice** — same voice with robotic effect (pitch shift + ring modulation + bitcrush) for system error messages (`ОШИБКА:`, `ПРЕДУПРЕЖДЕНИЕ:`)
## Setup
```bash
uv add torch torchaudio silero silero-stress soundfile librosa
```
## Usage
```bash
# Generate Silas audiobook
uv run tts/tts_silas.py Silas.md
# Generate any story
uv run tts/tts_silas.py Logan.md
```
Output goes to `tts/output/`:
- `<name>_full.wav` — complete audiobook
- `<name>_demo.wav` — first 4 segments (~1 min)
## Voice Options
Available Silero v5_ru voices:
- `aidar` — deep male (can be monotone)
- `eugene` — male, more expressive intonation ← default
- `baya` — female, soft
- `kseniya` — female, clear
- `xenia` — female, calm
Change `SPEAKER` in `tts_silas.py` to switch voices.
## Robotic Effect
System messages get these audio effects:
- Pitch shift: -4 semitones (deeper, more mechanical)
- Ring modulation: 50 Hz carrier (metallic timbre)
- Bitcrush: 200 levels (digital distortion)
Parameters in `make_robotic()` can be tuned.
## Known Issues
- Silero can misplace Russian word stress on complex sentences
- Long paragraphs are split at sentence boundaries
- Robotic effect uses librosa which has deprecation warnings (harmless)
+184
View File
@@ -0,0 +1,184 @@
#!/usr/bin/env python3
"""
TTS generator for ASFR stories with dual-voice effect.
Uses Silero TTS v5 (Russian) with the 'aidar' voice for narration,
applies a robotic effect (pitch shift + ring modulation + bitcrush)
to system error messages (lines starting with ОШИБКА:/ПРЕДУПРЕЖДЕНИЕ:).
Usage:
uv run tts/tts_silas.py # generate Silas audiobook
uv run tts/tts_silas.py <file.md> # generate for any story
Output:
tts/output/<name>_full.wav — full audiobook
tts/output/<name>_demo.wav — first 4 segments
Requirements (installed via uv):
torch, torchaudio, silero, silero-stress, soundfile, librosa
"""
import re
import sys
import torch
import soundfile as sf
import numpy as np
from pathlib import Path
# === Config ===
MODEL_ID = "v5_ru"
SPEAKER = "eugene" # male, more expressive intonation
USE_STRESS = True # silero-stress for correct word emphasis
SAMPLE_RATE = 48000
MAX_CHARS = 450 # silero works best with short segments
# === Load model ===
def load_model():
device = torch.device("cpu")
model, _ = torch.hub.load(
repo_or_dir="snakers4/silero-models",
model="silero_tts",
language="ru",
speaker=MODEL_ID,
)
model.to(device)
return model
def apply_stress(text):
"""Add stress marks for correct pronunciation using silero-stress."""
try:
from silero_stress import StressAccentor
accentor = StressAccentor()
return accentor.apply_stress(text)
except Exception:
return text
# === Robotic effect ===
def make_robotic(audio, sr, pitch_shift=-4, ring_freq=60, bit_depth=200):
"""Transform normal voice into robotic/cyborg voice."""
import librosa
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch_shift)
t = np.linspace(0, len(audio) / sr, len(audio))
ring = np.sin(2 * np.pi * ring_freq * t)
audio = audio * (0.75 + 0.25 * ring)
audio = np.round(audio * bit_depth) / bit_depth
audio = audio / np.max(np.abs(audio)) * 0.9
return audio
# === Split text into segments ===
def parse_segments(text):
"""Split text into (type, text) pairs. System errors get 'robot' type."""
text = re.sub(r"^---\n.*?---\n", "", text, flags=re.DOTALL)
text = text.strip()
segments = []
paragraphs = text.split("\n\n")
for para in paragraphs:
para = para.strip()
if not para:
continue
lines = para.split("\n")
is_system = all(
line.strip().startswith(("ПРЕДУПРЕЖДЕНИЕ:", "ОШИБКА:"))
for line in lines if line.strip()
)
if is_system:
clean = "\n".join(line.strip().lstrip("—").strip() for line in lines if line.strip())
segments.append(("robot", clean))
else:
segments.append(("normal", para))
return segments
# === Split long text for silero ===
def split_long(text, max_chars=MAX_CHARS):
parts = []
remaining = text
while remaining:
if len(remaining) <= max_chars:
parts.append(remaining)
break
cut = remaining.rfind(". ", 200, max_chars)
if cut == -1:
cut = remaining.rfind(" ", 300, max_chars)
if cut == -1:
cut = max_chars
parts.append(remaining[:cut+1])
remaining = remaining[cut+1:].lstrip()
return parts
# === Main generation ===
def generate(input_path, output_dir):
input_path = Path(input_path)
output_dir = Path(output_dir)
output_dir.mkdir(parents=True, exist_ok=True)
name = input_path.stem
print(f"Loading Silero TTS v5 (voice: {SPEAKER})...")
model = load_model()
text = input_path.read_text(encoding="utf-8")
segments = parse_segments(text)
print(f"Found {len(segments)} segments:")
robot_count = sum(1 for t, _ in segments if t == "robot")
print(f" Normal: {len(segments) - robot_count}, Robot: {robot_count}")
all_audio = []
demo_audio = []
demo_count = 0
for i, (stype, stext) in enumerate(segments):
print(f" [{i+1}/{len(segments)}] {stype:6s} {stext[:50]}...")
parts = split_long(stext)
segment_audio = []
for part in parts:
stressed = apply_stress(part) if USE_STRESS else part
audio = model.apply_tts(text=stressed, speaker=SPEAKER, sample_rate=SAMPLE_RATE)
audio_np = audio.cpu().numpy() if isinstance(audio, torch.Tensor) else np.array(audio)
segment_audio.append(audio_np)
combined = np.concatenate(segment_audio)
if stype == "robot":
try:
import librosa
combined = make_robotic(combined, SAMPLE_RATE)
except ImportError:
pass
pause = np.zeros(int(SAMPLE_RATE * 0.3))
all_audio.append(combined)
all_audio.append(pause)
if demo_count < 4:
demo_audio.append(combined)
demo_audio.append(np.zeros(int(SAMPLE_RATE * 0.3)))
demo_count += 1
# Save full
final = np.concatenate(all_audio)
final = final / np.max(np.abs(final)) * 0.9
full_path = output_dir / f"{name}_full.wav"
sf.write(str(full_path), final, SAMPLE_RATE)
dur = len(final) / SAMPLE_RATE
print(f"\n✅ {full_path} ({dur:.0f}s = {dur/60:.1f}min)")
# Save demo
demo = np.concatenate(demo_audio)
demo = demo / np.max(np.abs(demo)) * 0.9
demo_path = output_dir / f"{name}_demo.wav"
sf.write(str(demo_path), demo, SAMPLE_RATE)
demo_dur = len(demo) / SAMPLE_RATE
print(f"✅ {demo_path} ({demo_dur:.0f}s demo)")
if __name__ == "__main__":
if len(sys.argv) > 1:
input_file = sys.argv[1]
else:
input_file = "Silas.md"
generate(input_file, "tts/output")