Add TTS audiobook generator with dual-voice effect
Generates Russian audiobooks from ASFR stories using Silero TTS v5. Narrator voice (eugene) for prose, robotic voice for system errors. - tts/tts_silas.py: generator script with stress accentor + robotic effect - tts/README.md: documentation - .gitignore: exclude output WAV files
This commit is contained in:
@@ -0,0 +1 @@
|
|||||||
|
tts/output/
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
# TTS — Audio Generation for ASFR Stories
|
||||||
|
|
||||||
|
## What
|
||||||
|
|
||||||
|
Generates audiobooks from story files using **Silero TTS v5** (Russian). Two voices:
|
||||||
|
- **Narrator** — male voice (`eugene`) for the story prose
|
||||||
|
- **Cyborg voice** — same voice with robotic effect (pitch shift + ring modulation + bitcrush) for system error messages (`ОШИБКА:`, `ПРЕДУПРЕЖДЕНИЕ:`)
|
||||||
|
|
||||||
|
## Setup
|
||||||
|
|
||||||
|
```bash
|
||||||
|
uv add torch torchaudio silero silero-stress soundfile librosa
|
||||||
|
```
|
||||||
|
|
||||||
|
## Usage
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Generate Silas audiobook
|
||||||
|
uv run tts/tts_silas.py Silas.md
|
||||||
|
|
||||||
|
# Generate any story
|
||||||
|
uv run tts/tts_silas.py Logan.md
|
||||||
|
```
|
||||||
|
|
||||||
|
Output goes to `tts/output/`:
|
||||||
|
- `<name>_full.wav` — complete audiobook
|
||||||
|
- `<name>_demo.wav` — first 4 segments (~1 min)
|
||||||
|
|
||||||
|
## Voice Options
|
||||||
|
|
||||||
|
Available Silero v5_ru voices:
|
||||||
|
- `aidar` — deep male (can be monotone)
|
||||||
|
- `eugene` — male, more expressive intonation ← default
|
||||||
|
- `baya` — female, soft
|
||||||
|
- `kseniya` — female, clear
|
||||||
|
- `xenia` — female, calm
|
||||||
|
|
||||||
|
Change `SPEAKER` in `tts_silas.py` to switch voices.
|
||||||
|
|
||||||
|
## Robotic Effect
|
||||||
|
|
||||||
|
System messages get these audio effects:
|
||||||
|
- Pitch shift: -4 semitones (deeper, more mechanical)
|
||||||
|
- Ring modulation: 50 Hz carrier (metallic timbre)
|
||||||
|
- Bitcrush: 200 levels (digital distortion)
|
||||||
|
|
||||||
|
Parameters in `make_robotic()` can be tuned.
|
||||||
|
|
||||||
|
## Known Issues
|
||||||
|
|
||||||
|
- Silero can misplace Russian word stress on complex sentences
|
||||||
|
- Long paragraphs are split at sentence boundaries
|
||||||
|
- Robotic effect uses librosa which has deprecation warnings (harmless)
|
||||||
@@ -0,0 +1,184 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
TTS generator for ASFR stories with dual-voice effect.
|
||||||
|
|
||||||
|
Uses Silero TTS v5 (Russian) with the 'aidar' voice for narration,
|
||||||
|
applies a robotic effect (pitch shift + ring modulation + bitcrush)
|
||||||
|
to system error messages (lines starting with ОШИБКА:/ПРЕДУПРЕЖДЕНИЕ:).
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
uv run tts/tts_silas.py # generate Silas audiobook
|
||||||
|
uv run tts/tts_silas.py <file.md> # generate for any story
|
||||||
|
|
||||||
|
Output:
|
||||||
|
tts/output/<name>_full.wav — full audiobook
|
||||||
|
tts/output/<name>_demo.wav — first 4 segments
|
||||||
|
|
||||||
|
Requirements (installed via uv):
|
||||||
|
torch, torchaudio, silero, silero-stress, soundfile, librosa
|
||||||
|
"""
|
||||||
|
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
import torch
|
||||||
|
import soundfile as sf
|
||||||
|
import numpy as np
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
# === Config ===
|
||||||
|
MODEL_ID = "v5_ru"
|
||||||
|
SPEAKER = "eugene" # male, more expressive intonation
|
||||||
|
USE_STRESS = True # silero-stress for correct word emphasis
|
||||||
|
SAMPLE_RATE = 48000
|
||||||
|
MAX_CHARS = 450 # silero works best with short segments
|
||||||
|
|
||||||
|
# === Load model ===
|
||||||
|
def load_model():
|
||||||
|
device = torch.device("cpu")
|
||||||
|
model, _ = torch.hub.load(
|
||||||
|
repo_or_dir="snakers4/silero-models",
|
||||||
|
model="silero_tts",
|
||||||
|
language="ru",
|
||||||
|
speaker=MODEL_ID,
|
||||||
|
)
|
||||||
|
model.to(device)
|
||||||
|
return model
|
||||||
|
|
||||||
|
def apply_stress(text):
|
||||||
|
"""Add stress marks for correct pronunciation using silero-stress."""
|
||||||
|
try:
|
||||||
|
from silero_stress import StressAccentor
|
||||||
|
accentor = StressAccentor()
|
||||||
|
return accentor.apply_stress(text)
|
||||||
|
except Exception:
|
||||||
|
return text
|
||||||
|
|
||||||
|
# === Robotic effect ===
|
||||||
|
def make_robotic(audio, sr, pitch_shift=-4, ring_freq=60, bit_depth=200):
|
||||||
|
"""Transform normal voice into robotic/cyborg voice."""
|
||||||
|
import librosa
|
||||||
|
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch_shift)
|
||||||
|
t = np.linspace(0, len(audio) / sr, len(audio))
|
||||||
|
ring = np.sin(2 * np.pi * ring_freq * t)
|
||||||
|
audio = audio * (0.75 + 0.25 * ring)
|
||||||
|
audio = np.round(audio * bit_depth) / bit_depth
|
||||||
|
audio = audio / np.max(np.abs(audio)) * 0.9
|
||||||
|
return audio
|
||||||
|
|
||||||
|
# === Split text into segments ===
|
||||||
|
def parse_segments(text):
|
||||||
|
"""Split text into (type, text) pairs. System errors get 'robot' type."""
|
||||||
|
text = re.sub(r"^---\n.*?---\n", "", text, flags=re.DOTALL)
|
||||||
|
text = text.strip()
|
||||||
|
|
||||||
|
segments = []
|
||||||
|
paragraphs = text.split("\n\n")
|
||||||
|
|
||||||
|
for para in paragraphs:
|
||||||
|
para = para.strip()
|
||||||
|
if not para:
|
||||||
|
continue
|
||||||
|
lines = para.split("\n")
|
||||||
|
is_system = all(
|
||||||
|
line.strip().startswith(("ПРЕДУПРЕЖДЕНИЕ:", "ОШИБКА:"))
|
||||||
|
for line in lines if line.strip()
|
||||||
|
)
|
||||||
|
if is_system:
|
||||||
|
clean = "\n".join(line.strip().lstrip("—").strip() for line in lines if line.strip())
|
||||||
|
segments.append(("robot", clean))
|
||||||
|
else:
|
||||||
|
segments.append(("normal", para))
|
||||||
|
|
||||||
|
return segments
|
||||||
|
|
||||||
|
# === Split long text for silero ===
|
||||||
|
def split_long(text, max_chars=MAX_CHARS):
|
||||||
|
parts = []
|
||||||
|
remaining = text
|
||||||
|
while remaining:
|
||||||
|
if len(remaining) <= max_chars:
|
||||||
|
parts.append(remaining)
|
||||||
|
break
|
||||||
|
cut = remaining.rfind(". ", 200, max_chars)
|
||||||
|
if cut == -1:
|
||||||
|
cut = remaining.rfind(" ", 300, max_chars)
|
||||||
|
if cut == -1:
|
||||||
|
cut = max_chars
|
||||||
|
parts.append(remaining[:cut+1])
|
||||||
|
remaining = remaining[cut+1:].lstrip()
|
||||||
|
return parts
|
||||||
|
|
||||||
|
# === Main generation ===
|
||||||
|
def generate(input_path, output_dir):
|
||||||
|
input_path = Path(input_path)
|
||||||
|
output_dir = Path(output_dir)
|
||||||
|
output_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
name = input_path.stem
|
||||||
|
|
||||||
|
print(f"Loading Silero TTS v5 (voice: {SPEAKER})...")
|
||||||
|
model = load_model()
|
||||||
|
|
||||||
|
text = input_path.read_text(encoding="utf-8")
|
||||||
|
segments = parse_segments(text)
|
||||||
|
|
||||||
|
print(f"Found {len(segments)} segments:")
|
||||||
|
robot_count = sum(1 for t, _ in segments if t == "robot")
|
||||||
|
print(f" Normal: {len(segments) - robot_count}, Robot: {robot_count}")
|
||||||
|
|
||||||
|
all_audio = []
|
||||||
|
demo_audio = []
|
||||||
|
demo_count = 0
|
||||||
|
|
||||||
|
for i, (stype, stext) in enumerate(segments):
|
||||||
|
print(f" [{i+1}/{len(segments)}] {stype:6s} {stext[:50]}...")
|
||||||
|
|
||||||
|
parts = split_long(stext)
|
||||||
|
segment_audio = []
|
||||||
|
for part in parts:
|
||||||
|
stressed = apply_stress(part) if USE_STRESS else part
|
||||||
|
audio = model.apply_tts(text=stressed, speaker=SPEAKER, sample_rate=SAMPLE_RATE)
|
||||||
|
audio_np = audio.cpu().numpy() if isinstance(audio, torch.Tensor) else np.array(audio)
|
||||||
|
segment_audio.append(audio_np)
|
||||||
|
|
||||||
|
combined = np.concatenate(segment_audio)
|
||||||
|
|
||||||
|
if stype == "robot":
|
||||||
|
try:
|
||||||
|
import librosa
|
||||||
|
combined = make_robotic(combined, SAMPLE_RATE)
|
||||||
|
except ImportError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
pause = np.zeros(int(SAMPLE_RATE * 0.3))
|
||||||
|
all_audio.append(combined)
|
||||||
|
all_audio.append(pause)
|
||||||
|
|
||||||
|
if demo_count < 4:
|
||||||
|
demo_audio.append(combined)
|
||||||
|
demo_audio.append(np.zeros(int(SAMPLE_RATE * 0.3)))
|
||||||
|
demo_count += 1
|
||||||
|
|
||||||
|
# Save full
|
||||||
|
final = np.concatenate(all_audio)
|
||||||
|
final = final / np.max(np.abs(final)) * 0.9
|
||||||
|
full_path = output_dir / f"{name}_full.wav"
|
||||||
|
sf.write(str(full_path), final, SAMPLE_RATE)
|
||||||
|
dur = len(final) / SAMPLE_RATE
|
||||||
|
print(f"\n✅ {full_path} ({dur:.0f}s = {dur/60:.1f}min)")
|
||||||
|
|
||||||
|
# Save demo
|
||||||
|
demo = np.concatenate(demo_audio)
|
||||||
|
demo = demo / np.max(np.abs(demo)) * 0.9
|
||||||
|
demo_path = output_dir / f"{name}_demo.wav"
|
||||||
|
sf.write(str(demo_path), demo, SAMPLE_RATE)
|
||||||
|
demo_dur = len(demo) / SAMPLE_RATE
|
||||||
|
print(f"✅ {demo_path} ({demo_dur:.0f}s demo)")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
if len(sys.argv) > 1:
|
||||||
|
input_file = sys.argv[1]
|
||||||
|
else:
|
||||||
|
input_file = "Silas.md"
|
||||||
|
|
||||||
|
generate(input_file, "tts/output")
|
||||||
Reference in New Issue
Block a user