From e53cbdcc485fd5edc720d1b6863c51ebab2a93c1 Mon Sep 17 00:00:00 2001 From: Jason Eximoelle Date: Thu, 8 Oct 2026 09:51:41 +1000 Subject: [PATCH] Add TTS audiobook generator with dual-voice effect Generates Russian audiobooks from ASFR stories using Silero TTS v5. Narrator voice (eugene) for prose, robotic voice for system errors. - tts/tts_silas.py: generator script with stress accentor + robotic effect - tts/README.md: documentation - .gitignore: exclude output WAV files --- .gitignore | 1 + tts/README.md | 53 ++++++++++++++ tts/tts_silas.py | 184 +++++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 238 insertions(+) create mode 100644 .gitignore create mode 100644 tts/README.md create mode 100644 tts/tts_silas.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..dbe71c3 --- /dev/null +++ b/.gitignore @@ -0,0 +1 @@ +tts/output/ diff --git a/tts/README.md b/tts/README.md new file mode 100644 index 0000000..c594339 --- /dev/null +++ b/tts/README.md @@ -0,0 +1,53 @@ +# TTS — Audio Generation for ASFR Stories + +## What + +Generates audiobooks from story files using **Silero TTS v5** (Russian). Two voices: +- **Narrator** — male voice (`eugene`) for the story prose +- **Cyborg voice** — same voice with robotic effect (pitch shift + ring modulation + bitcrush) for system error messages (`ОШИБКА:`, `ПРЕДУПРЕЖДЕНИЕ:`) + +## Setup + +```bash +uv add torch torchaudio silero silero-stress soundfile librosa +``` + +## Usage + +```bash +# Generate Silas audiobook +uv run tts/tts_silas.py Silas.md + +# Generate any story +uv run tts/tts_silas.py Logan.md +``` + +Output goes to `tts/output/`: +- `_full.wav` — complete audiobook +- `_demo.wav` — first 4 segments (~1 min) + +## Voice Options + +Available Silero v5_ru voices: +- `aidar` — deep male (can be monotone) +- `eugene` — male, more expressive intonation ← default +- `baya` — female, soft +- `kseniya` — female, clear +- `xenia` — female, calm + +Change `SPEAKER` in `tts_silas.py` to switch voices. + +## Robotic Effect + +System messages get these audio effects: +- Pitch shift: -4 semitones (deeper, more mechanical) +- Ring modulation: 50 Hz carrier (metallic timbre) +- Bitcrush: 200 levels (digital distortion) + +Parameters in `make_robotic()` can be tuned. + +## Known Issues + +- Silero can misplace Russian word stress on complex sentences +- Long paragraphs are split at sentence boundaries +- Robotic effect uses librosa which has deprecation warnings (harmless) diff --git a/tts/tts_silas.py b/tts/tts_silas.py new file mode 100644 index 0000000..3e1a46e --- /dev/null +++ b/tts/tts_silas.py @@ -0,0 +1,184 @@ +#!/usr/bin/env python3 +""" +TTS generator for ASFR stories with dual-voice effect. + +Uses Silero TTS v5 (Russian) with the 'aidar' voice for narration, +applies a robotic effect (pitch shift + ring modulation + bitcrush) +to system error messages (lines starting with ОШИБКА:/ПРЕДУПРЕЖДЕНИЕ:). + +Usage: + uv run tts/tts_silas.py # generate Silas audiobook + uv run tts/tts_silas.py # generate for any story + +Output: + tts/output/_full.wav — full audiobook + tts/output/_demo.wav — first 4 segments + +Requirements (installed via uv): + torch, torchaudio, silero, silero-stress, soundfile, librosa +""" + +import re +import sys +import torch +import soundfile as sf +import numpy as np +from pathlib import Path + +# === Config === +MODEL_ID = "v5_ru" +SPEAKER = "eugene" # male, more expressive intonation +USE_STRESS = True # silero-stress for correct word emphasis +SAMPLE_RATE = 48000 +MAX_CHARS = 450 # silero works best with short segments + +# === Load model === +def load_model(): + device = torch.device("cpu") + model, _ = torch.hub.load( + repo_or_dir="snakers4/silero-models", + model="silero_tts", + language="ru", + speaker=MODEL_ID, + ) + model.to(device) + return model + +def apply_stress(text): + """Add stress marks for correct pronunciation using silero-stress.""" + try: + from silero_stress import StressAccentor + accentor = StressAccentor() + return accentor.apply_stress(text) + except Exception: + return text + +# === Robotic effect === +def make_robotic(audio, sr, pitch_shift=-4, ring_freq=60, bit_depth=200): + """Transform normal voice into robotic/cyborg voice.""" + import librosa + audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch_shift) + t = np.linspace(0, len(audio) / sr, len(audio)) + ring = np.sin(2 * np.pi * ring_freq * t) + audio = audio * (0.75 + 0.25 * ring) + audio = np.round(audio * bit_depth) / bit_depth + audio = audio / np.max(np.abs(audio)) * 0.9 + return audio + +# === Split text into segments === +def parse_segments(text): + """Split text into (type, text) pairs. System errors get 'robot' type.""" + text = re.sub(r"^---\n.*?---\n", "", text, flags=re.DOTALL) + text = text.strip() + + segments = [] + paragraphs = text.split("\n\n") + + for para in paragraphs: + para = para.strip() + if not para: + continue + lines = para.split("\n") + is_system = all( + line.strip().startswith(("ПРЕДУПРЕЖДЕНИЕ:", "ОШИБКА:")) + for line in lines if line.strip() + ) + if is_system: + clean = "\n".join(line.strip().lstrip("—").strip() for line in lines if line.strip()) + segments.append(("robot", clean)) + else: + segments.append(("normal", para)) + + return segments + +# === Split long text for silero === +def split_long(text, max_chars=MAX_CHARS): + parts = [] + remaining = text + while remaining: + if len(remaining) <= max_chars: + parts.append(remaining) + break + cut = remaining.rfind(". ", 200, max_chars) + if cut == -1: + cut = remaining.rfind(" ", 300, max_chars) + if cut == -1: + cut = max_chars + parts.append(remaining[:cut+1]) + remaining = remaining[cut+1:].lstrip() + return parts + +# === Main generation === +def generate(input_path, output_dir): + input_path = Path(input_path) + output_dir = Path(output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + + name = input_path.stem + + print(f"Loading Silero TTS v5 (voice: {SPEAKER})...") + model = load_model() + + text = input_path.read_text(encoding="utf-8") + segments = parse_segments(text) + + print(f"Found {len(segments)} segments:") + robot_count = sum(1 for t, _ in segments if t == "robot") + print(f" Normal: {len(segments) - robot_count}, Robot: {robot_count}") + + all_audio = [] + demo_audio = [] + demo_count = 0 + + for i, (stype, stext) in enumerate(segments): + print(f" [{i+1}/{len(segments)}] {stype:6s} {stext[:50]}...") + + parts = split_long(stext) + segment_audio = [] + for part in parts: + stressed = apply_stress(part) if USE_STRESS else part + audio = model.apply_tts(text=stressed, speaker=SPEAKER, sample_rate=SAMPLE_RATE) + audio_np = audio.cpu().numpy() if isinstance(audio, torch.Tensor) else np.array(audio) + segment_audio.append(audio_np) + + combined = np.concatenate(segment_audio) + + if stype == "robot": + try: + import librosa + combined = make_robotic(combined, SAMPLE_RATE) + except ImportError: + pass + + pause = np.zeros(int(SAMPLE_RATE * 0.3)) + all_audio.append(combined) + all_audio.append(pause) + + if demo_count < 4: + demo_audio.append(combined) + demo_audio.append(np.zeros(int(SAMPLE_RATE * 0.3))) + demo_count += 1 + + # Save full + final = np.concatenate(all_audio) + final = final / np.max(np.abs(final)) * 0.9 + full_path = output_dir / f"{name}_full.wav" + sf.write(str(full_path), final, SAMPLE_RATE) + dur = len(final) / SAMPLE_RATE + print(f"\n✅ {full_path} ({dur:.0f}s = {dur/60:.1f}min)") + + # Save demo + demo = np.concatenate(demo_audio) + demo = demo / np.max(np.abs(demo)) * 0.9 + demo_path = output_dir / f"{name}_demo.wav" + sf.write(str(demo_path), demo, SAMPLE_RATE) + demo_dur = len(demo) / SAMPLE_RATE + print(f"✅ {demo_path} ({demo_dur:.0f}s demo)") + +if __name__ == "__main__": + if len(sys.argv) > 1: + input_file = sys.argv[1] + else: + input_file = "Silas.md" + + generate(input_file, "tts/output")