#!/usr/bin/env python3 """ TTS generator for ASFR stories with dual-voice effect. Uses Silero TTS v5 (Russian) with the 'aidar' voice for narration, applies a robotic effect (pitch shift + ring modulation + bitcrush) to system error messages (lines starting with ОШИБКА:/ПРЕДУПРЕЖДЕНИЕ:). Usage: uv run tts/tts_silas.py # generate Silas audiobook uv run tts/tts_silas.py # generate for any story Output: tts/output/_full.wav — full audiobook tts/output/_demo.wav — first 4 segments Requirements (installed via uv): torch, torchaudio, silero, silero-stress, soundfile, librosa """ import re import sys import torch import soundfile as sf import numpy as np from pathlib import Path # === Config === MODEL_ID = "v5_ru" SPEAKER = "eugene" # male, more expressive intonation USE_STRESS = True # silero-stress for correct word emphasis SAMPLE_RATE = 48000 MAX_CHARS = 450 # silero works best with short segments # === Load model === def load_model(): device = torch.device("cpu") model, _ = torch.hub.load( repo_or_dir="snakers4/silero-models", model="silero_tts", language="ru", speaker=MODEL_ID, ) model.to(device) return model def apply_stress(text): """Add stress marks for correct pronunciation using silero-stress.""" try: from silero_stress import StressAccentor accentor = StressAccentor() return accentor.apply_stress(text) except Exception: return text # === Robotic effect === def make_robotic(audio, sr, pitch_shift=-4, ring_freq=60, bit_depth=200): """Transform normal voice into robotic/cyborg voice.""" import librosa audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch_shift) t = np.linspace(0, len(audio) / sr, len(audio)) ring = np.sin(2 * np.pi * ring_freq * t) audio = audio * (0.75 + 0.25 * ring) audio = np.round(audio * bit_depth) / bit_depth audio = audio / np.max(np.abs(audio)) * 0.9 return audio # === Split text into segments === def parse_segments(text): """Split text into (type, text) pairs. System errors get 'robot' type.""" text = re.sub(r"^---\n.*?---\n", "", text, flags=re.DOTALL) text = text.strip() segments = [] paragraphs = text.split("\n\n") for para in paragraphs: para = para.strip() if not para: continue lines = para.split("\n") is_system = all( line.strip().startswith(("ПРЕДУПРЕЖДЕНИЕ:", "ОШИБКА:")) for line in lines if line.strip() ) if is_system: clean = "\n".join(line.strip().lstrip("—").strip() for line in lines if line.strip()) segments.append(("robot", clean)) else: segments.append(("normal", para)) return segments # === Split long text for silero === def split_long(text, max_chars=MAX_CHARS): parts = [] remaining = text while remaining: if len(remaining) <= max_chars: parts.append(remaining) break cut = remaining.rfind(". ", 200, max_chars) if cut == -1: cut = remaining.rfind(" ", 300, max_chars) if cut == -1: cut = max_chars parts.append(remaining[:cut+1]) remaining = remaining[cut+1:].lstrip() return parts # === Main generation === def generate(input_path, output_dir): input_path = Path(input_path) output_dir = Path(output_dir) output_dir.mkdir(parents=True, exist_ok=True) name = input_path.stem print(f"Loading Silero TTS v5 (voice: {SPEAKER})...") model = load_model() text = input_path.read_text(encoding="utf-8") segments = parse_segments(text) print(f"Found {len(segments)} segments:") robot_count = sum(1 for t, _ in segments if t == "robot") print(f" Normal: {len(segments) - robot_count}, Robot: {robot_count}") all_audio = [] demo_audio = [] demo_count = 0 for i, (stype, stext) in enumerate(segments): print(f" [{i+1}/{len(segments)}] {stype:6s} {stext[:50]}...") parts = split_long(stext) segment_audio = [] for part in parts: stressed = apply_stress(part) if USE_STRESS else part audio = model.apply_tts(text=stressed, speaker=SPEAKER, sample_rate=SAMPLE_RATE) audio_np = audio.cpu().numpy() if isinstance(audio, torch.Tensor) else np.array(audio) segment_audio.append(audio_np) combined = np.concatenate(segment_audio) if stype == "robot": try: import librosa combined = make_robotic(combined, SAMPLE_RATE) except ImportError: pass pause = np.zeros(int(SAMPLE_RATE * 0.3)) all_audio.append(combined) all_audio.append(pause) if demo_count < 4: demo_audio.append(combined) demo_audio.append(np.zeros(int(SAMPLE_RATE * 0.3))) demo_count += 1 # Save full final = np.concatenate(all_audio) final = final / np.max(np.abs(final)) * 0.9 full_path = output_dir / f"{name}_full.wav" sf.write(str(full_path), final, SAMPLE_RATE) dur = len(final) / SAMPLE_RATE print(f"\n✅ {full_path} ({dur:.0f}s = {dur/60:.1f}min)") # Save demo demo = np.concatenate(demo_audio) demo = demo / np.max(np.abs(demo)) * 0.9 demo_path = output_dir / f"{name}_demo.wav" sf.write(str(demo_path), demo, SAMPLE_RATE) demo_dur = len(demo) / SAMPLE_RATE print(f"✅ {demo_path} ({demo_dur:.0f}s demo)") if __name__ == "__main__": if len(sys.argv) > 1: input_file = sys.argv[1] else: input_file = "Silas.md" generate(input_file, "tts/output")