""" NovaAi – TTS-Engine-Hub engines/kokoro.py Version: v0.1.0 Description: Kokoro TTS engine adapter: 82M parameter high-quality TTS model. Synthesizes 24kHz audio using Kokoro library, converts to OGG/MP3 via ffmpeg if needed. Supports 54 voices across 8 languages with GPU acceleration. Author: Claude Code (Anthropic) Date: 2025-12-05 """ import asyncio import subprocess import tempfile import os import shutil import logging from typing import Optional, List from .engine_base import TTSEngineBase from app.config import settings import ffmpeg logger = logging.getLogger(__name__) # Language code mapping for Kokoro KOKORO_LANG_CODES = { "kokoro-en-us": "a", # American English "kokoro-en-gb": "b", # British English "kokoro-fr": "fr", # French "kokoro-es": "es", # Spanish "kokoro-ja": "ja", # Japanese "kokoro-zh": "zh", # Chinese "kokoro-it": "it", # Italian "kokoro-pt": "pt", # Portuguese "kokoro-hi": "hi", # Hindi "kokoro-ko": "ko", # Korean } # Import voice metadata from .kokoro_voices import ALL_VOICES, get_voices_for_model, get_voice_info class KokoroEngine(TTSEngineBase): def __init__(self): self.kokoro_available = False self.pipeline = None self.current_lang = None self.ffmpeg_executable = shutil.which("ffmpeg") self.device = getattr(settings, "KOKORO_DEVICE", "cuda") self.timeout = getattr(settings, "KOKORO_TIMEOUT_SECONDS", 30) # Try to import and initialize Kokoro try: from kokoro import KPipeline self.KPipeline = KPipeline self.kokoro_available = True logger.info("Kokoro TTS library loaded successfully") except ImportError as e: logger.warning(f"Kokoro TTS library not available: {e}") self.kokoro_available = False def _get_pipeline(self, lang_code: str): """Get or create pipeline for specific language.""" if not self.kokoro_available: raise RuntimeError("Kokoro library not installed. Install with: pip install kokoro>=0.9.2") # Reuse pipeline if same language if self.pipeline is not None and self.current_lang == lang_code: return self.pipeline # Create new pipeline for language try: logger.info(f"Loading Kokoro pipeline for language code: {lang_code}") self.pipeline = self.KPipeline(lang_code=lang_code) self.current_lang = lang_code return self.pipeline except Exception as e: logger.error(f"Failed to load Kokoro pipeline: {e}") raise RuntimeError(f"Failed to load Kokoro pipeline for {lang_code}: {e}") def _run_ffmpeg_blocking(self, input_path: str, output_path: str): """ Wrapper for blocking ffmpeg call with error capture. Reused from Piper engine implementation. """ try: stdout, stderr = ( ffmpeg .input(input_path) .output(output_path) .run(overwrite_output=True, capture_stdout=True, capture_stderr=True) ) if stderr: logger.debug(f"FFmpeg output: {stderr.decode('utf-8', errors='replace')}") except ffmpeg.Error as e: stderr_output = e.stderr.decode('utf-8', errors='replace') if e.stderr else "No error output" logger.error(f"FFmpeg conversion failed: {input_path} -> {output_path}. Error: {stderr_output}") raise RuntimeError( f"FFmpeg conversion failed: {input_path} -> {output_path}. " f"Error: {stderr_output}" ) async def synthesize(self, text: str, speaker: str = None, model: str = None, fmt: str = "ogg") -> str: """ Synthesize speech from text using Kokoro TTS. Applies all bug fixes from Piper engine: - Timeout protection - Comprehensive temp file cleanup - FFmpeg error capture - Enhanced logging """ # Validation if not self.kokoro_available: raise RuntimeError("Kokoro library not installed. Install with: pip install kokoro>=0.9.2 soundfile") if not model: model = "kokoro-en-us" # Default to American English if model not in KOKORO_LANG_CODES: raise ValueError( f"Model '{model}' not supported. Available models: {list(KOKORO_LANG_CODES.keys())}" ) if not speaker: speaker = "af_bella" # Default voice if speaker not in ALL_VOICES: logger.warning( f"Voice '{speaker}' not in known voice list. Attempting anyway. " f"Known voices: {ALL_VOICES[:10]}..." ) # Get language code lang_code = KOKORO_LANG_CODES[model] # Track temp files for cleanup temp_files_to_cleanup = [] try: # Get pipeline for language pipeline = await asyncio.to_thread(self._get_pipeline, lang_code) # Create WAV temp file fd, output_wav_path = tempfile.mkstemp(suffix=".wav", prefix="kokoro_") os.close(fd) temp_files_to_cleanup.append(output_wav_path) # Log synthesis details text_preview = text[:100] + "..." if len(text) > 100 else text logger.debug(f"Kokoro synthesis: model={model}, voice={speaker}, text_len={len(text)}") logger.debug(f"Text preview: {text_preview}") # Generate audio with timeout try: audio_data = await asyncio.wait_for( asyncio.to_thread(self._synthesize_audio, pipeline, text, speaker), timeout=self.timeout ) except asyncio.TimeoutError: logger.error( f"Kokoro synthesis timed out after {self.timeout}s. " f"Model: {model}, Voice: {speaker}, Text length: {len(text)}" ) raise RuntimeError( f"Kokoro synthesis timed out after {self.timeout}s. " f"Text length: {len(text)} chars" ) # Save audio to WAV file import soundfile as sf await asyncio.to_thread(sf.write, output_wav_path, audio_data, 24000) # Verify output created if not os.path.exists(output_wav_path) or os.path.getsize(output_wav_path) == 0: raise RuntimeError("Kokoro synthesis failed: output file not created or empty") logger.info( f"Kokoro synthesis succeeded: {len(text)} chars -> " f"{os.path.getsize(output_wav_path)} bytes. Model: {model}, Voice: {speaker}" ) # Return WAV if requested fmt = (fmt or "ogg").lower() if fmt == "wav": temp_files_to_cleanup.remove(output_wav_path) return output_wav_path # FFmpeg conversion if not self.ffmpeg_executable: raise RuntimeError("ffmpeg not found, cannot convert audio format.") # Create converted file temp path fd_conv, output_other_path = tempfile.mkstemp(suffix=f'.{fmt}', prefix="kokoro_conv_") os.close(fd_conv) temp_files_to_cleanup.append(output_other_path) logger.debug(f"Converting WAV to {fmt}: {output_wav_path} -> {output_other_path}") # Convert with timeout try: await asyncio.wait_for( asyncio.to_thread(self._run_ffmpeg_blocking, output_wav_path, output_other_path), timeout=60 # FFmpeg timeout ) except asyncio.TimeoutError: logger.error( f"FFmpeg conversion timed out after 60s. " f"Input size: {os.path.getsize(output_wav_path)} bytes" ) raise RuntimeError( f"FFmpeg conversion timed out after 60s. " f"Input size: {os.path.getsize(output_wav_path)} bytes" ) # Verify conversion succeeded if not os.path.exists(output_other_path) or os.path.getsize(output_other_path) == 0: raise RuntimeError("FFmpeg conversion failed: output file not created or empty") logger.info( f"FFmpeg conversion succeeded: {os.path.getsize(output_wav_path)} bytes (WAV) -> " f"{os.path.getsize(output_other_path)} bytes ({fmt})" ) # Success! Remove converted file from cleanup (we're returning it) temp_files_to_cleanup.remove(output_other_path) return output_other_path finally: # Cleanup all temp files for temp_file in temp_files_to_cleanup: try: if os.path.exists(temp_file): os.remove(temp_file) logger.debug(f"Cleaned up temp file: {temp_file}") except Exception as e: logger.warning(f"Failed to cleanup temp file {temp_file}: {e}") def _synthesize_audio(self, pipeline, text: str, voice: str): """ Blocking synthesis function (runs in thread). Generates audio using Kokoro pipeline. """ import numpy as np # Generate audio using pipeline generator = pipeline(text, voice=voice) # Collect audio chunks audio_chunks = [] for gs, ps, audio in generator: audio_chunks.append(audio) # Concatenate all chunks if not audio_chunks: raise RuntimeError("Kokoro generated no audio chunks") full_audio = np.concatenate(audio_chunks) return full_audio def list_models(self) -> List[str]: """Return available Kokoro language models.""" return list(KOKORO_LANG_CODES.keys()) def list_voices(self, model: str = None) -> List[str]: """Return available Kokoro voices, optionally filtered by model/language.""" if model and model in KOKORO_LANG_CODES: # Return voices for specific language return sorted(get_voices_for_model(model)) else: # Return all voices return sorted(ALL_VOICES) def healthcheck(self): """Return health/status info for Kokoro engine.""" status = "ok" if self.kokoro_available else "not_available" details = { "status": status, "engine": "kokoro", "library_available": self.kokoro_available, "device": self.device if self.kokoro_available else None, } if not self.kokoro_available: details["error"] = "Kokoro library not installed. Install with: pip install kokoro>=0.9.2 soundfile" return details async def selftest(self): """Run self-test to verify Kokoro is working.""" if not self.kokoro_available: return { "selftest": False, "error": "Kokoro library not installed", "engine": "kokoro" } try: # Test synthesis with default model and voice test_text = "This is a Kokoro selftest." audio_file = await self.synthesize( test_text, speaker="af_bella", model="kokoro-en-us", fmt="wav" ) selftest_passed = os.path.exists(audio_file) and os.path.getsize(audio_file) > 0 if selftest_passed: os.remove(audio_file) return { "selftest": selftest_passed, "models": self.list_models(), "voices_count": len(self.list_voices()), "engine": "kokoro" } except Exception as e: return { "selftest": False, "error": str(e), "engine": "kokoro" } if __name__ == "__main__": async def main(): engine = KokoroEngine() print("Healthcheck:", engine.healthcheck()) print("Models:", engine.list_models()) print("Voices:", engine.list_voices()[:10], "...") print("Selftest:", await engine.selftest()) asyncio.run(main())