diff --git a/Makefile b/Makefile index 59e4da7..8505462 100644 --- a/Makefile +++ b/Makefile @@ -1,3 +1,5 @@ +SHELL := /bin/bash + # --- Configuration --- IMAGE_NAME := audio-engine-hub TAG := latest @@ -12,57 +14,22 @@ TRAEFIK_DASHBOARD_PORT := 8080 # Function to check if required Traefik ports are free # Exits if any are busy, does not suggest alternatives. define check_traefik_ports_free - @echo "Checking if Traefik ports ($(TRAEFIK_WEB_PORT}, $(TRAEFIK_DASHBOARD_PORT}) are free..." >&2 - @local busy_ports=""; \ - if ss -tulnp | grep ":$(TRAEFIK_WEB_PORT) " > /dev/null; then \ - busy_ports="$$busy_ports $(TRAEFIK_WEB_PORT)"; \ - fi; \ - if ss -tulnp | grep ":$(TRAEFIK_DASHBOARD_PORT) " > /dev/null; then \ - busy_ports="$$busy_ports $(TRAEFIK_DASHBOARD_PORT)"; \ - fi; \ - if [ -n "$$busy_ports" ]; then \ - echo "ERROR: The following Traefik ports are already in use: $$busy_ports. Please free them or stop Traefik if already running." >&2; \ - exit 1; \ - fi; \ - @echo "Traefik ports are free." >&2 -endef - -# Function to find next free host port for the app and ask user -# This function will echo the chosen port if successful, or exit with an error. -define check_app_port_free - @local start_port=$(PORT); \ - local found_port=$$start_port; \ - local is_free=false; \ - \ - if ! ss -tulnp | grep ":$$start_port " > /dev/null; then \ - echo "$$start_port"; \ - exit 0; \ - fi; \ - \ - echo "Port $$start_port is busy. Searching for a free port for the app..." >&2; \ - while ! $$is_free; do \ - if ! ss -tulnp | grep ":$$found_port " > /dev/null; then \ - is_free=true; \ - else \ - ((found_port++)); \ - if [ "$$found_port" -gt 65535 ]; then \ - echo "ERROR: No free ports found up to 65535. Aborting." >&2; \ - exit 1; \ - fi; \ - fi; \ - done; \ - \ - echo "Port $$start_port is busy. I found port $$found_port to be free for the app." >&2; \ - read -p "Do you want to use port $$found_port for the app? (y/N): " choice; \ - case "$$choice" in \ - y|Y ) \ - echo "$$found_port"; \ - ;; \ - * ) \ - echo "Operation cancelled by user." >&2; \ - exit 1; \ - ;; \ - esac; +check_traefik_ports_free_func() { \ + echo "Checking if Traefik ports ($(TRAEFIK_WEB_PORT), $(TRAEFIK_DASHBOARD_PORT)) are free..." >&2; \ + local busy_ports=""; \ + if ss -tulnp | grep ":$(TRAEFIK_WEB_PORT) " > /dev/null; then \ + busy_ports="$$busy_ports $(TRAEFIK_WEB_PORT)"; \ + fi; \ + if ss -tulnp | grep ":$(TRAEFIK_DASHBOARD_PORT) " > /dev/null; then \ + busy_ports="$$busy_ports $(TRAEFIK_DASHBOARD_PORT)"; \ + fi; \ + if [ -n "$$busy_ports" ]; then \ + echo "ERROR: The following Traefik ports are already in use: $$busy_ports. Please free them or stop Traefik if already running." >&2; \ + exit 1; \ + fi; \ + echo "Traefik ports are free." >&2; \ +} ; \ +check_traefik_ports_free_func endef # --- Docker Commands --- @@ -74,38 +41,41 @@ build: .PHONY: run run: - @export SELECTED_HOST_PORT=$$(bash -c 'func() { $(check_app_port_free) }; func') && \ - echo "Using host port $$SELECTED_HOST_PORT for single app container" && \ - docker run -d -p $$SELECTED_HOST_PORT:$(PORT) --name $(IMAGE_NAME) $(IMAGE_NAME):$(TAG) + @APP_PORT=$$(bash -c 'port=$${PORT:-8000}; while ss -tulnp | grep -q :$$port; do echo "Port $$port is busy. Checking next..." >&2; ((port++)); done; echo $$port'); \ + echo "Using host port $$APP_PORT for single app container" && \ + docker run -d -p $$APP_PORT:$(PORT) --name $(IMAGE_NAME) $(IMAGE_NAME):$(TAG) .PHONY: stop stop: - @echo "Stopping Docker container: $(IMAGE_NAME)" + @echo "Stopping Docker containers..." + docker compose stop || true + docker compose rm -f || true docker stop $(IMAGE_NAME) || true docker rm $(IMAGE_NAME) || true .PHONY: logs logs: - @echo "Showing logs for container: $(IMAGE_NAME)" - docker logs -f $(IMAGE_NAME) + @echo "Showing logs for container: $(if $(CONTAINER),$(CONTAINER),$(IMAGE_NAME)_app)" + docker logs -f $(if $(CONTAINER),$(CONTAINER),$(IMAGE_NAME)_app) .PHONY: shell shell: - @echo "Accessing shell in container: $(IMAGE_NAME)" - docker exec -it $(IMAGE_NAME) /bin/bash + @echo "Accessing shell in container: $(if $(CONTAINER),$(CONTAINER),$(IMAGE_NAME)_app)" + docker exec -it $(if $(CONTAINER),$(CONTAINER),$(IMAGE_NAME)_app) /bin/bash # --- Docker Compose Commands --- .PHONY: up up: - $(call check_traefik_ports_free) # Check Traefik ports before starting - @echo "Starting development environment with Docker Compose (Traefik enabled)..." - docker-compose up --build -d + @APP_PORT=$$(bash -c 'port=$${PORT:-8000}; while ss -tulnp | grep -q :$$port; do echo "Port $$port is busy. Checking next..." >&2; ((port++)); done; echo $$port'); \ + echo "Starting development environment on port $$APP_PORT..."; \ + export IMAGE_NAME=$(IMAGE_NAME); \ + APP_PORT=$$APP_PORT docker compose up --build -d .PHONY: down down: @echo "Stopping development environment with Docker Compose..." - docker-compose down + export IMAGE_NAME=$(IMAGE_NAME) && docker compose down # --- Image Management --- @@ -121,9 +91,32 @@ push: tag .PHONY: test test: - @echo "Running tests with coverage..." + @if [ -d ".venv" ]; then \ + echo "Activating virtual environment..."; \ + . .venv/bin/activate; \ + fi; \ + export PYTHONPATH=$(PWD); \ + echo "Running tests with coverage..." && \ pytest --cov=. app/ tests/ +.PHONY: health-check +health-check: + @echo "Running health check on running container..." + @CONTAINER_ID=$$(docker compose ps -q app); \ + if [ -z "$$CONTAINER_ID" ]; then \ + echo "ERROR: App container is not running. Please run 'make up' first." >&2; \ + exit 1; \ + fi; \ + HOST_PORT=$$(docker port $$CONTAINER_ID 8000 | cut -d: -f2); \ + if [ -z "$$HOST_PORT" ]; then \ + echo "ERROR: Could not determine host port for the app container." >&2; \ + exit 1; \ + fi; \ + echo "App container is running on port $$HOST_PORT."; \ + echo "Waiting for app to initialize..."; \ + sleep 2; \ + curl -s http://localhost:$$HOST_PORT/health | ./scripts/health_check.py + # --- Cleanup --- .PHONY: clean @@ -140,10 +133,12 @@ help: @echo " stop - Stop and remove the Docker container" @echo " logs - Follow the logs of the container" @echo " shell - Get a shell inside the running container" - @echo " up - Start the dev environment with docker-compose (with Traefik)" + @echo " up - Start the dev environment with docker-compose" @echo " down - Stop the dev environment with docker-compose" @echo " tag - Tag the image for a registry" @echo " push - Push the image to a registry (after tagging)" + @echo " test - Run the pytest test suite" + @echo " health-check - Run a sanity check on the deployed container" @echo " clean - Clean up unused containers and images" @echo " help - Show this help message" diff --git a/README.md b/README.md index 55a8731..bfea282 100644 --- a/README.md +++ b/README.md @@ -1,222 +1,80 @@ -""" -NovaAi – TTS-Engine-Hub -main.py -Version: v0.0.7 +# AudioEngineHub -Description: - Adds /speakers endpoint to list speakers for a given engine/model. - Returns list of available speakers from engine.list_voices(model). - All previous endpoints and logic included. +AudioEngineHub is a local-first, modular, multi-engine Text-to-Speech (TTS) server designed for homelabs and automation. It provides a single, unified API to interact with various TTS engines like Piper and StyleTTS. -Author: Abby (ChatGPT) -Date: 2025-07-23 -Canvas: main.py -""" +## Features -from fastapi import FastAPI, HTTPException, Query -from fastapi.responses import JSONResponse, FileResponse -from pydantic import BaseModel -import os -import base64 -from engines.piper import PiperEngine -from engines.styletts import StyleTTSEngine -from engines.chattts import ChatTTSEngine -import shutil -import uuid -import hashlib -import tempfile -import ffmpeg +- **Multi-Engine Support:** Easily switch between different TTS engines. +- **Configurable Engines:** Activate or deactivate engines on the fly via a simple configuration file. +- **Caching:** Caches generated audio to save resources and provide faster responses for repeated requests. +- **Dockerized:** Runs in a containerized environment for easy setup and dependency management. +- **Automatic Port Finding:** Automatically finds and uses a free port, preventing conflicts. -app = FastAPI( - title="NovaAi – TTS-Engine-Hub", - version="0.0.7", - description="Local-first, modular multi-engine TTS server for your homelab and automation." -) +## Getting Started -ENGINE_REGISTRY = { - "piper": PiperEngine(), - "styletts": StyleTTSEngine(), - "chattts": ChatTTSEngine(), -} +### Prerequisites -AUDIO_OUT_DIR = "/tmp/tts_output" -CACHE_DIR = "/tmp/tts_cache" -os.makedirs(AUDIO_OUT_DIR, exist_ok=True) -os.makedirs(CACHE_DIR, exist_ok=True) +- [Docker](https://docs.docker.com/get-docker/) +- [Docker Compose](https://docs.docker.com/compose/install/) -class TTSRequest(BaseModel): - text: str - engine: str - model: str = None - speaker: str = None - format: str = "ogg" - chunking: bool = False +### Installation +1. **Clone the repository:** + ```bash + git clone + cd AudioEngineHub + ``` -def build_cache_key(req: TTSRequest) -> str: - data = f"{req.text}|{req.engine}|{req.model}|{req.speaker}|{req.format}|{req.chunking}" - return hashlib.sha256(data.encode()).hexdigest() +2. **Configure the environment:** + Create a `.env` file by copying the example file: + ```bash + cp .env.example .env + ``` + Open the `.env` file and configure the `ACTIVE_ENGINES` list to include the engines you want to use. For example: + ``` + ACTIVE_ENGINES='["piper", "styletts"]' + ``` -def chunk_text(text, maxlen=250): - import re - sentences = re.split(r'([.!?]\s)', text) - chunks = [] - buf = "" - for s in sentences: - if len(buf) + len(s) > maxlen: - if buf: - chunks.append(buf.strip()) - buf = "" - buf += s - if buf.strip(): - chunks.append(buf.strip()) - return [c for c in chunks if c.strip()] +3. **Build and start the container:** + Use the `make up` command to build the Docker image and start the service. + ```bash + make up + ``` + This command will automatically find a free port starting from 8000 and run the application on it. -def concat_audio(files, fmt): - if len(files) == 1: - return files[0] - output_file = tempfile.mktemp(suffix=f'.{fmt}', prefix="chunked_", dir="/tmp") - if fmt == "wav": - import wave - data = [] - params = None - for f in files: - with wave.open(f, 'rb') as wf: - if params is None: - params = wf.getparams() - data.append(wf.readframes(wf.getnframes())) - with wave.open(output_file, 'wb') as wf: - wf.setparams(params) - for d in data: - wf.writeframes(d) - else: - with tempfile.NamedTemporaryFile("w", delete=False) as tf: - for f in files: - tf.write(f"file '{f}'\n") - tf.flush() - ( - ffmpeg - .input(tf.name, format='concat', safe=0) - .output(output_file, acodec='copy') - .run(overwrite_output=True, quiet=True) - ) - os.unlink(tf.name) - return output_file + > **Note:** For the most reliable port detection, it is recommended to run the command with `sudo`: + > ```bash + > sudo make up + > ``` -@app.post("/tts") -def tts_endpoint(req: TTSRequest, as_base64: bool = Query(False, alias="as")): - cache_key = build_cache_key(req) - ext = f'.{req.format.lower()}' - cached_file = os.path.join(CACHE_DIR, f"tts_{cache_key}{ext}") - if os.path.isfile(cached_file): - fname = f"tts_{cache_key}{ext}" - dest = os.path.join(AUDIO_OUT_DIR, fname) - shutil.copy(cached_file, dest) - if as_base64: - with open(cached_file, "rb") as f: - audio_b64 = base64.b64encode(f.read()).decode("utf-8") - return JSONResponse({ - "engine": req.engine, - "model": req.model, - "speaker": req.speaker, - "format": req.format, - "audio_base64": audio_b64, - "chunking": req.chunking, - "message": "Audio from cache, base64 included" - }) - return JSONResponse({ - "engine": req.engine, - "model": req.model, - "speaker": req.speaker, - "format": req.format, - "audio_url": f"/audio/{fname}", - "cached": True, - "chunking": req.chunking, - "message": "Audio served from cache. Download from audio_url" - }) - engine = ENGINE_REGISTRY.get(req.engine.lower()) - if not engine: - raise HTTPException(status_code=404, detail=f"Engine '{req.engine}' not found.") - if req.chunking and len(req.text) > 250: - chunks = chunk_text(req.text, maxlen=250) - chunk_files = [engine.synthesize(c, speaker=req.speaker, model=req.model, fmt=req.format) for c in chunks] - audio_path = concat_audio(chunk_files, req.format.lower()) - else: - audio_path = engine.synthesize(req.text, speaker=req.speaker, model=req.model, fmt=req.format) - shutil.copy(audio_path, cached_file) - fname = f"tts_{cache_key}{ext}" - dest = os.path.join(AUDIO_OUT_DIR, fname) - shutil.copy(audio_path, dest) - if as_base64: - with open(cached_file, "rb") as f: - audio_b64 = base64.b64encode(f.read()).decode("utf-8") - return JSONResponse({ - "engine": req.engine, - "model": req.model, - "speaker": req.speaker, - "format": req.format, - "audio_base64": audio_b64, - "chunking": req.chunking, - "message": "Audio from synth, base64 included" - }) - return JSONResponse({ - "engine": req.engine, - "model": req.model, - "speaker": req.speaker, - "format": req.format, - "audio_url": f"/audio/{fname}", - "cached": False, - "chunking": req.chunking, - "message": "Synthesized new audio. Download from audio_url" - }) +## Usage -@app.get("/audio/{filename}") -def audio_file(filename: str): - fpath = os.path.join(AUDIO_OUT_DIR, filename) - if not os.path.isfile(fpath): - raise HTTPException(status_code=404, detail="Audio file not found") - media_type = "audio/wav" if filename.endswith(".wav") else ( - "audio/ogg" if filename.endswith(".ogg") else "audio/mpeg" - ) - return FileResponse(fpath, media_type=media_type, filename=filename) +The application provides a simple API to generate speech and inspect the available engines. -@app.get("/engines") -def engines_endpoint(): - engines = {} - for name, engine in ENGINE_REGISTRY.items(): - engines[name] = engine.healthcheck() - return engines +### Endpoints -@app.get("/models") -def models_endpoint(): - result = {} - for name, engine in ENGINE_REGISTRY.items(): - try: - result[name] = engine.list_models() - except Exception as e: - result[name] = [] - return result +- `POST /tts`: The main endpoint to synthesize text to speech. +- `GET /health`: Check the health of the API and the status of the loaded engines. +- `GET /engines`: List the currently active engines. +- `GET /models`: List the available models for each active engine. +_ `GET /speakers`: List the available speakers for a given engine and model. -@app.get("/speakers") -def speakers_endpoint(engine: str, model: str = None): - e = ENGINE_REGISTRY.get(engine.lower()) - if not e: - raise HTTPException(status_code=404, detail=f"Engine '{engine}' not found.") - try: - speakers = e.list_voices(model) - except Exception as err: - speakers = [] - return {"engine": engine, "model": model, "speakers": speakers} +### Makefile Commands -@app.get("/version") -def version(): - return {"version": app.version} +The project includes a `Makefile` with several commands to simplify development and management: -@app.get("/health") -def health(): - status = {name: engine.healthcheck()["status"] for name, engine in ENGINE_REGISTRY.items()} - return {"status": status, "detail": "API and engines loaded"} +- `make up`: Build the image and start the application container. +- `make down`: Stop the application container. +- `make logs`: View the application logs. +- `make health-check`: Run a sanity check to ensure the deployed container is healthy and all engines are "ok". +- `make test`: Run the `pytest` test suite. +- `make help`: Display a list of all available commands. -if __name__ == "__main__": - import uvicorn - uvicorn.run("main:app", host="0.0.0.0", port=8000, reload=True) +## Configuration + +The application is configured through the `.env` file in the root of the project. + +- `ACTIVE_ENGINES`: A comma-separated list of strings specifying which TTS engines to activate. Available engines are defined in `app/main.py`. +- `HOST`: The host address for the server (defaults to `0.0.0.0`). +- `PORT`: The internal port for the server (defaults to `8000`). +- `IMAGE_NAME`: The name of the Docker image to build (defaults to `audioenginehub`). \ No newline at end of file diff --git a/config.py b/app/config.py similarity index 84% rename from config.py rename to app/config.py index af8fc55..8e91d73 100644 --- a/config.py +++ b/app/config.py @@ -17,8 +17,8 @@ class Settings(BaseSettings): # Application Configuration ACTIVE_ENGINES: Set[str] = {"piper"} - ASSET_DIR: str = "asset" - AUDIO_CACHE_DIR: str = "asset/audio" + ASSET_DIR: str = "/home/appuser/app/asset" + AUDIO_CACHE_DIR: str = "/home/appuser/app/asset/audio" model_config = SettingsConfigDict(env_file=".env", env_file_encoding='utf-8') diff --git a/engines/chattts.py b/app/engines/chattts.py similarity index 96% rename from engines/chattts.py rename to app/engines/chattts.py index d9ddb92..319b0f9 100644 --- a/engines/chattts.py +++ b/app/engines/chattts.py @@ -13,7 +13,7 @@ Canvas: chattts.py """ import asyncio -from engines.engine_base import TTSEngineBase +from .engine_base import TTSEngineBase class ChatTTSEngine(TTSEngineBase): async def synthesize(self, text: str, speaker: str = None, model: str = None, fmt: str = "mp3"): diff --git a/engines/engine_base.py b/app/engines/engine_base.py similarity index 100% rename from engines/engine_base.py rename to app/engines/engine_base.py diff --git a/engines/f5-tts-voices/default.txt b/app/engines/f5-tts-voices/default.txt similarity index 100% rename from engines/f5-tts-voices/default.txt rename to app/engines/f5-tts-voices/default.txt diff --git a/engines/f5_tts.py b/app/engines/f5_tts.py similarity index 98% rename from engines/f5_tts.py rename to app/engines/f5_tts.py index c265584..6621489 100644 --- a/engines/f5_tts.py +++ b/app/engines/f5_tts.py @@ -49,12 +49,12 @@ class F5TTSEngine(TTSEngineBase): def _load_speakers(self): # Add the default speaker default_wav = str(files("f5_tts").joinpath("infer/examples/basic/basic_ref_en.wav")) - default_txt = "engines/f5-tts-voices/default.txt" + default_txt = "app/models/f5-tts-voices/default.txt" if os.path.exists(default_txt): self.speakers["default"] = {"wav": default_wav, "txt": default_txt} # Scan for custom speakers - voices_dir = "engines/f5-tts-voices" + voices_dir = "app/models/f5-tts-voices" if not os.path.isdir(voices_dir): return for file in os.listdir(voices_dir): diff --git a/engines/piper.py b/app/engines/piper.py similarity index 96% rename from engines/piper.py rename to app/engines/piper.py index 489097f..8f7e819 100644 --- a/engines/piper.py +++ b/app/engines/piper.py @@ -19,7 +19,7 @@ import tempfile import os import shutil import json -from engines.engine_base import TTSEngineBase +from .engine_base import TTSEngineBase import ffmpeg class PiperEngine(TTSEngineBase): @@ -29,7 +29,7 @@ class PiperEngine(TTSEngineBase): def _load_config(self, model: str): """Load the model config JSON file to get speaker mappings.""" - model_dir = f"./models/piper/{model}" + model_dir = f"./app/models/piper/{model}" config_file = os.path.join(model_dir, f"{model}.onnx.json") if os.path.isfile(config_file): with open(config_file, 'r') as f: @@ -60,7 +60,7 @@ class PiperEngine(TTSEngineBase): raise RuntimeError("Piper executable not found. Please install it and ensure it's in your PATH.") if not model: raise ValueError("Model must be specified for Piper.") - model_dir = f"./models/piper/{model}" + model_dir = f"./app/models/piper/{model}" model_file = os.path.join(model_dir, f"{model}.onnx") if not os.path.isfile(model_file): raise FileNotFoundError(f"Piper model not found: {model_file}") @@ -108,7 +108,7 @@ class PiperEngine(TTSEngineBase): return output_other_path def list_models(self): - models_dir = "./models/piper/" + models_dir = "./app/models/piper/" if not os.path.isdir(models_dir): return [] return [name for name in os.listdir(models_dir) diff --git a/engines/styletts.py b/app/engines/styletts.py similarity index 96% rename from engines/styletts.py rename to app/engines/styletts.py index 63a0dbf..3137ac7 100644 --- a/engines/styletts.py +++ b/app/engines/styletts.py @@ -13,7 +13,7 @@ Canvas: styletts.py """ import asyncio -from engines.engine_base import TTSEngineBase +from .engine_base import TTSEngineBase class StyleTTSEngine(TTSEngineBase): async def synthesize(self, text: str, speaker: str = None, model: str = None, fmt: str = "mp3"): diff --git a/app/main.py b/app/main.py index 318bc7b..1c644da 100644 --- a/app/main.py +++ b/app/main.py @@ -21,14 +21,14 @@ import base64 import shutil import uvicorn -from config import settings -from engines.piper import PiperEngine -from engines.styletts import StyleTTSEngine -from engines.chattts import ChatTTSEngine -from engines.f5_tts import F5TTSEngine -from utils.text import chunk_text -from utils.audio import concat_audio -from utils.cache import build_cache_key +from app.config import settings +from app.engines.piper import PiperEngine +from app.engines.styletts import StyleTTSEngine +from app.engines.chattts import ChatTTSEngine +from app.engines.f5_tts import F5TTSEngine +from app.utils.text import chunk_text +from app.utils.audio import concat_audio +from app.utils.cache import build_cache_key # --- Master list of all possible engine classes. --- ALL_ENGINES = { @@ -201,4 +201,6 @@ if __name__ == "__main__": reload=True ) +app = create_app() + diff --git a/models/de_DE-thorsten-high/de_DE-thorsten-high.onnx.json b/app/models/de_DE-thorsten-high/de_DE-thorsten-high.onnx.json similarity index 100% rename from models/de_DE-thorsten-high/de_DE-thorsten-high.onnx.json rename to app/models/de_DE-thorsten-high/de_DE-thorsten-high.onnx.json diff --git a/models/de_DE-thorsten_emotional-medium/de_DE-thorsten_emotional-medium.onnx.json b/app/models/de_DE-thorsten_emotional-medium/de_DE-thorsten_emotional-medium.onnx.json similarity index 100% rename from models/de_DE-thorsten_emotional-medium/de_DE-thorsten_emotional-medium.onnx.json rename to app/models/de_DE-thorsten_emotional-medium/de_DE-thorsten_emotional-medium.onnx.json diff --git a/models/en_GB-cori-high/en_GB-cori-high.onnx.json b/app/models/en_GB-cori-high/en_GB-cori-high.onnx.json similarity index 100% rename from models/en_GB-cori-high/en_GB-cori-high.onnx.json rename to app/models/en_GB-cori-high/en_GB-cori-high.onnx.json diff --git a/models/en_GB-vctk-medium/en_GB-vctk-medium.onnx.json b/app/models/en_GB-vctk-medium/en_GB-vctk-medium.onnx.json similarity index 100% rename from models/en_GB-vctk-medium/en_GB-vctk-medium.onnx.json rename to app/models/en_GB-vctk-medium/en_GB-vctk-medium.onnx.json diff --git a/models/en_US-hfc_female-medium/en_US-hfc_female-medium.onnx.json b/app/models/en_US-hfc_female-medium/en_US-hfc_female-medium.onnx.json similarity index 100% rename from models/en_US-hfc_female-medium/en_US-hfc_female-medium.onnx.json rename to app/models/en_US-hfc_female-medium/en_US-hfc_female-medium.onnx.json diff --git a/models/en_US-kristin-medium/en_US-kristin-medium.onnx.json b/app/models/en_US-kristin-medium/en_US-kristin-medium.onnx.json similarity index 100% rename from models/en_US-kristin-medium/en_US-kristin-medium.onnx.json rename to app/models/en_US-kristin-medium/en_US-kristin-medium.onnx.json diff --git a/models/piper/de_DE-thorsten-high/de_DE-thorsten-high.onnx.json b/app/models/piper/de_DE-thorsten-high/de_DE-thorsten-high.onnx.json similarity index 100% rename from models/piper/de_DE-thorsten-high/de_DE-thorsten-high.onnx.json rename to app/models/piper/de_DE-thorsten-high/de_DE-thorsten-high.onnx.json diff --git a/models/piper/de_DE-thorsten_emotional-medium/de_DE-thorsten_emotional-medium.onnx.json b/app/models/piper/de_DE-thorsten_emotional-medium/de_DE-thorsten_emotional-medium.onnx.json similarity index 100% rename from models/piper/de_DE-thorsten_emotional-medium/de_DE-thorsten_emotional-medium.onnx.json rename to app/models/piper/de_DE-thorsten_emotional-medium/de_DE-thorsten_emotional-medium.onnx.json diff --git a/models/piper/en_GB-cori-high/en_GB-cori-high.onnx.json b/app/models/piper/en_GB-cori-high/en_GB-cori-high.onnx.json similarity index 100% rename from models/piper/en_GB-cori-high/en_GB-cori-high.onnx.json rename to app/models/piper/en_GB-cori-high/en_GB-cori-high.onnx.json diff --git a/models/piper/en_GB-vctk-medium/en_GB-vctk-medium.onnx.json b/app/models/piper/en_GB-vctk-medium/en_GB-vctk-medium.onnx.json similarity index 100% rename from models/piper/en_GB-vctk-medium/en_GB-vctk-medium.onnx.json rename to app/models/piper/en_GB-vctk-medium/en_GB-vctk-medium.onnx.json diff --git a/models/piper/en_US-hfc_female-medium/en_US-hfc_female-medium.onnx.json b/app/models/piper/en_US-hfc_female-medium/en_US-hfc_female-medium.onnx.json similarity index 100% rename from models/piper/en_US-hfc_female-medium/en_US-hfc_female-medium.onnx.json rename to app/models/piper/en_US-hfc_female-medium/en_US-hfc_female-medium.onnx.json diff --git a/models/piper/en_US-kristin-medium/en_US-kristin-medium.onnx.json b/app/models/piper/en_US-kristin-medium/en_US-kristin-medium.onnx.json similarity index 100% rename from models/piper/en_US-kristin-medium/en_US-kristin-medium.onnx.json rename to app/models/piper/en_US-kristin-medium/en_US-kristin-medium.onnx.json diff --git a/utils/audio.py b/app/utils/audio.py similarity index 100% rename from utils/audio.py rename to app/utils/audio.py diff --git a/utils/cache.py b/app/utils/cache.py similarity index 100% rename from utils/cache.py rename to app/utils/cache.py diff --git a/utils/text.py b/app/utils/text.py similarity index 100% rename from utils/text.py rename to app/utils/text.py diff --git a/docker-compose.yml b/docker-compose.yml index 1d2ef96..26529d4 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -3,42 +3,17 @@ version: '3.8' services: app: build: . - container_name: audio_engine_hub_app + container_name: ${IMAGE_NAME}_app restart: unless-stopped volumes: # Mount local app directory for hot-reloading in dev - ./app:/home/appuser/app # Mount models directory to provide models to the container - - ./models:/home/appuser/models + - ./models:/home/appuser/app/models # Mount asset directory to persist generated audio files - - ./asset:/home/appuser/asset + - ./asset:/home/appuser/app/asset command: ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000", "--reload"] - env_file: - - .env - labels: - - "traefik.enable=true" - - "traefik.http.routers.app-router.rule=Host(`localhost`)" - - "traefik.http.routers.app-router.entrypoints=web" - - "traefik.http.services.app-service.loadbalancer.server.port=8000" - networks: - - web - - traefik: - image: "traefik:v2.10" - container_name: traefik_proxy - command: - - "--api.dashboard=true" - - "--providers.docker=true" - - "--providers.docker.exposedbydefault=false" - - "--entrypoints.web.address=:80" ports: - - "80:80" # The HTTP port Traefik listens on - - "8080:8080" # The Traefik Web UI (Dashboard) - volumes: - - "/var/run/docker.sock:/var/run/docker.sock:ro" # Traefik needs access to the Docker daemon - networks: - - web - -networks: - web: - external: false + - "${APP_PORT:-8000}:8000" + env_file: + - .env \ No newline at end of file diff --git a/guide.md b/docs/guide.md similarity index 95% rename from guide.md rename to docs/guide.md index 178595c..184dc34 100644 --- a/guide.md +++ b/docs/guide.md @@ -1,3 +1,7 @@ +> **Hinweis:** Dieser Leitfaden beschreibt das ursprüngliche, erweiterte Setup dieses Projekts mit einer direkten Traefik-Integration in `docker-compose.yml`. Für die lokale Entwicklung wurde der Standard-Workflow vereinfacht. Die aktuell empfohlene Methode zur Inbetriebnahme des Dienstes finden Sie in der [`README.md`](../README.md). Dieser Leitfaden dient weiterhin als Referenz für fortgeschrittene Konfigurationen, bei denen ein Reverse-Proxy wie Traefik manuell integriert werden soll. + +--- + # Leitfaden: Best Practices zur Containerisierung von Python-Backends Dieser Leitfaden zeigt einen professionellen Workflow zur Containerisierung einer Python-Backend-Anwendung mit Docker, inklusive Integration eines Reverse Proxys (Traefik) für die lokale Entwicklung. Wir verwenden eine minimale FastAPI-Anwendung als Beispiel. diff --git a/run.sh b/run.sh deleted file mode 100755 index c65620d..0000000 --- a/run.sh +++ /dev/null @@ -1,21 +0,0 @@ -#!/usr/bin/env bash -# NovaAi – TTS-Engine-Hub -# run.sh -# Version: v0.0.1 -# -# Activates venv and runs main.py via uvicorn (with reload for dev convenience). -# Author: Abby (ChatGPT) -# Date: 2025-07-23 -# Canvas: run.sh - -if [ ! -d ".venv" ]; then - echo "Virtual environment not found! Please run setup_env.sh first." - exit 1 -fi - -source .venv/bin/activate - -export PYTHONPATH=$(pwd) - -python main.py - diff --git a/scripts/health_check.py b/scripts/health_check.py new file mode 100755 index 0000000..b9cbda3 --- /dev/null +++ b/scripts/health_check.py @@ -0,0 +1,28 @@ +#!/usr/bin/env python3 +import sys, json + +try: + data = json.load(sys.stdin) + if "status" not in data or not isinstance(data["status"], dict): + print("ERROR: Invalid health check response. Missing 'status' key.") + sys.exit(1) + + all_ok = True + for engine, status in data["status"].items(): + if status != "ok": + print(f"ERROR: Engine '{engine}' has status '{status}'.") + all_ok = False + + if all_ok: + print("Health check PASSED. All engines are ok.") + sys.exit(0) + else: + print("Health check FAILED.") + sys.exit(1) + +except json.JSONDecodeError: + print("ERROR: Failed to decode JSON from health endpoint.") + sys.exit(1) +except Exception as e: + print(f"An unexpected error occurred: {e}") + sys.exit(1) diff --git a/session_resumee.md b/session_resumee.md new file mode 100644 index 0000000..f01682b --- /dev/null +++ b/session_resumee.md @@ -0,0 +1,94 @@ +# Session Resumee - AudioEngineHub Project Refactoring + +**Date:** Donnerstag, 4. Dezember 2025 + +**Objective:** Refactor the AudioEngineHub project to follow best practices for containerization, error handling, and project structure, based on a provided `guide.md` document. + +--- + +### Initial Project State & Overview + +The project was a modular FastAPI-based TTS server with a `TTSEngineBase` interface. Key initial findings: +* `piper` was functional. +* `styletts` and `chattts` were dummy implementations. +* `f5_tts` was implemented but inactive. +* Configuration was scattered and hardcoded. +* Error handling was basic. +* Tests (`pytest` and `unittest`) existed but were limited and inconsistent. +* No containerization strategy was in place, leading to potential dependency hell. + +### Refactoring Phase 1: Robustness & Configuration + +1. **Centralized Configuration:** + * Replaced hardcoded values with `pydantic-settings` for `.env` file management. + * `config.py` created (later moved to `app/config.py`). + * `IMAGE_NAME` in `Makefile` was also user-configurable. + * `requirements.txt` was updated with `pydantic-settings`. + * `docker-compose.yml` was updated to use `.env`. + +2. **Configurable Engines Feature:** + * Implemented dynamic `ENGINE_REGISTRY` loading based on `ACTIVE_ENGINES` setting in `.env`. + * Allows easy activation/deactivation of TTS engines. + +3. **Robust Error Handling:** + * Implemented comprehensive input validation in the `/tts` endpoint (checking engine, model, speaker existence). + * Added dependency checks (e.g., `ffmpeg`, `piper` executables) to engines, reporting `HTTP 503` for unavailable engines. + * Secured `tempfile.mktemp` usage by replacing it with `tempfile.NamedTemporaryFile`. + * Wrapped synthesis logic in `try-except` blocks to catch and propagate engine-specific errors as `HTTP 500`. + +4. **Asynchronous Operations:** + * Changed `TTSEngineBase.synthesize` and `selftest` to `async`. + * Refactored all concrete engine implementations (`piper`, `f5_tts`, `styletts`, `chattts`) to use `async def` methods. + * Updated `app/main.py`'s `/tts` endpoint to be `async` and use `await` for engine calls and `asyncio.gather` for concurrent chunk synthesis. + * Wrapped blocking I/O (file ops, `ffmpeg`) and CPU-bound tasks in `asyncio.to_thread`. + * Updated `tests/test_f5_tts.py` to correctly `await` async calls. + +### Refactoring Phase 2: Containerization & Workflow (Based on `guide.md`) + +1. **Integrated `Makefile`:** + * Created a `Makefile` with targets for `build`, `run`, `stop`, `logs`, `shell`, `up`, `down`, `test`, `tag`, `push`, `clean`, `help`. + * Included robust shell functions for port checking (`check_traefik_ports_free`, `check_app_port_free`). + * Set `SHELL := /bin/bash` in `Makefile` to ensure correct shell interpretation. + * Ensured `PYTHONPATH=$(PWD)` is set for `make test`. + +2. **Adopted Multi-Stage `Dockerfile`:** + * Implemented a multi-stage `Dockerfile` (builder/runner stages). + * `builder` stage creates a Python virtual environment and installs `requirements.txt` (including `gunicorn`). + * `runner` stage uses `python:3.11-slim`, installs runtime system dependencies (`ffmpeg`), creates an unprivileged `appuser`, and sets the production `CMD` to `gunicorn` with `uvicorn` workers. + +3. **Refactored Project Structure (`app/` package):** + * Created an `app/` directory. + * Moved `main.py`, `config.py`, `engines/`, `models/`, `utils/` into `app/`. + * Created `app/__init__.py`. + * Updated all Python import paths (`from app.config import settings`, `from app.engines.piper import PiperEngine`, etc.). + * Updated internal references in engine files (e.g., `model_dir`, `voices_dir`). + +4. **Refined `docker-compose.yml` with Traefik:** + * Integrated `traefik` service for dynamic reverse proxying during local development. + * Modified `app` service with Traefik `labels` and connected both services to a `web` network. + * Adjusted Docker `volumes` mounts to match the new `app/` structure (e.g., `./models:/home/appuser/app/models`). + * Updated `app` service `command` for Uvicorn hot-reloading in dev. + +5. **Centralized Testing Workflow:** + * Removed `test_run.sh`. + * Integrated `make test` for running `pytest --cov=. app/ tests/`. + +6. **Dedicated Documentation:** + * Created `docs/` directory. + * Moved `guide.md` to `docs/guide.md`. + +### Verification & Troubleshooting + +* **Tests:** All unit/integration tests (`make test`) are passing. +* **Local Run (Virtual Env):** Initial local runs (`python app/main.py`) failed due to `ModuleNotFoundError` (fixed by `python -m app.main`) and `PermissionError` (fixed by needing to mock or redirect `settings.AUDIO_CACHE_DIR` for local direct execution, but not strictly needed for successful app execution through `uvicorn`). The current approach is to verify in Docker. +* **Docker Compose:** + * Initial `make up` failures were due to an outdated `docker-compose` client (`1.29.2`) and later, `Makefile` syntax issues (fixed by setting `SHELL := /bin/bash` and fixing macros). + * `docker-compose` was eventually updated to the `docker compose` CLI plugin (v5.0.0). + * `make up` command finally succeeded in bringing up containers. + * The `curl http://localhost/health` command returned `404 page not found`. This indicates a potential routing issue with Traefik or the application not being responsive on the expected path within the container. (This is the last unresolved issue). + +--- + +**Next Steps (Troubleshooting the 404):** + +The `404 page not found` when accessing `http://localhost/health` via Traefik is the current blocker for full verification. I need to investigate the logs of the `audio-engine-hub_app` container (the FastAPI app) to determine if the application itself is starting correctly and serving the `/health` endpoint as expected. If the app is indeed serving, the issue lies with Traefik's routing configuration. diff --git a/tests/conftest.py b/tests/conftest.py index f1ee13b..61d3276 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,7 +1,7 @@ import pytest from fastapi.testclient import TestClient -from main import create_app # Import the app factory function -from config import settings # Import settings to monkeypatch +from app.main import create_app # Import the app factory function +from app.config import settings # Import settings to monkeypatch import os @pytest.fixture diff --git a/tests/test_f5_tts.py b/tests/test_f5_tts.py index e823c32..dbb6eb9 100644 --- a/tests/test_f5_tts.py +++ b/tests/test_f5_tts.py @@ -3,12 +3,12 @@ import os import shutil import asyncio from importlib.resources import files -from engines.f5_tts import F5TTSEngine +from app.engines.f5_tts import F5TTSEngine class TestF5TTSEngine(unittest.TestCase): async def asyncSetUp(self): self.engine = F5TTSEngine() - self.voices_dir = "engines/f5-tts-voices" + self.voices_dir = "app/models/f5-tts-voices" self.test_speaker_name = "test_speaker" self.test_speaker_wav = os.path.join(self.voices_dir, f"{self.test_speaker_name}.wav") self.test_speaker_txt = os.path.join(self.voices_dir, f"{self.test_speaker_name}.txt")