diff --git a/Dockerfile.tts-bridge b/Dockerfile.tts-bridge new file mode 100644 index 0000000..8821eed --- /dev/null +++ b/Dockerfile.tts-bridge @@ -0,0 +1,10 @@ +FROM python:3.13-slim + +WORKDIR /app +COPY requirements.tts-bridge.txt ./requirements.txt +RUN pip install --no-cache-dir -r requirements.txt +COPY tts_bridge.py ./app.py +RUN useradd --system --uid 10001 --home /nonexistent --shell /usr/sbin/nologin ttsbridge +USER 10001:10001 +EXPOSE 8099 +CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8099", "--proxy-headers", "--no-server-header"] diff --git a/README.md b/README.md index 9dd5306..2c614c8 100644 --- a/README.md +++ b/README.md @@ -1,3 +1,11 @@ # automation1 services -Git-managed Compose stack for Vaultwarden and Uptime Kuma on `automation1`. Runtime variables are stored in Dockhand and are not committed. +Git-managed Docker Compose stack for the existing automation1 services and the authenticated TTS bridge. + +## TTS bridge + +- Listen address: `0.0.0.0:8099` +- Health: `GET /health` +- Voiceclone: `POST /v1/tts` with bearer authentication +- Default voice: `deep_thought.mp3` +- Secrets stay host-local under `/app-config/tts-bridge/` and are never committed. diff --git a/compose.yaml b/compose.yaml index 0e561e5..65a8d2b 100644 --- a/compose.yaml +++ b/compose.yaml @@ -32,3 +32,34 @@ services: timeout: 30s start_period: 180s retries: 5 + + tts-bridge: + build: + context: . + dockerfile: Dockerfile.tts-bridge + image: local/automation1-tts-bridge:1.0.0 + container_name: tts-bridge + restart: always + ports: + - "0.0.0.0:8099:8099" + environment: + BUTLER_URL: http://10.4.1.116:8888 + BUTLER_TOKEN_FILE: /run/secrets/butler_token + CLIENT_TOKEN_FILE: /run/secrets/client_token + DEFAULT_VOICE: deep_thought.mp3 + volumes: + - /app-config/tts-bridge/butler-token:/run/secrets/butler_token:ro + - /app-config/tts-bridge/client-token:/run/secrets/client_token:ro + read_only: true + tmpfs: + - /tmp:size=16m,mode=1777 + security_opt: + - no-new-privileges:true + cap_drop: + - ALL + healthcheck: + test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8099/health', timeout=5)"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 20s diff --git a/requirements.tts-bridge.txt b/requirements.tts-bridge.txt new file mode 100644 index 0000000..880ac0f --- /dev/null +++ b/requirements.tts-bridge.txt @@ -0,0 +1,3 @@ +fastapi==0.116.1 +httpx==0.28.1 +uvicorn==0.35.0 diff --git a/tts_bridge.py b/tts_bridge.py new file mode 100644 index 0000000..4462d97 --- /dev/null +++ b/tts_bridge.py @@ -0,0 +1,76 @@ +from __future__ import annotations + +import hmac +import os +from contextlib import asynccontextmanager +from pathlib import Path + +import httpx +from fastapi import Depends, FastAPI, Header, HTTPException +from fastapi.responses import Response +from pydantic import BaseModel, Field + +BUTLER_URL = os.environ.get("BUTLER_URL", "http://10.4.1.116:8888").rstrip("/") +BUTLER_TOKEN_FILE = Path(os.environ.get("BUTLER_TOKEN_FILE", "/run/secrets/butler_token")) +CLIENT_TOKEN_FILE = Path(os.environ.get("CLIENT_TOKEN_FILE", "/run/secrets/client_token")) +DEFAULT_VOICE = os.environ.get("DEFAULT_VOICE", "deep_thought.mp3") + + +def _read_secret(path: Path, label: str) -> str: + try: + value = path.read_text().strip() + except OSError as exc: + raise RuntimeError(f"{label} secret is unavailable") from exc + if not value: + raise RuntimeError(f"{label} secret is empty") + return value + + +@asynccontextmanager +async def lifespan(app: FastAPI): + app.state.butler_token = _read_secret(BUTLER_TOKEN_FILE, "Butler") + app.state.client_token = _read_secret(CLIENT_TOKEN_FILE, "Client") + app.state.http = httpx.AsyncClient(timeout=httpx.Timeout(190.0, connect=10.0)) + yield + await app.state.http.aclose() + + +app = FastAPI(title="Pfannkuchen TTS Bridge", version="1.0.0", lifespan=lifespan, docs_url=None, redoc_url=None, openapi_url=None) + + +class GenerateRequest(BaseModel): + text: str = Field(min_length=1, max_length=2000) + voice: str = Field(default=DEFAULT_VOICE, pattern=r"^[A-Za-z0-9_.-]+$") + language: str = Field(default="de", pattern=r"^[A-Za-z]{2,8}(?:-[A-Za-z0-9]{2,8})?$") + + +def verify_client(authorization: str | None = Header(default=None)) -> None: + if not authorization or not authorization.startswith("Bearer "): + raise HTTPException(status_code=401, detail="Missing bearer token") + supplied = authorization.removeprefix("Bearer ").strip() + if not hmac.compare_digest(supplied, app.state.client_token): + raise HTTPException(status_code=401, detail="Invalid bearer token") + + +@app.get("/health") +async def health(): + try: + response = await app.state.http.get(f"{BUTLER_URL}/tts/health", headers={"Authorization": f"Bearer {app.state.butler_token}"}) + response.raise_for_status() + chatterbox_ok = response.json().get("chatterbox") == "ok" + except (httpx.HTTPError, ValueError): + chatterbox_ok = False + return {"status": "ok" if chatterbox_ok else "degraded", "chatterbox": chatterbox_ok} + + +@app.post("/v1/tts", response_class=Response, dependencies=[Depends(verify_client)]) +async def generate(req: GenerateRequest): + try: + upstream = await app.state.http.post(f"{BUTLER_URL}/tts/generate", headers={"Authorization": f"Bearer {app.state.butler_token}"}, json=req.model_dump()) + except httpx.RequestError: + raise HTTPException(status_code=502, detail="Butler is unavailable") + if upstream.status_code != 200: + raise HTTPException(status_code=502, detail="Speech generation failed") + if not upstream.content.startswith(b"RIFF"): + raise HTTPException(status_code=502, detail="Invalid audio received") + return Response(content=upstream.content, media_type="audio/wav", headers={"Content-Disposition": 'inline; filename="voiceclone.wav"', "Cache-Control": "no-store", "X-Content-Type-Options": "nosniff"})