feat: deploy authenticated voiceclone bridge #1

Merged
sascha merged 7 commits from feat/tts-bridge-20260813125855 into main 2026-08-13 13:01:36 +02:00
5 changed files with 129 additions and 1 deletions

10
Dockerfile.tts-bridge Normal file
View file

@ -0,0 +1,10 @@
FROM python:3.13-slim
WORKDIR /app
COPY requirements.tts-bridge.txt ./requirements.txt
RUN pip install --no-cache-dir -r requirements.txt
COPY tts_bridge.py ./app.py
RUN useradd --system --uid 10001 --home /nonexistent --shell /usr/sbin/nologin ttsbridge
USER 10001:10001
EXPOSE 8099
CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8099", "--proxy-headers", "--no-server-header"]

View file

@ -1,3 +1,11 @@
# automation1 services # automation1 services
Git-managed Compose stack for Vaultwarden and Uptime Kuma on `automation1`. Runtime variables are stored in Dockhand and are not committed. Git-managed Docker Compose stack for the existing automation1 services and the authenticated TTS bridge.
## TTS bridge
- Listen address: `0.0.0.0:8099`
- Health: `GET /health`
- Voiceclone: `POST /v1/tts` with bearer authentication
- Default voice: `deep_thought.mp3`
- Secrets stay host-local under `/app-config/tts-bridge/` and are never committed.

View file

@ -32,3 +32,34 @@ services:
timeout: 30s timeout: 30s
start_period: 180s start_period: 180s
retries: 5 retries: 5
tts-bridge:
build:
context: .
dockerfile: Dockerfile.tts-bridge
image: local/automation1-tts-bridge:1.0.0
container_name: tts-bridge
restart: always
ports:
- "0.0.0.0:8099:8099"
environment:
BUTLER_URL: http://10.4.1.116:8888
BUTLER_TOKEN_FILE: /run/secrets/butler_token
CLIENT_TOKEN_FILE: /run/secrets/client_token
DEFAULT_VOICE: deep_thought.mp3
volumes:
- /app-config/tts-bridge/butler-token:/run/secrets/butler_token:ro
- /app-config/tts-bridge/client-token:/run/secrets/client_token:ro
read_only: true
tmpfs:
- /tmp:size=16m,mode=1777
security_opt:
- no-new-privileges:true
cap_drop:
- ALL
healthcheck:
test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8099/health', timeout=5)"]
interval: 30s
timeout: 10s
retries: 3
start_period: 20s

View file

@ -0,0 +1,3 @@
fastapi==0.116.1
httpx==0.28.1
uvicorn==0.35.0

76
tts_bridge.py Normal file
View file

@ -0,0 +1,76 @@
from __future__ import annotations
import hmac
import os
from contextlib import asynccontextmanager
from pathlib import Path
import httpx
from fastapi import Depends, FastAPI, Header, HTTPException
from fastapi.responses import Response
from pydantic import BaseModel, Field
BUTLER_URL = os.environ.get("BUTLER_URL", "http://10.4.1.116:8888").rstrip("/")
BUTLER_TOKEN_FILE = Path(os.environ.get("BUTLER_TOKEN_FILE", "/run/secrets/butler_token"))
CLIENT_TOKEN_FILE = Path(os.environ.get("CLIENT_TOKEN_FILE", "/run/secrets/client_token"))
DEFAULT_VOICE = os.environ.get("DEFAULT_VOICE", "deep_thought.mp3")
def _read_secret(path: Path, label: str) -> str:
try:
value = path.read_text().strip()
except OSError as exc:
raise RuntimeError(f"{label} secret is unavailable") from exc
if not value:
raise RuntimeError(f"{label} secret is empty")
return value
@asynccontextmanager
async def lifespan(app: FastAPI):
app.state.butler_token = _read_secret(BUTLER_TOKEN_FILE, "Butler")
app.state.client_token = _read_secret(CLIENT_TOKEN_FILE, "Client")
app.state.http = httpx.AsyncClient(timeout=httpx.Timeout(190.0, connect=10.0))
yield
await app.state.http.aclose()
app = FastAPI(title="Pfannkuchen TTS Bridge", version="1.0.0", lifespan=lifespan, docs_url=None, redoc_url=None, openapi_url=None)
class GenerateRequest(BaseModel):
text: str = Field(min_length=1, max_length=2000)
voice: str = Field(default=DEFAULT_VOICE, pattern=r"^[A-Za-z0-9_.-]+$")
language: str = Field(default="de", pattern=r"^[A-Za-z]{2,8}(?:-[A-Za-z0-9]{2,8})?$")
def verify_client(authorization: str | None = Header(default=None)) -> None:
if not authorization or not authorization.startswith("Bearer "):
raise HTTPException(status_code=401, detail="Missing bearer token")
supplied = authorization.removeprefix("Bearer ").strip()
if not hmac.compare_digest(supplied, app.state.client_token):
raise HTTPException(status_code=401, detail="Invalid bearer token")
@app.get("/health")
async def health():
try:
response = await app.state.http.get(f"{BUTLER_URL}/tts/health", headers={"Authorization": f"Bearer {app.state.butler_token}"})
response.raise_for_status()
chatterbox_ok = response.json().get("chatterbox") == "ok"
except (httpx.HTTPError, ValueError):
chatterbox_ok = False
return {"status": "ok" if chatterbox_ok else "degraded", "chatterbox": chatterbox_ok}
@app.post("/v1/tts", response_class=Response, dependencies=[Depends(verify_client)])
async def generate(req: GenerateRequest):
try:
upstream = await app.state.http.post(f"{BUTLER_URL}/tts/generate", headers={"Authorization": f"Bearer {app.state.butler_token}"}, json=req.model_dump())
except httpx.RequestError:
raise HTTPException(status_code=502, detail="Butler is unavailable")
if upstream.status_code != 200:
raise HTTPException(status_code=502, detail="Speech generation failed")
if not upstream.content.startswith(b"RIFF"):
raise HTTPException(status_code=502, detail="Invalid audio received")
return Response(content=upstream.content, media_type="audio/wav", headers={"Content-Disposition": 'inline; filename="voiceclone.wav"', "Cache-Control": "no-store", "X-Content-Type-Options": "nosniff"})