Critical frontend bugs: - Add TrackSubscribed/attach() for agent audio playback - Fix decodeToString TypeError with TextDecoder - XSS fix: innerHTML -> textContent in addMessage - Fresh token on reconnect retry Agent fixes: - GemmaLLM subclass with reasoning_content fallback wrapper - Disable Gemma 4 thinking mode via chat_template_kwargs (6.8s -> 0.5s) - Remove duplicate session-level LLM - Replace global _active_session with closure-based handler - asyncio.create_task instead of deprecated get_event_loop - Explicit silero VAD, topic filter on voice-control Infra: - supervisord: all programs log to /dev/stdout - Dockerfile: uv sync --frozen with committed uv.lock - nginx config moved to real file, token_server.py no longer served - entrypoint.sh: cert persisted, only regenerated on IP change - compose: healthcheck + cert volume - token_server: CORS removed, room pinned to voice-room UI upgrade: - Orb UI with state machine (idle/connecting/listening/thinking/speaking) - Streaming transcripts via lk.transcription text streams - Barge-in hint, thinking chip, audio visualizer - Glassmorphism, chat bubbles, settings sheet, light mode - PWA manifest, favicon, wake-lock, safe-area insets - localStorage conversation history Docs: AGENTS.md drift fixed
77 lines
2.9 KiB
Docker
77 lines
2.9 KiB
Docker
# syntax=docker/dockerfile:1
|
|
|
|
# ── Stage 1: Build Python agent deps with uv ───────────────────────────────
|
|
FROM ghcr.io/astral-sh/uv:python3.12-bookworm-slim AS build
|
|
|
|
ENV PYTHONUNBUFFERED=1
|
|
ENV UV_COMPILE_BYTECODE=1
|
|
|
|
WORKDIR /app
|
|
|
|
# Install build tools for native extensions (azure-cognitiveservices-speech)
|
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
|
gcc g++ python3-dev libasound2-dev \
|
|
&& rm -rf /var/lib/apt/lists/*
|
|
|
|
COPY agent/pyproject.toml ./agent/
|
|
COPY agent/uv.lock ./agent/
|
|
RUN cd /app/agent && uv sync --frozen
|
|
|
|
# ── Stage 2: Runtime (use same base for Python compat) ─────────────────────
|
|
FROM ghcr.io/astral-sh/uv:python3.12-bookworm-slim AS runtime
|
|
|
|
ENV PYTHONUNBUFFERED=1
|
|
# Fix SSL cert path for the slim base image (ca-certificates installs to /etc/ssl/certs)
|
|
ENV SSL_CERT_FILE=/etc/ssl/certs/ca-certificates.crt
|
|
|
|
# Install LiveKit server binary + supervisord + nginx
|
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
|
curl ca-certificates supervisor nginx libasound2 iproute2 \
|
|
&& rm -rf /var/lib/apt/lists/*
|
|
|
|
# Download LiveKit server (latest stable)
|
|
ARG LIVEKIT_VERSION=v1.13.5
|
|
ARG LIVEKIT_TAG=1.13.5
|
|
RUN curl -sSL "https://github.com/livekit/livekit/releases/download/${LIVEKIT_VERSION}/livekit_${LIVEKIT_TAG}_linux_amd64.tar.gz" \
|
|
| tar xz -C /usr/local/bin/ livekit-server \
|
|
&& chmod +x /usr/local/bin/livekit-server \
|
|
&& ln -sf /usr/local/bin/livekit-server /usr/local/bin/livekit
|
|
|
|
# Copy Python agent + venv from build stage
|
|
COPY --from=build /app/agent/.venv /opt/voice-agent/.venv
|
|
COPY agent/agent.py /opt/voice-agent/agent.py
|
|
COPY agent/web_mcp.py /opt/voice-agent/web_mcp.py
|
|
|
|
# Copy web frontend + token endpoint
|
|
COPY web/index.html /var/www/voice/
|
|
COPY web/app.js /var/www/voice/
|
|
COPY web/style.css /var/www/voice/
|
|
COPY web/livekit-client.umd.js /var/www/voice/
|
|
COPY web/manifest.json /var/www/voice/
|
|
COPY web/favicon.svg /var/www/voice/
|
|
COPY web/token_server.py /opt/voice/token_server.py
|
|
|
|
# Config files
|
|
COPY livekit.yaml /etc/livekit.yaml
|
|
COPY supervisord.conf /etc/supervisor/conf.d/voice.conf
|
|
|
|
# Configure nginx to serve the voice UI on port 8090 over HTTPS.
|
|
# Browsers require a secure context (HTTPS or localhost) for microphone access.
|
|
# The self-signed cert (with the LAN IP in the SAN) is generated at container
|
|
# start by entrypoint.sh.
|
|
COPY nginx.conf /etc/nginx/sites-available/voice
|
|
RUN rm -f /etc/nginx/sites-enabled/default \
|
|
&& ln -sf /etc/nginx/sites-available/voice /etc/nginx/sites-enabled/voice \
|
|
&& mkdir -p /etc/voice/certs
|
|
|
|
# Create non-root user for agent
|
|
RUN useradd -m -s /bin/bash voiceuser || true
|
|
|
|
COPY entrypoint.sh /entrypoint.sh
|
|
RUN chmod +x /entrypoint.sh
|
|
|
|
EXPOSE 7880 7881 7882/udp 8090
|
|
|
|
ENTRYPOINT ["/entrypoint.sh"]
|
|
CMD ["supervisord", "-n", "-c", "/etc/supervisor/conf.d/voice.conf"]
|