fix: ChatContext is not a sequence — use .items for len/list; add test suite
Bug fix: - GemmaLLM.chat(): len(chat_ctx) → len(chat_ctx.items) - GemmaLLM._compact_context(): list(chat_ctx) → list(chat_ctx.items) ChatContext in livekit-agents 1.7 is not iterable or sized directly; it exposes an .items list. The TypeError was silently killing all LLM responses (agent heard user but never spoke back). Test suite (tests/): - test_web_ui.py: static assets, token endpoint, LiveKit WS proxy - test_llm_api.py: completion, tool calling, thinking-disabled latency, streaming - test_mcp_tools.py: weather (Fahrenheit), time, memory, skills — all via docker exec + in-container MCP client - test_agent_integration.py: process alive, worker registered, supervisord, container health, LiveKit API - test_compaction.py: size-triggered compaction, system prompt preservation, short-context no-op - conftest.py: shared fixtures (HTTPS client, token, LLM, docker exec helpers) - Run with: .venv-tests/bin/python -m pytest tests/ -v
This commit is contained in:
@@ -5,3 +5,6 @@ __pycache__/
|
|||||||
node_modules/
|
node_modules/
|
||||||
*.log
|
*.log
|
||||||
certs/
|
certs/
|
||||||
|
.venv-tests/
|
||||||
|
__pycache__/
|
||||||
|
.pytest_cache/
|
||||||
|
|||||||
+5
-4
@@ -192,15 +192,16 @@ class GemmaLLM(openai.LLM):
|
|||||||
|
|
||||||
# Compact if context is large, OR if we've been idle for 5+ minutes
|
# Compact if context is large, OR if we've been idle for 5+ minutes
|
||||||
# (user came back after a break — summarize the old conversation).
|
# (user came back after a break — summarize the old conversation).
|
||||||
should_compact = len(chat_ctx) > self.COMPACTION_THRESHOLD
|
ctx_len = len(chat_ctx.items)
|
||||||
|
should_compact = ctx_len > self.COMPACTION_THRESHOLD
|
||||||
if not should_compact and idle_for > self.IDLE_COMPACTION_SECONDS:
|
if not should_compact and idle_for > self.IDLE_COMPACTION_SECONDS:
|
||||||
should_compact = len(chat_ctx) > 6 # minimal context worth summarizing
|
should_compact = ctx_len > 6 # minimal context worth summarizing
|
||||||
|
|
||||||
if should_compact:
|
if should_compact:
|
||||||
reason = "idle" if idle_for > self.IDLE_COMPACTION_SECONDS else "size"
|
reason = "idle" if idle_for > self.IDLE_COMPACTION_SECONDS else "size"
|
||||||
logger.info(
|
logger.info(
|
||||||
"Compacting context (%s): %d items, idle %.0fs",
|
"Compacting context (%s): %d items, idle %.0fs",
|
||||||
reason, len(chat_ctx), idle_for,
|
reason, ctx_len, idle_for,
|
||||||
)
|
)
|
||||||
chat_ctx = self._compact_context(chat_ctx, conn_options)
|
chat_ctx = self._compact_context(chat_ctx, conn_options)
|
||||||
|
|
||||||
@@ -221,7 +222,7 @@ class GemmaLLM(openai.LLM):
|
|||||||
"""
|
"""
|
||||||
from livekit.agents import llm as lk_llm
|
from livekit.agents import llm as lk_llm
|
||||||
|
|
||||||
items = list(chat_ctx)
|
items = list(chat_ctx.items)
|
||||||
# Identify the system/instruction message (first item usually)
|
# Identify the system/instruction message (first item usually)
|
||||||
sys_items = [it for it in items if getattr(it, "role", None) in ("system", "developer")]
|
sys_items = [it for it in items if getattr(it, "role", None) in ("system", "developer")]
|
||||||
non_sys = [it for it in items if getattr(it, "role", None) not in ("system", "developer")]
|
non_sys = [it for it in items if getattr(it, "role", None) not in ("system", "developer")]
|
||||||
|
|||||||
@@ -0,0 +1,38 @@
|
|||||||
|
# Hope test suite
|
||||||
|
|
||||||
|
End-to-end tests that drive the **actual running container** — no mocks.
|
||||||
|
They run on the host and hit the live web UI, token endpoint, LLM API, MCP
|
||||||
|
tool servers, and agent process.
|
||||||
|
|
||||||
|
## Prerequisites
|
||||||
|
|
||||||
|
- The container is up: `docker compose up -d` (and healthy)
|
||||||
|
- Host deps installed: `pip install -r tests/requirements.txt`
|
||||||
|
(or `uv pip install -r tests/requirements.txt`)
|
||||||
|
- The LLM at `GEMMA_BASE_URL` (default `http://192.168.86.2:8023/v1`) is reachable
|
||||||
|
- Internet access for the weather tool tests (wttr.in)
|
||||||
|
|
||||||
|
## Run
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd /home/shane/dev/voice
|
||||||
|
python -m pytest tests/ -v # everything
|
||||||
|
python -m pytest tests/ -v -m "not slow" # skip LLM/network-heavy tests
|
||||||
|
```
|
||||||
|
|
||||||
|
## What each file covers
|
||||||
|
|
||||||
|
| File | Target | Notes |
|
||||||
|
|------|--------|-------|
|
||||||
|
| `test_web_ui.py` | nginx HTTPS UI (:8090) + token endpoint (:8091) | static assets, `/token` (direct + via nginx), `/livekit/` proxy |
|
||||||
|
| `test_llm_api.py` | Gemma LLM API directly | completion, tool calling, `enable_thinking=false` latency, streaming |
|
||||||
|
| `test_mcp_tools.py` | MCP servers via `docker exec` | weather (Fahrenheit check), time, memory save/recall/list, skills save/recall/list — probe files are cleaned up |
|
||||||
|
| `test_agent_integration.py` | running container | agent process alive, worker registered with LiveKit, all 4 supervised processes up, container health, LiveKit HTTP API |
|
||||||
|
| `test_compaction.py` | `GemmaLLM._compact_context` in-container | compaction triggers >24 items, system prompt preserved, short context untouched |
|
||||||
|
|
||||||
|
## Notes
|
||||||
|
|
||||||
|
- LLM and MCP tests are marked `@pytest.mark.slow` (1–30s each).
|
||||||
|
- The token endpoint is **POST** `/token` → `{"token": "<JWT>"}`.
|
||||||
|
- Process checks use `/proc/*/cmdline` because the container image has no
|
||||||
|
`ps`, and the supervisorctl unix socket is not exposed in this build.
|
||||||
@@ -0,0 +1,136 @@
|
|||||||
|
"""Shared fixtures for the Hope voice-assistant test suite.
|
||||||
|
|
||||||
|
Tests run on the HOST machine against the live "voice" container:
|
||||||
|
- HTTPS web UI -> https://localhost:8090 (self-signed cert, verify=False)
|
||||||
|
- Token server -> http://127.0.0.1:8091 (host network mode)
|
||||||
|
- LLM API -> GEMMA_BASE_URL (default http://192.168.86.2:8023/v1)
|
||||||
|
- In-container -> docker exec voice ...
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
CONTAINER = "voice"
|
||||||
|
WEB_BASE_URL = os.environ.get("VOICE_WEB_URL", "https://localhost:8090")
|
||||||
|
TOKEN_URL = os.environ.get("VOICE_TOKEN_URL", "http://127.0.0.1:8091/token")
|
||||||
|
GEMMA_BASE_URL = os.environ.get("GEMMA_BASE_URL", "http://192.168.86.2:8023/v1").rstrip("/")
|
||||||
|
GEMMA_MODEL = os.environ.get("GEMMA_MODEL", "gemma-4-e4b")
|
||||||
|
|
||||||
|
# Path to the agent's venv python inside the container (has mcp, livekit.agents)
|
||||||
|
CONTAINER_PYTHON = "/opt/voice-agent/.venv/bin/python"
|
||||||
|
AGENT_DIR_IN_CONTAINER = "/opt/voice-agent"
|
||||||
|
|
||||||
|
|
||||||
|
def pytest_configure(config):
|
||||||
|
config.addinivalue_line("markers", "slow: marks tests as slow (LLM / network calls)")
|
||||||
|
|
||||||
|
|
||||||
|
# ── docker helpers ───────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def docker_exec(*args, timeout=60, env=None, check=True) -> subprocess.CompletedProcess:
|
||||||
|
"""Run a command inside the voice container. Returns CompletedProcess."""
|
||||||
|
cmd = ["docker", "exec"]
|
||||||
|
if env:
|
||||||
|
for k, v in env.items():
|
||||||
|
cmd += ["-e", f"{k}={v}"]
|
||||||
|
cmd += [CONTAINER, *args]
|
||||||
|
return subprocess.run(
|
||||||
|
cmd,
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
timeout=timeout,
|
||||||
|
check=check,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def docker_exec_python(
|
||||||
|
code: str, *, timeout=120, cwd=None, env=None, check=True
|
||||||
|
) -> subprocess.CompletedProcess:
|
||||||
|
"""Run a python snippet inside the container with the agent venv."""
|
||||||
|
return docker_exec(CONTAINER_PYTHON, "-c", code, timeout=timeout, env=env, check=check)
|
||||||
|
|
||||||
|
|
||||||
|
# ── fixtures ─────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(scope="session")
|
||||||
|
def base_url() -> str:
|
||||||
|
"""Base URL of the web UI (HTTPS, self-signed)."""
|
||||||
|
return WEB_BASE_URL
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(scope="session")
|
||||||
|
def client(base_url):
|
||||||
|
"""httpx client for the HTTPS web UI with TLS verification disabled."""
|
||||||
|
with httpx.Client(base_url=base_url, verify=False, timeout=15) as c:
|
||||||
|
yield c
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(scope="session")
|
||||||
|
def token() -> dict:
|
||||||
|
"""Fetch a fresh LiveKit token from the token endpoint (direct, no proxy)."""
|
||||||
|
resp = httpx.post(TOKEN_URL, json={}, timeout=10)
|
||||||
|
resp.raise_for_status()
|
||||||
|
data = resp.json()
|
||||||
|
assert "token" in data, f"token endpoint did not return a token: {data.keys()}"
|
||||||
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(scope="session")
|
||||||
|
def llm_client():
|
||||||
|
"""httpx client pointed at the Gemma LLM API (same one Hope uses)."""
|
||||||
|
with httpx.Client(base_url=GEMMA_BASE_URL, timeout=60) as c:
|
||||||
|
yield c
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(scope="session")
|
||||||
|
def mcp_client_factory():
|
||||||
|
"""Factory that runs an MCP stdio tool call inside the container.
|
||||||
|
|
||||||
|
Returns (result_text, is_error). Spawns the MCP server script via the
|
||||||
|
agent venv python and drives it with a raw JSON-RPC session using the
|
||||||
|
mcp client library (available in the container, not on the host).
|
||||||
|
"""
|
||||||
|
|
||||||
|
def _call(script: str, tool_name: str, args: dict | None = None, env: dict | None = None) -> tuple[str, bool]:
|
||||||
|
import os as host_os
|
||||||
|
|
||||||
|
merged_env = {**host_os.environ, **(env or {})}
|
||||||
|
code = f"""
|
||||||
|
import asyncio, json
|
||||||
|
from mcp import ClientSession, StdioServerParameters
|
||||||
|
from mcp.client.stdio import stdio_client
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
params = StdioServerParameters(
|
||||||
|
command={CONTAINER_PYTHON!r},
|
||||||
|
args=[{AGENT_DIR_IN_CONTAINER + "/" + script!r}],
|
||||||
|
env={json.dumps(merged_env)},
|
||||||
|
)
|
||||||
|
async with stdio_client(params) as (read, write):
|
||||||
|
async with ClientSession(read, write) as session:
|
||||||
|
await session.initialize()
|
||||||
|
result = await session.call_tool({tool_name!r}, {json.dumps(args or {})})
|
||||||
|
text = ""
|
||||||
|
for block in result.content:
|
||||||
|
if getattr(block, "type", None) == "text":
|
||||||
|
text += block.text
|
||||||
|
print(json.dumps({{"text": text, "isError": bool(result.isError)}}))
|
||||||
|
|
||||||
|
asyncio.run(main())
|
||||||
|
"""
|
||||||
|
proc = docker_exec_python(code, timeout=120, check=False)
|
||||||
|
if proc.returncode != 0:
|
||||||
|
raise AssertionError(
|
||||||
|
f"MCP call {tool_name} failed (rc={proc.returncode}): "
|
||||||
|
f"stderr={proc.stderr[-800:]}"
|
||||||
|
)
|
||||||
|
data = json.loads(proc.stdout.strip().splitlines()[-1])
|
||||||
|
return data["text"], data["isError"]
|
||||||
|
|
||||||
|
return _call
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
pytest
|
||||||
|
httpx
|
||||||
@@ -0,0 +1,69 @@
|
|||||||
|
"""Integration tests for the running container: processes, health, LiveKit."""
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
|
from conftest import docker_exec
|
||||||
|
|
||||||
|
|
||||||
|
def _proc_cmdlines() -> list[str]:
|
||||||
|
"""All process cmdlines in the container (no `ps` available)."""
|
||||||
|
proc = docker_exec(
|
||||||
|
"sh",
|
||||||
|
"-c",
|
||||||
|
'for p in /proc/[0-9]*/cmdline; do tr "\\0" " " < "$p" 2>/dev/null; echo; done',
|
||||||
|
)
|
||||||
|
return [line.strip() for line in proc.stdout.splitlines() if line.strip()]
|
||||||
|
|
||||||
|
|
||||||
|
def test_agent_process_alive():
|
||||||
|
cmdlines = _proc_cmdlines()
|
||||||
|
assert any("agent.py start" in c for c in cmdlines), (
|
||||||
|
"no agent.py process found in container"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_agent_registered_with_livekit():
|
||||||
|
"""`docker logs` runs on the host (not via docker exec)."""
|
||||||
|
import subprocess
|
||||||
|
|
||||||
|
proc = subprocess.run(
|
||||||
|
["docker", "logs", "--tail", "500", "voice"],
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
timeout=30,
|
||||||
|
)
|
||||||
|
assert "registered worker" in proc.stdout or "worker registered" in proc.stdout, (
|
||||||
|
"agent never registered a worker with LiveKit"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_all_supervisord_processes_up():
|
||||||
|
"""The supervisorctl socket is not exposed in this image, so verify each
|
||||||
|
supervised program's process directly via /proc."""
|
||||||
|
cmdlines = _proc_cmdlines()
|
||||||
|
|
||||||
|
def has(pattern: str) -> bool:
|
||||||
|
return any(re.search(pattern, c) for c in cmdlines)
|
||||||
|
|
||||||
|
assert has(r"livekit --config"), "livekit server not running"
|
||||||
|
assert has(r"agent\.py start"), "voice agent not running"
|
||||||
|
assert has(r"nginx.*daemon off"), "nginx (web UI) not running"
|
||||||
|
assert has(r"token_server\.py"), "token server not running"
|
||||||
|
|
||||||
|
|
||||||
|
def test_container_healthy():
|
||||||
|
import subprocess
|
||||||
|
|
||||||
|
proc = subprocess.run(
|
||||||
|
["docker", "inspect", "--format", "{{.State.Health.Status}}", "voice"],
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
timeout=30,
|
||||||
|
)
|
||||||
|
assert proc.stdout.strip() == "healthy"
|
||||||
|
|
||||||
|
|
||||||
|
def test_livekit_server_responds():
|
||||||
|
"""LiveKit HTTP API on :7880 (internal; reached via docker exec curl)."""
|
||||||
|
proc = docker_exec("curl", "-s", "-o", "/dev/null", "-w", "%{http_code}", "http://localhost:7880/")
|
||||||
|
assert proc.stdout.strip() in ("200", "404"), f"unexpected status {proc.stdout.strip()!r}"
|
||||||
@@ -0,0 +1,93 @@
|
|||||||
|
"""Tests for GemmaLLM context compaction (agent.GemmaLLM._compact_context).
|
||||||
|
|
||||||
|
Compaction runs a real side LLM call to the xNAS, so these are marked slow.
|
||||||
|
The agent module is imported inside the container via docker exec.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from conftest import GEMMA_BASE_URL, GEMMA_MODEL, docker_exec_python
|
||||||
|
|
||||||
|
# {num_messages} and {do_chat} are filled per test; all other braces are literal.
|
||||||
|
TEMPLATE = '''
|
||||||
|
import logging, sys
|
||||||
|
from livekit.agents import llm as lk_llm
|
||||||
|
sys.path.insert(0, "/opt/voice-agent")
|
||||||
|
from agent import GemmaLLM
|
||||||
|
|
||||||
|
logging.basicConfig(level=logging.INFO, stream=sys.stdout)
|
||||||
|
|
||||||
|
llm = GemmaLLM(
|
||||||
|
model="{model}",
|
||||||
|
base_url="{base_url}",
|
||||||
|
api_key="not-needed",
|
||||||
|
max_completion_tokens=100,
|
||||||
|
extra_body={{"chat_template_kwargs": {{"enable_thinking": False}}}},
|
||||||
|
)
|
||||||
|
|
||||||
|
items = [lk_llm.ChatMessage(role="system", content=["You are Hope, a voice assistant."])]
|
||||||
|
for i in range({num_messages}):
|
||||||
|
items.append(lk_llm.ChatMessage(role="user", content=["user message number " + str(i)]))
|
||||||
|
items.append(lk_llm.ChatMessage(role="assistant", content=["agent reply number " + str(i)]))
|
||||||
|
|
||||||
|
ctx = lk_llm.ChatContext(items)
|
||||||
|
print("BUILT ctx with " + str(len(ctx.items)) + " items")
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
try:
|
||||||
|
compacted = llm._compact_context(ctx, None)
|
||||||
|
except Exception as e:
|
||||||
|
print("COMPACT_ERROR " + repr(e))
|
||||||
|
return
|
||||||
|
print("AFTER items=" + str(len(compacted.items)))
|
||||||
|
for it in compacted.items:
|
||||||
|
role = getattr(it, "role", None)
|
||||||
|
content = GemmaLLM._msg_content_text(it)
|
||||||
|
print("ITEM " + str(role) + ": " + content[:120])
|
||||||
|
|
||||||
|
{do_chat}
|
||||||
|
|
||||||
|
asyncio.run(main())
|
||||||
|
'''
|
||||||
|
|
||||||
|
CHAT_SNIPPET = ' stream = llm.chat(chat_ctx=compacted)\n collected = await stream.collect()\n print("CHAT_OK text_len=" + str(len(collected.text)))'
|
||||||
|
|
||||||
|
|
||||||
|
def _run(num_messages: int, *, do_chat: bool = False) -> str:
|
||||||
|
code = TEMPLATE.format(
|
||||||
|
model=GEMMA_MODEL,
|
||||||
|
base_url=GEMMA_BASE_URL,
|
||||||
|
num_messages=num_messages,
|
||||||
|
do_chat=CHAT_SNIPPET if do_chat else "pass",
|
||||||
|
)
|
||||||
|
proc = docker_exec_python(code, timeout=120, check=False)
|
||||||
|
return proc.stdout + (("\n" + proc.stderr) if proc.stderr else "")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_compaction_triggers_on_size():
|
||||||
|
"""A context with > 24 items must compact without crashing and chat() must still work."""
|
||||||
|
# 13 pairs = 26 non-system + 1 system = 27 items (> threshold of 24)
|
||||||
|
out = _run(13, do_chat=True)
|
||||||
|
assert "BUILT ctx with 27 items" in out
|
||||||
|
assert "Context compacted:" in out
|
||||||
|
assert "CHAT_OK text_len=" in out
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_compaction_preserves_system_prompt():
|
||||||
|
out = _run(28)
|
||||||
|
lines = [l for l in out.splitlines() if l.startswith("ITEM system:")]
|
||||||
|
assert lines, f"no system message survived compaction:\n{out}"
|
||||||
|
assert "You are Hope" in lines[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_short_context_no_compaction():
|
||||||
|
"""Context with non_sys <= KEEP_RECENT must NOT be compacted."""
|
||||||
|
# system + 4 user/assistant pairs = 9 items, non_sys=8 <= KEEP_RECENT(10)
|
||||||
|
out = _run(4)
|
||||||
|
assert "BUILT ctx with 9 items" in out
|
||||||
|
# _compact_context should return the context unchanged (early return)
|
||||||
|
assert "AFTER items=9" in out, f"expected no compaction:\n{out}"
|
||||||
@@ -0,0 +1,106 @@
|
|||||||
|
"""Tests for the Gemma LLM API — the exact endpoint Hope's agent uses."""
|
||||||
|
|
||||||
|
import time
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from conftest import GEMMA_BASE_URL, GEMMA_MODEL
|
||||||
|
|
||||||
|
WEATHER_TOOL = {
|
||||||
|
"type": "function",
|
||||||
|
"function": {
|
||||||
|
"name": "get_weather",
|
||||||
|
"description": "Get the current weather and today's forecast for a location.",
|
||||||
|
"parameters": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {"location": {"type": "string"}},
|
||||||
|
"required": ["location"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _chat(llm_client, messages, *, tools=None, stream=False, max_tokens=200):
|
||||||
|
payload = {
|
||||||
|
"model": GEMMA_MODEL,
|
||||||
|
"messages": messages,
|
||||||
|
"max_tokens": max_tokens,
|
||||||
|
"temperature": 0.3,
|
||||||
|
"chat_template_kwargs": {"enable_thinking": False},
|
||||||
|
}
|
||||||
|
if tools:
|
||||||
|
payload["tools"] = tools
|
||||||
|
if stream:
|
||||||
|
payload["stream"] = True
|
||||||
|
return llm_client.post("/chat/completions", json=payload)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_simple_completion(llm_client):
|
||||||
|
resp = _chat(
|
||||||
|
llm_client,
|
||||||
|
[{"role": "user", "content": "What is 2+2? Answer with just the number."}],
|
||||||
|
max_tokens=50,
|
||||||
|
)
|
||||||
|
assert resp.status_code == 200
|
||||||
|
data = resp.json()
|
||||||
|
content = (data["choices"][0]["message"].get("content") or "").strip()
|
||||||
|
assert "4" in content
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_tool_call_weather(llm_client):
|
||||||
|
resp = _chat(
|
||||||
|
llm_client,
|
||||||
|
[{"role": "user", "content": "What is the weather like in Raleigh right now?"}],
|
||||||
|
tools=[WEATHER_TOOL],
|
||||||
|
max_tokens=100,
|
||||||
|
)
|
||||||
|
assert resp.status_code == 200
|
||||||
|
choice = resp.json()["choices"][0]
|
||||||
|
assert choice["finish_reason"] == "tool_calls"
|
||||||
|
tool_calls = choice["message"]["tool_calls"]
|
||||||
|
assert any(tc["function"]["name"] == "get_weather" for tc in tool_calls)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_thinking_disabled(llm_client):
|
||||||
|
"""A normally-slow prompt must answer fast when enable_thinking=false."""
|
||||||
|
messages = [
|
||||||
|
{"role": "user", "content": "Count from 1 to 20, then tell me the sum."},
|
||||||
|
]
|
||||||
|
start = time.monotonic()
|
||||||
|
resp = _chat(llm_client, messages, max_tokens=300)
|
||||||
|
elapsed = time.monotonic() - start
|
||||||
|
assert resp.status_code == 200
|
||||||
|
content = (resp.json()["choices"][0]["message"].get("content") or "").strip()
|
||||||
|
assert len(content) > 10
|
||||||
|
# With thinking enabled this prompt takes well over 3s on the xNAS.
|
||||||
|
assert elapsed < 3, f"response took {elapsed:.1f}s — thinking may be enabled"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_streaming(llm_client):
|
||||||
|
resp = _chat(
|
||||||
|
llm_client,
|
||||||
|
[{"role": "user", "content": "Say the words: the quick brown fox jumps over the lazy dog."}],
|
||||||
|
stream=True,
|
||||||
|
max_tokens=100,
|
||||||
|
)
|
||||||
|
assert resp.status_code == 200
|
||||||
|
|
||||||
|
chunks = []
|
||||||
|
for line in resp.iter_lines():
|
||||||
|
if not line or not line.startswith("data: "):
|
||||||
|
continue
|
||||||
|
payload = line[6:].strip()
|
||||||
|
if payload == "[DONE]":
|
||||||
|
break
|
||||||
|
chunk = __import__("json").loads(payload)
|
||||||
|
delta = (chunk.get("choices") or [{}])[0].get("delta", {})
|
||||||
|
if delta.get("content"):
|
||||||
|
chunks.append(delta["content"])
|
||||||
|
|
||||||
|
assert len(chunks) > 1, f"expected multiple streamed chunks, got {len(chunks)}"
|
||||||
|
assembled = "".join(chunks)
|
||||||
|
assert "quick brown fox" in assembled.lower()
|
||||||
@@ -0,0 +1,132 @@
|
|||||||
|
"""Tests for the MCP tool servers (weather, memory, skills) via docker exec.
|
||||||
|
|
||||||
|
Each test spawns the real MCP server script inside the container and drives it
|
||||||
|
with a JSON-RPC session using the mcp client library from the agent venv.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_number(text: str) -> float | None:
|
||||||
|
"""Pull the first plausible temperature number out of a weather summary."""
|
||||||
|
m = re.search(r"(\d+)\s*degrees", text)
|
||||||
|
if m:
|
||||||
|
return float(m.group(1))
|
||||||
|
m = re.search(r"\b(\d{2,3})\b", text)
|
||||||
|
return float(m.group(1)) if m else None
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_weather_get(mcp_client_factory):
|
||||||
|
text, is_error = mcp_client_factory("weather_mcp.py", "get_weather", {"location": "Raleigh"})
|
||||||
|
assert not is_error
|
||||||
|
assert "degrees" in text.lower()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_weather_fahrenheit(mcp_client_factory):
|
||||||
|
"""wttr.in reports Celsius; the tool must convert to Fahrenheit."""
|
||||||
|
text, is_error = mcp_client_factory("weather_mcp.py", "get_weather", {"location": "Raleigh"})
|
||||||
|
assert not is_error
|
||||||
|
temp = _extract_number(text)
|
||||||
|
assert temp is not None, f"no temperature found in: {text}"
|
||||||
|
# Fahrenheit for any inhabited place is > 50 (Celsius would be ~10-30).
|
||||||
|
assert temp > 50, f"temperature {temp} looks like Celsius, expected Fahrenheit"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_time_get(mcp_client_factory):
|
||||||
|
text, is_error = mcp_client_factory("weather_mcp.py", "get_time", {})
|
||||||
|
assert not is_error
|
||||||
|
assert ("AM" in text or "PM" in text)
|
||||||
|
days = ("Monday", "Tuesday", "Wednesday", "Thursday", "Friday", "Saturday", "Sunday")
|
||||||
|
assert any(d in text for d in days), f"no day name in: {text}"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_memory_save_and_recall(mcp_client_factory):
|
||||||
|
topic = "test-suite-probe"
|
||||||
|
content = "zebra42 is the probe fact for the voice test suite"
|
||||||
|
|
||||||
|
saved, err = mcp_client_factory(
|
||||||
|
"memory_mcp.py", "memory_save", {"topic": topic, "content": content}
|
||||||
|
)
|
||||||
|
assert not err
|
||||||
|
assert "Saved" in saved
|
||||||
|
|
||||||
|
try:
|
||||||
|
recalled, err = mcp_client_factory("memory_mcp.py", "memory_recall", {"query": "zebra42"})
|
||||||
|
assert not err
|
||||||
|
assert "zebra42" in recalled
|
||||||
|
assert topic in recalled # recall output is tagged with the topic name
|
||||||
|
finally:
|
||||||
|
# Clean up: remove the probe file so repeated runs stay green.
|
||||||
|
from conftest import docker_exec
|
||||||
|
|
||||||
|
docker_exec("rm", "-f", "/memory/test-suite-probe.md")
|
||||||
|
|
||||||
|
|
||||||
|
def test_memory_list(mcp_client_factory):
|
||||||
|
from conftest import docker_exec
|
||||||
|
|
||||||
|
topic = "test-suite-list-probe"
|
||||||
|
try:
|
||||||
|
saved, err = mcp_client_factory(
|
||||||
|
"memory_mcp.py", "memory_save", {"topic": topic, "content": "listing probe"}
|
||||||
|
)
|
||||||
|
assert not err
|
||||||
|
listed, err = mcp_client_factory("memory_mcp.py", "memory_list", {})
|
||||||
|
assert not err
|
||||||
|
assert topic in listed
|
||||||
|
finally:
|
||||||
|
docker_exec("rm", "-f", "/memory/test-suite-list-probe.md")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.slow
|
||||||
|
def test_skill_save_and_recall(mcp_client_factory):
|
||||||
|
name = "Test Suite Probe Skill"
|
||||||
|
description = "A probe skill used by the voice test suite."
|
||||||
|
steps = "Step 1: do the thing. Step 2: verify zebra42 was done."
|
||||||
|
|
||||||
|
saved, err = mcp_client_factory(
|
||||||
|
"skills_mcp.py",
|
||||||
|
"skill_save",
|
||||||
|
{"name": name, "description": description, "steps": steps},
|
||||||
|
)
|
||||||
|
assert not err
|
||||||
|
assert "Saved" in saved
|
||||||
|
|
||||||
|
try:
|
||||||
|
recalled, err = mcp_client_factory("skills_mcp.py", "skill_recall", {"query": "zebra42"})
|
||||||
|
assert not err
|
||||||
|
assert "zebra42" in recalled
|
||||||
|
assert "Steps" in recalled
|
||||||
|
finally:
|
||||||
|
from conftest import docker_exec
|
||||||
|
|
||||||
|
docker_exec("rm", "-f", "/skills/test-suite-probe-skill.md")
|
||||||
|
|
||||||
|
|
||||||
|
def test_skill_list(mcp_client_factory):
|
||||||
|
from conftest import docker_exec
|
||||||
|
|
||||||
|
name = "Test Suite List Skill"
|
||||||
|
try:
|
||||||
|
saved, err = mcp_client_factory(
|
||||||
|
"skills_mcp.py",
|
||||||
|
"skill_save",
|
||||||
|
{
|
||||||
|
"name": name,
|
||||||
|
"description": "listing probe skill",
|
||||||
|
"steps": "one step",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
assert not err
|
||||||
|
listed, err = mcp_client_factory("skills_mcp.py", "skill_list", {})
|
||||||
|
assert not err
|
||||||
|
# list output uses the slugified filename
|
||||||
|
assert "test-suite-list-skill" in listed
|
||||||
|
finally:
|
||||||
|
docker_exec("rm", "-f", "/skills/test-suite-list-skill.md")
|
||||||
@@ -0,0 +1,74 @@
|
|||||||
|
"""Tests for the web UI served by nginx (HTTPS :8090) and the token endpoint."""
|
||||||
|
|
||||||
|
import json
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
|
||||||
|
def test_index_serves(client):
|
||||||
|
resp = client.get("/")
|
||||||
|
assert resp.status_code == 200
|
||||||
|
body = resp.text
|
||||||
|
assert "Hope" in body
|
||||||
|
assert "livekit-client.umd.js" in body
|
||||||
|
assert "app.js" in body
|
||||||
|
|
||||||
|
|
||||||
|
def test_app_js_serves(client):
|
||||||
|
resp = client.get("/app.js")
|
||||||
|
assert resp.status_code == 200
|
||||||
|
assert "LiveKitClient" in resp.text or "Room" in resp.text
|
||||||
|
|
||||||
|
|
||||||
|
def test_style_css_serves(client):
|
||||||
|
resp = client.get("/style.css")
|
||||||
|
assert resp.status_code == 200
|
||||||
|
body = resp.text
|
||||||
|
assert "{" in body and "}" in body
|
||||||
|
assert "color" in body or "background" in body
|
||||||
|
|
||||||
|
|
||||||
|
def test_manifest_serves(client):
|
||||||
|
resp = client.get("/manifest.json")
|
||||||
|
assert resp.status_code == 200
|
||||||
|
manifest = json.loads(resp.text)
|
||||||
|
assert "name" in manifest
|
||||||
|
|
||||||
|
|
||||||
|
def test_favicon_serves(client):
|
||||||
|
resp = client.get("/favicon.svg")
|
||||||
|
assert resp.status_code == 200
|
||||||
|
assert "svg" in resp.headers.get("content-type", "")
|
||||||
|
assert "<svg" in resp.text
|
||||||
|
|
||||||
|
|
||||||
|
def test_livekit_client_bundle_serves(client):
|
||||||
|
resp = client.get("/livekit-client.umd.js")
|
||||||
|
assert resp.status_code == 200
|
||||||
|
assert len(resp.text) > 100_000 # vendored UMD bundle is large
|
||||||
|
|
||||||
|
|
||||||
|
def test_token_endpoint_direct():
|
||||||
|
"""Token server on 127.0.0.1:8091 (host network mode)."""
|
||||||
|
resp = httpx.post("http://127.0.0.1:8091/token", json={}, timeout=10)
|
||||||
|
assert resp.status_code == 200
|
||||||
|
data = resp.json()
|
||||||
|
assert "token" in data and data["token"].count(".") == 2 # JWT shape
|
||||||
|
|
||||||
|
|
||||||
|
def test_token_endpoint_via_nginx(client):
|
||||||
|
"""nginx proxies /token to the token server."""
|
||||||
|
resp = client.post("/token", json={})
|
||||||
|
assert resp.status_code == 200
|
||||||
|
data = resp.json()
|
||||||
|
assert "token" in data and data["token"].count(".") == 2
|
||||||
|
|
||||||
|
|
||||||
|
def test_livekit_ws_proxy(client):
|
||||||
|
"""The /livekit/ path is proxied to the LiveKit server (HTTP :7880)."""
|
||||||
|
resp = client.get("/livekit/")
|
||||||
|
# LiveKit's HTTP API answers on /; a 200 or an API error JSON both prove
|
||||||
|
# the proxy reaches the LiveKit server rather than nginx serving statics.
|
||||||
|
assert resp.status_code in (200, 404, 405)
|
||||||
|
if resp.headers.get("content-type", "").startswith("application/json"):
|
||||||
|
json.loads(resp.text)
|
||||||
Reference in New Issue
Block a user