v0.2.16 - Add Python test harness and fix increment script commit identity handling
This commit is contained in:
@@ -54,11 +54,11 @@ Skills compose by adoption-list order (`10123`) and trigger tags carry runtime e
|
||||
|
||||
Didactyl will support local inference, which is very privacy preserving. Remote inference does however have it's advantages, and in those cases Didactyl supports using Bitcoin Lightning and eCash inference providers.
|
||||
|
||||
## Current Status — v0.2.14
|
||||
## Current Status — v0.2.16
|
||||
|
||||
**Active build — this project is barely working. Experiment at your own risk.**
|
||||
|
||||
> Last release update: v0.2.14 — Follow-up push after context verification and startup-skill refactor
|
||||
> Last release update: v0.2.16 — Add Python test harness and fix increment script commit identity handling
|
||||
|
||||
- Connects to configured relays with auto-reconnect and relay state transition logging
|
||||
- Publishes configured startup events per relay as each relay becomes connected
|
||||
|
||||
@@ -111,6 +111,37 @@ check_git_repo() {
|
||||
fi
|
||||
}
|
||||
|
||||
# Ensure git identity exists so commits do not fail on clean machines
|
||||
ensure_git_identity() {
|
||||
local current_name=""
|
||||
local current_email=""
|
||||
|
||||
current_name=$(git config --get user.name 2>/dev/null || true)
|
||||
current_email=$(git config --get user.email 2>/dev/null || true)
|
||||
|
||||
if [[ -n "$current_name" && -n "$current_email" ]]; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
print_warning "Git user.name / user.email not fully configured for this repository"
|
||||
|
||||
local fallback_name
|
||||
local fallback_email
|
||||
|
||||
fallback_name="${GIT_AUTHOR_NAME:-Didactyl User}"
|
||||
fallback_email="${GIT_AUTHOR_EMAIL:-didactyl@local}"
|
||||
|
||||
if [[ -z "$current_name" ]]; then
|
||||
git config user.name "$fallback_name"
|
||||
print_status "Set local git user.name to '$fallback_name'"
|
||||
fi
|
||||
|
||||
if [[ -z "$current_email" ]]; then
|
||||
git config user.email "$fallback_email"
|
||||
print_status "Set local git user.email to '$fallback_email'"
|
||||
fi
|
||||
}
|
||||
|
||||
# Function to get current version and increment appropriately
|
||||
increment_version() {
|
||||
local increment_type="$1" # "patch", "minor", or "major"
|
||||
@@ -293,6 +324,8 @@ git_commit_and_push_no_tag() {
|
||||
print_success "Committed changes"
|
||||
else
|
||||
print_error "Failed to commit changes"
|
||||
print_error "git commit output:"
|
||||
git commit -m "$NEW_VERSION - $COMMIT_MESSAGE" 2>&1 || true
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
@@ -498,6 +531,7 @@ main() {
|
||||
|
||||
# Check prerequisites
|
||||
check_git_repo
|
||||
ensure_git_identity
|
||||
|
||||
if [[ "$RELEASE_MODE" == true ]]; then
|
||||
print_status "=== RELEASE MODE ==="
|
||||
|
||||
616
plans/automated_test_harness.md
Normal file
616
plans/automated_test_harness.md
Normal file
@@ -0,0 +1,616 @@
|
||||
# Didactyl Automated Test Harness
|
||||
|
||||
## Overview
|
||||
|
||||
An automated testing system that starts a Didactyl agent locally (debug build), converses with it via the HTTP API, exercises all tools, monitors logs in real-time, handles agent crashes/restarts, and produces a structured test results report.
|
||||
|
||||
**Language:** Python (stdlib only, no external dependencies)
|
||||
**Location:** `tests/`
|
||||
**Phase 1:** Scripted tests (no LLM driving the tester)
|
||||
**Phase 2 (future):** LLM-driven test agent that generates prompts, evaluates responses, and adapts
|
||||
|
||||
---
|
||||
|
||||
## Architecture
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
subgraph Test Harness - Python
|
||||
RUNNER[test_runner.py<br/>orchestrator]
|
||||
PROC[agent_process.py<br/>start/stop/restart]
|
||||
CLIENT[didactyl_client.py<br/>HTTP API wrapper]
|
||||
LOG[log_watcher.py<br/>tail debug.log]
|
||||
REPORT[reporter.py<br/>results output]
|
||||
|
||||
RUNNER --> PROC
|
||||
RUNNER --> CLIENT
|
||||
RUNNER --> LOG
|
||||
RUNNER --> REPORT
|
||||
end
|
||||
|
||||
subgraph Test Suites
|
||||
TH[test_health]
|
||||
TC[test_conversation]
|
||||
TI[test_tools_identity]
|
||||
TN[test_tools_nostr]
|
||||
TS[test_tools_skills]
|
||||
TSY[test_tools_system]
|
||||
TM[test_tools_memory]
|
||||
TCA[test_tools_cashu]
|
||||
TB[test_tools_blossom]
|
||||
TTO[test_timeouts]
|
||||
TE[test_errors]
|
||||
TR[test_restart]
|
||||
end
|
||||
|
||||
RUNNER --> TH & TC & TI & TN & TS & TSY & TM & TCA & TB & TTO & TE & TR
|
||||
|
||||
subgraph Didactyl Agent - debug build
|
||||
AGENT[didactyl process]
|
||||
API[HTTP API :8484]
|
||||
LOGFILE[debug.log]
|
||||
end
|
||||
|
||||
CLIENT -- HTTP --> API
|
||||
PROC -- subprocess --> AGENT
|
||||
LOG -- tail --> LOGFILE
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Design Decisions
|
||||
|
||||
| Decision | Choice | Rationale |
|
||||
|---|---|---|
|
||||
| Language | Python | Best subprocess/HTTP/threading support; already used in project |
|
||||
| Dependencies | stdlib only | No pip install needed; `urllib`, `subprocess`, `threading`, `json` |
|
||||
| Test framework | Standalone runner | Self-contained, no pytest dependency |
|
||||
| Relay strategy | Flexible genesis config | User provides their own test_genesis.jsonc |
|
||||
| LLM for agent | Real model | User provides API key in test genesis config |
|
||||
| Tool scope | All tools | Disposable test identity; full coverage |
|
||||
| Entry point | `tests/run_tests.py` | Single script to run everything |
|
||||
|
||||
---
|
||||
|
||||
## Directory Structure
|
||||
|
||||
```
|
||||
tests/
|
||||
├── harness/
|
||||
│ ├── __init__.py
|
||||
│ ├── agent_process.py # start/stop/restart didactyl subprocess
|
||||
│ ├── didactyl_client.py # HTTP API wrapper with timeouts
|
||||
│ ├── log_watcher.py # real-time log tail + marker system
|
||||
│ ├── test_runner.py # orchestrator
|
||||
│ └── reporter.py # results formatting (JSON + text)
|
||||
├── suites/
|
||||
│ ├── __init__.py
|
||||
│ ├── test_health.py # status, context endpoints
|
||||
│ ├── test_conversation.py # basic prompt/response
|
||||
│ ├── test_tools_identity.py # identity and context tools
|
||||
│ ├── test_tools_nostr.py # nostr event, messaging, relay tools
|
||||
│ ├── test_tools_skills.py # skill and trigger tools
|
||||
│ ├── test_tools_system.py # system, local, model, config tools
|
||||
│ ├── test_tools_memory.py # task and memory tools
|
||||
│ ├── test_tools_cashu.py # cashu wallet tools
|
||||
│ ├── test_tools_blossom.py # blossom tools
|
||||
│ ├── test_timeouts.py # response time assertions
|
||||
│ ├── test_errors.py # API error handling paths
|
||||
│ └── test_restart.py # crash recovery
|
||||
├── configs/
|
||||
│ └── test_genesis.jsonc # example test config (user fills in secrets)
|
||||
├── results/ # test run output (gitignored)
|
||||
├── run_tests.py # entry point
|
||||
├── test.sh # existing bash test (kept as-is)
|
||||
├── blossom_tool_validation_test.c # existing C test (kept as-is)
|
||||
└── blossom_tool_validation_test # existing compiled test (kept as-is)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Component Specifications
|
||||
|
||||
### 1. agent_process.py — Process Manager
|
||||
|
||||
Manages the didactyl subprocess lifecycle.
|
||||
|
||||
```python
|
||||
class AgentProcess:
|
||||
def __init__(self, binary_path, config_path, api_port=8484,
|
||||
api_bind="127.0.0.1", debug_level=5, log_file=None):
|
||||
"""Configure but don't start yet."""
|
||||
|
||||
def start(self, timeout=30) -> bool:
|
||||
"""
|
||||
Spawn didactyl as subprocess:
|
||||
./didactyl --config <config_path> --debug <level>
|
||||
--api-port <port> --api-bind <bind>
|
||||
|
||||
Set DIDACTYL_LOG_FILE env var to log_file path.
|
||||
Wait for GET /api/status to return 200 (poll with timeout).
|
||||
Returns True if agent started successfully.
|
||||
"""
|
||||
|
||||
def stop(self, timeout=10) -> bool:
|
||||
"""
|
||||
Send SIGTERM, wait for clean exit.
|
||||
If still alive after timeout, send SIGKILL.
|
||||
Returns True if stopped cleanly.
|
||||
"""
|
||||
|
||||
def restart(self, timeout=30) -> bool:
|
||||
"""stop() then start()."""
|
||||
|
||||
def is_alive(self) -> bool:
|
||||
"""Check process.poll() and optionally /api/status."""
|
||||
|
||||
def pid(self) -> int | None:
|
||||
"""Return PID if running."""
|
||||
|
||||
def return_code(self) -> int | None:
|
||||
"""Return exit code if stopped."""
|
||||
```
|
||||
|
||||
Key details:
|
||||
- Uses `subprocess.Popen` with `stdout=PIPE, stderr=PIPE`
|
||||
- Captures stdout/stderr for crash diagnostics
|
||||
- The health check polls `GET /api/status` every 500ms until success or timeout
|
||||
- Sets `DIDACTYL_LOG_FILE` to a test-run-specific path like `tests/results/<timestamp>/agent_debug.log`
|
||||
|
||||
### 2. didactyl_client.py — HTTP API Client
|
||||
|
||||
Thin wrapper around the Didactyl HTTP API using only `urllib`.
|
||||
|
||||
```python
|
||||
class DidactylClient:
|
||||
def __init__(self, base_url="https://127.0.0.1:8484", timeout=60,
|
||||
verify_tls=False):
|
||||
"""Configure base URL and default timeout."""
|
||||
|
||||
def status(self) -> dict:
|
||||
"""GET /api/status"""
|
||||
|
||||
def prompt(self, message: str, max_turns: int = 4,
|
||||
model: str = None) -> dict:
|
||||
"""POST /api/prompt/agent — full agent context conversation"""
|
||||
|
||||
def prompt_raw(self, messages: list, max_turns: int = 4,
|
||||
model: str = None) -> dict:
|
||||
"""POST /api/prompt/run — raw messages, no auto-context"""
|
||||
|
||||
def prompt_simple(self, system: str, user: str,
|
||||
model: str = None) -> dict:
|
||||
"""POST /api/prompt/run-simple — no tools"""
|
||||
|
||||
def context_current(self) -> dict:
|
||||
"""GET /api/context/current"""
|
||||
|
||||
def context_parts(self) -> dict:
|
||||
"""GET /api/context/parts"""
|
||||
|
||||
def fire_webhook(self, d_tag: str, payload: dict = None) -> dict:
|
||||
"""POST /api/trigger/<d_tag>"""
|
||||
```
|
||||
|
||||
Key details:
|
||||
- All methods return parsed JSON dict
|
||||
- Raises `TimeoutError` if response exceeds timeout
|
||||
- Raises `ConnectionError` if agent is unreachable
|
||||
- Raises `APIError(status_code, body)` for non-2xx responses
|
||||
- Uses `ssl._create_unverified_context()` for local TLS (same as chat CLI)
|
||||
|
||||
### 3. log_watcher.py — Real-time Log Monitor
|
||||
|
||||
Background thread that tails the agent's debug log file.
|
||||
|
||||
```python
|
||||
class LogWatcher:
|
||||
def __init__(self, log_path: str):
|
||||
"""Configure log file path."""
|
||||
|
||||
def start(self):
|
||||
"""
|
||||
Start background thread.
|
||||
Open file, seek to end, poll for new lines.
|
||||
Store all lines in memory with timestamps.
|
||||
"""
|
||||
|
||||
def stop(self):
|
||||
"""Stop background thread."""
|
||||
|
||||
def set_marker(self, name: str):
|
||||
"""Record current line count as a named marker."""
|
||||
|
||||
def get_lines_since(self, marker: str) -> list[str]:
|
||||
"""Return all lines captured since the named marker."""
|
||||
|
||||
def get_all_lines(self) -> list[str]:
|
||||
"""Return all captured lines."""
|
||||
|
||||
def search(self, pattern: str, since_marker: str = None) -> list[str]:
|
||||
"""Regex search through captured lines."""
|
||||
|
||||
def has_errors(self, since_marker: str = None) -> bool:
|
||||
"""Check for [ERROR] lines since marker."""
|
||||
|
||||
def has_warnings(self, since_marker: str = None) -> bool:
|
||||
"""Check for [WARN] lines since marker."""
|
||||
|
||||
def error_lines(self, since_marker: str = None) -> list[str]:
|
||||
"""Return all ERROR lines since marker."""
|
||||
```
|
||||
|
||||
Key details:
|
||||
- Uses `threading.Thread(daemon=True)` for background polling
|
||||
- Polls file every 100ms for new content
|
||||
- Handles file rotation (agent restart creates new file)
|
||||
- Thread-safe access to captured lines via `threading.Lock`
|
||||
|
||||
### 4. test_runner.py — Orchestrator
|
||||
|
||||
```python
|
||||
class TestResult:
|
||||
name: str
|
||||
suite: str
|
||||
status: str # "pass", "fail", "error", "skip", "timeout"
|
||||
message: str
|
||||
duration_seconds: float
|
||||
agent_errors: list[str] # ERROR lines from log during this test
|
||||
details: dict # arbitrary test-specific data
|
||||
|
||||
class TestCase:
|
||||
name: str
|
||||
description: str
|
||||
requires_restart: bool = False
|
||||
|
||||
def run(self, client: DidactylClient, log: LogWatcher) -> TestResult:
|
||||
"""Execute the test and return result."""
|
||||
|
||||
class TestRunner:
|
||||
def __init__(self, agent: AgentProcess, client: DidactylClient,
|
||||
log: LogWatcher):
|
||||
"""Configure with harness components."""
|
||||
|
||||
def discover_suites(self, suites_dir: str) -> list:
|
||||
"""Import all test_*.py modules from suites directory."""
|
||||
|
||||
def run_all(self, suites: list = None) -> list[TestResult]:
|
||||
"""
|
||||
For each suite:
|
||||
1. If test requires restart, restart agent
|
||||
2. Set log marker for this test
|
||||
3. Run test with timeout wrapper
|
||||
4. Capture result + any agent errors from log
|
||||
5. If agent crashed, restart and record error
|
||||
6. Collect all results
|
||||
Return list of TestResult.
|
||||
"""
|
||||
|
||||
def run_suite(self, suite_name: str) -> list[TestResult]:
|
||||
"""Run a single named suite."""
|
||||
```
|
||||
|
||||
Key details:
|
||||
- Each test gets a fresh log marker so errors can be correlated
|
||||
- If `client.prompt()` raises `TimeoutError`, the test is marked "timeout" and the agent is restarted
|
||||
- If `agent.is_alive()` returns False mid-suite, the agent is restarted and remaining tests continue
|
||||
- Supports filtering by suite name or test name via CLI args
|
||||
|
||||
### 5. reporter.py — Results Output
|
||||
|
||||
```python
|
||||
class Reporter:
|
||||
def __init__(self, results: list[TestResult], output_dir: str):
|
||||
"""Configure with results and output directory."""
|
||||
|
||||
def print_summary(self):
|
||||
"""Print pass/fail/error/skip/timeout counts to stdout."""
|
||||
|
||||
def print_details(self):
|
||||
"""Print each test result with details."""
|
||||
|
||||
def write_json(self, path: str):
|
||||
"""Write full results as JSON for programmatic consumption."""
|
||||
|
||||
def write_text(self, path: str):
|
||||
"""Write human-readable report."""
|
||||
```
|
||||
|
||||
Output format example:
|
||||
```
|
||||
== Didactyl Test Results ==
|
||||
Run: 2026-03-25T09:50:00Z
|
||||
Agent: v0.0.26
|
||||
Model: claude-haiku-4.5
|
||||
|
||||
Pass: 42
|
||||
Fail: 3
|
||||
Error: 1
|
||||
Timeout: 2
|
||||
Skip: 0
|
||||
Total: 48
|
||||
|
||||
-- Failures --
|
||||
[FAIL] test_tools_nostr::nostr_post_kind1
|
||||
Expected tool_calls to contain 'nostr_post', got: ['nostr_query']
|
||||
Agent errors during test: 0
|
||||
|
||||
[FAIL] test_conversation::multi_turn
|
||||
Agent returned empty final_response
|
||||
Agent errors during test: 2
|
||||
[ERROR] [llm.c:234] HTTP 429 rate limited
|
||||
[ERROR] [llm.c:240] LLM call failed after 3 retries
|
||||
|
||||
[TIMEOUT] test_tools_skills::skill_create
|
||||
No response within 60s
|
||||
Agent was restarted
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Test Suite Specifications
|
||||
|
||||
### test_health.py
|
||||
|
||||
| Test | Prompt/Action | Assertions |
|
||||
|---|---|---|
|
||||
| `status_returns_200` | `GET /api/status` | HTTP 200, `success=true` |
|
||||
| `status_has_fields` | `GET /api/status` | Has `name`, `version`, `pubkey`, `relay_count` |
|
||||
| `context_current_returns_messages` | `GET /api/context/current` | Has `messages` array, `total_chars > 0` |
|
||||
| `context_parts_has_system_prompt` | `GET /api/context/parts` | Has part named `system_prompt` |
|
||||
|
||||
### test_conversation.py
|
||||
|
||||
| Test | Prompt | Assertions |
|
||||
|---|---|---|
|
||||
| `simple_greeting` | "Hello, what is your name?" | `final_response` is non-empty string |
|
||||
| `agent_responds_about_itself` | "What are you? Describe yourself briefly." | `final_response` mentions agent/Didactyl/Nostr |
|
||||
| `empty_message_handling` | "" (empty string) | Returns error or handles gracefully |
|
||||
| `very_long_message` | 10000 char string | Returns response or graceful error, no crash |
|
||||
|
||||
### test_tools_identity.py
|
||||
|
||||
| Test | Prompt | Expected Tool | Assertions |
|
||||
|---|---|---|---|
|
||||
| `get_pubkey` | "What is your public key in hex?" | `nostr_pubkey` or `my_pubkey` | Tool called, result has hex pubkey |
|
||||
| `get_npub` | "What is your npub?" | `nostr_npub` or `my_npub` | Tool called, result has npub1... |
|
||||
| `agent_identity` | "Tell me about your identity" | `agent_identity` | Tool called, success |
|
||||
| `agent_version` | "What version are you?" | `agent_version` | Tool called, result has version string |
|
||||
| `admin_identity` | "Who is your administrator?" | `admin_identity` | Tool called, success |
|
||||
|
||||
### test_tools_nostr.py
|
||||
|
||||
| Test | Prompt | Expected Tool | Assertions |
|
||||
|---|---|---|---|
|
||||
| `nostr_post_kind1` | "Post a test note saying 'Automated test post'" | `nostr_post` | Tool called with kind=1, success, event_id returned |
|
||||
| `nostr_query_recent` | "Query the 3 most recent kind 1 notes from any author" | `nostr_query` | Tool called, returns events array |
|
||||
| `nostr_my_events` | "List your recent events" | `nostr_my_events` | Tool called, success |
|
||||
| `nostr_relay_status` | "What is the status of your relay connections?" | `nostr_relay_status` | Tool called, returns relay info |
|
||||
| `nostr_dm_send` | "Send a test DM to yourself" | `nostr_dm_send` | Tool called, success |
|
||||
| `nostr_encode_npub` | "Encode your pubkey as an npub" | `nostr_encode` | Tool called, returns npub |
|
||||
| `nostr_profile_get` | "Look up your own Nostr profile" | `nostr_profile_get` | Tool called, returns profile |
|
||||
|
||||
### test_tools_skills.py
|
||||
|
||||
| Test | Prompt | Expected Tool | Assertions |
|
||||
|---|---|---|---|
|
||||
| `skill_list` | "List your available skills" | `skill_list` | Tool called, returns skills array |
|
||||
| `trigger_list` | "List your active triggers" | `trigger_list` | Tool called, success |
|
||||
| `skill_create_and_remove` | "Create a test skill called 'test-harness-probe' with content 'Test skill' then remove it" | `skill_create`, `skill_remove` | Both tools called, success |
|
||||
|
||||
### test_tools_system.py
|
||||
|
||||
| Test | Prompt | Expected Tool | Assertions |
|
||||
|---|---|---|---|
|
||||
| `tool_list` | "List all your available tools" | `tool_list` | Tool called, returns tools array |
|
||||
| `model_get` | "What model are you currently using?" | `model_get` | Tool called, returns model info |
|
||||
| `model_list` | "List available models" | `model_list` | Tool called, success |
|
||||
| `local_http_fetch` | "Fetch https://httpbin.org/get" | `local_http_fetch` | Tool called, returns HTTP response |
|
||||
| `config_store_recall` | "Store a test config with d_tag 'test_harness_probe' containing 'hello', then recall it" | `config_store`, `config_recall` | Both called, recalled value matches |
|
||||
|
||||
### test_tools_memory.py
|
||||
|
||||
| Test | Prompt | Expected Tool | Assertions |
|
||||
|---|---|---|---|
|
||||
| `task_list` | "Show me your current task list" | `task_list` or `task_manage` | Tool called, success |
|
||||
| `task_manage_add_remove` | "Add a task 'test harness probe task' then remove it" | `task_manage` | Tool called with add then remove |
|
||||
| `memory_save_recall` | "Save 'test harness probe' to memory, then recall your memory" | `memory_save`, `memory_recall` | Both called, recalled contains probe text |
|
||||
|
||||
### test_tools_cashu.py
|
||||
|
||||
| Test | Prompt | Expected Tool | Assertions |
|
||||
|---|---|---|---|
|
||||
| `wallet_balance` | "Check your cashu wallet balance" | `cashu_wallet_balance` | Tool called, success (even if empty) |
|
||||
|
||||
Note: Most cashu tools require a configured mint and funded wallet. Phase 1 tests only the read-only balance check. Full cashu testing requires a test mint setup.
|
||||
|
||||
### test_tools_blossom.py
|
||||
|
||||
| Test | Prompt | Expected Tool | Assertions |
|
||||
|---|---|---|---|
|
||||
| `blossom_list` | "List your blossom blobs" | `blossom_list` | Tool called, success (even if empty) |
|
||||
|
||||
Note: Full blossom testing requires a configured blossom server. Phase 1 tests only the list operation.
|
||||
|
||||
### test_timeouts.py
|
||||
|
||||
| Test | Action | Assertions |
|
||||
|---|---|---|
|
||||
| `response_within_timeout` | Send simple prompt, measure time | Response received within 60s |
|
||||
| `status_responds_fast` | `GET /api/status`, measure time | Response within 2s |
|
||||
| `context_responds_fast` | `GET /api/context/current`, measure time | Response within 5s |
|
||||
|
||||
### test_errors.py
|
||||
|
||||
| Test | Action | Assertions |
|
||||
|---|---|---|
|
||||
| `invalid_json_body` | POST malformed JSON to `/api/prompt/agent` | Returns 400, `success=false` |
|
||||
| `missing_message_field` | POST `{}` to `/api/prompt/agent` | Returns 400, `success=false` |
|
||||
| `unknown_endpoint` | GET `/api/nonexistent` | Returns 404 |
|
||||
| `webhook_nonexistent_dtag` | POST to `/api/trigger/nonexistent-dtag` | Returns 404 |
|
||||
|
||||
### test_restart.py
|
||||
|
||||
| Test | Action | Assertions |
|
||||
|---|---|---|
|
||||
| `clean_restart` | Stop agent, start agent | Agent comes back, `/api/status` works |
|
||||
| `status_after_restart` | Restart, then `GET /api/status` | Same pubkey, version as before restart |
|
||||
| `conversation_after_restart` | Restart, then send prompt | Agent responds normally |
|
||||
|
||||
---
|
||||
|
||||
## Entry Point: run_tests.py
|
||||
|
||||
```
|
||||
Usage:
|
||||
python tests/run_tests.py [options]
|
||||
|
||||
Options:
|
||||
--config PATH Path to test genesis.jsonc (default: tests/configs/test_genesis.jsonc)
|
||||
--binary PATH Path to didactyl binary (default: ./didactyl)
|
||||
--suite NAME Run only this suite (can repeat)
|
||||
--test NAME Run only this test (can repeat)
|
||||
--api-port PORT API port (default: 8485)
|
||||
--timeout SECS Default response timeout (default: 60)
|
||||
--debug-level N Agent debug level 0-5 (default: 5)
|
||||
--output-dir PATH Results output directory (default: tests/results/<timestamp>)
|
||||
--verbose Print each test result as it runs
|
||||
--no-restart Don't auto-restart agent on crash (fail remaining tests)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Test Genesis Config
|
||||
|
||||
`tests/configs/test_genesis.jsonc` — an example config the user fills in:
|
||||
|
||||
```jsonc
|
||||
{
|
||||
// TEST CONFIGURATION — fill in your test identity secrets
|
||||
|
||||
"key": {
|
||||
"nsec": "nsec1REPLACE_WITH_DISPOSABLE_TEST_NSEC"
|
||||
},
|
||||
|
||||
"admin": {
|
||||
"pubkey": "npub1REPLACE_WITH_TEST_ADMIN_PUBKEY"
|
||||
},
|
||||
|
||||
"dm_protocol": "nip04",
|
||||
|
||||
"llm": {
|
||||
"provider": "openai",
|
||||
"api_key": "sk-REPLACE_WITH_API_KEY",
|
||||
"model": "claude-haiku-4.5",
|
||||
"base_url": "https://api.anthropic.com/v1",
|
||||
"max_tokens": 512,
|
||||
"temperature": 0.3
|
||||
},
|
||||
|
||||
"api": {
|
||||
"enabled": true,
|
||||
"port": 8485,
|
||||
"bind_address": "127.0.0.1"
|
||||
},
|
||||
|
||||
"startup_events": [
|
||||
{
|
||||
"kind": 0,
|
||||
"content_fields": {
|
||||
"name": "Didactyl Test Agent",
|
||||
"about": "Automated test instance"
|
||||
},
|
||||
"tags": []
|
||||
},
|
||||
{
|
||||
"kind": 10002,
|
||||
"content": "",
|
||||
"tags": [
|
||||
["r", "wss://relay.damus.io"],
|
||||
["r", "wss://relay.primal.net"]
|
||||
]
|
||||
},
|
||||
{
|
||||
"kind": 31124,
|
||||
"content": "# Test Agent\n\nYou are a test agent. Respond to all requests. Use tools when asked.\n\n{{my_kind0_profile}}\n\nYour npub: {{my_npub}}",
|
||||
"tags": [
|
||||
["d", "identity_and_rules"],
|
||||
["app", "didactyl"],
|
||||
["scope", "private"],
|
||||
["trigger", "dm"],
|
||||
["filter", "{\"from\":\"admin\"}"]
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Execution Flow
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
START[run_tests.py] --> PARSE[Parse CLI args]
|
||||
PARSE --> BUILD[Verify didactyl binary exists]
|
||||
BUILD --> CONFIG[Load test genesis config]
|
||||
CONFIG --> MKDIR[Create results output dir]
|
||||
MKDIR --> LOG_START[Start LogWatcher]
|
||||
LOG_START --> AGENT_START[Start AgentProcess]
|
||||
AGENT_START --> HEALTH[Wait for /api/status 200]
|
||||
|
||||
HEALTH -->|timeout| FAIL_STARTUP[Report startup failure and exit]
|
||||
HEALTH -->|success| DISCOVER[Discover test suites]
|
||||
|
||||
DISCOVER --> LOOP{Next test?}
|
||||
|
||||
LOOP -->|yes| CHECK_ALIVE{Agent alive?}
|
||||
CHECK_ALIVE -->|no| RESTART_MID[Restart agent]
|
||||
RESTART_MID --> CHECK_ALIVE
|
||||
CHECK_ALIVE -->|yes| MARKER[Set log marker]
|
||||
MARKER --> RUN_TEST[Run test with timeout]
|
||||
|
||||
RUN_TEST -->|pass| RECORD_PASS[Record PASS]
|
||||
RUN_TEST -->|fail| RECORD_FAIL[Record FAIL]
|
||||
RUN_TEST -->|timeout| RESTART_TIMEOUT[Restart agent]
|
||||
RESTART_TIMEOUT --> RECORD_TIMEOUT[Record TIMEOUT]
|
||||
RUN_TEST -->|error| RECORD_ERROR[Record ERROR]
|
||||
|
||||
RECORD_PASS & RECORD_FAIL & RECORD_TIMEOUT & RECORD_ERROR --> COLLECT_LOGS[Collect agent errors since marker]
|
||||
COLLECT_LOGS --> LOOP
|
||||
|
||||
LOOP -->|no more| STOP_AGENT[Stop agent]
|
||||
STOP_AGENT --> STOP_LOG[Stop LogWatcher]
|
||||
STOP_LOG --> REPORT[Generate report]
|
||||
REPORT --> EXIT[Exit with code 0 if all pass else 1]
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Implementation Order
|
||||
|
||||
1. **Harness infrastructure** (agent_process, didactyl_client, log_watcher)
|
||||
2. **Test runner + reporter** (orchestration layer)
|
||||
3. **test_health.py** (validates the harness itself works)
|
||||
4. **test_conversation.py** (validates basic agent interaction)
|
||||
5. **test_errors.py** (validates error handling)
|
||||
6. **test_timeouts.py** (validates timeout detection)
|
||||
7. **test_restart.py** (validates process management)
|
||||
8. **Tool test suites** (one at a time, identity → nostr → skills → system → memory → cashu → blossom)
|
||||
9. **Entry point + config** (run_tests.py, test_genesis.jsonc)
|
||||
10. **End-to-end validation** against a running agent
|
||||
|
||||
---
|
||||
|
||||
## Future: Phase 2 — LLM-Driven Testing
|
||||
|
||||
The architecture supports this naturally. In Phase 2:
|
||||
|
||||
- Add an `LLMTestAgent` class that wraps an LLM API call
|
||||
- The LLM receives: tool documentation, test objectives, previous results
|
||||
- It generates test prompts, evaluates responses, decides next actions
|
||||
- The `TestCase.run()` method delegates to the LLM agent instead of scripted logic
|
||||
- The LLM can also analyze agent logs for anomalies
|
||||
- Potentially: the LLM can generate code patches for bugs it finds
|
||||
|
||||
No architectural changes needed — just a new test suite type that uses LLM reasoning instead of scripted assertions.
|
||||
@@ -12,8 +12,8 @@
|
||||
// Using DIDACTYL_ prefix to avoid conflicts with nostr_core_lib VERSION macros
|
||||
#define DIDACTYL_VERSION_MAJOR 0
|
||||
#define DIDACTYL_VERSION_MINOR 2
|
||||
#define DIDACTYL_VERSION_PATCH 14
|
||||
#define DIDACTYL_VERSION "v0.2.14"
|
||||
#define DIDACTYL_VERSION_PATCH 16
|
||||
#define DIDACTYL_VERSION "v0.2.16"
|
||||
|
||||
// Agent metadata
|
||||
#define DIDACTYL_NAME "Didactyl"
|
||||
|
||||
BIN
tests/__pycache__/run_tests.cpython-313.pyc
Normal file
BIN
tests/__pycache__/run_tests.cpython-313.pyc
Normal file
Binary file not shown.
53
tests/configs/test_genesis.jsonc
Normal file
53
tests/configs/test_genesis.jsonc
Normal file
@@ -0,0 +1,53 @@
|
||||
{
|
||||
// TEST CONFIGURATION
|
||||
// Use disposable keys/accounts only.
|
||||
"key": {
|
||||
"nsec": "nsec1REPLACE_WITH_DISPOSABLE_TEST_NSEC"
|
||||
},
|
||||
"admin": {
|
||||
"pubkey": "npub1REPLACE_WITH_TEST_ADMIN_PUBKEY"
|
||||
},
|
||||
"dm_protocol": "nip04",
|
||||
"llm": {
|
||||
"provider": "openai",
|
||||
"api_key": "sk-REPLACE_WITH_API_KEY",
|
||||
"model": "claude-haiku-4.5",
|
||||
"base_url": "https://api.anthropic.com/v1",
|
||||
"max_tokens": 512,
|
||||
"temperature": 0.3
|
||||
},
|
||||
"api": {
|
||||
"enabled": true,
|
||||
"port": 8485,
|
||||
"bind_address": "127.0.0.1"
|
||||
},
|
||||
"startup_events": [
|
||||
{
|
||||
"kind": 0,
|
||||
"content_fields": {
|
||||
"name": "Didactyl Test Agent",
|
||||
"about": "Automated test instance"
|
||||
},
|
||||
"tags": []
|
||||
},
|
||||
{
|
||||
"kind": 10002,
|
||||
"content": "",
|
||||
"tags": [
|
||||
["r", "wss://relay.damus.io"],
|
||||
["r", "wss://relay.primal.net"]
|
||||
]
|
||||
},
|
||||
{
|
||||
"kind": 31124,
|
||||
"content": "# Test Agent\n\nYou are a test agent. Respond to requests and use tools when needed.",
|
||||
"tags": [
|
||||
["d", "identity_and_rules"],
|
||||
["app", "didactyl"],
|
||||
["scope", "private"],
|
||||
["trigger", "dm"],
|
||||
["filter", "{\"from\":\"admin\"}"]
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
26
tests/harness/__init__.py
Normal file
26
tests/harness/__init__.py
Normal file
@@ -0,0 +1,26 @@
|
||||
"""Didactyl automated test harness package."""
|
||||
|
||||
from .agent_process import AgentProcess
|
||||
from .didactyl_client import (
|
||||
APIError,
|
||||
DidactylClient,
|
||||
DidactylConnectionError,
|
||||
DidactylTimeoutError,
|
||||
)
|
||||
from .log_watcher import LogWatcher
|
||||
from .reporter import Reporter
|
||||
from .test_runner import SkipTest, TestCase, TestResult, TestRunner
|
||||
|
||||
__all__ = [
|
||||
"AgentProcess",
|
||||
"APIError",
|
||||
"DidactylClient",
|
||||
"DidactylConnectionError",
|
||||
"DidactylTimeoutError",
|
||||
"LogWatcher",
|
||||
"Reporter",
|
||||
"SkipTest",
|
||||
"TestCase",
|
||||
"TestResult",
|
||||
"TestRunner",
|
||||
]
|
||||
BIN
tests/harness/__pycache__/__init__.cpython-313.pyc
Normal file
BIN
tests/harness/__pycache__/__init__.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/harness/__pycache__/agent_process.cpython-313.pyc
Normal file
BIN
tests/harness/__pycache__/agent_process.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/harness/__pycache__/didactyl_client.cpython-313.pyc
Normal file
BIN
tests/harness/__pycache__/didactyl_client.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/harness/__pycache__/log_watcher.cpython-313.pyc
Normal file
BIN
tests/harness/__pycache__/log_watcher.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/harness/__pycache__/reporter.cpython-313.pyc
Normal file
BIN
tests/harness/__pycache__/reporter.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/harness/__pycache__/test_runner.cpython-313.pyc
Normal file
BIN
tests/harness/__pycache__/test_runner.cpython-313.pyc
Normal file
Binary file not shown.
113
tests/harness/agent_process.py
Normal file
113
tests/harness/agent_process.py
Normal file
@@ -0,0 +1,113 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import signal
|
||||
import subprocess
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from .didactyl_client import DidactylClient
|
||||
|
||||
|
||||
@dataclass
|
||||
class AgentProcess:
|
||||
binary_path: str
|
||||
config_path: str
|
||||
api_port: int = 8485
|
||||
api_bind: str = "127.0.0.1"
|
||||
debug_level: int = 5
|
||||
log_file: Optional[str] = None
|
||||
base_url: Optional[str] = None
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
self.binary_path = str(Path(self.binary_path))
|
||||
self.config_path = str(Path(self.config_path))
|
||||
self.log_file = self.log_file or "tests/results/agent_debug.log"
|
||||
scheme = "https"
|
||||
self.base_url = self.base_url or f"{scheme}://{self.api_bind}:{self.api_port}"
|
||||
self.process: Optional[subprocess.Popen[str]] = None
|
||||
|
||||
def _command(self) -> list[str]:
|
||||
return [
|
||||
self.binary_path,
|
||||
"--config",
|
||||
self.config_path,
|
||||
"--debug",
|
||||
str(self.debug_level),
|
||||
"--api-port",
|
||||
str(self.api_port),
|
||||
"--api-bind",
|
||||
self.api_bind,
|
||||
]
|
||||
|
||||
def start(self, timeout: float = 30.0) -> bool:
|
||||
if self.is_alive():
|
||||
return True
|
||||
|
||||
env = os.environ.copy()
|
||||
env["DIDACTYL_LOG_FILE"] = str(self.log_file)
|
||||
|
||||
Path(self.log_file).parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
self.process = subprocess.Popen(
|
||||
self._command(),
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
text=True,
|
||||
env=env,
|
||||
)
|
||||
|
||||
client = DidactylClient(base_url=self.base_url, timeout=2.0, verify_tls=False)
|
||||
deadline = time.time() + timeout
|
||||
while time.time() < deadline:
|
||||
if self.process and self.process.poll() is not None:
|
||||
return False
|
||||
try:
|
||||
data = client.status()
|
||||
if data.get("success"):
|
||||
return True
|
||||
except Exception:
|
||||
pass
|
||||
time.sleep(0.5)
|
||||
return False
|
||||
|
||||
def stop(self, timeout: float = 10.0) -> bool:
|
||||
if not self.process:
|
||||
return True
|
||||
if self.process.poll() is not None:
|
||||
return True
|
||||
|
||||
try:
|
||||
self.process.send_signal(signal.SIGTERM)
|
||||
self.process.wait(timeout=timeout)
|
||||
return True
|
||||
except subprocess.TimeoutExpired:
|
||||
self.process.kill()
|
||||
self.process.wait(timeout=5)
|
||||
return False
|
||||
|
||||
def restart(self, timeout: float = 30.0) -> bool:
|
||||
self.stop()
|
||||
return self.start(timeout=timeout)
|
||||
|
||||
def is_alive(self) -> bool:
|
||||
return self.process is not None and self.process.poll() is None
|
||||
|
||||
def pid(self) -> Optional[int]:
|
||||
return None if not self.process else self.process.pid
|
||||
|
||||
def return_code(self) -> Optional[int]:
|
||||
return None if not self.process else self.process.poll()
|
||||
|
||||
def read_pipes(self) -> tuple[str, str]:
|
||||
if not self.process:
|
||||
return "", ""
|
||||
out = ""
|
||||
err = ""
|
||||
if self.process.stdout:
|
||||
out = self.process.stdout.read() or ""
|
||||
if self.process.stderr:
|
||||
err = self.process.stderr.read() or ""
|
||||
return out, err
|
||||
104
tests/harness/didactyl_client.py
Normal file
104
tests/harness/didactyl_client.py
Normal file
@@ -0,0 +1,104 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import ssl
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Optional
|
||||
|
||||
|
||||
class DidactylTimeoutError(TimeoutError):
|
||||
pass
|
||||
|
||||
|
||||
class DidactylConnectionError(ConnectionError):
|
||||
pass
|
||||
|
||||
|
||||
@dataclass
|
||||
class APIError(Exception):
|
||||
status_code: int
|
||||
body: str
|
||||
|
||||
def __str__(self) -> str:
|
||||
return f"HTTP {self.status_code}: {self.body}"
|
||||
|
||||
|
||||
class DidactylClient:
|
||||
def __init__(self, base_url: str = "https://127.0.0.1:8485", timeout: float = 60.0, verify_tls: bool = False) -> None:
|
||||
self.base_url = base_url.rstrip("/")
|
||||
self.timeout = timeout
|
||||
self.verify_tls = verify_tls
|
||||
self.ssl_context = None
|
||||
if self.base_url.startswith("https://") and not verify_tls:
|
||||
self.ssl_context = ssl._create_unverified_context()
|
||||
|
||||
def _request(self, method: str, path: str, payload: Optional[dict[str, Any]] = None, raw_body: Optional[bytes] = None) -> dict[str, Any]:
|
||||
url = f"{self.base_url}{path}"
|
||||
body = raw_body
|
||||
if payload is not None:
|
||||
body = json.dumps(payload).encode("utf-8")
|
||||
req = urllib.request.Request(url=url, method=method.upper(), data=body)
|
||||
req.add_header("Content-Type", "application/json")
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=self.timeout, context=self.ssl_context) as resp:
|
||||
status = resp.getcode()
|
||||
text = resp.read().decode("utf-8", errors="replace")
|
||||
data = json.loads(text) if text.strip() else {}
|
||||
if status < 200 or status >= 300:
|
||||
raise APIError(status, text)
|
||||
return data
|
||||
except urllib.error.HTTPError as e:
|
||||
text = e.read().decode("utf-8", errors="replace")
|
||||
raise APIError(e.code, text) from e
|
||||
except urllib.error.URLError as e:
|
||||
reason = str(getattr(e, "reason", e))
|
||||
if "timed out" in reason.lower():
|
||||
raise DidactylTimeoutError(reason) from e
|
||||
raise DidactylConnectionError(reason) from e
|
||||
except TimeoutError as e:
|
||||
raise DidactylTimeoutError(str(e)) from e
|
||||
|
||||
def status(self) -> dict[str, Any]:
|
||||
return self._request("GET", "/api/status")
|
||||
|
||||
def context_current(self) -> dict[str, Any]:
|
||||
return self._request("GET", "/api/context/current")
|
||||
|
||||
def context_parts(self) -> dict[str, Any]:
|
||||
return self._request("GET", "/api/context/parts")
|
||||
|
||||
def prompt(self, message: str, max_turns: int = 4, model: Optional[str] = None) -> dict[str, Any]:
|
||||
payload: dict[str, Any] = {"message": message, "max_turns": max_turns}
|
||||
if model:
|
||||
payload["model"] = model
|
||||
return self._request("POST", "/api/prompt/agent", payload=payload)
|
||||
|
||||
def prompt_raw(self, messages: list[dict[str, str]], max_turns: int = 4, model: Optional[str] = None) -> dict[str, Any]:
|
||||
payload: dict[str, Any] = {"messages": messages, "max_turns": max_turns}
|
||||
if model:
|
||||
payload["model"] = model
|
||||
return self._request("POST", "/api/prompt/run", payload=payload)
|
||||
|
||||
def prompt_simple(self, system: str, user: str, model: Optional[str] = None) -> dict[str, Any]:
|
||||
payload: dict[str, Any] = {"system": system, "user": user}
|
||||
if model:
|
||||
payload["model"] = model
|
||||
return self._request("POST", "/api/prompt/run-simple", payload=payload)
|
||||
|
||||
def fire_webhook(self, d_tag: str, payload: Optional[dict[str, Any]] = None) -> dict[str, Any]:
|
||||
encoded = urllib.parse.quote(d_tag, safe="")
|
||||
return self._request("POST", f"/api/trigger/{encoded}", payload=payload or {})
|
||||
|
||||
def raw_post(self, path: str, body: bytes) -> tuple[int, str]:
|
||||
url = f"{self.base_url}{path}"
|
||||
req = urllib.request.Request(url=url, method="POST", data=body)
|
||||
req.add_header("Content-Type", "application/json")
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=self.timeout, context=self.ssl_context) as resp:
|
||||
return resp.getcode(), resp.read().decode("utf-8", errors="replace")
|
||||
except urllib.error.HTTPError as e:
|
||||
return e.code, e.read().decode("utf-8", errors="replace")
|
||||
77
tests/harness/log_watcher.py
Normal file
77
tests/harness/log_watcher.py
Normal file
@@ -0,0 +1,77 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import threading
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
|
||||
class LogWatcher:
|
||||
def __init__(self, log_path: str, poll_interval: float = 0.1) -> None:
|
||||
self.log_path = Path(log_path)
|
||||
self.poll_interval = poll_interval
|
||||
self._lines: list[str] = []
|
||||
self._markers: dict[str, int] = {}
|
||||
self._lock = threading.Lock()
|
||||
self._stop = threading.Event()
|
||||
self._thread: Optional[threading.Thread] = None
|
||||
|
||||
def start(self) -> None:
|
||||
self.log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
self.log_path.touch(exist_ok=True)
|
||||
self._stop.clear()
|
||||
self._thread = threading.Thread(target=self._run, daemon=True)
|
||||
self._thread.start()
|
||||
|
||||
def stop(self) -> None:
|
||||
self._stop.set()
|
||||
if self._thread and self._thread.is_alive():
|
||||
self._thread.join(timeout=2)
|
||||
|
||||
def _run(self) -> None:
|
||||
f = self.log_path.open("r", encoding="utf-8", errors="replace")
|
||||
f.seek(0, 2)
|
||||
try:
|
||||
while not self._stop.is_set():
|
||||
pos = f.tell()
|
||||
line = f.readline()
|
||||
if not line:
|
||||
if self.log_path.exists() and self.log_path.stat().st_size < pos:
|
||||
f.close()
|
||||
f = self.log_path.open("r", encoding="utf-8", errors="replace")
|
||||
time.sleep(self.poll_interval)
|
||||
continue
|
||||
with self._lock:
|
||||
self._lines.append(line.rstrip("\n"))
|
||||
finally:
|
||||
f.close()
|
||||
|
||||
def set_marker(self, name: str) -> None:
|
||||
with self._lock:
|
||||
self._markers[name] = len(self._lines)
|
||||
|
||||
def get_lines_since(self, marker: str) -> list[str]:
|
||||
with self._lock:
|
||||
idx = self._markers.get(marker, 0)
|
||||
return list(self._lines[idx:])
|
||||
|
||||
def get_all_lines(self) -> list[str]:
|
||||
with self._lock:
|
||||
return list(self._lines)
|
||||
|
||||
def search(self, pattern: str, since_marker: Optional[str] = None) -> list[str]:
|
||||
regex = re.compile(pattern)
|
||||
lines = self.get_all_lines() if since_marker is None else self.get_lines_since(since_marker)
|
||||
return [line for line in lines if regex.search(line)]
|
||||
|
||||
def error_lines(self, since_marker: Optional[str] = None) -> list[str]:
|
||||
lines = self.get_all_lines() if since_marker is None else self.get_lines_since(since_marker)
|
||||
return [line for line in lines if "[ERROR]" in line]
|
||||
|
||||
def warning_lines(self, since_marker: Optional[str] = None) -> list[str]:
|
||||
lines = self.get_all_lines() if since_marker is None else self.get_lines_since(since_marker)
|
||||
return [line for line in lines if "[WARN" in line]
|
||||
|
||||
def has_errors(self, since_marker: Optional[str] = None) -> bool:
|
||||
return len(self.error_lines(since_marker)) > 0
|
||||
77
tests/harness/reporter.py
Normal file
77
tests/harness/reporter.py
Normal file
@@ -0,0 +1,77 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections import Counter
|
||||
from dataclasses import asdict
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from .test_runner import TestResult
|
||||
|
||||
|
||||
class Reporter:
|
||||
def __init__(self, results: list[TestResult], run_meta: dict[str, Any], output_dir: str) -> None:
|
||||
self.results = results
|
||||
self.run_meta = run_meta
|
||||
self.output_dir = Path(output_dir)
|
||||
self.output_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
def summary_counts(self) -> Counter:
|
||||
return Counter(r.status for r in self.results)
|
||||
|
||||
def print_summary(self) -> None:
|
||||
c = self.summary_counts()
|
||||
total = len(self.results)
|
||||
print("\n== Didactyl Test Results ==")
|
||||
print(f"Run: {self.run_meta.get('run_timestamp', 'unknown')}")
|
||||
print(f"Agent base URL: {self.run_meta.get('base_url', 'unknown')}")
|
||||
print("")
|
||||
for k in ["pass", "fail", "error", "timeout", "skip"]:
|
||||
print(f"{k.capitalize():<8}: {c.get(k, 0)}")
|
||||
print(f"Total : {total}")
|
||||
|
||||
def print_details(self) -> None:
|
||||
bad = [r for r in self.results if r.status in {"fail", "error", "timeout"}]
|
||||
if not bad:
|
||||
return
|
||||
print("\n-- Non-passing tests --")
|
||||
for r in bad:
|
||||
print(f"[{r.status.upper()}] {r.suite}::{r.name}")
|
||||
print(f" {r.message}")
|
||||
if r.agent_errors:
|
||||
print(f" Agent errors: {len(r.agent_errors)}")
|
||||
|
||||
def write_json(self, path: str = "results.json") -> Path:
|
||||
p = self.output_dir / path
|
||||
payload = {
|
||||
"run_meta": self.run_meta,
|
||||
"results": [asdict(r) for r in self.results],
|
||||
"summary": dict(self.summary_counts()),
|
||||
}
|
||||
p.write_text(json.dumps(payload, indent=2), encoding="utf-8")
|
||||
return p
|
||||
|
||||
def write_text(self, path: str = "results.txt") -> Path:
|
||||
p = self.output_dir / path
|
||||
c = self.summary_counts()
|
||||
lines = [
|
||||
"== Didactyl Test Results ==",
|
||||
f"Run: {self.run_meta.get('run_timestamp', 'unknown')}",
|
||||
f"Agent base URL: {self.run_meta.get('base_url', 'unknown')}",
|
||||
"",
|
||||
f"Pass: {c.get('pass', 0)}",
|
||||
f"Fail: {c.get('fail', 0)}",
|
||||
f"Error: {c.get('error', 0)}",
|
||||
f"Timeout: {c.get('timeout', 0)}",
|
||||
f"Skip: {c.get('skip', 0)}",
|
||||
f"Total: {len(self.results)}",
|
||||
"",
|
||||
]
|
||||
for r in self.results:
|
||||
lines.append(f"[{r.status.upper()}] {r.suite}::{r.name} ({r.duration_seconds:.2f}s)")
|
||||
lines.append(f" {r.message}")
|
||||
if r.agent_errors:
|
||||
lines.append(f" Agent errors: {len(r.agent_errors)}")
|
||||
lines.append("")
|
||||
p.write_text("\n".join(lines), encoding="utf-8")
|
||||
return p
|
||||
141
tests/harness/test_runner.py
Normal file
141
tests/harness/test_runner.py
Normal file
@@ -0,0 +1,141 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import pkgutil
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from types import ModuleType
|
||||
from typing import Any, Callable
|
||||
|
||||
from .agent_process import AgentProcess
|
||||
from .didactyl_client import DidactylClient, DidactylTimeoutError
|
||||
from .log_watcher import LogWatcher
|
||||
|
||||
|
||||
class SkipTest(Exception):
|
||||
pass
|
||||
|
||||
|
||||
TestFn = Callable[["TestContext"], tuple[bool, str, dict[str, Any]]]
|
||||
|
||||
|
||||
@dataclass
|
||||
class TestCase:
|
||||
suite: str
|
||||
name: str
|
||||
description: str
|
||||
fn: TestFn
|
||||
requires_restart: bool = False
|
||||
|
||||
|
||||
@dataclass
|
||||
class TestResult:
|
||||
suite: str
|
||||
name: str
|
||||
status: str
|
||||
message: str
|
||||
duration_seconds: float
|
||||
agent_errors: list[str] = field(default_factory=list)
|
||||
details: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass
|
||||
class TestContext:
|
||||
client: DidactylClient
|
||||
agent: AgentProcess
|
||||
log: LogWatcher
|
||||
args: Any
|
||||
|
||||
|
||||
class TestRunner:
|
||||
def __init__(self, agent: AgentProcess, client: DidactylClient, log: LogWatcher, args: Any) -> None:
|
||||
self.agent = agent
|
||||
self.client = client
|
||||
self.log = log
|
||||
self.args = args
|
||||
|
||||
def discover_suites(self, package: str = "tests.suites") -> list[TestCase]:
|
||||
mod = importlib.import_module(package)
|
||||
tests: list[TestCase] = []
|
||||
for info in pkgutil.iter_modules(mod.__path__):
|
||||
if not info.name.startswith("test_"):
|
||||
continue
|
||||
if self.args.suite and info.name not in self.args.suite:
|
||||
continue
|
||||
module = importlib.import_module(f"{package}.{info.name}")
|
||||
tests.extend(self._tests_from_module(module))
|
||||
return tests
|
||||
|
||||
def _tests_from_module(self, module: ModuleType) -> list[TestCase]:
|
||||
if not hasattr(module, "get_tests"):
|
||||
return []
|
||||
suite_tests = module.get_tests()
|
||||
out: list[TestCase] = []
|
||||
for t in suite_tests:
|
||||
if self.args.test and t.name not in self.args.test:
|
||||
continue
|
||||
out.append(t)
|
||||
return out
|
||||
|
||||
def run_all(self, tests: list[TestCase]) -> list[TestResult]:
|
||||
results: list[TestResult] = []
|
||||
ctx = TestContext(client=self.client, agent=self.agent, log=self.log, args=self.args)
|
||||
|
||||
for tc in tests:
|
||||
marker = f"{tc.suite}.{tc.name}.{int(time.time() * 1000)}"
|
||||
if tc.requires_restart:
|
||||
self.agent.restart(timeout=30)
|
||||
if not self.agent.is_alive():
|
||||
ok = self.agent.restart(timeout=30)
|
||||
if not ok:
|
||||
results.append(
|
||||
TestResult(
|
||||
suite=tc.suite,
|
||||
name=tc.name,
|
||||
status="error",
|
||||
message="Agent not alive and restart failed",
|
||||
duration_seconds=0.0,
|
||||
)
|
||||
)
|
||||
continue
|
||||
|
||||
self.log.set_marker(marker)
|
||||
start = time.monotonic()
|
||||
try:
|
||||
passed, message, details = tc.fn(ctx)
|
||||
status = "pass" if passed else "fail"
|
||||
except SkipTest as e:
|
||||
status = "skip"
|
||||
message = str(e)
|
||||
details = {}
|
||||
except DidactylTimeoutError as e:
|
||||
status = "timeout"
|
||||
message = str(e)
|
||||
details = {}
|
||||
if not getattr(self.args, "no_restart", False):
|
||||
self.agent.restart(timeout=30)
|
||||
except Exception as e:
|
||||
status = "error"
|
||||
message = repr(e)
|
||||
details = {}
|
||||
if not self.agent.is_alive() and not getattr(self.args, "no_restart", False):
|
||||
self.agent.restart(timeout=30)
|
||||
|
||||
duration = time.monotonic() - start
|
||||
agent_errors = self.log.error_lines(marker)
|
||||
results.append(
|
||||
TestResult(
|
||||
suite=tc.suite,
|
||||
name=tc.name,
|
||||
status=status,
|
||||
message=message,
|
||||
duration_seconds=duration,
|
||||
agent_errors=agent_errors,
|
||||
details=details,
|
||||
)
|
||||
)
|
||||
|
||||
if getattr(self.args, "verbose", False):
|
||||
print(f"[{status.upper()}] {tc.suite}::{tc.name} - {message}")
|
||||
|
||||
return results
|
||||
107
tests/run_tests.py
Executable file
107
tests/run_tests.py
Executable file
@@ -0,0 +1,107 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import datetime as dt
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from tests.harness.agent_process import AgentProcess
|
||||
from tests.harness.didactyl_client import DidactylClient
|
||||
from tests.harness.log_watcher import LogWatcher
|
||||
from tests.harness.reporter import Reporter
|
||||
from tests.harness.test_runner import TestRunner
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="Didactyl automated test harness")
|
||||
parser.add_argument("--config", default="tests/configs/test_genesis.jsonc")
|
||||
parser.add_argument("--binary", default="./didactyl")
|
||||
parser.add_argument("--suite", action="append", help="Run only this suite (repeatable)")
|
||||
parser.add_argument("--test", action="append", help="Run only this test name (repeatable)")
|
||||
parser.add_argument("--api-port", type=int, default=8485)
|
||||
parser.add_argument("--api-bind", default="127.0.0.1")
|
||||
parser.add_argument("--timeout", type=float, default=60.0)
|
||||
parser.add_argument("--debug-level", type=int, default=5)
|
||||
parser.add_argument("--output-dir", default=None)
|
||||
parser.add_argument("--verbose", action="store_true")
|
||||
parser.add_argument("--no-restart", action="store_true")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
|
||||
run_ts = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%SZ")
|
||||
output_dir = args.output_dir or f"tests/results/{run_ts}"
|
||||
Path(output_dir).mkdir(parents=True, exist_ok=True)
|
||||
|
||||
binary = Path(args.binary)
|
||||
if not binary.exists():
|
||||
print(f"didactyl binary not found: {binary}")
|
||||
return 2
|
||||
|
||||
config = Path(args.config)
|
||||
if not config.exists():
|
||||
print(f"config not found: {config}")
|
||||
return 2
|
||||
|
||||
log_file = str(Path(output_dir) / "agent_debug.log")
|
||||
base_url = f"https://{args.api_bind}:{args.api_port}"
|
||||
|
||||
agent = AgentProcess(
|
||||
binary_path=str(binary),
|
||||
config_path=str(config),
|
||||
api_port=args.api_port,
|
||||
api_bind=args.api_bind,
|
||||
debug_level=args.debug_level,
|
||||
log_file=log_file,
|
||||
base_url=base_url,
|
||||
)
|
||||
client = DidactylClient(base_url=base_url, timeout=args.timeout, verify_tls=False)
|
||||
log = LogWatcher(log_file)
|
||||
|
||||
log.start()
|
||||
started = agent.start(timeout=30)
|
||||
if not started:
|
||||
print("Failed to start Didactyl agent")
|
||||
log.stop()
|
||||
return 1
|
||||
|
||||
try:
|
||||
runner = TestRunner(agent=agent, client=client, log=log, args=args)
|
||||
tests = runner.discover_suites("tests.suites")
|
||||
if not tests:
|
||||
print("No tests discovered")
|
||||
return 3
|
||||
|
||||
results = runner.run_all(tests)
|
||||
|
||||
run_meta = {
|
||||
"run_timestamp": dt.datetime.now(dt.timezone.utc).isoformat(),
|
||||
"base_url": base_url,
|
||||
"config": str(config),
|
||||
"binary": str(binary),
|
||||
"test_count": len(results),
|
||||
}
|
||||
reporter = Reporter(results=results, run_meta=run_meta, output_dir=output_dir)
|
||||
reporter.print_summary()
|
||||
reporter.print_details()
|
||||
json_path = reporter.write_json("results.json")
|
||||
txt_path = reporter.write_text("results.txt")
|
||||
print(f"\nWrote: {json_path}")
|
||||
print(f"Wrote: {txt_path}")
|
||||
|
||||
bad = [r for r in results if r.status in {"fail", "error", "timeout"}]
|
||||
return 1 if bad else 0
|
||||
finally:
|
||||
agent.stop(timeout=10)
|
||||
log.stop()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
1
tests/suites/__init__.py
Normal file
1
tests/suites/__init__.py
Normal file
@@ -0,0 +1 @@
|
||||
"""Test suite modules for Didactyl harness."""
|
||||
BIN
tests/suites/__pycache__/__init__.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/__init__.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/common.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/common.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_conversation.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_conversation.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_errors.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_errors.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_health.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_health.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_restart.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_restart.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_timeouts.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_timeouts.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_tools_blossom.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_tools_blossom.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_tools_cashu.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_tools_cashu.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_tools_identity.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_tools_identity.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_tools_memory.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_tools_memory.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_tools_nostr.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_tools_nostr.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_tools_skills.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_tools_skills.cpython-313.pyc
Normal file
Binary file not shown.
BIN
tests/suites/__pycache__/test_tools_system.cpython-313.pyc
Normal file
BIN
tests/suites/__pycache__/test_tools_system.cpython-313.pyc
Normal file
Binary file not shown.
34
tests/suites/common.py
Normal file
34
tests/suites/common.py
Normal file
@@ -0,0 +1,34 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from tests.harness.test_runner import TestCase, TestContext
|
||||
|
||||
|
||||
def assert_success(resp: dict[str, Any]) -> None:
|
||||
assert isinstance(resp, dict), "response must be dict"
|
||||
assert resp.get("success") is True, f"expected success=true, got: {resp}"
|
||||
|
||||
|
||||
def turns_tool_names(resp: dict[str, Any]) -> list[str]:
|
||||
names: list[str] = []
|
||||
for turn in resp.get("turns", []) or []:
|
||||
for tc in turn.get("tool_calls", []) or []:
|
||||
name = tc.get("name")
|
||||
if isinstance(name, str):
|
||||
names.append(name)
|
||||
return names
|
||||
|
||||
|
||||
def simple_prompt_test(suite: str, name: str, description: str, prompt: str, expected_tool: str | None = None) -> TestCase:
|
||||
def _run(ctx: TestContext):
|
||||
resp = ctx.client.prompt(prompt, max_turns=6)
|
||||
assert_success(resp)
|
||||
final_response = str(resp.get("final_response", "")).strip()
|
||||
assert final_response, "final_response is empty"
|
||||
tools = turns_tool_names(resp)
|
||||
if expected_tool:
|
||||
assert expected_tool in tools, f"expected tool {expected_tool}, got {tools}"
|
||||
return True, "ok", {"tool_calls": tools, "final_response": final_response[:200]}
|
||||
|
||||
return TestCase(suite=suite, name=name, description=description, fn=_run)
|
||||
45
tests/suites/test_conversation.py
Normal file
45
tests/suites/test_conversation.py
Normal file
@@ -0,0 +1,45 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from tests.harness.test_runner import TestCase
|
||||
from .common import assert_success
|
||||
|
||||
SUITE = "test_conversation"
|
||||
|
||||
|
||||
def _simple_greeting(ctx):
|
||||
resp = ctx.client.prompt("Hello, what is your name?", max_turns=4)
|
||||
assert_success(resp)
|
||||
text = str(resp.get("final_response", "")).strip()
|
||||
assert text, "empty final_response"
|
||||
return True, "greeting response ok", {"response": text[:200]}
|
||||
|
||||
|
||||
def _self_description(ctx):
|
||||
resp = ctx.client.prompt("What are you? Describe yourself briefly.", max_turns=4)
|
||||
assert_success(resp)
|
||||
text = str(resp.get("final_response", "")).lower()
|
||||
assert text, "empty final_response"
|
||||
assert any(token in text for token in ["didactyl", "agent", "nostr"]), "response missing expected identity terms"
|
||||
return True, "self description ok", {"response": text[:200]}
|
||||
|
||||
|
||||
def _empty_message(ctx):
|
||||
resp = ctx.client.prompt("", max_turns=2)
|
||||
assert isinstance(resp, dict), "response should be JSON object"
|
||||
return True, "empty message handled", {"success": resp.get("success")}
|
||||
|
||||
|
||||
def _long_message(ctx):
|
||||
long_msg = "A" * 10000
|
||||
resp = ctx.client.prompt(long_msg, max_turns=4)
|
||||
assert isinstance(resp, dict), "response should be JSON object"
|
||||
return True, "long message handled", {"success": resp.get("success")}
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
TestCase(SUITE, "simple_greeting", "Simple hello prompt", _simple_greeting),
|
||||
TestCase(SUITE, "agent_responds_about_itself", "Agent self description", _self_description),
|
||||
TestCase(SUITE, "empty_message_handling", "Empty message behavior", _empty_message),
|
||||
TestCase(SUITE, "very_long_message", "Very long input behavior", _long_message),
|
||||
]
|
||||
38
tests/suites/test_errors.py
Normal file
38
tests/suites/test_errors.py
Normal file
@@ -0,0 +1,38 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from tests.harness.test_runner import TestCase
|
||||
|
||||
SUITE = "test_errors"
|
||||
|
||||
|
||||
def _invalid_json(ctx):
|
||||
code, body = ctx.client.raw_post("/api/prompt/agent", b"{not json")
|
||||
assert code == 400, f"expected 400, got {code} body={body}"
|
||||
return True, "invalid json rejected", {"status": code}
|
||||
|
||||
|
||||
def _missing_message(ctx):
|
||||
code, body = ctx.client.raw_post("/api/prompt/agent", b"{}")
|
||||
assert code == 400, f"expected 400, got {code} body={body}"
|
||||
return True, "missing message rejected", {"status": code}
|
||||
|
||||
|
||||
def _unknown_endpoint(ctx):
|
||||
code, _body = ctx.client.raw_post("/api/nonexistent", b"{}")
|
||||
assert code == 404, f"expected 404, got {code}"
|
||||
return True, "unknown endpoint 404", {"status": code}
|
||||
|
||||
|
||||
def _webhook_nonexistent(ctx):
|
||||
code, _body = ctx.client.raw_post("/api/trigger/nonexistent-dtag", b"{}")
|
||||
assert code == 404, f"expected 404, got {code}"
|
||||
return True, "nonexistent d_tag 404", {"status": code}
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
TestCase(SUITE, "invalid_json_body", "Malformed JSON rejected", _invalid_json),
|
||||
TestCase(SUITE, "missing_message_field", "Missing message field rejected", _missing_message),
|
||||
TestCase(SUITE, "unknown_endpoint", "Unknown endpoint returns 404", _unknown_endpoint),
|
||||
TestCase(SUITE, "webhook_nonexistent_dtag", "Unknown webhook dtag returns 404", _webhook_nonexistent),
|
||||
]
|
||||
47
tests/suites/test_health.py
Normal file
47
tests/suites/test_health.py
Normal file
@@ -0,0 +1,47 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from tests.harness.test_runner import TestCase
|
||||
from .common import assert_success
|
||||
|
||||
|
||||
SUITE = "test_health"
|
||||
|
||||
|
||||
def _status_returns_200(ctx):
|
||||
resp = ctx.client.status()
|
||||
assert_success(resp)
|
||||
return True, "status success", {"keys": list(resp.keys())}
|
||||
|
||||
|
||||
def _status_has_fields(ctx):
|
||||
resp = ctx.client.status()
|
||||
for key in ["name", "version", "pubkey", "relay_count"]:
|
||||
assert key in resp, f"missing field: {key}"
|
||||
return True, "required fields present", {}
|
||||
|
||||
|
||||
def _context_current(ctx):
|
||||
resp = ctx.client.context_current()
|
||||
assert_success(resp)
|
||||
msgs = resp.get("messages", [])
|
||||
assert isinstance(msgs, list), "messages must be a list"
|
||||
assert resp.get("total_chars", 0) >= 0, "missing/invalid total_chars"
|
||||
return True, "context current ok", {"message_count": len(msgs)}
|
||||
|
||||
|
||||
def _context_parts(ctx):
|
||||
resp = ctx.client.context_parts()
|
||||
assert_success(resp)
|
||||
parts = resp.get("parts", [])
|
||||
names = [p.get("name") for p in parts if isinstance(p, dict)]
|
||||
assert "system_prompt" in names, "system_prompt part missing"
|
||||
return True, "context parts ok", {"parts": names}
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
TestCase(SUITE, "status_returns_200", "GET /api/status success", _status_returns_200),
|
||||
TestCase(SUITE, "status_has_fields", "GET /api/status has required fields", _status_has_fields),
|
||||
TestCase(SUITE, "context_current_returns_messages", "GET /api/context/current has messages", _context_current),
|
||||
TestCase(SUITE, "context_parts_has_system_prompt", "GET /api/context/parts has system_prompt", _context_parts),
|
||||
]
|
||||
41
tests/suites/test_restart.py
Normal file
41
tests/suites/test_restart.py
Normal file
@@ -0,0 +1,41 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from tests.harness.test_runner import TestCase
|
||||
|
||||
SUITE = "test_restart"
|
||||
|
||||
|
||||
def _clean_restart(ctx):
|
||||
ok = ctx.agent.restart(timeout=30)
|
||||
assert ok, "restart failed"
|
||||
resp = ctx.client.status()
|
||||
assert resp.get("success") is True
|
||||
return True, "restart succeeded", {}
|
||||
|
||||
|
||||
def _status_after_restart(ctx):
|
||||
before = ctx.client.status()
|
||||
assert before.get("success") is True
|
||||
ok = ctx.agent.restart(timeout=30)
|
||||
assert ok, "restart failed"
|
||||
after = ctx.client.status()
|
||||
assert after.get("success") is True
|
||||
assert before.get("pubkey") == after.get("pubkey"), "pubkey changed after restart"
|
||||
return True, "status stable after restart", {"pubkey": after.get("pubkey")}
|
||||
|
||||
|
||||
def _conversation_after_restart(ctx):
|
||||
ok = ctx.agent.restart(timeout=30)
|
||||
assert ok, "restart failed"
|
||||
resp = ctx.client.prompt("After restart, say hello.", max_turns=3)
|
||||
assert resp.get("success") is True
|
||||
assert str(resp.get("final_response", "")).strip(), "empty final response"
|
||||
return True, "conversation works post-restart", {}
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
TestCase(SUITE, "clean_restart", "Stop/start restart", _clean_restart, requires_restart=True),
|
||||
TestCase(SUITE, "status_after_restart", "Status stable after restart", _status_after_restart),
|
||||
TestCase(SUITE, "conversation_after_restart", "Prompt after restart", _conversation_after_restart),
|
||||
]
|
||||
42
tests/suites/test_timeouts.py
Normal file
42
tests/suites/test_timeouts.py
Normal file
@@ -0,0 +1,42 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
|
||||
from tests.harness.test_runner import TestCase
|
||||
|
||||
SUITE = "test_timeouts"
|
||||
|
||||
|
||||
def _response_within_timeout(ctx):
|
||||
start = time.monotonic()
|
||||
resp = ctx.client.prompt("Say hello in one sentence.", max_turns=2)
|
||||
elapsed = time.monotonic() - start
|
||||
assert isinstance(resp, dict)
|
||||
assert elapsed <= float(ctx.args.timeout), f"response exceeded timeout {ctx.args.timeout}s"
|
||||
return True, "response within timeout", {"elapsed": elapsed}
|
||||
|
||||
|
||||
def _status_fast(ctx):
|
||||
start = time.monotonic()
|
||||
resp = ctx.client.status()
|
||||
elapsed = time.monotonic() - start
|
||||
assert resp.get("success") is True
|
||||
assert elapsed <= 2.0, f"status too slow: {elapsed}s"
|
||||
return True, "status fast", {"elapsed": elapsed}
|
||||
|
||||
|
||||
def _context_fast(ctx):
|
||||
start = time.monotonic()
|
||||
resp = ctx.client.context_current()
|
||||
elapsed = time.monotonic() - start
|
||||
assert resp.get("success") is True
|
||||
assert elapsed <= 5.0, f"context too slow: {elapsed}s"
|
||||
return True, "context fast", {"elapsed": elapsed}
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
TestCase(SUITE, "response_within_timeout", "Prompt response within timeout", _response_within_timeout),
|
||||
TestCase(SUITE, "status_responds_fast", "Status endpoint latency", _status_fast),
|
||||
TestCase(SUITE, "context_responds_fast", "Context endpoint latency", _context_fast),
|
||||
]
|
||||
9
tests/suites/test_tools_blossom.py
Normal file
9
tests/suites/test_tools_blossom.py
Normal file
@@ -0,0 +1,9 @@
|
||||
from .common import simple_prompt_test
|
||||
|
||||
SUITE = "test_tools_blossom"
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
simple_prompt_test(SUITE, "blossom_list", "Blossom list", "List your blossom blobs", "blossom_list"),
|
||||
]
|
||||
9
tests/suites/test_tools_cashu.py
Normal file
9
tests/suites/test_tools_cashu.py
Normal file
@@ -0,0 +1,9 @@
|
||||
from .common import simple_prompt_test
|
||||
|
||||
SUITE = "test_tools_cashu"
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
simple_prompt_test(SUITE, "wallet_balance", "Wallet balance", "Check your cashu wallet balance", "cashu_wallet_balance"),
|
||||
]
|
||||
13
tests/suites/test_tools_identity.py
Normal file
13
tests/suites/test_tools_identity.py
Normal file
@@ -0,0 +1,13 @@
|
||||
from .common import simple_prompt_test
|
||||
|
||||
SUITE = "test_tools_identity"
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
simple_prompt_test(SUITE, "get_pubkey", "Fetch hex pubkey", "What is your public key in hex?", "nostr_pubkey"),
|
||||
simple_prompt_test(SUITE, "get_npub", "Fetch npub", "What is your npub?", "nostr_npub"),
|
||||
simple_prompt_test(SUITE, "agent_identity", "Agent identity tool", "Tell me about your identity", "agent_identity"),
|
||||
simple_prompt_test(SUITE, "agent_version", "Agent version tool", "What version are you?", "agent_version"),
|
||||
simple_prompt_test(SUITE, "admin_identity", "Admin identity tool", "Who is your administrator?", "admin_identity"),
|
||||
]
|
||||
23
tests/suites/test_tools_memory.py
Normal file
23
tests/suites/test_tools_memory.py
Normal file
@@ -0,0 +1,23 @@
|
||||
from .common import simple_prompt_test
|
||||
|
||||
SUITE = "test_tools_memory"
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
simple_prompt_test(SUITE, "task_list", "Task list", "Show me your current task list", "task_list"),
|
||||
simple_prompt_test(
|
||||
SUITE,
|
||||
"task_manage_add_remove",
|
||||
"Task add/remove",
|
||||
"Add a task test harness probe task then remove it",
|
||||
"task_manage",
|
||||
),
|
||||
simple_prompt_test(
|
||||
SUITE,
|
||||
"memory_save_recall",
|
||||
"Save and recall memory",
|
||||
"Save test harness probe to memory, then recall your memory",
|
||||
"memory_save",
|
||||
),
|
||||
]
|
||||
15
tests/suites/test_tools_nostr.py
Normal file
15
tests/suites/test_tools_nostr.py
Normal file
@@ -0,0 +1,15 @@
|
||||
from .common import simple_prompt_test
|
||||
|
||||
SUITE = "test_tools_nostr"
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
simple_prompt_test(SUITE, "nostr_post_kind1", "Post kind1 note", "Post a test note saying Automated test post", "nostr_post"),
|
||||
simple_prompt_test(SUITE, "nostr_query_recent", "Query recent notes", "Query the 3 most recent kind 1 notes from any author", "nostr_query"),
|
||||
simple_prompt_test(SUITE, "nostr_my_events", "List my events", "List your recent events", "nostr_my_events"),
|
||||
simple_prompt_test(SUITE, "nostr_relay_status", "Relay status", "What is the status of your relay connections?", "nostr_relay_status"),
|
||||
simple_prompt_test(SUITE, "nostr_dm_send", "Send DM", "Send a test DM to yourself", "nostr_dm_send"),
|
||||
simple_prompt_test(SUITE, "nostr_encode_npub", "Encode npub", "Encode your pubkey as an npub", "nostr_encode"),
|
||||
simple_prompt_test(SUITE, "nostr_profile_get", "Profile lookup", "Look up your own Nostr profile", "nostr_profile_get"),
|
||||
]
|
||||
17
tests/suites/test_tools_skills.py
Normal file
17
tests/suites/test_tools_skills.py
Normal file
@@ -0,0 +1,17 @@
|
||||
from .common import simple_prompt_test
|
||||
|
||||
SUITE = "test_tools_skills"
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
simple_prompt_test(SUITE, "skill_list", "List skills", "List your available skills", "skill_list"),
|
||||
simple_prompt_test(SUITE, "trigger_list", "List triggers", "List your active triggers", "trigger_list"),
|
||||
simple_prompt_test(
|
||||
SUITE,
|
||||
"skill_create_and_remove",
|
||||
"Create and remove skill",
|
||||
"Create a test skill called test-harness-probe with content Test skill then remove it",
|
||||
"skill_create",
|
||||
),
|
||||
]
|
||||
19
tests/suites/test_tools_system.py
Normal file
19
tests/suites/test_tools_system.py
Normal file
@@ -0,0 +1,19 @@
|
||||
from .common import simple_prompt_test
|
||||
|
||||
SUITE = "test_tools_system"
|
||||
|
||||
|
||||
def get_tests():
|
||||
return [
|
||||
simple_prompt_test(SUITE, "tool_list", "Tool list", "List all your available tools", "tool_list"),
|
||||
simple_prompt_test(SUITE, "model_get", "Current model", "What model are you currently using?", "model_get"),
|
||||
simple_prompt_test(SUITE, "model_list", "Available models", "List available models", "model_list"),
|
||||
simple_prompt_test(SUITE, "local_http_fetch", "HTTP fetch", "Fetch https://httpbin.org/get", "local_http_fetch"),
|
||||
simple_prompt_test(
|
||||
SUITE,
|
||||
"config_store_recall",
|
||||
"Store+recall config",
|
||||
"Store a test config with d_tag test_harness_probe containing hello, then recall it",
|
||||
"config_store",
|
||||
),
|
||||
]
|
||||
Reference in New Issue
Block a user