diff --git a/README.md b/README.md
index e62c402..3316c62 100644
--- a/README.md
+++ b/README.md
@@ -54,11 +54,11 @@ Skills compose by adoption-list order (`10123`) and trigger tags carry runtime e
Didactyl will support local inference, which is very privacy preserving. Remote inference does however have it's advantages, and in those cases Didactyl supports using Bitcoin Lightning and eCash inference providers.
-## Current Status — v0.2.14
+## Current Status — v0.2.16
**Active build — this project is barely working. Experiment at your own risk.**
-> Last release update: v0.2.14 — Follow-up push after context verification and startup-skill refactor
+> Last release update: v0.2.16 — Add Python test harness and fix increment script commit identity handling
- Connects to configured relays with auto-reconnect and relay state transition logging
- Publishes configured startup events per relay as each relay becomes connected
diff --git a/increment_and_push.sh b/increment_and_push.sh
index e728b6f..0087756 100755
--- a/increment_and_push.sh
+++ b/increment_and_push.sh
@@ -111,6 +111,37 @@ check_git_repo() {
fi
}
+# Ensure git identity exists so commits do not fail on clean machines
+ensure_git_identity() {
+ local current_name=""
+ local current_email=""
+
+ current_name=$(git config --get user.name 2>/dev/null || true)
+ current_email=$(git config --get user.email 2>/dev/null || true)
+
+ if [[ -n "$current_name" && -n "$current_email" ]]; then
+ return 0
+ fi
+
+ print_warning "Git user.name / user.email not fully configured for this repository"
+
+ local fallback_name
+ local fallback_email
+
+ fallback_name="${GIT_AUTHOR_NAME:-Didactyl User}"
+ fallback_email="${GIT_AUTHOR_EMAIL:-didactyl@local}"
+
+ if [[ -z "$current_name" ]]; then
+ git config user.name "$fallback_name"
+ print_status "Set local git user.name to '$fallback_name'"
+ fi
+
+ if [[ -z "$current_email" ]]; then
+ git config user.email "$fallback_email"
+ print_status "Set local git user.email to '$fallback_email'"
+ fi
+}
+
# Function to get current version and increment appropriately
increment_version() {
local increment_type="$1" # "patch", "minor", or "major"
@@ -293,6 +324,8 @@ git_commit_and_push_no_tag() {
print_success "Committed changes"
else
print_error "Failed to commit changes"
+ print_error "git commit output:"
+ git commit -m "$NEW_VERSION - $COMMIT_MESSAGE" 2>&1 || true
exit 1
fi
fi
@@ -498,6 +531,7 @@ main() {
# Check prerequisites
check_git_repo
+ ensure_git_identity
if [[ "$RELEASE_MODE" == true ]]; then
print_status "=== RELEASE MODE ==="
diff --git a/plans/automated_test_harness.md b/plans/automated_test_harness.md
new file mode 100644
index 0000000..a42564a
--- /dev/null
+++ b/plans/automated_test_harness.md
@@ -0,0 +1,616 @@
+# Didactyl Automated Test Harness
+
+## Overview
+
+An automated testing system that starts a Didactyl agent locally (debug build), converses with it via the HTTP API, exercises all tools, monitors logs in real-time, handles agent crashes/restarts, and produces a structured test results report.
+
+**Language:** Python (stdlib only, no external dependencies)
+**Location:** `tests/`
+**Phase 1:** Scripted tests (no LLM driving the tester)
+**Phase 2 (future):** LLM-driven test agent that generates prompts, evaluates responses, and adapts
+
+---
+
+## Architecture
+
+```mermaid
+flowchart TB
+ subgraph Test Harness - Python
+ RUNNER[test_runner.py
orchestrator]
+ PROC[agent_process.py
start/stop/restart]
+ CLIENT[didactyl_client.py
HTTP API wrapper]
+ LOG[log_watcher.py
tail debug.log]
+ REPORT[reporter.py
results output]
+
+ RUNNER --> PROC
+ RUNNER --> CLIENT
+ RUNNER --> LOG
+ RUNNER --> REPORT
+ end
+
+ subgraph Test Suites
+ TH[test_health]
+ TC[test_conversation]
+ TI[test_tools_identity]
+ TN[test_tools_nostr]
+ TS[test_tools_skills]
+ TSY[test_tools_system]
+ TM[test_tools_memory]
+ TCA[test_tools_cashu]
+ TB[test_tools_blossom]
+ TTO[test_timeouts]
+ TE[test_errors]
+ TR[test_restart]
+ end
+
+ RUNNER --> TH & TC & TI & TN & TS & TSY & TM & TCA & TB & TTO & TE & TR
+
+ subgraph Didactyl Agent - debug build
+ AGENT[didactyl process]
+ API[HTTP API :8484]
+ LOGFILE[debug.log]
+ end
+
+ CLIENT -- HTTP --> API
+ PROC -- subprocess --> AGENT
+ LOG -- tail --> LOGFILE
+```
+
+---
+
+## Design Decisions
+
+| Decision | Choice | Rationale |
+|---|---|---|
+| Language | Python | Best subprocess/HTTP/threading support; already used in project |
+| Dependencies | stdlib only | No pip install needed; `urllib`, `subprocess`, `threading`, `json` |
+| Test framework | Standalone runner | Self-contained, no pytest dependency |
+| Relay strategy | Flexible genesis config | User provides their own test_genesis.jsonc |
+| LLM for agent | Real model | User provides API key in test genesis config |
+| Tool scope | All tools | Disposable test identity; full coverage |
+| Entry point | `tests/run_tests.py` | Single script to run everything |
+
+---
+
+## Directory Structure
+
+```
+tests/
+├── harness/
+│ ├── __init__.py
+│ ├── agent_process.py # start/stop/restart didactyl subprocess
+│ ├── didactyl_client.py # HTTP API wrapper with timeouts
+│ ├── log_watcher.py # real-time log tail + marker system
+│ ├── test_runner.py # orchestrator
+│ └── reporter.py # results formatting (JSON + text)
+├── suites/
+│ ├── __init__.py
+│ ├── test_health.py # status, context endpoints
+│ ├── test_conversation.py # basic prompt/response
+│ ├── test_tools_identity.py # identity and context tools
+│ ├── test_tools_nostr.py # nostr event, messaging, relay tools
+│ ├── test_tools_skills.py # skill and trigger tools
+│ ├── test_tools_system.py # system, local, model, config tools
+│ ├── test_tools_memory.py # task and memory tools
+│ ├── test_tools_cashu.py # cashu wallet tools
+│ ├── test_tools_blossom.py # blossom tools
+│ ├── test_timeouts.py # response time assertions
+│ ├── test_errors.py # API error handling paths
+│ └── test_restart.py # crash recovery
+├── configs/
+│ └── test_genesis.jsonc # example test config (user fills in secrets)
+├── results/ # test run output (gitignored)
+├── run_tests.py # entry point
+├── test.sh # existing bash test (kept as-is)
+├── blossom_tool_validation_test.c # existing C test (kept as-is)
+└── blossom_tool_validation_test # existing compiled test (kept as-is)
+```
+
+---
+
+## Component Specifications
+
+### 1. agent_process.py — Process Manager
+
+Manages the didactyl subprocess lifecycle.
+
+```python
+class AgentProcess:
+ def __init__(self, binary_path, config_path, api_port=8484,
+ api_bind="127.0.0.1", debug_level=5, log_file=None):
+ """Configure but don't start yet."""
+
+ def start(self, timeout=30) -> bool:
+ """
+ Spawn didactyl as subprocess:
+ ./didactyl --config --debug
+ --api-port --api-bind
+
+ Set DIDACTYL_LOG_FILE env var to log_file path.
+ Wait for GET /api/status to return 200 (poll with timeout).
+ Returns True if agent started successfully.
+ """
+
+ def stop(self, timeout=10) -> bool:
+ """
+ Send SIGTERM, wait for clean exit.
+ If still alive after timeout, send SIGKILL.
+ Returns True if stopped cleanly.
+ """
+
+ def restart(self, timeout=30) -> bool:
+ """stop() then start()."""
+
+ def is_alive(self) -> bool:
+ """Check process.poll() and optionally /api/status."""
+
+ def pid(self) -> int | None:
+ """Return PID if running."""
+
+ def return_code(self) -> int | None:
+ """Return exit code if stopped."""
+```
+
+Key details:
+- Uses `subprocess.Popen` with `stdout=PIPE, stderr=PIPE`
+- Captures stdout/stderr for crash diagnostics
+- The health check polls `GET /api/status` every 500ms until success or timeout
+- Sets `DIDACTYL_LOG_FILE` to a test-run-specific path like `tests/results//agent_debug.log`
+
+### 2. didactyl_client.py — HTTP API Client
+
+Thin wrapper around the Didactyl HTTP API using only `urllib`.
+
+```python
+class DidactylClient:
+ def __init__(self, base_url="https://127.0.0.1:8484", timeout=60,
+ verify_tls=False):
+ """Configure base URL and default timeout."""
+
+ def status(self) -> dict:
+ """GET /api/status"""
+
+ def prompt(self, message: str, max_turns: int = 4,
+ model: str = None) -> dict:
+ """POST /api/prompt/agent — full agent context conversation"""
+
+ def prompt_raw(self, messages: list, max_turns: int = 4,
+ model: str = None) -> dict:
+ """POST /api/prompt/run — raw messages, no auto-context"""
+
+ def prompt_simple(self, system: str, user: str,
+ model: str = None) -> dict:
+ """POST /api/prompt/run-simple — no tools"""
+
+ def context_current(self) -> dict:
+ """GET /api/context/current"""
+
+ def context_parts(self) -> dict:
+ """GET /api/context/parts"""
+
+ def fire_webhook(self, d_tag: str, payload: dict = None) -> dict:
+ """POST /api/trigger/"""
+```
+
+Key details:
+- All methods return parsed JSON dict
+- Raises `TimeoutError` if response exceeds timeout
+- Raises `ConnectionError` if agent is unreachable
+- Raises `APIError(status_code, body)` for non-2xx responses
+- Uses `ssl._create_unverified_context()` for local TLS (same as chat CLI)
+
+### 3. log_watcher.py — Real-time Log Monitor
+
+Background thread that tails the agent's debug log file.
+
+```python
+class LogWatcher:
+ def __init__(self, log_path: str):
+ """Configure log file path."""
+
+ def start(self):
+ """
+ Start background thread.
+ Open file, seek to end, poll for new lines.
+ Store all lines in memory with timestamps.
+ """
+
+ def stop(self):
+ """Stop background thread."""
+
+ def set_marker(self, name: str):
+ """Record current line count as a named marker."""
+
+ def get_lines_since(self, marker: str) -> list[str]:
+ """Return all lines captured since the named marker."""
+
+ def get_all_lines(self) -> list[str]:
+ """Return all captured lines."""
+
+ def search(self, pattern: str, since_marker: str = None) -> list[str]:
+ """Regex search through captured lines."""
+
+ def has_errors(self, since_marker: str = None) -> bool:
+ """Check for [ERROR] lines since marker."""
+
+ def has_warnings(self, since_marker: str = None) -> bool:
+ """Check for [WARN] lines since marker."""
+
+ def error_lines(self, since_marker: str = None) -> list[str]:
+ """Return all ERROR lines since marker."""
+```
+
+Key details:
+- Uses `threading.Thread(daemon=True)` for background polling
+- Polls file every 100ms for new content
+- Handles file rotation (agent restart creates new file)
+- Thread-safe access to captured lines via `threading.Lock`
+
+### 4. test_runner.py — Orchestrator
+
+```python
+class TestResult:
+ name: str
+ suite: str
+ status: str # "pass", "fail", "error", "skip", "timeout"
+ message: str
+ duration_seconds: float
+ agent_errors: list[str] # ERROR lines from log during this test
+ details: dict # arbitrary test-specific data
+
+class TestCase:
+ name: str
+ description: str
+ requires_restart: bool = False
+
+ def run(self, client: DidactylClient, log: LogWatcher) -> TestResult:
+ """Execute the test and return result."""
+
+class TestRunner:
+ def __init__(self, agent: AgentProcess, client: DidactylClient,
+ log: LogWatcher):
+ """Configure with harness components."""
+
+ def discover_suites(self, suites_dir: str) -> list:
+ """Import all test_*.py modules from suites directory."""
+
+ def run_all(self, suites: list = None) -> list[TestResult]:
+ """
+ For each suite:
+ 1. If test requires restart, restart agent
+ 2. Set log marker for this test
+ 3. Run test with timeout wrapper
+ 4. Capture result + any agent errors from log
+ 5. If agent crashed, restart and record error
+ 6. Collect all results
+ Return list of TestResult.
+ """
+
+ def run_suite(self, suite_name: str) -> list[TestResult]:
+ """Run a single named suite."""
+```
+
+Key details:
+- Each test gets a fresh log marker so errors can be correlated
+- If `client.prompt()` raises `TimeoutError`, the test is marked "timeout" and the agent is restarted
+- If `agent.is_alive()` returns False mid-suite, the agent is restarted and remaining tests continue
+- Supports filtering by suite name or test name via CLI args
+
+### 5. reporter.py — Results Output
+
+```python
+class Reporter:
+ def __init__(self, results: list[TestResult], output_dir: str):
+ """Configure with results and output directory."""
+
+ def print_summary(self):
+ """Print pass/fail/error/skip/timeout counts to stdout."""
+
+ def print_details(self):
+ """Print each test result with details."""
+
+ def write_json(self, path: str):
+ """Write full results as JSON for programmatic consumption."""
+
+ def write_text(self, path: str):
+ """Write human-readable report."""
+```
+
+Output format example:
+```
+== Didactyl Test Results ==
+Run: 2026-03-25T09:50:00Z
+Agent: v0.0.26
+Model: claude-haiku-4.5
+
+Pass: 42
+Fail: 3
+Error: 1
+Timeout: 2
+Skip: 0
+Total: 48
+
+-- Failures --
+[FAIL] test_tools_nostr::nostr_post_kind1
+ Expected tool_calls to contain 'nostr_post', got: ['nostr_query']
+ Agent errors during test: 0
+
+[FAIL] test_conversation::multi_turn
+ Agent returned empty final_response
+ Agent errors during test: 2
+ [ERROR] [llm.c:234] HTTP 429 rate limited
+ [ERROR] [llm.c:240] LLM call failed after 3 retries
+
+[TIMEOUT] test_tools_skills::skill_create
+ No response within 60s
+ Agent was restarted
+```
+
+---
+
+## Test Suite Specifications
+
+### test_health.py
+
+| Test | Prompt/Action | Assertions |
+|---|---|---|
+| `status_returns_200` | `GET /api/status` | HTTP 200, `success=true` |
+| `status_has_fields` | `GET /api/status` | Has `name`, `version`, `pubkey`, `relay_count` |
+| `context_current_returns_messages` | `GET /api/context/current` | Has `messages` array, `total_chars > 0` |
+| `context_parts_has_system_prompt` | `GET /api/context/parts` | Has part named `system_prompt` |
+
+### test_conversation.py
+
+| Test | Prompt | Assertions |
+|---|---|---|
+| `simple_greeting` | "Hello, what is your name?" | `final_response` is non-empty string |
+| `agent_responds_about_itself` | "What are you? Describe yourself briefly." | `final_response` mentions agent/Didactyl/Nostr |
+| `empty_message_handling` | "" (empty string) | Returns error or handles gracefully |
+| `very_long_message` | 10000 char string | Returns response or graceful error, no crash |
+
+### test_tools_identity.py
+
+| Test | Prompt | Expected Tool | Assertions |
+|---|---|---|---|
+| `get_pubkey` | "What is your public key in hex?" | `nostr_pubkey` or `my_pubkey` | Tool called, result has hex pubkey |
+| `get_npub` | "What is your npub?" | `nostr_npub` or `my_npub` | Tool called, result has npub1... |
+| `agent_identity` | "Tell me about your identity" | `agent_identity` | Tool called, success |
+| `agent_version` | "What version are you?" | `agent_version` | Tool called, result has version string |
+| `admin_identity` | "Who is your administrator?" | `admin_identity` | Tool called, success |
+
+### test_tools_nostr.py
+
+| Test | Prompt | Expected Tool | Assertions |
+|---|---|---|---|
+| `nostr_post_kind1` | "Post a test note saying 'Automated test post'" | `nostr_post` | Tool called with kind=1, success, event_id returned |
+| `nostr_query_recent` | "Query the 3 most recent kind 1 notes from any author" | `nostr_query` | Tool called, returns events array |
+| `nostr_my_events` | "List your recent events" | `nostr_my_events` | Tool called, success |
+| `nostr_relay_status` | "What is the status of your relay connections?" | `nostr_relay_status` | Tool called, returns relay info |
+| `nostr_dm_send` | "Send a test DM to yourself" | `nostr_dm_send` | Tool called, success |
+| `nostr_encode_npub` | "Encode your pubkey as an npub" | `nostr_encode` | Tool called, returns npub |
+| `nostr_profile_get` | "Look up your own Nostr profile" | `nostr_profile_get` | Tool called, returns profile |
+
+### test_tools_skills.py
+
+| Test | Prompt | Expected Tool | Assertions |
+|---|---|---|---|
+| `skill_list` | "List your available skills" | `skill_list` | Tool called, returns skills array |
+| `trigger_list` | "List your active triggers" | `trigger_list` | Tool called, success |
+| `skill_create_and_remove` | "Create a test skill called 'test-harness-probe' with content 'Test skill' then remove it" | `skill_create`, `skill_remove` | Both tools called, success |
+
+### test_tools_system.py
+
+| Test | Prompt | Expected Tool | Assertions |
+|---|---|---|---|
+| `tool_list` | "List all your available tools" | `tool_list` | Tool called, returns tools array |
+| `model_get` | "What model are you currently using?" | `model_get` | Tool called, returns model info |
+| `model_list` | "List available models" | `model_list` | Tool called, success |
+| `local_http_fetch` | "Fetch https://httpbin.org/get" | `local_http_fetch` | Tool called, returns HTTP response |
+| `config_store_recall` | "Store a test config with d_tag 'test_harness_probe' containing 'hello', then recall it" | `config_store`, `config_recall` | Both called, recalled value matches |
+
+### test_tools_memory.py
+
+| Test | Prompt | Expected Tool | Assertions |
+|---|---|---|---|
+| `task_list` | "Show me your current task list" | `task_list` or `task_manage` | Tool called, success |
+| `task_manage_add_remove` | "Add a task 'test harness probe task' then remove it" | `task_manage` | Tool called with add then remove |
+| `memory_save_recall` | "Save 'test harness probe' to memory, then recall your memory" | `memory_save`, `memory_recall` | Both called, recalled contains probe text |
+
+### test_tools_cashu.py
+
+| Test | Prompt | Expected Tool | Assertions |
+|---|---|---|---|
+| `wallet_balance` | "Check your cashu wallet balance" | `cashu_wallet_balance` | Tool called, success (even if empty) |
+
+Note: Most cashu tools require a configured mint and funded wallet. Phase 1 tests only the read-only balance check. Full cashu testing requires a test mint setup.
+
+### test_tools_blossom.py
+
+| Test | Prompt | Expected Tool | Assertions |
+|---|---|---|---|
+| `blossom_list` | "List your blossom blobs" | `blossom_list` | Tool called, success (even if empty) |
+
+Note: Full blossom testing requires a configured blossom server. Phase 1 tests only the list operation.
+
+### test_timeouts.py
+
+| Test | Action | Assertions |
+|---|---|---|
+| `response_within_timeout` | Send simple prompt, measure time | Response received within 60s |
+| `status_responds_fast` | `GET /api/status`, measure time | Response within 2s |
+| `context_responds_fast` | `GET /api/context/current`, measure time | Response within 5s |
+
+### test_errors.py
+
+| Test | Action | Assertions |
+|---|---|---|
+| `invalid_json_body` | POST malformed JSON to `/api/prompt/agent` | Returns 400, `success=false` |
+| `missing_message_field` | POST `{}` to `/api/prompt/agent` | Returns 400, `success=false` |
+| `unknown_endpoint` | GET `/api/nonexistent` | Returns 404 |
+| `webhook_nonexistent_dtag` | POST to `/api/trigger/nonexistent-dtag` | Returns 404 |
+
+### test_restart.py
+
+| Test | Action | Assertions |
+|---|---|---|
+| `clean_restart` | Stop agent, start agent | Agent comes back, `/api/status` works |
+| `status_after_restart` | Restart, then `GET /api/status` | Same pubkey, version as before restart |
+| `conversation_after_restart` | Restart, then send prompt | Agent responds normally |
+
+---
+
+## Entry Point: run_tests.py
+
+```
+Usage:
+ python tests/run_tests.py [options]
+
+Options:
+ --config PATH Path to test genesis.jsonc (default: tests/configs/test_genesis.jsonc)
+ --binary PATH Path to didactyl binary (default: ./didactyl)
+ --suite NAME Run only this suite (can repeat)
+ --test NAME Run only this test (can repeat)
+ --api-port PORT API port (default: 8485)
+ --timeout SECS Default response timeout (default: 60)
+ --debug-level N Agent debug level 0-5 (default: 5)
+ --output-dir PATH Results output directory (default: tests/results/)
+ --verbose Print each test result as it runs
+ --no-restart Don't auto-restart agent on crash (fail remaining tests)
+```
+
+---
+
+## Test Genesis Config
+
+`tests/configs/test_genesis.jsonc` — an example config the user fills in:
+
+```jsonc
+{
+ // TEST CONFIGURATION — fill in your test identity secrets
+
+ "key": {
+ "nsec": "nsec1REPLACE_WITH_DISPOSABLE_TEST_NSEC"
+ },
+
+ "admin": {
+ "pubkey": "npub1REPLACE_WITH_TEST_ADMIN_PUBKEY"
+ },
+
+ "dm_protocol": "nip04",
+
+ "llm": {
+ "provider": "openai",
+ "api_key": "sk-REPLACE_WITH_API_KEY",
+ "model": "claude-haiku-4.5",
+ "base_url": "https://api.anthropic.com/v1",
+ "max_tokens": 512,
+ "temperature": 0.3
+ },
+
+ "api": {
+ "enabled": true,
+ "port": 8485,
+ "bind_address": "127.0.0.1"
+ },
+
+ "startup_events": [
+ {
+ "kind": 0,
+ "content_fields": {
+ "name": "Didactyl Test Agent",
+ "about": "Automated test instance"
+ },
+ "tags": []
+ },
+ {
+ "kind": 10002,
+ "content": "",
+ "tags": [
+ ["r", "wss://relay.damus.io"],
+ ["r", "wss://relay.primal.net"]
+ ]
+ },
+ {
+ "kind": 31124,
+ "content": "# Test Agent\n\nYou are a test agent. Respond to all requests. Use tools when asked.\n\n{{my_kind0_profile}}\n\nYour npub: {{my_npub}}",
+ "tags": [
+ ["d", "identity_and_rules"],
+ ["app", "didactyl"],
+ ["scope", "private"],
+ ["trigger", "dm"],
+ ["filter", "{\"from\":\"admin\"}"]
+ ]
+ }
+ ]
+}
+```
+
+---
+
+## Execution Flow
+
+```mermaid
+flowchart TD
+ START[run_tests.py] --> PARSE[Parse CLI args]
+ PARSE --> BUILD[Verify didactyl binary exists]
+ BUILD --> CONFIG[Load test genesis config]
+ CONFIG --> MKDIR[Create results output dir]
+ MKDIR --> LOG_START[Start LogWatcher]
+ LOG_START --> AGENT_START[Start AgentProcess]
+ AGENT_START --> HEALTH[Wait for /api/status 200]
+
+ HEALTH -->|timeout| FAIL_STARTUP[Report startup failure and exit]
+ HEALTH -->|success| DISCOVER[Discover test suites]
+
+ DISCOVER --> LOOP{Next test?}
+
+ LOOP -->|yes| CHECK_ALIVE{Agent alive?}
+ CHECK_ALIVE -->|no| RESTART_MID[Restart agent]
+ RESTART_MID --> CHECK_ALIVE
+ CHECK_ALIVE -->|yes| MARKER[Set log marker]
+ MARKER --> RUN_TEST[Run test with timeout]
+
+ RUN_TEST -->|pass| RECORD_PASS[Record PASS]
+ RUN_TEST -->|fail| RECORD_FAIL[Record FAIL]
+ RUN_TEST -->|timeout| RESTART_TIMEOUT[Restart agent]
+ RESTART_TIMEOUT --> RECORD_TIMEOUT[Record TIMEOUT]
+ RUN_TEST -->|error| RECORD_ERROR[Record ERROR]
+
+ RECORD_PASS & RECORD_FAIL & RECORD_TIMEOUT & RECORD_ERROR --> COLLECT_LOGS[Collect agent errors since marker]
+ COLLECT_LOGS --> LOOP
+
+ LOOP -->|no more| STOP_AGENT[Stop agent]
+ STOP_AGENT --> STOP_LOG[Stop LogWatcher]
+ STOP_LOG --> REPORT[Generate report]
+ REPORT --> EXIT[Exit with code 0 if all pass else 1]
+```
+
+---
+
+## Implementation Order
+
+1. **Harness infrastructure** (agent_process, didactyl_client, log_watcher)
+2. **Test runner + reporter** (orchestration layer)
+3. **test_health.py** (validates the harness itself works)
+4. **test_conversation.py** (validates basic agent interaction)
+5. **test_errors.py** (validates error handling)
+6. **test_timeouts.py** (validates timeout detection)
+7. **test_restart.py** (validates process management)
+8. **Tool test suites** (one at a time, identity → nostr → skills → system → memory → cashu → blossom)
+9. **Entry point + config** (run_tests.py, test_genesis.jsonc)
+10. **End-to-end validation** against a running agent
+
+---
+
+## Future: Phase 2 — LLM-Driven Testing
+
+The architecture supports this naturally. In Phase 2:
+
+- Add an `LLMTestAgent` class that wraps an LLM API call
+- The LLM receives: tool documentation, test objectives, previous results
+- It generates test prompts, evaluates responses, decides next actions
+- The `TestCase.run()` method delegates to the LLM agent instead of scripted logic
+- The LLM can also analyze agent logs for anomalies
+- Potentially: the LLM can generate code patches for bugs it finds
+
+No architectural changes needed — just a new test suite type that uses LLM reasoning instead of scripted assertions.
diff --git a/src/main.h b/src/main.h
index 142ba28..44aae13 100644
--- a/src/main.h
+++ b/src/main.h
@@ -12,8 +12,8 @@
// Using DIDACTYL_ prefix to avoid conflicts with nostr_core_lib VERSION macros
#define DIDACTYL_VERSION_MAJOR 0
#define DIDACTYL_VERSION_MINOR 2
-#define DIDACTYL_VERSION_PATCH 14
-#define DIDACTYL_VERSION "v0.2.14"
+#define DIDACTYL_VERSION_PATCH 16
+#define DIDACTYL_VERSION "v0.2.16"
// Agent metadata
#define DIDACTYL_NAME "Didactyl"
diff --git a/tests/__pycache__/run_tests.cpython-313.pyc b/tests/__pycache__/run_tests.cpython-313.pyc
new file mode 100644
index 0000000..7f78d71
Binary files /dev/null and b/tests/__pycache__/run_tests.cpython-313.pyc differ
diff --git a/tests/configs/test_genesis.jsonc b/tests/configs/test_genesis.jsonc
new file mode 100644
index 0000000..105941a
--- /dev/null
+++ b/tests/configs/test_genesis.jsonc
@@ -0,0 +1,53 @@
+{
+ // TEST CONFIGURATION
+ // Use disposable keys/accounts only.
+ "key": {
+ "nsec": "nsec1REPLACE_WITH_DISPOSABLE_TEST_NSEC"
+ },
+ "admin": {
+ "pubkey": "npub1REPLACE_WITH_TEST_ADMIN_PUBKEY"
+ },
+ "dm_protocol": "nip04",
+ "llm": {
+ "provider": "openai",
+ "api_key": "sk-REPLACE_WITH_API_KEY",
+ "model": "claude-haiku-4.5",
+ "base_url": "https://api.anthropic.com/v1",
+ "max_tokens": 512,
+ "temperature": 0.3
+ },
+ "api": {
+ "enabled": true,
+ "port": 8485,
+ "bind_address": "127.0.0.1"
+ },
+ "startup_events": [
+ {
+ "kind": 0,
+ "content_fields": {
+ "name": "Didactyl Test Agent",
+ "about": "Automated test instance"
+ },
+ "tags": []
+ },
+ {
+ "kind": 10002,
+ "content": "",
+ "tags": [
+ ["r", "wss://relay.damus.io"],
+ ["r", "wss://relay.primal.net"]
+ ]
+ },
+ {
+ "kind": 31124,
+ "content": "# Test Agent\n\nYou are a test agent. Respond to requests and use tools when needed.",
+ "tags": [
+ ["d", "identity_and_rules"],
+ ["app", "didactyl"],
+ ["scope", "private"],
+ ["trigger", "dm"],
+ ["filter", "{\"from\":\"admin\"}"]
+ ]
+ }
+ ]
+}
diff --git a/tests/harness/__init__.py b/tests/harness/__init__.py
new file mode 100644
index 0000000..5f50206
--- /dev/null
+++ b/tests/harness/__init__.py
@@ -0,0 +1,26 @@
+"""Didactyl automated test harness package."""
+
+from .agent_process import AgentProcess
+from .didactyl_client import (
+ APIError,
+ DidactylClient,
+ DidactylConnectionError,
+ DidactylTimeoutError,
+)
+from .log_watcher import LogWatcher
+from .reporter import Reporter
+from .test_runner import SkipTest, TestCase, TestResult, TestRunner
+
+__all__ = [
+ "AgentProcess",
+ "APIError",
+ "DidactylClient",
+ "DidactylConnectionError",
+ "DidactylTimeoutError",
+ "LogWatcher",
+ "Reporter",
+ "SkipTest",
+ "TestCase",
+ "TestResult",
+ "TestRunner",
+]
diff --git a/tests/harness/__pycache__/__init__.cpython-313.pyc b/tests/harness/__pycache__/__init__.cpython-313.pyc
new file mode 100644
index 0000000..3ad3b7b
Binary files /dev/null and b/tests/harness/__pycache__/__init__.cpython-313.pyc differ
diff --git a/tests/harness/__pycache__/agent_process.cpython-313.pyc b/tests/harness/__pycache__/agent_process.cpython-313.pyc
new file mode 100644
index 0000000..f5c75ad
Binary files /dev/null and b/tests/harness/__pycache__/agent_process.cpython-313.pyc differ
diff --git a/tests/harness/__pycache__/didactyl_client.cpython-313.pyc b/tests/harness/__pycache__/didactyl_client.cpython-313.pyc
new file mode 100644
index 0000000..c7db4cd
Binary files /dev/null and b/tests/harness/__pycache__/didactyl_client.cpython-313.pyc differ
diff --git a/tests/harness/__pycache__/log_watcher.cpython-313.pyc b/tests/harness/__pycache__/log_watcher.cpython-313.pyc
new file mode 100644
index 0000000..29a50d3
Binary files /dev/null and b/tests/harness/__pycache__/log_watcher.cpython-313.pyc differ
diff --git a/tests/harness/__pycache__/reporter.cpython-313.pyc b/tests/harness/__pycache__/reporter.cpython-313.pyc
new file mode 100644
index 0000000..fdd8f66
Binary files /dev/null and b/tests/harness/__pycache__/reporter.cpython-313.pyc differ
diff --git a/tests/harness/__pycache__/test_runner.cpython-313.pyc b/tests/harness/__pycache__/test_runner.cpython-313.pyc
new file mode 100644
index 0000000..f942e29
Binary files /dev/null and b/tests/harness/__pycache__/test_runner.cpython-313.pyc differ
diff --git a/tests/harness/agent_process.py b/tests/harness/agent_process.py
new file mode 100644
index 0000000..ff77ce5
--- /dev/null
+++ b/tests/harness/agent_process.py
@@ -0,0 +1,113 @@
+from __future__ import annotations
+
+import os
+import signal
+import subprocess
+import time
+from dataclasses import dataclass
+from pathlib import Path
+from typing import Optional
+
+from .didactyl_client import DidactylClient
+
+
+@dataclass
+class AgentProcess:
+ binary_path: str
+ config_path: str
+ api_port: int = 8485
+ api_bind: str = "127.0.0.1"
+ debug_level: int = 5
+ log_file: Optional[str] = None
+ base_url: Optional[str] = None
+
+ def __post_init__(self) -> None:
+ self.binary_path = str(Path(self.binary_path))
+ self.config_path = str(Path(self.config_path))
+ self.log_file = self.log_file or "tests/results/agent_debug.log"
+ scheme = "https"
+ self.base_url = self.base_url or f"{scheme}://{self.api_bind}:{self.api_port}"
+ self.process: Optional[subprocess.Popen[str]] = None
+
+ def _command(self) -> list[str]:
+ return [
+ self.binary_path,
+ "--config",
+ self.config_path,
+ "--debug",
+ str(self.debug_level),
+ "--api-port",
+ str(self.api_port),
+ "--api-bind",
+ self.api_bind,
+ ]
+
+ def start(self, timeout: float = 30.0) -> bool:
+ if self.is_alive():
+ return True
+
+ env = os.environ.copy()
+ env["DIDACTYL_LOG_FILE"] = str(self.log_file)
+
+ Path(self.log_file).parent.mkdir(parents=True, exist_ok=True)
+
+ self.process = subprocess.Popen(
+ self._command(),
+ stdout=subprocess.PIPE,
+ stderr=subprocess.PIPE,
+ text=True,
+ env=env,
+ )
+
+ client = DidactylClient(base_url=self.base_url, timeout=2.0, verify_tls=False)
+ deadline = time.time() + timeout
+ while time.time() < deadline:
+ if self.process and self.process.poll() is not None:
+ return False
+ try:
+ data = client.status()
+ if data.get("success"):
+ return True
+ except Exception:
+ pass
+ time.sleep(0.5)
+ return False
+
+ def stop(self, timeout: float = 10.0) -> bool:
+ if not self.process:
+ return True
+ if self.process.poll() is not None:
+ return True
+
+ try:
+ self.process.send_signal(signal.SIGTERM)
+ self.process.wait(timeout=timeout)
+ return True
+ except subprocess.TimeoutExpired:
+ self.process.kill()
+ self.process.wait(timeout=5)
+ return False
+
+ def restart(self, timeout: float = 30.0) -> bool:
+ self.stop()
+ return self.start(timeout=timeout)
+
+ def is_alive(self) -> bool:
+ return self.process is not None and self.process.poll() is None
+
+ def pid(self) -> Optional[int]:
+ return None if not self.process else self.process.pid
+
+ def return_code(self) -> Optional[int]:
+ return None if not self.process else self.process.poll()
+
+ def read_pipes(self) -> tuple[str, str]:
+ if not self.process:
+ return "", ""
+ out = ""
+ err = ""
+ if self.process.stdout:
+ out = self.process.stdout.read() or ""
+ if self.process.stderr:
+ err = self.process.stderr.read() or ""
+ return out, err
diff --git a/tests/harness/didactyl_client.py b/tests/harness/didactyl_client.py
new file mode 100644
index 0000000..fc17732
--- /dev/null
+++ b/tests/harness/didactyl_client.py
@@ -0,0 +1,104 @@
+from __future__ import annotations
+
+import json
+import ssl
+import urllib.error
+import urllib.parse
+import urllib.request
+from dataclasses import dataclass
+from typing import Any, Optional
+
+
+class DidactylTimeoutError(TimeoutError):
+ pass
+
+
+class DidactylConnectionError(ConnectionError):
+ pass
+
+
+@dataclass
+class APIError(Exception):
+ status_code: int
+ body: str
+
+ def __str__(self) -> str:
+ return f"HTTP {self.status_code}: {self.body}"
+
+
+class DidactylClient:
+ def __init__(self, base_url: str = "https://127.0.0.1:8485", timeout: float = 60.0, verify_tls: bool = False) -> None:
+ self.base_url = base_url.rstrip("/")
+ self.timeout = timeout
+ self.verify_tls = verify_tls
+ self.ssl_context = None
+ if self.base_url.startswith("https://") and not verify_tls:
+ self.ssl_context = ssl._create_unverified_context()
+
+ def _request(self, method: str, path: str, payload: Optional[dict[str, Any]] = None, raw_body: Optional[bytes] = None) -> dict[str, Any]:
+ url = f"{self.base_url}{path}"
+ body = raw_body
+ if payload is not None:
+ body = json.dumps(payload).encode("utf-8")
+ req = urllib.request.Request(url=url, method=method.upper(), data=body)
+ req.add_header("Content-Type", "application/json")
+
+ try:
+ with urllib.request.urlopen(req, timeout=self.timeout, context=self.ssl_context) as resp:
+ status = resp.getcode()
+ text = resp.read().decode("utf-8", errors="replace")
+ data = json.loads(text) if text.strip() else {}
+ if status < 200 or status >= 300:
+ raise APIError(status, text)
+ return data
+ except urllib.error.HTTPError as e:
+ text = e.read().decode("utf-8", errors="replace")
+ raise APIError(e.code, text) from e
+ except urllib.error.URLError as e:
+ reason = str(getattr(e, "reason", e))
+ if "timed out" in reason.lower():
+ raise DidactylTimeoutError(reason) from e
+ raise DidactylConnectionError(reason) from e
+ except TimeoutError as e:
+ raise DidactylTimeoutError(str(e)) from e
+
+ def status(self) -> dict[str, Any]:
+ return self._request("GET", "/api/status")
+
+ def context_current(self) -> dict[str, Any]:
+ return self._request("GET", "/api/context/current")
+
+ def context_parts(self) -> dict[str, Any]:
+ return self._request("GET", "/api/context/parts")
+
+ def prompt(self, message: str, max_turns: int = 4, model: Optional[str] = None) -> dict[str, Any]:
+ payload: dict[str, Any] = {"message": message, "max_turns": max_turns}
+ if model:
+ payload["model"] = model
+ return self._request("POST", "/api/prompt/agent", payload=payload)
+
+ def prompt_raw(self, messages: list[dict[str, str]], max_turns: int = 4, model: Optional[str] = None) -> dict[str, Any]:
+ payload: dict[str, Any] = {"messages": messages, "max_turns": max_turns}
+ if model:
+ payload["model"] = model
+ return self._request("POST", "/api/prompt/run", payload=payload)
+
+ def prompt_simple(self, system: str, user: str, model: Optional[str] = None) -> dict[str, Any]:
+ payload: dict[str, Any] = {"system": system, "user": user}
+ if model:
+ payload["model"] = model
+ return self._request("POST", "/api/prompt/run-simple", payload=payload)
+
+ def fire_webhook(self, d_tag: str, payload: Optional[dict[str, Any]] = None) -> dict[str, Any]:
+ encoded = urllib.parse.quote(d_tag, safe="")
+ return self._request("POST", f"/api/trigger/{encoded}", payload=payload or {})
+
+ def raw_post(self, path: str, body: bytes) -> tuple[int, str]:
+ url = f"{self.base_url}{path}"
+ req = urllib.request.Request(url=url, method="POST", data=body)
+ req.add_header("Content-Type", "application/json")
+ try:
+ with urllib.request.urlopen(req, timeout=self.timeout, context=self.ssl_context) as resp:
+ return resp.getcode(), resp.read().decode("utf-8", errors="replace")
+ except urllib.error.HTTPError as e:
+ return e.code, e.read().decode("utf-8", errors="replace")
diff --git a/tests/harness/log_watcher.py b/tests/harness/log_watcher.py
new file mode 100644
index 0000000..caef7d4
--- /dev/null
+++ b/tests/harness/log_watcher.py
@@ -0,0 +1,77 @@
+from __future__ import annotations
+
+import re
+import threading
+import time
+from pathlib import Path
+from typing import Optional
+
+
+class LogWatcher:
+ def __init__(self, log_path: str, poll_interval: float = 0.1) -> None:
+ self.log_path = Path(log_path)
+ self.poll_interval = poll_interval
+ self._lines: list[str] = []
+ self._markers: dict[str, int] = {}
+ self._lock = threading.Lock()
+ self._stop = threading.Event()
+ self._thread: Optional[threading.Thread] = None
+
+ def start(self) -> None:
+ self.log_path.parent.mkdir(parents=True, exist_ok=True)
+ self.log_path.touch(exist_ok=True)
+ self._stop.clear()
+ self._thread = threading.Thread(target=self._run, daemon=True)
+ self._thread.start()
+
+ def stop(self) -> None:
+ self._stop.set()
+ if self._thread and self._thread.is_alive():
+ self._thread.join(timeout=2)
+
+ def _run(self) -> None:
+ f = self.log_path.open("r", encoding="utf-8", errors="replace")
+ f.seek(0, 2)
+ try:
+ while not self._stop.is_set():
+ pos = f.tell()
+ line = f.readline()
+ if not line:
+ if self.log_path.exists() and self.log_path.stat().st_size < pos:
+ f.close()
+ f = self.log_path.open("r", encoding="utf-8", errors="replace")
+ time.sleep(self.poll_interval)
+ continue
+ with self._lock:
+ self._lines.append(line.rstrip("\n"))
+ finally:
+ f.close()
+
+ def set_marker(self, name: str) -> None:
+ with self._lock:
+ self._markers[name] = len(self._lines)
+
+ def get_lines_since(self, marker: str) -> list[str]:
+ with self._lock:
+ idx = self._markers.get(marker, 0)
+ return list(self._lines[idx:])
+
+ def get_all_lines(self) -> list[str]:
+ with self._lock:
+ return list(self._lines)
+
+ def search(self, pattern: str, since_marker: Optional[str] = None) -> list[str]:
+ regex = re.compile(pattern)
+ lines = self.get_all_lines() if since_marker is None else self.get_lines_since(since_marker)
+ return [line for line in lines if regex.search(line)]
+
+ def error_lines(self, since_marker: Optional[str] = None) -> list[str]:
+ lines = self.get_all_lines() if since_marker is None else self.get_lines_since(since_marker)
+ return [line for line in lines if "[ERROR]" in line]
+
+ def warning_lines(self, since_marker: Optional[str] = None) -> list[str]:
+ lines = self.get_all_lines() if since_marker is None else self.get_lines_since(since_marker)
+ return [line for line in lines if "[WARN" in line]
+
+ def has_errors(self, since_marker: Optional[str] = None) -> bool:
+ return len(self.error_lines(since_marker)) > 0
diff --git a/tests/harness/reporter.py b/tests/harness/reporter.py
new file mode 100644
index 0000000..264a3be
--- /dev/null
+++ b/tests/harness/reporter.py
@@ -0,0 +1,77 @@
+from __future__ import annotations
+
+import json
+from collections import Counter
+from dataclasses import asdict
+from pathlib import Path
+from typing import Any
+
+from .test_runner import TestResult
+
+
+class Reporter:
+ def __init__(self, results: list[TestResult], run_meta: dict[str, Any], output_dir: str) -> None:
+ self.results = results
+ self.run_meta = run_meta
+ self.output_dir = Path(output_dir)
+ self.output_dir.mkdir(parents=True, exist_ok=True)
+
+ def summary_counts(self) -> Counter:
+ return Counter(r.status for r in self.results)
+
+ def print_summary(self) -> None:
+ c = self.summary_counts()
+ total = len(self.results)
+ print("\n== Didactyl Test Results ==")
+ print(f"Run: {self.run_meta.get('run_timestamp', 'unknown')}")
+ print(f"Agent base URL: {self.run_meta.get('base_url', 'unknown')}")
+ print("")
+ for k in ["pass", "fail", "error", "timeout", "skip"]:
+ print(f"{k.capitalize():<8}: {c.get(k, 0)}")
+ print(f"Total : {total}")
+
+ def print_details(self) -> None:
+ bad = [r for r in self.results if r.status in {"fail", "error", "timeout"}]
+ if not bad:
+ return
+ print("\n-- Non-passing tests --")
+ for r in bad:
+ print(f"[{r.status.upper()}] {r.suite}::{r.name}")
+ print(f" {r.message}")
+ if r.agent_errors:
+ print(f" Agent errors: {len(r.agent_errors)}")
+
+ def write_json(self, path: str = "results.json") -> Path:
+ p = self.output_dir / path
+ payload = {
+ "run_meta": self.run_meta,
+ "results": [asdict(r) for r in self.results],
+ "summary": dict(self.summary_counts()),
+ }
+ p.write_text(json.dumps(payload, indent=2), encoding="utf-8")
+ return p
+
+ def write_text(self, path: str = "results.txt") -> Path:
+ p = self.output_dir / path
+ c = self.summary_counts()
+ lines = [
+ "== Didactyl Test Results ==",
+ f"Run: {self.run_meta.get('run_timestamp', 'unknown')}",
+ f"Agent base URL: {self.run_meta.get('base_url', 'unknown')}",
+ "",
+ f"Pass: {c.get('pass', 0)}",
+ f"Fail: {c.get('fail', 0)}",
+ f"Error: {c.get('error', 0)}",
+ f"Timeout: {c.get('timeout', 0)}",
+ f"Skip: {c.get('skip', 0)}",
+ f"Total: {len(self.results)}",
+ "",
+ ]
+ for r in self.results:
+ lines.append(f"[{r.status.upper()}] {r.suite}::{r.name} ({r.duration_seconds:.2f}s)")
+ lines.append(f" {r.message}")
+ if r.agent_errors:
+ lines.append(f" Agent errors: {len(r.agent_errors)}")
+ lines.append("")
+ p.write_text("\n".join(lines), encoding="utf-8")
+ return p
diff --git a/tests/harness/test_runner.py b/tests/harness/test_runner.py
new file mode 100644
index 0000000..5d7ae31
--- /dev/null
+++ b/tests/harness/test_runner.py
@@ -0,0 +1,141 @@
+from __future__ import annotations
+
+import importlib
+import pkgutil
+import time
+from dataclasses import dataclass, field
+from types import ModuleType
+from typing import Any, Callable
+
+from .agent_process import AgentProcess
+from .didactyl_client import DidactylClient, DidactylTimeoutError
+from .log_watcher import LogWatcher
+
+
+class SkipTest(Exception):
+ pass
+
+
+TestFn = Callable[["TestContext"], tuple[bool, str, dict[str, Any]]]
+
+
+@dataclass
+class TestCase:
+ suite: str
+ name: str
+ description: str
+ fn: TestFn
+ requires_restart: bool = False
+
+
+@dataclass
+class TestResult:
+ suite: str
+ name: str
+ status: str
+ message: str
+ duration_seconds: float
+ agent_errors: list[str] = field(default_factory=list)
+ details: dict[str, Any] = field(default_factory=dict)
+
+
+@dataclass
+class TestContext:
+ client: DidactylClient
+ agent: AgentProcess
+ log: LogWatcher
+ args: Any
+
+
+class TestRunner:
+ def __init__(self, agent: AgentProcess, client: DidactylClient, log: LogWatcher, args: Any) -> None:
+ self.agent = agent
+ self.client = client
+ self.log = log
+ self.args = args
+
+ def discover_suites(self, package: str = "tests.suites") -> list[TestCase]:
+ mod = importlib.import_module(package)
+ tests: list[TestCase] = []
+ for info in pkgutil.iter_modules(mod.__path__):
+ if not info.name.startswith("test_"):
+ continue
+ if self.args.suite and info.name not in self.args.suite:
+ continue
+ module = importlib.import_module(f"{package}.{info.name}")
+ tests.extend(self._tests_from_module(module))
+ return tests
+
+ def _tests_from_module(self, module: ModuleType) -> list[TestCase]:
+ if not hasattr(module, "get_tests"):
+ return []
+ suite_tests = module.get_tests()
+ out: list[TestCase] = []
+ for t in suite_tests:
+ if self.args.test and t.name not in self.args.test:
+ continue
+ out.append(t)
+ return out
+
+ def run_all(self, tests: list[TestCase]) -> list[TestResult]:
+ results: list[TestResult] = []
+ ctx = TestContext(client=self.client, agent=self.agent, log=self.log, args=self.args)
+
+ for tc in tests:
+ marker = f"{tc.suite}.{tc.name}.{int(time.time() * 1000)}"
+ if tc.requires_restart:
+ self.agent.restart(timeout=30)
+ if not self.agent.is_alive():
+ ok = self.agent.restart(timeout=30)
+ if not ok:
+ results.append(
+ TestResult(
+ suite=tc.suite,
+ name=tc.name,
+ status="error",
+ message="Agent not alive and restart failed",
+ duration_seconds=0.0,
+ )
+ )
+ continue
+
+ self.log.set_marker(marker)
+ start = time.monotonic()
+ try:
+ passed, message, details = tc.fn(ctx)
+ status = "pass" if passed else "fail"
+ except SkipTest as e:
+ status = "skip"
+ message = str(e)
+ details = {}
+ except DidactylTimeoutError as e:
+ status = "timeout"
+ message = str(e)
+ details = {}
+ if not getattr(self.args, "no_restart", False):
+ self.agent.restart(timeout=30)
+ except Exception as e:
+ status = "error"
+ message = repr(e)
+ details = {}
+ if not self.agent.is_alive() and not getattr(self.args, "no_restart", False):
+ self.agent.restart(timeout=30)
+
+ duration = time.monotonic() - start
+ agent_errors = self.log.error_lines(marker)
+ results.append(
+ TestResult(
+ suite=tc.suite,
+ name=tc.name,
+ status=status,
+ message=message,
+ duration_seconds=duration,
+ agent_errors=agent_errors,
+ details=details,
+ )
+ )
+
+ if getattr(self.args, "verbose", False):
+ print(f"[{status.upper()}] {tc.suite}::{tc.name} - {message}")
+
+ return results
diff --git a/tests/run_tests.py b/tests/run_tests.py
new file mode 100755
index 0000000..028629d
--- /dev/null
+++ b/tests/run_tests.py
@@ -0,0 +1,107 @@
+#!/usr/bin/env python3
+from __future__ import annotations
+
+import argparse
+import datetime as dt
+import sys
+from pathlib import Path
+
+ROOT = Path(__file__).resolve().parent.parent
+if str(ROOT) not in sys.path:
+ sys.path.insert(0, str(ROOT))
+
+from tests.harness.agent_process import AgentProcess
+from tests.harness.didactyl_client import DidactylClient
+from tests.harness.log_watcher import LogWatcher
+from tests.harness.reporter import Reporter
+from tests.harness.test_runner import TestRunner
+
+
+def parse_args() -> argparse.Namespace:
+ parser = argparse.ArgumentParser(description="Didactyl automated test harness")
+ parser.add_argument("--config", default="tests/configs/test_genesis.jsonc")
+ parser.add_argument("--binary", default="./didactyl")
+ parser.add_argument("--suite", action="append", help="Run only this suite (repeatable)")
+ parser.add_argument("--test", action="append", help="Run only this test name (repeatable)")
+ parser.add_argument("--api-port", type=int, default=8485)
+ parser.add_argument("--api-bind", default="127.0.0.1")
+ parser.add_argument("--timeout", type=float, default=60.0)
+ parser.add_argument("--debug-level", type=int, default=5)
+ parser.add_argument("--output-dir", default=None)
+ parser.add_argument("--verbose", action="store_true")
+ parser.add_argument("--no-restart", action="store_true")
+ return parser.parse_args()
+
+
+def main() -> int:
+ args = parse_args()
+
+ run_ts = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%SZ")
+ output_dir = args.output_dir or f"tests/results/{run_ts}"
+ Path(output_dir).mkdir(parents=True, exist_ok=True)
+
+ binary = Path(args.binary)
+ if not binary.exists():
+ print(f"didactyl binary not found: {binary}")
+ return 2
+
+ config = Path(args.config)
+ if not config.exists():
+ print(f"config not found: {config}")
+ return 2
+
+ log_file = str(Path(output_dir) / "agent_debug.log")
+ base_url = f"https://{args.api_bind}:{args.api_port}"
+
+ agent = AgentProcess(
+ binary_path=str(binary),
+ config_path=str(config),
+ api_port=args.api_port,
+ api_bind=args.api_bind,
+ debug_level=args.debug_level,
+ log_file=log_file,
+ base_url=base_url,
+ )
+ client = DidactylClient(base_url=base_url, timeout=args.timeout, verify_tls=False)
+ log = LogWatcher(log_file)
+
+ log.start()
+ started = agent.start(timeout=30)
+ if not started:
+ print("Failed to start Didactyl agent")
+ log.stop()
+ return 1
+
+ try:
+ runner = TestRunner(agent=agent, client=client, log=log, args=args)
+ tests = runner.discover_suites("tests.suites")
+ if not tests:
+ print("No tests discovered")
+ return 3
+
+ results = runner.run_all(tests)
+
+ run_meta = {
+ "run_timestamp": dt.datetime.now(dt.timezone.utc).isoformat(),
+ "base_url": base_url,
+ "config": str(config),
+ "binary": str(binary),
+ "test_count": len(results),
+ }
+ reporter = Reporter(results=results, run_meta=run_meta, output_dir=output_dir)
+ reporter.print_summary()
+ reporter.print_details()
+ json_path = reporter.write_json("results.json")
+ txt_path = reporter.write_text("results.txt")
+ print(f"\nWrote: {json_path}")
+ print(f"Wrote: {txt_path}")
+
+ bad = [r for r in results if r.status in {"fail", "error", "timeout"}]
+ return 1 if bad else 0
+ finally:
+ agent.stop(timeout=10)
+ log.stop()
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/tests/suites/__init__.py b/tests/suites/__init__.py
new file mode 100644
index 0000000..62e974f
--- /dev/null
+++ b/tests/suites/__init__.py
@@ -0,0 +1 @@
+"""Test suite modules for Didactyl harness."""
diff --git a/tests/suites/__pycache__/__init__.cpython-313.pyc b/tests/suites/__pycache__/__init__.cpython-313.pyc
new file mode 100644
index 0000000..e580da5
Binary files /dev/null and b/tests/suites/__pycache__/__init__.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/common.cpython-313.pyc b/tests/suites/__pycache__/common.cpython-313.pyc
new file mode 100644
index 0000000..799c0d5
Binary files /dev/null and b/tests/suites/__pycache__/common.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_conversation.cpython-313.pyc b/tests/suites/__pycache__/test_conversation.cpython-313.pyc
new file mode 100644
index 0000000..9037be1
Binary files /dev/null and b/tests/suites/__pycache__/test_conversation.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_errors.cpython-313.pyc b/tests/suites/__pycache__/test_errors.cpython-313.pyc
new file mode 100644
index 0000000..a378345
Binary files /dev/null and b/tests/suites/__pycache__/test_errors.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_health.cpython-313.pyc b/tests/suites/__pycache__/test_health.cpython-313.pyc
new file mode 100644
index 0000000..8597ec1
Binary files /dev/null and b/tests/suites/__pycache__/test_health.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_restart.cpython-313.pyc b/tests/suites/__pycache__/test_restart.cpython-313.pyc
new file mode 100644
index 0000000..c29432e
Binary files /dev/null and b/tests/suites/__pycache__/test_restart.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_timeouts.cpython-313.pyc b/tests/suites/__pycache__/test_timeouts.cpython-313.pyc
new file mode 100644
index 0000000..c11872f
Binary files /dev/null and b/tests/suites/__pycache__/test_timeouts.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_tools_blossom.cpython-313.pyc b/tests/suites/__pycache__/test_tools_blossom.cpython-313.pyc
new file mode 100644
index 0000000..48c6dff
Binary files /dev/null and b/tests/suites/__pycache__/test_tools_blossom.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_tools_cashu.cpython-313.pyc b/tests/suites/__pycache__/test_tools_cashu.cpython-313.pyc
new file mode 100644
index 0000000..0c0e2b2
Binary files /dev/null and b/tests/suites/__pycache__/test_tools_cashu.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_tools_identity.cpython-313.pyc b/tests/suites/__pycache__/test_tools_identity.cpython-313.pyc
new file mode 100644
index 0000000..2e58004
Binary files /dev/null and b/tests/suites/__pycache__/test_tools_identity.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_tools_memory.cpython-313.pyc b/tests/suites/__pycache__/test_tools_memory.cpython-313.pyc
new file mode 100644
index 0000000..df12822
Binary files /dev/null and b/tests/suites/__pycache__/test_tools_memory.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_tools_nostr.cpython-313.pyc b/tests/suites/__pycache__/test_tools_nostr.cpython-313.pyc
new file mode 100644
index 0000000..0e696e3
Binary files /dev/null and b/tests/suites/__pycache__/test_tools_nostr.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_tools_skills.cpython-313.pyc b/tests/suites/__pycache__/test_tools_skills.cpython-313.pyc
new file mode 100644
index 0000000..da279f3
Binary files /dev/null and b/tests/suites/__pycache__/test_tools_skills.cpython-313.pyc differ
diff --git a/tests/suites/__pycache__/test_tools_system.cpython-313.pyc b/tests/suites/__pycache__/test_tools_system.cpython-313.pyc
new file mode 100644
index 0000000..80d3426
Binary files /dev/null and b/tests/suites/__pycache__/test_tools_system.cpython-313.pyc differ
diff --git a/tests/suites/common.py b/tests/suites/common.py
new file mode 100644
index 0000000..bde43cc
--- /dev/null
+++ b/tests/suites/common.py
@@ -0,0 +1,34 @@
+from __future__ import annotations
+
+from typing import Any
+
+from tests.harness.test_runner import TestCase, TestContext
+
+
+def assert_success(resp: dict[str, Any]) -> None:
+ assert isinstance(resp, dict), "response must be dict"
+ assert resp.get("success") is True, f"expected success=true, got: {resp}"
+
+
+def turns_tool_names(resp: dict[str, Any]) -> list[str]:
+ names: list[str] = []
+ for turn in resp.get("turns", []) or []:
+ for tc in turn.get("tool_calls", []) or []:
+ name = tc.get("name")
+ if isinstance(name, str):
+ names.append(name)
+ return names
+
+
+def simple_prompt_test(suite: str, name: str, description: str, prompt: str, expected_tool: str | None = None) -> TestCase:
+ def _run(ctx: TestContext):
+ resp = ctx.client.prompt(prompt, max_turns=6)
+ assert_success(resp)
+ final_response = str(resp.get("final_response", "")).strip()
+ assert final_response, "final_response is empty"
+ tools = turns_tool_names(resp)
+ if expected_tool:
+ assert expected_tool in tools, f"expected tool {expected_tool}, got {tools}"
+ return True, "ok", {"tool_calls": tools, "final_response": final_response[:200]}
+
+ return TestCase(suite=suite, name=name, description=description, fn=_run)
diff --git a/tests/suites/test_conversation.py b/tests/suites/test_conversation.py
new file mode 100644
index 0000000..764f40d
--- /dev/null
+++ b/tests/suites/test_conversation.py
@@ -0,0 +1,45 @@
+from __future__ import annotations
+
+from tests.harness.test_runner import TestCase
+from .common import assert_success
+
+SUITE = "test_conversation"
+
+
+def _simple_greeting(ctx):
+ resp = ctx.client.prompt("Hello, what is your name?", max_turns=4)
+ assert_success(resp)
+ text = str(resp.get("final_response", "")).strip()
+ assert text, "empty final_response"
+ return True, "greeting response ok", {"response": text[:200]}
+
+
+def _self_description(ctx):
+ resp = ctx.client.prompt("What are you? Describe yourself briefly.", max_turns=4)
+ assert_success(resp)
+ text = str(resp.get("final_response", "")).lower()
+ assert text, "empty final_response"
+ assert any(token in text for token in ["didactyl", "agent", "nostr"]), "response missing expected identity terms"
+ return True, "self description ok", {"response": text[:200]}
+
+
+def _empty_message(ctx):
+ resp = ctx.client.prompt("", max_turns=2)
+ assert isinstance(resp, dict), "response should be JSON object"
+ return True, "empty message handled", {"success": resp.get("success")}
+
+
+def _long_message(ctx):
+ long_msg = "A" * 10000
+ resp = ctx.client.prompt(long_msg, max_turns=4)
+ assert isinstance(resp, dict), "response should be JSON object"
+ return True, "long message handled", {"success": resp.get("success")}
+
+
+def get_tests():
+ return [
+ TestCase(SUITE, "simple_greeting", "Simple hello prompt", _simple_greeting),
+ TestCase(SUITE, "agent_responds_about_itself", "Agent self description", _self_description),
+ TestCase(SUITE, "empty_message_handling", "Empty message behavior", _empty_message),
+ TestCase(SUITE, "very_long_message", "Very long input behavior", _long_message),
+ ]
diff --git a/tests/suites/test_errors.py b/tests/suites/test_errors.py
new file mode 100644
index 0000000..56b8323
--- /dev/null
+++ b/tests/suites/test_errors.py
@@ -0,0 +1,38 @@
+from __future__ import annotations
+
+from tests.harness.test_runner import TestCase
+
+SUITE = "test_errors"
+
+
+def _invalid_json(ctx):
+ code, body = ctx.client.raw_post("/api/prompt/agent", b"{not json")
+ assert code == 400, f"expected 400, got {code} body={body}"
+ return True, "invalid json rejected", {"status": code}
+
+
+def _missing_message(ctx):
+ code, body = ctx.client.raw_post("/api/prompt/agent", b"{}")
+ assert code == 400, f"expected 400, got {code} body={body}"
+ return True, "missing message rejected", {"status": code}
+
+
+def _unknown_endpoint(ctx):
+ code, _body = ctx.client.raw_post("/api/nonexistent", b"{}")
+ assert code == 404, f"expected 404, got {code}"
+ return True, "unknown endpoint 404", {"status": code}
+
+
+def _webhook_nonexistent(ctx):
+ code, _body = ctx.client.raw_post("/api/trigger/nonexistent-dtag", b"{}")
+ assert code == 404, f"expected 404, got {code}"
+ return True, "nonexistent d_tag 404", {"status": code}
+
+
+def get_tests():
+ return [
+ TestCase(SUITE, "invalid_json_body", "Malformed JSON rejected", _invalid_json),
+ TestCase(SUITE, "missing_message_field", "Missing message field rejected", _missing_message),
+ TestCase(SUITE, "unknown_endpoint", "Unknown endpoint returns 404", _unknown_endpoint),
+ TestCase(SUITE, "webhook_nonexistent_dtag", "Unknown webhook dtag returns 404", _webhook_nonexistent),
+ ]
diff --git a/tests/suites/test_health.py b/tests/suites/test_health.py
new file mode 100644
index 0000000..3854a07
--- /dev/null
+++ b/tests/suites/test_health.py
@@ -0,0 +1,47 @@
+from __future__ import annotations
+
+from tests.harness.test_runner import TestCase
+from .common import assert_success
+
+
+SUITE = "test_health"
+
+
+def _status_returns_200(ctx):
+ resp = ctx.client.status()
+ assert_success(resp)
+ return True, "status success", {"keys": list(resp.keys())}
+
+
+def _status_has_fields(ctx):
+ resp = ctx.client.status()
+ for key in ["name", "version", "pubkey", "relay_count"]:
+ assert key in resp, f"missing field: {key}"
+ return True, "required fields present", {}
+
+
+def _context_current(ctx):
+ resp = ctx.client.context_current()
+ assert_success(resp)
+ msgs = resp.get("messages", [])
+ assert isinstance(msgs, list), "messages must be a list"
+ assert resp.get("total_chars", 0) >= 0, "missing/invalid total_chars"
+ return True, "context current ok", {"message_count": len(msgs)}
+
+
+def _context_parts(ctx):
+ resp = ctx.client.context_parts()
+ assert_success(resp)
+ parts = resp.get("parts", [])
+ names = [p.get("name") for p in parts if isinstance(p, dict)]
+ assert "system_prompt" in names, "system_prompt part missing"
+ return True, "context parts ok", {"parts": names}
+
+
+def get_tests():
+ return [
+ TestCase(SUITE, "status_returns_200", "GET /api/status success", _status_returns_200),
+ TestCase(SUITE, "status_has_fields", "GET /api/status has required fields", _status_has_fields),
+ TestCase(SUITE, "context_current_returns_messages", "GET /api/context/current has messages", _context_current),
+ TestCase(SUITE, "context_parts_has_system_prompt", "GET /api/context/parts has system_prompt", _context_parts),
+ ]
diff --git a/tests/suites/test_restart.py b/tests/suites/test_restart.py
new file mode 100644
index 0000000..9989f31
--- /dev/null
+++ b/tests/suites/test_restart.py
@@ -0,0 +1,41 @@
+from __future__ import annotations
+
+from tests.harness.test_runner import TestCase
+
+SUITE = "test_restart"
+
+
+def _clean_restart(ctx):
+ ok = ctx.agent.restart(timeout=30)
+ assert ok, "restart failed"
+ resp = ctx.client.status()
+ assert resp.get("success") is True
+ return True, "restart succeeded", {}
+
+
+def _status_after_restart(ctx):
+ before = ctx.client.status()
+ assert before.get("success") is True
+ ok = ctx.agent.restart(timeout=30)
+ assert ok, "restart failed"
+ after = ctx.client.status()
+ assert after.get("success") is True
+ assert before.get("pubkey") == after.get("pubkey"), "pubkey changed after restart"
+ return True, "status stable after restart", {"pubkey": after.get("pubkey")}
+
+
+def _conversation_after_restart(ctx):
+ ok = ctx.agent.restart(timeout=30)
+ assert ok, "restart failed"
+ resp = ctx.client.prompt("After restart, say hello.", max_turns=3)
+ assert resp.get("success") is True
+ assert str(resp.get("final_response", "")).strip(), "empty final response"
+ return True, "conversation works post-restart", {}
+
+
+def get_tests():
+ return [
+ TestCase(SUITE, "clean_restart", "Stop/start restart", _clean_restart, requires_restart=True),
+ TestCase(SUITE, "status_after_restart", "Status stable after restart", _status_after_restart),
+ TestCase(SUITE, "conversation_after_restart", "Prompt after restart", _conversation_after_restart),
+ ]
diff --git a/tests/suites/test_timeouts.py b/tests/suites/test_timeouts.py
new file mode 100644
index 0000000..1e0a0ef
--- /dev/null
+++ b/tests/suites/test_timeouts.py
@@ -0,0 +1,42 @@
+from __future__ import annotations
+
+import time
+
+from tests.harness.test_runner import TestCase
+
+SUITE = "test_timeouts"
+
+
+def _response_within_timeout(ctx):
+ start = time.monotonic()
+ resp = ctx.client.prompt("Say hello in one sentence.", max_turns=2)
+ elapsed = time.monotonic() - start
+ assert isinstance(resp, dict)
+ assert elapsed <= float(ctx.args.timeout), f"response exceeded timeout {ctx.args.timeout}s"
+ return True, "response within timeout", {"elapsed": elapsed}
+
+
+def _status_fast(ctx):
+ start = time.monotonic()
+ resp = ctx.client.status()
+ elapsed = time.monotonic() - start
+ assert resp.get("success") is True
+ assert elapsed <= 2.0, f"status too slow: {elapsed}s"
+ return True, "status fast", {"elapsed": elapsed}
+
+
+def _context_fast(ctx):
+ start = time.monotonic()
+ resp = ctx.client.context_current()
+ elapsed = time.monotonic() - start
+ assert resp.get("success") is True
+ assert elapsed <= 5.0, f"context too slow: {elapsed}s"
+ return True, "context fast", {"elapsed": elapsed}
+
+
+def get_tests():
+ return [
+ TestCase(SUITE, "response_within_timeout", "Prompt response within timeout", _response_within_timeout),
+ TestCase(SUITE, "status_responds_fast", "Status endpoint latency", _status_fast),
+ TestCase(SUITE, "context_responds_fast", "Context endpoint latency", _context_fast),
+ ]
diff --git a/tests/suites/test_tools_blossom.py b/tests/suites/test_tools_blossom.py
new file mode 100644
index 0000000..8dc0d96
--- /dev/null
+++ b/tests/suites/test_tools_blossom.py
@@ -0,0 +1,9 @@
+from .common import simple_prompt_test
+
+SUITE = "test_tools_blossom"
+
+
+def get_tests():
+ return [
+ simple_prompt_test(SUITE, "blossom_list", "Blossom list", "List your blossom blobs", "blossom_list"),
+ ]
diff --git a/tests/suites/test_tools_cashu.py b/tests/suites/test_tools_cashu.py
new file mode 100644
index 0000000..572989e
--- /dev/null
+++ b/tests/suites/test_tools_cashu.py
@@ -0,0 +1,9 @@
+from .common import simple_prompt_test
+
+SUITE = "test_tools_cashu"
+
+
+def get_tests():
+ return [
+ simple_prompt_test(SUITE, "wallet_balance", "Wallet balance", "Check your cashu wallet balance", "cashu_wallet_balance"),
+ ]
diff --git a/tests/suites/test_tools_identity.py b/tests/suites/test_tools_identity.py
new file mode 100644
index 0000000..469f484
--- /dev/null
+++ b/tests/suites/test_tools_identity.py
@@ -0,0 +1,13 @@
+from .common import simple_prompt_test
+
+SUITE = "test_tools_identity"
+
+
+def get_tests():
+ return [
+ simple_prompt_test(SUITE, "get_pubkey", "Fetch hex pubkey", "What is your public key in hex?", "nostr_pubkey"),
+ simple_prompt_test(SUITE, "get_npub", "Fetch npub", "What is your npub?", "nostr_npub"),
+ simple_prompt_test(SUITE, "agent_identity", "Agent identity tool", "Tell me about your identity", "agent_identity"),
+ simple_prompt_test(SUITE, "agent_version", "Agent version tool", "What version are you?", "agent_version"),
+ simple_prompt_test(SUITE, "admin_identity", "Admin identity tool", "Who is your administrator?", "admin_identity"),
+ ]
diff --git a/tests/suites/test_tools_memory.py b/tests/suites/test_tools_memory.py
new file mode 100644
index 0000000..7e86b98
--- /dev/null
+++ b/tests/suites/test_tools_memory.py
@@ -0,0 +1,23 @@
+from .common import simple_prompt_test
+
+SUITE = "test_tools_memory"
+
+
+def get_tests():
+ return [
+ simple_prompt_test(SUITE, "task_list", "Task list", "Show me your current task list", "task_list"),
+ simple_prompt_test(
+ SUITE,
+ "task_manage_add_remove",
+ "Task add/remove",
+ "Add a task test harness probe task then remove it",
+ "task_manage",
+ ),
+ simple_prompt_test(
+ SUITE,
+ "memory_save_recall",
+ "Save and recall memory",
+ "Save test harness probe to memory, then recall your memory",
+ "memory_save",
+ ),
+ ]
diff --git a/tests/suites/test_tools_nostr.py b/tests/suites/test_tools_nostr.py
new file mode 100644
index 0000000..d245a8f
--- /dev/null
+++ b/tests/suites/test_tools_nostr.py
@@ -0,0 +1,15 @@
+from .common import simple_prompt_test
+
+SUITE = "test_tools_nostr"
+
+
+def get_tests():
+ return [
+ simple_prompt_test(SUITE, "nostr_post_kind1", "Post kind1 note", "Post a test note saying Automated test post", "nostr_post"),
+ simple_prompt_test(SUITE, "nostr_query_recent", "Query recent notes", "Query the 3 most recent kind 1 notes from any author", "nostr_query"),
+ simple_prompt_test(SUITE, "nostr_my_events", "List my events", "List your recent events", "nostr_my_events"),
+ simple_prompt_test(SUITE, "nostr_relay_status", "Relay status", "What is the status of your relay connections?", "nostr_relay_status"),
+ simple_prompt_test(SUITE, "nostr_dm_send", "Send DM", "Send a test DM to yourself", "nostr_dm_send"),
+ simple_prompt_test(SUITE, "nostr_encode_npub", "Encode npub", "Encode your pubkey as an npub", "nostr_encode"),
+ simple_prompt_test(SUITE, "nostr_profile_get", "Profile lookup", "Look up your own Nostr profile", "nostr_profile_get"),
+ ]
diff --git a/tests/suites/test_tools_skills.py b/tests/suites/test_tools_skills.py
new file mode 100644
index 0000000..e4be9be
--- /dev/null
+++ b/tests/suites/test_tools_skills.py
@@ -0,0 +1,17 @@
+from .common import simple_prompt_test
+
+SUITE = "test_tools_skills"
+
+
+def get_tests():
+ return [
+ simple_prompt_test(SUITE, "skill_list", "List skills", "List your available skills", "skill_list"),
+ simple_prompt_test(SUITE, "trigger_list", "List triggers", "List your active triggers", "trigger_list"),
+ simple_prompt_test(
+ SUITE,
+ "skill_create_and_remove",
+ "Create and remove skill",
+ "Create a test skill called test-harness-probe with content Test skill then remove it",
+ "skill_create",
+ ),
+ ]
diff --git a/tests/suites/test_tools_system.py b/tests/suites/test_tools_system.py
new file mode 100644
index 0000000..76ec666
--- /dev/null
+++ b/tests/suites/test_tools_system.py
@@ -0,0 +1,19 @@
+from .common import simple_prompt_test
+
+SUITE = "test_tools_system"
+
+
+def get_tests():
+ return [
+ simple_prompt_test(SUITE, "tool_list", "Tool list", "List all your available tools", "tool_list"),
+ simple_prompt_test(SUITE, "model_get", "Current model", "What model are you currently using?", "model_get"),
+ simple_prompt_test(SUITE, "model_list", "Available models", "List available models", "model_list"),
+ simple_prompt_test(SUITE, "local_http_fetch", "HTTP fetch", "Fetch https://httpbin.org/get", "local_http_fetch"),
+ simple_prompt_test(
+ SUITE,
+ "config_store_recall",
+ "Store+recall config",
+ "Store a test config with d_tag test_harness_probe containing hello, then recall it",
+ "config_store",
+ ),
+ ]