M7: polish + E2E + docs (layout pass, theming, states, e2e driver, schema, setup.md, security)
This commit is contained in:
1 parent
0cc8b7aafe
commit
bf6bf7e8bd
26 files changed
+2225
-327
No files matched your search
@@ -0,0 +1,373 @@
|
||||
#!/usr/bin/env python3
|
||||
"""E2E driver: docs/13-testing.md §13.4 scenarios 1-12 against the live gateway.
|
||||
|
||||
Drives ws_probe.py (and the hermes CLI for cron) as subprocesses. For each
|
||||
scenario prints PASS / PARTIAL / SKIP / FAIL with a one-line reason, then a
|
||||
summary table. Exit 0 if no FAIL, 1 otherwise.
|
||||
|
||||
Usage::
|
||||
|
||||
hermes-agent/.venv/bin/python gateway-plugin/tests/e2e.py
|
||||
hermes-agent/.venv/bin/python gateway-plugin/tests/e2e.py --skip 3,5,7
|
||||
hermes-agent/.venv/bin/python gateway-plugin/tests/e2e.py --url ws://host:8790/ws
|
||||
|
||||
The token is read from $ANDROID_TOKEN, else hermes-agent/.env, else
|
||||
~/.hermes/.env. The gateway must already be running (this driver never
|
||||
starts or stops it). Idempotent: channels/jobs it creates are cleaned up
|
||||
even on failure, and leftover "e2e-*" channels/jobs from earlier runs are
|
||||
removed at start.
|
||||
|
||||
Scenario notes:
|
||||
3 (reasoning) and 5 (commentary) are model-dependent: they SKIP (not
|
||||
FAIL) when the current model does not emit reasoning / commentary.
|
||||
11 (push) and 12 (reconnect) are PARTIAL by design: the WS leg is
|
||||
automated, the device-notification / gateway-kill leg is manual.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
import struct
|
||||
import subprocess
|
||||
import sys
|
||||
import uuid
|
||||
import zlib
|
||||
from pathlib import Path
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
REPO = HERE.parent.parent
|
||||
PY = REPO / "hermes-agent" / ".venv" / "bin" / "python"
|
||||
PROBE = HERE / "ws_probe.py"
|
||||
HERMES = REPO / "hermes-agent" / ".venv" / "bin" / "hermes"
|
||||
DEFAULT_URL = "ws://127.0.0.1:8790/ws"
|
||||
|
||||
PASS, PARTIAL, SKIP, FAIL = "PASS", "PARTIAL", "SKIP", "FAIL"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def find_token(cli_token: str) -> str:
|
||||
if cli_token:
|
||||
return cli_token
|
||||
env = os.getenv("ANDROID_TOKEN")
|
||||
if env:
|
||||
return env
|
||||
for p in (REPO / "hermes-agent" / ".env", Path.home() / ".hermes" / ".env"):
|
||||
try:
|
||||
for line in p.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if line.startswith("ANDROID_TOKEN="):
|
||||
return line.split("=", 1)[1].strip().strip('"').strip("'")
|
||||
except OSError:
|
||||
pass
|
||||
return ""
|
||||
|
||||
|
||||
def run_probe(env, url, token, *args, timeout=300):
|
||||
cmd = [str(PY), str(PROBE), "--url", url, "--token", token, *args]
|
||||
p = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout, env=env)
|
||||
return p.returncode, p.stdout, p.stderr
|
||||
|
||||
|
||||
def run_hermes(env, *args, timeout=120):
|
||||
cmd = [str(HERMES), *args]
|
||||
p = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout, env=env)
|
||||
return p.returncode, p.stdout, p.stderr
|
||||
|
||||
|
||||
def write_png(path: Path, color, size: int = 200) -> None:
|
||||
"""Write a solid-color RGB PNG using only the stdlib (no PIL needed)."""
|
||||
raw = b"".join(b"\x00" + bytes(color) * size for _ in range(size))
|
||||
|
||||
def chunk(tag: bytes, data: bytes) -> bytes:
|
||||
return (struct.pack(">I", len(data)) + tag + data
|
||||
+ struct.pack(">I", zlib.crc32(tag + data) & 0xFFFFFFFF))
|
||||
|
||||
ihdr = struct.pack(">IIBBBBB", size, size, 8, 2, 0, 0, 0)
|
||||
path.write_bytes(
|
||||
b"\x89PNG\r\n\x1a\n"
|
||||
+ chunk(b"IHDR", ihdr)
|
||||
+ chunk(b"IDAT", zlib.compress(raw))
|
||||
+ chunk(b"IEND", b"")
|
||||
)
|
||||
|
||||
|
||||
def parse_created_chat_id(out: str) -> str | None:
|
||||
m = re.search(r"== channel created: (\S+)", out)
|
||||
return m.group(1) if m else None
|
||||
|
||||
|
||||
def sweep_leftovers(env, url, token) -> None:
|
||||
"""Remove e2e-* channels / cron jobs left behind by earlier runs."""
|
||||
rc, out, _ = run_probe(env, url, token, "--channel-list")
|
||||
if rc == 0:
|
||||
for m in re.finditer(r"== channel: (\S+) name='(e2e-[^']*)'", out):
|
||||
chat_id, name = m.group(1), m.group(2)
|
||||
print(f" cleanup: removing leftover channel {chat_id} ({name})")
|
||||
run_probe(env, url, token, "--channel-delete", chat_id)
|
||||
rc, out, _ = run_hermes(env, "cron", "list")
|
||||
if rc == 0:
|
||||
for m in re.finditer(
|
||||
r"(\S+) \[(?:active|paused)\]\s*\n\s*Name:\s+(e2e-cron-[^ \n]*)", out
|
||||
):
|
||||
job_id, name = m.group(1), m.group(2)
|
||||
print(f" cleanup: removing leftover cron job {job_id} ({name})")
|
||||
run_hermes(env, "cron", "remove", job_id)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Scenarios (docs/13-testing.md §13.4)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def s1_pair(env, url, token):
|
||||
rc, _, _ = run_probe(env, url, "definitely-wrong-token", "--authfail", "--send", "")
|
||||
if rc != 0:
|
||||
return FAIL, f"wrong token was not rejected (rc={rc})"
|
||||
rc, _, _ = run_probe(env, url, token, "--send", "")
|
||||
if rc != 0:
|
||||
return FAIL, f"valid token did not pair (rc={rc})"
|
||||
return PASS, "wrong token rejected; hello.ack on valid token"
|
||||
|
||||
|
||||
def s2_text(env, url, token):
|
||||
prompt = "Write a short poem about the ocean, at least 8 lines"
|
||||
rc, _, _ = run_probe(env, url, token, "--send", prompt,
|
||||
"--assert-turn", "--timeout", "120")
|
||||
if rc == 0:
|
||||
return PASS, "message.start -> >=1 message.update -> message.stop"
|
||||
if rc == 10:
|
||||
return FAIL, "no ordered start/update/stop segment"
|
||||
return FAIL, f"probe rc={rc}"
|
||||
|
||||
|
||||
def s3_reasoning(env, url, token):
|
||||
prompt = "Work out step by step: what is 17 * 23? Show your reasoning."
|
||||
rc, _, _ = run_probe(env, url, token, "--send", prompt,
|
||||
"--assert-reasoning", "--timeout", "120")
|
||||
if rc == 0:
|
||||
return PASS, "final message.stop carries non-empty reasoning"
|
||||
if rc == 11:
|
||||
return SKIP, "model returned no reasoning (model-dependent)"
|
||||
return FAIL, f"probe rc={rc}"
|
||||
|
||||
|
||||
def s4_tools(env, url, token):
|
||||
prompt = ("List the files in your current working directory using your "
|
||||
"shell tool, then tell me how many there are")
|
||||
rc, _, _ = run_probe(env, url, token, "--send", prompt,
|
||||
"--assert-tools", "--timeout", "150")
|
||||
if rc == 0:
|
||||
return PASS, "tool.start with a matching tool.end"
|
||||
if rc == 12:
|
||||
return FAIL, "no tool.start/tool.end pair"
|
||||
return FAIL, f"probe rc={rc}"
|
||||
|
||||
|
||||
def s5_commentary(env, url, token):
|
||||
prompt = ("Research task: (1) use your shell tool to list the top-level "
|
||||
"directories in /tmp, (2) report your findings so far, "
|
||||
"(3) use your shell tool to count files in /tmp, "
|
||||
"(4) report those findings too, (5) give a final summary of both")
|
||||
rc, _, _ = run_probe(env, url, token, "--send", prompt,
|
||||
"--assert-commentary", "--timeout", "150")
|
||||
if rc == 0:
|
||||
return PASS, "commentary frame observed"
|
||||
if rc == 13:
|
||||
return SKIP, "no commentary (model/agent-dependent per M2)"
|
||||
return FAIL, f"probe rc={rc}"
|
||||
|
||||
|
||||
def s6_channels(env, url, token):
|
||||
name = f"e2e-chan-{uuid.uuid4().hex[:6]}"
|
||||
rc, out, _ = run_probe(env, url, token, "--channel-create", name)
|
||||
if rc != 0:
|
||||
return FAIL, f"channel.create failed (rc={rc})"
|
||||
chat_id = parse_created_chat_id(out)
|
||||
if not chat_id:
|
||||
return FAIL, "channel.created received but chat_id not parseable"
|
||||
rc, _, _ = run_probe(env, url, token, "--channel-delete", chat_id)
|
||||
if rc != 0:
|
||||
run_probe(env, url, token, "--channel-delete", chat_id) # best-effort
|
||||
return FAIL, f"channel.delete failed (rc={rc})"
|
||||
return PASS, f"created {chat_id} + deleted (cleanup)"
|
||||
|
||||
|
||||
def s7_cron(env, url, token):
|
||||
chan_name = f"e2e-cron-chan-{uuid.uuid4().hex[:6]}"
|
||||
rc, out, _ = run_probe(env, url, token, "--channel-create", chan_name)
|
||||
if rc != 0:
|
||||
return SKIP, f"could not create cron target channel (rc={rc})"
|
||||
chat_id = parse_created_chat_id(out)
|
||||
if not chat_id:
|
||||
return FAIL, "channel.created received but chat_id not parseable"
|
||||
job_name = f"e2e-cron-{uuid.uuid4().hex[:6]}"
|
||||
deliver = f"android:{chat_id}"
|
||||
rc, out, err = run_hermes(
|
||||
env, "cron", "create", "1m",
|
||||
"Reply with exactly: e2e cron delivery OK",
|
||||
"--deliver", deliver, "--name", job_name,
|
||||
)
|
||||
job_id = None
|
||||
if rc == 0:
|
||||
m = re.search(r"Created job: (\S+)", out)
|
||||
job_id = m.group(1) if m else None
|
||||
try:
|
||||
if rc != 0:
|
||||
return SKIP, f"hermes cron create failed: {(err or out).strip()[:120]}"
|
||||
rc, out, _ = run_probe(env, url, token, "--watch", chat_id,
|
||||
"--timeout", "330", timeout=400)
|
||||
if rc == 0:
|
||||
return PASS, f"one-shot cron job fired; message landed in {chat_id}"
|
||||
return FAIL, f"no message in {chat_id} within 330s (probe rc={rc})"
|
||||
finally:
|
||||
if job_id:
|
||||
run_hermes(env, "cron", "remove", job_id)
|
||||
else:
|
||||
# create succeeded but the id was not parseable: find by name.
|
||||
_, list_out, _ = run_hermes(env, "cron", "list")
|
||||
m = re.search(r"(\S+) \[active\]\s*\n\s*Name:\s+" + re.escape(job_name),
|
||||
list_out)
|
||||
if m:
|
||||
run_hermes(env, "cron", "remove", m.group(1))
|
||||
run_probe(env, url, token, "--channel-delete", chat_id)
|
||||
|
||||
|
||||
def s8_search(env, url, token):
|
||||
marker = f"e2emarker{uuid.uuid4().hex[:8]}"
|
||||
rc, _, _ = run_probe(env, url, token, "--send",
|
||||
f"Remember this marker phrase: {marker}. "
|
||||
"Just acknowledge it briefly.",
|
||||
"--timeout", "120")
|
||||
if rc != 0:
|
||||
return FAIL, f"setup message failed (rc={rc})"
|
||||
rc, _, _ = run_probe(env, url, token, "--send", "", "--search", marker)
|
||||
if rc == 0:
|
||||
return PASS, f"search for {marker!r} returned >=1 hit"
|
||||
if rc == 14:
|
||||
return FAIL, f"search for {marker!r} returned 0 hits"
|
||||
return FAIL, f"probe rc={rc}"
|
||||
|
||||
|
||||
def s9_media_in(env, url, token):
|
||||
png = Path(f"/tmp/e2e_in_{uuid.uuid4().hex[:6]}.png")
|
||||
write_png(png, (30, 120, 220))
|
||||
try:
|
||||
rc, _, _ = run_probe(env, url, token, "--upload", str(png),
|
||||
"--send", "describe this image briefly",
|
||||
"--timeout", "120")
|
||||
if rc == 0:
|
||||
return PASS, "upload + vision reply (final message)"
|
||||
if rc == 8:
|
||||
return FAIL, "media upload failed"
|
||||
if rc == 7:
|
||||
return FAIL, "no final message after upload"
|
||||
return FAIL, f"probe rc={rc}"
|
||||
finally:
|
||||
png.unlink(missing_ok=True)
|
||||
|
||||
|
||||
def s10_media_out(env, url, token):
|
||||
prompt = ("Create a 100x100 orange square PNG in /tmp with your tools. "
|
||||
"In your final reply, include the MEDIA:/absolute/path tag for "
|
||||
"that file so it is delivered to me.")
|
||||
rc, out, _ = run_probe(env, url, token, "--send", prompt,
|
||||
"--pull-offer", "--timeout", "150")
|
||||
m = re.search(r"== pulled (\d+) bytes", out)
|
||||
if rc == 0 and m and int(m.group(1)) > 0:
|
||||
return PASS, f"media.offer pulled ({m.group(1)} bytes)"
|
||||
if rc == 9:
|
||||
return FAIL, "media pull failed"
|
||||
return FAIL, "no media.offer pulled (agent did not deliver an image)"
|
||||
|
||||
|
||||
def s11_push(env, url, token):
|
||||
rc, out, _ = run_probe(env, url, token, "--fcm-token", "test-token-123",
|
||||
"--fcm-reg", "--send", "")
|
||||
if rc != 0:
|
||||
return FAIL, f"probe rc={rc}"
|
||||
if "<- error" in out:
|
||||
return FAIL, "error frame after fcm.register"
|
||||
return PARTIAL, ("fcm.register accepted (no error frame); "
|
||||
"device-notification leg is manual")
|
||||
|
||||
|
||||
def s12_sync(env, url, token):
|
||||
rc, out, _ = run_probe(env, url, token, "--sync", "0")
|
||||
if rc == 0 and "sync done" in out:
|
||||
return PARTIAL, ("sync replay + sync.done verified; "
|
||||
"gateway-kill/restart leg is manual")
|
||||
if rc == 8:
|
||||
return FAIL, "sync failed"
|
||||
return FAIL, f"probe rc={rc}"
|
||||
|
||||
|
||||
SCENARIOS = [
|
||||
(1, "pair", s1_pair),
|
||||
(2, "text round-trip", s2_text),
|
||||
(3, "reasoning", s3_reasoning),
|
||||
(4, "tools", s4_tools),
|
||||
(5, "commentary", s5_commentary),
|
||||
(6, "channels", s6_channels),
|
||||
(7, "cron delivery", s7_cron),
|
||||
(8, "search", s8_search),
|
||||
(9, "media in", s9_media_in),
|
||||
(10, "media out", s10_media_out),
|
||||
(11, "push", s11_push),
|
||||
(12, "reconnect/sync", s12_sync),
|
||||
]
|
||||
|
||||
|
||||
def main() -> int:
|
||||
p = argparse.ArgumentParser(
|
||||
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
|
||||
)
|
||||
p.add_argument("--url", default=os.getenv("ANDROID_WS_URL", DEFAULT_URL))
|
||||
p.add_argument("--token", default="")
|
||||
p.add_argument("--skip", default="",
|
||||
help="comma-separated scenario numbers to skip (e.g. 3,5,7)")
|
||||
args = p.parse_args()
|
||||
|
||||
token = find_token(args.token)
|
||||
if not token:
|
||||
print("!! ANDROID_TOKEN not found (env, hermes-agent/.env, or ~/.hermes/.env)")
|
||||
return 1
|
||||
skip = {int(x) for x in args.skip.split(",") if x.strip()}
|
||||
|
||||
env = dict(os.environ)
|
||||
env["ANDROID_TOKEN"] = token
|
||||
|
||||
print(f"== e2e: url={args.url} token={token[:6]}…")
|
||||
sweep_leftovers(env, args.url, token)
|
||||
|
||||
results = []
|
||||
for num, name, fn in SCENARIOS:
|
||||
if num in skip:
|
||||
results.append((num, name, SKIP, "skipped by --skip"))
|
||||
print(f"[{num:2d}] {name:<18} {SKIP:<7} skipped by --skip")
|
||||
continue
|
||||
print(f"[{num:2d}] {name:<18} running…", flush=True)
|
||||
try:
|
||||
status, reason = fn(env, args.url, token)
|
||||
except Exception as e:
|
||||
status, reason = FAIL, f"driver error: {e}"
|
||||
results.append((num, name, status, reason))
|
||||
print(f"[{num:2d}] {name:<18} {status:<7} {reason}")
|
||||
|
||||
print()
|
||||
print("=" * 78)
|
||||
print(f"{'#':<3} {'scenario':<18} {'status':<8} reason")
|
||||
print("-" * 78)
|
||||
for num, name, status, reason in results:
|
||||
print(f"{num:<3} {name:<18} {status:<8} {reason}")
|
||||
print("-" * 78)
|
||||
counts = {s: sum(1 for r in results if r[2] == s)
|
||||
for s in (PASS, PARTIAL, SKIP, FAIL)}
|
||||
print(f"total: {len(results)} PASS={counts[PASS]} PARTIAL={counts[PARTIAL]} "
|
||||
f"SKIP={counts[SKIP]} FAIL={counts[FAIL]}")
|
||||
return 1 if counts[FAIL] else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in new issue
Block a user