M1+M2: gateway core loop + agent transparency
M1 (gateway core loop / text round-trip): - WS server (ws_server.py): bind, hello auth (constant-time), hello.ack, heartbeat, connection registry - pairing.py: token generation, pairing store, QR payload - adapter.py: send() -> message frame; inbound message.send -> MessageEvent -> handle_message - app: Connect screen, GatewayClient (connect + reconnect), ChatScreen send/render, SecureStore (Android/Desktop) - tests/ws_probe.py: probe harness driving a real turn M2 (streaming + reasoning + tools + commentary): - protocol.py: M2 frame types (message.start/update/stop, tool.start/progress/end, commentary) - adapter.py: per-chat turn-state machine; classify outbound into frames; _split_reasoning; tool-line parsing - reasoning in streaming: capture via on_stream_delta hook (kind=reasoning, gated by plugins.stream_reasoning_deltas) with a FIFO barrier, attach to message.stop - app: live streaming bubble, ReasoningBlock (collapse + copy), ToolCard (Everything/Truncated/Nothing), dimmed commentary, typing - docs/14-milestones.md: M1/M2 marked done; reasoning note corrected
This commit is contained in:
1 parent
59acf66c89
commit
218c50d688
21 files changed
+3437
-129
No files matched your search
+762
-54
@@ -9,11 +9,17 @@ with a pairing token and talks to the agent over a single WS transport
|
||||
Zero new Python dependencies: ``websockets`` and ``httpx`` are hermes core
|
||||
deps. Zero hermes-core changes.
|
||||
|
||||
Milestone M0: this is a *skeleton* adapter. It registers the ``android``
|
||||
platform, resolves its configuration, and implements the abstract adapter
|
||||
contract as no-ops so that ``hermes gateway status`` lists ``android``. The
|
||||
WebSocket server, pairing, streaming, media, outbox, push, and search are
|
||||
wired in later milestones (see ``docs/14-milestones.md``).
|
||||
Milestone M1: the gateway core loop (text round-trip). The WS server binds
|
||||
and authenticates devices (``hello`` with constant-time token check), the
|
||||
adapter emits ``message`` frames from ``send()`` and turns inbound
|
||||
``message.send`` frames into ``MessageEvent``s for ``handle_message()``.
|
||||
|
||||
Milestone M2: agent transparency. ``send()``/``edit_message()`` are mapped to
|
||||
``message.start``/``message.update``/``message.stop`` (streaming), tool
|
||||
progress is classified into structured ``tool.start``/``tool.end`` frames,
|
||||
interim commentary becomes ``commentary`` frames, and the code-style
|
||||
reasoning prefix is split into a ``reasoning`` field. Media, outbox, push,
|
||||
and search land in later milestones (see ``docs/14-milestones.md``).
|
||||
|
||||
Configuration in config.yaml::
|
||||
|
||||
@@ -34,11 +40,15 @@ Or via environment variables (overrides config.yaml; secrets live in .env):
|
||||
ANDROID_PUSH_BACKEND, ANDROID_FCM_SERVICE_ACCOUNT, NTFY_TOPIC, ...
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from typing import Any, Dict, List, Optional
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from agent.secret_scope import UnscopedSecretError as _UnscopedSecretError
|
||||
from agent.secret_scope import get_secret as _scoped_get_secret
|
||||
@@ -79,6 +89,77 @@ from gateway.platforms.base import ( # noqa: E402
|
||||
MessageType,
|
||||
)
|
||||
from gateway.config import Platform # noqa: E402
|
||||
from hermes_constants import get_hermes_home # noqa: E402
|
||||
|
||||
from . import protocol # noqa: E402
|
||||
from .pairing import ( # noqa: E402
|
||||
DeviceRegistry,
|
||||
generate_token,
|
||||
pairing_url,
|
||||
qr_payload,
|
||||
)
|
||||
from .ws_server import WsServer # noqa: E402
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# M2 — reasoning capture (streaming)
|
||||
#
|
||||
# The gateway streams only ``content`` to the platform and suppresses the
|
||||
# final send (which would carry the prepended reasoning), so the model's
|
||||
# separate ``reasoning_content`` is otherwise lost in the streaming case.
|
||||
# hermes exposes a plugin ``on_stream_delta`` hook that fires reasoning
|
||||
# deltas with ``kind="reasoning"`` (gated by ``plugins.stream_reasoning_deltas``).
|
||||
# We accumulate those deltas here and attach the result to the turn's
|
||||
# ``message.stop`` frame. Single-chat for now (android:default), so a
|
||||
# module-level buffer suffices; it is reset at each turn start.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_reasoning_parts: List[str] = []
|
||||
_reasoning_lock = threading.Lock()
|
||||
# Barrier: set by the hook worker once it has processed the first content
|
||||
# delta (kind="text"). The worker drains a FIFO queue and reasoning deltas are
|
||||
# enqueued before content deltas, so at that point every reasoning delta has
|
||||
# already been appended -- a reliable "reasoning flushed" signal that avoids
|
||||
# racing message.stop against the async hook thread.
|
||||
_reasoning_flushed = threading.Event()
|
||||
|
||||
|
||||
def _on_stream_delta(**kwargs: Any) -> None:
|
||||
"""Plugin hook: capture reasoning deltas (kind="reasoning")."""
|
||||
kind = kwargs.get("kind")
|
||||
if kind == "reasoning":
|
||||
delta = kwargs.get("delta") or ""
|
||||
if delta:
|
||||
with _reasoning_lock:
|
||||
_reasoning_parts.append(delta)
|
||||
elif kind == "text":
|
||||
_reasoning_flushed.set()
|
||||
|
||||
|
||||
async def _wait_for_reasoning_flushed(timeout: float = 0.3) -> None:
|
||||
"""Wait (without blocking the event loop) until the hook worker has
|
||||
processed all reasoning deltas, or *timeout* seconds elapse."""
|
||||
loop = asyncio.get_running_loop()
|
||||
deadline = loop.time() + timeout
|
||||
while loop.time() < deadline:
|
||||
if _reasoning_flushed.is_set():
|
||||
return
|
||||
await asyncio.sleep(0.01)
|
||||
|
||||
|
||||
def _take_reasoning() -> str:
|
||||
"""Drain and return the accumulated reasoning (empty string if none)."""
|
||||
with _reasoning_lock:
|
||||
parts = _reasoning_parts[:]
|
||||
_reasoning_parts.clear()
|
||||
_reasoning_flushed.clear()
|
||||
return "".join(parts).strip()
|
||||
|
||||
|
||||
def _reset_reasoning() -> None:
|
||||
with _reasoning_lock:
|
||||
_reasoning_parts.clear()
|
||||
_reasoning_flushed.clear()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -88,6 +169,7 @@ from gateway.config import Platform # noqa: E402
|
||||
DEFAULT_HOST = "127.0.0.1"
|
||||
DEFAULT_PORT = 8790
|
||||
DEFAULT_HOME_CHANNEL = "android:default"
|
||||
DEFAULT_HOME_CHANNEL_NAME = "Default"
|
||||
DEFAULT_PUSH_BACKEND = "fcm"
|
||||
DEFAULT_OUTBOX_RETENTION_HOURS = 72
|
||||
DEFAULT_MAX_UPLOAD_BYTES = 100 * 1024 * 1024 # 100 MB
|
||||
@@ -97,6 +179,211 @@ def _truthy(value: Optional[str]) -> bool:
|
||||
return (value or "").strip().lower() in {"1", "true", "yes", "on"}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# M2 — turn state + outbound classification
|
||||
#
|
||||
# The main gateway delivers through the legacy callback path: the stream
|
||||
# consumer calls ``send()`` (first bubble of a segment) and ``edit_message()``
|
||||
# (updates), tool progress flows through ``send()``/``edit_message()`` of an
|
||||
# accumulated line buffer, and interim commentary arrives as a plain
|
||||
# ``send()``. We classify each outbound call into a structured frame using a
|
||||
# per-chat turn state machine + content markers:
|
||||
#
|
||||
# * ``metadata["expect_edits"] is True`` -> streaming segment start
|
||||
# * ``metadata["notify"] is True`` -> final message (or fallback final)
|
||||
# * tool-progress line format -> tool.start / tool.end
|
||||
# * anything else -> commentary
|
||||
#
|
||||
# Verified empirically against the live gateway with ``tests/ws_probe.py``.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Streaming cursor the gateway appends to in-progress edits (" ▉"). Stripped
|
||||
# before we forward text to the app (the app renders its own live indicator).
|
||||
_STREAMING_CURSOR = " ▉"
|
||||
|
||||
# Code-style reasoning prefix (gateway/run.py, reasoning_style="code"):
|
||||
# "💭 **Reasoning:**\n```\n<reasoning>\n```\n\n<response>"
|
||||
_REASONING_PREFIX = "💭 **Reasoning:**\n```\n"
|
||||
_REASONING_CLOSE = "\n```\n\n"
|
||||
|
||||
# A gateway tool-progress line begins with a (non-ASCII) tool emoji.
|
||||
_TOOL_LINE_RE = re.compile(r"^(\S+)\s+(.+)$")
|
||||
_TOOL_NAME_PREVIEW_RE = re.compile(r'^(\S+):\s*"(.*)"\s*$')
|
||||
_TOOL_NAME_BARE_RE = re.compile(r"^(\S+)\.\.\.\s*$")
|
||||
_TOOL_NAME_ARGS_RE = re.compile(r"^(\S+)\(([^)]*)\)\s*$")
|
||||
# Terminal code block: "💻 terminal\n```\n<cmd>\n```"
|
||||
_TOOL_CODEBLOCK_HEAD_RE = re.compile(r"^(\S+)\s+(\S+)\s*$")
|
||||
|
||||
# Reverse map of the gateway's friendly tool verbs (agent/display.py
|
||||
# _TOOL_VERBS) so a verb-form line ("🔍 Searching the web for …") can be
|
||||
# recovered to a structured (tool_name, preview). Longest-first matching is
|
||||
# done at parse time. Verbs shared by several tools map to the most common.
|
||||
_VERB_TO_TOOL: Dict[str, str] = {
|
||||
"Searching the web": "web_search",
|
||||
"Searching files": "search_files",
|
||||
"Searching past sessions": "session_search",
|
||||
"Running code": "execute_code",
|
||||
"Running": "terminal",
|
||||
"Reading skill": "skill_view",
|
||||
"Reading": "read_file",
|
||||
"Writing": "write_file",
|
||||
"Editing": "patch",
|
||||
"Browsing": "browser_navigate",
|
||||
"Clicking": "browser_click",
|
||||
"Typing": "browser_type",
|
||||
"Generating image": "image_generate",
|
||||
"Generating video": "video_generate",
|
||||
"Generating speech": "text_to_speech",
|
||||
"Looking at the image": "vision_analyze",
|
||||
"Listing skills": "skills_list",
|
||||
"Updating skill": "skill_manage",
|
||||
"Updating memory": "memory",
|
||||
"Updating tasks": "todo",
|
||||
"Delegating": "delegate_task",
|
||||
"Scheduling": "cronjob",
|
||||
"Asking": "clarify",
|
||||
}
|
||||
# Verbs that take a " for " connector before the preview.
|
||||
_VERB_FOR_CONNECTOR = {"web_search", "search_files"}
|
||||
|
||||
|
||||
def _mint_message_id() -> str:
|
||||
return f"m_{uuid.uuid4().hex[:16]}"
|
||||
|
||||
|
||||
def _thread_id_from_metadata(metadata: Optional[Dict[str, Any]]) -> Optional[str]:
|
||||
if not metadata:
|
||||
return None
|
||||
tid = metadata.get("thread_id")
|
||||
if isinstance(tid, str) and tid:
|
||||
return tid
|
||||
return None
|
||||
|
||||
|
||||
def _strip_streaming_cursor(text: str) -> str:
|
||||
if text and text.endswith(_STREAMING_CURSOR):
|
||||
return text[: -len(_STREAMING_CURSOR)]
|
||||
return text
|
||||
|
||||
|
||||
def _split_reasoning(text: str) -> Tuple[Optional[str], str]:
|
||||
"""Split a code-style reasoning prefix off the front of *text*.
|
||||
|
||||
Returns ``(reasoning, body)``; ``reasoning`` is ``None`` when no prefix is
|
||||
present (reasoning off / no reasoning / non-code style). Best-effort parse
|
||||
of a stable, gateway-owned format: on any mismatch the fallback is
|
||||
``(None, full text)`` so the answer still renders.
|
||||
"""
|
||||
if not text or not text.startswith(_REASONING_PREFIX):
|
||||
return None, text
|
||||
close_idx = text.find(_REASONING_CLOSE, len(_REASONING_PREFIX))
|
||||
if close_idx == -1:
|
||||
return None, text
|
||||
reasoning = text[len(_REASONING_PREFIX):close_idx]
|
||||
body = text[close_idx + len(_REASONING_CLOSE):]
|
||||
return reasoning, body
|
||||
|
||||
|
||||
def _parse_tool_line(line: str) -> Optional[Tuple[str, Optional[str]]]:
|
||||
"""Parse a single gateway tool-progress line into ``(name, preview)``.
|
||||
|
||||
Returns ``None`` when the line is not a tool line. The gateway formats
|
||||
tool lines as ``<emoji> <name>: "<preview>"``, ``<emoji> <name>...``,
|
||||
``<emoji> <name>(keys)``, or a friendly verb phrase (``<emoji> <verb> …``).
|
||||
The verb form is lossy (no tool name), so we surface the verb as the name.
|
||||
"""
|
||||
line = line.strip()
|
||||
if not line:
|
||||
return None
|
||||
m = _TOOL_LINE_RE.match(line)
|
||||
if not m:
|
||||
return None
|
||||
emoji, rest = m.group(1), m.group(2)
|
||||
if emoji.isascii():
|
||||
return None # a tool line always leads with a non-ASCII emoji
|
||||
mp = _TOOL_NAME_PREVIEW_RE.match(rest)
|
||||
if mp:
|
||||
return mp.group(1), mp.group(2)
|
||||
mb = _TOOL_NAME_BARE_RE.match(rest)
|
||||
if mb:
|
||||
return mb.group(1), None
|
||||
ma = _TOOL_NAME_ARGS_RE.match(rest)
|
||||
if ma:
|
||||
return ma.group(1), None
|
||||
# Friendly verb phrase: reverse-map to (tool_name, preview).
|
||||
verb_parsed = _parse_verb_phrase(rest)
|
||||
if verb_parsed is not None:
|
||||
return verb_parsed
|
||||
# Unrecognised: use the phrase as the label.
|
||||
return rest, None
|
||||
|
||||
|
||||
def _parse_verb_phrase(phrase: str) -> Optional[Tuple[str, Optional[str]]]:
|
||||
"""Reverse-map a friendly verb phrase to ``(tool_name, preview)``.
|
||||
|
||||
Matches the longest verb first so "Running code" wins over "Running".
|
||||
Returns ``None`` when no known verb leads the phrase.
|
||||
"""
|
||||
for verb in sorted(_VERB_TO_TOOL, key=len, reverse=True):
|
||||
tool = _VERB_TO_TOOL[verb]
|
||||
if phrase == verb:
|
||||
return tool, None
|
||||
if tool in _VERB_FOR_CONNECTOR and phrase.startswith(verb + " for "):
|
||||
return tool, phrase[len(verb) + len(" for "):].strip() or None
|
||||
if phrase.startswith(verb + " "):
|
||||
return tool, phrase[len(verb) + 1:].strip() or None
|
||||
return None
|
||||
|
||||
|
||||
def _extract_code_block(content: str) -> Optional[str]:
|
||||
"""Return the first fenced code block's body in *content*, else ``None``.
|
||||
|
||||
Used to recover the terminal command from a tool-progress code block
|
||||
(``<emoji> terminal`` head line + fenced command).
|
||||
"""
|
||||
m = re.search(r"```[^\n]*\n(.*?)\n```", content, re.DOTALL)
|
||||
if m:
|
||||
return m.group(1).strip() or None
|
||||
return None
|
||||
|
||||
|
||||
def _is_tool_progress(content: str) -> bool:
|
||||
"""Heuristic: does *content* look like gateway tool-progress line(s)?
|
||||
|
||||
Tool progress is delivered as one or more lines, each led by a tool emoji
|
||||
(or a terminal code block). Commentary is free-form prose. We classify on
|
||||
the first non-empty line; subsequent lines of the same bubble are tracked
|
||||
by message id, not re-classified.
|
||||
"""
|
||||
if not content:
|
||||
return False
|
||||
lines = [ln for ln in content.splitlines() if ln.strip()]
|
||||
if not lines:
|
||||
return False
|
||||
first = lines[0].strip()
|
||||
# Terminal code block: "<emoji> terminal" then a fenced command.
|
||||
if len(lines) > 1 and lines[1].strip().startswith("```"):
|
||||
return _TOOL_CODEBLOCK_HEAD_RE.match(first) is not None
|
||||
return _parse_tool_line(first) is not None
|
||||
|
||||
|
||||
@dataclass
|
||||
class _TurnState:
|
||||
"""Per-chat turn state for outbound frame classification (M2)."""
|
||||
|
||||
active: bool = False
|
||||
# message_id of the currently streaming segment (message.start open).
|
||||
stream_id: Optional[str] = None
|
||||
# message_id of the current tool-progress bubble (editable line buffer).
|
||||
tool_msg_id: Optional[str] = None
|
||||
# Monotonic per-turn tool counter (start -> end correlation).
|
||||
tool_index: int = 0
|
||||
# Index of the most recently started tool (awaiting tool.end).
|
||||
open_tool_index: Optional[int] = None
|
||||
# Tool lines already emitted as tool.start (dedup across edits).
|
||||
seen_tool_lines: set = field(default_factory=set)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Passive / config probes (called from status displays -- no side effects)
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -147,19 +434,26 @@ def _env_enablement() -> Optional[dict]:
|
||||
if not token:
|
||||
return None
|
||||
|
||||
seed: Dict[str, Any] = {
|
||||
"host": os.getenv("ANDROID_WS_HOST", "").strip() or DEFAULT_HOST,
|
||||
"port": _parse_port(os.getenv("ANDROID_WS_PORT", "")),
|
||||
"push_backend": (
|
||||
os.getenv("ANDROID_PUSH_BACKEND", "").strip().lower()
|
||||
or DEFAULT_PUSH_BACKEND
|
||||
),
|
||||
}
|
||||
home = os.getenv("ANDROID_HOME_CHANNEL", "").strip() or DEFAULT_HOME_CHANNEL
|
||||
seed["home_channel"] = {
|
||||
"chat_id": home,
|
||||
"name": os.getenv("ANDROID_HOME_CHANNEL_NAME", "").strip() or "Default",
|
||||
}
|
||||
# Seed ONLY explicitly-set env vars: the core commits this seed on top of
|
||||
# config.yaml (``extra.update(seed)``), so default values here would
|
||||
# clobber user YAML. Unset keys fall through to config.yaml / adapter
|
||||
# defaults.
|
||||
seed: Dict[str, Any] = {}
|
||||
host = os.getenv("ANDROID_WS_HOST", "").strip()
|
||||
if host:
|
||||
seed["host"] = host
|
||||
port_raw = os.getenv("ANDROID_WS_PORT", "").strip()
|
||||
if port_raw:
|
||||
seed["port"] = _parse_port(port_raw)
|
||||
push = os.getenv("ANDROID_PUSH_BACKEND", "").strip().lower()
|
||||
if push:
|
||||
seed["push_backend"] = push
|
||||
home = os.getenv("ANDROID_HOME_CHANNEL", "").strip()
|
||||
if home:
|
||||
seed["home_channel"] = {
|
||||
"chat_id": home,
|
||||
"name": os.getenv("ANDROID_HOME_CHANNEL_NAME", "").strip() or DEFAULT_HOME_CHANNEL_NAME,
|
||||
}
|
||||
return seed
|
||||
|
||||
|
||||
@@ -215,7 +509,7 @@ async def _standalone_send(
|
||||
|
||||
The outbox is served by the *running* gateway, so standalone delivery
|
||||
while the gateway process is fully down is best-effort only (see
|
||||
``docs/00-overview.md`` "Out of scope"). For M0 this is a stub that
|
||||
``docs/00-overview.md`` "Out of scope"). For M1 this is a stub that
|
||||
reports the gateway is required; the real implementation lands with the
|
||||
outbox (M3/M5).
|
||||
"""
|
||||
@@ -228,14 +522,14 @@ async def _standalone_send(
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Interactive setup (hermes gateway setup flow) -- full version in M1
|
||||
# Interactive setup (hermes gateway setup flow)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def interactive_setup() -> None:
|
||||
"""Prompt for the pairing token / host / port / push backend.
|
||||
|
||||
M0: minimal. M1 adds token generation, QR payload, and a live ``hello``
|
||||
connectivity test.
|
||||
M1: token generation, host/port/push prompts, and the pairing QR payload
|
||||
(``iris://pair?...``) + app URL printed for the Connect screen.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.config import (
|
||||
@@ -253,7 +547,7 @@ def interactive_setup() -> None:
|
||||
print_info("📱 Android / Desktop (Iris x Hermes)")
|
||||
token = get_env_value("ANDROID_TOKEN") or ""
|
||||
if not token:
|
||||
generated = uuid.uuid4().hex + uuid.uuid4().hex # 64 hex chars
|
||||
generated = generate_token()
|
||||
save_env_value("ANDROID_TOKEN", generated)
|
||||
print_success(f"Generated pairing token: {generated}")
|
||||
print_warning("Keep this secret -- the app presents it on connect.")
|
||||
@@ -267,6 +561,19 @@ def interactive_setup() -> None:
|
||||
backend = prompt("Push backend (fcm/ntfy)", default=get_env_value("ANDROID_PUSH_BACKEND") or DEFAULT_PUSH_BACKEND)
|
||||
save_env_value("ANDROID_PUSH_BACKEND", (backend or DEFAULT_PUSH_BACKEND).strip().lower())
|
||||
|
||||
# Pairing payload for the app's Connect screen (QR / manual entry).
|
||||
try:
|
||||
from hermes_cli.config import print_code
|
||||
url = pairing_url(host or DEFAULT_HOST, _parse_port(port))
|
||||
payload = qr_payload(host or DEFAULT_HOST, _parse_port(port), token)
|
||||
print_info("Pair your device (scan with the app or enter on the Connect screen):")
|
||||
print_code(payload)
|
||||
print_info(f"Server URL: {url}")
|
||||
except Exception:
|
||||
url = pairing_url(host or DEFAULT_HOST, _parse_port(port))
|
||||
print_info(f"Pairing URL: {qr_payload(host or DEFAULT_HOST, _parse_port(port), token)}")
|
||||
print_info(f"Server URL: {url}")
|
||||
|
||||
print_success("Android configuration saved to ~/.hermes/.env")
|
||||
print_info("Restart the gateway for changes to take effect: hermes gateway restart")
|
||||
|
||||
@@ -278,10 +585,10 @@ def interactive_setup() -> None:
|
||||
class AndroidAdapter(BasePlatformAdapter):
|
||||
"""WebSocket-backed adapter for the native Iris Android / Desktop app.
|
||||
|
||||
M0: skeleton. Implements the abstract adapter contract as no-ops and
|
||||
resolves configuration. The WebSocket server, connection registry,
|
||||
pairing, streaming, media, outbox, push, and search are added in later
|
||||
milestones.
|
||||
M1: the WS server (``ws_server.WsServer``) authenticates devices with the
|
||||
pairing token, the connection registry tracks live sockets, ``send()``
|
||||
emits ``message`` frames, and inbound ``message.send`` frames become
|
||||
``MessageEvent``s for ``handle_message()``.
|
||||
"""
|
||||
|
||||
def __init__(self, config, **kwargs):
|
||||
@@ -294,7 +601,6 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
self.host = os.getenv("ANDROID_WS_HOST", "").strip() or extra.get("host", DEFAULT_HOST)
|
||||
self.port = _parse_port(os.getenv("ANDROID_WS_PORT", "") or str(extra.get("port", DEFAULT_PORT)))
|
||||
self.token = _get_scoped_secret("ANDROID_TOKEN") or extra.get("token", "")
|
||||
self.home_channel = extra.get("home_channel", DEFAULT_HOME_CHANNEL)
|
||||
self.push_backend = (
|
||||
os.getenv("ANDROID_PUSH_BACKEND", "").strip().lower()
|
||||
or extra.get("push_backend", DEFAULT_PUSH_BACKEND)
|
||||
@@ -306,6 +612,25 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
extra.get("max_upload_bytes", DEFAULT_MAX_UPLOAD_BYTES)
|
||||
)
|
||||
|
||||
# Home channel: the core hook turns the env-seeded ``home_channel``
|
||||
# dict into a HomeChannel dataclass on the config; config.yaml may
|
||||
# also put it in extra (dict or bare string).
|
||||
home = getattr(config, "home_channel", None)
|
||||
if home is not None and getattr(home, "chat_id", None):
|
||||
self.home_channel = str(home.chat_id)
|
||||
self.home_channel_name = str(getattr(home, "name", "") or DEFAULT_HOME_CHANNEL_NAME)
|
||||
else:
|
||||
hc = extra.get("home_channel")
|
||||
if isinstance(hc, dict) and hc.get("chat_id"):
|
||||
self.home_channel = str(hc["chat_id"])
|
||||
self.home_channel_name = str(hc.get("name") or DEFAULT_HOME_CHANNEL_NAME)
|
||||
elif isinstance(hc, str) and hc.strip():
|
||||
self.home_channel = hc.strip()
|
||||
self.home_channel_name = DEFAULT_HOME_CHANNEL_NAME
|
||||
else:
|
||||
self.home_channel = DEFAULT_HOME_CHANNEL
|
||||
self.home_channel_name = DEFAULT_HOME_CHANNEL_NAME
|
||||
|
||||
# TLS (optional)
|
||||
self.ws_cert = _get_scoped_secret("ANDROID_WS_CERT") or extra.get("ws_cert", "")
|
||||
self.ws_key = _get_scoped_secret("ANDROID_WS_KEY") or extra.get("ws_key", "")
|
||||
@@ -317,10 +642,19 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
)
|
||||
self.allow_all = _truthy(os.getenv("ANDROID_ALLOW_ALL_USERS"))
|
||||
|
||||
# Runtime state (populated by the WS server in M1)
|
||||
self._ws_server = None
|
||||
self._connections: Dict[str, Any] = {}
|
||||
# Runtime state
|
||||
self._devices = DeviceRegistry(get_hermes_home() / "android" / "devices.db")
|
||||
self._ws_server = WsServer(self, self._devices)
|
||||
self._connected = False
|
||||
# M2: per-chat turn state for outbound frame classification.
|
||||
self._turns: Dict[str, _TurnState] = {}
|
||||
|
||||
def _turn_state(self, chat_id: str) -> _TurnState:
|
||||
st = self._turns.get(chat_id)
|
||||
if st is None:
|
||||
st = _TurnState()
|
||||
self._turns[chat_id] = st
|
||||
return st
|
||||
|
||||
@property
|
||||
def name(self) -> str:
|
||||
@@ -329,12 +663,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
# ── Connection lifecycle ──────────────────────────────────────────────
|
||||
|
||||
async def connect(self, *, is_reconnect: bool = False) -> bool:
|
||||
"""Bring the platform up.
|
||||
|
||||
M0: no WebSocket server yet -- just validate config and mark
|
||||
connected so ``hermes gateway status`` reflects the platform. M1
|
||||
starts the ``websockets`` server here.
|
||||
"""
|
||||
"""Bring the platform up: bind the WS server on host:port."""
|
||||
if not self.token:
|
||||
logger.error("android: ANDROID_TOKEN must be set")
|
||||
self._set_fatal_error(
|
||||
@@ -360,21 +689,33 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
except ImportError:
|
||||
self._lock_key = None # status module not available (e.g. tests)
|
||||
|
||||
# M1: start the websockets server on host:port (TLS if cert/key set).
|
||||
try:
|
||||
await self._ws_server.start()
|
||||
except Exception:
|
||||
self._connected = False
|
||||
return False
|
||||
|
||||
self._connected = True
|
||||
self._mark_connected()
|
||||
logger.info("android: connected (skeleton; WS server starts in M1) on %s:%s", self.host, self.port)
|
||||
logger.info("android: connected; WS server on %s:%s", self.host, self.port)
|
||||
return True
|
||||
|
||||
async def disconnect(self) -> None:
|
||||
"""Tear down the platform."""
|
||||
"""Tear down the platform: stop the server, close device sockets."""
|
||||
try:
|
||||
from gateway.status import release_scoped_lock
|
||||
if getattr(self, "_lock_key", None):
|
||||
release_scoped_lock("android", self._lock_key)
|
||||
except ImportError:
|
||||
pass
|
||||
# M1: stop the server and close all device sockets.
|
||||
try:
|
||||
await self._ws_server.stop()
|
||||
except Exception:
|
||||
logger.warning("android: WS server stop failed", exc_info=True)
|
||||
try:
|
||||
self._devices.close()
|
||||
except Exception:
|
||||
pass
|
||||
self._connected = False
|
||||
self._mark_disconnected()
|
||||
logger.info("android: disconnected")
|
||||
@@ -390,18 +731,277 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
) -> SendResult:
|
||||
"""Send a message to a chat.
|
||||
|
||||
M0: no live devices yet -- log and report success with a minted id.
|
||||
M1: broadcast a ``message`` frame to connected devices, else fall to
|
||||
the outbox + fire push.
|
||||
M2: classify the outbound call into a structured frame using the
|
||||
per-chat turn state machine (see module docstring):
|
||||
|
||||
* ``metadata["expect_edits"]`` -> ``message.start`` (streaming segment)
|
||||
* ``metadata["notify"]`` -> ``message`` / ``message.stop`` (final)
|
||||
* tool-progress line format -> ``tool.start`` (first tool bubble)
|
||||
* anything else -> ``commentary``
|
||||
|
||||
With no live devices the frame is dropped here (the outbox + push
|
||||
replay lands in M3/M5).
|
||||
"""
|
||||
message_id = f"msg_{uuid.uuid4().hex}"
|
||||
logger.debug("android: send to %s (%d chars) [skeleton no-op]", chat_id, len(content or ""))
|
||||
content = content or ""
|
||||
meta = metadata or {}
|
||||
thread_id = _thread_id_from_metadata(meta)
|
||||
state = self._turn_state(chat_id)
|
||||
|
||||
# 1. Streaming segment start (stream consumer first send).
|
||||
if meta.get("expect_edits") is True:
|
||||
# A new content segment means the tool the model was waiting on
|
||||
# has returned -> close it before the segment opens.
|
||||
await self._close_open_tool(chat_id, state, thread_id)
|
||||
message_id = _mint_message_id()
|
||||
state.active = True
|
||||
state.stream_id = message_id
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.message_start(chat_id, message_id, protocol.ROLE_ASSISTANT, thread_id=thread_id),
|
||||
)
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
|
||||
# 2. Final message (non-streaming final, or streaming fallback final).
|
||||
if meta.get("notify") is True:
|
||||
reasoning, body = _split_reasoning(content)
|
||||
# Non-streaming: reasoning is prepended to content (split above).
|
||||
# Streaming fallback: content has no reasoning, so use the
|
||||
# reasoning captured via the on_stream_delta hook (wait for the
|
||||
# async hook worker to flush it first).
|
||||
if not reasoning:
|
||||
await _wait_for_reasoning_flushed()
|
||||
reasoning = _take_reasoning() or None
|
||||
else:
|
||||
_reset_reasoning()
|
||||
if state.stream_id:
|
||||
# Fallback final: close the open streaming segment in place.
|
||||
message_id = state.stream_id
|
||||
state.stream_id = None
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.message_stop(
|
||||
chat_id, message_id, body,
|
||||
reasoning=reasoning, thread_id=thread_id,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
else:
|
||||
message_id = _mint_message_id()
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.message(
|
||||
chat_id=chat_id,
|
||||
message_id=message_id,
|
||||
role=protocol.ROLE_ASSISTANT,
|
||||
text=body,
|
||||
thread_id=thread_id,
|
||||
reasoning=reasoning,
|
||||
reply_to=reply_to,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
await self._close_open_tool(chat_id, state, thread_id)
|
||||
self._reset_tool_state(state)
|
||||
state.active = False
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
|
||||
# 3. Tool progress (first tool bubble of an editable line buffer).
|
||||
if _is_tool_progress(content):
|
||||
return await self._emit_tool_lines(chat_id, content, state, thread_id, is_edit=False)
|
||||
|
||||
# 4. Commentary (interim assistant beat).
|
||||
message_id = _mint_message_id()
|
||||
state.active = True
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.commentary(chat_id, message_id, content, thread_id=thread_id),
|
||||
)
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
|
||||
async def send_typing(self, chat_id: str, metadata: Optional[Dict[str, Any]] = None) -> None:
|
||||
"""Send a typing indicator. M0: no-op (M1 emits a ``typing`` frame)."""
|
||||
async def edit_message(
|
||||
self,
|
||||
chat_id: str,
|
||||
message_id: str,
|
||||
content: str,
|
||||
*,
|
||||
finalize: bool = False,
|
||||
metadata: Optional[Dict[str, Any]] = None,
|
||||
) -> SendResult:
|
||||
"""Edit a previously sent message (M2: drives streaming + tool updates).
|
||||
|
||||
* ``message_id == state.stream_id`` -> ``message.update``
|
||||
(``finalize=True`` -> ``message.stop``).
|
||||
* ``message_id == state.tool_msg_id`` -> tool-progress update
|
||||
(new lines -> ``tool.start``).
|
||||
* unknown id -> best-effort ``message.update``.
|
||||
"""
|
||||
content = content or ""
|
||||
thread_id = _thread_id_from_metadata(metadata)
|
||||
state = self._turn_state(chat_id)
|
||||
|
||||
if message_id and message_id == state.stream_id:
|
||||
if finalize:
|
||||
reasoning, body = _split_reasoning(_strip_streaming_cursor(content))
|
||||
# Streaming: the gateway drops the model's separate
|
||||
# reasoning_content (final send suppressed), so attach the
|
||||
# reasoning we captured via the on_stream_delta hook (wait
|
||||
# for the async hook worker to flush it first).
|
||||
if not reasoning:
|
||||
await _wait_for_reasoning_flushed()
|
||||
reasoning = _take_reasoning() or None
|
||||
else:
|
||||
_reset_reasoning()
|
||||
state.stream_id = None
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.message_stop(
|
||||
chat_id, message_id, body,
|
||||
reasoning=reasoning, thread_id=thread_id,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
await self._close_open_tool(chat_id, state, thread_id)
|
||||
self._reset_tool_state(state)
|
||||
state.active = False
|
||||
else:
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.message_update(
|
||||
chat_id, message_id, _strip_streaming_cursor(content),
|
||||
thread_id=thread_id,
|
||||
),
|
||||
)
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
|
||||
if message_id and message_id == state.tool_msg_id:
|
||||
return await self._emit_tool_lines(chat_id, content, state, thread_id, is_edit=True)
|
||||
|
||||
# Unknown id: treat as a streaming update (best effort).
|
||||
if finalize:
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.message_stop(
|
||||
chat_id, message_id, _strip_streaming_cursor(content),
|
||||
thread_id=thread_id, ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
else:
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.message_update(
|
||||
chat_id, message_id, _strip_streaming_cursor(content),
|
||||
thread_id=thread_id,
|
||||
),
|
||||
)
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
|
||||
# ── M2: tool-progress helpers ─────────────────────────────────────────
|
||||
|
||||
async def _emit_tool_lines(
|
||||
self,
|
||||
chat_id: str,
|
||||
content: str,
|
||||
state: _TurnState,
|
||||
thread_id: Optional[str],
|
||||
*,
|
||||
is_edit: bool,
|
||||
) -> SendResult:
|
||||
"""Emit ``tool.start`` for each NEW tool line in *content*.
|
||||
|
||||
The gateway accumulates tool lines in one editable bubble; on an edit
|
||||
the full buffer is re-sent, so we diff against ``seen_tool_lines`` to
|
||||
emit only the new ones. A new tool closes the previously-open tool.
|
||||
"""
|
||||
message_id = state.tool_msg_id or _mint_message_id()
|
||||
state.tool_msg_id = message_id
|
||||
state.active = True
|
||||
|
||||
lines = [ln for ln in content.splitlines() if ln.strip()]
|
||||
for line in lines:
|
||||
key = line.strip()
|
||||
if key in state.seen_tool_lines:
|
||||
continue
|
||||
state.seen_tool_lines.add(key)
|
||||
parsed = self._parse_tool_line_or_block(line, content)
|
||||
if parsed is None:
|
||||
continue
|
||||
name, preview = parsed
|
||||
# A new tool begins: close the previously-open one.
|
||||
if state.open_tool_index is not None:
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.tool_end(chat_id, state.open_tool_index, "", ok=True, thread_id=thread_id),
|
||||
)
|
||||
state.tool_index += 1
|
||||
state.open_tool_index = state.tool_index
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.tool_start(
|
||||
chat_id, state.tool_index, name,
|
||||
preview=preview, thread_id=thread_id,
|
||||
),
|
||||
)
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
|
||||
@staticmethod
|
||||
def _parse_tool_line_or_block(line: str, content: str) -> Optional[Tuple[str, Optional[str]]]:
|
||||
"""Parse a tool line, expanding a terminal code block to its command."""
|
||||
parsed = _parse_tool_line(line)
|
||||
if parsed is not None:
|
||||
name, preview = parsed
|
||||
# Terminal code block: the command lives in the fenced lines that
|
||||
# follow the "<emoji> terminal" head line.
|
||||
if name == "terminal" and preview is None and "```" in content:
|
||||
cmd = _extract_code_block(content)
|
||||
if cmd:
|
||||
return name, cmd
|
||||
return parsed
|
||||
return None
|
||||
|
||||
async def _close_open_tool(
|
||||
self, chat_id: str, state: _TurnState, thread_id: Optional[str]
|
||||
) -> None:
|
||||
"""Emit ``tool.end`` for the currently-open tool, if any.
|
||||
|
||||
A tool is considered complete when the next tool starts OR a new
|
||||
content segment begins (the model only produces content after the
|
||||
tool it was waiting on has returned).
|
||||
"""
|
||||
if state.open_tool_index is not None:
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.tool_end(chat_id, state.open_tool_index, "", ok=True, thread_id=thread_id),
|
||||
)
|
||||
state.open_tool_index = None
|
||||
|
||||
def _reset_tool_state(self, state: _TurnState) -> None:
|
||||
"""Clear per-turn tool bookkeeping (called at turn finalization)."""
|
||||
state.tool_msg_id = None
|
||||
state.seen_tool_lines = set()
|
||||
state.tool_index = 0
|
||||
state.open_tool_index = None
|
||||
|
||||
async def _broadcast_or_log(self, chat_id: str, frame: "protocol.Frame") -> None:
|
||||
delivered = await self._ws_server.broadcast(frame)
|
||||
if delivered == 0:
|
||||
logger.info(
|
||||
"android: no live devices for %s; %s frame not delivered (outbox lands in M3)",
|
||||
chat_id, frame.type,
|
||||
)
|
||||
|
||||
async def send_typing(self, chat_id: str, metadata: Optional[Dict[str, Any]] = None) -> None:
|
||||
"""Send a typing indicator (``typing`` frame, on=true)."""
|
||||
thread_id = None
|
||||
if metadata:
|
||||
tid = metadata.get("thread_id")
|
||||
if isinstance(tid, str) and tid:
|
||||
thread_id = tid
|
||||
await self._ws_server.broadcast(protocol.typing(chat_id, True, thread_id=thread_id))
|
||||
|
||||
async def stop_typing(self, chat_id: str) -> None:
|
||||
"""Clear the typing indicator (``typing`` frame, on=false)."""
|
||||
await self._ws_server.broadcast(protocol.typing(chat_id, False))
|
||||
|
||||
async def send_image(
|
||||
self,
|
||||
chat_id: str,
|
||||
@@ -410,19 +1010,120 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
reply_to: Optional[str] = None,
|
||||
metadata: Optional[Dict[str, Any]] = None,
|
||||
) -> SendResult:
|
||||
"""Send an image. M0: not implemented (M4)."""
|
||||
"""Send an image. M1: not implemented (M4)."""
|
||||
return SendResult(success=False, error="android: media not implemented yet (M4)")
|
||||
|
||||
# ── Inbound (app -> agent) ────────────────────────────────────────────
|
||||
|
||||
async def on_message_send(self, frame: protocol.Frame, device_id: str) -> None:
|
||||
"""Handle an inbound ``message.send`` frame.
|
||||
|
||||
Echoes the user message to all devices (multi-device sync + ack),
|
||||
then builds a ``MessageEvent`` and hands it to ``handle_message()``
|
||||
(the gateway's command pipeline + agent turn).
|
||||
"""
|
||||
payload = frame.payload
|
||||
text = payload.get("text")
|
||||
if not isinstance(text, str) or not text.strip():
|
||||
await self._ws_server.send_to(
|
||||
device_id,
|
||||
protocol.error(protocol.ERR_UNSUPPORTED, "message.send requires non-empty text", id=frame.id),
|
||||
)
|
||||
return
|
||||
|
||||
chat_id = frame.chat_id or payload.get("chat_id")
|
||||
if not isinstance(chat_id, str) or not chat_id.strip():
|
||||
chat_id = self.home_channel
|
||||
chat_id = chat_id.strip()
|
||||
|
||||
thread_id = frame.thread_id or payload.get("thread_id")
|
||||
if not isinstance(thread_id, str) or not thread_id.strip():
|
||||
thread_id = None
|
||||
|
||||
reply_to = payload.get("reply_to")
|
||||
if not isinstance(reply_to, str) or not reply_to.strip():
|
||||
reply_to = None
|
||||
|
||||
device = self._devices.get(device_id) or {}
|
||||
user_name = device.get("name") or device_id
|
||||
|
||||
# Echo to all devices: the sender confirms (server-assigned id),
|
||||
# other devices see the message too (single-user, multi-device).
|
||||
message_id = f"m_{uuid.uuid4().hex[:16]}"
|
||||
echo = protocol.message(
|
||||
chat_id=chat_id,
|
||||
message_id=message_id,
|
||||
role=protocol.ROLE_USER,
|
||||
text=text,
|
||||
thread_id=thread_id,
|
||||
reply_to=reply_to,
|
||||
ts=int(time.time() * 1000),
|
||||
)
|
||||
await self._ws_server.broadcast(echo)
|
||||
|
||||
source = self.build_source(
|
||||
chat_id=chat_id,
|
||||
chat_name=self._channel_name(chat_id),
|
||||
chat_type="dm",
|
||||
user_id=device_id,
|
||||
user_name=user_name,
|
||||
thread_id=thread_id,
|
||||
)
|
||||
event = MessageEvent(
|
||||
text=text,
|
||||
message_type=MessageType.TEXT,
|
||||
user_id=device_id,
|
||||
user_name=user_name,
|
||||
source=source,
|
||||
message_id=message_id,
|
||||
reply_to_message_id=reply_to,
|
||||
)
|
||||
await self.handle_message(event)
|
||||
|
||||
# ── Chat info ─────────────────────────────────────────────────────────
|
||||
|
||||
def _channel_name(self, chat_id: str) -> str:
|
||||
"""Channel display name. M1: home channel only (directory is M3)."""
|
||||
if chat_id in (self.home_channel, DEFAULT_HOME_CHANNEL):
|
||||
return self.home_channel_name
|
||||
return chat_id or "chat"
|
||||
|
||||
async def get_chat_info(self, chat_id: str) -> Dict[str, Any]:
|
||||
"""Return ``{name, type, chat_id}`` for a chat.
|
||||
|
||||
M0: the channel directory is not persisted yet, so report the home
|
||||
channel name for the default chat and a generic name otherwise.
|
||||
M1: the channel directory is not persisted yet, so report the home
|
||||
channel name for the default chat and the raw id otherwise.
|
||||
"""
|
||||
name = "Default" if chat_id in (self.home_channel, DEFAULT_HOME_CHANNEL) else (chat_id or "chat")
|
||||
return {"name": name, "type": "channel", "chat_id": chat_id}
|
||||
return {
|
||||
"name": self._channel_name(chat_id),
|
||||
"type": "channel",
|
||||
"chat_id": chat_id,
|
||||
}
|
||||
|
||||
# ── hello.ack helpers ─────────────────────────────────────────────────
|
||||
|
||||
def server_caps(self) -> Dict[str, Any]:
|
||||
"""Capability flags advertised in ``hello.ack`` (M2 surface)."""
|
||||
return {
|
||||
"streaming": True, # M2: message.start/update/stop
|
||||
"reasoning": True, # M2: reasoning field on message / message.stop
|
||||
"tools": True, # M2: tool.start/progress/end
|
||||
"media": False, # M4
|
||||
"search": False, # M3
|
||||
"push": self.push_backend,
|
||||
"pickers": False, # M2+
|
||||
}
|
||||
|
||||
def channel_list(self) -> List[Dict[str, Any]]:
|
||||
"""Channel directory for ``hello.ack``. M1: home channel only (M3)."""
|
||||
return [
|
||||
{
|
||||
"chat_id": self.home_channel,
|
||||
"name": self.home_channel_name,
|
||||
"kind": "default",
|
||||
"is_default": True,
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -431,6 +1132,13 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
|
||||
def register(ctx):
|
||||
"""Plugin entry point: called by the Hermes plugin system."""
|
||||
# M2: capture the model's separate reasoning_content during streaming so
|
||||
# it can be attached to the turn's message.stop frame (the gateway
|
||||
# otherwise drops it when streaming suppresses the final send).
|
||||
try:
|
||||
ctx.register_hook("on_stream_delta", _on_stream_delta)
|
||||
except Exception:
|
||||
logger.debug("android: on_stream_delta hook registration failed", exc_info=True)
|
||||
ctx.register_platform(
|
||||
name="android",
|
||||
label="Android",
|
||||
|
||||
Reference in new issue
Block a user