""" Android Platform Adapter for Hermes Agent (Iris x Hermes). A plugin-based gateway adapter that runs a WebSocket server *inside* the ``hermes gateway`` process. The native Android / Desktop app connects to it with a pairing token and talks to the agent over a single WS transport (chat, streaming, tools, media, pairing, push-token). Zero new Python dependencies: ``websockets`` and ``httpx`` are hermes core deps. Zero hermes-core changes. Milestone M1: the gateway core loop (text round-trip). The WS server binds and authenticates devices (``hello`` with constant-time token check), the adapter emits ``message`` frames from ``send()`` and turns inbound ``message.send`` frames into ``MessageEvent``s for ``handle_message()``. Milestone M2: agent transparency. ``send()``/``edit_message()`` are mapped to ``message.start``/``message.update``/``message.stop`` (streaming), tool progress is classified into structured ``tool.start``/``tool.end`` frames, interim commentary becomes ``commentary`` frames, and the code-style reasoning prefix is split into a ``reasoning`` field. Media, outbox, push, and search land in later milestones (see ``docs/14-milestones.md``). Configuration in config.yaml:: gateway: platforms: android: enabled: true extra: host: 127.0.0.1 port: 8790 home_channel: android:default push_backend: fcm outbox_retention_hours: 72 max_upload_bytes: 104857600 Or via environment variables (overrides config.yaml; secrets live in .env): ANDROID_TOKEN, ANDROID_WS_HOST, ANDROID_WS_PORT, ANDROID_HOME_CHANNEL, ANDROID_PUSH_BACKEND, ANDROID_FCM_SERVICE_ACCOUNT, NTFY_TOPIC, ... """ import asyncio import logging import os import re import threading import time import uuid from dataclasses import dataclass, field from typing import Any, Dict, List, Optional, Tuple from agent.secret_scope import UnscopedSecretError as _UnscopedSecretError from agent.secret_scope import get_secret as _scoped_get_secret def _get_scoped_secret(name, default=None): """Scope-aware credential read with the default-profile startup fallback. Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the IRC ``IRC_SERVER_PASSWORD`` read (``plugins/platforms/irc/adapter.py``). """ try: val = _scoped_get_secret(name, default) except _UnscopedSecretError: val = os.getenv(name) return val if val is not None else default logger = logging.getLogger(__name__) # --------------------------------------------------------------------------- # Lazy import: BasePlatformAdapter and friends live in the main repo. # We import at module level (as the bundled plugins do) but guard the heavy # gateway imports so the plugin can be discovered before the gateway is fully # initialised. # --------------------------------------------------------------------------- from gateway.platforms.base import ( # noqa: E402 BasePlatformAdapter, SendResult, MessageEvent, MessageType, ) from gateway.config import Platform # noqa: E402 from hermes_constants import get_hermes_home # noqa: E402 from . import protocol # noqa: E402 from .pairing import ( # noqa: E402 DeviceRegistry, generate_token, pairing_url, qr_payload, ) from .ws_server import WsServer # noqa: E402 # --------------------------------------------------------------------------- # M2 — reasoning capture (streaming) # # The gateway streams only ``content`` to the platform and suppresses the # final send (which would carry the prepended reasoning), so the model's # separate ``reasoning_content`` is otherwise lost in the streaming case. # hermes exposes a plugin ``on_stream_delta`` hook that fires reasoning # deltas with ``kind="reasoning"`` (gated by ``plugins.stream_reasoning_deltas``). # We accumulate those deltas here and attach the result to the turn's # ``message.stop`` frame. Single-chat for now (android:default), so a # module-level buffer suffices; it is reset at each turn start. # --------------------------------------------------------------------------- _reasoning_parts: List[str] = [] _reasoning_lock = threading.Lock() # Barrier: set by the hook worker once it has processed the first content # delta (kind="text"). The worker drains a FIFO queue and reasoning deltas are # enqueued before content deltas, so at that point every reasoning delta has # already been appended -- a reliable "reasoning flushed" signal that avoids # racing message.stop against the async hook thread. _reasoning_flushed = threading.Event() def _on_stream_delta(**kwargs: Any) -> None: """Plugin hook: capture reasoning deltas (kind="reasoning").""" kind = kwargs.get("kind") if kind == "reasoning": delta = kwargs.get("delta") or "" if delta: with _reasoning_lock: _reasoning_parts.append(delta) elif kind == "text": _reasoning_flushed.set() async def _wait_for_reasoning_flushed(timeout: float = 0.3) -> None: """Wait (without blocking the event loop) until the hook worker has processed all reasoning deltas, or *timeout* seconds elapse.""" loop = asyncio.get_running_loop() deadline = loop.time() + timeout while loop.time() < deadline: if _reasoning_flushed.is_set(): return await asyncio.sleep(0.01) def _take_reasoning() -> str: """Drain and return the accumulated reasoning (empty string if none).""" with _reasoning_lock: parts = _reasoning_parts[:] _reasoning_parts.clear() _reasoning_flushed.clear() return "".join(parts).strip() def _reset_reasoning() -> None: with _reasoning_lock: _reasoning_parts.clear() _reasoning_flushed.clear() # --------------------------------------------------------------------------- # Defaults # --------------------------------------------------------------------------- DEFAULT_HOST = "127.0.0.1" DEFAULT_PORT = 8790 DEFAULT_HOME_CHANNEL = "android:default" DEFAULT_HOME_CHANNEL_NAME = "Default" DEFAULT_PUSH_BACKEND = "fcm" DEFAULT_OUTBOX_RETENTION_HOURS = 72 DEFAULT_MAX_UPLOAD_BYTES = 100 * 1024 * 1024 # 100 MB def _truthy(value: Optional[str]) -> bool: return (value or "").strip().lower() in {"1", "true", "yes", "on"} # --------------------------------------------------------------------------- # M2 — turn state + outbound classification # # The main gateway delivers through the legacy callback path: the stream # consumer calls ``send()`` (first bubble of a segment) and ``edit_message()`` # (updates), tool progress flows through ``send()``/``edit_message()`` of an # accumulated line buffer, and interim commentary arrives as a plain # ``send()``. We classify each outbound call into a structured frame using a # per-chat turn state machine + content markers: # # * ``metadata["expect_edits"] is True`` -> streaming segment start # * ``metadata["notify"] is True`` -> final message (or fallback final) # * tool-progress line format -> tool.start / tool.end # * anything else -> commentary # # Verified empirically against the live gateway with ``tests/ws_probe.py``. # --------------------------------------------------------------------------- # Streaming cursor the gateway appends to in-progress edits (" ▉"). Stripped # before we forward text to the app (the app renders its own live indicator). _STREAMING_CURSOR = " ▉" # Code-style reasoning prefix (gateway/run.py, reasoning_style="code"): # "💭 **Reasoning:**\n```\n\n```\n\n" _REASONING_PREFIX = "💭 **Reasoning:**\n```\n" _REASONING_CLOSE = "\n```\n\n" # A gateway tool-progress line begins with a (non-ASCII) tool emoji. _TOOL_LINE_RE = re.compile(r"^(\S+)\s+(.+)$") _TOOL_NAME_PREVIEW_RE = re.compile(r'^(\S+):\s*"(.*)"\s*$') _TOOL_NAME_BARE_RE = re.compile(r"^(\S+)\.\.\.\s*$") _TOOL_NAME_ARGS_RE = re.compile(r"^(\S+)\(([^)]*)\)\s*$") # Terminal code block: "💻 terminal\n```\n\n```" _TOOL_CODEBLOCK_HEAD_RE = re.compile(r"^(\S+)\s+(\S+)\s*$") # Reverse map of the gateway's friendly tool verbs (agent/display.py # _TOOL_VERBS) so a verb-form line ("🔍 Searching the web for …") can be # recovered to a structured (tool_name, preview). Longest-first matching is # done at parse time. Verbs shared by several tools map to the most common. _VERB_TO_TOOL: Dict[str, str] = { "Searching the web": "web_search", "Searching files": "search_files", "Searching past sessions": "session_search", "Running code": "execute_code", "Running": "terminal", "Reading skill": "skill_view", "Reading": "read_file", "Writing": "write_file", "Editing": "patch", "Browsing": "browser_navigate", "Clicking": "browser_click", "Typing": "browser_type", "Generating image": "image_generate", "Generating video": "video_generate", "Generating speech": "text_to_speech", "Looking at the image": "vision_analyze", "Listing skills": "skills_list", "Updating skill": "skill_manage", "Updating memory": "memory", "Updating tasks": "todo", "Delegating": "delegate_task", "Scheduling": "cronjob", "Asking": "clarify", } # Verbs that take a " for " connector before the preview. _VERB_FOR_CONNECTOR = {"web_search", "search_files"} def _mint_message_id() -> str: return f"m_{uuid.uuid4().hex[:16]}" def _thread_id_from_metadata(metadata: Optional[Dict[str, Any]]) -> Optional[str]: if not metadata: return None tid = metadata.get("thread_id") if isinstance(tid, str) and tid: return tid return None def _strip_streaming_cursor(text: str) -> str: if text and text.endswith(_STREAMING_CURSOR): return text[: -len(_STREAMING_CURSOR)] return text def _split_reasoning(text: str) -> Tuple[Optional[str], str]: """Split a code-style reasoning prefix off the front of *text*. Returns ``(reasoning, body)``; ``reasoning`` is ``None`` when no prefix is present (reasoning off / no reasoning / non-code style). Best-effort parse of a stable, gateway-owned format: on any mismatch the fallback is ``(None, full text)`` so the answer still renders. """ if not text or not text.startswith(_REASONING_PREFIX): return None, text close_idx = text.find(_REASONING_CLOSE, len(_REASONING_PREFIX)) if close_idx == -1: return None, text reasoning = text[len(_REASONING_PREFIX):close_idx] body = text[close_idx + len(_REASONING_CLOSE):] return reasoning, body def _parse_tool_line(line: str) -> Optional[Tuple[str, Optional[str]]]: """Parse a single gateway tool-progress line into ``(name, preview)``. Returns ``None`` when the line is not a tool line. The gateway formats tool lines as `` : ""``, `` ...``, `` (keys)``, or a friendly verb phrase (`` …``). The verb form is lossy (no tool name), so we surface the verb as the name. """ line = line.strip() if not line: return None m = _TOOL_LINE_RE.match(line) if not m: return None emoji, rest = m.group(1), m.group(2) if emoji.isascii(): return None # a tool line always leads with a non-ASCII emoji mp = _TOOL_NAME_PREVIEW_RE.match(rest) if mp: return mp.group(1), mp.group(2) mb = _TOOL_NAME_BARE_RE.match(rest) if mb: return mb.group(1), None ma = _TOOL_NAME_ARGS_RE.match(rest) if ma: return ma.group(1), None # Friendly verb phrase: reverse-map to (tool_name, preview). verb_parsed = _parse_verb_phrase(rest) if verb_parsed is not None: return verb_parsed # Unrecognised: use the phrase as the label. return rest, None def _parse_verb_phrase(phrase: str) -> Optional[Tuple[str, Optional[str]]]: """Reverse-map a friendly verb phrase to ``(tool_name, preview)``. Matches the longest verb first so "Running code" wins over "Running". Returns ``None`` when no known verb leads the phrase. """ for verb in sorted(_VERB_TO_TOOL, key=len, reverse=True): tool = _VERB_TO_TOOL[verb] if phrase == verb: return tool, None if tool in _VERB_FOR_CONNECTOR and phrase.startswith(verb + " for "): return tool, phrase[len(verb) + len(" for "):].strip() or None if phrase.startswith(verb + " "): return tool, phrase[len(verb) + 1:].strip() or None return None def _extract_code_block(content: str) -> Optional[str]: """Return the first fenced code block's body in *content*, else ``None``. Used to recover the terminal command from a tool-progress code block (`` terminal`` head line + fenced command). """ m = re.search(r"```[^\n]*\n(.*?)\n```", content, re.DOTALL) if m: return m.group(1).strip() or None return None def _is_tool_progress(content: str) -> bool: """Heuristic: does *content* look like gateway tool-progress line(s)? Tool progress is delivered as one or more lines, each led by a tool emoji (or a terminal code block). Commentary is free-form prose. We classify on the first non-empty line; subsequent lines of the same bubble are tracked by message id, not re-classified. """ if not content: return False lines = [ln for ln in content.splitlines() if ln.strip()] if not lines: return False first = lines[0].strip() # Terminal code block: " terminal" then a fenced command. if len(lines) > 1 and lines[1].strip().startswith("```"): return _TOOL_CODEBLOCK_HEAD_RE.match(first) is not None return _parse_tool_line(first) is not None @dataclass class _TurnState: """Per-chat turn state for outbound frame classification (M2).""" active: bool = False # message_id of the currently streaming segment (message.start open). stream_id: Optional[str] = None # message_id of the current tool-progress bubble (editable line buffer). tool_msg_id: Optional[str] = None # Monotonic per-turn tool counter (start -> end correlation). tool_index: int = 0 # Index of the most recently started tool (awaiting tool.end). open_tool_index: Optional[int] = None # Tool lines already emitted as tool.start (dedup across edits). seen_tool_lines: set = field(default_factory=set) # --------------------------------------------------------------------------- # Passive / config probes (called from status displays -- no side effects) # --------------------------------------------------------------------------- def check_requirements() -> bool: """PASSIVE dependency probe: ``websockets`` importable + token set. Must be side-effect free (called from ``hermes setup`` / ``status`` / dashboard readiness). Never installs. """ try: import websockets # noqa: F401 (core dep) except Exception: return False return bool(_get_scoped_secret("ANDROID_TOKEN")) def validate_config(config) -> bool: """Given a PlatformConfig, is the platform properly configured?""" extra = getattr(config, "extra", {}) or {} token = _get_scoped_secret("ANDROID_TOKEN") or extra.get("token", "") return bool(token) def is_connected(config) -> bool: """Is the platform configured (env or config.yaml)?""" return validate_config(config) # --------------------------------------------------------------------------- # Env-driven auto-configuration (seeds PlatformConfig.extra pre-adapter) # --------------------------------------------------------------------------- def _env_enablement() -> Optional[dict]: """Seed ``PlatformConfig.extra`` from env vars during gateway config load. Called by the platform registry's env-enablement hook BEFORE adapter construction, so ``gateway status`` and ``get_connected_platforms()`` reflect env-only configuration without instantiating the adapter. Returns ``None`` when the platform isn't minimally configured (no token); the caller then skips auto-enabling. The special ``home_channel`` key in the returned dict is handled by the core hook -- it becomes a proper ``HomeChannel`` dataclass on the ``PlatformConfig`` rather than being merged into ``extra``. """ token = _get_scoped_secret("ANDROID_TOKEN", "") if not token: return None # Seed ONLY explicitly-set env vars: the core commits this seed on top of # config.yaml (``extra.update(seed)``), so default values here would # clobber user YAML. Unset keys fall through to config.yaml / adapter # defaults. seed: Dict[str, Any] = {} host = os.getenv("ANDROID_WS_HOST", "").strip() if host: seed["host"] = host port_raw = os.getenv("ANDROID_WS_PORT", "").strip() if port_raw: seed["port"] = _parse_port(port_raw) push = os.getenv("ANDROID_PUSH_BACKEND", "").strip().lower() if push: seed["push_backend"] = push home = os.getenv("ANDROID_HOME_CHANNEL", "").strip() if home: seed["home_channel"] = { "chat_id": home, "name": os.getenv("ANDROID_HOME_CHANNEL_NAME", "").strip() or DEFAULT_HOME_CHANNEL_NAME, } return seed def _parse_port(raw: str) -> int: try: return int((raw or "").strip()) except (ValueError, TypeError): return DEFAULT_PORT # --------------------------------------------------------------------------- # Target parsing: "android:[:]" # --------------------------------------------------------------------------- def _parse_target_ref(target_ref: str) -> Optional[tuple]: """Parse a raw target string into ``(chat_id, thread_id)`` or ``None``. Recognises the native syntax ``android:[:]``. Returns ``None`` for anything else so the target proceeds to channel-directory resolution. """ if not target_ref or not target_ref.startswith("android:"): return None body = target_ref[len("android:"):] if not body: return None if ":" in body: chat_id, thread_id = body.split(":", 1) thread_id = thread_id or None else: chat_id, thread_id = body, None chat_id = chat_id.strip() if not chat_id: return None return (chat_id, thread_id) # --------------------------------------------------------------------------- # Standalone (out-of-process) send -- best-effort, stretch for v1 # --------------------------------------------------------------------------- async def _standalone_send( pconfig, chat_id: str, message: str, *, thread_id: Optional[str] = None, media_files: Optional[List[str]] = None, force_document: bool = False, ) -> Dict[str, Any]: """Out-of-process delivery for cron jobs that run separately from the gateway. The outbox is served by the *running* gateway, so standalone delivery while the gateway process is fully down is best-effort only (see ``docs/00-overview.md`` "Out of scope"). For M1 this is a stub that reports the gateway is required; the real implementation lands with the outbox (M3/M5). """ return { "error": ( "android standalone send: the running gateway is required to serve " "the outbox (standalone delivery is best-effort only)" ) } # --------------------------------------------------------------------------- # Interactive setup (hermes gateway setup flow) # --------------------------------------------------------------------------- def interactive_setup() -> None: """Prompt for the pairing token / host / port / push backend. M1: token generation, host/port/push prompts, and the pairing QR payload (``iris://pair?...``) + app URL printed for the Connect screen. """ try: from hermes_cli.config import ( get_env_value, save_env_value, prompt, print_info, print_success, print_warning, ) except Exception: print("android: setup helpers unavailable; set ANDROID_TOKEN in ~/.hermes/.env") return print_info("📱 Android / Desktop (Iris x Hermes)") token = get_env_value("ANDROID_TOKEN") or "" if not token: generated = generate_token() save_env_value("ANDROID_TOKEN", generated) print_success(f"Generated pairing token: {generated}") print_warning("Keep this secret -- the app presents it on connect.") else: print_info("Existing ANDROID_TOKEN found (not shown).") host = prompt("WS bind host", default=get_env_value("ANDROID_WS_HOST") or DEFAULT_HOST) save_env_value("ANDROID_WS_HOST", host or DEFAULT_HOST) port = prompt("WS port", default=str(_parse_port(get_env_value("ANDROID_WS_PORT") or ""))) save_env_value("ANDROID_WS_PORT", str(_parse_port(port))) backend = prompt("Push backend (fcm/ntfy)", default=get_env_value("ANDROID_PUSH_BACKEND") or DEFAULT_PUSH_BACKEND) save_env_value("ANDROID_PUSH_BACKEND", (backend or DEFAULT_PUSH_BACKEND).strip().lower()) # Pairing payload for the app's Connect screen (QR / manual entry). try: from hermes_cli.config import print_code url = pairing_url(host or DEFAULT_HOST, _parse_port(port)) payload = qr_payload(host or DEFAULT_HOST, _parse_port(port), token) print_info("Pair your device (scan with the app or enter on the Connect screen):") print_code(payload) print_info(f"Server URL: {url}") except Exception: url = pairing_url(host or DEFAULT_HOST, _parse_port(port)) print_info(f"Pairing URL: {qr_payload(host or DEFAULT_HOST, _parse_port(port), token)}") print_info(f"Server URL: {url}") print_success("Android configuration saved to ~/.hermes/.env") print_info("Restart the gateway for changes to take effect: hermes gateway restart") # --------------------------------------------------------------------------- # Android Adapter # --------------------------------------------------------------------------- class AndroidAdapter(BasePlatformAdapter): """WebSocket-backed adapter for the native Iris Android / Desktop app. M1: the WS server (``ws_server.WsServer``) authenticates devices with the pairing token, the connection registry tracks live sockets, ``send()`` emits ``message`` frames, and inbound ``message.send`` frames become ``MessageEvent``s for ``handle_message()``. """ def __init__(self, config, **kwargs): platform = Platform("android") super().__init__(config=config, platform=platform) extra = getattr(config, "extra", {}) or {} # Connection settings (env vars override config.yaml) self.host = os.getenv("ANDROID_WS_HOST", "").strip() or extra.get("host", DEFAULT_HOST) self.port = _parse_port(os.getenv("ANDROID_WS_PORT", "") or str(extra.get("port", DEFAULT_PORT))) self.token = _get_scoped_secret("ANDROID_TOKEN") or extra.get("token", "") self.push_backend = ( os.getenv("ANDROID_PUSH_BACKEND", "").strip().lower() or extra.get("push_backend", DEFAULT_PUSH_BACKEND) ) self.outbox_retention_hours = int( extra.get("outbox_retention_hours", DEFAULT_OUTBOX_RETENTION_HOURS) ) self.max_upload_bytes = int( extra.get("max_upload_bytes", DEFAULT_MAX_UPLOAD_BYTES) ) # Home channel: the core hook turns the env-seeded ``home_channel`` # dict into a HomeChannel dataclass on the config; config.yaml may # also put it in extra (dict or bare string). home = getattr(config, "home_channel", None) if home is not None and getattr(home, "chat_id", None): self.home_channel = str(home.chat_id) self.home_channel_name = str(getattr(home, "name", "") or DEFAULT_HOME_CHANNEL_NAME) else: hc = extra.get("home_channel") if isinstance(hc, dict) and hc.get("chat_id"): self.home_channel = str(hc["chat_id"]) self.home_channel_name = str(hc.get("name") or DEFAULT_HOME_CHANNEL_NAME) elif isinstance(hc, str) and hc.strip(): self.home_channel = hc.strip() self.home_channel_name = DEFAULT_HOME_CHANNEL_NAME else: self.home_channel = DEFAULT_HOME_CHANNEL self.home_channel_name = DEFAULT_HOME_CHANNEL_NAME # TLS (optional) self.ws_cert = _get_scoped_secret("ANDROID_WS_CERT") or extra.get("ws_cert", "") self.ws_key = _get_scoped_secret("ANDROID_WS_KEY") or extra.get("ws_key", "") # Auth allowed = os.getenv("ANDROID_ALLOWED_USERS", "").strip() self.allowed_users: List[str] = ( [u.strip() for u in allowed.split(",") if u.strip()] if allowed else [] ) self.allow_all = _truthy(os.getenv("ANDROID_ALLOW_ALL_USERS")) # Runtime state self._devices = DeviceRegistry(get_hermes_home() / "android" / "devices.db") self._ws_server = WsServer(self, self._devices) self._connected = False # M2: per-chat turn state for outbound frame classification. self._turns: Dict[str, _TurnState] = {} def _turn_state(self, chat_id: str) -> _TurnState: st = self._turns.get(chat_id) if st is None: st = _TurnState() self._turns[chat_id] = st return st @property def name(self) -> str: return "Android" # ── Connection lifecycle ────────────────────────────────────────────── async def connect(self, *, is_reconnect: bool = False) -> bool: """Bring the platform up: bind the WS server on host:port.""" if not self.token: logger.error("android: ANDROID_TOKEN must be set") self._set_fatal_error( "config_missing", "ANDROID_TOKEN must be set", retryable=False, ) return False # Prevent two profiles from binding the same port/identity. try: from gateway.status import acquire_scoped_lock lock_key = f"{self.host}:{self.port}" if not acquire_scoped_lock("android", lock_key): logger.error("android: %s:%s already in use by another profile", self.host, self.port) self._set_fatal_error( "lock_conflict", "WS port in use by another profile", retryable=False, ) return False self._lock_key = lock_key except ImportError: self._lock_key = None # status module not available (e.g. tests) try: await self._ws_server.start() except Exception: self._connected = False return False self._connected = True self._mark_connected() logger.info("android: connected; WS server on %s:%s", self.host, self.port) return True async def disconnect(self) -> None: """Tear down the platform: stop the server, close device sockets.""" try: from gateway.status import release_scoped_lock if getattr(self, "_lock_key", None): release_scoped_lock("android", self._lock_key) except ImportError: pass try: await self._ws_server.stop() except Exception: logger.warning("android: WS server stop failed", exc_info=True) try: self._devices.close() except Exception: pass self._connected = False self._mark_disconnected() logger.info("android: disconnected") # ── Outbound (agent -> app) ─────────────────────────────────────────── async def send( self, chat_id: str, content: str, reply_to: Optional[str] = None, metadata: Optional[Dict[str, Any]] = None, ) -> SendResult: """Send a message to a chat. M2: classify the outbound call into a structured frame using the per-chat turn state machine (see module docstring): * ``metadata["expect_edits"]`` -> ``message.start`` (streaming segment) * ``metadata["notify"]`` -> ``message`` / ``message.stop`` (final) * tool-progress line format -> ``tool.start`` (first tool bubble) * anything else -> ``commentary`` With no live devices the frame is dropped here (the outbox + push replay lands in M3/M5). """ content = content or "" meta = metadata or {} thread_id = _thread_id_from_metadata(meta) state = self._turn_state(chat_id) # 1. Streaming segment start (stream consumer first send). if meta.get("expect_edits") is True: # A new content segment means the tool the model was waiting on # has returned -> close it before the segment opens. await self._close_open_tool(chat_id, state, thread_id) message_id = _mint_message_id() state.active = True state.stream_id = message_id await self._broadcast_or_log( chat_id, protocol.message_start(chat_id, message_id, protocol.ROLE_ASSISTANT, thread_id=thread_id), ) return SendResult(success=True, message_id=message_id) # 2. Final message (non-streaming final, or streaming fallback final). if meta.get("notify") is True: reasoning, body = _split_reasoning(content) # Non-streaming: reasoning is prepended to content (split above). # Streaming fallback: content has no reasoning, so use the # reasoning captured via the on_stream_delta hook (wait for the # async hook worker to flush it first). if not reasoning: await _wait_for_reasoning_flushed() reasoning = _take_reasoning() or None else: _reset_reasoning() if state.stream_id: # Fallback final: close the open streaming segment in place. message_id = state.stream_id state.stream_id = None await self._broadcast_or_log( chat_id, protocol.message_stop( chat_id, message_id, body, reasoning=reasoning, thread_id=thread_id, ts=int(time.time() * 1000), ), ) else: message_id = _mint_message_id() await self._broadcast_or_log( chat_id, protocol.message( chat_id=chat_id, message_id=message_id, role=protocol.ROLE_ASSISTANT, text=body, thread_id=thread_id, reasoning=reasoning, reply_to=reply_to, ts=int(time.time() * 1000), ), ) await self._close_open_tool(chat_id, state, thread_id) self._reset_tool_state(state) state.active = False return SendResult(success=True, message_id=message_id) # 3. Tool progress (first tool bubble of an editable line buffer). if _is_tool_progress(content): return await self._emit_tool_lines(chat_id, content, state, thread_id, is_edit=False) # 4. Commentary (interim assistant beat). message_id = _mint_message_id() state.active = True await self._broadcast_or_log( chat_id, protocol.commentary(chat_id, message_id, content, thread_id=thread_id), ) return SendResult(success=True, message_id=message_id) async def edit_message( self, chat_id: str, message_id: str, content: str, *, finalize: bool = False, metadata: Optional[Dict[str, Any]] = None, ) -> SendResult: """Edit a previously sent message (M2: drives streaming + tool updates). * ``message_id == state.stream_id`` -> ``message.update`` (``finalize=True`` -> ``message.stop``). * ``message_id == state.tool_msg_id`` -> tool-progress update (new lines -> ``tool.start``). * unknown id -> best-effort ``message.update``. """ content = content or "" thread_id = _thread_id_from_metadata(metadata) state = self._turn_state(chat_id) if message_id and message_id == state.stream_id: if finalize: reasoning, body = _split_reasoning(_strip_streaming_cursor(content)) # Streaming: the gateway drops the model's separate # reasoning_content (final send suppressed), so attach the # reasoning we captured via the on_stream_delta hook (wait # for the async hook worker to flush it first). if not reasoning: await _wait_for_reasoning_flushed() reasoning = _take_reasoning() or None else: _reset_reasoning() state.stream_id = None await self._broadcast_or_log( chat_id, protocol.message_stop( chat_id, message_id, body, reasoning=reasoning, thread_id=thread_id, ts=int(time.time() * 1000), ), ) await self._close_open_tool(chat_id, state, thread_id) self._reset_tool_state(state) state.active = False else: await self._broadcast_or_log( chat_id, protocol.message_update( chat_id, message_id, _strip_streaming_cursor(content), thread_id=thread_id, ), ) return SendResult(success=True, message_id=message_id) if message_id and message_id == state.tool_msg_id: return await self._emit_tool_lines(chat_id, content, state, thread_id, is_edit=True) # Unknown id: treat as a streaming update (best effort). if finalize: await self._broadcast_or_log( chat_id, protocol.message_stop( chat_id, message_id, _strip_streaming_cursor(content), thread_id=thread_id, ts=int(time.time() * 1000), ), ) else: await self._broadcast_or_log( chat_id, protocol.message_update( chat_id, message_id, _strip_streaming_cursor(content), thread_id=thread_id, ), ) return SendResult(success=True, message_id=message_id) # ── M2: tool-progress helpers ───────────────────────────────────────── async def _emit_tool_lines( self, chat_id: str, content: str, state: _TurnState, thread_id: Optional[str], *, is_edit: bool, ) -> SendResult: """Emit ``tool.start`` for each NEW tool line in *content*. The gateway accumulates tool lines in one editable bubble; on an edit the full buffer is re-sent, so we diff against ``seen_tool_lines`` to emit only the new ones. A new tool closes the previously-open tool. """ message_id = state.tool_msg_id or _mint_message_id() state.tool_msg_id = message_id state.active = True lines = [ln for ln in content.splitlines() if ln.strip()] for line in lines: key = line.strip() if key in state.seen_tool_lines: continue state.seen_tool_lines.add(key) parsed = self._parse_tool_line_or_block(line, content) if parsed is None: continue name, preview = parsed # A new tool begins: close the previously-open one. if state.open_tool_index is not None: await self._broadcast_or_log( chat_id, protocol.tool_end(chat_id, state.open_tool_index, "", ok=True, thread_id=thread_id), ) state.tool_index += 1 state.open_tool_index = state.tool_index await self._broadcast_or_log( chat_id, protocol.tool_start( chat_id, state.tool_index, name, preview=preview, thread_id=thread_id, ), ) return SendResult(success=True, message_id=message_id) @staticmethod def _parse_tool_line_or_block(line: str, content: str) -> Optional[Tuple[str, Optional[str]]]: """Parse a tool line, expanding a terminal code block to its command.""" parsed = _parse_tool_line(line) if parsed is not None: name, preview = parsed # Terminal code block: the command lives in the fenced lines that # follow the " terminal" head line. if name == "terminal" and preview is None and "```" in content: cmd = _extract_code_block(content) if cmd: return name, cmd return parsed return None async def _close_open_tool( self, chat_id: str, state: _TurnState, thread_id: Optional[str] ) -> None: """Emit ``tool.end`` for the currently-open tool, if any. A tool is considered complete when the next tool starts OR a new content segment begins (the model only produces content after the tool it was waiting on has returned). """ if state.open_tool_index is not None: await self._broadcast_or_log( chat_id, protocol.tool_end(chat_id, state.open_tool_index, "", ok=True, thread_id=thread_id), ) state.open_tool_index = None def _reset_tool_state(self, state: _TurnState) -> None: """Clear per-turn tool bookkeeping (called at turn finalization).""" state.tool_msg_id = None state.seen_tool_lines = set() state.tool_index = 0 state.open_tool_index = None async def _broadcast_or_log(self, chat_id: str, frame: "protocol.Frame") -> None: delivered = await self._ws_server.broadcast(frame) if delivered == 0: logger.info( "android: no live devices for %s; %s frame not delivered (outbox lands in M3)", chat_id, frame.type, ) async def send_typing(self, chat_id: str, metadata: Optional[Dict[str, Any]] = None) -> None: """Send a typing indicator (``typing`` frame, on=true).""" thread_id = None if metadata: tid = metadata.get("thread_id") if isinstance(tid, str) and tid: thread_id = tid await self._ws_server.broadcast(protocol.typing(chat_id, True, thread_id=thread_id)) async def stop_typing(self, chat_id: str) -> None: """Clear the typing indicator (``typing`` frame, on=false).""" await self._ws_server.broadcast(protocol.typing(chat_id, False)) async def send_image( self, chat_id: str, image_url: str, caption: Optional[str] = None, reply_to: Optional[str] = None, metadata: Optional[Dict[str, Any]] = None, ) -> SendResult: """Send an image. M1: not implemented (M4).""" return SendResult(success=False, error="android: media not implemented yet (M4)") # ── Inbound (app -> agent) ──────────────────────────────────────────── async def on_message_send(self, frame: protocol.Frame, device_id: str) -> None: """Handle an inbound ``message.send`` frame. Echoes the user message to all devices (multi-device sync + ack), then builds a ``MessageEvent`` and hands it to ``handle_message()`` (the gateway's command pipeline + agent turn). """ payload = frame.payload text = payload.get("text") if not isinstance(text, str) or not text.strip(): await self._ws_server.send_to( device_id, protocol.error(protocol.ERR_UNSUPPORTED, "message.send requires non-empty text", id=frame.id), ) return chat_id = frame.chat_id or payload.get("chat_id") if not isinstance(chat_id, str) or not chat_id.strip(): chat_id = self.home_channel chat_id = chat_id.strip() thread_id = frame.thread_id or payload.get("thread_id") if not isinstance(thread_id, str) or not thread_id.strip(): thread_id = None reply_to = payload.get("reply_to") if not isinstance(reply_to, str) or not reply_to.strip(): reply_to = None device = self._devices.get(device_id) or {} user_name = device.get("name") or device_id # Echo to all devices: the sender confirms (server-assigned id), # other devices see the message too (single-user, multi-device). message_id = f"m_{uuid.uuid4().hex[:16]}" echo = protocol.message( chat_id=chat_id, message_id=message_id, role=protocol.ROLE_USER, text=text, thread_id=thread_id, reply_to=reply_to, ts=int(time.time() * 1000), ) await self._ws_server.broadcast(echo) source = self.build_source( chat_id=chat_id, chat_name=self._channel_name(chat_id), chat_type="dm", user_id=device_id, user_name=user_name, thread_id=thread_id, ) event = MessageEvent( text=text, message_type=MessageType.TEXT, user_id=device_id, user_name=user_name, source=source, message_id=message_id, reply_to_message_id=reply_to, ) await self.handle_message(event) # ── Chat info ───────────────────────────────────────────────────────── def _channel_name(self, chat_id: str) -> str: """Channel display name. M1: home channel only (directory is M3).""" if chat_id in (self.home_channel, DEFAULT_HOME_CHANNEL): return self.home_channel_name return chat_id or "chat" async def get_chat_info(self, chat_id: str) -> Dict[str, Any]: """Return ``{name, type, chat_id}`` for a chat. M1: the channel directory is not persisted yet, so report the home channel name for the default chat and the raw id otherwise. """ return { "name": self._channel_name(chat_id), "type": "channel", "chat_id": chat_id, } # ── hello.ack helpers ───────────────────────────────────────────────── def server_caps(self) -> Dict[str, Any]: """Capability flags advertised in ``hello.ack`` (M2 surface).""" return { "streaming": True, # M2: message.start/update/stop "reasoning": True, # M2: reasoning field on message / message.stop "tools": True, # M2: tool.start/progress/end "media": False, # M4 "search": False, # M3 "push": self.push_backend, "pickers": False, # M2+ } def channel_list(self) -> List[Dict[str, Any]]: """Channel directory for ``hello.ack``. M1: home channel only (M3).""" return [ { "chat_id": self.home_channel, "name": self.home_channel_name, "kind": "default", "is_default": True, } ] # --------------------------------------------------------------------------- # Plugin entry point # --------------------------------------------------------------------------- def register(ctx): """Plugin entry point: called by the Hermes plugin system.""" # M2: capture the model's separate reasoning_content during streaming so # it can be attached to the turn's message.stop frame (the gateway # otherwise drops it when streaming suppresses the final send). try: ctx.register_hook("on_stream_delta", _on_stream_delta) except Exception: logger.debug("android: on_stream_delta hook registration failed", exc_info=True) ctx.register_platform( name="android", label="Android", adapter_factory=lambda cfg: AndroidAdapter(cfg), check_fn=check_requirements, validate_config=validate_config, is_connected=is_connected, required_env=["ANDROID_TOKEN"], install_hint="No extra packages needed (websockets + httpx are core deps)", setup_fn=interactive_setup, # Env-driven auto-configuration: seeds PlatformConfig.extra with # host/port/push_backend + home_channel so env-only setups show up in # gateway status without instantiating the adapter. env_enablement_fn=_env_enablement, # Cron home-channel delivery support (deliver=android:[:]). cron_deliver_env_var="ANDROID_HOME_CHANNEL", # Out-of-process cron delivery (best-effort; outbox is gateway-served). standalone_sender_fn=_standalone_send, # Native target syntax: "android:[:]". parse_target_ref_fn=_parse_target_ref, # Auth env vars for _is_user_authorized() integration. allowed_users_env="ANDROID_ALLOWED_USERS", allow_all_env="ANDROID_ALLOW_ALL_USERS", # WS has no message-size limit. max_message_length=0, # Display. emoji="📱", pii_safe=False, allow_update_command=True, # LLM guidance. platform_hint=( "You are chatting with the user through their native Iris app " "(Android/Desktop). It renders Markdown, inline code, images, " "audio and video, and shows your reasoning and tool activity. " "Conversations are organized into channels and optional threads. " "Keep formatting rich but readable." ), )