adapter.py was a 3,493-line monolith. Split it into focused modules with clear separation of responsibilities, bringing it down to ~857 lines: - Module-level helpers: hooks, classify, pickers, commands, setup, defaults, secrets - Frame-handler mixins: inbound, tool_frames, push_frames, media_frames, picker_frames, channel_frames, query_frames - mixin_base: IrisAdapterBase (declaration-only base for shared attrs) - adapter.py now holds only IrisAdapter (the composition of the 7 mixins + BasePlatformAdapter), register(), and test-facing re-exports The mixins come before BasePlatformAdapter in the MRO so their methods override the base; super() calls (e.g. send_image) still resolve to BasePlatformAdapter. No circular imports; dispatch.py and http_server.py (instance-method callers) are unaffected. Ruff complexity ceilings (PLR0911/0912/0913/0915) restored to Ruff's built-in defaults (12/50/6/5) instead of "just above the current maxima", which ratchets the bar down as code grows. The existing genuinely-complex functions (frame builders mirroring the wire schema, the QR matrix builder, the dispatch table) carry an explicit `# noqa: PLR09xx` marking them as reviewed, frozen exceptions; new code is held to the default ceilings. All 125 tests green (94 test_android + 31 test_android_http); no new ruff errors introduced.
336 lines
12 KiB
Python
336 lines
12 KiB
Python
"""Outbound frame classification (M2): turn state + content heuristics.
|
|
|
|
The main gateway delivers through the legacy callback path: the stream
|
|
consumer calls ``send()`` (first bubble of a segment) and ``edit_message()``
|
|
(updates), tool progress flows through ``send()``/``edit_message()`` of an
|
|
accumulated line buffer, and interim commentary arrives as a plain
|
|
``send()``. The adapter classifies each outbound call into a structured frame
|
|
using a per-chat turn state machine + the content markers below:
|
|
|
|
* ``metadata["expect_edits"] is True`` -> streaming segment start
|
|
* ``metadata["notify"] is True`` -> final message (or fallback final)
|
|
* tool-progress line format -> tool.start / tool.end
|
|
* anything else -> commentary
|
|
|
|
Verified empirically against the live gateway with ``tests/ws_probe.py``.
|
|
"""
|
|
|
|
import json
|
|
import logging
|
|
import re
|
|
import uuid
|
|
from dataclasses import dataclass, field
|
|
from typing import Any
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_STREAMING_CURSOR = " ▉"
|
|
|
|
# Code-style reasoning prefix (gateway/run.py, reasoning_style="code"):
|
|
# "💭 **Reasoning:**\n```\n<reasoning>\n```\n\n<response>"
|
|
_REASONING_PREFIX = "💭 **Reasoning:**\n```\n"
|
|
_REASONING_CLOSE = "\n```\n\n"
|
|
|
|
# A gateway tool-progress line begins with a (non-ASCII) tool emoji.
|
|
_TOOL_LINE_RE = re.compile(r"^(\S+)\s+(.+)$")
|
|
_TOOL_NAME_PREVIEW_RE = re.compile(r'^(\S+):\s*"(.*)"\s*$')
|
|
_TOOL_NAME_BARE_RE = re.compile(r"^(\S+)\.\.\.\s*$")
|
|
_TOOL_NAME_ARGS_RE = re.compile(r"^(\S+)\(([^)]*)\)\s*$")
|
|
# Terminal code block: "💻 terminal\n```\n<cmd>\n```"
|
|
_TOOL_CODEBLOCK_HEAD_RE = re.compile(r"^(\S+)\s+(\S+)\s*$")
|
|
|
|
# Reverse map of the gateway's friendly tool verbs (agent/display.py
|
|
# _TOOL_VERBS) so a verb-form line ("🔍 Searching the web for …") can be
|
|
# recovered to a structured (tool_name, preview). Longest-first matching is
|
|
# done at parse time. Verbs shared by several tools map to the most common.
|
|
_VERB_TO_TOOL: dict[str, str] = {
|
|
"Searching the web": "web_search",
|
|
"Searching files": "search_files",
|
|
"Searching past sessions": "session_search",
|
|
"Running code": "execute_code",
|
|
"Running": "terminal",
|
|
"Reading skill": "skill_view",
|
|
"Reading": "read_file",
|
|
"Writing": "write_file",
|
|
"Editing": "patch",
|
|
"Browsing": "browser_navigate",
|
|
"Clicking": "browser_click",
|
|
"Typing": "browser_type",
|
|
"Generating image": "image_generate",
|
|
"Generating video": "video_generate",
|
|
"Generating speech": "text_to_speech",
|
|
"Looking at the image": "vision_analyze",
|
|
"Listing skills": "skills_list",
|
|
"Updating skill": "skill_manage",
|
|
"Updating memory": "memory",
|
|
"Updating tasks": "todo",
|
|
"Delegating": "delegate_task",
|
|
"Scheduling": "cronjob",
|
|
"Asking": "clarify",
|
|
}
|
|
# Verbs that take a " for " connector before the preview.
|
|
_VERB_FOR_CONNECTOR = {"web_search", "search_files"}
|
|
|
|
|
|
def _mint_message_id() -> str:
|
|
return f"m_{uuid.uuid4().hex[:16]}"
|
|
|
|
|
|
def _mint_picker_id() -> str:
|
|
return f"pc_{uuid.uuid4().hex[:16]}"
|
|
|
|
|
|
def _thread_id_from_metadata(metadata: dict[str, Any] | None) -> str | None:
|
|
if not metadata:
|
|
return None
|
|
tid = metadata.get("thread_id")
|
|
if isinstance(tid, str) and tid:
|
|
return tid
|
|
return None
|
|
|
|
|
|
def _derive_thread_name(text: str) -> str:
|
|
"""Instant auto-thread name from the user's opening message (no model).
|
|
|
|
Reuses hermes' session-title derivation (``agent/title_generator.py``):
|
|
a deterministic slice of the user's own words, so the thread is named the
|
|
moment it is created. The LLM upgrade (``_schedule_thread_title_upgrade``)
|
|
replaces it moments later — the same two-stage titling hermes uses for
|
|
sessions (derived < llm < user).
|
|
"""
|
|
try:
|
|
from agent.title_generator import derive_title
|
|
|
|
title = derive_title(text)
|
|
except Exception:
|
|
logger.debug("Thread name derivation failed", exc_info=True)
|
|
title = None
|
|
return (title or "").strip() or "New thread"
|
|
|
|
|
|
def _strip_streaming_cursor(text: str) -> str:
|
|
if text and text.endswith(_STREAMING_CURSOR):
|
|
return text[: -len(_STREAMING_CURSOR)]
|
|
return text
|
|
|
|
|
|
# M5: coalesce back-to-back pushes for the same chat (a cron delivery parks
|
|
# a notification frame AND a message frame; only the first should push).
|
|
_PUSH_COALESCE_S = 5.0
|
|
|
|
|
|
def _push_preview(text: Any, limit: int = 120) -> str:
|
|
"""Short single-line preview for push bodies (lock-screen privacy: no
|
|
secrets, no full bodies -- full content arrives via ``sync``)."""
|
|
s = " ".join(str(text or "").split())
|
|
if len(s) > limit:
|
|
s = s[: limit - 1] + "…"
|
|
return s
|
|
|
|
|
|
# Cron delivery wrap (cron/scheduler.py ``_deliver_result``,
|
|
# cron.wrap_response: true):
|
|
# "Cronjob Response: <name>\n(job_id: <id>)\n-------------\n\n<content>\n\n
|
|
# To stop or manage this job, send me a new message (e.g. ...)."
|
|
_CRON_WRAP_RE = re.compile(r"^Cronjob Response: (.+?)\n\(job_id: [^)]*\)\n-+\n\n")
|
|
_CRON_FOOTER = "\n\nTo stop or manage this job"
|
|
|
|
|
|
def _cron_brief(content: str, job_id: str) -> tuple[str, str]:
|
|
"""Parse a cron delivery into ``(job_name, inner_text)``.
|
|
|
|
Falls back to ``(job_id, content)`` when the wrap is disabled
|
|
(``cron.wrap_response: false``) or unrecognised.
|
|
"""
|
|
m = _CRON_WRAP_RE.match(content or "")
|
|
if not m:
|
|
return str(job_id or "cron"), (content or "").strip()
|
|
name = m.group(1).strip()
|
|
body = content[m.end() :]
|
|
idx = body.rfind(_CRON_FOOTER)
|
|
if idx != -1:
|
|
body = body[:idx]
|
|
return name, body.strip()
|
|
|
|
|
|
def _split_reasoning(text: str) -> tuple[str | None, str]:
|
|
"""Split a code-style reasoning prefix off the front of *text*.
|
|
|
|
Returns ``(reasoning, body)``; ``reasoning`` is ``None`` when no prefix is
|
|
present (reasoning off / no reasoning / non-code style). Best-effort parse
|
|
of a stable, gateway-owned format: on any mismatch the fallback is
|
|
``(None, full text)`` so the answer still renders.
|
|
"""
|
|
if not text or not text.startswith(_REASONING_PREFIX):
|
|
return None, text
|
|
close_idx = text.find(_REASONING_CLOSE, len(_REASONING_PREFIX))
|
|
if close_idx == -1:
|
|
return None, text
|
|
reasoning = text[len(_REASONING_PREFIX) : close_idx]
|
|
body = text[close_idx + len(_REASONING_CLOSE) :]
|
|
return reasoning, body
|
|
|
|
|
|
def _parse_tool_line(line: str) -> tuple[str, str | None] | None: # noqa: PLR0911
|
|
"""Parse a single gateway tool-progress line into ``(name, preview)``.
|
|
|
|
Returns ``None`` when the line is not a tool line. The gateway formats
|
|
tool lines as ``<emoji> <name>: "<preview>"``, ``<emoji> <name>...``,
|
|
``<emoji> <name>(keys)``, or a friendly verb phrase (``<emoji> <verb> …``).
|
|
The verb form is lossy (no tool name), so we surface the verb as the name.
|
|
"""
|
|
line = line.strip()
|
|
if not line:
|
|
return None
|
|
m = _TOOL_LINE_RE.match(line)
|
|
if not m:
|
|
return None
|
|
emoji, rest = m.group(1), m.group(2)
|
|
if emoji.isascii():
|
|
return None # a tool line always leads with a non-ASCII emoji
|
|
mp = _TOOL_NAME_PREVIEW_RE.match(rest)
|
|
if mp:
|
|
return mp.group(1), mp.group(2)
|
|
mb = _TOOL_NAME_BARE_RE.match(rest)
|
|
if mb:
|
|
return mb.group(1), None
|
|
ma = _TOOL_NAME_ARGS_RE.match(rest)
|
|
if ma:
|
|
return ma.group(1), None
|
|
# Friendly verb phrase: reverse-map to (tool_name, preview).
|
|
verb_parsed = _parse_verb_phrase(rest)
|
|
if verb_parsed is not None:
|
|
return verb_parsed
|
|
# Unrecognised: use the phrase as the label.
|
|
return rest, None
|
|
|
|
|
|
def _parse_verb_phrase(phrase: str) -> tuple[str, str | None] | None:
|
|
"""Reverse-map a friendly verb phrase to ``(tool_name, preview)``.
|
|
|
|
Matches the longest verb first so "Running code" wins over "Running".
|
|
Returns ``None`` when no known verb leads the phrase.
|
|
"""
|
|
for verb in sorted(_VERB_TO_TOOL, key=len, reverse=True):
|
|
tool = _VERB_TO_TOOL[verb]
|
|
if phrase == verb:
|
|
return tool, None
|
|
if tool in _VERB_FOR_CONNECTOR and phrase.startswith(verb + " for "):
|
|
return tool, phrase[len(verb) + len(" for ") :].strip() or None
|
|
if phrase.startswith(verb + " "):
|
|
return tool, phrase[len(verb) + 1 :].strip() or None
|
|
return None
|
|
|
|
|
|
def _extract_code_block(content: str) -> str | None:
|
|
"""Return the first fenced code block's body in *content*, else ``None``.
|
|
|
|
Used to recover the terminal command from a tool-progress code block
|
|
(``<emoji> terminal`` head line + fenced command).
|
|
"""
|
|
m = re.search(r"```[^\n]*\n(.*?)\n```", content, re.DOTALL)
|
|
if m:
|
|
return m.group(1).strip() or None
|
|
return None
|
|
|
|
|
|
def _extract_verbose_args(line: str, content: str) -> dict[str, Any] | None:
|
|
"""Recover the full args dict from a verbose tool line, else ``None``.
|
|
|
|
In verbose mode the gateway renders ``<emoji> <name>(keys)`` on one line
|
|
and the full args JSON on the line that follows. When *line* is such a
|
|
header, return the parsed JSON object from the following line.
|
|
"""
|
|
parts = line.strip().split(None, 1)
|
|
# 2 == "tool name" + "args JSON" on the header line.
|
|
if len(parts) < 2 or not _TOOL_NAME_ARGS_RE.match(parts[1]): # noqa: PLR2004
|
|
return None
|
|
lines = content.splitlines()
|
|
for i, ln in enumerate(lines):
|
|
if ln.strip() != line.strip():
|
|
continue
|
|
for follow_line in lines[i + 1 :]:
|
|
follow = follow_line.strip()
|
|
if not follow:
|
|
continue
|
|
if follow.startswith("{"):
|
|
try:
|
|
obj = json.loads(follow)
|
|
return obj if isinstance(obj, dict) else None
|
|
except Exception:
|
|
return None
|
|
return None # next non-empty line is not the args JSON
|
|
return None
|
|
|
|
|
|
def _short_preview_from_args(args: dict[str, Any], cap: int = 60) -> str | None:
|
|
"""Derive a short one-line preview from a verbose args dict.
|
|
|
|
The verbose line carries no explicit preview, so the Truncated display
|
|
would otherwise lose its one-liner. Use the first non-empty string value
|
|
(whitespace-collapsed, capped) as a stand-in.
|
|
"""
|
|
if not isinstance(args, dict):
|
|
return None
|
|
for value in args.values():
|
|
if isinstance(value, str) and value.strip():
|
|
s = " ".join(value.split())
|
|
return s[: cap - 1] + "…" if len(s) > cap else s
|
|
return None
|
|
|
|
|
|
def _is_tool_progress(content: str) -> bool:
|
|
"""Heuristic: does *content* look like gateway tool-progress line(s)?
|
|
|
|
Tool progress is delivered as one or more lines, each led by a tool emoji
|
|
(or a terminal code block). Commentary is free-form prose. We classify on
|
|
the first non-empty line; subsequent lines of the same bubble are tracked
|
|
by message id, not re-classified.
|
|
"""
|
|
if not content:
|
|
return False
|
|
lines = [ln for ln in content.splitlines() if ln.strip()]
|
|
if not lines:
|
|
return False
|
|
first = lines[0].strip()
|
|
# Terminal code block: "<emoji> terminal" then a fenced command.
|
|
if len(lines) > 1 and lines[1].strip().startswith("```"):
|
|
return _TOOL_CODEBLOCK_HEAD_RE.match(first) is not None
|
|
return _parse_tool_line(first) is not None
|
|
|
|
|
|
def _is_gateway_lifecycle_notice(content: str) -> bool:
|
|
"""True for hermes gateway lifecycle notices (restart / shutdown / online).
|
|
|
|
These are system notices, not tool progress. Their leading ⚠️/♻️ emoji
|
|
would otherwise trip the tool-line heuristic and render them as a
|
|
never-completing tool card (an endless spinner, since no ``tool.end``
|
|
ever arrives for a notice that is not a real tool).
|
|
"""
|
|
if not content:
|
|
return False
|
|
c = content.strip()
|
|
return any(
|
|
marker in c for marker in ("Gateway restarting", "Gateway shutting down", "Gateway online")
|
|
)
|
|
|
|
|
|
@dataclass
|
|
class _TurnState:
|
|
"""Per-chat turn state for outbound frame classification (M2)."""
|
|
|
|
active: bool = False
|
|
# message_id of the currently streaming segment (message.start open).
|
|
stream_id: str | None = None
|
|
# message_id of the current tool-progress bubble (editable line buffer).
|
|
tool_msg_id: str | None = None
|
|
# Monotonic per-turn tool counter (start -> end correlation).
|
|
tool_index: int = 0
|
|
# Index of the most recently started tool (awaiting tool.end).
|
|
open_tool_index: int | None = None
|
|
# Name of the most recently started tool (matches the post_tool_call
|
|
# record when the tool completes, so tool.end can carry its output).
|
|
open_tool_name: str | None = None
|
|
# Tool lines already emitted as tool.start (dedup across edits).
|
|
seen_tool_lines: set = field(default_factory=set)
|