"""Outbound frame classification (M2): turn state + content heuristics. The main gateway delivers through the legacy callback path: the stream consumer calls ``send()`` (first bubble of a segment) and ``edit_message()`` (updates), tool progress flows through ``send()``/``edit_message()`` of an accumulated line buffer, and interim commentary arrives as a plain ``send()``. The adapter classifies each outbound call into a structured frame using a per-chat turn state machine + the content markers below: * ``metadata["expect_edits"] is True`` -> streaming segment start * ``metadata["notify"] is True`` -> final message (or fallback final) * tool-progress line format -> tool.start / tool.end * anything else -> commentary Verified empirically against the live gateway with ``tests/ws_probe.py``. """ import json import logging import re import uuid from dataclasses import dataclass, field from typing import Any logger = logging.getLogger(__name__) _STREAMING_CURSOR = " ▉" # Code-style reasoning prefix (gateway/run.py, reasoning_style="code"): # "💭 **Reasoning:**\n```\n\n```\n\n" _REASONING_PREFIX = "💭 **Reasoning:**\n```\n" _REASONING_CLOSE = "\n```\n\n" # A gateway tool-progress line begins with a (non-ASCII) tool emoji. _TOOL_LINE_RE = re.compile(r"^(\S+)\s+(.+)$") _TOOL_NAME_PREVIEW_RE = re.compile(r'^(\S+):\s*"(.*)"\s*$') _TOOL_NAME_BARE_RE = re.compile(r"^(\S+)\.\.\.\s*$") _TOOL_NAME_ARGS_RE = re.compile(r"^(\S+)\(([^)]*)\)\s*$") # Terminal code block: "💻 terminal\n```\n\n```" _TOOL_CODEBLOCK_HEAD_RE = re.compile(r"^(\S+)\s+(\S+)\s*$") # Reverse map of the gateway's friendly tool verbs (agent/display.py # _TOOL_VERBS) so a verb-form line ("🔍 Searching the web for …") can be # recovered to a structured (tool_name, preview). Longest-first matching is # done at parse time. Verbs shared by several tools map to the most common. _VERB_TO_TOOL: dict[str, str] = { "Searching the web": "web_search", "Searching files": "search_files", "Searching past sessions": "session_search", "Running code": "execute_code", "Running": "terminal", "Reading skill": "skill_view", "Reading": "read_file", "Writing": "write_file", "Editing": "patch", "Browsing": "browser_navigate", "Clicking": "browser_click", "Typing": "browser_type", "Generating image": "image_generate", "Generating video": "video_generate", "Generating speech": "text_to_speech", "Looking at the image": "vision_analyze", "Listing skills": "skills_list", "Updating skill": "skill_manage", "Updating memory": "memory", "Updating tasks": "todo", "Delegating": "delegate_task", "Scheduling": "cronjob", "Asking": "clarify", } # Verbs that take a " for " connector before the preview. _VERB_FOR_CONNECTOR = {"web_search", "search_files"} def _mint_message_id() -> str: return f"m_{uuid.uuid4().hex[:16]}" def _mint_picker_id() -> str: return f"pc_{uuid.uuid4().hex[:16]}" def _thread_id_from_metadata(metadata: dict[str, Any] | None) -> str | None: if not metadata: return None tid = metadata.get("thread_id") if isinstance(tid, str) and tid: return tid return None def _derive_thread_name(text: str) -> str: """Instant auto-thread name from the user's opening message (no model). Reuses hermes' session-title derivation (``agent/title_generator.py``): a deterministic slice of the user's own words, so the thread is named the moment it is created. The LLM upgrade (``_schedule_thread_title_upgrade``) replaces it moments later — the same two-stage titling hermes uses for sessions (derived < llm < user). """ try: from agent.title_generator import derive_title title = derive_title(text) except Exception: logger.debug("Thread name derivation failed", exc_info=True) title = None return (title or "").strip() or "New thread" def _strip_streaming_cursor(text: str) -> str: if text and text.endswith(_STREAMING_CURSOR): return text[: -len(_STREAMING_CURSOR)] return text # M5: coalesce back-to-back pushes for the same chat (a cron delivery parks # a notification frame AND a message frame; only the first should push). _PUSH_COALESCE_S = 5.0 def _push_preview(text: Any, limit: int = 120) -> str: """Short single-line preview for push bodies (lock-screen privacy: no secrets, no full bodies -- full content arrives via ``sync``).""" s = " ".join(str(text or "").split()) if len(s) > limit: s = s[: limit - 1] + "…" return s # Cron delivery wrap (cron/scheduler.py ``_deliver_result``, # cron.wrap_response: true): # "Cronjob Response: \n(job_id: )\n-------------\n\n\n\n # To stop or manage this job, send me a new message (e.g. ...)." _CRON_WRAP_RE = re.compile(r"^Cronjob Response: (.+?)\n\(job_id: [^)]*\)\n-+\n\n") _CRON_FOOTER = "\n\nTo stop or manage this job" def _cron_brief(content: str, job_id: str) -> tuple[str, str]: """Parse a cron delivery into ``(job_name, inner_text)``. Falls back to ``(job_id, content)`` when the wrap is disabled (``cron.wrap_response: false``) or unrecognised. """ m = _CRON_WRAP_RE.match(content or "") if not m: return str(job_id or "cron"), (content or "").strip() name = m.group(1).strip() body = content[m.end() :] idx = body.rfind(_CRON_FOOTER) if idx != -1: body = body[:idx] return name, body.strip() def _split_reasoning(text: str) -> tuple[str | None, str]: """Split a code-style reasoning prefix off the front of *text*. Returns ``(reasoning, body)``; ``reasoning`` is ``None`` when no prefix is present (reasoning off / no reasoning / non-code style). Best-effort parse of a stable, gateway-owned format: on any mismatch the fallback is ``(None, full text)`` so the answer still renders. """ if not text or not text.startswith(_REASONING_PREFIX): return None, text close_idx = text.find(_REASONING_CLOSE, len(_REASONING_PREFIX)) if close_idx == -1: return None, text reasoning = text[len(_REASONING_PREFIX) : close_idx] body = text[close_idx + len(_REASONING_CLOSE) :] return reasoning, body def _parse_tool_line(line: str) -> tuple[str, str | None] | None: # noqa: PLR0911 """Parse a single gateway tool-progress line into ``(name, preview)``. Returns ``None`` when the line is not a tool line. The gateway formats tool lines as `` : ""``, `` ...``, `` (keys)``, or a friendly verb phrase (`` …``). The verb form is lossy (no tool name), so we surface the verb as the name. """ line = line.strip() if not line: return None m = _TOOL_LINE_RE.match(line) if not m: return None emoji, rest = m.group(1), m.group(2) if emoji.isascii(): return None # a tool line always leads with a non-ASCII emoji mp = _TOOL_NAME_PREVIEW_RE.match(rest) if mp: return mp.group(1), mp.group(2) mb = _TOOL_NAME_BARE_RE.match(rest) if mb: return mb.group(1), None ma = _TOOL_NAME_ARGS_RE.match(rest) if ma: return ma.group(1), None # Friendly verb phrase: reverse-map to (tool_name, preview). verb_parsed = _parse_verb_phrase(rest) if verb_parsed is not None: return verb_parsed # Unrecognised: use the phrase as the label. return rest, None def _parse_verb_phrase(phrase: str) -> tuple[str, str | None] | None: """Reverse-map a friendly verb phrase to ``(tool_name, preview)``. Matches the longest verb first so "Running code" wins over "Running". Returns ``None`` when no known verb leads the phrase. """ for verb in sorted(_VERB_TO_TOOL, key=len, reverse=True): tool = _VERB_TO_TOOL[verb] if phrase == verb: return tool, None if tool in _VERB_FOR_CONNECTOR and phrase.startswith(verb + " for "): return tool, phrase[len(verb) + len(" for ") :].strip() or None if phrase.startswith(verb + " "): return tool, phrase[len(verb) + 1 :].strip() or None return None def _extract_code_block(content: str) -> str | None: """Return the first fenced code block's body in *content*, else ``None``. Used to recover the terminal command from a tool-progress code block (`` terminal`` head line + fenced command). """ m = re.search(r"```[^\n]*\n(.*?)\n```", content, re.DOTALL) if m: return m.group(1).strip() or None return None def _extract_verbose_args(line: str, content: str) -> dict[str, Any] | None: """Recover the full args dict from a verbose tool line, else ``None``. In verbose mode the gateway renders `` (keys)`` on one line and the full args JSON on the line that follows. When *line* is such a header, return the parsed JSON object from the following line. """ parts = line.strip().split(None, 1) # 2 == "tool name" + "args JSON" on the header line. if len(parts) < 2 or not _TOOL_NAME_ARGS_RE.match(parts[1]): # noqa: PLR2004 return None lines = content.splitlines() for i, ln in enumerate(lines): if ln.strip() != line.strip(): continue for follow_line in lines[i + 1 :]: follow = follow_line.strip() if not follow: continue if follow.startswith("{"): try: obj = json.loads(follow) return obj if isinstance(obj, dict) else None except Exception: return None return None # next non-empty line is not the args JSON return None def _short_preview_from_args(args: dict[str, Any], cap: int = 60) -> str | None: """Derive a short one-line preview from a verbose args dict. The verbose line carries no explicit preview, so the Truncated display would otherwise lose its one-liner. Use the first non-empty string value (whitespace-collapsed, capped) as a stand-in. """ if not isinstance(args, dict): return None for value in args.values(): if isinstance(value, str) and value.strip(): s = " ".join(value.split()) return s[: cap - 1] + "…" if len(s) > cap else s return None def _is_tool_progress(content: str) -> bool: """Heuristic: does *content* look like gateway tool-progress line(s)? Tool progress is delivered as one or more lines, each led by a tool emoji (or a terminal code block). Commentary is free-form prose. We classify on the first non-empty line; subsequent lines of the same bubble are tracked by message id, not re-classified. """ if not content: return False lines = [ln for ln in content.splitlines() if ln.strip()] if not lines: return False first = lines[0].strip() # Terminal code block: " terminal" then a fenced command. if len(lines) > 1 and lines[1].strip().startswith("```"): return _TOOL_CODEBLOCK_HEAD_RE.match(first) is not None return _parse_tool_line(first) is not None def _is_gateway_lifecycle_notice(content: str) -> bool: """True for hermes gateway lifecycle notices (restart / shutdown / online). These are system notices, not tool progress. Their leading ⚠️/♻️ emoji would otherwise trip the tool-line heuristic and render them as a never-completing tool card (an endless spinner, since no ``tool.end`` ever arrives for a notice that is not a real tool). """ if not content: return False c = content.strip() return any( marker in c for marker in ("Gateway restarting", "Gateway shutting down", "Gateway online") ) @dataclass class _TurnState: """Per-chat turn state for outbound frame classification (M2).""" active: bool = False # message_id of the currently streaming segment (message.start open). stream_id: str | None = None # message_id of the current tool-progress bubble (editable line buffer). tool_msg_id: str | None = None # Monotonic per-turn tool counter (start -> end correlation). tool_index: int = 0 # Index of the most recently started tool (awaiting tool.end). open_tool_index: int | None = None # Name of the most recently started tool (matches the post_tool_call # record when the tool completes, so tool.end can carry its output). open_tool_name: str | None = None # Tool lines already emitted as tool.start (dedup across edits). seen_tool_lines: set = field(default_factory=set)