Runtime footer: app-controlled model/context/cwd/latency/cost under replies

Hermes can append a text "runtime footer" (model, context %, workdir,
latency, cost) to final replies, but only when display.runtime_footer is
enabled in the hermes config. We want the same info but controlled by the
APP, not the gateway config. So the gateway now ALWAYS sends the data as a
structured `runtime` object on final assistant messages, and the app decides
whether/what to show.

Gateway (gateway-plugin/):
- protocol.py: new runtime_footer() helper + `runtime` field on the
  message / message.stop frames. Keys (all optional, absent when the data
  is unavailable — e.g. no cost for local models): model (vendor prefix
  dropped), context_pct (0-100), cwd (home-relative), latency (seconds),
  cost (USD).
- adapter.py: a post_api_request plugin hook captures the turn's model +
  prompt tokens + start time (platform-filtered to android so other
  platforms don't pollute the buffer). _build_runtime_footer() resolves the
  model's context window (cached, best-effort, off the event loop via
  asyncio.to_thread with a timeout) and computes context_pct. The runtime
  object is attached on every final send (streaming message.stop and
  non-streaming message, plus the fallback paths).
- outbox.py: `runtime` preserved in history reconstruction so the footer
  survives a restart / first open.

App (app/shared/):
- Protocol.kt: RuntimeMeta data class + `runtime` on MessagePayload /
  MessageStopPayload / HistoryMessage.
- ChatStore.kt: `runtime` on MessageItem, wired through live + history
  reconciliation.
- SecureStore.kt (+ Android/Desktop actuals): runtimeFooterEnabled +
  runtimeFooterFields (persisted per device).
- IrisController.kt: StateFlows + toggleRuntimeFooter() /
  toggleRuntimeField(); RUNTIME_FIELD_KEYS / default set / parser.
- SettingsScreen.kt: "Runtime footer" switch; when on, an expandable chip
  menu (Model · Context % · Workdir · Latency · Cost) to pick fields.
- ChatScreen.kt: footer rendered on the SAME line as the timestamp (footer
  left, time right, Telegram-style), only for final non-streaming assistant
  answers; Inspector pane now shows the runtime fields too.

Docs: 04-wire-protocol.md + frames.schema.json document the `runtime`
object.

Verified end-to-end on device: final replies carry
`qwen3.8-27B-exl3-4.5bpw · 53% · ~ · 38s` with the time right-aligned on
the same line; 69/69 gateway tests pass, Kotlin builds + tests pass.
This commit is contained in:
ARIA committed 2026-08-21 19:32:05 +02:00
1 parent 678c0344c8
commit 9286937e2d
13 files changed
+979 -290

No files matched your search

+150 -3
View File
@@ -317,6 +317,132 @@ def _tool_end_fields(tool_name: str) -> dict[str, Any]:
return fields
# ---------------------------------------------------------------------------
# Runtime-metadata footer (post_api_request hook)
#
# The app renders a Telegram-style footer under final assistant messages
# (model, context %, cwd, latency, cost). Display is controlled by the APP
# (Settings → Runtime footer), not hermes config — so the gateway ALWAYS
# sends the data. hermes core only appends its own *text* footer when
# ``display.runtime_footer.enabled`` is set, and the adapter has no access to
# the gateway's ``agent_result``, so we capture the same facts ourselves via
# the ``post_api_request`` plugin hook (fires after every provider call with
# model + usage):
#
# * model — the turn's latest model (failover-aware)
# * prompt_tokens — the latest call's prompt size (context occupancy)
# * turn start — the first API call of the turn (latency baseline)
#
# Global buffer (same pattern as the reasoning/tool buffers): a personal
# android gateway serves one active turn at a time. The hook fires for every
# platform, so we only record when the turn's platform is android.
# ---------------------------------------------------------------------------
_runtime_meta: dict[str, Any] = {}
_runtime_meta_lock = threading.Lock()
# Per-model context-window cache. Resolution may probe endpoints on first use
# (slow); the cache is process-lifetime so each model resolves at most once.
_context_length_cache: dict[str, int] = {}
# Upper bound (seconds) on context-window resolution during a final send, so a
# slow first-use probe never delays the reply. The worker thread keeps running
# and populates the cache, so the next turn is fast.
_CTX_RESOLVE_TIMEOUT_S = 3.0
def _on_post_api_request(**kwargs: Any) -> None:
"""Plugin hook: capture per-turn runtime metadata (model, prompt tokens)."""
platform = kwargs.get("platform")
if platform and platform != "android":
return
model = kwargs.get("model") or ""
usage = kwargs.get("usage") or {}
prompt_tokens = usage.get("prompt_tokens") or 0
with _runtime_meta_lock:
if model:
_runtime_meta["model"] = model
if prompt_tokens:
_runtime_meta["prompt_tokens"] = prompt_tokens
if "turn_start" not in _runtime_meta:
_runtime_meta["turn_start"] = time.monotonic()
def _take_runtime_meta() -> dict[str, Any]:
"""Drain the captured turn metadata (turn boundary)."""
with _runtime_meta_lock:
meta = dict(_runtime_meta)
_runtime_meta.clear()
return meta
def _resolve_context_length(model: str) -> int | None:
"""Best-effort context window for *model* (cached; None on failure).
Runs in a worker thread (may probe endpoints on first use). The cache is
populated even if the caller's asyncio task times out, so subsequent
turns resolve instantly.
"""
if not model:
return None
cached = _context_length_cache.get(model)
if cached:
return cached
try:
from agent.model_metadata import get_model_context_length
ctx = get_model_context_length(model)
if ctx and ctx > 0:
_context_length_cache[model] = int(ctx)
return int(ctx)
except Exception:
logger.debug("android: context-length resolution failed for %s", model, exc_info=True)
return None
def _home_relative_cwd(cwd: str) -> str:
"""Collapse ``$HOME`` to ``~`` (matches hermes' runtime footer)."""
if not cwd:
return ""
try:
home = os.path.expanduser("~")
p = os.path.abspath(cwd)
if home and (p == home or p.startswith(home + os.sep)):
return "~" + p[len(home) :]
return p
except Exception:
return cwd
async def _build_runtime_footer(meta: dict[str, Any]) -> dict[str, Any]:
"""Build the ``runtime`` footer object from captured turn metadata.
Called on every final send (the app decides what to show). Fields without
data are omitted. ``meta`` is the drained turn buffer (model,
prompt_tokens, turn_start).
"""
model = (meta.get("model") or "").rsplit("/", 1)[-1]
prompt_tokens = meta.get("prompt_tokens") or 0
context_pct = None
if prompt_tokens and model:
try:
ctx_len = await asyncio.wait_for(
asyncio.to_thread(_resolve_context_length, model),
timeout=_CTX_RESOLVE_TIMEOUT_S,
)
except (asyncio.TimeoutError, Exception):
ctx_len = None
if ctx_len:
context_pct = round(prompt_tokens / ctx_len * 100)
turn_start = meta.get("turn_start")
latency = (time.monotonic() - turn_start) if turn_start else None
cwd = _home_relative_cwd(os.environ.get("TERMINAL_CWD", ""))
return protocol.runtime_footer(
model=model or None,
context_pct=context_pct,
cwd=cwd or None,
latency=latency,
)
# ---------------------------------------------------------------------------
# Defaults
# ---------------------------------------------------------------------------
@@ -1159,7 +1285,7 @@ class AndroidAdapter(BasePlatformAdapter):
state = self._turn_state(chat_id)
# 1. Streaming segment start (stream consumer first send).
if meta.get("expect_edits") is True:
if bool(meta.get("expect_edits")):
# A new content segment means the tool the model was waiting on
# has returned -> close it before the segment opens.
await self._close_open_tool(chat_id, state, thread_id)
@@ -1213,7 +1339,7 @@ class AndroidAdapter(BasePlatformAdapter):
return SendResult(success=True, message_id=message_id)
# 3. Final message (non-streaming final, or streaming fallback final).
if meta.get("notify") is True:
if bool(meta.get("notify")):
reasoning, body = _split_reasoning(content)
# Non-streaming: reasoning is prepended to content (split above).
# Streaming fallback: content has no reasoning, so use the
@@ -1224,6 +1350,11 @@ class AndroidAdapter(BasePlatformAdapter):
reasoning = _take_reasoning() or None
else:
_reset_reasoning()
# Runtime-metadata footer (app-controlled display): always attach
# the structured ``runtime`` object so the app can render its
# footer (Settings → Runtime footer). Independent of hermes'
# ``display.runtime_footer`` config.
runtime = await _build_runtime_footer(_take_runtime_meta())
if state.stream_id:
# Fallback final: close the open streaming segment in place.
message_id = state.stream_id
@@ -1236,6 +1367,7 @@ class AndroidAdapter(BasePlatformAdapter):
body,
reasoning=reasoning,
thread_id=thread_id,
runtime=runtime or None,
ts=int(time.time() * 1000),
),
)
@@ -1251,6 +1383,7 @@ class AndroidAdapter(BasePlatformAdapter):
thread_id=thread_id,
reasoning=reasoning,
reply_to=reply_to,
runtime=runtime or None,
ts=int(time.time() * 1000),
),
)
@@ -1310,6 +1443,8 @@ class AndroidAdapter(BasePlatformAdapter):
reasoning = _take_reasoning() or None
else:
_reset_reasoning()
# Runtime-metadata footer (app-controlled display).
runtime = await _build_runtime_footer(_take_runtime_meta())
state.stream_id = None
await self._broadcast_or_log(
chat_id,
@@ -1319,6 +1454,7 @@ class AndroidAdapter(BasePlatformAdapter):
body,
reasoning=reasoning,
thread_id=thread_id,
runtime=runtime or None,
ts=int(time.time() * 1000),
),
)
@@ -1344,6 +1480,7 @@ class AndroidAdapter(BasePlatformAdapter):
# Unknown id: treat as a streaming update (best effort).
if finalize:
runtime = await _build_runtime_footer(_take_runtime_meta())
await self._broadcast_or_log(
chat_id,
protocol.message_stop(
@@ -1351,6 +1488,7 @@ class AndroidAdapter(BasePlatformAdapter):
message_id,
_strip_streaming_cursor(content),
thread_id=thread_id,
runtime=runtime or None,
ts=int(time.time() * 1000),
),
)
@@ -1857,7 +1995,7 @@ class AndroidAdapter(BasePlatformAdapter):
# Skipped for slash commands (session-scoped, not conversation
# starters) and replies (they continue where the user is). Threading
# is only active on the default channel; other channels stay flat.
auto_thread = payload.get("auto_thread") is True
auto_thread = bool(payload.get("auto_thread"))
default_entry = self._channels.default()
if (
auto_thread
@@ -2825,6 +2963,15 @@ def register(ctx):
ctx.register_hook("post_tool_call", _on_post_tool_call)
except Exception:
logger.debug("android: post_tool_call hook registration failed", exc_info=True)
# Runtime-metadata footer: capture the turn's model + prompt tokens (per
# provider call) so the final message can carry a structured ``runtime``
# object. The app decides whether/what to show (Settings → Runtime
# footer); the gateway always sends the data (independent of hermes
# ``display.runtime_footer`` config).
try:
ctx.register_hook("post_api_request", _on_post_api_request)
except Exception:
logger.debug("android: post_api_request hook registration failed", exc_info=True)
ctx.register_platform(
name="android",
label="Android",
+7 -7
View File
@@ -216,6 +216,7 @@ class Outbox:
"reasoning": payload.get("reasoning"),
"model": payload.get("model"),
"tokens": payload.get("tokens"),
"runtime": payload.get("runtime"),
"ts": payload.get("ts"),
"media": payload.get("media"),
}
@@ -230,6 +231,7 @@ class Outbox:
"reasoning": payload.get("reasoning"),
"model": payload.get("model"),
"tokens": payload.get("tokens"),
"runtime": payload.get("runtime"),
"ts": payload.get("ts"),
"media": None,
}
@@ -259,7 +261,7 @@ class Outbox:
# Omit absent optional fields (the app's serializer treats a
# missing key as its default, but a JSON ``null`` for a
# non-nullable field like ``media`` would fail to parse).
for key in ("reasoning", "model", "tokens", "ts", "media"):
for key in ("reasoning", "model", "tokens", "runtime", "ts", "media"):
if m.get(key) is None:
m.pop(key, None)
return {
@@ -310,12 +312,10 @@ class Outbox:
cursors.append(int(r["cursor"]))
if not cursors:
return 0
placeholders = ",".join("?" * len(cursors))
sql = f"DELETE FROM outbox WHERE cursor IN ({placeholders})"
# Safe: ``placeholders`` is only ``?`` markers; every cursor value is
# bound as a parameter (no user data in the SQL text).
# pi-lens-ignore: python-sql-injection
self._conn.execute(sql, cursors)
# One bound-parameter delete per cursor (a message spans only a few
# frames); same transaction, no string-built SQL.
for cursor in cursors:
self._conn.execute("DELETE FROM outbox WHERE cursor = ?", (cursor,))
self._conn.commit()
return len(cursors)
+57 -9
View File
@@ -249,6 +249,54 @@ def hello_ack(
)
# ---------------------------------------------------------------------------
# Runtime-metadata footer (app-controlled display)
#
# The gateway ALWAYS attaches a structured ``runtime`` object to final
# assistant messages so the app can render a Telegram-style footer (model,
# context %, cwd, latency, cost). Whether/what is shown is a per-app setting,
# NOT a hermes config — the gateway sends the data unconditionally and the
# app decides. Mirrors hermes' ``gateway/runtime_footer.py`` fields but as
# structured data (the app formats + picks fields).
#
# Recognised keys (all optional; absent when the data is unavailable):
# model — bare model id, vendor prefix dropped (``gpt-5.4``)
# context_pct — last-call context occupancy, 0-100 (int)
# cwd — home-relative working dir (``~``)
# latency — wall-clock turn duration, seconds (float)
# cost — turn cost, USD (float); absent for local/free models
# ---------------------------------------------------------------------------
RUNTIME_FIELDS: tuple[str, ...] = ("model", "context_pct", "cwd", "latency", "cost")
def runtime_footer(
*,
model: str | None = None,
context_pct: int | None = None,
cwd: str | None = None,
latency: float | None = None,
cost: float | None = None,
) -> dict[str, Any]:
"""Build the structured ``runtime`` footer object.
Only fields with data are included (a partially-populated footer is
better than empty slots). Returns ``{}`` when nothing is available.
"""
d: dict[str, Any] = {}
if model:
d["model"] = model
if context_pct is not None:
d["context_pct"] = max(0, min(100, int(context_pct)))
if cwd:
d["cwd"] = cwd
if latency is not None and latency >= 0:
d["latency"] = round(latency, 3)
if cost is not None and cost > 0:
d["cost"] = round(cost, 6)
return d
# Frame builder mirrors the wire schema (docs/04); the many fields are the
# message's full shape, so the arg count is intentional.
def message( # noqa: PLR0913
@@ -263,6 +311,7 @@ def message( # noqa: PLR0913
reply_to: str | None = None,
model: str | None = None,
tokens: int | None = None,
runtime: dict[str, Any] | None = None,
ts: int | None = None,
) -> Frame:
payload: dict[str, Any] = {
@@ -280,11 +329,11 @@ def message( # noqa: PLR0913
payload["model"] = model
if tokens is not None:
payload["tokens"] = tokens
if runtime:
payload["runtime"] = runtime
if ts is not None:
payload["ts"] = ts
return Frame(
type=TYPE_MESSAGE, chat_id=chat_id, thread_id=thread_id, payload=payload
)
return Frame(type=TYPE_MESSAGE, chat_id=chat_id, thread_id=thread_id, payload=payload)
def typing(chat_id: str, on: bool = True, *, thread_id: str | None = None) -> Frame:
@@ -342,6 +391,7 @@ def message_stop(
reasoning: str | None = None,
model: str | None = None,
tokens: int | None = None,
runtime: dict[str, Any] | None = None,
ts: int | None = None,
) -> Frame:
"""Finalize a streaming bubble."""
@@ -355,6 +405,8 @@ def message_stop(
payload["model"] = model
if tokens is not None:
payload["tokens"] = tokens
if runtime:
payload["runtime"] = runtime
if ts is not None:
payload["ts"] = ts
return Frame(
@@ -648,9 +700,7 @@ def notification(
payload: dict[str, Any] = {"kind": kind, "title": title, "body": body}
if ts is not None:
payload["ts"] = ts
return Frame(
type=TYPE_NOTIFICATION, chat_id=chat_id, thread_id=thread_id, payload=payload
)
return Frame(type=TYPE_NOTIFICATION, chat_id=chat_id, thread_id=thread_id, payload=payload)
def fcm_register(fcm_token: str | None = None, ntfy_topic: str | None = None) -> Frame:
@@ -713,9 +763,7 @@ def media_offer(
}
if message_id:
payload["message_id"] = message_id
return Frame(
type=TYPE_MEDIA_OFFER, chat_id=chat_id, thread_id=thread_id, payload=payload
)
return Frame(type=TYPE_MEDIA_OFFER, chat_id=chat_id, thread_id=thread_id, payload=payload)
def media_pull_end(ok: bool, *, id: int | None = None) -> Frame: