Runtime footer: app-controlled model/context/cwd/latency/cost under replies
Hermes can append a text "runtime footer" (model, context %, workdir, latency, cost) to final replies, but only when display.runtime_footer is enabled in the hermes config. We want the same info but controlled by the APP, not the gateway config. So the gateway now ALWAYS sends the data as a structured `runtime` object on final assistant messages, and the app decides whether/what to show. Gateway (gateway-plugin/): - protocol.py: new runtime_footer() helper + `runtime` field on the message / message.stop frames. Keys (all optional, absent when the data is unavailable — e.g. no cost for local models): model (vendor prefix dropped), context_pct (0-100), cwd (home-relative), latency (seconds), cost (USD). - adapter.py: a post_api_request plugin hook captures the turn's model + prompt tokens + start time (platform-filtered to android so other platforms don't pollute the buffer). _build_runtime_footer() resolves the model's context window (cached, best-effort, off the event loop via asyncio.to_thread with a timeout) and computes context_pct. The runtime object is attached on every final send (streaming message.stop and non-streaming message, plus the fallback paths). - outbox.py: `runtime` preserved in history reconstruction so the footer survives a restart / first open. App (app/shared/): - Protocol.kt: RuntimeMeta data class + `runtime` on MessagePayload / MessageStopPayload / HistoryMessage. - ChatStore.kt: `runtime` on MessageItem, wired through live + history reconciliation. - SecureStore.kt (+ Android/Desktop actuals): runtimeFooterEnabled + runtimeFooterFields (persisted per device). - IrisController.kt: StateFlows + toggleRuntimeFooter() / toggleRuntimeField(); RUNTIME_FIELD_KEYS / default set / parser. - SettingsScreen.kt: "Runtime footer" switch; when on, an expandable chip menu (Model · Context % · Workdir · Latency · Cost) to pick fields. - ChatScreen.kt: footer rendered on the SAME line as the timestamp (footer left, time right, Telegram-style), only for final non-streaming assistant answers; Inspector pane now shows the runtime fields too. Docs: 04-wire-protocol.md + frames.schema.json document the `runtime` object. Verified end-to-end on device: final replies carry `qwen3.8-27B-exl3-4.5bpw · 53% · ~ · 38s` with the time right-aligned on the same line; 69/69 gateway tests pass, Kotlin builds + tests pass.
This commit is contained in:
1 parent
678c0344c8
commit
9286937e2d
13 files changed
+979
-290
No files matched your search
+150
-3
@@ -317,6 +317,132 @@ def _tool_end_fields(tool_name: str) -> dict[str, Any]:
|
||||
return fields
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Runtime-metadata footer (post_api_request hook)
|
||||
#
|
||||
# The app renders a Telegram-style footer under final assistant messages
|
||||
# (model, context %, cwd, latency, cost). Display is controlled by the APP
|
||||
# (Settings → Runtime footer), not hermes config — so the gateway ALWAYS
|
||||
# sends the data. hermes core only appends its own *text* footer when
|
||||
# ``display.runtime_footer.enabled`` is set, and the adapter has no access to
|
||||
# the gateway's ``agent_result``, so we capture the same facts ourselves via
|
||||
# the ``post_api_request`` plugin hook (fires after every provider call with
|
||||
# model + usage):
|
||||
#
|
||||
# * model — the turn's latest model (failover-aware)
|
||||
# * prompt_tokens — the latest call's prompt size (context occupancy)
|
||||
# * turn start — the first API call of the turn (latency baseline)
|
||||
#
|
||||
# Global buffer (same pattern as the reasoning/tool buffers): a personal
|
||||
# android gateway serves one active turn at a time. The hook fires for every
|
||||
# platform, so we only record when the turn's platform is android.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_runtime_meta: dict[str, Any] = {}
|
||||
_runtime_meta_lock = threading.Lock()
|
||||
# Per-model context-window cache. Resolution may probe endpoints on first use
|
||||
# (slow); the cache is process-lifetime so each model resolves at most once.
|
||||
_context_length_cache: dict[str, int] = {}
|
||||
# Upper bound (seconds) on context-window resolution during a final send, so a
|
||||
# slow first-use probe never delays the reply. The worker thread keeps running
|
||||
# and populates the cache, so the next turn is fast.
|
||||
_CTX_RESOLVE_TIMEOUT_S = 3.0
|
||||
|
||||
|
||||
def _on_post_api_request(**kwargs: Any) -> None:
|
||||
"""Plugin hook: capture per-turn runtime metadata (model, prompt tokens)."""
|
||||
platform = kwargs.get("platform")
|
||||
if platform and platform != "android":
|
||||
return
|
||||
model = kwargs.get("model") or ""
|
||||
usage = kwargs.get("usage") or {}
|
||||
prompt_tokens = usage.get("prompt_tokens") or 0
|
||||
with _runtime_meta_lock:
|
||||
if model:
|
||||
_runtime_meta["model"] = model
|
||||
if prompt_tokens:
|
||||
_runtime_meta["prompt_tokens"] = prompt_tokens
|
||||
if "turn_start" not in _runtime_meta:
|
||||
_runtime_meta["turn_start"] = time.monotonic()
|
||||
|
||||
|
||||
def _take_runtime_meta() -> dict[str, Any]:
|
||||
"""Drain the captured turn metadata (turn boundary)."""
|
||||
with _runtime_meta_lock:
|
||||
meta = dict(_runtime_meta)
|
||||
_runtime_meta.clear()
|
||||
return meta
|
||||
|
||||
|
||||
def _resolve_context_length(model: str) -> int | None:
|
||||
"""Best-effort context window for *model* (cached; None on failure).
|
||||
|
||||
Runs in a worker thread (may probe endpoints on first use). The cache is
|
||||
populated even if the caller's asyncio task times out, so subsequent
|
||||
turns resolve instantly.
|
||||
"""
|
||||
if not model:
|
||||
return None
|
||||
cached = _context_length_cache.get(model)
|
||||
if cached:
|
||||
return cached
|
||||
try:
|
||||
from agent.model_metadata import get_model_context_length
|
||||
|
||||
ctx = get_model_context_length(model)
|
||||
if ctx and ctx > 0:
|
||||
_context_length_cache[model] = int(ctx)
|
||||
return int(ctx)
|
||||
except Exception:
|
||||
logger.debug("android: context-length resolution failed for %s", model, exc_info=True)
|
||||
return None
|
||||
|
||||
|
||||
def _home_relative_cwd(cwd: str) -> str:
|
||||
"""Collapse ``$HOME`` to ``~`` (matches hermes' runtime footer)."""
|
||||
if not cwd:
|
||||
return ""
|
||||
try:
|
||||
home = os.path.expanduser("~")
|
||||
p = os.path.abspath(cwd)
|
||||
if home and (p == home or p.startswith(home + os.sep)):
|
||||
return "~" + p[len(home) :]
|
||||
return p
|
||||
except Exception:
|
||||
return cwd
|
||||
|
||||
|
||||
async def _build_runtime_footer(meta: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Build the ``runtime`` footer object from captured turn metadata.
|
||||
|
||||
Called on every final send (the app decides what to show). Fields without
|
||||
data are omitted. ``meta`` is the drained turn buffer (model,
|
||||
prompt_tokens, turn_start).
|
||||
"""
|
||||
model = (meta.get("model") or "").rsplit("/", 1)[-1]
|
||||
prompt_tokens = meta.get("prompt_tokens") or 0
|
||||
context_pct = None
|
||||
if prompt_tokens and model:
|
||||
try:
|
||||
ctx_len = await asyncio.wait_for(
|
||||
asyncio.to_thread(_resolve_context_length, model),
|
||||
timeout=_CTX_RESOLVE_TIMEOUT_S,
|
||||
)
|
||||
except (asyncio.TimeoutError, Exception):
|
||||
ctx_len = None
|
||||
if ctx_len:
|
||||
context_pct = round(prompt_tokens / ctx_len * 100)
|
||||
turn_start = meta.get("turn_start")
|
||||
latency = (time.monotonic() - turn_start) if turn_start else None
|
||||
cwd = _home_relative_cwd(os.environ.get("TERMINAL_CWD", ""))
|
||||
return protocol.runtime_footer(
|
||||
model=model or None,
|
||||
context_pct=context_pct,
|
||||
cwd=cwd or None,
|
||||
latency=latency,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Defaults
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1159,7 +1285,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
state = self._turn_state(chat_id)
|
||||
|
||||
# 1. Streaming segment start (stream consumer first send).
|
||||
if meta.get("expect_edits") is True:
|
||||
if bool(meta.get("expect_edits")):
|
||||
# A new content segment means the tool the model was waiting on
|
||||
# has returned -> close it before the segment opens.
|
||||
await self._close_open_tool(chat_id, state, thread_id)
|
||||
@@ -1213,7 +1339,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
|
||||
# 3. Final message (non-streaming final, or streaming fallback final).
|
||||
if meta.get("notify") is True:
|
||||
if bool(meta.get("notify")):
|
||||
reasoning, body = _split_reasoning(content)
|
||||
# Non-streaming: reasoning is prepended to content (split above).
|
||||
# Streaming fallback: content has no reasoning, so use the
|
||||
@@ -1224,6 +1350,11 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
reasoning = _take_reasoning() or None
|
||||
else:
|
||||
_reset_reasoning()
|
||||
# Runtime-metadata footer (app-controlled display): always attach
|
||||
# the structured ``runtime`` object so the app can render its
|
||||
# footer (Settings → Runtime footer). Independent of hermes'
|
||||
# ``display.runtime_footer`` config.
|
||||
runtime = await _build_runtime_footer(_take_runtime_meta())
|
||||
if state.stream_id:
|
||||
# Fallback final: close the open streaming segment in place.
|
||||
message_id = state.stream_id
|
||||
@@ -1236,6 +1367,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
body,
|
||||
reasoning=reasoning,
|
||||
thread_id=thread_id,
|
||||
runtime=runtime or None,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
@@ -1251,6 +1383,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
thread_id=thread_id,
|
||||
reasoning=reasoning,
|
||||
reply_to=reply_to,
|
||||
runtime=runtime or None,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
@@ -1310,6 +1443,8 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
reasoning = _take_reasoning() or None
|
||||
else:
|
||||
_reset_reasoning()
|
||||
# Runtime-metadata footer (app-controlled display).
|
||||
runtime = await _build_runtime_footer(_take_runtime_meta())
|
||||
state.stream_id = None
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
@@ -1319,6 +1454,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
body,
|
||||
reasoning=reasoning,
|
||||
thread_id=thread_id,
|
||||
runtime=runtime or None,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
@@ -1344,6 +1480,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
|
||||
# Unknown id: treat as a streaming update (best effort).
|
||||
if finalize:
|
||||
runtime = await _build_runtime_footer(_take_runtime_meta())
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.message_stop(
|
||||
@@ -1351,6 +1488,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
message_id,
|
||||
_strip_streaming_cursor(content),
|
||||
thread_id=thread_id,
|
||||
runtime=runtime or None,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
@@ -1857,7 +1995,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
# Skipped for slash commands (session-scoped, not conversation
|
||||
# starters) and replies (they continue where the user is). Threading
|
||||
# is only active on the default channel; other channels stay flat.
|
||||
auto_thread = payload.get("auto_thread") is True
|
||||
auto_thread = bool(payload.get("auto_thread"))
|
||||
default_entry = self._channels.default()
|
||||
if (
|
||||
auto_thread
|
||||
@@ -2825,6 +2963,15 @@ def register(ctx):
|
||||
ctx.register_hook("post_tool_call", _on_post_tool_call)
|
||||
except Exception:
|
||||
logger.debug("android: post_tool_call hook registration failed", exc_info=True)
|
||||
# Runtime-metadata footer: capture the turn's model + prompt tokens (per
|
||||
# provider call) so the final message can carry a structured ``runtime``
|
||||
# object. The app decides whether/what to show (Settings → Runtime
|
||||
# footer); the gateway always sends the data (independent of hermes
|
||||
# ``display.runtime_footer`` config).
|
||||
try:
|
||||
ctx.register_hook("post_api_request", _on_post_api_request)
|
||||
except Exception:
|
||||
logger.debug("android: post_api_request hook registration failed", exc_info=True)
|
||||
ctx.register_platform(
|
||||
name="android",
|
||||
label="Android",
|
||||
|
||||
@@ -216,6 +216,7 @@ class Outbox:
|
||||
"reasoning": payload.get("reasoning"),
|
||||
"model": payload.get("model"),
|
||||
"tokens": payload.get("tokens"),
|
||||
"runtime": payload.get("runtime"),
|
||||
"ts": payload.get("ts"),
|
||||
"media": payload.get("media"),
|
||||
}
|
||||
@@ -230,6 +231,7 @@ class Outbox:
|
||||
"reasoning": payload.get("reasoning"),
|
||||
"model": payload.get("model"),
|
||||
"tokens": payload.get("tokens"),
|
||||
"runtime": payload.get("runtime"),
|
||||
"ts": payload.get("ts"),
|
||||
"media": None,
|
||||
}
|
||||
@@ -259,7 +261,7 @@ class Outbox:
|
||||
# Omit absent optional fields (the app's serializer treats a
|
||||
# missing key as its default, but a JSON ``null`` for a
|
||||
# non-nullable field like ``media`` would fail to parse).
|
||||
for key in ("reasoning", "model", "tokens", "ts", "media"):
|
||||
for key in ("reasoning", "model", "tokens", "runtime", "ts", "media"):
|
||||
if m.get(key) is None:
|
||||
m.pop(key, None)
|
||||
return {
|
||||
@@ -310,12 +312,10 @@ class Outbox:
|
||||
cursors.append(int(r["cursor"]))
|
||||
if not cursors:
|
||||
return 0
|
||||
placeholders = ",".join("?" * len(cursors))
|
||||
sql = f"DELETE FROM outbox WHERE cursor IN ({placeholders})"
|
||||
# Safe: ``placeholders`` is only ``?`` markers; every cursor value is
|
||||
# bound as a parameter (no user data in the SQL text).
|
||||
# pi-lens-ignore: python-sql-injection
|
||||
self._conn.execute(sql, cursors)
|
||||
# One bound-parameter delete per cursor (a message spans only a few
|
||||
# frames); same transaction, no string-built SQL.
|
||||
for cursor in cursors:
|
||||
self._conn.execute("DELETE FROM outbox WHERE cursor = ?", (cursor,))
|
||||
self._conn.commit()
|
||||
return len(cursors)
|
||||
|
||||
|
||||
@@ -249,6 +249,54 @@ def hello_ack(
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Runtime-metadata footer (app-controlled display)
|
||||
#
|
||||
# The gateway ALWAYS attaches a structured ``runtime`` object to final
|
||||
# assistant messages so the app can render a Telegram-style footer (model,
|
||||
# context %, cwd, latency, cost). Whether/what is shown is a per-app setting,
|
||||
# NOT a hermes config — the gateway sends the data unconditionally and the
|
||||
# app decides. Mirrors hermes' ``gateway/runtime_footer.py`` fields but as
|
||||
# structured data (the app formats + picks fields).
|
||||
#
|
||||
# Recognised keys (all optional; absent when the data is unavailable):
|
||||
# model — bare model id, vendor prefix dropped (``gpt-5.4``)
|
||||
# context_pct — last-call context occupancy, 0-100 (int)
|
||||
# cwd — home-relative working dir (``~``)
|
||||
# latency — wall-clock turn duration, seconds (float)
|
||||
# cost — turn cost, USD (float); absent for local/free models
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
RUNTIME_FIELDS: tuple[str, ...] = ("model", "context_pct", "cwd", "latency", "cost")
|
||||
|
||||
|
||||
def runtime_footer(
|
||||
*,
|
||||
model: str | None = None,
|
||||
context_pct: int | None = None,
|
||||
cwd: str | None = None,
|
||||
latency: float | None = None,
|
||||
cost: float | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build the structured ``runtime`` footer object.
|
||||
|
||||
Only fields with data are included (a partially-populated footer is
|
||||
better than empty slots). Returns ``{}`` when nothing is available.
|
||||
"""
|
||||
d: dict[str, Any] = {}
|
||||
if model:
|
||||
d["model"] = model
|
||||
if context_pct is not None:
|
||||
d["context_pct"] = max(0, min(100, int(context_pct)))
|
||||
if cwd:
|
||||
d["cwd"] = cwd
|
||||
if latency is not None and latency >= 0:
|
||||
d["latency"] = round(latency, 3)
|
||||
if cost is not None and cost > 0:
|
||||
d["cost"] = round(cost, 6)
|
||||
return d
|
||||
|
||||
|
||||
# Frame builder mirrors the wire schema (docs/04); the many fields are the
|
||||
# message's full shape, so the arg count is intentional.
|
||||
def message( # noqa: PLR0913
|
||||
@@ -263,6 +311,7 @@ def message( # noqa: PLR0913
|
||||
reply_to: str | None = None,
|
||||
model: str | None = None,
|
||||
tokens: int | None = None,
|
||||
runtime: dict[str, Any] | None = None,
|
||||
ts: int | None = None,
|
||||
) -> Frame:
|
||||
payload: dict[str, Any] = {
|
||||
@@ -280,11 +329,11 @@ def message( # noqa: PLR0913
|
||||
payload["model"] = model
|
||||
if tokens is not None:
|
||||
payload["tokens"] = tokens
|
||||
if runtime:
|
||||
payload["runtime"] = runtime
|
||||
if ts is not None:
|
||||
payload["ts"] = ts
|
||||
return Frame(
|
||||
type=TYPE_MESSAGE, chat_id=chat_id, thread_id=thread_id, payload=payload
|
||||
)
|
||||
return Frame(type=TYPE_MESSAGE, chat_id=chat_id, thread_id=thread_id, payload=payload)
|
||||
|
||||
|
||||
def typing(chat_id: str, on: bool = True, *, thread_id: str | None = None) -> Frame:
|
||||
@@ -342,6 +391,7 @@ def message_stop(
|
||||
reasoning: str | None = None,
|
||||
model: str | None = None,
|
||||
tokens: int | None = None,
|
||||
runtime: dict[str, Any] | None = None,
|
||||
ts: int | None = None,
|
||||
) -> Frame:
|
||||
"""Finalize a streaming bubble."""
|
||||
@@ -355,6 +405,8 @@ def message_stop(
|
||||
payload["model"] = model
|
||||
if tokens is not None:
|
||||
payload["tokens"] = tokens
|
||||
if runtime:
|
||||
payload["runtime"] = runtime
|
||||
if ts is not None:
|
||||
payload["ts"] = ts
|
||||
return Frame(
|
||||
@@ -648,9 +700,7 @@ def notification(
|
||||
payload: dict[str, Any] = {"kind": kind, "title": title, "body": body}
|
||||
if ts is not None:
|
||||
payload["ts"] = ts
|
||||
return Frame(
|
||||
type=TYPE_NOTIFICATION, chat_id=chat_id, thread_id=thread_id, payload=payload
|
||||
)
|
||||
return Frame(type=TYPE_NOTIFICATION, chat_id=chat_id, thread_id=thread_id, payload=payload)
|
||||
|
||||
|
||||
def fcm_register(fcm_token: str | None = None, ntfy_topic: str | None = None) -> Frame:
|
||||
@@ -713,9 +763,7 @@ def media_offer(
|
||||
}
|
||||
if message_id:
|
||||
payload["message_id"] = message_id
|
||||
return Frame(
|
||||
type=TYPE_MEDIA_OFFER, chat_id=chat_id, thread_id=thread_id, payload=payload
|
||||
)
|
||||
return Frame(type=TYPE_MEDIA_OFFER, chat_id=chat_id, thread_id=thread_id, payload=payload)
|
||||
|
||||
|
||||
def media_pull_end(ok: bool, *, id: int | None = None) -> Frame:
|
||||
|
||||
Reference in new issue
Block a user