Runtime footer: app-controlled model/context/cwd/latency/cost under replies
Hermes can append a text "runtime footer" (model, context %, workdir, latency, cost) to final replies, but only when display.runtime_footer is enabled in the hermes config. We want the same info but controlled by the APP, not the gateway config. So the gateway now ALWAYS sends the data as a structured `runtime` object on final assistant messages, and the app decides whether/what to show. Gateway (gateway-plugin/): - protocol.py: new runtime_footer() helper + `runtime` field on the message / message.stop frames. Keys (all optional, absent when the data is unavailable — e.g. no cost for local models): model (vendor prefix dropped), context_pct (0-100), cwd (home-relative), latency (seconds), cost (USD). - adapter.py: a post_api_request plugin hook captures the turn's model + prompt tokens + start time (platform-filtered to android so other platforms don't pollute the buffer). _build_runtime_footer() resolves the model's context window (cached, best-effort, off the event loop via asyncio.to_thread with a timeout) and computes context_pct. The runtime object is attached on every final send (streaming message.stop and non-streaming message, plus the fallback paths). - outbox.py: `runtime` preserved in history reconstruction so the footer survives a restart / first open. App (app/shared/): - Protocol.kt: RuntimeMeta data class + `runtime` on MessagePayload / MessageStopPayload / HistoryMessage. - ChatStore.kt: `runtime` on MessageItem, wired through live + history reconciliation. - SecureStore.kt (+ Android/Desktop actuals): runtimeFooterEnabled + runtimeFooterFields (persisted per device). - IrisController.kt: StateFlows + toggleRuntimeFooter() / toggleRuntimeField(); RUNTIME_FIELD_KEYS / default set / parser. - SettingsScreen.kt: "Runtime footer" switch; when on, an expandable chip menu (Model · Context % · Workdir · Latency · Cost) to pick fields. - ChatScreen.kt: footer rendered on the SAME line as the timestamp (footer left, time right, Telegram-style), only for final non-streaming assistant answers; Inspector pane now shows the runtime fields too. Docs: 04-wire-protocol.md + frames.schema.json document the `runtime` object. Verified end-to-end on device: final replies carry `qwen3.8-27B-exl3-4.5bpw · 53% · ~ · 38s` with the time right-aligned on the same line; 69/69 gateway tests pass, Kotlin builds + tests pass.
This commit is contained in:
1 parent
678c0344c8
commit
9286937e2d
13 files changed
+979
-290
No files matched your search
+150
-3
@@ -317,6 +317,132 @@ def _tool_end_fields(tool_name: str) -> dict[str, Any]:
|
||||
return fields
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Runtime-metadata footer (post_api_request hook)
|
||||
#
|
||||
# The app renders a Telegram-style footer under final assistant messages
|
||||
# (model, context %, cwd, latency, cost). Display is controlled by the APP
|
||||
# (Settings → Runtime footer), not hermes config — so the gateway ALWAYS
|
||||
# sends the data. hermes core only appends its own *text* footer when
|
||||
# ``display.runtime_footer.enabled`` is set, and the adapter has no access to
|
||||
# the gateway's ``agent_result``, so we capture the same facts ourselves via
|
||||
# the ``post_api_request`` plugin hook (fires after every provider call with
|
||||
# model + usage):
|
||||
#
|
||||
# * model — the turn's latest model (failover-aware)
|
||||
# * prompt_tokens — the latest call's prompt size (context occupancy)
|
||||
# * turn start — the first API call of the turn (latency baseline)
|
||||
#
|
||||
# Global buffer (same pattern as the reasoning/tool buffers): a personal
|
||||
# android gateway serves one active turn at a time. The hook fires for every
|
||||
# platform, so we only record when the turn's platform is android.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_runtime_meta: dict[str, Any] = {}
|
||||
_runtime_meta_lock = threading.Lock()
|
||||
# Per-model context-window cache. Resolution may probe endpoints on first use
|
||||
# (slow); the cache is process-lifetime so each model resolves at most once.
|
||||
_context_length_cache: dict[str, int] = {}
|
||||
# Upper bound (seconds) on context-window resolution during a final send, so a
|
||||
# slow first-use probe never delays the reply. The worker thread keeps running
|
||||
# and populates the cache, so the next turn is fast.
|
||||
_CTX_RESOLVE_TIMEOUT_S = 3.0
|
||||
|
||||
|
||||
def _on_post_api_request(**kwargs: Any) -> None:
|
||||
"""Plugin hook: capture per-turn runtime metadata (model, prompt tokens)."""
|
||||
platform = kwargs.get("platform")
|
||||
if platform and platform != "android":
|
||||
return
|
||||
model = kwargs.get("model") or ""
|
||||
usage = kwargs.get("usage") or {}
|
||||
prompt_tokens = usage.get("prompt_tokens") or 0
|
||||
with _runtime_meta_lock:
|
||||
if model:
|
||||
_runtime_meta["model"] = model
|
||||
if prompt_tokens:
|
||||
_runtime_meta["prompt_tokens"] = prompt_tokens
|
||||
if "turn_start" not in _runtime_meta:
|
||||
_runtime_meta["turn_start"] = time.monotonic()
|
||||
|
||||
|
||||
def _take_runtime_meta() -> dict[str, Any]:
|
||||
"""Drain the captured turn metadata (turn boundary)."""
|
||||
with _runtime_meta_lock:
|
||||
meta = dict(_runtime_meta)
|
||||
_runtime_meta.clear()
|
||||
return meta
|
||||
|
||||
|
||||
def _resolve_context_length(model: str) -> int | None:
|
||||
"""Best-effort context window for *model* (cached; None on failure).
|
||||
|
||||
Runs in a worker thread (may probe endpoints on first use). The cache is
|
||||
populated even if the caller's asyncio task times out, so subsequent
|
||||
turns resolve instantly.
|
||||
"""
|
||||
if not model:
|
||||
return None
|
||||
cached = _context_length_cache.get(model)
|
||||
if cached:
|
||||
return cached
|
||||
try:
|
||||
from agent.model_metadata import get_model_context_length
|
||||
|
||||
ctx = get_model_context_length(model)
|
||||
if ctx and ctx > 0:
|
||||
_context_length_cache[model] = int(ctx)
|
||||
return int(ctx)
|
||||
except Exception:
|
||||
logger.debug("android: context-length resolution failed for %s", model, exc_info=True)
|
||||
return None
|
||||
|
||||
|
||||
def _home_relative_cwd(cwd: str) -> str:
|
||||
"""Collapse ``$HOME`` to ``~`` (matches hermes' runtime footer)."""
|
||||
if not cwd:
|
||||
return ""
|
||||
try:
|
||||
home = os.path.expanduser("~")
|
||||
p = os.path.abspath(cwd)
|
||||
if home and (p == home or p.startswith(home + os.sep)):
|
||||
return "~" + p[len(home) :]
|
||||
return p
|
||||
except Exception:
|
||||
return cwd
|
||||
|
||||
|
||||
async def _build_runtime_footer(meta: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Build the ``runtime`` footer object from captured turn metadata.
|
||||
|
||||
Called on every final send (the app decides what to show). Fields without
|
||||
data are omitted. ``meta`` is the drained turn buffer (model,
|
||||
prompt_tokens, turn_start).
|
||||
"""
|
||||
model = (meta.get("model") or "").rsplit("/", 1)[-1]
|
||||
prompt_tokens = meta.get("prompt_tokens") or 0
|
||||
context_pct = None
|
||||
if prompt_tokens and model:
|
||||
try:
|
||||
ctx_len = await asyncio.wait_for(
|
||||
asyncio.to_thread(_resolve_context_length, model),
|
||||
timeout=_CTX_RESOLVE_TIMEOUT_S,
|
||||
)
|
||||
except (asyncio.TimeoutError, Exception):
|
||||
ctx_len = None
|
||||
if ctx_len:
|
||||
context_pct = round(prompt_tokens / ctx_len * 100)
|
||||
turn_start = meta.get("turn_start")
|
||||
latency = (time.monotonic() - turn_start) if turn_start else None
|
||||
cwd = _home_relative_cwd(os.environ.get("TERMINAL_CWD", ""))
|
||||
return protocol.runtime_footer(
|
||||
model=model or None,
|
||||
context_pct=context_pct,
|
||||
cwd=cwd or None,
|
||||
latency=latency,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Defaults
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1159,7 +1285,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
state = self._turn_state(chat_id)
|
||||
|
||||
# 1. Streaming segment start (stream consumer first send).
|
||||
if meta.get("expect_edits") is True:
|
||||
if bool(meta.get("expect_edits")):
|
||||
# A new content segment means the tool the model was waiting on
|
||||
# has returned -> close it before the segment opens.
|
||||
await self._close_open_tool(chat_id, state, thread_id)
|
||||
@@ -1213,7 +1339,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
return SendResult(success=True, message_id=message_id)
|
||||
|
||||
# 3. Final message (non-streaming final, or streaming fallback final).
|
||||
if meta.get("notify") is True:
|
||||
if bool(meta.get("notify")):
|
||||
reasoning, body = _split_reasoning(content)
|
||||
# Non-streaming: reasoning is prepended to content (split above).
|
||||
# Streaming fallback: content has no reasoning, so use the
|
||||
@@ -1224,6 +1350,11 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
reasoning = _take_reasoning() or None
|
||||
else:
|
||||
_reset_reasoning()
|
||||
# Runtime-metadata footer (app-controlled display): always attach
|
||||
# the structured ``runtime`` object so the app can render its
|
||||
# footer (Settings → Runtime footer). Independent of hermes'
|
||||
# ``display.runtime_footer`` config.
|
||||
runtime = await _build_runtime_footer(_take_runtime_meta())
|
||||
if state.stream_id:
|
||||
# Fallback final: close the open streaming segment in place.
|
||||
message_id = state.stream_id
|
||||
@@ -1236,6 +1367,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
body,
|
||||
reasoning=reasoning,
|
||||
thread_id=thread_id,
|
||||
runtime=runtime or None,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
@@ -1251,6 +1383,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
thread_id=thread_id,
|
||||
reasoning=reasoning,
|
||||
reply_to=reply_to,
|
||||
runtime=runtime or None,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
@@ -1310,6 +1443,8 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
reasoning = _take_reasoning() or None
|
||||
else:
|
||||
_reset_reasoning()
|
||||
# Runtime-metadata footer (app-controlled display).
|
||||
runtime = await _build_runtime_footer(_take_runtime_meta())
|
||||
state.stream_id = None
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
@@ -1319,6 +1454,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
body,
|
||||
reasoning=reasoning,
|
||||
thread_id=thread_id,
|
||||
runtime=runtime or None,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
@@ -1344,6 +1480,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
|
||||
# Unknown id: treat as a streaming update (best effort).
|
||||
if finalize:
|
||||
runtime = await _build_runtime_footer(_take_runtime_meta())
|
||||
await self._broadcast_or_log(
|
||||
chat_id,
|
||||
protocol.message_stop(
|
||||
@@ -1351,6 +1488,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
message_id,
|
||||
_strip_streaming_cursor(content),
|
||||
thread_id=thread_id,
|
||||
runtime=runtime or None,
|
||||
ts=int(time.time() * 1000),
|
||||
),
|
||||
)
|
||||
@@ -1857,7 +1995,7 @@ class AndroidAdapter(BasePlatformAdapter):
|
||||
# Skipped for slash commands (session-scoped, not conversation
|
||||
# starters) and replies (they continue where the user is). Threading
|
||||
# is only active on the default channel; other channels stay flat.
|
||||
auto_thread = payload.get("auto_thread") is True
|
||||
auto_thread = bool(payload.get("auto_thread"))
|
||||
default_entry = self._channels.default()
|
||||
if (
|
||||
auto_thread
|
||||
@@ -2825,6 +2963,15 @@ def register(ctx):
|
||||
ctx.register_hook("post_tool_call", _on_post_tool_call)
|
||||
except Exception:
|
||||
logger.debug("android: post_tool_call hook registration failed", exc_info=True)
|
||||
# Runtime-metadata footer: capture the turn's model + prompt tokens (per
|
||||
# provider call) so the final message can carry a structured ``runtime``
|
||||
# object. The app decides whether/what to show (Settings → Runtime
|
||||
# footer); the gateway always sends the data (independent of hermes
|
||||
# ``display.runtime_footer`` config).
|
||||
try:
|
||||
ctx.register_hook("post_api_request", _on_post_api_request)
|
||||
except Exception:
|
||||
logger.debug("android: post_api_request hook registration failed", exc_info=True)
|
||||
ctx.register_platform(
|
||||
name="android",
|
||||
label="Android",
|
||||
|
||||
Reference in new issue
Block a user