Runtime footer: app-controlled model/context/cwd/latency/cost under replies
Hermes can append a text "runtime footer" (model, context %, workdir, latency, cost) to final replies, but only when display.runtime_footer is enabled in the hermes config. We want the same info but controlled by the APP, not the gateway config. So the gateway now ALWAYS sends the data as a structured `runtime` object on final assistant messages, and the app decides whether/what to show. Gateway (gateway-plugin/): - protocol.py: new runtime_footer() helper + `runtime` field on the message / message.stop frames. Keys (all optional, absent when the data is unavailable — e.g. no cost for local models): model (vendor prefix dropped), context_pct (0-100), cwd (home-relative), latency (seconds), cost (USD). - adapter.py: a post_api_request plugin hook captures the turn's model + prompt tokens + start time (platform-filtered to android so other platforms don't pollute the buffer). _build_runtime_footer() resolves the model's context window (cached, best-effort, off the event loop via asyncio.to_thread with a timeout) and computes context_pct. The runtime object is attached on every final send (streaming message.stop and non-streaming message, plus the fallback paths). - outbox.py: `runtime` preserved in history reconstruction so the footer survives a restart / first open. App (app/shared/): - Protocol.kt: RuntimeMeta data class + `runtime` on MessagePayload / MessageStopPayload / HistoryMessage. - ChatStore.kt: `runtime` on MessageItem, wired through live + history reconciliation. - SecureStore.kt (+ Android/Desktop actuals): runtimeFooterEnabled + runtimeFooterFields (persisted per device). - IrisController.kt: StateFlows + toggleRuntimeFooter() / toggleRuntimeField(); RUNTIME_FIELD_KEYS / default set / parser. - SettingsScreen.kt: "Runtime footer" switch; when on, an expandable chip menu (Model · Context % · Workdir · Latency · Cost) to pick fields. - ChatScreen.kt: footer rendered on the SAME line as the timestamp (footer left, time right, Telegram-style), only for final non-streaming assistant answers; Inspector pane now shows the runtime fields too. Docs: 04-wire-protocol.md + frames.schema.json document the `runtime` object. Verified end-to-end on device: final replies carry `qwen3.8-27B-exl3-4.5bpw · 53% · ~ · 38s` with the time right-aligned on the same line; 69/69 gateway tests pass, Kotlin builds + tests pass.
This commit is contained in:
1 parent
678c0344c8
commit
9286937e2d
13 files changed
+979
-290
No files matched your search
@@ -249,6 +249,54 @@ def hello_ack(
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Runtime-metadata footer (app-controlled display)
|
||||
#
|
||||
# The gateway ALWAYS attaches a structured ``runtime`` object to final
|
||||
# assistant messages so the app can render a Telegram-style footer (model,
|
||||
# context %, cwd, latency, cost). Whether/what is shown is a per-app setting,
|
||||
# NOT a hermes config — the gateway sends the data unconditionally and the
|
||||
# app decides. Mirrors hermes' ``gateway/runtime_footer.py`` fields but as
|
||||
# structured data (the app formats + picks fields).
|
||||
#
|
||||
# Recognised keys (all optional; absent when the data is unavailable):
|
||||
# model — bare model id, vendor prefix dropped (``gpt-5.4``)
|
||||
# context_pct — last-call context occupancy, 0-100 (int)
|
||||
# cwd — home-relative working dir (``~``)
|
||||
# latency — wall-clock turn duration, seconds (float)
|
||||
# cost — turn cost, USD (float); absent for local/free models
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
RUNTIME_FIELDS: tuple[str, ...] = ("model", "context_pct", "cwd", "latency", "cost")
|
||||
|
||||
|
||||
def runtime_footer(
|
||||
*,
|
||||
model: str | None = None,
|
||||
context_pct: int | None = None,
|
||||
cwd: str | None = None,
|
||||
latency: float | None = None,
|
||||
cost: float | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build the structured ``runtime`` footer object.
|
||||
|
||||
Only fields with data are included (a partially-populated footer is
|
||||
better than empty slots). Returns ``{}`` when nothing is available.
|
||||
"""
|
||||
d: dict[str, Any] = {}
|
||||
if model:
|
||||
d["model"] = model
|
||||
if context_pct is not None:
|
||||
d["context_pct"] = max(0, min(100, int(context_pct)))
|
||||
if cwd:
|
||||
d["cwd"] = cwd
|
||||
if latency is not None and latency >= 0:
|
||||
d["latency"] = round(latency, 3)
|
||||
if cost is not None and cost > 0:
|
||||
d["cost"] = round(cost, 6)
|
||||
return d
|
||||
|
||||
|
||||
# Frame builder mirrors the wire schema (docs/04); the many fields are the
|
||||
# message's full shape, so the arg count is intentional.
|
||||
def message( # noqa: PLR0913
|
||||
@@ -263,6 +311,7 @@ def message( # noqa: PLR0913
|
||||
reply_to: str | None = None,
|
||||
model: str | None = None,
|
||||
tokens: int | None = None,
|
||||
runtime: dict[str, Any] | None = None,
|
||||
ts: int | None = None,
|
||||
) -> Frame:
|
||||
payload: dict[str, Any] = {
|
||||
@@ -280,11 +329,11 @@ def message( # noqa: PLR0913
|
||||
payload["model"] = model
|
||||
if tokens is not None:
|
||||
payload["tokens"] = tokens
|
||||
if runtime:
|
||||
payload["runtime"] = runtime
|
||||
if ts is not None:
|
||||
payload["ts"] = ts
|
||||
return Frame(
|
||||
type=TYPE_MESSAGE, chat_id=chat_id, thread_id=thread_id, payload=payload
|
||||
)
|
||||
return Frame(type=TYPE_MESSAGE, chat_id=chat_id, thread_id=thread_id, payload=payload)
|
||||
|
||||
|
||||
def typing(chat_id: str, on: bool = True, *, thread_id: str | None = None) -> Frame:
|
||||
@@ -342,6 +391,7 @@ def message_stop(
|
||||
reasoning: str | None = None,
|
||||
model: str | None = None,
|
||||
tokens: int | None = None,
|
||||
runtime: dict[str, Any] | None = None,
|
||||
ts: int | None = None,
|
||||
) -> Frame:
|
||||
"""Finalize a streaming bubble."""
|
||||
@@ -355,6 +405,8 @@ def message_stop(
|
||||
payload["model"] = model
|
||||
if tokens is not None:
|
||||
payload["tokens"] = tokens
|
||||
if runtime:
|
||||
payload["runtime"] = runtime
|
||||
if ts is not None:
|
||||
payload["ts"] = ts
|
||||
return Frame(
|
||||
@@ -648,9 +700,7 @@ def notification(
|
||||
payload: dict[str, Any] = {"kind": kind, "title": title, "body": body}
|
||||
if ts is not None:
|
||||
payload["ts"] = ts
|
||||
return Frame(
|
||||
type=TYPE_NOTIFICATION, chat_id=chat_id, thread_id=thread_id, payload=payload
|
||||
)
|
||||
return Frame(type=TYPE_NOTIFICATION, chat_id=chat_id, thread_id=thread_id, payload=payload)
|
||||
|
||||
|
||||
def fcm_register(fcm_token: str | None = None, ntfy_topic: str | None = None) -> Frame:
|
||||
@@ -713,9 +763,7 @@ def media_offer(
|
||||
}
|
||||
if message_id:
|
||||
payload["message_id"] = message_id
|
||||
return Frame(
|
||||
type=TYPE_MEDIA_OFFER, chat_id=chat_id, thread_id=thread_id, payload=payload
|
||||
)
|
||||
return Frame(type=TYPE_MEDIA_OFFER, chat_id=chat_id, thread_id=thread_id, payload=payload)
|
||||
|
||||
|
||||
def media_pull_end(ok: bool, *, id: int | None = None) -> Frame:
|
||||
|
||||
Reference in new issue
Block a user