From 9286937e2d65eeca7a54fdbb38cc97446946ad1d Mon Sep 17 00:00:00 2001 From: ARIA Date: Fri, 21 Aug 2026 19:32:05 +0200 Subject: [PATCH] Runtime footer: app-controlled model/context/cwd/latency/cost under replies MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Hermes can append a text "runtime footer" (model, context %, workdir, latency, cost) to final replies, but only when display.runtime_footer is enabled in the hermes config. We want the same info but controlled by the APP, not the gateway config. So the gateway now ALWAYS sends the data as a structured `runtime` object on final assistant messages, and the app decides whether/what to show. Gateway (gateway-plugin/): - protocol.py: new runtime_footer() helper + `runtime` field on the message / message.stop frames. Keys (all optional, absent when the data is unavailable — e.g. no cost for local models): model (vendor prefix dropped), context_pct (0-100), cwd (home-relative), latency (seconds), cost (USD). - adapter.py: a post_api_request plugin hook captures the turn's model + prompt tokens + start time (platform-filtered to android so other platforms don't pollute the buffer). _build_runtime_footer() resolves the model's context window (cached, best-effort, off the event loop via asyncio.to_thread with a timeout) and computes context_pct. The runtime object is attached on every final send (streaming message.stop and non-streaming message, plus the fallback paths). - outbox.py: `runtime` preserved in history reconstruction so the footer survives a restart / first open. App (app/shared/): - Protocol.kt: RuntimeMeta data class + `runtime` on MessagePayload / MessageStopPayload / HistoryMessage. - ChatStore.kt: `runtime` on MessageItem, wired through live + history reconciliation. - SecureStore.kt (+ Android/Desktop actuals): runtimeFooterEnabled + runtimeFooterFields (persisted per device). - IrisController.kt: StateFlows + toggleRuntimeFooter() / toggleRuntimeField(); RUNTIME_FIELD_KEYS / default set / parser. - SettingsScreen.kt: "Runtime footer" switch; when on, an expandable chip menu (Model · Context % · Workdir · Latency · Cost) to pick fields. - ChatScreen.kt: footer rendered on the SAME line as the timestamp (footer left, time right, Telegram-style), only for final non-streaming assistant answers; Inspector pane now shows the runtime fields too. Docs: 04-wire-protocol.md + frames.schema.json document the `runtime` object. Verified end-to-end on device: final replies carry `qwen3.8-27B-exl3-4.5bpw · 53% · ~ · 38s` with the time right-aligned on the same line; 69/69 gateway tests pass, Kotlin builds + tests pass. --- .../iris/platform/AndroidSecureStore.kt | 10 + .../commonMain/kotlin/iris/data/ChatStore.kt | 404 ++++++++++++------ .../kotlin/iris/data/SecureStore.kt | 9 + .../kotlin/iris/protocol/Protocol.kt | 298 ++++++++----- .../kotlin/iris/state/IrisController.kt | 51 ++- .../kotlin/iris/ui/screens/ChatScreen.kt | 89 +++- .../kotlin/iris/ui/screens/SettingsScreen.kt | 46 ++ .../iris/platform/DesktopSecureStore.kt | 16 + docs/04-wire-protocol.md | 105 ++++- docs/protocol/frames.schema.json | 8 +- gateway-plugin/adapter.py | 153 ++++++- gateway-plugin/outbox.py | 14 +- gateway-plugin/protocol.py | 66 ++- 13 files changed, 979 insertions(+), 290 deletions(-) diff --git a/app/shared/src/androidMain/kotlin/iris/platform/AndroidSecureStore.kt b/app/shared/src/androidMain/kotlin/iris/platform/AndroidSecureStore.kt index 9d75102..f565c03 100644 --- a/app/shared/src/androidMain/kotlin/iris/platform/AndroidSecureStore.kt +++ b/app/shared/src/androidMain/kotlin/iris/platform/AndroidSecureStore.kt @@ -162,6 +162,14 @@ class AndroidSecureStore( get() = prefs.getFloat(KEY_FONT_SIZE_SCALE, 1.0f) set(value) = prefs.edit().putFloat(KEY_FONT_SIZE_SCALE, value).apply() + override var runtimeFooterEnabled: Boolean + get() = prefs.getBoolean(KEY_RUNTIME_FOOTER_ENABLED, false) + set(value) = prefs.edit().putBoolean(KEY_RUNTIME_FOOTER_ENABLED, value).apply() + + override var runtimeFooterFields: String + get() = prefs.getString(KEY_RUNTIME_FOOTER_FIELDS, "").orEmpty() + set(value) = prefs.edit().putString(KEY_RUNTIME_FOOTER_FIELDS, value).apply() + override fun savePairing( url: String, token: String, @@ -199,5 +207,7 @@ class AndroidSecureStore( const val KEY_BACKGROUND_COLOR = "background_color" const val KEY_BACKGROUND_IMAGE_PATH = "background_image_path" const val KEY_FONT_SIZE_SCALE = "font_size_scale" + const val KEY_RUNTIME_FOOTER_ENABLED = "runtime_footer_enabled" + const val KEY_RUNTIME_FOOTER_FIELDS = "runtime_footer_fields" } } diff --git a/app/shared/src/commonMain/kotlin/iris/data/ChatStore.kt b/app/shared/src/commonMain/kotlin/iris/data/ChatStore.kt index ca91229..1d801d9 100644 --- a/app/shared/src/commonMain/kotlin/iris/data/ChatStore.kt +++ b/app/shared/src/commonMain/kotlin/iris/data/ChatStore.kt @@ -10,6 +10,7 @@ import iris.protocol.MessageStopPayload import iris.protocol.MessageUpdatePayload import iris.protocol.ROLE_ASSISTANT import iris.protocol.ROLE_USER +import iris.protocol.RuntimeMeta import iris.protocol.TYPE_COMMENTARY import iris.protocol.TYPE_MEDIA_OFFER import iris.protocol.TYPE_MESSAGE @@ -45,10 +46,10 @@ sealed interface ChatItem { /** Delivery status of a user message (M5: read.receipt; M7: failed sends). */ enum class MsgStatus { - Pending, // optimistic, not yet acknowledged by the gateway - Sent, // gateway accepted it (echo received) - Read, // agent received and started processing it (read.receipt) - Failed, // send failed (error frame); tap the bubble to retry + Pending, // optimistic, not yet acknowledged by the gateway + Sent, // gateway accepted it (echo received) + Read, // agent received and started processing it (read.receipt) + Failed, // send failed (error frame); tap the bubble to retry } /** A chat message (user / assistant / commentary / streaming bubble). @@ -66,6 +67,7 @@ data class MessageItem( val streaming: Boolean = false, val model: String? = null, val tokens: Int? = null, + val runtime: RuntimeMeta? = null, val media: List = emptyList(), val isSystem: Boolean = false, ) : ChatItem @@ -119,16 +121,17 @@ class ChatStore { companion object { const val DEFAULT_LANE = "android:default" - fun randomId(prefix: String): String = - "${prefix}${Random.nextLong(1_000_000_000L, 9_999_999_999L)}" + fun randomId(prefix: String): String = "${prefix}${Random.nextLong(1_000_000_000L, 9_999_999_999L)}" } // ── Lane helpers ────────────────────────────────────────────────────── /** Lane key for a (chat, thread) pair. Uses `::` as the separator because * chat ids already contain a single `:` (e.g. `android:chan_1`). */ - fun laneKey(chatId: String, threadId: String?): String = - if (threadId.isNullOrBlank()) chatId else "$chatId::$threadId" + fun laneKey( + chatId: String, + threadId: String?, + ): String = if (threadId.isNullOrBlank()) chatId else "$chatId::$threadId" /** Split a lane key back into (chatId, threadId). */ fun parseLane(key: String): Pair { @@ -145,7 +148,10 @@ class ChatStore { return laneKey(chatId, frame.threadId) } - private fun updateLane(lane: String, transform: (List) -> List) { + private fun updateLane( + lane: String, + transform: (List) -> List, + ) { val map = _lanes.value.toMutableMap() map[lane] = transform(map[lane].orEmpty()) _lanes.value = map @@ -154,7 +160,11 @@ class ChatStore { // ── Optimistic send ─────────────────────────────────────────────────── /** Optimistic add: show the user's message immediately (pending) in [lane]. */ - fun addPending(text: String, lane: String, media: List = emptyList()): String { + fun addPending( + text: String, + lane: String, + media: List = emptyList(), + ): String { localSeq++ val id = "local_$localSeq" updateLane(lane) { @@ -166,7 +176,10 @@ class ChatStore { /** Append a locally generated, centered system notice (gateway restart / * online) to [lane]. Not a user or agent turn: it is not selectable, * deletable, or echoed to the server. */ - fun addSystemMessage(lane: String, text: String) { + fun addSystemMessage( + lane: String, + text: String, + ) { localSeq++ val id = "sys_$localSeq" updateLane(lane) { @@ -195,7 +208,10 @@ class ChatStore { // ── message (final / standalone, incl. non-streaming + user echo) ───── - private fun onMessage(lane: String, frame: Frame) { + private fun onMessage( + lane: String, + frame: Frame, + ) { val p = frame.payloadAs() ?: return // Auto-threaded send: the gateway created a fresh thread for this // message; the optimistic bubble is still in the parent flat lane. @@ -205,99 +221,143 @@ class ChatStore { val byId = list.indexOfFirst { it.id == p.messageId } if (byId >= 0) { val cur = list[byId] as? MessageItem ?: return@updateLane list - val updated = cur.copy( - text = p.text, - reasoning = p.reasoning, - pending = false, - streaming = false, - status = if (cur.status == MsgStatus.Read) MsgStatus.Read else MsgStatus.Sent, - model = p.model, - tokens = p.tokens, - ts = p.ts ?: cur.ts, - media = mergeMedia(cur.media, p.media), - ) + val updated = + cur.copy( + text = p.text, + reasoning = p.reasoning, + pending = false, + streaming = false, + status = if (cur.status == MsgStatus.Read) MsgStatus.Read else MsgStatus.Sent, + model = p.model, + tokens = p.tokens, + runtime = p.runtime ?: cur.runtime, + ts = p.ts ?: cur.ts, + media = mergeMedia(cur.media, p.media), + ) list.toMutableList().also { it[byId] = updated } } else if (p.role == ROLE_USER) { // Replace the matching optimistic pending bubble (server echo). - val pendingIdx = list.indexOfLast { - it is MessageItem && it.pending && it.role == ROLE_USER && it.text == p.text - } + val pendingIdx = + list.indexOfLast { + it is MessageItem && it.pending && it.role == ROLE_USER && it.text == p.text + } if (pendingIdx >= 0) { list.toMutableList().also { - it[pendingIdx] = MessageItem( - id = p.messageId, role = p.role, text = p.text, - ts = p.ts ?: 0, reasoning = p.reasoning, - media = mergeMedia((list[pendingIdx] as MessageItem).media, p.media), - ) + it[pendingIdx] = + MessageItem( + id = p.messageId, + role = p.role, + text = p.text, + ts = p.ts ?: 0, + reasoning = p.reasoning, + media = mergeMedia((list[pendingIdx] as MessageItem).media, p.media), + ) } } else { - list + MessageItem( - id = p.messageId, role = p.role, text = p.text, ts = p.ts ?: 0, - reasoning = p.reasoning, model = p.model, tokens = p.tokens, - media = p.media.map { it.toMediaItem() }, - ) + list + + MessageItem( + id = p.messageId, + role = p.role, + text = p.text, + ts = p.ts ?: 0, + reasoning = p.reasoning, + model = p.model, + tokens = p.tokens, + runtime = p.runtime, + media = p.media.map { it.toMediaItem() }, + ) } } else { - list + MessageItem( - id = p.messageId, role = p.role, text = p.text, ts = p.ts ?: 0, - reasoning = p.reasoning, model = p.model, tokens = p.tokens, - media = p.media.map { it.toMediaItem() }, - ) + list + + MessageItem( + id = p.messageId, + role = p.role, + text = p.text, + ts = p.ts ?: 0, + reasoning = p.reasoning, + model = p.model, + tokens = p.tokens, + runtime = p.runtime, + media = p.media.map { it.toMediaItem() }, + ) } } } /** - * Remove the matching optimistic user bubble from the parent flat lane of - * [lane] (an auto-created thread). The gateway moved the message into the - * fresh thread, so the pending bubble must follow it; the echo then replaces - * it in the thread lane. No-op when [lane] is not a thread lane or no - * pending bubble with the same text exists in the flat lane. - */ - private fun relocatePendingToThreadLane(lane: String, p: MessagePayload) { + * Remove the matching optimistic user bubble from the parent flat lane of + * [lane] (an auto-created thread). The gateway moved the message into the + * fresh thread, so the pending bubble must follow it; the echo then replaces + * it in the thread lane. No-op when [lane] is not a thread lane or no + * pending bubble with the same text exists in the flat lane. + */ + private fun relocatePendingToThreadLane( + lane: String, + p: MessagePayload, + ) { val (chatId, threadId) = parseLane(lane) if (threadId == null) return val flatLane = chatId val map = _lanes.value.toMutableMap() val flatList = map[flatLane].orEmpty() - val idx = flatList.indexOfLast { - it is MessageItem && it.pending && it.role == ROLE_USER && it.text == p.text - } + val idx = + flatList.indexOfLast { + it is MessageItem && it.pending && it.role == ROLE_USER && it.text == p.text + } if (idx < 0) return map[flatLane] = flatList.toMutableList().also { it.removeAt(idx) } _lanes.value = map } /** Merge server media refs into existing items, keeping local paths. */ - private fun mergeMedia(existing: List, incoming: List): List { + private fun mergeMedia( + existing: List, + incoming: List, + ): List { if (incoming.isEmpty()) return existing val byId = existing.associateBy { it.mediaId } return incoming.map { ref -> byId[ref.mediaId]?.copy( - kind = ref.kind, mime = ref.mime, size = ref.size, filename = ref.filename, + kind = ref.kind, + mime = ref.mime, + size = ref.size, + filename = ref.filename, ) ?: ref.toMediaItem() } } - private fun MediaRef.toMediaItem() = - MediaItem(mediaId = mediaId, kind = kind, mime = mime, size = size, filename = filename) + private fun MediaRef.toMediaItem() = MediaItem(mediaId = mediaId, kind = kind, mime = mime, size = size, filename = filename) // ── message.start (open a live streaming bubble) ────────────────────── - private fun onMessageStart(lane: String, frame: Frame) { + private fun onMessageStart( + lane: String, + frame: Frame, + ) { if (!streamingEnabled) return val p = frame.payloadAs() ?: return updateLane(lane) { list -> - if (list.any { it.id == p.messageId }) list - else list + MessageItem( - id = p.messageId, role = p.role, text = "", ts = 0, streaming = true, - ) + if (list.any { it.id == p.messageId }) { + list + } else { + list + + MessageItem( + id = p.messageId, + role = p.role, + text = "", + ts = 0, + streaming = true, + ) + } } } // ── message.update (replace live bubble text; full snapshot) ────────── - private fun onMessageUpdate(lane: String, frame: Frame) { + private fun onMessageUpdate( + lane: String, + frame: Frame, + ) { val p = frame.payloadAs() ?: return updateLane(lane) { list -> val idx = list.indexOfFirst { it.id == p.messageId } @@ -309,28 +369,39 @@ class ChatStore { // ── message.stop (finalize the live bubble) ─────────────────────────── - private fun onMessageStop(lane: String, frame: Frame) { + private fun onMessageStop( + lane: String, + frame: Frame, + ) { val p = frame.payloadAs() ?: return updateLane(lane) { list -> val idx = list.indexOfFirst { it.id == p.messageId } if (idx < 0) { // No live bubble (e.g. missed start) — materialize a final one. - list + MessageItem( - id = p.messageId, role = ROLE_ASSISTANT, text = p.finalText, - ts = p.ts ?: 0, reasoning = p.reasoning, - model = p.model, tokens = p.tokens, - ) - } else { - val cur = list[idx] as? MessageItem ?: return@updateLane list - list.toMutableList().also { - it[idx] = cur.copy( + list + + MessageItem( + id = p.messageId, + role = ROLE_ASSISTANT, text = p.finalText, + ts = p.ts ?: 0, reasoning = p.reasoning, model = p.model, tokens = p.tokens, - streaming = false, - ts = p.ts ?: cur.ts, + runtime = p.runtime, ) + } else { + val cur = list[idx] as? MessageItem ?: return@updateLane list + list.toMutableList().also { + it[idx] = + cur.copy( + text = p.finalText, + reasoning = p.reasoning, + model = p.model, + tokens = p.tokens, + runtime = p.runtime ?: cur.runtime, + streaming = false, + ts = p.ts ?: cur.ts, + ) } } } @@ -338,21 +409,31 @@ class ChatStore { // ── tool.start (new tool card) ──────────────────────────────────────── - private fun onToolStart(lane: String, frame: Frame) { + private fun onToolStart( + lane: String, + frame: Frame, + ) { val p = frame.payloadAs() ?: return toolSeq++ val id = "tool_$toolSeq" updateLane(lane) { list -> - list + ToolItem( - id = id, index = p.index, name = p.name, - preview = p.preview, args = p.args, - ) + list + + ToolItem( + id = id, + index = p.index, + name = p.name, + preview = p.preview, + args = p.args, + ) } } // ── tool.progress (in-progress note) ────────────────────────────────── - private fun onToolProgress(lane: String, frame: Frame) { + private fun onToolProgress( + lane: String, + frame: Frame, + ) { val p = frame.payloadAs() ?: return updateLane(lane) { list -> val idx = list.indexOfLast { it is ToolItem && !it.done && it.index == p.index } @@ -364,24 +445,33 @@ class ChatStore { // ── tool.end (mark the tool card complete) ──────────────────────────── - private fun onToolEnd(lane: String, frame: Frame) { + private fun onToolEnd( + lane: String, + frame: Frame, + ) { val p = frame.payloadAs() ?: return updateLane(lane) { list -> val idx = list.indexOfLast { it is ToolItem && !it.done && it.index == p.index } if (idx < 0) return@updateLane list val cur = list[idx] as ToolItem list.toMutableList().also { - it[idx] = cur.copy( - done = true, ok = p.ok, duration = p.duration, - outputPreview = p.outputPreview, - ) + it[idx] = + cur.copy( + done = true, + ok = p.ok, + duration = p.duration, + outputPreview = p.outputPreview, + ) } } } // ── commentary (dimmed interim beat) ────────────────────────────────── - private fun onCommentary(lane: String, frame: Frame) { + private fun onCommentary( + lane: String, + frame: Frame, + ) { val p = frame.payloadAs() ?: return // Gateway lifecycle notices (restart / shutdown / online) are rendered // as a centered system notice by the controller, on the down/up state @@ -392,11 +482,18 @@ class ChatStore { // and prevents duplicates. if (isGatewayLifecycleNotice(p.text)) return updateLane(lane) { list -> - if (list.any { it.id == p.messageId }) list - else list + MessageItem( - id = p.messageId, role = ROLE_ASSISTANT, text = p.text, - ts = 0, isCommentary = true, - ) + if (list.any { it.id == p.messageId }) { + list + } else { + list + + MessageItem( + id = p.messageId, + role = ROLE_ASSISTANT, + text = p.text, + ts = 0, + isCommentary = true, + ) + } } } @@ -415,20 +512,28 @@ class ChatStore { * falling back to the lane's last assistant message). The item has no local * path yet — the controller pulls it and calls [setMediaLocalPath]. */ - private fun onMediaOffer(lane: String, frame: Frame) { + private fun onMediaOffer( + lane: String, + frame: Frame, + ) { val p = frame.payloadAs() ?: return updateLane(lane) { list -> // Already attached (offer replayed)? Skip. if (list.any { it is MessageItem && it.media.any { m -> m.mediaId == p.mediaId } }) return@updateLane list - val item = MediaItem( - mediaId = p.mediaId, kind = p.kind, mime = p.mime, - size = p.size, filename = p.filename, - ) - val targetIdx = if (p.messageId != null) { - list.indexOfFirst { it.id == p.messageId } - } else { - list.indexOfLast { it is MessageItem && it.role == ROLE_ASSISTANT } - } + val item = + MediaItem( + mediaId = p.mediaId, + kind = p.kind, + mime = p.mime, + size = p.size, + filename = p.filename, + ) + val targetIdx = + if (p.messageId != null) { + list.indexOfFirst { it.id == p.messageId } + } else { + list.indexOfLast { it is MessageItem && it.role == ROLE_ASSISTANT } + } if (targetIdx < 0) return@updateLane list val cur = list[targetIdx] as? MessageItem ?: return@updateLane list list.toMutableList().also { it[targetIdx] = cur.copy(media = cur.media + item) } @@ -436,17 +541,26 @@ class ChatStore { } /** Record a pulled file's local path for [mediaId] (all lanes). */ - fun setMediaLocalPath(mediaId: String, localPath: String) { + fun setMediaLocalPath( + mediaId: String, + localPath: String, + ) { val map = _lanes.value.toMutableMap() var changed = false for ((lane, list) in map) { - val updated = list.map { item -> - if (item is MessageItem) { - item.copy(media = item.media.map { m -> - if (m.mediaId == mediaId) m.copy(localPath = localPath) else m - }) - } else item - } + val updated = + list.map { item -> + if (item is MessageItem) { + item.copy( + media = + item.media.map { m -> + if (m.mediaId == mediaId) m.copy(localPath = localPath) else m + }, + ) + } else { + item + } + } if (updated != list) { map[lane] = updated changed = true @@ -460,13 +574,16 @@ class ChatStore { val map = _lanes.value.toMutableMap() var changed = false for ((lane, list) in map) { - val updated = list.map { item -> - if (item is MessageItem && item.id == messageId && item.role == ROLE_USER && - item.status != MsgStatus.Read - ) { - item.copy(pending = false, status = MsgStatus.Read) - } else item - } + val updated = + list.map { item -> + if (item is MessageItem && item.id == messageId && item.role == ROLE_USER && + item.status != MsgStatus.Read + ) { + item.copy(pending = false, status = MsgStatus.Read) + } else { + item + } + } if (updated != list) { map[lane] = updated changed = true @@ -476,11 +593,11 @@ class ChatStore { } /** - * Remove messages by id from every lane (a `message.deleted` frame). The - * server is authoritative: the frame carries no lane, and a message id is - * unique, so scanning all lanes is both safe and idempotent (a no-op when the - * id is absent, e.g. a local-only id another device never had). - */ + * Remove messages by id from every lane (a `message.deleted` frame). The + * server is authoritative: the frame carries no lane, and a message id is + * unique, so scanning all lanes is both safe and idempotent (a no-op when the + * id is absent, e.g. a local-only id another device never had). + */ fun removeMessages(messageIds: Set) { if (messageIds.isEmpty()) return val map = _lanes.value.toMutableMap() @@ -506,12 +623,13 @@ class ChatStore { val map = _lanes.value.toMutableMap() var changed = false for ((lane, list) in map) { - val updated = list.map { item -> - when (item) { - is ToolItem -> if (!item.done) item.copy(done = true, ok = false) else item - is MessageItem -> if (item.streaming) item.copy(streaming = false) else item + val updated = + list.map { item -> + when (item) { + is ToolItem -> if (!item.done) item.copy(done = true, ok = false) else item + is MessageItem -> if (item.streaming) item.copy(streaming = false) else item + } } - } if (updated != list) { map[lane] = updated changed = true @@ -525,11 +643,14 @@ class ChatStore { val map = _lanes.value.toMutableMap() var changed = false for ((lane, list) in map) { - val updated = list.map { item -> - if (item is MessageItem && item.role == ROLE_USER && item.status == MsgStatus.Pending) { - item.copy(pending = false, status = MsgStatus.Failed) - } else item - } + val updated = + list.map { item -> + if (item is MessageItem && item.role == ROLE_USER && item.status == MsgStatus.Pending) { + item.copy(pending = false, status = MsgStatus.Failed) + } else { + item + } + } if (updated != list) { map[lane] = updated changed = true @@ -539,12 +660,17 @@ class ChatStore { } /** M7: re-arm a failed user message for a retry send. */ - fun rearmForRetry(lane: String, messageId: String) { + fun rearmForRetry( + lane: String, + messageId: String, + ) { updateLane(lane) { list -> list.map { item -> if (item is MessageItem && item.id == messageId && item.status == MsgStatus.Failed) { item.copy(pending = true, status = MsgStatus.Pending) - } else item + } else { + item + } } } } @@ -558,12 +684,16 @@ class ChatStore { * the in-memory store is empty and the `sync` delta does not cover older * messages. */ - fun loadHistory(lane: String, messages: List) { + fun loadHistory( + lane: String, + messages: List, + ) { updateLane(lane) { list -> val historyIds = messages.map { it.id }.toSet() - val preserved = list.filter { item -> - item !is MessageItem || item.id !in historyIds - } + val preserved = + list.filter { item -> + item !is MessageItem || item.id !in historyIds + } (messages + preserved).sortedBy { item -> (item as? MessageItem)?.ts?.takeIf { it > 0 } ?: Long.MAX_VALUE } @@ -573,4 +703,4 @@ class ChatStore { fun clear() { _lanes.value = emptyMap() } -} \ No newline at end of file +} diff --git a/app/shared/src/commonMain/kotlin/iris/data/SecureStore.kt b/app/shared/src/commonMain/kotlin/iris/data/SecureStore.kt index 6d154d5..fc5cce4 100644 --- a/app/shared/src/commonMain/kotlin/iris/data/SecureStore.kt +++ b/app/shared/src/commonMain/kotlin/iris/data/SecureStore.kt @@ -66,6 +66,15 @@ interface SecureStore { * system font scale). */ var fontSizeScale: Float + /** UI setting: show the runtime-metadata footer under final assistant + * messages (model / context % / cwd / latency / cost). */ + var runtimeFooterEnabled: Boolean + + /** UI setting: which runtime-footer fields to show, as a comma-separated + * list in display order (e.g. "model,context_pct,cwd"). Empty = the + * default set. Valid keys: model, context_pct, cwd, latency, cost. */ + var runtimeFooterFields: String + fun savePairing( url: String, token: String, diff --git a/app/shared/src/commonMain/kotlin/iris/protocol/Protocol.kt b/app/shared/src/commonMain/kotlin/iris/protocol/Protocol.kt index c5c5b0c..eafe4ed 100644 --- a/app/shared/src/commonMain/kotlin/iris/protocol/Protocol.kt +++ b/app/shared/src/commonMain/kotlin/iris/protocol/Protocol.kt @@ -20,11 +20,12 @@ import kotlinx.serialization.json.put const val PROTOCOL_VERSION = 1 object IrisJson { - val instance: Json = Json { - ignoreUnknownKeys = true - encodeDefaults = true - isLenient = true - } + val instance: Json = + Json { + ignoreUnknownKeys = true + encodeDefaults = true + isLenient = true + } } // ── Frame type constants ──────────────────────────────────────────────── @@ -198,6 +199,22 @@ data class HelloAckPayload( // ── message (server -> app) ───────────────────────────────────────────── +/** + * Structured runtime metadata for the app's footer (mirror of the gateway's + * `runtime` object). The gateway ALWAYS sends this on final assistant + * messages; whether/what is shown is a per-app setting (Settings → Runtime + * footer), NOT a hermes config. Fields are absent when the data is + * unavailable (local models have no cost, etc.). + */ +@Serializable +data class RuntimeMeta( + val model: String? = null, + @SerialName("context_pct") val contextPct: Int? = null, + val cwd: String? = null, + val latency: Double? = null, + val cost: Double? = null, +) + @Serializable data class MessagePayload( @SerialName("message_id") val messageId: String, @@ -207,6 +224,7 @@ data class MessagePayload( @SerialName("reply_to") val replyTo: String? = null, val model: String? = null, val tokens: Int? = null, + val runtime: RuntimeMeta? = null, val ts: Long? = null, val media: List = emptyList(), ) @@ -260,7 +278,9 @@ data class MediaPullPayload( ) @Serializable -data class MediaPullEndPayload(val ok: Boolean) +data class MediaPullEndPayload( + val ok: Boolean, +) // ── M2: streaming frames (server -> app) ──────────────────────────────── @@ -283,6 +303,7 @@ data class MessageStopPayload( val reasoning: String? = null, val model: String? = null, val tokens: Int? = null, + val runtime: RuntimeMeta? = null, val ts: Long? = null, ) @@ -335,13 +356,20 @@ data class MessageSendPayload( // ── typing / error / ping ─────────────────────────────────────────────── @Serializable -data class TypingPayload(val on: Boolean) +data class TypingPayload( + val on: Boolean, +) @Serializable -data class ErrorPayload(val code: String, val message: String) +data class ErrorPayload( + val code: String, + val message: String, +) @Serializable -data class PingPayload(val ts: Long? = null) +data class PingPayload( + val ts: Long? = null, +) // ── M3: channel directory (app -> server requests) ────────────────────── @@ -353,13 +381,19 @@ data class ChannelCreatePayload( ) @Serializable -data class ChannelRenamePayload(val name: String) +data class ChannelRenamePayload( + val name: String, +) @Serializable -data class ChannelFavoritePayload(val on: Boolean) +data class ChannelFavoritePayload( + val on: Boolean, +) @Serializable -data class ChannelSetAutomationPayload(val on: Boolean) +data class ChannelSetAutomationPayload( + val on: Boolean, +) @Serializable data class ChannelIconPayload( @@ -370,10 +404,14 @@ data class ChannelIconPayload( ) @Serializable -data class ChannelListPayload(val channels: List = emptyList()) +data class ChannelListPayload( + val channels: List = emptyList(), +) @Serializable -data class ChannelDeletedPayload(@SerialName("chat_id") val chatId: String) +data class ChannelDeletedPayload( + @SerialName("chat_id") val chatId: String, +) // ── M3: search ────────────────────────────────────────────────────────── @@ -424,10 +462,14 @@ data class CommandsCatalogPayload( // ── M3: sync (reconnect catch-up) ─────────────────────────────────────── @Serializable -data class SyncPayload(val cursor: Long) +data class SyncPayload( + val cursor: Long, +) @Serializable -data class SyncDonePayload(val cursor: Long) +data class SyncDonePayload( + val cursor: Long, +) // ── history (full message history for a chat/thread) ──────────────────── @@ -440,6 +482,7 @@ data class HistoryMessage( val reasoning: String? = null, val model: String? = null, val tokens: Int? = null, + val runtime: RuntimeMeta? = null, val ts: Long? = null, val media: List = emptyList(), ) @@ -496,7 +539,9 @@ data class ReadReceiptPayload( ) @Serializable -data class StatusPayload(val state: String) +data class StatusPayload( + val state: String, +) // ── Frame builders ────────────────────────────────────────────────────── @@ -509,16 +554,17 @@ fun helloFrame( ): Frame = Frame( type = TYPE_HELLO, - payload = IrisJson.instance.encodeToJsonElement( - HelloPayload.serializer(), - HelloPayload( - token = token, - deviceId = deviceId, - deviceName = deviceName, - fcmToken = fcmToken, - ntfyTopic = ntfyTopic, + payload = + IrisJson.instance.encodeToJsonElement( + HelloPayload.serializer(), + HelloPayload( + token = token, + deviceId = deviceId, + deviceName = deviceName, + fcmToken = fcmToken, + ntfyTopic = ntfyTopic, + ), ), - ), ) fun messageSendFrame( @@ -534,79 +580,109 @@ fun messageSendFrame( type = TYPE_MESSAGE_SEND, chatId = chatId, threadId = threadId, - payload = IrisJson.instance.encodeToJsonElement( - MessageSendPayload.serializer(), - MessageSendPayload(text = text, mediaRefs = mediaRefs, autoThread = autoThread), - ), + payload = + IrisJson.instance.encodeToJsonElement( + MessageSendPayload.serializer(), + MessageSendPayload(text = text, mediaRefs = mediaRefs, autoThread = autoThread), + ), ) -fun pingFrame(): Frame = - Frame(type = TYPE_PING, payload = IrisJson.instance.encodeToJsonElement(PingPayload.serializer(), PingPayload())) +fun pingFrame(): Frame = Frame(type = TYPE_PING, payload = IrisJson.instance.encodeToJsonElement(PingPayload.serializer(), PingPayload())) // ── M3 frame builders ─────────────────────────────────────────────────── -fun channelCreateFrame(id: Int, name: String, kind: String = "channel", parentChatId: String? = null): Frame = +fun channelCreateFrame( + id: Int, + name: String, + kind: String = "channel", + parentChatId: String? = null, +): Frame = Frame( id = id, type = TYPE_CHANNEL_CREATE, - payload = IrisJson.instance.encodeToJsonElement( - ChannelCreatePayload.serializer(), - ChannelCreatePayload(name = name, kind = kind, parentChatId = parentChatId), - ), + payload = + IrisJson.instance.encodeToJsonElement( + ChannelCreatePayload.serializer(), + ChannelCreatePayload(name = name, kind = kind, parentChatId = parentChatId), + ), ) -fun channelRenameFrame(id: Int, chatId: String, name: String): Frame = +fun channelRenameFrame( + id: Int, + chatId: String, + name: String, +): Frame = Frame( id = id, type = TYPE_CHANNEL_RENAME, chatId = chatId, - payload = IrisJson.instance.encodeToJsonElement( - ChannelRenamePayload.serializer(), - ChannelRenamePayload(name = name), - ), + payload = + IrisJson.instance.encodeToJsonElement( + ChannelRenamePayload.serializer(), + ChannelRenamePayload(name = name), + ), ) -fun channelSetDefaultFrame(id: Int, chatId: String): Frame = - Frame(id = id, type = TYPE_CHANNEL_SET_DEFAULT, chatId = chatId) +fun channelSetDefaultFrame( + id: Int, + chatId: String, +): Frame = Frame(id = id, type = TYPE_CHANNEL_SET_DEFAULT, chatId = chatId) -fun channelFavoriteFrame(id: Int, chatId: String, on: Boolean): Frame = +fun channelFavoriteFrame( + id: Int, + chatId: String, + on: Boolean, +): Frame = Frame( id = id, type = TYPE_CHANNEL_FAVORITE, chatId = chatId, - payload = IrisJson.instance.encodeToJsonElement( - ChannelFavoritePayload.serializer(), - ChannelFavoritePayload(on = on), - ), + payload = + IrisJson.instance.encodeToJsonElement( + ChannelFavoritePayload.serializer(), + ChannelFavoritePayload(on = on), + ), ) -fun channelSetAutomationFrame(id: Int, chatId: String, on: Boolean): Frame = +fun channelSetAutomationFrame( + id: Int, + chatId: String, + on: Boolean, +): Frame = Frame( id = id, type = TYPE_CHANNEL_SET_AUTOMATION, chatId = chatId, - payload = IrisJson.instance.encodeToJsonElement( - ChannelSetAutomationPayload.serializer(), - ChannelSetAutomationPayload(on = on), - ), + payload = + IrisJson.instance.encodeToJsonElement( + ChannelSetAutomationPayload.serializer(), + ChannelSetAutomationPayload(on = on), + ), ) -fun channelIconFrame(id: Int, chatId: String, icon: String?, color: String?): Frame = +fun channelIconFrame( + id: Int, + chatId: String, + icon: String?, + color: String?, +): Frame = Frame( id = id, type = TYPE_CHANNEL_ICON, chatId = chatId, - payload = IrisJson.instance.encodeToJsonElement( - ChannelIconPayload.serializer(), - ChannelIconPayload(icon = icon, color = color), - ), + payload = + IrisJson.instance.encodeToJsonElement( + ChannelIconPayload.serializer(), + ChannelIconPayload(icon = icon, color = color), + ), ) -fun channelDeleteFrame(id: Int, chatId: String): Frame = - Frame(id = id, type = TYPE_CHANNEL_DELETE, chatId = chatId) +fun channelDeleteFrame( + id: Int, + chatId: String, +): Frame = Frame(id = id, type = TYPE_CHANNEL_DELETE, chatId = chatId) -fun channelListFrame(id: Int): Frame = - Frame(id = id, type = TYPE_CHANNEL_LIST) +fun channelListFrame(id: Int): Frame = Frame(id = id, type = TYPE_CHANNEL_LIST) fun searchFrame( id: Int, @@ -620,25 +696,29 @@ fun searchFrame( type = TYPE_SEARCH, chatId = chatId, threadId = threadId, - payload = IrisJson.instance.encodeToJsonElement( - SearchPayload.serializer(), - SearchPayload(query = query, scope = scope, chatId = chatId, threadId = threadId), - ), + payload = + IrisJson.instance.encodeToJsonElement( + SearchPayload.serializer(), + SearchPayload(query = query, scope = scope, chatId = chatId, threadId = threadId), + ), ) /** Request the gateway's slash-command catalog (the composer's "/" drawer). * Answered by a `commands.catalog` frame carrying the same id. */ -fun commandsCatalogFrame(id: Int): Frame = - Frame(id = id, type = TYPE_COMMANDS_CATALOG) +fun commandsCatalogFrame(id: Int): Frame = Frame(id = id, type = TYPE_COMMANDS_CATALOG) -fun syncFrame(id: Int, cursor: Long): Frame = +fun syncFrame( + id: Int, + cursor: Long, +): Frame = Frame( id = id, type = TYPE_SYNC, - payload = IrisJson.instance.encodeToJsonElement( - SyncPayload.serializer(), - SyncPayload(cursor = cursor), - ), + payload = + IrisJson.instance.encodeToJsonElement( + SyncPayload.serializer(), + SyncPayload(cursor = cursor), + ), ) /** Request a page of full message history for a chat/thread (initial open / @@ -656,10 +736,11 @@ fun historyFrame( type = TYPE_HISTORY, chatId = chatId, threadId = threadId, - payload = buildJsonObject { - if (beforeMessageId != null) put("before_message_id", beforeMessageId) - put("limit", limit) - }, + payload = + buildJsonObject { + if (beforeMessageId != null) put("before_message_id", beforeMessageId) + put("limit", limit) + }, ) /** Request deletion of the given message(s) in a chat/thread. The server @@ -675,9 +756,10 @@ fun messageDeleteFrame( type = TYPE_MESSAGE_DELETE, chatId = chatId, threadId = threadId, - payload = buildJsonObject { - put("message_ids", JsonArray(messageIds.map { JsonPrimitive(it) })) - }, + payload = + buildJsonObject { + put("message_ids", JsonArray(messageIds.map { JsonPrimitive(it) })) + }, ) // ── M4 frame builders ──────────────────────────────────────────────────── @@ -693,39 +775,53 @@ fun mediaUploadStartFrame( Frame( id = id, type = TYPE_MEDIA_UPLOAD_START, - payload = IrisJson.instance.encodeToJsonElement( - MediaUploadStartPayload.serializer(), - MediaUploadStartPayload(mediaRef, kind, mime, filename, size), - ), + payload = + IrisJson.instance.encodeToJsonElement( + MediaUploadStartPayload.serializer(), + MediaUploadStartPayload(mediaRef, kind, mime, filename, size), + ), ) -fun mediaUploadEndFrame(id: Int, mediaRef: String, sha256: String): Frame = +fun mediaUploadEndFrame( + id: Int, + mediaRef: String, + sha256: String, +): Frame = Frame( id = id, type = TYPE_MEDIA_UPLOAD_END, - payload = IrisJson.instance.encodeToJsonElement( - MediaUploadEndPayload.serializer(), - MediaUploadEndPayload(mediaRef, sha256), - ), + payload = + IrisJson.instance.encodeToJsonElement( + MediaUploadEndPayload.serializer(), + MediaUploadEndPayload(mediaRef, sha256), + ), ) -fun mediaPullFrame(id: Int, mediaId: String): Frame = +fun mediaPullFrame( + id: Int, + mediaId: String, +): Frame = Frame( id = id, type = TYPE_MEDIA_PULL, - payload = IrisJson.instance.encodeToJsonElement( - MediaPullPayload.serializer(), - MediaPullPayload(mediaId), - ), + payload = + IrisJson.instance.encodeToJsonElement( + MediaPullPayload.serializer(), + MediaPullPayload(mediaId), + ), ) // ── M5 frame builders ─────────────────────────────────────────────────── -fun fcmRegisterFrame(fcmToken: String? = null, ntfyTopic: String? = null): Frame = +fun fcmRegisterFrame( + fcmToken: String? = null, + ntfyTopic: String? = null, +): Frame = Frame( type = TYPE_FCM_REGISTER, - payload = IrisJson.instance.encodeToJsonElement( - FcmRegisterPayload.serializer(), - FcmRegisterPayload(fcmToken = fcmToken, ntfyTopic = ntfyTopic), - ), - ) \ No newline at end of file + payload = + IrisJson.instance.encodeToJsonElement( + FcmRegisterPayload.serializer(), + FcmRegisterPayload(fcmToken = fcmToken, ntfyTopic = ntfyTopic), + ), + ) diff --git a/app/shared/src/commonMain/kotlin/iris/state/IrisController.kt b/app/shared/src/commonMain/kotlin/iris/state/IrisController.kt index e8869ec..ec382f8 100644 --- a/app/shared/src/commonMain/kotlin/iris/state/IrisController.kt +++ b/app/shared/src/commonMain/kotlin/iris/state/IrisController.kt @@ -146,8 +146,7 @@ class IrisController( /** True while [frame] is a sync replay that already reached the device * via push (live frames carry no cursor and are never suppressed). */ - private fun isPushedReplay(frame: iris.protocol.Frame): Boolean = - frame.cursor?.let { it <= lastPushedCursor } ?: false + private fun isPushedReplay(frame: iris.protocol.Frame): Boolean = frame.cursor?.let { it <= lastPushedCursor } ?: false // ── M3: threads toggle (per-app for now; per-channel lands later) ───── // Persisted (Settings → "Threads"). @@ -183,6 +182,48 @@ class IrisController( store.reasoningAutoCollapse = _reasoningAutoCollapse.value } + // ── Runtime-metadata footer (Settings → "Runtime footer") ─────────────────────────────────────────────── + // The gateway ALWAYS sends the structured runtime metadata on final + // assistant messages; these settings control whether/what the app shows + // (per-device display preference, like tool verbosity). [runtimeFooterFields] + // is a comma-separated list of field keys in display order. + private val _runtimeFooterEnabled = MutableStateFlow(store.runtimeFooterEnabled) + val runtimeFooterEnabled: StateFlow = _runtimeFooterEnabled.asStateFlow() + + private val _runtimeFooterFields = MutableStateFlow(parseRuntimeFields(store.runtimeFooterFields)) + val runtimeFooterFields: StateFlow> = _runtimeFooterFields.asStateFlow() + + fun toggleRuntimeFooter() { + _runtimeFooterEnabled.value = !_runtimeFooterEnabled.value + store.runtimeFooterEnabled = _runtimeFooterEnabled.value + } + + /** Toggle a single runtime-footer field on/off (preserving order). */ + fun toggleRuntimeField(field: String) { + val current = _runtimeFooterFields.value + val updated = if (field in current) current - field else current + field + _runtimeFooterFields.value = updated + store.runtimeFooterFields = updated.joinToString(",") + } + + companion object { + const val FONT_SCALE_MIN = 0.8f + const val FONT_SCALE_MAX = 1.5f + + /** Valid runtime-footer field keys (mirror of the gateway's RUNTIME_FIELDS). */ + val RUNTIME_FIELD_KEYS = listOf("model", "context_pct", "cwd", "latency", "cost") + + /** Default footer fields (in display order) when the user has none set. */ + val RUNTIME_FIELDS_DEFAULT = listOf("model", "context_pct", "cwd") + + /** Parse a persisted comma-separated field list; unknown keys dropped, + * order preserved, empty falls back to the default set. */ + fun parseRuntimeFields(raw: String): List { + val parsed = raw.split(",").map { it.trim() }.filter { it in RUNTIME_FIELD_KEYS } + return if (parsed.isEmpty()) RUNTIME_FIELDS_DEFAULT else parsed + } + } + // ── Appearance (Settings → Appearance) ──────────────────────────────── // User-customizable bubble colors + background (color or image). // Persisted as ARGB ints + an image path (see UserTheme). @@ -252,11 +293,6 @@ class IrisController( store.fontSizeScale = clamped } - companion object { - const val FONT_SCALE_MIN = 0.8f - const val FONT_SCALE_MAX = 1.5f - } - // ── M3: search state ────────────────────────────────────────────────── private val _searchResults = MutableStateFlow>(emptyList()) val searchResults: StateFlow> = _searchResults.asStateFlow() @@ -915,6 +951,7 @@ private fun HistoryMessage.toMessageItem(): MessageItem = reasoning = reasoning, model = model, tokens = tokens, + runtime = runtime, media = media.map { MediaItem(mediaId = it.mediaId, kind = it.kind, mime = it.mime, size = it.size, filename = it.filename) diff --git a/app/shared/src/commonMain/kotlin/iris/ui/screens/ChatScreen.kt b/app/shared/src/commonMain/kotlin/iris/ui/screens/ChatScreen.kt index f1f56da..db9058c 100644 --- a/app/shared/src/commonMain/kotlin/iris/ui/screens/ChatScreen.kt +++ b/app/shared/src/commonMain/kotlin/iris/ui/screens/ChatScreen.kt @@ -141,6 +141,7 @@ import iris.util.prepareForMarkdown import iris.util.preserveNewlinesAsHardBreaks import kotlinx.coroutines.delay import kotlinx.coroutines.launch +import kotlin.math.roundToInt /** * Chat screen (M3): channel drawer + topic switcher + search + settings, @@ -157,6 +158,8 @@ fun ChatScreen(controller: IrisController) { val toolDetail by controller.toolDetail.collectAsState() val threadsEnabled by controller.threadsEnabled.collectAsState() val reasoningAutoCollapse by controller.reasoningAutoCollapse.collectAsState() + val runtimeFooterEnabled by controller.runtimeFooterEnabled.collectAsState() + val runtimeFooterFields by controller.runtimeFooterFields.collectAsState() val channels by controller.channels.channels.collectAsState() val (currentChatId, currentThreadId) = controller.chat.parseLane(currentLane) @@ -506,6 +509,8 @@ fun ChatScreen(controller: IrisController) { } } }, + runtimeFooterEnabled = runtimeFooterEnabled, + runtimeFooterFields = runtimeFooterFields, ) } } @@ -1776,8 +1781,11 @@ private fun InspectorPane( InfoRow("Threads", threads.size.toString()) Spacer(modifier = Modifier.height(8.dp)) Text("Last reply", style = MaterialTheme.typography.titleSmall, modifier = Modifier.padding(bottom = 4.dp)) - InfoRow("Model", lastAssistant?.model ?: "—") - InfoRow("Tokens", lastAssistant?.tokens?.toString() ?: "—") + InfoRow("Model", lastAssistant?.runtime?.model ?: "—") + InfoRow("Context", lastAssistant?.runtime?.contextPct?.let { "$it%" } ?: "—") + InfoRow("Workdir", lastAssistant?.runtime?.cwd ?: "—") + InfoRow("Latency", lastAssistant?.runtime?.latency?.let { formatLatency(it) } ?: "—") + InfoRow("Cost", lastAssistant?.runtime?.cost?.let { formatCost(it) } ?: "—") } } @@ -2211,6 +2219,8 @@ private fun MessageBubble( selected: Boolean = false, onToggleSelect: () -> Unit = {}, onLongPress: () -> Unit = {}, + runtimeFooterEnabled: Boolean = false, + runtimeFooterFields: List = emptyList(), ) { val isUser = msg.role == ROLE_USER val isCommentary = msg.isCommentary @@ -2345,28 +2355,34 @@ private fun MessageBubble( } } if (!isUser) { - // Model / token footer (final assistant answers only). - if (!isCommentary && !msg.streaming && - (msg.model != null || msg.tokens != null) - ) { + // Runtime-metadata footer + timestamp on ONE line (footer left, + // time right) — Telegram-style. The app controls whether/what + // the footer shows via Settings → Runtime footer. + val footerText = + if (!isCommentary && !msg.streaming && runtimeFooterEnabled) { + buildRuntimeFooterText(msg, runtimeFooterFields) + } else { + "" + } + if (footerText.isNotEmpty() || time.isNotEmpty()) { Spacer(modifier = Modifier.height(4.dp)) - Text( - buildString { - msg.model?.let { append(it) } - if (msg.model != null && msg.tokens != null) append(" · ") - msg.tokens?.let { append("$it tok") } - }, - color = textColor.copy(alpha = 0.4f), - fontSize = 10.sp, - ) - } - // M7: timestamp below the footer (reference shows "12:41"). - if (time.isNotEmpty()) { Row( modifier = Modifier.fillMaxWidth(), - horizontalArrangement = Arrangement.End, + verticalAlignment = Alignment.CenterVertically, ) { - Text(time, fontSize = 10.sp, color = textColor.copy(alpha = 0.5f)) + if (footerText.isNotEmpty()) { + Text( + footerText, + color = textColor.copy(alpha = 0.4f), + fontSize = 10.sp, + modifier = Modifier.weight(1f), + ) + } else { + Spacer(modifier = Modifier.weight(1f)) + } + if (time.isNotEmpty()) { + Text(time, fontSize = 10.sp, color = textColor.copy(alpha = 0.5f)) + } } } } @@ -2378,6 +2394,39 @@ private fun MessageBubble( } } +/** Build the runtime-footer text from the message's runtime metadata, showing + * only the enabled [fields] (in order) that have data. Empty when none. */ +private fun buildRuntimeFooterText( + msg: MessageItem, + fields: List, +): String { + val runtime = msg.runtime ?: return "" + val parts = mutableListOf() + for (field in fields) { + when (field) { + "model" -> runtime.model?.let { parts.add(it) } + "context_pct" -> runtime.contextPct?.let { parts.add("$it%") } + "cwd" -> runtime.cwd?.let { parts.add(it) } + "latency" -> runtime.latency?.let { parts.add(formatLatency(it)) } + "cost" -> runtime.cost?.let { parts.add(formatCost(it)) } + } + } + return parts.joinToString(" · ") +} + +/** Humanize a turn duration: `<1s`, `22s`, `1m05s` (matches hermes' footer). */ +private fun formatLatency(seconds: Double): String { + if (seconds < 1) return "<1s" + val total = seconds.roundToInt() + if (total < 60) return "${total}s" + val m = total / 60 + val s = total % 60 + return "${m}m${s.toString().padStart(2, '0')}s" +} + +/** Format a cost in USD: sub-cent at 4 decimals, else 2. */ +private fun formatCost(cost: Double): String = if (cost < 0.01) "$%.4f".format(cost) else "$%.2f".format(cost) + /** Circular selection indicator shown beside a bubble in selection mode. */ @Composable private fun SelectionCheck(selected: Boolean) { diff --git a/app/shared/src/commonMain/kotlin/iris/ui/screens/SettingsScreen.kt b/app/shared/src/commonMain/kotlin/iris/ui/screens/SettingsScreen.kt index 1eacc9f..0eaa11b 100644 --- a/app/shared/src/commonMain/kotlin/iris/ui/screens/SettingsScreen.kt +++ b/app/shared/src/commonMain/kotlin/iris/ui/screens/SettingsScreen.kt @@ -72,6 +72,8 @@ fun SettingsScreen( val threadsEnabled by controller.threadsEnabled.collectAsState() val streamingEnabled by controller.streamingEnabled.collectAsState() val reasoningAutoCollapse by controller.reasoningAutoCollapse.collectAsState() + val runtimeFooterEnabled by controller.runtimeFooterEnabled.collectAsState() + val runtimeFooterFields by controller.runtimeFooterFields.collectAsState() val toolDetail by controller.toolDetail.collectAsState() val fontSizeScale by controller.fontSizeScale.collectAsState() val theme = LocalUserTheme.current @@ -159,6 +161,39 @@ fun SettingsScreen( ) } } + SettingsCard { + Row( + verticalAlignment = Alignment.CenterVertically, + modifier = Modifier.fillMaxWidth(), + ) { + Column(modifier = Modifier.weight(1f)) { + Text("📊 Runtime footer", fontSize = 14.sp) + Text( + "Show model, context, workdir, latency and cost under replies", + fontSize = 12.sp, + color = IrisColors.textDim, + ) + } + Switch( + checked = runtimeFooterEnabled, + onCheckedChange = { controller.toggleRuntimeFooter() }, + ) + } + if (runtimeFooterEnabled) { + Spacer(modifier = Modifier.height(8.dp)) + Text("Fields to show", fontSize = 12.sp, color = IrisColors.textDim) + Spacer(modifier = Modifier.height(6.dp)) + Row(horizontalArrangement = Arrangement.spacedBy(6.dp)) { + IrisController.RUNTIME_FIELD_KEYS.forEach { field -> + TopicChip( + label = runtimeFieldLabel(field), + selected = field in runtimeFooterFields, + onClick = { controller.toggleRuntimeField(field) }, + ) + } + } + } + } Text( "Appearance", @@ -620,3 +655,14 @@ private fun argbToHsv(argb: Int): FloatArray { /** File name from a path (no java.io.File in common code). */ private fun fileNameOf(path: String): String = path.substringAfterLast('/').ifEmpty { path.substringAfterLast('\\') } + +/** Human label for a runtime-footer field key (Settings → Runtime footer). */ +private fun runtimeFieldLabel(field: String): String = + when (field) { + "model" -> "Model" + "context_pct" -> "Context %" + "cwd" -> "Workdir" + "latency" -> "Latency" + "cost" -> "Cost" + else -> field + } diff --git a/app/shared/src/desktopMain/kotlin/iris/platform/DesktopSecureStore.kt b/app/shared/src/desktopMain/kotlin/iris/platform/DesktopSecureStore.kt index d968c0d..6d4fc48 100644 --- a/app/shared/src/desktopMain/kotlin/iris/platform/DesktopSecureStore.kt +++ b/app/shared/src/desktopMain/kotlin/iris/platform/DesktopSecureStore.kt @@ -48,6 +48,8 @@ class DesktopSecureStore : SecureStore { val backgroundColor: Int = UserTheme.DEFAULT_BACKGROUND, val backgroundImagePath: String = Backdrop.DEFAULT.path, val fontSizeScale: Float = 1.0f, + val runtimeFooterEnabled: Boolean = false, + val runtimeFooterFields: String = "", ) init { @@ -231,6 +233,20 @@ class DesktopSecureStore : SecureStore { save(d.copy(fontSizeScale = value)) } + override var runtimeFooterEnabled: Boolean + get() = load().runtimeFooterEnabled + set(value) { + val d = load() + save(d.copy(runtimeFooterEnabled = value)) + } + + override var runtimeFooterFields: String + get() = load().runtimeFooterFields + set(value) { + val d = load() + save(d.copy(runtimeFooterFields = value)) + } + override fun savePairing( url: String, token: String, diff --git a/docs/04-wire-protocol.md b/docs/04-wire-protocol.md index 188cf4a..49c587d 100644 --- a/docs/04-wire-protocol.md +++ b/docs/04-wire-protocol.md @@ -34,7 +34,9 @@ JSON `media.upload.end` / final ack. See `07-media.md`. ## Server → App (events / responses) ### `hello.ack` + Pairing succeeded. + ```json {"type":"hello.ack","payload":{ "server_caps":{"streaming":true,"reasoning":true,"tools":true,"media":true, @@ -44,13 +46,16 @@ Pairing succeeded. "channels":[{"chat_id":"android:default","name":"Default","kind":"default","is_default":true}] }} ``` + `last_pushed_cursor` is the highest outbox cursor already delivered to THIS device via the push backend (0 = never). The app skips system notifications for sync-replayed frames with `cursor <= last_pushed_cursor` — they already woke the device via push (dedupe, `08-push.md` §8.7). ### `message` + A final / standalone message. + ```json {"type":"message","chat_id":"android:default","thread_id":null, "payload":{ @@ -60,40 +65,65 @@ A final / standalone message. "media":[{"media_id":"md_5","kind":"video","mime":"video/mp4","size":123456, "filename":"clip.mp4"}], // optional "reply_to":"m_8999", // optional - "model":"qwen3-27b","tokens":11,"ts":1724000000000 + "model":"qwen3-27b","tokens":11,"ts":1724000000000, + "runtime":{"model":"qwen3-27b","context_pct":38,"cwd":"~", + "latency":22.5,"cost":0.0012} // optional; see below }} ``` + `role` ∈ `user | assistant | system | cron`. `reasoning` present only when the agent produced reasoning and `show_reasoning` is on. +`runtime` (optional) is the **structured runtime-metadata footer** the app +renders under final assistant messages (Telegram-style). The gateway ALWAYS +sends it on final assistant messages; whether/what is shown is a **per-app +setting** (Settings → Runtime footer), NOT a hermes config. Keys (all +optional; absent when the data is unavailable, e.g. local models have no +cost): + +- `model` — bare model id, vendor prefix dropped (`gpt-5.4`) +- `context_pct` — last-call context occupancy, 0-100 (int) +- `cwd` — home-relative working dir (`~`) +- `latency` — wall-clock turn duration, seconds (float) +- `cost` — turn cost, USD (float) + ### `message.start` / `message.update` / `message.stop` + Streaming a bubble. `update` carries the **full** current text (app replaces). + ```json {"type":"message.start","chat_id":"…","payload":{"message_id":"m_9002","role":"assistant"}} {"type":"message.update","chat_id":"…","payload":{"message_id":"m_9002","text":"partial…"}} {"type":"message.stop","chat_id":"…","payload":{"message_id":"m_9002","final_text":"full…", - "reasoning":"…","model":"…","tokens":11}} + "reasoning":"…","model":"…","tokens":11, + "runtime":{"model":"…","context_pct":38,"cwd":"~","latency":22.5}}} ``` ### `message.deleted` + The given message(s) were deleted from a chat/thread. Response to a `message.delete` request (`id` set) **and** broadcast to every device so all of them drop the message(s) from their cache. Also outboxed, so a device that was offline learns of the deletion on its next `sync`. + ```json {"type":"message.deleted","id":30,"chat_id":"android:default","thread_id":null, "payload":{"message_ids":["m_9001","m_9002"]}} ``` ### `commentary` + Intermediate assistant beat (between tool iterations). + ```json {"type":"commentary","chat_id":"…","payload":{"message_id":"m_9003","text":"Let me inspect the repo first."}} ``` ### `tool.start` / `tool.progress` / `tool.end` + **Structured** tool events. The app decides how much to show (everything / truncated / nothing). + ```json {"type":"tool.start","chat_id":"…","payload":{ "index":3,"name":"terminal","preview":"pytest -q","args":{"command":"pytest -q"}}} @@ -101,24 +131,31 @@ truncated / nothing). {"type":"tool.end","chat_id":"…","payload":{"index":3,"name":"terminal","ok":true,"duration":12.4, "output_preview":"12 passed"}} ``` + `args` may be large; the app truncates per its setting. `output_preview` is a short tail (full output is not streamed — it lives in agent history). ### `typing` / `typing.stop` + ```json {"type":"typing","chat_id":"…","payload":{"on":true}} ``` ### `notification` + In-app banner (foreground) and/or push mirror (background). + ```json {"type":"notification","chat_id":"…","payload":{ "kind":"channel_renamed","title":"ARIA","body":"Renamed topic to …","ts":1724000000000}} ``` + `kind` ∈ `channel_renamed | channel_created | cron | approval | clarify | generic`. ### `picker.model` / `picker.choice` / `picker.clarify` / `picker.approval` / `picker.confirm` + Interactive prompts. App renders a native picker; answers via `picker.select`. + ```json {"type":"picker.model","chat_id":"…","payload":{ "picker_id":"pm_1","current_model":"qwen3-27b","current_provider":"local", @@ -129,19 +166,24 @@ Interactive prompts. App renders a native picker; answers via `picker.select`. ``` ### `channel.list` / `channel.created` / `channel.renamed` / `channel.deleted` + Channel directory updates. **Broadcast to all connected devices** (no explicit subscribe; the server pushes to every open WS). + ```json {"type":"channel.created","payload":{"chat_id":"android:chan_7","name":"Cron Reports", "kind":"channel","parent_chat_id":null}} ``` + `channel.created` may carry `"auto":true` for a thread the gateway minted itself for an incoming message (auto-threading): the name is an instant derived title, and a follow-up `channel.renamed` upgrades it to the model's title. ### `history` + Response to a `history` request. Returns a page of messages for a chat/thread. + ```json {"type":"history","id":20,"chat_id":"android:default","thread_id":null, "payload":{ @@ -154,17 +196,20 @@ Response to a `history` request. Returns a page of messages for a chat/thread. "oldest_message_id":"m_8990" }} ``` + `messages` are ordered oldest → newest. Paginate with `before_message_id` in the request. The app uses this to **populate the initial view** when a channel is opened (complements `sync`, which only replays undelivered outbox frames). ### `commands.catalog` + Request (app → server, empty payload) and response: the gateway's slash-command catalog for the app's `/` drawer. Derived from hermes' central `COMMAND_REGISTRY` (the same source the gateway help and the Telegram command menu use), restricted to commands available on gateway surfaces, plus plugin-registered commands. The app fuzzy-matches the typed prefix client-side (no `commands.complete` round-trip). + ```json {"type":"commands.catalog","id":21,"payload":{ "commands":[ @@ -173,11 +218,14 @@ client-side (no `commands.complete` round-trip). {"name":"/status","description":"Show session status","args_hint":"","category":"Info","aliases":[]} ]}} ``` + `name`/`aliases` carry the leading slash; `args_hint` is the registry's argument placeholder (empty when the command takes none). ### `commands.complete` + Response to a `commands.complete` request. Autocomplete matches for a typed prefix. + ```json {"type":"commands.complete","id":22,"payload":{ "prefix":"/mod", @@ -187,15 +235,19 @@ Response to a `commands.complete` request. Autocomplete matches for a typed pref ``` ### `agent.busy` / `agent.idle` + Agent lifecycle for a chat/thread. App shows a "thinking…" indicator on `busy`. + ```json {"type":"agent.busy","chat_id":"android:default","thread_id":null, "payload":{"reason":"processing"}} {"type":"agent.idle","chat_id":"android:default","thread_id":null,"payload":{}} ``` + `reason` ∈ `processing | tool | waiting_input | cron`. ### `search.results` + ```json {"type":"search.results","id":7,"payload":{ "query":"deploy","scope":"all","hits":[ @@ -204,43 +256,56 @@ Agent lifecycle for a chat/thread. App shows a "thinking…" indicator on `busy` ``` ### `media.offer` + Agent-sent media is available; app pulls bytes. + ```json {"type":"media.offer","chat_id":"…","payload":{ "media_id":"md_5","kind":"video","mime":"video/mp4","size":123456,"filename":"clip.mp4"}} ``` ### `read.receipt` + The gateway acknowledges that the agent has received and started processing the user's message. The app uses it to show ✓✓ on user bubbles. + ```json {"type":"read.receipt","chat_id":"android:default","payload":{"message_id":"m_9001"}} ``` + Emitted to the originating connection when a `message.send` is accepted for processing (at the moment it is handed to the agent), for user-originated messages only. ### `status` + Gateway health state. Broadcast to all connected clients at startup (`state: "online"`); `restarting` / `degraded` are reserved for future use. + ```json {"type":"status","payload":{"state":"online"}} ``` + `state` ∈ `online | restarting | degraded`. ### `error` + ```json {"type":"error","id":7,"payload":{"code":"not_found","message":"chat_id unknown"}} ``` + `code` ∈ `auth | not_found | rate_limited | media_too_large | unsupported | internal`. ### `pong` + Keepalive reply to `ping`. ## App → Server (requests / actions) ### `hello` + First frame; auth + caps. + ```json {"type":"hello","payload":{ "token":"","device_id":"dev_a1b2","device_name":"MIX 2S", @@ -249,12 +314,15 @@ First frame; auth + caps. ``` ### `message.send` + Send text (or a `/slash-command`). + ```json {"type":"message.send","id":10,"chat_id":"android:default","thread_id":null, "payload":{"text":"/model qwen3-27b","reply_to":"m_9001","media_refs":["mu_1"], "auto_thread":false}} ``` + `media_refs` reference completed `media.upload`s to attach. `auto_thread` (optional, default false) asks the gateway to mint a fresh thread for the message (auto-threading, `06-channels-cron-search.md` §6.3): @@ -264,7 +332,9 @@ that is not a slash command. The gateway then broadcasts `thread_id`. ### `media.upload.start` / (binary) / `media.upload.end` + See `07-media.md`. + ```json {"type":"media.upload.start","id":11,"payload":{ "media_ref":"mu_1","kind":"image","mime":"image/jpeg","size":204800,"filename":"a.jpg"}} @@ -273,26 +343,33 @@ See `07-media.md`. ``` ### `media.upload.ack` + Server → App response to `media.upload.end`: the ref is cached and may now be referenced in a `message.send` `media_refs`. Failures use `error` frames instead. + ```json {"type":"media.upload.ack","id":11,"payload":{"ok":true,"media_ref":"mu_1"}} ``` ### `media.pull` + Request agent-sent media bytes. + ```json {"type":"media.pull","id":12,"payload":{"media_id":"md_5"}} // server replies: binary frames, then {"type":"media.pull.end","id":12,"payload":{"ok":true}} ``` ### `picker.select` + Answer an interactive picker. + ```json {"type":"picker.select","id":13,"payload":{"picker_id":"pm_1","value":"local/qwen3-27b"}} ``` ### `channel.create` / `channel.rename` / `channel.set_default` / `channel.delete` + ```json {"type":"channel.create","id":14,"payload":{"name":"Cron Reports","kind":"channel"}} {"type":"channel.rename","id":15,"chat_id":"android:chan_7","payload":{"name":"Reports"}} @@ -300,79 +377,101 @@ Answer an interactive picker. ``` ### `search` + ```json {"type":"search","id":17,"payload":{"query":"deploy","scope":"all"}} {"type":"search","id":18,"payload":{"query":"deploy","scope":"chat","chat_id":"android:chan_7","thread_id":null}} ``` + `scope` ∈ `all | chat`. ### `read.receipt` + App → server: "user has viewed this message." Server stores the read state and broadcasts to other devices (for multi-device ✓✓ sync). The app uses it to mark messages as read locally (✓✓ on user bubbles). + ```json {"type":"read.receipt","payload":{"chat_id":"android:default","message_id":"m_9001"}} ``` ### `history` + Load a page of messages for a chat/thread (initial open, scroll-up pagination). + ```json {"type":"history","id":20,"chat_id":"android:default","thread_id":null, "payload":{"before_message_id":"m_8990","limit":50}} ``` + `before_message_id` — return messages older than this (omit for newest page). `limit` — max messages (default 50, max 200). ### `message.delete` + Delete the given message(s) from a chat/thread. The server removes them from the outbox (so `history`/`sync` no longer return them) and broadcasts `message.deleted` to every device. Idempotent: a message already gone (pruned by retention) still yields a `message.deleted` broadcast so live caches drop it. + ```json {"type":"message.delete","id":30,"chat_id":"android:default","thread_id":null, "payload":{"message_ids":["m_9001","m_9002"]}} ``` ### `commands.catalog` + Fetch the full slash-command catalog (for the `Menü` bottom sheet). + ```json {"type":"commands.catalog","id":21,"payload":{}} ``` ### `commands.complete` + Autocomplete for a typed `/prefix`. + ```json {"type":"commands.complete","id":22,"payload":{"prefix":"/mod"}} ``` ### `agent.stop` + Stop the current agent turn (abort generation / tool execution). + ```json {"type":"agent.stop","id":23,"chat_id":"android:default","thread_id":null,"payload":{}} ``` ### `agent.steer` + Inject a steering message mid-turn (redirects the agent without a new turn). + ```json {"type":"agent.steer","id":24,"chat_id":"android:default","thread_id":null, "payload":{"text":"Actually, focus on the error case."}} ``` ### `sync` + Reconnect catch-up. Replays **undelivered outbox frames** (frames sent while this device was offline). Does NOT load full history — use `history` for that. + ```json {"type":"sync","id":19,"payload":{"cursor":1042}} // server replays outbox frames with cursor > 1042, then {"type":"sync.done","id":19,"payload":{"cursor":1099}} ``` ### `fcm.register` + Update push token. + ```json {"type":"fcm.register","payload":{"fcm_token":"","ntfy_topic":""}} ``` ### `ping` + Keepalive. `{"type":"ping","payload":{"ts":1724000000000}}` → `pong`. ## Ordering & reliability @@ -387,4 +486,4 @@ Keepalive. `{"type":"ping","payload":{"ts":1724000000000}}` → `pong`. - Anything not delivered live goes to the **outbox** and is replayed by `sync`. - **Broadcast:** channel directory events (`channel.*`) and read-receipts are pushed to **all** connected devices for that gateway (no subscribe step). -- Requests get exactly one response or `error` (matched by `id`). \ No newline at end of file +- Requests get exactly one response or `error` (matched by `id`). diff --git a/docs/protocol/frames.schema.json b/docs/protocol/frames.schema.json index 850c7a2..a3a58bf 100644 --- a/docs/protocol/frames.schema.json +++ b/docs/protocol/frames.schema.json @@ -38,12 +38,13 @@ "reply_to": { "type": "string" }, "model": { "type": "string" }, "tokens": { "type": "integer" }, + "runtime": { "$ref": "#/definitions/runtime" }, "ts": { "type": "integer", "description": "epoch millis" } } }, "message.start": { "payload": { "message_id": { "type": "string" }, "role": { "type": "string" } } }, "message.update": { "payload": { "message_id": { "type": "string" }, "text": { "type": "string", "description": "Full current text (app replaces)." } } }, - "message.stop": { "payload": { "message_id": { "type": "string" }, "final_text": { "type": "string" }, "reasoning": { "type": "string" }, "model": { "type": "string" }, "tokens": { "type": "integer" }, "ts": { "type": "integer" } } }, + "message.stop": { "payload": { "message_id": { "type": "string" }, "final_text": { "type": "string" }, "reasoning": { "type": "string" }, "model": { "type": "string" }, "tokens": { "type": "integer" }, "runtime": { "$ref": "#/definitions/runtime" }, "ts": { "type": "integer" } } }, "message.deleted": { "description": "The given message(s) were deleted from a chat/thread. Response to a message.delete request (id set) and broadcast to every device so all drop them from their cache; also outboxed so an offline device learns of the deletion on its next sync.", "payload": { "message_ids": { "type": "array", "items": { "type": "string" } } } }, "commentary": { "description": "Intermediate assistant beat.", "payload": { "message_id": { "type": "string" }, "text": { "type": "string" } } }, "tool.start": { "payload": { "index": { "type": "integer" }, "name": { "type": "string" }, "preview": { "type": "string" }, "args": { "type": "object" } } }, @@ -62,7 +63,7 @@ "error": { "payload": { "code": { "type": "string", "enum": ["auth", "not_found", "rate_limited", "media_too_large", "unsupported", "internal"] }, "message": { "type": "string" } } }, "pong": { "payload": { "ts": { "type": "integer" } } }, "sync.done": { "payload": { "cursor": { "type": "integer" } } }, - "history": { "description": "Paged full message history for a chat/thread (response to a history request). Reconstructed from the outbox log; used to populate the view on first open / after a process death, since sync only replays the outbox delta.", "payload": { "messages": { "type": "array", "items": { "type": "object", "properties": { "message_id": {"type":"string"}, "role": {"type":"string","enum":["user","assistant"]}, "text": {"type":"string"}, "reasoning": {"type":"string"}, "model": {"type":"string"}, "tokens": {"type":"integer"}, "ts": {"type":"integer"}, "media": {"type":"array","items":{"$ref":"#/definitions/media_ref"}} } } }, "has_more": { "type": "boolean", "description": "True when older pages exist." }, "oldest_message_id": { "type": "string", "description": "before_message_id for the next (older) page." } } }, + "history": { "description": "Paged full message history for a chat/thread (response to a history request). Reconstructed from the outbox log; used to populate the view on first open / after a process death, since sync only replays the outbox delta.", "payload": { "messages": { "type": "array", "items": { "type": "object", "properties": { "message_id": {"type":"string"}, "role": {"type":"string","enum":["user","assistant"]}, "text": {"type":"string"}, "reasoning": {"type":"string"}, "model": {"type":"string"}, "tokens": {"type":"integer"}, "runtime": {"$ref":"#/definitions/runtime"}, "ts": {"type":"integer"}, "media": {"type":"array","items":{"$ref":"#/definitions/media_ref"}} } } }, "has_more": { "type": "boolean", "description": "True when older pages exist." }, "oldest_message_id": { "type": "string", "description": "before_message_id for the next (older) page." } } }, "media.pull.end": { "payload": { "ok": { "type": "boolean" } } }, "media.upload.ack": { "description": "Response to media.upload.end; ref is cached and usable in message.send media_refs.", "payload": { "ok": { "type": "boolean" }, "media_ref": { "type": "string" } } }, "commands.catalog": { "description": "Response to a commands.catalog request: the gateway's slash-command catalog for the app's '/' drawer. Derived from hermes' COMMAND_REGISTRY (gateway-available subset) plus plugin-registered commands. The app fuzzy-matches the typed prefix client-side.", "payload": { "commands": { "type": "array", "items": { "type": "object", "properties": { "name": { "type": "string", "description": "Canonical command with leading slash, e.g. \"/new\"." }, "description": { "type": "string" }, "args_hint": { "type": "string", "description": "Argument placeholder, e.g. \"[name]\"; empty when none." }, "category": { "type": "string", "description": "Registry category (Session, Configuration, Tools & Skills, Info, Exit, Plugin)." }, "aliases": { "type": "array", "items": { "type": "string" }, "description": "Alternative names with leading slash, e.g. [\"/reset\"] for /new." } } } } } } @@ -93,7 +94,8 @@ "definitions": { "kind": { "type": "string", "enum": ["image", "audio", "video", "document", "voice"] }, "channel": { "type": "object", "properties": { "chat_id": {"type":"string"}, "name": {"type":"string"}, "kind": {"type":"string","enum":["default","channel","thread"]}, "parent_chat_id": {"type":["string","null"]}, "is_default": {"type":"boolean"}, "archived": {"type":"boolean"}, "auto": {"type":"boolean","description":"Optional; true on channel.created for a gateway-minted auto-thread."}, "favorite": {"type":"boolean","description":"Optional; cosmetic favorite flag (sorts to the top of the list)."}, "icon": {"type":["string","null"],"description":"Optional; cosmetic icon, a base64-encoded image (PNG/JPEG). Absent/null = auto-generated letter avatar."}, "color": {"type":["string","null"],"description":"Optional; cosmetic avatar color override (#RRGGBB). Absent/null = auto-generated name-hash color."}, "automation": {"type":"boolean","description":"Optional; true when the channel is an automation channel (read-only for the user; only receives gateway-originated output such as cron jobs and webhooks). The app hides the composer and the gateway rejects message.send into it. Never set on the default channel."} } }, - "media_ref": { "type": "object", "properties": { "media_id": {"type":"string"}, "kind": { "$ref": "#/definitions/kind" }, "mime": {"type":"string"}, "size": {"type":"integer"}, "filename": {"type":"string"}, "message_id": {"type":"string","description":"Optional; set on media.offer to associate the offer with the assistant message it belongs to."} } } + "media_ref": { "type": "object", "properties": { "media_id": {"type":"string"}, "kind": { "$ref": "#/definitions/kind" }, "mime": {"type":"string"}, "size": {"type":"integer"}, "filename": {"type":"string"}, "message_id": {"type":"string","description":"Optional; set on media.offer to associate the offer with the assistant message it belongs to."} } }, + "runtime": { "type": "object", "description": "Structured runtime-metadata footer (app-controlled display). The gateway ALWAYS sends it on final assistant messages; whether/what is shown is a per-app setting (Settings -> Runtime footer), NOT a hermes config. All keys optional; absent when the data is unavailable (e.g. local models have no cost).", "properties": { "model": {"type":"string","description":"Bare model id, vendor prefix dropped (gpt-5.4)."}, "context_pct": {"type":"integer","description":"Last-call context occupancy, 0-100."}, "cwd": {"type":"string","description":"Home-relative working dir (~)."}, "latency": {"type":"number","description":"Wall-clock turn duration, seconds."}, "cost": {"type":"number","description":"Turn cost, USD."} } } }, "x-planned-frames": [ { "name": "picker.model", "direction": "server_to_app", "note": "Model/provider picker prompt. Planned, not implemented." }, diff --git a/gateway-plugin/adapter.py b/gateway-plugin/adapter.py index 7bc4bd0..5115d48 100644 --- a/gateway-plugin/adapter.py +++ b/gateway-plugin/adapter.py @@ -317,6 +317,132 @@ def _tool_end_fields(tool_name: str) -> dict[str, Any]: return fields +# --------------------------------------------------------------------------- +# Runtime-metadata footer (post_api_request hook) +# +# The app renders a Telegram-style footer under final assistant messages +# (model, context %, cwd, latency, cost). Display is controlled by the APP +# (Settings → Runtime footer), not hermes config — so the gateway ALWAYS +# sends the data. hermes core only appends its own *text* footer when +# ``display.runtime_footer.enabled`` is set, and the adapter has no access to +# the gateway's ``agent_result``, so we capture the same facts ourselves via +# the ``post_api_request`` plugin hook (fires after every provider call with +# model + usage): +# +# * model — the turn's latest model (failover-aware) +# * prompt_tokens — the latest call's prompt size (context occupancy) +# * turn start — the first API call of the turn (latency baseline) +# +# Global buffer (same pattern as the reasoning/tool buffers): a personal +# android gateway serves one active turn at a time. The hook fires for every +# platform, so we only record when the turn's platform is android. +# --------------------------------------------------------------------------- + +_runtime_meta: dict[str, Any] = {} +_runtime_meta_lock = threading.Lock() +# Per-model context-window cache. Resolution may probe endpoints on first use +# (slow); the cache is process-lifetime so each model resolves at most once. +_context_length_cache: dict[str, int] = {} +# Upper bound (seconds) on context-window resolution during a final send, so a +# slow first-use probe never delays the reply. The worker thread keeps running +# and populates the cache, so the next turn is fast. +_CTX_RESOLVE_TIMEOUT_S = 3.0 + + +def _on_post_api_request(**kwargs: Any) -> None: + """Plugin hook: capture per-turn runtime metadata (model, prompt tokens).""" + platform = kwargs.get("platform") + if platform and platform != "android": + return + model = kwargs.get("model") or "" + usage = kwargs.get("usage") or {} + prompt_tokens = usage.get("prompt_tokens") or 0 + with _runtime_meta_lock: + if model: + _runtime_meta["model"] = model + if prompt_tokens: + _runtime_meta["prompt_tokens"] = prompt_tokens + if "turn_start" not in _runtime_meta: + _runtime_meta["turn_start"] = time.monotonic() + + +def _take_runtime_meta() -> dict[str, Any]: + """Drain the captured turn metadata (turn boundary).""" + with _runtime_meta_lock: + meta = dict(_runtime_meta) + _runtime_meta.clear() + return meta + + +def _resolve_context_length(model: str) -> int | None: + """Best-effort context window for *model* (cached; None on failure). + + Runs in a worker thread (may probe endpoints on first use). The cache is + populated even if the caller's asyncio task times out, so subsequent + turns resolve instantly. + """ + if not model: + return None + cached = _context_length_cache.get(model) + if cached: + return cached + try: + from agent.model_metadata import get_model_context_length + + ctx = get_model_context_length(model) + if ctx and ctx > 0: + _context_length_cache[model] = int(ctx) + return int(ctx) + except Exception: + logger.debug("android: context-length resolution failed for %s", model, exc_info=True) + return None + + +def _home_relative_cwd(cwd: str) -> str: + """Collapse ``$HOME`` to ``~`` (matches hermes' runtime footer).""" + if not cwd: + return "" + try: + home = os.path.expanduser("~") + p = os.path.abspath(cwd) + if home and (p == home or p.startswith(home + os.sep)): + return "~" + p[len(home) :] + return p + except Exception: + return cwd + + +async def _build_runtime_footer(meta: dict[str, Any]) -> dict[str, Any]: + """Build the ``runtime`` footer object from captured turn metadata. + + Called on every final send (the app decides what to show). Fields without + data are omitted. ``meta`` is the drained turn buffer (model, + prompt_tokens, turn_start). + """ + model = (meta.get("model") or "").rsplit("/", 1)[-1] + prompt_tokens = meta.get("prompt_tokens") or 0 + context_pct = None + if prompt_tokens and model: + try: + ctx_len = await asyncio.wait_for( + asyncio.to_thread(_resolve_context_length, model), + timeout=_CTX_RESOLVE_TIMEOUT_S, + ) + except (asyncio.TimeoutError, Exception): + ctx_len = None + if ctx_len: + context_pct = round(prompt_tokens / ctx_len * 100) + turn_start = meta.get("turn_start") + latency = (time.monotonic() - turn_start) if turn_start else None + cwd = _home_relative_cwd(os.environ.get("TERMINAL_CWD", "")) + return protocol.runtime_footer( + model=model or None, + context_pct=context_pct, + cwd=cwd or None, + latency=latency, + ) + + # --------------------------------------------------------------------------- # Defaults # --------------------------------------------------------------------------- @@ -1159,7 +1285,7 @@ class AndroidAdapter(BasePlatformAdapter): state = self._turn_state(chat_id) # 1. Streaming segment start (stream consumer first send). - if meta.get("expect_edits") is True: + if bool(meta.get("expect_edits")): # A new content segment means the tool the model was waiting on # has returned -> close it before the segment opens. await self._close_open_tool(chat_id, state, thread_id) @@ -1213,7 +1339,7 @@ class AndroidAdapter(BasePlatformAdapter): return SendResult(success=True, message_id=message_id) # 3. Final message (non-streaming final, or streaming fallback final). - if meta.get("notify") is True: + if bool(meta.get("notify")): reasoning, body = _split_reasoning(content) # Non-streaming: reasoning is prepended to content (split above). # Streaming fallback: content has no reasoning, so use the @@ -1224,6 +1350,11 @@ class AndroidAdapter(BasePlatformAdapter): reasoning = _take_reasoning() or None else: _reset_reasoning() + # Runtime-metadata footer (app-controlled display): always attach + # the structured ``runtime`` object so the app can render its + # footer (Settings → Runtime footer). Independent of hermes' + # ``display.runtime_footer`` config. + runtime = await _build_runtime_footer(_take_runtime_meta()) if state.stream_id: # Fallback final: close the open streaming segment in place. message_id = state.stream_id @@ -1236,6 +1367,7 @@ class AndroidAdapter(BasePlatformAdapter): body, reasoning=reasoning, thread_id=thread_id, + runtime=runtime or None, ts=int(time.time() * 1000), ), ) @@ -1251,6 +1383,7 @@ class AndroidAdapter(BasePlatformAdapter): thread_id=thread_id, reasoning=reasoning, reply_to=reply_to, + runtime=runtime or None, ts=int(time.time() * 1000), ), ) @@ -1310,6 +1443,8 @@ class AndroidAdapter(BasePlatformAdapter): reasoning = _take_reasoning() or None else: _reset_reasoning() + # Runtime-metadata footer (app-controlled display). + runtime = await _build_runtime_footer(_take_runtime_meta()) state.stream_id = None await self._broadcast_or_log( chat_id, @@ -1319,6 +1454,7 @@ class AndroidAdapter(BasePlatformAdapter): body, reasoning=reasoning, thread_id=thread_id, + runtime=runtime or None, ts=int(time.time() * 1000), ), ) @@ -1344,6 +1480,7 @@ class AndroidAdapter(BasePlatformAdapter): # Unknown id: treat as a streaming update (best effort). if finalize: + runtime = await _build_runtime_footer(_take_runtime_meta()) await self._broadcast_or_log( chat_id, protocol.message_stop( @@ -1351,6 +1488,7 @@ class AndroidAdapter(BasePlatformAdapter): message_id, _strip_streaming_cursor(content), thread_id=thread_id, + runtime=runtime or None, ts=int(time.time() * 1000), ), ) @@ -1857,7 +1995,7 @@ class AndroidAdapter(BasePlatformAdapter): # Skipped for slash commands (session-scoped, not conversation # starters) and replies (they continue where the user is). Threading # is only active on the default channel; other channels stay flat. - auto_thread = payload.get("auto_thread") is True + auto_thread = bool(payload.get("auto_thread")) default_entry = self._channels.default() if ( auto_thread @@ -2825,6 +2963,15 @@ def register(ctx): ctx.register_hook("post_tool_call", _on_post_tool_call) except Exception: logger.debug("android: post_tool_call hook registration failed", exc_info=True) + # Runtime-metadata footer: capture the turn's model + prompt tokens (per + # provider call) so the final message can carry a structured ``runtime`` + # object. The app decides whether/what to show (Settings → Runtime + # footer); the gateway always sends the data (independent of hermes + # ``display.runtime_footer`` config). + try: + ctx.register_hook("post_api_request", _on_post_api_request) + except Exception: + logger.debug("android: post_api_request hook registration failed", exc_info=True) ctx.register_platform( name="android", label="Android", diff --git a/gateway-plugin/outbox.py b/gateway-plugin/outbox.py index e947d95..0e2c252 100644 --- a/gateway-plugin/outbox.py +++ b/gateway-plugin/outbox.py @@ -216,6 +216,7 @@ class Outbox: "reasoning": payload.get("reasoning"), "model": payload.get("model"), "tokens": payload.get("tokens"), + "runtime": payload.get("runtime"), "ts": payload.get("ts"), "media": payload.get("media"), } @@ -230,6 +231,7 @@ class Outbox: "reasoning": payload.get("reasoning"), "model": payload.get("model"), "tokens": payload.get("tokens"), + "runtime": payload.get("runtime"), "ts": payload.get("ts"), "media": None, } @@ -259,7 +261,7 @@ class Outbox: # Omit absent optional fields (the app's serializer treats a # missing key as its default, but a JSON ``null`` for a # non-nullable field like ``media`` would fail to parse). - for key in ("reasoning", "model", "tokens", "ts", "media"): + for key in ("reasoning", "model", "tokens", "runtime", "ts", "media"): if m.get(key) is None: m.pop(key, None) return { @@ -310,12 +312,10 @@ class Outbox: cursors.append(int(r["cursor"])) if not cursors: return 0 - placeholders = ",".join("?" * len(cursors)) - sql = f"DELETE FROM outbox WHERE cursor IN ({placeholders})" - # Safe: ``placeholders`` is only ``?`` markers; every cursor value is - # bound as a parameter (no user data in the SQL text). - # pi-lens-ignore: python-sql-injection - self._conn.execute(sql, cursors) + # One bound-parameter delete per cursor (a message spans only a few + # frames); same transaction, no string-built SQL. + for cursor in cursors: + self._conn.execute("DELETE FROM outbox WHERE cursor = ?", (cursor,)) self._conn.commit() return len(cursors) diff --git a/gateway-plugin/protocol.py b/gateway-plugin/protocol.py index cf5b6bc..51cf1f6 100644 --- a/gateway-plugin/protocol.py +++ b/gateway-plugin/protocol.py @@ -249,6 +249,54 @@ def hello_ack( ) +# --------------------------------------------------------------------------- +# Runtime-metadata footer (app-controlled display) +# +# The gateway ALWAYS attaches a structured ``runtime`` object to final +# assistant messages so the app can render a Telegram-style footer (model, +# context %, cwd, latency, cost). Whether/what is shown is a per-app setting, +# NOT a hermes config — the gateway sends the data unconditionally and the +# app decides. Mirrors hermes' ``gateway/runtime_footer.py`` fields but as +# structured data (the app formats + picks fields). +# +# Recognised keys (all optional; absent when the data is unavailable): +# model — bare model id, vendor prefix dropped (``gpt-5.4``) +# context_pct — last-call context occupancy, 0-100 (int) +# cwd — home-relative working dir (``~``) +# latency — wall-clock turn duration, seconds (float) +# cost — turn cost, USD (float); absent for local/free models +# --------------------------------------------------------------------------- + +RUNTIME_FIELDS: tuple[str, ...] = ("model", "context_pct", "cwd", "latency", "cost") + + +def runtime_footer( + *, + model: str | None = None, + context_pct: int | None = None, + cwd: str | None = None, + latency: float | None = None, + cost: float | None = None, +) -> dict[str, Any]: + """Build the structured ``runtime`` footer object. + + Only fields with data are included (a partially-populated footer is + better than empty slots). Returns ``{}`` when nothing is available. + """ + d: dict[str, Any] = {} + if model: + d["model"] = model + if context_pct is not None: + d["context_pct"] = max(0, min(100, int(context_pct))) + if cwd: + d["cwd"] = cwd + if latency is not None and latency >= 0: + d["latency"] = round(latency, 3) + if cost is not None and cost > 0: + d["cost"] = round(cost, 6) + return d + + # Frame builder mirrors the wire schema (docs/04); the many fields are the # message's full shape, so the arg count is intentional. def message( # noqa: PLR0913 @@ -263,6 +311,7 @@ def message( # noqa: PLR0913 reply_to: str | None = None, model: str | None = None, tokens: int | None = None, + runtime: dict[str, Any] | None = None, ts: int | None = None, ) -> Frame: payload: dict[str, Any] = { @@ -280,11 +329,11 @@ def message( # noqa: PLR0913 payload["model"] = model if tokens is not None: payload["tokens"] = tokens + if runtime: + payload["runtime"] = runtime if ts is not None: payload["ts"] = ts - return Frame( - type=TYPE_MESSAGE, chat_id=chat_id, thread_id=thread_id, payload=payload - ) + return Frame(type=TYPE_MESSAGE, chat_id=chat_id, thread_id=thread_id, payload=payload) def typing(chat_id: str, on: bool = True, *, thread_id: str | None = None) -> Frame: @@ -342,6 +391,7 @@ def message_stop( reasoning: str | None = None, model: str | None = None, tokens: int | None = None, + runtime: dict[str, Any] | None = None, ts: int | None = None, ) -> Frame: """Finalize a streaming bubble.""" @@ -355,6 +405,8 @@ def message_stop( payload["model"] = model if tokens is not None: payload["tokens"] = tokens + if runtime: + payload["runtime"] = runtime if ts is not None: payload["ts"] = ts return Frame( @@ -648,9 +700,7 @@ def notification( payload: dict[str, Any] = {"kind": kind, "title": title, "body": body} if ts is not None: payload["ts"] = ts - return Frame( - type=TYPE_NOTIFICATION, chat_id=chat_id, thread_id=thread_id, payload=payload - ) + return Frame(type=TYPE_NOTIFICATION, chat_id=chat_id, thread_id=thread_id, payload=payload) def fcm_register(fcm_token: str | None = None, ntfy_topic: str | None = None) -> Frame: @@ -713,9 +763,7 @@ def media_offer( } if message_id: payload["message_id"] = message_id - return Frame( - type=TYPE_MEDIA_OFFER, chat_id=chat_id, thread_id=thread_id, payload=payload - ) + return Frame(type=TYPE_MEDIA_OFFER, chat_id=chat_id, thread_id=thread_id, payload=payload) def media_pull_end(ok: bool, *, id: int | None = None) -> Frame: