diff --git a/CHANGELOG.md b/CHANGELOG.md
index 8fb1c80c83..3ad7827860 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -24,6 +24,11 @@
- **What:** the Sessions item moved from above the Monitoring label to directly under Agents (Home, Agents, Sessions, Activity, Cost, Models, Context usage). It is still the page the dashboard opens on and keeps the default highlight; the tab id and deep links are unchanged.
- **Verified:** `tests/test_beginner_nav_phase_a.py` and `tests/test_trail_tab_template.py` pin the new order.
+### Fixed: Home showed other runtimes' cards under the runtime switcher (2026-09-15)
+- **Why:** on app.clawmetry.com with Codex selected, Home still listed OpenClaw's "Gateway :18789" in System Health (the hosted snapshot names it plain "Gateway", which the OpenClaw filter missed), Run Health drew a `claude_code` row, and 30-day activity, "How independent is your agent?", Anomaly Detection, Reliability and the session quality tile counted every runtime on the machine. The snapshot's Run Health slice had no Codex row at all: the newest 60 sessions on that machine were all Claude Code.
+- **What:** each of those cards now asks for the selected runtime (`?runtime=` on `/api/activity-heatmap`, `/api/health-timeline`, `/api/autonomy`, `/api/evals/summary`, each echoing the runtime it counted) and hides rather than show a whole-machine answer under one runtime's name. Anomaly Detection keeps only that runtime's sessions; Reliability, which has no per-runtime form, is hidden under a runtime. The daemon adds each runtime's own recent sessions to the Run Health slice and ships `autonomyByRuntime` (both reused for a few minutes, so a sync cycle does not re-read them). A runtime switch reloads Home immediately instead of waiting for the next refresh.
+- **Verified:** `tests/test_overview_runtime_scope.py` (16 tests, all red on the previous code). The hosted chip row, gateway pill and autonomy/live-activity overrides are fixed in clawmetry-cloud.
+
### Release: enterprise readiness, batch 3 (2026-09-15)
- **Carries:** #5950 (fleet install for shared hosts and virtual desktops, refs #5942), #5965 (LiteLLM gateway spend by team, person and key, refs #5940), #5957 (the dashboard's first load no longer times out its own requests, refs #5935), #5996 (the hosted Cost Optimizer shows evidence-backed experiments, refs #5934; hosted rendering lands with clawmetry-cloud#2450 after this pin) #5967 (SECURITY.md: the DPA is not available and the sub-processor list is published; documentation only) and #6007 (Compliance tab shell, refs clawmetry-pro#250; the evaluation ships in clawmetry-pro 0.7.29). Any other change merged before this release carries its own entry below. Their entries follow.
diff --git a/clawmetry/local_store.py b/clawmetry/local_store.py
index 24082ab03a..958680ef5c 100644
--- a/clawmetry/local_store.py
+++ b/clawmetry/local_store.py
@@ -16502,6 +16502,7 @@ def query_eval_summary(
self,
*,
window_hours: int = 24,
+ runtime: str | None = None,
) -> dict[str, Any]:
"""Aggregate scores over the recent window. Drives
``/api/evals/summary``.
@@ -16510,24 +16511,40 @@ def query_eval_summary(
``total`` is sessions touched in the window (scored OR not);
``scored`` is the subset with a numeric eval_score. The ratio
``scored/total`` surfaces coverage on the overview tile.
+
+ ``runtime`` keeps only that runtime's sessions, bucketed by id prefix
+ like :func:`_runtime_of_session_id` (NemoClaw sessions carry OpenClaw
+ ids). ``None`` / ``"all"`` counts every runtime.
"""
try:
from datetime import datetime, timedelta, timezone
cutoff = (datetime.now(timezone.utc) - timedelta(hours=int(window_hours))).isoformat()
except Exception:
cutoff = ""
+ rt_sql = ""
+ rt_params: list[Any] = []
+ rt = str(runtime or "").strip().lower()
+ if rt and rt != "all":
+ placeholders = ", ".join(["?"] * len(_NON_OPENCLAW_RUNTIME_PREFIXES))
+ rt_sql = (
+ f" AND (CASE WHEN split_part(session_id, ':', 1) IN ({placeholders})"
+ f" THEN split_part(session_id, ':', 1) ELSE 'openclaw' END) = ?"
+ )
+ rt_params = list(_NON_OPENCLAW_RUNTIME_PREFIXES) + [
+ "openclaw" if rt == "nemoclaw" else rt
+ ]
# Two queries — one for totals (scored + un-scored), one for the
# quantile/avg over the scored subset. Keeps the SQL readable
# without a CTE that would have to handle NULLs in two places.
try:
total_row = self._fetch(
- """
+ f"""
SELECT COUNT(*) AS total,
COUNT(eval_score) AS scored
FROM sessions
- WHERE (? = '' OR COALESCE(last_active_at, started_at, '') >= ?)
+ WHERE (? = '' OR COALESCE(last_active_at, started_at, '') >= ?){rt_sql}
""",
- [cutoff, cutoff],
+ [cutoff, cutoff] + rt_params,
)
except Exception as e:
log.warning("local store: eval summary totals failed: %s", e)
@@ -16541,15 +16558,15 @@ def query_eval_summary(
if scored > 0:
try:
stats = self._fetch(
- """
+ f"""
SELECT AVG(eval_score) AS avg_score,
quantile_cont(eval_score, 0.5) AS p50,
quantile_cont(eval_score, 0.1) AS p10
FROM sessions
WHERE eval_score IS NOT NULL
- AND (? = '' OR COALESCE(last_active_at, started_at, '') >= ?)
+ AND (? = '' OR COALESCE(last_active_at, started_at, '') >= ?){rt_sql}
""",
- [cutoff, cutoff],
+ [cutoff, cutoff] + rt_params,
)
if stats:
avg = float(stats[0][0] or 0.0)
diff --git a/clawmetry/static/js/app.js b/clawmetry/static/js/app.js
index 3991a96247..b45c9e4994 100644
--- a/clawmetry/static/js/app.js
+++ b/clawmetry/static/js/app.js
@@ -1425,6 +1425,18 @@ async function loadAnomalyPanel() {
if (!panel) return;
var anomalies = data.anomalies || [];
var baselines = data.baselines || {};
+ // Under a selected runtime keep only that runtime's sessions. Node-wide
+ // aggregate rows (session_key "__error_rate__") and the node-wide
+ // baselines are not about this runtime, so they are left out.
+ var _anRt = (typeof _cmRuntimeFilter === 'function') ? _cmRuntimeFilter() : 'all';
+ var _anScoped = !!_anRt && _anRt !== 'all';
+ if (_anScoped) {
+ anomalies = anomalies.filter(function(a){
+ var sk = String(a && a.session_key || '');
+ return !!sk && sk.indexOf('__') !== 0 && _cmRuntimeOf({session_id: sk}) === _anRt;
+ });
+ baselines = {};
+ }
var active = anomalies.filter(function(a){ return !a.acknowledged; });
// Badge
@@ -1451,7 +1463,7 @@ async function loadAnomalyPanel() {
if (baselines.baseline_cost_7d > 0) blHtml += 'Avg cost: $' + Number(baselines.baseline_cost_7d).toFixed(4) + '/session';
if (baselines.baseline_tokens_7d > 0) blHtml += 'Avg tokens: ' + Math.round(baselines.baseline_tokens_7d).toLocaleString() + '/session';
if (baselines.baseline_sessions_per_day_7d > 0) blHtml += 'Sessions/day: ' + Number(baselines.baseline_sessions_per_day_7d).toFixed(1) + '';
- blEl.innerHTML = blHtml || 'Collecting baseline data...';
+ blEl.innerHTML = blHtml || (_anScoped ? '' : 'Collecting baseline data...');
}
// Anomaly list
@@ -2738,6 +2750,14 @@ async function loadReliabilityCard() {
var detEl = document.getElementById('reliability-detail-lt');
var iconEl = document.getElementById('reliability-icon-lt');
if (!dirEl) return;
+ // The trend is built from this machine's daemon heartbeats plus every
+ // runtime's error events; it has no per-runtime form, so it is not shown
+ // under a selected runtime.
+ var _relRt = (typeof _cmRuntimeFilter === 'function') ? _cmRuntimeFilter() : 'all';
+ var _relScoped = !!_relRt && _relRt !== 'all';
+ var _relCard = document.getElementById('reliability-card-lt');
+ if (_relCard) _relCard.style.display = _relScoped ? 'none' : '';
+ if (_relScoped) return;
try {
var d = await fetchJsonWithTimeout('/api/reliability', 5000);
d = d || {};
@@ -2793,9 +2813,12 @@ async function loadAutonomy() {
}
try {
+ // Check-in gaps are per runtime: Codex's cadence is not Claude Code's.
+ var _auRt = (typeof _cmRuntimeFilter === 'function') ? _cmRuntimeFilter() : 'all';
+ var _auUrl = '/api/autonomy' + ((_auRt && _auRt !== 'all') ? '?runtime=' + encodeURIComponent(_auRt) : '');
var d = await (typeof fetchJsonWithTimeout === 'function'
- ? fetchJsonWithTimeout('/api/autonomy', 5000)
- : fetch('/api/autonomy').then(function(r){return r.json();}));
+ ? fetchJsonWithTimeout(_auUrl, 5000)
+ : fetch(_auUrl).then(function(r){return r.json();}));
if (d.score == null) {
labelEl.textContent = t("app.just_getting_started", null, "Just getting started");
@@ -3963,13 +3986,23 @@ async function loadHealthTimeline() {
var card = document.getElementById('health-timeline-card');
var body = document.getElementById('health-timeline-body');
if (!card || !body) return;
+ var rt = (typeof _cmRuntimeFilter === 'function') ? _cmRuntimeFilter() : 'all';
+ var scoped = !!rt && rt !== 'all';
var data;
try {
- var resp = await fetch('/api/health-timeline');
+ var resp = await fetch('/api/health-timeline' + (scoped ? '?runtime=' + encodeURIComponent(rt) : ''));
if (!resp.ok) { card.style.display = 'none'; return; }
data = await resp.json();
} catch (e) { card.style.display = 'none'; return; }
var runtimes = (data && data.runtimes) || [];
+ // Under a selected runtime only its own row renders: an older server and
+ // the hosted snapshot both answer with every runtime they know. NemoClaw
+ // runs the OpenClaw adapter, so its sessions bucket as openclaw.
+ if (scoped) {
+ runtimes = runtimes.filter(function (r) {
+ return r && (r.runtime === rt || (rt === 'nemoclaw' && r.runtime === 'openclaw'));
+ });
+ }
if (!runtimes.length || !runtimes.some(function(r){ return (r.dots||[]).length; })) {
card.style.display = 'none';
return;
@@ -6805,8 +6838,11 @@ async function loadEvalSummary() {
function setTitleCheck(show) { if (checkEl) checkEl.style.display = show ? '' : 'none'; }
if (!avgEl) return;
try {
- var data = await fetch('/api/evals/summary?window=24h').then(function(r){return r.json();}).catch(function(){return null;});
- if (!data || typeof data.scored !== 'number') {
+ var _evRt = (typeof _cmRuntimeFilter === 'function') ? _cmRuntimeFilter() : 'all';
+ var _evQ = (_evRt && _evRt !== 'all') ? '&runtime=' + encodeURIComponent(_evRt) : '';
+ var data = await fetch('/api/evals/summary?window=24h' + _evQ).then(function(r){return r.json();}).catch(function(){return null;});
+ // A server that ignores ?runtime answers for the whole node.
+ if (!data || typeof data.scored !== 'number' || (_evQ && data.runtime !== _evRt)) {
setTitleCheck(false);
avgEl.textContent = '--';
if (covEl) covEl.textContent = '';
@@ -12883,6 +12919,9 @@ function _cmApplyRuntimeSelection(val) {
// Swap the Flow + Overview diagram to the selected runtime's topology.
try { if (typeof _applyRuntimeFlowDiagram === 'function') _applyRuntimeFlowDiagram(val); } catch (e) {}
// Reload the current tab so any runtime-aware view re-filters in place.
+ // loadAll coalesces calls 2 s apart; a switch must not be swallowed by that,
+ // or the Overview keeps the previous runtime's cards until the next refresh.
+ try { _loadAllLastFinishedMs = 0; } catch (e) {}
if (typeof switchTab === 'function' && _cmCurrentTab) switchTab(_cmCurrentTab);
// System Health refreshes on a 30s timer and is not part of loadAll, so
// re-scope it now or the previous runtime's checks linger.
@@ -17173,7 +17212,13 @@ async function loadSystemHealth() {
}
}
var services = Array.isArray(d.services) ? d.services : [];
- if (!isOc) services = services.filter(function (s) { return !/openclaw/i.test(String(s && s.name || '')); });
+ // OpenClaw's gateway arrives as "OpenClaw Gateway" locally and as a bare
+ // "Gateway" from the hosted snapshot; both, and anything on its port,
+ // belong to OpenClaw alone.
+ if (!isOc) services = services.filter(function (s) {
+ var name = String(s && s.name || '').trim();
+ return !(/openclaw/i.test(name) || /^gateway$/i.test(name) || Number(s && s.port) === 18789);
+ });
var channels = (scope.has('CHANNELS') && Array.isArray(d.channels)) ? d.channels : [];
var disks = Array.isArray(d.disks) ? d.disks : [];
var crons = (d.crons && typeof d.crons === 'object') ? d.crons : {enabled: 0, ok24h: 0, failed: []};
@@ -17884,9 +17929,13 @@ async function loadActivityHeatmap() {
var grid = document.getElementById('activity-heatmap-grid');
if (!card || !grid) return;
var data;
- try { data = await fetchJsonWithTimeout('/api/activity-heatmap', 5000); } catch(e) { return; }
+ var rt = (typeof _cmRuntimeFilter === 'function') ? _cmRuntimeFilter() : 'all';
+ var q = (rt && rt !== 'all') ? ('?runtime=' + encodeURIComponent(rt)) : '';
+ try { data = await fetchJsonWithTimeout('/api/activity-heatmap' + q, 5000); } catch(e) { card.style.display = 'none'; return; }
var days = (data && data.days) || [];
- if (!days.length) return;
+ // A server that ignores ?runtime answers for the whole node; hide the card
+ // rather than draw every runtime's days under this one's name.
+ if (!days.length || (q && data.runtime !== rt)) { card.style.display = 'none'; return; }
var maxSessions = Math.max.apply(null, days.map(function(d){ return d.sessions || 0; }));
var shades = ['#12122a','#1a3a2a','#2a6a3a','#4a9a2a','#6adb3a'];
var html = '';
diff --git a/clawmetry/sync.py b/clawmetry/sync.py
index 4384e255b5..6fca097da5 100644
--- a/clawmetry/sync.py
+++ b/clawmetry/sync.py
@@ -17797,18 +17797,19 @@ def _build_transcripts(limit_sessions=8, msg_cap=80, extra_sids=None):
return {}
-def _build_autonomy_snapshot():
+def _build_autonomy_snapshot(runtime=None):
"""Autonomy block for the cloud snapshot (same shape as /api/autonomy).
Trial-bug fix: the Overview "How independent is your agent?" card fetches
/api/autonomy, which is empty on the hosted dashboard (no DuckDB) because no
snapshot slice carried it -> the card was stuck on "Just getting started".
Reuses the store-backed compute from routes.autonomy (best-effort -> empty).
+ ``runtime`` computes one runtime's cadence for ``autonomyByRuntime``.
"""
try:
from routes.autonomy import _try_local_store_autonomy, _empty_response
try:
- r = _try_local_store_autonomy()
+ r = _try_local_store_autonomy(runtime=runtime) if runtime else _try_local_store_autonomy()
except Exception:
r = None
return r if r is not None else _empty_response()
@@ -17816,6 +17817,34 @@ def _build_autonomy_snapshot():
return {}
+_AUTONOMY_BY_RT_TTL_S = 300.0
+_autonomy_by_rt_memo: dict = {"ts": 0.0, "keys": None, "value": {}}
+
+
+def _build_autonomy_by_runtime(runtime_keys):
+ """``{runtime: autonomy}`` for the hosted Overview card under the switcher.
+
+ Three event scans per runtime is too much to repeat every sync cycle, and a
+ 7-day check-in cadence does not move in seconds, so the result is reused
+ for five minutes unless the set of runtimes changes.
+ """
+ keys = tuple(sorted(str(k) for k in (runtime_keys or []) if k))
+ now = time.time()
+ memo = _autonomy_by_rt_memo
+ if memo["keys"] == keys and (now - memo["ts"]) < _AUTONOMY_BY_RT_TTL_S:
+ return memo["value"]
+ out: dict = {}
+ for rt in keys:
+ try:
+ block = _build_autonomy_snapshot(runtime=rt)
+ except Exception:
+ continue
+ if block:
+ out[rt] = block
+ memo.update(ts=now, keys=keys, value=out)
+ return out
+
+
def _build_activity_heatmap_snapshot():
"""30-day (day x hour) activity grid, node-wide and per runtime.
@@ -18387,8 +18416,12 @@ def _build_waste_flags(session_limit: int = 60, events_per_session: int = 500):
return out
+_HEALTH_TIMELINE_TTL_S = 120.0
+_health_timeline_memo: dict = {"ts": 0.0, "key": None, "value": None}
+
+
def _build_health_timeline(session_limit: int = 60, events_per_session: int = 500,
- dots_per_runtime: int = 30):
+ dots_per_runtime: int = 30, per_runtime_floor: int = 8):
"""Per-runtime sparkline of recent runs (#2196 item #4).
Returns ``{"runtimes": [{"runtime": str, "dots": [...]}, …]}`` where each
@@ -18400,6 +18433,13 @@ def _build_health_timeline(session_limit: int = 60, events_per_session: int = 50
Daemon-only (uses the writer-owned store handle). Best-effort + bounded
so a busy store can't bloat the encrypted snapshot.
+
+ The newest ``session_limit`` sessions node-wide can all belong to one loud
+ runtime: a node with 1,700 Claude Code sessions shipped only
+ ``claude_code``, so hosted Run Health had no row for Codex. Each runtime's
+ own ``per_runtime_floor`` most recent sessions are added from the fair
+ per-runtime query. The result is reused for ``_HEALTH_TIMELINE_TTL_S``
+ because every added session costs one events read.
"""
try:
from clawmetry import local_store as _ls
@@ -18410,10 +18450,25 @@ def _build_health_timeline(session_limit: int = 60, events_per_session: int = 50
store = _ls.get_store()
if store is None:
return {"runtimes": []}
- sessions = store.query_sessions(limit=int(session_limit))
+ memo_key = (id(store), session_limit, events_per_session, dots_per_runtime, per_runtime_floor)
+ memo = _health_timeline_memo
+ if memo["key"] == memo_key and (time.time() - memo["ts"]) < _HEALTH_TIMELINE_TTL_S:
+ return memo["value"]
+ sessions = list(store.query_sessions(limit=int(session_limit)) or [])
except Exception:
return {"runtimes": []}
+ seen = {str(s.get("session_id")) for s in sessions if s.get("session_id")}
+ try:
+ fair = store.query_recent_sessions_by_runtime(per_runtime=int(per_runtime_floor)) or []
+ except Exception:
+ fair = []
+ for row in fair:
+ fsid = str((row or {}).get("session_id") or "")
+ if fsid and fsid not in seen:
+ seen.add(fsid)
+ sessions.append({"session_id": fsid})
+
buckets: dict[str, list] = {}
for s in sessions:
sid = s.get("session_id")
@@ -18423,6 +18478,17 @@ def _build_health_timeline(session_limit: int = 60, events_per_session: int = 50
events = store.query_events(session_id=sid, limit=int(events_per_session))
except Exception:
continue
+ # A session added by the per-runtime query has no rollup row: take its
+ # start, end and spend from the events just read.
+ if s.get("started_at") is None and events:
+ stamps = sorted(str(e.get("ts")) for e in events if e.get("ts"))
+ if stamps:
+ s = dict(s, started_at=stamps[0], updated_at=stamps[-1])
+ if s.get("cost_usd") is None and events:
+ try:
+ s = dict(s, cost_usd=sum(float(e.get("cost_usd") or 0.0) for e in events))
+ except (TypeError, ValueError):
+ pass
try:
signals = _wf.compute_signals_from_events(events)
flags = _wf.compute_flags(signals)
@@ -18460,7 +18526,9 @@ def _build_health_timeline(session_limit: int = 60, events_per_session: int = 50
),
reverse=True,
)
- return {"runtimes": runtimes_out}
+ result = {"runtimes": runtimes_out}
+ _health_timeline_memo.update(ts=time.time(), key=memo_key, value=result)
+ return result
def _build_resolved_errors(limit: int = 5000):
@@ -23872,6 +23940,13 @@ def sync_system_snapshot(config: dict, state: dict, paths: dict) -> int:
continue
except Exception as _e_rtb:
log.debug("snapshot: per-runtime breakdown failed: %s", _e_rtb)
+ try:
+ _autonomy_by_rt = _build_autonomy_by_runtime(
+ list(_runtime_summary.keys()) if isinstance(_runtime_summary, dict) else []
+ )
+ except Exception as _e_aut:
+ _autonomy_by_rt = {}
+ log.debug("snapshot: per-runtime autonomy failed: %s", _e_aut)
# Agent Inventory roster (single-pane control-tower view). Built ENTIRELY
# from rollups already computed above (runtime_summary / outcomes / activity
@@ -24128,6 +24203,9 @@ def _guard_call(method, **kw):
# surface as sp.cohortSuggested (WO-60).
"cohortSuggested": cohort_slice,
"autonomy": _build_autonomy_snapshot(),
+ # Per runtime, so the hosted "How independent is your agent?" card
+ # follows the switcher (cm-cloud autonomy interceptor reads ?runtime=).
+ "autonomyByRuntime": _autonomy_by_rt,
"flowRuns": _build_flow_runs_snapshot(),
"flowLanes": _build_flow_lanes_snapshot(),
"evals": evals_slice,
diff --git a/docs/ci_test_coverage_baseline.json b/docs/ci_test_coverage_baseline.json
index a53e9012f9..31480e3448 100644
--- a/docs/ci_test_coverage_baseline.json
+++ b/docs/ci_test_coverage_baseline.json
@@ -7,7 +7,7 @@
"Ratchet down by running --update-baseline after wiring new tests in.",
"Related: issue #5813"
],
- "total": 1188,
+ "total": 1189,
"listed": 275,
- "unlisted_max": 913
+ "unlisted_max": 914
}
diff --git a/routes/autonomy.py b/routes/autonomy.py
index 638d9eb49d..c11325fc43 100644
--- a/routes/autonomy.py
+++ b/routes/autonomy.py
@@ -18,13 +18,15 @@
from collections import defaultdict
from datetime import datetime, timezone
-from flask import Blueprint, jsonify
+from flask import Blueprint, jsonify, request
from clawmetry.config import is_local_store_read_enabled
bp_autonomy = Blueprint("autonomy", __name__)
_AUTONOMY_CACHE = {"ts": 0.0, "data": None}
_AUTONOMY_CACHE_TTL_SECONDS = 60
+# runtime -> (computed_at, payload) for /api/autonomy?runtime=.
+_AUTONOMY_RT_CACHE: dict = {}
# ---------------------------------------------------------------------------
@@ -275,9 +277,12 @@ def _empty_response() -> dict:
}
-def _try_local_store_autonomy() -> dict | None:
+def _try_local_store_autonomy(runtime: str | None = None) -> dict | None:
"""Tier-1 DuckDB fast path for /api/autonomy.
+ ``runtime`` reads only that runtime's user turns (the store's session-id
+ prefix filter), so each runtime's check-in cadence stands on its own.
+
Reads ``message`` events from the local store, filters to user-role
messages within the last 7 days, and runs the same per-session gap
aggregation the JSONL parser does. Returns the canonical autonomy
@@ -310,16 +315,18 @@ def _try_local_store_autonomy() -> dict | None:
# Overview "How independent is your agent?" widget rendered blank ``—``.
try:
from routes.local_query import local_store_via_daemon
+ # NemoClaw runs the OpenClaw adapter, so its turns carry OpenClaw ids.
+ rt_kw = {"runtime": "openclaw" if runtime == "nemoclaw" else runtime} if runtime else {}
rows = []
for et in ("message", "user", "prompt.submitted"):
- sub = local_store_via_daemon("query_events", event_type=et, limit=5000)
+ sub = local_store_via_daemon("query_events", event_type=et, limit=5000, **rt_kw)
if sub:
rows.extend(sub)
if not rows:
# Daemon unreachable → single-process fallback (tests/dev mode).
store = local_store.get_store(read_only=True)
for et in ("message", "user", "prompt.submitted"):
- sub = store.query_events(event_type=et, limit=5000) or []
+ sub = store.query_events(event_type=et, limit=5000, **rt_kw) or []
rows.extend(sub)
except Exception:
return None
@@ -472,10 +479,32 @@ def _try_local_store_autonomy() -> dict | None:
# Route
# ---------------------------------------------------------------------------
+def _autonomy_for_runtime(runtime: str, now: float) -> dict:
+ """``/api/autonomy?runtime=``: that runtime's user turns only.
+
+ No node-wide fallback: with no turns in the store for the runtime the
+ answer is the empty response, never another runtime's cadence (the legacy
+ JSONL scan reads OpenClaw's session files alone).
+ """
+ slot = _AUTONOMY_RT_CACHE.get(runtime)
+ if slot is not None and (now - slot[0]) < _AUTONOMY_CACHE_TTL_SECONDS:
+ return slot[1]
+ try:
+ result = _try_local_store_autonomy(runtime=runtime)
+ except Exception:
+ result = None
+ result = dict(result if result is not None else _empty_response(), runtime=runtime)
+ _AUTONOMY_RT_CACHE[runtime] = (now, result)
+ return result
+
+
@bp_autonomy.route("/api/autonomy")
def api_autonomy():
import dashboard as _d
now = datetime.now(tz=timezone.utc).timestamp()
+ runtime = (request.args.get("runtime") or "").strip().lower()
+ if runtime and runtime != "all":
+ return jsonify(_autonomy_for_runtime(runtime, now))
cached = _AUTONOMY_CACHE.get("data")
if cached is not None and (now - float(_AUTONOMY_CACHE.get("ts") or 0)) < _AUTONOMY_CACHE_TTL_SECONDS:
return jsonify(cached)
diff --git a/routes/evals.py b/routes/evals.py
index f89d4e8b63..c26af3ddb4 100644
--- a/routes/evals.py
+++ b/routes/evals.py
@@ -229,9 +229,15 @@ def evals_summary():
except (TypeError, ValueError):
hours = 24
hours = max(1, min(24 * 30, hours))
- payload = _store_via_daemon_or_direct(
- "query_eval_summary", window_hours=hours,
- )
+ # ``?runtime=`` scopes the tile to the runtime switcher; the response
+ # echoes it so the page can tell a scoped answer from a node-wide one.
+ runtime = (request.args.get("runtime") or "").strip().lower()
+ if runtime == "all":
+ runtime = ""
+ kwargs = {"window_hours": hours}
+ if runtime:
+ kwargs["runtime"] = runtime
+ payload = _store_via_daemon_or_direct("query_eval_summary", **kwargs)
if not payload:
payload = {
"avg_score": 0.0,
@@ -241,7 +247,7 @@ def evals_summary():
"p10": 0.0,
"window_hours": hours,
}
- return jsonify(payload)
+ return jsonify(dict(payload, runtime=runtime or "all"))
@bp_evals.route("/api/evals/rescore/", methods=["POST"])
diff --git a/routes/overview.py b/routes/overview.py
index 78bfd8407a..d5bfebcf86 100644
--- a/routes/overview.py
+++ b/routes/overview.py
@@ -1546,7 +1546,8 @@ def api_health_timeline():
events_per_session = max(1, min(int(request.args.get("events_per_session") or 500), 500))
dots_per_runtime = max(1, min(int(request.args.get("dots_per_runtime") or 30), 200))
- cache_key = (session_limit, events_per_session, dots_per_runtime)
+ runtime = _runtime_arg()
+ cache_key = (session_limit, events_per_session, dots_per_runtime, runtime)
with _health_timeline_cache_lock:
cached = _health_timeline_cache.get("value")
cached_key = _health_timeline_cache.get("key")
@@ -1558,9 +1559,18 @@ def api_health_timeline():
return jsonify(cached)
try:
- sessions = local_store_via_daemon("query_sessions", limit=session_limit) or []
+ # Under ?runtime the newest sessions node-wide can all belong to a
+ # louder runtime, so read deeper and keep this runtime's own newest.
+ sessions = local_store_via_daemon(
+ "query_sessions",
+ limit=min(session_limit * 10, 1000) if runtime else session_limit,
+ ) or []
except Exception as exc:
return jsonify({"runtimes": [], "error": f"sessions unavailable: {exc}"}), 503
+ if runtime:
+ sessions = [
+ s for s in sessions if _session_matches_runtime(s.get("session_id"), runtime)
+ ][:session_limit]
buckets: dict = {}
for s in sessions:
@@ -1607,7 +1617,7 @@ def api_health_timeline():
reverse=True,
)
- payload = {"runtimes": runtimes_out, "generated_at": _time.time()}
+ payload = {"runtimes": runtimes_out, "generated_at": _time.time(), "runtime": runtime or "all"}
with _health_timeline_cache_lock:
_health_timeline_cache["ts"] = _time.time()
_health_timeline_cache["key"] = cache_key
@@ -1707,14 +1717,39 @@ def api_prompt_errors():
return jsonify({"errors": errors, "count": len(errors)})
+def _runtime_arg() -> str:
+ """The ``?runtime=`` filter, lower-cased; ``""`` means every runtime."""
+ rt = (request.args.get("runtime") or "").strip().lower()
+ return "" if rt == "all" else rt
+
+
+def _session_matches_runtime(session_id, runtime: str) -> bool:
+ """Whether a session belongs to ``runtime``, from its id prefix.
+
+ NemoClaw runs the OpenClaw adapter, so its sessions carry OpenClaw ids.
+ """
+ from clawmetry.local_store import _runtime_of_session_id
+
+ got = _runtime_of_session_id(str(session_id or ""))
+ return got == runtime or (runtime == "nemoclaw" and got == "openclaw")
+
+
@bp_overview.route("/api/activity-heatmap")
def api_activity_heatmap():
- """30-day session activity heatmap data (#875)."""
+ """30-day session activity heatmap data (#875).
+
+ ``?runtime=`` keeps only that runtime's sessions so the Overview card
+ follows the runtime switcher. The response echoes the runtime it counted,
+ which lets the page tell a scoped answer from an older node-wide one.
+ """
from datetime import datetime, timedelta
+ runtime = _runtime_arg()
now = datetime.now()
cutoff = (now - timedelta(days=29)).strftime("%Y-%m-%d") + "T00:00:00"
rows = _ls_call("query_sessions", since=cutoff, limit=10000) or []
+ if runtime:
+ rows = [r for r in rows if _session_matches_runtime(r.get("session_id"), runtime)]
day_sessions: dict = {}
day_tokens: dict = {}
@@ -1738,7 +1773,7 @@ def api_activity_heatmap():
"tokens": day_tokens.get(ds, 0),
"cost": round(day_cost.get(ds, 0), 4),
})
- return jsonify({"days": days})
+ return jsonify({"days": days, "runtime": runtime or "all"})
@bp_overview.route("/api/cloud-cta/status")
diff --git a/tests/test_overview_runtime_scope.py b/tests/test_overview_runtime_scope.py
new file mode 100644
index 0000000000..c27e98a25a
--- /dev/null
+++ b/tests/test_overview_runtime_scope.py
@@ -0,0 +1,389 @@
+"""The Home (Overview) tab follows the runtime switcher, card by card.
+
+Burned 2026-09-15 on the hosted dashboard with Codex selected: System Health
+still listed OpenClaw's "Gateway :18789" (the snapshot names it a bare
+"Gateway", which the OpenClaw-name filter missed), Run Health drew a
+``claude_code`` row, and the 30-day activity, "How independent is your agent?",
+Anomaly Detection, Reliability and session-quality cards all counted every
+runtime on the node. The snapshot's Run Health slice had no Codex row at all,
+because the newest 60 sessions on that node were all Claude Code.
+"""
+from __future__ import annotations
+
+import importlib
+import json
+import os
+import re
+import shutil
+import subprocess
+import sys
+import time
+from datetime import datetime, timezone
+
+import pytest
+from flask import Flask
+
+_REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+if _REPO not in sys.path:
+ sys.path.insert(0, _REPO)
+
+_APP_JS = os.path.join(_REPO, "clawmetry", "static", "js", "app.js")
+
+
+def _src(path):
+ with open(path, encoding="utf-8") as fh:
+ return fh.read()
+
+
+def _function_body(src, header):
+ start = src.index(header)
+ nxt = re.search(r"\n(?:async )?function ", src[start + len(header):])
+ return src[start: start + len(header) + (nxt.start() if nxt else len(src))]
+
+
+def _fake_waste_flags(monkeypatch):
+ from clawmetry import waste_flags as wf
+
+ monkeypatch.setattr(wf, "runtime_from_session_id",
+ lambda sid: str(sid).split(":", 1)[0] if ":" in str(sid) else "openclaw")
+ monkeypatch.setattr(wf, "compute_signals_from_events", lambda events: {})
+ monkeypatch.setattr(wf, "compute_flags", lambda signals: [])
+ monkeypatch.setattr(wf, "event_is_real_error", lambda e: False)
+ monkeypatch.setattr(wf, "severity_from_counts", lambda errors, flags: "green")
+
+
+# ── /api/activity-heatmap ────────────────────────────────────────────────────
+
+
+@pytest.fixture
+def overview_client(monkeypatch):
+ import routes.overview as ov
+
+ today = datetime.now().strftime("%Y-%m-%dT10:00:00")
+ rows = [
+ {"session_id": "codex:1", "started_at": today, "token_count": 10, "cost_usd": 0.1},
+ {"session_id": "claude_code:2", "started_at": today, "token_count": 20, "cost_usd": 0.2},
+ {"session_id": "0b8f2c1e-openclaw", "started_at": today, "token_count": 30, "cost_usd": 0.3},
+ ]
+ monkeypatch.setattr(ov, "_ls_call", lambda method, **kw: list(rows))
+ app = Flask(__name__)
+ app.register_blueprint(ov.bp_overview)
+ return app.test_client()
+
+
+def _heatmap_totals(body):
+ return (sum(d["sessions"] for d in body["days"]), sum(d["tokens"] for d in body["days"]))
+
+
+def test_heatmap_counts_only_the_selected_runtime(overview_client):
+ body = overview_client.get("/api/activity-heatmap?runtime=codex").get_json()
+ assert body["runtime"] == "codex"
+ assert _heatmap_totals(body) == (1, 10)
+
+
+def test_heatmap_without_a_runtime_counts_every_runtime(overview_client):
+ body = overview_client.get("/api/activity-heatmap").get_json()
+ assert body["runtime"] == "all"
+ assert _heatmap_totals(body) == (3, 60)
+
+
+def test_heatmap_nemoclaw_counts_openclaw_adapter_sessions(overview_client):
+ body = overview_client.get("/api/activity-heatmap?runtime=nemoclaw").get_json()
+ assert _heatmap_totals(body) == (1, 30)
+
+
+# ── /api/health-timeline ─────────────────────────────────────────────────────
+
+
+def test_health_timeline_route_keeps_only_the_selected_runtime(monkeypatch):
+ import routes.local_query as lq
+ import routes.overview as ov
+
+ _fake_waste_flags(monkeypatch)
+ calls = []
+
+ def fake_proxy(method, **kw):
+ calls.append((method, kw))
+ if method == "query_sessions":
+ # The loud runtime owns the newest rows, the quiet one is older.
+ return [{"session_id": f"claude_code:{i}", "started_at": f"2026-09-15T10:{i:02d}"}
+ for i in range(8)] + [{"session_id": "codex:a", "started_at": "2026-09-14T09:00"}]
+ return []
+
+ monkeypatch.setattr(lq, "local_store_via_daemon", fake_proxy)
+ ov._health_timeline_cache.update(ts=0.0, value=None, key=None)
+ app = Flask(__name__)
+ app.register_blueprint(ov.bp_overview)
+ body = app.test_client().get("/api/health-timeline?runtime=codex&session_limit=2").get_json()
+ assert [r["runtime"] for r in body["runtimes"]] == ["codex"]
+ assert body["runtime"] == "codex"
+ # It read past the node's newest two sessions to find this runtime's.
+ assert ("query_sessions", {"limit": 20}) in calls
+
+
+# ── daemon snapshot: every runtime gets a Run Health row ─────────────────────
+
+
+class _TimelineStore:
+ def __init__(self):
+ self.event_reads = 0
+
+ def query_sessions(self, limit=60):
+ return [{"session_id": f"claude_code:{i}", "started_at": f"2026-09-15T10:{i:02d}:00",
+ "cost_usd": 0.1} for i in range(limit)]
+
+ def query_recent_sessions_by_runtime(self, per_runtime=2, max_runtimes=40):
+ return [{"runtime": "codex", "session_id": "codex:old", "last_ms": 1},
+ {"runtime": "claude_code", "session_id": "claude_code:0", "last_ms": 2}]
+
+ def query_events(self, session_id=None, limit=500):
+ self.event_reads += 1
+ if session_id == "codex:old":
+ return [{"ts": "2026-09-01T09:05:00", "cost_usd": 0.5},
+ {"ts": "2026-09-01T09:00:00", "cost_usd": 0.25}]
+ return []
+
+
+def test_snapshot_timeline_includes_a_quiet_runtime(monkeypatch):
+ from clawmetry import local_store as ls
+ from clawmetry import sync
+
+ _fake_waste_flags(monkeypatch)
+ store = _TimelineStore()
+ monkeypatch.setattr(ls, "get_store", lambda *a, **k: store)
+ sync._health_timeline_memo.update(ts=0.0, key=None, value=None)
+
+ out = sync._build_health_timeline(session_limit=5)
+ rows = {r["runtime"]: r["dots"] for r in out["runtimes"]}
+ assert set(rows) == {"claude_code", "codex"}
+ # claude_code:0 came back from both queries and is drawn once.
+ assert len(rows["claude_code"]) == 5
+ (dot,) = rows["codex"]
+ assert dot["started_at"] == "2026-09-01T09:00:00"
+ assert dot["cost_usd"] == pytest.approx(0.75)
+
+ reads = store.event_reads
+ sync._build_health_timeline(session_limit=5)
+ assert store.event_reads == reads, "a second cycle inside the TTL re-read every session"
+
+
+def test_autonomy_by_runtime_is_reused_within_its_ttl(monkeypatch):
+ from clawmetry import sync
+
+ calls = []
+ monkeypatch.setattr(sync, "_build_autonomy_snapshot",
+ lambda runtime=None: calls.append(runtime) or {"score": 0.5, "runtime": runtime})
+ sync._autonomy_by_rt_memo.update(ts=0.0, keys=None, value={})
+
+ out = sync._build_autonomy_by_runtime(["codex", "claude_code"])
+ assert set(out) == {"codex", "claude_code"}
+ assert out["codex"]["runtime"] == "codex"
+ sync._build_autonomy_by_runtime(["claude_code", "codex"])
+ assert sorted(calls) == ["claude_code", "codex"]
+ sync._build_autonomy_by_runtime(["codex"])
+ assert calls[-1] == "codex" and len(calls) == 3
+
+
+# ── /api/autonomy?runtime= ───────────────────────────────────────────────────
+
+
+@pytest.fixture
+def autonomy_app(tmp_path, monkeypatch):
+ monkeypatch.setenv("CLAWMETRY_LOCAL_STORE_PATH", str(tmp_path / "events.duckdb"))
+ monkeypatch.setenv("CLAWMETRY_LOCAL_FLUSH_SECS", "0.05")
+ monkeypatch.setenv("CLAWMETRY_LOCAL_FLUSH_BATCH", "1")
+ monkeypatch.setenv("CLAWMETRY_LOCAL_STORE_READ", "1")
+ import clawmetry.local_store as ls
+ importlib.reload(ls)
+ import routes.local_query as lq
+ importlib.reload(lq)
+ monkeypatch.setattr(lq, "local_store_via_daemon", lambda *a, **k: None)
+ monkeypatch.setattr(lq, "_read_discovery", lambda: None)
+ monkeypatch.setattr(lq, "_cached_discovery", lambda: None)
+ import routes.autonomy as aut
+ importlib.reload(aut)
+ a = Flask(__name__)
+ a.register_blueprint(aut.bp_autonomy)
+ yield a, ls
+ try:
+ ls.get_store().stop(flush=True)
+ except Exception:
+ pass
+
+
+def _ingest_user_turn(store, eid, sid, ts):
+ store.ingest({
+ "id": eid, "node_id": "agent+test", "agent_id": "main", "session_id": sid,
+ "event_type": "user", "ts": datetime.fromtimestamp(ts, tz=timezone.utc).isoformat(),
+ "data": {"message": {"role": "user", "content": eid}},
+ "cost_usd": None, "token_count": None, "model": None,
+ })
+
+
+def _wait_flush(store, t=2.0):
+ deadline = time.monotonic() + t
+ while time.monotonic() < deadline:
+ if store.health()["ring_depth"] == 0:
+ return
+ time.sleep(0.02)
+
+
+def test_autonomy_counts_only_the_selected_runtimes_turns(autonomy_app):
+ app, ls = autonomy_app
+ store = ls.get_store()
+ now = time.time()
+ _ingest_user_turn(store, "cx-1", "codex:s1", now - 900)
+ _ingest_user_turn(store, "cx-2", "codex:s1", now - 600)
+ for i in range(3):
+ _ingest_user_turn(store, f"cc-{i}", f"claude_code:s{i}", now - 1200 - i * 60)
+ _wait_flush(store)
+ client = app.test_client()
+ assert client.get("/api/autonomy").get_json()["samples_7d"] == 5
+ codex = client.get("/api/autonomy?runtime=codex").get_json()
+ assert codex["runtime"] == "codex"
+ assert codex["samples_7d"] == 2
+
+
+def test_autonomy_for_a_runtime_with_no_turns_is_empty_not_node_wide(autonomy_app):
+ app, ls = autonomy_app
+ store = ls.get_store()
+ _ingest_user_turn(store, "cc-1", "claude_code:s1", time.time() - 600)
+ _wait_flush(store)
+ body = app.test_client().get("/api/autonomy?runtime=cursor").get_json()
+ assert body["score"] is None
+ assert body["samples_7d"] == 0
+ assert body["runtime"] == "cursor"
+
+
+# ── /api/evals/summary?runtime= ──────────────────────────────────────────────
+
+
+def test_eval_summary_scopes_to_the_runtime(tmp_path, monkeypatch):
+ db_path = tmp_path / "clawmetry.duckdb"
+ monkeypatch.setenv("CLAWMETRY_LOCAL_STORE_PATH", str(db_path))
+ from clawmetry import local_store
+ monkeypatch.setattr(local_store, "DB_PATH", db_path)
+ monkeypatch.setattr(local_store, "_STORE", None, raising=False)
+ store = local_store.LocalStore()
+ now_iso = datetime.now(timezone.utc).isoformat()
+ with store._write_lock:
+ for sid in ("codex:a", "claude_code:b", "c0ffee-openclaw"):
+ store._conn.execute(
+ """
+ INSERT INTO sessions
+ (agent_type, session_id, node_id, agent_id, started_at,
+ last_active_at, ended_at, status, total_tokens, updated_at)
+ VALUES ('openclaw', ?, 'node-x', 'main', ?, ?, ?, 'completed', 1, ?)
+ """,
+ [sid, now_iso, now_iso, now_iso, int(time.time() * 1000)],
+ )
+ for sid, score in (("codex:a", 5.0), ("claude_code:b", 1.0), ("c0ffee-openclaw", 3.0)):
+ store.persist_eval_score(session_id=sid, score=score, reason="r", judge_model="m",
+ scored_at=1, rubric="default")
+ assert store.query_eval_summary(window_hours=24)["scored"] == 3
+ codex = store.query_eval_summary(window_hours=24, runtime="codex")
+ assert (codex["scored"], codex["total"], codex["avg_score"]) == (1, 1, 5.0)
+ assert store.query_eval_summary(window_hours=24, runtime="openclaw")["avg_score"] == 3.0
+ assert store.query_eval_summary(window_hours=24, runtime="nemoclaw")["avg_score"] == 3.0
+ assert store.query_eval_summary(window_hours=24, runtime="cursor")["total"] == 0
+
+
+def test_eval_summary_route_passes_and_echoes_the_runtime(monkeypatch):
+ import routes.evals as ev
+
+ seen = {}
+
+ def fake(method, **kw):
+ seen.update(kw)
+ return {"avg_score": 4.0, "total": 1, "scored": 1, "p50": 4.0, "p10": 4.0, "window_hours": 24}
+
+ monkeypatch.setattr(ev, "_store_via_daemon_or_direct", fake)
+ app = Flask(__name__)
+ app.register_blueprint(ev.bp_evals)
+ body = app.test_client().get("/api/evals/summary?window=24h&runtime=codex").get_json()
+ assert seen.get("runtime") == "codex"
+ assert body["runtime"] == "codex"
+ seen.clear()
+ body = app.test_client().get("/api/evals/summary?window=24h").get_json()
+ assert "runtime" not in seen
+ assert body["runtime"] == "all"
+
+
+# ── frontend ─────────────────────────────────────────────────────────────────
+
+
+def test_system_health_drops_the_hosted_gateway_pill_off_openclaw():
+ body = _function_body(_src(_APP_JS), "async function loadSystemHealth()")
+ seg = body[body.index("var services = "):body.index("var channels = ")]
+ assert "/^gateway$/i" in seg
+ assert "18789" in seg
+
+
+def test_cards_ask_for_the_selected_runtime():
+ src = _src(_APP_JS)
+ heat = _function_body(src, "async function loadActivityHeatmap()")
+ assert "'/api/activity-heatmap' + q" in heat
+ assert "data.runtime !== rt" in heat
+ auto = _function_body(src, "async function loadAutonomy()")
+ assert "fetchJsonWithTimeout(_auUrl, 5000)" in auto
+ ev = _function_body(src, "async function loadEvalSummary()")
+ assert "'/api/evals/summary?window=24h' + _evQ" in ev
+ assert "data.runtime !== _evRt" in ev
+
+
+def test_anomaly_panel_keeps_only_the_runtimes_sessions():
+ body = _function_body(_src(_APP_JS), "async function loadAnomalyPanel()")
+ scope = body[body.index("var _anScoped"):body.index("var active = ")]
+ assert "_cmRuntimeOf({session_id: sk}) === _anRt" in scope
+ assert "baselines = {};" in scope
+
+
+def test_reliability_card_is_hidden_under_a_runtime():
+ body = _function_body(_src(_APP_JS), "async function loadReliabilityCard()")
+ assert "_relCard.style.display = _relScoped ? 'none' : ''" in body
+ assert body.index("if (_relScoped) return;") < body.index("/api/reliability")
+
+
+def test_runtime_switch_is_not_swallowed_by_the_loadall_coalesce():
+ body = _function_body(_src(_APP_JS), "function _cmApplyRuntimeSelection(val)")
+ assert body.index("_loadAllLastFinishedMs = 0") < body.index("switchTab(_cmCurrentTab)")
+
+
+@pytest.mark.skipif(not shutil.which("node"), reason="node not installed")
+def test_run_health_renders_only_the_selected_runtime_under_node():
+ fn = _function_body(_src(_APP_JS), "async function loadHealthTimeline()")
+ script = fn + """
+var CURRENT, URL;
+function _cmRuntimeFilter() { return CURRENT; }
+function escapeHtmlSafe(s) { return String(s); }
+var els;
+var document = { getElementById: function (id) { return els[id]; } };
+function fetch(url) {
+ URL = url;
+ var rows = [{runtime: 'claude_code', dots: [{severity: 'red'}]},
+ {runtime: 'codex', dots: [{severity: 'green'}]},
+ {runtime: 'openclaw', dots: [{severity: 'green'}]}];
+ return Promise.resolve({ok: true, json: function () { return Promise.resolve({runtimes: rows}); }});
+}
+(async function () {
+ var out = {};
+ for (var rt of ['codex', 'cursor', 'nemoclaw', 'all']) {
+ CURRENT = rt;
+ els = {'health-timeline-card': {style: {display: 'none'}}, 'health-timeline-body': {innerHTML: ''}};
+ await loadHealthTimeline();
+ out[rt] = {url: URL, shown: els['health-timeline-card'].style.display !== 'none',
+ html: els['health-timeline-body'].innerHTML};
+ }
+ console.log(JSON.stringify(out));
+})();
+"""
+ res = subprocess.run(["node", "-e", script], capture_output=True, text=True, timeout=30)
+ assert res.returncode == 0, res.stderr
+ out = json.loads(res.stdout)
+ assert out["codex"]["url"] == "/api/health-timeline?runtime=codex"
+ assert out["codex"]["shown"] and "codex" in out["codex"]["html"]
+ assert "claude_code" not in out["codex"]["html"]
+ assert not out["cursor"]["shown"]
+ assert "openclaw" in out["nemoclaw"]["html"] and "claude_code" not in out["nemoclaw"]["html"]
+ assert out["all"]["url"] == "/api/health-timeline"
+ assert "claude_code" in out["all"]["html"] and "codex" in out["all"]["html"]