diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 328bb3974e..1b5d12a1c2 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -857,6 +857,7 @@ jobs:
tests/test_cost_basis_labels.py \
tests/test_provenance.py \
tests/test_provenance_render_coverage.py \
+ tests/test_cost_basis_remaining_surfaces.py \
tests/test_transcript_local_store.py \
tests/test_interceptor_daemon_tail.py \
tests/test_snapshot_transcripts_runtime_fair.py \
diff --git a/clawmetry/cost_basis_surfaces.py b/clawmetry/cost_basis_surfaces.py
new file mode 100644
index 0000000000..e0d6b958c5
--- /dev/null
+++ b/clawmetry/cost_basis_surfaces.py
@@ -0,0 +1,303 @@
+"""Financial-basis entries for the cost surfaces #5975 did not reach.
+
+vivekchand/clawmetry#5937, REQ-OBS-CEA-025 (AC-OBS-CEA-025.8 and .9).
+
+``clawmetry/cost_basis.py`` defines the vocabulary. This module applies it to
+the remaining dollar figures on the Usage tab (Where the money goes, the
+efficiency and cache cards, Cost Forecast, Cost Comparison, Spend
+Optimization, Cache Re-read Tax, compression potential, Cost By Plugin /
+Skill, Skill Cost Leaderboard, Cost by Team) and to the per-message cost a
+Sessions transcript carries.
+
+Every figure here is **usage value at published rates**: recorded token
+counts priced at a provider's list price, or the runtime's own per-call cost,
+which runtimes compute from published rates too. None is a bill, so none may
+claim ``contract`` or ``allocated_actual``.
+
+What varies is the arithmetic basis beside it:
+
+* a sum of recorded per-call cost over a window is ``derived``;
+* anything that assumes something unobserved is ``estimated``: a monthly
+ projection from a trailing window, a saving on a model nobody ran, a
+ session's cost split evenly across the skills it read, a category share
+ apportioned by token counts.
+
+The entries are built once per surface, here, so the local route and the
+hosted snapshot slice that reuses the same builder carry identical words
+(FLYWHEEL.md 0a, cloud parity). Each ``*_entries`` function returns a mapping
+ready for :func:`clawmetry.provenance.stamp`; :func:`stamp` applies one and
+never raises, because a labelling bug must not take a card down.
+"""
+from __future__ import annotations
+
+from typing import Any, Dict, Mapping
+
+from clawmetry import cost_basis as _cb
+from clawmetry import provenance as _prov
+
+Entries = Dict[str, Dict[str, Any]]
+
+
+def _pub(formula: str, source: str, *, basis: str = _prov.DERIVED,
+ **kw: Any) -> Dict[str, Any]:
+ return _cb.published_rate(formula, source, basis=basis, **kw)
+
+
+def stamp(payload: Any, entries: Mapping[str, Mapping[str, Any]]) -> Any:
+ """Attach ``entries`` to a dict payload. Never raises."""
+ try:
+ if isinstance(payload, dict):
+ _prov.stamp(payload, entries)
+ except Exception: # pragma: no cover - never-crash rule
+ pass
+ return payload
+
+
+# ── Efficiency grade, cache hit rate, savings ideas, routing advisor ────────
+
+_EFF_SRC = "DuckDB rollup_model_daily on this node (clawmetry/efficiency.py)"
+
+
+def efficiency_entries(days: int) -> Entries:
+ window = "the last %d days, scaled to 30 days" % int(days or 30)
+ return {
+ "projected_monthly_cost_usd": _pub(
+ "the recorded cost of the window divided by its days with data, "
+ "times 30. It assumes the next month looks like this window",
+ _EFF_SRC, basis=_prov.ESTIMATED, window=window),
+ "cache_saved_monthly_usd": _pub(
+ "cached-read tokens times the gap between the model's input rate "
+ "and its cache-read rate, scaled to a month. It assumes the "
+ "window's cache use continues",
+ _EFF_SRC, basis=_prov.ESTIMATED, window=window),
+ "left_on_table_monthly_usd": _pub(
+ "the projected monthly input cost times the share of input that "
+ "was not a cache hit, times an assumed 50% of it being cacheable, "
+ "times the cache-read discount",
+ _EFF_SRC, basis=_prov.ESTIMATED, window=window),
+ "metrics.window_cost_usd": _pub(
+ "sum of the recorded per-call cost over the window",
+ _EFF_SRC, window="the last %d days" % int(days or 30)),
+ "actions[].savings_monthly_usd": _pub(
+ "what the window would have cost after the suggested change, "
+ "subtracted from what it did cost, scaled to a month and capped "
+ "at 90% of projected spend. It assumes the change keeps the "
+ "results the same, which is the part that can be wrong",
+ _EFF_SRC, basis=_prov.ESTIMATED, window=window),
+ "actions[].data.window_cost_usd": _pub(
+ "sum of the recorded per-call cost for this model over the window",
+ _EFF_SRC, window="the last %d days" % int(days or 30)),
+ "actions[].data.target_window_cost_usd": _pub(
+ "the same token counts priced at the suggested model's published "
+ "rates", _EFF_SRC, basis=_prov.ESTIMATED),
+ "actions[].data.input_side_window_cost_usd": _pub(
+ "the input-token share of the window's recorded cost", _EFF_SRC),
+ "actions[].data.wasted_window_usd": _pub(
+ "cache writes that expired unread, priced at the write rate",
+ _EFF_SRC, basis=_prov.ESTIMATED),
+ "actions[].data.thinking_window_cost_usd": _pub(
+ "the thinking-token share of output cost over the window",
+ "clawmetry/spend_flow.py over the events table",
+ basis=_prov.ESTIMATED),
+ }
+
+
+# ── Where the money goes (spend flow) ────────────────────────────────────────
+
+_SF_SRC = "DuckDB events on this node (clawmetry/spend_flow.py)"
+
+
+def spend_flow_entries(days: int) -> Entries:
+ window = "the last %d days" % int(days or 7)
+ split = _pub(
+ "each call's recorded cost split between categories by their token "
+ "share of that call. Token shares are measured from event content "
+ "and the system-prompt share is the residual, so the split is an "
+ "estimate even where the total is not",
+ _SF_SRC, basis=_prov.ESTIMATED, window=window)
+ whole = _pub("sum of the recorded per-call cost", _SF_SRC, window=window)
+ return {
+ "totals.cost_usd": whole,
+ "totals.input_cost_usd": split,
+ "totals.output_cost_usd": split,
+ "runtimes[].cost_usd": whole,
+ "runtimes[].input_cost_usd": split,
+ "runtimes[].output_cost_usd": split,
+ "input_categories[].cost_usd": split,
+ "output_categories[].cost_usd": split,
+ "links[].cost_usd": split,
+ }
+
+
+# ── Cost Forecast ────────────────────────────────────────────────────────────
+
+_FC_SRC = "DuckDB daily rollups on this node"
+
+
+def forecast_entries(*, spent_so_far: float, daily_rate: float,
+ days_remaining: int) -> Entries:
+ return {
+ "projected_month_usd": _pub(
+ "spend so far this month, plus the average of the last 7 days "
+ "times the days left. It assumes the rest of the month looks "
+ "like the last week",
+ _FC_SRC, basis=_prov.ESTIMATED, window="the calendar month",
+ inputs={"spent_so_far_usd": round(spent_so_far, 4),
+ "daily_rate_usd": round(daily_rate, 4),
+ "days_remaining": days_remaining}),
+ "cost_this_month_usd": _pub(
+ "sum of the priced cost of every call this month", _FC_SRC,
+ window="this month, the local calendar month from the 1st"),
+ "daily_rate_usd": _pub(
+ "the priced cost of the last 7 days divided by 7", _FC_SRC,
+ window="the last 7 days"),
+ # A limit the operator typed, not usage value. It keeps its own
+ # arithmetic label and no financial basis, because it is not money
+ # anybody spent.
+ "monthly_budget_usd": _prov.measured(
+ "the monthly limit you set", "clawmetry budget config"),
+ }
+
+
+# ── Cost Comparison ──────────────────────────────────────────────────────────
+
+def cost_comparison_entries(source: str) -> Entries:
+ window = "the last 30 days"
+ return {
+ "actual.cost_usd": _pub(
+ "sum of the recorded per-call cost, with duplicate sibling rows "
+ "of the same turn removed. It is usage value, not an invoice",
+ source, window=window),
+ "alternatives[].estimated_cost": _pub(
+ "the same total tokens priced at the alternative's published "
+ "rates, assuming 60% input and 40% output. The real split and "
+ "the alternative's output length are not known",
+ source, basis=_prov.ESTIMATED, window=window),
+ "alternatives[].savings_usd": _pub(
+ "recorded usage value minus the alternative's estimate, on the "
+ "same assumptions", source, basis=_prov.ESTIMATED, window=window),
+ }
+
+
+# ── Spend Optimization ───────────────────────────────────────────────────────
+
+def spend_optimization_entries(tools_analysed: int) -> Entries:
+ saving = _pub(
+ "the measured cost of these tool calls over the window, times the "
+ "published price gap between the model they ran on and the cheaper "
+ "tier suggested. It assumes the cheaper model would have produced an "
+ "equivalent result, which is the part that can be wrong",
+ "duckdb spans, priced with the static model-tier ratio table",
+ basis=_prov.ESTIMATED, window="the last 30 days",
+ inputs={"tools_analysed": tools_analysed})
+ return {
+ "total_projected_savings_usd_30d": saving,
+ "recommendations[].projected_savings_usd_30d": saving,
+ "total_analyzed_cost_usd_30d": _pub(
+ "sum of the measured cost of the analysed tool calls",
+ "duckdb spans", window="the last 30 days"),
+ "recommendations[].current_cost_usd_30d": _pub(
+ "sum of the measured cost of this tool's calls",
+ "duckdb spans", window="the last 30 days"),
+ }
+
+
+def spend_optimization_unavailable() -> Entries:
+ return {
+ "total_projected_savings_usd_30d": _cb.unavailable(
+ "no spans were available to analyse, so there is nothing to "
+ "compare a cheaper tier against",
+ source="/api/usage/optimization-recommendations"),
+ "total_analyzed_cost_usd_30d": _cb.unavailable(
+ "no spans were available to analyse",
+ source="/api/usage/optimization-recommendations"),
+ }
+
+
+# ── Cache performance (cache trends) ─────────────────────────────────────────
+
+def cache_trends_entries(source: str, days: int) -> Entries:
+ window = "the last %d days" % int(days or 14)
+ parts = _pub("recorded tokens of this kind priced at the model's "
+ "published rate", source, window=window)
+ saved = _pub(
+ "cache-read tokens times the gap between the model's input rate and "
+ "its cache-read rate: what the same reads would have cost uncached",
+ source, basis=_prov.ESTIMATED, window=window)
+ out: Entries = {}
+ for scope in ("totals.", "daily[].", "by_model[]."):
+ for key in ("input_cost_usd", "output_cost_usd", "cache_read_cost_usd",
+ "cache_write_cost_usd", "total_cost_usd"):
+ out[scope + key] = parts
+ out[scope + "est_savings_usd"] = saved
+ return out
+
+
+# ── Cache Re-read Tax ────────────────────────────────────────────────────────
+
+def cache_risk_entries(source: str) -> Entries:
+ return {
+ "total_write_cost_usd": _pub(
+ "sum of each affected session's cache-write tokens priced at the "
+ "model's published cache-write rate", source),
+ "total_saved_usd": _pub(
+ "sum of each affected session's cache-read tokens times the gap "
+ "between the input rate and the cache-read rate",
+ source, basis=_prov.ESTIMATED),
+ }
+
+
+# ── Compression potential ────────────────────────────────────────────────────
+
+def compression_entries(source: str) -> Entries:
+ return {
+ "recoverable_usd": _pub(
+ "compressible tool-output tokens in qualifying sessions priced at "
+ "the session model's input rate. It assumes the output could be "
+ "summarised without losing what the agent needed",
+ source, basis=_prov.ESTIMATED),
+ }
+
+
+# ── Cost By Plugin / Skill, Skill Cost Leaderboard, Cost by Team ─────────────
+
+def by_plugin_entries(source: str) -> Entries:
+ return {
+ "plugins[].cost_usd": _pub(
+ "sum of the recorded cost of the events attributed to this "
+ "plugin or tool", source),
+ }
+
+
+def skill_attribution_entries(source: str) -> Entries:
+ split = _pub(
+ "each session's recorded cost split evenly between the skills it "
+ "read. A session that read two skills gives each half, whatever each "
+ "actually drove", source, basis=_prov.ESTIMATED)
+ return {
+ "skills[].total_cost_usd": split,
+ "skills[].avg_cost_usd": split,
+ "top5_week[].total_cost_usd": split,
+ "top5_week[].avg_cost_usd": split,
+ "total_cost": split,
+ }
+
+
+def by_team_entries(window_days: int) -> Entries:
+ return {
+ "teams[].cost_usd": _pub(
+ "sum of the recorded cost of each session, grouped by the team "
+ "label mapped to its runtime",
+ "DuckDB rollup_session and team_mapping",
+ window="the last %d days" % int(window_days or 7)),
+ }
+
+
+# ── Sessions transcript: per-message cost, summed per turn in the browser ───
+
+def transcript_entries(source: str) -> Entries:
+ return {
+ "messages[].cost_usd": _pub(
+ "the recorded cost of the event this message came from. A turn's "
+ "cost is the sum of its messages", source),
+ }
diff --git a/clawmetry/efficiency.py b/clawmetry/efficiency.py
index 2c160d19f9..2c77c106cf 100644
--- a/clawmetry/efficiency.py
+++ b/clawmetry/efficiency.py
@@ -27,6 +27,7 @@
import math
from typing import Any
+from clawmetry import cost_basis_surfaces as _cost_labels
from clawmetry.providers_pricing import (
_CACHE_READ_MULT,
_CACHE_WRITE_MULT,
@@ -170,6 +171,10 @@ def _empty_scope(days: int) -> dict[str, Any]:
"cache_saved_monthly_usd": 0.0,
"projected_monthly_cost_usd": 0.0,
"actions": [],
+ # What kind of money each figure is (REQ-OBS-CEA-025.8). Stamped in
+ # the engine so /api/efficiency and the hosted snapshot slice agree.
+ "provenance": _cost_labels.efficiency_entries(days),
+ "provenance_version": 1,
}
@@ -346,6 +351,8 @@ def _build_scope(rows: list[dict[str, Any]], days: int) -> dict[str, Any]:
"cache_saved_monthly_usd": round(cache_saved_window_usd * factor, 6),
"projected_monthly_cost_usd": round(projected_monthly_cost_usd, 6),
"actions": actions,
+ "provenance": _cost_labels.efficiency_entries(days),
+ "provenance_version": 1,
}
diff --git a/clawmetry/spend_flow.py b/clawmetry/spend_flow.py
index f8eae80f86..89fa2f0d7b 100644
--- a/clawmetry/spend_flow.py
+++ b/clawmetry/spend_flow.py
@@ -43,6 +43,7 @@
import logging
from typing import Any
+from clawmetry import cost_basis_surfaces as _cost_labels
from clawmetry.providers_pricing import estimate_event_cost_usd
log = logging.getLogger("clawmetry.spend_flow")
@@ -330,6 +331,11 @@ def _scope_payload(agg: dict[str, Any], days: int,
"input_categories": _categories_out(agg["input"], agg["input_cost_usd"], _BASIS_INPUT),
"output_categories": _categories_out(
agg["output"], agg["output_cost_usd"], basis_output or _BASIS_OUTPUT),
+ # What kind of money each figure is (REQ-OBS-CEA-025.8). Every scope,
+ # node-wide and per runtime, carries it, so the hosted interceptor's
+ # per-runtime copy keeps the basis too.
+ "provenance": _cost_labels.spend_flow_entries(days),
+ "provenance_version": 1,
}
diff --git a/clawmetry/static/js/app.js b/clawmetry/static/js/app.js
index 3991a96247..2a468a9985 100644
--- a/clawmetry/static/js/app.js
+++ b/clawmetry/static/js/app.js
@@ -4751,6 +4751,17 @@ function _cmRtRecentlyActive() {
var ts = (rt && rt !== 'all') ? (a.map[rt] || 0) : (a.max || 0);
return ts > 0 && (Date.now() - ts) < _CM_RT_ACTIVE_WINDOW_MS;
}
+// The hero's cost chip (REQ-OBS-CEA-025.8): the Spending tile's own number
+// and entry, through the shared component, with the basis beside it. The plan
+// note appears only for a real non-zero figure on a detected subscription: a
+// plan includes usage, it does not make it free (REQ-OBS-CEA-025.4).
+function _cmHeroCostChip(value, entry, onPlan) {
+ var included = !!onPlan && Number(value) !== 0;
+ return ''
+ + window.cmCostFigure(value, entry, { noBadge: true, label: 'Cost today' }) + ''
+ + (entry ? ' ' + window.cmProv.badge(entry, { label: 'Cost today' }) : '')
+ + (included ? ' included in your plan, not an extra bill' : '');
+}
function _renderOverviewHero() {
var hero = document.getElementById('overview-hero');
if (!hero) return;
@@ -4834,7 +4845,9 @@ function _renderOverviewHero() {
// window._cmCostTodayRaw is the number loadMiniWidgets actually rendered.
var _costRaw = window._cmCostTodayRaw;
var _costKnown = _scope ? true : (typeof _costRaw === 'number');
- var cost = _scope ? ('$' + _scope.cost.toFixed(2)) : (_txt('cost-today') || '$0.00');
+ // The number itself, not the tile's text read back off the DOM: the text
+ // is the formatter's output and cannot carry a basis.
+ var _costVal = _scope ? Number(_scope.cost || 0) : _costRaw;
var model = _scope ? (_txt('model-primary') || _scope.model || '—')
: (ov.model || _txt('model-primary') || 'your model');
// Node-wide, the chip is labelled "today" below, so it must BE today:
@@ -4846,8 +4859,10 @@ function _renderOverviewHero() {
: (_todayKnown ? ov.sessionsToday
: ((typeof ov.sessionCount === 'number') ? ov.sessionCount : null));
// Never assert 'free' from a number we have not actually read.
- var free = _costKnown && (cost === '$0.00' || cost === '$0' ||
- /oauth/i.test((document.getElementById('cost-trend') || {}).textContent || ''));
+ var _onPlan = /oauth/i.test((document.getElementById('cost-trend') || {}).textContent || '');
+ // Never assert 'free' (or 'included') from a number we have not read, and
+ // never from the tile's text: only from the value loadMiniWidgets rendered.
+ var free = _costKnown && (_costVal === 0 || _onPlan);
var say = window._cmLastAgentSay;
var sayText = say && say.text ? String(say.text).replace(/\s+/g, ' ').trim() : '';
if (sayText.length > 90) sayText = sayText.slice(0, 90) + '…';
@@ -4865,8 +4880,7 @@ function _renderOverviewHero() {
// Show nothing rather than a placeholder: an unlabelled '$0.00' next to
// live sessions reads as a real reading, not as 'still loading'.
// A plan includes usage; it does not make it free (REQ-OBS-CEA-025.4).
- var _heroIncluded = free && !/^\$0(\.00)?$/.test(cost);
- if (_costKnown) stats.push('💸 ' + escHtml(cost) + '' + (_heroIncluded ? ' included in your plan, not an extra bill' : ''));
+ if (_costKnown) stats.push('💸 ' + _cmHeroCostChip(_costVal, window._cmCostTodayEntry || null, free && _onPlan));
// Efficiency chip (design spec §1a): grade next to cost answers "what did it
// cost me, and is that reasonable?" in one read. Renders only when the
// daemon slice is fresh for the CURRENT runtime filter and passes the trust
@@ -5079,6 +5093,8 @@ async function loadMiniWidgets(overview, usage) {
// the tile: today, week and month all come out of the same rollup by the
// same rule, so they share a basis.
var _costEntry = window.cmProv ? window.cmProv.of(usage, 'todayCost') : null;
+ // The hero chip prints this same entry beside this same number.
+ window._cmCostTodayEntry = _costEntry;
var _costUnknown = window.cmProv ? window.cmProv.isUnknown(_costEntry) : false;
var _basisEl = document.getElementById('cost-basis-badge');
if (_basisEl && window.cmProv) {
@@ -5238,34 +5254,45 @@ async function loadMiniWidgets(overview, usage) {
_set('tokens-today', _fmtT(_scope.tokensToday));
_set('token-rate', _fmtT(_scope.tokensMonth));
window._cmCostTodayRaw = Number(_scope.cost || 0);
- _set('cost-today', fmtCost(_scope.cost));
- // SPENDING wk/mo sub-figures scope too (were node-wide projections).
- if (_scope.costWeek != null) _set('cost-week', fmtCost(_scope.costWeek));
- if (_scope.costMonth != null) _set('cost-month', fmtCost(_scope.costMonth));
// These three came from the runtime-scoped API, not the payload
// loadMiniWidgets badged, and the local-mode fallback above is NOT
// period-split: it repeats the runtime's all-time total in all three
- // slots. Re-badge from the source actually used, so the tooltip is
- // about the number on screen rather than the one it replaced.
+ // slots. One entry for the source actually used is shared by the
+ // tile, its badge and the hero chip, so they cannot disagree. This
+ // block used to call an fmtCost that does not exist in this scope:
+ // the ReferenceError was swallowed below, the tile kept node-wide
+ // figures and the hero printed the runtime's.
+ var _split = (_scope.costWeek !== _scope.costMonth);
+ var _scopeEntry = {
+ basis: _split ? 'derived' : 'estimated',
+ label: _split ? 'derived' : 'estimated',
+ hint: _split
+ ? 'Derived: computed from measured inputs by an exact rule.'
+ : 'Estimated: modelled, with an assumption that can be wrong.',
+ formula: _split
+ ? ('measured token counts for runtime ' + (_scope.runtime || '')
+ + ', priced against the provider\'s published rate card')
+ : ('this runtime\'s all-time total, standing in for all three '
+ + 'windows because the scoped source is not split by period'),
+ source: _split ? '/api/v1/usage?runtime=' + (_scope.runtime || '')
+ : '/api/runtime-summary',
+ cost_basis: 'published_rate',
+ rate_source: 'the runtime\'s own per-call cost when it reported one '
+ + '(computed by the runtime from published rates), otherwise '
+ + 'ClawMetry\'s published price table'
+ };
+ window._cmCostTodayEntry = _scopeEntry;
+ var _setScopedCost = function (id, v, label) {
+ var e = document.getElementById(id);
+ if (e) e.innerHTML = window.cmProv.figure(v, _scopeEntry, { label: label, noBadge: true });
+ };
+ _setScopedCost('cost-today', _scope.cost, 'Cost today');
+ // SPENDING wk/mo sub-figures scope too (were node-wide projections).
+ if (_scope.costWeek != null) _setScopedCost('cost-week', _scope.costWeek, 'Cost this week');
+ if (_scope.costMonth != null) _setScopedCost('cost-month', _scope.costMonth, 'Cost this month');
try {
var _sBadge = document.getElementById('cost-basis-badge');
- if (_sBadge && window.cmProv) {
- var _split = (_scope.costWeek !== _scope.costMonth);
- _sBadge.innerHTML = window.cmProv.badge({
- basis: _split ? 'derived' : 'estimated',
- label: _split ? 'derived' : 'estimated',
- hint: _split
- ? 'Derived: computed from measured inputs by an exact rule.'
- : 'Estimated: modelled, with an assumption that can be wrong.',
- formula: _split
- ? ('measured token counts for runtime ' + (_scope.runtime || '')
- + ', priced against the provider\'s published rate card')
- : ('this runtime\'s all-time total, standing in for all three '
- + 'windows because the scoped source is not split by period'),
- source: _split ? '/api/v1/usage?runtime=' + (_scope.runtime || '')
- : '/api/runtime-summary'
- }, { label: 'Cost' });
- }
+ if (_sBadge) _sBadge.innerHTML = window.cmProv.badge(_scopeEntry, { label: 'Cost' });
} catch (_eb) {}
window._cmTodayTokensRaw = _scope.tokensToday;
}
@@ -18028,7 +18055,12 @@ var _CM_EFF_IDEAS = {
// of output spend); evidence is the "Where the money goes" chart.
thinking_trim: { icon: '🧠', stem: 'think', evidenceTab: 'usage' },
};
-function _cmEffIdeaRowHtml(a) {
+// A translated sentence with a figure inside it. The figure is HTML from the
+// shared component, so it is spliced in after the sentence is escaped.
+function _cmI18nFig(key, fallback, figHtml) {
+ return escHtml(t(key, { amt: '\u0000' }, fallback)).split('\u0000').join(figHtml);
+}
+function _cmEffIdeaRowHtml(a, saveEntry) {
var m = _CM_EFF_IDEAS[a.id];
if (!m) return '';
var d = a.data || {};
@@ -18052,7 +18084,11 @@ function _cmEffIdeaRowHtml(a) {
+ ' ' + escHtml(t('efficiency.evidence', null, 'See the evidence')) + ' →'
+ ''
+ ''
- + '
' + escHtml(t('efficiency.save_mo', { amt: '$' + save }, 'save about $' + save + '/mo')) + '
'
+ // An estimate at published rates (REQ-OBS-CEA-025.9): the card heading
+ // carries the badge, the figure keeps the explanation on hover.
+ + '
';
- return ''
+ : '';
+ return caption + '' + tbl;
}
function renderEfficiencyCard() {
@@ -18291,17 +18338,19 @@ function _renderEfficiencyCardInner(card, eff) {
var sentence = t('efficiency.grade_sentence', { hit: hit, ctx: ctx },
'Your agent reuses ' + hit + '% of what it reads and carries about ' + ctx + ' tokens of history into each reply.');
var tip = t('efficiency.tooltip', null, 'A to F score of how much of your spend does useful work: how often your agent reuses what it already read, how much history each reply carries, and whether saved work pays for itself.');
- var rows = (eff.actions || []).map(_cmEffIdeaRowHtml).filter(Boolean);
+ var saveEntry = window.cmProv.of(eff, 'actions[].savings_monthly_usd');
+ var rows = (eff.actions || []).map(function (a) { return _cmEffIdeaRowHtml(a, saveEntry); }).filter(Boolean);
var total = Math.round(_cmEffTotalSavings(eff));
var saved = Math.round(Number(eff.cache_saved_monthly_usd) || 0);
var right;
if (rows.length) {
- right = '
'
@@ -18309,7 +18358,8 @@ function _renderEfficiencyCardInner(card, eff) {
}
if (saved >= 1) {
right += '
✨ '
- + escHtml(t('efficiency.already_saved', { amt: '$' + saved }, 'Reusing work already saved you about $' + saved + '/mo.')) + '
';
+ + _cmI18nFig('efficiency.already_saved', 'Reusing work already saved you about \u0000/mo.',
+ window.cmCostFigure(saved, window.cmProv.of(eff, 'cache_saved_monthly_usd'), { label: 'Saved by reusing cached work, per month' })) + '';
}
card.style.display = '';
card.innerHTML = '
'
@@ -18332,13 +18382,6 @@ function _renderEfficiencyCardInner(card, eff) {
// they are cloud-safe by construction (cm-cloud-efficiency serves that URL
// from the snapshot) and add ZERO fetches on tab load. Perf-first per
// FLYWHEEL §5 "share, don't duplicate."
-function _cmFmtUsd(n) {
- n = Number(n) || 0;
- if (n >= 1000) return '$' + Math.round(n).toLocaleString();
- if (n >= 10) return '$' + Math.round(n);
- if (n >= 1) return '$' + n.toFixed(1);
- return '$' + n.toFixed(2);
-}
var _CM_CACHE_LEFT_ON_TABLE_FRAC = 0.5;
var _CM_CACHE_READ_MULT = 0.1;
// Derive the Cache-Hit tile payload from an efficiency scope, mirroring the
@@ -18395,12 +18438,12 @@ function renderCacheHitRateCard() {
+ '
'
@@ -18527,7 +18571,6 @@ async function loadUsage() {
return;
}
function fmtTokens(n) { return n >= 1000000 ? (n/1000000).toFixed(1) + 'M' : n >= 1000 ? (n/1000).toFixed(0) + 'K' : String(n); }
- function fmtCost(c) { return c >= 0.01 ? '$' + c.toFixed(2) : c > 0 ? '<$0.01' : '$0.00'; }
// Subscription-coverage snapshot from /api/usage (dashboard.py
// _get_billing_coverage). When the user is on a subscription (e.g.
// Claude Max 20x), the headline API-equivalent cost is misleading —
@@ -18539,7 +18582,7 @@ async function loadUsage() {
function setUsageCard(valId, cost, tokens, periodKey) {
var v = document.getElementById(valId);
var s = document.getElementById(valId + '-cost');
- var costStr = fmtCost(cost || 0);
+ var costStr = window.cmProv.fmtMoney(cost || 0);
var tokStr = fmtTokens(tokens || 0);
var costKey = periodKey ? periodKey + 'Cost' : '';
var costEntry = (window.cmProv && costKey) ? window.cmProv.of(data, costKey) : null;
@@ -18679,9 +18722,7 @@ async function loadUsage() {
// version of what that heading was reaching for.
var costLabel = 'Cost';
var _cmCell = function (key, name) {
- return window.cmProv
- ? window.cmProv.money(data, key, { label: name })
- : fmtCost(data[key]);
+ return window.cmProv.money(data, key, { label: name });
};
var tableHtml = '
Period
Tokens
' + costLabel + '
';
tableHtml += '
Today
' + fmtTokens(data.today) + '
' + _cmCell('todayCost', 'Cost today') + '
';
@@ -18715,7 +18756,8 @@ async function loadUsage() {
} else {
otelExtra.style.display = 'none';
}
- renderPluginPieChart(byPlugin.plugins || [], byPlugin.store_available === false);
+ renderPluginPieChart(byPlugin.plugins || [], byPlugin.store_available === false,
+ window.cmProv.of(byPlugin, 'plugins[].cost_usd'));
// Load session cost breakdown
fetch('/api/sessions/cost-breakdown').then(r => r.json()).then(function(cbd) {
window._sessionCostData = cbd.top10 || [];
@@ -18904,6 +18946,8 @@ async function loadCacheRisk() {
var savedUsd = Number(d.total_saved_usd) || 0;
var affected = Number(d.affected_sessions) || 0;
var maxGap = Number(d.max_idle_gap_sec) || 0;
+ var _crWrite = window.cmProv.of(d, 'total_write_cost_usd');
+ var _crSaved = window.cmProv.of(d, 'total_saved_usd');
if (!expiries && !writeCost) return;
title.style.display = '';
card.style.display = '';
@@ -18912,7 +18956,7 @@ async function loadCacheRisk() {
var html = '
💰 Est. savings: '+window.cmCostFigure(savings, window.cmProv.of(d, 'totals.est_savings_usd'), { label: 'Estimated savings vs. uncached' })+' vs. uncached
';
}
+ var _ctSave = window.cmProv.of(d, 'by_model[].est_savings_usd');
var models = (d.by_model || []).filter(function(m) { return (m.cache_read_tokens||0)+(m.cache_write_tokens||0) > 0; });
if (models.length > 0) {
html += '
'
+ '
'
+ '
Model
'
+ '
Hit %
'
- + '
Saved
'
+ + '
Saved' + (_ctSave ? ' ' + window.cmProv.badge(_ctSave, { label: 'Saved per model (estimate)' }) : '') + '
'
+ '
';
models.forEach(function(m) {
var mhit = m.cache_hit_ratio_pct || 0;
@@ -19028,7 +19071,7 @@ async function loadCacheAnalytics() {
html += '
';
+ var _altCostEntry = window.cmProv.of(data, 'alternatives[].estimated_cost');
+ var _altSaveEntry = window.cmProv.of(data, 'alternatives[].savings_usd');
+ html += '
Estimates for the same tokens at each model\'s published rates'
+ + (_altCostEntry ? ' ' + window.cmProv.badge(_altCostEntry, { label: 'Alternative model estimates' }) : '') + '
';
html += '
';
alts.forEach(function(alt) {
var color = providerColors[alt.provider] || '#94a3b8';
var altCost = alt.estimated_cost || 0;
var savingsPct = alt.savings_pct || 0;
var savingsUsd = alt.savings_usd || 0;
- var costStr = altCost >= 0.01 ? '$' + altCost.toFixed(2) : altCost > 0 ? '<$0.01' : '$0.00';
+ var costStr = window.cmCostFigure(altCost, _altCostEntry, { noBadge: true, label: 'Estimated cost on ' + (alt.display_name || 'this model') });
var isCurrent = actualCost > 0 && Math.abs(altCost - actualCost) / (actualCost || 1) < 0.15;
var isCheaper = savingsPct > 5;
var isMoreExpensive = savingsPct < -5;
@@ -19153,9 +19200,9 @@ function renderCostComparison(data) {
if (isCurrent) {
html += '
about ' + window.cmCostFigure(Math.abs(savingsUsd), _altSaveEntry, { noBadge: true, label: 'Estimated extra cost' }) + ' more (' + Math.abs(savingsPct) + '%)
';
} else {
html += '
similar cost
';
}
@@ -19189,17 +19236,14 @@ function renderSpendOptimization(data) {
el.innerHTML = '' + t("app.no_optimization_suggestions_yet_run_more_agents_wi", null, "No optimization suggestions yet — run more agents with span data enabled to see recommendations.") + '';
return;
}
- var totalSave = data.total_projected_savings_usd_30d || 0;
- var saveFmt = totalSave >= 0.01 ? '$' + totalSave.toFixed(2) : totalSave > 0 ? '<$0.01' : '$0.00';
// This is the loudest number on the card and it is a counterfactual: what
// the window WOULD have cost on a cheaper tier, assuming that tier does the
// same job. Badged as an estimate so it does not read as banked money.
var saveEntry = window.cmProv
? window.cmProv.of(data, 'total_projected_savings_usd_30d') : null;
- var saveHtml = window.cmProv
- ? window.cmProv.money(data, 'total_projected_savings_usd_30d',
- { label: 'Projected 30-day savings' })
- : escHtml(saveFmt);
+ var curEntry = window.cmProv.of(data, 'recommendations[].current_cost_usd_30d');
+ var saveHtml = window.cmProv.money(data, 'total_projected_savings_usd_30d',
+ { label: 'Projected 30-day savings (estimate)' });
var html = '
';
html += '
Projected 30-day savings
';
html += '
' + saveHtml + '
';
@@ -19207,11 +19251,10 @@ function renderSpendOptimization(data) {
html += '
';
});
legend.innerHTML = lhtml;
@@ -19791,6 +19840,34 @@ async function loadModelAttribution() {
}
// ===== Skill Attribution =====
+// Leaderboard rows (REQ-OBS-CEA-025.8). A skill's cost is its sessions' cost
+// split evenly across the skills each one read, so the column badge says it
+// is an estimate at published rates. The local API sends total_cost_usd and
+// avg_cost_usd; the hosted synthesiser sends total_cost and avg_cost. Reading
+// only the second shape printed $0.00 on every local row.
+function _skillCostTableHtml(data, list) {
+ var totalEntry = window.cmProv.of(data, 'skills[].total_cost_usd');
+ var avgEntry = window.cmProv.of(data, 'skills[].avg_cost_usd');
+ function num(row, a, b) {
+ var v = row[a] != null ? row[a] : row[b];
+ return v == null ? null : Number(v);
+ }
+ var html = '
';
+}
async function loadSkillAttribution() {
var el = document.getElementById('skill-leaderboard-content');
if (!el) return;
@@ -19799,27 +19876,15 @@ async function loadSkillAttribution() {
var top5 = data.top5_week || [];
var allSkills = data.skills || [];
var totalCost = data.total_cost || 0;
- function fmtCost(c) { return c >= 0.01 ? '$' + c.toFixed(2) : c > 0 ? '<$0.01' : '$0.00'; }
if (top5.length === 0) {
el.innerHTML = '' + t("app.no_skill_invocations_detected_yet_skills_are_detec", null, "No skill invocations detected yet. Skills are detected when SKILL.md files are read during sessions.") + '';
return;
}
- var html = '
';
// Issue #564: decoding-config pill — small inline summary of the sampling
// params that produced this assistant turn (only present when the backend
// could extract at least one known key).
@@ -20753,7 +20807,7 @@ function _renderTurnChapter(turn, highlightOriginal) {
// Turn spend — same per-event token/cost stamps the Turn anatomy page sums,
// so the two figures agree.
if (turn.tokens > 0) pieces.push('🪙 ' + (turn.tokens >= 1000 ? (turn.tokens / 1000).toFixed(1) + 'K' : turn.tokens) + ' tok');
- if (turn.cost > 0) pieces.push('' + _taFmtCost(turn.cost) + '');
+ if (turn.cost > 0) pieces.push('' + window.cmCostFigure(turn.cost, window._replayCostEntry, { noBadge: true, label: 'Turn cost' }) + '');
var meta = pieces.join(' · ');
var html = '';
html += '';
@@ -21337,6 +21391,7 @@ async function viewTranscript(sessionId) {
window._replayFilter = 'all';
window._transcriptAllMessages = [];
window._transcriptPaging = null;
+ window._replayCostEntry = null;
_updateLoadEarlierBtn();
try {
// Fetch transcript, compaction markers, config-drift, lexical drift, and policy events in parallel
@@ -21437,6 +21492,14 @@ async function viewTranscript(sessionId) {
+ ''
+ ''
+ '';
+ // The per-turn and per-tool chips print message costs, so the header says
+ // once what kind of money they are (REQ-OBS-CEA-025.8). An older daemon or
+ // the JSONL fallback sends no entry, and then no label is invented.
+ window._replayCostEntry = window.cmProv.of(data, 'messages[].cost_usd');
+ if (window._replayCostEntry) {
+ metaHtml += '
Turn and tool costs: '
+ + window.cmProv.badge(window._replayCostEntry, { label: 'Turn and tool costs' }) + '
';
+ }
document.getElementById('transcript-meta').innerHTML = metaHtml;
_loadInputsPanel(sessionId);
_loadLifecycleCoverageLine(sessionId);
diff --git a/clawmetry/static/js/provenance.js b/clawmetry/static/js/provenance.js
index 58be19c55c..25f6118909 100644
--- a/clawmetry/static/js/provenance.js
+++ b/clawmetry/static/js/provenance.js
@@ -224,6 +224,20 @@
+ (opts.noBadge ? '' : badge(e, opts));
}
+ // A cost figure whose basis may not have arrived (REQ-OBS-CEA-025.8). With
+ // an entry this is figure(). Without one (a daemon older than the label,
+ // or a hosted answer the browser synthesised) the number still goes
+ // through the shared formatter and hover text, and no badge is invented.
+ function costFigure(value, entry, opts) {
+ var o = {};
+ var src = opts || {};
+ for (var k in src) {
+ if (Object.prototype.hasOwnProperty.call(src, k)) o[k] = src[k];
+ }
+ if (!entry) o.noBadge = true;
+ return figure(value, entry, o);
+ }
+
// Shorthands for the two shapes that appear most: a money figure looked up
// from a payload by key, and a score.
function money(payload, key, opts) {
@@ -306,8 +320,10 @@
LABEL: LABEL, HINT: HINT, COST_LABEL: COST_LABEL, COST_HINT: COST_HINT,
of: of, isUnknown: isUnknown, tip: tip, badge: badge,
figure: figure, money: money, score: score, text: text,
+ costFigure: costFigure,
fmtMoney: fmtMoney, fmtScore: fmtScore
};
+ window.cmCostFigure = costFigure;
// Terse aliases: these get called from inside string-concatenated table
// rows, where `window.cmProv.figure(...)` would be most of the line.
window.cmMoney = money;
diff --git a/docs/MODULE_MAP.md b/docs/MODULE_MAP.md
index b565bfc281..e997012a4e 100644
--- a/docs/MODULE_MAP.md
+++ b/docs/MODULE_MAP.md
@@ -4,7 +4,7 @@
> `python3 scripts/gen_module_map.py` (CI fails on drift via
> `tests/test_module_map_drift.py`).
-271 modules, 84 Flask blueprints. `CLAUDE.md` carries a short curated table of the ones you reach for most often; this is the whole list.
+272 modules, 84 Flask blueprints. `CLAUDE.md` carries a short curated table of the ones you reach for most often; this is the whole list.
Size bands are deliberately coarse so this file does not churn on every PR: **small** is under 200 lines, **medium** under 1k, **large** under 5k, **huge** is 5k and up.
@@ -158,6 +158,7 @@ The pip-installable package: CLI, sync daemon, DuckDB store, detectors, enforcem
| `clawmetry/context_coverage.py` | medium | Which context-blowout signals we can actually see, per runtime. |
| `clawmetry/context_windows.py` | medium | Context-window sizing across every runtime ClawMetry ingests. |
| `clawmetry/cost_basis.py` | medium | What kind of money a cost figure is. |
+| `clawmetry/cost_basis_surfaces.py` | medium | Financial-basis entries for the cost surfaces #5975 did not reach. |
| `clawmetry/cost_optimizer_advice.py` | medium | Cost Optimizer advice: observed provider routes, experiments, and cost basis. |
| `clawmetry/cost_optimizer_snapshot.py` | small | Cost Optimizer slice for the hosted dashboard (AC-OBS-CEA-023.9). |
| `clawmetry/cost_windows.py` | medium | One definition of "today", "this week" and "this month" for every cost surface. |
diff --git a/docs/acceptance_criteria.json b/docs/acceptance_criteria.json
index 8abc7234bb..942264b55e 100644
--- a/docs/acceptance_criteria.json
+++ b/docs/acceptance_criteria.json
@@ -690,6 +690,24 @@
"doc_id": "950d4687-45cc-45a8-9d58-0c0d82fd5d9b",
"text": "Hovering or keyboard-focusing a basis label shall show how the figure was computed: the formula, the source of the rate, the window, and the billing route where one applies."
},
+ {
+ "id": "AC-OBS-CEA-025.8",
+ "doc": "Cost basis labels: usage value at published rates, contract spend and actual spend are never confused",
+ "doc_id": "950d4687-45cc-45a8-9d58-0c0d82fd5d9b",
+ "text": "Each cost figure on the Overview hero chip; on the Usage tab's Where the money goes chart, Cost Forecast, Cache Re-read Tax, cache hit rate and cache performance cards, savings ideas and routing advisor cards, compression potential card, Cost By Plugin / Skill, Cost Comparison, Spend Optimization, Skill Cost Leaderboard and Cost by Team cards; and on the per-turn and per-tool cost chips of a Sessions tab transcript shall carry a financial basis of exactly one of the four in AC-OBS-CEA-025.1, shown as visible text beside the figure or on the heading or caption of the card, chart, list or transcript that holds it. The payloads that serve these figures, including the hosted snapshot slices the hosted dashboard reads them from (efficiency, spend flow, forecast, cost comparison, cache trends, spend optimization and transcripts), shall carry the same basis."
+ },
+ {
+ "id": "AC-OBS-CEA-025.9",
+ "doc": "Cost basis labels: usage value at published rates, contract spend and actual spend are never confused",
+ "doc_id": "950d4687-45cc-45a8-9d58-0c0d82fd5d9b",
+ "text": "A counterfactual cost figure (a projected saving, a month-end forecast, or what the same usage would cost on another model) shall be labelled an estimate of usage value at published rates, and no card shall describe usage value at published rates as actual spend."
+ },
+ {
+ "id": "AC-OBS-CEA-025.10",
+ "doc": "Cost basis labels: usage value at published rates, contract spend and actual spend are never confused",
+ "doc_id": "950d4687-45cc-45a8-9d58-0c0d82fd5d9b",
+ "text": "The count of dollar figures rendered without a basis on the Overview, Usage and Sessions tabs shall not increase; the check shall find those renders from the tabs' own templates rather than from a hand-kept list."
+ },
{
"id": "AC-OBS-GHA-001.1",
"doc": "Governance and Human Approval",
diff --git a/routes/sessions.py b/routes/sessions.py
index 9e545e7656..882d9c847f 100644
--- a/routes/sessions.py
+++ b/routes/sessions.py
@@ -5656,6 +5656,15 @@ def _try_local_store_transcript(session_id: str, _events=None, _msg_cap: int = 5
"intent_source": _intent.get("intent_source") or "",
"_source": "local_store",
}
+ # Per-message cost is what the Sessions tab's per-turn and per-tool chips
+ # sum and print (REQ-OBS-CEA-025.8). The hosted snapshot's transcripts
+ # slice reuses this dict, so the basis reaches the hosted replay too.
+ try:
+ from clawmetry import cost_basis_surfaces as _cost_labels
+ _cost_labels.stamp(ret, _cost_labels.transcript_entries(
+ "DuckDB events on this node"))
+ except Exception:
+ pass
if truncated:
ret["_truncated"] = True
ret["_oldest_contiguous_ts"] = oldest_contiguous_ts
@@ -5690,12 +5699,19 @@ def api_transcript_page(session_id):
if rows:
t = _try_local_store_transcript(session_id, _events=rows, _msg_cap=2000)
msgs = (t or {}).get("messages") or []
- return jsonify({
+ out = {
"messages": msgs,
"count": len(msgs),
"has_more": bool(page.get("has_more")),
"next_before_ts": page.get("next_before_ts"),
- })
+ }
+ try:
+ from clawmetry import cost_basis_surfaces as _cost_labels
+ _cost_labels.stamp(out, _cost_labels.transcript_entries(
+ "DuckDB events on this node"))
+ except Exception:
+ pass
+ return jsonify(out)
@bp_sessions.route("/api/transcript/")
diff --git a/routes/usage.py b/routes/usage.py
index d1210f5be9..6af2ea3ec8 100644
--- a/routes/usage.py
+++ b/routes/usage.py
@@ -46,6 +46,7 @@
from clawmetry._gate import gate
from clawmetry import provenance as _prov
from clawmetry import cost_basis as _cost_basis
+from clawmetry import cost_basis_surfaces as _cost_labels
from clawmetry.config import is_local_store_read_enabled
from routes._dedupe import build_sibling_bucket_max, is_sibling_dup
@@ -895,7 +896,9 @@ def _try_local_store_usage_by_plugin(threshold_pct, runtime=None):
"trend": "flat",
})
rows.sort(key=lambda r: r["total_tokens"], reverse=True)
- return {"plugins": rows, "warnings": warnings, "_source": "local_store"}
+ return _cost_labels.stamp(
+ {"plugins": rows, "warnings": warnings, "_source": "local_store"},
+ _cost_labels.by_plugin_entries("DuckDB events on this node"))
def _try_local_store_usage_by_plugin_trend(days_back):
@@ -1058,6 +1061,9 @@ def _try_local_store_cost_comparison():
"alternatives": alternatives,
"period": "30d",
"_source": "local_store",
+ "provenance": _cost_labels.cost_comparison_entries(
+ "DuckDB events on this node"),
+ "provenance_version": 1,
}
@@ -1175,28 +1181,9 @@ def _try_local_store_usage_forecast():
"window_days": window_days,
"daily_window": [round(c, 4) for c in reversed(window)],
"_source": "local_store",
- }, {
- "projected_month_usd": _prov.estimated(
- "spend so far this month, plus the average of the last 7 days "
- "times the days left. It assumes the rest of the month looks "
- "like the last week",
- "DuckDB daily rollups on this node",
- window="the calendar month",
- inputs={"spent_so_far_usd": round(cost_this_month, 4),
- "daily_rate_usd": round(daily_rate, 4),
- "days_remaining": days_remaining}),
- "cost_this_month_usd": _prov.derived(
- "sum of the priced cost of every call this month",
- "DuckDB daily rollups on this node",
- window="this month, the local calendar month from the 1st"),
- "daily_rate_usd": _prov.derived(
- "the priced cost of the last 7 days divided by 7",
- "DuckDB daily rollups on this node",
- window="the last 7 days"),
- "monthly_budget_usd": _prov.measured(
- "the monthly limit you set",
- "clawmetry budget config"),
- })
+ }, _cost_labels.forecast_entries(
+ spent_so_far=cost_this_month, daily_rate=daily_rate,
+ days_remaining=days_remaining))
# Known non-OpenClaw runtime prefixes (session-id prefix = runtime; agent_type
@@ -1513,6 +1500,9 @@ def _try_local_store_skill_attribution():
"note": "Skills detected from events table in the local DuckDB store.",
"clawhub": {"enabled": False, "url": None},
"_source": "local_store",
+ "provenance": _cost_labels.skill_attribution_entries(
+ "DuckDB events on this node"),
+ "provenance_version": 1,
}
@@ -2535,7 +2525,9 @@ def api_usage_by_plugin():
}
)
rows.sort(key=lambda r: r["total_tokens"], reverse=True)
- return jsonify({"plugins": rows, "warnings": warnings})
+ return jsonify(_cost_labels.stamp(
+ {"plugins": rows, "warnings": warnings},
+ _cost_labels.by_plugin_entries("session transcripts on disk")))
@bp_usage.route("/api/usage/by-plugin/trend")
@@ -3114,7 +3106,9 @@ def api_usage_cost_comparison():
return jsonify(fast)
try:
- return jsonify(_d._build_cost_comparison())
+ return jsonify(_cost_labels.stamp(
+ _d._build_cost_comparison(),
+ _cost_labels.cost_comparison_entries("session transcripts on disk")))
except Exception as e:
return jsonify({"error": str(e), "alternatives": [], "actual": {}}), 500
@@ -3693,11 +3687,14 @@ def api_skill_attribution():
sessions_dir = _d._get_sessions_dir()
if not sessions_dir or not os.path.isdir(sessions_dir):
- return jsonify({
+ # No skill read was found, so nothing is attributed to any skill: the
+ # total over zero skills is exactly 0 (the card shows "no skill
+ # invocations", not a dollar figure). Same label as the other paths.
+ return jsonify(_cost_labels.stamp({
"skills": [], "top5_week": [], "total_cost": 0.0,
"note": "No sessions directory found.",
"clawhub": {"enabled": False, "url": None},
- })
+ }, _cost_labels.skill_attribution_entries("session transcripts on disk")))
SKILL_MD_RE = _re.compile(r'[/\\]([^/\\]+)[/\\]SKILL\.md', _re.IGNORECASE)
# Also match bare "SKILL.md" references with skill name in path context
@@ -3792,13 +3789,13 @@ def api_skill_attribution():
note = 'Skills detected from SKILL.md file reads in session transcripts.'
- return jsonify({
+ return jsonify(_cost_labels.stamp({
'skills': skills_out,
'top5_week': top5_week,
'total_cost': round(total_cost, 6),
'note': note,
'clawhub': {'enabled': False, 'url': None},
- })
+ }, _cost_labels.skill_attribution_entries("session transcripts on disk")))
# ── Per-agent / per-team cost attribution (issue #3000) ──────────────────────
@@ -3857,7 +3854,9 @@ def api_usage_by_team():
gateway = _ls_call_team('query_gateway_usage', window_days=window_days)
if not isinstance(gateway, dict):
gateway = {'available': False}
- return jsonify({'teams': rows, 'window_days': window_days, 'gateway': gateway})
+ return jsonify(_cost_labels.stamp(
+ {'teams': rows, 'window_days': window_days, 'gateway': gateway},
+ _cost_labels.by_team_entries(window_days)))
@bp_usage.route('/api/usage/team-mappings', methods=['GET'])
@@ -4218,6 +4217,9 @@ def _try_local_store_cache_trends(days: int):
"totals": totals_out,
"recommendations": _cache_recommendations(totals_out, by_model_out),
"_source": "local_store",
+ "provenance": _cost_labels.cache_trends_entries(
+ "DuckDB events on this node", days),
+ "provenance_version": 1,
}
@@ -4438,13 +4440,13 @@ def api_usage_cache_trends():
totals_bucket[k] += b[k]
totals_out = _summarise_cache_bucket("totals", totals_bucket, key="label")
- return jsonify({
+ return jsonify(_cost_labels.stamp({
"days": days,
"daily": daily_out,
"by_model": by_model_out,
"totals": totals_out,
"recommendations": _cache_recommendations(totals_out, by_model_out),
- })
+ }, _cost_labels.cache_trends_entries("session transcripts on disk", days)))
# ── Cache Risk: per-session idle-gap re-write tax (issue #2839 Part 1) ──
@@ -4499,6 +4501,9 @@ def api_usage_cache_risk():
"total_saved_usd": round(total_saved, 4),
"max_idle_gap_sec": round(max_idle_gap, 1),
"_source": "local_store" if cb.get("_source") else "none",
+ "provenance": _cost_labels.cache_risk_entries(
+ "the per-session cost breakdown (DuckDB sessions)"),
+ "provenance_version": 1,
})
@@ -4979,30 +4984,13 @@ def _try_local_store_spend_optimization():
# table and on the assumption that the cheaper model does the same job.
# Both can be wrong, and the badge says so rather than letting a green
# 22px "Projected 30-day savings" read as money already in the bank.
- saving_entry = _prov.estimated(
- "the measured cost of these tool calls over the window, times the "
- "published price gap between the model they ran on and the cheaper "
- "tier suggested. It assumes the cheaper model would have produced an "
- "equivalent result, which is the part that can be wrong",
- "duckdb spans, priced with the static model-tier ratio table",
- window="the last 30 days",
- inputs={"tools_analysed": len(recs)})
return _prov.stamp({
"recommendations": recs,
"total_projected_savings_usd_30d": round(total_save, 4),
"total_analyzed_cost_usd_30d": round(total_cost, 4),
"window_days": 30,
"_source": "local_store",
- }, {
- "total_projected_savings_usd_30d": saving_entry,
- "recommendations[].projected_savings_usd_30d": saving_entry,
- "total_analyzed_cost_usd_30d": _prov.derived(
- "sum of the measured cost of the analysed tool calls",
- "duckdb spans", window="the last 30 days"),
- "recommendations[].current_cost_usd_30d": _prov.derived(
- "sum of the measured cost of this tool's calls",
- "duckdb spans", window="the last 30 days"),
- })
+ }, _cost_labels.spend_optimization_entries(len(recs)))
@bp_usage.route("/api/usage/optimization-recommendations")
@@ -5025,15 +5013,7 @@ def api_usage_optimization_recommendations():
"total_analyzed_cost_usd_30d": 0,
"window_days": 30,
"note": "Enable clawmetry connect to see recommendations.",
- }, {
- "total_projected_savings_usd_30d": _prov.unknown(
- "no spans were available to analyse, so there is nothing to "
- "compare a cheaper tier against",
- source="/api/usage/optimization-recommendations"),
- "total_analyzed_cost_usd_30d": _prov.unknown(
- "no spans were available to analyse",
- source="/api/usage/optimization-recommendations"),
- }))
+ }, _cost_labels.spend_optimization_unavailable()))
@bp_usage.route("/api/efficiency")
@@ -5331,7 +5311,11 @@ def api_usage_compression():
"""
try:
rows = _ls_call("query_sessions_table", limit=2000) or []
- return jsonify(_agg_compression(rows))
+ agg = _agg_compression(rows)
+ if agg:
+ _cost_labels.stamp(agg, _cost_labels.compression_entries(
+ "DuckDB sessions on this node"))
+ return jsonify(agg)
except Exception:
return jsonify({})
diff --git a/tests/test_cost_basis_remaining_surfaces.py b/tests/test_cost_basis_remaining_surfaces.py
new file mode 100644
index 0000000000..ae891e0752
--- /dev/null
+++ b/tests/test_cost_basis_remaining_surfaces.py
@@ -0,0 +1,613 @@
+"""Guard: the rest of the dollar figures say what kind of money they are.
+
+vivekchand/clawmetry#5937, REQ-OBS-CEA-025. #5975 labelled the Overview spend
+tile, the Usage period cards and the Flow brain panel. An audit on 2026-09-14
+found the Overview hero chip, most Usage cards and the Sessions transcript
+cost chips still printing bare dollar amounts. These tests hold the payloads
+(local routes and the hosted snapshot slices built from the same code) and the
+shipped renderers, run under node out of ``app.js`` and ``provenance.js``.
+
+Criteria declared here:
+
+* AC-OBS-CEA-025.8 -- the hero chip, the remaining Usage cards and the
+ transcript chips carry a financial basis, in payload and on screen:
+ ``test_the_efficiency_slice_labels_every_cost_figure``,
+ ``test_every_spend_flow_scope_labels_its_costs``,
+ ``test_the_usage_card_payloads_label_their_costs``,
+ ``test_the_snapshot_usage_slices_carry_the_basis``,
+ ``test_the_transcript_payload_labels_message_cost``,
+ ``test_the_snapshot_transcripts_carry_the_basis``,
+ ``test_the_hero_chip_shows_the_tile_figure_with_its_basis``,
+ ``test_the_spend_flow_chart_captions_its_basis``,
+ ``test_the_plugin_legend_labels_its_costs``,
+ ``test_the_skill_leaderboard_prints_real_costs_with_a_basis``,
+ ``test_the_team_and_cache_cards_label_their_costs``,
+ ``test_the_transcript_turn_and_tool_chips_carry_the_basis``.
+* AC-OBS-CEA-025.9 -- counterfactuals are estimates at published rates, and
+ usage value is never called actual spend:
+ ``test_counterfactual_figures_are_estimates_at_published_rates``,
+ ``test_the_cost_comparison_never_calls_usage_actual_spend``.
+"""
+import datetime as _dt
+import json
+import os
+import re
+import shutil
+import subprocess
+import sys
+
+import pytest
+
+REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+sys.path.insert(0, REPO)
+sys.path.insert(0, os.path.join(REPO, "tests"))
+
+from clawmetry import cost_basis, provenance # noqa: E402
+
+try: # absent before this change; each test then fails on its own
+ from clawmetry import cost_basis_surfaces # noqa: E402
+except ImportError: # pragma: no cover - only on a tree without the fix
+ cost_basis_surfaces = None
+
+APP_JS = os.path.join(REPO, "clawmetry", "static", "js", "app.js")
+PROV_JS = os.path.join(REPO, "clawmetry", "static", "js", "provenance.js")
+
+BILL_SHAPED = (cost_basis.CONTRACT, cost_basis.ALLOCATED_ACTUAL)
+
+
+def _money_entries(payload):
+ out = []
+
+ def walk(node):
+ if isinstance(node, dict):
+ prov = node.get("provenance")
+ if isinstance(prov, dict):
+ for key, entry in prov.items():
+ leaf = key.split(".")[-1].split("[]")[-1] or key
+ if (provenance.figure_kind(leaf) == "money"
+ or "_usd" in leaf or leaf.endswith("_cost")):
+ out.append((key, entry))
+ for k, v in node.items():
+ if k != "provenance":
+ walk(v)
+ elif isinstance(node, list):
+ for item in node[:50]:
+ walk(item)
+ walk(payload)
+ return out
+
+
+def _assert_cost_labelled(payload, where, *, not_costs=("monthly_budget_usd",)):
+ # Money only: a score beside the costs (the efficiency grade) is another
+ # requirement's figure.
+ gaps = [g for g in provenance.audit_payload(payload) if g["kind"] == "money"]
+ assert not gaps, "%s renders unlabelled money: %r" % (where, gaps)
+ entries = [(k, e) for k, e in _money_entries(payload) if k not in not_costs]
+ assert entries, "%s carries no cost provenance at all" % where
+ for key, entry in entries:
+ cb = entry.get("cost_basis")
+ assert cb in cost_basis.COST_BASES, (
+ "%s: %s has no financial basis (%r)" % (where, key, entry))
+ assert cb not in BILL_SHAPED, "%s: %s claims %s" % (where, key, cb)
+ assert entry.get("cost_basis_label") == cost_basis.COST_BASIS_LABEL[cb]
+
+
+def _entry(payload, key):
+ e = provenance.entry_for(payload, key)
+ assert e, "%s has no entry for %s" % (sorted(payload), key)
+ return e
+
+
+def _is_estimate_at_published_rates(entry):
+ return (entry.get("cost_basis") == cost_basis.PUBLISHED_RATE
+ and entry.get("basis") == provenance.ESTIMATED)
+
+
+# ── Payloads ────────────────────────────────────────────────────────────────
+
+_EFF_ROWS = [{"runtime": "claude_code", "model": "claude-sonnet-4-5",
+ "tokens_in": 200000, "tokens_out": 60000, "cache_read": 900000,
+ "cache_write": 120000, "cost_usd": 14.0, "calls": 500,
+ "days_with_data": 12}]
+
+
+def test_the_efficiency_slice_labels_every_cost_figure():
+ from clawmetry import efficiency
+ full = efficiency.build_efficiency_slice(_EFF_ROWS, days=30)
+ scopes = [full] + list(full.get("byRuntime", {}).values())
+ scopes.append(efficiency.build_efficiency_slice([], days=30))
+ assert len(scopes) >= 3
+ for scope in scopes:
+ _assert_cost_labelled(scope, "efficiency slice")
+ for key in ("projected_monthly_cost_usd", "cache_saved_monthly_usd",
+ "left_on_table_monthly_usd", "actions[].savings_monthly_usd"):
+ assert _is_estimate_at_published_rates(_entry(full, key)), key
+
+
+def test_every_spend_flow_scope_labels_its_costs():
+ from clawmetry import spend_flow
+ empty = spend_flow.build_spend_flow_slice([], days=7)
+ scope = spend_flow._scope_payload(spend_flow._new_agg(), 7)
+ for payload in (empty, scope):
+ populated = dict(payload)
+ populated["input_categories"] = [{"id": "user_prompts", "tokens": 10,
+ "cost_usd": 1.0, "basis": "measured"}]
+ populated["runtimes"] = [{"runtime": "codex", "cost_usd": 1.0,
+ "input_cost_usd": 1.0, "output_cost_usd": 0.0}]
+ populated["links"] = [{"source": "user_prompts",
+ "target": "runtime:codex", "cost_usd": 1.0}]
+ _assert_cost_labelled(populated, "spend flow scope")
+ assert _entry(scope, "totals.cost_usd")["basis"] == provenance.DERIVED
+ assert _is_estimate_at_published_rates(
+ _entry(scope, "input_categories[].cost_usd"))
+
+
+def _usage_mod():
+ import routes.usage as usage_mod
+ return usage_mod
+
+
+def _today():
+ return _dt.datetime.now().strftime("%Y-%m-%dT%H:%M:%S")
+
+
+def _fake_events():
+ return [{"id": "e%d" % i, "session_id": "claude_code:s%d" % (i % 2),
+ "ts": _today(), "event_type": "tool_call", "model": "claude-sonnet-4-5",
+ "token_count": 1000 + i, "cost_usd": 0.25,
+ "data": {"tool": "exec"}} for i in range(4)]
+
+
+def _spend_opt_spans(usage_mod):
+ tool, target = next(iter(usage_mod._SPEND_OPT_TOOL_DOWNGRADE.items()))
+ tgt_ratio = usage_mod._SPEND_OPT_MODEL_RATIO.get(target, 1.0)
+ for model in ("claude-opus-4-1", "claude-opus-4", "gpt-4o", "o1",
+ "claude-sonnet-4-5", "gemini-2.5-pro"):
+ tier = usage_mod._spend_opt_model_tier(model)
+ if usage_mod._SPEND_OPT_MODEL_RATIO.get(tier, 0) > tgt_ratio:
+ return [{"tool_name": tool, "model": model, "cost_usd": 2.0}
+ for _ in range(5)]
+ pytest.fail("no model in a tier dearer than %s" % target)
+
+
+def _fake_ls_call(usage_mod):
+ spans = _spend_opt_spans(usage_mod)
+
+ def call(method, **_kw):
+ if method == "query_cache_metrics":
+ return [{"day": _today()[:10], "model": "claude-sonnet-4-5",
+ "input_tokens": 1000, "output_tokens": 300,
+ "cache_read_tokens": 4000, "cache_write_tokens": 500,
+ "input_cost": 0.3, "output_cost": 0.45,
+ "cache_read_cost": 0.12, "cache_write_cost": 0.19,
+ "total_cost": 1.06}]
+ if method == "query_recent_spans":
+ return spans
+ if method == "query_sessions_table":
+ return [{"compression_potential_pct": 80,
+ "compressible_tool_tokens": 5000,
+ "compression_recoverable_usd": 1.2,
+ "dominant_compression_type": "json"}]
+ return None
+ return call
+
+
+@pytest.fixture
+def usage_client(monkeypatch):
+ from flask import Flask
+ usage_mod = _usage_mod()
+ import routes.sessions as sessions_mod
+ monkeypatch.setattr(usage_mod, "_ls_call", _fake_ls_call(usage_mod))
+ monkeypatch.setattr(usage_mod, "_ls_get_store", lambda: object())
+ monkeypatch.setattr(usage_mod, "_scan_events_slim",
+ lambda *a, **k: _fake_events())
+ monkeypatch.setattr(usage_mod, "_ls_event_skill",
+ lambda ev: "pdf" if ev.get("cost_usd") else None)
+ monkeypatch.setattr(usage_mod, "_ls_call_team", lambda method, **kw: [
+ {"label": "Eng", "cost_usd": 1.5, "tokens": 900, "sessions": 2,
+ "runtimes": ["codex"]}])
+ monkeypatch.setattr(sessions_mod, "_try_local_store_cost_breakdown", lambda: {
+ "_source": "local_store", "sessions": [
+ {"cache_expiry_count": 3, "cache_write_cost_usd": 0.4,
+ "cache_saved_usd": 0.1, "max_idle_gap_sec": 900}]})
+ app = Flask("cost-basis-remaining")
+ app.register_blueprint(usage_mod.bp_usage)
+ return app.test_client(), usage_mod
+
+
+def test_the_usage_card_payloads_label_their_costs(usage_client):
+ client, usage_mod = usage_client
+ for url in ("/api/usage/cache-risk", "/api/usage/compression",
+ "/api/usage/by-team"):
+ body = client.get(url).get_json()
+ _assert_cost_labelled(body, url)
+ built = {
+ "cost comparison": usage_mod._try_local_store_cost_comparison(),
+ "cache trends": usage_mod._try_local_store_cache_trends(7),
+ "spend optimization": usage_mod._try_local_store_spend_optimization(),
+ "by plugin": usage_mod._try_local_store_usage_by_plugin(50.0),
+ "skill attribution": usage_mod._try_local_store_skill_attribution(),
+ }
+ for where, payload in built.items():
+ assert payload is not None, "%s builder deferred" % where
+ _assert_cost_labelled(payload, where)
+ # Forecast: the builder needs a real store, so check it uses the shared
+ # entries and that those entries are right.
+ import inspect
+ assert "forecast_entries(" in inspect.getsource(
+ usage_mod._try_local_store_usage_forecast)
+ fc = cost_basis_surfaces.forecast_entries(
+ spent_so_far=10.0, daily_rate=1.0, days_remaining=5)
+ for key in ("projected_month_usd", "cost_this_month_usd", "daily_rate_usd"):
+ assert fc[key]["cost_basis"] == cost_basis.PUBLISHED_RATE, key
+ assert "cost_basis" not in fc["monthly_budget_usd"], (
+ "a budget the operator typed is not usage value")
+
+
+def test_counterfactual_figures_are_estimates_at_published_rates(usage_client,
+ monkeypatch):
+ client, usage_mod = usage_client
+ opt = usage_mod._try_local_store_spend_optimization()
+ cmp_ = usage_mod._try_local_store_cost_comparison()
+ fc = cost_basis_surfaces.forecast_entries(
+ spent_so_far=10.0, daily_rate=1.0, days_remaining=5)
+ for entry in (_entry(opt, "total_projected_savings_usd_30d"),
+ _entry(opt, "recommendations[].projected_savings_usd_30d"),
+ _entry(cmp_, "alternatives[].estimated_cost"),
+ _entry(cmp_, "alternatives[].savings_usd"),
+ fc["projected_month_usd"]):
+ assert _is_estimate_at_published_rates(entry), entry
+ # Nothing analysed: not "you could save $0.00", and not a published-rate
+ # figure either.
+ monkeypatch.setattr(usage_mod, "_ls_call", lambda *a, **k: None)
+ body = client.get("/api/usage/optimization-recommendations").get_json()
+ assert body["total_projected_savings_usd_30d"] is None
+ assert _entry(body, "total_projected_savings_usd_30d")["cost_basis"] == \
+ cost_basis.UNKNOWN
+
+
+def test_the_snapshot_usage_slices_carry_the_basis(usage_client):
+ from clawmetry import sync
+ snap = sync._build_usage_snapshot()
+ for key in ("costComparison", "cacheTrends", "spendOptimization"):
+ assert snap.get(key), "snapshot %s slice is empty" % key
+ _assert_cost_labelled(snap[key], "snapshot " + key)
+
+
+def _transcript_rows(sid):
+ # The store hands the builder ``data`` already decoded.
+ ts = "2026-09-14T10:00:0%dZ"
+ return [
+ {"id": "u1", "node_id": "n", "agent_id": "main", "session_id": sid,
+ "event_type": "message", "ts": ts % 1, "token_count": 0,
+ "cost_usd": 0.0,
+ "data": {"role": "user", "content": "do it", "timestamp": ts % 1}},
+ {"id": "a1", "node_id": "n", "agent_id": "main", "session_id": sid,
+ "event_type": "message", "ts": ts % 2, "token_count": 120,
+ "cost_usd": 0.02,
+ "data": {"role": "assistant", "content": "done", "timestamp": ts % 2}},
+ ]
+
+
+def test_the_transcript_payload_labels_message_cost(monkeypatch):
+ import routes.sessions as sessions_mod
+ monkeypatch.setattr(sessions_mod, "_fetch_session_intent", lambda sid: {})
+ sid = "claude_code:t1"
+ rows = list(reversed(_transcript_rows(sid))) # query_events is DESC
+ t = sessions_mod._try_local_store_transcript(sid, _events=rows)
+ assert t and any(m.get("cost_usd") == 0.02 for m in t["messages"]), t
+ _assert_cost_labelled(t, "/api/transcript")
+ assert _entry(t, "messages[].cost_usd")["basis"] == provenance.DERIVED
+
+ from flask import Flask
+ import routes.local_query as lq
+ monkeypatch.setattr(lq, "_dispatch", lambda name, args: {
+ "rows": rows, "has_more": False, "next_before_ts": None})
+ app = Flask("transcript-page")
+ app.register_blueprint(sessions_mod.bp_sessions)
+ page = app.test_client().get("/api/transcript-page/" + sid).get_json()
+ assert page["messages"], page
+ _assert_cost_labelled(page, "/api/transcript-page")
+
+
+def test_the_snapshot_transcripts_carry_the_basis(monkeypatch):
+ from unittest.mock import patch
+ from clawmetry import sync
+ import routes.sessions as sessions_mod
+ sid = "claude_code:t2"
+ rows = list(reversed(_transcript_rows(sid)))
+
+ class _Store:
+ def query_events(self, session_id=None, limit=None):
+ if session_id is None:
+ return [{"session_id": sid, "ts": rows[0]["ts"], "id": "a1"}]
+ return rows
+
+ def query_recent_sessions_by_runtime(self, per_runtime=2, **_kw):
+ return []
+
+ monkeypatch.setattr(sessions_mod, "_fetch_session_intent", lambda s: {})
+ sync._TRANSCRIPT_SNAP_CACHE.clear()
+ with patch("clawmetry.local_store.get_store", return_value=_Store()):
+ out = sync._build_transcripts()
+ assert sid in out, out
+ _assert_cost_labelled(out[sid], "snapshot transcripts")
+
+
+# ── The shipped renderers ───────────────────────────────────────────────────
+
+def _node():
+ node = shutil.which("node")
+ if not node:
+ pytest.skip("node is not installed")
+ return node
+
+
+def _app():
+ return open(APP_JS, encoding="utf-8").read()
+
+
+def _fn(src, name):
+ m = re.search(r"^(?:async )?function %s\(" % re.escape(name), src, re.M)
+ assert m, "%s is gone from app.js" % name
+ nxt = re.search(r"^(?:async )?function [\w$]+\(", src[m.end():], re.M)
+ end = m.end() + nxt.start() if nxt else len(src)
+ return src[m.start():end]
+
+
+_PRELUDE = "\n".join([
+ "var window = globalThis;",
+ "function t(k, v, fb) { return fb; }",
+ "function escHtml(s) { return String(s == null ? '' : s)"
+ ".replace(/&/g,'&').replace(//g,'>')"
+ ".replace(/\"/g,'"'); }",
+ "var console_error = console.error; console.error = function () {};",
+])
+
+_CTX_STUB = (
+ "var _ctx = new Proxy({}, {get: function (t, k) {"
+ " return (k in t) ? t[k] : function () {}; },"
+ " set: function (t, k, v) { t[k] = v; return true; }});")
+
+
+def _run(body, *, fns=(), els=None, fetch_json=None):
+ app = _app()
+ prog = "\n".join([
+ _PRELUDE,
+ open(PROV_JS, encoding="utf-8").read(),
+ _CTX_STUB,
+ "var els = %s;" % json.dumps(els or {}),
+ "Object.keys(els).forEach(function (k) { els[k].style = {};"
+ " els[k].getContext = function () { return _ctx; }; });",
+ "var document = {getElementById: function (id) { return els[id] || null; },"
+ " body: {}, createElement: function () { var o = {innerHTML: ''};"
+ " Object.defineProperty(o, 'textContent', {set: function (v) {"
+ " o.innerHTML = escHtml(v); }}); return o; }};",
+ "function getComputedStyle() { return {getPropertyValue: function () { return ''; }}; }",
+ "function fetch() { return Promise.resolve({ok: true, status: 200,"
+ " json: function () { return Promise.resolve(%s); }}); }"
+ % json.dumps(fetch_json),
+ ] + [_fn(app, f) for f in fns] + [body])
+ out = subprocess.run([_node(), "-e", prog], capture_output=True, text=True,
+ timeout=30)
+ assert out.returncode == 0, out.stderr[-2000:]
+ return json.loads(out.stdout.strip().splitlines()[-1])
+
+
+def _pub_entry(**kw):
+ return cost_basis.published_rate("tokens times rate", "duckdb", **kw)
+
+
+def _badges(html):
+ return re.findall(r']*>([^<]*)', html)
+
+
+def test_the_hero_chip_shows_the_tile_figure_with_its_basis():
+ entry = _pub_entry(window="today")
+ got = _run(
+ "console.log(JSON.stringify({"
+ " priced: _cmHeroCostChip(8.49, %s, false),"
+ " plan: _cmHeroCostChip(8.49, %s, true),"
+ " zeroPlan: _cmHeroCostChip(0, %s, true),"
+ " old: _cmHeroCostChip(8.49, null, false)}));"
+ % ((json.dumps(entry),) * 3), fns=("_cmHeroCostChip",))
+ assert "$8.49" in got["priced"]
+ assert _badges(got["priced"]) == ["published rates"]
+ assert "not an extra bill" in got["plan"]
+ assert "not an extra bill" not in got["zeroPlan"]
+ assert "$8.49" in got["old"] and not _badges(got["old"])
+ hero = _fn(_app(), "_renderOverviewHero")
+ assert "_cmHeroCostChip(" in hero
+ assert "_txt('cost-today')" not in hero, (
+ "the hero reads the tile's text back instead of its number and entry")
+
+
+def test_no_function_calls_a_formatter_that_only_another_function_defines():
+ """The runtime-scoped Spending tile called ``fmtCost`` inside
+ loadMiniWidgets, where no fmtCost exists. The ReferenceError was swallowed,
+ the tile kept node-wide figures, and the hero chip printed the runtime's:
+ two different numbers for one thing on one screen."""
+ import test_provenance_render_coverage as cov
+ lines = cov._lines()
+ funcs = cov._functions(lines)
+ fmts = cov._formatters(lines)
+ top_level = set(funcs)
+ src = "\n".join(lines)
+ offenders = []
+ for name in sorted(fmts - top_level):
+ if re.search(r"^(?:var|let|const) %s\b" % re.escape(name), src, re.M):
+ continue
+ call = re.compile(r"(?")[0]
+ assert _badges(head) == ["published rates"], head
+ assert "$0.00" not in got, "the local shape printed $0.00 on every row"
+ hosted = {"skills": [{"name": "pdf", "invocations": 2, "total_cost": 1.25,
+ "avg_cost": 0.625, "clawhub_url": ""}],
+ "total_cost": 1.25, "note": "heuristic"}
+ hosted["top5_week"] = list(hosted["skills"])
+ old = _run_async("loadSkillAttribution()", fns, els, hosted)["skill-leaderboard-content"]
+ assert "$1.25" in old and not _badges(old)
+
+
+def test_the_team_and_cache_cards_label_their_costs():
+ team = provenance.stamp({"teams": [{"label": "Eng", "cost_usd": 1.5,
+ "tokens": 9, "sessions": 2,
+ "runtimes": ["codex"]}],
+ "window_days": 7},
+ cost_basis_surfaces.by_team_entries(7))
+ els = {"usage-by-team-title": {}, "usage-by-team-card": {},
+ "usage-by-team-content": {"innerHTML": ""}}
+ got = _run_async("loadUsageByTeam()",
+ ("costCardText", "_e", "loadUsageByTeam"), els, team)
+ html = got["usage-by-team-content"]
+ assert _badges(html.split("
")[0]) == ["published rates"]
+ assert "$1.50" in html
+ risk = provenance.stamp({"affected_sessions": 1, "total_sessions": 3,
+ "total_expiry_count": 3,
+ "total_write_cost_usd": 0.4, "total_saved_usd": 0.1,
+ "max_idle_gap_sec": 900},
+ cost_basis_surfaces.cache_risk_entries("duckdb"))
+ els = {"cache-risk-title": {}, "cache-risk-card": {},
+ "cache-risk-content": {"innerHTML": ""}}
+ html = _run_async("loadCacheRisk()", ("loadCacheRisk",), els, risk)["cache-risk-content"]
+ assert "published rates" in _badges(html)
+ assert "$0.40" in html and "$0.10" in html and "$0.30" in html
+ assert "$0.400" not in html
+
+
+def test_the_transcript_turn_and_tool_chips_carry_the_basis():
+ entry = cost_basis_surfaces.transcript_entries("duckdb")["messages[].cost_usd"]
+ turn = {"turn": 1, "anchor": {"role": "user", "content": "do it"},
+ "firstTs": 0, "lastTs": 0, "toolCount": 1, "errorCount": 0,
+ "tokens": 120, "cost": 0.03,
+ "events": [{"role": "assistant", "type": "tool_use", "content": "",
+ "cost": 0.02, "tokens": 120, "originalIndex": 3}]}
+ stubs = ("function _renderCompactionEvent() { return ''; }\n"
+ "function _renderHistoryGap() { return ''; }\n"
+ "function _renderRawPayload() { return ''; }\n"
+ "function _renderToolDiveChip() { return ''; }\n")
+ got = _run(stubs + "window._replayCostEntry = %s;"
+ "var withBasis = _renderTurnChapter(%s, -1);"
+ "window._replayCostEntry = null;"
+ "console.log(JSON.stringify({html: withBasis, old: _renderTurnChapter(%s, -1)}));"
+ % (json.dumps(entry), json.dumps(turn), json.dumps(turn)),
+ fns=("_turnDurationLabel", "_turnAnchorPreview",
+ "_renderReplayEvent", "_renderTurnChapter"))
+ html = got["html"]
+ figs = re.findall(r']*title="([^"]*)"[^>]*>([^<]*)', html)
+ shown = {text for _tip, text in figs}
+ assert {"$0.03", "$0.02"} <= shown, figs
+ for tip, _text in figs:
+ assert "published rates" in tip, tip
+ # One badge in the transcript header, not one per chip.
+ assert not _badges(html)
+ assert "$0.03" in got["old"] and "$0.02" in got["old"]
+ assert "cmProv.of(data, 'messages[].cost_usd')" in _app()
diff --git a/tests/test_provenance_render_coverage.py b/tests/test_provenance_render_coverage.py
index 4a19a95772..4d107e134a 100644
--- a/tests/test_provenance_render_coverage.py
+++ b/tests/test_provenance_render_coverage.py
@@ -12,9 +12,33 @@
without a basis fails here on the day it lands, and every conversion of an
old one lowers the ceiling below.
-``UNBADGED_CEILING`` may only ever be edited DOWNWARD. If a change needs it
-raised, the change is adding an unlabelled figure, which is the thing this
-file exists to stop.
+``UNBADGED_CEILING`` and ``TAB_UNBADGED_CEILING`` may only ever be edited
+DOWNWARD. If a change needs one raised, the change is adding an unlabelled
+figure, which is the thing this file exists to stop.
+
+Per tab (REQ-OBS-CEA-025, vivekchand/clawmetry#5937)
+----------------------------------------------------
+The whole-file count cannot say WHERE an unlabelled figure is, and a count
+held under a ceiling can quietly trade a converted figure on a quiet panel
+for a new one on the Overview hero. So the Overview, Usage and Sessions tabs
+are also counted on their own, and their scope is discovered, not listed:
+
+* a function belongs to a tab when it touches an element id that tab's
+ template (``clawmetry/templates/tabs/.html``) defines;
+* so do the helpers those functions call, down three levels, unless the
+ helper touches ids of a different tab only;
+* a money render is a line that builds a currency string, or a call to a
+ local formatter that does (discovered the same way: a short function whose
+ ``return`` builds one).
+
+Criteria declared here:
+
+* AC-OBS-CEA-025.10 -- the unbadged count on Overview, Usage and Sessions
+ does not grow, discovered from the tab templates:
+ ``test_each_tab_unbadged_money_renders_do_not_grow``,
+ ``test_the_tab_discovery_reaches_the_cost_surfaces``.
+* AC-OBS-CEA-025.8 -- the named cost surfaces render every figure through
+ the shared component: ``test_the_named_cost_surfaces_route_every_figure``.
"""
import os
import re
@@ -24,6 +48,7 @@
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
APP_JS = os.path.join(REPO, "clawmetry", "static", "js", "app.js")
+TABS_DIR = os.path.join(REPO, "clawmetry", "templates", "tabs")
# A line that builds a currency string for the screen.
_MONEY_RENDER = re.compile(
@@ -32,18 +57,43 @@
# Routed through the shared component, so the figure carries its basis.
# ``cmFmtMoney`` alone does NOT count: sharing a formatter is reuse, not
-# provenance, and this guard is about the badge.
+# provenance, and this guard is about the badge. ``cmCostFigure`` is the
+# shared figure for a cost whose basis may not have arrived yet (it badges
+# whenever an entry exists and never invents one).
_BADGED = ("cmProv.figure", "cmProv.money", "cmProv.score", "cmProv.badge",
- "cmMoney(", "cmScore(", "cmFigure(", "cmProvBadge(")
+ "cmProv.costFigure", "cmMoney(", "cmScore(", "cmFigure(",
+ "cmProvBadge(", "cmCostFigure(")
# Measured 2026-08-25, when the shared badge shipped. Ratchet only downward.
+# 62 -> 43 (2026-09-14, #5937): the remaining Usage cards, the Overview hero
+# chip and the Sessions transcript chips.
#
# It over-counts slightly: a plan price in the upgrade overlay and the two
# lines that DETECT the old "$0.00" placeholder both match the pattern and
# are not figures. Over-counting is the safe direction for a ratchet, and
# ``test_the_ceiling_is_not_padded`` keeps the slack from growing into room
# for a real one to hide in.
-UNBADGED_CEILING = 62
+UNBADGED_CEILING = 43
+
+# Per tab, discovered from the tab templates. Ratchet only downward.
+# Overview's remainder is the run and cohort comparison panels, the anomaly
+# panel, the waste summary and loop sources; Sessions' is similar runs and
+# the orchestration panel. None of their payloads carries a basis yet.
+TAB_UNBADGED_CEILING = {"overview": 24, "usage": 0, "transcripts": 5}
+
+# The surfaces REQ-OBS-CEA-025.8 names, by the function that renders them.
+# Each must render every money figure through the shared component.
+NAMED_COST_SURFACES = {
+ "overview": ("_renderOverviewHero", "_cmHeroCostChip"),
+ "usage": ("_sfRender", "_renderEfficiencyCardInner", "_cmEffIdeaRowHtml",
+ "renderCacheHitRateCard", "renderRoutingAdvisorCard",
+ "loadCostForecast", "loadCacheRisk", "loadCompressionPotential",
+ "loadCacheAnalytics", "renderCostComparison",
+ "renderSpendOptimization", "renderPluginPieChart",
+ "loadSkillAttribution", "loadAllSkills", "_skillCostTableHtml",
+ "loadUsageByTeam"),
+ "transcripts": ("_renderReplayEvent", "_renderTurnChapter"),
+}
# A badged figure's legacy fallback branch usually lands a line or two below
# the shared call. Count the render as covered when the shared component
@@ -51,8 +101,17 @@
_CONTEXT_LINES = 3
+def _lines():
+ return open(APP_JS, encoding="utf-8").read().splitlines()
+
+
+def _is_badged(lines, i):
+ window = "\n".join(lines[max(0, i - _CONTEXT_LINES):i + 1])
+ return any(k in window for k in _BADGED)
+
+
def _unbadged():
- lines = open(APP_JS, encoding="utf-8").read().splitlines()
+ lines = _lines()
out = []
for i, line in enumerate(lines):
stripped = line.strip()
@@ -60,13 +119,116 @@ def _unbadged():
continue
if not _MONEY_RENDER.search(line):
continue
- window = "\n".join(lines[max(0, i - _CONTEXT_LINES):i + 1])
- if any(k in window for k in _BADGED):
+ if _is_badged(lines, i):
continue
out.append((i + 1, stripped))
return out
+# ── Discovery ────────────────────────────────────────────────────────────────
+
+def _functions(lines):
+ """Top-level function name -> list of (first line, end line) spans."""
+ starts = [(i, m.group(1)) for i, line in enumerate(lines)
+ for m in [re.match(r"^(?:async )?function ([\w$]+)\(", line)] if m]
+ spans = {}
+ for k, (i, name) in enumerate(starts):
+ end = starts[k + 1][0] if k + 1 < len(starts) else len(lines)
+ spans.setdefault(name, []).append((i, end))
+ return spans
+
+
+def _formatters(lines):
+ """Names of short functions, at any depth, whose return builds money."""
+ out = set()
+ for i, line in enumerate(lines):
+ m = re.search(r"\bfunction ([\w$]+)\s*\([^)]*\)\s*\{", line)
+ if not m:
+ continue
+ body = [line[m.end():]]
+ j = i
+ while "}" not in body[-1] or body[-1].strip() not in ("}", "};") and j == i and not body[-1].rstrip().endswith("}"):
+ j += 1
+ if j >= len(lines) or j - i > 8:
+ break
+ body.append(lines[j])
+ if lines[j].strip() in ("}", "};"):
+ break
+ if j - i > 8:
+ continue
+ text = "\n".join(body)
+ if re.search(r"\breturn\b", text) and _MONEY_RENDER.search(text):
+ out.add(m.group(1))
+ return out
+
+
+def _tab_ids():
+ tabs = {}
+ for fn in os.listdir(TABS_DIR):
+ if fn.endswith(".html"):
+ html = open(os.path.join(TABS_DIR, fn), encoding="utf-8").read()
+ tabs[fn[:-5]] = set(re.findall(r'\bid="([^"]+)"', html))
+ return tabs
+
+
+def _ids_touched(text):
+ return (set(re.findall(r"getElementById\(\s*['\"]([^'\"]+)['\"]", text))
+ | set(re.findall(r"querySelector(?:All)?\(\s*['\"]#([\w-]+)", text)))
+
+
+def _tab_scope(tab, lines=None):
+ """The functions that render ``tab``: seeds by template id, then callees."""
+ lines = lines or _lines()
+ funcs = _functions(lines)
+ tabs = _tab_ids()
+
+ def body(name):
+ return "\n".join("\n".join(lines[a:b]) for a, b in funcs[name])
+
+ owners = {}
+ for name in funcs:
+ touched = _ids_touched(body(name))
+ owners[name] = {t for t, ids in tabs.items() if touched & ids}
+ scope = {n for n, t in owners.items() if tab in t}
+ frontier = set(scope)
+ for _depth in range(3):
+ nxt = set()
+ for name in frontier:
+ for callee in set(re.findall(r"\b([\w$]+)\s*\(", body(name))) & set(funcs):
+ if callee in scope or (owners[callee] and tab not in owners[callee]):
+ continue
+ nxt.add(callee)
+ scope |= nxt
+ frontier = nxt
+ return scope, funcs
+
+
+def _unbadged_in(names, funcs, lines, fmts):
+ calls = [re.compile(r"(? ceiling:
+ report.append("%s: %d unbadged money renders, ceiling %d\n%s" % (
+ tab, len(found), ceiling,
+ "\n".join(" app.js:%d %s: %s" % f for f in found)))
+ # A ceiling far above the count is room for a new one to hide in.
+ assert ceiling - len(found) <= 3, (
+ "TAB_UNBADGED_CEILING[%r] is %d but only %d remain. Lower it."
+ % (tab, ceiling, len(found)))
+ assert not report, (
+ "A dollar figure on these tabs renders with no basis. Route it through "
+ "window.cmProv / window.cmCostFigure with the entry its payload "
+ "sends:\n" + "\n".join(report))
+
+
+def test_the_tab_discovery_reaches_the_cost_surfaces():
+ """A discovery that found nothing would pass the ratchet vacuously. The
+ renderers of every named surface must be inside their tab's scope."""
+ for tab, names in NAMED_COST_SURFACES.items():
+ scope, funcs = _tab_scope(tab)
+ assert len(scope) >= 5, "%s tab discovery found %r" % (tab, scope)
+ for name in names:
+ assert name in funcs, "%s() is gone from app.js" % name
+ assert name in scope, (
+ "%s() renders the %s tab but discovery does not reach it" % (name, tab))
+
+
+def test_the_named_cost_surfaces_route_every_figure():
+ """AC-OBS-CEA-025.8: no money render in these functions skips the shared
+ component, and each one actually calls it."""
+ lines = _lines()
+ funcs = _functions(lines)
+ fmts = _formatters(lines)
+ for tab, names in NAMED_COST_SURFACES.items():
+ found = _unbadged_in(names, funcs, lines, fmts)
+ assert not found, "%s tab: unlabelled cost renders:\n%s" % (
+ tab, "\n".join(" app.js:%d %s: %s" % f for f in found))
+ for name in names:
+ text = "\n".join("\n".join(lines[a:b]) for a, b in funcs[name])
+ # Either it calls the shared component itself, or it hands the
+ # figure to another named renderer of the same tab that does.
+ delegates = [n for n in names if n != name
+ and re.search(r"(?