diff --git a/docs/features.md b/docs/features.md index 4b138f3..1edf103 100644 --- a/docs/features.md +++ b/docs/features.md @@ -226,6 +226,19 @@ skill loads, filtered by machine or project. Missing usage is identified rather than guessed. These are usage metrics, not a billing dashboard or a measurement of task correctness; the chart below uses sample history. +The total and token breakdown show exact counts: uncached input + cache reads + +cache writes + output (including reasoning). Usage is counted across all recorded +providers and models in the selected projects and machines, not just the currently +selected model. Anthropic, OpenAI, Codex, DeepSeek, and compatible/local endpoints +use their reported usage; missing counts are never estimated from text. + +Cache writes are input tokens the provider reports saving for reuse, not files +written by Ava. Anthropic reports cache creation explicitly; many other endpoints +report cache reads but no separate write count. An unreported cache-write count is +shown as **—**, labeled **Not reported by these providers**, rather than zero. +When only some requests report cache writes, the breakdown shows how many did. +The total includes known counts only; it is not a billing estimate. + ![Seven-day sample usage with token trends, tool activity, and observed skill loads](assets/demo/session-information.png) ## Git and terminals diff --git a/src/ava/app/analytics.py b/src/ava/app/analytics.py index 3a77a68..4256b30 100644 --- a/src/ava/app/analytics.py +++ b/src/ava/app/analytics.py @@ -30,7 +30,7 @@ def merge_intervals(intervals: list) -> list[list[int]]: def totals(days: list[dict]) -> dict: - names = (*TOKEN_FIELDS, "responses", "missing_usage", "tools", "tool_errors", "skills", "runs", "run_ms", "active_ms") + names = (*TOKEN_FIELDS, "cache_write_reports", "responses", "missing_usage", "tools", "tool_errors", "skills", "runs", "run_ms", "active_ms") result = {name: sum(day.get(name, 0) for day in days) for name in names} result["tokens"] = sum(result[key] for key in TOKEN_FIELDS if key != "reasoning") return result @@ -45,7 +45,8 @@ def __init__(self, path: Path) -> None: os.chmod(path, 0o600) self.db.row_factory = sqlite3.Row try: - if self.db.execute("PRAGMA user_version").fetchone()[0] not in (0, 1): + version = self.db.execute("PRAGMA user_version").fetchone()[0] + if version not in (0, 1, 2): raise sqlite3.DatabaseError("Analytics cache uses a different aggregation version.") self.db.executescript(""" PRAGMA journal_mode=WAL; @@ -78,8 +79,13 @@ def __init__(self, path: Path) -> None: CREATE TABLE IF NOT EXISTS days ( zone TEXT, project TEXT, day TEXT, start INTEGER, end INTEGER, body TEXT, PRIMARY KEY(zone,project,day)); - PRAGMA user_version=1; """) + with self.db: + if version < 2: + # Facts retain the original disjoint token categories. Only + # derived days need rebuilding for inclusive output/reporting. + self.db.execute("DELETE FROM days") + self.db.execute("PRAGMA user_version=2") except sqlite3.DatabaseError: self.db.close() raise @@ -222,10 +228,15 @@ def _day(self, day: date, zone: ZoneInfo, project: str) -> dict: end = int(datetime.combine(day + timedelta(days=1), datetime.min.time(), zone).timestamp() * 1000) where = "logs.visible=1 AND logs.error=''" + (" AND logs.project=?" if project else "") params = (project,) if project else () - usage = self.db.execute(f"""SELECT COUNT(*) responses, SUM(input IS NULL OR output IS NULL) missing_usage, + usage = self.db.execute(f"""SELECT COUNT(*) responses, COUNT(cache_write) cache_write_reports, + SUM(input IS NULL OR output IS NULL) missing_usage, {','.join('SUM('+key+') '+key for key in TOKEN_FIELDS)} FROM attempts JOIN logs USING(path) WHERE {where} AND at>=? AND at=? AND at r["offset"] for r in rows), "active_sessions": len(active)} diff --git a/src/ava/app/desktop/analytics.py b/src/ava/app/desktop/analytics.py index 23f3448..575ec4d 100644 --- a/src/ava/app/desktop/analytics.py +++ b/src/ava/app/desktop/analytics.py @@ -184,12 +184,20 @@ def _assemble(self) -> None: notices.append(machine.name + ": indexing history; totals are still updating") if report["unavailable"] or report["incomplete"] or report.get("error"): notices.append(machine.name + ": some history is unavailable or incomplete") + legacy_tokens = report.get("token_accounting_version", 1) < 2 + if legacy_tokens: + notices.append("Update the Ava backend on " + machine.name + " for cache-write reporting availability") for source in report["days"][-self._days:]: if not first <= source["date"] <= last.isoformat(): continue target = days.setdefault(source["date"], {"date": source["date"], "intervals": [], "tool_counts": [], "skill_counts": []}) for name in (*TOKEN_FIELDS, "tokens", "responses", "missing_usage", "tools", "tool_errors", "skills", "runs", "run_ms"): - target[name] = target.get(name, 0) + source[name] + value = source[name] + if legacy_tokens and name in ("output", "tokens"): + value += source["reasoning"] + target[name] = target.get(name, 0) + value + # Older backends cannot distinguish unreported writes from zero. + target["cache_write_reports"] = target.get("cache_write_reports", 0) + source.get("cache_write_reports", 0) for name in ("intervals", "tool_counts", "skill_counts"): target[name].extend(source[name]) series = sorted(days.values(), key=lambda day: day["date"]) diff --git a/src/ava/app/desktop/qml/AnalyticsPane.qml b/src/ava/app/desktop/qml/AnalyticsPane.qml index 5629dd2..e48f4e8 100644 --- a/src/ava/app/desktop/qml/AnalyticsPane.qml +++ b/src/ava/app/desktop/qml/AnalyticsPane.qml @@ -75,7 +75,7 @@ Pane { contentItem: ColumnLayout { spacing: Theme.spaceSm Label { text: card.title; color: Theme.secondaryText; font.pixelSize: Theme.caption } - Label { text: card.value; color: Theme.text; font.pixelSize: 28; font.weight: Font.DemiBold } + Label { Layout.fillWidth: true; text: card.value; color: Theme.text; font.pixelSize: 28; font.weight: Font.DemiBold; fontSizeMode: Text.HorizontalFit; minimumPixelSize: Theme.body } Label { Layout.fillWidth: true text: pane.report.reports ? card.detail : pane.analytics.loading ? "Loading statistics…" : "No available statistics" @@ -253,7 +253,7 @@ Pane { uniformCellWidths: true columnSpacing: Theme.spaceMd rowSpacing: Theme.spaceMd - Card { objectName: "analyticsTokens"; Layout.fillWidth: true; Layout.fillHeight: true; title: "Total tokens"; value: pane.report.reports ? pane.compact(pane.report.totals.tokens) : "—"; detail: pane.number(pane.report.totals.responses) + " model requests" } + Card { objectName: "analyticsTokens"; Layout.fillWidth: true; Layout.fillHeight: true; title: "Total tokens"; value: pane.report.reports ? pane.number(pane.report.totals.tokens) : "—"; detail: pane.number(pane.report.totals.responses) + " model requests" } Card { objectName: "analyticsTime"; Layout.fillWidth: true; Layout.fillHeight: true; title: "Active time"; value: pane.report.reports ? pane.duration(pane.report.totals.active_ms) : "—"; detail: "Overlapping sessions counted once" } Card { Layout.fillWidth: true; Layout.fillHeight: true; title: "Tool calls"; value: pane.report.reports ? pane.compact(pane.report.totals.tools) : "—"; detail: pane.number(pane.report.totals.tool_errors) + " returned errors" } Card { Layout.fillWidth: true; Layout.fillHeight: true; title: "Skills loaded"; value: pane.report.reports ? pane.number(pane.report.totals.skills) : "—"; detail: "Recorded instruction loads" } @@ -409,7 +409,7 @@ Pane { contentItem: ColumnLayout { spacing: Theme.spaceLg Label { text: "Token breakdown"; font.pixelSize: Theme.sectionTitle; font.weight: Font.DemiBold } - Label { text: "How reported tokens are used"; color: Theme.secondaryText; font.pixelSize: Theme.caption } + Label { text: "Exact reported counts · sum to total tokens"; color: Theme.secondaryText; font.pixelSize: Theme.caption; Layout.fillWidth: true; wrapMode: Text.WordWrap } Row { id: tokenStack Layout.fillWidth: true @@ -432,6 +432,7 @@ Pane { id: token required property var modelData required property int index + readonly property bool unreported: modelData.key === "cache_write" && pane.report.totals.responses > 0 && !pane.report.totals.cache_write_reports && !pane.report.totals.cache_write Layout.fillWidth: true spacing: Theme.spaceXs RowLayout { @@ -439,12 +440,12 @@ Pane { spacing: Theme.spaceSm Rectangle { implicitWidth: 8; implicitHeight: 8; radius: 2; color: pane.colors[token.index] } Label { Layout.fillWidth: true; text: token.modelData.name; font.pixelSize: Theme.body } - Label { text: pane.report.reports ? pane.compact(pane.report.totals[token.modelData.key]) : "—"; font.pixelSize: Theme.body; font.weight: Font.DemiBold } - Label { text: pane.report.reports ? pane.percentage(pane.report.totals[token.modelData.key], pane.report.totals.tokens) : "—"; Layout.preferredWidth: 36; horizontalAlignment: Text.AlignRight; font.pixelSize: Theme.caption; color: Theme.secondaryText } + Label { objectName: "analyticsTokenValue_" + token.modelData.key; text: pane.report.reports && !token.unreported ? pane.number(pane.report.totals[token.modelData.key]) : "—"; font.pixelSize: Theme.body; font.weight: Font.DemiBold } + Label { text: pane.report.reports && !token.unreported ? pane.percentage(pane.report.totals[token.modelData.key], pane.report.totals.tokens) : "—"; Layout.preferredWidth: 36; horizontalAlignment: Text.AlignRight; font.pixelSize: Theme.caption; color: Theme.secondaryText } } - Label { text: token.modelData.detail; leftPadding: 16; color: Theme.secondaryText; font.pixelSize: Theme.captionSmall } + Label { text: token.unreported ? "Not reported by these providers" : token.modelData.key === "cache_write" && pane.report.totals.cache_write_reports > 0 ? "Reported on " + pane.number(pane.report.totals.cache_write_reports) + " of " + pane.number(pane.report.totals.responses) + " requests" : token.modelData.detail; Layout.fillWidth: true; wrapMode: Text.WordWrap; leftPadding: 16; color: Theme.secondaryText; font.pixelSize: Theme.captionSmall } HoverHandler { id: tokenHover } - NativeToolTip { visible: tokenHover.hovered && pane.report.reports > 0; text: pane.number(pane.report.totals[token.modelData.key]) + " " + token.modelData.name.toLowerCase() + " tokens" } + NativeToolTip { visible: tokenHover.hovered && pane.report.reports > 0 && !token.unreported; text: pane.number(pane.report.totals[token.modelData.key]) + " " + token.modelData.name.toLowerCase() + " tokens" } } } Item { Layout.fillHeight: true } diff --git a/src/ava/app/desktop/quick_chat.py b/src/ava/app/desktop/quick_chat.py index b8a9547..20cefaf 100644 --- a/src/ava/app/desktop/quick_chat.py +++ b/src/ava/app/desktop/quick_chat.py @@ -23,7 +23,7 @@ class QuickChatShortcut(QObject): def __init__(self, parent: QObject | None = None) -> None: super().__init__(parent) - self._carbon = None + self._carbon: ctypes.CDLL | None = None self._handler = ctypes.c_void_p() self._hotkey = ctypes.c_void_p() self.error = "" diff --git a/tests/test_analytics.py b/tests/test_analytics.py index d8295ab..168d482 100644 --- a/tests/test_analytics.py +++ b/tests/test_analytics.py @@ -61,14 +61,16 @@ def test_analytics_totals_overlap_dst_append_restart_and_replacement(tmp_path, s try: catch_up(index, sources) report = index.query(7, 'America/Los_Angeles', now=NOW) - assert report['totals']['tokens'] == 440 # 190 + 250; reasoning/1h cache write are subsets. + assert report['totals']['tokens'] == 450 # 200 + 250; only 1h cache write is a logged subset. + assert report['totals']['output'] == 90 + assert report['totals']['cache_write_reports'] == 1 assert report['totals']['responses'] == 2 and not report['totals']['missing_usage'] assert report['totals']['active_ms'] == 3*3600_000 assert report['totals']['run_ms'] == 4*3600_000 assert report['totals']['tools'] == report['totals']['skills'] == 1 dst = next(day for day in report['days'] if day['date'] == '2026-11-01') assert dst['end'] - dst['start'] == 25*3600_000 - assert index.query(project='p1', now=NOW)['totals']['tokens'] == 190 + assert index.query(project='p1', now=NOW)['totals']['tokens'] == 200 frames = index.frames_read computed = index.days_computed for _ in range(3): @@ -78,7 +80,7 @@ def test_analytics_totals_overlap_dst_append_restart_and_replacement(tmp_path, s put(first, [('2026-11-02T12:00:00.000Z', {'kind':'attempt/timing','attempt_id':'missing','elapsed_ms':20})], append=True, start=len(rows)) index.scan(sources) changed = index.query(7, 'America/Los_Angeles', now=NOW) - assert changed['totals']['missing_usage'] == 1 and changed['totals']['tokens'] == 440 + assert changed['totals']['missing_usage'] == 1 and changed['totals']['tokens'] == 450 assert index.days_computed == computed + 1, 'Only the affected day should be recalculated' finally: index.close() @@ -86,7 +88,7 @@ def test_analytics_totals_overlap_dst_append_restart_and_replacement(tmp_path, s try: catch_up(index, sources) assert index.frames_read == 0, 'Restart validates a physical checkpoint without decompressing history' - assert index.query(now=NOW)['totals']['tokens'] == 440 + assert index.query(now=NOW)['totals']['tokens'] == 450 replacement = first.with_name('replacement'+suffix) put(replacement, [header(), ('2026-11-01T10:00:00.000Z', {'kind':'usage','attempt_id':'new','tokens':{'input':1,'output':2}})]) replacement.replace(first) @@ -124,6 +126,101 @@ def test_analytics_checkpoints_only_complete_frames(tmp_path): index.close() +@pytest.mark.parametrize('provider,model,payloads,expected', [ + # Expected counts come from the wire contract, not the normalized Usage object. + ('anthropic', 'claude-sonnet', [{'input_tokens': 100, 'cache_read_input_tokens': 20, + 'cache_creation_input_tokens': 30, 'cache_creation': {'ephemeral_1h_input_tokens': 10}}, + {'output_tokens': 50}], (200, 50, 30, 1, 0)), + ('anthropic', 'claude-opus-thinking', [{'input_tokens': 100, 'output_tokens': 50, + 'cache_creation_input_tokens': 0}], (150, 50, 0, 1, 0)), + ('custom-anthropic', 'arbitrary-model', [{'input_tokens': 100}, {'output_tokens': 50}], (150, 50, 0, 0, 0)), + ('openai', 'gpt-reasoning', [{'prompt_tokens': 100, 'completion_tokens': 50, + 'prompt_tokens_details': {'cached_tokens': 20}, + 'completion_tokens_details': {'reasoning_tokens': 30}}], (150, 50, 0, 0, 0)), + ('openai', 'gpt-nonreasoning', [{'prompt_tokens': 100, 'completion_tokens': 50}], (150, 50, 0, 0, 0)), + ('codex', 'codex-reasoning', [{'input_tokens': 100, 'output_tokens': 50, + 'input_tokens_details': {'cached_tokens': 20}, + 'output_tokens_details': {'reasoning_tokens': 30}}], (150, 50, 0, 0, 0)), + ('codex', 'codex-no-details', [{'input_tokens': 100, 'output_tokens': 50}], (150, 50, 0, 0, 0)), + ('deepseek', 'deepseek-reasoner', [{'prompt_tokens': 100, 'prompt_cache_hit_tokens': 20, + 'prompt_cache_miss_tokens': 80, 'completion_tokens': 50, + 'completion_tokens_details': {'reasoning_tokens': 30}}], (150, 50, 0, 0, 0)), + ('deepseek', 'deepseek-chat', [{'prompt_cache_hit_tokens': 20, + 'prompt_cache_miss_tokens': 80, 'completion_tokens': 50}], (150, 50, 0, 0, 0)), + ('custom-openai', 'arbitrary-model', [{'prompt_tokens': 100, 'completion_tokens': 50}], (150, 50, 0, 0, 0)), + ('llamacpp', 'local-model', [{}], (0, 0, 0, 0, 1)), + ('custom-openai', 'partial-usage', [{'completion_tokens': 50, + 'completion_tokens_details': {'reasoning_tokens': 30}}], (50, 50, 0, 0, 1)), +]) +def test_provider_usage_survives_logging_and_analytics(home, project, tmp_path, provider, model, payloads, expected): + from types import SimpleNamespace + + from ava.agent.step import Timing, append_accounting, merge_usage + from ava.llm.anthropic import _emit_usage as anthropic_usage + from ava.llm.codex import _emit_usage as codex_usage + from ava.llm.openai import _emit_openai_usage + from ava.llm.provider import Usage + from ava.session import Log + + emit = (anthropic_usage if 'anthropic' in provider else codex_usage if provider == 'codex' + else _emit_openai_usage) + events = [] + for payload in payloads: + emit(payload, events.append) + usage = Usage() + for event in events: + merge_usage(usage, event.usage) + log = Log.create_default(project, provider, model) + try: + state = SimpleNamespace(append=log.append) + # Multiple turns with distinct attempts; timing must not double-count usage. + for attempt in ('first', 'second'): + append_accounting(state, attempt, usage, Timing(elapsed_ms=10)) + log.sync() + sources = {str(log.path): ('project', False)} + finally: + log.close() + index = AnalyticsIndex(tmp_path/'provider-analytics.sqlite3') + try: + index.scan(sources) + report = index.query() + total = report['totals'] + keys = ('tokens', 'output', 'cache_write', 'cache_write_reports', 'missing_usage') + assert tuple(total[key] for key in keys) == tuple(value * 2 for value in expected) + assert total['responses'] == 2 + assert total['tokens'] == sum(total[key] for key in ('input', 'cached_read', 'cache_write', 'output')) + assert report['token_accounting_version'] == 2 + finally: + index.close() + + +def test_analytics_upgrades_derived_days_without_replaying_history(tmp_path): + path = tmp_path/'session.jsonl' + put(path, [header(), ('2026-11-01T10:00:00.000Z', {'kind': 'usage', 'attempt_id': '1', + 'tokens': {'input': 100, 'output': 20, 'reasoning': 30}})]) + sources = {str(path): ('p1', False)} + cache = tmp_path/'analytics.sqlite3' + index = AnalyticsIndex(cache) + try: + catch_up(index, sources) + index.query(now=NOW) + # Reproduce an existing v1 cache with stale day aggregates and intact facts. + with index.db: + index.db.execute("UPDATE days SET body='{}'") + index.db.execute('PRAGMA user_version=1') + finally: + index.close() + index = AnalyticsIndex(cache) + try: + catch_up(index, sources) + assert index.query(now=NOW)['totals']['tokens'] == 150 + assert index.query(now=NOW)['totals']['output'] == 50 + assert index.frames_read == 0 + assert index.db.execute('PRAGMA user_version').fetchone()[0] == 2 + finally: + index.close() + + @pytest.mark.skipif(os.environ.get('AVA_ANALYTICS_PERF') != '1', reason='Explicit million-event performance run') def test_analytics_million_events(tmp_path): import threading diff --git a/tests/test_desktop.py b/tests/test_desktop.py index b88ce38..34a2d8b 100644 --- a/tests/test_desktop.py +++ b/tests/test_desktop.py @@ -347,7 +347,7 @@ def chunk(delta, reason=None): chunk({"content": "世界"}) chunk({}, "stop") if request.get("model") == "fixture-analytics": - self.wfile.write(b'data: {"choices":[],"usage":{"prompt_tokens":1200,"completion_tokens":80,"prompt_tokens_details":{"cached_tokens":200}}}\n\n') + self.wfile.write(b'data: {"choices":[],"usage":{"prompt_tokens":1200,"completion_tokens":80,"prompt_tokens_details":{"cached_tokens":200},"completion_tokens_details":{"reasoning_tokens":30}}}\n\n') self.wfile.write(b"data: [DONE]\n\n") self.wfile.flush() except (BrokenPipeError, ConnectionResetError): @@ -6787,7 +6787,17 @@ def test_desktop_analytics_history_filters_and_live_skill_usage(desktop, model_s assert first_ms < 500 and gap < 150, (first_ms, gap) assert analytics.data['totals']['active_ms'] == 14*60000 assert len(analytics.data['days']) == 7 - assert find_item(window, 'analyticsTokens').property('value') == '349.6k' + def displayed_tokens(text): + return int(''.join(character for character in text if character.isdecimal())) + + def assert_displayed_token_sum(): + values = [find_item(window, 'analyticsTokenValue_' + key).property('text') + for key in ('input', 'cached_read', 'cache_write', 'output')] + assert sum(displayed_tokens(value) for value in values) == displayed_tokens( + find_item(window, 'analyticsTokens').property('value')) + + assert displayed_tokens(find_item(window, 'analyticsTokens').property('value')) == 349650 + assert_displayed_token_sum() save_screenshot(window, 'analytics-week') assert find_item(window, 'pageHeader').property('title') == 'Session information' assert find_item(window, 'analytics7d').property('checked') @@ -6829,6 +6839,7 @@ def test_desktop_analytics_history_filters_and_live_skill_usage(desktop, model_s click(window, 'analytics30d') until(lambda: len(analytics.data['days']) == 30, analytics.changed) assert analytics.data['totals']['tokens'] == 860250 + assert_displayed_token_sum() assert find_item(window, 'analytics30d').property('checked') assert not find_item(window, 'analytics7d').property('checked') switch_ms = (time.perf_counter()-started)*1000 @@ -6881,6 +6892,45 @@ def test_desktop_analytics_history_filters_and_live_skill_usage(desktop, model_s assert analytics.data['totals']['tools'] == 1 assert analytics.data['skills'][0]['name'] == 'review' assert analytics.data['totals']['missing_usage'] == 1 + assert analytics.data['totals']['reasoning'] == 30 + assert_displayed_token_sum() + + # An older remote backend sends exclusive output. Normalize exactly once, + # without modifying its cached report on repeated assembly. + from copy import deepcopy + + saved_cache = deepcopy(analytics._cache) + expected_totals = dict(analytics.data['totals']) + for report in analytics._cache.values(): + report.pop('token_accounting_version', None) + for day in report['days']: + day['output'] -= day['reasoning'] + day['tokens'] -= day['reasoning'] + day.pop('cache_write_reports', None) + analytics._assemble() + assert analytics.data['totals']['tokens'] == expected_totals['tokens'] + assert analytics.data['totals']['output'] == expected_totals['output'] + assert 'Update the Ava backend' in analytics.data['notice'] + analytics._assemble() + assert analytics.data['totals']['tokens'] == expected_totals['tokens'] + assert_displayed_token_sum() + + analytics._cache = deepcopy(saved_cache) + for report in analytics._cache.values(): + for day in report['days']: + day['tokens'] -= day['cache_write'] + day['cache_write'] = day['cache_write_reports'] = 0 + analytics._assemble() + assert find_item(window, 'analyticsTokenValue_cache_write').property('text') == '—' + # A provider explicitly reporting zero is a different measurement. + for report in analytics._cache.values(): + for day in report['days']: + day['cache_write_reports'] = day['responses'] + analytics._assemble() + assert find_item(window, 'analyticsTokenValue_cache_write').property('text') == '0' + assert_displayed_token_sum() + analytics._cache = saved_cache + analytics._assemble() save_screenshot(window, 'analytics-live') for kind in ('tools', 'skills'): entry = find_item(window, f'analyticsRank_{kind}_0')