From 864db64a7bb168c22d8c65bd1a235c79927b3d15 Mon Sep 17 00:00:00 2001 From: Yuxin Qiao <104957188+Yuxin-Qiao@users.noreply.github.com> Date: Sat, 10 Oct 2026 23:58:29 +0800 Subject: [PATCH 1/2] Show cache reuse and sample coverage in session performance --- .../SpendSessionPerformanceView.swift | 49 ++++++++++++------- .../SpendSessionPerformanceTests.swift | 33 +++++++++++-- 2 files changed, 60 insertions(+), 22 deletions(-) diff --git a/Sources/CodexBar/SpendSessionPerformanceView.swift b/Sources/CodexBar/SpendSessionPerformanceView.swift index 582f8382ee..75f82cbd5c 100644 --- a/Sources/CodexBar/SpendSessionPerformanceView.swift +++ b/Sources/CodexBar/SpendSessionPerformanceView.swift @@ -46,6 +46,9 @@ private struct SpendPerformanceMetricStrip: View { Text(metric.value) .font(.system(.body, design: .rounded, weight: .semibold)) .foregroundStyle(.primary) + if let note = metric.note { + Text(note).font(.caption2).foregroundStyle(.secondary) + } } .fixedSize() .frame(maxWidth: .infinity, alignment: .leading) @@ -55,10 +58,15 @@ private struct SpendPerformanceMetricStrip: View { } VStack(spacing: 6) { ForEach(self.metrics) { metric in - HStack(alignment: .firstTextBaseline) { - Text(metric.label).foregroundStyle(.secondary) - Spacer(minLength: 12) - Text(metric.value).fontWeight(.semibold).foregroundStyle(.primary) + VStack(alignment: .leading, spacing: 3) { + HStack(alignment: .firstTextBaseline) { + Text(metric.label).foregroundStyle(.secondary) + Spacer(minLength: 12) + Text(metric.value).fontWeight(.semibold).foregroundStyle(.primary) + } + if let note = metric.note { + Text(note).font(.caption2).foregroundStyle(.secondary) + } } .font(.caption) .help(metric.help ?? "") @@ -193,15 +201,17 @@ private struct SpendPerformanceModelComparison: View { } func spendSessionPerformanceMetrics(_ summary: CostUsageTurnPerformanceSummary) -> [SpendPerformanceMetric] { - [ + let firstTokenCoverage = L( + "First-token samples: %@ / %@", + codexBarLocalizedInteger(summary.firstTokenSampleCount), + codexBarLocalizedInteger(summary.sampleCount)) + return [ SpendPerformanceMetric( id: "first-token", label: L("spend_performance_first_token"), value: spendPerformanceSeconds(summary.medianFirstTokenMilliseconds), - help: L( - "First-token samples: %@ / %@", - codexBarLocalizedInteger(summary.firstTokenSampleCount), - codexBarLocalizedInteger(summary.sampleCount)) + "\n" + + note: firstTokenCoverage, + help: firstTokenCoverage + "\n" + L("Model first token may be reasoning, before visible answer text.")), SpendPerformanceMetric( id: "output", @@ -211,6 +221,17 @@ func spendSessionPerformanceMetrics(_ summary: CostUsageTurnPerformanceSummary) id: "duration", label: L("spend_performance_duration"), value: spendPerformanceSeconds(summary.medianDurationMilliseconds)), + SpendPerformanceMetric( + id: "cached-input", + label: L("spend_performance_cached_input"), + value: summary.details.cachedInputFraction.map { + L("spend_performance_percent", spendPerformanceNumber($0 * 100)) + } ?? "—", + note: L( + "spend_performance_coverage", + codexBarLocalizedInteger(summary.details.cacheSampleCount), + codexBarLocalizedInteger(summary.sampleCount)), + help: L("spend_performance_cache_help")), ] } @@ -241,16 +262,6 @@ func spendSessionPerformanceDetailMetrics(_ summary: CostUsageTurnPerformanceSum } ?? "—", note: details.outputRateLowerQuartile == nil ? L("Speed range needs 4 completed turns.") : nil, help: L("spend_performance_speed_range_help")), - SpendPerformanceMetric( - id: "cached-input", - label: L("spend_performance_cached_input"), - value: details.cachedInputFraction.map { L("spend_performance_percent", spendPerformanceNumber($0 * 100)) } - ?? "—", - note: L( - "spend_performance_coverage", - codexBarLocalizedInteger(details.cacheSampleCount), - codexBarLocalizedInteger(summary.sampleCount)), - help: L("spend_performance_cache_help")), ] } diff --git a/Tests/CodexBarTests/SpendSessionPerformanceTests.swift b/Tests/CodexBarTests/SpendSessionPerformanceTests.swift index f3f2e4db2b..d989e277e9 100644 --- a/Tests/CodexBarTests/SpendSessionPerformanceTests.swift +++ b/Tests/CodexBarTests/SpendSessionPerformanceTests.swift @@ -123,11 +123,13 @@ struct SpendSessionPerformanceTests { let summary = try #require(CostUsageTurnPerformanceSummary(samples: [sample])) CodexBarLocalizationOverride.$appLanguage.withValue("en") { let metrics = spendSessionPerformanceMetrics(summary) - #expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 s"]) + #expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 s", "—"]) + #expect(metrics[0].note == "First-token samples: 0 / 1") + #expect(metrics[3].note == "0 / 1 turns with cache data") } CodexBarLocalizationOverride.$appLanguage.withValue("zh-Hans") { let metrics = spendSessionPerformanceMetrics(summary) - #expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 秒"]) + #expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 秒", "—"]) } } @@ -146,13 +148,38 @@ struct SpendSessionPerformanceTests { #expect(metrics[0].note != nil) #expect(metrics[1].value == "—") #expect(metrics[1].note != nil) - #expect(metrics[3].value == "80.0%") + let cachedInput = spendSessionPerformanceMetrics(summary).first { $0.id == "cached-input" } + #expect(cachedInput?.value == "80.0%") + #expect(cachedInput?.note == "1 / 1 turns with cache data") #expect(summary.firstTokenSampleCount == 0) #expect(summary.sampleCount == 1) #expect(summary.details.cacheSampleCount == 1) } } + @Test + func `primary observations distinguish measured zero cache reuse from missing records`() throws { + let measured = try #require(CostUsageTurnPerformanceSample( + completedAt: Date(), + outputTokens: 300, + durationMilliseconds: 1000, + firstTokenMilliseconds: 100, + inputTokens: 1000, + cachedInputTokens: 0)) + let unrecorded = try #require(CostUsageTurnPerformanceSample( + completedAt: Date(), + outputTokens: 200, + durationMilliseconds: 2000)) + let summary = try #require(CostUsageTurnPerformanceSummary(samples: [measured, unrecorded])) + CodexBarLocalizationOverride.$appLanguage.withValue("en") { + let metrics = spendSessionPerformanceMetrics(summary) + #expect(metrics.map(\.value) == ["0.1 s", "166.7 tok/s", "1.5 s", "0.0%"]) + #expect(metrics[0].note == "First-token samples: 1 / 2") + #expect(metrics[3].note == "1 / 2 turns with cache data") + #expect(!spendSessionPerformanceDetailMetrics(summary).contains { $0.id == "cached-input" }) + } + } + @Test func `render production session rows with synthetic timing`() throws { guard let path = ProcessInfo.processInfo.environment["CODEXBAR_PERFORMANCE_UI_PROOF_DIR"] else { return } From e9ed0849b32cd3bd135d021fae4dea8f3c5ccb8a Mon Sep 17 00:00:00 2001 From: Peter Steinberger Date: Sat, 10 Oct 2026 13:26:07 -0700 Subject: [PATCH 2/2] test(spend): verify visible cache and sample coverage Document the primary session metrics and add the 0.74.1 release note. Look up cache coverage by metric ID so the regression can fail cleanly against the previous layout. Co-authored-by: Yuxin Qiao <104957188+Yuxin-Qiao@users.noreply.github.com> --- CHANGELOG.md | 2 ++ Tests/CodexBarTests/SpendSessionPerformanceTests.swift | 2 +- docs/spend-turn-performance-validation.md | 7 +++++-- 3 files changed, 8 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f208f4aba1..73674597c2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,8 @@ ## 0.74.1 — Unreleased +- Usage & Spend: show cached-input reuse and first-token/cache sample coverage in the session performance strip, keeping missing cache records distinct from measured zero reuse (#4413). Thanks @Yuxin-Qiao! + ## 0.74.0 — 2026-10-10 ### Highlights diff --git a/Tests/CodexBarTests/SpendSessionPerformanceTests.swift b/Tests/CodexBarTests/SpendSessionPerformanceTests.swift index d989e277e9..c825e533c7 100644 --- a/Tests/CodexBarTests/SpendSessionPerformanceTests.swift +++ b/Tests/CodexBarTests/SpendSessionPerformanceTests.swift @@ -175,7 +175,7 @@ struct SpendSessionPerformanceTests { let metrics = spendSessionPerformanceMetrics(summary) #expect(metrics.map(\.value) == ["0.1 s", "166.7 tok/s", "1.5 s", "0.0%"]) #expect(metrics[0].note == "First-token samples: 1 / 2") - #expect(metrics[3].note == "1 / 2 turns with cache data") + #expect(metrics.first(where: { $0.id == "cached-input" })?.note == "1 / 2 turns with cache data") #expect(!spendSessionPerformanceDetailMetrics(summary).contains { $0.id == "cached-input" }) } } diff --git a/docs/spend-turn-performance-validation.md b/docs/spend-turn-performance-validation.md index 5bcf348e62..48500e20fc 100644 --- a/docs/spend-turn-performance-validation.md +++ b/docs/spend-turn-performance-validation.md @@ -8,8 +8,11 @@ read_when: # Native Codex turn performance Usage & Spend retains its cost-ranked session header and adds timing only when -validated native Codex samples exist. The three values are weighted whole-turn -output, median model-first-token latency, and median completed-turn duration. +validated native Codex samples exist. The primary strip shows weighted whole-turn +output, median model-first-token latency, median completed-turn duration, and +cached-input reuse. First-token and cache sample counts appear beside their +measurements; missing cache records show unavailable, while measured zero reuse +shows 0.0%. The initially collapsed **Performance details** disclosure shows the timed-turn count. First model token may be reasoning, before visible answer text. Output already includes reasoning tokens; elapsed time includes tools and waits. These