diff --git a/CHANGELOG.md b/CHANGELOG.md index b6b3d873cb..9d65768333 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ ## 0.74.1 — Unreleased +### Changed + +- Usage & Spend: show cached-input reuse and first-token/cache sample coverage in the session performance strip, keeping missing cache records distinct from measured zero reuse (#4413). Thanks @Yuxin-Qiao! + ### Fixed - Menu bar: stop the blank Settings placeholder window from appearing on launch (regression in 0.74.0) (#4415). Thanks @tcurdt, @ChuJiannn11 and @kcharlan! diff --git a/Sources/CodexBar/SpendSessionPerformanceView.swift b/Sources/CodexBar/SpendSessionPerformanceView.swift index 582f8382ee..75f82cbd5c 100644 --- a/Sources/CodexBar/SpendSessionPerformanceView.swift +++ b/Sources/CodexBar/SpendSessionPerformanceView.swift @@ -46,6 +46,9 @@ private struct SpendPerformanceMetricStrip: View { Text(metric.value) .font(.system(.body, design: .rounded, weight: .semibold)) .foregroundStyle(.primary) + if let note = metric.note { + Text(note).font(.caption2).foregroundStyle(.secondary) + } } .fixedSize() .frame(maxWidth: .infinity, alignment: .leading) @@ -55,10 +58,15 @@ private struct SpendPerformanceMetricStrip: View { } VStack(spacing: 6) { ForEach(self.metrics) { metric in - HStack(alignment: .firstTextBaseline) { - Text(metric.label).foregroundStyle(.secondary) - Spacer(minLength: 12) - Text(metric.value).fontWeight(.semibold).foregroundStyle(.primary) + VStack(alignment: .leading, spacing: 3) { + HStack(alignment: .firstTextBaseline) { + Text(metric.label).foregroundStyle(.secondary) + Spacer(minLength: 12) + Text(metric.value).fontWeight(.semibold).foregroundStyle(.primary) + } + if let note = metric.note { + Text(note).font(.caption2).foregroundStyle(.secondary) + } } .font(.caption) .help(metric.help ?? "") @@ -193,15 +201,17 @@ private struct SpendPerformanceModelComparison: View { } func spendSessionPerformanceMetrics(_ summary: CostUsageTurnPerformanceSummary) -> [SpendPerformanceMetric] { - [ + let firstTokenCoverage = L( + "First-token samples: %@ / %@", + codexBarLocalizedInteger(summary.firstTokenSampleCount), + codexBarLocalizedInteger(summary.sampleCount)) + return [ SpendPerformanceMetric( id: "first-token", label: L("spend_performance_first_token"), value: spendPerformanceSeconds(summary.medianFirstTokenMilliseconds), - help: L( - "First-token samples: %@ / %@", - codexBarLocalizedInteger(summary.firstTokenSampleCount), - codexBarLocalizedInteger(summary.sampleCount)) + "\n" + + note: firstTokenCoverage, + help: firstTokenCoverage + "\n" + L("Model first token may be reasoning, before visible answer text.")), SpendPerformanceMetric( id: "output", @@ -211,6 +221,17 @@ func spendSessionPerformanceMetrics(_ summary: CostUsageTurnPerformanceSummary) id: "duration", label: L("spend_performance_duration"), value: spendPerformanceSeconds(summary.medianDurationMilliseconds)), + SpendPerformanceMetric( + id: "cached-input", + label: L("spend_performance_cached_input"), + value: summary.details.cachedInputFraction.map { + L("spend_performance_percent", spendPerformanceNumber($0 * 100)) + } ?? "—", + note: L( + "spend_performance_coverage", + codexBarLocalizedInteger(summary.details.cacheSampleCount), + codexBarLocalizedInteger(summary.sampleCount)), + help: L("spend_performance_cache_help")), ] } @@ -241,16 +262,6 @@ func spendSessionPerformanceDetailMetrics(_ summary: CostUsageTurnPerformanceSum } ?? "—", note: details.outputRateLowerQuartile == nil ? L("Speed range needs 4 completed turns.") : nil, help: L("spend_performance_speed_range_help")), - SpendPerformanceMetric( - id: "cached-input", - label: L("spend_performance_cached_input"), - value: details.cachedInputFraction.map { L("spend_performance_percent", spendPerformanceNumber($0 * 100)) } - ?? "—", - note: L( - "spend_performance_coverage", - codexBarLocalizedInteger(details.cacheSampleCount), - codexBarLocalizedInteger(summary.sampleCount)), - help: L("spend_performance_cache_help")), ] } diff --git a/Tests/CodexBarTests/SpendSessionPerformanceTests.swift b/Tests/CodexBarTests/SpendSessionPerformanceTests.swift index f3f2e4db2b..c825e533c7 100644 --- a/Tests/CodexBarTests/SpendSessionPerformanceTests.swift +++ b/Tests/CodexBarTests/SpendSessionPerformanceTests.swift @@ -123,11 +123,13 @@ struct SpendSessionPerformanceTests { let summary = try #require(CostUsageTurnPerformanceSummary(samples: [sample])) CodexBarLocalizationOverride.$appLanguage.withValue("en") { let metrics = spendSessionPerformanceMetrics(summary) - #expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 s"]) + #expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 s", "—"]) + #expect(metrics[0].note == "First-token samples: 0 / 1") + #expect(metrics[3].note == "0 / 1 turns with cache data") } CodexBarLocalizationOverride.$appLanguage.withValue("zh-Hans") { let metrics = spendSessionPerformanceMetrics(summary) - #expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 秒"]) + #expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 秒", "—"]) } } @@ -146,13 +148,38 @@ struct SpendSessionPerformanceTests { #expect(metrics[0].note != nil) #expect(metrics[1].value == "—") #expect(metrics[1].note != nil) - #expect(metrics[3].value == "80.0%") + let cachedInput = spendSessionPerformanceMetrics(summary).first { $0.id == "cached-input" } + #expect(cachedInput?.value == "80.0%") + #expect(cachedInput?.note == "1 / 1 turns with cache data") #expect(summary.firstTokenSampleCount == 0) #expect(summary.sampleCount == 1) #expect(summary.details.cacheSampleCount == 1) } } + @Test + func `primary observations distinguish measured zero cache reuse from missing records`() throws { + let measured = try #require(CostUsageTurnPerformanceSample( + completedAt: Date(), + outputTokens: 300, + durationMilliseconds: 1000, + firstTokenMilliseconds: 100, + inputTokens: 1000, + cachedInputTokens: 0)) + let unrecorded = try #require(CostUsageTurnPerformanceSample( + completedAt: Date(), + outputTokens: 200, + durationMilliseconds: 2000)) + let summary = try #require(CostUsageTurnPerformanceSummary(samples: [measured, unrecorded])) + CodexBarLocalizationOverride.$appLanguage.withValue("en") { + let metrics = spendSessionPerformanceMetrics(summary) + #expect(metrics.map(\.value) == ["0.1 s", "166.7 tok/s", "1.5 s", "0.0%"]) + #expect(metrics[0].note == "First-token samples: 1 / 2") + #expect(metrics.first(where: { $0.id == "cached-input" })?.note == "1 / 2 turns with cache data") + #expect(!spendSessionPerformanceDetailMetrics(summary).contains { $0.id == "cached-input" }) + } + } + @Test func `render production session rows with synthetic timing`() throws { guard let path = ProcessInfo.processInfo.environment["CODEXBAR_PERFORMANCE_UI_PROOF_DIR"] else { return } diff --git a/docs/spend-turn-performance-validation.md b/docs/spend-turn-performance-validation.md index 5bcf348e62..48500e20fc 100644 --- a/docs/spend-turn-performance-validation.md +++ b/docs/spend-turn-performance-validation.md @@ -8,8 +8,11 @@ read_when: # Native Codex turn performance Usage & Spend retains its cost-ranked session header and adds timing only when -validated native Codex samples exist. The three values are weighted whole-turn -output, median model-first-token latency, and median completed-turn duration. +validated native Codex samples exist. The primary strip shows weighted whole-turn +output, median model-first-token latency, median completed-turn duration, and +cached-input reuse. First-token and cache sample counts appear beside their +measurements; missing cache records show unavailable, while measured zero reuse +shows 0.0%. The initially collapsed **Performance details** disclosure shows the timed-turn count. First model token may be reasoning, before visible answer text. Output already includes reasoning tokens; elapsed time includes tools and waits. These