Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,10 @@

## 0.74.1 — Unreleased

### Changed

- Usage & Spend: show cached-input reuse and first-token/cache sample coverage in the session performance strip, keeping missing cache records distinct from measured zero reuse (#4413). Thanks @Yuxin-Qiao!

### Fixed

- Menu bar: stop the blank Settings placeholder window from appearing on launch (regression in 0.74.0) (#4415). Thanks @tcurdt, @ChuJiannn11 and @kcharlan!
Expand Down
49 changes: 30 additions & 19 deletions Sources/CodexBar/SpendSessionPerformanceView.swift
Original file line number Diff line number Diff line change
Expand Up @@ -46,6 +46,9 @@ private struct SpendPerformanceMetricStrip: View {
Text(metric.value)
.font(.system(.body, design: .rounded, weight: .semibold))
.foregroundStyle(.primary)
if let note = metric.note {
Text(note).font(.caption2).foregroundStyle(.secondary)
}
}
.fixedSize()
.frame(maxWidth: .infinity, alignment: .leading)
Expand All @@ -55,10 +58,15 @@ private struct SpendPerformanceMetricStrip: View {
}
VStack(spacing: 6) {
ForEach(self.metrics) { metric in
HStack(alignment: .firstTextBaseline) {
Text(metric.label).foregroundStyle(.secondary)
Spacer(minLength: 12)
Text(metric.value).fontWeight(.semibold).foregroundStyle(.primary)
VStack(alignment: .leading, spacing: 3) {
HStack(alignment: .firstTextBaseline) {
Text(metric.label).foregroundStyle(.secondary)
Spacer(minLength: 12)
Text(metric.value).fontWeight(.semibold).foregroundStyle(.primary)
}
if let note = metric.note {
Text(note).font(.caption2).foregroundStyle(.secondary)
}
}
.font(.caption)
.help(metric.help ?? "")
Expand Down Expand Up @@ -193,15 +201,17 @@ private struct SpendPerformanceModelComparison: View {
}

func spendSessionPerformanceMetrics(_ summary: CostUsageTurnPerformanceSummary) -> [SpendPerformanceMetric] {
[
let firstTokenCoverage = L(
"First-token samples: %@ / %@",
codexBarLocalizedInteger(summary.firstTokenSampleCount),
codexBarLocalizedInteger(summary.sampleCount))
return [
SpendPerformanceMetric(
id: "first-token",
label: L("spend_performance_first_token"),
value: spendPerformanceSeconds(summary.medianFirstTokenMilliseconds),
help: L(
"First-token samples: %@ / %@",
codexBarLocalizedInteger(summary.firstTokenSampleCount),
codexBarLocalizedInteger(summary.sampleCount)) + "\n" +
note: firstTokenCoverage,
help: firstTokenCoverage + "\n" +
L("Model first token may be reasoning, before visible answer text.")),
SpendPerformanceMetric(
id: "output",
Expand All @@ -211,6 +221,17 @@ func spendSessionPerformanceMetrics(_ summary: CostUsageTurnPerformanceSummary)
id: "duration",
label: L("spend_performance_duration"),
value: spendPerformanceSeconds(summary.medianDurationMilliseconds)),
SpendPerformanceMetric(
id: "cached-input",
label: L("spend_performance_cached_input"),
value: summary.details.cachedInputFraction.map {
L("spend_performance_percent", spendPerformanceNumber($0 * 100))
} ?? "—",
note: L(
"spend_performance_coverage",
codexBarLocalizedInteger(summary.details.cacheSampleCount),
codexBarLocalizedInteger(summary.sampleCount)),
help: L("spend_performance_cache_help")),
]
}

Expand Down Expand Up @@ -241,16 +262,6 @@ func spendSessionPerformanceDetailMetrics(_ summary: CostUsageTurnPerformanceSum
} ?? "—",
note: details.outputRateLowerQuartile == nil ? L("Speed range needs 4 completed turns.") : nil,
help: L("spend_performance_speed_range_help")),
SpendPerformanceMetric(
id: "cached-input",
label: L("spend_performance_cached_input"),
value: details.cachedInputFraction.map { L("spend_performance_percent", spendPerformanceNumber($0 * 100)) }
?? "—",
note: L(
"spend_performance_coverage",
codexBarLocalizedInteger(details.cacheSampleCount),
codexBarLocalizedInteger(summary.sampleCount)),
help: L("spend_performance_cache_help")),
]
}

Expand Down
33 changes: 30 additions & 3 deletions Tests/CodexBarTests/SpendSessionPerformanceTests.swift
Original file line number Diff line number Diff line change
Expand Up @@ -123,11 +123,13 @@ struct SpendSessionPerformanceTests {
let summary = try #require(CostUsageTurnPerformanceSummary(samples: [sample]))
CodexBarLocalizationOverride.$appLanguage.withValue("en") {
let metrics = spendSessionPerformanceMetrics(summary)
#expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 s"])
#expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 s", "—"])
#expect(metrics[0].note == "First-token samples: 0 / 1")
#expect(metrics[3].note == "0 / 1 turns with cache data")
}
CodexBarLocalizationOverride.$appLanguage.withValue("zh-Hans") {
let metrics = spendSessionPerformanceMetrics(summary)
#expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 秒"])
#expect(metrics.map(\.value) == ["—", "2.0 tok/s", "10.0 秒", "—"])
}
}

Expand All @@ -146,13 +148,38 @@ struct SpendSessionPerformanceTests {
#expect(metrics[0].note != nil)
#expect(metrics[1].value == "—")
#expect(metrics[1].note != nil)
#expect(metrics[3].value == "80.0%")
let cachedInput = spendSessionPerformanceMetrics(summary).first { $0.id == "cached-input" }
#expect(cachedInput?.value == "80.0%")
#expect(cachedInput?.note == "1 / 1 turns with cache data")
#expect(summary.firstTokenSampleCount == 0)
#expect(summary.sampleCount == 1)
#expect(summary.details.cacheSampleCount == 1)
}
}

@Test
func `primary observations distinguish measured zero cache reuse from missing records`() throws {
let measured = try #require(CostUsageTurnPerformanceSample(
completedAt: Date(),
outputTokens: 300,
durationMilliseconds: 1000,
firstTokenMilliseconds: 100,
inputTokens: 1000,
cachedInputTokens: 0))
let unrecorded = try #require(CostUsageTurnPerformanceSample(
completedAt: Date(),
outputTokens: 200,
durationMilliseconds: 2000))
let summary = try #require(CostUsageTurnPerformanceSummary(samples: [measured, unrecorded]))
CodexBarLocalizationOverride.$appLanguage.withValue("en") {
let metrics = spendSessionPerformanceMetrics(summary)
#expect(metrics.map(\.value) == ["0.1 s", "166.7 tok/s", "1.5 s", "0.0%"])
#expect(metrics[0].note == "First-token samples: 1 / 2")
#expect(metrics.first(where: { $0.id == "cached-input" })?.note == "1 / 2 turns with cache data")
#expect(!spendSessionPerformanceDetailMetrics(summary).contains { $0.id == "cached-input" })
}
}

@Test
func `render production session rows with synthetic timing`() throws {
guard let path = ProcessInfo.processInfo.environment["CODEXBAR_PERFORMANCE_UI_PROOF_DIR"] else { return }
Expand Down
7 changes: 5 additions & 2 deletions docs/spend-turn-performance-validation.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,8 +8,11 @@ read_when:
# Native Codex turn performance

Usage & Spend retains its cost-ranked session header and adds timing only when
validated native Codex samples exist. The three values are weighted whole-turn
output, median model-first-token latency, and median completed-turn duration.
validated native Codex samples exist. The primary strip shows weighted whole-turn
output, median model-first-token latency, median completed-turn duration, and
cached-input reuse. First-token and cache sample counts appear beside their
measurements; missing cache records show unavailable, while measured zero reuse
shows 0.0%.
The initially collapsed **Performance details** disclosure shows the timed-turn
count. First model token may be reasoning, before visible answer text. Output
already includes reasoning tokens; elapsed time includes tools and waits. These
Expand Down
Loading