Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions harness/src/context_snapshot.rs
Original file line number Diff line number Diff line change
Expand Up @@ -134,6 +134,12 @@ pub struct ContextSnapshotV1 {
/// generation never completed).
#[serde(default, skip_serializing_if = "Option::is_none")]
pub usage: Option<Usage>,
/// Running cost of the whole session in USD, accumulated across every
/// generation step. `usage.cost_usd` is one step's bill — on providers
/// with steep cache discounts the per-step number swings two orders of
/// magnitude, so a chip showing it alone reads as a bouncing total.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub session_cost_usd: Option<f64>,
pub timestamp: i64,
}

Expand Down Expand Up @@ -419,6 +425,7 @@ mod tests {
reasoning: None,
cost_usd: Some(0.42),
}),
session_cost_usd: Some(1.37),
timestamp: 1_722_700_000_000,
};
let mut value = serde_json::to_value(&snapshot).unwrap();
Expand Down
34 changes: 34 additions & 0 deletions harness/src/turn_loop.rs
Original file line number Diff line number Diff line change
Expand Up @@ -739,6 +739,39 @@ pub async fn run_step(
// must never fail a turn that generated successfully.
if let Some(snapshot) = record.context_snapshot.as_mut() {
snapshot.usage = outcome.message.usage.clone();
// Accumulate the session's running cost on top of the stored
// snapshot's total: turns are serialized per session (the harness-turn
// queue is fifo grouped by session_id) and steps run sequentially
// within a turn, so the loop is the only writer and the read-back is
// race-free. Seeding from the store keeps the total honest across
// turns and harness restarts. A failed read leaves the total unknown
// rather than fabricating one that resets to the current step's cost.
let step_cost = outcome
.message
.usage
.as_ref()
.and_then(|u| u.cost_usd)
.unwrap_or(0.0);
snapshot.session_cost_usd = match crate::context_snapshot::get(
&deps.iii,
&record.session_id,
cfg.session_timeout_ms,
)
.await
{
Ok(prev) => {
let prior_cost = prev.and_then(|p| p.session_cost_usd).unwrap_or(0.0);
Some(prior_cost + step_cost)
}
Err(error) => {
tracing::warn!(
session_id = %record.session_id,
%error,
"prior context snapshot read failed; session cost total unknown this step"
);
None
}
};
crate::context_snapshot::exactify(snapshot, &router, gen_system_prompt.as_deref(), &tools)
.await;
if let Err(error) =
Expand Down Expand Up @@ -2217,6 +2250,7 @@ fn build_context_snapshot(
model: record.options.model.clone(),
provider: record.options.provider.clone(),
estimator: b.estimator,
session_cost_usd: None,
usable: assembled.usable,
effective_max_output_tokens: assembled.effective_max_output_tokens,
total: final_request_tokens,
Expand Down
8 changes: 8 additions & 0 deletions harness/tests/golden/schemas/harness.metrics.json
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,14 @@
"null"
]
},
"session_cost_usd": {
"description": "Running cost of the whole session in USD, accumulated across every generation step. `usage.cost_usd` is one step's bill — on providers with steep cache discounts the per-step number swings two orders of magnitude, so a chip showing it alone reads as a bouncing total.",
"format": "double",
"type": [
"number",
"null"
]
},
"session_id": {
"type": "string"
},
Expand Down
12 changes: 10 additions & 2 deletions harness/ui/src/context-chip/index.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -282,6 +282,7 @@ function ContextPopover({
<span>
last step {formatTokens(usage?.input ?? 0)} in · output{' '}
{formatTokens(usage?.output ?? 0)}
{usage?.cost_usd != null ? <> · {formatCost(usage.cost_usd)}</> : null}
</span>
) : null}
{cache ? (
Expand All @@ -298,8 +299,15 @@ function ContextPopover({
</span>
</span>
) : null}
{usage?.cost_usd != null ? (
<span>cost {formatCost(usage.cost_usd)}</span>
{snapshot.session_cost_usd != null ? (
<span
title={
'every generation step of this session summed. The per-step ' +
'cost on the line above swings with cache hits; this one only grows.'
}
>
session total {formatCost(snapshot.session_cost_usd)}
</span>
) : null}
</div>
</div>
Expand Down
2 changes: 2 additions & 0 deletions harness/ui/src/lib/metrics.ts
Original file line number Diff line number Diff line change
Expand Up @@ -46,6 +46,8 @@ export interface ContextSnapshot {
compacted: boolean
summarized_head_tokens?: number
usage?: SnapshotUsage
/** Running cost of the whole session, summed across generation steps. */
session_cost_usd?: number | null
timestamp: number
}

Expand Down
2 changes: 1 addition & 1 deletion provider-kimi/Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

24 changes: 16 additions & 8 deletions provider-kimi/src/sse.rs
Original file line number Diff line number Diff line change
Expand Up @@ -153,20 +153,28 @@ pub fn map_finish_reason(s: &str) -> StopReason {
/// values — overwriting is correct for both, adding double-counts.
pub fn merge_usage(raw: &Value, into: &mut Usage) {
let num = |k: &str| raw.get(k).and_then(Value::as_u64);
if let Some(v) = num("prompt_tokens").or_else(|| num("input_tokens")) {
into.input = Some(v);
}
if let Some(v) = num("completion_tokens").or_else(|| num("output_tokens")) {
into.output = Some(v);
}
// `Usage.input` is the cache-MISS slice, disjoint from `cache_read`
// (pricing bills the splits additively). The wire's `prompt_tokens` is a
// TOTAL that includes the cached slice, so the miss slice is derived —
// mapping the total verbatim would bill the cached prefix twice.
let mut cached = None;
for parent in ["prompt_tokens_details", "input_tokens_details"] {
if let Some(v) = raw
.pointer(&format!("/{parent}/cached_tokens"))
.and_then(Value::as_u64)
{
into.cache_read = Some(v);
cached = Some(v);
}
}
if let Some(v) = cached {
into.cache_read = Some(v);
}
if let Some(v) = num("prompt_tokens").or_else(|| num("input_tokens")) {
into.input = Some(v.saturating_sub(cached.unwrap_or(0)));
}
if let Some(v) = num("completion_tokens").or_else(|| num("output_tokens")) {
into.output = Some(v);
}
if let Some(v) = raw
.pointer("/completion_tokens_details/reasoning_tokens")
.and_then(Value::as_u64)
Expand Down Expand Up @@ -394,7 +402,7 @@ mod tests {
assert_eq!(final_msg.stop_reason, StopReason::End);
assert_eq!(final_msg.native_stop_reason.as_deref(), Some("stop"));
let usage = final_msg.usage.unwrap();
assert_eq!(usage.input, Some(12));
assert_eq!(usage.input, Some(8));
assert_eq!(usage.output, Some(2));
assert_eq!(usage.cache_read, Some(4));
assert_eq!(usage.reasoning, Some(0));
Expand Down
2 changes: 1 addition & 1 deletion provider-kimi/src/upstream.rs
Original file line number Diff line number Diff line change
Expand Up @@ -214,7 +214,7 @@ mod tests {
);
match events.last() {
Some(AssistantMessageEvent::Done { message }) => {
assert_eq!(message.usage.as_ref().unwrap().input, Some(12));
assert_eq!(message.usage.as_ref().unwrap().input, Some(8));
assert_eq!(message.usage.as_ref().unwrap().output, Some(2));
assert_eq!(message.usage.as_ref().unwrap().cache_read, Some(4));
assert_eq!(message.native_stop_reason.as_deref(), Some("stop"));
Expand Down
2 changes: 1 addition & 1 deletion provider-llamacpp/Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

24 changes: 16 additions & 8 deletions provider-llamacpp/src/sse.rs
Original file line number Diff line number Diff line change
Expand Up @@ -154,20 +154,28 @@ pub fn map_finish_reason(s: &str) -> StopReason {
/// correct whether it reports once at the end or cumulatively per chunk.
pub fn merge_usage(raw: &Value, into: &mut Usage) {
let num = |k: &str| raw.get(k).and_then(Value::as_u64);
if let Some(v) = num("prompt_tokens").or_else(|| num("input_tokens")) {
into.input = Some(v);
}
if let Some(v) = num("completion_tokens").or_else(|| num("output_tokens")) {
into.output = Some(v);
}
// `Usage.input` is the cache-MISS slice, disjoint from `cache_read`
// (pricing bills the splits additively). The wire's `prompt_tokens` is a
// TOTAL that includes the cached slice, so the miss slice is derived —
// mapping the total verbatim would bill the cached prefix twice.
let mut cached = None;
for parent in ["prompt_tokens_details", "input_tokens_details"] {
if let Some(v) = raw
.pointer(&format!("/{parent}/cached_tokens"))
.and_then(Value::as_u64)
{
into.cache_read = Some(v);
cached = Some(v);
}
}
if let Some(v) = cached {
into.cache_read = Some(v);
}
if let Some(v) = num("prompt_tokens").or_else(|| num("input_tokens")) {
into.input = Some(v.saturating_sub(cached.unwrap_or(0)));
}
if let Some(v) = num("completion_tokens").or_else(|| num("output_tokens")) {
into.output = Some(v);
}
if let Some(v) = raw
.pointer("/completion_tokens_details/reasoning_tokens")
.and_then(Value::as_u64)
Expand Down Expand Up @@ -427,7 +435,7 @@ mod tests {
assert_eq!(final_msg.stop_reason, StopReason::End);
assert_eq!(final_msg.native_stop_reason.as_deref(), Some("stop"));
let usage = final_msg.usage.unwrap();
assert_eq!(usage.input, Some(12));
assert_eq!(usage.input, Some(8));
assert_eq!(usage.output, Some(2));
assert_eq!(usage.cache_read, Some(4));
assert_eq!(usage.reasoning, Some(0));
Expand Down
2 changes: 1 addition & 1 deletion provider-llamacpp/src/upstream.rs
Original file line number Diff line number Diff line change
Expand Up @@ -221,7 +221,7 @@ mod tests {
);
match events.last() {
Some(AssistantMessageEvent::Done { message }) => {
assert_eq!(message.usage.as_ref().unwrap().input, Some(12));
assert_eq!(message.usage.as_ref().unwrap().input, Some(8));
assert_eq!(message.usage.as_ref().unwrap().output, Some(2));
assert_eq!(message.usage.as_ref().unwrap().cache_read, Some(4));
assert_eq!(message.native_stop_reason.as_deref(), Some("stop"));
Expand Down
Loading
Loading