- 1
//! Spend pricing (docs/design/15-reliability.md). Every dollar figure produced - 2
//! here is an ESTIMATE — providers do not return cost in responses, so - 3
//! rows are labeled `estimated` and never treated as settled fact. - 4
- 5
use crate::PriceEntry; - 6
use vak_llm::Usage; - 7
- 8
/// Rough per-model pricing (USD per million tokens), substring-matched on - 9
/// lowercase model names. Unknown models are UNKNOWN: callers omit dollar - 10
/// figures rather than guessing at zero. - 11
pub fn usd_per_mtok_heuristic(model: &str) -> Option<(f64, f64)> { - 12
let m = model.to_ascii_lowercase(); - 13
let has = |s: &str| m.contains(s); - 14
if has("ox-alpha") || has("opencode") { - 15
return Some((0.0, 0.0)); - 16
} - 17
if has("opus") { - 18
return Some((15.0, 75.0)); - 19
} - 20
if has("sonnet") { - 21
return Some((3.0, 15.0)); - 22
} - 23
if has("haiku") { - 24
return Some((0.8, 4.0)); - 25
} - 26
if has("4o-mini") || (has("mini") && has("gpt")) { - 27
return Some((0.15, 0.6)); - 28
} - 29
if has("gpt-4o") { - 30
return Some((2.5, 10.0)); - 31
} - 32
if has("gpt-4-turbo") { - 33
return Some((10.0, 30.0)); - 34
} - 35
if has("o3") { - 36
return Some((2.0, 8.0)); - 37
} - 38
if has("gemini") && has("flash") { - 39
return Some((0.30, 2.5)); - 40
} - 41
if has("gemini") { - 42
return Some((1.25, 10.0)); - 43
} - 44
if has("deepseek") { - 45
return Some((0.27, 1.1)); - 46
} - 47
None - 48
} - 49
- 50
/// Exact-id overrides win; heuristic substring table is the fallback. - 51
pub fn resolve_usd_per_mtok( - 52
model: &str, - 53
overrides: &std::collections::BTreeMap<String, PriceEntry>, - 54
) -> Option<(f64, f64)> { - 55
if let Some(e) = overrides.get(model) { - 56
return Some((e.input, e.output)); - 57
} - 58
usd_per_mtok_heuristic(model) - 59
} - 60
- 61
/// Estimated USD for a usage record. Cache-creation tokens bill at the - 62
/// input rate; cache reads at a tenth of it (provider-typical discount). - 63
pub fn estimate_cost_usd( - 64
model: &str, - 65
usage: &Usage, - 66
overrides: &std::collections::BTreeMap<String, PriceEntry>, - 67
) -> Option<f64> { - 68
let (i, o) = resolve_usd_per_mtok(model, overrides)?; - 69
let fresh_input = usage.input_tokens as f64; - 70
let created = usage.cache_creation_input_tokens.unwrap_or(0) as f64; - 71
let read = usage.cache_read_input_tokens.unwrap_or(0) as f64; - 72
let out = usage.output_tokens as f64; - 73
let mtok = 1e6; - 74
Some((fresh_input + created) / mtok * i + read / mtok * i * 0.1 + out / mtok * o) - 75
} - 76
- 77
#[cfg(test)] - 78
#[allow(clippy::unwrap_used, clippy::expect_used)] - 79
mod tests { - 80
use super::*; - 81
- 82
#[test] - 83
fn override_beats_heuristic_and_exact_key_wins() { - 84
let mut o = std::collections::BTreeMap::new(); - 85
o.insert( - 86
"claude-opus-4-x".to_string(), - 87
PriceEntry { - 88
input: 1.5, - 89
output: 7.5, - 90
}, - 91
); - 92
assert_eq!( - 93
resolve_usd_per_mtok("claude-opus-4-x", &o), - 94
Some((1.5, 7.5)) - 95
); - 96
// Unrelated model falls through to the heuristic. - 97
assert_eq!( - 98
resolve_usd_per_mtok("claude-sonnet-9", &o), - 99
Some((3.0, 15.0)) - 100
); - 101
// Unknown without override stays UNKNOWN. - 102
assert_eq!(resolve_usd_per_mtok("totally-novel-model", &o), None); - 103
} - 104
- 105
#[test] - 106
fn estimate_prices_all_token_classes() { - 107
let usage = Usage { - 108
input_tokens: 1_000_000, - 109
output_tokens: 100_000, - 110
cache_creation_input_tokens: Some(500_000), - 111
cache_read_input_tokens: Some(2_000_000), - 112
prefill_ms: None, - 113
load_ms: None, - 114
}; - 115
let usd = - 116
estimate_cost_usd("claude-sonnet-4", &usage, &Default::default()).expect("priced"); - 117
// input 1M*3 + creation 0.5M*3 + read 2M*3*0.1 + out 0.1M*15 - 118
let expected = 3.0 + 1.5 + 0.6 + 1.5; - 119
assert!((usd - expected).abs() < 1e-9, "{usd} vs {expected}"); - 120
} - 121
- 122
#[test] - 123
fn unpriced_model_estimates_nothing() { - 124
let u = Usage::default(); - 125
assert_eq!(estimate_cost_usd("mystery", &u, &Default::default()), None); - 126
} - 127
} - 128
Indexing the workspace…
Vakyartha documentation is discovering safe artifacts, anchors, and source references.