- 1
//! Frozen-ladder routing primitives (docs/design/15-reliability.md). - 2
//! - 3
//! The candidate SET is chosen by constraint satisfaction upstream (key - 4
//! present, model actually discovered on that provider); THIS module owns - 5
//! the versioned ORDERING function over that set. Ordering is pure and - 6
//! deterministic: identical inputs yield an identical ladder, so freezing - 7
//! the ladder into the frozen contract reproduces the decision exactly. - 8
//! - 9
//! Epistemics: outcomes are success/failure/**unknown**. Unknowns shrink - 10
//! confidence (denominator) without punishing direction — billing proves - 11
//! nothing about answer quality. Absent evidence is a neutral prior, - 12
//! never zero. - 13
//! - 14
//! Phase R (vakrouter adoption): v2 adds request-demand scoring (the - 15
//! difficulty of the work selects utility/balanced/quality-critical - 16
//! ordering instead of hardcoded model-name bands), belief demotion - 17
//! (domain-weighted doubt ranks a doubted leg below fully-trusted peers), - 18
//! and price honesty (known pricing outranks unknown before cost math). - 19
//! Model ids are NEVER hardcoded here; quality hints are caller-declared - 20
//! configuration, not baked-in knowledge. - 21
- 22
use serde::{Deserialize, Serialize}; - 23
use std::collections::HashMap; - 24
- 25
/// Wire contract selected for a model route. A provider credential is not a - 26
/// wire contract: one provider can expose multiple endpoint dialects whose - 27
/// feature combinations differ. This is frozen with the model so replay and - 28
/// fallback cannot silently change the request semantics. - 29
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Serialize, Deserialize, Default)] - 30
#[serde(rename_all = "snake_case")] - 31
pub enum EndpointDialect { - 32
#[default] - 33
ChatCompletions, - 34
Responses, - 35
AnthropicMessages, - 36
GoogleGenerateContent, - 37
} - 38
- 39
impl EndpointDialect { - 40
/// Select the native wire contract from provider identity and the turn's - 41
/// declared requirements. This deliberately contains no model-name - 42
/// knowledge; model-specific support comes from live catalogues/probes. - 43
pub fn for_provider(provider: &str, needs_tools_or_reasoning: bool) -> Self { - 44
match provider { - 45
"anthropic" => Self::AnthropicMessages, - 46
"google" => Self::GoogleGenerateContent, - 47
"openai" | "openai-responses" if needs_tools_or_reasoning => Self::Responses, - 48
"openrouter" | "openrouter-responses" if needs_tools_or_reasoning => Self::Responses, - 49
"openai-responses" | "openrouter-responses" => Self::Responses, - 50
_ => Self::ChatCompletions, - 51
} - 52
} - 53
} - 54
- 55
#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Serialize, Deserialize)] - 56
pub struct RouteLeg { - 57
pub provider: String, - 58
pub model: String, - 59
#[serde(default)] - 60
pub dialect: EndpointDialect, - 61
/// Non-secret credential fingerprint selected at admission. `None` is - 62
/// the legacy/default credential for callers that do not use a pool. - 63
#[serde(default, skip_serializing_if = "Option::is_none")] - 64
pub credential_id: Option<String>, - 65
} - 66
- 67
/// One provider/model observation aggregate. - 68
#[derive(Debug, Clone, Default)] - 69
pub struct ModelEvidence { - 70
pub success: u64, - 71
pub failure: u64, - 72
/// Settled-billing-without-verdict (or transport ambiguity). Shrinks - 73
/// confidence but does not count against the model. - 74
pub unknown: u64, - 75
pub p50_latency_ms: Option<u64>, - 76
} - 77
- 78
#[derive(Debug, Clone, Default)] - 79
pub struct EvidenceSnapshot { - 80
pub by_key: HashMap<(String, String), ModelEvidence>, - 81
} - 82
- 83
impl EvidenceSnapshot { - 84
pub fn get(&self, provider: &str, model: &str) -> ModelEvidence { - 85
self.by_key - 86
.get(&(provider.to_string(), model.to_string())) - 87
.cloned() - 88
.unwrap_or_default() - 89
} - 90
} - 91
- 92
/// Immutable belief-multiplier view for the pure ordering function. - 93
/// Multipliers live in [BELIEF_FLOOR, 1]; 1.0 means fully trusted. - 94
#[derive(Debug, Clone, Default)] - 95
pub struct BeliefMap { - 96
pub multipliers: HashMap<(String, String), f64>, - 97
} - 98
- 99
/// Accumulated doubt never fully zeroes a candidate. - 100
pub const BELIEF_FLOOR: f64 = 0.1; - 101
- 102
impl BeliefMap { - 103
pub fn get(&self, provider: &str, model: &str) -> f64 { - 104
self.multipliers - 105
.get(&(provider.to_string(), model.to_string())) - 106
.copied() - 107
.unwrap_or(1.0) - 108
} - 109
} - 110
- 111
/// Laplace-shrunk reliability in [0,1]. Unknown outcomes count as - 112
/// HALF-weight successes: they shrink confidence without punishing - 113
/// direction — billing proves nothing about quality, and an absent verdict - 114
/// is not a failure. - 115
fn reliability(e: &ModelEvidence) -> f64 { - 116
let trials = e.success + e.failure + e.unknown; - 117
(e.success as f64 + 0.5 * e.unknown as f64 + 1.0) / (trials as f64 + 2.0) - 118
} - 119
- 120
/// LEGACY band table, retained ONLY so replaying a v1-frozen contract - 121
/// reproduces its original ordering byte-for-byte. New admissions use - 122
/// order_ladder_v2, which takes caller-declared quality hints instead — - 123
/// model ids must not be hardcoded into routing knowledge. - 124
const FRONTIER_BANDS: [&str; 6] = ["opus", "gpt-5", "o3", "pro", "claude-4", "qwen3-max"]; - 125
- 126
fn frontier_band(model: &str) -> bool { - 127
let m = model.to_ascii_lowercase(); - 128
FRONTIER_BANDS.iter().any(|b| m.contains(b)) - 129
} - 130
- 131
fn latency_penalty(ms: Option<u64>) -> f64 { - 132
ms.unwrap_or(0).min(30_000) as f64 / 10_000.0 // ≤3.0 - 133
} - 134
- 135
/// Versioned ordering function. Cheap-first by default; `quality_first` - 136
/// promotes the frontier band for planning/verification purposes. - 137
/// `cost_usd_per_mtok` supplies the caller's pricing view (output rate); - 138
/// unpriced models sit at a middling rank — absent is UNKNOWN, never free. - 139
/// - 140
/// v1 score = reliability × band / (1 + cost/weight + latency); ties break - 141
/// deterministically on (cost, provider, model). - 142
pub fn order_ladder_v1( - 143
mut candidates: Vec<RouteLeg>, - 144
evidence: &EvidenceSnapshot, - 145
quality_first: bool, - 146
cost_usd_per_mtok: impl Fn(&str) -> Option<f64>, - 147
) -> Vec<RouteLeg> { - 148
const UNPRICED_RANK: f64 = 60.0; - 149
candidates.sort_by(|a, b| { - 150
let ea = evidence.get(&a.provider, &a.model); - 151
let eb = evidence.get(&b.provider, &b.model); - 152
let ra = reliability(&ea); - 153
let rb = reliability(&eb); - 154
let ca = cost_usd_per_mtok(&a.model).unwrap_or(UNPRICED_RANK); - 155
let cb = cost_usd_per_mtok(&b.model).unwrap_or(UNPRICED_RANK); - 156
let la = latency_penalty(ea.p50_latency_ms); - 157
let lb = latency_penalty(eb.p50_latency_ms); - 158
let score = |r: f64, c: f64, l: f64, m: &str| -> f64 { - 159
let boost = if quality_first && frontier_band(m) { - 160
1.5 - 161
} else { - 162
1.0 - 163
}; - 164
let cost_weight = if quality_first { 1000.0 } else { 100.0 }; - 165
(r * boost) / (1.0 + c / cost_weight + l) - 166
}; - 167
let sa = score(ra, ca, la, &a.model); - 168
let sb = score(rb, cb, lb, &b.model); - 169
sb.partial_cmp(&sa) - 170
.unwrap_or(std::cmp::Ordering::Equal) - 171
.then(ca.partial_cmp(&cb).unwrap_or(std::cmp::Ordering::Equal)) - 172
.then(a.provider.cmp(&b.provider)) - 173
.then(a.model.cmp(&b.model)) - 174
}); - 175
candidates - 176
} - 177
- 178
/// How hard the work is (Phase R demand scoring). Weights mirror the - 179
/// vakrouter study: context and output dominate because those are what a - 180
/// small model physically cannot do. Unavailable factors read as 0 — - 181
/// never fabricated. - 182
#[derive(Debug, Clone, Copy, Default)] - 183
pub struct DemandInput { - 184
pub estimated_input_tokens: u64, - 185
pub output_budget_tokens: u64, - 186
pub tool_count: usize, - 187
pub structured_output: bool, - 188
pub reasoning_required: bool, - 189
pub evidence_required: bool, - 190
} - 191
- 192
#[derive(Debug, Clone, Copy, PartialEq, Eq)] - 193
pub enum DemandBand { - 194
Low, - 195
Moderate, - 196
High, - 197
} - 198
- 199
impl DemandBand { - 200
pub fn as_str(&self) -> &'static str { - 201
match self { - 202
DemandBand::Low => "low", - 203
DemandBand::Moderate => "moderate", - 204
DemandBand::High => "high", - 205
} - 206
} - 207
} - 208
- 209
#[derive(Debug, Clone, Copy)] - 210
pub struct Demand { - 211
/// [0, 1] weighted saturation score. - 212
pub score: f64, - 213
pub band: DemandBand, - 214
} - 215
- 216
const CONTEXT_SATURATION_TOKENS: f64 = 32_768.0; - 217
const TOOL_BREADTH_SATURATION: f64 = 12.0; - 218
const OUTPUT_SATURATION_TOKENS: f64 = 8_192.0; - 219
- 220
fn saturate(value: f64, anchor: f64) -> f64 { - 221
if value <= 0.0 || !value.is_finite() { - 222
return 0.0; - 223
} - 224
(value / anchor).min(1.0) - 225
} - 226
- 227
/// Score request difficulty from saturating anchors. Unknown context reads - 228
/// as moderate (0.5), never zero — absent is not free. - 229
pub fn score_demand(input: DemandInput) -> Demand { - 230
const CONTEXT_W: f64 = 0.30; - 231
const OUTPUT_W: f64 = 0.25; - 232
const REASONING_W: f64 = 0.15; - 233
const TOOLS_W: f64 = 0.12; - 234
const STRUCTURED_W: f64 = 0.10; - 235
const EVIDENCE_W: f64 = 0.08; - 236
- 237
let ctx = if input.estimated_input_tokens == 0 { - 238
0.5 - 239
} else { - 240
saturate( - 241
input.estimated_input_tokens as f64, - 242
CONTEXT_SATURATION_TOKENS, - 243
) - 244
}; - 245
let out = saturate(input.output_budget_tokens as f64, OUTPUT_SATURATION_TOKENS); - 246
let reasoning = if input.reasoning_required { 1.0 } else { 0.0 }; - 247
let tools = saturate(input.tool_count as f64, TOOL_BREADTH_SATURATION); - 248
let structured = if input.structured_output { 1.0 } else { 0.0 }; - 249
let evidence = if input.evidence_required { 1.0 } else { 0.0 }; - 250
- 251
let score = CONTEXT_W * ctx - 252
+ OUTPUT_W * out - 253
+ REASONING_W * reasoning - 254
+ TOOLS_W * tools - 255
+ STRUCTURED_W * structured - 256
+ EVIDENCE_W * evidence; - 257
- 258
let band = if score < 0.25 { - 259
DemandBand::Low - 260
} else if score < 0.6 { - 261
DemandBand::Moderate - 262
} else { - 263
DemandBand::High - 264
}; - 265
Demand { score, band } - 266
} - 267
- 268
/// What the ladder optimizes for. `auto` resolves from the demand band: - 269
/// high → quality-critical, low → utility, moderate → balanced. - 270
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] - 271
#[serde(rename_all = "snake_case")] - 272
pub enum QualityObjective { - 273
Utility, - 274
Balanced, - 275
QualityCritical, - 276
} - 277
- 278
impl QualityObjective { - 279
pub fn as_str(&self) -> &'static str { - 280
match self { - 281
QualityObjective::Utility => "utility", - 282
QualityObjective::Balanced => "balanced", - 283
QualityObjective::QualityCritical => "quality-critical", - 284
} - 285
} - 286
- 287
/// Resolve the explicit config override ("auto" or unknown → None) - 288
/// against the demand-derived default. - 289
pub fn resolve(explicit: Option<&str>, band: DemandBand) -> Self { - 290
match explicit { - 291
Some("utility") => Self::Utility, - 292
Some("balanced") => Self::Balanced, - 293
Some("quality-critical") | Some("quality_critical") => Self::QualityCritical, - 294
_ => match band { - 295
DemandBand::High => Self::QualityCritical, - 296
DemandBand::Low => Self::Utility, - 297
DemandBand::Moderate => Self::Balanced, - 298
}, - 299
} - 300
} - 301
} - 302
- 303
/// Versioned ordering function v2 (Phase R). Differences from v1: - 304
/// - objective-driven instead of hardcoded frontier bands; `hints` are - 305
/// CALLER-declared model-id substrings treated as frontier tier. - 306
/// - belief demotion: a doubted leg ranks below every fully-trusted peer, - 307
/// whatever its cost advantage — retries are a cost too. - 308
/// - price honesty: known pricing outranks unknown before cost math. - 309
/// - 310
/// Ordering is a total order with deterministic tiebreaks - 311
/// (cost, provider, model), so identical inputs freeze identically. - 312
pub fn order_ladder_v2( - 313
mut candidates: Vec<RouteLeg>, - 314
evidence: &EvidenceSnapshot, - 315
beliefs: &BeliefMap, - 316
objective: QualityObjective, - 317
hints: &[String], - 318
cost_usd_per_mtok: impl Fn(&str) -> Option<f64>, - 319
) -> Vec<RouteLeg> { - 320
const UNPRICED_RANK: f64 = 60.0; - 321
let hint_boost = |model: &str| -> f64 { - 322
let m = model.to_ascii_lowercase(); - 323
if hints.iter().any(|h| m.contains(&h.to_ascii_lowercase())) { - 324
1.5 - 325
} else { - 326
1.0 - 327
} - 328
}; - 329
let quality = |leg: &RouteLeg| -> f64 { - 330
let e = evidence.get(&leg.provider, &leg.model); - 331
let r = reliability(&e); - 332
let c = cost_usd_per_mtok(&leg.model).unwrap_or(UNPRICED_RANK); - 333
let l = latency_penalty(e.p50_latency_ms); - 334
// Cost weight by objective: quality-critical is deliberately - 335
// cost-insensitive; balanced still respects spend. - 336
let cost_weight = match objective { - 337
QualityObjective::Utility => 100.0, - 338
QualityObjective::Balanced => 100.0, - 339
QualityObjective::QualityCritical => 1000.0, - 340
}; - 341
(r * hint_boost(&leg.model)) / (1.0 + c / cost_weight + l) - 342
}; - 343
candidates.sort_by(|a, b| { - 344
let ma = beliefs.get(&a.provider, &a.model); - 345
let mb = beliefs.get(&b.provider, &b.model); - 346
// Belief demotion precedes everything: doubted legs lose to trusted - 347
// peers regardless of price advantage. - 348
let a_doubted = ma < 1.0; - 349
let b_doubted = mb < 1.0; - 350
if a_doubted != b_doubted { - 351
// Trusted candidate sorts first. - 352
return if a_doubted { - 353
std::cmp::Ordering::Greater - 354
} else { - 355
std::cmp::Ordering::Less - 356
}; - 357
} - 358
let ka = cost_usd_per_mtok(&a.model); - 359
let kb = cost_usd_per_mtok(&b.model); - 360
// Price honesty before cost comparison: known beats unknown. - 361
if ka.is_some() != kb.is_some() { - 362
return ka.is_some().cmp(&kb.is_some()).reverse(); - 363
} - 364
let ca = ka.unwrap_or(UNPRICED_RANK); - 365
let cb = kb.unwrap_or(UNPRICED_RANK); - 366
let qa = quality(a) * ma; - 367
let qb = quality(b) * mb; - 368
match objective { - 369
// Utility: cost first, quality breaks ties. - 370
QualityObjective::Utility => ca - 371
.partial_cmp(&cb) - 372
.unwrap_or(std::cmp::Ordering::Equal) - 373
.then(qb.partial_cmp(&qa).unwrap_or(std::cmp::Ordering::Equal)), - 374
// Balanced / quality-critical: quality first, cost breaks ties. - 375
_ => qb - 376
.partial_cmp(&qa) - 377
.unwrap_or(std::cmp::Ordering::Equal) - 378
.then(ca.partial_cmp(&cb).unwrap_or(std::cmp::Ordering::Equal)), - 379
} - 380
.then(ca.partial_cmp(&cb).unwrap_or(std::cmp::Ordering::Equal)) - 381
.then(a.provider.cmp(&b.provider)) - 382
.then(a.model.cmp(&b.model)) - 383
}); - 384
candidates - 385
} - 386
- 387
#[cfg(test)] - 388
#[allow(clippy::unwrap_used, clippy::expect_used)] - 389
mod tests { - 390
use super::*; - 391
- 392
fn leg(p: &str, m: &str) -> RouteLeg { - 393
RouteLeg { - 394
provider: p.into(), - 395
model: m.into(), - 396
dialect: EndpointDialect::default(), - 397
credential_id: None, - 398
} - 399
} - 400
- 401
#[test] - 402
fn empty_evidence_orders_cheap_first_deterministically() { - 403
let cands = vec![ - 404
leg("p", "claude-opus-x"), - 405
leg("p", "some-haiku"), - 406
leg("q", "gpt-4o-mini"), - 407
]; - 408
let costs = |m: &str| -> Option<f64> { - 409
if m.contains("opus") { - 410
Some(75.0) - 411
} else if m.contains("haiku") { - 412
Some(4.0) - 413
} else { - 414
Some(0.6) - 415
} - 416
}; - 417
let a = order_ladder_v1(cands.clone(), &Default::default(), false, costs); - 418
assert_eq!(a[0].model, "gpt-4o-mini"); - 419
assert!(a.last().unwrap().model.contains("opus")); - 420
let b = order_ladder_v1( - 421
vec![ - 422
leg("p", "claude-opus-x"), - 423
leg("p", "some-haiku"), - 424
leg("q", "gpt-4o-mini"), - 425
], - 426
&Default::default(), - 427
false, - 428
costs, - 429
); - 430
assert_eq!(a, b, "pure function must be deterministic"); - 431
} - 432
- 433
#[test] - 434
fn failures_push_a_model_down_unknowns_only_shrink() { - 435
let mut ev = EvidenceSnapshot::default(); - 436
ev.by_key.insert( - 437
("p".into(), "m-fail".into()), - 438
ModelEvidence { - 439
failure: 9, - 440
..Default::default() - 441
}, - 442
); - 443
ev.by_key.insert( - 444
("p".into(), "m-unknown".into()), - 445
ModelEvidence { - 446
unknown: 9, - 447
..Default::default() - 448
}, - 449
); - 450
ev.by_key.insert( - 451
("p".into(), "m-good".into()), - 452
ModelEvidence { - 453
success: 9, - 454
..Default::default() - 455
}, - 456
); - 457
let cands = vec![ - 458
leg("p", "m-fail"), - 459
leg("p", "m-unknown"), - 460
leg("p", "m-good"), - 461
]; - 462
let ordered = order_ladder_v1(cands, &ev, false, |_m: &str| Some(3.0)); - 463
assert_eq!(ordered[0].model, "m-good"); - 464
assert_eq!( - 465
ordered[1].model, "m-unknown", - 466
"unknown must outrank failure" - 467
); - 468
assert_eq!(ordered[2].model, "m-fail"); - 469
} - 470
- 471
#[test] - 472
fn quality_first_promotes_frontier_band() { - 473
let cands = vec![leg("p", "cheap-haiku"), leg("q", "big-claude-opus")]; - 474
let cheap_first = order_ladder_v1(cands.clone(), &Default::default(), false, |_m: &str| { - 475
Some(3.0) - 476
}); - 477
assert_eq!(cheap_first[0].model, "cheap-haiku"); - 478
let quality_first = order_ladder_v1(cands, &Default::default(), true, |_m: &str| Some(3.0)); - 479
assert_eq!(quality_first[0].model, "big-claude-opus"); - 480
} - 481
- 482
#[test] - 483
fn high_latency_degrades_rank() { - 484
let mut ev = EvidenceSnapshot::default(); - 485
ev.by_key.insert( - 486
("p".into(), "slow".into()), - 487
ModelEvidence { - 488
success: 5, - 489
p50_latency_ms: Some(29_000), - 490
..Default::default() - 491
}, - 492
); - 493
ev.by_key.insert( - 494
("p".into(), "quick".into()), - 495
ModelEvidence { - 496
success: 5, - 497
p50_latency_ms: Some(300), - 498
..Default::default() - 499
}, - 500
); - 501
let ordered = order_ladder_v1( - 502
vec![leg("p", "slow"), leg("p", "quick")], - 503
&ev, - 504
false, - 505
|_m: &str| Some(3.0), - 506
); - 507
assert_eq!(ordered[0].model, "quick"); - 508
} - 509
- 510
fn leg2(p: &str, m: &str) -> RouteLeg { - 511
leg(p, m) - 512
} - 513
- 514
#[test] - 515
fn credential_identity_round_trips_and_legacy_route_defaults_empty() { - 516
let scoped = RouteLeg { - 517
provider: "openrouter".into(), - 518
model: "model-a".into(), - 519
dialect: EndpointDialect::Responses, - 520
credential_id: Some("deadbeef".into()), - 521
}; - 522
let encoded = serde_json::to_string(&scoped).unwrap(); - 523
let decoded: RouteLeg = serde_json::from_str(&encoded).unwrap(); - 524
assert_eq!(decoded, scoped); - 525
- 526
let legacy: RouteLeg = - 527
serde_json::from_str(r#"{"provider":"ollama","model":"local-model"}"#).unwrap(); - 528
assert_eq!(legacy.dialect, EndpointDialect::ChatCompletions); - 529
assert_eq!(legacy.credential_id, None); - 530
} - 531
- 532
#[test] - 533
fn agentic_openai_routes_use_responses_without_model_name_knowledge() { - 534
assert_eq!( - 535
EndpointDialect::for_provider("openai", true), - 536
EndpointDialect::Responses - 537
); - 538
assert_eq!( - 539
EndpointDialect::for_provider("openrouter", true), - 540
EndpointDialect::Responses - 541
); - 542
assert_eq!( - 543
EndpointDialect::for_provider("ollama", true), - 544
EndpointDialect::ChatCompletions - 545
); - 546
} - 547
- 548
#[test] - 549
fn v2_demand_scores_bands() { - 550
let light = score_demand(DemandInput { - 551
estimated_input_tokens: 1_000, - 552
output_budget_tokens: 1_000, - 553
tool_count: 0, - 554
..Default::default() - 555
}); - 556
assert_eq!(light.band, DemandBand::Low); - 557
let heavy = score_demand(DemandInput { - 558
estimated_input_tokens: 60_000, - 559
output_budget_tokens: 16_000, - 560
tool_count: 20, - 561
structured_output: true, - 562
reasoning_required: true, - 563
evidence_required: true, - 564
}); - 565
assert_eq!(heavy.band, DemandBand::High); - 566
// Unknown context reads as moderate, never zero. - 567
let unknown_ctx = score_demand(DemandInput { - 568
estimated_input_tokens: 0, - 569
output_budget_tokens: 4_000, - 570
tool_count: 6, - 571
..Default::default() - 572
}); - 573
assert_eq!(unknown_ctx.band, DemandBand::Moderate); - 574
} - 575
- 576
#[test] - 577
fn v2_resolves_objective_from_band_and_override() { - 578
assert_eq!( - 579
QualityObjective::resolve(None, DemandBand::High), - 580
QualityObjective::QualityCritical - 581
); - 582
assert_eq!( - 583
QualityObjective::resolve(None, DemandBand::Low), - 584
QualityObjective::Utility - 585
); - 586
assert_eq!( - 587
QualityObjective::resolve(Some("utility"), DemandBand::High), - 588
QualityObjective::Utility, - 589
"explicit override beats the band" - 590
); - 591
} - 592
- 593
#[test] - 594
fn v2_utility_orders_strictly_cheap_first() { - 595
let cands = vec![leg2("p", "big-expensive"), leg2("q", "cheap-mini")]; - 596
let ordered = order_ladder_v2( - 597
cands, - 598
&Default::default(), - 599
&BeliefMap::default(), - 600
QualityObjective::Utility, - 601
&[], - 602
|m: &str| { - 603
if m.contains("expensive") { - 604
Some(75.0) - 605
} else { - 606
Some(1.0) - 607
} - 608
}, - 609
); - 610
assert_eq!(ordered[0].model, "cheap-mini"); - 611
} - 612
- 613
#[test] - 614
fn v2_quality_critical_promotes_hinted_models_over_cost() { - 615
let cands = vec![leg2("p", "cheap-haiku"), leg2("q", "frontier-opus-xl")]; - 616
let ordered = order_ladder_v2( - 617
cands, - 618
&Default::default(), - 619
&BeliefMap::default(), - 620
QualityObjective::QualityCritical, - 621
&["opus".to_string()], - 622
|_m: &str| Some(3.0), - 623
); - 624
assert_eq!(ordered[0].model, "frontier-opus-xl"); - 625
} - 626
- 627
#[test] - 628
fn v2_doubted_leg_ranks_below_trusted_peers_even_when_cheaper() { - 629
let mut beliefs = BeliefMap::default(); - 630
beliefs.multipliers.insert(("p".into(), "m".into()), 0.5); - 631
let cands = vec![leg2("p", "m"), leg2("p", "trusted")]; - 632
for objective in [ - 633
QualityObjective::Utility, - 634
QualityObjective::Balanced, - 635
QualityObjective::QualityCritical, - 636
] { - 637
let ordered = order_ladder_v2( - 638
cands.clone(), - 639
&Default::default(), - 640
&beliefs, - 641
objective, - 642
&[], - 643
|_m: &str| Some(3.0), - 644
); - 645
assert_eq!(ordered[0].model, "trusted", "{objective:?}"); - 646
} - 647
} - 648
- 649
#[test] - 650
fn v2_known_pricing_outranks_unknown_before_cost_math() { - 651
let cands = vec![leg2("p", "priced-cheap"), leg2("q", "unpriced")]; - 652
let ordered = order_ladder_v2( - 653
cands, - 654
&Default::default(), - 655
&BeliefMap::default(), - 656
QualityObjective::Utility, - 657
&[], - 658
|m: &str| if m == "unpriced" { None } else { Some(0.5) }, - 659
); - 660
assert_eq!(ordered[0].model, "priced-cheap"); - 661
} - 662
- 663
#[test] - 664
fn v2_is_deterministic() { - 665
let mk = || vec![leg2("a", "z"), leg2("b", "y"), leg2("c", "x")]; - 666
let one = order_ladder_v2( - 667
mk(), - 668
&Default::default(), - 669
&BeliefMap::default(), - 670
QualityObjective::Balanced, - 671
&[], - 672
|_| None, - 673
); - 674
let two = order_ladder_v2( - 675
mk(), - 676
&Default::default(), - 677
&BeliefMap::default(), - 678
QualityObjective::Balanced, - 679
&[], - 680
|_| None, - 681
); - 682
assert_eq!(one, two); - 683
} - 684
} - 685
Indexing the workspace…
Vakyartha documentation is discovering safe artifacts, anchors, and source references.