- 10574
core.maybe_start_capacity_probe(&hosted_leg).await; - 10575
- 10576
assert!( - 10577
hosted_profile.provenance.rungs.is_empty(), - 10578
"no ladder rung should run for a hosted leg without opt-in" - 10579
); - 10580
assert_eq!(hosted_profile.instruction_horizon.confidence, 0.3); - 10581
assert_eq!( - 10582
hosted_profile.instruction_horizon.tokens, hosted_profile.declared_window, - 10583
"an unprobed hosted profile starts the horizon at the declared window" - 10584
); - 10585
- 10586
vak_config::clear_override("VAK_OLLAMA_BASE_URL"); - 10587
} - 10588
- 10589
/// Always follows the probe instruction; a request identical to the - 10590
/// previous one (a repeated sample inside a rung, or the cache rung's - 10591
/// second request) reports a cache hit, so the test can assert - 10592
/// `CacheBehaviour::ProviderReported` end to end through - 10593
/// `capacity_profile_for` without any network I/O. - 10594
struct FollowsProbeAndReportsCacheOnThirdCall { - 10595
calls: std::sync::atomic::AtomicU32, - 10596
last_fingerprint: std::sync::Mutex<Option<String>>, - 10597
} - 10598
- 10599
#[async_trait::async_trait] - 10600
impl Provider for FollowsProbeAndReportsCacheOnThirdCall { - 10601
fn name(&self) -> &str { - 10602
"ollama" - 10603
} - 10604
- 10605
async fn stream( - 10606
&self, - 10607
_request: ChatRequest, - 10608
_cancel: CancellationToken, - 10609
) -> Result<vak_llm::EventStream, LlmError> { - 10610
let call = self.calls.fetch_add(1, std::sync::atomic::Ordering::SeqCst); - 10611
let fingerprint = serde_json::to_string(&_request.messages).unwrap_or_default(); - 10612
let repeated = self - 10613
.last_fingerprint - 10614
.lock() - 10615
.map(|mut last| { - 10616
let same = last.as_deref() == Some(fingerprint.as_str()); - 10617
*last = Some(fingerprint); - 10618
same - 10619
}) - 10620
.unwrap_or(false); - 10621
let (mut sink, rx) = vak_llm::stream::channel(4); - 10622
sink.close_message(AssistantMessage { - 10623
content: vec![ContentBlock::ToolUse { - 10624
id: format!("probe-{call}"), - 10625
name: "probe_ack".into(), - 10626
input: serde_json::json!({"ok": true}), - 10627
}], - 10628
stop_reason: StopReason::ToolUse, - 10629
usage: Usage { - 10630
input_tokens: 4_000, - 10631
output_tokens: 5, - 10632
cache_read_input_tokens: repeated.then_some(3_500), - 10633
..Default::default() - 10634
}, - 10635
model: "fake-ollama-model".into(), - 10636
response_id: None, - 10637
}) - 10638
.await; - 10639
Ok(rx) - 10640
} - 10641
} - 10642
- 10643
#[tokio::test(flavor = "multi_thread", worker_threads = 4)] - 10644
async fn cache_rung_detects_a_provider_reported_hit_and_records_both_rungs_in_signals() { - 10645
super::isolate_global_config(); - 10646
let dir = tempfile::tempdir().unwrap(); - 10647
let core = Core::new_with_trust(dir.path().to_path_buf(), true).unwrap(); - 10648
let unused_port = { - 10649
let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); - 10650
listener.local_addr().unwrap().port() - 10651
}; - 10652
vak_config::set_override( - 10653
"VAK_OLLAMA_BASE_URL", - 10654
format!("http://127.0.0.1:{unused_port}/v1"), - 10655
); - 10656
core.inner.registry.register("ollama", |_auth| { - 10657
Ok(Arc::new(FollowsProbeAndReportsCacheOnThirdCall { - 10658
calls: std::sync::atomic::AtomicU32::new(0), - 10659
last_fingerprint: std::sync::Mutex::new(None), - 10660
}) as Arc<dyn Provider>) - 10661
}); - 10662
- 10663
let loopback_leg = vak_llm::RouteLeg { - 10664
provider: "ollama".into(), - 10665
model: "fake-ollama-model".into(), - 10666
dialect: vak_llm::EndpointDialect::default(), - 10667
credential_id: None, - 10668
}; - 10669
let session_path = dir.path().join("cache-rung-session.jsonl"); - 10670
let mut session = SessionLog::create(session_path, header()).unwrap(); - 10671
let _ = core.capacity_profile_for(&loopback_leg, &mut session).await; - 10672
core.maybe_start_capacity_probe(&loopback_leg).await; - 10673
let key = vak_context::capacity::ProfileKey { - 10674
provider: "ollama".into(), - 10675
model: "fake-ollama-model".into(), - 10676
quantisation: None, - 10677
}; - 10678
let probed = wait_for_capacity_probe(&core, &key).await; - 10679
- 10680
assert_eq!( - 10681
probed.cache, - 10682
vak_context::capacity::CacheBehaviour::ProviderReported - 10683
); - 10684
assert!( - 10685
probed - 10686
.provenance - 10687
.signals - 10688
.iter() - 10689
.any(|s| s.contains("cache rung") && s.contains("ProviderReported")), - 10690
"both cache-rung latencies and the outcome must be recorded as \ - 10691
provenance signals: {:?}", - 10692
probed.provenance.signals - 10693
); - 10694
- 10695
vak_config::clear_override("VAK_OLLAMA_BASE_URL"); - 10696
} - 10697
} - 10698
Indexing the workspace…
Vakyartha documentation is discovering safe artifacts, anchors, and source references.