- 1
//! Evaluating whether a commitment is actually done. - 2
//! - 3
//! # The separation of powers - 4
//! - 5
//! The model may **propose** criteria. It may never **mark one passed**. That - 6
//! is the same split as permission-before-dispatch, applied to completion: the - 7
//! thing being evaluated does not get to grade itself. - 8
//! - 9
//! Mechanically that means every criterion carries a [`Satisfaction`] derived - 10
//! from *how it was established*, not from how confident anyone is: - 11
//! - 12
//! | Criterion | Strength | Because | - 13
//! |---|---|---| - 14
//! | `Shell`, `FileExists`, `FileContains`, `ToolSucceeded`, `FlowCompleted` | `Observed` | the runtime ran it against the world | - 15
//! | `ExternalReceipt` | `Attested` | a third party confirmed it | - 16
//! | `Semantic` | `Asserted` | only the model's judgement backs it | - 17
//! - 18
//! A commitment whose `evidence` axis was `Verified` therefore cannot be - 19
//! closed by a pile of `Semantic` criteria, however emphatic the transcript - 20
//! is. That is the whole point. - 21
//! - 22
//! # Why the evaluator is a trait - 23
//! - 24
//! Running a shell command is a permissioned effect that belongs behind - 25
//! vak's tool broker and permission engine. This crate defines the contract - 26
//! and the strength mapping; `vak-core` supplies the implementation that - 27
//! actually crosses that boundary. - 28
- 29
use vak_intent::Satisfaction; - 30
use vak_session::types::{CriterionKind, CriterionResult, WorkCriterion}; - 31
- 32
/// How strongly a criterion of this kind is backed once it passes. - 33
/// - 34
/// This mapping is the closure invariant's factual half, and it is - 35
/// deliberately a total function over `CriterionKind` so that adding a new - 36
/// kind forces an explicit decision about its evidentiary weight. - 37
pub fn strength_of(kind: &CriterionKind) -> Satisfaction { - 38
match kind { - 39
// The runtime executed something and observed the result. - 40
CriterionKind::Shell { .. } - 41
| CriterionKind::FileExists { .. } - 42
| CriterionKind::FileContains { .. } - 43
| CriterionKind::ToolSucceeded { .. } - 44
| CriterionKind::FlowCompleted { .. } => Satisfaction::Observed, - 45
// Someone outside this runtime vouched for it. - 46
CriterionKind::ExternalReceipt { .. } => Satisfaction::Attested, - 47
// Nothing but the model's own judgement. - 48
CriterionKind::Semantic => Satisfaction::Asserted, - 49
} - 50
} - 51
- 52
/// Whether this criterion can be checked without a model. - 53
pub fn is_machine_checkable(kind: &CriterionKind) -> bool { - 54
strength_of(kind).rank() >= Satisfaction::Observed.rank() - 55
} - 56
- 57
/// One criterion's evaluation. - 58
#[derive(Debug, Clone, PartialEq)] - 59
pub struct Evaluation { - 60
pub criterion_id: String, - 61
pub result: CriterionResult, - 62
pub strength: Satisfaction, - 63
} - 64
- 65
impl Evaluation { - 66
/// Record a runtime-observed outcome. - 67
pub fn observed(criterion: &WorkCriterion, result: CriterionResult) -> Self { - 68
Evaluation { - 69
criterion_id: criterion.criterion_id.clone(), - 70
// A criterion that failed or could not be determined carries no - 71
// evidentiary weight, so its strength is the floor regardless of - 72
// kind. Only a pass earns the kind's strength. - 73
strength: match &result { - 74
CriterionResult::Passed { .. } => strength_of(&criterion.kind), - 75
_ => Satisfaction::Asserted, - 76
}, - 77
result, - 78
} - 79
} - 80
- 81
/// Record a human's attestation. The only path to `Attested` other than an - 82
/// external receipt. - 83
pub fn attested(criterion_id: impl Into<String>, by: &str, note: &str) -> Self { - 84
Evaluation { - 85
criterion_id: criterion_id.into(), - 86
result: CriterionResult::Passed { - 87
evidence: format!("attested by {by}: {note}"), - 88
}, - 89
strength: Satisfaction::Attested, - 90
} - 91
} - 92
- 93
/// Record that the runtime could not determine the outcome. - 94
/// - 95
/// Distinct from a failure: a check that could not run says nothing about - 96
/// the work, and recording it as a failure would be as dishonest as - 97
/// recording it as a pass. - 98
pub fn unknown(criterion_id: impl Into<String>, reason: impl Into<String>) -> Self { - 99
Evaluation { - 100
criterion_id: criterion_id.into(), - 101
result: CriterionResult::Unknown { - 102
reason: reason.into(), - 103
}, - 104
strength: Satisfaction::Asserted, - 105
} - 106
} - 107
- 108
pub fn passed(&self) -> bool { - 109
matches!(self.result, CriterionResult::Passed { .. }) - 110
} - 111
} - 112
- 113
/// Evaluates criteria against the world. - 114
/// - 115
/// Implemented in `vak-core`, where the tool broker, permission engine and - 116
/// sandbox live. A criterion evaluation is an ordinary permissioned effect and - 117
/// crosses exactly the same boundary as any other. - 118
#[allow(async_fn_in_trait)] - 119
pub trait CriterionEvaluator { - 120
/// Evaluate one criterion. Implementations must return - 121
/// [`CriterionResult::Unknown`] rather than a failure when the check could - 122
/// not be performed — a denied permission, a missing binary, a timeout. - 123
async fn evaluate(&self, criterion: &WorkCriterion) -> Evaluation; - 124
} - 125
- 126
/// Evaluate every machine-checkable criterion in a spec. - 127
/// - 128
/// `Semantic` criteria are skipped: there is nothing for the runtime to - 129
/// observe, and asking a model to confirm its own work would launder an - 130
/// assertion into something that looks stronger than it is. They stay at - 131
/// `Asserted` unless a human attests them. - 132
pub async fn evaluate_all<E: CriterionEvaluator>( - 133
evaluator: &E, - 134
criteria: &[WorkCriterion], - 135
) -> Vec<Evaluation> { - 136
let mut out = Vec::new(); - 137
for criterion in criteria { - 138
if !is_machine_checkable(&criterion.kind) { - 139
continue; - 140
} - 141
out.push(evaluator.evaluate(criterion).await); - 142
} - 143
out - 144
} - 145
- 146
#[cfg(test)] - 147
#[allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] - 148
mod tests { - 149
use super::*; - 150
use std::path::PathBuf; - 151
- 152
fn criterion(id: &str, kind: CriterionKind) -> WorkCriterion { - 153
WorkCriterion { - 154
criterion_id: id.into(), - 155
statement: format!("criterion {id}"), - 156
kind, - 157
required: true, - 158
} - 159
} - 160
- 161
#[test] - 162
fn semantic_criteria_are_only_ever_asserted() { - 163
assert_eq!( - 164
strength_of(&CriterionKind::Semantic), - 165
Satisfaction::Asserted - 166
); - 167
assert!(!is_machine_checkable(&CriterionKind::Semantic)); - 168
// And therefore cannot discharge a `Verified` requirement. - 169
assert!(!Satisfaction::Asserted.satisfies(Satisfaction::Observed)); - 170
} - 171
- 172
#[test] - 173
fn runtime_checked_criteria_are_observed() { - 174
for kind in [ - 175
CriterionKind::Shell { - 176
command: "cargo test".into(), - 177
}, - 178
CriterionKind::FileExists { - 179
path: PathBuf::from("out.txt"), - 180
}, - 181
CriterionKind::FileContains { - 182
path: PathBuf::from("out.txt"), - 183
pattern: "ok".into(), - 184
}, - 185
CriterionKind::ToolSucceeded { - 186
tool: "bash".into(), - 187
}, - 188
CriterionKind::FlowCompleted { - 189
flow: "release".into(), - 190
}, - 191
] { - 192
assert_eq!(strength_of(&kind), Satisfaction::Observed); - 193
assert!(is_machine_checkable(&kind)); - 194
} - 195
} - 196
- 197
#[test] - 198
fn external_receipts_are_attested() { - 199
let kind = CriterionKind::ExternalReceipt { - 200
integration: "stripe".into(), - 201
}; - 202
assert_eq!(strength_of(&kind), Satisfaction::Attested); - 203
} - 204
- 205
/// A failed or undetermined check must not carry its kind's strength — - 206
/// otherwise a failing shell criterion would count as `Observed` evidence. - 207
#[test] - 208
fn only_a_pass_earns_the_kinds_strength() { - 209
let shell = criterion( - 210
"c1", - 211
CriterionKind::Shell { - 212
command: "false".into(), - 213
}, - 214
); - 215
let failed = Evaluation::observed( - 216
&shell, - 217
CriterionResult::Failed { - 218
reason: "exit 1".into(), - 219
}, - 220
); - 221
assert_eq!(failed.strength, Satisfaction::Asserted); - 222
assert!(!failed.passed()); - 223
- 224
let passed = Evaluation::observed( - 225
&shell, - 226
CriterionResult::Passed { - 227
evidence: "exit 0".into(), - 228
}, - 229
); - 230
assert_eq!(passed.strength, Satisfaction::Observed); - 231
} - 232
- 233
#[test] - 234
fn an_undeterminable_check_is_unknown_not_failed() { - 235
let evaluation = Evaluation::unknown("c1", "permission denied"); - 236
assert!(matches!(evaluation.result, CriterionResult::Unknown { .. })); - 237
assert!(!evaluation.passed()); - 238
} - 239
- 240
#[test] - 241
fn human_attestation_reaches_the_top_of_the_lattice() { - 242
let evaluation = Evaluation::attested("c1", "nisheeth", "checked the dashboard"); - 243
assert_eq!(evaluation.strength, Satisfaction::Attested); - 244
assert!(evaluation.strength.satisfies(Satisfaction::Attested)); - 245
} - 246
- 247
struct StubEvaluator; - 248
- 249
impl CriterionEvaluator for StubEvaluator { - 250
async fn evaluate(&self, criterion: &WorkCriterion) -> Evaluation { - 251
Evaluation::observed( - 252
criterion, - 253
CriterionResult::Passed { - 254
evidence: "stub".into(), - 255
}, - 256
) - 257
} - 258
} - 259
- 260
#[tokio::test] - 261
async fn evaluate_all_skips_semantic_criteria() { - 262
let criteria = vec![ - 263
criterion("c1", CriterionKind::Semantic), - 264
criterion( - 265
"c2", - 266
CriterionKind::Shell { - 267
command: "true".into(), - 268
}, - 269
), - 270
]; - 271
let evaluations = evaluate_all(&StubEvaluator, &criteria).await; - 272
assert_eq!(evaluations.len(), 1); - 273
assert_eq!(evaluations[0].criterion_id, "c2"); - 274
} - 275
} - 276
Indexing the workspace…
Vakyartha documentation is discovering safe artifacts, anchors, and source references.