- 1
//! Strands: the parts of a request that are separate pieces of work. - 2
//! - 3
//! A request is rarely one thing. "Explain the parser, then refactor it, and - 4
//! also check whether the nightly job ran" is three pieces of work with three - 5
//! different readings, and the runtime has to know that: the tool surface must - 6
//! cover all three, the stop rule must be the strictest of the three, and the - 7
//! nightly-job question has nothing to do with the parser — it may well be a - 8
//! thread the user opened two turns ago. - 9
//! - 10
//! So a turn resolves to a list of [`Strand`]s. Each carries its own - 11
//! [`Reading`] and [`Engagement`], its relation to the strands beside it, and - 12
//! its lineage to threads from earlier turns. The turn's engagement is - 13
//! [`Engagement::compose`] over the strands, and the turn's composite reading - 14
//! (kept on [`crate::Intent::reading`] for everything that wants one answer) - 15
//! is the most consequential strand widened by the others. - 16
//! - 17
//! Segmentation is deterministic, like the rest of tier 1: sentence - 18
//! boundaries, enumerated items, and a short list of sequencing and addition - 19
//! markers. Plain "and" is deliberately not a boundary — "explain what this - 20
//! and that mean" is one question — and a clause that carries no act signal of - 21
//! its own is folded back into its neighbour rather than becoming a strand - 22
//! that says nothing. - 23
- 24
use std::collections::BTreeSet; - 25
- 26
use serde::{Deserialize, Serialize}; - 27
- 28
use crate::axes::Act; - 29
use crate::engage::Engagement; - 30
use crate::reading::Reading; - 31
- 32
/// How a strand relates to the strands beside it in the same request. - 33
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] - 34
#[serde(tag = "kind", rename_all = "kebab-case")] - 35
pub enum StrandRelation { - 36
/// Stands on its own; order does not matter. - 37
Independent, - 38
/// Must happen after `after` ("then", "after that", "finally"). - 39
Sequential { after: String }, - 40
/// Refers to `on`'s result ("… and summarise it"). - 41
Dependent { on: String }, - 42
} - 43
- 44
/// How a strand relates to work from earlier turns. - 45
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] - 46
#[serde(tag = "kind", rename_all = "kebab-case")] - 47
pub enum Lineage { - 48
/// A thread of its own. - 49
New, - 50
/// Carries on an open thread. - 51
Continues { thread_id: String }, - 52
/// Amends an open thread without discarding it. Only ever set from an - 53
/// explicit human command, never inferred (docs/design/47, control plane). - 54
Corrects { thread_id: String }, - 55
/// Supersedes an open thread. Same rule. The replacing strand starts a - 56
/// thread of its own; `thread_id` names the one it replaces. - 57
Replaces { thread_id: String }, - 58
} - 59
- 60
impl Lineage { - 61
/// The thread this strand carries on, if it carries one on. A - 62
/// replacement does not: it supersedes its thread and starts its own. - 63
pub fn continued_thread(&self) -> Option<&str> { - 64
match self { - 65
Lineage::New | Lineage::Replaces { .. } => None, - 66
Lineage::Continues { thread_id } | Lineage::Corrects { thread_id } => Some(thread_id), - 67
} - 68
} - 69
- 70
/// The thread this strand supersedes, if any. - 71
pub fn replaced_thread(&self) -> Option<&str> { - 72
match self { - 73
Lineage::Replaces { thread_id } => Some(thread_id), - 74
_ => None, - 75
} - 76
} - 77
} - 78
- 79
/// One piece of work inside a request. - 80
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] - 81
pub struct Strand { - 82
/// `{turn_id}.{index}`: unique across turns, sessions and workspaces, - 83
/// because the host mints the turn id (a UUIDv7). - 84
pub strand_id: String, - 85
/// The thread this strand belongs to across turns. Equal to `strand_id` - 86
/// for a new thread and for a replacement; inherited for a continuation - 87
/// or a correction. - 88
pub thread_id: String, - 89
/// The clause, verbatim after scaffolding removal. - 90
pub text: String, - 91
pub reading: Reading, - 92
pub relation: StrandRelation, - 93
pub lineage: Lineage, - 94
pub engagement: Engagement, - 95
} - 96
- 97
/// What the host knows about a thread that is still open, for lineage. - 98
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] - 99
pub struct ThreadFact { - 100
pub thread_id: String, - 101
pub act: Act, - 102
#[serde(default)] - 103
pub domains: BTreeSet<String>, - 104
/// A few content words from the thread's text, lowercase, for overlap. - 105
#[serde(default)] - 106
pub keywords: BTreeSet<String>, - 107
} - 108
- 109
/// A lineage the host already knows, from an explicit command. - 110
/// - 111
/// `/goal fix …` and `/goal replace …` are the only way a strand becomes a - 112
/// correction or a replacement; the resolver never infers either, because a - 113
/// wrongly inferred replacement discards work. - 114
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] - 115
#[serde(rename_all = "kebab-case")] - 116
pub enum LineageHint { - 117
Corrects, - 118
Replaces, - 119
} - 120
- 121
// ------------------------------------------------------------ segmentation --- - 122
- 123
/// Why a clause was split from the one before it. - 124
#[derive(Debug, Clone, Copy, PartialEq, Eq)] - 125
pub(crate) enum Boundary { - 126
/// The first clause. - 127
Start, - 128
/// A sequencing marker: the clause happens after the previous one. - 129
Sequence, - 130
/// An addition marker or a sentence boundary. - 131
Addition, - 132
} - 133
- 134
#[derive(Debug, Clone, PartialEq, Eq)] - 135
pub(crate) struct Clause { - 136
pub text: String, - 137
pub boundary: Boundary, - 138
/// The clause ended with a question mark. - 139
pub question: bool, - 140
} - 141
- 142
/// Markers that begin a new clause. Matched as whole words on the lowercase - 143
/// token stream, longest first, so "and then" is not read as "and" + "then". - 144
const SEQUENCE_MARKERS: &[&str] = &[ - 145
"and then", - 146
"then", - 147
"after that", - 148
"afterwards", - 149
"and finally", - 150
"finally", - 151
"and after that", - 152
"once that is done", - 153
"once done", - 154
]; - 155
// `plus` and `next` are not here: "plus-size", "the next release". - 156
const ADDITION_MARKERS: &[&str] = &["and also", "also", "additionally", "as well as"]; - 157
- 158
/// The segmentation vocabulary, for the lexicon digest. - 159
pub(crate) fn segmentation_fingerprint(out: &mut String) { - 160
for marker in SEQUENCE_MARKERS { - 161
out.push_str(&format!("seq:{marker}\n")); - 162
} - 163
for marker in ADDITION_MARKERS { - 164
out.push_str(&format!("add:{marker}\n")); - 165
} - 166
} - 167
- 168
/// Split a request into clauses. - 169
/// - 170
/// Boundaries, in priority order: enumerated list items; newlines; sentence - 171
/// terminators; `;`; sequencing markers; addition markers. Each marker must - 172
/// be a whole word (or run of words), and the marker itself is dropped from - 173
/// the clause that follows it. - 174
pub(crate) fn segment(text: &str) -> Vec<Clause> { - 175
let mut clauses: Vec<Clause> = Vec::new(); - 176
- 177
// Pass 1: hard boundaries — list items, newlines, sentence ends, `;`. - 178
let mut pieces: Vec<(String, Boundary, bool)> = Vec::new(); - 179
let mut current = String::new(); - 180
let mut chars = text.chars().peekable(); - 181
while let Some(c) = chars.next() { - 182
match c { - 183
'\n' | ';' => { - 184
flush(&mut pieces, &mut current, Boundary::Addition, false); - 185
} - 186
'.' | '!' | '?' => { - 187
// A terminator followed by whitespace or end ends a sentence; - 188
// "v1.2" and "e.g." do not. - 189
let ends = chars.peek().is_none_or(|next| next.is_whitespace()); - 190
// "1. run the tests": the dot belongs to the list marker. - 191
let list_marker = c == '.' && { - 192
let head = current.trim(); - 193
!head.is_empty() - 194
&& head.len() <= 3 - 195
&& head.chars().all(|ch| ch.is_ascii_digit()) - 196
}; - 197
// "e.g." / "i.e.": a dot after a single letter that itself - 198
// follows a dot. - 199
let abbreviation = c == '.' && { - 200
let mut tail = current.chars().rev(); - 201
tail.next().is_some_and(|ch| ch.is_ascii_alphabetic()) - 202
&& tail.next() == Some('.') - 203
}; - 204
if ends && !list_marker && !abbreviation { - 205
flush(&mut pieces, &mut current, Boundary::Addition, c == '?'); - 206
} else { - 207
current.push(c); - 208
} - 209
} - 210
_ => current.push(c), - 211
} - 212
} - 213
flush(&mut pieces, &mut current, Boundary::Addition, false); - 214
- 215
// Pass 2: soft boundaries inside each piece — sequencing and addition - 216
// markers, as whole words. A question mark belongs to the last clause of - 217
// its sentence. - 218
for (index, (piece, boundary, question)) in pieces.into_iter().enumerate() { - 219
let piece = strip_list_marker(&piece); - 220
let subs = split_on_markers(piece); - 221
let last = subs.len().saturating_sub(1); - 222
for (position, (sub, sub_boundary)) in subs.into_iter().enumerate() { - 223
let boundary = if position == 0 { - 224
if index == 0 { - 225
Boundary::Start - 226
} else { - 227
boundary - 228
} - 229
} else { - 230
sub_boundary - 231
}; - 232
let trimmed = sub.trim().trim_matches(',').trim(); - 233
if !trimmed.is_empty() { - 234
clauses.push(Clause { - 235
text: trimmed.to_string(), - 236
boundary, - 237
question: question && position == last, - 238
}); - 239
} - 240
} - 241
} - 242
if let Some(first) = clauses.first_mut() { - 243
first.boundary = Boundary::Start; - 244
} - 245
clauses - 246
} - 247
- 248
fn flush( - 249
pieces: &mut Vec<(String, Boundary, bool)>, - 250
current: &mut String, - 251
boundary: Boundary, - 252
question: bool, - 253
) { - 254
let trimmed = current.trim(); - 255
if !trimmed.is_empty() { - 256
pieces.push((trimmed.to_string(), boundary, question)); - 257
} - 258
current.clear(); - 259
} - 260
- 261
/// `- item`, `* item`, `1. item`, `1) item` → `item`. - 262
fn strip_list_marker(piece: &str) -> &str { - 263
let trimmed = piece.trim_start(); - 264
if let Some(rest) = trimmed - 265
.strip_prefix("- ") - 266
.or_else(|| trimmed.strip_prefix("* ")) - 267
{ - 268
return rest; - 269
} - 270
let digits = trimmed.chars().take_while(char::is_ascii_digit).count(); - 271
if digits > 0 && digits <= 3 { - 272
let rest = &trimmed[digits..]; - 273
if let Some(rest) = rest.strip_prefix(". ").or_else(|| rest.strip_prefix(") ")) { - 274
return rest; - 275
} - 276
} - 277
trimmed - 278
} - 279
- 280
/// Split one sentence on sequencing / addition markers. - 281
/// - 282
/// Works on the original text so the clauses keep their spelling; markers are - 283
/// located by scanning words with their byte offsets. - 284
fn split_on_markers(text: &str) -> Vec<(&str, Boundary)> { - 285
// Word spans: (start, end) byte offsets of alphanumeric runs. - 286
let mut spans: Vec<(usize, usize)> = Vec::new(); - 287
let mut start: Option<usize> = None; - 288
for (i, c) in text.char_indices() { - 289
if c.is_ascii_alphanumeric() || c == '\'' { - 290
if start.is_none() { - 291
start = Some(i); - 292
} - 293
} else if let Some(s) = start.take() { - 294
spans.push((s, i)); - 295
} - 296
} - 297
if let Some(s) = start { - 298
spans.push((s, text.len())); - 299
} - 300
let words: Vec<String> = spans - 301
.iter() - 302
.map(|(s, e)| text[*s..*e].to_ascii_lowercase()) - 303
.collect(); - 304
- 305
let mut out = Vec::new(); - 306
let mut clause_start = 0usize; - 307
let mut i = 0usize; - 308
while i < words.len() { - 309
let mut matched: Option<(usize, Boundary)> = None; - 310
for (markers, boundary) in [ - 311
(SEQUENCE_MARKERS, Boundary::Sequence), - 312
(ADDITION_MARKERS, Boundary::Addition), - 313
] { - 314
for marker in markers { - 315
let needle: Vec<&str> = marker.split(' ').collect(); - 316
if i + needle.len() <= words.len() - 317
&& needle - 318
.iter() - 319
.zip(&words[i..i + needle.len()]) - 320
.all(|(a, b)| *a == b.as_str()) - 321
{ - 322
// Longest match wins. - 323
if matched.is_none_or(|(len, _)| needle.len() > len) { - 324
matched = Some((needle.len(), boundary)); - 325
} - 326
} - 327
} - 328
} - 329
match matched { - 330
Some((len, boundary)) => { - 331
let (marker_start, _) = spans[i]; - 332
let (_, marker_end) = spans[i + len - 1]; - 333
let head = &text[clause_start..marker_start]; - 334
if head.trim().trim_matches(',').trim().is_empty() { - 335
// "then" as the first word of this clause: just drop it. - 336
} else { - 337
out.push((head, Boundary::Start)); - 338
} - 339
// The boundary applies to what follows the marker. - 340
clause_start = marker_end; - 341
i += len; - 342
// Record which boundary the *next* clause gets by pushing a - 343
// placeholder we fix up below. - 344
out.push(("", boundary)); - 345
} - 346
_ => i += 1, - 347
} - 348
} - 349
out.push((&text[clause_start..], Boundary::Start)); - 350
- 351
// Fold placeholders: ("", b) followed by (clause, _) → (clause, b). - 352
let mut folded: Vec<(&str, Boundary)> = Vec::new(); - 353
let mut pending: Option<Boundary> = None; - 354
for (clause, boundary) in out { - 355
if clause.is_empty() { - 356
pending = Some(boundary); - 357
continue; - 358
} - 359
let boundary = pending.take().unwrap_or(boundary); - 360
folded.push((clause, boundary)); - 361
} - 362
folded - 363
} - 364
- 365
/// Content words of a clause, for cross-turn overlap. Short and common words - 366
/// are dropped so "the" does not link every thread to every other. - 367
pub fn keywords(text: &str) -> BTreeSet<String> { - 368
const STOP: &[&str] = &[ - 369
"the", "a", "an", "and", "or", "to", "of", "in", "on", "for", "it", "this", "that", "is", - 370
"are", "be", "with", "as", "at", "by", "from", "me", "my", "we", "our", "you", "your", - 371
"please", "can", "could", "would", "should", "do", "does", "did", "then", "also", "just", - 372
"now", "so", "if", "but", "not", "no", "yes", "into", "about", - 373
]; - 374
text.to_ascii_lowercase() - 375
.split(|c: char| !c.is_ascii_alphanumeric()) - 376
.filter(|w| w.len() >= 3 && !STOP.contains(w)) - 377
.map(str::to_string) - 378
.collect() - 379
} - 380
- 381
#[cfg(test)] - 382
#[allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] - 383
mod tests { - 384
use super::*; - 385
- 386
fn texts(text: &str) -> Vec<(String, Boundary)> { - 387
segment(text) - 388
.into_iter() - 389
.map(|c| (c.text, c.boundary)) - 390
.collect() - 391
} - 392
- 393
#[test] - 394
fn a_single_clause_is_one_strand() { - 395
assert_eq!( - 396
texts("refactor the parser"), - 397
vec![("refactor the parser".to_string(), Boundary::Start)] - 398
); - 399
} - 400
- 401
#[test] - 402
fn then_splits_into_a_sequence_and_drops_the_marker() { - 403
assert_eq!( - 404
texts("first explain the parser, then refactor it"), - 405
vec![ - 406
("first explain the parser".to_string(), Boundary::Start), - 407
("refactor it".to_string(), Boundary::Sequence), - 408
] - 409
); - 410
assert_eq!( - 411
texts("explain the parser and then refactor it"), - 412
vec![ - 413
("explain the parser".to_string(), Boundary::Start), - 414
("refactor it".to_string(), Boundary::Sequence), - 415
] - 416
); - 417
} - 418
- 419
#[test] - 420
fn also_and_sentences_are_additions() { - 421
assert_eq!( - 422
texts("fix the login bug. Also check whether the nightly job ran"), - 423
vec![ - 424
("fix the login bug".to_string(), Boundary::Start), - 425
( - 426
"check whether the nightly job ran".to_string(), - 427
Boundary::Addition - 428
), - 429
] - 430
); - 431
} - 432
- 433
#[test] - 434
fn markers_inside_words_do_not_split() { - 435
assert_eq!( - 436
texts("fix the authentication bug in the login handler").len(), - 437
1 - 438
); - 439
assert_eq!(texts("strengthen the plus-size handling").len(), 1); - 440
} - 441
- 442
#[test] - 443
fn plain_and_is_not_a_boundary() { - 444
assert_eq!(texts("explain what this and that mean").len(), 1); - 445
} - 446
- 447
#[test] - 448
fn enumerated_items_split_and_lose_their_markers() { - 449
let clauses = - 450
texts("do these:\n1. run the tests\n2. update the changelog\n- tag the release"); - 451
assert_eq!( - 452
clauses.iter().map(|(t, _)| t.as_str()).collect::<Vec<_>>(), - 453
vec![ - 454
"do these:", - 455
"run the tests", - 456
"update the changelog", - 457
"tag the release" - 458
] - 459
); - 460
} - 461
- 462
#[test] - 463
fn version_numbers_and_abbreviations_are_not_sentence_ends() { - 464
assert_eq!(texts("upgrade to v1.2.3 e.g. via the installer").len(), 1); - 465
} - 466
- 467
#[test] - 468
fn keywords_drop_stop_words() { - 469
let k = keywords("Explain the parser and then refactor it"); - 470
assert!(k.contains("parser") && k.contains("refactor") && k.contains("explain")); - 471
assert!(!k.contains("the") && !k.contains("and") && !k.contains("it")); - 472
} - 473
} - 474
Indexing the workspace…
Vakyartha documentation is discovering safe artifacts, anchors, and source references.