- 1
//! Cross-session recall (docs/design/23-memory.md): deterministic scan+score - 2
//! over the per-cwd JSONL ledgers. Relevance comes from BM25-style term - 3
//! frequency with IDF weighting, a whole-phrase bonus, an entity-token bonus - 4
//! (named entities matching query terms score higher), and a small recency - 5
//! nudge so fresher context wins ties. An mtime-keyed per-ledger cache - 6
//! (crate::index) makes warm queries skip the rescan (M1); `search_all` - 7
//! extends the same ranking across every project hash dir (personal-os P1). - 8
- 9
use std::collections::HashSet; - 10
use std::path::{Path, PathBuf}; - 11
- 12
use chrono::{DateTime, Utc}; - 13
- 14
use crate::SessionPath; - 15
use crate::index; - 16
- 17
pub const DEFAULT_LIMIT: usize = 8; - 18
const SNIPPET_CHARS: usize = 240; - 19
const SNIPPET_CONTEXT: usize = 60; - 20
const PHRASE_BONUS: f32 = 3.0; - 21
const RECENCY_NUDGE: f32 = 0.01; - 22
/// Curated memory outranks equally-relevant transcript lines - 23
/// (docs/design/26-learning.md). - 24
pub const MEMORY_BONUS: f32 = 2.5; - 25
- 26
/// BM25 parameters (from vakyartha simulation). - 27
const BM25_K1: f32 = 1.2; - 28
const BM25_B: f32 = 0.75; - 29
/// Bonus multiplier for entity tokens — named entities matching query terms - 30
/// are strong relevance signals. - 31
const ENTITY_TOKEN_BONUS: f32 = 4.0; - 32
/// Saturation point for term frequency in BM25 denominator. - 33
const BM25_AVGDL_APPROX: f32 = 100.0; - 34
- 35
#[derive(Debug, Clone, PartialEq, serde::Serialize)] - 36
pub struct SessionHit { - 37
pub session_id: String, - 38
pub entry_id: String, - 39
pub ts: DateTime<Utc>, - 40
pub role: String, - 41
pub score: f32, - 42
pub snippet: String, - 43
} - 44
- 45
/// A hit from cross-project recall: the session hit plus the project hash - 46
/// directory (`<home>/sessions/<hash>/`) it was found in (personal-os P1). - 47
#[derive(Debug, Clone, PartialEq, serde::Serialize)] - 48
pub struct ProjectHit { - 49
pub project_hash: String, - 50
#[serde(flatten)] - 51
pub hit: SessionHit, - 52
} - 53
- 54
#[derive(Debug, Clone)] - 55
struct RankedHit { - 56
hit: SessionHit, - 57
project_hash: Option<String>, - 58
} - 59
- 60
/// A curated document fed into recall alongside raw transcripts — today, - 61
/// parsed MEMORY.md blocks (docs/design/26-learning.md). - 62
#[derive(Debug, Clone)] - 63
pub struct ExternalDoc { - 64
/// Stable identifier surfaced in `session_id` of the hit (e.g. the tag). - 65
pub id: String, - 66
pub text: String, - 67
/// Original document timestamp. `None` preserves the legacy caller - 68
/// contract; durable memory callers always provide it. - 69
pub ts: Option<DateTime<Utc>>, - 70
/// Result role (`memory` or `profile`) shown to surfaces. - 71
pub role: Option<String>, - 72
} - 73
- 74
#[derive(Debug, thiserror::Error)] - 75
pub enum SearchError { - 76
#[error("io error: {0}")] - 77
Io(#[from] std::io::Error), - 78
} - 79
/// Search every ledger under `home` for the given workspace `cwd`. - 80
/// A session in `excluded` never matches: the current one (its content is - 81
/// already in the caller's context) and every session in the trash, which - 82
/// is hidden everywhere. - 83
pub fn search( - 84
sessions_home: &Path, - 85
cwd: &Path, - 86
query: &str, - 87
limit: usize, - 88
excluded: &std::collections::HashSet<String>, - 89
) -> Result<Vec<SessionHit>, SearchError> { - 90
search_extended(sessions_home, cwd, query, limit, excluded, &[]) - 91
} - 92
- 93
/// Same scan with curated documents (memory blocks) folded into ranking. - 94
/// Extras carry a bonus so hand-curated knowledge outranks raw history. - 95
pub fn search_extended( - 96
sessions_home: &Path, - 97
cwd: &Path, - 98
query: &str, - 99
limit: usize, - 100
excluded: &std::collections::HashSet<String>, - 101
extras: &[ExternalDoc], - 102
) -> Result<Vec<SessionHit>, SearchError> { - 103
let terms = tokenize_impl(query); - 104
let phrase = normalize_impl(query); - 105
if terms.is_empty() || phrase.is_empty() { - 106
return Ok(Vec::new()); - 107
} - 108
let limit = limit.clamp(1, 50); - 109
- 110
let mut ranked: Vec<RankedHit> = Vec::new(); - 111
for doc in extras { - 112
let base = score_text(&doc.text, &terms, &phrase); - 113
if base <= 0.0 { - 114
continue; - 115
} - 116
ranked.push(RankedHit { - 117
hit: SessionHit { - 118
session_id: doc.id.clone(), - 119
entry_id: String::new(), - 120
ts: doc.ts.unwrap_or_else(Utc::now), - 121
role: doc.role.clone().unwrap_or_else(|| "memory".into()), - 122
score: base + MEMORY_BONUS, - 123
snippet: snippet_for(&doc.text, &terms), - 124
}, - 125
project_hash: None, - 126
}); - 127
} - 128
- 129
let mut dirs = vec![SessionPath::sessions_dir(sessions_home, cwd)]; - 130
if let Some(parent) = sessions_home.parent().and_then(|p| p.parent()) { - 131
dirs.push(SessionPath::sessions_dir(parent, cwd)); - 132
} - 133
let agents_dir = sessions_home.join("agents"); - 134
if let Ok(entries) = std::fs::read_dir(&agents_dir) { - 135
for entry in entries.flatten() { - 136
let p = entry.path(); - 137
if p.is_dir() { - 138
dirs.push(SessionPath::sessions_dir(&p, cwd)); - 139
} - 140
} - 141
} - 142
dirs.dedup(); - 143
let mut seen_sessions = std::collections::HashSet::new(); - 144
for dir in dirs { - 145
if let Ok(read) = std::fs::read_dir(&dir) { - 146
for file_entry in read.flatten() { - 147
let path = file_entry.path(); - 148
if path.extension().and_then(|e| e.to_str()) != Some("jsonl") { - 149
continue; - 150
} - 151
let Some(session_id) = path.file_stem().and_then(|s| s.to_str()).map(String::from) - 152
else { - 153
continue; - 154
}; - 155
if excluded.contains(&session_id) || !seen_sessions.insert(session_id.clone()) { - 156
continue; - 157
} - 158
collect_ranked(&path, &session_id, &terms, &phrase, None, &mut ranked)?; - 159
} - 160
} - 161
} - 162
- 163
finalize(&mut ranked, limit); - 164
Ok(ranked.into_iter().map(|r| r.hit).collect()) - 165
} - 166
- 167
/// Search every ledger of every project hash dir under `<home>/sessions/` - 168
/// with the identical scoring, exclusions, and snippet rules as `search`. - 169
/// Hits are annotated with the hash dir they came from; the current-session - 170
/// exclusion applies across all dirs (personal-os P1). - 171
pub fn search_all( - 172
home: &Path, - 173
query: &str, - 174
limit: usize, - 175
excluded: &std::collections::HashSet<String>, - 176
) -> Result<Vec<ProjectHit>, SearchError> { - 177
search_all_extended(home, query, limit, excluded, &[]) - 178
} - 179
- 180
/// Cross-project search with curated documents included in the same ranking. - 181
pub fn search_all_extended( - 182
home: &Path, - 183
query: &str, - 184
limit: usize, - 185
excluded: &std::collections::HashSet<String>, - 186
extras: &[ExternalDoc], - 187
) -> Result<Vec<ProjectHit>, SearchError> { - 188
let terms = tokenize_impl(query); - 189
let phrase = normalize_impl(query); - 190
if terms.is_empty() || phrase.is_empty() { - 191
return Ok(Vec::new()); - 192
} - 193
let limit = limit.clamp(1, 50); - 194
- 195
let mut ranked: Vec<RankedHit> = extras - 196
.iter() - 197
.filter_map(|doc| { - 198
let base = score_text(&doc.text, &terms, &phrase); - 199
(base > 0.0).then(|| RankedHit { - 200
hit: SessionHit { - 201
session_id: doc.id.clone(), - 202
entry_id: String::new(), - 203
ts: doc.ts.unwrap_or_else(Utc::now), - 204
role: doc.role.clone().unwrap_or_else(|| "memory".into()), - 205
score: base + MEMORY_BONUS, - 206
snippet: snippet_for(&doc.text, &terms), - 207
}, - 208
project_hash: None, - 209
}) - 210
}) - 211
.collect(); - 212
let mut session_roots = vec![home.join("sessions")]; - 213
if let Some(parent) = home.parent().and_then(|p| p.parent()) { - 214
session_roots.push(parent.join("sessions")); - 215
} - 216
let agents_dir = home.join("agents"); - 217
if let Ok(entries) = std::fs::read_dir(&agents_dir) { - 218
for entry in entries.flatten() { - 219
let p = entry.path(); - 220
if p.is_dir() { - 221
session_roots.push(p.join("sessions")); - 222
} - 223
} - 224
} - 225
session_roots.dedup(); - 226
let mut projects: Vec<PathBuf> = Vec::new(); - 227
for root in session_roots { - 228
if let Ok(read) = std::fs::read_dir(&root) { - 229
projects.extend(read.flatten().map(|e| e.path()).filter(|p| p.is_dir())); - 230
} - 231
} - 232
projects.sort(); - 233
projects.dedup(); - 234
for dir in projects { - 235
let Ok(read) = std::fs::read_dir(&dir) else { - 236
continue; - 237
}; - 238
let Some(project_hash) = dir.file_name().and_then(|n| n.to_str()).map(String::from) else { - 239
continue; - 240
}; - 241
let mut files: Vec<PathBuf> = read.flatten().map(|e| e.path()).collect(); - 242
files.sort(); - 243
for path in files { - 244
if path.extension().and_then(|e| e.to_str()) != Some("jsonl") { - 245
continue; - 246
} - 247
let Some(session_id) = path.file_stem().and_then(|s| s.to_str()).map(String::from) - 248
else { - 249
continue; - 250
}; - 251
if excluded.contains(&session_id) { - 252
continue; - 253
} - 254
collect_ranked( - 255
&path, - 256
&session_id, - 257
&terms, - 258
&phrase, - 259
Some(project_hash.as_str()), - 260
&mut ranked, - 261
)?; - 262
} - 263
} - 264
- 265
finalize(&mut ranked, limit); - 266
Ok(ranked - 267
.into_iter() - 268
.map(|r| ProjectHit { - 269
project_hash: r.project_hash.unwrap_or_default(), - 270
hit: r.hit, - 271
}) - 272
.collect()) - 273
} - 274
- 275
fn finalize(ranked: &mut Vec<RankedHit>, limit: usize) { - 276
// Score = relevance + tiny recency preference; newest wins exact ties. - 277
let now = Utc::now(); - 278
for r in ranked.iter_mut() { - 279
let hours = (now - r.hit.ts).num_hours().max(0) as f32; - 280
r.hit.score += RECENCY_NUDGE / (1.0 + hours); - 281
} - 282
// Total order so equal-score results are deterministic regardless of - 283
// directory iteration order. - 284
ranked.sort_by(|a, b| { - 285
b.hit - 286
.score - 287
.total_cmp(&a.hit.score) - 288
.then(b.hit.ts.cmp(&a.hit.ts)) - 289
.then(a.hit.session_id.cmp(&b.hit.session_id)) - 290
.then(a.hit.entry_id.cmp(&b.hit.entry_id)) - 291
.then(a.project_hash.cmp(&b.project_hash)) - 292
}); - 293
ranked.truncate(limit); - 294
} - 295
- 296
fn collect_ranked( - 297
path: &Path, - 298
session_id: &str, - 299
terms: &[String], - 300
phrase: &str, - 301
project_hash: Option<&str>, - 302
ranked: &mut Vec<RankedHit>, - 303
) -> Result<(), SearchError> { - 304
let messages = index::ledger(path)?; - 305
for m in messages.iter() { - 306
let entities = extract_entities(&m.text); - 307
let base = score_normalized(&m.normalized, terms, phrase, &entities); - 308
if base <= 0.0 { - 309
continue; - 310
} - 311
ranked.push(RankedHit { - 312
hit: SessionHit { - 313
session_id: session_id.to_string(), - 314
entry_id: m.entry_id.clone(), - 315
ts: m.ts, - 316
role: m.role.to_string(), - 317
score: base, - 318
snippet: snippet_for(&m.text, terms), - 319
}, - 320
project_hash: project_hash.map(String::from), - 321
}); - 322
} - 323
Ok(()) - 324
} - 325
- 326
/// Lowercase alphanumeric runs of length >= 2. Cheap and language-tolerant: - 327
/// CJK text yields long runs that behave like phrases, which is fine for - 328
/// substring matching below. Public for intra-crate use (e.g. SessionLog - 329
/// proactive retrieval). - 330
pub fn tokenize_impl(text: &str) -> Vec<String> { - 331
let mut out = Vec::new(); - 332
let mut seen = HashSet::new(); - 333
let normalized = normalize_impl(text); - 334
let mut current = String::new(); - 335
for ch in normalized.chars() { - 336
if ch.is_alphanumeric() { - 337
current.push(ch); - 338
} else if !current.is_empty() { - 339
push_token(&mut out, &mut seen, ¤t); - 340
current.clear(); - 341
} - 342
} - 343
if !current.is_empty() { - 344
push_token(&mut out, &mut seen, ¤t); - 345
} - 346
out - 347
} - 348
- 349
/// Extract named entities from text — capitalized proper nouns and acronyms. - 350
/// Matches vakyartha's entity extraction approach: - 351
/// - Proper noun: `[A-Z][a-z]{2,}` optionally followed by more such words - 352
/// (1-4 words, like "New York" → "new_york") - 353
/// - Acronym: `[A-Z]{2,6}` (like "NASA", "JSON") - 354
/// - Numbers with units (dates, sizes, currency) - 355
/// - 356
/// Excludes common noise words. - 357
pub fn extract_entities(text: &str) -> Vec<String> { - 358
let mut entities: Vec<String> = Vec::new(); - 359
let mut seen: HashSet<String> = HashSet::new(); - 360
- 361
// Split into words and look for proper noun sequences - 362
let words: Vec<&str> = text.split_whitespace().collect(); - 363
let mut i = 0; - 364
while i < words.len() { - 365
let word = words[i].trim_end_matches(|c: char| !c.is_alphanumeric()); - 366
if is_proper_noun_word(word) { - 367
// Collect consecutive proper noun words (up to 4) - 368
let mut term = word.to_lowercase(); - 369
let mut j = i + 1; - 370
while j < words.len() && j < i + 4 { - 371
let next = words[j].trim_end_matches(|c: char| !c.is_alphanumeric()); - 372
if is_proper_noun_word(next) { - 373
term.push('_'); - 374
term.push_str(&next.to_lowercase()); - 375
j += 1; - 376
} else { - 377
break; - 378
} - 379
} - 380
if !is_noisy_entity(&term) && seen.insert(term.clone()) { - 381
entities.push(term); - 382
} - 383
i = j; - 384
} else if is_acronym(word) { - 385
let term = word.to_lowercase(); - 386
if !is_noisy_entity(&term) && seen.insert(term.clone()) { - 387
entities.push(term); - 388
} - 389
i += 1; - 390
} else { - 391
i += 1; - 392
} - 393
} - 394
- 395
entities - 396
} - 397
- 398
/// Check if a word is a proper noun: starts with uppercase, followed by - 399
/// 2+ lowercase letters. Matches vakyartha's `[A-Z][a-z]{2,}`. - 400
fn is_proper_noun_word(word: &str) -> bool { - 401
let chars: Vec<char> = word.chars().collect(); - 402
chars.len() >= 4 - 403
&& chars[0].is_uppercase() - 404
&& chars[1].is_lowercase() - 405
&& chars[2].is_lowercase() - 406
&& chars[3].is_lowercase() - 407
} - 408
- 409
/// Check if a word is an acronym: 2-6 consecutive uppercase letters. - 410
fn is_acronym(word: &str) -> bool { - 411
let alpha: String = word.chars().filter(|c| c.is_alphabetic()).collect(); - 412
alpha.len() >= 2 && alpha.len() <= 6 && alpha.chars().all(|c| c.is_uppercase()) - 413
} - 414
- 415
fn is_noisy_entity(term: &str) -> bool { - 416
matches!( - 417
term, - 418
"the" - 419
| "this" - 420
| "that" - 421
| "these" - 422
| "those" - 423
| "current" - 424
| "latest" - 425
| "previous" - 426
| "next" - 427
| "key" - 428
| "summary" - 429
| "based" - 430
| "choice" - 431
| "recommended" - 432
| "actions" - 433
| "analysis" - 434
| "sources" - 435
| "related" - 436
| "source" - 437
| "tools" - 438
| "recovered" - 439
| "please" - 440
| "story" - 441
| "continues" - 442
| "copyright" - 443
| "article" - 444
| "body" - 445
) - 446
} - 447
- 448
fn push_token(out: &mut Vec<String>, seen: &mut HashSet<String>, token: &str) { - 449
if token.chars().count() >= 2 && seen.insert(token.to_string()) { - 450
out.push(token.to_string()); - 451
} - 452
} - 453
- 454
/// Public for intra-crate use (e.g. SessionLog proactive retrieval). - 455
pub(crate) fn normalize_impl(text: &str) -> String { - 456
text.to_lowercase() - 457
} - 458
- 459
fn score_text(text: &str, terms: &[String], phrase: &str) -> f32 { - 460
score_normalized(&normalize_impl(text), terms, phrase, &[]) - 461
} - 462
- 463
/// BM25-style scoring with entity bonus. Matches vakyartha's - 464
/// `score_normalized` + entity bonus approach. - 465
pub fn score_normalized(hay: &str, terms: &[String], phrase: &str, entities: &[String]) -> f32 { - 466
let doc_len = hay.chars().count().max(1) as f32; - 467
let avgdl = BM25_AVGDL_APPROX; - 468
let mut score = 0.0f32; - 469
let mut matched_any = false; - 470
let entity_set: HashSet<&str> = entities.iter().map(|s| s.as_str()).collect(); - 471
- 472
for term in terms { - 473
let mut count = 0usize; - 474
let mut from = 0usize; - 475
while let Some(pos) = hay[from..].find(term.as_str()) { - 476
count += 1; - 477
let abs = from + pos + term.len(); - 478
if abs >= hay.len() { - 479
break; - 480
} - 481
from = abs; - 482
if count >= 16 { - 483
break; - 484
} - 485
} - 486
if count > 0 { - 487
matched_any = true; - 488
let idf = 1.0f32.ln_1p(); - 489
// BM25: idf * (tf * (k1+1)) / (tf + k1 * (1 - b + b * dl/avgdl)) - 490
let tf = count as f32; - 491
let length_penalty = 1.0 - BM25_B + BM25_B * doc_len / avgdl; - 492
let denom = tf + BM25_K1 * length_penalty; - 493
let mut term_score = if denom > 0.0 { - 494
idf * tf * (BM25_K1 + 1.0) / denom - 495
} else { - 496
idf * tf - 497
}; - 498
// Entity bonus: if this term matches a named entity in the text, - 499
// multiply by ENTITY_TOKEN_BONUS. - 500
if entity_set.contains(term.as_str()) { - 501
term_score *= ENTITY_TOKEN_BONUS; - 502
} - 503
score += term_score; - 504
} - 505
} - 506
if !matched_any { - 507
return 0.0; - 508
} - 509
if hay.contains(phrase) { - 510
score += PHRASE_BONUS; - 511
} - 512
score - 513
} - 514
- 515
fn snippet_for(text: &str, terms: &[String]) -> String { - 516
let hay = normalize_impl(text); - 517
let mut first_byte: Option<usize> = None; - 518
for term in terms { - 519
if let Some(pos) = hay.find(term.as_str()) { - 520
first_byte = Some(match first_byte { - 521
Some(existing) => existing.min(pos), - 522
None => pos, - 523
}); - 524
} - 525
} - 526
let char_start = match first_byte { - 527
Some(byte) => text[..byte].chars().count(), - 528
None => 0, - 529
}; - 530
let start = char_start.saturating_sub(SNIPPET_CONTEXT); - 531
let total_chars = text.chars().count(); - 532
let end = (start + SNIPPET_CHARS).min(total_chars); - 533
- 534
let prefix = if start > 0 { "…" } else { "" }; - 535
let suffix = if end < total_chars { "…" } else { "" }; - 536
format!( - 537
"{prefix}{}{suffix}", - 538
text.chars() - 539
.skip(start) - 540
.take(end - start) - 541
.collect::<String>() - 542
.trim() - 543
) - 544
} - 545
- 546
#[cfg(test)] - 547
mod tests { - 548
#![allow(clippy::unwrap_used, clippy::expect_used)] - 549
use super::*; - 550
use crate::SessionLog; - 551
use crate::types::{Entry, EntryPayload, MessageRecord, SessionHeader}; - 552
use std::io::Write as _; - 553
use vak_llm::Role; - 554
use vak_llm::types::{ContentBlock, Message}; - 555
- 556
fn header_for(id: &str, cwd: &Path) -> SessionHeader { - 557
SessionHeader { - 558
agent: None, - 559
session_id: id.to_string(), - 560
created_at: Utc::now(), - 561
cwd: cwd.to_path_buf(), - 562
parent_session_id: None, - 563
contract_id: None, - 564
work_item_id: None, - 565
conversation: None, - 566
contract: crate::types::FrozenContract { - 567
app_version: "test".into(), - 568
provider: "scripted".into(), - 569
model: "m".into(), - 570
route_ladder: Vec::new(), - 571
route_objective: String::new(), - 572
route_annotations: Vec::new(), - 573
system_prompt: String::new(), - 574
permission_mode: "workspace-write".into(), - 575
capabilities: Vec::new(), - 576
prompt_layers: Vec::new(), - 577
}, - 578
} - 579
} - 580
- 581
fn user_msg(t: &str) -> MessageRecord { - 582
MessageRecord { - 583
message: Message { - 584
role: Role::User, - 585
content: vec![ContentBlock::text(t)], - 586
}, - 587
meta: None, - 588
} - 589
} - 590
- 591
fn assistant_msg(t: &str) -> MessageRecord { - 592
MessageRecord { - 593
message: Message { - 594
role: Role::Assistant, - 595
content: vec![ContentBlock::text(t)], - 596
}, - 597
meta: None, - 598
} - 599
} - 600
- 601
fn seed(home: &Path, cwd: &Path, id: &str, msgs: &[MessageRecord]) { - 602
let path = SessionPath::new_session_file(home, cwd, id); - 603
let mut log = SessionLog::create(path, header_for(id, cwd)).unwrap(); - 604
for m in msgs { - 605
log.append_message(m.clone()).unwrap(); - 606
} - 607
} - 608
- 609
#[test] - 610
fn finds_and_ranks_relevant_sessions() { - 611
let dir = tempfile::tempdir().unwrap(); - 612
let home = dir.path().join("home"); - 613
let cwd = dir.path().to_path_buf(); - 614
- 615
seed( - 616
&home, - 617
&cwd, - 618
"11111111-deploy-talks", - 619
&[ - 620
user_msg("how does the deploy script handle rollbacks?"), - 621
assistant_msg("the deploy script pauses before rollback windows"), - 622
], - 623
); - 624
seed( - 625
&home, - 626
&cwd, - 627
"22222222-unrelated", - 628
&[user_msg("favorite pizza toppings debate")], - 629
); - 630
seed( - 631
&home, - 632
&cwd, - 633
"33333333-phrase-match", - 634
&[user_msg("run the deploy script now")], - 635
); - 636
- 637
let hits = search( - 638
&home, - 639
&cwd, - 640
"deploy script", - 641
DEFAULT_LIMIT, - 642
&Default::default(), - 643
) - 644
.unwrap(); - 645
// Message-level hits: both messages of the first ledger match. - 646
assert_eq!(hits.len(), 3); - 647
assert_eq!( - 648
hits[0].session_id, "33333333-phrase-match", - 649
"verbatim phrase outranks scattered terms" - 650
); - 651
let sessions: HashSet<&str> = hits.iter().map(|h| h.session_id.as_str()).collect(); - 652
assert_eq!(sessions.len(), 2, "pizza debate never matches"); - 653
assert!(hits[0].snippet.contains("deploy script")); - 654
assert!(hits.iter().all(|h| h.score > 0.0)); - 655
assert!( - 656
hits.iter().any(|h| h.role == "user") && hits.iter().any(|h| h.role == "assistant") - 657
); - 658
} - 659
- 660
#[test] - 661
fn excludes_current_session_and_empty_queries() { - 662
let dir = tempfile::tempdir().unwrap(); - 663
let home = dir.path().join("home"); - 664
let cwd = dir.path().to_path_buf(); - 665
seed( - 666
&home, - 667
&cwd, - 668
"aaaaaaaa-current", - 669
&[user_msg("kubernetes ingress quirks")], - 670
); - 671
seed( - 672
&home, - 673
&cwd, - 674
"bbbbbbbb-other", - 675
&[user_msg("kubernetes ingress quirks from last week")], - 676
); - 677
- 678
let hits = search( - 679
&home, - 680
&cwd, - 681
"kubernetes ingress", - 682
DEFAULT_LIMIT, - 683
&std::collections::HashSet::from(["aaaaaaaa-current".to_string()]), - 684
) - 685
.unwrap(); - 686
assert_eq!(hits.len(), 1); - 687
assert_eq!(hits[0].session_id, "bbbbbbbb-other"); - 688
- 689
assert!( - 690
search(&home, &cwd, " ", DEFAULT_LIMIT, &Default::default()) - 691
.unwrap() - 692
.is_empty() - 693
); - 694
assert!( - 695
search( - 696
&home, - 697
&cwd, - 698
"zzzqqq nonexistent", - 699
DEFAULT_LIMIT, - 700
&Default::default() - 701
) - 702
.unwrap() - 703
.is_empty() - 704
); - 705
} - 706
- 707
#[test] - 708
fn memory_extras_outrank_equal_transcript_hits() { - 709
let dir = tempfile::tempdir().unwrap(); - 710
let home = dir.path().join("home"); - 711
let cwd = dir.path().to_path_buf(); - 712
seed( - 713
&home, - 714
&cwd, - 715
"dddddddd-transcript", - 716
&[user_msg( - 717
"rollback windows are configured in the deploy pipeline", - 718
)], - 719
); - 720
let extras = vec![ExternalDoc { - 721
id: "deploy".into(), - 722
text: "decision: rollback windows pause the deploy pipeline".into(), - 723
ts: None, - 724
role: None, - 725
}]; - 726
let hits = search_extended( - 727
&home, - 728
&cwd, - 729
"deploy rollback", - 730
5, - 731
&Default::default(), - 732
&extras, - 733
) - 734
.unwrap(); - 735
assert!(hits.len() >= 2); - 736
assert_eq!(hits[0].role, "memory"); - 737
assert_eq!(hits[0].session_id, "deploy"); - 738
assert!(hits.iter().skip(1).all(|h| h.role != "memory")); - 739
} - 740
- 741
#[test] - 742
fn entity_extraction_finds_camel_case_and_title_case() { - 743
let text = "The Kubernetes cluster runs the Docker container for PostgreSQL."; - 744
let entities = extract_entities(text); - 745
assert!(entities.contains(&"kubernetes".to_string())); - 746
assert!(entities.contains(&"docker".to_string())); - 747
assert!(entities.contains(&"postgresql".to_string())); - 748
} - 749
- 750
#[test] - 751
fn entity_bonus_raises_score_for_named_entities() { - 752
// Same phrase so PHRASE_BONUS cancels in the ratio. - 753
let terms = vec!["kubernetes".to_string()]; - 754
let phrase = "kubernetes"; - 755
let entities = vec!["kubernetes".to_string()]; - 756
let score_with_bonus = - 757
score_normalized("the kubernetes cluster", &terms, phrase, &entities); - 758
let score_without = score_normalized("the kubernetes cluster", &terms, phrase, &[]); - 759
assert!( - 760
score_with_bonus > score_without, - 761
"entity bonus should increase score" - 762
); - 763
// The entity bonus multiplies the term score by ENTITY_TOKEN_BONUS. - 764
// PHRASE_BONUS is added to both, so the ratio is diluted but - 765
// score_with_bonus - PHRASE > (score_without - PHRASE) * ENTITY_BONUS. - 766
assert!( - 767
(score_with_bonus - PHRASE_BONUS) > (score_without - PHRASE_BONUS) * 0.99, - 768
"entity bonus should multiply the term score" - 769
); - 770
} - 771
- 772
#[test] - 773
fn snippets_are_bounded_and_centered() { - 774
let dir = tempfile::tempdir().unwrap(); - 775
let home = dir.path().join("home"); - 776
let cwd = dir.path().to_path_buf(); - 777
let filler = "lorem ipsum ".repeat(500); - 778
seed( - 779
&home, - 780
&cwd, - 781
"cccccccc-long", - 782
&[assistant_msg(&format!( - 783
"{filler}NEEDLE-HAYSTACK trailing words" - 784
))], - 785
); - 786
let hits = search(&home, &cwd, "needle", 5, &Default::default()).unwrap(); - 787
assert_eq!(hits.len(), 1); - 788
let snip = &hits[0].snippet; - 789
assert!(snip.starts_with('…'), "long text gets a leading ellipsis"); - 790
assert!(snip.contains("NEEDLE")); - 791
assert!(snip.chars().count() <= SNIPPET_CHARS + 2); - 792
} - 793
- 794
#[test] - 795
fn index_detects_appended_lines_without_restart() { - 796
let dir = tempfile::tempdir().unwrap(); - 797
let home = dir.path().join("home"); - 798
let cwd = dir.path().to_path_buf(); - 799
seed( - 800
&home, - 801
&cwd, - 802
"eeeeeeee-append", - 803
&[user_msg("alpha bravo charlie baseline")], - 804
); - 805
- 806
assert!( - 807
search(&home, &cwd, "foxtrot", 5, &Default::default()) - 808
.unwrap() - 809
.is_empty() - 810
); - 811
- 812
let path = SessionPath::new_session_file(&home, &cwd, "eeeeeeee-append"); - 813
let mut log = SessionLog::open(path).unwrap(); - 814
log.append_message(user_msg("delta echo foxtrot followup")) - 815
.unwrap(); - 816
drop(log); - 817
- 818
let hits = search(&home, &cwd, "foxtrot", 5, &Default::default()).unwrap(); - 819
assert_eq!(hits.len(), 1, "append must invalidate the cached ledger"); - 820
assert_eq!(hits[0].session_id, "eeeeeeee-append"); - 821
assert_eq!(hits[0].role, "user"); - 822
- 823
// The warmed cache answers again without re-reading the file. - 824
let again = search(&home, &cwd, "foxtrot", 5, &Default::default()).unwrap(); - 825
assert_eq!(hits, again); - 826
} - 827
- 828
fn write_ledger(path: &Path, msgs: &[MessageRecord]) { - 829
std::fs::create_dir_all(path.parent().unwrap()).unwrap(); - 830
let file = std::fs::File::create(path).unwrap(); - 831
let mut w = std::io::BufWriter::new(file); - 832
for m in msgs { - 833
let entry = Entry::new(None, EntryPayload::Message(m.clone())); - 834
serde_json::to_writer(&mut w, &entry).unwrap(); - 835
w.write_all(b"\n").unwrap(); - 836
} - 837
w.flush().unwrap(); - 838
} - 839
- 840
#[test] - 841
fn warm_index_beats_cold_scan_on_large_store() { - 842
let dir = tempfile::tempdir().unwrap(); - 843
let home = dir.path().join("home"); - 844
let cwd = dir.path().to_path_buf(); - 845
- 846
const LEDGERS: usize = 10; - 847
const LINES: usize = 1000; - 848
for k in 0..LEDGERS { - 849
let mut msgs = Vec::with_capacity(LINES); - 850
for i in 0..LINES { - 851
msgs.push(user_msg(&format!( - 852
"note {i} about refactor planning and review cadence {k}" - 853
))); - 854
} - 855
if k == 0 { - 856
msgs[7] = user_msg("quantum tuning notes for ledger zero"); - 857
msgs[9] = user_msg("xylophone quantum alignment strategy"); - 858
} else if k < 3 { - 859
msgs[7] = user_msg(&format!("quantum tuning notes for ledger {k}")); - 860
msgs[9] = user_msg(&format!("quantum alignment strategy for ledger {k}")); - 861
} - 862
write_ledger( - 863
&SessionPath::new_session_file(&home, &cwd, &format!("{k:08}-bulk")), - 864
&msgs, - 865
); - 866
} - 867
- 868
let t0 = std::time::Instant::now(); - 869
let cold_hits = search(&home, &cwd, "xylophone quantum", 20, &Default::default()).unwrap(); - 870
let cold = t0.elapsed(); - 871
- 872
let mut warm_min = std::time::Duration::MAX; - 873
let mut last_hits = Vec::new(); - 874
for _ in 0..3 { - 875
let t = std::time::Instant::now(); - 876
last_hits = search(&home, &cwd, "xylophone quantum", 20, &Default::default()).unwrap(); - 877
warm_min = warm_min.min(t.elapsed()); - 878
} - 879
- 880
assert_eq!( - 881
cold_hits.len(), - 882
6, - 883
"two planted messages in each of three ledgers" - 884
); - 885
assert_eq!( - 886
cold_hits[0].session_id, "00000000-bulk", - 887
"the only full-phrase match outranks scattered term matches" - 888
); - 889
assert!(cold_hits[0].snippet.contains("xylophone")); - 890
assert_eq!( - 891
cold_hits, last_hits, - 892
"same input must yield identical output" - 893
); - 894
assert!( - 895
warm_min < cold, - 896
"warm query ({warm_min:?}) should beat cold scan ({cold:?})" - 897
); - 898
assert!( - 899
warm_min < std::time::Duration::from_millis(500), - 900
"warm query must stay fast even on slow CI ({warm_min:?})" - 901
); - 902
} - 903
- 904
#[test] - 905
fn search_all_spans_project_dirs_and_excludes_everywhere() { - 906
let dir = tempfile::tempdir().unwrap(); - 907
let home = dir.path().join("home"); - 908
let cwd_a = dir.path().join("proj-a"); - 909
let cwd_b = dir.path().join("proj-b"); - 910
let cwd_c = dir.path().join("proj-c"); - 911
- 912
seed( - 913
&home, - 914
&cwd_a, - 915
"aa-keeper", - 916
&[user_msg("needle in project a")], - 917
); - 918
seed(&home, &cwd_a, "zz-excluded", &[user_msg("needle hidden a")]); - 919
seed( - 920
&home, - 921
&cwd_b, - 922
"bb-keeper", - 923
&[user_msg("needle in project b")], - 924
); - 925
seed(&home, &cwd_b, "zz-excluded", &[user_msg("needle hidden b")]); - 926
seed( - 927
&home, - 928
&cwd_c, - 929
"cc-keeper", - 930
&[user_msg("needle in project c")], - 931
); - 932
- 933
let hash = |cwd: &Path| { - 934
SessionPath::sessions_dir(&home, cwd) - 935
.file_name() - 936
.unwrap() - 937
.to_string_lossy() - 938
.to_string() - 939
}; - 940
let (ha, hb, hc) = (hash(&cwd_a), hash(&cwd_b), hash(&cwd_c)); - 941
- 942
let first = search_all( - 943
&home, - 944
"needle", - 945
20, - 946
&std::collections::HashSet::from(["zz-excluded".to_string()]), - 947
) - 948
.unwrap(); - 949
let second = search_all( - 950
&home, - 951
"needle", - 952
20, - 953
&std::collections::HashSet::from(["zz-excluded".to_string()]), - 954
) - 955
.unwrap(); - 956
- 957
assert_eq!(first.len(), 3, "one keeper hit per project dir"); - 958
assert_eq!( - 959
first, second, - 960
"cross-project ordering must be deterministic" - 961
); - 962
assert!( - 963
first.iter().all(|h| h.hit.session_id != "zz-excluded"), - 964
"current-session exclusion applies in every hash dir" - 965
); - 966
let sessions: HashSet<&str> = first.iter().map(|h| h.hit.session_id.as_str()).collect(); - 967
assert_eq!( - 968
sessions, - 969
HashSet::from(["aa-keeper", "bb-keeper", "cc-keeper"]) - 970
); - 971
let hashes: HashSet<&str> = first.iter().map(|h| h.project_hash.as_str()).collect(); - 972
assert_eq!( - 973
hashes, - 974
HashSet::from([ha.as_str(), hb.as_str(), hc.as_str()]) - 975
); - 976
for h in &first { - 977
let source = match h.hit.session_id.as_str() { - 978
"aa-keeper" => &ha, - 979
"bb-keeper" => &hb, - 980
_ => &hc, - 981
}; - 982
assert_eq!(h.project_hash, *source); - 983
} - 984
} - 985
} - 986
Indexing the workspace…
Vakyartha documentation is discovering safe artifacts, anchors, and source references.