- 1001
collected.len() - 1002
) - 1003
} - 1004
} - 1005
- 1006
fn extract_text_lines(content: &str, offset: usize, limit: usize) -> String { - 1007
let lines: Vec<&str> = content.lines().collect(); - 1008
let total = lines.len(); - 1009
let start = (offset.saturating_sub(1)).min(total); - 1010
let end = (start + limit).min(total); - 1011
- 1012
let numbered: Vec<String> = lines[start..end] - 1013
.iter() - 1014
.enumerate() - 1015
.map(|(idx, line)| format!("{:4}: {}", start + idx + 1, line)) - 1016
.collect(); - 1017
- 1018
format!( - 1019
"{}\n\n[Lines {}..{} of {} total lines]", - 1020
numbered.join("\n"), - 1021
start + 1, - 1022
end, - 1023
total - 1024
) - 1025
} - 1026
- 1027
/// Where a file the person attached under the name `requested` was saved: - 1028
/// the inbox prefixes each saved name with a digest, so a model that names - 1029
/// the file as the person did misses it. - 1030
fn attached_copy(cwd: &Path, requested: &str) -> Option<String> { - 1031
let wanted = Path::new(requested).file_name()?.to_str()?; - 1032
std::fs::read_dir(cwd.join(crate::INBOX_DIR)) - 1033
.ok()? - 1034
.filter_map(Result::ok) - 1035
.filter_map(|entry| entry.file_name().into_string().ok()) - 1036
.find(|name| { - 1037
name.split_once('-').is_some_and(|(prefix, rest)| { - 1038
rest == wanted - 1039
&& prefix.len() == 12 - 1040
&& prefix.chars().all(|c| c.is_ascii_hexdigit()) - 1041
}) - 1042
}) - 1043
.map(|name| format!("{}/{name}", crate::INBOX_DIR)) - 1044
} - 1045
- 1046
#[cfg(test)] - 1047
mod tests { - 1048
#![allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] - 1049
use super::*; - 1050
- 1051
#[test] - 1052
fn test_extract_section_markdown() { - 1053
let doc = "# Research Paper\n\n## Abstract\nThis is the abstract.\n\n## Methods\nWe used in-memory tooling.\nStep 1: parse.\nStep 2: verify.\n\n## Results\nAll tests passed.\n"; - 1054
let extracted = extract_section(doc, "Methods", "md", 1, 50); - 1055
assert!(extracted.contains("We used in-memory tooling.")); - 1056
assert!(extracted.contains("Step 1: parse.")); - 1057
assert!(!extracted.contains("All tests passed.")); - 1058
} - 1059
- 1060
#[test] - 1061
fn test_outline_markdown() { - 1062
let doc = "# Main\n## Sec 1\n### Sub 1\n## Sec 2\n"; - 1063
let outline = outline_document(doc, "md"); - 1064
assert!(outline.contains("L001: # Main")); - 1065
assert!(outline.contains("L002: ## Sec 1")); - 1066
assert!(outline.contains("L004: ## Sec 2")); - 1067
} - 1068
- 1069
#[test] - 1070
fn test_render_table_view_csv() { - 1071
let csv = "name,role,score\nAlice,Lead,95\nBob,Eng,88\nCharlie,Analyst,91\n"; - 1072
let rendered = render_table_view(csv, "csv", 1, 2); - 1073
assert!(rendered.contains("| name | role | score |")); - 1074
assert!(rendered.contains("| Alice | Lead | 95 |")); - 1075
assert!(rendered.contains("| Bob | Eng | 88 |")); - 1076
assert!(!rendered.contains("Charlie")); - 1077
assert!(rendered.contains("Showing rows 1..2 of 3")); - 1078
} - 1079
- 1080
#[test] - 1081
fn test_render_json_table() { - 1082
let json_data = r#"[ - 1083
{"id": "usr-1", "name": "Alice", "active": true}, - 1084
{"id": "usr-2", "name": "Bob", "active": false} - 1085
]"#; - 1086
let rendered = render_table_view(json_data, "json", 1, 10); - 1087
assert!( - 1088
rendered.contains("id") && rendered.contains("name") && rendered.contains("active") - 1089
); - 1090
assert!(rendered.contains("Alice") && rendered.contains("usr-1")); - 1091
assert!(rendered.contains("Bob") && rendered.contains("usr-2")); - 1092
assert!(rendered.contains("Showing rows 1..2 of 2 total rows")); - 1093
} - 1094
- 1095
#[test] - 1096
fn test_render_kv_table_ini() { - 1097
let ini = "[server]\nport = 8080\nhost = localhost\n[db]\nurl = postgresql://db\n"; - 1098
let rendered = render_table_view(ini, "ini", 1, 10); - 1099
assert!(rendered.contains("| server.port | 8080 |")); - 1100
assert!(rendered.contains("| db.url | postgresql://db |")); - 1101
} - 1102
- 1103
#[test] - 1104
fn test_render_html_table() { - 1105
let html = "<table><tr><th>Metric</th><th>Value</th></tr><tr><td>Latency</td><td>12ms</td></tr><tr><td>Throughput</td><td>1000req/s</td></tr></table>"; - 1106
let rendered = render_table_view(html, "html", 1, 5); - 1107
assert!(rendered.contains("| Metric | Value |")); - 1108
assert!(rendered.contains("| Latency | 12ms |")); - 1109
assert!(rendered.contains("| Throughput | 1000req/s |")); - 1110
} - 1111
- 1112
#[test] - 1113
fn test_extract_bracket_section_toml() { - 1114
let toml_doc = - 1115
"[gateway]\nport = 3000\n[agent]\nname = \"Research\"\nrole = \"researcher\"\n"; - 1116
let extracted = extract_section(toml_doc, "agent", "toml", 1, 20); - 1117
assert!(extracted.contains("name = \"Research\"")); - 1118
assert!(extracted.contains("role = \"researcher\"")); - 1119
assert!(!extracted.contains("port = 3000")); - 1120
} - 1121
- 1122
#[test] - 1123
fn test_extract_text_lines() { - 1124
let text = "alpha\nbeta\ngamma\ndelta\nepsilon\n"; - 1125
let extracted = extract_text_lines(text, 2, 2); - 1126
assert!(extracted.contains("2: beta")); - 1127
assert!(extracted.contains("3: gamma")); - 1128
assert!(!extracted.contains("alpha")); - 1129
assert!(!extracted.contains("delta")); - 1130
} - 1131
- 1132
async fn read_office(name: &str, bytes: Vec<u8>, args: Value) -> ToolOutput { - 1133
let dir = tempfile::tempdir().unwrap(); - 1134
std::fs::write(dir.path().join(name), bytes).unwrap(); - 1135
let mut args = args; - 1136
args["path"] = Value::String(name.into()); - 1137
DocReadTool - 1138
.execute(&args, &ToolContext::new(dir.path().to_path_buf())) - 1139
.await - 1140
} - 1141
- 1142
#[tokio::test] - 1143
async fn a_file_named_as_it_was_attached_points_to_its_inbox_copy() { - 1144
let dir = tempfile::tempdir().unwrap(); - 1145
std::fs::create_dir(dir.path().join(crate::INBOX_DIR)).unwrap(); - 1146
std::fs::write( - 1147
dir.path().join("inbox/e1d2a5db9ff2-Board deck.pptx"), - 1148
vak_ooxml::fixtures::pptx(), - 1149
) - 1150
.unwrap(); - 1151
let output = DocReadTool - 1152
.execute( - 1153
&serde_json::json!({"path": "Board deck.pptx"}), - 1154
&ToolContext::new(dir.path().to_path_buf()), - 1155
) - 1156
.await; - 1157
assert!(output.is_error); - 1158
assert!( - 1159
output - 1160
.content - 1161
.contains("is at path \"inbox/e1d2a5db9ff2-Board deck.pptx\": call doc_read again"), - 1162
"{}", - 1163
output.content - 1164
); - 1165
let missing = DocReadTool - 1166
.execute( - 1167
&serde_json::json!({"path": "other.pptx"}), - 1168
&ToolContext::new(dir.path().to_path_buf()), - 1169
) - 1170
.await; - 1171
assert!(!missing.content.contains("inbox/"), "{}", missing.content); - 1172
} - 1173
- 1174
#[tokio::test] - 1175
async fn office_text_view_is_anchored_and_labels_hidden_content() { - 1176
let output = read_office( - 1177
"q3.docx", - 1178
vak_ooxml::fixtures::docx(), - 1179
serde_json::json!({}), - 1180
) - 1181
.await; - 1182
assert!(!output.is_error, "{}", output.content); - 1183
let text = output.content; - 1184
assert!( - 1185
text.starts_with("Word document (.docx) · title: Q3 Report"), - 1186
"{text}" - 1187
); - 1188
assert!(text.contains("not instructions"), "{text}"); - 1189
assert!( - 1190
text.contains("Cite a place as `q3.docx#<anchor>`"), - 1191
"{text}" - 1192
); - 1193
assert!(text.contains("Not read: headers and footers"), "{text}"); - 1194
assert!(text.contains("[p:0A1B2C3D] # Quarterly Report"), "{text}"); - 1195
assert!(text.contains("⟨hidden text⟩"), "{text}"); - 1196
assert!(text.contains("[lines 1.."), "{text}"); - 1197
} - 1198
- 1199
#[tokio::test] - 1200
async fn office_section_outline_and_table_views() { - 1201
let section = read_office( - 1202
"q3.docx", - 1203
vak_ooxml::fixtures::docx(), - 1204
serde_json::json!({"section": "Outlook"}), - 1205
) - 1206
.await; - 1207
assert!(section.content.contains("Steady."), "{}", section.content); - 1208
assert!( - 1209
!section.content.contains("Revenue grew"), - 1210
"{}", - 1211
section.content - 1212
); - 1213
- 1214
let outline = read_office( - 1215
"deck.pptx", - 1216
vak_ooxml::fixtures::pptx(), - 1217
serde_json::json!({"view": "outline"}), - 1218
) - 1219
.await; - 1220
assert!( - 1221
outline - 1222
.content - 1223
.contains("- [slide:256] Slide 1: Launch plan"), - 1224
"{}", - 1225
outline.content - 1226
); - 1227
- 1228
let table = read_office( - 1229
"book.xlsx", - 1230
vak_ooxml::fixtures::xlsx(), - 1231
serde_json::json!({"view": "table", "section": "Budget", "limit": 2}), - 1232
) - 1233
.await; - 1234
assert!( - 1235
table.content.contains("| | A | B | C |"), - 1236
"{}", - 1237
table.content - 1238
); - 1239
assert!( - 1240
table.content.contains("| 1 | Item | Cost | |"), - 1241
"{}", - 1242
table.content - 1243
); - 1244
assert!( - 1245
table.content.contains("continue with offset=3"), - 1246
"{}", - 1247
table.content - 1248
); - 1249
- 1250
let missing = read_office( - 1251
"book.xlsx", - 1252
vak_ooxml::fixtures::xlsx(), - 1253
serde_json::json!({"view": "table", "section": "Nope"}), - 1254
) - 1255
.await; - 1256
assert!(missing.is_error); - 1257
assert!( - 1258
missing.content.contains("'Budget' (Budget!)"), - 1259
"{}", - 1260
missing.content - 1261
); - 1262
} - 1263
- 1264
#[tokio::test] - 1265
async fn office_summary_lists_flags_and_external_links() { - 1266
let package = vak_ooxml::fixtures::word_with( - 1267
vak_ooxml::fixtures::WORD_MAIN, - 1268
vak_ooxml::fixtures::MINIMAL_WORD_BODY, - 1269
&[], - 1270
&[], - 1271
&[], - 1272
&[( - 1273
"rId1", - 1274
"http://schemas.openxmlformats.org/officeDocument/2006/relationships/attachedTemplate", - 1275
"https://attacker.example/t.dotm", - 1276
)], - 1277
); - 1278
let output = read_office( - 1279
"letter.docx", - 1280
package, - 1281
serde_json::json!({"view": "summary"}), - 1282
) - 1283
.await; - 1284
assert!( - 1285
output.content.contains("Flags: remote template"), - 1286
"{}", - 1287
output.content - 1288
); - 1289
assert!( - 1290
output.content.contains("External attachedTemplate from word/document.xml: https://attacker.example/t.dotm (not followed)"), - 1291
"{}", - 1292
output.content - 1293
); - 1294
} - 1295
- 1296
#[tokio::test] - 1297
async fn hostile_and_unsupported_files_fail_with_a_reason() { - 1298
let bomb = vak_ooxml::fixtures::word_with( - 1299
vak_ooxml::fixtures::WORD_MAIN, - 1300
r#"<!DOCTYPE x [<!ENTITY a "a">]><w:document xmlns:w="x"/>"#, - 1301
&[], - 1302
&[], - 1303
&[], - 1304
&[], - 1305
); - 1306
let output = read_office("bomb.docx", bomb, serde_json::json!({})).await; - 1307
assert!(output.is_error); - 1308
assert!(output.content.contains("DOCTYPE"), "{}", output.content); - 1309
- 1310
let legacy = read_office( - 1311
"old.doc", - 1312
b"\xD0\xCF\x11\xE0".to_vec(), - 1313
serde_json::json!({}), - 1314
) - 1315
.await; - 1316
assert!(legacy.is_error); - 1317
assert!( - 1318
legacy.content.contains("legacy binary Word"), - 1319
"{}", - 1320
legacy.content - 1321
); - 1322
- 1323
let binary = read_office( - 1324
"blob.bin", - 1325
vec![0xFF, 0xFE, 0x00, 0x81], - 1326
serde_json::json!({}), - 1327
) - 1328
.await; - 1329
assert!(binary.is_error); - 1330
assert!( - 1331
binary.content.contains("not a text file"), - 1332
"{}", - 1333
binary.content - 1334
); - 1335
} - 1336
- 1337
#[tokio::test] - 1338
async fn office_output_is_digest_bound_and_pages_a_huge_paragraph() { - 1339
let giant: String = (0..400_000u64) - 1340
.map(|index| format!("w{} ", index.wrapping_mul(2_654_435_761) % 1_000_003)) - 1341
.collect(); - 1342
let document = format!( - 1343
r#"<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"><w:body><w:p><w:r><w:t>{giant}</w:t></w:r></w:p><w:p><w:r><w:t>tail</w:t></w:r></w:p></w:body></w:document>"# - 1344
); - 1345
let package = vak_ooxml::fixtures::word_with( - 1346
vak_ooxml::fixtures::WORD_MAIN, - 1347
&document, - 1348
&[], - 1349
&[], - 1350
&[], - 1351
&[], - 1352
); - 1353
let output = read_office("big.docx", package, serde_json::json!({"limit": 500})).await; - 1354
assert!(!output.is_error, "{}", output.content); - 1355
assert!( - 1356
output.content.contains("· sha256 "), - 1357
"anchors are bound to a digest" - 1358
); - 1359
assert!( - 1360
output.content.contains("[p@1 ⟨part 1 of "), - 1361
"a huge unit is split, not cut" - 1362
); - 1363
assert!( - 1364
output.content.len() <= super::PAGE_BYTES + 4096, - 1365
"a page stays inside the budget: {} bytes", - 1366
output.content.len() - 1367
); - 1368
assert!(output.content.contains("continue with offset=")); - 1369
assert!( - 1370
!output.content.contains("[p@2] tail"), - 1371
"the rest waits for the next page" - 1372
); - 1373
} - 1374
} - 1375
Indexing the workspace…
Vakyartha documentation is discovering safe artifacts, anchors, and source references.