- 639
/// "Make sure" and "ensure" demand machine-checkable proof only when the - 640
/// thing to be sure of is checkable: "make sure the tests pass" is a - 641
/// predicate, "make sure it rhymes" is a style instruction, and reading the - 642
/// latter as `verified` made the stop gate demand a shell receipt for a poem. - 643
const ASSURANCE_PHRASES: &[(&str, f64)] = &[("make sure", 0.8), ("ensure", 0.7)]; - 644
const CHECKABLE_WORDS: &[&str] = &[ - 645
"test", - 646
"tests", - 647
"testing", - 648
"pass", - 649
"passes", - 650
"passing", - 651
"build", - 652
"builds", - 653
"compile", - 654
"compiles", - 655
"compiling", - 656
"lint", - 657
"ci", - 658
"green", - 659
"works", - 660
"working", - 661
"run", - 662
"runs", - 663
]; - 664
- 665
/// `live` is two words. As an adjective or adverb ("a live score", "is it - 666
/// live") it means *current*, and is a recency word like "right now": a - 667
/// request for a fact with it asks for a value observed this turn. As a verb - 668
/// ("we live in the city", "my kids live with me") it means *reside*, and - 669
/// says nothing about time: measured live, "we live in the city" in a - 670
/// weekend-planning request set `live-data`, the freshness check refused the - 671
/// plan card, and the person got no plan at all. "Go live" — launching — - 672
/// is a stakes phrase, not a recency word ([`STAKES_WORDS`]). - 673
/// - 674
/// The verb reading is recognised from its neighbours, never from a topic: - 675
/// a subject or auxiliary right before it, or — unless a copula or "go" - 676
/// right before it makes it the adjective — a residence preposition right - 677
/// after it. Only the bare form `live` is ever read in the current sense; - 678
/// `lives`, `lived` and `living` are always the verb. - 679
const LIVE_WORD: &str = "live"; - 680
/// A word before `live` that makes it the verb "reside". - 681
const RESIDE_SUBJECTS: &[&str] = &[ - 682
"i", "we", "you", "they", "he", "she", "who", "people", "both", "all", "to", "can", "could", - 683
"would", "will", "might", "should", "must", "not", "never", "t", "d", "ll", - 684
]; - 685
/// A word after `live` that makes it "reside", unless [`LIVE_COPULAS`] - 686
/// precedes it. - 687
const RESIDE_PREPOSITIONS: &[&str] = &[ - 688
"in", "near", "with", "nearby", "abroad", "alone", "together", "close", "downtown", "outside", - 689
]; - 690
/// A word before `live` that keeps it the adjective ("is live in prod", - 691
/// "go live in an hour"). - 692
const LIVE_COPULAS: &[&str] = &[ - 693
"is", "are", "was", "were", "be", "been", "being", "s", "re", "go", "goes", "going", "went", - 694
"gone", "now", - 695
]; - 696
- 697
/// Whether some occurrence of `live` in `tokens` means *current*, not - 698
/// *reside*. - 699
fn live_means_current(tokens: &[String]) -> bool { - 700
tokens.iter().enumerate().any(|(i, token)| { - 701
if token != LIVE_WORD { - 702
return false; - 703
} - 704
let prev = i.checked_sub(1).map(|p| tokens[p].as_str()); - 705
let next = tokens.get(i + 1).map(String::as_str); - 706
let copula = prev.is_some_and(|w| LIVE_COPULAS.contains(&w)); - 707
let subject = prev.is_some_and(|w| RESIDE_SUBJECTS.contains(&w)); - 708
let preposition = next.is_some_and(|w| RESIDE_PREPOSITIONS.contains(&w)); - 709
copula || !(subject || preposition) - 710
}) - 711
} - 712
- 713
/// Temporal deixis: the request asks for a value as it stands *now*, which - 714
/// no model knows from training and which must therefore be observed on this - 715
/// turn (`live-data` domain, docs/design/68-context-engine.md §7). References - 716
/// to time, never to a topic. Whether it applies also depends on the act — - 717
/// only a request for a fact asks for a current value — which the resolver - 718
/// decides once the act is known. - 719
const RECENCY_PHRASES: &[(&str, f64)] = &[ - 720
("right now", 1.0), - 721
("currently", 0.9), - 722
("current", 0.8), - 723
("as of today", 1.0), - 724
("as of now", 1.0), - 725
("today", 0.6), - 726
("tonight", 0.7), - 727
("this morning", 0.8), - 728
("this week", 0.5), - 729
("latest", 0.7), - 730
("real time", 0.8), - 731
("at the moment", 0.9), - 732
("up to date", 0.7), - 733
// Only in the sense of *current* ([`live_means_current`]). - 734
(LIVE_WORD, 0.5), - 735
]; - 736
- 737
/// Nouns that make a recency word local rather than live: "the current - 738
/// directory", "the latest changes", "live reload". The workspace, the - 739
/// conversation and the running session are observed with local tools and - 740
/// hold no value a model could carry over stale from training. - 741
const LOCAL_NOUNS: &[&str] = &[ - 742
"directory", - 743
"dir", - 744
"folder", - 745
"file", - 746
"files", - 747
"path", - 748
"branch", - 749
"commit", - 750
"commits", - 751
"diff", - 752
"changes", - 753
"change", - 754
"working", - 755
"workspace", - 756
"project", - 757
"repo", - 758
"repository", - 759
"codebase", - 760
"code", - 761
"implementation", - 762
"function", - 763
"method", - 764
"class", - 765
"module", - 766
"line", - 767
"lines", - 768
"cursor", - 769
"selection", - 770
"tab", - 771
"window", - 772
"page", - 773
"screen", - 774
"session", - 775
"conversation", - 776
"chat", - 777
"thread", - 778
"context", - 779
"task", - 780
"plan", - 781
"step", - 782
"turn", - 783
"user", - 784
"config", - 785
"configuration", - 786
"settings", - 787
"setup", - 788
"build", - 789
"test", - 790
"tests", - 791
"reload", - 792
"preview", - 793
"server", - 794
"share", - 795
"coding", - 796
"edit", - 797
"editing", - 798
"demo", - 799
"mode", - 800
"state", - 801
// The agent's own state is read with its own tools, not retrieved from - 802
// the world: "what are you holding right now", "your current plan". - 803
"you", - 804
"your", - 805
"yours", - 806
"yourself", - 807
"commitment", - 808
"commitments", - 809
"tasks", - 810
"plans", - 811
"inbox", - 812
"memory", - 813
"notes", - 814
"reminders", - 815
]; - 816
- 817
/// The runtime tells the model the time and date on every turn, so asking - 818
/// for them needs no retrieval. - 819
const TIME_WORDS: &[&str] = &[ - 820
"time", "date", "day", "weekday", "clock", "timezone", "hour", "year", "month", - 821
]; - 822
- 823
/// Phrases implying the work outlives this turn. Sequencing words ("then", - 824
/// "after that") are not here: they separate the parts of one request, which - 825
/// is what strands are for, not a claim that the work spans sessions. - 826
const HORIZON_PHRASES: &[(&str, Horizon, f64)] = &[ - 827
("every day", Horizon::Durable, 1.0), - 828
("every night", Horizon::Durable, 1.0), - 829
("every morning", Horizon::Durable, 1.0), - 830
("every evening", Horizon::Durable, 1.0), - 831
("every month", Horizon::Durable, 1.0), - 832
("every week", Horizon::Durable, 1.0), - 833
("every hour", Horizon::Durable, 1.0), - 834
("each day", Horizon::Durable, 1.0), - 835
("each week", Horizon::Durable, 1.0), - 836
("each month", Horizon::Durable, 1.0), - 837
("every monday", Horizon::Durable, 1.0), - 838
("every tuesday", Horizon::Durable, 1.0), - 839
("every wednesday", Horizon::Durable, 1.0), - 840
("every thursday", Horizon::Durable, 1.0), - 841
("every friday", Horizon::Durable, 1.0), - 842
("every saturday", Horizon::Durable, 1.0), - 843
("every sunday", Horizon::Durable, 1.0), - 844
("every weekday", Horizon::Durable, 1.0), - 845
("every weekend", Horizon::Durable, 1.0), - 846
("nightly", Horizon::Durable, 0.9), - 847
("monthly", Horizon::Durable, 0.9), - 848
("daily", Horizon::Durable, 0.9), - 849
("weekly", Horizon::Durable, 0.9), - 850
("hourly", Horizon::Durable, 0.9), - 851
("continuously", Horizon::Durable, 0.9), - 852
("whenever", Horizon::Durable, 0.7), - 853
("keep watching", Horizon::Durable, 1.0), - 854
("keep an eye", Horizon::Durable, 0.9), - 855
("from now on", Horizon::Durable, 0.9), - 856
("ongoing", Horizon::Durable, 0.8), - 857
("until", Horizon::Durable, 0.4), - 858
("over the next", Horizon::Durable, 0.7), - 859
]; - 860
- 861
/// Recurrence words that are adjectives after a determiner ("the nightly - 862
/// job") and adverbs otherwise ("check it nightly"). Only the adverb votes. - 863
const RECURRENCE_ADJECTIVES: &[&str] = &["nightly", "daily", "weekly", "monthly", "hourly"]; - 864
const DETERMINERS: &[&str] = &[ - 865
"the", "a", "an", "this", "that", "our", "my", "your", "its", "their", "each", "of", - 866
]; - 867
- 868
/// Deictic markers: the request points at something it does not contain. - 869
pub(crate) const DEICTIC_WORDS: &[&str] = &[ - 870
"this", "that", "it", "these", "those", "here", "there", "again", "same", - 871
]; - 872
- 873
/// Pronouns that leave a request with no object of its own when they end - 874
/// it: "fix it", "deploy that". "Write a function that parses dates" uses - 875
/// `that` as a relative pronoun and points at nothing. - 876
const BARE_PRONOUNS: &[&str] = &["it", "this", "that", "these", "those"]; - 877
- 878
/// Words before a verb that still leave it heading its clause. - 879
const HEAD_PREDECESSORS: &[&str] = &[ - 880
"and", "then", "also", "plus", "next", "after", "or", "now", "please", "first", "finally", "so", - 881
]; - 882
- 883
const IMPERATIVE_BONUS: f64 = 1.6; - 884
- 885
// ------------------------------------------------------------ scoring --- - 886
- 887
/// An axis value that can name itself, so vote tallies can break ties - 888
/// deterministically without depending on the enum's declaration order. - 889
/// - 890
/// `Ord` would have been shorter and is deliberately not used: the axes spell - 891
/// their `rank` out precisely so that reordering variants cannot change a - 892
/// safety decision, and a derived `Ord` would quietly reintroduce exactly that - 893
/// coupling here. - 894
pub trait AxisValue: Copy + PartialEq { - 895
fn axis_name(self) -> &'static str; - 896
- 897
/// Position on an ordered axis, or `None` for a categorical one. - 898
/// - 899
/// This distinction is load-bearing. `Act` is categorical: `modify` and - 900
/// `answer` are rival explanations, and evidence for one really is evidence - 901
/// against the other. `Stakes` is *ordered*: a signal saying "reversible" - 902
/// does not argue against "irreversible", it agrees with it more weakly. - 903
fn axis_rank(self) -> Option<u8> { - 904
None - 905
} - 906
} - 907
- 908
impl AxisValue for Act { - 909
fn axis_name(self) -> &'static str { - 910
self.as_str() - 911
} - 912
} - 913
impl AxisValue for Horizon { - 914
fn axis_name(self) -> &'static str { - 915
self.as_str() - 916
} - 917
- 918
fn axis_rank(self) -> Option<u8> { - 919
Some(self.rank()) - 920
} - 921
} - 922
impl AxisValue for Stakes { - 923
fn axis_name(self) -> &'static str { - 924
self.as_str() - 925
} - 926
- 927
fn axis_rank(self) -> Option<u8> { - 928
Some(self.rank()) - 929
} - 930
} - 931
impl AxisValue for Evidence { - 932
fn axis_name(self) -> &'static str { - 933
self.as_str() - 934
} - 935
- 936
fn axis_rank(self) -> Option<u8> { - 937
Some(self.rank()) - 938
} - 939
} - 940
impl AxisValue for Clarity { - 941
fn axis_name(self) -> &'static str { - 942
self.as_str() - 943
} - 944
} - 945
- 946
/// Weighted votes for one axis. - 947
#[derive(Debug, Clone)] - 948
pub struct Votes<T: AxisValue> { - 949
tally: Vec<(T, f64)>, - 950
} - 951
- 952
impl<T: AxisValue> Default for Votes<T> { - 953
fn default() -> Self { - 954
Votes { tally: Vec::new() } - 955
} - 956
} - 957
- 958
impl<T: AxisValue> Votes<T> { - 959
pub fn add(&mut self, value: T, weight: f64) { - 960
match self.tally.iter_mut().find(|(v, _)| *v == value) { - 961
Some((_, score)) => *score += weight, - 962
None => self.tally.push((value, weight)), - 963
} - 964
} - 965
- 966
pub fn is_empty(&self) -> bool { - 967
self.tally.is_empty() - 968
} - 969
- 970
/// Every value that received weight, strongest first. Ties break by name - 971
/// so the ordering is total and stable across runs. - 972
pub fn ranked(&self) -> Vec<(T, f64)> { - 973
let mut ranked = self.tally.clone(); - 974
ranked.sort_by(|a, b| { - 975
b.1.partial_cmp(&a.1) - 976
.unwrap_or(std::cmp::Ordering::Equal) - 977
.then_with(|| a.0.axis_name().cmp(b.0.axis_name())) - 978
}); - 979
ranked - 980
} - 981
- 982
/// Minimum weight before a signal may escalate an ordered axis. Filters - 983
/// out the incidental 0.3-weight hints so a stray word cannot promote a - 984
/// one-liner to durable multi-day work. - 985
pub(crate) const ESCALATION_FLOOR: f64 = 0.5; - 986
- 987
/// Every value scoring within `band` of the winner, strongest first. - 988
/// - 989
/// Used for the act axis, where two readings being close is often not - 990
/// ambiguity to resolve but a request that genuinely spans both: "fix the - 991
/// failing test" is a `modify` and a `verify`, and picking one loses the - 992
/// other's tools. - 993
pub fn contenders(&self, band: f64) -> Vec<T> { - 994
let ranked = self.ranked(); - 995
let Some((_, best)) = ranked.first().copied() else { - 996
return Vec::new(); - 997
}; - 998
if best <= 0.0 { - 999
return Vec::new(); - 1000
} - 1001
ranked - 1002
.into_iter() - 1003
.filter(|(_, weight)| { - 1004
// The winner is always retained. Others must clear the - 1005
// absolute floor, and either sit within the band of the - 1006
// winner or carry strong independent signal. - 1007
*weight >= best - 1008
|| (*weight >= Self::ESCALATION_FLOOR - 1009
&& (*weight >= best * (1.0 - band) || *weight >= 1.0)) - 1010
}) - 1011
.map(|(value, _)| value) - 1012
.collect() - 1013
} - 1014
- 1015
/// Winner and a [0,1] confidence. - 1016
/// - 1017
/// Two rules, because the axes are not all the same shape: - 1018
/// - 1019
/// * **Categorical** (`Act`, `Clarity`): argmax with confidence from the - 1020
/// margin over the runner-up. - 1021
/// * **Ordered** (`Stakes`, `Horizon`, `Evidence`): the highest level with - 1022
/// real support wins, and lower levels corroborate rather than compete. - 1023
/// Taking the maximum is also the safe direction on every ordered axis - 1024
/// here — more caution, a stricter proof standard, a longer horizon. - 1025
pub fn winner(&self) -> Option<(T, f64)> { - 1026
let ranked = self.ranked(); - 1027
let (best, best_score) = ranked.first().copied()?; - 1028
if best_score <= 0.0 { - 1029
return None; - 1030
} - 1031
if best.axis_rank().is_some() { - 1032
// Everything below the escalation floor abstains: a vote too weak - 1033
// to escalate on its own is not evidence of the level it names. - 1034
let (highest, weight) = ranked - 1035
.iter() - 1036
.filter(|(_, weight)| *weight >= Self::ESCALATION_FLOOR) - 1037
.max_by_key(|(value, _)| value.axis_rank().unwrap_or(0)) - 1038
.copied()?; - 1039
let mass = (weight / 1.5).min(1.0); - 1040
return Some((highest, (0.7 + mass * 0.3).clamp(0.0, 1.0))); - 1041
} - 1042
let runner_up = ranked.get(1).map(|(_, s)| *s).unwrap_or(0.0); - 1043
// How much better the winner is, as a fraction of itself — not its - 1044
// share of the total. Share-of-total punishes any second signal - 1045
// regardless of how weak. - 1046
let margin = 1.0 - (runner_up / best_score).min(1.0); - 1047
// A lone weak signal should not read as certainty either, so the - 1048
// margin is damped by how much evidence there was at all. - 1049
let mass = (best_score / 1.5).min(1.0); - 1050
Some((best, (margin * 0.7 + mass * 0.3).clamp(0.0, 1.0))) - 1051
} - 1052
} - 1053
- 1054
/// Everything tier 1 concluded about one part of a request, before the - 1055
/// resolver turns it into a reading. - 1056
#[derive(Debug, Clone, Default)] - 1057
pub struct Extraction { - 1058
pub signals: Vec<Signal>, - 1059
pub act: Votes<Act>, - 1060
pub horizon: Votes<Horizon>, - 1061
/// Stakes stated by the request's own words. Applied only to effectful - 1062
/// acts: a question about production has no blast radius. - 1063
pub stakes_from_words: Votes<Stakes>, - 1064
/// Stakes implied by the *environment* rather than by the request. - 1065
/// - 1066
/// Kept separate because it is only relevant to acts that actually touch - 1067
/// something: a dirty working tree makes an edit riskier, and says nothing - 1068
/// whatsoever about the risk of saying hello. - 1069
pub stakes_from_environment: Votes<Stakes>, - 1070
pub evidence: Votes<Evidence>, - 1071
pub clarity: Votes<Clarity>, - 1072
pub input_modalities: Vec<Modality>, - 1073
pub output_modalities: Vec<Modality>, - 1074
pub attendance: Attendance, - 1075
pub domains: Vec<String>, - 1076
/// Domains implied by the *environment* rather than by the request's own - 1077
/// words. A git repository says nothing about the subject of a question - 1078
/// asked inside it, so these apply only to effectful acts. - 1079
pub domains_from_environment: Vec<String>, - 1080
/// The strongest temporal reference to a current value, if any. The - 1081
/// resolver turns it into the `live-data` domain only when the act asks - 1082
/// for a fact; "refactor the current implementation" wants no retrieval. - 1083
pub recency: Option<(String, f64)>, - 1084
/// The part points at something it does not contain ("it", "that"). - 1085
pub deictic: bool, - 1086
} - 1087
- 1088
// ------------------------------------------------------------- tokens --- - 1089
- 1090
/// Split into lowercase alphanumeric words, preserving order. - 1091
fn words(text: &str) -> Vec<String> { - 1092
text.to_ascii_lowercase() - 1093
.split(|c: char| !c.is_ascii_alphanumeric()) - 1094
.filter(|w| !w.is_empty()) - 1095
.map(|w| w.to_string()) - 1096
.collect() - 1097
} - 1098
- 1099
/// Crude English de-inflection, enough to match a lexicon of bare verbs. - 1100
/// - 1101
/// Not a stemmer and not trying to be one: it only strips the suffixes that - 1102
/// make a request's actual verb invisible to an exact-match lexicon. - 1103
fn stems(word: &str) -> Vec<&str> { - 1104
let mut out = vec![word]; - 1105
for suffix in ["ing", "ed", "es", "s"] { - 1106
if let Some(stem) = word.strip_suffix(suffix) - 1107
&& stem.len() >= 3 - 1108
{ - 1109
out.push(stem); - 1110
// Doubled consonant before -ing/-ed: "running" -> "runn" -> "run". - 1111
if matches!(suffix, "ing" | "ed") - 1112
&& let Some(trimmed) = stem.strip_suffix(stem.chars().last().unwrap_or(' ')) - 1113
&& trimmed.len() >= 3 - 1114
&& stem.chars().rev().nth(1) == stem.chars().last() - 1115
{ - 1116
out.push(trimmed); - 1117
} - 1118
} - 1119
} - 1120
out - 1121
} - 1122
- 1123
/// A lexicon word with its de-inflected forms, computed once. - 1124
struct Target { - 1125
word: &'static str, - 1126
stems: Vec<&'static str>, - 1127
} - 1128
- 1129
impl Target { - 1130
fn new(word: &'static str) -> Self { - 1131
Target { - 1132
word, - 1133
stems: stems(word), - 1134
} - 1135
} - 1136
} - 1137
- 1138
/// A clause's tokens with their de-inflected forms, computed once, so - 1139
/// matching the whole lexicon against a clause allocates nothing per pair. - 1140
struct Tokens { - 1141
words: Vec<String>, - 1142
} - 1143
- 1144
impl Tokens { - 1145
fn new(text: &str) -> Self { - 1146
Tokens { words: words(text) } - 1147
} - 1148
- 1149
fn len(&self) -> usize { - 1150
self.words.len() - 1151
} - 1152
- 1153
fn get(&self, index: usize) -> Option<&str> { - 1154
self.words.get(index).map(String::as_str) - 1155
} - 1156
- 1157
/// Whether the token at `index` is `target`, allowing inflection and - 1158
/// silent-`e` alterations ("writing" → "write", "deploying" → "deploy"). - 1159
fn matches(&self, index: usize, target: &Target) -> bool { - 1160
let Some(token) = self.get(index) else { - 1161
return false; - 1162
}; - 1163
if token == target.word { - 1164
return true; - 1165
} - 1166
let token_stems = stems(token); - 1167
if token_stems.contains(&target.word) { - 1168
return true; - 1169
} - 1170
if let Some(without_e) = target.word.strip_suffix('e') - 1171
&& token_stems.contains(&without_e) - 1172
{ - 1173
return true; - 1174
} - 1175
for target_stem in &target.stems { - 1176
if token == *target_stem || token_stems.contains(target_stem) { - 1177
return true; - 1178
} - 1179
if token.strip_suffix('e') == Some(*target_stem) { - 1180
return true; - 1181
} - 1182
} - 1183
false - 1184
} - 1185
- 1186
/// First position matching `target`, skipping consumed tokens. - 1187
fn position(&self, target: &Target, consumed: &[bool]) -> Option<usize> { - 1188
(0..self.len()) - 1189
.find(|&i| !consumed.get(i).copied().unwrap_or(false) && self.matches(i, target)) - 1190
} - 1191
- 1192
/// Whether `phrase` (whole words) occurs, and where. - 1193
fn phrase_at(&self, phrase: &[&str]) -> Option<usize> { - 1194
if phrase.is_empty() || phrase.len() > self.len() { - 1195
return None; - 1196
} - 1197
(0..=self.len() - phrase.len()) - 1198
.find(|&start| (0..phrase.len()).all(|i| self.words[start + i] == phrase[i])) - 1199
} - 1200
} - 1201
- 1202
static VERB_TARGETS: LazyLock<Vec<(Target, Act, f64)>> = LazyLock::new(|| { - 1203
ACT_VERBS - 1204
.iter() - 1205
.map(|(word, act, weight)| (Target::new(word), *act, *weight)) - 1206
.collect() - 1207
}); - 1208
- 1209
fn is_lexicon_verb(tokens: &Tokens, index: usize) -> bool { - 1210
VERB_TARGETS - 1211
.iter() - 1212
.any(|(target, _, _)| tokens.matches(index, target)) - 1213
} - 1214
- 1215
fn split_phrase(phrase: &'static str) -> Vec<&'static str> { - 1216
phrase.split(' ').collect() - 1217
} - 1218
- 1219
// ------------------------------------------------------------ clauses --- - 1220
- 1221
/// What a clause is doing. - 1222
#[derive(Debug, Clone, Copy, PartialEq, Eq)] - 1223
pub(crate) enum ClauseKind { - 1224
/// A verb heads it: "fix the parser", "every day, check the bill". - 1225
Imperative, - 1226
/// Asks something: a wh-word or an auxiliary before a subject, or a - 1227
/// trailing question mark. - 1228
Question, - 1229
/// A request phrased as a wish ("I want the parser refactored"), or one - 1230
/// led by a verb the lexicon does not know ("sync the files daily"). - 1231
Request, - 1232
/// Greetings and thanks and nothing else. - 1233
Social, - 1234
/// Describes the situation: "it crashes on empty input". - 1235
Statement, - 1236
} - 1237
- 1238
impl ClauseKind { - 1239
/// Whether this clause asks for work, and so may stand as a strand. - 1240
pub(crate) fn is_work(self) -> bool { - 1241
matches!( - 1242
self, - 1243
ClauseKind::Imperative | ClauseKind::Question | ClauseKind::Request - 1244
) - 1245
} - 1246
} - 1247
- 1248
/// One clause, read once. - 1249
pub(crate) struct ClauseRead { - 1250
pub text: String, - 1251
pub boundary: crate::strand::Boundary, - 1252
pub kind: ClauseKind, - 1253
tokens: Tokens, - 1254
/// Tokens a social phrase already accounted for. - 1255
consumed: Vec<bool>, - 1256
/// Converse weight from greetings stripped off the front. - 1257
greeting: f64, - 1258
/// Signals for what was stripped, for the explain view. - 1259
stripped: Vec<String>, - 1260
question_mark: bool, - 1261
} - 1262
- 1263
impl ClauseRead { - 1264
/// Whether this clause asks for work of its own, and so starts a part: a - 1265
/// question, or an instruction or wish that names something to do. - 1266
/// Everything else is context for the part beside it. - 1267
pub(crate) fn starts_part(&self) -> bool { - 1268
match self.kind { - 1269
ClauseKind::Question => true, - 1270
ClauseKind::Imperative | ClauseKind::Request => self.has_act_votes(), - 1271
ClauseKind::Social | ClauseKind::Statement => false, - 1272
} - 1273
} - 1274
- 1275
/// Whether this clause votes for any act at all. - 1276
fn has_act_votes(&self) -> bool { - 1277
match self.kind { - 1278
ClauseKind::Statement | ClauseKind::Social => { - 1279
(1..self.tokens.len()).any(|i| self.is_head(i) && is_lexicon_verb(&self.tokens, i)) - 1280
} - 1281
_ => (0..self.tokens.len()) - 1282
.any(|i| !self.consumed[i] && is_lexicon_verb(&self.tokens, i)), - 1283
} - 1284
} - 1285
- 1286
/// Whether the token at `index` heads its (sub-)clause: first word, or - 1287
/// right after a conjunction or sequencing word, or after punctuation. - 1288
fn is_head(&self, index: usize) -> bool { - 1289
if index == 0 { - 1290
return true; - 1291
} - 1292
if let Some(prev) = self.tokens.get(index - 1) - 1293
&& HEAD_PREDECESSORS.contains(&prev) - 1294
{ - 1295
return true; - 1296
} - 1297
if index > 1 - 1298
&& self.tokens.get(index - 2) == Some("and") - 1299
&& self.tokens.get(index - 1) == Some("then") - 1300
{ - 1301
return true; - 1302
} - 1303
let Some(target) = self.tokens.get(index) else { - 1304
return false; - 1305
}; - 1306
let lower = self.text.to_ascii_lowercase(); - 1307
for (at, _) in lower.match_indices(target) { - 1308
let before = &lower[..at]; - 1309
let after = &lower[at + target.len()..]; - 1310
let word_start = - 1311
before.is_empty() || before.ends_with(|c: char| !c.is_ascii_alphanumeric()); - 1312
let word_end = - 1313
after.is_empty() || after.starts_with(|c: char| !c.is_ascii_alphanumeric()); - 1314
if word_start && word_end && before.trim_end().ends_with([',', ';', ':', '.', '\n']) { - 1315
return true; - 1316
} - 1317
} - 1318
false - 1319
} - 1320
} - 1321
- 1322
fn starts_with_words(tokens: &[String], phrase: &str) -> Option<usize> { - 1323
let needle: Vec<&str> = phrase.split(' ').collect(); - 1324
(tokens.len() >= needle.len() && needle.iter().zip(tokens).all(|(a, b)| a == b)) - 1325
.then_some(needle.len()) - 1326
} - 1327
- 1328
/// Drop `count` leading words from `text`, keeping the original spelling of - 1329
/// what remains. - 1330
fn drop_leading_words(text: &str, count: usize) -> &str { - 1331
let mut seen = 0usize; - 1332
let mut in_word = false; - 1333
for (i, c) in text.char_indices() { - 1334
let is_word = c.is_ascii_alphanumeric(); - 1335
if is_word && !in_word { - 1336
if seen == count { - 1337
return &text[i..]; - 1338
} - 1339
seen += 1; - 1340
} - 1341
in_word = is_word; - 1342
} - 1343
"" - 1344
} - 1345
- 1346
/// Read one clause: strip what leads it, then decide its kind. - 1347
pub(crate) fn read_clause( - 1348
text: &str, - 1349
boundary: crate::strand::Boundary, - 1350
question_mark: bool, - 1351
) -> ClauseRead { - 1352
let mut rest = text.trim(); - 1353
let mut greeting = 0.0f64; - 1354
let mut stripped = Vec::new(); - 1355
let mut acknowledged = false; - 1356
// Greetings, acknowledgements, polite openers and sequencing words, in - 1357
// any order, until none applies. - 1358
loop { - 1359
let tokens = words(rest); - 1360
let Some(first) = tokens.first() else { - 1361
break; - 1362
}; - 1363
if let Some((phrase, weight)) = SOCIAL_PHRASES - 1364
.iter() - 1365
.find(|(phrase, _)| starts_with_words(&tokens, phrase).is_some()) - 1366
.filter(|(phrase, _)| starts_with_words(&tokens, phrase) != Some(tokens.len())) - 1367
{ - 1368
// A social phrase leading a longer clause is a greeting. - 1369
if let Some(n) = starts_with_words(&tokens, phrase) { - 1370
greeting = greeting.max(*weight); - 1371
stripped.push(format!("greeting `{phrase}`")); - 1372
rest = drop_leading_words(rest, n); - 1373
continue; - 1374
} - 1375
} - 1376
if tokens.len() > 1 - 1377
&& let Some((word, weight)) = SOCIAL_WORDS.iter().find(|(word, _)| word == first) - 1378
{ - 1379
greeting = greeting.max(*weight); - 1380
stripped.push(format!("greeting `{word}`")); - 1381
rest = drop_leading_words(rest, 1); - 1382
continue; - 1383
} - 1384
if tokens.len() > 1 && ACKNOWLEDGEMENTS.contains(&first.as_str()) { - 1385
acknowledged = true; - 1386
stripped.push(format!("acknowledgement `{first}`")); - 1387
rest = drop_leading_words(rest, 1); - 1388
continue; - 1389
} - 1390
if let Some(n) = PREAMBLES - 1391
.iter() - 1392
.filter_map(|preamble| starts_with_words(&tokens, preamble)) - 1393
.max() - 1394
.filter(|n| *n < tokens.len()) - 1395
{ - 1396
stripped.push(format!("opener `{}`", tokens[..n].join(" "))); - 1397
rest = drop_leading_words(rest, n); - 1398
continue; - 1399
} - 1400
if tokens.len() > 1 && LEAD_FILLERS.contains(&first.as_str()) { - 1401
rest = drop_leading_words(rest, 1); - 1402
continue; - 1403
} - 1404
break; - 1405
} - 1406
let rest = rest.trim_start_matches(|c: char| !c.is_ascii_alphanumeric() && c.is_ascii()); - 1407
let tokens = Tokens::new(rest); - 1408
let mut consumed = vec![false; tokens.len()]; - 1409
let mut social_phrase = false; - 1410
for (phrase, _) in SOCIAL_PHRASES { - 1411
let words: Vec<&str> = split_phrase(phrase); - 1412
if let Some(at) = tokens.phrase_at(&words) { - 1413
social_phrase = true; - 1414
for flag in consumed.iter_mut().skip(at).take(words.len()) { - 1415
*flag = true; - 1416
} - 1417
} - 1418
} - 1419
let lone_social = tokens.len() == 1 - 1420
&& SOCIAL_WORDS - 1421
.iter() - 1422
.any(|(word, _)| tokens.get(0) == Some(*word)); - 1423
// A wish is a request, and the words that say so are not verbs of - 1424
// their own: the `what` of "what I need is…" asks nothing. - 1425
let desire = DESIRE_PREFIXES - 1426
.iter() - 1427
.filter_map(|prefix| starts_with_words(&tokens.words, prefix)) - 1428
.max(); - 1429
if let Some(length) = desire { - 1430
for flag in consumed.iter_mut().take(length) { - 1431
*flag = true; - 1432
} - 1433
} - 1434
let mut read = ClauseRead { - 1435
text: rest.to_string(), - 1436
boundary, - 1437
kind: ClauseKind::Statement, - 1438
tokens, - 1439
consumed, - 1440
greeting, - 1441
stripped, - 1442
question_mark, - 1443
}; - 1444
read.kind = classify( - 1445
&read, - 1446
social_phrase, - 1447
lone_social, - 1448
acknowledged, - 1449
desire.is_some(), - 1450
); - 1451
read - 1452
} - 1453
- 1454
fn classify( - 1455
read: &ClauseRead, - 1456
social_phrase: bool, - 1457
lone_social: bool, - 1458
acknowledged: bool, - 1459
desire: bool, - 1460
) -> ClauseKind { - 1461
let tokens = &read.tokens; - 1462
if tokens.len() == 0 { - 1463
return if read.greeting > 0.0 || acknowledged { - 1464
ClauseKind::Social - 1465
} else { - 1466
ClauseKind::Statement - 1467
}; - 1468
} - 1469
if lone_social || (social_phrase && read.consumed.iter().all(|c| *c)) { - 1470
return ClauseKind::Social; - 1471
} - 1472
if desire { - 1473
return ClauseKind::Request; - 1474
} - 1475
let first = tokens.get(0).unwrap_or(""); - 1476
let second = tokens.get(1).unwrap_or(""); - 1477
let asks = WH_WORDS.contains(&first) - 1478
|| (first == "do" && PERSONS.contains(&second)) - 1479
|| (first != "do" && AUX_WORDS.contains(&first) && QUESTION_SUBJECTS.contains(&second)) - 1480
|| read.question_mark; - 1481
if asks && !(social_phrase && read.consumed.first().copied().unwrap_or(false)) { - 1482
return ClauseKind::Question; - 1483
} - 1484
if social_phrase && read.consumed.first().copied().unwrap_or(false) { - 1485
return ClauseKind::Social; - 1486
} - 1487
// A verb heading a later sub-clause ("the build is broken, fix it") - 1488
// makes the clause an instruction whatever its first words say. - 1489
if (1..tokens.len()).any(|i| read.is_head(i) && is_lexicon_verb(tokens, i)) { - 1490
return ClauseKind::Imperative; - 1491
} - 1492
// "Build fails on CI", "the deploy failed", "move occurs because…": a - 1493
// subject, then its verb. The first word is the subject even when it - 1494
// could be a verb somewhere else. - 1495
if STATEMENT_VERBS.contains(&second) || first.chars().any(|c| c.is_ascii_digit()) { - 1496
return ClauseKind::Statement; - 1497
} - 1498
if read.is_head(0) && is_lexicon_verb(tokens, 0) { - 1499
return ClauseKind::Imperative; - 1500
} - 1501
if SUBJECT_STARTERS.contains(&first) { - 1502
return ClauseKind::Statement; - 1503
} - 1504
ClauseKind::Request - 1505
} - 1506
- 1507
// ------------------------------------------------------------ material --- - 1508
- 1509
/// A request with its pasted material set aside. - 1510
#[derive(Debug, Clone, Default, PartialEq)] - 1511
pub(crate) struct Prepared { - 1512
/// The text that may vote. - 1513
pub instruction: String, - 1514
/// Fenced code blocks removed. - 1515
pub fenced_blocks: usize, - 1516
/// Lines of unfenced pasted material removed. - 1517
pub pasted_lines: usize, - 1518
} - 1519
- 1520
fn is_list_item(line: &str) -> bool { - 1521
let trimmed = line.trim_start(); - 1522
if trimmed.starts_with("- ") || trimmed.starts_with("* ") || trimmed.starts_with("+ ") { - 1523
return true; - 1524
} - 1525
let digits = trimmed.chars().take_while(char::is_ascii_digit).count(); - 1526
digits > 0 - 1527
&& digits <= 3 - 1528
&& (trimmed[digits..].starts_with(". ") || trimmed[digits..].starts_with(") ")) - 1529
} - 1530
- 1531
/// Whether a line reads as something a person wrote to the agent, rather - 1532
/// than something pasted for it to look at. - 1533
fn reads_as_request(line: &str) -> bool { - 1534
let trimmed = line.trim(); - 1535
if trimmed.is_empty() { - 1536
return true; - 1537
} - 1538
// Indented lines are code, stack frames and tracebacks. - 1539
if line.starts_with('\t') || line.starts_with(" ") { - 1540
return false; - 1541
} - 1542
if is_list_item(line) { - 1543
return true; - 1544
} - 1545
let tokens = Tokens::new(trimmed); - 1546
if let Some(first) = tokens.get(0) - 1547
&& (is_lexicon_verb(&tokens, 0) - 1548
|| WH_WORDS.contains(&first) - 1549
|| AUX_WORDS.contains(&first) - 1550
|| SOCIAL_WORDS.iter().any(|(word, _)| *word == first) - 1551
|| PREAMBLES - 1552
.iter() - 1553
.any(|preamble| preamble.split(' ').next() == Some(first))) - 1554
{ - 1555
return true; - 1556
} - 1557
let visible: Vec<char> = trimmed.chars().filter(|c| !c.is_whitespace()).collect(); - 1558
if visible.is_empty() { - 1559
return true; - 1560
} - 1561
let prose = visible - 1562
.iter() - 1563
.filter(|c| { - 1564
c.is_alphabetic() - 1565
|| matches!( - 1566
c, - 1567
',' | '.' | '\'' | '?' | '!' | ':' | ';' | '-' | '"' | '(' | ')' - 1568
) - 1569
}) - 1570
.count(); - 1571
let letters = visible.iter().filter(|c| c.is_alphabetic()).count(); - 1572
let prose_like = prose as f64 / visible.len() as f64 >= 0.9 - 1573
&& letters as f64 / visible.len() as f64 >= 0.7 - 1574
&& tokens.len() >= 2; - 1575
let ends_like_prose = trimmed.ends_with(['.', '?', '!', ':']); - 1576
prose_like || (ends_like_prose && letters as f64 / visible.len() as f64 >= 0.6) - 1577
} - 1578
- 1579
/// Set pasted material aside: fenced blocks, and runs of two or more lines - 1580
/// that do not read as a request (or one very long one). What remains is the - 1581
/// request itself. - 1582
pub(crate) fn prepare(raw: &str) -> Prepared { - 1583
fn flush<'a>(run: &mut Vec<&'a str>, kept: &mut Vec<&'a str>, prepared: &mut Prepared) { - 1584
let material = run.len() >= 2 || run.iter().any(|line| line.len() >= 160); - 1585
if material { - 1586
prepared.pasted_lines += run.len(); - 1587
} else { - 1588
kept.append(run); - 1589
} - 1590
run.clear(); - 1591
} - 1592
let text = clean_request_text(raw); - 1593
let mut prepared = Prepared::default(); - 1594
let mut kept: Vec<&str> = Vec::new(); - 1595
let mut run: Vec<&str> = Vec::new(); - 1596
let mut in_fence = false; - 1597
for line in text.lines() { - 1598
let trimmed = line.trim_start(); - 1599
if trimmed.starts_with("```") || trimmed.starts_with("~~~") { - 1600
if !in_fence { - 1601
flush(&mut run, &mut kept, &mut prepared); - 1602
prepared.fenced_blocks += 1; - 1603
} - 1604
in_fence = !in_fence; - 1605
continue; - 1606
} - 1607
if in_fence { - 1608
continue; - 1609
} - 1610
if reads_as_request(line) { - 1611
flush(&mut run, &mut kept, &mut prepared); - 1612
kept.push(line); - 1613
} else { - 1614
run.push(line); - 1615
} - 1616
} - 1617
flush(&mut run, &mut kept, &mut prepared); - 1618
prepared.instruction = kept.join("\n"); - 1619
prepared - 1620
} - 1621
- 1622
/// Strip prompt scaffolding and runner control blocks before extracting - 1623
/// intent. Delegates to [`crate::control`], which owns the tag vocabulary. - 1624
pub(crate) fn clean_request_text(raw: &str) -> String { - 1625
let mut text = crate::control::strip_control_blocks(raw); - 1626
while let Some(start) = text.find("[Scheduled-run context:") { - 1627
if let Some(end_offset) = text[start..].find(']') { - 1628
let end = start + end_offset + 1; - 1629
text.replace_range(start..end, ""); - 1630
} else { - 1631
text.truncate(start); - 1632
break; - 1633
} - 1634
} - 1635
text.trim().to_string() - 1636
} - 1637
- 1638
// ------------------------------------------------------------- digest ---
Indexing the workspace…
Vakyartha documentation is discovering safe artifacts, anchors, and source references.