- 397
/// you" does not also vote for a factual answer. - 398
const SOCIAL_PHRASES: &[(&str, f64)] = &[ - 399
("how are you", 1.3), - 400
("how is it going", 1.3), - 401
("how s it going", 1.3), - 402
("what s up", 1.3), - 403
("good morning", 1.3), - 404
("good afternoon", 1.3), - 405
("good evening", 1.3), - 406
("good night", 1.3), - 407
("nice to meet you", 1.3), - 408
("nice to see you", 1.3), - 409
("long time no see", 1.3), - 410
("see you later", 1.3), - 411
("see you soon", 1.3), - 412
("thank you", 1.0), - 413
]; - 414
- 415
/// Acknowledgements stripped from the front of a clause. They carry no - 416
/// request of their own; alone they are social. - 417
const ACKNOWLEDGEMENTS: &[&str] = &["ok", "okay", "cool", "great", "sure", "alright", "right"]; - 418
- 419
/// Polite and conversational openers stripped so the operational verb lands - 420
/// in head position. - 421
const PREAMBLES: &[&str] = &[ - 422
"could you please", - 423
"can you please", - 424
"would you please", - 425
"would you mind", - 426
"could you", - 427
"can you", - 428
"would you", - 429
"will you", - 430
"can u", - 431
"can we", - 432
"could we", - 433
"please", - 434
"pls", - 435
"plz", - 436
"kindly", - 437
"just", - 438
"go ahead and", - 439
"help me to", - 440
"help me", - 441
"i would like you to", - 442
"i d like you to", - 443
"i want you to", - 444
"i need you to", - 445
"i want to", - 446
"i need to", - 447
"we need to", - 448
"let s", - 449
"lets", - 450
]; - 451
- 452
/// Sequencing words at the front of a clause. Structure, not verbs. - 453
const LEAD_FILLERS: &[&str] = &[ - 454
"first", "firstly", "second", "secondly", "third", "thirdly", "next", "now", "then", "finally", - 455
"lastly", "also", "and", "so", "actually", "well", "hmm", "oh", "btw", - 456
]; - 457
- 458
/// Openers that make a statement a request: "I want the parser refactored", - 459
/// "what I need is for you to refactor the module". Their words are - 460
/// consumed: the `what` of a pseudo-cleft does not ask a question. - 461
const DESIRE_PREFIXES: &[&str] = &[ - 462
"what i need", - 463
"what we need", - 464
"what i want", - 465
"what we want", - 466
"what i d like", - 467
"all i need", - 468
"i want", - 469
"i d like", - 470
"i would like", - 471
"we want", - 472
"we d like", - 473
"we would like", - 474
"i need", - 475
"we need", - 476
"i wish", - 477
]; - 478
- 479
/// Who the requester is. "Alert me", "send me the report": delivering a - 480
/// result to the person asking is an answer, not an irreversible operation - 481
/// on someone else. - 482
const REQUESTER: &[&str] = &["me", "us", "myself", "ourselves"]; - 483
- 484
/// A question's head: wh-words vote through [`ACT_VERBS`]; these auxiliaries - 485
/// open a question when a subject follows ("did the email send", "can I - 486
/// deploy on Friday", "is it done"). - 487
const AUX_WORDS: &[&str] = &[ - 488
"is", "are", "was", "were", "am", "does", "did", "has", "have", "had", "can", "could", "will", - 489
"would", "should", "shall", "may", "might", "isn", "aren", "wasn", "weren", "doesn", "didn", - 490
"hasn", "haven", "hadn", "couldn", "wouldn", "shouldn", "won", "do", - 491
]; - 492
const WH_WORDS: &[&str] = &[ - 493
"what", "why", "how", "where", "when", "which", "who", "whom", "whose", - 494
]; - 495
- 496
/// Words that follow an auxiliary in a question. `do` asks only before a - 497
/// person ("do you know"), because "do this every day" is an instruction. - 498
const QUESTION_SUBJECTS: &[&str] = &[ - 499
"i", - 500
"you", - 501
"we", - 502
"they", - 503
"he", - 504
"she", - 505
"it", - 506
"this", - 507
"that", - 508
"these", - 509
"those", - 510
"the", - 511
"there", - 512
"my", - 513
"our", - 514
"your", - 515
"their", - 516
"his", - 517
"her", - 518
"its", - 519
"any", - 520
"anyone", - 521
"anything", - 522
"anybody", - 523
"everyone", - 524
"everything", - 525
"someone", - 526
"something", - 527
"all", - 528
"each", - 529
"every", - 530
]; - 531
const PERSONS: &[&str] = &["i", "you", "we", "they", "he", "she"]; - 532
- 533
/// Verbs that, in second place, make the first word a subject: "build fails - 534
/// on CI", "deploy failed", "value moved here". A statement describes the - 535
/// situation; its first word is not an instruction even when it could be a - 536
/// verb somewhere else. - 537
const STATEMENT_VERBS: &[&str] = &[ - 538
"is", "are", "was", "were", "has", "have", "had", "does", "did", "doesn", "didn", "isn", - 539
"wasn", "won", "can", "cannot", "should", "will", "would", "fails", "failed", "failing", - 540
"occurs", "occurred", "happens", "happened", "crashes", "crashed", "crashing", "breaks", - 541
"broke", "broken", "works", "worked", "returns", "returned", "throws", "threw", "shows", - 542
"showed", "says", "said", "seems", "seemed", "looks", "looked", "keeps", "kept", "stopped", - 543
"started", "hangs", "hung", "errors", "errored", "times", "timed", "panics", "panicked", - 544
"moved", "passes", "passed", "runs", "ran", "went", "gets", "got", - 545
]; - 546
- 547
/// A clause that starts with one of these is a statement unless a verb heads - 548
/// one of its sub-clauses: "it crashes on empty input", "the deploy failed". - 549
const SUBJECT_STARTERS: &[&str] = &[ - 550
"i", - 551
"we", - 552
"you", - 553
"they", - 554
"he", - 555
"she", - 556
"it", - 557
"this", - 558
"that", - 559
"these", - 560
"those", - 561
"the", - 562
"a", - 563
"an", - 564
"my", - 565
"our", - 566
"your", - 567
"their", - 568
"his", - 569
"her", - 570
"its", - 571
"there", - 572
"here", - 573
"some", - 574
"all", - 575
"each", - 576
"every", - 577
"no", - 578
"none", - 579
"one", - 580
"someone", - 581
"something", - 582
"everyone", - 583
"everything", - 584
"nothing", - 585
"nobody", - 586
"anyone", - 587
"anything", - 588
]; - 589
- 590
/// Words that raise stakes when the act touches something. They name the - 591
/// target environment or the nature of the operation, never a topic: - 592
/// "customer", "payment", "live" and "everyone" read "rename the Customer - 593
/// struct" and "fix the payment form validation" as irreversible, capped - 594
/// approval at `ask`, and told the model to confirm an edit. - 595
const STAKES_WORDS: &[(&str, Stakes, f64)] = &[ - 596
("production", Stakes::Irreversible, 1.0), - 597
("prod", Stakes::Irreversible, 0.8), - 598
("irreversible", Stakes::Irreversible, 1.0), - 599
("irreversibly", Stakes::Irreversible, 1.0), - 600
("permanently", Stakes::Irreversible, 0.9), - 601
("force push", Stakes::Irreversible, 0.9), - 602
// The flag spellings of the same commands: `git push --force`, - 603
// `git push -f`, `git reset --hard` discard history or uncommitted work. - 604
("push force", Stakes::Irreversible, 0.9), - 605
("push f", Stakes::Irreversible, 0.9), - 606
("reset hard", Stakes::Irreversible, 0.9), - 607
("rm rf", Stakes::Irreversible, 1.0), - 608
("drop table", Stakes::Irreversible, 0.9), - 609
("drop database", Stakes::Irreversible, 1.0), - 610
// Launching: "we go live in an hour". Bare `live` is a topic word or a - 611
// recency word ([`LIVE_WORD`]), never stakes. - 612
("go live", Stakes::Irreversible, 0.8), - 613
("going live", Stakes::Irreversible, 0.8), - 614
]; - 615
- 616
/// Words that raise the evidence standard. - 617
const EVIDENCE_WORDS: &[(&str, Evidence, f64)] = &[ - 618
("cite", Evidence::Cited, 1.0), - 619
("cites", Evidence::Cited, 0.9), - 620
("citation", Evidence::Cited, 1.0), - 621
("citations", Evidence::Cited, 1.0), - 622
// Matched exactly (see `EXACT_EVIDENCE_WORDS`): inflection would fold - 623
// "source code" into it. - 624
("sources", Evidence::Cited, 0.8), - 625
("evidence", Evidence::Cited, 0.6), - 626
("prove", Evidence::Verified, 0.8), - 627
("proof", Evidence::Verified, 0.7), - 628
("passing", Evidence::Verified, 0.7), - 629
("sign off", Evidence::Audited, 0.9), - 630
("signed off", Evidence::Audited, 0.9), - 631
("acceptance test", Evidence::Audited, 0.8), - 632
("acceptance testing", Evidence::Audited, 0.8), - 633
]; - 634
- 635
/// Evidence words whose inflections mean something else. `sources` must not - 636
/// match `source` — "the source code" is not a citation request. - 637
const EXACT_EVIDENCE_WORDS: &[&str] = &["sources"]; - 638
- 639
/// "Make sure" and "ensure" demand machine-checkable proof only when the - 640
/// thing to be sure of is checkable: "make sure the tests pass" is a - 641
/// predicate, "make sure it rhymes" is a style instruction, and reading the - 642
/// latter as `verified` made the stop gate demand a shell receipt for a poem. - 643
const ASSURANCE_PHRASES: &[(&str, f64)] = &[("make sure", 0.8), ("ensure", 0.7)]; - 644
const CHECKABLE_WORDS: &[&str] = &[ - 645
"test", - 646
"tests", - 647
"testing", - 648
"pass", - 649
"passes", - 650
"passing", - 651
"build", - 652
"builds", - 653
"compile", - 654
"compiles", - 655
"compiling", - 656
"lint", - 657
"ci", - 658
"green", - 659
"works", - 660
"working", - 661
"run", - 662
"runs", - 663
]; - 664
- 665
/// `live` is two words. As an adjective or adverb ("a live score", "is it - 666
/// live") it means *current*, and is a recency word like "right now": a - 667
/// request for a fact with it asks for a value observed this turn. As a verb - 668
/// ("we live in the city", "my kids live with me") it means *reside*, and - 669
/// says nothing about time: measured live, "we live in the city" in a - 670
/// weekend-planning request set `live-data`, the freshness check refused the - 671
/// plan card, and the person got no plan at all. "Go live" — launching — - 672
/// is a stakes phrase, not a recency word ([`STAKES_WORDS`]). - 673
/// - 674
/// The verb reading is recognised from its neighbours, never from a topic: - 675
/// a subject or auxiliary right before it, or — unless a copula or "go" - 676
/// right before it makes it the adjective — a residence preposition right - 677
/// after it. Only the bare form `live` is ever read in the current sense; - 678
/// `lives`, `lived` and `living` are always the verb. - 679
const LIVE_WORD: &str = "live"; - 680
/// A word before `live` that makes it the verb "reside". - 681
const RESIDE_SUBJECTS: &[&str] = &[ - 682
"i", "we", "you", "they", "he", "she", "who", "people", "both", "all", "to", "can", "could", - 683
"would", "will", "might", "should", "must", "not", "never", "t", "d", "ll", - 684
]; - 685
/// A word after `live` that makes it "reside", unless [`LIVE_COPULAS`] - 686
/// precedes it. - 687
const RESIDE_PREPOSITIONS: &[&str] = &[ - 688
"in", "near", "with", "nearby", "abroad", "alone", "together", "close", "downtown", "outside", - 689
]; - 690
/// A word before `live` that keeps it the adjective ("is live in prod", - 691
/// "go live in an hour"). - 692
const LIVE_COPULAS: &[&str] = &[ - 693
"is", "are", "was", "were", "be", "been", "being", "s", "re", "go", "goes", "going", "went", - 694
"gone", "now", - 695
]; - 696
- 697
/// Whether some occurrence of `live` in `tokens` means *current*, not - 698
/// *reside*. - 699
fn live_means_current(tokens: &[String]) -> bool { - 700
tokens.iter().enumerate().any(|(i, token)| { - 701
if token != LIVE_WORD { - 702
return false; - 703
} - 704
let prev = i.checked_sub(1).map(|p| tokens[p].as_str()); - 705
let next = tokens.get(i + 1).map(String::as_str); - 706
let copula = prev.is_some_and(|w| LIVE_COPULAS.contains(&w)); - 707
let subject = prev.is_some_and(|w| RESIDE_SUBJECTS.contains(&w)); - 708
let preposition = next.is_some_and(|w| RESIDE_PREPOSITIONS.contains(&w)); - 709
copula || !(subject || preposition) - 710
}) - 711
} - 712
- 713
/// Temporal deixis: the request asks for a value as it stands *now*, which - 714
/// no model knows from training and which must therefore be observed on this - 715
/// turn (`live-data` domain, docs/design/68-context-engine.md §7). References - 716
/// to time, never to a topic. Whether it applies also depends on the act — - 717
/// only a request for a fact asks for a current value — which the resolver - 718
/// decides once the act is known. - 719
const RECENCY_PHRASES: &[(&str, f64)] = &[ - 720
("right now", 1.0), - 721
("currently", 0.9), - 722
("current", 0.8), - 723
("as of today", 1.0), - 724
("as of now", 1.0), - 725
("today", 0.6), - 726
("tonight", 0.7), - 727
("this morning", 0.8), - 728
("this week", 0.5), - 729
("latest", 0.7), - 730
("real time", 0.8), - 731
("at the moment", 0.9), - 732
("up to date", 0.7), - 733
// Only in the sense of *current* ([`live_means_current`]). - 734
(LIVE_WORD, 0.5), - 735
]; - 736
- 737
/// Nouns that make a recency word local rather than live: "the current - 738
/// directory", "the latest changes", "live reload". The workspace, the - 739
/// conversation and the running session are observed with local tools and - 740
/// hold no value a model could carry over stale from training. - 741
const LOCAL_NOUNS: &[&str] = &[ - 742
"directory", - 743
"dir", - 744
"folder", - 745
"file", - 746
"files", - 747
"path", - 748
"branch", - 749
"commit", - 750
"commits", - 751
"diff", - 752
"changes", - 753
"change", - 754
"working", - 755
"workspace", - 756
"project", - 757
"repo", - 758
"repository", - 759
"codebase", - 760
"code", - 761
"implementation", - 762
"function", - 763
"method", - 764
"class", - 765
"module", - 766
"line", - 767
"lines", - 768
"cursor", - 769
"selection", - 770
"tab", - 771
"window", - 772
"page", - 773
"screen", - 774
"session", - 775
"conversation", - 776
"chat", - 777
"thread", - 778
"context", - 779
"task", - 780
"plan", - 781
"step", - 782
"turn", - 783
"user", - 784
"config", - 785
"configuration", - 786
"settings", - 787
"setup", - 788
"build", - 789
"test", - 790
"tests", - 791
"reload", - 792
"preview", - 793
"server", - 794
"share", - 795
"coding", - 796
"edit", - 797
"editing", - 798
"demo", - 799
"mode", - 800
"state", - 801
// The agent's own state is read with its own tools, not retrieved from - 802
// the world: "what are you holding right now", "your current plan". - 803
"you", - 804
"your", - 805
"yours", - 806
"yourself", - 807
"commitment", - 808
"commitments", - 809
"tasks", - 810
"plans", - 811
"inbox", - 812
"memory", - 813
"notes", - 814
"reminders", - 815
]; - 816
- 817
/// The runtime tells the model the time and date on every turn, so asking - 818
/// for them needs no retrieval. - 819
const TIME_WORDS: &[&str] = &[ - 820
"time", "date", "day", "weekday", "clock", "timezone", "hour", "year", "month", - 821
]; - 822
- 823
/// Phrases implying the work outlives this turn. Sequencing words ("then", - 824
/// "after that") are not here: they separate the parts of one request, which - 825
/// is what strands are for, not a claim that the work spans sessions. - 826
const HORIZON_PHRASES: &[(&str, Horizon, f64)] = &[ - 827
("every day", Horizon::Durable, 1.0), - 828
("every night", Horizon::Durable, 1.0), - 829
("every morning", Horizon::Durable, 1.0), - 830
("every evening", Horizon::Durable, 1.0), - 831
("every month", Horizon::Durable, 1.0), - 832
("every week", Horizon::Durable, 1.0), - 833
("every hour", Horizon::Durable, 1.0), - 834
("each day", Horizon::Durable, 1.0), - 835
("each week", Horizon::Durable, 1.0), - 836
("each month", Horizon::Durable, 1.0), - 837
("every monday", Horizon::Durable, 1.0), - 838
("every tuesday", Horizon::Durable, 1.0), - 839
("every wednesday", Horizon::Durable, 1.0), - 840
("every thursday", Horizon::Durable, 1.0), - 841
("every friday", Horizon::Durable, 1.0), - 842
("every saturday", Horizon::Durable, 1.0), - 843
("every sunday", Horizon::Durable, 1.0), - 844
("every weekday", Horizon::Durable, 1.0), - 845
("every weekend", Horizon::Durable, 1.0), - 846
("nightly", Horizon::Durable, 0.9), - 847
("monthly", Horizon::Durable, 0.9), - 848
("daily", Horizon::Durable, 0.9), - 849
("weekly", Horizon::Durable, 0.9), - 850
("hourly", Horizon::Durable, 0.9), - 851
("continuously", Horizon::Durable, 0.9), - 852
("whenever", Horizon::Durable, 0.7), - 853
("keep watching", Horizon::Durable, 1.0), - 854
("keep an eye", Horizon::Durable, 0.9), - 855
("from now on", Horizon::Durable, 0.9), - 856
("ongoing", Horizon::Durable, 0.8), - 857
("until", Horizon::Durable, 0.4), - 858
("over the next", Horizon::Durable, 0.7), - 859
]; - 860
- 861
/// Recurrence words that are adjectives after a determiner ("the nightly - 862
/// job") and adverbs otherwise ("check it nightly"). Only the adverb votes. - 863
const RECURRENCE_ADJECTIVES: &[&str] = &["nightly", "daily", "weekly", "monthly", "hourly"]; - 864
const DETERMINERS: &[&str] = &[ - 865
"the", "a", "an", "this", "that", "our", "my", "your", "its", "their", "each", "of", - 866
]; - 867
- 868
/// Deictic markers: the request points at something it does not contain. - 869
pub(crate) const DEICTIC_WORDS: &[&str] = &[ - 870
"this", "that", "it", "these", "those", "here", "there", "again", "same", - 871
]; - 872
- 873
/// Pronouns that leave a request with no object of its own when they end - 874
/// it: "fix it", "deploy that". "Write a function that parses dates" uses - 875
/// `that` as a relative pronoun and points at nothing. - 876
const BARE_PRONOUNS: &[&str] = &["it", "this", "that", "these", "those"]; - 877
- 878
/// Words before a verb that still leave it heading its clause. - 879
const HEAD_PREDECESSORS: &[&str] = &[ - 880
"and", "then", "also", "plus", "next", "after", "or", "now", "please", "first", "finally", "so", - 881
]; - 882
- 883
const IMPERATIVE_BONUS: f64 = 1.6; - 884
- 885
// ------------------------------------------------------------ scoring --- - 886
- 887
/// An axis value that can name itself, so vote tallies can break ties - 888
/// deterministically without depending on the enum's declaration order. - 889
/// - 890
/// `Ord` would have been shorter and is deliberately not used: the axes spell - 891
/// their `rank` out precisely so that reordering variants cannot change a - 892
/// safety decision, and a derived `Ord` would quietly reintroduce exactly that - 893
/// coupling here. - 894
pub trait AxisValue: Copy + PartialEq { - 895
fn axis_name(self) -> &'static str; - 896
- 897
/// Position on an ordered axis, or `None` for a categorical one. - 898
/// - 899
/// This distinction is load-bearing. `Act` is categorical: `modify` and - 900
/// `answer` are rival explanations, and evidence for one really is evidence - 901
/// against the other. `Stakes` is *ordered*: a signal saying "reversible" - 902
/// does not argue against "irreversible", it agrees with it more weakly. - 903
fn axis_rank(self) -> Option<u8> { - 904
None - 905
} - 906
} - 907
- 908
impl AxisValue for Act { - 909
fn axis_name(self) -> &'static str { - 910
self.as_str() - 911
} - 912
} - 913
impl AxisValue for Horizon { - 914
fn axis_name(self) -> &'static str { - 915
self.as_str() - 916
} - 917
- 918
fn axis_rank(self) -> Option<u8> { - 919
Some(self.rank()) - 920
} - 921
} - 922
impl AxisValue for Stakes { - 923
fn axis_name(self) -> &'static str { - 924
self.as_str() - 925
} - 926
- 927
fn axis_rank(self) -> Option<u8> { - 928
Some(self.rank()) - 929
} - 930
} - 931
impl AxisValue for Evidence { - 932
fn axis_name(self) -> &'static str { - 933
self.as_str() - 934
} - 935
- 936
fn axis_rank(self) -> Option<u8> { - 937
Some(self.rank()) - 938
} - 939
} - 940
impl AxisValue for Clarity { - 941
fn axis_name(self) -> &'static str { - 942
self.as_str() - 943
} - 944
} - 945
- 946
/// Weighted votes for one axis. - 947
#[derive(Debug, Clone)] - 948
pub struct Votes<T: AxisValue> { - 949
tally: Vec<(T, f64)>, - 950
} - 951
- 952
impl<T: AxisValue> Default for Votes<T> { - 953
fn default() -> Self { - 954
Votes { tally: Vec::new() } - 955
} - 956
} - 957
- 958
impl<T: AxisValue> Votes<T> { - 959
pub fn add(&mut self, value: T, weight: f64) { - 960
match self.tally.iter_mut().find(|(v, _)| *v == value) { - 961
Some((_, score)) => *score += weight, - 962
None => self.tally.push((value, weight)), - 963
} - 964
} - 965
- 966
pub fn is_empty(&self) -> bool { - 967
self.tally.is_empty() - 968
} - 969
- 970
/// Every value that received weight, strongest first. Ties break by name - 971
/// so the ordering is total and stable across runs. - 972
pub fn ranked(&self) -> Vec<(T, f64)> { - 973
let mut ranked = self.tally.clone(); - 974
ranked.sort_by(|a, b| { - 975
b.1.partial_cmp(&a.1) - 976
.unwrap_or(std::cmp::Ordering::Equal) - 977
.then_with(|| a.0.axis_name().cmp(b.0.axis_name())) - 978
}); - 979
ranked - 980
} - 981
- 982
/// Minimum weight before a signal may escalate an ordered axis. Filters - 983
/// out the incidental 0.3-weight hints so a stray word cannot promote a - 984
/// one-liner to durable multi-day work. - 985
pub(crate) const ESCALATION_FLOOR: f64 = 0.5; - 986
- 987
/// Every value scoring within `band` of the winner, strongest first. - 988
/// - 989
/// Used for the act axis, where two readings being close is often not - 990
/// ambiguity to resolve but a request that genuinely spans both: "fix the - 991
/// failing test" is a `modify` and a `verify`, and picking one loses the - 992
/// other's tools. - 993
pub fn contenders(&self, band: f64) -> Vec<T> { - 994
let ranked = self.ranked(); - 995
let Some((_, best)) = ranked.first().copied() else { - 996
return Vec::new(); - 997
}; - 998
if best <= 0.0 { - 999
return Vec::new(); - 1000
} - 1001
ranked - 1002
.into_iter() - 1003
.filter(|(_, weight)| { - 1004
// The winner is always retained. Others must clear the - 1005
// absolute floor, and either sit within the band of the - 1006
// winner or carry strong independent signal. - 1007
*weight >= best - 1008
|| (*weight >= Self::ESCALATION_FLOOR - 1009
&& (*weight >= best * (1.0 - band) || *weight >= 1.0)) - 1010
}) - 1011
.map(|(value, _)| value) - 1012
.collect() - 1013
} - 1014
- 1015
/// Winner and a [0,1] confidence. - 1016
/// - 1017
/// Two rules, because the axes are not all the same shape: - 1018
/// - 1019
/// * **Categorical** (`Act`, `Clarity`): argmax with confidence from the - 1020
/// margin over the runner-up. - 1021
/// * **Ordered** (`Stakes`, `Horizon`, `Evidence`): the highest level with - 1022
/// real support wins, and lower levels corroborate rather than compete. - 1023
/// Taking the maximum is also the safe direction on every ordered axis - 1024
/// here — more caution, a stricter proof standard, a longer horizon. - 1025
pub fn winner(&self) -> Option<(T, f64)> { - 1026
let ranked = self.ranked(); - 1027
let (best, best_score) = ranked.first().copied()?; - 1028
if best_score <= 0.0 { - 1029
return None; - 1030
} - 1031
if best.axis_rank().is_some() { - 1032
// Everything below the escalation floor abstains: a vote too weak - 1033
// to escalate on its own is not evidence of the level it names. - 1034
let (highest, weight) = ranked - 1035
.iter() - 1036
.filter(|(_, weight)| *weight >= Self::ESCALATION_FLOOR) - 1037
.max_by_key(|(value, _)| value.axis_rank().unwrap_or(0)) - 1038
.copied()?; - 1039
let mass = (weight / 1.5).min(1.0); - 1040
return Some((highest, (0.7 + mass * 0.3).clamp(0.0, 1.0))); - 1041
} - 1042
let runner_up = ranked.get(1).map(|(_, s)| *s).unwrap_or(0.0); - 1043
// How much better the winner is, as a fraction of itself — not its - 1044
// share of the total. Share-of-total punishes any second signal - 1045
// regardless of how weak. - 1046
let margin = 1.0 - (runner_up / best_score).min(1.0); - 1047
// A lone weak signal should not read as certainty either, so the - 1048
// margin is damped by how much evidence there was at all. - 1049
let mass = (best_score / 1.5).min(1.0); - 1050
Some((best, (margin * 0.7 + mass * 0.3).clamp(0.0, 1.0))) - 1051
} - 1052
} - 1053
- 1054
/// Everything tier 1 concluded about one part of a request, before the - 1055
/// resolver turns it into a reading. - 1056
#[derive(Debug, Clone, Default)] - 1057
pub struct Extraction { - 1058
pub signals: Vec<Signal>, - 1059
pub act: Votes<Act>, - 1060
pub horizon: Votes<Horizon>, - 1061
/// Stakes stated by the request's own words. Applied only to effectful - 1062
/// acts: a question about production has no blast radius. - 1063
pub stakes_from_words: Votes<Stakes>, - 1064
/// Stakes implied by the *environment* rather than by the request. - 1065
/// - 1066
/// Kept separate because it is only relevant to acts that actually touch - 1067
/// something: a dirty working tree makes an edit riskier, and says nothing - 1068
/// whatsoever about the risk of saying hello. - 1069
pub stakes_from_environment: Votes<Stakes>, - 1070
pub evidence: Votes<Evidence>, - 1071
pub clarity: Votes<Clarity>, - 1072
pub input_modalities: Vec<Modality>, - 1073
pub output_modalities: Vec<Modality>, - 1074
pub attendance: Attendance, - 1075
pub domains: Vec<String>, - 1076
/// Domains implied by the *environment* rather than by the request's own - 1077
/// words. A git repository says nothing about the subject of a question - 1078
/// asked inside it, so these apply only to effectful acts. - 1079
pub domains_from_environment: Vec<String>, - 1080
/// The strongest temporal reference to a current value, if any. The - 1081
/// resolver turns it into the `live-data` domain only when the act asks - 1082
/// for a fact; "refactor the current implementation" wants no retrieval. - 1083
pub recency: Option<(String, f64)>, - 1084
/// The part points at something it does not contain ("it", "that"). - 1085
pub deictic: bool, - 1086
} - 1087
- 1088
// ------------------------------------------------------------- tokens --- - 1089
- 1090
/// Split into lowercase alphanumeric words, preserving order. - 1091
fn words(text: &str) -> Vec<String> { - 1092
text.to_ascii_lowercase() - 1093
.split(|c: char| !c.is_ascii_alphanumeric()) - 1094
.filter(|w| !w.is_empty()) - 1095
.map(|w| w.to_string()) - 1096
.collect() - 1097
} - 1098
- 1099
/// Crude English de-inflection, enough to match a lexicon of bare verbs. - 1100
/// - 1101
/// Not a stemmer and not trying to be one: it only strips the suffixes that - 1102
/// make a request's actual verb invisible to an exact-match lexicon. - 1103
fn stems(word: &str) -> Vec<&str> { - 1104
let mut out = vec![word]; - 1105
for suffix in ["ing", "ed", "es", "s"] { - 1106
if let Some(stem) = word.strip_suffix(suffix) - 1107
&& stem.len() >= 3 - 1108
{ - 1109
out.push(stem); - 1110
// Doubled consonant before -ing/-ed: "running" -> "runn" -> "run". - 1111
if matches!(suffix, "ing" | "ed") - 1112
&& let Some(trimmed) = stem.strip_suffix(stem.chars().last().unwrap_or(' ')) - 1113
&& trimmed.len() >= 3 - 1114
&& stem.chars().rev().nth(1) == stem.chars().last() - 1115
{ - 1116
out.push(trimmed); - 1117
} - 1118
} - 1119
} - 1120
out - 1121
} - 1122
- 1123
/// A lexicon word with its de-inflected forms, computed once. - 1124
struct Target { - 1125
word: &'static str, - 1126
stems: Vec<&'static str>, - 1127
} - 1128
- 1129
impl Target { - 1130
fn new(word: &'static str) -> Self { - 1131
Target { - 1132
word, - 1133
stems: stems(word), - 1134
} - 1135
} - 1136
} - 1137
- 1138
/// A clause's tokens with their de-inflected forms, computed once, so - 1139
/// matching the whole lexicon against a clause allocates nothing per pair. - 1140
struct Tokens { - 1141
words: Vec<String>, - 1142
} - 1143
- 1144
impl Tokens { - 1145
fn new(text: &str) -> Self { - 1146
Tokens { words: words(text) } - 1147
} - 1148
- 1149
fn len(&self) -> usize { - 1150
self.words.len() - 1151
} - 1152
- 1153
fn get(&self, index: usize) -> Option<&str> { - 1154
self.words.get(index).map(String::as_str) - 1155
} - 1156
- 1157
/// Whether the token at `index` is `target`, allowing inflection and - 1158
/// silent-`e` alterations ("writing" → "write", "deploying" → "deploy"). - 1159
fn matches(&self, index: usize, target: &Target) -> bool { - 1160
let Some(token) = self.get(index) else { - 1161
return false; - 1162
}; - 1163
if token == target.word { - 1164
return true; - 1165
} - 1166
let token_stems = stems(token); - 1167
if token_stems.contains(&target.word) { - 1168
return true; - 1169
} - 1170
if let Some(without_e) = target.word.strip_suffix('e') - 1171
&& token_stems.contains(&without_e) - 1172
{ - 1173
return true; - 1174
} - 1175
for target_stem in &target.stems { - 1176
if token == *target_stem || token_stems.contains(target_stem) { - 1177
return true; - 1178
} - 1179
if token.strip_suffix('e') == Some(*target_stem) { - 1180
return true; - 1181
} - 1182
} - 1183
false - 1184
} - 1185
- 1186
/// First position matching `target`, skipping consumed tokens. - 1187
fn position(&self, target: &Target, consumed: &[bool]) -> Option<usize> { - 1188
(0..self.len()) - 1189
.find(|&i| !consumed.get(i).copied().unwrap_or(false) && self.matches(i, target)) - 1190
} - 1191
- 1192
/// Whether `phrase` (whole words) occurs, and where. - 1193
fn phrase_at(&self, phrase: &[&str]) -> Option<usize> { - 1194
if phrase.is_empty() || phrase.len() > self.len() { - 1195
return None; - 1196
} - 1197
(0..=self.len() - phrase.len()) - 1198
.find(|&start| (0..phrase.len()).all(|i| self.words[start + i] == phrase[i])) - 1199
} - 1200
} - 1201
- 1202
static VERB_TARGETS: LazyLock<Vec<(Target, Act, f64)>> = LazyLock::new(|| { - 1203
ACT_VERBS - 1204
.iter() - 1205
.map(|(word, act, weight)| (Target::new(word), *act, *weight)) - 1206
.collect() - 1207
}); - 1208
- 1209
fn is_lexicon_verb(tokens: &Tokens, index: usize) -> bool { - 1210
VERB_TARGETS - 1211
.iter() - 1212
.any(|(target, _, _)| tokens.matches(index, target)) - 1213
} - 1214
- 1215
fn split_phrase(phrase: &'static str) -> Vec<&'static str> { - 1216
phrase.split(' ').collect() - 1217
} - 1218
- 1219
// ------------------------------------------------------------ clauses --- - 1220
- 1221
/// What a clause is doing. - 1222
#[derive(Debug, Clone, Copy, PartialEq, Eq)] - 1223
pub(crate) enum ClauseKind { - 1224
/// A verb heads it: "fix the parser", "every day, check the bill". - 1225
Imperative, - 1226
/// Asks something: a wh-word or an auxiliary before a subject, or a - 1227
/// trailing question mark. - 1228
Question, - 1229
/// A request phrased as a wish ("I want the parser refactored"), or one - 1230
/// led by a verb the lexicon does not know ("sync the files daily"). - 1231
Request, - 1232
/// Greetings and thanks and nothing else. - 1233
Social, - 1234
/// Describes the situation: "it crashes on empty input". - 1235
Statement, - 1236
} - 1237
- 1238
impl ClauseKind { - 1239
/// Whether this clause asks for work, and so may stand as a strand. - 1240
pub(crate) fn is_work(self) -> bool { - 1241
matches!( - 1242
self, - 1243
ClauseKind::Imperative | ClauseKind::Question | ClauseKind::Request - 1244
) - 1245
} - 1246
} - 1247
- 1248
/// One clause, read once. - 1249
pub(crate) struct ClauseRead { - 1250
pub text: String, - 1251
pub boundary: crate::strand::Boundary, - 1252
pub kind: ClauseKind, - 1253
tokens: Tokens, - 1254
/// Tokens a social phrase already accounted for. - 1255
consumed: Vec<bool>, - 1256
/// Converse weight from greetings stripped off the front. - 1257
greeting: f64, - 1258
/// Signals for what was stripped, for the explain view. - 1259
stripped: Vec<String>, - 1260
question_mark: bool, - 1261
} - 1262
- 1263
impl ClauseRead { - 1264
/// Whether this clause asks for work of its own, and so starts a part: a - 1265
/// question, or an instruction or wish that names something to do. - 1266
/// Everything else is context for the part beside it. - 1267
pub(crate) fn starts_part(&self) -> bool { - 1268
match self.kind { - 1269
ClauseKind::Question => true, - 1270
ClauseKind::Imperative | ClauseKind::Request => self.has_act_votes(), - 1271
ClauseKind::Social | ClauseKind::Statement => false, - 1272
} - 1273
} - 1274
- 1275
/// Whether this clause votes for any act at all. - 1276
fn has_act_votes(&self) -> bool { - 1277
match self.kind { - 1278
ClauseKind::Statement | ClauseKind::Social => { - 1279
(1..self.tokens.len()).any(|i| self.is_head(i) && is_lexicon_verb(&self.tokens, i)) - 1280
} - 1281
_ => (0..self.tokens.len()) - 1282
.any(|i| !self.consumed[i] && is_lexicon_verb(&self.tokens, i)), - 1283
} - 1284
} - 1285
- 1286
/// Whether the token at `index` heads its (sub-)clause: first word, or - 1287
/// right after a conjunction or sequencing word, or after punctuation. - 1288
fn is_head(&self, index: usize) -> bool { - 1289
if index == 0 { - 1290
return true; - 1291
} - 1292
if let Some(prev) = self.tokens.get(index - 1) - 1293
&& HEAD_PREDECESSORS.contains(&prev) - 1294
{ - 1295
return true; - 1296
} - 1297
if index > 1 - 1298
&& self.tokens.get(index - 2) == Some("and") - 1299
&& self.tokens.get(index - 1) == Some("then") - 1300
{ - 1301
return true; - 1302
} - 1303
let Some(target) = self.tokens.get(index) else { - 1304
return false; - 1305
}; - 1306
let lower = self.text.to_ascii_lowercase(); - 1307
for (at, _) in lower.match_indices(target) { - 1308
let before = &lower[..at]; - 1309
let after = &lower[at + target.len()..]; - 1310
let word_start = - 1311
before.is_empty() || before.ends_with(|c: char| !c.is_ascii_alphanumeric()); - 1312
let word_end = - 1313
after.is_empty() || after.starts_with(|c: char| !c.is_ascii_alphanumeric()); - 1314
if word_start && word_end && before.trim_end().ends_with([',', ';', ':', '.', '\n']) { - 1315
return true; - 1316
} - 1317
} - 1318
false - 1319
} - 1320
} - 1321
- 1322
fn starts_with_words(tokens: &[String], phrase: &str) -> Option<usize> { - 1323
let needle: Vec<&str> = phrase.split(' ').collect(); - 1324
(tokens.len() >= needle.len() && needle.iter().zip(tokens).all(|(a, b)| a == b)) - 1325
.then_some(needle.len()) - 1326
} - 1327
- 1328
/// Drop `count` leading words from `text`, keeping the original spelling of - 1329
/// what remains. - 1330
fn drop_leading_words(text: &str, count: usize) -> &str { - 1331
let mut seen = 0usize; - 1332
let mut in_word = false; - 1333
for (i, c) in text.char_indices() { - 1334
let is_word = c.is_ascii_alphanumeric(); - 1335
if is_word && !in_word { - 1336
if seen == count { - 1337
return &text[i..]; - 1338
} - 1339
seen += 1; - 1340
} - 1341
in_word = is_word; - 1342
} - 1343
"" - 1344
} - 1345
- 1346
/// Read one clause: strip what leads it, then decide its kind. - 1347
pub(crate) fn read_clause( - 1348
text: &str, - 1349
boundary: crate::strand::Boundary, - 1350
question_mark: bool, - 1351
) -> ClauseRead { - 1352
let mut rest = text.trim(); - 1353
let mut greeting = 0.0f64; - 1354
let mut stripped = Vec::new(); - 1355
let mut acknowledged = false; - 1356
// Greetings, acknowledgements, polite openers and sequencing words, in - 1357
// any order, until none applies. - 1358
loop { - 1359
let tokens = words(rest); - 1360
let Some(first) = tokens.first() else { - 1361
break; - 1362
}; - 1363
if let Some((phrase, weight)) = SOCIAL_PHRASES - 1364
.iter() - 1365
.find(|(phrase, _)| starts_with_words(&tokens, phrase).is_some()) - 1366
.filter(|(phrase, _)| starts_with_words(&tokens, phrase) != Some(tokens.len())) - 1367
{ - 1368
// A social phrase leading a longer clause is a greeting. - 1369
if let Some(n) = starts_with_words(&tokens, phrase) { - 1370
greeting = greeting.max(*weight); - 1371
stripped.push(format!("greeting `{phrase}`")); - 1372
rest = drop_leading_words(rest, n); - 1373
continue; - 1374
} - 1375
} - 1376
if tokens.len() > 1 - 1377
&& let Some((word, weight)) = SOCIAL_WORDS.iter().find(|(word, _)| word == first) - 1378
{ - 1379
greeting = greeting.max(*weight); - 1380
stripped.push(format!("greeting `{word}`")); - 1381
rest = drop_leading_words(rest, 1); - 1382
continue; - 1383
} - 1384
if tokens.len() > 1 && ACKNOWLEDGEMENTS.contains(&first.as_str()) { - 1385
acknowledged = true; - 1386
stripped.push(format!("acknowledgement `{first}`")); - 1387
rest = drop_leading_words(rest, 1); - 1388
continue; - 1389
} - 1390
if let Some(n) = PREAMBLES - 1391
.iter() - 1392
.filter_map(|preamble| starts_with_words(&tokens, preamble)) - 1393
.max() - 1394
.filter(|n| *n < tokens.len()) - 1395
{ - 1396
stripped.push(format!("opener `{}`", tokens[..n].join(" ")));
Indexing the workspace…
Vakyartha documentation is discovering safe artifacts, anchors, and source references.