Skip to main content

magi/
github_text.rs

1//! Posting gate shared by GitHub titles, descriptions and comments.
2use crate::run::RunState;
3use crate::scrub::{Identity, scrub};
4
5/// Fixed fallback for a rejected title.
6pub const NEUTRAL_TITLE: &str = "chore: update repository";
7/// Fixed fallback for a rejected description or comment.
8pub const NEUTRAL_BODY: &str = "Generated GitHub text was withheld by the magi posting gate. Review the branch diff and the local run report for details.";
9
10/// Categories only: diagnostics must never repeat the offending secret.
11#[derive(Debug, Clone, Copy, PartialEq, Eq)]
12pub enum Violation {
13    /// Title contains a meaningful share of non-English letters.
14    TitleLanguage,
15    /// Unquoted body prose contains a meaningful share of non-English letters.
16    BodyLanguage,
17    /// Shared redaction rules found sensitive or local data.
18    SensitiveData,
19}
20
21/// Pure checker. Language exemptions never exempt sensitive data.
22pub fn check(title: &str, body: &str) -> Vec<Violation> {
23    let mut out = Vec::new();
24    if non_english(title) {
25        out.push(Violation::TitleLanguage);
26    }
27    if non_english(&prose(body)) {
28        out.push(Violation::BodyLanguage);
29    }
30    let id = Identity::default();
31    if scrub(title, &id) != title || scrub(body, &id) != body {
32        out.push(Violation::SensitiveData);
33    }
34    out
35}
36
37/// Function words and common verbs of the Latin-script languages most likely
38/// to appear, chosen to avoid ordinary English words.
39const FOREIGN_WORDS: &[&str] = &[
40    "este",
41    "esta",
42    "para",
43    "los",
44    "las",
45    "del",
46    "una",
47    "que",
48    "por",
49    "errores",
50    "corregir",
51    "agrega",
52    "cambio",
53    "solicitudes",
54    "fallidas",
55    "reintentos",
56    "les",
57    "des",
58    "pour",
59    "avec",
60    "dans",
61    "est",
62    "une",
63    "pas",
64    "und",
65    "der",
66    "das",
67    "nicht",
68    "mit",
69    "ein",
70    "eine",
71    "für",
72    "wird",
73    "não",
74    "uma",
75    "della",
76    "che",
77    "con",
78    "fehler",
79    "corrigir",
80    "erreurs",
81    "bitte",
82    "anfrage",
83    "anfragen",
84    "wiederholen",
85    "korrigieren",
86];
87
88/// English function words and everyday development vocabulary. Text whose
89/// words never meet this list is not English, whatever FOREIGN_WORDS knows.
90/// Words spelled the same in German or Dutch (will, die, also) stay out.
91const ENGLISH_WORDS: &[&str] = &[
92    "the",
93    "a",
94    "an",
95    "to",
96    "of",
97    "and",
98    "or",
99    "for",
100    "in",
101    "on",
102    "at",
103    "by",
104    "with",
105    "from",
106    "when",
107    "while",
108    "this",
109    "that",
110    "these",
111    "those",
112    "is",
113    "are",
114    "be",
115    "was",
116    "were",
117    "been",
118    "not",
119    "no",
120    "it",
121    "its",
122    "as",
123    "if",
124    "so",
125    "but",
126    "than",
127    "then",
128    "into",
129    "after",
130    "before",
131    "only",
132    "can",
133    "should",
134    "must",
135    "may",
136    "has",
137    "have",
138    "had",
139    "do",
140    "does",
141    "all",
142    "any",
143    "each",
144    "one",
145    "two",
146    "new",
147    "old",
148    "more",
149    "less",
150    "which",
151    "what",
152    "why",
153    "how",
154    "now",
155    "never",
156    "always",
157    "instead",
158    "without",
159    "within",
160    "over",
161    "under",
162    "per",
163    "via",
164    "we",
165    "you",
166    "they",
167    "there",
168    "their",
169    "your",
170    "our",
171    "use",
172    "used",
173    "uses",
174    "make",
175    "makes",
176    "fix",
177    "add",
178    "update",
179    "remove",
180    "bump",
181    "refactor",
182    "test",
183    "retry",
184    "request",
185    "review",
186    "run",
187    "task",
188    "branch",
189    "merge",
190    "release",
191    "error",
192    "config",
193    "build",
194    "check",
195    "docs",
196    "feat",
197    "chore",
198    "change",
199    "changes",
200    "file",
201    "code",
202    "repository",
203    "repo",
204    "pull",
205    "commit",
206    "message",
207    "text",
208    "title",
209    "body",
210    "github",
211    "gate",
212    "default",
213    "value",
214    "key",
215    "name",
216    "path",
217    "state",
218    "agent",
219    "seat",
220    "fail",
221    "failure",
222    "failed",
223    "pass",
224    "read",
225    "write",
226    "set",
227    "get",
228    "send",
229    "post",
230    "open",
231    "close",
232    "start",
233    "stop",
234    "keep",
235    "drop",
236    "move",
237    "rename",
238    "handle",
239    "support",
240    "allow",
241    "avoid",
242    "ensure",
243    "prevent",
244    "background",
245    "summary",
246    "risk",
247    "risks",
248    "follow",
249    "verify",
250    "hand",
251    "version",
252    "bug",
253    "issue",
254    "finding",
255    "findings",
256    "round",
257    "rounds",
258    "queue",
259    "worktree",
260    "diff",
261    "line",
262    "lines",
263    "word",
264    "words",
265    "list",
266    "case",
267    "cases",
268    "input",
269    "output",
270    "result",
271    "results",
272    "fallback",
273    "replace",
274    "replaced",
275    "withheld",
276    "generated",
277    "neutral",
278    "english",
279    "language",
280    "detect",
281    "detection",
282    "match",
283    "matches",
284    "wrong",
285    "stale",
286    "missing",
287    "extra",
288    "small",
289    "let",
290    "settle",
291    "carry",
292    "note",
293    "help",
294    "colour",
295    "color",
296    "palette",
297    "deputy",
298    "brief",
299    "briefs",
300    "serde",
301    "tokio",
302];
303
304fn english_words(text: &str) -> (usize, usize) {
305    // A conventional-commit prefix (`fix(scope)!:`) says nothing about the
306    // language; drop it from the raw text, before punctuation is flattened.
307    let t = text.trim_start();
308    let kind = t.chars().take_while(|c| c.is_ascii_lowercase()).count();
309    let mut rest = &t[kind..];
310    if kind > 0 && rest.starts_with('(') {
311        rest = rest.find(')').map_or(rest, |i| &rest[i + 1..]);
312    }
313    let rest = rest.strip_prefix('!').unwrap_or(rest);
314    let text = if kind > 0 && rest.starts_with(':') {
315        &rest[1..]
316    } else {
317        text
318    };
319    let cleaned: String = text
320        .chars()
321        .map(|c| {
322            if "()[],;!?\"'*#<>`-=|".contains(c) {
323                ' '
324            } else {
325                c
326            }
327        })
328        .collect();
329    let (mut counted, mut hits) = (0, 0);
330    let tokens = cleaned.split_whitespace();
331    for raw in tokens {
332        let w = raw.trim_matches(|c| c == ':' || c == '.');
333        if w.len() < 2 || !w.chars().all(|c| c.is_ascii_alphabetic()) {
334            continue;
335        }
336        if w.chars().all(|c| c.is_ascii_uppercase())
337            || w.chars().skip(1).any(|c| c.is_ascii_uppercase())
338        {
339            continue;
340        }
341        counted += 1;
342        let w = w.to_ascii_lowercase();
343        let known = |x: &str| ENGLISH_WORDS.contains(&x);
344        let stem = |suffix: &str, add: &str| w.strip_suffix(suffix).map(|b| format!("{b}{add}"));
345        if known(&w)
346            || [
347                stem("ies", "y"),
348                stem("es", ""),
349                stem("s", ""),
350                stem("ed", ""),
351                stem("ed", "e"),
352                stem("ing", ""),
353                stem("ing", "e"),
354            ]
355            .iter()
356            .flatten()
357            .any(|x| known(x))
358        {
359            hits += 1;
360        }
361    }
362    (counted, hits)
363}
364
365/// Word endings that are common in German, Dutch, Spanish and Italian and
366/// rare in English. Only ever applied to long ASCII words (see `foreign_looking`).
367const FOREIGN_SUFFIXES: &[&str] = &[
368    "ieren", "ierung", "ungen", "ung", "keit", "heit", "lich", "zeit", "zeiten", "mente", "zione",
369];
370
371/// Prose shorter than this many counted words cannot be judged by a zero-hit
372/// result alone: a title like `Improve performance` has no word the list knows.
373const PROSE_WORDS: usize = 5;
374
375/// Positive evidence that a short text is not English, without needing a
376/// match against the English list: non-ASCII letters, any known foreign word,
377/// or a long ASCII word with a foreign ending (`Wartezeiten`, `reduzieren`).
378fn foreign_looking(text: &str) -> bool {
379    text.chars().any(|c| c.is_alphabetic() && !c.is_ascii())
380        || text
381            .split(|c: char| !c.is_alphabetic())
382            .filter(|w| !w.is_empty())
383            .map(str::to_lowercase)
384            .any(|w| {
385                FOREIGN_WORDS.contains(&w.as_str())
386                    || (w.len() >= 6
387                        && w.is_ascii()
388                        && FOREIGN_SUFFIXES.iter().any(|s| w.ends_with(s)))
389            })
390}
391
392fn lacks_english(text: &str) -> bool {
393    let (counted, hits) = english_words(text);
394    if counted < PROSE_WORDS {
395        // Too short for "no English word" to mean anything by itself.
396        return (hits == 0 || hits * 2 < counted) && foreign_looking(text);
397    }
398    hits == 0 || hits * 3 < counted
399}
400
401fn foreign_words(text: &str) -> bool {
402    let words: Vec<String> = text
403        .split(|c: char| !c.is_alphabetic())
404        .filter(|w| !w.is_empty())
405        .map(str::to_lowercase)
406        .collect();
407    let hits = words
408        .iter()
409        .filter(|w| FOREIGN_WORDS.contains(&w.as_str()))
410        .count();
411    hits >= 2 && hits * 4 >= words.len()
412}
413
414fn non_english(text: &str) -> bool {
415    if foreign_words(text)
416        || lacks_english(text)
417        || text
418            .split("\n\n")
419            .any(|p| p.split_whitespace().count() >= 5 && lacks_english(p))
420    {
421        return true;
422    }
423    let letters = text.chars().filter(|c| c.is_alphabetic()).count();
424    let foreign = text
425        .chars()
426        .filter(|c| c.is_alphabetic() && !c.is_ascii())
427        .count();
428    // Accents and isolated identifiers are common in otherwise English prose.
429    (foreign >= 4 && foreign * 10 >= letters.max(1))
430        || text.lines().any(|line| {
431            let letters = line.chars().filter(|c| c.is_alphabetic()).count();
432            let foreign = line
433                .chars()
434                .filter(|c| c.is_alphabetic() && !c.is_ascii())
435                .count();
436            foreign >= 4 && foreign * 2 >= letters.max(1)
437        })
438}
439
440/// Offset just past the first backtick run of exactly `n` in `text`, if any.
441/// An unmatched opener is literal text in Markdown, so it exempts nothing.
442fn closing_run(text: &str, n: usize) -> Option<usize> {
443    let mut at = 0;
444    while let Some(i) = text[at..].find('`') {
445        let start = at + i;
446        let len = text[start..].chars().take_while(|c| *c == '`').count();
447        if len == n {
448            return Some(start + len);
449        }
450        at = start + len;
451    }
452    None
453}
454
455fn prose(body: &str) -> String {
456    let mut out = String::new();
457    let mut details = 0usize;
458    let mut quote = false;
459    let mut fence: Option<(char, usize)> = None;
460    for line in body.lines() {
461        let trimmed = line.trim_start();
462        if let Some((marker, width)) = fence {
463            if trimmed.chars().take_while(|c| *c == marker).count() >= width {
464                fence = None;
465            }
466            continue;
467        }
468        let marker = trimmed.chars().next().unwrap_or(' ');
469        let width = trimmed.chars().take_while(|c| *c == marker).count();
470        if matches!(marker, '`' | '~') && width >= 3 {
471            fence = Some((marker, width));
472            continue;
473        }
474        if trimmed.starts_with('>') || line.starts_with("    ") || line.starts_with('\t') {
475            continue;
476        }
477        let mut rest = line;
478        while !rest.is_empty() {
479            if rest.starts_with('<')
480                && let Some(end) = rest.find('>')
481            {
482                let tag = rest[..=end].to_ascii_lowercase();
483                if tag.starts_with("<details") {
484                    details += 1;
485                } else if tag == "</details>" {
486                    details = details.saturating_sub(1);
487                } else if tag.starts_with("<blockquote") {
488                    quote = true;
489                } else if tag == "</blockquote>" {
490                    quote = false;
491                }
492                rest = &rest[end + 1..];
493                continue;
494            }
495            if rest.starts_with('`') {
496                let n = rest.chars().take_while(|c| *c == '`').count();
497                rest = match closing_run(&rest[n..], n) {
498                    Some(end) => &rest[n + end..],
499                    None => &rest[n..],
500                };
501                continue;
502            }
503            let c = rest.chars().next().expect("nonempty");
504            if details == 0 && !quote {
505                out.push(c);
506            }
507            rest = &rest[c.len_utf8()..];
508        }
509        out.push('\n');
510    }
511    out
512}
513
514/// Scrub local identity and pattern matches, then replace failing prose.
515/// Every intervention is recorded without copying the offending material.
516pub fn prepare(state: &mut RunState, title: &str, body: &str) -> (String, String) {
517    let id = Identity::current();
518    let clean_title = scrub(title, &id);
519    let clean_body = scrub(body, &id);
520    let violations = check(title, body);
521    if clean_title != title || clean_body != body {
522        state.event("github-text", "sensitive data removed before posting");
523    }
524    let language = state.config.graph.github_text_guard;
525    let title = if language && violations.contains(&Violation::TitleLanguage) {
526        state.event("github-text", "title replaced with neutral English text");
527        NEUTRAL_TITLE.to_owned()
528    } else {
529        clean_title
530    };
531    let body = if language && violations.contains(&Violation::BodyLanguage) {
532        state.event("github-text", "body replaced with neutral English text");
533        NEUTRAL_BODY.to_owned()
534    } else {
535        clean_body
536    };
537    (title, body)
538}
539
540#[cfg(test)]
541mod tests {
542    use super::*;
543
544    #[test]
545    fn github_text_language_and_exemptions() {
546        assert_eq!(
547            check("fix: retries", "日本語で変更の説明を書きます。"),
548            vec![Violation::BodyLanguage]
549        );
550        assert!(check("fix: retries", "Add retries for failed requests.\n<details>\n<summary>Original task</summary>\n日本語の元の依頼です。\n</details>").is_empty());
551        for body in [
552            "Fix `cache\n\n日本語の説明を書きます。",
553            "Fix `cache 日本語の説明を書きます。",
554        ] {
555            assert_eq!(
556                check("fix: retries", body),
557                vec![Violation::BodyLanguage],
558                "{body}"
559            );
560        }
561        for body in [
562            "Add retries. `日本語の識別子`",
563            "Add retries.\n```text\n日本語のコードです\n```",
564            "Add retries.\n> 日本語の引用です",
565            "Add retries.\n<blockquote>日本語の引用です</blockquote>",
566            "Add retries. ``日本語 ` の識別子``",
567            "Update café names.",
568            "Change src/graph.rs and tests/common/mod.rs.",
569        ] {
570            assert!(check("fix: retries", body).is_empty(), "{body}");
571        }
572        for title in [
573            "chore: release v0.95.0",
574            "feat(deputy): let a settle carry a note",
575            "feat(cli): colour --help in an Evangelion palette",
576            "Refactor deputy briefs",
577            "Bump tokio and serde",
578        ] {
579            assert!(check(title, "").is_empty(), "{title}");
580        }
581        assert!(check("Bitte Anfragen wiederholen", "").contains(&Violation::TitleLanguage));
582        assert!(
583            check("fix: retries", "Bitte Anfragen wiederholen").contains(&Violation::BodyLanguage)
584        );
585        assert!(
586            check(
587                "fix: retries",
588                "Add retries for failed requests.\n\nBitte Anfragen schnell wiederholen heute."
589            )
590            .contains(&Violation::BodyLanguage)
591        );
592        assert!(
593            check("fix: retries", "Zeitweise Sperren lösen Wartezeiten aus")
594                .contains(&Violation::BodyLanguage)
595        );
596    }
597
598    #[test]
599    fn short_english_without_list_hits_passes_but_foreign_short_text_fails() {
600        for text in [
601            "Improve performance",
602            "perf: speed cache",
603            "Trim idle sockets",
604            "Optimize memory consumption",
605            "ci: pin actions",
606            "Quicker warmup sprocket",
607        ] {
608            assert_eq!(english_words(text).1, 0, "{text} must have no list hit");
609            assert!(check(text, "").is_empty(), "{text}");
610            assert!(check("fix: retries", text).is_empty(), "{text}");
611        }
612        for text in [
613            "fix(request): Wartezeiten reduzieren",
614            "Leistung verbessern",
615            "Corrección rápida",
616            "fix: Wartezeiten im request reduzieren",
617            "fix: retries schneller wiederholen",
618            "fix(request): Wartezeiten bei retries reduzieren",
619        ] {
620            assert!(
621                check(text, "").contains(&Violation::TitleLanguage),
622                "{text}"
623            );
624            assert!(
625                check("fix: retries", text).contains(&Violation::BodyLanguage),
626                "{text}"
627            );
628        }
629        // Four zero-hit words are short; five are prose and need a hit.
630        let four = "Quicker warmup sprocket tweak";
631        let five = "Quicker warmup sprocket tweak gizmo";
632        assert_eq!(english_words(four), (4, 0));
633        assert_eq!(english_words(five), (5, 0));
634        assert!(check(four, "").is_empty());
635        assert!(check(five, "").contains(&Violation::TitleLanguage));
636    }
637
638    #[test]
639    fn github_text_sensitive_data_in_all_sections() {
640        for secret in [
641            "/Users/example/repo",
642            "/home/example/repo",
643            "C:\\Users\\Example\\repo",
644            "/private/tmp/work",
645            "dev@example.test",
646            "ghp_abcdefghijklmnopqrstuv",
647            "github_pat_abcdefghijklmnopqrstuv",
648            "sk-abcdefghijklmnopqrstuv",
649            "AKIAABCDEFGHIJKLMNOP",
650            "password=example",
651            "password=\"example\"",
652            "token aBcdEfgHijkLmn0123456789",
653            "buildbox.local",
654            "token=abcdefghijklmnop012345",
655            "Authorization: Bearer abcdefghijklmnop",
656            "hostname=buildbox",
657            "username=example",
658            "10.2.3.4",
659            "fe80::1",
660        ] {
661            assert!(
662                check(secret, "").contains(&Violation::SensitiveData),
663                "{secret}"
664            );
665            assert!(
666                check(
667                    "fix: retries",
668                    &format!("<details>\n`{secret}`\n</details>")
669                )
670                .contains(&Violation::SensitiveData),
671                "{secret}"
672            );
673        }
674        for clean in [
675            "https://github.com/example/repo",
676            "src/graph.rs",
677            "docs/home/example",
678            "/api/v1/runs",
679            "v1.2.3",
680            "std::io::Error",
681            "Use the token from the environment.",
682        ] {
683            assert!(check("fix: retries", clean).is_empty(), "{clean}");
684        }
685    }
686
687    #[test]
688    fn github_text_config_only_disables_language_and_records_interventions() {
689        let mut state = RunState::new(
690            ".".into(),
691            "main".into(),
692            "abc".into(),
693            "task".into(),
694            crate::config::Config::default(),
695        );
696        let (_, body) = prepare(
697            &mut state,
698            "fix: retries",
699            "日本語の説明を書きます。 token=secret",
700        );
701        assert_eq!(body, NEUTRAL_BODY);
702        assert!(!state.events.is_empty());
703        state.config.graph.github_text_guard = false;
704        let (_, body) = prepare(
705            &mut state,
706            "fix: retries",
707            "日本語の説明を書きます。 token=secret",
708        );
709        assert!(body.contains("日本語"));
710        assert!(!body.contains("secret"));
711        assert!(
712            check("fix: retries", &body)
713                .iter()
714                .all(|v| *v != Violation::SensitiveData)
715        );
716    }
717
718    #[test]
719    fn github_text_fixed_fallback_passes() {
720        assert!(check(NEUTRAL_TITLE, NEUTRAL_BODY).is_empty());
721    }
722}
723
724#[cfg(test)]
725mod review_round_tests {
726    use super::*;
727
728    #[test]
729    fn latin_script_non_english_is_flagged() {
730        assert!(check("Corregir errores", "fix: retry").contains(&Violation::TitleLanguage));
731        assert!(
732            check(
733                "fix: retries",
734                "Este cambio agrega reintentos para solicitudes fallidas."
735            )
736            .contains(&Violation::BodyLanguage)
737        );
738        assert!(
739            check(
740                "fix: retry failed requests",
741                "Adds retries for failed requests."
742            )
743            .is_empty()
744        );
745    }
746
747    #[test]
748    fn quoted_json_credentials_are_sensitive() {
749        let body = "Example: {\"password\": \"hunter2\"}";
750        assert!(check("t", body).contains(&Violation::SensitiveData));
751        assert!(!crate::scrub::scrub(body, &Identity::default()).contains("hunter2"));
752    }
753}
754
755#[cfg(test)]
756mod quoted_value_tests {
757    use crate::scrub::{Identity, scrub};
758
759    #[test]
760    fn quoted_values_are_redacted_whole() {
761        for v in ["correct horse battery staple", ",hunter2"] {
762            let out = scrub(
763                &format!("{{\"password\": \"{v}\"}} ok"),
764                &Identity::default(),
765            );
766            assert!(!out.contains("horse") && !out.contains("hunter2"), "{out}");
767            assert!(out.ends_with("ok"), "{out}");
768        }
769    }
770}