Skip to main content

magi/
github_text.rs

1//! Posting gate shared by GitHub titles, descriptions and comments.
2use crate::run::RunState;
3use crate::scrub::{Identity, scrub};
4
5/// Fixed fallback for a rejected title.
6pub const NEUTRAL_TITLE: &str = "chore: update repository";
7/// Fixed fallback for a rejected description or comment.
8pub const NEUTRAL_BODY: &str = "Generated GitHub text was withheld by the magi posting gate. Review the branch diff and the local run report for details.";
9
10/// Categories only: diagnostics must never repeat the offending secret.
11#[derive(Debug, Clone, Copy, PartialEq, Eq)]
12pub enum Violation {
13    /// Title contains a meaningful share of non-English letters.
14    TitleLanguage,
15    /// Unquoted body prose contains a meaningful share of non-English letters.
16    BodyLanguage,
17    /// Shared redaction rules found sensitive or local data.
18    SensitiveData,
19}
20
21/// Pure checker. Language exemptions never exempt sensitive data.
22pub fn check(title: &str, body: &str) -> Vec<Violation> {
23    let mut out = Vec::new();
24    if non_english(title) {
25        out.push(Violation::TitleLanguage);
26    }
27    if non_english(&prose(body)) {
28        out.push(Violation::BodyLanguage);
29    }
30    let id = Identity::default();
31    if scrub(title, &id) != title || scrub(body, &id) != body {
32        out.push(Violation::SensitiveData);
33    }
34    out
35}
36
37/// Function words and common verbs of the Latin-script languages most likely
38/// to appear, chosen to avoid ordinary English words.
39const FOREIGN_WORDS: &[&str] = &[
40    "este",
41    "esta",
42    "para",
43    "los",
44    "las",
45    "del",
46    "una",
47    "que",
48    "por",
49    "errores",
50    "corregir",
51    "agrega",
52    "cambio",
53    "solicitudes",
54    "fallidas",
55    "reintentos",
56    "les",
57    "des",
58    "pour",
59    "avec",
60    "dans",
61    "est",
62    "une",
63    "pas",
64    "und",
65    "der",
66    "das",
67    "nicht",
68    "mit",
69    "ein",
70    "eine",
71    "für",
72    "wird",
73    "não",
74    "uma",
75    "della",
76    "che",
77    "con",
78    "fehler",
79    "corrigir",
80    "erreurs",
81    "bitte",
82    "anfrage",
83    "anfragen",
84    "wiederholen",
85    "korrigieren",
86];
87
88/// English function words and everyday development vocabulary. Text whose
89/// words never meet this list is not English, whatever FOREIGN_WORDS knows.
90/// Words spelled the same in German or Dutch (will, die, also) stay out.
91const ENGLISH_WORDS: &[&str] = &[
92    "the",
93    "a",
94    "an",
95    "to",
96    "of",
97    "and",
98    "or",
99    "for",
100    "in",
101    "on",
102    "at",
103    "by",
104    "with",
105    "from",
106    "when",
107    "while",
108    "this",
109    "that",
110    "these",
111    "those",
112    "is",
113    "are",
114    "be",
115    "was",
116    "were",
117    "been",
118    "not",
119    "no",
120    "it",
121    "its",
122    "as",
123    "if",
124    "so",
125    "but",
126    "than",
127    "then",
128    "into",
129    "after",
130    "before",
131    "only",
132    "can",
133    "should",
134    "must",
135    "may",
136    "has",
137    "have",
138    "had",
139    "do",
140    "does",
141    "all",
142    "any",
143    "each",
144    "one",
145    "two",
146    "new",
147    "old",
148    "more",
149    "less",
150    "which",
151    "what",
152    "why",
153    "how",
154    "now",
155    "never",
156    "always",
157    "instead",
158    "without",
159    "within",
160    "over",
161    "under",
162    "per",
163    "via",
164    "we",
165    "you",
166    "they",
167    "there",
168    "their",
169    "your",
170    "our",
171    "use",
172    "used",
173    "uses",
174    "make",
175    "makes",
176    "fix",
177    "add",
178    "update",
179    "remove",
180    "bump",
181    "refactor",
182    "test",
183    "retry",
184    "request",
185    "review",
186    "run",
187    "task",
188    "branch",
189    "merge",
190    "release",
191    "error",
192    "config",
193    "build",
194    "check",
195    "docs",
196    "feat",
197    "chore",
198    "change",
199    "changes",
200    "file",
201    "code",
202    "repository",
203    "repo",
204    "pull",
205    "commit",
206    "message",
207    "text",
208    "title",
209    "body",
210    "github",
211    "gate",
212    "default",
213    "value",
214    "key",
215    "name",
216    "path",
217    "state",
218    "agent",
219    "seat",
220    "fail",
221    "failure",
222    "failed",
223    "pass",
224    "read",
225    "write",
226    "set",
227    "get",
228    "send",
229    "post",
230    "open",
231    "close",
232    "start",
233    "stop",
234    "keep",
235    "drop",
236    "move",
237    "rename",
238    "handle",
239    "support",
240    "allow",
241    "avoid",
242    "ensure",
243    "prevent",
244    "background",
245    "summary",
246    "risk",
247    "risks",
248    "follow",
249    "verify",
250    "hand",
251    "version",
252    "bug",
253    "issue",
254    "finding",
255    "findings",
256    "round",
257    "rounds",
258    "queue",
259    "worktree",
260    "diff",
261    "line",
262    "lines",
263    "word",
264    "words",
265    "list",
266    "case",
267    "cases",
268    "input",
269    "output",
270    "result",
271    "results",
272    "fallback",
273    "replace",
274    "replaced",
275    "withheld",
276    "generated",
277    "neutral",
278    "english",
279    "language",
280    "detect",
281    "detection",
282    "match",
283    "matches",
284    "wrong",
285    "stale",
286    "missing",
287    "extra",
288    "small",
289    "let",
290    "settle",
291    "carry",
292    "note",
293    "help",
294    "colour",
295    "color",
296    "palette",
297    "deputy",
298    "brief",
299    "briefs",
300    "serde",
301    "tokio",
302];
303
304fn english_words(text: &str) -> (usize, usize) {
305    // A conventional-commit prefix (`fix(scope)!:`) says nothing about the
306    // language; drop it from the raw text, before punctuation is flattened.
307    let t = text.trim_start();
308    let kind = t.chars().take_while(|c| c.is_ascii_lowercase()).count();
309    let mut rest = &t[kind..];
310    if kind > 0 && rest.starts_with('(') {
311        rest = rest.find(')').map_or(rest, |i| &rest[i + 1..]);
312    }
313    let rest = rest.strip_prefix('!').unwrap_or(rest);
314    let text = if kind > 0 && rest.starts_with(':') {
315        &rest[1..]
316    } else {
317        text
318    };
319    let cleaned: String = text
320        .chars()
321        .map(|c| {
322            if "()[],;!?\"'*#<>`-=|".contains(c) {
323                ' '
324            } else {
325                c
326            }
327        })
328        .collect();
329    let (mut counted, mut hits) = (0, 0);
330    let tokens = cleaned.split_whitespace();
331    for raw in tokens {
332        let w = raw.trim_matches(|c| c == ':' || c == '.');
333        if w.len() < 2 || !w.chars().all(|c| c.is_ascii_alphabetic()) {
334            continue;
335        }
336        if w.chars().all(|c| c.is_ascii_uppercase())
337            || w.chars().skip(1).any(|c| c.is_ascii_uppercase())
338        {
339            continue;
340        }
341        counted += 1;
342        let w = w.to_ascii_lowercase();
343        let known = |x: &str| ENGLISH_WORDS.contains(&x);
344        let stem = |suffix: &str, add: &str| w.strip_suffix(suffix).map(|b| format!("{b}{add}"));
345        if known(&w)
346            || [
347                stem("ies", "y"),
348                stem("es", ""),
349                stem("s", ""),
350                stem("ed", ""),
351                stem("ed", "e"),
352                stem("ing", ""),
353                stem("ing", "e"),
354            ]
355            .iter()
356            .flatten()
357            .any(|x| known(x))
358        {
359            hits += 1;
360        }
361    }
362    (counted, hits)
363}
364
365fn lacks_english(text: &str) -> bool {
366    let (counted, hits) = english_words(text);
367    (counted >= 2 && hits == 0) || (counted >= 4 && hits * 3 < counted)
368}
369
370fn foreign_words(text: &str) -> bool {
371    let words: Vec<String> = text
372        .split(|c: char| !c.is_alphabetic())
373        .filter(|w| !w.is_empty())
374        .map(str::to_lowercase)
375        .collect();
376    let hits = words
377        .iter()
378        .filter(|w| FOREIGN_WORDS.contains(&w.as_str()))
379        .count();
380    hits >= 2 && hits * 4 >= words.len()
381}
382
383fn non_english(text: &str) -> bool {
384    if foreign_words(text)
385        || lacks_english(text)
386        || text
387            .split("\n\n")
388            .any(|p| p.split_whitespace().count() >= 5 && lacks_english(p))
389    {
390        return true;
391    }
392    let letters = text.chars().filter(|c| c.is_alphabetic()).count();
393    let foreign = text
394        .chars()
395        .filter(|c| c.is_alphabetic() && !c.is_ascii())
396        .count();
397    // Accents and isolated identifiers are common in otherwise English prose.
398    (foreign >= 4 && foreign * 10 >= letters.max(1))
399        || text.lines().any(|line| {
400            let letters = line.chars().filter(|c| c.is_alphabetic()).count();
401            let foreign = line
402                .chars()
403                .filter(|c| c.is_alphabetic() && !c.is_ascii())
404                .count();
405            foreign >= 4 && foreign * 2 >= letters.max(1)
406        })
407}
408
409/// Offset just past the first backtick run of exactly `n` in `text`, if any.
410/// An unmatched opener is literal text in Markdown, so it exempts nothing.
411fn closing_run(text: &str, n: usize) -> Option<usize> {
412    let mut at = 0;
413    while let Some(i) = text[at..].find('`') {
414        let start = at + i;
415        let len = text[start..].chars().take_while(|c| *c == '`').count();
416        if len == n {
417            return Some(start + len);
418        }
419        at = start + len;
420    }
421    None
422}
423
424fn prose(body: &str) -> String {
425    let mut out = String::new();
426    let mut details = 0usize;
427    let mut quote = false;
428    let mut fence: Option<(char, usize)> = None;
429    for line in body.lines() {
430        let trimmed = line.trim_start();
431        if let Some((marker, width)) = fence {
432            if trimmed.chars().take_while(|c| *c == marker).count() >= width {
433                fence = None;
434            }
435            continue;
436        }
437        let marker = trimmed.chars().next().unwrap_or(' ');
438        let width = trimmed.chars().take_while(|c| *c == marker).count();
439        if matches!(marker, '`' | '~') && width >= 3 {
440            fence = Some((marker, width));
441            continue;
442        }
443        if trimmed.starts_with('>') || line.starts_with("    ") || line.starts_with('\t') {
444            continue;
445        }
446        let mut rest = line;
447        while !rest.is_empty() {
448            if rest.starts_with('<')
449                && let Some(end) = rest.find('>')
450            {
451                let tag = rest[..=end].to_ascii_lowercase();
452                if tag.starts_with("<details") {
453                    details += 1;
454                } else if tag == "</details>" {
455                    details = details.saturating_sub(1);
456                } else if tag.starts_with("<blockquote") {
457                    quote = true;
458                } else if tag == "</blockquote>" {
459                    quote = false;
460                }
461                rest = &rest[end + 1..];
462                continue;
463            }
464            if rest.starts_with('`') {
465                let n = rest.chars().take_while(|c| *c == '`').count();
466                rest = match closing_run(&rest[n..], n) {
467                    Some(end) => &rest[n + end..],
468                    None => &rest[n..],
469                };
470                continue;
471            }
472            let c = rest.chars().next().expect("nonempty");
473            if details == 0 && !quote {
474                out.push(c);
475            }
476            rest = &rest[c.len_utf8()..];
477        }
478        out.push('\n');
479    }
480    out
481}
482
483/// Scrub local identity and pattern matches, then replace failing prose.
484/// Every intervention is recorded without copying the offending material.
485pub fn prepare(state: &mut RunState, title: &str, body: &str) -> (String, String) {
486    let id = Identity::current();
487    let clean_title = scrub(title, &id);
488    let clean_body = scrub(body, &id);
489    let violations = check(title, body);
490    if clean_title != title || clean_body != body {
491        state.event("github-text", "sensitive data removed before posting");
492    }
493    let language = state.config.graph.github_text_guard;
494    let title = if language && violations.contains(&Violation::TitleLanguage) {
495        state.event("github-text", "title replaced with neutral English text");
496        NEUTRAL_TITLE.to_owned()
497    } else {
498        clean_title
499    };
500    let body = if language && violations.contains(&Violation::BodyLanguage) {
501        state.event("github-text", "body replaced with neutral English text");
502        NEUTRAL_BODY.to_owned()
503    } else {
504        clean_body
505    };
506    (title, body)
507}
508
509#[cfg(test)]
510mod tests {
511    use super::*;
512
513    #[test]
514    fn github_text_language_and_exemptions() {
515        assert_eq!(
516            check("fix: retries", "日本語で変更の説明を書きます。"),
517            vec![Violation::BodyLanguage]
518        );
519        assert!(check("fix: retries", "Add retries for failed requests.\n<details>\n<summary>Original task</summary>\n日本語の元の依頼です。\n</details>").is_empty());
520        for body in [
521            "Fix `cache\n\n日本語の説明を書きます。",
522            "Fix `cache 日本語の説明を書きます。",
523        ] {
524            assert_eq!(
525                check("fix: retries", body),
526                vec![Violation::BodyLanguage],
527                "{body}"
528            );
529        }
530        for body in [
531            "Add retries. `日本語の識別子`",
532            "Add retries.\n```text\n日本語のコードです\n```",
533            "Add retries.\n> 日本語の引用です",
534            "Add retries.\n<blockquote>日本語の引用です</blockquote>",
535            "Add retries. ``日本語 ` の識別子``",
536            "Update café names.",
537            "Change src/graph.rs and tests/common/mod.rs.",
538        ] {
539            assert!(check("fix: retries", body).is_empty(), "{body}");
540        }
541        for title in [
542            "chore: release v0.95.0",
543            "feat(deputy): let a settle carry a note",
544            "feat(cli): colour --help in an Evangelion palette",
545            "Refactor deputy briefs",
546            "Bump tokio and serde",
547        ] {
548            assert!(check(title, "").is_empty(), "{title}");
549        }
550        assert!(check("Bitte Anfragen wiederholen", "").contains(&Violation::TitleLanguage));
551        assert!(
552            check("fix: retries", "Bitte Anfragen wiederholen").contains(&Violation::BodyLanguage)
553        );
554        assert!(
555            check(
556                "fix: retries",
557                "Add retries for failed requests.\n\nBitte Anfragen schnell wiederholen heute."
558            )
559            .contains(&Violation::BodyLanguage)
560        );
561        assert!(
562            check("fix: retries", "Zeitweise Sperren lösen Wartezeiten aus")
563                .contains(&Violation::BodyLanguage)
564        );
565    }
566
567    #[test]
568    fn github_text_sensitive_data_in_all_sections() {
569        for secret in [
570            "/Users/example/repo",
571            "/home/example/repo",
572            "C:\\Users\\Example\\repo",
573            "/private/tmp/work",
574            "dev@example.test",
575            "ghp_abcdefghijklmnopqrstuv",
576            "github_pat_abcdefghijklmnopqrstuv",
577            "sk-abcdefghijklmnopqrstuv",
578            "AKIAABCDEFGHIJKLMNOP",
579            "password=example",
580            "password=\"example\"",
581            "token aBcdEfgHijkLmn0123456789",
582            "buildbox.local",
583            "token=abcdefghijklmnop012345",
584            "Authorization: Bearer abcdefghijklmnop",
585            "hostname=buildbox",
586            "username=example",
587            "10.2.3.4",
588            "fe80::1",
589        ] {
590            assert!(
591                check(secret, "").contains(&Violation::SensitiveData),
592                "{secret}"
593            );
594            assert!(
595                check(
596                    "fix: retries",
597                    &format!("<details>\n`{secret}`\n</details>")
598                )
599                .contains(&Violation::SensitiveData),
600                "{secret}"
601            );
602        }
603        for clean in [
604            "https://github.com/example/repo",
605            "src/graph.rs",
606            "docs/home/example",
607            "/api/v1/runs",
608            "v1.2.3",
609            "std::io::Error",
610            "Use the token from the environment.",
611        ] {
612            assert!(check("fix: retries", clean).is_empty(), "{clean}");
613        }
614    }
615
616    #[test]
617    fn github_text_config_only_disables_language_and_records_interventions() {
618        let mut state = RunState::new(
619            ".".into(),
620            "main".into(),
621            "abc".into(),
622            "task".into(),
623            crate::config::Config::default(),
624        );
625        let (_, body) = prepare(
626            &mut state,
627            "fix: retries",
628            "日本語の説明を書きます。 token=secret",
629        );
630        assert_eq!(body, NEUTRAL_BODY);
631        assert!(!state.events.is_empty());
632        state.config.graph.github_text_guard = false;
633        let (_, body) = prepare(
634            &mut state,
635            "fix: retries",
636            "日本語の説明を書きます。 token=secret",
637        );
638        assert!(body.contains("日本語"));
639        assert!(!body.contains("secret"));
640        assert!(
641            check("fix: retries", &body)
642                .iter()
643                .all(|v| *v != Violation::SensitiveData)
644        );
645    }
646
647    #[test]
648    fn github_text_fixed_fallback_passes() {
649        assert!(check(NEUTRAL_TITLE, NEUTRAL_BODY).is_empty());
650    }
651}
652
653#[cfg(test)]
654mod review_round_tests {
655    use super::*;
656
657    #[test]
658    fn latin_script_non_english_is_flagged() {
659        assert!(check("Corregir errores", "fix: retry").contains(&Violation::TitleLanguage));
660        assert!(
661            check(
662                "fix: retries",
663                "Este cambio agrega reintentos para solicitudes fallidas."
664            )
665            .contains(&Violation::BodyLanguage)
666        );
667        assert!(
668            check(
669                "fix: retry failed requests",
670                "Adds retries for failed requests."
671            )
672            .is_empty()
673        );
674    }
675
676    #[test]
677    fn quoted_json_credentials_are_sensitive() {
678        let body = "Example: {\"password\": \"hunter2\"}";
679        assert!(check("t", body).contains(&Violation::SensitiveData));
680        assert!(!crate::scrub::scrub(body, &Identity::default()).contains("hunter2"));
681    }
682}
683
684#[cfg(test)]
685mod quoted_value_tests {
686    use crate::scrub::{Identity, scrub};
687
688    #[test]
689    fn quoted_values_are_redacted_whole() {
690        for v in ["correct horse battery staple", ",hunter2"] {
691            let out = scrub(
692                &format!("{{\"password\": \"{v}\"}} ok"),
693                &Identity::default(),
694            );
695            assert!(!out.contains("horse") && !out.contains("hunter2"), "{out}");
696            assert!(out.ends_with("ok"), "{out}");
697        }
698    }
699}