pub const RAW_INJECTION_PATTERNS: &[(&str, &str)] = &[
(
"ignore_instructions",
r"(?i)ignore\s+(all\s+|any\s+|previous\s+|prior\s+)?instructions",
),
("role_override", r"(?i)you\s+are\s+now"),
(
"new_directive",
r"(?i)new\s+(instructions?|directives?)\s*:",
),
("developer_mode", r"(?i)developer\s+mode"),
(
"system_prompt_leak",
r"(?i)((reveal|show|print|output|display|repeat|expose|dump|leak|copy|give)\s+(me\s+)?(your\s+|the\s+|my\s+)?(full\s+|entire\s+|exact\s+|complete\s+)?system\s+prompt|what\s+(is|are|was)\s+(your\s+|the\s+)?system\s+prompt)",
),
(
"reveal_instructions",
r"(?i)(reveal|show|display|print)\s+your\s+(instructions?|prompts?|rules?)",
),
("jailbreak", r"(?i)\b(DAN|jailbreak)\b"),
("base64_payload", r"(?i)(decode|eval|execute).*base64"),
(
"xml_tag_injection",
r"(?i)</?\s*(system|assistant|user|tool_result|function_call)\s*>",
),
("markdown_image_exfil", r"(?i)!\[.*?\]\(https?://[^)]+\)"),
("forget_everything", r"(?i)forget\s+(everything|all)"),
(
"disregard_instructions",
r"(?i)disregard\s+(your|all|previous)",
),
(
"override_directives",
r"(?i)override\s+(your|all)\s+(directives?|instructions?|rules?)",
),
("act_as_if", r"(?i)act\s+as\s+if"),
("html_image_exfil", r"(?i)<img\s+[^>]*src\s*="),
("delimiter_escape_tool_output", r"(?i)</?tool-output[\s>]"),
(
"delimiter_escape_external_data",
r"(?i)</?external-data[\s>]",
),
];
pub const RAW_RESPONSE_PATTERNS: &[(&str, &str)] = &[
(
"autonomy_override",
r"(?i)\bset\s+(autonomy|trust)\s*(level|mode)\s*to\b",
),
(
"memory_write_instruction",
r"(?i)\b(now\s+)?(store|save|remember|write)\s+this\s+(to|in)\s+(memory|vault|database)\b",
),
(
"instruction_override",
r"(?i)\b(from\s+now\s+on|henceforth)\b.{0,80}\b(always|never|must)\b",
),
(
"config_manipulation",
r"(?i)\b(change|modify|update)\s+your\s+(config|configuration|settings)\b",
),
(
"ignore_instructions_response",
r"(?i)\bignore\s+(all\s+|any\s+|your\s+)?(previous\s+|prior\s+)?(instructions?|rules?|constraints?)\b",
),
(
"override_directives_response",
r"(?i)\boverride\s+(your\s+)?(directives?|instructions?|rules?|constraints?)\b",
),
(
"disregard_system",
r"(?i)\bdisregard\s+(your\s+|the\s+)?(system\s+prompt|instructions?|guidelines?)\b",
),
];
#[must_use]
pub fn strip_format_chars(text: &str) -> String {
text.chars()
.filter(|&c| {
if c == '\t' || c == '\n' {
return true;
}
if c.is_ascii_control() {
return false;
}
!matches!(
c,
'\u{00AD}' | '\u{034F}' | '\u{061C}' | '\u{115F}' | '\u{1160}' | '\u{17B4}' | '\u{17B5}' | '\u{180B}'..='\u{180D}' | '\u{180F}' | '\u{200B}'..='\u{200F}' | '\u{202A}'..='\u{202E}' | '\u{2060}'..='\u{2064}' | '\u{2066}'..='\u{206F}' | '\u{FEFF}' | '\u{FFF9}'..='\u{FFFB}' | '\u{1BCA0}'..='\u{1BCA3}' | '\u{1D173}'..='\u{1D17A}' | '\u{E0000}'..='\u{E007F}' )
})
.collect()
}