use std::path::Path;
use crate::finding::{Finding, Location, Origin, Severity};
use crate::rules::slop::ItemSpan;
pub(crate) struct CommentSpan {
pub(crate) start_line: usize,
pub(crate) end_line: usize,
pub(crate) text: String,
pub(crate) is_doc: bool,
}
#[derive(PartialEq, Eq)]
enum CommentMarker {
Line,
Block,
}
fn find_comment_start(text: &str) -> Option<(usize, CommentMarker)> {
let bytes = text.as_bytes();
let mut in_string = false;
let mut i = 0;
while i < bytes.len() {
let c = bytes[i];
if c == b'"' {
let mut backslashes = 0;
let mut j = i;
while j > 0 && bytes[j - 1] == b'\\' {
backslashes += 1;
j -= 1;
}
if backslashes % 2 == 0 {
in_string = !in_string;
}
i += 1;
continue;
}
if !in_string && c == b'/' && i + 1 < bytes.len() {
if bytes[i + 1] == b'/' {
return Some((i, CommentMarker::Line));
}
if bytes[i + 1] == b'*' {
return Some((i, CommentMarker::Block));
}
}
i += 1;
}
None
}
pub(crate) fn extract_comments(source: &str) -> Vec<CommentSpan> {
let mut spans = Vec::new();
let mut in_block_comment = false;
let mut block_start_line = 0usize;
let mut block_is_doc = false;
let mut block_text = String::new();
for (idx, line) in source.lines().enumerate() {
let line_no = idx + 1;
let mut rest = line;
loop {
if in_block_comment {
if let Some(end) = rest.find("*/") {
let before = rest[..end].trim();
if !before.is_empty() {
block_text.push(' ');
block_text.push_str(before);
}
spans.push(CommentSpan {
start_line: block_start_line,
end_line: line_no,
text: block_text.trim().to_string(),
is_doc: block_is_doc,
});
in_block_comment = false;
block_text.clear();
rest = &rest[end + 2..];
continue;
}
let trimmed = rest.trim();
if !trimmed.is_empty() {
block_text.push(' ');
block_text.push_str(trimmed);
}
break;
}
match find_comment_start(rest) {
Some((pos, CommentMarker::Line)) => {
let marker = &rest[pos..];
let is_doc = (marker.starts_with("///") && !marker.starts_with("////"))
|| marker.starts_with("//!");
spans.push(CommentSpan {
start_line: line_no,
end_line: line_no,
text: rest[pos + 2..].trim().to_string(),
is_doc,
});
break;
}
Some((pos, CommentMarker::Block)) => {
let marker = &rest[pos..];
block_is_doc = marker.starts_with("/**") || marker.starts_with("/*!");
block_start_line = line_no;
in_block_comment = true;
rest = &rest[pos + 2..];
}
None => break,
}
}
}
spans
}
fn nearest_item_path(item_spans: &[ItemSpan], line: usize, file: &Path) -> String {
item_spans
.iter()
.filter(|span| span.start_line <= line && line <= span.end_line)
.min_by_key(|span| span.end_line - span.start_line)
.map(|span| span.item_path.clone())
.unwrap_or_else(|| file.display().to_string())
}
fn build_finding(
rule: &'static str,
line: usize,
file: &Path,
item_spans: &[ItemSpan],
severity: Severity,
evidence: Option<serde_json::Value>,
) -> Finding {
let rule = crate::finding::RuleId::from(rule);
let evidence_class = crate::finding::evidence_class_for_rule(&rule);
Finding::new(
format!("{rule}:{}:{line}:1", file.display()),
rule,
severity,
Location {
file: file.to_path_buf(),
line: crate::finding::OneBasedLine::new(line)
.expect("text-scan line numbers are 1-based"),
item_path: nearest_item_path(item_spans, line, file),
},
evidence_class,
Origin::Code,
evidence,
)
}
fn tokenize(text: &str) -> Vec<String> {
text.to_lowercase()
.split(|c: char| !c.is_alphanumeric())
.filter(|token| !token.is_empty())
.map(str::to_string)
.collect()
}
pub(crate) fn scan_comments(source: &str, item_spans: &[ItemSpan], file: &Path) -> Vec<Finding> {
let comments = extract_comments(source);
let source_lines: Vec<&str> = source.lines().collect();
let mut findings = Vec::new();
findings.extend(conversational_artifact_findings(
&comments, item_spans, file,
));
findings.extend(step_comment_inflation_findings(
&comments,
&source_lines,
item_spans,
file,
));
findings.extend(restating_comment_findings(
&comments,
&source_lines,
item_spans,
file,
));
findings
}
pub(crate) const CONVERSATIONAL_TIER1: &[&str] = &[
"as an ai",
"as an ai language model",
"as a language model",
"i'm an ai",
"i cannot browse the internet",
];
pub(crate) const CONVERSATIONAL_TIER2: &[&str] = &[
"here is",
"here's",
"note that this is a simplified",
"in a real implementation",
"in a production implementation",
"for the purposes of this example",
];
fn word_index_at(text: &str, pos: usize) -> usize {
text[..pos].split_whitespace().count()
}
fn quote_char_nearby(mut chars: impl Iterator<Item = char>) -> bool {
for _ in 0..3 {
match chars.next() {
Some(c) if c == '"' || c == '`' => return true,
Some(c) if c.is_alphanumeric() => return false,
Some(_) => continue,
None => return false,
}
}
false
}
fn phrase_at_word_boundary(haystack: &str, phrase: &str) -> Option<usize> {
let mut start = 0;
while let Some(offset) = haystack[start..].find(phrase) {
let match_start = start + offset;
let match_end = match_start + phrase.len();
let before_ok = !haystack[..match_start]
.chars()
.next_back()
.is_some_and(|c| c.is_alphanumeric());
let after_ok = !haystack[match_end..]
.chars()
.next()
.is_some_and(|c| c.is_alphanumeric());
let quoted = quote_char_nearby(haystack[..match_start].chars().rev())
&& quote_char_nearby(haystack[match_end..].chars());
if before_ok && after_ok && !quoted {
return Some(match_start);
}
start = match_start + 1;
if start >= haystack.len() {
break;
}
}
None
}
fn conversational_artifact_findings(
comments: &[CommentSpan],
item_spans: &[ItemSpan],
file: &Path,
) -> Vec<Finding> {
let mut findings = Vec::new();
for comment in comments {
if comment.is_doc {
continue;
}
let lower = comment.text.to_lowercase();
if CONVERSATIONAL_TIER1
.iter()
.any(|phrase| phrase_at_word_boundary(&lower, phrase).is_some())
{
findings.push(build_finding(
crate::rules::slop::CONVERSATIONAL_ARTIFACT_RULE,
comment.start_line,
file,
item_spans,
Severity::Warn,
None,
));
continue;
}
let hit = CONVERSATIONAL_TIER2.iter().any(|phrase| {
phrase_at_word_boundary(&lower, phrase)
.is_some_and(|pos| word_index_at(&lower, pos) < 8)
});
if hit {
findings.push(build_finding(
crate::rules::slop::CONVERSATIONAL_ARTIFACT_RULE,
comment.start_line,
file,
item_spans,
Severity::Info,
None,
));
}
}
findings
}
fn step_number(text: &str) -> Option<u32> {
let trimmed = text.trim_start();
let prefix = trimmed.get(..4)?;
if !prefix.eq_ignore_ascii_case("step") {
return None;
}
let rest = trimmed[4..].trim_start();
let digits: String = rest.chars().take_while(char::is_ascii_digit).collect();
if digits.is_empty() {
return None;
}
digits.parse().ok()
}
struct StepComment {
number: u32,
start_line: usize,
item_path: String,
}
fn non_blank_lines_between(source_lines: &[&str], from_line: usize, to_line: usize) -> usize {
if to_line <= from_line + 1 {
return 0;
}
source_lines[from_line..to_line - 1]
.iter()
.filter(|line| !line.trim().is_empty())
.count()
}
fn step_comment_inflation_findings(
comments: &[CommentSpan],
source_lines: &[&str],
item_spans: &[ItemSpan],
file: &Path,
) -> Vec<Finding> {
let steps: Vec<StepComment> = comments
.iter()
.filter(|comment| !comment.is_doc)
.filter_map(|comment| {
step_number(&comment.text).map(|number| StepComment {
number,
start_line: comment.start_line,
item_path: nearest_item_path(item_spans, comment.start_line, file),
})
})
.collect();
let mut findings = Vec::new();
let mut chain: Vec<&StepComment> = Vec::new();
for step in &steps {
let continues_chain = match chain.last() {
Some(last) => {
last.item_path == step.item_path
&& step.number == last.number + 1
&& non_blank_lines_between(source_lines, last.start_line, step.start_line) <= 2
}
None => true,
};
if !continues_chain {
flush_step_chain(&mut chain, &mut findings, file, item_spans);
}
chain.push(step);
}
flush_step_chain(&mut chain, &mut findings, file, item_spans);
findings
}
fn flush_step_chain(
chain: &mut Vec<&StepComment>,
findings: &mut Vec<Finding>,
file: &Path,
item_spans: &[ItemSpan],
) {
if chain.len() >= 3 {
let lines: Vec<usize> = chain.iter().map(|step| step.start_line).collect();
findings.push(build_finding(
crate::rules::slop::STEP_COMMENT_INFLATION_RULE,
chain[0].start_line,
file,
item_spans,
Severity::Info,
Some(serde_json::json!({ "chain_length": chain.len(), "lines": lines })),
));
}
chain.clear();
}
const RESTATING_STOPWORDS: &[&str] = &[
"the", "a", "an", "to", "of", "and", "this", "is", "let", "mut", "fn", "if", "else", "return",
];
fn content_tokens(text: &str) -> std::collections::HashSet<String> {
tokenize(text)
.into_iter()
.filter(|token| !RESTATING_STOPWORDS.contains(&token.as_str()))
.collect()
}
fn next_non_blank_line<'a>(source_lines: &[&'a str], after_line: usize) -> Option<&'a str> {
source_lines
.iter()
.skip(after_line)
.find(|line| !line.trim().is_empty())
.copied()
}
const RESTATING_ITEM_KEYWORDS: &[&str] = &[
"fn ", "pub fn", "struct ", "enum ", "impl ", "trait ", "mod ", "#[",
];
fn restating_comment_findings(
comments: &[CommentSpan],
source_lines: &[&str],
item_spans: &[ItemSpan],
file: &Path,
) -> Vec<Finding> {
let mut findings = Vec::new();
for comment in comments {
if comment.is_doc || comment.start_line != comment.end_line {
continue;
}
let Some(code_line) = next_non_blank_line(source_lines, comment.end_line) else {
continue;
};
let trimmed_code = code_line.trim();
if trimmed_code.starts_with("//")
|| trimmed_code.starts_with("/*")
|| RESTATING_ITEM_KEYWORDS
.iter()
.any(|keyword| trimmed_code.starts_with(keyword))
{
continue;
}
let comment_tokens = content_tokens(&comment.text);
let code_tokens = content_tokens(trimmed_code);
if comment_tokens.len() < 4 || code_tokens.len() < 3 {
continue;
}
let intersection = comment_tokens.intersection(&code_tokens).count();
let union = comment_tokens.union(&code_tokens).count();
if union == 0 {
continue;
}
let similarity = intersection as f32 / union as f32;
if similarity >= 0.7 {
findings.push(build_finding(
crate::rules::slop::RESTATING_COMMENT_RULE,
comment.start_line,
file,
item_spans,
Severity::Info,
None,
));
}
}
findings
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn a_multibyte_char_at_byte_four_does_not_panic_step_number() {
assert_eq!(step_number("v1 — new format"), None);
assert_eq!(step_number("äöü"), None);
assert_eq!(step_number("Step 3:"), Some(3));
}
#[test]
fn string_literal_containing_comment_markers_is_not_a_comment() {
let spans = extract_comments("let url = \"http://example.com\";\n");
assert!(spans.is_empty());
}
#[test]
fn line_comment_after_a_string_literal_is_still_found() {
let spans = extract_comments("let url = \"http://example.com\"; // real comment\n");
assert_eq!(spans.len(), 1);
assert_eq!(spans[0].text, "real comment");
assert!(!spans[0].is_doc);
}
#[test]
fn triple_slash_is_doc_quadruple_slash_is_not() {
let spans = extract_comments("/// doc\n//// not doc\n");
assert_eq!(spans.len(), 2);
assert!(spans[0].is_doc);
assert!(!spans[1].is_doc);
}
#[test]
fn block_comment_spans_multiple_lines() {
let spans = extract_comments("/* line one\nline two */\ncode();\n");
assert_eq!(spans.len(), 1);
assert_eq!(spans[0].start_line, 1);
assert_eq!(spans[0].end_line, 2);
assert!(spans[0].text.contains("line one"));
assert!(spans[0].text.contains("line two"));
}
fn findings_for(source: &str, name: &str) -> Vec<Finding> {
let dir = crate::test_util::TempDir::new(name);
let file = dir.join("lib.rs");
std::fs::write(&file, source).unwrap();
crate::rules::slop::analyze_file(&file, false).unwrap()
}
fn rule_findings<'a>(findings: &'a [Finding], rule: &str) -> Vec<&'a Finding> {
findings.iter().filter(|f| f.rule == rule).collect()
}
#[test]
fn nested_block_comment_swallows_trailing_text_in_extract_comments() {
let spans = extract_comments(
"/* outer doc\n/* inner note */\nin a real implementation, this text sits inside the outer comment but the scanner already closed it\nstill more hidden text */\n",
);
assert_eq!(spans.len(), 1);
assert!(!spans[0].text.contains("in a real implementation"));
}
#[test]
fn nested_block_comment_hides_trigger_from_conversational_artifact() {
let findings = findings_for(
"/* outer doc\n/* inner note */\nin a real implementation, this text sits inside the outer comment but the scanner already closed it\nstill more hidden text */\nfn f() {}\n",
"slop-text-nested-block-comment",
);
assert!(
rule_findings(&findings, crate::rules::slop::CONVERSATIONAL_ARTIFACT_RULE).is_empty()
);
}
#[test]
fn consecutive_line_comments_are_never_joined_into_one_block() {
let findings = findings_for(
"fn f() {\n // I'm an\n // AI assistant explaining this limitation in detail here.\n let _ = 1;\n}\n",
"slop-text-consecutive-line-comments",
);
assert!(
rule_findings(&findings, crate::rules::slop::CONVERSATIONAL_ARTIFACT_RULE).is_empty()
);
}
#[test]
fn quoted_trigger_phrase_used_for_meta_discussion_does_not_fire() {
let findings = findings_for(
"// The trigger phrase (\"as an AI\") lives inside a Rust string literal\n// here, not a real comment, so this can't self-trigger the rule.\nfn f() {}\n",
"slop-text-quoted-trigger-phrase",
);
assert!(
rule_findings(&findings, crate::rules::slop::CONVERSATIONAL_ARTIFACT_RULE).is_empty()
);
}
#[test]
fn unquoted_genuine_disclaimer_still_fires() {
let findings = findings_for(
"fn f() {\n // As an AI, I can't verify this.\n let _ = 1;\n}\n",
"slop-text-unquoted-disclaimer",
);
assert_eq!(
rule_findings(&findings, crate::rules::slop::CONVERSATIONAL_ARTIFACT_RULE).len(),
1
);
}
}