use super::*;
#[derive(Debug, Clone, PartialEq, Eq, Hash)]
struct ContentSearchPairKey {
scenario: String,
path_label: Option<String>,
pattern_hash: String,
mode: String,
limit: u64,
context: u64,
}
#[derive(Debug, Clone, Copy)]
struct ContentSearchP95 {
p95: f64,
}
#[derive(Debug, Clone, Copy, Serialize)]
struct ContentSearchPairedSpeedupStats {
samples: usize,
wins: usize,
median: f64,
min: f64,
max: f64,
}
fn print_content_search_benchmark_summary(
cfg: &ContentSearchBenchConfig,
scenario_count: usize,
observed_values: &[Value],
) {
print_content_search_benchmark_line(content_search_summary_value(
cfg,
scenario_count,
observed_values,
));
}
fn content_search_summary_value(
cfg: &ContentSearchBenchConfig,
scenario_count: usize,
observed_values: &[Value],
) -> Value {
let mut fastest_p50_by_mode = serde_json::Map::new();
let mut fastest_p95_by_mode = serde_json::Map::new();
for mode in [ContentSearchMode::Plain, ContentSearchMode::Regex] {
let mode_name = mode.as_str();
let warm_values = observed_values
.iter()
.filter(|value| content_search_warm_success_value(value, mode_name))
.collect::<Vec<_>>();
if let Some(value) = warm_values
.iter()
.filter_map(|value| {
Some((
value["engine"].as_str()?,
value["elapsed_ms"]["p50"].as_f64()?,
))
})
.min_by(|left, right| left.1.total_cmp(&right.1))
{
fastest_p50_by_mode.insert(mode_name.to_string(), json!(value.0));
}
if let Some(value) = warm_values
.iter()
.filter_map(|value| {
Some((
value["engine"].as_str()?,
value["elapsed_ms"]["p95"].as_f64()?,
))
})
.min_by(|left, right| left.1.total_cmp(&right.1))
{
fastest_p95_by_mode.insert(mode_name.to_string(), json!(value.0));
}
}
let mut grep_by_key: HashMap<ContentSearchPairKey, ContentSearchP95> = HashMap::new();
let mut candidate_by_key: HashMap<ContentSearchPairKey, ContentSearchP95> = HashMap::new();
for value in observed_values {
if !content_search_warm_success_value(value, value["mode"].as_str().unwrap_or_default()) {
continue;
}
let Some(key) = content_search_pair_key(value) else {
continue;
};
let Some(engine) = value["engine"].as_str() else {
continue;
};
let Some(p95) = value["elapsed_ms"]["p95"].as_f64() else {
continue;
};
let p95 = ContentSearchP95 { p95 };
if engine.starts_with("ffgrep_") {
grep_by_key
.entry(key)
.and_modify(|existing| {
if p95.p95 < existing.p95 {
*existing = p95;
}
})
.or_insert(p95);
} else if engine.starts_with("candidate_") {
candidate_by_key
.entry(key)
.and_modify(|existing| {
if p95.p95 < existing.p95 {
*existing = p95;
}
})
.or_insert(p95);
}
}
let mut plain_ratios = Vec::new();
let mut regex_ratios = Vec::new();
for (key, candidate) in &candidate_by_key {
let Some(grep) = grep_by_key.get(key) else {
continue;
};
if candidate.p95 <= 0.0 {
continue;
}
let ratio = grep.p95 / candidate.p95;
match key.mode.as_str() {
"plain" => plain_ratios.push(ratio),
"regex" => regex_ratios.push(ratio),
_ => {}
}
}
let plain_paired = paired_speedup_stats(&mut plain_ratios);
let regex_paired = paired_speedup_stats(&mut regex_ratios);
let plain_speedup = plain_paired.map(|stats| stats.median);
let regex_speedup = regex_paired.map(|stats| stats.median);
let total_failures = observed_values
.iter()
.map(|value| value["failures"].as_u64().unwrap_or_default())
.sum::<u64>();
let failure_kind_counts = content_search_failure_kind_counts(observed_values);
let diagnostic_kind_counts = content_search_diagnostic_kind_counts(observed_values);
let failure_scenarios = observed_values
.iter()
.filter(|value| value["failures"].as_u64().unwrap_or_default() > 0)
.map(content_search_failure_scenario)
.collect::<Vec<_>>();
let resource_growth_observed = observed_values.iter().any(|value| {
value["fd_delta"].as_i64().is_some_and(|delta| delta > 16)
|| value["thread_delta"]
.as_i64()
.is_some_and(|delta| delta > 16)
});
let regex_fallback_observed = observed_values.iter().any(|value| {
value["regex_fallback_error_present"]
.as_bool()
.unwrap_or(false)
});
let adapter_limit_mismatch_observed = observed_values.iter().any(|value| {
value["adapter_limit_mismatch_observed"]
.as_bool()
.unwrap_or(false)
});
let raw_limit_overrun_observed = observed_values.iter().any(|value| {
value["raw_limit_overrun_observed"]
.as_bool()
.unwrap_or(false)
});
let zero_match_compatibility_risk = observed_values.iter().any(|candidate| {
let Some(engine) = candidate["engine"].as_str() else {
return false;
};
if !engine.starts_with("candidate_")
|| candidate["matches_returned"].as_u64().unwrap_or_default() != 0
{
return false;
}
observed_values.iter().any(|grep| {
grep["engine"]
.as_str()
.is_some_and(|engine| engine.starts_with("ffgrep_"))
&& grep["scenario"] == candidate["scenario"]
&& grep["path_label"] == candidate["path_label"]
&& grep["pattern_hash"] == candidate["pattern_hash"]
&& grep["mode"] == candidate["mode"]
&& grep["limit"] == candidate["limit"]
&& grep["context"] == candidate["context"]
&& grep["matches_returned"].as_u64().unwrap_or_default() > 0
})
});
let rss_known = observed_values
.iter()
.any(|value| !value["rss_kb_delta"].is_null());
let compatibility_risk_observed = total_failures > 0
|| regex_fallback_observed
|| zero_match_compatibility_risk
|| adapter_limit_mismatch_observed;
let recommended = total_failures == 0
&& plain_speedup.is_some_and(|speedup| speedup >= 1.2)
&& regex_speedup.is_some_and(|speedup| speedup >= 1.0)
&& !compatibility_risk_observed
&& !resource_growth_observed
&& rss_known;
json!({
"schema": 1,
"scenario": "summary",
"root_label": cfg.root_label,
"root_hash": cfg.root_hash,
"paths": cfg.paths,
"repetitions": cfg.repetitions,
"limits": cfg.limits,
"contexts": cfg.contexts,
"baseline": cfg.baseline.as_str(),
"enable_content_indexing": cfg.content_indexing,
"cwd_churn": cfg.cwd_churn,
"scenario_lines": scenario_count,
"fastest_warm_p50_engine_by_mode": fastest_p50_by_mode,
"fastest_warm_p95_engine_by_mode": fastest_p95_by_mode,
"candidate_plain_speedup_vs_ffgrep_native_p95": plain_speedup,
"candidate_regex_speedup_vs_ffgrep_native_p95": regex_speedup,
"paired_p95_speedup_vs_ffgrep_by_mode": {
"plain": plain_paired,
"regex": regex_paired,
},
"total_failures": total_failures,
"failure_kind_counts": failure_kind_counts,
"diagnostic_kind_counts": diagnostic_kind_counts,
"failure_scenario_count": failure_scenarios.len(),
"failure_scenarios": failure_scenarios,
"compatibility_risk_observed": compatibility_risk_observed,
"actual_failure_observed": total_failures > 0,
"regex_fallback_observed": regex_fallback_observed,
"semantic_mismatch_observed": zero_match_compatibility_risk,
"zero_match_compatibility_risk": zero_match_compatibility_risk,
"adapter_limit_mismatch_observed": adapter_limit_mismatch_observed,
"raw_limit_overrun_observed": raw_limit_overrun_observed,
"raw_limit_overrun_diagnostic": "performance_only_visible_output_capped",
"resource_growth_observed": resource_growth_observed,
"step4_exploration": {
"recommended": recommended,
"regex_equivalent_speedup_required": true,
"caveat": "benchmark approves only separate content-search candidate exploration, not production replacement"
}
})
}
fn content_search_warm_success_value(value: &Value, mode_name: &str) -> bool {
value["mode"].as_str() == Some(mode_name)
&& value["scenario"]
.as_str()
.is_some_and(|scenario| scenario.starts_with("warm_") || scenario == "high_limit_tail")
&& value["successes"].as_u64().unwrap_or_default() > 0
&& value["failures"].as_u64().unwrap_or_default() == 0
}
fn content_search_pair_key(value: &Value) -> Option<ContentSearchPairKey> {
Some(ContentSearchPairKey {
scenario: value["scenario"].as_str()?.to_string(),
path_label: value["path_label"].as_str().map(ToOwned::to_owned),
pattern_hash: value["pattern_hash"].as_str()?.to_string(),
mode: value["mode"].as_str()?.to_string(),
limit: value["limit"].as_u64()?,
context: value["context"].as_u64()?,
})
}
fn paired_speedup_stats(ratios: &mut [f64]) -> Option<ContentSearchPairedSpeedupStats> {
if ratios.is_empty() {
return None;
}
ratios.sort_by(f64::total_cmp);
let median = if ratios.len().is_multiple_of(2) {
let upper = ratios.len() / 2;
(ratios[upper - 1] + ratios[upper]) / 2.0
} else {
ratios[ratios.len() / 2]
};
Some(ContentSearchPairedSpeedupStats {
samples: ratios.len(),
wins: ratios.iter().filter(|ratio| **ratio > 1.0).count(),
median: round_millis(median),
min: round_millis(ratios[0]),
max: round_millis(ratios[ratios.len() - 1]),
})
}
fn content_search_failure_kind_counts(observed_values: &[Value]) -> BTreeMap<String, u64> {
let mut counts = BTreeMap::new();
for value in observed_values {
let Some(kinds) = value["failure_kind_counts"].as_object() else {
continue;
};
for (kind, count) in kinds {
*counts.entry(kind.clone()).or_default() += count.as_u64().unwrap_or_default();
}
}
counts
}
fn content_search_diagnostic_kind_counts(observed_values: &[Value]) -> BTreeMap<String, u64> {
let mut counts = BTreeMap::new();
for value in observed_values {
let Some(kinds) = value["diagnostic_kind_counts"].as_object() else {
continue;
};
for (kind, count) in kinds {
*counts.entry(kind.clone()).or_default() += count.as_u64().unwrap_or_default();
}
}
counts
}
fn content_search_failure_scenario(value: &Value) -> Value {
json!({
"scenario": value["scenario"],
"path_label": value["path_label"],
"pattern_label": value["pattern_label"],
"pattern_hash": value["pattern_hash"],
"mode": value["mode"],
"engine": value["engine"],
"limit": value["limit"],
"context": value["context"],
"failures": value["failures"],
"failure_kind_counts": value["failure_kind_counts"],
})
}
const CONTENT_SEARCH_BENCH_PREFIX: &str = "content_search_benchmark ";
const CONTENT_SEARCH_BENCH_DEFAULT_REPETITIONS: usize = 25;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum ContentSearchBaseline {
Grep,
Candidate,
Both,
}
impl ContentSearchBaseline {
fn from_env() -> Self {
match std::env::var("FFF_CONTENT_BENCH_BASELINE")
.unwrap_or_else(|_| "both".to_string())
.trim()
{
"grep" => Self::Grep,
"candidate" => Self::Candidate,
"both" | "" => Self::Both,
value => {
eprintln!(
"content_search_benchmark warning=unsupported_baseline value={value:?} using=both"
);
Self::Both
}
}
}
fn includes_grep(self) -> bool {
matches!(self, Self::Grep | Self::Both)
}
fn includes_candidate(self) -> bool {
matches!(self, Self::Candidate | Self::Both)
}
fn as_str(self) -> &'static str {
match self {
Self::Grep => "grep",
Self::Candidate => "candidate",
Self::Both => "both",
}
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
enum ContentSearchMode {
Plain,
Regex,
}
impl ContentSearchMode {
fn as_str(self) -> &'static str {
match self {
Self::Plain => "plain",
Self::Regex => "regex",
}
}
fn candidate_engine(self) -> &'static str {
match self {
Self::Plain => "candidate_plain",
Self::Regex => "candidate_regex",
}
}
}
#[derive(Debug, Clone)]
struct ContentSearchPattern {
label: String,
hash: String,
text: String,
mode: ContentSearchMode,
}
impl ContentSearchPattern {
fn grep_pattern(&self) -> String {
match self.mode {
ContentSearchMode::Plain => regex::escape(&self.text),
ContentSearchMode::Regex => self.text.clone(),
}
}
}
#[derive(Debug)]
struct ContentSearchBenchConfig {
root: PathBuf,
root_label: String,
root_hash: String,
paths: Vec<String>,
plain_patterns: Vec<ContentSearchPattern>,
regex_patterns: Vec<ContentSearchPattern>,
repetitions: usize,
limits: Vec<usize>,
contexts: Vec<usize>,
baseline: ContentSearchBaseline,
content_indexing: bool,
cwd_churn: usize,
show_patterns: bool,
}
impl ContentSearchBenchConfig {
fn from_env() -> anyhow::Result<Self> {
let root = std::env::var("FFF_CONTENT_BENCH_ROOT")
.map(PathBuf::from)
.unwrap_or_else(|_| PathBuf::from(env!("CARGO_MANIFEST_DIR")))
.canonicalize()?;
let root_label = root
.file_name()
.and_then(|name| name.to_str())
.filter(|name| !name.is_empty())
.unwrap_or("root")
.to_string();
let root_hash = hash_root_label(&root);
let paths = content_benchmark_paths_from_env()?;
let plain_patterns = content_patterns_from_env(
"FFF_CONTENT_BENCH_PATTERNS_PLAIN",
ContentSearchMode::Plain,
&[
"ToolRuntime",
"grep",
"serde_json",
"ProfileResult",
"content_search",
],
);
let regex_patterns = content_patterns_from_env(
"FFF_CONTENT_BENCH_PATTERNS_REGEX",
ContentSearchMode::Regex,
&[
"fn [a-zA-Z0-9_]+",
"pub\\(crate\\) struct",
"impl ToolRuntime",
],
);
let repetitions = parse_usize_env(
"FFF_CONTENT_BENCH_REPETITIONS",
CONTENT_SEARCH_BENCH_DEFAULT_REPETITIONS,
1,
usize::MAX,
);
let limits = parse_usize_list_env("FFF_CONTENT_BENCH_LIMITS", &[20, 100], 1);
let contexts = parse_usize_list_env("FFF_CONTENT_BENCH_CONTEXTS", &[0, 2], 0);
let content_indexing = parse_bool_env("FFF_CONTENT_BENCH_CONTENT_INDEXING", true);
let cwd_churn = parse_usize_env("FFF_CONTENT_BENCH_CWD_CHURN", 0, 0, usize::MAX);
let show_patterns = parse_bool_env("FFF_CONTENT_BENCH_SHOW_PATTERNS", false);
Ok(Self {
root,
root_label,
root_hash,
paths,
plain_patterns,
regex_patterns,
repetitions,
limits,
contexts,
baseline: ContentSearchBaseline::from_env(),
content_indexing,
cwd_churn,
show_patterns,
})
}
fn existing_paths(&self) -> Vec<String> {
self.paths
.iter()
.filter_map(|path| {
let full_path = self.root.join(path);
if full_path.is_dir() {
Some(path.clone())
} else {
eprintln!(
"content_search_benchmark warning=missing_path path={path:?} action=skipped"
);
None
}
})
.collect()
}
}
fn content_benchmark_paths_from_env() -> anyhow::Result<Vec<String>> {
comma_env("FFF_CONTENT_BENCH_PATHS")
.unwrap_or_else(|| ["src", "docs"].into_iter().map(ToOwned::to_owned).collect())
.into_iter()
.enumerate()
.map(|(index, path)| {
sanitize_content_benchmark_path(&path).map_err(|error| {
anyhow::anyhow!(
"invalid FFF_CONTENT_BENCH_PATHS entry {}: {error}",
index + 1
)
})
})
.collect()
}
fn sanitize_content_benchmark_path(path: &str) -> anyhow::Result<String> {
let path = Path::new(path);
if path.is_absolute() {
anyhow::bail!("absolute paths are not allowed");
}
let mut parts = Vec::new();
for component in path.components() {
match component {
std::path::Component::Normal(part) => {
let part = part
.to_str()
.ok_or_else(|| anyhow::anyhow!("non-utf8 paths are not allowed"))?;
parts.push(part.to_string());
}
std::path::Component::CurDir => {}
std::path::Component::ParentDir => {
anyhow::bail!("parent-directory paths are not allowed");
}
std::path::Component::RootDir | std::path::Component::Prefix(_) => {
anyhow::bail!("rooted paths are not allowed");
}
}
}
if parts.is_empty() {
anyhow::bail!("empty paths are not allowed");
}
Ok(parts.join("/"))
}
fn parse_bool_env(name: &str, default: bool) -> bool {
std::env::var(name)
.ok()
.map(|value| matches!(value.trim(), "1" | "true" | "TRUE" | "yes" | "YES"))
.unwrap_or(default)
}
fn parse_usize_list_env(name: &str, default: &[usize], min: usize) -> Vec<usize> {
let mut values = std::env::var(name)
.ok()
.map(|value| {
value
.split(',')
.filter_map(|value| value.trim().parse::<usize>().ok())
.filter(|value| *value >= min)
.collect::<Vec<_>>()
})
.filter(|values| !values.is_empty())
.unwrap_or_else(|| default.to_vec());
values.sort_unstable();
values.dedup();
values
}
fn content_patterns_from_env(
env_name: &str,
mode: ContentSearchMode,
defaults: &[&str],
) -> Vec<ContentSearchPattern> {
comma_env(env_name)
.unwrap_or_else(|| defaults.iter().map(|value| (*value).to_string()).collect())
.into_iter()
.enumerate()
.map(|(index, text)| ContentSearchPattern {
label: format!("{}_{:02}", mode.as_str(), index + 1),
hash: hash_content_search_pattern(&text),
text,
mode,
})
.collect()
}
fn hash_content_search_pattern(pattern: &str) -> String {
use sha2::{Digest, Sha256};
let mut hasher = Sha256::new();
hasher.update(pattern.as_bytes());
crate::hex::lower_hex(hasher.finalize())[..16].to_string()
}
#[derive(Debug)]
struct ContentCandidatePicker {
runtime: ToolRuntime,
index_ready: bool,
cold_index_wait_ms: f64,
content_indexing: bool,
}
impl ContentCandidatePicker {
fn new(scope: &Path, content_indexing: bool) -> anyhow::Result<Self> {
let started = Instant::now();
Ok(Self {
runtime: ToolRuntime::new(scope)?,
index_ready: true,
cold_index_wait_ms: round_millis(started.elapsed().as_secs_f64() * 1000.0),
content_indexing,
})
}
}
#[derive(Debug, Clone)]
struct ContentSearchObservation {
engine: String,
elapsed: Vec<Duration>,
successes: u64,
failures: u64,
matches_returned: u64,
raw_matches_returned: Option<u64>,
output_bytes: u64,
truncated: bool,
timed_out: bool,
exit_code: Option<i64>,
index_ready: Option<bool>,
cold_index_wait_ms: Option<f64>,
enable_content_indexing: Option<bool>,
files_searched: Option<u64>,
total_files: Option<u64>,
filtered_file_count: Option<u64>,
files_with_matches: Option<u64>,
regex_fallback_error_present: Option<bool>,
next_file_offset_nonzero: Option<bool>,
limit_truncation_observed: bool,
pipe_output_truncation_observed: bool,
adapter_limit_mismatch_observed: bool,
raw_limit_overrun_observed: bool,
failure_kind_counts: BTreeMap<&'static str, u64>,
diagnostic_kind_counts: BTreeMap<&'static str, u64>,
}
fn run_grep_content_observation(
runtime: &ToolRuntime,
pattern: &str,
path: Option<&str>,
limit: usize,
context: usize,
iterations: usize,
) -> ContentSearchObservation {
let mut elapsed = Vec::with_capacity(iterations);
let mut successes = 0;
let mut failures = 0;
let mut matches_returned = 0;
let mut output_bytes = 0;
let mut truncated = false;
let mut timed_out = false;
let mut exit_code = None;
let mut engine = "ffgrep_unknown".to_string();
let mut limit_truncation_observed = false;
let mut pipe_output_truncation_observed = false;
let mut failure_kind_counts = BTreeMap::new();
let diagnostic_kind_counts = BTreeMap::new();
for _ in 0..iterations {
let arguments = match path {
Some(path) => {
json!({"patterns": [pattern], "path": path, "limit": limit, "context": context})
}
None => json!({"patterns": [pattern], "limit": limit, "context": context}),
};
let started = Instant::now();
let result = runtime.dispatch("grep", arguments);
elapsed.push(started.elapsed());
if result.success {
successes += 1;
} else {
failures += 1;
increment_failure_kind(
&mut failure_kind_counts,
classify_grep_failure(&result.metadata),
);
}
engine = match result.metadata.get("engine").and_then(Value::as_str) {
Some(other) => format!("ffgrep_{other}"),
None => "ffgrep_unknown".to_string(),
};
matches_returned = metadata_u64(&result.metadata, "matches_returned");
output_bytes = result.content.len() as u64;
let current_truncated = metadata_bool(&result.metadata, "truncated");
truncated |= current_truncated;
timed_out |= metadata_bool(&result.metadata, "timed_out");
exit_code = result.metadata.get("exit_code").and_then(Value::as_i64);
limit_truncation_observed |= result.success
&& matches_returned >= limit as u64
&& (current_truncated || metadata_bool(&result.metadata, "has_more"));
pipe_output_truncation_observed |=
grep_pipe_output_truncation_observed(&result.metadata, result.success);
}
ContentSearchObservation {
engine,
elapsed,
successes,
failures,
matches_returned,
raw_matches_returned: Some(matches_returned),
output_bytes,
truncated,
timed_out,
exit_code,
index_ready: None,
cold_index_wait_ms: None,
enable_content_indexing: None,
files_searched: None,
total_files: None,
filtered_file_count: None,
files_with_matches: None,
regex_fallback_error_present: None,
next_file_offset_nonzero: None,
limit_truncation_observed,
pipe_output_truncation_observed,
adapter_limit_mismatch_observed: matches_returned > limit as u64,
raw_limit_overrun_observed: false,
failure_kind_counts,
diagnostic_kind_counts,
}
}
fn increment_failure_kind(
failure_kind_counts: &mut BTreeMap<&'static str, u64>,
kind: &'static str,
) {
*failure_kind_counts.entry(kind).or_default() += 1;
}
fn classify_grep_failure(metadata: &Value) -> &'static str {
if metadata_bool(metadata, "timed_out") {
return "timed_out";
}
if metadata
.get("cleanup_warning")
.and_then(Value::as_str)
.is_some_and(|warning| !warning.is_empty())
{
return "cleanup_warning";
}
if metadata_bool(metadata, "stdout_truncated") || metadata_bool(metadata, "truncated") {
return "pipe_or_output_truncation";
}
match metadata.get("exit_code").and_then(Value::as_i64) {
Some(2) => "rg_error_exit",
Some(code) if code != 0 && code != 1 => "nonzero_exit",
None => "missing_exit_status",
_ => "unknown_failure",
}
}
fn grep_pipe_output_truncation_observed(metadata: &Value, success: bool) -> bool {
metadata_bool(metadata, "stdout_truncated")
|| metadata_bool(metadata, "stderr_truncated")
|| (!success
&& metadata_bool(metadata, "truncated")
&& !metadata_bool(metadata, "timed_out"))
}
fn run_candidate_content_observation(
picker: &ContentCandidatePicker,
pattern: &str,
mode: ContentSearchMode,
limit: usize,
context: usize,
iterations: usize,
) -> anyhow::Result<ContentSearchObservation> {
let search_pattern = match mode {
ContentSearchMode::Plain => regex::escape(pattern),
ContentSearchMode::Regex => pattern.to_string(),
};
let mut observation = run_grep_content_observation(
&picker.runtime,
&search_pattern,
None,
limit,
context,
iterations,
);
observation.engine = mode.candidate_engine().to_string();
observation.index_ready = Some(picker.index_ready);
observation.cold_index_wait_ms = Some(picker.cold_index_wait_ms);
observation.enable_content_indexing = Some(picker.content_indexing);
observation.raw_matches_returned = Some(observation.matches_returned);
observation.files_searched = None;
observation.total_files = None;
observation.filtered_file_count = None;
observation.files_with_matches = None;
observation.regex_fallback_error_present = Some(false);
observation.next_file_offset_nonzero = Some(false);
observation.raw_limit_overrun_observed = false;
Ok(observation)
}
fn metadata_bool(metadata: &Value, key: &str) -> bool {
metadata.get(key).and_then(Value::as_bool).unwrap_or(false)
}
fn print_content_search_benchmark_line(value: Value) {
println!("{CONTENT_SEARCH_BENCH_PREFIX}{value}");
}
#[expect(
clippy::too_many_arguments,
reason = "benchmark output builder mirrors JSON fields for call-site clarity"
)]
fn content_search_benchmark_line(
cfg: &ContentSearchBenchConfig,
scenario: &str,
pattern: &ContentSearchPattern,
path_label: Option<&str>,
limit: usize,
context: usize,
observation: &ContentSearchObservation,
resources: ResourceDelta,
) -> Value {
let mut value = json!({
"schema": 1,
"scenario": scenario,
"root_label": cfg.root_label,
"root_hash": cfg.root_hash,
"path_label": path_label,
"pattern_label": pattern.label,
"pattern_hash": pattern.hash,
"mode": pattern.mode.as_str(),
"engine": observation.engine,
"iterations": observation.elapsed.len(),
"limit": limit,
"context": context,
"elapsed_ms": duration_stats(&observation.elapsed),
"successes": observation.successes,
"failures": observation.failures,
"matches_returned": observation.matches_returned,
"raw_matches_returned": observation.raw_matches_returned,
"output_bytes": observation.output_bytes,
"truncated": observation.truncated,
"timed_out": observation.timed_out,
"exit_code": observation.exit_code,
"index_ready": observation.index_ready,
"cold_index_wait_ms": observation.cold_index_wait_ms,
"enable_content_indexing": observation.enable_content_indexing,
"files_searched": observation.files_searched,
"total_files": observation.total_files,
"filtered_file_count": observation.filtered_file_count,
"files_with_matches": observation.files_with_matches,
"regex_fallback_error_present": observation.regex_fallback_error_present,
"next_file_offset_nonzero": observation.next_file_offset_nonzero,
"limit_truncation_observed": observation.limit_truncation_observed,
"pipe_output_truncation_observed": observation.pipe_output_truncation_observed,
"adapter_limit_mismatch_observed": observation.adapter_limit_mismatch_observed,
"raw_limit_overrun_observed": observation.raw_limit_overrun_observed,
"failure_kind_counts": observation.failure_kind_counts,
"diagnostic_kind_counts": observation.diagnostic_kind_counts,
"rss_kb_delta": resources.rss_kb_delta,
"fd_delta": resources.fd_delta,
"thread_delta": resources.thread_delta,
});
if cfg.show_patterns {
value["pattern"] = json!(pattern.text);
}
value
}
#[expect(
clippy::too_many_arguments,
reason = "benchmark harness records each scenario dimension explicitly"
)]
fn record_content_search_observation(
cfg: &ContentSearchBenchConfig,
scenario_count: &mut usize,
observed_values: &mut Vec<Value>,
scenario: &str,
pattern: &ContentSearchPattern,
path_label: Option<&str>,
limit: usize,
context: usize,
observation: ContentSearchObservation,
resources: ResourceDelta,
) {
let value = content_search_benchmark_line(
cfg,
scenario,
pattern,
path_label,
limit,
context,
&observation,
resources,
);
print_content_search_benchmark_line(value.clone());
observed_values.push(value);
*scenario_count += 1;
}
fn content_search_scenario_name(
root_scoped: bool,
mode: ContentSearchMode,
limit: usize,
context: usize,
max_limit: usize,
) -> &'static str {
if context > 0 {
return "warm_context";
}
if limit == max_limit && max_limit > 100 {
return "high_limit_tail";
}
match (root_scoped, mode) {
(false, ContentSearchMode::Plain) => "warm_root_plain",
(false, ContentSearchMode::Regex) => "warm_root_regex",
(true, ContentSearchMode::Plain) => "warm_scoped_plain",
(true, ContentSearchMode::Regex) => "warm_scoped_regex",
}
}
fn assert_content_sentinel_observation(
label: &str,
observation: &ContentSearchObservation,
min_matches: u64,
) {
assert!(
observation.successes > 0 && observation.failures == 0,
"content benchmark sentinel {label} failed: {observation:?}"
);
assert!(
observation.matches_returned >= min_matches,
"content benchmark sentinel {label} returned too few matches: {observation:?}"
);
}
fn run_content_search_benchmark_sentinel() -> anyhow::Result<()> {
assert_eq!(sanitize_content_benchmark_path("./src")?, "src");
assert!(sanitize_content_benchmark_path("/tmp").is_err());
assert!(sanitize_content_benchmark_path("../outside").is_err());
let temp = TempDir::new()?;
fs::create_dir_all(temp.path().join("src"))?;
fs::write(
temp.path().join("src/a.rs"),
"before context\nneedle_plain alpha\nfn sentinel_func() {}\nlimit_token one\nlimit_token two\nlimit_token three\nafter context\n",
)?;
fs::write(
temp.path().join("src/b.rs"),
"needle_plain beta\nfn sentinel_other() {}\nlimit_token two\n",
)?;
fs::write(
temp.path().join("src/c.rs"),
"CASE_NEEDLE should not match lower-case benchmark query\n",
)?;
let runtime = ToolRuntime::new(temp.path())?;
let picker = ContentCandidatePicker::new(temp.path(), true)?;
let grep_plain = run_grep_content_observation(&runtime, "needle_plain", Some("src"), 10, 0, 1);
assert_content_sentinel_observation("grep_plain", &grep_plain, 2);
let candidate_plain = run_candidate_content_observation(
&picker,
"needle_plain",
ContentSearchMode::Plain,
10,
0,
1,
)?;
assert_content_sentinel_observation("candidate_plain", &candidate_plain, 2);
let grep_regex =
run_grep_content_observation(&runtime, "fn sentinel_func", Some("src"), 10, 0, 1);
assert_content_sentinel_observation("grep_regex", &grep_regex, 1);
let candidate_regex = run_candidate_content_observation(
&picker,
"fn sentinel_func",
ContentSearchMode::Regex,
10,
0,
1,
)?;
assert_content_sentinel_observation("candidate_regex", &candidate_regex, 1);
let grep_limit = run_grep_content_observation(&runtime, "limit_token", Some("src"), 1, 0, 1);
assert_eq!(
grep_limit.matches_returned, 1,
"content benchmark sentinel grep limit failed: {grep_limit:?}"
);
assert!(
grep_limit.limit_truncation_observed,
"content benchmark sentinel grep limit truncation flag failed: {grep_limit:?}"
);
let candidate_limit = run_candidate_content_observation(
&picker,
"limit_token",
ContentSearchMode::Plain,
1,
0,
1,
)?;
assert_eq!(
candidate_limit.matches_returned, 1,
"content benchmark sentinel candidate limit cap failed: {candidate_limit:?}"
);
assert!(
candidate_limit
.raw_matches_returned
.is_some_and(|raw| raw >= candidate_limit.matches_returned),
"content benchmark sentinel candidate raw limit count failed: {candidate_limit:?}"
);
assert!(
candidate_limit.limit_truncation_observed,
"content benchmark sentinel candidate limit truncation flag failed: {candidate_limit:?}"
);
assert!(
!candidate_limit.adapter_limit_mismatch_observed,
"content benchmark sentinel candidate adapter-visible limit mismatch failed: {candidate_limit:?}"
);
let grep_context =
run_grep_content_observation(&runtime, "needle_plain", Some("src"), 10, 1, 1);
assert_content_sentinel_observation("grep_context", &grep_context, 2);
let candidate_context = run_candidate_content_observation(
&picker,
"needle_plain",
ContentSearchMode::Plain,
10,
1,
1,
)?;
assert_content_sentinel_observation("candidate_context", &candidate_context, 2);
let grep_case_sensitive =
run_grep_content_observation(&runtime, "case_needle", Some("src"), 10, 0, 1);
assert_eq!(
grep_case_sensitive.matches_returned, 0,
"content benchmark sentinel grep case parity failed: {grep_case_sensitive:?}"
);
let candidate_case_sensitive = run_candidate_content_observation(
&picker,
"case_needle",
ContentSearchMode::Plain,
10,
0,
1,
)?;
assert_eq!(
candidate_case_sensitive.matches_returned, 0,
"content benchmark sentinel candidate case parity failed: {candidate_case_sensitive:?}"
);
Ok(())
}
#[expect(
clippy::too_many_arguments,
reason = "benchmark harness coordinates paired engines with explicit dimensions"
)]
fn run_content_search_engine_pair(
cfg: &ContentSearchBenchConfig,
scenario_count: &mut usize,
observed_values: &mut Vec<Value>,
scenario: &str,
pattern: &ContentSearchPattern,
path_label: Option<&str>,
grep_path: Option<&str>,
grep_runtime: &ToolRuntime,
candidate_picker: Option<&ContentCandidatePicker>,
limit: usize,
context: usize,
iterations: usize,
) -> anyhow::Result<()> {
if cfg.baseline.includes_grep() {
let before = ResourceSnapshot::capture();
let grep_pattern = pattern.grep_pattern();
let observation = run_grep_content_observation(
grep_runtime,
&grep_pattern,
grep_path,
limit,
context,
iterations,
);
let resources = before.delta(ResourceSnapshot::capture());
record_content_search_observation(
cfg,
scenario_count,
observed_values,
scenario,
pattern,
path_label,
limit,
context,
observation,
resources,
);
}
if cfg.baseline.includes_candidate()
&& let Some(candidate_picker) = candidate_picker
{
let before = ResourceSnapshot::capture();
let observation = run_candidate_content_observation(
candidate_picker,
&pattern.text,
pattern.mode,
limit,
context,
iterations,
)?;
let resources = before.delta(ResourceSnapshot::capture());
record_content_search_observation(
cfg,
scenario_count,
observed_values,
scenario,
pattern,
path_label,
limit,
context,
observation,
resources,
);
}
Ok(())
}
#[test]
#[ignore = "diagnostic-only content-search benchmark; run explicitly with --ignored --nocapture"]
fn content_search_benchmark_compares_ffgrep_native_and_candidate_engine() {
run_content_search_benchmark_sentinel().expect("content search benchmark sentinel");
let cfg = ContentSearchBenchConfig::from_env().expect("content search benchmark config");
let runtime = ToolRuntime::new(&cfg.root).expect("content search benchmark runtime");
let root_picker = cfg
.baseline
.includes_candidate()
.then(|| ContentCandidatePicker::new(&cfg.root, cfg.content_indexing))
.transpose()
.expect("content search benchmark root picker");
let existing_paths = cfg.existing_paths();
let max_limit = cfg.limits.iter().copied().max().unwrap_or(100);
let mut scenario_count = 0;
let mut observed_values = Vec::new();
if let Some(cold_pattern) = cfg.plain_patterns.first() {
for limit in &cfg.limits {
for context in &cfg.contexts {
if cfg.baseline.includes_grep() {
let cold_runtime = ToolRuntime::new(&cfg.root).expect("cold grep runtime");
let before = ResourceSnapshot::capture();
let grep_pattern = cold_pattern.grep_pattern();
let observation = run_grep_content_observation(
&cold_runtime,
&grep_pattern,
None,
*limit,
*context,
1,
);
let resources = before.delta(ResourceSnapshot::capture());
record_content_search_observation(
&cfg,
&mut scenario_count,
&mut observed_values,
"cold_root_plain",
cold_pattern,
None,
*limit,
*context,
observation,
resources,
);
}
if cfg.baseline.includes_candidate() {
let before = ResourceSnapshot::capture();
let started = Instant::now();
let cold_picker = ContentCandidatePicker::new(&cfg.root, cfg.content_indexing)
.expect("cold candidate content picker");
let mut observation = run_candidate_content_observation(
&cold_picker,
&cold_pattern.text,
ContentSearchMode::Plain,
*limit,
*context,
1,
)
.expect("cold candidate content observation");
observation.elapsed = vec![started.elapsed()];
let resources = before.delta(ResourceSnapshot::capture());
record_content_search_observation(
&cfg,
&mut scenario_count,
&mut observed_values,
"cold_root_plain",
cold_pattern,
None,
*limit,
*context,
observation,
resources,
);
}
}
}
}
for pattern in cfg.plain_patterns.iter().chain(cfg.regex_patterns.iter()) {
for limit in &cfg.limits {
for context in &cfg.contexts {
let scenario =
content_search_scenario_name(false, pattern.mode, *limit, *context, max_limit);
run_content_search_engine_pair(
&cfg,
&mut scenario_count,
&mut observed_values,
scenario,
pattern,
None,
None,
&runtime,
root_picker.as_ref(),
*limit,
*context,
cfg.repetitions,
)
.expect("warm root content scenario");
}
}
}
for path in &existing_paths {
let scoped_picker = cfg
.baseline
.includes_candidate()
.then(|| ContentCandidatePicker::new(&cfg.root.join(path), cfg.content_indexing))
.transpose()
.expect("scoped content benchmark picker");
for pattern in cfg.plain_patterns.iter().chain(cfg.regex_patterns.iter()) {
for limit in &cfg.limits {
for context in &cfg.contexts {
let scenario = content_search_scenario_name(
true,
pattern.mode,
*limit,
*context,
max_limit,
);
run_content_search_engine_pair(
&cfg,
&mut scenario_count,
&mut observed_values,
scenario,
pattern,
Some(path),
Some(path),
&runtime,
scoped_picker.as_ref(),
*limit,
*context,
cfg.repetitions,
)
.expect("warm scoped content scenario");
}
}
}
}
if cfg.cwd_churn > 0
&& let Some(pattern) = cfg.plain_patterns.first()
{
for dir in collect_churn_dirs(&cfg.root, cfg.cwd_churn) {
let path_label = dir
.strip_prefix(&cfg.root)
.ok()
.and_then(|path| path.to_str())
.map(|path| path.replace('\\', "/"))
.unwrap_or_else(|| "child".to_string());
let churn_runtime = runtime
.clone_for_cwd_with_subagent_depth(&dir, 0)
.expect("cwd churn runtime");
let churn_picker = cfg
.baseline
.includes_candidate()
.then(|| ContentCandidatePicker::new(&dir, cfg.content_indexing))
.transpose()
.expect("cwd churn picker");
run_content_search_engine_pair(
&cfg,
&mut scenario_count,
&mut observed_values,
"cwd_churn",
pattern,
Some(&path_label),
None,
&churn_runtime,
churn_picker.as_ref(),
max_limit,
0,
1,
)
.expect("cwd churn content scenario");
}
}
print_content_search_benchmark_summary(&cfg, scenario_count, &observed_values);
}