bobbin-ai 0.25.2

Local-first context injection engine for AI coding agents
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
use serde::{Deserialize, Serialize};

/// A semantic chunk extracted from a source file
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Chunk {
    pub id: String,
    pub file_path: String,
    pub chunk_type: ChunkType,
    pub name: Option<String>,
    pub start_line: u32,
    pub end_line: u32,
    pub content: String,
    pub language: String,
    /// Comma-separated sorted tags (empty string = untagged)
    #[serde(default)]
    pub tags: String,
}

/// Types of semantic chunks that can be extracted
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ChunkType {
    Function,
    Method,
    Class,
    Struct,
    Enum,
    Interface,
    Module,
    Impl,
    Trait,
    Doc,
    Section,
    Table,
    CodeBlock,
    Commit,
    Issue,
    Other,
}

impl std::fmt::Display for ChunkType {
    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
        match self {
            ChunkType::Function => write!(f, "function"),
            ChunkType::Method => write!(f, "method"),
            ChunkType::Class => write!(f, "class"),
            ChunkType::Struct => write!(f, "struct"),
            ChunkType::Enum => write!(f, "enum"),
            ChunkType::Interface => write!(f, "interface"),
            ChunkType::Module => write!(f, "module"),
            ChunkType::Impl => write!(f, "impl"),
            ChunkType::Trait => write!(f, "trait"),
            ChunkType::Doc => write!(f, "doc"),
            ChunkType::Section => write!(f, "section"),
            ChunkType::Table => write!(f, "table"),
            ChunkType::CodeBlock => write!(f, "code_block"),
            ChunkType::Commit => write!(f, "commit"),
            ChunkType::Issue => write!(f, "issue"),
            ChunkType::Other => write!(f, "other"),
        }
    }
}

impl ChunkType {
    /// Whether a named chunk of this type carries a code symbol — the set
    /// the entity extractor mints `CodeSymbol` entities for and the mention
    /// emitter treats as symbol-bearing (they must agree, or a chunk can
    /// mention a name no entity extraction would ever mint).
    pub fn is_code_symbol(self) -> bool {
        matches!(
            self,
            ChunkType::Function
                | ChunkType::Method
                | ChunkType::Class
                | ChunkType::Struct
                | ChunkType::Enum
                | ChunkType::Interface
                | ChunkType::Trait
                | ChunkType::Impl
        )
    }
}

/// Metadata about an indexed file
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct FileMetadata {
    pub path: String,
    pub language: Option<String>,
    pub mtime: i64,
    pub hash: String,
    pub indexed_at: i64,
}

/// A search result with relevance score
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct SearchResult {
    pub chunk: Chunk,
    pub score: f32,
    #[serde(skip_serializing_if = "Option::is_none")]
    pub match_type: Option<MatchType>,
    /// Unix timestamp when this chunk was indexed (used for recency boosting)
    #[serde(skip_serializing_if = "Option::is_none")]
    pub indexed_at: Option<i64>,
    /// Repository name this result belongs to (from LanceDB repo column)
    #[serde(skip_serializing_if = "Option::is_none", default)]
    pub repo: Option<String>,
}

/// Derive the source kind from a chunk type for display purposes.
/// Returns "issue" for beads issues, "commit" for git commits, "code" for everything else.
pub fn source_kind(chunk_type: &ChunkType) -> &'static str {
    match chunk_type {
        ChunkType::Issue => "issue",
        ChunkType::Commit => "commit",
        _ => "code",
    }
}

/// How a result was matched
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum MatchType {
    Semantic,
    Keyword,
    Hybrid,
}

/// Temporal coupling between two files
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct FileCoupling {
    pub file_a: String,
    pub file_b: String,
    pub score: f32,
    pub co_changes: u32,
    pub last_co_change: i64,
}

/// Cross-repo coupling between two files in DIFFERENT repositories (bo-oqny).
///
/// Inferred from bead-reference co-occurrence: a bead id appearing in the git
/// history (bead trailers) of two repos in the same `GroupConfig` links the
/// files those commits touched. Both sides carry their repo because the stored
/// paths are repo-relative and collide across repos (e.g. `src/main.rs`).
/// Canonicalized so `(repo_a, path_a) <= (repo_b, path_b)` to dedupe.
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
pub struct CrossRepoCoupling {
    pub repo_a: String,
    pub path_a: String,
    pub repo_b: String,
    pub path_b: String,
    pub score: f32,
    pub co_changes: u32,
    pub last_co_change: i64,
}

/// A raw import statement extracted from source code
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct RawImport {
    /// The full verbatim import statement text (e.g., "use crate::auth::middleware;")
    pub statement: String,
    /// The extracted import path (e.g., "crate::auth::middleware")
    pub path: String,
    /// The categorized import type: "use", "import", "require", "from", "include"
    pub dep_type: String,
}

/// An import/dependency edge between two files (used during parsing/resolution)
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ImportEdge {
    /// The file that contains the import statement
    pub source_file: String,
    /// The raw import specifier as written in source
    pub import_specifier: String,
    /// The resolved file path (if resolution succeeded)
    pub resolved_path: Option<String>,
    /// The language of the source file
    pub language: String,
}

/// A stored import dependency edge between two files
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct ImportDependency {
    /// Importer (source file)
    pub file_a: String,
    /// Imported (target file or "unresolved:<path>")
    pub file_b: String,
    /// Dependency type: "use", "import", "require", "from", "include"
    pub dep_type: String,
    /// Raw import statement text
    pub import_statement: String,
    /// What's imported (nullable)
    pub symbol: Option<String>,
    /// True if file_b is a real file path
    pub resolved: bool,
}

/// A typed relationship between two chunks (symbol-level edges).
///
/// Unlike `ImportDependency` (file-to-file), chunk edges link specific symbols:
/// impl→trait, test→function, method→containing impl, etc.
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct ChunkEdge {
    /// Source chunk ID from the parser. Ordinary IDs hash the path and line
    /// range; colliding syntax spans carry a byte-range suffix. Treat IDs as opaque.
    pub source_chunk: String,
    /// Target chunk ID
    pub target_chunk: String,
    /// Source chunk name (for display/query without re-reading chunks table)
    pub source_name: String,
    /// Target chunk name
    pub target_name: String,
    /// Relationship type
    pub edge_type: ChunkEdgeType,
    /// Source file path (denormalized for queries)
    pub file_path: String,
}

/// Types of chunk-level relationships.
///
/// `Implements`/`ImplFor`/`Tests`/`Extends` are extracted by tree-sitter;
/// `NextChunk`/`PartOf` are deterministic structural edges derived from
/// chunk positions and markdown heading hierarchy; `SimilarTo` is written
/// opt-in by the near-duplicate scan (`bobbin similar --scan --persist`).
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ChunkEdgeType {
    /// impl block implements a trait (Rust: `impl Trait for Struct`)
    Implements,
    /// impl block targets a struct (Rust: `impl Struct`)
    ImplFor,
    /// Test function tests a production symbol (inferred from name/attributes)
    Tests,
    /// Class/struct extends another (Java/TS extends, Python inheritance)
    Extends,
    /// Source chunk is immediately followed by target chunk in document order
    NextChunk,
    /// Source chunk (child) is contained in target chunk (parent):
    /// fn in impl, section under parent heading, table/code block in section
    PartOf,
    /// Source chunk is a semantic near-duplicate of target chunk
    /// (cosine similarity above the scan threshold at persist time)
    SimilarTo,
}

impl ChunkEdgeType {
    /// Every variant, for storage sites that enumerate edge types.
    pub const ALL: [ChunkEdgeType; 7] = [
        ChunkEdgeType::Implements,
        ChunkEdgeType::ImplFor,
        ChunkEdgeType::Tests,
        ChunkEdgeType::Extends,
        ChunkEdgeType::NextChunk,
        ChunkEdgeType::PartOf,
        ChunkEdgeType::SimilarTo,
    ];
}

impl std::fmt::Display for ChunkEdgeType {
    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
        match self {
            ChunkEdgeType::Implements => write!(f, "implements"),
            ChunkEdgeType::ImplFor => write!(f, "impl_for"),
            ChunkEdgeType::Tests => write!(f, "tests"),
            ChunkEdgeType::Extends => write!(f, "extends"),
            ChunkEdgeType::NextChunk => write!(f, "next_chunk"),
            ChunkEdgeType::PartOf => write!(f, "part_of"),
            ChunkEdgeType::SimilarTo => write!(f, "similar_to"),
        }
    }
}

/// A knowledge graph entity with an embedding vector.
///
/// Entities represent higher-level concepts (CodeModule, CodeSymbol, Section,
/// Bundle) from the Quipu knowledge graph. Each entity gets a single embedding
/// in Bobbin's LanceDB `entities` table, enabling semantic search over the
/// knowledge graph.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Entity {
    /// Unique IRI identifying this entity (e.g., `bobbin:code/repo/path::symbol`)
    pub entity_iri: String,
    /// Text content used to generate the embedding
    pub text: String,
    /// Entity type (e.g., "CodeModule", "CodeSymbol", "Section", "Bundle")
    pub entity_type: String,
    /// Repository this entity belongs to (None for cross-repo entities)
    pub repo: Option<String>,
}

/// A search result from the entities table
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct EntitySearchResult {
    /// The matched entity IRI
    pub entity_iri: String,
    /// The entity's text content
    pub text: String,
    /// The entity type
    pub entity_type: String,
    /// Repository name (if any)
    pub repo: Option<String>,
    /// Similarity score (0.0 to 1.0, higher is better)
    pub score: f32,
}

/// Classification of a file by its role in the project.
///
/// The four built-in categories cover common cases. Custom categories can be
/// defined via `[[file_types]]` config rules for project-specific needs
/// (e.g., "generated", "vendor", "schema", "migration").
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum FileCategory {
    Source,
    Test,
    Documentation,
    Config,
    /// User-defined category from `[[file_types]]` config
    #[serde(untagged)]
    Custom(String),
}

impl FileCategory {
    /// Parse a category name string into a FileCategory.
    /// Recognizes built-in names; anything else becomes Custom.
    pub fn from_name(name: &str) -> Self {
        match name.to_lowercase().as_str() {
            "source" => FileCategory::Source,
            "test" => FileCategory::Test,
            "documentation" | "doc" | "docs" => FileCategory::Documentation,
            "config" | "configuration" => FileCategory::Config,
            _ => FileCategory::Custom(name.to_string()),
        }
    }

    /// Whether this category is treated as "documentation-like" for demotion purposes.
    /// Custom categories are NOT demoted by default — only Documentation and Config.
    pub fn is_doc_like(&self) -> bool {
        matches!(self, FileCategory::Documentation | FileCategory::Config)
    }
}

impl std::fmt::Display for FileCategory {
    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
        match self {
            FileCategory::Source => write!(f, "source"),
            FileCategory::Test => write!(f, "test"),
            FileCategory::Documentation => write!(f, "documentation"),
            FileCategory::Config => write!(f, "config"),
            FileCategory::Custom(name) => write!(f, "{}", name),
        }
    }
}

/// Classify a file using configurable rules, falling back to built-in heuristics.
///
/// Rules are evaluated in order; first matching glob wins.
/// If no rule matches, falls back to `classify_file()` built-in heuristics.
pub fn classify_file_with_rules(path: &str, rules: &[crate::config::FileTypeRule]) -> FileCategory {
    if !rules.is_empty() {
        let lower = path.to_lowercase();
        for rule in rules {
            for pattern in &rule.patterns {
                if let Ok(matcher) = glob::Pattern::new(&pattern.to_lowercase()) {
                    if matcher.matches(&lower) {
                        return FileCategory::from_name(&rule.name);
                    }
                }
            }
        }
    }
    classify_file(path)
}

/// Classify a file path into a category based on path heuristics.
/// Default is Source (conservative — only well-known patterns trigger other categories).
pub fn classify_file(path: &str) -> FileCategory {
    let lower = path.to_lowercase();
    let parts: Vec<&str> = lower.split('/').collect();
    let filename = parts.last().copied().unwrap_or("");

    // Documentation: known doc filenames and extensions
    let doc_names = [
        "changelog",
        "changelog.md",
        "changelog.rst",
        "changelog.txt",
        "changes",
        "changes.md",
        "changes.rst",
        "breaking_changes.md",
        "breaking_changes.rst",
        "history.md",
        "history.rst",
        "readme",
        "readme.md",
        "readme.rst",
        "readme.txt",
        "contributing.md",
        "contributing.rst",
        "license",
        "license.md",
        "license.txt",
        "code_of_conduct.md",
    ];
    if doc_names.contains(&filename) {
        return FileCategory::Documentation;
    }

    // Documentation: doc directories
    let doc_dirs = ["docs", "doc", "changelogs", "documentation"];
    for part in &parts[..parts.len().saturating_sub(1)] {
        if doc_dirs.contains(part) {
            // Files in doc dirs with code extensions are still source
            if has_code_extension(filename) {
                return FileCategory::Source;
            }
            return FileCategory::Documentation;
        }
    }

    // Documentation: doc extensions (only if not in a source-like directory)
    let doc_extensions = [".md", ".mdx", ".rst", ".txt"];
    for ext in &doc_extensions {
        if lower.ends_with(ext) {
            return FileCategory::Documentation;
        }
    }

    // Test: test directories and naming patterns
    let test_dirs = [
        "test",
        "tests",
        "spec",
        "specs",
        "__tests__",
        "test_fixtures",
        "testdata",
    ];
    for part in &parts[..parts.len().saturating_sub(1)] {
        if test_dirs.contains(part) {
            return FileCategory::Test;
        }
    }
    // Test: file naming patterns
    if filename.starts_with("test_")
        || filename.contains("_test.")
        || filename.contains("_spec.")
        || filename.contains(".test.")
        || filename.contains(".spec.")
        || filename.ends_with("_test.rs")
        || filename.ends_with("_test.py")
        || filename.ends_with("_test.go")
    {
        return FileCategory::Test;
    }
    // Snapshot directories
    if parts
        .iter()
        .any(|p| *p == "__snapshots__" || *p == "snapshots")
    {
        return FileCategory::Test;
    }

    // Config: known config files and extensions
    let config_names = [
        "cargo.toml",
        "cargo.lock",
        "package.json",
        "package-lock.json",
        "yarn.lock",
        "pnpm-lock.yaml",
        "makefile",
        "justfile",
        ".gitignore",
        ".gitattributes",
        ".editorconfig",
        "pyproject.toml",
        "setup.py",
        "setup.cfg",
        "tsconfig.json",
        "babel.config.js",
        "webpack.config.js",
        "dockerfile",
        "docker-compose.yml",
        "docker-compose.yaml",
        ".eslintrc.js",
        ".eslintrc.json",
        ".prettierrc",
        "renovate.json",
        "dependabot.yml",
        "rustfmt.toml",
        "clippy.toml",
        ".clippy.toml",
    ];
    if config_names.contains(&filename) {
        return FileCategory::Config;
    }
    // Config: config directories
    if parts
        .iter()
        .any(|p| *p == ".github" || *p == ".circleci" || *p == ".vscode")
    {
        return FileCategory::Config;
    }
    // Config: YAML/TOML at root level (1 part = just filename)
    let config_extensions = [".yaml", ".yml", ".toml", ".ini", ".cfg"];
    if parts.len() <= 2 {
        for ext in &config_extensions {
            if lower.ends_with(ext) {
                return FileCategory::Config;
            }
        }
    }

    FileCategory::Source
}

/// Check if a filename has a code extension (used to avoid classifying source in doc dirs)
fn has_code_extension(filename: &str) -> bool {
    let code_extensions = [
        ".rs", ".py", ".js", ".ts", ".tsx", ".jsx", ".go", ".java", ".c", ".cpp", ".h", ".hpp",
        ".cs", ".rb", ".swift", ".kt", ".scala", ".zig", ".hs", ".ml", ".ex", ".exs", ".sh",
        ".bash", ".zsh", ".fish", ".lua", ".r", ".jl", ".pl", ".php",
    ];
    code_extensions.iter().any(|ext| filename.ends_with(ext))
}

/// Statistics about the index
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct IndexStats {
    pub total_files: u64,
    pub total_chunks: u64,
    pub total_embeddings: u64,
    pub languages: Vec<LanguageStats>,
    pub last_indexed: Option<i64>,
    pub index_size_bytes: u64,
}

/// Per-language statistics
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct LanguageStats {
    pub language: String,
    pub file_count: u64,
    pub chunk_count: u64,
}

#[cfg(test)]
#[path = "types_tests.rs"]
mod tests;