bobbin-ai 0.25.2

Local-first context injection engine for AI coding agents
name: CI

on:
  push:
    branches: [main]
  pull_request:

env:
  CARGO_TERM_COLOR: always

jobs:
  test:
    name: Test
    runs-on: ubuntu-latest
    steps:
      - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4

      # A listed hook is not proof it executes: require both a benign control
      # and synthetic private-key refusals before scanning the tracked tree.
      - uses: actions/setup-python@v5
        with:
          python-version: "3.12"
      - name: Install private-key scanner
        run: python3 -m pip install pre-commit==4.6.1 pyyaml
      - name: Private-key refusal controls and tracked-file scan
        run: python3 scripts/check-private-keys.py

      - name: Release deploy checksum regression
        run: bash scripts/test-deploy-from-release.sh

      - name: Crates publication contract
        run: python3 scripts/test-crates-workflow.py

      - name: Install system dependencies
        run: sudo apt-get update && sudo apt-get install -y protobuf-compiler cmake g++

      # ONNX Runtime, without which `bobbin index` cannot run at all.
      #
      # WHY (aegis-pnm0uo). `ort` is built with `load-dynamic`, so it resolves
      # libonnxruntime.so at RUNTIME via ORT_DYLIB_PATH. A bare runner has no
      # such library, so every `bobbin index` panicked with
      #   Failed to load ONNX Runtime dylib: ... dlopen failed
      # and exited 101. Ten of the eleven cli_index tests opened with
      # `if !project.bobbin_index() { return; }`, and a returning #[test] has
      # PASSED — so the suite reported 11 passed while 1 had actually run, for as
      # long as that pattern existed. Indexing is bobbin's core function; a green
      # CI was not weak evidence about it, it was no evidence at all.
      #
      # Pinned to the version the deployed host installs (goldblum
      # bobbin_onnxruntime_gpu_version), so CI and production resolve the same
      # runtime rather than drifting apart silently. The CPU build is used here:
      # runners have no GPU and the tests do not need one.
      - name: Install ONNX Runtime
        run: |
          set -euo pipefail
          version=1.24.1
          curl -fsSL -o ort.tgz \
            "https://github.com/microsoft/onnxruntime/releases/download/v${version}/onnxruntime-linux-x64-${version}.tgz"
          tar xzf ort.tgz
          lib="$PWD/onnxruntime-linux-x64-${version}/lib/libonnxruntime.so"
          test -s "$lib"   # refuse to continue on a truncated or missing download
          echo "ORT_DYLIB_PATH=$lib" >> "$GITHUB_ENV"

      - name: Setup Rust
        uses: dtolnay/rust-toolchain@stable
        with:
          components: clippy

      - name: Cache cargo
        uses: Swatinem/rust-cache@779680da715d629ac1d338a641029a2f4372abb5 # v2

      # `knowledge` gates bobbin's entire Quipu integration (MCP surface, coupling
      # exporter, PPR reranking) and is NOT a cargo default feature. CI never passed
      # it, so none of that code was ever compiled here and its tests passed while
      # guarding nothing (bobbin-jdlkh). Build BOTH ways, every time:
      #   --features knowledge  -> the integration is actually compiled + tested
      #   (no features)         -> the default build, incl. the cfg(not(knowledge))
      #                            guard that refuses a PPR request it cannot serve
      # Dropping either arm re-creates the bug: whichever side goes unbuilt goes dark.
      - name: Check (knowledge)
        run: cargo check --features knowledge

      - name: Check (default, no features)
        run: cargo check

      - name: Test (knowledge)
        run: cargo test --features knowledge

      - name: Test (default, no features)
        run: cargo test

      - name: Clippy (knowledge)
        run: cargo clippy --features knowledge

      - name: Clippy (default, no features)
        run: cargo clippy

      # The file-size ratchet ran only from `just check`, i.e. only for whoever
      # remembered to run it. That is how `src/cli/hook.rs` grew 378 lines
      # inside its allowlist entry without a single red build. An allowlist
      # entry is a CEILING now, so the gate has something to enforce — enforce
      # it where nobody can skip it.
      - name: File size ratchet
        run: bash scripts/check-file-size.sh --all

      # A Linux release must become deployable when its own matrix leg finishes;
      # waiting for unrelated macOS legs caused a false deploy page on every
      # slow release (aegis-if8dp).  This is a static workflow contract because
      # the failure is orchestration order, not Rust behavior.
      - name: Release asset publication contract
        run: python3 scripts/test-release-workflow-contract.py

  # The eval framework's own 289 tests were run by NOTHING: `just test` is
  # cargo-only, `just check` does not reach them, and no workflow mentioned
  # them. Sixteen files with a declared pytest dev-dependency and a pyproject,
  # clearly meant to run, silently unexecuted.
  #
  # That matters more than the count. This framework is what measures injection
  # quality — the thing GH#54 proposes to A/B an embedding-model upgrade
  # against. An unrun scorer cannot be trusted to say a change helped, so the
  # measuring instrument needs a gate before the measurements do.
  #
  # Same defect as the file-size ratchet above, one language over: a real check
  # that only ran for whoever remembered to type the command.
  eval-tests:
    name: Eval framework tests
    runs-on: ubuntu-latest
    steps:
      - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4

      - uses: actions/setup-python@v5
        with:
          python-version: "3.11"

      # Installed explicitly rather than via `pip install -e .[dev]`: the
      # runtime deps include `anthropic`, and the tests neither need it nor
      # should a network-calling client be pulled into a gate that must stay
      # hermetic. matplotlib/pandas are here because test_mpl_charts.py imports
      # them at module scope, so their absence is a collection ERROR that takes
      # the whole run down rather than a skip.
      - name: Install test dependencies
        run: pip install pytest pyyaml click jinja2 matplotlib pandas

      - name: Run eval tests
        working-directory: eval
        run: python -m pytest tests/ scorer/test_scorer.py -q