Skip to main content

git_vdb/
model.rs

1//! Public request, response, configuration, filtering, and mutation types.
2
3use serde::{Deserialize, Serialize};
4use serde_json::{Map, Value};
5use std::fmt;
6use std::str::FromStr;
7
8/// A JSON object used for point payloads.
9pub type JsonObject = Map<String, Value>;
10
11/// A typed point identifier.
12///
13/// String and unsigned-integer IDs occupy distinct namespaces and have a stable
14/// canonical ordering used to break equal-score query ties.
15#[derive(Clone, Debug, Eq, PartialEq, Hash, Serialize, Deserialize)]
16#[serde(untagged)]
17pub enum PointId {
18    /// A UTF-8 string identifier.
19    String(String),
20    /// An unsigned 64-bit integer identifier.
21    UInt(u64),
22}
23
24impl PointId {
25    pub(crate) fn canonical_bytes(&self) -> Vec<u8> {
26        match self {
27            Self::String(value) => {
28                let mut bytes = vec![b's', 0];
29                bytes.extend_from_slice(value.as_bytes());
30                bytes
31            }
32            Self::UInt(value) => {
33                let mut bytes = vec![b'u', 0];
34                bytes.extend_from_slice(&value.to_be_bytes());
35                bytes
36            }
37        }
38    }
39}
40
41impl From<&str> for PointId {
42    fn from(value: &str) -> Self {
43        Self::String(value.to_owned())
44    }
45}
46
47impl From<String> for PointId {
48    fn from(value: String) -> Self {
49        Self::String(value)
50    }
51}
52
53impl From<u64> for PointId {
54    fn from(value: u64) -> Self {
55        Self::UInt(value)
56    }
57}
58
59impl Ord for PointId {
60    fn cmp(&self, other: &Self) -> std::cmp::Ordering {
61        self.canonical_bytes().cmp(&other.canonical_bytes())
62    }
63}
64
65impl PartialOrd for PointId {
66    fn partial_cmp(&self, other: &Self) -> Option<std::cmp::Ordering> {
67        Some(self.cmp(other))
68    }
69}
70
71impl fmt::Display for PointId {
72    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
73        match self {
74            Self::String(value) => write!(f, "{value}"),
75            Self::UInt(value) => write!(f, "{value}"),
76        }
77    }
78}
79
80/// A hexadecimal Git object ID that identifies a tree or commit.
81#[derive(Clone, Debug, Eq, PartialEq, Hash, Serialize, Deserialize)]
82#[serde(transparent)]
83pub struct ObjectId(pub String);
84
85impl fmt::Display for ObjectId {
86    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
87        f.write_str(&self.0)
88    }
89}
90
91impl FromStr for ObjectId {
92    type Err = git2::Error;
93
94    fn from_str(value: &str) -> Result<Self, Self::Err> {
95        git2::Oid::from_str(value)?;
96        Ok(Self(value.to_owned()))
97    }
98}
99
100impl From<git2::Oid> for ObjectId {
101    fn from(value: git2::Oid) -> Self {
102        Self(value.to_string())
103    }
104}
105
106impl AsRef<str> for ObjectId {
107    fn as_ref(&self) -> &str {
108        &self.0
109    }
110}
111
112/// A vector and its optional JSON payload, identified by a typed ID.
113#[derive(Clone, Debug, Serialize, Deserialize, PartialEq)]
114pub struct Point {
115    /// The stable typed identifier for this point.
116    pub id: PointId,
117    /// The vector components, whose length must match the collection dimension.
118    pub vector: Vec<f32>,
119    /// Application-defined metadata used by filters and optional result output.
120    #[serde(default)]
121    pub payload: JsonObject,
122}
123
124impl Point {
125    /// Creates a point with an empty payload.
126    pub fn new(id: impl Into<PointId>, vector: impl IntoIterator<Item = f32>) -> Self {
127        Self {
128            id: id.into(),
129            vector: vector.into_iter().collect(),
130            payload: JsonObject::new(),
131        }
132    }
133
134    /// Replaces the point payload and returns the updated point.
135    #[must_use]
136    pub fn with_payload(mut self, payload: JsonObject) -> Self {
137        self.payload = payload;
138        self
139    }
140}
141
142/// The vector distance used for ranking.
143#[derive(Clone, Copy, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
144#[serde(rename_all = "snake_case")]
145#[non_exhaustive]
146pub enum Distance {
147    /// Cosine similarity, returned as a descending score.
148    #[default]
149    Cosine,
150}
151
152/// Deterministic LSH index construction and query defaults.
153#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
154#[serde(default)]
155pub struct IndexConfig {
156    /// Number of independent LSH tables stored per point.
157    pub tables: usize,
158    /// Number of projection bits in each table signature.
159    pub signature_bits: usize,
160    /// Seed used to deterministically derive projection vectors.
161    pub projection_seed: u64,
162    /// Point-count threshold at or below which queries default to exact search.
163    pub full_scan_threshold: usize,
164    /// Default number of approximate buckets to probe.
165    pub default_probes: usize,
166    /// Default maximum number of approximate candidates to discover.
167    pub default_candidate_limit: usize,
168}
169
170impl Default for IndexConfig {
171    fn default() -> Self {
172        Self {
173            tables: 12,
174            signature_bits: 12,
175            projection_seed: 0x6769_742d_7664_6231,
176            full_scan_threshold: 1_000,
177            default_probes: 96,
178            default_candidate_limit: 10_000,
179        }
180    }
181}
182
183/// Collection-wide vector and index configuration.
184#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
185#[serde(default)]
186pub struct CollectionConfig {
187    /// Required number of components in every stored and query vector.
188    pub dimension: usize,
189    /// Distance used to score vectors.
190    pub distance: Distance,
191    /// Optional application-defined vector-space identity checked by queries.
192    pub vector_space: Option<String>,
193    /// Deterministic approximate-index configuration.
194    pub index: IndexConfig,
195}
196
197impl CollectionConfig {
198    /// Creates a cosine collection configuration for the given dimension.
199    pub fn new(dimension: usize) -> Self {
200        Self {
201            dimension,
202            ..Self::default()
203        }
204    }
205
206    /// Assigns an application-defined vector-space identity.
207    #[must_use]
208    pub fn with_vector_space(mut self, vector_space: impl Into<String>) -> Self {
209        self.vector_space = Some(vector_space.into());
210        self
211    }
212
213    /// Replaces the deterministic LSH configuration.
214    #[must_use]
215    pub fn with_index(mut self, index: IndexConfig) -> Self {
216        self.index = index;
217        self
218    }
219}
220
221impl Default for CollectionConfig {
222    fn default() -> Self {
223        Self {
224            dimension: 0,
225            distance: Distance::Cosine,
226            vector_space: None,
227            index: IndexConfig::default(),
228        }
229    }
230}
231
232/// A JSON scalar value required by a field-match condition.
233#[derive(Clone, Debug, Serialize, Deserialize)]
234pub struct MatchValue {
235    /// The scalar value that must equal the payload field.
236    pub value: Value,
237}
238
239/// Inclusive or exclusive numeric bounds for a payload field.
240#[derive(Clone, Debug, Default, Serialize, Deserialize)]
241pub struct Range {
242    /// Exclusive lower bound.
243    pub gt: Option<f64>,
244    /// Inclusive lower bound.
245    pub gte: Option<f64>,
246    /// Exclusive upper bound.
247    pub lt: Option<f64>,
248    /// Inclusive upper bound.
249    pub lte: Option<f64>,
250}
251
252/// A predicate used inside a [`Filter`].
253#[derive(Clone, Debug, Serialize, Deserialize)]
254#[serde(untagged)]
255#[non_exhaustive]
256pub enum Condition {
257    /// Matches or range-checks a dot-separated payload field.
258    Field {
259        /// Dot-separated payload path.
260        key: String,
261        /// Optional equality match.
262        #[serde(rename = "match", skip_serializing_if = "Option::is_none")]
263        matches: Option<MatchValue>,
264        /// Optional numeric range.
265        #[serde(skip_serializing_if = "Option::is_none")]
266        range: Option<Range>,
267    },
268    /// Matches points whose typed ID is in the supplied set.
269    HasId {
270        /// Accepted typed IDs.
271        has_id: Vec<PointId>,
272    },
273    /// Evaluates a nested Boolean filter.
274    Nested(Filter),
275}
276
277impl Condition {
278    /// Creates an equality condition for a dot-separated payload path.
279    pub fn matches(key: impl Into<String>, value: impl Into<Value>) -> Self {
280        Self::Field {
281            key: key.into(),
282            matches: Some(MatchValue {
283                value: value.into(),
284            }),
285            range: None,
286        }
287    }
288
289    /// Creates a numeric-range condition for a dot-separated payload path.
290    pub fn range(key: impl Into<String>, range: Range) -> Self {
291        Self::Field {
292            key: key.into(),
293            matches: None,
294            range: Some(range),
295        }
296    }
297
298    /// Creates a condition that accepts the supplied typed point IDs.
299    pub fn has_id(ids: impl IntoIterator<Item = PointId>) -> Self {
300        Self::HasId {
301            has_id: ids.into_iter().collect(),
302        }
303    }
304}
305
306/// A Boolean point filter.
307///
308/// Every `must` condition and no `must_not` condition must match. When `should`
309/// is non-empty, at least one `should` condition must also match.
310#[derive(Clone, Debug, Default, Serialize, Deserialize)]
311#[serde(default)]
312pub struct Filter {
313    /// Conditions that must all match.
314    pub must: Vec<Condition>,
315    /// Conditions of which at least one must match when the list is non-empty.
316    pub should: Vec<Condition>,
317    /// Conditions that must not match.
318    pub must_not: Vec<Condition>,
319}
320
321impl Filter {
322    /// Creates a filter containing only required conditions.
323    pub fn must(conditions: impl IntoIterator<Item = Condition>) -> Self {
324        Self {
325            must: conditions.into_iter().collect(),
326            ..Self::default()
327        }
328    }
329}
330
331/// Exact or approximate query execution parameters.
332#[derive(Clone, Debug, Default, Serialize, Deserialize)]
333#[serde(default)]
334pub struct QueryParams {
335    /// Explicit mode override; `None` selects exact search for small collections.
336    pub exact: Option<bool>,
337    /// Approximate buckets to probe, or zero to use the collection default.
338    pub probes: usize,
339    /// Approximate candidate limit, or zero to use the collection default.
340    pub candidate_limit: usize,
341}
342
343/// A vector similarity query.
344#[derive(Clone, Debug, Serialize, Deserialize)]
345#[serde(default)]
346pub struct Query {
347    /// Query vector, which must match the collection dimension.
348    pub vector: Vec<f32>,
349    /// Maximum number of scored points to return.
350    pub limit: usize,
351    /// Optional point filter applied before final ranking.
352    pub filter: Option<Filter>,
353    /// Whether returned winners include their JSON payloads.
354    pub with_payload: bool,
355    /// Whether returned winners include their stored vectors.
356    pub with_vector: bool,
357    /// Optional vector-space identity that must match collection metadata.
358    pub expected_vector_space: Option<String>,
359    /// Exact or approximate execution controls.
360    pub params: QueryParams,
361}
362
363impl Query {
364    /// Creates a query using collection defaults for exact or approximate mode.
365    pub fn new(vector: impl IntoIterator<Item = f32>, limit: usize) -> Self {
366        Self {
367            vector: vector.into_iter().collect(),
368            limit,
369            ..Self::default()
370        }
371    }
372
373    /// Creates a query that scores every eligible point.
374    pub fn exact(vector: impl IntoIterator<Item = f32>, limit: usize) -> Self {
375        let mut query = Self::new(vector, limit);
376        query.params.exact = Some(true);
377        query
378    }
379
380    /// Creates a query that uses deterministic LSH candidate discovery.
381    pub fn approximate(vector: impl IntoIterator<Item = f32>, limit: usize) -> Self {
382        let mut query = Self::new(vector, limit);
383        query.params.exact = Some(false);
384        query
385    }
386
387    /// Adds a point filter.
388    #[must_use]
389    pub fn with_filter(mut self, filter: Filter) -> Self {
390        self.filter = Some(filter);
391        self
392    }
393
394    /// Requests payloads for returned winners.
395    #[must_use]
396    pub fn with_payload(mut self) -> Self {
397        self.with_payload = true;
398        self
399    }
400
401    /// Requests stored vectors for returned winners.
402    #[must_use]
403    pub fn with_vector(mut self) -> Self {
404        self.with_vector = true;
405        self
406    }
407
408    /// Requires the collection to use the supplied vector-space identity.
409    #[must_use]
410    pub fn in_vector_space(mut self, vector_space: impl Into<String>) -> Self {
411        self.expected_vector_space = Some(vector_space.into());
412        self
413    }
414
415    /// Replaces the exact or approximate execution parameters.
416    #[must_use]
417    pub fn with_params(mut self, params: QueryParams) -> Self {
418        self.params = params;
419        self
420    }
421}
422
423impl Default for Query {
424    fn default() -> Self {
425        Self {
426            vector: Vec::new(),
427            limit: 10,
428            filter: None,
429            with_payload: false,
430            with_vector: false,
431            expected_vector_space: None,
432            params: QueryParams::default(),
433        }
434    }
435}
436
437/// A scored query winner.
438#[derive(Clone, Debug, Serialize, Deserialize)]
439pub struct ScoredPoint {
440    /// Typed point identifier.
441    pub id: PointId,
442    /// Descending cosine similarity score.
443    pub score: f32,
444    /// Payload when requested by [`Query::with_payload`].
445    #[serde(skip_serializing_if = "Option::is_none")]
446    pub payload: Option<JsonObject>,
447    /// Stored vector when requested by [`Query::with_vector`].
448    #[serde(skip_serializing_if = "Option::is_none")]
449    pub vector: Option<Vec<f32>>,
450}
451
452/// Query algorithm selected after applying collection defaults.
453#[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize, Deserialize)]
454#[serde(rename_all = "snake_case")]
455#[non_exhaustive]
456pub enum QueryMode {
457    /// Every filter-eligible point was scored.
458    Exact,
459    /// Candidates were discovered through deterministic LSH buckets.
460    Approximate,
461}
462
463/// Work counters for a completed query.
464#[derive(Clone, Debug, Serialize, Deserialize)]
465pub struct QueryStats {
466    /// Algorithm selected for the query.
467    pub mode: QueryMode,
468    /// Total points in the resolved collection root.
469    pub collection_points: usize,
470    /// LSH buckets visited; zero for exact search.
471    pub buckets_probed: usize,
472    /// Distinct approximate candidates discovered.
473    pub candidates_discovered: usize,
474    /// Point vectors actually scored.
475    pub vectors_scored: usize,
476    /// Whether approximate discovery consumed its probe budget.
477    pub probe_limit_exhausted: bool,
478    /// Whether approximate discovery consumed its candidate budget.
479    pub candidate_limit_exhausted: bool,
480}
481
482/// Ordered similarity-search results and execution statistics.
483#[derive(Clone, Debug, Serialize, Deserialize)]
484pub struct QueryResult {
485    /// Immutable root that was queried.
486    pub root: ObjectId,
487    /// Winners ordered by descending score and canonical typed ID.
488    pub points: Vec<ScoredPoint>,
489    /// Algorithm and work counters.
490    pub stats: QueryStats,
491}
492
493/// A deterministic point-retrieval request.
494#[derive(Clone, Debug, Default, Serialize, Deserialize)]
495#[serde(default)]
496pub struct GetRequest {
497    /// Optional typed IDs; when combined with a filter, both must match.
498    pub ids: Vec<PointId>,
499    /// Optional payload or ID filter.
500    pub filter: Option<Filter>,
501    /// Number of canonically ordered matches to skip.
502    pub offset: usize,
503    /// Maximum matches to return, or all remaining matches when `None`.
504    pub limit: Option<usize>,
505    /// Whether results include payloads.
506    pub with_payload: bool,
507    /// Whether results include vectors.
508    pub with_vector: bool,
509}
510
511/// A point returned without similarity scoring.
512#[derive(Clone, Debug, Serialize, Deserialize)]
513pub struct Record {
514    /// Typed point identifier.
515    pub id: PointId,
516    /// Payload when requested.
517    #[serde(skip_serializing_if = "Option::is_none")]
518    pub payload: Option<JsonObject>,
519    /// Stored vector when requested.
520    #[serde(skip_serializing_if = "Option::is_none")]
521    pub vector: Option<Vec<f32>>,
522}
523
524/// Canonically ordered point-retrieval results.
525#[derive(Clone, Debug, Serialize, Deserialize)]
526pub struct GetResult {
527    /// Immutable root that was read.
528    pub root: ObjectId,
529    /// Matching records after offset and limit are applied.
530    pub points: Vec<Record>,
531}
532
533/// Selects the union of typed IDs and filter matches for deletion.
534#[derive(Clone, Debug, Default, Serialize, Deserialize)]
535#[serde(default)]
536pub struct DeleteSelector {
537    /// Typed IDs to delete; missing IDs are ignored.
538    pub ids: Vec<PointId>,
539    /// Optional filter whose matches are also deleted.
540    pub filter: Option<Filter>,
541}
542
543/// The outcome of a named collection write.
544#[derive(Clone, Debug, Serialize, Deserialize)]
545pub struct WriteResult {
546    /// New deterministic collection root.
547    pub root: ObjectId,
548    /// Number of submitted upserts or points actually removed.
549    pub affected_points: usize,
550}
551
552/// A count and the immutable root from which it was read.
553#[derive(Clone, Debug, Serialize, Deserialize)]
554pub struct CountResult {
555    /// Immutable root that was counted.
556    pub root: ObjectId,
557    /// Number of matching points.
558    pub count: usize,
559}
560
561/// Metadata for a named or historical collection view.
562#[derive(Clone, Debug, Serialize, Deserialize)]
563pub struct CollectionInfo {
564    /// Resolved deterministic root.
565    pub root: ObjectId,
566    /// Named collection.
567    pub name: String,
568    /// Number of points at the resolved root.
569    pub point_count: usize,
570    /// Collection configuration stored in the root.
571    pub config: CollectionConfig,
572    /// Whether the view is historical and rejects writes.
573    pub read_only: bool,
574}
575
576/// Metadata for an immutable root snapshot.
577#[derive(Clone, Debug, Serialize, Deserialize)]
578pub struct SnapshotInfo {
579    /// Deterministic root tree ID.
580    pub root: ObjectId,
581    /// Number of points in the root.
582    pub point_count: usize,
583    /// Collection configuration stored in the root.
584    pub config: CollectionConfig,
585}
586
587/// One operation in an ordered immutable-root mutation batch.
588#[derive(Clone, Debug, Serialize, Deserialize)]
589#[serde(tag = "operation", rename_all = "snake_case")]
590#[non_exhaustive]
591pub enum SnapshotMutation {
592    /// Adds a new point or replaces an existing point with the same typed ID.
593    Upsert {
594        /// Complete replacement point.
595        point: Point,
596    },
597    /// Deletes the supplied typed IDs.
598    DeleteIds {
599        /// Typed IDs to delete; missing IDs are ignored.
600        ids: Vec<PointId>,
601    },
602    /// Deletes every point matching a filter.
603    DeleteFilter {
604        /// Filter evaluated against the preceding mutation state.
605        filter: Filter,
606    },
607}
608
609impl SnapshotMutation {
610    /// Creates an upsert mutation.
611    pub fn upsert(point: Point) -> Self {
612        Self::Upsert { point }
613    }
614
615    /// Creates a typed-ID deletion mutation.
616    pub fn delete_ids(ids: impl IntoIterator<Item = PointId>) -> Self {
617        Self::DeleteIds {
618            ids: ids.into_iter().collect(),
619        }
620    }
621
622    /// Creates a filter deletion mutation.
623    pub fn delete_filter(filter: Filter) -> Self {
624        Self::DeleteFilter { filter }
625    }
626}
627
628/// One named-collection commit in newest-first history order.
629#[derive(Clone, Debug, Serialize, Deserialize)]
630pub struct HistoryEntry {
631    /// Commit object ID.
632    pub commit: ObjectId,
633    /// Canonical root tree recorded by the commit.
634    pub root: ObjectId,
635    /// First parent commit, when present.
636    pub parent: Option<ObjectId>,
637    /// Commit message generated for the collection operation.
638    pub message: String,
639    /// Commit time in Unix seconds.
640    pub time_seconds: i64,
641}
642
643/// Count and logical size of a set of reachable Git objects.
644#[derive(Clone, Debug, Default, Serialize, Deserialize)]
645pub struct ObjectStats {
646    /// Number of Git objects.
647    pub objects: usize,
648    /// Sum of logical object bytes.
649    pub bytes: usize,
650}
651
652/// Logical point and structural-sharing differences between two roots.
653#[derive(Clone, Debug, Serialize, Deserialize)]
654pub struct DiffResult {
655    /// Resolved left root.
656    pub left_root: ObjectId,
657    /// Resolved right root.
658    pub right_root: ObjectId,
659    /// IDs present only on the right.
660    pub added: Vec<PointId>,
661    /// IDs present only on the left.
662    pub removed: Vec<PointId>,
663    /// IDs present in both roots with changed point content.
664    pub changed: Vec<PointId>,
665    /// Whether collection metadata differs.
666    pub configuration_changed: bool,
667    /// Whether any approximate-index buckets differ.
668    pub buckets_changed: bool,
669    /// Objects reachable from both roots.
670    pub shared: ObjectStats,
671    /// Objects reachable only from the left root.
672    pub left_unique: ObjectStats,
673    /// Objects reachable only from the right root.
674    pub right_unique: ObjectStats,
675}
676
677/// Result of basic or full canonical-root validation.
678#[derive(Clone, Debug, Serialize, Deserialize)]
679pub struct ValidationReport {
680    /// Root that was validated.
681    pub root: ObjectId,
682    /// Whether expensive index recomputation was requested.
683    pub full: bool,
684    /// Validated point count.
685    pub point_count: usize,
686    /// Approximate-index buckets checked during full validation.
687    pub checked_buckets: usize,
688    /// `true` when validation completed without finding corruption.
689    pub valid: bool,
690}