Skip to main content

git_vdb/
model.rs

1//! Public request, response, configuration, filtering, and mutation types.
2
3use serde::{Deserialize, Serialize};
4use serde_json::{Map, Value};
5use std::fmt;
6use std::str::FromStr;
7
8/// A JSON object used for point payloads.
9pub type JsonObject = Map<String, Value>;
10
11/// A typed point identifier.
12///
13/// String and unsigned-integer IDs occupy distinct namespaces and have a stable
14/// canonical ordering used to break equal-score query ties.
15#[derive(Clone, Debug, Eq, PartialEq, Hash, Serialize, Deserialize)]
16#[serde(untagged)]
17pub enum PointId {
18    /// A UTF-8 string identifier.
19    String(String),
20    /// An unsigned 64-bit integer identifier.
21    UInt(u64),
22}
23
24impl PointId {
25    pub(crate) fn canonical_bytes(&self) -> Vec<u8> {
26        match self {
27            Self::String(value) => {
28                let mut bytes = vec![b's', 0];
29                bytes.extend_from_slice(value.as_bytes());
30                bytes
31            }
32            Self::UInt(value) => {
33                let mut bytes = vec![b'u', 0];
34                bytes.extend_from_slice(&value.to_be_bytes());
35                bytes
36            }
37        }
38    }
39}
40
41impl From<&str> for PointId {
42    fn from(value: &str) -> Self {
43        Self::String(value.to_owned())
44    }
45}
46
47impl From<String> for PointId {
48    fn from(value: String) -> Self {
49        Self::String(value)
50    }
51}
52
53impl From<u64> for PointId {
54    fn from(value: u64) -> Self {
55        Self::UInt(value)
56    }
57}
58
59impl Ord for PointId {
60    fn cmp(&self, other: &Self) -> std::cmp::Ordering {
61        self.canonical_bytes().cmp(&other.canonical_bytes())
62    }
63}
64
65impl PartialOrd for PointId {
66    fn partial_cmp(&self, other: &Self) -> Option<std::cmp::Ordering> {
67        Some(self.cmp(other))
68    }
69}
70
71impl fmt::Display for PointId {
72    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
73        match self {
74            Self::String(value) => write!(f, "{value}"),
75            Self::UInt(value) => write!(f, "{value}"),
76        }
77    }
78}
79
80/// A hexadecimal Git object ID that identifies a tree or commit.
81#[derive(Clone, Debug, Eq, PartialEq, Hash, Serialize, Deserialize)]
82#[serde(transparent)]
83pub struct ObjectId(pub String);
84
85impl fmt::Display for ObjectId {
86    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
87        f.write_str(&self.0)
88    }
89}
90
91impl FromStr for ObjectId {
92    type Err = git2::Error;
93
94    fn from_str(value: &str) -> Result<Self, Self::Err> {
95        git2::Oid::from_str(value)?;
96        Ok(Self(value.to_owned()))
97    }
98}
99
100impl From<git2::Oid> for ObjectId {
101    fn from(value: git2::Oid) -> Self {
102        Self(value.to_string())
103    }
104}
105
106impl AsRef<str> for ObjectId {
107    fn as_ref(&self) -> &str {
108        &self.0
109    }
110}
111
112/// A vector and its optional JSON payload, identified by a typed ID.
113#[derive(Clone, Debug, Serialize, Deserialize, PartialEq)]
114pub struct Point {
115    /// The stable typed identifier for this point.
116    pub id: PointId,
117    /// The vector components, whose length must match the collection dimension.
118    pub vector: Vec<f32>,
119    /// Application-defined metadata used by filters and optional result output.
120    #[serde(default)]
121    pub payload: JsonObject,
122}
123
124impl Point {
125    /// Creates a point with an empty payload.
126    pub fn new(id: impl Into<PointId>, vector: impl IntoIterator<Item = f32>) -> Self {
127        Self {
128            id: id.into(),
129            vector: vector.into_iter().collect(),
130            payload: JsonObject::new(),
131        }
132    }
133
134    /// Replaces the point payload and returns the updated point.
135    #[must_use]
136    pub fn with_payload(mut self, payload: JsonObject) -> Self {
137        self.payload = payload;
138        self
139    }
140}
141
142/// The vector distance used for ranking.
143#[derive(Clone, Copy, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
144#[serde(rename_all = "snake_case")]
145#[non_exhaustive]
146pub enum Distance {
147    /// Cosine similarity, returned as a descending score.
148    #[default]
149    Cosine,
150}
151
152/// Deterministic approximate-query defaults.
153///
154/// `tables`, `signature_bits`, and `projection_seed` describe the legacy
155/// format-version-1 LSH index. New format-version-2 roots use deterministic
156/// IVF-flat construction and ignore those three compatibility fields.
157#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
158#[serde(default)]
159pub struct IndexConfig {
160    /// Legacy format-version-1 LSH table count.
161    pub tables: usize,
162    /// Legacy format-version-1 projection bits per table signature.
163    pub signature_bits: usize,
164    /// Legacy format-version-1 projection seed.
165    pub projection_seed: u64,
166    /// Point-count threshold at or below which queries default to exact search.
167    pub full_scan_threshold: usize,
168    /// Default number of approximate index partitions to probe.
169    pub default_probes: usize,
170    /// Default maximum number of approximate candidates to discover.
171    pub default_candidate_limit: usize,
172}
173
174impl Default for IndexConfig {
175    fn default() -> Self {
176        Self {
177            tables: 12,
178            signature_bits: 12,
179            projection_seed: 0x6769_742d_7664_6231,
180            full_scan_threshold: 1_000,
181            default_probes: 96,
182            default_candidate_limit: 10_000,
183        }
184    }
185}
186
187/// Collection-wide vector and index configuration.
188#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
189#[serde(default)]
190pub struct CollectionConfig {
191    /// Required number of components in every stored and query vector.
192    pub dimension: usize,
193    /// Distance used to score vectors.
194    pub distance: Distance,
195    /// Optional application-defined vector-space identity checked by queries.
196    pub vector_space: Option<String>,
197    /// Deterministic approximate-index configuration.
198    pub index: IndexConfig,
199}
200
201impl CollectionConfig {
202    /// Creates a cosine collection configuration for the given dimension.
203    pub fn new(dimension: usize) -> Self {
204        Self {
205            dimension,
206            ..Self::default()
207        }
208    }
209
210    /// Assigns an application-defined vector-space identity.
211    #[must_use]
212    pub fn with_vector_space(mut self, vector_space: impl Into<String>) -> Self {
213        self.vector_space = Some(vector_space.into());
214        self
215    }
216
217    /// Replaces deterministic approximate-query defaults.
218    ///
219    /// The legacy LSH construction fields only affect format-version-1 roots.
220    #[must_use]
221    pub fn with_index(mut self, index: IndexConfig) -> Self {
222        self.index = index;
223        self
224    }
225}
226
227impl Default for CollectionConfig {
228    fn default() -> Self {
229        Self {
230            dimension: 0,
231            distance: Distance::Cosine,
232            vector_space: None,
233            index: IndexConfig::default(),
234        }
235    }
236}
237
238/// A JSON scalar value required by a field-match condition.
239#[derive(Clone, Debug, Serialize, Deserialize)]
240pub struct MatchValue {
241    /// The scalar value that must equal the payload field.
242    pub value: Value,
243}
244
245/// Inclusive or exclusive numeric bounds for a payload field.
246#[derive(Clone, Debug, Default, Serialize, Deserialize)]
247pub struct Range {
248    /// Exclusive lower bound.
249    pub gt: Option<f64>,
250    /// Inclusive lower bound.
251    pub gte: Option<f64>,
252    /// Exclusive upper bound.
253    pub lt: Option<f64>,
254    /// Inclusive upper bound.
255    pub lte: Option<f64>,
256}
257
258/// A predicate used inside a [`Filter`].
259#[derive(Clone, Debug, Serialize, Deserialize)]
260#[serde(untagged)]
261#[non_exhaustive]
262pub enum Condition {
263    /// Matches or range-checks a dot-separated payload field.
264    Field {
265        /// Dot-separated payload path.
266        key: String,
267        /// Optional equality match.
268        #[serde(rename = "match", skip_serializing_if = "Option::is_none")]
269        matches: Option<MatchValue>,
270        /// Optional numeric range.
271        #[serde(skip_serializing_if = "Option::is_none")]
272        range: Option<Range>,
273    },
274    /// Matches points whose typed ID is in the supplied set.
275    HasId {
276        /// Accepted typed IDs.
277        has_id: Vec<PointId>,
278    },
279    /// Evaluates a nested Boolean filter.
280    Nested(Filter),
281}
282
283impl Condition {
284    /// Creates an equality condition for a dot-separated payload path.
285    pub fn matches(key: impl Into<String>, value: impl Into<Value>) -> Self {
286        Self::Field {
287            key: key.into(),
288            matches: Some(MatchValue {
289                value: value.into(),
290            }),
291            range: None,
292        }
293    }
294
295    /// Creates a numeric-range condition for a dot-separated payload path.
296    pub fn range(key: impl Into<String>, range: Range) -> Self {
297        Self::Field {
298            key: key.into(),
299            matches: None,
300            range: Some(range),
301        }
302    }
303
304    /// Creates a condition that accepts the supplied typed point IDs.
305    pub fn has_id(ids: impl IntoIterator<Item = PointId>) -> Self {
306        Self::HasId {
307            has_id: ids.into_iter().collect(),
308        }
309    }
310}
311
312/// A Boolean point filter.
313///
314/// Every `must` condition and no `must_not` condition must match. When `should`
315/// is non-empty, at least one `should` condition must also match.
316#[derive(Clone, Debug, Default, Serialize, Deserialize)]
317#[serde(default)]
318pub struct Filter {
319    /// Conditions that must all match.
320    pub must: Vec<Condition>,
321    /// Conditions of which at least one must match when the list is non-empty.
322    pub should: Vec<Condition>,
323    /// Conditions that must not match.
324    pub must_not: Vec<Condition>,
325}
326
327impl Filter {
328    /// Creates a filter containing only required conditions.
329    pub fn must(conditions: impl IntoIterator<Item = Condition>) -> Self {
330        Self {
331            must: conditions.into_iter().collect(),
332            ..Self::default()
333        }
334    }
335}
336
337/// Exact or approximate query execution parameters.
338#[derive(Clone, Debug, Default, Serialize, Deserialize)]
339#[serde(default)]
340pub struct QueryParams {
341    /// Explicit mode override; `None` selects exact search for small collections.
342    pub exact: Option<bool>,
343    /// Approximate index partitions to probe, or zero to use the default.
344    pub probes: usize,
345    /// Approximate candidate limit, or zero to use the collection default.
346    pub candidate_limit: usize,
347}
348
349/// A vector similarity query.
350#[derive(Clone, Debug, Serialize, Deserialize)]
351#[serde(default)]
352pub struct Query {
353    /// Query vector, which must match the collection dimension.
354    pub vector: Vec<f32>,
355    /// Maximum number of scored points to return.
356    pub limit: usize,
357    /// Optional point filter applied before final ranking.
358    pub filter: Option<Filter>,
359    /// Whether returned winners include their JSON payloads.
360    pub with_payload: bool,
361    /// Whether returned winners include their stored vectors.
362    pub with_vector: bool,
363    /// Optional vector-space identity that must match collection metadata.
364    pub expected_vector_space: Option<String>,
365    /// Exact or approximate execution controls.
366    pub params: QueryParams,
367}
368
369impl Query {
370    /// Creates a query using collection defaults for exact or approximate mode.
371    pub fn new(vector: impl IntoIterator<Item = f32>, limit: usize) -> Self {
372        Self {
373            vector: vector.into_iter().collect(),
374            limit,
375            ..Self::default()
376        }
377    }
378
379    /// Creates a query that scores every eligible point.
380    pub fn exact(vector: impl IntoIterator<Item = f32>, limit: usize) -> Self {
381        let mut query = Self::new(vector, limit);
382        query.params.exact = Some(true);
383        query
384    }
385
386    /// Creates a query that uses deterministic approximate candidate discovery.
387    pub fn approximate(vector: impl IntoIterator<Item = f32>, limit: usize) -> Self {
388        let mut query = Self::new(vector, limit);
389        query.params.exact = Some(false);
390        query
391    }
392
393    /// Adds a point filter.
394    #[must_use]
395    pub fn with_filter(mut self, filter: Filter) -> Self {
396        self.filter = Some(filter);
397        self
398    }
399
400    /// Requests payloads for returned winners.
401    #[must_use]
402    pub fn with_payload(mut self) -> Self {
403        self.with_payload = true;
404        self
405    }
406
407    /// Requests stored vectors for returned winners.
408    #[must_use]
409    pub fn with_vector(mut self) -> Self {
410        self.with_vector = true;
411        self
412    }
413
414    /// Requires the collection to use the supplied vector-space identity.
415    #[must_use]
416    pub fn in_vector_space(mut self, vector_space: impl Into<String>) -> Self {
417        self.expected_vector_space = Some(vector_space.into());
418        self
419    }
420
421    /// Replaces the exact or approximate execution parameters.
422    #[must_use]
423    pub fn with_params(mut self, params: QueryParams) -> Self {
424        self.params = params;
425        self
426    }
427}
428
429impl Default for Query {
430    fn default() -> Self {
431        Self {
432            vector: Vec::new(),
433            limit: 10,
434            filter: None,
435            with_payload: false,
436            with_vector: false,
437            expected_vector_space: None,
438            params: QueryParams::default(),
439        }
440    }
441}
442
443/// A scored query winner.
444#[derive(Clone, Debug, Serialize, Deserialize)]
445pub struct ScoredPoint {
446    /// Typed point identifier.
447    pub id: PointId,
448    /// Descending cosine similarity score.
449    pub score: f32,
450    /// Payload when requested by [`Query::with_payload`].
451    #[serde(skip_serializing_if = "Option::is_none")]
452    pub payload: Option<JsonObject>,
453    /// Stored vector when requested by [`Query::with_vector`].
454    #[serde(skip_serializing_if = "Option::is_none")]
455    pub vector: Option<Vec<f32>>,
456}
457
458/// Query algorithm selected after applying collection defaults.
459#[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize, Deserialize)]
460#[serde(rename_all = "snake_case")]
461#[non_exhaustive]
462pub enum QueryMode {
463    /// Every filter-eligible point was scored.
464    Exact,
465    /// Candidates were discovered through the root's deterministic index.
466    Approximate,
467}
468
469/// Work counters for a completed query.
470#[derive(Clone, Debug, Serialize, Deserialize)]
471pub struct QueryStats {
472    /// Algorithm selected for the query.
473    pub mode: QueryMode,
474    /// Total points in the resolved collection root.
475    pub collection_points: usize,
476    /// Index partitions visited; zero for exact search.
477    pub buckets_probed: usize,
478    /// Distinct approximate candidates discovered.
479    pub candidates_discovered: usize,
480    /// Point vectors actually scored.
481    pub vectors_scored: usize,
482    /// Whether approximate discovery consumed its probe budget.
483    pub probe_limit_exhausted: bool,
484    /// Whether approximate discovery consumed its candidate budget.
485    pub candidate_limit_exhausted: bool,
486}
487
488/// Ordered similarity-search results and execution statistics.
489#[derive(Clone, Debug, Serialize, Deserialize)]
490pub struct QueryResult {
491    /// Immutable root that was queried.
492    pub root: ObjectId,
493    /// Winners ordered by descending score and canonical typed ID.
494    pub points: Vec<ScoredPoint>,
495    /// Algorithm and work counters.
496    pub stats: QueryStats,
497}
498
499/// A deterministic point-retrieval request.
500#[derive(Clone, Debug, Default, Serialize, Deserialize)]
501#[serde(default)]
502pub struct GetRequest {
503    /// Optional typed IDs; when combined with a filter, both must match.
504    pub ids: Vec<PointId>,
505    /// Optional payload or ID filter.
506    pub filter: Option<Filter>,
507    /// Number of canonically ordered matches to skip.
508    pub offset: usize,
509    /// Maximum matches to return, or all remaining matches when `None`.
510    pub limit: Option<usize>,
511    /// Whether results include payloads.
512    pub with_payload: bool,
513    /// Whether results include vectors.
514    pub with_vector: bool,
515}
516
517/// A point returned without similarity scoring.
518#[derive(Clone, Debug, Serialize, Deserialize)]
519pub struct Record {
520    /// Typed point identifier.
521    pub id: PointId,
522    /// Payload when requested.
523    #[serde(skip_serializing_if = "Option::is_none")]
524    pub payload: Option<JsonObject>,
525    /// Stored vector when requested.
526    #[serde(skip_serializing_if = "Option::is_none")]
527    pub vector: Option<Vec<f32>>,
528}
529
530/// Canonically ordered point-retrieval results.
531#[derive(Clone, Debug, Serialize, Deserialize)]
532pub struct GetResult {
533    /// Immutable root that was read.
534    pub root: ObjectId,
535    /// Matching records after offset and limit are applied.
536    pub points: Vec<Record>,
537}
538
539/// Selects the union of typed IDs and filter matches for deletion.
540#[derive(Clone, Debug, Default, Serialize, Deserialize)]
541#[serde(default)]
542pub struct DeleteSelector {
543    /// Typed IDs to delete; missing IDs are ignored.
544    pub ids: Vec<PointId>,
545    /// Optional filter whose matches are also deleted.
546    pub filter: Option<Filter>,
547}
548
549/// The outcome of a named collection write.
550#[derive(Clone, Debug, Serialize, Deserialize)]
551pub struct WriteResult {
552    /// New deterministic collection root.
553    pub root: ObjectId,
554    /// Number of submitted upserts or points actually removed.
555    pub affected_points: usize,
556}
557
558/// A count and the immutable root from which it was read.
559#[derive(Clone, Debug, Serialize, Deserialize)]
560pub struct CountResult {
561    /// Immutable root that was counted.
562    pub root: ObjectId,
563    /// Number of matching points.
564    pub count: usize,
565}
566
567/// Metadata for a named or historical collection view.
568#[derive(Clone, Debug, Serialize, Deserialize)]
569pub struct CollectionInfo {
570    /// Resolved deterministic root.
571    pub root: ObjectId,
572    /// Named collection.
573    pub name: String,
574    /// Persisted format version used by the resolved root.
575    pub format_version: u32,
576    /// Number of points at the resolved root.
577    pub point_count: usize,
578    /// Collection configuration stored in the root.
579    pub config: CollectionConfig,
580    /// Whether the view is historical and rejects writes.
581    pub read_only: bool,
582}
583
584/// Metadata for an immutable root snapshot.
585#[derive(Clone, Debug, Serialize, Deserialize)]
586pub struct SnapshotInfo {
587    /// Deterministic root tree ID.
588    pub root: ObjectId,
589    /// Persisted format version used by the root.
590    pub format_version: u32,
591    /// Number of points in the root.
592    pub point_count: usize,
593    /// Collection configuration stored in the root.
594    pub config: CollectionConfig,
595}
596
597/// One operation in an ordered immutable-root mutation batch.
598#[derive(Clone, Debug, Serialize, Deserialize)]
599#[serde(tag = "operation", rename_all = "snake_case")]
600#[non_exhaustive]
601pub enum SnapshotMutation {
602    /// Adds a new point or replaces an existing point with the same typed ID.
603    Upsert {
604        /// Complete replacement point.
605        point: Point,
606    },
607    /// Deletes the supplied typed IDs.
608    DeleteIds {
609        /// Typed IDs to delete; missing IDs are ignored.
610        ids: Vec<PointId>,
611    },
612    /// Deletes every point matching a filter.
613    DeleteFilter {
614        /// Filter evaluated against the preceding mutation state.
615        filter: Filter,
616    },
617}
618
619impl SnapshotMutation {
620    /// Creates an upsert mutation.
621    pub fn upsert(point: Point) -> Self {
622        Self::Upsert { point }
623    }
624
625    /// Creates a typed-ID deletion mutation.
626    pub fn delete_ids(ids: impl IntoIterator<Item = PointId>) -> Self {
627        Self::DeleteIds {
628            ids: ids.into_iter().collect(),
629        }
630    }
631
632    /// Creates a filter deletion mutation.
633    pub fn delete_filter(filter: Filter) -> Self {
634        Self::DeleteFilter { filter }
635    }
636}
637
638/// One named-collection commit in newest-first history order.
639#[derive(Clone, Debug, Serialize, Deserialize)]
640pub struct HistoryEntry {
641    /// Commit object ID.
642    pub commit: ObjectId,
643    /// Canonical root tree recorded by the commit.
644    pub root: ObjectId,
645    /// First parent commit, when present.
646    pub parent: Option<ObjectId>,
647    /// Commit message generated for the collection operation.
648    pub message: String,
649    /// Commit time in Unix seconds.
650    pub time_seconds: i64,
651}
652
653/// Count and logical size of a set of reachable Git objects.
654#[derive(Clone, Debug, Default, Serialize, Deserialize)]
655pub struct ObjectStats {
656    /// Number of Git objects.
657    pub objects: usize,
658    /// Sum of logical object bytes.
659    pub bytes: usize,
660}
661
662/// Logical point and structural-sharing differences between two roots.
663#[derive(Clone, Debug, Serialize, Deserialize)]
664pub struct DiffResult {
665    /// Resolved left root.
666    pub left_root: ObjectId,
667    /// Resolved right root.
668    pub right_root: ObjectId,
669    /// IDs present only on the right.
670    pub added: Vec<PointId>,
671    /// IDs present only on the left.
672    pub removed: Vec<PointId>,
673    /// IDs present in both roots with changed point content.
674    pub changed: Vec<PointId>,
675    /// Whether collection metadata differs.
676    pub configuration_changed: bool,
677    /// Whether the persisted approximate index differs.
678    pub buckets_changed: bool,
679    /// Objects reachable from both roots.
680    pub shared: ObjectStats,
681    /// Objects reachable only from the left root.
682    pub left_unique: ObjectStats,
683    /// Objects reachable only from the right root.
684    pub right_unique: ObjectStats,
685}
686
687/// Result of basic or full canonical-root validation.
688#[derive(Clone, Debug, Serialize, Deserialize)]
689pub struct ValidationReport {
690    /// Root that was validated.
691    pub root: ObjectId,
692    /// Whether expensive index recomputation was requested.
693    pub full: bool,
694    /// Validated point count.
695    pub point_count: usize,
696    /// Approximate-index partitions checked during full validation.
697    pub checked_buckets: usize,
698    /// `true` when validation completed without finding corruption.
699    pub valid: bool,
700}