Skip to main content

git_vdb/
model.rs

1//! Public request, response, configuration, filtering, and mutation types.
2
3use serde::{Deserialize, Serialize};
4use serde_json::{Map, Value};
5use std::fmt;
6use std::str::FromStr;
7
8/// A JSON object used for point payloads.
9pub type JsonObject = Map<String, Value>;
10
11/// A typed point identifier.
12///
13/// String and unsigned-integer IDs occupy distinct namespaces and have a stable
14/// canonical ordering used to break equal-score query ties.
15#[derive(Clone, Debug, Eq, PartialEq, Hash, Serialize, Deserialize)]
16#[serde(untagged)]
17pub enum PointId {
18    /// A UTF-8 string identifier.
19    String(String),
20    /// An unsigned 64-bit integer identifier.
21    UInt(u64),
22}
23
24impl PointId {
25    pub(crate) fn canonical_bytes(&self) -> Vec<u8> {
26        match self {
27            Self::String(value) => {
28                let mut bytes = vec![b's', 0];
29                bytes.extend_from_slice(value.as_bytes());
30                bytes
31            }
32            Self::UInt(value) => {
33                let mut bytes = vec![b'u', 0];
34                bytes.extend_from_slice(&value.to_be_bytes());
35                bytes
36            }
37        }
38    }
39}
40
41impl From<&str> for PointId {
42    fn from(value: &str) -> Self {
43        Self::String(value.to_owned())
44    }
45}
46
47impl From<String> for PointId {
48    fn from(value: String) -> Self {
49        Self::String(value)
50    }
51}
52
53impl From<u64> for PointId {
54    fn from(value: u64) -> Self {
55        Self::UInt(value)
56    }
57}
58
59impl Ord for PointId {
60    fn cmp(&self, other: &Self) -> std::cmp::Ordering {
61        self.canonical_bytes().cmp(&other.canonical_bytes())
62    }
63}
64
65impl PartialOrd for PointId {
66    fn partial_cmp(&self, other: &Self) -> Option<std::cmp::Ordering> {
67        Some(self.cmp(other))
68    }
69}
70
71impl fmt::Display for PointId {
72    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
73        match self {
74            Self::String(value) => write!(f, "{value}"),
75            Self::UInt(value) => write!(f, "{value}"),
76        }
77    }
78}
79
80/// A hexadecimal Git object ID that identifies a tree or commit.
81#[derive(Clone, Debug, Eq, PartialEq, Hash, Serialize, Deserialize)]
82#[serde(transparent)]
83pub struct ObjectId(pub String);
84
85impl fmt::Display for ObjectId {
86    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
87        f.write_str(&self.0)
88    }
89}
90
91impl FromStr for ObjectId {
92    type Err = git2::Error;
93
94    fn from_str(value: &str) -> Result<Self, Self::Err> {
95        git2::Oid::from_str(value)?;
96        Ok(Self(value.to_owned()))
97    }
98}
99
100impl From<git2::Oid> for ObjectId {
101    fn from(value: git2::Oid) -> Self {
102        Self(value.to_string())
103    }
104}
105
106impl AsRef<str> for ObjectId {
107    fn as_ref(&self) -> &str {
108        &self.0
109    }
110}
111
112/// A vector and its optional JSON payload, identified by a typed ID.
113#[derive(Clone, Debug, Serialize, Deserialize, PartialEq)]
114pub struct Point {
115    /// The stable typed identifier for this point.
116    pub id: PointId,
117    /// The vector components, whose length must match the collection dimension.
118    pub vector: Vec<f32>,
119    /// Application-defined metadata used by filters and optional result output.
120    #[serde(default)]
121    pub payload: JsonObject,
122}
123
124impl Point {
125    /// Creates a point with an empty payload.
126    pub fn new(id: impl Into<PointId>, vector: impl IntoIterator<Item = f32>) -> Self {
127        Self {
128            id: id.into(),
129            vector: vector.into_iter().collect(),
130            payload: JsonObject::new(),
131        }
132    }
133
134    /// Replaces the point payload and returns the updated point.
135    #[must_use]
136    pub fn with_payload(mut self, payload: JsonObject) -> Self {
137        self.payload = payload;
138        self
139    }
140
141    /// Serializes object-shaped metadata as the point payload.
142    ///
143    /// Arrays, scalars, and null are rejected because persisted point payloads
144    /// are always JSON objects.
145    pub fn with_metadata(mut self, metadata: impl Serialize) -> crate::Result<Self> {
146        match serde_json::to_value(metadata)? {
147            Value::Object(payload) => {
148                self.payload = payload;
149                Ok(self)
150            }
151            _ => Err(crate::Error::Invalid(
152                "point metadata must serialize to a JSON object".into(),
153            )),
154        }
155    }
156}
157
158/// The vector distance used for ranking.
159#[derive(Clone, Copy, Debug, Default, Eq, PartialEq, Serialize, Deserialize)]
160#[serde(rename_all = "snake_case")]
161#[non_exhaustive]
162pub enum Distance {
163    /// Cosine similarity, returned as a descending score.
164    #[default]
165    Cosine,
166}
167
168/// Deterministic approximate-query defaults.
169///
170/// `tables`, `signature_bits`, and `projection_seed` describe the legacy
171/// format-version-1 LSH index. New format-version-2 roots use deterministic
172/// IVF-flat construction and ignore those three compatibility fields.
173#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
174#[serde(default)]
175pub struct IndexConfig {
176    /// Legacy format-version-1 LSH table count.
177    pub tables: usize,
178    /// Legacy format-version-1 projection bits per table signature.
179    pub signature_bits: usize,
180    /// Legacy format-version-1 projection seed.
181    pub projection_seed: u64,
182    /// Point-count threshold at or below which queries default to exact search.
183    pub full_scan_threshold: usize,
184    /// Default number of approximate index partitions to probe.
185    pub default_probes: usize,
186    /// Default maximum number of approximate candidates to discover.
187    pub default_candidate_limit: usize,
188}
189
190impl Default for IndexConfig {
191    fn default() -> Self {
192        Self {
193            tables: 12,
194            signature_bits: 12,
195            projection_seed: 0x6769_742d_7664_6231,
196            full_scan_threshold: 1_000,
197            default_probes: 96,
198            default_candidate_limit: 10_000,
199        }
200    }
201}
202
203/// Collection-wide vector and index configuration.
204#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
205#[serde(default)]
206pub struct CollectionConfig {
207    /// Required number of components in every stored and query vector.
208    pub dimension: usize,
209    /// Distance used to score vectors.
210    pub distance: Distance,
211    /// Optional application-defined vector-space identity checked by queries.
212    pub vector_space: Option<String>,
213    /// Deterministic approximate-index configuration.
214    pub index: IndexConfig,
215}
216
217impl CollectionConfig {
218    /// Creates a cosine collection configuration for the given dimension.
219    pub fn new(dimension: usize) -> Self {
220        Self {
221            dimension,
222            ..Self::default()
223        }
224    }
225
226    /// Assigns an application-defined vector-space identity.
227    #[must_use]
228    pub fn with_vector_space(mut self, vector_space: impl Into<String>) -> Self {
229        self.vector_space = Some(vector_space.into());
230        self
231    }
232
233    /// Replaces deterministic approximate-query defaults.
234    ///
235    /// The legacy LSH construction fields only affect format-version-1 roots.
236    #[must_use]
237    pub fn with_index(mut self, index: IndexConfig) -> Self {
238        self.index = index;
239        self
240    }
241}
242
243impl Default for CollectionConfig {
244    fn default() -> Self {
245        Self {
246            dimension: 0,
247            distance: Distance::Cosine,
248            vector_space: None,
249            index: IndexConfig::default(),
250        }
251    }
252}
253
254/// A JSON scalar value required by a field-match condition.
255#[derive(Clone, Debug, Serialize, Deserialize)]
256pub struct MatchValue {
257    /// The scalar value that must equal the payload field.
258    pub value: Value,
259}
260
261/// Inclusive or exclusive numeric bounds for a payload field.
262#[derive(Clone, Debug, Default, Serialize, Deserialize)]
263pub struct Range {
264    /// Exclusive lower bound.
265    pub gt: Option<f64>,
266    /// Inclusive lower bound.
267    pub gte: Option<f64>,
268    /// Exclusive upper bound.
269    pub lt: Option<f64>,
270    /// Inclusive upper bound.
271    pub lte: Option<f64>,
272}
273
274/// A predicate used inside a [`Filter`].
275#[derive(Clone, Debug, Serialize, Deserialize)]
276#[serde(untagged)]
277#[non_exhaustive]
278pub enum Condition {
279    /// Matches or range-checks a dot-separated payload field.
280    Field {
281        /// Dot-separated payload path.
282        key: String,
283        /// Optional equality match.
284        #[serde(rename = "match", skip_serializing_if = "Option::is_none")]
285        matches: Option<MatchValue>,
286        /// Optional numeric range.
287        #[serde(skip_serializing_if = "Option::is_none")]
288        range: Option<Range>,
289    },
290    /// Matches points whose typed ID is in the supplied set.
291    HasId {
292        /// Accepted typed IDs.
293        has_id: Vec<PointId>,
294    },
295    /// Evaluates a nested Boolean filter.
296    Nested(Filter),
297}
298
299impl Condition {
300    /// Creates an equality condition for a dot-separated payload path.
301    pub fn matches(key: impl Into<String>, value: impl Into<Value>) -> Self {
302        Self::Field {
303            key: key.into(),
304            matches: Some(MatchValue {
305                value: value.into(),
306            }),
307            range: None,
308        }
309    }
310
311    /// Creates a numeric-range condition for a dot-separated payload path.
312    pub fn range(key: impl Into<String>, range: Range) -> Self {
313        Self::Field {
314            key: key.into(),
315            matches: None,
316            range: Some(range),
317        }
318    }
319
320    /// Creates a condition that accepts the supplied typed point IDs.
321    pub fn has_id(ids: impl IntoIterator<Item = PointId>) -> Self {
322        Self::HasId {
323            has_id: ids.into_iter().collect(),
324        }
325    }
326}
327
328/// A Boolean point filter.
329///
330/// Every `must` condition and no `must_not` condition must match. When `should`
331/// is non-empty, at least one `should` condition must also match.
332#[derive(Clone, Debug, Default, Serialize, Deserialize)]
333#[serde(default)]
334pub struct Filter {
335    /// Conditions that must all match.
336    pub must: Vec<Condition>,
337    /// Conditions of which at least one must match when the list is non-empty.
338    pub should: Vec<Condition>,
339    /// Conditions that must not match.
340    pub must_not: Vec<Condition>,
341}
342
343impl Filter {
344    /// Creates a filter containing only required conditions.
345    pub fn must(conditions: impl IntoIterator<Item = Condition>) -> Self {
346        Self {
347            must: conditions.into_iter().collect(),
348            ..Self::default()
349        }
350    }
351}
352
353/// Exact or approximate query execution parameters.
354#[derive(Clone, Debug, Default, Serialize, Deserialize)]
355#[serde(default)]
356pub struct QueryParams {
357    /// Explicit mode override; `None` selects exact search for small collections.
358    pub exact: Option<bool>,
359    /// Approximate index partitions to probe, or zero to use the default.
360    pub probes: usize,
361    /// Approximate candidate limit, or zero to use the collection default.
362    pub candidate_limit: usize,
363}
364
365/// A vector similarity query.
366#[derive(Clone, Debug, Serialize, Deserialize)]
367#[serde(default)]
368pub struct Query {
369    /// Query vector, which must match the collection dimension.
370    pub vector: Vec<f32>,
371    /// Maximum number of scored points to return.
372    pub limit: usize,
373    /// Optional point filter applied before final ranking.
374    pub filter: Option<Filter>,
375    /// Whether returned winners include their JSON payloads.
376    pub with_payload: bool,
377    /// Whether returned winners include their stored vectors.
378    pub with_vector: bool,
379    /// Optional vector-space identity that must match collection metadata.
380    pub expected_vector_space: Option<String>,
381    /// Exact or approximate execution controls.
382    pub params: QueryParams,
383}
384
385impl Query {
386    /// Creates a query using collection defaults for exact or approximate mode.
387    pub fn new(vector: impl IntoIterator<Item = f32>, limit: usize) -> Self {
388        Self {
389            vector: vector.into_iter().collect(),
390            limit,
391            ..Self::default()
392        }
393    }
394
395    /// Creates a query that scores every eligible point.
396    pub fn exact(vector: impl IntoIterator<Item = f32>, limit: usize) -> Self {
397        let mut query = Self::new(vector, limit);
398        query.params.exact = Some(true);
399        query
400    }
401
402    /// Creates a query that uses deterministic approximate candidate discovery.
403    pub fn approximate(vector: impl IntoIterator<Item = f32>, limit: usize) -> Self {
404        let mut query = Self::new(vector, limit);
405        query.params.exact = Some(false);
406        query
407    }
408
409    /// Adds a point filter.
410    #[must_use]
411    pub fn with_filter(mut self, filter: Filter) -> Self {
412        self.filter = Some(filter);
413        self
414    }
415
416    /// Requests payloads for returned winners.
417    #[must_use]
418    pub fn with_payload(mut self) -> Self {
419        self.with_payload = true;
420        self
421    }
422
423    /// Requests stored vectors for returned winners.
424    #[must_use]
425    pub fn with_vector(mut self) -> Self {
426        self.with_vector = true;
427        self
428    }
429
430    /// Requires the collection to use the supplied vector-space identity.
431    #[must_use]
432    pub fn in_vector_space(mut self, vector_space: impl Into<String>) -> Self {
433        self.expected_vector_space = Some(vector_space.into());
434        self
435    }
436
437    /// Replaces the exact or approximate execution parameters.
438    #[must_use]
439    pub fn with_params(mut self, params: QueryParams) -> Self {
440        self.params = params;
441        self
442    }
443}
444
445impl Default for Query {
446    fn default() -> Self {
447        Self {
448            vector: Vec::new(),
449            limit: 10,
450            filter: None,
451            with_payload: false,
452            with_vector: false,
453            expected_vector_space: None,
454            params: QueryParams::default(),
455        }
456    }
457}
458
459/// A scored query winner.
460#[derive(Clone, Debug, Serialize, Deserialize)]
461pub struct ScoredPoint {
462    /// Typed point identifier.
463    pub id: PointId,
464    /// Descending cosine similarity score.
465    pub score: f32,
466    /// Payload when requested by [`Query::with_payload`].
467    #[serde(skip_serializing_if = "Option::is_none")]
468    pub payload: Option<JsonObject>,
469    /// Stored vector when requested by [`Query::with_vector`].
470    #[serde(skip_serializing_if = "Option::is_none")]
471    pub vector: Option<Vec<f32>>,
472}
473
474/// Query algorithm selected after applying collection defaults.
475#[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize, Deserialize)]
476#[serde(rename_all = "snake_case")]
477#[non_exhaustive]
478pub enum QueryMode {
479    /// Every filter-eligible point was scored.
480    Exact,
481    /// Candidates were discovered through the root's deterministic index.
482    Approximate,
483}
484
485/// Work counters for a completed query.
486#[derive(Clone, Debug, Serialize, Deserialize)]
487pub struct QueryStats {
488    /// Algorithm selected for the query.
489    pub mode: QueryMode,
490    /// Total points in the resolved collection root.
491    pub collection_points: usize,
492    /// Index partitions visited; zero for exact search.
493    pub buckets_probed: usize,
494    /// Distinct approximate candidates discovered.
495    pub candidates_discovered: usize,
496    /// Point vectors actually scored.
497    pub vectors_scored: usize,
498    /// Whether approximate discovery consumed its probe budget.
499    pub probe_limit_exhausted: bool,
500    /// Whether approximate discovery consumed its candidate budget.
501    pub candidate_limit_exhausted: bool,
502}
503
504/// Ordered similarity-search results and execution statistics.
505#[derive(Clone, Debug, Serialize, Deserialize)]
506pub struct QueryResult {
507    /// Immutable root that was queried.
508    pub root: ObjectId,
509    /// Winners ordered by descending score and canonical typed ID.
510    pub points: Vec<ScoredPoint>,
511    /// Algorithm and work counters.
512    pub stats: QueryStats,
513}
514
515/// A deterministic point-retrieval request.
516#[derive(Clone, Debug, Default, Serialize, Deserialize)]
517#[serde(default)]
518pub struct GetRequest {
519    /// Optional typed IDs; when combined with a filter, both must match.
520    pub ids: Vec<PointId>,
521    /// Optional payload or ID filter.
522    pub filter: Option<Filter>,
523    /// Number of canonically ordered matches to skip.
524    pub offset: usize,
525    /// Maximum matches to return, or all remaining matches when `None`.
526    pub limit: Option<usize>,
527    /// Whether results include payloads.
528    pub with_payload: bool,
529    /// Whether results include vectors.
530    pub with_vector: bool,
531}
532
533/// A point returned without similarity scoring.
534#[derive(Clone, Debug, Serialize, Deserialize)]
535pub struct Record {
536    /// Typed point identifier.
537    pub id: PointId,
538    /// Payload when requested.
539    #[serde(skip_serializing_if = "Option::is_none")]
540    pub payload: Option<JsonObject>,
541    /// Stored vector when requested.
542    #[serde(skip_serializing_if = "Option::is_none")]
543    pub vector: Option<Vec<f32>>,
544}
545
546/// Canonically ordered point-retrieval results.
547#[derive(Clone, Debug, Serialize, Deserialize)]
548pub struct GetResult {
549    /// Immutable root that was read.
550    pub root: ObjectId,
551    /// Matching records after offset and limit are applied.
552    pub points: Vec<Record>,
553}
554
555/// Selects the union of typed IDs and filter matches for deletion.
556#[derive(Clone, Debug, Default, Serialize, Deserialize)]
557#[serde(default)]
558pub struct DeleteSelector {
559    /// Typed IDs to delete; missing IDs are ignored.
560    pub ids: Vec<PointId>,
561    /// Optional filter whose matches are also deleted.
562    pub filter: Option<Filter>,
563}
564
565/// The outcome of a named collection write.
566#[derive(Clone, Debug, Serialize, Deserialize)]
567pub struct WriteResult {
568    /// New deterministic collection root.
569    pub root: ObjectId,
570    /// Number of submitted upserts or points actually removed.
571    pub affected_points: usize,
572}
573
574/// A count and the immutable root from which it was read.
575#[derive(Clone, Debug, Serialize, Deserialize)]
576pub struct CountResult {
577    /// Immutable root that was counted.
578    pub root: ObjectId,
579    /// Number of matching points.
580    pub count: usize,
581}
582
583/// Metadata for a named or historical collection view.
584#[derive(Clone, Debug, Serialize, Deserialize)]
585pub struct CollectionInfo {
586    /// Resolved deterministic root.
587    pub root: ObjectId,
588    /// Named collection.
589    pub name: String,
590    /// Persisted format version used by the resolved root.
591    pub format_version: u32,
592    /// Number of points at the resolved root.
593    pub point_count: usize,
594    /// Collection configuration stored in the root.
595    pub config: CollectionConfig,
596    /// Whether the view is historical and rejects writes.
597    pub read_only: bool,
598}
599
600/// Metadata for an immutable root snapshot.
601#[derive(Clone, Debug, Serialize, Deserialize)]
602pub struct SnapshotInfo {
603    /// Deterministic root tree ID.
604    pub root: ObjectId,
605    /// Persisted format version used by the root.
606    pub format_version: u32,
607    /// Number of points in the root.
608    pub point_count: usize,
609    /// Collection configuration stored in the root.
610    pub config: CollectionConfig,
611}
612
613/// One operation in an ordered immutable-root mutation batch.
614#[derive(Clone, Debug, Serialize, Deserialize)]
615#[serde(tag = "operation", rename_all = "snake_case")]
616#[non_exhaustive]
617pub enum SnapshotMutation {
618    /// Adds a new point or replaces an existing point with the same typed ID.
619    Upsert {
620        /// Complete replacement point.
621        point: Point,
622    },
623    /// Deletes the supplied typed IDs.
624    DeleteIds {
625        /// Typed IDs to delete; missing IDs are ignored.
626        ids: Vec<PointId>,
627    },
628    /// Deletes every point matching a filter.
629    DeleteFilter {
630        /// Filter evaluated against the preceding mutation state.
631        filter: Filter,
632    },
633}
634
635impl SnapshotMutation {
636    /// Creates an upsert mutation.
637    pub fn upsert(point: Point) -> Self {
638        Self::Upsert { point }
639    }
640
641    /// Creates a typed-ID deletion mutation.
642    pub fn delete_ids(ids: impl IntoIterator<Item = PointId>) -> Self {
643        Self::DeleteIds {
644            ids: ids.into_iter().collect(),
645        }
646    }
647
648    /// Creates a filter deletion mutation.
649    pub fn delete_filter(filter: Filter) -> Self {
650        Self::DeleteFilter { filter }
651    }
652}
653
654/// One named-collection commit in newest-first history order.
655#[derive(Clone, Debug, Serialize, Deserialize)]
656pub struct HistoryEntry {
657    /// Commit object ID.
658    pub commit: ObjectId,
659    /// Canonical root tree recorded by the commit.
660    pub root: ObjectId,
661    /// First parent commit, when present.
662    pub parent: Option<ObjectId>,
663    /// Commit message generated for the collection operation.
664    pub message: String,
665    /// Commit time in Unix seconds.
666    pub time_seconds: i64,
667}
668
669/// Count and logical size of a set of reachable Git objects.
670#[derive(Clone, Debug, Default, Serialize, Deserialize)]
671pub struct ObjectStats {
672    /// Number of Git objects.
673    pub objects: usize,
674    /// Sum of logical object bytes.
675    pub bytes: usize,
676}
677
678/// Logical point and structural-sharing differences between two roots.
679#[derive(Clone, Debug, Serialize, Deserialize)]
680pub struct DiffResult {
681    /// Resolved left root.
682    pub left_root: ObjectId,
683    /// Resolved right root.
684    pub right_root: ObjectId,
685    /// IDs present only on the right.
686    pub added: Vec<PointId>,
687    /// IDs present only on the left.
688    pub removed: Vec<PointId>,
689    /// IDs present in both roots with changed point content.
690    pub changed: Vec<PointId>,
691    /// Whether collection metadata differs.
692    pub configuration_changed: bool,
693    /// Whether the persisted approximate index differs.
694    pub buckets_changed: bool,
695    /// Objects reachable from both roots.
696    pub shared: ObjectStats,
697    /// Objects reachable only from the left root.
698    pub left_unique: ObjectStats,
699    /// Objects reachable only from the right root.
700    pub right_unique: ObjectStats,
701}
702
703/// Result of basic or full canonical-root validation.
704#[derive(Clone, Debug, Serialize, Deserialize)]
705pub struct ValidationReport {
706    /// Root that was validated.
707    pub root: ObjectId,
708    /// Whether expensive index recomputation was requested.
709    pub full: bool,
710    /// Validated point count.
711    pub point_count: usize,
712    /// Approximate-index partitions checked during full validation.
713    pub checked_buckets: usize,
714    /// `true` when validation completed without finding corruption.
715    pub valid: bool,
716}