Skip to main content

ifc_step/
index.rs

1//! Offset index: what is in a file, without decoding it.
2//!
3//! # Why this exists
4//!
5//! A type census, or picking 200 walls out of nine million records, needs
6//! to know where each record is and what type it has -- not what it says.
7//! [`Index::scan`] answers that from the borrowed source: it frames every
8//! record with `openbim_step::scan` and keeps its id, byte span and type,
9//! and nothing else. It does not copy the input and does not validate the
10//! inside of a record, which makes it the cheapest way to look into a file.
11//!
12//! For a model that decodes on access with every record validated up front,
13//! read with [`StepCodec`](crate::StepCodec): a strict read is lazy.
14//!
15//! # Trust
16//!
17//! Framing is `openbim_step::scan`: the header is parsed strictly,
18//! everything between records goes through the real lexer, and junk,
19//! missing section ends or a second file appended are errors rather than
20//! skipped. Decoding a record ([`Index::entity`]) runs the same parser and
21//! the same IFC conversion as a full read, on exactly the record's bytes.
22//! `tests/index_agreement.rs` checks both against the eager reader on every
23//! fixture in the repository.
24
25use std::collections::HashMap;
26
27use ifc_model::{Entity, EntityId, Model, Value};
28use openbim_step::{ParseOptions, Span};
29
30use crate::{parser, StepError};
31
32/// What is in a file, without what it says.
33///
34/// Borrows the source bytes and stores, per record, its id, span and an
35/// index into a table of distinct type names -- about 32 bytes per entity,
36/// two orders of magnitude less than a decoded [`Model`].
37///
38/// Use [`Index::entity`] to decode one record, or [`Index::materialize_closure`]
39/// to build a real [`Model`] from a chosen subset.
40pub struct Index<'a> {
41    source: &'a [u8],
42    header: ifc_model::header::Header,
43    ids: Vec<u64>,
44    spans: Vec<Span>,
45    type_ids: Vec<u32>,
46    type_names: Vec<String>,
47    /// Position of each id in the columns. Files list ids in increasing
48    /// order in practice, where a binary search over `ids` needs nothing
49    /// extra; this map exists only for a file that does not.
50    unordered: Option<HashMap<u64, usize>>,
51}
52
53impl<'a> Index<'a> {
54    /// Frames every record of `source`. Does not decode attributes.
55    ///
56    /// # Errors
57    ///
58    /// A header or framing defect: the wrong physical-file marker, a
59    /// malformed header, anything between records that is not a record, a
60    /// missing `ENDSEC`, content after the end marker, or an instance id or
61    /// type the IFC record model cannot hold (an id beyond `u64`, a complex
62    /// instance). Defects inside a record surface when it is decoded.
63    pub fn scan(source: &'a [u8]) -> Result<Self, StepError> {
64        let scanned =
65            openbim_step::scan(source).map_err(|error| StepError::from_step(source, error))?;
66        let mut header = ifc_model::header::Header::default();
67        parser::apply_header(&mut header, scanned.header().standard());
68        let mut ids = Vec::new();
69        let mut spans = Vec::new();
70        let mut type_ids = Vec::new();
71        let mut type_names: Vec<String> = Vec::new();
72        let mut seen: HashMap<String, u32> = HashMap::new();
73        for record in scanned.records() {
74            let record = record.map_err(|error| StepError::from_step(source, error))?;
75            let id = record.id.as_str().parse::<u64>().map_err(|_| {
76                unrepresentable(
77                    record.span,
78                    "instance id exceeds the IFC record model range",
79                )
80            })?;
81            let Some(name) = record.name else {
82                return Err(unrepresentable(
83                    record.span,
84                    "complex STEP instances are not representable in the IFC record model",
85                ));
86            };
87            let upper = name.to_ascii_uppercase();
88            let next = u32::try_from(type_names.len()).expect("fewer than 2^32 distinct types");
89            let type_id = *seen.entry(upper.clone()).or_insert(next);
90            if type_id == next {
91                type_names.push(upper);
92            }
93            ids.push(id);
94            spans.push(record.span);
95            type_ids.push(type_id);
96        }
97        let unordered = (!ids.windows(2).all(|pair| pair[0] < pair[1])).then(|| {
98            // Later records win, as they do in a model read.
99            ids.iter()
100                .enumerate()
101                .map(|(position, id)| (*id, position))
102                .collect()
103        });
104        Ok(Self {
105            source,
106            header,
107            ids,
108            spans,
109            type_ids,
110            type_names,
111            unordered,
112        })
113    }
114
115    /// Number of records found.
116    #[must_use]
117    pub fn len(&self) -> usize {
118        self.ids.len()
119    }
120
121    /// Whether the file has no data records.
122    #[must_use]
123    pub fn is_empty(&self) -> bool {
124        self.ids.is_empty()
125    }
126
127    /// The file header, as a model read would report it.
128    #[must_use]
129    pub fn header(&self) -> &ifc_model::header::Header {
130        &self.header
131    }
132
133    /// Every id, in file order.
134    pub fn ids(&self) -> impl Iterator<Item = EntityId> + '_ {
135        self.ids.iter().copied().map(EntityId)
136    }
137
138    /// Upper-cased type name of `id`, without decoding it.
139    #[must_use]
140    pub fn type_of(&self, id: EntityId) -> Option<&str> {
141        let position = self.position(id)?;
142        Some(&self.type_names[self.type_ids[position] as usize])
143    }
144
145    /// How many records of each type. The census a viewer opens with.
146    #[must_use]
147    pub fn count_by_type(&self) -> std::collections::BTreeMap<&str, usize> {
148        let mut out = std::collections::BTreeMap::new();
149        for type_id in &self.type_ids {
150            *out.entry(self.type_names[*type_id as usize].as_str())
151                .or_insert(0) += 1;
152        }
153        out
154    }
155
156    /// Ids of every record whose type matches `name`, case-insensitively,
157    /// in file order.
158    #[must_use]
159    pub fn ids_of_type(&self, name: &str) -> Vec<EntityId> {
160        let upper = name.to_ascii_uppercase();
161        let Some(type_id) = self.type_names.iter().position(|n| *n == upper) else {
162            return Vec::new();
163        };
164        let type_id = u32::try_from(type_id).expect("fewer than 2^32 distinct types");
165        self.type_ids
166            .iter()
167            .enumerate()
168            .filter(|(_, t)| **t == type_id)
169            .map(|(position, _)| EntityId(self.ids[position]))
170            .collect()
171    }
172
173    fn position(&self, id: EntityId) -> Option<usize> {
174        match &self.unordered {
175            Some(map) => map.get(&id.0).copied(),
176            None => self.ids.binary_search(&id.0).ok(),
177        }
178    }
179
180    /// Decodes one record into an [`Entity`]: the same parser and IFC
181    /// conversion a full read uses, on exactly the record's bytes.
182    ///
183    /// # Errors
184    ///
185    /// When the record is malformed or holds a value the IFC record model
186    /// cannot represent -- exactly what would make a strict read fail.
187    pub fn entity(&self, id: EntityId) -> Result<Option<Entity>, StepError> {
188        let Some(position) = self.position(id) else {
189            return Ok(None);
190        };
191        let (record, _) =
192            parser::decode(self.source, self.spans[position], ParseOptions::strict())?;
193        Ok(Some(parser::convert(record)?.1))
194    }
195
196    /// Builds a [`Model`] from exactly `wanted`, and nothing else, with the
197    /// file's header. Ids the file does not contain are skipped.
198    ///
199    /// # Dangling references
200    ///
201    /// The result is a real `Model` but not a self-contained one: an entity
202    /// that referenced `#7` still says `Ref(7)` even when `#7` was not
203    /// requested. Traversal will not resolve it. Use
204    /// [`Index::materialize_closure`] unless you specifically want the raw
205    /// subset and will handle missing targets yourself.
206    ///
207    /// # Errors
208    ///
209    /// The first wanted record that does not decode; see [`Index::entity`].
210    pub fn materialize(&self, wanted: &[EntityId]) -> Result<Model, StepError> {
211        self.build(wanted.iter().copied())
212    }
213
214    /// Builds a [`Model`] containing `wanted` and everything they reach.
215    ///
216    /// Follows references transitively, so the result has no dangling `Ref`
217    /// except to ids the file itself does not define. This is the
218    /// constructor to reach for: it is what makes a subset independently
219    /// usable.
220    ///
221    /// # Errors
222    ///
223    /// The first reached record that does not decode; see [`Index::entity`].
224    pub fn materialize_closure(&self, wanted: &[EntityId]) -> Result<Model, StepError> {
225        let mut needed: std::collections::BTreeSet<u64> = wanted.iter().map(|i| i.0).collect();
226        let mut frontier: Vec<u64> = needed.iter().copied().collect();
227        while let Some(id) = frontier.pop() {
228            let Some(entity) = self.entity(EntityId(id))? else {
229                continue;
230            };
231            for value in &entity.attributes {
232                collect_refs(value, &mut needed, &mut frontier);
233            }
234        }
235        self.build(needed.into_iter().map(EntityId))
236    }
237
238    /// Decodes `wanted` into a model, in file order, with the file's header.
239    fn build(&self, wanted: impl Iterator<Item = EntityId>) -> Result<Model, StepError> {
240        let mut positions: Vec<usize> = wanted.filter_map(|id| self.position(id)).collect();
241        positions.sort_unstable();
242        positions.dedup();
243        let mut model = Model::new();
244        *model.header_mut() = self.header.clone();
245        for position in positions {
246            let (record, _) =
247                parser::decode(self.source, self.spans[position], ParseOptions::strict())?;
248            let (id, entity) = parser::convert(record)?;
249            model.insert(id, entity);
250        }
251        Ok(model)
252    }
253}
254
255fn unrepresentable(span: Span, detail: &str) -> StepError {
256    StepError::Syntax {
257        offset: span.start,
258        detail: detail.into(),
259    }
260}
261
262/// Adds every `Ref` reachable from `value` to the work set.
263fn collect_refs(
264    value: &Value,
265    needed: &mut std::collections::BTreeSet<u64>,
266    frontier: &mut Vec<u64>,
267) {
268    match value {
269        Value::Ref(id) => {
270            if needed.insert(id.0) {
271                frontier.push(id.0);
272            }
273        }
274        Value::List(items) => {
275            for v in items {
276                collect_refs(v, needed, frontier);
277            }
278        }
279        Value::Typed { value, .. } => collect_refs(value, needed, frontier),
280        _ => {}
281    }
282}