ifc_step/index.rs
1//! Offset index: what is in a file, without decoding it.
2//!
3//! # Why this exists
4//!
5//! A type census, or picking 200 walls out of nine million records, needs
6//! to know where each record is and what type it has -- not what it says.
7//! [`Index::scan`] answers that from the borrowed source: it frames every
8//! record with `openbim_step::scan` and keeps its id, byte span and type,
9//! and nothing else. It does not copy the input and does not validate the
10//! inside of a record, which makes it the cheapest way to look into a file.
11//!
12//! For a model that decodes on access with every record validated up front,
13//! read with [`StepCodec`](crate::StepCodec): a strict read is lazy.
14//!
15//! # Trust
16//!
17//! Framing is `openbim_step::scan`: the header is parsed strictly,
18//! everything between records goes through the real lexer, and junk,
19//! missing section ends or a second file appended are errors rather than
20//! skipped. Decoding a record ([`Index::entity`]) runs the same parser and
21//! the same IFC conversion as a full read, on exactly the record's bytes.
22//! `tests/index_agreement.rs` checks both against the eager reader on every
23//! fixture in the repository.
24
25use std::collections::HashMap;
26
27use ifc_model::{Entity, EntityId, Model, Value};
28use openbim_step::{ParseOptions, Span};
29
30use crate::{parser, StepError};
31
32/// What is in a file, without what it says.
33///
34/// Borrows the source bytes and stores, per record, its id, span and an
35/// index into a table of distinct type names -- about 32 bytes per entity,
36/// two orders of magnitude less than a decoded [`Model`].
37///
38/// Use [`Index::entity`] to decode one record, or [`Index::materialize_closure`]
39/// to build a real [`Model`] from a chosen subset.
40pub struct Index<'a> {
41 source: &'a [u8],
42 header: ifc_model::header::Header,
43 ids: Vec<u64>,
44 spans: Vec<Span>,
45 type_ids: Vec<u32>,
46 type_names: Vec<String>,
47 /// Position of each id in the columns. Files list ids in increasing
48 /// order in practice, where a binary search over `ids` needs nothing
49 /// extra; this map exists only for a file that does not.
50 unordered: Option<HashMap<u64, usize>>,
51}
52
53impl<'a> Index<'a> {
54 /// Frames every record of `source`. Does not decode attributes.
55 ///
56 /// # Errors
57 ///
58 /// A header or framing defect: the wrong physical-file marker, a
59 /// malformed header, anything between records that is not a record, a
60 /// missing `ENDSEC`, content after the end marker, or an instance id or
61 /// type the IFC record model cannot hold (an id beyond `u64`, a complex
62 /// instance). Defects inside a record surface when it is decoded.
63 pub fn scan(source: &'a [u8]) -> Result<Self, StepError> {
64 let scanned =
65 openbim_step::scan(source).map_err(|error| StepError::from_step(source, error))?;
66 let mut header = ifc_model::header::Header::default();
67 parser::apply_header(&mut header, scanned.header().standard());
68 let mut ids = Vec::new();
69 let mut spans = Vec::new();
70 let mut type_ids = Vec::new();
71 let mut type_names: Vec<String> = Vec::new();
72 let mut seen: HashMap<String, u32> = HashMap::new();
73 for record in scanned.records() {
74 let record = record.map_err(|error| StepError::from_step(source, error))?;
75 let id = record.id.as_str().parse::<u64>().map_err(|_| {
76 unrepresentable(
77 record.span,
78 "instance id exceeds the IFC record model range",
79 )
80 })?;
81 let Some(name) = record.name else {
82 return Err(unrepresentable(
83 record.span,
84 "complex STEP instances are not representable in the IFC record model",
85 ));
86 };
87 let upper = name.to_ascii_uppercase();
88 let next = u32::try_from(type_names.len()).expect("fewer than 2^32 distinct types");
89 let type_id = *seen.entry(upper.clone()).or_insert(next);
90 if type_id == next {
91 type_names.push(upper);
92 }
93 ids.push(id);
94 spans.push(record.span);
95 type_ids.push(type_id);
96 }
97 let unordered = (!ids.windows(2).all(|pair| pair[0] < pair[1])).then(|| {
98 // Later records win, as they do in a model read.
99 ids.iter()
100 .enumerate()
101 .map(|(position, id)| (*id, position))
102 .collect()
103 });
104 Ok(Self {
105 source,
106 header,
107 ids,
108 spans,
109 type_ids,
110 type_names,
111 unordered,
112 })
113 }
114
115 /// Number of records found.
116 #[must_use]
117 pub fn len(&self) -> usize {
118 self.ids.len()
119 }
120
121 /// Whether the file has no data records.
122 #[must_use]
123 pub fn is_empty(&self) -> bool {
124 self.ids.is_empty()
125 }
126
127 /// The file header, as a model read would report it.
128 #[must_use]
129 pub fn header(&self) -> &ifc_model::header::Header {
130 &self.header
131 }
132
133 /// Every id, in file order.
134 pub fn ids(&self) -> impl Iterator<Item = EntityId> + '_ {
135 self.ids.iter().copied().map(EntityId)
136 }
137
138 /// Upper-cased type name of `id`, without decoding it.
139 #[must_use]
140 pub fn type_of(&self, id: EntityId) -> Option<&str> {
141 let position = self.position(id)?;
142 Some(&self.type_names[self.type_ids[position] as usize])
143 }
144
145 /// How many records of each type. The census a viewer opens with.
146 #[must_use]
147 pub fn count_by_type(&self) -> std::collections::BTreeMap<&str, usize> {
148 let mut out = std::collections::BTreeMap::new();
149 for type_id in &self.type_ids {
150 *out.entry(self.type_names[*type_id as usize].as_str())
151 .or_insert(0) += 1;
152 }
153 out
154 }
155
156 /// Ids of every record whose type matches `name`, case-insensitively,
157 /// in file order.
158 #[must_use]
159 pub fn ids_of_type(&self, name: &str) -> Vec<EntityId> {
160 let upper = name.to_ascii_uppercase();
161 let Some(type_id) = self.type_names.iter().position(|n| *n == upper) else {
162 return Vec::new();
163 };
164 let type_id = u32::try_from(type_id).expect("fewer than 2^32 distinct types");
165 self.type_ids
166 .iter()
167 .enumerate()
168 .filter(|(_, t)| **t == type_id)
169 .map(|(position, _)| EntityId(self.ids[position]))
170 .collect()
171 }
172
173 fn position(&self, id: EntityId) -> Option<usize> {
174 match &self.unordered {
175 Some(map) => map.get(&id.0).copied(),
176 None => self.ids.binary_search(&id.0).ok(),
177 }
178 }
179
180 /// Decodes one record into an [`Entity`]: the same parser and IFC
181 /// conversion a full read uses, on exactly the record's bytes.
182 ///
183 /// # Errors
184 ///
185 /// When the record is malformed or holds a value the IFC record model
186 /// cannot represent -- exactly what would make a strict read fail.
187 pub fn entity(&self, id: EntityId) -> Result<Option<Entity>, StepError> {
188 let Some(position) = self.position(id) else {
189 return Ok(None);
190 };
191 let (record, _) =
192 parser::decode(self.source, self.spans[position], ParseOptions::strict())?;
193 Ok(Some(parser::convert(record)?.1))
194 }
195
196 /// Builds a [`Model`] from exactly `wanted`, and nothing else, with the
197 /// file's header. Ids the file does not contain are skipped.
198 ///
199 /// # Dangling references
200 ///
201 /// The result is a real `Model` but not a self-contained one: an entity
202 /// that referenced `#7` still says `Ref(7)` even when `#7` was not
203 /// requested. Traversal will not resolve it. Use
204 /// [`Index::materialize_closure`] unless you specifically want the raw
205 /// subset and will handle missing targets yourself.
206 ///
207 /// # Errors
208 ///
209 /// The first wanted record that does not decode; see [`Index::entity`].
210 pub fn materialize(&self, wanted: &[EntityId]) -> Result<Model, StepError> {
211 self.build(wanted.iter().copied())
212 }
213
214 /// Builds a [`Model`] containing `wanted` and everything they reach.
215 ///
216 /// Follows references transitively, so the result has no dangling `Ref`
217 /// except to ids the file itself does not define. This is the
218 /// constructor to reach for: it is what makes a subset independently
219 /// usable.
220 ///
221 /// # Errors
222 ///
223 /// The first reached record that does not decode; see [`Index::entity`].
224 pub fn materialize_closure(&self, wanted: &[EntityId]) -> Result<Model, StepError> {
225 let mut needed: std::collections::BTreeSet<u64> = wanted.iter().map(|i| i.0).collect();
226 let mut frontier: Vec<u64> = needed.iter().copied().collect();
227 while let Some(id) = frontier.pop() {
228 let Some(entity) = self.entity(EntityId(id))? else {
229 continue;
230 };
231 for value in &entity.attributes {
232 collect_refs(value, &mut needed, &mut frontier);
233 }
234 }
235 self.build(needed.into_iter().map(EntityId))
236 }
237
238 /// Decodes `wanted` into a model, in file order, with the file's header.
239 fn build(&self, wanted: impl Iterator<Item = EntityId>) -> Result<Model, StepError> {
240 let mut positions: Vec<usize> = wanted.filter_map(|id| self.position(id)).collect();
241 positions.sort_unstable();
242 positions.dedup();
243 let mut model = Model::new();
244 *model.header_mut() = self.header.clone();
245 for position in positions {
246 let (record, _) =
247 parser::decode(self.source, self.spans[position], ParseOptions::strict())?;
248 let (id, entity) = parser::convert(record)?;
249 model.insert(id, entity);
250 }
251 Ok(model)
252 }
253}
254
255fn unrepresentable(span: Span, detail: &str) -> StepError {
256 StepError::Syntax {
257 offset: span.start,
258 detail: detail.into(),
259 }
260}
261
262/// Adds every `Ref` reachable from `value` to the work set.
263fn collect_refs(
264 value: &Value,
265 needed: &mut std::collections::BTreeSet<u64>,
266 frontier: &mut Vec<u64>,
267) {
268 match value {
269 Value::Ref(id) => {
270 if needed.insert(id.0) {
271 frontier.push(id.0);
272 }
273 }
274 Value::List(items) => {
275 for v in items {
276 collect_refs(v, needed, frontier);
277 }
278 }
279 Value::Typed { value, .. } => collect_refs(value, needed, frontier),
280 _ => {}
281 }
282}