ifc_step/index.rs
1//! Offset index: what is in a file, without decoding it.
2//!
3//! # Why this exists
4//!
5//! A type census, or picking 200 walls out of nine million records, needs
6//! to know where each record is and what type it has -- not what it says.
7//! [`Index::scan`] answers that from the borrowed source: it frames every
8//! record with `openbim_step::scan` and keeps its id, byte span and type,
9//! and nothing else. It does not copy the input and does not validate the
10//! inside of a record, which makes it the cheapest way to look into a file.
11//!
12//! For a model that decodes on access with every record validated up front,
13//! read with [`StepCodec`](crate::StepCodec): a strict read is lazy.
14//!
15//! # Trust
16//!
17//! Framing is `openbim_step::scan`: the header is parsed strictly,
18//! everything between records goes through the real lexer, and junk,
19//! missing section ends or a second file appended are errors rather than
20//! skipped. Decoding a record ([`Index::entity`]) runs the same parser and
21//! the same IFC conversion as a full read, on exactly the record's bytes.
22//! `tests/index_agreement.rs` checks both against the eager reader on every
23//! fixture in the repository.
24
25use std::collections::HashMap;
26
27use ifc_model::{Entity, EntityId, Model, Value};
28use openbim_step::Span;
29
30use crate::{parser, StepError};
31
32/// What is in a file, without what it says.
33///
34/// Borrows the source bytes and stores, per record, its id, span and an
35/// index into a table of distinct type names -- about 32 bytes per entity,
36/// two orders of magnitude less than a decoded [`Model`].
37///
38/// Use [`Index::entity`] to decode one record, or [`Index::materialize_closure`]
39/// to build a real [`Model`] from a chosen subset.
40pub struct Index<'a> {
41 source: &'a [u8],
42 header: ifc_model::header::Header,
43 ids: Vec<u64>,
44 spans: Vec<Span>,
45 type_ids: Vec<u32>,
46 type_names: Vec<String>,
47 /// Position of each id in the columns. Files list ids in increasing
48 /// order in practice, where a binary search over `ids` needs nothing
49 /// extra; this map exists only for a file that does not.
50 unordered: Option<HashMap<u64, usize>>,
51}
52
53impl<'a> Index<'a> {
54 /// Frames every record of `source`. Does not decode attributes.
55 ///
56 /// # Errors
57 ///
58 /// A header or framing defect: the wrong physical-file marker, a
59 /// malformed header, anything between records that is not a record, a
60 /// missing `ENDSEC`, content after the end marker, or an instance id or
61 /// type the IFC record model cannot hold (an id beyond `u64`, a complex
62 /// instance). Defects inside a record surface when it is decoded.
63 pub fn scan(source: &'a [u8]) -> Result<Self, StepError> {
64 let scanned = openbim_step::scan(source)?;
65 let mut header = ifc_model::header::Header::default();
66 parser::apply_header(&mut header, scanned.header().standard());
67 let mut ids = Vec::new();
68 let mut spans = Vec::new();
69 let mut type_ids = Vec::new();
70 let mut type_names: Vec<String> = Vec::new();
71 let mut seen: HashMap<String, u32> = HashMap::new();
72 for record in scanned.records() {
73 let record = record?;
74 let id = record.id.as_str().parse::<u64>().map_err(|_| {
75 unrepresentable(
76 record.span,
77 "instance id exceeds the IFC record model range",
78 )
79 })?;
80 let Some(name) = record.name else {
81 return Err(unrepresentable(
82 record.span,
83 "complex STEP instances are not representable in the IFC record model",
84 ));
85 };
86 let upper = name.to_ascii_uppercase();
87 let next = u32::try_from(type_names.len()).expect("fewer than 2^32 distinct types");
88 let type_id = *seen.entry(upper.clone()).or_insert(next);
89 if type_id == next {
90 type_names.push(upper);
91 }
92 ids.push(id);
93 spans.push(record.span);
94 type_ids.push(type_id);
95 }
96 let unordered = (!ids.windows(2).all(|pair| pair[0] < pair[1])).then(|| {
97 // Later records win, as they do in a model read.
98 ids.iter()
99 .enumerate()
100 .map(|(position, id)| (*id, position))
101 .collect()
102 });
103 Ok(Self {
104 source,
105 header,
106 ids,
107 spans,
108 type_ids,
109 type_names,
110 unordered,
111 })
112 }
113
114 /// Number of records found.
115 #[must_use]
116 pub fn len(&self) -> usize {
117 self.ids.len()
118 }
119
120 /// Whether the file has no data records.
121 #[must_use]
122 pub fn is_empty(&self) -> bool {
123 self.ids.is_empty()
124 }
125
126 /// The file header, as a model read would report it.
127 #[must_use]
128 pub fn header(&self) -> &ifc_model::header::Header {
129 &self.header
130 }
131
132 /// Every id, in file order.
133 pub fn ids(&self) -> impl Iterator<Item = EntityId> + '_ {
134 self.ids.iter().copied().map(EntityId)
135 }
136
137 /// Upper-cased type name of `id`, without decoding it.
138 #[must_use]
139 pub fn type_of(&self, id: EntityId) -> Option<&str> {
140 let position = self.position(id)?;
141 Some(&self.type_names[self.type_ids[position] as usize])
142 }
143
144 /// How many records of each type. The census a viewer opens with.
145 #[must_use]
146 pub fn count_by_type(&self) -> std::collections::BTreeMap<&str, usize> {
147 let mut out = std::collections::BTreeMap::new();
148 for type_id in &self.type_ids {
149 *out.entry(self.type_names[*type_id as usize].as_str())
150 .or_insert(0) += 1;
151 }
152 out
153 }
154
155 /// Ids of every record whose type matches `name`, case-insensitively,
156 /// in file order.
157 #[must_use]
158 pub fn ids_of_type(&self, name: &str) -> Vec<EntityId> {
159 let upper = name.to_ascii_uppercase();
160 let Some(type_id) = self.type_names.iter().position(|n| *n == upper) else {
161 return Vec::new();
162 };
163 let type_id = u32::try_from(type_id).expect("fewer than 2^32 distinct types");
164 self.type_ids
165 .iter()
166 .enumerate()
167 .filter(|(_, t)| **t == type_id)
168 .map(|(position, _)| EntityId(self.ids[position]))
169 .collect()
170 }
171
172 fn position(&self, id: EntityId) -> Option<usize> {
173 match &self.unordered {
174 Some(map) => map.get(&id.0).copied(),
175 None => self.ids.binary_search(&id.0).ok(),
176 }
177 }
178
179 /// Decodes one record into an [`Entity`]: the same parser and IFC
180 /// conversion a full read uses, on exactly the record's bytes.
181 ///
182 /// # Errors
183 ///
184 /// When the record is malformed or holds a value the IFC record model
185 /// cannot represent -- exactly what would make a strict read fail.
186 pub fn entity(&self, id: EntityId) -> Result<Option<Entity>, StepError> {
187 let Some(position) = self.position(id) else {
188 return Ok(None);
189 };
190 let record = openbim_step::decode_record_borrowed(self.source, self.spans[position])?;
191 Ok(Some(parser::convert(record)?.1))
192 }
193
194 /// Builds a [`Model`] from exactly `wanted`, and nothing else, with the
195 /// file's header. Ids the file does not contain are skipped.
196 ///
197 /// # Dangling references
198 ///
199 /// The result is a real `Model` but not a self-contained one: an entity
200 /// that referenced `#7` still says `Ref(7)` even when `#7` was not
201 /// requested. Traversal will not resolve it. Use
202 /// [`Index::materialize_closure`] unless you specifically want the raw
203 /// subset and will handle missing targets yourself.
204 ///
205 /// # Errors
206 ///
207 /// The first wanted record that does not decode; see [`Index::entity`].
208 pub fn materialize(&self, wanted: &[EntityId]) -> Result<Model, StepError> {
209 self.build(wanted.iter().copied())
210 }
211
212 /// Builds a [`Model`] containing `wanted` and everything they reach.
213 ///
214 /// Follows references transitively, so the result has no dangling `Ref`
215 /// except to ids the file itself does not define. This is the
216 /// constructor to reach for: it is what makes a subset independently
217 /// usable.
218 ///
219 /// # Errors
220 ///
221 /// The first reached record that does not decode; see [`Index::entity`].
222 pub fn materialize_closure(&self, wanted: &[EntityId]) -> Result<Model, StepError> {
223 let mut needed: std::collections::BTreeSet<u64> = wanted.iter().map(|i| i.0).collect();
224 let mut frontier: Vec<u64> = needed.iter().copied().collect();
225 while let Some(id) = frontier.pop() {
226 let Some(entity) = self.entity(EntityId(id))? else {
227 continue;
228 };
229 for value in &entity.attributes {
230 collect_refs(value, &mut needed, &mut frontier);
231 }
232 }
233 self.build(needed.into_iter().map(EntityId))
234 }
235
236 /// Decodes `wanted` into a model, in file order, with the file's header.
237 fn build(&self, wanted: impl Iterator<Item = EntityId>) -> Result<Model, StepError> {
238 let mut positions: Vec<usize> = wanted.filter_map(|id| self.position(id)).collect();
239 positions.sort_unstable();
240 positions.dedup();
241 let mut model = Model::new();
242 *model.header_mut() = self.header.clone();
243 for position in positions {
244 let record = openbim_step::decode_record_borrowed(self.source, self.spans[position])?;
245 let (id, entity) = parser::convert(record)?;
246 model.insert(id, entity);
247 }
248 Ok(model)
249 }
250}
251
252fn unrepresentable(span: Span, detail: &str) -> StepError {
253 StepError::Syntax {
254 offset: span.start,
255 detail: detail.into(),
256 }
257}
258
259/// Adds every `Ref` reachable from `value` to the work set.
260fn collect_refs(
261 value: &Value,
262 needed: &mut std::collections::BTreeSet<u64>,
263 frontier: &mut Vec<u64>,
264) {
265 match value {
266 Value::Ref(id) => {
267 if needed.insert(id.0) {
268 frontier.push(id.0);
269 }
270 }
271 Value::List(items) => {
272 for v in items {
273 collect_refs(v, needed, frontier);
274 }
275 }
276 Value::Typed { value, .. } => collect_refs(value, needed, frontier),
277 _ => {}
278 }
279}