Skip to main content

codehelion_core/
frontend.rs

1//! The Fast-frontend interface.
2//!
3//! A frontend turns a source file into a flat token stream with comments and
4//! whitespace removed, plus the coarse unit boundaries (functions, methods,
5//! `impl` blocks, closures) used later as clone-report anchors. Lexing is
6//! error-tolerant: a malformed span becomes a [`Diagnostic`] and lexing
7//! continues, so one broken construct never discards the rest of a file.
8//!
9//! The frontend deliberately stops at lexing. Macros and templates are not
10//! expanded; their invocations pass through as ordinary tokens. Normalization
11//! (identifier renaming, literal folding) is applied downstream at fragment
12//! scope, so the stream here carries each token's raw lexeme unchanged.
13//!
14//! Source positions are recorded for reporting only. They are never used as
15//! stable identifiers: fingerprints are built from token kinds and normalized
16//! text, never from a line number or token offset.
17//!
18//! The module also owns [`IrAssembly`], the parser-independent half of a
19//! Structural frontend: line mapping, token interning, byte-to-token lookup
20//! and the recovery data a depth-limited walk produces. Every Structural
21//! frontend assembles its file through it, so a fix to any of those concerns
22//! reaches all languages at once instead of one grammar at a time.
23
24use std::borrow::Borrow;
25use std::collections::HashSet;
26use std::fmt;
27use std::ops::Deref;
28use std::sync::Arc;
29
30use crate::discovery::Language;
31use crate::ir::{ByteRange, IrNode, Shape};
32
33/// A shared, immutable lexeme.
34///
35/// Token text is stored behind a shared pointer so that every occurrence of
36/// the same lexeme in a file shares one allocation instead of owning a copy;
37/// with millions of tokens in scope this is the difference between hundreds
38/// of megabytes and a few. Equality, ordering and hashing follow the text
39/// content, and the type dereferences to [`str`], so call sites treat it
40/// like a borrowed string.
41#[derive(Debug, Clone, Eq)]
42pub struct Lexeme(Arc<str>);
43
44impl Lexeme {
45    /// The lexeme text.
46    #[must_use]
47    pub fn as_str(&self) -> &str {
48        &self.0
49    }
50}
51
52impl PartialEq for Lexeme {
53    fn eq(&self, other: &Self) -> bool {
54        // Interned lexemes of one file share their allocation, so pointer
55        // identity settles most comparisons without touching the bytes.
56        Arc::ptr_eq(&self.0, &other.0) || self.0 == other.0
57    }
58}
59
60impl std::hash::Hash for Lexeme {
61    fn hash<H: std::hash::Hasher>(&self, state: &mut H) {
62        // Must agree with `str::hash` for `Borrow<str>` lookups.
63        self.0.hash(state);
64    }
65}
66
67impl Deref for Lexeme {
68    type Target = str;
69
70    fn deref(&self) -> &str {
71        &self.0
72    }
73}
74
75impl AsRef<str> for Lexeme {
76    fn as_ref(&self) -> &str {
77        &self.0
78    }
79}
80
81impl Borrow<str> for Lexeme {
82    fn borrow(&self) -> &str {
83        &self.0
84    }
85}
86
87impl From<&str> for Lexeme {
88    fn from(text: &str) -> Self {
89        Self(Arc::from(text))
90    }
91}
92
93impl PartialEq<str> for Lexeme {
94    fn eq(&self, other: &str) -> bool {
95        &*self.0 == other
96    }
97}
98
99impl PartialEq<&str> for Lexeme {
100    fn eq(&self, other: &&str) -> bool {
101        &*self.0 == *other
102    }
103}
104
105impl fmt::Display for Lexeme {
106    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
107        f.write_str(&self.0)
108    }
109}
110
111/// Deduplicating store of [`Lexeme`]s, typically one per lexed file.
112///
113/// Interning the same text twice returns two handles to one allocation. The
114/// interner is an implementation detail of memory layout: it never affects
115/// token equality or fingerprints, which follow text content only.
116#[derive(Debug, Default)]
117pub struct LexemeInterner {
118    known: HashSet<Lexeme>,
119}
120
121impl LexemeInterner {
122    /// Create an empty interner.
123    #[must_use]
124    pub fn new() -> Self {
125        Self::default()
126    }
127
128    /// Return the shared lexeme for `text`, allocating it once.
129    pub fn intern(&mut self, text: &str) -> Lexeme {
130        if let Some(found) = self.known.get(text) {
131            return found.clone();
132        }
133        let lexeme = Lexeme::from(text);
134        self.known.insert(lexeme.clone());
135        lexeme
136    }
137}
138
139/// Category of a literal token, used by literal-normalization strategies.
140#[derive(Debug, Clone, Copy, PartialEq, Eq)]
141pub enum LiteralKind {
142    /// Integer literal, e.g. `42`, `0xff`, `1_000`.
143    Integer,
144    /// Floating-point literal, e.g. `1.5`, `2e10`.
145    Float,
146    /// String literal, including raw and byte strings.
147    String,
148    /// Character or byte-character literal.
149    Char,
150    /// Boolean literal (`true` / `false`).
151    Bool,
152}
153
154/// The lexical category of a token.
155#[derive(Debug, Clone, Copy, PartialEq, Eq)]
156pub enum TokenKind {
157    /// An identifier (including raw identifiers).
158    Identifier,
159    /// A language keyword.
160    Keyword,
161    /// A literal of the given category.
162    Literal(LiteralKind),
163    /// A lifetime or label, e.g. `'a`.
164    Lifetime,
165    /// An operator or delimiter.
166    Punctuation,
167    /// Input that could not be lexed into any of the above.
168    Unknown,
169}
170
171impl TokenKind {
172    /// A stable one-byte tag for this kind, for use as fingerprint input.
173    ///
174    /// The literal sub-category does not affect the tag: whether two literals
175    /// are considered equal is a normalization decision, made downstream.
176    #[must_use]
177    pub const fn tag(self) -> u8 {
178        match self {
179            Self::Identifier => 1,
180            Self::Keyword => 2,
181            Self::Literal(_) => 3,
182            Self::Punctuation => 4,
183            Self::Lifetime => 5,
184            Self::Unknown => 6,
185        }
186    }
187}
188
189/// A source position span, recorded for reporting only.
190///
191/// `start_byte`/`end_byte` are byte offsets into the source; `start_line` and
192/// `start_column` are 1-based and counted in characters. None of these fields
193/// may be used to derive a stable identifier.
194#[derive(Debug, Clone, Copy, PartialEq, Eq)]
195pub struct SourceSpan {
196    /// Byte offset of the span start.
197    pub start_byte: usize,
198    /// Byte offset one past the span end.
199    pub end_byte: usize,
200    /// 1-based line of the span start.
201    pub start_line: u32,
202    /// 1-based column (in characters) of the span start.
203    pub start_column: u32,
204}
205
206/// One lexical token.
207#[derive(Debug, Clone, PartialEq, Eq)]
208pub struct Token {
209    /// Lexical category.
210    pub kind: TokenKind,
211    /// The raw lexeme exactly as it appeared in the source.
212    pub text: Lexeme,
213    /// Source position, for reporting only.
214    pub span: SourceSpan,
215}
216
217/// The kind of a recoverable Fast-frontend problem.
218#[derive(Debug, Clone, Copy, PartialEq, Eq)]
219pub enum DiagnosticKind {
220    /// A string literal was not closed before end of file.
221    UnterminatedString,
222    /// A character or byte literal was not closed before end of file.
223    UnterminatedChar,
224    /// A block comment was not closed before end of file.
225    UnterminatedBlockComment,
226    /// A byte that does not begin any valid token.
227    UnexpectedCharacter,
228    /// An opening delimiter needed by a unit boundary had no matching closer.
229    UnmatchedDelimiter,
230}
231
232/// A recoverable Fast-frontend problem. Analysis continues past it.
233#[derive(Debug, Clone, PartialEq, Eq)]
234pub struct Diagnostic {
235    /// What went wrong.
236    pub kind: DiagnosticKind,
237    /// Where it happened.
238    pub span: SourceSpan,
239}
240
241/// The kind of a coarse code unit used as a clone-report anchor.
242#[derive(Debug, Clone, Copy, PartialEq, Eq)]
243pub enum UnitKind {
244    /// A free function.
245    Function,
246    /// A method (a function inside an `impl` block or a record body).
247    Method,
248    /// An `impl` block.
249    Impl,
250    /// A record body: a `class`, `struct` or `union` definition.
251    Record,
252    /// A closure or lambda with a block body.
253    Closure,
254}
255
256impl UnitKind {
257    /// Stable lowercase identifier used in reports.
258    #[must_use]
259    pub const fn name(self) -> &'static str {
260        match self {
261            Self::Function => "function",
262            Self::Method => "method",
263            Self::Impl => "impl",
264            Self::Record => "record",
265            Self::Closure => "closure",
266        }
267    }
268}
269
270/// A coarse code unit: a token range plus its source span.
271#[derive(Debug, Clone, PartialEq, Eq)]
272pub struct Unit {
273    /// The unit's kind.
274    pub kind: UnitKind,
275    /// The unit's name, when the frontend can recover one.
276    pub name: Option<String>,
277    /// Index of the unit's first token in the stream.
278    pub token_start: usize,
279    /// Index one past the unit's last token in the stream.
280    pub token_end: usize,
281    /// Source span covering the unit, for reporting.
282    pub span: SourceSpan,
283}
284
285/// The tokens a recorded range covers, clamped to the stream it addresses.
286///
287/// An error-tolerant frontend may hand back a unit whose token range reaches
288/// past the tokens it recovered, and a range recorded against one stream may be
289/// read against another. Every reader clamps here rather than at its own call
290/// site: two call sites that clamp differently would fingerprint the same code
291/// under two identities, which is a worse defect than the out-of-range read the
292/// clamp exists to prevent. A start past the end yields an empty slice, as does
293/// an end before the start.
294#[must_use]
295pub fn tokens_in_range(tokens: &[Token], token_start: usize, token_end: usize) -> &[Token] {
296    let start = token_start.min(tokens.len());
297    let end = token_end.min(tokens.len()).max(start);
298    &tokens[start..end]
299}
300
301/// The result of lexing one source file.
302#[derive(Debug, Clone)]
303pub struct LexedFile {
304    /// Language the file was lexed as.
305    pub language: Language,
306    /// Version tag of the frontend that produced this result; a fingerprint
307    /// input, so a change to lexing that alters output must change it.
308    pub frontend_version: &'static str,
309    /// Tokens in source order, comments and whitespace removed.
310    pub tokens: Vec<Token>,
311    /// Coarse unit boundaries, in source order.
312    pub units: Vec<Unit>,
313    /// Recoverable problems encountered while lexing.
314    pub diagnostics: Vec<Diagnostic>,
315}
316
317/// The token stream and recovery data of one assembled Structural file.
318///
319/// Produced by [`IrAssembly::finish`]; the error ranges are sorted by start
320/// then end and deduplicated, which is the order every frontend records them
321/// in.
322#[derive(Debug, Clone)]
323pub struct AssembledIr {
324    /// Tokens in source order, comments and whitespace removed.
325    pub tokens: Vec<Token>,
326    /// Byte ranges the frontend could not map onto shapes.
327    pub error_ranges: Vec<ByteRange>,
328    /// Whether any subtree was cut short by the IR depth budget.
329    pub depth_truncated: bool,
330}
331
332/// The parser-independent half of a Structural frontend.
333///
334/// A frontend owns its grammar cursor and its node-classification table; the
335/// rest — interning lexemes, converting byte offsets to line/column, mapping a
336/// node's byte range onto token indices, and recording what a depth-limited
337/// walk had to leave out — is the same work in every language and lives here.
338///
339/// Token indices address the stream this type accumulates, so nodes must be
340/// built after the file's tokens have been pushed.
341#[derive(Debug)]
342pub struct IrAssembly<'s> {
343    source: &'s str,
344    /// Byte offset of the start of each source line.
345    line_starts: Vec<usize>,
346    interner: LexemeInterner,
347    tokens: Vec<Token>,
348    /// Byte start of each emitted token, for mapping node byte ranges onto
349    /// token index ranges by binary search.
350    token_starts: Vec<usize>,
351    error_ranges: Vec<ByteRange>,
352    depth_truncated: bool,
353}
354
355impl<'s> IrAssembly<'s> {
356    /// Start assembling the file `source`.
357    #[must_use]
358    pub fn new(source: &'s str) -> Self {
359        let mut line_starts = vec![0];
360        for (index, byte) in source.bytes().enumerate() {
361            if byte == b'\n' {
362                line_starts.push(index + 1);
363            }
364        }
365        Self {
366            source,
367            line_starts,
368            interner: LexemeInterner::new(),
369            tokens: Vec::new(),
370            token_starts: Vec::new(),
371            error_ranges: Vec::new(),
372            depth_truncated: false,
373        }
374    }
375
376    /// The source text being assembled.
377    #[must_use]
378    pub const fn source(&self) -> &'s str {
379        self.source
380    }
381
382    /// Return the shared lexeme for `text`, allocating it once per file.
383    pub fn intern(&mut self, text: &str) -> Lexeme {
384        self.interner.intern(text)
385    }
386
387    /// 1-based line and character column of a byte offset.
388    #[must_use]
389    pub fn line_column(&self, byte: usize) -> (u32, u32) {
390        let line_index = self
391            .line_starts
392            .partition_point(|&start| start <= byte)
393            .saturating_sub(1);
394        let line_start = self.line_starts.get(line_index).copied().unwrap_or(0);
395        let column_chars = self
396            .source
397            .get(line_start..byte)
398            .map_or(0, |prefix| prefix.chars().count());
399        (
400            u32::try_from(line_index + 1).unwrap_or(u32::MAX),
401            u32::try_from(column_chars + 1).unwrap_or(u32::MAX),
402        )
403    }
404
405    /// The reporting span of `start_byte..end_byte` in this source.
406    #[must_use]
407    pub fn span(&self, start_byte: usize, end_byte: usize) -> SourceSpan {
408        let (start_line, start_column) = self.line_column(start_byte);
409        SourceSpan {
410            start_byte,
411            end_byte,
412            start_line,
413            start_column,
414        }
415    }
416
417    /// Append a token covering `start_byte..end_byte`, computing its position.
418    ///
419    /// Tokens must be pushed in source order: the byte-to-token lookup binary
420    /// searches the starts recorded here.
421    pub fn push_token(&mut self, kind: TokenKind, text: &str, start_byte: usize, end_byte: usize) {
422        let span = self.span(start_byte, end_byte);
423        self.push_spanned_token(kind, text, span);
424    }
425
426    /// Append a token whose span was established elsewhere.
427    ///
428    /// Used where the positions come from a stream lexed over the whole file
429    /// while the assembly itself only covers a prefix of it.
430    pub fn push_spanned_token(&mut self, kind: TokenKind, text: &str, span: SourceSpan) {
431        let text = self.interner.intern(text);
432        self.token_starts.push(span.start_byte);
433        self.tokens.push(Token { kind, text, span });
434    }
435
436    /// Number of tokens appended so far, which is also the index the next one
437    /// will take.
438    #[must_use]
439    pub const fn token_count(&self) -> usize {
440        self.tokens.len()
441    }
442
443    /// Index of the first appended token starting at or after `byte`.
444    #[must_use]
445    pub fn token_index_at(&self, byte: usize) -> usize {
446        self.token_starts.partition_point(|&start| start < byte)
447    }
448
449    /// The token index range a node's byte range covers.
450    #[must_use]
451    pub fn token_bounds(&self, range: ByteRange) -> (usize, usize) {
452        (
453            self.token_index_at(range.start),
454            self.token_index_at(range.end),
455        )
456    }
457
458    /// Record a byte range the frontend could not map onto shapes.
459    pub fn record_error_range(&mut self, range: ByteRange) {
460        self.error_ranges.push(range);
461    }
462
463    /// Whether a subtree was cut short by the IR depth budget.
464    #[must_use]
465    pub const fn depth_truncated(&self) -> bool {
466        self.depth_truncated
467    }
468
469    /// Preserve an unvisited subtree as recoverable truncation data.
470    ///
471    /// Marks the file truncated, records `range` as an error range, and
472    /// returns the [`Shape::Error`] leaf that stands in for the subtree. The
473    /// leaf's token range covers the tokens the omitted subtree contributed,
474    /// so a truncated file still addresses its own token stream correctly.
475    pub fn truncate_at_depth(&mut self, range: ByteRange) -> IrNode {
476        self.depth_truncated = true;
477        self.record_error_range(range);
478        self.error_node(range)
479    }
480
481    /// A childless [`Shape::Error`] leaf over `range`.
482    #[must_use]
483    pub fn error_node(&self, range: ByteRange) -> IrNode {
484        let (token_start, token_end) = self.token_bounds(range);
485        IrNode {
486            shape: Shape::Error,
487            name: None,
488            token_start,
489            token_end,
490            range,
491            children: Vec::new(),
492        }
493    }
494
495    /// Finish the file: take the token stream and normalize the error ranges.
496    #[must_use]
497    pub fn finish(mut self) -> AssembledIr {
498        self.error_ranges
499            .sort_unstable_by_key(|range| (range.start, range.end));
500        self.error_ranges.dedup();
501        AssembledIr {
502            tokens: self.tokens,
503            error_ranges: self.error_ranges,
504            depth_truncated: self.depth_truncated,
505        }
506    }
507}
508
509/// A Fast-mode lexer for one language.
510pub trait Frontend {
511    /// The language this frontend lexes.
512    fn language(&self) -> Language;
513
514    /// The frontend's version tag, used as a fingerprint input.
515    fn frontend_version(&self) -> &'static str;
516
517    /// Lex `source` into a token stream with unit boundaries and diagnostics.
518    fn lex(&self, source: &str) -> LexedFile;
519}
520
521#[cfg(test)]
522mod tests {
523    use super::*;
524
525    #[test]
526    fn kind_tags_are_distinct_and_stable() {
527        let tags = [
528            TokenKind::Identifier.tag(),
529            TokenKind::Keyword.tag(),
530            TokenKind::Literal(LiteralKind::Integer).tag(),
531            TokenKind::Punctuation.tag(),
532            TokenKind::Lifetime.tag(),
533            TokenKind::Unknown.tag(),
534        ];
535        let mut sorted = tags.to_vec();
536        sorted.sort_unstable();
537        sorted.dedup();
538        assert_eq!(sorted.len(), tags.len(), "tags must be distinct");
539        // The literal sub-category does not change the tag.
540        assert_eq!(
541            TokenKind::Literal(LiteralKind::Integer).tag(),
542            TokenKind::Literal(LiteralKind::String).tag()
543        );
544    }
545
546    /// The shared clamp answers exactly what every call site computed for
547    /// itself before, including for the ranges an error-tolerant frontend can
548    /// hand back: a fingerprint that moved because the clamp moved would be a
549    /// worse defect than the out-of-range read the clamp prevents.
550    #[test]
551    fn the_shared_clamp_covers_the_tokens_each_call_site_clamped_to() {
552        let token = |text: &str| Token {
553            kind: TokenKind::Identifier,
554            text: text.into(),
555            span: SourceSpan {
556                start_byte: 0,
557                end_byte: 1,
558                start_line: 1,
559                start_column: 1,
560            },
561        };
562        let tokens = [token("a"), token("b"), token("c")];
563        // What the call sites computed inline before there was one helper.
564        let previous = |start: usize, end: usize| {
565            let start = start.min(tokens.len());
566            let end = end.min(tokens.len()).max(start);
567            &tokens[start..end]
568        };
569
570        for (start, end) in [
571            (0, 3),
572            (1, 2),
573            (2, 2),
574            (0, 9),
575            (5, 9),
576            (3, 3),
577            (2, 1),
578            (9, 1),
579        ] {
580            assert_eq!(
581                tokens_in_range(&tokens, start, end),
582                previous(start, end),
583                "range {start}..{end}"
584            );
585        }
586        assert!(
587            tokens_in_range(&tokens, 5, 9).is_empty(),
588            "{:?}",
589            tokens_in_range(&tokens, 5, 9)
590        );
591        assert!(
592            tokens_in_range(&tokens, 2, 1).is_empty(),
593            "{:?}",
594            tokens_in_range(&tokens, 2, 1)
595        );
596        assert!(
597            tokens_in_range(&[], 0, 4).is_empty(),
598            "{:?}",
599            tokens_in_range(&[], 0, 4)
600        );
601        assert_eq!(tokens_in_range(&tokens, 1, 9).len(), 2);
602    }
603
604    #[test]
605    fn interning_shares_one_allocation_per_text() {
606        let mut interner = LexemeInterner::new();
607        let a = interner.intern("alpha");
608        let b = interner.intern("alpha");
609        let c = interner.intern("beta");
610        assert!(Arc::ptr_eq(&a.0, &b.0), "same text must share storage");
611        assert!(!Arc::ptr_eq(&a.0, &c.0));
612        assert_eq!(a, b);
613        assert_ne!(a, c);
614    }
615
616    #[test]
617    fn lexeme_equality_and_hash_follow_content_across_interners() {
618        let a = LexemeInterner::new().intern("shared");
619        let b = LexemeInterner::new().intern("shared");
620        assert!(
621            !Arc::ptr_eq(&a.0, &b.0),
622            "distinct interners allocate separately"
623        );
624        assert_eq!(a, b, "equality is by content, not by pointer");
625        let set: HashSet<Lexeme> = [a].into();
626        assert!(set.contains("shared"), "str lookups must hash consistently");
627    }
628
629    #[test]
630    fn lexeme_compares_against_plain_strings() {
631        let lexeme = Lexeme::from("fn");
632        assert_eq!(lexeme, "fn");
633        assert_eq!(lexeme.as_str(), "fn");
634        assert_eq!(lexeme.to_string(), "fn");
635        assert_eq!(lexeme.as_bytes(), b"fn");
636    }
637
638    #[test]
639    fn assembly_columns_count_characters_not_bytes() {
640        let source = "let é = 1;\nlet b = 2;\n";
641        let assembly = IrAssembly::new(source);
642        let e_acute = source.find('é').unwrap_or_default();
643        assert_eq!(assembly.line_column(0), (1, 1));
644        assert_eq!(assembly.line_column(e_acute), (1, 5));
645        // The token after the two-byte character is one column further right,
646        // not two.
647        assert_eq!(assembly.line_column(e_acute + 'é'.len_utf8()), (1, 6));
648        let second_line = source.find("let b").unwrap_or_default();
649        assert_eq!(assembly.line_column(second_line), (2, 1));
650    }
651
652    #[test]
653    fn assembly_maps_byte_ranges_onto_token_indices() {
654        let source = "a bb ccc";
655        let mut assembly = IrAssembly::new(source);
656        for (start, end) in [(0, 1), (2, 4), (5, 8)] {
657            let text = source.get(start..end).unwrap_or_default();
658            assembly.push_token(TokenKind::Identifier, text, start, end);
659        }
660        assert_eq!(assembly.token_count(), 3);
661        // A range starting inside a token begins at the next whole token.
662        assert_eq!(assembly.token_index_at(1), 1);
663        assert_eq!(
664            assembly.token_bounds(ByteRange { start: 2, end: 8 }),
665            (1, 3)
666        );
667        assert_eq!(
668            assembly.token_bounds(ByteRange {
669                start: 0,
670                end: source.len()
671            }),
672            (0, 3)
673        );
674    }
675
676    #[test]
677    fn assembly_records_a_depth_limited_subtree_as_an_error_leaf() {
678        let source = "a bb ccc";
679        let mut assembly = IrAssembly::new(source);
680        for (start, end) in [(0, 1), (2, 4), (5, 8)] {
681            let text = source.get(start..end).unwrap_or_default();
682            assembly.push_token(TokenKind::Identifier, text, start, end);
683        }
684        assert!(!assembly.depth_truncated());
685        let omitted = ByteRange { start: 2, end: 8 };
686        let leaf = assembly.truncate_at_depth(omitted);
687        assert_eq!(leaf.shape, Shape::Error);
688        assert_eq!(leaf.name, None);
689        assert_eq!((leaf.token_start, leaf.token_end), (1, 3));
690        assert_eq!(leaf.range, omitted);
691        assert!(leaf.children.is_empty(), "children: {:?}", leaf.children);
692        assert!(assembly.depth_truncated());
693        let assembled = assembly.finish();
694        assert!(assembled.depth_truncated);
695        assert_eq!(assembled.error_ranges, vec![omitted]);
696    }
697
698    #[test]
699    fn assembly_finishes_with_ordered_unique_error_ranges() {
700        let mut assembly = IrAssembly::new("abc");
701        for range in [(2, 3), (0, 1), (2, 3), (0, 2)] {
702            assembly.record_error_range(ByteRange {
703                start: range.0,
704                end: range.1,
705            });
706        }
707        let assembled = assembly.finish();
708        assert_eq!(
709            assembled.error_ranges,
710            vec![
711                ByteRange { start: 0, end: 1 },
712                ByteRange { start: 0, end: 2 },
713                ByteRange { start: 2, end: 3 },
714            ]
715        );
716        assert!(!assembled.depth_truncated);
717    }
718
719    #[test]
720    fn assembly_keeps_spans_established_elsewhere() {
721        // Tokens of a region the assembly's own source does not cover carry
722        // the positions the whole-file stream gave them.
723        let mut assembly = IrAssembly::new("fn a() {}");
724        let span = SourceSpan {
725            start_byte: 900,
726            end_byte: 901,
727            start_line: 42,
728            start_column: 7,
729        };
730        assembly.push_spanned_token(TokenKind::Punctuation, "}", span);
731        let assembled = assembly.finish();
732        assert_eq!(assembled.tokens.len(), 1);
733        assert_eq!(assembled.tokens[0].span, span);
734        assert_eq!(assembled.tokens[0].text, "}");
735    }
736
737    #[test]
738    fn unit_kind_names_are_stable() {
739        assert_eq!(UnitKind::Function.name(), "function");
740        assert_eq!(UnitKind::Method.name(), "method");
741        assert_eq!(UnitKind::Impl.name(), "impl");
742        assert_eq!(UnitKind::Closure.name(), "closure");
743    }
744}