cinrs_core/capture.rs
1//! Input capture and source mapping.
2//!
3//! The whole point of `cinrs` is that a diagnostic produced anywhere in the
4//! pipeline — by our own lexer/parser/sema, or by `rustc` on the code we
5//! generate — must point at the exact *C* token that caused it. To make that
6//! possible the front end never works on a `TokenStream` directly: it first
7//! recovers a plain-text C translation unit plus a [`SourceMap`] that can turn
8//! any byte offset in that text back into a [`proc_macro2::Span`].
9//!
10//! # Capture strategies
11//!
12//! There are two shapes of input, and the raw-token shape has two strategies:
13//!
14//! * **String-literal mode** ([`InputMode::StringLiteral`]). The entire macro
15//! input is a single Rust string literal (`"…"`, `r"…"`, `r#"…"#`). We
16//! unescape it and use the text as-is. This accepts *any* C code, including
17//! the lexemes the Rust 2024 lexer rejects (hex float literals, `'ab'`,
18//! `L"…"`, `\` line continuations, `##`). The catch is that stable Rust has
19//! no way to build a span pointing *inside* a string literal
20//! (`Literal::subspan` is unstable), so every diagnostic is reported at the
21//! literal as a whole and the line/column inside the C text is appended to
22//! the message instead. Such a file is flagged as not [`SourceFile::precise`]
23//! — unless the caller hands capture a [`Subspan`] hook, which is what the
24//! `nightly` feature of `cinrs-macros` does; see [`Subspan`].
25//!
26//! * **Raw-token mode**, primary path ([`InputMode::FileSlice`]). We flatten
27//! the token trees, ask the first token for [`Span::local_file`], read that
28//! `.rs` file from disk and slice out `first.start() .. last.end()`. This
29//! gives back the *exact* original text, including whitespace, newlines and
30//! comments — which the [preprocessor](crate::pp) needs, since it is
31//! line-oriented and since `#error` reproduces what was written. The slice
32//! is validated against every token's [`Span::source_text`] before it is
33//! trusted.
34//!
35//! * **File mode** ([`InputMode::CFile`]), which is what
36//! `include_c99!("…")` captures with [`capture_c_file`]: the text is a `.c`
37//! file read from disk, and nothing in the `.rs` file corresponds to a
38//! position in it, so every range resolves to the span of the macro
39//! invocation and the file names its own position in the message — exactly
40//! as an `#include`d header does.
41//!
42//! * **Raw-token mode**, search path (also [`InputMode::FileSlice`]). A host
43//! may hand a procedural macro tokens with no positions at all:
44//! `rust-analyzer` reports no file, no source text and line 1 column 0 for
45//! every token alike. The text is still on disk, and which text it is can be
46//! *proved* — the `locate` module searches the crate's `.rs` files, the
47//! directory `CARGO_MANIFEST_DIR` names, for an invocation whose token
48//! sequence is exactly the one we were handed. A match gives the same three
49//! things the primary path gives (path, text, anchors), so everything
50//! downstream — diagnostics, `__FILE__`, a quoted `#include`, `include_str!`
51//! tracking, the unit id — is unchanged. See [`Origin`] for what capture has
52//! to be told for this to be possible, and note that it can only run when the
53//! primary path found no position whatsoever: in a normal build it costs
54//! nothing because it never happens.
55//!
56//! * **Raw-token mode**, fallback path ([`InputMode::Reconstructed`]). When
57//! there is no local file (macro-generated input, some IDE contexts such as
58//! rust-analyzer, or unit tests that build a `TokenStream` with
59//! `TokenStream::from_str`), or when the slice fails validation, we rebuild
60//! the text from the tokens. Where their positions are usable, every token
61//! goes at its original line/column, padded with newlines and spaces:
62//! comments are lost, but the line structure — and therefore all reported
63//! positions — survives. Where they are not (every token at the same
64//! position, which is what `rust-analyzer` gives an unsaved buffer), the text
65//! is rebuilt from the tokens alone: one space between two tokens, none where
66//! the host says they were written together, so that `->`, `<<=` and `++`
67//! survive and `- -` stays two tokens. A *directive* is a line, and tokens
68//! keep no lines; the forms whose end the tokens themselves give away
69//! (`#include <…>`, `#ifdef X`, `#endif`, …) are written on a line of their
70//! own and anything else — `#define`, `#if` — is one clear diagnostic rather
71//! than a guess.
72//!
73//! # Coordinates
74//!
75//! [`SourceMap`] owns any number of [`SourceFile`]s in a single, global byte
76//! offset space: file *i* occupies `base .. base + text.len()`. A [`Pos`] is
77//! therefore enough to identify both a file and an offset inside it, which is
78//! what will let `#include` drop extra files into the same map without
79//! changing a single signature in the lexer, parser or AST.
80//!
81//! A file also remembers which `.rs` file it came from and which line of it
82//! the text's own line 1 sits on, which is what the preprocessor's `__FILE__`
83//! and `__LINE__` are made of.
84
85use proc_macro2::{Delimiter, LineColumn, Spacing, Span, TokenStream, TokenTree};
86use std::cell::RefCell;
87use std::collections::HashMap;
88use std::ops::Range;
89use std::path::{Path, PathBuf};
90use std::rc::Rc;
91
92use crate::diag::{Diagnostic, Diagnostics};
93use crate::locate;
94
95/// A position in the global byte-offset space managed by a [`SourceMap`].
96pub type Pos = u32;
97
98/// Identifies one file inside a [`SourceMap`].
99#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)]
100pub struct FileId(u32);
101
102impl FileId {
103 /// The index of this file inside its [`SourceMap`].
104 pub fn index(self) -> usize {
105 self.0 as usize
106 }
107}
108
109/// A half-open range `[start, end)` in a [`SourceMap`]'s global offset space.
110#[derive(Clone, Copy, PartialEq, Eq, Debug)]
111pub struct SourceRange {
112 /// First byte of the range.
113 pub start: Pos,
114 /// One past the last byte of the range.
115 pub end: Pos,
116}
117
118impl SourceRange {
119 /// Builds a range, clamping `end` so that it is never before `start`.
120 pub fn new(start: Pos, end: Pos) -> Self {
121 Self {
122 start,
123 end: if end < start { start } else { end },
124 }
125 }
126
127 /// An empty range at `pos`.
128 pub fn at(pos: Pos) -> Self {
129 Self {
130 start: pos,
131 end: pos,
132 }
133 }
134
135 /// The smallest range covering both `self` and `other`.
136 pub fn join(self, other: Self) -> Self {
137 Self {
138 start: self.start.min(other.start),
139 end: self.end.max(other.end),
140 }
141 }
142
143 /// Length in bytes.
144 pub fn len(self) -> u32 {
145 self.end - self.start
146 }
147
148 /// Whether the range is empty.
149 pub fn is_empty(self) -> bool {
150 self.start == self.end
151 }
152}
153
154/// `(start, end, span)` triples describing where each Rust token landed in a
155/// captured file, in file-local byte offsets.
156pub(crate) type AnchorList = Vec<(u32, u32, Span)>;
157
158// ---------------------------------------------------------------------------
159// spans inside a string literal
160// ---------------------------------------------------------------------------
161
162/// Turns a byte range of the macro input's *source spelling* into a span
163/// pointing exactly at those bytes.
164///
165/// This is how [string-literal mode](InputMode::StringLiteral) stops being
166/// imprecise. `proc_macro::Literal::subspan` does exactly this job, but it is
167/// unstable (`proc_macro_span`) and `proc_macro2` does not re-export it, so
168/// this crate — which never touches `proc_macro` — takes it as a hook instead:
169/// `cinrs-macros`, compiled with its `nightly` feature, passes one in, and
170/// everything else passes [`None`] and keeps the ` (at line …)` suffix.
171///
172/// The range is measured in bytes of the literal *as written*, `r#"` prefix,
173/// closing `"#` and escape sequences and all, which is what `subspan` wants;
174/// [`SourceMap`] does the translation from C-text offsets. Returning [`None`]
175/// — which the real `subspan` does whenever the literal has no source of its
176/// own, as under `rust-analyzer` — falls back to the whole-literal span.
177///
178/// Results are memoised, because a span is asked for once per *generated
179/// token* and the real implementation is a round trip into the compiler.
180#[derive(Clone)]
181pub struct Subspan(Rc<SubspanInner>);
182
183struct SubspanInner {
184 resolve: Box<dyn Fn(Range<usize>) -> Option<Span>>,
185 cache: RefCell<HashMap<(u32, u32), Option<Span>>>,
186}
187
188impl Subspan {
189 /// Wraps a `subspan` implementation.
190 pub fn new(resolve: impl Fn(Range<usize>) -> Option<Span> + 'static) -> Self {
191 Self(Rc::new(SubspanInner {
192 resolve: Box::new(resolve),
193 cache: RefCell::new(HashMap::new()),
194 }))
195 }
196
197 /// The span of `start .. end` in the literal's spelling.
198 fn get(&self, start: u32, end: u32) -> Option<Span> {
199 if let Some(hit) = self.0.cache.borrow().get(&(start, end)) {
200 return *hit;
201 }
202 let span = (self.0.resolve)(start as usize..end as usize);
203 self.0.cache.borrow_mut().insert((start, end), span);
204 span
205 }
206}
207
208impl std::fmt::Debug for Subspan {
209 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
210 f.write_str("Subspan")
211 }
212}
213
214// ---------------------------------------------------------------------------
215// what the host says about the invocation itself
216// ---------------------------------------------------------------------------
217
218/// What the *host* can tell the front end about an invocation, over and above
219/// the tokens themselves.
220///
221/// Two of the three capture strategies need something a `TokenStream` does not
222/// carry:
223///
224/// * The search that finds the invocation in the crate's own sources needs the
225/// crate's directory — `CARGO_MANIFEST_DIR`, which Cargo sets for `rustc` and
226/// for `rust-analyzer` alike — and the name the invocation may be written
227/// with, which narrows the candidates.
228/// * String-literal mode needs the [`Subspan`] hook to put a caret inside the
229/// literal.
230///
231/// The default is an origin that says nothing: no directory, so no search, and
232/// no hook. That is what [`capture`], [`analyze`](crate::analyze) and
233/// [`expand`](crate::expand) use — a test building a `TokenStream` from a
234/// string must not depend on the developer's working directory, or on what
235/// happens to be written in the crate around it. The procedural macros pass
236/// [`Origin::new`], which is the real thing.
237#[derive(Clone, Default, Debug)]
238pub struct Origin {
239 entry_names: &'static [&'static str],
240 crate_dir: Option<PathBuf>,
241 subspan: Option<Subspan>,
242}
243
244impl Origin {
245 /// An origin that tells capture nothing; see the [type
246 /// documentation](Origin).
247 pub fn unknown() -> Self {
248 Self::default()
249 }
250
251 /// The origin of a real expansion of an entry point that may be written
252 /// with any of `entry_names` (`&["c89", "c90"]` for the two names one
253 /// entry point has), in the crate Cargo names.
254 ///
255 /// The names carry no `!` and no path: an invocation is looked for by its
256 /// last path segment, so `cinrs::c99!` is found by `"c99"`.
257 pub fn new(entry_names: &'static [&'static str]) -> Self {
258 Self {
259 entry_names,
260 crate_dir: std::env::var_os(crate::include::MANIFEST_DIR_VAR).map(PathBuf::from),
261 subspan: None,
262 }
263 }
264
265 /// This origin with a hook that can point inside a string literal; see
266 /// [`Subspan`].
267 pub fn with_subspan(mut self, subspan: Option<Subspan>) -> Self {
268 self.subspan = subspan;
269 self
270 }
271
272 /// This origin with the directory to search set explicitly.
273 ///
274 /// [`Origin::new`] reads `CARGO_MANIFEST_DIR`, which is process-global; a
275 /// test says which directory it means instead.
276 pub fn in_dir(mut self, dir: impl Into<PathBuf>) -> Self {
277 self.crate_dir = Some(dir.into());
278 self
279 }
280
281 /// The names an invocation of this entry point may be written with.
282 pub fn entry_names(&self) -> &'static [&'static str] {
283 self.entry_names
284 }
285
286 /// The directory the search walks, when there is one.
287 pub fn crate_dir(&self) -> Option<&Path> {
288 self.crate_dir.as_deref()
289 }
290
291 /// The entry point's own name, as a diagnostic spells it.
292 fn entry_name(&self) -> &str {
293 self.entry_names.first().copied().unwrap_or("c99")
294 }
295
296 /// Searches the crate's sources for the invocation `toks` came from.
297 ///
298 /// [`None`] whenever the search cannot run at all — no directory, which is
299 /// every context but a real expansion — or found nothing it could prove.
300 fn search(&self, toks: &[FlatTok], accept: impl FnMut(&Path) -> bool) -> Option<locate::Slice> {
301 let dir = self.crate_dir.as_deref()?;
302 locate::Search {
303 dir,
304 entry_names: self.entry_names,
305 caps: locate::Caps::default(),
306 }
307 .find(toks, accept)
308 }
309}
310
311/// Where each byte of a captured text sits in the input's source spelling.
312enum Spelling {
313 /// A raw string literal: the spelling is the text shifted by the length of
314 /// the `r#…"` prefix, since nothing between the quotes is decoded.
315 Shift(u32),
316 /// A string literal with escapes: the spelling offset of every byte of the
317 /// text, plus one entry past the end. All the bytes an escape decodes to
318 /// share the offset of the backslash it started at.
319 Table(Vec<u32>),
320}
321
322impl Spelling {
323 /// The spelling range a text range came from.
324 fn range(&self, start: u32, end: u32) -> Option<(u32, u32)> {
325 match self {
326 Spelling::Shift(shift) => Some((shift + start, shift + end)),
327 Spelling::Table(table) => {
328 Some((*table.get(start as usize)?, *table.get(end as usize)?))
329 }
330 }
331 }
332}
333
334/// A file whose spans come from a [`Subspan`] hook rather than from anchors.
335struct PreciseSpans {
336 hook: Subspan,
337 spelling: Spelling,
338}
339
340impl PreciseSpans {
341 /// The span of a file-local byte range, or [`None`] when the hook cannot
342 /// produce one.
343 fn span(&self, start: u32, end: u32, len: u32) -> Option<Span> {
344 let mut start = start.min(len);
345 let mut end = end.clamp(start, len);
346 // A zero-width caret shows nothing at all, so an empty range — the
347 // position of a token that is not there — grows by one byte.
348 if start == end {
349 if end < len {
350 end += 1;
351 } else if start > 0 {
352 start -= 1;
353 } else {
354 return None;
355 }
356 }
357 let (start, end) = self.spelling.range(start, end)?;
358 if end <= start {
359 return None;
360 }
361 self.hook.get(start, end)
362 }
363}
364
365/// One Rust token's footprint inside a captured file.
366#[derive(Clone, Copy)]
367struct Anchor {
368 start: Pos,
369 end: Pos,
370 span: Span,
371}
372
373/// How the C source text of a file was obtained.
374#[derive(Clone, Copy, PartialEq, Eq, Debug)]
375pub enum InputMode {
376 /// The macro input was a single Rust string literal.
377 StringLiteral,
378 /// The text was sliced out of the caller's `.rs` file on disk.
379 FileSlice,
380 /// The text was rebuilt from the token spans' line/column information.
381 Reconstructed,
382 /// The text came from an `#include`d file.
383 Included,
384 /// The text is a `.c` file an `include_c99!("…")` named.
385 ///
386 /// Like an [`Included`](InputMode::Included) file in every way that
387 /// matters to a diagnostic — there is no span inside it, so the caret goes
388 /// on the macro invocation and the message carries `path:line:col` — and
389 /// unlike one in being a translation unit rather than something pulled
390 /// into one.
391 CFile,
392}
393
394/// Everything [`SourceMap::add_file`] needs to know about a new file.
395struct FileSpec {
396 name: String,
397 text: String,
398 /// `(start, end, span)` triples in *file-local* byte offsets, sorted and
399 /// non-overlapping.
400 anchors: AnchorList,
401 /// Span used for positions that no anchor covers.
402 fallback_span: Span,
403 precise: bool,
404 mode: InputMode,
405 rust_path: Option<String>,
406 first_line: usize,
407 /// Set in string-literal mode when the caller supplied a [`Subspan`].
408 precise_spans: Option<PreciseSpans>,
409}
410
411/// A single file of C source together with its Rust-token anchors.
412pub struct SourceFile {
413 id: FileId,
414 name: String,
415 base: Pos,
416 text: String,
417 /// Byte offsets (file-local) at which each line starts.
418 line_starts: Vec<u32>,
419 /// Sorted, non-overlapping anchors in global coordinates.
420 anchors: Vec<Anchor>,
421 /// Span used for positions that no anchor covers.
422 fallback_span: Span,
423 /// Whether spans returned for this file actually point at the position
424 /// asked for (raw-token mode) or only at the file as a whole
425 /// (string-literal mode).
426 precise: bool,
427 /// Resolves a position inside a string literal exactly; see [`Subspan`].
428 precise_spans: Option<PreciseSpans>,
429 mode: InputMode,
430 /// The `.rs` file the invocation is written in, when the compiler knows
431 /// it. This is what `__FILE__` expands to.
432 rust_path: Option<String>,
433 /// The 1-based line of that `.rs` file that this file's own line 1 sits
434 /// on. This is what makes `__LINE__` a line number the user can find.
435 first_line: usize,
436}
437
438impl SourceFile {
439 /// This file's id.
440 pub fn id(&self) -> FileId {
441 self.id
442 }
443
444 /// A human readable name, used in diagnostics.
445 pub fn name(&self) -> &str {
446 &self.name
447 }
448
449 /// The global offset at which this file's text starts.
450 pub fn base(&self) -> Pos {
451 self.base
452 }
453
454 /// The recovered C source text.
455 pub fn text(&self) -> &str {
456 &self.text
457 }
458
459 /// The range this file occupies in the global offset space.
460 pub fn range(&self) -> SourceRange {
461 SourceRange::new(self.base, self.base + self.text.len() as Pos)
462 }
463
464 /// Whether diagnostics in this file can point at an exact position.
465 pub fn precise(&self) -> bool {
466 self.precise
467 }
468
469 /// How this file's text was captured.
470 pub fn mode(&self) -> InputMode {
471 self.mode
472 }
473
474 /// Number of lines in the file (at least 1).
475 pub fn line_count(&self) -> usize {
476 self.line_starts.len()
477 }
478
479 /// The `.rs` file this text was written in, when the compiler knows it.
480 ///
481 /// This is what the preprocessor's `__FILE__` expands to; it is `None`
482 /// outside a real macro expansion (a unit test building a `TokenStream`
483 /// from a string) and for `#include`d files, which name themselves.
484 pub fn rust_path(&self) -> Option<&str> {
485 self.rust_path.as_deref()
486 }
487
488 /// The 1-based line of the enclosing `.rs` file that this file's line 1
489 /// sits on; 1 when nothing better is known.
490 ///
491 /// The preprocessor adds this to a position's own line to make `__LINE__`
492 /// a number that matches what an editor shows.
493 pub fn first_line(&self) -> usize {
494 self.first_line
495 }
496
497 /// Whether this file is one whose diagnostics name it themselves: an
498 /// `#include`d header, or the `.c` file of an `include_c99!`.
499 ///
500 /// Neither has a span of its own — the caret lands on the `#include` or on
501 /// the macro invocation — so `file:line:column` goes into the message
502 /// instead; see [`SourceMap::header_position`].
503 pub fn is_included(&self) -> bool {
504 matches!(self.mode, InputMode::Included | InputMode::CFile)
505 }
506
507 fn local_line_col(&self, local: u32) -> (usize, usize) {
508 let local = local.min(self.text.len() as u32);
509 let line = match self.line_starts.binary_search(&local) {
510 Ok(i) => i,
511 // `line_starts` always begins with 0, so `i` is at least 1 here.
512 Err(i) => i.saturating_sub(1),
513 };
514 let line_start = self.line_starts[line] as usize;
515 // Columns are counted in characters, the way an editor shows them.
516 let col = self
517 .text
518 .get(line_start..local as usize)
519 .map_or(0, |s| s.chars().count())
520 + 1;
521 (line + 1, col)
522 }
523}
524
525/// Maps byte offsets back to `proc_macro2` spans.
526///
527/// See the [module documentation](self) for the coordinate scheme.
528pub struct SourceMap {
529 files: Vec<SourceFile>,
530 next_base: Pos,
531}
532
533impl Default for SourceMap {
534 fn default() -> Self {
535 Self::new()
536 }
537}
538
539impl SourceMap {
540 /// Creates an empty map.
541 pub fn new() -> Self {
542 Self {
543 files: Vec::new(),
544 next_base: 0,
545 }
546 }
547
548 /// Adds a file whose positions cannot be resolved better than
549 /// `spec.fallback_span`.
550 fn add_file(&mut self, spec: FileSpec) -> FileId {
551 let id = FileId(self.files.len() as u32);
552 let base = self.next_base;
553 let text = spec.text;
554 let mut line_starts = vec![0u32];
555 for (i, b) in text.bytes().enumerate() {
556 if b == b'\n' {
557 line_starts.push(i as u32 + 1);
558 }
559 }
560 let anchors = spec
561 .anchors
562 .into_iter()
563 .map(|(s, e, span)| Anchor {
564 start: base + s,
565 end: base + e,
566 span,
567 })
568 .collect();
569 // Leave a one-byte gap so that the end of one file and the start of
570 // the next are never the same position.
571 self.next_base = base.saturating_add(text.len() as Pos).saturating_add(1);
572 self.files.push(SourceFile {
573 id,
574 name: spec.name,
575 base,
576 text,
577 line_starts,
578 anchors,
579 fallback_span: spec.fallback_span,
580 precise: spec.precise,
581 precise_spans: spec.precise_spans,
582 mode: spec.mode,
583 rust_path: spec.rust_path,
584 first_line: spec.first_line.max(1),
585 });
586 id
587 }
588
589 /// The offset the next file added to this map will start at.
590 ///
591 /// The preprocessor runs on a thread of its own, where a
592 /// `proc_macro2::Span` — and therefore this map — cannot follow it, so it
593 /// allocates the offsets of the files it opens itself and hands them back
594 /// for [`SourceMap::add_included_file`] afterwards. This is what lets the
595 /// two agree; see [`crate::analyze`].
596 pub fn next_base(&self) -> Pos {
597 self.next_base
598 }
599
600 /// Adds an `#include`d file to the map.
601 ///
602 /// The preprocessor uses this to bring headers into the same offset space
603 /// as the macro body: positions inside the returned file resolve to
604 /// `directive_span` — the span of the `#include` that pulled it in — and,
605 /// since the file is not [`SourceFile::precise`], diagnostics inside it
606 /// carry their own file, line and column in the message.
607 pub fn add_included_file(
608 &mut self,
609 name: impl Into<String>,
610 text: String,
611 directive_span: Span,
612 ) -> FileId {
613 self.add_file(FileSpec {
614 name: name.into(),
615 text,
616 anchors: Vec::new(),
617 fallback_span: directive_span,
618 precise: false,
619 mode: InputMode::Included,
620 rust_path: None,
621 first_line: 1,
622 precise_spans: None,
623 })
624 }
625
626 /// All files, in insertion order.
627 pub fn files(&self) -> &[SourceFile] {
628 &self.files
629 }
630
631 /// Looks a file up by id.
632 ///
633 /// # Panics
634 ///
635 /// Panics if `id` did not come from this map.
636 pub fn file(&self, id: FileId) -> &SourceFile {
637 &self.files[id.index()]
638 }
639
640 /// Returns the file that contains `pos` (the nearest one if `pos` falls in
641 /// the gap between two files).
642 pub fn file_of(&self, pos: Pos) -> FileId {
643 debug_assert!(!self.files.is_empty(), "source map has no files");
644 let idx = self
645 .files
646 .partition_point(|f| f.base <= pos)
647 .saturating_sub(1);
648 FileId(idx as u32)
649 }
650
651 /// Whether diagnostics anywhere in the file holding `pos` can point at an
652 /// exact location.
653 ///
654 /// String-literal mode with a [`Subspan`] hook is precise range by range
655 /// rather than file by file, so a diagnostic asks [`SourceMap::is_precise_at`]
656 /// instead.
657 pub fn is_precise(&self, pos: Pos) -> bool {
658 self.file(self.file_of(pos)).precise
659 }
660
661 /// Whether the span returned for `range` points at `range` itself.
662 pub fn is_precise_at(&self, range: SourceRange) -> bool {
663 self.resolve(range).1
664 }
665
666 /// The 1-based line and (character-counted) column of `pos` inside its
667 /// file.
668 pub fn line_col(&self, pos: Pos) -> (usize, usize) {
669 let file = self.file(self.file_of(pos));
670 file.local_line_col(pos.saturating_sub(file.base))
671 }
672
673 /// `(name, line, column)` when `pos` is inside an `#include`d file.
674 ///
675 /// A diagnostic there is reported *at* the `#include` directive — that is
676 /// the only place a procedural macro can point at — so the position inside
677 /// the header travels in the message text instead, as
678 /// `header.h:12:5: message`.
679 pub fn header_position(&self, pos: Pos) -> Option<(&str, usize, usize)> {
680 let file = self.file(self.file_of(pos));
681 if !file.is_included() {
682 return None;
683 }
684 let (line, column) = file.local_line_col(pos.saturating_sub(file.base));
685 Some((&file.name, line, column))
686 }
687
688 /// The line number `pos` is named by in a diagnostic note, counted the way
689 /// `__LINE__` counts: a line of the enclosing `.rs` file for the macro's
690 /// own text, and a line of the header itself for an `#include`d file.
691 pub fn source_line(&self, pos: Pos) -> usize {
692 let file = self.file(self.file_of(pos));
693 let (line, _) = file.local_line_col(pos.saturating_sub(file.base));
694 file.first_line + line - 1
695 }
696
697 /// The Rust span to blame for `range`.
698 ///
699 /// See [`SourceMap::resolve`] for how it is found.
700 pub fn span(&self, range: SourceRange) -> Span {
701 self.resolve(range).0
702 }
703
704 /// The Rust span to blame for `range`, and whether it points at `range`
705 /// itself rather than at something enclosing it.
706 ///
707 /// A file whose caller supplied a [`Subspan`] hook asks it first: that is
708 /// string-literal mode under the `nightly` feature, where the span can
709 /// point at the exact bytes inside the literal.
710 ///
711 /// Otherwise the resolution order is: the anchor containing `range.start`,
712 /// then the first anchor overlapping `range`, then the anchor nearest to
713 /// `range.start`, then the file's fallback span.
714 /// `proc_macro::Span::join` is unstable on every channel, so a range
715 /// spanning several tokens resolves to its first token rather than to the
716 /// whole range.
717 pub fn resolve(&self, range: SourceRange) -> (Span, bool) {
718 if self.files.is_empty() {
719 return (Span::call_site(), false);
720 }
721 let file = self.file(self.file_of(range.start));
722 if let Some(precise) = &file.precise_spans
723 && let Some(span) = precise.span(
724 range.start.saturating_sub(file.base),
725 range.end.saturating_sub(file.base),
726 file.text.len() as Pos,
727 )
728 {
729 return (span, true);
730 }
731 (self.anchored_span(file, range), file.precise)
732 }
733
734 /// The span the anchors of `file` give `range`.
735 fn anchored_span(&self, file: &SourceFile, range: SourceRange) -> Span {
736 if file.anchors.is_empty() {
737 return file.fallback_span;
738 }
739 let idx = file.anchors.partition_point(|a| a.start <= range.start);
740 if idx > 0 {
741 let a = &file.anchors[idx - 1];
742 if a.end > range.start {
743 return a.span;
744 }
745 }
746 let before = idx.checked_sub(1).map(|i| &file.anchors[i]);
747 let after = file.anchors.get(idx);
748 // An anchor that starts inside the range still describes it well.
749 if let Some(a) = after
750 && a.start < range.end
751 {
752 return a.span;
753 }
754 match (before, after) {
755 (Some(b), Some(a)) => {
756 if range.start - b.end <= a.start - range.start {
757 b.span
758 } else {
759 a.span
760 }
761 }
762 (Some(b), None) => b.span,
763 (None, Some(a)) => a.span,
764 (None, None) => file.fallback_span,
765 }
766 }
767}
768
769/// The captured macro input: a [`SourceMap`] plus the id of the root file.
770pub struct Source {
771 /// All source files seen so far.
772 pub map: SourceMap,
773 /// The file holding the macro's own C text.
774 pub root: FileId,
775 /// How the root file was captured.
776 pub mode: InputMode,
777 /// A hash identifying this invocation; see [`Source::unit_id`].
778 unit_id: u64,
779}
780
781impl Source {
782 /// A number that identifies this macro invocation.
783 ///
784 /// Two `c99!` blocks in one Rust module generate items into the same
785 /// namespace, so every *synthetic* name — the module holding the `extern`
786 /// block, the mangling of a function-local `static`, the name given to an
787 /// anonymous `struct` — has to carry something that tells the two apart.
788 /// This is that something: a hash of where the invocation is (its file,
789 /// line and column) together with its text, which makes it deterministic
790 /// across compilations of the same code, unique between invocations, and
791 /// stable enough for a snapshot test to record.
792 pub fn unit_id(&self) -> u64 {
793 self.unit_id
794 }
795}
796
797impl Source {
798 /// The root file's C text.
799 pub fn text(&self) -> &str {
800 self.map.file(self.root).text()
801 }
802
803 /// The global offset of the root file's first byte.
804 pub fn base(&self) -> Pos {
805 self.map.file(self.root).base()
806 }
807
808 /// The range covering the whole root file.
809 pub fn root_range(&self) -> SourceRange {
810 self.map.file(self.root).range()
811 }
812}
813
814/// Recovers the C source text of a macro invocation.
815///
816/// Never fails outright: a malformed string literal produces an empty source
817/// file plus a diagnostic, so that the rest of the pipeline can run normally.
818pub fn capture(input: TokenStream, diags: &mut Diagnostics) -> Source {
819 capture_with(input, diags, &Origin::unknown())
820}
821
822/// Recovers the C source text of a macro invocation, with everything the host
823/// can say about the invocation itself.
824///
825/// [`Origin`] is what the two strategies that need more than the tokens are
826/// given: the [`Subspan`] hook that lets a diagnostic land inside a string
827/// literal, and the crate directory the search walks when the
828/// host reports no positions at all. [`capture`] is this with an origin that
829/// says nothing.
830pub fn capture_with(input: TokenStream, diags: &mut Diagnostics, origin: &Origin) -> Source {
831 let trees: Vec<TokenTree> = input.into_iter().collect();
832
833 // --- string-literal mode -------------------------------------------------
834 if trees.len() == 1
835 && let TokenTree::Literal(lit) = &trees[0]
836 {
837 let repr = lit.to_string();
838 if is_string_literal(&repr) {
839 let span = lit.span();
840 // The hook measures offsets in the literal's spelling, so it can
841 // only be trusted if the spelling we decoded is the source itself.
842 let hook = origin
843 .subspan
844 .clone()
845 .filter(|_| span.source_text().is_none_or(|text| text == repr));
846 let want_spelling = hook.is_some();
847 let (text, spelling, error) = match decode_string_literal(&repr, want_spelling) {
848 Ok((text, spelling)) => (text, spelling, None),
849 Err(msg) => (String::new(), None, Some(msg)),
850 };
851 let precise_spans = match (hook, spelling) {
852 (Some(hook), Some(spelling)) => Some(PreciseSpans { hook, spelling }),
853 _ => None,
854 };
855 // The text is complete either way; what the `.rs` file adds is its
856 // own name — which `__FILE__` expands to and a quoted `#include`
857 // searches beside — and the line the literal starts on. The
858 // compiler normally says both; where it says nothing, the
859 // invocation is looked for in the crate's sources like any other.
860 let mut written_in = rust_path_of(span).map(|path| (path, span.start()));
861 if written_in.is_none() {
862 let mut toks = Vec::new();
863 flatten(vec![trees[0].clone()], &mut toks);
864 written_in = origin.search(&toks, |_| true).map(|slice| {
865 (
866 slice.path.display().to_string(),
867 LineColumn {
868 line: slice.line,
869 column: slice.column,
870 },
871 )
872 });
873 }
874 let (rust_path, at) = match written_in {
875 Some((path, at)) => (Some(path), at),
876 None => (None, span.start()),
877 };
878 let mut map = SourceMap::new();
879 let root = map.add_file(FileSpec {
880 name: "<c99! string literal>".to_owned(),
881 text,
882 anchors: Vec::new(),
883 fallback_span: span,
884 precise: false,
885 precise_spans,
886 mode: InputMode::StringLiteral,
887 rust_path: rust_path.clone(),
888 // The C text starts just after the literal's opening quote, so
889 // its first line is the line the literal starts on.
890 first_line: at.line,
891 });
892 let unit_id = unit_id_of(rust_path.as_deref(), at, map.file(root).text());
893 let source = Source {
894 map,
895 root,
896 mode: InputMode::StringLiteral,
897 unit_id,
898 };
899 if let Some(msg) = error {
900 diags.error(SourceRange::at(source.base()), msg);
901 }
902 return source;
903 }
904 }
905
906 // --- raw-token mode ------------------------------------------------------
907 let mut toks = Vec::new();
908 flatten(trees, &mut toks);
909 let fallback_span = toks.first().map_or_else(Span::call_site, |t| t.span);
910 // Decided once for the whole stream rather than token by token, because a
911 // host that gives no positions gives none to any token; see
912 // [`positions_are_usable`].
913 let positioned = positions_are_usable(&toks);
914
915 // The `.rs` file this text is written in: sliced at the positions the
916 // compiler gave, or — where it gave none at all — found by searching the
917 // crate's sources for an invocation these very tokens spell out. The search
918 // is the fallback of a fallback: it cannot run while there is a position to
919 // be had, so a normal build never reaches it.
920 let located = capture_file_slice(&toks).or_else(|| {
921 if positioned {
922 return None;
923 }
924 origin.search(&toks, |_| true).map(Located::from)
925 });
926
927 if let Some(located) = located {
928 let name = located.path.display().to_string();
929 let at = LineColumn {
930 line: located.line,
931 column: located.column,
932 };
933 let mut map = SourceMap::new();
934 let root = map.add_file(FileSpec {
935 rust_path: Some(name.clone()),
936 name,
937 text: located.text,
938 anchors: located.anchors,
939 fallback_span,
940 precise: true,
941 precise_spans: None,
942 mode: InputMode::FileSlice,
943 first_line: located.line,
944 });
945 let unit_id = unit_id_of(
946 Some(map.file(root).rust_path().expect("just set")),
947 at,
948 map.file(root).text(),
949 );
950 return Source {
951 map,
952 root,
953 mode: InputMode::FileSlice,
954 unit_id,
955 };
956 }
957
958 let rebuilt = if positioned {
959 reconstruct(&toks)
960 } else {
961 reconstruct_from_tokens(&toks, origin.entry_name())
962 };
963 // A directive whose end the tokens do not give away stops the rebuild: the
964 // caret goes on its `#`, which is the one position the host does resolve.
965 let fallback_span = rebuilt
966 .blocked
967 .as_ref()
968 .map_or(fallback_span, |blocked| blocked.span);
969 let first_line = toks.first().map_or(1, |t| t.span.start().line);
970 let mut map = SourceMap::new();
971 let root = map.add_file(FileSpec {
972 name: "<c99! macro input>".to_owned(),
973 text: rebuilt.text,
974 anchors: rebuilt.anchors,
975 fallback_span,
976 precise: true,
977 precise_spans: None,
978 mode: InputMode::Reconstructed,
979 rust_path: rust_path_of(fallback_span),
980 first_line,
981 });
982 let unit_id = unit_id_of(
983 map.file(root).rust_path(),
984 fallback_span.start(),
985 map.file(root).text(),
986 );
987 let source = Source {
988 map,
989 root,
990 mode: InputMode::Reconstructed,
991 unit_id,
992 };
993 if let Some(blocked) = rebuilt.blocked {
994 let mut diag = Diagnostic::error(SourceRange::at(source.base()), blocked.message);
995 for note in blocked.notes {
996 diag = diag.with_note(note);
997 }
998 diags.push(diag);
999 }
1000 source
1001}
1002
1003/// The directory of the `.rs` file an invocation whose whole input is `input` is
1004/// written in, found by searching the crate's sources.
1005///
1006/// This is what `include_c99!("…")` needs where the compiler gives no position:
1007/// a relative path is resolved against the directory of the `.rs` file the macro
1008/// is written in, and that directory has to come from somewhere. `accept` is
1009/// asked about each candidate directory, so that of several invocations spelled
1010/// the same way the one whose directory really holds the file wins; [`None`] when
1011/// nothing can be proved, and the caller falls back to `CARGO_MANIFEST_DIR`.
1012pub fn invocation_directory(
1013 input: &TokenTree,
1014 origin: &Origin,
1015 mut accept: impl FnMut(&Path) -> bool,
1016) -> Option<PathBuf> {
1017 let mut toks = Vec::new();
1018 flatten(vec![input.clone()], &mut toks);
1019 let slice = origin.search(&toks, |path| path.parent().is_some_and(&mut accept))?;
1020 slice.path.parent().map(Path::to_path_buf)
1021}
1022
1023/// Makes a `.c` file read from disk the source of a translation unit.
1024///
1025/// This is what [`include_c99!`](crate::expand_include) captures instead of a
1026/// token stream: there is no C in the `.rs` file at all, so there is nothing to
1027/// slice and nothing to point a span into. The file becomes the map's root,
1028/// named by its own path, and every position in it resolves to `span` — the
1029/// macro invocation, which is the only place in the `.rs` file a caret can go.
1030/// Being [`InputMode::CFile`] is what puts `file.c:12:5: ` in front of the
1031/// message, exactly as a header's position travels there.
1032///
1033/// `name` is how diagnostics write the path — relative to the working
1034/// directory wherever it can be, since an absolute one differs between two
1035/// machines — and is also what `__FILE__` expands to and what the file's own
1036/// `#include "…"` searches beside.
1037pub fn capture_c_file(name: String, text: String, span: Span) -> Source {
1038 let mut map = SourceMap::new();
1039 let unit_id = unit_id_of(rust_path_of(span).as_deref(), span.start(), &text);
1040 let root = map.add_file(FileSpec {
1041 rust_path: Some(name.clone()),
1042 name,
1043 text,
1044 anchors: Vec::new(),
1045 fallback_span: span,
1046 precise: false,
1047 precise_spans: None,
1048 mode: InputMode::CFile,
1049 // Its own lines: `__LINE__` in a `.c` file names a line of that file.
1050 first_line: 1,
1051 });
1052 Source {
1053 map,
1054 root,
1055 mode: InputMode::CFile,
1056 unit_id,
1057 }
1058}
1059
1060// ---------------------------------------------------------------------------
1061// unit identity
1062// ---------------------------------------------------------------------------
1063
1064/// The path of the `.rs` file `span` points into, when the compiler knows it.
1065fn rust_path_of(span: Span) -> Option<String> {
1066 span.local_file().map(|p| p.display().to_string())
1067}
1068
1069/// Hashes where an invocation is and what it says into a number that
1070/// distinguishes it from every other invocation in the crate.
1071///
1072/// The position alone would do in a real expansion, but the file is not always
1073/// known (macro-generated input, some IDE contexts, unit tests that build a
1074/// `TokenStream` from a string), and a process-wide counter would make the
1075/// generated names depend on compilation order — which would in turn make
1076/// snapshot tests and incremental rebuilds unstable. Hashing the text as well
1077/// keeps the result deterministic in every context.
1078///
1079/// `path` and `at` are where the C text was written, whether the compiler said so
1080/// or the search worked it out — and the two have to agree, or the names an IDE's
1081/// expansion generates would differ from the names the build generates for the
1082/// same code.
1083fn unit_id_of(path: Option<&str>, at: LineColumn, text: &str) -> u64 {
1084 let mut hash = FNV_OFFSET;
1085 if let Some(path) = path {
1086 hash = fnv(hash, path.as_bytes());
1087 }
1088 hash = fnv(hash, &(at.line as u64).to_le_bytes());
1089 hash = fnv(hash, &(at.column as u64).to_le_bytes());
1090 fnv(hash, text.as_bytes())
1091}
1092
1093const FNV_OFFSET: u64 = 0xcbf2_9ce4_8422_2325;
1094
1095/// FNV-1a, which is short enough to keep here and has no business being a
1096/// dependency.
1097fn fnv(mut hash: u64, bytes: &[u8]) -> u64 {
1098 for b in bytes {
1099 hash ^= u64::from(*b);
1100 hash = hash.wrapping_mul(0x0000_0100_0000_01b3);
1101 }
1102 hash
1103}
1104
1105// ---------------------------------------------------------------------------
1106// string literals
1107// ---------------------------------------------------------------------------
1108
1109fn is_string_literal(repr: &str) -> bool {
1110 repr.starts_with('"') || repr.starts_with("r\"") || repr.starts_with("r#")
1111}
1112
1113/// The text a `Literal` denotes, when it is a plain or raw string literal.
1114///
1115/// [`crate::expand_include`] is what needs it: the argument of
1116/// `include_c99!("…")` is a path, and a path with a `\` in it has to be the
1117/// one the user wrote rather than the escape sequence it is spelled as.
1118/// Anything that is not a string literal — a byte string, a number, an
1119/// identifier — answers [`None`].
1120pub fn string_literal_value(literal: &proc_macro2::Literal) -> Option<String> {
1121 string_literal_text(&literal.to_string())
1122}
1123
1124/// The text a string literal's *spelling* denotes, for a caller that has the
1125/// spelling rather than the token.
1126///
1127/// The [search](self) is what needs it: a `#include "point.h"` in the input is a
1128/// `#`, an identifier and a literal, and the header's name is what that literal
1129/// says — not how it is written.
1130pub(crate) fn string_literal_text(repr: &str) -> Option<String> {
1131 if !is_string_literal(repr) {
1132 return None;
1133 }
1134 decode_string_literal(repr, false)
1135 .ok()
1136 .map(|(text, _)| text)
1137}
1138
1139/// Unescapes a Rust string literal (plain or raw) into the text it denotes.
1140///
1141/// With `spelling` set, it also records where every byte of the result was
1142/// written, which is what a [`Subspan`] hook needs; see [`Spelling`].
1143fn decode_string_literal(repr: &str, spelling: bool) -> Result<(String, Option<Spelling>), String> {
1144 const BAD: &str = "cannot decode this string literal; expected a plain or raw string literal";
1145
1146 if let Some(rest) = repr.strip_prefix('r') {
1147 let hashes = rest.bytes().take_while(|b| *b == b'#').count();
1148 let body = &rest[hashes..];
1149 if !body.starts_with('"') {
1150 return Err(BAD.to_owned());
1151 }
1152 let inner = &body[1..];
1153 let mut closing = String::with_capacity(1 + hashes);
1154 closing.push('"');
1155 closing.extend(std::iter::repeat_n('#', hashes));
1156 let text = inner
1157 .strip_suffix(&closing)
1158 .ok_or_else(|| BAD.to_owned())?
1159 .to_owned();
1160 // Nothing between the quotes is decoded, so the whole text sits at a
1161 // constant distance from the start of `r#…"`.
1162 let prefix = (repr.len() - inner.len()) as u32;
1163 return Ok((text, spelling.then_some(Spelling::Shift(prefix))));
1164 }
1165
1166 let inner = repr
1167 .strip_prefix('"')
1168 .and_then(|s| s.strip_suffix('"'))
1169 .ok_or_else(|| BAD.to_owned())?;
1170
1171 let mut out = String::with_capacity(inner.len());
1172 let mut table: Vec<u32> = Vec::new();
1173 // Every char of `out`, tagged with where in `repr` it was written.
1174 let mut push = |out: &mut String, c: char, at: usize| {
1175 if spelling {
1176 let mut buf = [0u8; 4];
1177 let encoded = c.encode_utf8(&mut buf).len();
1178 table.extend(std::iter::repeat_n(at as u32, encoded));
1179 }
1180 out.push(c);
1181 };
1182 // The opening quote is one byte, so an offset into `inner` is one less
1183 // than the same offset into `repr`.
1184 let mut chars = inner.char_indices().peekable();
1185 while let Some((index, c)) = chars.next() {
1186 let at = index + 1;
1187 if c != '\\' {
1188 push(&mut out, c, at);
1189 continue;
1190 }
1191 let Some((_, e)) = chars.next() else {
1192 return Err(BAD.to_owned());
1193 };
1194 match e {
1195 'n' => push(&mut out, '\n', at),
1196 'r' => push(&mut out, '\r', at),
1197 't' => push(&mut out, '\t', at),
1198 '0' => push(&mut out, '\0', at),
1199 '\\' => push(&mut out, '\\', at),
1200 '\'' => push(&mut out, '\'', at),
1201 '"' => push(&mut out, '"', at),
1202 'x' => {
1203 let mut v = 0u32;
1204 for _ in 0..2 {
1205 let Some(d) = chars.next().and_then(|(_, c)| c.to_digit(16)) else {
1206 return Err(BAD.to_owned());
1207 };
1208 v = v * 16 + d;
1209 }
1210 match char::from_u32(v) {
1211 Some(c) => push(&mut out, c, at),
1212 None => return Err(BAD.to_owned()),
1213 }
1214 }
1215 'u' => {
1216 if chars.next().map(|(_, c)| c) != Some('{') {
1217 return Err(BAD.to_owned());
1218 }
1219 let mut v = 0u32;
1220 loop {
1221 match chars.next().map(|(_, c)| c) {
1222 Some('}') => break,
1223 Some('_') => continue,
1224 Some(c) => match c.to_digit(16) {
1225 Some(d) => v = v.saturating_mul(16).saturating_add(d),
1226 None => return Err(BAD.to_owned()),
1227 },
1228 None => return Err(BAD.to_owned()),
1229 }
1230 }
1231 match char::from_u32(v) {
1232 Some(c) => push(&mut out, c, at),
1233 None => return Err(BAD.to_owned()),
1234 }
1235 }
1236 '\n' => {
1237 // Rust's string continuation: skip leading whitespace. It
1238 // produces nothing, so it takes no place in the table either.
1239 while matches!(chars.peek(), Some((_, ' ' | '\t' | '\n' | '\r'))) {
1240 chars.next();
1241 }
1242 }
1243 _ => return Err(BAD.to_owned()),
1244 }
1245 }
1246 let spelling = spelling.then(|| {
1247 // One past the last byte: the closing quote, which is where a range
1248 // ending at the end of the text stops.
1249 table.push((repr.len() - 1) as u32);
1250 Spelling::Table(table)
1251 });
1252 Ok((out, spelling))
1253}
1254
1255// ---------------------------------------------------------------------------
1256// raw tokens
1257// ---------------------------------------------------------------------------
1258
1259#[derive(Clone, Copy, PartialEq, Eq)]
1260pub(crate) enum FlatKind {
1261 Ident,
1262 Literal,
1263 Punct,
1264 Delimiter,
1265}
1266
1267/// One token of the input, flattened out of its groups.
1268pub(crate) struct FlatTok {
1269 span: Span,
1270 text: String,
1271 /// `true` when `text` came from [`Span::source_text`] and is therefore the
1272 /// verbatim source spelling of the whole span.
1273 exact: bool,
1274 kind: FlatKind,
1275 /// Whether the token after this one was written immediately after it, with
1276 /// nothing at all in between.
1277 ///
1278 /// Only a `Punct` knows — that is `Spacing::Joint`, and it is how one C
1279 /// operator made of several of them is told from several operators: `-` `>`
1280 /// jointly is `->`, and `-` `-` apart is two unary minuses. A host that
1281 /// reports no positions still reports this, which is what makes a text
1282 /// rebuilt from the tokens alone come out right.
1283 joint: bool,
1284}
1285
1286impl FlatTok {
1287 fn new(span: Span, fallback: String, kind: FlatKind, joint: bool) -> Self {
1288 match span.source_text() {
1289 Some(text) => FlatTok {
1290 span,
1291 text,
1292 exact: true,
1293 kind,
1294 joint,
1295 },
1296 None => FlatTok {
1297 span,
1298 text: fallback,
1299 exact: false,
1300 kind,
1301 joint,
1302 },
1303 }
1304 }
1305
1306 /// Where this token was written, as far as the host will say.
1307 pub(crate) fn span(&self) -> Span {
1308 self.span
1309 }
1310
1311 /// The token's spelling: its source text where the host gives one, and
1312 /// otherwise what the token itself says it is — a single character for a
1313 /// `Punct`, the bracket for a delimiter, `Literal::to_string` for a literal.
1314 pub(crate) fn text(&self) -> &str {
1315 &self.text
1316 }
1317
1318 /// Which kind of token this is.
1319 pub(crate) fn kind(&self) -> FlatKind {
1320 self.kind
1321 }
1322}
1323
1324/// Whether `tok` was already written out by `prev`.
1325///
1326/// Several `Punct`s may be handed over sharing the span — and therefore the
1327/// source text — of the one multi-character operator they make up, in which case
1328/// that text belongs to all of them and must be used once.
1329///
1330/// Only a token whose text *is* that source text can be dropped this way. With
1331/// the per-character fallback text, which is all a host reporting no source text
1332/// gives, every token carries its own single character and dropping one would
1333/// lose it — and under `rust-analyzer`, where every span equals every other,
1334/// dropping on the span alone would lose all but the first token of the unit.
1335pub(crate) fn shares_previous_span(prev: &FlatTok, tok: &FlatTok) -> bool {
1336 tok.exact && at_same_position(prev.span, tok.span)
1337}
1338
1339/// Whether two spans start and end at the same line and column.
1340fn at_same_position(a: Span, b: Span) -> bool {
1341 let (a_start, a_end) = (a.start(), a.end());
1342 let (b_start, b_end) = (b.start(), b.end());
1343 (a_start.line, a_start.column, a_end.line, a_end.column)
1344 == (b_start.line, b_start.column, b_end.line, b_end.column)
1345}
1346
1347/// Whether the host gave these tokens positions that mean anything.
1348///
1349/// Decided once for the whole stream, because a host either gives positions or
1350/// does not: `rust-analyzer` reports line 1, column 0 as both the start and the
1351/// end of every token alike, so a stream in which nothing sits anywhere but
1352/// where its first token does carries no positions at all. A single token is no
1353/// evidence either way, and both strategies write it identically.
1354fn positions_are_usable(toks: &[FlatTok]) -> bool {
1355 let Some(first) = toks.first() else {
1356 return false;
1357 };
1358 toks.iter().any(|t| !at_same_position(first.span, t.span))
1359}
1360
1361fn flatten(trees: Vec<TokenTree>, out: &mut Vec<FlatTok>) {
1362 for tt in trees {
1363 match tt {
1364 TokenTree::Group(g) => {
1365 let (open, close) = match g.delimiter() {
1366 Delimiter::Parenthesis => ("(", ")"),
1367 Delimiter::Brace => ("{", "}"),
1368 Delimiter::Bracket => ("[", "]"),
1369 // Invisible groups have no source text of their own.
1370 Delimiter::None => {
1371 flatten(g.stream().into_iter().collect(), out);
1372 continue;
1373 }
1374 };
1375 out.push(FlatTok::new(
1376 g.span_open(),
1377 open.to_owned(),
1378 FlatKind::Delimiter,
1379 false,
1380 ));
1381 flatten(g.stream().into_iter().collect(), out);
1382 out.push(FlatTok::new(
1383 g.span_close(),
1384 close.to_owned(),
1385 FlatKind::Delimiter,
1386 false,
1387 ));
1388 }
1389 TokenTree::Ident(i) => {
1390 out.push(FlatTok::new(
1391 i.span(),
1392 i.to_string(),
1393 FlatKind::Ident,
1394 false,
1395 ));
1396 }
1397 TokenTree::Punct(p) => {
1398 let joint = p.spacing() == Spacing::Joint;
1399 out.push(FlatTok::new(
1400 p.span(),
1401 p.as_char().to_string(),
1402 FlatKind::Punct,
1403 joint,
1404 ));
1405 }
1406 TokenTree::Literal(l) => {
1407 out.push(FlatTok::new(
1408 l.span(),
1409 l.to_string(),
1410 FlatKind::Literal,
1411 false,
1412 ));
1413 }
1414 }
1415 }
1416}
1417
1418/// The flattened tokens of an input stream, which is what the search in
1419/// `locate.rs` matches a candidate body against.
1420#[cfg(test)]
1421pub(crate) fn flat_tokens(input: TokenStream) -> Vec<FlatTok> {
1422 let mut toks = Vec::new();
1423 flatten(input.into_iter().collect(), &mut toks);
1424 toks
1425}
1426
1427/// Byte offsets of the start of every line of a string.
1428struct LineIndex {
1429 line_starts: Vec<usize>,
1430 len: usize,
1431}
1432
1433impl LineIndex {
1434 fn new(text: &str) -> Self {
1435 let mut line_starts = vec![0usize];
1436 for (i, b) in text.bytes().enumerate() {
1437 if b == b'\n' {
1438 line_starts.push(i + 1);
1439 }
1440 }
1441 Self {
1442 line_starts,
1443 len: text.len(),
1444 }
1445 }
1446
1447 /// Converts a 1-based line / 0-based character column into a byte offset.
1448 ///
1449 /// `proc_macro2` counts columns in `char`s, so this is where non-ASCII
1450 /// source (a Japanese comment, say) would otherwise go wrong.
1451 fn offset(&self, text: &str, lc: LineColumn) -> Option<usize> {
1452 let line = lc.line.checked_sub(1)?;
1453 let start = *self.line_starts.get(line)?;
1454 let end = self.line_starts.get(line + 1).copied().unwrap_or(self.len);
1455 let slice = text.get(start..end)?;
1456 let mut count = 0usize;
1457 for (i, _) in slice.char_indices() {
1458 if count == lc.column {
1459 return Some(start + i);
1460 }
1461 count += 1;
1462 }
1463 (count == lc.column).then_some(end)
1464 }
1465}
1466
1467/// The `.rs` file an invocation is written in, its text, and where in the file
1468/// that text begins.
1469///
1470/// What both strategies that answer with a real file produce: the primary one
1471/// from the positions the compiler gave, the search in `locate.rs` by proving
1472/// which text it is.
1473struct Located {
1474 path: PathBuf,
1475 text: String,
1476 anchors: AnchorList,
1477 /// The 1-based line of the `.rs` file the text starts on.
1478 line: usize,
1479 /// The 0-based column, in characters, the text starts at.
1480 column: usize,
1481}
1482
1483impl From<locate::Slice> for Located {
1484 fn from(slice: locate::Slice) -> Self {
1485 Self {
1486 path: slice.path,
1487 text: slice.text,
1488 anchors: slice.anchors,
1489 line: slice.line,
1490 column: slice.column,
1491 }
1492 }
1493}
1494
1495/// Primary raw-token strategy: slice the caller's `.rs` file.
1496///
1497/// Returns `None` (so that the caller falls back to the search and then to
1498/// rebuilding the text from the tokens) whenever anything at all looks
1499/// inconsistent, because a wrong slice would produce silently wrong code rather
1500/// than a diagnostic.
1501fn capture_file_slice(toks: &[FlatTok]) -> Option<Located> {
1502 let first = toks.first()?;
1503 let last = toks.last()?;
1504 let path = first.span.local_file()?;
1505 let content = std::fs::read_to_string(&path).ok()?;
1506 let index = LineIndex::new(&content);
1507
1508 let start = index.offset(&content, first.span.start())?;
1509 let end = index.offset(&content, last.span.end())?;
1510 if end < start || end - start > u32::MAX as usize {
1511 return None;
1512 }
1513
1514 let mut anchors = Vec::with_capacity(toks.len());
1515 let mut prev_end = start;
1516 for t in toks {
1517 let s = index.offset(&content, t.span.start())?;
1518 let e = index.offset(&content, t.span.end())?;
1519 if s < start || e > end || e < s {
1520 return None;
1521 }
1522 let slice = content.get(s..e)?;
1523 if t.exact {
1524 if slice != t.text {
1525 return None;
1526 }
1527 } else {
1528 match t.kind {
1529 FlatKind::Ident | FlatKind::Literal => {
1530 if slice != t.text {
1531 return None;
1532 }
1533 }
1534 // A multi-character operator may be delivered as several
1535 // `Punct`s sharing one span, so only require containment.
1536 FlatKind::Punct | FlatKind::Delimiter => {
1537 if !slice.contains(&t.text) {
1538 return None;
1539 }
1540 }
1541 }
1542 }
1543 if s < prev_end {
1544 continue;
1545 }
1546 prev_end = e;
1547 anchors.push(((s - start) as u32, (e - start) as u32, t.span));
1548 }
1549
1550 let at = first.span.start();
1551 Some(Located {
1552 path,
1553 text: content[start..end].to_owned(),
1554 anchors,
1555 line: at.line,
1556 column: at.column,
1557 })
1558}
1559
1560/// A text rebuilt from the tokens, with the anchors that map it back to them.
1561struct Rebuilt {
1562 text: String,
1563 anchors: AnchorList,
1564 /// Set when a preprocessing directive stopped the rebuild; see
1565 /// [`Blocked`].
1566 blocked: Option<Blocked>,
1567}
1568
1569/// A directive whose end the tokens alone do not give away, and the one
1570/// diagnostic it earns.
1571///
1572/// `#define X 1` is a *line*, and a token stream with no positions keeps no
1573/// lines: there is no way to tell the `1` that ends the replacement list from
1574/// the `int` that begins the next line. Guessing would silently mistranslate,
1575/// so the unit is refused with a message that says what is missing and how to
1576/// get it, and the text comes out empty so that this is the *only* thing
1577/// reported.
1578struct Blocked {
1579 /// The `#` the message is reported at. Under a host that gives no positions
1580 /// this is still a span it can resolve — that is how it maps a
1581 /// `compile_error!` back to the source — so the caret lands on the
1582 /// directive even though nothing could be read from the span itself.
1583 span: Span,
1584 message: String,
1585 notes: Vec<String>,
1586}
1587
1588/// Fallback raw-token strategy: rebuild the text from token positions.
1589fn reconstruct(toks: &[FlatTok]) -> Rebuilt {
1590 let mut out = String::new();
1591 let mut anchors = Vec::new();
1592 let Some(first) = toks.first() else {
1593 return Rebuilt {
1594 text: out,
1595 anchors,
1596 blocked: None,
1597 };
1598 };
1599 let line_base = first.span.start().line;
1600 let col_base = first.span.start().column;
1601 let mut cur_line = 1usize;
1602 let mut cur_col = 0usize;
1603 let mut prev: Option<&FlatTok> = None;
1604
1605 for t in toks {
1606 let s = t.span.start();
1607 // Several `Punct`s can share the span — and the source text — of one
1608 // multi-character operator (`->`, `==`, ...); write it once.
1609 if let Some(p) = prev
1610 && shares_previous_span(p, t)
1611 {
1612 continue;
1613 }
1614 prev = Some(t);
1615
1616 let target_line = s.line.saturating_sub(line_base) + 1;
1617 let target_col = if s.line == line_base {
1618 s.column.saturating_sub(col_base)
1619 } else {
1620 s.column
1621 };
1622
1623 if target_line < cur_line || (target_line == cur_line && target_col < cur_col) {
1624 // This token was written *before* the one in front of it, which a
1625 // stream assembled from more than one place can be. (A stream with
1626 // no positions at all never gets here: that is
1627 // `reconstruct_from_tokens`.) Keep the tokens; separate them with a
1628 // space so that they do not merge into one C token.
1629 if !out.is_empty() {
1630 out.push(' ');
1631 cur_col += 1;
1632 }
1633 } else {
1634 while cur_line < target_line {
1635 out.push('\n');
1636 cur_line += 1;
1637 cur_col = 0;
1638 }
1639 while cur_col < target_col {
1640 out.push(' ');
1641 cur_col += 1;
1642 }
1643 }
1644
1645 let anchor_start = out.len();
1646 out.push_str(&t.text);
1647 for ch in t.text.chars() {
1648 if ch == '\n' {
1649 cur_line += 1;
1650 cur_col = 0;
1651 } else {
1652 cur_col += 1;
1653 }
1654 }
1655 anchors.push((anchor_start as u32, out.len() as u32, t.span));
1656 }
1657
1658 Rebuilt {
1659 text: out,
1660 anchors,
1661 blocked: None,
1662 }
1663}
1664
1665/// Last-resort raw-token strategy: rebuild the text from the tokens alone.
1666///
1667/// Where the host gives no positions there is nothing to place tokens *at*, and
1668/// the one thing left to get right is which of them were written together. A
1669/// space goes between any two tokens except after a `Punct` the next token
1670/// followed immediately, so `->`, `==`, `<<=`, `&&`, `++` and `...` come out as
1671/// the single C operators they are while `- -` stays two tokens and `a + +b`
1672/// stays an addition of a unary plus. Nothing is ever dropped — only a token
1673/// that carries another's source text may be, and a host with no positions gives
1674/// no source text either.
1675///
1676/// Comments, line breaks and the columns are gone for good, which costs the C
1677/// text nothing that is not a *directive*: those are lines. The forms whose end
1678/// the tokens themselves give away are written on a line of their own, and any
1679/// other — where the end would have to be guessed — stops the rebuild with one
1680/// diagnostic; see [`Blocked`] and [`directive_shape`].
1681fn reconstruct_from_tokens(toks: &[FlatTok], entry: &str) -> Rebuilt {
1682 let mut out = Builder::default();
1683 // Whether the token just written was a `Punct` the next one follows
1684 // immediately, so that the two make up one C operator.
1685 let mut glued = false;
1686 let mut prev: Option<&FlatTok> = None;
1687 let mut i = 0;
1688 while i < toks.len() {
1689 let tok = &toks[i];
1690 if let Some(p) = prev
1691 && shares_previous_span(p, tok)
1692 {
1693 i += 1;
1694 continue;
1695 }
1696 prev = Some(tok);
1697
1698 if !is_hash(tok) {
1699 out.write(tok, !glued);
1700 glued = tok.kind == FlatKind::Punct && tok.joint;
1701 i += 1;
1702 continue;
1703 }
1704
1705 // A directive: its own line, and only if the tokens say where it ends.
1706 let rest = &toks[i..];
1707 let Some(shape) = directive_shape(rest) else {
1708 return Rebuilt {
1709 text: String::new(),
1710 anchors: AnchorList::new(),
1711 blocked: Some(blocked_directive(rest, entry)),
1712 };
1713 };
1714 out.newline();
1715 out.write(&rest[0], false);
1716 for (n, tok) in rest[1..shape.tokens()].iter().enumerate() {
1717 // One space after the directive's name, and none anywhere else: a
1718 // header name is a single token to the preprocessor, so
1719 // `<` `stdio` `.` `h` `>` has to come back out as `<stdio.h>`.
1720 out.write(tok, n == 1);
1721 }
1722 out.newline();
1723 glued = false;
1724 i += shape.tokens();
1725 }
1726 Rebuilt {
1727 text: out.text,
1728 anchors: out.anchors,
1729 blocked: None,
1730 }
1731}
1732
1733/// The text a token-only rebuild is writing, and where each token went.
1734#[derive(Default)]
1735struct Builder {
1736 text: String,
1737 anchors: AnchorList,
1738 /// Whether nothing has been written on the current line yet, in which case
1739 /// no separator may be.
1740 at_line_start: bool,
1741}
1742
1743impl Builder {
1744 /// Writes one token, with a space in front of it when `space` asks for one
1745 /// and it is not starting a line.
1746 fn write(&mut self, tok: &FlatTok, space: bool) {
1747 if space && !self.at_line_start && !self.text.is_empty() {
1748 self.text.push(' ');
1749 }
1750 let start = self.text.len() as u32;
1751 self.text.push_str(&tok.text);
1752 self.anchors.push((start, self.text.len() as u32, tok.span));
1753 self.at_line_start = false;
1754 }
1755
1756 /// Ends the current line, so that what comes next begins one.
1757 ///
1758 /// This is the only way a newline is ever written, because the only thing a
1759 /// line means here is "a directive starts here" — and the preprocessor
1760 /// decides that by the `#` being the first token on its line.
1761 fn newline(&mut self) {
1762 if !self.text.is_empty() && !self.at_line_start {
1763 self.text.push('\n');
1764 }
1765 self.at_line_start = true;
1766 }
1767}
1768
1769/// Whether this token is the `#` that opens a directive.
1770///
1771/// In C a `#` can be nothing else: the two preprocessor *operators* spelled
1772/// with one live in a `#define` replacement list, which is a directive itself.
1773fn is_hash(tok: &FlatTok) -> bool {
1774 tok.kind == FlatKind::Punct && tok.text == "#"
1775}
1776
1777/// A directive whose end is known from the tokens alone.
1778///
1779/// Everything a directive may be followed by is another line, and there are no
1780/// lines here — so a form is usable exactly when its own tokens say where it
1781/// stops. These do; `#define`, `#if`, `#elif`, `#error`, `#warning`, `#line`,
1782/// `#embed`, `#pragma` other than `once` and a `#` followed by anything but a
1783/// directive name do not, and are [`Blocked`].
1784#[derive(Clone, Copy)]
1785enum Shape {
1786 /// A `#` with nothing at all after it: the null directive.
1787 Null,
1788 /// `#` and the directive's name: `#else`, `#endif`.
1789 Bare,
1790 /// `#`, the name and one identifier: `#ifdef X`, `#undef X`, `#pragma once`.
1791 Word,
1792 /// `#include` and a quoted header name, which is one string literal.
1793 Quoted,
1794 /// `#include` and an angled header name, holding `tokens` of path.
1795 Angled(usize),
1796}
1797
1798impl Shape {
1799 /// How many tokens the directive is, `#` included.
1800 fn tokens(self) -> usize {
1801 match self {
1802 Shape::Null => 1,
1803 Shape::Bare => 2,
1804 Shape::Word | Shape::Quoted => 3,
1805 // `#`, `include`, `<`, the path, `>`.
1806 Shape::Angled(tokens) => 4 + tokens,
1807 }
1808 }
1809}
1810
1811/// The shape of the directive `toks` opens with — `toks[0]` is its `#` — or
1812/// [`None`] when its end cannot be known; see [`Shape`].
1813fn directive_shape(toks: &[FlatTok]) -> Option<Shape> {
1814 let Some(name) = toks.get(1) else {
1815 return Some(Shape::Null);
1816 };
1817 if name.kind != FlatKind::Ident {
1818 return None;
1819 }
1820 let operand = toks.get(2);
1821 let is_ident = |tok: Option<&FlatTok>, text: &str| {
1822 tok.is_some_and(|t| t.kind == FlatKind::Ident && (text.is_empty() || t.text == text))
1823 };
1824 match name.text.as_str() {
1825 "include" | "include_next" => {
1826 // A quoted name is one literal; an angled one runs to the `>`. A
1827 // name a macro stands for (`#include HEADER`) does not say where the
1828 // line ends, and neither does a raw string literal — which a C
1829 // preprocessor could not read anyway.
1830 if operand.is_some_and(|t| t.kind == FlatKind::Literal && t.text.starts_with('"')) {
1831 return Some(Shape::Quoted);
1832 }
1833 if operand.is_some_and(|t| t.kind == FlatKind::Punct && t.text == "<") {
1834 let path = toks
1835 .get(3..)?
1836 .iter()
1837 .position(|t| t.kind == FlatKind::Punct && t.text == ">")?;
1838 return Some(Shape::Angled(path));
1839 }
1840 None
1841 }
1842 // `#elifdef` and `#elifndef` are C23's, and take one name like the rest.
1843 "ifdef" | "ifndef" | "undef" | "elifdef" | "elifndef" => {
1844 is_ident(operand, "").then_some(Shape::Word)
1845 }
1846 "else" | "endif" => Some(Shape::Bare),
1847 // `#pragma once` is the only pragma whose operands are a fixed set.
1848 "pragma" => is_ident(operand, "once").then_some(Shape::Word),
1849 _ => None,
1850 }
1851}
1852
1853/// The one diagnostic a directive that cannot be delimited earns.
1854///
1855/// It has three jobs: say what is missing (the positions), say why it is likely
1856/// to be missing (an editor looking at an unsaved file, or input another macro
1857/// built), and say what to do about it — both ways out, since saving the file is
1858/// not always the one the reader wants.
1859fn blocked_directive(toks: &[FlatTok], entry: &str) -> Blocked {
1860 let what = match toks.get(1) {
1861 Some(name) if name.kind == FlatKind::Ident => format!("the '#{}' directive", name.text),
1862 _ => "this preprocessing directive".to_owned(),
1863 };
1864 Blocked {
1865 span: toks[0].span,
1866 message: format!(
1867 "cannot tell where {what} ends: this block's tokens carry no source positions, so its \
1868 text had to be rebuilt from the tokens alone — and a directive is a line, of which \
1869 tokens keep nothing"
1870 ),
1871 notes: vec![
1872 "the compiler that expanded this gave no position for any token, and the invocation \
1873 was not found in this crate's sources either. An editor analysing a file you have \
1874 not saved yet looks exactly like that — rust-analyzer gives a procedural macro no \
1875 positions, and what is on disk no longer matches what you are typing — and so does \
1876 input another macro built."
1877 .to_owned(),
1878 format!(
1879 "save the file and this block is read from disk again, directives and all; or \
1880 write it as a string literal — {entry}! {{ r#\"…\"# }} — which needs no \
1881 positions at all"
1882 ),
1883 ],
1884 }
1885}
1886
1887#[cfg(test)]
1888mod tests {
1889 use super::*;
1890
1891 /// The decoded text of a literal, ignoring the spelling map.
1892 fn unescape(repr: &str) -> Result<String, String> {
1893 decode_string_literal(repr, false).map(|(text, _)| text)
1894 }
1895
1896 /// The spelling range a text range maps back to.
1897 fn spelling_of(repr: &str, start: u32, end: u32) -> Option<(u32, u32)> {
1898 let (_, spelling) = decode_string_literal(repr, true).unwrap();
1899 spelling.unwrap().range(start, end)
1900 }
1901
1902 #[test]
1903 fn unescape_plain() {
1904 assert_eq!(unescape(r#""a\nb""#).unwrap(), "a\nb");
1905 assert_eq!(unescape(r#""\x41\u{3042}""#).unwrap(), "A\u{3042}");
1906 }
1907
1908 #[test]
1909 fn the_spelling_map_finds_the_bytes_a_character_was_written_as() {
1910 // `"a\nb"`: the `b` is the third byte of the text and the fifth of the
1911 // spelling, because `\n` was written as two.
1912 assert_eq!(spelling_of(r#""a\nb""#, 2, 3), Some((4, 5)));
1913 // The escape itself covers both of its bytes.
1914 assert_eq!(spelling_of(r#""a\nb""#, 1, 2), Some((2, 4)));
1915 // A raw literal is a constant shift past `r##"`.
1916 assert_eq!(spelling_of(r####"r##"abc"##"####, 1, 2), Some((5, 6)));
1917 // The end of the text is the closing quote.
1918 assert_eq!(spelling_of(r#""ab""#, 0, 2), Some((1, 3)));
1919 }
1920
1921 #[test]
1922 fn several_files_share_one_offset_space() {
1923 // What `#include` will need: extra files slot into the same
1924 // coordinate system without changing any signature downstream.
1925 let mut map = SourceMap::new();
1926 let root = map.add_file(FileSpec {
1927 name: "<input>".to_owned(),
1928 text: "int x;\n".to_owned(),
1929 anchors: Vec::new(),
1930 fallback_span: Span::call_site(),
1931 precise: true,
1932 precise_spans: None,
1933 mode: InputMode::Reconstructed,
1934 rust_path: None,
1935 first_line: 1,
1936 });
1937 let header = map.add_included_file(
1938 "stdio.h",
1939 "int printf();\nint puts();\n".to_owned(),
1940 Span::call_site(),
1941 );
1942 assert_ne!(root, header);
1943 assert_eq!(map.file_of(map.file(root).base()), root);
1944 assert_eq!(map.file_of(map.file(header).base()), header);
1945 assert_eq!(map.file_of(map.file(header).base() + 14), header);
1946 assert_eq!(map.line_col(map.file(header).base() + 14), (2, 1));
1947 assert!(map.is_precise(map.file(root).base()));
1948 assert!(!map.is_precise(map.file(header).base()));
1949 }
1950
1951 #[test]
1952 fn unescape_raw() {
1953 assert_eq!(unescape(r####"r#"a\nb"#"####).unwrap(), r"a\nb");
1954 assert_eq!(unescape(r#"r"x""#).unwrap(), "x");
1955 }
1956}