Skip to main content

workshop_rs/core/
source.rs

1//! Source model: files, positions, and spans.
2//!
3//! Conventions: positions are 1-based, and a span is a half-open interval
4//! (`end` is exclusive). Spans carry a typed [`FileId`] instead of a raw file
5//! index.
6
7use std::{ops::Range, slice::Iter};
8
9use super::ids::Id;
10
11/// A typed ID referencing a [`SourceFile`] in the program's file arena.
12pub type FileId = Id<SourceFile>;
13
14/// One source file in the program's file registry.
15#[derive(Debug, Clone, PartialEq, Eq)]
16pub struct SourceFile {
17    /// The file name as the frontend reported it (for diagnostics).
18    pub path: String,
19    file: Option<FileId>,
20    source: Option<SourceDocument>,
21}
22
23impl SourceFile {
24    /// Create a file entry with the given path.
25    pub fn new(path: impl Into<String>) -> Self {
26        SourceFile {
27            path: path.into(),
28            file: None,
29            source: None,
30        }
31    }
32
33    /// Create a file entry that retains its authored source text and comments.
34    pub fn with_source(path: impl Into<String>, source: impl Into<String>) -> Self {
35        SourceFile {
36            path: path.into(),
37            file: None,
38            source: Some(SourceDocument::new(source)),
39        }
40    }
41
42    /// Attach authored source to an existing file entry.
43    pub fn set_source(&mut self, source: impl Into<String>) {
44        let mut document = SourceDocument::new(source);
45        document.file = self.file;
46        self.source = Some(document);
47    }
48
49    pub(crate) fn bind_file(&mut self, file: FileId) {
50        self.file = Some(file);
51        if let Some(source) = &mut self.source {
52            source.file = Some(file);
53        }
54    }
55
56    /// The retained authored source, when this file was created source-aware.
57    pub fn source(&self) -> Option<&SourceDocument> {
58        self.source.as_ref()
59    }
60}
61
62/// A source document retained independently from canonical semantic nodes.
63///
64/// Whitespace and other non-comment trivia remain in [`Self::text`]. Comments
65/// are indexed as a convenience for stable span-based attachment; callers
66/// should use [`SourceEdit`] for local changes and reparse the edited text to
67/// obtain updated semantic spans.
68#[derive(Debug, Clone, PartialEq, Eq)]
69pub struct SourceDocument {
70    file: Option<FileId>,
71    text: String,
72    comments: Vec<SourceComment>,
73    line_starts: Vec<usize>,
74}
75
76impl SourceDocument {
77    /// Retain source text and index supported line comments (`// ...`).
78    pub fn new(text: impl Into<String>) -> Self {
79        let text = text.into();
80        let comments = find_line_comments(&text);
81        Self {
82            file: None,
83            line_starts: line_starts(&text),
84            text,
85            comments,
86        }
87    }
88
89    /// The exact authored source text.
90    pub fn text(&self) -> &str {
91        &self.text
92    }
93
94    /// All indexed comments in authored source order.
95    pub fn comments(&self) -> Iter<'_, SourceComment> {
96        self.comments.iter()
97    }
98
99    /// Comments whose complete source range is inside a semantic span.
100    ///
101    /// This is the attachment rule for the supported slice. Comments outside
102    /// a node span remain document-level preserved source and are never
103    /// guessed into a neighboring node.
104    pub fn comments_for(&self, span: Span) -> impl Iterator<Item = &SourceComment> {
105        let range = self.byte_range(span);
106        self.comments.iter().filter(move |comment| {
107            range.as_ref().is_some_and(|range| {
108                comment.range.start >= range.start && comment.range.end <= range.end
109            })
110        })
111    }
112
113    /// Convert a line/column span into a UTF-8 byte range in this document.
114    pub fn byte_range(&self, span: Span) -> Option<Range<usize>> {
115        if self.file != Some(span.file) {
116            return None;
117        }
118        let start = self.byte_offset(span.start)?;
119        let end = self.byte_offset(span.end)?;
120        (start <= end).then_some(start..end)
121    }
122
123    /// Create a checked replacement for a UTF-8 byte range.
124    pub fn edit(
125        &self,
126        range: Range<usize>,
127        replacement: impl Into<String>,
128    ) -> Result<SourceEdit, SourceEditError> {
129        SourceEdit::from_source(&self.text, range, replacement)
130    }
131
132    /// Create a checked replacement for a semantic span.
133    pub fn edit_span(
134        &self,
135        span: Span,
136        replacement: impl Into<String>,
137    ) -> Result<SourceEdit, SourceEditError> {
138        let range = self.byte_range(span).ok_or(SourceEditError::InvalidRange)?;
139        self.edit(range, replacement)
140    }
141
142    /// Apply non-overlapping edits against this exact document.
143    pub fn apply(&self, edits: &[SourceEdit]) -> Result<Self, SourceEditError> {
144        let mut ordered = edits.iter().collect::<Vec<_>>();
145        ordered.sort_by_key(|edit| edit.range.start);
146        for pair in ordered.windows(2) {
147            if pair[0].range.end > pair[1].range.start || pair[0].range.start == pair[1].range.start
148            {
149                return Err(SourceEditError::OverlappingEdits);
150            }
151        }
152        let mut text = self.text.clone();
153        for edit in ordered.into_iter().rev() {
154            edit.apply_to(&mut text)?;
155        }
156        Ok(Self::new(text))
157    }
158}
159
160/// A source comment retained from the authored document.
161#[derive(Debug, Clone, PartialEq, Eq)]
162pub struct SourceComment {
163    kind: CommentKind,
164    range: Range<usize>,
165}
166
167impl SourceComment {
168    pub fn kind(&self) -> CommentKind {
169        self.kind
170    }
171
172    /// The UTF-8 byte range including the `//` marker and excluding its line
173    /// ending.
174    pub fn range(&self) -> Range<usize> {
175        self.range.clone()
176    }
177
178    /// The exact comment text from the containing document.
179    pub fn text<'a>(&self, document: &'a SourceDocument) -> &'a str {
180        &document.text[self.range.clone()]
181    }
182}
183
184/// Comment kinds currently supported by the raw Workshop source contract.
185#[derive(Debug, Clone, Copy, PartialEq, Eq)]
186pub enum CommentKind {
187    Line,
188}
189
190/// A checked, byte-oriented source replacement.
191#[derive(Debug, Clone, PartialEq, Eq)]
192pub struct SourceEdit {
193    range: Range<usize>,
194    expected: String,
195    replacement: String,
196}
197
198impl SourceEdit {
199    /// Create a checked edit from an existing source buffer.
200    pub(crate) fn from_source(
201        source: &str,
202        range: Range<usize>,
203        replacement: impl Into<String>,
204    ) -> Result<Self, SourceEditError> {
205        if range.start > range.end
206            || !source.is_char_boundary(range.start)
207            || !source.is_char_boundary(range.end)
208            || range.end > source.len()
209        {
210            return Err(SourceEditError::InvalidRange);
211        }
212        Ok(Self {
213            expected: source[range.clone()].to_string(),
214            range,
215            replacement: replacement.into(),
216        })
217    }
218
219    pub fn range(&self) -> Range<usize> {
220        self.range.clone()
221    }
222
223    pub fn replacement(&self) -> &str {
224        &self.replacement
225    }
226
227    /// Apply this edit only when the original bytes still match.
228    pub fn apply(&self, source: &str) -> Result<String, SourceEditError> {
229        let mut result = source.to_string();
230        self.apply_to(&mut result)?;
231        Ok(result)
232    }
233
234    fn apply_to(&self, source: &mut String) -> Result<(), SourceEditError> {
235        if source.get(self.range.clone()) != Some(self.expected.as_str()) {
236            return Err(SourceEditError::SourceMismatch);
237        }
238        source.replace_range(self.range.clone(), &self.replacement);
239        Ok(())
240    }
241}
242
243/// Failure while creating or applying source edits.
244#[derive(Debug, Clone, Copy, PartialEq, Eq)]
245#[non_exhaustive]
246pub enum SourceEditError {
247    InvalidRange,
248    SourceMismatch,
249    OverlappingEdits,
250}
251
252impl std::fmt::Display for SourceEditError {
253    fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
254        formatter.write_str(match self {
255            Self::InvalidRange => "source edit range is not a valid UTF-8 range",
256            Self::SourceMismatch => "source no longer matches the edit",
257            Self::OverlappingEdits => "source edits overlap",
258        })
259    }
260}
261
262impl std::error::Error for SourceEditError {}
263
264fn find_line_comments(source: &str) -> Vec<SourceComment> {
265    let mut comments = Vec::new();
266    let mut index = 0;
267    let mut in_string = false;
268    let mut escaped = false;
269    while index < source.len() {
270        let character = source[index..].chars().next().unwrap();
271        if in_string {
272            if escaped {
273                escaped = false;
274            } else if character == '\\' {
275                escaped = true;
276            } else if character == '"' {
277                in_string = false;
278            }
279            index += character.len_utf8();
280            continue;
281        }
282        if character == '"' {
283            in_string = true;
284            index += character.len_utf8();
285        } else if character == '/' && source[index..].starts_with("//") {
286            let start = index;
287            index += 2;
288            while index < source.len()
289                && !source[index..].starts_with('\n')
290                && !source[index..].starts_with('\r')
291            {
292                index += source[index..].chars().next().unwrap().len_utf8();
293            }
294            comments.push(SourceComment {
295                kind: CommentKind::Line,
296                range: start..index,
297            });
298        } else {
299            index += character.len_utf8();
300        }
301    }
302    comments
303}
304
305/// Resolve a 1-based line/column [`Position`] to a UTF-8 byte offset in
306/// `source` by scanning from the start. [`SourceDocument`] callers should
307/// prefer its line-indexed `byte_offset` instead.
308pub(crate) fn byte_offset(source: &str, position: Position) -> Option<usize> {
309    if !position.is_valid() {
310        return None;
311    }
312    let mut line = 1;
313    let mut col = 1;
314    for (index, character) in source.char_indices() {
315        if line == position.line && col == position.col {
316            return Some(index);
317        }
318        if character == '\n' {
319            line += 1;
320            col = 1;
321        } else {
322            col += 1;
323        }
324    }
325    (line == position.line && col == position.col).then_some(source.len())
326}
327
328fn line_starts(source: &str) -> Vec<usize> {
329    let mut starts = Vec::with_capacity(source.len() / 40 + 1);
330    starts.push(0);
331    for (index, _) in source.match_indices('\n') {
332        starts.push(index + 1);
333    }
334    starts
335}
336
337impl SourceDocument {
338    fn byte_offset(&self, position: Position) -> Option<usize> {
339        self.byte_offset_scan(position).0
340    }
341
342    fn byte_offset_scan(&self, position: Position) -> (Option<usize>, usize) {
343        if !position.is_valid() {
344            return (None, 0);
345        }
346        let line_index = position.line as usize - 1;
347        let Some(&start) = self.line_starts.get(line_index) else {
348            return (None, 0);
349        };
350        let end = self
351            .line_starts
352            .get(line_index + 1)
353            .copied()
354            .unwrap_or(self.text.len());
355        let mut scanned = 0;
356        let mut col = 1;
357        for (offset, character) in self.text[start..end].char_indices() {
358            scanned += character.len_utf8();
359            if col == position.col {
360                return (Some(start + offset), scanned);
361            }
362            if character == '\n' {
363                return (None, scanned);
364            }
365            col += 1;
366        }
367        ((col == position.col).then_some(end), scanned)
368    }
369}
370
371/// A 1-based line/column position in a source file.
372#[derive(Debug, Clone, Copy, PartialEq, Eq)]
373pub struct Position {
374    pub line: u32,
375    pub col: u32,
376}
377
378impl Position {
379    /// A position at line `line`, column `col` (both 1-based).
380    pub const fn new(line: u32, col: u32) -> Self {
381        Position { line, col }
382    }
383
384    /// Whether this position is valid (1-based).
385    pub const fn is_valid(self) -> bool {
386        self.line >= 1 && self.col >= 1
387    }
388}
389
390/// A half-open, 1-based source interval in one file.
391#[derive(Debug, Clone, Copy, PartialEq, Eq)]
392pub struct Span {
393    pub file: FileId,
394    pub start: Position,
395    pub end: Position,
396}
397
398impl Span {
399    /// Create a span in `file` from `start` (inclusive) to `end` (exclusive).
400    pub const fn new(file: FileId, start: Position, end: Position) -> Self {
401        Span { file, start, end }
402    }
403
404    /// Whether the span is structurally valid: both positions are 1-based and
405    /// `end` is not before `start`.
406    pub const fn is_valid(self) -> bool {
407        self.start.is_valid()
408            && self.end.is_valid()
409            && (self.end.line > self.start.line
410                || (self.end.line == self.start.line && self.end.col >= self.start.col))
411    }
412}
413
414#[cfg(test)]
415mod tests {
416    use super::super::ids::Id;
417    use super::{CommentKind, Position, SourceDocument, SourceFile, Span};
418
419    #[test]
420    fn positions_are_one_based_and_validated() {
421        assert!(Position::new(1, 1).is_valid());
422        assert!(Position::new(10, 24).is_valid());
423        assert!(!Position::new(0, 1).is_valid());
424        assert!(!Position::new(1, 0).is_valid());
425    }
426
427    #[test]
428    fn spans_require_end_not_before_start() {
429        let file = Id::from_index(0);
430        assert!(Span::new(file, Position::new(1, 1), Position::new(1, 5)).is_valid());
431        assert!(Span::new(file, Position::new(1, 1), Position::new(2, 1)).is_valid());
432        assert!(Span::new(file, Position::new(1, 1), Position::new(1, 1)).is_valid());
433        assert!(!Span::new(file, Position::new(1, 5), Position::new(1, 1)).is_valid());
434        assert!(!Span::new(file, Position::new(2, 1), Position::new(1, 1)).is_valid());
435    }
436
437    #[test]
438    fn source_files_carry_paths() {
439        let file = SourceFile::new("source.opy");
440        assert_eq!(file.path, "source.opy");
441        assert!(file.source().is_none());
442    }
443
444    #[test]
445    fn source_documents_index_line_comments_but_not_string_contents() {
446        let document = SourceDocument::new("// before\nWait(\"// not a comment\"); // after\n");
447        let comments: Vec<_> = document.comments().collect();
448        assert_eq!(comments.len(), 2);
449        assert_eq!(comments[0].kind(), CommentKind::Line);
450        assert_eq!(comments[0].text(&document), "// before");
451        assert_eq!(comments[1].text(&document), "// after");
452    }
453
454    #[test]
455    fn source_edits_are_checked_and_reindex_comments() {
456        let document = SourceDocument::new("// keep\nvalue: 1\n");
457        let edit = document.edit(15..16, "2").expect("valid edit");
458        let updated = document.apply(&[edit]).expect("edit applies");
459        assert_eq!(updated.text(), "// keep\nvalue: 2\n");
460        assert_eq!(updated.comments().count(), 1);
461    }
462
463    #[test]
464    fn source_edits_reject_stale_and_overlapping_inputs() {
465        let document = SourceDocument::new("abcdef");
466        let edit = document.edit(1..3, "x").unwrap();
467        assert!(matches!(
468            edit.apply("aXcdef"),
469            Err(super::SourceEditError::SourceMismatch)
470        ));
471        let left = document.edit(1..3, "x").unwrap();
472        let right = document.edit(2..4, "y").unwrap();
473        assert!(matches!(
474            document.apply(&[left, right]),
475            Err(super::SourceEditError::OverlappingEdits)
476        ));
477    }
478
479    #[test]
480    fn source_comment_ranges_exclude_crlf_line_endings() {
481        let document = SourceDocument::new("// comment\r\nnext\r\n");
482        let comment = document.comments().next().unwrap();
483        assert_eq!(comment.text(&document), "// comment");
484    }
485
486    fn naive_byte_offset(source: &str, position: Position) -> Option<usize> {
487        if !position.is_valid() {
488            return None;
489        }
490        let mut line = 1;
491        let mut col = 1;
492        for (index, character) in source.char_indices() {
493            if line == position.line && col == position.col {
494                return Some(index);
495            }
496            if character == '\n' {
497                line += 1;
498                col = 1;
499            } else {
500                col += 1;
501            }
502        }
503        (line == position.line && col == position.col).then_some(source.len())
504    }
505
506    #[test]
507    fn byte_offsets_match_full_document_scan() {
508        let text = "one\r\ntwo\nthree\nlast é\u{301}\n";
509        let document = SourceDocument::new(text);
510        for line in 0..=8 {
511            for col in 0..=12 {
512                let position = Position::new(line, col);
513                assert_eq!(
514                    document.byte_offset_scan(position).0,
515                    naive_byte_offset(text, position),
516                    "position {position:?}"
517                );
518            }
519        }
520    }
521
522    #[test]
523    fn byte_offsets_scan_only_the_addressed_line() {
524        let line = "xxxxxxxxxxxxxxxx\n";
525        let document = SourceDocument::new(line.repeat(1000));
526        let (_, scanned) = document.byte_offset_scan(Position::new(1000, 9));
527        assert_eq!(document.line_starts.len(), 1001);
528        assert!(
529            scanned <= line.len(),
530            "resolving a late position scanned {scanned} bytes, expected at most one line ({})",
531            line.len()
532        );
533        let (_, scanned) = document.byte_offset_scan(Position::new(999, 20));
534        assert!(
535            scanned <= line.len(),
536            "an overshot column scanned {scanned} bytes, expected at most one line ({})",
537            line.len()
538        );
539        assert!(document.byte_offset(Position::new(1002, 1)).is_none());
540    }
541}