1use std::borrow::Cow;
4use std::collections::{BTreeMap, BTreeSet};
5use std::fmt;
6use std::ops::Range;
7use std::str::FromStr as _;
8use std::sync::{Arc, OnceLock, RwLock};
9use std::time::{Duration, Instant};
10
11use rhai::{CustomType, Dynamic, Engine, FuncRegistration, ImmutableString, TypeBuilder};
12use similar::{Algorithm, DiffOp, capture_diff_slices_deadline};
13use syntect::easy::ScopeRangeIterator;
14use syntect::highlighting::ScopeSelectors;
15use syntect::parsing::{ParseState, ScopeStack, SyntaxDefinition, SyntaxSet};
16use thiserror::Error;
17use unicode_segmentation::UnicodeSegmentation as _;
18
19use crate::ComponentInstancePath;
20
21pub const DEFAULT_DOCUMENT_MAX_BYTES: usize = 10 * 1024 * 1024;
22pub const DEFAULT_DOCUMENT_MAX_LINES: usize = 500_000;
23pub const DEFAULT_DIFF_MAX_HUNKS: usize = 100_000;
24pub const DEFAULT_DIFF_TIMEOUT: Duration = Duration::from_secs(2);
25const MAX_INLINE_DIFF_BYTES: usize = 256 * 1024;
26
27#[derive(Clone, Copy, Debug, Eq, PartialEq)]
28pub struct DocumentLimits {
29 pub max_document_bytes: usize,
30 pub max_document_lines: usize,
31 pub max_diff_total_bytes: usize,
32 pub max_diff_hunks: usize,
33 pub diff_timeout: Duration,
34}
35
36impl Default for DocumentLimits {
37 fn default() -> Self {
38 Self {
39 max_document_bytes: DEFAULT_DOCUMENT_MAX_BYTES,
40 max_document_lines: DEFAULT_DOCUMENT_MAX_LINES,
41 max_diff_total_bytes: DEFAULT_DOCUMENT_MAX_BYTES * 2,
42 max_diff_hunks: DEFAULT_DIFF_MAX_HUNKS,
43 diff_timeout: DEFAULT_DIFF_TIMEOUT,
44 }
45 }
46}
47
48impl DocumentLimits {
49 pub fn validate(self) -> Result<(), DocumentError> {
55 if self.max_document_bytes == 0
56 || self.max_document_lines == 0
57 || self.max_diff_total_bytes == 0
58 || self.max_diff_hunks == 0
59 || self.diff_timeout.is_zero()
60 {
61 Err(DocumentError::InvalidLimits)
62 } else {
63 Ok(())
64 }
65 }
66}
67
68#[derive(Clone, Debug)]
69pub struct DocumentRuntimeConfig {
70 limits: Arc<RwLock<DocumentLimits>>,
71}
72
73impl Default for DocumentRuntimeConfig {
74 fn default() -> Self {
75 Self::new()
76 }
77}
78
79impl DocumentRuntimeConfig {
80 #[must_use]
81 pub fn new() -> Self {
82 Self {
83 limits: Arc::new(RwLock::new(DocumentLimits::default())),
84 }
85 }
86
87 #[must_use]
88 pub fn limits(&self) -> DocumentLimits {
89 self.limits
90 .read()
91 .map_or_else(|poisoned| *poisoned.into_inner(), |limits| *limits)
92 }
93
94 pub fn set_limits(&self, limits: DocumentLimits) -> Result<(), DocumentError> {
100 limits.validate()?;
101 *self
102 .limits
103 .write()
104 .map_err(|_| DocumentError::DocumentConfigPoisoned)? = limits;
105 Ok(())
106 }
107}
108
109const RHAI_SYNTAX: &str = r#"%YAML 1.2
110---
111name: Rhai
112file_extensions: [rhai]
113scope: source.rhai
114contexts:
115 main:
116 - include: comments
117 - match: '\b(fn|let|const|if|else|switch|for|in|while|loop|do|until|break|continue|return|throw|try|catch|import|export|as|private)\b'
118 scope: keyword.control.rhai
119 - match: '\b(true|false)\b'
120 scope: constant.language.rhai
121 - match: '\b[0-9]+(?:\.[0-9]+)?\b'
122 scope: constant.numeric.rhai
123 - match: '"'
124 push: double-quoted-string
125 - match: "'"
126 push: single-quoted-string
127 - match: '`'
128 push: template-string
129 - match: '\b[A-Za-z_][A-Za-z0-9_]*(?=\s*\()'
130 scope: entity.name.function.rhai
131 - match: '[+\-*/%!=<>|&^~?:]+'
132 scope: keyword.operator.rhai
133 comments:
134 - match: '//.*$'
135 scope: comment.line.double-slash.rhai
136 - match: '/\*'
137 scope: punctuation.definition.comment.begin.rhai
138 push:
139 - meta_scope: comment.block.rhai
140 - match: '\*/'
141 scope: punctuation.definition.comment.end.rhai
142 pop: true
143 double-quoted-string:
144 - meta_scope: string.quoted.double.rhai
145 - match: '\\.'
146 scope: constant.character.escape.rhai
147 - match: '"'
148 pop: true
149 single-quoted-string:
150 - meta_scope: string.quoted.single.rhai
151 - match: '\\.'
152 scope: constant.character.escape.rhai
153 - match: "'"
154 pop: true
155 template-string:
156 - meta_scope: string.quoted.other.rhai
157 - match: '\\.'
158 scope: constant.character.escape.rhai
159 - match: '`'
160 pop: true
161"#;
162
163const TYPESCRIPT_SYNTAX: &str = r#"%YAML 1.2
164---
165name: TypeScript
166file_extensions: [ts, tsx]
167scope: source.ts
168contexts:
169 main:
170 - match: '//.*$'
171 scope: comment.line.double-slash.ts
172 - match: '/\*'
173 push:
174 - meta_scope: comment.block.ts
175 - match: '\*/'
176 pop: true
177 - match: '\b(async|await|break|case|catch|class|const|continue|debugger|default|delete|do|else|enum|export|extends|finally|for|from|function|get|if|implements|import|in|instanceof|interface|keyof|let|namespace|new|of|private|protected|public|readonly|return|set|static|super|switch|throw|try|type|typeof|var|void|while|with|yield)\b'
178 scope: keyword.control.ts
179 - match: '\b(any|bigint|boolean|never|number|object|string|symbol|unknown|undefined|null|true|false)\b'
180 scope: storage.type.ts
181 - match: '\b[0-9]+(?:\.[0-9]+)?\b'
182 scope: constant.numeric.ts
183 - match: '"'
184 push: double-string
185 - match: "'"
186 push: single-string
187 - match: '`'
188 push: template-string
189 - match: '\b[A-Za-z_$][A-Za-z0-9_$]*(?=\s*\()'
190 scope: entity.name.function.ts
191 - match: '</?[A-Za-z][A-Za-z0-9:.-]*'
192 scope: entity.name.tag.tsx
193 - match: '[+\-*/%!=<>|&^~?:]+'
194 scope: keyword.operator.ts
195 double-string:
196 - meta_scope: string.quoted.double.ts
197 - match: '\\.'
198 scope: constant.character.escape.ts
199 - match: '"'
200 pop: true
201 single-string:
202 - meta_scope: string.quoted.single.ts
203 - match: '\\.'
204 scope: constant.character.escape.ts
205 - match: "'"
206 pop: true
207 template-string:
208 - meta_scope: string.quoted.other.ts
209 - match: '\\.'
210 scope: constant.character.escape.ts
211 - match: '`'
212 pop: true
213"#;
214
215const JSONC_SYNTAX: &str = r#"%YAML 1.2
216---
217name: JSON with Comments
218file_extensions: [jsonc]
219scope: source.json.comments
220contexts:
221 main:
222 - match: '//.*$'
223 scope: comment.line.double-slash.json
224 - match: '/\*'
225 push:
226 - meta_scope: comment.block.json
227 - match: '\*/'
228 pop: true
229 - match: '"'
230 push: string
231 - match: '-?\b[0-9]+(?:\.[0-9]+)?(?:[eE][+-]?[0-9]+)?\b'
232 scope: constant.numeric.json
233 - match: '\b(true|false|null)\b'
234 scope: constant.language.json
235 - match: '[{}\[\],:]'
236 scope: punctuation.separator.json
237 string:
238 - meta_scope: string.quoted.double.json
239 - match: '\\(?:["\\/bfnrt]|u[0-9A-Fa-f]{4})'
240 scope: constant.character.escape.json
241 - match: '"'
242 pop: true
243"#;
244
245const TOML_SYNTAX: &str = r#"%YAML 1.2
246---
247name: TOML
248file_extensions: [toml]
249scope: source.toml
250contexts:
251 main:
252 - match: '#.*$'
253 scope: comment.line.number-sign.toml
254 - match: '^\s*\[\[?[^\]]+\]\]?'
255 scope: entity.name.section.toml
256 - match: '^[A-Za-z0-9_.-]+(?=\s*=)'
257 scope: variable.other.key.toml
258 - match: '"""'
259 push: multiline-string
260 - match: "'''"
261 push: multiline-literal
262 - match: '"'
263 push: string
264 - match: "'"
265 push: literal
266 - match: '\b(true|false)\b'
267 scope: constant.language.toml
268 - match: '[+-]?\b(?:0x[0-9A-Fa-f_]+|0o[0-7_]+|0b[01_]+|[0-9][0-9_]*(?:\.[0-9_]+)?)\b'
269 scope: constant.numeric.toml
270 - match: '[=,.{}\[\]]'
271 scope: punctuation.separator.toml
272 string:
273 - meta_scope: string.quoted.double.toml
274 - match: '\\.'
275 scope: constant.character.escape.toml
276 - match: '"'
277 pop: true
278 literal:
279 - meta_scope: string.quoted.single.toml
280 - match: "'"
281 pop: true
282 multiline-string:
283 - meta_scope: string.quoted.triple.toml
284 - match: '"""'
285 pop: true
286 multiline-literal:
287 - meta_scope: string.quoted.triple.toml
288 - match: "'''"
289 pop: true
290"#;
291
292const DOCKERFILE_SYNTAX: &str = r#"%YAML 1.2
293---
294name: Dockerfile
295file_extensions: [Dockerfile, dockerfile]
296scope: source.dockerfile
297contexts:
298 main:
299 - match: '#.*$'
300 scope: comment.line.number-sign.dockerfile
301 - match: '(?i)^\s*(ADD|ARG|CMD|COPY|ENTRYPOINT|ENV|EXPOSE|FROM|HEALTHCHECK|LABEL|MAINTAINER|ONBUILD|RUN|SHELL|STOPSIGNAL|USER|VOLUME|WORKDIR)(?=\s)'
302 scope: keyword.control.dockerfile
303 - match: '\$\{?[A-Za-z_][A-Za-z0-9_]*\}?'
304 scope: variable.other.dockerfile
305 - match: '"'
306 push: double-string
307 - match: "'"
308 push: single-string
309 double-string:
310 - meta_scope: string.quoted.double.dockerfile
311 - match: '\\.'
312 scope: constant.character.escape.dockerfile
313 - match: '"'
314 pop: true
315 single-string:
316 - meta_scope: string.quoted.single.dockerfile
317 - match: "'"
318 pop: true
319"#;
320
321const EXTRA_SYNTAXES: &[&str] = &[
322 RHAI_SYNTAX,
323 TYPESCRIPT_SYNTAX,
324 JSONC_SYNTAX,
325 TOML_SYNTAX,
326 DOCKERFILE_SYNTAX,
327];
328
329#[derive(Clone)]
330pub struct NativeTextDocument {
331 snapshot: Arc<NativeTextSnapshot>,
332}
333
334#[derive(Debug, Eq, PartialEq)]
335struct NativeTextSnapshot {
336 identity: String,
337 revision: u64,
338 text: Arc<str>,
339}
340
341impl NativeTextDocument {
342 pub fn new(
348 identity: impl Into<String>,
349 revision: u64,
350 text: impl Into<Arc<str>>,
351 ) -> Result<Self, DocumentError> {
352 let identity = identity.into();
353 validate_identity(&identity)?;
354 Ok(Self {
355 snapshot: Arc::new(NativeTextSnapshot {
356 identity,
357 revision,
358 text: text.into(),
359 }),
360 })
361 }
362
363 #[must_use]
364 pub fn identity(&self) -> &str {
365 &self.snapshot.identity
366 }
367
368 #[must_use]
369 pub fn revision(&self) -> u64 {
370 self.snapshot.revision
371 }
372
373 #[must_use]
374 pub fn text(&self) -> &str {
375 &self.snapshot.text
376 }
377
378 #[must_use]
379 pub fn text_arc(&self) -> Arc<str> {
380 Arc::clone(&self.snapshot.text)
381 }
382}
383
384impl PartialEq for NativeTextDocument {
385 fn eq(&self, other: &Self) -> bool {
386 Arc::ptr_eq(&self.snapshot, &other.snapshot) || self.snapshot == other.snapshot
387 }
388}
389
390impl Eq for NativeTextDocument {}
391
392impl fmt::Debug for NativeTextDocument {
393 fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
394 formatter
395 .debug_struct("NativeTextDocument")
396 .field("identity", &self.identity())
397 .field("revision", &self.revision())
398 .field("bytes", &self.text().len())
399 .finish()
400 }
401}
402
403impl CustomType for NativeTextDocument {
404 fn build(mut builder: TypeBuilder<Self>) {
405 builder
406 .with_name("NativeTextDocument")
407 .with_get("identity", |document: &mut Self| {
408 ImmutableString::from(document.identity().to_owned())
409 })
410 .with_get("revision", |document: &mut Self| {
411 i64::try_from(document.revision()).unwrap_or(i64::MAX)
412 })
413 .with_get("len", |document: &mut Self| {
414 i64::try_from(document.text().len()).unwrap_or(i64::MAX)
415 })
416 .with_fn("to_string", |document: &mut Self| {
417 format!(
418 "NativeTextDocument({}, revision={})",
419 document.identity(),
420 document.revision()
421 )
422 });
423 }
424}
425
426pub(crate) fn register_document_api(engine: &mut Engine) {
427 engine.build_type::<NativeTextDocument>();
428 FuncRegistration::new("is_native_text_document")
429 .in_global_namespace()
430 .register_into_engine(engine, |value: Dynamic| value.is::<NativeTextDocument>());
431}
432
433#[derive(Clone, Debug, Default)]
434pub struct NativeTextDocumentRegistry {
435 documents: BTreeMap<String, NativeTextDocument>,
436 readers: BTreeMap<String, BTreeSet<ComponentInstancePath>>,
437}
438
439impl NativeTextDocumentRegistry {
440 #[must_use]
441 pub fn new() -> Self {
442 Self::default()
443 }
444
445 pub fn register(
451 &mut self,
452 name: impl Into<String>,
453 document: NativeTextDocument,
454 ) -> Result<(), DocumentError> {
455 let name = name.into();
456 validate_identity(&name)?;
457 if self.documents.contains_key(&name) {
458 return Err(DocumentError::DuplicateDocument(name));
459 }
460 self.documents.insert(name, document);
461 Ok(())
462 }
463
464 pub fn get(&self, name: &str) -> Result<NativeTextDocument, DocumentError> {
473 self.documents
474 .get(name)
475 .cloned()
476 .ok_or_else(|| DocumentError::UnknownDocument(name.to_owned()))
477 }
478
479 pub fn replace(
485 &mut self,
486 name: &str,
487 document: NativeTextDocument,
488 ) -> Result<BTreeSet<ComponentInstancePath>, DocumentError> {
489 let current = self
490 .documents
491 .get_mut(name)
492 .ok_or_else(|| DocumentError::UnknownDocument(name.to_owned()))?;
493 if current == &document {
494 return Ok(BTreeSet::new());
495 }
496 if current.identity() == document.identity() && document.revision() <= current.revision() {
497 return Err(DocumentError::NonMonotonicRevision {
498 identity: document.identity().to_owned(),
499 current: current.revision(),
500 next: document.revision(),
501 });
502 }
503 *current = document;
504 Ok(self.readers.get(name).cloned().unwrap_or_default())
505 }
506
507 pub(crate) fn read_tracked(
508 &mut self,
509 reader: &ComponentInstancePath,
510 name: &str,
511 ) -> Result<NativeTextDocument, DocumentError> {
512 let document = self
513 .documents
514 .get(name)
515 .cloned()
516 .ok_or_else(|| DocumentError::UnknownDocument(name.to_owned()))?;
517 self.readers
518 .entry(name.to_owned())
519 .or_default()
520 .insert(reader.clone());
521 Ok(document)
522 }
523
524 pub(crate) fn reset_reader(&mut self, reader: &ComponentInstancePath) {
525 for readers in self.readers.values_mut() {
526 readers.remove(reader);
527 }
528 self.readers.retain(|_, readers| !readers.is_empty());
529 }
530
531 pub(crate) fn retain_reader_scope(
532 &mut self,
533 root: &ComponentInstancePath,
534 active: &BTreeSet<ComponentInstancePath>,
535 ) {
536 for readers in self.readers.values_mut() {
537 readers.retain(|reader| {
538 !reader.is_within(root) || reader == root || active.contains(reader)
539 });
540 }
541 self.readers.retain(|_, readers| !readers.is_empty());
542 }
543
544 pub(crate) fn remove_reader_scope(&mut self, root: &ComponentInstancePath) {
545 for readers in self.readers.values_mut() {
546 readers.retain(|reader| !reader.is_within(root));
547 }
548 self.readers.retain(|_, readers| !readers.is_empty());
549 }
550}
551
552#[derive(Clone)]
553pub struct SyntaxRegistry {
554 syntax_set: Arc<RwLock<Arc<SyntaxSet>>>,
555}
556
557impl fmt::Debug for SyntaxRegistry {
558 fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
559 formatter
560 .debug_struct("SyntaxRegistry")
561 .field("syntaxes", &self.snapshot().syntaxes().len())
562 .finish()
563 }
564}
565
566impl Default for SyntaxRegistry {
567 fn default() -> Self {
568 Self::new()
569 }
570}
571
572impl SyntaxRegistry {
573 #[must_use]
574 pub fn new() -> Self {
575 Self {
576 syntax_set: Arc::new(RwLock::new(Arc::clone(default_syntax_set()))),
577 }
578 }
579
580 pub fn register_sublime_syntax(&self, source: &str) -> Result<(), DocumentError> {
586 let syntax = SyntaxDefinition::load_from_str(source, true, None)
587 .map_err(|error| DocumentError::Syntax(error.to_string()))?;
588 let current = self
589 .syntax_set
590 .read()
591 .map_err(|_| DocumentError::SyntaxRegistryPoisoned)?
592 .clone();
593 let mut builder = current.as_ref().clone().into_builder();
594 builder.add(syntax);
595 let next = Arc::new(builder.build());
596 *self
597 .syntax_set
598 .write()
599 .map_err(|_| DocumentError::SyntaxRegistryPoisoned)? = next;
600 Ok(())
601 }
602
603 #[must_use]
604 pub fn snapshot(&self) -> Arc<SyntaxSet> {
605 self.syntax_set
606 .read()
607 .map_or_else(|poisoned| poisoned.into_inner().clone(), |set| set.clone())
608 }
609}
610
611fn default_syntax_set() -> &'static Arc<SyntaxSet> {
612 static SYNTAXES: OnceLock<Arc<SyntaxSet>> = OnceLock::new();
613 SYNTAXES.get_or_init(|| {
614 let mut builder = SyntaxSet::load_defaults_newlines().into_builder();
615 for source in EXTRA_SYNTAXES {
616 builder.add(
617 SyntaxDefinition::load_from_str(source, true, None)
618 .expect("built-in syntax is valid"),
619 );
620 }
621 Arc::new(builder.build())
622 })
623}
624
625#[derive(Clone, Debug, Eq, PartialEq)]
626pub enum DocumentSource {
627 Inline(Arc<str>),
628 Native(NativeTextDocument),
629}
630
631impl DocumentSource {
632 #[must_use]
633 pub fn identity(&self) -> Cow<'_, str> {
634 match self {
635 Self::Inline(_) => Cow::Borrowed("inline"),
636 Self::Native(document) => Cow::Borrowed(document.identity()),
637 }
638 }
639
640 #[must_use]
641 pub fn revision(&self) -> u64 {
642 match self {
643 Self::Inline(text) => stable_text_hash(text),
644 Self::Native(document) => document.revision(),
645 }
646 }
647
648 #[must_use]
649 pub fn text(&self) -> Arc<str> {
650 match self {
651 Self::Inline(text) => Arc::clone(text),
652 Self::Native(document) => document.text_arc(),
653 }
654 }
655}
656
657impl From<String> for DocumentSource {
658 fn from(value: String) -> Self {
659 Self::Inline(Arc::from(value))
660 }
661}
662
663impl From<NativeTextDocument> for DocumentSource {
664 fn from(value: NativeTextDocument) -> Self {
665 Self::Native(value)
666 }
667}
668
669#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
670pub enum DocumentWrap {
671 #[default]
672 None,
673 Viewport,
674 Column(usize),
675}
676
677#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
678pub enum DiffViewMode {
679 #[default]
680 Unified,
681 Split,
682}
683
684#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
685pub enum DiffWhitespace {
686 #[default]
687 Exact,
688 IgnoreChanges,
689 IgnoreAll,
690}
691
692#[derive(Clone, Debug, Eq, PartialEq)]
693pub struct DocumentDescriptor {
694 pub source: DocumentSource,
695 pub label: String,
696 pub file_name: Option<String>,
697 pub language: Option<String>,
698}
699
700#[derive(Clone, Copy, Debug, Eq, PartialEq)]
701pub enum SyntaxTokenKind {
702 Comment,
703 String,
704 Number,
705 Keyword,
706 Function,
707 Type,
708 Variable,
709 Constant,
710 Operator,
711 Punctuation,
712 Tag,
713 Attribute,
714}
715
716impl SyntaxTokenKind {
717 #[must_use]
718 pub const fn theme_token(self) -> &'static str {
719 match self {
720 Self::Comment => "syntax.comment",
721 Self::String => "syntax.string",
722 Self::Number => "syntax.number",
723 Self::Keyword => "syntax.keyword",
724 Self::Function => "syntax.function",
725 Self::Type => "syntax.type",
726 Self::Variable => "syntax.variable",
727 Self::Constant => "syntax.constant",
728 Self::Operator => "syntax.operator",
729 Self::Punctuation => "syntax.punctuation",
730 Self::Tag => "syntax.tag",
731 Self::Attribute => "syntax.attribute",
732 }
733 }
734}
735
736#[derive(Clone, Debug, Eq, PartialEq)]
737pub struct SyntaxSpan {
738 pub range: Range<usize>,
739 pub kind: SyntaxTokenKind,
740}
741
742#[derive(Clone, Debug, Eq, PartialEq)]
743pub struct DocumentLine {
744 pub range: Range<usize>,
745 pub syntax: Vec<SyntaxSpan>,
746}
747
748#[derive(Clone, Debug)]
749pub struct PreparedDocument {
750 identity: String,
751 revision: u64,
752 text: Arc<str>,
753 lines: Arc<[DocumentLine]>,
754 language: String,
755 max_line_chars: usize,
756 ends_with_newline: bool,
757 highlight_error: Option<String>,
758}
759
760impl PreparedDocument {
761 #[must_use]
762 pub fn identity(&self) -> &str {
763 &self.identity
764 }
765
766 #[must_use]
767 pub const fn revision(&self) -> u64 {
768 self.revision
769 }
770
771 #[must_use]
772 pub fn text(&self) -> &str {
773 &self.text
774 }
775
776 #[must_use]
777 pub fn text_arc(&self) -> Arc<str> {
778 Arc::clone(&self.text)
779 }
780
781 #[must_use]
782 pub fn lines(&self) -> &[DocumentLine] {
783 &self.lines
784 }
785
786 #[must_use]
787 pub fn line_text(&self, index: usize) -> Option<&str> {
788 let line = self.lines.get(index)?;
789 self.text.get(line.range.clone())
790 }
791
792 #[must_use]
793 pub fn language(&self) -> &str {
794 &self.language
795 }
796
797 #[must_use]
798 pub const fn max_line_chars(&self) -> usize {
799 self.max_line_chars
800 }
801
802 #[must_use]
803 pub const fn ends_with_newline(&self) -> bool {
804 self.ends_with_newline
805 }
806
807 #[must_use]
808 pub fn highlight_error(&self) -> Option<&str> {
809 self.highlight_error.as_deref()
810 }
811}
812
813pub fn prepare_document(
820 descriptor: &DocumentDescriptor,
821 syntaxes: &SyntaxRegistry,
822) -> Result<PreparedDocument, DocumentError> {
823 prepare_document_with_limits(descriptor, syntaxes, DocumentLimits::default())
824}
825
826pub fn prepare_document_with_limits(
832 descriptor: &DocumentDescriptor,
833 syntaxes: &SyntaxRegistry,
834 limits: DocumentLimits,
835) -> Result<PreparedDocument, DocumentError> {
836 let text = descriptor.source.text();
837 if text.len() > limits.max_document_bytes {
838 return Err(DocumentError::TooManyBytes {
839 actual: text.len(),
840 limit: limits.max_document_bytes,
841 });
842 }
843 let ranges = document_line_ranges(&text);
844 if ranges.len() > limits.max_document_lines {
845 return Err(DocumentError::TooManyLines {
846 actual: ranges.len(),
847 limit: limits.max_document_lines,
848 });
849 }
850 let syntax_set = syntaxes.snapshot();
851 let syntax = resolve_syntax(
852 &syntax_set,
853 descriptor.language.as_deref(),
854 descriptor.file_name.as_deref(),
855 );
856 let language = syntax.name.clone();
857 let mut parser = ParseState::new(syntax);
858 let mut stack = ScopeStack::new();
859 let selectors = syntax_selectors();
860 let mut lines = Vec::with_capacity(ranges.len());
861 let mut max_line_chars = 0usize;
862 let mut highlight_error = None;
863 let mut highlighting = true;
864 for range in ranges {
865 let line = &text[range.clone()];
866 max_line_chars = max_line_chars.max(line.chars().count());
867 let parse_end = text[range.end..]
868 .find('\n')
869 .map_or(range.end, |offset| range.end + offset + 1);
870 let parse_start = range.start;
871 let parse_line = &text[parse_start..parse_end];
872 let mut spans: Vec<SyntaxSpan> = Vec::new();
873 if highlighting {
874 match parser.parse_line(parse_line, &syntax_set) {
875 Ok(operations) => {
876 for (token_range, operation) in ScopeRangeIterator::new(&operations, parse_line)
877 {
878 if let Err(error) = stack.apply(operation) {
879 highlight_error.get_or_insert_with(|| error.to_string());
880 highlighting = false;
881 spans.clear();
882 break;
883 }
884 let start = token_range.start.min(line.len());
885 let end = token_range.end.min(line.len());
886 if start >= end {
887 continue;
888 }
889 if let Some(kind) = selectors.kind(stack.as_slice()) {
890 if let Some(previous) = spans.last_mut()
891 && previous.kind == kind
892 && previous.range.end == start
893 {
894 previous.range.end = end;
895 } else {
896 spans.push(SyntaxSpan {
897 range: start..end,
898 kind,
899 });
900 }
901 }
902 }
903 }
904 Err(error) => {
905 highlight_error.get_or_insert_with(|| error.to_string());
906 highlighting = false;
907 }
908 }
909 }
910 lines.push(DocumentLine {
911 range,
912 syntax: spans,
913 });
914 }
915 Ok(PreparedDocument {
916 identity: descriptor.source.identity().into_owned(),
917 revision: descriptor.source.revision(),
918 text,
919 lines: lines.into(),
920 language,
921 max_line_chars,
922 ends_with_newline: descriptor.source.text().ends_with('\n'),
923 highlight_error,
924 })
925}
926
927fn document_line_ranges(text: &str) -> Vec<Range<usize>> {
928 let mut ranges = Vec::new();
929 let mut start = 0usize;
930 for (index, byte) in text.bytes().enumerate() {
931 if byte == b'\n' {
932 let mut end = index;
933 if text.as_bytes().get(index.wrapping_sub(1)) == Some(&b'\r') && index > start {
934 end -= 1;
935 }
936 ranges.push(start..end);
937 start = index + 1;
938 }
939 }
940 if start < text.len() || text.is_empty() || text.ends_with('\n') {
941 ranges.push(start..text.len());
942 }
943 ranges
944}
945
946fn resolve_syntax<'a>(
947 syntaxes: &'a SyntaxSet,
948 language: Option<&str>,
949 file_name: Option<&str>,
950) -> &'a syntect::parsing::SyntaxReference {
951 language
952 .and_then(|language| {
953 let token = language_alias(language);
954 syntaxes.find_syntax_by_token(token.as_ref())
955 })
956 .or_else(|| {
957 file_name.and_then(|file_name| {
958 syntaxes.find_syntax_by_extension(file_name).or_else(|| {
959 file_name
960 .rsplit_once('.')
961 .and_then(|(_, ext)| syntaxes.find_syntax_by_extension(ext))
962 })
963 })
964 })
965 .unwrap_or_else(|| syntaxes.find_syntax_plain_text())
966}
967
968fn language_alias(language: &str) -> Cow<'_, str> {
969 let normalized = language.trim().to_ascii_lowercase();
970 match normalized.as_str() {
971 "plaintext" | "plain_text" | "text" => Cow::Borrowed("txt"),
972 "javascript" => Cow::Borrowed("js"),
973 "typescript" => Cow::Borrowed("ts"),
974 "shell" | "bash" => Cow::Borrowed("sh"),
975 "python" => Cow::Borrowed("py"),
976 "markdown" => Cow::Borrowed("md"),
977 "rust" => Cow::Borrowed("rs"),
978 "golang" => Cow::Borrowed("go"),
979 _ => Cow::Owned(normalized),
980 }
981}
982
983struct SyntaxSelectors {
984 comment: ScopeSelectors,
985 string: ScopeSelectors,
986 number: ScopeSelectors,
987 keyword: ScopeSelectors,
988 function: ScopeSelectors,
989 type_name: ScopeSelectors,
990 variable: ScopeSelectors,
991 constant: ScopeSelectors,
992 operator: ScopeSelectors,
993 punctuation: ScopeSelectors,
994 tag: ScopeSelectors,
995 attribute: ScopeSelectors,
996}
997
998impl SyntaxSelectors {
999 fn new() -> Self {
1000 let parse = |value: &str| ScopeSelectors::from_str(value).expect("static scope selector");
1001 Self {
1002 comment: parse("comment"),
1003 string: parse("string"),
1004 number: parse("constant.numeric"),
1005 keyword: parse("keyword, storage.modifier"),
1006 function: parse("entity.name.function, support.function"),
1007 type_name: parse("entity.name.type, support.type, storage.type"),
1008 variable: parse("variable"),
1009 constant: parse("constant"),
1010 operator: parse("keyword.operator"),
1011 punctuation: parse("punctuation"),
1012 tag: parse("entity.name.tag"),
1013 attribute: parse("entity.other.attribute-name"),
1014 }
1015 }
1016
1017 fn kind(&self, stack: &[syntect::parsing::Scope]) -> Option<SyntaxTokenKind> {
1018 [
1019 (&self.comment, SyntaxTokenKind::Comment),
1020 (&self.string, SyntaxTokenKind::String),
1021 (&self.number, SyntaxTokenKind::Number),
1022 (&self.function, SyntaxTokenKind::Function),
1023 (&self.type_name, SyntaxTokenKind::Type),
1024 (&self.tag, SyntaxTokenKind::Tag),
1025 (&self.attribute, SyntaxTokenKind::Attribute),
1026 (&self.operator, SyntaxTokenKind::Operator),
1027 (&self.keyword, SyntaxTokenKind::Keyword),
1028 (&self.variable, SyntaxTokenKind::Variable),
1029 (&self.constant, SyntaxTokenKind::Constant),
1030 (&self.punctuation, SyntaxTokenKind::Punctuation),
1031 ]
1032 .into_iter()
1033 .find_map(|(selector, kind)| selector.does_match(stack).map(|_| kind))
1034 }
1035}
1036
1037fn syntax_selectors() -> &'static SyntaxSelectors {
1038 static SELECTORS: OnceLock<SyntaxSelectors> = OnceLock::new();
1039 SELECTORS.get_or_init(SyntaxSelectors::new)
1040}
1041
1042#[derive(Clone, Copy, Debug, Eq, PartialEq)]
1043pub enum DiffRowKind {
1044 Equal,
1045 LeftOnly,
1046 RightOnly,
1047 Modified,
1048}
1049
1050#[derive(Clone, Debug, Eq, PartialEq)]
1051pub struct DiffAlignedRow {
1052 pub left: Option<usize>,
1053 pub right: Option<usize>,
1054 pub kind: DiffRowKind,
1055 pub hunk: Option<usize>,
1056 pub left_inline: Vec<Range<usize>>,
1057 pub right_inline: Vec<Range<usize>>,
1058}
1059
1060#[derive(Clone, Debug, Eq, PartialEq)]
1061pub enum DiffDisplayRow {
1062 Content(usize),
1063 Fold {
1064 id: usize,
1065 full_range: Range<usize>,
1066 hidden_rows: usize,
1067 },
1068}
1069
1070#[derive(Clone, Debug)]
1071pub struct PreparedDiff {
1072 pub left: PreparedDocument,
1073 pub right: PreparedDocument,
1074 pub rows: Arc<[DiffAlignedRow]>,
1075 pub collapsed: Arc<[DiffDisplayRow]>,
1076 pub hunk_count: usize,
1077}
1078
1079pub fn prepare_diff(
1085 left: &DocumentDescriptor,
1086 right: &DocumentDescriptor,
1087 whitespace: DiffWhitespace,
1088 context_lines: Option<usize>,
1089 syntaxes: &SyntaxRegistry,
1090) -> Result<PreparedDiff, DocumentError> {
1091 prepare_diff_with_limits(
1092 left,
1093 right,
1094 whitespace,
1095 context_lines,
1096 syntaxes,
1097 DocumentLimits::default(),
1098 )
1099}
1100
1101#[allow(clippy::too_many_lines)]
1107pub fn prepare_diff_with_limits(
1108 left: &DocumentDescriptor,
1109 right: &DocumentDescriptor,
1110 whitespace: DiffWhitespace,
1111 context_lines: Option<usize>,
1112 syntaxes: &SyntaxRegistry,
1113 limits: DocumentLimits,
1114) -> Result<PreparedDiff, DocumentError> {
1115 let total_bytes = left
1116 .source
1117 .text()
1118 .len()
1119 .saturating_add(right.source.text().len());
1120 if total_bytes > limits.max_diff_total_bytes {
1121 return Err(DocumentError::TooManyDiffBytes {
1122 actual: total_bytes,
1123 limit: limits.max_diff_total_bytes,
1124 });
1125 }
1126 let left = prepare_document_with_limits(left, syntaxes, limits)?;
1127 let right = prepare_document_with_limits(right, syntaxes, limits)?;
1128 let left_keys = diff_keys(&left, whitespace);
1129 let right_keys = diff_keys(&right, whitespace);
1130 let deadline = Instant::now().checked_add(limits.diff_timeout);
1131 let ops = capture_diff_slices_deadline(Algorithm::Patience, &left_keys, &right_keys, deadline);
1132 let mut rows = Vec::with_capacity(left.lines().len().max(right.lines().len()));
1133 let mut hunk = 0usize;
1134 let mut in_change = false;
1135 for operation in ops {
1136 match operation {
1137 DiffOp::Equal {
1138 old_index,
1139 new_index,
1140 len,
1141 } => {
1142 in_change = false;
1143 rows.extend((0..len).map(|offset| DiffAlignedRow {
1144 left: Some(old_index + offset),
1145 right: Some(new_index + offset),
1146 kind: DiffRowKind::Equal,
1147 hunk: None,
1148 left_inline: Vec::new(),
1149 right_inline: Vec::new(),
1150 }));
1151 }
1152 DiffOp::Delete {
1153 old_index, old_len, ..
1154 } => {
1155 let id = next_hunk(&mut hunk, &mut in_change, limits.max_diff_hunks)?;
1156 rows.extend((0..old_len).map(|offset| DiffAlignedRow {
1157 left: Some(old_index + offset),
1158 right: None,
1159 kind: DiffRowKind::LeftOnly,
1160 hunk: Some(id),
1161 left_inline: Vec::new(),
1162 right_inline: Vec::new(),
1163 }));
1164 }
1165 DiffOp::Insert {
1166 new_index, new_len, ..
1167 } => {
1168 let id = next_hunk(&mut hunk, &mut in_change, limits.max_diff_hunks)?;
1169 rows.extend((0..new_len).map(|offset| DiffAlignedRow {
1170 left: None,
1171 right: Some(new_index + offset),
1172 kind: DiffRowKind::RightOnly,
1173 hunk: Some(id),
1174 left_inline: Vec::new(),
1175 right_inline: Vec::new(),
1176 }));
1177 }
1178 DiffOp::Replace {
1179 old_index,
1180 old_len,
1181 new_index,
1182 new_len,
1183 } => {
1184 let id = next_hunk(&mut hunk, &mut in_change, limits.max_diff_hunks)?;
1185 let aligned = old_len.max(new_len);
1186 for offset in 0..aligned {
1187 let left_index = (offset < old_len).then_some(old_index + offset);
1188 let right_index = (offset < new_len).then_some(new_index + offset);
1189 let (left_inline, right_inline) = match (left_index, right_index) {
1190 (Some(left_index), Some(right_index)) => inline_ranges(
1191 left.line_text(left_index).unwrap_or_default(),
1192 right.line_text(right_index).unwrap_or_default(),
1193 deadline,
1194 ),
1195 _ => (Vec::new(), Vec::new()),
1196 };
1197 rows.push(DiffAlignedRow {
1198 left: left_index,
1199 right: right_index,
1200 kind: match (left_index, right_index) {
1201 (Some(_), Some(_)) => DiffRowKind::Modified,
1202 (Some(_), None) => DiffRowKind::LeftOnly,
1203 (None, Some(_)) => DiffRowKind::RightOnly,
1204 (None, None) => unreachable!(),
1205 },
1206 hunk: Some(id),
1207 left_inline,
1208 right_inline,
1209 });
1210 }
1211 }
1212 }
1213 }
1214 let collapsed = collapse_rows(&rows, context_lines);
1215 Ok(PreparedDiff {
1216 left,
1217 right,
1218 rows: rows.into(),
1219 collapsed: collapsed.into(),
1220 hunk_count: hunk,
1221 })
1222}
1223
1224fn next_hunk(
1225 hunks: &mut usize,
1226 in_change: &mut bool,
1227 limit: usize,
1228) -> Result<usize, DocumentError> {
1229 if !*in_change {
1230 *hunks = hunks.checked_add(1).ok_or(DocumentError::TooManyHunks {
1231 actual: usize::MAX,
1232 limit,
1233 })?;
1234 *in_change = true;
1235 }
1236 if *hunks > limit {
1237 return Err(DocumentError::TooManyHunks {
1238 actual: *hunks,
1239 limit,
1240 });
1241 }
1242 Ok(*hunks - 1)
1243}
1244
1245fn diff_keys(document: &PreparedDocument, whitespace: DiffWhitespace) -> Vec<Cow<'_, str>> {
1246 document
1247 .lines()
1248 .iter()
1249 .enumerate()
1250 .map(|(index, _)| {
1251 let line = document.line_text(index).unwrap_or_default();
1252 match whitespace {
1253 DiffWhitespace::Exact => Cow::Borrowed(line),
1254 DiffWhitespace::IgnoreChanges => {
1255 Cow::Owned(line.split_whitespace().collect::<Vec<_>>().join(" "))
1256 }
1257 DiffWhitespace::IgnoreAll => Cow::Owned(
1258 line.chars()
1259 .filter(|character| !character.is_whitespace())
1260 .collect(),
1261 ),
1262 }
1263 })
1264 .collect()
1265}
1266
1267fn inline_ranges(
1268 left: &str,
1269 right: &str,
1270 deadline: Option<Instant>,
1271) -> (Vec<Range<usize>>, Vec<Range<usize>>) {
1272 if left.len().saturating_add(right.len()) > MAX_INLINE_DIFF_BYTES {
1273 return (
1274 (!left.is_empty())
1275 .then_some(0..left.len())
1276 .into_iter()
1277 .collect(),
1278 (!right.is_empty())
1279 .then_some(0..right.len())
1280 .into_iter()
1281 .collect(),
1282 );
1283 }
1284 let left_graphemes = left.graphemes(true).collect::<Vec<_>>();
1285 let right_graphemes = right.graphemes(true).collect::<Vec<_>>();
1286 let operations = capture_diff_slices_deadline(
1287 Algorithm::Patience,
1288 &left_graphemes,
1289 &right_graphemes,
1290 deadline,
1291 );
1292 let mut left_offset = 0usize;
1293 let mut right_offset = 0usize;
1294 let mut left_ranges = Vec::new();
1295 let mut right_ranges = Vec::new();
1296 for operation in operations {
1297 for change in operation.iter_changes(&left_graphemes, &right_graphemes) {
1298 let len = change.value().len();
1299 match change.tag() {
1300 similar::ChangeTag::Equal => {
1301 left_offset += len;
1302 right_offset += len;
1303 }
1304 similar::ChangeTag::Delete => {
1305 push_range(&mut left_ranges, left_offset..left_offset + len);
1306 left_offset += len;
1307 }
1308 similar::ChangeTag::Insert => {
1309 push_range(&mut right_ranges, right_offset..right_offset + len);
1310 right_offset += len;
1311 }
1312 }
1313 }
1314 }
1315 (left_ranges, right_ranges)
1316}
1317
1318fn push_range(ranges: &mut Vec<Range<usize>>, range: Range<usize>) {
1319 if range.is_empty() {
1320 return;
1321 }
1322 if let Some(previous) = ranges.last_mut()
1323 && previous.end == range.start
1324 {
1325 previous.end = range.end;
1326 } else {
1327 ranges.push(range);
1328 }
1329}
1330
1331fn collapse_rows(rows: &[DiffAlignedRow], context: Option<usize>) -> Vec<DiffDisplayRow> {
1332 let Some(context) = context else {
1333 return (0..rows.len()).map(DiffDisplayRow::Content).collect();
1334 };
1335 if rows.iter().all(|row| row.kind == DiffRowKind::Equal) {
1336 return if rows.is_empty() {
1337 Vec::new()
1338 } else {
1339 vec![DiffDisplayRow::Fold {
1340 id: 0,
1341 full_range: 0..rows.len(),
1342 hidden_rows: rows.len(),
1343 }]
1344 };
1345 }
1346 let mut keep = vec![false; rows.len()];
1347 for (index, row) in rows.iter().enumerate() {
1348 if row.kind != DiffRowKind::Equal {
1349 let start = index.saturating_sub(context);
1350 let end = index
1351 .saturating_add(context)
1352 .saturating_add(1)
1353 .min(rows.len());
1354 keep[start..end].fill(true);
1355 }
1356 }
1357 let mut display = Vec::new();
1358 let mut index = 0usize;
1359 let mut fold = 0usize;
1360 while index < rows.len() {
1361 if keep[index] {
1362 display.push(DiffDisplayRow::Content(index));
1363 index += 1;
1364 continue;
1365 }
1366 let start = index;
1367 while index < rows.len() && !keep[index] {
1368 index += 1;
1369 }
1370 display.push(DiffDisplayRow::Fold {
1371 id: fold,
1372 full_range: start..index,
1373 hidden_rows: index - start,
1374 });
1375 fold += 1;
1376 }
1377 display
1378}
1379
1380fn stable_text_hash(text: &str) -> u64 {
1381 const OFFSET: u64 = 0xcbf2_9ce4_8422_2325;
1382 const PRIME: u64 = 0x0000_0100_0000_01b3;
1383 text.bytes().fold(OFFSET, |hash, byte| {
1384 (hash ^ u64::from(byte)).wrapping_mul(PRIME)
1385 })
1386}
1387
1388fn validate_identity(value: &str) -> Result<(), DocumentError> {
1389 if value.is_empty()
1390 || value.len() > 256
1391 || !value.chars().all(|character| {
1392 character.is_ascii_alphanumeric() || matches!(character, '_' | '-' | '.' | '/' | ':')
1393 })
1394 {
1395 Err(DocumentError::InvalidIdentity(value.to_owned()))
1396 } else {
1397 Ok(())
1398 }
1399}
1400
1401#[derive(Clone, Debug, Error, Eq, PartialEq)]
1402pub enum DocumentError {
1403 #[error("document identity `{0}` must contain 1-256 safe ASCII characters")]
1404 InvalidIdentity(String),
1405 #[error("native text document `{0}` is already registered")]
1406 DuplicateDocument(String),
1407 #[error("native text document `{0}` is not registered")]
1408 UnknownDocument(String),
1409 #[error(
1410 "native text document `{identity}` revision must increase: current {current}, next {next}"
1411 )]
1412 NonMonotonicRevision {
1413 identity: String,
1414 current: u64,
1415 next: u64,
1416 },
1417 #[error("document contains {actual} bytes, exceeding the {limit}-byte budget")]
1418 TooManyBytes { actual: usize, limit: usize },
1419 #[error("diff inputs contain {actual} bytes, exceeding the {limit}-byte budget")]
1420 TooManyDiffBytes { actual: usize, limit: usize },
1421 #[error("document contains {actual} lines, exceeding the {limit}-line budget")]
1422 TooManyLines { actual: usize, limit: usize },
1423 #[error("diff contains {actual} hunks, exceeding the {limit}-hunk budget")]
1424 TooManyHunks { actual: usize, limit: usize },
1425 #[error("syntax definition or parse failed: {0}")]
1426 Syntax(String),
1427 #[error("syntax registry lock was poisoned")]
1428 SyntaxRegistryPoisoned,
1429 #[error("document resource limits must all be greater than zero")]
1430 InvalidLimits,
1431 #[error("document runtime configuration lock was poisoned")]
1432 DocumentConfigPoisoned,
1433}
1434
1435#[cfg(test)]
1436mod tests {
1437 use super::*;
1438
1439 fn descriptor(text: &str, language: &str) -> DocumentDescriptor {
1440 DocumentDescriptor {
1441 source: DocumentSource::from(text.to_owned()),
1442 label: language.to_owned(),
1443 file_name: None,
1444 language: Some(language.to_owned()),
1445 }
1446 }
1447
1448 #[test]
1449 fn native_documents_are_revisioned_and_registry_reads_are_exact() {
1450 let first = NativeTextDocument::new("server/a", 1, Arc::<str>::from("port=80\n")).unwrap();
1451 let second =
1452 NativeTextDocument::new("server/a", 2, Arc::<str>::from("port=443\n")).unwrap();
1453 let reader = ComponentInstancePath::root("View", "main");
1454 let mut registry = NativeTextDocumentRegistry::new();
1455 registry.register("config", first).unwrap();
1456 assert_eq!(
1457 registry.read_tracked(&reader, "config").unwrap().revision(),
1458 1
1459 );
1460 assert_eq!(
1461 registry.replace("config", second).unwrap(),
1462 BTreeSet::from([reader])
1463 );
1464 let stale = NativeTextDocument::new("server/a", 2, Arc::<str>::from("stale")).unwrap();
1465 assert!(matches!(
1466 registry.replace("config", stale),
1467 Err(DocumentError::NonMonotonicRevision { .. })
1468 ));
1469 }
1470
1471 #[test]
1472 fn rhai_and_go_are_built_in_and_unknown_languages_fall_back_to_plain_text() {
1473 let registry = SyntaxRegistry::new();
1474 let rhai = prepare_document(&descriptor("fn view() { 42 }\n", "rhai"), ®istry).unwrap();
1475 assert_eq!(rhai.language(), "Rhai");
1476 assert!(
1477 rhai.lines()[0]
1478 .syntax
1479 .iter()
1480 .any(|span| span.kind == SyntaxTokenKind::Keyword),
1481 "{:?}",
1482 rhai.lines()[0].syntax
1483 );
1484
1485 let go = prepare_document(
1486 &descriptor("package main\nfunc main() {}\n", "go"),
1487 ®istry,
1488 )
1489 .unwrap();
1490 assert_eq!(go.language(), "Go");
1491 let plain = prepare_document(&descriptor("hello\n", "not-a-language"), ®istry).unwrap();
1492 assert_eq!(plain.language(), "Plain Text");
1493 }
1494
1495 #[test]
1496 fn public_launch_language_pack_resolves_every_declared_language() {
1497 let registry = SyntaxRegistry::new();
1498 let mut missing = Vec::new();
1499 for language in [
1500 "rhai",
1501 "rust",
1502 "go",
1503 "javascript",
1504 "typescript",
1505 "tsx",
1506 "json",
1507 "jsonc",
1508 "toml",
1509 "yaml",
1510 "markdown",
1511 "bash",
1512 "python",
1513 "html",
1514 "css",
1515 "sql",
1516 "dockerfile",
1517 "RUST",
1518 ] {
1519 let document = prepare_document(&descriptor("value = 1\n", language), ®istry)
1520 .unwrap_or_else(|error| panic!("{language}: {error}"));
1521 if document.language() == "Plain Text" {
1522 missing.push(language);
1523 }
1524 }
1525 assert!(missing.is_empty(), "missing {missing:?}");
1526 }
1527
1528 #[test]
1529 fn line_ranges_normalize_crlf_but_retain_terminal_empty_line() {
1530 let registry = SyntaxRegistry::new();
1531 let document = prepare_document(&descriptor("one\r\ntwo\n", "text"), ®istry).unwrap();
1532 assert_eq!(document.lines().len(), 3);
1533 assert_eq!(document.line_text(0), Some("one"));
1534 assert_eq!(document.line_text(1), Some("two"));
1535 assert_eq!(document.line_text(2), Some(""));
1536 assert!(document.ends_with_newline());
1537 }
1538
1539 #[test]
1540 fn diff_is_direction_neutral_refines_unicode_and_collapses_context() {
1541 let registry = SyntaxRegistry::new();
1542 let left = descriptor(
1543 "same\n城市 = 东京\nunchanged 1\nunchanged 2\nunchanged 3\nunchanged 4\n",
1544 "rhai",
1545 );
1546 let right = descriptor(
1547 "same\n城市 = 上海\nunchanged 1\nunchanged 2\nunchanged 3\nunchanged 4\n",
1548 "rhai",
1549 );
1550 let diff = prepare_diff(&left, &right, DiffWhitespace::Exact, Some(1), ®istry).unwrap();
1551 assert_eq!(diff.hunk_count, 1);
1552 let modified = diff
1553 .rows
1554 .iter()
1555 .find(|row| row.kind == DiffRowKind::Modified)
1556 .unwrap();
1557 assert!(!modified.left_inline.is_empty());
1558 assert!(!modified.right_inline.is_empty());
1559 assert!(
1560 diff.collapsed
1561 .iter()
1562 .any(|row| matches!(row, DiffDisplayRow::Fold { .. }))
1563 );
1564 }
1565
1566 #[test]
1567 fn whitespace_policy_is_explicit() {
1568 let registry = SyntaxRegistry::new();
1569 let left = descriptor("value = 1\n", "rhai");
1570 let right = descriptor("value = 1\n", "rhai");
1571 assert_eq!(
1572 prepare_diff(&left, &right, DiffWhitespace::Exact, None, ®istry)
1573 .unwrap()
1574 .hunk_count,
1575 1
1576 );
1577 assert_eq!(
1578 prepare_diff(
1579 &left,
1580 &right,
1581 DiffWhitespace::IgnoreChanges,
1582 None,
1583 ®istry,
1584 )
1585 .unwrap()
1586 .hunk_count,
1587 0
1588 );
1589 }
1590
1591 #[test]
1592 fn host_limits_fail_explicitly_before_unbounded_work() {
1593 let registry = SyntaxRegistry::new();
1594 let descriptor = descriptor("12345", "text");
1595 let limits = DocumentLimits {
1596 max_document_bytes: 4,
1597 ..DocumentLimits::default()
1598 };
1599 assert!(matches!(
1600 prepare_document_with_limits(&descriptor, ®istry, limits),
1601 Err(DocumentError::TooManyBytes {
1602 actual: 5,
1603 limit: 4
1604 })
1605 ));
1606 let runtime = DocumentRuntimeConfig::new();
1607 assert_eq!(runtime.limits(), DocumentLimits::default());
1608 assert_eq!(
1609 runtime.set_limits(DocumentLimits {
1610 max_diff_hunks: 0,
1611 ..DocumentLimits::default()
1612 }),
1613 Err(DocumentError::InvalidLimits)
1614 );
1615 }
1616
1617 #[test]
1618 fn host_registered_syntax_participates_in_background_safe_snapshots() {
1619 let registry = SyntaxRegistry::new();
1620 registry
1621 .register_sublime_syntax(
1622 r"%YAML 1.2
1623---
1624name: Probe
1625file_extensions: [probe]
1626scope: source.probe
1627contexts:
1628 main:
1629 - match: '\bprobe\b'
1630 scope: keyword.control.probe
1631",
1632 )
1633 .unwrap();
1634 let document = prepare_document(&descriptor("probe value\n", "probe"), ®istry).unwrap();
1635 assert_eq!(document.language(), "Probe");
1636 assert!(
1637 document.lines()[0]
1638 .syntax
1639 .iter()
1640 .any(|span| span.kind == SyntaxTokenKind::Keyword)
1641 );
1642 }
1643}