1use std::borrow::Cow;
4use std::collections::{BTreeMap, BTreeSet};
5use std::fmt;
6use std::ops::Range;
7use std::str::FromStr as _;
8use std::sync::{Arc, OnceLock, RwLock};
9use std::time::{Duration, Instant};
10
11use rhai::{CustomType, Dynamic, Engine, FuncRegistration, ImmutableString, TypeBuilder};
12use similar::{Algorithm, DiffOp, capture_diff_slices_deadline};
13use syntect::easy::ScopeRangeIterator;
14use syntect::highlighting::ScopeSelectors;
15use syntect::parsing::{ParseState, ScopeStack, SyntaxDefinition, SyntaxSet};
16use thiserror::Error;
17use unicode_segmentation::UnicodeSegmentation as _;
18
19use crate::ComponentInstancePath;
20
21pub const DEFAULT_DOCUMENT_MAX_BYTES: usize = 10 * 1024 * 1024;
22pub const DEFAULT_DOCUMENT_MAX_LINES: usize = 500_000;
23pub const DEFAULT_DIFF_MAX_HUNKS: usize = 100_000;
24pub const DEFAULT_DIFF_TIMEOUT: Duration = Duration::from_secs(2);
25const MAX_INLINE_DIFF_BYTES: usize = 256 * 1024;
26
27#[derive(Clone, Copy, Debug, Eq, PartialEq)]
28pub struct DocumentLimits {
29 pub max_document_bytes: usize,
30 pub max_document_lines: usize,
31 pub max_diff_total_bytes: usize,
32 pub max_diff_hunks: usize,
33 pub diff_timeout: Duration,
34}
35
36impl Default for DocumentLimits {
37 fn default() -> Self {
38 Self {
39 max_document_bytes: DEFAULT_DOCUMENT_MAX_BYTES,
40 max_document_lines: DEFAULT_DOCUMENT_MAX_LINES,
41 max_diff_total_bytes: DEFAULT_DOCUMENT_MAX_BYTES * 2,
42 max_diff_hunks: DEFAULT_DIFF_MAX_HUNKS,
43 diff_timeout: DEFAULT_DIFF_TIMEOUT,
44 }
45 }
46}
47
48impl DocumentLimits {
49 pub fn validate(self) -> Result<(), DocumentError> {
55 if self.max_document_bytes == 0
56 || self.max_document_lines == 0
57 || self.max_diff_total_bytes == 0
58 || self.max_diff_hunks == 0
59 || self.diff_timeout.is_zero()
60 {
61 Err(DocumentError::InvalidLimits)
62 } else {
63 Ok(())
64 }
65 }
66}
67
68#[derive(Clone, Debug)]
69pub struct DocumentRuntimeConfig {
70 limits: Arc<RwLock<DocumentLimits>>,
71}
72
73impl Default for DocumentRuntimeConfig {
74 fn default() -> Self {
75 Self::new()
76 }
77}
78
79impl DocumentRuntimeConfig {
80 #[must_use]
81 pub fn new() -> Self {
82 Self {
83 limits: Arc::new(RwLock::new(DocumentLimits::default())),
84 }
85 }
86
87 #[must_use]
88 pub fn limits(&self) -> DocumentLimits {
89 self.limits
90 .read()
91 .map_or_else(|poisoned| *poisoned.into_inner(), |limits| *limits)
92 }
93
94 pub fn set_limits(&self, limits: DocumentLimits) -> Result<(), DocumentError> {
100 limits.validate()?;
101 *self
102 .limits
103 .write()
104 .map_err(|_| DocumentError::DocumentConfigPoisoned)? = limits;
105 Ok(())
106 }
107}
108
109const RHAI_SYNTAX: &str = r#"%YAML 1.2
110---
111name: Rhai
112file_extensions: [rhai]
113scope: source.rhai
114contexts:
115 main:
116 - include: comments
117 - match: '\b(fn|let|const|if|else|switch|for|in|while|loop|do|until|break|continue|return|throw|try|catch|import|export|as|private)\b'
118 scope: keyword.control.rhai
119 - match: '\b(true|false)\b'
120 scope: constant.language.rhai
121 - match: '\b[0-9]+(?:\.[0-9]+)?\b'
122 scope: constant.numeric.rhai
123 - match: '"'
124 push: double-quoted-string
125 - match: "'"
126 push: single-quoted-string
127 - match: '`'
128 push: template-string
129 - match: '\b[A-Za-z_][A-Za-z0-9_]*(?=\s*\()'
130 scope: entity.name.function.rhai
131 - match: '[+\-*/%!=<>|&^~?:]+'
132 scope: keyword.operator.rhai
133 comments:
134 - match: '//.*$'
135 scope: comment.line.double-slash.rhai
136 - match: '/\*'
137 scope: punctuation.definition.comment.begin.rhai
138 push:
139 - meta_scope: comment.block.rhai
140 - match: '\*/'
141 scope: punctuation.definition.comment.end.rhai
142 pop: true
143 double-quoted-string:
144 - meta_scope: string.quoted.double.rhai
145 - match: '\\.'
146 scope: constant.character.escape.rhai
147 - match: '"'
148 pop: true
149 single-quoted-string:
150 - meta_scope: string.quoted.single.rhai
151 - match: '\\.'
152 scope: constant.character.escape.rhai
153 - match: "'"
154 pop: true
155 template-string:
156 - meta_scope: string.quoted.other.rhai
157 - match: '\\.'
158 scope: constant.character.escape.rhai
159 - match: '`'
160 pop: true
161"#;
162
163const TYPESCRIPT_SYNTAX: &str = r#"%YAML 1.2
164---
165name: TypeScript
166file_extensions: [ts, tsx]
167scope: source.ts
168contexts:
169 main:
170 - match: '//.*$'
171 scope: comment.line.double-slash.ts
172 - match: '/\*'
173 push:
174 - meta_scope: comment.block.ts
175 - match: '\*/'
176 pop: true
177 - match: '\b(async|await|break|case|catch|class|const|continue|debugger|default|delete|do|else|enum|export|extends|finally|for|from|function|get|if|implements|import|in|instanceof|interface|keyof|let|namespace|new|of|private|protected|public|readonly|return|set|static|super|switch|throw|try|type|typeof|var|void|while|with|yield)\b'
178 scope: keyword.control.ts
179 - match: '\b(any|bigint|boolean|never|number|object|string|symbol|unknown|undefined|null|true|false)\b'
180 scope: storage.type.ts
181 - match: '\b[0-9]+(?:\.[0-9]+)?\b'
182 scope: constant.numeric.ts
183 - match: '"'
184 push: double-string
185 - match: "'"
186 push: single-string
187 - match: '`'
188 push: template-string
189 - match: '\b[A-Za-z_$][A-Za-z0-9_$]*(?=\s*\()'
190 scope: entity.name.function.ts
191 - match: '</?[A-Za-z][A-Za-z0-9:.-]*'
192 scope: entity.name.tag.tsx
193 - match: '[+\-*/%!=<>|&^~?:]+'
194 scope: keyword.operator.ts
195 double-string:
196 - meta_scope: string.quoted.double.ts
197 - match: '\\.'
198 scope: constant.character.escape.ts
199 - match: '"'
200 pop: true
201 single-string:
202 - meta_scope: string.quoted.single.ts
203 - match: '\\.'
204 scope: constant.character.escape.ts
205 - match: "'"
206 pop: true
207 template-string:
208 - meta_scope: string.quoted.other.ts
209 - match: '\\.'
210 scope: constant.character.escape.ts
211 - match: '`'
212 pop: true
213"#;
214
215const JSONC_SYNTAX: &str = r#"%YAML 1.2
216---
217name: JSON with Comments
218file_extensions: [jsonc]
219scope: source.json.comments
220contexts:
221 main:
222 - match: '//.*$'
223 scope: comment.line.double-slash.json
224 - match: '/\*'
225 push:
226 - meta_scope: comment.block.json
227 - match: '\*/'
228 pop: true
229 - match: '"'
230 push: string
231 - match: '-?\b[0-9]+(?:\.[0-9]+)?(?:[eE][+-]?[0-9]+)?\b'
232 scope: constant.numeric.json
233 - match: '\b(true|false|null)\b'
234 scope: constant.language.json
235 - match: '[{}\[\],:]'
236 scope: punctuation.separator.json
237 string:
238 - meta_scope: string.quoted.double.json
239 - match: '\\(?:["\\/bfnrt]|u[0-9A-Fa-f]{4})'
240 scope: constant.character.escape.json
241 - match: '"'
242 pop: true
243"#;
244
245const TOML_SYNTAX: &str = r#"%YAML 1.2
246---
247name: TOML
248file_extensions: [toml]
249scope: source.toml
250contexts:
251 main:
252 - match: '#.*$'
253 scope: comment.line.number-sign.toml
254 - match: '^\s*\[\[?[^\]]+\]\]?'
255 scope: entity.name.section.toml
256 - match: '^[A-Za-z0-9_.-]+(?=\s*=)'
257 scope: variable.other.key.toml
258 - match: '"""'
259 push: multiline-string
260 - match: "'''"
261 push: multiline-literal
262 - match: '"'
263 push: string
264 - match: "'"
265 push: literal
266 - match: '\b(true|false)\b'
267 scope: constant.language.toml
268 - match: '[+-]?\b(?:0x[0-9A-Fa-f_]+|0o[0-7_]+|0b[01_]+|[0-9][0-9_]*(?:\.[0-9_]+)?)\b'
269 scope: constant.numeric.toml
270 - match: '[=,.{}\[\]]'
271 scope: punctuation.separator.toml
272 string:
273 - meta_scope: string.quoted.double.toml
274 - match: '\\.'
275 scope: constant.character.escape.toml
276 - match: '"'
277 pop: true
278 literal:
279 - meta_scope: string.quoted.single.toml
280 - match: "'"
281 pop: true
282 multiline-string:
283 - meta_scope: string.quoted.triple.toml
284 - match: '"""'
285 pop: true
286 multiline-literal:
287 - meta_scope: string.quoted.triple.toml
288 - match: "'''"
289 pop: true
290"#;
291
292const DOCKERFILE_SYNTAX: &str = r#"%YAML 1.2
293---
294name: Dockerfile
295file_extensions: [Dockerfile, dockerfile]
296scope: source.dockerfile
297contexts:
298 main:
299 - match: '#.*$'
300 scope: comment.line.number-sign.dockerfile
301 - match: '(?i)^\s*(ADD|ARG|CMD|COPY|ENTRYPOINT|ENV|EXPOSE|FROM|HEALTHCHECK|LABEL|MAINTAINER|ONBUILD|RUN|SHELL|STOPSIGNAL|USER|VOLUME|WORKDIR)(?=\s)'
302 scope: keyword.control.dockerfile
303 - match: '\$\{?[A-Za-z_][A-Za-z0-9_]*\}?'
304 scope: variable.other.dockerfile
305 - match: '"'
306 push: double-string
307 - match: "'"
308 push: single-string
309 double-string:
310 - meta_scope: string.quoted.double.dockerfile
311 - match: '\\.'
312 scope: constant.character.escape.dockerfile
313 - match: '"'
314 pop: true
315 single-string:
316 - meta_scope: string.quoted.single.dockerfile
317 - match: "'"
318 pop: true
319"#;
320
321const EXTRA_SYNTAXES: &[&str] = &[
322 RHAI_SYNTAX,
323 TYPESCRIPT_SYNTAX,
324 JSONC_SYNTAX,
325 TOML_SYNTAX,
326 DOCKERFILE_SYNTAX,
327];
328
329#[derive(Clone)]
330pub struct NativeTextDocument {
331 snapshot: Arc<NativeTextSnapshot>,
332}
333
334#[derive(Debug, Eq, PartialEq)]
335struct NativeTextSnapshot {
336 identity: String,
337 revision: u64,
338 text: Arc<str>,
339}
340
341impl NativeTextDocument {
342 pub fn new(
348 identity: impl Into<String>,
349 revision: u64,
350 text: impl Into<Arc<str>>,
351 ) -> Result<Self, DocumentError> {
352 let identity = identity.into();
353 validate_identity(&identity)?;
354 Ok(Self {
355 snapshot: Arc::new(NativeTextSnapshot {
356 identity,
357 revision,
358 text: text.into(),
359 }),
360 })
361 }
362
363 #[must_use]
364 pub fn identity(&self) -> &str {
365 &self.snapshot.identity
366 }
367
368 #[must_use]
369 pub fn revision(&self) -> u64 {
370 self.snapshot.revision
371 }
372
373 #[must_use]
374 pub fn text(&self) -> &str {
375 &self.snapshot.text
376 }
377
378 #[must_use]
379 pub fn text_arc(&self) -> Arc<str> {
380 Arc::clone(&self.snapshot.text)
381 }
382}
383
384impl PartialEq for NativeTextDocument {
385 fn eq(&self, other: &Self) -> bool {
386 Arc::ptr_eq(&self.snapshot, &other.snapshot) || self.snapshot == other.snapshot
387 }
388}
389
390impl Eq for NativeTextDocument {}
391
392impl fmt::Debug for NativeTextDocument {
393 fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
394 formatter
395 .debug_struct("NativeTextDocument")
396 .field("identity", &self.identity())
397 .field("revision", &self.revision())
398 .field("bytes", &self.text().len())
399 .finish()
400 }
401}
402
403impl CustomType for NativeTextDocument {
404 fn build(mut builder: TypeBuilder<Self>) {
405 builder
406 .with_name("NativeTextDocument")
407 .with_get("identity", |document: &mut Self| {
408 ImmutableString::from(document.identity().to_owned())
409 })
410 .with_get("revision", |document: &mut Self| {
411 i64::try_from(document.revision()).unwrap_or(i64::MAX)
412 })
413 .with_get("len", |document: &mut Self| {
414 i64::try_from(document.text().len()).unwrap_or(i64::MAX)
415 })
416 .with_fn("to_string", |document: &mut Self| {
417 format!(
418 "NativeTextDocument({}, revision={})",
419 document.identity(),
420 document.revision()
421 )
422 });
423 }
424}
425
426pub(crate) fn register_document_api(engine: &mut Engine) {
427 engine.build_type::<NativeTextDocument>();
428 FuncRegistration::new("is_native_text_document")
429 .in_global_namespace()
430 .register_into_engine(engine, |value: Dynamic| value.is::<NativeTextDocument>());
431}
432
433#[derive(Clone, Debug, Default)]
434pub struct NativeTextDocumentRegistry {
435 documents: BTreeMap<String, NativeTextDocument>,
436 readers: BTreeMap<String, BTreeSet<crate::read_dependency::ReadDependency>>,
437}
438
439impl NativeTextDocumentRegistry {
440 #[must_use]
441 pub fn new() -> Self {
442 Self::default()
443 }
444
445 pub fn register(
451 &mut self,
452 name: impl Into<String>,
453 document: NativeTextDocument,
454 ) -> Result<(), DocumentError> {
455 let name = name.into();
456 validate_identity(&name)?;
457 if self.documents.contains_key(&name) {
458 return Err(DocumentError::DuplicateDocument(name));
459 }
460 self.documents.insert(name, document);
461 Ok(())
462 }
463
464 pub fn get(&self, name: &str) -> Result<NativeTextDocument, DocumentError> {
473 self.documents
474 .get(name)
475 .cloned()
476 .ok_or_else(|| DocumentError::UnknownDocument(name.to_owned()))
477 }
478
479 pub fn replace(
485 &mut self,
486 name: &str,
487 document: NativeTextDocument,
488 ) -> Result<BTreeSet<ComponentInstancePath>, DocumentError> {
489 let current = self
490 .documents
491 .get_mut(name)
492 .ok_or_else(|| DocumentError::UnknownDocument(name.to_owned()))?;
493 if current == &document {
494 return Ok(BTreeSet::new());
495 }
496 if current.identity() == document.identity() && document.revision() <= current.revision() {
497 return Err(DocumentError::NonMonotonicRevision {
498 identity: document.identity().to_owned(),
499 current: current.revision(),
500 next: document.revision(),
501 });
502 }
503 *current = document;
504 Ok(crate::read_dependency::owners(
505 self.readers.get(name).cloned().unwrap_or_default(),
506 ))
507 }
508
509 #[cfg(test)]
510 pub(crate) fn read_tracked(
511 &mut self,
512 reader: &ComponentInstancePath,
513 name: &str,
514 ) -> Result<NativeTextDocument, DocumentError> {
515 self.read_dependency(
516 &crate::read_dependency::ReadDependency::component(reader),
517 name,
518 )
519 }
520
521 pub(crate) fn read_dependency(
522 &mut self,
523 reader: &crate::read_dependency::ReadDependency,
524 name: &str,
525 ) -> Result<NativeTextDocument, DocumentError> {
526 let document = self
527 .documents
528 .get(name)
529 .cloned()
530 .ok_or_else(|| DocumentError::UnknownDocument(name.to_owned()))?;
531 self.readers
532 .entry(name.to_owned())
533 .or_default()
534 .insert(reader.clone());
535 Ok(document)
536 }
537
538 pub(crate) fn reset_reader(&mut self, reader: &ComponentInstancePath) {
539 self.reset_contribution(&crate::read_dependency::ReadDependency::component(reader));
540 }
541 pub(crate) fn reset_contribution(&mut self, reader: &crate::read_dependency::ReadDependency) {
542 for readers in self.readers.values_mut() {
543 readers.remove(reader);
544 }
545 self.readers.retain(|_, readers| !readers.is_empty());
546 }
547
548 pub(crate) fn retain_reader_scope(
549 &mut self,
550 root: &ComponentInstancePath,
551 active: &BTreeSet<ComponentInstancePath>,
552 ) {
553 for readers in self.readers.values_mut() {
554 readers.retain(|reader| reader.retained_in_owner_scope(root, active));
555 }
556 self.readers.retain(|_, readers| !readers.is_empty());
557 }
558
559 pub(crate) fn remove_reader_scope(&mut self, root: &ComponentInstancePath) {
560 for readers in self.readers.values_mut() {
561 readers.retain(|reader| !reader.owner.is_within(root));
562 }
563 self.readers.retain(|_, readers| !readers.is_empty());
564 }
565
566 pub(crate) fn retain_contributions(
567 &mut self,
568 scope: &ComponentInstancePath,
569 active: &BTreeSet<crate::read_dependency::ReadContribution>,
570 ) {
571 crate::read_dependency::retain_readers(&mut self.readers, |reader| {
572 reader.retained_in_contribution_scope(scope, active)
573 });
574 }
575}
576
577#[derive(Clone)]
578pub struct SyntaxRegistry {
579 syntax_set: Arc<RwLock<Arc<SyntaxSet>>>,
580}
581
582impl fmt::Debug for SyntaxRegistry {
583 fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
584 formatter
585 .debug_struct("SyntaxRegistry")
586 .field("syntaxes", &self.snapshot().syntaxes().len())
587 .finish()
588 }
589}
590
591impl Default for SyntaxRegistry {
592 fn default() -> Self {
593 Self::new()
594 }
595}
596
597impl SyntaxRegistry {
598 #[must_use]
599 pub fn new() -> Self {
600 Self {
601 syntax_set: Arc::new(RwLock::new(Arc::clone(default_syntax_set()))),
602 }
603 }
604
605 pub fn register_sublime_syntax(&self, source: &str) -> Result<(), DocumentError> {
611 let syntax = SyntaxDefinition::load_from_str(source, true, None)
612 .map_err(|error| DocumentError::Syntax(error.to_string()))?;
613 let current = self
614 .syntax_set
615 .read()
616 .map_err(|_| DocumentError::SyntaxRegistryPoisoned)?
617 .clone();
618 let mut builder = current.as_ref().clone().into_builder();
619 builder.add(syntax);
620 let next = Arc::new(builder.build());
621 *self
622 .syntax_set
623 .write()
624 .map_err(|_| DocumentError::SyntaxRegistryPoisoned)? = next;
625 Ok(())
626 }
627
628 #[must_use]
629 pub fn snapshot(&self) -> Arc<SyntaxSet> {
630 self.syntax_set
631 .read()
632 .map_or_else(|poisoned| poisoned.into_inner().clone(), |set| set.clone())
633 }
634}
635
636fn default_syntax_set() -> &'static Arc<SyntaxSet> {
637 static SYNTAXES: OnceLock<Arc<SyntaxSet>> = OnceLock::new();
638 SYNTAXES.get_or_init(|| {
639 let mut builder = SyntaxSet::load_defaults_newlines().into_builder();
640 for source in EXTRA_SYNTAXES {
641 builder.add(
642 SyntaxDefinition::load_from_str(source, true, None)
643 .expect("built-in syntax is valid"),
644 );
645 }
646 Arc::new(builder.build())
647 })
648}
649
650#[derive(Clone, Debug, Eq, PartialEq)]
651pub enum DocumentSource {
652 Inline(Arc<str>),
653 Native(NativeTextDocument),
654}
655
656impl DocumentSource {
657 #[must_use]
658 pub fn identity(&self) -> Cow<'_, str> {
659 match self {
660 Self::Inline(_) => Cow::Borrowed("inline"),
661 Self::Native(document) => Cow::Borrowed(document.identity()),
662 }
663 }
664
665 #[must_use]
666 pub fn revision(&self) -> u64 {
667 match self {
668 Self::Inline(text) => stable_text_hash(text),
669 Self::Native(document) => document.revision(),
670 }
671 }
672
673 #[must_use]
674 pub fn text(&self) -> Arc<str> {
675 match self {
676 Self::Inline(text) => Arc::clone(text),
677 Self::Native(document) => document.text_arc(),
678 }
679 }
680}
681
682impl From<String> for DocumentSource {
683 fn from(value: String) -> Self {
684 Self::Inline(Arc::from(value))
685 }
686}
687
688impl From<NativeTextDocument> for DocumentSource {
689 fn from(value: NativeTextDocument) -> Self {
690 Self::Native(value)
691 }
692}
693
694#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
695pub enum DocumentWrap {
696 #[default]
697 None,
698 Viewport,
699 Column(usize),
700}
701
702#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
703pub enum DiffViewMode {
704 #[default]
705 Unified,
706 Split,
707}
708
709#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
710pub enum DiffWhitespace {
711 #[default]
712 Exact,
713 IgnoreChanges,
714 IgnoreAll,
715}
716
717#[derive(Clone, Debug, Eq, PartialEq)]
718pub struct DocumentDescriptor {
719 pub source: DocumentSource,
720 pub label: String,
721 pub file_name: Option<String>,
722 pub language: Option<String>,
723}
724
725#[derive(Clone, Copy, Debug, Eq, PartialEq)]
726pub enum SyntaxTokenKind {
727 Comment,
728 String,
729 Number,
730 Keyword,
731 Function,
732 Type,
733 Variable,
734 Constant,
735 Operator,
736 Punctuation,
737 Tag,
738 Attribute,
739}
740
741impl SyntaxTokenKind {
742 #[must_use]
743 pub const fn theme_token(self) -> &'static str {
744 match self {
745 Self::Comment => "syntax.comment",
746 Self::String => "syntax.string",
747 Self::Number => "syntax.number",
748 Self::Keyword => "syntax.keyword",
749 Self::Function => "syntax.function",
750 Self::Type => "syntax.type",
751 Self::Variable => "syntax.variable",
752 Self::Constant => "syntax.constant",
753 Self::Operator => "syntax.operator",
754 Self::Punctuation => "syntax.punctuation",
755 Self::Tag => "syntax.tag",
756 Self::Attribute => "syntax.attribute",
757 }
758 }
759}
760
761#[derive(Clone, Debug, Eq, PartialEq)]
762pub struct SyntaxSpan {
763 pub range: Range<usize>,
764 pub kind: SyntaxTokenKind,
765}
766
767#[derive(Clone, Debug, Eq, PartialEq)]
768pub struct DocumentLine {
769 pub range: Range<usize>,
770 pub syntax: Vec<SyntaxSpan>,
771}
772
773#[derive(Clone, Debug)]
774pub struct PreparedDocument {
775 identity: String,
776 revision: u64,
777 text: Arc<str>,
778 lines: Arc<[DocumentLine]>,
779 language: String,
780 max_line_chars: usize,
781 ends_with_newline: bool,
782 highlight_error: Option<String>,
783}
784
785impl PreparedDocument {
786 #[must_use]
787 pub fn identity(&self) -> &str {
788 &self.identity
789 }
790
791 #[must_use]
792 pub const fn revision(&self) -> u64 {
793 self.revision
794 }
795
796 #[must_use]
797 pub fn text(&self) -> &str {
798 &self.text
799 }
800
801 #[must_use]
802 pub fn text_arc(&self) -> Arc<str> {
803 Arc::clone(&self.text)
804 }
805
806 #[must_use]
807 pub fn lines(&self) -> &[DocumentLine] {
808 &self.lines
809 }
810
811 #[must_use]
812 pub fn line_text(&self, index: usize) -> Option<&str> {
813 let line = self.lines.get(index)?;
814 self.text.get(line.range.clone())
815 }
816
817 #[must_use]
818 pub fn language(&self) -> &str {
819 &self.language
820 }
821
822 #[must_use]
823 pub const fn max_line_chars(&self) -> usize {
824 self.max_line_chars
825 }
826
827 #[must_use]
828 pub const fn ends_with_newline(&self) -> bool {
829 self.ends_with_newline
830 }
831
832 #[must_use]
833 pub fn highlight_error(&self) -> Option<&str> {
834 self.highlight_error.as_deref()
835 }
836}
837
838pub fn prepare_document(
845 descriptor: &DocumentDescriptor,
846 syntaxes: &SyntaxRegistry,
847) -> Result<PreparedDocument, DocumentError> {
848 prepare_document_with_limits(descriptor, syntaxes, DocumentLimits::default())
849}
850
851pub fn prepare_document_with_limits(
857 descriptor: &DocumentDescriptor,
858 syntaxes: &SyntaxRegistry,
859 limits: DocumentLimits,
860) -> Result<PreparedDocument, DocumentError> {
861 let text = descriptor.source.text();
862 if text.len() > limits.max_document_bytes {
863 return Err(DocumentError::TooManyBytes {
864 actual: text.len(),
865 limit: limits.max_document_bytes,
866 });
867 }
868 let ranges = document_line_ranges(&text);
869 if ranges.len() > limits.max_document_lines {
870 return Err(DocumentError::TooManyLines {
871 actual: ranges.len(),
872 limit: limits.max_document_lines,
873 });
874 }
875 let syntax_set = syntaxes.snapshot();
876 let syntax = resolve_syntax(
877 &syntax_set,
878 descriptor.language.as_deref(),
879 descriptor.file_name.as_deref(),
880 );
881 let language = syntax.name.clone();
882 let mut parser = ParseState::new(syntax);
883 let mut stack = ScopeStack::new();
884 let selectors = syntax_selectors();
885 let mut lines = Vec::with_capacity(ranges.len());
886 let mut max_line_chars = 0usize;
887 let mut highlight_error = None;
888 let mut highlighting = true;
889 for range in ranges {
890 let line = &text[range.clone()];
891 max_line_chars = max_line_chars.max(line.chars().count());
892 let parse_end = text[range.end..]
893 .find('\n')
894 .map_or(range.end, |offset| range.end + offset + 1);
895 let parse_start = range.start;
896 let parse_line = &text[parse_start..parse_end];
897 let mut spans: Vec<SyntaxSpan> = Vec::new();
898 if highlighting {
899 match parser.parse_line(parse_line, &syntax_set) {
900 Ok(operations) => {
901 for (token_range, operation) in ScopeRangeIterator::new(&operations, parse_line)
902 {
903 if let Err(error) = stack.apply(operation) {
904 highlight_error.get_or_insert_with(|| error.to_string());
905 highlighting = false;
906 spans.clear();
907 break;
908 }
909 let start = token_range.start.min(line.len());
910 let end = token_range.end.min(line.len());
911 if start >= end {
912 continue;
913 }
914 if let Some(kind) = selectors.kind(stack.as_slice()) {
915 if let Some(previous) = spans.last_mut()
916 && previous.kind == kind
917 && previous.range.end == start
918 {
919 previous.range.end = end;
920 } else {
921 spans.push(SyntaxSpan {
922 range: start..end,
923 kind,
924 });
925 }
926 }
927 }
928 }
929 Err(error) => {
930 highlight_error.get_or_insert_with(|| error.to_string());
931 highlighting = false;
932 }
933 }
934 }
935 lines.push(DocumentLine {
936 range,
937 syntax: spans,
938 });
939 }
940 Ok(PreparedDocument {
941 identity: descriptor.source.identity().into_owned(),
942 revision: descriptor.source.revision(),
943 text,
944 lines: lines.into(),
945 language,
946 max_line_chars,
947 ends_with_newline: descriptor.source.text().ends_with('\n'),
948 highlight_error,
949 })
950}
951
952fn document_line_ranges(text: &str) -> Vec<Range<usize>> {
953 let mut ranges = Vec::new();
954 let mut start = 0usize;
955 for (index, byte) in text.bytes().enumerate() {
956 if byte == b'\n' {
957 let mut end = index;
958 if text.as_bytes().get(index.wrapping_sub(1)) == Some(&b'\r') && index > start {
959 end -= 1;
960 }
961 ranges.push(start..end);
962 start = index + 1;
963 }
964 }
965 if start < text.len() || text.is_empty() || text.ends_with('\n') {
966 ranges.push(start..text.len());
967 }
968 ranges
969}
970
971fn resolve_syntax<'a>(
972 syntaxes: &'a SyntaxSet,
973 language: Option<&str>,
974 file_name: Option<&str>,
975) -> &'a syntect::parsing::SyntaxReference {
976 language
977 .and_then(|language| {
978 let token = language_alias(language);
979 syntaxes.find_syntax_by_token(token.as_ref())
980 })
981 .or_else(|| {
982 file_name.and_then(|file_name| {
983 syntaxes.find_syntax_by_extension(file_name).or_else(|| {
984 file_name
985 .rsplit_once('.')
986 .and_then(|(_, ext)| syntaxes.find_syntax_by_extension(ext))
987 })
988 })
989 })
990 .unwrap_or_else(|| syntaxes.find_syntax_plain_text())
991}
992
993fn language_alias(language: &str) -> Cow<'_, str> {
994 let normalized = language.trim().to_ascii_lowercase();
995 match normalized.as_str() {
996 "plaintext" | "plain_text" | "text" => Cow::Borrowed("txt"),
997 "javascript" => Cow::Borrowed("js"),
998 "typescript" => Cow::Borrowed("ts"),
999 "shell" | "bash" => Cow::Borrowed("sh"),
1000 "python" => Cow::Borrowed("py"),
1001 "markdown" => Cow::Borrowed("md"),
1002 "rust" => Cow::Borrowed("rs"),
1003 "golang" => Cow::Borrowed("go"),
1004 _ => Cow::Owned(normalized),
1005 }
1006}
1007
1008struct SyntaxSelectors {
1009 comment: ScopeSelectors,
1010 string: ScopeSelectors,
1011 number: ScopeSelectors,
1012 keyword: ScopeSelectors,
1013 function: ScopeSelectors,
1014 type_name: ScopeSelectors,
1015 variable: ScopeSelectors,
1016 constant: ScopeSelectors,
1017 operator: ScopeSelectors,
1018 punctuation: ScopeSelectors,
1019 tag: ScopeSelectors,
1020 attribute: ScopeSelectors,
1021}
1022
1023impl SyntaxSelectors {
1024 fn new() -> Self {
1025 let parse = |value: &str| ScopeSelectors::from_str(value).expect("static scope selector");
1026 Self {
1027 comment: parse("comment"),
1028 string: parse("string"),
1029 number: parse("constant.numeric"),
1030 keyword: parse("keyword, storage.modifier"),
1031 function: parse("entity.name.function, support.function"),
1032 type_name: parse("entity.name.type, support.type, storage.type"),
1033 variable: parse("variable"),
1034 constant: parse("constant"),
1035 operator: parse("keyword.operator"),
1036 punctuation: parse("punctuation"),
1037 tag: parse("entity.name.tag"),
1038 attribute: parse("entity.other.attribute-name"),
1039 }
1040 }
1041
1042 fn kind(&self, stack: &[syntect::parsing::Scope]) -> Option<SyntaxTokenKind> {
1043 [
1044 (&self.comment, SyntaxTokenKind::Comment),
1045 (&self.string, SyntaxTokenKind::String),
1046 (&self.number, SyntaxTokenKind::Number),
1047 (&self.function, SyntaxTokenKind::Function),
1048 (&self.type_name, SyntaxTokenKind::Type),
1049 (&self.tag, SyntaxTokenKind::Tag),
1050 (&self.attribute, SyntaxTokenKind::Attribute),
1051 (&self.operator, SyntaxTokenKind::Operator),
1052 (&self.keyword, SyntaxTokenKind::Keyword),
1053 (&self.variable, SyntaxTokenKind::Variable),
1054 (&self.constant, SyntaxTokenKind::Constant),
1055 (&self.punctuation, SyntaxTokenKind::Punctuation),
1056 ]
1057 .into_iter()
1058 .find_map(|(selector, kind)| selector.does_match(stack).map(|_| kind))
1059 }
1060}
1061
1062fn syntax_selectors() -> &'static SyntaxSelectors {
1063 static SELECTORS: OnceLock<SyntaxSelectors> = OnceLock::new();
1064 SELECTORS.get_or_init(SyntaxSelectors::new)
1065}
1066
1067#[derive(Clone, Copy, Debug, Eq, PartialEq)]
1068pub enum DiffRowKind {
1069 Equal,
1070 LeftOnly,
1071 RightOnly,
1072 Modified,
1073}
1074
1075#[derive(Clone, Debug, Eq, PartialEq)]
1076pub struct DiffAlignedRow {
1077 pub left: Option<usize>,
1078 pub right: Option<usize>,
1079 pub kind: DiffRowKind,
1080 pub hunk: Option<usize>,
1081 pub left_inline: Vec<Range<usize>>,
1082 pub right_inline: Vec<Range<usize>>,
1083}
1084
1085#[derive(Clone, Debug, Eq, PartialEq)]
1086pub enum DiffDisplayRow {
1087 Content(usize),
1088 Fold {
1089 id: usize,
1090 full_range: Range<usize>,
1091 hidden_rows: usize,
1092 },
1093}
1094
1095#[derive(Clone, Debug)]
1096pub struct PreparedDiff {
1097 pub left: PreparedDocument,
1098 pub right: PreparedDocument,
1099 pub rows: Arc<[DiffAlignedRow]>,
1100 pub collapsed: Arc<[DiffDisplayRow]>,
1101 pub hunk_count: usize,
1102}
1103
1104pub fn prepare_diff(
1110 left: &DocumentDescriptor,
1111 right: &DocumentDescriptor,
1112 whitespace: DiffWhitespace,
1113 context_lines: Option<usize>,
1114 syntaxes: &SyntaxRegistry,
1115) -> Result<PreparedDiff, DocumentError> {
1116 prepare_diff_with_limits(
1117 left,
1118 right,
1119 whitespace,
1120 context_lines,
1121 syntaxes,
1122 DocumentLimits::default(),
1123 )
1124}
1125
1126#[allow(clippy::too_many_lines)]
1132pub fn prepare_diff_with_limits(
1133 left: &DocumentDescriptor,
1134 right: &DocumentDescriptor,
1135 whitespace: DiffWhitespace,
1136 context_lines: Option<usize>,
1137 syntaxes: &SyntaxRegistry,
1138 limits: DocumentLimits,
1139) -> Result<PreparedDiff, DocumentError> {
1140 let total_bytes = left
1141 .source
1142 .text()
1143 .len()
1144 .saturating_add(right.source.text().len());
1145 if total_bytes > limits.max_diff_total_bytes {
1146 return Err(DocumentError::TooManyDiffBytes {
1147 actual: total_bytes,
1148 limit: limits.max_diff_total_bytes,
1149 });
1150 }
1151 let left = prepare_document_with_limits(left, syntaxes, limits)?;
1152 let right = prepare_document_with_limits(right, syntaxes, limits)?;
1153 let left_keys = diff_keys(&left, whitespace);
1154 let right_keys = diff_keys(&right, whitespace);
1155 let deadline = Instant::now().checked_add(limits.diff_timeout);
1156 let ops = capture_diff_slices_deadline(Algorithm::Patience, &left_keys, &right_keys, deadline);
1157 let mut rows = Vec::with_capacity(left.lines().len().max(right.lines().len()));
1158 let mut hunk = 0usize;
1159 let mut in_change = false;
1160 for operation in ops {
1161 match operation {
1162 DiffOp::Equal {
1163 old_index,
1164 new_index,
1165 len,
1166 } => {
1167 in_change = false;
1168 rows.extend((0..len).map(|offset| DiffAlignedRow {
1169 left: Some(old_index + offset),
1170 right: Some(new_index + offset),
1171 kind: DiffRowKind::Equal,
1172 hunk: None,
1173 left_inline: Vec::new(),
1174 right_inline: Vec::new(),
1175 }));
1176 }
1177 DiffOp::Delete {
1178 old_index, old_len, ..
1179 } => {
1180 let id = next_hunk(&mut hunk, &mut in_change, limits.max_diff_hunks)?;
1181 rows.extend((0..old_len).map(|offset| DiffAlignedRow {
1182 left: Some(old_index + offset),
1183 right: None,
1184 kind: DiffRowKind::LeftOnly,
1185 hunk: Some(id),
1186 left_inline: Vec::new(),
1187 right_inline: Vec::new(),
1188 }));
1189 }
1190 DiffOp::Insert {
1191 new_index, new_len, ..
1192 } => {
1193 let id = next_hunk(&mut hunk, &mut in_change, limits.max_diff_hunks)?;
1194 rows.extend((0..new_len).map(|offset| DiffAlignedRow {
1195 left: None,
1196 right: Some(new_index + offset),
1197 kind: DiffRowKind::RightOnly,
1198 hunk: Some(id),
1199 left_inline: Vec::new(),
1200 right_inline: Vec::new(),
1201 }));
1202 }
1203 DiffOp::Replace {
1204 old_index,
1205 old_len,
1206 new_index,
1207 new_len,
1208 } => {
1209 let id = next_hunk(&mut hunk, &mut in_change, limits.max_diff_hunks)?;
1210 let aligned = old_len.max(new_len);
1211 for offset in 0..aligned {
1212 let left_index = (offset < old_len).then_some(old_index + offset);
1213 let right_index = (offset < new_len).then_some(new_index + offset);
1214 let (left_inline, right_inline) = match (left_index, right_index) {
1215 (Some(left_index), Some(right_index)) => inline_ranges(
1216 left.line_text(left_index).unwrap_or_default(),
1217 right.line_text(right_index).unwrap_or_default(),
1218 deadline,
1219 ),
1220 _ => (Vec::new(), Vec::new()),
1221 };
1222 rows.push(DiffAlignedRow {
1223 left: left_index,
1224 right: right_index,
1225 kind: match (left_index, right_index) {
1226 (Some(_), Some(_)) => DiffRowKind::Modified,
1227 (Some(_), None) => DiffRowKind::LeftOnly,
1228 (None, Some(_)) => DiffRowKind::RightOnly,
1229 (None, None) => unreachable!(),
1230 },
1231 hunk: Some(id),
1232 left_inline,
1233 right_inline,
1234 });
1235 }
1236 }
1237 }
1238 }
1239 let collapsed = collapse_rows(&rows, context_lines);
1240 Ok(PreparedDiff {
1241 left,
1242 right,
1243 rows: rows.into(),
1244 collapsed: collapsed.into(),
1245 hunk_count: hunk,
1246 })
1247}
1248
1249fn next_hunk(
1250 hunks: &mut usize,
1251 in_change: &mut bool,
1252 limit: usize,
1253) -> Result<usize, DocumentError> {
1254 if !*in_change {
1255 *hunks = hunks.checked_add(1).ok_or(DocumentError::TooManyHunks {
1256 actual: usize::MAX,
1257 limit,
1258 })?;
1259 *in_change = true;
1260 }
1261 if *hunks > limit {
1262 return Err(DocumentError::TooManyHunks {
1263 actual: *hunks,
1264 limit,
1265 });
1266 }
1267 Ok(*hunks - 1)
1268}
1269
1270fn diff_keys(document: &PreparedDocument, whitespace: DiffWhitespace) -> Vec<Cow<'_, str>> {
1271 document
1272 .lines()
1273 .iter()
1274 .enumerate()
1275 .map(|(index, _)| {
1276 let line = document.line_text(index).unwrap_or_default();
1277 match whitespace {
1278 DiffWhitespace::Exact => Cow::Borrowed(line),
1279 DiffWhitespace::IgnoreChanges => {
1280 Cow::Owned(line.split_whitespace().collect::<Vec<_>>().join(" "))
1281 }
1282 DiffWhitespace::IgnoreAll => Cow::Owned(
1283 line.chars()
1284 .filter(|character| !character.is_whitespace())
1285 .collect(),
1286 ),
1287 }
1288 })
1289 .collect()
1290}
1291
1292fn inline_ranges(
1293 left: &str,
1294 right: &str,
1295 deadline: Option<Instant>,
1296) -> (Vec<Range<usize>>, Vec<Range<usize>>) {
1297 if left.len().saturating_add(right.len()) > MAX_INLINE_DIFF_BYTES {
1298 return (
1299 (!left.is_empty())
1300 .then_some(0..left.len())
1301 .into_iter()
1302 .collect(),
1303 (!right.is_empty())
1304 .then_some(0..right.len())
1305 .into_iter()
1306 .collect(),
1307 );
1308 }
1309 let left_graphemes = left.graphemes(true).collect::<Vec<_>>();
1310 let right_graphemes = right.graphemes(true).collect::<Vec<_>>();
1311 let operations = capture_diff_slices_deadline(
1312 Algorithm::Patience,
1313 &left_graphemes,
1314 &right_graphemes,
1315 deadline,
1316 );
1317 let mut left_offset = 0usize;
1318 let mut right_offset = 0usize;
1319 let mut left_ranges = Vec::new();
1320 let mut right_ranges = Vec::new();
1321 for operation in operations {
1322 for change in operation.iter_changes(&left_graphemes, &right_graphemes) {
1323 let len = change.value().len();
1324 match change.tag() {
1325 similar::ChangeTag::Equal => {
1326 left_offset += len;
1327 right_offset += len;
1328 }
1329 similar::ChangeTag::Delete => {
1330 push_range(&mut left_ranges, left_offset..left_offset + len);
1331 left_offset += len;
1332 }
1333 similar::ChangeTag::Insert => {
1334 push_range(&mut right_ranges, right_offset..right_offset + len);
1335 right_offset += len;
1336 }
1337 }
1338 }
1339 }
1340 (left_ranges, right_ranges)
1341}
1342
1343fn push_range(ranges: &mut Vec<Range<usize>>, range: Range<usize>) {
1344 if range.is_empty() {
1345 return;
1346 }
1347 if let Some(previous) = ranges.last_mut()
1348 && previous.end == range.start
1349 {
1350 previous.end = range.end;
1351 } else {
1352 ranges.push(range);
1353 }
1354}
1355
1356fn collapse_rows(rows: &[DiffAlignedRow], context: Option<usize>) -> Vec<DiffDisplayRow> {
1357 let Some(context) = context else {
1358 return (0..rows.len()).map(DiffDisplayRow::Content).collect();
1359 };
1360 if rows.iter().all(|row| row.kind == DiffRowKind::Equal) {
1361 return if rows.is_empty() {
1362 Vec::new()
1363 } else {
1364 vec![DiffDisplayRow::Fold {
1365 id: 0,
1366 full_range: 0..rows.len(),
1367 hidden_rows: rows.len(),
1368 }]
1369 };
1370 }
1371 let mut keep = vec![false; rows.len()];
1372 for (index, row) in rows.iter().enumerate() {
1373 if row.kind != DiffRowKind::Equal {
1374 let start = index.saturating_sub(context);
1375 let end = index
1376 .saturating_add(context)
1377 .saturating_add(1)
1378 .min(rows.len());
1379 keep[start..end].fill(true);
1380 }
1381 }
1382 let mut display = Vec::new();
1383 let mut index = 0usize;
1384 let mut fold = 0usize;
1385 while index < rows.len() {
1386 if keep[index] {
1387 display.push(DiffDisplayRow::Content(index));
1388 index += 1;
1389 continue;
1390 }
1391 let start = index;
1392 while index < rows.len() && !keep[index] {
1393 index += 1;
1394 }
1395 display.push(DiffDisplayRow::Fold {
1396 id: fold,
1397 full_range: start..index,
1398 hidden_rows: index - start,
1399 });
1400 fold += 1;
1401 }
1402 display
1403}
1404
1405fn stable_text_hash(text: &str) -> u64 {
1406 const OFFSET: u64 = 0xcbf2_9ce4_8422_2325;
1407 const PRIME: u64 = 0x0000_0100_0000_01b3;
1408 text.bytes().fold(OFFSET, |hash, byte| {
1409 (hash ^ u64::from(byte)).wrapping_mul(PRIME)
1410 })
1411}
1412
1413fn validate_identity(value: &str) -> Result<(), DocumentError> {
1414 if value.is_empty()
1415 || value.len() > 256
1416 || !value.chars().all(|character| {
1417 character.is_ascii_alphanumeric() || matches!(character, '_' | '-' | '.' | '/' | ':')
1418 })
1419 {
1420 Err(DocumentError::InvalidIdentity(value.to_owned()))
1421 } else {
1422 Ok(())
1423 }
1424}
1425
1426#[derive(Clone, Debug, Error, Eq, PartialEq)]
1427pub enum DocumentError {
1428 #[error("document identity `{0}` must contain 1-256 safe ASCII characters")]
1429 InvalidIdentity(String),
1430 #[error("native text document `{0}` is already registered")]
1431 DuplicateDocument(String),
1432 #[error("native text document `{0}` is not registered")]
1433 UnknownDocument(String),
1434 #[error(
1435 "native text document `{identity}` revision must increase: current {current}, next {next}"
1436 )]
1437 NonMonotonicRevision {
1438 identity: String,
1439 current: u64,
1440 next: u64,
1441 },
1442 #[error("document contains {actual} bytes, exceeding the {limit}-byte budget")]
1443 TooManyBytes { actual: usize, limit: usize },
1444 #[error("diff inputs contain {actual} bytes, exceeding the {limit}-byte budget")]
1445 TooManyDiffBytes { actual: usize, limit: usize },
1446 #[error("document contains {actual} lines, exceeding the {limit}-line budget")]
1447 TooManyLines { actual: usize, limit: usize },
1448 #[error("diff contains {actual} hunks, exceeding the {limit}-hunk budget")]
1449 TooManyHunks { actual: usize, limit: usize },
1450 #[error("syntax definition or parse failed: {0}")]
1451 Syntax(String),
1452 #[error("syntax registry lock was poisoned")]
1453 SyntaxRegistryPoisoned,
1454 #[error("document resource limits must all be greater than zero")]
1455 InvalidLimits,
1456 #[error("document runtime configuration lock was poisoned")]
1457 DocumentConfigPoisoned,
1458}
1459
1460#[cfg(test)]
1461mod tests {
1462 use super::*;
1463
1464 fn descriptor(text: &str, language: &str) -> DocumentDescriptor {
1465 DocumentDescriptor {
1466 source: DocumentSource::from(text.to_owned()),
1467 label: language.to_owned(),
1468 file_name: None,
1469 language: Some(language.to_owned()),
1470 }
1471 }
1472
1473 #[test]
1474 fn native_documents_are_revisioned_and_registry_reads_are_exact() {
1475 let first = NativeTextDocument::new("server/a", 1, Arc::<str>::from("port=80\n")).unwrap();
1476 let second =
1477 NativeTextDocument::new("server/a", 2, Arc::<str>::from("port=443\n")).unwrap();
1478 let reader = ComponentInstancePath::root("View", "main");
1479 let mut registry = NativeTextDocumentRegistry::new();
1480 registry.register("config", first).unwrap();
1481 assert_eq!(
1482 registry.read_tracked(&reader, "config").unwrap().revision(),
1483 1
1484 );
1485 assert_eq!(
1486 registry.replace("config", second).unwrap(),
1487 BTreeSet::from([reader])
1488 );
1489 let stale = NativeTextDocument::new("server/a", 2, Arc::<str>::from("stale")).unwrap();
1490 assert!(matches!(
1491 registry.replace("config", stale),
1492 Err(DocumentError::NonMonotonicRevision { .. })
1493 ));
1494 }
1495
1496 #[test]
1497 fn rhai_and_go_are_built_in_and_unknown_languages_fall_back_to_plain_text() {
1498 let registry = SyntaxRegistry::new();
1499 let rhai = prepare_document(&descriptor("fn view() { 42 }\n", "rhai"), ®istry).unwrap();
1500 assert_eq!(rhai.language(), "Rhai");
1501 assert!(
1502 rhai.lines()[0]
1503 .syntax
1504 .iter()
1505 .any(|span| span.kind == SyntaxTokenKind::Keyword),
1506 "{:?}",
1507 rhai.lines()[0].syntax
1508 );
1509
1510 let go = prepare_document(
1511 &descriptor("package main\nfunc main() {}\n", "go"),
1512 ®istry,
1513 )
1514 .unwrap();
1515 assert_eq!(go.language(), "Go");
1516 let plain = prepare_document(&descriptor("hello\n", "not-a-language"), ®istry).unwrap();
1517 assert_eq!(plain.language(), "Plain Text");
1518 }
1519
1520 #[test]
1521 fn public_launch_language_pack_resolves_every_declared_language() {
1522 let registry = SyntaxRegistry::new();
1523 let mut missing = Vec::new();
1524 for language in [
1525 "rhai",
1526 "rust",
1527 "go",
1528 "javascript",
1529 "typescript",
1530 "tsx",
1531 "json",
1532 "jsonc",
1533 "toml",
1534 "yaml",
1535 "markdown",
1536 "bash",
1537 "python",
1538 "html",
1539 "css",
1540 "sql",
1541 "dockerfile",
1542 "RUST",
1543 ] {
1544 let document = prepare_document(&descriptor("value = 1\n", language), ®istry)
1545 .unwrap_or_else(|error| panic!("{language}: {error}"));
1546 if document.language() == "Plain Text" {
1547 missing.push(language);
1548 }
1549 }
1550 assert!(missing.is_empty(), "missing {missing:?}");
1551 }
1552
1553 #[test]
1554 fn line_ranges_normalize_crlf_but_retain_terminal_empty_line() {
1555 let registry = SyntaxRegistry::new();
1556 let document = prepare_document(&descriptor("one\r\ntwo\n", "text"), ®istry).unwrap();
1557 assert_eq!(document.lines().len(), 3);
1558 assert_eq!(document.line_text(0), Some("one"));
1559 assert_eq!(document.line_text(1), Some("two"));
1560 assert_eq!(document.line_text(2), Some(""));
1561 assert!(document.ends_with_newline());
1562 }
1563
1564 #[test]
1565 fn diff_is_direction_neutral_refines_unicode_and_collapses_context() {
1566 let registry = SyntaxRegistry::new();
1567 let left = descriptor(
1568 "same\n城市 = 东京\nunchanged 1\nunchanged 2\nunchanged 3\nunchanged 4\n",
1569 "rhai",
1570 );
1571 let right = descriptor(
1572 "same\n城市 = 上海\nunchanged 1\nunchanged 2\nunchanged 3\nunchanged 4\n",
1573 "rhai",
1574 );
1575 let diff = prepare_diff(&left, &right, DiffWhitespace::Exact, Some(1), ®istry).unwrap();
1576 assert_eq!(diff.hunk_count, 1);
1577 let modified = diff
1578 .rows
1579 .iter()
1580 .find(|row| row.kind == DiffRowKind::Modified)
1581 .unwrap();
1582 assert!(!modified.left_inline.is_empty());
1583 assert!(!modified.right_inline.is_empty());
1584 assert!(
1585 diff.collapsed
1586 .iter()
1587 .any(|row| matches!(row, DiffDisplayRow::Fold { .. }))
1588 );
1589 }
1590
1591 #[test]
1592 fn whitespace_policy_is_explicit() {
1593 let registry = SyntaxRegistry::new();
1594 let left = descriptor("value = 1\n", "rhai");
1595 let right = descriptor("value = 1\n", "rhai");
1596 assert_eq!(
1597 prepare_diff(&left, &right, DiffWhitespace::Exact, None, ®istry)
1598 .unwrap()
1599 .hunk_count,
1600 1
1601 );
1602 assert_eq!(
1603 prepare_diff(
1604 &left,
1605 &right,
1606 DiffWhitespace::IgnoreChanges,
1607 None,
1608 ®istry,
1609 )
1610 .unwrap()
1611 .hunk_count,
1612 0
1613 );
1614 }
1615
1616 #[test]
1617 fn host_limits_fail_explicitly_before_unbounded_work() {
1618 let registry = SyntaxRegistry::new();
1619 let descriptor = descriptor("12345", "text");
1620 let limits = DocumentLimits {
1621 max_document_bytes: 4,
1622 ..DocumentLimits::default()
1623 };
1624 assert!(matches!(
1625 prepare_document_with_limits(&descriptor, ®istry, limits),
1626 Err(DocumentError::TooManyBytes {
1627 actual: 5,
1628 limit: 4
1629 })
1630 ));
1631 let runtime = DocumentRuntimeConfig::new();
1632 assert_eq!(runtime.limits(), DocumentLimits::default());
1633 assert_eq!(
1634 runtime.set_limits(DocumentLimits {
1635 max_diff_hunks: 0,
1636 ..DocumentLimits::default()
1637 }),
1638 Err(DocumentError::InvalidLimits)
1639 );
1640 }
1641
1642 #[test]
1643 fn host_registered_syntax_participates_in_background_safe_snapshots() {
1644 let registry = SyntaxRegistry::new();
1645 registry
1646 .register_sublime_syntax(
1647 r"%YAML 1.2
1648---
1649name: Probe
1650file_extensions: [probe]
1651scope: source.probe
1652contexts:
1653 main:
1654 - match: '\bprobe\b'
1655 scope: keyword.control.probe
1656",
1657 )
1658 .unwrap();
1659 let document = prepare_document(&descriptor("probe value\n", "probe"), ®istry).unwrap();
1660 assert_eq!(document.language(), "Probe");
1661 assert!(
1662 document.lines()[0]
1663 .syntax
1664 .iter()
1665 .any(|span| span.kind == SyntaxTokenKind::Keyword)
1666 );
1667 }
1668}