tree_sitter_ruby_sqry/lib.rs
1//! Tree-sitter grammar for Ruby (vendored for sqry)
2//!
3//! The Rust binding in this crate is first-party (maintained in the sqry
4//! repository). The grammar under grammar-src/ is vendored THIRD-PARTY code
5//! reproduced under its upstream license; the full license text and copyright
6//! notice ship next to the sources in grammar-src/LICENSE and are also recorded
7//! in the repository-root THIRD-PARTY-LICENSES file.
8//!
9//! **Source Grammar**: <https://github.com/tree-sitter/tree-sitter-ruby>
10//! **Base release**: v0.23.1 (published crate `tree-sitter-ruby` 0.23.1)
11//! **License**: MIT
12//!
13//! # Why this grammar is vendored
14//!
15//! The published 0.23.1 external scanner overflows its serialization buffer on
16//! heredocs, which aborts the process through `assert(size == length)` in
17//! `deserialize`. Two defects combine: the bounds guard in `serialize`
18//! under-counts the bytes the loop body writes, and the heredoc identifier
19//! length is stored in a single byte, so an identifier longer than 255
20//! characters wraps. The abort is reachable in ordinary release builds, so any
21//! indexed Ruby file of that shape terminates the CLI, the daemon, the LSP
22//! server or the MCP server.
23//!
24//! Upstream fixed both in commit `ad907a69da0c` ("scanner: fix heredoc
25//! serialization buffer overflows", 2026-03-10) by widening the length field to
26//! `uint32_t` on the write and read sides and correcting the guard. That commit
27//! has never been released: 0.23.1 (2024-11-11) is still the newest version on
28//! crates.io, and the repository has taken four commits in the twenty-two months
29//! since. A git dependency is not an option because this workspace publishes to
30//! crates.io, so the grammar is vendored with the upstream fix applied verbatim.
31//!
32//! `grammar-src/scanner.c` is byte-identical to upstream master at
33//! `ad907a69da0c` (SHA-256 `88c1c036d5af7c22a1bc9cc5e50411d414ac9624afed676d42957c566ac4083d`).
34//! Every other file under grammar-src/ is unmodified from the published 0.23.1
35//! crate.
36//!
37//! # Retirement condition
38//!
39//! Upstream issue [#269](https://github.com/tree-sitter/tree-sitter-ruby/issues/269)
40//! tracks the crash. When a release containing `ad907a69da0c` reaches crates.io,
41//! delete this crate and point `sqry-lang-ruby` back at the published
42//! `tree-sitter-ruby`.
43
44use tree_sitter::Language;
45
46unsafe extern "C" {
47 fn tree_sitter_ruby() -> Language;
48}
49
50/// Returns the tree-sitter Language for Ruby
51#[must_use = "Language handles must be registered with tree-sitter consumers"]
52pub fn language() -> Language {
53 let lang = unsafe { tree_sitter_ruby() };
54 sqry_tree_sitter_support::validate_language_or_panic(lang, "Ruby")
55}
56
57/// Fallible alternative to [`language()`]
58#[allow(clippy::missing_errors_doc)] // Vendored tree-sitter binding
59pub fn try_language() -> Result<Language, sqry_tree_sitter_support::TreeSitterError> {
60 let lang = unsafe { tree_sitter_ruby() };
61 sqry_tree_sitter_support::validate_language(lang)
62}
63
64/// The content of the node-types.json file for this grammar.
65pub const NODE_TYPES: &str = include_str!("../grammar-src/node-types.json");
66
67#[cfg(test)]
68mod tests {
69 use super::*;
70
71 #[test]
72 fn test_can_load_grammar() {
73 let lang = language();
74 assert!(lang.abi_version() > 0);
75 }
76
77 #[test]
78 fn test_try_language_succeeds() {
79 assert!(try_language().is_ok());
80 }
81
82 #[test]
83 #[allow(clippy::const_is_empty)]
84 fn test_node_types_not_empty() {
85 assert!(!NODE_TYPES.is_empty());
86 }
87
88 /// Regression for the heredoc serialization overflow (sqry issue #747,
89 /// upstream tree-sitter-ruby issue #269).
90 ///
91 /// The published 0.23.1 scanner stored the heredoc identifier length in a
92 /// single byte, so an identifier longer than 255 characters wrapped and
93 /// `deserialize` then read a different number of bytes than `serialize`
94 /// wrote, tripping `assert(size == length)` and aborting the process.
95 #[test]
96 fn heredoc_identifier_longer_than_255_chars_does_not_abort() {
97 let word = "A".repeat(300);
98 let source = format!("<<~{word}\ncontent\n{word}\n");
99 let mut parser = tree_sitter::Parser::new();
100 parser
101 .set_language(&language())
102 .expect("Ruby grammar should load");
103 let tree = parser.parse(&source, None).expect("parse should return");
104 assert_eq!(tree.root_node().kind(), "program");
105 }
106
107 /// The other half of the same defect: the bounds guard in `serialize`
108 /// admitted a heredoc it did not have room for. The guard counted
109 /// `size + 2 + word` while the loop body wrote `4 + word` (three flags, a
110 /// one-byte length, then the identifier), so a state that ended two bytes
111 /// under the 1024-byte buffer was written one byte past it.
112 ///
113 /// The window is narrow. With no string literals on the stack the
114 /// serialized state starts at two bytes and each heredoc adds `4 + word`,
115 /// so only a seven-character identifier lands inside it, on the
116 /// ninety-third heredoc: the guard sees 1023 and the write ends at 1025.
117 /// The sweep covers the arithmetic being off by a heredoc either way.
118 #[test]
119 fn serialize_guard_admits_no_heredoc_it_cannot_fit() {
120 let mut parser = tree_sitter::Parser::new();
121 parser
122 .set_language(&language())
123 .expect("Ruby grammar should load");
124
125 for count in 88..=96 {
126 // Seven-character identifiers, and no string literals, so the
127 // serialized layout matches the accounting above.
128 let openers: Vec<String> = (0..count).map(|i| format!("<<~H{i:06}")).collect();
129 let source = format!("x = [{}]\n", openers.join(", "));
130 let tree = parser
131 .parse(&source, None)
132 .expect("parse should return for every heredoc count");
133 assert_eq!(tree.root_node().kind(), "program");
134 }
135 }
136
137 /// Incremental reparse, isolated to the incremental path.
138 ///
139 /// An earlier version of this test opened a long-identifier heredoc in its
140 /// very first parse, so under the previous revision it aborted before the
141 /// reparse ran. It passed for a reason unrelated to the name on it. Here
142 /// the first parse is deliberately benign, a short identifier the previous
143 /// revision's one-byte length field represents exactly, and only the edit
144 /// introduces an identifier past that limit. An abort in this test can
145 /// therefore only come from the reparse. Verified against the previous
146 /// revision: the first parse alone succeeds, and only the reparse aborts.
147 #[test]
148 fn incremental_reparse_round_trips_heredoc_scanner_state() {
149 let mut parser = tree_sitter::Parser::new();
150 parser
151 .set_language(&language())
152 .expect("Ruby grammar should load");
153
154 let first = "x = <<~SHORT\ncontent\nSHORT\n";
155 let mut tree = parser.parse(first, None).expect("parse should return");
156 assert_eq!(tree.root_node().kind(), "program");
157
158 let word = "B".repeat(300);
159 let appended = format!("y = <<~{word}\ncontent\n{word}\n");
160 let second = format!("{first}{appended}");
161 tree.edit(&tree_sitter::InputEdit {
162 start_byte: first.len(),
163 old_end_byte: first.len(),
164 new_end_byte: second.len(),
165 start_position: tree_sitter::Point::new(3, 0),
166 old_end_position: tree_sitter::Point::new(3, 0),
167 new_end_position: tree_sitter::Point::new(6, 0),
168 });
169 let reparsed = parser
170 .parse(&second, Some(&tree))
171 .expect("incremental parse should return");
172 assert_eq!(reparsed.root_node().kind(), "program");
173 }
174}