Skip to main content

fmt_lang/
lib.rs

1//! # fmt_lang
2//!
3//! A source formatter driven by declarative rules over a lossless syntax
4//! tree, rendering through [`pretty_lang`].
5//!
6//! A language gets a formatter from configuration, not code: describe how
7//! its nodes and tokens are laid out as [`Rules`] (data, keyed by kind names,
8//! the way a sketch names them), [`compile`](Rules::compile) them against the
9//! language's kinds once, and [`format()`] any [`syntax_lang`] tree of that
10//! language. Whatever the rules do not mention keeps its original whitespace.
11//!
12//! ## Quick start
13//!
14//! ```
15//! use fmt_lang::{format, Indent, NodeRule, Rules, Space, TokenRule, Trailing};
16//! use lang_forge::Language;
17//!
18//! let json = Language::from_lsf(
19//!     r#"
20//!     [language]
21//!     name = "json"
22//!
23//!     [lexer]
24//!     strings = ['"']
25//!     line_comments = ["//"]
26//!
27//!     [rules]
28//!     document = "value"
29//!     value    = "object | array | STRING | NUMBER | 'true' | 'false' | 'null'"
30//!     object   = "'{' (member (',' member)* ','?)? '}'"
31//!     member   = "STRING ':' value"
32//!     array    = "'[' (value (',' value)* ','?)? ']'"
33//!     "#,
34//! )?;
35//!
36//! let style = Rules::new()
37//!     .indent(2)
38//!     .verbatim("ERROR")
39//!     .token(TokenRule::new(":").before(Space::None).after(Space::Single))
40//!     .node(
41//!         NodeRule::new("object")
42//!             .group()
43//!             .indent(Indent::Block)
44//!             .delimiters("{", "}", Space::Line)
45//!             .separator(",", Space::Line, Trailing::Never),
46//!     )
47//!     .node(
48//!         NodeRule::new("array")
49//!             .group()
50//!             .indent(Indent::Block)
51//!             .delimiters("[", "]", Space::SoftLine)
52//!             .separator(",", Space::Line, Trailing::Never),
53//!     )
54//!     .compile(|name| json.kind(name))?;
55//!
56//! let parse = json.parse("{\"a\":[1,2,],   \"b\" :{}}");
57//! let wide = format(parse.tree(), parse.source(), &style, 80)?;
58//! assert_eq!(wide, "{ \"a\": [1, 2], \"b\": {} }\n");
59//!
60//! // Too wide for 16 columns: the object breaks, the array still fits.
61//! let narrow = format(parse.tree(), parse.source(), &style, 16)?;
62//! assert_eq!(narrow, "{\n  \"a\": [1, 2],\n  \"b\": {}\n}\n");
63//! # Ok::<(), Box<dyn std::error::Error>>(())
64//! ```
65//!
66//! ## The rule model
67//!
68//! - A [`NodeRule`] makes a node a *group* (flat if it fits, broken at its
69//!   line opportunities otherwise), *indents* its body or its whole content,
70//!   asks for spacing *before* and *after* the node, names the *delimiters*
71//!   and *separator* of a list (with a [`Trailing`] separator policy), caps
72//!   *blank lines*, and carries *token rules* for its direct child tokens.
73//! - A [`TokenRule`] asks for spacing before and after one token kind, at the
74//!   top level (a default everywhere) or inside a node rule (that context
75//!   only, and it wins).
76//! - Spacing is a [`Space`]; when several rules speak about one gap, the most
77//!   generous answer wins. When no rule speaks, the gap keeps its original
78//!   whitespace (with trailing spaces at line ends removed).
79//! - [`Rules::verbatim`] names node kinds written exactly as in the source,
80//!   such as a parser's `ERROR` nodes.
81//!
82//! ## Guarantees
83//!
84//! Property-tested over random valid and invalid sources of forged languages
85//! and over arbitrary trees (see `tests/`):
86//!
87//! - **Idempotent:** formatting formatted output changes nothing.
88//! - **Token-preserving:** the significant token sequence is unchanged (the
89//!   only exception is a trailing separator that a [`Trailing::Always`] or
90//!   [`Trailing::Never`] policy adds or removes), and every comment's text
91//!   appears exactly once, in its original order. A line comment always ends
92//!   its line. How comments are attached is documented on [`format()`].
93//! - **Total:** any tree whose tokens describe the source formats without
94//!   panicking, error nodes included; a tree that does not match its source is
95//!   a [`FormatError`], never a panic.
96//! - **Deterministic:** the same input gives the same output.
97//!
98//! For sources with lexical errors (an unterminated string or comment), the
99//! guarantees hold when the errors' regions are passed to [`format_keeping`]:
100//! such a token's extent depends on the whitespace after it, which the
101//! formatter cannot see from the tree.
102//!
103//! The walk is iterative and linear in the size of the tree, so trees nested
104//! hundreds of thousands of levels deep format without exhausting the stack;
105//! indentation stops growing at [`Rules::max_indent`], which bounds the output
106//! for hostile input.
107//!
108//! ## Limits
109//!
110//! - Whether two tokens may touch is decided by [`can_touch`], a conservative
111//!   heuristic, unless the style supplies an exact test
112//!   ([`Style::with_touch`]).
113//! - A parser's recovery that leaves no trace in the tree (an assumed missing
114//!   token) is invisible here. [`Trailing::Always`] and [`Trailing::Never`]
115//!   edit only lists with no error node and no unclosed delimited node inside,
116//!   but a recovery these checks miss could still read an edited list
117//!   differently; [`Trailing::Preserve`] never edits.
118//! - In languages whose line breaks are significant tokens, rules that break
119//!   lines ([`Space::Line`], [`Space::Hard`]) would add tokens; use them only
120//!   where the language allows a line break.
121//! - Indentation is spaces; widths are counted in `char`s (pretty-lang's
122//!   measure), not display columns.
123//!
124//! ## Features
125//!
126//! - `std` (default): the standard library, forwarded to `syntax-lang` and
127//!   `pretty-lang`. Without it the crate is `no_std` and needs only `alloc`.
128
129#![cfg_attr(not(feature = "std"), no_std)]
130#![cfg_attr(docsrs, feature(doc_cfg))]
131#![forbid(unsafe_code)]
132#![deny(missing_docs)]
133#![deny(unsafe_op_in_unsafe_fn)]
134#![deny(unused_must_use)]
135#![deny(unused_results)]
136#![deny(clippy::unwrap_used)]
137#![deny(clippy::expect_used)]
138#![deny(clippy::todo)]
139#![deny(clippy::unimplemented)]
140#![deny(clippy::print_stdout)]
141#![deny(clippy::print_stderr)]
142#![deny(clippy::dbg_macro)]
143#![deny(clippy::unreachable)]
144#![deny(clippy::undocumented_unsafe_blocks)]
145
146extern crate alloc;
147
148mod error;
149mod flat;
150mod rules;
151mod style;
152mod walk;
153
154use alloc::string::String;
155
156pub use error::{FormatError, RuleError};
157pub use rules::{Indent, MAX_INDENT_STEP, NodeRule, Rules, Space, TokenRule, Trailing};
158pub use style::Style;
159
160// Re-exported whole, so callers name the exact tree and document types this
161// crate's API is built on.
162pub use pretty_lang;
163pub use syntax_lang;
164
165use pretty_lang::Doc;
166use syntax_lang::{Node, Span, TokenKind};
167
168/// Compiles and runs the `rust` code blocks in `README.md` and `docs/API.md` as
169/// part of `cargo test`, so the published examples cannot drift from the API.
170///
171/// Present only while collecting doctests (`#[cfg(doctest)]`); it is not part of
172/// the public surface and does not appear in the built library or its docs.
173#[cfg(doctest)]
174#[doc = include_str!("../README.md")]
175#[doc = include_str!("../docs/API.md")]
176pub struct MarkdownDocTests;
177
178/// Formats the source `tree` was parsed from, at `width` columns.
179///
180/// `tree` must be a lossless tree of `source` (its tokens, trivia included,
181/// cover a contiguous range of `source`); it may be a whole file or any
182/// subtree, and may contain error nodes. The output covers the tree's range.
183///
184/// # Layout
185///
186/// Between every two neighbouring significant tokens (a node named by
187/// [`Rules::verbatim`] counts as one token), the formatter asks every rule that
188/// applies to the gap: the left token's `after`, the right token's `before`,
189/// the `after` of nodes ending there and the `before` of nodes starting there,
190/// and the delimiter and separator rules of the node holding both tokens. The
191/// most generous answer wins; if none applies, the source's whitespace is kept.
192/// Groups and indentation come from the node rules; [`pretty_lang`] then picks
193/// the line breaks that fit `width`.
194///
195/// # Comments
196///
197/// Comments (any trivia token that is not all whitespace) are attached by
198/// where they sit in the source:
199///
200/// - **trailing**: no line break between the comment and the token before it.
201///   It stays on that token's line, after one space.
202/// - **leading**: the comment started a line. It starts a line in the output,
203///   at the indentation of the gap it sits in, keeping up to the configured
204///   number of blank lines before it.
205/// - **dangling**: a leading comment just before the closing delimiter of an
206///   indented body ([`Indent::Block`]). It is indented with the body.
207///
208/// A comment followed by a line break in the source is followed by one in the
209/// output, so a line comment always ends its line. Comments never cross a
210/// token, so their order is preserved, and their text is written unchanged.
211///
212/// # Errors
213///
214/// A [`FormatError`] if the tree's tokens do not describe `source`: a span out
215/// of bounds or inside a character, or tokens with gaps or overlaps between
216/// them.
217///
218/// # Examples
219///
220/// ```
221/// use fmt_lang::syntax_lang::{Builder, Span, Token, TokenKind};
222/// use fmt_lang::{format, Rules, Space, TokenRule};
223///
224/// #[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)]
225/// enum K { Sum, Num, Plus, Ws }
226/// impl TokenKind for K {
227///     fn is_trivia(&self) -> bool { matches!(self, K::Ws) }
228/// }
229///
230/// // `1   +2`, lexed and built by hand.
231/// let source = "1   +2";
232/// let mut b = Builder::new();
233/// b.start_node(K::Sum);
234/// b.token(Token::new(K::Num, Span::new(0, 1)));
235/// b.token(Token::new(K::Ws, Span::new(1, 4)));
236/// b.token(Token::new(K::Plus, Span::new(4, 5)));
237/// b.token(Token::new(K::Num, Span::new(5, 6)));
238/// b.finish_node();
239/// let tree = b.finish()?;
240///
241/// let style = Rules::new()
242///     .final_newline(false)
243///     .token(TokenRule::new("+").around(Space::Single))
244///     .compile(|name| (name == "+").then_some(K::Plus))?;
245/// assert_eq!(format(&tree, source, &style, 80)?, "1 + 2");
246/// # Ok::<(), Box<dyn std::error::Error>>(())
247/// ```
248pub fn format<K: TokenKind + Ord>(
249    tree: &Node<K>,
250    source: &str,
251    style: &Style<K>,
252    width: usize,
253) -> Result<String, FormatError> {
254    Ok(format_doc(tree, source, style, &[])?.render(width))
255}
256
257/// Formats like [`format()`], but leaves the whitespace around and inside the
258/// `keep` regions exactly as written.
259///
260/// Pass the spans of lexical errors here (an LSP server has them as
261/// diagnostics). Some tokens end where the source's whitespace happens to be:
262/// an unterminated string usually ends at the line break, so removing that
263/// line break would pull the next token into the string. The formatter cannot
264/// tell such a token from a complete one, but every gap that touches a kept
265/// region keeps its original whitespace, no separator is inserted there, and a
266/// separator inside one is never removed, so the token stays as it was.
267///
268/// A gap touches a kept region when the region overlaps the token before the
269/// gap, the gap itself, or the token after it; an empty region (a diagnostic
270/// pointing between two tokens) keeps the gap it falls in. Regions may overlap
271/// and come in any order.
272///
273/// Idempotence holds as long as the regions mark the same tokens when the
274/// output is formatted again. Lexical errors do (they belong to a token);
275/// parse errors such as "expected `;`" may point elsewhere once whitespace has
276/// changed, so keeping their regions can shift between runs.
277///
278/// # Errors
279///
280/// As for [`format()`].
281///
282/// # Examples
283///
284/// ```
285/// use fmt_lang::{format, format_keeping, NodeRule, Rules, Space, Trailing};
286/// use lang_forge::Language;
287///
288/// let json = Language::from_lsf(
289///     "[language]\nname = \"j\"\n[lexer]\nstrings = ['\"']\n[rules]\n\
290///      list = \"'[' (STRING (',' STRING)*)? ']'\"\n",
291/// )?;
292/// let style = Rules::new()
293///     .node(NodeRule::new("list").delimiters("[", "]", Space::None).separator(",", Space::Single, Trailing::Preserve))
294///     .compile(|n| json.kind(n))?;
295///
296/// // `"open` is unterminated: the string ends at the line break.
297/// let parse = json.parse("[\"open\n, \"b\"]");
298/// let spans: Vec<_> = parse.diagnostics().iter().map(|d| d.primary().span()).collect();
299///
300/// let kept = format_keeping(parse.tree(), parse.source(), &style, 80, &spans)?;
301/// assert_eq!(kept, "[\"open\n, \"b\"]\n");
302///
303/// // Without the error spans the break goes, and the comma joins the string.
304/// let lost = format(parse.tree(), parse.source(), &style, 80)?;
305/// assert_eq!(lost, "[\"open, \"b\"]\n");
306/// # Ok::<(), Box<dyn std::error::Error>>(())
307/// ```
308pub fn format_keeping<K: TokenKind + Ord>(
309    tree: &Node<K>,
310    source: &str,
311    style: &Style<K>,
312    width: usize,
313    keep: &[Span],
314) -> Result<String, FormatError> {
315    Ok(format_doc(tree, source, style, keep)?.render(width))
316}
317
318/// Builds the [`Doc`] that [`format_keeping`] (or, with no `keep` regions,
319/// [`format()`]) renders, for callers that render it themselves: into an
320/// existing buffer or writer, or at several widths.
321///
322/// # Errors
323///
324/// As for [`format()`].
325///
326/// # Examples
327///
328/// ```
329/// use fmt_lang::syntax_lang::{Element, Node, Span, Token, TokenKind};
330/// use fmt_lang::{format_doc, Style};
331///
332/// #[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)]
333/// struct Word;
334/// impl TokenKind for Word {}
335///
336/// let tree = Node::new(Word, vec![Element::Token(Token::new(Word, Span::new(0, 5)))]);
337/// let doc = format_doc(&tree, "hello", &Style::default(), &[])?;
338/// let mut out = String::new();
339/// doc.render_into(80, &mut out)?;
340/// assert_eq!(out, "hello\n");
341/// # Ok::<(), Box<dyn std::error::Error>>(())
342/// ```
343pub fn format_doc<K: TokenKind + Ord>(
344    tree: &Node<K>,
345    source: &str,
346    style: &Style<K>,
347    keep: &[Span],
348) -> Result<Doc, FormatError> {
349    let flat = flat::flatten(tree, source, style)?;
350    Ok(walk::Walker::new(style, source, &flat, keep).run(&flat.events, flat.bom))
351}
352
353/// The default test for whether two tokens may be written touching, with no
354/// whitespace between them: `left` is the earlier token's text, `right` the
355/// later one's.
356///
357/// The formatter consults it only where a rule would remove whitespace the
358/// source had; tokens that touched in the source may always touch. It answers
359/// `false` when the touching characters could plausibly lex as one token:
360/// two word characters (`a` `b`), two ASCII punctuation characters (`-` `-`,
361/// `/` `/`), a quote next to a word character or another quote (`b` `"x"`,
362/// `"a"` `"b"`), or a dot next to a digit (`1` `.5`). A quote next to other
363/// punctuation may touch (`"k"` `:`). Brackets, parentheses, commas, and semicolons may touch
364/// anything except an identical bracket or brace (`[[`, `]]`, `{{`, `}}` are
365/// tokens in some languages).
366///
367/// It is a heuristic. A language that knows its own lexer supplies an exact
368/// test with [`Style::with_touch`].
369///
370/// # Examples
371///
372/// ```
373/// use fmt_lang::can_touch;
374///
375/// assert!(can_touch("f", "("));
376/// assert!(can_touch("x", ","));
377/// assert!(can_touch("a", "."));
378/// assert!(!can_touch("let", "x"));
379/// assert!(!can_touch("-", "-"));
380/// assert!(!can_touch("[", "["));
381/// ```
382#[must_use]
383pub fn can_touch(left: &str, right: &str) -> bool {
384    let (Some(a), Some(b)) = (left.chars().next_back(), right.chars().next()) else {
385        return true;
386    };
387    let fixed = |c: char| matches!(c, '(' | ')' | '[' | ']' | '{' | '}' | ',' | ';');
388    if fixed(a) || fixed(b) {
389        return !(a == b && matches!(a, '[' | ']' | '{' | '}'));
390    }
391    let word = |c: char| c.is_alphanumeric() || c == '_' || (!c.is_ascii() && !c.is_whitespace());
392    let quote = |c: char| matches!(c, '"' | '\'' | '`');
393    if quote(a) || quote(b) {
394        // A quote fuses with a word (a string prefix or suffix: `b"x"`, `"x"s`)
395        // or another quote (`""` is an escape in some languages), but other
396        // punctuation cannot reach into a string.
397        return !(word(a) || word(b) || (quote(a) && quote(b)));
398    }
399    if word(a) && word(b) {
400        return false;
401    }
402    if a.is_ascii_punctuation() && b.is_ascii_punctuation() {
403        return false;
404    }
405    !((a == '.' && b.is_ascii_digit()) || (a.is_ascii_digit() && b == '.'))
406}
407
408#[cfg(test)]
409mod tests {
410    use super::*;
411
412    #[test]
413    fn test_can_touch_rules() {
414        assert!(can_touch("", "x"));
415        assert!(can_touch("x", ""));
416        assert!(can_touch(")", ";"));
417        assert!(can_touch("(", "("));
418        assert!(!can_touch("}", "}"));
419        assert!(!can_touch("a", "b"));
420        assert!(!can_touch("a", "1"));
421        assert!(!can_touch("é", "x"));
422        assert!(!can_touch("+", "="));
423        assert!(!can_touch("b", "\"s\""));
424        assert!(!can_touch("'c'", "x"));
425        assert!(!can_touch("\"a\"", "\"b\""));
426        assert!(can_touch("\"k\"", ":"));
427        assert!(can_touch("=", "\"v\""));
428        assert!(!can_touch("1", ".5"));
429        assert!(!can_touch(".", "5"));
430        assert!(can_touch("-", "x"));
431        assert!(can_touch("x", "+"));
432    }
433}