hamelin_lib 0.21.9

Core library for Hamelin query language
Documentation
#![allow(dead_code)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
#![allow(nonstandard_style)]
#![allow(unused_imports)]
#![allow(unused_mut)]
#![allow(unused_braces)]
#![allow(unused_parens)]

use std::borrow::Cow;
use std::cmp::{max, min};
use std::ops::{Range, RangeInclusive};

use antlr_rust::parser_rule_context::ParserRuleContext;
use antlr_rust::token::Token;
use antlr_rust::tree::TerminalNode;
use antlr_rust::TidExt;

use crate::antlr::hamelinlexer::HamelinLexer;
use crate::antlr::hamelinparser::HamelinParserContextType;

pub mod hamelinlexer;
pub mod hamelinlistener;
pub mod hamelinparser;
pub mod hamelinvisitor;

pub mod parse;
pub mod trinotypelexer;
pub mod trinotypelistener;
pub mod trinotypeparser;
pub mod trinotypevisitor;

pub use parse::*;

/// Strip numeric separator underscores from ANTLR token text.
///
/// The grammar's `DECIMAL_INTEGER` fragment allows `_` between digits (e.g., `1_000`),
/// but Rust's `.parse()` methods don't accept them. Returns a borrowed reference when
/// no underscores are present (zero-allocation fast path).
pub fn strip_numeric_separators(s: &str) -> Cow<'_, str> {
    if s.contains('_') {
        Cow::Owned(s.replace('_', ""))
    } else {
        Cow::Borrowed(s)
    }
}

/// Calculate the completion interval for a tree node.
///
/// The completion interval is the cursor positions between which we think the the user is
/// trying to complete the target tree node. The completion interval includes the cursor
/// position directly after / cuddled up against, because when the cursor is in that position
/// the user might be still "trying to type" that.
pub fn completion_interval<'a, PRC>(tree: &PRC) -> RangeInclusive<usize>
where
    PRC: ParserRuleContext<'a>,
{
    interval_helper(tree, true, None)
}

/// Calculate the completion interval for a tree node that is allowed to cuddle its left neighbor.
pub fn cuddled_completion_interval<'a, PRC>(tree: &PRC) -> RangeInclusive<usize>
where
    PRC: ParserRuleContext<'a>,
{
    interval_helper(tree, true, Some(true))
}

/// Calculate the interval for a tree node.
///
/// This is the 0-indexed character offsets that "cover" the given tree.
pub fn interval<'a, PRC>(tree: &PRC) -> RangeInclusive<usize>
where
    PRC: ParserRuleContext<'a>,
{
    interval_helper(tree, false, None)
}

/// Fit the given interval to the given string.
/// Clamps to valid character indices only; upper bound is last character index
/// (char count - 1). Use for content ranges, not cursor positions.
/// Intervals are codepoint offsets, so we clamp against the char count, not byte length.
pub fn fit(interval: &RangeInclusive<usize>, s: &str) -> RangeInclusive<usize> {
    let max_pos = s.chars().count().saturating_sub(1);
    let end = min(*interval.end(), max_pos);
    let start = min(*interval.start(), end);
    start..=end
}

/// Fit the given exclusive range to the given string. The end is exclusive (one past the last
/// character to replace), so the maximum valid end is `chars().count()` (position after the
/// last character). Use for `Completion.at` ranges.
/// Intervals are codepoint offsets, so we clamp against the char count, not byte length.
pub fn fit_cursor(interval: &Range<usize>, s: &str) -> Range<usize> {
    let max_pos = s.chars().count();
    let end = min(interval.end, max_pos);
    let start = min(interval.start, end);
    start..end
}

/// Whether this is an error node cuddled up against its previous.
pub fn prepend_space<'a, PRC>(tree: &PRC) -> bool
where
    PRC: ParserRuleContext<'a>,
{
    let start_token = tree.start();
    let stop_token = tree.stop();

    if start_token.get_start() > stop_token.get_stop() {
        start_token.get_start() == stop_token.get_stop() + 1
    } else {
        false
    }
}

// Track which tokens "can cuddle."
// These are the tokens which won't accidentally masquerade as identifiers.
fn can_cuddle(typ: isize) -> bool {
    match typ {
        hamelinlexer::ARROW
        | hamelinlexer::PLUS
        | hamelinlexer::MINUS
        | hamelinlexer::ASTERISK
        | hamelinlexer::SLASH
        | hamelinlexer::PERCENT
        | hamelinlexer::LCURLY
        | hamelinlexer::RCURLY
        | hamelinlexer::COLON
        | hamelinlexer::QUESTIONMARK
        | hamelinlexer::EQ
        | hamelinlexer::NEQ
        | hamelinlexer::LT
        | hamelinlexer::LTE
        | hamelinlexer::GT
        | hamelinlexer::GTE
        | hamelinlexer::RANGE
        | hamelinlexer::ASSIGN
        | hamelinlexer::COMMA
        | hamelinlexer::PIPE
        | hamelinlexer::LPARENS
        | hamelinlexer::RPARENS
        | hamelinlexer::DOT
        | hamelinlexer::LBRACKET
        | hamelinlexer::RBRACKET => true,
        _ => false,
    }
}

fn interval_helper<'a, PRC>(
    tree: &PRC,
    completion: bool,
    cuddled: Option<bool>,
) -> RangeInclusive<usize>
where
    PRC: ParserRuleContext<'a>,
{
    if let Some(tok) = tree.downcast_ref::<TerminalNode<HamelinParserContextType>>() {
        return tok.symbol.get_start() as usize..=tok.symbol.get_stop() as usize;
    }

    let start_token = tree.start();
    let stop_token = tree.stop();

    if start_token.get_start() > stop_token.get_stop() {
        // If the starting token in the range actually appears after the stopping token --
        //   * This likely means that the ANTLR error recovery has gotten involved!
        //   * Be careful about figuring where the "end" is.
        let mut start = start_token.get_start() as usize;

        // Unless the caller has specifically asked me to override the behavior,
        // I check whether the thing we are trying to insert can be "cuddled" by
        // checking the type of the previous successful parsed node (which is the "stop token").
        // If it can be cuddled, I allow it to be, otherwise I bump the index up by one.
        let cuddled = cuddled.unwrap_or_else(|| can_cuddle(stop_token.get_token_type()));
        if completion && !cuddled && stop_token.get_stop() as usize + 1 == start {
            start += 1;
        }

        start..=start
    } else {
        let start = start_token.get_start() as usize;
        let mut stop = stop_token.get_stop() as usize;

        // If we're asking for the interval of a tree for completion purposes, and the tree
        // is not an error tree (it is a real tree that represents a character range), we
        // want to include the cursor position directly after this tree as "inside" its range,
        // because that's the place the cursor would be if the user was "still typing" that
        // thing.
        if completion {
            stop += 1;
        }

        start..=stop
    }
}