#[derive(Debug, Clone, Default, PartialEq, Eq)]
#[non_exhaustive]
pub struct ParsedSchema {
pub name: String,
pub entities: Vec<EntityDef>,
pub types: Vec<TypeDef>,
}
impl ParsedSchema {
#[must_use]
pub fn new(name: impl Into<String>, entities: Vec<EntityDef>, types: Vec<TypeDef>) -> Self {
Self {
name: name.into(),
entities,
types,
}
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
#[non_exhaustive]
pub enum AggregateKind {
List,
Set,
Bag,
Array,
}
impl AggregateKind {
fn from_keyword(keyword: &str) -> Option<Self> {
match keyword {
"LIST" => Some(Self::List),
"SET" => Some(Self::Set),
"BAG" => Some(Self::Bag),
"ARRAY" => Some(Self::Array),
_ => None,
}
}
}
#[derive(Debug, Clone, PartialEq, Eq, Hash)]
#[non_exhaustive]
pub enum Bound {
Integer(u64),
Unbounded,
Expression(String),
}
impl Bound {
fn parse(text: &str) -> Self {
let text = text.trim();
if text == "?" {
Self::Unbounded
} else if let Ok(value) = text.parse() {
Self::Integer(value)
} else {
Self::Expression(text.split_whitespace().collect::<Vec<_>>().join(" "))
}
}
}
#[derive(Debug, Clone, PartialEq, Eq, Hash)]
#[non_exhaustive]
pub struct Aggregation {
pub kind: AggregateKind,
pub lower: Bound,
pub upper: Bound,
pub unique: bool,
pub optional_elements: bool,
}
impl Aggregation {
#[must_use]
pub const fn new(kind: AggregateKind, lower: Bound, upper: Bound) -> Self {
Self {
kind,
lower,
upper,
unique: false,
optional_elements: false,
}
}
#[must_use]
pub const fn unique(mut self) -> Self {
self.unique = true;
self
}
#[must_use]
pub const fn optional_elements(mut self) -> Self {
self.optional_elements = true;
self
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
#[non_exhaustive]
pub struct Attribute {
pub name: String,
pub type_name: String,
pub optional: bool,
pub aggregate: bool,
pub aggregation: Vec<Aggregation>,
}
impl Attribute {
#[must_use]
pub fn new(name: impl Into<String>, type_name: impl Into<String>) -> Self {
Self {
name: name.into(),
type_name: type_name.into(),
optional: false,
aggregate: false,
aggregation: Vec::new(),
}
}
#[must_use]
pub const fn optional(mut self) -> Self {
self.optional = true;
self
}
#[must_use]
pub const fn aggregate(mut self) -> Self {
self.aggregate = true;
self
}
#[must_use]
pub fn with_aggregation(mut self, aggregation: Aggregation) -> Self {
self.aggregation.push(aggregation);
self.aggregate = true;
self
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
#[non_exhaustive]
pub struct InverseAttribute {
pub name: String,
pub redeclares: Option<String>,
pub entity: String,
pub for_attribute: String,
pub aggregation: Option<Aggregation>,
}
impl InverseAttribute {
#[must_use]
pub fn new(
name: impl Into<String>,
entity: impl Into<String>,
for_attribute: impl Into<String>,
) -> Self {
Self {
name: name.into(),
redeclares: None,
entity: entity.into(),
for_attribute: for_attribute.into(),
aggregation: None,
}
}
#[must_use]
pub fn with_aggregation(mut self, aggregation: Aggregation) -> Self {
self.aggregation = Some(aggregation);
self
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
#[non_exhaustive]
pub struct UniqueRule {
pub label: Option<String>,
pub attributes: Vec<String>,
}
impl UniqueRule {
#[must_use]
pub fn new(label: impl Into<String>, attributes: Vec<String>) -> Self {
Self {
label: Some(label.into()),
attributes,
}
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
#[non_exhaustive]
pub struct WhereRule {
pub label: String,
pub expression: String,
}
impl WhereRule {
#[must_use]
pub fn new(label: impl Into<String>, expression: impl Into<String>) -> Self {
Self {
label: label.into(),
expression: expression.into(),
}
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
#[non_exhaustive]
pub struct Redeclaration {
pub supertype: String,
pub name: String,
pub type_name: String,
}
impl Redeclaration {
#[must_use]
pub fn new(
supertype: impl Into<String>,
name: impl Into<String>,
type_name: impl Into<String>,
) -> Self {
Self {
supertype: supertype.into(),
name: name.into(),
type_name: type_name.into(),
}
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
#[non_exhaustive]
pub struct EntityDef {
pub name: String,
pub supertypes: Vec<String>,
pub abstract_: bool,
pub attributes: Vec<Attribute>,
pub derived: Vec<String>,
pub redeclared: Vec<Redeclaration>,
pub where_rules: Vec<WhereRule>,
pub inverses: Vec<InverseAttribute>,
pub unique_rules: Vec<UniqueRule>,
}
impl EntityDef {
#[must_use]
pub fn new(name: impl Into<String>) -> Self {
Self {
name: name.into(),
supertypes: Vec::new(),
abstract_: false,
attributes: Vec::new(),
derived: Vec::new(),
redeclared: Vec::new(),
where_rules: Vec::new(),
inverses: Vec::new(),
unique_rules: Vec::new(),
}
}
#[must_use]
pub const fn abstract_(mut self) -> Self {
self.abstract_ = true;
self
}
#[must_use]
pub fn with_where_rule(mut self, rule: WhereRule) -> Self {
self.where_rules.push(rule);
self
}
#[must_use]
pub fn with_inverse(mut self, inverse: InverseAttribute) -> Self {
self.inverses.push(inverse);
self
}
#[must_use]
pub fn with_unique_rule(mut self, rule: UniqueRule) -> Self {
self.unique_rules.push(rule);
self
}
#[must_use]
pub fn with_supertype(mut self, supertype: impl Into<String>) -> Self {
self.supertypes.push(supertype.into());
self
}
#[must_use]
pub fn supertype(&self) -> Option<&str> {
self.supertypes.first().map(String::as_str)
}
#[must_use]
pub fn with_redeclared(mut self, redeclaration: Redeclaration) -> Self {
self.redeclared.push(redeclaration);
self
}
#[must_use]
pub fn with_attribute(mut self, attribute: Attribute) -> Self {
self.attributes.push(attribute);
self
}
#[must_use]
pub fn with_derived(mut self, name: impl Into<String>) -> Self {
self.derived.push(name.into());
self
}
#[must_use]
pub fn is_derived(&self, name: &str) -> bool {
self.derived
.iter()
.any(|declared| declared.eq_ignore_ascii_case(name))
}
#[must_use]
pub fn is_redeclared(&self, name: &str) -> bool {
self.redeclared
.iter()
.any(|declared| declared.name.eq_ignore_ascii_case(name))
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
#[non_exhaustive]
pub enum TypeKind {
Defined(String),
Enumeration(Vec<String>),
Select(Vec<String>),
}
#[derive(Debug, Clone, PartialEq, Eq)]
#[non_exhaustive]
pub struct TypeDef {
pub name: String,
pub kind: TypeKind,
}
impl TypeDef {
#[must_use]
pub fn new(name: impl Into<String>, kind: TypeKind) -> Self {
Self {
name: name.into(),
kind,
}
}
#[must_use]
pub const fn is_defined(&self) -> bool {
matches!(self.kind, TypeKind::Defined(_))
}
}
#[must_use]
pub fn parse(source: &str) -> ParsedSchema {
let cleaned = strip_comments(source);
let upper = ascii_uppercase(&cleaned);
let name = schema_name(&cleaned, &upper).unwrap_or_default();
let entities = blocks(&cleaned, &upper, "ENTITY", "END_ENTITY")
.filter_map(parse_entity)
.collect();
let types = blocks(&cleaned, &upper, "TYPE", "END_TYPE")
.filter_map(parse_type)
.collect();
ParsedSchema {
name,
entities,
types,
}
}
fn strip_comments(source: &str) -> String {
let bytes = source.as_bytes();
let mut output = bytes.to_vec();
let mut position = 0;
let mut quoted = false;
while position < bytes.len() {
if bytes[position] == b'\'' {
if quoted && bytes.get(position + 1) == Some(&b'\'') {
position += 2;
continue;
}
quoted = !quoted;
position += 1;
continue;
}
if !quoted && bytes[position..].starts_with(b"(*") {
let start = position;
position += 2;
while position < bytes.len() && !bytes[position..].starts_with(b"*)") {
position += 1;
}
position = (position + 2).min(bytes.len());
blank_non_newlines(&mut output[start..position]);
continue;
}
if !quoted && bytes[position..].starts_with(b"--") {
let start = position;
position += 2;
while position < bytes.len() && bytes[position] != b'\n' {
position += 1;
}
blank_non_newlines(&mut output[start..position]);
continue;
}
position += 1;
}
String::from_utf8(output).expect("input was valid UTF-8")
}
fn blank_non_newlines(bytes: &mut [u8]) {
for byte in bytes {
if *byte != b'\n' && *byte != b'\r' {
*byte = b' ';
}
}
}
fn ascii_uppercase(source: &str) -> String {
let mut bytes = source.as_bytes().to_vec();
bytes.make_ascii_uppercase();
String::from_utf8(bytes).expect("ASCII case conversion preserves UTF-8")
}
fn schema_name(source: &str, upper: &str) -> Option<String> {
let start = find_keyword(upper, "SCHEMA", 0)? + "SCHEMA".len();
let end = source[start..].find(';')? + start;
source[start..end]
.split_whitespace()
.next()
.map(ToOwned::to_owned)
}
fn blocks<'a>(
source: &'a str,
upper: &'a str,
start_keyword: &'static str,
end_keyword: &'static str,
) -> impl Iterator<Item = &'a str> {
let mut cursor = 0;
std::iter::from_fn(move || {
let start = find_keyword(upper, start_keyword, cursor)?;
let end_start = find_keyword(upper, end_keyword, start + start_keyword.len())?;
let semicolon = source[end_start..]
.find(';')
.map_or(source.len(), |offset| end_start + offset + 1);
cursor = semicolon;
Some(&source[start..semicolon])
})
}
fn find_keyword(haystack: &str, needle: &str, from: usize) -> Option<usize> {
let bytes = haystack.as_bytes();
let mut cursor = from;
while let Some(relative) = haystack[cursor..].find(needle) {
let position = cursor + relative;
let before = position.checked_sub(1).and_then(|index| bytes.get(index));
let after = bytes.get(position + needle.len());
if before.is_none_or(|byte| !is_identifier_byte(*byte))
&& after.is_none_or(|byte| !is_identifier_byte(*byte))
{
return Some(position);
}
cursor = position + needle.len();
}
None
}
fn find_block_keyword(source: &str, upper: &str, needle: &str, from: usize) -> Option<usize> {
let mut cursor = from;
while let Some(position) = find_keyword(upper, needle, cursor) {
let preceding = source[from..position]
.rfind(';')
.map_or(&source[from..position], |offset| {
&source[from + offset + 1..position]
});
if preceding.trim().is_empty() {
return Some(position);
}
cursor = position + needle.len();
}
None
}
fn is_identifier_byte(byte: u8) -> bool {
byte.is_ascii_alphanumeric() || byte == b'_'
}
fn parse_entity(block: &str) -> Option<EntityDef> {
let upper = ascii_uppercase(block);
let header_end = block.find(';')?;
let header = &block[..header_end];
let header_upper = &upper[..header_end];
let entity_position = find_keyword(header_upper, "ENTITY", 0)? + "ENTITY".len();
let name = header[entity_position..]
.split_whitespace()
.next()?
.trim_matches(|character: char| !character.is_alphanumeric() && character != '_')
.to_owned();
let supertypes = clause_names(header, header_upper, "SUBTYPE OF");
let abstract_ = find_keyword(header_upper, "ABSTRACT", 0).is_some();
let body_end = ["DERIVE", "INVERSE", "UNIQUE", "WHERE", "END_ENTITY"]
.into_iter()
.filter_map(|keyword| find_block_keyword(block, &upper, keyword, header_end + 1))
.min()
.unwrap_or(block.len());
let mut attributes = Vec::new();
let mut redeclared = Vec::new();
for statement in block[header_end + 1..body_end].split(';') {
if let Some(redeclaration) = parse_redeclaration(statement) {
redeclared.push(redeclaration);
} else if let Some(attribute) = parse_attribute(statement) {
attributes.push(attribute);
}
}
let derived = parse_derive_block(block, &upper, header_end + 1);
let where_rules = parse_where_block(block, &upper, header_end + 1);
let inverses = block_statements(block, &upper, "INVERSE", header_end + 1)
.filter_map(parse_inverse)
.collect();
let unique_rules = block_statements(block, &upper, "UNIQUE", header_end + 1)
.filter_map(parse_unique_rule)
.collect();
Some(EntityDef {
name,
supertypes,
abstract_,
attributes,
derived,
redeclared,
where_rules,
inverses,
unique_rules,
})
}
fn block_statements<'a>(
block: &'a str,
upper: &str,
keyword: &'static str,
from: usize,
) -> impl Iterator<Item = &'a str> {
let range = find_block_keyword(block, upper, keyword, from).and_then(|start| {
let start = start + keyword.len();
let end = ["INVERSE", "UNIQUE", "WHERE", "END_ENTITY"]
.into_iter()
.filter(|next| *next != keyword)
.filter_map(|next| find_block_keyword(block, upper, next, start))
.min()
.unwrap_or(block.len());
(start < end).then_some(start..end)
});
range
.map_or("", |range| &block[range])
.split(';')
.filter(|statement| !statement.trim().is_empty())
}
fn parse_inverse(statement: &str) -> Option<InverseAttribute> {
let (target, declaration) = statement.split_once(':')?;
let target = target.trim();
let (redeclares, name) = match target
.get(..5)
.filter(|prefix| prefix.eq_ignore_ascii_case("SELF\\"))
{
Some(_) => {
let (supertype, name) = target[5..].split_once('.')?;
(Some(supertype.trim().to_owned()), name.trim())
}
None => (None, target),
};
if name.is_empty() || !name.bytes().all(is_identifier_byte) {
return None;
}
let upper = ascii_uppercase(declaration);
let for_position = find_keyword(&upper, "FOR", 0)?;
let for_attribute = declaration[for_position + "FOR".len()..].trim();
let (aggregation, element) = aggregation_levels(&declaration[..for_position]);
let entity = element.split_whitespace().next()?;
if for_attribute.is_empty() || aggregation.len() > 1 {
return None;
}
Some(InverseAttribute {
name: name.to_owned(),
redeclares,
entity: entity.to_owned(),
for_attribute: for_attribute.split_whitespace().collect(),
aggregation: aggregation.into_iter().next(),
})
}
fn parse_unique_rule(statement: &str) -> Option<UniqueRule> {
let (label, attributes) = match statement.split_once(':') {
Some((label, attributes)) => {
let label = label.trim();
if label.is_empty() || !label.bytes().all(is_identifier_byte) {
return None;
}
(Some(label.to_owned()), attributes)
}
None => (None, statement),
};
let attributes: Vec<String> = attributes
.split(',')
.map(|attribute| attribute.split_whitespace().collect::<String>())
.filter(|attribute| !attribute.is_empty())
.collect();
(!attributes.is_empty()).then_some(UniqueRule { label, attributes })
}
fn aggregation_levels(declaration: &str) -> (Vec<Aggregation>, &str) {
let mut levels = Vec::new();
let mut rest = declaration.trim_start();
loop {
let word_end = rest
.find(|character: char| !(character.is_ascii_alphanumeric() || character == '_'))
.unwrap_or(rest.len());
let Some(kind) = AggregateKind::from_keyword(&rest[..word_end].to_ascii_uppercase()) else {
return (levels, rest);
};
let mut after = rest[word_end..].trim_start();
let (lower, upper) = if let Some(inner) = after.strip_prefix('[') {
let Some(close) = inner.find(']') else {
return (levels, rest);
};
let bounds = &inner[..close];
after = inner[close + 1..].trim_start();
match bounds.split_once(':') {
Some((lower, upper)) => (Bound::parse(lower), Bound::parse(upper)),
None => (Bound::parse(bounds), Bound::parse(bounds)),
}
} else {
(Bound::Integer(0), Bound::Unbounded)
};
let Some(of) = after
.get(..2)
.filter(|word| word.eq_ignore_ascii_case("OF"))
.map(|_| after[2..].trim_start())
else {
return (levels, rest);
};
let mut level = Aggregation::new(kind, lower, upper);
rest = of;
for (keyword, mark) in [
(
"OPTIONAL",
Aggregation::optional_elements as fn(Aggregation) -> Aggregation,
),
("UNIQUE", Aggregation::unique),
] {
if rest
.get(..keyword.len())
.is_some_and(|word| word.eq_ignore_ascii_case(keyword))
&& !rest[keyword.len()..]
.starts_with(|c: char| c.is_ascii_alphanumeric() || c == '_')
{
level = mark(level);
rest = rest[keyword.len()..].trim_start();
}
}
levels.push(level);
}
}
fn parse_where_rule(statement: &str) -> Option<WhereRule> {
let (label, expression) = statement.split_once(':')?;
let label = label.trim();
if label.is_empty() || !label.bytes().all(is_identifier_byte) {
return None;
}
let expression = expression.split_whitespace().collect::<Vec<_>>().join(" ");
if expression.is_empty() {
return None;
}
Some(WhereRule {
label: label.to_owned(),
expression,
})
}
fn parse_where_block(block: &str, upper: &str, from: usize) -> Vec<WhereRule> {
let Some(start) = find_block_keyword(block, upper, "WHERE", from) else {
return Vec::new();
};
let start = start + "WHERE".len();
let end = find_keyword(upper, "END_ENTITY", start).unwrap_or(block.len());
if end <= start {
return Vec::new();
}
block[start..end]
.split(';')
.filter_map(parse_where_rule)
.collect()
}
fn parse_derive_block(block: &str, upper: &str, from: usize) -> Vec<String> {
let Some(start) = find_keyword(upper, "DERIVE", from) else {
return Vec::new();
};
let start = start + "DERIVE".len();
let end = ["INVERSE", "UNIQUE", "WHERE", "END_ENTITY"]
.into_iter()
.filter_map(|keyword| find_keyword(upper, keyword, start))
.min()
.unwrap_or(block.len());
if end <= start {
return Vec::new();
}
block[start..end]
.split(';')
.filter_map(derived_attribute_name)
.collect()
}
fn derived_attribute_name(statement: &str) -> Option<String> {
let (target, _) = statement.split_once(':')?;
let target = target.trim();
let name = target.rsplit('.').next()?.trim();
let name = name.rsplit('\\').next()?.trim();
if name.is_empty() || !name.bytes().all(is_identifier_byte) {
return None;
}
Some(name.to_owned())
}
fn clause_names(header: &str, upper: &str, clause: &str) -> Vec<String> {
let Some(position) = find_keyword(upper, clause, 0).map(|p| p + clause.len()) else {
return Vec::new();
};
let Some(open) = header[position..].find('(').map(|o| position + o + 1) else {
return Vec::new();
};
let Some(close) = header[open..].find(')').map(|o| open + o) else {
return Vec::new();
};
header[open..close]
.split(',')
.map(str::trim)
.filter(|name| !name.is_empty())
.map(ToOwned::to_owned)
.collect()
}
fn parse_redeclaration(statement: &str) -> Option<Redeclaration> {
let (target, _) = statement.split_once(':')?;
let target = target.trim();
let qualified = target
.get(..5)
.filter(|prefix| prefix.eq_ignore_ascii_case("SELF\\"))
.map(|_| &target[5..])?;
let (supertype, name) = qualified.split_once('.')?;
let (supertype, name) = (supertype.trim(), name.trim());
if supertype.is_empty()
|| name.is_empty()
|| !supertype.bytes().all(is_identifier_byte)
|| !name.bytes().all(is_identifier_byte)
{
return None;
}
let type_name =
parse_attribute(&format!("{name}{}", &statement[statement.find(':')?..]))?.type_name;
Some(Redeclaration {
supertype: supertype.to_owned(),
name: name.to_owned(),
type_name,
})
}
fn parse_attribute(statement: &str) -> Option<Attribute> {
let (name, declaration) = statement.split_once(':')?;
let name = name.trim();
if name.is_empty() {
return None;
}
let mut declaration = declaration.trim();
let optional = declaration
.get(.."OPTIONAL".len())
.is_some_and(|word| word.eq_ignore_ascii_case("OPTIONAL"))
&& !declaration["OPTIONAL".len()..]
.starts_with(|c: char| c.is_ascii_alphanumeric() || c == '_');
if optional {
declaration = declaration["OPTIONAL".len()..].trim_start();
}
let (aggregation, element) = aggregation_levels(declaration);
let type_name = element
.split_whitespace()
.next()?
.trim_matches(|character: char| matches!(character, '(' | ')' | ';'))
.to_owned();
Some(Attribute {
name: name.to_owned(),
type_name,
optional,
aggregate: !aggregation.is_empty(),
aggregation,
})
}
fn parse_type(block: &str) -> Option<TypeDef> {
let upper = ascii_uppercase(block);
let statement_end = block.find(';')?;
let statement = &block[..statement_end];
let statement_upper = &upper[..statement_end];
let type_position = find_keyword(statement_upper, "TYPE", 0)? + "TYPE".len();
let equals = statement[type_position..].find('=')? + type_position;
let name = statement[type_position..equals].trim().to_owned();
let right = statement[equals + 1..].trim();
let right_upper = ascii_uppercase(right);
let kind = if let Some(position) = find_keyword(&right_upper, "ENUMERATION", 0) {
TypeKind::Enumeration(parenthesized_names(right, position + "ENUMERATION".len()))
} else if let Some(position) = find_keyword(&right_upper, "SELECT", 0) {
TypeKind::Select(parenthesized_names(right, position + "SELECT".len()))
} else {
TypeKind::Defined(right.to_owned())
};
Some(TypeDef { name, kind })
}
fn parenthesized_names(source: &str, from: usize) -> Vec<String> {
let Some(open) = source[from..].find('(').map(|offset| from + offset + 1) else {
return Vec::new();
};
let close = source[open..]
.find(')')
.map_or(source.len(), |offset| open + offset);
source[open..close]
.split(',')
.map(str::trim)
.filter(|name| !name.is_empty())
.map(ToOwned::to_owned)
.collect()
}