use std::collections::BTreeMap;
use type_bridge_contract::diagnostic::{Diagnostic, DiagnosticCategory, DiagnosticCode};
use type_bridge_contract::schema::{
DocumentFingerprint, DocumentId, SchemaDiagnostic, SchemaDiagnostics,
SchemaDocumentSetFingerprint, SourceSpan,
};
use crate::schema_set::SchemaSetManifestDocument;
pub const DEFAULT_MAX_DOCUMENTS: usize = 4_096;
pub const DEFAULT_MAX_AGGREGATE_BYTES: usize = 64 * 1024 * 1024;
pub const DEFAULT_MAX_DOCUMENT_BYTES: usize = 16 * 1024 * 1024;
pub const DEFAULT_MAX_DEPTH: usize = 64;
pub const DEFAULT_MAX_NODES: usize = 65_536;
pub const DEFAULT_MAX_SCALAR_SOURCE_BYTES: usize = DEFAULT_MAX_DOCUMENT_BYTES;
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub struct SchemaParseLimits {
max_documents: usize,
max_aggregate_bytes: usize,
max_document_bytes: usize,
max_depth: usize,
max_nodes: usize,
max_scalar_source_bytes: usize,
}
impl SchemaParseLimits {
#[must_use]
pub const fn new(
max_documents: usize,
max_aggregate_bytes: usize,
max_document_bytes: usize,
max_depth: usize,
max_nodes: usize,
max_scalar_source_bytes: usize,
) -> Self {
Self {
max_documents,
max_aggregate_bytes,
max_document_bytes,
max_depth,
max_nodes,
max_scalar_source_bytes,
}
}
#[must_use]
pub const fn max_documents(self) -> usize {
self.max_documents
}
#[must_use]
pub const fn max_aggregate_bytes(self) -> usize {
self.max_aggregate_bytes
}
#[must_use]
pub const fn max_document_bytes(self) -> usize {
self.max_document_bytes
}
#[must_use]
pub const fn max_depth(self) -> usize {
self.max_depth
}
#[must_use]
pub const fn max_nodes(self) -> usize {
self.max_nodes
}
#[must_use]
pub const fn max_scalar_source_bytes(self) -> usize {
self.max_scalar_source_bytes
}
#[must_use]
pub const fn max_scalar_bytes(self) -> usize {
self.max_scalar_source_bytes()
}
}
impl Default for SchemaParseLimits {
fn default() -> Self {
Self::new(
DEFAULT_MAX_DOCUMENTS,
DEFAULT_MAX_AGGREGATE_BYTES,
DEFAULT_MAX_DOCUMENT_BYTES,
DEFAULT_MAX_DEPTH,
DEFAULT_MAX_NODES,
DEFAULT_MAX_SCALAR_SOURCE_BYTES,
)
}
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum YamlScalarStyle {
Plain,
SingleQuoted,
DoubleQuoted,
Literal,
Folded,
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum YamlCollectionStyle {
Block,
Flow,
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum CommentPlacement {
Above,
Right,
Free,
Last,
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct SchemaComment {
text: String,
placement: CommentPlacement,
span: SourceSpan,
}
impl SchemaComment {
pub(crate) fn new(text: String, placement: CommentPlacement, span: SourceSpan) -> Self {
Self {
text,
placement,
span,
}
}
#[must_use]
pub fn text(&self) -> &str {
&self.text
}
#[must_use]
pub const fn placement(&self) -> CommentPlacement {
self.placement
}
#[must_use]
pub const fn span(&self) -> &SourceSpan {
&self.span
}
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct YamlScalar {
value: String,
raw: String,
style: YamlScalarStyle,
span: SourceSpan,
}
impl YamlScalar {
pub(crate) fn new(
value: String,
raw: String,
style: YamlScalarStyle,
span: SourceSpan,
) -> Self {
Self {
value,
raw,
style,
span,
}
}
#[must_use]
pub fn value(&self) -> &str {
&self.value
}
#[must_use]
pub fn raw(&self) -> &str {
&self.raw
}
#[must_use]
pub const fn style(&self) -> YamlScalarStyle {
self.style
}
#[must_use]
pub const fn span(&self) -> &SourceSpan {
&self.span
}
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct YamlMappingEntry {
key: YamlScalar,
value: YamlNode,
}
impl YamlMappingEntry {
pub(crate) const fn new(key: YamlScalar, value: YamlNode) -> Self {
Self { key, value }
}
#[must_use]
pub const fn key(&self) -> &YamlScalar {
&self.key
}
#[must_use]
pub const fn value(&self) -> &YamlNode {
&self.value
}
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct YamlMapping {
entries: Vec<YamlMappingEntry>,
style: YamlCollectionStyle,
span: SourceSpan,
}
impl YamlMapping {
pub(crate) const fn new(
entries: Vec<YamlMappingEntry>,
style: YamlCollectionStyle,
span: SourceSpan,
) -> Self {
Self {
entries,
style,
span,
}
}
#[must_use]
pub fn entries(&self) -> &[YamlMappingEntry] {
&self.entries
}
#[must_use]
pub const fn style(&self) -> YamlCollectionStyle {
self.style
}
#[must_use]
pub const fn span(&self) -> &SourceSpan {
&self.span
}
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct YamlSequence {
items: Vec<YamlNode>,
style: YamlCollectionStyle,
span: SourceSpan,
}
impl YamlSequence {
pub(crate) const fn new(
items: Vec<YamlNode>,
style: YamlCollectionStyle,
span: SourceSpan,
) -> Self {
Self { items, style, span }
}
#[must_use]
pub fn items(&self) -> &[YamlNode] {
&self.items
}
#[must_use]
pub const fn style(&self) -> YamlCollectionStyle {
self.style
}
#[must_use]
pub const fn span(&self) -> &SourceSpan {
&self.span
}
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub enum YamlNode {
Scalar(YamlScalar),
Sequence(YamlSequence),
Mapping(YamlMapping),
}
impl YamlNode {
#[must_use]
pub const fn span(&self) -> &SourceSpan {
match self {
Self::Scalar(value) => value.span(),
Self::Sequence(value) => value.span(),
Self::Mapping(value) => value.span(),
}
}
#[must_use]
pub const fn as_mapping(&self) -> Option<&YamlMapping> {
match self {
Self::Mapping(value) => Some(value),
Self::Scalar(_) | Self::Sequence(_) => None,
}
}
#[must_use]
pub const fn as_sequence(&self) -> Option<&YamlSequence> {
match self {
Self::Sequence(value) => Some(value),
Self::Scalar(_) | Self::Mapping(_) => None,
}
}
#[must_use]
pub const fn as_scalar(&self) -> Option<&YamlScalar> {
match self {
Self::Scalar(value) => Some(value),
Self::Sequence(_) | Self::Mapping(_) => None,
}
}
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct SchemaDocument {
id: DocumentId,
source: String,
fingerprint: DocumentFingerprint,
root: YamlMapping,
comments: Vec<SchemaComment>,
}
impl SchemaDocument {
pub(crate) const fn new(
id: DocumentId,
source: String,
fingerprint: DocumentFingerprint,
root: YamlMapping,
comments: Vec<SchemaComment>,
) -> Self {
Self {
id,
source,
fingerprint,
root,
comments,
}
}
pub fn parse(id: DocumentId, source: impl Into<String>) -> Result<Self, SchemaDiagnostics> {
Self::parse_with_limits(id, source, SchemaParseLimits::default())
}
pub fn parse_with_limits(
id: DocumentId,
source: impl Into<String>,
limits: SchemaParseLimits,
) -> Result<Self, SchemaDiagnostics> {
crate::yaml::parse_document_with_limits(id, source.into(), limits)
}
#[must_use]
pub const fn id(&self) -> &DocumentId {
&self.id
}
#[must_use]
pub fn source(&self) -> &str {
&self.source
}
#[must_use]
pub const fn fingerprint(&self) -> &DocumentFingerprint {
&self.fingerprint
}
#[must_use]
pub const fn root(&self) -> &YamlMapping {
&self.root
}
#[must_use]
pub fn comments(&self) -> &[SchemaComment] {
&self.comments
}
}
#[derive(Clone, Debug, Default, Eq, PartialEq)]
pub struct SchemaDocumentSet {
documents: BTreeMap<DocumentId, SchemaDocument>,
manifest: Option<SchemaSetManifestDocument>,
}
impl SchemaDocumentSet {
pub fn parse<I, S>(sources: I) -> Result<Self, SchemaDiagnostics>
where
I: IntoIterator<Item = (DocumentId, S)>,
S: Into<String>,
{
Self::parse_with_limits(sources, SchemaParseLimits::default())
}
pub fn parse_with_limits<I, S>(
sources: I,
limits: SchemaParseLimits,
) -> Result<Self, SchemaDiagnostics>
where
I: IntoIterator<Item = (DocumentId, S)>,
S: Into<String>,
{
let mut documents: BTreeMap<DocumentId, SchemaDocument> = BTreeMap::new();
let mut aggregate_bytes = 0usize;
for (id, source) in sources {
if documents.len() >= limits.max_documents() {
return Err(resource_diagnostic(
"schema_document_count_limit",
format!(
"schema document count exceeds the limit of {}",
limits.max_documents()
),
None,
));
}
let source = source.into();
aggregate_bytes = aggregate_bytes.checked_add(source.len()).ok_or_else(|| {
resource_diagnostic(
"schema_aggregate_size_limit",
"schema aggregate source size overflowed",
None,
)
})?;
if aggregate_bytes > limits.max_aggregate_bytes() {
return Err(resource_diagnostic(
"schema_aggregate_size_limit",
format!(
"schema aggregate source size exceeds the limit of {} bytes",
limits.max_aggregate_bytes()
),
None,
));
}
if let Some(existing) = documents.get(&id) {
return Err(crate::yaml::diagnostic_with_related(
DiagnosticCategory::InvalidContract,
"duplicate_schema_document",
format!("schema document identifier `{}` is duplicated", id.as_str()),
existing.root().span().clone(),
existing.root().span().clone(),
"first document with this identifier",
));
}
let document = SchemaDocument::parse_with_limits(id.clone(), source, limits)?;
documents.insert(id, document);
}
Ok(Self {
documents,
manifest: None,
})
}
pub(crate) fn attach_manifest(&mut self, manifest: SchemaSetManifestDocument) {
self.manifest = Some(manifest);
}
#[must_use]
pub const fn manifest(&self) -> Option<&SchemaSetManifestDocument> {
self.manifest.as_ref()
}
pub fn fingerprint(&self) -> Result<SchemaDocumentSetFingerprint, SchemaDiagnostics> {
let mut canonical = Vec::new();
canonical.extend_from_slice(
&u64::try_from(self.documents.len())
.expect("schema document count is bounded below u64::MAX")
.to_be_bytes(),
);
for (id, document) in &self.documents {
canonical.extend_from_slice(
&u64::try_from(id.as_str().len())
.expect("document identifier length is bounded below u64::MAX")
.to_be_bytes(),
);
canonical.extend_from_slice(id.as_str().as_bytes());
canonical.extend_from_slice(&document.fingerprint().as_fingerprint().digest().bytes());
}
SchemaDocumentSetFingerprint::compute(&canonical)
.map_err(|error| SchemaDiagnostics::one(SchemaDiagnostic::new(error, None)))
}
#[must_use]
pub fn len(&self) -> usize {
self.documents.len()
}
#[must_use]
pub fn is_empty(&self) -> bool {
self.documents.is_empty()
}
#[must_use]
pub fn get(&self, id: &DocumentId) -> Option<&SchemaDocument> {
self.documents.get(id)
}
pub fn iter(&self) -> impl ExactSizeIterator<Item = (&DocumentId, &SchemaDocument)> {
self.documents.iter()
}
}
pub(crate) fn resource_diagnostic(
code: &'static str,
message: impl Into<String>,
primary: Option<SourceSpan>,
) -> SchemaDiagnostics {
let diagnostic = Diagnostic::new(
DiagnosticCategory::ResourceLimit,
DiagnosticCode::new(code).expect("static schema diagnostic code is valid"),
message,
);
SchemaDiagnostics::one(SchemaDiagnostic::new(diagnostic, primary))
}