use std::{borrow::Cow, collections::BTreeMap, path::Path};
use crate::engine::{
grammar::CompiledGrammar,
grammar_closure::{AvailabilityStep, ClosureMemberTraits, MAX_AVAILABILITY_STEPS},
grammar_ir::decode_compiled_grammar,
state::{GrammarId, RuleId},
};
pub const MAGIC: &[u8; 4] = b"MRKB";
pub const FORMAT_VERSION: u16 = 4;
pub const CODEC_NONE: u32 = 0;
pub const CODEC_DEFLATE_ZLIB: u32 = 1;
pub const GRAMMAR_BLOB_COMPILED_IR: u32 = 1;
pub const SECTION_STRINGS: u32 = 3;
pub const SECTION_SCOPES: u32 = 4;
pub const SECTION_LANGUAGES: u32 = 5;
pub const SECTION_GRAMMAR_BLOBS: u32 = 6;
pub const SECTION_LICENSES: u32 = 7;
pub const SECTION_GRAMMAR_GRAPHS: u32 = 8;
const CLOSURE_REPOSITORY_CONTEXTS: u32 = 1 << 31;
const CLOSURE_BASE_REFERENCE: u32 = 1 << 30;
const CLOSURE_INJECTS: u32 = 1 << 29;
const CLOSURE_BLOB_MASK: u32 = CLOSURE_INJECTS - 1;
const NO_AVAILABILITY: u32 = u32::MAX;
const AVAILABILITY_REPOSITORY: u32 = 1 << 31;
const NO_SKELETON: u32 = u32::MAX;
const HEADER_LEN: usize = 32;
const SECTION_ENTRY_LEN: usize = 24;
const NO_STRING: u32 = u32::MAX;
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct BundleHeader {
pub format_version: u16,
pub section_count: u16,
pub source_hash: u64,
pub bundle_hash: u64,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct SectionEntry {
pub id: u32,
pub offset: u64,
pub len: u64,
}
#[derive(Debug, Clone, PartialEq, Eq, Default)]
pub struct Bundle {
pub source_hash: u64,
pub bundle_hash: u64,
pub strings: StringTable,
pub scopes: ScopeTable,
pub languages: Vec<LanguageEntry>,
pub grammar_blobs: Vec<GrammarBlob>,
pub grammar_graphs: Vec<GrammarGraph>,
pub licenses: Vec<LicenseEntry>,
}
#[derive(Debug, Clone, PartialEq, Eq, Default)]
pub struct StringTable {
bytes: Cow<'static, [u8]>,
}
impl StringTable {
fn len(&self) -> usize {
self.bytes.get(..4).map_or(0, |bytes| {
u32::from_le_bytes(bytes.try_into().expect("four bytes")) as usize
})
}
fn get(&self, id: u32) -> Option<&str> {
let index = id as usize;
let count = self.len();
if index >= count {
return None;
}
let payload_start = 8 + count * 4;
let start = read_u32_at(&self.bytes, 8 + index * 4).expect("validated offset") as usize;
let end = if index + 1 == count {
self.bytes.len() - payload_start
} else {
read_u32_at(&self.bytes, 8 + (index + 1) * 4).expect("validated offset") as usize
};
Some(
std::str::from_utf8(&self.bytes[payload_start + start..payload_start + end])
.expect("validated string"),
)
}
}
#[derive(Debug, Clone, PartialEq, Eq, Default)]
pub struct ScopeTable {
bytes: Cow<'static, [u8]>,
}
impl ScopeTable {
pub fn len(&self) -> usize {
self.bytes.len().saturating_sub(4) / 4
}
fn iter(&self) -> impl Iterator<Item = u32> + '_ {
self.bytes
.get(4..)
.unwrap_or_default()
.as_chunks::<4>()
.0
.iter()
.map(|id| u32::from_le_bytes(*id))
}
}
#[derive(Debug, Clone, PartialEq, Eq, Default)]
pub struct GrammarGraph {
pub closure: Vec<ClosureMember>,
pub repository_walk_skeleton: Option<Cow<'static, [u8]>>,
pub top_level_availability: Option<Vec<AvailabilityStep>>,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct ClosureMember {
pub blob: u32,
pub traits: ClosureMemberTraits,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct LanguageEntry {
pub canonical: String,
pub scope_name: String,
pub aliases: Vec<String>,
pub extensions: Vec<String>,
pub basenames: Vec<String>,
pub first_line_pattern: Option<String>,
pub grammar_blob: u32,
pub license: u32,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct GrammarBlob {
pub language: String,
pub scope_name: String,
pub codec: u32,
pub flags: u32,
pub raw_len: u32,
pub bytes: Vec<u8>,
pub pattern_count: u32,
pub dfa_count: u32,
pub fallback_count: u32,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct LicenseEntry {
pub language: String,
pub source_path: String,
pub upstream_url: String,
pub spdx_id: String,
pub license_text: String,
pub source_revision: String,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum BundleError {
TooShort,
BadMagic,
UnsupportedVersion(u16),
SectionTableOutOfBounds,
SectionOutOfBounds { id: u32 },
MissingSection(u32),
BadUtf8,
UnknownLanguage(String),
BadStringId(u32),
BadGrammarBlobId(u32),
BadLicenseId(u32),
BadCodec(u32),
BadMetadata(&'static str),
BadGrammarFlags { language: String, flags: u32 },
Inflate { language: String },
Truncated(&'static str),
TrailingBytes(&'static str),
GrammarIr { language: String, message: String },
BadGrammarGraph(u32),
}
#[derive(Debug, Clone)]
pub struct BundleGrammarRegistry {
bundle: Bundle,
cache: BTreeMap<String, CompiledGrammar>,
}
impl BundleGrammarRegistry {
pub fn new(bundle: Bundle) -> Self {
Self {
bundle,
cache: BTreeMap::new(),
}
}
pub fn bundle(&self) -> &Bundle {
&self.bundle
}
pub fn cached_grammar_count(&self) -> usize {
self.cache.len()
}
pub fn grammar(&mut self, language: &str) -> Result<&CompiledGrammar, BundleError> {
let canonical = self
.bundle
.canonical_language(language)
.ok_or_else(|| BundleError::UnknownLanguage(language.to_owned()))?
.to_owned();
if !self.cache.contains_key(&canonical) {
let blob = self
.bundle
.grammar_blob_for_language(&canonical)
.ok_or(BundleError::BadGrammarBlobId(u32::MAX))?;
let id = GrammarId(self.cache.len() as u16);
let grammar = blob.compiled_grammar(id)?;
self.cache.insert(canonical.clone(), grammar);
}
Ok(self
.cache
.get(&canonical)
.expect("grammar inserted before lookup"))
}
pub fn grammar_by_scope(&mut self, scope_name: &str) -> Result<&CompiledGrammar, BundleError> {
let canonical = self
.bundle
.languages
.iter()
.find(|entry| entry.scope_name == scope_name)
.map(|entry| entry.canonical.clone())
.ok_or_else(|| BundleError::UnknownLanguage(scope_name.to_owned()))?;
self.grammar(&canonical)
}
}
impl GrammarBlob {
pub fn compiled_grammar(&self, id: GrammarId) -> Result<CompiledGrammar, BundleError> {
if self.flags != GRAMMAR_BLOB_COMPILED_IR {
return Err(BundleError::BadGrammarFlags {
language: self.language.clone(),
flags: self.flags,
});
}
let bytes = self.decoded_bytes()?;
decode_compiled_grammar(id, &bytes).map_err(|error| BundleError::GrammarIr {
language: self.language.clone(),
message: error.to_string(),
})
}
pub fn decoded_bytes(&self) -> Result<Cow<'_, [u8]>, BundleError> {
match self.codec {
CODEC_NONE => Ok(Cow::Borrowed(&self.bytes)),
CODEC_DEFLATE_ZLIB => {
let inflate_error = || BundleError::Inflate {
language: self.language.clone(),
};
let mut bytes = vec![0; self.raw_len as usize];
let len = miniz_oxide::inflate::decompress_slice_iter_to_slice(
&mut bytes,
std::iter::once(self.bytes.as_slice()),
true,
false,
)
.map_err(|_| inflate_error())?;
if len != bytes.len() {
return Err(inflate_error());
}
Ok(Cow::Owned(bytes))
}
other => Err(BundleError::BadCodec(other)),
}
}
}
impl Bundle {
pub fn parse(bytes: &[u8]) -> Result<Self, BundleError> {
Self::parse_with_storage(bytes, |bytes| Cow::Owned(bytes.to_vec()))
}
pub(crate) fn parse_static(bytes: &'static [u8]) -> Result<Self, BundleError> {
Self::parse_with_storage(bytes, Cow::Borrowed)
}
fn parse_with_storage<'a>(
bytes: &'a [u8],
store_bytes: impl Fn(&'a [u8]) -> Cow<'static, [u8]>,
) -> Result<Self, BundleError> {
let (header, sections) = read_header_and_sections(bytes)?;
if header.format_version != FORMAT_VERSION {
return Err(BundleError::UnsupportedVersion(header.format_version));
}
let strings =
decode_string_table(store_bytes(section(bytes, §ions, SECTION_STRINGS)?))?;
let scopes = decode_scope_table(
store_bytes(section(bytes, §ions, SECTION_SCOPES)?),
&strings,
)?;
let languages =
decode_language_table(section(bytes, §ions, SECTION_LANGUAGES)?, &strings)?;
let grammar_blobs =
decode_grammar_blobs(section(bytes, §ions, SECTION_GRAMMAR_BLOBS)?, &strings)?;
let licenses =
decode_license_table(section(bytes, §ions, SECTION_LICENSES)?, &strings)?;
let grammar_graphs = decode_grammar_graphs(
section(bytes, §ions, SECTION_GRAMMAR_GRAPHS)?,
grammar_blobs.len(),
store_bytes,
)?;
Ok(Self {
source_hash: header.source_hash,
bundle_hash: header.bundle_hash,
strings,
scopes,
languages,
grammar_blobs,
grammar_graphs,
licenses,
})
}
pub(crate) fn validate(&self) -> Result<(), BundleError> {
let bad = BundleError::BadMetadata;
if self.grammar_blobs.len() > 4096 || self.languages.len() > 4096 {
return Err(bad("catalog exceeds 4096 grammars or languages"));
}
let mut ids = std::collections::BTreeSet::new();
for entry in &self.languages {
if entry.canonical.is_empty()
|| normalize_token(&entry.canonical) != entry.canonical
|| !ids.insert(&entry.canonical)
{
return Err(bad("invalid or duplicate public language ID"));
}
let blob = self
.grammar_blobs
.get(entry.grammar_blob as usize)
.ok_or(BundleError::BadGrammarBlobId(entry.grammar_blob))?;
if blob.scope_name != entry.scope_name {
return Err(bad("language root scope mismatch"));
}
if entry.license as usize >= self.licenses.len() {
return Err(BundleError::BadLicenseId(entry.license));
}
}
let mut total = 0usize;
for (index, blob) in self.grammar_blobs.iter().enumerate() {
total = total.saturating_add(blob.raw_len as usize);
if blob.raw_len > 16 * 1024 * 1024 || total > 256 * 1024 * 1024 {
return Err(bad("decoded grammar size exceeds validation limit"));
}
let grammar = blob.compiled_grammar(GrammarId(0))?;
if grammar.scope_name != blob.scope_name {
return Err(bad("grammar root scope mismatch"));
}
if let Some(skeleton) = &self.grammar_graphs[index].repository_walk_skeleton {
decode_compiled_grammar(GrammarId(0), skeleton).map_err(|error| {
BundleError::GrammarIr {
language: blob.language.clone(),
message: error.to_string(),
}
})?;
}
}
Ok(())
}
pub fn to_bytes(&self) -> Vec<u8> {
let strings = interned_strings(self);
let sections = vec![
(SECTION_STRINGS, encode_string_table(&strings)),
(
SECTION_SCOPES,
encode_scope_table(&self.scopes, &self.strings, &strings),
),
(
SECTION_LANGUAGES,
encode_language_table(&self.languages, &strings),
),
(
SECTION_GRAMMAR_BLOBS,
encode_grammar_blobs(&self.grammar_blobs, &strings),
),
(
SECTION_LICENSES,
encode_license_table(&self.licenses, &strings),
),
(
SECTION_GRAMMAR_GRAPHS,
encode_grammar_graphs(&self.grammar_graphs),
),
];
let bundle_hash = hash_sections(§ions);
write_container(self.source_hash, bundle_hash, sections)
}
pub fn version_stamp(&self) -> String {
format!("{:016x}", self.bundle_hash)
}
pub fn available_languages(&self) -> Vec<&str> {
self.languages
.iter()
.map(|language| language.canonical.as_str())
.collect()
}
pub fn canonical_language(&self, language: &str) -> Option<&str> {
let language = normalize_token(language);
if let Some(entry) = self
.languages
.iter()
.find(|entry| entry.canonical == language)
{
return Some(entry.canonical.as_str());
}
self.languages.iter().find_map(|entry| {
entry
.aliases
.iter()
.any(|alias| alias == &language)
.then_some(entry.canonical.as_str())
})
}
pub fn has_language(&self, language: &str) -> bool {
self.canonical_language(language).is_some()
}
pub fn detect_language_from_path(&self, path: &str) -> Option<&str> {
let name = Path::new(path).file_name()?.to_str()?;
let lower_name = name.to_ascii_lowercase();
if let Some(language) = self.languages.iter().find_map(|entry| {
entry
.basenames
.iter()
.any(|basename| basename.eq_ignore_ascii_case(name))
.then_some(entry.canonical.as_str())
}) {
return Some(language);
}
self.languages
.iter()
.filter_map(|entry| {
entry
.extensions
.iter()
.chain(entry.basenames.iter().filter(|name| name.contains('.')))
.filter_map(|extension| extension_match_len(&lower_name, extension))
.max()
.map(|len| (len, entry.canonical.as_str()))
})
.max_by_key(|(len, _)| *len)
.map(|(_, language)| language)
}
pub fn grammar_blob_for_language(&self, language: &str) -> Option<&GrammarBlob> {
self.grammar_blobs
.get(self.grammar_blob_index_for_language(language)?)
}
pub fn grammar_blob_index_for_language(&self, language: &str) -> Option<usize> {
let canonical = self.canonical_language(language)?;
let entry = self
.languages
.iter()
.find(|entry| entry.canonical == canonical)?;
let index = entry.grammar_blob as usize;
(index < self.grammar_blobs.len()).then_some(index)
}
pub fn grammar_blob_for_scope(&self, scope_name: &str) -> Option<&GrammarBlob> {
self.grammar_blobs
.iter()
.find(|blob| blob.scope_name == scope_name)
}
}
pub fn read_header(bytes: &[u8]) -> Result<BundleHeader, BundleError> {
read_header_and_sections(bytes).map(|(header, _)| header)
}
pub fn fnv1a64(bytes: &[u8]) -> u64 {
let mut hash = 0xcbf2_9ce4_8422_2325u64;
for byte in bytes {
hash ^= u64::from(*byte);
hash = hash.wrapping_mul(0x0000_0100_0000_01b3);
}
hash
}
fn read_header_and_sections(
bytes: &[u8],
) -> Result<(BundleHeader, Vec<SectionEntry>), BundleError> {
if bytes.len() < HEADER_LEN {
return Err(BundleError::TooShort);
}
if &bytes[..4] != MAGIC {
return Err(BundleError::BadMagic);
}
let format_version = read_u16_at(bytes, 4)?;
let section_count = read_u16_at(bytes, 6)?;
let source_hash = read_u64_at(bytes, 8)?;
let bundle_hash = read_u64_at(bytes, 16)?;
let table_len = usize::from(section_count)
.checked_mul(SECTION_ENTRY_LEN)
.ok_or(BundleError::SectionTableOutOfBounds)?;
let table_end = HEADER_LEN
.checked_add(table_len)
.ok_or(BundleError::SectionTableOutOfBounds)?;
if table_end > bytes.len() {
return Err(BundleError::SectionTableOutOfBounds);
}
let mut sections = Vec::with_capacity(usize::from(section_count));
let mut cursor = HEADER_LEN;
for _ in 0..section_count {
let id = read_u32_at(bytes, cursor)?;
let offset = read_u64_at(bytes, cursor + 8)?;
let len = read_u64_at(bytes, cursor + 16)?;
let end = offset
.checked_add(len)
.ok_or(BundleError::SectionOutOfBounds { id })?;
if end > bytes.len() as u64 {
return Err(BundleError::SectionOutOfBounds { id });
}
sections.push(SectionEntry { id, offset, len });
cursor += SECTION_ENTRY_LEN;
}
Ok((
BundleHeader {
format_version,
section_count,
source_hash,
bundle_hash,
},
sections,
))
}
fn section<'a>(
bytes: &'a [u8],
sections: &[SectionEntry],
id: u32,
) -> Result<&'a [u8], BundleError> {
let section = sections
.iter()
.find(|section| section.id == id)
.ok_or(BundleError::MissingSection(id))?;
let start = section.offset as usize;
let end = start + section.len as usize;
Ok(&bytes[start..end])
}
fn write_container(source_hash: u64, bundle_hash: u64, sections: Vec<(u32, Vec<u8>)>) -> Vec<u8> {
let table_len = sections.len() * SECTION_ENTRY_LEN;
let mut offset = align_to(HEADER_LEN + table_len, 8);
let mut entries = Vec::with_capacity(sections.len());
for (id, bytes) in §ions {
entries.push(SectionEntry {
id: *id,
offset: offset as u64,
len: bytes.len() as u64,
});
offset = align_to(offset + bytes.len(), 8);
}
let mut out = Vec::with_capacity(offset);
out.extend_from_slice(MAGIC);
write_u16(&mut out, FORMAT_VERSION);
write_u16(&mut out, sections.len() as u16);
write_u64(&mut out, source_hash);
write_u64(&mut out, bundle_hash);
write_u64(&mut out, 0);
for entry in &entries {
write_u32(&mut out, entry.id);
write_u32(&mut out, 0);
write_u64(&mut out, entry.offset);
write_u64(&mut out, entry.len);
}
for ((_, bytes), entry) in sections.into_iter().zip(entries) {
while out.len() < entry.offset as usize {
out.push(0);
}
out.extend_from_slice(&bytes);
}
out
}
fn hash_sections(sections: &[(u32, Vec<u8>)]) -> u64 {
let mut bytes = Vec::new();
for (id, section) in sections {
write_u32(&mut bytes, *id);
write_u64(&mut bytes, section.len() as u64);
bytes.extend_from_slice(section);
}
fnv1a64(&bytes)
}
fn interned_strings(bundle: &Bundle) -> Vec<String> {
let mut strings = BTreeMap::<String, ()>::new();
let mut insert = |value: &str| {
strings.insert(value.to_owned(), ());
};
for scope in bundle.scopes.iter() {
insert(bundle.strings.get(scope).expect("validated scope string"));
}
for language in &bundle.languages {
insert(&language.canonical);
insert(&language.scope_name);
if let Some(first_line) = &language.first_line_pattern {
insert(first_line);
}
for value in language
.aliases
.iter()
.chain(language.extensions.iter())
.chain(language.basenames.iter())
{
insert(value);
}
}
for blob in &bundle.grammar_blobs {
insert(&blob.language);
insert(&blob.scope_name);
}
for license in &bundle.licenses {
insert(&license.language);
insert(&license.source_path);
insert(&license.upstream_url);
insert(&license.spdx_id);
insert(&license.license_text);
insert(&license.source_revision);
}
strings.into_keys().collect()
}
fn string_id(strings: &[String], value: &str) -> u32 {
strings
.binary_search_by(|candidate| candidate.as_str().cmp(value))
.expect("bundle string should be interned") as u32
}
fn optional_string_id(strings: &[String], value: Option<&str>) -> u32 {
value.map_or(NO_STRING, |value| string_id(strings, value))
}
fn string_by_id(strings: &StringTable, id: u32) -> Result<String, BundleError> {
strings
.get(id)
.map(str::to_owned)
.ok_or(BundleError::BadStringId(id))
}
fn optional_string_by_id(strings: &StringTable, id: u32) -> Result<Option<String>, BundleError> {
if id == NO_STRING {
Ok(None)
} else {
string_by_id(strings, id).map(Some)
}
}
fn encode_string_table(strings: &[String]) -> Vec<u8> {
let mut bytes = Vec::new();
let mut payload = Vec::new();
let mut offsets = Vec::with_capacity(strings.len());
for string in strings {
offsets.push(payload.len() as u32);
payload.extend_from_slice(string.as_bytes());
}
write_u32(&mut bytes, strings.len() as u32);
write_u32(&mut bytes, payload.len() as u32);
for offset in offsets {
write_u32(&mut bytes, offset);
}
bytes.extend_from_slice(&payload);
bytes
}
fn decode_string_table(bytes: Cow<'static, [u8]>) -> Result<StringTable, BundleError> {
let mut cursor = Cursor::new(&bytes, "string table");
let count = cursor.u32()? as usize;
let payload_len = cursor.u32()? as usize;
let index_len = count
.checked_mul(4)
.ok_or(BundleError::Truncated("string index"))?;
let offsets = cursor.bytes(index_len)?;
let payload = cursor.bytes(payload_len)?;
cursor.finish()?;
let text = std::str::from_utf8(payload).map_err(|_| BundleError::BadUtf8)?;
let mut previous = 0;
for index in 0..count {
let offset = read_u32_at(offsets, index * 4)? as usize;
if offset < previous || offset > payload_len || (index == 0 && offset != 0) {
return Err(BundleError::Truncated("string table payload"));
}
if !text.is_char_boundary(offset) {
return Err(BundleError::BadUtf8);
}
previous = offset;
}
if count == 0 && payload_len != 0 {
return Err(BundleError::TrailingBytes("string table payload"));
}
Ok(StringTable { bytes })
}
fn encode_scope_table(scopes: &ScopeTable, source: &StringTable, strings: &[String]) -> Vec<u8> {
let mut bytes = Vec::new();
write_u32(&mut bytes, scopes.len() as u32);
for scope in scopes.iter() {
write_u32(
&mut bytes,
string_id(strings, source.get(scope).expect("validated scope string")),
);
}
bytes
}
fn decode_scope_table(
bytes: Cow<'static, [u8]>,
strings: &StringTable,
) -> Result<ScopeTable, BundleError> {
let mut cursor = Cursor::new(&bytes, "scope table");
let count = cursor.u32()? as usize;
let ids = cursor.bytes(
count
.checked_mul(4)
.ok_or(BundleError::Truncated("scope table"))?,
)?;
cursor.finish()?;
for id in ids.as_chunks::<4>().0 {
let id = u32::from_le_bytes(*id);
if id as usize >= strings.len() {
return Err(BundleError::BadStringId(id));
}
}
Ok(ScopeTable { bytes })
}
fn encode_language_table(languages: &[LanguageEntry], strings: &[String]) -> Vec<u8> {
let mut bytes = Vec::new();
write_u32(&mut bytes, languages.len() as u32);
for language in languages {
write_u32(&mut bytes, string_id(strings, &language.canonical));
write_u32(&mut bytes, string_id(strings, &language.scope_name));
write_u32(&mut bytes, language.grammar_blob);
write_u32(&mut bytes, language.license);
write_u32(
&mut bytes,
optional_string_id(strings, language.first_line_pattern.as_deref()),
);
write_string_id_vec(&mut bytes, strings, &language.aliases);
write_string_id_vec(&mut bytes, strings, &language.extensions);
write_string_id_vec(&mut bytes, strings, &language.basenames);
}
bytes
}
fn decode_language_table(
bytes: &[u8],
strings: &StringTable,
) -> Result<Vec<LanguageEntry>, BundleError> {
let mut cursor = Cursor::new(bytes, "language table");
let count = cursor.count(32)?;
let mut languages = Vec::with_capacity(count as usize);
for _ in 0..count {
languages.push(LanguageEntry {
canonical: string_by_id(strings, cursor.u32()?)?,
scope_name: string_by_id(strings, cursor.u32()?)?,
grammar_blob: cursor.u32()?,
license: cursor.u32()?,
first_line_pattern: optional_string_by_id(strings, cursor.u32()?)?,
aliases: read_string_id_vec(&mut cursor, strings)?,
extensions: read_string_id_vec(&mut cursor, strings)?,
basenames: read_string_id_vec(&mut cursor, strings)?,
});
}
cursor.finish()?;
Ok(languages)
}
fn encode_grammar_blobs(blobs: &[GrammarBlob], strings: &[String]) -> Vec<u8> {
let record_len = 48usize;
let payload_start = 4 + blobs.len() * record_len;
let mut records = Vec::new();
let mut payload = Vec::new();
write_u32(&mut records, blobs.len() as u32);
for blob in blobs {
write_u32(&mut records, string_id(strings, &blob.language));
write_u32(&mut records, string_id(strings, &blob.scope_name));
write_u32(&mut records, blob.codec);
write_u32(&mut records, blob.flags);
write_u32(&mut records, blob.raw_len);
write_u64(&mut records, (payload_start + payload.len()) as u64);
write_u64(&mut records, blob.bytes.len() as u64);
write_u32(&mut records, blob.pattern_count);
write_u32(&mut records, blob.dfa_count);
write_u32(&mut records, blob.fallback_count);
payload.extend_from_slice(&blob.bytes);
}
records.extend_from_slice(&payload);
records
}
fn decode_grammar_blobs(
bytes: &[u8],
strings: &StringTable,
) -> Result<Vec<GrammarBlob>, BundleError> {
let mut cursor = Cursor::new(bytes, "grammar blobs");
let count = cursor.count(48)?;
let mut records = Vec::with_capacity(count as usize);
for _ in 0..count {
records.push((
cursor.u32()?,
cursor.u32()?,
cursor.u32()?,
cursor.u32()?,
cursor.u32()?,
cursor.u64()? as usize,
cursor.u64()? as usize,
cursor.u32()?,
cursor.u32()?,
cursor.u32()?,
));
}
let mut blobs = Vec::with_capacity(records.len());
for (
language,
scope_name,
codec,
flags,
raw_len,
offset,
len,
pattern_count,
dfa_count,
fallback_count,
) in records
{
if offset.checked_add(len).is_none_or(|end| end > bytes.len()) {
return Err(BundleError::Truncated("grammar blob payload"));
}
blobs.push(GrammarBlob {
language: string_by_id(strings, language)?,
scope_name: string_by_id(strings, scope_name)?,
codec,
flags,
raw_len,
bytes: bytes[offset..offset + len].to_vec(),
pattern_count,
dfa_count,
fallback_count,
});
}
Ok(blobs)
}
fn encode_grammar_graphs(graphs: &[GrammarGraph]) -> Vec<u8> {
let mut bytes = Vec::new();
write_u32(&mut bytes, graphs.len() as u32);
for graph in graphs {
write_u32(&mut bytes, graph.closure.len() as u32);
for member in &graph.closure {
assert!(
member.blob <= CLOSURE_BLOB_MASK,
"closure blob index fits below the flag bits"
);
let mut value = member.blob;
if member.traits.repository_contexts {
value |= CLOSURE_REPOSITORY_CONTEXTS;
}
if member.traits.base_reference {
value |= CLOSURE_BASE_REFERENCE;
}
if member.traits.injects {
value |= CLOSURE_INJECTS;
}
write_u32(&mut bytes, value);
}
if let Some(skeleton) = &graph.repository_walk_skeleton {
assert!(
skeleton.len() < NO_SKELETON as usize,
"walk skeleton length fits"
);
write_u32(&mut bytes, skeleton.len() as u32);
bytes.extend_from_slice(skeleton);
} else {
write_u32(&mut bytes, NO_SKELETON);
}
let Some(steps) = &graph.top_level_availability else {
write_u32(&mut bytes, NO_AVAILABILITY);
continue;
};
write_u32(&mut bytes, steps.len() as u32);
for step in steps {
match step {
AvailabilityStep::Rule(rule_id) => {
assert!(
rule_id.0 < AVAILABILITY_REPOSITORY,
"rule id fits the step tag"
);
write_u32(&mut bytes, rule_id.0);
}
AvailabilityStep::Repository(name) => {
write_u32(&mut bytes, AVAILABILITY_REPOSITORY | name.len() as u32);
bytes.extend_from_slice(name.as_bytes());
}
}
}
}
bytes
}
fn decode_grammar_graphs<'a>(
bytes: &'a [u8],
blob_count: usize,
store_bytes: impl Fn(&'a [u8]) -> Cow<'static, [u8]>,
) -> Result<Vec<GrammarGraph>, BundleError> {
let mut cursor = Cursor::new(bytes, "grammar graphs");
let count = cursor.u32()? as usize;
if count != blob_count {
return Err(BundleError::BadGrammarGraph(count as u32));
}
let mut graphs = Vec::with_capacity(count);
for root in 0..count {
let bad = || BundleError::BadGrammarGraph(root as u32);
let member_count = cursor.u32()? as usize;
if member_count > blob_count {
return Err(bad());
}
let mut closure = Vec::with_capacity(member_count);
for _ in 0..member_count {
let value = cursor.u32()?;
let blob = value & CLOSURE_BLOB_MASK;
let ascending = closure
.last()
.is_none_or(|previous: &ClosureMember| previous.blob < blob);
if blob as usize >= blob_count || !ascending {
return Err(bad());
}
closure.push(ClosureMember {
blob,
traits: ClosureMemberTraits {
repository_contexts: value & CLOSURE_REPOSITORY_CONTEXTS != 0,
base_reference: value & CLOSURE_BASE_REFERENCE != 0,
injects: value & CLOSURE_INJECTS != 0,
},
});
}
if !closure.iter().any(|member| member.blob as usize == root) {
return Err(bad());
}
let skeleton_len = cursor.u32()?;
let repository_walk_skeleton = (skeleton_len != NO_SKELETON)
.then(|| cursor.bytes(skeleton_len as usize).map(&store_bytes))
.transpose()?;
let step_count = cursor.u32()?;
let top_level_availability = if step_count == NO_AVAILABILITY {
None
} else {
if step_count as usize > MAX_AVAILABILITY_STEPS {
return Err(bad());
}
let mut steps = Vec::with_capacity(step_count as usize);
for _ in 0..step_count {
let value = cursor.u32()?;
steps.push(if value & AVAILABILITY_REPOSITORY == 0 {
AvailabilityStep::Rule(RuleId(value))
} else {
let name = cursor.bytes((value & !AVAILABILITY_REPOSITORY) as usize)?;
AvailabilityStep::Repository(
std::str::from_utf8(name)
.map_err(|_| BundleError::BadUtf8)?
.to_owned(),
)
});
}
Some(steps)
};
graphs.push(GrammarGraph {
closure,
repository_walk_skeleton,
top_level_availability,
});
}
cursor.finish()?;
Ok(graphs)
}
fn encode_license_table(licenses: &[LicenseEntry], strings: &[String]) -> Vec<u8> {
let mut bytes = Vec::new();
write_u32(&mut bytes, licenses.len() as u32);
for license in licenses {
write_u32(&mut bytes, string_id(strings, &license.language));
write_u32(&mut bytes, string_id(strings, &license.source_path));
write_u32(&mut bytes, string_id(strings, &license.upstream_url));
write_u32(&mut bytes, string_id(strings, &license.spdx_id));
write_u32(&mut bytes, string_id(strings, &license.license_text));
write_u32(&mut bytes, string_id(strings, &license.source_revision));
}
bytes
}
fn decode_license_table(
bytes: &[u8],
strings: &StringTable,
) -> Result<Vec<LicenseEntry>, BundleError> {
let mut cursor = Cursor::new(bytes, "license table");
let count = cursor.count(24)?;
let mut licenses = Vec::with_capacity(count as usize);
for _ in 0..count {
licenses.push(LicenseEntry {
language: string_by_id(strings, cursor.u32()?)?,
source_path: string_by_id(strings, cursor.u32()?)?,
upstream_url: string_by_id(strings, cursor.u32()?)?,
spdx_id: string_by_id(strings, cursor.u32()?)?,
license_text: string_by_id(strings, cursor.u32()?)?,
source_revision: string_by_id(strings, cursor.u32()?)?,
});
}
cursor.finish()?;
Ok(licenses)
}
fn write_string_id_vec(out: &mut Vec<u8>, strings: &[String], values: &[String]) {
write_u32(out, values.len() as u32);
for value in values {
write_u32(out, string_id(strings, value));
}
}
fn read_string_id_vec(
cursor: &mut Cursor<'_>,
strings: &StringTable,
) -> Result<Vec<String>, BundleError> {
let count = cursor.count(4)?;
let mut values = Vec::with_capacity(count as usize);
for _ in 0..count {
values.push(string_by_id(strings, cursor.u32()?)?);
}
Ok(values)
}
fn extension_match_len(filename: &str, extension: &str) -> Option<usize> {
let extension = extension.trim_start_matches('.').to_ascii_lowercase();
(!extension.is_empty() && filename.ends_with(&format!(".{extension}")))
.then_some(extension.len())
}
fn normalize_token(token: &str) -> String {
token.trim().trim_start_matches('.').to_ascii_lowercase()
}
fn align_to(value: usize, alignment: usize) -> usize {
value.div_ceil(alignment) * alignment
}
fn write_u16(out: &mut Vec<u8>, value: u16) {
out.extend_from_slice(&value.to_le_bytes());
}
fn write_u32(out: &mut Vec<u8>, value: u32) {
out.extend_from_slice(&value.to_le_bytes());
}
fn write_u64(out: &mut Vec<u8>, value: u64) {
out.extend_from_slice(&value.to_le_bytes());
}
fn read_u16_at(bytes: &[u8], offset: usize) -> Result<u16, BundleError> {
Ok(u16::from_le_bytes(
bytes
.get(offset..offset + 2)
.ok_or(BundleError::TooShort)?
.try_into()
.expect("slice length checked"),
))
}
fn read_u32_at(bytes: &[u8], offset: usize) -> Result<u32, BundleError> {
Ok(u32::from_le_bytes(
bytes
.get(offset..offset + 4)
.ok_or(BundleError::TooShort)?
.try_into()
.expect("slice length checked"),
))
}
fn read_u64_at(bytes: &[u8], offset: usize) -> Result<u64, BundleError> {
Ok(u64::from_le_bytes(
bytes
.get(offset..offset + 8)
.ok_or(BundleError::TooShort)?
.try_into()
.expect("slice length checked"),
))
}
struct Cursor<'a> {
bytes: &'a [u8],
cursor: usize,
name: &'static str,
}
impl<'a> Cursor<'a> {
fn new(bytes: &'a [u8], name: &'static str) -> Self {
Self {
bytes,
cursor: 0,
name,
}
}
fn count(&mut self, record_bytes: usize) -> Result<u32, BundleError> {
let count = self.u32()?;
if count as usize > (self.bytes.len() - self.cursor) / record_bytes {
return Err(BundleError::Truncated(self.name));
}
Ok(count)
}
fn u32(&mut self) -> Result<u32, BundleError> {
let value =
read_u32_at(self.bytes, self.cursor).map_err(|_| BundleError::Truncated(self.name))?;
self.cursor += 4;
Ok(value)
}
fn u64(&mut self) -> Result<u64, BundleError> {
let value =
read_u64_at(self.bytes, self.cursor).map_err(|_| BundleError::Truncated(self.name))?;
self.cursor += 8;
Ok(value)
}
fn bytes(&mut self, len: usize) -> Result<&'a [u8], BundleError> {
let end = self
.cursor
.checked_add(len)
.ok_or(BundleError::Truncated(self.name))?;
let bytes = self
.bytes
.get(self.cursor..end)
.ok_or(BundleError::Truncated(self.name))?;
self.cursor = end;
Ok(bytes)
}
fn finish(&self) -> Result<(), BundleError> {
if self.cursor == self.bytes.len() {
Ok(())
} else {
Err(BundleError::TrailingBytes(self.name))
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::engine::{grammar::load_dev_grammar_from_str, grammar_ir::encode_compiled_grammar};
fn grammar_ir(scope_name: &str, pattern: &str) -> Vec<u8> {
let source = format!(
r#"{{"scopeName":"{scope_name}","patterns":[{{"match":"{pattern}","name":"constant.language.fixture"}}]}}"#
);
let grammar = load_dev_grammar_from_str(GrammarId(0), &source).unwrap();
encode_compiled_grammar(&grammar).unwrap()
}
fn sample_bundle() -> Bundle {
let grammar_bytes = grammar_ir("source.rust", "true");
Bundle {
source_hash: 7,
bundle_hash: 0,
strings: decode_string_table(encode_string_table(&["source.rust".to_owned()]).into())
.unwrap(),
scopes: ScopeTable {
bytes: vec![1, 0, 0, 0, 0, 0, 0, 0].into(),
},
languages: vec![LanguageEntry {
canonical: "rust".to_owned(),
scope_name: "source.rust".to_owned(),
aliases: vec!["rs".to_owned()],
extensions: vec!["rs".to_owned()],
basenames: vec!["Cargo.toml".to_owned()],
first_line_pattern: None,
grammar_blob: 0,
license: 0,
}],
grammar_blobs: vec![GrammarBlob {
language: "rust".to_owned(),
scope_name: "source.rust".to_owned(),
codec: CODEC_NONE,
flags: GRAMMAR_BLOB_COMPILED_IR,
raw_len: grammar_bytes.len() as u32,
bytes: grammar_bytes,
pattern_count: 0,
dfa_count: 0,
fallback_count: 0,
}],
grammar_graphs: vec![GrammarGraph {
closure: vec![ClosureMember {
blob: 0,
traits: ClosureMemberTraits::default(),
}],
repository_walk_skeleton: Some(grammar_ir("source.rust", "").into()),
top_level_availability: Some(vec![
AvailabilityStep::Repository("entry".to_owned()),
AvailabilityStep::Rule(RuleId(0)),
]),
}],
licenses: vec![LicenseEntry {
language: "rust".to_owned(),
source_path: "assets/grammars/languages/rust.tmLanguage.json".to_owned(),
upstream_url: "https://example.invalid".to_owned(),
spdx_id: "MIT".to_owned(),
license_text: String::new(),
source_revision: "test".to_owned(),
}],
}
}
#[test]
fn reads_header_and_roundtrips_bundle() {
let bytes = sample_bundle().to_bytes();
let header = read_header(&bytes).unwrap();
assert_eq!(header.format_version, FORMAT_VERSION);
assert_eq!(header.section_count, 6);
assert_eq!(header.source_hash, 7);
assert_ne!(header.bundle_hash, 0);
let parsed = Bundle::parse(&bytes).unwrap();
assert_eq!(parsed.grammar_graphs, sample_bundle().grammar_graphs);
assert_eq!(
parsed.version_stamp(),
format!("{:016x}", header.bundle_hash)
);
assert_eq!(parsed.available_languages(), vec!["rust"]);
assert_eq!(parsed.canonical_language("RS"), Some("rust"));
assert_eq!(parsed.detect_language_from_path("src/lib.rs"), Some("rust"));
}
#[test]
fn static_bundle_borrows_tables_and_skeletons_and_matches_owned_parsing() {
let bytes = Box::leak(sample_bundle().to_bytes().into_boxed_slice());
let borrowed = Bundle::parse_static(bytes).unwrap();
let owned = Bundle::parse(bytes).unwrap();
assert_eq!(borrowed, owned);
assert_eq!(borrowed.to_bytes(), bytes);
let (_, sections) = read_header_and_sections(bytes).unwrap();
for (id, borrowed_bytes, owned_bytes) in [
(
SECTION_STRINGS,
&borrowed.strings.bytes,
&owned.strings.bytes,
),
(SECTION_SCOPES, &borrowed.scopes.bytes, &owned.scopes.bytes),
] {
assert!(matches!(borrowed_bytes, Cow::Borrowed(_)));
assert!(matches!(owned_bytes, Cow::Owned(_)));
assert_eq!(
borrowed_bytes.as_ptr(),
section(bytes, §ions, id).unwrap().as_ptr()
);
}
assert_eq!(
borrowed.scopes.iter().collect::<Vec<_>>(),
owned.scopes.iter().collect::<Vec<_>>()
);
assert!(matches!(
borrowed.grammar_graphs[0].repository_walk_skeleton,
Some(Cow::Borrowed(_))
));
assert!(matches!(
owned.grammar_graphs[0].repository_walk_skeleton,
Some(Cow::Owned(_))
));
for length in 0..bytes.len() {
assert_eq!(
Bundle::parse_static(&bytes[..length]),
Bundle::parse(&bytes[..length]),
"truncated bundle at {length}"
);
}
}
#[test]
fn rejects_bad_magic() {
let mut bytes = sample_bundle().to_bytes();
bytes[0] = b'X';
assert_eq!(Bundle::parse(&bytes), Err(BundleError::BadMagic));
}
#[test]
fn rejects_stale_bundle_version() {
let mut bytes = sample_bundle().to_bytes();
bytes[4..6].copy_from_slice(&(FORMAT_VERSION - 1).to_le_bytes());
assert_eq!(
Bundle::parse(&bytes),
Err(BundleError::UnsupportedVersion(FORMAT_VERSION - 1))
);
}
#[test]
fn registry_decodes_grammar_blob_lazily() {
let mut bundle = sample_bundle();
bundle.languages[0].canonical = "fixture".to_owned();
bundle.languages[0].aliases = vec!["fx".to_owned()];
bundle.grammar_blobs[0].language = "fixture".to_owned();
bundle.grammar_blobs[0].scope_name = "source.fixture".to_owned();
bundle.grammar_blobs[0].bytes = grammar_ir("source.fixture", "true");
bundle.grammar_blobs[0].raw_len = bundle.grammar_blobs[0].bytes.len() as u32;
let parsed = Bundle::parse(&bundle.to_bytes()).unwrap();
let mut registry = BundleGrammarRegistry::new(parsed);
let grammar = registry.grammar("fx").unwrap();
assert_eq!(grammar.scope_name, "source.fixture");
assert_eq!(grammar.patterns, vec![std::sync::Arc::<str>::from("true")]);
}
#[test]
fn registry_rejects_truncated_compiled_grammar_ir() {
let mut bundle = sample_bundle();
bundle.grammar_blobs[0].bytes.pop();
bundle.grammar_blobs[0].raw_len = bundle.grammar_blobs[0].bytes.len() as u32;
let parsed = Bundle::parse(&bundle.to_bytes()).unwrap();
let mut registry = BundleGrammarRegistry::new(parsed);
assert!(matches!(
registry.grammar("rust"),
Err(BundleError::GrammarIr { .. })
));
}
#[test]
fn rejects_malformed_grammar_graphs() {
let member = |blob| ClosureMember {
blob,
traits: ClosureMemberTraits::default(),
};
for closure in [vec![], vec![member(1)], vec![member(0), member(0)]] {
let mut bundle = sample_bundle();
bundle.grammar_graphs[0].closure = closure;
assert_eq!(
Bundle::parse(&bundle.to_bytes()),
Err(BundleError::BadGrammarGraph(0))
);
}
let mut bundle = sample_bundle();
bundle.grammar_graphs[0].top_level_availability =
Some(vec![
AvailabilityStep::Rule(RuleId(0));
MAX_AVAILABILITY_STEPS + 1
]);
assert_eq!(
Bundle::parse(&bundle.to_bytes()),
Err(BundleError::BadGrammarGraph(0))
);
let mut bundle = sample_bundle();
bundle.grammar_graphs.clear();
assert_eq!(
Bundle::parse(&bundle.to_bytes()),
Err(BundleError::BadGrammarGraph(0))
);
}
#[test]
#[cfg(any(feature = "bundled-grammars", feature = "bundle-tools"))]
fn inflates_compressed_blobs_to_their_recorded_length() {
let mut blob = sample_bundle().grammar_blobs.remove(0);
let raw = blob.bytes.clone();
blob.codec = CODEC_DEFLATE_ZLIB;
blob.bytes = miniz_oxide::deflate::compress_to_vec_zlib(&raw, 6);
assert_eq!(blob.decoded_bytes().unwrap().as_ref(), raw.as_slice());
assert_eq!(
blob.compiled_grammar(GrammarId(0)).unwrap().scope_name,
"source.rust"
);
for raw_len in [raw.len() - 1, raw.len() + 1] {
blob.raw_len = raw_len as u32;
assert!(matches!(
blob.decoded_bytes(),
Err(BundleError::Inflate { .. })
));
}
}
#[test]
fn string_table_borrows_utf8_and_handles_empty_strings() {
static BYTES: &[u8] = &[
3, 0, 0, 0, 2, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2, 0, 0, 0, 0xc3, 0xa9,
];
let table = decode_string_table(Cow::Borrowed(BYTES)).unwrap();
assert!(matches!(table.bytes, Cow::Borrowed(_)));
assert_eq!(table.get(0), Some(""));
assert_eq!(table.get(1), Some("é"));
assert_eq!(table.get(2), Some(""));
assert_eq!(table.get(3), None);
assert_eq!(table.get(1).unwrap().as_ptr(), BYTES[20..].as_ptr());
assert_eq!(decode_string_table(BYTES.to_vec().into()).unwrap(), table);
let empty = decode_string_table(encode_string_table(&[]).into()).unwrap();
assert_eq!(empty.len(), 0);
assert_eq!(empty.get(0), None);
}
#[test]
fn rejects_malformed_string_and_scope_tables() {
let raw = encode_string_table(&["é".to_owned(), "x".to_owned()]);
for (position, value) in [(0, u32::MAX), (4, u32::MAX), (8, 1), (12, 4), (12, 1)] {
let mut bytes = raw.clone();
bytes[position..position + 4].copy_from_slice(&value.to_le_bytes());
assert!(decode_string_table(bytes.into()).is_err());
}
let mut descending = encode_string_table(&["a".to_owned(), "b".to_owned(), "c".to_owned()]);
descending[16..20].copy_from_slice(&0u32.to_le_bytes());
assert!(decode_string_table(descending.into()).is_err());
let mut invalid_utf8 = raw.clone();
*invalid_utf8.last_mut().unwrap() = 0xff;
assert_eq!(
decode_string_table(invalid_utf8.into()),
Err(BundleError::BadUtf8)
);
let mut trailing = raw.clone();
trailing.push(0);
assert!(matches!(
decode_string_table(trailing.into()),
Err(BundleError::TrailingBytes(_))
));
for len in 0..raw.len() {
assert!(decode_string_table(raw[..len].to_vec().into()).is_err());
}
let strings = decode_string_table(raw.into()).unwrap();
assert_eq!(
decode_scope_table(Cow::Borrowed(&[1, 0, 0, 0, 2, 0, 0, 0]), &strings),
Err(BundleError::BadStringId(2))
);
assert!(decode_scope_table(u32::MAX.to_le_bytes().to_vec().into(), &strings).is_err());
}
#[test]
fn both_bundle_readers_reject_malformed_table_sections() {
let valid_strings = encode_string_table(&["é".to_owned(), "x".to_owned()]);
let mut bad_tables = Vec::new();
for (position, value) in [(0, u32::MAX), (4, u32::MAX), (8, 1), (12, 4), (12, 1)] {
let mut raw = valid_strings.clone();
raw[position..position + 4].copy_from_slice(&value.to_le_bytes());
bad_tables.push((SECTION_STRINGS, raw));
}
let mut invalid_utf8 = valid_strings.clone();
*invalid_utf8.last_mut().unwrap() = 0xff;
bad_tables.push((SECTION_STRINGS, invalid_utf8));
for (id, raw) in [
(SECTION_STRINGS, valid_strings),
(SECTION_SCOPES, vec![1, 0, 0, 0, 0, 0, 0, 0]),
] {
for len in 0..raw.len() {
bad_tables.push((id, raw[..len].to_vec()));
}
let mut trailing = raw;
trailing.push(0);
bad_tables.push((id, trailing));
}
bad_tables.push((SECTION_SCOPES, vec![1, 0, 0, 0, 255, 255, 255, 255]));
bad_tables.push((SECTION_SCOPES, u32::MAX.to_le_bytes().to_vec()));
let original = sample_bundle().to_bytes();
let (_, sections) = read_header_and_sections(&original).unwrap();
for (bad_id, raw) in bad_tables {
let tables = sections
.iter()
.map(|entry| {
let bytes = if entry.id == bad_id {
raw.clone()
} else {
section(&original, §ions, entry.id).unwrap().to_vec()
};
(entry.id, bytes)
})
.collect();
let bytes = Box::leak(write_container(0, 0, tables).into_boxed_slice());
let owned = Bundle::parse(bytes);
assert!(owned.is_err(), "malformed section {bad_id}");
assert_eq!(Bundle::parse_static(bytes), owned);
}
}
#[test]
#[cfg(feature = "bundled-grammars")]
fn embedded_metadata_roundtrips_through_owned_and_static_readers() {
let bytes = crate::grammars::embedded_bundle_bytes();
let input = bytes.to_vec();
let owned = Bundle::parse(&input).unwrap();
drop(input);
let borrowed = Bundle::parse_static(bytes).unwrap();
assert_eq!(owned, borrowed);
assert_eq!(owned.to_bytes(), bytes);
assert!(matches!(borrowed.strings.bytes, Cow::Borrowed(_)));
assert!(matches!(borrowed.scopes.bytes, Cow::Borrowed(_)));
assert!(matches!(owned.strings.bytes, Cow::Owned(_)));
assert!(matches!(owned.scopes.bytes, Cow::Owned(_)));
assert!(
borrowed
.grammar_graphs
.iter()
.filter_map(|graph| graph.repository_walk_skeleton.as_ref())
.all(|skeleton| matches!(skeleton, Cow::Borrowed(_)))
);
}
#[test]
fn deterministic_output() {
let bundle = sample_bundle();
assert_eq!(bundle.to_bytes(), bundle.to_bytes());
}
}