use std::collections::{BTreeMap, HashMap, VecDeque};
use byteorder::{ByteOrder, LittleEndian};
use crate::blocks::{BlockHeader, BLOCK_HEADER_SIZE, ID_BLOCK_SIZE};
use crate::io::ByteSource;
const SUMMARY_PREFIX: usize = 320;
const MAX_BLOCKS: usize = 2_000_000;
#[derive(Debug, Clone)]
pub struct BlockInfo {
pub address: u64,
pub block_type: String,
pub length: u64,
pub link_count: u64,
pub links: Vec<u64>,
pub link_labels: Vec<String>,
pub summary: String,
pub referenced_by: Vec<u64>,
pub data_size: u64,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct Gap {
pub address: u64,
pub length: u64,
}
#[derive(Debug, Clone)]
pub struct BlockMap {
pub blocks: Vec<BlockInfo>,
pub gaps: Vec<Gap>,
pub file_size: u64,
pub covered_bytes: u64,
pub warnings: Vec<String>,
pub unfinalized: bool,
}
impl BlockMap {
pub fn scan<S: ByteSource + ?Sized>(source: &S) -> Self {
Walk::new(source).run()
}
pub fn block_at(&self, address: u64) -> Option<&BlockInfo> {
self.blocks
.binary_search_by_key(&address, |b| b.address)
.ok()
.map(|i| &self.blocks[i])
}
pub fn type_counts(&self) -> Vec<(String, usize)> {
let mut counts: HashMap<&str, usize> = HashMap::new();
for block in &self.blocks {
*counts.entry(block.block_type.as_str()).or_default() += 1;
}
let mut counts: Vec<(String, usize)> = counts
.into_iter()
.map(|(k, v)| (k.to_string(), v))
.collect();
counts.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
counts
}
}
struct Raw {
address: u64,
block_type: [u8; 4],
length: u64,
link_count: u64,
links: Vec<u64>,
data_size: u64,
prefix: Vec<u8>,
referenced_by: Vec<u64>,
}
struct Walk<'a, S: ByteSource + ?Sized> {
source: &'a S,
found: BTreeMap<u64, Raw>,
queue: VecDeque<(u64, u64)>,
warnings: Vec<String>,
unfinalized: bool,
}
impl<'a, S: ByteSource + ?Sized> Walk<'a, S> {
fn new(source: &'a S) -> Self {
Self {
source,
found: BTreeMap::new(),
queue: VecDeque::new(),
warnings: Vec::new(),
unfinalized: false,
}
}
fn run(mut self) -> BlockMap {
let file_size = self.source.len();
self.read_id_block(file_size);
self.queue.push_back((ID_BLOCK_SIZE as u64, 0));
while let Some((address, referrer)) = self.queue.pop_front() {
if let Some(existing) = self.found.get_mut(&address) {
if referrer != 0 && !existing.referenced_by.contains(&referrer) {
existing.referenced_by.push(referrer);
}
continue;
}
if self.found.len() >= MAX_BLOCKS {
self.warnings.push(format!(
"stopped after {MAX_BLOCKS} blocks; the file's link graph is larger than this walk will follow"
));
break;
}
self.visit(address, referrer);
}
self.finish(file_size)
}
fn read_id_block(&mut self, file_size: u64) {
if file_size < ID_BLOCK_SIZE as u64 {
self.warnings.push(format!(
"the file is {file_size} bytes, shorter than the {ID_BLOCK_SIZE}-byte identification block"
));
return;
}
let Ok(data) = self.source.read_bytes(0, ID_BLOCK_SIZE) else {
self.warnings
.push("the identification block at 0x0 could not be read".to_string());
return;
};
let mut block_type = [0u8; 4];
block_type.copy_from_slice(&data[..4]);
self.unfinalized = &data[..8] == b"UnFinMF " || LittleEndian::read_u16(&data[60..62]) != 0;
self.found.insert(
0,
Raw {
address: 0,
block_type,
length: ID_BLOCK_SIZE as u64,
link_count: 0,
links: Vec::new(),
data_size: ID_BLOCK_SIZE as u64,
prefix: data[..ID_BLOCK_SIZE].to_vec(),
referenced_by: Vec::new(),
},
);
}
fn visit(&mut self, address: u64, referrer: u64) {
let file_size = self.source.len();
if address.saturating_add(BLOCK_HEADER_SIZE as u64) > file_size {
self.warnings.push(format!(
"the link at {referrer:#x} points to {address:#x}, past the end of the {file_size}-byte file"
));
return;
}
let raw_header = match self.source.read_bytes(address, BLOCK_HEADER_SIZE) {
Ok(bytes) => bytes,
Err(e) => {
self.warnings
.push(format!("the block at {address:#x} could not be read: {e}"));
return;
}
};
if &raw_header[..2] != b"##" {
let id = String::from_utf8_lossy(&raw_header[..4]).into_owned();
self.warnings.push(format!(
"the link at {referrer:#x} points to {address:#x}, which starts with {id:?} rather than a block identifier"
));
return;
}
let header = match BlockHeader::parse(&raw_header, address) {
Ok(header) => header,
Err(e) => {
self.warnings
.push(format!("the block at {address:#x} does not parse: {e}"));
return;
}
};
if address.saturating_add(header.length) > file_size {
self.warnings.push(format!(
"the block at {address:#x} declares {} bytes, which runs past the end of the file",
header.length
));
return;
}
let links = self.read_links(address, &header);
let prefix_len = header.data_size().min(SUMMARY_PREFIX as u64) as usize;
let prefix = self
.source
.read_bytes(address + header.data_offset() as u64, prefix_len)
.map(|b| b.to_vec())
.unwrap_or_default();
for (index, &link) in links.iter().enumerate() {
if link == 0 {
continue;
}
if link == address {
self.warnings.push(format!(
"link {index} of the block at {address:#x} points at itself"
));
continue;
}
self.queue.push_back((link, address));
}
self.found.insert(
address,
Raw {
address,
block_type: header.block_type,
length: header.length,
link_count: header.link_count,
links,
data_size: header.data_size(),
prefix,
referenced_by: if referrer == 0 {
Vec::new()
} else {
vec![referrer]
},
},
);
}
fn read_links(&mut self, address: u64, header: &BlockHeader) -> Vec<u64> {
let count = header.link_count as usize;
if count == 0 {
return Vec::new();
}
let bytes = match self
.source
.read_bytes(address + BLOCK_HEADER_SIZE as u64, count * 8)
{
Ok(bytes) => bytes,
Err(e) => {
self.warnings.push(format!(
"the {count} links of the block at {address:#x} could not be read: {e}"
));
return Vec::new();
}
};
(0..count)
.map(|i| LittleEndian::read_u64(&bytes[i * 8..i * 8 + 8]))
.collect()
}
fn finish(self, file_size: u64) -> BlockMap {
let Walk {
found,
warnings,
unfinalized,
..
} = self;
let texts: HashMap<u64, String> = found
.values()
.filter(|raw| matches!(&raw.block_type, b"##TX" | b"##MD"))
.map(|raw| (raw.address, block_text(&raw.prefix)))
.collect();
let mut covered_bytes = 0u64;
let mut blocks: Vec<BlockInfo> = Vec::with_capacity(found.len());
for raw in found.values() {
covered_bytes = covered_bytes.saturating_add(raw.length);
blocks.push(BlockInfo {
address: raw.address,
block_type: String::from_utf8_lossy(&raw.block_type).into_owned(),
length: raw.length,
link_count: raw.link_count,
link_labels: link_labels(&raw.block_type, raw.links.len()),
links: raw.links.clone(),
summary: summarize(raw, &texts, unfinalized),
referenced_by: raw.referenced_by.clone(),
data_size: raw.data_size,
});
}
let mut gaps = Vec::new();
let mut cursor = 0u64;
for block in &blocks {
if block.address > cursor {
gaps.push(Gap {
address: cursor,
length: block.address - cursor,
});
}
cursor = cursor.max(block.address.saturating_add(block.length));
}
if cursor < file_size {
gaps.push(Gap {
address: cursor,
length: file_size - cursor,
});
}
BlockMap {
blocks,
gaps,
file_size,
covered_bytes,
warnings,
unfinalized,
}
}
}
fn block_text(prefix: &[u8]) -> String {
let end = prefix.iter().position(|&b| b == 0).unwrap_or(prefix.len());
String::from_utf8_lossy(&prefix[..end]).into_owned()
}
fn one_line(text: &str, limit: usize) -> String {
let collapsed = text.split_whitespace().collect::<Vec<_>>().join(" ");
if collapsed.chars().count() <= limit {
return collapsed;
}
let cut: String = collapsed.chars().take(limit).collect();
format!("{cut}\u{2026}")
}
fn linked_text(links: &[u64], index: usize, texts: &HashMap<u64, String>) -> String {
links
.get(index)
.and_then(|link| texts.get(link))
.map(|text| one_line(text, 60))
.unwrap_or_default()
}
fn quoted_or(name: &str, fallback: &str) -> String {
if name.is_empty() {
fallback.to_string()
} else {
format!("\"{name}\"")
}
}
fn u8_at(data: &[u8], offset: usize) -> Option<u8> {
data.get(offset).copied()
}
fn u16_at(data: &[u8], offset: usize) -> Option<u16> {
data.get(offset..offset + 2).map(LittleEndian::read_u16)
}
fn u32_at(data: &[u8], offset: usize) -> Option<u32> {
data.get(offset..offset + 4).map(LittleEndian::read_u32)
}
fn u64_at(data: &[u8], offset: usize) -> Option<u64> {
data.get(offset..offset + 8).map(LittleEndian::read_u64)
}
fn f64_at(data: &[u8], offset: usize) -> Option<f64> {
data.get(offset..offset + 8).map(LittleEndian::read_f64)
}
fn summarize(raw: &Raw, texts: &HashMap<u64, String>, unfinalized: bool) -> String {
let data = &raw.prefix[..];
let links = &raw.links[..];
match &raw.block_type {
b"MDF " => {
let version = one_line(&String::from_utf8_lossy(data.get(8..16).unwrap_or(&[])), 8);
let program = one_line(&String::from_utf8_lossy(data.get(16..24).unwrap_or(&[])), 8);
let unfinalized = u16_at(data, 60).unwrap_or(0);
let mut summary = format!(
"MDF {version}, written by {}",
quoted_or(&program, "an unnamed tool")
);
if unfinalized != 0 {
summary.push_str(&format!(", unfinalized (flags {unfinalized:#06x})"));
}
summary
}
b"##HD" => {
let start_ns = u64_at(data, 0).unwrap_or(0);
let flags = u8_at(data, 12).unwrap_or(0);
format!("recording starts at {start_ns} ns, time flags {flags:#04x}")
}
b"##FH" => {
let time_ns = u64_at(data, 0).unwrap_or(0);
let comment = linked_text(links, 1, texts);
if comment.is_empty() {
format!("history entry at {time_ns} ns")
} else {
format!("history entry at {time_ns} ns: {comment}")
}
}
b"##DG" => {
let rec_id_size = u8_at(data, 0).unwrap_or(0);
if rec_id_size == 0 {
"sorted data group (no record IDs)".to_string()
} else {
format!("unsorted data group, {rec_id_size}-byte record IDs")
}
}
b"##CG" => {
let name = linked_text(links, 2, texts);
let record_id = u64_at(data, 0).unwrap_or(0);
let cycles = u64_at(data, 8).unwrap_or(0);
let flags = u16_at(data, 16).unwrap_or(0);
let data_bytes = u32_at(data, 20).unwrap_or(0);
let inval_bytes = u32_at(data, 24).unwrap_or(0);
let mut summary = format!(
"{} \u{2014} record {record_id}, {cycles} cycles, {data_bytes} data bytes",
quoted_or(&name, "unnamed group")
);
if inval_bytes > 0 {
summary.push_str(&format!(" + {inval_bytes} invalidation bytes"));
}
if flags & 0x1 != 0 {
summary.push_str(", variable-length data");
}
if flags & 0x2 != 0 {
summary.push_str(", bus events");
}
summary
}
b"##CN" => {
let name = linked_text(links, 2, texts);
let cn_type = u8_at(data, 0).unwrap_or(0);
let sync_type = u8_at(data, 1).unwrap_or(0);
let data_type = u8_at(data, 2).unwrap_or(0);
let bit_offset = u8_at(data, 3).unwrap_or(0);
let byte_offset = u32_at(data, 4).unwrap_or(0);
let bit_count = u32_at(data, 8).unwrap_or(0);
format!(
"{} \u{2014} {}, data type {data_type}, {bit_count} bits at byte {byte_offset}+{bit_offset}",
quoted_or(&name, "unnamed channel"),
channel_type_name(cn_type, sync_type)
)
}
b"##CC" => {
let name = linked_text(links, 0, texts);
let cc_type = u8_at(data, 0).unwrap_or(0);
let ref_count = u16_at(data, 4).unwrap_or(0);
let val_count = u16_at(data, 6).unwrap_or(0);
let mut summary = conversion_type_name(cc_type).to_string();
if !name.is_empty() {
summary = format!("\"{name}\" \u{2014} {summary}");
}
summary.push_str(&format!(", {val_count} values, {ref_count} references"));
summary
}
b"##SI" => {
let name = linked_text(links, 0, texts);
let path = linked_text(links, 1, texts);
let si_type = u8_at(data, 0).unwrap_or(0);
let bus_type = u8_at(data, 1).unwrap_or(0);
let flags = u8_at(data, 2).unwrap_or(0);
let mut summary = format!(
"{} \u{2014} {} source",
quoted_or(&name, "unnamed source"),
source_type_name(si_type)
);
if bus_type != 0 {
summary.push_str(&format!(" on {}", bus_type_name(bus_type)));
}
if !path.is_empty() {
summary.push_str(&format!(", path {path}"));
}
if flags & 0x1 != 0 {
summary.push_str(", simulated");
}
summary
}
b"##TX" | b"##MD" => {
let text = one_line(&block_text(data), 100);
if text.is_empty() {
"(empty)".to_string()
} else {
text
}
}
b"##DT" | b"##SD" | b"##RD" | b"##DV" | b"##DI" => {
if raw.data_size == 0 && unfinalized {
"no length recorded \u{2014} the file was never finalized, so the records run to the end of it".to_string()
} else {
format!("{} bytes of records", raw.data_size)
}
}
b"##DZ" => {
let original = data
.get(..2)
.map(|b| String::from_utf8_lossy(b).into_owned())
.unwrap_or_default();
let zip_type = u8_at(data, 2).unwrap_or(0);
let original_len = u64_at(data, 8).unwrap_or(0);
let stored_len = u64_at(data, 16).unwrap_or(0);
let ratio = if original_len > 0 {
stored_len as f64 / original_len as f64 * 100.0
} else {
0.0
};
format!(
"##{original} compressed with {}: {original_len} \u{2192} {stored_len} bytes ({ratio:.0}%)",
zip_type_name(zip_type)
)
}
b"##DL" => {
let flags = u8_at(data, 0).unwrap_or(0);
let count = u32_at(data, 4).unwrap_or(0);
let mut summary = format!("{count} data blocks");
if flags & 0x1 != 0 {
if let Some(equal_length) = u64_at(data, 8) {
summary.push_str(&format!(", {equal_length} bytes each"));
}
}
if flags & 0x2 != 0 {
summary.push_str(", time-indexed");
}
summary
}
b"##LD" => {
let flags = u32_at(data, 0).unwrap_or(0);
let count = u32_at(data, 4).unwrap_or(0);
let mut summary = format!("{count} data blocks");
if flags & 0x1 != 0 {
if let Some(equal_length) = u64_at(data, 8) {
summary.push_str(&format!(", {equal_length} samples/bytes each"));
}
}
if flags & 0x2 != 0 {
summary.push_str(", time-indexed");
}
if flags & 0x8000_0000 != 0 {
summary.push_str(", invalidation bits");
}
summary
}
b"##HL" => {
let flags = u16_at(data, 0).unwrap_or(0);
let zip_type = u8_at(data, 2).unwrap_or(0);
format!(
"header of a {} list, flags {flags:#06x}",
zip_type_name(zip_type)
)
}
b"##AT" => {
let name = linked_text(links, 1, texts);
let mime = linked_text(links, 2, texts);
let flags = u16_at(data, 16).unwrap_or(0);
let original_size = u64_at(data, 24).unwrap_or(0);
let embedded_size = u64_at(data, 32).unwrap_or(0);
let embedded = flags & 0x1 != 0;
let mut summary = format!(
"{} \u{2014} {}",
quoted_or(&name, "unnamed attachment"),
if embedded { "embedded" } else { "external" }
);
if embedded {
summary.push_str(&format!(", {embedded_size} bytes stored"));
if flags & 0x2 != 0 {
summary.push_str(&format!(" for {original_size} original"));
}
} else {
summary.push_str(&format!(", {original_size} bytes at the named path"));
}
if !mime.is_empty() {
summary.push_str(&format!(", {mime}"));
}
summary
}
b"##EV" => {
let name = linked_text(links, 3, texts);
let ev_type = u8_at(data, 0).unwrap_or(0);
let sync_type = u8_at(data, 1).unwrap_or(0);
let scope_count = u32_at(data, 8).unwrap_or(0);
let base = u64_at(data, 16).unwrap_or(0) as i64;
let factor = f64_at(data, 24).unwrap_or(0.0);
format!(
"{} \u{2014} {} event at {} {}, {scope_count} in scope",
quoted_or(&name, "unnamed event"),
event_type_name(ev_type),
base as f64 * factor,
sync_unit_name(sync_type)
)
}
b"##CA" => {
let ca_type = u8_at(data, 0).unwrap_or(0);
let storage = u8_at(data, 1).unwrap_or(0);
let ndim = u16_at(data, 2).unwrap_or(0);
let dims: Vec<String> = (0..ndim as usize)
.filter_map(|i| u64_at(data, 16 + i * 8))
.map(|d| d.to_string())
.collect();
format!(
"{} array, {} storage, shape [{}]",
array_type_name(ca_type),
array_storage_name(storage),
dims.join(", ")
)
}
b"##CH" => {
let name = linked_text(links, 2, texts);
let element_count = u32_at(data, 0).unwrap_or(0);
let ch_type = u8_at(data, 4).unwrap_or(0);
format!(
"{} \u{2014} {} node, {element_count} channels",
quoted_or(&name, "unnamed node"),
hierarchy_type_name(ch_type)
)
}
b"##SR" => {
let cycles = u64_at(data, 0).unwrap_or(0);
let interval = f64_at(data, 8).unwrap_or(0.0);
let sync_type = u8_at(data, 16).unwrap_or(0);
format!(
"{cycles} reduced cycles every {interval} {}",
sync_unit_name(sync_type)
)
}
_ => String::new(),
}
}
fn channel_type_name(cn_type: u8, sync_type: u8) -> String {
let kind = match cn_type {
0 => "fixed-length value",
1 => "variable-length value",
2 => "master",
3 => "virtual master",
4 => "synchronisation",
5 => "maximum-length value",
6 => "virtual value",
_ => "unknown type",
};
if matches!(cn_type, 2..=4) {
format!("{kind} ({})", sync_name(sync_type))
} else {
kind.to_string()
}
}
fn sync_name(sync_type: u8) -> &'static str {
match sync_type {
0 => "none",
1 => "time",
2 => "angle",
3 => "distance",
4 => "index",
_ => "unknown",
}
}
fn sync_unit_name(sync_type: u8) -> &'static str {
match sync_type {
1 => "s",
2 => "rad",
3 => "m",
4 => "samples",
_ => "",
}
}
fn conversion_type_name(cc_type: u8) -> &'static str {
match cc_type {
0 => "identity conversion",
1 => "linear conversion",
2 => "rational conversion",
3 => "algebraic conversion",
4 => "value-to-value table, interpolating",
5 => "value-to-value table",
6 => "value-range-to-value table",
7 => "value-to-text table",
8 => "value-range-to-text table",
9 => "text-to-value table",
10 => "text-to-text table",
11 => "bitfield text table",
_ => "unknown conversion",
}
}
fn source_type_name(si_type: u8) -> &'static str {
match si_type {
0 => "other",
1 => "ECU",
2 => "bus",
3 => "I/O",
4 => "tool",
5 => "user",
_ => "unknown",
}
}
fn bus_type_name(bus_type: u8) -> &'static str {
match bus_type {
1 => "CAN",
2 => "LIN",
3 => "MOST",
4 => "FlexRay",
5 => "K-Line",
6 => "Ethernet",
7 => "USB",
_ => "an unknown bus",
}
}
fn zip_type_name(zip_type: u8) -> &'static str {
match zip_type {
0 => "deflate",
1 => "transposed deflate",
_ => "an unknown compression",
}
}
fn event_type_name(ev_type: u8) -> &'static str {
match ev_type {
0 => "recording",
1 => "recording interrupt",
2 => "acquisition interrupt",
3 => "start recording trigger",
4 => "stop recording trigger",
5 => "trigger",
6 => "marker",
_ => "unknown",
}
}
fn array_type_name(ca_type: u8) -> &'static str {
match ca_type {
0 => "value",
1 => "scaling axis",
2 => "look-up",
3 => "interval axis",
4 => "classification result",
_ => "unknown",
}
}
fn array_storage_name(storage: u8) -> &'static str {
match storage {
0 => "in-record",
1 => "per-element signal data",
2 => "per-array signal data",
_ => "unknown",
}
}
fn hierarchy_type_name(ch_type: u8) -> &'static str {
match ch_type {
0 => "group",
1 => "function",
2 => "structure",
3 => "map list",
4 => "input",
5 => "output",
6 => "local",
7 => "calibration definition",
8 => "calibration reference",
_ => "unknown",
}
}
fn link_labels(block_type: &[u8; 4], count: usize) -> Vec<String> {
let (named, tail): (&[&str], &str) = match block_type {
b"##HD" => (
&[
"hd_dg_first",
"hd_fh_first",
"hd_ch_first",
"hd_at_first",
"hd_ev_first",
"hd_md_comment",
],
"hd_link",
),
b"##FH" => (&["fh_fh_next", "fh_md_comment"], "fh_link"),
b"##DG" => (
&["dg_dg_next", "dg_cg_first", "dg_data", "dg_md_comment"],
"dg_link",
),
b"##CG" => (
&[
"cg_cg_next",
"cg_cn_first",
"cg_tx_acq_name",
"cg_si_acq_source",
"cg_sr_first",
"cg_md_comment",
],
"cg_link",
),
b"##CN" => (
&[
"cn_cn_next",
"cn_composition",
"cn_tx_name",
"cn_si_source",
"cn_cc_conversion",
"cn_data",
"cn_md_unit",
"cn_md_comment",
],
"cn_at_reference",
),
b"##CC" => (
&["cc_tx_name", "cc_md_unit", "cc_md_comment", "cc_cc_inverse"],
"cc_ref",
),
b"##SI" => (&["si_tx_name", "si_tx_path", "si_md_comment"], "si_link"),
b"##CA" => (&["ca_composition"], "ca_data"),
b"##DL" => (&["dl_dl_next"], "dl_data"),
b"##LD" => (&["ld_ld_next"], "ld_data"),
b"##HL" => (&["hl_dl_first"], "hl_link"),
b"##AT" => (
&[
"at_at_next",
"at_tx_filename",
"at_tx_mimetype",
"at_md_comment",
],
"at_link",
),
b"##EV" => (
&[
"ev_ev_next",
"ev_ev_parent",
"ev_ev_range",
"ev_tx_name",
"ev_md_comment",
],
"ev_scope",
),
b"##CH" => (
&["ch_ch_next", "ch_ch_first", "ch_tx_name", "ch_md_comment"],
"ch_element",
),
b"##SR" => (&["sr_sr_next", "sr_data"], "sr_link"),
_ => (&[], "link"),
};
(0..count)
.map(|i| match named.get(i) {
Some(name) => (*name).to_string(),
None => format!("{tail}[{}]", i - named.len()),
})
.collect()
}
#[cfg(test)]
mod tests {
use super::*;
use crate::io::ByteSlice;
struct Bytes(Vec<u8>);
impl ByteSource for Bytes {
fn len(&self) -> u64 {
self.0.len() as u64
}
fn read_bytes(&self, offset: u64, len: usize) -> crate::error::Result<ByteSlice<'_>> {
let start = offset as usize;
let end = start.saturating_add(len);
if end > self.0.len() {
return Err(crate::error::Mf4Error::truncated(
offset,
len,
self.0.len().saturating_sub(start),
));
}
Ok(ByteSlice::borrowed(&self.0[start..end]))
}
}
fn build(blocks: &[(&[u8; 4], Vec<u64>, Vec<u8>)]) -> (Bytes, Vec<u64>) {
let mut file = vec![0u8; ID_BLOCK_SIZE];
file[..8].copy_from_slice(b"MDF ");
file[8..16].copy_from_slice(b"4.10 ");
let mut addresses = Vec::new();
for (id, links, data) in blocks {
addresses.push(file.len() as u64);
let length = (BLOCK_HEADER_SIZE + links.len() * 8 + data.len()) as u64;
file.extend_from_slice(*id);
file.extend_from_slice(&[0u8; 4]);
file.extend_from_slice(&length.to_le_bytes());
file.extend_from_slice(&(links.len() as u64).to_le_bytes());
for link in links {
file.extend_from_slice(&link.to_le_bytes());
}
file.extend_from_slice(data);
}
(Bytes(file), addresses)
}
#[test]
fn walks_from_the_id_block_to_the_last_block() {
let text = b"a comment\0".to_vec();
let hd_data = vec![0u8; 24];
let (source, addresses) = build(&[
(b"##HD", vec![0, 0, 0, 0, 0, 0], hd_data),
(b"##TX", vec![], text),
]);
let mut bytes = source.0;
let hd_links = addresses[0] as usize + BLOCK_HEADER_SIZE;
bytes[hd_links + 5 * 8..hd_links + 6 * 8].copy_from_slice(&addresses[1].to_le_bytes());
let map = BlockMap::scan(&Bytes(bytes));
let types: Vec<&str> = map.blocks.iter().map(|b| b.block_type.as_str()).collect();
assert_eq!(types, ["MDF ", "##HD", "##TX"]);
assert_eq!(map.blocks[0].address, 0);
assert_eq!(map.blocks[1].address, 64);
assert_eq!(map.blocks[2].summary, "a comment");
assert_eq!(map.blocks[2].referenced_by, vec![addresses[0]]);
assert_eq!(map.blocks[1].link_labels[5], "hd_md_comment");
assert!(map.warnings.is_empty(), "{:?}", map.warnings);
}
#[test]
fn a_block_linked_twice_is_listed_once_with_both_referrers() {
let hd_data = vec![0u8; 24];
let (source, addresses) = build(&[
(b"##HD", vec![0, 0, 0, 0, 0, 0], hd_data),
(b"##FH", vec![0, 0], vec![0u8; 16]),
(b"##TX", vec![], b"shared\0".to_vec()),
]);
let mut bytes = source.0;
let hd_links = addresses[0] as usize + BLOCK_HEADER_SIZE;
bytes[hd_links + 8..hd_links + 16].copy_from_slice(&addresses[1].to_le_bytes());
bytes[hd_links + 5 * 8..hd_links + 6 * 8].copy_from_slice(&addresses[2].to_le_bytes());
let fh_links = addresses[1] as usize + BLOCK_HEADER_SIZE;
bytes[fh_links + 8..fh_links + 16].copy_from_slice(&addresses[2].to_le_bytes());
let map = BlockMap::scan(&Bytes(bytes));
let tx: Vec<&BlockInfo> = map
.blocks
.iter()
.filter(|b| b.block_type == "##TX")
.collect();
assert_eq!(tx.len(), 1);
assert_eq!(tx[0].referenced_by, vec![addresses[0], addresses[1]]);
}
#[test]
fn a_link_into_nothing_is_a_warning_not_a_block() {
let hd_data = vec![0u8; 24];
let (source, addresses) = build(&[
(b"##HD", vec![0, 0, 0, 0, 0, 0], hd_data),
(b"##DT", vec![], vec![0xAB; 64]),
]);
let mut bytes = source.0;
let hd_links = addresses[0] as usize + BLOCK_HEADER_SIZE;
let bogus = addresses[1] + 40;
bytes[hd_links..hd_links + 8].copy_from_slice(&bogus.to_le_bytes());
let map = BlockMap::scan(&Bytes(bytes));
assert!(map.blocks.iter().all(|b| b.address != bogus));
assert_eq!(map.warnings.len(), 1, "{:?}", map.warnings);
assert!(map.warnings[0].contains("rather than a block identifier"));
}
#[test]
fn bytes_no_block_covers_are_reported_as_gaps() {
let (source, addresses) = build(&[
(b"##HD", vec![0, 0, 0, 0, 0, 0], vec![0u8; 24]),
(b"##DT", vec![], vec![0u8; 32]),
]);
let map = BlockMap::scan(&source);
assert_eq!(map.blocks.len(), 2);
assert_eq!(
map.gaps,
[Gap {
address: addresses[1],
length: BLOCK_HEADER_SIZE as u64 + 32,
}]
);
assert_eq!(map.covered_bytes + map.gaps[0].length, map.file_size);
}
#[test]
fn a_truncated_block_is_reported_rather_than_read() {
let (source, addresses) = build(&[(b"##HD", vec![0, 0, 0, 0, 0, 0], vec![0u8; 24])]);
let mut bytes = source.0;
let length_at = addresses[0] as usize + 8;
bytes[length_at..length_at + 8].copy_from_slice(&100_000u64.to_le_bytes());
let map = BlockMap::scan(&Bytes(bytes));
assert_eq!(map.blocks.len(), 1, "only the ID block should be listed");
assert!(map.warnings.iter().any(|w| w.contains("past the end")));
}
#[test]
fn a_cycle_between_blocks_terminates() {
let (source, addresses) = build(&[
(b"##HD", vec![0, 0, 0, 0, 0, 0], vec![0u8; 24]),
(b"##FH", vec![0, 0], vec![0u8; 16]),
]);
let mut bytes = source.0;
let hd_links = addresses[0] as usize + BLOCK_HEADER_SIZE;
bytes[hd_links + 8..hd_links + 16].copy_from_slice(&addresses[1].to_le_bytes());
let fh_links = addresses[1] as usize + BLOCK_HEADER_SIZE;
bytes[fh_links..fh_links + 8].copy_from_slice(&addresses[0].to_le_bytes());
let map = BlockMap::scan(&Bytes(bytes));
assert_eq!(map.blocks.len(), 3);
}
#[test]
fn type_counts_are_ordered_by_frequency() {
let (source, addresses) = build(&[
(b"##HD", vec![0, 0, 0, 0, 0, 0], vec![0u8; 24]),
(b"##FH", vec![0, 0], vec![0u8; 16]),
(b"##TX", vec![], b"one\0".to_vec()),
(b"##TX", vec![], b"two\0".to_vec()),
]);
let mut bytes = source.0;
let hd_links = addresses[0] as usize + BLOCK_HEADER_SIZE;
bytes[hd_links + 8..hd_links + 16].copy_from_slice(&addresses[1].to_le_bytes());
bytes[hd_links + 5 * 8..hd_links + 6 * 8].copy_from_slice(&addresses[2].to_le_bytes());
let fh_links = addresses[1] as usize + BLOCK_HEADER_SIZE;
bytes[fh_links + 8..fh_links + 16].copy_from_slice(&addresses[3].to_le_bytes());
let map = BlockMap::scan(&Bytes(bytes));
let counts = map.type_counts();
assert_eq!(counts[0], ("##TX".to_string(), 2));
}
}