use crate::{read_u32_le, read_u64_le, write_u32_le, write_u64_le};
pub const SHORT_STRING_THRESHOLD: usize = 12;
pub const GERMAN_INLINE_OFF: usize = 4;
#[inline]
pub fn encode_german_string(s: &[u8], blob: &mut Vec<u8>) -> [u8; 16] {
let len = s.len();
assert!(
len <= u32::MAX as usize,
"encode_german_string: length {len} exceeds u32::MAX"
);
let mut st = [0u8; 16];
write_u32_le(&mut st, 0, len as u32);
if len > SHORT_STRING_THRESHOLD {
st[4..8].copy_from_slice(&s[..4]);
write_u64_le(&mut st, 8, blob.len() as u64);
blob.extend_from_slice(s);
} else {
st[GERMAN_INLINE_OFF..GERMAN_INLINE_OFF + len].copy_from_slice(s);
}
st
}
#[inline]
pub fn shift_german_string_heaps(cells: &mut [u8], delta: usize) {
if delta == 0 {
return;
}
for cell in cells.as_chunks_mut::<16>().0 {
if german_string_inline(cell).is_none() {
let off = read_u64_le(cell, 8) + delta as u64;
write_u64_le(cell, 8, off);
}
}
}
#[inline]
pub fn blob_extent(blob_len: usize, heap_offset: u64, length: usize) -> Option<std::ops::Range<usize>> {
let end = heap_offset.checked_add(length as u64)?;
if end > blob_len as u64 {
return None;
}
Some(heap_offset as usize..end as usize)
}
#[inline(always)]
pub fn german_string_inline(cell: &[u8]) -> Option<&[u8]> {
let length = read_u32_le(cell, 0) as usize;
if length <= SHORT_STRING_THRESHOLD {
Some(&cell[GERMAN_INLINE_OFF..GERMAN_INLINE_OFF + length])
} else {
None
}
}
#[inline(always)]
pub fn german_string_heap(cell: &[u8], blob_len: usize) -> Option<std::ops::Range<usize>> {
let None = german_string_inline(cell) else {
return None;
};
blob_extent(blob_len, read_u64_le(cell, 8), read_u32_le(cell, 0) as usize)
}
pub fn try_decode_german_string(st: &[u8], blob: &[u8]) -> Option<Vec<u8>> {
match german_string_inline(st) {
Some(inline) => Some(inline.to_vec()),
None => Some(blob[german_string_heap(st, blob.len())?].to_vec()),
}
}
pub fn german_string_cell_ok(cell: &[u8], blob: &[u8]) -> bool {
let cell: &[u8; 16] = cell[..16].try_into().unwrap();
if let Some(canonical) = canonical_short_cell(cell) {
return *cell == canonical;
}
match german_string_heap(cell, blob.len()) {
Some(r) => blob[r].starts_with(&cell[4..8]),
None => false,
}
}
#[inline]
pub fn canonical_short_cell(src: &[u8; 16]) -> Option<[u8; 16]> {
if let Some(content) = german_string_inline(src) {
let keep = u128::MAX >> (128 - 8 * (4 + content.len()));
return Some((u128::from_le_bytes(*src) & keep).to_le_bytes());
}
None
}
#[inline]
pub fn german_string_short_ascii(cell: &[u8; 16]) -> bool {
german_string_inline(cell).is_some() && u128::from_le_bytes(*cell) & u128::from_le_bytes([0x80; 16]) == 0
}
#[inline]
pub fn german_string_content<'a>(s: &'a [u8], blob: &'a [u8]) -> &'a [u8] {
match german_string_inline(s) {
Some(inline) => inline,
None => match german_string_heap(s, blob.len()) {
Some(r) => &blob[r],
None => &[],
},
}
}
#[inline(always)]
pub fn compare_german_strings(a: &[u8], blob_a: &[u8], b: &[u8], blob_b: &[u8]) -> std::cmp::Ordering {
let pfx_a = u32::from_be_bytes(a[4..8].try_into().unwrap());
let pfx_b = u32::from_be_bytes(b[4..8].try_into().unwrap());
if pfx_a != pfx_b {
return pfx_a.cmp(&pfx_b);
}
german_string_content(a, blob_a).cmp(german_string_content(b, blob_b))
}
#[cfg(test)]
#[path = "tests/german_string.rs"]
mod tests;