#![forbid(unsafe_code)]
use kcode_k1_web_package::{PackageError, SourceFile, SourcePackage, WebFamily, WebId};
use semver::Version;
use serde::{Deserialize, Serialize};
use std::{
fmt::{Display, Formatter},
ops::Range,
};
const HTML_ENTRY: &str = "const frame=document.createElement(\"iframe\");\nframe.hidden=true;\nframe.src=new URL(\"./Code.html\",import.meta.url).href;\ndocument.body.append(frame);\nawait new Promise((resolve,reject)=>{frame.addEventListener(\"load\",resolve,{once:true});frame.addEventListener(\"error\",()=>reject(new Error(\"Code.html failed to load\")),{once:true});});\nexport const codeWindow=frame.contentWindow;\n";
const HTML_TESTS: &str = "import {codeWindow} from \"./k1-entry.js\";\nexport async function runTests(){const test=codeWindow.globalThis.runTests;if(typeof test!==\"function\")throw new Error(\"Code.html must define globalThis.runTests\");return await test.call(codeWindow);}\n";
const CSS_ENTRY: &str = "const link=document.createElement(\"link\");\nlink.rel=\"stylesheet\";\nlink.href=new URL(\"./Code.css\",import.meta.url).href;\ndocument.head.append(link);\nawait new Promise((resolve,reject)=>{link.addEventListener(\"load\",resolve,{once:true});link.addEventListener(\"error\",()=>reject(new Error(\"Code.css failed to load\")),{once:true});});\nexport const sheet=link.sheet;\nvoid sheet.cssRules;\n";
const CSS_TESTS: &str = "import {sheet} from \"./k1-entry.js\";\nexport async function runTests(){void sheet.cssRules;}\n";
#[derive(Clone, Copy, Debug, Eq, Hash, Ord, PartialEq, PartialOrd)]
pub enum Language {
JavaScript,
Html,
Css,
}
impl Language {
pub const fn code_path(self) -> &'static str {
match self {
Self::JavaScript => "Code.js",
Self::Html => "Code.html",
Self::Css => "Code.css",
}
}
pub const fn as_str(self) -> &'static str {
match self {
Self::JavaScript => "javascript",
Self::Html => "html",
Self::Css => "css",
}
}
}
#[derive(Clone, Debug, Deserialize, Eq, Hash, Ord, PartialEq, PartialOrd, Serialize)]
#[serde(deny_unknown_fields)]
pub struct Dependency {
authority: String,
name: String,
selector: String,
}
impl Dependency {
pub fn authority(&self) -> &str {
&self.authority
}
pub fn name(&self) -> &str {
&self.name
}
pub fn selector(&self) -> &str {
&self.selector
}
}
#[derive(Deserialize, Serialize)]
#[serde(deny_unknown_fields)]
struct Header {
dependencies: Vec<Dependency>,
}
#[derive(Serialize)]
struct Manifest<'a> {
name: &'a str,
version: String,
entry: &'a str,
tests: &'a str,
dependencies: Vec<&'a Dependency>,
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct DocumentError(String);
impl DocumentError {
pub fn message(&self) -> &str {
&self.0
}
}
impl Display for DocumentError {
fn fmt(&self, formatter: &mut Formatter<'_>) -> std::fmt::Result {
formatter.write_str(&self.0)
}
}
impl std::error::Error for DocumentError {}
impl From<PackageError> for DocumentError {
fn from(value: PackageError) -> Self {
Self(value.to_string())
}
}
fn fail<T>(message: impl Into<String>) -> Result<T, DocumentError> {
Err(DocumentError(message.into()))
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct CodeDocument {
family: WebFamily,
documentation: Vec<u8>,
language: Language,
code: Vec<u8>,
dependencies: Vec<Dependency>,
}
impl CodeDocument {
pub fn new(
family: WebFamily,
documentation: Vec<u8>,
language: Language,
code: Vec<u8>,
) -> Result<Self, DocumentError> {
let dependencies = parse_header(&documentation)?;
std::str::from_utf8(&code).map_err(|_| DocumentError("code must be UTF-8".into()))?;
let value = Self {
family,
documentation,
language,
code,
dependencies,
};
value.to_source_package(Version::new(0, 0, 0))?;
Ok(value)
}
pub fn family(&self) -> &WebFamily {
&self.family
}
pub fn documentation(&self) -> &[u8] {
&self.documentation
}
pub const fn language(&self) -> Language {
self.language
}
pub fn code(&self) -> &[u8] {
&self.code
}
pub fn dependencies(&self) -> &[Dependency] {
&self.dependencies
}
pub fn to_source_package(&self, version: Version) -> Result<SourcePackage, DocumentError> {
let id = WebId::new(self.family.clone(), version)?;
let (entry, tests) = if self.language == Language::JavaScript {
("Code.js", "Code.js")
} else {
("k1-entry.js", "k1-tests.js")
};
let mut dependencies = self.dependencies.iter().collect::<Vec<_>>();
dependencies.sort_by(|a, b| {
(&a.authority, &a.name, &a.selector).cmp(&(&b.authority, &b.name, &b.selector))
});
let manifest = Manifest {
name: self.family.logical_name(),
version: id.version().to_string(),
entry,
tests,
dependencies,
};
let mut files = vec![
SourceFile::new("Documentation.md", self.documentation.clone()),
SourceFile::new(self.language.code_path(), self.code.clone()),
SourceFile::new(
"k1-web.json",
serde_json::to_vec(&manifest).map_err(|error| DocumentError(error.to_string()))?,
),
];
match self.language {
Language::JavaScript => {}
Language::Html => {
files.push(SourceFile::new(
"k1-entry.js",
HTML_ENTRY.as_bytes().to_vec(),
));
files.push(SourceFile::new(
"k1-tests.js",
HTML_TESTS.as_bytes().to_vec(),
));
}
Language::Css => {
files.push(SourceFile::new(
"k1-entry.js",
CSS_ENTRY.as_bytes().to_vec(),
));
files.push(SourceFile::new(
"k1-tests.js",
CSS_TESTS.as_bytes().to_vec(),
));
}
}
SourcePackage::new(id, files).map_err(Into::into)
}
pub fn from_source_package(source: &SourcePackage) -> Result<Self, DocumentError> {
let documentation = source
.files()
.iter()
.find(|file| file.path() == "Documentation.md")
.ok_or_else(|| DocumentError("noncanonical code-document projection".into()))?
.bytes()
.to_vec();
for language in [Language::JavaScript, Language::Html, Language::Css] {
let Some(file) = source
.files()
.iter()
.find(|file| file.path() == language.code_path())
else {
continue;
};
let Ok(value) = Self::new(
source.id().family().clone(),
documentation.clone(),
language,
file.bytes().to_vec(),
) else {
continue;
};
if value
.to_source_package(source.id().version().clone())
.is_ok_and(|made| made.files() == source.files())
{
return Ok(value);
}
}
fail("noncanonical code-document projection")
}
pub fn chunk_ranges(&self) -> Vec<Range<usize>> {
let text = std::str::from_utf8(&self.code).expect("CodeDocument maintains UTF-8");
let points = match self.language {
Language::JavaScript => js_points(text),
Language::Css => css_points(text),
Language::Html => html_points(text),
};
points.map_or_else(|| whole(self.code.len()), |points| pack(text, points))
}
}
fn parse_header(bytes: &[u8]) -> Result<Vec<Dependency>, DocumentError> {
let text = std::str::from_utf8(bytes)
.map_err(|_| DocumentError("Documentation.md must be UTF-8".into()))?;
let rest = text
.strip_prefix("<!-- k1-web/v1\n")
.ok_or_else(|| DocumentError("missing exact documentation header".into()))?;
let (json, rest) = rest
.split_once('\n')
.ok_or_else(|| DocumentError("incomplete documentation header".into()))?;
if !(rest == "-->" || rest.starts_with("-->\n")) {
return fail("documentation header must end with an exact --> line");
}
let header: Header = serde_json::from_str(json)
.map_err(|error| DocumentError(format!("invalid documentation header: {error}")))?;
if serde_json::to_string(&header).map_err(|error| DocumentError(error.to_string()))? != json {
return fail("documentation header JSON must be canonical and compact");
}
if header.dependencies.iter().any(|dependency| {
dependency.authority.is_empty()
|| dependency.name.is_empty()
|| dependency.selector.is_empty()
}) {
return fail("dependency fields must be nonempty");
}
let mut unique = header.dependencies.clone();
unique.sort();
unique.dedup();
if unique.len() != header.dependencies.len() {
return fail("duplicate dependency declaration");
}
Ok(header.dependencies)
}
fn whole(length: usize) -> Vec<Range<usize>> {
std::iter::once(0..length).collect()
}
fn pack(text: &str, mut points: Vec<usize>) -> Vec<Range<usize>> {
if text.is_empty() {
return whole(0);
}
if text.chars().count() < 800 {
return whole(text.len());
}
points.push(0);
points.push(text.len());
points.sort_unstable();
points.dedup();
let mut output = Vec::new();
let mut start = 0;
while start < text.len() {
let end = points
.iter()
.copied()
.filter(|point| *point > start)
.find(|point| {
text[start..*point].chars().count() >= 800
&& (*point == text.len() || text[*point..].chars().count() >= 800)
})
.unwrap_or(text.len());
output.push(start..end);
start = end;
}
output
}
#[derive(Clone, Copy)]
enum Brace {
Class(bool),
Function,
Member,
Other,
}
fn js_points(text: &str) -> Option<Vec<usize>> {
let bytes = text.as_bytes();
let mut output = vec![0];
let (mut index, mut paren, mut bracket) = (0, 0usize, 0usize);
let mut braces = Vec::new();
let mut pending = None;
let mut regex = true;
let mut statement = true;
let mut last = 0;
while index < bytes.len() {
match bytes[index] {
b'\'' | b'"' => {
index = skip_quote(bytes, index)?;
regex = false;
statement = false;
}
b'`' => {
index = skip_template(bytes, index)?;
regex = false;
statement = false;
}
b'/' if bytes.get(index + 1) == Some(&b'/') => {
index = bytes[index + 2..]
.iter()
.position(|byte| *byte == b'\n')
.map_or(bytes.len(), |next| index + next + 3);
}
b'/' if bytes.get(index + 1) == Some(&b'*') => {
index = find_bytes(bytes, index + 2, b"*/")? + 2;
}
b'/' if regex => {
index = skip_regex(bytes, index)?;
regex = false;
statement = false;
}
byte if ident_start(byte) => {
let start = index;
index += 1;
while index < bytes.len() && ident_continue(bytes[index]) {
index += 1;
}
let word = &text[start..index];
if word == "class" {
pending = Some(Brace::Class(statement));
}
if word == "function" && statement {
pending = Some(Brace::Function);
}
regex = matches!(
word,
"return"
| "throw"
| "case"
| "delete"
| "void"
| "typeof"
| "new"
| "yield"
| "await"
| "else"
| "do"
| "in"
| "of"
| "instanceof"
);
if !matches!(word, "export" | "default" | "async") {
statement = false;
}
last = b'x';
}
b'(' => {
paren += 1;
index += 1;
regex = true;
last = b'(';
}
b')' => {
paren = paren.checked_sub(1)?;
index += 1;
regex = false;
last = b')';
}
b'[' => {
bracket += 1;
index += 1;
regex = true;
last = b'[';
}
b']' => {
bracket = bracket.checked_sub(1)?;
index += 1;
regex = false;
last = b']';
}
b'{' => {
let kind = if paren == 0 && bracket == 0 {
pending.take().unwrap_or_else(|| {
if matches!(braces.last(), Some(Brace::Class(_))) && last == b')' {
Brace::Member
} else {
Brace::Other
}
})
} else {
Brace::Other
};
braces.push(kind);
index += 1;
regex = true;
last = b'{';
}
b'}' => {
if paren != 0 || bracket != 0 {
return None;
}
let kind = braces.pop()?;
index += 1;
regex = false;
last = b'}';
let boundary = matches!(kind, Brace::Function) && braces.is_empty()
|| matches!(kind, Brace::Class(true)) && braces.is_empty()
|| matches!(kind, Brace::Member)
&& matches!(braces.last(), Some(Brace::Class(_)));
if boundary {
output.push(index);
statement = braces.is_empty();
}
}
b';' => {
index += 1;
regex = true;
pending = None;
if paren == 0
&& bracket == 0
&& (braces.is_empty() || matches!(braces.last(), Some(Brace::Class(_))))
{
output.push(index);
statement = braces.is_empty();
}
last = b';';
}
byte if byte.is_ascii_whitespace() => index += 1,
byte if byte.is_ascii_digit() => {
index += 1;
while index < bytes.len()
&& (ident_continue(bytes[index]) || b"._".contains(&bytes[index]))
{
index += 1;
}
regex = false;
statement = false;
last = b'n';
}
b'/' => {
index += 1;
regex = true;
statement = false;
last = b'/';
}
byte => {
index += 1;
regex = byte != b'.';
if !matches!(byte, b',' | b':') {
statement = false;
}
last = byte;
}
}
}
if paren != 0 || bracket != 0 || !braces.is_empty() {
None
} else {
output.push(bytes.len());
Some(output)
}
}
fn ident_start(byte: u8) -> bool {
byte.is_ascii_alphabetic() || matches!(byte, b'_' | b'$')
}
fn ident_continue(byte: u8) -> bool {
ident_start(byte) || byte.is_ascii_digit()
}
fn skip_quote(bytes: &[u8], mut index: usize) -> Option<usize> {
let quote = bytes[index];
index += 1;
while index < bytes.len() {
match bytes[index] {
b'\\' => index = (index + 2).min(bytes.len()),
byte if byte == quote => return Some(index + 1),
b'\n' | b'\r' => return None,
_ => index += 1,
}
}
None
}
fn skip_regex(bytes: &[u8], mut index: usize) -> Option<usize> {
index += 1;
let mut class = false;
while index < bytes.len() {
match bytes[index] {
b'\\' => index = (index + 2).min(bytes.len()),
b'[' => {
class = true;
index += 1;
}
b']' => {
class = false;
index += 1;
}
b'/' if !class => {
index += 1;
while index < bytes.len() && bytes[index].is_ascii_alphabetic() {
index += 1;
}
return Some(index);
}
b'\n' | b'\r' => return None,
_ => index += 1,
}
}
None
}
fn skip_template(bytes: &[u8], mut index: usize) -> Option<usize> {
index += 1;
while index < bytes.len() {
match bytes[index] {
b'\\' => index = (index + 2).min(bytes.len()),
b'`' => return Some(index + 1),
b'$' if bytes.get(index + 1) == Some(&b'{') => {
index = skip_expression(bytes, index + 2)?
}
_ => index += 1,
}
}
None
}
fn skip_expression(bytes: &[u8], mut index: usize) -> Option<usize> {
let (mut depth, mut paren, mut bracket) = (1usize, 0usize, 0usize);
let mut regex = true;
while index < bytes.len() {
match bytes[index] {
b'\'' | b'"' => {
index = skip_quote(bytes, index)?;
regex = false;
}
b'`' => {
index = skip_template(bytes, index)?;
regex = false;
}
b'/' if bytes.get(index + 1) == Some(&b'/') => {
index = bytes[index + 2..]
.iter()
.position(|byte| *byte == b'\n')
.map_or(bytes.len(), |next| index + next + 3)
}
b'/' if bytes.get(index + 1) == Some(&b'*') => {
index = find_bytes(bytes, index + 2, b"*/")? + 2
}
b'/' if regex => {
index = skip_regex(bytes, index)?;
regex = false;
}
b'{' => {
depth += 1;
index += 1;
regex = true;
}
b'}' => {
depth -= 1;
index += 1;
if depth == 0 {
return (paren == 0 && bracket == 0).then_some(index);
}
regex = false;
}
b'(' => {
paren += 1;
index += 1;
regex = true;
}
b')' => {
paren = paren.checked_sub(1)?;
index += 1;
regex = false;
}
b'[' => {
bracket += 1;
index += 1;
regex = true;
}
b']' => {
bracket = bracket.checked_sub(1)?;
index += 1;
regex = false;
}
byte if byte.is_ascii_whitespace() => index += 1,
byte if ident_continue(byte) => {
index += 1;
regex = false;
}
_ => {
index += 1;
regex = true;
}
}
}
None
}
fn find_bytes(haystack: &[u8], start: usize, needle: &[u8]) -> Option<usize> {
haystack[start..]
.windows(needle.len())
.position(|window| window == needle)
.map(|offset| start + offset)
}
fn css_points(text: &str) -> Option<Vec<usize>> {
let bytes = text.as_bytes();
let mut output = vec![0];
let (mut index, mut braces, mut paren, mut bracket) = (0, 0usize, 0usize, 0usize);
while index < bytes.len() {
match bytes[index] {
b'\'' | b'"' => index = skip_quote(bytes, index)?,
b'/' if bytes.get(index + 1) == Some(&b'*') => {
index = find_bytes(bytes, index + 2, b"*/")? + 2
}
b'{' => {
braces += 1;
index += 1;
}
b'}' => {
braces = braces.checked_sub(1)?;
index += 1;
output.push(index);
}
b'(' => {
paren += 1;
index += 1;
}
b')' => {
paren = paren.checked_sub(1)?;
index += 1;
}
b'[' => {
bracket += 1;
index += 1;
}
b']' => {
bracket = bracket.checked_sub(1)?;
index += 1;
}
b';' if braces == 0 && paren == 0 && bracket == 0 => {
index += 1;
output.push(index);
}
_ => index += 1,
}
}
if braces + paren + bracket == 0 {
output.push(bytes.len());
Some(output)
} else {
None
}
}
fn html_points(text: &str) -> Option<Vec<usize>> {
let bytes = text.as_bytes();
let mut output = vec![0];
let mut stack: Vec<String> = Vec::new();
let mut index = 0;
while index < bytes.len() {
if bytes[index] != b'<' {
index = bytes[index..]
.iter()
.position(|byte| *byte == b'<')
.map_or(bytes.len(), |next| index + next);
output.push(index);
continue;
}
if starts_ci(bytes, index, b"<!--") {
index = find_bytes(bytes, index + 4, b"-->")? + 3;
output.push(index);
continue;
}
if starts_ci(bytes, index, b"</") {
let (name, end) = close_tag(bytes, index)?;
if stack.pop().as_deref() != Some(&name) {
return None;
}
index = end;
output.push(index);
continue;
}
if starts_ci(bytes, index, b"<!") || starts_ci(bytes, index, b"<?") {
index = tag_end(bytes, index + 2)?;
output.push(index);
continue;
}
let (name, end, closed, script_type) = start_tag(bytes, index)?;
if !closed && (name == "script" || name == "style") {
let (raw_end, close_end) = raw_close(bytes, end, &name)?;
let raw = std::str::from_utf8(&bytes[end..raw_end]).ok()?;
let points = if name == "style" {
css_points(raw)?
} else if script_type.as_deref().is_none_or(is_javascript_type) {
js_points(raw)?
} else {
vec![0, raw.len()]
};
output.extend(
points
.into_iter()
.filter(|point| *point > 0 && *point < raw.len())
.map(|point| end + point),
);
index = close_end;
output.push(index);
continue;
}
index = end;
if closed || is_void(&name) {
output.push(index);
} else {
stack.push(name);
}
}
if stack.is_empty() {
output.push(bytes.len());
Some(output)
} else {
None
}
}
fn starts_ci(bytes: &[u8], at: usize, needle: &[u8]) -> bool {
bytes
.get(at..at + needle.len())
.is_some_and(|value| value.eq_ignore_ascii_case(needle))
}
fn tag_end(bytes: &[u8], mut index: usize) -> Option<usize> {
let mut quote = None;
while index < bytes.len() {
let byte = bytes[index];
if let Some(mark) = quote {
if byte == mark {
quote = None;
}
} else if matches!(byte, b'\'' | b'"') {
quote = Some(byte);
} else if byte == b'>' {
return Some(index + 1);
}
index += 1;
}
None
}
fn start_tag(bytes: &[u8], at: usize) -> Option<(String, usize, bool, Option<String>)> {
let mut index = at + 1;
let start = index;
while index < bytes.len()
&& (ident_continue(bytes[index]) || matches!(bytes[index], b':' | b'-'))
{
index += 1;
}
if index == start {
return None;
}
let name = std::str::from_utf8(&bytes[start..index])
.ok()?
.to_ascii_lowercase();
let mut kind = None;
loop {
while bytes.get(index).is_some_and(u8::is_ascii_whitespace) {
index += 1;
}
if bytes.get(index) == Some(&b'>') {
return Some((name, index + 1, false, kind));
}
if bytes.get(index) == Some(&b'/') && bytes.get(index + 1) == Some(&b'>') {
return Some((name, index + 2, true, kind));
}
let attribute_start = index;
while index < bytes.len()
&& (ident_continue(bytes[index]) || matches!(bytes[index], b':' | b'-'))
{
index += 1;
}
if index == attribute_start {
return None;
}
let attribute = std::str::from_utf8(&bytes[attribute_start..index])
.ok()?
.to_ascii_lowercase();
while bytes.get(index).is_some_and(u8::is_ascii_whitespace) {
index += 1;
}
let mut value = "";
if bytes.get(index) == Some(&b'=') {
index += 1;
while bytes.get(index).is_some_and(u8::is_ascii_whitespace) {
index += 1;
}
let quote = *bytes.get(index)?;
let value_start;
if matches!(quote, b'\'' | b'"') {
index += 1;
value_start = index;
while bytes.get(index).is_some_and(|byte| *byte != quote) {
index += 1;
}
value = std::str::from_utf8(&bytes[value_start..index]).ok()?;
index += 1;
} else {
value_start = index;
while bytes
.get(index)
.is_some_and(|byte| !byte.is_ascii_whitespace() && !matches!(byte, b'>' | b'/'))
{
index += 1;
}
value = std::str::from_utf8(&bytes[value_start..index]).ok()?;
}
}
if attribute == "type" {
kind = Some(value.trim().to_ascii_lowercase());
}
}
}
fn close_tag(bytes: &[u8], at: usize) -> Option<(String, usize)> {
let mut index = at + 2;
let start = index;
while index < bytes.len()
&& (ident_continue(bytes[index]) || matches!(bytes[index], b':' | b'-'))
{
index += 1;
}
let name = std::str::from_utf8(&bytes[start..index])
.ok()?
.to_ascii_lowercase();
while bytes.get(index).is_some_and(u8::is_ascii_whitespace) {
index += 1;
}
(bytes.get(index) == Some(&b'>') && !name.is_empty()).then_some((name, index + 1))
}
fn raw_close(bytes: &[u8], mut index: usize, name: &str) -> Option<(usize, usize)> {
while index < bytes.len() {
let next = bytes[index..].iter().position(|byte| *byte == b'<')? + index;
if starts_ci(bytes, next, b"</")
&& let Some((found, end)) = close_tag(bytes, next)
&& found == name
{
return Some((next, end));
}
index = next + 1;
}
None
}
fn is_javascript_type(value: &str) -> bool {
matches!(
value,
"" | "module"
| "text/javascript"
| "application/javascript"
| "text/ecmascript"
| "application/ecmascript"
)
}
fn is_void(name: &str) -> bool {
matches!(
name,
"area"
| "base"
| "br"
| "col"
| "embed"
| "hr"
| "img"
| "input"
| "link"
| "meta"
| "param"
| "source"
| "track"
| "wbr"
)
}