1use crate::{cli::PageArgs, error::Error};
2use roxmltree::{Document, Node};
3use serde_json::{Value, json};
4
5const NS: &str = "http://schemas.microsoft.com/office/onenote/2013/onenote";
6pub const FIELDS: &[&str] = &[
7 "id",
8 "name",
9 "kind",
10 "parent_id",
11 "path",
12 "last_modified",
13 "locked",
14];
15
16fn parse(xml: &str) -> Result<Document<'_>, Error> {
17 let doc =
18 Document::parse(xml).map_err(|e| Error::desktop(format!("Invalid OneNote XML: {e}")))?;
19 if doc.root_element().tag_name().namespace() != Some(NS) {
20 return Err(Error::desktop("Unexpected OneNote XML namespace"));
21 }
22 Ok(doc)
23}
24fn is(node: Node<'_, '_>, name: &str) -> bool {
25 node.has_tag_name((NS, name))
26}
27fn hidden(node: Node<'_, '_>) -> bool {
28 node.ancestors().any(|n| {
29 n.attribute("isRecycleBin") == Some("true") || n.attribute("isInRecycleBin") == Some("true")
30 })
31}
32fn record(node: Node<'_, '_>) -> Result<Value, Error> {
33 let id = node
34 .attribute("ID")
35 .filter(|id| !id.is_empty())
36 .ok_or_else(|| Error::desktop("OneNote returned a resource without an ID"))?;
37 let path: Vec<_> = node
38 .ancestors()
39 .filter(|n| n.is_element())
40 .filter_map(|n| n.attribute("name"))
41 .collect();
42 Ok(json!({
43 "id": id, "name": node.attribute("name").unwrap_or("Untitled"),
44 "kind": node.tag_name().name().to_ascii_lowercase(),
45 "parent_id": node.parent_element().and_then(|n| n.attribute("ID")),
46 "path": path.into_iter().rev().collect::<Vec<_>>().join(" / "),
47 "last_modified": node.attribute("lastModifiedTime"),
48 "locked": node.ancestors().any(|n| n.attribute("locked") == Some("true")),
49 }))
50}
51pub fn validate_fields(page: &PageArgs) -> Result<(), Error> {
52 for field in &page.fields {
53 if !FIELDS.contains(&field.as_str()) {
54 return Err(Error::invalid(format!(
55 "Unknown field '{field}'; choose from {}",
56 FIELDS.join(", ")
57 )));
58 }
59 }
60 Ok(())
61}
62pub fn collection(xml: &str, kind: &str, page: &PageArgs) -> Result<Value, Error> {
63 collect(xml, kind, page, false)
64}
65pub fn search(xml: &str, page: &PageArgs) -> Result<Value, Error> {
66 collect(xml, "Page", page, true)
67}
68fn collect(xml: &str, kind: &str, page: &PageArgs, search: bool) -> Result<Value, Error> {
69 validate_fields(page)?;
70 let doc = parse(xml)?;
71 if kind == "Page" && doc.root_element().attribute("locked") == Some("true") {
72 return Err(Error::desktop(
73 "This section is password-protected. Unlock it in OneNote desktop and retry.",
74 ));
75 }
76 let mut nodes: Vec<_> = doc
77 .descendants()
78 .filter(|n| is(*n, kind) && !hidden(*n))
79 .collect();
80 let unindexed_count = if search {
83 let before = nodes.len();
84 nodes.retain(|node| node.attribute("isIndexed") != Some("false"));
85 before - nodes.len()
86 } else {
87 0
88 };
89 let total = nodes.len();
90 let start = page.offset as usize;
91 let mut items = Vec::new();
92 for node in nodes.into_iter().skip(start).take(page.limit as usize) {
93 let mut item = record(node)?;
94 if !page.fields.is_empty() {
95 item.as_object_mut()
96 .unwrap()
97 .retain(|k, _| page.fields.contains(k));
98 }
99 items.push(item);
100 }
101 let next = start + items.len();
102 let mut result = json!({"items":items, "total":total, "next_offset": if next < total { Some(next) } else { None }, "truncated":next < total});
103 if search {
104 result["indexing_pending"] = json!(unindexed_count > 0);
105 result["unindexed_count"] = json!(unindexed_count);
106 }
107 Ok(result)
108}
109pub fn read(xml: &str, expected_id: &str, include_xml: bool) -> Result<Value, Error> {
110 let doc = parse(xml)?;
111 let root = doc.root_element();
112 if !is(root, "Page") || root.attribute("ID") != Some(expected_id) {
113 return Err(Error::desktop(
114 "OneNote returned a different page than requested",
115 ));
116 }
117 let mut result = record(root)?;
118 let mut fragments = Vec::new();
119 for node in root.descendants().filter(|n| is(*n, "T")) {
120 if node.ancestors().any(|n| is(n, "Title")) {
121 continue;
122 }
123 let html: String = node.children().filter_map(|n| n.text()).collect();
124 let rendered = html2md::parse_html(&html);
125 if !rendered.trim().is_empty() {
126 fragments.push(rendered.trim().to_owned());
127 }
128 }
129 result["markdown"] = json!(fragments.join("\n\n"));
130 result["has_images"] = json!(root.descendants().any(|n| is(n, "Image")));
131 result["has_attachments"] = json!(root.descendants().any(|n| is(n, "InsertedFile")));
132 if include_xml {
133 result["xml"] = json!(xml);
134 }
135 Ok(result)
136}
137
138fn normalized(text: &str) -> String {
139 text.split_whitespace()
140 .collect::<Vec<_>>()
141 .join(" ")
142 .to_lowercase()
143}
144
145fn html_text(html: &str) -> String {
147 use html5ever::{parse_document, tendril::TendrilSink};
148 use markup5ever_rcdom::{NodeData, RcDom};
149 let dom = parse_document(RcDom::default(), Default::default()).one(html);
150 let mut stack = vec![(dom.document.clone(), false)];
151 let mut text = String::new();
152 while let Some((node, closing)) = stack.pop() {
153 if closing {
154 text.push(' ');
155 continue;
156 }
157 match &node.data {
158 NodeData::Text { contents } => text.push_str(&contents.borrow()),
159 NodeData::Element { name, .. } => {
160 if matches!(name.local.as_ref(), "script" | "style") {
161 continue;
162 }
163 if matches!(name.local.as_ref(), "br" | "p" | "div" | "li" | "tr" | "td") {
164 text.push(' ');
165 stack.push((node.clone(), true));
166 }
167 }
168 _ => {}
169 }
170 stack.extend(
171 node.children
172 .borrow()
173 .iter()
174 .rev()
175 .cloned()
176 .map(|node| (node, false)),
177 );
178 }
179 text
180}
181
182pub fn scan(batch: &Value, query: &str, page: &PageArgs) -> Result<Value, Error> {
183 validate_fields(page)?;
184 let hierarchy = parse(
185 batch["xml"]
186 .as_str()
187 .ok_or_else(|| Error::desktop("Scan response has no hierarchy"))?,
188 )?;
189 let pages = batch["pages"]
190 .as_array()
191 .ok_or_else(|| Error::desktop("Scan response has no pages"))?;
192 let mut skipped = batch["skipped_pages"]
193 .as_array()
194 .ok_or_else(|| Error::desktop("Scan response has no skipped-page metadata"))?
195 .clone();
196 let query = normalized(query);
197 let mut matches = Vec::new();
198 let mut scanned = 0;
199 for entry in pages {
200 let id = entry["id"]
201 .as_str()
202 .ok_or_else(|| Error::desktop("Scan page has no ID"))?;
203 let node = hierarchy
204 .descendants()
205 .find(|n| is(*n, "Page") && n.attribute("ID") == Some(id) && !hidden(*n))
206 .ok_or_else(|| {
207 Error::desktop("Scan returned a page outside the requested hierarchy")
208 })?;
209 if node
210 .ancestors()
211 .any(|n| n.attribute("locked") == Some("true"))
212 {
213 return Err(Error::desktop(
214 "Scan returned content from a locked section",
215 ));
216 }
217 let content = entry["xml"]
218 .as_str()
219 .ok_or_else(|| Error::desktop("Scan page has no XML"))?;
220 let doc = match parse(content) {
221 Ok(doc)
222 if is(doc.root_element(), "Page")
223 && doc.root_element().attribute("ID") == Some(id) =>
224 {
225 doc
226 }
227 _ => {
228 skipped.push(json!({"id":id,"reason":"invalid_page_xml"}));
229 continue;
230 }
231 };
232 scanned += 1;
233 let root = doc.root_element();
234 let mut text = root.attribute("name").unwrap_or("").to_owned();
235 for part in root.descendants().filter(|n| is(*n, "T")) {
236 text.push('\n');
237 let html: String = part.children().filter_map(|n| n.text()).collect();
238 text.push_str(&html_text(&html));
239 }
240 if normalized(&text).contains(&query) {
241 let mut item = record(node)?;
242 if !page.fields.is_empty() {
243 item.as_object_mut()
244 .unwrap()
245 .retain(|key, _| page.fields.contains(key));
246 }
247 matches.push(item);
248 }
249 }
250 let total = matches.len();
251 let start = page.offset as usize;
252 let items: Vec<_> = matches
253 .into_iter()
254 .skip(start)
255 .take(page.limit as usize)
256 .collect();
257 let next = start + items.len();
258 let incomplete = !batch["next_scan_offset"].is_null()
259 || !skipped.is_empty()
260 || batch["skipped_locked_sections"].as_u64().unwrap_or(0) > 0;
261 Ok(
262 json!({"items":items,"total":total,"next_offset":if next < total {Some(next)} else {None},
263 "truncated":next < total,"search_mode":"scan","incomplete":incomplete,
264 "scan_offset":batch["scan_offset"],"next_scan_offset":batch["next_scan_offset"],
265 "candidate_pages":batch["candidate_pages"],"attempted_pages":batch["attempted_pages"],
266 "scanned_pages":scanned,"skipped_locked_sections":batch["skipped_locked_sections"],"skipped_pages":skipped}),
267 )
268}