1use std::fmt;
21use std::ops::Range;
22
23use crate::xml::{Attribute, Element, Name, Node, Ns};
24
25#[derive(Debug, Clone, PartialEq, Eq)]
28pub enum Refused {
29 Covered,
31 Formula,
33 NotFound,
35}
36
37impl fmt::Display for Refused {
38 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
39 match self {
40 Self::Covered => write!(f, "the cell is covered by a neighbour's span"),
41 Self::Formula => write!(f, "the cell holds a formula"),
42 Self::NotFound => write!(f, "nothing is there to edit"),
43 }
44 }
45}
46
47impl std::error::Error for Refused {}
48
49#[derive(Debug, Clone, Copy, PartialEq, Eq)]
51enum Kind {
52 Text(usize),
54 Spaces(usize),
56 Tab,
58 LineBreak,
60 Marker,
62}
63
64impl Kind {
65 fn len(self) -> usize {
66 match self {
67 Self::Text(n) | Self::Spaces(n) => n,
68 Self::Tab | Self::LineBreak => 1,
69 Self::Marker => 0,
70 }
71 }
72}
73
74#[derive(Debug)]
76struct Segment {
77 path: Vec<usize>,
79 start: usize,
81 kind: Kind,
82}
83
84fn is_inline_container(element: &Element) -> bool {
89 element.name.ns == Ns::Text
90 && !element.children.is_empty()
91 && matches!(
92 &*element.name.local,
93 "span" | "a" | "bibliography-mark" | "ruby" | "ruby-base" | "meta" | "meta-field"
94 )
95}
96
97fn is_inline_container_name(element: &Element) -> bool {
100 element.name.ns == Ns::Text
101 && matches!(
102 &*element.name.local,
103 "span" | "a" | "bibliography-mark" | "ruby" | "ruby-base" | "meta" | "meta-field"
104 )
105}
106
107fn segments(paragraph: &Element) -> Vec<Segment> {
108 let mut out = Vec::new();
109 let mut path = Vec::new();
110 let mut at = 0;
111 collect(paragraph, &mut path, &mut at, &mut out);
112 out
113}
114
115fn collect(parent: &Element, path: &mut Vec<usize>, at: &mut usize, out: &mut Vec<Segment>) {
116 for (index, child) in parent.children.iter().enumerate() {
117 let kind = match child {
118 Node::Text(t) | Node::CData(t) => Kind::Text(t.chars().count()),
119 Node::Comment(_) | Node::ProcessingInstruction(_) => continue,
120 Node::Element(e) if e.is(&Ns::Text, "s") => {
121 Kind::Spaces(e.attr_usize(&Ns::Text, "c").unwrap_or(1))
122 }
123 Node::Element(e) if e.is(&Ns::Text, "tab") => Kind::Tab,
124 Node::Element(e) if e.is(&Ns::Text, "line-break") => Kind::LineBreak,
125 Node::Element(e) if is_inline_container(e) => {
126 path.push(index);
127 collect(e, path, at, out);
128 path.pop();
129 continue;
130 }
131 Node::Element(_) => Kind::Marker,
132 };
133 path.push(index);
134 out.push(Segment {
135 path: path.clone(),
136 start: *at,
137 kind,
138 });
139 path.pop();
140 *at += kind.len();
141 }
142}
143
144pub fn text(paragraph: &Element) -> String {
148 let mut out = String::new();
149 write_text(paragraph, &mut out);
150 out
151}
152
153fn write_text(parent: &Element, out: &mut String) {
154 for child in &parent.children {
155 match child {
156 Node::Text(t) | Node::CData(t) => out.push_str(t),
157 Node::Element(e) if e.is(&Ns::Text, "s") => {
158 for _ in 0..e.attr_usize(&Ns::Text, "c").unwrap_or(1) {
159 out.push(' ');
160 }
161 }
162 Node::Element(e) if e.is(&Ns::Text, "tab") => out.push('\t'),
163 Node::Element(e) if e.is(&Ns::Text, "line-break") => out.push('\n'),
164 Node::Element(e) if is_inline_container(e) => write_text(e, out),
165 Node::Element(_) | Node::Comment(_) | Node::ProcessingInstruction(_) => {}
166 }
167 }
168}
169
170pub fn replace(paragraph: &mut Element, range: Range<usize>, with: &str) {
181 let len = text(paragraph).chars().count();
182 let start = range.start.min(len);
183 let end = range.end.clamp(start, len);
184 let added = with.chars().count();
185 let holder = segments(paragraph).into_iter().find(|s| {
189 matches!(s.kind, Kind::Text(_)) && s.start <= start && start < s.start + s.kind.len()
190 });
191 if start < end
192 && added > 0
193 && let Some(segment) = holder
194 {
195 insert_into(paragraph, &segment.path, start - segment.start, with);
196 cut(paragraph, start + added..end + added, |_| false);
197 } else {
198 cut(paragraph, start..end, |_| false);
199 insert(paragraph, start, with);
200 }
201 normalize(paragraph);
202}
203
204pub fn split(paragraph: &Element, at: usize) -> (Element, Element) {
211 let len = text(paragraph).chars().count();
212 let at = at.min(len);
213 let mut first = paragraph.clone();
214 cut(&mut first, at..len, |start| start > at);
215 let mut second = paragraph.clone();
216 cut(&mut second, 0..at, |start| start <= at);
217 second.attrs.retain(|a| {
218 !(a.name.is(&Ns::Text, "id")
219 || a.name.local.as_ref() == "id" && a.name.prefix.as_deref() == Some("xml"))
220 });
221 normalize(&mut first);
222 normalize(&mut second);
223 (first, second)
224}
225
226pub fn join(first: &mut Element, second: Element) {
229 first.children.extend(second.children);
230 first.self_closing = false;
231 normalize(first);
232}
233
234pub fn rewrite(paragraph: &Element, edited: &str) -> Vec<Element> {
245 let before = text(paragraph);
246 let old: Vec<char> = before.chars().collect();
247 let new: Vec<char> = edited.chars().collect();
248 let prefix = old.iter().zip(&new).take_while(|(a, b)| a == b).count();
249 let suffix = old[prefix..]
250 .iter()
251 .rev()
252 .zip(new[prefix..].iter().rev())
253 .take_while(|(a, b)| a == b)
254 .count();
255 let inserted: String = new[prefix..new.len() - suffix].iter().collect();
256
257 let mut whole = paragraph.clone();
258 replace(&mut whole, prefix..old.len() - suffix, &inserted);
259
260 let mut breaks: Vec<usize> = inserted
263 .chars()
264 .enumerate()
265 .filter(|(_, c)| *c == '\n')
266 .map(|(i, _)| prefix + i)
267 .collect();
268 breaks.reverse();
269 let mut after = Vec::new();
270 for at in breaks {
271 let (first, mut second) = split(&whole, at);
272 replace(&mut second, 0..1, "");
274 after.push(second);
275 whole = first;
276 }
277 after.push(whole);
278 after.reverse();
279 after
280}
281
282pub fn apply(root: &mut Element, path: &[usize], edited: &str) -> Result<usize, Refused> {
289 let (last, above) = path.split_last().ok_or(Refused::NotFound)?;
290 let parent = root.at_mut(above).ok_or(Refused::NotFound)?;
291 let Some(Node::Element(paragraph)) = parent.children.get(*last) else {
292 return Err(Refused::NotFound);
293 };
294 if !(paragraph.is(&Ns::Text, "p") || paragraph.is(&Ns::Text, "h")) {
295 return Err(Refused::NotFound);
296 }
297 let paragraphs = rewrite(paragraph, edited);
298 let count = paragraphs.len();
299 parent
300 .children
301 .splice(*last..=*last, paragraphs.into_iter().map(Node::Element));
302 Ok(count)
303}
304
305pub fn join_with_previous(root: &mut Element, path: &[usize]) -> Result<Vec<usize>, Refused> {
312 let (last, above) = path.split_last().ok_or(Refused::NotFound)?;
313 let parent = root.at_mut(above).ok_or(Refused::NotFound)?;
314 let is_paragraph = |node: &Node| matches!(node, Node::Element(e) if e.is(&Ns::Text, "p") || e.is(&Ns::Text, "h"));
315 if !parent.children.get(*last).is_some_and(is_paragraph) {
316 return Err(Refused::NotFound);
317 }
318 let previous = parent.children[..*last]
319 .iter()
320 .rposition(is_paragraph)
321 .ok_or(Refused::NotFound)?;
322 let Node::Element(second) = parent.children.remove(*last) else {
325 return Err(Refused::NotFound);
326 };
327 parent.children.drain(previous + 1..*last);
328 let Some(Node::Element(first)) = parent.children.get_mut(previous) else {
329 return Err(Refused::NotFound);
330 };
331 join(first, second);
332 let mut at = above.to_vec();
333 at.push(previous);
334 Ok(at)
335}
336
337fn cut(paragraph: &mut Element, range: Range<usize>, drop_marker: impl Fn(usize) -> bool) {
343 for segment in segments(paragraph).into_iter().rev() {
344 let end = segment.start + segment.kind.len();
345 let overlap_start = range.start.max(segment.start);
346 let overlap_end = range.end.min(end);
347 let overlaps = overlap_start < overlap_end;
348 let remove = match segment.kind {
349 Kind::Marker => drop_marker(segment.start),
350 Kind::Tab | Kind::LineBreak => overlaps,
351 Kind::Spaces(n) => {
352 if !overlaps {
353 continue;
354 }
355 let left = n - (overlap_end - overlap_start);
356 if left > 0
357 && let Some(e) = paragraph.at_mut(&segment.path)
358 {
359 set_count(e, left);
360 }
361 left == 0
362 }
363 Kind::Text(_) => {
364 if !overlaps {
365 continue;
366 }
367 let Some((parent, index)) = parent_of(paragraph, &segment.path) else {
368 continue;
369 };
370 let Some(Node::Text(t) | Node::CData(t)) = parent.children.get_mut(index) else {
371 continue;
372 };
373 let kept: String = t
374 .chars()
375 .enumerate()
376 .filter(|(i, _)| {
377 let at = segment.start + i;
378 !(overlap_start..overlap_end).contains(&at)
379 })
380 .map(|(_, c)| c)
381 .collect();
382 *t = kept;
383 t.is_empty()
384 }
385 };
386 if remove && let Some((parent, index)) = parent_of(paragraph, &segment.path) {
387 parent.children.remove(index);
388 prune_emptied(paragraph, &segment.path[..segment.path.len() - 1]);
389 }
390 }
391}
392
393fn prune_emptied(paragraph: &mut Element, container: &[usize]) {
401 let mut path = container.to_vec();
402 while !path.is_empty() {
403 let Some((parent, index)) = parent_of(paragraph, &path) else {
404 return;
405 };
406 match parent.children.get(index) {
407 Some(Node::Element(e)) if is_inline_container_name(e) && e.children.is_empty() => {
408 parent.children.remove(index);
409 path.pop();
410 }
411 _ => return,
412 }
413 }
414}
415
416fn insert(paragraph: &mut Element, at: usize, with: &str) {
423 if with.is_empty() {
424 return;
425 }
426 let segments = segments(paragraph);
427 let text_at = |predicate: &dyn Fn(&Segment, usize) -> bool| {
428 segments
429 .iter()
430 .find(|s| matches!(s.kind, Kind::Text(_)) && predicate(s, s.start + s.kind.len()))
431 };
432 let target = text_at(&|s, end| s.start < at && at < end)
433 .or_else(|| text_at(&|_, end| end == at))
434 .or_else(|| text_at(&|s, _| s.start == at));
435 if let Some(segment) = target {
436 insert_into(paragraph, &segment.path, at - segment.start, with);
437 return;
438 }
439 let after = segments
443 .iter()
444 .rev()
445 .find(|s| s.start + s.kind.len() == at)
446 .map(|s| s.path.clone());
447 let before = segments
448 .iter()
449 .find(|s| s.start >= at)
450 .map(|s| s.path.clone());
451 let node = Node::Text(with.to_owned());
452 if let Some(path) = after
453 && let Some((parent, index)) = parent_of(paragraph, &path)
454 {
455 parent.children.insert(index + 1, node);
456 } else if let Some(path) = before
457 && let Some((parent, index)) = parent_of(paragraph, &path)
458 {
459 parent.children.insert(index, node);
460 } else {
461 paragraph.children.push(node);
462 paragraph.self_closing = false;
463 }
464}
465
466fn insert_into(paragraph: &mut Element, path: &[usize], offset: usize, with: &str) {
468 if let Some((parent, index)) = parent_of(paragraph, path)
469 && let Some(Node::Text(t) | Node::CData(t)) = parent.children.get_mut(index)
470 {
471 let byte = t.char_indices().nth(offset).map_or(t.len(), |(b, _)| b);
472 t.insert_str(byte, with);
473 }
474}
475
476fn parent_of<'a>(paragraph: &'a mut Element, path: &[usize]) -> Option<(&'a mut Element, usize)> {
478 let (last, above) = path.split_last()?;
479 Some((paragraph.at_mut(above)?, *last))
480}
481
482fn set_count(space: &mut Element, count: usize) {
483 if count <= 1 {
484 space.remove_attr(&Ns::Text, "c");
485 } else {
486 let name = space
487 .attrs
488 .iter()
489 .find(|a| a.name.is(&Ns::Text, "c"))
490 .map_or_else(
491 || {
492 Name::new(
493 space.name.prefix.as_deref().unwrap_or("text"),
494 "c",
495 Ns::Text,
496 )
497 },
498 |a| a.name.clone(),
499 );
500 space.set_attr(name, count.to_string());
501 }
502}
503
504fn normalize(paragraph: &mut Element) {
513 let prefix = paragraph
514 .name
515 .prefix
516 .as_deref()
517 .unwrap_or("text")
518 .to_owned();
519 let mut previous_was_space = true;
520 normalize_in(paragraph, &prefix, &mut previous_was_space);
521}
522
523fn normalize_in(parent: &mut Element, prefix: &str, previous_was_space: &mut bool) {
524 merge_text(parent);
525 let mut index = 0;
526 while index < parent.children.len() {
527 match &mut parent.children[index] {
528 Node::Text(t) | Node::CData(t) => {
529 let replacement = encode(t, prefix, previous_was_space);
530 match replacement {
531 None => index += 1,
532 Some(nodes) => {
533 let count = nodes.len();
534 parent.children.splice(index..=index, nodes);
535 index += count;
536 }
537 }
538 }
539 Node::Element(e) if is_inline_container(e) => {
540 normalize_in(e, prefix, previous_was_space);
541 index += 1;
542 }
543 Node::Element(e) => {
544 if e.is(&Ns::Text, "s") || e.is(&Ns::Text, "tab") || e.is(&Ns::Text, "line-break") {
547 *previous_was_space = true;
548 }
549 index += 1;
550 }
551 Node::Comment(_) | Node::ProcessingInstruction(_) => index += 1,
552 }
553 }
554}
555
556fn merge_text(parent: &mut Element) {
563 let mut index = 1;
564 while index < parent.children.len() {
565 let joinable = matches!(
566 (&parent.children[index - 1], &parent.children[index]),
567 (Node::Text(_), Node::Text(_))
568 );
569 if joinable {
570 let Node::Text(tail) = parent.children.remove(index) else {
571 unreachable!("matched a text node");
572 };
573 if let Node::Text(head) = &mut parent.children[index - 1] {
574 head.push_str(&tail);
575 }
576 } else {
577 index += 1;
578 }
579 }
580}
581
582fn encode(text: &str, prefix: &str, previous_was_space: &mut bool) -> Option<Vec<Node>> {
585 let needs_work = text.contains(['\t', '\n', '\r'])
586 || text.contains(" ")
587 || (*previous_was_space && text.starts_with(' '));
588 if !needs_work {
589 if let Some(last) = text.chars().last() {
590 *previous_was_space = last == ' ';
591 }
592 return None;
593 }
594 let mut nodes = Vec::new();
595 let mut run = String::new();
596 let mut spaces = 0usize;
597 let flush_spaces = |nodes: &mut Vec<Node>,
598 run: &mut String,
599 spaces: &mut usize,
600 previous_was_space: &mut bool| {
601 if *spaces == 0 {
602 return;
603 }
604 let mut counted = *spaces;
607 if !*previous_was_space {
608 run.push(' ');
609 counted -= 1;
610 }
611 if counted > 0 {
612 if !run.is_empty() {
613 nodes.push(Node::Text(std::mem::take(run)));
614 }
615 let mut s = Element::new(prefix, "s", Ns::Text);
616 if counted > 1 {
617 s.attrs.push(Attribute {
618 name: Name::new(prefix, "c", Ns::Text),
619 value: counted.to_string(),
620 });
621 }
622 nodes.push(Node::Element(s));
623 }
624 *spaces = 0;
625 *previous_was_space = true;
626 };
627 for c in text.chars() {
628 match c {
629 ' ' => spaces += 1,
630 '\r' => {}
633 '\t' | '\n' => {
634 flush_spaces(&mut nodes, &mut run, &mut spaces, previous_was_space);
635 if !run.is_empty() {
636 nodes.push(Node::Text(std::mem::take(&mut run)));
637 }
638 let local = if c == '\t' { "tab" } else { "line-break" };
639 nodes.push(Node::Element(Element::new(prefix, local, Ns::Text)));
640 *previous_was_space = true;
641 }
642 other => {
643 flush_spaces(&mut nodes, &mut run, &mut spaces, previous_was_space);
644 run.push(other);
645 *previous_was_space = false;
646 }
647 }
648 }
649 flush_spaces(&mut nodes, &mut run, &mut spaces, previous_was_space);
650 if !run.is_empty() {
651 nodes.push(Node::Text(run));
652 }
653 Some(nodes)
654}