Skip to main content

odox_core/
edit.rs

1//! Changing the text of a paragraph, and what an edit may refuse.
2//!
3//! A paragraph's text is not one string in the tree. It is text nodes, inside
4//! spans and links and marks, with `text:s`, `text:tab` and `text:line-break`
5//! standing for the whitespace ODF will not write literally, and with elements
6//! that contribute no characters at all — a bookmark, a frame, a note, a field
7//! — sitting between them. An editor works on the flat string and this module
8//! maps the edit back: characters are deleted from the nodes that hold them,
9//! new text goes into the node at the insertion point, and every zero-length
10//! element stays where it was. DESIGN.md §11.
11//!
12//! The string the editor sees, [`text`], is built from the tree and not from
13//! the renderer's layout, because the two differ: the renderer draws a tab as
14//! four spaces and a footnote as its citation. Here a tab is a tab, and a
15//! footnote contributes nothing and is kept.
16//
17// Author: David M. Anderson
18// Built with AI assistance (Claude, Anthropic)
19
20mod format;
21
22use std::fmt;
23use std::ops::Range;
24
25use crate::xml::{Attribute, Element, Name, Node, Ns};
26
27pub use format::{Mark, format, marked};
28
29/// Why an edit was not made. Each is a state of the document rather than a
30/// failure, and the window says which.
31#[derive(Debug, Clone, PartialEq, Eq)]
32pub enum Refused {
33    /// The cell is under a neighbour's span; the neighbour is the one to edit.
34    Covered,
35    /// The cell holds a formula, which this version does not edit.
36    Formula,
37    /// The document does not declare a namespace the edit would write in.
38    Namespace,
39    /// There is no such sheet, slide, shape or paragraph.
40    NotFound,
41    /// An end of the range is inside a table, a cell or a frame the range
42    /// does not wholly contain, which a join of text does not take apart.
43    Structure,
44}
45
46impl fmt::Display for Refused {
47    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
48        match self {
49            Self::Covered => write!(f, "the cell is covered by a neighbour's span"),
50            Self::Formula => write!(f, "the cell holds a formula"),
51            Self::Namespace => write!(f, "the document does not declare the namespace needed"),
52            Self::NotFound => write!(f, "nothing is there to edit"),
53            Self::Structure => write!(f, "the range crosses a table or a frame"),
54        }
55    }
56}
57
58impl std::error::Error for Refused {}
59
60/// What one node contributes to a paragraph's text.
61#[derive(Debug, Clone, Copy, PartialEq, Eq)]
62enum Kind {
63    /// A text node, this many characters long.
64    Text(usize),
65    /// `text:s`, this many spaces.
66    Spaces(usize),
67    /// `text:tab`.
68    Tab,
69    /// `text:line-break`.
70    LineBreak,
71    /// An element that contributes nothing and is kept where it is.
72    Marker,
73}
74
75impl Kind {
76    fn len(self) -> usize {
77        match self {
78            Self::Text(n) | Self::Spaces(n) => n,
79            Self::Tab | Self::LineBreak => 1,
80            Self::Marker => 0,
81        }
82    }
83}
84
85/// One node's place in the paragraph's text.
86#[derive(Debug)]
87struct Segment {
88    /// The route from the paragraph to the node.
89    path: Vec<usize>,
90    /// Where its characters begin.
91    start: usize,
92    kind: Kind,
93}
94
95/// Whether an element holds text on the paragraph's behalf: a span, a link, a
96/// ruby or a field whose contents are the paragraph's own characters. Anything
97/// else inside a paragraph is a marker, and stays; so is one of these with
98/// nothing in it, which is what lets a split carry it to one side.
99fn is_inline_container(element: &Element) -> bool {
100    element.name.ns == Ns::Text
101        && !element.children.is_empty()
102        && matches!(
103            &*element.name.local,
104            "span" | "a" | "bibliography-mark" | "ruby" | "ruby-base" | "meta" | "meta-field"
105        )
106}
107
108/// Whether an element is a paragraph or a heading: what an editor edits.
109pub fn is_paragraph(element: &Element) -> bool {
110    element.is(&Ns::Text, "p") || element.is(&Ns::Text, "h")
111}
112
113/// Whether an element inside a paragraph holds characters of the paragraph's
114/// text, as a span or a link does, rather than being a mark or a field that
115/// contributes none of its own.
116pub fn holds_text(element: &Element) -> bool {
117    is_inline_container_name(element)
118}
119
120/// The containers above, whether or not they hold anything: what an edit may
121/// drop once it has emptied one.
122fn is_inline_container_name(element: &Element) -> bool {
123    element.name.ns == Ns::Text
124        && matches!(
125            &*element.name.local,
126            "span" | "a" | "bibliography-mark" | "ruby" | "ruby-base" | "meta" | "meta-field"
127        )
128}
129
130fn segments(paragraph: &Element) -> Vec<Segment> {
131    let mut out = Vec::new();
132    let mut path = Vec::new();
133    let mut at = 0;
134    collect(paragraph, &mut path, &mut at, &mut out);
135    out
136}
137
138fn collect(parent: &Element, path: &mut Vec<usize>, at: &mut usize, out: &mut Vec<Segment>) {
139    for (index, child) in parent.children.iter().enumerate() {
140        let kind = match child {
141            Node::Text(t) | Node::CData(t) => Kind::Text(t.chars().count()),
142            Node::Comment(_) | Node::ProcessingInstruction(_) => continue,
143            Node::Element(e) if e.is(&Ns::Text, "s") => {
144                Kind::Spaces(e.attr_usize(&Ns::Text, "c").unwrap_or(1))
145            }
146            Node::Element(e) if e.is(&Ns::Text, "tab") => Kind::Tab,
147            Node::Element(e) if e.is(&Ns::Text, "line-break") => Kind::LineBreak,
148            Node::Element(e) if is_inline_container(e) => {
149                path.push(index);
150                collect(e, path, at, out);
151                path.pop();
152                continue;
153            }
154            Node::Element(_) => Kind::Marker,
155        };
156        path.push(index);
157        out.push(Segment {
158            path: path.clone(),
159            start: *at,
160            kind,
161        });
162        path.pop();
163        *at += kind.len();
164    }
165}
166
167/// The paragraph's text as an editor sees it: every character the tree
168/// holds, in order, with ODF's whitespace elements as the characters they
169/// stand for.
170pub fn text(paragraph: &Element) -> String {
171    let mut out = String::new();
172    write_text(paragraph, &mut out);
173    out
174}
175
176fn write_text(parent: &Element, out: &mut String) {
177    for child in &parent.children {
178        match child {
179            Node::Text(t) | Node::CData(t) => out.push_str(t),
180            Node::Element(e) if e.is(&Ns::Text, "s") => {
181                for _ in 0..e.attr_usize(&Ns::Text, "c").unwrap_or(1) {
182                    out.push(' ');
183                }
184            }
185            Node::Element(e) if e.is(&Ns::Text, "tab") => out.push('\t'),
186            Node::Element(e) if e.is(&Ns::Text, "line-break") => out.push('\n'),
187            Node::Element(e) if is_inline_container(e) => write_text(e, out),
188            Node::Element(_) | Node::Comment(_) | Node::ProcessingInstruction(_) => {}
189        }
190    }
191}
192
193/// Replace a range of the paragraph's text, in characters, with new text.
194///
195/// The characters are removed from the nodes that hold them and the new text
196/// goes into the text node at the start of the range, or into a new one where
197/// none is there. Every element that contributes no characters stays where it
198/// was, so a bookmark inside the deleted range is a bookmark still. Whitespace
199/// is then written the way ODF requires, so the new text may hold tabs,
200/// newlines and runs of spaces.
201///
202/// A range past the end of the text is clamped to it.
203pub fn replace(paragraph: &mut Element, range: Range<usize>, with: &str) {
204    let len = text(paragraph).chars().count();
205    let start = range.start.min(len);
206    let end = range.end.clamp(start, len);
207    let added = with.chars().count();
208    // Where the replaced range begins inside a text node, the new text goes
209    // into that node before the old comes out, so that what replaces a bold
210    // word is bold and a span emptied by the edit is not emptied first.
211    let holder = segments(paragraph).into_iter().find(|s| {
212        matches!(s.kind, Kind::Text(_)) && s.start <= start && start < s.start + s.kind.len()
213    });
214    if start < end
215        && added > 0
216        && let Some(segment) = holder
217    {
218        insert_into(paragraph, &segment.path, start - segment.start, with);
219        cut(paragraph, start + added..end + added, |_| false);
220    } else {
221        cut(paragraph, start..end, |_| false);
222        insert(paragraph, start, with);
223    }
224    normalize(paragraph);
225}
226
227/// Split a paragraph at a character offset into the paragraph before it and
228/// the paragraph after, each with the original's style and attributes.
229///
230/// A zero-length element at the split point stays with the first half. The
231/// second half drops any `xml:id` or `text:id`, which name one element and
232/// cannot name two.
233pub fn split(paragraph: &Element, at: usize) -> (Element, Element) {
234    let len = text(paragraph).chars().count();
235    let at = at.min(len);
236    let mut first = paragraph.clone();
237    cut(&mut first, at..len, |start| start > at);
238    let mut second = paragraph.clone();
239    cut(&mut second, 0..at, |start| start <= at);
240    second.attrs.retain(|a| {
241        !(a.name.is(&Ns::Text, "id")
242            || a.name.local.as_ref() == "id" && a.name.prefix.as_deref() == Some("xml"))
243    });
244    normalize(&mut first);
245    normalize(&mut second);
246    (first, second)
247}
248
249/// Join a paragraph onto the end of another. The first keeps its style; the
250/// second's contents, markers included, follow the first's.
251pub fn join(first: &mut Element, second: Element) {
252    first.children.extend(second.children);
253    first.self_closing = false;
254    normalize(first);
255}
256
257/// What a paragraph becomes when its whole text is replaced by what an editor
258/// hands back: one paragraph, or several where the new text holds newlines
259/// the old one did not.
260///
261/// The old and the new text are compared from both ends, and only the middle
262/// they differ in is replaced, so spans and markers outside it are untouched
263/// and a marker inside it stays where it was. A newline inside that middle is
264/// a paragraph break: the paragraph is split there, each half with the
265/// original's style, and the newline itself is not kept. A newline outside
266/// it is a `text:line-break` the editor was shown and left alone.
267pub fn rewrite(paragraph: &Element, edited: &str) -> Vec<Element> {
268    let before = text(paragraph);
269    let old: Vec<char> = before.chars().collect();
270    let new: Vec<char> = edited.chars().collect();
271    let prefix = old.iter().zip(&new).take_while(|(a, b)| a == b).count();
272    let suffix = old[prefix..]
273        .iter()
274        .rev()
275        .zip(new[prefix..].iter().rev())
276        .take_while(|(a, b)| a == b)
277        .count();
278    let inserted: String = new[prefix..new.len() - suffix].iter().collect();
279
280    let mut whole = paragraph.clone();
281    replace(&mut whole, prefix..old.len() - suffix, &inserted);
282
283    // The breaks, as offsets into the rewritten text, last first so that each
284    // split leaves the earlier offsets where they were.
285    let mut breaks: Vec<usize> = inserted
286        .chars()
287        .enumerate()
288        .filter(|(_, c)| *c == '\n')
289        .map(|(i, _)| prefix + i)
290        .collect();
291    breaks.reverse();
292    let mut after = Vec::new();
293    for at in breaks {
294        let (first, mut second) = split(&whole, at);
295        // The newline itself, which the split left at the head of the second.
296        replace(&mut second, 0..1, "");
297        after.push(second);
298        whole = first;
299    }
300    after.push(whole);
301    after.reverse();
302    after
303}
304
305/// Replace the paragraph at a path under a root with what the editor hands
306/// back, which may be several paragraphs. Answers how many now stand there.
307///
308/// # Errors
309///
310/// The path leads to nothing, or to something that is not a paragraph.
311pub fn apply(root: &mut Element, path: &[usize], edited: &str) -> Result<usize, Refused> {
312    let (last, above) = path.split_last().ok_or(Refused::NotFound)?;
313    let parent = root.at_mut(above).ok_or(Refused::NotFound)?;
314    let Some(Node::Element(paragraph)) = parent.children.get(*last) else {
315        return Err(Refused::NotFound);
316    };
317    if !is_paragraph(paragraph) {
318        return Err(Refused::NotFound);
319    }
320    let paragraphs = rewrite(paragraph, edited);
321    let count = paragraphs.len();
322    parent
323        .children
324        .splice(*last..=*last, paragraphs.into_iter().map(Node::Element));
325    Ok(count)
326}
327
328/// Split the paragraph at a path under a root at a character offset, the
329/// second half becoming the paragraph after it, and answer where the second
330/// half is.
331///
332/// In a list item the second half begins a new item after it, as Enter does
333/// in a list, and whatever followed the paragraph in the item goes with it.
334///
335/// # Errors
336///
337/// The path leads to nothing, or to something that is not a paragraph.
338pub fn split_at(root: &mut Element, path: &[usize], at: usize) -> Result<Vec<usize>, Refused> {
339    let (last, above) = path.split_last().ok_or(Refused::NotFound)?;
340    let parent = root.at_mut(above).ok_or(Refused::NotFound)?;
341    let Some(Node::Element(paragraph)) = parent.children.get(*last) else {
342        return Err(Refused::NotFound);
343    };
344    if !is_paragraph(paragraph) {
345        return Err(Refused::NotFound);
346    }
347    let (first, second) = split(paragraph, at);
348    if !parent.is(&Ns::Text, "list-item") {
349        parent
350            .children
351            .splice(*last..=*last, [Node::Element(first), Node::Element(second)]);
352        let mut second_at = above.to_vec();
353        second_at.push(last + 1);
354        return Ok(second_at);
355    }
356
357    // A new item, named the way the document names its items, holding the
358    // second half and what followed it. The item's attributes stay with the
359    // item: a start value or an id belongs to one item and not to two.
360    let mut item = Element {
361        name: parent.name.clone(),
362        attrs: Vec::new(),
363        children: vec![Node::Element(second)],
364        self_closing: false,
365    };
366    item.children.extend(parent.children.drain(last + 1..));
367    parent.children[*last] = Node::Element(first);
368    let (item_at, list_path) = above.split_last().ok_or(Refused::NotFound)?;
369    let enclosing = root.at_mut(list_path).ok_or(Refused::NotFound)?;
370    enclosing.children.insert(item_at + 1, Node::Element(item));
371    let mut second_at = list_path.to_vec();
372    second_at.extend([item_at + 1, 0]);
373    Ok(second_at)
374}
375
376/// Replace the text from one position to another with new text, and join
377/// what is left of the last paragraph onto the first. A position is a
378/// paragraph's path under the root and a character offset into its text.
379///
380/// Everything between the two goes, as a selection over it takes it:
381/// paragraphs, tables and frames the range wholly contains, and list items
382/// and lists left with nothing in them. The first paragraph keeps its style
383/// and its place, and the last one's remaining text follows the new text in
384/// it.
385///
386/// # Errors
387///
388/// Either path does not lead to a paragraph, or the first is not before the
389/// last; or an end is inside a table, a cell or a frame that the range does
390/// not wholly contain ([`Refused::Structure`]), which no join takes apart.
391/// Each is refused before anything changes.
392pub fn replace_range(
393    root: &mut Element,
394    from: (&[usize], usize),
395    to: (&[usize], usize),
396    with: &str,
397) -> Result<(), Refused> {
398    let ((from, start), (to, end)) = (from, to);
399    if !root.at(from).is_some_and(is_paragraph) || !root.at(to).is_some_and(is_paragraph) {
400        return Err(Refused::NotFound);
401    }
402    if from == to {
403        let paragraph = root.at_mut(from).ok_or(Refused::NotFound)?;
404        replace(paragraph, start..end, with);
405        return Ok(());
406    }
407    // The deepest element holding both ends, and which of its children each
408    // end is in.
409    let common = from.iter().zip(to).take_while(|(a, b)| a == b).count();
410    if common >= from.len() || common >= to.len() || from[common] >= to[common] {
411        return Err(Refused::NotFound);
412    }
413    if end_inside_structure(root, common, from, to) {
414        return Err(Refused::Structure);
415    }
416    let (first_child, last_child) = (from[common], to[common]);
417
418    // The last paragraph first: it is after everything else touched, so
419    // taking it apart moves no path still to be used.
420    let last = root.at_mut(to).ok_or(Refused::NotFound)?;
421    replace(last, 0..end, "");
422    let tail = std::mem::take(&mut last.children);
423
424    let holder = root.at_mut(&from[..common]).ok_or(Refused::NotFound)?;
425    let emptied = match holder.children.get_mut(last_child) {
426        Some(Node::Element(child)) if common + 1 < to.len() => {
427            strip_before(child, &to[common + 1..])
428        }
429        _ => true,
430    };
431    if emptied {
432        holder.children.remove(last_child);
433    }
434    holder.children.drain(first_child + 1..last_child);
435    if let Some(Node::Element(child)) = holder.children.get_mut(first_child) {
436        strip_after(child, &from[common + 1..]);
437    }
438
439    let first = root.at_mut(from).ok_or(Refused::NotFound)?;
440    let len = text(first).chars().count();
441    replace(first, start..len, with);
442    let mut rest = Element::new("text", "p", Ns::Text);
443    rest.children = tail;
444    join(first, rest);
445    Ok(())
446}
447
448/// Whether either end of a range, whose two paths agree for their first
449/// `common` steps, sits inside a table, a cell or a frame that the range
450/// does not wholly contain. One wholly between the ends goes with the rest.
451fn end_inside_structure(root: &Element, common: usize, from: &[usize], to: &[usize]) -> bool {
452    let inside = |path: &[usize]| {
453        (common + 1..path.len()).any(|depth| root.at(&path[..depth]).is_some_and(is_structure))
454    };
455    inside(from) || inside(to)
456}
457
458fn is_structure(element: &Element) -> bool {
459    element.is(&Ns::Table, "table")
460        || element.is(&Ns::Table, "table-row")
461        || element.is(&Ns::Table, "table-cell")
462        || element.is(&Ns::Table, "covered-table-cell")
463        || element.is(&Ns::Draw, "frame")
464        || element.is(&Ns::Draw, "text-box")
465}
466
467/// Remove, under an element, everything before the end of a path, the end
468/// itself, and every element that leaves empty. Answers whether the element
469/// is left with no elements in it.
470fn strip_before(element: &mut Element, path: &[usize]) -> bool {
471    let Some((&at, rest)) = path.split_first() else {
472        return true;
473    };
474    let emptied = match element.children.get_mut(at) {
475        Some(Node::Element(child)) if !rest.is_empty() => strip_before(child, rest),
476        _ => true,
477    };
478    let through = if emptied { at + 1 } else { at };
479    element
480        .children
481        .drain(..through.min(element.children.len()));
482    !element
483        .children
484        .iter()
485        .any(|n| matches!(n, Node::Element(_)))
486}
487
488/// Remove, under an element, everything after the end of a path, at every
489/// level down to it.
490fn strip_after(element: &mut Element, path: &[usize]) {
491    let Some((&at, rest)) = path.split_first() else {
492        return;
493    };
494    if let Some(Node::Element(child)) = element.children.get_mut(at) {
495        strip_after(child, rest);
496    }
497    element.children.truncate(at + 1);
498}
499
500/// Join the paragraph at a path onto the paragraph before it among its
501/// parent's children, and answer where the joined paragraph is.
502///
503/// # Errors
504///
505/// The path leads to nothing, or the element before it is not a paragraph:
506/// there is none, or a table or a list stands between, which a join would
507/// otherwise delete.
508pub fn join_with_previous(root: &mut Element, path: &[usize]) -> Result<Vec<usize>, Refused> {
509    let (last, above) = path.split_last().ok_or(Refused::NotFound)?;
510    let parent = root.at_mut(above).ok_or(Refused::NotFound)?;
511    let paragraph_node = |node: &Node| matches!(node, Node::Element(e) if is_paragraph(e));
512    if !parent.children.get(*last).is_some_and(paragraph_node) {
513        return Err(Refused::NotFound);
514    }
515    let previous = parent.children[..*last]
516        .iter()
517        .rposition(|node| matches!(node, Node::Element(_)))
518        .filter(|&index| paragraph_node(&parent.children[index]))
519        .ok_or(Refused::NotFound)?;
520    // Whitespace between the two, which was between them and is now inside
521    // neither, goes too.
522    let Node::Element(second) = parent.children.remove(*last) else {
523        return Err(Refused::NotFound);
524    };
525    parent.children.drain(previous + 1..*last);
526    let Some(Node::Element(first)) = parent.children.get_mut(previous) else {
527        return Err(Refused::NotFound);
528    };
529    join(first, second);
530    let mut at = above.to_vec();
531    at.push(previous);
532    Ok(at)
533}
534
535/// Delete a range of characters, and any zero-length element whose position
536/// the predicate names.
537///
538/// Segments are visited last to first, so removing a node never moves one
539/// that is still to be visited.
540fn cut(paragraph: &mut Element, range: Range<usize>, drop_marker: impl Fn(usize) -> bool) {
541    for segment in segments(paragraph).into_iter().rev() {
542        let end = segment.start + segment.kind.len();
543        let overlap_start = range.start.max(segment.start);
544        let overlap_end = range.end.min(end);
545        let overlaps = overlap_start < overlap_end;
546        let remove = match segment.kind {
547            Kind::Marker => drop_marker(segment.start),
548            Kind::Tab | Kind::LineBreak => overlaps,
549            Kind::Spaces(n) => {
550                if !overlaps {
551                    continue;
552                }
553                let left = n - (overlap_end - overlap_start);
554                if left > 0
555                    && let Some(e) = paragraph.at_mut(&segment.path)
556                {
557                    set_count(e, left);
558                }
559                left == 0
560            }
561            Kind::Text(_) => {
562                if !overlaps {
563                    continue;
564                }
565                let Some((parent, index)) = parent_of(paragraph, &segment.path) else {
566                    continue;
567                };
568                let Some(Node::Text(t) | Node::CData(t)) = parent.children.get_mut(index) else {
569                    continue;
570                };
571                let kept: String = t
572                    .chars()
573                    .enumerate()
574                    .filter(|(i, _)| {
575                        let at = segment.start + i;
576                        !(overlap_start..overlap_end).contains(&at)
577                    })
578                    .map(|(_, c)| c)
579                    .collect();
580                *t = kept;
581                t.is_empty()
582            }
583        };
584        if remove && let Some((parent, index)) = parent_of(paragraph, &segment.path) {
585            parent.children.remove(index);
586            prune_emptied(paragraph, &segment.path[..segment.path.len() - 1]);
587        }
588    }
589}
590
591/// A span or a link left with nothing in it says nothing, and goes; and so
592/// does the one it was in, if that is now empty too. One that was empty
593/// before the edit is not touched, because it is not this edit's: it never
594/// had a child removed.
595///
596/// Done as soon as the container empties, while the segments still to be
597/// visited are all before it and its own index is still right.
598fn prune_emptied(paragraph: &mut Element, container: &[usize]) {
599    let mut path = container.to_vec();
600    while !path.is_empty() {
601        let Some((parent, index)) = parent_of(paragraph, &path) else {
602            return;
603        };
604        match parent.children.get(index) {
605            Some(Node::Element(e)) if is_inline_container_name(e) && e.children.is_empty() => {
606                parent.children.remove(index);
607                path.pop();
608            }
609            _ => return,
610        }
611    }
612}
613
614/// Put text at a character offset.
615///
616/// Into the text node whose characters surround the offset; failing that, the
617/// one that ends there, then the one that begins there; failing that, as a new
618/// text node after whatever ends there, which puts new text after a bookmark
619/// at the same offset rather than before it.
620fn insert(paragraph: &mut Element, at: usize, with: &str) {
621    if with.is_empty() {
622        return;
623    }
624    let segments = segments(paragraph);
625    let text_at = |predicate: &dyn Fn(&Segment, usize) -> bool| {
626        segments
627            .iter()
628            .find(|s| matches!(s.kind, Kind::Text(_)) && predicate(s, s.start + s.kind.len()))
629    };
630    let target = text_at(&|s, end| s.start < at && at < end)
631        .or_else(|| text_at(&|_, end| end == at))
632        .or_else(|| text_at(&|s, _| s.start == at));
633    if let Some(segment) = target {
634        insert_into(paragraph, &segment.path, at - segment.start, with);
635        return;
636    }
637    // No text node touches the offset: a new one goes after the last segment
638    // ending at it, before the first beginning at it, or at the paragraph's
639    // end.
640    let after = segments
641        .iter()
642        .rev()
643        .find(|s| s.start + s.kind.len() == at)
644        .map(|s| s.path.clone());
645    let before = segments
646        .iter()
647        .find(|s| s.start >= at)
648        .map(|s| s.path.clone());
649    let node = Node::Text(with.to_owned());
650    if let Some(path) = after
651        && let Some((parent, index)) = parent_of(paragraph, &path)
652    {
653        parent.children.insert(index + 1, node);
654    } else if let Some(path) = before
655        && let Some((parent, index)) = parent_of(paragraph, &path)
656    {
657        parent.children.insert(index, node);
658    } else {
659        paragraph.children.push(node);
660        paragraph.self_closing = false;
661    }
662}
663
664/// Put text into the text node at a path, at a character offset within it.
665fn insert_into(paragraph: &mut Element, path: &[usize], offset: usize, with: &str) {
666    if let Some((parent, index)) = parent_of(paragraph, path)
667        && let Some(Node::Text(t) | Node::CData(t)) = parent.children.get_mut(index)
668    {
669        let byte = t.char_indices().nth(offset).map_or(t.len(), |(b, _)| b);
670        t.insert_str(byte, with);
671    }
672}
673
674/// The parent of the node a path leads to, and the node's index in it.
675fn parent_of<'a>(paragraph: &'a mut Element, path: &[usize]) -> Option<(&'a mut Element, usize)> {
676    let (last, above) = path.split_last()?;
677    Some((paragraph.at_mut(above)?, *last))
678}
679
680fn set_count(space: &mut Element, count: usize) {
681    if count <= 1 {
682        space.remove_attr(&Ns::Text, "c");
683    } else {
684        let name = space
685            .attrs
686            .iter()
687            .find(|a| a.name.is(&Ns::Text, "c"))
688            .map_or_else(
689                || {
690                    Name::new(
691                        space.name.prefix.as_deref().unwrap_or("text"),
692                        "c",
693                        Ns::Text,
694                    )
695                },
696                |a| a.name.clone(),
697            );
698        space.set_attr(name, count.to_string());
699    }
700}
701
702/// Write whitespace the way ODF requires.
703///
704/// A conforming reader collapses a run of spaces to one and drops the spaces
705/// at the start of a paragraph, and reads a literal tab or newline as a
706/// space. So a tab becomes `text:tab`, a newline `text:line-break`, a run of
707/// spaces one literal space and a `text:s` for the rest, and a run at the very
708/// start a `text:s` for all of them. Text nodes are split where an element has
709/// to go; nothing else in the tree is touched.
710fn normalize(paragraph: &mut Element) {
711    let prefix = paragraph
712        .name
713        .prefix
714        .as_deref()
715        .unwrap_or("text")
716        .to_owned();
717    let mut previous_was_space = true;
718    normalize_in(paragraph, &prefix, &mut previous_was_space);
719}
720
721fn normalize_in(parent: &mut Element, prefix: &str, previous_was_space: &mut bool) {
722    merge_text(parent);
723    let mut index = 0;
724    while index < parent.children.len() {
725        match &mut parent.children[index] {
726            Node::Text(t) | Node::CData(t) => {
727                let replacement = encode(t, prefix, previous_was_space);
728                match replacement {
729                    None => index += 1,
730                    Some(nodes) => {
731                        let count = nodes.len();
732                        parent.children.splice(index..=index, nodes);
733                        index += count;
734                    }
735                }
736            }
737            Node::Element(e) if is_inline_container(e) => {
738                normalize_in(e, prefix, previous_was_space);
739                index += 1;
740            }
741            Node::Element(e) => {
742                // Whitespace elements are whitespace; a marker is nothing,
743                // and what came before it still stands.
744                if e.is(&Ns::Text, "s") || e.is(&Ns::Text, "tab") || e.is(&Ns::Text, "line-break") {
745                    *previous_was_space = true;
746                }
747                index += 1;
748            }
749            Node::Comment(_) | Node::ProcessingInstruction(_) => index += 1,
750        }
751    }
752}
753
754/// Join text nodes that sit side by side into one.
755///
756/// The parser keeps an entity reference as a text node of its own, so a
757/// paragraph reads `a`, `&lt;`, `b` as three nodes; delete the middle one and
758/// the two left would serialize as one and parse back as one, which is a tree
759/// that is not equal to itself across a round trip. Joined here, it is.
760fn merge_text(parent: &mut Element) {
761    let mut index = 1;
762    while index < parent.children.len() {
763        let joinable = matches!(
764            (&parent.children[index - 1], &parent.children[index]),
765            (Node::Text(_), Node::Text(_))
766        );
767        if joinable {
768            let Node::Text(tail) = parent.children.remove(index) else {
769                unreachable!("matched a text node");
770            };
771            if let Node::Text(head) = &mut parent.children[index - 1] {
772                head.push_str(&tail);
773            }
774        } else {
775            index += 1;
776        }
777    }
778}
779
780/// The nodes a text node becomes, or `None` where it is already as ODF
781/// writes it.
782fn encode(text: &str, prefix: &str, previous_was_space: &mut bool) -> Option<Vec<Node>> {
783    let needs_work = text.contains(['\t', '\n', '\r'])
784        || text.contains("  ")
785        || (*previous_was_space && text.starts_with(' '));
786    if !needs_work {
787        if let Some(last) = text.chars().last() {
788            *previous_was_space = last == ' ';
789        }
790        return None;
791    }
792    let mut nodes = Vec::new();
793    let mut run = String::new();
794    let mut spaces = 0usize;
795    let flush_spaces = |nodes: &mut Vec<Node>,
796                        run: &mut String,
797                        spaces: &mut usize,
798                        previous_was_space: &mut bool| {
799        if *spaces == 0 {
800            return;
801        }
802        // One literal space where a space may be literal, and `text:s` for
803        // whatever a reader would otherwise collapse.
804        let mut counted = *spaces;
805        if !*previous_was_space {
806            run.push(' ');
807            counted -= 1;
808        }
809        if counted > 0 {
810            if !run.is_empty() {
811                nodes.push(Node::Text(std::mem::take(run)));
812            }
813            let mut s = Element::new(prefix, "s", Ns::Text);
814            if counted > 1 {
815                s.attrs.push(Attribute {
816                    name: Name::new(prefix, "c", Ns::Text),
817                    value: counted.to_string(),
818                });
819            }
820            nodes.push(Node::Element(s));
821        }
822        *spaces = 0;
823        *previous_was_space = true;
824    };
825    for c in text.chars() {
826        match c {
827            ' ' => spaces += 1,
828            // A carriage return is not a character ODF has a spelling for; a
829            // newline follows it wherever it was typed.
830            '\r' => {}
831            '\t' | '\n' => {
832                flush_spaces(&mut nodes, &mut run, &mut spaces, previous_was_space);
833                if !run.is_empty() {
834                    nodes.push(Node::Text(std::mem::take(&mut run)));
835                }
836                let local = if c == '\t' { "tab" } else { "line-break" };
837                nodes.push(Node::Element(Element::new(prefix, local, Ns::Text)));
838                *previous_was_space = true;
839            }
840            other => {
841                flush_spaces(&mut nodes, &mut run, &mut spaces, previous_was_space);
842                run.push(other);
843                *previous_was_space = false;
844            }
845        }
846    }
847    flush_spaces(&mut nodes, &mut run, &mut spaces, previous_was_space);
848    if !run.is_empty() {
849        nodes.push(Node::Text(run));
850    }
851    Some(nodes)
852}