1use std::collections::{HashMap, HashSet};
2
3use quick_xml::{
4 XmlVersion,
5 escape::resolve_predefined_entity,
6 events::{BytesStart, Event},
7 name::{Namespace, ResolveResult},
8 reader::NsReader,
9};
10
11use super::{DocumentProjection, TextMaterialization};
12use crate::projection::ooxml::TextControl;
13
14const WORDPROCESSING_TRANSITIONAL: &[u8] =
15 b"http://schemas.openxmlformats.org/wordprocessingml/2006/main";
16const WORDPROCESSING_STRICT: &[u8] = b"http://purl.oclc.org/ooxml/wordprocessingml/main";
17const WORDPROCESSING_2010: &[u8] = b"http://schemas.microsoft.com/office/word/2010/wordml";
18const WORDPROCESSING_2012: &[u8] = b"http://schemas.microsoft.com/office/word/2012/wordml";
19
20#[derive(Clone, Copy, Debug, Eq, PartialEq)]
21pub struct ReviewFactLimits {
22 pub maximum_comments_xml_bytes: usize,
23 pub maximum_comments_extended_xml_bytes: usize,
24 pub maximum_facts_per_family: usize,
25 pub maximum_review_detail_bytes: usize,
26}
27
28impl Default for ReviewFactLimits {
29 fn default() -> Self {
30 Self {
31 maximum_comments_xml_bytes: 16 * 1024 * 1024,
32 maximum_comments_extended_xml_bytes: 16 * 1024 * 1024,
33 maximum_facts_per_family: 1_000_000,
34 maximum_review_detail_bytes: 32 * 1024 * 1024,
35 }
36 }
37}
38
39#[derive(Clone, Copy, Debug, Eq, PartialEq)]
40pub enum ReviewFactUnknownReason {
41 InvalidDocument,
42 InvalidComments,
43 InvalidCommentsExtended,
44 ResourceLimit,
45 UnsupportedLocation,
46}
47
48#[derive(Clone, Debug, Eq, PartialEq)]
49pub enum ReviewFactSet<T> {
50 Known(Vec<T>),
51 Unknown(ReviewFactUnknownReason),
52}
53
54#[derive(Clone, Debug, Eq, PartialEq)]
55pub enum ReviewDetail<T> {
56 Known(T),
57 Unknown(ReviewFactUnknownReason),
58}
59
60#[derive(Clone, Copy, Debug, Eq, PartialEq)]
61pub struct ReviewPoint {
62 pub paragraph_ordinal: usize,
63 pub utf8: u32,
64 pub utf16: u32,
65}
66
67#[derive(Clone, Copy, Debug, Eq, PartialEq)]
68pub struct ReviewSpan {
69 pub start: ReviewPoint,
70 pub end: ReviewPoint,
71}
72
73#[derive(Clone, Debug, Eq, PartialEq)]
74pub struct RevisionContent {
75 pub span: ReviewSpan,
76 pub payload: RevisionPayload,
77}
78
79#[derive(Clone, Debug, Eq, PartialEq)]
81pub enum RevisionPayload {
82 Text(String),
85 FormattingOnly,
87 ParagraphMark,
91}
92
93impl RevisionPayload {
94 pub(super) fn from_text(text: String) -> Self {
95 if text.is_empty() {
96 Self::FormattingOnly
97 } else {
98 Self::Text(text)
99 }
100 }
101
102 #[must_use]
103 pub fn text(&self) -> &str {
104 match self {
105 Self::Text(text) => text,
106 Self::FormattingOnly | Self::ParagraphMark => "",
107 }
108 }
109}
110
111#[derive(Clone, Debug, Eq, PartialEq)]
112pub struct CommentContent {
113 pub anchor: ReviewSpan,
114 pub comment_text: String,
115 pub referenced_text: String,
116}
117
118#[derive(Clone, Copy, Debug, Eq, PartialEq)]
119pub enum RevisionFactKind {
120 Insertion,
121 Deletion,
122 MoveFrom,
123 MoveTo,
124 CellInsertion,
125 CellDeletion,
126 CellMerge,
127 ParagraphPropertiesChange,
128 RunPropertiesChange,
129 SectionPropertiesChange,
130 TablePropertiesChange,
131 TablePropertiesExceptionChange,
132 TableRowPropertiesChange,
133 TableCellPropertiesChange,
134 TableGridChange,
135 CustomXmlDeletionRangeStart,
136 CustomXmlDeletionRangeEnd,
137 CustomXmlInsertionRangeStart,
138 CustomXmlInsertionRangeEnd,
139 CustomXmlMoveFromRangeStart,
140 CustomXmlMoveFromRangeEnd,
141 CustomXmlMoveToRangeStart,
142 CustomXmlMoveToRangeEnd,
143}
144
145#[derive(Clone, Debug, Eq, PartialEq)]
146pub struct AttributedRevision {
147 pub kind: RevisionFactKind,
148 pub author: String,
149 pub date: Option<String>,
150 pub revision_id: Option<String>,
151 pub content: ReviewDetail<RevisionContent>,
152}
153
154#[derive(Clone, Debug, Eq, PartialEq)]
155pub struct AttributedComment {
156 pub comment_id: String,
157 pub author: String,
158 pub initials: Option<String>,
159 pub date: Option<String>,
160 pub parent_comment_id: Option<String>,
161 pub resolved: bool,
162 pub content: ReviewDetail<CommentContent>,
163}
164
165#[derive(Clone, Debug, Eq, PartialEq)]
166pub struct DocumentReviewFacts {
167 pub revisions: ReviewFactSet<AttributedRevision>,
168 pub comments: ReviewFactSet<AttributedComment>,
169}
170
171#[derive(Clone)]
172struct CommentRow {
173 comment_id: String,
174 author: String,
175 initials: Option<String>,
176 date: Option<String>,
177 paragraph_id: Option<String>,
178 paragraph_id_from_wrapper: bool,
179 content: String,
180 paragraph_seen: bool,
181}
182
183#[derive(Clone)]
184struct CommentExtension {
185 parent_paragraph_id: Option<String>,
186 resolved: bool,
187}
188
189#[derive(Clone, Copy, Eq, PartialEq)]
190enum NamespaceKind {
191 Other,
192 Wordprocessing,
193 Wordprocessing2010,
194 Wordprocessing2012,
195}
196
197pub(super) fn project_review_facts(
198 revisions: ReviewFactSet<AttributedRevision>,
199 comment_anchors: Option<&HashMap<String, ReviewSpan>>,
200 document: &DocumentProjection,
201 comments_xml: Result<Option<Vec<u8>>, ReviewFactUnknownReason>,
202 comments_extended_xml: Result<Option<Vec<u8>>, ReviewFactUnknownReason>,
203 limits: ReviewFactLimits,
204 text_materialization: TextMaterialization,
205) -> DocumentReviewFacts {
206 let (revisions, remaining_detail_bytes) =
207 bound_revision_details(revisions, limits.maximum_review_detail_bytes);
208 let comments = match comments_xml {
209 Ok(None) => ReviewFactSet::Known(Vec::new()),
210 Ok(Some(comments_xml)) => match comments_extended_xml {
211 Ok(comments_extended_xml) => parse_comments(
212 &comments_xml,
213 comments_extended_xml.as_deref(),
214 comment_anchors,
215 document,
216 limits.maximum_facts_per_family,
217 remaining_detail_bytes,
218 text_materialization,
219 )
220 .map_or_else(ReviewFactSet::Unknown, ReviewFactSet::Known),
221 Err(reason) => ReviewFactSet::Unknown(reason),
222 },
223 Err(reason) => ReviewFactSet::Unknown(reason),
224 };
225 DocumentReviewFacts {
226 revisions,
227 comments,
228 }
229}
230
231fn bound_revision_details(
232 revisions: ReviewFactSet<AttributedRevision>,
233 maximum_detail_bytes: usize,
234) -> (ReviewFactSet<AttributedRevision>, usize) {
235 let ReviewFactSet::Known(facts) = &revisions else {
236 return (revisions, maximum_detail_bytes);
237 };
238 let detail_bytes = facts.iter().try_fold(0_usize, |total, fact| {
239 let bytes = match &fact.content {
240 ReviewDetail::Known(content) => content.payload.text().len(),
241 ReviewDetail::Unknown(_) => 0,
242 };
243 total.checked_add(bytes)
244 });
245 let Some(remaining) = detail_bytes.and_then(|bytes| maximum_detail_bytes.checked_sub(bytes))
246 else {
247 return (
248 ReviewFactSet::Unknown(ReviewFactUnknownReason::ResourceLimit),
249 maximum_detail_bytes,
250 );
251 };
252 (revisions, remaining)
253}
254
255fn namespace_kind(namespace: &ResolveResult<'_>) -> NamespaceKind {
256 let ResolveResult::Bound(Namespace(value)) = namespace else {
257 return NamespaceKind::Other;
258 };
259 match *value {
260 WORDPROCESSING_TRANSITIONAL | WORDPROCESSING_STRICT => NamespaceKind::Wordprocessing,
261 WORDPROCESSING_2010 => NamespaceKind::Wordprocessing2010,
262 WORDPROCESSING_2012 => NamespaceKind::Wordprocessing2012,
263 _ => NamespaceKind::Other,
264 }
265}
266
267fn attribute(
268 reader: &NsReader<&[u8]>,
269 element: &BytesStart<'_>,
270 namespace: NamespaceKind,
271 local_name: &[u8],
272) -> Result<Option<String>, ReviewFactUnknownReason> {
273 for item in element.attributes() {
274 let item = item.map_err(|_| ReviewFactUnknownReason::InvalidDocument)?;
275 let (resolved, name) = reader.resolver().resolve_attribute(item.key);
276 if namespace_kind(&resolved) == namespace && name.as_ref() == local_name {
277 return item
278 .decoded_and_normalized_value(XmlVersion::default(), reader.decoder())
279 .map(|value| Some(value.into_owned()))
280 .map_err(|_| ReviewFactUnknownReason::InvalidDocument);
281 }
282 }
283 Ok(None)
284}
285
286fn comment_attribute(
287 reader: &NsReader<&[u8]>,
288 element: &BytesStart<'_>,
289 namespace: NamespaceKind,
290 local_name: &[u8],
291) -> Result<Option<String>, ReviewFactUnknownReason> {
292 attribute(reader, element, namespace, local_name)
293 .map_err(|_| ReviewFactUnknownReason::InvalidComments)
294}
295
296fn comment_extension_attribute(
297 reader: &NsReader<&[u8]>,
298 element: &BytesStart<'_>,
299 local_name: &[u8],
300) -> Result<Option<String>, ReviewFactUnknownReason> {
301 attribute(
302 reader,
303 element,
304 NamespaceKind::Wordprocessing2012,
305 local_name,
306 )
307 .map_err(|_| ReviewFactUnknownReason::InvalidCommentsExtended)
308}
309
310fn comment_paragraph_id(
311 reader: &NsReader<&[u8]>,
312 element: &BytesStart<'_>,
313) -> Result<Option<String>, ReviewFactUnknownReason> {
314 for namespace in [
315 NamespaceKind::Wordprocessing2010,
316 NamespaceKind::Wordprocessing2012,
317 NamespaceKind::Wordprocessing,
318 ] {
319 if let Some(value) = comment_attribute(reader, element, namespace, b"paraId")? {
320 return Ok(Some(value.to_ascii_uppercase()));
321 }
322 }
323 Ok(None)
324}
325
326fn parse_comments(
327 comments_xml: &[u8],
328 comments_extended_xml: Option<&[u8]>,
329 comment_anchors: Option<&HashMap<String, ReviewSpan>>,
330 document: &DocumentProjection,
331 maximum_facts: usize,
332 maximum_detail_bytes: usize,
333 text_materialization: TextMaterialization,
334) -> Result<Vec<AttributedComment>, ReviewFactUnknownReason> {
335 let comments = parse_comment_rows(comments_xml, maximum_facts, text_materialization)?;
336 let mut detail_budget = ReviewDetailBudget::new(maximum_detail_bytes);
337 let Some(comments_extended_xml) = comments_extended_xml else {
338 let mut output = Vec::with_capacity(comments.len());
339 for comment in comments {
340 let content = comment_content(&comment, comment_anchors, document, &mut detail_budget)?;
341 output.push(attributed_comment(comment, None, false, content));
342 }
343 return Ok(output);
344 };
345 let mut extensions = parse_comment_extensions(comments_extended_xml, maximum_facts)?;
346 let comment_ids_by_paragraph = comments
347 .iter()
348 .filter_map(|comment| {
349 comment
350 .paragraph_id
351 .as_ref()
352 .map(|paragraph_id| (paragraph_id.clone(), comment.comment_id.clone()))
353 })
354 .collect::<HashMap<_, _>>();
355 let comments_with_paragraph_ids = comments
356 .iter()
357 .filter(|comment| comment.paragraph_id.is_some())
358 .count();
359 if comment_ids_by_paragraph.len() != comments_with_paragraph_ids {
360 return Err(ReviewFactUnknownReason::InvalidCommentsExtended);
361 }
362 let mut output = Vec::with_capacity(comments.len());
363 for comment in comments {
364 let content = comment_content(&comment, comment_anchors, document, &mut detail_budget)?;
365 let extension = comment
366 .paragraph_id
367 .as_ref()
368 .and_then(|paragraph_id| extensions.remove(paragraph_id));
369 let (parent_comment_id, resolved) = extension.map_or(Ok((None, false)), |extension| {
370 let parent = extension
371 .parent_paragraph_id
372 .as_ref()
373 .map(|parent| {
374 comment_ids_by_paragraph
375 .get(parent)
376 .cloned()
377 .ok_or(ReviewFactUnknownReason::InvalidCommentsExtended)
378 })
379 .transpose()?;
380 Ok((parent, extension.resolved))
381 })?;
382 output.push(attributed_comment(
383 comment,
384 parent_comment_id,
385 resolved,
386 content,
387 ));
388 }
389 if !extensions.is_empty() || contains_parent_cycle(&output) {
390 return Err(ReviewFactUnknownReason::InvalidCommentsExtended);
391 }
392 Ok(output)
393}
394
395fn attributed_comment(
396 comment: CommentRow,
397 parent_comment_id: Option<String>,
398 resolved: bool,
399 content: ReviewDetail<CommentContent>,
400) -> AttributedComment {
401 AttributedComment {
402 comment_id: comment.comment_id,
403 author: comment.author,
404 initials: comment.initials,
405 date: comment.date,
406 parent_comment_id,
407 resolved,
408 content,
409 }
410}
411
412fn comment_content(
413 comment: &CommentRow,
414 anchors: Option<&HashMap<String, ReviewSpan>>,
415 document: &DocumentProjection,
416 detail_budget: &mut ReviewDetailBudget,
417) -> Result<ReviewDetail<CommentContent>, ReviewFactUnknownReason> {
418 let Some(anchor) = anchors
419 .and_then(|anchors| anchors.get(&comment.comment_id))
420 .copied()
421 else {
422 return Ok(ReviewDetail::Unknown(
423 ReviewFactUnknownReason::UnsupportedLocation,
424 ));
425 };
426 let Some(referenced_text_bytes) = text_for_span_bytes(document, anchor) else {
427 return Ok(ReviewDetail::Unknown(
428 ReviewFactUnknownReason::UnsupportedLocation,
429 ));
430 };
431 let detail_bytes = comment
432 .content
433 .len()
434 .checked_add(referenced_text_bytes)
435 .ok_or(ReviewFactUnknownReason::ResourceLimit)?;
436 detail_budget.reserve(detail_bytes)?;
437 let Some(referenced_text) = text_for_span(document, anchor, referenced_text_bytes) else {
438 return Ok(ReviewDetail::Unknown(
439 ReviewFactUnknownReason::UnsupportedLocation,
440 ));
441 };
442 Ok(ReviewDetail::Known(CommentContent {
443 anchor,
444 comment_text: comment.content.clone(),
445 referenced_text,
446 }))
447}
448
449struct ReviewDetailBudget {
450 remaining: usize,
451}
452
453impl ReviewDetailBudget {
454 const fn new(maximum: usize) -> Self {
455 Self { remaining: maximum }
456 }
457
458 fn reserve(&mut self, bytes: usize) -> Result<(), ReviewFactUnknownReason> {
459 self.remaining = self
460 .remaining
461 .checked_sub(bytes)
462 .ok_or(ReviewFactUnknownReason::ResourceLimit)?;
463 Ok(())
464 }
465}
466
467fn text_for_span_bytes(document: &DocumentProjection, span: ReviewSpan) -> Option<usize> {
468 let start = document.paragraphs.get(span.start.paragraph_ordinal)?;
469 let end = document.paragraphs.get(span.end.paragraph_ordinal)?;
470 let start_utf8 = usize::try_from(span.start.utf8).ok()?;
471 let end_utf8 = usize::try_from(span.end.utf8).ok()?;
472 if span.start.paragraph_ordinal == span.end.paragraph_ordinal {
473 return Some(start.text.get(start_utf8..end_utf8)?.len());
474 }
475 let mut bytes = start.text.get(start_utf8..)?.len();
476 for paragraph in document
477 .paragraphs
478 .get(span.start.paragraph_ordinal.checked_add(1)?..span.end.paragraph_ordinal)?
479 {
480 bytes = bytes.checked_add(1)?.checked_add(paragraph.text.len())?;
481 }
482 bytes
483 .checked_add(1)?
484 .checked_add(end.text.get(..end_utf8)?.len())
485}
486
487fn text_for_span(
488 document: &DocumentProjection,
489 span: ReviewSpan,
490 byte_length: usize,
491) -> Option<String> {
492 let start = document.paragraphs.get(span.start.paragraph_ordinal)?;
493 let end = document.paragraphs.get(span.end.paragraph_ordinal)?;
494 let start_utf8 = usize::try_from(span.start.utf8).ok()?;
495 let end_utf8 = usize::try_from(span.end.utf8).ok()?;
496 let mut text = String::with_capacity(byte_length);
497 if span.start.paragraph_ordinal == span.end.paragraph_ordinal {
498 text.push_str(start.text.get(start_utf8..end_utf8)?);
499 return Some(text);
500 }
501 text.push_str(start.text.get(start_utf8..)?);
502 for paragraph in document
503 .paragraphs
504 .get(span.start.paragraph_ordinal.checked_add(1)?..span.end.paragraph_ordinal)?
505 {
506 text.push('\n');
507 text.push_str(¶graph.text);
508 }
509 text.push('\n');
510 text.push_str(end.text.get(..end_utf8)?);
511 Some(text)
512}
513
514#[allow(clippy::too_many_lines)] fn parse_comment_rows(
516 xml: &[u8],
517 maximum_facts: usize,
518 text_materialization: TextMaterialization,
519) -> Result<Vec<CommentRow>, ReviewFactUnknownReason> {
520 let invalid = ReviewFactUnknownReason::InvalidComments;
521 let mut reader = NsReader::from_reader(xml);
522 reader.config_mut().expand_empty_elements = true;
523 reader.config_mut().check_end_names = true;
524 let mut output = Vec::new();
525 let mut current: Option<(usize, CommentRow)> = None;
526 let mut text_depth: Option<usize> = None;
527 let mut property_depth: Option<usize> = None;
528 let mut depth = 0_usize;
529 let mut root_seen = false;
530 let mut comment_ids = HashSet::new();
531 let mut paragraph_ids = HashSet::new();
532 loop {
533 match reader.read_resolved_event() {
534 Ok((_, Event::Eof)) => {
535 if depth != 0
536 || current.is_some()
537 || text_depth.is_some()
538 || property_depth.is_some()
539 || !root_seen
540 {
541 return Err(invalid);
542 }
543 return Ok(output);
544 }
545 Ok((namespace, Event::Start(element))) => {
546 let kind = namespace_kind(&namespace);
547 if depth == 0 {
548 if root_seen
549 || kind != NamespaceKind::Wordprocessing
550 || element.local_name().as_ref() != b"comments"
551 {
552 return Err(invalid);
553 }
554 root_seen = true;
555 } else if depth == 1
556 && kind == NamespaceKind::Wordprocessing
557 && element.local_name().as_ref() == b"comment"
558 {
559 if current.is_some() || output.len() >= maximum_facts {
560 return Err(if output.len() >= maximum_facts {
561 ReviewFactUnknownReason::ResourceLimit
562 } else {
563 invalid
564 });
565 }
566 let comment_id =
567 comment_attribute(&reader, &element, NamespaceKind::Wordprocessing, b"id")?
568 .ok_or(invalid)?;
569 if !comment_ids.insert(comment_id.clone()) {
570 return Err(invalid);
571 }
572 let wrapper_paragraph_id = comment_paragraph_id(&reader, &element)?;
573 let paragraph_id_from_wrapper = wrapper_paragraph_id.is_some();
574 current = Some((
575 depth,
576 CommentRow {
577 comment_id,
578 author: comment_attribute(
579 &reader,
580 &element,
581 NamespaceKind::Wordprocessing,
582 b"author",
583 )?
584 .unwrap_or_default(),
585 initials: comment_attribute(
586 &reader,
587 &element,
588 NamespaceKind::Wordprocessing,
589 b"initials",
590 )?,
591 date: comment_attribute(
592 &reader,
593 &element,
594 NamespaceKind::Wordprocessing,
595 b"date",
596 )?,
597 paragraph_id: wrapper_paragraph_id,
598 paragraph_id_from_wrapper,
599 content: String::new(),
600 paragraph_seen: false,
601 },
602 ));
603 } else if kind == NamespaceKind::Wordprocessing
604 && element.local_name().as_ref() == b"p"
605 && let Some((comment_depth, comment)) = current.as_mut()
606 && depth == comment_depth.saturating_add(1)
607 {
608 if !comment.paragraph_id_from_wrapper {
611 comment.paragraph_id = comment_paragraph_id(&reader, &element)?;
612 }
613 if comment.paragraph_seen {
614 comment.content.push('\n');
615 }
616 comment.paragraph_seen = true;
617 } else if kind == NamespaceKind::Wordprocessing
618 && matches!(element.local_name().as_ref(), b"pPr" | b"rPr")
619 && current.is_some()
620 && property_depth.is_none()
621 {
622 property_depth = Some(depth);
623 } else if kind == NamespaceKind::Wordprocessing
624 && matches!(element.local_name().as_ref(), b"t" | b"delText")
625 && current.is_some()
626 && property_depth.is_none()
627 {
628 if text_depth.replace(depth).is_some() {
629 return Err(invalid);
630 }
631 } else if kind == NamespaceKind::Wordprocessing
632 && let Some((_, comment)) = current.as_mut()
633 && property_depth.is_none()
634 {
635 let control = match element.local_name().as_ref() {
636 b"tab" | b"ptab" => Some(TextControl::Tab),
637 b"br" => Some(
638 match comment_attribute(
639 &reader,
640 &element,
641 NamespaceKind::Wordprocessing,
642 b"type",
643 )?
644 .as_deref()
645 {
646 Some("page") => TextControl::PageBreak,
647 Some("column") => TextControl::ColumnBreak,
648 _ => TextControl::LineBreak,
649 },
650 ),
651 b"cr" => Some(TextControl::CarriageReturn),
652 b"softHyphen" => Some(TextControl::SoftHyphen),
653 b"noBreakHyphen" => Some(TextControl::NoBreakHyphen),
654 _ => None,
655 };
656 if let Some(text) =
657 control.and_then(|control| control.materialize(text_materialization))
658 {
659 comment.content.push_str(text);
660 }
661 }
662 depth = depth.checked_add(1).ok_or(invalid)?;
663 }
664 Ok((_, Event::DocType(_))) | Err(_) => return Err(invalid),
665 Ok((_, Event::Text(text))) if text_depth.is_some() => {
666 let decoded = text.xml10_content().map_err(|_| invalid)?;
667 current
668 .as_mut()
669 .ok_or(invalid)?
670 .1
671 .content
672 .push_str(&decoded);
673 }
674 Ok((_, Event::CData(text))) if text_depth.is_some() => {
675 let decoded = text.xml10_content().map_err(|_| invalid)?;
676 current
677 .as_mut()
678 .ok_or(invalid)?
679 .1
680 .content
681 .push_str(&decoded);
682 }
683 Ok((_, Event::GeneralRef(reference))) if text_depth.is_some() => {
684 let resolved_character = reference.resolve_char_ref().map_err(|_| invalid)?;
685 let resolved = if let Some(character) = resolved_character {
686 character.to_string()
687 } else {
688 let name = reference.decode().map_err(|_| invalid)?;
689 resolve_predefined_entity(&name).ok_or(invalid)?.to_owned()
690 };
691 current
692 .as_mut()
693 .ok_or(invalid)?
694 .1
695 .content
696 .push_str(&resolved);
697 }
698 Ok((namespace, Event::End(element))) => {
699 depth = depth.checked_sub(1).ok_or(invalid)?;
700 if text_depth == Some(depth) {
701 text_depth = None;
702 }
703 if property_depth == Some(depth) {
704 property_depth = None;
705 }
706 if namespace_kind(&namespace) == NamespaceKind::Wordprocessing
707 && element.local_name().as_ref() == b"comment"
708 {
709 let (comment_depth, comment) = current.take().ok_or(invalid)?;
710 if depth != comment_depth || text_depth.is_some() || property_depth.is_some() {
711 return Err(invalid);
712 }
713 if let Some(paragraph_id) = &comment.paragraph_id
714 && !paragraph_ids.insert(paragraph_id.clone())
715 {
716 return Err(invalid);
717 }
718 output.push(comment);
719 }
720 }
721 Ok(_) => {}
722 }
723 }
724}
725
726fn parse_comment_extensions(
727 xml: &[u8],
728 maximum_facts: usize,
729) -> Result<HashMap<String, CommentExtension>, ReviewFactUnknownReason> {
730 let invalid = ReviewFactUnknownReason::InvalidCommentsExtended;
731 let mut reader = NsReader::from_reader(xml);
732 reader.config_mut().expand_empty_elements = true;
733 reader.config_mut().check_end_names = true;
734 let mut output = HashMap::new();
735 let mut depth = 0_usize;
736 let mut root_seen = false;
737 loop {
738 match reader.read_resolved_event() {
739 Ok((_, Event::Eof)) => {
740 if depth != 0 || !root_seen {
741 return Err(invalid);
742 }
743 return Ok(output);
744 }
745 Ok((namespace, Event::Start(element))) => {
746 let kind = namespace_kind(&namespace);
747 if depth == 0 {
748 if root_seen
749 || kind != NamespaceKind::Wordprocessing2012
750 || element.local_name().as_ref() != b"commentsEx"
751 {
752 return Err(invalid);
753 }
754 root_seen = true;
755 } else if depth == 1
756 && kind == NamespaceKind::Wordprocessing2012
757 && element.local_name().as_ref() == b"commentEx"
758 {
759 if output.len() >= maximum_facts {
760 return Err(ReviewFactUnknownReason::ResourceLimit);
761 }
762 let paragraph_id = comment_extension_attribute(&reader, &element, b"paraId")?
763 .ok_or(invalid)?
764 .to_ascii_uppercase();
765 let parent_paragraph_id =
766 comment_extension_attribute(&reader, &element, b"paraIdParent")?
767 .map(|value| value.to_ascii_uppercase());
768 let resolved = comment_extension_attribute(&reader, &element, b"done")?
769 .map(|value| parse_on_off(&value))
770 .transpose()?
771 .unwrap_or(false);
772 if output
773 .insert(
774 paragraph_id,
775 CommentExtension {
776 parent_paragraph_id,
777 resolved,
778 },
779 )
780 .is_some()
781 {
782 return Err(invalid);
783 }
784 }
785 depth = depth.checked_add(1).ok_or(invalid)?;
786 }
787 Ok((_, Event::End(_))) => depth = depth.checked_sub(1).ok_or(invalid)?,
788 Ok((_, Event::DocType(_))) | Err(_) => return Err(invalid),
789 Ok(_) => {}
790 }
791 }
792}
793
794fn parse_on_off(value: &str) -> Result<bool, ReviewFactUnknownReason> {
795 match value.to_ascii_lowercase().as_str() {
796 "1" | "true" | "on" => Ok(true),
797 "0" | "false" | "off" => Ok(false),
798 _ => Err(ReviewFactUnknownReason::InvalidCommentsExtended),
799 }
800}
801
802fn contains_parent_cycle(comments: &[AttributedComment]) -> bool {
803 let parents = comments
804 .iter()
805 .map(|comment| {
806 (
807 comment.comment_id.as_str(),
808 comment.parent_comment_id.as_deref(),
809 )
810 })
811 .collect::<HashMap<_, _>>();
812 let mut complete = HashSet::new();
813 for comment_id in parents.keys() {
814 let mut path = Vec::new();
815 let mut visiting = HashSet::new();
816 let mut current = Some(*comment_id);
817 while let Some(id) = current {
818 if complete.contains(id) {
819 break;
820 }
821 if !visiting.insert(id) {
822 return true;
823 }
824 path.push(id);
825 current = parents.get(id).copied().flatten();
826 }
827 complete.extend(path);
828 }
829 false
830}