#![allow(
clippy::arithmetic_side_effects,
clippy::expect_used,
clippy::indexing_slicing,
clippy::panic,
clippy::unwrap_used
)]
use crate::analyzer::readability::flesch_kincaid_grade_level;
use crate::io::document::to_markdown;
use crate::test::utils::fixture_path;
use std::io::Write;
use std::path::PathBuf;
use tempfile::{Builder, NamedTempFile};
use zip::write::SimpleFileOptions;
use zip::{CompressionMethod, ZipWriter};
fn write_docx(document_xml: &str) -> NamedTempFile {
let file = Builder::new().suffix(".docx").tempfile().unwrap();
let mut archive = ZipWriter::new(file);
let options = SimpleFileOptions::default().compression_method(CompressionMethod::Stored);
let content_types = r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types">
<Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/>
<Default Extension="xml" ContentType="application/xml"/>
<Override PartName="/word/document.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml"/>
</Types>"#;
let root_rels = r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">
<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="word/document.xml"/>
</Relationships>"#;
let document_rels = r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">
<Relationship Id="rIdHyper" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/hyperlink" Target="https://example.org" TargetMode="External"/>
</Relationships>"#;
archive.start_file("[Content_Types].xml", options).unwrap();
archive.write_all(content_types.as_bytes()).unwrap();
archive.start_file("_rels/.rels", options).unwrap();
archive.write_all(root_rels.as_bytes()).unwrap();
archive.start_file("word/document.xml", options).unwrap();
archive.write_all(document_xml.as_bytes()).unwrap();
archive.start_file("word/_rels/document.xml.rels", options).unwrap();
archive.write_all(document_rels.as_bytes()).unwrap();
archive.finish().unwrap()
}
#[test]
fn test_extract_sample_fixture() {
let result = to_markdown(fixture_path("sample.docx"));
assert!(result.is_err(), "Sample fixture should currently fail parsing");
}
#[test]
fn test_extract_acorn_fixture() {
let result = to_markdown(fixture_path("acorn.docx"));
assert!(result.is_ok(), "Acorn fixture should parse");
let text = result.unwrap();
assert!(text.contains("ACORN"), "Acorn fixture text should contain ACORN");
assert!(
text.contains("Research Activity Data (RAD)"),
"Acorn fixture text should contain Research Activity Data (RAD)"
);
}
#[test]
fn test_extract_acorn_fixture_readability() {
let text = to_markdown(fixture_path("acorn.docx")).unwrap();
let grade = flesch_kincaid_grade_level(&text);
assert!((grade - 17.12).abs() < 0.2, "Expected readability near 17.12, got {grade}");
}
#[test]
fn test_extract_nonexistent_file() {
let nonexistent_path = PathBuf::from("/nonexistent/path/to/file.docx");
let result = to_markdown(&nonexistent_path);
assert!(result.is_err(), "Should fail for nonexistent file");
}
#[test]
fn test_extract_invalid_docx() {
let mut temporary = Builder::new().suffix(".docx").tempfile().unwrap();
temporary.write_all(b"Not a valid DOCX file").unwrap();
let result = to_markdown(temporary.path());
assert!(result.is_err(), "Should fail for invalid DOCX format");
}
#[test]
fn test_extract_empty_zip() {
let mut temporary = Builder::new().suffix(".docx").tempfile().unwrap();
temporary.write_all(&[0x50, 0x4b, 0x03, 0x04]).unwrap();
let result = to_markdown(temporary.path());
let _ = result;
}
#[test]
fn test_extract_function_signature() {
let nonexistent = "/fake/path.docx";
let result = to_markdown(nonexistent);
assert!(result.is_err(), "Should accept &str and return error for missing file");
}
#[test]
fn test_extract_generated_docx_paragraph_children() {
let document_xml = r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"
xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships">
<w:body>
<w:p>
<w:r><w:t>Alpha</w:t></w:r>
<w:r><w:tab/></w:r>
<w:r><w:t>Beta</w:t></w:r>
<w:r><w:br/></w:r>
<w:r><w:instrText>Instr</w:instrText></w:r>
<w:hyperlink r:id="rIdHyper"><w:r><w:t>Link</w:t></w:r></w:hyperlink>
<w:ins w:id="1"><w:r><w:t>InsertText</w:t></w:r></w:ins>
<w:del w:id="2"><w:r><w:t>DeleteText</w:t></w:r></w:del>
<w:moveFrom w:id="3" w:author="Author" w:date="2026-08-01T00:00:00Z"><w:r><w:t>GhostText</w:t></w:r></w:moveFrom>
<w:moveTo w:id="3" w:author="Author" w:date="2026-08-01T00:00:00Z"><w:r><w:t>MovedText</w:t></w:r></w:moveTo>
<w:r><w:cr/></w:r>
<w:r><w:t>AfterCarriageReturn</w:t></w:r>
</w:p>
<w:p>
<w:r><w:t>Tail</w:t></w:r>
</w:p>
</w:body>
</w:document>"#;
let document = write_docx(document_xml);
let result = to_markdown(document.path());
assert!(result.is_ok(), "Generated DOCX should parse");
let text = result.unwrap();
["Alpha", "Beta", "Link", "InsertText", "MovedText", "AfterCarriageReturn"]
.into_iter()
.for_each(|expected| assert!(text.contains(expected), "Expected extracted text to contain {expected}: {text:?}"));
["Instr", "DeleteText", "GhostText"]
.into_iter()
.for_each(|excluded| assert!(!text.contains(excluded), "Expected extracted text to omit {excluded}: {text:?}"));
assert!(text.contains("Tail"));
}
#[test]
fn test_extract_generated_docx_table_children() {
let document_xml = r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body>
<w:tbl>
<w:tr>
<w:tc>
<w:p><w:r><w:t>CellOne</w:t></w:r></w:p>
<w:tbl>
<w:tr>
<w:tc>
<w:p><w:r><w:t>NestedCell</w:t></w:r></w:p>
</w:tc>
</w:tr>
</w:tbl>
</w:tc>
</w:tr>
</w:tbl>
</w:body>
</w:document>"#;
let document = write_docx(document_xml);
let result = to_markdown(document.path());
assert!(result.is_ok(), "Generated table DOCX should parse");
let text = result.unwrap();
assert!(text.contains("CellOne"));
assert!(text.contains("NestedCell"));
}