use crate::Workspace;
use crate::document::DocOutcome;
use crate::tools::{ImagePayload, ImagePayloadSource, ToolOutput};
use anyhow::Context;
use std::path::{Path, PathBuf};
const MAX_INJECTED_IMAGES: usize = 5;
const HEAD_SNIFF_BYTES: usize = 16;
const NO_ARTIFACT_DIR_REASON: &str = "could not be converted";
pub(super) async fn read_document(
ws: &Workspace,
res: &super::read::ResolvedRead,
strict: bool,
) -> anyhow::Result<Option<ToolOutput>> {
let Ok(meta) = tokio::fs::metadata(&res.path).await else {
return Ok(None);
};
if !meta.is_file() {
return Ok(None);
}
let Some(name) = res.path.file_name().and_then(|n| n.to_str()) else {
return Ok(None);
};
let Some(head) = peek_head(&res.path).await else {
return Ok(None);
};
if !crate::document::needs_extraction(&head, name) {
return Ok(None);
}
crate::tools::check_size_within(
&meta,
crate::util::FILE_MAX_BYTES,
"File too large to convert",
)?;
let dir = match create_artifact_dir(ws, strict).await {
Ok(dir) => dir,
Err(e) => {
tracing::warn!(error = %e, "Failed to create a document artifact directory");
return Ok(Some(plain_answer(res, NO_ARTIFACT_DIR_REASON)));
}
};
crate::tools::shell::record_spill_owner(dir.clone());
let outcome = match crate::document::convert_document_file(&res.path, name, &dir).await {
DocOutcome::Unsupported => None,
DocOutcome::Unreadable { reason } => Some(plain_answer(res, &reason)),
DocOutcome::Text {
text,
images,
notes,
all_page_text_lost,
} => Some(compose_output(res, &dir, text, images, notes, all_page_text_lost).await),
};
let _ = tokio::fs::remove_dir(&dir).await;
Ok(outcome)
}
fn plain_answer(res: &super::read::ResolvedRead, reason: &str) -> ToolOutput {
ToolOutput {
text: crate::tools::with_recovery_note(
res.recovery_note.as_deref(),
answer_line(&res.path.display().to_string(), reason),
),
image_payloads: Vec::new(),
text_is_content: true,
}
}
fn answer_line(display: &str, body: impl std::fmt::Display) -> String {
format!("[{display}: {body}]")
}
async fn create_artifact_dir(ws: &Workspace, strict: bool) -> anyhow::Result<PathBuf> {
let parent = if strict {
ws.as_path().join("uploads")
} else {
crate::tools::shell::agent_temp_dir()
.ok_or_else(|| anyhow::anyhow!("no temp directory available for document conversion"))?
};
tokio::fs::create_dir_all(&parent)
.await
.with_context(|| format!("failed to create {}", parent.display()))?;
let dir = parent.join(format!("read_{:016x}", rand::random::<u64>()));
tokio::fs::create_dir(&dir)
.await
.with_context(|| format!("failed to create {}", dir.display()))?;
Ok(dir)
}
async fn peek_head(path: &Path) -> Option<Vec<u8>> {
use tokio::io::AsyncReadExt;
let mut file = tokio::fs::File::open(path).await.ok()?;
let mut head = [0u8; HEAD_SNIFF_BYTES];
let mut filled = 0;
while filled < HEAD_SNIFF_BYTES {
match file.read(&mut head[filled..]).await {
Ok(0) => break,
Ok(n) => filled += n,
Err(_) => return None,
}
}
Some(head[..filled].to_vec())
}
async fn compose_output(
res: &super::read::ResolvedRead,
dir: &Path,
text: String,
images: Vec<PathBuf>,
notes: Vec<String>,
all_page_text_lost: bool,
) -> ToolOutput {
let display = res.path.display().to_string();
let text = crate::util::scrub_credentials(text.trim());
let over_cap = images.len() > MAX_INJECTED_IMAGES;
let mut attached = Vec::new();
let mut unattached = 0usize;
if !over_cap {
for image in &images {
match crate::util::local_image_to_compressed_data_uri_with_meta(image).await {
Ok(meta) => attached.push(ImagePayload::from_compressed_meta(
image,
meta,
None,
ImagePayloadSource::Generated,
)),
Err(e) => {
tracing::warn!(
path = %image.display(),
error = %e,
"Failed to encode an extracted document image"
);
unattached += 1;
}
}
}
}
let note = res.recovery_note.as_deref();
let mut text_block = if text.is_empty() {
if all_page_text_lost {
String::new()
} else if attached.is_empty() {
answer_line(&display, crate::document::NO_TEXT_NOTE)
} else {
answer_line(&display, crate::document::NO_TEXT_LAYER_NOTE)
}
} else {
format!(
"{}\n\n{text}",
answer_line(&display, "extracted text follows")
)
};
let mut trailing: Vec<String> = notes.iter().map(|n| answer_line(&display, n)).collect();
if unattached > 0 {
trailing.push(answer_line(
&display,
format!(
"{unattached} extracted image(s) could not be attached; they are in {} — read them individually if you need them",
dir.display()
),
));
}
let (full_listing, folder_listing) = if over_cap {
let head = format!(
"[{display}: {} extracted image(s) were not attached ({MAX_INJECTED_IMAGES} is the per-call limit);",
images.len()
);
let mut listing = vec![format!("{head} read them individually as images:]")];
listing.extend(images.iter().map(|image| image.display().to_string()));
let folder = vec![format!(
"{head} they are all in {} — list that folder, then read the images you need.]",
dir.display()
)];
(listing, folder)
} else {
(Vec::new(), Vec::new())
};
let compose = |text_block: &str, listing: &[String]| {
let mut lines: Vec<&str> = Vec::with_capacity(2 + trailing.len() + listing.len());
if !text_block.is_empty() {
lines.push(text_block);
}
lines.extend(trailing.iter().map(String::as_str));
lines.extend(listing.iter().map(String::as_str));
crate::tools::with_recovery_note(note, lines.join("\n"))
};
let fits = |text_block: &str, listing: &[String]| {
compose(text_block, listing).len() <= crate::util::TOOL_OUTPUT_BUDGET_BYTES
};
let pick_listing = |text_block: &str| {
if fits(text_block, &full_listing) {
&full_listing
} else {
&folder_listing
}
};
let mut listing = pick_listing(&text_block);
if !fits(&text_block, listing) && !text.is_empty() {
text_block = spill_text(dir, &display, &text).await;
listing = pick_listing(&text_block);
}
ToolOutput {
text: compose(&text_block, listing),
image_payloads: attached,
text_is_content: true,
}
}
async fn spill_text(dir: &Path, display: &str, text: &str) -> String {
let path = dir.join(crate::tools::path::format_spill_filename());
match tokio::fs::write(&path, text.as_bytes()).await {
Ok(()) => answer_line(
display,
crate::document::spilled_text_note(text.chars().count(), &path),
),
Err(e) => {
tracing::warn!(
path = %path.display(),
error = %e,
"Failed to save the extracted document text"
);
answer_line(display, "the extracted text could not be saved")
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::document::test_fixtures::{DOCX_BODY, multi_page_pdf, zip_fixture};
use tempfile::TempDir;
fn temp_workspace(files: &[(&str, &[u8])]) -> (TempDir, Workspace) {
let dir = TempDir::new().expect("tempdir");
for (name, bytes) in files {
std::fs::write(dir.path().join(name), bytes).expect("write fixture");
}
let ws = crate::workspace::test_ws(dir.path());
(dir, ws)
}
struct SpillOwner(String);
impl SpillOwner {
fn new() -> Self {
Self(format!("read-document-test-{:016x}", rand::random::<u64>()))
}
async fn scope<T>(&self, future: impl std::future::Future<Output = T>) -> T {
crate::agent::CURRENT_TOOL_AGENT_ID
.scope(Some(self.0.clone()), future)
.await
}
}
impl Drop for SpillOwner {
fn drop(&mut self) {
crate::tools::shell::cleanup_agent_spills(&self.0);
}
}
async fn convert(
owner: &SpillOwner,
ws: &Workspace,
name: &str,
strict: bool,
) -> Option<ToolOutput> {
let res = super::super::read::resolve_content_read(ws, name, strict)
.await
.expect("resolve the fixture");
owner
.scope(read_document(ws, &res, strict))
.await
.expect("read the document")
}
fn png_fixture(size: u32) -> Vec<u8> {
let image = image::RgbImage::from_pixel(size, size, image::Rgb([10, 200, 10]));
let mut bytes = std::io::Cursor::new(Vec::new());
image::DynamicImage::ImageRgb8(image)
.write_to(&mut bytes, image::ImageFormat::Png)
.expect("encode fixture png");
bytes.into_inner()
}
fn raster_only_pdf() -> Vec<u8> {
multi_page_pdf(&[(
"/MediaBox [0 0 612 792]",
b"0.1 0.5 0.9 rg 0 0 612 792 re f",
)])
}
fn pdf_with_no_media_box() -> Vec<u8> {
multi_page_pdf(&[("", b"BT /F1 24 Tf 72 700 Td (Page text long enough) Tj ET")])
}
fn spilled_path(text: &str) -> PathBuf {
let start = text.find("saved to ").expect("a spill line") + "saved to ".len();
let end = start + text[start..].find(" — read that file").expect("a pointer");
PathBuf::from(&text[start..end])
}
const EMPTY_DOCX_BODY: &[u8] = br#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"><w:body></w:body></w:document>"#;
#[tokio::test]
async fn text_free_documents_name_their_pages_only_when_attached() {
let empty = zip_fixture(&[("word/document.xml", EMPTY_DOCX_BODY)]);
let owner = SpillOwner::new();
let (_dir, ws) = temp_workspace(&[("empty.docx", &empty)]);
let out = convert(&owner, &ws, "empty.docx", false)
.await
.expect("a .docx is a document");
assert_eq!(out.image_payloads.len(), 0);
assert!(
out.text.contains("empty.docx: no text could be extracted]"),
"{:?}",
out.text
);
let with_media = docx_with_images(EMPTY_DOCX_BODY, 6);
let (_dir, ws) = temp_workspace(&[("scanned.docx", &with_media)]);
let out = convert(&owner, &ws, "scanned.docx", false)
.await
.expect("a .docx is a document");
assert_eq!(
out.image_payloads.len(),
0,
"over the cap nothing is attached"
);
assert!(
out.text
.contains("scanned.docx: no text could be extracted]\n"),
"{:?}",
out.text
);
assert!(
!out.text.contains(crate::document::NO_TEXT_LAYER_NOTE),
"pages that were not attached are not called attached: {:?}",
out.text
);
}
#[tokio::test]
async fn image_only_pdf_pages_come_back_as_images() {
let owner = SpillOwner::new();
let (_dir, ws) = temp_workspace(&[("scan.pdf", &raster_only_pdf())]);
let out = convert(&owner, &ws, "scan.pdf", false)
.await
.expect("a .pdf is a document");
assert_eq!(
out.image_payloads.len(),
1,
"one rasterized page: {}",
out.text
);
assert!(
out.text
.contains("no text could be extracted — the pages were provided as images"),
"{}",
out.text
);
assert!(
out.text_is_content,
"the page images supplement, never replace, the answer"
);
}
#[tokio::test]
async fn unread_text_is_named_not_reported_as_absent() {
let owner = SpillOwner::new();
let (_dir, ws) = temp_workspace(&[("broken.pdf", &pdf_with_no_media_box())]);
let out = convert(&owner, &ws, "broken.pdf", false)
.await
.expect("a .pdf is a document");
assert!(
out.text
.contains("broken.pdf: the text of page 1 could not be read (1 of 1 pages)"),
"{}",
out.text
);
assert!(
!out.text.contains(crate::document::NO_TEXT_NOTE),
"a reader failure is not a document without text: {}",
out.text
);
assert_eq!(
out.image_payloads.len(),
1,
"the page that could not be read comes back as an image: {}",
out.text
);
}
#[tokio::test]
async fn corrupt_pdf_is_reported_not_failed() {
crate::util::test::init_test_stores().await;
let owner = SpillOwner::new();
let (_dir, ws) = temp_workspace(&[("broken.pdf", b"%PDF-1.4\ngarbage")]);
let out = convert(&owner, &ws, "brokn.pdf", false)
.await
.expect("a .pdf is a document");
assert!(out.text.starts_with("[Recovered path: "), "{}", out.text);
assert!(out.text.contains("could not be parsed"), "{}", out.text);
assert!(out.image_payloads.is_empty());
}
fn docx_with_images(body: &[u8], count: usize) -> Vec<u8> {
let png = png_fixture(16);
let names: Vec<String> = (1..=count)
.map(|page| format!("word/media/page_{page}.png"))
.collect();
let mut entries: Vec<(&str, &[u8])> = vec![("word/document.xml", body)];
entries.extend(names.iter().map(|name| (name.as_str(), png.as_slice())));
zip_fixture(&entries)
}
#[tokio::test]
async fn images_over_the_cap_are_delivered_as_paths() {
let fixture = docx_with_images(DOCX_BODY, 6);
let owner = SpillOwner::new();
let (_dir, ws) = temp_workspace(&[("report.docx", &fixture)]);
let out = convert(&owner, &ws, "report.docx", false)
.await
.expect("a .docx is a document");
assert!(
out.image_payloads.is_empty(),
"nothing may be attached: {}",
out.text
);
assert!(
out.text
.contains("6 extracted image(s) were not attached (5 is the per-call limit); read them individually as images:]"),
"the annotation is one closed envelope: {}",
out.text
);
let listed: Vec<&str> = out
.text
.lines()
.skip_while(|line| !line.contains("were not attached"))
.skip(1)
.collect();
assert_eq!(listed.len(), 6, "one line per produced image: {listed:?}");
assert!(
listed
.iter()
.all(|path| std::path::Path::new(path).is_absolute()),
"each image is named by its absolute path: {listed:?}"
);
}
#[tokio::test]
async fn a_listing_too_long_to_fit_names_the_folder() {
let fixture = docx_with_images(DOCX_BODY, 120);
let owner = SpillOwner::new();
let (_dir, ws) = temp_workspace(&[("many.docx", &fixture)]);
let out = convert(&owner, &ws, "many.docx", false)
.await
.expect("a .docx is a document");
assert!(
out.text.len() <= crate::util::TOOL_OUTPUT_BUDGET_BYTES,
"the answer must fit the budget it is formatted against: {} bytes",
out.text.len()
);
assert!(
out.text.contains("they are all in "),
"the folder is named rather than a truncated listing: {}",
out.text
);
}
#[tokio::test]
async fn plain_text_is_not_a_document() {
use crate::Tool;
let owner = SpillOwner::new();
let (_dir, ws) = temp_workspace(&[("notes.md", b"# Notes\n")]);
assert!(convert(&owner, &ws, "notes.md", false).await.is_none());
let out = owner
.scope(
crate::tools::read::ReadTool::general()
.execute_with_payloads(&ws, serde_json::json!({"path": "notes.md"})),
)
.await
.expect("read the text file");
assert!(out.text.contains("1: # Notes"), "{}", out.text);
assert!(out.image_payloads.is_empty());
}
#[tokio::test]
async fn read_tool_delivers_the_converted_document() {
use crate::Tool;
let owner = SpillOwner::new();
let fixture = zip_fixture(&[("word/document.xml", DOCX_BODY)]);
let (_dir, ws) = temp_workspace(&[("report.docx", &fixture)]);
let out = owner
.scope(
crate::tools::read::ReadTool::general()
.execute_with_payloads(&ws, serde_json::json!({"path": "report.docx"})),
)
.await
.expect("read the document");
assert!(out.text.contains("extracted text follows"), "{}", out.text);
assert!(
out.text.contains("First paragraph\nSecond paragraph"),
"{}",
out.text
);
assert!(
!out.text.contains("1: "),
"a document is converted, not line-numbered: {}",
out.text
);
assert!(
!out.text.contains("too long to inline"),
"short text is inlined, never spilled: {}",
out.text
);
assert!(
out.image_payloads.is_empty(),
"the fixture has no media entries"
);
}
#[tokio::test]
async fn strict_read_keeps_artifacts_inside_the_workspace() {
let (dir, ws) = temp_workspace(&[("scan.pdf", &raster_only_pdf())]);
let out = convert(&SpillOwner::new(), &ws, "scan.pdf", true)
.await
.expect("a .pdf is a document");
let image = out.image_payloads.first().expect("the page is attached");
let uploads = std::fs::canonicalize(dir.path())
.expect("canonical workspace")
.join("uploads");
assert!(
crate::util::is_within(Path::new(&image.path), &uploads),
"a restricted read must place artifacts where it can open them: {}",
image.path
);
}
#[tokio::test]
async fn long_text_with_many_images_is_spilled_not_truncated() {
let filler = "lorem ipsum dolor sit amet ".repeat(1_200);
let body = format!(
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"><w:body><w:p><w:r><w:t>FIRST-MARKER</w:t></w:r></w:p><w:p><w:r><w:t>{filler}</w:t></w:r></w:p><w:p><w:r><w:t>LAST-MARKER</w:t></w:r></w:p></w:body></w:document>"#
);
let fixture = docx_with_images(body.as_bytes(), 6);
let owner = SpillOwner::new();
let (_dir, ws) = temp_workspace(&[("long.docx", &fixture)]);
let out = convert(&owner, &ws, "long.docx", false)
.await
.expect("a .docx is a document");
assert!(
out.text.len() <= crate::util::TOOL_OUTPUT_BUDGET_BYTES,
"the answer must fit the budget it is formatted against: {} bytes",
out.text.len()
);
assert!(
!out.text.contains("FIRST-MARKER"),
"long extracted text is spilled, not inlined: {}",
out.text
);
let spilled =
std::fs::read_to_string(spilled_path(&out.text)).expect("read the spill file");
assert!(
spilled.contains("FIRST-MARKER") && spilled.contains("LAST-MARKER"),
"the spill file holds the whole text, head to tail"
);
assert_eq!(
out.text
.lines()
.filter(|line| std::path::Path::new(line).extension() == Some("png".as_ref()))
.count(),
6,
"spilling the text frees room for the per-image paths again: {}",
out.text
);
}
}