1use async_trait::async_trait;
8use ragfs_core::{
9 ContentElement, ContentExtractor, ContentMetadataInfo, ExtractError, ExtractedContent,
10};
11use std::cmp::Ordering;
12use std::io::{Cursor, Read};
13use std::path::Path;
14use tracing::debug;
15use zip::ZipArchive;
16
17const DOCX_MIME: &str = "application/vnd.openxmlformats-officedocument.wordprocessingml.document";
18const XLSX_MIME: &str = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet";
19const PPTX_MIME: &str = "application/vnd.openxmlformats-officedocument.presentationml.presentation";
20const ODT_MIME: &str = "application/vnd.oasis.opendocument.text";
21
22const MAX_ENTRY_UNCOMPRESSED: u64 = 8 * 1024 * 1024;
24const MAX_TOTAL_UNCOMPRESSED: u64 = 32 * 1024 * 1024;
26const MAX_ODT_SPACES: usize = 255;
28
29pub struct OfficeExtractor;
31
32impl OfficeExtractor {
33 #[must_use]
35 pub fn new() -> Self {
36 Self
37 }
38}
39
40impl Default for OfficeExtractor {
41 fn default() -> Self {
42 Self::new()
43 }
44}
45
46#[derive(Clone, Copy, Debug, PartialEq, Eq)]
47enum OfficeKind {
48 Docx,
49 Xlsx,
50 Pptx,
51 Odt,
52}
53
54impl OfficeKind {
55 fn from_ext(ext: &str) -> Option<Self> {
56 match ext.to_ascii_lowercase().as_str() {
57 "docx" => Some(Self::Docx),
58 "xlsx" => Some(Self::Xlsx),
59 "pptx" => Some(Self::Pptx),
60 "odt" => Some(Self::Odt),
61 _ => None,
62 }
63 }
64
65 fn from_mime(mime: &str) -> Option<Self> {
66 match mime {
67 DOCX_MIME => Some(Self::Docx),
68 XLSX_MIME => Some(Self::Xlsx),
69 PPTX_MIME => Some(Self::Pptx),
70 ODT_MIME => Some(Self::Odt),
71 _ => None,
72 }
73 }
74
75 fn mime(self) -> &'static str {
76 match self {
77 Self::Docx => DOCX_MIME,
78 Self::Xlsx => XLSX_MIME,
79 Self::Pptx => PPTX_MIME,
80 Self::Odt => ODT_MIME,
81 }
82 }
83}
84
85#[async_trait]
86impl ContentExtractor for OfficeExtractor {
87 fn supported_types(&self) -> &[&str] {
88 &[DOCX_MIME, XLSX_MIME, PPTX_MIME, ODT_MIME]
89 }
90
91 fn can_extract_by_extension(&self, path: &Path) -> bool {
92 path.extension()
93 .and_then(|ext| ext.to_str())
94 .and_then(OfficeKind::from_ext)
95 .is_some()
96 }
97
98 async fn extract(&self, path: &Path) -> Result<ExtractedContent, ExtractError> {
99 debug!("Extracting office document: {:?}", path);
100 let bytes = tokio::fs::read(path).await?;
101 let kind = path
102 .extension()
103 .and_then(|ext| ext.to_str())
104 .and_then(OfficeKind::from_ext)
105 .ok_or_else(|| ExtractError::UnsupportedType(path.display().to_string()))?;
106 extract_kind(&bytes, kind)
107 }
108
109 async fn extract_bytes(
110 &self,
111 data: &[u8],
112 mime_type: &str,
113 ) -> Result<ExtractedContent, ExtractError> {
114 let kind = OfficeKind::from_mime(mime_type)
115 .ok_or_else(|| ExtractError::UnsupportedType(mime_type.to_string()))?;
116 extract_kind(data, kind)
117 }
118}
119
120fn extract_kind(bytes: &[u8], kind: OfficeKind) -> Result<ExtractedContent, ExtractError> {
121 let text = match kind {
122 OfficeKind::Docx => extract_named_parts(bytes, |name| {
123 name == "word/document.xml"
124 || name.starts_with("word/header")
125 || name.starts_with("word/footer")
126 })?,
127 OfficeKind::Xlsx => extract_xlsx(bytes)?,
128 OfficeKind::Pptx => extract_named_parts(bytes, |name| {
129 name.starts_with("ppt/slides/slide")
130 && Path::new(name)
131 .extension()
132 .is_some_and(|ext| ext.eq_ignore_ascii_case("xml"))
133 })?,
134 OfficeKind::Odt => extract_named_parts(bytes, |name| name == "content.xml")?,
135 };
136
137 if text.trim().is_empty() {
138 return Err(ExtractError::Failed(format!(
139 "no text extracted from {}",
140 kind.mime()
141 )));
142 }
143
144 let elements = text
145 .split('\n')
146 .filter(|line| !line.trim().is_empty())
147 .scan(0u64, |offset, line| {
148 let element = ContentElement::Paragraph {
149 text: line.to_string(),
150 byte_offset: *offset,
151 };
152 *offset += line.len() as u64 + 1;
153 Some(element)
154 })
155 .collect();
156
157 Ok(ExtractedContent {
158 text,
159 elements,
160 images: vec![],
161 metadata: ContentMetadataInfo::default(),
162 })
163}
164
165fn open_zip(bytes: &[u8]) -> Result<ZipArchive<Cursor<&[u8]>>, ExtractError> {
166 ZipArchive::new(Cursor::new(bytes))
167 .map_err(|e| ExtractError::Parse(format!("not a ZIP office document: {e}")))
168}
169
170fn collect_part_names(
171 archive: &mut ZipArchive<Cursor<&[u8]>>,
172 include: impl Fn(&str) -> bool,
173) -> Vec<String> {
174 let mut names: Vec<String> = (0..archive.len())
175 .filter_map(|i| {
176 let file = archive.by_index(i).ok()?;
177 let name = file.name().to_string();
178 include(&name).then_some(name)
179 })
180 .collect();
181 names.sort_by(|a, b| natural_cmp(a, b));
182 names
183}
184
185fn extract_named_parts(
186 bytes: &[u8],
187 include: impl Fn(&str) -> bool,
188) -> Result<String, ExtractError> {
189 let mut archive = open_zip(bytes)?;
190 let names = collect_part_names(&mut archive, include);
191 let mut remaining = MAX_TOTAL_UNCOMPRESSED;
192 let mut parts = Vec::new();
193 for name in names {
194 let mut file = archive
195 .by_name(&name)
196 .map_err(|e| ExtractError::Parse(format!("missing {name}: {e}")))?;
197 let xml = read_xml_part(&mut file, &mut remaining, MAX_ENTRY_UNCOMPRESSED)?;
198 let part = xml_to_text(&xml);
199 if !part.is_empty() {
200 parts.push(part);
201 }
202 }
203 Ok(parts.join("\n"))
204}
205
206fn extract_xlsx(bytes: &[u8]) -> Result<String, ExtractError> {
207 let mut archive = open_zip(bytes)?;
208 let mut remaining = MAX_TOTAL_UNCOMPRESSED;
209
210 let sst = if archive.by_name("xl/sharedStrings.xml").is_ok() {
211 let mut file = archive
212 .by_name("xl/sharedStrings.xml")
213 .map_err(|e| ExtractError::Parse(format!("missing sharedStrings: {e}")))?;
214 let xml = read_xml_part(&mut file, &mut remaining, MAX_ENTRY_UNCOMPRESSED)?;
215 drop(file);
216 parse_shared_strings(&xml)
217 } else {
218 Vec::new()
219 };
220
221 let sheets = collect_part_names(&mut archive, |name| {
222 name.starts_with("xl/worksheets/")
223 && Path::new(name)
224 .extension()
225 .is_some_and(|ext| ext.eq_ignore_ascii_case("xml"))
226 });
227
228 let mut parts = Vec::new();
229 for name in sheets {
230 let mut file = archive
231 .by_name(&name)
232 .map_err(|e| ExtractError::Parse(format!("missing {name}: {e}")))?;
233 let xml = read_xml_part(&mut file, &mut remaining, MAX_ENTRY_UNCOMPRESSED)?;
234 drop(file);
235 let part = xlsx_sheet_to_text(&xml, &sst);
236 if !part.is_empty() {
237 parts.push(part);
238 }
239 }
240 Ok(parts.join("\n"))
241}
242
243fn read_xml_part<R: Read + ?Sized>(
244 file: &mut zip::read::ZipFile<'_, R>,
245 remaining: &mut u64,
246 max_entry: u64,
247) -> Result<String, ExtractError> {
248 let bytes = read_entry_bytes(file, remaining, max_entry)?;
249 decode_xml_part(&bytes)
250}
251
252fn read_entry_bytes<R: Read + ?Sized>(
253 file: &mut zip::read::ZipFile<'_, R>,
254 remaining: &mut u64,
255 max_entry: u64,
256) -> Result<Vec<u8>, ExtractError> {
257 let name = file.name().to_string();
258 let declared = file.size();
259 if declared > max_entry {
260 return Err(ExtractError::Failed(format!(
261 "office ZIP entry {name} exceeds {max_entry} uncompressed bytes"
262 )));
263 }
264 if declared > *remaining {
265 return Err(ExtractError::Failed(format!(
266 "office ZIP aggregate uncompressed limit exceeded at {name}"
267 )));
268 }
269
270 let mut buf = Vec::new();
271 let mut limited = file.take(max_entry.saturating_add(1));
272 limited
273 .read_to_end(&mut buf)
274 .map_err(|e| ExtractError::Parse(format!("failed to read {name}: {e}")))?;
275 let len = buf.len() as u64;
276 if len > max_entry {
277 return Err(ExtractError::Failed(format!(
278 "office ZIP entry {name} exceeds {max_entry} uncompressed bytes"
279 )));
280 }
281 if len > *remaining {
282 return Err(ExtractError::Failed(format!(
283 "office ZIP aggregate uncompressed limit exceeded at {name}"
284 )));
285 }
286 *remaining -= len;
287 Ok(buf)
288}
289
290fn decode_xml_part(bytes: &[u8]) -> Result<String, ExtractError> {
291 if bytes.starts_with(&[0xFF, 0xFE]) {
292 return decode_utf16(&bytes[2..], true);
293 }
294 if bytes.starts_with(&[0xFE, 0xFF]) {
295 return decode_utf16(&bytes[2..], false);
296 }
297 match std::str::from_utf8(bytes) {
298 Ok(s) => Ok(s.to_string()),
299 Err(_) => decode_utf16(bytes, true).or_else(|_| decode_utf16(bytes, false)),
300 }
301}
302
303fn decode_utf16(bytes: &[u8], little_endian: bool) -> Result<String, ExtractError> {
304 if !bytes.len().is_multiple_of(2) {
305 return Err(ExtractError::Parse(
306 "UTF-16 XML part has an odd number of bytes".into(),
307 ));
308 }
309 let units: Vec<u16> = bytes
310 .as_chunks::<2>()
311 .0
312 .iter()
313 .map(|&c| {
314 if little_endian {
315 u16::from_le_bytes(c)
316 } else {
317 u16::from_be_bytes(c)
318 }
319 })
320 .collect();
321 String::from_utf16(&units)
322 .map_err(|e| ExtractError::Parse(format!("invalid UTF-16 XML part: {e}")))
323}
324
325fn natural_cmp(a: &str, b: &str) -> Ordering {
326 let mut ai = a.chars().peekable();
327 let mut bi = b.chars().peekable();
328 loop {
329 match (ai.peek(), bi.peek()) {
330 (None, None) => return Ordering::Equal,
331 (None, Some(_)) => return Ordering::Less,
332 (Some(_), None) => return Ordering::Greater,
333 (Some(ac), Some(bc)) if ac.is_ascii_digit() && bc.is_ascii_digit() => {
334 let mut an = 0u64;
335 while matches!(ai.peek(), Some(c) if c.is_ascii_digit()) {
336 let d = ai.next().expect("digit") as u32 - u32::from(b'0');
337 an = an.saturating_mul(10).saturating_add(u64::from(d));
338 }
339 let mut bn = 0u64;
340 while matches!(bi.peek(), Some(c) if c.is_ascii_digit()) {
341 let d = bi.next().expect("digit") as u32 - u32::from(b'0');
342 bn = bn.saturating_mul(10).saturating_add(u64::from(d));
343 }
344 match an.cmp(&bn) {
345 Ordering::Equal => {}
346 other => return other,
347 }
348 }
349 (Some(_), Some(_)) => {
350 let ac = ai.next().expect("char");
351 let bc = bi.next().expect("char");
352 match ac.cmp(&bc) {
353 Ordering::Equal => {}
354 other => return other,
355 }
356 }
357 }
358 }
359}
360
361fn parse_shared_strings(xml: &str) -> Vec<String> {
362 inner_elements(xml, "si")
363 .into_iter()
364 .map(xml_to_text)
365 .collect()
366}
367
368fn xlsx_sheet_to_text(xml: &str, sst: &[String]) -> String {
369 let mut out = String::new();
370 let mut rest = xml;
371 let mut shared = false;
372 let mut in_v = false;
373 let mut in_f = false;
374 let mut index_buf = String::new();
375
376 while let Some(start) = rest.find('<') {
377 if start > 0 {
378 if shared && in_v {
379 index_buf.push_str(&rest[..start]);
380 } else if !shared && !in_f {
381 push_decoded(&mut out, &rest[..start]);
382 }
383 }
384 let after = &rest[start + 1..];
385 let Some(end_rel) = after.find('>') else {
386 break;
387 };
388 let tag = &after[..end_rel];
389 let local = local_name(tag_name(tag));
390 let is_end = tag.starts_with('/');
391
392 if !is_end && local == "c" {
393 shared = cell_is_shared_string(tag);
394 in_f = false;
395 index_buf.clear();
396 } else if is_end && local == "c" {
397 shared = false;
398 in_v = false;
399 in_f = false;
400 index_buf.clear();
401 out.push('\n');
402 } else if !is_end && local == "f" {
403 in_f = true;
404 } else if is_end && local == "f" {
405 in_f = false;
406 } else if !is_end && local == "v" {
407 in_v = true;
408 index_buf.clear();
409 } else if is_end && local == "v" {
410 if shared {
411 if let Ok(i) = index_buf.trim().parse::<usize>()
412 && let Some(s) = sst.get(i)
413 {
414 if !out.is_empty() && !out.ends_with(['\n', ' ']) {
415 out.push(' ');
416 }
417 out.push_str(s);
418 }
419 index_buf.clear();
420 }
421 in_v = false;
422 } else if is_end && is_block_local(local) {
423 out.push('\n');
424 }
425 rest = &after[end_rel + 1..];
426 }
427 if !rest.is_empty() && !shared && !in_f {
428 push_decoded(&mut out, rest);
429 }
430 normalize_ws(&out)
431}
432
433fn cell_is_shared_string(tag: &str) -> bool {
434 tag.contains("t=\"s\"") || tag.contains("t='s'")
435}
436
437fn inner_elements<'a>(xml: &'a str, local: &str) -> Vec<&'a str> {
438 let mut rest = xml;
439 let mut out = Vec::new();
440 while let Some(i) = rest.find('<') {
441 let after = &rest[i + 1..];
442 let Some(gt) = after.find('>') else {
443 break;
444 };
445 let tag = &after[..gt];
446 let is_end = tag.starts_with('/');
447 let is_empty = tag.ends_with('/');
448 if !is_end && !is_empty && local_name(tag_name(tag)) == local {
449 let inner = &after[gt + 1..];
450 if let Some(end) = find_close_local(inner, local) {
451 out.push(&inner[..end]);
452 rest = &inner[end..];
453 continue;
454 }
455 }
456 rest = &after[gt + 1..];
457 }
458 out
459}
460
461fn find_close_local(inner: &str, local: &str) -> Option<usize> {
462 let mut offset = 0;
463 let mut rest = inner;
464 while let Some(i) = rest.find("</") {
465 let after = &rest[i + 2..];
466 let gt = after.find('>')?;
467 if local_name(tag_name(&after[..gt])) == local {
468 return Some(offset + i);
469 }
470 offset += i + 2 + gt + 1;
471 rest = &inner[offset..];
472 }
473 None
474}
475
476fn xml_to_text(xml: &str) -> String {
477 let mut out = String::new();
478 let mut rest = xml;
479 let mut skip_depth = 0u32;
480 let mut vanish_run = false;
481 while let Some(start) = rest.find('<') {
482 if start > 0 && skip_depth == 0 && !vanish_run {
483 push_decoded(&mut out, &rest[..start]);
484 }
485 let after = &rest[start + 1..];
486 let Some(end_rel) = after.find('>') else {
487 break;
488 };
489 let tag = &after[..end_rel];
490 let local = local_name(tag_name(tag));
491 let is_end = tag.starts_with('/');
492 let is_empty = tag.ends_with('/');
493
494 if !is_end && matches!(local, "delText" | "del") {
495 if !is_empty {
496 skip_depth = skip_depth.saturating_add(1);
497 }
498 } else if is_end && matches!(local, "delText" | "del") {
499 skip_depth = skip_depth.saturating_sub(1);
500 }
501
502 if !is_end && local == "vanish" && !vanish_disabled(tag) {
503 vanish_run = true;
504 }
505 if is_end && local == "r" {
506 vanish_run = false;
507 }
508
509 if skip_depth == 0 && !vanish_run {
510 if is_end && is_block_local(local) {
511 out.push('\n');
512 } else if !is_end && local == "s" {
513 for _ in 0..odt_space_count(tag) {
514 out.push(' ');
515 }
516 } else if matches!(local, "tab" | "br") {
517 out.push(' ');
518 } else if local == "line-break" {
519 out.push('\n');
520 }
521 }
522 rest = &after[end_rel + 1..];
523 }
524 if !rest.is_empty() && skip_depth == 0 && !vanish_run {
525 push_decoded(&mut out, rest);
526 }
527 normalize_ws(&out)
528}
529
530fn vanish_disabled(tag: &str) -> bool {
531 tag.contains("val=\"0\"")
532 || tag.contains("val='0'")
533 || tag.contains("val=\"false\"")
534 || tag.contains("val='false'")
535}
536
537fn tag_name(tag: &str) -> &str {
538 let trimmed = tag.trim_start_matches('/').trim_start_matches('?');
539 trimmed
540 .split(|c: char| c.is_whitespace() || c == '/')
541 .next()
542 .unwrap_or("")
543}
544
545fn local_name(qname: &str) -> &str {
546 qname.rsplit_once(':').map_or(qname, |(_, local)| local)
547}
548
549fn is_block_local(local: &str) -> bool {
550 matches!(local, "p" | "h" | "tr" | "si" | "c")
551}
552
553fn odt_space_count(tag: &str) -> usize {
554 for key in ["text:c=", "c="] {
555 for quote in ['"', '\''] {
556 let pat = format!("{key}{quote}");
557 if let Some(i) = tag.find(&pat) {
558 let rest = &tag[i + pat.len()..];
559 if let Some(end) = rest.find(quote)
560 && let Ok(n) = rest[..end].parse::<usize>()
561 {
562 return n.clamp(1, MAX_ODT_SPACES);
563 }
564 }
565 }
566 }
567 1
568}
569
570fn push_decoded(out: &mut String, raw: &str) {
571 let mut chars = raw.chars().peekable();
572 while let Some(c) = chars.next() {
573 if c == '&' {
574 let mut entity = String::new();
575 while let Some(&next) = chars.peek() {
576 chars.next();
577 if next == ';' {
578 break;
579 }
580 entity.push(next);
581 if entity.len() > 10 {
582 break;
583 }
584 }
585 match entity.as_str() {
586 "amp" => out.push('&'),
587 "lt" => out.push('<'),
588 "gt" => out.push('>'),
589 "quot" => out.push('"'),
590 "apos" => out.push('\''),
591 "nbsp" => out.push(' '),
592 other if other.starts_with('#') => {
593 let code = if let Some(hex) = other.strip_prefix("#x") {
594 u32::from_str_radix(hex, 16).ok()
595 } else {
596 other.strip_prefix('#').and_then(|n| n.parse().ok())
597 };
598 if let Some(ch) = code.and_then(char::from_u32) {
599 out.push(ch);
600 }
601 }
602 _ => {
603 out.push('&');
604 out.push_str(&entity);
605 }
606 }
607 } else {
608 out.push(c);
609 }
610 }
611}
612
613fn normalize_ws(text: &str) -> String {
614 text.lines()
615 .map(|line| line.split_whitespace().collect::<Vec<_>>().join(" "))
616 .filter(|line| !line.is_empty())
617 .collect::<Vec<_>>()
618 .join("\n")
619}
620
621#[cfg(test)]
622mod tests {
623 use super::*;
624 use std::io::{Cursor, Write};
625 use zip::ZipWriter;
626 use zip::write::SimpleFileOptions;
627
628 fn zip_with(files: &[(&str, &str)]) -> Vec<u8> {
629 let owned: Vec<(&str, Vec<u8>)> = files
630 .iter()
631 .map(|(n, b)| (*n, b.as_bytes().to_vec()))
632 .collect();
633 let refs: Vec<(&str, &[u8])> = owned.iter().map(|(n, b)| (*n, b.as_slice())).collect();
634 zip_with_bytes(&refs)
635 }
636
637 fn zip_with_bytes(files: &[(&str, &[u8])]) -> Vec<u8> {
638 let mut cursor = Cursor::new(Vec::new());
639 {
640 let mut zip = ZipWriter::new(&mut cursor);
641 let opts = SimpleFileOptions::default();
642 for (name, body) in files {
643 zip.start_file(*name, opts).unwrap();
644 zip.write_all(body).unwrap();
645 }
646 zip.finish().unwrap();
647 }
648 cursor.into_inner()
649 }
650
651 fn utf16_le_bom(s: &str) -> Vec<u8> {
652 let mut out = vec![0xFF, 0xFE];
653 for unit in s.encode_utf16() {
654 out.extend_from_slice(&unit.to_le_bytes());
655 }
656 out
657 }
658
659 #[test]
660 fn xml_to_text_strips_tags_and_entities() {
661 let xml = r"<w:p><w:r><w:t>Hello & world</w:t></w:r></w:p>";
662 assert_eq!(xml_to_text(xml), "Hello & world");
663 }
664
665 #[test]
666 fn xml_to_text_emits_odt_whitespace() {
667 assert_eq!(xml_to_text("A<text:s/>B"), "A B");
668 assert_eq!(xml_to_text(r#"A<text:s text:c="3"/>B"#), "A B");
669 assert_eq!(xml_to_text("A<text:tab/>B"), "A B");
670 assert_eq!(xml_to_text("A<text:line-break/>B"), "A\nB");
671 }
672
673 #[test]
674 fn odt_space_count_honors_text_c_but_clamps() {
675 assert_eq!(odt_space_count(r#"text:s text:c="3""#), 3);
676 assert_eq!(odt_space_count(r#"text:s text:c="0""#), 1);
677 assert_eq!(
678 odt_space_count(r#"text:s text:c="18446744073709551615""#),
679 255
680 );
681 assert_eq!(
682 xml_to_text(r#"A<text:s text:c="18446744073709551615"/>B"#),
683 "A B"
684 );
685 }
686
687 #[test]
688 fn xml_to_text_skips_deleted_and_vanished() {
689 let xml = r"<w:p><w:r><w:t>Keep</w:t></w:r><w:del><w:r><w:delText>Gone</w:delText></w:r></w:del><w:r><w:rPr><w:vanish/></w:rPr><w:t>Hidden</w:t></w:r><w:r><w:t>Visible</w:t></w:r></w:p>";
690 let text = xml_to_text(xml);
691 assert!(text.contains("Keep"));
692 assert!(text.contains("Visible"));
693 assert!(!text.contains("Gone"));
694 assert!(!text.contains("Hidden"));
695 }
696
697 #[test]
698 fn natural_cmp_orders_slide10_after_slide2() {
699 assert_eq!(
700 natural_cmp("ppt/slides/slide2.xml", "ppt/slides/slide10.xml"),
701 Ordering::Less
702 );
703 let mut names = vec![
704 "ppt/slides/slide10.xml".to_string(),
705 "ppt/slides/slide2.xml".to_string(),
706 ];
707 names.sort_by(|a, b| natural_cmp(a, b));
708 assert_eq!(
709 names,
710 vec![
711 "ppt/slides/slide2.xml".to_string(),
712 "ppt/slides/slide10.xml".to_string()
713 ]
714 );
715 }
716
717 #[test]
718 fn rejects_legacy_doc_extension() {
719 let extractor = OfficeExtractor::new();
720 assert!(!extractor.can_extract_by_extension(Path::new("report.doc")));
721 assert!(extractor.can_extract_by_extension(Path::new("report.docx")));
722 }
723
724 #[tokio::test]
725 async fn extracts_docx_paragraph() {
726 let xml = r#"<?xml version="1.0"?>
727<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
728 <w:body>
729 <w:p><w:r><w:t>Indexed from DOCX</w:t></w:r></w:p>
730 </w:body>
731</w:document>"#;
732 let bytes = zip_with(&[("word/document.xml", xml)]);
733 let extractor = OfficeExtractor::new();
734 let content = extractor.extract_bytes(&bytes, DOCX_MIME).await.unwrap();
735 assert!(content.text.contains("Indexed from DOCX"));
736 }
737
738 #[tokio::test]
739 async fn extracts_xlsx_skips_formula_keeps_cached_value() {
740 let sheet = r#"<?xml version="1.0"?>
741<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
742 <sheetData>
743 <row><c><f>SUM(A1:A2)</f><v>3</v></c></row>
744 </sheetData>
745</worksheet>"#;
746 let bytes = zip_with(&[("xl/worksheets/sheet1.xml", sheet)]);
747 let extractor = OfficeExtractor::new();
748 let content = extractor.extract_bytes(&bytes, XLSX_MIME).await.unwrap();
749 assert!(content.text.contains('3'));
750 assert!(!content.text.contains("SUM"));
751 }
752
753 #[tokio::test]
754 async fn extracts_xlsx_prefixed_shared_string_cells() {
755 let sst = r#"<?xml version="1.0"?>
756<x:sst xmlns:x="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
757 <x:si><x:t>Prefixed</x:t></x:si>
758</x:sst>"#;
759 let sheet = r#"<?xml version="1.0"?>
760<x:worksheet xmlns:x="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
761 <x:sheetData>
762 <x:row><x:c t="s"><x:v>0</x:v></x:c></x:row>
763 </x:sheetData>
764</x:worksheet>"#;
765 let bytes = zip_with(&[
766 ("xl/sharedStrings.xml", sst),
767 ("xl/worksheets/sheet1.xml", sheet),
768 ]);
769 let extractor = OfficeExtractor::new();
770 let content = extractor.extract_bytes(&bytes, XLSX_MIME).await.unwrap();
771 assert!(content.text.contains("Prefixed"));
772 assert!(!content.text.split_whitespace().any(|w| w == "0"));
773 }
774
775 #[test]
776 fn xml_to_text_skips_prefixed_deleted_text() {
777 let xml = r"<ns:p><ns:r><ns:t>Keep</ns:t></ns:r><ns:del><ns:r><ns:delText>Gone</ns:delText></ns:r></ns:del></ns:p>";
778 let text = xml_to_text(xml);
779 assert!(text.contains("Keep"));
780 assert!(!text.contains("Gone"));
781 }
782
783 #[tokio::test]
784 async fn extracts_xlsx_shared_strings_in_sheet_order() {
785 let sst = r#"<?xml version="1.0"?>
786<sst xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
787 <si><t>Revenue</t></si>
788 <si><t>Q1 actuals</t></si>
789</sst>"#;
790 let sheet = r#"<?xml version="1.0"?>
791<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
792 <sheetData>
793 <row>
794 <c t="s"><v>1</v></c>
795 <c t="s"><v>0</v></c>
796 </row>
797 </sheetData>
798</worksheet>"#;
799 let bytes = zip_with(&[
800 ("xl/sharedStrings.xml", sst),
801 ("xl/worksheets/sheet1.xml", sheet),
802 ]);
803 let extractor = OfficeExtractor::new();
804 let content = extractor.extract_bytes(&bytes, XLSX_MIME).await.unwrap();
805 assert!(content.text.contains("Q1 actuals"));
806 assert!(content.text.contains("Revenue"));
807 let q1 = content.text.find("Q1 actuals").unwrap();
808 let rev = content.text.find("Revenue").unwrap();
809 assert!(q1 < rev, "worksheet order should resolve index 1 then 0");
810 assert!(
811 !content
812 .text
813 .split_whitespace()
814 .any(|w| w == "0" || w == "1")
815 );
816 }
817
818 #[tokio::test]
819 async fn extracts_pptx_slides_in_numeric_order() {
820 let slide = |title: &str| {
821 format!(
822 r#"<?xml version="1.0"?>
823<p:sld xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">
824 <a:p><a:r><a:t>{title}</a:t></a:r></a:p>
825</p:sld>"#
826 )
827 };
828 let s10 = slide("Slide ten");
829 let s2 = slide("Slide two");
830 let bytes = zip_with(&[
831 ("ppt/slides/slide10.xml", s10.as_str()),
832 ("ppt/slides/slide2.xml", s2.as_str()),
833 ]);
834 let extractor = OfficeExtractor::new();
835 let content = extractor.extract_bytes(&bytes, PPTX_MIME).await.unwrap();
836 let two = content.text.find("Slide two").unwrap();
837 let ten = content.text.find("Slide ten").unwrap();
838 assert!(two < ten);
839 }
840
841 #[tokio::test]
842 async fn extracts_pptx_slide() {
843 let xml = r#"<?xml version="1.0"?>
844<p:sld xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">
845 <a:p><a:r><a:t>Slide title</a:t></a:r></a:p>
846</p:sld>"#;
847 let bytes = zip_with(&[("ppt/slides/slide1.xml", xml)]);
848 let extractor = OfficeExtractor::new();
849 let content = extractor.extract_bytes(&bytes, PPTX_MIME).await.unwrap();
850 assert!(content.text.contains("Slide title"));
851 }
852
853 #[tokio::test]
854 async fn extracts_odt_content() {
855 let xml = r#"<?xml version="1.0"?>
856<office:document-content xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0">
857 <text:p>OpenDocument text</text:p>
858</office:document-content>"#;
859 let bytes = zip_with(&[("content.xml", xml)]);
860 let extractor = OfficeExtractor::new();
861 let content = extractor.extract_bytes(&bytes, ODT_MIME).await.unwrap();
862 assert!(content.text.contains("OpenDocument text"));
863 }
864
865 #[tokio::test]
866 async fn extracts_utf16_docx_part() {
867 let xml = r#"<?xml version="1.0" encoding="UTF-16"?>
868<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
869 <w:body>
870 <w:p><w:r><w:t>UTF16 body</w:t></w:r></w:p>
871 </w:body>
872</w:document>"#;
873 let encoded = utf16_le_bom(xml);
874 let bytes = zip_with_bytes(&[("word/document.xml", encoded.as_slice())]);
875 let extractor = OfficeExtractor::new();
876 let content = extractor.extract_bytes(&bytes, DOCX_MIME).await.unwrap();
877 assert!(content.text.contains("UTF16 body"));
878 }
879
880 #[test]
881 fn rejects_oversized_zip_entry() {
882 let bytes = zip_with(&[("word/document.xml", "hello")]);
883 let mut archive = ZipArchive::new(Cursor::new(bytes.as_slice())).unwrap();
884 let mut file = archive.by_name("word/document.xml").unwrap();
885 let mut remaining = 1024u64;
886 let err = read_entry_bytes(&mut file, &mut remaining, 2).unwrap_err();
887 assert!(matches!(err, ExtractError::Failed(_)));
888 }
889
890 #[tokio::test]
891 async fn extract_bytes_rejects_unknown_mime() {
892 let extractor = OfficeExtractor::new();
893 let err = extractor
894 .extract_bytes(b"not zip", "application/msword")
895 .await
896 .unwrap_err();
897 assert!(matches!(err, ExtractError::UnsupportedType(_)));
898 }
899}