Skip to main content

ragfs_extract/
office.rs

1//! Office document extractor (OOXML and ODT).
2//!
3//! Reads ZIP containers and pulls visible text from XML parts.
4//! Supported: `.docx`, `.xlsx`, `.pptx`, `.odt`.
5//! Not supported: legacy binary `.doc` / `.xls` / `.ppt`, RTF, EPUB.
6
7use async_trait::async_trait;
8use ragfs_core::{
9    ContentElement, ContentExtractor, ContentMetadataInfo, ExtractError, ExtractedContent,
10};
11use std::cmp::Ordering;
12use std::io::{Cursor, Read};
13use std::path::Path;
14use tracing::debug;
15use zip::ZipArchive;
16
17const DOCX_MIME: &str = "application/vnd.openxmlformats-officedocument.wordprocessingml.document";
18const XLSX_MIME: &str = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet";
19const PPTX_MIME: &str = "application/vnd.openxmlformats-officedocument.presentationml.presentation";
20const ODT_MIME: &str = "application/vnd.oasis.opendocument.text";
21
22/// Per-entry uncompressed cap (zip-bomb guard).
23const MAX_ENTRY_UNCOMPRESSED: u64 = 8 * 1024 * 1024;
24/// Aggregate uncompressed cap across extracted XML parts.
25const MAX_TOTAL_UNCOMPRESSED: u64 = 32 * 1024 * 1024;
26/// Upper bound on spaces expanded from one ODT `text:c` count.
27const MAX_ODT_SPACES: usize = 255;
28
29/// Office/OpenDocument text extractor.
30pub struct OfficeExtractor;
31
32impl OfficeExtractor {
33    /// Create a new office extractor.
34    #[must_use]
35    pub fn new() -> Self {
36        Self
37    }
38}
39
40impl Default for OfficeExtractor {
41    fn default() -> Self {
42        Self::new()
43    }
44}
45
46#[derive(Clone, Copy, Debug, PartialEq, Eq)]
47enum OfficeKind {
48    Docx,
49    Xlsx,
50    Pptx,
51    Odt,
52}
53
54impl OfficeKind {
55    fn from_ext(ext: &str) -> Option<Self> {
56        match ext.to_ascii_lowercase().as_str() {
57            "docx" => Some(Self::Docx),
58            "xlsx" => Some(Self::Xlsx),
59            "pptx" => Some(Self::Pptx),
60            "odt" => Some(Self::Odt),
61            _ => None,
62        }
63    }
64
65    fn from_mime(mime: &str) -> Option<Self> {
66        match mime {
67            DOCX_MIME => Some(Self::Docx),
68            XLSX_MIME => Some(Self::Xlsx),
69            PPTX_MIME => Some(Self::Pptx),
70            ODT_MIME => Some(Self::Odt),
71            _ => None,
72        }
73    }
74
75    fn mime(self) -> &'static str {
76        match self {
77            Self::Docx => DOCX_MIME,
78            Self::Xlsx => XLSX_MIME,
79            Self::Pptx => PPTX_MIME,
80            Self::Odt => ODT_MIME,
81        }
82    }
83}
84
85#[async_trait]
86impl ContentExtractor for OfficeExtractor {
87    fn supported_types(&self) -> &[&str] {
88        &[DOCX_MIME, XLSX_MIME, PPTX_MIME, ODT_MIME]
89    }
90
91    fn can_extract_by_extension(&self, path: &Path) -> bool {
92        path.extension()
93            .and_then(|ext| ext.to_str())
94            .and_then(OfficeKind::from_ext)
95            .is_some()
96    }
97
98    async fn extract(&self, path: &Path) -> Result<ExtractedContent, ExtractError> {
99        debug!("Extracting office document: {:?}", path);
100        let bytes = tokio::fs::read(path).await?;
101        let kind = path
102            .extension()
103            .and_then(|ext| ext.to_str())
104            .and_then(OfficeKind::from_ext)
105            .ok_or_else(|| ExtractError::UnsupportedType(path.display().to_string()))?;
106        extract_kind(&bytes, kind)
107    }
108
109    async fn extract_bytes(
110        &self,
111        data: &[u8],
112        mime_type: &str,
113    ) -> Result<ExtractedContent, ExtractError> {
114        let kind = OfficeKind::from_mime(mime_type)
115            .ok_or_else(|| ExtractError::UnsupportedType(mime_type.to_string()))?;
116        extract_kind(data, kind)
117    }
118}
119
120fn extract_kind(bytes: &[u8], kind: OfficeKind) -> Result<ExtractedContent, ExtractError> {
121    let text = match kind {
122        OfficeKind::Docx => extract_named_parts(bytes, |name| {
123            name == "word/document.xml"
124                || name.starts_with("word/header")
125                || name.starts_with("word/footer")
126        })?,
127        OfficeKind::Xlsx => extract_xlsx(bytes)?,
128        OfficeKind::Pptx => extract_named_parts(bytes, |name| {
129            name.starts_with("ppt/slides/slide")
130                && Path::new(name)
131                    .extension()
132                    .is_some_and(|ext| ext.eq_ignore_ascii_case("xml"))
133        })?,
134        OfficeKind::Odt => extract_named_parts(bytes, |name| name == "content.xml")?,
135    };
136
137    if text.trim().is_empty() {
138        return Err(ExtractError::Failed(format!(
139            "no text extracted from {}",
140            kind.mime()
141        )));
142    }
143
144    let elements = text
145        .split('\n')
146        .filter(|line| !line.trim().is_empty())
147        .scan(0u64, |offset, line| {
148            let element = ContentElement::Paragraph {
149                text: line.to_string(),
150                byte_offset: *offset,
151            };
152            *offset += line.len() as u64 + 1;
153            Some(element)
154        })
155        .collect();
156
157    Ok(ExtractedContent {
158        text,
159        elements,
160        images: vec![],
161        metadata: ContentMetadataInfo::default(),
162    })
163}
164
165fn open_zip(bytes: &[u8]) -> Result<ZipArchive<Cursor<&[u8]>>, ExtractError> {
166    ZipArchive::new(Cursor::new(bytes))
167        .map_err(|e| ExtractError::Parse(format!("not a ZIP office document: {e}")))
168}
169
170fn collect_part_names(
171    archive: &mut ZipArchive<Cursor<&[u8]>>,
172    include: impl Fn(&str) -> bool,
173) -> Vec<String> {
174    let mut names: Vec<String> = (0..archive.len())
175        .filter_map(|i| {
176            let file = archive.by_index(i).ok()?;
177            let name = file.name().to_string();
178            include(&name).then_some(name)
179        })
180        .collect();
181    names.sort_by(|a, b| natural_cmp(a, b));
182    names
183}
184
185fn extract_named_parts(
186    bytes: &[u8],
187    include: impl Fn(&str) -> bool,
188) -> Result<String, ExtractError> {
189    let mut archive = open_zip(bytes)?;
190    let names = collect_part_names(&mut archive, include);
191    let mut remaining = MAX_TOTAL_UNCOMPRESSED;
192    let mut parts = Vec::new();
193    for name in names {
194        let mut file = archive
195            .by_name(&name)
196            .map_err(|e| ExtractError::Parse(format!("missing {name}: {e}")))?;
197        let xml = read_xml_part(&mut file, &mut remaining, MAX_ENTRY_UNCOMPRESSED)?;
198        let part = xml_to_text(&xml);
199        if !part.is_empty() {
200            parts.push(part);
201        }
202    }
203    Ok(parts.join("\n"))
204}
205
206fn extract_xlsx(bytes: &[u8]) -> Result<String, ExtractError> {
207    let mut archive = open_zip(bytes)?;
208    let mut remaining = MAX_TOTAL_UNCOMPRESSED;
209
210    let sst = if archive.by_name("xl/sharedStrings.xml").is_ok() {
211        let mut file = archive
212            .by_name("xl/sharedStrings.xml")
213            .map_err(|e| ExtractError::Parse(format!("missing sharedStrings: {e}")))?;
214        let xml = read_xml_part(&mut file, &mut remaining, MAX_ENTRY_UNCOMPRESSED)?;
215        drop(file);
216        parse_shared_strings(&xml)
217    } else {
218        Vec::new()
219    };
220
221    let sheets = collect_part_names(&mut archive, |name| {
222        name.starts_with("xl/worksheets/")
223            && Path::new(name)
224                .extension()
225                .is_some_and(|ext| ext.eq_ignore_ascii_case("xml"))
226    });
227
228    let mut parts = Vec::new();
229    for name in sheets {
230        let mut file = archive
231            .by_name(&name)
232            .map_err(|e| ExtractError::Parse(format!("missing {name}: {e}")))?;
233        let xml = read_xml_part(&mut file, &mut remaining, MAX_ENTRY_UNCOMPRESSED)?;
234        drop(file);
235        let part = xlsx_sheet_to_text(&xml, &sst);
236        if !part.is_empty() {
237            parts.push(part);
238        }
239    }
240    Ok(parts.join("\n"))
241}
242
243fn read_xml_part<R: Read + ?Sized>(
244    file: &mut zip::read::ZipFile<'_, R>,
245    remaining: &mut u64,
246    max_entry: u64,
247) -> Result<String, ExtractError> {
248    let bytes = read_entry_bytes(file, remaining, max_entry)?;
249    decode_xml_part(&bytes)
250}
251
252fn read_entry_bytes<R: Read + ?Sized>(
253    file: &mut zip::read::ZipFile<'_, R>,
254    remaining: &mut u64,
255    max_entry: u64,
256) -> Result<Vec<u8>, ExtractError> {
257    let name = file.name().to_string();
258    let declared = file.size();
259    if declared > max_entry {
260        return Err(ExtractError::Failed(format!(
261            "office ZIP entry {name} exceeds {max_entry} uncompressed bytes"
262        )));
263    }
264    if declared > *remaining {
265        return Err(ExtractError::Failed(format!(
266            "office ZIP aggregate uncompressed limit exceeded at {name}"
267        )));
268    }
269
270    let mut buf = Vec::new();
271    let mut limited = file.take(max_entry.saturating_add(1));
272    limited
273        .read_to_end(&mut buf)
274        .map_err(|e| ExtractError::Parse(format!("failed to read {name}: {e}")))?;
275    let len = buf.len() as u64;
276    if len > max_entry {
277        return Err(ExtractError::Failed(format!(
278            "office ZIP entry {name} exceeds {max_entry} uncompressed bytes"
279        )));
280    }
281    if len > *remaining {
282        return Err(ExtractError::Failed(format!(
283            "office ZIP aggregate uncompressed limit exceeded at {name}"
284        )));
285    }
286    *remaining -= len;
287    Ok(buf)
288}
289
290fn decode_xml_part(bytes: &[u8]) -> Result<String, ExtractError> {
291    if bytes.starts_with(&[0xFF, 0xFE]) {
292        return decode_utf16(&bytes[2..], true);
293    }
294    if bytes.starts_with(&[0xFE, 0xFF]) {
295        return decode_utf16(&bytes[2..], false);
296    }
297    match std::str::from_utf8(bytes) {
298        Ok(s) => Ok(s.to_string()),
299        Err(_) => decode_utf16(bytes, true).or_else(|_| decode_utf16(bytes, false)),
300    }
301}
302
303fn decode_utf16(bytes: &[u8], little_endian: bool) -> Result<String, ExtractError> {
304    if !bytes.len().is_multiple_of(2) {
305        return Err(ExtractError::Parse(
306            "UTF-16 XML part has an odd number of bytes".into(),
307        ));
308    }
309    let units: Vec<u16> = bytes
310        .as_chunks::<2>()
311        .0
312        .iter()
313        .map(|&c| {
314            if little_endian {
315                u16::from_le_bytes(c)
316            } else {
317                u16::from_be_bytes(c)
318            }
319        })
320        .collect();
321    String::from_utf16(&units)
322        .map_err(|e| ExtractError::Parse(format!("invalid UTF-16 XML part: {e}")))
323}
324
325fn natural_cmp(a: &str, b: &str) -> Ordering {
326    let mut ai = a.chars().peekable();
327    let mut bi = b.chars().peekable();
328    loop {
329        match (ai.peek(), bi.peek()) {
330            (None, None) => return Ordering::Equal,
331            (None, Some(_)) => return Ordering::Less,
332            (Some(_), None) => return Ordering::Greater,
333            (Some(ac), Some(bc)) if ac.is_ascii_digit() && bc.is_ascii_digit() => {
334                let mut an = 0u64;
335                while matches!(ai.peek(), Some(c) if c.is_ascii_digit()) {
336                    let d = ai.next().expect("digit") as u32 - u32::from(b'0');
337                    an = an.saturating_mul(10).saturating_add(u64::from(d));
338                }
339                let mut bn = 0u64;
340                while matches!(bi.peek(), Some(c) if c.is_ascii_digit()) {
341                    let d = bi.next().expect("digit") as u32 - u32::from(b'0');
342                    bn = bn.saturating_mul(10).saturating_add(u64::from(d));
343                }
344                match an.cmp(&bn) {
345                    Ordering::Equal => {}
346                    other => return other,
347                }
348            }
349            (Some(_), Some(_)) => {
350                let ac = ai.next().expect("char");
351                let bc = bi.next().expect("char");
352                match ac.cmp(&bc) {
353                    Ordering::Equal => {}
354                    other => return other,
355                }
356            }
357        }
358    }
359}
360
361fn parse_shared_strings(xml: &str) -> Vec<String> {
362    inner_elements(xml, "si")
363        .into_iter()
364        .map(xml_to_text)
365        .collect()
366}
367
368fn xlsx_sheet_to_text(xml: &str, sst: &[String]) -> String {
369    let mut out = String::new();
370    let mut rest = xml;
371    let mut shared = false;
372    let mut in_v = false;
373    let mut in_f = false;
374    let mut index_buf = String::new();
375
376    while let Some(start) = rest.find('<') {
377        if start > 0 {
378            if shared && in_v {
379                index_buf.push_str(&rest[..start]);
380            } else if !shared && !in_f {
381                push_decoded(&mut out, &rest[..start]);
382            }
383        }
384        let after = &rest[start + 1..];
385        let Some(end_rel) = after.find('>') else {
386            break;
387        };
388        let tag = &after[..end_rel];
389        let local = local_name(tag_name(tag));
390        let is_end = tag.starts_with('/');
391
392        if !is_end && local == "c" {
393            shared = cell_is_shared_string(tag);
394            in_f = false;
395            index_buf.clear();
396        } else if is_end && local == "c" {
397            shared = false;
398            in_v = false;
399            in_f = false;
400            index_buf.clear();
401            out.push('\n');
402        } else if !is_end && local == "f" {
403            in_f = true;
404        } else if is_end && local == "f" {
405            in_f = false;
406        } else if !is_end && local == "v" {
407            in_v = true;
408            index_buf.clear();
409        } else if is_end && local == "v" {
410            if shared {
411                if let Ok(i) = index_buf.trim().parse::<usize>()
412                    && let Some(s) = sst.get(i)
413                {
414                    if !out.is_empty() && !out.ends_with(['\n', ' ']) {
415                        out.push(' ');
416                    }
417                    out.push_str(s);
418                }
419                index_buf.clear();
420            }
421            in_v = false;
422        } else if is_end && is_block_local(local) {
423            out.push('\n');
424        }
425        rest = &after[end_rel + 1..];
426    }
427    if !rest.is_empty() && !shared && !in_f {
428        push_decoded(&mut out, rest);
429    }
430    normalize_ws(&out)
431}
432
433fn cell_is_shared_string(tag: &str) -> bool {
434    tag.contains("t=\"s\"") || tag.contains("t='s'")
435}
436
437fn inner_elements<'a>(xml: &'a str, local: &str) -> Vec<&'a str> {
438    let mut rest = xml;
439    let mut out = Vec::new();
440    while let Some(i) = rest.find('<') {
441        let after = &rest[i + 1..];
442        let Some(gt) = after.find('>') else {
443            break;
444        };
445        let tag = &after[..gt];
446        let is_end = tag.starts_with('/');
447        let is_empty = tag.ends_with('/');
448        if !is_end && !is_empty && local_name(tag_name(tag)) == local {
449            let inner = &after[gt + 1..];
450            if let Some(end) = find_close_local(inner, local) {
451                out.push(&inner[..end]);
452                rest = &inner[end..];
453                continue;
454            }
455        }
456        rest = &after[gt + 1..];
457    }
458    out
459}
460
461fn find_close_local(inner: &str, local: &str) -> Option<usize> {
462    let mut offset = 0;
463    let mut rest = inner;
464    while let Some(i) = rest.find("</") {
465        let after = &rest[i + 2..];
466        let gt = after.find('>')?;
467        if local_name(tag_name(&after[..gt])) == local {
468            return Some(offset + i);
469        }
470        offset += i + 2 + gt + 1;
471        rest = &inner[offset..];
472    }
473    None
474}
475
476fn xml_to_text(xml: &str) -> String {
477    let mut out = String::new();
478    let mut rest = xml;
479    let mut skip_depth = 0u32;
480    let mut vanish_run = false;
481    while let Some(start) = rest.find('<') {
482        if start > 0 && skip_depth == 0 && !vanish_run {
483            push_decoded(&mut out, &rest[..start]);
484        }
485        let after = &rest[start + 1..];
486        let Some(end_rel) = after.find('>') else {
487            break;
488        };
489        let tag = &after[..end_rel];
490        let local = local_name(tag_name(tag));
491        let is_end = tag.starts_with('/');
492        let is_empty = tag.ends_with('/');
493
494        if !is_end && matches!(local, "delText" | "del") {
495            if !is_empty {
496                skip_depth = skip_depth.saturating_add(1);
497            }
498        } else if is_end && matches!(local, "delText" | "del") {
499            skip_depth = skip_depth.saturating_sub(1);
500        }
501
502        if !is_end && local == "vanish" && !vanish_disabled(tag) {
503            vanish_run = true;
504        }
505        if is_end && local == "r" {
506            vanish_run = false;
507        }
508
509        if skip_depth == 0 && !vanish_run {
510            if is_end && is_block_local(local) {
511                out.push('\n');
512            } else if !is_end && local == "s" {
513                for _ in 0..odt_space_count(tag) {
514                    out.push(' ');
515                }
516            } else if matches!(local, "tab" | "br") {
517                out.push(' ');
518            } else if local == "line-break" {
519                out.push('\n');
520            }
521        }
522        rest = &after[end_rel + 1..];
523    }
524    if !rest.is_empty() && skip_depth == 0 && !vanish_run {
525        push_decoded(&mut out, rest);
526    }
527    normalize_ws(&out)
528}
529
530fn vanish_disabled(tag: &str) -> bool {
531    tag.contains("val=\"0\"")
532        || tag.contains("val='0'")
533        || tag.contains("val=\"false\"")
534        || tag.contains("val='false'")
535}
536
537fn tag_name(tag: &str) -> &str {
538    let trimmed = tag.trim_start_matches('/').trim_start_matches('?');
539    trimmed
540        .split(|c: char| c.is_whitespace() || c == '/')
541        .next()
542        .unwrap_or("")
543}
544
545fn local_name(qname: &str) -> &str {
546    qname.rsplit_once(':').map_or(qname, |(_, local)| local)
547}
548
549fn is_block_local(local: &str) -> bool {
550    matches!(local, "p" | "h" | "tr" | "si" | "c")
551}
552
553fn odt_space_count(tag: &str) -> usize {
554    for key in ["text:c=", "c="] {
555        for quote in ['"', '\''] {
556            let pat = format!("{key}{quote}");
557            if let Some(i) = tag.find(&pat) {
558                let rest = &tag[i + pat.len()..];
559                if let Some(end) = rest.find(quote)
560                    && let Ok(n) = rest[..end].parse::<usize>()
561                {
562                    return n.clamp(1, MAX_ODT_SPACES);
563                }
564            }
565        }
566    }
567    1
568}
569
570fn push_decoded(out: &mut String, raw: &str) {
571    let mut chars = raw.chars().peekable();
572    while let Some(c) = chars.next() {
573        if c == '&' {
574            let mut entity = String::new();
575            while let Some(&next) = chars.peek() {
576                chars.next();
577                if next == ';' {
578                    break;
579                }
580                entity.push(next);
581                if entity.len() > 10 {
582                    break;
583                }
584            }
585            match entity.as_str() {
586                "amp" => out.push('&'),
587                "lt" => out.push('<'),
588                "gt" => out.push('>'),
589                "quot" => out.push('"'),
590                "apos" => out.push('\''),
591                "nbsp" => out.push(' '),
592                other if other.starts_with('#') => {
593                    let code = if let Some(hex) = other.strip_prefix("#x") {
594                        u32::from_str_radix(hex, 16).ok()
595                    } else {
596                        other.strip_prefix('#').and_then(|n| n.parse().ok())
597                    };
598                    if let Some(ch) = code.and_then(char::from_u32) {
599                        out.push(ch);
600                    }
601                }
602                _ => {
603                    out.push('&');
604                    out.push_str(&entity);
605                }
606            }
607        } else {
608            out.push(c);
609        }
610    }
611}
612
613fn normalize_ws(text: &str) -> String {
614    text.lines()
615        .map(|line| line.split_whitespace().collect::<Vec<_>>().join(" "))
616        .filter(|line| !line.is_empty())
617        .collect::<Vec<_>>()
618        .join("\n")
619}
620
621#[cfg(test)]
622mod tests {
623    use super::*;
624    use std::io::{Cursor, Write};
625    use zip::ZipWriter;
626    use zip::write::SimpleFileOptions;
627
628    fn zip_with(files: &[(&str, &str)]) -> Vec<u8> {
629        let owned: Vec<(&str, Vec<u8>)> = files
630            .iter()
631            .map(|(n, b)| (*n, b.as_bytes().to_vec()))
632            .collect();
633        let refs: Vec<(&str, &[u8])> = owned.iter().map(|(n, b)| (*n, b.as_slice())).collect();
634        zip_with_bytes(&refs)
635    }
636
637    fn zip_with_bytes(files: &[(&str, &[u8])]) -> Vec<u8> {
638        let mut cursor = Cursor::new(Vec::new());
639        {
640            let mut zip = ZipWriter::new(&mut cursor);
641            let opts = SimpleFileOptions::default();
642            for (name, body) in files {
643                zip.start_file(*name, opts).unwrap();
644                zip.write_all(body).unwrap();
645            }
646            zip.finish().unwrap();
647        }
648        cursor.into_inner()
649    }
650
651    fn utf16_le_bom(s: &str) -> Vec<u8> {
652        let mut out = vec![0xFF, 0xFE];
653        for unit in s.encode_utf16() {
654            out.extend_from_slice(&unit.to_le_bytes());
655        }
656        out
657    }
658
659    #[test]
660    fn xml_to_text_strips_tags_and_entities() {
661        let xml = r"<w:p><w:r><w:t>Hello &amp; world</w:t></w:r></w:p>";
662        assert_eq!(xml_to_text(xml), "Hello & world");
663    }
664
665    #[test]
666    fn xml_to_text_emits_odt_whitespace() {
667        assert_eq!(xml_to_text("A<text:s/>B"), "A B");
668        assert_eq!(xml_to_text(r#"A<text:s text:c="3"/>B"#), "A B");
669        assert_eq!(xml_to_text("A<text:tab/>B"), "A B");
670        assert_eq!(xml_to_text("A<text:line-break/>B"), "A\nB");
671    }
672
673    #[test]
674    fn odt_space_count_honors_text_c_but_clamps() {
675        assert_eq!(odt_space_count(r#"text:s text:c="3""#), 3);
676        assert_eq!(odt_space_count(r#"text:s text:c="0""#), 1);
677        assert_eq!(
678            odt_space_count(r#"text:s text:c="18446744073709551615""#),
679            255
680        );
681        assert_eq!(
682            xml_to_text(r#"A<text:s text:c="18446744073709551615"/>B"#),
683            "A B"
684        );
685    }
686
687    #[test]
688    fn xml_to_text_skips_deleted_and_vanished() {
689        let xml = r"<w:p><w:r><w:t>Keep</w:t></w:r><w:del><w:r><w:delText>Gone</w:delText></w:r></w:del><w:r><w:rPr><w:vanish/></w:rPr><w:t>Hidden</w:t></w:r><w:r><w:t>Visible</w:t></w:r></w:p>";
690        let text = xml_to_text(xml);
691        assert!(text.contains("Keep"));
692        assert!(text.contains("Visible"));
693        assert!(!text.contains("Gone"));
694        assert!(!text.contains("Hidden"));
695    }
696
697    #[test]
698    fn natural_cmp_orders_slide10_after_slide2() {
699        assert_eq!(
700            natural_cmp("ppt/slides/slide2.xml", "ppt/slides/slide10.xml"),
701            Ordering::Less
702        );
703        let mut names = vec![
704            "ppt/slides/slide10.xml".to_string(),
705            "ppt/slides/slide2.xml".to_string(),
706        ];
707        names.sort_by(|a, b| natural_cmp(a, b));
708        assert_eq!(
709            names,
710            vec![
711                "ppt/slides/slide2.xml".to_string(),
712                "ppt/slides/slide10.xml".to_string()
713            ]
714        );
715    }
716
717    #[test]
718    fn rejects_legacy_doc_extension() {
719        let extractor = OfficeExtractor::new();
720        assert!(!extractor.can_extract_by_extension(Path::new("report.doc")));
721        assert!(extractor.can_extract_by_extension(Path::new("report.docx")));
722    }
723
724    #[tokio::test]
725    async fn extracts_docx_paragraph() {
726        let xml = r#"<?xml version="1.0"?>
727<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
728  <w:body>
729    <w:p><w:r><w:t>Indexed from DOCX</w:t></w:r></w:p>
730  </w:body>
731</w:document>"#;
732        let bytes = zip_with(&[("word/document.xml", xml)]);
733        let extractor = OfficeExtractor::new();
734        let content = extractor.extract_bytes(&bytes, DOCX_MIME).await.unwrap();
735        assert!(content.text.contains("Indexed from DOCX"));
736    }
737
738    #[tokio::test]
739    async fn extracts_xlsx_skips_formula_keeps_cached_value() {
740        let sheet = r#"<?xml version="1.0"?>
741<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
742  <sheetData>
743    <row><c><f>SUM(A1:A2)</f><v>3</v></c></row>
744  </sheetData>
745</worksheet>"#;
746        let bytes = zip_with(&[("xl/worksheets/sheet1.xml", sheet)]);
747        let extractor = OfficeExtractor::new();
748        let content = extractor.extract_bytes(&bytes, XLSX_MIME).await.unwrap();
749        assert!(content.text.contains('3'));
750        assert!(!content.text.contains("SUM"));
751    }
752
753    #[tokio::test]
754    async fn extracts_xlsx_prefixed_shared_string_cells() {
755        let sst = r#"<?xml version="1.0"?>
756<x:sst xmlns:x="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
757  <x:si><x:t>Prefixed</x:t></x:si>
758</x:sst>"#;
759        let sheet = r#"<?xml version="1.0"?>
760<x:worksheet xmlns:x="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
761  <x:sheetData>
762    <x:row><x:c t="s"><x:v>0</x:v></x:c></x:row>
763  </x:sheetData>
764</x:worksheet>"#;
765        let bytes = zip_with(&[
766            ("xl/sharedStrings.xml", sst),
767            ("xl/worksheets/sheet1.xml", sheet),
768        ]);
769        let extractor = OfficeExtractor::new();
770        let content = extractor.extract_bytes(&bytes, XLSX_MIME).await.unwrap();
771        assert!(content.text.contains("Prefixed"));
772        assert!(!content.text.split_whitespace().any(|w| w == "0"));
773    }
774
775    #[test]
776    fn xml_to_text_skips_prefixed_deleted_text() {
777        let xml = r"<ns:p><ns:r><ns:t>Keep</ns:t></ns:r><ns:del><ns:r><ns:delText>Gone</ns:delText></ns:r></ns:del></ns:p>";
778        let text = xml_to_text(xml);
779        assert!(text.contains("Keep"));
780        assert!(!text.contains("Gone"));
781    }
782
783    #[tokio::test]
784    async fn extracts_xlsx_shared_strings_in_sheet_order() {
785        let sst = r#"<?xml version="1.0"?>
786<sst xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
787  <si><t>Revenue</t></si>
788  <si><t>Q1 actuals</t></si>
789</sst>"#;
790        let sheet = r#"<?xml version="1.0"?>
791<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
792  <sheetData>
793    <row>
794      <c t="s"><v>1</v></c>
795      <c t="s"><v>0</v></c>
796    </row>
797  </sheetData>
798</worksheet>"#;
799        let bytes = zip_with(&[
800            ("xl/sharedStrings.xml", sst),
801            ("xl/worksheets/sheet1.xml", sheet),
802        ]);
803        let extractor = OfficeExtractor::new();
804        let content = extractor.extract_bytes(&bytes, XLSX_MIME).await.unwrap();
805        assert!(content.text.contains("Q1 actuals"));
806        assert!(content.text.contains("Revenue"));
807        let q1 = content.text.find("Q1 actuals").unwrap();
808        let rev = content.text.find("Revenue").unwrap();
809        assert!(q1 < rev, "worksheet order should resolve index 1 then 0");
810        assert!(
811            !content
812                .text
813                .split_whitespace()
814                .any(|w| w == "0" || w == "1")
815        );
816    }
817
818    #[tokio::test]
819    async fn extracts_pptx_slides_in_numeric_order() {
820        let slide = |title: &str| {
821            format!(
822                r#"<?xml version="1.0"?>
823<p:sld xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">
824  <a:p><a:r><a:t>{title}</a:t></a:r></a:p>
825</p:sld>"#
826            )
827        };
828        let s10 = slide("Slide ten");
829        let s2 = slide("Slide two");
830        let bytes = zip_with(&[
831            ("ppt/slides/slide10.xml", s10.as_str()),
832            ("ppt/slides/slide2.xml", s2.as_str()),
833        ]);
834        let extractor = OfficeExtractor::new();
835        let content = extractor.extract_bytes(&bytes, PPTX_MIME).await.unwrap();
836        let two = content.text.find("Slide two").unwrap();
837        let ten = content.text.find("Slide ten").unwrap();
838        assert!(two < ten);
839    }
840
841    #[tokio::test]
842    async fn extracts_pptx_slide() {
843        let xml = r#"<?xml version="1.0"?>
844<p:sld xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">
845  <a:p><a:r><a:t>Slide title</a:t></a:r></a:p>
846</p:sld>"#;
847        let bytes = zip_with(&[("ppt/slides/slide1.xml", xml)]);
848        let extractor = OfficeExtractor::new();
849        let content = extractor.extract_bytes(&bytes, PPTX_MIME).await.unwrap();
850        assert!(content.text.contains("Slide title"));
851    }
852
853    #[tokio::test]
854    async fn extracts_odt_content() {
855        let xml = r#"<?xml version="1.0"?>
856<office:document-content xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0">
857  <text:p>OpenDocument text</text:p>
858</office:document-content>"#;
859        let bytes = zip_with(&[("content.xml", xml)]);
860        let extractor = OfficeExtractor::new();
861        let content = extractor.extract_bytes(&bytes, ODT_MIME).await.unwrap();
862        assert!(content.text.contains("OpenDocument text"));
863    }
864
865    #[tokio::test]
866    async fn extracts_utf16_docx_part() {
867        let xml = r#"<?xml version="1.0" encoding="UTF-16"?>
868<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
869  <w:body>
870    <w:p><w:r><w:t>UTF16 body</w:t></w:r></w:p>
871  </w:body>
872</w:document>"#;
873        let encoded = utf16_le_bom(xml);
874        let bytes = zip_with_bytes(&[("word/document.xml", encoded.as_slice())]);
875        let extractor = OfficeExtractor::new();
876        let content = extractor.extract_bytes(&bytes, DOCX_MIME).await.unwrap();
877        assert!(content.text.contains("UTF16 body"));
878    }
879
880    #[test]
881    fn rejects_oversized_zip_entry() {
882        let bytes = zip_with(&[("word/document.xml", "hello")]);
883        let mut archive = ZipArchive::new(Cursor::new(bytes.as_slice())).unwrap();
884        let mut file = archive.by_name("word/document.xml").unwrap();
885        let mut remaining = 1024u64;
886        let err = read_entry_bytes(&mut file, &mut remaining, 2).unwrap_err();
887        assert!(matches!(err, ExtractError::Failed(_)));
888    }
889
890    #[tokio::test]
891    async fn extract_bytes_rejects_unknown_mime() {
892        let extractor = OfficeExtractor::new();
893        let err = extractor
894            .extract_bytes(b"not zip", "application/msword")
895            .await
896            .unwrap_err();
897        assert!(matches!(err, ExtractError::UnsupportedType(_)));
898    }
899}