Skip to main content

ragfs_core/
types.rs

1//! Core types for RAGFS.
2//!
3//! This module contains all shared data structures used across RAGFS:
4//!
5//! ## File Management
6//! - [`FileRecord`]: Metadata about an indexed file
7//! - [`FileStatus`]: Current indexing state of a file
8//! - [`FileEvent`]: File system events for the watcher
9//!
10//! ## Content Chunks
11//! - [`Chunk`]: A segment of content with its embedding
12//! - [`ContentType`]: Type classification for chunk content
13//! - [`ChunkConfig`]: Configuration for chunking behavior
14//!
15//! ## Extraction
16//! - [`ExtractedContent`]: Content extracted from a file
17//! - [`ContentElement`]: Structural elements (headings, paragraphs, etc.)
18//!
19//! ## Embeddings
20//! - [`Modality`]: Supported embedding modalities (text, image, audio)
21//! - [`EmbeddingConfig`]: Configuration for embedding generation
22//! - [`EmbeddingOutput`]: Result of embedding a text
23//!
24//! ## Search
25//! - [`SearchQuery`]: Parameters for a vector search
26//! - [`SearchResult`]: A matching chunk with similarity score
27//! - [`SearchFilter`]: Filters to narrow search results
28//! - [`DirectoryScope`]: Relative directory metadata for scoped search
29//! - [`DistanceMetric`]: Vector distance calculation method
30
31use chrono::{DateTime, Utc};
32use serde::{Deserialize, Serialize};
33use std::collections::HashMap;
34use std::ops::Range;
35use std::path::PathBuf;
36use uuid::Uuid;
37
38// ============================================================================
39// File Records
40// ============================================================================
41
42/// Metadata about an indexed file.
43#[derive(Debug, Clone, Serialize, Deserialize)]
44pub struct FileRecord {
45    /// Unique file identifier
46    pub id: Uuid,
47    /// Absolute path to the file
48    pub path: PathBuf,
49    /// File size in bytes
50    pub size_bytes: u64,
51    /// MIME type
52    pub mime_type: String,
53    /// Content hash for change detection (blake3)
54    pub content_hash: String,
55    /// Last modification time
56    pub modified_at: DateTime<Utc>,
57    /// When the file was indexed (None if not yet indexed)
58    pub indexed_at: Option<DateTime<Utc>>,
59    /// Number of chunks produced
60    pub chunk_count: u32,
61    /// Current indexing status
62    pub status: FileStatus,
63    /// Error message if status is Error
64    pub error_message: Option<String>,
65}
66
67/// File indexing status.
68#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
69#[serde(rename_all = "lowercase")]
70pub enum FileStatus {
71    /// Waiting to be indexed
72    Pending,
73    /// Currently being indexed
74    Indexing,
75    /// Successfully indexed
76    Indexed,
77    /// Indexing failed
78    Error,
79    /// File was deleted
80    Deleted,
81}
82
83// ============================================================================
84// Chunks
85// ============================================================================
86
87/// A chunk of content from a file.
88#[derive(Debug, Clone, Serialize, Deserialize)]
89pub struct Chunk {
90    /// Unique chunk identifier
91    pub id: Uuid,
92    /// Parent file identifier
93    pub file_id: Uuid,
94    /// Path to the source file
95    pub file_path: PathBuf,
96    /// The actual content
97    pub content: String,
98    /// Type of content
99    pub content_type: ContentType,
100    /// MIME type of the source file
101    pub mime_type: Option<String>,
102    /// Position in file (0-indexed)
103    pub chunk_index: u32,
104    /// Byte range in source file
105    pub byte_range: Range<u64>,
106    /// Line range (if applicable)
107    pub line_range: Option<Range<u32>>,
108    /// Parent chunk ID (for hierarchical chunking)
109    pub parent_chunk_id: Option<Uuid>,
110    /// Depth in hierarchy (0 = root)
111    pub depth: u8,
112    /// Embedding vector (if computed)
113    pub embedding: Option<Vec<f32>>,
114    /// Relative directory of the source file (e.g. `src/auth`). Empty at the index root.
115    pub dir_path: String,
116    /// Number of directory components in `dir_path`.
117    pub dir_depth: u16,
118    /// Comma-separated relative path components, including the filename.
119    pub path_components: String,
120    /// Additional metadata
121    pub metadata: ChunkMetadata,
122}
123
124/// Directory-scope metadata stored on each indexed chunk.
125///
126/// This is a path prefix, not a trie. Search uses exact `dir_path` or a
127/// subdirectory (`src/auth` matches `src/auth` and `src/auth/oauth`).
128#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
129pub struct DirectoryScope {
130    /// Relative directory containing the file (e.g. `src/auth`).
131    pub dir_path: String,
132    /// Number of directory components in `dir_path`.
133    pub dir_depth: u16,
134    /// Comma-separated relative path components including the filename.
135    pub path_components: String,
136}
137
138impl DirectoryScope {
139    /// Derive scope fields from `file_path`, optionally relative to an index `root`.
140    #[must_use]
141    pub fn from_paths(file_path: &std::path::Path, root: Option<&std::path::Path>) -> Self {
142        let rel = root
143            .and_then(|r| file_path.strip_prefix(r).ok())
144            .unwrap_or(file_path);
145        let parts: Vec<String> = rel
146            .components()
147            .filter_map(|c| match c {
148                std::path::Component::Normal(s) => Some(s.to_string_lossy().into_owned()),
149                _ => None,
150            })
151            .collect();
152
153        let dir_parts = parts.get(..parts.len().saturating_sub(1)).unwrap_or(&[]);
154        let dir_path = dir_parts.join("/");
155        let dir_depth = dir_parts.len() as u16;
156        let path_components = parts.join(",");
157
158        Self {
159            dir_path,
160            dir_depth,
161            path_components,
162        }
163    }
164
165    /// Derive scope from an already-relative or absolute file path (no index root).
166    #[must_use]
167    pub fn from_file_path(file_path: &std::path::Path) -> Self {
168        Self::from_paths(file_path, None)
169    }
170
171    /// Normalize a CLI/query scope (`src/auth/`, `src\\auth` → `src/auth`).
172    #[must_use]
173    pub fn normalize_prefix(scope: &str) -> String {
174        scope.replace('\\', "/").trim_matches('/').to_string()
175    }
176
177    /// Whether this directory is `scope` or a subdirectory of it.
178    #[must_use]
179    pub fn matches_prefix(&self, scope: &str) -> bool {
180        dir_path_matches_scope(&self.dir_path, scope)
181    }
182
183    /// Infer the index root from an absolute file path and its stored relative `dir_path`.
184    #[must_use]
185    pub fn infer_root(file_path: &std::path::Path, dir_path: &str) -> Option<std::path::PathBuf> {
186        let parent = file_path.parent()?;
187        let dir = Self::normalize_prefix(dir_path);
188        if dir.is_empty() {
189            return Some(parent.to_path_buf());
190        }
191        let parent_norm = parent.to_string_lossy().replace('\\', "/");
192        parent_norm
193            .strip_suffix(&dir)
194            .map(|root| std::path::PathBuf::from(root.trim_end_matches('/')))
195    }
196}
197
198/// Match a stored `dir_path` against a scope prefix.
199///
200/// Empty / `.` scope matches everything (no restriction).
201///
202/// Besides an exact directory or a subdirectory (`src/auth` matches
203/// `src/auth/oauth`), a path-component-aligned suffix also matches. That
204/// covers schema-v1 indexes whose `dir_path` was backfilled from an
205/// absolute `file_path` (`project/src/auth` still matches `--scope src/auth`).
206#[must_use]
207pub fn dir_path_matches_scope(dir_path: &str, scope: &str) -> bool {
208    let scope = DirectoryScope::normalize_prefix(scope);
209    if scope.is_empty() || scope == "." {
210        return true;
211    }
212    let dir = DirectoryScope::normalize_prefix(dir_path);
213    dir == scope
214        || dir.starts_with(&format!("{scope}/"))
215        || dir.ends_with(&format!("/{scope}"))
216        || dir.contains(&format!("/{scope}/"))
217}
218
219/// Type of chunk content.
220#[derive(Debug, Clone, Serialize, Deserialize)]
221#[serde(tag = "type", rename_all = "snake_case")]
222pub enum ContentType {
223    /// Plain text content
224    Text,
225    /// Source code
226    Code {
227        /// Programming language
228        language: String,
229        /// Code symbol information
230        symbol: Option<CodeSymbol>,
231    },
232    /// Caption for an image
233    ImageCaption,
234    /// Content from a PDF page
235    PdfPage {
236        /// Page number (1-indexed)
237        page_num: u32,
238    },
239    /// Markdown content
240    Markdown,
241}
242
243/// Code symbol information.
244#[derive(Debug, Clone, Serialize, Deserialize)]
245pub struct CodeSymbol {
246    /// Type of symbol
247    pub kind: SymbolKind,
248    /// Symbol name
249    pub name: String,
250}
251
252/// Types of code symbols.
253#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
254#[serde(rename_all = "snake_case")]
255pub enum SymbolKind {
256    Function,
257    Method,
258    Class,
259    Struct,
260    Enum,
261    Module,
262    Constant,
263    Variable,
264    Interface,
265    Trait,
266}
267
268/// Metadata associated with a chunk.
269#[derive(Debug, Clone, Default, Serialize, Deserialize)]
270pub struct ChunkMetadata {
271    /// Embedding model used
272    pub embedding_model: Option<String>,
273    /// When chunk was indexed
274    pub indexed_at: Option<DateTime<Utc>>,
275    /// Token count (approximate)
276    pub token_count: Option<usize>,
277    /// Additional key-value metadata
278    #[serde(flatten)]
279    pub extra: HashMap<String, String>,
280}
281
282// ============================================================================
283// Extraction
284// ============================================================================
285
286/// Content extracted from a file.
287#[derive(Debug, Clone)]
288pub struct ExtractedContent {
289    /// Main text content
290    pub text: String,
291    /// Structured elements
292    pub elements: Vec<ContentElement>,
293    /// Extracted images
294    pub images: Vec<ExtractedImage>,
295    /// File-level metadata
296    pub metadata: ContentMetadataInfo,
297}
298
299/// A structural element in extracted content.
300#[derive(Debug, Clone)]
301pub enum ContentElement {
302    Heading {
303        level: u8,
304        text: String,
305        byte_offset: u64,
306    },
307    Paragraph {
308        text: String,
309        byte_offset: u64,
310    },
311    CodeBlock {
312        language: Option<String>,
313        code: String,
314        byte_offset: u64,
315    },
316    List {
317        items: Vec<String>,
318        ordered: bool,
319        byte_offset: u64,
320    },
321    Table {
322        headers: Vec<String>,
323        rows: Vec<Vec<String>>,
324        byte_offset: u64,
325    },
326}
327
328/// An image extracted from a document.
329#[derive(Debug, Clone)]
330pub struct ExtractedImage {
331    /// Raw image data
332    pub data: Vec<u8>,
333    /// MIME type
334    pub mime_type: String,
335    /// Caption if available
336    pub caption: Option<String>,
337    /// Page number (for PDFs)
338    pub page: Option<u32>,
339}
340
341/// Metadata extracted from file content.
342#[derive(Debug, Clone, Default)]
343pub struct ContentMetadataInfo {
344    /// Document title
345    pub title: Option<String>,
346    /// Author
347    pub author: Option<String>,
348    /// Language
349    pub language: Option<String>,
350    /// Page count (for PDFs)
351    pub page_count: Option<u32>,
352    /// Creation date
353    pub created_at: Option<DateTime<Utc>>,
354}
355
356// ============================================================================
357// Chunking
358// ============================================================================
359
360/// Configuration for chunking.
361#[derive(Debug, Clone, Serialize, Deserialize)]
362pub struct ChunkConfig {
363    /// Target chunk size in tokens
364    pub target_size: usize,
365    /// Maximum chunk size in tokens
366    pub max_size: usize,
367    /// Overlap between chunks in tokens
368    pub overlap: usize,
369    /// Enable hierarchical chunking
370    pub hierarchical: bool,
371    /// Maximum hierarchy depth
372    pub max_depth: u8,
373}
374
375impl Default for ChunkConfig {
376    fn default() -> Self {
377        Self {
378            target_size: 512,
379            max_size: 1024,
380            overlap: 64,
381            hierarchical: true,
382            max_depth: 2,
383        }
384    }
385}
386
387/// Output from a chunker.
388#[derive(Debug, Clone)]
389pub struct ChunkOutput {
390    /// Chunk content
391    pub content: String,
392    /// Byte range in source
393    pub byte_range: Range<u64>,
394    /// Line range if applicable
395    pub line_range: Option<Range<u32>>,
396    /// Index of parent chunk (in output array)
397    pub parent_index: Option<usize>,
398    /// Depth in hierarchy
399    pub depth: u8,
400    /// Additional metadata
401    pub metadata: ChunkOutputMetadata,
402}
403
404/// Metadata for chunk output.
405#[derive(Debug, Clone, Default)]
406pub struct ChunkOutputMetadata {
407    /// Symbol type (for code)
408    pub symbol_type: Option<String>,
409    /// Symbol name (for code)
410    pub symbol_name: Option<String>,
411    /// Programming language
412    pub language: Option<String>,
413}
414
415// ============================================================================
416// Embedding
417// ============================================================================
418
419/// Supported modalities for embedding.
420#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
421#[serde(rename_all = "lowercase")]
422pub enum Modality {
423    Text,
424    Image,
425    Audio,
426}
427
428/// Configuration for embedding.
429#[derive(Debug, Clone, Serialize, Deserialize)]
430pub struct EmbeddingConfig {
431    /// Normalize embeddings to unit length
432    pub normalize: bool,
433    /// Instruction prefix for models that support it
434    pub instruction: Option<String>,
435    /// Batch size for processing
436    pub batch_size: usize,
437}
438
439impl Default for EmbeddingConfig {
440    fn default() -> Self {
441        Self {
442            normalize: true,
443            instruction: None,
444            batch_size: 32,
445        }
446    }
447}
448
449/// Output from embedding.
450#[derive(Debug, Clone)]
451pub struct EmbeddingOutput {
452    /// The embedding vector
453    pub embedding: Vec<f32>,
454    /// Number of tokens in input
455    pub token_count: usize,
456}
457
458// ============================================================================
459// Search
460// ============================================================================
461
462/// A search query.
463#[derive(Debug, Clone)]
464pub struct SearchQuery {
465    /// Query embedding
466    pub embedding: Vec<f32>,
467    /// Optional text for hybrid search
468    pub text: Option<String>,
469    /// Maximum results to return
470    pub limit: usize,
471    /// Search filters
472    pub filters: Vec<SearchFilter>,
473    /// Distance metric
474    pub metric: DistanceMetric,
475    /// Optional directory scope, relative to the index root.
476    ///
477    /// When set, only chunks whose `dir_path` is this directory or a
478    /// subdirectory are considered. Applied as a filter alongside ANN.
479    pub scope_prefix: Option<String>,
480}
481
482/// Search filters.
483#[derive(Debug, Clone)]
484pub enum SearchFilter {
485    /// Match files with path prefix
486    PathPrefix(String),
487    /// Match files by glob pattern
488    PathGlob(String),
489    /// Match by MIME type
490    MimeType(String),
491    /// Match by programming language
492    Language(String),
493    /// Files modified after date
494    ModifiedAfter(DateTime<Utc>),
495    /// Files modified before date
496    ModifiedBefore(DateTime<Utc>),
497    /// Minimum hierarchy depth
498    MinDepth(u8),
499    /// Maximum hierarchy depth
500    MaxDepth(u8),
501}
502
503/// Distance metric for vector search.
504#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
505#[serde(rename_all = "lowercase")]
506pub enum DistanceMetric {
507    #[default]
508    Cosine,
509    L2,
510    Dot,
511}
512
513/// A search result.
514#[derive(Debug, Clone, Serialize, Deserialize)]
515pub struct SearchResult {
516    /// Chunk ID
517    pub chunk_id: Uuid,
518    /// File path
519    pub file_path: PathBuf,
520    /// Chunk content
521    pub content: String,
522    /// Similarity score
523    pub score: f32,
524    /// Byte range in file
525    pub byte_range: Range<u64>,
526    /// Line range if available
527    pub line_range: Option<Range<u32>>,
528    /// Additional metadata
529    pub metadata: HashMap<String, String>,
530}
531
532/// Vector store statistics.
533#[derive(Debug, Clone, Serialize, Deserialize)]
534pub struct StoreStats {
535    /// Total number of chunks
536    pub total_chunks: u64,
537    /// Total number of files
538    pub total_files: u64,
539    /// Index size in bytes
540    pub index_size_bytes: u64,
541    /// Last update time
542    pub last_updated: Option<DateTime<Utc>>,
543}
544
545// ============================================================================
546// Index Status
547// ============================================================================
548
549/// Overall index statistics.
550#[derive(Debug, Clone, Default, Serialize, Deserialize)]
551pub struct IndexStats {
552    /// Total files tracked
553    pub total_files: u64,
554    /// Successfully indexed files
555    pub indexed_files: u64,
556    /// Files pending indexing
557    pub pending_files: u64,
558    /// Files with errors
559    pub error_files: u64,
560    /// Total chunks stored
561    pub total_chunks: u64,
562    /// Last update time
563    pub last_update: Option<DateTime<Utc>>,
564}
565
566// ============================================================================
567// File Events
568// ============================================================================
569
570/// File system event for indexing.
571#[derive(Debug, Clone)]
572pub enum FileEvent {
573    Created(PathBuf),
574    Modified(PathBuf),
575    Deleted(PathBuf),
576    Renamed { from: PathBuf, to: PathBuf },
577}
578
579#[cfg(test)]
580mod tests {
581    use super::*;
582
583    // ==================== FileRecord Tests ====================
584
585    #[test]
586    fn test_file_record_serialization() {
587        let record = FileRecord {
588            id: Uuid::new_v4(),
589            path: PathBuf::from("/test/file.txt"),
590            size_bytes: 1024,
591            mime_type: "text/plain".to_string(),
592            content_hash: "abc123".to_string(),
593            modified_at: Utc::now(),
594            indexed_at: Some(Utc::now()),
595            chunk_count: 5,
596            status: FileStatus::Indexed,
597            error_message: None,
598        };
599
600        let json = serde_json::to_string(&record).unwrap();
601        let deserialized: FileRecord = serde_json::from_str(&json).unwrap();
602
603        assert_eq!(record.id, deserialized.id);
604        assert_eq!(record.path, deserialized.path);
605        assert_eq!(record.size_bytes, deserialized.size_bytes);
606        assert_eq!(record.status, deserialized.status);
607    }
608
609    #[test]
610    fn test_file_status_serialization() {
611        assert_eq!(
612            serde_json::to_string(&FileStatus::Pending).unwrap(),
613            "\"pending\""
614        );
615        assert_eq!(
616            serde_json::to_string(&FileStatus::Indexed).unwrap(),
617            "\"indexed\""
618        );
619        assert_eq!(
620            serde_json::to_string(&FileStatus::Error).unwrap(),
621            "\"error\""
622        );
623    }
624
625    #[test]
626    fn test_file_status_equality() {
627        assert_eq!(FileStatus::Pending, FileStatus::Pending);
628        assert_ne!(FileStatus::Pending, FileStatus::Indexed);
629    }
630
631    // ==================== Chunk Tests ====================
632
633    #[test]
634    fn test_chunk_serialization() {
635        let chunk = Chunk {
636            id: Uuid::new_v4(),
637            file_id: Uuid::new_v4(),
638            file_path: PathBuf::from("/test/file.rs"),
639            content: "fn main() {}".to_string(),
640            content_type: ContentType::Code {
641                language: "rust".to_string(),
642                symbol: Some(CodeSymbol {
643                    kind: SymbolKind::Function,
644                    name: "main".to_string(),
645                }),
646            },
647            mime_type: Some("text/x-rust".to_string()),
648            chunk_index: 0,
649            byte_range: 0..12,
650            line_range: Some(0..1),
651            parent_chunk_id: None,
652            depth: 0,
653            embedding: None,
654            dir_path: "test".to_string(),
655            dir_depth: 1,
656            path_components: "test,file.rs".to_string(),
657            metadata: ChunkMetadata::default(),
658        };
659
660        let json = serde_json::to_string(&chunk).unwrap();
661        let deserialized: Chunk = serde_json::from_str(&json).unwrap();
662
663        assert_eq!(chunk.id, deserialized.id);
664        assert_eq!(chunk.content, deserialized.content);
665    }
666
667    #[test]
668    fn test_content_type_text() {
669        let ct = ContentType::Text;
670        let json = serde_json::to_string(&ct).unwrap();
671        assert!(json.contains("\"type\":\"text\""));
672    }
673
674    #[test]
675    fn test_content_type_code() {
676        let ct = ContentType::Code {
677            language: "python".to_string(),
678            symbol: None,
679        };
680        let json = serde_json::to_string(&ct).unwrap();
681        assert!(json.contains("\"type\":\"code\""));
682        assert!(json.contains("\"language\":\"python\""));
683    }
684
685    #[test]
686    fn test_content_type_pdf_page() {
687        let ct = ContentType::PdfPage { page_num: 5 };
688        let json = serde_json::to_string(&ct).unwrap();
689        assert!(json.contains("\"type\":\"pdf_page\""));
690        assert!(json.contains("\"page_num\":5"));
691    }
692
693    #[test]
694    fn test_content_type_markdown() {
695        let ct = ContentType::Markdown;
696        let json = serde_json::to_string(&ct).unwrap();
697        assert!(json.contains("\"type\":\"markdown\""));
698    }
699
700    #[test]
701    fn test_symbol_kind_serialization() {
702        assert_eq!(
703            serde_json::to_string(&SymbolKind::Function).unwrap(),
704            "\"function\""
705        );
706        assert_eq!(
707            serde_json::to_string(&SymbolKind::Struct).unwrap(),
708            "\"struct\""
709        );
710        assert_eq!(
711            serde_json::to_string(&SymbolKind::Trait).unwrap(),
712            "\"trait\""
713        );
714    }
715
716    // ==================== ChunkConfig Tests ====================
717
718    #[test]
719    fn test_chunk_config_default() {
720        let config = ChunkConfig::default();
721        assert_eq!(config.target_size, 512);
722        assert_eq!(config.max_size, 1024);
723        assert_eq!(config.overlap, 64);
724        assert!(config.hierarchical);
725        assert_eq!(config.max_depth, 2);
726    }
727
728    #[test]
729    fn test_chunk_config_serialization() {
730        let config = ChunkConfig::default();
731        let json = serde_json::to_string(&config).unwrap();
732        let deserialized: ChunkConfig = serde_json::from_str(&json).unwrap();
733
734        assert_eq!(config.target_size, deserialized.target_size);
735        assert_eq!(config.max_size, deserialized.max_size);
736    }
737
738    // ==================== EmbeddingConfig Tests ====================
739
740    #[test]
741    fn test_embedding_config_default() {
742        let config = EmbeddingConfig::default();
743        assert!(config.normalize);
744        assert!(config.instruction.is_none());
745        assert_eq!(config.batch_size, 32);
746    }
747
748    #[test]
749    fn test_embedding_config_serialization() {
750        let config = EmbeddingConfig {
751            normalize: false,
752            instruction: Some("Search: ".to_string()),
753            batch_size: 16,
754        };
755        let json = serde_json::to_string(&config).unwrap();
756        let deserialized: EmbeddingConfig = serde_json::from_str(&json).unwrap();
757
758        assert_eq!(config.normalize, deserialized.normalize);
759        assert_eq!(config.instruction, deserialized.instruction);
760        assert_eq!(config.batch_size, deserialized.batch_size);
761    }
762
763    // ==================== Modality Tests ====================
764
765    #[test]
766    fn test_modality_serialization() {
767        assert_eq!(serde_json::to_string(&Modality::Text).unwrap(), "\"text\"");
768        assert_eq!(
769            serde_json::to_string(&Modality::Image).unwrap(),
770            "\"image\""
771        );
772        assert_eq!(
773            serde_json::to_string(&Modality::Audio).unwrap(),
774            "\"audio\""
775        );
776    }
777
778    #[test]
779    fn test_modality_equality() {
780        assert_eq!(Modality::Text, Modality::Text);
781        assert_ne!(Modality::Text, Modality::Image);
782    }
783
784    // ==================== DistanceMetric Tests ====================
785
786    #[test]
787    fn test_distance_metric_default() {
788        let metric = DistanceMetric::default();
789        assert_eq!(metric, DistanceMetric::Cosine);
790    }
791
792    #[test]
793    fn test_distance_metric_serialization() {
794        assert_eq!(
795            serde_json::to_string(&DistanceMetric::Cosine).unwrap(),
796            "\"cosine\""
797        );
798        assert_eq!(
799            serde_json::to_string(&DistanceMetric::L2).unwrap(),
800            "\"l2\""
801        );
802        assert_eq!(
803            serde_json::to_string(&DistanceMetric::Dot).unwrap(),
804            "\"dot\""
805        );
806    }
807
808    // ==================== SearchResult Tests ====================
809
810    #[test]
811    fn test_search_result_serialization() {
812        let result = SearchResult {
813            chunk_id: Uuid::new_v4(),
814            file_path: PathBuf::from("/test/file.txt"),
815            content: "Test content".to_string(),
816            score: 0.95,
817            byte_range: 0..12,
818            line_range: Some(0..1),
819            metadata: HashMap::new(),
820        };
821
822        let json = serde_json::to_string(&result).unwrap();
823        let deserialized: SearchResult = serde_json::from_str(&json).unwrap();
824
825        assert_eq!(result.chunk_id, deserialized.chunk_id);
826        assert_eq!(result.score, deserialized.score);
827        assert_eq!(result.content, deserialized.content);
828    }
829
830    // ==================== StoreStats Tests ====================
831
832    #[test]
833    fn test_store_stats_serialization() {
834        let stats = StoreStats {
835            total_chunks: 100,
836            total_files: 10,
837            index_size_bytes: 1024 * 1024,
838            last_updated: Some(Utc::now()),
839        };
840
841        let json = serde_json::to_string(&stats).unwrap();
842        let deserialized: StoreStats = serde_json::from_str(&json).unwrap();
843
844        assert_eq!(stats.total_chunks, deserialized.total_chunks);
845        assert_eq!(stats.total_files, deserialized.total_files);
846    }
847
848    // ==================== IndexStats Tests ====================
849
850    #[test]
851    fn test_index_stats_default() {
852        let stats = IndexStats::default();
853        assert_eq!(stats.total_files, 0);
854        assert_eq!(stats.indexed_files, 0);
855        assert_eq!(stats.pending_files, 0);
856        assert_eq!(stats.error_files, 0);
857        assert_eq!(stats.total_chunks, 0);
858        assert!(stats.last_update.is_none());
859    }
860
861    #[test]
862    fn test_index_stats_serialization() {
863        let stats = IndexStats {
864            total_files: 50,
865            indexed_files: 45,
866            pending_files: 3,
867            error_files: 2,
868            total_chunks: 500,
869            last_update: Some(Utc::now()),
870        };
871
872        let json = serde_json::to_string(&stats).unwrap();
873        let deserialized: IndexStats = serde_json::from_str(&json).unwrap();
874
875        assert_eq!(stats.total_files, deserialized.total_files);
876        assert_eq!(stats.indexed_files, deserialized.indexed_files);
877    }
878
879    // ==================== ChunkMetadata Tests ====================
880
881    #[test]
882    fn test_chunk_metadata_default() {
883        let meta = ChunkMetadata::default();
884        assert!(meta.embedding_model.is_none());
885        assert!(meta.indexed_at.is_none());
886        assert!(meta.token_count.is_none());
887        assert!(meta.extra.is_empty());
888    }
889
890    // ==================== ChunkOutput Tests ====================
891
892    #[test]
893    fn test_chunk_output_metadata_default() {
894        let meta = ChunkOutputMetadata::default();
895        assert!(meta.symbol_type.is_none());
896        assert!(meta.symbol_name.is_none());
897        assert!(meta.language.is_none());
898    }
899
900    // ==================== ContentMetadataInfo Tests ====================
901
902    #[test]
903    fn test_content_metadata_info_default() {
904        let meta = ContentMetadataInfo::default();
905        assert!(meta.title.is_none());
906        assert!(meta.author.is_none());
907        assert!(meta.language.is_none());
908        assert!(meta.page_count.is_none());
909        assert!(meta.created_at.is_none());
910    }
911
912    // ==================== DirectoryScope Tests ====================
913
914    #[test]
915    fn test_directory_scope_is_relative_to_root() {
916        let root = PathBuf::from("/project");
917        let file = PathBuf::from("/project/src/auth/login.rs");
918        let scope = DirectoryScope::from_paths(&file, Some(&root));
919
920        assert_eq!(scope.dir_path, "src/auth");
921        assert_eq!(scope.dir_depth, 2);
922        assert_eq!(scope.path_components, "src,auth,login.rs");
923    }
924
925    #[test]
926    fn test_directory_scope_index_root_file() {
927        let root = PathBuf::from("/project");
928        let file = PathBuf::from("/project/README.md");
929        let scope = DirectoryScope::from_paths(&file, Some(&root));
930
931        assert_eq!(scope.dir_path, "");
932        assert_eq!(scope.dir_depth, 0);
933        assert_eq!(scope.path_components, "README.md");
934    }
935
936    #[test]
937    fn test_directory_scope_without_root_strips_only_prefix_components() {
938        let file = PathBuf::from("/test/file.rs");
939        let scope = DirectoryScope::from_file_path(&file);
940        assert_eq!(scope.dir_path, "test");
941        assert_eq!(scope.path_components, "test,file.rs");
942    }
943
944    #[test]
945    fn test_dir_path_matches_scope_exact_and_subdirectory() {
946        assert!(dir_path_matches_scope("src/auth", "src/auth"));
947        assert!(dir_path_matches_scope("src/auth/oauth", "src/auth"));
948        assert!(dir_path_matches_scope("src/auth", "src/auth/"));
949        assert!(!dir_path_matches_scope("src/other", "src/auth"));
950        assert!(!dir_path_matches_scope("src/auth-backup", "src/auth"));
951        assert!(dir_path_matches_scope("src", ""));
952        assert!(dir_path_matches_scope("src", "."));
953    }
954
955    #[test]
956    fn test_dir_path_matches_scope_migrated_absolute_suffix() {
957        // from_file_path("/project/src/auth/login.rs") → "project/src/auth"
958        assert!(dir_path_matches_scope("project/src/auth", "src/auth"));
959        assert!(dir_path_matches_scope("project/src/auth/oauth", "src/auth"));
960        assert!(dir_path_matches_scope("project/src", "src"));
961        assert!(dir_path_matches_scope("project/src/auth", "src"));
962        assert!(!dir_path_matches_scope("project/src-backup", "src"));
963        assert!(!dir_path_matches_scope("project/other", "src/auth"));
964        assert!(!dir_path_matches_scope("project/author/notes", "auth"));
965    }
966
967    // ==================== FileEvent Tests ====================
968
969    #[test]
970    fn test_file_event_created() {
971        let event = FileEvent::Created(PathBuf::from("/test/new.txt"));
972        match event {
973            FileEvent::Created(path) => assert_eq!(path, PathBuf::from("/test/new.txt")),
974            _ => panic!("Expected Created event"),
975        }
976    }
977
978    #[test]
979    fn test_file_event_modified() {
980        let event = FileEvent::Modified(PathBuf::from("/test/changed.txt"));
981        match event {
982            FileEvent::Modified(path) => assert_eq!(path, PathBuf::from("/test/changed.txt")),
983            _ => panic!("Expected Modified event"),
984        }
985    }
986
987    #[test]
988    fn test_file_event_deleted() {
989        let event = FileEvent::Deleted(PathBuf::from("/test/removed.txt"));
990        match event {
991            FileEvent::Deleted(path) => assert_eq!(path, PathBuf::from("/test/removed.txt")),
992            _ => panic!("Expected Deleted event"),
993        }
994    }
995
996    #[test]
997    fn test_file_event_renamed() {
998        let event = FileEvent::Renamed {
999            from: PathBuf::from("/test/old.txt"),
1000            to: PathBuf::from("/test/new.txt"),
1001        };
1002        match event {
1003            FileEvent::Renamed { from, to } => {
1004                assert_eq!(from, PathBuf::from("/test/old.txt"));
1005                assert_eq!(to, PathBuf::from("/test/new.txt"));
1006            }
1007            _ => panic!("Expected Renamed event"),
1008        }
1009    }
1010}