fix: detect and report Ollama embedding dimension inconsistency

- Add validation for embedding dimension consistency in Ollama mode - Provide clear error message with troubleshooting steps when dimensions mismatch - Fail fast instead of silent fallback to prevent data corruption Fixes #31
2025-08-11 17:36:44 -07:00
5 changed files with 222 additions and 403 deletions
@@ -468,7 +468,7 @@ leann --help
 ### Usage Examples
 ```bash
-# build from a specific directory, and my_docs is the index name(Here you can also build from multiple dict or multiple files)
+# build from a specific directory, and my_docs is the index name
 leann build my-docs --docs ./your_documents
 # Search your documents
@@ -5,7 +5,6 @@ from typing import Union
 from llama_index.core import SimpleDirectoryReader
 from llama_index.core.node_parser import SentenceSplitter
 from tqdm import tqdm
 from .api import LeannBuilder, LeannChat, LeannSearcher
@@ -76,10 +75,7 @@ class LeannCLI:
            formatter_class=argparse.RawDescriptionHelpFormatter,
            epilog="""
 Examples:
-  leann build my-docs --docs ./documents                                  # Build index from directory
+  leann build my-docs --docs ./documents                    # Build index named my-docs
  leann build my-code --docs ./src ./tests ./config                      # Build index from multiple directories
  leann build my-files --docs ./file1.py ./file2.txt ./docs/             # Build index from files and directories
  leann build my-mixed --docs ./readme.md ./src/ ./config.json           # Build index from mixed files/dirs
  leann build my-ppts --docs ./ --file-types .pptx,.pdf    # Index only PowerPoint and PDF files
  leann search my-docs "query"                             # Search in my-docs index
  leann ask my-docs "question"                             # Ask my-docs index
@@ -95,11 +91,7 @@ Examples:
            "index_name", nargs="?", help="Index name (default: current directory name)"
        )
        build_parser.add_argument(
-            "--docs",
+            "--docs", type=str, default=".", help="Documents directory (default: current directory)"
            type=str,
            nargs="+",
            default=["."],
            help="Documents directories and/or files (default: current directory)",
        )
        build_parser.add_argument(
            "--backend", type=str, default="hnsw", choices=["hnsw", "diskann"]
@@ -243,32 +235,6 @@ Examples:
        """Check if a file should be excluded using gitignore parser."""
        return gitignore_matches(str(relative_path))
    def _is_git_submodule(self, path: Path) -> bool:
        """Check if a path is a git submodule."""
        try:
            # Find the git repo root
            current_dir = Path.cwd()
            while current_dir != current_dir.parent:
                if (current_dir / ".git").exists():
                    gitmodules_path = current_dir / ".gitmodules"
                    if gitmodules_path.exists():
                        # Read .gitmodules to check if this path is a submodule
                        gitmodules_content = gitmodules_path.read_text()
                        # Convert path to relative to git root
                        try:
                            relative_path = path.resolve().relative_to(current_dir)
                            # Check if this path appears in .gitmodules
                            return f"path = {relative_path}" in gitmodules_content
                        except ValueError:
                            # Path is not under git root
                            return False
                    break
                current_dir = current_dir.parent
            return False
        except Exception:
            # If anything goes wrong, assume it's not a submodule
            return False
    def list_indexes(self):
        print("Stored LEANN indexes:")
@@ -298,9 +264,7 @@ Examples:
            valid_projects.append(current_path)
        if not valid_projects:
-            print(
+            print("No indexes found. Use 'leann build <name> --docs <dir>' to create one.")
                "No indexes found. Use 'leann build <name> --docs <dir> [<dir2> ...]' to create one."
            )
            return
        total_indexes = 0
@@ -347,88 +311,56 @@ Examples:
                    print(f'  leann search {example_name} "your query"')
                    print(f"  leann ask {example_name} --interactive")
-    def load_documents(
+    def load_documents(self, docs_dir: str, custom_file_types: Union[str, None] = None):
-        self, docs_paths: Union[str, list], custom_file_types: Union[str, None] = None
+        print(f"Loading documents from {docs_dir}...")
    ):
        # Handle both single path (string) and multiple paths (list) for backward compatibility
        if isinstance(docs_paths, str):
            docs_paths = [docs_paths]
        # Separate files and directories
        files = []
        directories = []
        for path in docs_paths:
            path_obj = Path(path)
            if path_obj.is_file():
                files.append(str(path_obj))
            elif path_obj.is_dir():
                # Check if this is a git submodule - if so, skip it
                if self._is_git_submodule(path_obj):
                    print(f"⚠️  Skipping git submodule: {path}")
                    continue
                directories.append(str(path_obj))
            else:
                print(f"⚠️  Warning: Path '{path}' does not exist, skipping...")
                continue
        # Print summary of what we're processing
        total_items = len(files) + len(directories)
        items_desc = []
        if files:
            items_desc.append(f"{len(files)} file{'s' if len(files) > 1 else ''}")
        if directories:
            items_desc.append(
                f"{len(directories)} director{'ies' if len(directories) > 1 else 'y'}"
            )
        print(f"Loading documents from {' and '.join(items_desc)} ({total_items} total):")
        if files:
            print(f"  📄 Files: {', '.join([Path(f).name for f in files])}")
        if directories:
            print(f"  📁 Directories: {', '.join(directories)}")
        if custom_file_types:
            print(f"Using custom file types: {custom_file_types}")
-        all_documents = []
+        # Build gitignore parser
        gitignore_matches = self._build_gitignore_parser(docs_dir)
-        # First, process individual files if any
+        # Try to use better PDF parsers first, but only if PDFs are requested
-        if files:
+        documents = []
-            print(f"\n🔄 Processing {len(files)} individual file{'s' if len(files) > 1 else ''}...")
+        docs_path = Path(docs_dir)
-            # Load individual files using SimpleDirectoryReader with input_files
+        # Check if we should process PDFs
-            # Note: We skip gitignore filtering for explicitly specified files
+        should_process_pdfs = custom_file_types is None or ".pdf" in custom_file_types
        if should_process_pdfs:
            for file_path in docs_path.rglob("*.pdf"):
                # Check if file matches any exclude pattern
                relative_path = file_path.relative_to(docs_path)
                if self._should_exclude_file(relative_path, gitignore_matches):
                    continue
                print(f"Processing PDF: {file_path}")
                # Try PyMuPDF first (best quality)
                text = extract_pdf_text_with_pymupdf(str(file_path))
                if text is None:
                    # Try pdfplumber
                    text = extract_pdf_text_with_pdfplumber(str(file_path))
                if text:
                    # Create a simple document structure
                    from llama_index.core import Document
                    doc = Document(text=text, metadata={"source": str(file_path)})
                    documents.append(doc)
                else:
                    # Fallback to default reader
                    print(f"Using default reader for {file_path}")
                    try:
-                # Group files by their parent directory for efficient loading
+                        default_docs = SimpleDirectoryReader(
-                from collections import defaultdict
+                            str(file_path.parent),
                files_by_dir = defaultdict(list)
                for file_path in files:
                    parent_dir = str(Path(file_path).parent)
                    files_by_dir[parent_dir].append(file_path)
                # Load files from each parent directory
                for parent_dir, file_list in files_by_dir.items():
                    print(
                        f"  Loading {len(file_list)} file{'s' if len(file_list) > 1 else ''} from {parent_dir}"
                    )
                    try:
                        file_docs = SimpleDirectoryReader(
                            parent_dir,
                            input_files=file_list,
                            filename_as_id=True,
                            required_exts=[file_path.suffix],
                        ).load_data()
-                        all_documents.extend(file_docs)
+                        documents.extend(default_docs)
                        print(
                            f"    ✅ Loaded {len(file_docs)} document{'s' if len(file_docs) > 1 else ''}"
                        )
                    except Exception as e:
-                        print(f"    ❌ Warning: Could not load files from {parent_dir}: {e}")
+                        print(f"Warning: Could not process {file_path}: {e}")
-            except Exception as e:
+        # Load other file types with default reader
                print(f"❌ Error processing individual files: {e}")
        # Define file extensions to process
        if custom_file_types:
            # Parse custom file types from comma-separated string
            code_extensions = [ext.strip() for ext in custom_file_types.split(",") if ext.strip()]
@@ -490,70 +422,10 @@ Examples:
                ".py",
                ".jl",
            ]
-
+        # Try to load other file types, but don't fail if none are found
        # Process each directory
        if directories:
            print(
                f"\n🔄 Processing {len(directories)} director{'ies' if len(directories) > 1 else 'y'}..."
            )
        for docs_dir in directories:
            print(f"Processing directory: {docs_dir}")
            # Build gitignore parser for each directory
            gitignore_matches = self._build_gitignore_parser(docs_dir)
            # Try to use better PDF parsers first, but only if PDFs are requested
            documents = []
            docs_path = Path(docs_dir)
            # Check if we should process PDFs
            should_process_pdfs = custom_file_types is None or ".pdf" in custom_file_types
            if should_process_pdfs:
                for file_path in docs_path.rglob("*.pdf"):
                    # Check if file matches any exclude pattern
                    try:
                        relative_path = file_path.relative_to(docs_path)
                        if self._should_exclude_file(relative_path, gitignore_matches):
                            continue
                    except ValueError:
                        # Skip files that can't be made relative to docs_path
                        print(f"⚠️  Skipping file outside directory scope: {file_path}")
                        continue
                    print(f"Processing PDF: {file_path}")
                    # Try PyMuPDF first (best quality)
                    text = extract_pdf_text_with_pymupdf(str(file_path))
                    if text is None:
                        # Try pdfplumber
                        text = extract_pdf_text_with_pdfplumber(str(file_path))
                    if text:
                        # Create a simple document structure
                        from llama_index.core import Document
                        doc = Document(text=text, metadata={"source": str(file_path)})
                        documents.append(doc)
                    else:
                        # Fallback to default reader
                        print(f"Using default reader for {file_path}")
                        try:
                            default_docs = SimpleDirectoryReader(
                                str(file_path.parent),
                                filename_as_id=True,
                                required_exts=[file_path.suffix],
                            ).load_data()
                            documents.extend(default_docs)
                        except Exception as e:
                            print(f"Warning: Could not process {file_path}: {e}")
            # Load other file types with default reader
        try:
            # Create a custom file filter function using our PathSpec
-                def file_filter(
+            def file_filter(file_path: str) -> bool:
                    file_path: str, docs_dir=docs_dir, gitignore_matches=gitignore_matches
                ) -> bool:
                """Return True if file should be included (not excluded)"""
                try:
                    docs_path_obj = Path(docs_dir)
@@ -582,15 +454,10 @@ Examples:
            documents.extend(filtered_docs)
        except ValueError as e:
            if "No files found" in str(e):
-                    print(f"No additional files found for other supported types in {docs_dir}.")
+                print("No additional files found for other supported types.")
            else:
                raise e
            all_documents.extend(documents)
            print(f"Loaded {len(documents)} documents from {docs_dir}")
        documents = all_documents
        all_texts = []
        # Define code file extensions for intelligent chunking
@@ -640,9 +507,7 @@ Examples:
            ".jl",
        }
-        print("start chunking documents")
+        for doc in documents:
        # Add progress bar for document chunking
        for doc in tqdm(documents, desc="Chunking documents", unit="doc"):
            # Check if this is a code file based on source path
            source_path = doc.metadata.get("source", "")
            is_code_file = any(source_path.endswith(ext) for ext in code_file_exts)
@@ -658,7 +523,7 @@ Examples:
        return all_texts
    async def build_index(self, args):
-        docs_paths = args.docs
+        docs_dir = args.docs
        # Use current directory name if index_name not provided
        if args.index_name:
            index_name = args.index_name
@@ -669,25 +534,13 @@ Examples:
        index_dir = self.indexes_dir / index_name
        index_path = self.get_index_path(index_name)
-        # Display all paths being indexed with file/directory distinction
+        print(f"📂 Indexing: {Path(docs_dir).resolve()}")
        files = [p for p in docs_paths if Path(p).is_file()]
        directories = [p for p in docs_paths if Path(p).is_dir()]
        print(f"📂 Indexing {len(docs_paths)} path{'s' if len(docs_paths) > 1 else ''}:")
        if files:
            print(f"  📄 Files ({len(files)}):")
            for i, file_path in enumerate(files, 1):
                print(f"    {i}. {Path(file_path).resolve()}")
        if directories:
            print(f"  📁 Directories ({len(directories)}):")
            for i, dir_path in enumerate(directories, 1):
                print(f"    {i}. {Path(dir_path).resolve()}")
        if index_dir.exists() and not args.force:
            print(f"Index '{index_name}' already exists. Use --force to rebuild.")
            return
-        all_texts = self.load_documents(docs_paths, args.file_types)
+        all_texts = self.load_documents(docs_dir, args.file_types)
        if not all_texts:
            print("No documents found")
            return
@@ -723,7 +576,7 @@ Examples:
        if not self.index_exists(index_name):
            print(
-                f"Index '{index_name}' not found. Use 'leann build {index_name} --docs <dir> [<dir2> ...]' to create it."
+                f"Index '{index_name}' not found. Use 'leann build {index_name} --docs <dir>' to create it."
            )
            return
@@ -750,7 +603,7 @@ Examples:
        if not self.index_exists(index_name):
            print(
-                f"Index '{index_name}' not found. Use 'leann build {index_name} --docs <dir> [<dir2> ...]' to create it."
+                f"Index '{index_name}' not found. Use 'leann build {index_name} --docs <dir>' to create it."
            )
            return
@@ -6,6 +6,7 @@ Preserves all optimization parameters to ensure performance
 import logging
 import os
 from concurrent.futures import ThreadPoolExecutor, as_completed
 from typing import Any
 import numpy as np
@@ -373,9 +374,7 @@ def compute_embeddings_ollama(
    texts: list[str], model_name: str, is_build: bool = False, host: str = "http://localhost:11434"
 ) -> np.ndarray:
    """
-    Compute embeddings using Ollama API with simplified batch processing.
+    Compute embeddings using Ollama API.
    Uses batch size of 32 for MPS/CPU and 128 for CUDA to optimize performance.
    Args:
        texts: List of texts to compute embeddings for
@@ -439,19 +438,12 @@ def compute_embeddings_ollama(
            if any(emb in base_name for emb in ["embed", "bge", "minilm", "e5"]):
                embedding_models.append(model)
-        # Check if model exists (handle versioned names) and resolve to full name
+        # Check if model exists (handle versioned names)
-        resolved_model_name = None
+        model_found = any(
-        for name in model_names:
+            model_name == name.split(":")[0] or model_name == name for name in model_names
-            # Exact match
+        )
            if model_name == name:
                resolved_model_name = name
                break
            # Match without version tag (use the versioned name)
            elif model_name == name.split(":")[0]:
                resolved_model_name = name
                break
-        if not resolved_model_name:
+        if not model_found:
            error_msg = f"❌ Model '{model_name}' not found in local Ollama.\n\n"
            # Suggest pulling the model
@@ -473,11 +465,6 @@ def compute_embeddings_ollama(
            error_msg += "\n📚 Browse more: https://ollama.com/library"
            raise ValueError(error_msg)
        # Use the resolved model name for all subsequent operations
        if resolved_model_name != model_name:
            logger.info(f"Resolved model name '{model_name}' to '{resolved_model_name}'")
        model_name = resolved_model_name
        # Verify the model supports embeddings by testing it
        try:
            test_response = requests.post(
@@ -498,33 +485,18 @@ def compute_embeddings_ollama(
    except requests.exceptions.RequestException as e:
        logger.warning(f"Could not verify model existence: {e}")
-    # Determine batch size based on device availability
+    # Process embeddings with optimized concurrent processing
-    # Check for CUDA/MPS availability using torch if available
+    import requests
    batch_size = 32  # Default for MPS/CPU
    try:
        import torch
-        if torch.cuda.is_available():
+    def get_single_embedding(text_idx_tuple):
-            batch_size = 128  # CUDA gets larger batch size
+        """Helper function to get embedding for a single text."""
-        elif hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
+        text, idx = text_idx_tuple
            batch_size = 32  # MPS gets smaller batch size
    except ImportError:
        # If torch is not available, use conservative batch size
        batch_size = 32
    logger.info(f"Using batch size: {batch_size}")
    def get_batch_embeddings(batch_texts):
        """Get embeddings for a batch of texts."""
        all_embeddings = []
        failed_indices = []
        for i, text in enumerate(batch_texts):
        max_retries = 3
        retry_count = 0
        # Truncate very long texts to avoid API issues
        truncated_text = text[:8000] if len(text) > 8000 else text
        while retry_count < max_retries:
            try:
                response = requests.post(
@@ -538,87 +510,115 @@ def compute_embeddings_ollama(
                embedding = result.get("embedding")
                if embedding is None:
-                        raise ValueError(f"No embedding returned for text {i}")
+                    raise ValueError(f"No embedding returned for text {idx}")
-                    if not isinstance(embedding, list) or len(embedding) == 0:
+                return idx, embedding
                        raise ValueError(f"Invalid embedding format for text {i}")
                    all_embeddings.append(embedding)
                    break
            except requests.exceptions.Timeout:
                retry_count += 1
                if retry_count >= max_retries:
-                        logger.warning(f"Timeout for text {i} after {max_retries} retries")
+                    logger.warning(f"Timeout for text {idx} after {max_retries} retries")
-                        failed_indices.append(i)
+                    return idx, None
                        all_embeddings.append(None)
                        break
            except Exception as e:
                if retry_count >= max_retries - 1:
                    logger.error(f"Failed to get embedding for text {idx}: {e}")
                    return idx, None
                retry_count += 1
                    if retry_count >= max_retries:
                        logger.error(f"Failed to get embedding for text {i}: {e}")
                        failed_indices.append(i)
                        all_embeddings.append(None)
                        break
        return all_embeddings, failed_indices
-    # Process texts in batches
+        return idx, None
    all_embeddings = []
    all_failed_indices = []
-    # Setup progress bar if needed
+    # Determine if we should use concurrent processing
    use_concurrent = (
        len(texts) > 5 and not is_build
    )  # Don't use concurrent in build mode to avoid overwhelming
    max_workers = min(4, len(texts))  # Limit concurrent requests to avoid overwhelming Ollama
    all_embeddings = [None] * len(texts)  # Pre-allocate list to maintain order
    failed_indices = []
    if use_concurrent:
        logger.info(
            f"Using concurrent processing with {max_workers} workers for {len(texts)} texts"
        )
        with ThreadPoolExecutor(max_workers=max_workers) as executor:
            # Submit all tasks
            future_to_idx = {
                executor.submit(get_single_embedding, (text, idx)): idx
                for idx, text in enumerate(texts)
            }
            # Add progress bar for concurrent processing
            try:
                if is_build or len(texts) > 10:
                    from tqdm import tqdm
                    futures_iterator = tqdm(
                        as_completed(future_to_idx),
                        total=len(texts),
                        desc="Computing Ollama embeddings",
                    )
                else:
                    futures_iterator = as_completed(future_to_idx)
            except ImportError:
                futures_iterator = as_completed(future_to_idx)
            # Collect results as they complete
            for future in futures_iterator:
                try:
                    idx, embedding = future.result()
                    if embedding is not None:
                        all_embeddings[idx] = embedding
                    else:
                        failed_indices.append(idx)
                except Exception as e:
                    idx = future_to_idx[future]
                    logger.error(f"Exception for text {idx}: {e}")
                    failed_indices.append(idx)
    else:
        # Sequential processing with progress bar
        show_progress = is_build or len(texts) > 10
        try:
            if show_progress:
                from tqdm import tqdm
    except ImportError:
        show_progress = False
-    # Process batches
+                iterator = tqdm(
-    num_batches = (len(texts) + batch_size - 1) // batch_size
+                    enumerate(texts), total=len(texts), desc="Computing Ollama embeddings"
-
+                )
    if show_progress:
        batch_iterator = tqdm(range(num_batches), desc="Computing Ollama embeddings")
            else:
-        batch_iterator = range(num_batches)
+                iterator = enumerate(texts)
        except ImportError:
            iterator = enumerate(texts)
-    for batch_idx in batch_iterator:
+        for idx, text in iterator:
-        start_idx = batch_idx * batch_size
+            result_idx, embedding = get_single_embedding((text, idx))
-        end_idx = min(start_idx + batch_size, len(texts))
+            if embedding is not None:
-        batch_texts = texts[start_idx:end_idx]
+                all_embeddings[idx] = embedding
-
+            else:
-        batch_embeddings, batch_failed = get_batch_embeddings(batch_texts)
+                failed_indices.append(idx)
        # Adjust failed indices to global indices
        global_failed = [start_idx + idx for idx in batch_failed]
        all_failed_indices.extend(global_failed)
        all_embeddings.extend(batch_embeddings)
    # Handle failed embeddings
-    if all_failed_indices:
+    if failed_indices:
-        if len(all_failed_indices) == len(texts):
+        if len(failed_indices) == len(texts):
            raise RuntimeError("Failed to compute any embeddings")
-        logger.warning(
+        logger.warning(f"Failed to compute embeddings for {len(failed_indices)}/{len(texts)} texts")
            f"Failed to compute embeddings for {len(all_failed_indices)}/{len(texts)} texts"
        )
        # Use zero embeddings as fallback for failed ones
        valid_embedding = next((e for e in all_embeddings if e is not None), None)
        if valid_embedding:
            embedding_dim = len(valid_embedding)
-            for i, embedding in enumerate(all_embeddings):
+            for idx in failed_indices:
-                if embedding is None:
+                all_embeddings[idx] = [0.0] * embedding_dim
                    all_embeddings[i] = [0.0] * embedding_dim
-    # Remove None values
+    # Remove None values and convert to numpy array
    all_embeddings = [e for e in all_embeddings if e is not None]
-    if not all_embeddings:
+    # Validate embedding dimensions before creating numpy array
-        raise RuntimeError("No valid embeddings were computed")
+    if all_embeddings:
    # Validate embedding dimensions
        expected_dim = len(all_embeddings[0])
        inconsistent_dims = []
        for i, embedding in enumerate(all_embeddings):
@@ -631,7 +631,9 @@ def compute_embeddings_ollama(
                error_msg += f"  - Text {idx}: {dim} dimensions\n"
            if len(inconsistent_dims) > 10:
                error_msg += f"  ... and {len(inconsistent_dims) - 10} more\n"
-        error_msg += f"\nThis is likely an Ollama API bug with model '{model_name}'. Please try:\n"
+            error_msg += (
                f"\nThis is likely an Ollama API bug with model '{model_name}'. Please try:\n"
            )
            error_msg += "1. Restart Ollama service: 'ollama serve'\n"
            error_msg += f"2. Re-pull the model: 'ollama pull {model_name}'\n"
            error_msg += (
@@ -45,42 +45,6 @@ leann build my-project --docs ./
 claude
 ```
 ## 🚀 Advanced Usage Examples
 ### Index Entire Git Repository
 ```bash
 # Index all tracked files in your git repository, note right now we will skip submodules, but we can add it back easily if you want
 leann build my-repo --docs $(git ls-files) --embedding-mode sentence-transformers --embedding-model all-MiniLM-L6-v2 --backend hnsw
 # Index only specific file types from git
 leann build my-python-code --docs $(git ls-files "*.py") --embedding-mode sentence-transformers --embedding-model all-MiniLM-L6-v2 --backend hnsw
 ```
 ### Multiple Directories and Files
 ```bash
 # Index multiple directories
 leann build my-codebase --docs ./src ./tests ./docs ./config --embedding-mode sentence-transformers --embedding-model all-MiniLM-L6-v2 --backend hnsw
 # Mix files and directories
 leann build my-project --docs ./README.md ./src/ ./package.json ./docs/ --embedding-mode sentence-transformers --embedding-model all-MiniLM-L6-v2 --backend hnsw
 # Specific files only
 leann build my-configs --docs ./tsconfig.json ./package.json ./webpack.config.js --embedding-mode sentence-transformers --embedding-model all-MiniLM-L6-v2 --backend hnsw
 ```
 ### Advanced Git Integration
 ```bash
 # Index recently modified files
 leann build recent-changes --docs $(git diff --name-only HEAD~10..HEAD) --embedding-mode sentence-transformers --embedding-model all-MiniLM-L6-v2 --backend hnsw
 # Index files matching pattern
 leann build frontend --docs $(git ls-files "*.tsx" "*.ts" "*.jsx" "*.js") --embedding-mode sentence-transformers --embedding-model all-MiniLM-L6-v2 --backend hnsw
 # Index documentation and config files
 leann build docs-and-configs --docs $(git ls-files "*.md" "*.yml" "*.yaml" "*.json" "*.toml") --embedding-mode sentence-transformers --embedding-model all-MiniLM-L6-v2 --backend hnsw
 ```
 **Try this in Claude Code:**
 ```
 Help me understand this codebase. List available indexes and search for authentication patterns.