Merge remote-tracking branch 'origin/main' into financebench

2025-09-15 19:52:37 -07:00
parent 3578680cb6 3b8dc6368e
commit b1daf021e0
22 changed files with 519 additions and 43 deletions
--- a/packages/astchunk-leann
+++ b/packages/astchunk-leann
--- a/packages/leann-backend-diskann/pyproject.toml
+++ b/packages/leann-backend-diskann/pyproject.toml
@@ -4,8 +4,8 @@ build-backend = "scikit_build_core.build"

 [project]
 name = "leann-backend-diskann"
-version = "0.3.2"
-dependencies = ["leann-core==0.3.2", "numpy", "protobuf>=3.19.0"]
+version = "0.3.3"
+dependencies = ["leann-core==0.3.3", "numpy", "protobuf>=3.19.0"]

 [tool.scikit-build]
 # Key: simplified CMake path
--- a/packages/leann-backend-diskann/third_party/DiskANN
+++ b/packages/leann-backend-diskann/third_party/DiskANN
--- a/packages/leann-backend-hnsw/CMakeLists.txt
+++ b/packages/leann-backend-hnsw/CMakeLists.txt
@@ -49,9 +49,28 @@ set(BUILD_TESTING OFF CACHE BOOL "" FORCE)
 set(FAISS_ENABLE_C_API OFF CACHE BOOL "" FORCE)
 set(FAISS_OPT_LEVEL "generic" CACHE STRING "" FORCE)

-# Disable additional SIMD versions to speed up compilation
+# Disable x86-specific SIMD optimizations (important for ARM64 compatibility)
 set(FAISS_ENABLE_AVX2 OFF CACHE BOOL "" FORCE)
 set(FAISS_ENABLE_AVX512 OFF CACHE BOOL "" FORCE)
+set(FAISS_ENABLE_SSE4_1 OFF CACHE BOOL "" FORCE)
+
+# ARM64-specific configuration
+if(CMAKE_SYSTEM_PROCESSOR MATCHES "aarch64|arm64")
+    message(STATUS "Configuring Faiss for ARM64 architecture")
+
+    if(CMAKE_SYSTEM_NAME STREQUAL "Linux")
+        # Use SVE optimization level for ARM64 Linux (as seen in Faiss conda build)
+        set(FAISS_OPT_LEVEL "sve" CACHE STRING "" FORCE)
+        message(STATUS "Setting FAISS_OPT_LEVEL to 'sve' for ARM64 Linux")
+    else()
+        # Use generic optimization for other ARM64 platforms (like macOS)
+        set(FAISS_OPT_LEVEL "generic" CACHE STRING "" FORCE)
+        message(STATUS "Setting FAISS_OPT_LEVEL to 'generic' for ARM64 ${CMAKE_SYSTEM_NAME}")
+    endif()
+
+    # ARM64 compatibility: Faiss submodule has been modified to fix x86 header inclusion
+    message(STATUS "Using ARM64-compatible Faiss submodule")
+endif()

 # Additional optimization options from INSTALL.md
 set(CMAKE_BUILD_TYPE "Release" CACHE STRING "" FORCE)
--- a/packages/leann-backend-hnsw/pyproject.toml
+++ b/packages/leann-backend-hnsw/pyproject.toml
@@ -6,10 +6,10 @@ build-backend = "scikit_build_core.build"

 [project]
 name = "leann-backend-hnsw"
-version = "0.3.2"
+version = "0.3.3"
 description = "Custom-built HNSW (Faiss) backend for the Leann toolkit."
 dependencies = [
-    "leann-core==0.3.2",
+    "leann-core==0.3.3",
    "numpy",
    "pyzmq>=23.0.0",
    "msgpack>=1.0.0",
--- a/packages/leann-backend-hnsw/third_party/faiss
+++ b/packages/leann-backend-hnsw/third_party/faiss
--- a/packages/leann-core/pyproject.toml
+++ b/packages/leann-core/pyproject.toml
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"

 [project]
 name = "leann-core"
-version = "0.3.2"
+version = "0.3.3"
 description = "Core API and plugin system for LEANN"
 readme = "README.md"
 requires-python = ">=3.9"
--- a/packages/leann-core/src/leann/api.py
+++ b/packages/leann-core/src/leann/api.py
@@ -6,6 +6,8 @@ with the correct, original embedding logic from the user's reference code.
 import json
 import logging
 import pickle
+import re
+import subprocess
 import time
 import warnings
 from dataclasses import dataclass, field
@@ -675,6 +677,7 @@ class LeannSearcher:
        expected_zmq_port: int = 5557,
        metadata_filters: Optional[dict[str, dict[str, Union[str, int, float, bool, list]]]] = None,
        batch_size: int = 0,
+        use_grep: bool = False,
        **kwargs,
    ) -> list[SearchResult]:
        """
@@ -701,6 +704,10 @@ class LeannSearcher:
        Returns:
            List of SearchResult objects with text, metadata, and similarity scores
        """
+        # Handle grep search
+        if use_grep:
+            return self._grep_search(query, top_k)
+
        logger.info("🔍 LeannSearcher.search() called:")
        logger.info(f"  Query: '{query}'")
        logger.info(f"  Top_k: {top_k}")
@@ -817,9 +824,96 @@ class LeannSearcher:
        logger.info(f"  {GREEN}✓ Final enriched results: {len(enriched_results)} passages{RESET}")
        return enriched_results

+    def _find_jsonl_file(self) -> Optional[str]:
+        """Find the .jsonl file containing raw passages for grep search"""
+        index_path = Path(self.meta_path_str).parent
+        potential_files = [
+            index_path / "documents.leann.passages.jsonl",
+            index_path.parent / "documents.leann.passages.jsonl",
+        ]
+
+        for file_path in potential_files:
+            if file_path.exists():
+                return str(file_path)
+        return None
+
+    def _grep_search(self, query: str, top_k: int = 5) -> list[SearchResult]:
+        """Perform grep-based search on raw passages"""
+        jsonl_file = self._find_jsonl_file()
+        if not jsonl_file:
+            raise FileNotFoundError("No .jsonl passages file found for grep search")
+
+        try:
+            cmd = ["grep", "-i", "-n", query, jsonl_file]
+            result = subprocess.run(cmd, capture_output=True, text=True, check=False)
+
+            if result.returncode == 1:
+                return []
+            elif result.returncode != 0:
+                raise RuntimeError(f"Grep failed: {result.stderr}")
+
+            matches = []
+            for line in result.stdout.strip().split("\n"):
+                if not line:
+                    continue
+                parts = line.split(":", 1)
+                if len(parts) != 2:
+                    continue
+
+                try:
+                    data = json.loads(parts[1])
+                    text = data.get("text", "")
+                    score = text.lower().count(query.lower())
+
+                    matches.append(
+                        SearchResult(
+                            id=data.get("id", parts[0]),
+                            text=text,
+                            metadata=data.get("metadata", {}),
+                            score=float(score),
+                        )
+                    )
+                except json.JSONDecodeError:
+                    continue
+
+            matches.sort(key=lambda x: x.score, reverse=True)
+            return matches[:top_k]
+
+        except FileNotFoundError:
+            raise RuntimeError(
+                "grep command not found. Please install grep or use semantic search."
+            )
+
+    def _python_regex_search(self, query: str, top_k: int = 5) -> list[SearchResult]:
+        """Fallback regex search"""
+        jsonl_file = self._find_jsonl_file()
+        if not jsonl_file:
+            raise FileNotFoundError("No .jsonl file found")
+
+        pattern = re.compile(re.escape(query), re.IGNORECASE)
+        matches = []
+
+        with open(jsonl_file, encoding="utf-8") as f:
+            for line_num, line in enumerate(f, 1):
+                if pattern.search(line):
+                    try:
+                        data = json.loads(line.strip())
+                        matches.append(
+                            SearchResult(
+                                id=data.get("id", str(line_num)),
+                                text=data.get("text", ""),
+                                metadata=data.get("metadata", {}),
+                                score=float(len(pattern.findall(data.get("text", "")))),
+                            )
+                        )
+                    except json.JSONDecodeError:
+                        continue
+
+        matches.sort(key=lambda x: x.score, reverse=True)
+        return matches[:top_k]
+
    def cleanup(self):
        """Explicitly cleanup embedding server resources.
-
        This method should be called after you're done using the searcher,
        especially in test environments or batch processing scenarios.
        """
@@ -875,6 +969,7 @@ class LeannChat:
        expected_zmq_port: int = 5557,
        metadata_filters: Optional[dict[str, dict[str, Union[str, int, float, bool, list]]]] = None,
        batch_size: int = 0,
+        use_grep: bool = False,
        **search_kwargs,
    ):
        if llm_kwargs is None:
--- a/packages/leann-core/src/leann/cli.py
+++ b/packages/leann-core/src/leann/cli.py
@@ -322,9 +322,17 @@ Examples:

        return basic_matches

-    def _should_exclude_file(self, relative_path: Path, gitignore_matches) -> bool:
-        """Check if a file should be excluded using gitignore parser."""
-        return gitignore_matches(str(relative_path))
+    def _should_exclude_file(self, file_path: Path, gitignore_matches) -> bool:
+        """Check if a file should be excluded using gitignore parser.
+
+        Always match against absolute, posix-style paths for consistency with
+        gitignore_parser expectations.
+        """
+        try:
+            absolute_path = file_path.resolve()
+        except Exception:
+            absolute_path = Path(str(file_path))
+        return gitignore_matches(absolute_path.as_posix())

    def _is_git_submodule(self, path: Path) -> bool:
        """Check if a path is a git submodule."""
@@ -396,7 +404,9 @@ Examples:
        print(f"   {current_path}")
        print("   " + "─" * 45)

-        current_indexes = self._discover_indexes_in_project(current_path)
+        current_indexes = self._discover_indexes_in_project(
+            current_path, exclude_dirs=other_projects
+        )
        if current_indexes:
            for idx in current_indexes:
                total_indexes += 1
@@ -435,9 +445,14 @@ Examples:
            print("   leann build my-docs --docs ./documents")
        else:
            # Count only projects that have at least one discoverable index
-            projects_count = sum(
-                1 for p in valid_projects if len(self._discover_indexes_in_project(p)) > 0
-            )
+            projects_count = 0
+            for p in valid_projects:
+                if p == current_path:
+                    discovered = self._discover_indexes_in_project(p, exclude_dirs=other_projects)
+                else:
+                    discovered = self._discover_indexes_in_project(p)
+                if len(discovered) > 0:
+                    projects_count += 1
            print(f"📊 Total: {total_indexes} indexes across {projects_count} projects")

            if current_indexes_count > 0:
@@ -454,9 +469,22 @@ Examples:
                print("\n💡 Create your first index:")
                print("   leann build my-docs --docs ./documents")

-    def _discover_indexes_in_project(self, project_path: Path):
-        """Discover all indexes in a project directory (both CLI and apps formats)"""
+    def _discover_indexes_in_project(
+        self, project_path: Path, exclude_dirs: Optional[list[Path]] = None
+    ):
+        """Discover all indexes in a project directory (both CLI and apps formats)
+
+        exclude_dirs: when provided, skip any APP-format index files that are
+        located under these directories. This prevents duplicates when the
+        current project is a parent directory of other registered projects.
+        """
        indexes = []
+        exclude_dirs = exclude_dirs or []
+        # normalize to resolved paths once for comparison
+        try:
+            exclude_dirs_resolved = [p.resolve() for p in exclude_dirs]
+        except Exception:
+            exclude_dirs_resolved = exclude_dirs

        # 1. CLI format: .leann/indexes/index_name/
        cli_indexes_dir = project_path / ".leann" / "indexes"
@@ -495,6 +523,17 @@ Examples:
                        continue
                except Exception:
                    pass
+                # Skip meta files that live under excluded directories
+                try:
+                    meta_parent_resolved = meta_file.parent.resolve()
+                    if any(
+                        meta_parent_resolved.is_relative_to(ex_dir)
+                        for ex_dir in exclude_dirs_resolved
+                    ):
+                        continue
+                except Exception:
+                    # best effort; if resolve or comparison fails, do not exclude
+                    pass
                # Use the parent directory name as the app index display name
                display_name = meta_file.parent.name
                # Extract file base used to store files
@@ -1022,7 +1061,8 @@ Examples:

            # Try to use better PDF parsers first, but only if PDFs are requested
            documents = []
-            docs_path = Path(docs_dir)
+            # Use resolved absolute paths to avoid mismatches (symlinks, relative vs absolute)
+            docs_path = Path(docs_dir).resolve()

            # Check if we should process PDFs
            should_process_pdfs = custom_file_types is None or ".pdf" in custom_file_types
@@ -1031,10 +1071,15 @@ Examples:
                for file_path in docs_path.rglob("*.pdf"):
                    # Check if file matches any exclude pattern
                    try:
+                        # Ensure both paths are resolved before computing relativity
+                        file_path_resolved = file_path.resolve()
+                        # Determine directory scope using the non-resolved path to avoid
+                        # misclassifying symlinked entries as outside the docs directory
                        relative_path = file_path.relative_to(docs_path)
                        if not include_hidden and _path_has_hidden_segment(relative_path):
                            continue
-                        if self._should_exclude_file(relative_path, gitignore_matches):
+                        # Use absolute path for gitignore matching
+                        if self._should_exclude_file(file_path_resolved, gitignore_matches):
                            continue
                    except ValueError:
                        # Skip files that can't be made relative to docs_path
@@ -1077,10 +1122,11 @@ Examples:
                ) -> bool:
                    """Return True if file should be included (not excluded)"""
                    try:
-                        docs_path_obj = Path(docs_dir)
-                        file_path_obj = Path(file_path)
-                        relative_path = file_path_obj.relative_to(docs_path_obj)
-                        return not self._should_exclude_file(relative_path, gitignore_matches)
+                        docs_path_obj = Path(docs_dir).resolve()
+                        file_path_obj = Path(file_path).resolve()
+                        # Use absolute path for gitignore matching
+                        _ = file_path_obj.relative_to(docs_path_obj)  # validate scope
+                        return not self._should_exclude_file(file_path_obj, gitignore_matches)
                    except (ValueError, OSError):
                        return True  # Include files that can't be processed

--- a/packages/leann-mcp/README.md
+++ b/packages/leann-mcp/README.md
@@ -2,6 +2,8 @@

 Transform your development workflow with intelligent code assistance using LEANN's semantic search directly in Claude Code.

+For agent-facing discovery details, see `llms.txt` in the repository root.
+
 ## Prerequisites

 Install LEANN globally for MCP integration (with default backend):
--- a/packages/leann/pyproject.toml
+++ b/packages/leann/pyproject.toml
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"

 [project]
 name = "leann"
-version = "0.3.2"
+version = "0.3.3"
 description = "LEANN - The smallest vector index in the world. RAG Everything with LEANN!"
 readme = "README.md"
 requires-python = ">=3.9"