fix: resolve all ruff linting errors and add lint CI check

- Fix ambiguous fullwidth characters (commas, parentheses) in strings and comments - Replace Chinese comments with English equivalents - Fix unused imports with proper noqa annotations for intentional imports - Fix bare except clauses with specific exception types - Fix redefined variables and undefined names - Add ruff noqa annotations for generated protobuf files - Add lint and format check to GitHub Actions CI pipeline
2025-07-26 22:35:12 -07:00
parent 8537a6b17e
commit b3e9ee96fa
53 changed files with 5655 additions and 5220 deletions
--- a/packages/leann-backend-diskann/init.py
+++ b/packages/leann-backend-diskann/init.py
@@ -1 +1 @@
-# This file makes the directory a Python package 
+# This file makes the directory a Python package
--- a/packages/leann-backend-diskann/leann_backend_diskann/init.py
+++ b/packages/leann-backend-diskann/leann_backend_diskann/init.py
@@ -1 +1 @@
-from . import diskann_backend
+from . import diskann_backend as diskann_backend
--- a/packages/leann-backend-diskann/leann_backend_diskann/diskann_backend.py
+++ b/packages/leann-backend-diskann/leann_backend_diskann/diskann_backend.py
@@ -1,20 +1,19 @@
-import numpy as np
+import contextlib
+import logging
 import os
 import struct
 import sys
 from pathlib import Path
-from typing import Dict, Any, List, Literal, Optional
-import contextlib
+from typing import Any, Literal

-import logging
-
-from leann.searcher_base import BaseSearcher
-from leann.registry import register_backend
+import numpy as np
 from leann.interface import (
-    LeannBackendFactoryInterface,
    LeannBackendBuilderInterface,
+    LeannBackendFactoryInterface,
    LeannBackendSearcherInterface,
 )
+from leann.registry import register_backend
+from leann.searcher_base import BaseSearcher

 logger = logging.getLogger(__name__)

@@ -100,7 +99,7 @@ class DiskannBuilder(LeannBackendBuilderInterface):
    def __init__(self, **kwargs):
        self.build_params = kwargs

-    def build(self, data: np.ndarray, ids: List[str], index_path: str, **kwargs):
+    def build(self, data: np.ndarray, ids: list[str], index_path: str, **kwargs):
        path = Path(index_path)
        index_dir = path.parent
        index_prefix = path.stem
@@ -186,11 +185,11 @@ class DiskannSearcher(BaseSearcher):
        prune_ratio: float = 0.0,
        recompute_embeddings: bool = False,
        pruning_strategy: Literal["global", "local", "proportional"] = "global",
-        zmq_port: Optional[int] = None,
+        zmq_port: int | None = None,
        batch_recompute: bool = False,
        dedup_node_dis: bool = False,
        **kwargs,
-    ) -> Dict[str, Any]:
+    ) -> dict[str, Any]:
        """
        Search for nearest neighbors using DiskANN index.

@@ -216,14 +215,10 @@ class DiskannSearcher(BaseSearcher):
        # Handle zmq_port compatibility: DiskANN can now update port at runtime
        if recompute_embeddings:
            if zmq_port is None:
-                raise ValueError(
-                    "zmq_port must be provided if recompute_embeddings is True"
-                )
+                raise ValueError("zmq_port must be provided if recompute_embeddings is True")
            current_port = self._index.get_zmq_port()
            if zmq_port != current_port:
-                logger.debug(
-                    f"Updating DiskANN zmq_port from {current_port} to {zmq_port}"
-                )
+                logger.debug(f"Updating DiskANN zmq_port from {current_port} to {zmq_port}")
                self._index.set_zmq_port(zmq_port)

        # DiskANN doesn't support "proportional" strategy
@@ -259,8 +254,6 @@ class DiskannSearcher(BaseSearcher):
                use_global_pruning,
            )

-        string_labels = [
-            [str(int_label) for int_label in batch_labels] for batch_labels in labels
-        ]
+        string_labels = [[str(int_label) for int_label in batch_labels] for batch_labels in labels]

        return {"labels": string_labels, "distances": distances}
--- a/packages/leann-backend-diskann/leann_backend_diskann/diskann_embedding_server.py
+++ b/packages/leann-backend-diskann/leann_backend_diskann/diskann_embedding_server.py
@@ -3,16 +3,16 @@ DiskANN-specific embedding server
 """

 import argparse
+import json
+import logging
+import os
+import sys
 import threading
 import time
-import os
-import zmq
-import numpy as np
-import json
 from pathlib import Path
-from typing import Optional
-import sys
-import logging
+
+import numpy as np
+import zmq

 # Set up logging based on environment variable
 LOG_LEVEL = os.getenv("LEANN_LOG_LEVEL", "WARNING").upper()
@@ -32,7 +32,7 @@ if not logger.handlers:


 def create_diskann_embedding_server(
-    passages_file: Optional[str] = None,
+    passages_file: str | None = None,
    zmq_port: int = 5555,
    model_name: str = "sentence-transformers/all-mpnet-base-v2",
    embedding_mode: str = "sentence-transformers",
@@ -50,8 +50,8 @@ def create_diskann_embedding_server(
    sys.path.insert(0, str(leann_core_path))

    try:
-        from leann.embedding_compute import compute_embeddings
        from leann.api import PassageManager
+        from leann.embedding_compute import compute_embeddings

        logger.info("Successfully imported unified embedding computation module")
    except ImportError as e:
@@ -76,7 +76,7 @@ def create_diskann_embedding_server(
        raise ValueError("Only metadata files (.meta.json) are supported")

    # Load metadata to get passage sources
-    with open(passages_file, "r") as f:
+    with open(passages_file) as f:
        meta = json.load(f)

    passages = PassageManager(meta["passage_sources"])
@@ -150,9 +150,7 @@ def create_diskann_embedding_server(
                        ):
                            texts = request
                            is_text_request = True
-                            logger.info(
-                                f"✅ MSGPACK: Direct text request for {len(texts)} texts"
-                            )
+                            logger.info(f"✅ MSGPACK: Direct text request for {len(texts)} texts")
                        else:
                            raise ValueError("Not a valid msgpack text request")
                    except Exception as msgpack_error:
@@ -167,9 +165,7 @@ def create_diskann_embedding_server(
                            passage_data = passages.get_passage(str(nid))
                            txt = passage_data["text"]
                            if not txt:
-                                raise RuntimeError(
-                                    f"FATAL: Empty text for passage ID {nid}"
-                                )
+                                raise RuntimeError(f"FATAL: Empty text for passage ID {nid}")
                            texts.append(txt)
                        except KeyError as e:
                            logger.error(f"Passage ID {nid} not found: {e}")
@@ -180,9 +176,7 @@ def create_diskann_embedding_server(

                    # Debug logging
                    logger.debug(f"Processing {len(texts)} texts")
-                    logger.debug(
-                        f"Text lengths: {[len(t) for t in texts[:5]]}"
-                    )  # Show first 5
+                    logger.debug(f"Text lengths: {[len(t) for t in texts[:5]]}")  # Show first 5

                # Process embeddings using unified computation
                embeddings = compute_embeddings(texts, model_name, mode=embedding_mode)
@@ -199,9 +193,7 @@ def create_diskann_embedding_server(
                else:
                    # For DiskANN C++ compatibility: return protobuf format
                    resp_proto = embedding_pb2.NodeEmbeddingResponse()
-                    hidden_contiguous = np.ascontiguousarray(
-                        embeddings, dtype=np.float32
-                    )
+                    hidden_contiguous = np.ascontiguousarray(embeddings, dtype=np.float32)

                    # Serialize embeddings data
                    resp_proto.embeddings_data = hidden_contiguous.tobytes()
--- a/packages/leann-backend-diskann/leann_backend_diskann/embedding_pb2.py
+++ b/packages/leann-backend-diskann/leann_backend_diskann/embedding_pb2.py
@@ -1,27 +1,28 @@
-# -*- coding: utf-8 -*-
 # Generated by the protocol buffer compiler.  DO NOT EDIT!
 # source: embedding.proto
+# ruff: noqa
 """Generated protocol buffer code."""
-from google.protobuf.internal import builder as _builder
+
 from google.protobuf import descriptor as _descriptor
 from google.protobuf import descriptor_pool as _descriptor_pool
 from google.protobuf import symbol_database as _symbol_database
+from google.protobuf.internal import builder as _builder
+
 # @@protoc_insertion_point(imports)

 _sym_db = _symbol_database.Default()


-
-
-DESCRIPTOR = _descriptor_pool.Default().AddSerializedFile(b'\n\x0f\x65mbedding.proto\x12\x0eprotoembedding\"(\n\x14NodeEmbeddingRequest\x12\x10\n\x08node_ids\x18\x01 \x03(\r\"Y\n\x15NodeEmbeddingResponse\x12\x17\n\x0f\x65mbeddings_data\x18\x01 \x01(\x0c\x12\x12\n\ndimensions\x18\x02 \x03(\x05\x12\x13\n\x0bmissing_ids\x18\x03 \x03(\rb\x06proto3')
+DESCRIPTOR = _descriptor_pool.Default().AddSerializedFile(
+    b'\n\x0f\x65mbedding.proto\x12\x0eprotoembedding"(\n\x14NodeEmbeddingRequest\x12\x10\n\x08node_ids\x18\x01 \x03(\r"Y\n\x15NodeEmbeddingResponse\x12\x17\n\x0f\x65mbeddings_data\x18\x01 \x01(\x0c\x12\x12\n\ndimensions\x18\x02 \x03(\x05\x12\x13\n\x0bmissing_ids\x18\x03 \x03(\rb\x06proto3'
+)

 _builder.BuildMessageAndEnumDescriptors(DESCRIPTOR, globals())
-_builder.BuildTopDescriptorsAndMessages(DESCRIPTOR, 'embedding_pb2', globals())
-if _descriptor._USE_C_DESCRIPTORS == False:
-
-  DESCRIPTOR._options = None
-  _NODEEMBEDDINGREQUEST._serialized_start=35
-  _NODEEMBEDDINGREQUEST._serialized_end=75
-  _NODEEMBEDDINGRESPONSE._serialized_start=77
-  _NODEEMBEDDINGRESPONSE._serialized_end=166
+_builder.BuildTopDescriptorsAndMessages(DESCRIPTOR, "embedding_pb2", globals())
+if not _descriptor._USE_C_DESCRIPTORS:
+    DESCRIPTOR._options = None
+    _NODEEMBEDDINGREQUEST._serialized_start = 35
+    _NODEEMBEDDINGREQUEST._serialized_end = 75
+    _NODEEMBEDDINGRESPONSE._serialized_start = 77
+    _NODEEMBEDDINGRESPONSE._serialized_end = 166
 # @@protoc_insertion_point(module_scope)
--- a/packages/leann-backend-hnsw/leann_backend_hnsw/init.py
+++ b/packages/leann-backend-hnsw/leann_backend_hnsw/init.py
@@ -1 +1 @@
-from . import hnsw_backend
+from . import hnsw_backend as hnsw_backend
--- a/packages/leann-backend-hnsw/leann_backend_hnsw/convert_to_csr.py
+++ b/packages/leann-backend-hnsw/leann_backend_hnsw/convert_to_csr.py
@@ -1,87 +1,111 @@
+import argparse
+import gc  # Import garbage collector interface
+import os
 import struct
 import sys
-import numpy as np
-import os
-import argparse
-import gc # Import garbage collector interface
 import time
+
+import numpy as np
+
 # --- FourCCs (add more if needed) ---
-INDEX_HNSW_FLAT_FOURCC = int.from_bytes(b'IHNf', 'little')
+INDEX_HNSW_FLAT_FOURCC = int.from_bytes(b"IHNf", "little")
 # Add other HNSW fourccs if you expect different storage types inside HNSW
 # INDEX_HNSW_PQ_FOURCC = int.from_bytes(b'IHNp', 'little')
 # INDEX_HNSW_SQ_FOURCC = int.from_bytes(b'IHNs', 'little')
 # INDEX_HNSW_CAGRA_FOURCC = int.from_bytes(b'IHNc', 'little') # Example

-EXPECTED_HNSW_FOURCCS = {INDEX_HNSW_FLAT_FOURCC} # Modify if needed
-NULL_INDEX_FOURCC = int.from_bytes(b'null', 'little')
+EXPECTED_HNSW_FOURCCS = {INDEX_HNSW_FLAT_FOURCC}  # Modify if needed
+NULL_INDEX_FOURCC = int.from_bytes(b"null", "little")

 # --- Helper functions for reading/writing binary data ---

+
 def read_struct(f, fmt):
    """Reads data according to the struct format."""
    size = struct.calcsize(fmt)
    data = f.read(size)
    if len(data) != size:
-        raise EOFError(f"File ended unexpectedly reading struct fmt '{fmt}'. Expected {size} bytes, got {len(data)}.")
+        raise EOFError(
+            f"File ended unexpectedly reading struct fmt '{fmt}'. Expected {size} bytes, got {len(data)}."
+        )
    return struct.unpack(fmt, data)[0]

+
 def read_vector_raw(f, element_fmt_char):
    """Reads a vector (size followed by data), returns count and raw bytes."""
-    count = -1 # Initialize count
-    total_bytes = -1 # Initialize total_bytes
+    count = -1  # Initialize count
+    total_bytes = -1  # Initialize total_bytes
    try:
-        count = read_struct(f, '<Q') # size_t usually 64-bit unsigned
+        count = read_struct(f, "<Q")  # size_t usually 64-bit unsigned
        element_size = struct.calcsize(element_fmt_char)
        # --- FIX for MemoryError: Check for unreasonably large count ---
-        max_reasonable_count = 10 * (10**9) # ~10 billion elements limit
+        max_reasonable_count = 10 * (10**9)  # ~10 billion elements limit
        if count > max_reasonable_count or count < 0:
-            raise MemoryError(f"Vector count {count} seems unreasonably large, possibly due to file corruption or incorrect format read.")
+            raise MemoryError(
+                f"Vector count {count} seems unreasonably large, possibly due to file corruption or incorrect format read."
+            )

        total_bytes = count * element_size
        # --- FIX for MemoryError: Check for huge byte size before allocation ---
-        max_reasonable_bytes = 50 * (1024**3) # ~50 GB limit
-        if total_bytes > max_reasonable_bytes or total_bytes < 0: # Check for overflow
-             raise MemoryError(f"Attempting to read {total_bytes} bytes ({count} elements * {element_size} bytes/element), which exceeds the safety limit. File might be corrupted or format mismatch.")
+        max_reasonable_bytes = 50 * (1024**3)  # ~50 GB limit
+        if total_bytes > max_reasonable_bytes or total_bytes < 0:  # Check for overflow
+            raise MemoryError(
+                f"Attempting to read {total_bytes} bytes ({count} elements * {element_size} bytes/element), which exceeds the safety limit. File might be corrupted or format mismatch."
+            )

        data_bytes = f.read(total_bytes)

        if len(data_bytes) != total_bytes:
-             raise EOFError(f"File ended unexpectedly reading vector data. Expected {total_bytes} bytes, got {len(data_bytes)}.")
+            raise EOFError(
+                f"File ended unexpectedly reading vector data. Expected {total_bytes} bytes, got {len(data_bytes)}."
+            )
        return count, data_bytes
    except (MemoryError, OverflowError) as e:
-         # Add context to the error message
-         print(f"\nError during raw vector read (element_fmt='{element_fmt_char}', count={count}, total_bytes={total_bytes}): {e}", file=sys.stderr)
-         raise e # Re-raise the original error type
+        # Add context to the error message
+        print(
+            f"\nError during raw vector read (element_fmt='{element_fmt_char}', count={count}, total_bytes={total_bytes}): {e}",
+            file=sys.stderr,
+        )
+        raise e  # Re-raise the original error type
+

 def read_numpy_vector(f, np_dtype, struct_fmt_char):
    """Reads a vector into a NumPy array."""
-    count = -1 # Initialize count for robust error handling
-    print(f"  Reading vector (dtype={np_dtype}, fmt='{struct_fmt_char}')... ", end='', flush=True)
+    count = -1  # Initialize count for robust error handling
+    print(f"  Reading vector (dtype={np_dtype}, fmt='{struct_fmt_char}')... ", end="", flush=True)
    try:
        count, data_bytes = read_vector_raw(f, struct_fmt_char)
        print(f"Count={count}, Bytes={len(data_bytes)}")
        if count > 0 and len(data_bytes) > 0:
            arr = np.frombuffer(data_bytes, dtype=np_dtype)
            if arr.size != count:
-                raise ValueError(f"Inconsistent array size after reading. Expected {count}, got {arr.size}")
+                raise ValueError(
+                    f"Inconsistent array size after reading. Expected {count}, got {arr.size}"
+                )
            return arr
        elif count == 0:
-             return np.array([], dtype=np_dtype)
+            return np.array([], dtype=np_dtype)
        else:
-             raise ValueError("Read zero bytes but count > 0.")
+            raise ValueError("Read zero bytes but count > 0.")
    except MemoryError as e:
        # Now count should be defined (or -1 if error was in read_struct)
-        print(f"\nMemoryError creating NumPy array (dtype={np_dtype}, count={count}). {e}", file=sys.stderr)
+        print(
+            f"\nMemoryError creating NumPy array (dtype={np_dtype}, count={count}). {e}",
+            file=sys.stderr,
+        )
        raise e
-    except Exception as e: # Catch other potential errors like ValueError
-        print(f"\nError reading numpy vector (dtype={np_dtype}, fmt='{struct_fmt_char}', count={count}): {e}", file=sys.stderr)
+    except Exception as e:  # Catch other potential errors like ValueError
+        print(
+            f"\nError reading numpy vector (dtype={np_dtype}, fmt='{struct_fmt_char}', count={count}): {e}",
+            file=sys.stderr,
+        )
        raise e


 def write_numpy_vector(f, arr, struct_fmt_char):
    """Writes a NumPy array as a vector (size followed by data)."""
    count = arr.size
-    f.write(struct.pack('<Q', count))
+    f.write(struct.pack("<Q", count))
    try:
        expected_dtype = np.dtype(struct_fmt_char)
        if arr.dtype != expected_dtype:
@@ -89,23 +113,30 @@ def write_numpy_vector(f, arr, struct_fmt_char):
        else:
            data_to_write = arr.tobytes()
        f.write(data_to_write)
-        del data_to_write # Hint GC
+        del data_to_write  # Hint GC
    except MemoryError as e:
-         print(f"\nMemoryError converting NumPy array to bytes for writing (size={count}, dtype={arr.dtype}). {e}", file=sys.stderr)
-         raise e
+        print(
+            f"\nMemoryError converting NumPy array to bytes for writing (size={count}, dtype={arr.dtype}). {e}",
+            file=sys.stderr,
+        )
+        raise e
+

 def write_list_vector(f, lst, struct_fmt_char):
    """Writes a Python list as a vector iteratively."""
    count = len(lst)
-    f.write(struct.pack('<Q', count))
-    fmt = '<' + struct_fmt_char
+    f.write(struct.pack("<Q", count))
+    fmt = "<" + struct_fmt_char
    chunk_size = 1024 * 1024
    element_size = struct.calcsize(fmt)
    # Allocate buffer outside the loop if possible, or handle MemoryError during allocation
    try:
        buffer = bytearray(chunk_size * element_size)
    except MemoryError:
-        print(f"MemoryError: Cannot allocate buffer for writing list vector chunk (size {chunk_size * element_size} bytes).", file=sys.stderr)
+        print(
+            f"MemoryError: Cannot allocate buffer for writing list vector chunk (size {chunk_size * element_size} bytes).",
+            file=sys.stderr,
+        )
        raise
    buffer_count = 0

@@ -116,66 +147,80 @@ def write_list_vector(f, lst, struct_fmt_char):
            buffer_count += 1

            if buffer_count == chunk_size or i == count - 1:
-                f.write(buffer[:buffer_count * element_size])
+                f.write(buffer[: buffer_count * element_size])
                buffer_count = 0

        except struct.error as e:
-            print(f"\nStruct packing error for item {item} at index {i} with format '{fmt}'. {e}", file=sys.stderr)
+            print(
+                f"\nStruct packing error for item {item} at index {i} with format '{fmt}'. {e}",
+                file=sys.stderr,
+            )
            raise e


 def get_cum_neighbors(cum_nneighbor_per_level_np, level):
    """Helper to get cumulative neighbors count, matching C++ logic."""
-    if level < 0: return 0
+    if level < 0:
+        return 0
    if level < len(cum_nneighbor_per_level_np):
        return cum_nneighbor_per_level_np[level]
    else:
        return cum_nneighbor_per_level_np[-1] if len(cum_nneighbor_per_level_np) > 0 else 0

-def write_compact_format(f_out, original_hnsw_data, assign_probas_np, cum_nneighbor_per_level_np, 
-                        levels_np, compact_level_ptr, compact_node_offsets_np, 
-                        compact_neighbors_data, storage_fourcc, storage_data):
+
+def write_compact_format(
+    f_out,
+    original_hnsw_data,
+    assign_probas_np,
+    cum_nneighbor_per_level_np,
+    levels_np,
+    compact_level_ptr,
+    compact_node_offsets_np,
+    compact_neighbors_data,
+    storage_fourcc,
+    storage_data,
+):
    """Write HNSW data in compact format following C++ read order exactly."""
    # Write IndexHNSW Header
-    f_out.write(struct.pack('<I', original_hnsw_data['index_fourcc']))
-    f_out.write(struct.pack('<i', original_hnsw_data['d']))
-    f_out.write(struct.pack('<q', original_hnsw_data['ntotal']))
-    f_out.write(struct.pack('<q', original_hnsw_data['dummy1']))
-    f_out.write(struct.pack('<q', original_hnsw_data['dummy2']))
-    f_out.write(struct.pack('<?', original_hnsw_data['is_trained']))
-    f_out.write(struct.pack('<i', original_hnsw_data['metric_type']))
-    if original_hnsw_data['metric_type'] > 1:
-         f_out.write(struct.pack('<f', original_hnsw_data['metric_arg']))
+    f_out.write(struct.pack("<I", original_hnsw_data["index_fourcc"]))
+    f_out.write(struct.pack("<i", original_hnsw_data["d"]))
+    f_out.write(struct.pack("<q", original_hnsw_data["ntotal"]))
+    f_out.write(struct.pack("<q", original_hnsw_data["dummy1"]))
+    f_out.write(struct.pack("<q", original_hnsw_data["dummy2"]))
+    f_out.write(struct.pack("<?", original_hnsw_data["is_trained"]))
+    f_out.write(struct.pack("<i", original_hnsw_data["metric_type"]))
+    if original_hnsw_data["metric_type"] > 1:
+        f_out.write(struct.pack("<f", original_hnsw_data["metric_arg"]))

    # Write HNSW struct parts (standard order)
-    write_numpy_vector(f_out, assign_probas_np, 'd')
-    write_numpy_vector(f_out, cum_nneighbor_per_level_np, 'i')
-    write_numpy_vector(f_out, levels_np, 'i')
+    write_numpy_vector(f_out, assign_probas_np, "d")
+    write_numpy_vector(f_out, cum_nneighbor_per_level_np, "i")
+    write_numpy_vector(f_out, levels_np, "i")

    # Write compact format flag
-    f_out.write(struct.pack('<?', True)) # storage_is_compact = True
+    f_out.write(struct.pack("<?", True))  # storage_is_compact = True

    # Write compact data in CORRECT C++ read order: level_ptr, node_offsets FIRST
    if isinstance(compact_level_ptr, np.ndarray):
-        write_numpy_vector(f_out, compact_level_ptr, 'Q')
+        write_numpy_vector(f_out, compact_level_ptr, "Q")
    else:
-        write_list_vector(f_out, compact_level_ptr, 'Q')
-    
-    write_numpy_vector(f_out, compact_node_offsets_np, 'Q')
+        write_list_vector(f_out, compact_level_ptr, "Q")
+
+    write_numpy_vector(f_out, compact_node_offsets_np, "Q")

    # Write HNSW scalar parameters
-    f_out.write(struct.pack('<i', original_hnsw_data['entry_point']))
-    f_out.write(struct.pack('<i', original_hnsw_data['max_level']))
-    f_out.write(struct.pack('<i', original_hnsw_data['efConstruction']))
-    f_out.write(struct.pack('<i', original_hnsw_data['efSearch']))
-    f_out.write(struct.pack('<i', original_hnsw_data['dummy_upper_beam']))
+    f_out.write(struct.pack("<i", original_hnsw_data["entry_point"]))
+    f_out.write(struct.pack("<i", original_hnsw_data["max_level"]))
+    f_out.write(struct.pack("<i", original_hnsw_data["efConstruction"]))
+    f_out.write(struct.pack("<i", original_hnsw_data["efSearch"]))
+    f_out.write(struct.pack("<i", original_hnsw_data["dummy_upper_beam"]))

    # Write storage fourcc (this determines how to read what follows)
-    f_out.write(struct.pack('<I', storage_fourcc))
-    
+    f_out.write(struct.pack("<I", storage_fourcc))
+
    # Write compact neighbors data AFTER storage fourcc
-    write_list_vector(f_out, compact_neighbors_data, 'i')
-    
+    write_list_vector(f_out, compact_neighbors_data, "i")
+
    # Write storage data if not NULL (only after neighbors)
    if storage_fourcc != NULL_INDEX_FOURCC and storage_data:
        f_out.write(storage_data)
@@ -183,11 +228,12 @@ def write_compact_format(f_out, original_hnsw_data, assign_probas_np, cum_nneigh

 # --- Main Conversion Logic ---

+
 def convert_hnsw_graph_to_csr(input_filename, output_filename, prune_embeddings=True):
    """
    Converts an HNSW graph file to the CSR format.
    Supports both original and already-compact formats (backward compatibility).
-    
+
    Args:
        input_filename: Input HNSW index file
        output_filename: Output CSR index file
@@ -196,172 +242,228 @@ def convert_hnsw_graph_to_csr(input_filename, output_filename, prune_embeddings=
    print(f"Starting conversion: {input_filename} -> {output_filename}")
    start_time = time.time()
    original_hnsw_data = {}
-    neighbors_np = None # Initialize to allow check in finally block
+    neighbors_np = None  # Initialize to allow check in finally block
    try:
-        with open(input_filename, 'rb') as f_in, open(output_filename, 'wb') as f_out:
-
+        with open(input_filename, "rb") as f_in, open(output_filename, "wb") as f_out:
            # --- Read IndexHNSW FourCC and Header ---
            print(f"[{time.time() - start_time:.2f}s] Reading Index HNSW header...")
            # ... (Keep the header reading logic as before) ...
-            hnsw_index_fourcc = read_struct(f_in, '<I')
+            hnsw_index_fourcc = read_struct(f_in, "<I")
            if hnsw_index_fourcc not in EXPECTED_HNSW_FOURCCS:
-                 print(f"Error: Expected HNSW Index FourCC ({list(EXPECTED_HNSW_FOURCCS)}), got {hnsw_index_fourcc:08x}.", file=sys.stderr)
-                 return False
-            original_hnsw_data['index_fourcc'] = hnsw_index_fourcc
-            original_hnsw_data['d'] = read_struct(f_in, '<i')
-            original_hnsw_data['ntotal'] = read_struct(f_in, '<q')
-            original_hnsw_data['dummy1'] = read_struct(f_in, '<q')
-            original_hnsw_data['dummy2'] = read_struct(f_in, '<q')
-            original_hnsw_data['is_trained'] = read_struct(f_in, '?')
-            original_hnsw_data['metric_type'] = read_struct(f_in, '<i')
-            original_hnsw_data['metric_arg'] = 0.0
-            if original_hnsw_data['metric_type'] > 1:
-                 original_hnsw_data['metric_arg'] = read_struct(f_in, '<f')
-            print(f"[{time.time() - start_time:.2f}s]   Header read: d={original_hnsw_data['d']}, ntotal={original_hnsw_data['ntotal']}")
-
+                print(
+                    f"Error: Expected HNSW Index FourCC ({list(EXPECTED_HNSW_FOURCCS)}), got {hnsw_index_fourcc:08x}.",
+                    file=sys.stderr,
+                )
+                return False
+            original_hnsw_data["index_fourcc"] = hnsw_index_fourcc
+            original_hnsw_data["d"] = read_struct(f_in, "<i")
+            original_hnsw_data["ntotal"] = read_struct(f_in, "<q")
+            original_hnsw_data["dummy1"] = read_struct(f_in, "<q")
+            original_hnsw_data["dummy2"] = read_struct(f_in, "<q")
+            original_hnsw_data["is_trained"] = read_struct(f_in, "?")
+            original_hnsw_data["metric_type"] = read_struct(f_in, "<i")
+            original_hnsw_data["metric_arg"] = 0.0
+            if original_hnsw_data["metric_type"] > 1:
+                original_hnsw_data["metric_arg"] = read_struct(f_in, "<f")
+            print(
+                f"[{time.time() - start_time:.2f}s]   Header read: d={original_hnsw_data['d']}, ntotal={original_hnsw_data['ntotal']}"
+            )

            # --- Read original HNSW struct data ---
            print(f"[{time.time() - start_time:.2f}s] Reading HNSW struct vectors...")
-            assign_probas_np = read_numpy_vector(f_in, np.float64, 'd')
-            print(f"[{time.time() - start_time:.2f}s]   Read assign_probas ({assign_probas_np.size})")
+            assign_probas_np = read_numpy_vector(f_in, np.float64, "d")
+            print(
+                f"[{time.time() - start_time:.2f}s]   Read assign_probas ({assign_probas_np.size})"
+            )
            gc.collect()

-            cum_nneighbor_per_level_np = read_numpy_vector(f_in, np.int32, 'i')
-            print(f"[{time.time() - start_time:.2f}s]   Read cum_nneighbor_per_level ({cum_nneighbor_per_level_np.size})")
+            cum_nneighbor_per_level_np = read_numpy_vector(f_in, np.int32, "i")
+            print(
+                f"[{time.time() - start_time:.2f}s]   Read cum_nneighbor_per_level ({cum_nneighbor_per_level_np.size})"
+            )
            gc.collect()

-            levels_np = read_numpy_vector(f_in, np.int32, 'i')
+            levels_np = read_numpy_vector(f_in, np.int32, "i")
            print(f"[{time.time() - start_time:.2f}s]   Read levels ({levels_np.size})")
            gc.collect()

            ntotal = len(levels_np)
-            if ntotal != original_hnsw_data['ntotal']:
-                 print(f"Warning: ntotal mismatch! Header says {original_hnsw_data['ntotal']}, levels vector size is {ntotal}. Using levels vector size.", file=sys.stderr)
-                 original_hnsw_data['ntotal'] = ntotal
+            if ntotal != original_hnsw_data["ntotal"]:
+                print(
+                    f"Warning: ntotal mismatch! Header says {original_hnsw_data['ntotal']}, levels vector size is {ntotal}. Using levels vector size.",
+                    file=sys.stderr,
+                )
+                original_hnsw_data["ntotal"] = ntotal

            # --- Check for compact format flag ---
            print(f"[{time.time() - start_time:.2f}s]   Probing for compact storage flag...")
            pos_before_compact = f_in.tell()
            try:
-                is_compact_flag = read_struct(f_in, '<?')
+                is_compact_flag = read_struct(f_in, "<?")
                print(f"[{time.time() - start_time:.2f}s]   Found compact flag: {is_compact_flag}")
-                
+
                if is_compact_flag:
                    # Input is already in compact format - read compact data
-                    print(f"[{time.time() - start_time:.2f}s]   Input is already in compact format, reading compact data...")
-                    
-                    compact_level_ptr = read_numpy_vector(f_in, np.uint64, 'Q')
-                    print(f"[{time.time() - start_time:.2f}s]   Read compact_level_ptr ({compact_level_ptr.size})")
-                    
-                    compact_node_offsets_np = read_numpy_vector(f_in, np.uint64, 'Q')
-                    print(f"[{time.time() - start_time:.2f}s]   Read compact_node_offsets ({compact_node_offsets_np.size})")
-                    
+                    print(
+                        f"[{time.time() - start_time:.2f}s]   Input is already in compact format, reading compact data..."
+                    )
+
+                    compact_level_ptr = read_numpy_vector(f_in, np.uint64, "Q")
+                    print(
+                        f"[{time.time() - start_time:.2f}s]   Read compact_level_ptr ({compact_level_ptr.size})"
+                    )
+
+                    compact_node_offsets_np = read_numpy_vector(f_in, np.uint64, "Q")
+                    print(
+                        f"[{time.time() - start_time:.2f}s]   Read compact_node_offsets ({compact_node_offsets_np.size})"
+                    )
+
                    # Read scalar parameters
-                    original_hnsw_data['entry_point'] = read_struct(f_in, '<i')
-                    original_hnsw_data['max_level'] = read_struct(f_in, '<i')
-                    original_hnsw_data['efConstruction'] = read_struct(f_in, '<i')
-                    original_hnsw_data['efSearch'] = read_struct(f_in, '<i')
-                    original_hnsw_data['dummy_upper_beam'] = read_struct(f_in, '<i')
-                    print(f"[{time.time() - start_time:.2f}s]   Read scalar params (ep={original_hnsw_data['entry_point']}, max_lvl={original_hnsw_data['max_level']})")
+                    original_hnsw_data["entry_point"] = read_struct(f_in, "<i")
+                    original_hnsw_data["max_level"] = read_struct(f_in, "<i")
+                    original_hnsw_data["efConstruction"] = read_struct(f_in, "<i")
+                    original_hnsw_data["efSearch"] = read_struct(f_in, "<i")
+                    original_hnsw_data["dummy_upper_beam"] = read_struct(f_in, "<i")
+                    print(
+                        f"[{time.time() - start_time:.2f}s]   Read scalar params (ep={original_hnsw_data['entry_point']}, max_lvl={original_hnsw_data['max_level']})"
+                    )

                    # Read storage fourcc
-                    storage_fourcc = read_struct(f_in, '<I')
-                    print(f"[{time.time() - start_time:.2f}s]   Found storage fourcc: {storage_fourcc:08x}")
-                    
+                    storage_fourcc = read_struct(f_in, "<I")
+                    print(
+                        f"[{time.time() - start_time:.2f}s]   Found storage fourcc: {storage_fourcc:08x}"
+                    )
+
                    if prune_embeddings and storage_fourcc != NULL_INDEX_FOURCC:
                        # Read compact neighbors data
-                        compact_neighbors_data_np = read_numpy_vector(f_in, np.int32, 'i')
-                        print(f"[{time.time() - start_time:.2f}s]   Read compact neighbors data ({compact_neighbors_data_np.size})")
+                        compact_neighbors_data_np = read_numpy_vector(f_in, np.int32, "i")
+                        print(
+                            f"[{time.time() - start_time:.2f}s]   Read compact neighbors data ({compact_neighbors_data_np.size})"
+                        )
                        compact_neighbors_data = compact_neighbors_data_np.tolist()
                        del compact_neighbors_data_np
-                        
+
                        # Skip storage data and write with NULL marker
-                        print(f"[{time.time() - start_time:.2f}s]   Pruning embeddings: Writing NULL storage marker.")
+                        print(
+                            f"[{time.time() - start_time:.2f}s]   Pruning embeddings: Writing NULL storage marker."
+                        )
                        storage_fourcc = NULL_INDEX_FOURCC
                    elif not prune_embeddings:
                        # Read and preserve compact neighbors and storage
-                        compact_neighbors_data_np = read_numpy_vector(f_in, np.int32, 'i')
+                        compact_neighbors_data_np = read_numpy_vector(f_in, np.int32, "i")
                        compact_neighbors_data = compact_neighbors_data_np.tolist()
                        del compact_neighbors_data_np
-                        
+
                        # Read remaining storage data
                        storage_data = f_in.read()
                    else:
                        # Already pruned (NULL storage)
-                        compact_neighbors_data_np = read_numpy_vector(f_in, np.int32, 'i')
+                        compact_neighbors_data_np = read_numpy_vector(f_in, np.int32, "i")
                        compact_neighbors_data = compact_neighbors_data_np.tolist()
                        del compact_neighbors_data_np
-                        storage_data = b''
-                    
+                        storage_data = b""
+
                    # Write the updated compact format
                    print(f"[{time.time() - start_time:.2f}s] Writing updated compact format...")
-                    write_compact_format(f_out, original_hnsw_data, assign_probas_np, cum_nneighbor_per_level_np, 
-                                       levels_np, compact_level_ptr, compact_node_offsets_np, 
-                                       compact_neighbors_data, storage_fourcc, storage_data if not prune_embeddings else b'')
-                    
+                    write_compact_format(
+                        f_out,
+                        original_hnsw_data,
+                        assign_probas_np,
+                        cum_nneighbor_per_level_np,
+                        levels_np,
+                        compact_level_ptr,
+                        compact_node_offsets_np,
+                        compact_neighbors_data,
+                        storage_fourcc,
+                        storage_data if not prune_embeddings else b"",
+                    )
+
                    print(f"[{time.time() - start_time:.2f}s] Conversion complete.")
                    return True
-                    
+
                else:
                    # is_compact=False, rewind and read original format
                    f_in.seek(pos_before_compact)
-                    print(f"[{time.time() - start_time:.2f}s]   Compact flag is False, reading original format...")
-                    
+                    print(
+                        f"[{time.time() - start_time:.2f}s]   Compact flag is False, reading original format..."
+                    )
+
            except EOFError:
                # No compact flag found, assume original format
                f_in.seek(pos_before_compact)
-                print(f"[{time.time() - start_time:.2f}s]   No compact flag found, assuming original format...")
+                print(
+                    f"[{time.time() - start_time:.2f}s]   No compact flag found, assuming original format..."
+                )

            # --- Handle potential extra byte in original format (like C++ code) ---
-            print(f"[{time.time() - start_time:.2f}s]   Probing for potential extra byte before non-compact offsets...")
+            print(
+                f"[{time.time() - start_time:.2f}s]   Probing for potential extra byte before non-compact offsets..."
+            )
            pos_before_probe = f_in.tell()
            try:
-                suspected_flag = read_struct(f_in, '<B')  # Read 1 byte
+                suspected_flag = read_struct(f_in, "<B")  # Read 1 byte
                if suspected_flag == 0x00:
-                    print(f"[{time.time() - start_time:.2f}s]   Found and consumed an unexpected 0x00 byte.")
+                    print(
+                        f"[{time.time() - start_time:.2f}s]   Found and consumed an unexpected 0x00 byte."
+                    )
                elif suspected_flag == 0x01:
-                    print(f"[{time.time() - start_time:.2f}s]   ERROR: Found 0x01 but is_compact should be False")
+                    print(
+                        f"[{time.time() - start_time:.2f}s]   ERROR: Found 0x01 but is_compact should be False"
+                    )
                    raise ValueError("Inconsistent compact flag state")
                else:
                    # Rewind - this byte is part of offsets data
                    f_in.seek(pos_before_probe)
-                    print(f"[{time.time() - start_time:.2f}s]   Rewound to original position (byte was 0x{suspected_flag:02x})")
+                    print(
+                        f"[{time.time() - start_time:.2f}s]   Rewound to original position (byte was 0x{suspected_flag:02x})"
+                    )
            except EOFError:
                f_in.seek(pos_before_probe)
-                print(f"[{time.time() - start_time:.2f}s]   No extra byte found (EOF), proceeding with offsets read")
+                print(
+                    f"[{time.time() - start_time:.2f}s]   No extra byte found (EOF), proceeding with offsets read"
+                )

            # --- Read original format data ---
-            offsets_np = read_numpy_vector(f_in, np.uint64, 'Q')
+            offsets_np = read_numpy_vector(f_in, np.uint64, "Q")
            print(f"[{time.time() - start_time:.2f}s]   Read offsets ({offsets_np.size})")
            if len(offsets_np) != ntotal + 1:
-                 raise ValueError(f"Inconsistent offsets size: len(levels)={ntotal} but len(offsets)={len(offsets_np)}")
+                raise ValueError(
+                    f"Inconsistent offsets size: len(levels)={ntotal} but len(offsets)={len(offsets_np)}"
+                )
            gc.collect()

            print(f"[{time.time() - start_time:.2f}s]   Attempting to read neighbors vector...")
-            neighbors_np = read_numpy_vector(f_in, np.int32, 'i')
+            neighbors_np = read_numpy_vector(f_in, np.int32, "i")
            print(f"[{time.time() - start_time:.2f}s]   Read neighbors ({neighbors_np.size})")
            expected_neighbors_size = offsets_np[-1] if ntotal > 0 else 0
            if neighbors_np.size != expected_neighbors_size:
-                 print(f"Warning: neighbors vector size mismatch. Expected {expected_neighbors_size} based on offsets, got {neighbors_np.size}.")
+                print(
+                    f"Warning: neighbors vector size mismatch. Expected {expected_neighbors_size} based on offsets, got {neighbors_np.size}."
+                )
            gc.collect()

-            original_hnsw_data['entry_point'] = read_struct(f_in, '<i')
-            original_hnsw_data['max_level'] = read_struct(f_in, '<i')
-            original_hnsw_data['efConstruction'] = read_struct(f_in, '<i')
-            original_hnsw_data['efSearch'] = read_struct(f_in, '<i')
-            original_hnsw_data['dummy_upper_beam'] = read_struct(f_in, '<i')
-            print(f"[{time.time() - start_time:.2f}s]   Read scalar params (ep={original_hnsw_data['entry_point']}, max_lvl={original_hnsw_data['max_level']})")
+            original_hnsw_data["entry_point"] = read_struct(f_in, "<i")
+            original_hnsw_data["max_level"] = read_struct(f_in, "<i")
+            original_hnsw_data["efConstruction"] = read_struct(f_in, "<i")
+            original_hnsw_data["efSearch"] = read_struct(f_in, "<i")
+            original_hnsw_data["dummy_upper_beam"] = read_struct(f_in, "<i")
+            print(
+                f"[{time.time() - start_time:.2f}s]   Read scalar params (ep={original_hnsw_data['entry_point']}, max_lvl={original_hnsw_data['max_level']})"
+            )

            print(f"[{time.time() - start_time:.2f}s] Checking for storage data...")
            storage_fourcc = None
            try:
-                storage_fourcc = read_struct(f_in, '<I')
-                print(f"[{time.time() - start_time:.2f}s]   Found storage fourcc: {storage_fourcc:08x}.")
+                storage_fourcc = read_struct(f_in, "<I")
+                print(
+                    f"[{time.time() - start_time:.2f}s]   Found storage fourcc: {storage_fourcc:08x}."
+                )
            except EOFError:
-                 print(f"[{time.time() - start_time:.2f}s]   No storage data found (EOF).")
+                print(f"[{time.time() - start_time:.2f}s]   No storage data found (EOF).")
            except Exception as e:
-                 print(f"[{time.time() - start_time:.2f}s]   Error reading potential storage data: {e}")
-
+                print(
+                    f"[{time.time() - start_time:.2f}s]   Error reading potential storage data: {e}"
+                )

            # --- Perform Conversion ---
            print(f"[{time.time() - start_time:.2f}s] Converting to CSR format...")
@@ -373,17 +475,21 @@ def convert_hnsw_graph_to_csr(input_filename, output_filename, prune_embeddings=

            current_level_ptr_idx = 0
            current_data_idx = 0
-            total_valid_neighbors_counted = 0 # For validation
+            total_valid_neighbors_counted = 0  # For validation

            # Optimize calculation by getting slices once per node if possible
            for i in range(ntotal):
-                if i > 0 and i % (ntotal // 100 or 1) == 0: # Log progress roughly every 1%
+                if i > 0 and i % (ntotal // 100 or 1) == 0:  # Log progress roughly every 1%
                    progress = (i / ntotal) * 100
                    elapsed = time.time() - start_time
-                    print(f"\r[{elapsed:.2f}s]   Converting node {i}/{ntotal} ({progress:.1f}%)...", end="")
+                    print(
+                        f"\r[{elapsed:.2f}s]   Converting node {i}/{ntotal} ({progress:.1f}%)...",
+                        end="",
+                    )

                node_max_level = levels_np[i] - 1
-                if node_max_level < -1: node_max_level = -1
+                if node_max_level < -1:
+                    node_max_level = -1

                node_ptr_start_index = current_level_ptr_idx
                compact_node_offsets_np[i] = node_ptr_start_index
@@ -394,13 +500,17 @@ def convert_hnsw_graph_to_csr(input_filename, output_filename, prune_embeddings=
                for level in range(node_max_level + 1):
                    compact_level_ptr.append(current_data_idx)

-                    begin_orig_np = original_offset_start + get_cum_neighbors(cum_nneighbor_per_level_np, level)
-                    end_orig_np = original_offset_start + get_cum_neighbors(cum_nneighbor_per_level_np, level + 1)
+                    begin_orig_np = original_offset_start + get_cum_neighbors(
+                        cum_nneighbor_per_level_np, level
+                    )
+                    end_orig_np = original_offset_start + get_cum_neighbors(
+                        cum_nneighbor_per_level_np, level + 1
+                    )

                    begin_orig = int(begin_orig_np)
                    end_orig = int(end_orig_np)

-                    neighbors_len = len(neighbors_np) # Cache length
+                    neighbors_len = len(neighbors_np)  # Cache length
                    begin_orig = min(max(0, begin_orig), neighbors_len)
                    end_orig = min(max(begin_orig, end_orig), neighbors_len)

@@ -413,83 +523,117 @@ def convert_hnsw_graph_to_csr(input_filename, output_filename, prune_embeddings=

                        if num_valid > 0:
                            # Append valid neighbors
-                            compact_neighbors_data.extend(level_neighbors_slice[valid_neighbors_mask])
+                            compact_neighbors_data.extend(
+                                level_neighbors_slice[valid_neighbors_mask]
+                            )
                            current_data_idx += num_valid
                            total_valid_neighbors_counted += num_valid

-
                compact_level_ptr.append(current_data_idx)
                current_level_ptr_idx += num_pointers_expected

            compact_node_offsets_np[ntotal] = current_level_ptr_idx
-            print(f"\r[{time.time() - start_time:.2f}s]   Conversion loop finished.                        ") # Clear progress line
+            print(
+                f"\r[{time.time() - start_time:.2f}s]   Conversion loop finished.                        "
+            )  # Clear progress line

            # --- Validation Checks ---
            print(f"[{time.time() - start_time:.2f}s] Running validation checks...")
            valid_check_passed = True
            # Check 1: Total valid neighbors count
-            print(f"    Checking total valid neighbor count...")
+            print("    Checking total valid neighbor count...")
            expected_valid_count = np.sum(neighbors_np >= 0)
            if total_valid_neighbors_counted != len(compact_neighbors_data):
-                 print(f"Error: Mismatch between counted valid neighbors ({total_valid_neighbors_counted}) and final compact_data size ({len(compact_neighbors_data)})!", file=sys.stderr)
-                 valid_check_passed = False
+                print(
+                    f"Error: Mismatch between counted valid neighbors ({total_valid_neighbors_counted}) and final compact_data size ({len(compact_neighbors_data)})!",
+                    file=sys.stderr,
+                )
+                valid_check_passed = False
            if expected_valid_count != len(compact_neighbors_data):
-                 print(f"Error: Mismatch between NumPy count of valid neighbors ({expected_valid_count}) and final compact_data size ({len(compact_neighbors_data)})!", file=sys.stderr)
-                 valid_check_passed = False
+                print(
+                    f"Error: Mismatch between NumPy count of valid neighbors ({expected_valid_count}) and final compact_data size ({len(compact_neighbors_data)})!",
+                    file=sys.stderr,
+                )
+                valid_check_passed = False
            else:
-                 print(f"    OK: Total valid neighbors = {len(compact_neighbors_data)}")
+                print(f"    OK: Total valid neighbors = {len(compact_neighbors_data)}")

            # Check 2: Final pointer indices consistency
-            print(f"    Checking final pointer indices...")
+            print("    Checking final pointer indices...")
            if compact_node_offsets_np[ntotal] != len(compact_level_ptr):
-                 print(f"Error: Final node offset ({compact_node_offsets_np[ntotal]}) doesn't match level_ptr size ({len(compact_level_ptr)})!", file=sys.stderr)
-                 valid_check_passed = False
-            if (len(compact_level_ptr) > 0 and compact_level_ptr[-1] != len(compact_neighbors_data)) or \
-               (len(compact_level_ptr) == 0 and len(compact_neighbors_data) != 0):
-                 last_ptr = compact_level_ptr[-1] if len(compact_level_ptr) > 0 else -1
-                 print(f"Error: Last level pointer ({last_ptr}) doesn't match compact_data size ({len(compact_neighbors_data)})!", file=sys.stderr)
-                 valid_check_passed = False
+                print(
+                    f"Error: Final node offset ({compact_node_offsets_np[ntotal]}) doesn't match level_ptr size ({len(compact_level_ptr)})!",
+                    file=sys.stderr,
+                )
+                valid_check_passed = False
+            if (
+                len(compact_level_ptr) > 0 and compact_level_ptr[-1] != len(compact_neighbors_data)
+            ) or (len(compact_level_ptr) == 0 and len(compact_neighbors_data) != 0):
+                last_ptr = compact_level_ptr[-1] if len(compact_level_ptr) > 0 else -1
+                print(
+                    f"Error: Last level pointer ({last_ptr}) doesn't match compact_data size ({len(compact_neighbors_data)})!",
+                    file=sys.stderr,
+                )
+                valid_check_passed = False
            else:
-                 print(f"    OK: Final pointers match data size.")
+                print("    OK: Final pointers match data size.")

            if not valid_check_passed:
-                print("Error: Validation checks failed. Output file might be incorrect.", file=sys.stderr)
+                print(
+                    "Error: Validation checks failed. Output file might be incorrect.",
+                    file=sys.stderr,
+                )
                # Optional: Exit here if validation fails
                # return False

            # --- Explicitly delete large intermediate arrays ---
-            print(f"[{time.time() - start_time:.2f}s] Deleting original neighbors and offsets arrays...")
+            print(
+                f"[{time.time() - start_time:.2f}s] Deleting original neighbors and offsets arrays..."
+            )
            del neighbors_np
            del offsets_np
            gc.collect()

-            print(f"    CSR Stats: |data|={len(compact_neighbors_data)}, |level_ptr|={len(compact_level_ptr)}")
+            print(
+                f"    CSR Stats: |data|={len(compact_neighbors_data)}, |level_ptr|={len(compact_level_ptr)}"
+            )

            # --- Write CSR HNSW graph data using unified function ---
-            print(f"[{time.time() - start_time:.2f}s] Writing CSR HNSW graph data in FAISS-compatible order...")
-            
+            print(
+                f"[{time.time() - start_time:.2f}s] Writing CSR HNSW graph data in FAISS-compatible order..."
+            )
+
            # Determine storage fourcc and data based on prune_embeddings
            if prune_embeddings:
-                print(f"   Pruning embeddings: Writing NULL storage marker.")
+                print("   Pruning embeddings: Writing NULL storage marker.")
                output_storage_fourcc = NULL_INDEX_FOURCC
-                storage_data = b''
+                storage_data = b""
            else:
                # Keep embeddings - read and preserve original storage data
                if storage_fourcc and storage_fourcc != NULL_INDEX_FOURCC:
-                    print(f"   Preserving embeddings: Reading original storage data...")
+                    print("   Preserving embeddings: Reading original storage data...")
                    storage_data = f_in.read()  # Read remaining storage data
                    output_storage_fourcc = storage_fourcc
                    print(f"   Read {len(storage_data)} bytes of storage data")
                else:
-                    print(f"   No embeddings found in original file (NULL storage)")
+                    print("   No embeddings found in original file (NULL storage)")
                    output_storage_fourcc = NULL_INDEX_FOURCC
-                    storage_data = b''
-            
+                    storage_data = b""
+
            # Use the unified write function
-            write_compact_format(f_out, original_hnsw_data, assign_probas_np, cum_nneighbor_per_level_np, 
-                               levels_np, compact_level_ptr, compact_node_offsets_np, 
-                               compact_neighbors_data, output_storage_fourcc, storage_data)
-            
+            write_compact_format(
+                f_out,
+                original_hnsw_data,
+                assign_probas_np,
+                cum_nneighbor_per_level_np,
+                levels_np,
+                compact_level_ptr,
+                compact_node_offsets_np,
+                compact_neighbors_data,
+                output_storage_fourcc,
+                storage_data,
+            )
+
            # Clean up memory
            del assign_probas_np, cum_nneighbor_per_level_np, levels_np
            del compact_neighbors_data, compact_level_ptr, compact_node_offsets_np
@@ -503,40 +647,63 @@ def convert_hnsw_graph_to_csr(input_filename, output_filename, prune_embeddings=
        print(f"Error: Input file not found: {input_filename}", file=sys.stderr)
        return False
    except MemoryError as e:
-         print(f"\nFatal MemoryError during conversion: {e}. Insufficient RAM.", file=sys.stderr)
-         # Clean up potentially partially written output file?
-         try: os.remove(output_filename)
-         except OSError: pass
-         return False
+        print(f"\nFatal MemoryError during conversion: {e}. Insufficient RAM.", file=sys.stderr)
+        # Clean up potentially partially written output file?
+        try:
+            os.remove(output_filename)
+        except OSError:
+            pass
+        return False
    except EOFError as e:
-        print(f"Error: Reached end of file unexpectedly reading {input_filename}. {e}", file=sys.stderr)
-        try: os.remove(output_filename)
-        except OSError: pass
+        print(
+            f"Error: Reached end of file unexpectedly reading {input_filename}. {e}",
+            file=sys.stderr,
+        )
+        try:
+            os.remove(output_filename)
+        except OSError:
+            pass
        return False
    except Exception as e:
        print(f"An unexpected error occurred during conversion: {e}", file=sys.stderr)
        import traceback
+
        traceback.print_exc()
        try:
            os.remove(output_filename)
-        except OSError: pass
+        except OSError:
+            pass
        return False
    # Ensure neighbors_np is deleted even if an error occurs after its allocation
    finally:
-        if 'neighbors_np' in locals() and neighbors_np is not None:
-            del neighbors_np
-            gc.collect()
+        try:
+            if "neighbors_np" in locals() and neighbors_np is not None:
+                del neighbors_np
+                gc.collect()
+        except NameError:
+            pass


 # --- Script Execution ---
 if __name__ == "__main__":
-    parser = argparse.ArgumentParser(description="Convert a Faiss IndexHNSWFlat file to a CSR-based HNSW graph file.")
+    parser = argparse.ArgumentParser(
+        description="Convert a Faiss IndexHNSWFlat file to a CSR-based HNSW graph file."
+    )
    parser.add_argument("input_index_file", help="Path to the input IndexHNSWFlat file")
-    parser.add_argument("output_csr_graph_file", help="Path to write the output CSR HNSW graph file")
-    parser.add_argument("--prune-embeddings", action="store_true", default=True, 
-                       help="Prune embedding storage (write NULL storage marker)")
-    parser.add_argument("--keep-embeddings", action="store_true", 
-                       help="Keep embedding storage (overrides --prune-embeddings)")
+    parser.add_argument(
+        "output_csr_graph_file", help="Path to write the output CSR HNSW graph file"
+    )
+    parser.add_argument(
+        "--prune-embeddings",
+        action="store_true",
+        default=True,
+        help="Prune embedding storage (write NULL storage marker)",
+    )
+    parser.add_argument(
+        "--keep-embeddings",
+        action="store_true",
+        help="Keep embedding storage (overrides --prune-embeddings)",
+    )

    args = parser.parse_args()

@@ -545,10 +712,12 @@ if __name__ == "__main__":
        sys.exit(1)

    if os.path.abspath(args.input_index_file) == os.path.abspath(args.output_csr_graph_file):
-         print(f"Error: Input and output filenames cannot be the same.", file=sys.stderr)
-         sys.exit(1)
+        print("Error: Input and output filenames cannot be the same.", file=sys.stderr)
+        sys.exit(1)

    prune_embeddings = args.prune_embeddings and not args.keep_embeddings
-    success = convert_hnsw_graph_to_csr(args.input_index_file, args.output_csr_graph_file, prune_embeddings)
+    success = convert_hnsw_graph_to_csr(
+        args.input_index_file, args.output_csr_graph_file, prune_embeddings
+    )
    if not success:
-        sys.exit(1)
+        sys.exit(1)
--- a/packages/leann-backend-hnsw/leann_backend_hnsw/hnsw_backend.py
+++ b/packages/leann-backend-hnsw/leann_backend_hnsw/hnsw_backend.py
@@ -1,19 +1,19 @@
-import numpy as np
-import os
-from pathlib import Path
-from typing import Dict, Any, List, Literal, Optional
-import shutil
 import logging
+import os
+import shutil
+from pathlib import Path
+from typing import Any, Literal

-from leann.searcher_base import BaseSearcher
-from .convert_to_csr import convert_hnsw_graph_to_csr
-
-from leann.registry import register_backend
+import numpy as np
 from leann.interface import (
-    LeannBackendFactoryInterface,
    LeannBackendBuilderInterface,
+    LeannBackendFactoryInterface,
    LeannBackendSearcherInterface,
 )
+from leann.registry import register_backend
+from leann.searcher_base import BaseSearcher
+
+from .convert_to_csr import convert_hnsw_graph_to_csr

 logger = logging.getLogger(__name__)

@@ -51,9 +51,11 @@ class HNSWBuilder(LeannBackendBuilderInterface):
        if not self.is_recompute:
            if self.is_compact:
                # TODO: support this case @andy
-                raise ValueError("is_recompute is False, but is_compact is True. This is not compatible now. change is compact to False and you can use the original HNSW index.")
+                raise ValueError(
+                    "is_recompute is False, but is_compact is True. This is not compatible now. change is compact to False and you can use the original HNSW index."
+                )

-    def build(self, data: np.ndarray, ids: List[str], index_path: str, **kwargs):
+    def build(self, data: np.ndarray, ids: list[str], index_path: str, **kwargs):
        from . import faiss  # type: ignore

        path = Path(index_path)
@@ -99,16 +101,12 @@ class HNSWBuilder(LeannBackendBuilderInterface):
            # index_file_old = index_file.with_suffix(".old")
            # shutil.move(str(index_file), str(index_file_old))
            shutil.move(str(csr_temp_file), str(index_file))
-            logger.info(
-                f"INFO: Replaced original index with {mode_str} version at '{index_file}'"
-            )
+            logger.info(f"INFO: Replaced original index with {mode_str} version at '{index_file}'")
        else:
            # Clean up and fail fast
            if csr_temp_file.exists():
                os.remove(csr_temp_file)
-            raise RuntimeError(
-                "CSR conversion failed - cannot proceed with compact format"
-            )
+            raise RuntimeError("CSR conversion failed - cannot proceed with compact format")


 class HNSWSearcher(BaseSearcher):
@@ -146,7 +144,7 @@ class HNSWSearcher(BaseSearcher):
        self,
        query: np.ndarray,
        top_k: int,
-        zmq_port: Optional[int] = None,
+        zmq_port: int | None = None,
        complexity: int = 64,
        beam_width: int = 1,
        prune_ratio: float = 0.0,
@@ -154,7 +152,7 @@ class HNSWSearcher(BaseSearcher):
        pruning_strategy: Literal["global", "local", "proportional"] = "global",
        batch_size: int = 0,
        **kwargs,
-    ) -> Dict[str, Any]:
+    ) -> dict[str, Any]:
        """
        Search for nearest neighbors using HNSW index.

@@ -183,9 +181,7 @@ class HNSWSearcher(BaseSearcher):
                raise RuntimeError("Recompute is required for pruned index.")
        if recompute_embeddings:
            if zmq_port is None:
-                raise ValueError(
-                    "zmq_port must be provided if recompute_embeddings is True"
-                )
+                raise ValueError("zmq_port must be provided if recompute_embeddings is True")

        if query.dtype != np.float32:
            query = query.astype(np.float32)
@@ -194,9 +190,7 @@ class HNSWSearcher(BaseSearcher):

        params = faiss.SearchParametersHNSW()
        if zmq_port is not None:
-            params.zmq_port = (
-                zmq_port  # C++ code won't use this if recompute_embeddings is False
-            )
+            params.zmq_port = zmq_port  # C++ code won't use this if recompute_embeddings is False
        params.efSearch = complexity
        params.beam_size = beam_width

@@ -209,9 +203,7 @@ class HNSWSearcher(BaseSearcher):
            params.send_neigh_times_ratio = 0.0
        elif pruning_strategy == "proportional":
            params.local_prune = False
-            params.send_neigh_times_ratio = (
-                1.0  # Any value > 1e-6 triggers proportional mode
-            )
+            params.send_neigh_times_ratio = 1.0  # Any value > 1e-6 triggers proportional mode
        else:  # "global"
            params.local_prune = False
            params.send_neigh_times_ratio = 0.0
@@ -232,8 +224,6 @@ class HNSWSearcher(BaseSearcher):
            params,
        )

-        string_labels = [
-            [str(int_label) for int_label in batch_labels] for batch_labels in labels
-        ]
+        string_labels = [[str(int_label) for int_label in batch_labels] for batch_labels in labels]

        return {"labels": string_labels, "distances": distances}
--- a/packages/leann-backend-hnsw/leann_backend_hnsw/hnsw_embedding_server.py
+++ b/packages/leann-backend-hnsw/leann_backend_hnsw/hnsw_embedding_server.py
@@ -3,17 +3,17 @@ HNSW-specific embedding server
 """

 import argparse
+import json
+import logging
+import os
+import sys
 import threading
 import time
-import os
-import zmq
-import numpy as np
-import msgpack
-import json
 from pathlib import Path
-from typing import Optional
-import sys
-import logging
+
+import msgpack
+import numpy as np
+import zmq

 # Set up logging based on environment variable
 LOG_LEVEL = os.getenv("LEANN_LOG_LEVEL", "WARNING").upper()
@@ -33,7 +33,7 @@ if not logger.handlers:


 def create_hnsw_embedding_server(
-    passages_file: Optional[str] = None,
+    passages_file: str | None = None,
    zmq_port: int = 5555,
    model_name: str = "sentence-transformers/all-mpnet-base-v2",
    distance_metric: str = "mips",
@@ -52,8 +52,8 @@ def create_hnsw_embedding_server(
    sys.path.insert(0, str(leann_core_path))

    try:
-        from leann.embedding_compute import compute_embeddings
        from leann.api import PassageManager
+        from leann.embedding_compute import compute_embeddings

        logger.info("Successfully imported unified embedding computation module")
    except ImportError as e:
@@ -78,13 +78,11 @@ def create_hnsw_embedding_server(
        raise ValueError("Only metadata files (.meta.json) are supported")

    # Load metadata to get passage sources
-    with open(passages_file, "r") as f:
+    with open(passages_file) as f:
        meta = json.load(f)

    # Convert relative paths to absolute paths based on metadata file location
-    metadata_dir = Path(
-        passages_file
-    ).parent.parent  # Go up one level from the metadata file
+    metadata_dir = Path(passages_file).parent.parent  # Go up one level from the metadata file
    passage_sources = []
    for source in meta["passage_sources"]:
        source_copy = source.copy()
@@ -134,9 +132,7 @@ def create_hnsw_embedding_server(
                        response = embeddings.tolist()
                        socket.send(msgpack.packb(response))
                        e2e_end = time.time()
-                        logger.info(
-                            f"⏱️  Text embedding E2E time: {e2e_end - e2e_start:.6f}s"
-                        )
+                        logger.info(f"⏱️  Text embedding E2E time: {e2e_end - e2e_start:.6f}s")
                        continue

                # Handle distance calculation requests
@@ -162,17 +158,13 @@ def create_hnsw_embedding_server(
                            texts.append(txt)
                        except KeyError:
                            logger.error(f"Passage ID {nid} not found")
-                            raise RuntimeError(
-                                f"FATAL: Passage with ID {nid} not found"
-                            )
+                            raise RuntimeError(f"FATAL: Passage with ID {nid} not found")
                        except Exception as e:
                            logger.error(f"Exception looking up passage ID {nid}: {e}")
                            raise

                    # Process embeddings
-                    embeddings = compute_embeddings(
-                        texts, model_name, mode=embedding_mode
-                    )
+                    embeddings = compute_embeddings(texts, model_name, mode=embedding_mode)
                    logger.info(
                        f"Computed embeddings for {len(texts)} texts, shape: {embeddings.shape}"
                    )
@@ -186,18 +178,12 @@ def create_hnsw_embedding_server(
                        distances = -np.dot(embeddings, query_vector)

                    response_payload = distances.flatten().tolist()
-                    response_bytes = msgpack.packb(
-                        [response_payload], use_single_float=True
-                    )
-                    logger.debug(
-                        f"Sending distance response with {len(distances)} distances"
-                    )
+                    response_bytes = msgpack.packb([response_payload], use_single_float=True)
+                    logger.debug(f"Sending distance response with {len(distances)} distances")

                    socket.send(response_bytes)
                    e2e_end = time.time()
-                    logger.info(
-                        f"⏱️  Distance calculation E2E time: {e2e_end - e2e_start:.6f}s"
-                    )
+                    logger.info(f"⏱️  Distance calculation E2E time: {e2e_end - e2e_start:.6f}s")
                    continue

                # Standard embedding request (passage ID lookup)
@@ -222,9 +208,7 @@ def create_hnsw_embedding_server(
                        passage_data = passages.get_passage(str(nid))
                        txt = passage_data["text"]
                        if not txt:
-                            raise RuntimeError(
-                                f"FATAL: Empty text for passage ID {nid}"
-                            )
+                            raise RuntimeError(f"FATAL: Empty text for passage ID {nid}")
                        texts.append(txt)
                    except KeyError:
                        raise RuntimeError(f"FATAL: Passage with ID {nid} not found")
@@ -243,11 +227,9 @@ def create_hnsw_embedding_server(
                    logger.error(
                        f"NaN or Inf detected in embeddings! Requested IDs: {node_ids[:5]}..."
                    )
-                    assert False
+                    raise AssertionError()

-                hidden_contiguous_f32 = np.ascontiguousarray(
-                    embeddings, dtype=np.float32
-                )
+                hidden_contiguous_f32 = np.ascontiguousarray(embeddings, dtype=np.float32)
                response_payload = [
                    list(hidden_contiguous_f32.shape),
                    hidden_contiguous_f32.flatten().tolist(),
--- a/packages/leann-core/pyproject.toml
+++ b/packages/leann-core/pyproject.toml
@@ -35,9 +35,9 @@ dependencies = [

 [project.optional-dependencies]
 colab = [
-    "torch>=2.0.0,<3.0.0",  # 限制torch版本避免冲突
-    "transformers>=4.30.0,<5.0.0",  # 限制transformers版本
-    "accelerate>=0.20.0,<1.0.0",  # 限制accelerate版本
+    "torch>=2.0.0,<3.0.0",  # Limit torch version to avoid conflicts
+    "transformers>=4.30.0,<5.0.0",  # Limit transformers version
+    "accelerate>=0.20.0,<1.0.0",  # Limit accelerate version
 ]

 [project.scripts]
--- a/packages/leann-core/src/leann/init.py
+++ b/packages/leann-core/src/leann/init.py
@@ -14,4 +14,4 @@ from .registry import BACKEND_REGISTRY, autodiscover_backends

 autodiscover_backends()

-__all__ = ["LeannBuilder", "LeannSearcher", "LeannChat", "BACKEND_REGISTRY"]
+__all__ = ["BACKEND_REGISTRY", "LeannBuilder", "LeannChat", "LeannSearcher"]
--- a/packages/leann-core/src/leann/api.py
+++ b/packages/leann-core/src/leann/api.py
@@ -4,27 +4,30 @@ with the correct, original embedding logic from the user's reference code.
 """

 import json
-import pickle
-from leann.interface import LeannBackendSearcherInterface
-import numpy as np
-import time
-from pathlib import Path
-from typing import List, Dict, Any, Optional, Literal
-from dataclasses import dataclass, field
-from .registry import BACKEND_REGISTRY
-from .interface import LeannBackendFactoryInterface
-from .chat import get_llm
 import logging
+import pickle
+import time
+from dataclasses import dataclass, field
+from pathlib import Path
+from typing import Any, Literal
+
+import numpy as np
+
+from leann.interface import LeannBackendSearcherInterface
+
+from .chat import get_llm
+from .interface import LeannBackendFactoryInterface
+from .registry import BACKEND_REGISTRY

 logger = logging.getLogger(__name__)


 def compute_embeddings(
-    chunks: List[str],
+    chunks: list[str],
    model_name: str,
    mode: str = "sentence-transformers",
    use_server: bool = True,
-    port: Optional[int] = None,
+    port: int | None = None,
    is_build=False,
 ) -> np.ndarray:
    """
@@ -61,9 +64,7 @@ def compute_embeddings(
        )


-def compute_embeddings_via_server(
-    chunks: List[str], model_name: str, port: int
-) -> np.ndarray:
+def compute_embeddings_via_server(chunks: list[str], model_name: str, port: int) -> np.ndarray:
    """Computes embeddings using sentence-transformers.

    Args:
@@ -73,9 +74,9 @@ def compute_embeddings_via_server(
    logger.info(
        f"Computing embeddings for {len(chunks)} chunks using SentenceTransformer model '{model_name}' (via embedding server)..."
    )
-    import zmq
    import msgpack
    import numpy as np
+    import zmq

    # Connect to embedding server
    context = zmq.Context()
@@ -104,11 +105,11 @@ class SearchResult:
    id: str
    score: float
    text: str
-    metadata: Dict[str, Any] = field(default_factory=dict)
+    metadata: dict[str, Any] = field(default_factory=dict)


 class PassageManager:
-    def __init__(self, passage_sources: List[Dict[str, Any]]):
+    def __init__(self, passage_sources: list[dict[str, Any]]):
        self.offset_maps = {}
        self.passage_files = {}
        self.global_offset_map = {}  # Combined map for fast lookup
@@ -117,15 +118,15 @@ class PassageManager:
            assert source["type"] == "jsonl", "only jsonl is supported"
            passage_file = source["path"]
            index_file = source["index_path"]  # .idx file
-            
+
            # Fix path resolution for Colab and other environments
            if not Path(index_file).is_absolute():
                # If relative path, try to resolve it properly
                index_file = str(Path(index_file).resolve())
-            
+
            if not Path(index_file).exists():
                raise FileNotFoundError(f"Passage index file not found: {index_file}")
-            
+
            with open(index_file, "rb") as f:
                offset_map = pickle.load(f)
                self.offset_maps[passage_file] = offset_map
@@ -135,11 +136,11 @@ class PassageManager:
                for passage_id, offset in offset_map.items():
                    self.global_offset_map[passage_id] = (passage_file, offset)

-    def get_passage(self, passage_id: str) -> Dict[str, Any]:
+    def get_passage(self, passage_id: str) -> dict[str, Any]:
        if passage_id in self.global_offset_map:
            passage_file, offset = self.global_offset_map[passage_id]
            # Lazy file opening - only open when needed
-            with open(passage_file, "r", encoding="utf-8") as f:
+            with open(passage_file, encoding="utf-8") as f:
                f.seek(offset)
                return json.loads(f.readline())
        raise KeyError(f"Passage ID not found: {passage_id}")
@@ -150,14 +151,12 @@ class LeannBuilder:
        self,
        backend_name: str,
        embedding_model: str = "facebook/contriever",
-        dimensions: Optional[int] = None,
+        dimensions: int | None = None,
        embedding_mode: str = "sentence-transformers",
        **backend_kwargs,
    ):
        self.backend_name = backend_name
-        backend_factory: LeannBackendFactoryInterface | None = BACKEND_REGISTRY.get(
-            backend_name
-        )
+        backend_factory: LeannBackendFactoryInterface | None = BACKEND_REGISTRY.get(backend_name)
        if backend_factory is None:
            raise ValueError(f"Backend '{backend_name}' not found or not registered.")
        self.backend_factory = backend_factory
@@ -165,9 +164,9 @@ class LeannBuilder:
        self.dimensions = dimensions
        self.embedding_mode = embedding_mode
        self.backend_kwargs = backend_kwargs
-        self.chunks: List[Dict[str, Any]] = []
+        self.chunks: list[dict[str, Any]] = []

-    def add_text(self, text: str, metadata: Optional[Dict[str, Any]] = None):
+    def add_text(self, text: str, metadata: dict[str, Any] | None = None):
        if metadata is None:
            metadata = {}
        passage_id = metadata.get("id", str(len(self.chunks)))
@@ -197,9 +196,7 @@ class LeannBuilder:
            try:
                from tqdm import tqdm

-                chunk_iterator = tqdm(
-                    self.chunks, desc="Writing passages", unit="chunk"
-                )
+                chunk_iterator = tqdm(self.chunks, desc="Writing passages", unit="chunk")
            except ImportError:
                chunk_iterator = self.chunks

@@ -229,9 +226,7 @@ class LeannBuilder:
        string_ids = [chunk["id"] for chunk in self.chunks]
        current_backend_kwargs = {**self.backend_kwargs, "dimensions": self.dimensions}
        builder_instance = self.backend_factory.builder(**current_backend_kwargs)
-        builder_instance.build(
-            embeddings, string_ids, index_path, **current_backend_kwargs
-        )
+        builder_instance.build(embeddings, string_ids, index_path, **current_backend_kwargs)
        leann_meta_path = index_dir / f"{index_name}.meta.json"
        meta_data = {
            "version": "1.0",
@@ -280,9 +275,7 @@ class LeannBuilder:
        ids, embeddings = data

        if not isinstance(embeddings, np.ndarray):
-            raise ValueError(
-                f"Expected embeddings to be numpy array, got {type(embeddings)}"
-            )
+            raise ValueError(f"Expected embeddings to be numpy array, got {type(embeddings)}")

        if len(ids) != embeddings.shape[0]:
            raise ValueError(
@@ -294,9 +287,7 @@ class LeannBuilder:
        if self.dimensions is None:
            self.dimensions = embedding_dim
        elif self.dimensions != embedding_dim:
-            raise ValueError(
-                f"Dimension mismatch: expected {self.dimensions}, got {embedding_dim}"
-            )
+            raise ValueError(f"Dimension mismatch: expected {self.dimensions}, got {embedding_dim}")

        logger.info(
            f"Building index from precomputed embeddings: {len(ids)} items, {embedding_dim} dimensions"
@@ -381,9 +372,7 @@ class LeannBuilder:
        with open(leann_meta_path, "w", encoding="utf-8") as f:
            json.dump(meta_data, f, indent=2)

-        logger.info(
-            f"Index built successfully from precomputed embeddings: {index_path}"
-        )
+        logger.info(f"Index built successfully from precomputed embeddings: {index_path}")


 class LeannSearcher:
@@ -391,20 +380,16 @@ class LeannSearcher:
        # Fix path resolution for Colab and other environments
        if not Path(index_path).is_absolute():
            index_path = str(Path(index_path).resolve())
-        
+
        self.meta_path_str = f"{index_path}.meta.json"
        if not Path(self.meta_path_str).exists():
-            raise FileNotFoundError(
-                f"Leann metadata file not found at {self.meta_path_str}"
-            )
-        with open(self.meta_path_str, "r", encoding="utf-8") as f:
+            raise FileNotFoundError(f"Leann metadata file not found at {self.meta_path_str}")
+        with open(self.meta_path_str, encoding="utf-8") as f:
            self.meta_data = json.load(f)
        backend_name = self.meta_data["backend_name"]
        self.embedding_model = self.meta_data["embedding_model"]
        # Support both old and new format
-        self.embedding_mode = self.meta_data.get(
-            "embedding_mode", "sentence-transformers"
-        )
+        self.embedding_mode = self.meta_data.get("embedding_mode", "sentence-transformers")
        self.passage_manager = PassageManager(self.meta_data.get("passage_sources", []))
        backend_factory = BACKEND_REGISTRY.get(backend_name)
        if backend_factory is None:
@@ -426,7 +411,7 @@ class LeannSearcher:
        pruning_strategy: Literal["global", "local", "proportional"] = "global",
        expected_zmq_port: int = 5557,
        **kwargs,
-    ) -> List[SearchResult]:
+    ) -> list[SearchResult]:
        logger.info("🔍 LeannSearcher.search() called:")
        logger.info(f"  Query: '{query}'")
        logger.info(f"  Top_k: {top_k}")
@@ -453,7 +438,7 @@ class LeannSearcher:
            zmq_port=zmq_port,
        )
        # logger.info(f"  Generated embedding shape: {query_embedding.shape}")
-        embedding_time = time.time() - start_time
+        time.time() - start_time
        # logger.info(f"  Embedding time: {embedding_time} seconds")

        start_time = time.time()
@@ -468,17 +453,15 @@ class LeannSearcher:
            zmq_port=zmq_port,
            **kwargs,
        )
-        search_time = time.time() - start_time
+        time.time() - start_time
        # logger.info(f"  Search time: {search_time} seconds")
-        logger.info(
-            f"  Backend returned: labels={len(results.get('labels', [[]])[0])} results"
-        )
+        logger.info(f"  Backend returned: labels={len(results.get('labels', [[]])[0])} results")

        enriched_results = []
        if "labels" in results and "distances" in results:
            logger.info(f"  Processing {len(results['labels'][0])} passage IDs:")
            for i, (string_id, dist) in enumerate(
-                zip(results["labels"][0], results["distances"][0])
+                zip(results["labels"][0], results["distances"][0], strict=False)
            ):
                try:
                    passage_data = self.passage_manager.get_passage(string_id)
@@ -490,15 +473,15 @@ class LeannSearcher:
                            metadata=passage_data.get("metadata", {}),
                        )
                    )
-                    
+
                    # Color codes for better logging
                    GREEN = "\033[92m"
                    BLUE = "\033[94m"
                    YELLOW = "\033[93m"
                    RESET = "\033[0m"
-                    
+
                    # Truncate text for display (first 100 chars)
-                    display_text = passage_data['text']
+                    display_text = passage_data["text"]
                    logger.info(
                        f"   {GREEN}✓{RESET} {BLUE}[{i + 1:2d}]{RESET} {YELLOW}ID:{RESET} '{string_id}' {YELLOW}Score:{RESET} {dist:.4f} {YELLOW}Text:{RESET} {display_text}"
                    )
@@ -516,7 +499,7 @@ class LeannChat:
    def __init__(
        self,
        index_path: str,
-        llm_config: Optional[Dict[str, Any]] = None,
+        llm_config: dict[str, Any] | None = None,
        enable_warmup: bool = False,
        **kwargs,
    ):
@@ -532,7 +515,7 @@ class LeannChat:
        prune_ratio: float = 0.0,
        recompute_embeddings: bool = True,
        pruning_strategy: Literal["global", "local", "proportional"] = "global",
-        llm_kwargs: Optional[Dict[str, Any]] = None,
+        llm_kwargs: dict[str, Any] | None = None,
        expected_zmq_port: int = 5557,
        **search_kwargs,
    ):
--- a/packages/leann-core/src/leann/chat.py
+++ b/packages/leann-core/src/leann/chat.py
@@ -4,11 +4,12 @@ This file contains the chat generation logic for the LEANN project,
 supporting different backends like Ollama, Hugging Face Transformers, and a simulation mode.
 """

-from abc import ABC, abstractmethod
-from typing import Dict, Any, Optional, List
+import difflib
 import logging
 import os
-import difflib
+from abc import ABC, abstractmethod
+from typing import Any
+
 import torch

 # Configure logging
@@ -16,10 +17,11 @@ logging.basicConfig(level=logging.INFO)
 logger = logging.getLogger(__name__)


-def check_ollama_models() -> List[str]:
+def check_ollama_models() -> list[str]:
    """Check available Ollama models and return a list"""
    try:
        import requests
+
        response = requests.get("http://localhost:11434/api/tags", timeout=5)
        if response.status_code == 200:
            data = response.json()
@@ -31,51 +33,52 @@ def check_ollama_models() -> List[str]:

 def check_ollama_model_exists_remotely(model_name: str) -> tuple[bool, list[str]]:
    """Check if a model exists in Ollama's remote library and return available tags
-    
+
    Returns:
        (model_exists, available_tags): bool and list of matching tags
    """
    try:
-        import requests
        import re
-        
+
+        import requests
+
        # Split model name and tag
-        if ':' in model_name:
-            base_model, requested_tag = model_name.split(':', 1)
+        if ":" in model_name:
+            base_model, requested_tag = model_name.split(":", 1)
        else:
            base_model, requested_tag = model_name, None
-        
+
        # First check if base model exists in library
        library_response = requests.get("https://ollama.com/library", timeout=8)
        if library_response.status_code != 200:
            return True, []  # Assume exists if can't check
-            
+
        # Extract model names from library page
        models_in_library = re.findall(r'href="/library/([^"]+)"', library_response.text)
-        
+
        if base_model not in models_in_library:
            return False, []  # Base model doesn't exist
-        
+
        # If base model exists, get available tags
        tags_response = requests.get(f"https://ollama.com/library/{base_model}/tags", timeout=8)
        if tags_response.status_code != 200:
            return True, []  # Base model exists but can't get tags
-            
+
        # Extract tags for this model - be more specific to avoid HTML artifacts
-        tag_pattern = rf'{re.escape(base_model)}:[a-zA-Z0-9\.\-_]+'
+        tag_pattern = rf"{re.escape(base_model)}:[a-zA-Z0-9\.\-_]+"
        raw_tags = re.findall(tag_pattern, tags_response.text)
-        
+
        # Clean up tags - remove HTML artifacts and duplicates
        available_tags = []
        seen = set()
        for tag in raw_tags:
            # Skip if it looks like HTML (contains < or >)
-            if '<' in tag or '>' in tag:
+            if "<" in tag or ">" in tag:
                continue
            if tag not in seen:
                seen.add(tag)
                available_tags.append(tag)
-        
+
        # Check if exact model exists
        if requested_tag is None:
            # User just requested base model, suggest tags
@@ -83,76 +86,80 @@ def check_ollama_model_exists_remotely(model_name: str) -> tuple[bool, list[str]
        else:
            exact_match = model_name in available_tags
            return exact_match, available_tags[:10]
-            
+
    except Exception:
        pass
-    
+
    # If scraping fails, assume model might exist (don't block user)
    return True, []


-def search_ollama_models_fuzzy(query: str, available_models: List[str]) -> List[str]:
+def search_ollama_models_fuzzy(query: str, available_models: list[str]) -> list[str]:
    """Use intelligent fuzzy search for Ollama models"""
    if not available_models:
        return []
-    
+
    query_lower = query.lower()
    suggestions = []
-    
+
    # 1. Exact matches first
    exact_matches = [m for m in available_models if query_lower == m.lower()]
    suggestions.extend(exact_matches)
-    
+
    # 2. Starts with query
-    starts_with = [m for m in available_models if m.lower().startswith(query_lower) and m not in suggestions]
+    starts_with = [
+        m for m in available_models if m.lower().startswith(query_lower) and m not in suggestions
+    ]
    suggestions.extend(starts_with)
-    
+
    # 3. Contains query
    contains = [m for m in available_models if query_lower in m.lower() and m not in suggestions]
    suggestions.extend(contains)
-    
+
    # 4. Base model name matching (remove version numbers)
    def get_base_name(model_name: str) -> str:
        """Extract base name without version (e.g., 'llama3:8b' -> 'llama3')"""
-        return model_name.split(':')[0].split('-')[0]
-    
+        return model_name.split(":")[0].split("-")[0]
+
    query_base = get_base_name(query_lower)
    base_matches = [
-        m for m in available_models 
+        m
+        for m in available_models
        if get_base_name(m.lower()) == query_base and m not in suggestions
    ]
    suggestions.extend(base_matches)
-    
+
    # 5. Family/variant matching
    model_families = {
-        'llama': ['llama2', 'llama3', 'alpaca', 'vicuna', 'codellama'],
-        'qwen': ['qwen', 'qwen2', 'qwen3'],
-        'gemma': ['gemma', 'gemma2'],
-        'phi': ['phi', 'phi2', 'phi3'],
-        'mistral': ['mistral', 'mixtral', 'openhermes'],
-        'dolphin': ['dolphin', 'openchat'],
-        'deepseek': ['deepseek', 'deepseek-coder']
+        "llama": ["llama2", "llama3", "alpaca", "vicuna", "codellama"],
+        "qwen": ["qwen", "qwen2", "qwen3"],
+        "gemma": ["gemma", "gemma2"],
+        "phi": ["phi", "phi2", "phi3"],
+        "mistral": ["mistral", "mixtral", "openhermes"],
+        "dolphin": ["dolphin", "openchat"],
+        "deepseek": ["deepseek", "deepseek-coder"],
    }
-    
+
    query_family = None
    for family, variants in model_families.items():
        if any(variant in query_lower for variant in variants):
            query_family = family
            break
-    
+
    if query_family:
        family_variants = model_families[query_family]
        family_matches = [
-            m for m in available_models
+            m
+            for m in available_models
            if any(variant in m.lower() for variant in family_variants) and m not in suggestions
        ]
        suggestions.extend(family_matches)
-    
+
    # 6. Use difflib for remaining fuzzy matches
    remaining_models = [m for m in available_models if m not in suggestions]
    difflib_matches = difflib.get_close_matches(query_lower, remaining_models, n=3, cutoff=0.4)
    suggestions.extend(difflib_matches)
-    
+
    return suggestions[:8]  # Return top 8 suggestions


@@ -162,15 +169,13 @@ def search_ollama_models_fuzzy(query: str, available_models: List[str]) -> List[
 # Remove this too - no need for fallback


-def suggest_similar_models(invalid_model: str, available_models: List[str]) -> List[str]:
+def suggest_similar_models(invalid_model: str, available_models: list[str]) -> list[str]:
    """Use difflib to find similar model names"""
    if not available_models:
        return []
-    
+
    # Get close matches using fuzzy matching
-    suggestions = difflib.get_close_matches(
-        invalid_model, available_models, n=3, cutoff=0.3
-    )
+    suggestions = difflib.get_close_matches(invalid_model, available_models, n=3, cutoff=0.3)
    return suggestions


@@ -178,49 +183,50 @@ def check_hf_model_exists(model_name: str) -> bool:
    """Quick check if HuggingFace model exists without downloading"""
    try:
        from huggingface_hub import model_info
+
        model_info(model_name)
        return True
    except Exception:
        return False


-def get_popular_hf_models() -> List[str]:
+def get_popular_hf_models() -> list[str]:
    """Return a list of popular HuggingFace models for suggestions"""
    try:
        from huggingface_hub import list_models
-        
+
        # Get popular text-generation models, sorted by downloads
        models = list_models(
            filter="text-generation",
            sort="downloads",
            direction=-1,
-            limit=20  # Get top 20 most downloaded
+            limit=20,  # Get top 20 most downloaded
        )
-        
+
        # Extract model names and filter for chat/conversation models
        model_names = []
-        chat_keywords = ['chat', 'instruct', 'dialog', 'conversation', 'assistant']
-        
+        chat_keywords = ["chat", "instruct", "dialog", "conversation", "assistant"]
+
        for model in models:
-            model_name = model.id if hasattr(model, 'id') else str(model)
+            model_name = model.id if hasattr(model, "id") else str(model)
            # Prioritize models with chat-related keywords
            if any(keyword in model_name.lower() for keyword in chat_keywords):
                model_names.append(model_name)
            elif len(model_names) < 10:  # Fill up with other popular models
                model_names.append(model_name)
-                
+
        return model_names[:10] if model_names else _get_fallback_hf_models()
-        
+
    except Exception:
        # Fallback to static list if API call fails
        return _get_fallback_hf_models()


-def _get_fallback_hf_models() -> List[str]:
+def _get_fallback_hf_models() -> list[str]:
    """Fallback list of popular HuggingFace models"""
    return [
        "microsoft/DialoGPT-medium",
-        "microsoft/DialoGPT-large", 
+        "microsoft/DialoGPT-large",
        "facebook/blenderbot-400M-distill",
        "microsoft/phi-2",
        "deepseek-ai/deepseek-llm-7b-chat",
@@ -228,44 +234,40 @@ def _get_fallback_hf_models() -> List[str]:
        "facebook/blenderbot_small-90M",
        "microsoft/phi-1_5",
        "facebook/opt-350m",
-        "EleutherAI/gpt-neo-1.3B"
+        "EleutherAI/gpt-neo-1.3B",
    ]


-def search_hf_models_fuzzy(query: str, limit: int = 10) -> List[str]:
+def search_hf_models_fuzzy(query: str, limit: int = 10) -> list[str]:
    """Use HuggingFace Hub's native fuzzy search for model suggestions"""
    try:
        from huggingface_hub import list_models
-        
+
        # HF Hub's search is already fuzzy! It handles typos and partial matches
        models = list_models(
-            search=query,
-            filter="text-generation",
-            sort="downloads", 
-            direction=-1,
-            limit=limit
+            search=query, filter="text-generation", sort="downloads", direction=-1, limit=limit
        )
-        
-        model_names = [model.id if hasattr(model, 'id') else str(model) for model in models]
-        
+
+        model_names = [model.id if hasattr(model, "id") else str(model) for model in models]
+
        # If direct search doesn't return enough results, try some variations
        if len(model_names) < 3:
            # Try searching for partial matches or common variations
            variations = []
-            
+
            # Extract base name (e.g., "gpt3" from "gpt-3.5")
-            base_query = query.lower().replace('-', '').replace('.', '').replace('_', '')
+            base_query = query.lower().replace("-", "").replace(".", "").replace("_", "")
            if base_query != query.lower():
                variations.append(base_query)
-            
+
            # Try common model name patterns
-            if 'gpt' in query.lower():
-                variations.extend(['gpt2', 'gpt-neo', 'gpt-j', 'dialoGPT'])
-            elif 'llama' in query.lower():
-                variations.extend(['llama2', 'alpaca', 'vicuna'])
-            elif 'bert' in query.lower():
-                variations.extend(['roberta', 'distilbert', 'albert'])
-            
+            if "gpt" in query.lower():
+                variations.extend(["gpt2", "gpt-neo", "gpt-j", "dialoGPT"])
+            elif "llama" in query.lower():
+                variations.extend(["llama2", "alpaca", "vicuna"])
+            elif "bert" in query.lower():
+                variations.extend(["roberta", "distilbert", "albert"])
+
            # Search with variations
            for var in variations[:2]:  # Limit to 2 variations to avoid too many API calls
                try:
@@ -274,13 +276,15 @@ def search_hf_models_fuzzy(query: str, limit: int = 10) -> List[str]:
                        filter="text-generation",
                        sort="downloads",
                        direction=-1,
-                        limit=3
+                        limit=3,
                    )
-                    var_names = [model.id if hasattr(model, 'id') else str(model) for model in var_models]
+                    var_names = [
+                        model.id if hasattr(model, "id") else str(model) for model in var_models
+                    ]
                    model_names.extend(var_names)
-                except:
+                except Exception:
                    continue
-        
+
        # Remove duplicates while preserving order
        seen = set()
        unique_models = []
@@ -288,65 +292,67 @@ def search_hf_models_fuzzy(query: str, limit: int = 10) -> List[str]:
            if model not in seen:
                seen.add(model)
                unique_models.append(model)
-        
+
        return unique_models[:limit]
-        
+
    except Exception:
        # If search fails, return empty list
        return []


-def search_hf_models(query: str, limit: int = 10) -> List[str]:
+def search_hf_models(query: str, limit: int = 10) -> list[str]:
    """Simple search for HuggingFace models based on query (kept for backward compatibility)"""
    return search_hf_models_fuzzy(query, limit)


-def validate_model_and_suggest(model_name: str, llm_type: str) -> Optional[str]:
+def validate_model_and_suggest(model_name: str, llm_type: str) -> str | None:
    """Validate model name and provide suggestions if invalid"""
    if llm_type == "ollama":
        available_models = check_ollama_models()
        if available_models and model_name not in available_models:
            error_msg = f"Model '{model_name}' not found in your local Ollama installation."
-            
+
            # Check if the model exists remotely and get available tags
            model_exists_remotely, available_tags = check_ollama_model_exists_remotely(model_name)
-            
+
            if model_exists_remotely and model_name in available_tags:
                # Exact model exists remotely - suggest pulling it
-                error_msg += f"\n\nTo install the requested model:\n"
+                error_msg += "\n\nTo install the requested model:\n"
                error_msg += f"  ollama pull {model_name}\n"
-                
+
                # Show local alternatives
                suggestions = search_ollama_models_fuzzy(model_name, available_models)
                if suggestions:
                    error_msg += "\nOr use one of these similar installed models:\n"
                    for i, suggestion in enumerate(suggestions, 1):
                        error_msg += f"  {i}. {suggestion}\n"
-                        
+
            elif model_exists_remotely and available_tags:
                # Base model exists but requested tag doesn't - suggest correct tags
-                base_model = model_name.split(':')[0]
-                requested_tag = model_name.split(':', 1)[1] if ':' in model_name else None
-                
-                error_msg += f"\n\nModel '{base_model}' exists, but tag '{requested_tag}' is not available."
+                base_model = model_name.split(":")[0]
+                requested_tag = model_name.split(":", 1)[1] if ":" in model_name else None
+
+                error_msg += (
+                    f"\n\nModel '{base_model}' exists, but tag '{requested_tag}' is not available."
+                )
                error_msg += f"\n\nAvailable {base_model} models you can install:\n"
                for i, tag in enumerate(available_tags[:8], 1):
                    error_msg += f"  {i}. ollama pull {tag}\n"
                if len(available_tags) > 8:
                    error_msg += f"  ... and {len(available_tags) - 8} more variants\n"
-                    
+
                # Also show local alternatives
                suggestions = search_ollama_models_fuzzy(model_name, available_models)
                if suggestions:
                    error_msg += "\nOr use one of these similar installed models:\n"
                    for i, suggestion in enumerate(suggestions, 1):
                        error_msg += f"  {i}. {suggestion}\n"
-                        
+
            else:
                # Model doesn't exist remotely - show fuzzy suggestions
                suggestions = search_ollama_models_fuzzy(model_name, available_models)
                error_msg += f"\n\nModel '{model_name}' was not found in Ollama's library."
-                
+
                if suggestions:
                    error_msg += "\n\nDid you mean one of these installed models?\n"
                    for i, suggestion in enumerate(suggestions, 1):
@@ -357,23 +363,25 @@ def validate_model_and_suggest(model_name: str, llm_type: str) -> Optional[str]:
                        error_msg += f"  {i}. {model}\n"
                    if len(available_models) > 8:
                        error_msg += f"  ... and {len(available_models) - 8} more\n"
-            
+
            error_msg += "\n\nCommands:"
            error_msg += "\n  ollama list                    # List installed models"
            if model_exists_remotely and available_tags:
                if model_name in available_tags:
                    error_msg += f"\n  ollama pull {model_name}          # Install requested model"
                else:
-                    error_msg += f"\n  ollama pull {available_tags[0]}    # Install recommended variant"
+                    error_msg += (
+                        f"\n  ollama pull {available_tags[0]}    # Install recommended variant"
+                    )
            error_msg += "\n  https://ollama.com/library     # Browse available models"
            return error_msg
-            
+
    elif llm_type == "hf":
        # For HF models, we can do a quick existence check
        if not check_hf_model_exists(model_name):
            # Use HF Hub's native fuzzy search directly
            search_suggestions = search_hf_models_fuzzy(model_name, limit=8)
-            
+
            error_msg = f"Model '{model_name}' not found on HuggingFace Hub."
            if search_suggestions:
                error_msg += "\n\nDid you mean one of these?\n"
@@ -385,10 +393,10 @@ def validate_model_and_suggest(model_name: str, llm_type: str) -> Optional[str]:
                error_msg += "\n\nPopular chat models:\n"
                for i, model in enumerate(popular_models[:5], 1):
                    error_msg += f"  {i}. {model}\n"
-            
+
            error_msg += f"\nSearch more: https://huggingface.co/models?search={model_name}&pipeline_tag=text-generation"
            return error_msg
-    
+
    return None  # Model is valid or we can't check


@@ -451,28 +459,27 @@ class OllamaChat(LLMInterface):
            # Check if the Ollama server is responsive
            if host:
                requests.get(host)
-                
+
            # Pre-check model availability with helpful suggestions
            model_error = validate_model_and_suggest(model, "ollama")
            if model_error:
                raise ValueError(model_error)
-                
+
        except ImportError:
            raise ImportError(
                "The 'requests' library is required for Ollama. Please install it with 'pip install requests'."
            )
        except requests.exceptions.ConnectionError:
-            logger.error(
-                f"Could not connect to Ollama at {host}. Please ensure Ollama is running."
-            )
+            logger.error(f"Could not connect to Ollama at {host}. Please ensure Ollama is running.")
            raise ConnectionError(
                f"Could not connect to Ollama at {host}. Please ensure Ollama is running."
            )

    def ask(self, prompt: str, **kwargs) -> str:
-        import requests
        import json

+        import requests
+
        full_url = f"{self.host}/api/generate"
        payload = {
            "model": self.model,
@@ -482,7 +489,7 @@ class OllamaChat(LLMInterface):
        }
        logger.debug(f"Sending request to Ollama: {payload}")
        try:
-            logger.info(f"Sending request to Ollama and waiting for response...")
+            logger.info("Sending request to Ollama and waiting for response...")
            response = requests.post(full_url, data=json.dumps(payload))
            response.raise_for_status()

@@ -506,15 +513,15 @@ class HFChat(LLMInterface):

    def __init__(self, model_name: str = "deepseek-ai/deepseek-llm-7b-chat"):
        logger.info(f"Initializing HFChat with model='{model_name}'")
-        
+
        # Pre-check model availability with helpful suggestions
        model_error = validate_model_and_suggest(model_name, "hf")
        if model_error:
            raise ValueError(model_error)
-            
+
        try:
-            from transformers import AutoTokenizer, AutoModelForCausalLM
            import torch
+            from transformers import AutoModelForCausalLM, AutoTokenizer
        except ImportError:
            raise ImportError(
                "The 'transformers' and 'torch' libraries are required for Hugging Face models. Please install them with 'pip install transformers torch'."
@@ -537,36 +544,34 @@ class HFChat(LLMInterface):
            model_name,
            torch_dtype=torch.float16 if self.device != "cpu" else torch.float32,
            device_map="auto" if self.device != "cpu" else None,
-            trust_remote_code=True
+            trust_remote_code=True,
        )
-        
+
        # Move model to device if not using device_map
        if self.device != "cpu" and "device_map" not in str(self.model):
            self.model = self.model.to(self.device)
-        
+
        # Set pad token if not present
        if self.tokenizer.pad_token is None:
            self.tokenizer.pad_token = self.tokenizer.eos_token

    def ask(self, prompt: str, **kwargs) -> str:
-        print('kwargs in HF: ', kwargs)
+        print("kwargs in HF: ", kwargs)
        # Check if this is a Qwen model and add /no_think by default
        is_qwen_model = "qwen" in self.model.config._name_or_path.lower()
-        
+
        # For Qwen models, automatically add /no_think to the prompt
        if is_qwen_model and "/no_think" not in prompt and "/think" not in prompt:
            prompt = prompt + " /no_think"
-        
+
        # Prepare chat template
        messages = [{"role": "user", "content": prompt}]
-        
+
        # Apply chat template if available
        if hasattr(self.tokenizer, "apply_chat_template"):
            try:
                formatted_prompt = self.tokenizer.apply_chat_template(
-                    messages, 
-                    tokenize=False, 
-                    add_generation_prompt=True
+                    messages, tokenize=False, add_generation_prompt=True
                )
            except Exception as e:
                logger.warning(f"Chat template failed, using raw prompt: {e}")
@@ -577,13 +582,9 @@ class HFChat(LLMInterface):

        # Tokenize input
        inputs = self.tokenizer(
-            formatted_prompt, 
-            return_tensors="pt", 
-            padding=True,
-            truncation=True,
-            max_length=2048
+            formatted_prompt, return_tensors="pt", padding=True, truncation=True, max_length=2048
        )
-        
+
        # Move inputs to device
        if self.device != "cpu":
            inputs = {k: v.to(self.device) for k, v in inputs.items()}
@@ -597,32 +598,29 @@ class HFChat(LLMInterface):
            "pad_token_id": self.tokenizer.eos_token_id,
            "eos_token_id": self.tokenizer.eos_token_id,
        }
-        
+
        # Handle temperature=0 for greedy decoding
        if generation_config["temperature"] == 0.0:
            generation_config["do_sample"] = False
            generation_config.pop("temperature")

        logger.info(f"Generating with HuggingFace model, config: {generation_config}")
-        
+
        # Generate
        with torch.no_grad():
-            outputs = self.model.generate(
-                **inputs,
-                **generation_config
-            )
+            outputs = self.model.generate(**inputs, **generation_config)

        # Decode response
-        generated_tokens = outputs[0][inputs["input_ids"].shape[1]:]
+        generated_tokens = outputs[0][inputs["input_ids"].shape[1] :]
        response = self.tokenizer.decode(generated_tokens, skip_special_tokens=True)
-        
+
        return response.strip()


 class OpenAIChat(LLMInterface):
    """LLM interface for OpenAI models."""

-    def __init__(self, model: str = "gpt-4o", api_key: Optional[str] = None):
+    def __init__(self, model: str = "gpt-4o", api_key: str | None = None):
        self.model = model
        self.api_key = api_key or os.getenv("OPENAI_API_KEY")

@@ -649,11 +647,7 @@ class OpenAIChat(LLMInterface):
            "messages": [{"role": "user", "content": prompt}],
            "max_tokens": kwargs.get("max_tokens", 1000),
            "temperature": kwargs.get("temperature", 0.7),
-            **{
-                k: v
-                for k, v in kwargs.items()
-                if k not in ["max_tokens", "temperature"]
-            },
+            **{k: v for k, v in kwargs.items() if k not in ["max_tokens", "temperature"]},
        }

        logger.info(f"Sending request to OpenAI with model {self.model}")
@@ -675,7 +669,7 @@ class SimulatedChat(LLMInterface):
        return "This is a simulated answer from the LLM based on the retrieved context."


-def get_llm(llm_config: Optional[Dict[str, Any]] = None) -> LLMInterface:
+def get_llm(llm_config: dict[str, Any] | None = None) -> LLMInterface:
    """
    Factory function to get an LLM interface based on configuration.

--- a/packages/leann-core/src/leann/cli.py
+++ b/packages/leann-core/src/leann/cli.py
@@ -5,12 +5,14 @@ from pathlib import Path
 from llama_index.core import SimpleDirectoryReader
 from llama_index.core.node_parser import SentenceSplitter

-from .api import LeannBuilder, LeannSearcher, LeannChat
+from .api import LeannBuilder, LeannChat, LeannSearcher
+

 def extract_pdf_text_with_pymupdf(file_path: str) -> str:
    """Extract text from PDF using PyMuPDF for better quality."""
    try:
        import fitz  # PyMuPDF
+
        doc = fitz.open(file_path)
        text = ""
        for page in doc:
@@ -21,10 +23,12 @@ def extract_pdf_text_with_pymupdf(file_path: str) -> str:
        # Fallback to default reader
        return None

+
 def extract_pdf_text_with_pdfplumber(file_path: str) -> str:
    """Extract text from PDF using pdfplumber for better quality."""
    try:
        import pdfplumber
+
        text = ""
        with pdfplumber.open(file_path) as pdf:
            for page in pdf.pages:
@@ -72,18 +76,12 @@ Examples:
        # Build command
        build_parser = subparsers.add_parser("build", help="Build document index")
        build_parser.add_argument("index_name", help="Index name")
-        build_parser.add_argument(
-            "--docs", type=str, required=True, help="Documents directory"
-        )
+        build_parser.add_argument("--docs", type=str, required=True, help="Documents directory")
        build_parser.add_argument(
            "--backend", type=str, default="hnsw", choices=["hnsw", "diskann"]
        )
-        build_parser.add_argument(
-            "--embedding-model", type=str, default="facebook/contriever"
-        )
-        build_parser.add_argument(
-            "--force", "-f", action="store_true", help="Force rebuild"
-        )
+        build_parser.add_argument("--embedding-model", type=str, default="facebook/contriever")
+        build_parser.add_argument("--force", "-f", action="store_true", help="Force rebuild")
        build_parser.add_argument("--graph-degree", type=int, default=32)
        build_parser.add_argument("--complexity", type=int, default=64)
        build_parser.add_argument("--num-threads", type=int, default=1)
@@ -129,7 +127,7 @@ Examples:
        )

        # List command
-        list_parser = subparsers.add_parser("list", help="List all indexes")
+        subparsers.add_parser("list", help="List all indexes")

        return parser

@@ -137,17 +135,13 @@ Examples:
        print("Stored LEANN indexes:")

        if not self.indexes_dir.exists():
-            print(
-                "No indexes found. Use 'leann build <name> --docs <dir>' to create one."
-            )
+            print("No indexes found. Use 'leann build <name> --docs <dir>' to create one.")
            return

        index_dirs = [d for d in self.indexes_dir.iterdir() if d.is_dir()]

        if not index_dirs:
-            print(
-                "No indexes found. Use 'leann build <name> --docs <dir>' to create one."
-            )
+            print("No indexes found. Use 'leann build <name> --docs <dir>' to create one.")
            return

        print(f"Found {len(index_dirs)} indexes:")
@@ -157,15 +151,15 @@ Examples:

            print(f"  {i}. {index_name} [{status}]")
            if self.index_exists(index_name):
-                meta_file = index_dir / "documents.leann.meta.json"
-                size_mb = sum(
-                    f.stat().st_size for f in index_dir.iterdir() if f.is_file()
-                ) / (1024 * 1024)
+                index_dir / "documents.leann.meta.json"
+                size_mb = sum(f.stat().st_size for f in index_dir.iterdir() if f.is_file()) / (
+                    1024 * 1024
+                )
                print(f"     Size: {size_mb:.1f} MB")

        if index_dirs:
            example_name = index_dirs[0].name
-            print(f"\nUsage:")
+            print("\nUsage:")
            print(f'  leann search {example_name} "your query"')
            print(f"  leann ask {example_name} --interactive")

@@ -175,19 +169,20 @@ Examples:
        # Try to use better PDF parsers first
        documents = []
        docs_path = Path(docs_dir)
-        
+
        for file_path in docs_path.rglob("*.pdf"):
            print(f"Processing PDF: {file_path}")
-            
+
            # Try PyMuPDF first (best quality)
            text = extract_pdf_text_with_pymupdf(str(file_path))
            if text is None:
                # Try pdfplumber
                text = extract_pdf_text_with_pdfplumber(str(file_path))
-            
+
            if text:
                # Create a simple document structure
                from llama_index.core import Document
+
                doc = Document(text=text, metadata={"source": str(file_path)})
                documents.append(doc)
            else:
--- a/packages/leann-core/src/leann/embedding_compute.py
+++ b/packages/leann-core/src/leann/embedding_compute.py
@@ -4,11 +4,12 @@ Consolidates all embedding computation logic using SentenceTransformer
 Preserves all optimization parameters to ensure performance
 """

-import numpy as np
-import torch
-from typing import List, Dict, Any
 import logging
 import os
+from typing import Any
+
+import numpy as np
+import torch

 # Set up logger with proper level
 logger = logging.getLogger(__name__)
@@ -17,11 +18,11 @@ log_level = getattr(logging, LOG_LEVEL, logging.WARNING)
 logger.setLevel(log_level)

 # Global model cache to avoid repeated loading
-_model_cache: Dict[str, Any] = {}
+_model_cache: dict[str, Any] = {}


 def compute_embeddings(
-    texts: List[str],
+    texts: list[str],
    model_name: str,
    mode: str = "sentence-transformers",
    is_build: bool = False,
@@ -59,7 +60,7 @@ def compute_embeddings(


 def compute_embeddings_sentence_transformers(
-    texts: List[str],
+    texts: list[str],
    model_name: str,
    use_fp16: bool = True,
    device: str = "auto",
@@ -114,9 +115,7 @@ def compute_embeddings_sentence_transformers(
        logger.info(f"Using cached optimized model: {model_name}")
        model = _model_cache[cache_key]
    else:
-        logger.info(
-            f"Loading and caching optimized SentenceTransformer model: {model_name}"
-        )
+        logger.info(f"Loading and caching optimized SentenceTransformer model: {model_name}")
        from sentence_transformers import SentenceTransformer

        logger.info(f"Using device: {device}")
@@ -134,9 +133,7 @@ def compute_embeddings_sentence_transformers(
                if hasattr(torch.mps, "set_per_process_memory_fraction"):
                    torch.mps.set_per_process_memory_fraction(0.9)
            except AttributeError:
-                logger.warning(
-                    "Some MPS optimizations not available in this PyTorch version"
-                )
+                logger.warning("Some MPS optimizations not available in this PyTorch version")
        elif device == "cpu":
            # TODO: Haven't tested this yet
            torch.set_num_threads(min(8, os.cpu_count() or 4))
@@ -226,25 +223,22 @@ def compute_embeddings_sentence_transformers(
            device=device,
        )

-    logger.info(
-        f"Generated {len(embeddings)} embeddings, dimension: {embeddings.shape[1]}"
-    )
+    logger.info(f"Generated {len(embeddings)} embeddings, dimension: {embeddings.shape[1]}")

    # Validate results
    if np.isnan(embeddings).any() or np.isinf(embeddings).any():
-        raise RuntimeError(
-            f"Detected NaN or Inf values in embeddings, model: {model_name}"
-        )
+        raise RuntimeError(f"Detected NaN or Inf values in embeddings, model: {model_name}")

    return embeddings


-def compute_embeddings_openai(texts: List[str], model_name: str) -> np.ndarray:
+def compute_embeddings_openai(texts: list[str], model_name: str) -> np.ndarray:
    # TODO: @yichuan-w add progress bar only in build mode
    """Compute embeddings using OpenAI API"""
    try:
-        import openai
        import os
+
+        import openai
    except ImportError as e:
        raise ImportError(f"OpenAI package not installed: {e}")

@@ -294,16 +288,12 @@ def compute_embeddings_openai(texts: List[str], model_name: str) -> np.ndarray:
            raise

    embeddings = np.array(all_embeddings, dtype=np.float32)
-    logger.info(
-        f"Generated {len(embeddings)} embeddings, dimension: {embeddings.shape[1]}"
-    )
+    logger.info(f"Generated {len(embeddings)} embeddings, dimension: {embeddings.shape[1]}")
    print(f"len of embeddings: {len(embeddings)}")
    return embeddings


-def compute_embeddings_mlx(
-    chunks: List[str], model_name: str, batch_size: int = 16
-) -> np.ndarray:
+def compute_embeddings_mlx(chunks: list[str], model_name: str, batch_size: int = 16) -> np.ndarray:
    # TODO: @yichuan-w add progress bar only in build mode
    """Computes embeddings using an MLX model."""
    try:
--- a/packages/leann-core/src/leann/embedding_server_manager.py
+++ b/packages/leann-core/src/leann/embedding_server_manager.py
@@ -1,12 +1,12 @@
-import time
 import atexit
+import logging
+import os
 import socket
 import subprocess
 import sys
-import os
-import logging
+import time
 from pathlib import Path
-from typing import Optional
+
 import psutil

 # Set up logging based on environment variable
@@ -33,7 +33,7 @@ def _get_available_port(start_port: int = 5557) -> int:
                return port
        except OSError:
            port += 1
-    raise RuntimeError(f"No available ports found in range {start_port}-{start_port+100}")
+    raise RuntimeError(f"No available ports found in range {start_port}-{start_port + 100}")


 def _check_port(port: int) -> bool:
@@ -182,8 +182,8 @@ class EmbeddingServerManager:
                                       e.g., "leann_backend_diskann.embedding_server"
        """
        self.backend_module_name = backend_module_name
-        self.server_process: Optional[subprocess.Popen] = None
-        self.server_port: Optional[int] = None
+        self.server_process: subprocess.Popen | None = None
+        self.server_port: int | None = None
        self._atexit_registered = False

    def start_server(
@@ -234,10 +234,10 @@ class EmbeddingServerManager:
            return False, port

        logger.info(f"Starting server on port {actual_port} for Colab environment")
-        
+
        # Use a simpler startup strategy for Colab
        command = self._build_server_command(actual_port, model_name, embedding_mode, **kwargs)
-        
+
        try:
            # In Colab, we'll use a more direct approach
            self._launch_server_process_colab(command, actual_port)
@@ -246,26 +246,16 @@ class EmbeddingServerManager:
            logger.error(f"Failed to start embedding server in Colab: {e}")
            return False, actual_port

-    def _has_compatible_running_server(
-        self, model_name: str, passages_file: str
-    ) -> bool:
+    def _has_compatible_running_server(self, model_name: str, passages_file: str) -> bool:
        """Check if we have a compatible running server."""
-        if not (
-            self.server_process
-            and self.server_process.poll() is None
-            and self.server_port
-        ):
+        if not (self.server_process and self.server_process.poll() is None and self.server_port):
            return False

        if _check_process_matches_config(self.server_port, model_name, passages_file):
-            logger.info(
-                f"Existing server process (PID {self.server_process.pid}) is compatible"
-            )
+            logger.info(f"Existing server process (PID {self.server_process.pid}) is compatible")
            return True

-        logger.info(
-            "Existing server process is incompatible. Should start a new server."
-        )
+        logger.info("Existing server process is incompatible. Should start a new server.")
        return False

    def _start_new_server(
@@ -400,7 +390,7 @@ class EmbeddingServerManager:
    def _wait_for_server_ready_colab(self, port: int) -> tuple[bool, int]:
        """Wait for the server to be ready with Colab-specific timeout."""
        max_wait, wait_interval = 30, 0.5  # Shorter timeout for Colab
-        
+
        for _ in range(int(max_wait / wait_interval)):
            if _check_port(port):
                logger.info("Colab embedding server is ready!")
@@ -409,7 +399,7 @@ class EmbeddingServerManager:
            if self.server_process and self.server_process.poll() is not None:
                # Check for error output
                stdout, stderr = self.server_process.communicate()
-                logger.error(f"Colab server terminated during startup.")
+                logger.error("Colab server terminated during startup.")
                logger.error(f"stdout: {stdout}")
                logger.error(f"stderr: {stderr}")
                return False, port
--- a/packages/leann-core/src/leann/interface.py
+++ b/packages/leann-core/src/leann/interface.py
@@ -1,15 +1,14 @@
 from abc import ABC, abstractmethod
+from typing import Any, Literal
+
 import numpy as np
-from typing import Dict, Any, List, Literal, Optional


 class LeannBackendBuilderInterface(ABC):
    """Backend interface for building indexes"""

    @abstractmethod
-    def build(
-        self, data: np.ndarray, ids: List[str], index_path: str, **kwargs
-    ) -> None:
+    def build(self, data: np.ndarray, ids: list[str], index_path: str, **kwargs) -> None:
        """Build index

        Args:
@@ -35,9 +34,7 @@ class LeannBackendSearcherInterface(ABC):
        pass

    @abstractmethod
-    def _ensure_server_running(
-        self, passages_source_file: str, port: Optional[int], **kwargs
-    ) -> int:
+    def _ensure_server_running(self, passages_source_file: str, port: int | None, **kwargs) -> int:
        """Ensure server is running"""
        pass

@@ -51,9 +48,9 @@ class LeannBackendSearcherInterface(ABC):
        prune_ratio: float = 0.0,
        recompute_embeddings: bool = False,
        pruning_strategy: Literal["global", "local", "proportional"] = "global",
-        zmq_port: Optional[int] = None,
+        zmq_port: int | None = None,
        **kwargs,
-    ) -> Dict[str, Any]:
+    ) -> dict[str, Any]:
        """Search for nearest neighbors

        Args:
@@ -77,7 +74,7 @@ class LeannBackendSearcherInterface(ABC):
        self,
        query: str,
        use_server_if_available: bool = True,
-        zmq_port: Optional[int] = None,
+        zmq_port: int | None = None,
    ) -> np.ndarray:
        """Compute embedding for a query string

--- a/packages/leann-core/src/leann/registry.py
+++ b/packages/leann-core/src/leann/registry.py
@@ -1,13 +1,13 @@
 # packages/leann-core/src/leann/registry.py

-from typing import Dict, TYPE_CHECKING
 import importlib
 import importlib.metadata
+from typing import TYPE_CHECKING

 if TYPE_CHECKING:
    from leann.interface import LeannBackendFactoryInterface

-BACKEND_REGISTRY: Dict[str, "LeannBackendFactoryInterface"] = {}
+BACKEND_REGISTRY: dict[str, "LeannBackendFactoryInterface"] = {}


 def register_backend(name: str):
@@ -31,13 +31,11 @@ def autodiscover_backends():
            backend_module_name = dist_name.replace("-", "_")
            discovered_backends.append(backend_module_name)

-    for backend_module_name in sorted(
-        discovered_backends
-    ):  # sort for deterministic loading
+    for backend_module_name in sorted(discovered_backends):  # sort for deterministic loading
        try:
            importlib.import_module(backend_module_name)
            # Registration message is printed by the decorator
-        except ImportError as e:
+        except ImportError:
            # print(f"WARN: Could not import backend module '{backend_module_name}': {e}")
            pass
    # print("INFO: Backend auto-discovery finished.")
--- a/packages/leann-core/src/leann/searcher_base.py
+++ b/packages/leann-core/src/leann/searcher_base.py
@@ -1,7 +1,7 @@
 import json
 from abc import ABC, abstractmethod
 from pathlib import Path
-from typing import Dict, Any, Literal, Optional
+from typing import Any, Literal

 import numpy as np

@@ -38,9 +38,7 @@ class BaseSearcher(LeannBackendSearcherInterface, ABC):

        self.embedding_model = self.meta.get("embedding_model")
        if not self.embedding_model:
-            print(
-                "WARNING: embedding_model not found in meta.json. Recompute will fail."
-            )
+            print("WARNING: embedding_model not found in meta.json. Recompute will fail.")

        self.embedding_mode = self.meta.get("embedding_mode", "sentence-transformers")

@@ -48,26 +46,22 @@ class BaseSearcher(LeannBackendSearcherInterface, ABC):
            backend_module_name=backend_module_name,
        )

-    def _load_meta(self) -> Dict[str, Any]:
+    def _load_meta(self) -> dict[str, Any]:
        """Loads the metadata file associated with the index."""
        # This is the corrected logic for finding the meta file.
        meta_path = self.index_dir / f"{self.index_path.name}.meta.json"
        if not meta_path.exists():
            raise FileNotFoundError(f"Leann metadata file not found at {meta_path}")
-        with open(meta_path, "r", encoding="utf-8") as f:
+        with open(meta_path, encoding="utf-8") as f:
            return json.load(f)

-    def _ensure_server_running(
-        self, passages_source_file: str, port: int, **kwargs
-    ) -> int:
+    def _ensure_server_running(self, passages_source_file: str, port: int, **kwargs) -> int:
        """
        Ensures the embedding server is running if recompute is needed.
        This is a helper for subclasses.
        """
        if not self.embedding_model:
-            raise ValueError(
-                "Cannot use recompute mode without 'embedding_model' in meta.json."
-            )
+            raise ValueError("Cannot use recompute mode without 'embedding_model' in meta.json.")

        server_started, actual_port = self.embedding_server_manager.start_server(
            port=port,
@@ -78,9 +72,7 @@ class BaseSearcher(LeannBackendSearcherInterface, ABC):
            enable_warmup=kwargs.get("enable_warmup", False),
        )
        if not server_started:
-            raise RuntimeError(
-                f"Failed to start embedding server on port {actual_port}"
-            )
+            raise RuntimeError(f"Failed to start embedding server on port {actual_port}")

        return actual_port

@@ -109,9 +101,7 @@ class BaseSearcher(LeannBackendSearcherInterface, ABC):
                # on that port?

                # Ensure we have a server with passages_file for compatibility
-                passages_source_file = (
-                    self.index_dir / f"{self.index_path.name}.meta.json"
-                )
+                passages_source_file = self.index_dir / f"{self.index_path.name}.meta.json"
                # Convert to absolute path to ensure server can find it
                zmq_port = self._ensure_server_running(
                    str(passages_source_file.resolve()), zmq_port
@@ -132,8 +122,8 @@ class BaseSearcher(LeannBackendSearcherInterface, ABC):

    def _compute_embedding_via_server(self, chunks: list, zmq_port: int) -> np.ndarray:
        """Compute embeddings using the ZMQ embedding server."""
-        import zmq
        import msgpack
+        import zmq

        try:
            context = zmq.Context()
@@ -172,9 +162,9 @@ class BaseSearcher(LeannBackendSearcherInterface, ABC):
        prune_ratio: float = 0.0,
        recompute_embeddings: bool = False,
        pruning_strategy: Literal["global", "local", "proportional"] = "global",
-        zmq_port: Optional[int] = None,
+        zmq_port: int | None = None,
        **kwargs,
-    ) -> Dict[str, Any]:
+    ) -> dict[str, Any]:
        """
        Search for the top_k nearest neighbors of the query vector.

--- a/packages/leann/init.py
+++ b/packages/leann/init.py
@@ -7,6 +7,6 @@ A revolutionary vector database that democratizes personal AI.
 __version__ = "0.1.0"

 # Re-export main API from leann-core
-from leann_core import LeannBuilder, LeannSearcher, LeannChat
+from leann_core import LeannBuilder, LeannChat, LeannSearcher

-__all__ = ["LeannBuilder", "LeannSearcher", "LeannChat"]
+__all__ = ["LeannBuilder", "LeannChat", "LeannSearcher"]
--- a/packages/wechat-exporter/main.py
+++ b/packages/wechat-exporter/main.py
@@ -1,22 +1,23 @@
 import json
-import typer
-from pathlib import Path
-import requests
-from tqdm import tqdm
-import xml.etree.ElementTree as ET
-from typing_extensions import Annotated
 import sqlite3
+import xml.etree.ElementTree as ET
+from pathlib import Path
+from typing import Annotated
+
+import requests
+import typer
+from tqdm import tqdm

 app = typer.Typer()

+
 def get_safe_path(s: str) -> str:
    """
    Remove invalid characters to sanitize a path.
    :param s: str to sanitize
    :returns: sanitized str
    """
-    ban_chars = "\\  /  :  *  ?  \"  '  <  >  |  $  \r  \n".replace(
-        ' ', '')
+    ban_chars = "\\  /  :  *  ?  \"  '  <  >  |  $  \r  \n".replace(" ", "")
    for i in ban_chars:
        s = s.replace(i, "")
    return s
@@ -26,35 +27,38 @@ def process_history(history: str):
    if history.startswith("<?xml") or history.startswith("<msg>"):
        try:
            root = ET.fromstring(history)
-            title = root.find('.//title').text if root.find('.//title') is not None else None
-            quoted = root.find('.//refermsg/content').text if root.find('.//refermsg/content') is not None else None
+            title = root.find(".//title").text if root.find(".//title") is not None else None
+            quoted = (
+                root.find(".//refermsg/content").text
+                if root.find(".//refermsg/content") is not None
+                else None
+            )
            if title and quoted:
-                return {
-                    "title": title,
-                    "quoted": process_history(quoted)
-                }
+                return {"title": title, "quoted": process_history(quoted)}
            if title:
                return title
        except Exception:
            return history
    return history

+
 def get_message(history: dict | str):
    if isinstance(history, dict):
-        if 'title' in history:
-            return history['title']
+        if "title" in history:
+            return history["title"]
    else:
        return history

+
 def export_chathistory(user_id: str):
-    res = requests.get("http://localhost:48065/wechat/chatlog", params={
-        "userId": user_id,
-        "count": 100000
-    }).json()
-    for i in range(len(res['chatLogs'])):
-        res['chatLogs'][i]['content'] = process_history(res['chatLogs'][i]['content'])
-        res['chatLogs'][i]['message'] = get_message(res['chatLogs'][i]['content'])
-    return res['chatLogs']
+    res = requests.get(
+        "http://localhost:48065/wechat/chatlog", params={"userId": user_id, "count": 100000}
+    ).json()
+    for i in range(len(res["chatLogs"])):
+        res["chatLogs"][i]["content"] = process_history(res["chatLogs"][i]["content"])
+        res["chatLogs"][i]["message"] = get_message(res["chatLogs"][i]["content"])
+    return res["chatLogs"]
+

@app.command()
 def export_all(dest: Annotated[Path, typer.Argument(help="Destination path to export to.")]):
@@ -64,7 +68,7 @@ def export_all(dest: Annotated[Path, typer.Argument(help="Destination path to ex
    if not dest.is_dir():
        if not dest.exists():
            inp = typer.prompt("Destination path does not exist, create it? (y/n)")
-            if inp.lower() == 'y':
+            if inp.lower() == "y":
                dest.mkdir(parents=True)
            else:
                typer.echo("Aborted.", err=True)
@@ -77,12 +81,12 @@ def export_all(dest: Annotated[Path, typer.Argument(help="Destination path to ex
    exported_count = 0
    for user in tqdm(all_users):
        try:
-            usr_chatlog = export_chathistory(user['arg'])
-            
+            usr_chatlog = export_chathistory(user["arg"])
+
            # Only write file if there are messages
            if len(usr_chatlog) > 0:
-                out_path = dest/get_safe_path((user['title'] or "")+"-"+user['arg']+'.json')
-                with open(out_path, 'w', encoding='utf-8') as f:
+                out_path = dest / get_safe_path((user["title"] or "") + "-" + user["arg"] + ".json")
+                with open(out_path, "w", encoding="utf-8") as f:
                    json.dump(usr_chatlog, f, ensure_ascii=False, indent=2)
                exported_count += 1
        except Exception as e:
@@ -91,23 +95,42 @@ def export_all(dest: Annotated[Path, typer.Argument(help="Destination path to ex

    print(f"Exported {exported_count} users' chat history to {dest} in json.")

+
@app.command()
-def export_sqlite(dest: Annotated[Path, typer.Argument(help="Destination path to export to.")] = Path("chatlog.db")):
+def export_sqlite(
+    dest: Annotated[Path, typer.Argument(help="Destination path to export to.")] = Path(
+        "chatlog.db"
+    ),
+):
    """
    Export all users' chat history to a sqlite database.
    """
    connection = sqlite3.connect(dest)
    cursor = connection.cursor()
-    cursor.execute("CREATE TABLE IF NOT EXISTS chatlog (id INTEGER PRIMARY KEY AUTOINCREMENT, with_id TEXT, from_user TEXT, to_user TEXT, message TEXT, timest DATETIME, auxiliary TEXT)")
+    cursor.execute(
+        "CREATE TABLE IF NOT EXISTS chatlog (id INTEGER PRIMARY KEY AUTOINCREMENT, with_id TEXT, from_user TEXT, to_user TEXT, message TEXT, timest DATETIME, auxiliary TEXT)"
+    )
    cursor.execute("CREATE INDEX IF NOT EXISTS chatlog_with_id_index ON chatlog (with_id)")
    cursor.execute("CREATE TABLE iF NOT EXISTS users (id TEXT PRIMARY KEY, name TEXT)")

    all_users = requests.get("http://localhost:48065/wechat/allcontacts").json()
    for user in tqdm(all_users):
-        cursor.execute("INSERT OR IGNORE INTO users (id, name) VALUES (?, ?)", (user['arg'], user['title']))
-        usr_chatlog = export_chathistory(user['arg'])
+        cursor.execute(
+            "INSERT OR IGNORE INTO users (id, name) VALUES (?, ?)", (user["arg"], user["title"])
+        )
+        usr_chatlog = export_chathistory(user["arg"])
        for msg in usr_chatlog:
-            cursor.execute("INSERT INTO chatlog (with_id, from_user, to_user, message, timest, auxiliary) VALUES (?, ?, ?, ?, ?, ?)", (user['arg'], msg['fromUser'], msg['toUser'], msg['message'], msg['createTime'], str(msg['content'])))
+            cursor.execute(
+                "INSERT INTO chatlog (with_id, from_user, to_user, message, timest, auxiliary) VALUES (?, ?, ?, ?, ?, ?)",
+                (
+                    user["arg"],
+                    msg["fromUser"],
+                    msg["toUser"],
+                    msg["message"],
+                    msg["createTime"],
+                    str(msg["content"]),
+                ),
+            )
    connection.commit()