[chat] update huggingface chat and make qwen no thinking

2025-07-24 00:11:42 -07:00
parent d502fa24b0
commit efd6373b32
3 changed files with 308 additions and 47 deletions
--- a/README.md
+++ b/README.md
@@ -152,7 +152,7 @@ python ./examples/main_cli_example.py

 **Note:** You need to grant full disk access to your terminal/VS Code in System Preferences → Privacy & Security → Full Disk Access.
 ```bash
-python examples/mail_reader_leann.py --query "What's the food I ordered by doordash or Uber eat?"
+python examples/mail_reader_leann.py --query "What's the food I ordered by doordash or Uber eat mostly?"
 ```
 **780K email chunks → 78MB storage** Finally, search your email like you search Google.

@@ -187,7 +187,7 @@ Once the index is built, you can ask questions like:
 - "Show me emails about travel expenses"
 </details>

-### 🔍 Time Machine for the Web: RAG Your Entire Browser History!
+### 🔍 Time Machine for the Web: RAG Your Entire Google Browser History!
 ```bash
 python examples/google_history_reader_leann.py --query "Tell me my browser history about machine learning?"
 ```
--- a/demo.ipynb
+++ b/demo.ipynb
@@ -7,6 +7,16 @@
    "# Quick Start in 30s"
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "# install this if you areusing colab\n",
+    "! pip install leann"
+   ]
+  },
  {
   "cell_type": "markdown",
   "metadata": {},
@@ -16,9 +26,81 @@
  },
  {
   "cell_type": "code",
-   "execution_count": null,
+   "execution_count": 1,
   "metadata": {},
-   "outputs": [],
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "INFO: Registering backend 'hnsw'\n"
+     ]
+    },
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "/Users/yichuan/Desktop/code/LEANN/leann/.venv/lib/python3.11/site-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html\n",
+      "  from .autonotebook import tqdm as notebook_tqdm\n",
+      "INFO:sentence_transformers.SentenceTransformer:Load pretrained SentenceTransformer: facebook/contriever\n",
+      "WARNING:sentence_transformers.SentenceTransformer:No sentence-transformers model found with name facebook/contriever. Creating a new one with mean pooling.\n",
+      "Writing passages: 100%|██████████| 5/5 [00:00<00:00, 31254.13chunk/s]\n",
+      "Batches: 100%|██████████| 1/1 [00:00<00:00, 12.19it/s]\n",
+      "WARNING:leann_backend_hnsw.hnsw_backend:Converting data to float32, shape: (5, 768)\n",
+      "INFO:leann_backend_hnsw.hnsw_backend:INFO: Converting HNSW index to CSR-pruned format...\n"
+     ]
+    },
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "M: 64 for level: 0\n",
+      "Starting conversion: knowledge.index -> knowledge.csr.tmp\n",
+      "[0.00s] Reading Index HNSW header...\n",
+      "[0.00s]   Header read: d=768, ntotal=5\n",
+      "[0.00s] Reading HNSW struct vectors...\n",
+      "  Reading vector (dtype=<class 'numpy.float64'>, fmt='d')... Count=6, Bytes=48\n",
+      "[0.00s]   Read assign_probas (6)\n",
+      "  Reading vector (dtype=<class 'numpy.int32'>, fmt='i')... Count=7, Bytes=28\n",
+      "[0.11s]   Read cum_nneighbor_per_level (7)\n",
+      "  Reading vector (dtype=<class 'numpy.int32'>, fmt='i')... Count=5, Bytes=20\n",
+      "[0.23s]   Read levels (5)\n",
+      "[0.34s]   Probing for compact storage flag...\n",
+      "[0.34s]   Found compact flag: False\n",
+      "[0.34s]   Compact flag is False, reading original format...\n",
+      "[0.34s]   Probing for potential extra byte before non-compact offsets...\n",
+      "[0.34s]   Found and consumed an unexpected 0x00 byte.\n",
+      "  Reading vector (dtype=<class 'numpy.uint64'>, fmt='Q')... Count=6, Bytes=48\n",
+      "[0.34s]   Read offsets (6)\n",
+      "[0.44s]   Attempting to read neighbors vector...\n",
+      "  Reading vector (dtype=<class 'numpy.int32'>, fmt='i')... Count=320, Bytes=1280\n",
+      "[0.44s]   Read neighbors (320)\n",
+      "[0.54s]   Read scalar params (ep=4, max_lvl=0)\n",
+      "[0.54s] Checking for storage data...\n",
+      "[0.54s]   Found storage fourcc: 49467849.\n",
+      "[0.54s] Converting to CSR format...\n",
+      "[0.54s]   Conversion loop finished.                        \n",
+      "[0.54s] Running validation checks...\n",
+      "    Checking total valid neighbor count...\n",
+      "    OK: Total valid neighbors = 20\n",
+      "    Checking final pointer indices...\n",
+      "    OK: Final pointers match data size.\n",
+      "[0.54s] Deleting original neighbors and offsets arrays...\n",
+      "    CSR Stats: |data|=20, |level_ptr|=10\n",
+      "[0.63s] Writing CSR HNSW graph data in FAISS-compatible order...\n",
+      "   Pruning embeddings: Writing NULL storage marker.\n",
+      "[0.73s] Conversion complete.\n"
+     ]
+    },
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "INFO:leann_backend_hnsw.hnsw_backend:✅ CSR conversion successful.\n",
+      "INFO:leann_backend_hnsw.hnsw_backend:INFO: Replaced original index with CSR-pruned version at 'knowledge.index'\n"
+     ]
+    }
+   ],
   "source": [
    "from leann.api import LeannBuilder\n",
    "\n",
@@ -40,9 +122,66 @@
  },
  {
   "cell_type": "code",
-   "execution_count": null,
+   "execution_count": 2,
   "metadata": {},
-   "outputs": [],
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "INFO:leann.api:🔍 LeannSearcher.search() called:\n",
+      "INFO:leann.api:  Query: 'programming languages'\n",
+      "INFO:leann.api:  Top_k: 2\n",
+      "INFO:leann.api:  Additional kwargs: {}\n",
+      "INFO:leann.embedding_server_manager:Port 5557 has incompatible server, trying next port...\n",
+      "INFO:leann.embedding_server_manager:Port 5558 has incompatible server, trying next port...\n",
+      "INFO:leann.embedding_server_manager:Port 5559 has incompatible server, trying next port...\n",
+      "INFO:leann.embedding_server_manager:Found compatible server on port 5560\n",
+      "INFO:leann.embedding_server_manager:Using existing compatible server on port 5560\n",
+      "INFO:leann.api:  Launching server time: 0.05758476257324219 seconds\n",
+      "INFO:leann.embedding_server_manager:Found compatible server on port 5560\n",
+      "INFO:leann.embedding_server_manager:Using existing compatible server on port 5560\n",
+      "INFO:leann.api:  Generated embedding shape: (1, 768)\n",
+      "INFO:leann.api:  Embedding time: 0.05983591079711914 seconds\n",
+      "INFO:leann.api:  Search time: 0.039762258529663086 seconds\n",
+      "INFO:leann.api:  Backend returned: labels=2 results\n",
+      "INFO:leann.api:  Processing 2 passage IDs:\n",
+      "INFO:leann.api:    1. passage_id='0' -> SUCCESS: C# is a powerful programming language...\n",
+      "INFO:leann.api:    2. passage_id='1' -> SUCCESS: Python is a powerful programming language and it is very popular...\n",
+      "INFO:leann.api:  Final enriched results: 2 passages\n"
+     ]
+    },
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "[read_HNSW - CSR NL v4] Reading metadata & CSR indices (manual offset)...\n",
+      "[read_HNSW NL v4] Read levels vector, size: 5\n",
+      "[read_HNSW NL v4] Reading Compact Storage format indices...\n",
+      "[read_HNSW NL v4] Read compact_level_ptr, size: 10\n",
+      "[read_HNSW NL v4] Read compact_node_offsets, size: 6\n",
+      "[read_HNSW NL v4] Read entry_point: 4, max_level: 0\n",
+      "[read_HNSW NL v4] Read storage fourcc: 0x6c6c756e\n",
+      "[read_HNSW NL v4 FIX] Detected FileIOReader. Neighbors size field offset: 326\n",
+      "[read_HNSW NL v4] Reading neighbors data into memory.\n",
+      "[read_HNSW NL v4] Read neighbors data, size: 20\n",
+      "[read_HNSW NL v4] Finished reading metadata and CSR indices.\n",
+      "INFO: Skipping external storage loading, since is_recompute is true.\n",
+      "ZmqDistanceComputer initialized: d=768, metric=0\n"
+     ]
+    },
+    {
+     "data": {
+      "text/plain": [
+       "[SearchResult(id='0', score=np.float32(0.9646692), text='C# is a powerful programming language', metadata={}),\n",
+       " SearchResult(id='1', score=np.float32(0.91955304), text='Python is a powerful programming language and it is very popular', metadata={})]"
+      ]
+     },
+     "execution_count": 2,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
   "source": [
    "from leann.api import LeannSearcher\n",
    "\n",
@@ -60,15 +199,90 @@
  },
  {
   "cell_type": "code",
-   "execution_count": null,
+   "execution_count": 1,
   "metadata": {},
-   "outputs": [],
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "INFO:leann.chat:Attempting to create LLM of type='hf' with model='Qwen/Qwen3-0.6B'\n",
+      "INFO:leann.chat:Initializing HFChat with model='Qwen/Qwen3-0.6B'\n"
+     ]
+    },
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "INFO: Registering backend 'hnsw'\n",
+      "[read_HNSW - CSR NL v4] Reading metadata & CSR indices (manual offset)...\n",
+      "[read_HNSW NL v4] Read levels vector, size: 5\n",
+      "[read_HNSW NL v4] Reading Compact Storage format indices...\n",
+      "[read_HNSW NL v4] Read compact_level_ptr, size: 10\n",
+      "[read_HNSW NL v4] Read compact_node_offsets, size: 6\n",
+      "[read_HNSW NL v4] Read entry_point: 4, max_level: 0\n",
+      "[read_HNSW NL v4] Read storage fourcc: 0x6c6c756e\n",
+      "[read_HNSW NL v4 FIX] Detected FileIOReader. Neighbors size field offset: 326\n",
+      "[read_HNSW NL v4] Reading neighbors data into memory.\n",
+      "[read_HNSW NL v4] Read neighbors data, size: 20\n",
+      "[read_HNSW NL v4] Finished reading metadata and CSR indices.\n",
+      "INFO: Skipping external storage loading, since is_recompute is true.\n"
+     ]
+    },
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "/Users/yichuan/Desktop/code/LEANN/leann/.venv/lib/python3.11/site-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html\n",
+      "  from .autonotebook import tqdm as notebook_tqdm\n",
+      "INFO:leann.chat:MPS is available. Using Apple Silicon GPU.\n",
+      "INFO:leann.api:🔍 LeannSearcher.search() called:\n",
+      "INFO:leann.api:  Query: 'Compare the two retrieved programming languages and say which one is more popular today.'\n",
+      "INFO:leann.api:  Top_k: 2\n",
+      "INFO:leann.api:  Additional kwargs: {}\n",
+      "INFO:leann.embedding_server_manager:Port 5557 has incompatible server, trying next port...\n",
+      "INFO:leann.embedding_server_manager:Port 5558 has incompatible server, trying next port...\n",
+      "INFO:leann.embedding_server_manager:Port 5559 has incompatible server, trying next port...\n",
+      "INFO:leann.embedding_server_manager:Found compatible server on port 5560\n",
+      "INFO:leann.embedding_server_manager:Using existing compatible server on port 5560\n",
+      "INFO:leann.api:  Launching server time: 0.11421084403991699 seconds\n",
+      "INFO:leann.embedding_server_manager:Found compatible server on port 5560\n",
+      "INFO:leann.embedding_server_manager:Using existing compatible server on port 5560\n",
+      "INFO:leann.api:  Generated embedding shape: (1, 768)\n",
+      "INFO:leann.api:  Embedding time: 0.1147918701171875 seconds\n",
+      "INFO:leann.api:  Search time: 0.05468583106994629 seconds\n",
+      "INFO:leann.api:  Backend returned: labels=2 results\n",
+      "INFO:leann.api:  Processing 2 passage IDs:\n",
+      "INFO:leann.api:    1. passage_id='1' -> SUCCESS: Python is a powerful programming language and it is very popular...\n",
+      "INFO:leann.api:    2. passage_id='0' -> SUCCESS: C# is a powerful programming language...\n",
+      "INFO:leann.api:  Final enriched results: 2 passages\n",
+      "INFO:leann.chat:Generating with HuggingFace model, config: {'max_new_tokens': 512, 'temperature': 0.7, 'top_p': 0.9, 'do_sample': True, 'pad_token_id': 151645, 'eos_token_id': 151645}\n"
+     ]
+    },
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "ZmqDistanceComputer initialized: d=768, metric=0\n"
+     ]
+    },
+    {
+     "data": {
+      "text/plain": [
+       "'<think>\\n\\n</think>\\n\\nBased on the context provided, both Python and C# are mentioned as powerful programming languages, but no specific information is given about their popularity today. However, generally, Python is more popular for data science, web development, and other tasks, while C# is widely used in enterprise applications and game development. Since the context does not explicitly state which is more popular, but Python is often considered more popular in many cases, the best answer would be:\\n\\n**Python is more popular today.**'"
+      ]
+     },
+     "execution_count": 1,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
   "source": [
    "from leann.api import LeannChat\n",
    "\n",
    "llm_config = {\n",
-    "    \"type\": \"ollama\",\n",
-    "    \"model\": \"llama3.2:1b\"\n",
+    "    \"type\": \"hf\",\n",
+    "    \"model\": \"Qwen/Qwen3-0.6B\"\n",
    "}\n",
    "\n",
    "chat = LeannChat(index_path=\"knowledge.leann\", llm_config=llm_config)\n",
--- a/packages/leann-core/src/leann/chat.py
+++ b/packages/leann-core/src/leann/chat.py
@@ -9,6 +9,7 @@ from typing import Dict, Any, Optional, List
 import logging
 import os
 import difflib
+import torch

 # Configure logging
 logging.basicConfig(level=logging.INFO)
@@ -501,7 +502,7 @@ class OllamaChat(LLMInterface):


 class HFChat(LLMInterface):
-    """LLM interface for local Hugging Face Transformers models."""
+    """LLM interface for local Hugging Face Transformers models with proper chat templates."""

    def __init__(self, model_name: str = "deepseek-ai/deepseek-llm-7b-chat"):
        logger.info(f"Initializing HFChat with model='{model_name}'")
@@ -512,7 +513,7 @@ class HFChat(LLMInterface):
            raise ValueError(model_error)
            
        try:
-            from transformers.pipelines import pipeline
+            from transformers import AutoTokenizer, AutoModelForCausalLM
            import torch
        except ImportError:
            raise ImportError(
@@ -521,54 +522,100 @@ class HFChat(LLMInterface):

        # Auto-detect device
        if torch.cuda.is_available():
-            device = "cuda"
+            self.device = "cuda"
            logger.info("CUDA is available. Using GPU.")
        elif hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
-            device = "mps"
+            self.device = "mps"
            logger.info("MPS is available. Using Apple Silicon GPU.")
        else:
-            device = "cpu"
+            self.device = "cpu"
            logger.info("No GPU detected. Using CPU.")

-        self.pipeline = pipeline("text-generation", model=model_name, device=device)
+        # Load tokenizer and model
+        self.tokenizer = AutoTokenizer.from_pretrained(model_name)
+        self.model = AutoModelForCausalLM.from_pretrained(
+            model_name,
+            torch_dtype=torch.float16 if self.device != "cpu" else torch.float32,
+            device_map="auto" if self.device != "cpu" else None,
+            trust_remote_code=True
+        )
+        
+        # Move model to device if not using device_map
+        if self.device != "cpu" and "device_map" not in str(self.model):
+            self.model = self.model.to(self.device)
+        
+        # Set pad token if not present
+        if self.tokenizer.pad_token is None:
+            self.tokenizer.pad_token = self.tokenizer.eos_token

    def ask(self, prompt: str, **kwargs) -> str:
-        # Map OpenAI-style arguments to Hugging Face equivalents
-        if "max_tokens" in kwargs:
-            # Prefer user-provided max_new_tokens if both are present
-            kwargs.setdefault("max_new_tokens", kwargs["max_tokens"])
-            # Remove the unsupported key to avoid errors in Transformers
-            kwargs.pop("max_tokens")
+        # Check if this is a Qwen model and add /no_think by default
+        is_qwen_model = "qwen" in self.model.config._name_or_path.lower()
+        
+        # For Qwen models, automatically add /no_think to the prompt
+        if is_qwen_model and "/no_think" not in prompt and "/think" not in prompt:
+            prompt = prompt + " /no_think"
+        
+        # Prepare chat template
+        messages = [{"role": "user", "content": prompt}]
+        
+        # Apply chat template if available
+        if hasattr(self.tokenizer, "apply_chat_template"):
+            try:
+                formatted_prompt = self.tokenizer.apply_chat_template(
+                    messages, 
+                    tokenize=False, 
+                    add_generation_prompt=True
+                )
+            except Exception as e:
+                logger.warning(f"Chat template failed, using raw prompt: {e}")
+                formatted_prompt = prompt
+        else:
+            # Fallback for models without chat template
+            formatted_prompt = prompt

-        # Handle temperature=0 edge-case for greedy decoding
-        if "temperature" in kwargs and kwargs["temperature"] == 0.0:
-            # Remove unsupported zero temperature and use deterministic generation
-            kwargs.pop("temperature")
-            kwargs.setdefault("do_sample", False)
+        # Tokenize input
+        inputs = self.tokenizer(
+            formatted_prompt, 
+            return_tensors="pt", 
+            padding=True,
+            truncation=True,
+            max_length=2048
+        )
+        
+        # Move inputs to device
+        if self.device != "cpu":
+            inputs = {k: v.to(self.device) for k, v in inputs.items()}

-        # Sensible defaults for text generation
-        params = {"max_length": 500, "num_return_sequences": 1, **kwargs}
-        logger.info(f"Generating text with Hugging Face model with params: {params}")
-        results = self.pipeline(prompt, **params)
+        # Set generation parameters
+        generation_config = {
+            "max_new_tokens": kwargs.get("max_tokens", kwargs.get("max_new_tokens", 512)),
+            "temperature": kwargs.get("temperature", 0.7),
+            "top_p": kwargs.get("top_p", 0.9),
+            "do_sample": kwargs.get("temperature", 0.7) > 0,
+            "pad_token_id": self.tokenizer.eos_token_id,
+            "eos_token_id": self.tokenizer.eos_token_id,
+        }
+        
+        # Handle temperature=0 for greedy decoding
+        if generation_config["temperature"] == 0.0:
+            generation_config["do_sample"] = False
+            generation_config.pop("temperature")

-        # Handle different response formats from transformers
-        if isinstance(results, list) and len(results) > 0:
-            generated_text = (
-                results[0].get("generated_text", "")
-                if isinstance(results[0], dict)
-                else str(results[0])
+        logger.info(f"Generating with HuggingFace model, config: {generation_config}")
+        
+        # Generate
+        with torch.no_grad():
+            outputs = self.model.generate(
+                **inputs,
+                **generation_config
            )
-        else:
-            generated_text = str(results)

-        # Extract only the newly generated portion by removing the original prompt
-        if isinstance(generated_text, str) and generated_text.startswith(prompt):
-            response = generated_text[len(prompt) :].strip()
-        else:
-            # Fallback: return the full response if prompt removal fails
-            response = str(generated_text)
-
-        return response
+        # Decode response
+        generated_tokens = outputs[0][inputs["input_ids"].shape[1]:]
+        response = self.tokenizer.decode(generated_tokens, skip_special_tokens=True)
+        
+        return response.strip()


 class OpenAIChat(LLMInterface):