add google hostory api

2025-07-11 21:21:36 -07:00
parent 16ee9d0422
commit 8239bbb48f
4 changed files with 497 additions and 76 deletions
--- a/examples/mail_reader_leann.py
+++ b/examples/mail_reader_leann.py
@@ -25,54 +25,55 @@ def create_leann_index_from_multiple_sources(messages_dirs: List[Path], index_pa
    # from email_data.email import EmlxMboxReader
    # from pathlib import Path
    # reader = EmlxMboxReader()
-    
-    all_documents = []
-    total_processed = 0
-    
-    # Process each Messages directory
-    for i, messages_dir in enumerate(messages_dirs):
-        print(f"\nProcessing Messages directory {i+1}/{len(messages_dirs)}: {messages_dir}")
-        
-        try:
-            documents = reader.load_data(messages_dir)
-            if documents:
-                print(f"Loaded {len(documents)} email documents from {messages_dir}")
-                all_documents.extend(documents)
-                total_processed += len(documents)
-                
-                # Check if we've reached the max count
-                if max_count > 0 and total_processed >= max_count:
-                    print(f"Reached max count of {max_count} documents")
-                    break
-            else:
-                print(f"No documents loaded from {messages_dir}")
-        except Exception as e:
-            print(f"Error processing {messages_dir}: {e}")
-            continue
-    
-    if not all_documents:
-        print("No documents loaded from any source. Exiting.")
-        return None
-    
-    print(f"\nTotal loaded {len(all_documents)} email documents from {len(messages_dirs)} directories")
-    
-    # Create text splitter with 256 chunk size
-    text_splitter = SentenceSplitter(chunk_size=256, chunk_overlap=25)
-    
-    # Convert Documents to text strings and chunk them
-    all_texts = []
-    for doc in all_documents:
-        # Split the document into chunks
-        nodes = text_splitter.get_nodes_from_documents([doc])
-        for node in nodes:
-            all_texts.append(node.get_content())
-    
-    print(f"Created {len(all_texts)} text chunks from {len(all_documents)} documents")
-    
-    # Create LEANN index directory
    INDEX_DIR = Path(index_path).parent
    
    if not INDEX_DIR.exists():
+        print(f"--- Index directory not found, building new index ---")
+        all_documents = []
+        total_processed = 0
+        
+        # Process each Messages directory
+        for i, messages_dir in enumerate(messages_dirs):
+            print(f"\nProcessing Messages directory {i+1}/{len(messages_dirs)}: {messages_dir}")
+            
+            try:
+                documents = reader.load_data(messages_dir)
+                if documents:
+                    print(f"Loaded {len(documents)} email documents from {messages_dir}")
+                    all_documents.extend(documents)
+                    total_processed += len(documents)
+                    
+                    # Check if we've reached the max count
+                    if max_count > 0 and total_processed >= max_count:
+                        print(f"Reached max count of {max_count} documents")
+                        break
+                else:
+                    print(f"No documents loaded from {messages_dir}")
+            except Exception as e:
+                print(f"Error processing {messages_dir}: {e}")
+                continue
+        
+        if not all_documents:
+            print("No documents loaded from any source. Exiting.")
+            return None
+        
+        print(f"\nTotal loaded {len(all_documents)} email documents from {len(messages_dirs)} directories")
+        
+        # Create text splitter with 256 chunk size
+        text_splitter = SentenceSplitter(chunk_size=256, chunk_overlap=25)
+        
+        # Convert Documents to text strings and chunk them
+        all_texts = []
+        for doc in all_documents:
+            # Split the document into chunks
+            nodes = text_splitter.get_nodes_from_documents([doc])
+            for node in nodes:
+                all_texts.append(node.get_content())
+        
+        print(f"Created {len(all_texts)} text chunks from {len(all_documents)} documents")
+        
+        # Create LEANN index directory
+
        print(f"--- Index directory not found, building new index ---")
        INDEX_DIR.mkdir(exist_ok=True)

@@ -112,35 +113,6 @@ def create_leann_index(mail_path: str, index_path: str = "mail_index.leann", max
        max_count: Maximum number of emails to process
    """
    print("Creating LEANN index from mail data...")
-    
-    # Load documents using EmlxReader from LEANN_email_reader
-    from LEANN_email_reader import EmlxReader
-    reader = EmlxReader()
-    # from email_data.email import EmlxMboxReader
-    # from pathlib import Path
-    # reader = EmlxMboxReader()
-    documents = reader.load_data(Path(mail_path))
-    
-    if not documents:
-        print("No documents loaded. Exiting.")
-        return None
-    
-    print(f"Loaded {len(documents)} email documents")
-    
-    # Create text splitter with 256 chunk size
-    text_splitter = SentenceSplitter(chunk_size=256, chunk_overlap=25)
-    
-    # Convert Documents to text strings and chunk them
-    all_texts = []
-    for doc in documents:
-        # Split the document into chunks
-        nodes = text_splitter.get_nodes_from_documents([doc])
-        for node in nodes:
-            all_texts.append(node.get_content())
-    
-    print(f"Created {len(all_texts)} text chunks from {len(documents)} documents")
-    
-    # Create LEANN index directory
    INDEX_DIR = Path(index_path).parent
    
    if not INDEX_DIR.exists():
@@ -151,6 +123,42 @@ def create_leann_index(mail_path: str, index_path: str = "mail_index.leann", max
        
        print(f"\n[PHASE 1] Building Leann index...")

+        # Load documents using EmlxReader from LEANN_email_reader
+        from LEANN_email_reader import EmlxReader
+        reader = EmlxReader()
+        # from email_data.email import EmlxMboxReader
+        # from pathlib import Path
+        # reader = EmlxMboxReader()
+        documents = reader.load_data(Path(mail_path))
+        
+        if not documents:
+            print("No documents loaded. Exiting.")
+            return None
+        
+        print(f"Loaded {len(documents)} email documents")
+        
+        # Create text splitter with 256 chunk size
+        text_splitter = SentenceSplitter(chunk_size=256, chunk_overlap=25)
+        
+        # Convert Documents to text strings and chunk them
+        all_texts = []
+        for doc in documents:
+            # Split the document into chunks
+            nodes = text_splitter.get_nodes_from_documents([doc])
+            for node in nodes:
+                all_texts.append(node.get_content())
+        
+        print(f"Created {len(all_texts)} text chunks from {len(documents)} documents")
+        
+        # Create LEANN index directory
+
+        print(f"--- Index directory not found, building new index ---")
+        INDEX_DIR.mkdir(exist_ok=True)
+
+        print(f"--- Building new LEANN index ---")
+        
+        print(f"\n[PHASE 1] Building Leann index...")
+
        # Use HNSW backend for better macOS compatibility
        builder = LeannBuilder(
            backend_name="hnsw",
@@ -189,7 +197,7 @@ async def query_leann_index(index_path: str, query: str):
        query, 
        top_k=5, 
        recompute_beighbor_embeddings=True,
-        complexity=32,
+        complexity=128,
        beam_width=1
    )
    print(f"Leann: {chat_response}")
@@ -198,7 +206,7 @@ async def main():
    # Base path to the mail data directory
    base_mail_path = "/Users/yichuan/Library/Mail/V10/0FCA0879-FD8C-4B7E-83BF-FDDA930791C5/[Gmail].mbox/All Mail.mbox/78BA5BE1-8819-4F9A-9613-EB63772F1DD0/Data"
    
-    INDEX_DIR = Path("./mail_index_leann_raw_text_all")
+    INDEX_DIR = Path("./mail_index_leann_raw_text_all_dicts")
    INDEX_PATH = str(INDEX_DIR / "mail_documents.leann")
    
    # Find all Messages directories