update readme and add timer

2025-07-16 17:15:51 -07:00
parent f77c4e38cb
commit 51255bdffa
8 changed files with 32531 additions and 107 deletions
--- a/examples/email_data/LEANN_email_reader.py
+++ b/examples/email_data/LEANN_email_reader.py
@@ -12,9 +12,14 @@ class EmlxReader(BaseReader):
    Reads individual .emlx files from Apple Mail's storage format.
    """
    
-    def __init__(self) -> None:
-        """Initialize."""
-        pass
+    def __init__(self, include_html: bool = False) -> None:
+        """
+        Initialize.
+        
+        Args:
+            include_html: Whether to include HTML content in the email body (default: False)
+        """
+        self.include_html = include_html
    
    def load_data(self, input_dir: str, **load_kwargs: Any) -> List[Document]:
        """
@@ -66,8 +71,8 @@ class EmlxReader(BaseReader):
                                if msg.is_multipart():
                                    for part in msg.walk():
                                        if part.get_content_type() == "text/plain" or part.get_content_type() == "text/html":
-                                            # if part.get_content_type() == "text/html":
-                                            #     continue
+                                            if part.get_content_type() == "text/html" and not self.include_html:
+                                                continue
                                            body += part.get_payload(decode=True).decode('utf-8', errors='ignore')
                                            # break
                                else:
--- a/examples/mail_reader_leann.py
+++ b/examples/mail_reader_leann.py
@@ -1,9 +1,15 @@
 import os
+import sys
 import asyncio
 import dotenv
 import argparse
 from pathlib import Path
 from typing import List, Any
+
+# Add the project root to Python path so we can import from examples
+project_root = Path(__file__).parent.parent
+sys.path.insert(0, str(project_root))
+
 from leann.api import LeannBuilder, LeannSearcher, LeannChat
 from llama_index.core.node_parser import SentenceSplitter

@@ -12,7 +18,7 @@ dotenv.load_dotenv()
 # Default mail path for macOS
 DEFAULT_MAIL_PATH = "/Users/yichuan/Library/Mail/V10/0FCA0879-FD8C-4B7E-83BF-FDDA930791C5/[Gmail].mbox/All Mail.mbox/78BA5BE1-8819-4F9A-9613-EB63772F1DD0/Data"

-def create_leann_index_from_multiple_sources(messages_dirs: List[Path], index_path: str = "mail_index.leann", max_count: int = -1):
+def create_leann_index_from_multiple_sources(messages_dirs: List[Path], index_path: str = "mail_index.leann", max_count: int = -1, include_html: bool = False):
    """
    Create LEANN index from multiple mail data sources.
    
@@ -20,12 +26,13 @@ def create_leann_index_from_multiple_sources(messages_dirs: List[Path], index_pa
        messages_dirs: List of Path objects pointing to Messages directories
        index_path: Path to save the LEANN index
        max_count: Maximum number of emails to process per directory
+        include_html: Whether to include HTML content in email processing
    """
    print("Creating LEANN index from multiple mail data sources...")
    
    # Load documents using EmlxReader from LEANN_email_reader
-    from LEANN_email_reader import EmlxReader
-    reader = EmlxReader()
+    from examples.email_data.LEANN_email_reader import EmlxReader
+    reader = EmlxReader(include_html=include_html)
    # from email_data.email import EmlxMboxReader
    # from pathlib import Path
    # reader = EmlxMboxReader()
@@ -107,7 +114,7 @@ def create_leann_index_from_multiple_sources(messages_dirs: List[Path], index_pa
    
    return index_path

-def create_leann_index(mail_path: str, index_path: str = "mail_index.leann", max_count: int = 1000):
+def create_leann_index(mail_path: str, index_path: str = "mail_index.leann", max_count: int = 1000, include_html: bool = False):
    """
    Create LEANN index from mail data.
    
@@ -115,6 +122,7 @@ def create_leann_index(mail_path: str, index_path: str = "mail_index.leann", max
        mail_path: Path to the mail directory
        index_path: Path to save the LEANN index
        max_count: Maximum number of emails to process
+        include_html: Whether to include HTML content in email processing
    """
    print("Creating LEANN index from mail data...")
    INDEX_DIR = Path(index_path).parent
@@ -128,8 +136,8 @@ def create_leann_index(mail_path: str, index_path: str = "mail_index.leann", max
        print(f"\n[PHASE 1] Building Leann index...")

        # Load documents using EmlxReader from LEANN_email_reader
-        from LEANN_email_reader import EmlxReader
-        reader = EmlxReader()
+        from examples.email_data.LEANN_email_reader import EmlxReader
+        reader = EmlxReader(include_html=include_html)
        # from email_data.email import EmlxMboxReader
        # from pathlib import Path
        # reader = EmlxMboxReader()
@@ -194,16 +202,22 @@ async def query_leann_index(index_path: str, query: str):
        query: The query string
    """
    print(f"\n[PHASE 2] Starting Leann chat session...")
-    chat = LeannChat(index_path=index_path)
+    chat = LeannChat(index_path=index_path,
+                     llm_config={"type": "openai", "model": "gpt-4o"})
    
    print(f"You: {query}")
+    import time
+    start_time = time.time()
    chat_response = chat.ask(
        query, 
-        top_k=5, 
+        top_k=10, 
        recompute_beighbor_embeddings=True,
-        complexity=128,
-        beam_width=1
+        complexity=12,
+        beam_width=1,
+        
    )
+    end_time = time.time()
+    print(f"Time taken: {end_time - start_time} seconds")
    print(f"Leann: {chat_response}")

 async def main():
@@ -217,6 +231,8 @@ async def main():
                       help='Maximum number of emails to process (-1 means all)')
    parser.add_argument('--query', type=str, default="Give me some funny advertisement about apple or other companies",
                       help='Single query to run (default: runs example queries)')
+    parser.add_argument('--include-html', action='store_true', default=False,
+                       help='Include HTML content in email processing (default: False)')
    
    args = parser.parse_args()

@@ -232,7 +248,8 @@ async def main():
    print(f"Index directory: {INDEX_DIR}")
    
    # Find all Messages directories
-    from LEANN_email_reader import EmlxReader
+    
+    from examples.email_data.LEANN_email_reader import EmlxReader
    messages_dirs = EmlxReader.find_all_messages_directories(base_mail_path)
    
    if not messages_dirs:
@@ -240,7 +257,7 @@ async def main():
        return
    
    # Create or load the LEANN index from all sources
-    index_path = create_leann_index_from_multiple_sources(messages_dirs, INDEX_PATH, args.max_emails)
+    index_path = create_leann_index_from_multiple_sources(messages_dirs, INDEX_PATH, args.max_emails, args.include_html)
    
    if index_path:
        if args.query:
--- a/examples/mail_reader_llamaindex.py
+++ b/examples/mail_reader_llamaindex.py
@@ -1,6 +1,13 @@
 import os
+import sys
+import argparse
 from pathlib import Path
 from typing import List, Any
+
+# Add the project root to Python path so we can import from examples
+project_root = Path(__file__).parent.parent
+sys.path.insert(0, str(project_root))
+
 from llama_index.core import VectorStoreIndex, StorageContext
 from llama_index.core.node_parser import SentenceSplitter

@@ -11,11 +18,11 @@ import torch
 # --- END EMBEDDING MODEL ---

 # Import EmlxReader from the new module
-from LEANN_email_reader import EmlxReader
+from examples.email_data.LEANN_email_reader import EmlxReader

-def create_and_save_index(mail_path: str, save_dir: str = "mail_index_embedded", max_count: int = 1000):
+def create_and_save_index(mail_path: str, save_dir: str = "mail_index_embedded", max_count: int = 1000, include_html: bool = False):
    print("Creating index from mail data with embedded metadata...")
-    documents = EmlxReader().load_data(mail_path, max_count=max_count)
+    documents = EmlxReader(include_html=include_html).load_data(mail_path, max_count=max_count)
    if not documents:
        print("No documents loaded. Exiting.")
        return None
@@ -64,18 +71,33 @@ def query_index(index, query: str):
    print(f"Response: {response}")

 def main():
-    mail_path = "/Users/yichuan/Library/Mail/V10/0FCA0879-FD8C-4B7E-83BF-FDDA930791C5/[Gmail].mbox/All Mail.mbox/78BA5BE1-8819-4F9A-9613-EB63772F1DD0/Data/9/Messages"
-    save_dir = "mail_index_embedded"
+    # Parse command line arguments
+    parser = argparse.ArgumentParser(description='LlamaIndex Mail Reader - Create and query email index')
+    parser.add_argument('--mail-path', type=str, 
+                       default="/Users/yichuan/Library/Mail/V10/0FCA0879-FD8C-4B7E-83BF-FDDA930791C5/[Gmail].mbox/All Mail.mbox/78BA5BE1-8819-4F9A-9613-EB63772F1DD0/Data/9/Messages",
+                       help='Path to mail data directory')
+    parser.add_argument('--save-dir', type=str, default="mail_index_embedded",
+                       help='Directory to store the index (default: mail_index_embedded)')
+    parser.add_argument('--max-emails', type=int, default=10000,
+                       help='Maximum number of emails to process')
+    parser.add_argument('--include-html', action='store_true', default=False,
+                       help='Include HTML content in email processing (default: False)')
+    
+    args = parser.parse_args()
+    
+    mail_path = args.mail_path
+    save_dir = args.save_dir
+    
    if os.path.exists(save_dir) and os.path.exists(os.path.join(save_dir, "vector_store.json")):
        print("Loading existing index...")
        index = load_index(save_dir)
    else:
        print("Creating new index...")
-        index = create_and_save_index(mail_path, save_dir, max_count=10000)
+        index = create_and_save_index(mail_path, save_dir, max_count=args.max_emails, include_html=args.include_html)
    if index:
        queries = [
            "Hows Berkeley Graduate Student Instructor",
-            "how's the icloud related advertisement saying"
+            "how's the icloud related advertisement saying",
            "Whats the number of class recommend to take per semester for incoming EECS students"
        ]
        for query in queries: