Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 13 additions & 11 deletions packages/leann-core/src/leann/readers.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,17 +32,20 @@ def load_data(
)

history_db_path = os.path.join(chrome_profile_path, "History")
temp_db_path = "/tmp/leann_history_index_copy"

if not os.path.exists(history_db_path):
print(f"⚠️ Browser history database not found at: {history_db_path}")
return docs

source_conn = None
conn = None
try:
# Create a temporary copy to avoid "database is locked"
shutil.copy2(history_db_path, temp_db_path)

conn = sqlite3.connect(temp_db_path)
# SQLite's backup API captures committed rows from the main database and its WAL
# while keeping the live browser database untouched.
source_uri = f"{Path(history_db_path).resolve().as_uri()}?mode=ro"
source_conn = sqlite3.connect(source_uri, uri=True)
conn = sqlite3.connect(":memory:")
source_conn.backup(conn)
cursor = conn.cursor()

query = """
Expand Down Expand Up @@ -77,15 +80,14 @@ def load_data(
doc = Document(text=doc_content, metadata={"title": title[0:150], "url": url})
docs.append(doc)

conn.close()
if os.path.exists(temp_db_path):
os.remove(temp_db_path)

except Exception as e:
print(f"❌ Error reading browser history: {e}")
if os.path.exists(temp_db_path):
os.remove(temp_db_path)
return docs
finally:
if conn is not None:
conn.close()
if source_conn is not None:
source_conn.close()

return docs

Expand Down
92 changes: 92 additions & 0 deletions tests/test_chrome_history_reader.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
import importlib.util
import os
import sqlite3
import sys
import types
from pathlib import Path

try:
from llama_index.core import Document as _Document
except ModuleNotFoundError:
llama_index = types.ModuleType("llama_index")
llama_index_core = types.ModuleType("llama_index.core")
llama_index_readers = types.ModuleType("llama_index.core.readers")
llama_index_base = types.ModuleType("llama_index.core.readers.base")

class _Document:
def __init__(self, text, metadata):
self.text = text
self.metadata = metadata

class _BaseReader:
pass

llama_index_core.Document = _Document
llama_index_base.BaseReader = _BaseReader
sys.modules["llama_index"] = llama_index
sys.modules["llama_index.core"] = llama_index_core
sys.modules["llama_index.core.readers"] = llama_index_readers
sys.modules["llama_index.core.readers.base"] = llama_index_base

readers_path = Path(__file__).parents[1] / "packages/leann-core/src/leann/readers.py"
readers_spec = importlib.util.spec_from_file_location("leann_readers", readers_path)
assert readers_spec is not None and readers_spec.loader is not None
readers = importlib.util.module_from_spec(readers_spec)
readers_spec.loader.exec_module(readers)
ChromeHistoryReader = readers.ChromeHistoryReader


def test_chrome_history_reader_includes_rows_still_in_wal(tmp_path, monkeypatch):
profile_dir = tmp_path / "profile"
profile_dir.mkdir()
history_path = profile_dir / "History"

source = sqlite3.connect(history_path)
assert source.execute("PRAGMA journal_mode=WAL").fetchone() == ("wal",)
source.execute(
"""
CREATE TABLE urls (
last_visit_time INTEGER,
url TEXT,
title TEXT,
visit_count INTEGER,
typed_count INTEGER,
hidden INTEGER
)
"""
)
source.commit()
source.execute("PRAGMA wal_checkpoint(TRUNCATE)")
source.execute(
"INSERT INTO urls VALUES (?, ?, ?, ?, ?, ?)",
(13300000000000000, "https://example.com/recent", "Recent page", 1, 0, 0),
)
source.commit()

isolated_copy = tmp_path / "main-database-only"
legacy_temp_path = "/tmp/leann_history_index_copy"
real_connect = sqlite3.connect
real_copy2 = readers.shutil.copy2
real_exists = os.path.exists

def redirected_connect(database, *args, **kwargs):
if os.fspath(database) == legacy_temp_path:
database = isolated_copy
return real_connect(database, *args, **kwargs)

def redirected_copy(source_path, _destination, *args, **kwargs):
return real_copy2(source_path, isolated_copy, *args, **kwargs)

def redirected_exists(path):
if os.fspath(path) == legacy_temp_path:
return False
return real_exists(path)

monkeypatch.setattr(readers.sqlite3, "connect", redirected_connect)
monkeypatch.setattr(readers.shutil, "copy2", redirected_copy)
monkeypatch.setattr(readers.os.path, "exists", redirected_exists)

documents = ChromeHistoryReader().load_data(str(profile_dir))
source.close()

assert [document.metadata["url"] for document in documents] == ["https://example.com/recent"]
Loading