diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index aeeb5a2..f372e09 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -94,6 +94,7 @@ jobs: - name: Upload an export as a visitor, then download and delete it run: | + docker exec demo bsdtar --version curl -fs -c visitor.txt http://127.0.0.1:7860/library/import \ -H 'X-ChatLore: 1' -H 'X-Filename: conversations.json' \ --data-binary @tests/fixtures/claude/conversations.json diff --git a/CHANGELOG.md b/CHANGELOG.md index b640677..b6b5d38 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -19,6 +19,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - A `Dockerfile` for the hosted demo, with the demo library and embedding model built in. CI builds it, starts it, and checks the web interface, the API, and a tool call over MCP. Guide in `docs/hosting.md`. +- Import anything: `chatlore import` takes any number of files, folders, and archives, and + **Your data** in the web interface takes many files or a whole folder. Archives (zip, tar, + gz, and with bsdtar 7z and rar) are unpacked, nested ones too; chat exports are recognised by + their content wherever they sit; documents become notes: PDF, Word, PowerPoint, Excel, + OpenDocument, EPUB, RTF, web pages, CSV, JSON, XML, code, and any plain text; email becomes a + conversation per message. Skipped files are listed with the reason. New sources `document` + and `email`. Guide in `docs/importers.md`. +- `POST /library/files` adds a file, with its path, to a batch that `POST /library/import` + then imports. - **Your data** in the web interface: upload a ChatGPT, Claude, or Gemini export, Markdown notes, or a ChatLore archive, and watch it be imported, embedded, and read into the knowledge graph in the background; download the library as a ChatLore archive or as Markdown. diff --git a/Dockerfile b/Dockerfile index 6e07976..6efcdf1 100644 --- a/Dockerfile +++ b/Dockerfile @@ -11,6 +11,9 @@ FROM ghcr.io/astral-sh/uv:0.12-python3.12-trixie-slim ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy UV_PYTHON_DOWNLOADS=never WORKDIR /app +# bsdtar unpacks 7z and rar uploads; zip and tar need nothing extra. +RUN apt-get update && apt-get install -y --no-install-recommends libarchive-tools && rm -rf /var/lib/apt/lists/* + # Dependencies first, so changing the code does not reinstall them. COPY pyproject.toml uv.lock ./ RUN uv sync --locked --no-dev --no-install-project diff --git a/README.md b/README.md index 00b25ee..3b8cea0 100644 --- a/README.md +++ b/README.md @@ -6,8 +6,9 @@ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE) [![Python 3.12+](https://img.shields.io/badge/python-3.12%2B-blue.svg)](pyproject.toml) -**Status: pre-alpha.** Importing and search work today: ChatGPT, Claude, Gemini, -and Markdown exports land in a local library and a SQLite graph store you can +**Status: pre-alpha.** Importing and search work today: ChatGPT, Claude, and +Gemini exports, notes, documents (PDF, Word, Excel, and more), email, and +archives of any of them land in a local library and a SQLite graph store you can search from the terminal by words, by meaning, or both. A language model turns them into a knowledge graph of entities, relationships, and topics that you can browse and search, and you can ask questions and get answers with sources, in @@ -34,7 +35,7 @@ Asking questions also needs a model key; see [docs/models.md](docs/models.md). ```bash uv tool install chatlore # or: pipx install chatlore -chatlore import path/to/chatgpt-export.zip +chatlore import path/to/chatgpt-export.zip ~/Documents/notes # exports, documents, archives chatlore search "postgres index" chatlore process # chunk and embed, local model chatlore search "why was my query slow" --semantic @@ -102,6 +103,7 @@ extraction is an optional enrichment you can re-run with a better model later. | M9 | Hosted demo | public mode, MCP over HTTP, and Docker image done; deployment next | | M10 | FalkorDB backend | done | | M11 | Your own data in the web interface: upload, export, private visitor libraries | done | +| M12 | Import anything: documents, email, data files, folders, and archives | done | ## Development setup diff --git a/docs/importers.md b/docs/importers.md index 2748726..480d40b 100644 --- a/docs/importers.md +++ b/docs/importers.md @@ -1,15 +1,18 @@ # Importing your data ```bash -chatlore import # detects the source from the content +chatlore import ... # anything: files, folders, archives chatlore import --source chatgpt chatlore import --dry-run # parse and report, write nothing chatlore stats # what the library holds ``` -`` can be the zip exactly as you downloaded it, the extracted folder, or -the JSON file itself. Importing is idempotent: run it again after a fresh -export and only new or changed conversations are written. +Give it as many files, folders, and archives as you like, and it works out what +each one is from its content: the export zip exactly as you downloaded it, the +extracted folder, a folder of documents, or all of them at once. The web +interface does the same under **Your data**, for files or a whole folder. +Importing is idempotent: run it again after a fresh export and only new or +changed conversations are written. Everything lands in `~/.chatlore/conversations//.json` (override the location with `--home` or `CHATLORE_HOME`). The files are plain JSON and stay readable @@ -19,6 +22,42 @@ SQLite database holding the graph and the search indexes. An archive written by `chatlore export` is recognised too, and brings its knowledge graph and caches along; see [export.md](export.md). +## What can be imported + +| What | Files | Becomes | +|---|---|---| +| Chat exports | ChatGPT and Claude `conversations.json`, Gemini `MyActivity.json`, wherever they sit | conversations | +| Notes | Markdown and text (`.md`, `.markdown`, `.txt`), Obsidian vaults | notes, source `markdown` | +| Documents | PDF, Word (`.docx`), PowerPoint (`.pptx`), Excel (`.xlsx`), OpenDocument (`.odt`, `.odp`, `.ods`), EPUB, RTF, web pages (`.html`) | notes, source `document` | +| Data | CSV and TSV, JSON and JSON Lines, XML | notes: a line per row, or `key.path: value` lines | +| Code and text | source code, `.log`, `.rst`, `.org`, subtitles, and any other file that is plain text | notes; code keeps its language | +| Email | `.eml`, and `.mbox` mailboxes | one conversation per email, source `email` | +| Archives | zip, tar, tar.gz, tgz, tar.bz2, tar.xz, gz, 7z, rar | whatever they hold | + +**Archives** are unpacked, and archives inside them too, four levels deep, so a +zip of folders of zips works, and so does a Google Takeout split into several +zips. zip, tar, and gz need nothing extra. 7z and rar are unpacked with +`bsdtar`, which comes with Windows 10 and later and with macOS; on Linux, +install `libarchive-tools`. Together they may unpack to at most 2 GB, 100,000 +files, and nothing is ever written outside the folder they are unpacked into. + +**Chat exports** are found by their content, not their names or places. The +other files of an export, such as ChatGPT's `chat.html`, which holds the same +chats again, are skipped. In a Google Takeout, Gemini's activity is imported +and other products' activity is skipped, while other files, such as Drive +documents, are read like any others. + +**Documents** become notes of one message each, titled from the document when +it names itself (a Word title, a PDF's metadata, a web page's ``) and +otherwise by file name. Each is known by its path within what was imported, so +importing the same folder again updates the same notes. A scanned PDF with no +text layer has no text to read and is skipped. + +**Skipped files** are listed at the end with the reason: pictures, audio, +video, programs, files that are not text, and documents that could not be read. +Folders such as `.git` and `node_modules` are left out, and so are system files +like `.DS_Store`. + ## Searching ```bash diff --git a/docs/web.md b/docs/web.md index 8f20046..b984c9a 100644 --- a/docs/web.md +++ b/docs/web.md @@ -32,15 +32,19 @@ follows the system's light or dark setting. **Your data**, in the navigation, imports an export and downloads the library: -- Drop a file on the dialog, or choose one: a ChatGPT or Claude export (.zip), - Gemini Takeout (.zip or MyActivity.json), Markdown notes (.zip or .md), or a - ChatLore archive. Uploads may be up to 200 MB, or what `--max-upload-mb` sets. +- Drop files or folders on the dialog, or choose files or a whole folder: chat + exports, documents, notes, email, data files, and archives of any of them, as + listed in [importers.md](importers.md#what-can-be-imported). Each file is sent + with its path in its folder, then all of them are imported together. Uploads + may be up to 200 MB in all, or what `--max-upload-mb` sets; folders such as + `.git` and `node_modules` are left out before anything is sent. - The import runs in the background, like `chatlore import`, `chatlore process`, and `chatlore extract` one after another, and the dialog shows each step with its progress: importing, preparing search, reading with the language - model, summarising, linking names for the same thing, and topics. Without a - model key, everything but the knowledge graph is built. When it is done, - **Show the library** reloads the page on it. + model, summarising, linking names for the same thing, and topics. It then + says what came from where, and lists the files it skipped with the reason. + Without a model key, everything but the knowledge graph is built. When it is + done, **Show the library** reloads the page on it. - **ChatLore archive** downloads the whole library, graph and caches included, to import anywhere; **Markdown** downloads one readable file per conversation. See [export.md](export.md). diff --git a/pyproject.toml b/pyproject.toml index d6da464..4798c63 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -36,6 +36,7 @@ dependencies = [ "networkx>=3.7", "openai>=3.17", "pydantic>=2.9", + "pypdf>=6.0", "python-dotenv>=1.2", "pyyaml>=6.0", "rich>=13.7", diff --git a/src/chatlore/api.py b/src/chatlore/api.py index a2e3f64..cf17750 100644 --- a/src/chatlore/api.py +++ b/src/chatlore/api.py @@ -25,6 +25,7 @@ from __future__ import annotations import functools +import re import shutil import tempfile import threading @@ -35,7 +36,7 @@ from contextlib import asynccontextmanager from contextvars import ContextVar from http.cookies import SimpleCookie -from pathlib import Path +from pathlib import Path, PurePosixPath from typing import Annotated, Any from urllib.parse import unquote @@ -54,7 +55,7 @@ from chatlore.archive import export_archive, export_markdown from chatlore.chat import MAX_SOURCES, ChatLimits, Context, Source, answer, cited, retrieve from chatlore.embeddings import Embedder, EmbeddingError, make_embedder, normalise -from chatlore.imports import Imports, upload_suffix +from chatlore.imports import Imports, safe_relative from chatlore.library import Library from chatlore.llm import LLMError, make_llm from chatlore.mcp_server import create_server @@ -176,6 +177,10 @@ def _topic(node: Node, entities: int | None = None) -> dict[str, Any]: without the browser first asking this server, which does not allow it, so it cannot upload or delete through a visitor's browser.""" _SWEEP_SECONDS = 600 +MAX_BATCH_FILES = 20_000 +"""Files one upload may hold.""" +_BATCH = re.compile(r"[0-9a-f]{32}") +_STALE_BATCH_SECONDS = 24 * 3600 # The visitor's own library, when the request carries a token for one. _visitor: ContextVar[Space | None] = ContextVar("chatlore_visitor", default=None) @@ -212,6 +217,24 @@ def _cookie(scope: Scope, name: str) -> str | None: return None +def _forget_stale_batches(uploads: Path) -> None: + """Delete batches that were uploaded but never imported, a day on.""" + if not uploads.exists(): + return + now = time.time() + for folder in uploads.iterdir(): + try: + stale = folder.is_dir() and now - folder.stat().st_mtime > _STALE_BATCH_SECONDS + except OSError: + continue + if stale: + shutil.rmtree(folder, ignore_errors=True) + + +def _only_file(folder: Path) -> str: + return next((path.name for path in folder.rglob("*") if path.is_file()), "1 file") + + def _changes_allowed(request: Request) -> None: if request.headers.get(REQUEST_HEADER) != "1": raise HTTPException(403, f"changes need the {REQUEST_HEADER} header") @@ -334,65 +357,118 @@ def library_info() -> dict[str, Any]: "import": status.as_dict() if status is not None else None, } - @app.post("/library/import", status_code=202) - async def import_upload(request: Request, response: Response) -> dict[str, Any]: - """Take an export as the request body and import it in the background. + staged: dict[tuple[Path, str], list[int]] = {} + staging = threading.Lock() - Send the file's name in ``X-Filename``. On a public server the upload goes - into the visitor's own library, made on the first upload, and a cookie - remembers it. ``GET /library`` shows how far the import got. - """ - _changes_allowed(request) + def _target(request: Request, response: Response) -> tuple[Path, Callable[[], GraphStore]]: + """The library uploads go into: the server's, or the visitor's, made on first use.""" if not uploads: raise HTTPException(403, "this server does not take uploads") - declared = request.headers.get("content-length") - if declared is not None and declared.isdigit() and int(declared) > max_upload: - raise HTTPException(413, f"uploads may be up to {max_upload // 1_000_000} MB") + if not public: + return library, lambda: open_store(library) + assert spaces is not None space = _visitor.get() - created: Space | None = None - if public: - assert spaces is not None - if space is None: - token, space = spaces.create() - created = space - response.set_cookie( - COOKIE, - token, - max_age=int(spaces.keep.total_seconds()), - httponly=True, - samesite="lax", - secure=request.url.scheme == "https", - ) - target, opener = space.home, space.open_store - else: - target, opener = library, lambda: open_store(library) + if space is None: + token, space = spaces.create() + response.set_cookie( + COOKIE, + token, + max_age=int(spaces.keep.total_seconds()), + httponly=True, + samesite="lax", + secure=request.url.scheme == "https", + ) + return space.home, space.open_store + + async def _receive(request: Request, path: Path, room: int) -> int: + """Write the request body to ``path``, refusing more than ``room`` bytes.""" + declared = request.headers.get("content-length") + if declared is not None and declared.isdigit() and int(declared) > room: + raise HTTPException(413, f"uploads may be up to {max_upload // 1_000_000} MB in all") + path.parent.mkdir(parents=True, exist_ok=True) + size = 0 try: - if imports.running(target): - raise HTTPException(409, "an import is already running for this library") - name = unquote(request.headers.get("x-filename") or "upload")[:200] - folder = target / "uploads" - folder.mkdir(parents=True, exist_ok=True) - path = folder / f"{uuid.uuid4().hex}{upload_suffix(name)}" - size = 0 + with path.open("wb") as handle: + async for piece in request.stream(): + size += len(piece) + if size > room: + raise HTTPException( + 413, f"uploads may be up to {max_upload // 1_000_000} MB in all" + ) + handle.write(piece) + except BaseException: + path.unlink(missing_ok=True) + raise + return size + + def _batch_folder(target: Path, batch: str) -> Path: + if not _BATCH.fullmatch(batch): + raise HTTPException(422, "a batch is 32 hexadecimal digits") + return target / "uploads" / batch + + @app.post("/library/files") + async def upload_file( + request: Request, response: Response, batch: Annotated[str, Query()] + ) -> dict[str, Any]: + """Add one file to a batch of uploads; ``POST /library/import`` imports the batch. + + Send the file as the request body and its path in ``X-Filename``, such as + ``Export/conversations.json`` for a file chosen with its folder. ``batch`` + is any 32 hexadecimal digits the client picks for the whole upload. The + files of a batch may be up to the upload limit in all. + """ + _changes_allowed(request) + target, _ = _target(request, response) + folder = _batch_folder(target, batch) + with staging: + used = staged.setdefault((target, batch), [0, 0]) + if used[1] >= MAX_BATCH_FILES: + raise HTTPException(413, f"a batch may hold up to {MAX_BATCH_FILES:,} files") + used[1] += 1 + if not folder.exists(): + _forget_stale_batches(target / "uploads") + name = safe_relative(unquote(request.headers.get("x-filename") or "upload")) + size = await _receive(request, folder / name, max_upload - used[0]) + with staging: + used[0] += size + return {"batch": batch, "files": used[1], "bytes": used[0]} + + @app.post("/library/import", status_code=202) + async def import_upload( + request: Request, response: Response, batch: Annotated[str | None, Query()] = None + ) -> dict[str, Any]: + """Import a batch of files sent to ``POST /library/files``, or one file sent here. + + Without ``batch``, the request body is the file and ``X-Filename`` its name. + Archives are unpacked, chat exports recognised, and other files read as + notes, in the background. On a public server the upload goes into the + visitor's own library, made on the first upload, and a cookie remembers it. + ``GET /library`` shows how far the import got. + """ + _changes_allowed(request) + target, opener = _target(request, response) + if imports.running(target): + raise HTTPException(409, "an import is already running for this library") + if batch is not None: + folder = _batch_folder(target, batch) + with staging: + used = staged.pop((target, batch), [0, 0]) + if not folder.exists() or not any(folder.iterdir()): + raise HTTPException(400, "the batch holds no files") + name = f"{used[1]:,} files" if used[1] != 1 else _only_file(folder) + else: + name = safe_relative(unquote(request.headers.get("x-filename") or "upload")) + folder = _batch_folder(target, uuid.uuid4().hex) + _forget_stale_batches(target / "uploads") try: - with path.open("wb") as handle: - async for piece in request.stream(): - size += len(piece) - if size > max_upload: - raise HTTPException( - 413, f"uploads may be up to {max_upload // 1_000_000} MB" - ) - handle.write(piece) + size = await _receive(request, folder / name, max_upload) if size == 0: raise HTTPException(400, "the upload is empty") except BaseException: - path.unlink(missing_ok=True) + shutil.rmtree(folder, ignore_errors=True) raise - except BaseException: - if created is not None and spaces is not None: - spaces.delete(created) - raise - return imports.start(target, path, name, opener).as_dict() + name = PurePosixPath(name).name + return imports.start(target, folder, name, opener).as_dict() @app.get("/library/export") def export_library( diff --git a/src/chatlore/cli.py b/src/chatlore/cli.py index faa80bc..95c8840 100644 --- a/src/chatlore/cli.py +++ b/src/chatlore/cli.py @@ -6,6 +6,7 @@ import platform import re import sys +import tempfile import threading import webbrowser from collections import Counter @@ -36,10 +37,10 @@ from chatlore.importers import ( ImporterError, ImportIssue, - detect_source, get_importer, make_note, ) +from chatlore.importers.intake import Intake, describe from chatlore.library import AddOutcome, Library from chatlore.llm import OPENROUTER_KEY_ENV, LLMError, llm_settings, make_llm from chatlore.paths import HOME_ENV, default_home, demo_home, spaces_home @@ -183,9 +184,14 @@ def _describe_llm() -> str: @app.command("import") def import_( - path: Annotated[ - Path, - typer.Argument(help="Export zip, extracted folder, JSON file, or Markdown folder."), + paths: Annotated[ + list[Path], + typer.Argument( + help=( + "Files, folders, or archives: chat exports, documents, notes, email, " + "data files, or zip, tar, 7z, and rar archives holding any of them." + ), + ), ], source: Annotated[ str, @@ -194,7 +200,7 @@ def import_( "-s", help=( "chatgpt, claude, gemini, markdown, chatlore for an archive from " - "`chatlore export`, or auto to detect it from the content." + "`chatlore export`, or auto to recognise everything from its content." ), ), ] = "auto", @@ -203,16 +209,31 @@ def import_( typer.Option("--dry-run", help="Parse and report without writing anything."), ] = False, ) -> None: - """Import an export into the local library. Safe to run repeatedly.""" - if source == "chatlore" or (source == "auto" and is_archive(path)): - _import_archive(path, dry_run) + """Import exports, documents, and notes into the local library. Safe to run repeatedly. + + Folders are searched and archives unpacked, archives inside them too. Chat + exports are recognised wherever they sit, other files become notes, and every + file that is skipped is listed with the reason. + """ + if source == "chatlore" or (source == "auto" and len(paths) == 1 and is_archive(paths[0])): + for path in paths: + _import_archive(path, dry_run) + return + if source != "auto": + for path in paths: + _import_as(path, source, dry_run) return + _import_anything(paths, dry_run) + + +def _import_as(path: Path, source: str, dry_run: bool) -> None: + """Import one path with the importer named, as chatlore import always has.""" issues: list[ImportIssue] = [] outcomes: Counter[str] = Counter() messages = 0 try: - kind = detect_source(path) if source == "auto" else get_importer(source).kind + kind = get_importer(source).kind importer = get_importer(kind.value) with ( Library(default_home()) as library, @@ -233,8 +254,62 @@ def import_( console.print(f"[red]Import failed:[/red] {error}") raise typer.Exit(code=1) from error - title = f"{kind.value} import" + (" (dry run)" if dry_run else "") - table = Table(title=title, show_header=False) + _import_summary(f"{kind.value} import", outcomes, messages, issues, dry_run) + + +def _import_anything(paths: list[Path], dry_run: bool) -> None: + """Import whatever the paths hold, recognising each file from its content.""" + issues: list[ImportIssue] = [] + outcomes: Counter[str] = Counter() + messages = 0 + home = default_home() + with tempfile.TemporaryDirectory(prefix="chatlore-import-") as folder: + intake = Intake(Path(folder)) + try: + intake.scan(paths) + if not intake.found_anything(): + console.print("[red]Import failed:[/red] nothing ChatLore can read was found.") + for line in describe(intake.skipped): + console.print(f" [yellow]skipped[/yellow] {escape(line)}") + raise typer.Exit(code=1) + with ( + Library(home) as library, + _store(dry_run) as store, + _writes(store), + _writer(store) as writer, + ): + for conversation in intake.conversations(issues.append): + messages += len(conversation.messages) + if dry_run: + outcomes["parsed"] += 1 + continue + outcome = library.add(conversation) + outcomes[outcome.value] += 1 + if outcome is not AddOutcome.UNCHANGED and writer is not None: + writer.add(conversation) + for archive, _ in intake.archives: + report = import_archive(archive, home, dry_run=dry_run) + outcomes.update(report.outcomes) + messages += report.messages + intake.sources["archive"] += sum(report.outcomes.values()) + except (ImporterError, ArchiveError) as error: + console.print(f"[red]Import failed:[/red] {error}") + raise typer.Exit(code=1) from error + + kinds = [kind for kind, count in intake.sources.items() if count] + title = f"{kinds[0]} import" if len(kinds) == 1 else "import" + _import_summary(title, outcomes, messages, issues, dry_run, intake) + + +def _import_summary( + title: str, + outcomes: Counter[str], + messages: int, + issues: list[ImportIssue], + dry_run: bool, + intake: Intake | None = None, +) -> None: + table = Table(title=title + (" (dry run)" if dry_run else ""), show_header=False) table.add_column("key", style="bold") table.add_column("value", justify="right") if dry_run: @@ -243,10 +318,18 @@ def import_( for outcome in AddOutcome: table.add_row(outcome.value, str(outcomes[outcome.value])) table.add_row("messages", str(messages)) + if intake is not None and len([count for count in intake.sources.values() if count]) > 1: + for kind, count in sorted(intake.sources.items()): + table.add_row(f"from {kind}", str(count)) table.add_row("skipped records", str(len(issues))) + if intake is not None: + table.add_row("skipped files", str(len(intake.skipped))) console.print(table) _print_issues(issues) + if intake is not None: + for line in describe(intake.skipped): + console.print(f" [yellow]skipped[/yellow] {escape(line)}") if not dry_run: console.print(f"Library: {display_path(default_home())}") diff --git a/src/chatlore/importers/documents.py b/src/chatlore/importers/documents.py new file mode 100644 index 0000000..579b960 --- /dev/null +++ b/src/chatlore/importers/documents.py @@ -0,0 +1,578 @@ +"""Reading documents of many kinds into notes: text, data, office files, PDF, and email. + +Each document becomes a conversation holding one message with its text, like a +Markdown note, so it is chunked, searched, and read into the knowledge graph +the same way. An email becomes a conversation of its own, and a mailbox one per +email. The file's path within what was imported is its identity, so importing +the same files again changes nothing. + +Office files (Word, PowerPoint, Excel, OpenDocument) and EPUB books are zipped +XML or HTML and are read with the standard library; only PDF needs a library. +A document with no text, such as a scanned PDF without a text layer, is +skipped with that reason. +""" + +from __future__ import annotations + +import csv +import email +import email.policy +import io +import json +import mailbox +import re +import zipfile +from collections.abc import Callable, Iterator +from datetime import UTC, datetime +from email.message import EmailMessage +from email.utils import parsedate_to_datetime +from html.parser import HTMLParser +from pathlib import Path, PurePosixPath +from typing import Any +from xml.etree import ElementTree + +from pydantic import ValidationError + +from chatlore.ids import conversation_id, message_id +from chatlore.importers.markdown import _conversation as markdown_note +from chatlore.models import ContentPart, Conversation, Message, PartType, Role, SourceKind + +MAX_DOCUMENT_BYTES = 200_000_000 +"""Larger files are skipped rather than read.""" + +NOTES = frozenset({".md", ".markdown", ".txt"}) +"""Read as Markdown notes, as ``chatlore import`` always has.""" +TEXT = frozenset({".text", ".rst", ".org", ".log", ".adoc", ".tex", ".srt", ".vtt"}) +CODE = { + ".py": "python", ".js": "javascript", ".mjs": "javascript", ".ts": "typescript", + ".tsx": "tsx", ".jsx": "jsx", ".java": "java", ".kt": "kotlin", ".go": "go", + ".rs": "rust", ".c": "c", ".h": "c", ".cpp": "cpp", ".hpp": "cpp", ".cs": "csharp", + ".rb": "ruby", ".php": "php", ".swift": "swift", ".scala": "scala", ".sh": "bash", + ".ps1": "powershell", ".sql": "sql", ".r": "r", ".lua": "lua", ".dart": "dart", + ".css": "css", ".scss": "scss", ".yaml": "yaml", ".yml": "yaml", ".toml": "toml", + ".ini": "ini", ".cfg": "ini", ".ipynb": "json", +} # fmt: skip +DATA = frozenset({".csv", ".tsv", ".json", ".jsonl", ".ndjson", ".xml"}) +MARKUP = frozenset({".html", ".htm", ".xhtml"}) +OFFICE = frozenset({".docx", ".pptx", ".xlsx", ".odt", ".odp", ".ods", ".epub", ".rtf", ".pdf"}) +MAIL = frozenset({".eml", ".mbox"}) +MEDIA = frozenset( + {".png", ".jpg", ".jpeg", ".gif", ".webp", ".heic", ".bmp", ".tif", ".tiff", ".svg", + ".ico", ".mp3", ".wav", ".m4a", ".aac", ".ogg", ".flac", ".opus", ".mp4", ".mov", + ".avi", ".mkv", ".webm", ".wmv", ".ttf", ".otf", ".woff", ".woff2", ".exe", ".dll", + ".so", ".dylib", ".bin", ".iso", ".dmg", ".pyc", ".class", ".jar", ".db", ".sqlite"} +) # fmt: skip +READABLE = NOTES | TEXT | frozenset(CODE) | DATA | MARKUP | OFFICE | MAIL + +_SNIFF_BYTES = 8192 +_TITLE_LENGTH = 120 + + +class DocumentError(Exception): + """A document cannot be read; the message says why.""" + + +def readable(path: Path) -> bool: + """Whether the file is a kind ChatLore reads, by its name or, failing that, by looking.""" + suffix = path.suffix.lower() + if suffix in READABLE: + return True + if suffix in MEDIA: + return False + return _looks_like_text(path) + + +def read_document(path: Path, relative: str) -> list[Conversation]: + """The conversations in one document: usually one, one per email in a mailbox. + + Raises ``DocumentError`` when it cannot be read or holds no text. + """ + if path.stat().st_size > MAX_DOCUMENT_BYTES: + raise DocumentError("too large to read") + suffix = path.suffix.lower() + try: + if suffix in NOTES: + try: + note = markdown_note(path, relative) + except UnicodeDecodeError: + note = None # not UTF-8: read below as text in whatever encoding it has + else: + if note is None: + raise DocumentError("it is empty") + return [note] + if suffix in MAIL: + return _emails(path, relative) + reader = _READERS.get(suffix) + if reader is not None: + title, text, created = reader(path) + parts = [ContentPart(text=text)] + elif suffix in CODE: + title, created = None, None + parts = [ContentPart(type=PartType.CODE, language=CODE[suffix], text=_text(path))] + else: + title, text, created = None, _text(path), None + parts = [ContentPart(text=text)] + except DocumentError: + raise + except (OSError, ValueError, KeyError, IndexError, zipfile.BadZipFile) as error: + raise DocumentError(f"it could not be read: {error}") from error + except ElementTree.ParseError as error: + raise DocumentError(f"its XML could not be read: {error}") from error + except Exception as error: # a library failing on one odd file must not stop an import + raise DocumentError(f"it could not be read: {error}") from error + if not any(part.text.strip() for part in parts): + raise DocumentError("it holds no text" + (" (a scan?)" if suffix == ".pdf" else "")) + return [_note(relative, title or PurePosixPath(relative).stem, parts, created, suffix)] + + +# -- building notes -------------------------------------------------------------- + + +def _note( + relative: str, + title: str, + parts: list[ContentPart], + created: datetime | None, + suffix: str, + source: SourceKind = SourceKind.DOCUMENT, + key: str | None = None, +) -> Conversation: + external = key or relative + conv_id = conversation_id(source, external) + try: + return Conversation( + id=conv_id, + source=source, + external_id=external, + title=" ".join(title.split())[:_TITLE_LENGTH] or PurePosixPath(relative).stem, + created_at=created, + updated_at=created, + messages=[ + Message( + id=message_id(conv_id, "body"), + role=Role.USER, + created_at=created, + content=[part for part in parts if part.text.strip()], + ) + ], + metadata={"kind": "document", "path": relative, "type": suffix.lstrip(".")}, + ) + except ValidationError as error: + raise DocumentError(str(error).splitlines()[0]) from error + + +def _emails(path: Path, relative: str) -> list[Conversation]: + if path.suffix.lower() == ".eml": + messages = [email.message_from_bytes(path.read_bytes(), policy=email.policy.default)] + else: + box = mailbox.mbox(path, factory=None, create=False) + try: + messages = [ + email.message_from_bytes(bytes(item), policy=email.policy.default) for item in box + ] + finally: + box.close() + found: list[Conversation] = [] + for number, message in enumerate(messages): + body = _email_body(message) + if not body.strip(): + continue + subject = str(message.get("subject") or "") or "(no subject)" + sender = str(message.get("from") or "") + header = "\n".join( + f"{name}: {value}" + for name, value in (("From", sender), ("To", str(message.get("to") or ""))) + if value + ) + key = str(message.get("message-id") or "") or f"{relative}#{number}" + found.append( + _note( + relative, + subject, + [ContentPart(text=f"{header}\n\n{body}".strip())], + _email_date(message), + path.suffix.lower(), + source=SourceKind.EMAIL, + key=key if len(messages) > 1 or key != f"{relative}#0" else relative, + ) + ) + if not found: + raise DocumentError("it holds no email with text") + return found + + +def _email_body(message: EmailMessage) -> str: + part = message.get_body(preferencelist=("plain", "html")) + if part is None: + return "" + content = part.get_content() + text = content if isinstance(content, str) else "" + return _html_text(text)[1] if part.get_content_subtype() == "html" else text + + +def _email_date(message: EmailMessage) -> datetime | None: + try: + return parsedate_to_datetime(str(message.get("date"))) + except (TypeError, ValueError, IndexError): + return None + + +# -- plain text and data --------------------------------------------------------- + + +def _decode(data: bytes) -> str: + if data.startswith((b"\xff\xfe", b"\xfe\xff")): + return data.decode("utf-16") + for encoding in ("utf-8-sig", "cp1252"): + try: + return data.decode(encoding) + except UnicodeDecodeError: + continue + return data.decode("latin-1") + + +def _text(path: Path) -> str: + return _decode(path.read_bytes()) + + +def _looks_like_text(path: Path) -> bool: + try: + with path.open("rb") as handle: + head = handle.read(_SNIFF_BYTES) + except OSError: + return False + if not head or b"\x00" in head: + return False + try: + head.decode("utf-8") + except UnicodeDecodeError as error: + return error.start > len(head) - 4 # a character cut off at the end + return True + + +def _table(path: Path) -> tuple[str | None, str, datetime | None]: + text = _text(path) + dialect = csv.excel_tab if path.suffix.lower() == ".tsv" else csv.excel + rows = [ + row for row in csv.reader(io.StringIO(text), dialect) if any(cell.strip() for cell in row) + ] + if not rows: + return None, "", None + header, *body = rows + if not body: + return None, " | ".join(header), None + lines = [ + " | ".join(f"{name}: {value}" for name, value in zip(header, row, strict=False) if value) + for row in body + ] + return None, "\n".join(lines), None + + +def _json(path: Path) -> tuple[str | None, str, datetime | None]: + text = _text(path) + if path.suffix.lower() in {".jsonl", ".ndjson"}: + values = [json.loads(line) for line in text.splitlines() if line.strip()] + return None, "\n\n".join("\n".join(_flatten(value)) for value in values), None + return None, "\n".join(_flatten(json.loads(text))), None + + +def _flatten(value: Any, prefix: str = "") -> Iterator[str]: + """``key.path: value`` lines, so any JSON reads as text.""" + if isinstance(value, dict): + for key, inner in value.items(): + yield from _flatten(inner, f"{prefix}.{key}" if prefix else str(key)) + elif isinstance(value, list): + for index, inner in enumerate(value): + yield from _flatten(inner, f"{prefix}[{index}]") + elif value is not None and str(value).strip(): + yield f"{prefix}: {value}" if prefix else str(value) + + +def _xml(path: Path) -> tuple[str | None, str, datetime | None]: + root = ElementTree.fromstring(path.read_bytes()) + return None, _joined(root.itertext()), None + + +def _joined(pieces: Iterator[str] | list[str]) -> str: + lines = (" ".join(piece.split()) for piece in pieces) + return "\n".join(line for line in lines if line) + + +# -- HTML and markup --------------------------------------------------------------- + + +class _TextOfHtml(HTMLParser): + """The visible text of a page, a paragraph per block, and its title.""" + + _BLOCKS = frozenset( + {"p", "div", "br", "li", "tr", "h1", "h2", "h3", "h4", "h5", "h6", "section", + "article", "blockquote", "pre", "td", "th", "dt", "dd", "header", "footer"} + ) # fmt: skip + _HIDDEN = frozenset({"script", "style", "noscript", "template", "svg", "head"}) + + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self.pieces: list[str] = [] + self.title = "" + self._hidden = 0 + self._in_title = False + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + if tag == "title": + self._in_title = True + elif tag in self._HIDDEN: + self._hidden += 1 + elif tag in self._BLOCKS: + self.pieces.append("\n") + + def handle_endtag(self, tag: str) -> None: + if tag == "title": + self._in_title = False + elif tag in self._HIDDEN and self._hidden: + self._hidden -= 1 + elif tag in self._BLOCKS: + self.pieces.append("\n") + + def handle_data(self, data: str) -> None: + if self._in_title: + self.title += data + elif not self._hidden: + self.pieces.append(data) + + +def _html_text(html: str) -> tuple[str, str]: + parser = _TextOfHtml() + parser.feed(html) + parser.close() + return " ".join(parser.title.split()), _joined("".join(parser.pieces).split("\n")) + + +def _html(path: Path) -> tuple[str | None, str, datetime | None]: + title, text = _html_text(_text(path)) + return title or None, text, None + + +def _rtf(path: Path) -> tuple[str | None, str, datetime | None]: + text = _text(path) + text = re.sub(r"\\par[d]?\b", "\n", text) + text = re.sub(r"\{\\\*[^{}]*\}", "", text) + text = re.sub(r"\\'[0-9a-fA-F]{2}", "", text) + text = re.sub(r"\\[a-zA-Z]+-?\d* ?", "", text) + text = text.replace("{", "").replace("}", "") + return None, _joined(text.split("\n")), None + + +# -- office files, EPUB, and PDF --------------------------------------------------- + +_W = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}" +_A = "{http://schemas.openxmlformats.org/drawingml/2006/main}" +_S = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}" +_REL = "{http://schemas.openxmlformats.org/officeDocument/2006/relationships}" +_DC = "{http://purl.org/dc/elements/1.1/}" +_TERMS = "{http://purl.org/dc/terms/}" +_TEXT_NS = "{urn:oasis:names:tc:opendocument:xmlns:text:1.0}" +_TABLE_NS = "{urn:oasis:names:tc:opendocument:xmlns:table:1.0}" + + +def _core(archive: zipfile.ZipFile) -> tuple[str | None, datetime | None]: + """Title and creation date from an Office file's core properties.""" + if "docProps/core.xml" not in archive.namelist(): + return None, None + root = ElementTree.fromstring(archive.read("docProps/core.xml")) + title = (root.findtext(f"{_DC}title") or "").strip() or None + created = _iso((root.findtext(f"{_TERMS}created") or "").strip()) + return title, created + + +def _iso(value: str) -> datetime | None: + if not value: + return None + try: + parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) + except ValueError: + return None + return parsed if parsed.tzinfo else parsed.replace(tzinfo=UTC) + + +def _docx(path: Path) -> tuple[str | None, str, datetime | None]: + with zipfile.ZipFile(path) as archive: + root = ElementTree.fromstring(archive.read("word/document.xml")) + title, created = _core(archive) + paragraphs = [ + "".join(node.text or "" for node in p.iter(f"{_W}t")) for p in root.iter(f"{_W}p") + ] + return title, _joined(paragraphs), created + + +def _pptx(path: Path) -> tuple[str | None, str, datetime | None]: + with zipfile.ZipFile(path) as archive: + slides = sorted( + ( + name + for name in archive.namelist() + if re.fullmatch(r"ppt/slides/slide\d+\.xml", name) + ), + key=lambda name: int(re.findall(r"\d+", name)[-1]), + ) + pages: list[str] = [] + for number, name in enumerate(slides, start=1): + root = ElementTree.fromstring(archive.read(name)) + lines = [ + "".join(node.text or "" for node in p.iter(f"{_A}t")) for p in root.iter(f"{_A}p") + ] + body = _joined(lines) + if body: + pages.append(f"Slide {number}\n{body}") + title, created = _core(archive) + return title, "\n\n".join(pages), created + + +def _xlsx(path: Path) -> tuple[str | None, str, datetime | None]: + with zipfile.ZipFile(path) as archive: + names = set(archive.namelist()) + shared: list[str] = [] + if "xl/sharedStrings.xml" in names: + root = ElementTree.fromstring(archive.read("xl/sharedStrings.xml")) + shared = [ + "".join(t.text or "" for t in item.iter(f"{_S}t")) for item in root.iter(f"{_S}si") + ] + workbook = ElementTree.fromstring(archive.read("xl/workbook.xml")) + relations = ElementTree.fromstring(archive.read("xl/_rels/workbook.xml.rels")) + targets = { + rel.get("Id"): rel.get("Target", "") + for rel in relations + if rel.get("Target", "").startswith(("worksheets/", "/xl/worksheets/")) + } + sheets: list[str] = [] + for sheet in workbook.iter(f"{_S}sheet"): + target = targets.get(sheet.get(f"{_REL}id")) + if not target: + continue + member = target.lstrip("/") if target.startswith("/") else f"xl/{target}" + if member not in names: + continue + rows = _sheet_rows(ElementTree.fromstring(archive.read(member)), shared) + if rows: + sheets.append(f"{sheet.get('name', 'Sheet')}\n" + "\n".join(rows)) + title, created = _core(archive) + return title, "\n\n".join(sheets), created + + +def _sheet_rows(root: ElementTree.Element, shared: list[str]) -> list[str]: + rows: list[list[str]] = [] + for row in root.iter(f"{_S}row"): + cells: list[str] = [] + for cell in row.iter(f"{_S}c"): + kind = cell.get("t") + if kind == "inlineStr": + value = "".join(t.text or "" for t in cell.iter(f"{_S}t")) + else: + value = cell.findtext(f"{_S}v") or "" + if kind == "s" and value.isdigit() and int(value) < len(shared): + value = shared[int(value)] + cells.append(value.strip()) + if any(cells): + rows.append(cells) + if not rows: + return [] + header, *body = rows + if not body: + return [" | ".join(cell for cell in header if cell)] + return [ + " | ".join( + f"{name}: {value}" if name else value + for name, value in zip(header, row, strict=False) + if value + ) + for row in body + ] + + +def _opendocument(path: Path) -> tuple[str | None, str, datetime | None]: + with zipfile.ZipFile(path) as archive: + root = ElementTree.fromstring(archive.read("content.xml")) + title = None + if "meta.xml" in archive.namelist(): + meta = ElementTree.fromstring(archive.read("meta.xml")) + node = next(meta.iter(f"{_DC}title"), None) + title = (node.text or "").strip() or None if node is not None else None + lines: list[str] = [] + for node in root.iter(): + if node.tag in (f"{_TEXT_NS}p", f"{_TEXT_NS}h"): + lines.append("".join(node.itertext())) + elif node.tag == f"{_TABLE_NS}table-row": + lines.append( + " | ".join( + "".join(cell.itertext()) for cell in node if "".join(cell.itertext()).strip() + ) + ) + # Table cells hold paragraphs too; keep each line once, in order. + return title, _joined(list(dict.fromkeys(line for line in lines if line.strip()))), None + + +def _epub(path: Path) -> tuple[str | None, str, datetime | None]: + with zipfile.ZipFile(path) as archive: + container = ElementTree.fromstring(archive.read("META-INF/container.xml")) + rootfile = next(node for node in container.iter() if node.tag.endswith("rootfile")) + package_path = rootfile.get("full-path", "") + package = ElementTree.fromstring(archive.read(package_path)) + base = str(PurePosixPath(package_path).parent) + title = next((node.text for node in package.iter() if node.tag == f"{_DC}title"), None) + items = { + node.get("id"): node.get("href", "") + for node in package.iter() + if node.tag.endswith("item") + } + chapters: list[str] = [] + for node in package.iter(): + if not node.tag.endswith("itemref"): + continue + href = items.get(node.get("idref")) + if not href: + continue + member = str(PurePosixPath(base) / href) if base not in ("", ".") else href + if member in archive.namelist(): + chapters.append(_html_text(_decode(archive.read(member)))[1]) + return title, "\n\n".join(chapter for chapter in chapters if chapter), None + + +def _pdf(path: Path) -> tuple[str | None, str, datetime | None]: + from pypdf import PdfReader # loaded only when a PDF is imported + from pypdf.errors import PdfReadError + + try: + reader = PdfReader(path) + if reader.is_encrypted and not reader.decrypt(""): + raise DocumentError("it is protected by a password") + pages = [page.extract_text() or "" for page in reader.pages] + except PdfReadError as error: + raise DocumentError(f"it is not a readable PDF: {error}") from error + meta = reader.metadata + title = (meta.title or "").strip() if meta is not None else "" + created = meta.creation_date if meta is not None else None + if created is not None and created.tzinfo is None: + created = created.replace(tzinfo=UTC) + text = "\n\n".join(page.strip() for page in pages if page.strip()) + return title or None, text, created + + +_READERS: dict[str, Callable[[Path], tuple[str | None, str, datetime | None]]] = { + ".csv": _table, + ".tsv": _table, + ".json": _json, + ".jsonl": _json, + ".ndjson": _json, + ".xml": _xml, + ".html": _html, + ".htm": _html, + ".xhtml": _html, + ".rtf": _rtf, + ".docx": _docx, + ".pptx": _pptx, + ".xlsx": _xlsx, + ".odt": _opendocument, + ".odp": _opendocument, + ".ods": _opendocument, + ".epub": _epub, + ".pdf": _pdf, +} diff --git a/src/chatlore/importers/intake.py b/src/chatlore/importers/intake.py new file mode 100644 index 0000000..ff6610d --- /dev/null +++ b/src/chatlore/importers/intake.py @@ -0,0 +1,377 @@ +"""Importing whatever is dropped: files, folders, and archives, holding anything. + +``Intake`` walks every path it is given. Archives (zip, tar, gz, and, with the +system's bsdtar, 7z and rar) are unpacked, archives inside them too, up to a few +levels deep. Chat exports are recognised by their content wherever they sit and +go to their importers: a ChatGPT or Claude ``conversations.json``, a Gemini +``MyActivity.json``, and ChatLore archives. Every other file ChatLore can read +becomes a note (``chatlore.importers.documents``). The rest is skipped, and each +skipped file is listed with the reason, so nothing disappears silently. + +Unpacking is limited in total size and file count and never writes outside its +folder, so a hostile archive can neither fill the disk nor overwrite anything. +""" + +from __future__ import annotations + +import gzip +import os +import platform +import re +import shutil +import subprocess +import tarfile +import zipfile +from collections import Counter +from collections.abc import Iterator, Sequence +from dataclasses import dataclass, field +from functools import cache +from pathlib import Path + +from chatlore.archive import is_archive +from chatlore.importers import IMPORTERS, ImportIssue, IssueSink, detect_source +from chatlore.importers.base import ImporterError, load_json, report +from chatlore.importers.documents import MEDIA, DocumentError, read_document, readable +from chatlore.importers.gemini import _is_gemini +from chatlore.models import Conversation, SourceKind + +MAX_UNPACKED = 2_000_000_000 +"""Bytes all archives together may unpack to.""" +MAX_FILES = 100_000 +"""Files one import may hold.""" +MAX_DEPTH = 4 +"""How deep archives may sit inside archives.""" + +ARCHIVES = ( + ".tar.gz", + ".tar.bz2", + ".tar.xz", + ".tgz", + ".tbz2", + ".txz", + ".tar", + ".zip", + ".7z", + ".rar", + ".gz", +) +_SKIPPED_DIRS = frozenset( + {".git", ".svn", ".hg", "node_modules", "__pycache__", ".venv", "venv", ".obsidian", + ".trash", ".Trash", "__MACOSX", ".idea", ".vscode"} +) # fmt: skip +_SKIPPED_FILES = frozenset({".ds_store", "thumbs.db", "desktop.ini", "archive_browser.html"}) +_EXPORT_FILES = {"conversations.json", "myactivity.json"} + + +@dataclass(frozen=True, slots=True) +class Skipped: + """A file that was not imported, and why.""" + + path: str + reason: str + + +@dataclass(frozen=True, slots=True) +class _Export: + kind: SourceKind + path: Path + shown: str + + +@dataclass(slots=True) +class Intake: + """What is in the paths given, found by ``scan`` and read by ``conversations``. + + ``workdir`` is an empty folder for unpacked archives; the caller removes it. + """ + + workdir: Path + max_unpacked: int = MAX_UNPACKED + max_files: int = MAX_FILES + exports: list[_Export] = field(default_factory=list) + archives: list[tuple[Path, str]] = field(default_factory=list) + documents: list[tuple[Path, str]] = field(default_factory=list) + skipped: list[Skipped] = field(default_factory=list) + sources: Counter[str] = field(default_factory=Counter) + _unpacked: int = 0 + _files: int = 0 + _folders: int = 0 + + # -- finding what is there --------------------------------------------------- + + def scan(self, paths: Sequence[Path]) -> None: + """Find everything importable under ``paths``, unpacking archives on the way.""" + for path in paths: + if not path.exists(): + raise ImporterError(f"{path} does not exist") + # Files keep their path within the folder given, as the Markdown importer + # always named them, so importing a folder again updates the same notes. + if path.is_dir(): + self._folder(path, "", 0) + else: + self._file(path, path.name, 0) + + def _folder(self, root: Path, shown: str, depth: int) -> None: + for folder, dirnames, filenames in os.walk(root): + dirnames[:] = sorted( + name for name in dirnames if name not in _SKIPPED_DIRS and not name.startswith(".") + ) + here = Path(folder) + prefix = _join(shown, here.relative_to(root).as_posix()) + claimed = self._exports_in(here, filenames, prefix) + for name in sorted(filenames): + if name.lower() in _EXPORT_FILES: + continue + if claimed is not None and Path(name).suffix.lower() in {".json", ".html"}: + self.skipped.append( + Skipped(_join(prefix, name), f"part of the {claimed} export") + ) + continue + self._file(here / name, _join(prefix, name), depth) + + def _exports_in(self, folder: Path, filenames: list[str], prefix: str) -> str | None: + """Take the chat exports in one folder. Returns the export's name when one is there.""" + claimed: str | None = None + for name in filenames: + lower = name.lower() + if lower not in _EXPORT_FILES: + continue + path, shown = folder / name, _join(prefix, name) + if not self._count(): + return claimed + if lower == "myactivity.json": + if _has_gemini(path): + self.exports.append(_Export(SourceKind.GEMINI, path, shown)) + claimed = "Gemini" + else: + self.skipped.append(Skipped(shown, "activity for another Google product")) + continue + try: + kind = detect_source(path) + except ImporterError: + self.documents.append((path, shown)) # a conversations.json of some other kind + continue + self.exports.append(_Export(kind, path, shown)) + claimed = "ChatGPT" if kind is SourceKind.CHATGPT else kind.value.title() + return claimed + + def _file(self, path: Path, shown: str, depth: int) -> None: + lower = path.name.lower() + if lower in _SKIPPED_FILES or lower.startswith("._"): + return + if not self._count(): + return + if lower in _EXPORT_FILES: + self._exports_in( + path.parent, [path.name], shown.rsplit("/", 1)[0] if "/" in shown else "" + ) + return + if lower.endswith(ARCHIVES) and not lower.endswith((".docx", ".xlsx", ".pptx")): + self._archive(path, shown, depth) + return + if path.suffix.lower() in MEDIA: + self.skipped.append(Skipped(shown, "images, audio, video, and programs are not read")) + return + if not readable(path): + self.skipped.append(Skipped(shown, "not a kind of file ChatLore reads")) + return + self.documents.append((path, shown)) + + def _archive(self, path: Path, shown: str, depth: int) -> None: + if lower_is_zip(path) and is_archive(path): + self.archives.append((path, shown)) + return + if depth >= MAX_DEPTH: + self.skipped.append(Skipped(shown, "archives nested too deep")) + return + self._folders += 1 + target = self.workdir / f"{self._folders:05d}" + target.mkdir(parents=True) + try: + self._unpack(path, target) + except UnpackError as error: + self.skipped.append(Skipped(shown, str(error))) + return + if path.name.lower().endswith(".gz") and not path.name.lower().endswith(".tar.gz"): + for inner in sorted(target.iterdir()): + self._file( + inner, + _join(shown.rsplit("/", 1)[0] if "/" in shown else "", inner.name), + depth + 1, + ) + return + self._folder(target, shown, depth + 1) + + def _unpack(self, path: Path, target: Path) -> None: + lower = path.name.lower() + if zipfile.is_zipfile(path): + with zipfile.ZipFile(path) as archive: + members = [member for member in archive.infolist() if not member.is_dir()] + self._reserve(sum(member.file_size for member in members), len(members)) + for member in members: + archive.extract(member, target) # zipfile keeps members inside the target + return + if lower.endswith((".tar", ".tgz", ".tbz2", ".txz", ".tar.gz", ".tar.bz2", ".tar.xz")): + try: + with tarfile.open(path) as bundle: + entries = [entry for entry in bundle.getmembers() if entry.isfile()] + self._reserve(sum(entry.size for entry in entries), len(entries)) + bundle.extractall(target, members=entries, filter="data") + except (tarfile.TarError, OSError) as error: + raise UnpackError(f"the archive could not be unpacked: {error}") from error + return + if lower.endswith(".gz"): + name = path.name[: -len(".gz")] or "unpacked" + with gzip.open(path, "rb") as source, (target / name).open("wb") as out: + while piece := source.read(1 << 20): + self._reserve(len(piece), 0) + out.write(piece) + self._reserve(0, 1) + return + self._bsdtar(path, target) + + def _bsdtar(self, path: Path, target: Path) -> None: + tool = bsdtar() + if tool is None: + raise UnpackError( + f"{path.suffix} archives need bsdtar, which is not installed here " + "(it comes with Windows and macOS; on Linux install libarchive-tools)" + ) + listing = subprocess.run( + [tool, "-tvf", str(path)], capture_output=True, text=True, errors="replace", check=False + ) + if listing.returncode != 0: + raise UnpackError(f"the archive could not be read: {listing.stderr.strip()[:200]}") + sizes = [_listed_size(line) for line in listing.stdout.splitlines() if line.startswith("-")] + self._reserve(sum(sizes), len(sizes)) + # bsdtar refuses absolute paths and ".." by default, so nothing lands outside target. + unpacked = subprocess.run( + [tool, "-xf", str(path), "-C", str(target)], + capture_output=True, + text=True, + errors="replace", + check=False, + ) + if unpacked.returncode != 0: + raise UnpackError(f"the archive could not be unpacked: {unpacked.stderr.strip()[:200]}") + + def _reserve(self, size: int, files: int) -> None: + self._unpacked += size + self._files += files + if self._unpacked > self.max_unpacked: + raise ImporterError( + f"the archives unpack to more than {self.max_unpacked // 1_000_000_000} GB" + ) + if self._files > self.max_files: + raise ImporterError(f"more than {self.max_files:,} files in one import") + + def _count(self) -> bool: + self._files += 1 + if self._files > self.max_files: + raise ImporterError(f"more than {self.max_files:,} files in one import") + return True + + # -- reading it ---------------------------------------------------------------- + + def conversations(self, on_issue: IssueSink | None = None) -> Iterator[Conversation]: + """Every conversation found, chat exports first, then documents as notes. + + ChatLore archives are listed in ``archives`` and imported by the caller, + since they bring a graph and caches along. ``sources`` counts what each + source yielded. + """ + for export in self.exports: + try: + for conversation in IMPORTERS[export.kind].parse(export.path, on_issue): + self.sources[conversation.source.value] += 1 + yield conversation + except ImporterError as error: + self.skipped.append(Skipped(export.shown, str(error))) + for path, shown in self.documents: + try: + found = read_document(path, shown) + except DocumentError as error: + self.skipped.append(Skipped(shown, str(error))) + continue + for conversation in found: + self.sources[conversation.source.value] += 1 + yield conversation + + def found_anything(self) -> bool: + return bool(self.exports or self.archives or self.documents) + + +class UnpackError(Exception): + """One archive cannot be unpacked; the import goes on without it.""" + + +def lower_is_zip(path: Path) -> bool: + return path.name.lower().endswith(".zip") and zipfile.is_zipfile(path) + + +def _join(prefix: str, name: str) -> str: + parts = [part for part in (prefix, name) if part and part != "."] + return "/".join(parts) + + +def _has_gemini(path: Path) -> bool: + try: + data = load_json(str(path), path.read_bytes()) + except ImporterError: + return False + return isinstance(data, list) and any(_is_gemini(entry) for entry in data) + + +def _listed_size(line: str) -> int: + fields = line.split() + return int(fields[4]) if len(fields) > 4 and fields[4].isdigit() else 0 + + +@cache +def bsdtar() -> str | None: + """A bsdtar on this machine, which unpacks 7z and rar as well as the rest.""" + candidates = [shutil.which("bsdtar")] + if platform.system() == "Windows": + candidates.append( + str(Path(os.environ.get("SYSTEMROOT", r"C:\Windows")) / "System32" / "tar.exe") + ) + candidates += ["/usr/bin/tar", shutil.which("tar")] + for candidate in candidates: + if not candidate or not Path(candidate).exists(): + continue + try: + version = subprocess.run( + [candidate, "--version"], capture_output=True, text=True, timeout=10, check=False + ).stdout + except (OSError, subprocess.SubprocessError): + continue + if "bsdtar" in version: + return candidate + return None + + +def describe(skipped: Sequence[Skipped], limit: int = 10) -> list[str]: + """Skipped files for people to read, with identical reasons grouped.""" + grouped: dict[str, list[str]] = {} + for item in skipped: + grouped.setdefault(re.sub(r"\d+", "N", item.reason), []).append(item.path) + lines = [] + for reason, paths in list(grouped.items())[:limit]: + lines.append( + f"{paths[0]}: {reason}" if len(paths) == 1 else f"{len(paths)} files: {reason}" + ) + return lines + + +__all__ = [ + "ARCHIVES", + "MAX_FILES", + "MAX_UNPACKED", + "ImportIssue", + "Intake", + "Skipped", + "UnpackError", + "bsdtar", + "describe", + "report", +] diff --git a/src/chatlore/imports.py b/src/chatlore/imports.py index 81c9f12..a0e2af0 100644 --- a/src/chatlore/imports.py +++ b/src/chatlore/imports.py @@ -11,23 +11,24 @@ from __future__ import annotations import logging +import re +import shutil import tempfile import threading -import zipfile from collections.abc import Callable from concurrent.futures import ThreadPoolExecutor from dataclasses import asdict, dataclass, field from datetime import UTC, datetime -from pathlib import Path, PurePosixPath +from pathlib import Path from typing import Any, Protocol -from chatlore.archive import ArchiveError, import_archive, is_archive +from chatlore.archive import ArchiveError, import_archive from chatlore.embeddings import Embedder, EmbeddingCache, EmbeddingError from chatlore.extraction import ExtractionCache -from chatlore.importers import ImporterError, detect_source, get_importer +from chatlore.importers import ImporterError +from chatlore.importers.intake import Intake from chatlore.library import AddOutcome, Library from chatlore.llm import LLM, LLMError -from chatlore.models import SourceKind from chatlore.pipeline import ( EmbeddingModelMismatchError, pending_duplicates, @@ -46,12 +47,6 @@ logger = logging.getLogger(__name__) -SUFFIXES = frozenset({".zip", ".json", ".md", ".markdown", ".txt"}) -"""File types an upload may have; they tell the importers what they are reading.""" -MAX_UNPACKED = 2_000_000_000 -"""Bytes of text a zip may unpack to, so a small zip cannot fill the disk.""" -_TEXT = (".json", ".jsonl", ".md", ".markdown", ".txt") - class _ClosableLLM(LLM, Protocol): def close(self) -> None: ... @@ -77,6 +72,12 @@ class ImportStatus: conversations: int = 0 messages: int = 0 skipped: int = 0 + """Records the importers could not read, inside files they did read.""" + sources: dict[str, int] = field(default_factory=dict) + """Conversations found, by where they came from: chatgpt, document, email, ...""" + skipped_files: int = 0 + skipped_shown: list[list[str]] = field(default_factory=list) + """The first skipped files, each with the reason.""" notes: list[str] = field(default_factory=list) error: str | None = None started_at: str = field(default_factory=lambda: datetime.now(UTC).isoformat()) @@ -119,7 +120,10 @@ def running(self, home: Path) -> bool: def start( self, home: Path, upload: Path, name: str, store: Callable[[], GraphStore] ) -> ImportStatus: - """Queue ``upload`` for the library at ``home``, whose graph ``store`` opens.""" + """Queue the files in the folder ``upload`` for the library at ``home``. + + ``store`` opens the library's graph. The folder is deleted when the import ends. + """ status = ImportStatus(file=name) with self._lock: self._status[home] = status @@ -191,7 +195,7 @@ def advance(count: int) -> None: finally: if graph is not None: graph.close() - upload.unlink(missing_ok=True) + shutil.rmtree(upload, ignore_errors=True) status.finished_at = datetime.now(UTC).isoformat() with self._lock: self._cancelled.discard(home) @@ -254,70 +258,58 @@ def _build_graph( llm.close() -def upload_suffix(name: str) -> str: - """The suffix to save an upload with, taken from its name when it is one ChatLore reads.""" - suffix = PurePosixPath(name.replace("\\", "/")).suffix.lower() - return suffix if suffix in SUFFIXES else "" +_UNSAFE = re.compile(r'[<>:"|?*\x00-\x1f]') +_MAX_PART = 120 +_MAX_PARTS = 24 +_SKIPPED_SHOWN = 50 -def _import(upload: Path, home: Path, graph: GraphStore, status: ImportStatus) -> None: - """Add the upload's conversations to the library and its graph.""" - if zipfile.is_zipfile(upload): - _check_unpacked_size(upload) - if is_archive(upload): - report = import_archive(upload, home, store=graph) - status.conversations = report.outcomes[AddOutcome.NEW] + report.outcomes[AddOutcome.UPDATED] - status.messages = report.messages - return - with tempfile.TemporaryDirectory() as folder: - try: - path, kind = upload, detect_source(upload) - except ImporterError: - unpacked = _markdown_folder(upload, Path(folder)) - if unpacked is None: - raise - path, kind = unpacked, SourceKind.MARKDOWN +def safe_relative(name: str) -> str: + """An upload's path within its batch, made safe: no absolute paths, no ``..``. + + Browsers send a file's path within the folder it was chosen from, such as + ``Export/conversations.json``; that path is kept, since it tells the + importers what they are reading. + """ + parts: list[str] = [] + for part in name.replace("\\", "/").split("/"): + part = _UNSAFE.sub("_", part).strip().rstrip(". ") + if not part or part in {".", ".."}: + continue + parts.append(part[:_MAX_PART]) + return "/".join(parts[-_MAX_PARTS:]) or "upload" + + +def _import(folder: Path, home: Path, graph: GraphStore, status: ImportStatus) -> None: + """Add everything the uploaded files hold to the library and its graph.""" + with tempfile.TemporaryDirectory(prefix="chatlore-unpacked-") as workdir: + intake = Intake(Path(workdir)) issues: list[object] = [] - with ( - Library(home) as library, - graph.transaction(), - ConversationWriter(graph) as writer, - ): - for conversation in get_importer(kind.value).parse(path, issues.append): - status.messages += len(conversation.messages) - if library.add(conversation) is not AddOutcome.UNCHANGED: - status.conversations += 1 - writer.add(conversation) - status.skipped = len(issues) - if status.conversations == 0 and status.messages == 0: - raise ImporterError("the upload holds no conversations ChatLore can read") - - -def _check_unpacked_size(upload: Path) -> None: - with zipfile.ZipFile(upload) as archive: - size = sum( - member.file_size - for member in archive.infolist() - if member.filename.lower().endswith(_TEXT) - ) - if size > MAX_UNPACKED: - raise ImporterError("the zip unpacks to more text than this server accepts") - - -def _markdown_folder(upload: Path, folder: Path) -> Path | None: - """Unpack a zip of Markdown or text notes, as a vault is shared; None for any other zip.""" - if not zipfile.is_zipfile(upload): - return None - with zipfile.ZipFile(upload) as archive: - notes = [ - member - for member in archive.infolist() - if not member.is_dir() - and member.filename.lower().endswith((".md", ".markdown", ".txt")) - ] - if not notes: - return None - target = folder / "notes" - for member in notes: - archive.extract(member, target) # zipfile keeps members inside the target - return target + try: + intake.scan([folder]) + with ( + Library(home) as library, + graph.transaction(), + ConversationWriter(graph) as writer, + ): + for conversation in intake.conversations(issues.append): + status.messages += len(conversation.messages) + if library.add(conversation) is not AddOutcome.UNCHANGED: + status.conversations += 1 + writer.add(conversation) + for archive, _ in intake.archives: + report = import_archive(archive, home, store=graph) + status.conversations += ( + report.outcomes[AddOutcome.NEW] + report.outcomes[AddOutcome.UPDATED] + ) + status.messages += report.messages + intake.sources["archive"] += sum(report.outcomes.values()) + finally: + status.sources = {kind: count for kind, count in intake.sources.items() if count} + status.skipped = len(issues) + status.skipped_files = len(intake.skipped) + status.skipped_shown = [ + [item.path, item.reason] for item in intake.skipped[:_SKIPPED_SHOWN] + ] + if not intake.found_anything(): + raise ImporterError("nothing ChatLore can read was found in the upload") diff --git a/src/chatlore/models.py b/src/chatlore/models.py index 92d08c4..cea6c36 100644 --- a/src/chatlore/models.py +++ b/src/chatlore/models.py @@ -24,6 +24,8 @@ class SourceKind(StrEnum): CLAUDE = "claude" GEMINI = "gemini" MARKDOWN = "markdown" + DOCUMENT = "document" + EMAIL = "email" NOTE = "note" CHATLORE = "chatlore" CLAUDE_CODE = "claude_code" diff --git a/src/chatlore/web/app.js b/src/chatlore/web/app.js index f1211bd..6c86e32 100644 --- a/src/chatlore/web/app.js +++ b/src/chatlore/web/app.js @@ -213,7 +213,18 @@ const Data = { const summary = []; if (status.file) summary.push(status.file); if (status.conversations) summary.push(`${number(status.conversations)} conversations added`); + const sources = Object.entries(status.sources || {}); + if (sources.length) summary.push(sources.map(([source, count]) => `${number(count)} from ${source}`).join(", ")); if (status.skipped) summary.push(`${number(status.skipped)} records skipped`); + const skipped = $("#data-skipped"); + skipped.hidden = !status.skipped_files; + if (status.skipped_files) { + $("summary", skipped).textContent = `${number(status.skipped_files)} ${status.skipped_files === 1 ? "file" : "files"} skipped`; + const shown = status.skipped_shown || []; + $("ul", skipped).innerHTML = + shown.map(([path, reason]) => `<li>${esc(path)}: ${esc(reason)}</li>`).join("") + + (status.skipped_files > shown.length ? `<li>and ${number(status.skipped_files - shown.length)} more</li>` : ""); + } $("#data-summary").innerHTML = esc(summary.join(" · ")) + (status.error ? `<br><span class="error">${esc(status.error)}</span>` : ""); const notes = [...(status.notes || [])]; $("#data-notes").innerHTML = notes.map((note) => `<li>${esc(note)}</li>`).join(""); @@ -240,22 +251,59 @@ const Data = { }, 1500); }, - upload(file) { - if (!file) return; - if (file.size > this.info.max_upload_mb * 1_000_000) { - $("#data-progress").hidden = false; - this.show({ file: file.name, stage: "failed", running: false, error: `The file is larger than ${this.info.max_upload_mb} MB.` }); + /** Upload files, each ``{file, path}`` with its path in the folder it came from, then import them. */ + async upload(files) { + files = files.filter(({ path }) => !path.split("/").some((part) => SKIPPED_FOLDERS.has(part))); + if (!files.length) return; + $("#data-progress").hidden = false; + const fail = (error) => this.show({ file: label(files), stage: "failed", running: false, error }); + const total = files.reduce((sum, { file }) => sum + file.size, 0); + if (total > this.info.max_upload_mb * 1_000_000) { + fail(`That is ${number(Math.ceil(total / 1_000_000))} MB; this server takes up to ${number(this.info.max_upload_mb)} MB at a time.`); return; } - $("#data-progress").hidden = false; + const batch = Array.from(crypto.getRandomValues(new Uint8Array(16)), (byte) => byte.toString(16).padStart(2, "0")).join(""); + let sent = 0; + try { + for (const [index, item] of files.entries()) { + const shown = files.length > 1 ? `${index + 1} of ${number(files.length)}: ${item.path}` : item.path; + await send(`library/files?batch=${batch}`, item.file, item.path, (loaded) => + this.show({ file: shown, running: true }, total ? (sent + loaded) / total : 1), + ); + sent += item.file.size; + } + this.show(await send(`library/import?batch=${batch}`, null, label(files))); + this.poll(); + } catch (error) { + fail(error.message); + } + }, + + async remove() { + if (!confirm("Delete your library from this server now? This cannot be undone.")) return; + const response = await fetch("library", { method: "DELETE", headers: { "X-ChatLore": "1" } }); + if (response.ok) location.reload(); + }, +}; + +$("#data-open").addEventListener("click", () => { + Data.refresh().catch(() => {}); + $("#data").showModal(); +}); +// Folders nobody means to import, left out before anything is uploaded. +const SKIPPED_FOLDERS = new Set([".git", "node_modules", "__MACOSX", ".venv", "__pycache__", ".Trash"]); + +const label = (files) => (files.length === 1 ? files[0].path : `${number(files.length)} files`); + +/** POST a file (or nothing) with the interface's header, reporting upload progress; the answer's JSON. */ +function send(url, file, path, progress) { + return new Promise((resolve, reject) => { const request = new XMLHttpRequest(); - request.open("POST", "library/import"); + request.open("POST", url); request.setRequestHeader("X-ChatLore", "1"); - request.setRequestHeader("X-Filename", encodeURIComponent(file.name)); + request.setRequestHeader("X-Filename", encodeURIComponent(path)); request.setRequestHeader("Content-Type", "application/octet-stream"); - request.upload.onprogress = (event) => { - if (event.lengthComputable) this.show({ file: file.name, running: true }, event.loaded / event.total); - }; + if (progress) request.upload.onprogress = (event) => event.lengthComputable && progress(event.loaded); request.onload = () => { let body = {}; try { @@ -263,29 +311,46 @@ const Data = { } catch { // an empty or plain-text answer } - if (request.status === 202) { - this.show(body); - this.poll(); - } else { - this.show({ file: file.name, stage: "failed", running: false, error: body.detail || `The server answered ${request.status}.` }); - } + if (request.status >= 200 && request.status < 300) resolve(body); + else reject(new Error(body.detail || `The server answered ${request.status}.`)); }; - request.onerror = () => this.show({ file: file.name, stage: "failed", running: false, error: "The upload did not reach the server." }); + request.onerror = () => reject(new Error("The upload did not reach the server.")); request.send(file); - }, + }); +} - async remove() { - if (!confirm("Delete your library from this server now? This cannot be undone.")) return; - const response = await fetch("library", { method: "DELETE", headers: { "X-ChatLore": "1" } }); - if (response.ok) location.reload(); - }, -}; +/** The files of a drop, folders walked, each with its path; read before the drop event ends. */ +async function dropped(transfer) { + const entries = [...transfer.items].map((item) => item.webkitGetAsEntry?.()).filter(Boolean); + if (!entries.length) return [...transfer.files].map((file) => ({ file, path: file.name })); + const found = []; + const walk = async (entry, prefix) => { + if (entry.isFile) { + found.push({ file: await new Promise((done, failed) => entry.file(done, failed)), path: prefix + entry.name }); + } else if (entry.isDirectory) { + const reader = entry.createReader(); + for (;;) { + const children = await new Promise((done, failed) => reader.readEntries(done, failed)); + if (!children.length) break; + for (const child of children) await walk(child, `${prefix}${entry.name}/`); + } + } + }; + for (const entry of entries) await walk(entry, ""); + return found; +} -$("#data-open").addEventListener("click", () => { - Data.refresh().catch(() => {}); - $("#data").showModal(); +const chosen = (input) => [...input.files].map((file) => ({ file, path: file.webkitRelativePath || file.name })); + +$("#data-file").addEventListener("change", (event) => { + Data.upload(chosen(event.target)); + event.target.value = ""; +}); +$("#data-folder-pick").addEventListener("click", () => $("#data-folder").click()); +$("#data-folder").addEventListener("change", (event) => { + Data.upload(chosen(event.target)); + event.target.value = ""; }); -$("#data-file").addEventListener("change", (event) => Data.upload(event.target.files[0])); $("#data-delete").addEventListener("click", () => Data.remove()); const drop = $("#data-drop"); drop.addEventListener("dragover", (event) => { @@ -296,7 +361,7 @@ drop.addEventListener("dragleave", () => drop.classList.remove("over")); drop.addEventListener("drop", (event) => { event.preventDefault(); drop.classList.remove("over"); - Data.upload(event.dataTransfer.files[0]); + dropped(event.dataTransfer).then((files) => Data.upload(files)); }); // -- conversations --------------------------------------------------------------------- diff --git a/src/chatlore/web/index.html b/src/chatlore/web/index.html index 15281ec..7b83e11 100644 --- a/src/chatlore/web/index.html +++ b/src/chatlore/web/index.html @@ -198,18 +198,27 @@ <h2>Your data</h2> </header> <div class="sheet-body"> <label id="data-drop" class="dropzone"> - <input id="data-file" type="file" accept=".zip,.json,.md,.markdown,.txt"> + <input id="data-file" type="file" multiple> <svg viewBox="0 0 24 24"><path d="M12 15V4M7 9l5-5 5 5"/><path d="M4 15v4a1 1 0 0 0 1 1h14a1 1 0 0 0 1-1v-4"/></svg> - <strong>Drop an export here, or choose a file</strong> - <span>ChatGPT or Claude export (.zip), Gemini Takeout (.zip or MyActivity.json), - Markdown notes (.zip or .md), or a ChatLore archive. Up to <span id="data-max">200</span> MB.</span> + <strong>Drop files or folders here, or choose files</strong> + <span>Chat exports from ChatGPT, Claude, and Gemini, documents (PDF, Word, PowerPoint, + Excel, EPUB, web pages), notes and text, email, CSV and JSON, and zip, tar, 7z, or rar + archives of any of them. Up to <span id="data-max">200</span> MB in all.</span> </label> + <p class="data-alt"> + <button type="button" id="data-folder-pick" class="link">Choose a whole folder</button> + <input id="data-folder" type="file" webkitdirectory multiple hidden> + </p> <p id="data-privacy" class="data-note" hidden></p> <div id="data-progress" class="data-progress" hidden> <div class="data-stage"><span id="data-stage"></span><span id="data-count" class="muted"></span></div> <progress id="data-bar" max="1" value="0"></progress> <p id="data-summary" class="muted"></p> <ul id="data-notes"></ul> + <details id="data-skipped" hidden> + <summary></summary> + <ul></ul> + </details> </div> <h4>Download</h4> <p class="muted data-help">Everything in this library, to import into ChatLore anywhere, or as Markdown files to read.</p> diff --git a/src/chatlore/web/style.css b/src/chatlore/web/style.css index 2aeb55e..8adef66 100644 --- a/src/chatlore/web/style.css +++ b/src/chatlore/web/style.css @@ -406,6 +406,12 @@ button.danger:hover { border-color: var(--danger); } .data-progress p { margin: 0; font-size: 13px; } .data-progress ul { margin: 8px 0 0; font-size: 13px; } .data-help { margin: 0; font-size: 13px; } +.data-alt { margin: 8px 0 0; text-align: center; font-size: 13px; } +button.link { padding: 0; border: 0; background: none; color: var(--accent-text); text-decoration: underline; cursor: pointer; } +button.link:hover { background: none; color: var(--accent-hover); } +.data-progress details { margin-top: 8px; font-size: 13px; color: var(--muted); } +.data-progress details summary { cursor: pointer; } +.data-progress details li { overflow-wrap: anywhere; } .chat { display: flex; flex-direction: column; gap: 12px; background: var(--bg); } .bubble { max-width: 88%; padding: 10px 14px; border-radius: 16px; background: var(--surface); border: 1px solid var(--border); color: var(--text-2); overflow-wrap: anywhere; } diff --git a/tests/importers/test_documents.py b/tests/importers/test_documents.py new file mode 100644 index 0000000..eec3ec2 --- /dev/null +++ b/tests/importers/test_documents.py @@ -0,0 +1,135 @@ +"""Tests for reading documents of every kind into notes.""" + +from __future__ import annotations + +from datetime import UTC, datetime +from pathlib import Path + +import pytest + +from chatlore.importers.documents import DocumentError, read_document, readable +from chatlore.models import PartType, SourceKind +from tests import samples + + +def one(path: Path) -> tuple[str | None, str, SourceKind]: + [conversation] = read_document(path, path.name) + return conversation.title, conversation.messages[0].text, conversation.source + + +def test_office_files_become_notes_with_their_titles(tmp_path: Path) -> None: + docx = samples.docx( + tmp_path / "plan.docx", "Garden plan", ["Plant tomatoes in May.", "Water daily."] + ) + [note] = read_document(docx, "plans/plan.docx") + + assert note.source is SourceKind.DOCUMENT + assert note.title == "Garden plan" + assert note.created_at == datetime(2026, 3, 4, 10, tzinfo=UTC) + assert note.messages[0].text == "Plant tomatoes in May.\nWater daily." + assert note.metadata == {"kind": "document", "path": "plans/plan.docx", "type": "docx"} + assert one(samples.pptx(tmp_path / "talk.pptx", [["Intro", "Why"], ["Demo"]]))[1] == ( + "Slide 1\nIntro\nWhy\n\nSlide 2\nDemo" + ) + assert one(samples.xlsx(tmp_path / "b.xlsx", "Costs", [["Item", "Price"], ["Kayak", "900"]]))[ + 1 + ] == ("Costs\nItem: Kayak | Price: 900") + assert one(samples.odt(tmp_path / "l.odt", "Letter", ["Dear Sam,", "See you."])) == ( + "Letter", + "Dear Sam,\nSee you.", + SourceKind.DOCUMENT, + ) + + +def test_books_and_pdfs_are_read(tmp_path: Path) -> None: + book = one(samples.epub(tmp_path / "b.epub", "Tide Tales", ["Chapter one.", "Chapter two."])) + report = one( + samples.pdf(tmp_path / "r.pdf", "Tide report", ["High tide at 14:05", "Low at 20:10"]) + ) + + assert book[:2] == ("Tide Tales", "Chapter one.\n\nChapter two.") + assert report[:2] == ("Tide report", "High tide at 14:05\nLow at 20:10") + + +def test_email_becomes_conversations_one_per_message(tmp_path: Path) -> None: + [mail] = read_document( + samples.email(tmp_path / "trip.eml", "Kayak trip", "Meet at 8."), "trip.eml" + ) + box = read_document( + samples.mbox(tmp_path / "box.mbox", [("First", "Hello one"), ("Second", "Hello two")]), + "box.mbox", + ) + + assert mail.source is SourceKind.EMAIL + assert mail.title == "Kayak trip" + assert mail.created_at == datetime(2026, 3, 3, 9, 30, tzinfo=UTC) + assert "From: Maya <maya@example.com>" in mail.messages[0].text + assert "Meet at 8." in mail.messages[0].text + assert [conversation.title for conversation in box] == ["First", "Second"] + assert len({conversation.id for conversation in box}) == 2 + + +def test_text_and_data_files_are_read(tmp_path: Path) -> None: + (tmp_path / "people.csv").write_text("name,city\nMaya,Lisbon\nSam,,\n", encoding="utf-8") + (tmp_path / "settings.json").write_text( + '{"trip": {"to": "Porto", "days": [1, 2]}}', encoding="utf-8" + ) + (tmp_path / "page.html").write_text( + "<html><head><title>Tides" + "

High tide

At 14:05

", + encoding="utf-8", + ) + (tmp_path / "notes.rtf").write_text(r"{\rtf1\ansi {\b Bold} plain\par next}", encoding="utf-8") + (tmp_path / "old.txt").write_bytes("Caf\xe9 in Porto".encode("cp1252")) + + assert one(tmp_path / "people.csv")[1] == "name: Maya | city: Lisbon\nname: Sam" + assert one(tmp_path / "settings.json")[1] == "trip.to: Porto\ntrip.days[0]: 1\ntrip.days[1]: 2" + assert one(tmp_path / "page.html")[:2] == ("Tides", "High tide\nAt 14:05") + assert one(tmp_path / "notes.rtf")[1] == "Bold plain\nnext" + assert one(tmp_path / "old.txt")[1] == "Café in Porto" + + +def test_code_keeps_its_language_and_markdown_stays_a_note(tmp_path: Path) -> None: + (tmp_path / "tide.py").write_text("def high():\n return 14\n", encoding="utf-8") + (tmp_path / "idea.md").write_text("# Idea\n\nA graph of chats.", encoding="utf-8") + + [code] = read_document(tmp_path / "tide.py", "tide.py") + [note] = read_document(tmp_path / "idea.md", "idea.md") + + assert code.messages[0].content[0].type is PartType.CODE + assert code.messages[0].content[0].language == "python" + assert note.source is SourceKind.MARKDOWN and note.title == "Idea" + + +def test_files_are_recognised_by_name_or_by_looking(tmp_path: Path) -> None: + (tmp_path / "server.conf").write_text("port = 80\n", encoding="utf-8") + (tmp_path / "blob.dat").write_bytes(b"\x00\x01\x02binary") + (tmp_path / "photo.jpg").write_bytes(b"not really a photo") + + assert readable(tmp_path / "server.conf") + assert not readable(tmp_path / "blob.dat") + assert not readable(tmp_path / "photo.jpg") + assert readable(tmp_path / "missing.pdf") + + +def test_documents_without_text_are_refused_with_a_reason(tmp_path: Path) -> None: + (tmp_path / "empty.md").write_text(" \n", encoding="utf-8") + samples.pdf(tmp_path / "scan.pdf", "Scan", []) + (tmp_path / "broken.docx").write_bytes(b"not a zip") + + with pytest.raises(DocumentError, match="empty"): + read_document(tmp_path / "empty.md", "empty.md") + with pytest.raises(DocumentError, match="no text"): + read_document(tmp_path / "scan.pdf", "scan.pdf") + with pytest.raises(DocumentError, match="could not be read"): + read_document(tmp_path / "broken.docx", "broken.docx") + + +def test_a_documents_identity_is_its_path(tmp_path: Path) -> None: + (tmp_path / "a.csv").write_text("x\n1\n", encoding="utf-8") + + first = read_document(tmp_path / "a.csv", "data/a.csv")[0] + again = read_document(tmp_path / "a.csv", "data/a.csv")[0] + elsewhere = read_document(tmp_path / "a.csv", "other/a.csv")[0] + + assert first.id == again.id != elsewhere.id diff --git a/tests/importers/test_intake.py b/tests/importers/test_intake.py new file mode 100644 index 0000000..8c1370e --- /dev/null +++ b/tests/importers/test_intake.py @@ -0,0 +1,199 @@ +"""Tests for importing whatever is dropped: files, folders, and archives.""" + +from __future__ import annotations + +import gzip +import json +import subprocess +import tarfile +import tempfile +import zipfile +from pathlib import Path + +import pytest + +from chatlore.importers.base import ImporterError +from chatlore.importers.intake import MAX_FILES, MAX_UNPACKED, Intake, bsdtar, describe +from tests import samples +from tests.conftest import make_zip + + +def take( + tmp_path: Path, *paths: Path, max_unpacked: int = MAX_UNPACKED, max_files: int = MAX_FILES +) -> tuple[Intake, dict[str, str]]: + """Scan ``paths`` and read everything: the intake and each conversation's title by source.""" + work = Path(tempfile.mkdtemp(prefix="work", dir=tmp_path)) + intake = Intake(work, max_unpacked=max_unpacked, max_files=max_files) + intake.scan(list(paths)) + titles = { + f"{c.source.value}: {c.title}": c.metadata.get("path", "") for c in intake.conversations() + } + return intake, titles + + +def export_folder(root: Path) -> Path: + """A ChatGPT export as it unzips, with its companions and a picture.""" + root.mkdir(parents=True) + (root / "conversations.json").write_text(samples.CHATGPT, encoding="utf-8") + (root / "chat.html").write_text("the same chats again", encoding="utf-8") + (root / "user.json").write_text('{"id": "user-1"}', encoding="utf-8") + (root / "file-abc.png").write_bytes(b"\x89PNG") + return root + + +def test_a_folder_of_everything_is_sorted_out(tmp_path: Path) -> None: + drop = tmp_path / "drop" + export_folder(drop / "ChatGPT") + (drop / "notes").mkdir() + (drop / "notes" / "idea.md").write_text("# Idea\n\nA graph of chats.", encoding="utf-8") + samples.pdf(drop / "report.pdf", "Tide report", ["High tide at 14:05"]) + samples.email(drop / "trip.eml", "Kayak trip", "Meet at 8.") + (drop / ".git").mkdir() + (drop / ".git" / "config").write_text("secret", encoding="utf-8") + (drop / ".DS_Store").write_bytes(b"\x00") + + intake, titles = take(tmp_path, drop) + + assert titles == { + "chatgpt: Trip ideas": "", + "markdown: Idea": "notes/idea.md", + "document: Tide report": "report.pdf", + "email: Kayak trip": "trip.eml", + } + assert dict(intake.sources) == {"chatgpt": 1, "markdown": 1, "document": 1, "email": 1} + assert {(item.path, item.reason) for item in intake.skipped} == { + ("ChatGPT/chat.html", "part of the ChatGPT export"), + ("ChatGPT/user.json", "part of the ChatGPT export"), + ("ChatGPT/file-abc.png", "images, audio, video, and programs are not read"), + } + + +def test_archives_inside_archives_are_unpacked(tmp_path: Path) -> None: + inner = make_zip(tmp_path / "inner.zip", {"deep/plan.txt": "Plant tomatoes in May."}) + tarball = tmp_path / "notes.tar.gz" + with tarfile.open(tarball, "w:gz") as archive: + archive.add(inner, arcname="inner.zip") + outer = make_zip( + tmp_path / "outer.zip", + {"ChatGPT/conversations.json": samples.CHATGPT, "notes.tar.gz": tarball}, + ) + compressed = tmp_path / "log.txt.gz" + compressed.write_bytes(gzip.compress(b"Deploy went fine.")) + + intake, titles = take(tmp_path, outer, compressed) + + assert titles == { + "chatgpt: Trip ideas": "", + "markdown: plan": "outer.zip/notes.tar.gz/inner.zip/deep/plan.txt", + "markdown: log": "log.txt", + } + assert intake.skipped == [] + + +def test_google_takeout_brings_gemini_and_skips_other_activity( + tmp_path: Path, fixtures: Path +) -> None: + gemini = (fixtures / "gemini" / "MyActivity.json").read_text(encoding="utf-8") + search = json.dumps( + [{"header": "Search", "title": "Searched for tides", "time": "2026-01-01T00:00:00Z"}] + ) + takeout = make_zip( + tmp_path / "takeout-001.zip", + { + "Takeout/My Activity/Gemini Apps/MyActivity.json": gemini, + "Takeout/My Activity/Search/MyActivity.json": search, + "Takeout/archive_browser.html": "index", + "Takeout/Drive/Trip.docx": samples.docx( + tmp_path / "Trip.docx", "Trip", ["Porto in May."] + ), + }, + ) + + intake, titles = take(tmp_path, takeout) + + assert intake.sources["gemini"] > 0 + assert "document: Trip" in titles + assert [(item.path, item.reason) for item in intake.skipped] == [ + ( + "takeout-001.zip/Takeout/My Activity/Search/MyActivity.json", + "activity for another Google product", + ) + ] + + +def test_chatlore_archives_are_set_aside_for_the_caller(tmp_path: Path) -> None: + archive = make_zip( + tmp_path / "library.zip", + { + "manifest.json": json.dumps({"format": "chatlore-archive", "version": 1}), + "conversations.jsonl": "", + }, + ) + folder = tmp_path / "backups" + folder.mkdir() + (folder / "library.zip").write_bytes(archive.read_bytes()) + + intake, titles = take(tmp_path, folder) + + assert [shown for _, shown in intake.archives] == ["library.zip"] + assert titles == {} + assert intake.found_anything() + + +def test_unreadable_files_are_listed_and_nothing_found_is_said(tmp_path: Path) -> None: + (tmp_path / "photos").mkdir() + for name in ("a.jpg", "b.jpg", "c.mp4"): + (tmp_path / "photos" / name).write_bytes(b"\x00\x01") + (tmp_path / "photos" / "blob.bin2").write_bytes(b"\x00binary") + + intake, titles = take(tmp_path, tmp_path / "photos") + + assert titles == {} and not intake.found_anything() + assert describe(intake.skipped) == [ + "3 files: images, audio, video, and programs are not read", + "blob.bin2: not a kind of file ChatLore reads", + ] + + +def test_unpacking_is_limited_and_stays_inside_its_folder(tmp_path: Path) -> None: + sneaky = tmp_path / "sneaky.zip" + with zipfile.ZipFile(sneaky, "w") as archive: + archive.writestr("../../escaped.txt", "outside") + archive.writestr("/etc/absolute.txt", "outside") + big = make_zip(tmp_path / "big.zip", {"a.txt": "x" * 5000}) + + intake, titles = take(tmp_path, sneaky) + with pytest.raises(ImporterError, match="unpack to more than"): + take(tmp_path, big, max_unpacked=1000) + with pytest.raises(ImporterError, match="more than 2 files"): + take( + tmp_path, + make_zip(tmp_path / "many.zip", {"1.txt": "a", "2.txt": "b", "3.txt": "c"}), + max_files=2, + ) + + assert not (tmp_path / "escaped.txt").exists() and not Path("/etc/absolute.txt").exists() + assert sorted(titles) == ["markdown: absolute", "markdown: escaped"] + assert all(intake.workdir in path.parents for path, _ in intake.documents) + + +def test_a_missing_path_is_an_error(tmp_path: Path) -> None: + with pytest.raises(ImporterError, match="does not exist"): + take(tmp_path, tmp_path / "nowhere") + + +@pytest.mark.skipif(bsdtar() is None, reason="needs bsdtar") +def test_7z_archives_are_unpacked_with_bsdtar(tmp_path: Path) -> None: + source = tmp_path / "src" + source.mkdir() + (source / "plan.md").write_text("# Plan\n\nKayak on Sunday.", encoding="utf-8") + tool = bsdtar() + assert tool is not None + subprocess.run( + [tool, "--format", "7zip", "-cf", str(tmp_path / "notes.7z"), "-C", str(source), "plan.md"], + check=True, + ) + + _, titles = take(tmp_path, tmp_path / "notes.7z") + + assert titles == {"markdown: Plan": "notes.7z/plan.md"} diff --git a/tests/samples.py b/tests/samples.py new file mode 100644 index 0000000..ef4b2f8 --- /dev/null +++ b/tests/samples.py @@ -0,0 +1,181 @@ +"""Small documents of every kind ChatLore reads, written for tests.""" + +from __future__ import annotations + +import zipfile +from pathlib import Path + +W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" +A = "http://schemas.openxmlformats.org/drawingml/2006/main" +S = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +CORE = ( + '{title}' + "2026-03-04T10:00:00Z" +) + + +def _zip(path: Path, members: dict[str, str]) -> Path: + with zipfile.ZipFile(path, "w") as archive: + for name, text in members.items(): + archive.writestr(name, text) + return path + + +def docx(path: Path, title: str, paragraphs: list[str]) -> Path: + body = "".join(f"{text}" for text in paragraphs) + return _zip( + path, + { + "word/document.xml": f'{body}', + "docProps/core.xml": CORE.format(title=title), + }, + ) + + +def pptx(path: Path, slides: list[list[str]]) -> Path: + members = {} + for number, lines in enumerate(slides, start=1): + paragraphs = "".join(f"{line}" for line in lines) + members[f"ppt/slides/slide{number}.xml"] = ( + f'{paragraphs}' + ) + return _zip(path, members) + + +def xlsx(path: Path, sheet: str, rows: list[list[str]]) -> Path: + shared: list[str] = [] + cells = [] + for number, row in enumerate(rows, start=1): + values = [] + for value in row: + shared.append(value) + values.append(f'{len(shared) - 1}') + cells.append(f'{"".join(values)}') + strings = "".join(f"{value}" for value in shared) + return _zip( + path, + { + "xl/workbook.xml": ( + f'' + f'' + ), + "xl/_rels/workbook.xml.rels": ( + '' + ), + "xl/sharedStrings.xml": f'{strings}', + "xl/worksheets/sheet1.xml": ( + f'{"".join(cells)}' + ), + }, + ) + + +def odt(path: Path, title: str, paragraphs: list[str]) -> Path: + body = "".join(f"{text}" for text in paragraphs) + return _zip( + path, + { + "content.xml": ( + '' + f"{body}" + "" + ), + "meta.xml": ( + '' + f"{title}" + ), + }, + ) + + +def epub(path: Path, title: str, chapters: list[str]) -> Path: + items = "".join( + f'' + for number in range(len(chapters)) + ) + spine = "".join(f'' for number in range(len(chapters))) + members = { + "META-INF/container.xml": ( + '' + '' + ), + "OEBPS/content.opf": ( + '' + f"{title}" + f"{items}{spine}" + ), + } + for number, text in enumerate(chapters): + members[f"OEBPS/c{number}.xhtml"] = f"

{text}

" + return _zip(path, members) + + +def pdf(path: Path, title: str, lines: list[str]) -> Path: + """A one-page PDF with text in Helvetica, its offsets worked out.""" + text = " ".join( + f"BT /F1 12 Tf 72 {720 - 20 * number} Td ({line}) Tj ET" + for number, line in enumerate(lines) + ) + objects = [ + "<< /Type /Catalog /Pages 2 0 R >>", + "<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Contents 4 0 R " + "/Resources << /Font << /F1 5 0 R >> >> >>", + f"<< /Length {len(text)} >>\nstream\n{text}\nendstream", + "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>", + f"<< /Title ({title}) >>", + ] + out = b"%PDF-1.4\n" + offsets = [] + for number, body in enumerate(objects, start=1): + offsets.append(len(out)) + out += f"{number} 0 obj\n{body}\nendobj\n".encode("latin-1") + xref = len(out) + out += f"xref\n0 {len(objects) + 1}\n0000000000 65535 f \n".encode() + out += "".join(f"{offset:010d} 00000 n \n" for offset in offsets).encode() + out += ( + f"trailer\n<< /Size {len(objects) + 1} /Root 1 0 R /Info 6 0 R >>\n" + f"startxref\n{xref}\n%%EOF\n" + ).encode() + path.write_bytes(out) + return path + + +def email(path: Path, subject: str, body: str, message_id: str = "") -> Path: + path.write_text( + f"From: Maya \nTo: Sam \nSubject: {subject}\n" + f"Date: Tue, 03 Mar 2026 09:30:00 +0000\nMessage-ID: {message_id}\n" + f"Content-Type: text/plain; charset=utf-8\n\n{body}\n", + encoding="utf-8", + ) + return path + + +def mbox(path: Path, messages: list[tuple[str, str]]) -> Path: + text = "" + for number, (subject, body) in enumerate(messages): + text += ( + "From maya@example.com Tue Mar 3 09:30:00 2026\n" + f"From: maya@example.com\nSubject: {subject}\n" + f"Message-ID: \n\n{body}\n\n" + ) + path.write_text(text, encoding="utf-8") + return path + + +CHATGPT = """[{"title": "Trip ideas", "create_time": 1767225600, "update_time": 1767225700, +"id": "c-1", "current_node": "b", "mapping": { +"a": {"id": "a", "parent": null, "children": ["b"], "message": {"id": "a", +"author": {"role": "user"}, "create_time": 1767225600, +"content": {"content_type": "text", "parts": ["Where should I go in spring?"]}}}, +"b": {"id": "b", "parent": "a", "children": [], "message": {"id": "b", +"author": {"role": "assistant"}, "create_time": 1767225660, +"content": {"content_type": "text", "parts": ["Kyoto for the blossoms."]}}}}}]""" diff --git a/tests/test_cli_import.py b/tests/test_cli_import.py index d60baf6..d19f948 100644 --- a/tests/test_cli_import.py +++ b/tests/test_cli_import.py @@ -2,6 +2,7 @@ from __future__ import annotations +import re from pathlib import Path import pytest @@ -76,3 +77,37 @@ def test_note_and_stats(home: Path, fixtures: Path) -> None: assert totals[SourceKind.MARKDOWN.value].conversations == 4 assert totals[SourceKind.NOTE.value].conversations == 1 assert "markdown" in stats.output and "total" in stats.output + + +def test_import_takes_many_paths_of_any_kind(home: Path, fixtures: Path, tmp_path: Path) -> None: + from tests import samples + + samples.pdf(tmp_path / "report.pdf", "Tide report", ["High tide at 14:05"]) + (tmp_path / "photo.jpg").write_bytes(b"\x00") + + result = runner.invoke( + app, + [ + "import", + str(fixtures / "claude"), + str(tmp_path / "report.pdf"), + str(tmp_path / "photo.jpg"), + ], + ) + + output = " ".join(re.sub(r"[^\w.:,/ -]", " ", result.output).split()) + assert result.exit_code == 0, result.output + assert "from claude 3" in output and "from document 1" in output + assert "skipped files 1" in output + assert "photo.jpg: images, audio, video, and programs are not read" in output + assert Library(home).stats()["document"].conversations == 1 + + +def test_import_says_when_nothing_can_be_read(home: Path, tmp_path: Path) -> None: + (tmp_path / "photo.jpg").write_bytes(b"\x00") + + result = runner.invoke(app, ["import", str(tmp_path / "photo.jpg")]) + + assert result.exit_code == 1 + assert "nothing ChatLore can read was found" in result.output + assert not home.exists() diff --git a/tests/test_uploads.py b/tests/test_uploads.py index 577a888..fc0ee60 100644 --- a/tests/test_uploads.py +++ b/tests/test_uploads.py @@ -139,7 +139,10 @@ def test_an_unreadable_upload_fails_with_a_reason(private: TestClient, tmp_path: status = finished(private) assert status["stage"] == "failed" - assert "could not recognise" in status["error"] + assert status["error"] == "nothing ChatLore can read was found in the upload" + assert status["skipped_shown"] == [ + ["photo.zip/photo.jpg", "images, audio, video, and programs are not read"] + ] def test_a_zip_of_markdown_notes_and_an_archive_are_imported( @@ -355,3 +358,62 @@ def test_the_archive_downloaded_from_a_visitor_library_imports_anywhere( (tmp_path / "mine.zip").write_bytes(download.content) assert read_manifest(tmp_path / "mine.zip")["conversations"] == 3 + + +def send(client: TestClient, batch: str, name: str, content: bytes) -> Any: + return client.post( + "/library/files", + params={"batch": batch}, + content=content, + headers={**CHANGE, "X-Filename": name}, + ) + + +def test_many_files_with_their_folders_import_together( + private: TestClient, home: Path, fixtures: Path, tmp_path: Path +) -> None: + from tests import samples + + batch = "0123456789abcdef0123456789abcdef" + report = samples.pdf(tmp_path / "report.pdf", "Tide report", ["High tide at 14:05"]) + sent = [ + send( + private, + batch, + "Claude%20export/conversations.json", + (fixtures / "claude" / "conversations.json").read_bytes(), + ), + send(private, batch, "Docs/report.pdf", report.read_bytes()), + send(private, batch, "../../outside.md", b"# Outside\n\nStill inside."), + send(private, batch, "Docs/photo.jpg", b"\x00"), + ] + + started = private.post("/library/import", params={"batch": batch}, headers=CHANGE) + status = finished(private) + + assert [response.status_code for response in sent] == [200, 200, 200, 200] + assert sent[-1].json()["files"] == 4 + assert started.status_code == 202 and started.json()["file"] == "4 files" + assert status["stage"] == "done", status + assert status["sources"] == {"claude": 3, "document": 1, "markdown": 1} + assert status["skipped_files"] == 1 + assert status["skipped_shown"] == [ + ["Docs/photo.jpg", "images, audio, video, and programs are not read"] + ] + assert not (home / "outside.md").exists() + assert list((home / "uploads").iterdir()) == [] + + +def test_batches_are_checked(home: Path) -> None: + with TestClient(create_app(home, max_upload=10)) as client: + bad = send(client, "not-hex", "a.txt", b"x") + empty = client.post("/library/import", params={"batch": "f" * 32}, headers=CHANGE) + first = send(client, "a" * 32, "a.txt", b"12345") + over = send(client, "a" * 32, "b.txt", b"1234567") + unmarked = client.post("/library/files", params={"batch": "a" * 32}, content=b"x") + + assert bad.status_code == 422 + assert empty.status_code == 400 + assert first.status_code == 200 + assert over.status_code == 413 + assert unmarked.status_code == 403 diff --git a/uv.lock b/uv.lock index 0d1624d..1d5d045 100644 --- a/uv.lock +++ b/uv.lock @@ -357,6 +357,7 @@ dependencies = [ { name = "networkx" }, { name = "openai" }, { name = "pydantic" }, + { name = "pypdf" }, { name = "python-dotenv" }, { name = "pyyaml" }, { name = "rich" }, @@ -390,6 +391,7 @@ requires-dist = [ { name = "networkx", specifier = ">=3.7" }, { name = "openai", specifier = ">=3.17" }, { name = "pydantic", specifier = ">=2.9" }, + { name = "pypdf", specifier = ">=6.0" }, { name = "python-dotenv", specifier = ">=1.2" }, { name = "pyyaml", specifier = ">=6.0" }, { name = "rich", specifier = ">=13.7" }, @@ -1736,6 +1738,15 @@ crypto = [ { name = "cryptography" }, ] +[[package]] +name = "pypdf" +version = "6.19.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/1f/ac/63d71aaedb59acbcdef491e6ca6469165e3771c9c74358204818fd9bc5a6/pypdf-6.19.0.tar.gz", hash = "sha256:bbc43aca292369ccc6cbc8a921991ecf2538a3587ab5a116eff06c321d647155", size = 7033266, upload-time = "2026-09-16T09:32:05.946Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/3c/2c/c43c03eaf630435f023f1dc61ec4a4a78951ad5530a62c71cc89bde307b7/pypdf-6.19.0-py3-none-any.whl", hash = "sha256:7e5d6e730e7dae87d560a2cee218b852f6498c8be61966f3cd02ead971e48d14", size = 395480, upload-time = "2026-09-16T09:32:04.087Z" }, +] + [[package]] name = "pytest" version = "9.1.1"