diff --git a/CHANGELOG.md b/CHANGELOG.md index 27d2a28..9404849 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,11 @@ All notable changes to this project are documented here, following [Keep a Changelog](https://keepachangelog.com/) and semantic versioning. +## [0.1.20] - 2026-09-23 + +### Fixed + +- Classify `.ipynb` notebooks as data trim candidates, including notebooks below the large-data threshold. ## [0.1.19] - 2026-09-23 ### Fixed diff --git a/README.md b/README.md index 59332a9..cc269f0 100644 --- a/README.md +++ b/README.md @@ -67,7 +67,11 @@ ctxtrim · my-repo · 412 files Known binary media, including AVIF, HEIC/HEIF, APNG, WebM, Ogg, FLAC and M4A, are not read as text and contribute no text tokens. They remain explicit ignore candidates. +<<<<<<< HEAD +Jupyter notebooks (`.ipynb`) are treated as data and suggested for trimming regardless of size. This excludes the whole notebook, including its source cells, not only its outputs; review these suggestions before using `--write`. +======= Loose `.snap` files are treated as generated output and suggested for trimming, just like files inside `__snapshots__/` directories. +>>>>>>> origin/main Estimates use the widely-cited ~4-chars-per-token rule (great for ranking and relative savings; pass `--price` to match your model). diff --git a/package-lock.json b/package-lock.json index 125adc3..ff1f2a8 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "ctxtrim", - "version": "0.1.19", + "version": "0.1.20", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "ctxtrim", - "version": "0.1.19", + "version": "0.1.20", "license": "MIT", "bin": { "ctxtrim": "bin/ctxtrim.js" diff --git a/package.json b/package.json index 2a851ac..b7f719d 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "ctxtrim", - "version": "0.1.19", + "version": "0.1.20", "description": "Trim what bloats your AI coding context. Scan a repo, find the high-cost/low-value files ballooning your Claude Code / Cursor / Codex context, and write ignore files to cut token cost. Zero dependencies.", "type": "module", "bin": { diff --git a/src/classify.js b/src/classify.js index 8900dec..8c461d8 100644 --- a/src/classify.js +++ b/src/classify.js @@ -21,7 +21,7 @@ const LOCKFILES = new Set([ ]); const DATA_EXT = new Set([ - ".csv", ".tsv", ".parquet", ".ndjson", ".jsonl", ".sqlite", ".sqlite3", + ".ipynb", ".csv", ".tsv", ".parquet", ".ndjson", ".jsonl", ".sqlite", ".sqlite3", ".db", ".dump", ".log", ".pkl", ".npy", ".npz", ".arrow", ".feather", ".geojson", ]); diff --git a/test/notebooks.test.js b/test/notebooks.test.js new file mode 100644 index 0000000..1fdcec3 --- /dev/null +++ b/test/notebooks.test.js @@ -0,0 +1,37 @@ +import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { classify } from "../src/classify.js"; +import { scanRepo } from "../src/scan.js"; + +test("notebooks are data trim candidates below the large-data threshold", () => { + for (const name of ["analysis.ipynb", "analysis.IPYNB"]) { + const result = classify(name, { tokens: 1510, maxTokens: 2000 }); + assert.equal(result.category, "data", name); + assert.equal(result.trim, true, name); + assert.equal(result.binary, false, name); + } + assert.equal(classify("analysis.py", { tokens: 1510 }).trim, false); +}); + +test("scan accounts for output-heavy notebooks and suggests their ignore pattern", (t) => { + const root = mkdtempSync(join(tmpdir(), "ctxtrim-notebook-")); + t.after(() => rmSync(root, { recursive: true, force: true })); + const notebook = { + cells: [{ cell_type: "code", execution_count: 1, metadata: {}, source: ["print('x')"], + outputs: [{ output_type: "stream", name: "stdout", text: ["x".repeat(6000)] }] }], + metadata: {}, nbformat: 4, nbformat_minor: 5, + }; + writeFileSync(join(root, "analysis.ipynb"), JSON.stringify(notebook)); + writeFileSync(join(root, "analysis.py"), "print('x')\n"); + const result = scanRepo(root); + const file = result.files.find((entry) => entry.rel === "analysis.ipynb"); + assert.ok(file.tokens > 1500 && file.tokens < 2000); + assert.equal(file.category, "data"); + assert.equal(file.trim, true); + assert.equal(result.totals.trimTokens, file.tokens); + assert.deepEqual(result.patterns, ["analysis.ipynb"]); + assert.equal(result.files.find((entry) => entry.rel === "analysis.py").trim, false); +});