From dd6f5ade2729de98b1084fe059860ab5b0aa8eb3 Mon Sep 17 00:00:00 2001 From: Yurii Bakurov <45154988+Yurii201811@users.noreply.github.com> Date: Wed, 23 Sep 2026 08:17:25 +0200 Subject: [PATCH] fix: classify notebooks as trimmable data --- CHANGELOG.md | 7 +++++++ README.md | 2 ++ package-lock.json | 12 ++---------- package.json | 2 +- src/classify.js | 2 +- test/notebooks.test.js | 37 +++++++++++++++++++++++++++++++++++++ 6 files changed, 50 insertions(+), 12 deletions(-) create mode 100644 test/notebooks.test.js diff --git a/CHANGELOG.md b/CHANGELOG.md index 966d236..ea73774 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,13 @@ All notable changes to this project are documented here, following [Keep a Changelog](https://keepachangelog.com/) and semantic versioning. +## [0.1.17] - 2026-09-23 + +### Fixed + +- Classify `.ipynb` notebooks as data trim candidates, including notebooks below the large-data threshold. +- Repair unresolved merge markers in the package lockfile while updating release metadata. + ## [0.1.16] - 2026-09-17 ### Fixed diff --git a/README.md b/README.md index 89e3050..f81cd0e 100644 --- a/README.md +++ b/README.md @@ -65,6 +65,8 @@ ctxtrim · my-repo · 412 files ✓ .cursorignore (created, 34 patterns), .aiexclude (created, 34 patterns) ``` +Jupyter notebooks (`.ipynb`) are treated as data and suggested for trimming regardless of size. This excludes the whole notebook, including its source cells, not only its outputs; review these suggestions before using `--write`. + Estimates use the widely-cited ~4-chars-per-token rule (great for ranking and relative savings; pass `--price` to match your model). Vendored dependency and build output directories (`node_modules`, `dist`, `build`, ...) are reported as a single entry keyed by directory name — sized from a cheap stat walk without reading the files inside. Directory-level categories are decided by the directory name, not by reading its contents. diff --git a/package-lock.json b/package-lock.json index 20a1a7d..4f11004 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,20 +1,12 @@ { "name": "ctxtrim", -<<<<<<< HEAD - "version": "0.1.16", -======= - "version": "0.1.15", ->>>>>>> origin/main + "version": "0.1.17", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "ctxtrim", -<<<<<<< HEAD - "version": "0.1.16", -======= - "version": "0.1.15", ->>>>>>> origin/main + "version": "0.1.17", "license": "MIT", "bin": { "ctxtrim": "bin/ctxtrim.js" diff --git a/package.json b/package.json index 8f855ca..726111b 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "ctxtrim", - "version": "0.1.16", + "version": "0.1.17", "description": "Trim what bloats your AI coding context. Scan a repo, find the high-cost/low-value files ballooning your Claude Code / Cursor / Codex context, and write ignore files to cut token cost. Zero dependencies.", "type": "module", "bin": { diff --git a/src/classify.js b/src/classify.js index 8a1bc2b..e9230fd 100644 --- a/src/classify.js +++ b/src/classify.js @@ -21,7 +21,7 @@ const LOCKFILES = new Set([ ]); const DATA_EXT = new Set([ - ".csv", ".tsv", ".parquet", ".ndjson", ".jsonl", ".sqlite", ".sqlite3", + ".ipynb", ".csv", ".tsv", ".parquet", ".ndjson", ".jsonl", ".sqlite", ".sqlite3", ".db", ".dump", ".log", ".pkl", ".npy", ".npz", ".arrow", ".feather", ".geojson", ]); diff --git a/test/notebooks.test.js b/test/notebooks.test.js new file mode 100644 index 0000000..1fdcec3 --- /dev/null +++ b/test/notebooks.test.js @@ -0,0 +1,37 @@ +import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { classify } from "../src/classify.js"; +import { scanRepo } from "../src/scan.js"; + +test("notebooks are data trim candidates below the large-data threshold", () => { + for (const name of ["analysis.ipynb", "analysis.IPYNB"]) { + const result = classify(name, { tokens: 1510, maxTokens: 2000 }); + assert.equal(result.category, "data", name); + assert.equal(result.trim, true, name); + assert.equal(result.binary, false, name); + } + assert.equal(classify("analysis.py", { tokens: 1510 }).trim, false); +}); + +test("scan accounts for output-heavy notebooks and suggests their ignore pattern", (t) => { + const root = mkdtempSync(join(tmpdir(), "ctxtrim-notebook-")); + t.after(() => rmSync(root, { recursive: true, force: true })); + const notebook = { + cells: [{ cell_type: "code", execution_count: 1, metadata: {}, source: ["print('x')"], + outputs: [{ output_type: "stream", name: "stdout", text: ["x".repeat(6000)] }] }], + metadata: {}, nbformat: 4, nbformat_minor: 5, + }; + writeFileSync(join(root, "analysis.ipynb"), JSON.stringify(notebook)); + writeFileSync(join(root, "analysis.py"), "print('x')\n"); + const result = scanRepo(root); + const file = result.files.find((entry) => entry.rel === "analysis.ipynb"); + assert.ok(file.tokens > 1500 && file.tokens < 2000); + assert.equal(file.category, "data"); + assert.equal(file.trim, true); + assert.equal(result.totals.trimTokens, file.tokens); + assert.deepEqual(result.patterns, ["analysis.ipynb"]); + assert.equal(result.files.find((entry) => entry.rel === "analysis.py").trim, false); +});