Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,11 @@
All notable changes to this project are documented here, following
[Keep a Changelog](https://keepachangelog.com/) and semantic versioning.

## [0.1.20] - 2026-09-23

### Fixed

- Classify `.ipynb` notebooks as data trim candidates, including notebooks below the large-data threshold.
## [0.1.19] - 2026-09-23

### Fixed
Expand Down
4 changes: 4 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -67,7 +67,11 @@ ctxtrim · my-repo · 412 files

Known binary media, including AVIF, HEIC/HEIF, APNG, WebM, Ogg, FLAC and M4A, are not read as text and contribute no text tokens. They remain explicit ignore candidates.

<<<<<<< HEAD
Jupyter notebooks (`.ipynb`) are treated as data and suggested for trimming regardless of size. This excludes the whole notebook, including its source cells, not only its outputs; review these suggestions before using `--write`.
=======
Loose `.snap` files are treated as generated output and suggested for trimming, just like files inside `__snapshots__/` directories.
>>>>>>> origin/main

Estimates use the widely-cited ~4-chars-per-token rule (great for ranking and relative savings; pass `--price` to match your model).

Expand Down
4 changes: 2 additions & 2 deletions package-lock.json

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 1 addition & 1 deletion package.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "ctxtrim",
"version": "0.1.19",
"version": "0.1.20",
"description": "Trim what bloats your AI coding context. Scan a repo, find the high-cost/low-value files ballooning your Claude Code / Cursor / Codex context, and write ignore files to cut token cost. Zero dependencies.",
"type": "module",
"bin": {
Expand Down
2 changes: 1 addition & 1 deletion src/classify.js
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@ const LOCKFILES = new Set([
]);

const DATA_EXT = new Set([
".csv", ".tsv", ".parquet", ".ndjson", ".jsonl", ".sqlite", ".sqlite3",
".ipynb", ".csv", ".tsv", ".parquet", ".ndjson", ".jsonl", ".sqlite", ".sqlite3",
".db", ".dump", ".log", ".pkl", ".npy", ".npz", ".arrow", ".feather", ".geojson",
]);

Expand Down
37 changes: 37 additions & 0 deletions test/notebooks.test.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,37 @@
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { test } from "node:test";
import assert from "node:assert/strict";
import { classify } from "../src/classify.js";
import { scanRepo } from "../src/scan.js";

test("notebooks are data trim candidates below the large-data threshold", () => {
for (const name of ["analysis.ipynb", "analysis.IPYNB"]) {
const result = classify(name, { tokens: 1510, maxTokens: 2000 });
assert.equal(result.category, "data", name);
assert.equal(result.trim, true, name);
assert.equal(result.binary, false, name);
}
assert.equal(classify("analysis.py", { tokens: 1510 }).trim, false);
});

test("scan accounts for output-heavy notebooks and suggests their ignore pattern", (t) => {
const root = mkdtempSync(join(tmpdir(), "ctxtrim-notebook-"));
t.after(() => rmSync(root, { recursive: true, force: true }));
const notebook = {
cells: [{ cell_type: "code", execution_count: 1, metadata: {}, source: ["print('x')"],
outputs: [{ output_type: "stream", name: "stdout", text: ["x".repeat(6000)] }] }],
metadata: {}, nbformat: 4, nbformat_minor: 5,
};
writeFileSync(join(root, "analysis.ipynb"), JSON.stringify(notebook));
writeFileSync(join(root, "analysis.py"), "print('x')\n");
const result = scanRepo(root);
const file = result.files.find((entry) => entry.rel === "analysis.ipynb");
assert.ok(file.tokens > 1500 && file.tokens < 2000);
assert.equal(file.category, "data");
assert.equal(file.trim, true);
assert.equal(result.totals.trimTokens, file.tokens);
assert.deepEqual(result.patterns, ["analysis.ipynb"]);
assert.equal(result.files.find((entry) => entry.rel === "analysis.py").trim, false);
});
Loading