Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,13 @@
All notable changes to this project are documented here, following
[Keep a Changelog](https://keepachangelog.com/) and semantic versioning.

## [0.1.17] - 2026-09-23

### Fixed

- Recognize AVIF, HEIC, HEIF, APNG, WebM, Ogg, FLAC and M4A as binary assets, avoiding text reads and inflated token estimates.
- Repair unresolved merge markers in the package lockfile while updating release metadata.

## [0.1.16] - 2026-09-17

### Fixed
Expand Down
2 changes: 2 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -65,6 +65,8 @@ ctxtrim · my-repo · 412 files
✓ .cursorignore (created, 34 patterns), .aiexclude (created, 34 patterns)
```

Known binary media, including AVIF, HEIC/HEIF, APNG, WebM, Ogg, FLAC and M4A, are not read as text and contribute no text tokens. They remain explicit ignore candidates.

Estimates use the widely-cited ~4-chars-per-token rule (great for ranking and relative savings; pass `--price` to match your model).

Vendored dependency and build output directories (`node_modules`, `dist`, `build`, ...) are reported as a single entry keyed by directory name — sized from a cheap stat walk without reading the files inside. Directory-level categories are decided by the directory name, not by reading its contents.
Expand Down
12 changes: 2 additions & 10 deletions package-lock.json

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 1 addition & 1 deletion package.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "ctxtrim",
"version": "0.1.16",
"version": "0.1.17",
"description": "Trim what bloats your AI coding context. Scan a repo, find the high-cost/low-value files ballooning your Claude Code / Cursor / Codex context, and write ignore files to cut token cost. Zero dependencies.",
"type": "module",
"bin": {
Expand Down
1 change: 1 addition & 0 deletions src/classify.js
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@ const DATA_EXT = new Set([

const BINARY_EXT = new Set([
".png", ".jpg", ".jpeg", ".gif", ".webp", ".ico", ".bmp", ".tiff",
".avif", ".heic", ".heif", ".apng", ".webm", ".ogg", ".flac", ".m4a",
".pdf", ".zip", ".gz", ".tar", ".tgz", ".7z", ".rar", ".jar", ".war",
".woff", ".woff2", ".ttf", ".otf", ".eot", ".mp3", ".mp4", ".mov", ".avi",
".wasm", ".so", ".dylib", ".dll", ".exe", ".bin", ".class", ".pyc",
Expand Down
57 changes: 57 additions & 0 deletions test/modern-binary-formats.test.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,57 @@
import fs from "node:fs";
import { syncBuiltinESMExports } from "node:module";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { test } from "node:test";
import assert from "node:assert/strict";
import { classify, classifyPath } from "../src/classify.js";
import { scanRepo } from "../src/scan.js";

const extensions = ["avif", "heic", "heif", "apng", "webm", "ogg", "flac", "m4a"];

test("modern image, video and audio formats are binary regardless of case", () => {
for (const extension of extensions) {
for (const ext of [extension, extension.toUpperCase()]) {
const name = `media/asset.${ext}`;
const result = classify(name, { tokens: 100 });
assert.equal(result.category, "binary", name);
assert.equal(result.trim, true, name);
assert.equal(result.binary, true, name);
assert.deepEqual(classifyPath(name), result);
}
}
assert.equal(classify("src/audio.js", { tokens: 100 }).trim, false);
});

test("scan never reads modern binary media or counts it as text tokens", (t) => {
const root = fs.mkdtempSync(join(tmpdir(), "ctxtrim-modern-media-"));
t.after(() => fs.rmSync(root, { recursive: true, force: true }));
const names = extensions.map((ext) => `asset.${ext}`);
for (const name of names) fs.writeFileSync(join(root, name), Buffer.from([0xff, 0x00, 0xfe]));
// Exercise both the full-read and partial-read paths if classification regresses.
fs.truncateSync(join(root, "asset.webm"), 5_000_001);
fs.writeFileSync(join(root, "source.js"), "export const value = 1;\n");
const reads = [];
for (const method of ["readFileSync", "openSync"]) {
const original = fs[method];
t.mock.method(fs, method, (...args) => {
reads.push(String(args[0]));
return original(...args);
});
}
syncBuiltinESMExports();
let result;
try { result = scanRepo(root); }
finally { t.mock.restoreAll(); syncBuiltinESMExports(); }

for (const name of names) {
assert.equal(reads.includes(join(root, name)), false, name);
const file = result.files.find((entry) => entry.rel === name);
assert.equal(file.category, "binary", name);
assert.equal(file.tokens, 0, name);
assert.equal(file.trim, true, name);
assert.ok(result.patterns.includes(name));
}
assert.equal(result.totals.textFiles, 1);
assert.equal(result.totals.totalTokens, result.files.find((file) => file.rel === "source.js").tokens);
});
Loading