diff --git a/CHANGELOG.md b/CHANGELOG.md index 966d236..6905142 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,13 @@ All notable changes to this project are documented here, following [Keep a Changelog](https://keepachangelog.com/) and semantic versioning. +## [0.1.17] - 2026-09-23 + +### Fixed + +- Recognize AVIF, HEIC, HEIF, APNG, WebM, Ogg, FLAC and M4A as binary assets, avoiding text reads and inflated token estimates. +- Repair unresolved merge markers in the package lockfile while updating release metadata. + ## [0.1.16] - 2026-09-17 ### Fixed diff --git a/README.md b/README.md index 89e3050..6b38317 100644 --- a/README.md +++ b/README.md @@ -65,6 +65,8 @@ ctxtrim · my-repo · 412 files ✓ .cursorignore (created, 34 patterns), .aiexclude (created, 34 patterns) ``` +Known binary media, including AVIF, HEIC/HEIF, APNG, WebM, Ogg, FLAC and M4A, are not read as text and contribute no text tokens. They remain explicit ignore candidates. + Estimates use the widely-cited ~4-chars-per-token rule (great for ranking and relative savings; pass `--price` to match your model). Vendored dependency and build output directories (`node_modules`, `dist`, `build`, ...) are reported as a single entry keyed by directory name — sized from a cheap stat walk without reading the files inside. Directory-level categories are decided by the directory name, not by reading its contents. diff --git a/package-lock.json b/package-lock.json index 20a1a7d..4f11004 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,20 +1,12 @@ { "name": "ctxtrim", -<<<<<<< HEAD - "version": "0.1.16", -======= - "version": "0.1.15", ->>>>>>> origin/main + "version": "0.1.17", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "ctxtrim", -<<<<<<< HEAD - "version": "0.1.16", -======= - "version": "0.1.15", ->>>>>>> origin/main + "version": "0.1.17", "license": "MIT", "bin": { "ctxtrim": "bin/ctxtrim.js" diff --git a/package.json b/package.json index 8f855ca..726111b 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "ctxtrim", - "version": "0.1.16", + "version": "0.1.17", "description": "Trim what bloats your AI coding context. Scan a repo, find the high-cost/low-value files ballooning your Claude Code / Cursor / Codex context, and write ignore files to cut token cost. Zero dependencies.", "type": "module", "bin": { diff --git a/src/classify.js b/src/classify.js index 8a1bc2b..6859387 100644 --- a/src/classify.js +++ b/src/classify.js @@ -27,6 +27,7 @@ const DATA_EXT = new Set([ const BINARY_EXT = new Set([ ".png", ".jpg", ".jpeg", ".gif", ".webp", ".ico", ".bmp", ".tiff", + ".avif", ".heic", ".heif", ".apng", ".webm", ".ogg", ".flac", ".m4a", ".pdf", ".zip", ".gz", ".tar", ".tgz", ".7z", ".rar", ".jar", ".war", ".woff", ".woff2", ".ttf", ".otf", ".eot", ".mp3", ".mp4", ".mov", ".avi", ".wasm", ".so", ".dylib", ".dll", ".exe", ".bin", ".class", ".pyc", diff --git a/test/modern-binary-formats.test.js b/test/modern-binary-formats.test.js new file mode 100644 index 0000000..b878708 --- /dev/null +++ b/test/modern-binary-formats.test.js @@ -0,0 +1,57 @@ +import fs from "node:fs"; +import { syncBuiltinESMExports } from "node:module"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { classify, classifyPath } from "../src/classify.js"; +import { scanRepo } from "../src/scan.js"; + +const extensions = ["avif", "heic", "heif", "apng", "webm", "ogg", "flac", "m4a"]; + +test("modern image, video and audio formats are binary regardless of case", () => { + for (const extension of extensions) { + for (const ext of [extension, extension.toUpperCase()]) { + const name = `media/asset.${ext}`; + const result = classify(name, { tokens: 100 }); + assert.equal(result.category, "binary", name); + assert.equal(result.trim, true, name); + assert.equal(result.binary, true, name); + assert.deepEqual(classifyPath(name), result); + } + } + assert.equal(classify("src/audio.js", { tokens: 100 }).trim, false); +}); + +test("scan never reads modern binary media or counts it as text tokens", (t) => { + const root = fs.mkdtempSync(join(tmpdir(), "ctxtrim-modern-media-")); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + const names = extensions.map((ext) => `asset.${ext}`); + for (const name of names) fs.writeFileSync(join(root, name), Buffer.from([0xff, 0x00, 0xfe])); + // Exercise both the full-read and partial-read paths if classification regresses. + fs.truncateSync(join(root, "asset.webm"), 5_000_001); + fs.writeFileSync(join(root, "source.js"), "export const value = 1;\n"); + const reads = []; + for (const method of ["readFileSync", "openSync"]) { + const original = fs[method]; + t.mock.method(fs, method, (...args) => { + reads.push(String(args[0])); + return original(...args); + }); + } + syncBuiltinESMExports(); + let result; + try { result = scanRepo(root); } + finally { t.mock.restoreAll(); syncBuiltinESMExports(); } + + for (const name of names) { + assert.equal(reads.includes(join(root, name)), false, name); + const file = result.files.find((entry) => entry.rel === name); + assert.equal(file.category, "binary", name); + assert.equal(file.tokens, 0, name); + assert.equal(file.trim, true, name); + assert.ok(result.patterns.includes(name)); + } + assert.equal(result.totals.textFiles, 1); + assert.equal(result.totals.totalTokens, result.files.find((file) => file.rel === "source.js").tokens); +});