From 76006a67a8fd8a8b2af734229127610ed8e951bc Mon Sep 17 00:00:00 2001 From: ranxianglei Date: Thu, 30 Jul 2026 10:05:21 +0800 Subject: [PATCH] fix: re-add HOW_TO_COMPRESS_RULES to nudge injection for high-attention summary guidance MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit v1.14.7 removed HOW_TO_COMPRESS_RULES from nudge to save ~2.4K tokens/turn, keeping it in system prompt only. But in 8700+ message sessions, the system prompt is 12K chars buried at the START of 1M+ token context — the model's attention to the rules degrades ('lost in the middle' effect). When the nudge fires and the model needs to write summaries, the detailed rules (KEEP VERBATIM, DROP, PRIORITY) are not at high attention. Re-add HOW_TO_COMPRESS_RULES to nudge text: non-maxLimit path (before ranges list) and maxLimit path (in strong alert). +5K chars per nudge, but nudges are gated by nudgeGrowthTokens — not every turn. Better summaries = fewer recompressions = net token savings. --- .../REQ.md | 21 +++++++++++++++++++ .../WORKLOG.md | 20 ++++++++++++++++++ lib/messages/inject/inject.ts | 6 +++--- 3 files changed, 44 insertions(+), 3 deletions(-) create mode 100644 devlog/2026-07-30_how-to-compress-investigation/REQ.md create mode 100644 devlog/2026-07-30_how-to-compress-investigation/WORKLOG.md diff --git a/devlog/2026-07-30_how-to-compress-investigation/REQ.md b/devlog/2026-07-30_how-to-compress-investigation/REQ.md new file mode 100644 index 0000000..186320d --- /dev/null +++ b/devlog/2026-07-30_how-to-compress-investigation/REQ.md @@ -0,0 +1,21 @@ +# REQ: Re-add HOW_TO_COMPRESS_RULES to nudge injection + +## Problem + +v1.14.7 (PR #228) removed `HOW_TO_COMPRESS_RULES` from nudge templates and the breakdown block, keeping it only in the system prompt (injected every turn). The rationale was token savings (~2.4-3.6K per nudge turn) and deduplication. + +However, in long sessions (8,700+ messages), the system prompt is 12K chars buried at the START of a 1M+ token context. The model's attention to rules in the system prompt degrades significantly — the "lost in the middle" effect. When the nudge fires and the model is about to compress, the detailed summary-writing rules (KEEP VERBATIM, DROP, PRIORITY) are not at high attention where they're needed most. + +The nudge message IS at the END of the context (high attention). Before v1.14.7, it contained HOW_TO_COMPRESS_RULES. After v1.14.7, it only contains COMPRESS_PHILOSOPHY (a short 889-char high-level guide) — missing the detailed 4936-char rules for HOW to write good summaries. + +## Fix + +Re-add HOW_TO_COMPRESS_RULES to the nudge message in two places: + +1. **Non-maxLimit path** (`inject.ts:534`): Between COMPRESS_PHILOSOPHY and the compressible ranges list. Only when `recommendedRanges.length > 0` (no point adding rules if nothing to compress). + +2. **MaxLimit path** (`inject.ts:545`): In the strong alert tipsText, before the JSON example. + +## Tradeoff + ++5K chars when a nudge fires (gated by nudgeGrowthTokens — not every turn). Better summaries = fewer recompressions needed = net token savings. The rules appear at HIGH attention (end of context) exactly when the model needs them. diff --git a/devlog/2026-07-30_how-to-compress-investigation/WORKLOG.md b/devlog/2026-07-30_how-to-compress-investigation/WORKLOG.md new file mode 100644 index 0000000..569a61a --- /dev/null +++ b/devlog/2026-07-30_how-to-compress-investigation/WORKLOG.md @@ -0,0 +1,20 @@ +# WORKLOG: Re-add HOW_TO_COMPRESS_RULES to nudge injection + +## 2026-07-30 + +### Investigation +- User reported "how to compress" injection getting lost (session ses_0b89319b, floor 8683) +- Checked deployed bundle: HOW_TO_COMPRESS_RULES IS in system prompt (system.ts:58) +- Checked debug nudge message: COMPRESS_PHILOSOPHY present but HOW_TO_COMPRESS_RULES missing +- Root cause: v1.14.7 (PR #228) removed rules from nudge to save tokens; kept in system prompt only +- "Lost in the middle" effect: system prompt is 12K chars at START of 1M+ token context — low attention +- Nudge message is at END of context — high attention — but rules are no longer there + +### Implementation +- `inject.ts:44`: Added `HOW_TO_COMPRESS_RULES` to import from cc-alg +- `inject.ts:534`: Added rules before compressible ranges list (non-maxLimit path) +- `inject.ts:545`: Added rules to strong alert tipsText (maxLimit path) + +### Verification +- typecheck: clean +- 934 tests pass (0 failures) diff --git a/lib/messages/inject/inject.ts b/lib/messages/inject/inject.ts index 953f823..49564ba 100644 --- a/lib/messages/inject/inject.ts +++ b/lib/messages/inject/inject.ts @@ -41,7 +41,7 @@ import { resolveAdaptiveNudgeGrowth, } from "./utils" import { buildCompressedBlockGuidance } from "../../prompts/extensions/nudge" -import { COMPRESS_PHILOSOPHY, TIER2_DISTILL_RULES, TIER3_CONDENSE_RULES } from "context-compress-algorithms/prompts" +import { COMPRESS_PHILOSOPHY, HOW_TO_COMPRESS_RULES, TIER2_DISTILL_RULES, TIER3_CONDENSE_RULES } from "context-compress-algorithms/prompts" import { getTierTokenUsage } from "../../state/utils" /** @@ -532,7 +532,7 @@ export const injectCompressNudges = ( } if (recommendedRanges.length > 0) { - breakdown += `\n\n${formatCompressibleRanges(recommendedRanges, contextRanges.protected)}` + breakdown += `\n\n${HOW_TO_COMPRESS_RULES}\n\n${formatCompressibleRanges(recommendedRanges, contextRanges.protected)}` breakdown += `\n💡 Compress all ranges in one call (pass multiple content entries: \`content: [{...}, {...}]\`).` } breakdown += `\nUse \`acp_status({scope:"uncompressed"})\` to re-fetch compressible ranges after compressing, or \`acp_status\` for compressed block details.` @@ -543,7 +543,7 @@ export const injectCompressNudges = ( // maxLimit strong alert + lastNudgeShownTokens + block aging guidance if (effectiveTipsVariant === "maxLimit") { tipsText = - '\n\n⚠️ Context limit reached — compress now. Prioritize consumed tool outputs.\n\n{ "topic": "...", "content": [{ "startId": "", "endId": "", "summary": "..." }] }\n\nOnly use IDs from visible messages above. Compress older work first.' + '\n\n⚠️ Context limit reached — compress now. Prioritize consumed tool outputs.\n\n' + HOW_TO_COMPRESS_RULES + '\n\n{ "topic": "...", "content": [{ "startId": "", "endId": "", "summary": "..." }] }\n\nOnly use IDs from visible messages above. Compress older work first.' } // Intentionally do NOT update lastPerMessageNudgeTokens here — nudges // repeat every turn until the model actually compresses.