From 4f71fee74ba6b943473a63bc4762e3e4f5bbe528 Mon Sep 17 00:00:00 2001 From: antra-tess Date: Tue, 15 Sep 2026 12:55:41 -0700 Subject: [PATCH 1/8] feat(modules): add HistoryModule for full uncompressed message history MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Exposes three read-only agent tools backed by context-manager's new native chronicle secondary-index queries (queryMessagesByTime / queryMessagesByChannel / queryMessagesByTimeAndChannel / getChannelMessageCounts / getChannelTokenStats, @animalabs/context-manager >= 0.6.0): - `stats` — per-channel message counts (all-time) + token totals (range-scoped), for orienting before pulling raw content. - `extract` — paginated raw messages for a time range and/or channel, rendered as readable text by default or raw content blocks on request. - `search` — substring/regex match over a narrowed candidate window, with an explicit truncation signal when the caller's filter was too broad for maxScan rather than silently missing matches past the cap. These queries are O(log n + k) against chronicle's native field indexes, not a full-store scan, so the module stays fast against multi-million-message production stores. Kept deliberately read-only (no side effects), same posture as HealthModule — bind(contextManager) after AgentFramework.create(), mirroring HealthModule.bind(). Requires the companion chronicle (native secondary field index) and context-manager (query wrappers) changes to actually exercise the native path; degrades to a clear tool error (not a crash) against an older chronicle build lacking the capability. Co-Authored-By: Claude Sonnet 5 --- src/modules/history/index.ts | 443 +++++++++++++++++++++++++++++++++++ src/modules/index.ts | 2 + test/history-module.test.ts | 355 ++++++++++++++++++++++++++++ 3 files changed, 800 insertions(+) create mode 100644 src/modules/history/index.ts create mode 100644 test/history-module.test.ts diff --git a/src/modules/history/index.ts b/src/modules/history/index.ts new file mode 100644 index 00000000..0c0f0d8b --- /dev/null +++ b/src/modules/history/index.ts @@ -0,0 +1,443 @@ +/** + * HistoryModule — read-only access to an agent's full uncompressed message + * history, backed by context-manager's native chronicle secondary index + * (`queryMessagesByTime` / `queryMessagesByChannel` / + * `queryMessagesByTimeAndChannel` / `getChannelMessageCounts` / + * `getChannelTokenStats`, @animalabs/context-manager >= 0.6.0). Those calls + * are O(log n + k) against chronicle's `/timestamp` and + * `/metadata/external/channelId` field indexes, not a full-store scan, so + * this module can answer "what happened in #foo last Tuesday" against a + * multi-million-message store without walking the whole log. + * + * Three tools: + * - `stats` — per-channel message counts (all-time) + token totals + * (range-scoped), for orienting before pulling raw content. + * - `extract` — paginated raw messages for a time range and/or channel. + * - `search` — substring/regex match over a narrowed candidate window, + * with an explicit truncation signal when the caller's + * filter was too broad for `maxScan`. + * + * Kept deliberately read-only, same posture as HealthModule: no side + * effects, no message mutation. `onProcess` is a no-op. + * + * Every query dispatch collapses onto `queryMessagesByTimeAndChannel` — its + * own implementation already handles "only a time range", "only a channel", + * or "neither" by delegating to the single-filter native queries (see + * context-manager's `test/message-store-history-index.test.ts`), so this + * module doesn't need to re-decide which of the three query methods to call. + */ + +import type { ContextManager, StoredMessage, ChannelCount, ChannelTokenStats } from '@animalabs/context-manager'; +import type { ContentBlock } from '@animalabs/membrane'; +import type { Module, ModuleContext } from '../../types/module.js'; +import type { ToolDefinition, ToolCall, ToolResult, ProcessEvent } from '../../types/events.js'; +import type { EventResponse, ProcessState } from '../../types/module.js'; + +// ============================================================================ +// Tool input shapes +// ============================================================================ + +interface StatsInput { + from?: string; + to?: string; + channelId?: string; +} + +interface ExtractInput { + from?: string; + to?: string; + channelId?: string; + limit?: number; + offset?: number; + format?: 'text' | 'raw'; +} + +interface SearchInput { + query: string; + regex?: boolean; + caseSensitive?: boolean; + from?: string; + to?: string; + channelId?: string; + limit?: number; + maxScan?: number; +} + +// ============================================================================ +// Limits +// ============================================================================ + +const EXTRACT_DEFAULT_LIMIT = 50; +const EXTRACT_MAX_LIMIT = 200; + +const SEARCH_DEFAULT_LIMIT = 20; +const SEARCH_MAX_LIMIT = 100; +const SEARCH_DEFAULT_MAX_SCAN = 5000; +const SEARCH_MAX_MAX_SCAN = 50000; + +/** Characters of surrounding context kept on each side of a search snippet. */ +const SNIPPET_CONTEXT_CHARS = 80; +/** Fallback snippet length when a match position isn't meaningful to center on. */ +const SNIPPET_FALLBACK_CHARS = 160; + +export class HistoryModule implements Module { + readonly name = 'history'; + + private ctx: ModuleContext | null = null; + private cm: ContextManager | null = null; + + /** + * Wire the context-manager instance. Host calls this after + * ContextManager.open() so the module can issue the native index-backed + * queries — ModuleContext itself exposes no store/context-manager + * reference, only the narrower message CRUD surface (addMessage/getMessage/ + * queryMessages), which can't do range or channel queries. + */ + bind(contextManager: ContextManager): void { + this.cm = contextManager; + } + + async start(ctx: ModuleContext): Promise { + this.ctx = ctx; + } + + async stop(): Promise { + this.ctx = null; + } + + getTools(): ToolDefinition[] { + return [ + { + name: 'stats', + description: + 'Orient before pulling raw history: per-channel message counts and token totals. ' + + 'messageCountsAllTime is always all-time and all-channel-scope (the underlying index has no ' + + 'range dimension) — it is NOT limited by from/to, only optionally filtered to one channelId. ' + + 'tokenStatsForRange IS scoped to from/to when given. The two are reported separately, never ' + + 'merged, because their range semantics differ.', + inputSchema: { + type: 'object' as const, + properties: { + from: { type: 'string', description: 'ISO 8601 inclusive lower bound for tokenStatsForRange. Omit for open-ended.' }, + to: { type: 'string', description: 'ISO 8601 inclusive upper bound for tokenStatsForRange. Omit for open-ended.' }, + channelId: { type: 'string', description: 'Restrict both parts of the response to one channel.' }, + }, + }, + }, + { + name: 'extract', + description: + 'Fetch raw, uncompressed messages for a time range and/or channel, paginated oldest-first. ' + + 'Use after `stats` to pull the actual content. format:"text" flattens each message to a short ' + + 'readable string (tool_use/tool_result/thinking/images rendered as bracketed labels); ' + + 'format:"raw" returns the content blocks unmodified.', + inputSchema: { + type: 'object' as const, + properties: { + from: { type: 'string', description: 'ISO 8601 inclusive lower bound. Omit for open-ended.' }, + to: { type: 'string', description: 'ISO 8601 inclusive upper bound. Omit for open-ended.' }, + channelId: { type: 'string', description: 'Restrict to one channel.' }, + limit: { type: 'number', description: `Max messages to return (default ${EXTRACT_DEFAULT_LIMIT}, hard cap ${EXTRACT_MAX_LIMIT}).` }, + offset: { type: 'number', description: 'Number of matching messages to skip (default 0).' }, + format: { type: 'string', enum: ['text', 'raw'], description: 'Content rendering (default "text").' }, + }, + }, + }, + { + name: 'search', + description: + 'Search message history for a substring (default) or regex match, within an optional time ' + + 'range/channel window. Narrows candidates via the same query as `extract` before matching, up ' + + 'to maxScan candidates — if the narrowed window is larger than maxScan, the response reports ' + + 'truncated:true up front rather than silently missing later matches; narrow the filter or raise ' + + 'maxScan and retry.', + inputSchema: { + type: 'object' as const, + properties: { + query: { type: 'string', description: 'Substring (or regex source, when regex:true) to search for.' }, + regex: { type: 'boolean', description: 'Treat query as a regular expression (default false).' }, + caseSensitive: { type: 'boolean', description: 'Case-sensitive match (default false).' }, + from: { type: 'string', description: 'ISO 8601 inclusive lower bound. Omit for open-ended.' }, + to: { type: 'string', description: 'ISO 8601 inclusive upper bound. Omit for open-ended.' }, + channelId: { type: 'string', description: 'Restrict to one channel.' }, + limit: { type: 'number', description: `Max matches to return (default ${SEARCH_DEFAULT_LIMIT}, hard cap ${SEARCH_MAX_LIMIT}).` }, + maxScan: { type: 'number', description: `Max candidate messages to scan (default ${SEARCH_DEFAULT_MAX_SCAN}, hard cap ${SEARCH_MAX_MAX_SCAN}).` }, + }, + required: ['query'], + }, + }, + ]; + } + + async handleToolCall(call: ToolCall): Promise { + try { + if (!this.cm) { + throw new Error('HistoryModule not bound — host must call bind(contextManager) before tool dispatch.'); + } + switch (call.name) { + case 'stats': + return this.handleStats((call.input ?? {}) as StatsInput); + case 'extract': + return this.handleExtract((call.input ?? {}) as ExtractInput); + case 'search': + return this.handleSearch((call.input ?? {}) as SearchInput); + default: + return { success: false, isError: true, error: `Unknown tool: ${call.name}` }; + } + } catch (error) { + // Catches both our own validation errors (bad ISO date, invalid regex, + // unbound module) and context-manager's capability-absent error + // ("Chronicle history index unsupported...", thrown by + // queryMessagesByTime/queryMessagesByChannel/queryMessagesByTimeAndChannel/ + // getChannelMessageCounts/getChannelTokenStats on a chronicle build that + // predates the native index-query capability) — surfaced as a normal + // tool error rather than crashing the module. + return { + success: false, + isError: true, + error: error instanceof Error ? error.message : String(error), + }; + } + } + + async onProcess(_event: ProcessEvent, _state: ProcessState): Promise { + return {}; + } + + // ========================================================================== + // stats + // ========================================================================== + + private handleStats(input: StatsInput): ToolResult { + const fromMs = parseIsoDate(input.from, 'from'); + const toMs = parseIsoDate(input.to, 'to'); + const cm = this.cm as ContextManager; + + let messageCountsAllTime: ChannelCount[] = cm.getChannelMessageCounts(); + let tokenStatsForRange: ChannelTokenStats = cm.getChannelTokenStats({ fromMs, toMs }); + + if (input.channelId) { + messageCountsAllTime = messageCountsAllTime.filter((c) => c.channelId === input.channelId); + tokenStatsForRange = { + // Totals stay whole-range (they're documented as store-wide-for-the-range, + // not per-channel); only the byChannel breakdown is narrowed. Labeling the + // field `byChannel` (rather than folding a channel-filtered total into + // totalMessages/totalTokensEstimate) keeps this from reading as a + // misleadingly-merged single-channel total. + totalMessages: tokenStatsForRange.totalMessages, + totalTokensEstimate: tokenStatsForRange.totalTokensEstimate, + byChannel: tokenStatsForRange.byChannel.filter((c) => c.channelId === input.channelId), + }; + } + + return { + success: true, + data: { + query: { from: input.from ?? null, to: input.to ?? null, channelId: input.channelId ?? null }, + messageCountsAllTime, + tokenStatsForRange, + }, + }; + } + + // ========================================================================== + // extract + // ========================================================================== + + private handleExtract(input: ExtractInput): ToolResult { + const fromMs = parseIsoDate(input.from, 'from'); + const toMs = parseIsoDate(input.to, 'to'); + const limit = Math.min(input.limit ?? EXTRACT_DEFAULT_LIMIT, EXTRACT_MAX_LIMIT); + const offset = input.offset ?? 0; + const format = input.format ?? 'text'; + + const cm = this.cm as ContextManager; + const result = cm.queryMessagesByTimeAndChannel({ + fromMs, + toMs, + channelId: input.channelId, + limit, + offset, + }); + + return { + success: true, + data: { + matchedCount: result.matchedCount, + returned: result.messages.length, + messages: result.messages.map((msg) => projectMessage(msg, format)), + }, + }; + } + + // ========================================================================== + // search + // ========================================================================== + + private handleSearch(input: SearchInput): ToolResult { + if (!input.query) { + throw new Error('search requires a non-empty "query".'); + } + const fromMs = parseIsoDate(input.from, 'from'); + const toMs = parseIsoDate(input.to, 'to'); + const limit = Math.min(input.limit ?? SEARCH_DEFAULT_LIMIT, SEARCH_MAX_LIMIT); + const maxScan = Math.min(input.maxScan ?? SEARCH_DEFAULT_MAX_SCAN, SEARCH_MAX_MAX_SCAN); + const caseSensitive = input.caseSensitive ?? false; + + // Compile the matcher up front — an invalid regex is a clean tool error, + // not a crash mid-scan. + let matcher: RegExp | null = null; + if (input.regex) { + try { + matcher = new RegExp(input.query, caseSensitive ? '' : 'i'); + } catch (error) { + throw new Error(`Invalid regex "${input.query}": ${error instanceof Error ? error.message : String(error)}`); + } + } + const needle = caseSensitive ? input.query : input.query.toLowerCase(); + + const cm = this.cm as ContextManager; + // Fetch one more than maxScan so an oversized candidate window is + // detected up front (before any matching is attempted), rather than + // silently scanning maxScan candidates and returning as if that were the + // whole window. See the tool description's `truncated` contract. + const probe = cm.queryMessagesByTimeAndChannel({ + fromMs, + toMs, + channelId: input.channelId, + limit: maxScan + 1, + offset: 0, + }); + const truncated = probe.messages.length > maxScan; + const candidates = truncated ? probe.messages.slice(0, maxScan) : probe.messages; + + const matches: Array<{ id: string; timestamp: string; participant: string; channelId: string | null; snippet: string }> = []; + let scanned = 0; + for (const msg of candidates) { + scanned++; + const text = flattenContent(msg.content); + const hit = matcher ? matchRegex(matcher, text) : matchSubstring(text, needle, caseSensitive); + if (!hit) continue; + matches.push({ + id: String(msg.id), + timestamp: msg.timestamp.toISOString(), + participant: msg.participant, + channelId: getChannelId(msg) ?? null, + snippet: snippetAround(text, hit.index, hit.length), + }); + if (matches.length >= limit) break; + } + + return { + success: true, + data: { + scanned, + candidatePoolSize: candidates.length, + truncated, + matches, + }, + }; + } +} + +// ============================================================================ +// Helpers +// ============================================================================ + +/** Parse an ISO 8601 date string to Unix ms. Throws a clear error (not a + * crash) on an unparseable string; returns undefined for an omitted field + * (open-ended bound). */ +function parseIsoDate(value: string | undefined, field: string): number | undefined { + if (value === undefined) return undefined; + const ms = Date.parse(value); + if (Number.isNaN(ms)) { + throw new Error(`Invalid ISO 8601 date for "${field}": ${JSON.stringify(value)}`); + } + return ms; +} + +/** channelId lives at metadata.external.channelId — same path context-manager's + * own native index reads from (message-store.ts's CHANNEL_FIELD). Metadata's + * index-signature type means this needs an explicit narrow, same as + * MessageStore.getChannelTokenStats does internally. */ +function getChannelId(msg: StoredMessage): string | undefined { + const external = msg.metadata?.external as { channelId?: string } | undefined; + return external?.channelId; +} + +function projectMessage(msg: StoredMessage, format: 'text' | 'raw'): Record { + return { + id: String(msg.id), + timestamp: msg.timestamp.toISOString(), + participant: msg.participant, + channelId: getChannelId(msg) ?? null, + content: format === 'raw' ? msg.content : flattenContent(msg.content), + }; +} + +/** Flatten a message's content blocks to one short, readable string. Text + * blocks verbatim; everything else (tool_use/tool_result/thinking/media) as + * a short bracketed label — this is for agent readability and search + * matching, not byte-faithful reconstruction. */ +function flattenContent(content: ContentBlock[]): string { + return content.map(blockLabel).join(' ').trim(); +} + +function blockLabel(block: ContentBlock): string { + switch (block.type) { + case 'text': + return block.text; + case 'tool_use': + return `[tool_use: ${block.name}]`; + case 'tool_result': + return block.toolName ? `[tool_result: ${block.toolName}]` : '[tool_result]'; + case 'thinking': + case 'redacted_thinking': + return '[thinking]'; + case 'image': + return '[image]'; + case 'generated_image': + return '[image]'; + case 'document': + return '[document]'; + case 'audio': + return '[audio]'; + case 'video': + return '[video]'; + default: + // Exhaustiveness guard: a future ContentBlock variant falls back to a + // generic label instead of a compile error at a call site far from here. + return `[${(block as { type: string }).type}]`; + } +} + +interface MatchHit { + index: number; + length: number; +} + +function matchSubstring(text: string, needle: string, caseSensitive: boolean): MatchHit | null { + const haystack = caseSensitive ? text : text.toLowerCase(); + const index = haystack.indexOf(needle); + if (index === -1) return null; + return { index, length: needle.length }; +} + +function matchRegex(matcher: RegExp, text: string): MatchHit | null { + const m = matcher.exec(text); + if (!m) return null; + return { index: m.index, length: m[0].length }; +} + +/** ~SNIPPET_CONTEXT_CHARS of surrounding context on each side of a match, or + * the first ~SNIPPET_FALLBACK_CHARS of content when no meaningful match + * position is given (kept simple on purpose — this is a readability aid, + * not a highlighting engine). */ +function snippetAround(text: string, index: number, length: number): string { + if (index < 0) return text.slice(0, SNIPPET_FALLBACK_CHARS); + const start = Math.max(0, index - SNIPPET_CONTEXT_CHARS); + const end = Math.min(text.length, index + length + SNIPPET_CONTEXT_CHARS); + const prefix = start > 0 ? '…' : ''; + const suffix = end < text.length ? '…' : ''; + return prefix + text.slice(start, end) + suffix; +} diff --git a/src/modules/index.ts b/src/modules/index.ts index 4d8aedbb..12059676 100644 --- a/src/modules/index.ts +++ b/src/modules/index.ts @@ -8,6 +8,8 @@ export type { ApiEvent } from './api/index.js'; export { HealthModule } from './health/index.js'; export type { HealthModuleConfig } from './health/index.js'; +export { HistoryModule } from './history/index.js'; + export { WorkspaceModule, WorkspaceReadError } from './workspace/index.js'; export type { WorkspaceReadErrorCode, WorkspaceReadStage, WorkspaceDiskReadResult, ReadFileFromDiskOptions } from './workspace/index.js'; export type { diff --git a/test/history-module.test.ts b/test/history-module.test.ts new file mode 100644 index 00000000..0f815198 --- /dev/null +++ b/test/history-module.test.ts @@ -0,0 +1,355 @@ +import { describe, it } from 'node:test'; +import assert from 'node:assert/strict'; +import { HistoryModule } from '../src/modules/history/index.js'; +import type { ContextManager, StoredMessage, IndexedMessageQueryResult, ChannelCount, ChannelTokenStats } from '@animalabs/context-manager'; +import type { ContentBlock } from '@animalabs/membrane'; +import type { ToolCall } from '../src/types/events.js'; + +/** + * HistoryModule's own logic is dispatch (always via + * queryMessagesByTimeAndChannel, per context-manager's own test evidence + * that it already handles "only one filter" and "neither filter"), input + * validation (ISO dates, regex, limit capping), and presentation (content + * flattening, snippet extraction, truncation detection). The underlying + * index-query *correctness* is context-manager's own responsibility and is + * covered by its test suite (test/message-store-history-index.test.ts) — so + * this stub implements realistic-but-simple in-memory filtering over a small + * fixture rather than standing up a real chronicle store, and a `calls` log + * lets tests assert on exactly what HistoryModule asked for (e.g. the capped + * limit). + */ + +function textBlock(text: string): ContentBlock { + return { type: 'text', text }; +} + +function metaFor(channelId?: string): Record | undefined { + return channelId ? { external: { channelId } } : undefined; +} + +function msg(id: string, ms: number, participant: string, content: ContentBlock[], channelId?: string): StoredMessage { + return { + id, + sequence: Number(id.replace(/\D/g, '')), + participant, + content, + metadata: metaFor(channelId), + timestamp: new Date(ms), + } as StoredMessage; +} + +function channelIdOf(m: StoredMessage): string | undefined { + return (m.metadata as { external?: { channelId?: string } } | undefined)?.external?.channelId; +} + +/** Trivial, deterministic stand-in for real token estimation — tests only + * assert on shape/filtering, never on the exact number. */ +function estimate(m: StoredMessage): number { + return m.content.reduce((n, b) => n + (b.type === 'text' ? b.text.length : 4), 0); +} + +interface StubOptions { + /** When set, every query/stats method throws this — simulates a chronicle + * build that predates the native index-query capability. */ + throwUnsupported?: boolean; +} + +function buildStub(messages: StoredMessage[], opts: StubOptions = {}): { cm: ContextManager; calls: Array<{ method: string; args: unknown }> } { + const calls: Array<{ method: string; args: unknown }> = []; + const unsupported = () => { + throw new Error('Chronicle history index unsupported: native field-index capability absent on this chronicle build.'); + }; + + const cm = { + queryMessagesByTimeAndChannel(args: { fromMs?: number; toMs?: number; channelId?: string; limit?: number; offset?: number }): IndexedMessageQueryResult { + calls.push({ method: 'queryMessagesByTimeAndChannel', args }); + if (opts.throwUnsupported) return unsupported(); + const filtered = messages.filter((m) => { + const ts = m.timestamp.getTime(); + if (args.fromMs !== undefined && ts < args.fromMs) return false; + if (args.toMs !== undefined && ts > args.toMs) return false; + if (args.channelId !== undefined && channelIdOf(m) !== args.channelId) return false; + return true; + }); + const matchedCount = filtered.length; + const offset = args.offset ?? 0; + const limit = args.limit ?? filtered.length; + return { messages: filtered.slice(offset, offset + limit), matchedCount }; + }, + getChannelMessageCounts(): ChannelCount[] { + calls.push({ method: 'getChannelMessageCounts', args: undefined }); + if (opts.throwUnsupported) return unsupported(); + const counts = new Map(); + for (const m of messages) { + const c = channelIdOf(m); + if (c) counts.set(c, (counts.get(c) ?? 0) + 1); + } + return [...counts.entries()].map(([channelId, count]) => ({ channelId, messages: count })); + }, + getChannelTokenStats(args?: { fromMs?: number; toMs?: number }): ChannelTokenStats { + calls.push({ method: 'getChannelTokenStats', args }); + if (opts.throwUnsupported) return unsupported(); + const inRange = messages.filter((m) => { + const ts = m.timestamp.getTime(); + if (args?.fromMs !== undefined && ts < args.fromMs) return false; + if (args?.toMs !== undefined && ts > args.toMs) return false; + return true; + }); + const byChannel = new Map(); + let totalTokensEstimate = 0; + for (const m of inRange) { + const est = estimate(m); + totalTokensEstimate += est; + const c = channelIdOf(m); + if (c) { + const agg = byChannel.get(c) ?? { messages: 0, tokensEstimate: 0 }; + agg.messages++; + agg.tokensEstimate += est; + byChannel.set(c, agg); + } + } + return { + totalMessages: inRange.length, + totalTokensEstimate, + byChannel: [...byChannel.entries()].map(([channelId, agg]) => ({ channelId, ...agg })), + }; + }, + } as unknown as ContextManager; + + return { cm, calls }; +} + +function call(name: string, input: Record): ToolCall { + return { id: `call-${name}`, name, input } as unknown as ToolCall; +} + +// Fixture: two channels (c1, c2), one channel-less message, spread across +// distinct timestamps so time-range filtering is exercisable. +const FIXTURE: StoredMessage[] = [ + msg('m1', 1000, 'User', [textBlock('hello world')], 'c1'), + msg('m2', 2000, 'Claude', [{ type: 'tool_use', id: 't1', name: 'search', input: {} }], 'c1'), + msg('m3', 3000, 'User', [textBlock('foo BAR baz')], 'c2'), + msg('m4', 4000, 'Claude', [{ type: 'thinking', thinking: 'hmm' }, textBlock('final answer')], 'c2'), + msg('m5', 5000, 'User', [textBlock('unchanneled message')]), + msg('m6', 6000, 'User', [textBlock('another c1 message with searchterm inside')], 'c1'), +]; + +describe('HistoryModule', () => { + describe('stats', () => { + it('reports all-time message counts and range-scoped token stats separately', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + // Range covers only m3..m5 (3000-5000), but messageCountsAllTime must + // still reflect the whole store. + const result = await h.handleToolCall(call('stats', { from: new Date(3000).toISOString(), to: new Date(5000).toISOString() })); + assert.equal(result.success, true, result.error); + const data = result.data as { + messageCountsAllTime: ChannelCount[]; + tokenStatsForRange: ChannelTokenStats; + }; + + const allTimeByChannel = new Map(data.messageCountsAllTime.map((c) => [c.channelId, c.messages])); + assert.equal(allTimeByChannel.get('c1'), 3); // m1, m2, m6 — unaffected by the range + assert.equal(allTimeByChannel.get('c2'), 2); // m3, m4 + + assert.equal(data.tokenStatsForRange.totalMessages, 3); // m3, m4, m5 + const rangeByChannel = new Map(data.tokenStatsForRange.byChannel.map((c) => [c.channelId, c.messages])); + assert.equal(rangeByChannel.get('c2'), 2); // m3 and m4 both fall in [3000,5000] + assert.equal(rangeByChannel.has('c1'), false); // no c1 message falls in [3000,5000] + }); + + it('filters both parts of the response to one channelId when given', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('stats', { channelId: 'c1' })); + assert.equal(result.success, true, result.error); + const data = result.data as { messageCountsAllTime: ChannelCount[]; tokenStatsForRange: ChannelTokenStats }; + assert.deepEqual(data.messageCountsAllTime.map((c) => c.channelId), ['c1']); + assert.deepEqual(data.tokenStatsForRange.byChannel.map((c) => c.channelId), ['c1']); + }); + + it('rejects an invalid ISO date cleanly instead of throwing', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('stats', { from: 'not-a-date' })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Invalid ISO 8601 date/); + }); + + it('surfaces the capability-absent error as a clean tool error, not a crash', async () => { + const { cm } = buildStub(FIXTURE, { throwUnsupported: true }); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('stats', {})); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Chronicle history index unsupported/); + }); + }); + + describe('extract', () => { + it('filters by channel and date range', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall( + call('extract', { channelId: 'c1', from: new Date(1500).toISOString() }), + ); + assert.equal(result.success, true, result.error); + const data = result.data as { matchedCount: number; returned: number; messages: Array<{ id: string }> }; + // c1 messages at ts >= 1500: m2 (2000), m6 (6000) — m1 (1000) excluded. + assert.deepEqual(data.messages.map((m) => m.id), ['m2', 'm6']); + assert.equal(data.returned, 2); + assert.equal(data.matchedCount, 2); + }); + + it('format:"text" flattens content; format:"raw" returns blocks as-is', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const textResult = await h.handleToolCall(call('extract', { channelId: 'c2', format: 'text' })); + const textData = textResult.data as { messages: Array<{ id: string; content: unknown }> }; + const m4 = textData.messages.find((m) => m.id === 'm4')!; + assert.equal(m4.content, '[thinking] final answer'); + + const rawResult = await h.handleToolCall(call('extract', { channelId: 'c2', format: 'raw' })); + const rawData = rawResult.data as { messages: Array<{ id: string; content: unknown }> }; + const m4Raw = rawData.messages.find((m) => m.id === 'm4')!; + assert.ok(Array.isArray(m4Raw.content)); + assert.equal((m4Raw.content as ContentBlock[]).length, 2); + assert.equal((m4Raw.content as ContentBlock[])[0]?.type, 'thinking'); + }); + + it('caps limit at the hard maximum before it reaches context-manager', async () => { + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + await h.handleToolCall(call('extract', { limit: 999999 })); + const queryCall = calls.find((c) => c.method === 'queryMessagesByTimeAndChannel'); + assert.equal((queryCall?.args as { limit?: number }).limit, 200); + }); + + it('surfaces the capability-absent error cleanly', async () => { + const { cm } = buildStub(FIXTURE, { throwUnsupported: true }); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('extract', {})); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Chronicle history index unsupported/); + }); + }); + + describe('search', () => { + it('matches a case-insensitive substring by default', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'bar' })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: Array<{ id: string; snippet: string }> }; + assert.deepEqual(data.matches.map((m) => m.id), ['m3']); + assert.match(data.matches[0]!.snippet, /BAR/); + }); + + it('caseSensitive:true respects case', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'bar', caseSensitive: true })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: unknown[] }; + assert.equal(data.matches.length, 0); // fixture only has "BAR", not "bar" + }); + + it('matches a regex', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: '\\bworld\\b', regex: true })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: Array<{ id: string }> }; + assert.deepEqual(data.matches.map((m) => m.id), ['m1']); + }); + + it('returns a clean error for an invalid regex instead of throwing', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: '(unclosed', regex: true })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Invalid regex/); + }); + + it('reports truncated:true up front when the candidate window exceeds maxScan', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'm', maxScan: 3 })); + assert.equal(result.success, true, result.error); + const data = result.data as { truncated: boolean; candidatePoolSize: number; scanned: number }; + assert.equal(data.truncated, true); + assert.equal(data.candidatePoolSize, 3); + assert.ok(data.scanned <= 3); + }); + + it('does not report truncated when the candidate window fits within maxScan', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'searchterm', maxScan: 5000 })); + assert.equal(result.success, true, result.error); + const data = result.data as { truncated: boolean; matches: Array<{ id: string }> }; + assert.equal(data.truncated, false); + assert.deepEqual(data.matches.map((m) => m.id), ['m6']); + }); + + it('surfaces the capability-absent error cleanly', async () => { + const { cm } = buildStub(FIXTURE, { throwUnsupported: true }); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'x' })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Chronicle history index unsupported/); + }); + }); + + it('rejects tool calls before bind()', async () => { + const h = new HistoryModule(); + const result = await h.handleToolCall(call('stats', {})); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /not bound/); + }); + + it('rejects an unknown tool name', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('bogus', {})); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Unknown tool/); + }); +}); From 9ba878c4d22f19cd29d27f47825274e659d42bb8 Mon Sep 17 00:00:00 2001 From: antra-tess Date: Wed, 16 Sep 2026 14:41:19 -0700 Subject: [PATCH 2/8] changelog: add fragment for HistoryModule (#151) Co-Authored-By: Claude Sonnet 5 --- changelog.d/151-history-module.added.md | 5 +++++ 1 file changed, 5 insertions(+) create mode 100644 changelog.d/151-history-module.added.md diff --git a/changelog.d/151-history-module.added.md b/changelog.d/151-history-module.added.md new file mode 100644 index 00000000..17616cc4 --- /dev/null +++ b/changelog.d/151-history-module.added.md @@ -0,0 +1,5 @@ +- `HistoryModule` adds `stats`/`extract`/`search` tools for querying an + agent's full uncompressed message history by time range and/or channel + (#151), backed by context-manager's native chronicle secondary-index + queries — O(log n + k) against multi-million-message stores, not a full + scan. Read-only, `bind(contextManager)` after `AgentFramework.create()`. From 9485f16d2fb00a0984035a6014f2025fa3afc51d Mon Sep 17 00:00:00 2001 From: antra-tess Date: Thu, 17 Sep 2026 08:40:13 -0700 Subject: [PATCH 3/8] chore(deps): bump @animalabs/context-manager to ^0.9.0, @animalabs/chronicle to ^0.4.0 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit context-manager 0.9.0 published (anima-research/context-manager#99) — ships the queryMessagesByTime/queryMessagesByChannel/etc. this module's HistoryModule depends on. Also bumps this package's own direct chronicle dependency to ^0.4.0: leaving it at ^0.3.0 while context-manager pulled in 0.4.0 produced two separate installed chronicle copies (npm can't dedupe across an unsatisfied range), which made TypeScript see two structurally-identical-but-distinct JsStore types and fail to compile anywhere this package's own code touches a JsStore. Bumping both to ^0.4.0 lets npm dedupe to one copy. Unblocks CI: tsc --noEmit was failing outright against the old published context-manager (missing exports); now clean, and the full suite passes (677/681, 4 pre-existing unrelated skips) against the real published dependency chain. Co-Authored-By: Claude Sonnet 5 --- package.json | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/package.json b/package.json index d5945c3e..b1a0f959 100644 --- a/package.json +++ b/package.json @@ -46,8 +46,8 @@ "author": "Anima Research", "license": "MIT", "dependencies": { - "@animalabs/chronicle": "^0.3.0", - "@animalabs/context-manager": "^0.8.0", + "@animalabs/chronicle": "^0.4.0", + "@animalabs/context-manager": "^0.9.0", "@animalabs/membrane": "^0.5.78", "chokidar": "^4.0.3", "discord.js": "^14.25.1", From 80f1c8ce643a863bc719f579ed87167643ca008e Mon Sep 17 00:00:00 2001 From: antra-tess Date: Thu, 17 Sep 2026 08:59:37 -0700 Subject: [PATCH 4/8] fix(history): worker-sandbox regex search, reject negative pagination MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two review findings on #151's HistoryModule (head 9485f16): - [P1] search's regex:true path ran the caller's pattern via a synchronous RegExp.exec() on the main thread — a catastrophically-backtracking pattern (e.g. (a+)+$) blocks the framework's single-threaded event loop for every agent, not just the calling search. A Promise.race/setTimeout can't interrupt this (nothing on the same thread can preempt synchronous JS), so regex matching now runs on a one-shot node:worker_threads worker (src/modules/history/search-regex-worker.ts) with a hard wall-clock deadline; a stuck worker is forcibly terminate()'d and the call surfaces as a clean tool error, never a silently-empty result. Follows the existing gate-script.ts/gate-script-worker.ts precedent for sandboxing script-shaped execution in this repo, simplified to a one-shot request/response (handleSearch was already async, so no need for gate-script's Atomics.wait synchronous-emulation machinery). Plain substring search is unaffected (linear, in-process, no ReDoS risk) — only regex mode pays the worker-spawn cost. - [P2] extract/search capped limit/offset/maxScan with bare Math.min(value, max), which only enforces an upper bound — Math.min(-1, 200) is -1. A negative limit/offset/maxScan reached context-manager's native pagination calls unbounded (repro: extract({limit:-1}) against a 250-message store returned all 250). Replaced with clampCount(), which rejects any non-finite/non-integer/ negative value as a clean tool error before any native call is made. Adds regression coverage: the reviewer's ReDoS repro (asserts the call returns in bounded time with a clear failure, not a hung test or an empty success), a control regex to prove the worker path still matches correctly, and 18 negative/non-integer/NaN/Infinity cases across extract's limit/offset and search's limit/maxScan, including the reviewer's exact 250-message/248-candidate repro shapes. Co-Authored-By: Claude Sonnet 5 --- src/modules/history/index.ts | 212 +++++++++++++++++---- src/modules/history/search-regex-worker.ts | 72 +++++++ test/history-module.test.ts | 119 ++++++++++++ 3 files changed, 367 insertions(+), 36 deletions(-) create mode 100644 src/modules/history/search-regex-worker.ts diff --git a/src/modules/history/index.ts b/src/modules/history/index.ts index 0c0f0d8b..224c55bb 100644 --- a/src/modules/history/index.ts +++ b/src/modules/history/index.ts @@ -27,11 +27,15 @@ * module doesn't need to re-decide which of the three query methods to call. */ +import { Worker } from 'node:worker_threads'; +import { fileURLToPath } from 'node:url'; +import { dirname, join } from 'node:path'; import type { ContextManager, StoredMessage, ChannelCount, ChannelTokenStats } from '@animalabs/context-manager'; import type { ContentBlock } from '@animalabs/membrane'; import type { Module, ModuleContext } from '../../types/module.js'; import type { ToolDefinition, ToolCall, ToolResult, ProcessEvent } from '../../types/events.js'; import type { EventResponse, ProcessState } from '../../types/module.js'; +import type { SearchWorkerMessage, SearchWorkerMatch } from './search-regex-worker.js'; // ============================================================================ // Tool input shapes @@ -80,6 +84,22 @@ const SNIPPET_CONTEXT_CHARS = 80; /** Fallback snippet length when a match position isn't meaningful to center on. */ const SNIPPET_FALLBACK_CHARS = 160; +/** + * Wall-clock deadline for a single regex-mode search call, enforced by + * forcibly terminating the worker thread doing the matching (see + * search-regex-worker.ts's header for why a worker — not a + * Promise.race/setTimeout — is required to actually interrupt a stuck + * synchronous RegExp.exec()). Generous enough for any legitimate pattern + * against a few thousand short strings; short enough that a catastrophic + * pattern doesn't tie up a worker (or an agent's turn) for long. + */ +const SEARCH_REGEX_TIMEOUT_MS = 2000; + +/** Compiled sibling of search-regex-worker.ts — resolved at runtime the same + * way gate-script.ts locates gate-script-worker.js, so it tracks whatever + * directory this module's own compiled output lives in. */ +const SEARCH_WORKER_PATH = join(dirname(fileURLToPath(import.meta.url)), 'search-regex-worker.js'); + export class HistoryModule implements Module { readonly name = 'history'; @@ -137,8 +157,8 @@ export class HistoryModule implements Module { from: { type: 'string', description: 'ISO 8601 inclusive lower bound. Omit for open-ended.' }, to: { type: 'string', description: 'ISO 8601 inclusive upper bound. Omit for open-ended.' }, channelId: { type: 'string', description: 'Restrict to one channel.' }, - limit: { type: 'number', description: `Max messages to return (default ${EXTRACT_DEFAULT_LIMIT}, hard cap ${EXTRACT_MAX_LIMIT}).` }, - offset: { type: 'number', description: 'Number of matching messages to skip (default 0).' }, + limit: { type: 'number', description: `Max messages to return (default ${EXTRACT_DEFAULT_LIMIT}, hard cap ${EXTRACT_MAX_LIMIT}). Must be a non-negative integer.` }, + offset: { type: 'number', description: 'Number of matching messages to skip (default 0). Must be a non-negative integer.' }, format: { type: 'string', enum: ['text', 'raw'], description: 'Content rendering (default "text").' }, }, }, @@ -150,7 +170,8 @@ export class HistoryModule implements Module { 'range/channel window. Narrows candidates via the same query as `extract` before matching, up ' + 'to maxScan candidates — if the narrowed window is larger than maxScan, the response reports ' + 'truncated:true up front rather than silently missing later matches; narrow the filter or raise ' + - 'maxScan and retry.', + 'maxScan and retry. regex:true matching runs under a wall-clock deadline and is cleanly failed ' + + `(not silently empty) if a pattern is too slow — avoid nested-quantifier patterns like (a+)+.`, inputSchema: { type: 'object' as const, properties: { @@ -160,8 +181,8 @@ export class HistoryModule implements Module { from: { type: 'string', description: 'ISO 8601 inclusive lower bound. Omit for open-ended.' }, to: { type: 'string', description: 'ISO 8601 inclusive upper bound. Omit for open-ended.' }, channelId: { type: 'string', description: 'Restrict to one channel.' }, - limit: { type: 'number', description: `Max matches to return (default ${SEARCH_DEFAULT_LIMIT}, hard cap ${SEARCH_MAX_LIMIT}).` }, - maxScan: { type: 'number', description: `Max candidate messages to scan (default ${SEARCH_DEFAULT_MAX_SCAN}, hard cap ${SEARCH_MAX_MAX_SCAN}).` }, + limit: { type: 'number', description: `Max matches to return (default ${SEARCH_DEFAULT_LIMIT}, hard cap ${SEARCH_MAX_LIMIT}). Must be a non-negative integer.` }, + maxScan: { type: 'number', description: `Max candidate messages to scan (default ${SEARCH_DEFAULT_MAX_SCAN}, hard cap ${SEARCH_MAX_MAX_SCAN}). Must be a non-negative integer.` }, }, required: ['query'], }, @@ -180,18 +201,22 @@ export class HistoryModule implements Module { case 'extract': return this.handleExtract((call.input ?? {}) as ExtractInput); case 'search': - return this.handleSearch((call.input ?? {}) as SearchInput); + return await this.handleSearch((call.input ?? {}) as SearchInput); default: return { success: false, isError: true, error: `Unknown tool: ${call.name}` }; } } catch (error) { - // Catches both our own validation errors (bad ISO date, invalid regex, - // unbound module) and context-manager's capability-absent error - // ("Chronicle history index unsupported...", thrown by - // queryMessagesByTime/queryMessagesByChannel/queryMessagesByTimeAndChannel/ - // getChannelMessageCounts/getChannelTokenStats on a chronicle build that - // predates the native index-query capability) — surfaced as a normal - // tool error rather than crashing the module. + // Catches our own validation errors (bad ISO date, invalid regex, + // out-of-range limit/offset/maxScan, unbound module), a regex-search + // worker timeout or worker-side error (see handleSearch/ + // searchWithRegexWorker — a ReDoS-shaped pattern surfaces here as a + // clean timeout error, never a hang), and context-manager's + // capability-absent error ("Chronicle history index unsupported...", + // thrown by queryMessagesByTime/queryMessagesByChannel/ + // queryMessagesByTimeAndChannel/getChannelMessageCounts/ + // getChannelTokenStats on a chronicle build that predates the native + // index-query capability) — all surfaced as a normal tool error + // rather than crashing the module. return { success: false, isError: true, @@ -247,8 +272,8 @@ export class HistoryModule implements Module { private handleExtract(input: ExtractInput): ToolResult { const fromMs = parseIsoDate(input.from, 'from'); const toMs = parseIsoDate(input.to, 'to'); - const limit = Math.min(input.limit ?? EXTRACT_DEFAULT_LIMIT, EXTRACT_MAX_LIMIT); - const offset = input.offset ?? 0; + const limit = clampCount(input.limit, EXTRACT_DEFAULT_LIMIT, EXTRACT_MAX_LIMIT, 'limit'); + const offset = clampCount(input.offset, 0, Number.MAX_SAFE_INTEGER, 'offset'); const format = input.format ?? 'text'; const cm = this.cm as ContextManager; @@ -274,22 +299,25 @@ export class HistoryModule implements Module { // search // ========================================================================== - private handleSearch(input: SearchInput): ToolResult { + private async handleSearch(input: SearchInput): Promise { if (!input.query) { throw new Error('search requires a non-empty "query".'); } const fromMs = parseIsoDate(input.from, 'from'); const toMs = parseIsoDate(input.to, 'to'); - const limit = Math.min(input.limit ?? SEARCH_DEFAULT_LIMIT, SEARCH_MAX_LIMIT); - const maxScan = Math.min(input.maxScan ?? SEARCH_DEFAULT_MAX_SCAN, SEARCH_MAX_MAX_SCAN); + const limit = clampCount(input.limit, SEARCH_DEFAULT_LIMIT, SEARCH_MAX_LIMIT, 'limit'); + const maxScan = clampCount(input.maxScan, SEARCH_DEFAULT_MAX_SCAN, SEARCH_MAX_MAX_SCAN, 'maxScan'); const caseSensitive = input.caseSensitive ?? false; + const flags = caseSensitive ? '' : 'i'; - // Compile the matcher up front — an invalid regex is a clean tool error, - // not a crash mid-scan. - let matcher: RegExp | null = null; + // Validate regex SYNTAX up front — an invalid pattern is a clean tool + // error, not a crash mid-scan. This does NOT bound match TIME (a + // syntactically valid pattern can still backtrack catastrophically), + // which is why regex-mode matching itself runs on a worker below rather + // than here. if (input.regex) { try { - matcher = new RegExp(input.query, caseSensitive ? '' : 'i'); + new RegExp(input.query, flags); } catch (error) { throw new Error(`Invalid regex "${input.query}": ${error instanceof Error ? error.message : String(error)}`); } @@ -311,12 +339,30 @@ export class HistoryModule implements Module { const truncated = probe.messages.length > maxScan; const candidates = truncated ? probe.messages.slice(0, maxScan) : probe.messages; - const matches: Array<{ id: string; timestamp: string; participant: string; channelId: string | null; snippet: string }> = []; + if (input.regex) { + // Regex matching against caller-supplied patterns is ReDoS-shaped: a + // pathological pattern (e.g. `(a+)+$`) can take catastrophically long + // against one candidate string, and a synchronous RegExp.exec() on + // this thread would block the WHOLE framework's event loop — every + // agent's turns, health checks, timers — for as long as it runs, with + // no way to interrupt it from this same thread. Route matching + // through a worker thread instead, which can be forcibly terminated + // on a deadline. See search-regex-worker.ts's header for the full + // rationale. + const { matches, scanned } = await this.searchWithRegexWorker(candidates, input.query, flags, limit); + return { success: true, data: { scanned, candidatePoolSize: candidates.length, truncated, matches } }; + } + + // Plain substring search (String.prototype.indexOf) is inherently + // linear in input length — no ReDoS-equivalent risk — so it stays + // in-process. This is also the common case, so it pays no worker-spawn + // overhead. + const matches: SearchMatch[] = []; let scanned = 0; for (const msg of candidates) { scanned++; const text = flattenContent(msg.content); - const hit = matcher ? matchRegex(matcher, text) : matchSubstring(text, needle, caseSensitive); + const hit = matchSubstring(text, needle, caseSensitive); if (!hit) continue; matches.push({ id: String(msg.id), @@ -330,14 +376,78 @@ export class HistoryModule implements Module { return { success: true, - data: { - scanned, - candidatePoolSize: candidates.length, - truncated, - matches, - }, + data: { scanned, candidatePoolSize: candidates.length, truncated, matches }, }; } + + /** + * Run regex matching for `search` on a worker thread with a hard + * wall-clock deadline, so a catastrophically-backtracking pattern can be + * forcibly killed instead of hanging the framework. One worker per call + * (not per candidate — spawn overhead would dominate at scale; not a + * persistent pool — a fresh worker per call means one bad pattern can + * never contaminate a later search). Always terminated on the way out, + * success or failure, so nothing lingers. + * + * On timeout or a worker-side error this THROWS (caught by + * handleToolCall's try/catch, same as every other error path in this + * module) rather than returning an empty match list — a timed-out search + * must never be indistinguishable from a clean "no matches" result. + */ + private async searchWithRegexWorker( + candidates: StoredMessage[], + pattern: string, + flags: string, + limit: number, + ): Promise<{ matches: SearchMatch[]; scanned: number }> { + const texts = candidates.map((msg) => flattenContent(msg.content)); + let worker: Worker | undefined; + try { + const { matches: rawMatches, scanned } = await new Promise<{ matches: SearchWorkerMatch[]; scanned: number }>( + (resolve, reject) => { + worker = new Worker(SEARCH_WORKER_PATH, { workerData: { texts, pattern, flags, limit } }); + const timer = setTimeout(() => { + reject( + new Error( + `search timed out after ${SEARCH_REGEX_TIMEOUT_MS}ms while matching regex "${pattern}" against ` + + `up to ${texts.length} candidate(s) — the pattern may be catastrophically slow (exponential ` + + 'backtracking) against this data; try a simpler pattern, a literal substring search ' + + '(regex:false), or a smaller maxScan.', + ), + ); + }, SEARCH_REGEX_TIMEOUT_MS); + timer.unref?.(); + worker.once('message', (msg: SearchWorkerMessage) => { + clearTimeout(timer); + if (msg.type === 'error') reject(new Error(msg.error)); + else resolve({ matches: msg.matches, scanned: msg.scanned }); + }); + worker.once('error', (err) => { + clearTimeout(timer); + reject(err); + }); + }, + ); + + const matches: SearchMatch[] = rawMatches.map(({ candidateIndex, matchIndex, matchLength }) => { + const msg = candidates[candidateIndex]!; + const text = texts[candidateIndex]!; + return { + id: String(msg.id), + timestamp: msg.timestamp.toISOString(), + participant: msg.participant, + channelId: getChannelId(msg) ?? null, + snippet: snippetAround(text, matchIndex, matchLength), + }; + }); + return { matches, scanned }; + } finally { + // Always kill the worker — whether it finished, errored, or is still + // stuck mid-backtrack when the deadline hit. terminate() on an + // already-exited worker is a harmless no-op. + if (worker) void worker.terminate().catch(() => {}); + } + } } // ============================================================================ @@ -356,6 +466,32 @@ function parseIsoDate(value: string | undefined, field: string): number | undefi return ms; } +/** + * Validate a pagination-ish numeric input (limit/offset/maxScan) and clamp + * it to an upper bound. `Math.min(value, max)` alone is NOT sufficient here: + * it only enforces an upper bound, so `Math.min(-1, 200)` is `-1`, not a + * sane value — a negative limit/offset/maxScan would sail straight past the + * "hard cap" and reach a native chronicle call expecting an unsigned + * pagination argument (observed: `extract({limit:-1})` returning every + * message in the store instead of being capped). Rejects (rather than + * silently clamping) anything that isn't a finite non-negative integer, so + * a caller mistake is surfaced as a clean tool error instead of silently + * doing something other than what was asked. + */ +function clampCount(value: number | undefined, def: number, max: number, field: string): number { + if (value === undefined) return def; + if (typeof value !== 'number' || !Number.isFinite(value)) { + throw new Error(`"${field}" must be a finite number, got ${JSON.stringify(value)}.`); + } + if (!Number.isInteger(value)) { + throw new Error(`"${field}" must be an integer, got ${value}.`); + } + if (value < 0) { + throw new Error(`"${field}" must be >= 0, got ${value}.`); + } + return Math.min(value, max); +} + /** channelId lives at metadata.external.channelId — same path context-manager's * own native index reads from (message-store.ts's CHANNEL_FIELD). Metadata's * index-signature type means this needs an explicit narrow, same as @@ -416,6 +552,16 @@ interface MatchHit { length: number; } +/** One `search` result entry — shared shape between the in-process substring + * path and the worker-backed regex path. */ +interface SearchMatch { + id: string; + timestamp: string; + participant: string; + channelId: string | null; + snippet: string; +} + function matchSubstring(text: string, needle: string, caseSensitive: boolean): MatchHit | null { const haystack = caseSensitive ? text : text.toLowerCase(); const index = haystack.indexOf(needle); @@ -423,12 +569,6 @@ function matchSubstring(text: string, needle: string, caseSensitive: boolean): M return { index, length: needle.length }; } -function matchRegex(matcher: RegExp, text: string): MatchHit | null { - const m = matcher.exec(text); - if (!m) return null; - return { index: m.index, length: m[0].length }; -} - /** ~SNIPPET_CONTEXT_CHARS of surrounding context on each side of a match, or * the first ~SNIPPET_FALLBACK_CHARS of content when no meaningful match * position is given (kept simple on purpose — this is a readability aid, diff --git a/src/modules/history/search-regex-worker.ts b/src/modules/history/search-regex-worker.ts new file mode 100644 index 00000000..eb827dd8 --- /dev/null +++ b/src/modules/history/search-regex-worker.ts @@ -0,0 +1,72 @@ +/** + * search-regex-worker — runs a caller-supplied regex against a batch of + * message text on a separate thread, so HistoryModule's `search` tool can + * forcibly terminate a catastrophically-backtracking pattern (ReDoS) + * instead of blocking the framework's single-threaded event loop. + * + * Why a worker and not a Promise.race/setTimeout "timeout": JS is + * single-threaded, so nothing on the SAME thread can interrupt a + * synchronous RegExp.exec() call already in flight — a timeout callback + * racing it can't even fire until the blocking call returns, by which point + * it's too late. A worker thread is real OS-level concurrency: the parent's + * event loop keeps running, and `worker.terminate()` actually kills the + * stuck thread rather than just giving up on waiting for it. + * + * Only regex-mode search routes through this worker. Plain substring search + * (String.prototype.indexOf) is inherently linear in input length and has + * no equivalent risk, so it stays in-process — the common case pays no + * worker-spawn overhead. + * + * One-shot: the whole job (already-flattened text, already syntax-validated + * pattern/flags, match limit) arrives via `workerData` at construction; this + * script runs it to completion and posts exactly one message, then the + * parent always terminates the worker (whether it finished normally or had + * to be killed for running long) — no persistent pool, so a wedged worker + * from one bad pattern can never contaminate a later call. + */ +import { parentPort, workerData } from 'node:worker_threads'; + +export interface SearchWorkerJob { + /** Flattened candidate text, aligned by index with the parent's candidate array. */ + texts: string[]; + pattern: string; + flags: string; + /** Stop scanning once this many matches are found. */ + limit: number; +} + +export interface SearchWorkerMatch { + /** Index into `texts` / the parent's candidate array. */ + candidateIndex: number; + matchIndex: number; + matchLength: number; +} + +export type SearchWorkerMessage = + | { type: 'done'; matches: SearchWorkerMatch[]; scanned: number } + | { type: 'error'; error: string }; + +const { texts, pattern, flags, limit } = workerData as SearchWorkerJob; + +try { + // Re-construct (not re-validate for safety — RegExp syntax is what it is; + // the parent already rejected an uncompilable pattern before ever getting + // here) the matcher on this thread. Compilation succeeding doesn't bound + // MATCHING time — that's exactly the risk this worker exists to contain. + const matcher = new RegExp(pattern, flags); + const matches: SearchWorkerMatch[] = []; + let scanned = 0; + for (let i = 0; i < texts.length; i++) { + scanned++; + const m = matcher.exec(texts[i] ?? ''); + if (m) { + matches.push({ candidateIndex: i, matchIndex: m.index, matchLength: m[0].length }); + if (matches.length >= limit) break; + } + } + const done: SearchWorkerMessage = { type: 'done', matches, scanned }; + parentPort?.postMessage(done); +} catch (e) { + const errMsg: SearchWorkerMessage = { type: 'error', error: e instanceof Error ? e.message : String(e) }; + parentPort?.postMessage(errMsg); +} diff --git a/test/history-module.test.ts b/test/history-module.test.ts index 0f815198..1a267022 100644 --- a/test/history-module.test.ts +++ b/test/history-module.test.ts @@ -287,6 +287,46 @@ describe('HistoryModule', () => { assert.deepEqual(data.matches.map((m) => m.id), ['m1']); }); + it('forcibly terminates a catastrophically-backtracking (ReDoS) regex instead of hanging', async () => { + // Reviewer's repro: 32 'a's followed by a non-matching character, plus + // a classic exponential-backtracking pattern. `(a+)+$` against this + // input backtracks exponentially and (verified separately in an + // isolated child process) has to be force-killed after several + // seconds; the equivalent linear pattern `^a+$` returns instantly. + // This must never block the framework's single event loop — matching + // has to happen off-thread with a real, forcible deadline. + const evil = 'a'.repeat(32) + '!'; + const { cm } = buildStub([msg('r1', 1000, 'User', [textBlock(evil)], 'c1')]); + const h = new HistoryModule(); + h.bind(cm); + + const start = Date.now(); + const result = await h.handleToolCall(call('search', { query: '(a+)+$', regex: true, maxScan: 1 })); + const elapsed = Date.now() - start; + + // Returns well within a bounded time (module deadline is 2s; generous + // slack here for worker spawn/CI jitter) — proves the test runner + // itself never hung waiting on this call. + assert.ok(elapsed < 8000, `search took ${elapsed}ms — should have been forcibly terminated near the deadline`); + // A cut-short match MUST be a clean, unambiguous failure, never + // success:true with an empty matches array — that would be + // indistinguishable from "no matches found", which is a lie. + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /timed out/i); + }); + + it('a normal regex against the same kind of input still matches correctly (worker path is not just failing everything)', async () => { + const { cm } = buildStub([msg('r1', 1000, 'User', [textBlock('aaaa!')], 'c1')]); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: '^a+!$', regex: true })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: Array<{ id: string }> }; + assert.deepEqual(data.matches.map((m) => m.id), ['r1']); + }); + it('returns a clean error for an invalid regex instead of throwing', async () => { const { cm } = buildStub(FIXTURE); const h = new HistoryModule(); @@ -335,6 +375,85 @@ describe('HistoryModule', () => { }); }); + describe('pagination input validation', () => { + // Math.min(value, max) alone only enforces an UPPER bound — + // Math.min(-1, 200) is -1, not clamped up to anything — so a negative + // (or non-integer/non-finite) limit/offset/maxScan would previously + // reach context-manager's native query unbounded. Each of these must + // now be rejected before the native call is ever made. + const badValues: Array<[string, number]> = [ + ['negative', -1], + ['non-integer', 1.5], + ['NaN', NaN], + ['Infinity', Infinity], + ]; + + for (const [label, value] of badValues) { + it(`extract: rejects a ${label} limit before it reaches context-manager`, async () => { + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('extract', { limit: value })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.equal(calls.length, 0, 'must reject before ever calling context-manager'); + }); + + it(`extract: rejects a ${label} offset before it reaches context-manager`, async () => { + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('extract', { offset: value })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.equal(calls.length, 0); + }); + + it(`search: rejects a ${label} limit before it reaches context-manager`, async () => { + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('search', { query: 'm', limit: value })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.equal(calls.length, 0); + }); + + it(`search: rejects a ${label} maxScan before it reaches context-manager`, async () => { + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('search', { query: 'm', maxScan: value })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.equal(calls.length, 0); + }); + } + + it('extract: a real 250-message store with limit:-1 does NOT return everything (reviewer repro)', async () => { + // Reviewer's exact repro shape: a store larger than the hard cap, and + // a negative limit that must not bypass it. + const many = Array.from({ length: 250 }, (_, i) => msg(`p${i}`, 1000 + i, 'User', [textBlock(`m${i}`)], 'c1')); + const { cm } = buildStub(many); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('extract', { limit: -1 })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + }); + + it('search: a real large candidate pool with maxScan:-2 does NOT fetch everything (reviewer repro)', async () => { + const many = Array.from({ length: 248 }, (_, i) => msg(`q${i}`, 1000 + i, 'User', [textBlock(`m${i}`)], 'c1')); + const { cm, calls } = buildStub(many); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('search', { query: 'm', maxScan: -2 })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.equal(calls.length, 0); + }); + }); + it('rejects tool calls before bind()', async () => { const h = new HistoryModule(); const result = await h.handleToolCall(call('stats', {})); From a55b7883314addab65030304b065b9f6c179f354 Mon Sep 17 00:00:00 2001 From: antra-tess Date: Thu, 17 Sep 2026 09:07:56 -0700 Subject: [PATCH 5/8] fix(history): sandbox regex search in a worker, reject negative pagination MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two review findings on the HistoryModule PR: - ReDoS: an arbitrary caller-supplied regex was matched synchronously on the framework's event loop. A catastrophically-backtracking pattern (e.g. (a+)+$ against 32 'a's) blocks not just this tool call but every agent's turns, health checks, and timers, since Node is single-threaded and a Promise timeout cannot interrupt an in-flight synchronous RegExp.exec. Fixed by moving regex-mode matching (only regex mode — substring search is inherently linear and stays in-process) to a one-shot worker thread (search-regex-worker.ts) with a hard wall-clock deadline; the worker is always terminate()'d in a finally block whether it finished, errored, or had to be killed mid-backtrack. Follows this repo's existing gate-script.ts/gate-script-worker.ts precedent for the same class of problem (bounding caller-influenced execution) rather than inventing a new mechanism. - Negative pagination: Math.min(value, max) only enforces an upper bound, so limit:-1/maxScan:-2 reached native chronicle calls unbounded, defeating the advertised hard caps (verified: a 250-message store returned all 250 on limit:-1). Fixed with clampCount(), which rejects any limit/offset/maxScan that isn't a finite non-negative integer before any context-manager call. Also: getChannelId() now checks metadata.channelId before metadata.external.channelId, matching context-manager 0.9.1's dual-schema channel index (bumped below) — a channel-filtered result found via the newer schema no longer displays channelId: null. Bumps @animalabs/context-manager to ^0.9.1 (anima-research/context-manager now indexes and merges both channel-id metadata shapes — 0.9.0 only indexed metadata.external.channelId, which real agent-framework MCPL ingestion never writes, so HistoryModule silently found nothing for real framework history; see that repo's 0.9.1 changelog). Co-Authored-By: Claude Sonnet 5 --- src/modules/history/index.ts | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/src/modules/history/index.ts b/src/modules/history/index.ts index 224c55bb..b1a30b3a 100644 --- a/src/modules/history/index.ts +++ b/src/modules/history/index.ts @@ -496,9 +496,20 @@ function clampCount(value: number | undefined, def: number, max: number, field: * own native index reads from (message-store.ts's CHANNEL_FIELD). Metadata's * index-signature type means this needs an explicit narrow, same as * MessageStore.getChannelTokenStats does internally. */ +/** + * Two channel-id metadata shapes exist in the wild: `metadata.channelId` + * (what agent-framework's real MCPL ingestion — `handleMcplChannelIncoming` + * — actually writes) and `metadata.external.channelId` (an older + * convention). context-manager 0.9.1+ indexes and queries BOTH (see + * MessageStore.extractChannelId there), so a channel-filtered result can + * come from either shape — check the same order here so a displayed + * `channelId` never reads `null` on a message that was, in fact, matched by + * channel. + */ function getChannelId(msg: StoredMessage): string | undefined { - const external = msg.metadata?.external as { channelId?: string } | undefined; - return external?.channelId; + const metadata = msg.metadata as { channelId?: unknown; external?: { channelId?: unknown } } | undefined; + if (typeof metadata?.channelId === 'string') return metadata.channelId; + return typeof metadata?.external?.channelId === 'string' ? metadata.external.channelId : undefined; } function projectMessage(msg: StoredMessage, format: 'text' | 'raw'): Record { From 465c80b1a4fad8ce4500515520a7ef0b7f497253 Mon Sep 17 00:00:00 2001 From: antra-tess Date: Thu, 17 Sep 2026 09:08:08 -0700 Subject: [PATCH 6/8] chore(deps): bump @animalabs/context-manager to ^0.9.1 Co-Authored-By: Claude Sonnet 5 --- package.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/package.json b/package.json index b1a0f959..9ebaa4c7 100644 --- a/package.json +++ b/package.json @@ -47,7 +47,7 @@ "license": "MIT", "dependencies": { "@animalabs/chronicle": "^0.4.0", - "@animalabs/context-manager": "^0.9.0", + "@animalabs/context-manager": "^0.9.1", "@animalabs/membrane": "^0.5.78", "chokidar": "^4.0.3", "discord.js": "^14.25.1", From b0058ef977b8a6292a667d78d9c9e3b50036e5cf Mon Sep 17 00:00:00 2001 From: antra-tess Date: Thu, 17 Sep 2026 09:30:41 -0700 Subject: [PATCH 7/8] fix(history): honor search limit:0, bound offset to native u32 range MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two more small review findings on #151's HistoryModule (head 465c80b): - [P3] search's limit:0 wasn't honored in either matching loop (the in-process substring loop in index.ts, and search-regex-worker.ts's own independent copy of the same loop): both pushed a match onto the results array BEFORE checking matches.length >= limit, so with limit:0 (a value clampCount already accepts — it's >= 0) and at least one matching candidate, both paths still returned exactly 1 match instead of 0. Moved the limit check to before any per-candidate work in both loops. - [P3] extract's offset validation ceiling was Number.MAX_SAFE_INTEGER, but the native chronicle call it reaches takes a u32 offset (max 4294967295). A value between the u32 max and MAX_SAFE_INTEGER passed clampCount unchanged, then wrapped/truncated crossing the N-API boundary — repro: extract({offset: 4294967296, limit: 1}) against a real 1-message store "succeeded" and returned page 0 instead of an empty page or a validation error. Added NATIVE_OFFSET_MAX (0xFFFFFFFF) as the offset ceiling, clamped the same way limit already is (consistent clamp-not-reject behavior). Adds regression coverage: limit:0 in both substring and regex mode (asserts zero matches, not one), and offset:4294967296 asserted clamped to 4294967295 before it ever reaches context-manager. Co-Authored-By: Claude Sonnet 5 --- src/modules/history/index.ts | 25 +++++++++++-- src/modules/history/search-regex-worker.ts | 6 ++- test/history-module.test.ts | 43 ++++++++++++++++++++++ 3 files changed, 70 insertions(+), 4 deletions(-) diff --git a/src/modules/history/index.ts b/src/modules/history/index.ts index b1a30b3a..5119af8d 100644 --- a/src/modules/history/index.ts +++ b/src/modules/history/index.ts @@ -74,6 +74,20 @@ interface SearchInput { const EXTRACT_DEFAULT_LIMIT = 50; const EXTRACT_MAX_LIMIT = 200; +/** + * Real ceiling of the native chronicle pagination `offset` argument (a u32 + * at the N-API boundary). `Number.MAX_SAFE_INTEGER` is NOT a safe ceiling + * for this field: a value between 0xFFFFFFFF and MAX_SAFE_INTEGER passes a + * naive upper-bound check unchanged, then wraps/truncates when it crosses + * into an unsigned 32-bit int on the native side — silently wrong + * pagination (repro: offset:4294967296 against a real store "succeeded" + * and returned page 0) instead of a clean failure. Clamping to this ceiling + * (same clamp-not-reject behavior `limit` already has) means anything + * beyond it lands on an offset no real store's message count will ever + * reach, i.e. a correctly-empty page, rather than wrapping onto a wrong one. + */ +const NATIVE_OFFSET_MAX = 0xffffffff; // 4294967295 + const SEARCH_DEFAULT_LIMIT = 20; const SEARCH_MAX_LIMIT = 100; const SEARCH_DEFAULT_MAX_SCAN = 5000; @@ -158,7 +172,7 @@ export class HistoryModule implements Module { to: { type: 'string', description: 'ISO 8601 inclusive upper bound. Omit for open-ended.' }, channelId: { type: 'string', description: 'Restrict to one channel.' }, limit: { type: 'number', description: `Max messages to return (default ${EXTRACT_DEFAULT_LIMIT}, hard cap ${EXTRACT_MAX_LIMIT}). Must be a non-negative integer.` }, - offset: { type: 'number', description: 'Number of matching messages to skip (default 0). Must be a non-negative integer.' }, + offset: { type: 'number', description: `Number of matching messages to skip (default 0, capped to ${NATIVE_OFFSET_MAX}). Must be a non-negative integer.` }, format: { type: 'string', enum: ['text', 'raw'], description: 'Content rendering (default "text").' }, }, }, @@ -273,7 +287,7 @@ export class HistoryModule implements Module { const fromMs = parseIsoDate(input.from, 'from'); const toMs = parseIsoDate(input.to, 'to'); const limit = clampCount(input.limit, EXTRACT_DEFAULT_LIMIT, EXTRACT_MAX_LIMIT, 'limit'); - const offset = clampCount(input.offset, 0, Number.MAX_SAFE_INTEGER, 'offset'); + const offset = clampCount(input.offset, 0, NATIVE_OFFSET_MAX, 'offset'); const format = input.format ?? 'text'; const cm = this.cm as ContextManager; @@ -360,6 +374,12 @@ export class HistoryModule implements Module { const matches: SearchMatch[] = []; let scanned = 0; for (const msg of candidates) { + // Check the limit BEFORE doing any work for this candidate — not + // after pushing a match — so limit:0 (a valid clampCount value: it's + // >= 0) correctly yields zero matches instead of one. Checking + // post-push would always let through the match that first reaches + // the limit. Same fix mirrored in search-regex-worker.ts's loop. + if (matches.length >= limit) break; scanned++; const text = flattenContent(msg.content); const hit = matchSubstring(text, needle, caseSensitive); @@ -371,7 +391,6 @@ export class HistoryModule implements Module { channelId: getChannelId(msg) ?? null, snippet: snippetAround(text, hit.index, hit.length), }); - if (matches.length >= limit) break; } return { diff --git a/src/modules/history/search-regex-worker.ts b/src/modules/history/search-regex-worker.ts index eb827dd8..48993284 100644 --- a/src/modules/history/search-regex-worker.ts +++ b/src/modules/history/search-regex-worker.ts @@ -57,11 +57,15 @@ try { const matches: SearchWorkerMatch[] = []; let scanned = 0; for (let i = 0; i < texts.length; i++) { + // Check the limit BEFORE doing any work for this candidate — not after + // pushing a match — so limit:0 (a valid clampCount value: it's >= 0) + // correctly yields zero matches instead of one. Checking post-push would + // always let through the match that first reaches the limit. + if (matches.length >= limit) break; scanned++; const m = matcher.exec(texts[i] ?? ''); if (m) { matches.push({ candidateIndex: i, matchIndex: m.index, matchLength: m[0].length }); - if (matches.length >= limit) break; } } const done: SearchWorkerMessage = { type: 'done', matches, scanned }; diff --git a/test/history-module.test.ts b/test/history-module.test.ts index 1a267022..e87e29c5 100644 --- a/test/history-module.test.ts +++ b/test/history-module.test.ts @@ -240,6 +240,21 @@ describe('HistoryModule', () => { assert.equal((queryCall?.args as { limit?: number }).limit, 200); }); + it('clamps an out-of-u32-range offset to the native ceiling instead of passing it through to wrap (reviewer repro)', async () => { + // Number.MAX_SAFE_INTEGER is not a safe ceiling for a value that + // eventually crosses into a native u32 argument: 4294967296 (one past + // u32 max) must be clamped to 4294967295, not passed through as-is — + // otherwise it wraps/truncates at the N-API boundary into a small + // offset and silently returns the wrong page instead of an empty one. + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + await h.handleToolCall(call('extract', { offset: 4294967296, limit: 1 })); + const queryCall = calls.find((c) => c.method === 'queryMessagesByTimeAndChannel'); + assert.equal((queryCall?.args as { offset?: number }).offset, 4294967295); + }); + it('surfaces the capability-absent error cleanly', async () => { const { cm } = buildStub(FIXTURE, { throwUnsupported: true }); const h = new HistoryModule(); @@ -265,6 +280,34 @@ describe('HistoryModule', () => { assert.match(data.matches[0]!.snippet, /BAR/); }); + it('limit:0 (substring) returns zero matches instead of one (reviewer repro)', async () => { + // clampCount accepts 0 as a valid value (it's >= 0). Both matching + // loops used to push a match onto the results array BEFORE checking + // the limit, so with a matching candidate present, limit:0 still + // returned exactly 1 match. + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'bar', limit: 0 })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: unknown[] }; + assert.equal(data.matches.length, 0); + }); + + it('limit:0 (regex) returns zero matches instead of one (reviewer repro)', async () => { + // Same bug, independently, in search-regex-worker.ts's own copy of + // the match loop. + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: '\\bworld\\b', regex: true, limit: 0 })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: unknown[] }; + assert.equal(data.matches.length, 0); + }); + it('caseSensitive:true respects case', async () => { const { cm } = buildStub(FIXTURE); const h = new HistoryModule(); From 57ec6765a14a5af4ab07929970956371fbf84ede Mon Sep 17 00:00:00 2001 From: antra-tess Date: Thu, 17 Sep 2026 09:37:37 -0700 Subject: [PATCH 8/8] chore(deps): bump @animalabs/context-manager to ^0.9.2 Co-Authored-By: Claude Sonnet 5 --- package.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/package.json b/package.json index 9ebaa4c7..958fcbb8 100644 --- a/package.json +++ b/package.json @@ -47,7 +47,7 @@ "license": "MIT", "dependencies": { "@animalabs/chronicle": "^0.4.0", - "@animalabs/context-manager": "^0.9.1", + "@animalabs/context-manager": "^0.9.2", "@animalabs/membrane": "^0.5.78", "chokidar": "^4.0.3", "discord.js": "^14.25.1",