diff --git a/changelog.d/151-history-module.added.md b/changelog.d/151-history-module.added.md new file mode 100644 index 00000000..17616cc4 --- /dev/null +++ b/changelog.d/151-history-module.added.md @@ -0,0 +1,5 @@ +- `HistoryModule` adds `stats`/`extract`/`search` tools for querying an + agent's full uncompressed message history by time range and/or channel + (#151), backed by context-manager's native chronicle secondary-index + queries — O(log n + k) against multi-million-message stores, not a full + scan. Read-only, `bind(contextManager)` after `AgentFramework.create()`. diff --git a/package.json b/package.json index d5945c3e..958fcbb8 100644 --- a/package.json +++ b/package.json @@ -46,8 +46,8 @@ "author": "Anima Research", "license": "MIT", "dependencies": { - "@animalabs/chronicle": "^0.3.0", - "@animalabs/context-manager": "^0.8.0", + "@animalabs/chronicle": "^0.4.0", + "@animalabs/context-manager": "^0.9.2", "@animalabs/membrane": "^0.5.78", "chokidar": "^4.0.3", "discord.js": "^14.25.1", diff --git a/src/modules/history/index.ts b/src/modules/history/index.ts new file mode 100644 index 00000000..5119af8d --- /dev/null +++ b/src/modules/history/index.ts @@ -0,0 +1,613 @@ +/** + * HistoryModule — read-only access to an agent's full uncompressed message + * history, backed by context-manager's native chronicle secondary index + * (`queryMessagesByTime` / `queryMessagesByChannel` / + * `queryMessagesByTimeAndChannel` / `getChannelMessageCounts` / + * `getChannelTokenStats`, @animalabs/context-manager >= 0.6.0). Those calls + * are O(log n + k) against chronicle's `/timestamp` and + * `/metadata/external/channelId` field indexes, not a full-store scan, so + * this module can answer "what happened in #foo last Tuesday" against a + * multi-million-message store without walking the whole log. + * + * Three tools: + * - `stats` — per-channel message counts (all-time) + token totals + * (range-scoped), for orienting before pulling raw content. + * - `extract` — paginated raw messages for a time range and/or channel. + * - `search` — substring/regex match over a narrowed candidate window, + * with an explicit truncation signal when the caller's + * filter was too broad for `maxScan`. + * + * Kept deliberately read-only, same posture as HealthModule: no side + * effects, no message mutation. `onProcess` is a no-op. + * + * Every query dispatch collapses onto `queryMessagesByTimeAndChannel` — its + * own implementation already handles "only a time range", "only a channel", + * or "neither" by delegating to the single-filter native queries (see + * context-manager's `test/message-store-history-index.test.ts`), so this + * module doesn't need to re-decide which of the three query methods to call. + */ + +import { Worker } from 'node:worker_threads'; +import { fileURLToPath } from 'node:url'; +import { dirname, join } from 'node:path'; +import type { ContextManager, StoredMessage, ChannelCount, ChannelTokenStats } from '@animalabs/context-manager'; +import type { ContentBlock } from '@animalabs/membrane'; +import type { Module, ModuleContext } from '../../types/module.js'; +import type { ToolDefinition, ToolCall, ToolResult, ProcessEvent } from '../../types/events.js'; +import type { EventResponse, ProcessState } from '../../types/module.js'; +import type { SearchWorkerMessage, SearchWorkerMatch } from './search-regex-worker.js'; + +// ============================================================================ +// Tool input shapes +// ============================================================================ + +interface StatsInput { + from?: string; + to?: string; + channelId?: string; +} + +interface ExtractInput { + from?: string; + to?: string; + channelId?: string; + limit?: number; + offset?: number; + format?: 'text' | 'raw'; +} + +interface SearchInput { + query: string; + regex?: boolean; + caseSensitive?: boolean; + from?: string; + to?: string; + channelId?: string; + limit?: number; + maxScan?: number; +} + +// ============================================================================ +// Limits +// ============================================================================ + +const EXTRACT_DEFAULT_LIMIT = 50; +const EXTRACT_MAX_LIMIT = 200; + +/** + * Real ceiling of the native chronicle pagination `offset` argument (a u32 + * at the N-API boundary). `Number.MAX_SAFE_INTEGER` is NOT a safe ceiling + * for this field: a value between 0xFFFFFFFF and MAX_SAFE_INTEGER passes a + * naive upper-bound check unchanged, then wraps/truncates when it crosses + * into an unsigned 32-bit int on the native side — silently wrong + * pagination (repro: offset:4294967296 against a real store "succeeded" + * and returned page 0) instead of a clean failure. Clamping to this ceiling + * (same clamp-not-reject behavior `limit` already has) means anything + * beyond it lands on an offset no real store's message count will ever + * reach, i.e. a correctly-empty page, rather than wrapping onto a wrong one. + */ +const NATIVE_OFFSET_MAX = 0xffffffff; // 4294967295 + +const SEARCH_DEFAULT_LIMIT = 20; +const SEARCH_MAX_LIMIT = 100; +const SEARCH_DEFAULT_MAX_SCAN = 5000; +const SEARCH_MAX_MAX_SCAN = 50000; + +/** Characters of surrounding context kept on each side of a search snippet. */ +const SNIPPET_CONTEXT_CHARS = 80; +/** Fallback snippet length when a match position isn't meaningful to center on. */ +const SNIPPET_FALLBACK_CHARS = 160; + +/** + * Wall-clock deadline for a single regex-mode search call, enforced by + * forcibly terminating the worker thread doing the matching (see + * search-regex-worker.ts's header for why a worker — not a + * Promise.race/setTimeout — is required to actually interrupt a stuck + * synchronous RegExp.exec()). Generous enough for any legitimate pattern + * against a few thousand short strings; short enough that a catastrophic + * pattern doesn't tie up a worker (or an agent's turn) for long. + */ +const SEARCH_REGEX_TIMEOUT_MS = 2000; + +/** Compiled sibling of search-regex-worker.ts — resolved at runtime the same + * way gate-script.ts locates gate-script-worker.js, so it tracks whatever + * directory this module's own compiled output lives in. */ +const SEARCH_WORKER_PATH = join(dirname(fileURLToPath(import.meta.url)), 'search-regex-worker.js'); + +export class HistoryModule implements Module { + readonly name = 'history'; + + private ctx: ModuleContext | null = null; + private cm: ContextManager | null = null; + + /** + * Wire the context-manager instance. Host calls this after + * ContextManager.open() so the module can issue the native index-backed + * queries — ModuleContext itself exposes no store/context-manager + * reference, only the narrower message CRUD surface (addMessage/getMessage/ + * queryMessages), which can't do range or channel queries. + */ + bind(contextManager: ContextManager): void { + this.cm = contextManager; + } + + async start(ctx: ModuleContext): Promise { + this.ctx = ctx; + } + + async stop(): Promise { + this.ctx = null; + } + + getTools(): ToolDefinition[] { + return [ + { + name: 'stats', + description: + 'Orient before pulling raw history: per-channel message counts and token totals. ' + + 'messageCountsAllTime is always all-time and all-channel-scope (the underlying index has no ' + + 'range dimension) — it is NOT limited by from/to, only optionally filtered to one channelId. ' + + 'tokenStatsForRange IS scoped to from/to when given. The two are reported separately, never ' + + 'merged, because their range semantics differ.', + inputSchema: { + type: 'object' as const, + properties: { + from: { type: 'string', description: 'ISO 8601 inclusive lower bound for tokenStatsForRange. Omit for open-ended.' }, + to: { type: 'string', description: 'ISO 8601 inclusive upper bound for tokenStatsForRange. Omit for open-ended.' }, + channelId: { type: 'string', description: 'Restrict both parts of the response to one channel.' }, + }, + }, + }, + { + name: 'extract', + description: + 'Fetch raw, uncompressed messages for a time range and/or channel, paginated oldest-first. ' + + 'Use after `stats` to pull the actual content. format:"text" flattens each message to a short ' + + 'readable string (tool_use/tool_result/thinking/images rendered as bracketed labels); ' + + 'format:"raw" returns the content blocks unmodified.', + inputSchema: { + type: 'object' as const, + properties: { + from: { type: 'string', description: 'ISO 8601 inclusive lower bound. Omit for open-ended.' }, + to: { type: 'string', description: 'ISO 8601 inclusive upper bound. Omit for open-ended.' }, + channelId: { type: 'string', description: 'Restrict to one channel.' }, + limit: { type: 'number', description: `Max messages to return (default ${EXTRACT_DEFAULT_LIMIT}, hard cap ${EXTRACT_MAX_LIMIT}). Must be a non-negative integer.` }, + offset: { type: 'number', description: `Number of matching messages to skip (default 0, capped to ${NATIVE_OFFSET_MAX}). Must be a non-negative integer.` }, + format: { type: 'string', enum: ['text', 'raw'], description: 'Content rendering (default "text").' }, + }, + }, + }, + { + name: 'search', + description: + 'Search message history for a substring (default) or regex match, within an optional time ' + + 'range/channel window. Narrows candidates via the same query as `extract` before matching, up ' + + 'to maxScan candidates — if the narrowed window is larger than maxScan, the response reports ' + + 'truncated:true up front rather than silently missing later matches; narrow the filter or raise ' + + 'maxScan and retry. regex:true matching runs under a wall-clock deadline and is cleanly failed ' + + `(not silently empty) if a pattern is too slow — avoid nested-quantifier patterns like (a+)+.`, + inputSchema: { + type: 'object' as const, + properties: { + query: { type: 'string', description: 'Substring (or regex source, when regex:true) to search for.' }, + regex: { type: 'boolean', description: 'Treat query as a regular expression (default false).' }, + caseSensitive: { type: 'boolean', description: 'Case-sensitive match (default false).' }, + from: { type: 'string', description: 'ISO 8601 inclusive lower bound. Omit for open-ended.' }, + to: { type: 'string', description: 'ISO 8601 inclusive upper bound. Omit for open-ended.' }, + channelId: { type: 'string', description: 'Restrict to one channel.' }, + limit: { type: 'number', description: `Max matches to return (default ${SEARCH_DEFAULT_LIMIT}, hard cap ${SEARCH_MAX_LIMIT}). Must be a non-negative integer.` }, + maxScan: { type: 'number', description: `Max candidate messages to scan (default ${SEARCH_DEFAULT_MAX_SCAN}, hard cap ${SEARCH_MAX_MAX_SCAN}). Must be a non-negative integer.` }, + }, + required: ['query'], + }, + }, + ]; + } + + async handleToolCall(call: ToolCall): Promise { + try { + if (!this.cm) { + throw new Error('HistoryModule not bound — host must call bind(contextManager) before tool dispatch.'); + } + switch (call.name) { + case 'stats': + return this.handleStats((call.input ?? {}) as StatsInput); + case 'extract': + return this.handleExtract((call.input ?? {}) as ExtractInput); + case 'search': + return await this.handleSearch((call.input ?? {}) as SearchInput); + default: + return { success: false, isError: true, error: `Unknown tool: ${call.name}` }; + } + } catch (error) { + // Catches our own validation errors (bad ISO date, invalid regex, + // out-of-range limit/offset/maxScan, unbound module), a regex-search + // worker timeout or worker-side error (see handleSearch/ + // searchWithRegexWorker — a ReDoS-shaped pattern surfaces here as a + // clean timeout error, never a hang), and context-manager's + // capability-absent error ("Chronicle history index unsupported...", + // thrown by queryMessagesByTime/queryMessagesByChannel/ + // queryMessagesByTimeAndChannel/getChannelMessageCounts/ + // getChannelTokenStats on a chronicle build that predates the native + // index-query capability) — all surfaced as a normal tool error + // rather than crashing the module. + return { + success: false, + isError: true, + error: error instanceof Error ? error.message : String(error), + }; + } + } + + async onProcess(_event: ProcessEvent, _state: ProcessState): Promise { + return {}; + } + + // ========================================================================== + // stats + // ========================================================================== + + private handleStats(input: StatsInput): ToolResult { + const fromMs = parseIsoDate(input.from, 'from'); + const toMs = parseIsoDate(input.to, 'to'); + const cm = this.cm as ContextManager; + + let messageCountsAllTime: ChannelCount[] = cm.getChannelMessageCounts(); + let tokenStatsForRange: ChannelTokenStats = cm.getChannelTokenStats({ fromMs, toMs }); + + if (input.channelId) { + messageCountsAllTime = messageCountsAllTime.filter((c) => c.channelId === input.channelId); + tokenStatsForRange = { + // Totals stay whole-range (they're documented as store-wide-for-the-range, + // not per-channel); only the byChannel breakdown is narrowed. Labeling the + // field `byChannel` (rather than folding a channel-filtered total into + // totalMessages/totalTokensEstimate) keeps this from reading as a + // misleadingly-merged single-channel total. + totalMessages: tokenStatsForRange.totalMessages, + totalTokensEstimate: tokenStatsForRange.totalTokensEstimate, + byChannel: tokenStatsForRange.byChannel.filter((c) => c.channelId === input.channelId), + }; + } + + return { + success: true, + data: { + query: { from: input.from ?? null, to: input.to ?? null, channelId: input.channelId ?? null }, + messageCountsAllTime, + tokenStatsForRange, + }, + }; + } + + // ========================================================================== + // extract + // ========================================================================== + + private handleExtract(input: ExtractInput): ToolResult { + const fromMs = parseIsoDate(input.from, 'from'); + const toMs = parseIsoDate(input.to, 'to'); + const limit = clampCount(input.limit, EXTRACT_DEFAULT_LIMIT, EXTRACT_MAX_LIMIT, 'limit'); + const offset = clampCount(input.offset, 0, NATIVE_OFFSET_MAX, 'offset'); + const format = input.format ?? 'text'; + + const cm = this.cm as ContextManager; + const result = cm.queryMessagesByTimeAndChannel({ + fromMs, + toMs, + channelId: input.channelId, + limit, + offset, + }); + + return { + success: true, + data: { + matchedCount: result.matchedCount, + returned: result.messages.length, + messages: result.messages.map((msg) => projectMessage(msg, format)), + }, + }; + } + + // ========================================================================== + // search + // ========================================================================== + + private async handleSearch(input: SearchInput): Promise { + if (!input.query) { + throw new Error('search requires a non-empty "query".'); + } + const fromMs = parseIsoDate(input.from, 'from'); + const toMs = parseIsoDate(input.to, 'to'); + const limit = clampCount(input.limit, SEARCH_DEFAULT_LIMIT, SEARCH_MAX_LIMIT, 'limit'); + const maxScan = clampCount(input.maxScan, SEARCH_DEFAULT_MAX_SCAN, SEARCH_MAX_MAX_SCAN, 'maxScan'); + const caseSensitive = input.caseSensitive ?? false; + const flags = caseSensitive ? '' : 'i'; + + // Validate regex SYNTAX up front — an invalid pattern is a clean tool + // error, not a crash mid-scan. This does NOT bound match TIME (a + // syntactically valid pattern can still backtrack catastrophically), + // which is why regex-mode matching itself runs on a worker below rather + // than here. + if (input.regex) { + try { + new RegExp(input.query, flags); + } catch (error) { + throw new Error(`Invalid regex "${input.query}": ${error instanceof Error ? error.message : String(error)}`); + } + } + const needle = caseSensitive ? input.query : input.query.toLowerCase(); + + const cm = this.cm as ContextManager; + // Fetch one more than maxScan so an oversized candidate window is + // detected up front (before any matching is attempted), rather than + // silently scanning maxScan candidates and returning as if that were the + // whole window. See the tool description's `truncated` contract. + const probe = cm.queryMessagesByTimeAndChannel({ + fromMs, + toMs, + channelId: input.channelId, + limit: maxScan + 1, + offset: 0, + }); + const truncated = probe.messages.length > maxScan; + const candidates = truncated ? probe.messages.slice(0, maxScan) : probe.messages; + + if (input.regex) { + // Regex matching against caller-supplied patterns is ReDoS-shaped: a + // pathological pattern (e.g. `(a+)+$`) can take catastrophically long + // against one candidate string, and a synchronous RegExp.exec() on + // this thread would block the WHOLE framework's event loop — every + // agent's turns, health checks, timers — for as long as it runs, with + // no way to interrupt it from this same thread. Route matching + // through a worker thread instead, which can be forcibly terminated + // on a deadline. See search-regex-worker.ts's header for the full + // rationale. + const { matches, scanned } = await this.searchWithRegexWorker(candidates, input.query, flags, limit); + return { success: true, data: { scanned, candidatePoolSize: candidates.length, truncated, matches } }; + } + + // Plain substring search (String.prototype.indexOf) is inherently + // linear in input length — no ReDoS-equivalent risk — so it stays + // in-process. This is also the common case, so it pays no worker-spawn + // overhead. + const matches: SearchMatch[] = []; + let scanned = 0; + for (const msg of candidates) { + // Check the limit BEFORE doing any work for this candidate — not + // after pushing a match — so limit:0 (a valid clampCount value: it's + // >= 0) correctly yields zero matches instead of one. Checking + // post-push would always let through the match that first reaches + // the limit. Same fix mirrored in search-regex-worker.ts's loop. + if (matches.length >= limit) break; + scanned++; + const text = flattenContent(msg.content); + const hit = matchSubstring(text, needle, caseSensitive); + if (!hit) continue; + matches.push({ + id: String(msg.id), + timestamp: msg.timestamp.toISOString(), + participant: msg.participant, + channelId: getChannelId(msg) ?? null, + snippet: snippetAround(text, hit.index, hit.length), + }); + } + + return { + success: true, + data: { scanned, candidatePoolSize: candidates.length, truncated, matches }, + }; + } + + /** + * Run regex matching for `search` on a worker thread with a hard + * wall-clock deadline, so a catastrophically-backtracking pattern can be + * forcibly killed instead of hanging the framework. One worker per call + * (not per candidate — spawn overhead would dominate at scale; not a + * persistent pool — a fresh worker per call means one bad pattern can + * never contaminate a later search). Always terminated on the way out, + * success or failure, so nothing lingers. + * + * On timeout or a worker-side error this THROWS (caught by + * handleToolCall's try/catch, same as every other error path in this + * module) rather than returning an empty match list — a timed-out search + * must never be indistinguishable from a clean "no matches" result. + */ + private async searchWithRegexWorker( + candidates: StoredMessage[], + pattern: string, + flags: string, + limit: number, + ): Promise<{ matches: SearchMatch[]; scanned: number }> { + const texts = candidates.map((msg) => flattenContent(msg.content)); + let worker: Worker | undefined; + try { + const { matches: rawMatches, scanned } = await new Promise<{ matches: SearchWorkerMatch[]; scanned: number }>( + (resolve, reject) => { + worker = new Worker(SEARCH_WORKER_PATH, { workerData: { texts, pattern, flags, limit } }); + const timer = setTimeout(() => { + reject( + new Error( + `search timed out after ${SEARCH_REGEX_TIMEOUT_MS}ms while matching regex "${pattern}" against ` + + `up to ${texts.length} candidate(s) — the pattern may be catastrophically slow (exponential ` + + 'backtracking) against this data; try a simpler pattern, a literal substring search ' + + '(regex:false), or a smaller maxScan.', + ), + ); + }, SEARCH_REGEX_TIMEOUT_MS); + timer.unref?.(); + worker.once('message', (msg: SearchWorkerMessage) => { + clearTimeout(timer); + if (msg.type === 'error') reject(new Error(msg.error)); + else resolve({ matches: msg.matches, scanned: msg.scanned }); + }); + worker.once('error', (err) => { + clearTimeout(timer); + reject(err); + }); + }, + ); + + const matches: SearchMatch[] = rawMatches.map(({ candidateIndex, matchIndex, matchLength }) => { + const msg = candidates[candidateIndex]!; + const text = texts[candidateIndex]!; + return { + id: String(msg.id), + timestamp: msg.timestamp.toISOString(), + participant: msg.participant, + channelId: getChannelId(msg) ?? null, + snippet: snippetAround(text, matchIndex, matchLength), + }; + }); + return { matches, scanned }; + } finally { + // Always kill the worker — whether it finished, errored, or is still + // stuck mid-backtrack when the deadline hit. terminate() on an + // already-exited worker is a harmless no-op. + if (worker) void worker.terminate().catch(() => {}); + } + } +} + +// ============================================================================ +// Helpers +// ============================================================================ + +/** Parse an ISO 8601 date string to Unix ms. Throws a clear error (not a + * crash) on an unparseable string; returns undefined for an omitted field + * (open-ended bound). */ +function parseIsoDate(value: string | undefined, field: string): number | undefined { + if (value === undefined) return undefined; + const ms = Date.parse(value); + if (Number.isNaN(ms)) { + throw new Error(`Invalid ISO 8601 date for "${field}": ${JSON.stringify(value)}`); + } + return ms; +} + +/** + * Validate a pagination-ish numeric input (limit/offset/maxScan) and clamp + * it to an upper bound. `Math.min(value, max)` alone is NOT sufficient here: + * it only enforces an upper bound, so `Math.min(-1, 200)` is `-1`, not a + * sane value — a negative limit/offset/maxScan would sail straight past the + * "hard cap" and reach a native chronicle call expecting an unsigned + * pagination argument (observed: `extract({limit:-1})` returning every + * message in the store instead of being capped). Rejects (rather than + * silently clamping) anything that isn't a finite non-negative integer, so + * a caller mistake is surfaced as a clean tool error instead of silently + * doing something other than what was asked. + */ +function clampCount(value: number | undefined, def: number, max: number, field: string): number { + if (value === undefined) return def; + if (typeof value !== 'number' || !Number.isFinite(value)) { + throw new Error(`"${field}" must be a finite number, got ${JSON.stringify(value)}.`); + } + if (!Number.isInteger(value)) { + throw new Error(`"${field}" must be an integer, got ${value}.`); + } + if (value < 0) { + throw new Error(`"${field}" must be >= 0, got ${value}.`); + } + return Math.min(value, max); +} + +/** channelId lives at metadata.external.channelId — same path context-manager's + * own native index reads from (message-store.ts's CHANNEL_FIELD). Metadata's + * index-signature type means this needs an explicit narrow, same as + * MessageStore.getChannelTokenStats does internally. */ +/** + * Two channel-id metadata shapes exist in the wild: `metadata.channelId` + * (what agent-framework's real MCPL ingestion — `handleMcplChannelIncoming` + * — actually writes) and `metadata.external.channelId` (an older + * convention). context-manager 0.9.1+ indexes and queries BOTH (see + * MessageStore.extractChannelId there), so a channel-filtered result can + * come from either shape — check the same order here so a displayed + * `channelId` never reads `null` on a message that was, in fact, matched by + * channel. + */ +function getChannelId(msg: StoredMessage): string | undefined { + const metadata = msg.metadata as { channelId?: unknown; external?: { channelId?: unknown } } | undefined; + if (typeof metadata?.channelId === 'string') return metadata.channelId; + return typeof metadata?.external?.channelId === 'string' ? metadata.external.channelId : undefined; +} + +function projectMessage(msg: StoredMessage, format: 'text' | 'raw'): Record { + return { + id: String(msg.id), + timestamp: msg.timestamp.toISOString(), + participant: msg.participant, + channelId: getChannelId(msg) ?? null, + content: format === 'raw' ? msg.content : flattenContent(msg.content), + }; +} + +/** Flatten a message's content blocks to one short, readable string. Text + * blocks verbatim; everything else (tool_use/tool_result/thinking/media) as + * a short bracketed label — this is for agent readability and search + * matching, not byte-faithful reconstruction. */ +function flattenContent(content: ContentBlock[]): string { + return content.map(blockLabel).join(' ').trim(); +} + +function blockLabel(block: ContentBlock): string { + switch (block.type) { + case 'text': + return block.text; + case 'tool_use': + return `[tool_use: ${block.name}]`; + case 'tool_result': + return block.toolName ? `[tool_result: ${block.toolName}]` : '[tool_result]'; + case 'thinking': + case 'redacted_thinking': + return '[thinking]'; + case 'image': + return '[image]'; + case 'generated_image': + return '[image]'; + case 'document': + return '[document]'; + case 'audio': + return '[audio]'; + case 'video': + return '[video]'; + default: + // Exhaustiveness guard: a future ContentBlock variant falls back to a + // generic label instead of a compile error at a call site far from here. + return `[${(block as { type: string }).type}]`; + } +} + +interface MatchHit { + index: number; + length: number; +} + +/** One `search` result entry — shared shape between the in-process substring + * path and the worker-backed regex path. */ +interface SearchMatch { + id: string; + timestamp: string; + participant: string; + channelId: string | null; + snippet: string; +} + +function matchSubstring(text: string, needle: string, caseSensitive: boolean): MatchHit | null { + const haystack = caseSensitive ? text : text.toLowerCase(); + const index = haystack.indexOf(needle); + if (index === -1) return null; + return { index, length: needle.length }; +} + +/** ~SNIPPET_CONTEXT_CHARS of surrounding context on each side of a match, or + * the first ~SNIPPET_FALLBACK_CHARS of content when no meaningful match + * position is given (kept simple on purpose — this is a readability aid, + * not a highlighting engine). */ +function snippetAround(text: string, index: number, length: number): string { + if (index < 0) return text.slice(0, SNIPPET_FALLBACK_CHARS); + const start = Math.max(0, index - SNIPPET_CONTEXT_CHARS); + const end = Math.min(text.length, index + length + SNIPPET_CONTEXT_CHARS); + const prefix = start > 0 ? '…' : ''; + const suffix = end < text.length ? '…' : ''; + return prefix + text.slice(start, end) + suffix; +} diff --git a/src/modules/history/search-regex-worker.ts b/src/modules/history/search-regex-worker.ts new file mode 100644 index 00000000..48993284 --- /dev/null +++ b/src/modules/history/search-regex-worker.ts @@ -0,0 +1,76 @@ +/** + * search-regex-worker — runs a caller-supplied regex against a batch of + * message text on a separate thread, so HistoryModule's `search` tool can + * forcibly terminate a catastrophically-backtracking pattern (ReDoS) + * instead of blocking the framework's single-threaded event loop. + * + * Why a worker and not a Promise.race/setTimeout "timeout": JS is + * single-threaded, so nothing on the SAME thread can interrupt a + * synchronous RegExp.exec() call already in flight — a timeout callback + * racing it can't even fire until the blocking call returns, by which point + * it's too late. A worker thread is real OS-level concurrency: the parent's + * event loop keeps running, and `worker.terminate()` actually kills the + * stuck thread rather than just giving up on waiting for it. + * + * Only regex-mode search routes through this worker. Plain substring search + * (String.prototype.indexOf) is inherently linear in input length and has + * no equivalent risk, so it stays in-process — the common case pays no + * worker-spawn overhead. + * + * One-shot: the whole job (already-flattened text, already syntax-validated + * pattern/flags, match limit) arrives via `workerData` at construction; this + * script runs it to completion and posts exactly one message, then the + * parent always terminates the worker (whether it finished normally or had + * to be killed for running long) — no persistent pool, so a wedged worker + * from one bad pattern can never contaminate a later call. + */ +import { parentPort, workerData } from 'node:worker_threads'; + +export interface SearchWorkerJob { + /** Flattened candidate text, aligned by index with the parent's candidate array. */ + texts: string[]; + pattern: string; + flags: string; + /** Stop scanning once this many matches are found. */ + limit: number; +} + +export interface SearchWorkerMatch { + /** Index into `texts` / the parent's candidate array. */ + candidateIndex: number; + matchIndex: number; + matchLength: number; +} + +export type SearchWorkerMessage = + | { type: 'done'; matches: SearchWorkerMatch[]; scanned: number } + | { type: 'error'; error: string }; + +const { texts, pattern, flags, limit } = workerData as SearchWorkerJob; + +try { + // Re-construct (not re-validate for safety — RegExp syntax is what it is; + // the parent already rejected an uncompilable pattern before ever getting + // here) the matcher on this thread. Compilation succeeding doesn't bound + // MATCHING time — that's exactly the risk this worker exists to contain. + const matcher = new RegExp(pattern, flags); + const matches: SearchWorkerMatch[] = []; + let scanned = 0; + for (let i = 0; i < texts.length; i++) { + // Check the limit BEFORE doing any work for this candidate — not after + // pushing a match — so limit:0 (a valid clampCount value: it's >= 0) + // correctly yields zero matches instead of one. Checking post-push would + // always let through the match that first reaches the limit. + if (matches.length >= limit) break; + scanned++; + const m = matcher.exec(texts[i] ?? ''); + if (m) { + matches.push({ candidateIndex: i, matchIndex: m.index, matchLength: m[0].length }); + } + } + const done: SearchWorkerMessage = { type: 'done', matches, scanned }; + parentPort?.postMessage(done); +} catch (e) { + const errMsg: SearchWorkerMessage = { type: 'error', error: e instanceof Error ? e.message : String(e) }; + parentPort?.postMessage(errMsg); +} diff --git a/src/modules/index.ts b/src/modules/index.ts index 4d8aedbb..12059676 100644 --- a/src/modules/index.ts +++ b/src/modules/index.ts @@ -8,6 +8,8 @@ export type { ApiEvent } from './api/index.js'; export { HealthModule } from './health/index.js'; export type { HealthModuleConfig } from './health/index.js'; +export { HistoryModule } from './history/index.js'; + export { WorkspaceModule, WorkspaceReadError } from './workspace/index.js'; export type { WorkspaceReadErrorCode, WorkspaceReadStage, WorkspaceDiskReadResult, ReadFileFromDiskOptions } from './workspace/index.js'; export type { diff --git a/test/history-module.test.ts b/test/history-module.test.ts new file mode 100644 index 00000000..e87e29c5 --- /dev/null +++ b/test/history-module.test.ts @@ -0,0 +1,517 @@ +import { describe, it } from 'node:test'; +import assert from 'node:assert/strict'; +import { HistoryModule } from '../src/modules/history/index.js'; +import type { ContextManager, StoredMessage, IndexedMessageQueryResult, ChannelCount, ChannelTokenStats } from '@animalabs/context-manager'; +import type { ContentBlock } from '@animalabs/membrane'; +import type { ToolCall } from '../src/types/events.js'; + +/** + * HistoryModule's own logic is dispatch (always via + * queryMessagesByTimeAndChannel, per context-manager's own test evidence + * that it already handles "only one filter" and "neither filter"), input + * validation (ISO dates, regex, limit capping), and presentation (content + * flattening, snippet extraction, truncation detection). The underlying + * index-query *correctness* is context-manager's own responsibility and is + * covered by its test suite (test/message-store-history-index.test.ts) — so + * this stub implements realistic-but-simple in-memory filtering over a small + * fixture rather than standing up a real chronicle store, and a `calls` log + * lets tests assert on exactly what HistoryModule asked for (e.g. the capped + * limit). + */ + +function textBlock(text: string): ContentBlock { + return { type: 'text', text }; +} + +function metaFor(channelId?: string): Record | undefined { + return channelId ? { external: { channelId } } : undefined; +} + +function msg(id: string, ms: number, participant: string, content: ContentBlock[], channelId?: string): StoredMessage { + return { + id, + sequence: Number(id.replace(/\D/g, '')), + participant, + content, + metadata: metaFor(channelId), + timestamp: new Date(ms), + } as StoredMessage; +} + +function channelIdOf(m: StoredMessage): string | undefined { + return (m.metadata as { external?: { channelId?: string } } | undefined)?.external?.channelId; +} + +/** Trivial, deterministic stand-in for real token estimation — tests only + * assert on shape/filtering, never on the exact number. */ +function estimate(m: StoredMessage): number { + return m.content.reduce((n, b) => n + (b.type === 'text' ? b.text.length : 4), 0); +} + +interface StubOptions { + /** When set, every query/stats method throws this — simulates a chronicle + * build that predates the native index-query capability. */ + throwUnsupported?: boolean; +} + +function buildStub(messages: StoredMessage[], opts: StubOptions = {}): { cm: ContextManager; calls: Array<{ method: string; args: unknown }> } { + const calls: Array<{ method: string; args: unknown }> = []; + const unsupported = () => { + throw new Error('Chronicle history index unsupported: native field-index capability absent on this chronicle build.'); + }; + + const cm = { + queryMessagesByTimeAndChannel(args: { fromMs?: number; toMs?: number; channelId?: string; limit?: number; offset?: number }): IndexedMessageQueryResult { + calls.push({ method: 'queryMessagesByTimeAndChannel', args }); + if (opts.throwUnsupported) return unsupported(); + const filtered = messages.filter((m) => { + const ts = m.timestamp.getTime(); + if (args.fromMs !== undefined && ts < args.fromMs) return false; + if (args.toMs !== undefined && ts > args.toMs) return false; + if (args.channelId !== undefined && channelIdOf(m) !== args.channelId) return false; + return true; + }); + const matchedCount = filtered.length; + const offset = args.offset ?? 0; + const limit = args.limit ?? filtered.length; + return { messages: filtered.slice(offset, offset + limit), matchedCount }; + }, + getChannelMessageCounts(): ChannelCount[] { + calls.push({ method: 'getChannelMessageCounts', args: undefined }); + if (opts.throwUnsupported) return unsupported(); + const counts = new Map(); + for (const m of messages) { + const c = channelIdOf(m); + if (c) counts.set(c, (counts.get(c) ?? 0) + 1); + } + return [...counts.entries()].map(([channelId, count]) => ({ channelId, messages: count })); + }, + getChannelTokenStats(args?: { fromMs?: number; toMs?: number }): ChannelTokenStats { + calls.push({ method: 'getChannelTokenStats', args }); + if (opts.throwUnsupported) return unsupported(); + const inRange = messages.filter((m) => { + const ts = m.timestamp.getTime(); + if (args?.fromMs !== undefined && ts < args.fromMs) return false; + if (args?.toMs !== undefined && ts > args.toMs) return false; + return true; + }); + const byChannel = new Map(); + let totalTokensEstimate = 0; + for (const m of inRange) { + const est = estimate(m); + totalTokensEstimate += est; + const c = channelIdOf(m); + if (c) { + const agg = byChannel.get(c) ?? { messages: 0, tokensEstimate: 0 }; + agg.messages++; + agg.tokensEstimate += est; + byChannel.set(c, agg); + } + } + return { + totalMessages: inRange.length, + totalTokensEstimate, + byChannel: [...byChannel.entries()].map(([channelId, agg]) => ({ channelId, ...agg })), + }; + }, + } as unknown as ContextManager; + + return { cm, calls }; +} + +function call(name: string, input: Record): ToolCall { + return { id: `call-${name}`, name, input } as unknown as ToolCall; +} + +// Fixture: two channels (c1, c2), one channel-less message, spread across +// distinct timestamps so time-range filtering is exercisable. +const FIXTURE: StoredMessage[] = [ + msg('m1', 1000, 'User', [textBlock('hello world')], 'c1'), + msg('m2', 2000, 'Claude', [{ type: 'tool_use', id: 't1', name: 'search', input: {} }], 'c1'), + msg('m3', 3000, 'User', [textBlock('foo BAR baz')], 'c2'), + msg('m4', 4000, 'Claude', [{ type: 'thinking', thinking: 'hmm' }, textBlock('final answer')], 'c2'), + msg('m5', 5000, 'User', [textBlock('unchanneled message')]), + msg('m6', 6000, 'User', [textBlock('another c1 message with searchterm inside')], 'c1'), +]; + +describe('HistoryModule', () => { + describe('stats', () => { + it('reports all-time message counts and range-scoped token stats separately', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + // Range covers only m3..m5 (3000-5000), but messageCountsAllTime must + // still reflect the whole store. + const result = await h.handleToolCall(call('stats', { from: new Date(3000).toISOString(), to: new Date(5000).toISOString() })); + assert.equal(result.success, true, result.error); + const data = result.data as { + messageCountsAllTime: ChannelCount[]; + tokenStatsForRange: ChannelTokenStats; + }; + + const allTimeByChannel = new Map(data.messageCountsAllTime.map((c) => [c.channelId, c.messages])); + assert.equal(allTimeByChannel.get('c1'), 3); // m1, m2, m6 — unaffected by the range + assert.equal(allTimeByChannel.get('c2'), 2); // m3, m4 + + assert.equal(data.tokenStatsForRange.totalMessages, 3); // m3, m4, m5 + const rangeByChannel = new Map(data.tokenStatsForRange.byChannel.map((c) => [c.channelId, c.messages])); + assert.equal(rangeByChannel.get('c2'), 2); // m3 and m4 both fall in [3000,5000] + assert.equal(rangeByChannel.has('c1'), false); // no c1 message falls in [3000,5000] + }); + + it('filters both parts of the response to one channelId when given', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('stats', { channelId: 'c1' })); + assert.equal(result.success, true, result.error); + const data = result.data as { messageCountsAllTime: ChannelCount[]; tokenStatsForRange: ChannelTokenStats }; + assert.deepEqual(data.messageCountsAllTime.map((c) => c.channelId), ['c1']); + assert.deepEqual(data.tokenStatsForRange.byChannel.map((c) => c.channelId), ['c1']); + }); + + it('rejects an invalid ISO date cleanly instead of throwing', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('stats', { from: 'not-a-date' })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Invalid ISO 8601 date/); + }); + + it('surfaces the capability-absent error as a clean tool error, not a crash', async () => { + const { cm } = buildStub(FIXTURE, { throwUnsupported: true }); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('stats', {})); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Chronicle history index unsupported/); + }); + }); + + describe('extract', () => { + it('filters by channel and date range', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall( + call('extract', { channelId: 'c1', from: new Date(1500).toISOString() }), + ); + assert.equal(result.success, true, result.error); + const data = result.data as { matchedCount: number; returned: number; messages: Array<{ id: string }> }; + // c1 messages at ts >= 1500: m2 (2000), m6 (6000) — m1 (1000) excluded. + assert.deepEqual(data.messages.map((m) => m.id), ['m2', 'm6']); + assert.equal(data.returned, 2); + assert.equal(data.matchedCount, 2); + }); + + it('format:"text" flattens content; format:"raw" returns blocks as-is', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const textResult = await h.handleToolCall(call('extract', { channelId: 'c2', format: 'text' })); + const textData = textResult.data as { messages: Array<{ id: string; content: unknown }> }; + const m4 = textData.messages.find((m) => m.id === 'm4')!; + assert.equal(m4.content, '[thinking] final answer'); + + const rawResult = await h.handleToolCall(call('extract', { channelId: 'c2', format: 'raw' })); + const rawData = rawResult.data as { messages: Array<{ id: string; content: unknown }> }; + const m4Raw = rawData.messages.find((m) => m.id === 'm4')!; + assert.ok(Array.isArray(m4Raw.content)); + assert.equal((m4Raw.content as ContentBlock[]).length, 2); + assert.equal((m4Raw.content as ContentBlock[])[0]?.type, 'thinking'); + }); + + it('caps limit at the hard maximum before it reaches context-manager', async () => { + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + await h.handleToolCall(call('extract', { limit: 999999 })); + const queryCall = calls.find((c) => c.method === 'queryMessagesByTimeAndChannel'); + assert.equal((queryCall?.args as { limit?: number }).limit, 200); + }); + + it('clamps an out-of-u32-range offset to the native ceiling instead of passing it through to wrap (reviewer repro)', async () => { + // Number.MAX_SAFE_INTEGER is not a safe ceiling for a value that + // eventually crosses into a native u32 argument: 4294967296 (one past + // u32 max) must be clamped to 4294967295, not passed through as-is — + // otherwise it wraps/truncates at the N-API boundary into a small + // offset and silently returns the wrong page instead of an empty one. + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + await h.handleToolCall(call('extract', { offset: 4294967296, limit: 1 })); + const queryCall = calls.find((c) => c.method === 'queryMessagesByTimeAndChannel'); + assert.equal((queryCall?.args as { offset?: number }).offset, 4294967295); + }); + + it('surfaces the capability-absent error cleanly', async () => { + const { cm } = buildStub(FIXTURE, { throwUnsupported: true }); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('extract', {})); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Chronicle history index unsupported/); + }); + }); + + describe('search', () => { + it('matches a case-insensitive substring by default', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'bar' })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: Array<{ id: string; snippet: string }> }; + assert.deepEqual(data.matches.map((m) => m.id), ['m3']); + assert.match(data.matches[0]!.snippet, /BAR/); + }); + + it('limit:0 (substring) returns zero matches instead of one (reviewer repro)', async () => { + // clampCount accepts 0 as a valid value (it's >= 0). Both matching + // loops used to push a match onto the results array BEFORE checking + // the limit, so with a matching candidate present, limit:0 still + // returned exactly 1 match. + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'bar', limit: 0 })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: unknown[] }; + assert.equal(data.matches.length, 0); + }); + + it('limit:0 (regex) returns zero matches instead of one (reviewer repro)', async () => { + // Same bug, independently, in search-regex-worker.ts's own copy of + // the match loop. + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: '\\bworld\\b', regex: true, limit: 0 })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: unknown[] }; + assert.equal(data.matches.length, 0); + }); + + it('caseSensitive:true respects case', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'bar', caseSensitive: true })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: unknown[] }; + assert.equal(data.matches.length, 0); // fixture only has "BAR", not "bar" + }); + + it('matches a regex', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: '\\bworld\\b', regex: true })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: Array<{ id: string }> }; + assert.deepEqual(data.matches.map((m) => m.id), ['m1']); + }); + + it('forcibly terminates a catastrophically-backtracking (ReDoS) regex instead of hanging', async () => { + // Reviewer's repro: 32 'a's followed by a non-matching character, plus + // a classic exponential-backtracking pattern. `(a+)+$` against this + // input backtracks exponentially and (verified separately in an + // isolated child process) has to be force-killed after several + // seconds; the equivalent linear pattern `^a+$` returns instantly. + // This must never block the framework's single event loop — matching + // has to happen off-thread with a real, forcible deadline. + const evil = 'a'.repeat(32) + '!'; + const { cm } = buildStub([msg('r1', 1000, 'User', [textBlock(evil)], 'c1')]); + const h = new HistoryModule(); + h.bind(cm); + + const start = Date.now(); + const result = await h.handleToolCall(call('search', { query: '(a+)+$', regex: true, maxScan: 1 })); + const elapsed = Date.now() - start; + + // Returns well within a bounded time (module deadline is 2s; generous + // slack here for worker spawn/CI jitter) — proves the test runner + // itself never hung waiting on this call. + assert.ok(elapsed < 8000, `search took ${elapsed}ms — should have been forcibly terminated near the deadline`); + // A cut-short match MUST be a clean, unambiguous failure, never + // success:true with an empty matches array — that would be + // indistinguishable from "no matches found", which is a lie. + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /timed out/i); + }); + + it('a normal regex against the same kind of input still matches correctly (worker path is not just failing everything)', async () => { + const { cm } = buildStub([msg('r1', 1000, 'User', [textBlock('aaaa!')], 'c1')]); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: '^a+!$', regex: true })); + assert.equal(result.success, true, result.error); + const data = result.data as { matches: Array<{ id: string }> }; + assert.deepEqual(data.matches.map((m) => m.id), ['r1']); + }); + + it('returns a clean error for an invalid regex instead of throwing', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: '(unclosed', regex: true })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Invalid regex/); + }); + + it('reports truncated:true up front when the candidate window exceeds maxScan', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'm', maxScan: 3 })); + assert.equal(result.success, true, result.error); + const data = result.data as { truncated: boolean; candidatePoolSize: number; scanned: number }; + assert.equal(data.truncated, true); + assert.equal(data.candidatePoolSize, 3); + assert.ok(data.scanned <= 3); + }); + + it('does not report truncated when the candidate window fits within maxScan', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'searchterm', maxScan: 5000 })); + assert.equal(result.success, true, result.error); + const data = result.data as { truncated: boolean; matches: Array<{ id: string }> }; + assert.equal(data.truncated, false); + assert.deepEqual(data.matches.map((m) => m.id), ['m6']); + }); + + it('surfaces the capability-absent error cleanly', async () => { + const { cm } = buildStub(FIXTURE, { throwUnsupported: true }); + const h = new HistoryModule(); + h.bind(cm); + + const result = await h.handleToolCall(call('search', { query: 'x' })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Chronicle history index unsupported/); + }); + }); + + describe('pagination input validation', () => { + // Math.min(value, max) alone only enforces an UPPER bound — + // Math.min(-1, 200) is -1, not clamped up to anything — so a negative + // (or non-integer/non-finite) limit/offset/maxScan would previously + // reach context-manager's native query unbounded. Each of these must + // now be rejected before the native call is ever made. + const badValues: Array<[string, number]> = [ + ['negative', -1], + ['non-integer', 1.5], + ['NaN', NaN], + ['Infinity', Infinity], + ]; + + for (const [label, value] of badValues) { + it(`extract: rejects a ${label} limit before it reaches context-manager`, async () => { + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('extract', { limit: value })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.equal(calls.length, 0, 'must reject before ever calling context-manager'); + }); + + it(`extract: rejects a ${label} offset before it reaches context-manager`, async () => { + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('extract', { offset: value })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.equal(calls.length, 0); + }); + + it(`search: rejects a ${label} limit before it reaches context-manager`, async () => { + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('search', { query: 'm', limit: value })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.equal(calls.length, 0); + }); + + it(`search: rejects a ${label} maxScan before it reaches context-manager`, async () => { + const { cm, calls } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('search', { query: 'm', maxScan: value })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.equal(calls.length, 0); + }); + } + + it('extract: a real 250-message store with limit:-1 does NOT return everything (reviewer repro)', async () => { + // Reviewer's exact repro shape: a store larger than the hard cap, and + // a negative limit that must not bypass it. + const many = Array.from({ length: 250 }, (_, i) => msg(`p${i}`, 1000 + i, 'User', [textBlock(`m${i}`)], 'c1')); + const { cm } = buildStub(many); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('extract', { limit: -1 })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + }); + + it('search: a real large candidate pool with maxScan:-2 does NOT fetch everything (reviewer repro)', async () => { + const many = Array.from({ length: 248 }, (_, i) => msg(`q${i}`, 1000 + i, 'User', [textBlock(`m${i}`)], 'c1')); + const { cm, calls } = buildStub(many); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('search', { query: 'm', maxScan: -2 })); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.equal(calls.length, 0); + }); + }); + + it('rejects tool calls before bind()', async () => { + const h = new HistoryModule(); + const result = await h.handleToolCall(call('stats', {})); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /not bound/); + }); + + it('rejects an unknown tool name', async () => { + const { cm } = buildStub(FIXTURE); + const h = new HistoryModule(); + h.bind(cm); + const result = await h.handleToolCall(call('bogus', {})); + assert.equal(result.success, false); + assert.equal(result.isError, true); + assert.match(result.error ?? '', /Unknown tool/); + }); +});