From 71c8fb6bbed5bf642a653b27d80e1ee59aeb3f67 Mon Sep 17 00:00:00 2001 From: daewoongoh Date: Fri, 2 Oct 2026 15:41:18 +0900 Subject: [PATCH 1/3] fix(webview): omit originalContent from webview messages and fetch on demand Strip originalContent (the whole pre-edit file) from file-edit tool messages before they are posted to the webview, and let FileChangesPanel request it on demand by messageId and taskId. This keeps the webview heap from growing with the total size of edited files, which could end in a gray screen (OOM). Also adds the scripts/gray-screen heap-measurement tooling. --- packages/types/src/vscode-extension-host.ts | 7 + scripts/gray-screen/analyze-session.mjs | 129 ++++++ scripts/gray-screen/generate-large-task.mjs | 124 ++++++ scripts/gray-screen/mock-openai-server.mjs | 416 ++++++++++++++++++ scripts/gray-screen/webview-heap-matrix.mjs | 341 ++++++++++++++ scripts/gray-screen/webview-render-stress.mjs | 394 +++++++++++++++++ src/core/webview/ClineProvider.ts | 3 +- .../webview/__tests__/ClineProvider.spec.ts | 54 +++ .../__tests__/stripOriginalContent.spec.ts | 245 +++++++++++ ...MessageHandler.readOriginalContent.spec.ts | 170 +++++++ src/core/webview/stripOriginalContent.ts | 82 ++++ src/core/webview/webviewMessageHandler.ts | 24 + .../src/__tests__/FileChangesPanel.spec.tsx | 211 ++++++++- .../__tests__/fileChangesFromMessages.spec.ts | 44 +- webview-ui/src/components/chat/ChatView.tsx | 2 +- .../src/components/chat/FileChangesPanel.tsx | 51 ++- .../chat/utils/fileChangesFromMessages.ts | 11 + 17 files changed, 2294 insertions(+), 14 deletions(-) create mode 100644 scripts/gray-screen/analyze-session.mjs create mode 100644 scripts/gray-screen/generate-large-task.mjs create mode 100644 scripts/gray-screen/mock-openai-server.mjs create mode 100644 scripts/gray-screen/webview-heap-matrix.mjs create mode 100644 scripts/gray-screen/webview-render-stress.mjs create mode 100644 src/core/webview/__tests__/stripOriginalContent.spec.ts create mode 100644 src/core/webview/__tests__/webviewMessageHandler.readOriginalContent.spec.ts create mode 100644 src/core/webview/stripOriginalContent.ts diff --git a/packages/types/src/vscode-extension-host.ts b/packages/types/src/vscode-extension-host.ts index 5f6b579779..b9fdaf5b79 100644 --- a/packages/types/src/vscode-extension-host.ts +++ b/packages/types/src/vscode-extension-host.ts @@ -106,11 +106,14 @@ export interface ExtensionMessage { | "skills" | "rules" | "fileContent" + | "originalContent" | "rooHistoryImportProgress" | "themeFixtureProbeRequest" text?: string /** For fileContent: { path, content, error? } */ fileContent?: { path: string; content: string | null; error?: string } + /** For originalContent: the pre-edit file content of the requested tool message (null when unavailable) */ + originalContentInfo?: { ts: number; messageId?: string; taskId?: string; content: string | null } payload?: any // eslint-disable-line @typescript-eslint/no-explicit-any checkpointWarning?: { type: "WAIT_TIMEOUT" | "INIT_TIMEOUT" @@ -498,6 +501,7 @@ export interface WebviewMessage { | "saveImage" | "openFile" | "readFileContent" + | "readOriginalContent" | "openMention" | "cancelTask" | "cancelAutoApproval" @@ -696,6 +700,7 @@ export interface WebviewMessage { ids?: string[] terminalOperation?: "continue" | "abort" messageTs?: number + messageId?: string restoreCheckpoint?: boolean historyPreviewCollapsed?: boolean filters?: { type?: string; search?: string; tags?: string[] } @@ -858,6 +863,8 @@ export interface ClineSayTool { content?: string // Original file content before first edit (for merged diff display in FileChangesPanel) originalContent?: string + // Length of the originalContent the extension left out of the webview state (request it with readOriginalContent) + originalContentLength?: number // Unified diff statistics computed by the extension diffStats?: { added: number; removed: number } regex?: string diff --git a/scripts/gray-screen/analyze-session.mjs b/scripts/gray-screen/analyze-session.mjs new file mode 100644 index 0000000000..ee395590e8 --- /dev/null +++ b/scripts/gray-screen/analyze-session.mjs @@ -0,0 +1,129 @@ +#!/usr/bin/env node +// Measures what a saved Zoo Code task is made of, to see what the webview has to hold: sizes only, never prints message content. +// +// node scripts/gray-screen/analyze-session.mjs [--storage ] [--top 15] [--task ] [--include-generated true] +// +// Per task: file size, message counts, bytes per message kind and per tool, bytes per JSON field of tool payloads +// (originalContent / content / diff ...), share of strings V8 stores as 2 bytes/char (any char above U+00FF), and the +// longest run of consecutive file-edit asks that ChatView's batchNearby would merge into one message. +import fs from "node:fs" +import os from "node:os" +import path from "node:path" + +const args = Object.fromEntries( + process.argv.slice(2).reduce((acc, cur, i, all) => { + if (cur.startsWith("--")) acc.push([cur.slice(2), all[i + 1]]) + return acc + }, []), +) +const storage = + args["storage"] ?? path.join(os.homedir(), ".vscode-server", "data", "User", "globalStorage", "codemate.zoo-code") +const top = Number(args["top"] ?? 15) +const includeGenerated = args["include-generated"] === "true" +const tasksDir = path.join(storage, "tasks") + +const EDIT_TOOLS = new Set(["editedExistingFile", "appliedDiff", "newFileCreated", "insertContent", "searchAndReplace"]) +const BOUNDARY_SAY = new Set(["user_feedback", "user_feedback_diff", "completion_result", "checkpoint_saved", "error", "condense_context", "codebase_search_result"]) +const NON_LATIN1 = /[^\u0000-ÿ]/ +const mb = (n) => (n / 1048576).toFixed(1) + +const isIgnorable = (m) => m.type === "say" && (m.say === "api_req_started" || (m.say === "text" && !m.text?.trim()) || m.say === "reasoning") +const isBoundary = (m) => m.type === "say" && (BOUNDARY_SAY.has(m.say) || (m.say === "text" && !!m.text?.trim())) + +function analyze(taskId) { + const file = path.join(tasksDir, taskId, "ui_messages.json") + const raw = fs.readFileSync(file, "utf8") + const messages = JSON.parse(raw) + const r = { taskId, fileBytes: Buffer.byteLength(raw), count: messages.length, kinds: {}, tools: {}, fields: {}, imageBytes: 0, twoByteBytes: 0, latin1Bytes: 0, maxRun: 0, maxRunBytes: 0, runs2plus: 0 } + + const add = (obj, key, bytes) => { + const e = (obj[key] ??= { n: 0, bytes: 0 }) + e.n++ + e.bytes += bytes + } + const classify = (str) => { + if (NON_LATIN1.test(str)) r.twoByteBytes += str.length * 2 + else r.latin1Bytes += str.length + } + + for (const m of messages) { + const text = typeof m.text === "string" ? m.text : "" + add(r.kinds, `${m.type}:${m.say ?? m.ask ?? "?"}`, text.length) + if (text) classify(text) + if (Array.isArray(m.images)) for (const img of m.images) r.imageBytes += typeof img === "string" ? img.length : 0 + if (m.type === "ask" && m.ask === "tool" && text) { + try { + const t = JSON.parse(text) + add(r.tools, String(t.tool), text.length) + for (const [k, v] of Object.entries(t)) add(r.fields, k, typeof v === "string" ? v.length : JSON.stringify(v)?.length ?? 0) + } catch {} + } + } + + // longest run of consecutive edit asks that batchNearby merges (same rules as batchNearby.ts) + const isEditAsk = (m) => { + if (m.type !== "ask" || m.ask !== "tool" || !m.text) return false + try { + const t = JSON.parse(m.text) + return EDIT_TOOLS.has(t.tool) && !t.batchDiffs + } catch { + return false + } + } + const items = messages.slice(1) + for (let i = 0; i < items.length; ) { + if (isBoundary(items[i])) { i++; continue } + if (!isEditAsk(items[i])) { i++; continue } + let j = i + 1, len = 1, bytes = items[i].text.length + while (j < items.length) { + if (isBoundary(items[j])) break + if (isEditAsk(items[j])) { len++; bytes += items[j].text.length; j++ } + else if (isIgnorable(items[j])) j++ + else break + } + if (len > 1) r.runs2plus++ + if (len > r.maxRun) { r.maxRun = len; r.maxRunBytes = bytes } + i = j + } + return r +} + +const taskMeta = (id) => { + try { return JSON.parse(fs.readFileSync(path.join(tasksDir, id, "history_item.json"), "utf8")) } catch { return {} } +} +const ids = args["task"] ? [args["task"]] : fs.readdirSync(tasksDir).filter((id) => fs.existsSync(path.join(tasksDir, id, "ui_messages.json"))) +const sizes = ids.map((id) => ({ id, size: fs.statSync(path.join(tasksDir, id, "ui_messages.json")).size, generated: (taskMeta(id).task ?? "").startsWith("[LOAD TEST]") })) +const real = sizes.filter((s) => includeGenerated || !s.generated) + +const sorted = real.map((s) => s.size).sort((a, b) => a - b) +const pct = (p) => sorted[Math.min(sorted.length - 1, Math.floor(sorted.length * p))] ?? 0 +console.log(`storage: ${storage}`) +console.log(`tasks: ${ids.length} (excluded ${sizes.length - real.length} generated [LOAD TEST] tasks${includeGenerated ? " - included" : ""}); analyzed set: ${real.length}`) +console.log(`ui_messages.json size: p50=${mb(pct(0.5))}MB p90=${mb(pct(0.9))}MB p99=${mb(pct(0.99))}MB max=${mb(sorted.at(-1) ?? 0)}MB; over 10MB: ${sorted.filter((s) => s > 10 * 1048576).length}, over 5MB: ${sorted.filter((s) => s > 5 * 1048576).length}`) + +const results = real.sort((a, b) => b.size - a.size).slice(0, top).map((s) => analyze(s.id)) +console.log(`\nTop ${results.length} tasks by ui_messages.json size (sizes only, no content):`) +console.log("size(MB) msgs toolTxt(MB) editTools topField(share) 2byte% imgMB maxBatchRun(bytes MB) task") +for (const r of results) { + const toolBytes = Object.values(r.tools).reduce((a, b) => a + b.bytes, 0) + const editCount = Object.entries(r.tools).filter(([k]) => EDIT_TOOLS.has(k)).reduce((a, [, v]) => a + v.n, 0) + const fieldTotal = Object.values(r.fields).reduce((a, b) => a + b.bytes, 0) || 1 + const [topKey, topVal] = Object.entries(r.fields).sort((a, b) => b[1].bytes - a[1].bytes)[0] ?? ["-", { bytes: 0 }] + const text = r.latin1Bytes + r.twoByteBytes || 1 + const meta = taskMeta(r.taskId) + console.log( + `${mb(r.fileBytes).padStart(7)} ${String(r.count).padStart(5)} ${mb(toolBytes).padStart(9)} ${String(editCount).padStart(9)} ${`${topKey} ${(100 * topVal.bytes / fieldTotal).toFixed(0)}%`.padEnd(30)} ${(100 * r.twoByteBytes / text).toFixed(0).padStart(5)}% ${mb(r.imageBytes).padStart(5)} ${`${r.maxRun} (${mb(r.maxRunBytes)})`.padEnd(20)} ${(meta.task ?? "").slice(0, 28).replace(/\s+/g, " ")} [${r.taskId.slice(0, 8)}]`, + ) +} + +const biggest = results[0] +if (biggest) { + console.log(`\nBreakdown of the largest analyzed task [${biggest.taskId.slice(0, 8)}], ${mb(biggest.fileBytes)}MB, ${biggest.count} messages:`) + const show = (title, obj, n = 8) => { + console.log(` ${title}`) + for (const [k, v] of Object.entries(obj).sort((a, b) => b[1].bytes - a[1].bytes).slice(0, n)) console.log(` ${k.padEnd(28)} n=${String(v.n).padStart(5)} ${mb(v.bytes).padStart(7)}MB`) + } + show("by message kind (text bytes):", biggest.kinds) + show("by tool (payload bytes):", biggest.tools) + show("by tool payload field (string chars):", biggest.fields) +} diff --git a/scripts/gray-screen/generate-large-task.mjs b/scripts/gray-screen/generate-large-task.mjs new file mode 100644 index 0000000000..3ef5df814e --- /dev/null +++ b/scripts/gray-screen/generate-large-task.mjs @@ -0,0 +1,124 @@ +#!/usr/bin/env node +// Generates a synthetic task with thousands of messages to reproduce webview +// memory pressure (gray screen / OOM). Usage: +// node scripts/gray-screen/generate-large-task.mjs [--messages 5000] [--text-bytes 800] [--tool-bytes ] +// [--tool-kind mixed|newFileCreated|appliedDiff|readFile|listFilesRecursive] [--batchable true] [--two-byte true] +// [--image-every 0] [--image-kb 200] [--storage ] [--workspace ] +// Then reload VS Code and open the task named "[LOAD TEST] ..." from history. +import fs from "node:fs" +import os from "node:os" +import path from "node:path" +import crypto from "node:crypto" + +const args = Object.fromEntries( + process.argv.slice(2).reduce((acc, cur, i, all) => { + if (cur.startsWith("--")) acc.push([cur.slice(2), all[i + 1]]) + return acc + }, []), +) + +const messageCount = Number(args["messages"] ?? 5000) +const textBytes = Number(args["text-bytes"] ?? 800) +const toolBytes = Number(args["tool-bytes"] ?? textBytes) +const toolKind = args["tool-kind"] ?? "mixed" +const batchable = args["batchable"] === "true" +const twoByte = args["two-byte"] === "true" // Korean chars in tool text -> UTF-16 strings (2 bytes/char) in V8 +const imageEvery = Number(args["image-every"] ?? 0) +const imageKb = Number(args["image-kb"] ?? 200) +const workspace = args["workspace"] ?? process.cwd() +const storageCandidates = [ + path.join(os.homedir(), ".vscode-server", "data", "User", "globalStorage", "codemate.zoo-code"), + path.join(os.homedir(), ".config", "Code", "User", "globalStorage", "codemate.zoo-code"), +] +const storage = args["storage"] ?? storageCandidates.find((p) => fs.existsSync(p)) ?? storageCandidates[1] + +const taskId = crypto.randomUUID() +const taskDir = path.join(storage, "tasks", taskId) +fs.mkdirSync(taskDir, { recursive: true }) + +const filler = (n, seed) => { + const base = `line ${seed}: ${twoByte ? "\uD55C\uAE00 " : ""}The quick brown fox jumps over the lazy dog. ` + return base.repeat(Math.ceil(n / base.length)).slice(0, n) +} +const fakeImage = (kb) => `data:image/png;base64,${crypto.randomBytes(kb * 768).toString("base64")}` + +// Large tool payloads (new files, diffs) are what the webview re-parses on every streamed chunk. +const KINDS = ["newFileCreated", "appliedDiff", "readFile", "listFilesRecursive"] +const toolPayload = (round) => { + const kind = toolKind === "mixed" ? KINDS[round % KINDS.length] : toolKind + const file = `src/gen/file-${round}.ts` + switch (kind) { + case "newFileCreated": + return { tool: "newFileCreated", path: file, content: filler(toolBytes, round) } + case "appliedDiff": + return { tool: "appliedDiff", path: file, diff: filler(toolBytes, round), diffStats: { added: 12, removed: 4 } } + case "listFilesRecursive": + return { tool: "listFilesRecursive", path: "src", content: filler(Math.min(toolBytes, 2000), round) } + default: + return { tool: "readFile", path: file, content: filler(Math.min(toolBytes, 400), round) } + } +} + +const taskText = `[LOAD TEST] ${messageCount} messages` +const start = Date.now() - messageCount * 1000 +const messages = [{ ts: start, type: "say", say: "text", text: taskText }] +const apiHistory = [{ role: "user", content: [{ type: "text", text: `\n${taskText}\n` }], ts: start }] + +let ts = start + 1 +let round = 0 +while (messages.length < messageCount) { + round++ + const images = imageEvery > 0 && round % imageEvery === 0 ? [fakeImage(imageKb)] : undefined + messages.push({ + ts: ts++, + type: "say", + say: "api_req_started", + text: JSON.stringify({ apiProtocol: "openai", tokensIn: 1000, tokensOut: 200, cost: 0.001 }), + }) + // batchable: tool-only turns (no visible text/feedback between edits), which ChatView merges into one giant batch message + if (!batchable) messages.push({ ts: ts++, type: "say", say: "text", text: filler(textBytes, round), images }) + messages.push({ + ts: ts++, + type: "ask", + ask: "tool", + text: JSON.stringify(toolPayload(round)), + isAnswered: true, + }) + if (!batchable) messages.push({ ts: ts++, type: "say", say: "user_feedback", text: `ok ${round}` }) + apiHistory.push( + { role: "assistant", content: [{ type: "text", text: filler(textBytes, round) }], ts }, + { role: "user", content: [{ type: "text", text: `ok ${round}` }], ts }, + ) +} +messages.length = messageCount + +const historyItem = { + id: taskId, + number: 9999, + ts: Date.now(), + task: taskText, + tokensIn: round * 1000, + tokensOut: round * 200, + totalCost: 0, + workspace, + status: "completed", +} + +// Write every file under a temporary name and rename only after all writes succeeded, so an interrupted run +// (or a full disk) never leaves a truncated or partial task; history_item.json goes last. +const taskFiles = [ + ["ui_messages.json", messages], + ["api_conversation_history.json", apiHistory], + ["history_item.json", historyItem], +] +try { + for (const [name, value] of taskFiles) fs.writeFileSync(path.join(taskDir, `${name}.tmp`), JSON.stringify(value)) + for (const [name] of taskFiles) fs.renameSync(path.join(taskDir, `${name}.tmp`), path.join(taskDir, name)) +} catch (error) { + fs.rmSync(taskDir, { recursive: true, force: true }) + throw error +} + +const mb = (f) => (fs.statSync(path.join(taskDir, f)).size / 1048576).toFixed(1) +console.log(`Task ${taskId}: ${messages.length} messages, ui_messages.json ${mb("ui_messages.json")} MB`) +console.log(`Written to ${taskDir}\nReload VS Code, then open "${taskText}" from history.`) diff --git a/scripts/gray-screen/mock-openai-server.mjs b/scripts/gray-screen/mock-openai-server.mjs new file mode 100644 index 0000000000..9a3b8c7cff --- /dev/null +++ b/scripts/gray-screen/mock-openai-server.mjs @@ -0,0 +1,416 @@ +#!/usr/bin/env node +// LLM-free OpenAI-compatible server for reproducing Zoo Code sessions in real VS Code (gray screen / OOM work). +// Every request gets a scripted reply: optional streamed reasoning, a markdown explanation (headings, lists, tables, +// fenced code, optional mermaid) and ONE native tool call. After --max-requests turns it ends with attempt_completion. +// +// Usage (run on the machine hosting the extension host; sv01 for Remote-SSH): +// node scripts/gray-screen/mock-openai-server.mjs [--scenario rapid] [--port 8989] [--max-requests 500] [options] +// +// Scenarios (--scenario): +// rapid Turns as fast as possible: ~300B text, no reasoning, no delays, tiny update_todo_list call. +// Each new message makes the host push the full state to the webview. Use with a pre-generated big task +// (generate-large-task.mjs ... --batchable true) and Resume. This is the gray-screen reproduction. +// md ~15KB markdown (+ reasoning) per turn with a tiny update_todo_list call (isolates markdown/streaming cost). +// flood Like md but ~60KB in 4-16 char chunks every ~1ms (maximum partial-message updates). +// coding Realistic coding loop in a sandbox dir: update_todo_list, write_to_file, read_file, apply_diff, +// search_files, test-runner execute_command, list_files (real files/commands run; see --dir). +// chatty Only execute_command with ANSI-colored streaming output (no text). +// churn Phase 1 (--accumulate-turns) piles up large write_to_file/read_file/apply_diff messages, phase 2 is md. +// --fast true: 100 turns of ~55KB write_to_file, 4ms chunks. +// +// Options (defaults vary per scenario): +// --text-bytes N markdown size per turn (3000; md/churn 15000; flood 60000; rapid 300) +// --reasoning-bytes N streamed reasoning_content per turn (600; flood 3000; rapid 0) +// --chunk-ms N delay between streamed chunks (15; flood 1; rapid 0; --fast 4) +// --chunk-min/--chunk-max N characters per streamed chunk (8/47; flood 4/16) +// --tool-chunk N characters per tool-call argument chunk (120) +// --code-lines N lines per generated source file (150; churn 500; --fast 1500) +// --cmd-lines N lines printed by the simulated test/build runner (400) +// --cmd-sleep S seconds between runner lines (0.005) +// --mermaid-every N mermaid diagram every N turns (10; rapid 0; 0 disables) +// --dir PATH sandbox folder relative to the workspace (.mock-session); delete it afterwards +// +// Zoo Code settings: provider "OpenAI Compatible", base URL http://127.0.0.1:/v1, any API key, model id "mock", +// context window 1000000 (avoid condensing); auto-approve read/write/execute with allowed command "*". +import http from "node:http" + +const args = Object.fromEntries( + process.argv.slice(2).reduce((acc, cur, i, all) => { + if (cur.startsWith("--")) acc.push([cur.slice(2), all[i + 1]]) + return acc + }, []), +) +const fast = args["fast"] === "true" +const cfg = { + fast, + port: Number(args["port"] ?? 8989), + maxRequests: Number(args["max-requests"] ?? 500), + scenario: args["scenario"] ?? "coding", + textBytes: Number(args["text-bytes"] ?? (args["scenario"] === "rapid" ? 300 : args["scenario"] === "flood" ? 60000 : args["scenario"] === "md" || args["scenario"] === "churn" ? 15000 : 3000)), + reasoningBytes: Number(args["reasoning-bytes"] ?? (args["scenario"] === "rapid" ? 0 : args["scenario"] === "flood" ? 3000 : 600)), + chunkMs: Number(args["chunk-ms"] ?? (args["scenario"] === "rapid" ? 0 : args["scenario"] === "flood" ? 1 : fast ? 4 : 15)), + chunkMin: Number(args["chunk-min"] ?? (args["scenario"] === "flood" ? 4 : 8)), + chunkMax: Number(args["chunk-max"] ?? (args["scenario"] === "flood" ? 16 : 47)), + toolChunk: Number(args["tool-chunk"] ?? 120), + codeLines: Number(args["code-lines"] ?? (fast ? 1500 : args["scenario"] === "churn" ? 500 : 150)), + accumulateTurns: Number(args["accumulate-turns"] ?? (fast ? 100 : 300)), + cmdLines: Number(args["cmd-lines"] ?? 400), + cmdSleep: Number(args["cmd-sleep"] ?? 0.005), + mermaidEvery: Number(args["mermaid-every"] ?? (args["scenario"] === "rapid" ? 0 : 10)), + dir: (args["dir"] ?? ".mock-session").replace(/\/+$/, ""), +} + +// cfg.dir is interpolated into the shell commands the mock asks the extension to run, so only plain relative paths are allowed. +if (!/^[A-Za-z0-9._-]+(\/[A-Za-z0-9._-]+)*$/.test(cfg.dir) || cfg.dir.split("/").some((part) => part === "..")) { + console.error(`Invalid --dir "${cfg.dir}": use a relative path of letters, digits, ".", "_" and "-" without ".." segments.`) + process.exit(1) +} + +const sleep = (ms) => new Promise((r) => setTimeout(r, ms)) +let requestCount = 0 + +// --------------------------------------------------------------------------------------------- +// Deterministic pseudo-random helpers (same turn number -> same content) +// --------------------------------------------------------------------------------------------- +const rng = (seed) => { + let s = (seed * 2654435761) >>> 0 + return () => { + s = (s + 0x6d2b79f5) >>> 0 + let t = s + t = Math.imul(t ^ (t >>> 15), t | 1) + t ^= t + Math.imul(t ^ (t >>> 7), t | 61) + return ((t ^ (t >>> 14)) >>> 0) / 4294967296 + } +} +const pick = (r, arr) => arr[Math.floor(r() * arr.length)] + +const NOUNS = ["cache", "scheduler", "parser", "router", "session", "queue", "tokenizer", "pipeline", "registry", "validator", "watcher", "emitter"] +const VERBS = ["normalize", "resolve", "dispatch", "compress", "merge", "retry", "flush", "hydrate", "serialize", "throttle"] +const feature = (k) => `${NOUNS[k % NOUNS.length]}_${Math.floor(k / NOUNS.length)}` +const camel = (s) => s.replace(/_(\w)/g, (_, c) => c.toUpperCase()) +const pascal = (s) => camel(s).replace(/^\w/, (c) => c.toUpperCase()) + +// --------------------------------------------------------------------------------------------- +// Generated source files (the server knows exact contents so apply_diff blocks match) +// --------------------------------------------------------------------------------------------- +const files = new Map() // path -> { k, rev, retry } + +function fileContent(k, rev, retry) { + const name = feature(k) + const r = rng(k + 7) + const lines = [ + `// Auto-generated module: ${name}`, + ``, + `export const REVISION = ${rev}`, + `export const RETRY_LIMIT = ${retry}`, + ``, + `export interface ${pascal(name)}Options {`, + `\tname: string`, + `\tmaxItems?: number`, + `\ttimeoutMs?: number`, + `}`, + ``, + ] + let fn = 0 + while (lines.length < cfg.codeLines) { + const v = pick(r, VERBS) + lines.push( + `/** ${v} the ${name} payload, attempt ${fn} */`, + `export async function ${v}${pascal(name)}${fn}(input: string[], opts: ${pascal(name)}Options): Promise {`, + `\tconst out: string[] = []`, + `\tfor (const [index, item] of input.entries()) {`, + `\t\tif (opts.maxItems !== undefined && index >= opts.maxItems) break`, + `\t\tconst trimmed = item.trim().toLowerCase()`, + `\t\tif (trimmed.length === 0) continue`, + `\t\tout.push(\`\${opts.name}:\${trimmed}:\${index * ${fn + 2}}\`)`, + `\t}`, + `\treturn out`, + `}`, + ``, + ) + fn++ + } + return lines.join("\n") + "\n" +} + +const srcPath = (k) => `${cfg.dir}/src/${feature(k)}.ts` + +// --------------------------------------------------------------------------------------------- +// Markdown explanation generator +// --------------------------------------------------------------------------------------------- +const PARAS = [ + "I looked at how the current code paths interact and there are a few things worth calling out before making changes.", + "The existing implementation works for the happy path, but the error handling around retries is inconsistent, so I want to tighten that up first.", + "Before touching anything else, I'll keep the change small and verify it with the test runner so we can see regressions early.", + "This keeps the public surface unchanged while making the internals easier to reason about and to test in isolation.", + "One thing to watch: the `timeoutMs` option is optional, so every consumer has to handle `undefined` explicitly.", + "The trade-off here is a little more code in exchange for much clearer failure modes when the upstream service is slow.", +] +const BULLETS = [ + "Keep `REVISION` monotonically increasing so consumers can detect stale caches", + "Guard against empty input before iterating (`trimmed.length === 0`)", + "Prefer `for...of` with `entries()` over index loops for readability", + "Add a regression test for the `maxItems` boundary", + "Document the retry policy in the module header", + "Avoid mutating the input array; return a fresh `out` instead", +] + +function codeBlock(r, k) { + const name = feature(k) + const kind = pick(r, ["ts", "py", "bash", "json", "diff", "sql", "yaml"]) + switch (kind) { + case "py": + return `\`\`\`python\ndef ${name}_summary(items: list[str], limit: int = 10) -> dict[str, int]:\n counts: dict[str, int] = {}\n for item in items[:limit]:\n key = item.strip().lower()\n counts[key] = counts.get(key, 0) + 1\n return counts\n\`\`\`` + case "bash": + return `\`\`\`bash\n# run the focused suite\npnpm --dir src exec vitest run core/${name} --reporter=verbose\npnpm lint && pnpm check-types\n\`\`\`` + case "json": + return `\`\`\`json\n{\n "name": "${name}",\n "retry": { "limit": ${3 + (k % 4)}, "backoffMs": [100, 250, 500] },\n "features": ["cache", "metrics", "tracing"]\n}\n\`\`\`` + case "diff": + return `\`\`\`diff\n--- a/${srcPath(k)}\n+++ b/${srcPath(k)}\n@@ -3,2 +3,2 @@\n-export const REVISION = 0\n-export const RETRY_LIMIT = 3\n+export const REVISION = 1\n+export const RETRY_LIMIT = 5\n\`\`\`` + case "sql": + return `\`\`\`sql\nSELECT id, status, COUNT(*) AS attempts\nFROM ${name}_jobs\nWHERE created_at > NOW() - INTERVAL '1 day'\nGROUP BY id, status\nORDER BY attempts DESC\nLIMIT 50;\n\`\`\`` + case "yaml": + return `\`\`\`yaml\nservice: ${name}\nreplicas: ${2 + (k % 3)}\nenv:\n - name: LOG_LEVEL\n value: info\n - name: RETRY_LIMIT\n value: "${3 + (k % 4)}"\n\`\`\`` + default: + return `\`\`\`ts\nexport function ${camel(name)}Key(id: string, revision = REVISION): string {\n\treturn \`${name}:\${id}:v\${revision}\`\n}\n\nconst result = await ${pick(r, VERBS)}${pascal(name)}0(["a", " B ", ""], { name: "${name}", maxItems: 2 })\nconsole.log(result) // ["${name}:a:0", "${name}:b:2"]\n\`\`\`` + } +} + +function mermaidBlock(k) { + const a = feature(k) + const b = feature(k + 1) + return `\`\`\`mermaid\nflowchart TD\n A[Request] --> B{${a} cache hit?}\n B -- yes --> C[Return cached]\n B -- no --> D[${b} resolver]\n D --> E[(Store)]\n E --> C\n\`\`\`` +} + +function table(r, k) { + const rows = [["Module", "Lines", "Status"]] + for (let i = 0; i < 3 + Math.floor(r() * 3); i++) rows.push([`\`${feature(k - i < 0 ? 0 : k - i)}.ts\``, String(cfg.codeLines + i), pick(r, ["ok", "updated", "new"])]) + return [`| ${rows[0].join(" | ")} |`, `| --- | ---: | --- |`, ...rows.slice(1).map((row) => `| ${row.join(" | ")} |`)].join("\n") +} + +function markdownFor(n, k, action) { + const r = rng(n * 31 + 5) + const parts = [`## Step ${n}: ${action.title}`, "", pick(r, PARAS), ""] + let size = parts.join("\n").length + const sections = [ + () => ["**Key points**", "", ...Array.from({ length: 3 + Math.floor(r() * 3) }, () => `- ${pick(r, BULLETS)}`), ""].join("\n"), + () => [pick(r, PARAS), "", codeBlock(r, k), ""].join("\n"), + () => [table(r, k), ""].join("\n"), + () => [`Here is the relevant snippet from \`${srcPath(k)}\`:`, "", codeBlock(r, k), "", pick(r, PARAS), ""].join("\n"), + () => [`> **Note:** ${pick(r, PARAS)}`, ""].join("\n"), + () => [`1. Inspect \`${srcPath(k)}\``, `2. ${pick(r, BULLETS)}`, `3. Re-run the suite and compare output`, ""].join("\n"), + ] + if (cfg.mermaidEvery > 0 && n % cfg.mermaidEvery === 0) { + parts.push("The data flow looks like this:", "", mermaidBlock(k), "") + size += 200 + } + const targetBytes = action.textBytes ?? cfg.textBytes + while (size < targetBytes) { + const s = pick(r, sections)() + parts.push(s) + size += s.length + } + parts.push(action.closing) + return parts.join("\n") +} + +function reasoningFor(n, skip) { + if (skip || cfg.reasoningBytes <= 0) return "" + const r = rng(n * 13 + 1) + let out = "" + while (out.length < cfg.reasoningBytes) out += `${pick(r, PARAS)} ${pick(r, BULLETS)}. ` + return out.slice(0, cfg.reasoningBytes) +} + +// --------------------------------------------------------------------------------------------- +// Simulated commands (ANSI colors, streaming output, stack traces) +// --------------------------------------------------------------------------------------------- +const sleepPart = () => (cfg.cmdSleep > 0 ? `sleep ${cfg.cmdSleep}; ` : "") + +const testRunCommand = (k) => + `echo "RUN v2.1.0 ${cfg.dir}"; for i in $(seq 1 ${cfg.cmdLines}); do printf "\\033[32m ✓\\033[0m src/${feature(k)} case %d \\033[90m(%d ms)\\033[0m\\n" $i $((i % 37 + 3)); ${sleepPart()}done; printf "\\n\\033[32m Test Files 1 passed (1)\\033[0m\\n\\033[32m Tests ${cfg.cmdLines} passed (${cfg.cmdLines})\\033[0m\\n"` + +const buildErrorCommand = (k) => + `for i in $(seq 1 ${Math.max(20, Math.floor(cfg.cmdLines / 4))}); do printf "\\033[31merror\\033[0m TS2322: Type 'string' is not assignable to type 'number'.\\n \\033[36m${cfg.dir}/src/${feature(k)}.ts\\033[0m:%d:%d\\n %d | const value: number = input[%d]\\n\\n" $i $((i % 80 + 1)) $i $i; ${sleepPart()}done; echo "Found ${Math.max(20, Math.floor(cfg.cmdLines / 4))} errors."` + +const listCommand = () => `find ${cfg.dir} -type f -name "*.ts" | sort | head -100 && wc -l ${cfg.dir}/src/*.ts | tail -5` + +// --------------------------------------------------------------------------------------------- +// Turn planner: repeating coding cycle over successive "features" +// --------------------------------------------------------------------------------------------- +const CYCLE = ["todo", "write", "read", "apply_diff", "search", "test", "apply_diff", "test", "list", "build_check"] + +function planFor(n) { + if (n > cfg.maxRequests) { + return { + title: "Wrap up", + closing: "All planned changes are in place.", + tool: { name: "attempt_completion", args: { result: `## Summary\n\nFinished ${cfg.maxRequests} scripted turns in \`${cfg.dir}/\`.\n\n- Created and revised generated modules\n- Ran the simulated test suite repeatedly\n\nRun \`rm -rf ${cfg.dir}\` to clean up.` } }, + } + } + if (cfg.scenario === "churn") { + if (n <= cfg.accumulateTurns) { + // Phase 1: pile up large tool messages (new files, diffs, reads) with minimal text, quickly. + const k = cfg.fast ? n - 1 : Math.floor((n - 1) / 3) + const path = srcPath(k) + const step = cfg.fast ? 0 : (n - 1) % 3 + const fast = { fast: true, textBytes: 200 } + if (step === 0) { + files.set(path, { k, rev: 0, retry: 3 }) + return { ...fast, title: `Create \`${feature(k)}.ts\``, closing: `Creating \`${path}\`.`, tool: { name: "write_to_file", args: { path, content: fileContent(k, 0, 3) } } } + } + if (step === 1) return { ...fast, title: "Review the new file", closing: `Reading \`${path}\`.`, tool: { name: "read_file", args: { path } } } + const f = files.get(path) ?? { k, rev: 0, retry: 3 } + const next = { k, rev: f.rev + 1, retry: f.retry + 1 } + const diff = `<<<<<<< SEARCH\n:start_line:3\n-------\nexport const REVISION = ${f.rev}\nexport const RETRY_LIMIT = ${f.retry}\n=======\nexport const REVISION = ${next.rev}\nexport const RETRY_LIMIT = ${next.retry}\n>>>>>>> REPLACE` + files.set(path, next) + return { ...fast, title: "Apply a targeted fix", closing: `Applying the change to \`${path}\`.`, tool: { name: "apply_diff", args: { path, diff } } } + } + // Phase 2: long streamed markdown with a tiny tool call; every chunk re-derives state from all the tool messages above. + if (n === cfg.accumulateTurns + 1) { + console.log("[mock] ===== phase 2: streaming long markdown; watch the webview heap peak now =====") + } + return { title: "Notes", closing: "Updating the checklist.", tool: { name: "update_todo_list", args: { todos: `[-] Review notes for turn ${n}\n[ ] Summarize findings` } } } + } + if (cfg.scenario === "md" || cfg.scenario === "flood" || cfg.scenario === "rapid") { + const todos = `[-] Review notes for turn ${n}\n[ ] Summarize findings` + return { title: "Notes", closing: "Updating the checklist.", tool: { name: "update_todo_list", args: { todos } } } + } + if (cfg.scenario === "chatty") { + const k = n + return { title: "Run the suite", closing: "Running it again.", tool: { name: "execute_command", args: { command: testRunCommand(k), cwd: null, timeout: null } } } + } + + const cycleIndex = (n - 1) % CYCLE.length + const k = Math.floor((n - 1) / CYCLE.length) + const kind = CYCLE[cycleIndex] + const path = srcPath(k) + + switch (kind) { + case "todo": { + const done = Math.min(k, 6) + const todos = Array.from({ length: 8 }, (_, i) => `${i < done ? "[x]" : i === done ? "[-]" : "[ ]"} Implement and verify ${feature(i)}`).join("\n") + return { title: "Update the plan", closing: "Updating the checklist now.", tool: { name: "update_todo_list", args: { todos } } } + } + case "write": { + const retry = 3 + files.set(path, { k, rev: 0, retry }) + return { title: `Create \`${feature(k)}.ts\``, closing: `Creating \`${path}\`.`, tool: { name: "write_to_file", args: { path, content: fileContent(k, 0, retry) } } } + } + case "read": + return { title: "Review the new file", closing: `Reading \`${path}\` back to double-check it.`, tool: { name: "read_file", args: { path } } } + case "apply_diff": { + const f = files.get(path) ?? { k, rev: 0, retry: 3 } + const next = { k, rev: f.rev + 1, retry: f.retry + 1 } + const diff = `<<<<<<< SEARCH\n:start_line:3\n-------\nexport const REVISION = ${f.rev}\nexport const RETRY_LIMIT = ${f.retry}\n=======\nexport const REVISION = ${next.rev}\nexport const RETRY_LIMIT = ${next.retry}\n>>>>>>> REPLACE` + files.set(path, next) + return { title: "Apply a targeted fix", closing: `Applying the change to \`${path}\`.`, tool: { name: "apply_diff", args: { path, diff } } } + } + case "search": + return { title: "Find related usages", closing: "Searching the sandbox for other consumers.", tool: { name: "search_files", args: { path: `${cfg.dir}/src`, regex: "RETRY_LIMIT|REVISION", file_pattern: "*.ts" } } } + case "test": + return { title: "Run the tests", closing: "Running the suite.", tool: { name: "execute_command", args: { command: testRunCommand(k), cwd: null, timeout: null } } } + case "list": + return { title: "Check the workspace layout", closing: "Listing what we have so far.", tool: { name: "list_files", args: { path: cfg.dir, recursive: true } } } + default: + return { title: "Type-check and summarize", closing: "Collecting the diagnostics.", tool: { name: "execute_command", args: { command: k % 2 === 0 ? buildErrorCommand(k) : listCommand(), cwd: null, timeout: null } } } + } +} + +// --------------------------------------------------------------------------------------------- +// OpenAI-compatible streaming +// --------------------------------------------------------------------------------------------- +function sseChunk(res, id, delta, finishReason = null) { + res.write(`data: ${JSON.stringify({ id, object: "chat.completion.chunk", created: Math.floor(Date.now() / 1000), model: "mock", choices: [{ index: 0, delta, finish_reason: finishReason }] })}\n\n`) +} + +async function streamText(res, id, text, field, chunkMs = cfg.chunkMs) { + const r = rng(text.length) + for (let i = 0; i < text.length; ) { + const size = cfg.chunkMin + Math.floor(r() * (cfg.chunkMax - cfg.chunkMin + 1)) + sseChunk(res, id, { [field]: text.slice(i, i + size) }) + i += size + if (chunkMs > 0) await sleep(chunkMs * (0.5 + r())) + } +} + +async function handleCompletion(req, res, body) { + const messages = Array.isArray(body.messages) ? body.messages : [] + const hasTools = Array.isArray(body.tools) && body.tools.length > 0 + const id = `chatcmpl-mock-${Date.now()}` + + // Requests without tools (condense, prompt enhance, title generation): plain answer, not part of the script. + if (!hasTools) { + console.log(`[mock] auxiliary request messages=${messages.length} stream=${body.stream === true}`) + const text = "## Summary\n\nThe conversation so far covered creating, reviewing and revising generated modules.\n" + if (body.stream !== true) { + res.writeHead(200, { "content-type": "application/json" }) + res.end(JSON.stringify({ id, object: "chat.completion", created: Math.floor(Date.now() / 1000), model: "mock", choices: [{ index: 0, message: { role: "assistant", content: text }, finish_reason: "stop" }], usage: { prompt_tokens: 100, completion_tokens: 30, total_tokens: 130 } })) + return + } + res.writeHead(200, { "content-type": "text/event-stream", "cache-control": "no-cache", connection: "keep-alive" }) + sseChunk(res, id, { role: "assistant", content: text }) + sseChunk(res, id, {}, "stop") + res.write(`data: ${JSON.stringify({ id, object: "chat.completion.chunk", created: 0, model: "mock", choices: [], usage: { prompt_tokens: 100, completion_tokens: 30, total_tokens: 130 } })}\n\n`) + res.write("data: [DONE]\n\n") + res.end() + return + } + + const n = ++requestCount + const plan = planFor(n) + console.log(`[mock] turn #${n} tool=${plan.tool.name} messages=${messages.length} bodyBytes=${JSON.stringify(body).length}`) + + if (body.stream !== true) { + res.writeHead(200, { "content-type": "application/json" }) + res.end(JSON.stringify({ id, object: "chat.completion", created: Math.floor(Date.now() / 1000), model: "mock", choices: [{ index: 0, message: { role: "assistant", content: markdownFor(n, Math.floor((n - 1) / CYCLE.length), plan), tool_calls: [{ id: `call_${n}`, type: "function", function: { name: plan.tool.name, arguments: JSON.stringify(plan.tool.args) } }] }, finish_reason: "tool_calls" }], usage: { prompt_tokens: 100, completion_tokens: 30, total_tokens: 130 } })) + return + } + + res.writeHead(200, { "content-type": "text/event-stream", "cache-control": "no-cache", connection: "keep-alive" }) + sseChunk(res, id, { role: "assistant", content: "" }) + const reasoning = reasoningFor(n, plan.fast) + if (reasoning) await streamText(res, id, reasoning, "reasoning_content") + if (cfg.scenario !== "chatty") await streamText(res, id, markdownFor(n, Math.floor((n - 1) / CYCLE.length), plan), "content", plan.fast ? 1 : cfg.chunkMs) + + const argsJson = JSON.stringify(plan.tool.args) + sseChunk(res, id, { tool_calls: [{ index: 0, id: `call_${n}`, type: "function", function: { name: plan.tool.name, arguments: "" } }] }) + for (let i = 0; i < argsJson.length; i += cfg.toolChunk) { + sseChunk(res, id, { tool_calls: [{ index: 0, function: { arguments: argsJson.slice(i, i + cfg.toolChunk) } }] }) + if (cfg.chunkMs > 0) await sleep(Math.max(1, cfg.chunkMs / 3)) + } + sseChunk(res, id, {}, "tool_calls") + const promptTokens = 1000 + messages.length * 400 + res.write(`data: ${JSON.stringify({ id, object: "chat.completion.chunk", created: Math.floor(Date.now() / 1000), model: "mock", choices: [], usage: { prompt_tokens: promptTokens, completion_tokens: 400, total_tokens: promptTokens + 400 } })}\n\n`) + res.write("data: [DONE]\n\n") + res.end() +} + +const server = http.createServer((req, res) => { + const url = new URL(req.url, "http://x") + if (req.method === "GET" && url.pathname.endsWith("/models")) { + res.writeHead(200, { "content-type": "application/json" }) + res.end(JSON.stringify({ object: "list", data: [{ id: "mock", object: "model", created: 0, owned_by: "mock" }] })) + return + } + if (req.method === "POST" && url.pathname.endsWith("/chat/completions")) { + const chunks = [] + req.on("data", (c) => chunks.push(c)) + req.on("end", () => { + let body = {} + try { + body = JSON.parse(Buffer.concat(chunks).toString("utf8")) + } catch {} + handleCompletion(req, res, body).catch((e) => { + console.error("[mock] handler error:", e) + res.end() + }) + }) + return + } + res.writeHead(404).end() +}) + +server.listen(cfg.port, "0.0.0.0", () => { + console.log(`[mock] OpenAI-compatible server listening on 0.0.0.0:${cfg.port} -> use http://127.0.0.1:${cfg.port}/v1 (scenario=${cfg.scenario}, maxRequests=${cfg.maxRequests}, sandbox=${cfg.dir}/)`) +}) diff --git a/scripts/gray-screen/webview-heap-matrix.mjs b/scripts/gray-screen/webview-heap-matrix.mjs new file mode 100644 index 0000000000..67e944e51b --- /dev/null +++ b/scripts/gray-screen/webview-heap-matrix.mjs @@ -0,0 +1,341 @@ +#!/usr/bin/env node +// Finds which update pattern grows the webview JS heap fastest, in real headless Chromium with the real +// webview-ui (production build in /tmp/zoo-webview-stress-build; run scripts/gray-screen/webview-render-stress.mjs once to build it). +// Each experiment reloads the page (fresh heap), hydrates a history of large tool messages, then drives updates +// for --seconds while sampling JSHeapUsedSize every 100ms. The hydration peak is reported separately (hydrPk); +// peakMB covers only the driven updates after a forced GC (baseMB). +// +// node scripts/gray-screen/webview-heap-matrix.mjs [--seconds 20] [--heap-mb 4096] [--only name1,name2] +import fs from "node:fs" +import http from "node:http" +import { createRequire } from "node:module" +import path from "node:path" +import { fileURLToPath } from "node:url" + +const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..") +const require = createRequire(path.join(root, "webview-ui", "package.json")) +const { chromium } = require("@playwright/test") + +const args = Object.fromEntries( + process.argv.slice(2).reduce((acc, cur, i, all) => { + if (cur.startsWith("--")) acc.push([cur.slice(2), all[i + 1]]) + return acc + }, []), +) +const seconds = Number(args["seconds"] ?? 20) +const heapMb = Number(args["heap-mb"] ?? 4096) +const only = args["only"]?.split(",") +const buildDir = args["build-dir"] ?? "/tmp/zoo-webview-stress-build" +if (!fs.existsSync(path.join(buildDir, "index.html"))) { + console.error(`No build at ${buildDir}. Run: node scripts/gray-screen/webview-render-stress.mjs --scenario session --start 100 --pushes 1`) + process.exit(1) +} + +// mode: stream = messageUpdated chunks only | turns = new messages + full state push per turn | both = turns + chunks +// Find the smallest crashing data size, or lower the cap with --heap-mb to see the live-set boundary. +const EXPERIMENTS = [ + { name: "baseline-small-stream", mode: "stream", toolMb: 0.5, hz: 60 }, + { name: "T100-stream-30hz", mode: "stream", toolMb: 100, hz: 30 }, + { name: "T100-turns-2hz", mode: "turns", toolMb: 100, hz: 2 }, + // batchable: tool-only turns -> ChatView merges consecutive edit asks into one giant message (~2x memory) + { name: "T50-batch-turns-2hz", mode: "turns", toolMb: 50, hz: 2, batchable: true }, + { name: "T100-batch-turns-2hz", mode: "turns", toolMb: 100, hz: 2, batchable: true }, + { name: "T200-batch-turns-2hz", mode: "turns", toolMb: 200, hz: 2, batchable: true }, + { name: "T300-batch-turns-2hz", mode: "turns", toolMb: 300, hz: 2, batchable: true }, // crashes at the default 4096MB cap + { name: "T700-turns-2hz", mode: "turns", toolMb: 700, hz: 2 }, // crashes without batching + // realistic appliedDiff messages (diff + patch + whole pre-edit file of origBytes): originalContent inline (before) vs omitted (now) + { name: "E2000-orig50k-inline-turns-2hz", mode: "turns", tools: 2000, origBytes: 50000, hz: 2 }, + { name: "E2000-orig50k-omitted-turns-2hz", mode: "turns", tools: 2000, origBytes: 50000, omitOriginal: true, hz: 2 }, + { name: "E2000-orig50k-inline-batch-turns-2hz", mode: "turns", tools: 2000, origBytes: 50000, batchable: true, hz: 2 }, + { name: "E2000-orig50k-omitted-batch-turns-2hz", mode: "turns", tools: 2000, origBytes: 50000, omitOriginal: true, batchable: true, hz: 2 }, + { name: "E2000-orig50k-inline-stream-30hz", mode: "stream", tools: 2000, origBytes: 50000, hz: 30 }, + { name: "E2000-orig50k-omitted-stream-30hz", mode: "stream", tools: 2000, origBytes: 50000, omitOriginal: true, hz: 30 }, + { name: "E6000-orig50k-inline-turns-2hz", mode: "turns", tools: 6000, origBytes: 50000, hz: 2 }, + { name: "E6000-orig50k-omitted-turns-2hz", mode: "turns", tools: 6000, origBytes: 50000, omitOriginal: true, hz: 2 }, + // async: pushes come from a Worker at a fixed rate regardless of how fast the webview processes them (like the real extension host) + { name: "T50-batch-async-2hz", mode: "async", toolMb: 50, hz: 2, batchable: true }, + { name: "T100-batch-async-2hz", mode: "async", toolMb: 100, hz: 2, batchable: true }, + { name: "T100-batch-async-5hz", mode: "async", toolMb: 100, hz: 5, batchable: true }, + { name: "T100-async-2hz", mode: "async", toolMb: 100, hz: 2 }, + { name: "T200-batch-async-2hz", mode: "async", toolMb: 200, hz: 2, batchable: true }, + { name: "T300-batch-async-2hz", mode: "async", toolMb: 300, hz: 2, batchable: true }, + { name: "T200-batch-async-5hz", mode: "async", toolMb: 200, hz: 5, batchable: true }, + // twoByte: tool text contains Korean chars -> UTF-16 strings (2 bytes/char) for the same character count + { name: "T50-batch-twobyte-2hz", mode: "turns", toolMb: 50, hz: 2, batchable: true, twoByte: true }, + { name: "T100-batch-twobyte-2hz", mode: "turns", toolMb: 100, hz: 2, batchable: true, twoByte: true }, + { name: "T150-batch-twobyte-2hz", mode: "turns", toolMb: 150, hz: 2, batchable: true, twoByte: true }, + { name: "T100-turns-2hz+ballast2500", mode: "turns", toolMb: 100, hz: 2, ballastMb: 2500 }, // artificial retained heap +] + +const types = { ".html": "text/html", ".js": "text/javascript", ".css": "text/css", ".json": "application/json", ".wasm": "application/wasm", ".svg": "image/svg+xml", ".map": "application/json", ".woff2": "font/woff2", ".ttf": "font/ttf" } +const serveRoot = fs.realpathSync(buildDir) +// Resolves a request path to a regular file inside serveRoot (after symlink resolution), or undefined. +const resolveServedFile = (rawPath) => { + try { + const decoded = decodeURIComponent(rawPath) + const real = fs.realpathSync(path.resolve(serveRoot, "." + (decoded === "/" ? "/index.html" : decoded))) + const rel = path.relative(serveRoot, real) + if (rel === "" || rel === ".." || rel.startsWith(".." + path.sep) || path.isAbsolute(rel)) return undefined + return fs.statSync(real).isFile() ? real : undefined + } catch { + return undefined + } +} +const httpServer = http.createServer((req, res) => { + const file = resolveServedFile(new URL(req.url, "http://x").pathname) + if (!file) return void res.writeHead(404).end() + res.writeHead(200, { + "content-type": types[path.extname(file)] ?? "application/octet-stream", + "cross-origin-opener-policy": "same-origin", + "cross-origin-embedder-policy": "require-corp", + }) + fs.createReadStream(file).pipe(res) +}) +await new Promise((r) => httpServer.listen(0, "127.0.0.1", r)) +const url = `http://127.0.0.1:${httpServer.address().port}/` + +const initScript = () => { + window.acquireVsCodeApi = () => ({ postMessage: () => {}, getState: () => undefined, setState: () => {} }) + window.IMAGES_BASE_URI = "/" + window.MATERIAL_ICONS_BASE_URI = "/" +} + +const pageDriver = async ({ mode, toolMb, hz, chunkHz, seconds, ballastMb, batchable, twoByte, tools, origBytes, omitOriginal }) => { + const sleep = (ms) => new Promise((r) => setTimeout(r, ms)) + // twoByte: one non-Latin1 char makes V8 store the whole string as UTF-16 (2 bytes/char), like Korean comments/logs. + const filler = twoByte + ? (n, seed) => { + const base = `line ${seed}: \uD55C\uAE00 The quick brown fox jumps over the lazy dog. ` + return base.repeat(Math.ceil(n / base.length)).slice(0, n) + } + : (n, seed) => `line ${seed}: The quick brown fox jumps over the lazy dog. `.repeat(Math.ceil(n / 50)).slice(0, n) + const kinds = ["newFileCreated", "appliedDiff"] + const bigBytes = 50000 + const toolCount = tools ?? Math.max(1, Math.round((toolMb * 1048576) / bigBytes)) + let ts = Date.now() - 1e6 + let seq = 0 + const messages = [{ ts: ts++, type: "say", say: "text", text: "[STRESS] heap matrix" }] + const addTurn = (i, big) => { + messages.push({ ts: ts++, type: "say", say: "api_req_started", text: JSON.stringify({ apiProtocol: "openai", tokensIn: 1000, tokensOut: 200, cost: 0.001 }) }) + // batchable: tool-only turns (no visible text between edits) -> ChatView merges them into one giant batch message + if (!(batchable && big)) messages.push({ ts: ts++, type: "say", say: "text", text: filler(300, i) }) + // origBytes: realistic appliedDiff (diff + unified patch + the whole pre-edit file). omitOriginal: what the extension sends now + // (originalContent left out, only its length kept). + const tool = big && origBytes + ? { + tool: "appliedDiff", + path: `src/gen/f${i}.ts`, + diff: filler(1500, i), + content: filler(2500, i), + diffStats: { added: 20, removed: 5 }, + ...(omitOriginal ? { originalContentLength: origBytes } : { originalContent: filler(origBytes, i) }), + } + : big + ? { tool: kinds[i % 2], path: `src/gen/f${i}.ts`, [i % 2 === 0 ? "content" : "diff"]: filler(bigBytes, i) } + : { tool: "updateTodoList", todos: [{ id: String(i), content: `item ${i}`, status: "pending" }] } + messages.push({ ts: ts++, type: "ask", ask: "tool", isAnswered: true, text: JSON.stringify(tool) }) + if (!(batchable && big)) messages.push({ ts: ts++, type: "say", say: "user_feedback", text: `ok ${i}` }) + } + for (let i = 0; i < toolCount; i++) addTurn(i, true) + const state = () => ({ type: "state", state: { version: "0.0.0", clineMessages: messages, clineMessagesSeq: ++seq, apiConfiguration: { apiProvider: "fake-ai" }, mode: "code", customModes: [], taskHistory: [], terminalShellIntegrationDisabled: true } }) + window.postMessage(state(), "*") + await sleep(3000) + if (!document.body.innerText.includes("[STRESS]")) return { error: "not rendered" } + // Retained heap that is not chat data (stands in for everything else a long session keeps alive). + window.__ballast = [] + for (let i = 0; i < (ballastMb ?? 0); i++) { + const chunk = i + "y".repeat(1048576) + chunk.charCodeAt(5000) + window.__ballast.push(chunk) + } + await window.phaseStart() + + const live = { ts: ts++, type: "say", say: "text", text: "", partial: true } + messages.push(live) + window.postMessage(state(), "*") + let posts = 0 + let turns = 0 + let chunkText = "" + const end = performance.now() + seconds * 1000 + const chunk = () => { + chunkText += "streamed markdown chunk with `code` and **bold** text. " + live.text = chunkText + window.postMessage({ type: "messageUpdated", clineMessage: { ...live } }, "*") + posts++ + } + const turn = () => { + addTurn(toolCount + turns++, false) + window.postMessage(state(), "*") + posts++ + } + if (mode === "stream") { + while (performance.now() < end) { + chunk() + await sleep(1000 / hz) + } + } else if (mode === "turns") { + while (performance.now() < end) { + turn() + await sleep(1000 / hz) + } + } else { + let next = performance.now() + const chunkEvery = 1000 / chunkHz + let nextChunk = performance.now() + while (performance.now() < end) { + const now = performance.now() + if (now >= next) { + turn() + next = now + 1000 / hz + } + if (now >= nextChunk) { + chunk() + nextChunk = now + chunkEvery + } + await sleep(Math.min(chunkEvery, 1000 / hz) / 2) + } + } + return { posts, messages: messages.length } +} + + +// Decoupled sender: a Worker posts full-state pushes at a fixed rate whether or not the webview has finished the previous one +// (the real extension host is a separate process). Reports how many pushes were sent vs processed (backlog). +const pageDriverAsync = async ({ toolMb, hz, seconds, batchable, twoByte }) => { + const sleep = (ms) => new Promise((r) => setTimeout(r, ms)) + const workerMain = () => { + let messages = [] + let ts = Date.now() - 1e6 + let seq = 0 + let turns = 0 + let toolCount = 0 + let p + let counters + const filler = (n, seed) => { + if (p.twoByte) { + const base = `line ${seed}: \uD55C\uAE00 The quick brown fox jumps over the lazy dog. ` + return base.repeat(Math.ceil(n / base.length)).slice(0, n) + } + return `line ${seed}: The quick brown fox jumps over the lazy dog. `.repeat(Math.ceil(n / 50)).slice(0, n) + } + const bigBytes = 50000 + const addTurn = (i, big) => { + messages.push({ ts: ts++, type: "say", say: "api_req_started", text: JSON.stringify({ apiProtocol: "openai", tokensIn: 1000, tokensOut: 200, cost: 0.001 }) }) + if (!(p.batchable && big)) messages.push({ ts: ts++, type: "say", say: "text", text: filler(300, i) }) + const tool = big + ? { tool: i % 2 === 0 ? "newFileCreated" : "appliedDiff", path: `src/gen/f${i}.ts`, [i % 2 === 0 ? "content" : "diff"]: filler(bigBytes, i) } + : { tool: "updateTodoList", todos: [{ id: String(i), content: `item ${i}`, status: "pending" }] } + messages.push({ ts: ts++, type: "ask", ask: "tool", isAnswered: true, text: JSON.stringify(tool) }) + if (!(p.batchable && big)) messages.push({ ts: ts++, type: "say", say: "user_feedback", text: `ok ${i}` }) + } + const stateMsg = () => ({ type: "state", state: { version: "0.0.0", clineMessages: messages, clineMessagesSeq: ++seq, apiConfiguration: { apiProvider: "fake-ai" }, mode: "code", customModes: [], taskHistory: [], terminalShellIntegrationDisabled: true } }) + let timer + self.onmessage = (e) => { + const m = e.data + if (m.type === "init") { + p = m.params + counters = m.sab ? new Int32Array(m.sab) : null + messages = [{ ts: ts++, type: "say", say: "text", text: "[STRESS] heap matrix" }] + toolCount = Math.max(1, Math.round((p.toolMb * 1048576) / bigBytes)) + for (let i = 0; i < toolCount; i++) addTurn(i, true) + self.postMessage({ kind: "state", data: stateMsg() }) + self.postMessage({ kind: "ready" }) + } else if (m.type === "start") { + timer = setInterval(() => { + addTurn(toolCount + turns++, false) + self.postMessage({ kind: "state", data: stateMsg() }) + if (counters) Atomics.add(counters, 0, 1) + }, 1000 / p.hz) + } else if (m.type === "stop") { + clearInterval(timer) + } + } + } + const sab = typeof SharedArrayBuffer !== "undefined" ? new SharedArrayBuffer(8) : null + const counters = sab ? new Int32Array(sab) : null + const worker = new Worker(URL.createObjectURL(new Blob(["(" + workerMain.toString() + ")()"], { type: "text/javascript" }))) + let received = 0 + let readyResolve + const ready = new Promise((r) => (readyResolve = r)) + worker.onmessage = (e) => { + if (e.data.kind === "state") { + window.dispatchEvent(new MessageEvent("message", { data: e.data.data })) + received++ + } else if (e.data.kind === "ready") { + readyResolve() + } + } + worker.postMessage({ type: "init", params: { toolMb, hz, batchable, twoByte }, sab }) + await ready + await sleep(3000) + if (!document.body.innerText.includes("[STRESS]")) return { error: "not rendered" } + await window.phaseStart() + received = 0 + worker.postMessage({ type: "start" }) + await sleep(seconds * 1000) + const sent = counters ? Atomics.load(counters, 0) : -1 + worker.postMessage({ type: "stop" }) + return { posts: received, sent, backlog: sent - received } +} + +const browser = await chromium.launch({ headless: true, args: [`--js-flags=--max-old-space-size=${heapMb}`] }) +const rows = [] +for (const exp of EXPERIMENTS.filter((e) => !only || only.includes(e.name))) { + const page = await browser.newPage() + let crashed = false + let crashAt = null + page.on("crash", () => { + crashed = true + crashAt = Date.now() + }) + await page.addInitScript(initScript) + await page.goto(url) + const cdp = await page.context().newCDPSession(page) + await cdp.send("Performance.enable") + await cdp.send("HeapProfiler.enable") + const heap = async () => Math.round((await cdp.send("Performance.getMetrics")).metrics.find((m) => m.name === "JSHeapUsedSize").value / 1048576) + + let peak = 0 + let hydratePeak = 0 + let baseMb = 0 + let tTo1G = null + let samplingFrom = null + await page.exposeFunction("phaseStart", async () => { + hydratePeak = peak + await cdp.send("HeapProfiler.collectGarbage") + baseMb = await heap() + peak = 0 + samplingFrom = Date.now() + }) + const sampler = setInterval(async () => { + if (crashed) return + try { + const h = await heap() + peak = Math.max(peak, h) + if (samplingFrom !== null && tTo1G === null && h >= 1024) tTo1G = ((Date.now() - samplingFrom) / 1000).toFixed(1) + } catch {} + }, 100) + + let result = {} + try { + result = await page.evaluate(exp.mode === "async" ? pageDriverAsync : pageDriver, { mode: exp.mode, toolMb: exp.toolMb, hz: exp.hz, chunkHz: exp.chunkHz, seconds, ballastMb: exp.ballastMb, batchable: exp.batchable, twoByte: exp.twoByte, tools: exp.tools, origBytes: exp.origBytes, omitOriginal: exp.omitOriginal }) + } catch (e) { + result = { error: crashed ? "CRASHED (renderer OOM)" : e.message.split("\n")[0] } + } + clearInterval(sampler) + let floorMb = "-" + if (!crashed && !result.error) { + await cdp.send("HeapProfiler.collectGarbage") + floorMb = await heap() + } + rows.push({ name: exp.name, mode: exp.mode, toolMb: exp.toolMb, twoByte: !!exp.twoByte, ballastMb: exp.ballastMb ?? 0, rate: exp.mode === "both" ? `${exp.hz}t+${exp.chunkHz}c` : exp.hz, hydratePeakMB: hydratePeak, baseMB: baseMb, peakMB: peak, afterGC: floorMb, crashAfterSec: crashed ? (samplingFrom === null ? "during-hydration" : ((crashAt - samplingFrom) / 1000).toFixed(1)) : "-", to1GBsec: tTo1G ?? "-", posts: result.posts ?? "-", sent: result.sent ?? "-", backlog: result.backlog ?? "-", result: result.error ?? "ok" }) + console.log(JSON.stringify(rows.at(-1))) + await page.close().catch(() => {}) +} +await browser.close() +httpServer.close() + +console.log("\nname".padEnd(28) + "mode".padEnd(8) + "toolMB".padEnd(8) + "rate".padEnd(10) + "hydrPk".padEnd(8) + "baseMB".padEnd(8) + "peakMB".padEnd(9) + "afterGC".padEnd(9) + "to1GB(s)".padEnd(10) + "posts".padEnd(7) + "sent".padEnd(6) + "backlog".padEnd(9) + "crashAfter(s)".padEnd(15) + "result") +for (const r of rows) { + console.log(r.name.padEnd(28) + r.mode.padEnd(8) + String(r.toolMb).padEnd(8) + String(r.rate).padEnd(10) + String(r.hydratePeakMB).padEnd(8) + String(r.baseMB).padEnd(8) + String(r.peakMB).padEnd(9) + String(r.afterGC).padEnd(9) + String(r.to1GBsec).padEnd(10) + String(r.posts).padEnd(7) + String(r.sent).padEnd(6) + String(r.backlog).padEnd(9) + String(r.crashAfterSec).padEnd(15) + r.result) +} diff --git a/scripts/gray-screen/webview-render-stress.mjs b/scripts/gray-screen/webview-render-stress.mjs new file mode 100644 index 0000000000..25fb4c528d --- /dev/null +++ b/scripts/gray-screen/webview-render-stress.mjs @@ -0,0 +1,394 @@ +#!/usr/bin/env node +// Real-render stress harness: boots the actual webview-ui (webview-ui) in headless Chromium, +// fakes acquireVsCodeApi, and feeds it extension-host-style messages while measuring real +// rendering latency (rAF), DOM size and V8 heap (before/after forced GC) via CDP. +// +// Usage (from repo root): +// node scripts/gray-screen/webview-render-stress.mjs --scenario session|command|md [options] +// --start 5000 initial message count (session) +// --pushes 300 number of full-state pushes (session) +// --grow 4 messages added per push (session) +// --chunks 20 streaming chunks per push (session) +// --rate 200 output messages/second (command) +// --seconds 60 duration (command) +// --turns 40 streamed markdown turns (md) +// --text-bytes 15000 markdown size per turn (md; also session) +// --chunk-ms 15 ms between streamed chunks (md) +// --mermaid-every 10 mermaid diagram every N turns (md; 0 disables) +// --alloc-profile true print top allocators (CDP sampling heap profiler, md) +// --build-mode development unminified build (readable function names) in a separate dir +// --heap-mb 1024 V8 old-space cap (default 1024; 4096 for md) +// --skip-build true reuse the previous build in /tmp/zoo-webview-stress-build +// --messages-file seed history from a generated task (generate-large-task.mjs) +// --url http://... use an already running server instead of building/serving +// --headed show the browser +import { spawn } from "node:child_process" +import fs from "node:fs" +import http from "node:http" +import { createRequire } from "node:module" +import path from "node:path" +import { fileURLToPath } from "node:url" + +const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..") +const webviewDir = path.join(root, "webview-ui") +const require = createRequire(path.join(webviewDir, "package.json")) +const { chromium } = require("@playwright/test") + +const args = Object.fromEntries( + process.argv.slice(2).reduce((acc, cur, i, all) => { + if (cur.startsWith("--")) acc.push([cur.slice(2), all[i + 1]?.startsWith("--") || all[i + 1] === undefined ? "true" : all[i + 1]]) + return acc + }, []), +) +const cfg = { + scenario: args["scenario"] ?? "session", + start: Number(args["start"] ?? 5000), + pushes: Number(args["pushes"] ?? 300), + grow: Number(args["grow"] ?? 4), + chunks: Number(args["chunks"] ?? 20), + rate: Number(args["rate"] ?? 200), + seconds: Number(args["seconds"] ?? 60), + heapMb: Number(args["heap-mb"] ?? (args["scenario"] === "md" ? 4096 : 1024)), + textBytes: Number(args["text-bytes"] ?? (args["scenario"] === "md" ? 15000 : 800)), + turns: Number(args["turns"] ?? 40), + chunkMs: Number(args["chunk-ms"] ?? 15), + mermaidEvery: Number(args["mermaid-every"] ?? 10), + url: args["url"], + headed: args["headed"] === "true", +} + +const sleep = (ms) => new Promise((r) => setTimeout(r, ms)) + +const buildMode = args["build-mode"] ?? "production" +const buildDir = "/tmp/zoo-webview-stress-build" + (buildMode === "production" ? "" : `-${buildMode}`) + +function run(cmd, cmdArgs, opts) { + return new Promise((resolve, reject) => { + const child = spawn(cmd, cmdArgs, { stdio: ["ignore", "ignore", "inherit"], ...opts }) + child.on("exit", (code) => (code === 0 ? resolve() : reject(new Error(`${cmd} exited ${code}`)))) + }) +} + +// Production build into a temp dir (never touches src/webview-ui/build), then serve it statically. +async function startServer() { + if (args["skip-build"] !== "true" || !fs.existsSync(path.join(buildDir, "index.html"))) { + console.log("building webview-ui (production) into", buildDir, "...") + await run("pnpm", ["exec", "vite", "build", "--outDir", buildDir, "--emptyOutDir", "--mode", buildMode], { + cwd: webviewDir, + }) + } + const types = { ".html": "text/html", ".js": "text/javascript", ".css": "text/css", ".json": "application/json", ".wasm": "application/wasm", ".svg": "image/svg+xml", ".map": "application/json", ".woff2": "font/woff2", ".ttf": "font/ttf" } + const serveRoot = fs.realpathSync(buildDir) + // Resolves a request path to a regular file inside serveRoot (after symlink resolution), or undefined. + const resolveServedFile = (rawPath) => { + try { + const decoded = decodeURIComponent(rawPath) + const real = fs.realpathSync(path.resolve(serveRoot, "." + (decoded === "/" ? "/index.html" : decoded))) + const rel = path.relative(serveRoot, real) + if (rel === "" || rel === ".." || rel.startsWith(".." + path.sep) || path.isAbsolute(rel)) return undefined + return fs.statSync(real).isFile() ? real : undefined + } catch { + return undefined + } + } + const httpServer = http.createServer((req, res) => { + const file = resolveServedFile(new URL(req.url, "http://x").pathname) + if (!file) { + res.writeHead(404).end() + return + } + res.writeHead(200, { "content-type": types[path.extname(file)] ?? "application/octet-stream" }) + fs.createReadStream(file).pipe(res) + }) + await new Promise((r) => httpServer.listen(0, "127.0.0.1", r)) + return { url: `http://127.0.0.1:${httpServer.address().port}/`, child: { kill: () => httpServer.close() } } +} + +const initScript = () => { + window.__posted = 0 + window.acquireVsCodeApi = () => ({ + postMessage: () => { + window.__posted++ + }, + getState: () => undefined, + setState: () => {}, + }) + window.IMAGES_BASE_URI = "/" + window.MATERIAL_ICONS_BASE_URI = "/" +} + +const baseState = (extra) => ({ + version: "0.0.0", + clineMessages: [], + taskHistory: [], + apiConfiguration: { apiProvider: "fake-ai" }, + mode: "code", + customModes: [], + terminalShellIntegrationDisabled: true, + ...extra, +}) + +const pageHelpers = () => { + window.__stress = { + filler: (n, seed) => `line ${seed}: The quick brown fox jumps over the lazy dog. `.repeat(Math.ceil(n / 50)).slice(0, n), + raf: () => new Promise((r) => requestAnimationFrame(() => requestAnimationFrame(r))), + emit: (data) => window.postMessage(data, "*"), + } +} + +async function main() { + const server = cfg.url ? null : await startServer() + const url = cfg.url ?? server.url + const browser = await chromium.launch({ + headless: !cfg.headed, + args: [`--js-flags=--max-old-space-size=${cfg.heapMb}`, "--enable-precise-memory-info"], + }) + const page = await browser.newPage() + let crashed = false + page.on("crash", () => { + crashed = true + console.error("\n*** PAGE CRASHED (renderer died: this is the 'gray screen') ***") + }) + page.on("pageerror", (e) => console.error("[pageerror]", e.message.split("\n")[0])) + await page.addInitScript(initScript) + await page.goto(url) + await page.evaluate(pageHelpers) + const cdp = await page.context().newCDPSession(page) + await cdp.send("Performance.enable") + await cdp.send("HeapProfiler.enable") + + const heap = async (gc) => { + if (gc) await cdp.send("HeapProfiler.collectGarbage") + const { metrics } = await cdp.send("Performance.getMetrics") + return Math.round(metrics.find((m) => m.name === "JSHeapUsedSize").value / 1048576) + } + const dom = () => page.evaluate(() => document.getElementsByTagName("*").length) + const frame = async () => { + const t = Date.now() + await page.evaluate(() => window.__stress.raf()) + return Date.now() - t + } + + const emit = (data) => page.evaluate((d) => window.__stress.emit(d), data) + const ts0 = Date.now() - 1e6 + let ts = ts0 + let seq = 0 + const messages = [{ ts: ts++, type: "say", say: "text", text: "[STRESS] simulated long session" }] + const round = (i) => [ + { ts: ts++, type: "say", say: "api_req_started", text: JSON.stringify({ apiProtocol: "openai", tokensIn: 1000, tokensOut: 200, cost: 0.001 }) }, + { ts: ts++, type: "say", say: "text", text: `line ${i}: ${"The quick brown fox jumps over the lazy dog. ".repeat(Math.ceil(cfg.textBytes / 45))}`.slice(0, cfg.textBytes) }, + { ts: ts++, type: "ask", ask: "tool", isAnswered: true, text: JSON.stringify({ tool: "appliedDiff", path: `src/f${i}.ts`, diff: "x".repeat(cfg.textBytes) }) }, + { ts: ts++, type: "say", say: "user_feedback", text: `ok ${i}` }, + ] + if (args["messages-file"]) { + // Seed with a task generated by scripts/gray-screen/generate-large-task.mjs (ui_messages.json). + messages.length = 0 + messages.push(...JSON.parse(fs.readFileSync(args["messages-file"], "utf8"))) + messages[0].text = `[STRESS] ${messages[0].text}` + ts = messages.at(-1).ts + 1 + } else { + for (let i = 0; messages.length < cfg.start; i++) messages.push(...round(i)) + } + + // Hydrate (with the initial history) and verify real rendering happened. + await emit({ type: "state", state: baseState({ clineMessages: messages, clineMessagesSeq: ++seq }) }) + await sleep(3000) + const rendered = await page.evaluate(() => document.body.innerText.includes("[STRESS]")) + if (!rendered) { + console.error("Render check FAILED: '[STRESS]' text not in DOM. Aborting (results would be meaningless).") + await browser.close() + server?.child.kill() + process.exit(2) + } + console.log(`render check passed: ${await dom()} DOM nodes, ${messages.length} msgs, heap ${await heap(true)}MB after GC`) + console.log(`scenario=${cfg.scenario} heapCap=${cfg.heapMb}MB (headless Chromium, ${buildMode} build)\n`) + + const t0 = Date.now() + const floor = [] + try { + if (cfg.scenario === "session") { + for (let p = 0; p < cfg.pushes && !crashed; p++) { + for (let k = 0; k < cfg.grow; k += 4) messages.push(...round(messages.length)) + await emit({ type: "state", state: baseState({ clineMessages: messages, clineMessagesSeq: ++seq }) }) + const streamTs = ts++ + const live = { ts: streamTs, type: "say", say: "text", text: "", partial: true } + messages.push(live) + await emit({ type: "state", state: baseState({ clineMessages: messages, clineMessagesSeq: ++seq }) }) + for (let c = 0; c < cfg.chunks; c++) { + live.text += `chunk ${c}: ${"streamed text ".repeat(15)}\n` + await emit({ type: "messageUpdated", clineMessage: { ...live } }) + await sleep(50) + } + live.partial = false + if (p % 10 === 0) { + const lat = await frame() + const gcHeap = await heap(true) + floor.push(gcHeap) + console.log(`push ${String(p).padStart(4)} msgs=${messages.length} dom=${await dom()} frameLatency=${lat}ms heapAfterGC=${gcHeap}MB t=${Math.round((Date.now() - t0) / 1000)}s`) + } + } + } else if (cfg.scenario === "command") { + const execTs = ts++ + messages.push({ ts: execTs, type: "ask", ask: "command", text: "npm run dev", isAnswered: true }) + await emit({ type: "state", state: baseState({ clineMessages: messages, clineMessagesSeq: ++seq }) }) + const executionId = String(execTs) + const post = (s) => emit({ type: "commandExecutionStatus", text: JSON.stringify({ executionId, ...s }) }) + await post({ status: "started", pid: 1, command: "npm run dev" }) + const lineText = (n) => `[${n}] ${"build output line ".repeat(10)}`.slice(0, 100) + await post({ status: "output", output: Array.from({ length: 500 }, (_, k) => lineText(k)).join("\n") }) + await sleep(1500) + const shown = await page.evaluate(() => document.body.innerText.includes("build output line")) + if (!shown) { + console.error("Render check FAILED: command output is not in the DOM (row collapsed or not rendered). Aborting.") + await browser.close() + server?.child.kill() + process.exit(2) + } + console.log("command output render check passed (TerminalOutput is mounted and visible)") + let counter = 0 + const end = Date.now() + cfg.seconds * 1000 + let sent = 0 + let lastLog = Date.now() + while (Date.now() < end && !crashed) { + const batchStart = Date.now() + const perBatch = Math.max(1, Math.round(cfg.rate / 20)) + for (let i = 0; i < perBatch; i++) { + const start = counter++ + await post({ status: "output", output: Array.from({ length: 500 }, (_, k) => lineText(start + k)).join("\n") }) + sent++ + } + const wait = 50 - (Date.now() - batchStart) + if (wait > 0) await sleep(wait) + if (Date.now() - lastLog > 5000) { + lastLog = Date.now() + const lat = await frame() + const gcHeap = await heap(true) + floor.push(gcHeap) + console.log(`t=${Math.round((Date.now() - t0) / 1000)}s sent=${sent} dom=${await dom()} frameLatency=${lat}ms heapAfterGC=${gcHeap}MB`) + } + } + } else if (cfg.scenario === "md") { + // Streams realistic markdown (code blocks, tables, lists, mermaid) the way Zoo Code does: + // every chunk re-sends the FULL accumulated text of the partial message. + const rngFor = (seed) => { + let a = (seed * 2654435761) >>> 0 + return () => { + a = (a + 0x6d2b79f5) >>> 0 + let t = a + t = Math.imul(t ^ (t >>> 15), t | 1) + t ^= t + Math.imul(t ^ (t >>> 7), t | 61) + return ((t ^ (t >>> 14)) >>> 0) / 4294967296 + } + } + const PARAS = [ + "I looked at how the current code paths interact and there are a few things worth calling out before making changes.", + "The existing implementation works for the happy path, but the error handling around retries is inconsistent.", + "This keeps the public surface unchanged while making the internals easier to reason about and to test in isolation.", + "One thing to watch: the `timeoutMs` option is optional, so every consumer has to handle `undefined` explicitly.", + ] + const BULLETS = ["Keep `REVISION` monotonically increasing", "Guard against empty input before iterating", "Prefer `for...of` with `entries()`", "Add a regression test for the boundary"] + const CODE = [ + (n) => "```ts\nexport async function resolve" + n + "(input: string[]): Promise {\n\tconst out: string[] = []\n\tfor (const [i, item] of input.entries()) {\n\t\tout.push(`${item.trim()}:${i}`)\n\t}\n\treturn out\n}\n```", + (n) => "```python\ndef summary_" + n + "(items):\n counts = {}\n for item in items:\n counts[item] = counts.get(item, 0) + 1\n return counts\n```", + (n) => "```bash\npnpm --dir src exec vitest run core/feature_" + n + "\npnpm lint && pnpm check-types\n```", + (n) => "```json\n{\n \"name\": \"feature_" + n + "\",\n \"retry\": { \"limit\": 3, \"backoffMs\": [100, 250, 500] }\n}\n```", + (n) => "```diff\n--- a/src/f" + n + ".ts\n+++ b/src/f" + n + ".ts\n@@ -3,2 +3,2 @@\n-export const REVISION = 0\n+export const REVISION = 1\n```", + ] + const mdFor = (n) => { + const r = rngFor(n * 31 + 5) + const pick = (a) => a[Math.floor(r() * a.length)] + let out = `## Step ${n}: review\n\n${pick(PARAS)}\n\n` + if (cfg.mermaidEvery > 0 && n % cfg.mermaidEvery === 0) { + out += "```mermaid\nflowchart TD\n A[Request] --> B{cache hit?}\n B -- yes --> C[Return cached]\n B -- no --> D[Resolver]\n D --> C\n```\n\n" + } + while (out.length < cfg.textBytes) { + const kind = Math.floor(r() * 4) + if (kind === 0) out += `**Key points**\n\n${[0, 1, 2].map(() => `- ${pick(BULLETS)}`).join("\n")}\n\n` + else if (kind === 1) out += `${pick(PARAS)}\n\n${pick(CODE)(n)}\n\n` + else if (kind === 2) out += "| Module | Lines | Status |\n| --- | ---: | --- |\n| `a.ts` | 150 | ok |\n| `b.ts` | 151 | updated |\n\n" + else out += `> **Note:** ${pick(PARAS)}\n\n` + } + return out + } + + const allocProfile = args["alloc-profile"] === "true" + if (allocProfile) { + await cdp.send("HeapProfiler.startSampling", { + samplingInterval: 16384, + includeObjectsCollectedByMajorGC: true, + includeObjectsCollectedByMinorGC: true, + }) + } + let peak = 0 + const sampler = setInterval(async () => { + try { + const { metrics } = await cdp.send("Performance.getMetrics") + peak = Math.max(peak, Math.round(metrics.find((m) => m.name === "JSHeapUsedSize").value / 1048576)) + } catch {} + }, 250) + for (let turn = 1; turn <= cfg.turns && !crashed; turn++) { + messages.push({ ts: ts++, type: "say", say: "api_req_started", text: JSON.stringify({ apiProtocol: "openai", tokensIn: 1000, tokensOut: 200, cost: 0.001 }) }) + const live = { ts: ts++, type: "say", say: "text", text: "", partial: true } + messages.push(live) + await emit({ type: "state", state: baseState({ clineMessages: messages, clineMessagesSeq: ++seq }) }) + const full = mdFor(turn) + let updates = 0 + for (let i = 0; i < full.length; ) { + i += 8 + Math.floor(Math.random() * 40) + live.text = full.slice(0, i) + await emit({ type: "messageUpdated", clineMessage: { ...live } }) + updates++ + if (cfg.chunkMs > 0) await sleep(cfg.chunkMs * (0.5 + Math.random())) + } + live.text = full + live.partial = false + await emit({ type: "messageUpdated", clineMessage: { ...live } }) + const lat = await frame() + const turnPeak = peak + peak = 0 + const gcHeap = await heap(true) + floor.push(gcHeap) + console.log(`turn ${String(turn).padStart(3)} msgs=${messages.length} updates=${updates} dom=${await dom()} frameLatency=${lat}ms peakHeap=${turnPeak}MB heapAfterGC=${gcHeap}MB t=${Math.round((Date.now() - t0) / 1000)}s`) + } + clearInterval(sampler) + if (allocProfile) { + const { profile } = await cdp.send("HeapProfiler.stopSampling") + const totals = new Map() + const walk = (node) => { + const f = node.callFrame + const key = `${f.functionName || "(anonymous)"} @ ${f.url.split("/").pop()}:${f.lineNumber + 1}` + totals.set(key, (totals.get(key) ?? 0) + node.selfSize) + node.children.forEach(walk) + } + walk(profile.head) + const sum = [...totals.values()].reduce((a, b) => a + b, 0) + console.log(`\nAllocation sampling (self size incl. collected objects), total ${Math.round(sum / 1048576)}MB:`) + for (const [k, v] of [...totals.entries()].sort((a, b) => b[1] - a[1]).slice(0, 25)) { + console.log(` ${String(Math.round(v / 1048576)).padStart(6)}MB ${((v / sum) * 100).toFixed(1).padStart(5)}% ${k}`) + } + } + } else { + throw new Error(`unknown scenario ${cfg.scenario}`) + } + } catch (e) { + if (!crashed) console.error("scenario error:", e.message.split("\n")[0]) + } + + console.log("") + if (crashed) { + console.log("RESULT: renderer crashed (OOM reproduced).") + } else if (floor.length >= 4) { + const q = Math.max(1, Math.floor(floor.length / 4)) + const first = Math.round(floor.slice(0, q).reduce((a, b) => a + b, 0) / q) + const last = Math.round(floor.slice(-q).reduce((a, b) => a + b, 0) / q) + console.log(`RESULT: heap-after-GC first quarter avg ${first}MB -> last quarter avg ${last}MB (${last > first * 1.3 ? "GROWING: likely retained memory" : "flat: no leak on this path"})`) + } + await browser.close().catch(() => {}) + server?.child.kill() + process.exit(crashed ? 1 : 0) +} + +main().catch((e) => { + console.error(e) + process.exit(1) +}) diff --git a/src/core/webview/ClineProvider.ts b/src/core/webview/ClineProvider.ts index 34848ab860..a8cbfbdd00 100644 --- a/src/core/webview/ClineProvider.ts +++ b/src/core/webview/ClineProvider.ts @@ -129,6 +129,7 @@ import { import { readTaskMessages } from "../task-persistence/taskMessages" import { getNonce } from "./getNonce" import { getUri } from "./getUri" +import { omitOriginalContentFromExtensionMessage } from "./stripOriginalContent" import { REQUESTY_BASE_URL } from "../../shared/utils/requesty" import { validateAndFixToolResultIds } from "../task/validateToolResultIds" import { PendingEditOperationStore, type PendingEditOperationInput } from "./PendingEditOperationStore" @@ -1427,7 +1428,7 @@ export class ClineProvider } try { - await this.view?.webview.postMessage(message) + await this.view?.webview.postMessage(omitOriginalContentFromExtensionMessage(message)) } catch { // View disposed, drop message silently } diff --git a/src/core/webview/__tests__/ClineProvider.spec.ts b/src/core/webview/__tests__/ClineProvider.spec.ts index b312853e76..2ace012534 100644 --- a/src/core/webview/__tests__/ClineProvider.spec.ts +++ b/src/core/webview/__tests__/ClineProvider.spec.ts @@ -838,6 +838,60 @@ describe("ClineProvider", () => { await expect(provider.postMessageToWebview(message)).resolves.toBeUndefined() }) + test("postMessageToWebview leaves originalContent of file-edit tool messages out of the state it posts", async () => { + await provider.resolveWebviewView(mockWebviewView) + + const originalFile = "line of the original file\n".repeat(500) + const toolText = JSON.stringify({ + tool: "appliedDiff", + path: "a.ts", + diff: "@@ d", + originalContent: originalFile, + }) + // Only the field under test is populated; the rest of ExtensionState is irrelevant here. + const message = { + type: "state", + state: { clineMessages: [{ ts: 1, type: "ask", ask: "tool", text: toolText }] }, + } as unknown as ExtensionMessage + + await provider.postMessageToWebview(message) + + const posted = mockPostMessage.mock.calls.at(-1)![0] as ExtensionMessage + const postedText = posted.state!.clineMessages![0]!.text! + + expect(JSON.parse(postedText)).toEqual({ + tool: "appliedDiff", + path: "a.ts", + diff: "@@ d", + originalContentLength: originalFile.length, + }) + // the extension's own message (and so the persisted task) keeps the full content + expect(message.state!.clineMessages![0]!.text).toBe(toolText) + }) + + test("postMessageToWebview leaves originalContent out of messageUpdated", async () => { + await provider.resolveWebviewView(mockWebviewView) + + const toolText = JSON.stringify({ + tool: "appliedDiff", + path: "a.ts", + originalContent: "original file\n".repeat(100), + }) + + await provider.postMessageToWebview({ + type: "messageUpdated", + clineMessage: { ts: 2, type: "ask", ask: "tool", text: toolText }, + }) + + const posted = mockPostMessage.mock.calls.at(-1)![0] as ExtensionMessage + + expect(JSON.parse(posted.clineMessage!.text!)).toEqual({ + tool: "appliedDiff", + path: "a.ts", + originalContentLength: "original file\n".length * 100, + }) + }) + describe("theme fixture probes", () => { const fixture = { themeId: "Default Dark Modern", diff --git a/src/core/webview/__tests__/stripOriginalContent.spec.ts b/src/core/webview/__tests__/stripOriginalContent.spec.ts new file mode 100644 index 0000000000..ed68a0be51 --- /dev/null +++ b/src/core/webview/__tests__/stripOriginalContent.spec.ts @@ -0,0 +1,245 @@ +import type { ClineMessage, ExtensionMessage } from "@roo-code/types" + +import { + findOriginalContent, + omitOriginalContent, + omitOriginalContentFromExtensionMessage, +} from "../stripOriginalContent" + +let ts = 0 +const toolAsk = (payload: unknown, extra: Partial = {}): ClineMessage => ({ + ts: ++ts, + type: "ask", + ask: "tool", + text: typeof payload === "string" ? payload : JSON.stringify(payload), + ...extra, +}) + +const bigOriginal = "line of the original file\n".repeat(2000) + +describe("omitOriginalContent", () => { + it("removes a non-empty originalContent and records its length", () => { + const message = toolAsk({ + tool: "appliedDiff", + path: "a.ts", + diff: "@@ d", + content: "patch", + originalContent: bigOriginal, + }) + + const result = omitOriginalContent(message) + const payload = JSON.parse(result.text!) + + expect(payload).toEqual({ + tool: "appliedDiff", + path: "a.ts", + diff: "@@ d", + content: "patch", + originalContentLength: bigOriginal.length, + }) + expect(result.text!.length).toBeLessThan(message.text!.length / 10) + expect(result).not.toBe(message) + expect(result.ts).toBe(message.ts) + }) + + it("does not modify the original message object", () => { + const message = toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal }) + const before = message.text + + omitOriginalContent(message) + + expect(message.text).toBe(before) + }) + + it("keeps an empty originalContent (a new file) inline", () => { + const message = toolAsk({ tool: "newFileCreated", path: "new.ts", content: "x", originalContent: "" }) + + expect(omitOriginalContent(message)).toBe(message) + }) + + it("works for say tool messages", () => { + const message: ClineMessage = { + ts: ++ts, + type: "say", + say: "tool", + text: JSON.stringify({ tool: "editedExistingFile", path: "a.ts", originalContent: bigOriginal }), + } + + expect(JSON.parse(omitOriginalContent(message).text!).originalContentLength).toBe(bigOriginal.length) + }) + + it("leaves messages without originalContent untouched", () => { + const messages = [ + toolAsk({ tool: "readFile", path: "a.ts" }), + toolAsk({ tool: "appliedDiff", path: "a.ts", diff: "d" }), + { ts: ++ts, type: "say", say: "text", text: "originalContent is only a word here" } as ClineMessage, + { ts: ++ts, type: "ask", ask: "command", text: '{"originalContent":"not a tool message"}' } as ClineMessage, + { ts: ++ts, type: "ask", ask: "tool" } as ClineMessage, + ] + + for (const message of messages) { + expect(omitOriginalContent(message)).toBe(message) + } + }) + + it("leaves unparsable text untouched", () => { + const message = toolAsk('{"tool":"appliedDiff","originalContent":"cut off') + + expect(omitOriginalContent(message)).toBe(message) + }) + + it("leaves a non-string originalContent untouched", () => { + const message = toolAsk({ tool: "appliedDiff", originalContent: 42 }) + + expect(omitOriginalContent(message)).toBe(message) + }) + + it("reuses the result while the text is unchanged and recomputes after an in-place update", () => { + const parse = vi.spyOn(JSON, "parse") + const message = toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal }) + + const first = omitOriginalContent(message) + const second = omitOriginalContent(message) + + expect(second).toEqual(first) + expect(parse).toHaveBeenCalledTimes(1) + parse.mockRestore() + + // partial messages are updated in place by the task, so a new text must not return a stale result + message.text = JSON.stringify({ tool: "appliedDiff", path: "b.ts", originalContent: bigOriginal + "more" }) + const third = omitOriginalContent(message) + + expect(JSON.parse(third.text!)).toMatchObject({ path: "b.ts", originalContentLength: bigOriginal.length + 4 }) + }) + + it("takes metadata from the current message when the cached text is reused", () => { + const message = toolAsk( + { tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal }, + { partial: false, isAnswered: false }, + ) + + expect(omitOriginalContent(message).isAnswered).toBe(false) + + // approval only flips metadata; the text is unchanged + message.isAnswered = true + message.partial = true + const result = omitOriginalContent(message) + + expect(result).toMatchObject({ isAnswered: true, partial: true }) + expect(JSON.parse(result.text!)).toMatchObject({ originalContentLength: bigOriginal.length }) + expect(result.text).not.toContain(bigOriginal.slice(0, 50)) + }) + + it("is idempotent", () => { + const once = omitOriginalContent(toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal })) + + expect(omitOriginalContent(once)).toBe(once) + }) +}) + +describe("findOriginalContent", () => { + it("returns the originalContent of the tool message with that ts", () => { + const messages = [ + toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal }), + toolAsk({ tool: "appliedDiff", path: "b.ts", originalContent: "other" }), + ] + + expect(findOriginalContent(messages, { ts: messages[0]!.ts })).toBe(bigOriginal) + expect(findOriginalContent(messages, { ts: messages[1]!.ts })).toBe("other") + }) + + it("tells messages created in the same millisecond apart by messageId", () => { + const first = toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: "first" }, { messageId: "id-1" }) + const second = toolAsk( + { tool: "appliedDiff", path: "b.ts", originalContent: "second" }, + { messageId: "id-2", ts: first.ts }, + ) + const messages = [first, second] + + expect(findOriginalContent(messages, { messageId: "id-2", ts: first.ts })).toBe("second") + expect(findOriginalContent(messages, { messageId: "id-1", ts: first.ts })).toBe("first") + expect(findOriginalContent(messages, { messageId: "missing", ts: first.ts })).toBeNull() + // messages persisted without an id are still found by ts + expect(findOriginalContent(messages, { ts: first.ts })).toBe("first") + }) + + it("finds say tool messages too", () => { + const message: ClineMessage = { + ts: ++ts, + type: "say", + say: "tool", + text: JSON.stringify({ tool: "editedExistingFile", originalContent: bigOriginal }), + } + + expect(findOriginalContent([message], { ts: message.ts })).toBe(bigOriginal) + }) + + it("returns null when there is nothing to return", () => { + const noOriginal = toolAsk({ tool: "readFile", path: "a.ts" }) + const notTool: ClineMessage = { + ts: ++ts, + type: "say", + say: "text", + text: JSON.stringify({ originalContent: "x" }), + } + const unparsable = toolAsk('{"originalContent":"cut off') + const notAString = toolAsk({ tool: "appliedDiff", originalContent: 42 }) + const noText: ClineMessage = { ts: ++ts, type: "ask", ask: "tool" } + const all = [noOriginal, notTool, unparsable, notAString, noText] + + for (const message of all) { + expect(findOriginalContent(all, { ts: message.ts })).toBeNull() + } + + expect(findOriginalContent(all, { ts: -1 })).toBeNull() + expect(findOriginalContent(undefined, { ts: 1 })).toBeNull() + }) + + it("returns an empty original as an empty string", () => { + const message = toolAsk({ tool: "newFileCreated", content: "x", originalContent: "" }) + + expect(findOriginalContent([message], { ts: message.ts })).toBe("") + }) +}) + +describe("omitOriginalContentFromExtensionMessage", () => { + it("applies to the chat messages inside a state message without touching other state", () => { + const message = { + type: "state", + state: { + version: "1", + mode: "code", + clineMessages: [toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal })], + }, + } as unknown as ExtensionMessage + + const result = omitOriginalContentFromExtensionMessage(message) + + expect(result.state).toMatchObject({ version: "1", mode: "code" }) + expect(JSON.parse(result.state!.clineMessages![0]!.text!).originalContentLength).toBe(bigOriginal.length) + }) + + it("applies to messageUpdated", () => { + const message = { + type: "messageUpdated", + clineMessage: toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal }), + } as ExtensionMessage + + const result = omitOriginalContentFromExtensionMessage(message) + + expect(JSON.parse(result.clineMessage!.text!).originalContentLength).toBe(bigOriginal.length) + }) + + it("passes every other message through unchanged", () => { + const others = [ + { type: "action", action: "didBecomeVisible" }, + { type: "state", state: { clineMessages: [] } }, + { type: "state" }, + { type: "fileContent", fileContent: { path: "a", content: bigOriginal } }, + ] as unknown as ExtensionMessage[] + + for (const message of others) { + expect(omitOriginalContentFromExtensionMessage(message)).toBe(message) + } + }) +}) diff --git a/src/core/webview/__tests__/webviewMessageHandler.readOriginalContent.spec.ts b/src/core/webview/__tests__/webviewMessageHandler.readOriginalContent.spec.ts new file mode 100644 index 0000000000..b3d083c212 --- /dev/null +++ b/src/core/webview/__tests__/webviewMessageHandler.readOriginalContent.spec.ts @@ -0,0 +1,170 @@ +// npx vitest core/webview/__tests__/webviewMessageHandler.readOriginalContent.spec.ts + +import { describe, it, expect, vi, beforeEach } from "vitest" + +vi.mock("../../../api/providers/fetchers/modelCache") + +vi.mock("vscode", () => ({ + window: { + showInformationMessage: vi.fn(), + showErrorMessage: vi.fn(), + showTextDocument: vi.fn(), + }, + workspace: { + workspaceFolders: [{ uri: { fsPath: "/mock/workspace" } }], + openTextDocument: vi.fn().mockResolvedValue({}), + }, +})) + +vi.mock("../../../i18n", () => ({ + t: vi.fn((key: string) => key), +})) + +vi.mock("fs/promises", () => { + const readFile = vi.fn().mockResolvedValue("file content here") + return { + default: { + rm: vi.fn(), + mkdir: vi.fn(), + readFile, + writeFile: vi.fn(), + }, + rm: vi.fn(), + mkdir: vi.fn(), + readFile, + writeFile: vi.fn(), + } +}) + +vi.mock("../../../utils/fs") +vi.mock("../../../utils/path") +vi.mock("../../../utils/globalContext") + +vi.mock("../../../utils/pathUtils", () => ({ + isPathOutsideWorkspace: vi.fn((filePath: string) => { + const nodePath = require("path") + const normalized = nodePath.resolve(filePath) + const workspaceRoot = nodePath.resolve("/mock/workspace") + // Path is inside workspace if it equals or is under workspace root + if (normalized === workspaceRoot) return false + if (normalized.startsWith(workspaceRoot + nodePath.sep)) return false + return true + }), +})) + +vi.mock("../../mentions/resolveImageMentions", () => ({ + resolveImageMentions: vi.fn(async ({ text, images }: { text: string; images?: string[] }) => ({ + text, + images: [...(images ?? [])], + })), +})) + +import { webviewMessageHandler } from "../webviewMessageHandler" +import type { ClineProvider } from "../ClineProvider" +import type { ClineMessage } from "@roo-code/types" + +const originalFile = "const a = 1\nconst b = 2\n" + +const toolMessage = (ts: number, payload: unknown, type: "ask" | "say" = "ask", messageId?: string): ClineMessage => + type === "ask" + ? { ts, type: "ask", ask: "tool", text: JSON.stringify(payload), ...(messageId && { messageId }) } + : { ts, type: "say", say: "tool", text: JSON.stringify(payload), ...(messageId && { messageId }) } + +function createProvider(clineMessages: ClineMessage[] | undefined) { + const postMessageToWebview = vi.fn() + // Only the members the handler touches for this message type. + const provider = { + postMessageToWebview, + getCurrentTask: vi.fn().mockReturnValue(clineMessages ? { taskId: "task-1", clineMessages } : undefined), + } as unknown as ClineProvider + + return { provider, postMessageToWebview } +} + +describe("webviewMessageHandler - readOriginalContent", () => { + beforeEach(() => { + vi.clearAllMocks() + }) + + it("answers with the originalContent of the requested tool message", async () => { + const { provider, postMessageToWebview } = createProvider([ + toolMessage(10, { tool: "appliedDiff", path: "a.ts", diff: "d", originalContent: originalFile }), + toolMessage(11, { tool: "appliedDiff", path: "b.ts", diff: "d", originalContent: "other" }), + ]) + + await webviewMessageHandler(provider, { type: "readOriginalContent", messageTs: 10 }) + + expect(postMessageToWebview).toHaveBeenCalledWith({ + type: "originalContent", + originalContentInfo: { ts: 10, messageId: undefined, taskId: undefined, content: originalFile }, + }) + }) + + it("answers with null content for an unknown ts or when there is no current task", async () => { + const unknown = createProvider([toolMessage(40, { tool: "appliedDiff", originalContent: "x" })]) + await webviewMessageHandler(unknown.provider, { type: "readOriginalContent", messageTs: 999 }) + expect(unknown.postMessageToWebview).toHaveBeenCalledWith({ + type: "originalContent", + originalContentInfo: { ts: 999, messageId: undefined, taskId: undefined, content: null }, + }) + + const noTask = createProvider(undefined) + await webviewMessageHandler(noTask.provider, { type: "readOriginalContent", messageTs: 10 }) + expect(noTask.postMessageToWebview).toHaveBeenCalledWith({ + type: "originalContent", + originalContentInfo: { ts: 10, messageId: undefined, taskId: undefined, content: null }, + }) + }) + + it("picks the message by messageId when two messages share a ts", async () => { + const { provider, postMessageToWebview } = createProvider([ + toolMessage(10, { tool: "appliedDiff", path: "a.ts", originalContent: "first" }, "ask", "id-1"), + toolMessage(10, { tool: "appliedDiff", path: "b.ts", originalContent: "second" }, "ask", "id-2"), + ]) + + await webviewMessageHandler(provider, { type: "readOriginalContent", messageTs: 10, messageId: "id-2" }) + + expect(postMessageToWebview).toHaveBeenCalledWith({ + type: "originalContent", + originalContentInfo: { ts: 10, messageId: "id-2", taskId: undefined, content: "second" }, + }) + }) + + it("echoes the task id and answers null for a request made for another task", async () => { + const { provider, postMessageToWebview } = createProvider([ + toolMessage(10, { tool: "appliedDiff", originalContent: originalFile }, "ask", "id-1"), + ]) + + await webviewMessageHandler(provider, { + type: "readOriginalContent", + messageTs: 10, + messageId: "id-1", + taskId: "task-1", + }) + await webviewMessageHandler(provider, { + type: "readOriginalContent", + messageTs: 10, + messageId: "id-1", + taskId: "task-2", + }) + + expect(postMessageToWebview).toHaveBeenNthCalledWith(1, { + type: "originalContent", + originalContentInfo: { ts: 10, messageId: "id-1", taskId: "task-1", content: originalFile }, + }) + expect(postMessageToWebview).toHaveBeenNthCalledWith(2, { + type: "originalContent", + originalContentInfo: { ts: 10, messageId: "id-1", taskId: "task-2", content: null }, + }) + }) + + it("ignores a request without a ts", async () => { + const { provider, postMessageToWebview } = createProvider([ + toolMessage(10, { tool: "appliedDiff", originalContent: "x" }), + ]) + + await webviewMessageHandler(provider, { type: "readOriginalContent" }) + + expect(postMessageToWebview).not.toHaveBeenCalled() + }) +}) diff --git a/src/core/webview/stripOriginalContent.ts b/src/core/webview/stripOriginalContent.ts new file mode 100644 index 0000000000..7d966282dc --- /dev/null +++ b/src/core/webview/stripOriginalContent.ts @@ -0,0 +1,82 @@ +import type { ClineMessage, ExtensionMessage } from "@roo-code/types" + +// Keyed by message object; only the transformed text is cached (valid while the source text is unchanged), so +// metadata such as `isAnswered` and `partial` is always taken from the current message. +const cache = new WeakMap() + +function isToolMessage(message: ClineMessage): boolean { + return (message.type === "ask" && message.ask === "tool") || (message.type === "say" && message.say === "tool") +} + +/** Replaces a file-edit tool message's `originalContent` (the whole pre-edit file) with its length. */ +export function omitOriginalContent(message: ClineMessage): ClineMessage { + const text = message.text + + if (typeof text !== "string" || !isToolMessage(message) || !text.includes('"originalContent"')) { + return message + } + + let entry = cache.get(message) + + if (!entry || entry.source !== text) { + entry = { source: text, strippedText: stripOriginalContentFromText(text) } + cache.set(message, entry) + } + + return entry.strippedText === undefined ? message : { ...message, text: entry.strippedText } +} + +function stripOriginalContentFromText(text: string): string | undefined { + try { + const { originalContent, ...rest } = JSON.parse(text) as Record + + // An empty original (new file) is free and still means "has an original" to the webview, so it stays. + if (typeof originalContent === "string" && originalContent.length > 0) { + return JSON.stringify({ ...rest, originalContentLength: originalContent.length }) + } + } catch { + // Not valid JSON (e.g. a truncated partial message): leave it untouched. + } + + return undefined +} + +/** + * The `originalContent` of a tool message, or null when there is none. `ts` is not unique (two messages can be + * created in the same millisecond), so `messageId` is preferred; `ts` only serves messages persisted without one. + */ +export function findOriginalContent( + messages: ClineMessage[] | undefined, + id: { messageId?: string; ts: number }, +): string | null { + const message = messages?.find( + (m) => isToolMessage(m) && (id.messageId !== undefined ? m.messageId === id.messageId : m.ts === id.ts), + ) + + if (!message?.text) { + return null + } + + try { + const { originalContent } = JSON.parse(message.text) as { originalContent?: unknown } + return typeof originalContent === "string" ? originalContent : null + } catch { + return null + } +} + +/** The webview gets `originalContent` on demand (`readOriginalContent`) instead of inside every chat message. */ +export function omitOriginalContentFromExtensionMessage(message: ExtensionMessage): ExtensionMessage { + if (message.type === "state" && message.state?.clineMessages?.length) { + return { + ...message, + state: { ...message.state, clineMessages: message.state.clineMessages.map(omitOriginalContent) }, + } + } + + if (message.type === "messageUpdated" && message.clineMessage) { + return { ...message, clineMessage: omitOriginalContent(message.clineMessage) } + } + + return message +} diff --git a/src/core/webview/webviewMessageHandler.ts b/src/core/webview/webviewMessageHandler.ts index 4ba94d454c..193540455b 100644 --- a/src/core/webview/webviewMessageHandler.ts +++ b/src/core/webview/webviewMessageHandler.ts @@ -40,6 +40,7 @@ import { saveTaskMessages } from "../task-persistence" import { importRooTaskHistory } from "../task-persistence/importRooTaskHistory" import { ClineProvider } from "./ClineProvider" +import { findOriginalContent } from "./stripOriginalContent" import { handleCheckpointRestoreOperation } from "./checkpointRestoreHandler" import { generateErrorDiagnostics } from "./diagnosticsHandler" import { @@ -1617,6 +1618,29 @@ export const webviewMessageHandler = async ( } break } + case "readOriginalContent": { + const ts = message.messageTs + + if (typeof ts !== "number") { + break + } + + const task = provider.getCurrentTask() + // A request made for another task must not be answered from the current one. + const isRequestedTask = message.taskId === undefined || message.taskId === task?.taskId + const id = { messageId: message.messageId, ts } + + await provider.postMessageToWebview({ + type: "originalContent", + originalContentInfo: { + ts, + messageId: message.messageId, + taskId: message.taskId, + content: isRequestedTask ? findOriginalContent(task?.clineMessages, id) : null, + }, + }) + break + } case "openMention": await openMention(getCurrentCwd(), message.text) break diff --git a/webview-ui/src/__tests__/FileChangesPanel.spec.tsx b/webview-ui/src/__tests__/FileChangesPanel.spec.tsx index 2208bc127d..3e5f2e26f1 100644 --- a/webview-ui/src/__tests__/FileChangesPanel.spec.tsx +++ b/webview-ui/src/__tests__/FileChangesPanel.spec.tsx @@ -1,4 +1,5 @@ import React from "react" +import { act } from "@testing-library/react" import { fireEvent, render, screen } from "@/utils/test-utils" import type { ClineMessage } from "@roo-code/types" import { TranslationProvider } from "@/i18n/__mocks__/TranslationContext" @@ -28,15 +29,18 @@ vi.mock("react-i18next", () => ({ vi.mock("@src/components/common/CodeAccordion", () => ({ default: ({ path, + code, isExpanded, onToggleExpand, }: { path?: string + code?: string isExpanded: boolean onToggleExpand: () => void }) => (
{path} +
{code}
@@ -64,10 +68,10 @@ function createFileEditMessage( } } -function renderPanel(messages: ClineMessage[] | undefined) { +function renderPanel(messages: ClineMessage[] | undefined, taskId?: string) { return render( - + , ) } @@ -196,4 +200,207 @@ describe("FileChangesPanel", () => { expect(screen.getByTestId("total-added")).toHaveTextContent("+5") expect(screen.getByTestId("total-removed")).toHaveTextContent("-6") }) + describe("original content omitted by the extension", () => { + const TS = 1234 + + function createEditWithOriginal(payload: Record): ClineMessage { + return { + type: "ask", + ask: "tool", + ts: TS, + partial: false, + isAnswered: true, + text: JSON.stringify({ + tool: "appliedDiff", + path: "src/foo.ts", + diff: "the recorded diff", + ...payload, + }), + } + } + + function expandRow() { + fireEvent.click(screen.getByText("1 file(s) changed in this conversation").closest("button")!) + fireEvent.click(screen.getByTestId("accordian-toggle")) + } + + function respond(message: Record) { + act(() => { + window.dispatchEvent(new MessageEvent("message", { data: message })) + }) + } + + const requestsOfType = (type: string) => + mockPostMessage.mock.calls.map(([m]) => m).filter((m: { type: string }) => m.type === type) + + it("requests nothing until a row is expanded, then asks for the final and the original content", () => { + renderPanel([createEditWithOriginal({ originalContentLength: 5000 })]) + fireEvent.click(screen.getByText("1 file(s) changed in this conversation").closest("button")!) + + expect(mockPostMessage).not.toHaveBeenCalled() + + fireEvent.click(screen.getByTestId("accordian-toggle")) + + expect(requestsOfType("readFileContent")).toEqual([{ type: "readFileContent", text: "src/foo.ts" }]) + expect(requestsOfType("readOriginalContent")).toEqual([ + { type: "readOriginalContent", messageTs: TS, messageId: undefined, taskId: undefined }, + ]) + }) + + it("shows the merged diff once both the original and the final content arrive", () => { + renderPanel([createEditWithOriginal({ originalContentLength: 5000 })]) + expandRow() + + expect(screen.getByTestId("accordian-code")).toHaveTextContent("the recorded diff") + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + expect(screen.getByTestId("accordian-code")).toHaveTextContent("the recorded diff") + + respond({ type: "originalContent", originalContentInfo: { ts: TS, content: "old line\n" } }) + expect(screen.getByTestId("accordian-code")).toHaveTextContent("-old line") + expect(screen.getByTestId("accordian-code")).toHaveTextContent("+new line") + }) + + it("keeps the recorded diff when the original cannot be loaded", () => { + renderPanel([createEditWithOriginal({ originalContentLength: 5000 })]) + expandRow() + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + respond({ type: "originalContent", originalContentInfo: { ts: TS, content: null } }) + + expect(screen.getByTestId("accordian-code")).toHaveTextContent("the recorded diff") + expect(requestsOfType("readOriginalContent")).toHaveLength(1) + }) + + it("does not request the original when it is already inline", () => { + renderPanel([createEditWithOriginal({ originalContent: "old line\n" })]) + expandRow() + + expect(requestsOfType("readOriginalContent")).toHaveLength(0) + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + expect(screen.getByTestId("accordian-code")).toHaveTextContent("-old line") + expect(screen.getByTestId("accordian-code")).toHaveTextContent("+new line") + }) + + it("requests nothing for an edit that has no original", () => { + renderPanel([createEditWithOriginal({})]) + expandRow() + + expect(mockPostMessage).not.toHaveBeenCalled() + }) + + it("requests and matches originals by messageId when messages share a ts", () => { + const edit = (path: string, messageId: string): ClineMessage => ({ + ...createEditWithOriginal({ path, originalContentLength: 5000 }), + messageId, + }) + renderPanel([edit("src/a.ts", "id-a"), edit("src/b.ts", "id-b")], "task-1") + fireEvent.click(screen.getByText("2 file(s) changed in this conversation").closest("button")!) + screen.getAllByTestId("accordian-toggle").forEach((toggle) => fireEvent.click(toggle)) + + expect(requestsOfType("readOriginalContent")).toEqual([ + { type: "readOriginalContent", messageTs: TS, messageId: "id-a", taskId: "task-1" }, + { type: "readOriginalContent", messageTs: TS, messageId: "id-b", taskId: "task-1" }, + ]) + + respond({ type: "fileContent", fileContent: { path: "src/a.ts", content: "new a\n" } }) + respond({ type: "fileContent", fileContent: { path: "src/b.ts", content: "new b\n" } }) + respond({ + type: "originalContent", + originalContentInfo: { ts: TS, messageId: "id-b", taskId: "task-1", content: "old b\n" }, + }) + + const [a, b] = screen.getAllByTestId("accordian-code") + expect(a).toHaveTextContent("the recorded diff") + expect(b).toHaveTextContent("-old b") + }) + + it("requests the original again when the task id arrives after a request is already pending", () => { + const messages = [createEditWithOriginal({ originalContentLength: 5000 })] + const { rerender } = renderPanel(messages) + expandRow() + expect(requestsOfType("readOriginalContent")).toHaveLength(1) + + rerender( + + + , + ) + fireEvent.click(screen.getByTestId("accordian-toggle")) + + const requests = requestsOfType("readOriginalContent") + expect(requests).toHaveLength(2) + expect(requests[1]).toMatchObject({ taskId: "task-1" }) + }) + + it("does not send a duplicate request when switching A -> B -> A before the first response arrives", () => { + const messages = [createEditWithOriginal({ originalContentLength: 5000 })] + const panel = (taskId: string) => ( + + + + ) + const { rerender } = renderPanel(messages, "task-A") + expandRow() + expect(requestsOfType("readOriginalContent")).toHaveLength(1) + + rerender(panel("task-B")) + rerender(panel("task-A")) + fireEvent.click(screen.getByTestId("accordian-toggle")) + + expect(requestsOfType("readOriginalContent").filter((m) => m.taskId === "task-A")).toHaveLength(1) + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + respond({ + type: "originalContent", + originalContentInfo: { ts: TS, taskId: "task-A", content: "old line\n" }, + }) + expect(screen.getByTestId("accordian-code")).toHaveTextContent("-old line") + }) + + it("requests again after a response that arrived for a task that is no longer current", () => { + const messages = [createEditWithOriginal({ originalContentLength: 5000 })] + const panel = (taskId: string) => ( + + + + ) + const { rerender } = renderPanel(messages, "task-A") + expandRow() + + rerender(panel("task-B")) + respond({ + type: "originalContent", + originalContentInfo: { ts: TS, taskId: "task-A", content: "old line\n" }, + }) + rerender(panel("task-A")) + fireEvent.click(screen.getByTestId("accordian-toggle")) + + expect(requestsOfType("readOriginalContent").filter((m) => m.taskId === "task-A")).toHaveLength(2) + }) + + it("ignores an original answered for a different task", () => { + renderPanel([createEditWithOriginal({ originalContentLength: 5000 })], "task-1") + expandRow() + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + respond({ + type: "originalContent", + originalContentInfo: { ts: TS, taskId: "task-2", content: "old line\n" }, + }) + + expect(screen.getByTestId("accordian-code")).toHaveTextContent("the recorded diff") + }) + + it("ignores an original for a different message", () => { + renderPanel([createEditWithOriginal({ originalContentLength: 5000 })]) + expandRow() + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + respond({ type: "originalContent", originalContentInfo: { ts: TS + 1, content: "old line\n" } }) + + expect(screen.getByTestId("accordian-code")).toHaveTextContent("the recorded diff") + }) + }) }) diff --git a/webview-ui/src/__tests__/fileChangesFromMessages.spec.ts b/webview-ui/src/__tests__/fileChangesFromMessages.spec.ts index 8fab8b14d5..b09f4978c4 100644 --- a/webview-ui/src/__tests__/fileChangesFromMessages.spec.ts +++ b/webview-ui/src/__tests__/fileChangesFromMessages.spec.ts @@ -105,6 +105,8 @@ describe("fileChangesFromMessages", () => { path: "src/foo.ts", diff: "@@ -1 +1 @@\n+line", diffStats: { added: 1, removed: 0 }, + ts: messages[0].ts, + hasOriginalContent: false, }) }) @@ -190,7 +192,7 @@ describe("fileChangesFromMessages", () => { ] const result = fileChangesFromMessages(messages) expect(result).toHaveLength(2) - expect(result[0]).toEqual({ path: "a.ts", diff: "content a" }) + expect(result[0]).toEqual({ path: "a.ts", diff: "content a", ts: messages[0].ts, hasOriginalContent: false }) expect(result[1].path).toBe("b.ts") expect(result[1].diff).toBe("content b") }) @@ -277,4 +279,44 @@ describe("fileChangesFromMessages", () => { ] expect(fileChangesFromMessages(messages)).toEqual([]) }) + describe("original content (omitted by the extension to keep the webview small)", () => { + const editMessage = (payload: Record) => + msg({ + type: "ask", + ask: "tool", + isAnswered: true, + text: JSON.stringify({ tool: "appliedDiff", path: "src/foo.ts", diff: "d", ...payload }), + }) + + it("keeps an inline originalContent and marks the entry as having one", () => { + const message = editMessage({ originalContent: "old file" }) + const [entry] = fileChangesFromMessages([message]) + + expect(entry.originalContent).toBe("old file") + expect(entry.hasOriginalContent).toBe(true) + expect(entry.ts).toBe(message.ts) + }) + + it("marks an entry whose originalContent was omitted (originalContentLength) as requestable", () => { + const message = editMessage({ originalContentLength: 4096 }) + const [entry] = fileChangesFromMessages([message]) + + expect(entry.originalContent).toBeUndefined() + expect(entry.hasOriginalContent).toBe(true) + expect(entry.ts).toBe(message.ts) + }) + + it("treats an empty inline originalContent (a new file) as an original", () => { + const [entry] = fileChangesFromMessages([editMessage({ originalContent: "" })]) + + expect(entry.originalContent).toBe("") + expect(entry.hasOriginalContent).toBe(true) + }) + + it("marks an entry without any original as not having one", () => { + const [entry] = fileChangesFromMessages([editMessage({})]) + + expect(entry.hasOriginalContent).toBe(false) + }) + }) }) diff --git a/webview-ui/src/components/chat/ChatView.tsx b/webview-ui/src/components/chat/ChatView.tsx index 4694eeabf7..e75041c457 100644 --- a/webview-ui/src/components/chat/ChatView.tsx +++ b/webview-ui/src/components/chat/ChatView.tsx @@ -1738,7 +1738,7 @@ const ChatViewComponent: React.ForwardRefRenderFunction
- + {areButtonsVisible && (
{ +// `ts` is not unique, so the message's own id identifies it; `ts` only covers messages persisted without one. +const originalKey = (entry: { messageId?: string; ts: number }) => entry.messageId ?? `ts:${entry.ts}` + +const FileChangesPanel = memo(({ clineMessages, taskId, className }: FileChangesPanelProps) => { const { t } = useTranslation() const [panelExpanded, setPanelExpanded] = useState(false) const [expandedPaths, setExpandedPaths] = useState>(new Set()) const [finalContentByPath, setFinalContentByPath] = useState>({}) const pendingPathsRef = useRef>(new Set()) + // The extension omits `originalContent` from the messages it posts; it is requested when a row is expanded. + const [originalContentByKey, setOriginalContentByKey] = useState>({}) + // In-flight requests, keyed by task and message. Deliberately not reset with the caches below: a request cannot + // be cancelled, so a reset would let a task switch (A -> B -> A) send a duplicate while the first is still open. + const pendingOriginalRequestsRef = useRef>(new Set()) // Reset expanded file rows and final content cache when switching to a different task useEffect(() => { setExpandedPaths(new Set()) setFinalContentByPath({}) pendingPathsRef.current = new Set() - }, [clineMessages]) + setOriginalContentByKey({}) + }, [clineMessages, taskId]) const fileChanges = useMemo(() => fileChangesFromMessages(clineMessages), [clineMessages]) @@ -65,25 +75,40 @@ const FileChangesPanel = memo(({ clineMessages, className }: FileChangesPanelPro }) }, []) - // Request final file content when a row is expanded and we have originalContent + // Request the final file content (and the omitted original content) when a row is expanded and the edit has an original useEffect(() => { for (const path of expandedPaths) { const entries = byPath.get(path) if (!entries?.length) continue - const originalContent = entries[0].originalContent + const first = entries[0] const lookupPath = path.startsWith("./") ? path.slice(2) : path if ( - originalContent !== undefined && + first.hasOriginalContent && !(lookupPath in finalContentByPath) && !pendingPathsRef.current.has(lookupPath) ) { pendingPathsRef.current.add(lookupPath) vscode.postMessage({ type: "readFileContent", text: lookupPath }) } + const requestKey = `${taskId ?? ""}|${originalKey(first)}` + if ( + first.hasOriginalContent && + first.originalContent === undefined && + !(originalKey(first) in originalContentByKey) && + !pendingOriginalRequestsRef.current.has(requestKey) + ) { + pendingOriginalRequestsRef.current.add(requestKey) + vscode.postMessage({ + type: "readOriginalContent", + messageTs: first.ts, + messageId: first.messageId, + taskId, + }) + } } - }, [expandedPaths, byPath, finalContentByPath]) + }, [expandedPaths, byPath, finalContentByPath, originalContentByKey, taskId]) - // Listen for fileContent responses + // Listen for fileContent and originalContent responses useEffect(() => { const handler = (event: MessageEvent) => { const message: ExtensionMessage = event.data @@ -91,11 +116,18 @@ const FileChangesPanel = memo(({ clineMessages, className }: FileChangesPanelPro const fc = message.fileContent pendingPathsRef.current.delete(fc.path) setFinalContentByPath((prev) => ({ ...prev, [fc.path]: fc.content ?? null })) + } else if (message.type === "originalContent" && message.originalContentInfo) { + const { taskId: responseTaskId, content, ...id } = message.originalContentInfo + const key = originalKey(id) + pendingOriginalRequestsRef.current.delete(`${responseTaskId ?? ""}|${key}`) + // A late response for another task must not populate this task's cache. + if (responseTaskId !== taskId) return + setOriginalContentByKey((prev) => ({ ...prev, [key]: content })) } } window.addEventListener("message", handler) return () => window.removeEventListener("message", handler) - }, []) + }, [taskId]) if (fileChanges.length === 0) return null @@ -133,7 +165,8 @@ const FileChangesPanel = memo(({ clineMessages, className }: FileChangesPanelPro
{Array.from(byPath.entries()).map(([path, entries]) => { - const originalContent = entries[0].originalContent + const originalContent = + entries[0].originalContent ?? originalContentByKey[originalKey(entries[0])] ?? undefined const lookupPath = path.startsWith("./") ? path.slice(2) : path const finalContent = finalContentByPath[lookupPath] const hasMergedDiff = diff --git a/webview-ui/src/components/chat/utils/fileChangesFromMessages.ts b/webview-ui/src/components/chat/utils/fileChangesFromMessages.ts index 738305ad15..35c55e496d 100644 --- a/webview-ui/src/components/chat/utils/fileChangesFromMessages.ts +++ b/webview-ui/src/components/chat/utils/fileChangesFromMessages.ts @@ -10,6 +10,11 @@ export interface FileChangeEntry { diffStats?: { added: number; removed: number } /** Original file content before first edit (for merged diff display) */ originalContent?: string + /** Identity of the message this entry comes from; used to request `originalContent` when the extension omitted it */ + ts: number + messageId?: string + /** True when the edit has an original (inline, or omitted by the extension and requestable by `ts`) */ + hasOriginalContent: boolean } /** @@ -44,6 +49,9 @@ export function fileChangesFromMessages(messages: ClineMessage[] | undefined): F path: file.path, diff: content, diffStats: file.diffStats, + ts: msg.ts, + messageId: msg.messageId, + hasOriginalContent: false, }) } } @@ -59,6 +67,9 @@ export function fileChangesFromMessages(messages: ClineMessage[] | undefined): F diff, diffStats: tool.diffStats, originalContent: tool.originalContent, + ts: msg.ts, + messageId: msg.messageId, + hasOriginalContent: tool.originalContent !== undefined || (tool.originalContentLength ?? 0) > 0, }) } } From ae5662ad622ea255bcb57ad21ea3e03a89a4da82 Mon Sep 17 00:00:00 2001 From: daewoongoh Date: Fri, 2 Oct 2026 16:08:23 +0900 Subject: [PATCH 2/3] fix: address review on original-content fetch and gray-screen tooling Keep loaded originals across message updates and skip requests for rows expanded under a previous task. Pass currentTaskId to FileChangesPanel. Move shared tooling helpers to lib.mjs: validate numeric flags and --dir (no "." or option-like segments), stage generated tasks in a separate directory before renaming them into place, and add node:test coverage. --- scripts/gray-screen/__tests__/lib.test.mjs | 104 ++++++++++++++++++ scripts/gray-screen/generate-large-task.mjs | 41 +++---- scripts/gray-screen/lib.mjs | 77 +++++++++++++ scripts/gray-screen/mock-openai-server.mjs | 7 +- scripts/gray-screen/webview-heap-matrix.mjs | 16 +-- scripts/gray-screen/webview-render-stress.mjs | 16 +-- .../src/__tests__/FileChangesPanel.spec.tsx | 38 +++++++ webview-ui/src/components/chat/ChatView.tsx | 2 +- .../src/components/chat/FileChangesPanel.tsx | 33 ++++-- 9 files changed, 266 insertions(+), 68 deletions(-) create mode 100644 scripts/gray-screen/__tests__/lib.test.mjs create mode 100644 scripts/gray-screen/lib.mjs diff --git a/scripts/gray-screen/__tests__/lib.test.mjs b/scripts/gray-screen/__tests__/lib.test.mjs new file mode 100644 index 0000000000..af47a20eab --- /dev/null +++ b/scripts/gray-screen/__tests__/lib.test.mjs @@ -0,0 +1,104 @@ +import assert from "node:assert/strict" +import fs from "node:fs" +import os from "node:os" +import path from "node:path" +import { afterEach, beforeEach, describe, it } from "node:test" + +import { integerFlag, parseFlagArgs, resolveServedFile, validateRelativeDir, writeTaskAtomically } from "../lib.mjs" + +let tmp + +beforeEach(() => { + tmp = fs.mkdtempSync(path.join(os.tmpdir(), "gray-screen-test-")) +}) + +afterEach(() => { + fs.rmSync(tmp, { recursive: true, force: true }) +}) + +describe("parseFlagArgs / integerFlag", () => { + it("parses flag/value pairs", () => { + assert.deepEqual(parseFlagArgs(["--messages", "10", "--two-byte", "true"]), { messages: "10", "two-byte": "true" }) + }) + + it("rejects a flag with a missing operand", () => { + assert.throws(() => parseFlagArgs(["--messages", "--text-bytes", "5"]), /Missing value for --messages/) + assert.throws(() => parseFlagArgs(["--messages"]), /Missing value for --messages/) + }) + + it("accepts integers at or above the minimum and rejects everything else", () => { + assert.equal(integerFlag("messages", "5000", 1), 5000) + assert.equal(integerFlag("image-every", "0", 0), 0) + for (const bad of ["0", "-3", "1.5", "abc", "", "NaN"]) { + assert.throws(() => integerFlag("messages", bad, 1), /--messages must be an integer >= 1/) + } + }) +}) + +describe("validateRelativeDir", () => { + it("accepts plain relative paths", () => { + for (const ok of [".mock-session", "a", "src/mock_1", "a.b/c-d"]) assert.equal(validateRelativeDir(ok), ok) + }) + + it("rejects shell metacharacters, absolute paths, traversal, dot and option-like segments", () => { + const bad = ['safe"; touch /tmp/pwned; #', "a b", "$(id)", "/abs", "../x", "a/../b", ".", "a/./b", "-rf", "a/-rf", "", "a//b"] + for (const dir of bad) assert.throws(() => validateRelativeDir(dir), /Invalid --dir/, JSON.stringify(dir)) + }) +}) + +describe("resolveServedFile", () => { + beforeEach(() => { + fs.mkdirSync(path.join(tmp, "build", "assets"), { recursive: true }) + fs.mkdirSync(path.join(tmp, "build-secret")) + fs.writeFileSync(path.join(tmp, "build", "index.html"), "index") + fs.writeFileSync(path.join(tmp, "build", "assets", "app.js"), "js") + fs.writeFileSync(path.join(tmp, "build-secret", "credentials.json"), "secret") + fs.symlinkSync(path.join(tmp, "build-secret"), path.join(tmp, "build", "link")) + }) + + const root = () => path.join(tmp, "build") + + it("serves files inside the build directory, with / mapping to index.html", () => { + assert.equal(resolveServedFile(root(), "/"), fs.realpathSync(path.join(root(), "index.html"))) + assert.equal(resolveServedFile(root(), "/assets/app.js"), fs.realpathSync(path.join(root(), "assets", "app.js"))) + }) + + it("returns undefined for traversal (plain, encoded and sibling-prefix), symlink escapes, directories and bad input", () => { + for (const p of [ + "/../build-secret/credentials.json", + "/%2e%2e%2fbuild-secret%2fcredentials.json", + "/link/credentials.json", + "/assets", + "/missing.js", + "/%E0%A4%A", + "/a\0b", + ]) { + assert.equal(resolveServedFile(root(), p), undefined, p) + } + }) + + it("returns undefined when the build directory does not exist", () => { + assert.equal(resolveServedFile(path.join(tmp, "nope"), "/"), undefined) + }) +}) + +describe("writeTaskAtomically", () => { + it("writes every file into tasks/ and leaves no staging directory", () => { + const taskDir = writeTaskAtomically(tmp, "t1", { "a.json": { a: 1 }, "b.json": [2] }) + + assert.equal(taskDir, path.join(tmp, "tasks", "t1")) + assert.deepEqual(JSON.parse(fs.readFileSync(path.join(taskDir, "a.json"), "utf8")), { a: 1 }) + assert.deepEqual(JSON.parse(fs.readFileSync(path.join(taskDir, "b.json"), "utf8")), [2]) + assert.deepEqual(fs.readdirSync(tmp).sort(), ["tasks"]) + }) + + it("leaves neither a task directory nor staging files when a write fails", () => { + const circular = {} + circular.self = circular + + assert.throws(() => writeTaskAtomically(tmp, "t2", { "ok.json": { ok: true }, "bad.json": circular }), /circular/i) + + assert.equal(fs.existsSync(path.join(tmp, "tasks", "t2")), false) + assert.deepEqual(fs.readdirSync(tmp), []) + }) +}) diff --git a/scripts/gray-screen/generate-large-task.mjs b/scripts/gray-screen/generate-large-task.mjs index 3ef5df814e..c2db6330f5 100644 --- a/scripts/gray-screen/generate-large-task.mjs +++ b/scripts/gray-screen/generate-large-task.mjs @@ -9,22 +9,19 @@ import fs from "node:fs" import os from "node:os" import path from "node:path" import crypto from "node:crypto" +import { integerFlag, parseFlagArgs, writeTaskAtomically } from "./lib.mjs" -const args = Object.fromEntries( - process.argv.slice(2).reduce((acc, cur, i, all) => { - if (cur.startsWith("--")) acc.push([cur.slice(2), all[i + 1]]) - return acc - }, []), -) +const args = parseFlagArgs(process.argv.slice(2)) -const messageCount = Number(args["messages"] ?? 5000) -const textBytes = Number(args["text-bytes"] ?? 800) -const toolBytes = Number(args["tool-bytes"] ?? textBytes) +// Validate every numeric flag up front, before anything is written. +const messageCount = integerFlag("messages", args["messages"] ?? 5000, 1) +const textBytes = integerFlag("text-bytes", args["text-bytes"] ?? 800, 1) +const toolBytes = integerFlag("tool-bytes", args["tool-bytes"] ?? textBytes, 1) const toolKind = args["tool-kind"] ?? "mixed" const batchable = args["batchable"] === "true" const twoByte = args["two-byte"] === "true" // Korean chars in tool text -> UTF-16 strings (2 bytes/char) in V8 -const imageEvery = Number(args["image-every"] ?? 0) -const imageKb = Number(args["image-kb"] ?? 200) +const imageEvery = integerFlag("image-every", args["image-every"] ?? 0, 0) +const imageKb = integerFlag("image-kb", args["image-kb"] ?? 200, 1) const workspace = args["workspace"] ?? process.cwd() const storageCandidates = [ path.join(os.homedir(), ".vscode-server", "data", "User", "globalStorage", "codemate.zoo-code"), @@ -33,8 +30,6 @@ const storageCandidates = [ const storage = args["storage"] ?? storageCandidates.find((p) => fs.existsSync(p)) ?? storageCandidates[1] const taskId = crypto.randomUUID() -const taskDir = path.join(storage, "tasks", taskId) -fs.mkdirSync(taskDir, { recursive: true }) const filler = (n, seed) => { const base = `line ${seed}: ${twoByte ? "\uD55C\uAE00 " : ""}The quick brown fox jumps over the lazy dog. ` @@ -104,20 +99,12 @@ const historyItem = { status: "completed", } -// Write every file under a temporary name and rename only after all writes succeeded, so an interrupted run -// (or a full disk) never leaves a truncated or partial task; history_item.json goes last. -const taskFiles = [ - ["ui_messages.json", messages], - ["api_conversation_history.json", apiHistory], - ["history_item.json", historyItem], -] -try { - for (const [name, value] of taskFiles) fs.writeFileSync(path.join(taskDir, `${name}.tmp`), JSON.stringify(value)) - for (const [name] of taskFiles) fs.renameSync(path.join(taskDir, `${name}.tmp`), path.join(taskDir, name)) -} catch (error) { - fs.rmSync(taskDir, { recursive: true, force: true }) - throw error -} +// The task becomes visible under tasks/ only once all three files are written. +const taskDir = writeTaskAtomically(storage, taskId, { + "ui_messages.json": messages, + "api_conversation_history.json": apiHistory, + "history_item.json": historyItem, +}) const mb = (f) => (fs.statSync(path.join(taskDir, f)).size / 1048576).toFixed(1) console.log(`Task ${taskId}: ${messages.length} messages, ui_messages.json ${mb("ui_messages.json")} MB`) diff --git a/scripts/gray-screen/lib.mjs b/scripts/gray-screen/lib.mjs new file mode 100644 index 0000000000..fb9c0be48d --- /dev/null +++ b/scripts/gray-screen/lib.mjs @@ -0,0 +1,77 @@ +// Shared helpers for the gray-screen tooling. Tests: node --test scripts/gray-screen/__tests__/lib.test.mjs +import fs from "node:fs" +import path from "node:path" + +/** `--flag value` pairs; a flag without a value (or followed by another flag) is an error. */ +export function parseFlagArgs(argv) { + const entries = [] + + argv.forEach((cur, i) => { + if (!cur.startsWith("--")) return + const value = argv[i + 1] + if (value === undefined || value.startsWith("--")) throw new Error(`Missing value for ${cur}`) + entries.push([cur.slice(2), value]) + }) + + return Object.fromEntries(entries) +} + +export function integerFlag(name, value, min) { + const number = Number(value) + if (!Number.isInteger(number) || number < min) throw new Error(`--${name} must be an integer >= ${min}`) + return number +} + +/** A plain relative path that is safe to interpolate into shell commands and to resolve under a workspace. */ +export function validateRelativeDir(dir) { + const segments = dir.split("/") + const isSafe = + /^[A-Za-z0-9._-]+(\/[A-Za-z0-9._-]+)*$/.test(dir) && + segments.every((part) => part !== "." && part !== ".." && !part.startsWith("-")) + + if (!isSafe) { + throw new Error( + `Invalid --dir "${dir}": use a relative path of letters, digits, ".", "_" and "-" without "." or ".." segments and without segments starting with "-".`, + ) + } + + return dir +} + +/** Resolves a request path to a regular file inside `root` (after symlink resolution), or undefined. */ +export function resolveServedFile(root, rawPath) { + try { + const realRoot = fs.realpathSync(root) + const decoded = decodeURIComponent(rawPath) + const real = fs.realpathSync(path.resolve(realRoot, "." + (decoded === "/" ? "/index.html" : decoded))) + const rel = path.relative(realRoot, real) + if (rel === "" || rel === ".." || rel.startsWith(".." + path.sep) || path.isAbsolute(rel)) return undefined + return fs.statSync(real).isFile() ? real : undefined + } catch { + return undefined + } +} + +/** + * Writes a task into `/tasks/` all-or-nothing: the files are written into a staging directory + * next to `tasks/`, which is renamed into place only after every write succeeded. + * `files` maps a file name to the value to serialize. Returns the final task directory. + */ +export function writeTaskAtomically(storage, taskId, files) { + const staging = path.join(storage, `.task-staging-${taskId}`) + const taskDir = path.join(storage, "tasks", taskId) + + try { + fs.mkdirSync(staging, { recursive: true }) + for (const [name, value] of Object.entries(files)) { + fs.writeFileSync(path.join(staging, name), JSON.stringify(value)) + } + fs.mkdirSync(path.dirname(taskDir), { recursive: true }) + fs.renameSync(staging, taskDir) + } catch (error) { + fs.rmSync(staging, { recursive: true, force: true }) + throw error + } + + return taskDir +} diff --git a/scripts/gray-screen/mock-openai-server.mjs b/scripts/gray-screen/mock-openai-server.mjs index 9a3b8c7cff..8a887c930f 100644 --- a/scripts/gray-screen/mock-openai-server.mjs +++ b/scripts/gray-screen/mock-openai-server.mjs @@ -33,6 +33,7 @@ // Zoo Code settings: provider "OpenAI Compatible", base URL http://127.0.0.1:/v1, any API key, model id "mock", // context window 1000000 (avoid condensing); auto-approve read/write/execute with allowed command "*". import http from "node:http" +import { validateRelativeDir } from "./lib.mjs" const args = Object.fromEntries( process.argv.slice(2).reduce((acc, cur, i, all) => { @@ -61,8 +62,10 @@ const cfg = { } // cfg.dir is interpolated into the shell commands the mock asks the extension to run, so only plain relative paths are allowed. -if (!/^[A-Za-z0-9._-]+(\/[A-Za-z0-9._-]+)*$/.test(cfg.dir) || cfg.dir.split("/").some((part) => part === "..")) { - console.error(`Invalid --dir "${cfg.dir}": use a relative path of letters, digits, ".", "_" and "-" without ".." segments.`) +try { + validateRelativeDir(cfg.dir) +} catch (error) { + console.error(error.message) process.exit(1) } diff --git a/scripts/gray-screen/webview-heap-matrix.mjs b/scripts/gray-screen/webview-heap-matrix.mjs index 67e944e51b..8f368a1022 100644 --- a/scripts/gray-screen/webview-heap-matrix.mjs +++ b/scripts/gray-screen/webview-heap-matrix.mjs @@ -11,6 +11,7 @@ import http from "node:http" import { createRequire } from "node:module" import path from "node:path" import { fileURLToPath } from "node:url" +import { resolveServedFile } from "./lib.mjs" const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..") const require = createRequire(path.join(root, "webview-ui", "package.json")) @@ -68,21 +69,8 @@ const EXPERIMENTS = [ ] const types = { ".html": "text/html", ".js": "text/javascript", ".css": "text/css", ".json": "application/json", ".wasm": "application/wasm", ".svg": "image/svg+xml", ".map": "application/json", ".woff2": "font/woff2", ".ttf": "font/ttf" } -const serveRoot = fs.realpathSync(buildDir) -// Resolves a request path to a regular file inside serveRoot (after symlink resolution), or undefined. -const resolveServedFile = (rawPath) => { - try { - const decoded = decodeURIComponent(rawPath) - const real = fs.realpathSync(path.resolve(serveRoot, "." + (decoded === "/" ? "/index.html" : decoded))) - const rel = path.relative(serveRoot, real) - if (rel === "" || rel === ".." || rel.startsWith(".." + path.sep) || path.isAbsolute(rel)) return undefined - return fs.statSync(real).isFile() ? real : undefined - } catch { - return undefined - } -} const httpServer = http.createServer((req, res) => { - const file = resolveServedFile(new URL(req.url, "http://x").pathname) + const file = resolveServedFile(buildDir, new URL(req.url, "http://x").pathname) if (!file) return void res.writeHead(404).end() res.writeHead(200, { "content-type": types[path.extname(file)] ?? "application/octet-stream", diff --git a/scripts/gray-screen/webview-render-stress.mjs b/scripts/gray-screen/webview-render-stress.mjs index 25fb4c528d..bd80ca52f6 100644 --- a/scripts/gray-screen/webview-render-stress.mjs +++ b/scripts/gray-screen/webview-render-stress.mjs @@ -28,6 +28,7 @@ import http from "node:http" import { createRequire } from "node:module" import path from "node:path" import { fileURLToPath } from "node:url" +import { resolveServedFile } from "./lib.mjs" const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..") const webviewDir = path.join(root, "webview-ui") @@ -78,21 +79,8 @@ async function startServer() { }) } const types = { ".html": "text/html", ".js": "text/javascript", ".css": "text/css", ".json": "application/json", ".wasm": "application/wasm", ".svg": "image/svg+xml", ".map": "application/json", ".woff2": "font/woff2", ".ttf": "font/ttf" } - const serveRoot = fs.realpathSync(buildDir) - // Resolves a request path to a regular file inside serveRoot (after symlink resolution), or undefined. - const resolveServedFile = (rawPath) => { - try { - const decoded = decodeURIComponent(rawPath) - const real = fs.realpathSync(path.resolve(serveRoot, "." + (decoded === "/" ? "/index.html" : decoded))) - const rel = path.relative(serveRoot, real) - if (rel === "" || rel === ".." || rel.startsWith(".." + path.sep) || path.isAbsolute(rel)) return undefined - return fs.statSync(real).isFile() ? real : undefined - } catch { - return undefined - } - } const httpServer = http.createServer((req, res) => { - const file = resolveServedFile(new URL(req.url, "http://x").pathname) + const file = resolveServedFile(buildDir, new URL(req.url, "http://x").pathname) if (!file) { res.writeHead(404).end() return diff --git a/webview-ui/src/__tests__/FileChangesPanel.spec.tsx b/webview-ui/src/__tests__/FileChangesPanel.spec.tsx index 3e5f2e26f1..a1c3701955 100644 --- a/webview-ui/src/__tests__/FileChangesPanel.spec.tsx +++ b/webview-ui/src/__tests__/FileChangesPanel.spec.tsx @@ -380,6 +380,44 @@ describe("FileChangesPanel", () => { expect(requestsOfType("readOriginalContent").filter((m) => m.taskId === "task-A")).toHaveLength(2) }) + it("does not request for rows expanded under the previous task when the task id changes", () => { + const messages = [createEditWithOriginal({ originalContentLength: 5000 })] + const { rerender } = renderPanel(messages, "task-A") + expandRow() + mockPostMessage.mockClear() + + rerender( + + + , + ) + + expect(mockPostMessage).not.toHaveBeenCalled() + }) + + it("keeps a loaded original when the messages are replaced by an update of the same task", () => { + const first = [createEditWithOriginal({ originalContentLength: 5000 })] + const { rerender } = renderPanel(first, "task-A") + expandRow() + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + respond({ + type: "originalContent", + originalContentInfo: { ts: TS, taskId: "task-A", content: "old line\n" }, + }) + expect(requestsOfType("readOriginalContent")).toHaveLength(1) + + rerender( + + + , + ) + fireEvent.click(screen.getByTestId("accordian-toggle")) + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + + expect(requestsOfType("readOriginalContent")).toHaveLength(1) + expect(screen.getByTestId("accordian-code")).toHaveTextContent("-old line") + }) + it("ignores an original answered for a different task", () => { renderPanel([createEditWithOriginal({ originalContentLength: 5000 })], "task-1") expandRow() diff --git a/webview-ui/src/components/chat/ChatView.tsx b/webview-ui/src/components/chat/ChatView.tsx index e75041c457..c4c4369ad1 100644 --- a/webview-ui/src/components/chat/ChatView.tsx +++ b/webview-ui/src/components/chat/ChatView.tsx @@ -1738,7 +1738,7 @@ const ChatViewComponent: React.ForwardRefRenderFunction
- + {areButtonsVisible && (
B -> A) send a duplicate while the first is still open. const pendingOriginalRequestsRef = useRef>(new Set()) - // Reset expanded file rows and final content cache when switching to a different task + // Task the expanded rows belong to; the reset below only lands after the render in which `taskId` changed, so + // the request effect must not act on rows expanded under the previous task. + const expandedTaskIdRef = useRef(taskId) + + // Reset expanded file rows and final content cache when the messages change useEffect(() => { setExpandedPaths(new Set()) setFinalContentByPath({}) pendingPathsRef.current = new Set() - setOriginalContentByKey({}) }, [clineMessages, taskId]) + // Originals are keyed by message id, which is stable across message updates, so they only reset per task + useEffect(() => { + setOriginalContentByKey({}) + }, [taskId]) + const fileChanges = useMemo(() => fileChangesFromMessages(clineMessages), [clineMessages]) // Group by path so we show one row per file (multiple edits to same file combined for display) @@ -66,17 +74,22 @@ const FileChangesPanel = memo(({ clineMessages, taskId, className }: FileChanges ) }, [fileChanges]) - const togglePath = useCallback((path: string) => { - setExpandedPaths((prev) => { - const next = new Set(prev) - if (next.has(path)) next.delete(path) - else next.add(path) - return next - }) - }, []) + const togglePath = useCallback( + (path: string) => { + expandedTaskIdRef.current = taskId + setExpandedPaths((prev) => { + const next = new Set(prev) + if (next.has(path)) next.delete(path) + else next.add(path) + return next + }) + }, + [taskId], + ) // Request the final file content (and the omitted original content) when a row is expanded and the edit has an original useEffect(() => { + if (expandedTaskIdRef.current !== taskId) return for (const path of expandedPaths) { const entries = byPath.get(path) if (!entries?.length) continue From 43159fe0283145eca93dc131b90c9be260af295b Mon Sep 17 00:00:00 2001 From: daewoongoh Date: Fri, 2 Oct 2026 16:28:14 +0900 Subject: [PATCH 3/3] fix(scripts): validate --build-mode and add tests for the gray-screen tools Only build into the fixed production/development temp directories (Vite empties the output directory), never into a symlink. Add subprocess and HTTP tests for generate-large-task, analyze-session and mock-openai-server, and make analyze-session skip tasks whose ui_messages.json cannot be parsed. --- scripts/gray-screen/__tests__/lib.test.mjs | 29 ++- scripts/gray-screen/__tests__/tools.test.mjs | 183 ++++++++++++++++++ scripts/gray-screen/analyze-session.mjs | 14 +- scripts/gray-screen/lib.mjs | 20 ++ scripts/gray-screen/webview-render-stress.mjs | 4 +- 5 files changed, 246 insertions(+), 4 deletions(-) create mode 100644 scripts/gray-screen/__tests__/tools.test.mjs diff --git a/scripts/gray-screen/__tests__/lib.test.mjs b/scripts/gray-screen/__tests__/lib.test.mjs index af47a20eab..868987d541 100644 --- a/scripts/gray-screen/__tests__/lib.test.mjs +++ b/scripts/gray-screen/__tests__/lib.test.mjs @@ -4,7 +4,14 @@ import os from "node:os" import path from "node:path" import { afterEach, beforeEach, describe, it } from "node:test" -import { integerFlag, parseFlagArgs, resolveServedFile, validateRelativeDir, writeTaskAtomically } from "../lib.mjs" +import { + integerFlag, + parseFlagArgs, + resolveBuildDir, + resolveServedFile, + validateRelativeDir, + writeTaskAtomically, +} from "../lib.mjs" let tmp @@ -46,6 +53,26 @@ describe("validateRelativeDir", () => { }) }) +describe("resolveBuildDir", () => { + it("maps the documented modes to fixed directories under the temp root", () => { + assert.equal(resolveBuildDir("production", tmp), path.join(tmp, "zoo-webview-stress-build")) + assert.equal(resolveBuildDir("development", tmp), path.join(tmp, "zoo-webview-stress-build-development")) + }) + + it("rejects any other mode, including path traversal", () => { + for (const mode of ["../../../tmp/victim", "staging", "", "production/../x"]) { + assert.throws(() => resolveBuildDir(mode, tmp), /Invalid --build-mode/, mode) + } + }) + + it("refuses an output path that is a symlink", () => { + fs.mkdirSync(path.join(tmp, "elsewhere")) + fs.symlinkSync(path.join(tmp, "elsewhere"), path.join(tmp, "zoo-webview-stress-build")) + + assert.throws(() => resolveBuildDir("production", tmp), /symlink/) + }) +}) + describe("resolveServedFile", () => { beforeEach(() => { fs.mkdirSync(path.join(tmp, "build", "assets"), { recursive: true }) diff --git a/scripts/gray-screen/__tests__/tools.test.mjs b/scripts/gray-screen/__tests__/tools.test.mjs new file mode 100644 index 0000000000..69031e5fa5 --- /dev/null +++ b/scripts/gray-screen/__tests__/tools.test.mjs @@ -0,0 +1,183 @@ +// Subprocess tests for the executable gray-screen tools. Run: node --test scripts/gray-screen/__tests__/tools.test.mjs +// The browser harnesses (webview-heap-matrix, webview-render-stress) need a built webview and Chromium, so only +// their pure parts (static serving, build directory) are covered, in lib.test.mjs. +import assert from "node:assert/strict" +import { spawn, spawnSync } from "node:child_process" +import fs from "node:fs" +import net from "node:net" +import os from "node:os" +import path from "node:path" +import { afterEach, beforeEach, describe, it } from "node:test" +import { fileURLToPath } from "node:url" + +const dir = path.dirname(fileURLToPath(import.meta.url)) +const script = (name) => path.join(dir, "..", name) +const node = (name, args) => spawnSync(process.execPath, [script(name), ...args], { encoding: "utf8" }) + +let tmp + +beforeEach(() => { + tmp = fs.mkdtempSync(path.join(os.tmpdir(), "gray-screen-tools-")) +}) + +afterEach(() => { + fs.rmSync(tmp, { recursive: true, force: true }) +}) + +const generate = (...args) => node("generate-large-task.mjs", ["--storage", tmp, ...args]) + +function readTask() { + const [id] = fs.readdirSync(path.join(tmp, "tasks")) + const read = (name) => JSON.parse(fs.readFileSync(path.join(tmp, "tasks", id, name), "utf8")) + return { id, messages: read("ui_messages.json"), api: read("api_conversation_history.json"), item: read("history_item.json") } +} + +describe("generate-large-task", () => { + it("writes a complete task with the requested number of messages", () => { + const result = generate("--messages", "40", "--text-bytes", "100") + + assert.equal(result.status, 0, result.stderr) + const { id, messages, item } = readTask() + assert.ok(messages.length >= 40) + assert.equal(item.id, id) + assert.match(item.task, /\[LOAD TEST\]/) + assert.deepEqual(fs.readdirSync(tmp).sort(), ["tasks"]) + }) + + it("adds images when asked to", () => { + assert.equal(generate("--messages", "30", "--text-bytes", "100", "--image-every", "5", "--image-kb", "1").status, 0) + + assert.ok(readTask().messages.some((m) => m.images?.length)) + }) + + it("produces non-Latin-1 text with --two-byte true only", () => { + assert.equal(generate("--messages", "30", "--text-bytes", "100", "--two-byte", "true").status, 0) + assert.match(JSON.stringify(readTask().messages), /한글/) + + fs.rmSync(path.join(tmp, "tasks"), { recursive: true }) + assert.equal(generate("--messages", "30", "--text-bytes", "100").status, 0) + assert.doesNotMatch(JSON.stringify(readTask().messages), /한글/) + }) + + it("creates nothing when a flag has no value or is not a positive integer", () => { + for (const args of [["--messages", "--text-bytes", "5"], ["--messages", "abc"], ["--messages", "0"], ["--image-kb", "-1"]]) { + const result = generate(...args) + + assert.notEqual(result.status, 0, args.join(" ")) + assert.deepEqual(fs.readdirSync(tmp), [], args.join(" ")) + } + }) +}) + +describe("analyze-session", () => { + const analyze = (...args) => node("analyze-session.mjs", ["--storage", tmp, ...args]) + + it("excludes generated [LOAD TEST] tasks unless asked, and reports sizes for the rest", () => { + assert.equal(generate("--messages", "40", "--text-bytes", "100").status, 0) + + const excluded = analyze() + assert.equal(excluded.status, 0, excluded.stderr) + assert.match(excluded.stdout, /excluded 1 generated/) + assert.match(excluded.stdout, /analyzed set: 0/) + + const included = analyze("--include-generated", "true") + assert.equal(included.status, 0, included.stderr) + assert.match(included.stdout, /analyzed set: 1/) + assert.match(included.stdout, /size\(MB\)\s+msgs\s+toolTxt/) + }) + + it("does not print message content", () => { + const taskDir = path.join(tmp, "tasks", "real-task") + fs.mkdirSync(taskDir, { recursive: true }) + fs.writeFileSync( + path.join(taskDir, "ui_messages.json"), + JSON.stringify([{ ts: 1, type: "say", say: "text", text: "TOP-SECRET-CONTENT" }]), + ) + + const result = analyze() + + assert.equal(result.status, 0, result.stderr) + assert.doesNotMatch(result.stdout, /TOP-SECRET-CONTENT/) + }) + + it("survives a malformed ui_messages.json", () => { + const taskDir = path.join(tmp, "tasks", "broken") + fs.mkdirSync(taskDir, { recursive: true }) + fs.writeFileSync(path.join(taskDir, "ui_messages.json"), "{not json") + + assert.equal(analyze().status, 0) + }) +}) + +describe("mock-openai-server", () => { + it("rejects an unsafe --dir before listening", () => { + for (const bad of ['x"; touch /tmp/pwned; #', ".", "../x", "-rf"]) { + const result = node("mock-openai-server.mjs", ["--dir", bad, "--port", "0"]) + + assert.equal(result.status, 1, bad) + assert.match(result.stderr, /Invalid --dir/) + } + }) + + describe("HTTP", () => { + let child + let base + + beforeEach(async () => { + const port = await new Promise((resolve) => { + const probe = net.createServer().listen(0, "127.0.0.1", () => { + const { port } = probe.address() + probe.close(() => resolve(port)) + }) + }) + base = `http://127.0.0.1:${port}` + child = spawn( + process.execPath, + [script("mock-openai-server.mjs"), "--scenario", "rapid", "--port", String(port), "--max-requests", "1"], + { stdio: ["ignore", "pipe", "inherit"] }, + ) + await new Promise((resolve, reject) => { + child.once("error", reject) + child.stdout.on("data", (chunk) => String(chunk).includes("listening") && resolve()) + }) + }) + + afterEach(() => child.kill()) + + const complete = (body) => + fetch(`${base}/v1/chat/completions`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify(body), + }) + + it("lists the mock model and 404s elsewhere", async () => { + const models = await (await fetch(`${base}/v1/models`)).json() + assert.deepEqual( + models.data.map((m) => m.id), + ["mock"], + ) + assert.equal((await fetch(`${base}/nope`)).status, 404) + }) + + it("answers a request without tools with plain JSON", async () => { + const body = await (await complete({ messages: [{ role: "user", content: "hi" }] })).json() + + assert.equal(body.object, "chat.completion") + assert.match(body.choices[0].message.content, /Summary/) + }) + + it("streams a scripted tool call, then attempt_completion once --max-requests is exceeded", async () => { + const request = { stream: true, messages: [{ role: "user", content: "go" }], tools: [{ type: "function", function: { name: "x" } }] } + + const first = await (await complete(request)).text() + assert.match(first, /"tool_calls"/) + assert.ok(first.trimEnd().endsWith("data: [DONE]")) + assert.doesNotMatch(first, /attempt_completion/) + + const second = await (await complete(request)).text() + assert.match(second, /attempt_completion/) + assert.ok(second.trimEnd().endsWith("data: [DONE]")) + }) + }) +}) diff --git a/scripts/gray-screen/analyze-session.mjs b/scripts/gray-screen/analyze-session.mjs index ee395590e8..bacd1c602d 100644 --- a/scripts/gray-screen/analyze-session.mjs +++ b/scripts/gray-screen/analyze-session.mjs @@ -101,7 +101,19 @@ console.log(`storage: ${storage}`) console.log(`tasks: ${ids.length} (excluded ${sizes.length - real.length} generated [LOAD TEST] tasks${includeGenerated ? " - included" : ""}); analyzed set: ${real.length}`) console.log(`ui_messages.json size: p50=${mb(pct(0.5))}MB p90=${mb(pct(0.9))}MB p99=${mb(pct(0.99))}MB max=${mb(sorted.at(-1) ?? 0)}MB; over 10MB: ${sorted.filter((s) => s > 10 * 1048576).length}, over 5MB: ${sorted.filter((s) => s > 5 * 1048576).length}`) -const results = real.sort((a, b) => b.size - a.size).slice(0, top).map((s) => analyze(s.id)) +let unreadable = 0 +const results = real + .sort((a, b) => b.size - a.size) + .slice(0, top) + .flatMap((s) => { + try { + return [analyze(s.id)] + } catch { + unreadable++ + return [] + } + }) +if (unreadable) console.log(`skipped ${unreadable} task(s) whose ui_messages.json could not be parsed`) console.log(`\nTop ${results.length} tasks by ui_messages.json size (sizes only, no content):`) console.log("size(MB) msgs toolTxt(MB) editTools topField(share) 2byte% imgMB maxBatchRun(bytes MB) task") for (const r of results) { diff --git a/scripts/gray-screen/lib.mjs b/scripts/gray-screen/lib.mjs index fb9c0be48d..25e58eb9aa 100644 --- a/scripts/gray-screen/lib.mjs +++ b/scripts/gray-screen/lib.mjs @@ -38,6 +38,26 @@ export function validateRelativeDir(dir) { return dir } +const BUILD_MODES = ["production", "development"] + +/** + * The temp directory a stress-build of the given Vite mode is written to (Vite empties it first, so it must never + * be derived from unchecked input). Refuses modes other than the documented ones and a symlinked output path. + */ +export function resolveBuildDir(mode, tmpRoot = "/tmp") { + if (!BUILD_MODES.includes(mode)) throw new Error(`Invalid --build-mode "${mode}": use ${BUILD_MODES.join(" or ")}.`) + + const dir = path.join(tmpRoot, "zoo-webview-stress-build" + (mode === "production" ? "" : `-${mode}`)) + + try { + if (fs.lstatSync(dir).isSymbolicLink()) throw new Error(`Refusing to build into a symlink: ${dir}`) + } catch (error) { + if (error?.code !== "ENOENT") throw error + } + + return dir +} + /** Resolves a request path to a regular file inside `root` (after symlink resolution), or undefined. */ export function resolveServedFile(root, rawPath) { try { diff --git a/scripts/gray-screen/webview-render-stress.mjs b/scripts/gray-screen/webview-render-stress.mjs index bd80ca52f6..38e0241a89 100644 --- a/scripts/gray-screen/webview-render-stress.mjs +++ b/scripts/gray-screen/webview-render-stress.mjs @@ -28,7 +28,7 @@ import http from "node:http" import { createRequire } from "node:module" import path from "node:path" import { fileURLToPath } from "node:url" -import { resolveServedFile } from "./lib.mjs" +import { resolveBuildDir, resolveServedFile } from "./lib.mjs" const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..") const webviewDir = path.join(root, "webview-ui") @@ -61,7 +61,7 @@ const cfg = { const sleep = (ms) => new Promise((r) => setTimeout(r, ms)) const buildMode = args["build-mode"] ?? "production" -const buildDir = "/tmp/zoo-webview-stress-build" + (buildMode === "production" ? "" : `-${buildMode}`) +const buildDir = resolveBuildDir(buildMode) function run(cmd, cmdArgs, opts) { return new Promise((resolve, reject) => {