diff --git a/packages/types/src/vscode-extension-host.ts b/packages/types/src/vscode-extension-host.ts index 5f6b579779..b9fdaf5b79 100644 --- a/packages/types/src/vscode-extension-host.ts +++ b/packages/types/src/vscode-extension-host.ts @@ -106,11 +106,14 @@ export interface ExtensionMessage { | "skills" | "rules" | "fileContent" + | "originalContent" | "rooHistoryImportProgress" | "themeFixtureProbeRequest" text?: string /** For fileContent: { path, content, error? } */ fileContent?: { path: string; content: string | null; error?: string } + /** For originalContent: the pre-edit file content of the requested tool message (null when unavailable) */ + originalContentInfo?: { ts: number; messageId?: string; taskId?: string; content: string | null } payload?: any // eslint-disable-line @typescript-eslint/no-explicit-any checkpointWarning?: { type: "WAIT_TIMEOUT" | "INIT_TIMEOUT" @@ -498,6 +501,7 @@ export interface WebviewMessage { | "saveImage" | "openFile" | "readFileContent" + | "readOriginalContent" | "openMention" | "cancelTask" | "cancelAutoApproval" @@ -696,6 +700,7 @@ export interface WebviewMessage { ids?: string[] terminalOperation?: "continue" | "abort" messageTs?: number + messageId?: string restoreCheckpoint?: boolean historyPreviewCollapsed?: boolean filters?: { type?: string; search?: string; tags?: string[] } @@ -858,6 +863,8 @@ export interface ClineSayTool { content?: string // Original file content before first edit (for merged diff display in FileChangesPanel) originalContent?: string + // Length of the originalContent the extension left out of the webview state (request it with readOriginalContent) + originalContentLength?: number // Unified diff statistics computed by the extension diffStats?: { added: number; removed: number } regex?: string diff --git a/scripts/gray-screen/__tests__/lib.test.mjs b/scripts/gray-screen/__tests__/lib.test.mjs new file mode 100644 index 0000000000..868987d541 --- /dev/null +++ b/scripts/gray-screen/__tests__/lib.test.mjs @@ -0,0 +1,131 @@ +import assert from "node:assert/strict" +import fs from "node:fs" +import os from "node:os" +import path from "node:path" +import { afterEach, beforeEach, describe, it } from "node:test" + +import { + integerFlag, + parseFlagArgs, + resolveBuildDir, + resolveServedFile, + validateRelativeDir, + writeTaskAtomically, +} from "../lib.mjs" + +let tmp + +beforeEach(() => { + tmp = fs.mkdtempSync(path.join(os.tmpdir(), "gray-screen-test-")) +}) + +afterEach(() => { + fs.rmSync(tmp, { recursive: true, force: true }) +}) + +describe("parseFlagArgs / integerFlag", () => { + it("parses flag/value pairs", () => { + assert.deepEqual(parseFlagArgs(["--messages", "10", "--two-byte", "true"]), { messages: "10", "two-byte": "true" }) + }) + + it("rejects a flag with a missing operand", () => { + assert.throws(() => parseFlagArgs(["--messages", "--text-bytes", "5"]), /Missing value for --messages/) + assert.throws(() => parseFlagArgs(["--messages"]), /Missing value for --messages/) + }) + + it("accepts integers at or above the minimum and rejects everything else", () => { + assert.equal(integerFlag("messages", "5000", 1), 5000) + assert.equal(integerFlag("image-every", "0", 0), 0) + for (const bad of ["0", "-3", "1.5", "abc", "", "NaN"]) { + assert.throws(() => integerFlag("messages", bad, 1), /--messages must be an integer >= 1/) + } + }) +}) + +describe("validateRelativeDir", () => { + it("accepts plain relative paths", () => { + for (const ok of [".mock-session", "a", "src/mock_1", "a.b/c-d"]) assert.equal(validateRelativeDir(ok), ok) + }) + + it("rejects shell metacharacters, absolute paths, traversal, dot and option-like segments", () => { + const bad = ['safe"; touch /tmp/pwned; #', "a b", "$(id)", "/abs", "../x", "a/../b", ".", "a/./b", "-rf", "a/-rf", "", "a//b"] + for (const dir of bad) assert.throws(() => validateRelativeDir(dir), /Invalid --dir/, JSON.stringify(dir)) + }) +}) + +describe("resolveBuildDir", () => { + it("maps the documented modes to fixed directories under the temp root", () => { + assert.equal(resolveBuildDir("production", tmp), path.join(tmp, "zoo-webview-stress-build")) + assert.equal(resolveBuildDir("development", tmp), path.join(tmp, "zoo-webview-stress-build-development")) + }) + + it("rejects any other mode, including path traversal", () => { + for (const mode of ["../../../tmp/victim", "staging", "", "production/../x"]) { + assert.throws(() => resolveBuildDir(mode, tmp), /Invalid --build-mode/, mode) + } + }) + + it("refuses an output path that is a symlink", () => { + fs.mkdirSync(path.join(tmp, "elsewhere")) + fs.symlinkSync(path.join(tmp, "elsewhere"), path.join(tmp, "zoo-webview-stress-build")) + + assert.throws(() => resolveBuildDir("production", tmp), /symlink/) + }) +}) + +describe("resolveServedFile", () => { + beforeEach(() => { + fs.mkdirSync(path.join(tmp, "build", "assets"), { recursive: true }) + fs.mkdirSync(path.join(tmp, "build-secret")) + fs.writeFileSync(path.join(tmp, "build", "index.html"), "index") + fs.writeFileSync(path.join(tmp, "build", "assets", "app.js"), "js") + fs.writeFileSync(path.join(tmp, "build-secret", "credentials.json"), "secret") + fs.symlinkSync(path.join(tmp, "build-secret"), path.join(tmp, "build", "link")) + }) + + const root = () => path.join(tmp, "build") + + it("serves files inside the build directory, with / mapping to index.html", () => { + assert.equal(resolveServedFile(root(), "/"), fs.realpathSync(path.join(root(), "index.html"))) + assert.equal(resolveServedFile(root(), "/assets/app.js"), fs.realpathSync(path.join(root(), "assets", "app.js"))) + }) + + it("returns undefined for traversal (plain, encoded and sibling-prefix), symlink escapes, directories and bad input", () => { + for (const p of [ + "/../build-secret/credentials.json", + "/%2e%2e%2fbuild-secret%2fcredentials.json", + "/link/credentials.json", + "/assets", + "/missing.js", + "/%E0%A4%A", + "/a\0b", + ]) { + assert.equal(resolveServedFile(root(), p), undefined, p) + } + }) + + it("returns undefined when the build directory does not exist", () => { + assert.equal(resolveServedFile(path.join(tmp, "nope"), "/"), undefined) + }) +}) + +describe("writeTaskAtomically", () => { + it("writes every file into tasks/ and leaves no staging directory", () => { + const taskDir = writeTaskAtomically(tmp, "t1", { "a.json": { a: 1 }, "b.json": [2] }) + + assert.equal(taskDir, path.join(tmp, "tasks", "t1")) + assert.deepEqual(JSON.parse(fs.readFileSync(path.join(taskDir, "a.json"), "utf8")), { a: 1 }) + assert.deepEqual(JSON.parse(fs.readFileSync(path.join(taskDir, "b.json"), "utf8")), [2]) + assert.deepEqual(fs.readdirSync(tmp).sort(), ["tasks"]) + }) + + it("leaves neither a task directory nor staging files when a write fails", () => { + const circular = {} + circular.self = circular + + assert.throws(() => writeTaskAtomically(tmp, "t2", { "ok.json": { ok: true }, "bad.json": circular }), /circular/i) + + assert.equal(fs.existsSync(path.join(tmp, "tasks", "t2")), false) + assert.deepEqual(fs.readdirSync(tmp), []) + }) +}) diff --git a/scripts/gray-screen/__tests__/tools.test.mjs b/scripts/gray-screen/__tests__/tools.test.mjs new file mode 100644 index 0000000000..69031e5fa5 --- /dev/null +++ b/scripts/gray-screen/__tests__/tools.test.mjs @@ -0,0 +1,183 @@ +// Subprocess tests for the executable gray-screen tools. Run: node --test scripts/gray-screen/__tests__/tools.test.mjs +// The browser harnesses (webview-heap-matrix, webview-render-stress) need a built webview and Chromium, so only +// their pure parts (static serving, build directory) are covered, in lib.test.mjs. +import assert from "node:assert/strict" +import { spawn, spawnSync } from "node:child_process" +import fs from "node:fs" +import net from "node:net" +import os from "node:os" +import path from "node:path" +import { afterEach, beforeEach, describe, it } from "node:test" +import { fileURLToPath } from "node:url" + +const dir = path.dirname(fileURLToPath(import.meta.url)) +const script = (name) => path.join(dir, "..", name) +const node = (name, args) => spawnSync(process.execPath, [script(name), ...args], { encoding: "utf8" }) + +let tmp + +beforeEach(() => { + tmp = fs.mkdtempSync(path.join(os.tmpdir(), "gray-screen-tools-")) +}) + +afterEach(() => { + fs.rmSync(tmp, { recursive: true, force: true }) +}) + +const generate = (...args) => node("generate-large-task.mjs", ["--storage", tmp, ...args]) + +function readTask() { + const [id] = fs.readdirSync(path.join(tmp, "tasks")) + const read = (name) => JSON.parse(fs.readFileSync(path.join(tmp, "tasks", id, name), "utf8")) + return { id, messages: read("ui_messages.json"), api: read("api_conversation_history.json"), item: read("history_item.json") } +} + +describe("generate-large-task", () => { + it("writes a complete task with the requested number of messages", () => { + const result = generate("--messages", "40", "--text-bytes", "100") + + assert.equal(result.status, 0, result.stderr) + const { id, messages, item } = readTask() + assert.ok(messages.length >= 40) + assert.equal(item.id, id) + assert.match(item.task, /\[LOAD TEST\]/) + assert.deepEqual(fs.readdirSync(tmp).sort(), ["tasks"]) + }) + + it("adds images when asked to", () => { + assert.equal(generate("--messages", "30", "--text-bytes", "100", "--image-every", "5", "--image-kb", "1").status, 0) + + assert.ok(readTask().messages.some((m) => m.images?.length)) + }) + + it("produces non-Latin-1 text with --two-byte true only", () => { + assert.equal(generate("--messages", "30", "--text-bytes", "100", "--two-byte", "true").status, 0) + assert.match(JSON.stringify(readTask().messages), /한글/) + + fs.rmSync(path.join(tmp, "tasks"), { recursive: true }) + assert.equal(generate("--messages", "30", "--text-bytes", "100").status, 0) + assert.doesNotMatch(JSON.stringify(readTask().messages), /한글/) + }) + + it("creates nothing when a flag has no value or is not a positive integer", () => { + for (const args of [["--messages", "--text-bytes", "5"], ["--messages", "abc"], ["--messages", "0"], ["--image-kb", "-1"]]) { + const result = generate(...args) + + assert.notEqual(result.status, 0, args.join(" ")) + assert.deepEqual(fs.readdirSync(tmp), [], args.join(" ")) + } + }) +}) + +describe("analyze-session", () => { + const analyze = (...args) => node("analyze-session.mjs", ["--storage", tmp, ...args]) + + it("excludes generated [LOAD TEST] tasks unless asked, and reports sizes for the rest", () => { + assert.equal(generate("--messages", "40", "--text-bytes", "100").status, 0) + + const excluded = analyze() + assert.equal(excluded.status, 0, excluded.stderr) + assert.match(excluded.stdout, /excluded 1 generated/) + assert.match(excluded.stdout, /analyzed set: 0/) + + const included = analyze("--include-generated", "true") + assert.equal(included.status, 0, included.stderr) + assert.match(included.stdout, /analyzed set: 1/) + assert.match(included.stdout, /size\(MB\)\s+msgs\s+toolTxt/) + }) + + it("does not print message content", () => { + const taskDir = path.join(tmp, "tasks", "real-task") + fs.mkdirSync(taskDir, { recursive: true }) + fs.writeFileSync( + path.join(taskDir, "ui_messages.json"), + JSON.stringify([{ ts: 1, type: "say", say: "text", text: "TOP-SECRET-CONTENT" }]), + ) + + const result = analyze() + + assert.equal(result.status, 0, result.stderr) + assert.doesNotMatch(result.stdout, /TOP-SECRET-CONTENT/) + }) + + it("survives a malformed ui_messages.json", () => { + const taskDir = path.join(tmp, "tasks", "broken") + fs.mkdirSync(taskDir, { recursive: true }) + fs.writeFileSync(path.join(taskDir, "ui_messages.json"), "{not json") + + assert.equal(analyze().status, 0) + }) +}) + +describe("mock-openai-server", () => { + it("rejects an unsafe --dir before listening", () => { + for (const bad of ['x"; touch /tmp/pwned; #', ".", "../x", "-rf"]) { + const result = node("mock-openai-server.mjs", ["--dir", bad, "--port", "0"]) + + assert.equal(result.status, 1, bad) + assert.match(result.stderr, /Invalid --dir/) + } + }) + + describe("HTTP", () => { + let child + let base + + beforeEach(async () => { + const port = await new Promise((resolve) => { + const probe = net.createServer().listen(0, "127.0.0.1", () => { + const { port } = probe.address() + probe.close(() => resolve(port)) + }) + }) + base = `http://127.0.0.1:${port}` + child = spawn( + process.execPath, + [script("mock-openai-server.mjs"), "--scenario", "rapid", "--port", String(port), "--max-requests", "1"], + { stdio: ["ignore", "pipe", "inherit"] }, + ) + await new Promise((resolve, reject) => { + child.once("error", reject) + child.stdout.on("data", (chunk) => String(chunk).includes("listening") && resolve()) + }) + }) + + afterEach(() => child.kill()) + + const complete = (body) => + fetch(`${base}/v1/chat/completions`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify(body), + }) + + it("lists the mock model and 404s elsewhere", async () => { + const models = await (await fetch(`${base}/v1/models`)).json() + assert.deepEqual( + models.data.map((m) => m.id), + ["mock"], + ) + assert.equal((await fetch(`${base}/nope`)).status, 404) + }) + + it("answers a request without tools with plain JSON", async () => { + const body = await (await complete({ messages: [{ role: "user", content: "hi" }] })).json() + + assert.equal(body.object, "chat.completion") + assert.match(body.choices[0].message.content, /Summary/) + }) + + it("streams a scripted tool call, then attempt_completion once --max-requests is exceeded", async () => { + const request = { stream: true, messages: [{ role: "user", content: "go" }], tools: [{ type: "function", function: { name: "x" } }] } + + const first = await (await complete(request)).text() + assert.match(first, /"tool_calls"/) + assert.ok(first.trimEnd().endsWith("data: [DONE]")) + assert.doesNotMatch(first, /attempt_completion/) + + const second = await (await complete(request)).text() + assert.match(second, /attempt_completion/) + assert.ok(second.trimEnd().endsWith("data: [DONE]")) + }) + }) +}) diff --git a/scripts/gray-screen/analyze-session.mjs b/scripts/gray-screen/analyze-session.mjs new file mode 100644 index 0000000000..bacd1c602d --- /dev/null +++ b/scripts/gray-screen/analyze-session.mjs @@ -0,0 +1,141 @@ +#!/usr/bin/env node +// Measures what a saved Zoo Code task is made of, to see what the webview has to hold: sizes only, never prints message content. +// +// node scripts/gray-screen/analyze-session.mjs [--storage ] [--top 15] [--task ] [--include-generated true] +// +// Per task: file size, message counts, bytes per message kind and per tool, bytes per JSON field of tool payloads +// (originalContent / content / diff ...), share of strings V8 stores as 2 bytes/char (any char above U+00FF), and the +// longest run of consecutive file-edit asks that ChatView's batchNearby would merge into one message. +import fs from "node:fs" +import os from "node:os" +import path from "node:path" + +const args = Object.fromEntries( + process.argv.slice(2).reduce((acc, cur, i, all) => { + if (cur.startsWith("--")) acc.push([cur.slice(2), all[i + 1]]) + return acc + }, []), +) +const storage = + args["storage"] ?? path.join(os.homedir(), ".vscode-server", "data", "User", "globalStorage", "codemate.zoo-code") +const top = Number(args["top"] ?? 15) +const includeGenerated = args["include-generated"] === "true" +const tasksDir = path.join(storage, "tasks") + +const EDIT_TOOLS = new Set(["editedExistingFile", "appliedDiff", "newFileCreated", "insertContent", "searchAndReplace"]) +const BOUNDARY_SAY = new Set(["user_feedback", "user_feedback_diff", "completion_result", "checkpoint_saved", "error", "condense_context", "codebase_search_result"]) +const NON_LATIN1 = /[^\u0000-ÿ]/ +const mb = (n) => (n / 1048576).toFixed(1) + +const isIgnorable = (m) => m.type === "say" && (m.say === "api_req_started" || (m.say === "text" && !m.text?.trim()) || m.say === "reasoning") +const isBoundary = (m) => m.type === "say" && (BOUNDARY_SAY.has(m.say) || (m.say === "text" && !!m.text?.trim())) + +function analyze(taskId) { + const file = path.join(tasksDir, taskId, "ui_messages.json") + const raw = fs.readFileSync(file, "utf8") + const messages = JSON.parse(raw) + const r = { taskId, fileBytes: Buffer.byteLength(raw), count: messages.length, kinds: {}, tools: {}, fields: {}, imageBytes: 0, twoByteBytes: 0, latin1Bytes: 0, maxRun: 0, maxRunBytes: 0, runs2plus: 0 } + + const add = (obj, key, bytes) => { + const e = (obj[key] ??= { n: 0, bytes: 0 }) + e.n++ + e.bytes += bytes + } + const classify = (str) => { + if (NON_LATIN1.test(str)) r.twoByteBytes += str.length * 2 + else r.latin1Bytes += str.length + } + + for (const m of messages) { + const text = typeof m.text === "string" ? m.text : "" + add(r.kinds, `${m.type}:${m.say ?? m.ask ?? "?"}`, text.length) + if (text) classify(text) + if (Array.isArray(m.images)) for (const img of m.images) r.imageBytes += typeof img === "string" ? img.length : 0 + if (m.type === "ask" && m.ask === "tool" && text) { + try { + const t = JSON.parse(text) + add(r.tools, String(t.tool), text.length) + for (const [k, v] of Object.entries(t)) add(r.fields, k, typeof v === "string" ? v.length : JSON.stringify(v)?.length ?? 0) + } catch {} + } + } + + // longest run of consecutive edit asks that batchNearby merges (same rules as batchNearby.ts) + const isEditAsk = (m) => { + if (m.type !== "ask" || m.ask !== "tool" || !m.text) return false + try { + const t = JSON.parse(m.text) + return EDIT_TOOLS.has(t.tool) && !t.batchDiffs + } catch { + return false + } + } + const items = messages.slice(1) + for (let i = 0; i < items.length; ) { + if (isBoundary(items[i])) { i++; continue } + if (!isEditAsk(items[i])) { i++; continue } + let j = i + 1, len = 1, bytes = items[i].text.length + while (j < items.length) { + if (isBoundary(items[j])) break + if (isEditAsk(items[j])) { len++; bytes += items[j].text.length; j++ } + else if (isIgnorable(items[j])) j++ + else break + } + if (len > 1) r.runs2plus++ + if (len > r.maxRun) { r.maxRun = len; r.maxRunBytes = bytes } + i = j + } + return r +} + +const taskMeta = (id) => { + try { return JSON.parse(fs.readFileSync(path.join(tasksDir, id, "history_item.json"), "utf8")) } catch { return {} } +} +const ids = args["task"] ? [args["task"]] : fs.readdirSync(tasksDir).filter((id) => fs.existsSync(path.join(tasksDir, id, "ui_messages.json"))) +const sizes = ids.map((id) => ({ id, size: fs.statSync(path.join(tasksDir, id, "ui_messages.json")).size, generated: (taskMeta(id).task ?? "").startsWith("[LOAD TEST]") })) +const real = sizes.filter((s) => includeGenerated || !s.generated) + +const sorted = real.map((s) => s.size).sort((a, b) => a - b) +const pct = (p) => sorted[Math.min(sorted.length - 1, Math.floor(sorted.length * p))] ?? 0 +console.log(`storage: ${storage}`) +console.log(`tasks: ${ids.length} (excluded ${sizes.length - real.length} generated [LOAD TEST] tasks${includeGenerated ? " - included" : ""}); analyzed set: ${real.length}`) +console.log(`ui_messages.json size: p50=${mb(pct(0.5))}MB p90=${mb(pct(0.9))}MB p99=${mb(pct(0.99))}MB max=${mb(sorted.at(-1) ?? 0)}MB; over 10MB: ${sorted.filter((s) => s > 10 * 1048576).length}, over 5MB: ${sorted.filter((s) => s > 5 * 1048576).length}`) + +let unreadable = 0 +const results = real + .sort((a, b) => b.size - a.size) + .slice(0, top) + .flatMap((s) => { + try { + return [analyze(s.id)] + } catch { + unreadable++ + return [] + } + }) +if (unreadable) console.log(`skipped ${unreadable} task(s) whose ui_messages.json could not be parsed`) +console.log(`\nTop ${results.length} tasks by ui_messages.json size (sizes only, no content):`) +console.log("size(MB) msgs toolTxt(MB) editTools topField(share) 2byte% imgMB maxBatchRun(bytes MB) task") +for (const r of results) { + const toolBytes = Object.values(r.tools).reduce((a, b) => a + b.bytes, 0) + const editCount = Object.entries(r.tools).filter(([k]) => EDIT_TOOLS.has(k)).reduce((a, [, v]) => a + v.n, 0) + const fieldTotal = Object.values(r.fields).reduce((a, b) => a + b.bytes, 0) || 1 + const [topKey, topVal] = Object.entries(r.fields).sort((a, b) => b[1].bytes - a[1].bytes)[0] ?? ["-", { bytes: 0 }] + const text = r.latin1Bytes + r.twoByteBytes || 1 + const meta = taskMeta(r.taskId) + console.log( + `${mb(r.fileBytes).padStart(7)} ${String(r.count).padStart(5)} ${mb(toolBytes).padStart(9)} ${String(editCount).padStart(9)} ${`${topKey} ${(100 * topVal.bytes / fieldTotal).toFixed(0)}%`.padEnd(30)} ${(100 * r.twoByteBytes / text).toFixed(0).padStart(5)}% ${mb(r.imageBytes).padStart(5)} ${`${r.maxRun} (${mb(r.maxRunBytes)})`.padEnd(20)} ${(meta.task ?? "").slice(0, 28).replace(/\s+/g, " ")} [${r.taskId.slice(0, 8)}]`, + ) +} + +const biggest = results[0] +if (biggest) { + console.log(`\nBreakdown of the largest analyzed task [${biggest.taskId.slice(0, 8)}], ${mb(biggest.fileBytes)}MB, ${biggest.count} messages:`) + const show = (title, obj, n = 8) => { + console.log(` ${title}`) + for (const [k, v] of Object.entries(obj).sort((a, b) => b[1].bytes - a[1].bytes).slice(0, n)) console.log(` ${k.padEnd(28)} n=${String(v.n).padStart(5)} ${mb(v.bytes).padStart(7)}MB`) + } + show("by message kind (text bytes):", biggest.kinds) + show("by tool (payload bytes):", biggest.tools) + show("by tool payload field (string chars):", biggest.fields) +} diff --git a/scripts/gray-screen/generate-large-task.mjs b/scripts/gray-screen/generate-large-task.mjs new file mode 100644 index 0000000000..c2db6330f5 --- /dev/null +++ b/scripts/gray-screen/generate-large-task.mjs @@ -0,0 +1,111 @@ +#!/usr/bin/env node +// Generates a synthetic task with thousands of messages to reproduce webview +// memory pressure (gray screen / OOM). Usage: +// node scripts/gray-screen/generate-large-task.mjs [--messages 5000] [--text-bytes 800] [--tool-bytes ] +// [--tool-kind mixed|newFileCreated|appliedDiff|readFile|listFilesRecursive] [--batchable true] [--two-byte true] +// [--image-every 0] [--image-kb 200] [--storage ] [--workspace ] +// Then reload VS Code and open the task named "[LOAD TEST] ..." from history. +import fs from "node:fs" +import os from "node:os" +import path from "node:path" +import crypto from "node:crypto" +import { integerFlag, parseFlagArgs, writeTaskAtomically } from "./lib.mjs" + +const args = parseFlagArgs(process.argv.slice(2)) + +// Validate every numeric flag up front, before anything is written. +const messageCount = integerFlag("messages", args["messages"] ?? 5000, 1) +const textBytes = integerFlag("text-bytes", args["text-bytes"] ?? 800, 1) +const toolBytes = integerFlag("tool-bytes", args["tool-bytes"] ?? textBytes, 1) +const toolKind = args["tool-kind"] ?? "mixed" +const batchable = args["batchable"] === "true" +const twoByte = args["two-byte"] === "true" // Korean chars in tool text -> UTF-16 strings (2 bytes/char) in V8 +const imageEvery = integerFlag("image-every", args["image-every"] ?? 0, 0) +const imageKb = integerFlag("image-kb", args["image-kb"] ?? 200, 1) +const workspace = args["workspace"] ?? process.cwd() +const storageCandidates = [ + path.join(os.homedir(), ".vscode-server", "data", "User", "globalStorage", "codemate.zoo-code"), + path.join(os.homedir(), ".config", "Code", "User", "globalStorage", "codemate.zoo-code"), +] +const storage = args["storage"] ?? storageCandidates.find((p) => fs.existsSync(p)) ?? storageCandidates[1] + +const taskId = crypto.randomUUID() + +const filler = (n, seed) => { + const base = `line ${seed}: ${twoByte ? "\uD55C\uAE00 " : ""}The quick brown fox jumps over the lazy dog. ` + return base.repeat(Math.ceil(n / base.length)).slice(0, n) +} +const fakeImage = (kb) => `data:image/png;base64,${crypto.randomBytes(kb * 768).toString("base64")}` + +// Large tool payloads (new files, diffs) are what the webview re-parses on every streamed chunk. +const KINDS = ["newFileCreated", "appliedDiff", "readFile", "listFilesRecursive"] +const toolPayload = (round) => { + const kind = toolKind === "mixed" ? KINDS[round % KINDS.length] : toolKind + const file = `src/gen/file-${round}.ts` + switch (kind) { + case "newFileCreated": + return { tool: "newFileCreated", path: file, content: filler(toolBytes, round) } + case "appliedDiff": + return { tool: "appliedDiff", path: file, diff: filler(toolBytes, round), diffStats: { added: 12, removed: 4 } } + case "listFilesRecursive": + return { tool: "listFilesRecursive", path: "src", content: filler(Math.min(toolBytes, 2000), round) } + default: + return { tool: "readFile", path: file, content: filler(Math.min(toolBytes, 400), round) } + } +} + +const taskText = `[LOAD TEST] ${messageCount} messages` +const start = Date.now() - messageCount * 1000 +const messages = [{ ts: start, type: "say", say: "text", text: taskText }] +const apiHistory = [{ role: "user", content: [{ type: "text", text: `\n${taskText}\n` }], ts: start }] + +let ts = start + 1 +let round = 0 +while (messages.length < messageCount) { + round++ + const images = imageEvery > 0 && round % imageEvery === 0 ? [fakeImage(imageKb)] : undefined + messages.push({ + ts: ts++, + type: "say", + say: "api_req_started", + text: JSON.stringify({ apiProtocol: "openai", tokensIn: 1000, tokensOut: 200, cost: 0.001 }), + }) + // batchable: tool-only turns (no visible text/feedback between edits), which ChatView merges into one giant batch message + if (!batchable) messages.push({ ts: ts++, type: "say", say: "text", text: filler(textBytes, round), images }) + messages.push({ + ts: ts++, + type: "ask", + ask: "tool", + text: JSON.stringify(toolPayload(round)), + isAnswered: true, + }) + if (!batchable) messages.push({ ts: ts++, type: "say", say: "user_feedback", text: `ok ${round}` }) + apiHistory.push( + { role: "assistant", content: [{ type: "text", text: filler(textBytes, round) }], ts }, + { role: "user", content: [{ type: "text", text: `ok ${round}` }], ts }, + ) +} +messages.length = messageCount + +const historyItem = { + id: taskId, + number: 9999, + ts: Date.now(), + task: taskText, + tokensIn: round * 1000, + tokensOut: round * 200, + totalCost: 0, + workspace, + status: "completed", +} + +// The task becomes visible under tasks/ only once all three files are written. +const taskDir = writeTaskAtomically(storage, taskId, { + "ui_messages.json": messages, + "api_conversation_history.json": apiHistory, + "history_item.json": historyItem, +}) + +const mb = (f) => (fs.statSync(path.join(taskDir, f)).size / 1048576).toFixed(1) +console.log(`Task ${taskId}: ${messages.length} messages, ui_messages.json ${mb("ui_messages.json")} MB`) +console.log(`Written to ${taskDir}\nReload VS Code, then open "${taskText}" from history.`) diff --git a/scripts/gray-screen/lib.mjs b/scripts/gray-screen/lib.mjs new file mode 100644 index 0000000000..25e58eb9aa --- /dev/null +++ b/scripts/gray-screen/lib.mjs @@ -0,0 +1,97 @@ +// Shared helpers for the gray-screen tooling. Tests: node --test scripts/gray-screen/__tests__/lib.test.mjs +import fs from "node:fs" +import path from "node:path" + +/** `--flag value` pairs; a flag without a value (or followed by another flag) is an error. */ +export function parseFlagArgs(argv) { + const entries = [] + + argv.forEach((cur, i) => { + if (!cur.startsWith("--")) return + const value = argv[i + 1] + if (value === undefined || value.startsWith("--")) throw new Error(`Missing value for ${cur}`) + entries.push([cur.slice(2), value]) + }) + + return Object.fromEntries(entries) +} + +export function integerFlag(name, value, min) { + const number = Number(value) + if (!Number.isInteger(number) || number < min) throw new Error(`--${name} must be an integer >= ${min}`) + return number +} + +/** A plain relative path that is safe to interpolate into shell commands and to resolve under a workspace. */ +export function validateRelativeDir(dir) { + const segments = dir.split("/") + const isSafe = + /^[A-Za-z0-9._-]+(\/[A-Za-z0-9._-]+)*$/.test(dir) && + segments.every((part) => part !== "." && part !== ".." && !part.startsWith("-")) + + if (!isSafe) { + throw new Error( + `Invalid --dir "${dir}": use a relative path of letters, digits, ".", "_" and "-" without "." or ".." segments and without segments starting with "-".`, + ) + } + + return dir +} + +const BUILD_MODES = ["production", "development"] + +/** + * The temp directory a stress-build of the given Vite mode is written to (Vite empties it first, so it must never + * be derived from unchecked input). Refuses modes other than the documented ones and a symlinked output path. + */ +export function resolveBuildDir(mode, tmpRoot = "/tmp") { + if (!BUILD_MODES.includes(mode)) throw new Error(`Invalid --build-mode "${mode}": use ${BUILD_MODES.join(" or ")}.`) + + const dir = path.join(tmpRoot, "zoo-webview-stress-build" + (mode === "production" ? "" : `-${mode}`)) + + try { + if (fs.lstatSync(dir).isSymbolicLink()) throw new Error(`Refusing to build into a symlink: ${dir}`) + } catch (error) { + if (error?.code !== "ENOENT") throw error + } + + return dir +} + +/** Resolves a request path to a regular file inside `root` (after symlink resolution), or undefined. */ +export function resolveServedFile(root, rawPath) { + try { + const realRoot = fs.realpathSync(root) + const decoded = decodeURIComponent(rawPath) + const real = fs.realpathSync(path.resolve(realRoot, "." + (decoded === "/" ? "/index.html" : decoded))) + const rel = path.relative(realRoot, real) + if (rel === "" || rel === ".." || rel.startsWith(".." + path.sep) || path.isAbsolute(rel)) return undefined + return fs.statSync(real).isFile() ? real : undefined + } catch { + return undefined + } +} + +/** + * Writes a task into `/tasks/` all-or-nothing: the files are written into a staging directory + * next to `tasks/`, which is renamed into place only after every write succeeded. + * `files` maps a file name to the value to serialize. Returns the final task directory. + */ +export function writeTaskAtomically(storage, taskId, files) { + const staging = path.join(storage, `.task-staging-${taskId}`) + const taskDir = path.join(storage, "tasks", taskId) + + try { + fs.mkdirSync(staging, { recursive: true }) + for (const [name, value] of Object.entries(files)) { + fs.writeFileSync(path.join(staging, name), JSON.stringify(value)) + } + fs.mkdirSync(path.dirname(taskDir), { recursive: true }) + fs.renameSync(staging, taskDir) + } catch (error) { + fs.rmSync(staging, { recursive: true, force: true }) + throw error + } + + return taskDir +} diff --git a/scripts/gray-screen/mock-openai-server.mjs b/scripts/gray-screen/mock-openai-server.mjs new file mode 100644 index 0000000000..8a887c930f --- /dev/null +++ b/scripts/gray-screen/mock-openai-server.mjs @@ -0,0 +1,419 @@ +#!/usr/bin/env node +// LLM-free OpenAI-compatible server for reproducing Zoo Code sessions in real VS Code (gray screen / OOM work). +// Every request gets a scripted reply: optional streamed reasoning, a markdown explanation (headings, lists, tables, +// fenced code, optional mermaid) and ONE native tool call. After --max-requests turns it ends with attempt_completion. +// +// Usage (run on the machine hosting the extension host; sv01 for Remote-SSH): +// node scripts/gray-screen/mock-openai-server.mjs [--scenario rapid] [--port 8989] [--max-requests 500] [options] +// +// Scenarios (--scenario): +// rapid Turns as fast as possible: ~300B text, no reasoning, no delays, tiny update_todo_list call. +// Each new message makes the host push the full state to the webview. Use with a pre-generated big task +// (generate-large-task.mjs ... --batchable true) and Resume. This is the gray-screen reproduction. +// md ~15KB markdown (+ reasoning) per turn with a tiny update_todo_list call (isolates markdown/streaming cost). +// flood Like md but ~60KB in 4-16 char chunks every ~1ms (maximum partial-message updates). +// coding Realistic coding loop in a sandbox dir: update_todo_list, write_to_file, read_file, apply_diff, +// search_files, test-runner execute_command, list_files (real files/commands run; see --dir). +// chatty Only execute_command with ANSI-colored streaming output (no text). +// churn Phase 1 (--accumulate-turns) piles up large write_to_file/read_file/apply_diff messages, phase 2 is md. +// --fast true: 100 turns of ~55KB write_to_file, 4ms chunks. +// +// Options (defaults vary per scenario): +// --text-bytes N markdown size per turn (3000; md/churn 15000; flood 60000; rapid 300) +// --reasoning-bytes N streamed reasoning_content per turn (600; flood 3000; rapid 0) +// --chunk-ms N delay between streamed chunks (15; flood 1; rapid 0; --fast 4) +// --chunk-min/--chunk-max N characters per streamed chunk (8/47; flood 4/16) +// --tool-chunk N characters per tool-call argument chunk (120) +// --code-lines N lines per generated source file (150; churn 500; --fast 1500) +// --cmd-lines N lines printed by the simulated test/build runner (400) +// --cmd-sleep S seconds between runner lines (0.005) +// --mermaid-every N mermaid diagram every N turns (10; rapid 0; 0 disables) +// --dir PATH sandbox folder relative to the workspace (.mock-session); delete it afterwards +// +// Zoo Code settings: provider "OpenAI Compatible", base URL http://127.0.0.1:/v1, any API key, model id "mock", +// context window 1000000 (avoid condensing); auto-approve read/write/execute with allowed command "*". +import http from "node:http" +import { validateRelativeDir } from "./lib.mjs" + +const args = Object.fromEntries( + process.argv.slice(2).reduce((acc, cur, i, all) => { + if (cur.startsWith("--")) acc.push([cur.slice(2), all[i + 1]]) + return acc + }, []), +) +const fast = args["fast"] === "true" +const cfg = { + fast, + port: Number(args["port"] ?? 8989), + maxRequests: Number(args["max-requests"] ?? 500), + scenario: args["scenario"] ?? "coding", + textBytes: Number(args["text-bytes"] ?? (args["scenario"] === "rapid" ? 300 : args["scenario"] === "flood" ? 60000 : args["scenario"] === "md" || args["scenario"] === "churn" ? 15000 : 3000)), + reasoningBytes: Number(args["reasoning-bytes"] ?? (args["scenario"] === "rapid" ? 0 : args["scenario"] === "flood" ? 3000 : 600)), + chunkMs: Number(args["chunk-ms"] ?? (args["scenario"] === "rapid" ? 0 : args["scenario"] === "flood" ? 1 : fast ? 4 : 15)), + chunkMin: Number(args["chunk-min"] ?? (args["scenario"] === "flood" ? 4 : 8)), + chunkMax: Number(args["chunk-max"] ?? (args["scenario"] === "flood" ? 16 : 47)), + toolChunk: Number(args["tool-chunk"] ?? 120), + codeLines: Number(args["code-lines"] ?? (fast ? 1500 : args["scenario"] === "churn" ? 500 : 150)), + accumulateTurns: Number(args["accumulate-turns"] ?? (fast ? 100 : 300)), + cmdLines: Number(args["cmd-lines"] ?? 400), + cmdSleep: Number(args["cmd-sleep"] ?? 0.005), + mermaidEvery: Number(args["mermaid-every"] ?? (args["scenario"] === "rapid" ? 0 : 10)), + dir: (args["dir"] ?? ".mock-session").replace(/\/+$/, ""), +} + +// cfg.dir is interpolated into the shell commands the mock asks the extension to run, so only plain relative paths are allowed. +try { + validateRelativeDir(cfg.dir) +} catch (error) { + console.error(error.message) + process.exit(1) +} + +const sleep = (ms) => new Promise((r) => setTimeout(r, ms)) +let requestCount = 0 + +// --------------------------------------------------------------------------------------------- +// Deterministic pseudo-random helpers (same turn number -> same content) +// --------------------------------------------------------------------------------------------- +const rng = (seed) => { + let s = (seed * 2654435761) >>> 0 + return () => { + s = (s + 0x6d2b79f5) >>> 0 + let t = s + t = Math.imul(t ^ (t >>> 15), t | 1) + t ^= t + Math.imul(t ^ (t >>> 7), t | 61) + return ((t ^ (t >>> 14)) >>> 0) / 4294967296 + } +} +const pick = (r, arr) => arr[Math.floor(r() * arr.length)] + +const NOUNS = ["cache", "scheduler", "parser", "router", "session", "queue", "tokenizer", "pipeline", "registry", "validator", "watcher", "emitter"] +const VERBS = ["normalize", "resolve", "dispatch", "compress", "merge", "retry", "flush", "hydrate", "serialize", "throttle"] +const feature = (k) => `${NOUNS[k % NOUNS.length]}_${Math.floor(k / NOUNS.length)}` +const camel = (s) => s.replace(/_(\w)/g, (_, c) => c.toUpperCase()) +const pascal = (s) => camel(s).replace(/^\w/, (c) => c.toUpperCase()) + +// --------------------------------------------------------------------------------------------- +// Generated source files (the server knows exact contents so apply_diff blocks match) +// --------------------------------------------------------------------------------------------- +const files = new Map() // path -> { k, rev, retry } + +function fileContent(k, rev, retry) { + const name = feature(k) + const r = rng(k + 7) + const lines = [ + `// Auto-generated module: ${name}`, + ``, + `export const REVISION = ${rev}`, + `export const RETRY_LIMIT = ${retry}`, + ``, + `export interface ${pascal(name)}Options {`, + `\tname: string`, + `\tmaxItems?: number`, + `\ttimeoutMs?: number`, + `}`, + ``, + ] + let fn = 0 + while (lines.length < cfg.codeLines) { + const v = pick(r, VERBS) + lines.push( + `/** ${v} the ${name} payload, attempt ${fn} */`, + `export async function ${v}${pascal(name)}${fn}(input: string[], opts: ${pascal(name)}Options): Promise {`, + `\tconst out: string[] = []`, + `\tfor (const [index, item] of input.entries()) {`, + `\t\tif (opts.maxItems !== undefined && index >= opts.maxItems) break`, + `\t\tconst trimmed = item.trim().toLowerCase()`, + `\t\tif (trimmed.length === 0) continue`, + `\t\tout.push(\`\${opts.name}:\${trimmed}:\${index * ${fn + 2}}\`)`, + `\t}`, + `\treturn out`, + `}`, + ``, + ) + fn++ + } + return lines.join("\n") + "\n" +} + +const srcPath = (k) => `${cfg.dir}/src/${feature(k)}.ts` + +// --------------------------------------------------------------------------------------------- +// Markdown explanation generator +// --------------------------------------------------------------------------------------------- +const PARAS = [ + "I looked at how the current code paths interact and there are a few things worth calling out before making changes.", + "The existing implementation works for the happy path, but the error handling around retries is inconsistent, so I want to tighten that up first.", + "Before touching anything else, I'll keep the change small and verify it with the test runner so we can see regressions early.", + "This keeps the public surface unchanged while making the internals easier to reason about and to test in isolation.", + "One thing to watch: the `timeoutMs` option is optional, so every consumer has to handle `undefined` explicitly.", + "The trade-off here is a little more code in exchange for much clearer failure modes when the upstream service is slow.", +] +const BULLETS = [ + "Keep `REVISION` monotonically increasing so consumers can detect stale caches", + "Guard against empty input before iterating (`trimmed.length === 0`)", + "Prefer `for...of` with `entries()` over index loops for readability", + "Add a regression test for the `maxItems` boundary", + "Document the retry policy in the module header", + "Avoid mutating the input array; return a fresh `out` instead", +] + +function codeBlock(r, k) { + const name = feature(k) + const kind = pick(r, ["ts", "py", "bash", "json", "diff", "sql", "yaml"]) + switch (kind) { + case "py": + return `\`\`\`python\ndef ${name}_summary(items: list[str], limit: int = 10) -> dict[str, int]:\n counts: dict[str, int] = {}\n for item in items[:limit]:\n key = item.strip().lower()\n counts[key] = counts.get(key, 0) + 1\n return counts\n\`\`\`` + case "bash": + return `\`\`\`bash\n# run the focused suite\npnpm --dir src exec vitest run core/${name} --reporter=verbose\npnpm lint && pnpm check-types\n\`\`\`` + case "json": + return `\`\`\`json\n{\n "name": "${name}",\n "retry": { "limit": ${3 + (k % 4)}, "backoffMs": [100, 250, 500] },\n "features": ["cache", "metrics", "tracing"]\n}\n\`\`\`` + case "diff": + return `\`\`\`diff\n--- a/${srcPath(k)}\n+++ b/${srcPath(k)}\n@@ -3,2 +3,2 @@\n-export const REVISION = 0\n-export const RETRY_LIMIT = 3\n+export const REVISION = 1\n+export const RETRY_LIMIT = 5\n\`\`\`` + case "sql": + return `\`\`\`sql\nSELECT id, status, COUNT(*) AS attempts\nFROM ${name}_jobs\nWHERE created_at > NOW() - INTERVAL '1 day'\nGROUP BY id, status\nORDER BY attempts DESC\nLIMIT 50;\n\`\`\`` + case "yaml": + return `\`\`\`yaml\nservice: ${name}\nreplicas: ${2 + (k % 3)}\nenv:\n - name: LOG_LEVEL\n value: info\n - name: RETRY_LIMIT\n value: "${3 + (k % 4)}"\n\`\`\`` + default: + return `\`\`\`ts\nexport function ${camel(name)}Key(id: string, revision = REVISION): string {\n\treturn \`${name}:\${id}:v\${revision}\`\n}\n\nconst result = await ${pick(r, VERBS)}${pascal(name)}0(["a", " B ", ""], { name: "${name}", maxItems: 2 })\nconsole.log(result) // ["${name}:a:0", "${name}:b:2"]\n\`\`\`` + } +} + +function mermaidBlock(k) { + const a = feature(k) + const b = feature(k + 1) + return `\`\`\`mermaid\nflowchart TD\n A[Request] --> B{${a} cache hit?}\n B -- yes --> C[Return cached]\n B -- no --> D[${b} resolver]\n D --> E[(Store)]\n E --> C\n\`\`\`` +} + +function table(r, k) { + const rows = [["Module", "Lines", "Status"]] + for (let i = 0; i < 3 + Math.floor(r() * 3); i++) rows.push([`\`${feature(k - i < 0 ? 0 : k - i)}.ts\``, String(cfg.codeLines + i), pick(r, ["ok", "updated", "new"])]) + return [`| ${rows[0].join(" | ")} |`, `| --- | ---: | --- |`, ...rows.slice(1).map((row) => `| ${row.join(" | ")} |`)].join("\n") +} + +function markdownFor(n, k, action) { + const r = rng(n * 31 + 5) + const parts = [`## Step ${n}: ${action.title}`, "", pick(r, PARAS), ""] + let size = parts.join("\n").length + const sections = [ + () => ["**Key points**", "", ...Array.from({ length: 3 + Math.floor(r() * 3) }, () => `- ${pick(r, BULLETS)}`), ""].join("\n"), + () => [pick(r, PARAS), "", codeBlock(r, k), ""].join("\n"), + () => [table(r, k), ""].join("\n"), + () => [`Here is the relevant snippet from \`${srcPath(k)}\`:`, "", codeBlock(r, k), "", pick(r, PARAS), ""].join("\n"), + () => [`> **Note:** ${pick(r, PARAS)}`, ""].join("\n"), + () => [`1. Inspect \`${srcPath(k)}\``, `2. ${pick(r, BULLETS)}`, `3. Re-run the suite and compare output`, ""].join("\n"), + ] + if (cfg.mermaidEvery > 0 && n % cfg.mermaidEvery === 0) { + parts.push("The data flow looks like this:", "", mermaidBlock(k), "") + size += 200 + } + const targetBytes = action.textBytes ?? cfg.textBytes + while (size < targetBytes) { + const s = pick(r, sections)() + parts.push(s) + size += s.length + } + parts.push(action.closing) + return parts.join("\n") +} + +function reasoningFor(n, skip) { + if (skip || cfg.reasoningBytes <= 0) return "" + const r = rng(n * 13 + 1) + let out = "" + while (out.length < cfg.reasoningBytes) out += `${pick(r, PARAS)} ${pick(r, BULLETS)}. ` + return out.slice(0, cfg.reasoningBytes) +} + +// --------------------------------------------------------------------------------------------- +// Simulated commands (ANSI colors, streaming output, stack traces) +// --------------------------------------------------------------------------------------------- +const sleepPart = () => (cfg.cmdSleep > 0 ? `sleep ${cfg.cmdSleep}; ` : "") + +const testRunCommand = (k) => + `echo "RUN v2.1.0 ${cfg.dir}"; for i in $(seq 1 ${cfg.cmdLines}); do printf "\\033[32m ✓\\033[0m src/${feature(k)} case %d \\033[90m(%d ms)\\033[0m\\n" $i $((i % 37 + 3)); ${sleepPart()}done; printf "\\n\\033[32m Test Files 1 passed (1)\\033[0m\\n\\033[32m Tests ${cfg.cmdLines} passed (${cfg.cmdLines})\\033[0m\\n"` + +const buildErrorCommand = (k) => + `for i in $(seq 1 ${Math.max(20, Math.floor(cfg.cmdLines / 4))}); do printf "\\033[31merror\\033[0m TS2322: Type 'string' is not assignable to type 'number'.\\n \\033[36m${cfg.dir}/src/${feature(k)}.ts\\033[0m:%d:%d\\n %d | const value: number = input[%d]\\n\\n" $i $((i % 80 + 1)) $i $i; ${sleepPart()}done; echo "Found ${Math.max(20, Math.floor(cfg.cmdLines / 4))} errors."` + +const listCommand = () => `find ${cfg.dir} -type f -name "*.ts" | sort | head -100 && wc -l ${cfg.dir}/src/*.ts | tail -5` + +// --------------------------------------------------------------------------------------------- +// Turn planner: repeating coding cycle over successive "features" +// --------------------------------------------------------------------------------------------- +const CYCLE = ["todo", "write", "read", "apply_diff", "search", "test", "apply_diff", "test", "list", "build_check"] + +function planFor(n) { + if (n > cfg.maxRequests) { + return { + title: "Wrap up", + closing: "All planned changes are in place.", + tool: { name: "attempt_completion", args: { result: `## Summary\n\nFinished ${cfg.maxRequests} scripted turns in \`${cfg.dir}/\`.\n\n- Created and revised generated modules\n- Ran the simulated test suite repeatedly\n\nRun \`rm -rf ${cfg.dir}\` to clean up.` } }, + } + } + if (cfg.scenario === "churn") { + if (n <= cfg.accumulateTurns) { + // Phase 1: pile up large tool messages (new files, diffs, reads) with minimal text, quickly. + const k = cfg.fast ? n - 1 : Math.floor((n - 1) / 3) + const path = srcPath(k) + const step = cfg.fast ? 0 : (n - 1) % 3 + const fast = { fast: true, textBytes: 200 } + if (step === 0) { + files.set(path, { k, rev: 0, retry: 3 }) + return { ...fast, title: `Create \`${feature(k)}.ts\``, closing: `Creating \`${path}\`.`, tool: { name: "write_to_file", args: { path, content: fileContent(k, 0, 3) } } } + } + if (step === 1) return { ...fast, title: "Review the new file", closing: `Reading \`${path}\`.`, tool: { name: "read_file", args: { path } } } + const f = files.get(path) ?? { k, rev: 0, retry: 3 } + const next = { k, rev: f.rev + 1, retry: f.retry + 1 } + const diff = `<<<<<<< SEARCH\n:start_line:3\n-------\nexport const REVISION = ${f.rev}\nexport const RETRY_LIMIT = ${f.retry}\n=======\nexport const REVISION = ${next.rev}\nexport const RETRY_LIMIT = ${next.retry}\n>>>>>>> REPLACE` + files.set(path, next) + return { ...fast, title: "Apply a targeted fix", closing: `Applying the change to \`${path}\`.`, tool: { name: "apply_diff", args: { path, diff } } } + } + // Phase 2: long streamed markdown with a tiny tool call; every chunk re-derives state from all the tool messages above. + if (n === cfg.accumulateTurns + 1) { + console.log("[mock] ===== phase 2: streaming long markdown; watch the webview heap peak now =====") + } + return { title: "Notes", closing: "Updating the checklist.", tool: { name: "update_todo_list", args: { todos: `[-] Review notes for turn ${n}\n[ ] Summarize findings` } } } + } + if (cfg.scenario === "md" || cfg.scenario === "flood" || cfg.scenario === "rapid") { + const todos = `[-] Review notes for turn ${n}\n[ ] Summarize findings` + return { title: "Notes", closing: "Updating the checklist.", tool: { name: "update_todo_list", args: { todos } } } + } + if (cfg.scenario === "chatty") { + const k = n + return { title: "Run the suite", closing: "Running it again.", tool: { name: "execute_command", args: { command: testRunCommand(k), cwd: null, timeout: null } } } + } + + const cycleIndex = (n - 1) % CYCLE.length + const k = Math.floor((n - 1) / CYCLE.length) + const kind = CYCLE[cycleIndex] + const path = srcPath(k) + + switch (kind) { + case "todo": { + const done = Math.min(k, 6) + const todos = Array.from({ length: 8 }, (_, i) => `${i < done ? "[x]" : i === done ? "[-]" : "[ ]"} Implement and verify ${feature(i)}`).join("\n") + return { title: "Update the plan", closing: "Updating the checklist now.", tool: { name: "update_todo_list", args: { todos } } } + } + case "write": { + const retry = 3 + files.set(path, { k, rev: 0, retry }) + return { title: `Create \`${feature(k)}.ts\``, closing: `Creating \`${path}\`.`, tool: { name: "write_to_file", args: { path, content: fileContent(k, 0, retry) } } } + } + case "read": + return { title: "Review the new file", closing: `Reading \`${path}\` back to double-check it.`, tool: { name: "read_file", args: { path } } } + case "apply_diff": { + const f = files.get(path) ?? { k, rev: 0, retry: 3 } + const next = { k, rev: f.rev + 1, retry: f.retry + 1 } + const diff = `<<<<<<< SEARCH\n:start_line:3\n-------\nexport const REVISION = ${f.rev}\nexport const RETRY_LIMIT = ${f.retry}\n=======\nexport const REVISION = ${next.rev}\nexport const RETRY_LIMIT = ${next.retry}\n>>>>>>> REPLACE` + files.set(path, next) + return { title: "Apply a targeted fix", closing: `Applying the change to \`${path}\`.`, tool: { name: "apply_diff", args: { path, diff } } } + } + case "search": + return { title: "Find related usages", closing: "Searching the sandbox for other consumers.", tool: { name: "search_files", args: { path: `${cfg.dir}/src`, regex: "RETRY_LIMIT|REVISION", file_pattern: "*.ts" } } } + case "test": + return { title: "Run the tests", closing: "Running the suite.", tool: { name: "execute_command", args: { command: testRunCommand(k), cwd: null, timeout: null } } } + case "list": + return { title: "Check the workspace layout", closing: "Listing what we have so far.", tool: { name: "list_files", args: { path: cfg.dir, recursive: true } } } + default: + return { title: "Type-check and summarize", closing: "Collecting the diagnostics.", tool: { name: "execute_command", args: { command: k % 2 === 0 ? buildErrorCommand(k) : listCommand(), cwd: null, timeout: null } } } + } +} + +// --------------------------------------------------------------------------------------------- +// OpenAI-compatible streaming +// --------------------------------------------------------------------------------------------- +function sseChunk(res, id, delta, finishReason = null) { + res.write(`data: ${JSON.stringify({ id, object: "chat.completion.chunk", created: Math.floor(Date.now() / 1000), model: "mock", choices: [{ index: 0, delta, finish_reason: finishReason }] })}\n\n`) +} + +async function streamText(res, id, text, field, chunkMs = cfg.chunkMs) { + const r = rng(text.length) + for (let i = 0; i < text.length; ) { + const size = cfg.chunkMin + Math.floor(r() * (cfg.chunkMax - cfg.chunkMin + 1)) + sseChunk(res, id, { [field]: text.slice(i, i + size) }) + i += size + if (chunkMs > 0) await sleep(chunkMs * (0.5 + r())) + } +} + +async function handleCompletion(req, res, body) { + const messages = Array.isArray(body.messages) ? body.messages : [] + const hasTools = Array.isArray(body.tools) && body.tools.length > 0 + const id = `chatcmpl-mock-${Date.now()}` + + // Requests without tools (condense, prompt enhance, title generation): plain answer, not part of the script. + if (!hasTools) { + console.log(`[mock] auxiliary request messages=${messages.length} stream=${body.stream === true}`) + const text = "## Summary\n\nThe conversation so far covered creating, reviewing and revising generated modules.\n" + if (body.stream !== true) { + res.writeHead(200, { "content-type": "application/json" }) + res.end(JSON.stringify({ id, object: "chat.completion", created: Math.floor(Date.now() / 1000), model: "mock", choices: [{ index: 0, message: { role: "assistant", content: text }, finish_reason: "stop" }], usage: { prompt_tokens: 100, completion_tokens: 30, total_tokens: 130 } })) + return + } + res.writeHead(200, { "content-type": "text/event-stream", "cache-control": "no-cache", connection: "keep-alive" }) + sseChunk(res, id, { role: "assistant", content: text }) + sseChunk(res, id, {}, "stop") + res.write(`data: ${JSON.stringify({ id, object: "chat.completion.chunk", created: 0, model: "mock", choices: [], usage: { prompt_tokens: 100, completion_tokens: 30, total_tokens: 130 } })}\n\n`) + res.write("data: [DONE]\n\n") + res.end() + return + } + + const n = ++requestCount + const plan = planFor(n) + console.log(`[mock] turn #${n} tool=${plan.tool.name} messages=${messages.length} bodyBytes=${JSON.stringify(body).length}`) + + if (body.stream !== true) { + res.writeHead(200, { "content-type": "application/json" }) + res.end(JSON.stringify({ id, object: "chat.completion", created: Math.floor(Date.now() / 1000), model: "mock", choices: [{ index: 0, message: { role: "assistant", content: markdownFor(n, Math.floor((n - 1) / CYCLE.length), plan), tool_calls: [{ id: `call_${n}`, type: "function", function: { name: plan.tool.name, arguments: JSON.stringify(plan.tool.args) } }] }, finish_reason: "tool_calls" }], usage: { prompt_tokens: 100, completion_tokens: 30, total_tokens: 130 } })) + return + } + + res.writeHead(200, { "content-type": "text/event-stream", "cache-control": "no-cache", connection: "keep-alive" }) + sseChunk(res, id, { role: "assistant", content: "" }) + const reasoning = reasoningFor(n, plan.fast) + if (reasoning) await streamText(res, id, reasoning, "reasoning_content") + if (cfg.scenario !== "chatty") await streamText(res, id, markdownFor(n, Math.floor((n - 1) / CYCLE.length), plan), "content", plan.fast ? 1 : cfg.chunkMs) + + const argsJson = JSON.stringify(plan.tool.args) + sseChunk(res, id, { tool_calls: [{ index: 0, id: `call_${n}`, type: "function", function: { name: plan.tool.name, arguments: "" } }] }) + for (let i = 0; i < argsJson.length; i += cfg.toolChunk) { + sseChunk(res, id, { tool_calls: [{ index: 0, function: { arguments: argsJson.slice(i, i + cfg.toolChunk) } }] }) + if (cfg.chunkMs > 0) await sleep(Math.max(1, cfg.chunkMs / 3)) + } + sseChunk(res, id, {}, "tool_calls") + const promptTokens = 1000 + messages.length * 400 + res.write(`data: ${JSON.stringify({ id, object: "chat.completion.chunk", created: Math.floor(Date.now() / 1000), model: "mock", choices: [], usage: { prompt_tokens: promptTokens, completion_tokens: 400, total_tokens: promptTokens + 400 } })}\n\n`) + res.write("data: [DONE]\n\n") + res.end() +} + +const server = http.createServer((req, res) => { + const url = new URL(req.url, "http://x") + if (req.method === "GET" && url.pathname.endsWith("/models")) { + res.writeHead(200, { "content-type": "application/json" }) + res.end(JSON.stringify({ object: "list", data: [{ id: "mock", object: "model", created: 0, owned_by: "mock" }] })) + return + } + if (req.method === "POST" && url.pathname.endsWith("/chat/completions")) { + const chunks = [] + req.on("data", (c) => chunks.push(c)) + req.on("end", () => { + let body = {} + try { + body = JSON.parse(Buffer.concat(chunks).toString("utf8")) + } catch {} + handleCompletion(req, res, body).catch((e) => { + console.error("[mock] handler error:", e) + res.end() + }) + }) + return + } + res.writeHead(404).end() +}) + +server.listen(cfg.port, "0.0.0.0", () => { + console.log(`[mock] OpenAI-compatible server listening on 0.0.0.0:${cfg.port} -> use http://127.0.0.1:${cfg.port}/v1 (scenario=${cfg.scenario}, maxRequests=${cfg.maxRequests}, sandbox=${cfg.dir}/)`) +}) diff --git a/scripts/gray-screen/webview-heap-matrix.mjs b/scripts/gray-screen/webview-heap-matrix.mjs new file mode 100644 index 0000000000..8f368a1022 --- /dev/null +++ b/scripts/gray-screen/webview-heap-matrix.mjs @@ -0,0 +1,329 @@ +#!/usr/bin/env node +// Finds which update pattern grows the webview JS heap fastest, in real headless Chromium with the real +// webview-ui (production build in /tmp/zoo-webview-stress-build; run scripts/gray-screen/webview-render-stress.mjs once to build it). +// Each experiment reloads the page (fresh heap), hydrates a history of large tool messages, then drives updates +// for --seconds while sampling JSHeapUsedSize every 100ms. The hydration peak is reported separately (hydrPk); +// peakMB covers only the driven updates after a forced GC (baseMB). +// +// node scripts/gray-screen/webview-heap-matrix.mjs [--seconds 20] [--heap-mb 4096] [--only name1,name2] +import fs from "node:fs" +import http from "node:http" +import { createRequire } from "node:module" +import path from "node:path" +import { fileURLToPath } from "node:url" +import { resolveServedFile } from "./lib.mjs" + +const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..") +const require = createRequire(path.join(root, "webview-ui", "package.json")) +const { chromium } = require("@playwright/test") + +const args = Object.fromEntries( + process.argv.slice(2).reduce((acc, cur, i, all) => { + if (cur.startsWith("--")) acc.push([cur.slice(2), all[i + 1]]) + return acc + }, []), +) +const seconds = Number(args["seconds"] ?? 20) +const heapMb = Number(args["heap-mb"] ?? 4096) +const only = args["only"]?.split(",") +const buildDir = args["build-dir"] ?? "/tmp/zoo-webview-stress-build" +if (!fs.existsSync(path.join(buildDir, "index.html"))) { + console.error(`No build at ${buildDir}. Run: node scripts/gray-screen/webview-render-stress.mjs --scenario session --start 100 --pushes 1`) + process.exit(1) +} + +// mode: stream = messageUpdated chunks only | turns = new messages + full state push per turn | both = turns + chunks +// Find the smallest crashing data size, or lower the cap with --heap-mb to see the live-set boundary. +const EXPERIMENTS = [ + { name: "baseline-small-stream", mode: "stream", toolMb: 0.5, hz: 60 }, + { name: "T100-stream-30hz", mode: "stream", toolMb: 100, hz: 30 }, + { name: "T100-turns-2hz", mode: "turns", toolMb: 100, hz: 2 }, + // batchable: tool-only turns -> ChatView merges consecutive edit asks into one giant message (~2x memory) + { name: "T50-batch-turns-2hz", mode: "turns", toolMb: 50, hz: 2, batchable: true }, + { name: "T100-batch-turns-2hz", mode: "turns", toolMb: 100, hz: 2, batchable: true }, + { name: "T200-batch-turns-2hz", mode: "turns", toolMb: 200, hz: 2, batchable: true }, + { name: "T300-batch-turns-2hz", mode: "turns", toolMb: 300, hz: 2, batchable: true }, // crashes at the default 4096MB cap + { name: "T700-turns-2hz", mode: "turns", toolMb: 700, hz: 2 }, // crashes without batching + // realistic appliedDiff messages (diff + patch + whole pre-edit file of origBytes): originalContent inline (before) vs omitted (now) + { name: "E2000-orig50k-inline-turns-2hz", mode: "turns", tools: 2000, origBytes: 50000, hz: 2 }, + { name: "E2000-orig50k-omitted-turns-2hz", mode: "turns", tools: 2000, origBytes: 50000, omitOriginal: true, hz: 2 }, + { name: "E2000-orig50k-inline-batch-turns-2hz", mode: "turns", tools: 2000, origBytes: 50000, batchable: true, hz: 2 }, + { name: "E2000-orig50k-omitted-batch-turns-2hz", mode: "turns", tools: 2000, origBytes: 50000, omitOriginal: true, batchable: true, hz: 2 }, + { name: "E2000-orig50k-inline-stream-30hz", mode: "stream", tools: 2000, origBytes: 50000, hz: 30 }, + { name: "E2000-orig50k-omitted-stream-30hz", mode: "stream", tools: 2000, origBytes: 50000, omitOriginal: true, hz: 30 }, + { name: "E6000-orig50k-inline-turns-2hz", mode: "turns", tools: 6000, origBytes: 50000, hz: 2 }, + { name: "E6000-orig50k-omitted-turns-2hz", mode: "turns", tools: 6000, origBytes: 50000, omitOriginal: true, hz: 2 }, + // async: pushes come from a Worker at a fixed rate regardless of how fast the webview processes them (like the real extension host) + { name: "T50-batch-async-2hz", mode: "async", toolMb: 50, hz: 2, batchable: true }, + { name: "T100-batch-async-2hz", mode: "async", toolMb: 100, hz: 2, batchable: true }, + { name: "T100-batch-async-5hz", mode: "async", toolMb: 100, hz: 5, batchable: true }, + { name: "T100-async-2hz", mode: "async", toolMb: 100, hz: 2 }, + { name: "T200-batch-async-2hz", mode: "async", toolMb: 200, hz: 2, batchable: true }, + { name: "T300-batch-async-2hz", mode: "async", toolMb: 300, hz: 2, batchable: true }, + { name: "T200-batch-async-5hz", mode: "async", toolMb: 200, hz: 5, batchable: true }, + // twoByte: tool text contains Korean chars -> UTF-16 strings (2 bytes/char) for the same character count + { name: "T50-batch-twobyte-2hz", mode: "turns", toolMb: 50, hz: 2, batchable: true, twoByte: true }, + { name: "T100-batch-twobyte-2hz", mode: "turns", toolMb: 100, hz: 2, batchable: true, twoByte: true }, + { name: "T150-batch-twobyte-2hz", mode: "turns", toolMb: 150, hz: 2, batchable: true, twoByte: true }, + { name: "T100-turns-2hz+ballast2500", mode: "turns", toolMb: 100, hz: 2, ballastMb: 2500 }, // artificial retained heap +] + +const types = { ".html": "text/html", ".js": "text/javascript", ".css": "text/css", ".json": "application/json", ".wasm": "application/wasm", ".svg": "image/svg+xml", ".map": "application/json", ".woff2": "font/woff2", ".ttf": "font/ttf" } +const httpServer = http.createServer((req, res) => { + const file = resolveServedFile(buildDir, new URL(req.url, "http://x").pathname) + if (!file) return void res.writeHead(404).end() + res.writeHead(200, { + "content-type": types[path.extname(file)] ?? "application/octet-stream", + "cross-origin-opener-policy": "same-origin", + "cross-origin-embedder-policy": "require-corp", + }) + fs.createReadStream(file).pipe(res) +}) +await new Promise((r) => httpServer.listen(0, "127.0.0.1", r)) +const url = `http://127.0.0.1:${httpServer.address().port}/` + +const initScript = () => { + window.acquireVsCodeApi = () => ({ postMessage: () => {}, getState: () => undefined, setState: () => {} }) + window.IMAGES_BASE_URI = "/" + window.MATERIAL_ICONS_BASE_URI = "/" +} + +const pageDriver = async ({ mode, toolMb, hz, chunkHz, seconds, ballastMb, batchable, twoByte, tools, origBytes, omitOriginal }) => { + const sleep = (ms) => new Promise((r) => setTimeout(r, ms)) + // twoByte: one non-Latin1 char makes V8 store the whole string as UTF-16 (2 bytes/char), like Korean comments/logs. + const filler = twoByte + ? (n, seed) => { + const base = `line ${seed}: \uD55C\uAE00 The quick brown fox jumps over the lazy dog. ` + return base.repeat(Math.ceil(n / base.length)).slice(0, n) + } + : (n, seed) => `line ${seed}: The quick brown fox jumps over the lazy dog. `.repeat(Math.ceil(n / 50)).slice(0, n) + const kinds = ["newFileCreated", "appliedDiff"] + const bigBytes = 50000 + const toolCount = tools ?? Math.max(1, Math.round((toolMb * 1048576) / bigBytes)) + let ts = Date.now() - 1e6 + let seq = 0 + const messages = [{ ts: ts++, type: "say", say: "text", text: "[STRESS] heap matrix" }] + const addTurn = (i, big) => { + messages.push({ ts: ts++, type: "say", say: "api_req_started", text: JSON.stringify({ apiProtocol: "openai", tokensIn: 1000, tokensOut: 200, cost: 0.001 }) }) + // batchable: tool-only turns (no visible text between edits) -> ChatView merges them into one giant batch message + if (!(batchable && big)) messages.push({ ts: ts++, type: "say", say: "text", text: filler(300, i) }) + // origBytes: realistic appliedDiff (diff + unified patch + the whole pre-edit file). omitOriginal: what the extension sends now + // (originalContent left out, only its length kept). + const tool = big && origBytes + ? { + tool: "appliedDiff", + path: `src/gen/f${i}.ts`, + diff: filler(1500, i), + content: filler(2500, i), + diffStats: { added: 20, removed: 5 }, + ...(omitOriginal ? { originalContentLength: origBytes } : { originalContent: filler(origBytes, i) }), + } + : big + ? { tool: kinds[i % 2], path: `src/gen/f${i}.ts`, [i % 2 === 0 ? "content" : "diff"]: filler(bigBytes, i) } + : { tool: "updateTodoList", todos: [{ id: String(i), content: `item ${i}`, status: "pending" }] } + messages.push({ ts: ts++, type: "ask", ask: "tool", isAnswered: true, text: JSON.stringify(tool) }) + if (!(batchable && big)) messages.push({ ts: ts++, type: "say", say: "user_feedback", text: `ok ${i}` }) + } + for (let i = 0; i < toolCount; i++) addTurn(i, true) + const state = () => ({ type: "state", state: { version: "0.0.0", clineMessages: messages, clineMessagesSeq: ++seq, apiConfiguration: { apiProvider: "fake-ai" }, mode: "code", customModes: [], taskHistory: [], terminalShellIntegrationDisabled: true } }) + window.postMessage(state(), "*") + await sleep(3000) + if (!document.body.innerText.includes("[STRESS]")) return { error: "not rendered" } + // Retained heap that is not chat data (stands in for everything else a long session keeps alive). + window.__ballast = [] + for (let i = 0; i < (ballastMb ?? 0); i++) { + const chunk = i + "y".repeat(1048576) + chunk.charCodeAt(5000) + window.__ballast.push(chunk) + } + await window.phaseStart() + + const live = { ts: ts++, type: "say", say: "text", text: "", partial: true } + messages.push(live) + window.postMessage(state(), "*") + let posts = 0 + let turns = 0 + let chunkText = "" + const end = performance.now() + seconds * 1000 + const chunk = () => { + chunkText += "streamed markdown chunk with `code` and **bold** text. " + live.text = chunkText + window.postMessage({ type: "messageUpdated", clineMessage: { ...live } }, "*") + posts++ + } + const turn = () => { + addTurn(toolCount + turns++, false) + window.postMessage(state(), "*") + posts++ + } + if (mode === "stream") { + while (performance.now() < end) { + chunk() + await sleep(1000 / hz) + } + } else if (mode === "turns") { + while (performance.now() < end) { + turn() + await sleep(1000 / hz) + } + } else { + let next = performance.now() + const chunkEvery = 1000 / chunkHz + let nextChunk = performance.now() + while (performance.now() < end) { + const now = performance.now() + if (now >= next) { + turn() + next = now + 1000 / hz + } + if (now >= nextChunk) { + chunk() + nextChunk = now + chunkEvery + } + await sleep(Math.min(chunkEvery, 1000 / hz) / 2) + } + } + return { posts, messages: messages.length } +} + + +// Decoupled sender: a Worker posts full-state pushes at a fixed rate whether or not the webview has finished the previous one +// (the real extension host is a separate process). Reports how many pushes were sent vs processed (backlog). +const pageDriverAsync = async ({ toolMb, hz, seconds, batchable, twoByte }) => { + const sleep = (ms) => new Promise((r) => setTimeout(r, ms)) + const workerMain = () => { + let messages = [] + let ts = Date.now() - 1e6 + let seq = 0 + let turns = 0 + let toolCount = 0 + let p + let counters + const filler = (n, seed) => { + if (p.twoByte) { + const base = `line ${seed}: \uD55C\uAE00 The quick brown fox jumps over the lazy dog. ` + return base.repeat(Math.ceil(n / base.length)).slice(0, n) + } + return `line ${seed}: The quick brown fox jumps over the lazy dog. `.repeat(Math.ceil(n / 50)).slice(0, n) + } + const bigBytes = 50000 + const addTurn = (i, big) => { + messages.push({ ts: ts++, type: "say", say: "api_req_started", text: JSON.stringify({ apiProtocol: "openai", tokensIn: 1000, tokensOut: 200, cost: 0.001 }) }) + if (!(p.batchable && big)) messages.push({ ts: ts++, type: "say", say: "text", text: filler(300, i) }) + const tool = big + ? { tool: i % 2 === 0 ? "newFileCreated" : "appliedDiff", path: `src/gen/f${i}.ts`, [i % 2 === 0 ? "content" : "diff"]: filler(bigBytes, i) } + : { tool: "updateTodoList", todos: [{ id: String(i), content: `item ${i}`, status: "pending" }] } + messages.push({ ts: ts++, type: "ask", ask: "tool", isAnswered: true, text: JSON.stringify(tool) }) + if (!(p.batchable && big)) messages.push({ ts: ts++, type: "say", say: "user_feedback", text: `ok ${i}` }) + } + const stateMsg = () => ({ type: "state", state: { version: "0.0.0", clineMessages: messages, clineMessagesSeq: ++seq, apiConfiguration: { apiProvider: "fake-ai" }, mode: "code", customModes: [], taskHistory: [], terminalShellIntegrationDisabled: true } }) + let timer + self.onmessage = (e) => { + const m = e.data + if (m.type === "init") { + p = m.params + counters = m.sab ? new Int32Array(m.sab) : null + messages = [{ ts: ts++, type: "say", say: "text", text: "[STRESS] heap matrix" }] + toolCount = Math.max(1, Math.round((p.toolMb * 1048576) / bigBytes)) + for (let i = 0; i < toolCount; i++) addTurn(i, true) + self.postMessage({ kind: "state", data: stateMsg() }) + self.postMessage({ kind: "ready" }) + } else if (m.type === "start") { + timer = setInterval(() => { + addTurn(toolCount + turns++, false) + self.postMessage({ kind: "state", data: stateMsg() }) + if (counters) Atomics.add(counters, 0, 1) + }, 1000 / p.hz) + } else if (m.type === "stop") { + clearInterval(timer) + } + } + } + const sab = typeof SharedArrayBuffer !== "undefined" ? new SharedArrayBuffer(8) : null + const counters = sab ? new Int32Array(sab) : null + const worker = new Worker(URL.createObjectURL(new Blob(["(" + workerMain.toString() + ")()"], { type: "text/javascript" }))) + let received = 0 + let readyResolve + const ready = new Promise((r) => (readyResolve = r)) + worker.onmessage = (e) => { + if (e.data.kind === "state") { + window.dispatchEvent(new MessageEvent("message", { data: e.data.data })) + received++ + } else if (e.data.kind === "ready") { + readyResolve() + } + } + worker.postMessage({ type: "init", params: { toolMb, hz, batchable, twoByte }, sab }) + await ready + await sleep(3000) + if (!document.body.innerText.includes("[STRESS]")) return { error: "not rendered" } + await window.phaseStart() + received = 0 + worker.postMessage({ type: "start" }) + await sleep(seconds * 1000) + const sent = counters ? Atomics.load(counters, 0) : -1 + worker.postMessage({ type: "stop" }) + return { posts: received, sent, backlog: sent - received } +} + +const browser = await chromium.launch({ headless: true, args: [`--js-flags=--max-old-space-size=${heapMb}`] }) +const rows = [] +for (const exp of EXPERIMENTS.filter((e) => !only || only.includes(e.name))) { + const page = await browser.newPage() + let crashed = false + let crashAt = null + page.on("crash", () => { + crashed = true + crashAt = Date.now() + }) + await page.addInitScript(initScript) + await page.goto(url) + const cdp = await page.context().newCDPSession(page) + await cdp.send("Performance.enable") + await cdp.send("HeapProfiler.enable") + const heap = async () => Math.round((await cdp.send("Performance.getMetrics")).metrics.find((m) => m.name === "JSHeapUsedSize").value / 1048576) + + let peak = 0 + let hydratePeak = 0 + let baseMb = 0 + let tTo1G = null + let samplingFrom = null + await page.exposeFunction("phaseStart", async () => { + hydratePeak = peak + await cdp.send("HeapProfiler.collectGarbage") + baseMb = await heap() + peak = 0 + samplingFrom = Date.now() + }) + const sampler = setInterval(async () => { + if (crashed) return + try { + const h = await heap() + peak = Math.max(peak, h) + if (samplingFrom !== null && tTo1G === null && h >= 1024) tTo1G = ((Date.now() - samplingFrom) / 1000).toFixed(1) + } catch {} + }, 100) + + let result = {} + try { + result = await page.evaluate(exp.mode === "async" ? pageDriverAsync : pageDriver, { mode: exp.mode, toolMb: exp.toolMb, hz: exp.hz, chunkHz: exp.chunkHz, seconds, ballastMb: exp.ballastMb, batchable: exp.batchable, twoByte: exp.twoByte, tools: exp.tools, origBytes: exp.origBytes, omitOriginal: exp.omitOriginal }) + } catch (e) { + result = { error: crashed ? "CRASHED (renderer OOM)" : e.message.split("\n")[0] } + } + clearInterval(sampler) + let floorMb = "-" + if (!crashed && !result.error) { + await cdp.send("HeapProfiler.collectGarbage") + floorMb = await heap() + } + rows.push({ name: exp.name, mode: exp.mode, toolMb: exp.toolMb, twoByte: !!exp.twoByte, ballastMb: exp.ballastMb ?? 0, rate: exp.mode === "both" ? `${exp.hz}t+${exp.chunkHz}c` : exp.hz, hydratePeakMB: hydratePeak, baseMB: baseMb, peakMB: peak, afterGC: floorMb, crashAfterSec: crashed ? (samplingFrom === null ? "during-hydration" : ((crashAt - samplingFrom) / 1000).toFixed(1)) : "-", to1GBsec: tTo1G ?? "-", posts: result.posts ?? "-", sent: result.sent ?? "-", backlog: result.backlog ?? "-", result: result.error ?? "ok" }) + console.log(JSON.stringify(rows.at(-1))) + await page.close().catch(() => {}) +} +await browser.close() +httpServer.close() + +console.log("\nname".padEnd(28) + "mode".padEnd(8) + "toolMB".padEnd(8) + "rate".padEnd(10) + "hydrPk".padEnd(8) + "baseMB".padEnd(8) + "peakMB".padEnd(9) + "afterGC".padEnd(9) + "to1GB(s)".padEnd(10) + "posts".padEnd(7) + "sent".padEnd(6) + "backlog".padEnd(9) + "crashAfter(s)".padEnd(15) + "result") +for (const r of rows) { + console.log(r.name.padEnd(28) + r.mode.padEnd(8) + String(r.toolMb).padEnd(8) + String(r.rate).padEnd(10) + String(r.hydratePeakMB).padEnd(8) + String(r.baseMB).padEnd(8) + String(r.peakMB).padEnd(9) + String(r.afterGC).padEnd(9) + String(r.to1GBsec).padEnd(10) + String(r.posts).padEnd(7) + String(r.sent).padEnd(6) + String(r.backlog).padEnd(9) + String(r.crashAfterSec).padEnd(15) + r.result) +} diff --git a/scripts/gray-screen/webview-render-stress.mjs b/scripts/gray-screen/webview-render-stress.mjs new file mode 100644 index 0000000000..38e0241a89 --- /dev/null +++ b/scripts/gray-screen/webview-render-stress.mjs @@ -0,0 +1,382 @@ +#!/usr/bin/env node +// Real-render stress harness: boots the actual webview-ui (webview-ui) in headless Chromium, +// fakes acquireVsCodeApi, and feeds it extension-host-style messages while measuring real +// rendering latency (rAF), DOM size and V8 heap (before/after forced GC) via CDP. +// +// Usage (from repo root): +// node scripts/gray-screen/webview-render-stress.mjs --scenario session|command|md [options] +// --start 5000 initial message count (session) +// --pushes 300 number of full-state pushes (session) +// --grow 4 messages added per push (session) +// --chunks 20 streaming chunks per push (session) +// --rate 200 output messages/second (command) +// --seconds 60 duration (command) +// --turns 40 streamed markdown turns (md) +// --text-bytes 15000 markdown size per turn (md; also session) +// --chunk-ms 15 ms between streamed chunks (md) +// --mermaid-every 10 mermaid diagram every N turns (md; 0 disables) +// --alloc-profile true print top allocators (CDP sampling heap profiler, md) +// --build-mode development unminified build (readable function names) in a separate dir +// --heap-mb 1024 V8 old-space cap (default 1024; 4096 for md) +// --skip-build true reuse the previous build in /tmp/zoo-webview-stress-build +// --messages-file seed history from a generated task (generate-large-task.mjs) +// --url http://... use an already running server instead of building/serving +// --headed show the browser +import { spawn } from "node:child_process" +import fs from "node:fs" +import http from "node:http" +import { createRequire } from "node:module" +import path from "node:path" +import { fileURLToPath } from "node:url" +import { resolveBuildDir, resolveServedFile } from "./lib.mjs" + +const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..") +const webviewDir = path.join(root, "webview-ui") +const require = createRequire(path.join(webviewDir, "package.json")) +const { chromium } = require("@playwright/test") + +const args = Object.fromEntries( + process.argv.slice(2).reduce((acc, cur, i, all) => { + if (cur.startsWith("--")) acc.push([cur.slice(2), all[i + 1]?.startsWith("--") || all[i + 1] === undefined ? "true" : all[i + 1]]) + return acc + }, []), +) +const cfg = { + scenario: args["scenario"] ?? "session", + start: Number(args["start"] ?? 5000), + pushes: Number(args["pushes"] ?? 300), + grow: Number(args["grow"] ?? 4), + chunks: Number(args["chunks"] ?? 20), + rate: Number(args["rate"] ?? 200), + seconds: Number(args["seconds"] ?? 60), + heapMb: Number(args["heap-mb"] ?? (args["scenario"] === "md" ? 4096 : 1024)), + textBytes: Number(args["text-bytes"] ?? (args["scenario"] === "md" ? 15000 : 800)), + turns: Number(args["turns"] ?? 40), + chunkMs: Number(args["chunk-ms"] ?? 15), + mermaidEvery: Number(args["mermaid-every"] ?? 10), + url: args["url"], + headed: args["headed"] === "true", +} + +const sleep = (ms) => new Promise((r) => setTimeout(r, ms)) + +const buildMode = args["build-mode"] ?? "production" +const buildDir = resolveBuildDir(buildMode) + +function run(cmd, cmdArgs, opts) { + return new Promise((resolve, reject) => { + const child = spawn(cmd, cmdArgs, { stdio: ["ignore", "ignore", "inherit"], ...opts }) + child.on("exit", (code) => (code === 0 ? resolve() : reject(new Error(`${cmd} exited ${code}`)))) + }) +} + +// Production build into a temp dir (never touches src/webview-ui/build), then serve it statically. +async function startServer() { + if (args["skip-build"] !== "true" || !fs.existsSync(path.join(buildDir, "index.html"))) { + console.log("building webview-ui (production) into", buildDir, "...") + await run("pnpm", ["exec", "vite", "build", "--outDir", buildDir, "--emptyOutDir", "--mode", buildMode], { + cwd: webviewDir, + }) + } + const types = { ".html": "text/html", ".js": "text/javascript", ".css": "text/css", ".json": "application/json", ".wasm": "application/wasm", ".svg": "image/svg+xml", ".map": "application/json", ".woff2": "font/woff2", ".ttf": "font/ttf" } + const httpServer = http.createServer((req, res) => { + const file = resolveServedFile(buildDir, new URL(req.url, "http://x").pathname) + if (!file) { + res.writeHead(404).end() + return + } + res.writeHead(200, { "content-type": types[path.extname(file)] ?? "application/octet-stream" }) + fs.createReadStream(file).pipe(res) + }) + await new Promise((r) => httpServer.listen(0, "127.0.0.1", r)) + return { url: `http://127.0.0.1:${httpServer.address().port}/`, child: { kill: () => httpServer.close() } } +} + +const initScript = () => { + window.__posted = 0 + window.acquireVsCodeApi = () => ({ + postMessage: () => { + window.__posted++ + }, + getState: () => undefined, + setState: () => {}, + }) + window.IMAGES_BASE_URI = "/" + window.MATERIAL_ICONS_BASE_URI = "/" +} + +const baseState = (extra) => ({ + version: "0.0.0", + clineMessages: [], + taskHistory: [], + apiConfiguration: { apiProvider: "fake-ai" }, + mode: "code", + customModes: [], + terminalShellIntegrationDisabled: true, + ...extra, +}) + +const pageHelpers = () => { + window.__stress = { + filler: (n, seed) => `line ${seed}: The quick brown fox jumps over the lazy dog. `.repeat(Math.ceil(n / 50)).slice(0, n), + raf: () => new Promise((r) => requestAnimationFrame(() => requestAnimationFrame(r))), + emit: (data) => window.postMessage(data, "*"), + } +} + +async function main() { + const server = cfg.url ? null : await startServer() + const url = cfg.url ?? server.url + const browser = await chromium.launch({ + headless: !cfg.headed, + args: [`--js-flags=--max-old-space-size=${cfg.heapMb}`, "--enable-precise-memory-info"], + }) + const page = await browser.newPage() + let crashed = false + page.on("crash", () => { + crashed = true + console.error("\n*** PAGE CRASHED (renderer died: this is the 'gray screen') ***") + }) + page.on("pageerror", (e) => console.error("[pageerror]", e.message.split("\n")[0])) + await page.addInitScript(initScript) + await page.goto(url) + await page.evaluate(pageHelpers) + const cdp = await page.context().newCDPSession(page) + await cdp.send("Performance.enable") + await cdp.send("HeapProfiler.enable") + + const heap = async (gc) => { + if (gc) await cdp.send("HeapProfiler.collectGarbage") + const { metrics } = await cdp.send("Performance.getMetrics") + return Math.round(metrics.find((m) => m.name === "JSHeapUsedSize").value / 1048576) + } + const dom = () => page.evaluate(() => document.getElementsByTagName("*").length) + const frame = async () => { + const t = Date.now() + await page.evaluate(() => window.__stress.raf()) + return Date.now() - t + } + + const emit = (data) => page.evaluate((d) => window.__stress.emit(d), data) + const ts0 = Date.now() - 1e6 + let ts = ts0 + let seq = 0 + const messages = [{ ts: ts++, type: "say", say: "text", text: "[STRESS] simulated long session" }] + const round = (i) => [ + { ts: ts++, type: "say", say: "api_req_started", text: JSON.stringify({ apiProtocol: "openai", tokensIn: 1000, tokensOut: 200, cost: 0.001 }) }, + { ts: ts++, type: "say", say: "text", text: `line ${i}: ${"The quick brown fox jumps over the lazy dog. ".repeat(Math.ceil(cfg.textBytes / 45))}`.slice(0, cfg.textBytes) }, + { ts: ts++, type: "ask", ask: "tool", isAnswered: true, text: JSON.stringify({ tool: "appliedDiff", path: `src/f${i}.ts`, diff: "x".repeat(cfg.textBytes) }) }, + { ts: ts++, type: "say", say: "user_feedback", text: `ok ${i}` }, + ] + if (args["messages-file"]) { + // Seed with a task generated by scripts/gray-screen/generate-large-task.mjs (ui_messages.json). + messages.length = 0 + messages.push(...JSON.parse(fs.readFileSync(args["messages-file"], "utf8"))) + messages[0].text = `[STRESS] ${messages[0].text}` + ts = messages.at(-1).ts + 1 + } else { + for (let i = 0; messages.length < cfg.start; i++) messages.push(...round(i)) + } + + // Hydrate (with the initial history) and verify real rendering happened. + await emit({ type: "state", state: baseState({ clineMessages: messages, clineMessagesSeq: ++seq }) }) + await sleep(3000) + const rendered = await page.evaluate(() => document.body.innerText.includes("[STRESS]")) + if (!rendered) { + console.error("Render check FAILED: '[STRESS]' text not in DOM. Aborting (results would be meaningless).") + await browser.close() + server?.child.kill() + process.exit(2) + } + console.log(`render check passed: ${await dom()} DOM nodes, ${messages.length} msgs, heap ${await heap(true)}MB after GC`) + console.log(`scenario=${cfg.scenario} heapCap=${cfg.heapMb}MB (headless Chromium, ${buildMode} build)\n`) + + const t0 = Date.now() + const floor = [] + try { + if (cfg.scenario === "session") { + for (let p = 0; p < cfg.pushes && !crashed; p++) { + for (let k = 0; k < cfg.grow; k += 4) messages.push(...round(messages.length)) + await emit({ type: "state", state: baseState({ clineMessages: messages, clineMessagesSeq: ++seq }) }) + const streamTs = ts++ + const live = { ts: streamTs, type: "say", say: "text", text: "", partial: true } + messages.push(live) + await emit({ type: "state", state: baseState({ clineMessages: messages, clineMessagesSeq: ++seq }) }) + for (let c = 0; c < cfg.chunks; c++) { + live.text += `chunk ${c}: ${"streamed text ".repeat(15)}\n` + await emit({ type: "messageUpdated", clineMessage: { ...live } }) + await sleep(50) + } + live.partial = false + if (p % 10 === 0) { + const lat = await frame() + const gcHeap = await heap(true) + floor.push(gcHeap) + console.log(`push ${String(p).padStart(4)} msgs=${messages.length} dom=${await dom()} frameLatency=${lat}ms heapAfterGC=${gcHeap}MB t=${Math.round((Date.now() - t0) / 1000)}s`) + } + } + } else if (cfg.scenario === "command") { + const execTs = ts++ + messages.push({ ts: execTs, type: "ask", ask: "command", text: "npm run dev", isAnswered: true }) + await emit({ type: "state", state: baseState({ clineMessages: messages, clineMessagesSeq: ++seq }) }) + const executionId = String(execTs) + const post = (s) => emit({ type: "commandExecutionStatus", text: JSON.stringify({ executionId, ...s }) }) + await post({ status: "started", pid: 1, command: "npm run dev" }) + const lineText = (n) => `[${n}] ${"build output line ".repeat(10)}`.slice(0, 100) + await post({ status: "output", output: Array.from({ length: 500 }, (_, k) => lineText(k)).join("\n") }) + await sleep(1500) + const shown = await page.evaluate(() => document.body.innerText.includes("build output line")) + if (!shown) { + console.error("Render check FAILED: command output is not in the DOM (row collapsed or not rendered). Aborting.") + await browser.close() + server?.child.kill() + process.exit(2) + } + console.log("command output render check passed (TerminalOutput is mounted and visible)") + let counter = 0 + const end = Date.now() + cfg.seconds * 1000 + let sent = 0 + let lastLog = Date.now() + while (Date.now() < end && !crashed) { + const batchStart = Date.now() + const perBatch = Math.max(1, Math.round(cfg.rate / 20)) + for (let i = 0; i < perBatch; i++) { + const start = counter++ + await post({ status: "output", output: Array.from({ length: 500 }, (_, k) => lineText(start + k)).join("\n") }) + sent++ + } + const wait = 50 - (Date.now() - batchStart) + if (wait > 0) await sleep(wait) + if (Date.now() - lastLog > 5000) { + lastLog = Date.now() + const lat = await frame() + const gcHeap = await heap(true) + floor.push(gcHeap) + console.log(`t=${Math.round((Date.now() - t0) / 1000)}s sent=${sent} dom=${await dom()} frameLatency=${lat}ms heapAfterGC=${gcHeap}MB`) + } + } + } else if (cfg.scenario === "md") { + // Streams realistic markdown (code blocks, tables, lists, mermaid) the way Zoo Code does: + // every chunk re-sends the FULL accumulated text of the partial message. + const rngFor = (seed) => { + let a = (seed * 2654435761) >>> 0 + return () => { + a = (a + 0x6d2b79f5) >>> 0 + let t = a + t = Math.imul(t ^ (t >>> 15), t | 1) + t ^= t + Math.imul(t ^ (t >>> 7), t | 61) + return ((t ^ (t >>> 14)) >>> 0) / 4294967296 + } + } + const PARAS = [ + "I looked at how the current code paths interact and there are a few things worth calling out before making changes.", + "The existing implementation works for the happy path, but the error handling around retries is inconsistent.", + "This keeps the public surface unchanged while making the internals easier to reason about and to test in isolation.", + "One thing to watch: the `timeoutMs` option is optional, so every consumer has to handle `undefined` explicitly.", + ] + const BULLETS = ["Keep `REVISION` monotonically increasing", "Guard against empty input before iterating", "Prefer `for...of` with `entries()`", "Add a regression test for the boundary"] + const CODE = [ + (n) => "```ts\nexport async function resolve" + n + "(input: string[]): Promise {\n\tconst out: string[] = []\n\tfor (const [i, item] of input.entries()) {\n\t\tout.push(`${item.trim()}:${i}`)\n\t}\n\treturn out\n}\n```", + (n) => "```python\ndef summary_" + n + "(items):\n counts = {}\n for item in items:\n counts[item] = counts.get(item, 0) + 1\n return counts\n```", + (n) => "```bash\npnpm --dir src exec vitest run core/feature_" + n + "\npnpm lint && pnpm check-types\n```", + (n) => "```json\n{\n \"name\": \"feature_" + n + "\",\n \"retry\": { \"limit\": 3, \"backoffMs\": [100, 250, 500] }\n}\n```", + (n) => "```diff\n--- a/src/f" + n + ".ts\n+++ b/src/f" + n + ".ts\n@@ -3,2 +3,2 @@\n-export const REVISION = 0\n+export const REVISION = 1\n```", + ] + const mdFor = (n) => { + const r = rngFor(n * 31 + 5) + const pick = (a) => a[Math.floor(r() * a.length)] + let out = `## Step ${n}: review\n\n${pick(PARAS)}\n\n` + if (cfg.mermaidEvery > 0 && n % cfg.mermaidEvery === 0) { + out += "```mermaid\nflowchart TD\n A[Request] --> B{cache hit?}\n B -- yes --> C[Return cached]\n B -- no --> D[Resolver]\n D --> C\n```\n\n" + } + while (out.length < cfg.textBytes) { + const kind = Math.floor(r() * 4) + if (kind === 0) out += `**Key points**\n\n${[0, 1, 2].map(() => `- ${pick(BULLETS)}`).join("\n")}\n\n` + else if (kind === 1) out += `${pick(PARAS)}\n\n${pick(CODE)(n)}\n\n` + else if (kind === 2) out += "| Module | Lines | Status |\n| --- | ---: | --- |\n| `a.ts` | 150 | ok |\n| `b.ts` | 151 | updated |\n\n" + else out += `> **Note:** ${pick(PARAS)}\n\n` + } + return out + } + + const allocProfile = args["alloc-profile"] === "true" + if (allocProfile) { + await cdp.send("HeapProfiler.startSampling", { + samplingInterval: 16384, + includeObjectsCollectedByMajorGC: true, + includeObjectsCollectedByMinorGC: true, + }) + } + let peak = 0 + const sampler = setInterval(async () => { + try { + const { metrics } = await cdp.send("Performance.getMetrics") + peak = Math.max(peak, Math.round(metrics.find((m) => m.name === "JSHeapUsedSize").value / 1048576)) + } catch {} + }, 250) + for (let turn = 1; turn <= cfg.turns && !crashed; turn++) { + messages.push({ ts: ts++, type: "say", say: "api_req_started", text: JSON.stringify({ apiProtocol: "openai", tokensIn: 1000, tokensOut: 200, cost: 0.001 }) }) + const live = { ts: ts++, type: "say", say: "text", text: "", partial: true } + messages.push(live) + await emit({ type: "state", state: baseState({ clineMessages: messages, clineMessagesSeq: ++seq }) }) + const full = mdFor(turn) + let updates = 0 + for (let i = 0; i < full.length; ) { + i += 8 + Math.floor(Math.random() * 40) + live.text = full.slice(0, i) + await emit({ type: "messageUpdated", clineMessage: { ...live } }) + updates++ + if (cfg.chunkMs > 0) await sleep(cfg.chunkMs * (0.5 + Math.random())) + } + live.text = full + live.partial = false + await emit({ type: "messageUpdated", clineMessage: { ...live } }) + const lat = await frame() + const turnPeak = peak + peak = 0 + const gcHeap = await heap(true) + floor.push(gcHeap) + console.log(`turn ${String(turn).padStart(3)} msgs=${messages.length} updates=${updates} dom=${await dom()} frameLatency=${lat}ms peakHeap=${turnPeak}MB heapAfterGC=${gcHeap}MB t=${Math.round((Date.now() - t0) / 1000)}s`) + } + clearInterval(sampler) + if (allocProfile) { + const { profile } = await cdp.send("HeapProfiler.stopSampling") + const totals = new Map() + const walk = (node) => { + const f = node.callFrame + const key = `${f.functionName || "(anonymous)"} @ ${f.url.split("/").pop()}:${f.lineNumber + 1}` + totals.set(key, (totals.get(key) ?? 0) + node.selfSize) + node.children.forEach(walk) + } + walk(profile.head) + const sum = [...totals.values()].reduce((a, b) => a + b, 0) + console.log(`\nAllocation sampling (self size incl. collected objects), total ${Math.round(sum / 1048576)}MB:`) + for (const [k, v] of [...totals.entries()].sort((a, b) => b[1] - a[1]).slice(0, 25)) { + console.log(` ${String(Math.round(v / 1048576)).padStart(6)}MB ${((v / sum) * 100).toFixed(1).padStart(5)}% ${k}`) + } + } + } else { + throw new Error(`unknown scenario ${cfg.scenario}`) + } + } catch (e) { + if (!crashed) console.error("scenario error:", e.message.split("\n")[0]) + } + + console.log("") + if (crashed) { + console.log("RESULT: renderer crashed (OOM reproduced).") + } else if (floor.length >= 4) { + const q = Math.max(1, Math.floor(floor.length / 4)) + const first = Math.round(floor.slice(0, q).reduce((a, b) => a + b, 0) / q) + const last = Math.round(floor.slice(-q).reduce((a, b) => a + b, 0) / q) + console.log(`RESULT: heap-after-GC first quarter avg ${first}MB -> last quarter avg ${last}MB (${last > first * 1.3 ? "GROWING: likely retained memory" : "flat: no leak on this path"})`) + } + await browser.close().catch(() => {}) + server?.child.kill() + process.exit(crashed ? 1 : 0) +} + +main().catch((e) => { + console.error(e) + process.exit(1) +}) diff --git a/src/core/webview/ClineProvider.ts b/src/core/webview/ClineProvider.ts index 34848ab860..a8cbfbdd00 100644 --- a/src/core/webview/ClineProvider.ts +++ b/src/core/webview/ClineProvider.ts @@ -129,6 +129,7 @@ import { import { readTaskMessages } from "../task-persistence/taskMessages" import { getNonce } from "./getNonce" import { getUri } from "./getUri" +import { omitOriginalContentFromExtensionMessage } from "./stripOriginalContent" import { REQUESTY_BASE_URL } from "../../shared/utils/requesty" import { validateAndFixToolResultIds } from "../task/validateToolResultIds" import { PendingEditOperationStore, type PendingEditOperationInput } from "./PendingEditOperationStore" @@ -1427,7 +1428,7 @@ export class ClineProvider } try { - await this.view?.webview.postMessage(message) + await this.view?.webview.postMessage(omitOriginalContentFromExtensionMessage(message)) } catch { // View disposed, drop message silently } diff --git a/src/core/webview/__tests__/ClineProvider.spec.ts b/src/core/webview/__tests__/ClineProvider.spec.ts index b312853e76..2ace012534 100644 --- a/src/core/webview/__tests__/ClineProvider.spec.ts +++ b/src/core/webview/__tests__/ClineProvider.spec.ts @@ -838,6 +838,60 @@ describe("ClineProvider", () => { await expect(provider.postMessageToWebview(message)).resolves.toBeUndefined() }) + test("postMessageToWebview leaves originalContent of file-edit tool messages out of the state it posts", async () => { + await provider.resolveWebviewView(mockWebviewView) + + const originalFile = "line of the original file\n".repeat(500) + const toolText = JSON.stringify({ + tool: "appliedDiff", + path: "a.ts", + diff: "@@ d", + originalContent: originalFile, + }) + // Only the field under test is populated; the rest of ExtensionState is irrelevant here. + const message = { + type: "state", + state: { clineMessages: [{ ts: 1, type: "ask", ask: "tool", text: toolText }] }, + } as unknown as ExtensionMessage + + await provider.postMessageToWebview(message) + + const posted = mockPostMessage.mock.calls.at(-1)![0] as ExtensionMessage + const postedText = posted.state!.clineMessages![0]!.text! + + expect(JSON.parse(postedText)).toEqual({ + tool: "appliedDiff", + path: "a.ts", + diff: "@@ d", + originalContentLength: originalFile.length, + }) + // the extension's own message (and so the persisted task) keeps the full content + expect(message.state!.clineMessages![0]!.text).toBe(toolText) + }) + + test("postMessageToWebview leaves originalContent out of messageUpdated", async () => { + await provider.resolveWebviewView(mockWebviewView) + + const toolText = JSON.stringify({ + tool: "appliedDiff", + path: "a.ts", + originalContent: "original file\n".repeat(100), + }) + + await provider.postMessageToWebview({ + type: "messageUpdated", + clineMessage: { ts: 2, type: "ask", ask: "tool", text: toolText }, + }) + + const posted = mockPostMessage.mock.calls.at(-1)![0] as ExtensionMessage + + expect(JSON.parse(posted.clineMessage!.text!)).toEqual({ + tool: "appliedDiff", + path: "a.ts", + originalContentLength: "original file\n".length * 100, + }) + }) + describe("theme fixture probes", () => { const fixture = { themeId: "Default Dark Modern", diff --git a/src/core/webview/__tests__/stripOriginalContent.spec.ts b/src/core/webview/__tests__/stripOriginalContent.spec.ts new file mode 100644 index 0000000000..ed68a0be51 --- /dev/null +++ b/src/core/webview/__tests__/stripOriginalContent.spec.ts @@ -0,0 +1,245 @@ +import type { ClineMessage, ExtensionMessage } from "@roo-code/types" + +import { + findOriginalContent, + omitOriginalContent, + omitOriginalContentFromExtensionMessage, +} from "../stripOriginalContent" + +let ts = 0 +const toolAsk = (payload: unknown, extra: Partial = {}): ClineMessage => ({ + ts: ++ts, + type: "ask", + ask: "tool", + text: typeof payload === "string" ? payload : JSON.stringify(payload), + ...extra, +}) + +const bigOriginal = "line of the original file\n".repeat(2000) + +describe("omitOriginalContent", () => { + it("removes a non-empty originalContent and records its length", () => { + const message = toolAsk({ + tool: "appliedDiff", + path: "a.ts", + diff: "@@ d", + content: "patch", + originalContent: bigOriginal, + }) + + const result = omitOriginalContent(message) + const payload = JSON.parse(result.text!) + + expect(payload).toEqual({ + tool: "appliedDiff", + path: "a.ts", + diff: "@@ d", + content: "patch", + originalContentLength: bigOriginal.length, + }) + expect(result.text!.length).toBeLessThan(message.text!.length / 10) + expect(result).not.toBe(message) + expect(result.ts).toBe(message.ts) + }) + + it("does not modify the original message object", () => { + const message = toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal }) + const before = message.text + + omitOriginalContent(message) + + expect(message.text).toBe(before) + }) + + it("keeps an empty originalContent (a new file) inline", () => { + const message = toolAsk({ tool: "newFileCreated", path: "new.ts", content: "x", originalContent: "" }) + + expect(omitOriginalContent(message)).toBe(message) + }) + + it("works for say tool messages", () => { + const message: ClineMessage = { + ts: ++ts, + type: "say", + say: "tool", + text: JSON.stringify({ tool: "editedExistingFile", path: "a.ts", originalContent: bigOriginal }), + } + + expect(JSON.parse(omitOriginalContent(message).text!).originalContentLength).toBe(bigOriginal.length) + }) + + it("leaves messages without originalContent untouched", () => { + const messages = [ + toolAsk({ tool: "readFile", path: "a.ts" }), + toolAsk({ tool: "appliedDiff", path: "a.ts", diff: "d" }), + { ts: ++ts, type: "say", say: "text", text: "originalContent is only a word here" } as ClineMessage, + { ts: ++ts, type: "ask", ask: "command", text: '{"originalContent":"not a tool message"}' } as ClineMessage, + { ts: ++ts, type: "ask", ask: "tool" } as ClineMessage, + ] + + for (const message of messages) { + expect(omitOriginalContent(message)).toBe(message) + } + }) + + it("leaves unparsable text untouched", () => { + const message = toolAsk('{"tool":"appliedDiff","originalContent":"cut off') + + expect(omitOriginalContent(message)).toBe(message) + }) + + it("leaves a non-string originalContent untouched", () => { + const message = toolAsk({ tool: "appliedDiff", originalContent: 42 }) + + expect(omitOriginalContent(message)).toBe(message) + }) + + it("reuses the result while the text is unchanged and recomputes after an in-place update", () => { + const parse = vi.spyOn(JSON, "parse") + const message = toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal }) + + const first = omitOriginalContent(message) + const second = omitOriginalContent(message) + + expect(second).toEqual(first) + expect(parse).toHaveBeenCalledTimes(1) + parse.mockRestore() + + // partial messages are updated in place by the task, so a new text must not return a stale result + message.text = JSON.stringify({ tool: "appliedDiff", path: "b.ts", originalContent: bigOriginal + "more" }) + const third = omitOriginalContent(message) + + expect(JSON.parse(third.text!)).toMatchObject({ path: "b.ts", originalContentLength: bigOriginal.length + 4 }) + }) + + it("takes metadata from the current message when the cached text is reused", () => { + const message = toolAsk( + { tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal }, + { partial: false, isAnswered: false }, + ) + + expect(omitOriginalContent(message).isAnswered).toBe(false) + + // approval only flips metadata; the text is unchanged + message.isAnswered = true + message.partial = true + const result = omitOriginalContent(message) + + expect(result).toMatchObject({ isAnswered: true, partial: true }) + expect(JSON.parse(result.text!)).toMatchObject({ originalContentLength: bigOriginal.length }) + expect(result.text).not.toContain(bigOriginal.slice(0, 50)) + }) + + it("is idempotent", () => { + const once = omitOriginalContent(toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal })) + + expect(omitOriginalContent(once)).toBe(once) + }) +}) + +describe("findOriginalContent", () => { + it("returns the originalContent of the tool message with that ts", () => { + const messages = [ + toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal }), + toolAsk({ tool: "appliedDiff", path: "b.ts", originalContent: "other" }), + ] + + expect(findOriginalContent(messages, { ts: messages[0]!.ts })).toBe(bigOriginal) + expect(findOriginalContent(messages, { ts: messages[1]!.ts })).toBe("other") + }) + + it("tells messages created in the same millisecond apart by messageId", () => { + const first = toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: "first" }, { messageId: "id-1" }) + const second = toolAsk( + { tool: "appliedDiff", path: "b.ts", originalContent: "second" }, + { messageId: "id-2", ts: first.ts }, + ) + const messages = [first, second] + + expect(findOriginalContent(messages, { messageId: "id-2", ts: first.ts })).toBe("second") + expect(findOriginalContent(messages, { messageId: "id-1", ts: first.ts })).toBe("first") + expect(findOriginalContent(messages, { messageId: "missing", ts: first.ts })).toBeNull() + // messages persisted without an id are still found by ts + expect(findOriginalContent(messages, { ts: first.ts })).toBe("first") + }) + + it("finds say tool messages too", () => { + const message: ClineMessage = { + ts: ++ts, + type: "say", + say: "tool", + text: JSON.stringify({ tool: "editedExistingFile", originalContent: bigOriginal }), + } + + expect(findOriginalContent([message], { ts: message.ts })).toBe(bigOriginal) + }) + + it("returns null when there is nothing to return", () => { + const noOriginal = toolAsk({ tool: "readFile", path: "a.ts" }) + const notTool: ClineMessage = { + ts: ++ts, + type: "say", + say: "text", + text: JSON.stringify({ originalContent: "x" }), + } + const unparsable = toolAsk('{"originalContent":"cut off') + const notAString = toolAsk({ tool: "appliedDiff", originalContent: 42 }) + const noText: ClineMessage = { ts: ++ts, type: "ask", ask: "tool" } + const all = [noOriginal, notTool, unparsable, notAString, noText] + + for (const message of all) { + expect(findOriginalContent(all, { ts: message.ts })).toBeNull() + } + + expect(findOriginalContent(all, { ts: -1 })).toBeNull() + expect(findOriginalContent(undefined, { ts: 1 })).toBeNull() + }) + + it("returns an empty original as an empty string", () => { + const message = toolAsk({ tool: "newFileCreated", content: "x", originalContent: "" }) + + expect(findOriginalContent([message], { ts: message.ts })).toBe("") + }) +}) + +describe("omitOriginalContentFromExtensionMessage", () => { + it("applies to the chat messages inside a state message without touching other state", () => { + const message = { + type: "state", + state: { + version: "1", + mode: "code", + clineMessages: [toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal })], + }, + } as unknown as ExtensionMessage + + const result = omitOriginalContentFromExtensionMessage(message) + + expect(result.state).toMatchObject({ version: "1", mode: "code" }) + expect(JSON.parse(result.state!.clineMessages![0]!.text!).originalContentLength).toBe(bigOriginal.length) + }) + + it("applies to messageUpdated", () => { + const message = { + type: "messageUpdated", + clineMessage: toolAsk({ tool: "appliedDiff", path: "a.ts", originalContent: bigOriginal }), + } as ExtensionMessage + + const result = omitOriginalContentFromExtensionMessage(message) + + expect(JSON.parse(result.clineMessage!.text!).originalContentLength).toBe(bigOriginal.length) + }) + + it("passes every other message through unchanged", () => { + const others = [ + { type: "action", action: "didBecomeVisible" }, + { type: "state", state: { clineMessages: [] } }, + { type: "state" }, + { type: "fileContent", fileContent: { path: "a", content: bigOriginal } }, + ] as unknown as ExtensionMessage[] + + for (const message of others) { + expect(omitOriginalContentFromExtensionMessage(message)).toBe(message) + } + }) +}) diff --git a/src/core/webview/__tests__/webviewMessageHandler.readOriginalContent.spec.ts b/src/core/webview/__tests__/webviewMessageHandler.readOriginalContent.spec.ts new file mode 100644 index 0000000000..b3d083c212 --- /dev/null +++ b/src/core/webview/__tests__/webviewMessageHandler.readOriginalContent.spec.ts @@ -0,0 +1,170 @@ +// npx vitest core/webview/__tests__/webviewMessageHandler.readOriginalContent.spec.ts + +import { describe, it, expect, vi, beforeEach } from "vitest" + +vi.mock("../../../api/providers/fetchers/modelCache") + +vi.mock("vscode", () => ({ + window: { + showInformationMessage: vi.fn(), + showErrorMessage: vi.fn(), + showTextDocument: vi.fn(), + }, + workspace: { + workspaceFolders: [{ uri: { fsPath: "/mock/workspace" } }], + openTextDocument: vi.fn().mockResolvedValue({}), + }, +})) + +vi.mock("../../../i18n", () => ({ + t: vi.fn((key: string) => key), +})) + +vi.mock("fs/promises", () => { + const readFile = vi.fn().mockResolvedValue("file content here") + return { + default: { + rm: vi.fn(), + mkdir: vi.fn(), + readFile, + writeFile: vi.fn(), + }, + rm: vi.fn(), + mkdir: vi.fn(), + readFile, + writeFile: vi.fn(), + } +}) + +vi.mock("../../../utils/fs") +vi.mock("../../../utils/path") +vi.mock("../../../utils/globalContext") + +vi.mock("../../../utils/pathUtils", () => ({ + isPathOutsideWorkspace: vi.fn((filePath: string) => { + const nodePath = require("path") + const normalized = nodePath.resolve(filePath) + const workspaceRoot = nodePath.resolve("/mock/workspace") + // Path is inside workspace if it equals or is under workspace root + if (normalized === workspaceRoot) return false + if (normalized.startsWith(workspaceRoot + nodePath.sep)) return false + return true + }), +})) + +vi.mock("../../mentions/resolveImageMentions", () => ({ + resolveImageMentions: vi.fn(async ({ text, images }: { text: string; images?: string[] }) => ({ + text, + images: [...(images ?? [])], + })), +})) + +import { webviewMessageHandler } from "../webviewMessageHandler" +import type { ClineProvider } from "../ClineProvider" +import type { ClineMessage } from "@roo-code/types" + +const originalFile = "const a = 1\nconst b = 2\n" + +const toolMessage = (ts: number, payload: unknown, type: "ask" | "say" = "ask", messageId?: string): ClineMessage => + type === "ask" + ? { ts, type: "ask", ask: "tool", text: JSON.stringify(payload), ...(messageId && { messageId }) } + : { ts, type: "say", say: "tool", text: JSON.stringify(payload), ...(messageId && { messageId }) } + +function createProvider(clineMessages: ClineMessage[] | undefined) { + const postMessageToWebview = vi.fn() + // Only the members the handler touches for this message type. + const provider = { + postMessageToWebview, + getCurrentTask: vi.fn().mockReturnValue(clineMessages ? { taskId: "task-1", clineMessages } : undefined), + } as unknown as ClineProvider + + return { provider, postMessageToWebview } +} + +describe("webviewMessageHandler - readOriginalContent", () => { + beforeEach(() => { + vi.clearAllMocks() + }) + + it("answers with the originalContent of the requested tool message", async () => { + const { provider, postMessageToWebview } = createProvider([ + toolMessage(10, { tool: "appliedDiff", path: "a.ts", diff: "d", originalContent: originalFile }), + toolMessage(11, { tool: "appliedDiff", path: "b.ts", diff: "d", originalContent: "other" }), + ]) + + await webviewMessageHandler(provider, { type: "readOriginalContent", messageTs: 10 }) + + expect(postMessageToWebview).toHaveBeenCalledWith({ + type: "originalContent", + originalContentInfo: { ts: 10, messageId: undefined, taskId: undefined, content: originalFile }, + }) + }) + + it("answers with null content for an unknown ts or when there is no current task", async () => { + const unknown = createProvider([toolMessage(40, { tool: "appliedDiff", originalContent: "x" })]) + await webviewMessageHandler(unknown.provider, { type: "readOriginalContent", messageTs: 999 }) + expect(unknown.postMessageToWebview).toHaveBeenCalledWith({ + type: "originalContent", + originalContentInfo: { ts: 999, messageId: undefined, taskId: undefined, content: null }, + }) + + const noTask = createProvider(undefined) + await webviewMessageHandler(noTask.provider, { type: "readOriginalContent", messageTs: 10 }) + expect(noTask.postMessageToWebview).toHaveBeenCalledWith({ + type: "originalContent", + originalContentInfo: { ts: 10, messageId: undefined, taskId: undefined, content: null }, + }) + }) + + it("picks the message by messageId when two messages share a ts", async () => { + const { provider, postMessageToWebview } = createProvider([ + toolMessage(10, { tool: "appliedDiff", path: "a.ts", originalContent: "first" }, "ask", "id-1"), + toolMessage(10, { tool: "appliedDiff", path: "b.ts", originalContent: "second" }, "ask", "id-2"), + ]) + + await webviewMessageHandler(provider, { type: "readOriginalContent", messageTs: 10, messageId: "id-2" }) + + expect(postMessageToWebview).toHaveBeenCalledWith({ + type: "originalContent", + originalContentInfo: { ts: 10, messageId: "id-2", taskId: undefined, content: "second" }, + }) + }) + + it("echoes the task id and answers null for a request made for another task", async () => { + const { provider, postMessageToWebview } = createProvider([ + toolMessage(10, { tool: "appliedDiff", originalContent: originalFile }, "ask", "id-1"), + ]) + + await webviewMessageHandler(provider, { + type: "readOriginalContent", + messageTs: 10, + messageId: "id-1", + taskId: "task-1", + }) + await webviewMessageHandler(provider, { + type: "readOriginalContent", + messageTs: 10, + messageId: "id-1", + taskId: "task-2", + }) + + expect(postMessageToWebview).toHaveBeenNthCalledWith(1, { + type: "originalContent", + originalContentInfo: { ts: 10, messageId: "id-1", taskId: "task-1", content: originalFile }, + }) + expect(postMessageToWebview).toHaveBeenNthCalledWith(2, { + type: "originalContent", + originalContentInfo: { ts: 10, messageId: "id-1", taskId: "task-2", content: null }, + }) + }) + + it("ignores a request without a ts", async () => { + const { provider, postMessageToWebview } = createProvider([ + toolMessage(10, { tool: "appliedDiff", originalContent: "x" }), + ]) + + await webviewMessageHandler(provider, { type: "readOriginalContent" }) + + expect(postMessageToWebview).not.toHaveBeenCalled() + }) +}) diff --git a/src/core/webview/stripOriginalContent.ts b/src/core/webview/stripOriginalContent.ts new file mode 100644 index 0000000000..7d966282dc --- /dev/null +++ b/src/core/webview/stripOriginalContent.ts @@ -0,0 +1,82 @@ +import type { ClineMessage, ExtensionMessage } from "@roo-code/types" + +// Keyed by message object; only the transformed text is cached (valid while the source text is unchanged), so +// metadata such as `isAnswered` and `partial` is always taken from the current message. +const cache = new WeakMap() + +function isToolMessage(message: ClineMessage): boolean { + return (message.type === "ask" && message.ask === "tool") || (message.type === "say" && message.say === "tool") +} + +/** Replaces a file-edit tool message's `originalContent` (the whole pre-edit file) with its length. */ +export function omitOriginalContent(message: ClineMessage): ClineMessage { + const text = message.text + + if (typeof text !== "string" || !isToolMessage(message) || !text.includes('"originalContent"')) { + return message + } + + let entry = cache.get(message) + + if (!entry || entry.source !== text) { + entry = { source: text, strippedText: stripOriginalContentFromText(text) } + cache.set(message, entry) + } + + return entry.strippedText === undefined ? message : { ...message, text: entry.strippedText } +} + +function stripOriginalContentFromText(text: string): string | undefined { + try { + const { originalContent, ...rest } = JSON.parse(text) as Record + + // An empty original (new file) is free and still means "has an original" to the webview, so it stays. + if (typeof originalContent === "string" && originalContent.length > 0) { + return JSON.stringify({ ...rest, originalContentLength: originalContent.length }) + } + } catch { + // Not valid JSON (e.g. a truncated partial message): leave it untouched. + } + + return undefined +} + +/** + * The `originalContent` of a tool message, or null when there is none. `ts` is not unique (two messages can be + * created in the same millisecond), so `messageId` is preferred; `ts` only serves messages persisted without one. + */ +export function findOriginalContent( + messages: ClineMessage[] | undefined, + id: { messageId?: string; ts: number }, +): string | null { + const message = messages?.find( + (m) => isToolMessage(m) && (id.messageId !== undefined ? m.messageId === id.messageId : m.ts === id.ts), + ) + + if (!message?.text) { + return null + } + + try { + const { originalContent } = JSON.parse(message.text) as { originalContent?: unknown } + return typeof originalContent === "string" ? originalContent : null + } catch { + return null + } +} + +/** The webview gets `originalContent` on demand (`readOriginalContent`) instead of inside every chat message. */ +export function omitOriginalContentFromExtensionMessage(message: ExtensionMessage): ExtensionMessage { + if (message.type === "state" && message.state?.clineMessages?.length) { + return { + ...message, + state: { ...message.state, clineMessages: message.state.clineMessages.map(omitOriginalContent) }, + } + } + + if (message.type === "messageUpdated" && message.clineMessage) { + return { ...message, clineMessage: omitOriginalContent(message.clineMessage) } + } + + return message +} diff --git a/src/core/webview/webviewMessageHandler.ts b/src/core/webview/webviewMessageHandler.ts index 4ba94d454c..193540455b 100644 --- a/src/core/webview/webviewMessageHandler.ts +++ b/src/core/webview/webviewMessageHandler.ts @@ -40,6 +40,7 @@ import { saveTaskMessages } from "../task-persistence" import { importRooTaskHistory } from "../task-persistence/importRooTaskHistory" import { ClineProvider } from "./ClineProvider" +import { findOriginalContent } from "./stripOriginalContent" import { handleCheckpointRestoreOperation } from "./checkpointRestoreHandler" import { generateErrorDiagnostics } from "./diagnosticsHandler" import { @@ -1617,6 +1618,29 @@ export const webviewMessageHandler = async ( } break } + case "readOriginalContent": { + const ts = message.messageTs + + if (typeof ts !== "number") { + break + } + + const task = provider.getCurrentTask() + // A request made for another task must not be answered from the current one. + const isRequestedTask = message.taskId === undefined || message.taskId === task?.taskId + const id = { messageId: message.messageId, ts } + + await provider.postMessageToWebview({ + type: "originalContent", + originalContentInfo: { + ts, + messageId: message.messageId, + taskId: message.taskId, + content: isRequestedTask ? findOriginalContent(task?.clineMessages, id) : null, + }, + }) + break + } case "openMention": await openMention(getCurrentCwd(), message.text) break diff --git a/webview-ui/src/__tests__/FileChangesPanel.spec.tsx b/webview-ui/src/__tests__/FileChangesPanel.spec.tsx index 2208bc127d..a1c3701955 100644 --- a/webview-ui/src/__tests__/FileChangesPanel.spec.tsx +++ b/webview-ui/src/__tests__/FileChangesPanel.spec.tsx @@ -1,4 +1,5 @@ import React from "react" +import { act } from "@testing-library/react" import { fireEvent, render, screen } from "@/utils/test-utils" import type { ClineMessage } from "@roo-code/types" import { TranslationProvider } from "@/i18n/__mocks__/TranslationContext" @@ -28,15 +29,18 @@ vi.mock("react-i18next", () => ({ vi.mock("@src/components/common/CodeAccordion", () => ({ default: ({ path, + code, isExpanded, onToggleExpand, }: { path?: string + code?: string isExpanded: boolean onToggleExpand: () => void }) => (
{path} +
{code}
@@ -64,10 +68,10 @@ function createFileEditMessage( } } -function renderPanel(messages: ClineMessage[] | undefined) { +function renderPanel(messages: ClineMessage[] | undefined, taskId?: string) { return render( - + , ) } @@ -196,4 +200,245 @@ describe("FileChangesPanel", () => { expect(screen.getByTestId("total-added")).toHaveTextContent("+5") expect(screen.getByTestId("total-removed")).toHaveTextContent("-6") }) + describe("original content omitted by the extension", () => { + const TS = 1234 + + function createEditWithOriginal(payload: Record): ClineMessage { + return { + type: "ask", + ask: "tool", + ts: TS, + partial: false, + isAnswered: true, + text: JSON.stringify({ + tool: "appliedDiff", + path: "src/foo.ts", + diff: "the recorded diff", + ...payload, + }), + } + } + + function expandRow() { + fireEvent.click(screen.getByText("1 file(s) changed in this conversation").closest("button")!) + fireEvent.click(screen.getByTestId("accordian-toggle")) + } + + function respond(message: Record) { + act(() => { + window.dispatchEvent(new MessageEvent("message", { data: message })) + }) + } + + const requestsOfType = (type: string) => + mockPostMessage.mock.calls.map(([m]) => m).filter((m: { type: string }) => m.type === type) + + it("requests nothing until a row is expanded, then asks for the final and the original content", () => { + renderPanel([createEditWithOriginal({ originalContentLength: 5000 })]) + fireEvent.click(screen.getByText("1 file(s) changed in this conversation").closest("button")!) + + expect(mockPostMessage).not.toHaveBeenCalled() + + fireEvent.click(screen.getByTestId("accordian-toggle")) + + expect(requestsOfType("readFileContent")).toEqual([{ type: "readFileContent", text: "src/foo.ts" }]) + expect(requestsOfType("readOriginalContent")).toEqual([ + { type: "readOriginalContent", messageTs: TS, messageId: undefined, taskId: undefined }, + ]) + }) + + it("shows the merged diff once both the original and the final content arrive", () => { + renderPanel([createEditWithOriginal({ originalContentLength: 5000 })]) + expandRow() + + expect(screen.getByTestId("accordian-code")).toHaveTextContent("the recorded diff") + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + expect(screen.getByTestId("accordian-code")).toHaveTextContent("the recorded diff") + + respond({ type: "originalContent", originalContentInfo: { ts: TS, content: "old line\n" } }) + expect(screen.getByTestId("accordian-code")).toHaveTextContent("-old line") + expect(screen.getByTestId("accordian-code")).toHaveTextContent("+new line") + }) + + it("keeps the recorded diff when the original cannot be loaded", () => { + renderPanel([createEditWithOriginal({ originalContentLength: 5000 })]) + expandRow() + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + respond({ type: "originalContent", originalContentInfo: { ts: TS, content: null } }) + + expect(screen.getByTestId("accordian-code")).toHaveTextContent("the recorded diff") + expect(requestsOfType("readOriginalContent")).toHaveLength(1) + }) + + it("does not request the original when it is already inline", () => { + renderPanel([createEditWithOriginal({ originalContent: "old line\n" })]) + expandRow() + + expect(requestsOfType("readOriginalContent")).toHaveLength(0) + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + expect(screen.getByTestId("accordian-code")).toHaveTextContent("-old line") + expect(screen.getByTestId("accordian-code")).toHaveTextContent("+new line") + }) + + it("requests nothing for an edit that has no original", () => { + renderPanel([createEditWithOriginal({})]) + expandRow() + + expect(mockPostMessage).not.toHaveBeenCalled() + }) + + it("requests and matches originals by messageId when messages share a ts", () => { + const edit = (path: string, messageId: string): ClineMessage => ({ + ...createEditWithOriginal({ path, originalContentLength: 5000 }), + messageId, + }) + renderPanel([edit("src/a.ts", "id-a"), edit("src/b.ts", "id-b")], "task-1") + fireEvent.click(screen.getByText("2 file(s) changed in this conversation").closest("button")!) + screen.getAllByTestId("accordian-toggle").forEach((toggle) => fireEvent.click(toggle)) + + expect(requestsOfType("readOriginalContent")).toEqual([ + { type: "readOriginalContent", messageTs: TS, messageId: "id-a", taskId: "task-1" }, + { type: "readOriginalContent", messageTs: TS, messageId: "id-b", taskId: "task-1" }, + ]) + + respond({ type: "fileContent", fileContent: { path: "src/a.ts", content: "new a\n" } }) + respond({ type: "fileContent", fileContent: { path: "src/b.ts", content: "new b\n" } }) + respond({ + type: "originalContent", + originalContentInfo: { ts: TS, messageId: "id-b", taskId: "task-1", content: "old b\n" }, + }) + + const [a, b] = screen.getAllByTestId("accordian-code") + expect(a).toHaveTextContent("the recorded diff") + expect(b).toHaveTextContent("-old b") + }) + + it("requests the original again when the task id arrives after a request is already pending", () => { + const messages = [createEditWithOriginal({ originalContentLength: 5000 })] + const { rerender } = renderPanel(messages) + expandRow() + expect(requestsOfType("readOriginalContent")).toHaveLength(1) + + rerender( + + + , + ) + fireEvent.click(screen.getByTestId("accordian-toggle")) + + const requests = requestsOfType("readOriginalContent") + expect(requests).toHaveLength(2) + expect(requests[1]).toMatchObject({ taskId: "task-1" }) + }) + + it("does not send a duplicate request when switching A -> B -> A before the first response arrives", () => { + const messages = [createEditWithOriginal({ originalContentLength: 5000 })] + const panel = (taskId: string) => ( + + + + ) + const { rerender } = renderPanel(messages, "task-A") + expandRow() + expect(requestsOfType("readOriginalContent")).toHaveLength(1) + + rerender(panel("task-B")) + rerender(panel("task-A")) + fireEvent.click(screen.getByTestId("accordian-toggle")) + + expect(requestsOfType("readOriginalContent").filter((m) => m.taskId === "task-A")).toHaveLength(1) + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + respond({ + type: "originalContent", + originalContentInfo: { ts: TS, taskId: "task-A", content: "old line\n" }, + }) + expect(screen.getByTestId("accordian-code")).toHaveTextContent("-old line") + }) + + it("requests again after a response that arrived for a task that is no longer current", () => { + const messages = [createEditWithOriginal({ originalContentLength: 5000 })] + const panel = (taskId: string) => ( + + + + ) + const { rerender } = renderPanel(messages, "task-A") + expandRow() + + rerender(panel("task-B")) + respond({ + type: "originalContent", + originalContentInfo: { ts: TS, taskId: "task-A", content: "old line\n" }, + }) + rerender(panel("task-A")) + fireEvent.click(screen.getByTestId("accordian-toggle")) + + expect(requestsOfType("readOriginalContent").filter((m) => m.taskId === "task-A")).toHaveLength(2) + }) + + it("does not request for rows expanded under the previous task when the task id changes", () => { + const messages = [createEditWithOriginal({ originalContentLength: 5000 })] + const { rerender } = renderPanel(messages, "task-A") + expandRow() + mockPostMessage.mockClear() + + rerender( + + + , + ) + + expect(mockPostMessage).not.toHaveBeenCalled() + }) + + it("keeps a loaded original when the messages are replaced by an update of the same task", () => { + const first = [createEditWithOriginal({ originalContentLength: 5000 })] + const { rerender } = renderPanel(first, "task-A") + expandRow() + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + respond({ + type: "originalContent", + originalContentInfo: { ts: TS, taskId: "task-A", content: "old line\n" }, + }) + expect(requestsOfType("readOriginalContent")).toHaveLength(1) + + rerender( + + + , + ) + fireEvent.click(screen.getByTestId("accordian-toggle")) + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + + expect(requestsOfType("readOriginalContent")).toHaveLength(1) + expect(screen.getByTestId("accordian-code")).toHaveTextContent("-old line") + }) + + it("ignores an original answered for a different task", () => { + renderPanel([createEditWithOriginal({ originalContentLength: 5000 })], "task-1") + expandRow() + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + respond({ + type: "originalContent", + originalContentInfo: { ts: TS, taskId: "task-2", content: "old line\n" }, + }) + + expect(screen.getByTestId("accordian-code")).toHaveTextContent("the recorded diff") + }) + + it("ignores an original for a different message", () => { + renderPanel([createEditWithOriginal({ originalContentLength: 5000 })]) + expandRow() + + respond({ type: "fileContent", fileContent: { path: "src/foo.ts", content: "new line\n" } }) + respond({ type: "originalContent", originalContentInfo: { ts: TS + 1, content: "old line\n" } }) + + expect(screen.getByTestId("accordian-code")).toHaveTextContent("the recorded diff") + }) + }) }) diff --git a/webview-ui/src/__tests__/fileChangesFromMessages.spec.ts b/webview-ui/src/__tests__/fileChangesFromMessages.spec.ts index 8fab8b14d5..b09f4978c4 100644 --- a/webview-ui/src/__tests__/fileChangesFromMessages.spec.ts +++ b/webview-ui/src/__tests__/fileChangesFromMessages.spec.ts @@ -105,6 +105,8 @@ describe("fileChangesFromMessages", () => { path: "src/foo.ts", diff: "@@ -1 +1 @@\n+line", diffStats: { added: 1, removed: 0 }, + ts: messages[0].ts, + hasOriginalContent: false, }) }) @@ -190,7 +192,7 @@ describe("fileChangesFromMessages", () => { ] const result = fileChangesFromMessages(messages) expect(result).toHaveLength(2) - expect(result[0]).toEqual({ path: "a.ts", diff: "content a" }) + expect(result[0]).toEqual({ path: "a.ts", diff: "content a", ts: messages[0].ts, hasOriginalContent: false }) expect(result[1].path).toBe("b.ts") expect(result[1].diff).toBe("content b") }) @@ -277,4 +279,44 @@ describe("fileChangesFromMessages", () => { ] expect(fileChangesFromMessages(messages)).toEqual([]) }) + describe("original content (omitted by the extension to keep the webview small)", () => { + const editMessage = (payload: Record) => + msg({ + type: "ask", + ask: "tool", + isAnswered: true, + text: JSON.stringify({ tool: "appliedDiff", path: "src/foo.ts", diff: "d", ...payload }), + }) + + it("keeps an inline originalContent and marks the entry as having one", () => { + const message = editMessage({ originalContent: "old file" }) + const [entry] = fileChangesFromMessages([message]) + + expect(entry.originalContent).toBe("old file") + expect(entry.hasOriginalContent).toBe(true) + expect(entry.ts).toBe(message.ts) + }) + + it("marks an entry whose originalContent was omitted (originalContentLength) as requestable", () => { + const message = editMessage({ originalContentLength: 4096 }) + const [entry] = fileChangesFromMessages([message]) + + expect(entry.originalContent).toBeUndefined() + expect(entry.hasOriginalContent).toBe(true) + expect(entry.ts).toBe(message.ts) + }) + + it("treats an empty inline originalContent (a new file) as an original", () => { + const [entry] = fileChangesFromMessages([editMessage({ originalContent: "" })]) + + expect(entry.originalContent).toBe("") + expect(entry.hasOriginalContent).toBe(true) + }) + + it("marks an entry without any original as not having one", () => { + const [entry] = fileChangesFromMessages([editMessage({})]) + + expect(entry.hasOriginalContent).toBe(false) + }) + }) }) diff --git a/webview-ui/src/components/chat/ChatView.tsx b/webview-ui/src/components/chat/ChatView.tsx index 4694eeabf7..c4c4369ad1 100644 --- a/webview-ui/src/components/chat/ChatView.tsx +++ b/webview-ui/src/components/chat/ChatView.tsx @@ -1738,7 +1738,7 @@ const ChatViewComponent: React.ForwardRefRenderFunction
- + {areButtonsVisible && (
{ +// `ts` is not unique, so the message's own id identifies it; `ts` only covers messages persisted without one. +const originalKey = (entry: { messageId?: string; ts: number }) => entry.messageId ?? `ts:${entry.ts}` + +const FileChangesPanel = memo(({ clineMessages, taskId, className }: FileChangesPanelProps) => { const { t } = useTranslation() const [panelExpanded, setPanelExpanded] = useState(false) const [expandedPaths, setExpandedPaths] = useState>(new Set()) const [finalContentByPath, setFinalContentByPath] = useState>({}) const pendingPathsRef = useRef>(new Set()) + // The extension omits `originalContent` from the messages it posts; it is requested when a row is expanded. + const [originalContentByKey, setOriginalContentByKey] = useState>({}) + // In-flight requests, keyed by task and message. Deliberately not reset with the caches below: a request cannot + // be cancelled, so a reset would let a task switch (A -> B -> A) send a duplicate while the first is still open. + const pendingOriginalRequestsRef = useRef>(new Set()) + + // Task the expanded rows belong to; the reset below only lands after the render in which `taskId` changed, so + // the request effect must not act on rows expanded under the previous task. + const expandedTaskIdRef = useRef(taskId) - // Reset expanded file rows and final content cache when switching to a different task + // Reset expanded file rows and final content cache when the messages change useEffect(() => { setExpandedPaths(new Set()) setFinalContentByPath({}) pendingPathsRef.current = new Set() - }, [clineMessages]) + }, [clineMessages, taskId]) + + // Originals are keyed by message id, which is stable across message updates, so they only reset per task + useEffect(() => { + setOriginalContentByKey({}) + }, [taskId]) const fileChanges = useMemo(() => fileChangesFromMessages(clineMessages), [clineMessages]) @@ -56,34 +74,54 @@ const FileChangesPanel = memo(({ clineMessages, className }: FileChangesPanelPro ) }, [fileChanges]) - const togglePath = useCallback((path: string) => { - setExpandedPaths((prev) => { - const next = new Set(prev) - if (next.has(path)) next.delete(path) - else next.add(path) - return next - }) - }, []) + const togglePath = useCallback( + (path: string) => { + expandedTaskIdRef.current = taskId + setExpandedPaths((prev) => { + const next = new Set(prev) + if (next.has(path)) next.delete(path) + else next.add(path) + return next + }) + }, + [taskId], + ) - // Request final file content when a row is expanded and we have originalContent + // Request the final file content (and the omitted original content) when a row is expanded and the edit has an original useEffect(() => { + if (expandedTaskIdRef.current !== taskId) return for (const path of expandedPaths) { const entries = byPath.get(path) if (!entries?.length) continue - const originalContent = entries[0].originalContent + const first = entries[0] const lookupPath = path.startsWith("./") ? path.slice(2) : path if ( - originalContent !== undefined && + first.hasOriginalContent && !(lookupPath in finalContentByPath) && !pendingPathsRef.current.has(lookupPath) ) { pendingPathsRef.current.add(lookupPath) vscode.postMessage({ type: "readFileContent", text: lookupPath }) } + const requestKey = `${taskId ?? ""}|${originalKey(first)}` + if ( + first.hasOriginalContent && + first.originalContent === undefined && + !(originalKey(first) in originalContentByKey) && + !pendingOriginalRequestsRef.current.has(requestKey) + ) { + pendingOriginalRequestsRef.current.add(requestKey) + vscode.postMessage({ + type: "readOriginalContent", + messageTs: first.ts, + messageId: first.messageId, + taskId, + }) + } } - }, [expandedPaths, byPath, finalContentByPath]) + }, [expandedPaths, byPath, finalContentByPath, originalContentByKey, taskId]) - // Listen for fileContent responses + // Listen for fileContent and originalContent responses useEffect(() => { const handler = (event: MessageEvent) => { const message: ExtensionMessage = event.data @@ -91,11 +129,18 @@ const FileChangesPanel = memo(({ clineMessages, className }: FileChangesPanelPro const fc = message.fileContent pendingPathsRef.current.delete(fc.path) setFinalContentByPath((prev) => ({ ...prev, [fc.path]: fc.content ?? null })) + } else if (message.type === "originalContent" && message.originalContentInfo) { + const { taskId: responseTaskId, content, ...id } = message.originalContentInfo + const key = originalKey(id) + pendingOriginalRequestsRef.current.delete(`${responseTaskId ?? ""}|${key}`) + // A late response for another task must not populate this task's cache. + if (responseTaskId !== taskId) return + setOriginalContentByKey((prev) => ({ ...prev, [key]: content })) } } window.addEventListener("message", handler) return () => window.removeEventListener("message", handler) - }, []) + }, [taskId]) if (fileChanges.length === 0) return null @@ -133,7 +178,8 @@ const FileChangesPanel = memo(({ clineMessages, className }: FileChangesPanelPro
{Array.from(byPath.entries()).map(([path, entries]) => { - const originalContent = entries[0].originalContent + const originalContent = + entries[0].originalContent ?? originalContentByKey[originalKey(entries[0])] ?? undefined const lookupPath = path.startsWith("./") ? path.slice(2) : path const finalContent = finalContentByPath[lookupPath] const hasMergedDiff = diff --git a/webview-ui/src/components/chat/utils/fileChangesFromMessages.ts b/webview-ui/src/components/chat/utils/fileChangesFromMessages.ts index 738305ad15..35c55e496d 100644 --- a/webview-ui/src/components/chat/utils/fileChangesFromMessages.ts +++ b/webview-ui/src/components/chat/utils/fileChangesFromMessages.ts @@ -10,6 +10,11 @@ export interface FileChangeEntry { diffStats?: { added: number; removed: number } /** Original file content before first edit (for merged diff display) */ originalContent?: string + /** Identity of the message this entry comes from; used to request `originalContent` when the extension omitted it */ + ts: number + messageId?: string + /** True when the edit has an original (inline, or omitted by the extension and requestable by `ts`) */ + hasOriginalContent: boolean } /** @@ -44,6 +49,9 @@ export function fileChangesFromMessages(messages: ClineMessage[] | undefined): F path: file.path, diff: content, diffStats: file.diffStats, + ts: msg.ts, + messageId: msg.messageId, + hasOriginalContent: false, }) } } @@ -59,6 +67,9 @@ export function fileChangesFromMessages(messages: ClineMessage[] | undefined): F diff, diffStats: tool.diffStats, originalContent: tool.originalContent, + ts: msg.ts, + messageId: msg.messageId, + hasOriginalContent: tool.originalContent !== undefined || (tool.originalContentLength ?? 0) > 0, }) } }