diff --git a/icon128.png b/icon128.png new file mode 100644 index 00000000..f3beebf7 Binary files /dev/null and b/icon128.png differ diff --git a/new-logo.png b/new-logo.png new file mode 100644 index 00000000..ed937f7e Binary files /dev/null and b/new-logo.png differ diff --git a/runners/extension/chatLocator.js b/runners/extension/chatLocator.js index b24b4623..22b26680 100644 --- a/runners/extension/chatLocator.js +++ b/runners/extension/chatLocator.js @@ -30,6 +30,38 @@ export async function locateChatWidget(tabId, readerCfg, options = {}) { const clickedLaunchers = []; let lastErr = ""; + // Deterministic fast-path: if the in-frame heuristic already surfaced a + // high-confidence, non-search chat input, use it directly and skip the + // (non-deterministic) LLM planner — this removes most run-to-run variance in + // input detection. Falls through to the planner on any miss (e.g. widget not + // open yet, or only ambiguous/low-score inputs present). + try { + const detFrames = await collectFrames(tabId); + const ranked = detFrames + .filter((f) => f.bestInputSelector && (f.bestInputScore || 0) >= 12) + .sort((a, b) => (b.bestInputScore || 0) - (a.bestInputScore || 0)); + for (const f of ranked) { + if (state.OPFOR_STOP) return { ok: false, error: "Run stopped." }; + const visible = await actVerifyInputVisible(tabId, f.frameId, f.bestInputSelector); + if (!visible) continue; + const siteSnapshot = detFrames.find((x) => x.frameId === 0)?.snapshot || f.snapshot || ""; + return { + ok: true, + plan: { + inputSelector: f.bestInputSelector, + submit: f.bestSendSelector + ? { method: "click", buttonSelector: f.bestSendSelector } + : { method: "enter" }, + confidence: 0.9, + }, + best: { frameId: f.frameId, frameUrl: f.frameUrl || "", snapshot: siteSnapshot }, + siteSnapshot, + }; + } + } catch { + /* fall through to the LLM planner */ + } + const collectAxSnapshots = async () => { let results; try { diff --git a/runners/extension/frameDiscovery.js b/runners/extension/frameDiscovery.js index 8fd80ce8..da3fb6b7 100644 --- a/runners/extension/frameDiscovery.js +++ b/runners/extension/frameDiscovery.js @@ -13,6 +13,9 @@ export async function collectFrames(tabId) { frameUrl: r.result?.frameUrl, inputCount: r.result?.inputCount ?? 0, chatScore: r.result?.chatScore ?? 0, + bestInputSelector: r.result?.bestInputSelector ?? "", + bestInputScore: r.result?.bestInputScore ?? 0, + bestSendSelector: r.result?.bestSendSelector ?? "", })) .filter((f) => typeof f.snapshot === "string" && f.snapshot.length > 0); } @@ -57,11 +60,7 @@ export async function pickChatFrame(tabId) { const frames = await collectFrames(tabId); if (!frames.length) throw new Error("No frames collected."); - frames.sort((a, b) => { - const boost = embeddedChatBoost(b) - embeddedChatBoost(a); - if (boost !== 0) return boost; - return (b.chatScore || 0) - (a.chatScore || 0) || (b.inputCount || 0) - (a.inputCount || 0); - }); + frames.sort(sortByChatRelevance); return { frames, best: frames[0] }; } @@ -74,14 +73,25 @@ export async function pickChatFrameWithRetry(tabId, { maxRetries = 6, intervalMs for (let i = 0; i < maxRetries; i++) { await sleep(i === 0 ? 600 : intervalMs); frames = await collectFrames(tabId); - if (frames.some((f) => (f.chatScore || 0) > 0)) break; + // Wait for the chat INPUT itself, not merely a chat signal. chatScore can go + // positive from container markup while the composer is still mounting (open + // animation, lazy render, input shown only after a greeting) — breaking on + // chatScore alone snapshots too early and misses the input. + if (frames.some((f) => (f.bestInputScore || 0) > 0)) break; } if (!frames.length) throw new Error("No frames collected."); - frames.sort((a, b) => { - const boost = embeddedChatBoost(b) - embeddedChatBoost(a); - if (boost !== 0) return boost; - return (b.chatScore || 0) - (a.chatScore || 0) || (b.inputCount || 0) - (a.inputCount || 0); - }); + frames.sort(sortByChatRelevance); return { frames, best: frames[0] }; } + +// Shared ranking: embedded-chat URL boost, then a real input, then chat signal. +function sortByChatRelevance(a, b) { + const boost = embeddedChatBoost(b) - embeddedChatBoost(a); + if (boost !== 0) return boost; + return ( + (b.bestInputScore || 0) - (a.bestInputScore || 0) || + (b.chatScore || 0) - (a.chatScore || 0) || + (b.inputCount || 0) - (a.inputCount || 0) + ); +} diff --git a/runners/extension/frame_collect.js b/runners/extension/frame_collect.js index 8a95eee0..084fbe82 100644 --- a/runners/extension/frame_collect.js +++ b/runners/extension/frame_collect.js @@ -12,6 +12,17 @@ function escapeCssValue(v) { return String(v).replace(/["\\]/g, "\\$&"); } +function nthOfTypeSelector(el) { + const tag = el.tagName.toLowerCase(); + const parent = el.parentElement || el.getRootNode?.()?.host?.shadowRoot; + const siblings = parent ? Array.from(parent.children || []) : []; + const sameTag = siblings.filter((c) => c.tagName === el.tagName); + if (sameTag.length <= 1) return tag; + const idx = sameTag.indexOf(el); + // :nth-of-type is 1-based among same-type element siblings — equals our index+1. + return idx >= 0 ? `${tag}:nth-of-type(${idx + 1})` : tag; +} + function selectorFromEl(el) { if (!(el instanceof Element)) return null; const testid = el.getAttribute("data-testid"); @@ -19,10 +30,13 @@ function selectorFromEl(el) { const aria = el.getAttribute("aria-label"); if (aria) return `${el.tagName.toLowerCase()}[aria-label="${escapeCssValue(aria)}"]`; const id = el.getAttribute("id"); - if (id) return `#${escapeCssValue(id)}`; + // Skip dynamic ids (e.g. "radix-:r3:") that aren't valid/stable CSS identifiers. + if (id && /^[A-Za-z][\w-]*$/.test(id)) return `#${escapeCssValue(id)}`; const name = el.getAttribute("name"); if (name) return `${el.tagName.toLowerCase()}[name="${escapeCssValue(name)}"]`; - return el.tagName.toLowerCase(); + // No stable attribute — fall back to a positional selector so we don't return a + // bare tag that matches (and later actuates) the wrong element. + return nthOfTypeSelector(el); } function getShadowRoot(el) { @@ -602,6 +616,15 @@ function collectSanitizedDomSnapshot() { return true; }); + // Track the single best non-search input deterministically, so callers can use + // the heuristic directly instead of relying on a (non-deterministic) LLM pick. + let bestInput = null; + for (const el of inputs) { + if (looksLikeSiteSearch(el)) continue; + const sc = scoreInput(el); + if (!bestInput || sc > bestInput.score) bestInput = { el, score: sc }; + } + lines.push(""); lines.push("CANDIDATE_INPUTS:"); for (const el of inputs.slice(0, 40)) { @@ -700,13 +723,16 @@ function collectSanitizedDomSnapshot() { lines.push(""); lines.push("CANDIDATE_BUTTONS:"); + let bestSendSelector = ""; for (const el of buttons) { const aria = (el.getAttribute?.("aria-label") || "").toLowerCase(); const text = (el.textContent || "").trim().toLowerCase(); const looksLikeSend = aria.includes("send") || text === "send" || text.includes("send") || text.includes("submit"); + const selStr = deepPathSelector(el); + if (looksLikeSend && !bestSendSelector) bestSendSelector = selStr; lines.push( - `- ${looksLikeSend ? "sendish=1" : "sendish=0"} selector="${deepPathSelector(el)}" ${describeEl(el)}` + `- ${looksLikeSend ? "sendish=1" : "sendish=0"} selector="${selStr}" ${describeEl(el)}` ); } @@ -744,6 +770,10 @@ function collectSanitizedDomSnapshot() { snapshot: lines.join("\n").slice(0, 60_000), inputCount: inputs.length, chatScore, + // Deterministic best chat input + send button for this frame (no LLM needed). + bestInputSelector: bestInput ? deepPathSelector(bestInput.el) : "", + bestInputScore: bestInput ? bestInput.score : 0, + bestSendSelector, }; } diff --git a/runners/extension/frame_snapshot.js b/runners/extension/frame_snapshot.js index 8ad5261b..0c6cd25a 100644 --- a/runners/extension/frame_snapshot.js +++ b/runners/extension/frame_snapshot.js @@ -21,8 +21,77 @@ "menuitem", "option", ]); - const MIN_MSG = 20; - const MAX_MSG = 1000; + // Lowered from 20 → 2 so terse bot replies ("Done.", "Order #4521 shipped.") + // are captured instead of being silently dropped. MAX raised to a large sanity + // bound because per-message granularity now comes from block-structure (see + // collectText), not an arbitrary character cap. + const MIN_MSG = 2; + const MAX_MSG = 20000; + + // Inline tags never count as block-level message containers — their text is + // part of the surrounding block, so we must NOT recurse past their parent + // (that would lose the plain-text siblings around , , etc.). + const INLINE_TAGS = new Set([ + "a", + "b", + "i", + "em", + "strong", + "span", + "code", + "small", + "sub", + "sup", + "mark", + "u", + "s", + "abbr", + "cite", + "q", + "time", + "label", + "kbd", + "samp", + "var", + "wbr", + "bdi", + "bdo", + "del", + "ins", + "br", + "img", + "svg", + "picture", + ]); + + function isBlockish(el) { + const tag = el.tagName?.toLowerCase(); + if (tag && INLINE_TAGS.has(tag)) return false; + try { + const d = window.getComputedStyle(el).display; + return !(d === "inline" || d === "inline-block" || d === "contents" || d === "none"); + } catch { + return true; // assume block when style is unavailable + } + } + + // Count block-level child elements that carry their own text. ≥1 means this node + // is a CONTAINER of sub-blocks (transcript, message list, message row) and we + // should recurse into it; 0 means it's a leaf text block (a single message / + // paragraph) safe to capture as one node. This is what stops a short transcript + // from collapsing into a single mega-node (which broke the diff + echo filter). + function blockTextChildren(node) { + let n = 0; + for (const c of node.children || []) { + if (!isBlockish(c)) continue; + const t = (c.textContent || "").replace(/\s+/g, " ").trim(); + if (t.length >= MIN_MSG) { + n++; + if (n >= 1) break; + } + } + return n; + } // ── Text collector (shared by both fast and full paths) ────────────────────── function collectText(node, depth, out) { @@ -46,7 +115,9 @@ const role = (node.getAttribute?.("role") || "").toLowerCase(); if (SKIP_ROLES.has(role)) return; const text = (node.textContent || "").replace(/\s+/g, " ").trim(); - if (text.length >= MIN_MSG && text.length <= MAX_MSG) { + // Capture only LEAF text blocks (no block-level child carries its own text). + // Containers fall through and recurse, so each message becomes its own node. + if (text.length >= MIN_MSG && text.length <= MAX_MSG && blockTextChildren(node) === 0) { out.push(text); return; } @@ -75,18 +146,38 @@ }; } - // ── Fast path: reuse selector from previous full scan ──────────────────────── + // ── Fast path: reuse the container from a previous full scan ───────────────── + // Prefer a direct reference to the element over re-running querySelector on a + // cached string. String selectors are brittle: hashed/CSS-module class names, + // dynamic ids (radix-:r3:, which is also invalid CSS and throws), and + // first-match ambiguity all make the string resolve to the wrong element or + // none. Holding the node itself is stable until the framework re-renders it. + const cachedEl = globalThis.__OPFOR_CONTAINER_EL__; + if (cachedEl && cachedEl.isConnected) { + try { + const result = snapshotEl(cachedEl, globalThis.__OPFOR_CONTAINER_SEL__ || ""); + if (result.nodeCount > 0) return result; + // Element still present but empty — fall through to full scan. + } catch { + /* swallowed */ + } + } + + // Secondary fast path: the cached selector string (used when the page-world + // element ref was lost, e.g. seeded by the orchestrator across calls). const cachedSel = globalThis.__OPFOR_CONTAINER_SEL__; if (cachedSel) { try { const el = document.querySelector(cachedSel); if (el) { const result = snapshotEl(el, cachedSel); - if (result.nodeCount > 0) return result; - // Container found but empty — fall through to full scan + if (result.nodeCount > 0) { + globalThis.__OPFOR_CONTAINER_EL__ = el; + return result; + } } } catch { - /* swallowed */ + /* swallowed (invalid/dynamic selector) — fall through to full scan */ } } @@ -216,14 +307,18 @@ const best = candidates[0].el; - // Build selector and cache it for fast subsequent polls + // Cache the element itself for fast subsequent polls (stable across re-renders + // that only change attributes). Keep a best-effort selector string as backup. let sel = best.tagName.toLowerCase(); - if (best.id) { + if (best.id && /^[A-Za-z][\w-]*$/.test(best.id)) { + // Only use #id when it's a static, valid CSS identifier (skip dynamic ids + // like "radix-:r3:" / ":r0:" that change per render and break querySelector). sel = "#" + best.id; } else if (typeof best.className === "string" && best.className.trim()) { const parts = best.className.trim().split(/\s+/).slice(0, 2); sel = best.tagName.toLowerCase() + "." + parts.join("."); } + globalThis.__OPFOR_CONTAINER_EL__ = best; globalThis.__OPFOR_CONTAINER_SEL__ = sel; const result = snapshotEl(best, sel); diff --git a/runners/extension/icons/icon128.png b/runners/extension/icons/icon128.png index 68e391f5..4f4def9f 100644 Binary files a/runners/extension/icons/icon128.png and b/runners/extension/icons/icon128.png differ diff --git a/runners/extension/icons/icon16.png b/runners/extension/icons/icon16.png index 68e391f5..0f3bb0c9 100644 Binary files a/runners/extension/icons/icon16.png and b/runners/extension/icons/icon16.png differ diff --git a/runners/extension/icons/icon32.png b/runners/extension/icons/icon32.png index 68e391f5..7d4a103b 100644 Binary files a/runners/extension/icons/icon32.png and b/runners/extension/icons/icon32.png differ diff --git a/runners/extension/icons/icon48.png b/runners/extension/icons/icon48.png index 68e391f5..746ab09e 100644 Binary files a/runners/extension/icons/icon48.png and b/runners/extension/icons/icon48.png differ diff --git a/runners/extension/icons/logo.png b/runners/extension/icons/logo.png index 8d42ab83..a5d675f3 100644 Binary files a/runners/extension/icons/logo.png and b/runners/extension/icons/logo.png differ diff --git a/runners/extension/popup.html b/runners/extension/popup.html index 1f139ce8..8b1da394 100644 --- a/runners/extension/popup.html +++ b/runners/extension/popup.html @@ -61,9 +61,9 @@ --muted: #9ba3b8; --muted-2: rgb(106, 116, 145); --muted-3: #5a6478; - --accent: #f5ad5c; - --accent-soft: rgba(245, 173, 92, 0.12); - --accent-ring: rgba(245, 173, 92, 0.22); + --accent: #ff4d4f; + --accent-soft: rgba(255, 77, 79, 0.12); + --accent-ring: rgba(255, 77, 79, 0.22); --pass: #34d399; --warn: #fbbf24; --fail: #ff5a6e; @@ -205,7 +205,7 @@ background: var(--line-3); } ::selection { - background: rgba(245, 173, 92, 0.3); + background: rgba(255, 77, 79, 0.3); color: #fff; } ::placeholder { @@ -295,9 +295,9 @@ background: currentColor; } .status-pill[data-screen="running"] { - color: #f5ad5c; - background: rgba(245, 173, 92, 0.08); - border-color: rgba(245, 173, 92, 0.25); + color: #ff4d4f; + background: rgba(255, 77, 79, 0.08); + border-color: rgba(255, 77, 79, 0.25); } .status-pill[data-screen="running"] .dot { box-shadow: 0 0 6px currentColor; @@ -980,7 +980,7 @@ gap: 8px; letter-spacing: 0.01em; box-shadow: - 0 4px 14px rgba(245, 173, 92, 0.25), + 0 4px 14px rgba(255, 77, 79, 0.25), inset 0 1px 0 rgba(255, 255, 255, 0.2); transition: transform 0.08s, @@ -1149,7 +1149,7 @@ .turn-fill { position: absolute; inset: 0; - background: linear-gradient(90deg, rgba(245, 173, 92, 0.5), var(--accent)); + background: linear-gradient(90deg, rgba(255, 77, 79, 0.5), var(--accent)); transition: width 0.4s; box-shadow: 0 0 12px var(--accent); width: 0%; @@ -1212,8 +1212,8 @@ word-break: break-word; } .bubble[data-who="attacker"] .body { - background: rgba(245, 173, 92, 0.08); - border-color: rgba(245, 173, 92, 0.25); + background: rgba(255, 77, 79, 0.08); + border-color: rgba(255, 77, 79, 0.25); color: var(--text); } .bubble .body[data-pending="true"]::after { @@ -1439,7 +1439,7 @@ width: 80px; height: 80px; border-radius: 50%; - background: rgba(245, 173, 92, 0.13); + background: rgba(255, 77, 79, 0.13); filter: blur(30px); } .verdict-card[data-verdict="PASS"] .glow { @@ -1455,8 +1455,8 @@ width: 38px; height: 38px; border-radius: 10px; - background: rgba(245, 173, 92, 0.13); - border: 1px solid rgba(245, 173, 92, 0.33); + background: rgba(255, 77, 79, 0.13); + border: 1px solid rgba(255, 77, 79, 0.33); color: var(--accent); display: flex; align-items: center; @@ -2012,7 +2012,12 @@
- OPFOR + Agent OPFOR
diff --git a/runners/extension/popup.js b/runners/extension/popup.js index a02627f7..0a4fe450 100644 --- a/runners/extension/popup.js +++ b/runners/extension/popup.js @@ -1321,7 +1321,7 @@ function generateHtmlReport(report) { --line:#E2E8F0;--line-2:#CBD5E1; --pass:#059669;--pass-bg:#D1FAE5;--pass-border:#6EE7B7; --fail:#DC2626;--fail-bg:#FEE2E2;--fail-border:#FCA5A5; - --accent:#f5ad5c; + --accent:#FF4D4F; } *{box-sizing:border-box;margin:0;padding:0} html{background:var(--bg)} @@ -1337,7 +1337,7 @@ function generateHtmlReport(report) { .cover-inner{max-width:960px;margin:0 auto;padding:36px 24px 32px} .cover-top{display:flex;align-items:flex-start;justify-content:space-between;gap:24px;margin-bottom:28px} .cover-brand{display:flex;align-items:center;gap:10px} - .cover-brand-icon{width:36px;height:36px;background:linear-gradient(135deg,#f5ad5c,#c47a2a);border-radius:8px;display:flex;align-items:center;justify-content:center;flex-shrink:0} + .cover-brand-icon{width:36px;height:36px;background:linear-gradient(135deg,#FF4D4F,#c4302a);border-radius:8px;display:flex;align-items:center;justify-content:center;flex-shrink:0} .cover-brand-name{font-size:15px;font-weight:700;letter-spacing:0.04em;color:#fff} .cover-brand-sub{font-size:11px;color:#94A3B8;letter-spacing:0.08em;text-transform:uppercase;margin-top:1px} .cover-classification{padding:4px 12px;border:1px solid rgba(255,255,255,0.15);border-radius:4px;font-size:11px;font-weight:600;letter-spacing:0.1em;text-transform:uppercase;color:#CBD5E1} diff --git a/runners/extension/responseExtractor.js b/runners/extension/responseExtractor.js index ebb3e5ca..51e556ed 100644 --- a/runners/extension/responseExtractor.js +++ b/runners/extension/responseExtractor.js @@ -28,16 +28,20 @@ function diffTextNodes(pre, post) { let i = 0; while (i < preNorm.length && i < postNorm.length && preNorm[i] === postNorm[i]) i++; - const candidates = post.slice(i); - const preNormSet = new Set(preNorm); - const filtered = candidates.filter((t) => !preNormSet.has(stripLeadingTimestamp(t))); - const result = filtered.length > 0 ? filtered : candidates; - - if (result.length > post.length * 0.8 && post.length > 5) { - const fullFiltered = post.filter((t) => !preNormSet.has(stripLeadingTimestamp(t))); - return { text: fullFiltered.join("\n") }; + // Clean append: the common prefix covered every prior node, so the new nodes + // are exactly the appended tail. Use this POSITIONAL diff directly — do NOT + // value-filter against the whole prior transcript, or a reply that repeats an + // earlier message verbatim (e.g. the same canned answer on two turns) gets + // deduped away, leaving only the user echo → "could not extract". + if (i >= preNorm.length) { + return { text: post.slice(i).join("\n") }; } - return { text: result.join("\n") }; + + // Prefix diverged (full re-render / reordered / virtualized transcript): fall + // back to a value diff — surface post nodes whose text isn't present in pre. + const preSet = new Set(preNorm); + const valueNew = post.filter((t) => !preSet.has(stripLeadingTimestamp(t))); + return { text: (valueNew.length ? valueNew : post.slice(i)).join("\n") }; } // ── Scan all frames, return the best container snapshot ─────────────────────── @@ -58,24 +62,47 @@ async function scanBestFrame(tabId) { return iframeHits.length > 0 ? iframeHits[0] : hits[0] || null; } +// Snapshot one specific frame (the frame the message is actually sent into). +async function snapshotFrame(tabId, frameId) { + const results = await chrome.scripting.executeScript({ + target: { tabId, frameIds: [frameId] }, + files: ["frame_snapshot.js"], + }); + const r = results?.[0]; + return r?.result ? { frameId: r.frameId, ...r.result } : null; +} + // ── Public API ──────────────────────────────────────────────────────────────── /** * Take a pre-send snapshot of the chat container. * Returns an object suitable for passing as prevSnapshot to extractResponse. */ -export async function snapshotCurrentResponse(tabId, _frameId) { +export async function snapshotCurrentResponse(tabId, frameId) { + const empty = (containerFrameId) => ({ + text: "", + messageCount: 0, + botCount: 0, + textNodes: [], + nodeCount: 0, + containerFrameId, + }); try { - const snap = await scanBestFrame(tabId); - if (!snap) - return { - text: "", - messageCount: 0, - botCount: 0, - textNodes: [], - nodeCount: 0, - containerFrameId: null, - }; + // Anchor the response read to the SAME frame the message is sent into. The + // response always appears in the send frame's transcript, so reading from a + // different (higher-"scoring") iframe is the main cause of missed replies. + // Only when no send frame is known do we fall back to scanning all frames. + let snap = null; + const hasFrame = frameId !== null && frameId !== undefined; + if (hasFrame) { + snap = await snapshotFrame(tabId, frameId); + // Send frame has no chat container yet (e.g. empty transcript on turn 1): + // still anchor reads to it — polling picks up the container once a reply renders. + if (!snap?.ok) return empty(frameId); + } else { + snap = await scanBestFrame(tabId); + if (!snap) return empty(null); + } return { text: snap.fullText || "", messageCount: snap.nodeCount, @@ -84,21 +111,18 @@ export async function snapshotCurrentResponse(tabId, _frameId) { nodeCount: snap.nodeCount, fullText: snap.fullText, lastNodeText: snap.lastNodeText, - containerFrameId: snap.frameId, + containerFrameId: snap.frameId ?? (hasFrame ? frameId : null), containerSel: snap.sel, }; } catch { - return { - text: "", - messageCount: 0, - botCount: 0, - textNodes: [], - nodeCount: 0, - containerFrameId: null, - }; + return empty(hasFrameIdSafe(frameId)); } } +function hasFrameIdSafe(frameId) { + return frameId === null || frameId === undefined ? null : frameId; +} + /** * Poll until a new, complete bot response appears, then return it via text-node diff. * @@ -269,7 +293,12 @@ export async function extractResponse(tabId, frameId, lastUserText = "", prevSna if (stableCount >= neededStable) { const { text: rawDiff } = diffTextNodes(baseTextNodes, snap.textNodes); const diffLines = rawDiff.split("\n").filter((l) => l.trim()); - const botLines = diffLines.filter((l) => !isUserEcho(l) && !isTypingIndicator(l)); + // Drop pure-timestamp lines (e.g. message-bubble "3:45 PM" footers) that now + // surface as their own nodes — they strip to empty and aren't real reply text. + const isTimestampOnly = (l) => stripLeadingTimestamp(l).trim() === ""; + const botLines = diffLines.filter( + (l) => !isUserEcho(l) && !isTypingIndicator(l) && !isTimestampOnly(l) + ); if (botLines.length > 0) { return {