From b0170d4066d8d9ae2df1c400c116d5f710403495 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 21:13:45 -0400 Subject: [PATCH 01/24] Tell search, storefront and seller pages apart from product, cart and account pages --- scripts/lib/page-kind.js | 115 +++++++++++++++++++++++++++++++++++ tests/unit/page-kind.test.js | 72 ++++++++++++++++++++++ 2 files changed, 187 insertions(+) create mode 100644 scripts/lib/page-kind.js create mode 100644 tests/unit/page-kind.test.js diff --git a/scripts/lib/page-kind.js b/scripts/lib/page-kind.js new file mode 100644 index 0000000..ba551a7 --- /dev/null +++ b/scripts/lib/page-kind.js @@ -0,0 +1,115 @@ +/** + * @fileoverview What kind of Amazon page a URL is, for the on-page dock. + * + * The dock offers Scrape only where a run can start (search results and + * seller storefronts) and points seller pages at their storefront. It never + * offers anything on product, cart, checkout, sign-in or account pages. + * + * Kinds: + * - search: /s with a keyword or a category + * - storefront: /s?me=SELLER, the seller's product list + * - seller: /sp?seller=SELLER, the seller profile; its storefront is /s?me=SELLER + * - store: /stores/..., a brand store; a run cannot start there + * - never: product, cart, checkout, sign-in, account and order pages + * - other: any other Amazon page + * + * Loaded as a plain global (`PageKind`) in content scripts and the popup, + * and as a CommonJS module in Jest. + * + * @module PageKind + */ + +const PageKind = (() => { + // Paths where the dock must stay quiet, whatever the query says. + const NEVER = [ + /\/dp\//i, + /^\/gp\/product\//i, + /^\/gp\/aw\/d\//i, + /^\/gp\/offer-listing\//i, + /^\/gp\/aod\//i, + /^\/gp\/cart\//i, + /^\/cart(\/|$)/i, + /^\/gp\/buy\//i, + /^\/checkout(\/|$)/i, + /^\/gp\/checkout/i, + /^\/ap\//i, + /^\/ax\//i, + /^\/gp\/css\//i, + /^\/gp\/your-account/i, + /^\/your-account/i, + /^\/your-orders/i, + /^\/gp\/yourstore/i, + /^\/hz\//i, + /^\/a\/addresses/i, + /^\/cpe\/yourpayments/i, + /^\/mn\/dcw\/myx/i + ]; + + const SELLER_ID = /^[A-Z0-9]{8,20}$/i; + + function parse(url) { + try { + const u = new URL(url); + return /(^|\.)amazon\.com$/i.test(u.hostname) ? u : null; + } catch (e) { + return null; + } + } + + /** A seller id from a query value, or null when it does not look like one. */ + function sellerId(value) { + const v = String(value || '').trim(); + return SELLER_ID.test(v) ? v.toUpperCase() : null; + } + + /** The storefront URL for a seller id. */ + function storefrontUrl(id) { + return `https://www.amazon.com/s?me=${encodeURIComponent(id)}&marketplaceID=ATVPDKIKX0DER`; + } + + /** + * @param {string} url + * @returns {{kind: string, sellerId: ?string, keyword: ?string}} + */ + function classify(url) { + const u = parse(url); + const out = { kind: 'other', sellerId: null, keyword: null }; + if (!u) return { ...out, kind: 'never' }; + const path = u.pathname; + if (NEVER.some((re) => re.test(path))) return { ...out, kind: 'never' }; + + if (path === '/s' || path.startsWith('/s/')) { + const me = sellerId(u.searchParams.get('me')); + if (me) return { ...out, kind: 'storefront', sellerId: me }; + const k = u.searchParams.get('k') || u.searchParams.get('field-keywords'); + if (k || u.searchParams.get('rh') || u.searchParams.get('i') || u.searchParams.get('node')) { + return { ...out, kind: 'search', keyword: k ? k.trim() : null }; + } + return out; + } + if (path === '/sp' || path.startsWith('/sp/')) { + const id = sellerId(u.searchParams.get('seller')); + return id ? { ...out, kind: 'seller', sellerId: id } : out; + } + if (path.startsWith('/stores/')) { + return { ...out, kind: 'store', sellerId: sellerId(u.searchParams.get('seller') || u.searchParams.get('me')) }; + } + return out; + } + + /** True where the dock may suggest something: a scrape, or the way to one. */ + function suggests(url) { + return ['search', 'storefront', 'seller', 'store'].includes(classify(url).kind); + } + + /** True where a run can start from the page itself. */ + function scrapable(url) { + return ['search', 'storefront'].includes(classify(url).kind); + } + + return { classify, suggests, scrapable, storefrontUrl, sellerId }; +})(); + +if (typeof module !== 'undefined' && module.exports) { + module.exports = PageKind; +} diff --git a/tests/unit/page-kind.test.js b/tests/unit/page-kind.test.js new file mode 100644 index 0000000..6977239 --- /dev/null +++ b/tests/unit/page-kind.test.js @@ -0,0 +1,72 @@ +const PageKind = require('../../scripts/lib/page-kind'); + +const A = 'https://www.amazon.com'; + +describe('pages that get the Scrape suggestion', () => { + test.each([ + [`${A}/s?k=yoga+mat`, 'search', { keyword: 'yoga mat' }], + [`${A}/s?k=usb+c&page=3&ref=sr_pg_3`, 'search', { keyword: 'usb c' }], + [`${A}/s?rh=n%3A3407731&fs=true`, 'search', { keyword: null }], + [`${A}/s?me=A1B2C3D4E5F6G7&marketplaceID=ATVPDKIKX0DER`, 'storefront', { sellerId: 'A1B2C3D4E5F6G7' }], + [`${A}/s?me=a1b2c3d4e5f6g7&k=lamp`, 'storefront', { sellerId: 'A1B2C3D4E5F6G7' }], + [`https://smile.amazon.com/s?k=desk`, 'search', {}], + ])('%s is %s', (url, kind, extra) => { + const got = PageKind.classify(url); + expect(got).toMatchObject({ kind, ...extra }); + expect(PageKind.scrapable(url)).toBe(true); + expect(PageKind.suggests(url)).toBe(true); + }); + + test('a seller profile points at its storefront but cannot start a run itself', () => { + const url = `${A}/sp?ie=UTF8&seller=A1B2C3D4E5F6G7&isAmazonFulfilled=1`; + expect(PageKind.classify(url)).toMatchObject({ kind: 'seller', sellerId: 'A1B2C3D4E5F6G7' }); + expect(PageKind.suggests(url)).toBe(true); + expect(PageKind.scrapable(url)).toBe(false); + expect(PageKind.storefrontUrl('A1B2C3D4E5F6G7')).toBe(`${A}/s?me=A1B2C3D4E5F6G7&marketplaceID=ATVPDKIKX0DER`); + }); + + test('a brand store is a store page, not a scrapable one', () => { + const url = `${A}/stores/Gaiam/page/4F1C1B7A-6C0B-4D0B-9E1F-3B7B2C8D9E10`; + expect(PageKind.classify(url)).toMatchObject({ kind: 'store', sellerId: null }); + expect(PageKind.scrapable(url)).toBe(false); + }); +}); + +describe('pages that never get it', () => { + test.each([ + `${A}/dp/B09B8V1LZ3`, + `${A}/Gaiam-Thick-Yoga-Mat/dp/B09B8V1LZ3/ref=sr_1_1?k=yoga+mat`, + `${A}/gp/product/B09B8V1LZ3`, + `${A}/gp/aw/d/B09B8V1LZ3`, + `${A}/gp/offer-listing/B09B8V1LZ3`, + `${A}/gp/aod/ajax?asin=B09B8V1LZ3`, + `${A}/gp/cart/view.html?ref_=nav_cart`, + `${A}/cart`, + `${A}/cart/smart-wagon?newItems=1`, + `${A}/gp/buy/spc/handlers/display.html`, + `${A}/checkout/p/p-123/spc`, + `${A}/ap/signin?openid.return_to=%2Fs%3Fk%3Dmat`, + `${A}/gp/css/homepage.html`, + `${A}/gp/your-account/order-history`, + `${A}/your-orders/orders`, + `${A}/hz/wishlist/ls`, + `${A}/a/addresses`, + `${A}/cpe/yourpayments/wallet`, + 'https://example.com/s?k=yoga+mat', + 'not a url', + ])('%s', (url) => { + expect(PageKind.classify(url).kind).toBe('never'); + expect(PageKind.suggests(url)).toBe(false); + expect(PageKind.scrapable(url)).toBe(false); + }); + + test('the home page and a bare /s are quiet', () => { + for (const url of [`${A}/`, `${A}/s`, `${A}/gp/bestsellers`, `${A}/sp?seller=`]) { + expect(PageKind.suggests(url)).toBe(false); + } + }); + + test('a me= value that is not a seller id is not a storefront', () => { + expect(PageKind.classify(`${A}/s?me= @@ -9,151 +10,118 @@ + -
- -

ProScan

+
+ + ProScan + +
- -
-
-
-
0
-
Items
-
-
-
-
-
Avg Rating
-
-
-
-
-
Avg Price
-
-
-
+
+ +
+ This tab +

Open a search or a storefront

+

ProScan scrapes Amazon search results and seller storefronts.

+ + + +
- -
diff --git a/popup/popup.js b/popup/popup.js index 8753165..6a24350 100644 --- a/popup/popup.js +++ b/popup/popup.js @@ -1,9 +1,10 @@ /** * @fileoverview Popup UI Controller * - * Manages the extension popup interface including the analytics dashboard, - * scraping controls, export buttons, and real-time progress updates. - * Depends on Storage, Analyzer, and Exporter modules loaded via popup.html. + * The popup is the fallback to the dock on the page (scripts/content/dock.js). + * It says what the current tab is, offers the same Scrape, follows the run, + * downloads the latest run's files, and holds the two things that must never + * be typed on an Amazon page: the Gemini key and the ProScan sign-in. * * State Management: * The service worker owns the run and everything it found. The popup never @@ -13,9 +14,12 @@ * * Communication: * - START_RUN / STOP_RUN / GET_STATE to the worker (scripts/lib/messages.js) - * - PING to a tab that did not answer, to start after a reload + * - PING to the active tab: what kind of page it is, and to start after a reload * - SPREAD_PROGRESS and SPREAD_ANALYSIS_COMPLETE from the offer fetcher * + * Opened in a tab as popup.html#key or #sync (the dock's settings links), + * it unfolds that setting. + * * @module PopupUI * @requires Storage * @requires Analyzer @@ -31,6 +35,9 @@ let currentResults = []; /** @type {Object} Spread data of the latest run, ASIN to offers */ let currentSpread = {}; +/** @type {?Object} The latest run record, from the worker */ +let currentRun = null; + /** @type {number} Page of the live run the results were last fetched for */ let renderedPage = -1; @@ -43,32 +50,58 @@ let currentProScanUser = null; /** @type {boolean} Whether a cloud export to ProScan is in progress */ let isExporting = false; +/** @type {?{kind: string, count: number, url: string, where: Object}} The active tab, as last seen */ +let currentTab = null; + +/** Inline icons: the popup loads nothing from the network. */ +const ICONS = { + play: '', + stop: '', + down: '', + spin: '' +}; + +/** Sets a button to an icon and a text label; the label is text, never HTML. */ +function setButton(button, icon, label) { + button.innerHTML = icon ? ICONS[icon] : ''; + button.appendChild(document.createTextNode((icon ? ' ' : '') + label)); +} + /** * Cached references to DOM elements used throughout the popup lifecycle. - * Resolved once at module load time for performance. * @const {Object} */ const elements = { itemCount: document.getElementById('itemCount'), + itemWord: document.getElementById('itemWord'), avgRating: document.getElementById('avgRating'), avgPrice: document.getElementById('avgPrice'), status: document.getElementById('status'), + tabChip: document.getElementById('tabChip'), + tabChipText: document.getElementById('tabChipText'), + hereTitle: document.getElementById('hereTitle'), + hereSub: document.getElementById('hereSub'), actionButton: document.getElementById('actionButton'), reloadTabButton: document.getElementById('reloadTabButton'), + lastScrape: document.getElementById('lastScrape'), + lastWhen: document.getElementById('lastWhen'), + lastSource: document.getElementById('lastSource'), + lastTicks: document.getElementById('lastTicks'), + lastLine: document.getElementById('lastLine'), exportButtons: document.getElementById('exportButtons'), downloadExcel: document.getElementById('downloadExcel'), downloadCSV: document.getElementById('downloadCSV'), downloadJSON: document.getElementById('downloadJSON'), - insightsPreview: document.getElementById('insightsPreview'), - insightBadge: document.getElementById('insightBadge'), - insightText: document.getElementById('insightText'), spreadButton: document.getElementById('spreadButton'), + spreadLabel: document.getElementById('spreadLabel'), + spreadHint: document.getElementById('spreadHint'), spreadProgress: document.getElementById('spreadProgress'), spreadProgressFill: document.getElementById('spreadProgressFill'), spreadProgressText: document.getElementById('spreadProgressText'), spreadResults: document.getElementById('spreadResults'), spreadHighCount: document.getElementById('spreadHighCount'), spreadAvgCV: document.getElementById('spreadAvgCV'), + aiKeySummary: document.getElementById('aiKeySummary'), // ProScan cloud auth + export authForm: document.getElementById('authForm'), authEmail: document.getElementById('authEmail'), @@ -81,131 +114,174 @@ const elements = { authNotice: document.getElementById('authNotice'), authResetBtn: document.getElementById('authResetBtn'), authSignUpLink: document.getElementById('authSignUpLink'), + authSummarySub: document.getElementById('authSummarySub'), authSync: document.getElementById('authSync'), exportToProScanBtn: document.getElementById('exportToProScanBtn') }; +const fmt = (n) => Number(n || 0).toLocaleString('en-US'); +const plural = (n, one, many = one + 's') => `${fmt(n)} ${n === 1 ? one : many}`; + /** - * Update the dashboard stats display with current results. - * Calculates average price and rating using the Analyzer module - * and triggers the insights preview update. + * Update the stats in the last-scrape card: count, average price and rating. * * @param {Object[]} results - Array of product objects */ function updateStats(results) { currentResults = results; - elements.itemCount.textContent = results.length; - - if (results.length > 0) { - const prices = results.map(r => Analyzer.parsePrice(r.price)).filter(p => p > 0); - const ratings = results.map(r => Analyzer.parseRating(r.rating)).filter(r => r > 0); - - if (prices.length > 0) { - const avgPrice = (prices.reduce((a, b) => a + b, 0) / prices.length).toFixed(2); - elements.avgPrice.textContent = '$' + avgPrice; - } - - if (ratings.length > 0) { - const avgRating = (ratings.reduce((a, b) => a + b, 0) / ratings.length).toFixed(1); - elements.avgRating.textContent = avgRating; - } + elements.itemCount.textContent = fmt(results.length); + elements.itemWord.textContent = results.length === 1 ? 'product' : 'products'; + elements.avgPrice.textContent = '-'; + elements.avgRating.textContent = '-'; + if (results.length === 0) return; - updateInsightsPreview(results); + const prices = results.map(r => Analyzer.parsePrice(r.price)).filter(p => p > 0); + const ratings = results.map(r => Analyzer.parseRating(r.rating)).filter(r => r > 0); + if (prices.length > 0) { + elements.avgPrice.textContent = '$' + (prices.reduce((a, b) => a + b, 0) / prices.length).toFixed(2); } -} - -/** - * Show the top insight badge in the popup. - * Requires at least 5 products for meaningful analysis. - * Prioritizes high-priority insights (opportunities) over informational ones. - * - * @param {Object[]} results - Array of product objects - */ -function updateInsightsPreview(results) { - if (results.length < 5) return; - - const insights = Analyzer.generateInsights(results); - const topInsight = insights.find(i => i.priority === 'high') || insights[0]; - - if (topInsight) { - elements.insightText.textContent = topInsight.title; - elements.insightBadge.className = 'insight-badge ' + (topInsight.type === 'opportunity' ? 'opportunity' : ''); - elements.insightsPreview.classList.remove('hidden'); + if (ratings.length > 0) { + elements.avgRating.textContent = (ratings.reduce((a, b) => a + b, 0) / ratings.length).toFixed(1); } } /** - * Update the status bar message with an icon and styling. + * Show a status line under the tab description, or hide it with ''. * * @param {string} message - Status message text * @param {'info'|'success'|'error'|'warning'} [type='info'] - Status type for styling */ function updateStatus(message, type = 'info') { - elements.status.innerHTML = ` ${message}`; - elements.status.className = 'status ' + type; + elements.status.textContent = message || ''; + elements.status.className = 'status ' + type + (message ? '' : ' hidden'); } /** - * Map status type to Font Awesome icon name. - * - * @param {string} type - Status type - * @returns {string} Font Awesome icon identifier - */ -function getStatusIcon(type) { - const icons = { - info: 'info-circle', - success: 'check-circle', - error: 'exclamation-circle', - warning: 'exclamation-triangle' - }; - return icons[type] || 'info-circle'; -} - -/** - * Toggle the UI between scraping and idle states. - * Updates the action button text/style and shows/hides export buttons. + * Toggle the main button between Scrape and Stop. * * @param {boolean} isActive - Whether scraping is in progress */ function setScrapingState(isActive) { isScrapingActive = isActive; - + elements.actionButton.classList.toggle('stop', isActive); if (isActive) { - elements.actionButton.innerHTML = ' Stop Scraping'; - elements.actionButton.classList.add('stop'); - elements.exportButtons.classList.add('hidden'); + const n = currentRun ? currentRun.itemCount || 0 : currentResults.length; + setButton(elements.actionButton, 'stop', `Stop and keep ${plural(n, 'product')}`); } else { - elements.actionButton.innerHTML = ' Start Scraping'; - elements.actionButton.classList.remove('stop'); - if (currentResults.length > 0) { - elements.exportButtons.classList.remove('hidden'); - } + setButton(elements.actionButton, 'play', 'Scrape this page'); } + elements.exportButtons.classList.toggle('hidden', isActive || currentResults.length === 0); } +const DOWNLOAD_LABELS = { downloadExcel: 'Excel', downloadCSV: 'CSV', downloadJSON: 'JSON' }; + /** - * Toggle loading spinner on a download button during export generation. + * Toggle loading state on a download button during export generation. * * @param {HTMLButtonElement} button - The download button element - * @param {boolean} loading - True to show spinner, false to restore original content + * @param {boolean} loading - True while the file is made */ function setDownloadLoading(button, loading) { - if (loading) { - button.innerHTML = ' Preparing...'; - button.disabled = true; + button.disabled = loading; + if (loading) setButton(button, 'spin', 'Making'); + else setButton(button, button.id === 'downloadExcel' ? 'down' : null, DOWNLOAD_LABELS[button.id]); +} + +/** The pages strip: saved pages, the one in progress, and hatched pages a store never had. */ +function drawTicks(run) { + const box = elements.lastTicks; + box.textContent = ''; + if (!run || !run.maxPages) return; + const max = run.maxPages; + const n = Math.min(max, 20); + const per = max / n; + const done = run.page || 0; + const live = Run.isActive(run); + const early = !live && run.reason === 'complete' && done < max; + for (let i = 0; i < n; i++) { + const from = i * per; + const to = (i + 1) * per; + const tick = document.createElement('i'); + if (done >= to - 1e-9) tick.className = 'done'; + else if (live && done >= from - 1e-9) tick.className = 'now'; + else if (!live && done > from) tick.className = 'done'; + else if (early) tick.className = 'skip'; + box.appendChild(tick); + } +} + +/** What a run scraped, in words: a storefront or a search. */ +function sourceLabel(run) { + const src = (run && run.source) || {}; + if (src.type === 'storefront') return src.sellerId ? `storefront ${src.sellerId}` : 'a storefront'; + return src.keyword ? `search "${src.keyword}"` : 'search results'; +} + +function clock(ms) { + try { return new Date(ms).toLocaleTimeString('en-US', { hour: 'numeric', minute: '2-digit' }); } catch (e) { return ''; } +} + +/** The last-scrape card: count, source, pages strip, how it ended. */ +function renderLast(run) { + const has = !!run && (currentResults.length > 0 || Run.isActive(run)); + elements.lastScrape.classList.toggle('hidden', !has); + if (!has) return; + elements.lastSource.textContent = 'from ' + sourceLabel(run); + elements.lastWhen.textContent = Run.isActive(run) ? 'Running now' : run.finishedAt ? clock(run.finishedAt) : ''; + drawTicks(run); + let line = ''; + if (Run.isActive(run)) { + line = `Page ${fmt(Math.min((run.page || 0) + 1, run.maxPages))} of ${fmt(run.maxPages)}`; + } else if (run.reason === 'complete' && run.maxPages) { + line = run.page < run.maxPages + ? `Reached the last page. Page ${fmt(run.page)} was the last one, so the run ended early.` + : `Stopped at the ${fmt(run.maxPages)}-page limit.`; + } else if (run.reason === 'stopped') { + line = `You stopped it after page ${fmt(run.page)}.`; + } + elements.lastLine.textContent = line; + elements.lastLine.classList.toggle('hidden', !line); +} + +/** The top of the popup: what the active tab is and whether Scrape works there. */ +function renderHere() { + const run = currentRun; + const chip = elements.tabChip; + chip.className = 'chip'; + if (Run.isActive(run)) { + chip.classList.add('live'); + elements.tabChipText.textContent = 'Scraping now'; + elements.hereTitle.textContent = `Page ${fmt(Math.min((run.page || 0) + 1, run.maxPages))} of ${fmt(run.maxPages)}`; + elements.hereSub.textContent = `${plural(run.itemCount || 0, 'product')} saved so far from ${sourceLabel(run)}. Keep its tab open.`; + return; + } + const tab = currentTab; + const where = tab && tab.where ? tab.where : { kind: 'other' }; + const startable = !!tab && Run.STARTABLE.includes(tab.kind); + const hint = 'You can also press Scrape in the bottom-right corner of the page.'; + elements.actionButton.classList.toggle('muted', !startable); + if (startable && where.kind === 'storefront') { + chip.classList.add('on'); + elements.tabChipText.textContent = 'This tab: seller storefront'; + elements.hereTitle.textContent = `${plural(tab.count, 'product')} on this page`; + elements.hereSub.textContent = `A seller's storefront. ${hint}`; + } else if (startable) { + chip.classList.add('on'); + elements.tabChipText.textContent = 'This tab: search results'; + elements.hereTitle.textContent = `${plural(tab.count, 'product')} on this page`; + elements.hereSub.textContent = (where.keyword ? `Search for "${where.keyword}". ` : '') + hint; + } else if (tab && tab.kind && (where.kind === 'search' || where.kind === 'storefront')) { + elements.tabChipText.textContent = 'This tab: cannot scrape yet'; + elements.hereTitle.textContent = 'Nothing to scrape here yet'; + elements.hereSub.textContent = Run.refusal(tab.kind); + } else if (where.kind === 'seller' && where.sellerId) { + elements.tabChipText.textContent = 'This tab: seller profile'; + elements.hereTitle.textContent = 'Open the storefront to scrape'; + elements.hereSub.textContent = 'The page has a ProScan button bottom-right that opens this seller\'s storefront.'; } else { - const icons = { - downloadExcel: 'file-excel', - downloadCSV: 'file-csv', - downloadJSON: 'file-code' - }; - const labels = { - downloadExcel: 'Excel', - downloadCSV: 'CSV', - downloadJSON: 'JSON' - }; - button.innerHTML = ` ${labels[button.id]}`; - button.disabled = false; + elements.tabChipText.textContent = tab && tab.url ? 'This tab: not a search' : 'This tab'; + elements.hereTitle.textContent = 'Open a search or a storefront'; + elements.hereSub.textContent = 'ProScan scrapes Amazon search results and seller storefronts, then saves them as Excel, CSV or JSON.'; } } @@ -217,27 +293,24 @@ function setDownloadLoading(button, loading) { */ function render(state) { const run = (state && state.run) || null; + currentRun = run; currentResults = (state && state.results) || []; currentSpread = (state && state.spread) || {}; isScrapingActive = Run.isActive(run); renderedPage = run ? run.page : -1; - elements.itemCount.textContent = currentResults.length; - elements.avgRating.textContent = '-'; - elements.avgPrice.textContent = '-'; - elements.insightsPreview.classList.add('hidden'); updateStats(currentResults); setScrapingState(isScrapingActive); + renderHere(); + renderLast(run); const line = Run.describe(run, currentResults.length); - if (line && (isScrapingActive || run.reason !== 'complete')) { - updateStatus(line.text, line.type); - } else if (currentResults.length > 0) { - updateStatus('Ready to download ' + currentResults.length + ' products', 'success'); - } + if (line && !isScrapingActive && run.reason !== 'complete') updateStatus(line.text, line.type); + else updateStatus(''); const showSpread = !isScrapingActive && currentResults.length > 0; elements.spreadButton.classList.toggle('hidden', !showSpread); + elements.spreadHint.textContent = `Checks other sellers' offers, about ${Math.max(1, Math.round(currentResults.length * 2.5 / 60))} min`; if (showSpread && Object.keys(currentSpread).length > 0) displaySpreadResults(); } @@ -261,16 +334,6 @@ async function warnIfNearlyFull() { } catch (e) { /* no estimate in this context */ } } -/** - * Initialize the popup UI from the worker's state. - * - * @async - */ -async function initializeUI() { - await refresh(); - await warnIfNearlyFull(); -} - /** The tab the popup was opened over, or null. */ function activeTab() { return new Promise(resolve => { @@ -294,6 +357,34 @@ function sendToTab(tabId, message) { const AMAZON_URL = /^https:\/\/([a-z0-9-]+\.)*amazon\.com\//i; +/** Looks at the active tab (PING) and redraws the top of the popup. */ +async function refreshTab() { + const tab = await activeTab(); + if (!tab || !AMAZON_URL.test(tab.url || '')) { + currentTab = tab ? { kind: null, count: 0, url: tab.url || '', where: { kind: 'other' } } : null; + } else { + const pong = await sendToTab(tab.id, { type: Msg.T.PING }); + currentTab = { + kind: pong && pong.ok ? pong.kind : null, + count: pong && pong.ok ? pong.count : 0, + url: tab.url, + where: PageKind.classify(tab.url) + }; + } + renderHere(); +} + +/** + * Initialize the popup UI from the worker's state. + * + * @async + */ +async function initializeUI() { + await refresh(); + await refreshTab(); + await warnIfNearlyFull(); +} + /** * Start a new scraping session. * @@ -320,6 +411,7 @@ async function startScraping() { offerReload(tab.id); return; } + refreshTab(); updateStatus((resp && (resp.message || resp.error)) || 'Could not start the run.', 'warning'); } @@ -369,6 +461,13 @@ async function stopScraping() { // --- Spread Analysis --- +function setSpreadButton(running) { + elements.spreadButton.classList.toggle('stop', running); + elements.spreadLabel.textContent = running ? 'Stop checking seller prices' : 'Compare seller prices'; + // Starting a run navigates the tab and would cut the analysis off. + elements.actionButton.disabled = running; +} + /** * Start the price spread analysis. * Sends a message to the offer-fetcher content script to begin @@ -381,13 +480,12 @@ async function startSpreadAnalysis() { } isSpreadAnalyzing = true; - elements.spreadButton.innerHTML = ' Stop Analysis'; - elements.spreadButton.classList.add('stop'); + setSpreadButton(true); elements.spreadProgress.classList.remove('hidden'); elements.spreadResults.classList.add('hidden'); elements.spreadProgressFill.style.width = '0%'; elements.spreadProgressText.textContent = `0/${currentResults.length}`; - updateStatus('Analyzing price spreads...', 'info'); + updateStatus('Checking seller prices. Keep the Amazon tab on its page until it finishes.', 'info'); chrome.tabs.query({ active: true, currentWindow: true }, tabs => { chrome.tabs.sendMessage(tabs[0].id, { type: Msg.T.START_SPREAD_ANALYSIS }); @@ -399,9 +497,8 @@ async function startSpreadAnalysis() { */ function stopSpreadAnalysis() { isSpreadAnalyzing = false; - elements.spreadButton.innerHTML = ' Analyze Price Spreads'; - elements.spreadButton.classList.remove('stop'); - updateStatus('Spread analysis stopped.', 'warning'); + setSpreadButton(false); + updateStatus('Seller price check stopped.', 'info'); chrome.tabs.query({ active: true, currentWindow: true }, tabs => { chrome.tabs.sendMessage(tabs[0].id, { type: Msg.T.STOP_SPREAD_ANALYSIS }); @@ -422,22 +519,16 @@ function displaySpreadResults() { elements.spreadResults.classList.remove('hidden'); elements.spreadProgress.classList.add('hidden'); - // Show insight if there are high-spread products if (summary.highSpreadCount > 0) { - updateStatus( - `Found ${summary.highSpreadCount} products with high price spread!`, - 'success' - ); + updateStatus(`${plural(summary.highSpreadCount, 'product')} with a wide price spread. They are marked in the Excel and CSV files.`, 'success'); } else if (summary.withSpreadData > 0) { - updateStatus( - `Analyzed ${summary.withSpreadData} products. No high spreads found.`, - 'info' - ); + updateStatus(`Checked ${plural(summary.withSpreadData, 'product')}. No wide spreads found.`, 'info'); } } // --- ProScan Cloud Auth + Export --- + /** * Send a message to the background service worker and resolve with its * response. The sync/auth handlers are async (return true), so responses may @@ -467,7 +558,8 @@ function sendToWorker(message) { function showSignedIn(user) { currentProScanUser = user; elements.authStatusEmail.textContent = user.email || user.displayName || 'Signed in'; - document.getElementById('authSummary').textContent = 'Syncing to dashboard'; + elements.authSummarySub.textContent = 'Signed in'; + elements.authSummarySub.classList.add('on'); elements.authForm.classList.add('hidden'); elements.authAccount.classList.remove('hidden'); elements.authError.classList.add('hidden'); @@ -488,12 +580,12 @@ function showSyncState(state) { let text = 'Scans sync to your ProScan dashboard automatically.'; if (pending > 0) { text = `${pending} page${pending === 1 ? '' : 's'} waiting to sync.`; - if (last && last.error) text += ' The last try failed; it will try again in a while, or now with Export to ProScan.'; + if (last && last.error) text += ' The last try failed; it will try again in a while, or now with Send to my ProScan dashboard.'; } else if (last && !last.error) { text = 'Everything is synced.'; } if (failed > 0) { - text += ` ${failed} page${failed === 1 ? '' : 's'} could not sync: ProScan refused ${failed === 1 ? 'it' : 'them'}. Export to ProScan tries again.`; + text += ` ${failed} page${failed === 1 ? '' : 's'} could not sync: ProScan refused ${failed === 1 ? 'it' : 'them'}. Send to my ProScan dashboard tries again.`; } elements.authSync.textContent = text; elements.authSync.classList.remove('hidden'); @@ -511,7 +603,8 @@ function showSignInForm(notice = null) { ? 'Your session expired. Sign in again to keep syncing; your scans are kept.' : ''; elements.authNotice.classList.toggle('hidden', notice !== 'expired'); - document.getElementById('authSummary').textContent = 'Dashboard sync (optional)'; + elements.authSummarySub.textContent = 'Not signed in'; + elements.authSummarySub.classList.remove('on'); // An expired session is the one case worth unfolding the panel for. if (notice === 'expired') document.getElementById('authPanel').open = true; elements.authForm.classList.remove('hidden'); @@ -560,7 +653,7 @@ async function handleResetPassword() { */ function resetSignInButton() { elements.authSignInBtn.disabled = false; - elements.authSignInBtn.innerHTML = ' Sign in to ProScan'; + setButton(elements.authSignInBtn, null, 'Sign in to ProScan'); } /** @@ -601,7 +694,7 @@ async function handleSignIn() { } elements.authSignInBtn.disabled = true; - elements.authSignInBtn.innerHTML = ' Signing in…'; + setButton(elements.authSignInBtn, 'spin', 'Signing in'); const response = await sendToWorker({ type: Msg.T.PROSCAN_SIGN_IN, email, password }); @@ -648,8 +741,8 @@ async function handleExportToProScan() { isExporting = true; elements.exportToProScanBtn.disabled = true; - elements.exportToProScanBtn.innerHTML = ' Exporting…'; - updateStatus('Exporting to ProScan…', 'info'); + setButton(elements.exportToProScanBtn, 'spin', 'Sending'); + updateStatus('Sending to your ProScan dashboard.', 'info'); const response = await sendToWorker({ type: Msg.T.PROSCAN_EXPORT }); @@ -673,7 +766,7 @@ async function handleExportToProScan() { isExporting = false; elements.exportToProScanBtn.disabled = false; - elements.exportToProScanBtn.innerHTML = ' Export to ProScan'; + setButton(elements.exportToProScanBtn, null, 'Send to my ProScan dashboard'); } // --- Event Listeners --- @@ -714,92 +807,37 @@ elements.actionButton.addEventListener('click', async () => { console.error('[ProScan] Start or stop failed:', err && err.message); updateStatus('Something went wrong. Close and reopen the popup, then try again.', 'error'); } finally { - elements.actionButton.disabled = false; - } -}); - -// Excel export with full analytics report -elements.downloadExcel.addEventListener('click', async () => { - if (currentResults.length === 0) { - updateStatus('No results to download!', 'warning'); - return; - } - - setDownloadLoading(elements.downloadExcel, true); - updateStatus('Generating Excel report...', 'info'); - - try { - const analysisReport = Analyzer.generateFullReport(currentResults); - const { blob, filename } = Exporter.exportToExcel(currentResults, analysisReport); - - Exporter.triggerDownload(blob, filename, (downloadId) => { - if (downloadId) { - updateStatus('Download started!', 'success'); - } else { - updateStatus('Download failed. Try again.', 'error'); - } - setDownloadLoading(elements.downloadExcel, false); - }); - } catch (error) { - console.error('Excel export error:', error); - updateStatus('Error generating Excel file', 'error'); - setDownloadLoading(elements.downloadExcel, false); + elements.actionButton.disabled = isSpreadAnalyzing; } }); -// CSV export -elements.downloadCSV.addEventListener('click', async () => { +/** + * Makes the file for `format` with Exporter.build (the same code the + * worker uses for the page) and opens the save dialog. + */ +function downloadFile(button, format) { if (currentResults.length === 0) { - updateStatus('No results to download!', 'warning'); + updateStatus('There is nothing to download yet.', 'warning'); return; } - - setDownloadLoading(elements.downloadCSV, true); - + setDownloadLoading(button, true); try { - const { blob, filename } = Exporter.exportToCSV(currentResults); - + const { blob, filename } = Exporter.build(format, currentResults, currentSpread); Exporter.triggerDownload(blob, filename, (downloadId) => { - if (downloadId) { - updateStatus('CSV download started!', 'success'); - } else { - updateStatus('Download failed. Try again.', 'error'); - } - setDownloadLoading(elements.downloadCSV, false); + if (downloadId) updateStatus(`${DOWNLOAD_LABELS[button.id]} download started.`, 'success'); + else updateStatus('The download did not start. Try again.', 'error'); + setDownloadLoading(button, false); }); } catch (error) { - console.error('CSV export error:', error); - updateStatus('Error generating CSV file', 'error'); - setDownloadLoading(elements.downloadCSV, false); - } -}); - -// JSON export with analytics -elements.downloadJSON.addEventListener('click', async () => { - if (currentResults.length === 0) { - updateStatus('No results to download!', 'warning'); - return; + console.error('[ProScan] Export failed:', error); + updateStatus(`ProScan could not make the ${DOWNLOAD_LABELS[button.id]} file.`, 'error'); + setDownloadLoading(button, false); } +} - setDownloadLoading(elements.downloadJSON, true); - - try { - const { blob, filename } = Exporter.exportToJSON(currentResults, true); - - Exporter.triggerDownload(blob, filename, (downloadId) => { - if (downloadId) { - updateStatus('JSON download started!', 'success'); - } else { - updateStatus('Download failed. Try again.', 'error'); - } - setDownloadLoading(elements.downloadJSON, false); - }); - } catch (error) { - console.error('JSON export error:', error); - updateStatus('Error generating JSON file', 'error'); - setDownloadLoading(elements.downloadJSON, false); - } -}); +elements.downloadExcel.addEventListener('click', () => downloadFile(elements.downloadExcel, 'xlsx')); +elements.downloadCSV.addEventListener('click', () => downloadFile(elements.downloadCSV, 'csv')); +elements.downloadJSON.addEventListener('click', () => downloadFile(elements.downloadJSON, 'json')); /** * Messages from the offer fetcher in the tab: @@ -814,8 +852,7 @@ chrome.runtime.onMessage.addListener((request) => { elements.spreadProgressText.textContent = `${request.current}/${request.total}`; } else if (request.type === Msg.T.SPREAD_ANALYSIS_COMPLETE) { isSpreadAnalyzing = false; - elements.spreadButton.innerHTML = ' Analyze Price Spreads'; - elements.spreadButton.classList.remove('stop'); + setSpreadButton(false); refresh(); } return false; @@ -824,18 +861,45 @@ chrome.runtime.onMessage.addListener((request) => { // The worker writes the run record to session storage on every change, so // the view follows it: a new page, the end of the run, a closed tab. chrome.storage.onChanged.addListener((changes, area) => { + if (area === 'local' && changes.geminiApiKey) showKeySummary(!!changes.geminiApiKey.newValue); if (area !== 'session' || !changes[Run.KEY]) return; const run = changes[Run.KEY].newValue; if (!run) return; if (!Run.isActive(run) || run.page !== renderedPage || !isScrapingActive) { refresh(); } else { - elements.itemCount.textContent = run.itemCount || 0; + currentRun = run; + renderHere(); + setScrapingState(true); } }); +/** The one-line state under "AI chat settings". */ +function showKeySummary(set) { + elements.aiKeySummary.textContent = set ? 'Gemini key added' : 'No key yet'; + elements.aiKeySummary.classList.toggle('on', set); +} + +/** popup.html#key or #sync, opened from the dock: unfold that setting. */ +function openFromHash() { + const hash = (location.hash || '').slice(1); + if (!hash) return; + document.body.classList.add('in-tab'); + const panel = document.getElementById(hash === 'sync' ? 'authPanel' : 'aiSettings'); + if (!panel || panel.classList.contains('hidden')) return; + panel.open = true; + const field = hash === 'sync' ? elements.authEmail : document.getElementById('geminiKeyInput'); + if (field) field.focus(); + panel.scrollIntoView({ block: 'nearest' }); +} + +// For scripts that aim the popup at a tab after it opened (the screenshot harness). +window.ProScanPopup = { refreshTab, refresh }; + // Initialize on DOM load document.addEventListener('DOMContentLoaded', () => { + try { document.getElementById('version').textContent = chrome.runtime.getManifest().version; } catch (e) { /* no manifest */ } + chrome.storage.local.get('geminiApiKey').then((d) => showKeySummary(!!(d && d.geminiApiKey)), () => {}); initializeUI(); - initializeAuthUI(); + initializeAuthUI().then(openFromHash, openFromHash); }); From 10e1f3167441de6aebb1c8c426d7c58775cedfee Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 21:42:09 -0400 Subject: [PATCH 08/24] Drive the dock in Chromium: suggest, scrape in its tab, stop, dismiss, download --- tests/e2e/dock.spec.mjs | 153 ++++++++++++++++++++++++++++++++++++++++ tests/e2e/lib/dock.mjs | 104 +++++++++++++++++++++++++++ 2 files changed, 257 insertions(+) create mode 100644 tests/e2e/dock.spec.mjs create mode 100644 tests/e2e/lib/dock.mjs diff --git a/tests/e2e/dock.spec.mjs b/tests/e2e/dock.spec.mjs new file mode 100644 index 0000000..729d9ad --- /dev/null +++ b/tests/e2e/dock.spec.mjs @@ -0,0 +1,153 @@ +// The on-page dock in real Chromium, against saved pages. The dock sits in +// a closed shadow root; tests/e2e/lib/dock.mjs reaches it through CDP. +import fs from 'node:fs'; +import { createRequire } from 'node:module'; +import { test as base, expect } from '@playwright/test'; +import { launch, extPage, getState, waitForState, endReason, sleep, ended } from './lib/extension.mjs'; +import { serveAmazon, simplePlan, corpusPage, asinFor } from './lib/amazon.mjs'; +import { dock } from './lib/dock.mjs'; + +const require = createRequire(import.meta.url); + +const test = base.extend({ + // eslint-disable-next-line no-empty-pattern + ext: async ({}, use) => { + const ext = await launch(); + await use(ext); + await ext.close(); + }, +}); + +const searchUrl = (k, p) => `https://www.amazon.com/s?k=${encodeURIComponent(k).replace(/%20/g, '+')}${p ? `&page=${p}` : ''}`; +const pagesOf = (served, k) => served.filter((p) => p.keyword === k).map((p) => p.page); + +async function open(ext, url) { + const tab = await ext.context.newPage(); + await tab.goto(url); + const d = await dock(tab); + await d.waitFor('.launcher'); + return { tab, d }; +} + +/** The id Chrome gave `tab`, asked from an extension page. */ +async function tabIdOf(store, tab) { + return store.evaluate((url) => new Promise((r) => chrome.tabs.query({ url }, (t) => r(t[0] && t[0].id))), tab.url()); +} + +test('the dock suggests Scrape on a saved search page, with its product count', async ({ ext }) => { + const yoga = corpusPage('2026-09/search-yoga-mat.html'); + await serveAmazon(ext.context, () => ({ body: yoga })); + const { d } = await open(ext, searchUrl('yoga mat')); + expect(await d.has('.launcher.suggest')).toBe(true); + const Parsers = require('../../scripts/lib/parsers.js'); + const { parseDoc } = require('../setup/corpus.js'); + const n = Parsers.parseSearchPage(parseDoc(yoga, searchUrl('yoga mat')), searchUrl('yoga mat')).products.length; + expect(await d.text('.launcher')).toContain(`"yoga mat"`); + expect(await d.text('.launcher')).toContain(`${n} products on this page`); + expect(await d.attr('[data-k="notnow"]', 'aria-label')).toBe('Not now, hide for this tab'); +}); + +test('Scrape in the dock runs in that tab, page by page, and completes', async ({ ext }) => { + const served = await serveAmazon(ext.context, simplePlan(3)); + const store = await extPage(ext); + const other = await ext.context.newPage(); + await other.goto(searchUrl('elsewhere')); + const { tab, d } = await open(ext, searchUrl('garden hose')); + await d.click('[data-k="scrape-quick"]'); + const s = await waitForState(store, (x) => x.scrapeRunPages?.length >= 3 && !x.isScrapingActive, { timeout: 40000 }); + + expect(s.run.tabId).toBe(await tabIdOf(store, tab)); + expect(pagesOf(served, 'garden hose')).toEqual([1, 2, 3]); + expect(pagesOf(served, 'elsewhere')).toEqual([1]); + expect(s.results.map((r) => r.asin)).toEqual([1, 2, 3].flatMap((p) => [0, 1, 2, 3].map((i) => asinFor('garden hose', p, i)))); + expect(endReason(s)).toBe('complete'); + + // The last page's dock rebuilt itself from the worker and shows the result. + const after = await dock(tab); + await after.waitForText(/12 products saved/, { timeout: 10000 }); + await after.waitForText(/Reached the last page\. Page 3 was the last one/); + expect(await after.count('.ticks i.skip')).toBe(17); +}); + +test('Stop in the dock halts the run before the next page', async ({ ext }) => { + const served = await serveAmazon(ext.context, simplePlan(4)); + const store = await extPage(ext); + const { tab, d } = await open(ext, searchUrl('stop me')); + await d.click('[data-k="launcher"]'); + await d.waitFor('[data-k="scrape"]'); + await d.click('[data-k="scrape"]'); + await waitForState(store, (x) => x.scrapeRunPages?.length >= 1, { timeout: 15000, interval: 100 }); + const live = await dock(tab); + await live.waitFor('[data-k="stop"]'); + expect(await live.text('[data-k="stop"]')).toMatch(/Stop and keep 4 products/); + await live.click('[data-k="stop"]'); + await sleep(4500); + const s = await getState(store); + + expect(s.isScrapingActive).toBe(false); + expect(endReason(s)).toBe('stopped'); + expect(pagesOf(served, 'stop me')).toEqual([1]); + await live.waitForText(/Stopped with 4 products/); +}); + +test('a product page gets the plain launcher and no Scrape; the cart gets no dock', async ({ ext }) => { + await serveAmazon(ext.context, simplePlan(1)); + const { tab, d } = await open(ext, 'https://www.amazon.com/dp/B09B8V1LZ3'); + expect(await d.has('.launcher.plain')).toBe(true); + expect(await d.has('[data-k="scrape-quick"]')).toBe(false); + await d.click('[data-k="launcher"]'); + await d.waitForText(/Open a search or a storefront/); + expect(await d.has('[data-k="scrape"]')).toBe(false); + + await ext.context.route('https://www.amazon.com/gp/cart/**', (r) => r.fulfill({ status: 200, contentType: 'text/html', body: '

Cart

' })); + await tab.goto('https://www.amazon.com/gp/cart/view.html'); + await sleep(1200); + expect(await (await dock(tab)).present()).toBe(false); +}); + +test('Not now hides the suggestion for the rest of the tab, not in a new tab', async ({ ext }) => { + await serveAmazon(ext.context, simplePlan(2)); + const { tab, d } = await open(ext, searchUrl('dismiss')); + await d.click('[data-k="notnow"]'); + await d.waitFor('.launcher.plain'); + await tab.goto(searchUrl('dismiss', 2)); + const again = await dock(tab); + await again.waitFor('.launcher'); + expect(await again.has('.launcher.plain')).toBe(true); + + const fresh = await open(ext, searchUrl('dismiss')); + expect(await fresh.d.has('.launcher.suggest')).toBe(true); +}); + +test('Download Excel in the dock saves the run through chrome.downloads', async ({ ext }) => { + await serveAmazon(ext.context, simplePlan(2)); + const store = await extPage(ext); + const { tab, d } = await open(ext, searchUrl('export me')); + await d.click('[data-k="scrape-quick"]'); + await waitForState(store, (x) => ended(x) && x.scrapeRunPages?.length >= 2, { timeout: 40000 }); + const done = await dock(tab); + await done.waitFor('[data-k="xlsx"]', { timeout: 10000 }); + await done.click('[data-k="xlsx"]'); + await done.waitForText(/Excel download started/, { timeout: 15000 }); + await done.click('[data-k="csv"]'); + await done.waitForText(/CSV download started/, { timeout: 15000 }); + + const items = await store.evaluate(async () => { + const shape = (f) => ({ url: f.url.slice(0, 60), mime: f.mime, state: f.state, error: f.error || null, bytes: f.totalBytes, file: f.filename }); + for (let i = 0; i < 40; i++) { + const found = await chrome.downloads.search({}); + if (found.length >= 2 && found.every((f) => f.state !== 'in_progress')) return found.map(shape); + await new Promise((r) => setTimeout(r, 250)); + } + return (await chrome.downloads.search({})).map(shape); + }); + // Playwright keeps downloads under its own names, so check type and bytes. + expect(items.map((i) => [i.mime, i.state, i.error]).sort()).toEqual([ + ['application/vnd.openxmlformats-officedocument.spreadsheetml.sheet', 'complete', null], + ['text/csv', 'complete', null], + ]); + const csv = fs.readFileSync(items.find((i) => i.mime === 'text/csv').file, 'utf8'); + expect(csv).toContain(asinFor('export me', 2, 3)); + const xlsx = fs.readFileSync(items.find((i) => i.mime !== 'text/csv').file); + expect(xlsx.subarray(0, 2).toString()).toBe('PK'); +}); diff --git a/tests/e2e/lib/dock.mjs b/tests/e2e/lib/dock.mjs new file mode 100644 index 0000000..2f858c4 --- /dev/null +++ b/tests/e2e/lib/dock.mjs @@ -0,0 +1,104 @@ +// Reaches the dock inside its closed shadow root through CDP, which can +// pierce closed roots. Page scripts cannot, and neither can Playwright's +// locators, so the tests read and click it here. Clicks are real mouse +// clicks at the element's center. + +import { sleep } from './extension.mjs'; + +const HOST = 'proscan-dock-host'; + +function findHost(node) { + const a = node.attributes || []; + for (let i = 0; i < a.length; i += 2) if (a[i] === 'id' && a[i + 1] === HOST) return node; + for (const c of node.children || []) { + const f = findHost(c); + if (f) return f; + } + return null; +} + +export async function dock(page) { + const cdp = await page.context().newCDPSession(page); + + async function shadowId() { + const { root } = await cdp.send('DOM.getDocument', { depth: -1, pierce: true }); + const host = findHost(root); + return host && host.shadowRoots && host.shadowRoots[0] ? host.shadowRoots[0].nodeId : null; + } + + async function query(selector) { + const id = await shadowId(); + if (!id) return null; + const { nodeId } = await cdp.send('DOM.querySelector', { nodeId: id, selector }); + return nodeId || null; + } + + async function call(selector, fn) { + const nodeId = await query(selector); + if (!nodeId) return undefined; + const { object } = await cdp.send('DOM.resolveNode', { nodeId }); + const { result } = await cdp.send('Runtime.callFunctionOn', { + objectId: object.objectId, functionDeclaration: fn, returnByValue: true, + }); + return result.value; + } + + const api = { + /** True when the page has a dock host at all. */ + async present() { + return page.evaluate((id) => !!document.getElementById(id), HOST); + }, + has: async (selector) => !!(await query(selector)), + text: async (selector = '.dock') => { + const t = await call(selector, 'function () { return this.textContent; }'); + return t === undefined ? null : t.replace(/\s+/g, ' ').trim(); + }, + attr: (selector, name) => call(selector, `function () { return this.getAttribute(${JSON.stringify(name)}); }`), + count: async (selector) => { + const id = await shadowId(); + if (!id) return 0; + const { nodeIds } = await cdp.send('DOM.querySelectorAll', { nodeId: id, selector }); + return nodeIds.length; + }, + async box(selector) { + const nodeId = await query(selector); + if (!nodeId) return null; + const { model } = await cdp.send('DOM.getBoxModel', { nodeId }); + const q = model.border; + const xs = [q[0], q[2], q[4], q[6]]; + const ys = [q[1], q[3], q[5], q[7]]; + return { x: Math.min(...xs), y: Math.min(...ys), width: Math.max(...xs) - Math.min(...xs), height: Math.max(...ys) - Math.min(...ys) }; + }, + async click(selector) { + const b = await api.box(selector); + if (!b) throw new Error(`dock has no ${selector}`); + await page.mouse.click(b.x + b.width / 2, b.y + b.height / 2); + }, + /** Focuses the element as a keyboard user would, so :focus-visible shows. */ + async focus(selector) { + await page.keyboard.press('Shift'); + const nodeId = await query(selector); + if (!nodeId) throw new Error(`dock has no ${selector}`); + await cdp.send('DOM.focus', { nodeId }); + }, + async waitFor(selector, { timeout = 15000 } = {}) { + const end = Date.now() + timeout; + while (Date.now() < end) { + if (await query(selector)) return true; + await sleep(150); + } + throw new Error(`dock never showed ${selector}`); + }, + async waitForText(re, { selector = '.dock', timeout = 15000 } = {}) { + const end = Date.now() + timeout; + let last = null; + while (Date.now() < end) { + last = await api.text(selector); + if (last && re.test(last)) return last; + await sleep(150); + } + throw new Error(`dock text never matched ${re}; last: ${last}`); + }, + }; + return api; +} From fae3ca7d9cf9e69639a0263a503af7c0142018bc Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 21:43:17 -0400 Subject: [PATCH 09/24] ProScan 2.4.0 --- manifest.json | 2 +- tests/e2e/upgrade.spec.mjs | 5 ++++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/manifest.json b/manifest.json index 603289a..1062eb8 100644 --- a/manifest.json +++ b/manifest.json @@ -1,7 +1,7 @@ { "manifest_version": 3, "name": "ProScan - Amazon Product Scraper", - "version": "2.3.0", + "version": "2.4.0", "description": "Scrape Amazon seller products with analytics and AI-powered product insights", "permissions": [ "storage", diff --git a/tests/e2e/upgrade.spec.mjs b/tests/e2e/upgrade.spec.mjs index 6a694c1..995d71a 100644 --- a/tests/e2e/upgrade.spec.mjs +++ b/tests/e2e/upgrade.spec.mjs @@ -3,8 +3,10 @@ // auto-update does. From the live store build (v2.0, commit 7c2ba1c), and // from 2.1 (commit 40be432) in the middle of a run. Based on the audit's // upgrade harness. +import fs from 'node:fs'; import path from 'node:path'; import { execFileSync } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; import { test, expect } from '@playwright/test'; import { launch, getState, waitForState, sleep, checkoutRevision, copyDir, enableDeveloperMode, clickStart, @@ -17,7 +19,8 @@ const V21_REV = '40be432'; const V20 = path.join(BUILD_DIR, 'v2.0'); const V21 = path.join(BUILD_DIR, 'v2.1'); const SLOT = path.join(BUILD_DIR, 'upgrade-slot'); -const CURRENT_VERSION = '2.3.0'; +// The build under test, whatever the manifest says it is. +const CURRENT_VERSION = JSON.parse(fs.readFileSync(fileURLToPath(new URL('../../manifest.json', import.meta.url)), 'utf8')).version; function bug(fid, what) { test.fail(!process.env.PROSCAN_SHOW_KNOWN, `${fid}: ${what}`); From 7a7380d9aad3298089d07ed98f4225541e4c8594 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 21:49:25 -0400 Subject: [PATCH 10/24] Document the dock: README and CLAUDE.md for 2.4 --- CLAUDE.md | 79 ++++++++++++++++++++++++++++---------------- README.md | 98 ++++++++++++++++++++++++++++++++++--------------------- 2 files changed, 111 insertions(+), 66 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index eb5a535..d7380e0 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -2,27 +2,29 @@ ## Project Overview -Chrome extension (Manifest V3) that scrapes Amazon seller product listings, provides analytics for resellers/arbitrage, and includes a floating AI chatbot on Amazon pages powered by Gemini API. No server required — everything runs client-side. +Chrome extension (Manifest V3) that scrapes Amazon search results and seller storefronts and saves them as Excel, CSV or JSON. Its main surface is an on-page dock, bottom-right on Amazon, that suggests Scrape on search and storefront pages, follows the run, downloads the files, and holds the Gemini chat (Ask). The popup is a fallback and holds the key and the optional sign-in. No server required; everything runs client-side. -## Architecture (v2.3) +## Architecture (v2.4) ```text AmazonSellerScraper/ -├── manifest.json # Extension config (v2.3) +├── manifest.json # Extension config (v2.4) ├── popup/ # UI Layer -│ ├── popup.html # Popup interface (dashboard + settings) +│ ├── popup.html # Popup: this tab, fallback Scrape, last scrape, settings │ ├── popup.css # Popup styling │ ├── popup.js # UI logic │ └── ai-key.js # Gemini key field (AI chat settings) ├── scripts/ │ ├── content/ │ │ ├── scraper.js # Parses a search page, reports it to the SW (parse only) -│ │ ├── chatbot.js # Floating AI chatbot widget (Shadow DOM) -│ │ └── offer-fetcher.js # Seller price fetching for spread analysis +│ │ ├── offer-fetcher.js # Seller price fetching for spread analysis (watchSpread for the dock) +│ │ ├── dock-styles.js # DOCK_CSS, adopted as a constructed stylesheet +│ │ └── dock.js # The on-page dock: Scrape, Ask, settings (closed Shadow DOM) │ ├── lib/ │ │ ├── parsers.js # Pure search/offer parsing (global Parsers) │ │ ├── messages.js # Every message type and who may send it (global Msg) │ │ ├── run.js # Run state machine, bound to one tab (global Run) +│ │ ├── page-kind.js # search / storefront / seller / store / never, and hidden pages (global PageKind) │ │ ├── flags.js # Build flags (global Flags); CLOUD_SYNC is on from 2.3 │ │ ├── migrate.js # Storage schema migrations, run by the SW │ │ └── chat.js # Gemini request builder, run scoping, error text @@ -30,6 +32,7 @@ AmazonSellerScraper/ │ │ ├── service-worker.js # Wires router, engine, chat, auth and migrations │ │ ├── router.js # The one typed message router │ │ ├── engine.js # The run engine; the only writer of run data +│ │ ├── download.js # DOWNLOAD for the dock: data: URL to chrome.downloads, or back to the page │ │ ├── db.js # IndexedDB: runs, products, placements, lastValues, outbox, spread │ │ ├── sync-plan.js # Outbox entry -> cloud writes (pure) │ │ └── sync.js # Drains the outbox into Firestore, one entry at a time @@ -40,8 +43,6 @@ AmazonSellerScraper/ │ └── exporter.js # Excel/CSV/JSON export ├── packages/ │ └── schema/index.js # Cloud schema: types, validators, ids (shared with the dashboard) -├── styles/ -│ └── chatbot.css # Chatbot widget styles (loaded into Shadow DOM) ├── tests/ # Test suite (Jest) │ ├── setup/ │ │ ├── chrome-mock.js # In-memory Chrome API mock @@ -69,9 +70,10 @@ AmazonSellerScraper/ | -------------------- | --------------------------------------------------------- | | `scraper.js` | Parses a page with `Parsers`, sends `PAGE_RESULT` | | `engine.js` | Run state machine, page saves, navigation, heartbeat | -| `chatbot.js` | Floating AI chatbot widget on Amazon pages (Shadow DOM) | +| `dock.js` | On-page dock: Scrape, progress, downloads, Ask, settings | +| `page-kind.js` | Which pages get Scrape, a storefront link, or no dock | | `offer-fetcher.js` | Fetches seller offer pages for price spread analysis | -| `chatbot.css` | Widget styles loaded into Shadow DOM | +| `download.js` | Files for the dock, made in the worker | | `storage.js` | Async wrapper for chrome.storage.local | | `analyzer.js` | Opportunity scoring, insights, statistics | | `spread-analyzer.js` | Price spread statistics (CV, std dev, arbitrage scoring) | @@ -84,7 +86,8 @@ AmazonSellerScraper/ The service worker is the only writer (`scripts/background/engine.js`). -1. User clicks "Start Scraping" in popup; `popup.js` sends `START_RUN {tabId}` to the SW +1. User presses Scrape in the dock; `dock.js` sends `START_RUN_HERE {maxPages?}` and the SW takes the tab + from `sender.tab.id` (`engine.startHere`). The popup's fallback sends `START_RUN {tabId}` 2. The SW pings the tab. No answer: the popup offers `chrome.tabs.reload` and starts after the reload. A page that is not a search (captcha, sign-in, product page) is refused and nothing changes 3. The SW writes the run record (`scripts/lib/run.js`) to `chrome.storage.session` under `run`, @@ -100,7 +103,11 @@ The service worker is the only writer (`scripts/background/engine.js`). 8. The run ends with a reason; Stop goes to the SW, which cancels the pending page. Closing the run's tab or taking it elsewhere ends the run as `interrupted` 9. `analyzer.js` generates insights and opportunity scores -10. User exports via `exporter.js` (Excel/CSV/JSON) +10. After each saved page the SW sends `RUN_PROGRESS` to the run's tab; the dock also polls `RUN_STATUS` + every 2 s while a run is live. `RUN_STATUS` (`engine.status`) is the run in brief (`brief`), whether it is + the asking tab's, and a `summarize` of the results once it ended. Never products, the key or the email +11. User downloads from the dock (`DOWNLOAD {format}`, made in the SW by `Exporter.build`) or the popup + (same `Exporter.build`, `Exporter.triggerDownload`) Run states: idle, starting, running, stopping, then stopped, blocked, failed or done. `reason` says why it ended: complete, stopped, blocked, selectors_broken, storage_full, @@ -137,22 +144,36 @@ Cloud paths and shapes: `packages/schema/index.js` (keep the dashboard's copy id on a shape change). Run id `{sourceId}_{startMs}`, minted once in `engine.start`; page id `p0001`; `dayKey` is the local date the run started. -### AI Chatbot (client-side, no server) - -1. `chatbot.js` injects a floating widget (bottom-right) on Amazon pages with product listings -2. Widget uses Shadow DOM to isolate styles from Amazon's CSS -3. User types a question (e.g. "What's the best deal under $30?") -4. `chatbot.js` sends `CHAT_MESSAGE` to `service-worker.js` with the question and the last few turns -5. The service worker reads the user's key from `chrome.storage.local` and the latest run from IndexedDB; the content script never sees the key -6. `scripts/lib/chat.js` builds the request: model id in `GEMINI_MODEL`, key in the `x-goog-api-key` header, titles in a fenced JSON block marked untrusted -7. Response displayed in chat bubble as text, never HTML +### The dock (2.4) + +1. `dock.js` runs on every Amazon page except cart, checkout, sign-in and account pages (`PageKind.hidden`). + Closed shadow root on `#proscan-dock-host`, `DOCK_CSS` adopted as a constructed stylesheet, system fonts, + inline SVG. No `web_accessible_resources` +2. Launcher: search or storefront suggests Scrape (one click starts); `/sp?seller=` and brand stores link to + `/s?me=ID`; elsewhere a plain ProScan button with Ask. A live run shows page x of y and a labeled Stop; + a finished run in this tab offers the Excel download. "Not now" sets `proscan.dock.notNow` in the tab's + sessionStorage; the card's open state is `proscan.dock.open` so it survives each page of a run +3. Card: Scrape tab (ready, running, done with the pages strip, hatched pages the store never had, downloads, + Compare seller prices through `runSpreadAnalysis` and `watchSpread`), Ask tab, settings (pages per run and + the suggestion via `SAVE_SETTINGS`; the key and sign-in open `popup.html#key` or `#sync` in a tab via + `OPEN_SETTINGS`) +4. Keys and passwords are never typed on Amazon. Key events stop at the dock so Amazon shortcuts do not fire + +### AI chat (client-side, no server) + +1. The dock's Ask tab asks `CHAT_STATUS` for the key state and the scan it covers +2. User types a question (e.g. "What's the best deal under $30?") +3. `dock.js` sends `CHAT_MESSAGE` to `service-worker.js` with the question and the last few turns +4. The service worker reads the user's key from `chrome.storage.local` and the latest run from IndexedDB; the content script never sees the key +5. `scripts/lib/chat.js` builds the request: model id in `GEMINI_MODEL`, key in the `x-goog-api-key` header, titles in a fenced JSON block marked untrusted +6. Response displayed in chat bubble as text, never HTML ## Setup 1. Load unpacked extension in `chrome://extensions` -2. Click the ProScan popup, expand AI chat settings, paste your Gemini API key (free at [aistudio.google.com/apikey](https://aistudio.google.com/apikey)) -3. Navigate to Amazon seller/search page → scrape → export -4. The AI chatbot button appears in the bottom-right corner on Amazon pages with product listings +2. Open an Amazon search or storefront and press Scrape on the dock (bottom-right), then Download Excel +3. For Ask, add a Gemini API key (free at [aistudio.google.com/apikey](https://aistudio.google.com/apikey)) from the + dock's settings (Add key) or the popup's AI chat settings ## Testing @@ -173,6 +194,8 @@ npm run test:coverage # With coverage report `firestore.rules` (`tools/contract.mjs`, `tests/contract/`); ESM files under `packages/` and `scripts/background/` load in Jest through `tests/setup/esm-to-cjs-transform.js` - `npm run test:e2e` runs the same scenarios in Chromium (`tests/e2e/`), including the worker stopped via CDP between pages and 2.0 and 2.1 builds updated mid-run +- `tests/unit/dock.test.js` runs `dock.js` in JSDOM with the shadow root forced open; `tests/e2e/dock.spec.mjs` drives + the real dock through CDP, which can pierce the closed root (`tests/e2e/lib/dock.mjs`) - HTML fixtures in `tests/fixtures/` match the exact CSS selectors the code uses - Chrome APIs (`storage`, `runtime`, `tabs`, `downloads`) are mocked in `tests/setup/chrome-mock.js` - XLSX is mocked with jest.fn() stubs in exporter.test.js; export-xlsx.test.js uses the real libs/xlsx.full.min.js @@ -183,8 +206,8 @@ npm run test:coverage # With coverage report - `chrome.storage.session`: the live run record (content scripts cannot read it) - IndexedDB: run data, in the extension origin; needs no permission - `chrome.runtime.sendMessage/onMessage`: messages, all in `scripts/lib/messages.js` -- `chrome.downloads`: file downloads -- `chrome.tabs`: `sendMessage`, `update`, `onRemoved`, `onUpdated`; none need the `tabs` permission +- `chrome.downloads`: file downloads, from the popup and from the SW for the dock (data: URL up to about 1.5 MB) +- `chrome.tabs`: `sendMessage`, `update`, `onRemoved`, `onUpdated`, `create` (settings page); none need the `tabs` permission No permission was added for the run engine: no `alarms`, `scripting`, `offscreen` or `unlimitedStorage`. `npm run lock` fails on any addition. @@ -300,8 +323,8 @@ See `docs/PRICE_SPREAD_ANALYSIS.md` for the full specification. - [x] In-popup analytics dashboard - [x] Opportunity scoring - [x] Price spread analysis (seller price variability detection) -- [x] Floating AI chatbot on Amazon pages (Gemini API, no server needed) -- [x] Shadow DOM isolation for chatbot widget +- [x] On-page dock with Scrape, progress, downloads and Ask (2.4) +- [x] Closed Shadow DOM isolation, no web-accessible files - [x] API key management in popup settings - [x] Comprehensive test suite (214 Jest tests) diff --git a/README.md b/README.md index 49d10c2..3d48e79 100644 --- a/README.md +++ b/README.md @@ -21,17 +21,17 @@ ## Overview -ProScan is a Chrome extension that scrapes Amazon product listings across multiple pages, runs analytics to identify arbitrage opportunities, and includes a floating AI chatbot powered by Google Gemini for real-time product Q&A -- all running client-side with no server required. +ProScan is a Chrome extension that scrapes Amazon search results and seller storefronts across multiple pages and saves them as Excel, CSV or JSON. It lives on the page: a small dock in the bottom-right corner offers **Scrape** when you are on a search or a storefront, follows the run page by page, and hands you the file when it is done. The same dock answers questions about what you scraped with Google Gemini. Everything runs client-side with no server required. ## Key Features +- **On-page dock** -- One button bottom-right on Amazon. On a search or a storefront it suggests Scrape, so there is nothing to pin or open. It shows the page it is on, how many products it has saved and a Stop that keeps them. "Not now" hides the suggestion for the rest of the tab - **Multi-page scraping** -- Automatically navigates and extracts product data (name, ASIN, price, rating, reviews, Prime status) across paginated Amazon results - **Opportunity scoring** -- Proprietary formula identifies high-value arbitrage opportunities based on rating, review velocity, and price positioning - **Price spread analysis** -- Fetches competing seller prices for each product and calculates variability (Coefficient of Variation) to identify pricing disagreement -- a strong arbitrage signal -- **AI chatbot** -- Floating widget on Amazon pages answers questions about your last scan using Gemini Flash and your own free API key (e.g., "What's the best deal under $30?") -- **Analytics dashboard** -- Real-time stats, underpriced product detection, and quality distribution analysis -- **Multi-format export** -- Excel (with styled sheets and charts), CSV, and JSON with full analytics -- **Shadow DOM isolation** -- Chatbot widget styles are fully isolated from Amazon's CSS +- **Ask** -- The dock's second tab answers questions about your last scan using Gemini Flash and your own free API key (e.g., "What's the best deal under $30?") +- **Multi-format export** -- Excel (with styled sheets and analytics), CSV, and JSON, straight from the dock or the popup. Spread data goes into the files when you have it +- **Stays out of Amazon's way** -- The dock is a closed Shadow DOM with system fonts and inline icons. It loads nothing from the network, follows your light or dark OS theme, and never appears on cart, checkout, sign-in or account pages ## Architecture @@ -41,14 +41,14 @@ ProScan is a Chrome extension that scrapes Amazon product listings across multip │ │ │ popup/ scripts/content/ │ │ ├── popup.html ├── scraper.js (DOM extraction) │ - │ ├── popup.css └── chatbot.js (AI widget) │ - │ └── popup.js │ - │ scripts/background/ │ - │ scripts/modules/ └── service-worker.js │ - │ ├── storage.js (message routing + Gemini API) │ - │ ├── analyzer.js │ - │ └── exporter.js styles/ │ - │ └── chatbot.css │ + │ ├── popup.css ├── dock.js (on-page UI) │ + │ └── popup.js └── dock-styles.js │ + │ (fallback, keys, │ + │ sign-in) scripts/background/ │ + │ ├── service-worker.js │ + │ scripts/modules/ │ (routing, Gemini, downloads) │ + │ ├── analyzer.js └── engine.js (runs scrapes) │ + │ └── exporter.js │ └──────────────────────────────────────────────────────────┘ ``` @@ -57,8 +57,9 @@ ProScan is a Chrome extension that scrapes Amazon product listings across multip ### Scraping Pipeline ``` -User clicks "Start Scraping" - → popup.js sends START_RUN to the service worker +User presses Scrape in the dock (or in the popup) + → dock.js sends START_RUN_HERE; the worker takes the tab from + sender.tab.id, never from the message. The popup sends START_RUN {tabId} → The worker pings the tab (the popup offers a reload if ProScan is not loaded there), makes a run bound to that tab and asks it to parse → scraper.js classifies the page, extracts products and sends PAGE_RESULT; @@ -70,10 +71,28 @@ User clicks "Start Scraping" → The run ends with a reason: complete, stopped, blocked (captcha, bot check, sign-in), selectors_broken, storage_full, interrupted or updated (the extension updated mid-run) + → After each saved page the worker sends RUN_PROGRESS to the run's tab. + Every page is a new document, so the dock rebuilds from RUN_STATUS → analyzer.js generates insights and opportunity scores - → User exports via exporter.js (Excel/CSV/JSON) + → Download in the dock sends DOWNLOAD {format}; the worker makes the file + with Exporter.build and saves it through chrome.downloads. A file over + about 1.5 MB goes back to the page, which saves it with a download link ``` +### The dock + +``` +On every Amazon page but cart, checkout, sign-in and account pages: + search results (/s?k=) or a storefront (/s?me=) → "Scrape" suggested + seller profile (/sp?seller=) or a brand store → "Open storefront" + anywhere else, product pages included → ProScan + Ask +Open, it is a card: Scrape (start, progress, stop, downloads, compare +seller prices) and Ask (the chat), plus settings (pages per run, the +suggestion on or off, links to the key and sign-in on ProScan's own page). +``` + +Keys and passwords are never typed on Amazon: keystrokes inside a shadow root still reach the page's own listeners. The dock's settings open ProScan's settings page instead. + ### Cloud Sync ``` @@ -101,8 +120,8 @@ The document shapes, ids and validators are in `packages/schema/index.js`, which ### AI Chatbot Flow ``` -User types question in floating widget - → chatbot.js sends CHAT_MESSAGE (question + last few turns) to service-worker.js +User types a question in the dock's Ask tab + → dock.js sends CHAT_MESSAGE (question + last few turns) to service-worker.js → Service worker reads the user's key and the latest run from its storage → scripts/lib/chat.js builds the Gemini request (model id is GEMINI_MODEL there) → Response displayed in chat bubble as plain text @@ -133,7 +152,7 @@ Normalized to a 1-10 scale. Higher score = better arbitrage opportunity. The for ### Price Spread Analysis -After scraping, click **Analyze Price Spreads** to fetch competing seller prices for each product. The system: +After scraping, click **Compare seller prices** in the dock or the popup to fetch competing seller prices for each product. The system: 1. Fetches the offer listing page for each ASIN (2-second delay between requests) 2. Extracts all seller prices using cascading DOM selectors @@ -167,7 +186,7 @@ See [docs/PRICE_SPREAD_ANALYSIS.md](docs/PRICE_SPREAD_ANALYSIS.md) for the full 3. Enable **Developer mode** (top-right toggle) 4. Click **Load unpacked** and select the `dist/` folder. The repo root does not load on its own, because the service worker has to be bundled. -To use the AI chat, open the ProScan popup, expand **AI chat settings** and paste a Gemini API key (free at [aistudio.google.com/apikey](https://aistudio.google.com/apikey)). The key stays in `chrome.storage.local`; only the service worker reads it and sends it to Google. +To use Ask, open the dock's settings and press **Add key** (or open the popup and expand **AI chat settings**), then paste a Gemini API key (free at [aistudio.google.com/apikey](https://aistudio.google.com/apikey)). The key stays in `chrome.storage.local`; only the service worker reads it and sends it to Google. For local Firebase work, `npm run build:dev` points the build at the emulators under the `demo-proscan` project. @@ -175,20 +194,19 @@ For local Firebase work, `npm run build:dev` points the build at the emulators u `npm test` runs the Jest suite and the tool tests. `npm run check` builds, then runs the permission lock (nothing may be added over `tools/live-manifest.json`, the published v2.0 manifest), the version gate and the secret scan. `npm run zip` writes the store package to `dist-zips/` and refuses a dev build, a stray file, uncommitted changes (`node tools/zip.mjs --allow-dirty` overrides that for local tries), or any gate failure. CI runs all of these. -The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, see its README). `npm run test:e2e` loads the built extension into Chromium and runs scrape scenarios against those pages, with every request answered locally. Run `npx playwright install --no-shell chromium` once first. Known bugs run as expected failures tagged with their audit finding id; `PROSCAN_SHOW_KNOWN=1 npm run test:e2e` shows what they fail on. +The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, see its README). `npm run test:e2e` loads the built extension into Chromium and runs scrape scenarios against those pages, with every request answered locally. `tests/e2e/dock.spec.mjs` drives the dock through its closed shadow root with CDP (`tests/e2e/lib/dock.mjs`). Run `npx playwright install --no-shell chromium` once first. Known bugs run as expected failures tagged with their audit finding id; `PROSCAN_SHOW_KNOWN=1 npm run test:e2e` shows what they fail on. `npm run test:contract` runs the real sync module against the Firebase emulators with the dashboard's `firestore.rules` (from `PROSCAN_RULES`, or a `web` or `proscan-web` checkout next to this repo): queues of 1, 201 and 600 products, a page added mid-flush, replace semantics across runs, create-only `firstSeenAt`, the write count per run, a product in two sources and an entry the rules refuse. It refuses to start if any of its ports is taken, and only stops the emulator processes it started. It needs the Firebase CLI and Java, and uses the `demo-proscan` project only. -The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, see its README). `npm run test:e2e` loads the built extension into Chromium and runs scrape scenarios against those pages, with every request answered locally. Run `npx playwright install --no-shell chromium` once first. Known bugs run as expected failures tagged with their audit finding id; `PROSCAN_SHOW_KNOWN=1 npm run test:e2e` shows what they fail on. - ## Usage -1. Navigate to any Amazon search results or seller page -2. Click the ProScan extension icon -3. Hit **Start Scraping** -- it will automatically paginate through results -4. View analytics in the dashboard (item count, average rating, average price) -5. Use the floating AI chatbot (bottom-right) to ask questions about products -6. Export data as **Excel** (multi-sheet with analytics), **CSV**, or **JSON** +1. Open an Amazon search or a seller storefront +2. Press **Scrape** on the ProScan dock in the bottom-right corner. It turns the pages in that tab +3. Watch the page count and the products saved, or minimize the dock and keep browsing in other tabs. **Stop** keeps what it has +4. Press **Download Excel**, or CSV or JSON +5. Switch to **Ask** to question the products you scraped + +The toolbar popup does the same Scrape and downloads, and holds the Gemini key and the optional dashboard sign-in. ## Tech Stack @@ -196,7 +214,7 @@ The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, s |-------|-----------|---------| | Extension | Vanilla JavaScript | UI, DOM scraping, export | | Extension | Chrome Manifest V3 | Extension framework | -| Extension | Shadow DOM | Chatbot style isolation | +| Extension | Shadow DOM (closed) + constructed stylesheet | Keeps the dock and Amazon's CSS apart | | Extension | XLSX.js | Excel generation | | AI | Google Gemini Flash (bring your own key) | Chatbot | @@ -206,8 +224,10 @@ The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, s - `chrome.storage.session` -- The live run record, written by the service worker - IndexedDB (the extension's own origin, no permission) -- Runs, products, pages, lastValues and the sync outbox. Starting a run keeps the 10 newest runs and removes older ones, except runs still waiting to sync. - `chrome.runtime.sendMessage` / `onMessage` -- Messages, all listed in `scripts/lib/messages.js` -- `chrome.downloads` -- File export downloads -- `chrome.tabs` -- Messages to the run's tab, `tabs.update` for the next page, and `onRemoved` / `onUpdated` to notice the tab going away (none of these need the `tabs` permission) +- `chrome.downloads` -- File downloads, from the popup and, for the dock, from the service worker +- `chrome.tabs` -- Messages to the run's tab, `tabs.update` for the next page, `onRemoved` / `onUpdated` to notice the tab going away, and `tabs.create` to open ProScan's settings page from the dock (none of these need the `tabs` permission) + +2.4 adds no permission, host or match pattern, and drops `web_accessible_resources`: the dock needs no file from the extension. ## Project Structure @@ -216,17 +236,20 @@ AmazonSellerScraper/ ├── manifest.json # Extension config (Manifest V3) ├── popup/ │ ├── popup.html # Extension popup interface -│ ├── popup.css # Popup styling (dark theme) -│ └── popup.js # UI state management and export handling +│ ├── popup.css # Popup styling (light and dark) +│ ├── popup.js # Fallback Scrape, downloads, settings +│ └── ai-key.js # The Gemini key field ├── scripts/ │ ├── content/ │ │ ├── scraper.js # Parses a search page and reports it to the worker -│ │ ├── chatbot.js # Floating AI chatbot (Shadow DOM) -│ │ └── offer-fetcher.js # Seller offer page fetching for spread analysis +│ │ ├── offer-fetcher.js # Seller offer page fetching for spread analysis +│ │ ├── dock-styles.js # The dock's stylesheet, as a string +│ │ └── dock.js # The on-page dock: Scrape, Ask, settings (closed Shadow DOM) │ ├── lib/ │ │ ├── parsers.js # Pure search and offer page parsing │ │ ├── messages.js # Every message type and who may send it │ │ ├── run.js # The run state machine and its end reasons +│ │ ├── page-kind.js # Search, storefront, seller or a page to stay off │ │ ├── flags.js # Build flags (cloud sync is on from 2.3) │ │ ├── migrate.js # Storage schema migrations (schemaVersion) │ │ └── chat.js # Gemini request builder and run scoping @@ -234,6 +257,7 @@ AmazonSellerScraper/ │ │ ├── service-worker.js # Wires the router, the engine, the chat and migrations │ │ ├── router.js # The one message router │ │ ├── engine.js # Runs scrapes; the only writer of run data +│ │ ├── download.js # Files for the dock, through chrome.downloads │ │ ├── db.js # IndexedDB stores │ │ ├── sync-plan.js # What each outbox entry writes (pure) │ │ └── sync.js # Drains the outbox into Firestore @@ -244,8 +268,6 @@ AmazonSellerScraper/ │ └── exporter.js # Multi-format export (Excel/CSV/JSON) ├── packages/ │ └── schema/index.js # Cloud schema shared with the dashboard -├── styles/ -│ └── chatbot.css # Chatbot widget styles (Shadow DOM) ├── libs/ │ └── xlsx.full.min.js # Excel generation library └── assets/ From 0f28aafc7f4d019f61a57e5f57d825357d6ef574 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 21:50:01 -0400 Subject: [PATCH 11/24] Poll less from background tabs while a run goes --- scripts/content/dock.js | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/scripts/content/dock.js b/scripts/content/dock.js index 7ea1b62..506a332 100644 --- a/scripts/content/dock.js +++ b/scripts/content/dock.js @@ -252,7 +252,8 @@ ui.orphaned = true; } render(); - if (isLive(run())) pollTimer = setTimeout(refresh, POLL_MS); + // A background tab checks in less often; RUN_PROGRESS still reaches the run's own tab. + if (isLive(run())) pollTimer = setTimeout(refresh, document.hidden ? POLL_MS * 5 : POLL_MS); } async function startHere(maxPages) { From 258ee59872518c08a32a9b1a47fb5106ea62395d Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 21:51:05 -0400 Subject: [PATCH 12/24] Show the download icon on the popup's Excel button from the start --- popup/popup.html | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/popup/popup.html b/popup/popup.html index 14bab56..3694fb6 100644 --- a/popup/popup.html +++ b/popup/popup.html @@ -49,7 +49,7 @@

Last scrape

-
Avg rating
From da1990b03417172cc2de7891ce5ec123db22ec9c Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 22:09:03 -0400 Subject: [PATCH 13/24] Name a storefront run by the store, not its seller id The run keeps the store name from the tab title, so the dock, the chat scope line and the popup say "Cedar & Pine Home" instead of a raw seller id. The name stays local; the cloud source is built field by field. --- scripts/background/engine.js | 20 +++++++++++++++++--- scripts/content/scraper.js | 2 +- scripts/lib/chat.js | 1 + tests/unit/chat.test.js | 7 +++++++ tests/unit/engine-dock.test.js | 12 ++++++++++-- 5 files changed, 36 insertions(+), 6 deletions(-) diff --git a/scripts/background/engine.js b/scripts/background/engine.js index 10a39cf..fb57f1f 100644 --- a/scripts/background/engine.js +++ b/scripts/background/engine.js @@ -111,6 +111,19 @@ function summarize(results) { return { medianCents, avgRating, sponsoredPct }; } +/** + * A storefront's name from its tab title ("Amazon.com: Northfield Goods"), + * or null. Kept on the run for labels only; it never goes to the cloud. + */ +function storeNameOf(title) { + const t = String(title == null ? '' : title) + .replace(/[\u0000-\u001f\u007f-\u009f\u200b-\u200f\u202a-\u202e\u2066-\u2069]/g, ' ') + .replace(/\s+/g, ' ') + .replace(/^\s*Amazon\.com\s*[:|-]?\s*/i, '') + .trim(); + return t && t.length <= 60 && !/^amazon/i.test(t) ? t : null; +} + /** The run as the page may see it: no product rows, no tab ids but its own. */ function brief(run, tabId) { if (!run) return null; @@ -124,7 +137,7 @@ function brief(run, tabId) { itemCount: run.itemCount || 0, startedAt: run.startedAt || null, finishedAt: run.finishedAt || null, - source: { type: src.type || null, sellerId: src.sellerId || null, keyword: src.keyword || null }, + source: { type: src.type || null, sellerId: src.sellerId || null, keyword: src.keyword || null, name: src.name || null }, thisTab: typeof tabId === 'number' && run.tabId === tabId }; } @@ -287,10 +300,11 @@ function createEngine({ // One id for the run, here and in the cloud: {sourceId}_{startMs}. const found = Schema.sourceOf(pong.url); const sourceId = Schema.sourceIdOf(found); + const name = found.type === 'storefront' ? storeNameOf(pong.title) : null; const runId = Schema.runIdOf(sourceId, t); let run = Run.create({ runId, tabId, maxPages: settings.maxPages, now: t, - source: { ...found, sourceId, startedAt: new Date(t).toISOString() } + source: { ...found, sourceId, name, startedAt: new Date(t).toISOString() } }); run = { ...run, sourceId, dayKey: Schema.dayKeyOf(t, new Date(t).getTimezoneOffset()) }; await saveRun(run); @@ -668,4 +682,4 @@ function createEngine({ }; } -module.exports = { createEngine, foldPage, durable, prevFor, summarize, brief, PAGE_TIMEOUT_MS, KEEP_RUNS }; +module.exports = { createEngine, foldPage, durable, prevFor, summarize, brief, storeNameOf, PAGE_TIMEOUT_MS, KEEP_RUNS }; diff --git a/scripts/content/scraper.js b/scripts/content/scraper.js index a285920..32dfa86 100644 --- a/scripts/content/scraper.js +++ b/scripts/content/scraper.js @@ -131,7 +131,7 @@ chrome.runtime.onMessage.addListener((request, sender, sendResponse) => { if (request.type === Msg.T.PING) { const page = parsePage(); - sendResponse({ ok: true, kind: page.kind, count: page.products.length, url: window.location.href }); + sendResponse({ ok: true, kind: page.kind, count: page.products.length, url: window.location.href, title: String(document.title || '').slice(0, 200) }); return false; } diff --git a/scripts/lib/chat.js b/scripts/lib/chat.js index 88919db..c97d05b 100644 --- a/scripts/lib/chat.js +++ b/scripts/lib/chat.js @@ -76,6 +76,7 @@ const Chat = (() => { function sourceLabel(meta) { if (!meta) return ''; if (meta.keyword) return 'search "' + cleanText(meta.keyword, 80) + '"'; + if (meta.name) return cleanText(meta.name, 60); if (meta.sellerId) return 'storefront ' + cleanText(meta.sellerId, 40); return ''; } diff --git a/tests/unit/chat.test.js b/tests/unit/chat.test.js index a40c5e8..ac5c549 100644 --- a/tests/unit/chat.test.js +++ b/tests/unit/chat.test.js @@ -245,3 +245,10 @@ describe('cleanText', () => { expect(/[\u200b-\u200f\u202a-\u202e\u2066-\u2069]/.test(src)).toBe(false); }); }); + +test('a storefront scan is labeled by its name when the run has one', () => { + const named = Chat.buildContext({ ...baseData, scrapeRunMeta: { type: 'storefront', sellerId: 'A1B2C3', name: 'Northfield Goods' } }); + expect(named.source.label).toBe('Northfield Goods'); + const bare = Chat.buildContext({ ...baseData, scrapeRunMeta: { type: 'storefront', sellerId: 'A1B2C3' } }); + expect(bare.source.label).toBe('storefront A1B2C3'); +}); diff --git a/tests/unit/engine-dock.test.js b/tests/unit/engine-dock.test.js index fabacbf..9cf9e9e 100644 --- a/tests/unit/engine-dock.test.js +++ b/tests/unit/engine-dock.test.js @@ -6,7 +6,7 @@ * the products or the key, and the run's tab hears about each saved page. */ const { createRig, settle } = require('../setup/engine-rig'); -const { summarize, brief } = require('../../scripts/background/engine'); +const { summarize, brief, storeNameOf } = require('../../scripts/background/engine'); const Run = require('../../scripts/lib/run'); const asinFor = (k, p, i) => `B0${k.slice(0, 3).toUpperCase().padEnd(3, 'X')}${String(p).padStart(2, '0')}${String(i).padStart(3, '0')}`; @@ -154,7 +154,15 @@ test('brief keeps the run shape small', () => { const run = { runId: 'r', tabId: 3, state: 'running', page: 2, maxPages: 10, itemCount: 5, source: { type: 'keyword', keyword: 'x', url: 'u' } }; expect(brief(run, 3)).toEqual({ runId: 'r', state: 'running', reason: null, page: 2, maxPages: 10, itemCount: 5, startedAt: null, finishedAt: null, - source: { type: 'keyword', sellerId: null, keyword: 'x' }, thisTab: true, + source: { type: 'keyword', sellerId: null, keyword: 'x', name: null }, thisTab: true, }); expect(brief(null, 3)).toBeNull(); }); + +test('storeNameOf reads a storefront name from its tab title', () => { + expect(storeNameOf('Amazon.com: Northfield Goods')).toBe('Northfield Goods'); + expect(storeNameOf('Amazon.com: Cedar‮ & Pine')).toBe('Cedar & Pine'); + expect(storeNameOf('Amazon.com')).toBeNull(); + expect(storeNameOf('x'.repeat(61))).toBeNull(); + expect(storeNameOf(undefined)).toBeNull(); +}); From f8dff7a888bc4370eff10c02871c3a7ad2206531 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 22:09:04 -0400 Subject: [PATCH 14/24] Tidy the dock after the first visual review Ask on the seller profile launcher too, say "Reached the last page" once in the done header and keep the strip line to why the run ended early, shorten "Scrape this page again", and draw the compose focus as one ring. --- scripts/content/dock-styles.js | 2 +- scripts/content/dock.js | 11 ++++++++--- tests/e2e/dock.spec.mjs | 2 +- tests/unit/dock.test.js | 2 +- 4 files changed, 11 insertions(+), 6 deletions(-) diff --git a/scripts/content/dock-styles.js b/scripts/content/dock-styles.js index eb17f57..946ad51 100644 --- a/scripts/content/dock-styles.js +++ b/scripts/content/dock-styles.js @@ -193,7 +193,7 @@ svg { display: block; flex: none; } .chip { border: 1px solid var(--line-strong); border-radius: 999px; padding: 6px 12px; font-size: 13px; background: var(--surface); } .chip:hover { background: var(--sunken); } .compose { display: flex; gap: 8px; align-items: center; border: 1px solid var(--line-strong); border-radius: 14px; padding: 5px 5px 5px 14px; background: var(--surface); } -.compose:focus-within { outline: 2px solid var(--ring); outline-offset: 2px; } +.compose:focus-within { border-color: var(--ring); box-shadow: 0 0 0 1px var(--ring); } .compose input { flex: 1; border: 0; background: none; color: var(--ink); min-width: 0; padding: 6px 0; } .compose input:focus-visible { outline: none; } .compose input::placeholder { color: var(--muted); } diff --git a/scripts/content/dock.js b/scripts/content/dock.js index 506a332..5853fd4 100644 --- a/scripts/content/dock.js +++ b/scripts/content/dock.js @@ -219,6 +219,7 @@ function runLabel(r) { const src = (r && r.source) || {}; if (src.type === 'storefront') { + if (src.name) return src.name; if (page.kind === 'storefront' && page.sellerId === src.sellerId && page.name) return page.name; return 'this storefront'; } @@ -440,7 +441,7 @@ function endLine(r) { if (r.reason !== 'complete') return null; if ((r.page || 0) < r.maxPages) { - return `Reached the last page. Page ${fmt(r.page)} was the last one, so the run ended early.`; + return `Page ${fmt(r.page)} was the last one, so the run ended early.`; } return `Stopped at your ${fmt(r.maxPages)}-page limit.`; } @@ -536,6 +537,7 @@ return launcherShell('suggest', mainButton('Open ProScan', page.name || 'This seller', 'Has a storefront to scrape', mark()), h('a', { class: 'pill', 'data-k': 'storefront', href: PageKind.storefrontUrl(page.sellerId) }, 'Open storefront'), + chatButton(), hideButton('Not now, hide for this tab')); } return launcherShell('plain', @@ -665,7 +667,10 @@ if (r.reason === 'complete') { kind = 'ok'; title = `${plural(n, 'product')} saved`; - sub = `${plural(r.page, 'page')}${r.finishedAt ? `, finished at ${clock(r.finishedAt)}` : ''}`; + const at = r.finishedAt ? ` at ${clock(r.finishedAt)}` : ''; + sub = (r.page || 0) < r.maxPages + ? `Reached the last page${at}, ${plural(r.page, 'page')} in all.` + : `${plural(r.page, 'page')}, finished${at}.`; } else if (r.reason === 'stopped') { kind = 'neutral'; title = `Stopped with ${plural(n, 'product')}`; @@ -691,7 +696,7 @@ if (scrapable && page.startable && !ui.spread) { out.push(h('div', { class: 'split' }, h('span', {}), - h('button', { class: 'quiet', 'data-k': 'again', onclick: () => { ui.fresh = true; ui.notice = null; pendingFocus = 'first'; render(); }, text: 'Scrape this page again' }))); + h('button', { class: 'quiet', 'data-k': 'again', onclick: () => { ui.fresh = true; ui.notice = null; pendingFocus = 'first'; render(); }, text: 'Scrape again' }))); } return out; } diff --git a/tests/e2e/dock.spec.mjs b/tests/e2e/dock.spec.mjs index 729d9ad..06cd302 100644 --- a/tests/e2e/dock.spec.mjs +++ b/tests/e2e/dock.spec.mjs @@ -65,7 +65,7 @@ test('Scrape in the dock runs in that tab, page by page, and completes', async ( // The last page's dock rebuilt itself from the worker and shows the result. const after = await dock(tab); await after.waitForText(/12 products saved/, { timeout: 10000 }); - await after.waitForText(/Reached the last page\. Page 3 was the last one/); + await after.waitForText(/Reached the last page.*Page 3 was the last one/); expect(await after.count('.ticks i.skip')).toBe(17); }); diff --git a/tests/unit/dock.test.js b/tests/unit/dock.test.js index b7e0607..00a13f3 100644 --- a/tests/unit/dock.test.js +++ b/tests/unit/dock.test.js @@ -173,7 +173,7 @@ test('a run that ran out of pages hatches the rest and says so', async () => { const d = mount(SEARCH, YOGA, { RUN_STATUS: done, DOWNLOAD: { ok: true, via: 'downloads', filename: 'x.xlsx' } }, { session: { 'proscan.dock.open': '1' } }); await flush(); expect(d.text()).toContain('912 products saved'); - expect(d.text()).toContain('Reached the last page. Page 19 was the last one, so the run ended early.'); + expect(d.text()).toContain('Page 19 was the last one, so the run ended early.'); expect(d.$$('.ticks i.done')).toHaveLength(19); expect(d.$$('.ticks i.skip')).toHaveLength(1); expect(d.text()).toContain('$27.40'); From 2336f2212f445e760d62cad3b57f502db53d6bd5 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 22:09:05 -0400 Subject: [PATCH 15/24] Match the popup to the dock: median price, named Scrape button The last-scrape card shows the median price like the dock does, the fallback button says Scrape this storefront or Scrape this search, the saved-key placeholder fits the field, and fields focus with one ring. --- popup/ai-key.js | 2 +- popup/popup.css | 2 +- popup/popup.html | 2 +- popup/popup.js | 27 +++++++++++++++++++++------ tests/unit/ai-key.test.js | 2 +- 5 files changed, 25 insertions(+), 10 deletions(-) diff --git a/popup/ai-key.js b/popup/ai-key.js index 0cecc56..9b39b02 100644 --- a/popup/ai-key.js +++ b/popup/ai-key.js @@ -42,7 +42,7 @@ const AiKey = (() => { const show = (set, message) => { status.textContent = message || (set ? 'Key saved. The chat button on Amazon is ready.' : 'No key set. AI chat is off.'); - input.placeholder = set ? 'Key saved (hidden). Paste a new one to replace it.' : 'Paste your Gemini API key'; + input.placeholder = set ? 'Paste a new key to replace it' : 'Paste your Gemini API key'; clearBtn.classList.toggle('hidden', !set); }; diff --git a/popup/popup.css b/popup/popup.css index 2cd0c34..9e406ee 100644 --- a/popup/popup.css +++ b/popup/popup.css @@ -155,7 +155,7 @@ main { padding: 8px 18px 16px; display: flex; flex-direction: column; gap: 18px; .fold-body { padding: 0 14px 14px; display: flex; flex-direction: column; gap: 8px; } .field { width: 100%; height: 38px; border-radius: 10px; border: 1px solid var(--line-strong); background: var(--bg); padding: 0 12px; color: var(--ink); } .field::placeholder { color: var(--muted); } -.field:focus-visible { outline: 2px solid var(--ring); outline-offset: 1px; } +.field:focus-visible { outline: none; border-color: var(--ring); box-shadow: 0 0 0 1px var(--ring); } .ai-key-actions { display: flex; gap: 8px; align-items: center; } .ai-key-actions .secondary { width: auto; } .help { font-size: 12px; color: var(--muted); } diff --git a/popup/popup.html b/popup/popup.html index 3694fb6..4719582 100644 --- a/popup/popup.html +++ b/popup/popup.html @@ -45,7 +45,7 @@

Last scrape

-
-
Avg price
+
-
Median price
-
Avg rating