From cb5b19d2bb2d1f6d300a8aa7f24157a635cef081 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:15:14 -0400 Subject: [PATCH 01/25] Add a message catalog and a typed router for the service worker --- scripts/background/router.js | 61 +++++++++++++++++++++++++++ scripts/lib/messages.js | 81 ++++++++++++++++++++++++++++++++++++ tests/unit/router.test.js | 73 ++++++++++++++++++++++++++++++++ 3 files changed, 215 insertions(+) create mode 100644 scripts/background/router.js create mode 100644 scripts/lib/messages.js create mode 100644 tests/unit/router.test.js diff --git a/scripts/background/router.js b/scripts/background/router.js new file mode 100644 index 0000000..7017e92 --- /dev/null +++ b/scripts/background/router.js @@ -0,0 +1,61 @@ +/** + * @fileoverview The service worker's one message router. + * + * Handlers are registered by message type from scripts/lib/messages.js and + * may return a value or a promise. The router answers with what the handler + * returns, or {error} when it throws. A message meant for someone else, an + * unknown type or a sender of the wrong kind is not answered, and the + * channel is not held open for it. + * + * @module Router + */ + +const Msg = require('../lib/messages.js'); + +function createRouter({ extensionId = null, log = console } = {}) { + const handlers = new Map(); + + function on(type, handler) { + if (!Msg.CATALOG[type] || Msg.CATALOG[type].to !== 'worker') { + throw new Error(`not a worker message: ${type}`); + } + if (handlers.has(type)) throw new Error(`handler already set for ${type}`); + handlers.set(type, handler); + return api; + } + + /** The chrome.runtime.onMessage listener. */ + function listener(message, sender, sendResponse) { + const entry = Msg.spec(message); + const handler = entry && handlers.get(message.type); + if (!handler) return false; + if (!Msg.senderAllowed(entry, sender, extensionId)) { + log.warn('[ProScan] Refused', message.type, 'from the wrong sender'); + return false; + } + let result; + try { + result = handler(message, sender); + } catch (err) { + sendResponse({ error: (err && err.message) || String(err) }); + return false; + } + if (!result || typeof result.then !== 'function') { + sendResponse(result === undefined ? { ok: true } : result); + return false; + } + result.then( + (value) => sendResponse(value === undefined ? { ok: true } : value), + (err) => { + log.error('[ProScan]', message.type, 'failed:', err && err.message); + sendResponse({ error: (err && err.message) || String(err) }); + } + ); + return true; + } + + const api = { on, listener, handles: (type) => handlers.has(type) }; + return api; +} + +module.exports = { createRouter }; diff --git a/scripts/lib/messages.js b/scripts/lib/messages.js new file mode 100644 index 0000000..260b377 --- /dev/null +++ b/scripts/lib/messages.js @@ -0,0 +1,81 @@ +/** + * @fileoverview Every message the extension sends, in one place. + * + * `to` says who handles a message: the service worker ('worker'), the + * content script in a tab ('tab'), or any open popup ('popup'). `from` says + * who may send a worker message: 'tab' (a content script, so sender.tab is + * set), 'page' (an extension page such as the popup), or 'any'. The worker + * router refuses a message from the wrong kind of sender. + * + * Loaded as a plain global (`Msg`) in the popup and content scripts, and as + * a CommonJS module in Jest and the bundled service worker. + * + * @module Msg + */ + +const Msg = (() => { + const CATALOG = { + // Popup -> worker: the run + START_RUN: { to: 'worker', from: 'page' }, + STOP_RUN: { to: 'worker', from: 'page' }, + GET_STATE: { to: 'worker', from: 'page' }, + + // Content script -> worker: the run + PAGE_READY: { to: 'worker', from: 'tab' }, + PAGE_RESULT: { to: 'worker', from: 'tab' }, + HEARTBEAT: { to: 'worker', from: 'tab' }, + + // Spread analysis in the tab + GET_RESULTS: { to: 'worker', from: 'any' }, + SPREAD_RESULT: { to: 'worker', from: 'tab' }, + + // Chat + CHAT_STATUS: { to: 'worker', from: 'any' }, + CHAT_MESSAGE: { to: 'worker', from: 'tab' }, + + // Cloud account and export + PROSCAN_AUTH_STATE: { to: 'worker', from: 'page' }, + PROSCAN_SIGN_IN: { to: 'worker', from: 'page' }, + PROSCAN_SIGN_OUT: { to: 'worker', from: 'page' }, + PROSCAN_EXPORT: { to: 'worker', from: 'page' }, + + // Worker or popup -> content script + PING: { to: 'tab' }, + PARSE_PAGE: { to: 'tab' }, + RUN_ENDED: { to: 'tab' }, + START_SPREAD_ANALYSIS: { to: 'tab' }, + STOP_SPREAD_ANALYSIS: { to: 'tab' }, + + // Content script -> popup + SPREAD_PROGRESS: { to: 'popup' }, + SPREAD_ANALYSIS_COMPLETE: { to: 'popup' } + }; + + const T = Object.freeze(Object.fromEntries(Object.keys(CATALOG).map(k => [k, k]))); + + /** The catalog entry for a message, or null for an unknown or malformed one. */ + function spec(message) { + if (!message || typeof message !== 'object' || typeof message.type !== 'string') return null; + return Object.prototype.hasOwnProperty.call(CATALOG, message.type) ? CATALOG[message.type] : null; + } + + /** + * Whether `sender` may send a worker message with this spec. A content + * script sender always has a tab; an extension page never does. + */ + function senderAllowed(entry, sender, extensionId) { + if (!entry || entry.to !== 'worker') return false; + const fromTab = !!(sender && sender.tab && typeof sender.tab.id === 'number'); + const ours = !sender || !sender.id || !extensionId || sender.id === extensionId; + if (!ours) return false; + if (entry.from === 'tab') return fromTab; + if (entry.from === 'page') return !fromTab; + return true; + } + + return { T, CATALOG, spec, senderAllowed }; +})(); + +if (typeof module !== 'undefined' && module.exports) { + module.exports = Msg; +} diff --git a/tests/unit/router.test.js b/tests/unit/router.test.js new file mode 100644 index 0000000..762ac32 --- /dev/null +++ b/tests/unit/router.test.js @@ -0,0 +1,73 @@ +const Msg = require('../../scripts/lib/messages'); +const { createRouter } = require('../../scripts/background/router'); + +const quiet = { warn() {}, error() {} }; +const TAB = { id: 'ext', tab: { id: 4 } }; +const PAGE = { id: 'ext', url: 'chrome-extension://ext/popup/popup.html' }; + +function call(router, message, sender) { + return new Promise((resolve) => { + const held = router.listener(message, sender, resolve); + if (!held) setTimeout(() => resolve('no answer'), 0); + }); +} + +test('every worker message in the catalog says who may send it', () => { + for (const [type, entry] of Object.entries(Msg.CATALOG)) { + expect(Msg.T[type]).toBe(type); + if (entry.to === 'worker') expect(['tab', 'page', 'any']).toContain(entry.from); + } +}); + +test('a handler answers with its value, sync or async', async () => { + const r = createRouter({ extensionId: 'ext', log: quiet }); + r.on('GET_STATE', () => ({ run: null })); + r.on('HEARTBEAT', async () => ({ active: true })); + expect(await call(r, { type: 'GET_STATE' }, PAGE)).toEqual({ run: null }); + expect(await call(r, { type: 'HEARTBEAT' }, TAB)).toEqual({ active: true }); +}); + +test('a throwing or rejecting handler answers with an error', async () => { + const r = createRouter({ extensionId: 'ext', log: quiet }); + r.on('STOP_RUN', () => { throw new Error('boom'); }); + r.on('START_RUN', async () => { throw new Error('later'); }); + expect(await call(r, { type: 'STOP_RUN' }, PAGE)).toEqual({ error: 'boom' }); + expect(await call(r, { type: 'START_RUN' }, PAGE)).toEqual({ error: 'later' }); +}); + +test('page-only messages are refused from a tab, and tab-only ones from a page', async () => { + const r = createRouter({ extensionId: 'ext', log: quiet }); + const start = jest.fn(() => ({ ok: true })); + const result = jest.fn(() => ({ ok: true })); + r.on('START_RUN', start); + r.on('PAGE_RESULT', result); + expect(await call(r, { type: 'START_RUN' }, TAB)).toBe('no answer'); + expect(await call(r, { type: 'PAGE_RESULT' }, PAGE)).toBe('no answer'); + expect(start).not.toHaveBeenCalled(); + expect(result).not.toHaveBeenCalled(); +}); + +test('another extension is refused', async () => { + const r = createRouter({ extensionId: 'ext', log: quiet }); + const h = jest.fn(); + r.on('GET_RESULTS', h); + expect(await call(r, { type: 'GET_RESULTS' }, { id: 'other', tab: { id: 1 } })).toBe('no answer'); + expect(h).not.toHaveBeenCalled(); +}); + +test('unknown, malformed and popup-bound messages do not hold the channel', () => { + const r = createRouter({ extensionId: 'ext', log: quiet }); + const respond = jest.fn(); + for (const m of [null, 'PING', {}, { type: 'NOPE' }, { type: 'SPREAD_PROGRESS' }, { type: 'PING' }]) { + expect(r.listener(m, TAB, respond)).toBe(false); + } + expect(respond).not.toHaveBeenCalled(); +}); + +test('only worker messages can be registered, once each', () => { + const r = createRouter({ log: quiet }); + expect(() => r.on('PING', () => {})).toThrow(/not a worker message/); + expect(() => r.on('MADE_UP', () => {})).toThrow(/not a worker message/); + r.on('GET_STATE', () => {}); + expect(() => r.on('GET_STATE', () => {})).toThrow(/already/); +}); From 24f2b0024b3828015f9882acb9858d048f7cfece Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:17:04 -0400 Subject: [PATCH 02/25] Add an IndexedDB store for runs, products, pages, lastValues and the outbox --- package-lock.json | 11 +++ package.json | 1 + scripts/background/db.js | 206 +++++++++++++++++++++++++++++++++++++++ tests/unit/db.test.js | 82 ++++++++++++++++ 4 files changed, 300 insertions(+) create mode 100644 scripts/background/db.js create mode 100644 tests/unit/db.test.js diff --git a/package-lock.json b/package-lock.json index 6732d39..a1e089c 100644 --- a/package-lock.json +++ b/package-lock.json @@ -12,6 +12,7 @@ "@playwright/test": "1.63.0", "adm-zip": "^0.5.17", "esbuild": "^0.28.0", + "fake-indexeddb": "^6.2.5", "jest": "^29.7.0", "jest-environment-jsdom": "^29.7.0", "jsdom": "^20.0.3" @@ -3196,6 +3197,16 @@ "node": "^14.15.0 || ^16.10.0 || >=18.0.0" } }, + "node_modules/fake-indexeddb": { + "version": "6.2.5", + "resolved": "https://registry.npmjs.org/fake-indexeddb/-/fake-indexeddb-6.2.5.tgz", + "integrity": "sha512-CGnyrvbhPlWYMngksqrSSUT1BAVP49dZocrHuK0SvtR0D5TMs5wP0o3j7jexDJW01KSadjBp1M/71o/KR3nD1w==", + "dev": true, + "license": "Apache-2.0", + "engines": { + "node": ">=18" + } + }, "node_modules/fast-json-stable-stringify": { "version": "2.1.0", "resolved": "https://registry.npmjs.org/fast-json-stable-stringify/-/fast-json-stable-stringify-2.1.0.tgz", diff --git a/package.json b/package.json index 279823d..dbad2d1 100644 --- a/package.json +++ b/package.json @@ -19,6 +19,7 @@ "@playwright/test": "1.63.0", "adm-zip": "^0.5.17", "esbuild": "^0.28.0", + "fake-indexeddb": "^6.2.5", "jest": "^29.7.0", "jest-environment-jsdom": "^29.7.0", "jsdom": "^20.0.3" diff --git a/scripts/background/db.js b/scripts/background/db.js new file mode 100644 index 0000000..07de098 --- /dev/null +++ b/scripts/background/db.js @@ -0,0 +1,206 @@ +/** + * @fileoverview IndexedDB for everything a run collects. + * + * The service worker is the only user. chrome.storage.local keeps only + * settings and schemaVersion; this database keeps the rest, in the + * extension's own origin, which needs no permission and has far more room. + * + * Stores: + * - runs {runId, source, state, reason, pages[], itemCount, ...} + * - products one record per ASIN per run, key [runId, n] where n is the + * order it was found in; index runId + * - placements one page of a run, key [runId, pageIndex]; index runId + * - lastValues {asin, priceCents, rating, reviewCount, scrapedAt, ...}; + * index scrapedAt + * - outbox products waiting for cloud sync, key seq; index runId + * - spread offer prices per product, key [runId, asin]; index runId + * - meta {key, value}, e.g. latestRunId + * + * Reads and writes each use one transaction. A write that runs out of room + * rejects with a StorageError whose code is 'storage_full'. + * + * @module DB + */ + +const NAME = 'proscan'; +const VERSION = 1; + +function storageError(err) { + const e = new Error((err && err.message) || String(err || 'IndexedDB failed')); + e.name = 'StorageError'; + e.code = err && (err.name === 'QuotaExceededError' || /quota/i.test(err.message || '')) ? 'storage_full' : 'storage_error'; + return e; +} + +function upgrade(db) { + const has = (n) => db.objectStoreNames.contains(n); + if (!has('runs')) db.createObjectStore('runs', { keyPath: 'runId' }); + if (!has('products')) db.createObjectStore('products', { keyPath: ['runId', 'n'] }).createIndex('runId', 'runId'); + if (!has('placements')) db.createObjectStore('placements', { keyPath: ['runId', 'pageIndex'] }).createIndex('runId', 'runId'); + if (!has('lastValues')) db.createObjectStore('lastValues', { keyPath: 'asin' }).createIndex('scrapedAt', 'scrapedAt'); + if (!has('outbox')) db.createObjectStore('outbox', { keyPath: 'seq', autoIncrement: true }).createIndex('runId', 'runId'); + if (!has('spread')) db.createObjectStore('spread', { keyPath: ['runId', 'asin'] }).createIndex('runId', 'runId'); + if (!has('meta')) db.createObjectStore('meta', { keyPath: 'key' }); +} + +/** + * Opens the database. `indexedDB` is injectable for tests. + * @returns {Promise} the DB api + */ +function open({ indexedDB = globalThis.indexedDB, name = NAME } = {}) { + return new Promise((resolve, reject) => { + const req = indexedDB.open(name, VERSION); + req.onupgradeneeded = () => upgrade(req.result); + req.onerror = () => reject(storageError(req.error)); + req.onblocked = () => reject(storageError(new Error('database upgrade blocked'))); + req.onsuccess = () => { + const idb = req.result; + // Another context upgrading the schema should not hang on us. + idb.onversionchange = () => idb.close(); + resolve(api(idb)); + }; + }); +} + +function api(idb) { + /** Runs `fn(tx)` in one transaction; resolves with fn's result once it commits. */ + function run(stores, mode, fn) { + return new Promise((resolve, reject) => { + let tx; + try { + tx = idb.transaction(stores, mode); + } catch (err) { + reject(storageError(err)); + return; + } + let out; + tx.oncomplete = () => resolve(typeof out === 'function' ? out() : out); + tx.onabort = () => reject(storageError(tx.error)); + tx.onerror = (ev) => { if (ev && ev.preventDefault) ev.preventDefault(); }; + try { + out = fn(tx); + } catch (err) { + try { tx.abort(); } catch (e) { /* already done */ } + reject(storageError(err)); + } + }); + } + + /** A request's result, collected when the transaction commits. */ + function collect(request) { + let value; + request.onsuccess = () => { value = request.result; }; + return () => value; + } + + function get(store, key) { + return run([store], 'readonly', (tx) => collect(tx.objectStore(store).get(key))); + } + + /** All records of `store`, or those whose `index` equals `key`. */ + function getAll(store, index, key) { + return run([store], 'readonly', (tx) => { + const s = tx.objectStore(store); + return collect(index ? s.index(index).getAll(key) : s.getAll()); + }); + } + + function getMany(store, keys) { + return run([store], 'readonly', (tx) => { + const s = tx.objectStore(store); + const got = keys.map((k) => collect(s.get(k))); + return () => got.map((g) => g()); + }); + } + + function count(store) { + return run([store], 'readonly', (tx) => collect(tx.objectStore(store).count())); + } + + /** + * Applies `ops` in one transaction, all or nothing. Each op is + * {store, put}, {store, delete} or {store, deleteIndex: [index, key]}. + */ + function write(ops) { + const stores = [...new Set(ops.map((op) => op.store))]; + if (stores.length === 0) return Promise.resolve(); + return run(stores, 'readwrite', (tx) => { + for (const op of ops) { + const s = tx.objectStore(op.store); + if ('put' in op) s.put(op.put); + else if ('delete' in op) s.delete(op.delete); + else if (op.deleteIndex) { + s.index(op.deleteIndex[0]).openKeyCursor(op.deleteIndex[1]).onsuccess = (ev) => { + const c = ev.target.result; + if (!c) return; + s.delete(c.primaryKey); + c.continue(); + }; + } + } + }); + } + + async function getMeta(key) { + const rec = await get('meta', key); + return rec ? rec.value : undefined; + } + + /** Products of run `runId`, in the order they were found. */ + async function runProducts(runId) { + const rows = await getAll('products', 'runId', runId); + return rows.sort((a, b) => a.n - b.n); + } + + async function runPages(runId) { + const rows = await getAll('placements', 'runId', runId); + return rows.sort((a, b) => a.pageIndex - b.pageIndex); + } + + /** + * Bounds lastValues: drops snapshots last seen more than `maxAgeDays` + * ago, then keeps the `max` most recently seen. A snapshot with no date + * is never dropped for age but goes first when over the cap. + */ + function pruneLastValues({ max, maxAgeDays, now = Date.now() }) { + const cutoff = new Date(now - maxAgeDays * 86400000).toISOString(); + return run(['lastValues'], 'readwrite', (tx) => { + const s = tx.objectStore('lastValues'); + const byDate = s.index('scrapedAt'); + let removed = 0; + byDate.openKeyCursor(IDBKeyRange.upperBound(cutoff, true)).onsuccess = (ev) => { + const c = ev.target.result; + if (c) { s.delete(c.primaryKey); removed++; c.continue(); return; } + s.count().onsuccess = (e) => { + let over = e.target.result - max; + if (over <= 0) return; + // Undated first: they are not in the index. + s.openCursor().onsuccess = (e2) => { + const cur = e2.target.result; + if (cur && over > 0) { + const t = cur.value.scrapedAt; + if (typeof t !== 'string') { cur.delete(); over--; removed++; } + cur.continue(); + return; + } + if (over <= 0) return; + byDate.openKeyCursor().onsuccess = (e3) => { + const k = e3.target.result; + if (!k || over <= 0) return; + s.delete(k.primaryKey); over--; removed++; + k.continue(); + }; + }; + }; + }; + return () => removed; + }); + } + + return { + idb, run, get, getAll, getMany, count, write, getMeta, runProducts, runPages, pruneLastValues, + close: () => idb.close() + }; +} + +module.exports = { open, NAME, VERSION, storageError }; diff --git a/tests/unit/db.test.js b/tests/unit/db.test.js new file mode 100644 index 0000000..9983f1a --- /dev/null +++ b/tests/unit/db.test.js @@ -0,0 +1,82 @@ +/** + * @jest-environment node + */ +require('fake-indexeddb/auto'); +const { IDBFactory } = require('fake-indexeddb'); +const DB = require('../../scripts/background/db'); + +let db; +beforeEach(async () => { + db = await DB.open({ indexedDB: new IDBFactory() }); +}); +afterEach(() => db.close()); + +test('opens with every store', () => { + expect([...db.idb.objectStoreNames].sort()).toEqual( + ['lastValues', 'meta', 'outbox', 'placements', 'products', 'runs', 'spread']); +}); + +test('a write is all or nothing', async () => { + await db.write([{ store: 'runs', put: { runId: 'r1' } }]); + await expect(db.write([ + { store: 'runs', put: { runId: 'r2' } }, + { store: 'products', put: { runId: 'r2' } }, // no n: the key path fails + ])).rejects.toMatchObject({ name: 'StorageError' }); + expect((await db.getAll('runs')).map((r) => r.runId)).toEqual(['r1']); +}); + +test('products come back in the order they were found', async () => { + await db.write([3, 1, 2].map((n) => ({ store: 'products', put: { runId: 'r1', n, asin: `B0${n}` } }))); + await db.write([{ store: 'products', put: { runId: 'r0', n: 0, asin: 'B0OLD' } }]); + expect((await db.runProducts('r1')).map((p) => p.asin)).toEqual(['B01', 'B02', 'B03']); +}); + +test('deleteIndex removes one run and leaves the others', async () => { + await db.write([ + { store: 'outbox', put: { runId: 'a', asin: '1' } }, + { store: 'outbox', put: { runId: 'b', asin: '2' } }, + { store: 'outbox', put: { runId: 'a', asin: '3' } }, + ]); + await db.write([{ store: 'outbox', deleteIndex: ['runId', 'a'] }]); + expect((await db.getAll('outbox')).map((o) => o.asin)).toEqual(['2']); +}); + +test('meta values round trip', async () => { + await db.write([{ store: 'meta', put: { key: 'latestRunId', value: 'r9' } }]); + expect(await db.getMeta('latestRunId')).toBe('r9'); + expect(await db.getMeta('nothing')).toBeUndefined(); +}); + +describe('pruneLastValues', () => { + const now = Date.parse('2026-09-01T00:00:00Z'); + const day = 86400000; + const snap = (asin, daysAgo) => ({ + store: 'lastValues', + put: { asin, priceCents: 100, scrapedAt: daysAgo === null ? null : new Date(now - daysAgo * day).toISOString() }, + }); + + test('drops snapshots older than the age limit, keeps undated ones', async () => { + await db.write([snap('OLD', 400), snap('NEW', 3), snap('UNDATED', null)]); + expect(await db.pruneLastValues({ max: 10, maxAgeDays: 365, now })).toBe(1); + expect((await db.getAll('lastValues')).map((s) => s.asin).sort()).toEqual(['NEW', 'UNDATED']); + }); + + test('over the cap, drops undated then the least recently seen', async () => { + await db.write([snap('A', 1), snap('B', 5), snap('C', 2), snap('U', null), snap('D', 9)]); + await db.pruneLastValues({ max: 3, maxAgeDays: 365, now }); + expect((await db.getAll('lastValues')).map((s) => s.asin).sort()).toEqual(['A', 'B', 'C']); + }); + + test('under the cap nothing goes', async () => { + await db.write([snap('A', 1), snap('B', 2)]); + expect(await db.pruneLastValues({ max: 5, maxAgeDays: 365, now })).toBe(0); + expect(await db.count('lastValues')).toBe(2); + }); +}); + +test('a quota error is reported as storage_full', () => { + const quota = new Error('The quota has been exceeded.'); + quota.name = 'QuotaExceededError'; + expect(DB.storageError(quota).code).toBe('storage_full'); + expect(DB.storageError(new Error('other')).code).toBe('storage_error'); +}); From 90d503603c3aabad520a8eab23485c9df6052dc0 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:32:32 -0400 Subject: [PATCH 03/25] Give the run record states from idle to done and a reason for how it ended --- scripts/lib/run.js | 110 +++++++++++++++++++++++++++++++++++------ tests/unit/run.test.js | 60 ++++++++++++++++++++-- 2 files changed, 150 insertions(+), 20 deletions(-) diff --git a/scripts/lib/run.js b/scripts/lib/run.js index ac933ae..48b5a58 100644 --- a/scripts/lib/run.js +++ b/scripts/lib/run.js @@ -1,11 +1,14 @@ /** - * @fileoverview The scrape run record. + * @fileoverview The scrape run record and its state machine. * - * One run at a time, bound to the tab it started in. It is kept in - * chrome.storage.local under `run` so the popup, the service worker and the - * content script all see the same thing. The popup starts and stops it, the - * content script in that tab moves it page by page and ends it with a - * reason, and the service worker ends it if the tab goes away. + * One run at a time, bound to the tab it started in. The service worker is + * the only writer: it keeps the live record in chrome.storage.session under + * `run` and a durable copy in IndexedDB. The popup reads it to render, and + * content scripts never see it. + * + * States: idle (no run), starting, running, stopping, and the ends stopped, + * blocked, failed and done. `reason` says why a run ended; `END_STATE` maps + * each reason to its end state. * * Loaded as a plain global (`Run`) in the popup and content scripts, and as * a CommonJS module in Jest and the bundled service worker. @@ -29,9 +32,38 @@ const Run = (() => { selectors_broken: 'ProScan could not read this page. Amazon may have shown an error or changed its layout.', storage_full: 'Browser storage is full, so the run stopped. Download your results. Starting a new scan clears them.', interrupted: 'The run ended early because its tab was closed or left the search.', + storage_error: 'ProScan could not save a page, so the run stopped.', updated: 'ProScan was updated during the run, so it stopped. The products found before the update are kept.' }; + /** The state each end reason leaves a run in. */ + const END_STATE = { + complete: 'done', + stopped: 'stopped', + blocked: 'blocked', + selectors_broken: 'failed', + storage_full: 'failed', + storage_error: 'failed', + interrupted: 'failed', + updated: 'failed' + }; + + const STATES = ['idle', 'starting', 'running', 'stopping', 'stopped', 'blocked', 'failed', 'done']; + const ENDS = ['stopped', 'blocked', 'failed', 'done']; + const LIVE = ['starting', 'running', 'stopping']; + + /** Where each state may go. An ended run is never reopened; a new run is a new record. */ + const TRANSITIONS = { + idle: ['starting'], + starting: ['running', 'stopping', ...ENDS], + running: ['stopping', ...ENDS], + stopping: ENDS, + stopped: [], + blocked: [], + failed: [], + done: [] + }; + /** Page kinds a run can start on. */ const STARTABLE = ['results', 'last']; @@ -40,30 +72,55 @@ const Run = (() => { return Number.isInteger(v) && v > 0 ? Math.min(v, MAX_PAGES_LIMIT) : DEFAULT_MAX_PAGES; } - function create({ runId, tabId, maxPages, now = Date.now() }) { + function create({ runId, tabId, maxPages, source = null, now = Date.now() }) { return { runId, tabId, - status: 'running', + state: 'starting', + reason: null, + source, page: 0, + itemCount: 0, maxPages: clampMaxPages(maxPages), startedAt: now, heartbeat: now, finishedAt: null, lastUrl: null, - nextHref: null + nextHref: null, + navAt: null, + awaiting: false }; } + /** The state of `run`, or idle when there is none. */ + function stateOf(run) { + return run && STATES.includes(run.state) ? run.state : 'idle'; + } + + function canTransition(from, to) { + return (TRANSITIONS[from] || []).includes(to); + } + + /** `run` moved to state `to` with `patch` applied. Throws on a move the machine does not allow. */ + function transition(run, to, patch = {}) { + const from = stateOf(run); + if (!canTransition(from, to)) throw new Error(`run cannot go from ${from} to ${to}`); + return { ...run, ...patch, state: to }; + } + function isActive(run) { - return !!run && run.status === 'running'; + return LIVE.includes(stateOf(run)); + } + + function isEnded(run) { + return ENDS.includes(stateOf(run)); } function isStale(run, now = Date.now()) { return isActive(run) && now - (run.heartbeat || 0) > STALE_MS; } - /** True when `tabId` is the tab this running run belongs to. */ + /** True when `tabId` is the tab this live run belongs to. */ function owns(run, tabId) { return isActive(run) && typeof tabId === 'number' && run.tabId === tabId; } @@ -93,8 +150,10 @@ const Run = (() => { return isActive(run) && !!run.nextHref && samePage(run.nextHref, href); } - function finish(run, status, now = Date.now()) { - return { ...run, status, finishedAt: now, nextHref: null }; + /** `run` ended for `reason`. */ + function finish(run, reason, now = Date.now()) { + const to = END_STATE[reason] || 'failed'; + return transition(run, to, { reason, finishedAt: now, nextHref: null, navAt: null, awaiting: false }); } /** @@ -129,6 +188,20 @@ const Run = (() => { return null; } + /** What a search URL scans: a storefront (me=) or a keyword (k=). */ + function sourceOf(url, now = Date.now()) { + let type = 'keyword'; + let sellerId = null; + let keyword = null; + try { + const u = new URL(url); + const me = u.searchParams.get('me'); + const k = u.searchParams.get('k'); + if (me) { type = 'storefront'; sellerId = me; } else if (k) keyword = k; + } catch (e) { /* not a URL */ } + return { type, sellerId, keyword, url: url || null, startedAt: new Date(now).toISOString() }; + } + /** Wait before the next page: 2 to 4 seconds. */ function pageDelay(rand = Math.random()) { return 2000 + Math.floor(rand * 2000); @@ -155,8 +228,8 @@ const Run = (() => { function describe(run, count) { if (!run) return null; if (isActive(run)) return { text: 'Scraping in progress... ' + count + ' items', type: 'info' }; - const text = (MESSAGES[run.status] || 'Scraping ended.') + ' ' + count + ' products found.'; - return { text, type: run.status === 'complete' ? 'success' : 'warning' }; + const text = (MESSAGES[run.reason] || 'Scraping ended.') + ' ' + count + ' products found.'; + return { text, type: run.reason === 'complete' ? 'success' : 'warning' }; } return { @@ -164,10 +237,16 @@ const Run = (() => { DEFAULT_MAX_PAGES, STALE_MS, MESSAGES, + END_STATE, + STATES, STARTABLE, clampMaxPages, create, + stateOf, + canTransition, + transition, isActive, + isEnded, isStale, owns, samePage, @@ -175,6 +254,7 @@ const Run = (() => { finish, outcome, pageDelay, + sourceOf, refusal, describe }; diff --git a/tests/unit/run.test.js b/tests/unit/run.test.js index df3da93..de9b53a 100644 --- a/tests/unit/run.test.js +++ b/tests/unit/run.test.js @@ -8,10 +8,12 @@ const page = (kind, extra = {}) => ({ ...extra, }); +const running = (opts) => Run.transition(Run.create(opts), 'running'); + describe('Run.create', () => { - test('starts running, bound to its tab', () => { + test('starts in starting, bound to its tab', () => { const r = Run.create({ runId: 'r1', tabId: 7, maxPages: 5, now: 1000 }); - expect(r).toMatchObject({ runId: 'r1', tabId: 7, status: 'running', page: 0, maxPages: 5, heartbeat: 1000 }); + expect(r).toMatchObject({ runId: 'r1', tabId: 7, state: 'starting', reason: null, page: 0, maxPages: 5, heartbeat: 1000 }); }); test.each([[undefined], [0], [-3], ['x'], [2.5]])('maxPages %p falls back to the default', (n) => { @@ -23,8 +25,47 @@ describe('Run.create', () => { }); }); +describe('the state machine', () => { + test('no run is idle', () => { + expect(Run.stateOf(null)).toBe('idle'); + expect(Run.stateOf({ status: 'running' })).toBe('idle'); + }); + + test('a run goes starting, running, stopping, stopped', () => { + let r = Run.create({ runId: 'r', tabId: 1 }); + r = Run.transition(r, 'running'); + r = Run.transition(r, 'stopping'); + r = Run.finish(r, 'stopped', 9); + expect(r).toMatchObject({ state: 'stopped', reason: 'stopped', finishedAt: 9 }); + }); + + test.each(Object.entries(Run.END_STATE))('ending for %s leaves the run %s', (reason, state) => { + expect(Run.finish(running({ runId: 'r', tabId: 1 }), reason)).toMatchObject({ state, reason }); + expect(Run.isEnded(Run.finish(running({ runId: 'r', tabId: 1 }), reason))).toBe(true); + }); + + test('an ended run cannot be reopened or ended twice', () => { + const ended = Run.finish(running({ runId: 'r', tabId: 1 }), 'complete'); + expect(() => Run.transition(ended, 'running')).toThrow(/cannot go from done to running/); + expect(() => Run.finish(ended, 'stopped')).toThrow(); + }); + + test('stopping cannot go back to running', () => { + const r = Run.transition(running({ runId: 'r', tabId: 1 }), 'stopping'); + expect(Run.canTransition('stopping', 'running')).toBe(false); + expect(() => Run.transition(r, 'running')).toThrow(); + }); + + test('every state is known and only live states are active', () => { + expect(Run.STATES).toEqual(['idle', 'starting', 'running', 'stopping', 'stopped', 'blocked', 'failed', 'done']); + for (const s of Run.STATES) { + expect(Run.isActive({ state: s })).toBe(['starting', 'running', 'stopping'].includes(s)); + } + }); +}); + describe('Run ownership and staleness', () => { - const run = Run.create({ runId: 'r1', tabId: 7, now: 1000 }); + const run = running({ runId: 'r1', tabId: 7, now: 1000 }); test('only the run tab owns a running run', () => { expect(Run.owns(run, 7)).toBe(true); @@ -41,7 +82,7 @@ describe('Run ownership and staleness', () => { }); test('finish records the reason and the time', () => { - expect(Run.finish(run, 'blocked', 5000)).toMatchObject({ status: 'blocked', finishedAt: 5000, nextHref: null }); + expect(Run.finish(run, 'blocked', 5000)).toMatchObject({ state: 'blocked', reason: 'blocked', finishedAt: 5000, nextHref: null }); }); }); @@ -92,7 +133,7 @@ describe('Run text', () => { }); test('only complete reads as success', () => { - const run = Run.create({ runId: 'r', tabId: 1 }); + const run = running({ runId: 'r', tabId: 1 }); expect(Run.describe(Run.finish(run, 'complete'), 3)).toEqual({ text: 'Scraping complete! 3 products found.', type: 'success' }); for (const reason of ['stopped', 'blocked', 'selectors_broken', 'storage_full', 'interrupted', 'updated']) { expect(Run.describe(Run.finish(run, reason), 3).type).toBe('warning'); @@ -128,3 +169,12 @@ describe('Run.expects', () => { expect(Run.expects(Run.finish(run, 'complete'), run.nextHref)).toBe(false); }); }); + +describe('Run.sourceOf', () => { + const now = Date.parse('2026-09-01T00:00:00Z'); + test('a storefront, a keyword search and junk', () => { + expect(Run.sourceOf('https://www.amazon.com/s?me=A1B2&k=x', now)).toMatchObject({ type: 'storefront', sellerId: 'A1B2', keyword: null }); + expect(Run.sourceOf('https://www.amazon.com/s?k=yoga+mat', now)).toMatchObject({ type: 'keyword', sellerId: null, keyword: 'yoga mat', startedAt: '2026-09-01T00:00:00.000Z' }); + expect(Run.sourceOf('not a url', now)).toMatchObject({ type: 'keyword', sellerId: null, keyword: null }); + }); +}); From c6b38ab84aaf07178191a80e777fbadb1d3de2b0 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:32:33 -0400 Subject: [PATCH 04/25] Run scrapes from a service worker engine that owns the run and its data --- scripts/background/engine.js | 453 ++++++++++++++++++++++++++++++++ tests/setup/chrome-mock.js | 36 ++- tests/setup/dom-helpers.js | 6 +- tests/setup/engine-rig.js | 206 +++++++++++++++ tests/unit/engine.test.js | 489 +++++++++++++++++++++++++++++++++++ 5 files changed, 1184 insertions(+), 6 deletions(-) create mode 100644 scripts/background/engine.js create mode 100644 tests/setup/engine-rig.js create mode 100644 tests/unit/engine.test.js diff --git a/scripts/background/engine.js b/scripts/background/engine.js new file mode 100644 index 0000000..46e5444 --- /dev/null +++ b/scripts/background/engine.js @@ -0,0 +1,453 @@ +/** + * @fileoverview The run engine. The service worker is the only writer. + * + * The live run record is in chrome.storage.session, so it survives the + * worker being stopped between pages but not a browser restart. What the + * run collects goes to IndexedDB (db.js) in one transaction per page. + * + * A run moves page by page: + * 1. START_RUN from the popup: ping the tab, make the run, ask the tab to + * parse (PARSE_PAGE). + * 2. The content script parses and answers with PAGE_RESULT. The engine + * folds the page into the run and saves it. + * 3. After a 2 to 4 second delay the engine opens the page's Next link with + * tabs.update. The new page says PAGE_READY and is asked to parse. + * + * The delay is a timer in the worker, which dies with it. The content + * script's HEARTBEAT every few seconds wakes a stopped worker, and every + * wake runs tick(), which opens the next page once it is due. All changes + * go through one queue, so two messages never interleave their writes. + * + * @module Engine + */ + +const Run = require('../lib/run.js'); +const Msg = require('../lib/messages.js'); +const Flags = require('../lib/flags.js'); +const Delta = require('../modules/delta.js'); + +/** A page asked for but not reported this long is asked for again. */ +const PAGE_TIMEOUT_MS = 20000; + +/** + * Folds one page's products into a run. Placements get the page number and + * a run-wide organic rank. An ASIN the run already has only adds its + * placements to that record. Pure: `runProducts` is not changed. + * + * @returns {{fresh: Object[], changed: Object[]}} new records, and copies + * of existing records that changed + */ +function foldPage(runProducts, pageProducts, pageIndex) { + const organicBefore = runProducts.reduce( + (n, r) => n + (r.placements || []).filter(pl => !pl.sponsored).length, 0); + const known = new Map(runProducts.map(r => [r.asin, r])); + const changed = new Map(); + const fresh = []; + + for (const product of pageProducts) { + const placements = (product.placements || []).map(pl => ({ + page: pageIndex, + ...pl, + rank: pl.rank === null ? null : pl.rank + organicBefore + })); + const firstRank = (placements.find(pl => pl.rank !== null) || {}).rank; + const organicRank = firstRank === undefined ? null : firstRank; + + const seen = changed.get(product.asin) || known.get(product.asin); + if (!seen) { + fresh.push({ ...product, placements, organicRank }); + continue; + } + changed.set(product.asin, { + ...seen, + placements: [...(seen.placements || []), ...placements], + sponsored: !!seen.sponsored || !!product.sponsored, + organicRank: seen.organicRank == null ? organicRank : seen.organicRank + }); + } + return { fresh, changed: [...changed.values()] }; +} + +/** The run as kept in IndexedDB: without the fields that only matter live. */ +function durable(run) { + const { heartbeat, navAt, awaiting, expectUrl, navigatedAt, ...rest } = run; + return rest; +} + +function runSuffix(random) { + return Math.floor(random() * 0x100000000).toString(36).padStart(7, '0').slice(0, 8); +} + +/** + * @param {Object} deps + * @param {Object} deps.chrome - chrome.* (storage.session, storage.local, tabs) + * @param {function(): Promise} deps.openDb - resolves to the db.js api + */ +function createEngine({ + chrome, + openDb, + now = Date.now, + random = Math.random, + setTimer = setTimeout, + clearTimer = clearTimeout, + flags = Flags, + log = console +}) { + let dbPromise = null; + let timer = null; + let chain = Promise.resolve(); + + const db = () => { + if (!dbPromise) dbPromise = openDb().catch((err) => { dbPromise = null; throw err; }); + return dbPromise; + }; + + /** Runs `fn` after every change queued before it. */ + function serial(fn) { + const p = chain.then(fn, fn); + chain = p.catch(() => {}); + return p; + } + + async function getRun() { + const data = await chrome.storage.session.get(Run.KEY); + return (data && data[Run.KEY]) || null; + } + + const saveRun = (run) => chrome.storage.session.set({ [Run.KEY]: run }); + + function sendToTab(tabId, message) { + return new Promise((resolve) => { + try { + chrome.tabs.sendMessage(tabId, message, { frameId: 0 }, (response) => { + if (chrome.runtime.lastError) return resolve(null); + resolve(response || null); + }); + } catch (e) { + resolve(null); + } + }); + } + + function schedule(ms) { + if (timer) clearTimer(timer); + timer = setTimer(() => { timer = null; tick(); }, Math.max(0, ms)); + } + + function cancelTimer() { + if (timer) clearTimer(timer); + timer = null; + } + + /** Ends `run` for `reason` and tells its tab. Returns the ended run. */ + async function end(run, reason) { + const ended = Run.finish(run, reason, now()); + cancelTimer(); + await saveRun(ended); + try { + await (await db()).write([{ store: 'runs', put: durable(ended) }]); + } catch (err) { + log.warn('[ProScan] Could not record the end of the run:', err.message); + } + log.log(`[ProScan] Run ended: ${reason}. ${ended.itemCount} items`); + sendToTab(ended.tabId, { type: Msg.T.RUN_ENDED, runId: ended.runId, reason }); + return ended; + } + + /** The live run, after ending it if its tab stopped checking in. */ + async function liveRun() { + const run = await getRun(); + if (Run.isStale(run, now())) return end(run, 'interrupted'); + return run; + } + + function start({ tabId }) { + return serial(async () => { + if (typeof tabId !== 'number') return { error: 'refused', message: Run.refusal('unknown') }; + const current = await liveRun(); + if (Run.isActive(current)) { + return { error: 'busy', message: 'A run is already going. Stop it before starting another.' }; + } + const pong = await sendToTab(tabId, { type: Msg.T.PING }); + if (!pong || !pong.ok) return { error: 'no_receiver' }; + if (!Run.STARTABLE.includes(pong.kind)) { + return { error: 'refused', kind: pong.kind, message: Run.refusal(pong.kind) }; + } + + const local = await chrome.storage.local.get('settings'); + const settings = (local && local.settings) || {}; + const t = now(); + const runId = `${t}-${runSuffix(random)}`; + let run = Run.create({ runId, tabId, maxPages: settings.maxPages, source: Run.sourceOf(pong.url, t), now: t }); + await saveRun(run); + try { + await (await db()).write([ + { store: 'runs', put: durable(run) }, + { store: 'meta', put: { key: 'latestRunId', value: runId } } + ]); + } catch (err) { + await end(run, err.code === 'storage_full' ? 'storage_full' : 'storage_error'); + return { error: err.code || 'storage_error', message: Run.MESSAGES[err.code] || Run.MESSAGES.storage_error }; + } + + run = Run.transition(run, 'running', { awaiting: true, expectUrl: pong.url, navigatedAt: t }); + await saveRun(run); + const ack = await sendToTab(tabId, { type: Msg.T.PARSE_PAGE, runId, page: 1 }); + if (!ack || !ack.ok) { + await end(run, 'interrupted'); + return { error: 'no_receiver' }; + } + return { ok: true, runId }; + }); + } + + function stop() { + return serial(async () => { + const run = await getRun(); + if (!Run.isActive(run)) return { ok: true, stopped: false }; + const stopping = Run.transition(run, 'stopping'); + await saveRun(stopping); + await end(stopping, 'stopped'); + return { ok: true, stopped: true }; + }); + } + + /** A page in some tab finished loading. Only the run's tab is asked to parse. */ + function pageReady({ url }, sender) { + return serial(async () => { + const run = await liveRun(); + const tabId = sender && sender.tab ? sender.tab.id : null; + if (!Run.owns(run, tabId) || run.state !== 'running') return { idle: true }; + if (run.awaiting) return { parse: true, runId: run.runId, page: run.page + 1 }; + if (url === run.lastUrl) return { heartbeat: true, runId: run.runId }; + // The tab went somewhere else while the next page was pending. + await end(run, 'interrupted'); + return { idle: true }; + }); + } + + function pageResult({ runId, page, url, result }, sender) { + return serial(async () => { + const run = await liveRun(); + const tabId = sender && sender.tab ? sender.tab.id : null; + if (!Run.owns(run, tabId) || run.runId !== runId || run.state !== 'running' || + !run.awaiting || page !== run.page + 1 || !result || !Array.isArray(result.products)) { + return { ok: false, ignored: true }; + } + const ending = Run.outcome(result, page, run.maxPages); + if (result.products.length === 0 || ending === 'selectors_broken') { + const ended = await end(run, ending || 'complete'); + return { ok: true, next: 'end', reason: ended.reason }; + } + + const store = await db(); + const t = now(); + let ops; + let next; + try { + const existing = await store.runProducts(runId); + const { fresh, changed } = foldPage(existing, result.products, page); + const prevs = await store.getMany('lastValues', fresh.map(p => p.asin)); + ops = []; + fresh.forEach((p, i) => { + p.runId = runId; + p.pageIndex = page; + p.n = existing.length + i; + const prev = prevs[i] || null; + // Only an ASIN's first sighting in the run gets a delta. + p.delta = Delta.computeDeltas(p, prev); + ops.push({ store: 'lastValues', put: { asin: p.asin, ...Delta.snapshot(p, prev) } }); + ops.push({ store: 'products', put: p }); + }); + changed.forEach(p => ops.push({ store: 'products', put: p })); + ops.push({ + store: 'placements', + put: { + runId, pageIndex: page, count: fresh.length, placements: result.placements || 0, + kind: result.kind, fill: result.fill || null, scrapedAt: new Date(t).toISOString(), url + } + }); + if (flags.CLOUD_SYNC) ops.push({ store: 'outbox', put: { runId, pageIndex: page, queuedAt: t } }); + + next = { + ...run, page, itemCount: run.itemCount + fresh.length, heartbeat: t, + lastUrl: url, nextHref: result.nextHref || null, awaiting: false + }; + if (ending) next = Run.finish(next, ending, t); + else next.navAt = t + Run.pageDelay(random()); + ops.push({ store: 'runs', put: durable(next) }); + await store.write(ops); + } catch (err) { + log.warn('[ProScan] Could not save the page:', err.message); + const ended = await end(run, err.code === 'storage_full' ? 'storage_full' : 'storage_error'); + return { ok: false, next: 'end', reason: ended.reason }; + } + + await saveRun(next); + store.pruneLastValues({ max: Delta.MAX_ENTRIES, maxAgeDays: Delta.MAX_AGE_DAYS, now: t }) + .catch(err => log.warn('[ProScan] Could not prune lastValues:', err.message)); + if (ending) { + log.log(`[ProScan] Run ended: ${ending}. ${next.itemCount} items`); + sendToTab(next.tabId, { type: Msg.T.RUN_ENDED, runId, reason: ending }); + return { ok: true, next: 'end', reason: ending }; + } + schedule(next.navAt - t); + return { ok: true, next: 'wait' }; + }); + } + + /** Opens the next page once it is due, and re-asks a page that never reported. */ + function tick() { + return serial(async () => { + let run = await liveRun(); + if (!run || run.state !== 'running') return; + const t = now(); + if (!run.awaiting && run.nextHref && run.navAt != null) { + if (run.navAt > t) { + if (!timer) schedule(run.navAt - t); + return; + } + run = { ...run, awaiting: true, expectUrl: run.nextHref, navigatedAt: t, navAt: null }; + await saveRun(run); + log.log(`[ProScan] Navigating to next page: ${run.expectUrl}`); + try { + await chrome.tabs.update(run.tabId, { url: run.expectUrl }); + } catch (err) { + await end(run, 'interrupted'); + } + return; + } + if (run.awaiting && t - (run.navigatedAt || 0) > PAGE_TIMEOUT_MS) { + await saveRun({ ...run, navigatedAt: t }); + sendToTab(run.tabId, { type: Msg.T.PARSE_PAGE, runId: run.runId, page: run.page + 1 }); + } + }); + } + + function heartbeat({ runId }, sender) { + return serial(async () => { + const run = await liveRun(); + const tabId = sender && sender.tab ? sender.tab.id : null; + if (!Run.owns(run, tabId) || run.runId !== runId) return { active: false }; + await saveRun({ ...run, heartbeat: now() }); + return { active: true }; + }).then((out) => { tick(); return out; }); + } + + function tabRemoved(tabId) { + return serial(async () => { + const run = await getRun(); + if (Run.isActive(run) && run.tabId === tabId) await end(run, 'interrupted'); + }); + } + + /** + * The run's tab started loading a page the engine did not open: a reload + * is fine, anything else means the user took the tab elsewhere. Without + * the tabs permission a non-Amazon URL is hidden, which also counts. + */ + function tabUpdated(tabId, info, tab) { + if (!info || info.status !== 'loading') return Promise.resolve(); + return serial(async () => { + const run = await getRun(); + if (!Run.owns(run, tabId) || run.state !== 'running' || run.awaiting) return; + const url = info.url || (tab && tab.url); + if (url && url === run.lastUrl) return; + await end(run, 'interrupted'); + }); + } + + /** After a browser restart the session is empty; a run IndexedDB still has as live was cut off. */ + function recover() { + return serial(async () => { + if (await getRun()) return; + const store = await db(); + const latest = await store.getMeta('latestRunId'); + const rec = latest ? await store.get('runs', latest) : null; + if (Run.isActive(rec)) { + await store.write([{ store: 'runs', put: durable(Run.finish(rec, 'interrupted', now())) }]); + } + }); + } + + /** The latest run and what it found, for the popup and the chat. */ + async function latest() { + const live = await liveRun(); + const store = await db(); + const runId = (live && live.runId) || await store.getMeta('latestRunId'); + if (!runId) return { run: null, results: [], pages: [], spread: {} }; + const [rec, results, pages, spreadRows] = await Promise.all([ + store.get('runs', runId), + store.runProducts(runId), + store.runPages(runId), + store.getAll('spread', 'runId', runId) + ]); + const spread = {}; + spreadRows.forEach(r => { spread[r.asin] = r.data; }); + const run = live && live.runId === runId ? live : (rec || null); + return { run, results, pages, spread }; + } + + function getState() { + return serial(latest); + } + + async function getResults() { + const { run, results } = await getState(); + return { runId: run ? run.runId : null, results }; + } + + function spreadResult({ asin, data }) { + return serial(async () => { + const store = await db(); + const runId = await store.getMeta('latestRunId'); + if (!runId || typeof asin !== 'string') return { ok: false }; + await store.write([{ store: 'spread', put: { runId, asin, data: data || null } }]); + return { ok: true }; + }); + } + + /** What chat.js reads, in the shape it was written for. */ + async function chatData(keys) { + const local = await chrome.storage.local.get(keys.filter(k => k === 'geminiApiKey')); + const { run, results, pages } = await getState(); + return { + ...local, + results, + scrapeRunId: run ? run.runId : null, + scrapeRunMeta: run ? run.source : null, + scrapeRunPages: pages + }; + } + + /** + * The sync bundle for every run in the outbox, built from the products + * as they are now, and the outbox keys to delete once it is written. + */ + async function outboxBundle() { + const store = await db(); + const entries = await store.getAll('outbox'); + const runIds = [...new Set(entries.map(e => e.runId))]; + const syncQueue = []; + const scrapeRunPages = []; + let scrapeRunMeta = null; + for (const runId of runIds) { + syncQueue.push(...await store.runProducts(runId)); + scrapeRunPages.push(...await store.runPages(runId)); + const rec = await store.get('runs', runId); + if (rec) scrapeRunMeta = rec.source; + } + return { bundle: { syncQueue, scrapeRunMeta, scrapeRunPages }, seqs: entries.map(e => e.seq) }; + } + + async function clearOutbox(seqs) { + await (await db()).write(seqs.map(seq => ({ store: 'outbox', delete: seq }))); + } + + return { + start, stop, pageReady, pageResult, heartbeat, tick, tabRemoved, tabUpdated, recover, + getState, getResults, spreadResult, chatData, outboxBundle, clearOutbox, db + }; +} + +module.exports = { createEngine, foldPage, durable, PAGE_TIMEOUT_MS }; diff --git a/tests/setup/chrome-mock.js b/tests/setup/chrome-mock.js index dadfca0..4a9df2e 100644 --- a/tests/setup/chrome-mock.js +++ b/tests/setup/chrome-mock.js @@ -9,8 +9,31 @@ const { TextEncoder, TextDecoder } = require('util'); if (!global.TextEncoder) global.TextEncoder = TextEncoder; if (!global.TextDecoder) global.TextDecoder = TextDecoder; +// fake-indexeddb needs structuredClone, which the jsdom environment leaves out. +if (!global.structuredClone) { + const v8 = require('v8'); + global.structuredClone = (value) => v8.deserialize(v8.serialize(value)); +} let store = {}; + +/** chrome.storage.session: promise-only, cleared by _reset(). */ +function sessionArea() { + let data = {}; + const copy = (v) => JSON.parse(JSON.stringify(v)); + return { + async get(keys) { + const list = keys == null ? Object.keys(data) : [].concat(keys); + const out = {}; + list.forEach(k => { if (data[k] !== undefined) out[k] = copy(data[k]); }); + return out; + }, + async set(items) { Object.assign(data, copy(items)); }, + async remove(keys) { [].concat(keys).forEach(k => delete data[k]); }, + _getStore() { return copy(data); }, + _reset() { data = {}; } + }; +} // Set by _failNext: the next set() fails with this runtime.lastError message. let failNext = null; @@ -65,7 +88,8 @@ global.chrome = { }, _getStore() { return { ...store }; }, _reset() { store = {}; failNext = null; } - } + }, + session: sessionArea() }, runtime: { sendMessage: function(msg, callback) { @@ -83,9 +107,13 @@ global.chrome = { query: function(opts, callback) { if (callback) callback([{ id: 1, url: 'https://www.amazon.com/s?k=test' }]); }, - sendMessage: function(tabId, msg, callback) { - if (callback) callback({}); - } + sendMessage: function(tabId, msg, opts, callback) { + const cb = typeof opts === 'function' ? opts : callback; + if (cb) cb({}); + }, + update: async function(tabId, props) { return { id: tabId, ...props }; }, + onRemoved: { addListener() {} }, + onUpdated: { addListener() {} } }, downloads: { download: function(opts, callback) { diff --git a/tests/setup/dom-helpers.js b/tests/setup/dom-helpers.js index eef1e27..6fe649f 100644 --- a/tests/setup/dom-helpers.js +++ b/tests/setup/dom-helpers.js @@ -27,6 +27,7 @@ function quietConsole() { * @param {string} [url='https://www.amazon.com/s?k=test&page=1'] - Page URL * @param {Object} [options] * @param {Object} [options.flags] - Overrides for scripts/lib/flags.js, e.g. { CLOUD_SYNC: true } + * @param {Object} [options.globals] - Replaces context globals, e.g. a per-tab chrome and timers * @returns {Object} VM context with all script functions accessible */ function loadContentScript(scriptPath, html, url = 'https://www.amazon.com/s?k=test&page=1', options = {}) { @@ -75,7 +76,8 @@ function loadContentScript(scriptPath, html, url = 'https://www.amazon.com/s?k=t RegExp: global.RegExp, String: global.String, Number: global.Number, - JSON: global.JSON + JSON: global.JSON, + ...(options.globals || {}) }); const absolutePath = path.resolve(__dirname, '../../', scriptPath); @@ -86,7 +88,7 @@ function loadContentScript(scriptPath, html, url = 'https://www.amazon.com/s?k=t // declarations share one lexical environment only within a single // runInContext call, so they are concatenated ahead of the target script. let preamble = ''; - for (const rel of ['scripts/modules/price.js', 'scripts/lib/parsers.js', 'scripts/lib/run.js', 'scripts/lib/flags.js', 'scripts/modules/delta.js']) { + for (const rel of ['scripts/modules/price.js', 'scripts/lib/parsers.js', 'scripts/lib/messages.js', 'scripts/lib/run.js', 'scripts/lib/flags.js', 'scripts/modules/delta.js']) { const dep = path.resolve(__dirname, '../../', rel); if (fs.existsSync(dep)) preamble += fs.readFileSync(dep, 'utf8') + '\n'; } diff --git a/tests/setup/engine-rig.js b/tests/setup/engine-rig.js new file mode 100644 index 0000000..821591f --- /dev/null +++ b/tests/setup/engine-rig.js @@ -0,0 +1,206 @@ +/** + * An in-process extension for the run engine tests: the real router and + * engine over fake-indexeddb, and the real scraper.js loaded into a JSDOM + * page per tab. Messages go between them the way Chrome routes them. + * tabs.update loads the next page from `site(url)`. + * + * Use jest fake timers with setImmediate left real (fake-indexeddb runs on + * it), then drive time with rig.advance(ms). + */ +require('fake-indexeddb/auto'); +const { IDBFactory } = require('fake-indexeddb'); +const { loadContentScript } = require('./dom-helpers'); +const DB = require('../../scripts/background/db'); +const { createEngine } = require('../../scripts/background/engine'); +const { createRouter } = require('../../scripts/background/router'); + +const EXT_ID = 'test-extension'; + +function memoryArea(initial = {}) { + let store = JSON.parse(JSON.stringify(initial)); + let failNext = null; + return { + async get(keys) { + const list = keys == null ? Object.keys(store) : [].concat(keys); + const out = {}; + list.forEach((k) => { if (store[k] !== undefined) out[k] = JSON.parse(JSON.stringify(store[k])); }); + return out; + }, + async set(items) { + if (failNext) { const m = failNext; failNext = null; throw new Error(m); } + Object.assign(store, JSON.parse(JSON.stringify(items))); + }, + async remove(keys) { [].concat(keys).forEach((k) => delete store[k]); }, + dump: () => JSON.parse(JSON.stringify(store)), + _failNext(m = 'QUOTA_BYTES quota exceeded') { failNext = m; }, + }; +} + +/** Timers that can all be cleared at once, as when a page or the worker goes away. */ +function timerSet() { + const live = new Set(); + return { + setTimeout(fn, ms) { const id = setTimeout(() => { live.delete(id); fn(); }, ms); live.add(id); return id; }, + clearTimeout(id) { live.delete(id); clearTimeout(id); }, + setInterval(fn, ms) { const id = setInterval(fn, ms); live.add(id); return id; }, + clearInterval(id) { live.delete(id); clearInterval(id); }, + clearAll() { live.forEach((id) => { clearTimeout(id); clearInterval(id); }); live.clear(); }, + }; +} + +function createRig({ site, flags, local = {}, now = () => Date.now(), random = () => 0, wrapDb = (db) => db } = {}) { + const factory = new IDBFactory(); + const session = memoryArea(); + const localArea = memoryArea(local); + const tabs = new Map(); + const served = []; + const sent = []; + const log = { log() {}, warn() {}, error() {} }; + let nextTabId = 1; + let worker = null; + let lastError = null; + + const withError = (message, fn) => { + lastError = { message }; + try { fn(); } finally { lastError = null; } + }; + + // The worker's view of chrome. + const workerChrome = { + storage: { session, local: localArea }, + runtime: { get lastError() { return lastError; }, id: EXT_ID }, + tabs: { + sendMessage(tabId, message, opts, cb) { + const tab = tabs.get(tabId); + setImmediate(() => { + if (!tab || !tab.listener) return withError('Could not establish connection. Receiving end does not exist.', () => cb()); + let answered = false; + const respond = (r) => { if (!answered) { answered = true; cb(r); } }; + const held = tab.listener(message, { id: EXT_ID }, respond); + if (!held && !answered) withError('The message port closed before a response was received.', () => cb()); + }); + }, + async update(tabId, { url }) { + if (!tabs.has(tabId)) throw new Error(`No tab with id: ${tabId}.`); + setImmediate(() => navigate(tabId, url)); + return { id: tabId }; + }, + }, + }; + + function startWorker() { + const timers = timerSet(); + const engine = createEngine({ + chrome: workerChrome, + openDb: () => DB.open({ indexedDB: factory }).then(wrapDb), + now, random, log, flags, + setTimer: timers.setTimeout, + clearTimer: timers.clearTimeout, + }); + const router = createRouter({ extensionId: EXT_ID, log }); + router.on('START_RUN', (m) => engine.start(m)); + router.on('STOP_RUN', () => engine.stop()); + router.on('GET_STATE', () => engine.getState()); + router.on('PAGE_READY', (m, s) => engine.pageReady(m, s)); + router.on('PAGE_RESULT', (m, s) => engine.pageResult(m, s)); + router.on('HEARTBEAT', (m, s) => engine.heartbeat(m, s)); + router.on('GET_RESULTS', () => engine.getResults()); + router.on('SPREAD_RESULT', (m) => engine.spreadResult(m)); + worker = { engine, router, timers }; + return worker; + } + + /** A content script message to the worker, from tab `tabId`. Starts a stopped worker. */ + function fromTab(tabId, message, cb) { + sent.push({ tabId, type: message && message.type }); + setImmediate(() => { + const w = worker || startWorker(); + let answered = false; + const respond = (r) => { if (!answered) { answered = true; if (cb) cb(r); } }; + const held = w.router.listener(message, { id: EXT_ID, tab: { id: tabId } }, respond); + if (!held && !answered && cb) withError('The message port closed before a response was received.', () => cb()); + }); + } + + function tabChrome(tabId) { + return { + runtime: { + id: EXT_ID, + get lastError() { return lastError; }, + sendMessage(message, cb) { fromTab(tabId, message, cb); }, + onMessage: { addListener(fn) { tabs.get(tabId).listener = fn; } }, + getURL: (p) => `chrome-extension://${EXT_ID}/${p}`, + }, + }; + } + + function navigate(tabId, url) { + const tab = tabs.get(tabId); + if (!tab) return; + if (tab.timers) tab.timers.clearAll(); + tab.listener = null; + tab.url = url; + (worker || startWorker()).engine.tabUpdated(tabId, { status: 'loading', url }, { id: tabId, url }); + const html = site(url); + served.push({ tabId, url }); + if (html == null) return; + tab.timers = timerSet(); + tab.ctx = loadContentScript('scripts/content/scraper.js', html, url, { + flags, + globals: { chrome: tabChrome(tabId), ...tab.timers }, + }); + } + + return { + served, + sent, + session, + local: localArea, + get worker() { return worker || startWorker(); }, + engine: () => (worker || startWorker()).engine, + openTab(url) { + const id = nextTabId++; + tabs.set(id, { id }); + navigate(id, url); + return id; + }, + closeTab(id) { + const tab = tabs.get(id); + if (tab && tab.timers) tab.timers.clearAll(); + tabs.delete(id); + return (worker || startWorker()).engine.tabRemoved(id); + }, + goTo(id, url) { navigate(id, url); }, + tabCtx: (id) => tabs.get(id).ctx, + /** A popup message to the worker. */ + popup(message) { + return new Promise((resolve) => { + const w = worker || startWorker(); + const held = w.router.listener(message, { id: EXT_ID, url: `chrome-extension://${EXT_ID}/popup/popup.html` }, resolve); + if (!held) setImmediate(() => resolve(undefined)); + }); + }, + /** Stops the worker the way Chrome does: its timers and memory go, storage stays. */ + killWorker() { + if (worker) worker.timers.clearAll(); + worker = null; + }, + run: async () => (await session.get('run')).run || null, + db: () => DB.open({ indexedDB: factory }), + }; +} + +/** Lets messages, IndexedDB and timers settle; with fake timers, moves time on by `ms`. */ +async function settle(ms = 0, step = 250) { + const spin = async () => { for (let i = 0; i < 40; i++) await new Promise((r) => setImmediate(r)); }; + for (let i = 0; i < 3; i++) { + jest.advanceTimersByTime(0); + await spin(); + } + for (let t = 0; t < ms; t += step) { + jest.advanceTimersByTime(Math.min(step, ms - t)); + await spin(); + } +} + +module.exports = { createRig, settle, memoryArea, EXT_ID }; diff --git a/tests/unit/engine.test.js b/tests/unit/engine.test.js new file mode 100644 index 0000000..4bc0cae --- /dev/null +++ b/tests/unit/engine.test.js @@ -0,0 +1,489 @@ +/** + * @jest-environment node + * + * The service worker run engine with the real scraper.js in each tab + * (tests/setup/engine-rig.js). The Chromium harness in tests/e2e runs the + * same scenarios in a real browser. + */ +const fs = require('fs'); +const path = require('path'); +const { createRig, settle } = require('../setup/engine-rig'); +const { foldPage } = require('../../scripts/background/engine'); +const Run = require('../../scripts/lib/run'); + +const CAPTCHA = fs.readFileSync(path.join(__dirname, '../pages/2026-09/captcha-synthetic.html'), 'utf8'); +const PRODUCT = fs.readFileSync(path.join(__dirname, '../pages/2026-09/product-dp.html'), 'utf8'); + +const asinFor = (k, page, i) => `B0${k.slice(0, 3).toUpperCase().padEnd(3, 'X')}${String(page).padStart(2, '0')}${String(i).padStart(3, '0')}`; +const card = (asin, price, { ad = false, rating = true } = {}) => ` +
+

Item ${asin}

+
${price}
+ ${rating ? '
4.5 out of 5 stars
(1K)' : ''} +
`; +const page = (cards, next) => `
${cards}
${ + next ? `Next` : 'Next' +}`; +const searchUrl = (k, p = 1) => `https://www.amazon.com/s?k=${k}${p > 1 ? `&page=${p}` : ''}`; +const pageNo = (url) => Number(new URL(url).searchParams.get('page') || 1); + +/** A site of `pages` generated search pages per keyword. */ +function simpleSite(pages, { captchaAt = null } = {}) { + return (url) => { + const u = new URL(url); + if (u.pathname.startsWith('/dp/')) return PRODUCT; + const k = u.searchParams.get('k'); + const p = pageNo(url); + if (p === captchaAt) return CAPTCHA; + if (p > pages) return null; + const cards = [0, 1, 2, 3].map((i) => card(asinFor(k, p, i), `$${10 + i}.0${p}`)).join(''); + return page(cards, p < pages ? `/s?k=${k}&page=${p + 1}` : null); + }; +} + +const served = (rig, k) => rig.served.filter((s) => s.url.includes(`k=${k}`)).map((s) => pageNo(s.url)); + +async function results(rig) { + return (await rig.popup({ type: 'GET_STATE' })).results; +} + +/** Opens a search tab and starts a run in it. */ +async function startIn(rig, k) { + const tab = rig.openTab(searchUrl(k)); + await settle(); + const resp = await rig.popup({ type: 'START_RUN', tabId: tab }); + return { tab, resp }; +} + +/** Moves time on until the run ends or `ms` passes. */ +async function runUntilEnd(rig, ms = 30000) { + for (let t = 0; t < ms; t += 500) { + const run = await rig.run(); + if (run && !Run.isActive(run)) return run; + await settle(500); + } + return rig.run(); +} + +beforeEach(() => jest.useFakeTimers({ doNotFake: ['nextTick', 'setImmediate', 'queueMicrotask'] })); +afterEach(() => { + jest.clearAllTimers(); + jest.useRealTimers(); +}); + +describe('a run', () => { + test('a clean 3 page run scrapes every page once and completes', async () => { + const rig = createRig({ site: simpleSite(3) }); + const { resp } = await startIn(rig, 'hose'); + expect(resp).toMatchObject({ ok: true }); + const run = await runUntilEnd(rig); + expect(served(rig, 'hose')).toEqual([1, 2, 3]); + expect(run).toMatchObject({ state: 'done', reason: 'complete', page: 3, itemCount: 12 }); + const st = await rig.popup({ type: 'GET_STATE' }); + expect(st.results.map((r) => r.asin)).toEqual([1, 2, 3].flatMap((p) => [0, 1, 2, 3].map((i) => asinFor('hose', p, i)))); + expect(st.results[0]).toMatchObject({ priceCents: 1001, rating: 4.5, reviewCount: 1000, runId: run.runId, pageIndex: 1 }); + expect(st.pages.map((p) => [p.pageIndex, p.count, p.kind])).toEqual([[1, 4, 'results'], [2, 4, 'results'], [3, 4, 'last']]); + expect(st.run.source).toMatchObject({ type: 'keyword', keyword: 'hose' }); + }); + + test('waits 2 to 4 seconds between pages', async () => { + const rig = createRig({ site: simpleSite(3), random: () => 0.5 }); + await startIn(rig, 'wait'); + await settle(2500); + expect(served(rig, 'wait')).toEqual([1]); + await settle(1000); + expect(served(rig, 'wait')).toEqual([1, 2]); + }); + + test('a captcha at page 2 ends the run as blocked', async () => { + const rig = createRig({ site: simpleSite(4, { captchaAt: 2 }) }); + await startIn(rig, 'usb'); + const run = await runUntilEnd(rig); + await settle(5000); + + expect(served(rig, 'usb')).toEqual([1, 2]); + expect(run).toMatchObject({ state: 'blocked', reason: 'blocked' }); + expect((await results(rig)).map((r) => r.asin)).toEqual([0, 1, 2, 3].map((i) => asinFor('usb', 1, i))); + }); + + test('the page cap in settings ends the run on that page', async () => { + const rig = createRig({ site: simpleSite(5), local: { settings: { maxPages: 2 } } }); + await startIn(rig, 'cap'); + const run = await runUntilEnd(rig); + await settle(5000); + expect(served(rig, 'cap')).toEqual([1, 2]); + expect(run).toMatchObject({ state: 'done', reason: 'complete', maxPages: 2 }); + }); + + test('a page whose cards all fail to parse ends the run as selectors_broken', async () => { + const broken = '
'; + const rig = createRig({ site: (url) => (pageNo(url) === 1 ? simpleSite(3)(url) : page(broken, null)) }); + await startIn(rig, 'broke'); + const run = await runUntilEnd(rig); + expect(run).toMatchObject({ state: 'failed', reason: 'selectors_broken', page: 1 }); + }); +}); + +describe('Stop', () => { + test('Stop halts the run before the next page loads', async () => { + const rig = createRig({ site: simpleSite(4) }); + await startIn(rig, 'stop'); + await settle(); + expect((await rig.run()).page).toBe(1); + expect(await rig.popup({ type: 'STOP_RUN' })).toEqual({ ok: true, stopped: true }); + await settle(8000); + + expect(served(rig, 'stop')).toEqual([1]); + expect(await rig.run()).toMatchObject({ state: 'stopped', reason: 'stopped' }); + expect((await results(rig))).toHaveLength(4); + }); + + test('Stop that lands while a page is being saved still ends the run as stopped', async () => { + const rig = createRig({ site: simpleSite(4) }); + const tab = rig.openTab(searchUrl('race')); + await settle(); + await rig.popup({ type: 'START_RUN', tabId: tab }); + // The page result is on its way; Stop queues behind its save. + const stopped = rig.popup({ type: 'STOP_RUN' }); + await settle(); + await stopped; + await settle(8000); + const run = await rig.run(); + expect(run).toMatchObject({ state: 'stopped' }); + expect(served(rig, 'race')).toEqual([1]); + }); + + test('Stop with no run changes nothing', async () => { + const rig = createRig({ site: simpleSite(1) }); + expect(await rig.popup({ type: 'STOP_RUN' })).toEqual({ ok: true, stopped: false }); + expect(await rig.run()).toBeNull(); + }); +}); + +describe('the run belongs to its tab', () => { + test('a second search tab opened mid-run does not join the run', async () => { + const rig = createRig({ site: simpleSite(3) }); + await startIn(rig, 'alpha'); + await settle(); + const other = rig.openTab(searchUrl('beta')); + const run = await runUntilEnd(rig); + + expect(served(rig, 'beta')).toEqual([1]); + expect(run).toMatchObject({ reason: 'complete' }); + const got = await results(rig); + expect(got.filter((r) => r.asin.startsWith('B0BET'))).toEqual([]); + expect(got).toHaveLength(12); + expect(other).not.toBe(run.tabId); + }); + + test('a product page opened in another tab does not end the run', async () => { + const rig = createRig({ site: simpleSite(3) }); + await startIn(rig, 'gamma'); + await settle(); + rig.openTab('https://www.amazon.com/dp/B09B8V1LZ3'); + const run = await runUntilEnd(rig); + expect(run).toMatchObject({ reason: 'complete', page: 3 }); + }); + + test('closing the run tab ends the run as interrupted', async () => { + const rig = createRig({ site: simpleSite(4) }); + const { tab } = await startIn(rig, 'close'); + await settle(); + await rig.closeTab(tab); + await settle(8000); + expect(await rig.run()).toMatchObject({ state: 'failed', reason: 'interrupted' }); + expect(served(rig, 'close')).toEqual([1]); + }); + + test('taking the run tab elsewhere ends the run, and the tab is not pulled back', async () => { + const rig = createRig({ site: simpleSite(4) }); + const { tab } = await startIn(rig, 'left'); + await settle(); + rig.goTo(tab, 'https://www.amazon.com/dp/B09B8V1LZ3'); + await settle(8000); + expect(await rig.run()).toMatchObject({ reason: 'interrupted' }); + expect(served(rig, 'left')).toEqual([1]); + }); + + test('a reload of the run tab between pages does not end the run', async () => { + const rig = createRig({ site: simpleSite(3) }); + const { tab } = await startIn(rig, 'reload'); + await settle(); + rig.goTo(tab, searchUrl('reload')); + const run = await runUntilEnd(rig); + expect(run).toMatchObject({ reason: 'complete', page: 3 }); + expect(served(rig, 'reload')).toEqual([1, 1, 2, 3]); + }); +}); + +describe('starting', () => { + test('Start on a captcha page changes nothing and says why', async () => { + const rig = createRig({ site: () => CAPTCHA }); + const tab = rig.openTab(searchUrl('robot')); + await settle(); + const resp = await rig.popup({ type: 'START_RUN', tabId: tab }); + expect(resp).toMatchObject({ error: 'refused', kind: 'captcha' }); + expect(resp.message).toMatch(/captcha/); + expect(await rig.run()).toBeNull(); + }); + + test('Start on a tab with no content script asks for a reload', async () => { + const rig = createRig({ site: simpleSite(1) }); + expect(await rig.popup({ type: 'START_RUN', tabId: 99 })).toEqual({ error: 'no_receiver' }); + expect(await rig.run()).toBeNull(); + }); + + test('a second Start while a run is going is refused', async () => { + const rig = createRig({ site: simpleSite(3) }); + await startIn(rig, 'one'); + const tab2 = rig.openTab(searchUrl('two')); + await settle(); + expect(await rig.popup({ type: 'START_RUN', tabId: tab2 })).toMatchObject({ error: 'busy' }); + }); + + test('START_RUN from a content script is refused', async () => { + const rig = createRig({ site: simpleSite(1) }); + const respond = jest.fn(); + expect(rig.worker.router.listener({ type: 'START_RUN', tabId: 1 }, { id: 'test-extension', tab: { id: 1 } }, respond)).toBe(false); + expect(respond).not.toHaveBeenCalled(); + }); +}); + +describe('the worker stopped between pages', () => { + test('the heartbeat wakes it and the run resumes from session storage', async () => { + const rig = createRig({ site: simpleSite(3) }); + await startIn(rig, 'sleepy'); + await settle(); + expect((await rig.run()).page).toBe(1); + rig.killWorker(); + const run = await runUntilEnd(rig); + expect(served(rig, 'sleepy')).toEqual([1, 2, 3]); + expect(run).toMatchObject({ reason: 'complete', itemCount: 12 }); + }); + + test('killed on every page, the run still finishes once each', async () => { + const rig = createRig({ site: simpleSite(4) }); + await startIn(rig, 'again'); + for (let i = 0; i < 12; i++) { + await settle(500); + rig.killWorker(); + } + const run = await runUntilEnd(rig); + expect(served(rig, 'again')).toEqual([1, 2, 3, 4]); + expect(run).toMatchObject({ reason: 'complete', page: 4 }); + expect(await results(rig)).toHaveLength(16); + }); + + test('a run whose tab stopped checking in ends as interrupted on the next wake', async () => { + let clock = Date.parse('2026-09-24T10:00:00Z'); + const rig = createRig({ site: simpleSite(3), now: () => clock }); + await startIn(rig, 'stale'); + await settle(); + expect(await rig.run()).toMatchObject({ state: 'running', page: 1 }); + // No heartbeat for longer than STALE_MS: time moves, the page's timers do not. + clock += Run.STALE_MS + 1; + rig.killWorker(); + const st = await rig.popup({ type: 'GET_STATE' }); + expect(st.run).toMatchObject({ state: 'failed', reason: 'interrupted' }); + }); + + test('after a browser restart a run IndexedDB still has as live is ended', async () => { + const rig = createRig({ site: simpleSite(3) }); + await startIn(rig, 'restart'); + await settle(); + await rig.session.remove('run'); + rig.killWorker(); + await rig.engine().recover(); + const st = await rig.popup({ type: 'GET_STATE' }); + expect(st.run).toMatchObject({ state: 'failed', reason: 'interrupted' }); + expect(st.results).toHaveLength(4); + }); +}); + +describe('page results', () => { + test('a repeated or out of order PAGE_RESULT is ignored', async () => { + const rig = createRig({ site: simpleSite(3) }); + const { tab } = await startIn(rig, 'latch'); + await settle(); + const run = await rig.run(); + const send = (msg) => new Promise((r) => rig.worker.router.listener(msg, { id: 'test-extension', tab: { id: tab } }, r)); + const result = { kind: 'results', products: [{ asin: 'B0XXXXXXX1', placements: [] }], placements: 1, nextHref: 'x', fill: { title: 1, price: 1 } }; + expect(await send({ type: 'PAGE_RESULT', runId: run.runId, page: 1, url: 'u', result })).toMatchObject({ ignored: true }); + expect(await send({ type: 'PAGE_RESULT', runId: run.runId, page: 5, url: 'u', result })).toMatchObject({ ignored: true }); + expect(await send({ type: 'PAGE_RESULT', runId: 'other', page: 2, url: 'u', result })).toMatchObject({ ignored: true }); + expect(await results(rig)).toHaveLength(4); + }); + + test('PAGE_RESULT from another tab is ignored', async () => { + const rig = createRig({ site: simpleSite(3) }); + await startIn(rig, 'owner'); + await settle(); + const run = await rig.run(); + const resp = await new Promise((r) => rig.worker.router.listener( + { type: 'PAGE_RESULT', runId: run.runId, page: 2, url: 'u', result: { kind: 'results', products: [] } }, + { id: 'test-extension', tab: { id: run.tabId + 1 } }, r)); + expect(resp).toMatchObject({ ignored: true }); + }); + + test('a page that cannot be saved ends the run loudly as storage_full', async () => { + let writes = 0; + const wrapDb = (db) => ({ + ...db, + write: (ops) => { + if (ops.some((o) => o.store === 'placements') && ++writes === 2) { + return Promise.reject(Object.assign(new Error('QuotaExceededError'), { name: 'StorageError', code: 'storage_full' })); + } + return db.write(ops); + }, + }); + const rig = createRig({ site: simpleSite(4), wrapDb }); + await startIn(rig, 'full'); + const run = await runUntilEnd(rig); + await settle(5000); + expect(run).toMatchObject({ state: 'failed', reason: 'storage_full', page: 1 }); + expect(served(rig, 'full')).toEqual([1, 2]); + expect(Run.describe(run, 4).text).toMatch(/storage is full/); + }); +}); + +describe('products across pages (F-27, F-28)', () => { + const PAGE1 = page(card('B0A', '$5.00', { ad: true }) + card('B0B', '$2.00') + card('B0A', '$5.00'), '/s?k=w&page=2'); + const PAGE2 = page(card('B0C', '$3.00') + card('B0A', '$5.00') + card('B0B', '$2.00', { rating: false }), null); + const site = (url) => (pageNo(url) === 1 ? PAGE1 : PAGE2); + + async function twoPages({ lastValues = [], flags } = {}) { + const rig = createRig({ site, flags }); + const db = await rig.db(); + await db.write(lastValues.map((v) => ({ store: 'lastValues', put: v }))); + db.close(); + await startIn(rig, 'w'); + await runUntilEnd(rig); + return rig; + } + const prevRun = (asin, priceCents) => ({ asin, priceCents, rating: 4.5, reviewCount: 900, runId: 'r0', scrapedAt: null }); + + test('each ASIN is stored once per run, with every placement', async () => { + const rig = await twoPages(); + const st = await rig.popup({ type: 'GET_STATE' }); + expect(st.results.map((r) => r.asin)).toEqual(['B0A', 'B0B', 'B0C']); + expect(st.run.itemCount).toBe(3); + const a = st.results[0]; + expect(a.url).toBe('https://www.amazon.com/dp/B0A'); + expect(a.sponsored).toBe(true); + expect(a.organicRank).toBe(2); + expect(a.placements).toEqual([ + { page: 1, position: 1, sponsored: true, rank: null }, + { page: 1, position: 3, sponsored: false, rank: 2 }, + { page: 2, position: 2, sponsored: false, rank: 4 }, + ]); + expect(st.results[2]).toMatchObject({ asin: 'B0C', organicRank: 3, sponsored: false }); + expect(st.pages.map((p) => [p.count, p.placements])).toEqual([[2, 3], [1, 3]]); + }); + + test('with cloud sync off nothing goes into the outbox (F-26)', async () => { + const rig = await twoPages({ flags: { CLOUD_SYNC: false } }); + const db = await rig.db(); + expect(await db.count('outbox')).toBe(0); + db.close(); + }); + + test('an outbox left by an older build is not grown while sync is off', async () => { + const rig = createRig({ site, flags: { CLOUD_SYNC: false } }); + const db = await rig.db(); + await db.write([{ store: 'outbox', put: { seq: 1, runId: 'r0', pageIndex: null } }]); + db.close(); + await startIn(rig, 'w'); + await runUntilEnd(rig); + const after = await rig.db(); + expect(await after.getAll('outbox')).toEqual([{ seq: 1, runId: 'r0', pageIndex: null }]); + after.close(); + }); + + test('with cloud sync on each page is queued once, and the bundle has the products as they are now', async () => { + const rig = await twoPages({ flags: { CLOUD_SYNC: true } }); + const { bundle, seqs } = await rig.engine().outboxBundle(); + expect(seqs).toHaveLength(2); + expect(bundle.syncQueue.map((p) => p.asin)).toEqual(['B0A', 'B0B', 'B0C']); + expect(bundle.syncQueue[0].placements).toHaveLength(3); + expect(bundle.scrapeRunMeta).toMatchObject({ keyword: 'w' }); + await rig.engine().clearOutbox(seqs); + expect((await rig.engine().outboxBundle()).seqs).toEqual([]); + }); + + test('each page drops lastValues older than the age limit (F-26)', async () => { + const stale = { asin: 'B0GONE', priceCents: 1, runId: 'r0', scrapedAt: new Date(Date.now() - 400 * 86400000).toISOString() }; + const rig = await twoPages({ lastValues: [stale] }); + const db = await rig.db(); + expect((await db.getAll('lastValues')).map((s) => s.asin).sort()).toEqual(['B0A', 'B0B', 'B0C']); + db.close(); + }); + + test('a repeat in the same run keeps the delta against the last run', async () => { + const rig = await twoPages({ lastValues: [prevRun('B0A', 600), prevRun('B0B', 250)] }); + const [a, b] = await results(rig); + expect(a.delta).toEqual({ isNew: false, dPriceCents: -100, dRating: 0, dReviews: 100 }); + // B0B lost its rating markup on page 2; the page 1 delta stays + expect(b.delta).toEqual({ isNew: false, dPriceCents: -50, dRating: 0, dReviews: 100 }); + const db = await rig.db(); + expect(await db.get('lastValues', 'B0B')).toMatchObject({ priceCents: 200, rating: 4.5, reviewCount: 1000 }); + db.close(); + }); + + test('a card with no rating gives a null delta, not a fake drop (F-28)', async () => { + const rig = createRig({ site: () => page(card('B0B', '$2.00', { rating: false }), null) }); + const db = await rig.db(); + await db.write([{ store: 'lastValues', put: prevRun('B0B', 200) }]); + db.close(); + await startIn(rig, 'w'); + await runUntilEnd(rig); + const [b] = await results(rig); + expect(b).toMatchObject({ rating: null, reviewCount: null }); + expect(b.delta).toEqual({ isNew: false, dPriceCents: 0, dRating: null, dReviews: null }); + }); + + test('two runs keep their own products, and organic ranks start at 1 in each', async () => { + const rig = createRig({ site: simpleSite(1) }); + const first = await startIn(rig, 'aaa'); + await runUntilEnd(rig); + const firstRun = await rig.run(); + await startIn(rig, 'bbb'); + await runUntilEnd(rig); + const st = await rig.popup({ type: 'GET_STATE' }); + expect(st.results.map((r) => r.asin)).toEqual([0, 1, 2, 3].map((i) => asinFor('bbb', 1, i))); + expect(st.results.map((r) => r.organicRank)).toEqual([1, 2, 3, 4]); + const db = await rig.db(); + expect(await db.runProducts(firstRun.runId)).toHaveLength(4); + expect(await db.count('lastValues')).toBe(8); + db.close(); + expect(first.resp.ok).toBe(true); + }); +}); + +describe('spread results', () => { + test('are kept per run and come back with the state', async () => { + const rig = createRig({ site: simpleSite(1) }); + const { tab } = await startIn(rig, 'spread'); + await runUntilEnd(rig); + const send = (msg) => new Promise((r) => rig.worker.router.listener(msg, { id: 'test-extension', tab: { id: tab } }, r)); + const got = await send({ type: 'GET_RESULTS' }); + expect(got.results).toHaveLength(4); + await send({ type: 'SPREAD_RESULT', asin: got.results[0].asin, data: { sellerPrices: [1, 2] } }); + const st = await rig.popup({ type: 'GET_STATE' }); + expect(st.spread).toEqual({ [got.results[0].asin]: { sellerPrices: [1, 2] } }); + }); +}); + +describe('foldPage', () => { + test('does not change the records it is given', () => { + const before = [{ asin: 'B0A', organicRank: 1, placements: [{ page: 1, position: 1, sponsored: false, rank: 1 }] }]; + const frozen = JSON.stringify(before); + const { fresh, changed } = foldPage(before, [ + { asin: 'B0A', sponsored: true, placements: [{ position: 1, sponsored: true, rank: null }] }, + { asin: 'B0B', sponsored: false, placements: [{ position: 2, sponsored: false, rank: 1 }] }, + ], 2); + expect(JSON.stringify(before)).toBe(frozen); + expect(fresh.map((p) => [p.asin, p.organicRank])).toEqual([['B0B', 2]]); + expect(changed[0]).toMatchObject({ asin: 'B0A', sponsored: true, organicRank: 1 }); + expect(changed[0].placements).toHaveLength(2); + }); +}); From 077e61cfa35bb58f0e6e13df0bab6d0624c61d23 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:32:33 -0400 Subject: [PATCH 05/25] Move run data out of storage.local into IndexedDB as schema 4 --- scripts/lib/migrate.js | 121 +++++++++++++++++++++++++++---- tests/unit/migrate.test.js | 145 ++++++++++++++++++++++++++++++++----- 2 files changed, 235 insertions(+), 31 deletions(-) diff --git a/scripts/lib/migrate.js b/scripts/lib/migrate.js index a08d74c..5e8ef36 100644 --- a/scripts/lib/migrate.js +++ b/scripts/lib/migrate.js @@ -14,6 +14,16 @@ * - a 2.0 run cut off by the update is recorded as ended, not resumed * - the untouched 2.0 default settings are dropped, so 2.1 defaults apply * + * 3 to 4 (2.2, the service worker run engine): + * - everything but settings moves out of chrome.storage.local into + * IndexedDB (scripts/background/db.js): result rows become the products + * of their run, scrapeRunPages its pages, lastValues and spread data + * their own stores + * - a 2.1 run still marked running was cut off by the update and ends as + * updated + * - the IndexedDB writes commit before anything is removed, so a failure + * leaves version 3 in place to be retried + * * Loaded as a CommonJS module in Jest and the bundled service worker. * * @module Migrate @@ -24,7 +34,14 @@ const Migrate = (() => { const RunLib = typeof Run !== 'undefined' ? Run : require('./run.js'); const DeltaLib = typeof Delta !== 'undefined' ? Delta : require('../modules/delta.js'); - const CURRENT = 3; + const CURRENT = 4; + + /** Keys 3 to 4 moves out of chrome.storage.local. */ + const MOVED = [ + 'results', 'currentItemCount', 'isScrapingActive', 'scrapeRunId', 'scrapeRunPageIndex', + 'scrapeRunMeta', 'scrapeRunPages', 'syncQueue', 'lastValues', 'spreadResults', + 'isSpreadAnalyzing', 'run' + ]; const LEGACY_RUN = 'legacy-2.0'; const ASIN = /^[A-Z0-9]{10}$/; @@ -89,52 +106,128 @@ const Migrate = (() => { remove.push('settings'); } return { set, remove }; + }, + + /** 3 -> 4. Returns {set, remove, idb}, idb being db.js write ops. */ + 4(store, now) { + const remove = MOVED.filter(k => k in store); + return { set: {}, remove, idb: toIdb(store, now) }; } }; + /** A run record 2.1 kept in chrome.storage.local, in the 2.2 shape. */ + function upgradeRun(old, now) { + if (!old || typeof old.runId !== 'string') return null; + if (RunLib.STATES.includes(old.state)) return old; + const reason = old.status === 'running' ? 'updated' : (RunLib.END_STATE[old.status] ? old.status : 'complete'); + const { status, heartbeat, ...rest } = old; + return { ...rest, state: RunLib.END_STATE[reason], reason, finishedAt: old.finishedAt || now }; + } + + /** The IndexedDB writes that hold what `store` kept in chrome.storage.local. */ + function toIdb(store, now) { + const ops = []; + const results = Array.isArray(store.results) ? store.results.filter(r => r && typeof r === 'object') : []; + const oldRun = upgradeRun(store[RunLib.KEY], now); + const runOf = (row) => row.runId || store.scrapeRunId || LEGACY_RUN; + + const byRun = new Map(); + const add = (row) => { + const runId = runOf(row); + if (!byRun.has(runId)) byRun.set(runId, []); + byRun.get(runId).push({ ...row, runId }); + }; + results.forEach(add); + // Queued products of runs whose results are gone. Only unreleased builds queued any. + const queued = Array.isArray(store.syncQueue) ? store.syncQueue.filter(r => r && ASIN.test(r.asin || '')) : []; + const inResults = new Set(results.map(r => runOf(r) + ':' + r.asin)); + queued.filter(r => !inResults.has(runOf(r) + ':' + r.asin)).forEach(add); + + const latestRunId = results.length ? runOf(results[0]) : (oldRun ? oldRun.runId : null); + const runIds = new Set([...byRun.keys(), ...(oldRun ? [oldRun.runId] : [])]); + for (const runId of runIds) { + const rows = byRun.get(runId) || []; + rows.forEach((row, n) => ops.push({ store: 'products', put: { ...row, n } })); + const base = oldRun && oldRun.runId === runId + ? oldRun + : { runId, state: 'done', reason: 'complete', startedAt: null, finishedAt: null }; + const source = runId === store.scrapeRunId && store.scrapeRunMeta ? store.scrapeRunMeta : (base.source || null); + ops.push({ store: 'runs', put: { ...base, source, itemCount: rows.length } }); + } + + (Array.isArray(store.scrapeRunPages) ? store.scrapeRunPages : []).forEach(pg => { + if (!pg || !Number.isInteger(pg.pageIndex)) return; + ops.push({ store: 'placements', put: { ...pg, runId: pg.runId || store.scrapeRunId || LEGACY_RUN } }); + }); + + Object.entries(store.lastValues || {}).forEach(([asin, snap]) => { + if (snap && typeof snap === 'object') ops.push({ store: 'lastValues', put: { ...snap, asin } }); + }); + + [...new Set(queued.map(runOf))].forEach((runId, i) => { + ops.push({ store: 'outbox', put: { seq: i + 1, runId, pageIndex: null, queuedAt: now } }); + }); + + if (latestRunId) { + Object.entries(store.spreadResults || {}).forEach(([asin, data]) => { + ops.push({ store: 'spread', put: { runId: latestRunId, asin, data: data || null } }); + }); + ops.push({ store: 'meta', put: { key: 'latestRunId', value: latestRunId } }); + } + return ops; + } + /** * What it takes to bring `store` to the current version, or null when it * is already there. Pure: reads `store`, returns the writes. * * @param {Object} store - everything in chrome.storage.local - * @param {{now?: number}} [opts] - * @returns {{from: number, to: number, set: Object, remove: string[]}|null} + * @param {{now?: number, to?: number}} [opts] - `to` stops at an earlier version + * @returns {{from: number, to: number, set: Object, remove: string[], idb: Object[]}|null} */ - function plan(store, { now = Date.now() } = {}) { + function plan(store, { now = Date.now(), to = CURRENT } = {}) { const from = versionOf(store); - if (from >= CURRENT) return null; + if (from >= to) return null; let state = { ...store }; const remove = new Set(); - for (let v = from + 1; v <= CURRENT; v++) { + const idb = []; + for (let v = from + 1; v <= to; v++) { const step = STEPS[v](state, now); state = { ...state, ...step.set }; step.remove.forEach(k => { remove.add(k); delete state[k]; }); + if (step.idb) idb.push(...step.idb); } const set = {}; Object.keys(state).forEach(k => { if (state[k] !== store[k]) set[k] = state[k]; }); - set.schemaVersion = CURRENT; - return { from, to: CURRENT, set, remove: [...remove] }; + set.schemaVersion = to; + return { from, to, set, remove: [...remove].filter(k => k in store), idb }; } /** - * Runs the migration against a chrome.storage area. Removals go first and - * schemaVersion is written last with the rest, so a failed write leaves - * the old version in place to be retried. + * Runs the migration against a chrome.storage area. The IndexedDB writes + * commit first, in one transaction, then the removals, and schemaVersion + * is written last with the rest. A failure at any point leaves the old + * version in place to be retried, and a retry rewrites the same keyed + * records. * * @param {chrome.storage.StorageArea} area - * @param {{now?: number}} [opts] + * @param {{now?: number, to?: number, openDb?: function(): Promise}} [opts] * @returns {Promise} the plan that was applied, or null */ - async function run(area, opts) { + async function run(area, opts = {}) { const store = await area.get(null); const p = plan(store || {}, opts); if (!p) return null; + if (p.idb.length) { + if (!opts.openDb) throw new Error('migration needs IndexedDB'); + await (await opts.openDb()).write(p.idb); + } if (p.remove.length) await area.remove(p.remove); await area.set(p.set); return p; } - return { CURRENT, LEGACY_RUN, versionOf, upgradeRow, plan, run }; + return { CURRENT, LEGACY_RUN, MOVED, versionOf, upgradeRow, upgradeRun, toIdb, plan, run }; })(); if (typeof module !== 'undefined' && module.exports) { diff --git a/tests/unit/migrate.test.js b/tests/unit/migrate.test.js index e62ea66..aa60463 100644 --- a/tests/unit/migrate.test.js +++ b/tests/unit/migrate.test.js @@ -1,5 +1,5 @@ /** - * schemaVersion 2 to 3 (F-100, F-101), over storage the live 2.0 build + * schemaVersion 2 to 3 (F-100, F-101) and 3 to 4, over storage the live 2.0 build * (commit 7c2ba1c) wrote in Chromium: tests/fixtures/v2.0-storage.json, * captured by tests/e2e/capture-v20-storage.mjs from the saved yoga mat * page plus one generated page. @@ -26,7 +26,7 @@ describe('the captured v2.0 snapshot', () => { }); describe('plan 2 -> 3 on the v2.0 snapshot', () => { - const p = Migrate.plan(clone(V20), { now: NOW }); + const p = Migrate.plan(clone(V20), { now: NOW, to: 3 }); const asins = [...new Set(V20.results.map((r) => r.asin))]; test('moves to version 3', () => { @@ -80,7 +80,7 @@ describe('plan 2 -> 3 on the v2.0 snapshot', () => { describe('the first 2.1 scrape after the update', () => { test('diffs against the 2.0 values instead of calling every product new', () => { - const p = Migrate.plan(clone(V20), { now: NOW }); + const p = Migrate.plan(clone(V20), { now: NOW, to: 3 }); const seeded = p.set.lastValues; const asin = p.set.results.find((r) => r.priceCents !== null).asin; const old = seeded[asin]; @@ -96,9 +96,9 @@ describe('the first 2.1 scrape after the update', () => { describe('edge cases', () => { test('a run cut off by the update ends as updated and is not resumed (F-101)', () => { const store = { ...clone(V20), isScrapingActive: true }; - const p = Migrate.plan(store, { now: NOW }); + const p = Migrate.plan(store, { now: NOW, to: 3 }); expect(p.set.isScrapingActive).toBe(false); - expect(p.set[Run.KEY]).toMatchObject({ runId: 'legacy-2.0', status: 'updated', finishedAt: NOW }); + expect(p.set[Run.KEY]).toMatchObject({ runId: 'legacy-2.0', state: 'failed', reason: 'updated', finishedAt: NOW }); expect(Run.isActive(p.set[Run.KEY])).toBe(false); expect(Run.describe(p.set[Run.KEY], 74).text).toMatch(/updated during the run/); }); @@ -109,7 +109,7 @@ describe('edge cases', () => { { name: 'Zero', asin: 'B0V1000002', price: '$5.00', rating: 0, reviewCount: 0, url: 'x', scrapedAt: '2026-02-20T15:00:00.000Z' }, { name: 'Euro', asin: 'B0V1000003', price: '€19,99', rating: 4, reviewCount: 3, url: 'x' }, { name: 'Number', asin: 'B0V1000004', price: 12.5, rating: 4, reviewCount: 3, url: 'x' }, - ] }, { now: NOW }); + ] }, { now: NOW, to: 3 }); expect(p.set.results.map((r) => [r.name, r.price, r.priceCents, r.rating, r.reviewCount])).toEqual([ [null, null, null, null, null], ['Zero', '$5.00', 500, null, null], @@ -125,12 +125,12 @@ describe('edge cases', () => { test('a February 2.0 scrape survives the lastValues age limit', () => { const p = Migrate.plan({ results: [ { name: 'Old', asin: 'B0FEB00001', price: '$9.99', rating: 4, reviewCount: 3, url: 'x', scrapedAt: '2026-02-20T15:00:00.000Z' }, - ] }, { now: NOW }); + ] }, { now: NOW, to: 3 }); expect(Object.keys(p.set.lastValues)).toEqual(['B0FEB00001']); }); test('rows with a bad ASIN are kept but not seeded', () => { - const p = Migrate.plan({ results: [{ name: 'x', asin: 'bad', price: '$1.00', rating: 4, reviewCount: 1, url: 'u' }] }, { now: NOW }); + const p = Migrate.plan({ results: [{ name: 'x', asin: 'bad', price: '$1.00', rating: 4, reviewCount: 1, url: 'u' }] }, { now: NOW, to: 3 }); expect(p.set.results).toHaveLength(1); expect(p.set.results[0].url).toBe('u'); expect(p.set.lastValues).toEqual({}); @@ -138,28 +138,29 @@ describe('edge cases', () => { test('an existing lastValues entry is never replaced', () => { const mine = { priceCents: 100, rating: 4, reviewCount: 1, runId: 'r9', scrapedAt: '2026-09-01T00:00:00.000Z' }; - const p = Migrate.plan({ ...clone(V20), lastValues: { [V20.results[0].asin]: mine } }, { now: NOW }); + const p = Migrate.plan({ ...clone(V20), lastValues: { [V20.results[0].asin]: mine } }, { now: NOW, to: 3 }); expect(p.set.lastValues[V20.results[0].asin]).toEqual(mine); }); test('rows written by a newer build pass through untouched', () => { const row = { asin: 'B0NEW00001', runId: 'r1', price: null, priceCents: null, url: 'https://www.amazon.com/dp/B0NEW00001', rating: 0 }; - const p = Migrate.plan({ results: [row] }, { now: NOW }); + const p = Migrate.plan({ results: [row] }, { now: NOW, to: 3 }); expect(p.set.results[0]).toEqual(row); expect(p.set.lastValues).toEqual({}); }); test('changed settings are kept', () => { - const p = Migrate.plan({ settings: { pageDelay: 2000, maxPages: 5 } }, { now: NOW }); + const p = Migrate.plan({ settings: { pageDelay: 2000, maxPages: 5 } }, { now: NOW, to: 3 }); expect(p.remove).toEqual([]); }); test('empty storage just gets the version', () => { - expect(Migrate.plan({}, { now: NOW })).toEqual({ from: 2, to: 3, set: { lastValues: {}, schemaVersion: 3 }, remove: [] }); + expect(Migrate.plan({}, { now: NOW, to: 3 })).toEqual({ from: 2, to: 3, set: { lastValues: {}, schemaVersion: 3 }, remove: [], idb: [] }); }); test('storage already at version 3 is left alone', () => { - expect(Migrate.plan({ schemaVersion: 3, results: V20.results })).toBeNull(); + expect(Migrate.plan({ schemaVersion: 3, results: V20.results }, { to: 3 })).toBeNull(); + expect(Migrate.plan({ schemaVersion: 4, settings: {} })).toBeNull(); }); }); @@ -168,22 +169,132 @@ describe('run against chrome.storage.local', () => { test('applies the plan and is idempotent', async () => { await chrome.storage.local.set(clone(V20)); - const first = await Migrate.run(chrome.storage.local, { now: NOW }); + const first = await Migrate.run(chrome.storage.local, { now: NOW, to: 3 }); expect(first.from).toBe(2); const after = chrome.storage.local._getStore(); expect(after.schemaVersion).toBe(3); expect(after).not.toHaveProperty('settings'); expect(Object.keys(after.lastValues).length).toBeGreaterThan(60); - expect(await Migrate.run(chrome.storage.local, { now: NOW })).toBeNull(); + expect(await Migrate.run(chrome.storage.local, { now: NOW, to: 3 })).toBeNull(); expect(chrome.storage.local._getStore()).toEqual(after); }); test('a failed write leaves version 2 in place, to retry next time', async () => { await chrome.storage.local.set(clone(V20)); chrome.storage.local._failNext(); - await expect(Migrate.run(chrome.storage.local, { now: NOW })).rejects.toThrow(/quota/); + await expect(Migrate.run(chrome.storage.local, { now: NOW, to: 3 })).rejects.toThrow(/quota/); expect(chrome.storage.local._getStore()).not.toHaveProperty('schemaVersion'); - expect((await Migrate.run(chrome.storage.local, { now: NOW })).to).toBe(3); + expect((await Migrate.run(chrome.storage.local, { now: NOW, to: 3 })).to).toBe(3); + }); +}); + +describe('3 -> 4: into IndexedDB', () => { + require('fake-indexeddb/auto'); + const { IDBFactory } = require('fake-indexeddb'); + const DB = require('../../scripts/background/db'); + + let factory; + const openDb = () => DB.open({ indexedDB: factory }); + beforeEach(() => { + factory = new IDBFactory(); + chrome.storage.local._reset(); + }); + + // What 2.1 leaves after a two page run that is still going. + const V21 = () => ({ + schemaVersion: 3, + settings: { maxPages: 5 }, + geminiApiKey: 'k', + results: [ + { asin: 'B0AAAAAAA1', runId: 'r21', name: 'One', priceCents: 100, placements: [] }, + { asin: 'B0AAAAAAA2', runId: 'r21', name: 'Two', priceCents: 200, placements: [] }, + ], + currentItemCount: 2, + isScrapingActive: true, + scrapeRunId: 'r21', + scrapeRunPageIndex: 2, + scrapeRunMeta: { type: 'keyword', keyword: 'mats', sellerId: null, url: 'u', startedAt: 's' }, + scrapeRunPages: [{ runId: 'r21', pageIndex: 1, count: 1 }, { runId: 'r21', pageIndex: 2, count: 1 }], + lastValues: { B0AAAAAAA1: { priceCents: 100, scrapedAt: '2026-09-20T00:00:00.000Z' } }, + spreadResults: { B0AAAAAAA1: { sellerPrices: [1, 2] } }, + run: { runId: 'r21', tabId: 7, status: 'running', page: 2, maxPages: 5, startedAt: 1, heartbeat: 2, finishedAt: null }, + }); + + test('leaves only settings, the key and schemaVersion in chrome.storage.local', async () => { + await chrome.storage.local.set(V21()); + await Migrate.run(chrome.storage.local, { now: NOW, openDb }); + expect(chrome.storage.local._getStore()).toEqual({ schemaVersion: 4, settings: { maxPages: 5 }, geminiApiKey: 'k' }); + }); + + test('moves the run, its products, pages, lastValues and spread data', async () => { + await chrome.storage.local.set(V21()); + await Migrate.run(chrome.storage.local, { now: NOW, openDb }); + const db = await openDb(); + expect(await db.getMeta('latestRunId')).toBe('r21'); + expect((await db.runProducts('r21')).map((p) => [p.n, p.asin])).toEqual([[0, 'B0AAAAAAA1'], [1, 'B0AAAAAAA2']]); + expect((await db.runPages('r21')).map((p) => p.pageIndex)).toEqual([1, 2]); + expect(await db.get('lastValues', 'B0AAAAAAA1')).toMatchObject({ asin: 'B0AAAAAAA1', priceCents: 100 }); + expect(await db.get('spread', ['r21', 'B0AAAAAAA1'])).toMatchObject({ data: { sellerPrices: [1, 2] } }); + const run = await db.get('runs', 'r21'); + expect(run).toMatchObject({ state: 'failed', reason: 'updated', itemCount: 2, source: { keyword: 'mats' } }); + expect(run).not.toHaveProperty('status'); + expect(Run.describe(run, 2).text).toMatch(/updated during the run/); + db.close(); + }); + + test('a finished 2.1 run keeps how it ended', async () => { + await chrome.storage.local.set({ ...V21(), isScrapingActive: false, run: { ...V21().run, status: 'blocked', finishedAt: 5 } }); + await Migrate.run(chrome.storage.local, { now: NOW, openDb }); + const db = await openDb(); + expect(await db.get('runs', 'r21')).toMatchObject({ state: 'blocked', reason: 'blocked', finishedAt: 5 }); + db.close(); + }); + + test('a 2.0 user goes to 4 in one go and keeps every row, repeats included', async () => { + await chrome.storage.local.set(clone(V20)); + const p = await Migrate.run(chrome.storage.local, { now: NOW, openDb }); + expect(p).toMatchObject({ from: 2, to: 4 }); + expect(chrome.storage.local._getStore()).toEqual({ schemaVersion: 4 }); + const db = await openDb(); + expect(await db.getMeta('latestRunId')).toBe('legacy-2.0'); + const rows = await db.runProducts('legacy-2.0'); + expect(rows.map((r) => r.asin)).toEqual(V20.results.map((r) => r.asin)); + expect(rows.every((r) => r.url === `https://www.amazon.com/dp/${r.asin}`)).toBe(true); + expect(await db.count('lastValues')).toBe(new Set(V20.results.map((r) => r.asin)).size); + expect(await db.get('runs', 'legacy-2.0')).toMatchObject({ state: 'done', reason: 'complete' }); + db.close(); + }); + + test('without IndexedDB nothing is removed', async () => { + await chrome.storage.local.set(V21()); + await expect(Migrate.run(chrome.storage.local, { now: NOW })).rejects.toThrow(/IndexedDB/); + expect(chrome.storage.local._getStore()).toEqual(V21()); + }); + + test('a failed IndexedDB write leaves version 3 in place, and the retry is clean', async () => { + await chrome.storage.local.set(V21()); + const failing = async () => ({ write: async () => { throw Object.assign(new Error('quota'), { code: 'storage_full' }); } }); + await expect(Migrate.run(chrome.storage.local, { now: NOW, openDb: failing })).rejects.toThrow(/quota/); + expect(chrome.storage.local._getStore().schemaVersion).toBe(3); + expect(chrome.storage.local._getStore().results).toHaveLength(2); + + await Migrate.run(chrome.storage.local, { now: NOW, openDb }); + // A second copy of the data, as if the first removal had failed, lands on the same keys. + await Migrate.run({ get: async () => V21(), remove: async () => {}, set: async () => {} }, { now: NOW, openDb }); + const db = await openDb(); + expect(await db.count('products')).toBe(2); + expect(await db.count('placements')).toBe(2); + db.close(); + }); + + test('queued products from an unreleased build keep their run and are queued once', async () => { + const queued = [{ asin: 'B0QQQQQQQ1', runId: 'r0' }, { asin: 'B0AAAAAAA1', runId: 'r21' }]; + await chrome.storage.local.set({ ...V21(), syncQueue: queued }); + await Migrate.run(chrome.storage.local, { now: NOW, openDb }); + const db = await openDb(); + expect((await db.runProducts('r0')).map((p) => p.asin)).toEqual(['B0QQQQQQQ1']); + expect((await db.getAll('outbox')).map((o) => o.runId).sort()).toEqual(['r0', 'r21']); + db.close(); }); }); From ec83596321ab286fccd7b3ad533888bba164951c Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:32:33 -0400 Subject: [PATCH 06/25] Answer every worker message through the router and wire in the engine --- scripts/background/service-worker.js | 276 +++++++++++---------------- tests/unit/service-worker.test.js | 63 ++++-- 2 files changed, 155 insertions(+), 184 deletions(-) diff --git a/scripts/background/service-worker.js b/scripts/background/service-worker.js index 45960ed..7297ee6 100644 --- a/scripts/background/service-worker.js +++ b/scripts/background/service-worker.js @@ -1,128 +1,97 @@ -import { auth, db } from './firebase-init.js'; +import { auth } from './firebase-init.js'; import { onAuthStateChanged, signInWithEmailAndPassword, signOut, } from 'firebase/auth/web-extension'; import { syncToCloud } from './sync.js'; +import { createRouter } from './router.js'; +import { createEngine } from './engine.js'; +import DB from './db.js'; import Chat from '../lib/chat.js'; import Run from '../lib/run.js'; import Flags from '../lib/flags.js'; import Migrate from '../lib/migrate.js'; +import Msg from '../lib/messages.js'; /** * @fileoverview Background Service Worker * - * Central message router for the ProScan extension. Handles: - * - Message routing between popup, content scripts, and external APIs - * - Gemini calls for the AI chatbot (see scripts/lib/chat.js) - * - Extension lifecycle events (install, update, startup) + * The only writer. It runs scrapes (engine.js), keeps what they collect in + * IndexedDB (db.js), answers the popup and the content scripts through one + * router (router.js), calls Gemini for the chat, and migrates storage on + * install, update and browser start. * - * Runs as a Manifest V3 service worker -- no persistent background page. - * Wakes on message events and API calls, then goes idle. + * Runs as a Manifest V3 service worker: no persistent background page. + * Chrome stops it when idle, so nothing here keeps state in memory that a + * run needs. The run lives in chrome.storage.session and IndexedDB. * * @module ServiceWorker */ +const openDb = () => DB.open(); +const engine = createEngine({ chrome, openDb }); +const router = createRouter({ extensionId: chrome.runtime.id }); + const chatDeps = { - getStorage: (keys) => chrome.storage.local.get(keys), - fetchFn: (url, init) => fetch(url, init), + getStorage: (keys) => engine.chatData(keys), + fetchFn: (url, init) => fetch(url, init), }; -/** - * Main message listener -- routes messages between extension components. - * - * Message types handled: - * - WHO_AM_I: Tells a content script its tab id, so it can check the run is its own - * - SCRAPING_COMPLETE: Logs scrape completion - * - CHAT_STATUS: Whether a Gemini key is set, and which run the chat covers - * - CHAT_MESSAGE: Answers a question about the current run with Gemini - * - * The chat handlers read the key and the run from storage here, so the key - * never passes through the content script. - */ -chrome.runtime.onMessage.addListener((request, sender, sendResponse) => { - if (request.type === 'WHO_AM_I') { - sendResponse({ tabId: sender.tab ? sender.tab.id : null }); - return false; - } - - // Log scrape completion - if (request.type === 'SCRAPING_COMPLETE') { - console.log('[ProScan] Run ended:', request.reason, request.itemCount, 'items'); - } - - if (request.type === 'CHAT_STATUS') { - Chat.status(chatDeps).then(sendResponse, () => sendResponse({ hasKey: false, productCount: 0 })); - return true; - } - - if (request.type === 'CHAT_MESSAGE') { - Chat.answerQuestion({ question: request.question, history: request.history }, chatDeps) - .then(sendResponse, (err) => sendResponse({ error: 'Chat failed: ' + err.message })); - return true; // keep channel open for async response - } - - return true; -}); +// The run +router.on(Msg.T.START_RUN, (m) => engine.start(m)); +router.on(Msg.T.STOP_RUN, () => engine.stop()); +router.on(Msg.T.GET_STATE, () => engine.getState()); +router.on(Msg.T.PAGE_READY, (m, sender) => engine.pageReady(m, sender)); +router.on(Msg.T.PAGE_RESULT, (m, sender) => engine.pageResult(m, sender)); +router.on(Msg.T.HEARTBEAT, (m, sender) => engine.heartbeat(m, sender)); + +// Spread analysis +router.on(Msg.T.GET_RESULTS, () => engine.getResults()); +router.on(Msg.T.SPREAD_RESULT, (m) => engine.spreadResult(m)); + +// Chat: the key and the run are read here, so the key never passes through the page. +router.on(Msg.T.CHAT_STATUS, () => + Chat.status(chatDeps).catch(() => ({ hasKey: false, productCount: 0 }))); +router.on(Msg.T.CHAT_MESSAGE, (m) => + Chat.answerQuestion({ question: m.question, history: m.history }, chatDeps) + .catch((err) => ({ error: 'Chat failed: ' + err.message }))); /** Brings stored data to the current schema. Safe to call any number of times. */ function migrate() { - return Migrate.run(chrome.storage.local).then((done) => { - if (done) console.log(`[ProScan] Storage migrated from schema ${done.from} to ${done.to}`); - }, (err) => { - console.error('[ProScan] Storage migration failed, will retry:', err && err.message); - }); + return Migrate.run(chrome.storage.local, { openDb }).then((done) => { + if (done) console.log(`[ProScan] Storage migrated from schema ${done.from} to ${done.to}`); + }, (err) => { + console.error('[ProScan] Storage migration failed, will retry:', err && err.message); + }); } /** - * Handle extension installation and update events. - * On fresh install, initializes chrome.storage with default values. - * On update, migrates what the previous version stored. + * On a fresh install only the settings and the schema version go into + * chrome.storage.local. On an update, whatever the previous version stored + * is migrated. */ chrome.runtime.onInstalled.addListener((details) => { - if (details.reason === 'install') { - console.log('[ProScan] Extension installed'); - - // Initialize storage with defaults - chrome.storage.local.set({ - results: [], - currentItemCount: 0, - isScrapingActive: false, - settings: { - pageDelay: 2000, - maxPages: Run.DEFAULT_MAX_PAGES - }, - schemaVersion: Migrate.CURRENT - }); - } else if (details.reason === 'update') { - console.log('[ProScan] Extension updated to version', chrome.runtime.getManifest().version); - migrate(); - } + if (details.reason === 'install') { + console.log('[ProScan] Extension installed'); + chrome.storage.local.set({ + settings: { maxPages: Run.DEFAULT_MAX_PAGES }, + schemaVersion: Migrate.CURRENT, + }); + } else if (details.reason === 'update') { + console.log('[ProScan] Extension updated to version', chrome.runtime.getManifest().version); + migrate().then(() => engine.recover()); + } }); -/** Ends the running run as `reason` if `test(run)` holds. */ -async function endRunIf(test, reason) { - const { run } = await chrome.storage.local.get(Run.KEY); - if (Run.isActive(run) && test(run)) { - await chrome.storage.local.set({ [Run.KEY]: Run.finish(run, reason), isScrapingActive: false }); - } -} - -/** - * Clean up on browser startup. - * A run cannot survive a browser restart, since its tab id is gone. - */ +// A run cannot survive a browser restart, since its tab id is gone. chrome.runtime.onStartup.addListener(() => { - migrate(); - chrome.storage.local.set({ isScrapingActive: false }); - endRunIf(() => true, 'interrupted'); + migrate().then(() => engine.recover()); }); -// Closing the run's tab ends the run. Needs no tabs permission. -chrome.tabs.onRemoved.addListener((tabId) => { - endRunIf((run) => run.tabId === tabId, 'interrupted'); -}); +// Neither listener needs the tabs permission. +chrome.tabs.onRemoved.addListener((tabId) => { engine.tabRemoved(tabId); }); +chrome.tabs.onUpdated.addListener((tabId, info, tab) => { engine.tabUpdated(tabId, info, tab); }); // ════════════════════════════════════════════════════════════════════════════ // M3 cloud sync — Firebase Auth (extension-native) + Firestore write path. @@ -133,88 +102,67 @@ chrome.tabs.onRemoved.addListener((tabId) => { /** Resolve the current Firebase user, waiting for auth to rehydrate from * IndexedDB after a cold service-worker start. */ function currentUser() { - return new Promise((resolve) => { - if (auth.currentUser) return resolve(auth.currentUser); - const unsub = onAuthStateChanged(auth, (u) => { - unsub(); - resolve(u); - }); + return new Promise((resolve) => { + if (auth.currentUser) return resolve(auth.currentUser); + const unsub = onAuthStateChanged(auth, (u) => { + unsub(); + resolve(u); }); + }); } /** Trim a Firebase user to the popup-safe shape. */ const publicUser = (u) => - u ? { uid: u.uid, email: u.email, displayName: u.displayName } : null; + u ? { uid: u.uid, email: u.email, displayName: u.displayName } : null; /** Map Firebase auth error codes to friendly popup messages. */ function friendlyAuthError(err) { - switch (err && err.code) { - case 'auth/invalid-credential': - case 'auth/wrong-password': - case 'auth/user-not-found': - return 'Email or password is incorrect.'; - case 'auth/invalid-email': - return 'That email address does not look valid.'; - case 'auth/too-many-requests': - return 'Too many attempts. Wait a minute and try again.'; - case 'auth/network-request-failed': - return 'Network error — check your connection.'; - case 'auth/operation-not-allowed': - return 'Email sign-in is not enabled for this project yet.'; - default: - return ((err && err.message) || 'Sign-in failed.').replace(/^Firebase:\s*/, ''); - } + switch (err && err.code) { + case 'auth/invalid-credential': + case 'auth/wrong-password': + case 'auth/user-not-found': + return 'Email or password is incorrect.'; + case 'auth/invalid-email': + return 'That email address does not look valid.'; + case 'auth/too-many-requests': + return 'Too many attempts. Wait a minute and try again.'; + case 'auth/network-request-failed': + return 'Network error. Check your connection.'; + case 'auth/operation-not-allowed': + return 'Email sign-in is not enabled for this project yet.'; + default: + return ((err && err.message) || 'Sign-in failed.').replace(/^Firebase:\s*/, ''); + } } -chrome.runtime.onMessage.addListener((request, sender, sendResponse) => { - if (request.type === 'PROSCAN_AUTH_STATE') { - currentUser().then((u) => sendResponse({ user: publicUser(u) })); - return true; - } - - if (request.type === 'PROSCAN_SIGN_IN') { - signInWithEmailAndPassword(auth, request.email, request.password) - .then((cred) => sendResponse({ user: publicUser(cred.user) })) - .catch((err) => sendResponse({ error: friendlyAuthError(err) })); - return true; - } - - if (request.type === 'PROSCAN_SIGN_OUT') { - signOut(auth) - .then(() => sendResponse({ ok: true })) - .catch((err) => sendResponse({ error: err.message })); - return true; - } - - if (request.type === 'PROSCAN_EXPORT') { - if (!Flags.CLOUD_SYNC) { - sendResponse({ error: 'Export to ProScan is not available in this version.' }); - return false; - } - (async () => { - const user = await currentUser(); - if (!user) return sendResponse({ error: 'Sign in to ProScan first.' }); - const bundle = await chrome.storage.local.get([ - 'syncQueue', - 'scrapeRunMeta', - 'scrapeRunPages', - ]); - if (!bundle.syncQueue || bundle.syncQueue.length === 0) { - return sendResponse({ ok: true, written: 0, products: 0 }); - } - try { - const result = await syncToCloud(user.uid, bundle); - // Clear only the drained queue; keep lastValues so future scrapes - // still compute month-over-month deltas. - await chrome.storage.local.set({ syncQueue: [] }); - sendResponse({ ok: true, ...result }); - } catch (err) { - console.error('[ProScan] cloud export failed', err); - sendResponse({ error: (err && err.message) || 'Export failed.' }); - } - })(); - return true; - } - - return false; // not a sync message — let the other listener handle it +router.on(Msg.T.PROSCAN_AUTH_STATE, async () => ({ user: publicUser(await currentUser()) })); + +router.on(Msg.T.PROSCAN_SIGN_IN, (m) => + signInWithEmailAndPassword(auth, m.email, m.password) + .then((cred) => ({ user: publicUser(cred.user) })) + .catch((err) => ({ error: friendlyAuthError(err) }))); + +router.on(Msg.T.PROSCAN_SIGN_OUT, () => + signOut(auth).then(() => ({ ok: true }), (err) => ({ error: err.message }))); + +router.on(Msg.T.PROSCAN_EXPORT, async () => { + if (!Flags.CLOUD_SYNC) return { error: 'Export to ProScan is not available in this version.' }; + const user = await currentUser(); + if (!user) return { error: 'Sign in to ProScan first.' }; + const { bundle, seqs } = await engine.outboxBundle(); + if (bundle.syncQueue.length === 0) return { ok: true, written: 0, products: 0 }; + try { + const result = await syncToCloud(user.uid, bundle); + // Only what was written leaves the outbox; lastValues stays for deltas. + await engine.clearOutbox(seqs); + return { ok: true, ...result }; + } catch (err) { + console.error('[ProScan] cloud export failed', err); + return { error: (err && err.message) || 'Export failed.' }; + } }); + +chrome.runtime.onMessage.addListener(router.listener); + +// A worker started by any event picks up a run that is due its next page. +engine.tick(); diff --git a/tests/unit/service-worker.test.js b/tests/unit/service-worker.test.js index e2693ee..43a15d7 100644 --- a/tests/unit/service-worker.test.js +++ b/tests/unit/service-worker.test.js @@ -1,5 +1,8 @@ /** - * The service worker's own handlers, with Firebase and sync mocked out. + * @jest-environment node + * + * The service worker's own handlers, with Firebase and sync mocked out and + * fake-indexeddb standing in for IndexedDB. */ jest.mock('../../scripts/background/firebase-init.js', () => ({ auth: { currentUser: { uid: 'u1', email: 'u1@example.test', displayName: null } }, @@ -12,7 +15,9 @@ jest.mock('firebase/auth/web-extension', () => ({ }), { virtual: true }); jest.mock('../../scripts/background/sync.js', () => ({ syncToCloud: jest.fn(async () => ({ written: 1 })) })); +require('fake-indexeddb/auto'); const { syncToCloud } = require('../../scripts/background/sync.js'); +const DB = require('../../scripts/background/db'); const fs = require('fs'); const path = require('path'); @@ -23,53 +28,71 @@ const installedListeners = []; const startupListeners = []; beforeAll(() => { + chrome.runtime.id = 'test-extension'; chrome.runtime.onMessage.addListener = (fn) => messageListeners.push(fn); chrome.runtime.onInstalled = { addListener: (fn) => installedListeners.push(fn) }; chrome.runtime.onStartup = { addListener: (fn) => startupListeners.push(fn) }; - chrome.tabs.onRemoved = { addListener() {} }; require('../../scripts/background/service-worker.js'); }); -beforeEach(() => chrome.storage.local._reset()); +beforeEach(() => { + chrome.storage.local._reset(); + chrome.storage.session._reset(); +}); -function send(msg) { +function send(msg, sender = { id: 'test-extension' }) { return new Promise((resolve) => { - // The sync listener is the second one registered. - messageListeners[1](msg, {}, resolve); + const held = messageListeners[0](msg, sender, resolve); + if (!held) setTimeout(() => resolve('no answer'), 0); }); } -test('Export to ProScan is refused while cloud sync is off (2.1)', async () => { - chrome.storage.local.set({ syncQueue: [{ asin: 'B000000001', runId: 'r1' }] }); +async function settle() { + for (let i = 0; i < 400; i++) await new Promise((r) => setImmediate(r)); +} + +test('one router answers every message', () => { + expect(messageListeners).toHaveLength(1); +}); + +test('Export to ProScan is refused while cloud sync is off', async () => { const resp = await send({ type: 'PROSCAN_EXPORT' }); expect(resp.error).toMatch(/not available/); expect(syncToCloud).not.toHaveBeenCalled(); - expect(chrome.storage.local._getStore().syncQueue).toHaveLength(1); }); -async function settle() { - for (let i = 0; i < 20; i++) await Promise.resolve(); -} +test('messages meant for the popup or a tab are left alone', async () => { + expect(await send({ type: 'SPREAD_PROGRESS', current: 1, total: 2 })).toBe('no answer'); + expect(await send({ type: 'PING' })).toBe('no answer'); +}); -test('an update from 2.0 migrates storage to schema 3 (F-100)', async () => { +test('an update from 2.0 migrates storage to schema 4, into IndexedDB (F-100)', async () => { chrome.storage.local.set(JSON.parse(JSON.stringify(V20))); installedListeners.forEach((fn) => fn({ reason: 'update', previousVersion: '2.0' })); await settle(); - const s = chrome.storage.local._getStore(); - expect(s.schemaVersion).toBe(3); - expect(s.results).toHaveLength(V20.results.length); - expect(Object.keys(s.lastValues).sort()).toEqual([...new Set(V20.results.map((r) => r.asin))].sort()); + expect(chrome.storage.local._getStore()).toEqual({ schemaVersion: 4 }); + const st = await send({ type: 'GET_STATE' }); + expect(st.results).toHaveLength(V20.results.length); + const db = await DB.open(); + expect(await db.count('lastValues')).toBe(new Set(V20.results.map((r) => r.asin)).size); + db.close(); }); test('a browser start finishes a migration an update could not', async () => { chrome.storage.local.set(JSON.parse(JSON.stringify(V20))); startupListeners.forEach((fn) => fn()); await settle(); - expect(chrome.storage.local._getStore().schemaVersion).toBe(3); + expect(chrome.storage.local._getStore().schemaVersion).toBe(4); }); -test('a fresh install starts at schema 3', async () => { +test('a fresh install writes only settings and the schema version', async () => { installedListeners.forEach((fn) => fn({ reason: 'install' })); await settle(); - expect(chrome.storage.local._getStore()).toMatchObject({ schemaVersion: 3, results: [], isScrapingActive: false }); + expect(chrome.storage.local._getStore()).toEqual({ schemaVersion: 4, settings: { maxPages: 20 } }); +}); + +test('the chat reads the key from settings and the run from IndexedDB', async () => { + chrome.storage.local.set({ geminiApiKey: 'k', schemaVersion: 4 }); + const st = await send({ type: 'CHAT_STATUS' }, { id: 'test-extension', tab: { id: 3 } }); + expect(st).toMatchObject({ hasKey: true }); }); From c3b18fc61e36ad008b1d36dcfa78fd8c98efd452 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:32:41 -0400 Subject: [PATCH 07/25] Make the scraper parse only and report pages to the worker The content script no longer writes storage or navigates. It says PAGE_READY on load, parses when the worker asks, sends PAGE_RESULT and keeps a heartbeat while the next page is pending. The run scenarios of scraper-run.test.js and every test of scraper-dedupe.test.js now run in engine.test.js, over this script and the engine, with the same names and checks. scraper-run.test.js keeps the content script's own contract. The characterization test drives the engine instead of seeding storage. Products, pages, next links and helpers are unchanged on every corpus page. The snapshot changes only in the messages sent and in non-search pages, which Start now refuses instead of being forced into a run. --- scripts/content/scraper.js | 347 +++++------------- .../__snapshots__/characterize.test.js.snap | 64 ++-- tests/golden/characterize.test.js | 82 ++--- tests/unit/scraper-dedupe.test.js | 136 ------- tests/unit/scraper-run.test.js | 182 ++++----- 5 files changed, 256 insertions(+), 555 deletions(-) delete mode 100644 tests/unit/scraper-dedupe.test.js diff --git a/scripts/content/scraper.js b/scripts/content/scraper.js index ed15351..e3c2032 100644 --- a/scripts/content/scraper.js +++ b/scripts/content/scraper.js @@ -1,35 +1,28 @@ /** - * @fileoverview Amazon Product DOM Scraper + * @fileoverview Amazon search page parser, in the page. * - * Content script that extracts product data from Amazon search results - * and seller pages. Uses a cascading selector strategy to handle - * Amazon's frequently changing DOM structure. + * Parse only: this script never writes storage and never navigates. The + * service worker runs the scrape (scripts/background/engine.js). * - * Selector Strategy: - * Each data point (title, price, rating, reviews) has multiple selectors - * ordered from most-stable to least-stable. The scraper tries each in - * sequence and uses the first successful match. This makes the extension - * resilient to Amazon A/B tests and layout changes. + * On load the script says PAGE_READY. When this tab owns the running run + * and the worker is waiting for a page, it answers with the page number, + * and the script parses the page (scripts/lib/parsers.js) and sends it back + * as PAGE_RESULT. While the worker waits to open the next page, the script + * sends a HEARTBEAT every few seconds, which also wakes a stopped worker. + * Other Amazon tabs get told they have no part in the run and stay quiet. * - * Runs: - * A run belongs to the tab it started in (see scripts/lib/run.js). On each - * page load the script asks the service worker for its tab id and scrapes - * only when that tab owns the running run, so other Amazon tabs never join - * it. Each page is classified first, so a captcha ends the run as blocked - * rather than complete. The next page is the page's own Next link, opened - * after a 2 to 4 second delay, up to the run's page cap. + * Every call into the extension is guarded by alive(): after an update or + * a removal the old script stays in open tabs with no extension behind it. * * @module Scraper */ -/** Tab id of this page, once known. */ -let myTabId = null; -/** Pending navigation to the next page. */ -let navTimer = null; -/** Runs this document has already scraped, so no page is scraped twice. */ -const scrapedRuns = new Set(); -/** Runs stopped while this page was being saved. */ -const stoppedRuns = new Set(); +const HEARTBEAT_MS = 2000; +const RETRY_MS = 1000; + +/** Pages this document already reported, by run and page number. */ +const reported = new Set(); +let heartbeatTimer = null; /** False once the extension was updated or removed under this page. */ function alive() { @@ -40,19 +33,14 @@ function alive() { } } -function getRun() { - return new Promise(resolve => { - chrome.storage.local.get([Run.KEY], data => resolve((data && data[Run.KEY]) || null)); - }); -} - -/** Asks the service worker which tab this page is in. */ -function whoAmI() { +/** Sends `message` to the worker; resolves null when nothing answers. */ +function send(message) { return new Promise(resolve => { + if (!alive()) return resolve(null); try { - chrome.runtime.sendMessage({ type: 'WHO_AM_I' }, response => { + chrome.runtime.sendMessage(message, response => { if (chrome.runtime.lastError) return resolve(null); - resolve(response && typeof response.tabId === 'number' ? response.tabId : null); + resolve(response || null); }); } catch (e) { resolve(null); @@ -60,39 +48,8 @@ function whoAmI() { }); } -/** - * Resumes a run on page load when this tab owns it. Most Amazon pages have - * no run, so storage is checked before the service worker is asked. - */ -async function initialize() { - if (!alive()) return; - const run = await getRun(); - if (!Run.isActive(run)) return; - const tabId = await whoAmI(); - if (!Run.owns(run, tabId)) return; - myTabId = tabId; - - // The tab went quiet for too long (it left Amazon, or the browser slept). - if (Run.isStale(run)) { - finishRun(run.runId, 'interrupted'); - return; - } - // A reload of a page already scraped: go on from where the run was. - if (run.lastUrl === location.href) { - if (run.nextHref) scheduleNext(run.runId, run.nextHref); - return; - } - // Only the page the run opened is scraped. A search the user typed in - // this tab ends the run instead of joining it. - if (!Run.expects(run, location.href)) { - finishRun(run.runId, 'interrupted'); - return; - } - scrapeCurrentPage(run.runId); -} - -// Pure parsing lives in scripts/lib/parsers.js. These names stay for the page -// logic below and for the unit tests that load this file. +// Pure parsing lives in scripts/lib/parsers.js. These names stay for the +// unit tests that load this file. var { getText, extractPrice, parseRatingText, extractRating, parseReviewText, extractReviewCount, hasPrimeBadge, scrapeProduct @@ -110,225 +67,85 @@ function hasNextPage() { return Parsers.hasNextPage(document); } -/** - * Scrapes the current page into run `runId`. - * - * The page is parsed and classified first. A page that ends the run (the - * last page, the page cap, a captcha) is saved together with the run's end - * in one write. Otherwise the next page is opened after a short delay. - * - * @param {string} runId - */ -function scrapeCurrentPage(runId) { - if (scrapedRuns.has(runId)) return; - scrapedRuns.add(runId); - - const page = Parsers.parseSearchPage(document, window.location.href); - const results = page.products; - console.log(`[ProScan] Page kind ${page.kind}, ${results.length} product listings`); - - const syncing = Flags.CLOUD_SYNC; - const keys = [Run.KEY, 'currentItemCount', 'results', 'scrapeRunId', 'scrapeRunPageIndex', 'lastValues', 'scrapeRunPages']; - chrome.storage.local.get( - syncing ? [...keys, 'syncQueue'] : keys, - (data) => { - const run = data[Run.KEY]; - if (!run || run.runId !== runId || !Run.isActive(run)) return; - - const pageIndex = (data.scrapeRunPageIndex || 0) + 1; // 1-based page number - const ending = Run.outcome(page, pageIndex, run.maxPages); - const previousCount = data.currentItemCount || 0; - - if (results.length === 0 || ending === 'selectors_broken') { - console.log(`[ProScan] Nothing to save on this page, run ends: ${ending}`); - finishRun(runId, ending || 'complete'); - return; - } - - const runKey = data.scrapeRunId || runId; - const runResults = data.results || []; - // With sync off nothing is queued, so the queue cannot grow. - const syncQueue = syncing ? (data.syncQueue || []) : []; - const fresh = mergeRepeats(results, pageIndex, runKey, runResults, syncQueue); - - chrome.runtime.sendMessage({ - type: 'UPDATE_PROGRESS', - itemCount: fresh.length, - results: fresh - }); - - // Only an ASIN's first sighting in the run gets a delta, taken - // against the last run's snapshot, and rolls lastValues forward. - const lastValues = data.lastValues || {}; - const stampedAt = new Date().toISOString(); - fresh.forEach(product => { - product.runId = runKey; - product.pageIndex = pageIndex; - const prev = lastValues[product.asin] || null; - product.delta = Delta.computeDeltas(product, prev); - lastValues[product.asin] = Delta.snapshot(product, prev); - }); - - const newCount = previousCount + fresh.length; - if (syncing) syncQueue.push(...fresh); - const runPages = data.scrapeRunPages || []; - runPages.push({ - runId: runKey, pageIndex, count: fresh.length, placements: page.placements, - scrapedAt: stampedAt, url: location.href - }); - - const now = Date.now(); - let nextRun = { ...run, page: pageIndex, heartbeat: now, lastUrl: location.href, nextHref: page.nextHref }; - if (ending) nextRun = Run.finish(nextRun, ending, now); - - const save = { - results: [...runResults, ...fresh], - currentItemCount: newCount, - scrapeRunPageIndex: pageIndex, - lastValues: Delta.prune(lastValues), - scrapeRunPages: runPages, - [Run.KEY]: nextRun, - isScrapingActive: !ending - }; - if (syncing) save.syncQueue = syncQueue; - chrome.storage.local.set(save, () => { - if (chrome.runtime.lastError) { - console.warn('[ProScan] Could not save the page:', chrome.runtime.lastError.message); - finishRun(runId, 'storage_full'); - return; - } - if (syncing) chrome.runtime.sendMessage({ type: 'ENQUEUE_SYNC', runId: runKey, pageIndex: pageIndex }); - - // Stop landed between the read and this write, which put the run back. - if (!ending && stoppedRuns.has(runId)) { - finishRun(runId, 'stopped'); - } else if (ending) { - announceEnd(ending, newCount); - } else { - scheduleNext(runId, page.nextHref); - } - }); - } - ); +function parsePage() { + return Parsers.parseSearchPage(document, window.location.href); } -/** - * Folds this page's products into the run. Placements get the page number - * and a run-wide organic rank. An ASIN already in the run only adds its - * placements to the stored record (and its sync queue copy); the rest are - * returned as new. - */ -function mergeRepeats(products, pageIndex, runKey, runResults, syncQueue) { - const thisRun = runResults.filter(r => r.runId === runKey); - const organicBefore = thisRun.reduce( - (n, r) => n + (r.placements || []).filter(pl => !pl.sponsored).length, 0); - const known = new Map(thisRun.map(r => [r.asin, r])); - - const fresh = []; - products.forEach(product => { - const placements = (product.placements || []).map(pl => ({ - page: pageIndex, - ...pl, - rank: pl.rank === null ? null : pl.rank + organicBefore - })); - const firstRank = (placements.find(pl => pl.rank !== null) || {}).rank; - product.placements = placements; - product.organicRank = firstRank === undefined ? null : firstRank; +function stopHeartbeat() { + clearInterval(heartbeatTimer); + heartbeatTimer = null; +} - const seen = known.get(product.asin); - if (!seen) { - fresh.push(product); - return; - } - const queued = syncQueue.find(q => q.asin === product.asin && q.runId === runKey); - // The two can be one object when storage hands back shared references - new Set([seen, queued]).forEach(rec => { - if (!rec) return; - rec.placements = [...(rec.placements || []), ...placements]; - rec.sponsored = !!rec.sponsored || product.sponsored; - if (rec.organicRank == null) rec.organicRank = product.organicRank; - }); - }); - return fresh; +function startHeartbeat(runId) { + stopHeartbeat(); + heartbeatTimer = setInterval(async () => { + if (!alive()) return stopHeartbeat(); + const reply = await send({ type: Msg.T.HEARTBEAT, runId }); + if (reply && reply.active === false) stopHeartbeat(); + }, HEARTBEAT_MS); } -/** - * Opens `nextHref` after a 2 to 4 second delay, unless the run was stopped - * or handed to another tab in the meantime. - */ -function scheduleNext(runId, nextHref) { - clearTimeout(navTimer); - const delay = Run.pageDelay(); - console.log(`[ProScan] Navigating to next page: ${nextHref}`); - navTimer = setTimeout(async () => { - navTimer = null; - if (!alive()) return; - const run = await getRun(); - if (!run || run.runId !== runId || !Run.owns(run, myTabId)) return; - window.location.href = nextHref; - }, delay); +/** Parses this page as page `page` of run `runId` and reports it, once. */ +function parseAndReport(runId, page) { + const key = runId + ':' + page; + if (reported.has(key)) return; + reported.add(key); + const result = parsePage(); + console.log(`[ProScan] Page ${page} kind ${result.kind}, ${result.products.length} product listings`); + report({ type: Msg.T.PAGE_RESULT, runId, page, url: window.location.href, result }, 3); } -/** - * Ends run `runId` with `reason`, unless it already ended or another run - * replaced it. Resolves once storage has the result. - */ -function finishRun(runId, reason) { - clearTimeout(navTimer); - navTimer = null; - return new Promise(resolve => { - chrome.storage.local.get([Run.KEY, 'currentItemCount'], (data) => { - const run = data[Run.KEY]; - const count = data.currentItemCount || 0; - if (!run || run.runId !== runId || !Run.isActive(run)) return resolve(false); - chrome.storage.local.set({ [Run.KEY]: Run.finish(run, reason), isScrapingActive: false }, () => { - console.log(`[ProScan] Run ended: ${reason}. Total items: ${count}`); - announceEnd(reason, count); - resolve(true); - }); - }); - }); +async function report(message, tries) { + const reply = await send(message); + if (!reply) { + // The worker may be starting up; it drops a page it already has. + if (tries > 1 && alive()) setTimeout(() => report(message, tries - 1), RETRY_MS); + return; + } + if (reply.next === 'wait') startHeartbeat(message.runId); + else stopHeartbeat(); } -function announceEnd(reason, count) { - chrome.runtime.sendMessage({ type: 'SCRAPING_COMPLETE', reason, itemCount: count }); +/** Tells the worker this page is up, and does what it says. */ +async function announce(tries = 3) { + if (!alive()) return; + const reply = await send({ type: Msg.T.PAGE_READY, url: window.location.href }); + if (!reply) { + if (tries > 1) setTimeout(() => announce(tries - 1), RETRY_MS); + return; + } + if (reply.parse) parseAndReport(reply.runId, reply.page); + else if (reply.heartbeat) startHeartbeat(reply.runId); } /** - * Messages from the popup: + * Messages from the worker and the popup: * - PING: is this script alive, and what kind of page is this - * - START_SCRAPING {runId, tabId}: the popup made run `runId` for this tab - * - STOP_SCRAPING {runId}: cancel the pending page and end the run as stopped + * - PARSE_PAGE {runId, page}: parse this page for the run + * - RUN_ENDED: the run is over, stop the heartbeat */ chrome.runtime.onMessage.addListener((request, sender, sendResponse) => { - if (request.type === 'PING') { - const page = Parsers.parseSearchPage(document, window.location.href); - sendResponse({ ok: true, kind: page.kind, count: page.products.length }); + if (!alive() || !request) return false; + + if (request.type === Msg.T.PING) { + const page = parsePage(); + sendResponse({ ok: true, kind: page.kind, count: page.products.length, url: window.location.href }); return false; } - if (request.type === 'START_SCRAPING') { - if (!request.runId) { - sendResponse({ status: 'refused' }); + if (request.type === Msg.T.PARSE_PAGE) { + if (!request.runId || !request.page) { + sendResponse({ ok: false }); return false; } - console.log('[ProScan] Starting scrape...'); - if (typeof request.tabId === 'number') myTabId = request.tabId; - sendResponse({ status: 'started' }); - scrapeCurrentPage(request.runId); + sendResponse({ ok: true }); + setTimeout(() => parseAndReport(request.runId, request.page), 0); return false; } - if (request.type === 'STOP_SCRAPING') { - console.log('[ProScan] Stopping scrape...'); - clearTimeout(navTimer); - navTimer = null; - getRun().then(run => { - const runId = request.runId || (run && run.runId); - if (runId) stoppedRuns.add(runId); - return finishRun(runId, 'stopped'); - }).then(() => sendResponse({ stopped: true })); - return true; + if (request.type === Msg.T.RUN_ENDED) { + stopHeartbeat(); + return false; } return false; @@ -336,7 +153,7 @@ chrome.runtime.onMessage.addListener((request, sender, sendResponse) => { // At document_idle the load events may already have fired, so run now. if (document.readyState === 'loading') { - document.addEventListener('DOMContentLoaded', initialize, { once: true }); + document.addEventListener('DOMContentLoaded', () => announce(), { once: true }); } else { - initialize(); + announce(); } diff --git a/tests/golden/__snapshots__/characterize.test.js.snap b/tests/golden/__snapshots__/characterize.test.js.snap index ebe4170..bb09899 100644 --- a/tests/golden/__snapshots__/characterize.test.js.snap +++ b/tests/golden/__snapshots__/characterize.test.js.snap @@ -221,7 +221,7 @@ exports[`current price, rating and review parsers rating and review strings 1`] exports[`current scraper behavior on the corpus 2026-02/aod-synthetic 1`] = ` { - "currentItemCount": undefined, + "currentItemCount": 0, "helpers": { "hasNext": false, "nextUrl": "https://www.amazon.com/gp/aod/ajax?asin=B0TEST001&page=2&ref=sr_pg_2", @@ -232,16 +232,17 @@ exports[`current scraper behavior on the corpus 2026-02/aod-synthetic 1`] = ` "results": [], "runPage": 0, "runPages": [], - "runStatus": "interrupted", + "runStatus": null, "sent": [ - "SCRAPING_COMPLETE", + "PAGE_READY", ], + "started": "refused", } `; exports[`current scraper behavior on the corpus 2026-02/offer-classic-synthetic 1`] = ` { - "currentItemCount": undefined, + "currentItemCount": 0, "helpers": { "hasNext": false, "nextUrl": "https://www.amazon.com/gp/offer-listing/B0TEST001?page=2&ref=sr_pg_2", @@ -252,10 +253,11 @@ exports[`current scraper behavior on the corpus 2026-02/offer-classic-synthetic "results": [], "runPage": 0, "runPages": [], - "runStatus": "interrupted", + "runStatus": null, "sent": [ - "SCRAPING_COMPLETE", + "PAGE_READY", ], + "started": "refused", } `; @@ -311,9 +313,10 @@ exports[`current scraper behavior on the corpus 2026-02/search-lastpage 1`] = ` ], "runStatus": "complete", "sent": [ - "UPDATE_PROGRESS", - "SCRAPING_COMPLETE", + "PAGE_READY", + "PAGE_RESULT", ], + "started": "started", } `; @@ -461,14 +464,16 @@ exports[`current scraper behavior on the corpus 2026-02/search-results 1`] = ` ], "runStatus": "running", "sent": [ - "UPDATE_PROGRESS", + "PAGE_READY", + "PAGE_RESULT", ], + "started": "started", } `; exports[`current scraper behavior on the corpus 2026-09/aod-pinned-only 1`] = ` { - "currentItemCount": undefined, + "currentItemCount": 0, "helpers": { "hasNext": false, "nextUrl": "https://www.amazon.com/gp/product/ajax/aodAjaxMain/?asin=B09B8V1LZ3&page=2&ref=sr_pg_2", @@ -479,16 +484,17 @@ exports[`current scraper behavior on the corpus 2026-09/aod-pinned-only 1`] = ` "results": [], "runPage": 0, "runPages": [], - "runStatus": "interrupted", + "runStatus": null, "sent": [ - "SCRAPING_COMPLETE", + "PAGE_READY", ], + "started": "refused", } `; exports[`current scraper behavior on the corpus 2026-09/captcha-synthetic 1`] = ` { - "currentItemCount": undefined, + "currentItemCount": 0, "helpers": { "hasNext": false, "nextUrl": "https://www.amazon.com/s?k=yoga+mat&page=4&ref=sr_pg_4", @@ -499,16 +505,17 @@ exports[`current scraper behavior on the corpus 2026-09/captcha-synthetic 1`] = "results": [], "runPage": 0, "runPages": [], - "runStatus": "blocked", + "runStatus": null, "sent": [ - "SCRAPING_COMPLETE", + "PAGE_READY", ], + "started": "refused", } `; exports[`current scraper behavior on the corpus 2026-09/interstitial-akamai 1`] = ` { - "currentItemCount": undefined, + "currentItemCount": 0, "helpers": { "hasNext": false, "nextUrl": "https://www.amazon.com/s?k=usb+c+cable&page=3&ref=sr_pg_3", @@ -519,16 +526,17 @@ exports[`current scraper behavior on the corpus 2026-09/interstitial-akamai 1`] "results": [], "runPage": 0, "runPages": [], - "runStatus": "blocked", + "runStatus": null, "sent": [ - "SCRAPING_COMPLETE", + "PAGE_READY", ], + "started": "refused", } `; exports[`current scraper behavior on the corpus 2026-09/product-dp 1`] = ` { - "currentItemCount": undefined, + "currentItemCount": 0, "helpers": { "hasNext": false, "nextUrl": "https://www.amazon.com/dp/B09B8V1LZ3?page=2&ref=sr_pg_2", @@ -539,10 +547,11 @@ exports[`current scraper behavior on the corpus 2026-09/product-dp 1`] = ` "results": [], "runPage": 0, "runPages": [], - "runStatus": "interrupted", + "runStatus": null, "sent": [ - "SCRAPING_COMPLETE", + "PAGE_READY", ], + "started": "refused", } `; @@ -628,9 +637,10 @@ exports[`current scraper behavior on the corpus 2026-09/search-no-pagination-syn ], "runStatus": "complete", "sent": [ - "UPDATE_PROGRESS", - "SCRAPING_COMPLETE", + "PAGE_READY", + "PAGE_RESULT", ], + "started": "started", } `; @@ -754,8 +764,10 @@ exports[`current scraper behavior on the corpus 2026-09/search-title-recipe-synt ], "runStatus": "running", "sent": [ - "UPDATE_PROGRESS", + "PAGE_READY", + "PAGE_RESULT", ], + "started": "started", } `; @@ -2559,7 +2571,9 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` ], "runStatus": "running", "sent": [ - "UPDATE_PROGRESS", + "PAGE_READY", + "PAGE_RESULT", ], + "started": "started", } `; diff --git a/tests/golden/characterize.test.js b/tests/golden/characterize.test.js index 34542c0..7e1027c 100644 --- a/tests/golden/characterize.test.js +++ b/tests/golden/characterize.test.js @@ -1,9 +1,12 @@ /** + * @jest-environment node + * * Records what today's code does on every corpus page, bugs included, so * refactors can prove they changed nothing. The snapshot is the contract: * update it only in a commit that means to change behavior. */ const { loadContentScript } = require('../setup/dom-helpers'); +const { createRig, settle } = require('../setup/engine-rig'); const { corpus } = require('../setup/corpus'); const Analyzer = require('../../scripts/modules/analyzer'); const Price = require('../../scripts/modules/price'); @@ -16,56 +19,49 @@ const PRICE_STRINGS = [ 'Was: $30.00 Now: $20.00', 'N/A', '', null, 19.99, ]; -// Fake timers keep the 2 second page navigation from firing. -beforeAll(() => jest.useFakeTimers({ doNotFake: ['nextTick', 'setImmediate', 'queueMicrotask', 'Date'] })); +beforeAll(() => jest.useFakeTimers({ doNotFake: ['nextTick', 'setImmediate', 'queueMicrotask'] })); afterAll(() => jest.useRealTimers()); function attempt(fn) { try { return fn(); } catch (e) { return `throws ${e.constructor.name}`; } } -// Lets the scraper's promise chains settle; the storage mock is synchronous. -async function flush() { - for (let i = 0; i < 20; i++) await Promise.resolve(); -} - +/** + * Starts a run on the page through the service worker engine and the real + * content script (tests/setup/engine-rig.js), then lets the page delay pass + * once to see where the run goes next. Other pages answer 404. + */ async function runScrape(page) { - let listener = null; - const sent = []; - const origAdd = chrome.runtime.onMessage.addListener; - const origSend = chrome.runtime.sendMessage; - chrome.runtime.onMessage.addListener = (fn) => { listener = fn; }; - chrome.runtime.sendMessage = (msg) => { sent.push(msg.type); }; - chrome.storage.local._reset(); - try { - const ctx = loadContentScript('scripts/content/scraper.js', page.html, page.expected.url); - const run = { ...Run.create({ runId: 'run-golden', tabId: 1, now: 0 }), startedAt: 0, heartbeat: 0 }; - chrome.storage.local.set({ scrapeRunId: 'run-golden', scrapeRunPageIndex: 0, run, isScrapingActive: true }); - listener({ type: 'START_SCRAPING', runId: 'run-golden', tabId: 1 }, {}, () => {}); - await flush(); - const store = chrome.storage.local._getStore(); - const nav = ctx.console.log.mock.calls.map((c) => c[0]).filter((l) => /Navigating/.test(l)); - return { - sent, - nav, - isScrapingActive: store.isScrapingActive, - runStatus: store.run.status, - runPage: store.run.page, - currentItemCount: store.currentItemCount, - runPages: (store.scrapeRunPages || []).map(({ pageIndex, count, url }) => ({ pageIndex, count, url })), - results: (store.results || []).map((r) => ({ ...r, scrapedAt: typeof r.scrapedAt })), - helpers: { - total: ctx.getTotalResults(), - hasNext: ctx.hasNextPage(), - nextUrl: ctx.getNextPageUrl(), - }, - }; - } finally { - chrome.storage.local._reset(); - chrome.runtime.onMessage.addListener = origAdd; - chrome.runtime.sendMessage = origSend; - jest.clearAllTimers(); - } + const rig = createRig({ site: (url) => (url === page.expected.url ? page.html : null), random: () => 0 }); + const tabId = rig.openTab(page.expected.url); + await settle(); + const started = await rig.popup({ type: 'START_RUN', tabId }); + await settle(); + const st = await rig.popup({ type: 'GET_STATE' }); + const sent = rig.sent.map((m) => m.type); + const before = st.run || {}; + await settle(4000); + const nav = rig.served.slice(1).map((s) => `[ProScan] Navigating to next page: ${s.url}`); + const ctx = rig.tabCtx(tabId); + const runId = before.runId; + return { + started: started.ok ? 'started' : started.error, + sent, + nav, + isScrapingActive: Run.isActive(before), + runStatus: before.reason || before.state || null, + runPage: before.page || 0, + currentItemCount: before.itemCount || 0, + runPages: (st.pages || []).map(({ pageIndex, count, url }) => ({ pageIndex, count, url })), + results: (st.results || []).map(({ n, ...r }) => ({ + ...r, runId: r.runId === runId ? 'run-golden' : r.runId, scrapedAt: typeof r.scrapedAt, + })), + helpers: ctx ? { + total: ctx.getTotalResults(), + hasNext: ctx.hasNextPage(), + nextUrl: ctx.getNextPageUrl(), + } : null, + }; } describe('current scraper behavior on the corpus', () => { diff --git a/tests/unit/scraper-dedupe.test.js b/tests/unit/scraper-dedupe.test.js deleted file mode 100644 index 2ccdc9b..0000000 --- a/tests/unit/scraper-dedupe.test.js +++ /dev/null @@ -1,136 +0,0 @@ -/** - * ASIN dedupe across the pages of one run (F-27) and null-safe deltas - * (F-28), through the real scraper with the storage mock. - */ -const { loadContentScript } = require('../setup/dom-helpers'); -const Run = require('../../scripts/lib/run'); - -const URL1 = 'https://www.amazon.com/s?k=w'; -const URL2 = 'https://www.amazon.com/s?k=w&page=2'; - -const card = (asin, price, { ad = false, rating = true } = {}) => ` -
-

Item ${asin}

-
${price}
- ${rating ? '
4.5 out of 5 stars
(1K)' : ''} -
`; -const page = (cards, next) => `
${cards}
${ - next ? `Next` : 'Next' -}`; - -const PAGE1 = page(card('B0A', '$5.00', { ad: true }) + card('B0B', '$2.00') + card('B0A', '$5.00'), '/s?k=w&page=2'); -const PAGE2 = page(card('B0C', '$3.00') + card('B0A', '$5.00') + card('B0B', '$2.00', { rating: false }), null); - -const settle = async () => { - for (let i = 0; i < 30; i++) await Promise.resolve(); -}; - -const orig = {}; -beforeEach(() => { - jest.useFakeTimers({ doNotFake: ['nextTick', 'setImmediate', 'queueMicrotask', 'Date'] }); - orig.add = chrome.runtime.onMessage.addListener; - orig.send = chrome.runtime.sendMessage; - orig.id = chrome.runtime.id; - chrome.runtime.id = 'test-extension'; - chrome.runtime.onMessage.addListener = () => {}; - chrome.runtime.sendMessage = (msg, cb) => { - if (msg.type === 'WHO_AM_I' && cb) cb({ tabId: 5 }); - }; - chrome.storage.local._reset(); -}); - -afterEach(() => { - chrome.runtime.onMessage.addListener = orig.add; - chrome.runtime.sendMessage = orig.send; - chrome.runtime.id = orig.id; - chrome.storage.local._reset(); - jest.clearAllTimers(); - jest.useRealTimers(); -}); - -async function scrapeTwoPages(lastValues, flags = { CLOUD_SYNC: true }) { - const run = { ...Run.create({ runId: 'r1', tabId: 5 }), nextHref: URL1 }; - chrome.storage.local.set({ run, scrapeRunId: 'r1', scrapeRunPageIndex: 0, isScrapingActive: true, lastValues }); - loadContentScript('scripts/content/scraper.js', PAGE1, URL1, { flags }); - await settle(); - jest.clearAllTimers(); - loadContentScript('scripts/content/scraper.js', PAGE2, URL2, { flags }); - await settle(); - return chrome.storage.local._getStore(); -} - -const prevRun = (priceCents) => ({ priceCents, rating: 4.5, reviewCount: 900, runId: 'r0', scrapedAt: null }); - -test('each ASIN is stored once per run, with every placement', async () => { - const store = await scrapeTwoPages({}); - expect(store.results.map((r) => r.asin)).toEqual(['B0A', 'B0B', 'B0C']); - expect(store.currentItemCount).toBe(3); - expect(store.syncQueue.map((r) => r.asin)).toEqual(['B0A', 'B0B', 'B0C']); - - const a = store.results[0]; - expect(a.url).toBe('https://www.amazon.com/dp/B0A'); - expect(a.sponsored).toBe(true); - expect(a.organicRank).toBe(2); - expect(a.placements).toEqual([ - { page: 1, position: 1, sponsored: true, rank: null }, - { page: 1, position: 3, sponsored: false, rank: 2 }, - { page: 2, position: 2, sponsored: false, rank: 4 }, - ]); - expect(store.syncQueue[0].placements).toEqual(a.placements); - - expect(store.results[2]).toMatchObject({ asin: 'B0C', organicRank: 3, sponsored: false }); - expect(store.scrapeRunPages.map((p) => [p.count, p.placements])).toEqual([[2, 3], [1, 3]]); -}); - -test('with cloud sync off (2.1) nothing is queued (F-26)', async () => { - const store = await scrapeTwoPages({}, { CLOUD_SYNC: false }); - expect(store.results.map((r) => r.asin)).toEqual(['B0A', 'B0B', 'B0C']); - expect(store.results[0].placements).toHaveLength(3); - expect(store).not.toHaveProperty('syncQueue'); -}); - -test('a queue left by an older build is not grown while sync is off', async () => { - chrome.storage.local.set({ syncQueue: [{ asin: 'B0OLD', runId: 'r0' }] }); - const store = await scrapeTwoPages({}, { CLOUD_SYNC: false }); - expect(store.syncQueue).toEqual([{ asin: 'B0OLD', runId: 'r0' }]); -}); - -test('each page write drops lastValues older than the age limit (F-26)', async () => { - const stale = { priceCents: 1, rating: null, reviewCount: null, runId: 'r0', - scrapedAt: new Date(Date.now() - 400 * 86400000).toISOString() }; - const store = await scrapeTwoPages({ B0GONE: stale }); - expect(store.lastValues).not.toHaveProperty('B0GONE'); - expect(Object.keys(store.lastValues).sort()).toEqual(['B0A', 'B0B', 'B0C']); -}); - -test('a repeat in the same run keeps the delta against the last run', async () => { - const store = await scrapeTwoPages({ B0A: prevRun(600), B0B: prevRun(250) }); - const [a, b] = store.results; - expect(a.delta).toEqual({ isNew: false, dPriceCents: -100, dRating: 0, dReviews: 100 }); - // B0B lost its rating markup on page 2; the page 1 delta stays - expect(b.delta).toEqual({ isNew: false, dPriceCents: -50, dRating: 0, dReviews: 100 }); - expect(store.syncQueue[0].delta.dPriceCents).toBe(-100); - expect(store.lastValues.B0B).toMatchObject({ priceCents: 200, rating: 4.5, reviewCount: 1000, runId: 'r1' }); -}); - -test('a card with no rating gives a null delta, not a fake drop (F-28)', async () => { - const run = { ...Run.create({ runId: 'r1', tabId: 5 }), nextHref: URL1 }; - chrome.storage.local.set({ run, scrapeRunId: 'r1', scrapeRunPageIndex: 0, isScrapingActive: true, lastValues: { B0B: prevRun(200) } }); - loadContentScript('scripts/content/scraper.js', page(card('B0B', '$2.00', { rating: false }), null), URL1); - await settle(); - const store = chrome.storage.local._getStore(); - expect(store.results[0]).toMatchObject({ rating: null, reviewCount: null }); - expect(store.results[0].delta).toEqual({ isNew: false, dPriceCents: 0, dRating: null, dReviews: null }); -}); - -test('organic ranks start at 1 even when an older run is still stored', async () => { - const old = { asin: 'B0Z', runId: 'r0', placements: [{ page: 1, position: 1, sponsored: false, rank: 1 }] }; - const run = { ...Run.create({ runId: 'r1', tabId: 5 }), nextHref: URL1 }; - chrome.storage.local.set({ run, scrapeRunId: 'r1', scrapeRunPageIndex: 0, isScrapingActive: true, results: [old] }); - loadContentScript('scripts/content/scraper.js', page(card('B0B', '$2.00') + card('B0Z', '$1.00'), null), URL1); - await settle(); - const store = chrome.storage.local._getStore(); - expect(store.results.map((r) => [r.asin, r.runId, r.organicRank])).toEqual([ - ['B0Z', 'r0', undefined], ['B0B', 'r1', 1], ['B0Z', 'r1', 2], - ]); -}); diff --git a/tests/unit/scraper-run.test.js b/tests/unit/scraper-run.test.js index 0e40851..d07b110 100644 --- a/tests/unit/scraper-run.test.js +++ b/tests/unit/scraper-run.test.js @@ -1,11 +1,11 @@ /** - * The scraper's run handling: tab ownership, Stop and the page cap. The - * Chromium harness in tests/e2e covers the same in a real browser. + * The content script's side of a run: it parses and reports, and nothing + * else. Tab ownership, Stop and the page cap now live in the service + * worker and are tested in engine.test.js, over this same script. */ const fs = require('fs'); const path = require('path'); const { loadContentScript } = require('../setup/dom-helpers'); -const Run = require('../../scripts/lib/run'); const html = fs.readFileSync(path.join(__dirname, '../pages/2026-09/search-title-recipe-synthetic.html'), 'utf8'); const URL1 = 'https://www.amazon.com/s?k=widget'; @@ -16,7 +16,7 @@ const settle = async () => { let listener; let sent; -let tabIdForWorker; +let replies; const orig = {}; beforeEach(() => { @@ -26,12 +26,13 @@ beforeEach(() => { orig.id = chrome.runtime.id; listener = null; sent = []; - tabIdForWorker = null; + replies = {}; chrome.runtime.id = 'test-extension'; chrome.runtime.onMessage.addListener = (fn) => { listener = fn; }; chrome.runtime.sendMessage = (msg, cb) => { - sent.push(msg.type); - if (msg.type === 'WHO_AM_I' && cb) cb({ tabId: tabIdForWorker }); + sent.push(msg); + const r = replies[msg.type]; + if (cb) cb(typeof r === 'function' ? r(msg) : r); }; chrome.storage.local._reset(); }); @@ -40,120 +41,129 @@ afterEach(() => { chrome.runtime.onMessage.addListener = orig.add; chrome.runtime.sendMessage = orig.send; chrome.runtime.id = orig.id; - chrome.storage.local._reset(); jest.clearAllTimers(); jest.useRealTimers(); }); -function seedRun(extra = {}) { - const run = { ...{ ...Run.create({ runId: 'r1', tabId: 5 }), nextHref: URL1 }, ...extra }; - chrome.storage.local.set({ run, scrapeRunId: 'r1', scrapeRunPageIndex: 0, isScrapingActive: true }); - return run; -} +const types = () => sent.map((m) => m.type); -test('a page load in another tab does not join the run', async () => { - seedRun(); - tabIdForWorker = 9; +test('on load it says PAGE_READY and nothing more when the tab has no run', async () => { + replies.PAGE_READY = { idle: true }; loadContentScript('scripts/content/scraper.js', html, URL1); await settle(); - expect(chrome.storage.local._getStore().results).toBeUndefined(); - expect(sent).toEqual(['WHO_AM_I']); -}); - -test('a page load in the run tab scrapes and schedules the next page', async () => { - seedRun(); - tabIdForWorker = 5; - const ctx = loadContentScript('scripts/content/scraper.js', html, URL1); + jest.advanceTimersByTime(10000); await settle(); - const store = chrome.storage.local._getStore(); - expect(store.results.length).toBeGreaterThan(0); - expect(store.run).toMatchObject({ status: 'running', page: 1, lastUrl: URL1 }); - expect(ctx.console.log.mock.calls.some((c) => /Navigating to next page/.test(c[0]))).toBe(true); + expect(types()).toEqual(['PAGE_READY']); + expect(sent[0].url).toBe(URL1); }); -test('Stop cancels the pending page and records the run as stopped', async () => { - seedRun(); - tabIdForWorker = 5; +test('asked to parse on load, it reports the page and then keeps a heartbeat', async () => { + replies.PAGE_READY = { parse: true, runId: 'r1', page: 2 }; + replies.PAGE_RESULT = { ok: true, next: 'wait' }; + replies.HEARTBEAT = { active: true }; loadContentScript('scripts/content/scraper.js', html, URL1); await settle(); - const respond = jest.fn(); - expect(listener({ type: 'STOP_SCRAPING', runId: 'r1' }, {}, respond)).toBe(true); + const report = sent.find((m) => m.type === 'PAGE_RESULT'); + expect(report).toMatchObject({ runId: 'r1', page: 2, url: URL1 }); + expect(report.result.kind).toBe('results'); + expect(report.result.products.length).toBeGreaterThan(0); + jest.advanceTimersByTime(4100); await settle(); - expect(respond).toHaveBeenCalledWith({ stopped: true }); - expect(jest.getTimerCount()).toBe(0); - expect(chrome.storage.local._getStore().run.status).toBe('stopped'); - expect(chrome.storage.local._getStore().isScrapingActive).toBe(false); + expect(types().filter((t) => t === 'HEARTBEAT')).toHaveLength(2); }); -test('Stop during a page save still ends the run as stopped', async () => { - seedRun(); - tabIdForWorker = 5; - const realSet = chrome.storage.local.set; - let held = null; - chrome.storage.local.set = (items, cb) => { - if (items.results && !held) { held = () => realSet.call(chrome.storage.local, items, cb); return; } - return realSet.call(chrome.storage.local, items, cb); - }; - try { - loadContentScript('scripts/content/scraper.js', html, URL1); - await settle(); - expect(held).not.toBeNull(); - const respond = jest.fn(); - listener({ type: 'STOP_SCRAPING', runId: 'r1' }, {}, respond); - await settle(); - held(); +test('the heartbeat stops when the worker says the run is over', async () => { + replies.PAGE_READY = { parse: true, runId: 'r1', page: 1 }; + replies.PAGE_RESULT = { ok: true, next: 'wait' }; + replies.HEARTBEAT = { active: false }; + loadContentScript('scripts/content/scraper.js', html, URL1); + await settle(); + for (let i = 0; i < 3; i++) { + jest.advanceTimersByTime(2000); await settle(); - } finally { - chrome.storage.local.set = realSet; } - expect(chrome.storage.local._getStore().run.status).toBe('stopped'); - expect(jest.getTimerCount()).toBe(0); + expect(types().filter((t) => t === 'HEARTBEAT')).toHaveLength(1); +}); + +test('the last page stops the heartbeat, and RUN_ENDED does too', async () => { + replies.PAGE_READY = { parse: true, runId: 'r1', page: 1 }; + replies.PAGE_RESULT = { ok: true, next: 'end' }; + loadContentScript('scripts/content/scraper.js', html, URL1); + await settle(); + jest.advanceTimersByTime(6000); + await settle(); + expect(types()).toEqual(['PAGE_READY', 'PAGE_RESULT']); + + replies.PAGE_RESULT = { ok: true, next: 'wait' }; + listener({ type: 'PARSE_PAGE', runId: 'r2', page: 1 }, {}, () => {}); + jest.advanceTimersByTime(0); + await settle(); + listener({ type: 'RUN_ENDED', runId: 'r2' }, {}, () => {}); + jest.advanceTimersByTime(6000); + await settle(); + expect(types().filter((t) => t === 'HEARTBEAT')).toEqual([]); +}); + +test('a page is reported once per run, however often it is asked', async () => { + replies.PAGE_READY = { parse: true, runId: 'r1', page: 1 }; + replies.PAGE_RESULT = { ok: true, next: 'wait' }; + loadContentScript('scripts/content/scraper.js', html, URL1); + await settle(); + const respond = jest.fn(); + expect(listener({ type: 'PARSE_PAGE', runId: 'r1', page: 1 }, {}, respond)).toBe(false); + expect(respond).toHaveBeenCalledWith({ ok: true }); + jest.advanceTimersByTime(0); + await settle(); + expect(types().filter((t) => t === 'PAGE_RESULT')).toHaveLength(1); }); -test('the page cap ends the run on that page', async () => { - seedRun({ maxPages: 1 }); - tabIdForWorker = 5; +test('with no answer from a starting worker it tries again', async () => { loadContentScript('scripts/content/scraper.js', html, URL1); await settle(); - const store = chrome.storage.local._getStore(); - expect(store.run.status).toBe('complete'); - expect(store.isScrapingActive).toBe(false); - expect(jest.getTimerCount()).toBe(0); + expect(types()).toEqual(['PAGE_READY']); + replies.PAGE_READY = { idle: true }; + jest.advanceTimersByTime(1000); + await settle(); + expect(types()).toEqual(['PAGE_READY', 'PAGE_READY']); }); -test('START without a run id is refused', () => { +test('PARSE_PAGE without a run id is refused', () => { loadContentScript('scripts/content/scraper.js', html, URL1); const respond = jest.fn(); - listener({ type: 'START_SCRAPING' }, {}, respond); - expect(respond).toHaveBeenCalledWith({ status: 'refused' }); + listener({ type: 'PARSE_PAGE' }, {}, respond); + expect(respond).toHaveBeenCalledWith({ ok: false }); }); -test('PING answers with the page kind', () => { +test('PING answers with the page kind and URL', () => { loadContentScript('scripts/content/scraper.js', html, URL1); const respond = jest.fn(); listener({ type: 'PING' }, {}, respond); - expect(respond).toHaveBeenCalledWith(expect.objectContaining({ ok: true, kind: 'results' })); + expect(respond).toHaveBeenCalledWith(expect.objectContaining({ ok: true, kind: 'results', url: URL1 })); }); -test('a different search typed in the run tab ends the run instead of joining it', async () => { - seedRun({ page: 1, lastUrl: URL1, nextHref: 'https://www.amazon.com/s?k=widget&page=2&ref=sr_pg_1' }); - tabIdForWorker = 5; - loadContentScript('scripts/content/scraper.js', html, 'https://www.amazon.com/s?k=garden+hose'); +test('it never writes storage', async () => { + replies.PAGE_READY = { parse: true, runId: 'r1', page: 1 }; + replies.PAGE_RESULT = { ok: true, next: 'wait' }; + const set = jest.spyOn(chrome.storage.local, 'set'); + loadContentScript('scripts/content/scraper.js', html, URL1); await settle(); - const store = chrome.storage.local._getStore(); - expect(store.results).toBeUndefined(); - expect(store.run.status).toBe('interrupted'); - expect(store.isScrapingActive).toBe(false); - expect(jest.getTimerCount()).toBe(0); + jest.advanceTimersByTime(5000); + await settle(); + expect(set).not.toHaveBeenCalled(); + set.mockRestore(); }); -test('a stale run is ended, not resumed, when its tab loads Amazon again', async () => { - seedRun({ heartbeat: Date.now() - Run.STALE_MS - 1000 }); - tabIdForWorker = 5; +test('after the extension is updated under the page it goes quiet (alive guard)', async () => { + replies.PAGE_READY = { parse: true, runId: 'r1', page: 1 }; + replies.PAGE_RESULT = { ok: true, next: 'wait' }; loadContentScript('scripts/content/scraper.js', html, URL1); await settle(); - const store = chrome.storage.local._getStore(); - expect(store.results).toBeUndefined(); - expect(store.run.status).toBe('interrupted'); - expect(jest.getTimerCount()).toBe(0); + const before = sent.length; + chrome.runtime.id = undefined; + jest.advanceTimersByTime(6000); + await settle(); + expect(sent.length).toBe(before); + const respond = jest.fn(); + expect(listener({ type: 'PING' }, {}, respond)).toBe(false); + expect(respond).not.toHaveBeenCalled(); }); From 5caeaf7a2a8d27e09b5e51e5b4ab717c015b1d71 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:32:41 -0400 Subject: [PATCH 08/25] Send spread results to the worker instead of writing storage from the page --- scripts/content/offer-fetcher.js | 76 +++++++++++++++++++------------- 1 file changed, 46 insertions(+), 30 deletions(-) diff --git a/scripts/content/offer-fetcher.js b/scripts/content/offer-fetcher.js index bbb6eff..c2be592 100644 --- a/scripts/content/offer-fetcher.js +++ b/scripts/content/offer-fetcher.js @@ -7,10 +7,11 @@ * * Flow: * 1. Receives START_SPREAD_ANALYSIS message from popup - * 2. Reads product ASINs from chrome.storage + * 2. Asks the service worker for the last run's products (GET_RESULTS) * 3. For each ASIN, fetches the offer listing page * 4. Parses seller prices using DOMParser + cascading selectors - * 5. Stores spread data back in chrome.storage + * 5. Sends each product's spread data to the worker (SPREAD_RESULT), which + * stores it; this script never writes storage * 6. Sends progress updates to popup * * Rate limiting: 2-second delay between requests to avoid Amazon throttling. @@ -22,6 +23,30 @@ /** @type {boolean} Whether spread analysis is currently running */ let isAnalyzing = false; +/** False once the extension was updated or removed under this page. */ +function offersAlive() { + try { + return !!(chrome.runtime && chrome.runtime.id); + } catch (e) { + return false; + } +} + +/** Sends `message` to the extension; resolves null when nothing answers. */ +function tell(message) { + return new Promise(resolve => { + if (!offersAlive()) return resolve(null); + try { + chrome.runtime.sendMessage(message, response => { + if (chrome.runtime.lastError) return resolve(null); + resolve(response || null); + }); + } catch (e) { + resolve(null); + } + }); +} + // Pure parsing lives in scripts/lib/parsers.js. These names stay for the fetch // loop below and for the unit tests that load this file. var { buildOfferUrl, buildAodUrl, parseOfferPrice, extractPricesFromDocument } = Parsers; @@ -112,8 +137,8 @@ async function runSpreadAnalysis(products) { console.log(`[ProScan Spread] Starting analysis for ${total} products`); for (let i = 0; i < total; i++) { - // Check if user cancelled - if (!isAnalyzing) { + // Check if user cancelled, or the extension went away + if (!isAnalyzing || !offersAlive()) { console.log('[ProScan Spread] Analysis cancelled by user'); break; } @@ -139,14 +164,12 @@ async function runSpreadAnalysis(products) { console.log(`[ProScan Spread] No offers found for ${asin}`); } - // Save progress to storage incrementally - await new Promise(resolve => { - chrome.storage.local.set({ spreadResults }, resolve); - }); + // The worker stores it + await tell({ type: Msg.T.SPREAD_RESULT, asin, data: spreadResults[asin] }); // Send progress update to popup - chrome.runtime.sendMessage({ - type: 'SPREAD_PROGRESS', + tell({ + type: Msg.T.SPREAD_PROGRESS, current: i + 1, total: total, asin: asin, @@ -162,13 +185,8 @@ async function runSpreadAnalysis(products) { // Analysis complete isAnalyzing = false; - // Final save - await new Promise(resolve => { - chrome.storage.local.set({ spreadResults, isSpreadAnalyzing: false }, resolve); - }); - - chrome.runtime.sendMessage({ - type: 'SPREAD_ANALYSIS_COMPLETE', + tell({ + type: Msg.T.SPREAD_ANALYSIS_COMPLETE, totalAnalyzed: Object.keys(spreadResults).length }); @@ -183,34 +201,32 @@ async function runSpreadAnalysis(products) { * - STOP_SPREAD_ANALYSIS: Cancel the running analysis */ chrome.runtime.onMessage.addListener((request, sender, sendResponse) => { - if (request.type === 'START_SPREAD_ANALYSIS') { + if (!offersAlive() || !request) return false; + + if (request.type === Msg.T.START_SPREAD_ANALYSIS) { if (isAnalyzing) { sendResponse({ status: 'already_running' }); - return true; + return false; } - // Read products from storage and start analysis - chrome.storage.local.get(['results'], (data) => { - const products = data.results || []; + tell({ type: Msg.T.GET_RESULTS }).then((data) => { + const products = (data && data.results) || []; if (products.length === 0) { sendResponse({ status: 'no_products' }); return; } - - chrome.storage.local.set({ isSpreadAnalyzing: true }, () => { - runSpreadAnalysis(products); - sendResponse({ status: 'started', total: products.length }); - }); + runSpreadAnalysis(products); + sendResponse({ status: 'started', total: products.length }); }); return true; // Keep channel open for async response } - if (request.type === 'STOP_SPREAD_ANALYSIS') { + if (request.type === Msg.T.STOP_SPREAD_ANALYSIS) { isAnalyzing = false; sendResponse({ status: 'stopping' }); - return true; + return false; } - return true; + return false; }); From cb6363166620aafa1dcd276f26fa75976ee8eb76 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:32:41 -0400 Subject: [PATCH 09/25] Render the popup from the worker's state and start and stop through it --- popup/popup.html | 1 + popup/popup.js | 259 ++++++++++++++++++----------------------------- 2 files changed, 102 insertions(+), 158 deletions(-) diff --git a/popup/popup.html b/popup/popup.html index 9e800b8..d906be8 100644 --- a/popup/popup.html +++ b/popup/popup.html @@ -6,6 +6,7 @@ + diff --git a/popup/popup.js b/popup/popup.js index 7cb30b9..0cd0fd1 100644 --- a/popup/popup.js +++ b/popup/popup.js @@ -6,14 +6,15 @@ * Depends on Storage, Analyzer, and Exporter modules loaded via popup.html. * * State Management: - * - isScrapingActive: tracks whether a scrape is in progress - * - scrapedItemCount: running total across paginated pages - * - currentResults: full array of scraped product objects + * The service worker owns the run and everything it found. The popup never + * writes either: it asks the worker (GET_STATE) and renders the whole view + * from the run record it gets back, and again whenever the record in + * chrome.storage.session changes. * * Communication: - * - Pings the tab, then sends START_SCRAPING / STOP_SCRAPING to the run's tab - * - Receives UPDATE_PROGRESS, SCRAPING_COMPLETE, PAGE_COMPLETE from content script - * - Renders the run's end from the run record (scripts/lib/run.js) + * - START_RUN / STOP_RUN / GET_STATE to the worker (scripts/lib/messages.js) + * - PING to a tab that did not answer, to start after a reload + * - SPREAD_PROGRESS and SPREAD_ANALYSIS_COMPLETE from the offer fetcher * * @module PopupUI * @requires Storage @@ -24,12 +25,15 @@ /** @type {boolean} Whether scraping is currently in progress */ let isScrapingActive = false; -/** @type {number} Running count of scraped items across all pages */ -let scrapedItemCount = 0; - -/** @type {Object[]} Full array of scraped product objects */ +/** @type {Object[]} Products of the latest run, from the worker */ let currentResults = []; +/** @type {Object} Spread data of the latest run, ASIN to offers */ +let currentSpread = {}; + +/** @type {number} Page of the live run the results were last fetched for */ +let renderedPage = -1; + /** @type {boolean} Whether spread analysis is currently running */ let isSpreadAnalyzing = false; @@ -202,56 +206,65 @@ function setDownloadLoading(button, loading) { } /** - * Initialize the popup UI from persisted storage state. - * Called on DOMContentLoaded to restore scraping state, results, - * and dashboard stats from the previous session. + * Render the whole view from the worker's state: the run record, its + * products and its spread data. * - * @async + * @param {{run: ?Object, results: Object[], spread: Object}} state */ -async function initializeUI() { - const state = await Storage.getScrapingState(); - let run = await Storage.get(Run.KEY); - - // A run whose tab stopped checking in, or a flag left by an older - // version with no run record, is not running. - if (Run.isStale(run)) { - run = Run.finish(run, 'interrupted'); - await Storage.setMultiple({ [Run.KEY]: run, [Storage.KEYS.IS_SCRAPING]: false }); - } else if (state.isActive && !Run.isActive(run)) { - await Storage.set(Storage.KEYS.IS_SCRAPING, false); - } - +function render(state) { + const run = (state && state.run) || null; + currentResults = (state && state.results) || []; + currentSpread = (state && state.spread) || {}; isScrapingActive = Run.isActive(run); - scrapedItemCount = state.itemCount; - currentResults = state.results; + renderedPage = run ? run.page : -1; + elements.itemCount.textContent = currentResults.length; + elements.avgRating.textContent = '-'; + elements.avgPrice.textContent = '-'; + elements.insightsPreview.classList.add('hidden'); updateStats(currentResults); setScrapingState(isScrapingActive); const line = Run.describe(run, currentResults.length); - if (line && (isScrapingActive || run.status !== 'complete')) { + if (line && (isScrapingActive || run.reason !== 'complete')) { updateStatus(line.text, line.type); - } else if (!isScrapingActive && currentResults.length > 0) { + } else if (currentResults.length > 0) { updateStatus('Ready to download ' + currentResults.length + ' products', 'success'); } - const usage = await Storage.usage().catch(() => null); - if (usage && usage.nearFull && !isScrapingActive) { - const pct = Math.round((usage.bytes / usage.quota) * 100); - updateStatus(`Browser storage is ${pct}% full. Download your results before the next run.`, 'warning'); - } + const showSpread = !isScrapingActive && currentResults.length > 0; + elements.spreadButton.classList.toggle('hidden', !showSpread); + if (showSpread && Object.keys(currentSpread).length > 0) displaySpreadResults(); +} - if (!isScrapingActive && currentResults.length > 0) { - elements.spreadButton.classList.remove('hidden'); +/** Fetch the worker's state and render it. */ +async function refresh() { + const state = await sendToWorker({ type: Msg.T.GET_STATE }); + if (state && !state.error) render(state); + else updateStatus('Could not reach ProScan. Close and reopen the popup.', 'error'); + return state; +} - // Check if spread results already exist - const spreadData = await new Promise(resolve => { - chrome.storage.local.get(['spreadResults'], resolve); - }); - if (spreadData.spreadResults && Object.keys(spreadData.spreadResults).length > 0) { - displaySpreadResults(); +/** Warn when the extension's storage is nearly full. */ +async function warnIfNearlyFull() { + if (isScrapingActive) return; + try { + const { usage, quota } = await navigator.storage.estimate(); + if (quota && usage >= quota * Storage.NEAR_FULL) { + const pct = Math.round((usage / quota) * 100); + updateStatus(`Browser storage is ${pct}% full. Download your results before the next run.`, 'warning'); } - } + } catch (e) { /* no estimate in this context */ } +} + +/** + * Initialize the popup UI from the worker's state. + * + * @async + */ +async function initializeUI() { + await refresh(); + await warnIfNearlyFull(); } /** The tab the popup was opened over, or null. */ @@ -280,10 +293,9 @@ const AMAZON_URL = /^https:\/\/([a-z0-9-]+\.)*amazon\.com\//i; /** * Start a new scraping session. * - * Pings the tab first and changes nothing if the content script is not + * The worker pings the tab and changes nothing if the content script is not * there (an open tab keeps the old script after an update) or the page is - * not a search. Only then are the old results cleared and a run bound to - * this tab. + * not a search. Old runs are kept; the view moves to the new one. * * @async */ @@ -295,40 +307,16 @@ async function startScraping() { return; } - const pong = await sendToTab(tab.id, { type: 'PING' }); - if (!pong || !pong.ok) { - offerReload(tab.id); + const resp = await sendToWorker({ type: Msg.T.START_RUN, tabId: tab.id }); + if (resp && resp.ok) { + await refresh(); return; } - if (!Run.STARTABLE.includes(pong.kind)) { - updateStatus(Run.refusal(pong.kind), 'warning'); + if (resp && resp.error === 'no_receiver') { + offerReload(tab.id); return; } - - const settings = (await Storage.get(Storage.KEYS.SETTINGS)) || {}; - await Storage.resetForNewScrape(); - // Mint a run id (with storefront/keyword source metadata from the tab - // URL) before the content script begins; it persists across pagination. - const runId = await Storage.beginRun(tab.url); - const run = Run.create({ runId, tabId: tab.id, maxPages: settings.maxPages }); - await Storage.set(Run.KEY, run); - - // Reset UI - elements.itemCount.textContent = '0'; - elements.avgRating.textContent = '-'; - elements.avgPrice.textContent = '-'; - elements.insightsPreview.classList.add('hidden'); - currentResults = []; - scrapedItemCount = 0; - updateStatus('Scraping in progress...', 'info'); - setScrapingState(true); - - const ack = await sendToTab(tab.id, { type: 'START_SCRAPING', runId, tabId: tab.id }); - if (!ack || ack.status !== 'started') { - await Storage.setMultiple({ [Run.KEY]: Run.finish(run, 'interrupted'), [Storage.KEYS.IS_SCRAPING]: false }); - updateStatus('The tab stopped responding. Reload it and try again.', 'error'); - setScrapingState(false); - } + updateStatus((resp && (resp.message || resp.error)) || 'Could not start the run.', 'warning'); } /** @@ -355,7 +343,7 @@ function offerReload(tabId) { /** Starts once the reloaded tab answers a ping, trying a few times. */ async function startWhenReady(tabId, tries) { - const pong = await sendToTab(tabId, { type: 'PING' }); + const pong = await sendToTab(tabId, { type: Msg.T.PING }); if (pong && pong.ok) return startScraping(); if (tries <= 1) { updateStatus('The tab still does not answer. Close it and open the search again.', 'error'); @@ -365,40 +353,14 @@ async function startWhenReady(tabId, tries) { } /** - * Stop the current run. STOP goes to the run's own tab, which cancels its - * pending page and records the run as stopped. If the tab is gone, the - * popup records it. + * Stop the current run. The worker cancels the pending page and records + * the run as stopped. * * @async */ async function stopScraping() { - const run = await Storage.get(Run.KEY); - const ack = Run.isActive(run) ? await sendToTab(run.tabId, { type: 'STOP_SCRAPING', runId: run.runId }) : null; - if (!ack || !ack.stopped) { - const latest = await Storage.get(Run.KEY); - const updates = { [Storage.KEYS.IS_SCRAPING]: false }; - if (Run.isActive(latest)) updates[Run.KEY] = Run.finish(latest, 'stopped'); - await Storage.setMultiple(updates); - } - showRunEnd(); -} - -/** - * Show how the last run ended, from the run record, and return the UI to - * the idle state. Safe to call more than once. - */ -async function showRunEnd() { - const run = await Storage.get(Run.KEY); - if (Run.isActive(run)) return; - const results = await Storage.getResults(); - currentResults = results; - updateStats(currentResults); - const line = Run.describe(run, results.length) || { text: 'Scraping stopped.', type: 'warning' }; - updateStatus(line.text, line.type); - setScrapingState(false); - if (results.length > 0) { - elements.spreadButton.classList.remove('hidden'); - } + await sendToWorker({ type: Msg.T.STOP_RUN }); + await refresh(); } // --- Spread Analysis --- @@ -424,7 +386,7 @@ async function startSpreadAnalysis() { updateStatus('Analyzing price spreads...', 'info'); chrome.tabs.query({ active: true, currentWindow: true }, tabs => { - chrome.tabs.sendMessage(tabs[0].id, { type: 'START_SPREAD_ANALYSIS' }); + chrome.tabs.sendMessage(tabs[0].id, { type: Msg.T.START_SPREAD_ANALYSIS }); }); } @@ -438,20 +400,16 @@ function stopSpreadAnalysis() { updateStatus('Spread analysis stopped.', 'warning'); chrome.tabs.query({ active: true, currentWindow: true }, tabs => { - chrome.tabs.sendMessage(tabs[0].id, { type: 'STOP_SPREAD_ANALYSIS' }); + chrome.tabs.sendMessage(tabs[0].id, { type: Msg.T.STOP_SPREAD_ANALYSIS }); }); } /** - * Display the spread analysis results in the popup. - * Reads spread data from storage and runs it through SpreadAnalyzer. + * Display the spread analysis results in the popup, from the worker's + * spread data for the latest run. */ -async function displaySpreadResults() { - const data = await new Promise(resolve => { - chrome.storage.local.get(['spreadResults'], resolve); - }); - - const spreadResults = data.spreadResults || {}; +function displaySpreadResults() { + const spreadResults = currentSpread || {}; const analyzed = SpreadAnalyzer.analyzeAll(spreadResults); const summary = SpreadAnalyzer.generateSummary(analyzed); @@ -554,7 +512,7 @@ async function initializeAuthUI() { // Sign-in only serves the cloud export, which is off in this build. if (!Flags.CLOUD_SYNC) return; document.getElementById('authPanel').classList.remove('hidden'); - const response = await sendToWorker({ type: 'PROSCAN_AUTH_STATE' }); + const response = await sendToWorker({ type: Msg.T.PROSCAN_AUTH_STATE }); if (response && response.user) { showSignedIn(response.user); } else { @@ -582,7 +540,7 @@ async function handleSignIn() { elements.authSignInBtn.disabled = true; elements.authSignInBtn.innerHTML = ' Signing in…'; - const response = await sendToWorker({ type: 'PROSCAN_SIGN_IN', email, password }); + const response = await sendToWorker({ type: Msg.T.PROSCAN_SIGN_IN, email, password }); if (response && response.user) { elements.authPassword.value = ''; @@ -600,7 +558,7 @@ async function handleSignIn() { */ async function handleSignOut() { elements.authSignOutBtn.disabled = true; - const response = await sendToWorker({ type: 'PROSCAN_SIGN_OUT' }); + const response = await sendToWorker({ type: Msg.T.PROSCAN_SIGN_OUT }); elements.authSignOutBtn.disabled = false; if (response && response.ok) { @@ -629,7 +587,7 @@ async function handleExportToProScan() { elements.exportToProScanBtn.innerHTML = ' Exporting…'; updateStatus('Exporting to ProScan…', 'info'); - const response = await sendToWorker({ type: 'PROSCAN_EXPORT' }); + const response = await sendToWorker({ type: Msg.T.PROSCAN_EXPORT }); if (response && response.ok) { const count = response.products || 0; @@ -681,12 +639,8 @@ elements.actionButton.addEventListener('click', async () => { await stopScraping(); } } catch (err) { - if (err && err.name !== 'StorageError') throw err; - console.error('[ProScan] Storage write failed:', err.message); - updateStatus(err.code === 'storage_full' - ? 'Browser storage is full, so the run could not be saved. Download your results first.' - : 'Could not save to browser storage: ' + err.message, 'error'); - setScrapingState(false); + console.error('[ProScan] Start or stop failed:', err && err.message); + updateStatus('Something went wrong. Close and reopen the popup, then try again.', 'error'); } finally { elements.actionButton.disabled = false; } @@ -776,47 +730,36 @@ elements.downloadJSON.addEventListener('click', async () => { }); /** - * Listen for real-time messages from the content script. - * - * Message types: - * - UPDATE_PROGRESS: Incremental results from each page - * - SCRAPING_COMPLETE: Final notification when all pages are done - * - PAGE_COMPLETE: Per-page item count update + * Messages from the offer fetcher in the tab: + * - SPREAD_PROGRESS: one more product analyzed + * - SPREAD_ANALYSIS_COMPLETE: the analysis is done */ -chrome.runtime.onMessage.addListener((request, sender, sendResponse) => { - if (request.type === 'UPDATE_PROGRESS') { - scrapedItemCount += request.itemCount; - if (request.results) { - currentResults = [...currentResults, ...request.results]; - updateStats(currentResults); - } else { - elements.itemCount.textContent = scrapedItemCount; - } - updateStatus('Scraping in progress... ' + scrapedItemCount + ' items', 'info'); - } else if (request.type === 'SCRAPING_COMPLETE') { - showRunEnd(); - } else if (request.type === 'PAGE_COMPLETE') { - scrapedItemCount += request.itemsScraped; - elements.itemCount.textContent = scrapedItemCount; - } else if (request.type === 'SPREAD_PROGRESS') { - // Update spread analysis progress bar +chrome.runtime.onMessage.addListener((request) => { + if (!request) return false; + if (request.type === Msg.T.SPREAD_PROGRESS) { const percent = Math.round((request.current / request.total) * 100); elements.spreadProgressFill.style.width = percent + '%'; elements.spreadProgressText.textContent = `${request.current}/${request.total}`; - } else if (request.type === 'SPREAD_ANALYSIS_COMPLETE') { + } else if (request.type === Msg.T.SPREAD_ANALYSIS_COMPLETE) { isSpreadAnalyzing = false; elements.spreadButton.innerHTML = ' Analyze Price Spreads'; elements.spreadButton.classList.remove('stop'); - displaySpreadResults(); + refresh(); } + return false; }); -// A run can end while the popup is open without a message reaching it -// (the tab closed, another popup stopped it), so follow the record too. +// The worker writes the run record to session storage on every change, so +// the view follows it: a new page, the end of the run, a closed tab. chrome.storage.onChanged.addListener((changes, area) => { - if (area !== 'local' || !changes[Run.KEY]) return; + if (area !== 'session' || !changes[Run.KEY]) return; const run = changes[Run.KEY].newValue; - if (isScrapingActive && run && !Run.isActive(run)) showRunEnd(); + if (!run) return; + if (!Run.isActive(run) || run.page !== renderedPage || !isScrapingActive) { + refresh(); + } else { + elements.itemCount.textContent = run.itemCount || 0; + } }); // Initialize on DOM load From a2c26a3986d3cf027d40c926f27086480751be33 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:32:42 -0400 Subject: [PATCH 10/25] Ship messages.js to pages, drop unused content globals, bump to 2.2.0 --- manifest.json | 6 ++---- tools/build.mjs | 1 + 2 files changed, 3 insertions(+), 4 deletions(-) diff --git a/manifest.json b/manifest.json index b194f18..fd66f52 100644 --- a/manifest.json +++ b/manifest.json @@ -1,7 +1,7 @@ { "manifest_version": 3, "name": "ProScan - Amazon Product Scraper", - "version": "2.1.0", + "version": "2.2.0", "description": "Scrape Amazon seller products with analytics and AI-powered product insights", "permissions": [ "storage", @@ -30,9 +30,7 @@ "js": [ "scripts/modules/price.js", "scripts/lib/parsers.js", - "scripts/lib/run.js", - "scripts/lib/flags.js", - "scripts/modules/delta.js", + "scripts/lib/messages.js", "scripts/content/scraper.js", "scripts/content/chatbot.js", "scripts/content/offer-fetcher.js" diff --git a/tools/build.mjs b/tools/build.mjs index 2b4cfb9..49c59dc 100644 --- a/tools/build.mjs +++ b/tools/build.mjs @@ -30,6 +30,7 @@ export const COPY_FILES = [ 'scripts/content/chatbot.js', 'scripts/content/offer-fetcher.js', 'scripts/lib/parsers.js', + 'scripts/lib/messages.js', 'scripts/lib/run.js', 'scripts/lib/flags.js', 'scripts/modules/price.js', From 0cc1f7e8006b9742aab4a5554d9926174a953c57 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:43:42 -0400 Subject: [PATCH 11/25] Tell extension pages from content scripts by URL, since a popup can open in a tab --- scripts/lib/messages.js | 18 +++++++++++------- tests/unit/router.test.js | 10 ++++++++++ 2 files changed, 21 insertions(+), 7 deletions(-) diff --git a/scripts/lib/messages.js b/scripts/lib/messages.js index 260b377..90d08be 100644 --- a/scripts/lib/messages.js +++ b/scripts/lib/messages.js @@ -60,16 +60,20 @@ const Msg = (() => { } /** - * Whether `sender` may send a worker message with this spec. A content - * script sender always has a tab; an extension page never does. + * Whether `sender` may send a worker message with this spec. An + * extension page has the extension's own URL, even when it is open in a + * tab; a content script is in a tab with a web page URL. */ function senderAllowed(entry, sender, extensionId) { if (!entry || entry.to !== 'worker') return false; - const fromTab = !!(sender && sender.tab && typeof sender.tab.id === 'number'); - const ours = !sender || !sender.id || !extensionId || sender.id === extensionId; - if (!ours) return false; - if (entry.from === 'tab') return fromTab; - if (entry.from === 'page') return !fromTab; + const s = sender || {}; + if (s.id && extensionId && s.id !== extensionId) return false; + const url = typeof s.url === 'string' ? s.url : ''; + const page = extensionId ? url.startsWith(`chrome-extension://${extensionId}/`) : url.startsWith('chrome-extension://'); + const inTab = !!(s.tab && typeof s.tab.id === 'number'); + const content = inTab && !page; + if (entry.from === 'tab') return content; + if (entry.from === 'page') return page || (!inTab && !url); return true; } diff --git a/tests/unit/router.test.js b/tests/unit/router.test.js index 762ac32..9a2fa7b 100644 --- a/tests/unit/router.test.js +++ b/tests/unit/router.test.js @@ -71,3 +71,13 @@ test('only worker messages can be registered, once each', () => { r.on('GET_STATE', () => {}); expect(() => r.on('GET_STATE', () => {})).toThrow(/already/); }); + +test('the popup open in a tab is still a page, not a content script', async () => { + const r = createRouter({ extensionId: 'ext', log: quiet }); + r.on('START_RUN', () => ({ ok: true })); + r.on('PAGE_READY', () => ({ idle: true })); + const popupTab = { id: 'ext', tab: { id: 9 }, url: 'chrome-extension://ext/popup/popup.html' }; + expect(await call(r, { type: 'START_RUN' }, popupTab)).toEqual({ ok: true }); + expect(await call(r, { type: 'PAGE_READY' }, popupTab)).toBe('no answer'); + expect(await call(r, { type: 'PAGE_READY' }, { id: 'ext', tab: { id: 9 }, url: 'https://www.amazon.com/s?k=a' })).toEqual({ idle: true }); +}); From 01b58b306f5c4eebe45c36994fd64782bebff182 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:47:12 -0400 Subject: [PATCH 12/25] Open the database again when another context closes it, and end the run if it cannot --- scripts/background/db.js | 15 ++++++++++----- scripts/background/engine.js | 21 ++++++++++++++++----- tests/unit/db.test.js | 11 +++++++++++ tests/unit/engine.test.js | 32 ++++++++++++++++++++++++++++++++ 4 files changed, 69 insertions(+), 10 deletions(-) diff --git a/scripts/background/db.js b/scripts/background/db.js index 07de098..d3a7846 100644 --- a/scripts/background/db.js +++ b/scripts/background/db.js @@ -55,9 +55,12 @@ function open({ indexedDB = globalThis.indexedDB, name = NAME } = {}) { req.onblocked = () => reject(storageError(new Error('database upgrade blocked'))); req.onsuccess = () => { const idb = req.result; - // Another context upgrading the schema should not hang on us. - idb.onversionchange = () => idb.close(); - resolve(api(idb)); + const db = api(idb); + // Another context upgrading or deleting the database should not + // hang on us; the caller opens it again. + idb.onversionchange = () => { idb.close(); db.closed = true; }; + idb.onclose = () => { db.closed = true; }; + resolve(db); }; }); } @@ -197,10 +200,12 @@ function api(idb) { }); } - return { + const db = { idb, run, get, getAll, getMany, count, write, getMeta, runProducts, runPages, pruneLastValues, - close: () => idb.close() + closed: false, + close() { idb.close(); db.closed = true; } }; + return db; } module.exports = { open, NAME, VERSION, storageError }; diff --git a/scripts/background/engine.js b/scripts/background/engine.js index 46e5444..2803c08 100644 --- a/scripts/background/engine.js +++ b/scripts/background/engine.js @@ -97,10 +97,20 @@ function createEngine({ let timer = null; let chain = Promise.resolve(); - const db = () => { - if (!dbPromise) dbPromise = openDb().catch((err) => { dbPromise = null; throw err; }); - return dbPromise; - }; + /** The open database, opened again if it was closed under us. */ + async function db() { + if (dbPromise) { + const open = await dbPromise.catch(() => null); + if (open && !open.closed) return open; + } + dbPromise = openDb(); + try { + return await dbPromise; + } catch (err) { + dbPromise = null; + throw err; + } + } /** Runs `fn` after every change queued before it. */ function serial(fn) { @@ -240,11 +250,12 @@ function createEngine({ return { ok: true, next: 'end', reason: ended.reason }; } - const store = await db(); const t = now(); + let store; let ops; let next; try { + store = await db(); const existing = await store.runProducts(runId); const { fresh, changed } = foldPage(existing, result.products, page); const prevs = await store.getMany('lastValues', fresh.map(p => p.asin)); diff --git a/tests/unit/db.test.js b/tests/unit/db.test.js index 9983f1a..d3a2747 100644 --- a/tests/unit/db.test.js +++ b/tests/unit/db.test.js @@ -80,3 +80,14 @@ test('a quota error is reported as storage_full', () => { expect(DB.storageError(quota).code).toBe('storage_full'); expect(DB.storageError(new Error('other')).code).toBe('storage_error'); }); + +test('a database closed by another context says so', async () => { + const factory = new IDBFactory(); + const mine = await DB.open({ indexedDB: factory }); + expect(mine.closed).toBe(false); + await new Promise((resolve) => { + const req = factory.deleteDatabase('proscan'); + req.onsuccess = resolve; + }); + expect(mine.closed).toBe(true); +}); diff --git a/tests/unit/engine.test.js b/tests/unit/engine.test.js index 4bc0cae..94d19ac 100644 --- a/tests/unit/engine.test.js +++ b/tests/unit/engine.test.js @@ -346,6 +346,38 @@ describe('page results', () => { }); }); +describe('the database', () => { + test('closed under the worker between pages, it is opened again and the run goes on', async () => { + let opened = 0; + let first = null; + const rig = createRig({ site: simpleSite(3), wrapDb: (db) => { opened++; if (!first) first = db; return db; } }); + await startIn(rig, 'reopen'); + await settle(); + first.closed = true; + const run = await runUntilEnd(rig); + expect(run).toMatchObject({ reason: 'complete', page: 3 }); + expect(opened).toBe(2); + }); + + test('that cannot be opened again, it ends the run loudly as storage_error', async () => { + let opened = 0; + let first = null; + const wrapDb = (db) => { + if (++opened > 1) throw Object.assign(new Error('VersionError'), { name: 'StorageError', code: 'storage_error' }); + first = db; + return db; + }; + const rig = createRig({ site: simpleSite(3), wrapDb }); + await startIn(rig, 'gone'); + await settle(); + first.closed = true; + const run = await runUntilEnd(rig); + await settle(5000); + expect(run).toMatchObject({ state: 'failed', reason: 'storage_error', page: 1 }); + expect(served(rig, 'gone')).toEqual([1, 2]); + }); +}); + describe('products across pages (F-27, F-28)', () => { const PAGE1 = page(card('B0A', '$5.00', { ad: true }) + card('B0B', '$2.00') + card('B0A', '$5.00'), '/s?k=w&page=2'); const PAGE2 = page(card('B0C', '$3.00') + card('B0A', '$5.00') + card('B0B', '$2.00', { rating: false }), null); From 14a25282d78268fb7ab29fa1a90e39b05c92418b Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:48:17 -0400 Subject: [PATCH 13/25] Show how a run ended even when its database cannot be opened --- popup/popup.js | 2 +- scripts/background/engine.js | 8 +++++++- tests/unit/engine.test.js | 2 ++ 3 files changed, 10 insertions(+), 2 deletions(-) diff --git a/popup/popup.js b/popup/popup.js index 0cd0fd1..0927f99 100644 --- a/popup/popup.js +++ b/popup/popup.js @@ -240,7 +240,7 @@ function render(state) { /** Fetch the worker's state and render it. */ async function refresh() { const state = await sendToWorker({ type: Msg.T.GET_STATE }); - if (state && !state.error) render(state); + if (state && (state.run || !state.error)) render(state); else updateStatus('Could not reach ProScan. Close and reopen the popup.', 'error'); return state; } diff --git a/scripts/background/engine.js b/scripts/background/engine.js index 2803c08..5c0b52c 100644 --- a/scripts/background/engine.js +++ b/scripts/background/engine.js @@ -384,7 +384,13 @@ function createEngine({ /** The latest run and what it found, for the popup and the chat. */ async function latest() { const live = await liveRun(); - const store = await db(); + let store; + try { + store = await db(); + } catch (err) { + // The run record is in session storage, so the popup can still say how it ended. + return { run: live, results: [], pages: [], spread: {}, error: err.code || 'storage_error' }; + } const runId = (live && live.runId) || await store.getMeta('latestRunId'); if (!runId) return { run: null, results: [], pages: [], spread: {} }; const [rec, results, pages, spreadRows] = await Promise.all([ diff --git a/tests/unit/engine.test.js b/tests/unit/engine.test.js index 94d19ac..ddec08b 100644 --- a/tests/unit/engine.test.js +++ b/tests/unit/engine.test.js @@ -375,6 +375,8 @@ describe('the database', () => { await settle(5000); expect(run).toMatchObject({ state: 'failed', reason: 'storage_error', page: 1 }); expect(served(rig, 'gone')).toEqual([1, 2]); + const st = await rig.popup({ type: 'GET_STATE' }); + expect(st).toMatchObject({ run: { reason: 'storage_error' }, results: [], error: 'storage_error' }); }); }); From 51329a7b90958adaa530fff850d910474db2ef66 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:48:17 -0400 Subject: [PATCH 14/25] Read run state from session storage and IndexedDB in the Chromium harness --- tests/e2e/lib/extension.mjs | 79 ++++++++++++++++++++++++++++++++++++- tests/e2e/scrape.spec.mjs | 76 +++++++++++++++++------------------ 2 files changed, 115 insertions(+), 40 deletions(-) diff --git a/tests/e2e/lib/extension.mjs b/tests/e2e/lib/extension.mjs index 5e91d39..051c172 100644 --- a/tests/e2e/lib/extension.mjs +++ b/tests/e2e/lib/extension.mjs @@ -45,8 +45,83 @@ export async function extPage(ext) { return page; } +/** + * Everything the extension has stored, read from an extension page without + * waking the service worker. From 2.2 the run is in chrome.storage.session + * and its data in IndexedDB; this folds them into the keys older builds kept + * in chrome.storage.local, so one scenario reads the same on any build: + * results and scrapeRunPages are the latest run's, run is its record, + * lastValues is keyed by ASIN and outbox lists what waits for sync. + */ export async function getState(page) { - return page.evaluate(() => chrome.storage.local.get(null)); + return page.evaluate(async () => { + const state = await chrome.storage.local.get(null); + const session = chrome.storage.session ? await chrome.storage.session.get(null) : {}; + // Opening a database that does not exist would create an empty one. + const names = indexedDB.databases ? (await indexedDB.databases()).map((d) => d.name) : []; + if (!names.includes('proscan')) return { ...state, ...session }; + const db = await new Promise((resolve, reject) => { + const req = indexedDB.open('proscan'); + req.onsuccess = () => resolve(req.result); + req.onerror = () => reject(req.error); + }); + const all = (store, index, key) => new Promise((resolve, reject) => { + if (!db.objectStoreNames.contains(store)) return resolve([]); + const s = db.transaction(store).objectStore(store); + const req = index ? s.index(index).getAll(key) : s.getAll(); + req.onsuccess = () => resolve(req.result); + req.onerror = () => reject(req.error); + }); + try { + const meta = await all('meta'); + const latest = (meta.find((m) => m.key === 'latestRunId') || {}).value; + const runs = await all('runs'); + const liveRun = session.run || null; + const runId = (liveRun && liveRun.runId) || latest; + const run = liveRun && liveRun.runId === runId ? liveRun : runs.find((r) => r.runId === runId); + const results = runId ? (await all('products', 'runId', runId)).sort((a, b) => a.n - b.n) : []; + const pages = runId ? (await all('placements', 'runId', runId)).sort((a, b) => a.pageIndex - b.pageIndex) : []; + const lastValues = {}; + (await all('lastValues')).forEach(({ asin, ...snap }) => { lastValues[asin] = snap; }); + const out = { + ...state, + results, + scrapeRunPages: pages, + lastValues, + outbox: await all('outbox'), + scrapeRuns: Object.fromEntries(runs.map((r) => [r.runId, r.source])), + isScrapingActive: !!run && ['starting', 'running', 'stopping'].includes(run.state), + }; + if (run) out.run = run; + return out; + } finally { + db.close(); + } + }); +} + +/** True once the run in `state` has ended. */ +export const ended = (state) => !!(state.run && (state.run.reason || (state.run.status && state.run.status !== 'running'))); + +/** + * Seeds a finished run with `rows` as the latest run, the way a past scrape + * leaves it. Asks the worker for its state first, so the database exists. + */ +export async function seedRun(page, rows, runId = 'seeded') { + await page.evaluate(() => new Promise((r) => chrome.runtime.sendMessage({ type: 'GET_STATE' }, r))); + await page.evaluate(({ rows, runId }) => new Promise((resolve, reject) => { + const req = indexedDB.open('proscan'); + req.onsuccess = () => { + const db = req.result; + const tx = db.transaction(['runs', 'products', 'meta'], 'readwrite'); + tx.objectStore('runs').put({ runId, state: 'done', reason: 'complete', itemCount: rows.length, source: null }); + rows.forEach((row, n) => tx.objectStore('products').put({ ...row, runId, n })); + tx.objectStore('meta').put({ key: 'latestRunId', value: runId }); + tx.oncomplete = () => { db.close(); resolve(); }; + tx.onerror = () => reject(tx.error); + }; + req.onerror = () => reject(req.error); + }), { rows, runId }); } /** @@ -90,6 +165,8 @@ export async function waitForState(page, fn, { timeout = 20000, interval = 250 } */ export function endReason(state) { if (state.scrapeEndReason) return state.scrapeEndReason; + if (state.run && state.run.reason) return state.run.reason; + if (state.run && state.run.state) return state.run.state; if (state.run && state.run.status) return state.run.status; return state.isScrapingActive ? 'running' : 'complete'; } diff --git a/tests/e2e/scrape.spec.mjs b/tests/e2e/scrape.spec.mjs index 5ffefce..c7c1cc7 100644 --- a/tests/e2e/scrape.spec.mjs +++ b/tests/e2e/scrape.spec.mjs @@ -9,7 +9,7 @@ import { createRequire } from 'node:module'; import { test as base, expect } from '@playwright/test'; import { launch, extPage, getState, clickStart, aimPopupAt, waitForState, endReason, - runMetaFor, killServiceWorker, enableDeveloperMode, sleep, + runMetaFor, killServiceWorker, enableDeveloperMode, sleep, ended, seedRun, workerTargets, } from './lib/extension.mjs'; import { serveAmazon, simplePlan, searchPage, card, asinFor, corpusPage, CAPTCHA } from './lib/amazon.mjs'; @@ -192,30 +192,29 @@ test('a slow page is scraped once and the run still completes', async ({ ext }) expect(endReason(s)).toBe('complete'); }); -test('a full storage quota fails the run loudly', async ({ ext }) => { - const served = await serveAmazon(ext.context, simplePlan(2)); - const store = await extPage(ext); - // Fill storage with queued products until about 2 KB is left. - const left = await store.evaluate(async () => { - const one = { name: 'x'.repeat(100), asin: 'B0FILL0000', price: '$1.00', priceCents: 100, url: 'https://www.amazon.com/dp/B0FILL0000', scrapedAt: new Date().toISOString(), runId: 'old' }; - const quota = chrome.storage.local.QUOTA_BYTES; - const per = JSON.stringify(one).length + 1; - let n = Math.floor((quota - 4096) / per); - for (;;) { - try { await chrome.storage.local.set({ syncQueue: Array.from({ length: n }, () => one) }); break; } catch { n -= 50; } - } - return quota - await chrome.storage.local.getBytesInUse(null); - }); - expect(left).toBeLessThan(4096); - +// Run data lives in IndexedDB now. Chrome gives an extension origin most of +// the disk and ignores a CDP quota override for it (checked: the override +// reports active, and a 200 KB write under a 1 KB quota still commits), so +// a real quota error cannot be forced here. engine.test.js covers the +// storage_full path; this breaks the database for real instead. +test('a page that cannot be saved fails the run loudly (F-26)', async ({ ext }) => { + const served = await serveAmazon(ext.context, simplePlan(4)); const tab = await openSearch(ext, 'full disk'); - await clickStart(ext, tab); - const s = await waitForState(store, (x) => x.run && x.run.status !== 'running', { timeout: 30000 }); - await sleep(2500); - - expect(pagesOf(served, 'full disk').length).toBeGreaterThanOrEqual(1); + const store = await extPage(ext); + const popup = await clickStart(ext, tab); + await waitForState(store, (x) => x.scrapeRunPages?.length >= 1, { timeout: 15000, interval: 100 }); + // A newer schema from elsewhere: the worker's connection closes and it can no longer open version 1. + await store.evaluate(() => new Promise((resolve, reject) => { + const req = indexedDB.open('proscan', 99); + req.onsuccess = () => { req.result.close(); resolve(); }; + req.onerror = () => reject(req.error); + })); + await sleep(6000); + const run = await popup.evaluate(async () => (await chrome.storage.session.get('run')).run); - expect(endReason(s)).toBe('storage_full'); + expect(pagesOf(served, 'full disk')).toEqual([1, 2]); + expect(run).toMatchObject({ state: 'failed', reason: 'storage_error', page: 1 }); + await expect(popup.locator('#status')).toContainText(/could not save a page/); }); const Flags = require('../../scripts/lib/flags.js'); @@ -236,7 +235,7 @@ async function twoRuns(ext, store, done = () => true) { (x.results || []).some((r) => r.name.startsWith('yoga mat')) && done(x), { timeout: 30000 }); } -test('with cloud sync off (2.1), two runs queue nothing and keep their deltas (F-26)', async ({ ext }) => { +test('with cloud sync off, two runs queue nothing and keep their deltas (F-26)', async ({ ext }) => { test.skip(Flags.CLOUD_SYNC, 'cloud sync is on in this build'); await serveAmazon(ext.context, simplePlan(2)); const store = await extPage(ext); @@ -244,10 +243,11 @@ test('with cloud sync off (2.1), two runs queue nothing and keep their deltas (F expect(s.results.map((r) => r.asin)).toEqual(allAsins('yoga mat', [1, 2])); expect(s).not.toHaveProperty('syncQueue'); + expect(s.outbox).toEqual([]); expect(Object.keys(s.lastValues).sort()).toEqual([...allAsins('garden hose', [1, 2]), ...allAsins('yoga mat', [1, 2])].sort()); }); -test('with cloud sync off (2.1), the popup shows no sign-in and no Export to ProScan', async ({ ext }) => { +test('with cloud sync off, the popup shows no sign-in and no Export to ProScan', async ({ ext }) => { test.skip(Flags.CLOUD_SYNC, 'cloud sync is on in this build'); const popup = await extPage(ext); await sleep(500); @@ -259,18 +259,15 @@ test('with cloud sync off (2.1), the popup shows no sign-in and no Export to Pro }); test('two runs without an export keep their own attribution', async ({ ext }) => { - test.skip(!Flags.CLOUD_SYNC, 'F-20: cloud sync is off in 2.1, so nothing is queued to attribute'); + test.skip(!Flags.CLOUD_SYNC, 'F-20: cloud sync is off until 2.3, so nothing is queued to attribute'); await serveAmazon(ext.context, simplePlan(2)); const store = await extPage(ext); - const s = await twoRuns(ext, store, (x) => (x.syncQueue || []).length >= 16); + const s = await twoRuns(ext, store, (x) => (x.outbox || []).length >= 4); - const runIds = [...new Set(s.syncQueue.map((p) => p.runId))]; + const runIds = [...new Set(s.outbox.map((p) => p.runId))]; expect(runIds).toHaveLength(2); - - bug('F-20', 'only the newest run keeps its source; older queued items lose theirs'); - for (const item of s.syncQueue) { - const meta = runMetaFor(s, item.runId); - expect(meta && meta.keyword).toBe(item.name.startsWith('garden hose') ? 'garden hose' : 'yoga mat'); + for (const runId of runIds) { + expect(runMetaFor(s, runId)).toBeTruthy(); } }); @@ -281,7 +278,8 @@ test('a service worker stopped between pages does not break the run', async ({ e await clickStart(ext, tab); await waitForState(store, (x) => x.scrapeRunPages?.length >= 1, { timeout: 15000, interval: 100 }); expect(await killServiceWorker(ext, tab)).toBe(true); - + // The pending page lived in the dead worker's timer. Nothing here wakes + // it: the state reads go straight to storage, so the tab's heartbeat must. const s = await waitForState(store, (x) => x.scrapeRunPages?.length >= 3 && !x.isScrapingActive, { timeout: 30000 }); expect(pagesOf(served, 'sleepy')).toEqual([1, 2, 3]); expect(s.results.map((r) => r.asin)).toEqual(allAsins('sleepy', [1, 2, 3])); @@ -294,7 +292,7 @@ test('the page cap in settings ends the run as complete', async ({ ext }) => { await store.evaluate(() => chrome.storage.local.set({ settings: { pageDelay: 2000, maxPages: 2 } })); const tab = await openSearch(ext, 'capped'); await clickStart(ext, tab); - const s = await waitForState(store, (x) => x.run && x.run.status !== 'running', { timeout: 30000 }); + const s = await waitForState(store, ended, { timeout: 30000 }); await sleep(4500); expect(pagesOf(served, 'capped')).toEqual([1, 2]); @@ -305,14 +303,14 @@ test('the page cap in settings ends the run as complete', async ({ ext }) => { test('Start on a captcha page changes nothing and says why', async ({ ext }) => { await serveAmazon(ext.context, () => ({ body: CAPTCHA() })); const store = await extPage(ext); - await store.evaluate(() => chrome.storage.local.set({ results: [{ asin: 'B0KEEP0001', name: 'kept' }] })); + await seedRun(store, [{ asin: 'B0KEEP0001', name: 'kept' }]); const tab = await openSearch(ext, 'robot'); const popup = await clickStart(ext, tab); await sleep(1000); const s = await getState(store); expect(s.results.map((r) => r.asin)).toEqual(['B0KEEP0001']); - expect(s.run).toBeUndefined(); + expect(s.run.runId).toBe('seeded'); expect(s.isScrapingActive).toBeFalsy(); expect(await popup.textContent('#status')).toMatch(/captcha/); }); @@ -321,7 +319,7 @@ test('Start on a tab left over from an update offers a reload, then runs', async const served = await serveAmazon(ext.context, simplePlan(2)); const tab = await openSearch(ext, 'orphan'); const before = await extPage(ext); - await before.evaluate(() => chrome.storage.local.set({ results: [{ asin: 'B0KEEP0001', name: 'kept' }] })); + await seedRun(before, [{ asin: 'B0KEEP0001', name: 'kept' }]); // Reloading the extension orphans the content script already in the tab, // as a Chrome Web Store update does. @@ -338,7 +336,7 @@ test('Start on a tab left over from an update offers a reload, then runs', async expect(await popup.isVisible('#reloadTabButton')).toBe(true); await popup.click('#reloadTabButton'); - s = await waitForState(store, (x) => x.run && x.run.status !== 'running', { timeout: 30000 }); + s = await waitForState(store, (x) => ended(x) && x.run.runId !== 'seeded', { timeout: 30000 }); expect(pagesOf(served, 'orphan')).toEqual([1, 1, 2]); expect(s.results.map((r) => r.asin)).toEqual(allAsins('orphan', [1, 2])); From 2029a6f8bc672c390a83ddd896bfc5676f22ea8a Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:49:49 -0400 Subject: [PATCH 15/25] Upgrade harness: expect 2.2.0 and schema 4, and cut a 2.1 run off mid-scrape --- tests/e2e/upgrade.spec.mjs | 84 ++++++++++++++++++++++++++++++-------- 1 file changed, 68 insertions(+), 16 deletions(-) diff --git a/tests/e2e/upgrade.spec.mjs b/tests/e2e/upgrade.spec.mjs index fb2da3b..8a5ac97 100644 --- a/tests/e2e/upgrade.spec.mjs +++ b/tests/e2e/upgrade.spec.mjs @@ -1,16 +1,23 @@ -// Upgrade in place from the live store build (v2.0, commit 7c2ba1c): load it -// unpacked, scrape, then swap the folder to the current build and reload, as -// a Chrome Web Store auto-update does. Based on the audit's upgrade harness. +// Upgrade in place from an older build: load it unpacked, scrape, then swap +// the folder to the current build and reload, as a Chrome Web Store +// auto-update does. From the live store build (v2.0, commit 7c2ba1c), and +// from 2.1 (commit 40be432) in the middle of a run. Based on the audit's +// upgrade harness. import path from 'node:path'; +import { execFileSync } from 'node:child_process'; import { test, expect } from '@playwright/test'; import { - launch, getState, waitForState, sleep, checkoutRevision, copyDir, enableDeveloperMode, BUILD_DIR, EXT_DIR, + launch, getState, waitForState, sleep, checkoutRevision, copyDir, enableDeveloperMode, clickStart, + BUILD_DIR, EXT_DIR, } from './lib/extension.mjs'; import { serveAmazon, simplePlan, asinFor } from './lib/amazon.mjs'; const LIVE_REV = '7c2ba1c'; +const V21_REV = '40be432'; const V20 = path.join(BUILD_DIR, 'v2.0'); +const V21 = path.join(BUILD_DIR, 'v2.1'); const SLOT = path.join(BUILD_DIR, 'upgrade-slot'); +const CURRENT_VERSION = '2.2.0'; function bug(fid, what) { test.fail(!process.env.PROSCAN_SHOW_KNOWN, `${fid}: ${what}`); @@ -18,8 +25,23 @@ function bug(fid, what) { test.beforeAll(() => { checkoutRevision(LIVE_REV, V20); + // 2.1 bundles its worker, so build it; node_modules resolves from this repo. + checkoutRevision(V21_REV, V21); + execFileSync(process.execPath, [path.join(V21, 'tools/build.mjs')], { cwd: V21, stdio: 'pipe' }); }); +/** Swaps the loaded extension to the current build, as an auto-update does. */ +async function update(ext) { + await enableDeveloperMode(ext); + copyDir(EXT_DIR, SLOT); + await ext.sw.evaluate(() => chrome.runtime.reload()).catch(() => {}); + await sleep(4000); + const after = await ext.context.newPage(); + await after.goto(`chrome-extension://${ext.extId}/popup/popup.html`); + expect(await after.evaluate(() => chrome.runtime.getManifest().version)).toBe(CURRENT_VERSION); + return after; +} + /** * Starts a v2.0 scrape of `keyword` and waits for `pages` pages, then * updates the extension. Returns a page of the updated extension. @@ -41,15 +63,7 @@ async function scrapeThenUpdate(ext, keyword, pages) { expect(started).toEqual({ status: 'started' }); const before = await waitForState(v20, (s) => (s.currentItemCount || 0) >= pages * 4, { timeout: 30000 }); - // The auto-update: new files in the same folder, then a reload. - await enableDeveloperMode(ext); - copyDir(EXT_DIR, SLOT); - await ext.sw.evaluate(() => chrome.runtime.reload()).catch(() => {}); - await sleep(4000); - - const after = await ext.context.newPage(); - await after.goto(`chrome-extension://${ext.extId}/popup/popup.html`); - expect(await after.evaluate(() => chrome.runtime.getManifest().version)).toBe('2.1.0'); + const after = await update(ext); return { before, after, tab }; } @@ -65,8 +79,10 @@ test('a v2.0 user keeps their last scrape through the update', async () => { expect(before.isScrapingActive).toBe(false); expect(s.results.map((r) => r.asin)).toEqual(asins); - expect(s.schemaVersion).toBe(3); + expect(s.schemaVersion).toBe(4); expect(Object.keys(s.lastValues || {}).sort()).toEqual([...asins].sort()); + // Only settings stay in chrome.storage.local; 2.0 left its defaults untouched, so none. + expect(await after.evaluate(() => chrome.storage.local.get(null))).toEqual({ schemaVersion: 4 }); } finally { await ext.close(); } @@ -82,8 +98,8 @@ test('a scrape running during the update is stopped, not resumed without a run', const s = await getState(after); expect(s.results.length).toBeGreaterThanOrEqual(4); - expect(s.schemaVersion).toBe(3); - expect(s.run).toMatchObject({ runId: 'legacy-2.0', status: 'updated' }); + expect(s.schemaVersion).toBe(4); + expect(s.run).toMatchObject({ runId: 'legacy-2.0', state: 'failed', reason: 'updated' }); expect((s.syncQueue || []).filter((p) => !p.runId).map((p) => p.asin)).toEqual([]); expect((s.scrapeRunPages || []).filter((p) => !p.runId)).toEqual([]); @@ -91,3 +107,39 @@ test('a scrape running during the update is stopped, not resumed without a run', await ext.close(); } }); + +test('a 2.1 run cut off by the update ends as updated, keeps its pages and pages no further', async () => { + copyDir(path.join(V21, 'dist'), SLOT); + const ext = await launch(SLOT); + try { + const served = await serveAmazon(ext.context, simplePlan(8)); + const tab = await ext.context.newPage(); + await tab.goto('https://www.amazon.com/s?k=wires'); + await sleep(800); + const v21 = await clickStart(ext, tab); + expect(await v21.evaluate(() => chrome.runtime.getManifest().version)).toBe('2.1.0'); + const local = () => v21.evaluate(() => chrome.storage.local.get(null)); + let before = await local(); + for (let i = 0; i < 100 && !((before.scrapeRunPages || []).length >= 2); i++) { + await sleep(200); + before = await local(); + } + expect(before.run).toMatchObject({ status: 'running' }); + + const after = await update(ext); + const servedAtUpdate = served.length; + await sleep(6000); + const s = await getState(after); + + expect(s.schemaVersion).toBe(4); + expect(Object.keys(await after.evaluate(() => chrome.storage.local.get(null))).sort()).toEqual(['schemaVersion', 'settings']); + expect(s.run).toMatchObject({ runId: before.run.runId, state: 'failed', reason: 'updated' }); + expect(s.results.length).toBeGreaterThanOrEqual(before.results.length); + expect(s.results.every((r) => r.runId === before.run.runId)).toBe(true); + expect(s.scrapeRunPages.map((p) => p.pageIndex).slice(0, 2)).toEqual([1, 2]); + // The 2.1 script left in the tab is dead and 2.2 has no run, so nothing pages on. + expect(served.length).toBe(servedAtUpdate); + } finally { + await ext.close(); + } +}); From 0dfb3a95118470465242d3e748c527a08b34d043 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 04:58:40 -0400 Subject: [PATCH 16/25] Keep the specs and tests of checked-out older builds out of the test runs --- jest.config.js | 3 +++ playwright.config.mjs | 2 ++ 2 files changed, 5 insertions(+) diff --git a/jest.config.js b/jest.config.js index 765efae..30665ce 100644 --- a/jest.config.js +++ b/jest.config.js @@ -3,6 +3,9 @@ module.exports = { roots: ['/tests'], setupFiles: ['/tests/setup/chrome-mock.js'], testMatch: ['**/*.test.js'], + // Older builds checked out by the e2e upgrade harness have their own tests. + testPathIgnorePatterns: ['/node_modules/', '/tests/e2e/\\.build/'], + modulePathIgnorePatterns: ['/tests/e2e/\\.build/'], // sync.js and service-worker.js are authored as ESM (esbuild bundles // them). A tiny scoped transform rewrites their import/export to // CommonJS so they can be unit-tested; every other file keeps the diff --git a/playwright.config.mjs b/playwright.config.mjs index 68d68c3..f730810 100644 --- a/playwright.config.mjs +++ b/playwright.config.mjs @@ -4,6 +4,8 @@ import { defineConfig } from '@playwright/test'; export default defineConfig({ testDir: 'tests/e2e', testMatch: '*.spec.mjs', + // Older builds checked out for the upgrade harness bring their own specs. + testIgnore: '**/.build/**', globalSetup: './tests/e2e/global-setup.mjs', workers: 1, fullyParallel: false, From be49263a73817737dac6a229df76e3af6f7a8033 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:02:56 -0400 Subject: [PATCH 17/25] Chromium harness: a run tab taken elsewhere ends the run and stays where it went --- tests/e2e/scrape.spec.mjs | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/tests/e2e/scrape.spec.mjs b/tests/e2e/scrape.spec.mjs index c7c1cc7..6526bd2 100644 --- a/tests/e2e/scrape.spec.mjs +++ b/tests/e2e/scrape.spec.mjs @@ -177,6 +177,26 @@ test('a product page opened mid-run does not end the run', async ({ ext }) => { expect(s.results.map((r) => r.asin)).toEqual(allAsins('gamma', [1, 2, 3])); }); +for (const [where, url] of [['an Amazon product page', 'https://www.amazon.com/dp/B09B8V1LZ3'], ['another site', 'https://example.com/']]) { + test(`the run tab taken to ${where} ends the run and is not pulled back`, async ({ ext }) => { + const served = await serveAmazon(ext.context, simplePlan(4)); + const tab = await openSearch(ext, 'wander'); + const store = await extPage(ext); + await clickStart(ext, tab); + await waitForState(store, (x) => x.scrapeRunPages?.length >= 1, { timeout: 15000, interval: 100 }); + // Nothing leaves the machine: the other site is served here too. + await ext.context.route('https://example.com/**', (r) => r.fulfill({ status: 200, contentType: 'text/html', body: '

elsewhere

' })); + await tab.goto(url); + await sleep(6000); + const s = await getState(store); + + expect(tab.url()).toBe(url); + expect(pagesOf(served, 'wander')).toEqual([1]); + expect(endReason(s)).toBe('interrupted'); + expect(s.results.map((r) => r.asin)).toEqual(allAsins('wander', [1])); + }); +} + test('a slow page is scraped once and the run still completes', async ({ ext }) => { const served = await serveAmazon(ext.context, ({ keyword, page }) => ({ body: searchPage(keyword, page, { last: page >= 3 }), delayMs: page === 2 ? 4000 : 0 })); From 7fde97c254a1fe1704f1e9c389d6ab1023012c14 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:04:12 -0400 Subject: [PATCH 18/25] Document the service worker run engine, IndexedDB storage and schema 4 --- CLAUDE.md | 89 ++++++++++++++++++++++++++++++-------------- README.md | 35 ++++++++++------- scripts/lib/flags.js | 6 +-- scripts/lib/run.js | 4 +- 4 files changed, 88 insertions(+), 46 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index b9fc128..62ede2c 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -4,11 +4,11 @@ Chrome extension (Manifest V3) that scrapes Amazon seller product listings, provides analytics for resellers/arbitrage, and includes a floating AI chatbot on Amazon pages powered by Gemini API. No server required — everything runs client-side. -## Architecture (v2.0) +## Architecture (v2.2) ```text AmazonSellerScraper/ -├── manifest.json # Extension config (v2.0) +├── manifest.json # Extension config (v2.2) ├── popup/ # UI Layer │ ├── popup.html # Popup interface (dashboard + settings) │ ├── popup.css # Popup styling @@ -16,17 +16,21 @@ AmazonSellerScraper/ │ └── ai-key.js # Gemini key field (AI chat settings) ├── scripts/ │ ├── content/ -│ │ ├── scraper.js # DOM scraping on Amazon pages +│ │ ├── scraper.js # Parses a search page, reports it to the SW (parse only) │ │ ├── chatbot.js # Floating AI chatbot widget (Shadow DOM) │ │ └── offer-fetcher.js # Seller price fetching for spread analysis │ ├── lib/ │ │ ├── parsers.js # Pure search/offer parsing (global Parsers) -│ │ ├── run.js # Scrape run record, bound to one tab (global Run) -│ │ ├── flags.js # Build flags (global Flags); CLOUD_SYNC is off in 2.1 +│ │ ├── messages.js # Every message type and who may send it (global Msg) +│ │ ├── run.js # Run state machine, bound to one tab (global Run) +│ │ ├── flags.js # Build flags (global Flags); CLOUD_SYNC is off until 2.3 │ │ ├── migrate.js # Storage schema migrations, run by the SW │ │ └── chat.js # Gemini request builder, run scoping, error text │ ├── background/ -│ │ └── service-worker.js # Message routing + Gemini API calls +│ │ ├── service-worker.js # Wires router, engine, chat, auth and migrations +│ │ ├── router.js # The one typed message router +│ │ ├── engine.js # The run engine; the only writer of run data +│ │ └── db.js # IndexedDB: runs, products, placements, lastValues, outbox, spread │ └── modules/ │ ├── storage.js # Chrome storage wrapper │ ├── analyzer.js # Data analysis & insights @@ -59,7 +63,8 @@ AmazonSellerScraper/ | Module | Purpose | | -------------------- | --------------------------------------------------------- | -| `scraper.js` | DOM scraping with cascading fallback selectors | +| `scraper.js` | Parses a page with `Parsers`, sends `PAGE_RESULT` | +| `engine.js` | Run state machine, page saves, navigation, heartbeat | | `chatbot.js` | Floating AI chatbot widget on Amazon pages (Shadow DOM) | | `offer-fetcher.js` | Fetches seller offer pages for price spread analysis | | `chatbot.css` | Widget styles loaded into Shadow DOM | @@ -67,21 +72,35 @@ AmazonSellerScraper/ | `analyzer.js` | Opportunity scoring, insights, statistics | | `spread-analyzer.js` | Price spread statistics (CV, std dev, arbitrage scoring) | | `exporter.js` | Multi-format export (Excel, CSV, JSON) | -| `service-worker.js` | Message routing, Gemini API calls | +| `service-worker.js` | Router wiring, Gemini API calls, migrations | ## Data Flow ### Scraping -1. User clicks "Start Scraping" in popup -2. `popup.js` pings the tab; with no answer it offers `chrome.tabs.reload` and starts after the reload -3. `popup.js` writes a run record (`scripts/lib/run.js`, key `run`) bound to the tab id, then sends `START_SCRAPING` -4. `scraper.js` classifies the page (results, last, empty, captcha, interstitial, signin, unknown) and extracts products -5. Results and run progress stored in `chrome.storage.local` in one write per page -6. Follows the page's Next link after 2 to 4 seconds, up to `settings.maxPages`; on each load the content script asks the SW for its tab id (`WHO_AM_I`) and acts only if its tab owns the run -7. The run ends with a reason; Stop goes to the run's tab and cancels the pending navigation -8. `analyzer.js` generates insights and opportunity scores -9. User exports via `exporter.js` (Excel/CSV/JSON) +The service worker is the only writer (`scripts/background/engine.js`). + +1. User clicks "Start Scraping" in popup; `popup.js` sends `START_RUN {tabId}` to the SW +2. The SW pings the tab. No answer: the popup offers `chrome.tabs.reload` and starts after the reload. + A page that is not a search (captcha, sign-in, product page) is refused and nothing changes +3. The SW writes the run record (`scripts/lib/run.js`) to `chrome.storage.session` under `run`, + bound to the tab id, and asks the tab to parse (`PARSE_PAGE`) +4. `scraper.js` classifies the page (results, last, empty, captcha, interstitial, signin, unknown), + extracts products and sends `PAGE_RESULT`. It never writes storage and never navigates +5. The SW folds the page into the run and writes products, the page, lastValues and the run + to IndexedDB in one transaction, then prunes lastValues +6. After 2 to 4 seconds the SW opens the page's Next link with `tabs.update`, up to `settings.maxPages`. + The new page says `PAGE_READY`; only the run's tab is asked to parse +7. While a page is pending the tab sends `HEARTBEAT` every 2 seconds. That wakes a stopped SW, + which reads the run from session storage and opens the next page when it is due +8. The run ends with a reason; Stop goes to the SW, which cancels the pending page. Closing the + run's tab or taking it elsewhere ends the run as `interrupted` +9. `analyzer.js` generates insights and opportunity scores +10. User exports via `exporter.js` (Excel/CSV/JSON) + +Run states: idle, starting, running, stopping, then stopped, blocked, failed or done. +`reason` says why it ended: complete, stopped, blocked, selectors_broken, storage_full, +storage_error, interrupted or updated. ### AI Chatbot (client-side, no server) @@ -89,7 +108,7 @@ AmazonSellerScraper/ 2. Widget uses Shadow DOM to isolate styles from Amazon's CSS 3. User types a question (e.g. "What's the best deal under $30?") 4. `chatbot.js` sends `CHAT_MESSAGE` to `service-worker.js` with the question and the last few turns -5. The service worker reads the user's key and the current run from `chrome.storage.local`; the content script never sees the key +5. The service worker reads the user's key from `chrome.storage.local` and the latest run from IndexedDB; the content script never sees the key 6. `scripts/lib/chat.js` builds the request: model id in `GEMINI_MODEL`, key in the `x-goog-api-key` header, titles in a fenced JSON block marked untrusted 7. Response displayed in chat bubble as text, never HTML @@ -114,16 +133,23 @@ npm run test:coverage # With coverage report **Test architecture:** - Pure logic modules (`analyzer.js`, `spread-analyzer.js`) are tested via `require()` directly - Content scripts (`scraper.js`, `offer-fetcher.js`) have no `module.exports` — loaded via `vm.runInContext` into a JSDOM context with Chrome API mocks and an `innerText` polyfill +- `tests/setup/engine-rig.js` runs the real router and engine over fake-indexeddb with the real `scraper.js` in a JSDOM page per tab; `tests/unit/engine.test.js` drives runs through it +- `npm run test:e2e` runs the same scenarios in Chromium (`tests/e2e/`), including the worker stopped via CDP between pages and 2.0 and 2.1 builds updated mid-run - HTML fixtures in `tests/fixtures/` match the exact CSS selectors the code uses - Chrome APIs (`storage`, `runtime`, `tabs`, `downloads`) are mocked in `tests/setup/chrome-mock.js` - XLSX is mocked with jest.fn() stubs in exporter.test.js; export-xlsx.test.js uses the real libs/xlsx.full.min.js ## Chrome APIs Used -- `chrome.storage.local` — state persistence -- `chrome.runtime.sendMessage/onMessage` — inter-script communication -- `chrome.downloads` — file downloads -- `chrome.tabs` — active tab messaging +- `chrome.storage.local`: settings, the Gemini key and `schemaVersion` only +- `chrome.storage.session`: the live run record (content scripts cannot read it) +- IndexedDB: run data, in the extension origin; needs no permission +- `chrome.runtime.sendMessage/onMessage`: messages, all in `scripts/lib/messages.js` +- `chrome.downloads`: file downloads +- `chrome.tabs`: `sendMessage`, `update`, `onRemoved`, `onUpdated`; none need the `tabs` permission + +No permission was added for the run engine: no `alarms`, `scripting`, `offscreen` or +`unlimitedStorage`. `npm run lock` fails on any addition. ## DOM Selectors (Amazon-specific, updated Sep 2026) @@ -175,17 +201,24 @@ One record per ASIN per run, from `Parsers.parseSearchPage` and the scraper: ## Storage -- `chrome.storage.local` carries `schemaVersion` (3). Storage without it came from 2.0. +- `chrome.storage.local` carries `schemaVersion` (4) and settings. Storage without it came from 2.0. `scripts/lib/migrate.js` brings it up to date on install, update and browser start; each step is idempotent. 2 to 3 adds `priceCents`, nulls and `/dp/` URLs to 2.0 rows, seeds `lastValues` from them (with `firstSeenAt`), ends a run the update cut off as - `updated`, and drops the untouched 2.0 default settings. + `updated`, and drops the untouched 2.0 default settings. 3 to 4 moves results, pages, + lastValues, spread data and the run into IndexedDB (a run still running is `updated`); + the IndexedDB writes commit before anything is removed. +- IndexedDB `proscan` (`scripts/background/db.js`): `runs`, `products` (key `[runId, n]`, + n is the order found), `placements` (one record per page), `lastValues` (by ASIN), + `outbox` (pages waiting for sync), `spread` and `meta` (`latestRunId`). The popup shows + the latest run's products. `tests/fixtures/v2.0-storage.json` is what the live 2.0 build stores, captured with `node tests/e2e/capture-v20-storage.mjs`. -- `lastValues` is capped by `Delta.prune`: 5,000 ASINs, none older than a year. -- `Storage` rejects when `chrome.runtime.lastError` is set (code `storage_full` on a - quota error). The popup warns from 90% of the quota. -- Cloud sync is off in 2.1 (`Flags.CLOUD_SYNC`): nothing goes into `syncQueue`, the SW +- `lastValues` is pruned after every page: 5,000 ASINs, none older than a year + (`Delta.MAX_ENTRIES`, `Delta.MAX_AGE_DAYS`). +- A page write that fails ends the run as `storage_full` (quota) or `storage_error`. + The popup warns from 90% of `navigator.storage.estimate()`. +- Cloud sync is off in 2.1 and 2.2 (`Flags.CLOUD_SYNC`): nothing goes into the outbox, the SW refuses `PROSCAN_EXPORT`, and the popup hides sign-in and Export to ProScan. ## Export diff --git a/README.md b/README.md index 2339ef9..8282519 100644 --- a/README.md +++ b/README.md @@ -58,11 +58,14 @@ ProScan is a Chrome extension that scrapes Amazon product listings across multip ``` User clicks "Start Scraping" - → popup.js pings the tab (offers a reload if ProScan is not loaded there) - → popup.js creates a run bound to that tab and sends START_SCRAPING - → scraper.js classifies the page, then extracts products - → Results and run progress stored in chrome.storage.local - → Follows the page's Next link after a 2 to 4 second delay + → popup.js sends START_RUN to the service worker + → The worker pings the tab (the popup offers a reload if ProScan is not + loaded there), makes a run bound to that tab and asks it to parse + → scraper.js classifies the page, extracts products and sends PAGE_RESULT; + it never writes storage and never navigates + → The worker saves the page to IndexedDB in one transaction + → After a 2 to 4 second delay the worker opens the page's Next link + with tabs.update, and the new page reports in → Repeats until the last page or the page cap (settings.maxPages, default 20) → The run ends with a reason: complete, stopped, blocked (captcha, bot check, sign-in), selectors_broken, storage_full, interrupted or @@ -76,7 +79,7 @@ User clicks "Start Scraping" ``` User types question in floating widget → chatbot.js sends CHAT_MESSAGE (question + last few turns) to service-worker.js - → Service worker reads the user's key and the current run from storage + → Service worker reads the user's key and the latest run from its storage → scripts/lib/chat.js builds the Gemini request (model id is GEMINI_MODEL there) → Response displayed in chat bubble as plain text ``` @@ -173,10 +176,12 @@ The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, s ## Chrome APIs Used -- `chrome.storage.local` -- Persistent state and product data -- `chrome.runtime.sendMessage` / `onMessage` -- Inter-script messaging +- `chrome.storage.local` -- Settings and the schema version only +- `chrome.storage.session` -- The live run record, written by the service worker +- IndexedDB (the extension's own origin, no permission) -- Runs, products, pages, lastValues and the sync outbox +- `chrome.runtime.sendMessage` / `onMessage` -- Messages, all listed in `scripts/lib/messages.js` - `chrome.downloads` -- File export downloads -- `chrome.tabs` -- Active tab communication +- `chrome.tabs` -- Messages to the run's tab, `tabs.update` for the next page, and `onRemoved` / `onUpdated` to notice the tab going away (none of these need the `tabs` permission) ## Project Structure @@ -189,17 +194,21 @@ AmazonSellerScraper/ │ └── popup.js # UI state management and export handling ├── scripts/ │ ├── content/ -│ │ ├── scraper.js # Scrape loop: storage, messages, pagination +│ │ ├── scraper.js # Parses a search page and reports it to the worker │ │ ├── chatbot.js # Floating AI chatbot (Shadow DOM) │ │ └── offer-fetcher.js # Seller offer page fetching for spread analysis │ ├── lib/ │ │ ├── parsers.js # Pure search and offer page parsing -│ │ ├── run.js # The scrape run record and its end reasons -│ │ ├── flags.js # Build flags (cloud sync is off in 2.1) +│ │ ├── messages.js # Every message type and who may send it +│ │ ├── run.js # The run state machine and its end reasons +│ │ ├── flags.js # Build flags (cloud sync is off until 2.3) │ │ ├── migrate.js # Storage schema migrations (schemaVersion) │ │ └── chat.js # Gemini request builder and run scoping │ ├── background/ -│ │ └── service-worker.js # Message routing + Gemini API +│ │ ├── service-worker.js # Wires the router, the engine, the chat and migrations +│ │ ├── router.js # The one message router +│ │ ├── engine.js # Runs scrapes; the only writer of run data +│ │ └── db.js # IndexedDB stores │ └── modules/ │ ├── storage.js # Chrome storage abstraction layer │ ├── analyzer.js # Analytics engine + opportunity scoring diff --git a/scripts/lib/flags.js b/scripts/lib/flags.js index 79ea62d..1a5a98b 100644 --- a/scripts/lib/flags.js +++ b/scripts/lib/flags.js @@ -1,13 +1,13 @@ /** * @fileoverview Build flags. * - * CLOUD_SYNC is off in 2.1. Scrapes are not queued for the cloud, the + * CLOUD_SYNC is off in 2.1 and 2.2. Nothing goes into the sync outbox, the * service worker refuses PROSCAN_EXPORT, and the popup hides the ProScan * sign-in and the Export to ProScan button. Sync comes back with the * per-run outbox (audit report, phase 4). * - * Loaded as a plain global (`Flags`) in the popup and content scripts, and - * as a CommonJS module in Jest and the bundled service worker. + * Loaded as a plain global (`Flags`) in the popup, and as a CommonJS + * module in Jest and the bundled service worker. */ const Flags = { diff --git a/scripts/lib/run.js b/scripts/lib/run.js index 48b5a58..433c8a2 100644 --- a/scripts/lib/run.js +++ b/scripts/lib/run.js @@ -10,8 +10,8 @@ * blocked, failed and done. `reason` says why a run ended; `END_STATE` maps * each reason to its end state. * - * Loaded as a plain global (`Run`) in the popup and content scripts, and as - * a CommonJS module in Jest and the bundled service worker. + * Loaded as a plain global (`Run`) in the popup, and as a CommonJS module + * in Jest and the bundled service worker. * * @module Run */ From 75027218807b6a3570256ead637d323c1137203e Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:05:16 -0400 Subject: [PATCH 19/25] End a run cut off by an update as updated, not interrupted --- scripts/background/engine.js | 10 +++++++--- scripts/background/service-worker.js | 2 +- tests/unit/engine.test.js | 11 +++++++++++ 3 files changed, 19 insertions(+), 4 deletions(-) diff --git a/scripts/background/engine.js b/scripts/background/engine.js index 5c0b52c..3043cae 100644 --- a/scripts/background/engine.js +++ b/scripts/background/engine.js @@ -368,15 +368,19 @@ function createEngine({ }); } - /** After a browser restart the session is empty; a run IndexedDB still has as live was cut off. */ - function recover() { + /** + * A browser restart or an update empties session storage, so a run + * IndexedDB still has as live was cut off: `interrupted` after a + * restart, `updated` after an update. + */ + function recover(reason = 'interrupted') { return serial(async () => { if (await getRun()) return; const store = await db(); const latest = await store.getMeta('latestRunId'); const rec = latest ? await store.get('runs', latest) : null; if (Run.isActive(rec)) { - await store.write([{ store: 'runs', put: durable(Run.finish(rec, 'interrupted', now())) }]); + await store.write([{ store: 'runs', put: durable(Run.finish(rec, reason, now())) }]); } }); } diff --git a/scripts/background/service-worker.js b/scripts/background/service-worker.js index 7297ee6..13cdecb 100644 --- a/scripts/background/service-worker.js +++ b/scripts/background/service-worker.js @@ -80,7 +80,7 @@ chrome.runtime.onInstalled.addListener((details) => { }); } else if (details.reason === 'update') { console.log('[ProScan] Extension updated to version', chrome.runtime.getManifest().version); - migrate().then(() => engine.recover()); + migrate().then(() => engine.recover('updated')); } }); diff --git a/tests/unit/engine.test.js b/tests/unit/engine.test.js index ddec08b..97296ce 100644 --- a/tests/unit/engine.test.js +++ b/tests/unit/engine.test.js @@ -287,6 +287,17 @@ describe('the worker stopped between pages', () => { expect(st.run).toMatchObject({ state: 'failed', reason: 'interrupted' }); }); + test('after an update empties session storage, the live run is ended as updated', async () => { + const rig = createRig({ site: simpleSite(3) }); + await startIn(rig, 'upd'); + await settle(); + await rig.session.remove('run'); + rig.killWorker(); + await rig.engine().recover('updated'); + const st = await rig.popup({ type: 'GET_STATE' }); + expect(st.run).toMatchObject({ state: 'failed', reason: 'updated' }); + }); + test('after a browser restart a run IndexedDB still has as live is ended', async () => { const rig = createRig({ site: simpleSite(3) }); await startIn(rig, 'restart'); From f9b9b733fe75166051402b2bc3744e312605b13d Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:05:16 -0400 Subject: [PATCH 20/25] Ask a page again when it has not reported 20 seconds after the worker opened it --- scripts/background/engine.js | 3 +++ tests/setup/engine-rig.js | 7 +++++++ tests/unit/engine.test.js | 15 +++++++++++++++ 3 files changed, 25 insertions(+) diff --git a/scripts/background/engine.js b/scripts/background/engine.js index 3043cae..72e4e25 100644 --- a/scripts/background/engine.js +++ b/scripts/background/engine.js @@ -325,7 +325,10 @@ function createEngine({ await chrome.tabs.update(run.tabId, { url: run.expectUrl }); } catch (err) { await end(run, 'interrupted'); + return; } + // If the page never reports, ask it again. + schedule(PAGE_TIMEOUT_MS + 1000); return; } if (run.awaiting && t - (run.navigatedAt || 0) > PAGE_TIMEOUT_MS) { diff --git a/tests/setup/engine-rig.js b/tests/setup/engine-rig.js index 821591f..45f975f 100644 --- a/tests/setup/engine-rig.js +++ b/tests/setup/engine-rig.js @@ -55,6 +55,7 @@ function createRig({ site, flags, local = {}, now = () => Date.now(), random = ( const tabs = new Map(); const served = []; const sent = []; + const dropped = new Set(); const log = { log() {}, warn() {}, error() {} }; let nextTabId = 1; let worker = null; @@ -113,6 +114,10 @@ function createRig({ site, flags, local = {}, now = () => Date.now(), random = ( /** A content script message to the worker, from tab `tabId`. Starts a stopped worker. */ function fromTab(tabId, message, cb) { sent.push({ tabId, type: message && message.type }); + if (dropped.has(message && message.type)) { + setImmediate(() => { if (cb) withError('Could not establish connection. Receiving end does not exist.', () => cb()); }); + return; + } setImmediate(() => { const w = worker || startWorker(); let answered = false; @@ -171,6 +176,8 @@ function createRig({ site, flags, local = {}, now = () => Date.now(), random = ( return (worker || startWorker()).engine.tabRemoved(id); }, goTo(id, url) { navigate(id, url); }, + /** Messages of these types from tabs are lost, as when the worker is not up. */ + drop(...types) { dropped.clear(); types.forEach((t) => dropped.add(t)); }, tabCtx: (id) => tabs.get(id).ctx, /** A popup message to the worker. */ popup(message) { diff --git a/tests/unit/engine.test.js b/tests/unit/engine.test.js index 97296ce..5f5984a 100644 --- a/tests/unit/engine.test.js +++ b/tests/unit/engine.test.js @@ -311,6 +311,21 @@ describe('the worker stopped between pages', () => { }); }); +describe('a page that never reports', () => { + test('is asked again after the page timeout, and the run goes on', async () => { + const rig = createRig({ site: simpleSite(3) }); + await startIn(rig, 'lost'); + await settle(); + rig.drop('PAGE_READY'); + await settle(4000); + expect(served(rig, 'lost')).toEqual([1, 2]); + expect((await rig.run()).page).toBe(1); + rig.drop(); + const run = await runUntilEnd(rig, 60000); + expect(run).toMatchObject({ reason: 'complete', page: 3 }); + }); +}); + describe('page results', () => { test('a repeated or out of order PAGE_RESULT is ignored', async () => { const rig = createRig({ site: simpleSite(3) }); From 433a79424dd946595fe8769e6d70d0d20c099536 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:06:14 -0400 Subject: [PATCH 21/25] End a run whose tab left Amazon while its next page was loading --- scripts/background/engine.js | 7 ++++++- tests/setup/chrome-mock.js | 1 + tests/setup/engine-rig.js | 6 ++++++ tests/unit/engine.test.js | 28 ++++++++++++++++++++++++++++ 4 files changed, 41 insertions(+), 1 deletion(-) diff --git a/scripts/background/engine.js b/scripts/background/engine.js index 72e4e25..5cd6487 100644 --- a/scripts/background/engine.js +++ b/scripts/background/engine.js @@ -333,7 +333,12 @@ function createEngine({ } if (run.awaiting && t - (run.navigatedAt || 0) > PAGE_TIMEOUT_MS) { await saveRun({ ...run, navigatedAt: t }); - sendToTab(run.tabId, { type: Msg.T.PARSE_PAGE, runId: run.runId, page: run.page + 1 }); + schedule(PAGE_TIMEOUT_MS + 1000); + const ack = await sendToTab(run.tabId, { type: Msg.T.PARSE_PAGE, runId: run.runId, page: run.page + 1 }); + if (ack) return; + // No script answers. Still on Amazon means still loading; a hidden URL means the tab left. + const tab = await chrome.tabs.get(run.tabId).catch(() => null); + if (!tab || !tab.url) await end(run, 'interrupted'); } }); } diff --git a/tests/setup/chrome-mock.js b/tests/setup/chrome-mock.js index 4a9df2e..aebce9e 100644 --- a/tests/setup/chrome-mock.js +++ b/tests/setup/chrome-mock.js @@ -112,6 +112,7 @@ global.chrome = { if (cb) cb({}); }, update: async function(tabId, props) { return { id: tabId, ...props }; }, + get: async function(tabId) { return { id: tabId, url: 'https://www.amazon.com/s?k=test' }; }, onRemoved: { addListener() {} }, onUpdated: { addListener() {} } }, diff --git a/tests/setup/engine-rig.js b/tests/setup/engine-rig.js index 45f975f..623f00d 100644 --- a/tests/setup/engine-rig.js +++ b/tests/setup/engine-rig.js @@ -81,6 +81,12 @@ function createRig({ site, flags, local = {}, now = () => Date.now(), random = ( if (!held && !answered) withError('The message port closed before a response was received.', () => cb()); }); }, + /** Without the tabs permission Chrome shows the URL of Amazon tabs only. */ + async get(tabId) { + const tab = tabs.get(tabId); + if (!tab) throw new Error(`No tab with id: ${tabId}.`); + return { id: tabId, url: /^https:\/\/www\.amazon\.com\//.test(tab.url) ? tab.url : undefined }; + }, async update(tabId, { url }) { if (!tabs.has(tabId)) throw new Error(`No tab with id: ${tabId}.`); setImmediate(() => navigate(tabId, url)); diff --git a/tests/unit/engine.test.js b/tests/unit/engine.test.js index 5f5984a..10d7040 100644 --- a/tests/unit/engine.test.js +++ b/tests/unit/engine.test.js @@ -31,6 +31,7 @@ const pageNo = (url) => Number(new URL(url).searchParams.get('page') || 1); function simpleSite(pages, { captchaAt = null } = {}) { return (url) => { const u = new URL(url); + if (u.hostname !== 'www.amazon.com') return null; if (u.pathname.startsWith('/dp/')) return PRODUCT; const k = u.searchParams.get('k'); const p = pageNo(url); @@ -326,6 +327,33 @@ describe('a page that never reports', () => { }); }); +describe('a page that never reports, continued', () => { + test('when the tab is no longer on Amazon, the run ends as interrupted', async () => { + const site = simpleSite(3); + const rig = createRig({ site: (url) => (pageNo(url) === 2 ? null : site(url)) }); + await startIn(rig, 'away'); + await settle(4000); + expect(served(rig, 'away')).toEqual([1, 2]); + rig.goTo((await rig.run()).tabId, 'https://example.com/'); + const run = await runUntilEnd(rig, 60000); + expect(run).toMatchObject({ reason: 'interrupted', page: 1 }); + }); + + test('a slow page still on Amazon is waited for', async () => { + const site = simpleSite(3); + let slow = true; + const rig = createRig({ site: (url) => (pageNo(url) === 2 && slow ? null : site(url)) }); + const { tab } = await startIn(rig, 'slowly'); + await settle(4000); + await settle(25000); + expect(await rig.run()).toMatchObject({ state: 'running', awaiting: true }); + slow = false; + rig.goTo(tab, searchUrl('slowly', 2)); + const run = await runUntilEnd(rig); + expect(run).toMatchObject({ reason: 'complete', page: 3 }); + }); +}); + describe('page results', () => { test('a repeated or out of order PAGE_RESULT is ignored', async () => { const rig = createRig({ site: simpleSite(3) }); From 95e020432fa34ac0ce0440566c8add905af982d2 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:15:45 -0400 Subject: [PATCH 22/25] Report a page again when the worker re-asks after every try failed --- scripts/content/scraper.js | 2 ++ tests/unit/scraper-run.test.js | 16 ++++++++++++++++ 2 files changed, 18 insertions(+) diff --git a/scripts/content/scraper.js b/scripts/content/scraper.js index e3c2032..a285920 100644 --- a/scripts/content/scraper.js +++ b/scripts/content/scraper.js @@ -100,6 +100,8 @@ async function report(message, tries) { if (!reply) { // The worker may be starting up; it drops a page it already has. if (tries > 1 && alive()) setTimeout(() => report(message, tries - 1), RETRY_MS); + // Never got through, so a later PARSE_PAGE may try again. + else reported.delete(message.runId + ':' + message.page); return; } if (reply.next === 'wait') startHeartbeat(message.runId); diff --git a/tests/unit/scraper-run.test.js b/tests/unit/scraper-run.test.js index d07b110..d3b9d1f 100644 --- a/tests/unit/scraper-run.test.js +++ b/tests/unit/scraper-run.test.js @@ -117,6 +117,22 @@ test('a page is reported once per run, however often it is asked', async () => { expect(types().filter((t) => t === 'PAGE_RESULT')).toHaveLength(1); }); +test('a page whose report never got through is reported again when asked', async () => { + replies.PAGE_READY = { parse: true, runId: 'r1', page: 1 }; + loadContentScript('scripts/content/scraper.js', html, URL1); + await settle(); + for (let i = 0; i < 3; i++) { + jest.advanceTimersByTime(1000); + await settle(); + } + expect(types().filter((t) => t === 'PAGE_RESULT')).toHaveLength(3); + replies.PAGE_RESULT = { ok: true, next: 'wait' }; + listener({ type: 'PARSE_PAGE', runId: 'r1', page: 1 }, {}, jest.fn()); + jest.advanceTimersByTime(0); + await settle(); + expect(types().filter((t) => t === 'PAGE_RESULT')).toHaveLength(4); +}); + test('with no answer from a starting worker it tries again', async () => { loadContentScript('scripts/content/scraper.js', html, URL1); await settle(); From b96d1fa39436248f921719a7f32245dc8464ea0f Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 06:58:15 -0400 Subject: [PATCH 23/25] Let the install defaults land before the page cap test sets its own --- tests/e2e/scrape.spec.mjs | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/e2e/scrape.spec.mjs b/tests/e2e/scrape.spec.mjs index 6526bd2..3658ec4 100644 --- a/tests/e2e/scrape.spec.mjs +++ b/tests/e2e/scrape.spec.mjs @@ -309,6 +309,8 @@ test('a service worker stopped between pages does not break the run', async ({ e test('the page cap in settings ends the run as complete', async ({ ext }) => { const served = await serveAmazon(ext.context, simplePlan(5)); const store = await extPage(ext); + // Wait for the install defaults first, or they can overwrite this on a slow runner. + await waitForState(store, (x) => x.schemaVersion, { timeout: 15000, interval: 100 }); await store.evaluate(() => chrome.storage.local.set({ settings: { pageDelay: 2000, maxPages: 2 } })); const tab = await openSearch(ext, 'capped'); await clickStart(ext, tab); From e2f5ad48f2bf429ce4a704fa17b94554c8fce366 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 16:30:07 -0400 Subject: [PATCH 24/25] End runs left live only in IndexedDB, keep the newest 10 runs, and order the 3 to 4 migration With no session record (the extension was disabled and enabled, or its process crashed), GET_STATE now ends the durable run as interrupted and Stop ends it as stopped, so the popup is no longer stuck on a run that cannot be stopped. Starting a run removes runs past the newest 10, except ones still in the outbox, and a full disk at start drops the rest and tries once more, so the storage full message is true. The 3 to 4 migration no longer overwrites latestRunId or outbox rows the engine wrote first. The engine also checks that the page a tab reports is the one it opened. --- README.md | 2 +- popup/popup.js | 4 +- scripts/background/db.js | 12 +++++- scripts/background/engine.js | 80 +++++++++++++++++++++++++++++++--- scripts/lib/migrate.js | 13 ++++-- scripts/lib/run.js | 2 +- tests/unit/engine.test.js | 83 ++++++++++++++++++++++++++++++++++++ tests/unit/migrate.test.js | 19 +++++++++ 8 files changed, 200 insertions(+), 15 deletions(-) diff --git a/README.md b/README.md index 8282519..9bd19dc 100644 --- a/README.md +++ b/README.md @@ -178,7 +178,7 @@ The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, s - `chrome.storage.local` -- Settings and the schema version only - `chrome.storage.session` -- The live run record, written by the service worker -- IndexedDB (the extension's own origin, no permission) -- Runs, products, pages, lastValues and the sync outbox +- IndexedDB (the extension's own origin, no permission) -- Runs, products, pages, lastValues and the sync outbox. Starting a run keeps the 10 newest runs and removes older ones, except runs still waiting to sync. - `chrome.runtime.sendMessage` / `onMessage` -- Messages, all listed in `scripts/lib/messages.js` - `chrome.downloads` -- File export downloads - `chrome.tabs` -- Messages to the run's tab, `tabs.update` for the next page, and `onRemoved` / `onUpdated` to notice the tab going away (none of these need the `tabs` permission) diff --git a/popup/popup.js b/popup/popup.js index 0927f99..2ce64c6 100644 --- a/popup/popup.js +++ b/popup/popup.js @@ -252,7 +252,7 @@ async function warnIfNearlyFull() { const { usage, quota } = await navigator.storage.estimate(); if (quota && usage >= quota * Storage.NEAR_FULL) { const pct = Math.round((usage / quota) * 100); - updateStatus(`Browser storage is ${pct}% full. Download your results before the next run.`, 'warning'); + updateStatus(`Browser storage is ${pct}% full. Download your results; the next run removes older saved scans.`, 'warning'); } } catch (e) { /* no estimate in this context */ } } @@ -295,7 +295,7 @@ const AMAZON_URL = /^https:\/\/([a-z0-9-]+\.)*amazon\.com\//i; * * The worker pings the tab and changes nothing if the content script is not * there (an open tab keeps the old script after an update) or the page is - * not a search. Old runs are kept; the view moves to the new one. + * not a search. The 10 newest runs are kept; the view moves to the new one. * * @async */ diff --git a/scripts/background/db.js b/scripts/background/db.js index d3a7846..8f161c6 100644 --- a/scripts/background/db.js +++ b/scripts/background/db.js @@ -122,7 +122,10 @@ function api(idb) { /** * Applies `ops` in one transaction, all or nothing. Each op is - * {store, put}, {store, delete} or {store, deleteIndex: [index, key]}. + * {store, put}, {store, delete}, {store, deleteIndex: [index, key]}, or + * {store, putIfAbsent, unlessIndex?: [index, key]}, which writes only + * when no record has its key (or, with unlessIndex, none is in that + * index under that key). */ function write(ops) { const stores = [...new Set(ops.map((op) => op.store))]; @@ -131,6 +134,13 @@ function api(idb) { for (const op of ops) { const s = tx.objectStore(op.store); if ('put' in op) s.put(op.put); + else if ('putIfAbsent' in op) { + const v = op.putIfAbsent; + const probe = op.unlessIndex + ? s.index(op.unlessIndex[0]).count(op.unlessIndex[1]) + : s.count([].concat(s.keyPath).length > 1 ? s.keyPath.map((k) => v[k]) : v[s.keyPath]); + probe.onsuccess = () => { if (!probe.result) s.put(v); }; + } else if ('delete' in op) s.delete(op.delete); else if (op.deleteIndex) { s.index(op.deleteIndex[0]).openKeyCursor(op.deleteIndex[1]).onsuccess = (ev) => { diff --git a/scripts/background/engine.js b/scripts/background/engine.js index 5cd6487..329fbfc 100644 --- a/scripts/background/engine.js +++ b/scripts/background/engine.js @@ -29,6 +29,9 @@ const Delta = require('../modules/delta.js'); /** A page asked for but not reported this long is asked for again. */ const PAGE_TIMEOUT_MS = 20000; +/** Runs kept in IndexedDB. Starting a run removes the oldest past this. */ +const KEEP_RUNS = 10; + /** * Folds one page's products into a run. Placements get the page number and * a run-wide organic rank. An ASIN the run already has only adds its @@ -171,6 +174,28 @@ function createEngine({ return run; } + /** + * Keeps the newest `keep` runs (KEEP_RUNS - 1 by default, so the one + * starting makes KEEP_RUNS). A run still in the outbox or still live is + * never removed. + */ + async function pruneRuns(store, keep = KEEP_RUNS - 1) { + const [runs, outbox] = await Promise.all([store.getAll('runs'), store.getAll('outbox')]); + const pending = new Set(outbox.map(e => e.runId)); + const old = runs + .sort((a, b) => (b.startedAt || 0) - (a.startedAt || 0)) + .slice(keep) + .filter(r => !pending.has(r.runId) && !Run.isActive(r)); + if (old.length === 0) return 0; + const ops = []; + for (const { runId } of old) { + ops.push({ store: 'runs', delete: runId }); + for (const s of ['products', 'placements', 'spread']) ops.push({ store: s, deleteIndex: ['runId', runId] }); + } + await store.write(ops); + return old.length; + } + function start({ tabId }) { return serial(async () => { if (typeof tabId !== 'number') return { error: 'refused', message: Run.refusal('unknown') }; @@ -190,11 +215,24 @@ function createEngine({ const runId = `${t}-${runSuffix(random)}`; let run = Run.create({ runId, tabId, maxPages: settings.maxPages, source: Run.sourceOf(pong.url, t), now: t }); await saveRun(run); + const first = [ + { store: 'runs', put: durable(run) }, + { store: 'meta', put: { key: 'latestRunId', value: runId } } + ]; try { - await (await db()).write([ - { store: 'runs', put: durable(run) }, - { store: 'meta', put: { key: 'latestRunId', value: runId } } - ]); + await pruneRuns(await db()); + } catch (err) { + log.warn('[ProScan] Could not remove old runs:', err.message); + } + try { + try { + await (await db()).write(first); + } catch (err) { + if (err.code !== 'storage_full') throw err; + // Full: drop every older run that is not waiting to sync, then try once more. + await pruneRuns(await db(), 0); + await (await db()).write(first); + } } catch (err) { await end(run, err.code === 'storage_full' ? 'storage_full' : 'storage_error'); return { error: err.code || 'storage_error', message: Run.MESSAGES[err.code] || Run.MESSAGES.storage_error }; @@ -211,9 +249,29 @@ function createEngine({ }); } + /** + * The latest run IndexedDB still has as live when session storage has + * no run: the extension was disabled and enabled again, or its process + * crashed. Nothing can resume it, so it is ended as `reason`. Returns + * the ended record, or null. Call inside serial(). + */ + async function endOrphan(store, reason) { + if (await getRun()) return null; + const latestId = await store.getMeta('latestRunId'); + const rec = latestId ? await store.get('runs', latestId) : null; + if (!Run.isActive(rec)) return null; + const ended = Run.finish(rec, reason, now()); + await store.write([{ store: 'runs', put: durable(ended) }]); + return ended; + } + function stop() { return serial(async () => { const run = await getRun(); + if (!run) { + const ended = await endOrphan(await db(), 'stopped').catch(() => null); + return { ok: true, stopped: !!ended }; + } if (!Run.isActive(run)) return { ok: true, stopped: false }; const stopping = Run.transition(run, 'stopping'); await saveRun(stopping); @@ -228,7 +286,11 @@ function createEngine({ const run = await liveRun(); const tabId = sender && sender.tab ? sender.tab.id : null; if (!Run.owns(run, tabId) || run.state !== 'running') return { idle: true }; - if (run.awaiting) return { parse: true, runId: run.runId, page: run.page + 1 }; + // Only the page the engine opened is parsed. A search the user + // typed in the tab while it was loading ends the run instead. + if (run.awaiting && (!run.expectUrl || Run.samePage(run.expectUrl, url))) { + return { parse: true, runId: run.runId, page: run.page + 1 }; + } if (url === run.lastUrl) return { heartbeat: true, runId: run.runId }; // The tab went somewhere else while the next page was pending. await end(run, 'interrupted'); @@ -413,7 +475,11 @@ function createEngine({ ]); const spread = {}; spreadRows.forEach(r => { spread[r.asin] = r.data; }); - const run = live && live.runId === runId ? live : (rec || null); + let run = live && live.runId === runId ? live : (rec || null); + if (!live && Run.isActive(rec)) { + // Live in IndexedDB only: no session record means nothing drives it. + run = (await endOrphan(store, 'interrupted').catch(() => null)) || Run.finish(rec, 'interrupted', now()); + } return { run, results, pages, spread }; } @@ -479,4 +545,4 @@ function createEngine({ }; } -module.exports = { createEngine, foldPage, durable, PAGE_TIMEOUT_MS }; +module.exports = { createEngine, foldPage, durable, PAGE_TIMEOUT_MS, KEEP_RUNS }; diff --git a/scripts/lib/migrate.js b/scripts/lib/migrate.js index 5e8ef36..3e4d02b 100644 --- a/scripts/lib/migrate.js +++ b/scripts/lib/migrate.js @@ -164,15 +164,22 @@ const Migrate = (() => { if (snap && typeof snap === 'object') ops.push({ store: 'lastValues', put: { ...snap, asin } }); }); - [...new Set(queued.map(runOf))].forEach((runId, i) => { - ops.push({ store: 'outbox', put: { seq: i + 1, runId, pageIndex: null, queuedAt: now } }); + // The outbox numbers its own entries, so ones the engine already + // added are never overwritten, and a retry adds nothing twice. + [...new Set(queued.map(runOf))].forEach((runId) => { + ops.push({ + store: 'outbox', + putIfAbsent: { runId, pageIndex: null, queuedAt: now }, + unlessIndex: ['runId', runId] + }); }); if (latestRunId) { Object.entries(store.spreadResults || {}).forEach(([asin, data]) => { ops.push({ store: 'spread', put: { runId: latestRunId, asin, data: data || null } }); }); - ops.push({ store: 'meta', put: { key: 'latestRunId', value: latestRunId } }); + // A retry that runs after a 2.2 run must not point back at 2.1. + ops.push({ store: 'meta', putIfAbsent: { key: 'latestRunId', value: latestRunId } }); } return ops; } diff --git a/scripts/lib/run.js b/scripts/lib/run.js index 433c8a2..06d8f0e 100644 --- a/scripts/lib/run.js +++ b/scripts/lib/run.js @@ -30,7 +30,7 @@ const Run = (() => { stopped: 'Scraping stopped.', blocked: 'Amazon showed a captcha or a sign-in page, so the run stopped. Solve it in the tab, then start again.', selectors_broken: 'ProScan could not read this page. Amazon may have shown an error or changed its layout.', - storage_full: 'Browser storage is full, so the run stopped. Download your results. Starting a new scan clears them.', + storage_full: 'Browser storage is full, so the run stopped. Download your results. Starting a new scan removes older saved scans.', interrupted: 'The run ended early because its tab was closed or left the search.', storage_error: 'ProScan could not save a page, so the run stopped.', updated: 'ProScan was updated during the run, so it stopped. The products found before the update are kept.' diff --git a/tests/unit/engine.test.js b/tests/unit/engine.test.js index 10d7040..82f41a3 100644 --- a/tests/unit/engine.test.js +++ b/tests/unit/engine.test.js @@ -575,3 +575,86 @@ describe('foldPage', () => { expect(changed[0].placements).toHaveLength(2); }); }); + +describe('a run left live only in IndexedDB', () => { + test('session storage wiped with no recover (disable and enable): the run shows as ended and Start works', async () => { + const rig = createRig({ site: simpleSite(3) }); + await startIn(rig, 'orphan'); + await settle(1000); + await rig.session.remove('run'); + rig.killWorker(); + const st = await rig.popup({ type: 'GET_STATE' }); + expect(st.run).toMatchObject({ state: 'failed', reason: 'interrupted' }); + expect(Run.isActive(st.run)).toBe(false); + expect(st.results).toHaveLength(4); + const db = await rig.db(); + expect(Run.isActive(await db.get('runs', st.run.runId))).toBe(false); + db.close(); + const tab2 = rig.openTab(searchUrl('after')); + await settle(); + expect(await rig.popup({ type: 'START_RUN', tabId: tab2 })).toMatchObject({ ok: true }); + }); + + test('Stop ends it when session storage has no run', async () => { + const rig = createRig({ site: simpleSite(3) }); + await startIn(rig, 'orphan2'); + await settle(1000); + await rig.session.remove('run'); + rig.killWorker(); + expect(await rig.popup({ type: 'STOP_RUN' })).toEqual({ ok: true, stopped: true }); + const st = await rig.popup({ type: 'GET_STATE' }); + expect(st.run).toMatchObject({ reason: 'stopped' }); + }); +}); + +describe('the run tab', () => { + test('a new search that loads while the next page is pending ends the run and is not parsed', async () => { + const rig = createRig({ site: simpleSite(3) }); + const { tab } = await startIn(rig, 'mine'); + await settle(); + const run = await rig.run(); + await rig.session.set({ run: { ...run, awaiting: true, expectUrl: searchUrl('mine', 2), navAt: null } }); + const out = await rig.engine().pageReady({ url: searchUrl('theirs') }, { tab: { id: tab } }); + expect(out).toEqual({ idle: true }); + expect(await rig.run()).toMatchObject({ reason: 'interrupted' }); + expect((await results(rig)).map((r) => r.asin)).toEqual([0, 1, 2, 3].map((i) => asinFor('mine', 1, i))); + }); + + test('the page it opened is parsed even when Amazon rewrites qid and ref', async () => { + const rig = createRig({ site: simpleSite(3) }); + const { tab } = await startIn(rig, 'mine'); + await settle(); + const run = await rig.run(); + await rig.session.set({ run: { ...run, awaiting: true, expectUrl: `${searchUrl('mine', 2)}&ref=sr_pg_1`, navAt: null } }); + const out = await rig.engine().pageReady({ url: `${searchUrl('mine', 2)}&qid=5&ref=sr_pg_2` }, { tab: { id: tab } }); + expect(out).toMatchObject({ parse: true, page: 2 }); + }); +}); + +describe('old runs', () => { + test(`starting a run keeps the newest ${require('../../scripts/background/engine').KEEP_RUNS}, and never one waiting to sync`, async () => { + const { KEEP_RUNS } = require('../../scripts/background/engine'); + const rig = createRig({ site: simpleSite(1) }); + const db = await rig.db(); + const ops = []; + for (let i = 0; i < KEEP_RUNS + 3; i++) { + const runId = `old-${String(i).padStart(2, '0')}`; + ops.push({ store: 'runs', put: { runId, state: 'done', reason: 'complete', startedAt: 1000 + i } }); + ops.push({ store: 'products', put: { runId, n: 0, asin: 'B0OLD00000' } }); + ops.push({ store: 'placements', put: { runId, pageIndex: 1, count: 1 } }); + } + ops.push({ store: 'outbox', put: { runId: 'old-00', pageIndex: 1, queuedAt: 1 } }); + await db.write(ops); + await startIn(rig, 'fresh'); + await runUntilEnd(rig); + const runs = (await db.getAll('runs')).map((r) => r.runId).sort(); + // The run just made, the 9 newest old ones, and old-00 kept for the outbox. + expect(runs).toHaveLength(KEEP_RUNS + 1); + expect(runs).toContain('old-00'); + expect(runs).not.toContain('old-01'); + expect(await db.runProducts('old-01')).toEqual([]); + expect(await db.runPages('old-01')).toEqual([]); + expect(await db.runProducts(`old-${KEEP_RUNS + 2}`)).toHaveLength(1); + db.close(); + }); +}); diff --git a/tests/unit/migrate.test.js b/tests/unit/migrate.test.js index aa60463..98514ec 100644 --- a/tests/unit/migrate.test.js +++ b/tests/unit/migrate.test.js @@ -297,4 +297,23 @@ describe('3 -> 4: into IndexedDB', () => { expect((await db.getAll('outbox')).map((o) => o.runId).sort()).toEqual(['r0', 'r21']); db.close(); }); + + test('a retry after a 2.2 run keeps the newer latest run and its outbox entries', async () => { + const queued = [{ asin: 'B0QQQQQQQ1', runId: 'r0' }]; + await chrome.storage.local.set({ ...V21(), syncQueue: queued }); + const db = await openDb(); + // The engine got there first: a 2.2 run and its queued page. + await db.write([ + { store: 'runs', put: { runId: 'r22', state: 'done', reason: 'complete', startedAt: 9 } }, + { store: 'meta', put: { key: 'latestRunId', value: 'r22' } }, + { store: 'outbox', put: { runId: 'r22', pageIndex: 1, queuedAt: 9 } }, + ]); + await Migrate.run(chrome.storage.local, { now: NOW, openDb }); + await Migrate.run({ get: async () => ({ ...V21(), syncQueue: queued }), remove: async () => {}, set: async () => {} }, { now: NOW, openDb }); + expect(await db.getMeta('latestRunId')).toBe('r22'); + const outbox = await db.getAll('outbox'); + expect(outbox.map((o) => o.runId).sort()).toEqual(['r0', 'r22']); + expect(outbox.find((o) => o.runId === 'r22')).toMatchObject({ pageIndex: 1 }); + db.close(); + }); }); From 7ce2963b5c8f0167af0ea4e74b67bf89180b10e8 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 16:47:46 -0400 Subject: [PATCH 25/25] Read the run-tab and 503 scenarios through the 2.2 run record --- tests/e2e/scrape.spec.mjs | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/tests/e2e/scrape.spec.mjs b/tests/e2e/scrape.spec.mjs index 3658ec4..833a124 100644 --- a/tests/e2e/scrape.spec.mjs +++ b/tests/e2e/scrape.spec.mjs @@ -372,7 +372,7 @@ test('a new search typed in the run tab ends the run and is not scraped into it' await clickStart(ext, tab); await waitForState(store, (x) => x.scrapeRunPages?.length >= 1, { timeout: 15000, interval: 100 }); await tab.goto(searchUrl('second')); - const s = await waitForState(store, (x) => x.run && x.run.status !== 'running', { timeout: 15000 }); + const s = await waitForState(store, (x) => ended(x), { timeout: 15000 }); await sleep(5000); expect(pagesOf(served, 'second')).toEqual([1]); @@ -391,12 +391,14 @@ test('a run tab that left Amazon and comes back after a minute does not resume', await tab.route('https://example.com/**', (r) => r.fulfill({ status: 200, contentType: 'text/html', body: '

elsewhere

' })); await tab.goto('https://example.com/'); // Age the heartbeat instead of waiting out the 60 s. + // From 2.2 the live run is in session storage. await store.evaluate(async () => { - const { run } = await chrome.storage.local.get('run'); - await chrome.storage.local.set({ run: { ...run, heartbeat: Date.now() - 120000 } }); + const area = (await chrome.storage.session.get('run')).run ? chrome.storage.session : chrome.storage.local; + const { run } = await area.get('run'); + if (run) await area.set({ run: { ...run, heartbeat: Date.now() - 120000 } }); }); await tab.goto(searchUrl('wander', 2)); - const s = await waitForState(store, (x) => x.run && x.run.status !== 'running', { timeout: 15000 }); + const s = await waitForState(store, (x) => ended(x), { timeout: 15000 }); await sleep(5000); expect(pagesOf(served, 'wander')).toEqual([1, 2]); @@ -412,7 +414,7 @@ test('a 503 error page at page 2 ends the run as a warning, not complete', async const tab = await openSearch(ext, 'throttled'); const store = await extPage(ext); await clickStart(ext, tab); - const s = await waitForState(store, (x) => x.run && x.run.status !== 'running', { timeout: 30000 }); + const s = await waitForState(store, (x) => ended(x), { timeout: 30000 }); await sleep(2500); expect(pagesOf(served, 'throttled')).toEqual([1, 2]);