From 9699abe32ffb0246f0d93cd2f29b0557185051c5 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:24:23 -0400 Subject: [PATCH 01/18] Add a shared cloud schema with validators and deterministic ids --- jest.config.js | 12 +- packages/schema/index.js | 480 ++++++++++++++++++++++++++++ tests/setup/esm-to-cjs-transform.js | 6 + tests/unit/schema.test.js | 130 ++++++++ 4 files changed, 623 insertions(+), 5 deletions(-) create mode 100644 packages/schema/index.js create mode 100644 tests/unit/schema.test.js diff --git a/jest.config.js b/jest.config.js index 30665ce..61d2d1f 100644 --- a/jest.config.js +++ b/jest.config.js @@ -6,12 +6,13 @@ module.exports = { // Older builds checked out by the e2e upgrade harness have their own tests. testPathIgnorePatterns: ['/node_modules/', '/tests/e2e/\\.build/'], modulePathIgnorePatterns: ['/tests/e2e/\\.build/'], - // sync.js and service-worker.js are authored as ESM (esbuild bundles - // them). A tiny scoped transform rewrites their import/export to - // CommonJS so they can be unit-tested; every other file keeps the - // default babel-jest transform. + // sync.js, sync-plan.js, service-worker.js and the shared schema are + // authored as ESM (esbuild bundles them). A tiny scoped transform + // rewrites their import/export to CommonJS so they can be unit-tested; + // every other file keeps the default babel-jest transform. transform: { - '[\\\\/]scripts[\\\\/]background[\\\\/](sync|service-worker)\\.js$': '/tests/setup/esm-to-cjs-transform.js', + '[\\\\/]scripts[\\\\/]background[\\\\/](sync|sync-plan|service-worker)\\.js$': '/tests/setup/esm-to-cjs-transform.js', + '[\\\\/]packages[\\\\/]schema[\\\\/]index\\.js$': '/tests/setup/esm-to-cjs-transform.js', '\\.[jt]sx?$': 'babel-jest' }, coverageDirectory: 'coverage', @@ -20,6 +21,7 @@ module.exports = { 'scripts/lib/*.js', 'scripts/content/scraper.js', 'scripts/content/offer-fetcher.js', + 'packages/schema/index.js', '!**/node_modules/**' ] }; diff --git a/packages/schema/index.js b/packages/schema/index.js new file mode 100644 index 0000000..edb605e --- /dev/null +++ b/packages/schema/index.js @@ -0,0 +1,480 @@ +/** + * @fileoverview The ProScan cloud schema: types, validators and id builders. + * + * One file shared by the extension (which writes) and the dashboard (which + * reads). It has no imports and no Firebase code, so the dashboard can copy + * it as is; keep the copies byte-identical and bump SV on any change to a + * document's shape. + * + * Conventions: + * - Money is integer cents. + * - Unknown is null, never 0. In compact points (latest, prev, history + * days, page items) an unknown value is left out, so read it as + * `point.p ?? null`. The Firestore rules check `p is int` when present. + * - Every document carries `sv`, the schema version it was written with. + * - Times are Firestore Timestamps in the cloud. Builders and validators + * here take anything `timeMs` understands: a number of ms, a Date, or an + * object with toMillis(). + * + * Paths, all under workspaces/{uid}: + * sources/{sourceId} SourceDoc + * runs/{runId} RunDoc + * runs/{runId}/pages/{pageId} PageDoc + * products/{asin} ProductDoc (extension fields; lead, + * tags and verdict are the dashboard's) + * products/{asin}/history/daily HistoryDoc + * + * @module Schema + */ + +/** Schema version written into every document. */ +export const SV = 1; + +/** The one marketplace so far. */ +export const MK = 'US'; + +export const ASIN_RE = /^[A-Z0-9]{10}$/; +export const DAY_KEY_RE = /^\d{4}-\d{2}-\d{2}$/; + +/** Page chunks expire this long after the run started (TTL policy on expireAt). */ +export const PAGE_TTL_DAYS = 400; + +/** The rules cap a page chunk's items map at this size. */ +export const MAX_PAGE_ITEMS = 120; + +export const RUN_STATUS = ['active', 'complete', 'stopped', 'dead']; +export const SOURCE_TYPES = ['storefront', 'keyword']; + +/** + * @typedef {Object} Point one observation of a product + * @property {number} [p] price, cents + * @property {number} [r] rating, 0 to 5, one decimal + * @property {number} [v] review count + * @property {0|1} [pr] Prime badge + * @property {number} [rk] organic rank within the run, from 1 + * @property {0|1} [sp] sponsored card (page items only) + */ + +/** + * @typedef {Object} Delta change against the previous observation + * @property {number} [p] price change, cents + * @property {number} [pPct] price change, percent of the previous price + * @property {number} [r] rating change + * @property {number} [v] review count change + * @property {number} [days] days between the two observations + */ + +/** + * @typedef {Object} Source + * @property {'storefront'|'keyword'} type + * @property {?string} sellerId + * @property {?string} keyword + * @property {?string} url + */ + +/** + * @typedef {Object} SourceDoc + * @property {number} sv + * @property {string} sourceId + * @property {'storefront'|'keyword'} type + * @property {?string} sellerId + * @property {?string} keyword + * @property {?string} url + * @property {string} lastRunId + * @property {*} lastScrapedAt + * @property {number} [catalogSize] result count Amazon showed, when known + */ + +/** + * @typedef {Object} RunDoc + * @property {number} sv + * @property {string} runId + * @property {string} sourceId + * @property {Source} source + * @property {string} mk + * @property {string} dayKey local date the run started, YYYY-MM-DD + * @property {*} startedAt + * @property {*} finishedAt null while active + * @property {'active'|'complete'|'stopped'|'dead'} status + * @property {?string} reason how it ended, in the extension's words + * @property {number} pagesDone + * @property {number} maxPages + * @property {?number} pagesPlanned known only once a run completes + * @property {?number} totalResultsOnSerp + * @property {{placements:number, uniqueAsins:number, sponsored:number, priceParseFailures:number, newSeen:number}} counters + */ + +/** + * @typedef {Object} PageDoc + * @property {number} sv + * @property {string} runId + * @property {number} page + * @property {*} scrapedAt + * @property {*} expireAt + * @property {number} count products first seen on this page + * @property {number} placements cards on this page + * @property {string} kind results or last + * @property {Object} items every ASIN on the page + * @property {boolean} [truncated] more than MAX_PAGE_ITEMS ASINs + */ + +/** + * @typedef {Object} ProductDoc + * @property {number} sv + * @property {string} asin + * @property {string} mk + * @property {?string} name + * @property {string} url + * @property {?string} img + * @property {Point & {at:*, runId:string, dayKey:string}} latest + * @property {?(Point & {at:*})} prev + * @property {?Delta} delta + * @property {string[]} sourceIds + * @property {*} [firstSeenAt] set once, when the document is created + * @property {string} [firstRunId] + */ + +/** + * @typedef {Object} HistoryDoc + * @property {number} sv + * @property {string} asin + * @property {Object} d dayKey to point + */ + +// ── Ids ───────────────────────────────────────────────────────── + +/** 32-bit FNV-1a of a string, base 36. */ +export function hash(s) { + let h = 0x811c9dc5; + for (const ch of String(s)) { + const c = ch.codePointAt(0); + h ^= c & 0xff; + h = Math.imul(h, 0x01000193); + if (c > 0xff) { + h ^= c >>> 8; + h = Math.imul(h, 0x01000193); + } + } + return (h >>> 0).toString(36); +} + +export function slugify(s) { + return String(s).toLowerCase().trim().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, ''); +} + +const SELLER_RE = /^[A-Z0-9]{6,20}$/; +// Query params that change per visit, not per search. +const NOISE = new Set(['page', 'ref', 'qid', 'xpid', 'crid', 'sprefix', 'dib', 'dib_tag', 'sr', 'ds', 'pd_rd_r', 'pf_rd_r']); + +/** The seller in a search URL: me=, seller=, or p_6:/me: inside rh=. */ +function sellerIn(params) { + const direct = params.get('me') || params.get('seller'); + if (direct) return direct.toUpperCase(); + const m = (params.get('rh') || '').match(/(?:^|,)(?:p_6|me):([A-Za-z0-9]+)/); + return m ? m[1].toUpperCase() : null; +} + +/** + * What an Amazon search URL scans. + * @param {string} url + * @returns {Source} + */ +export function sourceOf(url) { + let params = null; + try { + params = new URL(url).searchParams; + } catch (e) { + return { type: 'keyword', sellerId: null, keyword: null, url: url || null }; + } + const seller = sellerIn(params); + if (seller && SELLER_RE.test(seller)) { + return { type: 'storefront', sellerId: seller, keyword: params.get('k') || null, url }; + } + const keyword = params.get('k') || params.get('field-keywords') || null; + return { type: 'keyword', sellerId: null, keyword: keyword ? keyword.trim() : null, url }; +} + +/** + * The source id: s_{sellerId} for a storefront, k_{slug} for a keyword. + * A keyword with non-ASCII letters, or one too long, keeps what it can of + * the slug and adds a hash, so two keywords never share an id. A search + * with neither gets k_x{hash} of its query. + * @param {Source} source + * @returns {string} + */ +export function sourceIdOf(source) { + if (source && source.type === 'storefront' && source.sellerId) return `s_${source.sellerId}`; + const kw = source && source.keyword ? String(source.keyword).normalize('NFC').trim().toLowerCase() : ''; + if (kw) { + const slug = slugify(kw); + // eslint-disable-next-line no-control-regex + const plain = !/[^\x00-\x7f]/.test(kw) && slug.length > 0 && slug.length <= 80; + return plain ? `k_${slug}` : `k_${slug ? slug.slice(0, 60) + '-' : ''}h${hash(kw)}`; + } + let query = ''; + try { + const u = new URL(source.url); + query = [...u.searchParams.entries()] + .filter(([k]) => !NOISE.has(k)) + .map(([k, v]) => `${k}=${v}`) + .sort() + .join('&'); + query = u.pathname + '?' + query; + } catch (e) { /* no URL */ } + return query ? `k_x${hash(query)}` : 'k_unknown'; +} + +/** Run id: {sourceId}_{startMs}, minted once when the run starts. */ +export function runIdOf(sourceId, startMs) { + return `${sourceId}_${startMs}`; +} + +/** Page chunk id: p0001 for page 1. */ +export function pageIdOf(page) { + return 'p' + String(page).padStart(4, '0'); +} + +/** + * The local date at `ms` as YYYY-MM-DD. `tzOffsetMin` is what + * Date#getTimezoneOffset gives where the scan ran (minutes behind UTC), so + * an evening scan in the US files under its own day. + */ +export function dayKeyOf(ms, tzOffsetMin = 0) { + return new Date(ms - tzOffsetMin * 60000).toISOString().slice(0, 10); +} + +export function expireAtMs(startMs) { + return startMs + PAGE_TTL_DAYS * 86400000; +} + +/** + * The cloud status for an extension run: active while it runs, then + * complete, stopped (by the user or a block page) or dead. + */ +export function runStatusOf(state) { + switch (state) { + case 'starting': + case 'running': + case 'stopping': + return 'active'; + case 'done': + return 'complete'; + case 'stopped': + case 'blocked': + return 'stopped'; + default: + return 'dead'; + } +} + +// ── Values ────────────────────────────────────────────────────── + +/** Milliseconds from a number, Date, ISO string or Timestamp; null otherwise. */ +export function timeMs(v) { + if (typeof v === 'number') return Number.isFinite(v) ? v : null; + if (v instanceof Date) return Number.isNaN(v.getTime()) ? null : v.getTime(); + if (typeof v === 'string') { + const t = Date.parse(v); + return Number.isNaN(t) ? null : t; + } + if (v && typeof v.toMillis === 'function') return v.toMillis(); + return null; +} + +const isInt = (v) => Number.isInteger(v); +const round1 = (n) => Math.round(n * 10) / 10; + +/** + * A compact point from an extension product record, leaving out what is + * unknown. + * @param {{priceCents?:?number, rating?:?number, reviewCount?:?number, isPrime?:boolean, organicRank?:?number}} rec + * @returns {Point} + */ +export function pointOf(rec) { + const pt = {}; + if (isInt(rec.priceCents) && rec.priceCents >= 0) pt.p = rec.priceCents; + if (typeof rec.rating === 'number' && rec.rating > 0 && rec.rating <= 5) pt.r = round1(rec.rating); + if (isInt(rec.reviewCount) && rec.reviewCount >= 0) pt.v = rec.reviewCount; + if (typeof rec.isPrime === 'boolean') pt.pr = rec.isPrime ? 1 : 0; + if (isInt(rec.organicRank) && rec.organicRank > 0) pt.rk = rec.organicRank; + return pt; +} + +/** + * The delta between two points, or null when nothing can be compared. + * @param {Point} now + * @param {?Point} prev + * @param {?number} [days] + * @returns {?Delta} + */ +export function deltaOf(now, prev, days = null) { + if (!prev) return null; + const d = {}; + if (isInt(now.p) && isInt(prev.p)) { + d.p = now.p - prev.p; + if (prev.p > 0) d.pPct = round1((d.p / prev.p) * 100); + } + if (typeof now.r === 'number' && typeof prev.r === 'number') d.r = round1(now.r - prev.r); + if (isInt(now.v) && isInt(prev.v)) d.v = now.v - prev.v; + if (Object.keys(d).length === 0) return null; + if (isInt(days) && days >= 0) d.days = days; + return d; +} + +// ── Validators ────────────────────────────────────────────────── +// Each returns a list of problems; empty means valid. + +const isStr = (v) => typeof v === 'string'; +const isStrOrNull = (v) => v === null || isStr(v); +const isTime = (v) => timeMs(v) !== null && typeof v !== 'string'; + +function checkPoint(pt, where, errs, { sp = false } = {}) { + if (!pt || typeof pt !== 'object') { errs.push(`${where}: not a map`); return; } + for (const [k, v] of Object.entries(pt)) { + if (v === null || v === undefined) errs.push(`${where}.${k}: unknown values are left out, not null`); + } + if ('p' in pt && !(isInt(pt.p) && pt.p >= 0)) errs.push(`${where}.p: cents must be a whole number`); + if ('r' in pt && !(typeof pt.r === 'number' && pt.r >= 0 && pt.r <= 5)) errs.push(`${where}.r: rating out of range`); + if ('v' in pt && !(isInt(pt.v) && pt.v >= 0)) errs.push(`${where}.v: review count must be a whole number`); + if ('pr' in pt && pt.pr !== 0 && pt.pr !== 1) errs.push(`${where}.pr: must be 0 or 1`); + if ('rk' in pt && !(isInt(pt.rk) && pt.rk > 0)) errs.push(`${where}.rk: rank must be 1 or more`); + if ('sp' in pt && (!sp || (pt.sp !== 0 && pt.sp !== 1))) errs.push(`${where}.sp: only page items carry sp, as 0 or 1`); +} + +function checkSv(doc, errs) { + if (!doc || typeof doc !== 'object') { errs.push('not a document'); return false; } + if (doc.sv !== SV) errs.push(`sv: expected ${SV}`); + return true; +} + +/** @returns {string[]} */ +export function validateSource(doc) { + const errs = []; + if (!checkSv(doc, errs)) return errs; + if (!isStr(doc.sourceId) || !/^[sk]_/.test(doc.sourceId)) errs.push('sourceId: expected s_ or k_'); + if (!SOURCE_TYPES.includes(doc.type)) errs.push('type: storefront or keyword'); + if (!isStrOrNull(doc.sellerId)) errs.push('sellerId: string or null'); + if (!isStrOrNull(doc.keyword)) errs.push('keyword: string or null'); + if (!isStrOrNull(doc.url)) errs.push('url: string or null'); + if (!isStr(doc.lastRunId)) errs.push('lastRunId: string'); + if (!isTime(doc.lastScrapedAt)) errs.push('lastScrapedAt: time'); + if ('catalogSize' in doc && !(isInt(doc.catalogSize) && doc.catalogSize >= 0)) errs.push('catalogSize: whole number'); + return errs; +} + +/** @returns {string[]} */ +export function validateRun(doc) { + const errs = []; + if (!checkSv(doc, errs)) return errs; + if (!isStr(doc.runId)) errs.push('runId: string'); + if (!isStr(doc.sourceId)) errs.push('sourceId: string'); + if (isStr(doc.runId) && isStr(doc.sourceId) && !doc.runId.startsWith(doc.sourceId + '_')) errs.push('runId: must start with sourceId_'); + if (!doc.source || !SOURCE_TYPES.includes(doc.source.type)) errs.push('source.type: storefront or keyword'); + if (doc.mk !== MK) errs.push(`mk: ${MK}`); + if (!isStr(doc.dayKey) || !DAY_KEY_RE.test(doc.dayKey)) errs.push('dayKey: YYYY-MM-DD'); + if (!isTime(doc.startedAt)) errs.push('startedAt: time'); + if (doc.finishedAt !== null && !isTime(doc.finishedAt)) errs.push('finishedAt: time or null'); + if (!RUN_STATUS.includes(doc.status)) errs.push(`status: one of ${RUN_STATUS.join(', ')}`); + if (doc.status === 'active' && doc.finishedAt !== null) errs.push('finishedAt: null while active'); + if (!isStrOrNull(doc.reason)) errs.push('reason: string or null'); + if (!isInt(doc.pagesDone) || doc.pagesDone < 0) errs.push('pagesDone: whole number'); + if (!isInt(doc.maxPages) || doc.maxPages < 1) errs.push('maxPages: whole number'); + if (doc.pagesPlanned !== null && !isInt(doc.pagesPlanned)) errs.push('pagesPlanned: whole number or null'); + if (doc.totalResultsOnSerp !== null && !isInt(doc.totalResultsOnSerp)) errs.push('totalResultsOnSerp: whole number or null'); + const c = doc.counters || {}; + for (const k of ['placements', 'uniqueAsins', 'sponsored', 'priceParseFailures', 'newSeen']) { + if (!isInt(c[k]) || c[k] < 0) errs.push(`counters.${k}: whole number`); + } + return errs; +} + +/** @returns {string[]} */ +export function validatePage(doc) { + const errs = []; + if (!checkSv(doc, errs)) return errs; + if (!isStr(doc.runId)) errs.push('runId: string'); + if (!isInt(doc.page) || doc.page < 1) errs.push('page: 1 or more'); + if (!isTime(doc.scrapedAt)) errs.push('scrapedAt: time'); + if (!isTime(doc.expireAt)) errs.push('expireAt: time'); + if (!isInt(doc.count) || doc.count < 0) errs.push('count: whole number'); + if (!isInt(doc.placements) || doc.placements < 0) errs.push('placements: whole number'); + if (!doc.items || typeof doc.items !== 'object') { + errs.push('items: map'); + } else { + const keys = Object.keys(doc.items); + if (keys.length > MAX_PAGE_ITEMS) errs.push(`items: at most ${MAX_PAGE_ITEMS}`); + for (const asin of keys) { + if (!ASIN_RE.test(asin)) errs.push(`items.${asin}: not an ASIN`); + checkPoint(doc.items[asin], `items.${asin}`, errs, { sp: true }); + } + } + return errs; +} + +/** @returns {string[]} */ +export function validateProduct(doc) { + const errs = []; + if (!checkSv(doc, errs)) return errs; + if (!isStr(doc.asin) || !ASIN_RE.test(doc.asin)) errs.push('asin: 10 letters and digits'); + if (doc.mk !== MK) errs.push(`mk: ${MK}`); + if ('name' in doc && !isStrOrNull(doc.name)) errs.push('name: string or null'); + if ('img' in doc && doc.img !== null && !(isStr(doc.img) && /^https:\/\//.test(doc.img))) errs.push('img: https URL or null'); + if (!isStr(doc.url)) errs.push('url: string'); + if (!doc.latest) { + errs.push('latest: map'); + } else { + const { at, runId, dayKey, ...pt } = doc.latest; + checkPoint(pt, 'latest', errs); + if (!isTime(at)) errs.push('latest.at: time'); + if (!isStr(runId)) errs.push('latest.runId: string'); + if (!isStr(dayKey) || !DAY_KEY_RE.test(dayKey)) errs.push('latest.dayKey: YYYY-MM-DD'); + } + if (doc.prev !== null && doc.prev !== undefined) { + const { at, ...pt } = doc.prev; + checkPoint(pt, 'prev', errs); + if (at !== undefined && !isTime(at)) errs.push('prev.at: time'); + } + if (doc.delta !== null && doc.delta !== undefined) { + const d = doc.delta; + if ('p' in d && !isInt(d.p)) errs.push('delta.p: cents must be a whole number'); + if ('pPct' in d && typeof d.pPct !== 'number') errs.push('delta.pPct: number'); + if ('r' in d && typeof d.r !== 'number') errs.push('delta.r: number'); + if ('v' in d && !isInt(d.v)) errs.push('delta.v: whole number'); + if ('days' in d && !isInt(d.days)) errs.push('delta.days: whole number'); + for (const [k, v] of Object.entries(d)) if (v === null) errs.push(`delta.${k}: left out, not null`); + } + if ('sourceIds' in doc && !(Array.isArray(doc.sourceIds) && doc.sourceIds.every(isStr))) errs.push('sourceIds: list of strings'); + if ('firstSeenAt' in doc && !isTime(doc.firstSeenAt)) errs.push('firstSeenAt: time'); + return errs; +} + +/** @returns {string[]} */ +export function validateHistory(doc) { + const errs = []; + if (!checkSv(doc, errs)) return errs; + if (!isStr(doc.asin) || !ASIN_RE.test(doc.asin)) errs.push('asin: 10 letters and digits'); + if (!doc.d || typeof doc.d !== 'object') { + errs.push('d: map'); + } else { + for (const [day, pt] of Object.entries(doc.d)) { + if (!DAY_KEY_RE.test(day)) errs.push(`d.${day}: not a day key`); + checkPoint(pt, `d.${day}`, errs); + } + } + return errs; +} + +const VALIDATORS = { + source: validateSource, + run: validateRun, + page: validatePage, + product: validateProduct, + history: validateHistory +}; + +/** Throws when `doc` is not a valid document of `kind`. */ +export function assertValid(kind, doc) { + const errs = VALIDATORS[kind](doc); + if (errs.length) throw new Error(`invalid ${kind} document: ${errs.join('; ')}`); + return doc; +} diff --git a/tests/setup/esm-to-cjs-transform.js b/tests/setup/esm-to-cjs-transform.js index 582d0eb..b2ebbf5 100644 --- a/tests/setup/esm-to-cjs-transform.js +++ b/tests/setup/esm-to-cjs-transform.js @@ -15,6 +15,7 @@ * import x from 'mod'; -> const x = require('mod'); * export async function f() {} -> async function f() {}; module.exports.f = f; * export function f() {} -> function f() {}; module.exports.f = f; + * export const X = ...; -> const X = ...; module.exports.X = X; */ function rewrite(src) { const named = []; @@ -37,6 +38,11 @@ function rewrite(src) { /import\s+([A-Za-z_$][\w$]*)\s+from\s*['"]([^'"]+)['"]\s*;?/g, (_m, name, mod) => `const ${name} = require('${mod}');` ) + // export const NAME = ... -> strip `export`, remember the name + .replace(/^export\s+const\s+([A-Za-z_$][\w$]*)/gm, (_m, name) => { + named.push(name); + return `const ${name}`; + }) // export (async) function name(...) -> strip `export`, remember the name .replace( /export\s+(async\s+)?function\s+([A-Za-z_$][\w$]*)/g, diff --git a/tests/unit/schema.test.js b/tests/unit/schema.test.js new file mode 100644 index 0000000..41c0a87 --- /dev/null +++ b/tests/unit/schema.test.js @@ -0,0 +1,130 @@ +/** + * @jest-environment node + * + * The shared cloud schema: id builders, points, deltas and validators. + */ +const S = require('../../packages/schema/index.js'); + +describe('source ids (F-29d)', () => { + const idOf = (url) => S.sourceIdOf(S.sourceOf(url)); + + test('a storefront is s_{seller}, from me=, seller= or rh', () => { + expect(idOf('https://www.amazon.com/s?me=A3K9XELT4QZ6M2&marketplaceID=ATVPDKIKX0DER')).toBe('s_A3K9XELT4QZ6M2'); + expect(idOf('https://www.amazon.com/s?seller=a3k9xelt4qz6m2')).toBe('s_A3K9XELT4QZ6M2'); + expect(idOf('https://www.amazon.com/s?i=merchant-items&rh=p_6%3AA3K9XELT4QZ6M2')).toBe('s_A3K9XELT4QZ6M2'); + expect(idOf('https://www.amazon.com/s?rh=n%3A1055398%2Cme%3AA3K9XELT4QZ6M2')).toBe('s_A3K9XELT4QZ6M2'); + expect(S.sourceOf('https://www.amazon.com/s?me=A3K9XELT4QZ6M2&k=mug')).toMatchObject({ type: 'storefront', keyword: 'mug' }); + }); + + test('a keyword is k_{slug}, the same for the same search', () => { + expect(idOf('https://www.amazon.com/s?k=Yoga+Mat&page=2')).toBe('k_yoga-mat'); + expect(idOf('https://www.amazon.com/s?k=yoga%20mat&ref=sr_pg_3')).toBe('k_yoga-mat'); + }); + + test('non-ASCII and long keywords get a hash, so they never collide', () => { + const a = idOf('https://www.amazon.com/s?k=%E6%9D%AF%E5%AD%90'); + const b = idOf('https://www.amazon.com/s?k=%E7%A2%97'); + expect(a).toMatch(/^k_h[0-9a-z]+$/); + expect(a).not.toBe(b); + expect(idOf('https://www.amazon.com/s?k=caf%C3%A9+mug')).toMatch(/^k_caf-mug-h[0-9a-z]+$/); + const long = idOf('https://www.amazon.com/s?k=' + 'a'.repeat(90)); + expect(long.length).toBeLessThan(80); + expect(long).not.toBe(idOf('https://www.amazon.com/s?k=' + 'a'.repeat(91))); + }); + + test('a search with neither gets an id from its query, not a shared k_unknown', () => { + const a = idOf('https://www.amazon.com/s?i=kitchen&rh=n%3A289814&page=2&qid=1'); + expect(a).toMatch(/^k_x[0-9a-z]+$/); + expect(a).toBe(idOf('https://www.amazon.com/s?rh=n%3A289814&i=kitchen&qid=9')); + expect(a).not.toBe(idOf('https://www.amazon.com/s?i=kitchen&rh=n%3A289815')); + expect(S.sourceIdOf(S.sourceOf('not a url'))).toBe('k_unknown'); + }); + + test('run and page ids', () => { + expect(S.runIdOf('k_mug', 1749477731000)).toBe('k_mug_1749477731000'); + expect(S.pageIdOf(3)).toBe('p0003'); + }); +}); + +test('the day key is the local date of the scan', () => { + const ms = Date.parse('2026-06-10T02:30:00Z'); + expect(S.dayKeyOf(ms, 0)).toBe('2026-06-10'); + // 10:30 pm in New York (UTC-4) is still June 9 there + expect(S.dayKeyOf(ms, 240)).toBe('2026-06-09'); +}); + +test('run status from the extension state', () => { + expect(S.runStatusOf('running')).toBe('active'); + expect(S.runStatusOf('done')).toBe('complete'); + expect(S.runStatusOf('stopped')).toBe('stopped'); + expect(S.runStatusOf('blocked')).toBe('stopped'); + expect(S.runStatusOf('failed')).toBe('dead'); +}); + +describe('points and deltas', () => { + test('unknown values are left out, never 0 (F-28)', () => { + expect(S.pointOf({ priceCents: null, rating: null, reviewCount: null, isPrime: false, organicRank: null })).toEqual({ pr: 0 }); + expect(S.pointOf({ priceCents: 1999, rating: 4.56, reviewCount: 0, isPrime: true, organicRank: 3 })) + .toEqual({ p: 1999, r: 4.6, v: 0, pr: 1, rk: 3 }); + }); + + test('a delta compares only what both sides know, and is null otherwise (F-21)', () => { + expect(S.deltaOf({ p: 900, r: 4.5 }, { p: 1000, r: 4.4, v: 10 }, 7)).toEqual({ p: -100, pPct: -10, r: 0.1, days: 7 }); + expect(S.deltaOf({ r: 4.5 }, { p: 1000 })).toBeNull(); + expect(S.deltaOf({ p: 900 }, null)).toBeNull(); + expect(S.deltaOf({ p: 900 }, { p: 0 })).toEqual({ p: 900 }); + }); + + test('time from numbers, dates, ISO strings and Timestamps', () => { + expect(S.timeMs(5)).toBe(5); + expect(S.timeMs(new Date(7))).toBe(7); + expect(S.timeMs('1970-01-01T00:00:00.009Z')).toBe(9); + expect(S.timeMs({ toMillis: () => 11 })).toBe(11); + expect(S.timeMs('nope')).toBeNull(); + }); +}); + +describe('validators', () => { + const run = { + sv: S.SV, runId: 'k_mug_1', sourceId: 'k_mug', + source: { type: 'keyword', sellerId: null, keyword: 'mug', url: 'https://www.amazon.com/s?k=mug' }, + mk: 'US', dayKey: '2026-06-09', startedAt: 1, finishedAt: null, status: 'active', reason: null, + pagesDone: 1, maxPages: 20, pagesPlanned: null, totalResultsOnSerp: 412, + counters: { placements: 3, uniqueAsins: 2, sponsored: 1, priceParseFailures: 0, newSeen: 2 }, + }; + const product = { + sv: S.SV, asin: 'B0C8XL4N2P', mk: 'US', name: 'Mug', url: 'https://www.amazon.com/dp/B0C8XL4N2P', img: null, + latest: { p: 2399, pr: 1, at: 5, runId: 'k_mug_1', dayKey: '2026-06-09' }, + prev: null, delta: null, sourceIds: ['k_mug'], + }; + + test('good documents pass', () => { + expect(S.validateRun(run)).toEqual([]); + expect(S.validateProduct(product)).toEqual([]); + expect(S.validateSource({ sv: S.SV, sourceId: 'k_mug', type: 'keyword', sellerId: null, keyword: 'mug', url: null, lastRunId: 'k_mug_1', lastScrapedAt: 1 })).toEqual([]); + expect(S.validatePage({ sv: S.SV, runId: 'k_mug_1', page: 1, scrapedAt: 1, expireAt: 2, count: 1, placements: 1, kind: 'last', items: { B0C8XL4N2P: { p: 1, sp: 0 } } })).toEqual([]); + expect(S.validateHistory({ sv: S.SV, asin: 'B0C8XL4N2P', d: { '2026-06-09': { p: 1 } } })).toEqual([]); + }); + + test('every document carries sv', () => { + expect(S.validateRun({ ...run, sv: undefined })).toContain(`sv: expected ${S.SV}`); + }); + + test('money must be whole cents and unknowns must be left out', () => { + const bad = { ...product, latest: { ...product.latest, p: 23.99, r: null } }; + const errs = S.validateProduct(bad); + expect(errs).toContain('latest.p: cents must be a whole number'); + expect(errs).toContain('latest.r: unknown values are left out, not null'); + expect(() => S.assertValid('product', bad)).toThrow(/invalid product document/); + }); + + test('the ASIN, status and the page size limit are checked', () => { + expect(S.validateProduct({ ...product, asin: 'b0-bad' })).toContain('asin: 10 letters and digits'); + expect(S.validateRun({ ...run, status: 'blocked' })[0]).toMatch(/^status/); + expect(S.validateRun({ ...run, finishedAt: 9 })).toContain('finishedAt: null while active'); + const items = {}; + for (let i = 0; i <= S.MAX_PAGE_ITEMS; i++) items[`B${String(i).padStart(9, '0')}`] = { p: 1 }; + expect(S.validatePage({ sv: S.SV, runId: 'r', page: 1, scrapedAt: 1, expireAt: 2, count: 0, placements: 0, items })) + .toContain(`items: at most ${S.MAX_PAGE_ITEMS}`); + }); +}); From 276c9f8e8850de51dc2dee7f2d019930afb1334a Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:25:30 -0400 Subject: [PATCH 02/18] Read each card's product image from Amazon's image host --- scripts/lib/parsers.js | 10 +++ .../__snapshots__/characterize.test.js.snap | 69 +++++++++++++++++++ tests/unit/parsers.test.js | 10 ++- 3 files changed, 88 insertions(+), 1 deletion(-) diff --git a/scripts/lib/parsers.js b/scripts/lib/parsers.js index 5a524c2..d242d4b 100644 --- a/scripts/lib/parsers.js +++ b/scripts/lib/parsers.js @@ -47,6 +47,7 @@ const Parsers = (() => { productLink: '.a-link-normal.s-no-outline', sponsored: '.puis-sponsored-label-text, .s-sponsored-label-text, [data-component-type="sp-sponsored-result"], a[href*="/sspa/"]', primeBadge: '.a-icon-prime, .s-prime', + image: 'img.s-image', nextPage: '.s-pagination-next:not(.s-pagination-disabled)', nextPageLink: 'a.s-pagination-next[href]:not(.s-pagination-disabled)', nextPageDisabled: '.s-pagination-next.s-pagination-disabled', @@ -188,6 +189,13 @@ const Parsers = (() => { element.querySelector(SEARCH_SELECTORS.sponsored) !== null; } + /** The card's product image on Amazon's image host, or null. */ + function extractImage(element) { + const img = element.querySelector(SEARCH_SELECTORS.image); + const src = img ? img.getAttribute('src') : null; + return src && /^https:\/\/(m\.media-amazon\.com|images-na\.ssl-images-amazon\.com)\//.test(src) ? src : null; + } + /** The product page for an ASIN. Card links can be expiring sspa ad redirects. */ function productUrl(asin) { return `https://www.amazon.com/dp/${asin}`; @@ -219,6 +227,7 @@ const Parsers = (() => { isPrime: isPrime, sponsored: isSponsored(listing), url: productUrl(asin), + img: extractImage(listing), scrapedAt: new Date().toISOString() }; } @@ -476,6 +485,7 @@ const Parsers = (() => { hasPrimeBadge, isSponsored, productUrl, + extractImage, scrapeProduct, getTotalResults, getNextPageUrl, diff --git a/tests/golden/__snapshots__/characterize.test.js.snap b/tests/golden/__snapshots__/characterize.test.js.snap index bb09899..8ca32f9 100644 --- a/tests/golden/__snapshots__/characterize.test.js.snap +++ b/tests/golden/__snapshots__/characterize.test.js.snap @@ -281,6 +281,7 @@ exports[`current scraper behavior on the corpus 2026-02/search-lastpage 1`] = ` "dReviews": null, "isNew": true, }, + "img": null, "isPrime": false, "name": "Last Page Product", "organicRank": 1, @@ -342,6 +343,7 @@ exports[`current scraper behavior on the corpus 2026-02/search-results 1`] = ` "dReviews": null, "isNew": true, }, + "img": null, "isPrime": true, "name": "Test Widget Pro 2000", "organicRank": 1, @@ -372,6 +374,7 @@ exports[`current scraper behavior on the corpus 2026-02/search-results 1`] = ` "dReviews": null, "isNew": true, }, + "img": null, "isPrime": false, "name": "Budget Gadget Basic", "organicRank": 2, @@ -402,6 +405,7 @@ exports[`current scraper behavior on the corpus 2026-02/search-results 1`] = ` "dReviews": null, "isNew": true, }, + "img": null, "isPrime": false, "name": "Mystery Item No Data", "organicRank": 3, @@ -432,6 +436,7 @@ exports[`current scraper behavior on the corpus 2026-02/search-results 1`] = ` "dReviews": null, "isNew": true, }, + "img": null, "isPrime": true, "name": "Legacy Star Product", "organicRank": 4, @@ -575,6 +580,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-no-pagination-syn "dReviews": null, "isNew": true, }, + "img": null, "isPrime": false, "name": "Rare Widget, single result", "organicRank": 1, @@ -605,6 +611,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-no-pagination-syn "dReviews": null, "isNew": true, }, + "img": null, "isPrime": false, "name": "Rare Widget, second result", "organicRank": 2, @@ -666,6 +673,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-title-recipe-synt "dReviews": null, "isNew": true, }, + "img": null, "isPrime": false, "name": "Acme Widget Pro, 6 ft, Braided", "organicRank": 3, @@ -702,6 +710,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-title-recipe-synt "dReviews": null, "isNew": true, }, + "img": null, "isPrime": false, "name": "Widget Basic, 3 ft", "organicRank": 1, @@ -732,6 +741,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-title-recipe-synt "dReviews": null, "isNew": true, }, + "img": null, "isPrime": false, "name": "Bulk pack of widgets, 24 count", "organicRank": 2, @@ -793,6 +803,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81s9TLcupfL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Thick Yoga Mat, Exercise & Fitness Mat with Easy-Cinch Carrying Strap", "organicRank": null, @@ -823,6 +834,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61JebJK2odL._AC_UL320_.jpg", "isPrime": false, "name": "Retrospec Solana Yoga Mat Thick 1/2in Non-Slip Workout Mat with Nylon Strap - 72x24in Exercise Mat for Pilates, Stretching & Fitness - BPA Free, Easy Clean", "organicRank": 10, @@ -859,6 +871,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61v1tPJufHL._AC_UL320_.jpg", "isPrime": false, "name": "Yoga Mat 72" x 24" x 6mm – Non Slip Exercise Mat", "organicRank": null, @@ -889,6 +902,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71aaK-w9tbL._AC_UL320_.jpg", "isPrime": false, "name": "Retrospec Solana Yoga Mat 1" Thick w/Nylon Strap for Men & Women - Non Slip Exercise Mat for Home Yoga, Pilates, Stretching, Floor & Fitness Workouts", "organicRank": null, @@ -919,6 +933,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61KZlPKYscL._AC_UL320_.jpg", "isPrime": false, "name": "Gruper Yoga Mat Non Slip, Eco Friendly Exercise Mat with Carrying Strap", "organicRank": 1, @@ -949,6 +964,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71Dw6U5ZNVL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Thick Yoga Mat, Exercise & Fitness Mat with Easy-Cinch Carrying Strap", "organicRank": 2, @@ -979,6 +995,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/611L9oYa0DL._AC_UL320_.jpg", "isPrime": false, "name": "Yoga Knee Pad Cushion, 0.6 Inch (15mm) Mini Yoga Mat for Home Workout", "organicRank": 3, @@ -1009,6 +1026,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71U3oP3IqsL._AC_UL320_.jpg", "isPrime": false, "name": "Amazon Basics Extra Thick Exercise Yoga Mat with Carrying Strap", "organicRank": 4, @@ -1039,6 +1057,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81mao7QZ1iL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Thick Yoga Mat, Exercise & Fitness Mat with Easy-Cinch Carrying Strap", "organicRank": 5, @@ -1069,6 +1088,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81gN-e265yL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Thick Yoga Mat, Exercise & Fitness Mat with Easy-Cinch Carrying Strap", "organicRank": 6, @@ -1099,6 +1119,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71aAvY+4ZqL._AC_UL320_.jpg", "isPrime": false, "name": "HAPBEAR Foldable Yoga Mat with Towel Set - 72"x24"x0.24" (6mm), Non-Slip TPE Exercise Mat with Absorbent Yoga Towel for Home Workout, Yoga, Pilates, Stretching & Travel, Durable & Eco-Friendly, Includes Carry Bag", "organicRank": null, @@ -1129,6 +1150,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81chpv2CQLL._AC_UL320_.jpg", "isPrime": false, "name": "COOLMOON Yoga Mat Non Slip, Anti-Tear 1/4 Thick TPE Yoga Mats for Women and Men, 72"x24" Exercise & Fitness Mat with Carrying Strap, Workout Mats for Yoga, Pilates and Floor Exercise", "organicRank": null, @@ -1159,6 +1181,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/710jyzqZ06L._AC_UL320_.jpg", "isPrime": false, "name": "PAIDU Foldable Yoga Mat, 70"x24"x0.31" TPE Non-Slip Mat with Bag", "organicRank": null, @@ -1189,6 +1212,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61HbxvB0gXL._AC_UL320_.jpg", "isPrime": false, "name": "PAIDU Foldable Yoga Mat 10mm Thick, 75"x26" Non-Slip Fitness Exercise Mat", "organicRank": null, @@ -1219,6 +1243,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61wBZtl1D5L._AC_UL320_.jpg", "isPrime": false, "name": "6.2x3ft Folding Gym Mat, 3" Thick", "organicRank": 7, @@ -1249,6 +1274,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71MQ8weHJOL._AC_UL320_.jpg", "isPrime": false, "name": "Gruper Yoga Mat Non Slip, Eco Friendly Exercise Mat with Carrying Strap", "organicRank": 8, @@ -1279,6 +1305,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81oQh5H8mWL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Yoga Mat - Premium 6mm Print Reversible Extra Thick Non Slip Exercise & Fitness Mat for All Types of Yoga, Pilates & Floor Workouts (68" x 24" x 6mm Thick)", "organicRank": 9, @@ -1309,6 +1336,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/41J4tyGjZzL._AC_UL320_.jpg", "isPrime": false, "name": "Stakt Foldable Yoga Mat Pro - Non Slip 2 in 1 Fitness Exercise Mat", "organicRank": null, @@ -1339,6 +1367,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61evs7yyd4L._AC_UL320_.jpg", "isPrime": false, "name": "Foldable Yoga Mat Extra Large 74"×31.5"", "organicRank": null, @@ -1369,6 +1398,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/712NJegtSpL._AC_UL320_.jpg", "isPrime": false, "name": "Yoga Mat, 80cm Wide, 1/3 Inch Thick Exercise Mat, Carry Strap, Fitness", "organicRank": null, @@ -1399,6 +1429,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/614eJq9mV0L._AC_UL320_.jpg", "isPrime": false, "name": "KEEP 72" x 24" Yoga Mat, 7mm Thick Non Slip TPE Mat", "organicRank": null, @@ -1429,6 +1460,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71f4oCDw3KL._AC_UL320_.jpg", "isPrime": false, "name": "CAP Barbell 1/2-Inch High Density Exercise Yoga Mat with Strap | Multiple Options", "organicRank": 11, @@ -1459,6 +1491,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71qgt7f5j5L._AC_UL320_.jpg", "isPrime": false, "name": "CAP Barbell 1/2-Inch High Density Exercise Yoga Mat with Strap | Multiple Options", "organicRank": 12, @@ -1489,6 +1522,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61KH78eW9lL._AC_UL320_.jpg", "isPrime": false, "name": "KEEP Extra Wide Yoga Mat, 7mm Thick 72" x 32" Exercise Mat", "organicRank": 13, @@ -1519,6 +1553,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/713U7n-k-3L._AC_UL320_.jpg", "isPrime": false, "name": "18 Pcs EVA Foam Floor Tiles, Puzzle Exercise Mat Set for Gym", "organicRank": 14, @@ -1549,6 +1584,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61T8eKzs2VL._AC_UL320_.jpg", "isPrime": false, "name": "Retrospec Solana Yoga Mat 1" Thick w/Nylon Strap for Men & Women - Non Slip Exercise Mat for Home Yoga, Pilates, Stretching, Floor & Fitness Workouts", "organicRank": 15, @@ -1579,6 +1615,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71v1o-kvwjL._AC_UL320_.jpg", "isPrime": false, "name": "CAP Barbell Folding Exercise Mat – Durable, Anti-Tear, Thick Padding for Fitness, Aerobics, Gymnastics & Home Workouts. 72"L x 24"W x 2"Thick. BLACK", "organicRank": 16, @@ -1609,6 +1646,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71iqfgjEfGL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Thick Yoga Mat, Exercise & Fitness Mat with Easy-Cinch Carrying Strap", "organicRank": 17, @@ -1639,6 +1677,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71pBRbyl-TL._AC_UL320_.jpg", "isPrime": false, "name": "Gruper Yoga Mat Non Slip, Anti-Tear Workout Mat with Carrying Strap and Bag", "organicRank": 18, @@ -1669,6 +1708,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/7171WaS6I3L._AC_UL320_.jpg", "isPrime": false, "name": "Fitvids All Purpose 1/4-Inch High Density Anti-Tear Exercise Yoga Mat with Carrying Strap with Optional Yoga Blocks, Multiple Colors", "organicRank": 19, @@ -1699,6 +1739,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81YjZmUPryL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Yoga Mat - Premium 5mm Solid Thick Non Slip Exercise & Fitness Mat for All Types of Yoga, Pilates & Floor Workouts (68" x 24" x 5mm)", "organicRank": 20, @@ -1729,6 +1770,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81U56-TDp+L._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Premium Yoga Mat, 6mm Thick Print Mat", "organicRank": 21, @@ -1759,6 +1801,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/811FZqWCxIL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Premium Yoga Mat, 6mm Thick Print Mat", "organicRank": 22, @@ -1789,6 +1832,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71jk4ZHwsdL._AC_UL320_.jpg", "isPrime": false, "name": "CAP Barbell 1/2-Inch High Density Exercise Yoga Mat with Strap | Multiple Options", "organicRank": 23, @@ -1819,6 +1863,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61I+rnqoMcL._AC_UL320_.jpg", "isPrime": false, "name": "Retrospec Solana Yoga Mat Thick 1/2in Non-Slip Workout Mat with Nylon Strap - 72x24in Exercise Mat for Pilates, Stretching & Fitness - BPA Free, Easy Clean", "organicRank": 24, @@ -1849,6 +1894,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/7156TmPL+sL._AC_UL320_.jpg", "isPrime": false, "name": "KEEP Extra Thick NBR Yoga Mat for Women & Men, 10mm Thick, 72”x32” Large Exercise Mat with Non-Slip Workout Mat for Yoga, Pilates, Stretching, Meditation – Wide & Cushioned Fitness Mat", "organicRank": 25, @@ -1879,6 +1925,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81T1NUS+UwL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Yoga Mat Classic Print, 4mm Non-Slip Mat", "organicRank": 26, @@ -1909,6 +1956,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/91WnKjpclwL._AC_UL320_.jpg", "isPrime": false, "name": "Manduka PRO Yoga Mat - 6mm | Lifetime Durability | Hygienic Construction | Premium Studio Quality | Teacher Approved", "organicRank": 27, @@ -1939,6 +1987,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/819cPiAoMuL._AC_UL320_.jpg", "isPrime": false, "name": "YOGATI® Yoga Mat with Strap with Alignment Lines. Home Workout Mat for Women, Men and Kids. Thick Non Slip Yoga Mat for Pilates and Fitness. Brown, Pink and Purple Yoga Mats", "organicRank": 28, @@ -1969,6 +2018,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/711NCt1h9zL._AC_UL320_.jpg", "isPrime": false, "name": "Voyage 5mm Studio Yoga Mats, 6 Pack, 72x24in", "organicRank": 29, @@ -1999,6 +2049,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81YdT2ovoPL._AC_UL320_.jpg", "isPrime": false, "name": "Economy 3mm Bulk Yoga Mats, 10 Pack, 68x24in", "organicRank": 30, @@ -2029,6 +2080,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/91hz7hERYjS._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Yoga Mat - Premium 5mm Solid Thick Non Slip Exercise & Fitness Mat for All Types of Yoga, Pilates & Floor Workouts (68" x 24" x 5mm)", "organicRank": 31, @@ -2059,6 +2111,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/91UdCyPWQLL._AC_UL320_.jpg", "isPrime": false, "name": "Manduka PROlite Yoga Mat 4.7mm | Lifetime Durability | Hygienic Construction | Teacher Approved", "organicRank": 32, @@ -2089,6 +2142,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61WrjbRYC3L._AC_UL320_.jpg", "isPrime": false, "name": "Amazon Basics Extra Thick Exercise Yoga Mat with Carrying Strap", "organicRank": 33, @@ -2119,6 +2173,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71WJvgWQy-L._AC_UL320_.jpg", "isPrime": false, "name": "Fitvids 1/2-inch Thick High Density Exercise Yoga Mat with Carrying Strap", "organicRank": 34, @@ -2149,6 +2204,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/61rFL8O9XZS._AC_UL320_.jpg", "isPrime": false, "name": "Manduka PRO Yoga Mat - 6mm | Lifetime Durability | Hygienic Construction | Premium Studio Quality | Teacher Approved", "organicRank": 35, @@ -2179,6 +2235,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81KHLWfRt5L._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Essentials 1/4" Thin (6mm) Yoga & Pilates, Fitness & Exercise Mat with Easy-Cinch Carrier Strap Cusion Support For Fitness and Gym Workouts", "organicRank": 36, @@ -2209,6 +2266,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81PfuUvfvkL._AC_UL320_.jpg", "isPrime": false, "name": "Funtery 3 mm Thin Yoga Mats Bulk, 68 x 24 in", "organicRank": 37, @@ -2239,6 +2297,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71+n2fvgNoL._AC_UL320_.jpg", "isPrime": false, "name": "Sunny Health & Fitness Exercise Yoga Mats - Non-Slip Support for Home & Gym", "organicRank": 38, @@ -2269,6 +2328,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/91TL-Sk5aNL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Premium Yoga Mat, 6mm Thick Print Mat", "organicRank": 39, @@ -2299,6 +2359,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71S+4POpCiL._AC_UL320_.jpg", "isPrime": false, "name": "Cork Yoga Mat with Natural Rubber Base, Extra Size, Thickness and Support, Excellent Cushion & Grip, Non-Slip, Non-Toxic, Sweat-Resistant, Sustainable, Eco-friendly Exercise Mat", "organicRank": 40, @@ -2329,6 +2390,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81xWBR2159L._AC_UL320_.jpg", "isPrime": false, "name": "SPRI Hanging Exercise Mats for Gyms, Closed-Cell Foam Yoga & Fitness Mats with Grommets, Thick Workout Mats for Studios & Group Fitness Classes", "organicRank": 41, @@ -2359,6 +2421,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71EhT3gPpNL._AC_UL320_.jpg", "isPrime": false, "name": "Amazon Basics 1/4 Inch Thick TPE Exercise Yoga Mat with Carrying Strap", "organicRank": 42, @@ -2389,6 +2452,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71zC5MVG2xL._AC_UL320_.jpg", "isPrime": false, "name": "CAMBIVO Extra Long Wide Yoga Mat, Thick PVC Workout Mat", "organicRank": 43, @@ -2419,6 +2483,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81P7dfAuglL._AC_UL320_.jpg", "isPrime": false, "name": "Gaiam Yoga Mat - Premium 5mm Solid Thick Non Slip Exercise & Fitness Mat for All Types of Yoga, Pilates & Floor Workouts (68" x 24" x 5mm)", "organicRank": 44, @@ -2449,6 +2514,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81PK-dlA1aL._AC_UL320_.jpg", "isPrime": false, "name": "Heathyoga Yoga Mat Non Slip TPE Pilates Exercise Mats for Home Workout Gym Fitness Eco Friendly Thick Hot Yoga Mat with Strap Body Alignment,72"x 26" Thickness 1/4"", "organicRank": 45, @@ -2479,6 +2545,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/71YPoM9x53L._AC_UL320_.jpg", "isPrime": false, "name": "ProsourceFit Exercise Balance Pad, Non-Slip Cushioned Foam Mat & Knee Pad for Fitness and Stability Training, Yoga, Physical Therapy", "organicRank": 46, @@ -2509,6 +2576,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/81ZKYeFRa8L._AC_UL320_.jpg", "isPrime": false, "name": "CAP Non-Slip Yoga Mat for Home Workout & Pilates, Lightweight Exercise Fitness Mat with Textured Grip Surface, Portable Roll-Up Yoga Mat for Women & Men", "organicRank": 47, @@ -2539,6 +2607,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "dReviews": null, "isNew": true, }, + "img": "https://m.media-amazon.com/images/I/51M7L1s50VL._AC_UL320_.jpg", "isPrime": false, "name": "Liforme Travel Yoga Mat - Patented Alignment Design, Advanced Non-Slip Grip", "organicRank": 48, diff --git a/tests/unit/parsers.test.js b/tests/unit/parsers.test.js index da52ea3..20fca4c 100644 --- a/tests/unit/parsers.test.js +++ b/tests/unit/parsers.test.js @@ -184,7 +184,15 @@ describe('Parsers.scrapeProduct fields', () => { test('a card with no title, rating or reviews has nulls, not 0 or N/A', () => { const p = Parsers.scrapeProduct(one('
')); - expect(p).toMatchObject({ name: null, price: null, priceCents: null, rating: null, reviewCount: null }); + expect(p).toMatchObject({ name: null, price: null, priceCents: null, rating: null, reviewCount: null, img: null }); + }); + + test('the image comes from Amazon\'s image host only (F-25)', () => { + const img = (src) => `
`; + const good = 'https://m.media-amazon.com/images/I/81s9TLcupfL._AC_UL320_.jpg'; + expect(Parsers.scrapeProduct(one(img(good))).img).toBe(good); + expect(Parsers.scrapeProduct(one(img('data:image/gif;base64,R0lGOD'))).img).toBeNull(); + expect(Parsers.scrapeProduct(one(img('https://evil.example/x.jpg'))).img).toBeNull(); }); }); From 108074c86465c6fc5c05bf4dbaaa9bc36a14c0db Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:27:16 -0400 Subject: [PATCH 03/18] Mint one run id per run and queue pages for the signed-in account only --- scripts/background/engine.js | 97 +++++++++++-------- .../__snapshots__/characterize.test.js.snap | 69 +++++++++++++ tests/unit/engine.test.js | 55 +++++++++-- 3 files changed, 172 insertions(+), 49 deletions(-) diff --git a/scripts/background/engine.js b/scripts/background/engine.js index 329fbfc..58be276 100644 --- a/scripts/background/engine.js +++ b/scripts/background/engine.js @@ -25,6 +25,7 @@ const Run = require('../lib/run.js'); const Msg = require('../lib/messages.js'); const Flags = require('../lib/flags.js'); const Delta = require('../modules/delta.js'); +const Schema = require('../../packages/schema/index.js'); /** A page asked for but not reported this long is asked for again. */ const PAGE_TIMEOUT_MS = 20000; @@ -77,14 +78,26 @@ function durable(run) { return rest; } -function runSuffix(random) { - return Math.floor(random() * 0x100000000).toString(36).padStart(7, '0').slice(0, 8); +/** The signed-in account's uid, as the service worker keeps it in storage.local. */ +async function accountUid(chrome) { + const data = await chrome.storage.local.get('account'); + return (data && data.account && data.account.uid) || null; +} + +/** + * A lastValues snapshot counts for `uid` when this account took it, or when + * it was taken signed out. Another account's snapshot does not. + */ +function prevFor(snap, uid) { + if (!snap) return null; + return !snap.uid || snap.uid === uid ? snap : null; } /** * @param {Object} deps * @param {Object} deps.chrome - chrome.* (storage.session, storage.local, tabs) * @param {function(): Promise} deps.openDb - resolves to the db.js api + * @param {function(): Promise} [deps.owner] - uid of the signed-in account */ function createEngine({ chrome, @@ -94,7 +107,8 @@ function createEngine({ setTimer = setTimeout, clearTimer = clearTimeout, flags = Flags, - log = console + log = console, + owner = () => accountUid(chrome) }) { let dbPromise = null; let timer = null; @@ -152,13 +166,23 @@ function createEngine({ timer = null; } + /** Outbox entries for a signed-in account; none when sync is off or nobody is signed in. */ + async function outbox(runId, kind, extra = {}) { + if (!flags.CLOUD_SYNC) return []; + const uid = await owner(); + if (!uid) return []; + return [{ store: 'outbox', put: { runId, kind, uid, queuedAt: now(), ...extra } }]; + } + /** Ends `run` for `reason` and tells its tab. Returns the ended run. */ async function end(run, reason) { const ended = Run.finish(run, reason, now()); cancelTimer(); await saveRun(ended); try { - await (await db()).write([{ store: 'runs', put: durable(ended) }]); + // A run with saved pages sends its final header to the cloud. + const queued = ended.page > 0 ? await outbox(ended.runId, 'run') : []; + await (await db()).write([{ store: 'runs', put: durable(ended) }, ...queued]); } catch (err) { log.warn('[ProScan] Could not record the end of the run:', err.message); } @@ -212,8 +236,15 @@ function createEngine({ const local = await chrome.storage.local.get('settings'); const settings = (local && local.settings) || {}; const t = now(); - const runId = `${t}-${runSuffix(random)}`; - let run = Run.create({ runId, tabId, maxPages: settings.maxPages, source: Run.sourceOf(pong.url, t), now: t }); + // One id for the run, here and in the cloud: {sourceId}_{startMs}. + const found = Schema.sourceOf(pong.url); + const sourceId = Schema.sourceIdOf(found); + const runId = Schema.runIdOf(sourceId, t); + let run = Run.create({ + runId, tabId, maxPages: settings.maxPages, now: t, + source: { ...found, sourceId, startedAt: new Date(t).toISOString() } + }); + run = { ...run, sourceId, dayKey: Schema.dayKeyOf(t, new Date(t).getTimezoneOffset()) }; await saveRun(run); const first = [ { store: 'runs', put: durable(run) }, @@ -320,16 +351,23 @@ function createEngine({ store = await db(); const existing = await store.runProducts(runId); const { fresh, changed } = foldPage(existing, result.products, page); - const prevs = await store.getMany('lastValues', fresh.map(p => p.asin)); + const snaps = await store.getMany('lastValues', fresh.map(p => p.asin)); + const uid = await owner(); ops = []; fresh.forEach((p, i) => { p.runId = runId; p.pageIndex = page; p.n = existing.length + i; - const prev = prevs[i] || null; + const prev = prevFor(snaps[i], uid); // Only an ASIN's first sighting in the run gets a delta. p.delta = Delta.computeDeltas(p, prev); - ops.push({ store: 'lastValues', put: { asin: p.asin, ...Delta.snapshot(p, prev) } }); + p.prev = prev ? { + priceCents: prev.priceCents ?? null, rating: prev.rating ?? null, + reviewCount: prev.reviewCount ?? null, scrapedAt: prev.scrapedAt || null, uid: prev.uid || null + } : null; + const snap = { asin: p.asin, ...Delta.snapshot(p, prev) }; + if (uid) snap.uid = uid; + ops.push({ store: 'lastValues', put: snap }); ops.push({ store: 'products', put: p }); }); changed.forEach(p => ops.push({ store: 'products', put: p })); @@ -337,17 +375,22 @@ function createEngine({ store: 'placements', put: { runId, pageIndex: page, count: fresh.length, placements: result.placements || 0, - kind: result.kind, fill: result.fill || null, scrapedAt: new Date(t).toISOString(), url + kind: result.kind, fill: result.fill || null, scrapedAt: new Date(t).toISOString(), url, + total: Number.isInteger(result.total) && result.total > 0 ? result.total : null } }); - if (flags.CLOUD_SYNC) ops.push({ store: 'outbox', put: { runId, pageIndex: page, queuedAt: t } }); + ops.push(...await outbox(runId, 'page', { pageIndex: page })); next = { ...run, page, itemCount: run.itemCount + fresh.length, heartbeat: t, lastUrl: url, nextHref: result.nextHref || null, awaiting: false }; - if (ending) next = Run.finish(next, ending, t); - else next.navAt = t + Run.pageDelay(random()); + if (ending) { + next = Run.finish(next, ending, t); + ops.push(...await outbox(runId, 'run')); + } else { + next.navAt = t + Run.pageDelay(random()); + } ops.push({ store: 'runs', put: durable(next) }); await store.write(ops); } catch (err) { @@ -515,34 +558,10 @@ function createEngine({ }; } - /** - * The sync bundle for every run in the outbox, built from the products - * as they are now, and the outbox keys to delete once it is written. - */ - async function outboxBundle() { - const store = await db(); - const entries = await store.getAll('outbox'); - const runIds = [...new Set(entries.map(e => e.runId))]; - const syncQueue = []; - const scrapeRunPages = []; - let scrapeRunMeta = null; - for (const runId of runIds) { - syncQueue.push(...await store.runProducts(runId)); - scrapeRunPages.push(...await store.runPages(runId)); - const rec = await store.get('runs', runId); - if (rec) scrapeRunMeta = rec.source; - } - return { bundle: { syncQueue, scrapeRunMeta, scrapeRunPages }, seqs: entries.map(e => e.seq) }; - } - - async function clearOutbox(seqs) { - await (await db()).write(seqs.map(seq => ({ store: 'outbox', delete: seq }))); - } - return { start, stop, pageReady, pageResult, heartbeat, tick, tabRemoved, tabUpdated, recover, - getState, getResults, spreadResult, chatData, outboxBundle, clearOutbox, db + getState, getResults, spreadResult, chatData, db }; } -module.exports = { createEngine, foldPage, durable, PAGE_TIMEOUT_MS, KEEP_RUNS }; +module.exports = { createEngine, foldPage, durable, prevFor, PAGE_TIMEOUT_MS, KEEP_RUNS }; diff --git a/tests/golden/__snapshots__/characterize.test.js.snap b/tests/golden/__snapshots__/characterize.test.js.snap index 8ca32f9..4080920 100644 --- a/tests/golden/__snapshots__/characterize.test.js.snap +++ b/tests/golden/__snapshots__/characterize.test.js.snap @@ -294,6 +294,7 @@ exports[`current scraper behavior on the corpus 2026-02/search-lastpage 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$19.99", "priceCents": 1999, "rating": 4, @@ -356,6 +357,7 @@ exports[`current scraper behavior on the corpus 2026-02/search-results 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$29.99", "priceCents": 2999, "rating": 4.5, @@ -387,6 +389,7 @@ exports[`current scraper behavior on the corpus 2026-02/search-results 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$9.99", "priceCents": 999, "rating": 3.8, @@ -418,6 +421,7 @@ exports[`current scraper behavior on the corpus 2026-02/search-results 1`] = ` "sponsored": false, }, ], + "prev": null, "price": null, "priceCents": null, "rating": null, @@ -449,6 +453,7 @@ exports[`current scraper behavior on the corpus 2026-02/search-results 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$49.99", "priceCents": 4999, "rating": 4.2, @@ -593,6 +598,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-no-pagination-syn "sponsored": false, }, ], + "prev": null, "price": "$41.00", "priceCents": 4100, "rating": null, @@ -624,6 +630,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-no-pagination-syn "sponsored": false, }, ], + "prev": null, "price": "$1,249.00", "priceCents": 124900, "rating": null, @@ -692,6 +699,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-title-recipe-synt "sponsored": false, }, ], + "prev": null, "price": "$12.99", "priceCents": 1299, "rating": 4.4, @@ -723,6 +731,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-title-recipe-synt "sponsored": false, }, ], + "prev": null, "price": "$7.49", "priceCents": 749, "rating": 4.1, @@ -754,6 +763,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-title-recipe-synt "sponsored": false, }, ], + "prev": null, "price": null, "priceCents": null, "rating": null, @@ -816,6 +826,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$21.23", "priceCents": 2123, "rating": 4.6, @@ -853,6 +864,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$24.99", "priceCents": 2499, "rating": 4.4, @@ -884,6 +896,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$29.99", "priceCents": 2999, "rating": 4.8, @@ -915,6 +928,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$39.99", "priceCents": 3999, "rating": 4.5, @@ -946,6 +960,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$25.19", "priceCents": 2519, "rating": 4.4, @@ -977,6 +992,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$24.98", "priceCents": 2498, "rating": 4.6, @@ -1008,6 +1024,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$8.99", "priceCents": 899, "rating": 4.5, @@ -1039,6 +1056,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": null, "priceCents": null, "rating": 4.6, @@ -1070,6 +1088,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$25.02", "priceCents": 2502, "rating": 4.6, @@ -1101,6 +1120,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$21.11", "priceCents": 2111, "rating": 4.6, @@ -1132,6 +1152,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$39.99", "priceCents": 3999, "rating": 4.7, @@ -1163,6 +1184,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$36.99", "priceCents": 3699, "rating": 4.2, @@ -1194,6 +1216,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$42.65", "priceCents": 4265, "rating": 4.4, @@ -1225,6 +1248,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$51.49", "priceCents": 5149, "rating": 4.3, @@ -1256,6 +1280,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$59.99", "priceCents": 5999, "rating": 5, @@ -1287,6 +1312,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$28.16", "priceCents": 2816, "rating": 4.4, @@ -1318,6 +1344,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$22.75", "priceCents": 2275, "rating": 4.6, @@ -1349,6 +1376,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$128.00", "priceCents": 12800, "rating": 4.2, @@ -1380,6 +1408,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$45.99", "priceCents": 4599, "rating": 4.4, @@ -1411,6 +1440,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$29.99", "priceCents": 2999, "rating": 4.5, @@ -1442,6 +1472,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": true, }, ], + "prev": null, "price": "$24.99", "priceCents": 2499, "rating": 4.3, @@ -1473,6 +1504,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$15.99", "priceCents": 1599, "rating": 4.6, @@ -1504,6 +1536,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$15.99", "priceCents": 1599, "rating": 4.6, @@ -1535,6 +1568,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$29.98", "priceCents": 2998, "rating": 4.5, @@ -1566,6 +1600,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$22.49", "priceCents": 2249, "rating": 4.2, @@ -1597,6 +1632,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$39.99", "priceCents": 3999, "rating": 4.5, @@ -1628,6 +1664,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$36.97", "priceCents": 3697, "rating": 4.7, @@ -1659,6 +1696,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$24.98", "priceCents": 2498, "rating": 4.6, @@ -1690,6 +1728,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$22.49", "priceCents": 2249, "rating": 4.4, @@ -1721,6 +1760,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$16.99", "priceCents": 1699, "rating": 4.4, @@ -1752,6 +1792,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$20.81", "priceCents": 2081, "rating": 4.4, @@ -1783,6 +1824,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$19.29", "priceCents": 1929, "rating": 4.7, @@ -1814,6 +1856,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$21.00", "priceCents": 2100, "rating": 4.7, @@ -1845,6 +1888,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$18.99", "priceCents": 1899, "rating": 4.6, @@ -1876,6 +1920,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$24.99", "priceCents": 2499, "rating": 4.4, @@ -1907,6 +1952,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$21.99", "priceCents": 2199, "rating": 4.5, @@ -1938,6 +1984,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$24.38", "priceCents": 2438, "rating": 4.5, @@ -1969,6 +2016,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$144.00", "priceCents": 14400, "rating": 4.5, @@ -2000,6 +2048,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$19.98", "priceCents": 1998, "rating": 4.5, @@ -2031,6 +2080,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$99.99", "priceCents": 9999, "rating": 4.8, @@ -2062,6 +2112,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$89.99", "priceCents": 8999, "rating": 4.7, @@ -2093,6 +2144,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$22.65", "priceCents": 2265, "rating": 4.4, @@ -2124,6 +2176,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$99.99", "priceCents": 9999, "rating": 4.5, @@ -2155,6 +2208,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": null, "priceCents": null, "rating": 4.6, @@ -2186,6 +2240,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$20.99", "priceCents": 2099, "rating": 4.5, @@ -2217,6 +2272,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$144.00", "priceCents": 14400, "rating": 4.5, @@ -2248,6 +2304,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$20.69", "priceCents": 2069, "rating": 4.6, @@ -2279,6 +2336,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$129.99", "priceCents": 12999, "rating": 3.6, @@ -2310,6 +2368,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$18.99", "priceCents": 1899, "rating": 4.4, @@ -2341,6 +2400,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$39.99", "priceCents": 3999, "rating": 4.7, @@ -2372,6 +2432,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$89.00", "priceCents": 8900, "rating": 4.6, @@ -2403,6 +2464,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$233.01", "priceCents": 23301, "rating": 4.7, @@ -2434,6 +2496,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$19.27", "priceCents": 1927, "rating": 4.6, @@ -2465,6 +2528,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$39.99", "priceCents": 3999, "rating": 4.6, @@ -2496,6 +2560,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$23.47", "priceCents": 2347, "rating": 4.4, @@ -2527,6 +2592,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$36.99", "priceCents": 3699, "rating": 4.5, @@ -2558,6 +2624,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$26.99", "priceCents": 2699, "rating": 4.8, @@ -2589,6 +2656,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$11.99", "priceCents": 1199, "rating": 4.2, @@ -2620,6 +2688,7 @@ exports[`current scraper behavior on the corpus 2026-09/search-yoga-mat 1`] = ` "sponsored": false, }, ], + "prev": null, "price": "$160.00", "priceCents": 16000, "rating": 4.4, diff --git a/tests/unit/engine.test.js b/tests/unit/engine.test.js index 82f41a3..c2e66c5 100644 --- a/tests/unit/engine.test.js +++ b/tests/unit/engine.test.js @@ -439,8 +439,8 @@ describe('products across pages (F-27, F-28)', () => { const PAGE2 = page(card('B0C', '$3.00') + card('B0A', '$5.00') + card('B0B', '$2.00', { rating: false }), null); const site = (url) => (pageNo(url) === 1 ? PAGE1 : PAGE2); - async function twoPages({ lastValues = [], flags } = {}) { - const rig = createRig({ site, flags }); + async function twoPages({ lastValues = [], flags, local } = {}) { + const rig = createRig({ site, flags, local }); const db = await rig.db(); await db.write(lastValues.map((v) => ({ store: 'lastValues', put: v }))); db.close(); @@ -487,15 +487,50 @@ describe('products across pages (F-27, F-28)', () => { after.close(); }); - test('with cloud sync on each page is queued once, and the bundle has the products as they are now', async () => { + test('with cloud sync on but nobody signed in nothing is queued (F-26)', async () => { const rig = await twoPages({ flags: { CLOUD_SYNC: true } }); - const { bundle, seqs } = await rig.engine().outboxBundle(); - expect(seqs).toHaveLength(2); - expect(bundle.syncQueue.map((p) => p.asin)).toEqual(['B0A', 'B0B', 'B0C']); - expect(bundle.syncQueue[0].placements).toHaveLength(3); - expect(bundle.scrapeRunMeta).toMatchObject({ keyword: 'w' }); - await rig.engine().clearOutbox(seqs); - expect((await rig.engine().outboxBundle()).seqs).toEqual([]); + const db = await rig.db(); + expect(await db.count('outbox')).toBe(0); + db.close(); + }); + + test('signed in, each page is queued once for that account, then the run (F-20, F-29e)', async () => { + const rig = await twoPages({ flags: { CLOUD_SYNC: true }, local: { account: { uid: 'u1', email: 'a@b.c' } } }); + const run = await rig.run(); + const db = await rig.db(); + const entries = await db.getAll('outbox'); + expect(entries.map((e) => [e.kind, e.pageIndex, e.uid, e.runId])).toEqual([ + ['page', 1, 'u1', run.runId], + ['page', 2, 'u1', run.runId], + ['run', undefined, 'u1', run.runId], + ]); + db.close(); + }); + + test('the run id is minted once as {sourceId}_{startMs}, with the local day key (F-20, F-29d)', async () => { + const rig = await twoPages(); + const run = await rig.run(); + expect(run.runId).toBe(`k_w_${run.startedAt}`); + expect(run.sourceId).toBe('k_w'); + expect(run.source).toMatchObject({ type: 'keyword', keyword: 'w', sourceId: 'k_w' }); + expect(run.dayKey).toMatch(/^\d{4}-\d{2}-\d{2}$/); + const db = await rig.db(); + expect((await db.runPages(run.runId)).map((p) => p.pageIndex)).toEqual([1, 2]); + db.close(); + }); + + test("another account's last values give no delta, and new ones carry the uid (F-29e)", async () => { + const theirs = { ...prevRun('B0A', 600), uid: 'someone-else' }; + const anon = prevRun('B0B', 250); + const rig = await twoPages({ lastValues: [theirs, anon], local: { account: { uid: 'u1' } } }); + const [a, b] = await results(rig); + expect(a.delta).toMatchObject({ isNew: true }); + expect(a.prev).toBeNull(); + expect(b.delta).toMatchObject({ isNew: false, dPriceCents: -50 }); + expect(b.prev).toMatchObject({ priceCents: 250, rating: 4.5, reviewCount: 900, uid: null }); + const db = await rig.db(); + expect(await db.get('lastValues', 'B0A')).toMatchObject({ priceCents: 500, uid: 'u1' }); + db.close(); }); test('each page drops lastValues older than the age limit (F-26)', async () => { From d71aba99ee1e44fd6d6c04309191d3aa24ac6a26 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:28:34 -0400 Subject: [PATCH 04/18] Plan each outbox entry's cloud writes as whole-field replacements --- scripts/background/sync-plan.js | 211 ++++++++++++++++++++++++++++++++ tests/unit/sync-plan.test.js | 169 +++++++++++++++++++++++++ 2 files changed, 380 insertions(+) create mode 100644 scripts/background/sync-plan.js create mode 100644 tests/unit/sync-plan.test.js diff --git a/scripts/background/sync-plan.js b/scripts/background/sync-plan.js new file mode 100644 index 0000000..f0261ff --- /dev/null +++ b/scripts/background/sync-plan.js @@ -0,0 +1,211 @@ +/** + * @fileoverview What one outbox entry writes to the cloud. Pure: records + * from IndexedDB in, a list of document writes out. + * + * An entry is one page of a run ({kind:'page', pageIndex}) or the end of a + * run ({kind:'run'}). A page writes its chunk, the product and history + * documents of every ASIN on it, the run header and the source. The end of + * a run writes the header and the source again with the final status. + * + * Each write is {path, data, fields}. `fields` null means replace the whole + * document; otherwise only those fields are written, each replaced whole + * (Firestore mergeFields), so a price that failed to parse is gone from + * `latest` instead of kept from last week. A field given as an array is a + * field path, e.g. ['d', '2026-06-09']. + * + * @module SyncPlan + */ + +import { + SV, MK, MAX_PAGE_ITEMS, pageIdOf, sourceOf, sourceIdOf, dayKeyOf, expireAtMs, + runStatusOf, pointOf, deltaOf, timeMs, assertValid +} from '../../packages/schema/index.js'; + +const DAY_MS = 86400000; +const same = (v) => v; + +/** Placements of `product` on page `pageIndex`. */ +function onPage(product, pageIndex) { + return (product.placements || []).filter((pl) => pl.page === pageIndex); +} + +/** The run's source id, day key and source, also for a run from before 2.3. */ +export function runKeys(run) { + const source = run.source || {}; + const found = source.url ? sourceOf(source.url) : { type: source.type || 'keyword', sellerId: source.sellerId || null, keyword: source.keyword || null, url: null }; + const sourceId = run.sourceId || source.sourceId || sourceIdOf(found); + const dayKey = run.dayKey || dayKeyOf(run.startedAt, new Date(run.startedAt).getTimezoneOffset()); + return { + sourceId, + dayKey, + source: { type: found.type, sellerId: found.sellerId || null, keyword: found.keyword || null, url: found.url || source.url || null } + }; +} + +/** + * ASINs on this page whose product document may not exist yet: never seen, + * or last seen by nobody signed in to this account. The caller reads them + * and passes back the ones that are missing, which alone get firstSeenAt. + */ +export function firstSeenCandidates(entry, products) { + if (entry.kind !== 'page') return []; + return products + .filter((p) => onPage(p, entry.pageIndex).length > 0) + .filter((p) => !p.prev || p.prev.uid !== entry.uid) + .map((p) => p.asin); +} + +function counters(products) { + const placements = products.flatMap((p) => p.placements || []); + return { + placements: placements.length, + uniqueAsins: products.length, + sponsored: placements.filter((pl) => pl.sponsored).length, + priceParseFailures: products.filter((p) => !Number.isInteger(p.priceCents)).length, + newSeen: products.filter((p) => p.delta && p.delta.isNew).length + }; +} + +function runDoc(run, keys, products, pages, time) { + const ended = !['starting', 'running', 'stopping'].includes(run.state); + const first = pages.find((pg) => pg.pageIndex === 1) || pages[0]; + const total = first && Number.isInteger(first.total) ? first.total : null; + return { + sv: SV, + runId: run.runId, + sourceId: keys.sourceId, + source: keys.source, + mk: MK, + dayKey: keys.dayKey, + startedAt: time(run.startedAt), + finishedAt: ended && run.finishedAt ? time(run.finishedAt) : null, + status: runStatusOf(run.state), + reason: ended ? run.reason || null : null, + pagesDone: pages.length, + maxPages: run.maxPages, + pagesPlanned: ended && run.reason === 'complete' ? pages.length : null, + totalResultsOnSerp: total, + counters: counters(products) + }; +} + +function sourceDoc(run, keys, pages, time) { + const first = pages.find((pg) => pg.pageIndex === 1) || pages[0]; + const doc = { + sv: SV, + sourceId: keys.sourceId, + type: keys.source.type, + sellerId: keys.source.sellerId, + keyword: keys.source.keyword, + url: keys.source.url, + lastRunId: run.runId, + lastScrapedAt: time(run.startedAt) + }; + if (first && Number.isInteger(first.total)) doc.catalogSize = first.total; + return doc; +} + +function pageDoc(run, pageRec, products, time) { + const items = {}; + let truncated = false; + for (const p of products) { + const here = onPage(p, pageRec.pageIndex); + if (here.length === 0) continue; + if (Object.keys(items).length >= MAX_PAGE_ITEMS) { truncated = true; break; } + const organic = here.filter((pl) => !pl.sponsored && Number.isInteger(pl.rank)).map((pl) => pl.rank); + const pt = pointOf({ ...p, organicRank: organic.length ? Math.min(...organic) : null }); + pt.sp = organic.length ? 0 : 1; + items[p.asin] = pt; + } + const doc = { + sv: SV, + runId: run.runId, + page: pageRec.pageIndex, + scrapedAt: time(timeMs(pageRec.scrapedAt) ?? run.startedAt), + expireAt: time(expireAtMs(run.startedAt)), + count: pageRec.count || 0, + placements: pageRec.placements || 0, + kind: pageRec.kind || 'results', + items + }; + if (Number.isInteger(pageRec.total)) doc.total = pageRec.total; + if (truncated) doc.truncated = true; + return doc; +} + +function productDoc(run, keys, p, time, union) { + const at = timeMs(p.scrapedAt) ?? run.startedAt; + const now = pointOf(p); + const prevAt = p.prev ? timeMs(p.prev.scrapedAt) : null; + const prevPt = p.prev ? pointOf({ priceCents: p.prev.priceCents, rating: p.prev.rating, reviewCount: p.prev.reviewCount }) : null; + const days = prevAt !== null ? Math.floor((at - prevAt) / DAY_MS) : null; + const doc = { + sv: SV, + asin: p.asin, + mk: MK, + url: `https://www.amazon.com/dp/${p.asin}`, + latest: { ...now, at: time(at), runId: run.runId, dayKey: keys.dayKey }, + prev: prevPt ? { ...prevPt, ...(prevAt !== null ? { at: time(prevAt) } : {}) } : null, + delta: deltaOf(now, prevPt, days), + sourceIds: union([keys.sourceId]) + }; + // A name or image this card lacked keeps the last one written. + if (typeof p.name === 'string' && p.name) doc.name = p.name; + if (typeof p.img === 'string' && p.img) doc.img = p.img; + return doc; +} + +/** + * The writes for one outbox entry. + * + * @param {Object} args + * @param {{kind:string, runId:string, uid:string, pageIndex?:number}} args.entry + * @param {Object} args.run the run record from IndexedDB + * @param {Object[]} args.products the run's products + * @param {Object[]} args.pages the run's page records + * @param {Set} [args.missing] ASINs with no product document yet + * @param {function(number):*} [args.time] ms to a Firestore Timestamp + * @param {function(string[]):*} [args.union] values to arrayUnion + * @returns {{path: string[], data: Object, fields: ?Array}[]} + */ +export function planEntry({ entry, run: rec, products, pages, missing = new Set(), time = same, union = same }) { + const run = { ...rec, startedAt: timeMs(rec.startedAt), finishedAt: timeMs(rec.finishedAt) }; + const ws = ['workspaces', entry.uid]; + const keys = runKeys(run); + const writes = []; + + if (entry.kind === 'page') { + const pageRec = pages.find((pg) => pg.pageIndex === entry.pageIndex); + if (pageRec) { + const chunk = pageDoc(run, pageRec, products, time); + assertValid('page', chunk); + writes.push({ path: [...ws, 'runs', run.runId, 'pages', pageIdOf(pageRec.pageIndex)], data: chunk, fields: null }); + } + for (const p of products) { + if (onPage(p, entry.pageIndex).length === 0) continue; + const doc = productDoc(run, keys, p, time, union); + if (missing.has(p.asin)) { + doc.firstSeenAt = doc.latest.at; + doc.firstRunId = run.runId; + } + assertValid('product', { ...doc, sourceIds: [keys.sourceId] }); + writes.push({ path: [...ws, 'products', p.asin], data: doc, fields: Object.keys(doc) }); + + const point = pointOf(p); + assertValid('history', { sv: SV, asin: p.asin, d: { [keys.dayKey]: point } }); + writes.push({ + path: [...ws, 'products', p.asin, 'history', 'daily'], + data: { sv: SV, asin: p.asin, d: { [keys.dayKey]: point } }, + fields: ['sv', 'asin', ['d', keys.dayKey]] + }); + } + } + + const header = runDoc(run, keys, products, pages, time); + assertValid('run', header); + writes.push({ path: [...ws, 'runs', run.runId], data: header, fields: Object.keys(header) }); + const source = sourceDoc(run, keys, pages, time); + assertValid('source', source); + writes.push({ path: [...ws, 'sources', keys.sourceId], data: source, fields: Object.keys(source) }); + return writes; +} diff --git a/tests/unit/sync-plan.test.js b/tests/unit/sync-plan.test.js new file mode 100644 index 0000000..47f8853 --- /dev/null +++ b/tests/unit/sync-plan.test.js @@ -0,0 +1,169 @@ +/** + * @jest-environment node + * + * The pure sync plan: outbox entry and IndexedDB records in, cloud writes out. + */ +const { planEntry, firstSeenCandidates, runKeys } = require('../../scripts/background/sync-plan.js'); +const S = require('../../packages/schema/index.js'); + +const START = Date.parse('2026-06-09T14:02:11Z'); +const RUN_ID = `k_mug_${START}`; + +const run = (patch = {}) => ({ + runId: RUN_ID, sourceId: 'k_mug', dayKey: '2026-06-09', state: 'running', reason: null, + source: { type: 'keyword', sellerId: null, keyword: 'mug', url: 'https://www.amazon.com/s?k=mug', sourceId: 'k_mug' }, + startedAt: START, finishedAt: null, page: 2, maxPages: 20, itemCount: 3, + ...patch, +}); + +const product = (asin, patch = {}) => ({ + asin, name: `Mug ${asin}`, priceCents: 1999, rating: 4.5, reviewCount: 100, isPrime: true, + sponsored: false, organicRank: 1, img: null, runId: RUN_ID, pageIndex: 1, + scrapedAt: '2026-06-09T14:03:00.000Z', + placements: [{ page: 1, position: 1, sponsored: false, rank: 1 }], + delta: { isNew: true, dPriceCents: null, dRating: null, dReviews: null }, prev: null, + ...patch, +}); + +const pages = [ + { runId: RUN_ID, pageIndex: 1, count: 2, placements: 3, kind: 'results', scrapedAt: '2026-06-09T14:03:00.000Z', total: 412 }, + { runId: RUN_ID, pageIndex: 2, count: 1, placements: 2, kind: 'last', scrapedAt: '2026-06-09T14:03:05.000Z', total: 412 }, +]; + +const products = [ + product('B0AAAAAAA1', { + sponsored: true, + placements: [ + { page: 1, position: 1, sponsored: true, rank: null }, + { page: 2, position: 2, sponsored: false, rank: 3 }, + ], + organicRank: 3, + }), + product('B0AAAAAAA2', { + priceCents: null, organicRank: 1, placements: [{ page: 1, position: 2, sponsored: false, rank: 1 }], + delta: { isNew: false, dPriceCents: null, dRating: 0, dReviews: 5 }, + prev: { priceCents: 2100, rating: 4.5, reviewCount: 95, scrapedAt: '2026-06-02T14:00:00.000Z', uid: 'u1' }, + }), + product('B0AAAAAAA3', { pageIndex: 2, organicRank: 2, placements: [{ page: 2, position: 1, sponsored: false, rank: 2 }] }), +]; + +const entry = (patch) => ({ seq: 1, runId: RUN_ID, uid: 'u1', ...patch }); +const byPath = (writes) => Object.fromEntries(writes.map((w) => [w.path.join('/'), w])); + +describe('a page entry', () => { + const writes = planEntry({ entry: entry({ kind: 'page', pageIndex: 1 }), run: run(), products, pages }); + const w = byPath(writes); + + test('writes the chunk, each ASIN on the page twice, the run and the source', () => { + expect(writes.map((x) => x.path.slice(2).join('/'))).toEqual([ + `runs/${RUN_ID}/pages/p0001`, + 'products/B0AAAAAAA1', 'products/B0AAAAAAA1/history/daily', + 'products/B0AAAAAAA2', 'products/B0AAAAAAA2/history/daily', + `runs/${RUN_ID}`, + 'sources/k_mug', + ]); + expect(writes.every((x) => x.path[0] === 'workspaces' && x.path[1] === 'u1')).toBe(true); + }); + + test('the chunk keeps every ASIN on the page with its sponsored flag and rank', () => { + const chunk = w[`workspaces/u1/runs/${RUN_ID}/pages/p0001`]; + expect(chunk.fields).toBeNull(); + expect(chunk.data).toMatchObject({ sv: S.SV, page: 1, count: 2, placements: 3, kind: 'results', total: 412 }); + expect(chunk.data.items).toEqual({ + B0AAAAAAA1: { p: 1999, r: 4.5, v: 100, pr: 1, sp: 1 }, + B0AAAAAAA2: { r: 4.5, v: 100, pr: 1, rk: 1, sp: 0 }, + }); + expect(chunk.data.expireAt).toBe(START + S.PAGE_TTL_DAYS * 86400000); + }); + + test('latest, prev and delta are replaced whole, and a failed price leaves no stale value (F-21)', () => { + const doc = w['workspaces/u1/products/B0AAAAAAA2']; + expect(doc.fields).toEqual(expect.arrayContaining(['latest', 'prev', 'delta', 'sourceIds'])); + expect(doc.data.latest).toEqual({ r: 4.5, v: 100, pr: 1, rk: 1, at: Date.parse('2026-06-09T14:03:00.000Z'), runId: RUN_ID, dayKey: '2026-06-09' }); + expect(doc.data.prev).toEqual({ p: 2100, r: 4.5, v: 95, at: Date.parse('2026-06-02T14:00:00.000Z') }); + // No price this run, so no price delta; the rest compares + expect(doc.data.delta).toEqual({ r: 0, v: 5, days: 7 }); + }); + + test('a product seen for the first time has a null delta and no prev', () => { + const doc = w['workspaces/u1/products/B0AAAAAAA1'].data; + expect(doc.delta).toBeNull(); + expect(doc.prev).toBeNull(); + expect(doc.latest.rk).toBe(3); + expect(doc.name).toBe('Mug B0AAAAAAA1'); + expect(doc).not.toHaveProperty('img'); + expect(doc).not.toHaveProperty('firstSeenAt'); + }); + + test('history writes only its own day', () => { + const h = w['workspaces/u1/products/B0AAAAAAA1/history/daily']; + expect(h.fields).toEqual(['sv', 'asin', ['d', '2026-06-09']]); + expect(h.data).toEqual({ sv: S.SV, asin: 'B0AAAAAAA1', d: { '2026-06-09': { p: 1999, r: 4.5, v: 100, pr: 1, rk: 3 } } }); + }); + + test('the run header is active with counters from the whole run', () => { + const h = w[`workspaces/u1/runs/${RUN_ID}`].data; + expect(h).toMatchObject({ + sv: S.SV, runId: RUN_ID, sourceId: 'k_mug', mk: 'US', dayKey: '2026-06-09', status: 'active', + finishedAt: null, reason: null, pagesDone: 2, maxPages: 20, pagesPlanned: null, totalResultsOnSerp: 412, + counters: { placements: 4, uniqueAsins: 3, sponsored: 1, priceParseFailures: 1, newSeen: 2 }, + }); + expect(h.source).toEqual({ type: 'keyword', sellerId: null, keyword: 'mug', url: 'https://www.amazon.com/s?k=mug' }); + }); + + test('the source never touches the fields the dashboard owns', () => { + const src = w['workspaces/u1/sources/k_mug']; + expect(src.fields.sort()).toEqual(['catalogSize', 'keyword', 'lastRunId', 'lastScrapedAt', 'sellerId', 'sourceId', 'sv', 'type', 'url']); + }); +}); + +test('a repeat on a later page is in that page\'s chunk and product writes', () => { + const writes = planEntry({ entry: entry({ kind: 'page', pageIndex: 2 }), run: run(), products, pages }); + const chunk = byPath(writes)[`workspaces/u1/runs/${RUN_ID}/pages/p0002`].data; + expect(Object.keys(chunk.items)).toEqual(['B0AAAAAAA1', 'B0AAAAAAA3']); + expect(chunk.items.B0AAAAAAA1).toMatchObject({ sp: 0, rk: 3 }); +}); + +test('firstSeenAt goes only on documents the caller found missing (F-29b)', () => { + const writes = planEntry({ entry: entry({ kind: 'page', pageIndex: 1 }), run: run(), products, pages, missing: new Set(['B0AAAAAAA1']) }); + const w = byPath(writes); + expect(w['workspaces/u1/products/B0AAAAAAA1'].data).toMatchObject({ firstSeenAt: Date.parse('2026-06-09T14:03:00.000Z'), firstRunId: RUN_ID }); + expect(w['workspaces/u1/products/B0AAAAAAA1'].fields).toEqual(expect.arrayContaining(['firstSeenAt', 'firstRunId'])); + expect(w['workspaces/u1/products/B0AAAAAAA2'].fields).not.toContain('firstSeenAt'); +}); + +test('first-seen candidates: new here, or last seen by nobody signed in to this account', () => { + const list = [ + product('B0NEW00001'), + product('B0MINE0001', { prev: { uid: 'u1' } }), + product('B0ANON0001', { prev: { uid: null } }), + product('B0ELSE0001', { placements: [{ page: 2, position: 1, sponsored: false, rank: 1 }] }), + ]; + expect(firstSeenCandidates(entry({ kind: 'page', pageIndex: 1 }), list)).toEqual(['B0NEW00001', 'B0ANON0001']); + expect(firstSeenCandidates(entry({ kind: 'run' }), list)).toEqual([]); +}); + +test('a run entry writes the final header and the source only', () => { + const ended = run({ state: 'blocked', reason: 'blocked', finishedAt: START + 60000 }); + const writes = planEntry({ entry: entry({ kind: 'run' }), run: ended, products, pages }); + expect(writes.map((x) => x.path.slice(2).join('/'))).toEqual([`runs/${RUN_ID}`, 'sources/k_mug']); + expect(writes[0].data).toMatchObject({ status: 'stopped', reason: 'blocked', finishedAt: START + 60000, pagesPlanned: null }); + const done = planEntry({ entry: entry({ kind: 'run' }), run: run({ state: 'done', reason: 'complete', finishedAt: START + 1 }), products, pages }); + expect(done[0].data).toMatchObject({ status: 'complete', pagesPlanned: 2 }); +}); + +test('times and array unions go through the converters the sync module passes', () => { + const writes = planEntry({ + entry: entry({ kind: 'page', pageIndex: 1 }), run: run(), products, pages, + time: (ms) => ({ toMillis: () => ms, ms }), union: (vals) => ({ union: vals }), + }); + const p = byPath(writes)['workspaces/u1/products/B0AAAAAAA1'].data; + expect(p.sourceIds).toEqual({ union: ['k_mug'] }); + expect(p.latest.at.ms).toBe(Date.parse('2026-06-09T14:03:00.000Z')); +}); + +test('a run from before 2.3 still gets its source id and day key', () => { + const old = { runId: '1749477731000-abc', source: { type: 'storefront', sellerId: 'A3K9XELT4QZ6M2', url: 'https://www.amazon.com/s?me=A3K9XELT4QZ6M2' }, startedAt: START }; + expect(runKeys(old)).toMatchObject({ sourceId: 's_A3K9XELT4QZ6M2', source: { type: 'storefront', sellerId: 'A3K9XELT4QZ6M2' } }); + expect(runKeys(old).dayKey).toBe(S.dayKeyOf(START, new Date(START).getTimezoneOffset())); +}); From d102a1a332637bef2177a868aa812925bdbccf09 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:32:28 -0400 Subject: [PATCH 05/18] Drain the outbox one entry at a time, deleting each only after its commit --- scripts/background/service-worker.js | 155 +++++++++-- scripts/background/sync.js | 286 +++++++++----------- scripts/lib/messages.js | 1 + tests/setup/esm-to-cjs-transform.js | 10 +- tests/unit/service-worker.test.js | 10 +- tests/unit/sync.test.js | 383 +++++++++++++-------------- 6 files changed, 459 insertions(+), 386 deletions(-) diff --git a/scripts/background/service-worker.js b/scripts/background/service-worker.js index 13cdecb..bd68af3 100644 --- a/scripts/background/service-worker.js +++ b/scripts/background/service-worker.js @@ -1,10 +1,11 @@ -import { auth } from './firebase-init.js'; +import { auth, db as firestore } from './firebase-init.js'; import { onAuthStateChanged, signInWithEmailAndPassword, + sendPasswordResetEmail, signOut, } from 'firebase/auth/web-extension'; -import { syncToCloud } from './sync.js'; +import { createSync, isAuthError } from './sync.js'; import { createRouter } from './router.js'; import { createEngine } from './engine.js'; import DB from './db.js'; @@ -39,11 +40,14 @@ const chatDeps = { }; // The run +// Saved pages and ended runs wake the sync; see scheduleFlush below. +const thenFlush = (p) => p.then((out) => { scheduleFlush(); return out; }); + router.on(Msg.T.START_RUN, (m) => engine.start(m)); -router.on(Msg.T.STOP_RUN, () => engine.stop()); +router.on(Msg.T.STOP_RUN, () => thenFlush(engine.stop())); router.on(Msg.T.GET_STATE, () => engine.getState()); router.on(Msg.T.PAGE_READY, (m, sender) => engine.pageReady(m, sender)); -router.on(Msg.T.PAGE_RESULT, (m, sender) => engine.pageResult(m, sender)); +router.on(Msg.T.PAGE_RESULT, (m, sender) => thenFlush(engine.pageResult(m, sender))); router.on(Msg.T.HEARTBEAT, (m, sender) => engine.heartbeat(m, sender)); // Spread analysis @@ -80,24 +84,32 @@ chrome.runtime.onInstalled.addListener((details) => { }); } else if (details.reason === 'update') { console.log('[ProScan] Extension updated to version', chrome.runtime.getManifest().version); - migrate().then(() => engine.recover('updated')); + migrate().then(() => engine.recover('updated')).then(() => scheduleFlush()); } }); // A run cannot survive a browser restart, since its tab id is gone. chrome.runtime.onStartup.addListener(() => { - migrate().then(() => engine.recover()); + migrate().then(() => engine.recover()).then(() => scheduleFlush()); }); // Neither listener needs the tabs permission. -chrome.tabs.onRemoved.addListener((tabId) => { engine.tabRemoved(tabId); }); +chrome.tabs.onRemoved.addListener((tabId) => { thenFlush(engine.tabRemoved(tabId)); }); chrome.tabs.onUpdated.addListener((tabId, info, tab) => { engine.tabUpdated(tabId, info, tab); }); -// ════════════════════════════════════════════════════════════════════════════ -// M3 cloud sync — Firebase Auth (extension-native) + Firestore write path. -// The popup is a plain (unbundled) page; it drives sign-in / export by sending -// these messages to the worker, which owns the single Firebase instance. -// ════════════════════════════════════════════════════════════════════════════ +// ── ProScan account and cloud sync ────────────────────────────────────────── +// The popup is a plain page; it signs in and exports through these messages, +// and the worker owns the one Firebase instance. +// +// storage.local keeps `account` {uid, email} while signed in, so the engine +// queues pages for that account, and `authNotice` 'expired' when Firebase +// dropped the session without the user signing out. + +const DASHBOARD_URL = 'https://proscanbot.web.app/dashboard/'; +const FLUSH_DELAY_MS = 3000; +let flushTimer = null; + +const sync = createSync({ db: firestore, openStore: () => engine.db() }); /** Resolve the current Firebase user, waiting for auth to rehydrate from * IndexedDB after a cold service-worker start. */ @@ -115,6 +127,63 @@ function currentUser() { const publicUser = (u) => u ? { uid: u.uid, email: u.email, displayName: u.displayName } : null; +let signingOut = false; + +async function rememberAccount(user) { + if (user) { + await chrome.storage.local.set({ account: { uid: user.uid, email: user.email || null } }); + await chrome.storage.local.remove('authNotice'); + return; + } + const { account } = await chrome.storage.local.get('account'); + if (!account) return; + await chrome.storage.local.remove('account'); + // Signed out without asking: the refresh token was revoked or expired. + if (!signingOut) await chrome.storage.local.set({ authNotice: 'expired' }); +} + +onAuthStateChanged(auth, (user) => { + rememberAccount(user).catch(() => {}); + if (user) scheduleFlush(); +}); + +/** Firebase refused the session: sign out and say so in the popup. */ +async function expireSession() { + await chrome.storage.local.set({ authNotice: 'expired' }); + await chrome.storage.local.remove('account'); + await signOut(auth).catch(() => {}); +} + +/** Writes the outbox for the signed-in account now. */ +async function flushNow() { + if (!Flags.CLOUD_SYNC) return { skipped: true }; + const user = await currentUser(); + if (!user) return { skipped: true }; + try { + const out = await sync.flush(user.uid); + await chrome.storage.local.set({ lastSync: { at: Date.now(), error: null } }); + return out; + } catch (err) { + await chrome.storage.local.set({ lastSync: { at: Date.now(), error: (err && err.code) || 'unknown' } }); + if (isAuthError(err)) await expireSession(); + throw err; + } +} + +/** + * Flushes a few seconds after the last page or run end, on wake events the + * worker already gets. No alarm: the timer dies with the worker, and what + * it missed goes out on the next wake. + */ +function scheduleFlush(ms = FLUSH_DELAY_MS) { + if (!Flags.CLOUD_SYNC) return; + if (flushTimer) clearTimeout(flushTimer); + flushTimer = setTimeout(() => { + flushTimer = null; + flushNow().catch(() => {}); + }, ms); +} + /** Map Firebase auth error codes to friendly popup messages. */ function friendlyAuthError(err) { switch (err && err.code) { @@ -131,34 +200,68 @@ function friendlyAuthError(err) { case 'auth/operation-not-allowed': return 'Email sign-in is not enabled for this project yet.'; default: - return ((err && err.message) || 'Sign-in failed.').replace(/^Firebase:\s*/, ''); + return 'Sign-in failed. Try again.'; } } -router.on(Msg.T.PROSCAN_AUTH_STATE, async () => ({ user: publicUser(await currentUser()) })); +router.on(Msg.T.PROSCAN_AUTH_STATE, async () => { + const user = await currentUser(); + const { authNotice, lastSync } = await chrome.storage.local.get(['authNotice', 'lastSync']); + // Opening the popup is a wake event too. + if (user) scheduleFlush(0); + return { + user: publicUser(user), + notice: user ? null : authNotice || null, + pending: user ? await sync.pending(user.uid).catch(() => 0) : 0, + lastSync: lastSync || null, + dashboardUrl: DASHBOARD_URL, + }; +}); router.on(Msg.T.PROSCAN_SIGN_IN, (m) => - signInWithEmailAndPassword(auth, m.email, m.password) - .then((cred) => ({ user: publicUser(cred.user) })) + signInWithEmailAndPassword(auth, String(m.email || ''), String(m.password || '')) + .then(async (cred) => { + await rememberAccount(cred.user); + scheduleFlush(0); + return { user: publicUser(cred.user) }; + }) .catch((err) => ({ error: friendlyAuthError(err) }))); -router.on(Msg.T.PROSCAN_SIGN_OUT, () => - signOut(auth).then(() => ({ ok: true }), (err) => ({ error: err.message }))); +// The same answer whether or not the account exists. +router.on(Msg.T.PROSCAN_RESET_PASSWORD, async (m) => { + const email = String(m.email || '').trim(); + if (!email) return { error: 'Enter your email first.' }; + try { + await sendPasswordResetEmail(auth, email, { url: DASHBOARD_URL }); + } catch (err) { + if (err && err.code === 'auth/invalid-email') return { error: 'That email address does not look valid.' }; + if (err && err.code === 'auth/network-request-failed') return { error: 'Network error. Check your connection.' }; + } + return { ok: true, message: `If an account exists for ${email}, a reset link is on its way.` }; +}); + +router.on(Msg.T.PROSCAN_SIGN_OUT, async () => { + signingOut = true; + try { + await signOut(auth); + await chrome.storage.local.remove(['account', 'authNotice']); + return { ok: true }; + } catch (err) { + return { error: err.message }; + } finally { + signingOut = false; + } +}); router.on(Msg.T.PROSCAN_EXPORT, async () => { if (!Flags.CLOUD_SYNC) return { error: 'Export to ProScan is not available in this version.' }; const user = await currentUser(); if (!user) return { error: 'Sign in to ProScan first.' }; - const { bundle, seqs } = await engine.outboxBundle(); - if (bundle.syncQueue.length === 0) return { ok: true, written: 0, products: 0 }; try { - const result = await syncToCloud(user.uid, bundle); - // Only what was written leaves the outbox; lastValues stays for deltas. - await engine.clearOutbox(seqs); - return { ok: true, ...result }; + return { ok: true, ...(await flushNow()) }; } catch (err) { - console.error('[ProScan] cloud export failed', err); - return { error: (err && err.message) || 'Export failed.' }; + if (isAuthError(err)) return { error: 'Your session expired. Sign in again to keep syncing.', expired: true }; + return { error: 'Could not reach ProScan. Your scans are kept and will sync later.' }; } }); diff --git a/scripts/background/sync.js b/scripts/background/sync.js index 4dedd8d..eefd004 100644 --- a/scripts/background/sync.js +++ b/scripts/background/sync.js @@ -1,187 +1,157 @@ /** - * @fileoverview Sync consumer — drains the durable syncQueue into the - * signed-in user's Firestore workspace (workspaces/{uid}), converting the - * producer's shapes to the cloud schema (proscan-web docs/architecture/ - * data-model.md §4) so every write passes firestore.rules. + * @fileoverview Cloud sync: drains the IndexedDB outbox into the signed-in + * user's workspace, one entry at a time. * - * The producer (scraper.js/storage.js/delta.js) stamps each product with - * priceCents + a delta block + runId/pageIndex and pushes it to - * chrome.storage.local 'syncQueue'. It does NOT carry several fields the cloud - * schema requires — this module derives them at write time: - * - canonical sourceId (s_{sellerId} | k_{slug}) from scrapeRunMeta - * - canonical runId ({sourceId}_{startEpochMs}) - * - dayKey (UTC date of run start; rules REQUIRE it as string) - * - mk ('US' — only marketplace today) - * - delta.pPct (percent change; producer only has absolute cents) - * - ISO strings -> Firestore Timestamp - * - isPrime boolean -> pr 0/1 - * Money stays integer cents. Unknown numerics are OMITTED (never written as - * null) so they never violate the rules' `is int` checks. Dashboard-owned - * fields (lead/verdict/tags/spread) are never touched — set(merge) preserves - * them. + * Each entry (one page of a run, or the end of a run) is planned by + * sync-plan.js, committed, and only then deleted. Entries queued while a + * flush runs are picked up before it returns. An entry belongs to the + * account that was signed in when it was queued and is never written into + * another account's workspace. + * + * Every write is idempotent, so an entry that fails halfway is simply + * written again on the next flush. + * + * Authored as ESM and bundled into the service worker by esbuild. The + * contract test runs this same file in Node against the emulators. * - * Bundled into the service worker by esbuild. * @module Sync */ import { - writeBatch, doc, - serverTimestamp, + getDoc, + writeBatch, Timestamp, arrayUnion, + FieldPath, } from 'firebase/firestore'; -import { db } from './firebase-init.js'; +import { planEntry, firstSeenCandidates } from './sync-plan.js'; -const slugify = (s) => - String(s).toLowerCase().trim().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, ''); -const tsOf = (iso) => Timestamp.fromDate(new Date(iso)); -const round1 = (n) => Math.round(n * 10) / 10; -const intOrUndef = (n) => - typeof n === 'number' && Number.isFinite(n) ? Math.round(n) : undefined; +const FIRESTORE = { doc, getDoc, writeBatch, Timestamp, arrayUnion, FieldPath }; -function deriveSourceId(meta) { - if (meta && meta.sellerId) return `s_${meta.sellerId}`; - if (meta && meta.keyword) return `k_${slugify(meta.keyword)}`; - return 'k_unknown'; -} +/** Firestore allows 500 writes per batch; stay under it. */ +export const BATCH_LIMIT = 450; -/** Compact observation point {p,r,v,pr} — schema's history/latest shape. - * Omits unknown numerics so the rules' `is int` checks never see a null. */ -function pointFrom(q) { - const pt = { - p: intOrUndef(q.priceCents), - r: typeof q.rating === 'number' && q.rating > 0 ? round1(q.rating) : undefined, - v: typeof q.reviewCount === 'number' && q.reviewCount > 0 ? Math.round(q.reviewCount) : undefined, - pr: q.isPrime ? 1 : 0, - }; - Object.keys(pt).forEach((k) => pt[k] === undefined && delete pt[k]); - return pt; -} +/** Errors that mean the account's session is gone, not the network. */ +const AUTH_CODES = ['permission-denied', 'unauthenticated', 'auth/user-token-expired', 'auth/user-disabled', 'auth/invalid-user-token']; -/** Convert the producer delta {dPriceCents,dRating,dReviews} → schema - * {p,pPct,r,v}. Returns undefined for first-sight / empty deltas. */ -function deltaBlock(q) { - const d = q.delta; - if (!d || d.isNew) return undefined; - const out = {}; - if (typeof d.dPriceCents === 'number') { - out.p = Math.round(d.dPriceCents); - const prev = typeof q.priceCents === 'number' ? q.priceCents - d.dPriceCents : null; - if (prev && prev !== 0) out.pPct = round1((d.dPriceCents / prev) * 100); - } - if (typeof d.dRating === 'number') out.r = round1(d.dRating); - if (typeof d.dReviews === 'number') out.v = Math.round(d.dReviews); - return Object.keys(out).length ? out : undefined; +export function isAuthError(err) { + return !!err && AUTH_CODES.includes(err.code); } /** - * Drain the queue into workspaces/{uid}. Idempotent: every write is - * set(merge) of absolute values + one arrayUnion, so a retried export - * rewrites byte-identical docs. - * - * @param {string} uid signed-in user's uid (== workspace id) - * @param {{syncQueue?: object[], scrapeRunMeta?: object, scrapeRunPages?: object[]}} bundle - * @returns {Promise<{written: number, runId: string|null, products: number}>} + * @param {Object} deps + * @param {Object} deps.db - the Firestore instance + * @param {function(): Promise} deps.openStore - resolves to the db.js api + * @param {Object} [deps.fs] - firebase/firestore functions, for tests + * @param {function(Object): Promise} [deps.onBatch] - called before each commit, for tests */ -export async function syncToCloud(uid, bundle) { - const queue = (bundle && bundle.syncQueue) || []; - if (!queue.length) return { written: 0, runId: null, products: 0 }; +export function createSync({ db, openStore, fs = FIRESTORE, batchLimit = BATCH_LIMIT, onBatch = null, log = console }) { + let running = null; + let again = false; + + const ref = (path) => fs.doc(db, ...path); + const time = (ms) => fs.Timestamp.fromMillis(ms); + const union = (vals) => fs.arrayUnion(...vals); + const field = (f) => (Array.isArray(f) ? new fs.FieldPath(...f) : f); + + /** Entries queued for `uid`, oldest first. */ + async function mine(store, uid) { + const all = await store.getAll('outbox'); + return all + .filter((e) => e.uid === uid && (e.kind === 'page' || e.kind === 'run')) + .sort((a, b) => a.seq - b.seq); + } - const meta = (bundle && bundle.scrapeRunMeta) || {}; - const sourceId = deriveSourceId(meta); - const startMs = meta.startedAt ? Date.parse(meta.startedAt) : Date.now(); - const runId = `${sourceId}_${startMs}`; - const dayKey = new Date(startMs).toISOString().slice(0, 10); - const mk = 'US'; + /** Product documents among `asins` that do not exist yet. */ + async function missingOf(uid, asins) { + const missing = new Set(); + for (let i = 0; i < asins.length; i += 25) { + const part = asins.slice(i, i + 25); + const snaps = await Promise.all(part.map((a) => fs.getDoc(ref(['workspaces', uid, 'products', a])))); + snaps.forEach((snap, j) => { if (!snap.exists()) missing.add(part[j]); }); + } + return missing; + } - // ── products + history, chunked under the 500-writes/batch limit ── - const CHUNK = 200; // 2 writes per product → 400 writes/batch - let products = 0; - for (let i = 0; i < queue.length; i += CHUNK) { - const batch = writeBatch(db); - for (const q of queue.slice(i, i + CHUNK)) { - if (!q || !q.asin) continue; - const at = q.scrapedAt ? tsOf(q.scrapedAt) : serverTimestamp(); + /** Writes one entry. Returns how many writes it took, or null if it had nothing to write. */ + async function commitEntry(store, entry) { + const run = await store.get('runs', entry.runId); + if (!run) return null; + const [products, pages] = await Promise.all([store.runProducts(entry.runId), store.runPages(entry.runId)]); + const missing = await missingOf(entry.uid, firstSeenCandidates(entry, products)); + const writes = planEntry({ entry, run, products, pages, missing, time, union }); + + for (let i = 0; i < writes.length; i += batchLimit) { + const batch = fs.writeBatch(db); + for (const w of writes.slice(i, i + batchLimit)) { + if (w.fields) batch.set(ref(w.path), w.data, { mergeFields: w.fields.map(field) }); + else batch.set(ref(w.path), w.data); + } + if (onBatch) await onBatch({ entry, writes: Math.min(batchLimit, writes.length - i) }); + await batch.commit(); + } + return { writes: writes.length, products: products.filter((p) => (p.placements || []).some((pl) => pl.page === entry.pageIndex)).length }; + } - const payload = { - asin: q.asin, // rules: doc id must equal this - mk, - url: `https://www.amazon.com/dp/${q.asin}`, - latest: { ...pointFrom(q), at, runId, dayKey }, // rules require latest.dayKey:string - sourceIds: arrayUnion(sourceId), - }; - if (typeof q.name === 'string' && q.name) payload.name = q.name; - const d = deltaBlock(q); - if (d) payload.delta = d; - if (q.delta && q.delta.isNew) { - // immutable first-sight stamps — only on first sight, else merge would clobber - payload.firstSeenAt = at; - payload.firstRunId = runId; + async function drain(uid) { + const store = await openStore(); + const totals = { entries: 0, pages: 0, runs: 0, products: 0, writes: 0 }; + const runs = new Set(); + // New entries can arrive while we write; keep going until none are left. + for (let round = 0; round < 1000; round++) { + const entries = await mine(store, uid); + if (entries.length === 0) break; + for (const entry of entries) { + const done = await commitEntry(store, entry); + await store.write([{ store: 'outbox', delete: entry.seq }]); + totals.entries++; + if (!done) continue; + runs.add(entry.runId); + totals.writes += done.writes; + if (entry.kind === 'page') { + totals.pages++; + totals.products += done.products; + } } - batch.set(doc(db, 'workspaces', uid, 'products', q.asin), payload, { merge: true }); + } + totals.runs = runs.size; + return totals; + } - // date-keyed history point (deep-merges into the d-map) - batch.set( - doc(db, 'workspaces', uid, 'products', q.asin, 'history', 'daily'), - { asin: q.asin, d: { [dayKey]: pointFrom(q) } }, - { merge: true }, - ); - products++; + /** + * Writes everything queued for `uid`. One flush at a time; a call during + * a flush makes it look again once more before it resolves. + * + * @param {string} uid + * @returns {Promise<{entries:number, pages:number, runs:number, products:number, writes:number}>} + */ + function flush(uid) { + if (!uid) return Promise.resolve({ entries: 0, pages: 0, runs: 0, products: 0, writes: 0 }); + if (running) { + again = true; + return running; } - await batch.commit(); + running = (async () => { + const totals = { entries: 0, pages: 0, runs: 0, products: 0, writes: 0 }; + do { + again = false; + const t = await drain(uid); + for (const k of Object.keys(totals)) totals[k] += t[k]; + } while (again); + return totals; + })().catch((err) => { + log.warn('[ProScan] Sync stopped:', err && (err.code || err.message)); + throw err; + }).finally(() => { running = null; }); + return running; } - // ── run header + source spine (once per drain) ── - const head = writeBatch(db); - head.set( - doc(db, 'workspaces', uid, 'runs', runId), - { - runId, - sourceId, // rules require string - source: { - type: meta.type || 'keyword', - sellerId: meta.sellerId ?? null, - keyword: meta.keyword ?? null, - url: meta.url ?? null, - }, - mk, - dayKey, // rules require string - startedAt: meta.startedAt ? tsOf(meta.startedAt) : serverTimestamp(), - finishedAt: serverTimestamp(), - status: 'complete', // rules enum - pagesDone: bundle.scrapeRunPages ? bundle.scrapeRunPages.length : null, - pagesPlanned: bundle.scrapeRunPages ? bundle.scrapeRunPages.length : null, - counters: { - // Each queued product is one ASIN; its placements list every card it had - placements: queue.reduce((n, q) => n + (Array.isArray(q.placements) ? q.placements.length : 1), 0), - uniqueAsins: new Set(queue.map((q) => q.asin)).size, - sponsored: queue.reduce( - (n, q) => n + (Array.isArray(q.placements) ? q.placements.filter((pl) => pl.sponsored).length : q.sponsored ? 1 : 0), - 0, - ), - priceParseFailures: queue.filter((q) => q.priceCents == null).length, - newSeen: queue.filter((q) => q.delta && q.delta.isNew).length, - }, - }, - { merge: true }, - ); - head.set( - doc(db, 'workspaces', uid, 'sources', sourceId), - { - sourceId, - type: meta.type === 'storefront' ? 'storefront' : 'keyword', // rules enum - sellerId: meta.sellerId ?? null, - keyword: meta.keyword ?? null, - url: meta.url ?? null, - lastRunId: runId, - lastScrapedAt: meta.startedAt ? tsOf(meta.startedAt) : serverTimestamp(), - // cadenceDays intentionally omitted — rules default it; never stomp a - // dashboard-set cadence on re-scan. - }, - { merge: true }, - ); - await head.commit(); + /** How many entries are waiting for `uid`. */ + async function pending(uid) { + if (!uid) return 0; + return (await mine(await openStore(), uid)).length; + } - return { written: products * 2 + 2, runId, products }; + return { flush, pending }; } diff --git a/scripts/lib/messages.js b/scripts/lib/messages.js index 90d08be..eade46b 100644 --- a/scripts/lib/messages.js +++ b/scripts/lib/messages.js @@ -37,6 +37,7 @@ const Msg = (() => { PROSCAN_AUTH_STATE: { to: 'worker', from: 'page' }, PROSCAN_SIGN_IN: { to: 'worker', from: 'page' }, PROSCAN_SIGN_OUT: { to: 'worker', from: 'page' }, + PROSCAN_RESET_PASSWORD: { to: 'worker', from: 'page' }, PROSCAN_EXPORT: { to: 'worker', from: 'page' }, // Worker or popup -> content script diff --git a/tests/setup/esm-to-cjs-transform.js b/tests/setup/esm-to-cjs-transform.js index b2ebbf5..1ccc5f3 100644 --- a/tests/setup/esm-to-cjs-transform.js +++ b/tests/setup/esm-to-cjs-transform.js @@ -12,6 +12,7 @@ * * Supported forms (all that sync.js uses): * import { a, b } from 'mod'; -> const { a, b } = require('mod'); + * import { a as b } from 'mod'; -> const { a: b } = require('mod'); * import x from 'mod'; -> const x = require('mod'); * export async function f() {} -> async function f() {}; module.exports.f = f; * export function f() {} -> function f() {}; module.exports.f = f; @@ -27,7 +28,7 @@ function rewrite(src) { (_m, names, mod) => { const clean = names .split(',') - .map((s) => s.trim()) + .map((s) => s.trim().replace(/\s+as\s+/, ': ')) .filter(Boolean) .join(', '); return `const { ${clean} } = require('${mod}');`; @@ -58,8 +59,15 @@ function rewrite(src) { return out; } +const crypto = require('crypto'); +const self = require('fs').readFileSync(__filename); + module.exports = { process(sourceText) { return { code: rewrite(sourceText) }; }, + // Jest caches transformed files; a change here must invalidate them. + getCacheKey(sourceText, filename) { + return crypto.createHash('md5').update(self).update(sourceText).update(filename).digest('hex'); + }, }; diff --git a/tests/unit/service-worker.test.js b/tests/unit/service-worker.test.js index 43a15d7..040e9c0 100644 --- a/tests/unit/service-worker.test.js +++ b/tests/unit/service-worker.test.js @@ -11,12 +11,16 @@ jest.mock('../../scripts/background/firebase-init.js', () => ({ jest.mock('firebase/auth/web-extension', () => ({ onAuthStateChanged: (auth, cb) => { cb(auth.currentUser); return () => {}; }, signInWithEmailAndPassword: async () => ({}), + sendPasswordResetEmail: async () => {}, signOut: async () => {}, }), { virtual: true }); -jest.mock('../../scripts/background/sync.js', () => ({ syncToCloud: jest.fn(async () => ({ written: 1 })) })); +jest.mock('../../scripts/background/sync.js', () => { + const flush = jest.fn(async () => ({ entries: 0, pages: 0, runs: 0, products: 0, writes: 0 })); + return { createSync: () => ({ flush, pending: async () => 0 }), isAuthError: () => false, __flush: flush }; +}); require('fake-indexeddb/auto'); -const { syncToCloud } = require('../../scripts/background/sync.js'); +const { __flush: flush } = require('../../scripts/background/sync.js'); const DB = require('../../scripts/background/db'); const fs = require('fs'); @@ -58,7 +62,7 @@ test('one router answers every message', () => { test('Export to ProScan is refused while cloud sync is off', async () => { const resp = await send({ type: 'PROSCAN_EXPORT' }); expect(resp.error).toMatch(/not available/); - expect(syncToCloud).not.toHaveBeenCalled(); + expect(flush).not.toHaveBeenCalled(); }); test('messages meant for the popup or a tab are left alone', async () => { diff --git a/tests/unit/sync.test.js b/tests/unit/sync.test.js index d597226..799195a 100644 --- a/tests/unit/sync.test.js +++ b/tests/unit/sync.test.js @@ -1,217 +1,204 @@ /** - * @fileoverview Offline unit test for scripts/background/sync.js syncToCloud. + * @jest-environment node * - * sync.js is authored as ESM and bundled into the service worker by esbuild. - * Here it is loaded through a tiny scoped Jest transform (see jest.config.js) - * that rewrites its import/export to CommonJS, with firebase/firestore and the - * firebase-init db fully mocked so the test never touches a real Firebase. - * - * We assert syncToCloud issues the cloud-schema writes the dashboard depends on - * (products/{asin} with latest + delta, products/{asin}/history/daily, - * runs/{runId}, sources/{sourceId}) and returns {written, runId, products}. + * scripts/background/sync.js over fake-indexeddb and an in-memory Firestore + * that applies set and mergeFields the way Firestore does. The contract test + * (tests/contract) runs the same module against the real emulator and rules. */ - -// ── Mock firebase/firestore: record every doc path and set() payload. ── -// All recording state lives INSIDE the factory (Jest hoists jest.mock above -// the file, so the factory may not close over outer variables) and is exposed -// via __writes / __committed / __reset on the mocked module. -jest.mock('firebase/firestore', () => { - const writes = []; - const state = { committed: 0 }; - const makeBatch = () => ({ - set(ref, data, opts) { - writes.push({ path: ref.__path, data, opts }); - }, - async commit() { - state.committed++; - }, - }); - return { - // doc(db, 'workspaces', uid, 'products', asin, ...) -> ref carrying its path - doc: (_db, ...segments) => ({ __path: segments.join('/') }), - writeBatch: () => makeBatch(), - serverTimestamp: () => ({ __sentinel: 'serverTimestamp' }), - arrayUnion: (...vals) => ({ __arrayUnion: vals }), - Timestamp: { fromDate: (d) => ({ __ts: d.toISOString() }) }, - // test-only handles - __writes: writes, - __state: state, - __reset: () => { - writes.length = 0; - state.committed = 0; +jest.mock('firebase/firestore', () => ({})); + +require('fake-indexeddb/auto'); +const DB = require('../../scripts/background/db'); +const { createSync, isAuthError } = require('../../scripts/background/sync.js'); + +const START = Date.parse('2026-06-21T10:00:00.000Z'); +const RUN_ID = `k_wireless-mouse_${START}`; + +/** A tiny Firestore: documents in a Map, keyed by path. */ +function fakeFirestore({ failCommit = () => false } = {}) { + const docs = new Map(); + let commits = 0; + class FieldPath { constructor(...segs) { this.segs = segs; } } + const setPath = (obj, segs, value) => { + let o = obj; + segs.slice(0, -1).forEach((s) => { o[s] = o[s] && typeof o[s] === 'object' ? o[s] : {}; o = o[s]; }); + o[segs[segs.length - 1]] = value; + }; + const getPath = (obj, segs) => segs.reduce((o, s) => (o == null ? undefined : o[s]), obj); + const resolve = (v) => (v && v.__union ? v.__union : v); + const copy = (v) => (Array.isArray(v) ? v.map(copy) + : v && typeof v === 'object' ? Object.fromEntries(Object.entries(v).map(([k, x]) => [k, copy(x)])) : v); + const fs = { + doc: (_db, ...path) => ({ path: path.join('/') }), + getDoc: async (ref) => ({ exists: () => docs.has(ref.path) }), + Timestamp: { fromMillis: (ms) => ({ toMillis: () => ms, ms }) }, + arrayUnion: (...vals) => ({ __union: vals }), + FieldPath, + writeBatch: () => { + const ops = []; + return { + set(ref, data, opts) { ops.push({ ref, data, opts }); }, + async commit() { + if (failCommit(ops)) throw Object.assign(new Error('unavailable'), { code: 'unavailable' }); + commits++; + for (const { ref, data, opts } of ops) { + if (!opts) { docs.set(ref.path, copy(data)); continue; } + const cur = docs.get(ref.path) || {}; + for (const f of opts.mergeFields) { + const segs = f instanceof FieldPath ? f.segs : [f]; + let v = getPath(data, segs); + if (v && v.__union) v = [...new Set([...(getPath(cur, segs) || []), ...resolve(v)])]; + setPath(cur, segs, copy(v)); + } + docs.set(ref.path, cur); + } + }, + }; }, }; -}); - -// ── Mock the firebase-init db (sync.js imports { db } from it) ── -jest.mock('../../scripts/background/firebase-init.js', () => ({ db: { __db: true } }), { - virtual: true, -}); - -const firestoreMock = require('firebase/firestore'); -const writes = firestoreMock.__writes; -const { syncToCloud } = require('../../scripts/background/sync.js'); - -const UID = 'user_abc'; + return { fs, docs, commits: () => commits }; +} -function seedBundle() { - return { - scrapeRunMeta: { - type: 'keyword', - keyword: 'wireless mouse', - sellerId: null, - url: 'https://www.amazon.com/s?k=wireless+mouse', - startedAt: '2026-06-21T10:00:00.000Z', +let dbCount = 0; +async function storeWith({ products = 3, perPage = 2, uid = 'u1', state = 'running' } = {}) { + const store = await DB.open({ name: `sync-test-${++dbCount}` }); + const pages = Math.ceil(products / perPage); + const ops = [{ + store: 'runs', + put: { + runId: RUN_ID, sourceId: 'k_wireless-mouse', dayKey: '2026-06-21', state, reason: state === 'done' ? 'complete' : null, + source: { type: 'keyword', sellerId: null, keyword: 'wireless mouse', url: 'https://www.amazon.com/s?k=wireless+mouse' }, + startedAt: START, finishedAt: state === 'done' ? START + 1000 : null, page: pages, maxPages: 20, }, - scrapeRunPages: [{ pageIndex: 0 }, { pageIndex: 1 }], - syncQueue: [ - { - // first-sight product - asin: 'B0NEW00001', - name: 'New Mouse', - url: 'https://www.amazon.com/dp/B0NEW00001', - priceCents: 1999, - rating: 4.5, - reviewCount: 120, - isPrime: true, - scrapedAt: '2026-06-21T10:00:05.000Z', - delta: { isNew: true, dPriceCents: null, dRating: null, dReviews: null }, - }, - { - // returning product with a price drop - asin: 'B0SEEN00002', - name: 'Seen Mouse', - url: 'https://www.amazon.com/dp/B0SEEN00002', - priceCents: 1799, - rating: 4.2, - reviewCount: 300, - isPrime: false, - scrapedAt: '2026-06-21T10:00:06.000Z', - delta: { isNew: false, dPriceCents: -200, dRating: 0, dReviews: 20 }, + }]; + for (let n = 0; n < products; n++) { + const page = Math.floor(n / perPage) + 1; + ops.push({ + store: 'products', + put: { + runId: RUN_ID, n, pageIndex: page, asin: `B0${String(n).padStart(8, '0')}`, name: `Mouse ${n}`, + priceCents: 1000 + n, rating: 4.5, reviewCount: 10, isPrime: true, sponsored: false, organicRank: n + 1, + scrapedAt: new Date(START + n).toISOString(), placements: [{ page, position: 1, sponsored: false, rank: n + 1 }], + delta: { isNew: true, dPriceCents: null, dRating: null, dReviews: null }, prev: null, }, - ], - }; + }); + } + for (let page = 1; page <= pages; page++) { + ops.push({ store: 'placements', put: { runId: RUN_ID, pageIndex: page, count: perPage, placements: perPage, kind: 'results', scrapedAt: new Date(START).toISOString(), total: products } }); + ops.push({ store: 'outbox', put: { runId: RUN_ID, kind: 'page', pageIndex: page, uid, queuedAt: START } }); + } + if (state === 'done') ops.push({ store: 'outbox', put: { runId: RUN_ID, kind: 'run', uid, queuedAt: START } }); + await store.write(ops); + return store; } -const findWrite = (path) => writes.find((w) => w.path === path); - -describe('sync.js — syncToCloud', () => { - beforeEach(() => { - firestoreMock.__reset(); - }); - - test('returns {written:0, runId:null, products:0} for an empty queue', async () => { - const res = await syncToCloud(UID, { syncQueue: [] }); - expect(res).toEqual({ written: 0, runId: null, products: 0 }); - expect(writes).toHaveLength(0); - }); - - test('derives a canonical keyword runId/sourceId from run meta', async () => { - const res = await syncToCloud(UID, seedBundle()); - const sourceId = 'k_wireless-mouse'; - const startMs = Date.parse('2026-06-21T10:00:00.000Z'); - expect(res.runId).toBe(`${sourceId}_${startMs}`); - }); +const quiet = { warn() {}, log() {}, error() {} }; + +test('a flush writes every queued page and empties the outbox', async () => { + const store = await storeWith({ products: 3, perPage: 2, state: 'done' }); + const cloud = fakeFirestore(); + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, log: quiet }); + const totals = await sync.flush('u1'); + // page 1: chunk + 2x2 + run + source; page 2: chunk + 1x2 + run + source; end: run + source + expect(totals).toEqual({ entries: 3, pages: 2, runs: 1, products: 3, writes: 7 + 5 + 2 }); + expect(await store.count('outbox')).toBe(0); + const run = cloud.docs.get(`workspaces/u1/runs/${RUN_ID}`); + expect(run).toMatchObject({ status: 'complete', pagesDone: 2, pagesPlanned: 2, counters: { uniqueAsins: 3 } }); + expect(cloud.docs.get('workspaces/u1/products/B000000002').sourceIds).toEqual(['k_wireless-mouse']); + expect(cloud.docs.has(`workspaces/u1/runs/${RUN_ID}/pages/p0002`)).toBe(true); + expect(cloud.docs.get('workspaces/u1/sources/k_wireless-mouse')).toMatchObject({ lastRunId: RUN_ID, catalogSize: 3 }); +}); - test('writes products/{asin} with latest (incl. dayKey) + sourceIds', async () => { - await syncToCloud(UID, seedBundle()); - const p = findWrite(`workspaces/${UID}/products/B0NEW00001`); - expect(p).toBeDefined(); - expect(p.opts).toEqual({ merge: true }); - expect(p.data.asin).toBe('B0NEW00001'); - expect(p.data.mk).toBe('US'); - expect(p.data.latest.p).toBe(1999); - expect(p.data.latest.pr).toBe(1); // isPrime -> 1 - expect(p.data.latest.dayKey).toBe('2026-06-21'); - expect(p.data.sourceIds).toEqual({ __arrayUnion: ['k_wireless-mouse'] }); - // first-sight stamps present on a new product - expect(p.data.firstRunId).toBeDefined(); - expect(p.data.firstSeenAt).toBeDefined(); - }); +test('a new product gets firstSeenAt, and a second flush never moves it (F-29b)', async () => { + const store = await storeWith({ products: 1, perPage: 1 }); + const cloud = fakeFirestore(); + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, log: quiet }); + await sync.flush('u1'); + const first = cloud.docs.get('workspaces/u1/products/B000000000'); + expect(first.firstSeenAt.ms).toBe(START); + expect(first.firstRunId).toBe(RUN_ID); + + // The same product again, as after a reinstall: new locally, known in the cloud. + cloud.docs.get('workspaces/u1/products/B000000000').firstSeenAt = { ms: 1, toMillis: () => 1 }; + await store.write([{ store: 'outbox', put: { runId: RUN_ID, kind: 'page', pageIndex: 1, uid: 'u1', queuedAt: START } }]); + await sync.flush('u1'); + expect(cloud.docs.get('workspaces/u1/products/B000000000').firstSeenAt.ms).toBe(1); +}); - test('writes a delta block for a returning product, not for a new one', async () => { - await syncToCloud(UID, seedBundle()); - const seen = findWrite(`workspaces/${UID}/products/B0SEEN00002`); - const fresh = findWrite(`workspaces/${UID}/products/B0NEW00001`); - expect(seen.data.delta).toMatchObject({ p: -200, v: 20 }); - expect(seen.data.delta.pPct).toBeCloseTo(-10, 1); // -200 on prior 1999 - expect(fresh.data.delta).toBeUndefined(); - expect(fresh.data.firstRunId).toBeDefined(); - // returning product gets no first-sight stamp - expect(seen.data.firstRunId).toBeUndefined(); - }); +test('an entry leaves the outbox only after its own commit (F-22)', async () => { + const store = await storeWith({ products: 4, perPage: 2 }); + let calls = 0; + const cloud = fakeFirestore({ failCommit: () => ++calls === 2 }); + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, log: quiet }); + await expect(sync.flush('u1')).rejects.toMatchObject({ code: 'unavailable' }); + expect((await store.getAll('outbox')).map((e) => e.pageIndex)).toEqual([2]); + await sync.flush('u1'); + expect(await store.count('outbox')).toBe(0); + expect(cloud.docs.has(`workspaces/u1/runs/${RUN_ID}/pages/p0002`)).toBe(true); +}); - test('writes the date-keyed history/daily point per product', async () => { - await syncToCloud(UID, seedBundle()); - const h = findWrite(`workspaces/${UID}/products/B0NEW00001/history/daily`); - expect(h).toBeDefined(); - expect(h.opts).toEqual({ merge: true }); - expect(h.data.d['2026-06-21']).toMatchObject({ p: 1999, r: 4.5, v: 120, pr: 1 }); - }); +test('a page queued during a flush is written before it returns (F-22)', async () => { + const store = await storeWith({ products: 2, perPage: 2 }); + const cloud = fakeFirestore(); + let appended = false; + const onBatch = async () => { + if (appended) return; + appended = true; + await store.write([ + { store: 'products', put: { runId: RUN_ID, n: 2, pageIndex: 2, asin: 'B0LATE0001', priceCents: 5, isPrime: false, placements: [{ page: 2, position: 1, sponsored: false, rank: 3 }], delta: { isNew: true }, prev: null, scrapedAt: new Date(START).toISOString() } }, + { store: 'placements', put: { runId: RUN_ID, pageIndex: 2, count: 1, placements: 1, kind: 'last', scrapedAt: new Date(START).toISOString() } }, + { store: 'outbox', put: { runId: RUN_ID, kind: 'page', pageIndex: 2, uid: 'u1', queuedAt: START } }, + ]); + }; + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, onBatch, log: quiet }); + const totals = await sync.flush('u1'); + expect(totals.pages).toBe(2); + expect(await store.count('outbox')).toBe(0); + expect(cloud.docs.get('workspaces/u1/products/B0LATE0001').latest.p).toBe(5); +}); - test('writes the run header doc with counters', async () => { - await syncToCloud(UID, seedBundle()); - const startMs = Date.parse('2026-06-21T10:00:00.000Z'); - const runId = `k_wireless-mouse_${startMs}`; - const run = findWrite(`workspaces/${UID}/runs/${runId}`); - expect(run).toBeDefined(); - expect(run.data.runId).toBe(runId); - expect(run.data.sourceId).toBe('k_wireless-mouse'); - expect(run.data.status).toBe('complete'); - expect(run.data.dayKey).toBe('2026-06-21'); - expect(run.data.counters.placements).toBe(2); - expect(run.data.counters.uniqueAsins).toBe(2); - expect(run.data.counters.newSeen).toBe(1); - expect(run.data.pagesDone).toBe(2); - }); +test('a second flush while one runs waits for it and does not write twice', async () => { + const store = await storeWith({ products: 2, perPage: 1 }); + const cloud = fakeFirestore(); + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, log: quiet }); + const [a, b] = await Promise.all([sync.flush('u1'), sync.flush('u1')]); + expect(a).toBe(b); + expect(a.pages).toBe(2); + expect(cloud.commits()).toBe(2); +}); - test('counts every placement and sponsored card, and skips a missing name (F-27, F-28)', async () => { - const bundle = seedBundle(); - bundle.syncQueue[0].placements = [ - { page: 1, position: 1, sponsored: true, rank: null }, - { page: 1, position: 4, sponsored: false, rank: 3 }, - ]; - bundle.syncQueue[1].name = null; - bundle.syncQueue[1].url = 'https://www.amazon.com/sspa/click?url=x'; - await syncToCloud(UID, bundle); - const startMs = Date.parse('2026-06-21T10:00:00.000Z'); - const run = findWrite(`workspaces/${UID}/runs/k_wireless-mouse_${startMs}`); - expect(run.data.counters).toMatchObject({ placements: 3, uniqueAsins: 2, sponsored: 1 }); - const seen = findWrite(`workspaces/${UID}/products/B0SEEN00002`); - expect(seen.data).not.toHaveProperty('name'); - expect(seen.data.url).toBe('https://www.amazon.com/dp/B0SEEN00002'); - }); +test('entries of another account are left alone (F-29e)', async () => { + const store = await storeWith({ products: 2, perPage: 2, uid: 'someone-else' }); + const cloud = fakeFirestore(); + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, log: quiet }); + expect(await sync.pending('u1')).toBe(0); + expect(await sync.flush('u1')).toMatchObject({ entries: 0, writes: 0 }); + expect(cloud.docs.size).toBe(0); + expect(await sync.pending('someone-else')).toBe(1); + expect(await sync.flush(null)).toMatchObject({ entries: 0 }); +}); - test('writes the source spine doc with lastRunId', async () => { - await syncToCloud(UID, seedBundle()); - const src = findWrite(`workspaces/${UID}/sources/k_wireless-mouse`); - expect(src).toBeDefined(); - expect(src.data.sourceId).toBe('k_wireless-mouse'); - expect(src.data.type).toBe('keyword'); - expect(src.data.keyword).toBe('wireless mouse'); - expect(src.data.lastRunId).toMatch(/^k_wireless-mouse_/); - }); +test('big pages are split into batches under the Firestore limit', async () => { + const store = await storeWith({ products: 60, perPage: 60 }); + const cloud = fakeFirestore(); + const sizes = []; + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, batchLimit: 50, onBatch: async ({ writes }) => sizes.push(writes), log: quiet }); + const totals = await sync.flush('u1'); + expect(totals.writes).toBe(1 + 120 + 2); + expect(sizes).toEqual([50, 50, 23]); +}); - test('returns the expected write tally (2 per product + 2 headers)', async () => { - const res = await syncToCloud(UID, seedBundle()); - expect(res.products).toBe(2); - expect(res.written).toBe(2 * 2 + 2); // products*2 + run + source - // every batch was committed (1 product batch + 1 header batch) - expect(firestoreMock.__state.committed).toBe(2); - }); +test('an entry whose run is gone is dropped', async () => { + const store = await DB.open({ name: `sync-test-${++dbCount}` }); + await store.write([{ store: 'outbox', put: { runId: 'gone', kind: 'page', pageIndex: 1, uid: 'u1' } }]); + const sync = createSync({ db: {}, openStore: async () => store, fs: fakeFirestore().fs, log: quiet }); + expect(await sync.flush('u1')).toMatchObject({ entries: 1, pages: 0 }); + expect(await store.count('outbox')).toBe(0); +}); - test('derives a storefront sourceId (s_{sellerId}) from seller meta', async () => { - const bundle = seedBundle(); - bundle.scrapeRunMeta = { - type: 'storefront', - sellerId: 'A123XYZ', - keyword: null, - startedAt: '2026-06-21T10:00:00.000Z', - }; - await syncToCloud(UID, bundle); - const src = findWrite(`workspaces/${UID}/sources/s_A123XYZ`); - expect(src).toBeDefined(); - expect(src.data.type).toBe('storefront'); - expect(src.data.sellerId).toBe('A123XYZ'); - }); +test('auth errors are told apart from network errors', () => { + expect(isAuthError({ code: 'permission-denied' })).toBe(true); + expect(isAuthError({ code: 'auth/user-token-expired' })).toBe(true); + expect(isAuthError({ code: 'unavailable' })).toBe(false); + expect(isAuthError(null)).toBe(false); }); From 65fda0586eabe497cc67ce4e2f30a7bbf60b25e5 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:35:23 -0400 Subject: [PATCH 06/18] Contract test the real sync module against the emulator and the dashboard rules --- package.json | 1 + tests/contract/sync.contract.mjs | 288 +++++++++++++++++++++++++++++++ tools/contract.mjs | 109 ++++++++++++ 3 files changed, 398 insertions(+) create mode 100644 tests/contract/sync.contract.mjs create mode 100644 tools/contract.mjs diff --git a/package.json b/package.json index dbad2d1..af2da63 100644 --- a/package.json +++ b/package.json @@ -10,6 +10,7 @@ "test:integration": "jest tests/integration --verbose", "test:tools": "node --test --test-concurrency=1 \"tools/tests/*.test.mjs\"", "test:e2e": "playwright test", + "test:contract": "node tools/contract.mjs", "test:coverage": "jest --coverage", "lock": "node tools/permission-lock.mjs && node tools/version-gate.mjs", "scan:secrets": "node tools/secret-scan.mjs", diff --git a/tests/contract/sync.contract.mjs b/tests/contract/sync.contract.mjs new file mode 100644 index 0000000..87a0ced --- /dev/null +++ b/tests/contract/sync.contract.mjs @@ -0,0 +1,288 @@ +// The sync contract: the real engine fills IndexedDB (fake-indexeddb), the +// real sync module drains it into the Firestore emulator, and the dashboard +// repo's firestore.rules judge every write. Run through tools/contract.mjs, +// which starts the emulators. +// +// Queue sizes 1, 201 and 600, a page appended in the middle of a flush, +// replace semantics across runs, create-only firstSeenAt, and a write count +// per run. + +import 'fake-indexeddb/auto'; +import { test, before, after } from 'node:test'; +import assert from 'node:assert/strict'; +import { createRequire } from 'node:module'; +import { initializeApp, deleteApp } from 'firebase/app'; +import { + getAuth, connectAuthEmulator, createUserWithEmailAndPassword, signInWithEmailAndPassword, signOut, +} from 'firebase/auth'; +import { + initializeFirestore, connectFirestoreEmulator, doc, getDoc, collection, getCountFromServer, terminate, +} from 'firebase/firestore'; +import { createSync, isAuthError } from '../../scripts/background/sync.js'; +import * as Schema from '../../packages/schema/index.js'; + +const require = createRequire(import.meta.url); +const DB = require('../../scripts/background/db.js'); +const { createEngine } = require('../../scripts/background/engine.js'); + +const AUTH_HOST = process.env.FIREBASE_AUTH_EMULATOR_HOST; +const FS_HOST = process.env.FIRESTORE_EMULATOR_HOST; +if (!AUTH_HOST || !FS_HOST) throw new Error('run through tools/contract.mjs, which starts the emulators'); + +const app = initializeApp({ projectId: 'demo-proscan', apiKey: 'demo-key' }); +const auth = getAuth(app); +const db = initializeFirestore(app, {}); +connectAuthEmulator(auth, `http://${AUTH_HOST}`, { disableWarnings: true }); +connectFirestoreEmulator(db, FS_HOST.split(':')[0], Number(FS_HOST.split(':')[1])); + +const quiet = { log() {}, warn() {}, error() {} }; +const DAY = 86400000; +let dbSeq = 0; + +function memoryArea(init = {}) { + let data = JSON.parse(JSON.stringify(init)); + return { + async get(keys) { + const list = keys == null ? Object.keys(data) : [].concat(keys); + return Object.fromEntries(list.filter((k) => data[k] !== undefined).map((k) => [k, data[k]])); + }, + async set(items) { Object.assign(data, JSON.parse(JSON.stringify(items))); }, + async remove(keys) { [].concat(keys).forEach((k) => delete data[k]); }, + }; +} + +/** The real engine over a fresh IndexedDB, with a tab that always answers. */ +function extension(uid) { + const name = `contract-${process.pid}-${++dbSeq}`; + const clock = { t: 0 }; + let url = ''; + const chrome = { + storage: { session: memoryArea(), local: memoryArea({ account: { uid }, settings: { maxPages: 400 } }) }, + runtime: { lastError: null }, + tabs: { + sendMessage(tabId, message, opts, cb) { + setImmediate(() => cb(message.type === 'PING' ? { ok: true, kind: 'results', url } : { ok: true })); + }, + async update() { return {}; }, + async get(id) { return { id, url }; }, + }, + }; + const engine = createEngine({ + chrome, openDb: () => DB.open({ name }), now: () => clock.t, random: () => 0, + setTimer: () => 0, clearTimer: () => {}, flags: { CLOUD_SYNC: true }, log: quiet, + }); + const sender = { tab: { id: 1 } }; + + /** Starts a keyword run at `startMs`. */ + async function start(keyword, startMs) { + clock.t = startMs; + url = `https://www.amazon.com/s?k=${encodeURIComponent(keyword)}`; + const resp = await engine.start({ tabId: 1 }); + assert.equal(resp.ok, true, JSON.stringify(resp)); + return resp.runId; + } + + /** Reports page `page` of the live run; `last` ends the run. */ + async function page(runId, pageNo, products, { last = false, total = 0 } = {}) { + if (pageNo > 1) { + clock.t += 5000; + await engine.tick(); + } + const result = { + kind: last ? 'last' : 'results', + products: products.map((p) => ({ ...p, scrapedAt: new Date(clock.t).toISOString() })), + placements: products.length, + nextHref: last ? null : `${url}&page=${pageNo + 1}`, + total, + fill: { asin: 1, title: 1, price: 1 }, + }; + const resp = await engine.pageResult({ runId, page: pageNo, url: pageNo === 1 ? url : `${url}&page=${pageNo}`, result }, sender); + assert.equal(resp.ok, true, JSON.stringify(resp)); + } + + /** A whole run of `products`, `perPage` to a page. */ + async function scrape(keyword, startMs, products, perPage) { + const runId = await start(keyword, startMs); + const pages = Math.ceil(products.length / perPage); + for (let i = 0; i < pages; i++) { + await page(runId, i + 1, products.slice(i * perPage, (i + 1) * perPage), { last: i === pages - 1, total: products.length }); + } + return runId; + } + + const sync = (opts = {}) => createSync({ db, openStore: () => engine.db(), log: quiet, ...opts }); + return { engine, start, page, scrape, sync, store: () => engine.db() }; +} + +const asinOf = (i) => `B0${i.toString(36).toUpperCase().padStart(8, '0')}`; +function products(n, { price = (i) => 1000 + i } = {}) { + return Array.from({ length: n }, (_, i) => { + const cents = price(i); + return { + asin: asinOf(i), name: `Product ${i}`, price: cents === null ? null : `$${(cents / 100).toFixed(2)}`, + priceCents: cents, currency: cents === null ? null : 'USD', rating: 4.5, reviewCount: 100 + i, isPrime: i % 2 === 0, + sponsored: false, url: `https://www.amazon.com/dp/${asinOf(i)}`, + img: `https://m.media-amazon.com/images/I/${i}.jpg`, + placements: [{ position: 1, sponsored: false, rank: 1 }], + }; + }); +} + +/** Writes an entry takes: chunk + 2 per product + run + source for a page; run + source at the end. */ +const expectedWrites = (pageSizes) => pageSizes.reduce((n, k) => n + 1 + 2 * k + 2, 0) + 2; + +async function account(email) { + const cred = await createUserWithEmailAndPassword(auth, email, 'contract-pass-1'); + return cred.user.uid; +} + +const ws = (uid, ...rest) => doc(db, 'workspaces', uid, ...rest); +const count = async (uid, ...path) => (await getCountFromServer(collection(db, 'workspaces', uid, ...path))).data().count; + +before(async () => { + const rules = process.env.PROSCAN_CONTRACT_RULES || '(unknown)'; + console.log(`# rules: ${rules}`); +}); + +after(async () => { + await signOut(auth).catch(() => {}); + await terminate(db).catch(() => {}); + await deleteApp(app).catch(() => {}); +}); + +test('queue size 1: one product on one page', async () => { + const uid = await account(`one-${Date.now()}@contract.test`); + const ext = extension(uid); + const start = Date.parse('2026-06-09T14:00:00Z'); + const runId = await ext.scrape('single mug', start, products(1), 60); + assert.equal(runId, `k_single-mug_${start}`); + + const totals = await ext.sync().flush(uid); + assert.deepEqual(totals, { entries: 2, pages: 1, runs: 1, products: 1, writes: expectedWrites([1]) }); + assert.equal(await (await ext.store()).count('outbox'), 0); + + const p = (await getDoc(ws(uid, 'products', asinOf(0)))).data(); + assert.deepEqual(Schema.validateProduct({ ...p }), []); + assert.equal(p.latest.p, 1000); + assert.equal(p.latest.runId, runId); + assert.equal(p.firstRunId, runId); + assert.equal(p.img, 'https://m.media-amazon.com/images/I/0.jpg'); + const run = (await getDoc(ws(uid, 'runs', runId))).data(); + assert.deepEqual(Schema.validateRun(run), []); + assert.equal(run.status, 'complete'); + assert.equal(run.pagesPlanned, 1); + const chunk = (await getDoc(ws(uid, 'runs', runId, 'pages', 'p0001'))).data(); + assert.deepEqual(Schema.validatePage(chunk), []); + assert.deepEqual(Object.keys(chunk.items), [asinOf(0)]); + const hist = (await getDoc(ws(uid, 'products', asinOf(0), 'history', 'daily'))).data(); + assert.deepEqual(Schema.validateHistory(hist), []); + const src = (await getDoc(ws(uid, 'sources', 'k_single-mug'))).data(); + assert.deepEqual(Schema.validateSource(src), []); + assert.equal(src.lastRunId, runId); +}); + +test('queue size 201, with a page appended in the middle of the flush', async () => { + const uid = await account(`mid-${Date.now()}@contract.test`); + const ext = extension(uid); + const all = products(201); + const start = Date.parse('2026-06-10T09:00:00Z'); + const runId = await ext.start('yoga mat', start); + for (let i = 0; i < 4; i++) await ext.page(runId, i + 1, all.slice(i * 48, (i + 1) * 48), { total: 201 }); + + let appended = false; + const sync = ext.sync({ + onBatch: async () => { + if (appended) return; + appended = true; + // The last page lands while page 1 is being written. + await ext.page(runId, 5, all.slice(192), { last: true, total: 201 }); + }, + }); + const totals = await sync.flush(uid); + assert.equal(appended, true); + assert.equal(totals.pages, 5); + assert.equal(totals.products, 201); + assert.equal(totals.writes, expectedWrites([48, 48, 48, 48, 9])); + assert.equal(await (await ext.store()).count('outbox'), 0); + + assert.equal(await count(uid, 'products'), 201); + assert.equal(await count(uid, 'runs', runId, 'pages'), 5); + const run = (await getDoc(ws(uid, 'runs', runId))).data(); + assert.equal(run.status, 'complete'); + assert.equal(run.pagesDone, 5); + assert.equal(run.counters.uniqueAsins, 201); + assert.equal(run.totalResultsOnSerp, 201); + const last = (await getDoc(ws(uid, 'products', asinOf(200)))).data(); + assert.equal(last.latest.p, 1200); +}); + +test('queue size 600: two runs a week apart replace latest and delta, and keep firstSeenAt', async () => { + const uid = await account(`six-${Date.now()}@contract.test`); + const ext = extension(uid); + const day1 = Date.parse('2026-06-01T15:00:00Z'); + const day8 = day1 + 7 * DAY; + const run1 = await ext.scrape('garden hose', day1, products(300), 60); + // Week two: every 10th price fails to parse, the rest drop by 100 cents. + const run2 = await ext.scrape('garden hose', day8, products(300, { price: (i) => (i % 10 === 0 ? null : 900 + i) }), 60); + + const totals = await ext.sync().flush(uid); + assert.equal(totals.pages, 10); + assert.equal(totals.runs, 2); + assert.equal(totals.products, 600); + assert.equal(totals.writes, 2 * expectedWrites([60, 60, 60, 60, 60])); + assert.equal(await count(uid, 'products'), 300); + + const failed = (await getDoc(ws(uid, 'products', asinOf(10)))).data(); + assert.equal(failed.latest.runId, run2); + assert.equal('p' in failed.latest, false, 'a price that failed to parse is not last week\'s price (F-21)'); + assert.equal('p' in (failed.delta || {}), false, 'no price delta without a price (F-21)'); + assert.equal(failed.firstRunId, run1); + + const moved = (await getDoc(ws(uid, 'products', asinOf(11)))).data(); + assert.equal(moved.latest.p, 911); + assert.equal(moved.prev.p, 1011); + assert.deepEqual(moved.delta, { p: -100, pPct: -9.9, r: 0, v: 0, days: 7 }); + assert.equal(moved.firstRunId, run1); + assert.equal(moved.firstSeenAt.toMillis(), day1); + + const hist = (await getDoc(ws(uid, 'products', asinOf(11), 'history', 'daily'))).data(); + assert.deepEqual(Object.keys(hist.d).sort(), [Schema.dayKeyOf(day1, new Date(day1).getTimezoneOffset()), Schema.dayKeyOf(day8, new Date(day8).getTimezoneOffset())].sort()); + + for (const runId of [run1, run2]) { + const run = (await getDoc(ws(uid, 'runs', runId))).data(); + assert.equal(run.status, 'complete'); + assert.equal(run.sourceId, 'k_garden-hose'); + assert.equal(await count(uid, 'runs', runId, 'pages'), 5); + } + assert.equal((await getDoc(ws(uid, 'runs', run1))).data().counters.newSeen, 300); + assert.equal((await getDoc(ws(uid, 'runs', run2))).data().counters.newSeen, 0); + assert.equal((await getDoc(ws(uid, 'sources', 'k_garden-hose'))).data().lastRunId, run2); +}); + +test('a reinstall does not move firstSeenAt (F-29b)', async () => { + const uid = await account(`re-${Date.now()}@contract.test`); + const first = extension(uid); + const day1 = Date.parse('2026-05-01T12:00:00Z'); + const run1 = await first.scrape('desk lamp', day1, products(5), 60); + await first.sync().flush(uid); + + // A fresh install on another computer: no lastValues, every product looks new. + const second = extension(uid); + await second.scrape('desk lamp', day1 + 30 * DAY, products(5), 60); + await second.sync().flush(uid); + + const p = (await getDoc(ws(uid, 'products', asinOf(3)))).data(); + assert.equal(p.firstRunId, run1); + assert.equal(p.firstSeenAt.toMillis(), day1); +}); + +test('another account cannot write this account\'s queue, and the entries stay', async () => { + const owner = await account(`a-${Date.now()}@contract.test`); + const ext = extension(owner); + await ext.scrape('phone case', Date.parse('2026-06-12T10:00:00Z'), products(3), 60); + await signOut(auth); + await account(`b-${Date.now()}@contract.test`); + + await assert.rejects(ext.sync().flush(owner), (err) => isAuthError(err)); + assert.equal(await (await ext.store()).count('outbox'), 2); +}); diff --git a/tools/contract.mjs b/tools/contract.mjs new file mode 100644 index 0000000..a6139bf --- /dev/null +++ b/tools/contract.mjs @@ -0,0 +1,109 @@ +// tools/contract.mjs: runs the sync contract test against the Firebase +// emulators, with the dashboard repo's firestore.rules. +// +// node tools/contract.mjs +// +// The rules come from PROSCAN_RULES, else ../web/firestore.rules or +// ../proscan-web/firestore.rules next to this repo. Ports default away from +// the dashboard's (EMU_AUTH_PORT 9299, EMU_FIRESTORE_PORT 8288) so both can +// be checked out side by side. Project is demo-proscan: emulators only. + +import { spawn, execSync } from 'node:child_process'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); +const TEST = path.join(ROOT, 'tests', 'contract', 'sync.contract.mjs'); +const TIMEOUT_MS = 7 * 60 * 1000; + +const port = (name, fallback) => { + const n = Number(process.env[name] || fallback); + if (!Number.isInteger(n) || n <= 0) throw new Error(`${name} is not a port`); + return n; +}; +const ports = { + auth: port('EMU_AUTH_PORT', 9299), + firestore: port('EMU_FIRESTORE_PORT', 8288), + websocket: port('EMU_WEBSOCKET_PORT', 9250), + hub: port('EMU_HUB_PORT', 4488), + logging: port('EMU_LOGGING_PORT', 4588), +}; + +export function findRules(env = process.env) { + const candidates = env.PROSCAN_RULES + ? [path.resolve(env.PROSCAN_RULES)] + : [path.join(ROOT, '..', 'web', 'firestore.rules'), path.join(ROOT, '..', 'proscan-web', 'firestore.rules')]; + const found = candidates.find((p) => fs.existsSync(p)); + if (!found) throw new Error(`firestore.rules not found; tried ${candidates.join(', ')}. Set PROSCAN_RULES.`); + return found; +} + +/** Kills whatever still listens on our ports; on Windows java outlives firebase. */ +function freePorts() { + if (process.platform !== 'win32') return; + let out = ''; + try { + out = execSync('netstat -ano -p tcp', { encoding: 'utf8' }); + } catch { + return; + } + const wanted = new Set(Object.values(ports).map(String)); + const pids = new Set(); + for (const line of out.split('\n')) { + const m = line.trim().match(/^TCP\s+127\.0\.0\.1:(\d+)\s+\S+\s+LISTENING\s+(\d+)/); + if (m && wanted.has(m[1])) pids.add(m[2]); + } + for (const pid of pids) { + try { + execSync(`taskkill /PID ${pid} /F`, { stdio: 'ignore' }); + } catch { + /* already gone */ + } + } +} + +function main() { + const rules = findRules(); + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'proscan-contract-')); + fs.copyFileSync(rules, path.join(dir, 'firestore.rules')); + fs.writeFileSync(path.join(dir, 'firebase.json'), JSON.stringify({ + firestore: { rules: 'firestore.rules' }, + emulators: { + auth: { host: '127.0.0.1', port: ports.auth }, + firestore: { host: '127.0.0.1', port: ports.firestore, websocketPort: ports.websocket }, + hub: { host: '127.0.0.1', port: ports.hub }, + logging: { host: '127.0.0.1', port: ports.logging }, + ui: { enabled: false }, + singleProjectMode: true, + }, + }, null, 2)); + console.log(`[contract] rules ${rules}`); + console.log(`[contract] auth ${ports.auth}, firestore ${ports.firestore}`); + + const cmd = `node --test --test-concurrency=1 "${TEST}"`; + const child = spawn('firebase', [ + 'emulators:exec', '--only', 'auth,firestore', '--project', 'demo-proscan', + '--config', path.join(dir, 'firebase.json'), JSON.stringify(cmd), + ], { cwd: dir, stdio: 'inherit', shell: true, env: { ...process.env, PROSCAN_CONTRACT_RULES: rules } }); + + const timer = setTimeout(() => { + console.error('[contract] timed out'); + child.kill('SIGTERM'); + }, TIMEOUT_MS); + + const cleanup = () => { + clearTimeout(timer); + freePorts(); + fs.rmSync(dir, { recursive: true, force: true }); + }; + child.on('exit', (code, signal) => { + cleanup(); + process.exit(signal ? 1 : (code ?? 1)); + }); + for (const sig of ['SIGINT', 'SIGTERM']) process.on(sig, () => child.kill(sig)); +} + +const invokedDirectly = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url); +if (invokedDirectly) main(); From 78fff62e06825f292e0e319b7f7065fce41acbc3 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:36:24 -0400 Subject: [PATCH 07/18] Add sign-up and reset links, a session-expired notice and a sync line to the popup --- popup/popup.css | 30 ++++++++++++++++++++ popup/popup.html | 6 ++++ popup/popup.js | 71 ++++++++++++++++++++++++++++++++++++++++++++---- 3 files changed, 102 insertions(+), 5 deletions(-) diff --git a/popup/popup.css b/popup/popup.css index 4933826..eeb8c1b 100644 --- a/popup/popup.css +++ b/popup/popup.css @@ -412,6 +412,36 @@ h1 { text-align: center; } +.auth-notice { + margin-bottom: var(--spacing-sm); + font-size: var(--font-size-sm); + color: var(--color-status-warning-text); +} + +.auth-links { + display: flex; + justify-content: space-between; + margin-top: var(--spacing-sm); + font-size: var(--font-size-sm); +} + +.auth-links a, +.auth-link-button { + color: var(--color-accent-green); + background: none; + border: none; + padding: 0; + font: inherit; + cursor: pointer; + text-decoration: underline; +} + +.auth-sync { + margin-top: var(--spacing-sm); + font-size: var(--font-size-sm); + color: var(--color-text-muted); +} + .auth-account { display: flex; align-items: center; diff --git a/popup/popup.html b/popup/popup.html index d906be8..7db2899 100644 --- a/popup/popup.html +++ b/popup/popup.html @@ -52,12 +52,17 @@

ProScan

Sign in to ProScan
+ + @@ -70,6 +75,7 @@

ProScan

Sign out + diff --git a/popup/popup.js b/popup/popup.js index 2ce64c6..8d844ba 100644 --- a/popup/popup.js +++ b/popup/popup.js @@ -78,6 +78,10 @@ const elements = { authAccount: document.getElementById('authAccount'), authStatusEmail: document.getElementById('authStatusEmail'), authSignOutBtn: document.getElementById('authSignOutBtn'), + authNotice: document.getElementById('authNotice'), + authResetBtn: document.getElementById('authResetBtn'), + authSignUpLink: document.getElementById('authSignUpLink'), + authSync: document.getElementById('authSync'), exportToProScanBtn: document.getElementById('exportToProScanBtn') }; @@ -467,16 +471,41 @@ function showSignedIn(user) { elements.authAccount.classList.remove('hidden'); elements.authError.classList.add('hidden'); elements.authError.textContent = ''; + elements.authNotice.classList.add('hidden'); elements.exportToProScanBtn.classList.toggle('hidden', !Flags.CLOUD_SYNC); } +/** + * The sync line under the account: what is waiting, or when it last synced. + * + * @param {{pending?: number, lastSync?: ?{at:number, error:?string}}} state + */ +function showSyncState(state) { + const pending = (state && state.pending) || 0; + const last = state && state.lastSync; + let text = 'Scans sync to your ProScan dashboard automatically.'; + if (pending > 0) { + text = `${pending} page${pending === 1 ? '' : 's'} waiting to sync.`; + if (last && last.error) text += ' The last try failed; it will retry.'; + } else if (last && !last.error) { + text = 'Everything is synced.'; + } + elements.authSync.textContent = text; + elements.authSync.classList.remove('hidden'); +} + /** * Render the signed-out state: show the sign-in form, hide the account state, * and hide the cloud-export button. */ -function showSignInForm() { +function showSignInForm(notice = null) { currentProScanUser = null; elements.authAccount.classList.add('hidden'); + elements.authSync.classList.add('hidden'); + elements.authNotice.textContent = notice === 'expired' + ? 'Your session expired. Sign in again to keep syncing; your scans are kept.' + : ''; + elements.authNotice.classList.toggle('hidden', notice !== 'expired'); elements.authForm.classList.remove('hidden'); elements.authError.classList.add('hidden'); elements.authError.textContent = ''; @@ -494,6 +523,30 @@ function showAuthError(message) { elements.authError.classList.remove('hidden'); } +/** + * Ask the worker to send a password reset email for the typed address. + * The answer is the same whether or not the account exists. + * + * @async + */ +async function handleResetPassword() { + const email = elements.authEmail.value.trim(); + if (!email) { + showAuthError('Enter your email above, then tap Forgot password.'); + return; + } + elements.authResetBtn.disabled = true; + const response = await sendToWorker({ type: Msg.T.PROSCAN_RESET_PASSWORD, email }); + elements.authResetBtn.disabled = false; + if (response && response.ok) { + elements.authError.classList.add('hidden'); + elements.authNotice.textContent = response.message; + elements.authNotice.classList.remove('hidden'); + } else { + showAuthError((response && response.error) || 'Could not send the reset email.'); + } +} + /** * Restore the sign-in button to its idle state. */ @@ -513,10 +566,12 @@ async function initializeAuthUI() { if (!Flags.CLOUD_SYNC) return; document.getElementById('authPanel').classList.remove('hidden'); const response = await sendToWorker({ type: Msg.T.PROSCAN_AUTH_STATE }); + if (response && response.dashboardUrl) elements.authSignUpLink.href = response.dashboardUrl; if (response && response.user) { showSignedIn(response.user); + showSyncState(response); } else { - showSignInForm(); + showSignInForm(response && response.notice); } } @@ -545,6 +600,7 @@ async function handleSignIn() { if (response && response.user) { elements.authPassword.value = ''; showSignedIn(response.user); + showSyncState({ pending: 0 }); } else { showAuthError((response && response.error) || 'Sign-in failed.'); resetSignInButton(); @@ -591,11 +647,15 @@ async function handleExportToProScan() { if (response && response.ok) { const count = response.products || 0; - if (!response.written || count === 0) { - updateStatus('No new products to export', 'info'); + if (!response.entries) { + updateStatus('Everything is already synced.', 'info'); } else { - updateStatus(`Exported ${count} product${count === 1 ? '' : 's'} to ProScan`, 'success'); + updateStatus(`Synced ${count} product${count === 1 ? '' : 's'} to ProScan`, 'success'); } + showSyncState({ pending: 0, lastSync: { at: Date.now(), error: null } }); + } else if (response && response.expired) { + showSignInForm('expired'); + updateStatus(response.error, 'warning'); } else { updateStatus((response && response.error) || 'Export failed.', 'error'); } @@ -610,6 +670,7 @@ async function handleExportToProScan() { // ProScan cloud auth + export elements.authSignInBtn.addEventListener('click', handleSignIn); elements.authSignOutBtn.addEventListener('click', handleSignOut); +elements.authResetBtn.addEventListener('click', handleResetPassword); elements.exportToProScanBtn.addEventListener('click', handleExportToProScan); // Enter-to-submit from the password field From 740644d3fe192782f0a5603baf32562983073562 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:42:38 -0400 Subject: [PATCH 08/18] Turn on auto-sync and Export to ProScan --- scripts/lib/flags.js | 9 ++++----- tests/e2e/scrape.spec.mjs | 28 +++++++++++++++++++++++----- tests/unit/service-worker.test.js | 18 +++++++++++++++--- 3 files changed, 42 insertions(+), 13 deletions(-) diff --git a/scripts/lib/flags.js b/scripts/lib/flags.js index 1a5a98b..883bcef 100644 --- a/scripts/lib/flags.js +++ b/scripts/lib/flags.js @@ -1,17 +1,16 @@ /** * @fileoverview Build flags. * - * CLOUD_SYNC is off in 2.1 and 2.2. Nothing goes into the sync outbox, the - * service worker refuses PROSCAN_EXPORT, and the popup hides the ProScan - * sign-in and the Export to ProScan button. Sync comes back with the - * per-run outbox (audit report, phase 4). + * CLOUD_SYNC turns on the ProScan account and sync: signed-in scans go + * into the outbox and flush to the dashboard, and the popup shows sign-in + * and Export to ProScan. It was off in 2.1 and 2.2 and is on from 2.3. * * Loaded as a plain global (`Flags`) in the popup, and as a CommonJS * module in Jest and the bundled service worker. */ const Flags = { - CLOUD_SYNC: false + CLOUD_SYNC: true }; if (typeof module !== 'undefined' && module.exports) { diff --git a/tests/e2e/scrape.spec.mjs b/tests/e2e/scrape.spec.mjs index 833a124..19eebc1 100644 --- a/tests/e2e/scrape.spec.mjs +++ b/tests/e2e/scrape.spec.mjs @@ -255,8 +255,7 @@ async function twoRuns(ext, store, done = () => true) { (x.results || []).some((r) => r.name.startsWith('yoga mat')) && done(x), { timeout: 30000 }); } -test('with cloud sync off, two runs queue nothing and keep their deltas (F-26)', async ({ ext }) => { - test.skip(Flags.CLOUD_SYNC, 'cloud sync is on in this build'); +test('signed out, two runs queue nothing and keep their deltas (F-26)', async ({ ext }) => { await serveAmazon(ext.context, simplePlan(2)); const store = await extPage(ext); const s = await twoRuns(ext, store); @@ -278,14 +277,33 @@ test('with cloud sync off, the popup shows no sign-in and no Export to ProScan', expect(resp.error).toMatch(/not available/); }); -test('two runs without an export keep their own attribution', async ({ ext }) => { - test.skip(!Flags.CLOUD_SYNC, 'F-20: cloud sync is off until 2.3, so nothing is queued to attribute'); +test('with cloud sync on, the popup offers sign-in, sign-up and a password reset', async ({ ext }) => { + test.skip(!Flags.CLOUD_SYNC, 'cloud sync is off in this build'); + const popup = await extPage(ext); + await expect(popup.locator('#authForm')).toBeVisible(); + await expect(popup.locator('#authSignUpLink')).toHaveAttribute('href', /^https:\/\/proscanbot\.web\.app\/dashboard\//); + await expect(popup.locator('#authResetBtn')).toBeVisible(); + await expect(popup.locator('#exportToProScanBtn')).toBeHidden(); + await popup.click('#authResetBtn'); + await expect(popup.locator('#authError')).toContainText('Enter your email'); +}); + +test('two runs without an export keep their own attribution (F-20)', async ({ ext }) => { + test.skip(!Flags.CLOUD_SYNC, 'cloud sync is off in this build'); await serveAmazon(ext.context, simplePlan(2)); const store = await extPage(ext); - const s = await twoRuns(ext, store, (x) => (x.outbox || []).length >= 4); + // Signed in as far as the engine knows. Firebase has no user in this + // profile, so nothing is sent anywhere and the outbox keeps it all. Seed + // after the worker's first auth check, which clears a stale account. + await store.evaluate(() => new Promise((r) => chrome.runtime.sendMessage({ type: 'PROSCAN_AUTH_STATE' }, r))); + await store.evaluate(() => chrome.storage.local.set({ account: { uid: 'e2e-user' } })); + const s = await twoRuns(ext, store, (x) => (x.outbox || []).length >= 6); const runIds = [...new Set(s.outbox.map((p) => p.runId))]; expect(runIds).toHaveLength(2); + expect(runIds.map((id) => id.replace(/_\d+$/, '')).sort()).toEqual(['k_garden-hose', 'k_yoga-mat']); + expect(s.outbox.every((e) => e.uid === 'e2e-user')).toBe(true); + expect(s.outbox.map((e) => e.kind)).toEqual(['page', 'page', 'run', 'page', 'page', 'run']); for (const runId of runIds) { expect(runMetaFor(s, runId)).toBeTruthy(); } diff --git a/tests/unit/service-worker.test.js b/tests/unit/service-worker.test.js index 040e9c0..deafb42 100644 --- a/tests/unit/service-worker.test.js +++ b/tests/unit/service-worker.test.js @@ -59,10 +59,22 @@ test('one router answers every message', () => { expect(messageListeners).toHaveLength(1); }); -test('Export to ProScan is refused while cloud sync is off', async () => { +test("Export to ProScan flushes the signed-in account's outbox", async () => { + flush.mockClear(); const resp = await send({ type: 'PROSCAN_EXPORT' }); - expect(resp.error).toMatch(/not available/); - expect(flush).not.toHaveBeenCalled(); + expect(resp).toMatchObject({ ok: true, entries: 0 }); + expect(flush).toHaveBeenCalledWith('u1'); +}); + +test('the auth state says who is signed in and what waits to sync', async () => { + const st = await send({ type: 'PROSCAN_AUTH_STATE' }); + expect(st).toMatchObject({ user: { uid: 'u1' }, notice: null, pending: 0, dashboardUrl: expect.stringMatching(/^https:/) }); +}); + +test('a password reset answers the same whether or not the account exists (F-56)', async () => { + const st = await send({ type: 'PROSCAN_RESET_PASSWORD', email: 'nobody@example.test' }); + expect(st).toEqual({ ok: true, message: 'If an account exists for nobody@example.test, a reset link is on its way.' }); + expect(await send({ type: 'PROSCAN_RESET_PASSWORD', email: '' })).toMatchObject({ error: expect.any(String) }); }); test('messages meant for the popup or a tab are left alone', async () => { From 534efcebf8611ba743762071f8bda727c6cb6678 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:42:38 -0400 Subject: [PATCH 09/18] Bump the version to 2.3.0 --- manifest.json | 2 +- tests/e2e/upgrade.spec.mjs | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/manifest.json b/manifest.json index fd66f52..71326e9 100644 --- a/manifest.json +++ b/manifest.json @@ -1,7 +1,7 @@ { "manifest_version": 3, "name": "ProScan - Amazon Product Scraper", - "version": "2.2.0", + "version": "2.3.0", "description": "Scrape Amazon seller products with analytics and AI-powered product insights", "permissions": [ "storage", diff --git a/tests/e2e/upgrade.spec.mjs b/tests/e2e/upgrade.spec.mjs index 8a5ac97..6a694c1 100644 --- a/tests/e2e/upgrade.spec.mjs +++ b/tests/e2e/upgrade.spec.mjs @@ -17,7 +17,7 @@ const V21_REV = '40be432'; const V20 = path.join(BUILD_DIR, 'v2.0'); const V21 = path.join(BUILD_DIR, 'v2.1'); const SLOT = path.join(BUILD_DIR, 'upgrade-slot'); -const CURRENT_VERSION = '2.2.0'; +const CURRENT_VERSION = '2.3.0'; function bug(fid, what) { test.fail(!process.env.PROSCAN_SHOW_KNOWN, `${fid}: ${what}`); From ad86dc2fb3f5b8796dbc5fa31431692e407e594d Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:43:39 -0400 Subject: [PATCH 10/18] Send the final header of a run a restart or update ended, and flush after tab changes --- scripts/background/engine.js | 3 ++- scripts/background/service-worker.js | 2 +- tests/unit/engine.test.js | 12 ++++++++++++ 3 files changed, 15 insertions(+), 2 deletions(-) diff --git a/scripts/background/engine.js b/scripts/background/engine.js index 58be276..87e78ee 100644 --- a/scripts/background/engine.js +++ b/scripts/background/engine.js @@ -493,7 +493,8 @@ function createEngine({ const latest = await store.getMeta('latestRunId'); const rec = latest ? await store.get('runs', latest) : null; if (Run.isActive(rec)) { - await store.write([{ store: 'runs', put: durable(Run.finish(rec, reason, now())) }]); + const queued = rec.page > 0 ? await outbox(rec.runId, 'run') : []; + await store.write([{ store: 'runs', put: durable(Run.finish(rec, reason, now())) }, ...queued]); } }); } diff --git a/scripts/background/service-worker.js b/scripts/background/service-worker.js index bd68af3..3864d55 100644 --- a/scripts/background/service-worker.js +++ b/scripts/background/service-worker.js @@ -95,7 +95,7 @@ chrome.runtime.onStartup.addListener(() => { // Neither listener needs the tabs permission. chrome.tabs.onRemoved.addListener((tabId) => { thenFlush(engine.tabRemoved(tabId)); }); -chrome.tabs.onUpdated.addListener((tabId, info, tab) => { engine.tabUpdated(tabId, info, tab); }); +chrome.tabs.onUpdated.addListener((tabId, info, tab) => { thenFlush(engine.tabUpdated(tabId, info, tab)); }); // ── ProScan account and cloud sync ────────────────────────────────────────── // The popup is a plain page; it signs in and exports through these messages, diff --git a/tests/unit/engine.test.js b/tests/unit/engine.test.js index c2e66c5..63cbc8c 100644 --- a/tests/unit/engine.test.js +++ b/tests/unit/engine.test.js @@ -310,6 +310,18 @@ describe('the worker stopped between pages', () => { expect(st.run).toMatchObject({ state: 'failed', reason: 'interrupted' }); expect(st.results).toHaveLength(4); }); + + test('a run ended after a restart queues its final header for the signed-in account', async () => { + const rig = createRig({ site: simpleSite(3), flags: { CLOUD_SYNC: true }, local: { account: { uid: 'u1' } } }); + await startIn(rig, 'restart'); + await settle(); + await rig.session.remove('run'); + rig.killWorker(); + await rig.engine().recover(); + const db = await rig.db(); + expect((await db.getAll('outbox')).map((e) => e.kind)).toEqual(['page', 'run']); + db.close(); + }); }); describe('a page that never reports', () => { From 7d3c5b714177bd62968f414f3a8c8778d984759e Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:44:56 -0400 Subject: [PATCH 11/18] Test the expired-session notice, and let currentUser survive an early auth callback --- scripts/background/service-worker.js | 10 ++++-- tests/unit/service-worker.test.js | 46 ++++++++++++++++++++++++---- 2 files changed, 48 insertions(+), 8 deletions(-) diff --git a/scripts/background/service-worker.js b/scripts/background/service-worker.js index 3864d55..7a9cb5c 100644 --- a/scripts/background/service-worker.js +++ b/scripts/background/service-worker.js @@ -116,10 +116,16 @@ const sync = createSync({ db: firestore, openStore: () => engine.db() }); function currentUser() { return new Promise((resolve) => { if (auth.currentUser) return resolve(auth.currentUser); - const unsub = onAuthStateChanged(auth, (u) => { - unsub(); + let settled = false; + let unsub = null; + unsub = onAuthStateChanged(auth, (u) => { + if (settled) return; + settled = true; + if (unsub) unsub(); resolve(u); }); + // The first answer can come before onAuthStateChanged returns. + if (settled) unsub(); }); } diff --git a/tests/unit/service-worker.test.js b/tests/unit/service-worker.test.js index deafb42..a34e64a 100644 --- a/tests/unit/service-worker.test.js +++ b/tests/unit/service-worker.test.js @@ -8,12 +8,20 @@ jest.mock('../../scripts/background/firebase-init.js', () => ({ auth: { currentUser: { uid: 'u1', email: 'u1@example.test', displayName: null } }, db: {}, })); -jest.mock('firebase/auth/web-extension', () => ({ - onAuthStateChanged: (auth, cb) => { cb(auth.currentUser); return () => {}; }, - signInWithEmailAndPassword: async () => ({}), - sendPasswordResetEmail: async () => {}, - signOut: async () => {}, -}), { virtual: true }); +jest.mock('firebase/auth/web-extension', () => { + const listeners = []; + return { + __listeners: listeners, + onAuthStateChanged: (auth, cb) => { + listeners.push(cb); + cb(auth.currentUser); + return () => {}; + }, + signInWithEmailAndPassword: async () => ({}), + sendPasswordResetEmail: async () => {}, + signOut: async () => {}, + }; +}, { virtual: true }); jest.mock('../../scripts/background/sync.js', () => { const flush = jest.fn(async () => ({ entries: 0, pages: 0, runs: 0, products: 0, writes: 0 })); return { createSync: () => ({ flush, pending: async () => 0 }), isAuthError: () => false, __flush: flush }; @@ -112,3 +120,29 @@ test('the chat reads the key from settings and the run from IndexedDB', async () const st = await send({ type: 'CHAT_STATUS' }, { id: 'test-extension', tab: { id: 3 } }); expect(st).toMatchObject({ hasKey: true }); }); + +test('a session Firebase dropped without a sign-out shows as expired (F-57)', async () => { + const { auth } = require('../../scripts/background/firebase-init.js'); + const authMod = require('firebase/auth/web-extension'); + const user = auth.currentUser; + await chrome.storage.local.set({ account: { uid: 'u1', email: 'u1@example.test' } }); + auth.currentUser = null; + try { + // The worker's own listener is the first one registered. + authMod.__listeners[0](null); + await settle(); + expect(chrome.storage.local._getStore()).toMatchObject({ authNotice: 'expired' }); + expect(chrome.storage.local._getStore()).not.toHaveProperty('account'); + const st = await send({ type: 'PROSCAN_AUTH_STATE' }); + expect(st).toMatchObject({ user: null, notice: 'expired' }); + } finally { + auth.currentUser = user; + } +}); + +test('signing out on purpose clears the account without an expired notice', async () => { + await chrome.storage.local.set({ account: { uid: 'u1' }, authNotice: 'expired' }); + expect(await send({ type: 'PROSCAN_SIGN_OUT' })).toEqual({ ok: true }); + expect(chrome.storage.local._getStore()).not.toHaveProperty('account'); + expect(chrome.storage.local._getStore()).not.toHaveProperty('authNotice'); +}); From d36b613f3b7b16a2e8fd5db9a134f9caac6339d7 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:45:32 -0400 Subject: [PATCH 12/18] Document cloud sync, the shared schema and the contract test --- CLAUDE.md | 46 ++++++++++++++++++++++++++++++++++++++-------- README.md | 30 +++++++++++++++++++++++++++--- 2 files changed, 65 insertions(+), 11 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 62ede2c..a03aefc 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -4,11 +4,11 @@ Chrome extension (Manifest V3) that scrapes Amazon seller product listings, provides analytics for resellers/arbitrage, and includes a floating AI chatbot on Amazon pages powered by Gemini API. No server required — everything runs client-side. -## Architecture (v2.2) +## Architecture (v2.3) ```text AmazonSellerScraper/ -├── manifest.json # Extension config (v2.2) +├── manifest.json # Extension config (v2.3) ├── popup/ # UI Layer │ ├── popup.html # Popup interface (dashboard + settings) │ ├── popup.css # Popup styling @@ -23,19 +23,23 @@ AmazonSellerScraper/ │ │ ├── parsers.js # Pure search/offer parsing (global Parsers) │ │ ├── messages.js # Every message type and who may send it (global Msg) │ │ ├── run.js # Run state machine, bound to one tab (global Run) -│ │ ├── flags.js # Build flags (global Flags); CLOUD_SYNC is off until 2.3 +│ │ ├── flags.js # Build flags (global Flags); CLOUD_SYNC is on from 2.3 │ │ ├── migrate.js # Storage schema migrations, run by the SW │ │ └── chat.js # Gemini request builder, run scoping, error text │ ├── background/ │ │ ├── service-worker.js # Wires router, engine, chat, auth and migrations │ │ ├── router.js # The one typed message router │ │ ├── engine.js # The run engine; the only writer of run data -│ │ └── db.js # IndexedDB: runs, products, placements, lastValues, outbox, spread +│ │ ├── db.js # IndexedDB: runs, products, placements, lastValues, outbox, spread +│ │ ├── sync-plan.js # Outbox entry -> cloud writes (pure) +│ │ └── sync.js # Drains the outbox into Firestore, one entry at a time │ └── modules/ │ ├── storage.js # Chrome storage wrapper │ ├── analyzer.js # Data analysis & insights │ ├── spread-analyzer.js # Price spread & arbitrage scoring │ └── exporter.js # Excel/CSV/JSON export +├── packages/ +│ └── schema/index.js # Cloud schema: types, validators, ids (shared with the dashboard) ├── styles/ │ └── chatbot.css # Chatbot widget styles (loaded into Shadow DOM) ├── tests/ # Test suite (Jest) @@ -102,6 +106,27 @@ Run states: idle, starting, running, stopping, then stopped, blocked, failed or `reason` says why it ended: complete, stopped, blocked, selectors_broken, storage_full, storage_error, interrupted or updated. +### Cloud sync (2.3) + +1. The popup signs in with email and password through the SW (`PROSCAN_SIGN_IN`); sign-up opens + the dashboard, and "Forgot password?" sends `PROSCAN_RESET_PASSWORD` (same answer either way) +2. The SW keeps `account` {uid, email} in `chrome.storage.local` while signed in. The engine + queues `{kind:'page', pageIndex, uid}` per saved page and `{kind:'run', uid}` when a run ends, + only when an account is signed in +3. `lastValues` snapshots carry the uid that took them; another account's snapshot gives no delta +4. `scheduleFlush()` runs a few seconds after PAGE_RESULT, STOP_RUN, a tab change, popup open, + sign-in, browser start and update. No alarms +5. `sync.js` drains the outbox for the signed-in uid, oldest first. `sync-plan.js` turns an entry + into writes; each is `set` with `mergeFields`, so `latest`, `prev` and `delta` are replaced whole. + `firstSeenAt` is written only when a read shows the product document does not exist +6. An entry is deleted after its own commit. Entries of another uid are left alone +7. `permission-denied` or an invalid token signs out and sets `authNotice: 'expired'`; + the popup shows "Your session expired" + +Cloud paths and shapes: `packages/schema/index.js` (keep the dashboard's copy identical, bump `SV` +on a shape change). Run id `{sourceId}_{startMs}`, minted once in `engine.start`; page id `p0001`; +`dayKey` is the local date the run started. + ### AI Chatbot (client-side, no server) 1. `chatbot.js` injects a floating widget (bottom-right) on Amazon pages with product listings @@ -134,6 +159,9 @@ npm run test:coverage # With coverage report - Pure logic modules (`analyzer.js`, `spread-analyzer.js`) are tested via `require()` directly - Content scripts (`scraper.js`, `offer-fetcher.js`) have no `module.exports` — loaded via `vm.runInContext` into a JSDOM context with Chrome API mocks and an `innerText` polyfill - `tests/setup/engine-rig.js` runs the real router and engine over fake-indexeddb with the real `scraper.js` in a JSDOM page per tab; `tests/unit/engine.test.js` drives runs through it +- `npm run test:contract` runs the real `sync.js` against the Firebase emulators with the dashboard's + `firestore.rules` (`tools/contract.mjs`, `tests/contract/`); ESM files under `packages/` and + `scripts/background/` load in Jest through `tests/setup/esm-to-cjs-transform.js` - `npm run test:e2e` runs the same scenarios in Chromium (`tests/e2e/`), including the worker stopped via CDP between pages and 2.0 and 2.1 builds updated mid-run - HTML fixtures in `tests/fixtures/` match the exact CSS selectors the code uses - Chrome APIs (`storage`, `runtime`, `tabs`, `downloads`) are mocked in `tests/setup/chrome-mock.js` @@ -141,7 +169,7 @@ npm run test:coverage # With coverage report ## Chrome APIs Used -- `chrome.storage.local`: settings, the Gemini key and `schemaVersion` only +- `chrome.storage.local`: settings, the Gemini key, `schemaVersion`, and `account`, `authNotice`, `lastSync` - `chrome.storage.session`: the live run record (content scripts cannot read it) - IndexedDB: run data, in the extension origin; needs no permission - `chrome.runtime.sendMessage/onMessage`: messages, all in `scripts/lib/messages.js` @@ -196,6 +224,8 @@ One record per ASIN per run, from `Parsers.parseSearchPage` and the scraper: - `url`: always `https://www.amazon.com/dp/{asin}`, never the sspa ad link - `sponsored`: true if any placement was an ad; `organicRank`: run-wide rank of the first organic card, or null - `placements`: every card the ASIN had, as `{page, position, sponsored, rank}` +- `img`: the card image on Amazon's image host, or null +- `prev`: the `lastValues` snapshot the delta was taken against, or null - `delta`: taken once, on the ASIN's first sighting in the run, against `lastValues` from earlier runs. A field that fails to parse keeps its last good value in `lastValues`, with the time in `carried` @@ -210,7 +240,7 @@ One record per ASIN per run, from `Parsers.parseSearchPage` and the scraper: the IndexedDB writes commit before anything is removed. - IndexedDB `proscan` (`scripts/background/db.js`): `runs`, `products` (key `[runId, n]`, n is the order found), `placements` (one record per page), `lastValues` (by ASIN), - `outbox` (pages waiting for sync), `spread` and `meta` (`latestRunId`). The popup shows + `outbox` (`{kind, runId, uid, pageIndex}` entries waiting for sync), `spread` and `meta` (`latestRunId`). The popup shows the latest run's products. `tests/fixtures/v2.0-storage.json` is what the live 2.0 build stores, captured with `node tests/e2e/capture-v20-storage.mjs`. @@ -218,8 +248,8 @@ One record per ASIN per run, from `Parsers.parseSearchPage` and the scraper: (`Delta.MAX_ENTRIES`, `Delta.MAX_AGE_DAYS`). - A page write that fails ends the run as `storage_full` (quota) or `storage_error`. The popup warns from 90% of `navigator.storage.estimate()`. -- Cloud sync is off in 2.1 and 2.2 (`Flags.CLOUD_SYNC`): nothing goes into the outbox, the SW - refuses `PROSCAN_EXPORT`, and the popup hides sign-in and Export to ProScan. +- Cloud sync was off in 2.1 and 2.2 and is on from 2.3 (`Flags.CLOUD_SYNC`). Signed out, nothing + goes into the outbox. ## Export diff --git a/README.md b/README.md index 9bd19dc..e098cd2 100644 --- a/README.md +++ b/README.md @@ -74,6 +74,24 @@ User clicks "Start Scraping" → User exports via exporter.js (Excel/CSV/JSON) ``` +### Cloud Sync + +``` +Signed in to a ProScan account in the popup (email and password) + → Each saved page goes into the IndexedDB outbox, tagged with the account's uid; + the end of the run goes in after its pages + → A few seconds after a page, a run end, a popup open or a browser start, + the worker flushes the outbox (no alarms) + → sync.js writes one entry at a time: the page chunk, a product and a history + document per ASIN on the page, the run header and the source. latest, prev + and delta are replaced whole, never merged + → An entry leaves the outbox only once its own writes commit; a page saved + during a flush goes out before the flush returns + → Export to ProScan flushes right away +``` + +The document shapes, ids and validators are in `packages/schema/index.js`, which the dashboard copies. Money is integer cents, unknown values are null (left out of compact points), and every document carries `sv`. A signed-out user queues nothing. If Firebase drops the session, the popup says so and the queued pages wait for the same account to sign in again. + ### AI Chatbot Flow ``` @@ -153,6 +171,8 @@ For local Firebase work, `npm run build:dev` points the build at the emulators u The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, see its README). `npm run test:e2e` loads the built extension into Chromium and runs scrape scenarios against those pages, with every request answered locally. Run `npx playwright install --no-shell chromium` once first. Known bugs run as expected failures tagged with their audit finding id; `PROSCAN_SHOW_KNOWN=1 npm run test:e2e` shows what they fail on. +`npm run test:contract` runs the real sync module against the Firebase emulators with the dashboard's `firestore.rules` (from `PROSCAN_RULES`, or a `web` or `proscan-web` checkout next to this repo): queues of 1, 201 and 600 products, a page added mid-flush, replace semantics across runs, create-only `firstSeenAt` and the write count per run. It needs the Firebase CLI and Java, and uses the `demo-proscan` project only. + The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, see its README). `npm run test:e2e` loads the built extension into Chromium and runs scrape scenarios against those pages, with every request answered locally. Run `npx playwright install --no-shell chromium` once first. Known bugs run as expected failures tagged with their audit finding id; `PROSCAN_SHOW_KNOWN=1 npm run test:e2e` shows what they fail on. ## Usage @@ -176,7 +196,7 @@ The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, s ## Chrome APIs Used -- `chrome.storage.local` -- Settings and the schema version only +- `chrome.storage.local` -- Settings, the schema version, and the signed-in account (`account`, `authNotice`, `lastSync`) - `chrome.storage.session` -- The live run record, written by the service worker - IndexedDB (the extension's own origin, no permission) -- Runs, products, pages, lastValues and the sync outbox. Starting a run keeps the 10 newest runs and removes older ones, except runs still waiting to sync. - `chrome.runtime.sendMessage` / `onMessage` -- Messages, all listed in `scripts/lib/messages.js` @@ -201,19 +221,23 @@ AmazonSellerScraper/ │ │ ├── parsers.js # Pure search and offer page parsing │ │ ├── messages.js # Every message type and who may send it │ │ ├── run.js # The run state machine and its end reasons -│ │ ├── flags.js # Build flags (cloud sync is off until 2.3) +│ │ ├── flags.js # Build flags (cloud sync is on from 2.3) │ │ ├── migrate.js # Storage schema migrations (schemaVersion) │ │ └── chat.js # Gemini request builder and run scoping │ ├── background/ │ │ ├── service-worker.js # Wires the router, the engine, the chat and migrations │ │ ├── router.js # The one message router │ │ ├── engine.js # Runs scrapes; the only writer of run data -│ │ └── db.js # IndexedDB stores +│ │ ├── db.js # IndexedDB stores +│ │ ├── sync-plan.js # What each outbox entry writes (pure) +│ │ └── sync.js # Drains the outbox into Firestore │ └── modules/ │ ├── storage.js # Chrome storage abstraction layer │ ├── analyzer.js # Analytics engine + opportunity scoring │ ├── spread-analyzer.js # Price spread statistics (CV, arbitrage score) │ └── exporter.js # Multi-format export (Excel/CSV/JSON) +├── packages/ +│ └── schema/index.js # Cloud schema shared with the dashboard ├── styles/ │ └── chatbot.css # Chatbot widget styles (Shadow DOM) ├── libs/ From 6bfea41003fb38f2e324ba6214db5915b63a8b62 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:56:54 -0400 Subject: [PATCH 13/18] Do not sign the user out when the rules refuse a sync write A permission-denied from Firestore means the rules rejected a document, not that the session is gone. Treating it as expired signed the user out on every flush. --- scripts/background/sync.js | 7 +++++-- tests/contract/sync.contract.mjs | 2 +- tests/unit/sync.test.js | 5 +++-- 3 files changed, 9 insertions(+), 5 deletions(-) diff --git a/scripts/background/sync.js b/scripts/background/sync.js index eefd004..046b9ec 100644 --- a/scripts/background/sync.js +++ b/scripts/background/sync.js @@ -32,8 +32,11 @@ const FIRESTORE = { doc, getDoc, writeBatch, Timestamp, arrayUnion, FieldPath }; /** Firestore allows 500 writes per batch; stay under it. */ export const BATCH_LIMIT = 450; -/** Errors that mean the account's session is gone, not the network. */ -const AUTH_CODES = ['permission-denied', 'unauthenticated', 'auth/user-token-expired', 'auth/user-disabled', 'auth/invalid-user-token']; +/** + * Errors that mean the account's session is gone, not the network. A + * permission-denied is the rules refusing a document, not a lost session. + */ +const AUTH_CODES = ['unauthenticated', 'auth/user-token-expired', 'auth/user-disabled', 'auth/invalid-user-token']; export function isAuthError(err) { return !!err && AUTH_CODES.includes(err.code); diff --git a/tests/contract/sync.contract.mjs b/tests/contract/sync.contract.mjs index 87a0ced..b88aee1 100644 --- a/tests/contract/sync.contract.mjs +++ b/tests/contract/sync.contract.mjs @@ -283,6 +283,6 @@ test('another account cannot write this account\'s queue, and the entries stay', await signOut(auth); await account(`b-${Date.now()}@contract.test`); - await assert.rejects(ext.sync().flush(owner), (err) => isAuthError(err)); + await assert.rejects(ext.sync().flush(owner), (err) => err.code === 'permission-denied' && !isAuthError(err)); assert.equal(await (await ext.store()).count('outbox'), 2); }); diff --git a/tests/unit/sync.test.js b/tests/unit/sync.test.js index 799195a..313a887 100644 --- a/tests/unit/sync.test.js +++ b/tests/unit/sync.test.js @@ -196,8 +196,9 @@ test('an entry whose run is gone is dropped', async () => { expect(await store.count('outbox')).toBe(0); }); -test('auth errors are told apart from network errors', () => { - expect(isAuthError({ code: 'permission-denied' })).toBe(true); +test('auth errors are told apart from network and rules errors', () => { + expect(isAuthError({ code: 'permission-denied' })).toBe(false); + expect(isAuthError({ code: 'unauthenticated' })).toBe(true); expect(isAuthError({ code: 'auth/user-token-expired' })).toBe(true); expect(isAuthError({ code: 'unavailable' })).toBe(false); expect(isAuthError(null)).toBe(false); From e5c4757eaea3b535881df2604b86d99d2691d31a Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:56:57 -0400 Subject: [PATCH 14/18] Drop an outbox entry that fails validation instead of retrying it forever Planning is pure, so an entry that cannot be planned, such as the end of a run from before 2.3, would fail on every flush and hold up the rest of the queue. --- scripts/background/sync.js | 9 ++++++++- tests/unit/sync.test.js | 13 +++++++++++++ 2 files changed, 21 insertions(+), 1 deletion(-) diff --git a/scripts/background/sync.js b/scripts/background/sync.js index 046b9ec..ace2117 100644 --- a/scripts/background/sync.js +++ b/scripts/background/sync.js @@ -83,7 +83,14 @@ export function createSync({ db, openStore, fs = FIRESTORE, batchLimit = BATCH_L if (!run) return null; const [products, pages] = await Promise.all([store.runProducts(entry.runId), store.runPages(entry.runId)]); const missing = await missingOf(entry.uid, firstSeenCandidates(entry, products)); - const writes = planEntry({ entry, run, products, pages, missing, time, union }); + let writes; + try { + writes = planEntry({ entry, run, products, pages, missing, time, union }); + } catch (err) { + // Planning is pure, so a retry would fail the same way and hold up the queue. + log.warn('[ProScan] Dropped an outbox entry that cannot be written:', err.message); + return null; + } for (let i = 0; i < writes.length; i += batchLimit) { const batch = fs.writeBatch(db); diff --git a/tests/unit/sync.test.js b/tests/unit/sync.test.js index 313a887..b48f9b6 100644 --- a/tests/unit/sync.test.js +++ b/tests/unit/sync.test.js @@ -196,6 +196,19 @@ test('an entry whose run is gone is dropped', async () => { expect(await store.count('outbox')).toBe(0); }); +test('an entry that fails validation is dropped instead of blocking the queue', async () => { + const store = await DB.open({ name: `sync-test-${++dbCount}` }); + // A run from before 2.3: its id does not start with its source id. + await store.write([ + { store: 'runs', put: { runId: '1717000000000-abc1234', state: 'updated', reason: 'updated', startedAt: 1717000000000, finishedAt: 1717000100000, maxPages: 5, page: 1, source: { type: 'keyword', keyword: 'lamp', sourceId: 'k_lamp' } } }, + { store: 'outbox', put: { runId: '1717000000000-abc1234', kind: 'run', uid: 'u1' } } + ]); + const cloud = fakeFirestore(); + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, log: quiet }); + expect(await sync.flush('u1')).toMatchObject({ entries: 1, writes: 0 }); + expect(await store.count('outbox')).toBe(0); +}); + test('auth errors are told apart from network and rules errors', () => { expect(isAuthError({ code: 'permission-denied' })).toBe(false); expect(isAuthError({ code: 'unauthenticated' })).toBe(true); From c2b90cb49c1ebd5d752363466bc3013ce71bb43e Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 05:57:08 -0400 Subject: [PATCH 15/18] Note how sync treats rules refusals and invalid entries --- CLAUDE.md | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index a03aefc..7b063d4 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -119,9 +119,11 @@ storage_error, interrupted or updated. 5. `sync.js` drains the outbox for the signed-in uid, oldest first. `sync-plan.js` turns an entry into writes; each is `set` with `mergeFields`, so `latest`, `prev` and `delta` are replaced whole. `firstSeenAt` is written only when a read shows the product document does not exist -6. An entry is deleted after its own commit. Entries of another uid are left alone -7. `permission-denied` or an invalid token signs out and sets `authNotice: 'expired'`; - the popup shows "Your session expired" +6. An entry is deleted after its own commit. Entries of another uid are left alone. An entry + that fails schema validation is dropped, since retrying it would fail the same way +7. An expired or invalid token signs out and sets `authNotice: 'expired'`; the popup shows + "Your session expired". A `permission-denied` (the rules refusing a write) is not a sign-out; + the entry stays queued and `lastSync.error` records it Cloud paths and shapes: `packages/schema/index.js` (keep the dashboard's copy identical, bump `SV` on a shape change). Run id `{sourceId}_{startMs}`, minted once in `engine.start`; page id `p0001`; From 2176e375cb1a090447edde67dbc2c26892ce6bcf Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 16:46:37 -0400 Subject: [PATCH 16/18] Refuse taken ports in the contract runner and stop only what it started The runner used to taskkill every process listening on its ports on exit, even when the emulators never started because another process held them. It now checks the ports first and exits without starting or stopping anything if one is taken, and on exit kills only listeners that started after it did. --- tools/contract.mjs | 96 ++++++++++++++++++++++++++++------- tools/tests/contract.test.mjs | 16 ++++++ 2 files changed, 93 insertions(+), 19 deletions(-) create mode 100644 tools/tests/contract.test.mjs diff --git a/tools/contract.mjs b/tools/contract.mjs index a6139bf..6d8aed0 100644 --- a/tools/contract.mjs +++ b/tools/contract.mjs @@ -7,9 +7,14 @@ // ../proscan-web/firestore.rules next to this repo. Ports default away from // the dashboard's (EMU_AUTH_PORT 9299, EMU_FIRESTORE_PORT 8288) so both can // be checked out side by side. Project is demo-proscan: emulators only. +// +// It refuses to start when any of its ports is already taken, and on exit it +// only kills processes that listen on its ports and started after it did, +// so another emulator, dev server or agent on those ports is never touched. import { spawn, execSync } from 'node:child_process'; import fs from 'node:fs'; +import net from 'node:net'; import os from 'node:os'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; @@ -40,32 +45,84 @@ export function findRules(env = process.env) { return found; } -/** Kills whatever still listens on our ports; on Windows java outlives firebase. */ -function freePorts() { - if (process.platform !== 'win32') return; - let out = ''; +/** Resolves true when nothing can be bound to 127.0.0.1:`port`. */ +function portTaken(port) { + return new Promise((resolve) => { + const srv = net.createServer(); + srv.once('error', () => resolve(true)); + srv.listen(port, '127.0.0.1', () => srv.close(() => resolve(false))); + }); +} + +/** The ports of `list` something already listens on. */ +export async function busyPorts(list) { + const taken = []; + for (const p of list) if (await portTaken(p)) taken.push(p); + return taken; +} + +/** {port: Set(pid)} for every LISTENING socket on `list`, on any address (Windows only). */ +function listeners(list) { + const out = new Map(); + if (process.platform !== 'win32') return out; + let text = ''; try { - out = execSync('netstat -ano -p tcp', { encoding: 'utf8' }); + text = execSync('netstat -ano -p tcp', { encoding: 'utf8' }); } catch { - return; + return out; } - const wanted = new Set(Object.values(ports).map(String)); - const pids = new Set(); - for (const line of out.split('\n')) { - const m = line.trim().match(/^TCP\s+127\.0\.0\.1:(\d+)\s+\S+\s+LISTENING\s+(\d+)/); - if (m && wanted.has(m[1])) pids.add(m[2]); + const wanted = new Set(list.map(String)); + for (const line of text.split(/\r?\n/)) { + const m = line.trim().match(/^TCP\s+\S+:(\d+)\s+\S+\s+LISTENING\s+(\d+)/); + if (m && wanted.has(m[1])) { + if (!out.has(m[1])) out.set(m[1], new Set()); + out.get(m[1]).add(m[2]); + } } - for (const pid of pids) { - try { - execSync(`taskkill /PID ${pid} /F`, { stdio: 'ignore' }); - } catch { - /* already gone */ + return out; +} + +/** When process `pid` started, in ms, or null when that cannot be read. */ +function startedAt(pid) { + try { + const iso = execSync( + `powershell -NoProfile -Command "(Get-Process -Id ${Number(pid)}).StartTime.ToUniversalTime().ToString('o')"`, + { encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] }, + ).trim(); + const t = Date.parse(iso); + return Number.isFinite(t) ? t : null; + } catch { + return null; + } +} + +/** + * On Windows java outlives firebase, so the emulators this run started can + * keep our ports. Kills only listeners on our ports that started after + * `since`; anything older belongs to someone else. + */ +function freeOurPorts(since) { + for (const pids of listeners(Object.values(ports)).values()) { + for (const pid of pids) { + const t = startedAt(pid); + if (t === null || t < since) continue; + try { + execSync(`taskkill /PID ${pid} /T /F`, { stdio: 'ignore' }); + } catch { + /* already gone */ + } } } } -function main() { +async function main() { const rules = findRules(); + const taken = await busyPorts(Object.values(ports)); + if (taken.length) { + console.error(`[contract] FAIL: port${taken.length > 1 ? 's' : ''} ${taken.join(', ')} already in use. Nothing was started or stopped.`); + console.error('[contract] Pick free ports with EMU_AUTH_PORT, EMU_FIRESTORE_PORT, EMU_WEBSOCKET_PORT, EMU_HUB_PORT and EMU_LOGGING_PORT.'); + process.exit(2); + } const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'proscan-contract-')); fs.copyFileSync(rules, path.join(dir, 'firestore.rules')); fs.writeFileSync(path.join(dir, 'firebase.json'), JSON.stringify({ @@ -83,6 +140,7 @@ function main() { console.log(`[contract] auth ${ports.auth}, firestore ${ports.firestore}`); const cmd = `node --test --test-concurrency=1 "${TEST}"`; + const spawnedAt = Date.now() - 1000; const child = spawn('firebase', [ 'emulators:exec', '--only', 'auth,firestore', '--project', 'demo-proscan', '--config', path.join(dir, 'firebase.json'), JSON.stringify(cmd), @@ -95,7 +153,7 @@ function main() { const cleanup = () => { clearTimeout(timer); - freePorts(); + freeOurPorts(spawnedAt); fs.rmSync(dir, { recursive: true, force: true }); }; child.on('exit', (code, signal) => { @@ -106,4 +164,4 @@ function main() { } const invokedDirectly = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url); -if (invokedDirectly) main(); +if (invokedDirectly) main().catch((err) => { console.error(`[contract] FAIL: ${err.message}`); process.exit(1); }); diff --git a/tools/tests/contract.test.mjs b/tools/tests/contract.test.mjs new file mode 100644 index 0000000..bf43945 --- /dev/null +++ b/tools/tests/contract.test.mjs @@ -0,0 +1,16 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import net from 'node:net'; +import { busyPorts } from '../contract.mjs'; + +test('busyPorts names a port someone else listens on, and nothing else', async () => { + const srv = net.createServer(); + await new Promise((r) => srv.listen(0, '127.0.0.1', r)); + const { port } = srv.address(); + try { + assert.deepEqual(await busyPorts([port]), [port]); + } finally { + await new Promise((r) => srv.close(r)); + } + assert.deepEqual(await busyPorts([port]), []); +}); From e9873c1d4c95ba41a115b74fa9c154a299cc438b Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 16:46:43 -0400 Subject: [PATCH 17/18] Keep every source on a product, set refused entries aside, back off, and write the header once per flush Product documents are now one merge write with deletes for the keys latest, prev and delta lack, so the sourceIds arrayUnion keeps earlier sources (NEW-SYNC-1) without an extra write. An entry the rules refuse is marked failed and skipped so the rest still sync; Export retries it and the popup counts it. Automatic flushes back off after a failure, and tab events flush only when they ended the run. The run header and source are written once per run per flush instead of on every page, and the rules' length caps are checked before a commit. A run's end entry goes to the account its pages were queued for, also for runs ended from IndexedDB. --- CLAUDE.md | 20 +++-- README.md | 16 ++-- popup/popup.js | 15 +++- scripts/background/engine.js | 36 ++++++--- scripts/background/service-worker.js | 40 ++++++++-- scripts/background/sync-plan.js | 107 ++++++++++++++++++++++++--- scripts/background/sync.js | 104 +++++++++++++++++++++----- tests/contract/sync.contract.mjs | 59 ++++++++++++--- tests/unit/engine.test.js | 27 +++++++ tests/unit/service-worker.test.js | 46 +++++++++++- tests/unit/sync-plan.test.js | 54 ++++++++++++-- tests/unit/sync.test.js | 84 ++++++++++++++++++++- 12 files changed, 519 insertions(+), 89 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 7b063d4..eb5a535 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -114,16 +114,24 @@ storage_error, interrupted or updated. queues `{kind:'page', pageIndex, uid}` per saved page and `{kind:'run', uid}` when a run ends, only when an account is signed in 3. `lastValues` snapshots carry the uid that took them; another account's snapshot gives no delta -4. `scheduleFlush()` runs a few seconds after PAGE_RESULT, STOP_RUN, a tab change, popup open, - sign-in, browser start and update. No alarms +4. `scheduleFlush()` runs a few seconds after PAGE_RESULT, STOP_RUN, a tab change that ended the + run, popup open, sign-in, browser start and update. No alarms. After a failed flush the + automatic ones back off (30 s, doubling to 1 h, `lastSync.failures`); Export does not wait 5. `sync.js` drains the outbox for the signed-in uid, oldest first. `sync-plan.js` turns an entry - into writes; each is `set` with `mergeFields`, so `latest`, `prev` and `delta` are replaced whole. + into writes; most are `set` with `mergeFields`, each field replaced whole. + A product document is a `merge: true` write instead, so the `sourceIds` arrayUnion keeps earlier + sources; the keys its `latest`, `prev` and `delta` lack are sent as deletes. + The run header and source go on each run's last entry in a flush round. `firstSeenAt` is written only when a read shows the product document does not exist 6. An entry is deleted after its own commit. Entries of another uid are left alone. An entry - that fails schema validation is dropped, since retrying it would fail the same way + that fails schema validation or a rules cap (`RULE_CAPS` in sync-plan.js) is dropped, since + retrying it would fail the same way. The run's end entry goes to `run.syncUid`, the account its + pages were queued for 7. An expired or invalid token signs out and sets `authNotice: 'expired'`; the popup shows - "Your session expired". A `permission-denied` (the rules refusing a write) is not a sign-out; - the entry stays queued and `lastSync.error` records it + "Your session expired". A `permission-denied` or `invalid-argument` (the rules refusing a + write) is not a sign-out: the entry is marked `failed` and skipped so the rest go on, and + Export retries it. If every entry of a flush is refused, nothing is marked and the error is + thrown, since the account or the rules are the problem Cloud paths and shapes: `packages/schema/index.js` (keep the dashboard's copy identical, bump `SV` on a shape change). Run id `{sourceId}_{startMs}`, minted once in `engine.start`; page id `p0001`; diff --git a/README.md b/README.md index e098cd2..49d10c2 100644 --- a/README.md +++ b/README.md @@ -81,15 +81,21 @@ Signed in to a ProScan account in the popup (email and password) → Each saved page goes into the IndexedDB outbox, tagged with the account's uid; the end of the run goes in after its pages → A few seconds after a page, a run end, a popup open or a browser start, - the worker flushes the outbox (no alarms) - → sync.js writes one entry at a time: the page chunk, a product and a history - document per ASIN on the page, the run header and the source. latest, prev - and delta are replaced whole, never merged + the worker flushes the outbox (no alarms). After a failed flush the next + automatic one waits 30 s, doubling up to an hour + → sync.js writes one entry at a time: the page chunk, and a product and a + history document per ASIN on the page. The run header and the source go + once per run per flush. latest, prev and delta are replaced whole, never + merged; sourceIds keeps every source → An entry leaves the outbox only once its own writes commit; a page saved during a flush goes out before the flush returns + → An entry the rules refuse is set aside, so the entries behind it still sync; + the popup counts them and Export to ProScan tries them again → Export to ProScan flushes right away ``` +Write cost: a page of k products is 1 + 2k writes, plus 2 per run per flush. A 20 page run of 48 products a page is about 1,960 writes. Spark allows 20,000 writes a day for the whole project, shared by every user, so about 10 such runs a day fill it (audit F-23). + The document shapes, ids and validators are in `packages/schema/index.js`, which the dashboard copies. Money is integer cents, unknown values are null (left out of compact points), and every document carries `sv`. A signed-out user queues nothing. If Firebase drops the session, the popup says so and the queued pages wait for the same account to sign in again. ### AI Chatbot Flow @@ -171,7 +177,7 @@ For local Firebase work, `npm run build:dev` points the build at the emulators u The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, see its README). `npm run test:e2e` loads the built extension into Chromium and runs scrape scenarios against those pages, with every request answered locally. Run `npx playwright install --no-shell chromium` once first. Known bugs run as expected failures tagged with their audit finding id; `PROSCAN_SHOW_KNOWN=1 npm run test:e2e` shows what they fail on. -`npm run test:contract` runs the real sync module against the Firebase emulators with the dashboard's `firestore.rules` (from `PROSCAN_RULES`, or a `web` or `proscan-web` checkout next to this repo): queues of 1, 201 and 600 products, a page added mid-flush, replace semantics across runs, create-only `firstSeenAt` and the write count per run. It needs the Firebase CLI and Java, and uses the `demo-proscan` project only. +`npm run test:contract` runs the real sync module against the Firebase emulators with the dashboard's `firestore.rules` (from `PROSCAN_RULES`, or a `web` or `proscan-web` checkout next to this repo): queues of 1, 201 and 600 products, a page added mid-flush, replace semantics across runs, create-only `firstSeenAt`, the write count per run, a product in two sources and an entry the rules refuse. It refuses to start if any of its ports is taken, and only stops the emulator processes it started. It needs the Firebase CLI and Java, and uses the `demo-proscan` project only. The Jest suite includes a golden corpus of saved Amazon pages (`tests/pages/`, see its README). `npm run test:e2e` loads the built extension into Chromium and runs scrape scenarios against those pages, with every request answered locally. Run `npx playwright install --no-shell chromium` once first. Known bugs run as expected failures tagged with their audit finding id; `PROSCAN_SHOW_KNOWN=1 npm run test:e2e` shows what they fail on. diff --git a/popup/popup.js b/popup/popup.js index 8d844ba..0ce44f5 100644 --- a/popup/popup.js +++ b/popup/popup.js @@ -478,18 +478,22 @@ function showSignedIn(user) { /** * The sync line under the account: what is waiting, or when it last synced. * - * @param {{pending?: number, lastSync?: ?{at:number, error:?string}}} state + * @param {{pending?: number, failed?: number, lastSync?: ?{at:number, error:?string}}} state */ function showSyncState(state) { const pending = (state && state.pending) || 0; + const failed = (state && state.failed) || 0; const last = state && state.lastSync; let text = 'Scans sync to your ProScan dashboard automatically.'; if (pending > 0) { text = `${pending} page${pending === 1 ? '' : 's'} waiting to sync.`; - if (last && last.error) text += ' The last try failed; it will retry.'; + if (last && last.error) text += ' The last try failed; it will try again in a while, or now with Export to ProScan.'; } else if (last && !last.error) { text = 'Everything is synced.'; } + if (failed > 0) { + text += ` ${failed} page${failed === 1 ? '' : 's'} could not sync: ProScan refused ${failed === 1 ? 'it' : 'them'}. Export to ProScan tries again.`; + } elements.authSync.textContent = text; elements.authSync.classList.remove('hidden'); } @@ -647,12 +651,15 @@ async function handleExportToProScan() { if (response && response.ok) { const count = response.products || 0; - if (!response.entries) { + const failed = response.failed || 0; + if (failed > 0) { + updateStatus(`Synced ${count} product${count === 1 ? '' : 's'}; ${failed} page${failed === 1 ? '' : 's'} could not sync.`, 'warning'); + } else if (!response.entries) { updateStatus('Everything is already synced.', 'info'); } else { updateStatus(`Synced ${count} product${count === 1 ? '' : 's'} to ProScan`, 'success'); } - showSyncState({ pending: 0, lastSync: { at: Date.now(), error: null } }); + showSyncState({ pending: 0, failed, lastSync: { at: Date.now(), error: null } }); } else if (response && response.expired) { showSignInForm('expired'); updateStatus(response.error, 'warning'); diff --git a/scripts/background/engine.js b/scripts/background/engine.js index 87e78ee..f366113 100644 --- a/scripts/background/engine.js +++ b/scripts/background/engine.js @@ -174,6 +174,18 @@ function createEngine({ return [{ store: 'outbox', put: { runId, kind, uid, queuedAt: now(), ...extra } }]; } + /** + * The end-of-run entry for `run`. It goes to the account its pages were + * queued for (run.syncUid), even if that account signed out mid-run, so + * the cloud header does not stay active forever. + */ + async function endEntry(run) { + if (!flags.CLOUD_SYNC || !(run.page > 0)) return []; + const uid = run.syncUid || await owner(); + if (!uid) return []; + return [{ store: 'outbox', put: { runId: run.runId, kind: 'run', uid, queuedAt: now() } }]; + } + /** Ends `run` for `reason` and tells its tab. Returns the ended run. */ async function end(run, reason) { const ended = Run.finish(run, reason, now()); @@ -181,7 +193,7 @@ function createEngine({ await saveRun(ended); try { // A run with saved pages sends its final header to the cloud. - const queued = ended.page > 0 ? await outbox(ended.runId, 'run') : []; + const queued = await endEntry(ended); await (await db()).write([{ store: 'runs', put: durable(ended) }, ...queued]); } catch (err) { log.warn('[ProScan] Could not record the end of the run:', err.message); @@ -292,7 +304,7 @@ function createEngine({ const rec = latestId ? await store.get('runs', latestId) : null; if (!Run.isActive(rec)) return null; const ended = Run.finish(rec, reason, now()); - await store.write([{ store: 'runs', put: durable(ended) }]); + await store.write([{ store: 'runs', put: durable(ended) }, ...await endEntry(ended)]); return ended; } @@ -379,15 +391,17 @@ function createEngine({ total: Number.isInteger(result.total) && result.total > 0 ? result.total : null } }); - ops.push(...await outbox(runId, 'page', { pageIndex: page })); + const queuedPage = await outbox(runId, 'page', { pageIndex: page }); + ops.push(...queuedPage); next = { ...run, page, itemCount: run.itemCount + fresh.length, heartbeat: t, lastUrl: url, nextHref: result.nextHref || null, awaiting: false }; + if (queuedPage.length && !next.syncUid) next.syncUid = queuedPage[0].put.uid; if (ending) { next = Run.finish(next, ending, t); - ops.push(...await outbox(runId, 'run')); + ops.push(...await endEntry(next)); } else { next.navAt = t + Run.pageDelay(random()); } @@ -458,10 +472,13 @@ function createEngine({ }).then((out) => { tick(); return out; }); } + /** Resolves true when this ended the run. */ function tabRemoved(tabId) { return serial(async () => { const run = await getRun(); - if (Run.isActive(run) && run.tabId === tabId) await end(run, 'interrupted'); + if (!Run.isActive(run) || run.tabId !== tabId) return false; + await end(run, 'interrupted'); + return true; }); } @@ -471,13 +488,14 @@ function createEngine({ * the tabs permission a non-Amazon URL is hidden, which also counts. */ function tabUpdated(tabId, info, tab) { - if (!info || info.status !== 'loading') return Promise.resolve(); + if (!info || info.status !== 'loading') return Promise.resolve(false); return serial(async () => { const run = await getRun(); - if (!Run.owns(run, tabId) || run.state !== 'running' || run.awaiting) return; + if (!Run.owns(run, tabId) || run.state !== 'running' || run.awaiting) return false; const url = info.url || (tab && tab.url); - if (url && url === run.lastUrl) return; + if (url && url === run.lastUrl) return false; await end(run, 'interrupted'); + return true; }); } @@ -493,7 +511,7 @@ function createEngine({ const latest = await store.getMeta('latestRunId'); const rec = latest ? await store.get('runs', latest) : null; if (Run.isActive(rec)) { - const queued = rec.page > 0 ? await outbox(rec.runId, 'run') : []; + const queued = await endEntry(rec); await store.write([{ store: 'runs', put: durable(Run.finish(rec, reason, now())) }, ...queued]); } }); diff --git a/scripts/background/service-worker.js b/scripts/background/service-worker.js index 7a9cb5c..936f601 100644 --- a/scripts/background/service-worker.js +++ b/scripts/background/service-worker.js @@ -42,6 +42,8 @@ const chatDeps = { // The run // Saved pages and ended runs wake the sync; see scheduleFlush below. const thenFlush = (p) => p.then((out) => { scheduleFlush(); return out; }); +// A tab event wakes it only when it ended the run. +const flushIfEnded = (p) => p.then((ended) => { if (ended) scheduleFlush(); return ended; }); router.on(Msg.T.START_RUN, (m) => engine.start(m)); router.on(Msg.T.STOP_RUN, () => thenFlush(engine.stop())); @@ -94,8 +96,8 @@ chrome.runtime.onStartup.addListener(() => { }); // Neither listener needs the tabs permission. -chrome.tabs.onRemoved.addListener((tabId) => { thenFlush(engine.tabRemoved(tabId)); }); -chrome.tabs.onUpdated.addListener((tabId, info, tab) => { thenFlush(engine.tabUpdated(tabId, info, tab)); }); +chrome.tabs.onRemoved.addListener((tabId) => { flushIfEnded(engine.tabRemoved(tabId)); }); +chrome.tabs.onUpdated.addListener((tabId, info, tab) => { flushIfEnded(engine.tabUpdated(tabId, info, tab)); }); // ── ProScan account and cloud sync ────────────────────────────────────────── // The popup is a plain page; it signs in and exports through these messages, @@ -109,6 +111,16 @@ const DASHBOARD_URL = 'https://proscanbot.web.app/dashboard/'; const FLUSH_DELAY_MS = 3000; let flushTimer = null; +/** + * After a failed automatic flush the next one waits 30 s, doubling per + * failure up to an hour, so a refused or throttled entry is not replayed + * (with its product reads) on every page load. Export ignores the wait. + */ +function retryDelayMs(failures) { + if (!failures) return 0; + return Math.min(30000 * 2 ** (failures - 1), 60 * 60 * 1000); +} + const sync = createSync({ db: firestore, openStore: () => engine.db() }); /** Resolve the current Firebase user, waiting for auth to rehydrate from @@ -160,17 +172,27 @@ async function expireSession() { await signOut(auth).catch(() => {}); } -/** Writes the outbox for the signed-in account now. */ -async function flushNow() { +/** + * Writes the outbox for the signed-in account now. `manual` (Export) also + * retries entries the rules refused and skips the backoff. + */ +async function flushNow({ manual = false } = {}) { if (!Flags.CLOUD_SYNC) return { skipped: true }; const user = await currentUser(); if (!user) return { skipped: true }; + const { lastSync } = await chrome.storage.local.get('lastSync'); + const failures = (lastSync && lastSync.error && lastSync.failures) || 0; + if (!manual && lastSync && lastSync.error && Date.now() < lastSync.at + retryDelayMs(failures)) { + return { skipped: true, backoff: true }; + } try { - const out = await sync.flush(user.uid); - await chrome.storage.local.set({ lastSync: { at: Date.now(), error: null } }); + const out = await sync.flush(user.uid, { retryFailed: manual }); + await chrome.storage.local.set({ lastSync: { at: Date.now(), error: null, failures: 0 } }); return out; } catch (err) { - await chrome.storage.local.set({ lastSync: { at: Date.now(), error: (err && err.code) || 'unknown' } }); + await chrome.storage.local.set({ + lastSync: { at: Date.now(), error: (err && err.code) || 'unknown', failures: failures + 1 }, + }); if (isAuthError(err)) await expireSession(); throw err; } @@ -219,6 +241,7 @@ router.on(Msg.T.PROSCAN_AUTH_STATE, async () => { user: publicUser(user), notice: user ? null : authNotice || null, pending: user ? await sync.pending(user.uid).catch(() => 0) : 0, + failed: user ? await sync.failed(user.uid).catch(() => 0) : 0, lastSync: lastSync || null, dashboardUrl: DASHBOARD_URL, }; @@ -264,7 +287,8 @@ router.on(Msg.T.PROSCAN_EXPORT, async () => { const user = await currentUser(); if (!user) return { error: 'Sign in to ProScan first.' }; try { - return { ok: true, ...(await flushNow()) }; + const out = await flushNow({ manual: true }); + return { ok: true, ...out, failed: await sync.failed(user.uid).catch(() => out.failed || 0) }; } catch (err) { if (isAuthError(err)) return { error: 'Your session expired. Sign in again to keep syncing.', expired: true }; return { error: 'Could not reach ProScan. Your scans are kept and will sync later.' }; diff --git a/scripts/background/sync-plan.js b/scripts/background/sync-plan.js index f0261ff..1c9ed77 100644 --- a/scripts/background/sync-plan.js +++ b/scripts/background/sync-plan.js @@ -7,11 +7,23 @@ * documents of every ASIN on it, the run header and the source. The end of * a run writes the header and the source again with the final status. * - * Each write is {path, data, fields}. `fields` null means replace the whole - * document; otherwise only those fields are written, each replaced whole - * (Firestore mergeFields), so a price that failed to parse is gone from - * `latest` instead of kept from last week. A field given as an array is a - * field path, e.g. ['d', '2026-06-09']. + * Each write is {path, data, fields} or {path, data, merge: true}. `fields` + * null means replace the whole document; otherwise only those fields are + * written, each replaced whole (Firestore mergeFields). A field given as an + * array is a field path, e.g. ['d', '2026-06-09']. + * + * A product document is a `merge` write, so its sourceIds arrayUnion keeps + * the sources written before (in a mergeFields mask it would be replaced + * whole). `latest`, `prev` and `delta` are still replaced whole: every key + * they can have and this write lacks is sent as a delete, so a price that + * failed to parse is gone from `latest` instead of kept from last week. + * + * The run header and the source are written once per flush for each run, + * on its last entry in that flush (`header`), not once per page. + * + * Firestore rules cap string lengths and list sizes the schema validators + * do not know about. checkRuleCaps() applies the same caps here, so an + * entry the rules would refuse on every try fails planning instead. * * @module SyncPlan */ @@ -23,12 +35,71 @@ import { const DAY_MS = 86400000; const same = (v) => v; +/** What a field delete looks like when the caller passes no converter. */ +export const DELETE = Object.freeze({ delete: true }); + +/** Caps from the dashboard's firestore.rules that the schema does not check. */ +export const RULE_CAPS = { + keyword: 300, url: 2000, sellerId: 40, name: 1000, productUrl: 500, img: 1000, + runId: 260, reason: 100, maxPages: 1000, page: 1000, kind: 20, sourceIds: 200 +}; + +const tooLong = (v, max) => typeof v === 'string' && v.length > max; + +/** Throws when `doc` of `kind` breaks a rules cap. */ +export function checkRuleCaps(kind, doc) { + const errs = []; + const c = RULE_CAPS; + if (kind === 'source') { + if (tooLong(doc.keyword, c.keyword)) errs.push(`keyword: at most ${c.keyword} characters`); + if (tooLong(doc.url, c.url)) errs.push(`url: at most ${c.url} characters`); + if (tooLong(doc.sellerId, c.sellerId)) errs.push(`sellerId: at most ${c.sellerId} characters`); + if (tooLong(doc.lastRunId, c.runId)) errs.push(`lastRunId: at most ${c.runId} characters`); + if (!/^[sk]_[A-Za-z0-9_-]{1,200}$/.test(doc.sourceId || '')) errs.push('sourceId: s_ or k_ and 1 to 200 id characters'); + } else if (kind === 'run') { + if (tooLong(doc.reason, c.reason)) errs.push(`reason: at most ${c.reason} characters`); + if (doc.maxPages > c.maxPages) errs.push(`maxPages: at most ${c.maxPages}`); + } else if (kind === 'page') { + if (doc.page > c.page) errs.push(`page: at most ${c.page}`); + if (tooLong(doc.kind, c.kind)) errs.push(`kind: at most ${c.kind} characters`); + } else if (kind === 'product') { + if (tooLong(doc.name, c.name)) errs.push(`name: at most ${c.name} characters`); + if (tooLong(doc.url, c.productUrl)) errs.push(`url: at most ${c.productUrl} characters`); + if (tooLong(doc.img, c.img)) errs.push(`img: at most ${c.img} characters`); + if (doc.latest && tooLong(doc.latest.runId, c.runId)) errs.push(`latest.runId: at most ${c.runId} characters`); + if (tooLong(doc.firstRunId, c.runId)) errs.push(`firstRunId: at most ${c.runId} characters`); + } + if (errs.length) throw new Error(`${kind} document breaks a rules cap: ${errs.join('; ')}`); + return doc; +} + +/** Every key each replaced map of a product document can hold. */ +const MAP_KEYS = { + latest: ['p', 'r', 'v', 'pr', 'rk', 'at', 'runId', 'dayKey'], + prev: ['p', 'r', 'v', 'pr', 'rk', 'at'], + delta: ['p', 'pPct', 'r', 'v', 'days'] +}; + +/** `doc` with the keys its replaced maps lack set to `del()`, for a merge write. */ +function withDeletes(doc, del) { + const out = { ...doc }; + for (const [field, keys] of Object.entries(MAP_KEYS)) { + const v = doc[field]; + if (!v || typeof v !== 'object') continue; + const filled = { ...v }; + for (const k of keys) if (!(k in filled)) filled[k] = del(); + out[field] = filled; + } + return out; +} /** Placements of `product` on page `pageIndex`. */ function onPage(product, pageIndex) { return (product.placements || []).filter((pl) => pl.page === pageIndex); } +const clamp = (v, max) => (typeof v === 'string' && v.length > max ? v.slice(0, max) : v); + /** The run's source id, day key and source, also for a run from before 2.3. */ export function runKeys(run) { const source = run.source || {}; @@ -38,7 +109,12 @@ export function runKeys(run) { return { sourceId, dayKey, - source: { type: found.type, sellerId: found.sellerId || null, keyword: found.keyword || null, url: found.url || source.url || null } + source: { + type: found.type, + sellerId: found.sellerId || null, + keyword: clamp(found.keyword || null, RULE_CAPS.keyword), + url: clamp(found.url || source.url || null, RULE_CAPS.url) + } }; } @@ -166,9 +242,11 @@ function productDoc(run, keys, p, time, union) { * @param {Set} [args.missing] ASINs with no product document yet * @param {function(number):*} [args.time] ms to a Firestore Timestamp * @param {function(string[]):*} [args.union] values to arrayUnion - * @returns {{path: string[], data: Object, fields: ?Array}[]} + * @param {function():*} [args.del] a field delete, for merge writes + * @param {boolean} [args.header] also write the run header and the source + * @returns {{path: string[], data: Object, fields?: ?Array, merge?: boolean}[]} */ -export function planEntry({ entry, run: rec, products, pages, missing = new Set(), time = same, union = same }) { +export function planEntry({ entry, run: rec, products, pages, missing = new Set(), time = same, union = same, del = () => DELETE, header: withHeader = true }) { const run = { ...rec, startedAt: timeMs(rec.startedAt), finishedAt: timeMs(rec.finishedAt) }; const ws = ['workspaces', entry.uid]; const keys = runKeys(run); @@ -179,6 +257,7 @@ export function planEntry({ entry, run: rec, products, pages, missing = new Set( if (pageRec) { const chunk = pageDoc(run, pageRec, products, time); assertValid('page', chunk); + checkRuleCaps('page', chunk); writes.push({ path: [...ws, 'runs', run.runId, 'pages', pageIdOf(pageRec.pageIndex)], data: chunk, fields: null }); } for (const p of products) { @@ -189,7 +268,8 @@ export function planEntry({ entry, run: rec, products, pages, missing = new Set( doc.firstRunId = run.runId; } assertValid('product', { ...doc, sourceIds: [keys.sourceId] }); - writes.push({ path: [...ws, 'products', p.asin], data: doc, fields: Object.keys(doc) }); + checkRuleCaps('product', doc); + writes.push({ path: [...ws, 'products', p.asin], data: withDeletes(doc, del), merge: true }); const point = pointOf(p); assertValid('history', { sv: SV, asin: p.asin, d: { [keys.dayKey]: point } }); @@ -201,11 +281,14 @@ export function planEntry({ entry, run: rec, products, pages, missing = new Set( } } - const header = runDoc(run, keys, products, pages, time); - assertValid('run', header); - writes.push({ path: [...ws, 'runs', run.runId], data: header, fields: Object.keys(header) }); + if (!withHeader && entry.kind !== 'run') return writes; + const head = runDoc(run, keys, products, pages, time); + assertValid('run', head); + checkRuleCaps('run', head); + writes.push({ path: [...ws, 'runs', run.runId], data: head, fields: Object.keys(head) }); const source = sourceDoc(run, keys, pages, time); assertValid('source', source); + checkRuleCaps('source', source); writes.push({ path: [...ws, 'sources', keys.sourceId], data: source, fields: Object.keys(source) }); return writes; } diff --git a/scripts/background/sync.js b/scripts/background/sync.js index ace2117..4411a1f 100644 --- a/scripts/background/sync.js +++ b/scripts/background/sync.js @@ -11,6 +11,13 @@ * Every write is idempotent, so an entry that fails halfway is simply * written again on the next flush. * + * The rules can refuse one entry for good (permission-denied or + * invalid-argument on a document the schema allowed). That entry is marked + * failed and skipped, so the entries behind it still sync; a flush with + * `retryFailed` (Export to ProScan) tries the failed ones again. When every + * entry of a flush is refused, the cause is the account or the rules, not + * one entry, so nothing is marked and the error goes to the caller. + * * Authored as ESM and bundled into the service worker by esbuild. The * contract test runs this same file in Node against the emulators. * @@ -23,11 +30,12 @@ import { writeBatch, Timestamp, arrayUnion, + deleteField, FieldPath, } from 'firebase/firestore'; import { planEntry, firstSeenCandidates } from './sync-plan.js'; -const FIRESTORE = { doc, getDoc, writeBatch, Timestamp, arrayUnion, FieldPath }; +const FIRESTORE = { doc, getDoc, writeBatch, Timestamp, arrayUnion, deleteField, FieldPath }; /** Firestore allows 500 writes per batch; stay under it. */ export const BATCH_LIMIT = 450; @@ -42,6 +50,15 @@ export function isAuthError(err) { return !!err && AUTH_CODES.includes(err.code); } +/** The rules or the backend refused this write for good; a retry would fail the same way. */ +const REFUSAL_CODES = ['permission-denied', 'invalid-argument']; + +export function isRefusal(err) { + return !!err && REFUSAL_CODES.includes(err.code); +} + +const EMPTY = () => ({ entries: 0, pages: 0, runs: 0, products: 0, writes: 0, failed: 0 }); + /** * @param {Object} deps * @param {Object} deps.db - the Firestore instance @@ -55,14 +72,17 @@ export function createSync({ db, openStore, fs = FIRESTORE, batchLimit = BATCH_L const ref = (path) => fs.doc(db, ...path); const time = (ms) => fs.Timestamp.fromMillis(ms); + // Product writes need field deletes; without them every page would fail planning and be dropped. + if (typeof fs.deleteField !== 'function') throw new Error('createSync: fs.deleteField is missing'); const union = (vals) => fs.arrayUnion(...vals); + const del = () => fs.deleteField(); const field = (f) => (Array.isArray(f) ? new fs.FieldPath(...f) : f); - /** Entries queued for `uid`, oldest first. */ - async function mine(store, uid) { + /** Entries queued for `uid`, oldest first; failed ones only when `failed` is true. */ + async function mine(store, uid, { failed = false } = {}) { const all = await store.getAll('outbox'); return all - .filter((e) => e.uid === uid && (e.kind === 'page' || e.kind === 'run')) + .filter((e) => e.uid === uid && (e.kind === 'page' || e.kind === 'run') && !!e.failed === failed) .sort((a, b) => a.seq - b.seq); } @@ -77,15 +97,18 @@ export function createSync({ db, openStore, fs = FIRESTORE, batchLimit = BATCH_L return missing; } - /** Writes one entry. Returns how many writes it took, or null if it had nothing to write. */ - async function commitEntry(store, entry) { + /** + * Writes one entry, with the run header and source when `header` is true. + * Returns how many writes it took, or null if it had nothing to write. + */ + async function commitEntry(store, entry, header) { const run = await store.get('runs', entry.runId); if (!run) return null; const [products, pages] = await Promise.all([store.runProducts(entry.runId), store.runPages(entry.runId)]); const missing = await missingOf(entry.uid, firstSeenCandidates(entry, products)); let writes; try { - writes = planEntry({ entry, run, products, pages, missing, time, union }); + writes = planEntry({ entry, run, products, pages, missing, time, union, del, header }); } catch (err) { // Planning is pure, so a retry would fail the same way and hold up the queue. log.warn('[ProScan] Dropped an outbox entry that cannot be written:', err.message); @@ -95,7 +118,8 @@ export function createSync({ db, openStore, fs = FIRESTORE, batchLimit = BATCH_L for (let i = 0; i < writes.length; i += batchLimit) { const batch = fs.writeBatch(db); for (const w of writes.slice(i, i + batchLimit)) { - if (w.fields) batch.set(ref(w.path), w.data, { mergeFields: w.fields.map(field) }); + if (w.merge) batch.set(ref(w.path), w.data, { merge: true }); + else if (w.fields) batch.set(ref(w.path), w.data, { mergeFields: w.fields.map(field) }); else batch.set(ref(w.path), w.data); } if (onBatch) await onBatch({ entry, writes: Math.min(batchLimit, writes.length - i) }); @@ -104,16 +128,28 @@ export function createSync({ db, openStore, fs = FIRESTORE, batchLimit = BATCH_L return { writes: writes.length, products: products.filter((p) => (p.placements || []).some((pl) => pl.page === entry.pageIndex)).length }; } - async function drain(uid) { + async function drain(uid, state) { const store = await openStore(); - const totals = { entries: 0, pages: 0, runs: 0, products: 0, writes: 0 }; + const totals = EMPTY(); const runs = new Set(); // New entries can arrive while we write; keep going until none are left. for (let round = 0; round < 1000; round++) { const entries = await mine(store, uid); if (entries.length === 0) break; - for (const entry of entries) { - const done = await commitEntry(store, entry); + // The header and source go with each run's last entry in this round. + const lastOf = new Map(entries.map((e, i) => [e.runId, i])); + const refused = []; + for (let i = 0; i < entries.length; i++) { + const entry = entries[i]; + let done; + try { + done = await commitEntry(store, entry, lastOf.get(entry.runId) === i); + } catch (err) { + if (!isRefusal(err)) throw err; + refused.push({ entry, err }); + continue; + } + state.through = true; await store.write([{ store: 'outbox', delete: entry.seq }]); totals.entries++; if (!done) continue; @@ -124,29 +160,53 @@ export function createSync({ db, openStore, fs = FIRESTORE, batchLimit = BATCH_L totals.products += done.products; } } + if (refused.length === 0) continue; + // Entries refused before stay failed. For the others, nothing getting + // through at all points at the account or the rules, not these entries. + const before = refused.filter(({ entry }) => state.retried.has(entry.seq)); + const fresh = refused.filter(({ entry }) => !state.retried.has(entry.seq)); + const mark = state.through ? refused : before; + if (mark.length) { + await store.write(mark.map(({ entry, err }) => ({ + store: 'outbox', put: { ...entry, failed: { code: err.code, at: Date.now() } }, + }))); + } + if (!state.through && fresh.length) throw fresh[0].err; + totals.failed += mark.length; + log.warn(`[ProScan] The rules refused ${refused.length} outbox entr${refused.length === 1 ? 'y' : 'ies'}; the rest go on.`); } totals.runs = runs.size; return totals; } + /** Puts `uid`'s failed entries back in line. */ + async function unfail(uid) { + const store = await openStore(); + const failed = await mine(store, uid, { failed: true }); + if (failed.length) await store.write(failed.map(({ failed: _f, ...entry }) => ({ store: 'outbox', put: entry }))); + return new Set(failed.map((e) => e.seq)); + } + /** * Writes everything queued for `uid`. One flush at a time; a call during * a flush makes it look again once more before it resolves. * * @param {string} uid - * @returns {Promise<{entries:number, pages:number, runs:number, products:number, writes:number}>} + * @param {{retryFailed?: boolean}} [opts] - also try entries the rules refused before + * @returns {Promise<{entries:number, pages:number, runs:number, products:number, writes:number, failed:number}>} */ - function flush(uid) { - if (!uid) return Promise.resolve({ entries: 0, pages: 0, runs: 0, products: 0, writes: 0 }); + function flush(uid, { retryFailed = false } = {}) { + if (!uid) return Promise.resolve(EMPTY()); if (running) { again = true; return running; } running = (async () => { - const totals = { entries: 0, pages: 0, runs: 0, products: 0, writes: 0 }; + const totals = EMPTY(); + const state = { through: false, retried: retryFailed ? await unfail(uid) : new Set() }; do { again = false; - const t = await drain(uid); + const t = await drain(uid, state); for (const k of Object.keys(totals)) totals[k] += t[k]; } while (again); return totals; @@ -157,11 +217,17 @@ export function createSync({ db, openStore, fs = FIRESTORE, batchLimit = BATCH_L return running; } - /** How many entries are waiting for `uid`. */ + /** How many entries are waiting for `uid`, not counting failed ones. */ async function pending(uid) { if (!uid) return 0; return (await mine(await openStore(), uid)).length; } - return { flush, pending }; + /** How many of `uid`'s entries the rules refused. */ + async function failedCount(uid) { + if (!uid) return 0; + return (await mine(await openStore(), uid, { failed: true })).length; + } + + return { flush, pending, failed: failedCount }; } diff --git a/tests/contract/sync.contract.mjs b/tests/contract/sync.contract.mjs index b88aee1..2a8f533 100644 --- a/tests/contract/sync.contract.mjs +++ b/tests/contract/sync.contract.mjs @@ -4,8 +4,8 @@ // which starts the emulators. // // Queue sizes 1, 201 and 600, a page appended in the middle of a flush, -// replace semantics across runs, create-only firstSeenAt, and a write count -// per run. +// replace semantics across runs, create-only firstSeenAt, a write count +// per run, a product in two sources, and an entry the rules refuse. import 'fake-indexeddb/auto'; import { test, before, after } from 'node:test'; @@ -73,10 +73,10 @@ function extension(uid) { }); const sender = { tab: { id: 1 } }; - /** Starts a keyword run at `startMs`. */ - async function start(keyword, startMs) { + /** Starts a keyword run at `startMs`, or a run on `at` when given. */ + async function start(keyword, startMs, at = null) { clock.t = startMs; - url = `https://www.amazon.com/s?k=${encodeURIComponent(keyword)}`; + url = at || `https://www.amazon.com/s?k=${encodeURIComponent(keyword)}`; const resp = await engine.start({ tabId: 1 }); assert.equal(resp.ok, true, JSON.stringify(resp)); return resp.runId; @@ -101,8 +101,8 @@ function extension(uid) { } /** A whole run of `products`, `perPage` to a page. */ - async function scrape(keyword, startMs, products, perPage) { - const runId = await start(keyword, startMs); + async function scrape(keyword, startMs, products, perPage, at = null) { + const runId = await start(keyword, startMs, at); const pages = Math.ceil(products.length / perPage); for (let i = 0; i < pages; i++) { await page(runId, i + 1, products.slice(i * perPage, (i + 1) * perPage), { last: i === pages - 1, total: products.length }); @@ -128,8 +128,11 @@ function products(n, { price = (i) => 1000 + i } = {}) { }); } -/** Writes an entry takes: chunk + 2 per product + run + source for a page; run + source at the end. */ -const expectedWrites = (pageSizes) => pageSizes.reduce((n, k) => n + 1 + 2 * k + 2, 0) + 2; +/** + * Writes a run takes: per page the chunk and 2 per product (document and + * history); the header and source once per flush round the run had entries in. + */ +const expectedWrites = (pageSizes, rounds = 1) => pageSizes.reduce((n, k) => n + 1 + 2 * k, 0) + 2 * rounds; async function account(email) { const cred = await createUserWithEmailAndPassword(auth, email, 'contract-pass-1'); @@ -158,7 +161,7 @@ test('queue size 1: one product on one page', async () => { assert.equal(runId, `k_single-mug_${start}`); const totals = await ext.sync().flush(uid); - assert.deepEqual(totals, { entries: 2, pages: 1, runs: 1, products: 1, writes: expectedWrites([1]) }); + assert.deepEqual(totals, { entries: 2, pages: 1, runs: 1, products: 1, writes: expectedWrites([1]), failed: 0 }); assert.equal(await (await ext.store()).count('outbox'), 0); const p = (await getDoc(ws(uid, 'products', asinOf(0)))).data(); @@ -202,7 +205,7 @@ test('queue size 201, with a page appended in the middle of the flush', async () assert.equal(appended, true); assert.equal(totals.pages, 5); assert.equal(totals.products, 201); - assert.equal(totals.writes, expectedWrites([48, 48, 48, 48, 9])); + assert.equal(totals.writes, expectedWrites([48, 48, 48, 48, 9], 2)); assert.equal(await (await ext.store()).count('outbox'), 0); assert.equal(await count(uid, 'products'), 201); @@ -286,3 +289,37 @@ test('another account cannot write this account\'s queue, and the entries stay', await assert.rejects(ext.sync().flush(owner), (err) => err.code === 'permission-denied' && !isAuthError(err)); assert.equal(await (await ext.store()).count('outbox'), 2); }); + +test('a product seen by a storefront and a keyword run lists both sources (NEW-SYNC-1)', async () => { + const uid = await account(`two-${Date.now()}@contract.test`); + const ext = extension(uid); + const day = Date.parse('2026-06-14T10:00:00Z'); + await ext.scrape('store', day, products(6), 60, 'https://www.amazon.com/s?me=A3K9XELT4QZ6M2'); + await ext.sync().flush(uid); + // The keyword run sees the first 3 again. + await ext.scrape('water bottle', day + 60000, products(3), 60); + await ext.sync().flush(uid); + + const both = (await getDoc(ws(uid, 'products', asinOf(1)))).data(); + assert.deepEqual([...both.sourceIds].sort(), ['k_water-bottle', 's_A3K9XELT4QZ6M2']); + const one = (await getDoc(ws(uid, 'products', asinOf(5)))).data(); + assert.deepEqual(one.sourceIds, ['s_A3K9XELT4QZ6M2']); +}); + +test('an entry the rules refuse is set aside and the next scan still syncs (EXT9-1)', async () => { + const uid = await account(`skew-${Date.now()}@contract.test`); + const ext = extension(uid); + // A clock two hours fast: the rules refuse times over an hour ahead. + const ahead = Date.now() + 2 * 60 * 60 * 1000; + const bad = await ext.scrape('fast clock', ahead, products(2), 60); + // Other ASINs: these would carry the fast clock in prev.at. + const good = await ext.scrape('garden hose', Date.now() - 60000, products(4).slice(2), 60); + + const totals = await ext.sync().flush(uid); + assert.equal(totals.failed, 2); + assert.equal((await getDoc(ws(uid, 'runs', good))).exists(), true); + assert.equal((await getDoc(ws(uid, 'runs', bad))).exists(), false); + const sync = ext.sync(); + assert.equal(await sync.pending(uid), 0); + assert.equal(await sync.failed(uid), 2); +}); diff --git a/tests/unit/engine.test.js b/tests/unit/engine.test.js index 63cbc8c..1c10dcd 100644 --- a/tests/unit/engine.test.js +++ b/tests/unit/engine.test.js @@ -705,3 +705,30 @@ describe('old runs', () => { db.close(); }); }); + +describe('the end of a synced run (EXT9-5)', () => { + test('signed out mid-run, the final header still goes to the account its pages went to', async () => { + const rig = createRig({ site: simpleSite(3), flags: { CLOUD_SYNC: true }, local: { account: { uid: 'u1' } } }); + await startIn(rig, 'leave'); + await settle(); + await rig.local.remove('account'); + expect(await rig.popup({ type: 'STOP_RUN' })).toMatchObject({ stopped: true }); + const db = await rig.db(); + const outbox = await db.getAll('outbox'); + db.close(); + expect(outbox.map((e) => [e.kind, e.uid])).toEqual([['page', 'u1'], ['run', 'u1']]); + }); + + test('a run left live only in IndexedDB queues its final header when it is ended', async () => { + const rig = createRig({ site: simpleSite(3), flags: { CLOUD_SYNC: true }, local: { account: { uid: 'u1' } } }); + await startIn(rig, 'gone'); + await settle(1000); + await rig.session.remove('run'); + rig.killWorker(); + await rig.popup({ type: 'GET_STATE' }); + const db = await rig.db(); + const kinds = (await db.getAll('outbox')).map((e) => e.kind); + db.close(); + expect(kinds).toContain('run'); + }); +}); diff --git a/tests/unit/service-worker.test.js b/tests/unit/service-worker.test.js index a34e64a..4de1b4a 100644 --- a/tests/unit/service-worker.test.js +++ b/tests/unit/service-worker.test.js @@ -24,7 +24,7 @@ jest.mock('firebase/auth/web-extension', () => { }, { virtual: true }); jest.mock('../../scripts/background/sync.js', () => { const flush = jest.fn(async () => ({ entries: 0, pages: 0, runs: 0, products: 0, writes: 0 })); - return { createSync: () => ({ flush, pending: async () => 0 }), isAuthError: () => false, __flush: flush }; + return { createSync: () => ({ flush, pending: async () => 0, failed: async () => 0 }), isAuthError: () => false, __flush: flush }; }); require('fake-indexeddb/auto'); @@ -38,12 +38,14 @@ const V20 = JSON.parse(fs.readFileSync(path.join(__dirname, '../fixtures/v2.0-st const messageListeners = []; const installedListeners = []; const startupListeners = []; +const tabUpdatedListeners = []; beforeAll(() => { chrome.runtime.id = 'test-extension'; chrome.runtime.onMessage.addListener = (fn) => messageListeners.push(fn); chrome.runtime.onInstalled = { addListener: (fn) => installedListeners.push(fn) }; chrome.runtime.onStartup = { addListener: (fn) => startupListeners.push(fn) }; + chrome.tabs.onUpdated = { addListener: (fn) => tabUpdatedListeners.push(fn) }; require('../../scripts/background/service-worker.js'); }); @@ -71,12 +73,13 @@ test("Export to ProScan flushes the signed-in account's outbox", async () => { flush.mockClear(); const resp = await send({ type: 'PROSCAN_EXPORT' }); expect(resp).toMatchObject({ ok: true, entries: 0 }); - expect(flush).toHaveBeenCalledWith('u1'); + // Export also retries entries the rules refused before. + expect(flush).toHaveBeenCalledWith('u1', { retryFailed: true }); }); test('the auth state says who is signed in and what waits to sync', async () => { const st = await send({ type: 'PROSCAN_AUTH_STATE' }); - expect(st).toMatchObject({ user: { uid: 'u1' }, notice: null, pending: 0, dashboardUrl: expect.stringMatching(/^https:/) }); + expect(st).toMatchObject({ user: { uid: 'u1' }, notice: null, pending: 0, failed: 0, dashboardUrl: expect.stringMatching(/^https:/) }); }); test('a password reset answers the same whether or not the account exists (F-56)', async () => { @@ -146,3 +149,40 @@ test('signing out on purpose clears the account without an expired notice', asyn expect(chrome.storage.local._getStore()).not.toHaveProperty('account'); expect(chrome.storage.local._getStore()).not.toHaveProperty('authNotice'); }); + +describe('automatic flushes (EXT9-2)', () => { + const wait = (ms) => new Promise((r) => setTimeout(r, ms)); + + test('a page load in some tab with no run does not flush', async () => { + // Let flushes asked for before this test go out first. + await wait(3200); + await settle(); + flush.mockClear(); + tabUpdatedListeners.forEach((fn) => fn(42, { status: 'loading', url: 'https://www.amazon.com/s?k=x' }, { id: 42 })); + await settle(); + await wait(3200); + expect(flush).not.toHaveBeenCalled(); + }, 15000); + + test('after a failed flush the automatic one waits, and Export does not', async () => { + await chrome.storage.local.set({ lastSync: { at: Date.now(), error: 'unavailable', failures: 2 } }); + flush.mockClear(); + // Opening the popup asks for a flush right away. + await send({ type: 'PROSCAN_AUTH_STATE' }); + await wait(50); + await settle(); + expect(flush).not.toHaveBeenCalled(); + await send({ type: 'PROSCAN_EXPORT' }); + expect(flush).toHaveBeenCalledWith('u1', { retryFailed: true }); + expect(chrome.storage.local._getStore().lastSync).toMatchObject({ error: null, failures: 0 }); + }); + + test('without a recent failure the popup flush goes out', async () => { + await chrome.storage.local.set({ lastSync: { at: Date.now() - 2 * 60 * 60 * 1000, error: 'unavailable', failures: 9 } }); + flush.mockClear(); + await send({ type: 'PROSCAN_AUTH_STATE' }); + await wait(50); + await settle(); + expect(flush).toHaveBeenCalledWith('u1', { retryFailed: false }); + }); +}); diff --git a/tests/unit/sync-plan.test.js b/tests/unit/sync-plan.test.js index 47f8853..884c2f3 100644 --- a/tests/unit/sync-plan.test.js +++ b/tests/unit/sync-plan.test.js @@ -3,7 +3,7 @@ * * The pure sync plan: outbox entry and IndexedDB records in, cloud writes out. */ -const { planEntry, firstSeenCandidates, runKeys } = require('../../scripts/background/sync-plan.js'); +const { planEntry, firstSeenCandidates, runKeys, checkRuleCaps, RULE_CAPS, DELETE } = require('../../scripts/background/sync-plan.js'); const S = require('../../packages/schema/index.js'); const START = Date.parse('2026-06-09T14:02:11Z'); @@ -49,6 +49,11 @@ const products = [ const entry = (patch) => ({ seq: 1, runId: RUN_ID, uid: 'u1', ...patch }); const byPath = (writes) => Object.fromEntries(writes.map((w) => [w.path.join('/'), w])); +/** A merge write's data without the deletes that clear stale keys. */ +const kept = (data) => Object.fromEntries(Object.entries(data) + .filter(([, v]) => v !== DELETE) + .map(([k, v]) => [k, v && typeof v === 'object' && !Array.isArray(v) && Object.values(v).includes(DELETE) + ? Object.fromEntries(Object.entries(v).filter(([, x]) => x !== DELETE)) : v])); describe('a page entry', () => { const writes = planEntry({ entry: entry({ kind: 'page', pageIndex: 1 }), run: run(), products, pages }); @@ -78,15 +83,21 @@ describe('a page entry', () => { test('latest, prev and delta are replaced whole, and a failed price leaves no stale value (F-21)', () => { const doc = w['workspaces/u1/products/B0AAAAAAA2']; - expect(doc.fields).toEqual(expect.arrayContaining(['latest', 'prev', 'delta', 'sourceIds'])); - expect(doc.data.latest).toEqual({ r: 4.5, v: 100, pr: 1, rk: 1, at: Date.parse('2026-06-09T14:03:00.000Z'), runId: RUN_ID, dayKey: '2026-06-09' }); - expect(doc.data.prev).toEqual({ p: 2100, r: 4.5, v: 95, at: Date.parse('2026-06-02T14:00:00.000Z') }); + // A merge write, so sourceIds accumulates (NEW-SYNC-1); every key a map lacks is deleted. + expect(doc.merge).toBe(true); + expect(doc.data.latest.p).toBe(DELETE); + expect(doc.data.delta.p).toBe(DELETE); + expect(doc.data.delta.pPct).toBe(DELETE); + expect(doc.data.sourceIds).toEqual(['k_mug']); + const d = kept(doc.data); + expect(d.latest).toEqual({ r: 4.5, v: 100, pr: 1, rk: 1, at: Date.parse('2026-06-09T14:03:00.000Z'), runId: RUN_ID, dayKey: '2026-06-09' }); + expect(d.prev).toEqual({ p: 2100, r: 4.5, v: 95, at: Date.parse('2026-06-02T14:00:00.000Z') }); // No price this run, so no price delta; the rest compares - expect(doc.data.delta).toEqual({ r: 0, v: 5, days: 7 }); + expect(d.delta).toEqual({ r: 0, v: 5, days: 7 }); }); test('a product seen for the first time has a null delta and no prev', () => { - const doc = w['workspaces/u1/products/B0AAAAAAA1'].data; + const doc = kept(w['workspaces/u1/products/B0AAAAAAA1'].data); expect(doc.delta).toBeNull(); expect(doc.prev).toBeNull(); expect(doc.latest.rk).toBe(3); @@ -128,8 +139,7 @@ test('firstSeenAt goes only on documents the caller found missing (F-29b)', () = const writes = planEntry({ entry: entry({ kind: 'page', pageIndex: 1 }), run: run(), products, pages, missing: new Set(['B0AAAAAAA1']) }); const w = byPath(writes); expect(w['workspaces/u1/products/B0AAAAAAA1'].data).toMatchObject({ firstSeenAt: Date.parse('2026-06-09T14:03:00.000Z'), firstRunId: RUN_ID }); - expect(w['workspaces/u1/products/B0AAAAAAA1'].fields).toEqual(expect.arrayContaining(['firstSeenAt', 'firstRunId'])); - expect(w['workspaces/u1/products/B0AAAAAAA2'].fields).not.toContain('firstSeenAt'); + expect(w['workspaces/u1/products/B0AAAAAAA2'].data).not.toHaveProperty('firstSeenAt'); }); test('first-seen candidates: new here, or last seen by nobody signed in to this account', () => { @@ -167,3 +177,31 @@ test('a run from before 2.3 still gets its source id and day key', () => { expect(runKeys(old)).toMatchObject({ sourceId: 's_A3K9XELT4QZ6M2', source: { type: 'storefront', sellerId: 'A3K9XELT4QZ6M2' } }); expect(runKeys(old).dayKey).toBe(S.dayKeyOf(START, new Date(START).getTimezoneOffset())); }); + +test('with header false a page entry leaves out the run header and the source; a run entry keeps them', () => { + const page = planEntry({ entry: entry({ kind: 'page', pageIndex: 1 }), run: run(), products, pages, header: false }); + expect(page.map((x) => x.path.slice(2).join('/'))).not.toContain(`runs/${RUN_ID}`); + expect(page.map((x) => x.path.slice(2).join('/'))).not.toContain('sources/k_mug'); + const end = planEntry({ entry: entry({ kind: 'run' }), run: run({ state: 'done', reason: 'complete', finishedAt: START + 1 }), products, pages, header: false }); + expect(end.map((x) => x.path.slice(2).join('/'))).toEqual([`runs/${RUN_ID}`, 'sources/k_mug']); +}); + +describe('rules caps (EXT9-1)', () => { + test('a keyword or URL longer than the rules allow is cut to the cap', () => { + const long = 'a'.repeat(310); + const r = run({ source: { type: 'keyword', sellerId: null, keyword: long, url: `https://www.amazon.com/s?k=${long}`, sourceId: 'k_mug' } }); + const src = byPath(planEntry({ entry: entry({ kind: 'run' }), run: r, products, pages }))['workspaces/u1/sources/k_mug'].data; + expect(src.keyword.length).toBeLessThanOrEqual(RULE_CAPS.keyword); + expect(src.url.length).toBeLessThanOrEqual(RULE_CAPS.url); + }); + + test('a product the rules would refuse fails planning', () => { + const big = [product('B0AAAAAAA1', { name: 'n'.repeat(RULE_CAPS.name + 1) })]; + expect(() => planEntry({ entry: entry({ kind: 'page', pageIndex: 1 }), run: run(), products: big, pages })).toThrow(/rules cap/); + }); + + test('checkRuleCaps passes documents inside the caps', () => { + expect(() => checkRuleCaps('source', { sourceId: 'k_mug', keyword: 'mug', url: 'u', sellerId: null, lastRunId: RUN_ID })).not.toThrow(); + expect(() => checkRuleCaps('run', { reason: 'r'.repeat(101), maxPages: 5 })).toThrow(/reason/); + }); +}); diff --git a/tests/unit/sync.test.js b/tests/unit/sync.test.js index b48f9b6..852a126 100644 --- a/tests/unit/sync.test.js +++ b/tests/unit/sync.test.js @@ -9,7 +9,7 @@ jest.mock('firebase/firestore', () => ({})); require('fake-indexeddb/auto'); const DB = require('../../scripts/background/db'); -const { createSync, isAuthError } = require('../../scripts/background/sync.js'); +const { createSync, isAuthError, isRefusal } = require('../../scripts/background/sync.js'); const START = Date.parse('2026-06-21T10:00:00.000Z'); const RUN_ID = `k_wireless-mouse_${START}`; @@ -33,17 +33,38 @@ function fakeFirestore({ failCommit = () => false } = {}) { getDoc: async (ref) => ({ exists: () => docs.has(ref.path) }), Timestamp: { fromMillis: (ms) => ({ toMillis: () => ms, ms }) }, arrayUnion: (...vals) => ({ __union: vals }), + deleteField: () => ({ __delete: true }), FieldPath, writeBatch: () => { const ops = []; return { set(ref, data, opts) { ops.push({ ref, data, opts }); }, async commit() { - if (failCommit(ops)) throw Object.assign(new Error('unavailable'), { code: 'unavailable' }); + const fail = failCommit(ops); + if (fail) { + const code = fail === true ? 'unavailable' : fail; + throw Object.assign(new Error(code), { code }); + } commits++; for (const { ref, data, opts } of ops) { if (!opts) { docs.set(ref.path, copy(data)); continue; } const cur = docs.get(ref.path) || {}; + if (opts.merge) { + // Deep merge, as Firestore does: maps key by key, deletes remove. + const merge = (into, from) => { + for (const [k, v] of Object.entries(from)) { + if (v && v.__delete) delete into[k]; + else if (v && v.__union) into[k] = [...new Set([...(into[k] || []), ...resolve(v)])]; + else if (v && typeof v === 'object' && !Array.isArray(v) && !v.toMillis) { + into[k] = into[k] && typeof into[k] === 'object' ? into[k] : {}; + merge(into[k], v); + } else into[k] = copy(v); + } + }; + merge(cur, data); + docs.set(ref.path, cur); + continue; + } for (const f of opts.mergeFields) { const segs = f instanceof FieldPath ? f.segs : [f]; let v = getPath(data, segs); @@ -99,8 +120,8 @@ test('a flush writes every queued page and empties the outbox', async () => { const cloud = fakeFirestore(); const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, log: quiet }); const totals = await sync.flush('u1'); - // page 1: chunk + 2x2 + run + source; page 2: chunk + 1x2 + run + source; end: run + source - expect(totals).toEqual({ entries: 3, pages: 2, runs: 1, products: 3, writes: 7 + 5 + 2 }); + // page 1: chunk + 2x2; page 2: chunk + 2x1; the header and source once, with the end + expect(totals).toEqual({ entries: 3, pages: 2, runs: 1, products: 3, writes: 5 + 3 + 2, failed: 0 }); expect(await store.count('outbox')).toBe(0); const run = cloud.docs.get(`workspaces/u1/runs/${RUN_ID}`); expect(run).toMatchObject({ status: 'complete', pagesDone: 2, pagesPlanned: 2, counters: { uniqueAsins: 3 } }); @@ -216,3 +237,58 @@ test('auth errors are told apart from network and rules errors', () => { expect(isAuthError({ code: 'unavailable' })).toBe(false); expect(isAuthError(null)).toBe(false); }); + +test('a product seen by two sources lists both (NEW-SYNC-1)', async () => { + const store = await storeWith({ products: 1, perPage: 1, state: 'done' }); + const cloud = fakeFirestore(); + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, log: quiet }); + await sync.flush('u1'); + // The same ASIN in a storefront run. + const RUN2 = `s_A3K9XELT4QZ6M2_${START + 5000}`; + await store.write([ + { store: 'runs', put: { runId: RUN2, sourceId: 's_A3K9XELT4QZ6M2', dayKey: '2026-06-21', state: 'done', reason: 'complete', source: { type: 'storefront', sellerId: 'A3K9XELT4QZ6M2', keyword: null, url: 'https://www.amazon.com/s?me=A3K9XELT4QZ6M2' }, startedAt: START + 5000, finishedAt: START + 6000, page: 1, maxPages: 20 } }, + { store: 'products', put: { runId: RUN2, n: 0, pageIndex: 1, asin: 'B000000000', name: 'Mouse 0', priceCents: 999, rating: 4.5, reviewCount: 10, isPrime: true, sponsored: false, organicRank: 1, scrapedAt: new Date(START + 5000).toISOString(), placements: [{ page: 1, position: 1, sponsored: false, rank: 1 }], delta: { isNew: false }, prev: null } }, + { store: 'placements', put: { runId: RUN2, pageIndex: 1, count: 1, placements: 1, kind: 'last', scrapedAt: new Date(START + 5000).toISOString() } }, + { store: 'outbox', put: { runId: RUN2, kind: 'page', pageIndex: 1, uid: 'u1', queuedAt: START } }, + { store: 'outbox', put: { runId: RUN2, kind: 'run', uid: 'u1', queuedAt: START } }, + ]); + await sync.flush('u1'); + expect(cloud.docs.get('workspaces/u1/products/B000000000').sourceIds.sort()).toEqual(['k_wireless-mouse', 's_A3K9XELT4QZ6M2']); + expect(cloud.docs.get('workspaces/u1/products/B000000000').latest.p).toBe(999); +}); + +test('an entry the rules refuse is marked failed and the entries behind it still sync (EXT9-1)', async () => { + const store = await storeWith({ products: 4, perPage: 2, state: 'done' }); + // The rules refuse page 1 of the run, every time. + const cloud = fakeFirestore({ failCommit: (ops) => ops.some((o) => o.ref.path.endsWith('/pages/p0001')) && 'permission-denied' }); + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, log: quiet }); + const totals = await sync.flush('u1'); + expect(totals).toMatchObject({ entries: 2, pages: 1, failed: 1 }); + expect(cloud.docs.has(`workspaces/u1/runs/${RUN_ID}/pages/p0002`)).toBe(true); + expect(cloud.docs.get(`workspaces/u1/runs/${RUN_ID}`)).toMatchObject({ status: 'complete' }); + const left = await store.getAll('outbox'); + expect(left).toHaveLength(1); + expect(left[0]).toMatchObject({ pageIndex: 1, failed: { code: 'permission-denied' } }); + expect(await sync.pending('u1')).toBe(0); + expect(await sync.failed('u1')).toBe(1); + + // An automatic flush leaves it alone; Export tries it again. + expect(await sync.flush('u1')).toMatchObject({ entries: 0, failed: 0 }); + expect(await sync.flush('u1', { retryFailed: true })).toMatchObject({ entries: 0, failed: 1 }); +}); + +test('when every entry is refused nothing is marked and the error goes to the caller', async () => { + const store = await storeWith({ products: 2, perPage: 1 }); + const cloud = fakeFirestore({ failCommit: () => 'permission-denied' }); + const sync = createSync({ db: {}, openStore: async () => store, fs: cloud.fs, log: quiet }); + await expect(sync.flush('u1')).rejects.toMatchObject({ code: 'permission-denied' }); + expect(await sync.pending('u1')).toBe(2); + expect(await sync.failed('u1')).toBe(0); +}); + +test('refusals are told apart from errors worth retrying', () => { + expect(isRefusal({ code: 'permission-denied' })).toBe(true); + expect(isRefusal({ code: 'invalid-argument' })).toBe(true); + expect(isRefusal({ code: 'unavailable' })).toBe(false); + expect(isRefusal({ code: 'resource-exhausted' })).toBe(false); +}); From 0a412d2746f2a368894446c4e7f0d8d6c5993294 Mon Sep 17 00:00:00 2001 From: Enes Yilmaz Date: Thu, 24 Sep 2026 17:39:17 -0400 Subject: [PATCH 18/18] Fold dashboard sign-in away so the popup opens on scraping Sign-in is only for sending scans to the dashboard, so it now sits in a collapsed "Dashboard sync (optional)" row under the export buttons, like the AI chat settings. It unfolds on its own only when a session expires. --- popup/popup.css | 10 ++++ popup/popup.html | 105 +++++++++++++++++++------------------- popup/popup.js | 4 ++ tests/e2e/scrape.spec.mjs | 6 ++- 4 files changed, 72 insertions(+), 53 deletions(-) diff --git a/popup/popup.css b/popup/popup.css index eeb8c1b..2ef101a 100644 --- a/popup/popup.css +++ b/popup/popup.css @@ -490,6 +490,16 @@ h1 { color: var(--color-text-primary); } +#authPanel { + margin-top: var(--spacing-md); +} + +.auth-help { + font-size: var(--font-size-sm); + opacity: 0.7; + margin-bottom: var(--spacing-sm); +} + /* AI chat settings */ .ai-settings summary { cursor: pointer; diff --git a/popup/popup.html b/popup/popup.html index 7db2899..d80473f 100644 --- a/popup/popup.html +++ b/popup/popup.html @@ -45,58 +45,6 @@

ProScan

- - - - -
- - AI chat settings - - -
- - -
-
-
- Get a free key at aistudio.google.com/apikey. - It stays on this computer and is sent only to Google. -
-
-
Ready to start
@@ -152,6 +100,59 @@

ProScan

Export to ProScan + + + + +
+ + AI chat settings + + +
+ + +
+
+
+ Get a free key at aistudio.google.com/apikey. + It stays on this computer and is sent only to Google. +
+
+ diff --git a/popup/popup.js b/popup/popup.js index 0ce44f5..8753165 100644 --- a/popup/popup.js +++ b/popup/popup.js @@ -467,6 +467,7 @@ function sendToWorker(message) { function showSignedIn(user) { currentProScanUser = user; elements.authStatusEmail.textContent = user.email || user.displayName || 'Signed in'; + document.getElementById('authSummary').textContent = 'Syncing to dashboard'; elements.authForm.classList.add('hidden'); elements.authAccount.classList.remove('hidden'); elements.authError.classList.add('hidden'); @@ -510,6 +511,9 @@ function showSignInForm(notice = null) { ? 'Your session expired. Sign in again to keep syncing; your scans are kept.' : ''; elements.authNotice.classList.toggle('hidden', notice !== 'expired'); + document.getElementById('authSummary').textContent = 'Dashboard sync (optional)'; + // An expired session is the one case worth unfolding the panel for. + if (notice === 'expired') document.getElementById('authPanel').open = true; elements.authForm.classList.remove('hidden'); elements.authError.classList.add('hidden'); elements.authError.textContent = ''; diff --git a/tests/e2e/scrape.spec.mjs b/tests/e2e/scrape.spec.mjs index 19eebc1..349c325 100644 --- a/tests/e2e/scrape.spec.mjs +++ b/tests/e2e/scrape.spec.mjs @@ -277,9 +277,13 @@ test('with cloud sync off, the popup shows no sign-in and no Export to ProScan', expect(resp.error).toMatch(/not available/); }); -test('with cloud sync on, the popup offers sign-in, sign-up and a password reset', async ({ ext }) => { +test('with cloud sync on, sign-in is folded away and offers sign-up and a password reset', async ({ ext }) => { test.skip(!Flags.CLOUD_SYNC, 'cloud sync is off in this build'); const popup = await extPage(ext); + await expect(popup.locator('#actionButton')).toBeVisible(); + await expect(popup.locator('#authPanel')).toBeVisible(); + await expect(popup.locator('#authForm')).toBeHidden(); + await popup.click('#authPanel summary'); await expect(popup.locator('#authForm')).toBeVisible(); await expect(popup.locator('#authSignUpLink')).toHaveAttribute('href', /^https:\/\/proscanbot\.web\.app\/dashboard\//); await expect(popup.locator('#authResetBtn')).toBeVisible();