From 92ff73170fbd869402a0449da4bffca17741792b Mon Sep 17 00:00:00 2001 From: Asapteejo Date: Wed, 16 Sep 2026 00:17:32 +0100 Subject: [PATCH 1/4] fix(ops): generate the expected-migrations list and add an E2E route sweep MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Migration 0042_site_content_cms was never applied to production, so pages reading tenant site content threw P2022 "column does not exist". The real bug was the safety net: /api/readyz compared applied migrations against a HAND-MAINTAINED list that stopped at 0035 while the repo had grown to 0049, so readiness reported a false green with 14 migrations missing. - Generate EXPECTED_PRODUCTION_MIGRATIONS from prisma/migrations (scripts/generate-migration-manifest.mjs). A generated list cannot go stale; a hand-maintained one did. - Regenerate the manifest in run-build.mjs before next build, so every deploy ships a current list. The build still needs no database access. - CI guard: `npm run migrations:check` fails when a migration was added without regenerating the manifest. - Degrade gracefully: the site-content editor renders an explanatory banner instead of a 500 when the columns are missing. - Playwright suite: load every public/admin/portal route in a real browser and fail on any 5xx, error boundary, or console error — the class of failure unit tests and typecheck cannot see. health.spec.ts asserts migrations.missing === [] against a deployment. Docs: incident writeup, deploy pipeline (migrations run in a release step, before app code), provisioning spec, UI/Paystack fix notes. Co-Authored-By: Claude Opus 5 --- .github/workflows/ci.yml | 75 +++++++++ .gitignore | 6 + DEPLOY-PIPELINE.md | 62 ++++++++ FIXES-UI-AND-PAYSTACK.md | 63 ++++++++ INCIDENT-2026-09-15-MIGRATION-DRIFT.md | 97 +++++++++++ USER-PROVISIONING-SPEC.md | 115 ++++++++++++++ e2e/authenticated.spec.ts | 150 ++++++++++++++++++ e2e/health.spec.ts | 49 ++++++ e2e/smoke.spec.ts | 107 +++++++++++++ eslint.config.mjs | 3 + package-lock.json | 47 ++++++ package.json | 7 + playwright.config.ts | 53 +++++++ scripts/generate-migration-manifest.mjs | 89 +++++++++++ scripts/run-build.mjs | 7 + .../admin/site-content-management.tsx | 28 ++++ src/lib/ops/health.test.ts | 65 +++++--- src/lib/ops/health.ts | 18 ++- src/lib/ops/migration-manifest.ts | 61 +++++++ 19 files changed, 1074 insertions(+), 28 deletions(-) create mode 100644 DEPLOY-PIPELINE.md create mode 100644 FIXES-UI-AND-PAYSTACK.md create mode 100644 INCIDENT-2026-09-15-MIGRATION-DRIFT.md create mode 100644 USER-PROVISIONING-SPEC.md create mode 100644 e2e/authenticated.spec.ts create mode 100644 e2e/health.spec.ts create mode 100644 e2e/smoke.spec.ts create mode 100644 playwright.config.ts create mode 100644 scripts/generate-migration-manifest.mjs create mode 100644 src/lib/ops/migration-manifest.ts diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 509fa24..3d0c358 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -40,6 +40,12 @@ jobs: - name: Install dependencies run: npm ci + # Fails when a migration was added without regenerating the manifest, + # which would leave /api/readyz blind to the new migration and let a + # database that is behind the code report a false green. + - name: Migration manifest up to date + run: npm run migrations:check + - name: Tests run: npm run test @@ -51,3 +57,72 @@ jobs: - name: Build run: npm run build + + e2e: + name: E2E smoke (browser) + runs-on: ubuntu-latest + timeout-minutes: 30 + # Loads every route in a real browser and fails on 5xx / error boundaries / + # console errors. This is the layer that catches runtime-only faults such + # as a database behind the deployed code (Prisma P2022), which typecheck + # and unit tests cannot see. + services: + postgres: + image: postgres:16 + env: + POSTGRES_USER: postgres + POSTGRES_PASSWORD: postgres + POSTGRES_DB: estateos_e2e + ports: + - 5432:5432 + options: >- + --health-cmd pg_isready + --health-interval 10s + --health-timeout 5s + --health-retries 5 + env: + DATABASE_URL: postgresql://postgres:postgres@localhost:5432/estateos_e2e + DIRECT_URL: postgresql://postgres:postgres@localhost:5432/estateos_e2e + # Lets the authenticated sweep mint sessions without Clerk. Safe here: + # the database is a throwaway container, never production. + ESTATEOS_ENABLE_DEV_BYPASS: "true" + NODE_ENV: development + # Public marketing routes resolve their tenant from the host; on + # localhost there is no tenant subdomain, so fall back to the seeded + # demo company or every public route 404s instead of rendering. + DEFAULT_COMPANY_SLUG: acme-realty + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Setup Node + uses: actions/setup-node@v4 + with: + node-version: 22 + cache: npm + + - name: Install dependencies + run: npm ci + + - name: Install Playwright browser + run: npm run e2e:install + + # The E2E database is migrated from the same migration files production + # uses, so a broken or missing migration fails here instead of in prod. + - name: Apply migrations + run: npm run db:migrate:deploy + + - name: Seed demo tenant + run: npm run db:seed + continue-on-error: true + + - name: Run E2E smoke suite + run: npm run e2e + + - name: Upload Playwright report + if: failure() + uses: actions/upload-artifact@v4 + with: + name: playwright-report + path: playwright-report/ + retention-days: 7 diff --git a/.gitignore b/.gitignore index 502d5f5..ed90b2e 100644 --- a/.gitignore +++ b/.gitignore @@ -44,3 +44,9 @@ next-env.d.ts # generated local audit outputs artifacts/ + +# playwright e2e output (the specs and config ARE committed) +/test-results/ +/playwright-report/ +/blob-report/ +/playwright/.cache/ diff --git a/DEPLOY-PIPELINE.md b/DEPLOY-PIPELINE.md new file mode 100644 index 0000000..142af05 --- /dev/null +++ b/DEPLOY-PIPELINE.md @@ -0,0 +1,62 @@ +# Deploy pipeline — apply migrations + block drift + +Wire these so the `SiteSettings.draftSiteContent` class of error (schema drift) +can never ship again. The build stays DB-free; migrations run in a release step. + +## Option A — GitHub Actions (recommended if you deploy from GitHub) + +Create `.github/workflows/deploy-migrate.yml`: + +```yaml +name: DB migrate + drift gate +on: + push: + branches: [main] # or your production branch + workflow_dispatch: + +jobs: + migrate: + runs-on: ubuntu-latest + environment: production # holds the production DB secrets + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-node@v4 + with: + node-version: 20 + cache: npm + - run: npm ci + # 1) apply any pending migrations to the production DB + - run: npm run db:migrate:deploy + env: + DATABASE_URL: ${{ secrets.PROD_DATABASE_URL }} + DIRECT_URL: ${{ secrets.PROD_DIRECT_URL }} + # 2) fail the job if the DB is still behind (defense in depth) + - run: npm run db:migrate:check + env: + DATABASE_URL: ${{ secrets.PROD_DATABASE_URL }} + DIRECT_URL: ${{ secrets.PROD_DIRECT_URL }} +``` + +Make the Vercel production deploy **depend on this job succeeding** (e.g. trigger +Vercel via a deploy hook only after this workflow passes, or run the app deploy +as a later step in the same workflow). The point: migrations land BEFORE the new +code serves traffic. + +## Option B — Vercel only (no GitHub Actions) + +Vercel builds must not touch the DB, so add a **release step** you run before +promoting a build (locally or from a small CI job) with the prod connection: + +``` +DATABASE_URL="$PROD_DB" DIRECT_URL="$PROD_DIRECT" npm run db:migrate:deploy +DATABASE_URL="$PROD_DB" DIRECT_URL="$PROD_DIRECT" npm run db:migrate:check # exits 1 if behind +``` + +Then promote the Vercel deployment. Do NOT put `migrate deploy` in the Vercel +build command — the build has no guaranteed DB access and that's by design +(`scripts/run-build.mjs`). + +## The one rule +Any PR that adds a file under `prisma/migrations/` must be paired with a +`db:migrate:deploy` in the release for the environment it targets. `db:migrate:check` +is the gate that enforces it. diff --git a/FIXES-UI-AND-PAYSTACK.md b/FIXES-UI-AND-PAYSTACK.md new file mode 100644 index 0000000..7bf9ce6 --- /dev/null +++ b/FIXES-UI-AND-PAYSTACK.md @@ -0,0 +1,63 @@ +# Three fixes — company name clipping, superadmin leakage, Paystack + +## 1. Company name cut off in the dashboard sidebar — FIXED (code) +`src/components/shared/logo.tsx` had `sm:min-w-[11rem]` on the name wrapper. That +hard 176px minimum is wider than the sidebar's inner card, so long names like +"Blueprint Urban Residences LTD" overflowed and were clipped by the parent's +`overflow-hidden`. Removed the min-width, kept `min-w-0` so the flex child can +shrink, and added `break-words` so the full name wraps instead of being cut. + +## 2. Superadmin-only content on the tenant admin dashboard — FIXED (code) +Two separate leaks: + +**a) "Paystack platform readiness" card** (`owner: "Platform"`, links to +`/superadmin/settings`). The readiness checklist mixed tenant-owned and +platform-owned items and rendered the whole list to tenant admins. +- `src/components/shared/tenant-readiness-checklist.tsx`: added an `audience` + prop. `audience="tenant"` filters out any item owned by **Platform** or + **Superadmin**, or whose action link points at `/superadmin/*`. +- `src/app/(admin)/admin/settings/page.tsx` now passes `audience="tenant"`. +- The superadmin company page still sees the full list (default audience). +This also removes the other platform-owned rows (R2 storage, public-site +reachability) from the tenant view — a tenant can't action those either. + +**b) "This account cannot access the platform owner dashboard"** +(`src/app/app/access/page.tsx`, `?status=forbidden`). This page is what an +ordinary tenant admin sees if they land on a `/superadmin` URL. The guard was +correct, but the page was alarming and wrong: its "what this means" bullets are +hardcoded SUSPENSION copy ("Admin dashboards, portal access, and payment actions +are currently blocked") — untrue for this case, nothing is blocked. +- The `forbidden` case now has its own reassuring bullets ("Your company + workspace is unaffected…") and its button goes to **/admin** ("Go to your + dashboard") instead of the marketing homepage. + +NOTE: nothing is being *exposed* here — the superadmin guard works, it just +explained itself badly. If a tenant admin is hitting this page repeatedly, +something is linking them to /superadmin; tell me where you clicked from and +I'll remove that link. + +## 3. Paystack not working — CONFIG, not code +`src/modules/readiness/service.ts` sets `paystackConfigured: featureFlags.hasPaystack`, +and in `src/lib/config.ts`: + +``` +hasPaystack = PAYSTACK_SECRET_KEY && PAYSTACK_PUBLIC_KEY && PAYSTACK_WEBHOOK_SECRET +``` + +All **three** must be present. So "MISSING" = at least one is not set in the +**Vercel production** environment. Two gotchas: +- `normalizeRuntimeServerEnv()` **clears the entire group** if only some are set — + a partial config reads as completely missing. +- `getProductionReadinessIssues()` also flags it if `PAYSTACK_WEBHOOK_SECRET` + looks like a URL: it must be the **signing secret**, not the webhook URL. + +### To fix +In Vercel → Project → Settings → Environment Variables (**Production**), set all three: +- `PAYSTACK_SECRET_KEY` → `sk_live_…` +- `PAYSTACK_PUBLIC_KEY` → `pk_live_…` +- `PAYSTACK_WEBHOOK_SECRET` → the signing secret from the Paystack dashboard + (Settings → API Keys & Webhooks), **not** the webhook URL. + +Then **redeploy** (env changes need a new build). The card flips to complete. +Separately, each tenant still connects their own Paystack **subaccount** under +Admin → Settings ("Payment account" item) so buyer payments settle to them. diff --git a/INCIDENT-2026-09-15-MIGRATION-DRIFT.md b/INCIDENT-2026-09-15-MIGRATION-DRIFT.md new file mode 100644 index 0000000..9bb183e --- /dev/null +++ b/INCIDENT-2026-09-15-MIGRATION-DRIFT.md @@ -0,0 +1,97 @@ +# Incident — P2022 "column does not exist" in production (2026-09-15) + +## What users saw + +Server errors on pages that read tenant site content: + +``` +PrismaClientKnownRequestError: The column `SiteSettings.draftSiteContent` +does not exist in the current database. +code: 'P2022' +``` + +## Root cause (two layers) + +**1. The database was behind the deployed code.** +`draftSiteContent` / `publishedSiteContent` are added by migration +`0042_site_content_cms`. That migration is in the repo but was never applied +to the production database, so the deployed code queried columns that do not +exist. + +**2. The safety net was blind — this is the real bug.** +`/api/readyz` already compares applied migrations against an expected list. +That list (`EXPECTED_PRODUCTION_MIGRATIONS`) was **hand-maintained and stopped +at `0035`**, while the repo had grown to `0049`. Production therefore reported: + +```json +"migrations": { "ok": true, "missing": [] } +``` + +a **false green**, while 14 migrations — including 0042 — were missing. The +monitoring said healthy right up until users hit the broken pages. + +## Immediate fix (operator action) + +```bash +# 1. See exactly what is missing +npx prisma migrate status # with DATABASE_URL pointed at production + +# 2. Apply the pending migrations +npm run db:migrate:deploy + +# 3. Confirm the gap is closed — `missing` must be [] +curl -s https://estateos.tech/api/readyz | jq '.checks.database.migrations' +``` + +Expect several migrations to apply, not just 0042. `0049_webhook_event_dedup_unique` +deletes duplicate historical `WebhookEvent` rows before adding its unique index; +it is additive and safe, but read it before running if the table is large. + +## Prevention shipped with this change + +| Layer | What it does | +|---|---| +| **Generated manifest** | `scripts/generate-migration-manifest.mjs` derives the expected-migrations list from `prisma/migrations/`. A hand-maintained list can go stale; a generated one cannot. | +| **Build step** | `run-build.mjs` regenerates the manifest before `next build`, so every deploy ships a current list. | +| **CI guard** | `npm run migrations:check` fails the build when a migration was added without regenerating the manifest. | +| **Graceful degradation** | `getTenantSiteContentState` catches the read failure. The site-content editor shows an explanatory banner instead of a 500; the public site was already falling back to default copy. | +| **E2E browser sweep** | Playwright loads every public / admin / portal route and fails on any 5xx, error boundary, or console error. | +| **Health gate** | `e2e/health.spec.ts` asserts `migrations.missing === []`. Run it against a deployment right after release. | + +## Running the E2E suite + +```bash +npm install # first time — installs @playwright/test +npm run e2e:install # first time — downloads the browser + +npm run e2e # full sweep against a local dev server +npm run e2e:ui # interactive runner + +# Against a deployed environment (no local server is started): +E2E_BASE_URL=https://estateos.tech npm run e2e:health +``` + +The authenticated sweep signs in through `/api/dev/session`, which only mints +a session when `ESTATEOS_ENABLE_DEV_BYPASS` is on and the database is not +production. Against production those tests **skip themselves** rather than +fail, so `e2e:health` is the production-safe subset. + +## Post-deploy check (add to the release runbook) + +```bash +E2E_BASE_URL=https://estateos.tech npm run e2e:health +``` + +Green means the database matches the deployed code. Red names the exact +missing migrations. + +## Separate findings from this investigation + +- **`R2_PUBLIC_BASE_URL` is not configured in production.** `/api/readyz` + warns about it. Consequence: public assets use the signed-proxy fallback, + and because the `next/image` host allowlist is derived from this variable at + **build time**, property photos are not being optimised in production. Set + it in the build environment and redeploy to activate AVIF/WebP. +- **Sentry is disabled in production** (`sentry: "disabled"`). This incident + was found by a user, not by alerting. Enabling `SENTRY_DSN` would have + surfaced the P2022 on the first occurrence. diff --git a/USER-PROVISIONING-SPEC.md b/USER-PROVISIONING-SPEC.md new file mode 100644 index 0000000..61487c9 --- /dev/null +++ b/USER-PROVISIONING-SPEC.md @@ -0,0 +1,115 @@ +# Build spec for Claude Code — operator-created users (buyers + staff) + +Run locally where you have the shell, `.env.local` with `CLERK_SECRET_KEY`, and can +run `npm run check`. This is security-sensitive (account + credential creation) — read +the whole spec, reuse the existing infra called out, and test before committing. + +## Goal +Let operators create accounts that the person can log in with later: +- **Front desk** → create a **buyer** with a full profile (a walk-in / offline buyer + who didn't purchase through the portal). They can log in with the email provided. +- **CEO (admin)** → add company staff to roles: **Marketer / Finance (accountant) / + Front desk (STAFF) / Legal**. +- For any created account, the operator **chooses the credential delivery**: + 1. **Invite / set-password link** (primary; works with any Clerk config). + 2. **Temporary password shown once on screen** (only when Clerk password auth is + enabled — gate behind a flag, default OFF). + +## Auth reality (do not fight this) +Auth is **Clerk**; the DB has no password column. Assume **password sign-in is OFF** +until confirmed in the Clerk Dashboard, so the **invite/set-password link is the default +and the only path enabled by default.** The temp-password path is implemented but gated +behind a new flag (`featureFlags.hasClerkPassword`, from env `CLERK_PASSWORD_ENABLED`, +default false) so it can be switched on later without code changes. + +## Reuse this existing infrastructure (don't reinvent) +- `src/lib/auth/clerk-user-sync.ts` — `syncAuthenticatedClerkUser` already **links a + placeholder** `clerkUserId` that starts with `manual:` to the real Clerk account on + first sign-in (matched by email). This is THE mechanism for operator-created users: + create the local `User` now with `clerkUserId = "manual:" + randomUUID()`, and Clerk + links it when they sign in. Confirm the email-match path and extend if needed. +- `src/modules/invitations/team-invitations.ts` — the invite create/accept pattern + (`createTeamMemberInvitation` / `acceptTeamMemberInvitation`, `TeamMemberInvitation` + table, TTL, audit). Generalize it to also cover **BUYER** and to link to a + pre-created profile. Prefer reusing your own invitation table over Clerk invitations + so the accept flow + local linkage you already have keeps working. +- `src/modules/admin/user-actions.ts` (`setUserRoleAction`) + `src/modules/admin/users.ts` + (`OPERATOR_ROLES`, `ROLE_LABELS`, `CompanyUserRow`) — reuse for role assignment; the + Marketer path already auto-provisions a `StaffProfile`. +- Buyer profile fields: the `Profile` model + `src/modules/kyc/service.ts` + (`getBuyerProfileRecord`) — reuse for the front-desk buyer form fields. +- `src/lib/audit/service.ts` (`writeAuditLog`), `@clerk/nextjs/server` (`clerkClient`). + +## Backend: `src/modules/provisioning/provision-user.ts` (new) +`provisionCompanyUser(input)`: +- Input: `{ companyId, email, fullName, phone?, role, branchId?, buyerProfile?, delivery, actor }` + where `role ∈ {BUYER, MARKETER, FINANCE, STAFF, LEGAL}`, `delivery ∈ {"invite","password"}`. +- **Authorize the actor**: front-desk (STAFF) may create **BUYER only**; ADMIN/SUPER_ADMIN + may create any of the roles above; never `SUPER_ADMIN`/`ADMIN` through this flow. +- **Validate**: normalize email; reject if a User with that email already exists in the + company; company-scope everything; full name ≥ 2 chars. +- **Create local records** in a transaction: + - `User` with `clerkUserId = "manual:" + randomUUID()`, email, firstName/lastName split + from fullName, phone, companyId, branchId, isActive true. + - Role via `Role`/`UserRole` upsert (reuse the pattern in `setUserRoleAction`). + - For BUYER: create the buyer `Profile` from `buyerProfile` fields. + - For MARKETER: `StaffProfile` (isAssignable) — mirror the existing auto-provision. +- **Delivery**: + - `"invite"` (default/only-enabled): create a `TeamMemberInvitation`-style record + (generalized to BUYER) and email the accept/set-password link (reuse the email + sender + template; add a buyer variant). Return `{ mode: "invite", email }`. + - `"password"` (only if `featureFlags.hasClerkPassword`): call + `clerkClient().users.createUser({ emailAddress:[email], password: , + publicMetadata:{ mustResetPassword:true, companyId } })`, set the local + `User.clerkUserId` to the real Clerk id (replace the placeholder), and return + `{ mode: "password", oneTimePassword: }`. If disabled, reject with a + clear message so the UI only offers "invite". +- **Password generation**: strong (≥14 chars, upper/lower/digit/symbol) that satisfies + Clerk's policy. **Never persist it, never put it in audit payloads or logs.** +- **Audit**: `writeAuditLog` the creation (email, role, delivery mode, actor) — NO password. + +## Force change on first login (password path only) +Set `publicMetadata.mustResetPassword = true` at creation. In the portal/admin post-auth +resolution (or middleware), if the signed-in Clerk user has `mustResetPassword`, redirect +to a change-password page until it's cleared; clear the flag after they change it. + +## UI +- **Front desk** — add an "Add buyer" entry on `/admin/front-desk` (or a new + `/admin/front-desk/new-buyer`). Full buyer form (name, email, phone, address/city/state, + occupation, notes) + a delivery choice (Invite link / Temp password — the latter only + shown when `hasClerkPassword`). Submit → server action → **result modal**. +- **CEO** — add an "Add person" button on the Users tab (`users-management.tsx`). Modal: + name, email, phone, role picker (Marketer / Finance / Front desk / Legal) + delivery + choice → same result modal. +- **Credential result modal** (shared): on `mode:"password"` show the email + one-time + password with a Copy button and a bold "This is shown once. Ask them to change it after + first login." note; on `mode:"invite"` show "Invite sent to ." Both re-fetch the + list so the new user appears. + +## Security guardrails (must all hold) +- Actor-role gating (front-desk → BUYER only) enforced **server-side**, not just UI. +- Company-scoped; email unique per company. +- One-time password shown once, never stored, never logged/audited. +- Invite tokens single-use + TTL (reuse existing). +- Force-change-on-first-login for the password path. +- Rate-limit / audit every creation. + +## Schema / migration +Aim for **no migration**: use the `manual:` placeholder clerkUserId + `publicMetadata` +for the reset flag + audit logs. If you must add a column (e.g. `User.createdByUserId`), +add a migration and follow `DEPLOY-PIPELINE.md` (`db:migrate:deploy` + `db:migrate:check`). + +## Testing (before committing) +1. `npm run check` green. +2. Unit tests for `provisionCompanyUser`: role gating (front-desk can't create staff), + email-uniqueness, password strength, delivery branching. +3. Manual E2E (dev): front-desk creates a buyer → invite path → accept/sign-in → buyer + lands in `/portal` with the profile populated. CEO creates a Marketer → appears in + Users with the role + a StaffProfile → assignable on `/admin/leads`. +4. If password auth is enabled in Clerk: create with temp password → sign in with it → + forced to change on first login. + +## Clerk config note +Temp-password path needs Clerk Dashboard → User & Authentication → **Email+Password ON** +and `CLERK_SECRET_KEY` set. Keep `CLERK_PASSWORD_ENABLED=false` until then; the invite +path needs neither and should ship first. diff --git a/e2e/authenticated.spec.ts b/e2e/authenticated.spec.ts new file mode 100644 index 0000000..7f0fd59 --- /dev/null +++ b/e2e/authenticated.spec.ts @@ -0,0 +1,150 @@ +import { expect, test, type Page } from "@playwright/test"; + +/** + * Authenticated route sweep (operator + buyer surfaces). + * + * Uses the dev-bypass session endpoint so the suite needs no Clerk + * credentials. That endpoint only mints a session when + * ESTATEOS_ENABLE_DEV_BYPASS is on AND the database is not the production + * one, so these tests are inert against production and skip themselves. + * + * This sweep is what would have caught the reported incident directly: + * /admin/settings/site-content threw P2022 because the database was missing + * migration 0042, and no unit test loads that page. + */ + +const ADMIN_ROUTES = [ + "/admin", + "/admin/overview", + "/admin/listings", + "/admin/leads", + "/admin/clients", + "/admin/pipeline", + "/admin/transactions", + "/admin/payments", + "/admin/invoices", + "/admin/contracts", + "/admin/documents", + "/admin/team", + "/admin/users", + "/admin/analytics", + "/admin/messages", + "/admin/announcements", + "/admin/testimonials", + "/admin/audit-logs", + "/admin/settings", + "/admin/settings/site-content", + "/admin/settings/branding", +] as const; + +const PORTAL_ROUTES = [ + "/portal", + "/portal/profile", + "/portal/saved", + "/portal/inspections", + "/portal/reservations", + "/portal/messages", + "/portal/payments", + "/portal/invoices", + "/portal/timeline", + "/portal/contracts", + "/portal/notifications", + "/portal/documents", + "/portal/support", +] as const; + +const IGNORABLE_CONSOLE = [ + /favicon/i, + /googletagmanager|posthog|sentry|clerk-telemetry/i, + /ERR_BLOCKED_BY_CLIENT/i, + /Download the React DevTools/i, +]; + +function watchForErrors(page: Page) { + const consoleErrors: string[] = []; + const pageErrors: string[] = []; + + page.on("console", (message) => { + if (message.type() !== "error") return; + const text = message.text(); + if (IGNORABLE_CONSOLE.some((pattern) => pattern.test(text))) return; + consoleErrors.push(text); + }); + page.on("pageerror", (error) => pageErrors.push(error.message)); + + return { consoleErrors, pageErrors }; +} + +/** + * Establishes a dev session. Returns false when dev bypass is unavailable + * (production, or the flag is off), so the caller can skip rather than fail. + */ +async function signInAs(page: Page, role: "admin" | "buyer"): Promise { + const landing = role === "admin" ? "/admin" : "/portal"; + await page.goto(`/api/dev/session?role=${role}&redirectTo=${encodeURIComponent(landing)}`); + + // Dev bypass off → the endpoint clears cookies and we land unauthenticated. + const url = page.url(); + return url.includes(landing) && !url.includes("/sign-in") && !url.includes("/auth/"); +} + +async function assertRouteHealthy(page: Page, route: string) { + const { consoleErrors, pageErrors } = watchForErrors(page); + const response = await page.goto(route, { waitUntil: "domcontentloaded" }); + + expect( + response?.status() ?? 0, + `${route} returned a server error — this is the signature of a missing migration or a broken query.`, + ).toBeLessThan(500); + + const body = (await page.locator("body").innerText()).toLowerCase(); + expect(body, `${route} rendered a server-side exception`).not.toContain( + "a server-side exception has occurred", + ); + expect(body, `${route} rendered an application error`).not.toContain("application error"); + + expect(pageErrors, `${route} threw in the browser`).toEqual([]); + expect(consoleErrors, `${route} logged console errors`).toEqual([]); +} + +test.describe("operator surfaces", () => { + test.beforeEach(async ({ page }) => { + const signedIn = await signInAs(page, "admin"); + test.skip(!signedIn, "Dev bypass unavailable (production or flag off) — skipping."); + }); + + for (const route of ADMIN_ROUTES) { + test(`admin route renders: ${route}`, async ({ page }) => { + await assertRouteHealthy(page, route); + }); + } + + test("site content editor loads its form, not a failure banner", async ({ page }) => { + await page.goto("/admin/settings/site-content"); + + const body = await page.locator("body").innerText(); + + // When the database is behind the code the editor degrades to a banner + // instead of throwing. That is correct behavior, but in a healthy + // environment it means migrations are pending and must be applied. + expect( + body, + "Site content editor is in its degraded state — the database is missing migrations. Run `npm run db:migrate:deploy`.", + ).not.toContain("Site content unavailable"); + + await expect(page.getByText("Search & SEO")).toBeVisible(); + }); +}); + +test.describe("buyer portal", () => { + test.beforeEach(async ({ page }) => { + const signedIn = await signInAs(page, "buyer"); + test.skip(!signedIn, "Dev bypass unavailable (production or flag off) — skipping."); + }); + + for (const route of PORTAL_ROUTES) { + test(`portal route renders: ${route}`, async ({ page }) => { + await assertRouteHealthy(page, route); + }); + } +}); diff --git a/e2e/health.spec.ts b/e2e/health.spec.ts new file mode 100644 index 0000000..ddab198 --- /dev/null +++ b/e2e/health.spec.ts @@ -0,0 +1,49 @@ +import { expect, test } from "@playwright/test"; + +/** + * Deployment health gates. + * + * The first test here is the one that matters most: it is the check that would + * have caught the production incident where the database was 14 migrations + * behind the deployed code and every page touching the new columns threw + * Prisma P2022 ("column does not exist"). + * + * Run this against a deployment immediately after release: + * E2E_BASE_URL=https://estateos.tech npx playwright test e2e/health.spec.ts + */ + +test("database has every migration the deployed code expects", async ({ request }) => { + const response = await request.get("/api/readyz"); + + expect( + response.status(), + "readyz must return 200; a 503 means the deployment is not serviceable", + ).toBe(200); + + const body = await response.json(); + + // The explicit assertion: no migration in the repo may be missing from the + // live database. `missing` is surfaced so a failure names the exact gap. + expect( + body?.checks?.database?.migrations?.missing ?? [], + "Database is behind the deployed code. Run `npm run db:migrate:deploy` before serving traffic.", + ).toEqual([]); + + expect(body?.checks?.database?.migrations?.ok).toBe(true); + expect(body?.ok).toBe(true); +}); + +test("liveness endpoint responds", async ({ request }) => { + const response = await request.get("/api/health"); + expect(response.status()).toBe(200); +}); + +test("readyz never leaks credentials", async ({ request }) => { + const response = await request.get("/api/readyz"); + const raw = await response.text(); + + // Host names are fine; secrets are not. + for (const secretish of ["password", "secret", "sk_live", "sk_test", "Bearer "]) { + expect(raw.toLowerCase()).not.toContain(secretish.toLowerCase()); + } +}); diff --git a/e2e/smoke.spec.ts b/e2e/smoke.spec.ts new file mode 100644 index 0000000..c637af6 --- /dev/null +++ b/e2e/smoke.spec.ts @@ -0,0 +1,107 @@ +import { expect, test, type Page } from "@playwright/test"; + +/** + * Public route smoke sweep. + * + * Loads every public page in a real browser and fails on anything that a + * typecheck cannot see: a 5xx response, a Next.js error boundary, or a + * server-side exception surfaced in the console. This is the net that catches + * runtime-only faults such as Prisma P2022 ("column does not exist") when the + * database is behind the deployed code. + */ + +const PUBLIC_ROUTES = [ + "/", + "/properties", + "/about", + "/team", + "/agents", + "/contact", + "/faq", + "/blog", + "/testimonials", + "/careers", +] as const; + +/** Text Next.js renders when a server component throws. */ +const ERROR_BOUNDARY_MARKERS = [ + "Application error", + "a server-side exception has occurred", + "This page could not be found", // only unexpected on routes we expect to exist + "Internal Server Error", +]; + +/** + * Console noise that is not a defect: third-party embeds, favicon 404s, and + * analytics blocked by the test browser. + */ +const IGNORABLE_CONSOLE = [ + /favicon/i, + /googletagmanager|posthog|sentry|clerk-telemetry/i, + /ERR_BLOCKED_BY_CLIENT/i, + /Download the React DevTools/i, +]; + +function watchForErrors(page: Page) { + const consoleErrors: string[] = []; + const pageErrors: string[] = []; + + page.on("console", (message) => { + if (message.type() !== "error") return; + const text = message.text(); + if (IGNORABLE_CONSOLE.some((pattern) => pattern.test(text))) return; + consoleErrors.push(text); + }); + + page.on("pageerror", (error) => { + pageErrors.push(error.message); + }); + + return { consoleErrors, pageErrors }; +} + +for (const route of PUBLIC_ROUTES) { + test(`public route renders without server error: ${route}`, async ({ page }) => { + const { consoleErrors, pageErrors } = watchForErrors(page); + + const response = await page.goto(route, { waitUntil: "domcontentloaded" }); + + // A 5xx is the exact signature of the production incident. + expect( + response?.status() ?? 0, + `${route} returned a server error — check the server logs for the underlying exception.`, + ).toBeLessThan(500); + + const body = (await page.locator("body").innerText()).toLowerCase(); + for (const marker of ERROR_BOUNDARY_MARKERS) { + expect(body, `${route} rendered an error boundary ("${marker}")`).not.toContain( + marker.toLowerCase(), + ); + } + + expect(pageErrors, `${route} threw in the browser`).toEqual([]); + expect(consoleErrors, `${route} logged console errors`).toEqual([]); + }); +} + +test("homepage shows the property search and a working listings link", async ({ page }) => { + await page.goto("/"); + + // The hero search is the front door of the site; it must submit to /properties. + const search = page.locator('form[action="/properties"]').first(); + await expect(search).toBeVisible(); + + await search.locator('input[name="location"]').fill("Lekki"); + await Promise.all([page.waitForURL(/\/properties/), search.locator("button[type=submit]").click()]); + + expect(page.url()).toContain("location=Lekki"); +}); + +test("tenant site content renders copy rather than blank sections", async ({ page }) => { + await page.goto("/"); + + // If the CMS read silently failed, the hero would collapse to empty text. + const heading = page.locator("h1").first(); + await expect(heading).toBeVisible(); + expect((await heading.innerText()).trim().length).toBeGreaterThan(10); +}); diff --git a/eslint.config.mjs b/eslint.config.mjs index 05e726d..64f7a4b 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -12,6 +12,9 @@ const eslintConfig = defineConfig([ "out/**", "build/**", "next-env.d.ts", + // Playwright run output (the specs themselves are linted). + "test-results/**", + "playwright-report/**", ]), ]); diff --git a/package-lock.json b/package-lock.json index 7a0cfa0..143c08c 100644 --- a/package-lock.json +++ b/package-lock.json @@ -43,6 +43,7 @@ "zod": "^4.3.6" }, "devDependencies": { + "@playwright/test": "^1.50.0", "@tailwindcss/postcss": "^4", "@types/mapbox-gl": "^3.4.1", "@types/node": "^20", @@ -5284,6 +5285,23 @@ "@opentelemetry/api": "^1.1.0" } }, + "node_modules/@playwright/test": { + "version": "1.63.0", + "resolved": "https://registry.npmjs.org/@playwright/test/-/test-1.63.0.tgz", + "integrity": "sha512-oxMK4vllB9RK5NQ2l1pq1IfOf2AvnEuj/vYGDj0H2nMtmtZpKtCwt/l00GEO6xjGfpBNAvjovvYdCm50dRQkpQ==", + "devOptional": true, + "license": "Apache-2.0", + "peer": true, + "dependencies": { + "playwright": "1.63.0" + }, + "bin": { + "playwright": "cli.js" + }, + "engines": { + "node": ">=20" + } + }, "node_modules/@prisma/client": { "version": "6.19.2", "resolved": "https://registry.npmjs.org/@prisma/client/-/client-6.19.2.tgz", @@ -13244,6 +13262,35 @@ "pathe": "^2.0.3" } }, + "node_modules/playwright": { + "version": "1.63.0", + "resolved": "https://registry.npmjs.org/playwright/-/playwright-1.63.0.tgz", + "integrity": "sha512-+7ziBLidS4NaNCdt57SUDT+wYmmd5fmiQejUic/kb+YsYSCPyOOE9sebzMjNmQrsnNpDJqd4WHvV/8lfKfUDUg==", + "devOptional": true, + "license": "Apache-2.0", + "dependencies": { + "playwright-core": "1.63.0" + }, + "bin": { + "playwright": "cli.js" + }, + "engines": { + "node": ">=20" + } + }, + "node_modules/playwright-core": { + "version": "1.63.0", + "resolved": "https://registry.npmjs.org/playwright-core/-/playwright-core-1.63.0.tgz", + "integrity": "sha512-rYCsBF/M5HjUch52bbtVONEFjv6Xu8sm8h72dNlR5bzIE1fvC/bxgspzkjSfU+MweEMmPM8KJebG6nnyxo5mCg==", + "devOptional": true, + "license": "Apache-2.0", + "bin": { + "playwright-core": "cli.js" + }, + "engines": { + "node": ">=20" + } + }, "node_modules/pngjs": { "version": "5.0.0", "resolved": "https://registry.npmjs.org/pngjs/-/pngjs-5.0.0.tgz", diff --git a/package.json b/package.json index a5c4969..490174d 100644 --- a/package.json +++ b/package.json @@ -10,6 +10,12 @@ "start": "next start", "lint": "node --max-old-space-size=4096 ./node_modules/eslint/bin/eslint.js .", "encoding:check": "node scripts/check-utf8.mjs", + "migrations:manifest": "node scripts/generate-migration-manifest.mjs", + "migrations:check": "node scripts/generate-migration-manifest.mjs --check", + "e2e": "playwright test", + "e2e:ui": "playwright test --ui", + "e2e:install": "playwright install --with-deps chromium", + "e2e:health": "playwright test e2e/health.spec.ts", "test": "tsx --test \"src/**/*.test.ts\"", "typecheck": "node --max-old-space-size=4096 ./node_modules/typescript/bin/tsc --noEmit --incremental false", "db:generate": "prisma generate", @@ -65,6 +71,7 @@ "zod": "^4.3.6" }, "devDependencies": { + "@playwright/test": "^1.50.0", "@tailwindcss/postcss": "^4", "@types/mapbox-gl": "^3.4.1", "@types/node": "^20", diff --git a/playwright.config.ts b/playwright.config.ts new file mode 100644 index 0000000..8c3d736 --- /dev/null +++ b/playwright.config.ts @@ -0,0 +1,53 @@ +import { defineConfig, devices } from "@playwright/test"; + +/** + * End-to-end smoke configuration. + * + * Purpose: catch the class of failure that reached production as a P2022 + * "column does not exist" 500 — a page that renders fine in the developer's + * head but throws when actually loaded. Unit tests and typecheck cannot see + * this; only loading every route in a real browser can. + * + * Runs against a local dev server by default. Point at a deployed environment + * with E2E_BASE_URL (then the local server is not started): + * E2E_BASE_URL=https://staging.example.com npm run e2e + */ +const baseURL = process.env.E2E_BASE_URL ?? "http://127.0.0.1:3000"; +const usingExternalTarget = Boolean(process.env.E2E_BASE_URL); + +export default defineConfig({ + testDir: "./e2e", + // Route sweeps are I/O bound; a little parallelism keeps the suite quick + // without overwhelming a single dev server. + workers: process.env.CI ? 2 : 4, + fullyParallel: true, + forbidOnly: Boolean(process.env.CI), + retries: process.env.CI ? 1 : 0, + reporter: process.env.CI ? [["github"], ["list"]] : [["list"]], + timeout: 60_000, + expect: { timeout: 15_000 }, + + use: { + baseURL, + trace: "retain-on-failure", + screenshot: "only-on-failure", + video: "off", + }, + + projects: [ + { name: "chromium", use: { ...devices["Desktop Chrome"] } }, + // Buyers are mobile-first; the portal must survive a phone viewport too. + { name: "mobile", use: { ...devices["Pixel 7"] }, testMatch: /smoke\.spec\.ts/ }, + ], + + webServer: usingExternalTarget + ? undefined + : { + command: "npm run dev", + url: baseURL, + reuseExistingServer: !process.env.CI, + timeout: 180_000, + stdout: "pipe", + stderr: "pipe", + }, +}); diff --git a/scripts/generate-migration-manifest.mjs b/scripts/generate-migration-manifest.mjs new file mode 100644 index 0000000..ae99b2b --- /dev/null +++ b/scripts/generate-migration-manifest.mjs @@ -0,0 +1,89 @@ +#!/usr/bin/env node +/** + * Generates src/lib/ops/migration-manifest.ts from the folders in + * prisma/migrations. + * + * WHY THIS EXISTS + * --------------- + * The readiness endpoint (/api/readyz) compares the migrations applied in the + * live database against a list of expected migrations. That list used to be + * hand-maintained and silently went stale: it stopped at 0035 while the repo + * grew to 0049, so readyz reported `migrations: ok` while production was + * missing 14 migrations — including 0042, whose absent columns produced a + * P2022 "column does not exist" 500 in production. + * + * A generated manifest cannot go stale. `--check` fails CI when someone adds + * a migration without regenerating, so the drift detector always knows about + * every migration in the repo. + * + * Usage: + * node scripts/generate-migration-manifest.mjs # write the file + * node scripts/generate-migration-manifest.mjs --check # verify, exit 1 if stale + */ + +import fs from "node:fs"; +import path from "node:path"; + +const ROOT = process.cwd(); +const MIGRATIONS_DIR = path.join(ROOT, "prisma", "migrations"); +const OUTPUT_FILE = path.join(ROOT, "src", "lib", "ops", "migration-manifest.ts"); + +function readMigrationNames() { + if (!fs.existsSync(MIGRATIONS_DIR)) { + return []; + } + + return fs + .readdirSync(MIGRATIONS_DIR, { withFileTypes: true }) + .filter((entry) => entry.isDirectory()) + // A directory is a real migration only when it holds a migration.sql. + .filter((entry) => + fs.existsSync(path.join(MIGRATIONS_DIR, entry.name, "migration.sql")), + ) + .map((entry) => entry.name) + .sort((a, b) => a.localeCompare(b)); +} + +function renderManifest(names) { + const entries = names.map((name) => ` ${JSON.stringify(name)},`).join("\n"); + + return `// GENERATED FILE — DO NOT EDIT BY HAND. +// Regenerate with: node scripts/generate-migration-manifest.mjs +// +// Every migration directory in prisma/migrations, used by the readiness +// endpoint to detect a database that is behind the deployed code. Generated +// so the list can never silently go stale (see the script header for the +// production incident that motivated this). + +export const EXPECTED_PRODUCTION_MIGRATIONS = [ +${entries} +] as const; +`; +} + +const names = readMigrationNames(); +const contents = renderManifest(names); +const isCheck = process.argv.includes("--check"); + +if (isCheck) { + const existing = fs.existsSync(OUTPUT_FILE) + ? fs.readFileSync(OUTPUT_FILE, "utf8") + : ""; + + // Normalize line endings so the check passes on both Windows and Linux/CI. + if (existing.replace(/\r\n/g, "\n") !== contents.replace(/\r\n/g, "\n")) { + console.error( + "Migration manifest is out of date.\n" + + "Run: node scripts/generate-migration-manifest.mjs\n" + + `Expected ${names.length} migration(s) from prisma/migrations.`, + ); + process.exit(1); + } + + console.log(`Migration manifest is up to date (${names.length} migrations).`); + process.exit(0); +} + +fs.mkdirSync(path.dirname(OUTPUT_FILE), { recursive: true }); +fs.writeFileSync(OUTPUT_FILE, contents); +console.log(`Wrote ${OUTPUT_FILE} (${names.length} migrations).`); diff --git a/scripts/run-build.mjs b/scripts/run-build.mjs index 74207dd..06b7ca0 100644 --- a/scripts/run-build.mjs +++ b/scripts/run-build.mjs @@ -10,12 +10,19 @@ const env = { // Builds must not require direct database access. Run `npm run db:migrate:deploy` // separately in a controlled CI/release step before deploying application code. +// +// The migration manifest is regenerated first so the readiness endpoint always +// ships knowing about every migration in the repo. Without this the expected +// list drifts and /api/readyz reports a false green while the database is +// behind the code. const steps = [ + ["migrationManifest", []], ["prisma", ["generate"]], ["next", ["build"]], ]; const cliEntrypoints = { + migrationManifest: path.join(process.cwd(), "scripts", "generate-migration-manifest.mjs"), prisma: path.join(process.cwd(), "node_modules", "prisma", "build", "index.js"), next: path.join(process.cwd(), "node_modules", "next", "dist", "bin", "next"), }; diff --git a/src/components/admin/site-content-management.tsx b/src/components/admin/site-content-management.tsx index 125796f..f812750 100644 --- a/src/components/admin/site-content-management.tsx +++ b/src/components/admin/site-content-management.tsx @@ -292,6 +292,34 @@ export function SiteContentManagement({ router.refresh(); } + // The content columns could not be read — almost always a database that is + // behind the deployed code. Explain it instead of showing a broken editor. + if (state.unavailable) { + return ( + +
+
+ Site content unavailable +
+

+ Your site editor is temporarily unavailable. +

+

+ Your public website is still online and is showing its default copy — visitors are + unaffected. The editor needs a pending database update before your saved content can + be loaded. +

+

+ If you administer this deployment, run the pending database migrations + (npm run db:migrate:deploy) + and reload. /api/readyz{" "} + lists exactly which migrations are missing. +

+
+
+ ); + } + return (
diff --git a/src/lib/ops/health.test.ts b/src/lib/ops/health.test.ts index 783068e..332f4dd 100644 --- a/src/lib/ops/health.test.ts +++ b/src/lib/ops/health.test.ts @@ -7,6 +7,7 @@ import { buildHealthSnapshot, buildRuntimeReadinessSummary, getMissingExpectedMigrations, + EXPECTED_PRODUCTION_MIGRATIONS, } from "@/lib/ops/health"; test("health snapshot returns safe operational metadata", () => { @@ -47,26 +48,50 @@ test("database readiness metadata stays sanitized", () => { assert.equal(serialized.includes("@"), false); }); -test("migration readiness reports missing production contract migrations", () => { - assert.deepEqual( - getMissingExpectedMigrations([ - "0030_communication_wallet_ledger", - "0031_communication_topups", - "0032_buyer_portal_kyc_review_metadata", - "0033_buyer_testimonial_moderation", - "0034_contract_generation_mvp", - ]), - ["0035_contract_template_version_locking"], +test("migration readiness reports every migration missing from the database", () => { + // A database with nothing applied is missing the whole manifest. + assert.equal( + getMissingExpectedMigrations([]).length, + EXPECTED_PRODUCTION_MIGRATIONS.length, ); - assert.deepEqual( - getMissingExpectedMigrations([ - "0030_communication_wallet_ledger", - "0031_communication_topups", - "0032_buyer_portal_kyc_review_metadata", - "0033_buyer_testimonial_moderation", - "0034_contract_generation_mvp", - "0035_contract_template_version_locking", - ]), - [], + + // A database with everything applied is clean. + assert.deepEqual(getMissingExpectedMigrations([...EXPECTED_PRODUCTION_MIGRATIONS]), []); + + // Exactly one gap is reported as exactly one missing migration. + const allButLast = EXPECTED_PRODUCTION_MIGRATIONS.slice(0, -1); + assert.deepEqual(getMissingExpectedMigrations([...allButLast]), [ + EXPECTED_PRODUCTION_MIGRATIONS[EXPECTED_PRODUCTION_MIGRATIONS.length - 1], + ]); +}); + +/** + * Regression test for the production incident: the expected-migrations list + * was hand-maintained, stopped at 0035, and reported a false green while the + * database was missing 0042 — whose absent columns produced P2022 + * "column does not exist" 500s. + */ +test("migration manifest covers the CMS migration that caused the P2022 incident", () => { + assert.ok( + EXPECTED_PRODUCTION_MIGRATIONS.includes("0042_site_content_cms"), + "0042_site_content_cms must be in the manifest or readyz cannot detect the drift that broke production.", + ); + + // A database stuck at 0035 must now be reported as drifted, not healthy. + const stuckAt0035 = EXPECTED_PRODUCTION_MIGRATIONS.slice( + 0, + EXPECTED_PRODUCTION_MIGRATIONS.indexOf("0035_contract_template_version_locking") + 1, ); + const missing = getMissingExpectedMigrations([...stuckAt0035]); + + assert.ok(missing.length > 0, "A database behind the code must never report zero missing migrations."); + assert.ok(missing.includes("0042_site_content_cms")); +}); + +test("migration manifest is generated, sorted, and free of duplicates", () => { + const names = [...EXPECTED_PRODUCTION_MIGRATIONS]; + + assert.ok(names.length > 40, "Manifest looks truncated; regenerate it."); + assert.equal(new Set(names).size, names.length, "Duplicate migration names in the manifest."); + assert.deepEqual(names, [...names].sort((a, b) => a.localeCompare(b))); }); diff --git a/src/lib/ops/health.ts b/src/lib/ops/health.ts index be1e00b..0996bb3 100644 --- a/src/lib/ops/health.ts +++ b/src/lib/ops/health.ts @@ -9,16 +9,18 @@ import { buildSafeErrorLogContext, logError } from "@/lib/ops/logger"; import { parseSuperadminEmails } from "@/lib/auth/superadmin"; import { getProductionDatabaseSafetyStatus } from "@/lib/db/production-db-guard"; import { resolveRealtimeRuntimeStatus } from "@/lib/realtime/config"; +import { EXPECTED_PRODUCTION_MIGRATIONS } from "@/lib/ops/migration-manifest"; -export const EXPECTED_PRODUCTION_MIGRATIONS = [ - "0030_communication_wallet_ledger", - "0031_communication_topups", - "0032_buyer_portal_kyc_review_metadata", - "0033_buyer_testimonial_moderation", - "0034_contract_generation_mvp", - "0035_contract_template_version_locking", -] as const; +export { EXPECTED_PRODUCTION_MIGRATIONS }; +/** + * Migrations present in the repo but not yet applied to the connected + * database — i.e. the deployed code expects columns the database does not + * have. The expected list is GENERATED from prisma/migrations + * (scripts/generate-migration-manifest.mjs) rather than hand-maintained, + * because a hand-maintained list silently went stale at 0035 and let a + * 14-migration gap reach production as P2022 "column does not exist" 500s. + */ export function getMissingExpectedMigrations(appliedMigrations: string[]) { const applied = new Set(appliedMigrations); return EXPECTED_PRODUCTION_MIGRATIONS.filter((migration) => !applied.has(migration)); diff --git a/src/lib/ops/migration-manifest.ts b/src/lib/ops/migration-manifest.ts new file mode 100644 index 0000000..663027b --- /dev/null +++ b/src/lib/ops/migration-manifest.ts @@ -0,0 +1,61 @@ +// GENERATED FILE — DO NOT EDIT BY HAND. +// Regenerate with: node scripts/generate-migration-manifest.mjs +// +// Every migration directory in prisma/migrations, used by the readiness +// endpoint to detect a database that is behind the deployed code. Generated +// so the list can never silently go stale (see the script header for the +// production incident that motivated this). + +export const EXPECTED_PRODUCTION_MIGRATIONS = [ + "0001_init", + "0002_billing_monetization", + "0003_team_marketers_payment_progress", + "0004_crm_inspections_notifications", + "0005_staff_directory_id_cards", + "0006_wishlist_crm_intelligence", + "0007_property_verification_trust_layer", + "0008_operating_system_dashboard_and_automation", + "0009_payment_requests_and_verification_runtime", + "0010_tenant_branding_workflow", + "0011_marketer_relations_and_snapshots", + "0012_activation_and_collections_follow_up", + "0013_collections_priority_and_quick_actions", + "0014_company_lifecycle_realtime_analytics", + "0015_development_feasibility_calculator", + "0016_development_calculation_phasing", + "0017_development_calculation_versioning", + "0018_deal_risk_scoring", + "0019_support_requests_linear", + "0020_support_request_retry_metadata", + "0021_support_sync_backoff_and_revenue_snapshot", + "0022_incident_aggregation_and_escalation", + "0023_incident_windows_cooldown_and_resolution", + "0024_incident_pruning_caching_and_linear_updates", + "0025_buyer_profile_image", + "0026_development_decision_status", + "0027_property_land_options_countdown", + "0028_property_global_land_units", + "0029_whatsapp_usage_logs", + "0030_communication_wallet_ledger", + "0031_communication_topups", + "0032_buyer_portal_kyc_review_metadata", + "0033_buyer_testimonial_moderation", + "0034_contract_generation_mvp", + "0035_contract_template_version_locking", + "0035a_team_member_invitation_base", + "0036_team_invitation_branch", + "0037_property_location_mapbox", + "0038_property_location_boundary", + "0039_team_invitation_invited_by_user_id", + "0040_drop_legacy_team_invitation_invited_by_id", + "0041_sync_schema_drift", + "0042_site_content_cms", + "0043_team_member_staff_code_unique", + "0044_front_desk_visitor_call_log", + "0045_property_invoices", + "0046_in_app_messaging", + "0047_announcements", + "0048_marketer_role", + "0049_webhook_event_dedup_unique", + "20260701000111_applies_migration_0044_regenerates_the_prisma_client", +] as const; From 6007007981eece8868039516535d54d9e42ad600 Mon Sep 17 00:00:00 2001 From: Asapteejo Date: Wed, 16 Sep 2026 00:17:51 +0100 Subject: [PATCH 2/4] fix(admin): Add person always submitted Front Desk whatever role was picked MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Role is controlled, and its onChange deferred setSelectedRole + * into a setTimeout that read e.target.value lazily. React restores a + * controlled select to its current state right after the event, so by the time + * the timeout ran the DOM value was back to STAFF and that is what got saved. + * + * The server action is intercepted rather than executed so the test proves the + * submitted payload without creating a real Clerk account or sending an invite. + */ + +test("Add person submits the role that was selected", async ({ page }) => { + await page.goto(`/api/dev/session?role=admin&redirectTo=${encodeURIComponent("/admin/users")}`); + const url = page.url(); + test.skip( + !url.includes("/admin") || url.includes("/sign-in") || url.includes("/auth/"), + "Dev bypass unavailable (production or flag off) — skipping.", + ); + + await page.goto("/admin/users"); + await page.getByRole("button", { name: "Add person" }).click(); + + const roleSelect = page.locator("#person-role"); + await expect(roleSelect).toHaveValue("STAFF"); + await roleSelect.selectOption("FINANCE"); + + // Give a deferred (buggy) update every chance to run and revert before asserting. + await page.waitForTimeout(250); + await expect(roleSelect).toHaveValue("FINANCE"); + + await page.locator("#person-firstName").fill("E2E"); + await page.locator("#person-lastName").fill("Finance"); + await page.locator("#person-email").fill(`e2e-finance-${Date.now()}@example.test`); + + let submittedBody: string | null = null; + await page.route("**/admin/users**", async (route) => { + const request = route.request(); + if (request.method() === "POST" && request.headers()["next-action"]) { + submittedBody = request.postData(); + await route.abort(); + return; + } + await route.continue(); + }); + + const actionRequest = page.waitForRequest( + (request) => request.method() === "POST" && Boolean(request.headers()["next-action"]), + ); + await page.getByRole("button", { name: "Create account" }).click(); + await actionRequest; + + expect(submittedBody, "server action request was not captured").not.toBeNull(); + // Next.js encodes useActionState form fields as multipart, optionally prefixed (e.g. "1_role"). + const role = /name="(?:\d+_)?role"\r\n\r\n([A-Z_]+)/.exec(submittedBody ?? "")?.[1]; + expect(role).toBe("FINANCE"); +}); diff --git a/src/components/admin/add-buyer-form.tsx b/src/components/admin/add-buyer-form.tsx index 7630995..6582ad4 100644 --- a/src/components/admin/add-buyer-form.tsx +++ b/src/components/admin/add-buyer-form.tsx @@ -113,10 +113,7 @@ export function AddBuyerButton({ hasClerkPassword }: { hasClerkPassword: boolean diff --git a/src/components/admin/users-management.tsx b/src/components/admin/users-management.tsx index b57a777..1c5e01d 100644 --- a/src/components/admin/users-management.tsx +++ b/src/components/admin/users-management.tsx @@ -109,10 +109,7 @@ function AddPersonDialog({ name="role" required value={selectedRole} - onChange={(e) => { - const id = setTimeout(() => setSelectedRole(e.target.value as AppRole), 0); - return () => clearTimeout(id); - }} + onChange={(e) => setSelectedRole(e.target.value as AppRole)} className={`${inputCls} mt-1`} > {PROVISIONABLE_ROLES.map((r) => ( From 5eb22ebb84a9cc5737b3ad6d70ba0de2985a1265 Mon Sep 17 00:00:00 2001 From: Asapteejo Date: Wed, 16 Sep 2026 23:23:40 +0100 Subject: [PATCH 3/4] fix(e2e): point the suite at a tenant host and stop failing on dev-server noise MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The suite could never pass in CI. Three harness faults, no app changes: - Public pages resolve their tenant from the HOST, and marketing routes deliberately ignore the DEFAULT_COMPANY_SLUG fallback — only a tenant host counts. Against 127.0.0.1 every tenant page rendered the platform site or 404'd, so the sweep failed on nearly every public route. Drive the seeded demo tenant through .localhost, which is the supported dev tenant host (resolveTenantSubdomainFromHost). - Windows, and some Linux setups, do not resolve *.localhost, so map it to loopback in the browser itself rather than trusting the OS resolver. The webServer readiness probe keeps using 127.0.0.1, which Node can resolve. - `next dev` logs a hot-reload WebSocket console error in headless CI, and the sweep counts any console error as a defect. It is dev-server noise, not an application fault, so it joins the ignore list. Also raise the per-test timeout to 120s: a route's first request compiles it (30s+ for admin routes on a cold machine), and the old 60s budget made the first visit time out and abort the rest of the file's navigations. Co-Authored-By: Claude Opus 5 --- e2e/authenticated.spec.ts | 3 +++ e2e/smoke.spec.ts | 3 +++ playwright.config.ts | 27 ++++++++++++++++++++++++--- 3 files changed, 30 insertions(+), 3 deletions(-) diff --git a/e2e/authenticated.spec.ts b/e2e/authenticated.spec.ts index 7f0fd59..a9e6ff6 100644 --- a/e2e/authenticated.spec.ts +++ b/e2e/authenticated.spec.ts @@ -58,6 +58,9 @@ const IGNORABLE_CONSOLE = [ /googletagmanager|posthog|sentry|clerk-telemetry/i, /ERR_BLOCKED_BY_CLIENT/i, /Download the React DevTools/i, + // Dev-server noise: the suite runs against `next dev`, whose hot-reload + // socket logs a console error in headless CI. Not an application fault. + /webpack-hmr|hot-update|__nextjs|turbopack-hmr/i, ]; function watchForErrors(page: Page) { diff --git a/e2e/smoke.spec.ts b/e2e/smoke.spec.ts index c637af6..d7c91c5 100644 --- a/e2e/smoke.spec.ts +++ b/e2e/smoke.spec.ts @@ -40,6 +40,9 @@ const IGNORABLE_CONSOLE = [ /googletagmanager|posthog|sentry|clerk-telemetry/i, /ERR_BLOCKED_BY_CLIENT/i, /Download the React DevTools/i, + // Dev-server noise: the suite runs against `next dev`, whose hot-reload + // socket logs a console error in headless CI. Not an application fault. + /webpack-hmr|hot-update|__nextjs|turbopack-hmr/i, ]; function watchForErrors(page: Page) { diff --git a/playwright.config.ts b/playwright.config.ts index 8c3d736..79354ab 100644 --- a/playwright.config.ts +++ b/playwright.config.ts @@ -12,7 +12,16 @@ import { defineConfig, devices } from "@playwright/test"; * with E2E_BASE_URL (then the local server is not started): * E2E_BASE_URL=https://staging.example.com npm run e2e */ -const baseURL = process.env.E2E_BASE_URL ?? "http://127.0.0.1:3000"; +/** + * Public pages resolve their tenant from the HOST, and for marketing routes a + * DEFAULT_COMPANY_SLUG fallback is deliberately NOT applied — only a tenant + * host counts. On 127.0.0.1 every tenant page therefore renders the platform + * site or 404s. `.localhost` is the supported dev tenant host (see + * resolveTenantSubdomainFromHost), and Chromium resolves *.localhost to + * loopback itself, so the suite drives the seeded demo tenant by default. + */ +const localTenantHost = `http://${process.env.E2E_TENANT_SLUG ?? "acme-realty"}.localhost:3000`; +const baseURL = process.env.E2E_BASE_URL ?? localTenantHost; const usingExternalTarget = Boolean(process.env.E2E_BASE_URL); export default defineConfig({ @@ -24,7 +33,11 @@ export default defineConfig({ forbidOnly: Boolean(process.env.CI), retries: process.env.CI ? 1 : 0, reporter: process.env.CI ? [["github"], ["list"]] : [["list"]], - timeout: 60_000, + // The suite runs against `next dev`, which compiles each route on its first + // request — 30s+ per admin route on a cold, slow machine. 60s was tight + // enough that the first visit to a route timed out and took the rest of the + // file's navigations down with it (net::ERR_ABORTED on teardown). + timeout: 120_000, expect: { timeout: 15_000 }, use: { @@ -32,6 +45,12 @@ export default defineConfig({ trace: "retain-on-failure", screenshot: "only-on-failure", video: "off", + // Windows (and some Linux setups) do not resolve *.localhost, so the + // tenant host is mapped to loopback in the browser itself rather than + // relying on the OS resolver. Harmless against an external target. + launchOptions: usingExternalTarget + ? undefined + : { args: ["--host-resolver-rules=MAP *.localhost 127.0.0.1"] }, }, projects: [ @@ -44,7 +63,9 @@ export default defineConfig({ ? undefined : { command: "npm run dev", - url: baseURL, + // Node (unlike Chromium) does not resolve *.localhost on every + // platform, so the readiness probe uses loopback directly. + url: "http://127.0.0.1:3000", reuseExistingServer: !process.env.CI, timeout: 180_000, stdout: "pipe", From fc6eac4d2667e5c7fe9e056bc55ed2c1a5866b6f Mon Sep 17 00:00:00 2001 From: Asapteejo Date: Thu, 17 Sep 2026 00:27:14 +0100 Subject: [PATCH 4/4] fix(ci): run the E2E sweep against a production build instead of next dev MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The suite ran against `next dev`, which compiles each route on its first request. On a 2-core runner that is 30s+ per admin route, so tests timed out one after another and the job hit its 30-minute limit — 49 failures, none of them real. Build once, then serve that build with NODE_ENV=test. The test value keeps the dev-session bypass available (allowDevBypass in src/lib/config.ts treats "test" as non-production), so the authenticated sweep still runs rather than skipping itself on a production build. Locally the default stays `npm run dev` so the suite still picks up edits; CI overrides it with E2E_SERVER_COMMAND. Measured on the same machine, same suite: next dev 49 failed, 14 passed, 33 min production build 1 failed, 61 passed, 3.8 min (the one failure is this machine's database being behind — exactly what health.spec.ts exists to report; CI migrates a throwaway database first.) Also: accept either submit transport in the Add person regression test. A hydrated page sends a server action (Next-Action header, multipart); before hydration the same form posts natively (url-encoded). The test asserts which role was submitted, so it should not care which transport carried it — it only failed on a production build because hydration timing differs. The *.localhost resolver mapping now applies to external targets too; it only affects that suffix, so it is inert against a deployed environment. Co-Authored-By: Claude Opus 5 --- .github/workflows/ci.yml | 20 ++++++++++++++++---- e2e/add-person-role.spec.ts | 22 ++++++++++++++-------- playwright.config.ts | 14 +++++++++----- 3 files changed, 39 insertions(+), 17 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3d0c358..07cfed4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -86,10 +86,9 @@ jobs: # Lets the authenticated sweep mint sessions without Clerk. Safe here: # the database is a throwaway container, never production. ESTATEOS_ENABLE_DEV_BYPASS: "true" - NODE_ENV: development - # Public marketing routes resolve their tenant from the host; on - # localhost there is no tenant subdomain, so fall back to the seeded - # demo company or every public route 404s instead of rendering. + # NODE_ENV is deliberately NOT set here: the build below must run as a + # normal production build, and only the test run overrides it (see + # "Run E2E smoke suite"). DEFAULT_COMPANY_SLUG: acme-realty steps: - name: Checkout @@ -116,8 +115,21 @@ jobs: run: npm run db:seed continue-on-error: true + # Build once, then serve that build. Running the suite against `next dev` + # made every route compile on its first request (30s+ per admin route on + # a runner), which exhausted the job's time budget before the sweep + # finished. A production build serves each route in well under a second. + - name: Build + run: npm run build + + # NODE_ENV=test keeps the dev-session bypass available on a production + # build (see allowDevBypass in src/lib/config.ts), so the authenticated + # sweep still runs instead of skipping itself. - name: Run E2E smoke suite run: npm run e2e + env: + NODE_ENV: test + E2E_SERVER_COMMAND: npm run start - name: Upload Playwright report if: failure() diff --git a/e2e/add-person-role.spec.ts b/e2e/add-person-role.spec.ts index 3cea1d8..8b64ef6 100644 --- a/e2e/add-person-role.spec.ts +++ b/e2e/add-person-role.spec.ts @@ -35,10 +35,14 @@ test("Add person submits the role that was selected", async ({ page }) => { await page.locator("#person-lastName").fill("Finance"); await page.locator("#person-email").fill(`e2e-finance-${Date.now()}@example.test`); + // Any POST to this route carries the form. A hydrated page sends a server + // action (Next-Action header, multipart body); before hydration the same + // form posts natively (url-encoded), so both shapes are accepted — the point + // is what `role` the form submitted, not which transport carried it. let submittedBody: string | null = null; await page.route("**/admin/users**", async (route) => { const request = route.request(); - if (request.method() === "POST" && request.headers()["next-action"]) { + if (request.method() === "POST") { submittedBody = request.postData(); await route.abort(); return; @@ -46,14 +50,16 @@ test("Add person submits the role that was selected", async ({ page }) => { await route.continue(); }); - const actionRequest = page.waitForRequest( - (request) => request.method() === "POST" && Boolean(request.headers()["next-action"]), - ); + const actionRequest = page.waitForRequest((request) => request.method() === "POST"); await page.getByRole("button", { name: "Create account" }).click(); await actionRequest; - expect(submittedBody, "server action request was not captured").not.toBeNull(); - // Next.js encodes useActionState form fields as multipart, optionally prefixed (e.g. "1_role"). - const role = /name="(?:\d+_)?role"\r\n\r\n([A-Z_]+)/.exec(submittedBody ?? "")?.[1]; - expect(role).toBe("FINANCE"); + expect(submittedBody, "form submission was not captured").not.toBeNull(); + const body = submittedBody ?? ""; + // Multipart (server action, field optionally prefixed e.g. "1_role") or + // url-encoded (pre-hydration native submit). + const role = + /name="(?:\d+_)?role"\r\n\r\n([A-Z_]+)/.exec(body)?.[1] ?? + /(?:^|&)(?:\d+_)?role=([A-Z_]+)/.exec(body)?.[1]; + expect(role, `role not found in submitted body: ${body.slice(0, 300)}`).toBe("FINANCE"); }); diff --git a/playwright.config.ts b/playwright.config.ts index 79354ab..65e4a77 100644 --- a/playwright.config.ts +++ b/playwright.config.ts @@ -47,10 +47,9 @@ export default defineConfig({ video: "off", // Windows (and some Linux setups) do not resolve *.localhost, so the // tenant host is mapped to loopback in the browser itself rather than - // relying on the OS resolver. Harmless against an external target. - launchOptions: usingExternalTarget - ? undefined - : { args: ["--host-resolver-rules=MAP *.localhost 127.0.0.1"] }, + // relying on the OS resolver. Only *.localhost is affected, so this is + // inert when E2E_BASE_URL points at a deployed environment. + launchOptions: { args: ["--host-resolver-rules=MAP *.localhost 127.0.0.1"] }, }, projects: [ @@ -62,7 +61,12 @@ export default defineConfig({ webServer: usingExternalTarget ? undefined : { - command: "npm run dev", + // CI serves a production build (`npm run start` with NODE_ENV=test, + // which keeps the dev-session bypass available) because `next dev` + // compiles every route on its first request — 30s+ per admin route on + // a CI runner, which blew the job's time budget. Locally the default + // stays `npm run dev` so the suite picks up edits. + command: process.env.E2E_SERVER_COMMAND ?? "npm run dev", // Node (unlike Chromium) does not resolve *.localhost on every // platform, so the readiness probe uses loopback directly. url: "http://127.0.0.1:3000",