diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 509fa24..07cfed4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -40,6 +40,12 @@ jobs: - name: Install dependencies run: npm ci + # Fails when a migration was added without regenerating the manifest, + # which would leave /api/readyz blind to the new migration and let a + # database that is behind the code report a false green. + - name: Migration manifest up to date + run: npm run migrations:check + - name: Tests run: npm run test @@ -51,3 +57,84 @@ jobs: - name: Build run: npm run build + + e2e: + name: E2E smoke (browser) + runs-on: ubuntu-latest + timeout-minutes: 30 + # Loads every route in a real browser and fails on 5xx / error boundaries / + # console errors. This is the layer that catches runtime-only faults such + # as a database behind the deployed code (Prisma P2022), which typecheck + # and unit tests cannot see. + services: + postgres: + image: postgres:16 + env: + POSTGRES_USER: postgres + POSTGRES_PASSWORD: postgres + POSTGRES_DB: estateos_e2e + ports: + - 5432:5432 + options: >- + --health-cmd pg_isready + --health-interval 10s + --health-timeout 5s + --health-retries 5 + env: + DATABASE_URL: postgresql://postgres:postgres@localhost:5432/estateos_e2e + DIRECT_URL: postgresql://postgres:postgres@localhost:5432/estateos_e2e + # Lets the authenticated sweep mint sessions without Clerk. Safe here: + # the database is a throwaway container, never production. + ESTATEOS_ENABLE_DEV_BYPASS: "true" + # NODE_ENV is deliberately NOT set here: the build below must run as a + # normal production build, and only the test run overrides it (see + # "Run E2E smoke suite"). + DEFAULT_COMPANY_SLUG: acme-realty + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Setup Node + uses: actions/setup-node@v4 + with: + node-version: 22 + cache: npm + + - name: Install dependencies + run: npm ci + + - name: Install Playwright browser + run: npm run e2e:install + + # The E2E database is migrated from the same migration files production + # uses, so a broken or missing migration fails here instead of in prod. + - name: Apply migrations + run: npm run db:migrate:deploy + + - name: Seed demo tenant + run: npm run db:seed + continue-on-error: true + + # Build once, then serve that build. Running the suite against `next dev` + # made every route compile on its first request (30s+ per admin route on + # a runner), which exhausted the job's time budget before the sweep + # finished. A production build serves each route in well under a second. + - name: Build + run: npm run build + + # NODE_ENV=test keeps the dev-session bypass available on a production + # build (see allowDevBypass in src/lib/config.ts), so the authenticated + # sweep still runs instead of skipping itself. + - name: Run E2E smoke suite + run: npm run e2e + env: + NODE_ENV: test + E2E_SERVER_COMMAND: npm run start + + - name: Upload Playwright report + if: failure() + uses: actions/upload-artifact@v4 + with: + name: playwright-report + path: playwright-report/ + retention-days: 7 diff --git a/.gitignore b/.gitignore index 502d5f5..ed90b2e 100644 --- a/.gitignore +++ b/.gitignore @@ -44,3 +44,9 @@ next-env.d.ts # generated local audit outputs artifacts/ + +# playwright e2e output (the specs and config ARE committed) +/test-results/ +/playwright-report/ +/blob-report/ +/playwright/.cache/ diff --git a/DEPLOY-PIPELINE.md b/DEPLOY-PIPELINE.md new file mode 100644 index 0000000..142af05 --- /dev/null +++ b/DEPLOY-PIPELINE.md @@ -0,0 +1,62 @@ +# Deploy pipeline — apply migrations + block drift + +Wire these so the `SiteSettings.draftSiteContent` class of error (schema drift) +can never ship again. The build stays DB-free; migrations run in a release step. + +## Option A — GitHub Actions (recommended if you deploy from GitHub) + +Create `.github/workflows/deploy-migrate.yml`: + +```yaml +name: DB migrate + drift gate +on: + push: + branches: [main] # or your production branch + workflow_dispatch: + +jobs: + migrate: + runs-on: ubuntu-latest + environment: production # holds the production DB secrets + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-node@v4 + with: + node-version: 20 + cache: npm + - run: npm ci + # 1) apply any pending migrations to the production DB + - run: npm run db:migrate:deploy + env: + DATABASE_URL: ${{ secrets.PROD_DATABASE_URL }} + DIRECT_URL: ${{ secrets.PROD_DIRECT_URL }} + # 2) fail the job if the DB is still behind (defense in depth) + - run: npm run db:migrate:check + env: + DATABASE_URL: ${{ secrets.PROD_DATABASE_URL }} + DIRECT_URL: ${{ secrets.PROD_DIRECT_URL }} +``` + +Make the Vercel production deploy **depend on this job succeeding** (e.g. trigger +Vercel via a deploy hook only after this workflow passes, or run the app deploy +as a later step in the same workflow). The point: migrations land BEFORE the new +code serves traffic. + +## Option B — Vercel only (no GitHub Actions) + +Vercel builds must not touch the DB, so add a **release step** you run before +promoting a build (locally or from a small CI job) with the prod connection: + +``` +DATABASE_URL="$PROD_DB" DIRECT_URL="$PROD_DIRECT" npm run db:migrate:deploy +DATABASE_URL="$PROD_DB" DIRECT_URL="$PROD_DIRECT" npm run db:migrate:check # exits 1 if behind +``` + +Then promote the Vercel deployment. Do NOT put `migrate deploy` in the Vercel +build command — the build has no guaranteed DB access and that's by design +(`scripts/run-build.mjs`). + +## The one rule +Any PR that adds a file under `prisma/migrations/` must be paired with a +`db:migrate:deploy` in the release for the environment it targets. `db:migrate:check` +is the gate that enforces it. diff --git a/FIXES-UI-AND-PAYSTACK.md b/FIXES-UI-AND-PAYSTACK.md new file mode 100644 index 0000000..7bf9ce6 --- /dev/null +++ b/FIXES-UI-AND-PAYSTACK.md @@ -0,0 +1,63 @@ +# Three fixes — company name clipping, superadmin leakage, Paystack + +## 1. Company name cut off in the dashboard sidebar — FIXED (code) +`src/components/shared/logo.tsx` had `sm:min-w-[11rem]` on the name wrapper. That +hard 176px minimum is wider than the sidebar's inner card, so long names like +"Blueprint Urban Residences LTD" overflowed and were clipped by the parent's +`overflow-hidden`. Removed the min-width, kept `min-w-0` so the flex child can +shrink, and added `break-words` so the full name wraps instead of being cut. + +## 2. Superadmin-only content on the tenant admin dashboard — FIXED (code) +Two separate leaks: + +**a) "Paystack platform readiness" card** (`owner: "Platform"`, links to +`/superadmin/settings`). The readiness checklist mixed tenant-owned and +platform-owned items and rendered the whole list to tenant admins. +- `src/components/shared/tenant-readiness-checklist.tsx`: added an `audience` + prop. `audience="tenant"` filters out any item owned by **Platform** or + **Superadmin**, or whose action link points at `/superadmin/*`. +- `src/app/(admin)/admin/settings/page.tsx` now passes `audience="tenant"`. +- The superadmin company page still sees the full list (default audience). +This also removes the other platform-owned rows (R2 storage, public-site +reachability) from the tenant view — a tenant can't action those either. + +**b) "This account cannot access the platform owner dashboard"** +(`src/app/app/access/page.tsx`, `?status=forbidden`). This page is what an +ordinary tenant admin sees if they land on a `/superadmin` URL. The guard was +correct, but the page was alarming and wrong: its "what this means" bullets are +hardcoded SUSPENSION copy ("Admin dashboards, portal access, and payment actions +are currently blocked") — untrue for this case, nothing is blocked. +- The `forbidden` case now has its own reassuring bullets ("Your company + workspace is unaffected…") and its button goes to **/admin** ("Go to your + dashboard") instead of the marketing homepage. + +NOTE: nothing is being *exposed* here — the superadmin guard works, it just +explained itself badly. If a tenant admin is hitting this page repeatedly, +something is linking them to /superadmin; tell me where you clicked from and +I'll remove that link. + +## 3. Paystack not working — CONFIG, not code +`src/modules/readiness/service.ts` sets `paystackConfigured: featureFlags.hasPaystack`, +and in `src/lib/config.ts`: + +``` +hasPaystack = PAYSTACK_SECRET_KEY && PAYSTACK_PUBLIC_KEY && PAYSTACK_WEBHOOK_SECRET +``` + +All **three** must be present. So "MISSING" = at least one is not set in the +**Vercel production** environment. Two gotchas: +- `normalizeRuntimeServerEnv()` **clears the entire group** if only some are set — + a partial config reads as completely missing. +- `getProductionReadinessIssues()` also flags it if `PAYSTACK_WEBHOOK_SECRET` + looks like a URL: it must be the **signing secret**, not the webhook URL. + +### To fix +In Vercel → Project → Settings → Environment Variables (**Production**), set all three: +- `PAYSTACK_SECRET_KEY` → `sk_live_…` +- `PAYSTACK_PUBLIC_KEY` → `pk_live_…` +- `PAYSTACK_WEBHOOK_SECRET` → the signing secret from the Paystack dashboard + (Settings → API Keys & Webhooks), **not** the webhook URL. + +Then **redeploy** (env changes need a new build). The card flips to complete. +Separately, each tenant still connects their own Paystack **subaccount** under +Admin → Settings ("Payment account" item) so buyer payments settle to them. diff --git a/INCIDENT-2026-09-15-MIGRATION-DRIFT.md b/INCIDENT-2026-09-15-MIGRATION-DRIFT.md new file mode 100644 index 0000000..9bb183e --- /dev/null +++ b/INCIDENT-2026-09-15-MIGRATION-DRIFT.md @@ -0,0 +1,97 @@ +# Incident — P2022 "column does not exist" in production (2026-09-15) + +## What users saw + +Server errors on pages that read tenant site content: + +``` +PrismaClientKnownRequestError: The column `SiteSettings.draftSiteContent` +does not exist in the current database. +code: 'P2022' +``` + +## Root cause (two layers) + +**1. The database was behind the deployed code.** +`draftSiteContent` / `publishedSiteContent` are added by migration +`0042_site_content_cms`. That migration is in the repo but was never applied +to the production database, so the deployed code queried columns that do not +exist. + +**2. The safety net was blind — this is the real bug.** +`/api/readyz` already compares applied migrations against an expected list. +That list (`EXPECTED_PRODUCTION_MIGRATIONS`) was **hand-maintained and stopped +at `0035`**, while the repo had grown to `0049`. Production therefore reported: + +```json +"migrations": { "ok": true, "missing": [] } +``` + +a **false green**, while 14 migrations — including 0042 — were missing. The +monitoring said healthy right up until users hit the broken pages. + +## Immediate fix (operator action) + +```bash +# 1. See exactly what is missing +npx prisma migrate status # with DATABASE_URL pointed at production + +# 2. Apply the pending migrations +npm run db:migrate:deploy + +# 3. Confirm the gap is closed — `missing` must be [] +curl -s https://estateos.tech/api/readyz | jq '.checks.database.migrations' +``` + +Expect several migrations to apply, not just 0042. `0049_webhook_event_dedup_unique` +deletes duplicate historical `WebhookEvent` rows before adding its unique index; +it is additive and safe, but read it before running if the table is large. + +## Prevention shipped with this change + +| Layer | What it does | +|---|---| +| **Generated manifest** | `scripts/generate-migration-manifest.mjs` derives the expected-migrations list from `prisma/migrations/`. A hand-maintained list can go stale; a generated one cannot. | +| **Build step** | `run-build.mjs` regenerates the manifest before `next build`, so every deploy ships a current list. | +| **CI guard** | `npm run migrations:check` fails the build when a migration was added without regenerating the manifest. | +| **Graceful degradation** | `getTenantSiteContentState` catches the read failure. The site-content editor shows an explanatory banner instead of a 500; the public site was already falling back to default copy. | +| **E2E browser sweep** | Playwright loads every public / admin / portal route and fails on any 5xx, error boundary, or console error. | +| **Health gate** | `e2e/health.spec.ts` asserts `migrations.missing === []`. Run it against a deployment right after release. | + +## Running the E2E suite + +```bash +npm install # first time — installs @playwright/test +npm run e2e:install # first time — downloads the browser + +npm run e2e # full sweep against a local dev server +npm run e2e:ui # interactive runner + +# Against a deployed environment (no local server is started): +E2E_BASE_URL=https://estateos.tech npm run e2e:health +``` + +The authenticated sweep signs in through `/api/dev/session`, which only mints +a session when `ESTATEOS_ENABLE_DEV_BYPASS` is on and the database is not +production. Against production those tests **skip themselves** rather than +fail, so `e2e:health` is the production-safe subset. + +## Post-deploy check (add to the release runbook) + +```bash +E2E_BASE_URL=https://estateos.tech npm run e2e:health +``` + +Green means the database matches the deployed code. Red names the exact +missing migrations. + +## Separate findings from this investigation + +- **`R2_PUBLIC_BASE_URL` is not configured in production.** `/api/readyz` + warns about it. Consequence: public assets use the signed-proxy fallback, + and because the `next/image` host allowlist is derived from this variable at + **build time**, property photos are not being optimised in production. Set + it in the build environment and redeploy to activate AVIF/WebP. +- **Sentry is disabled in production** (`sentry: "disabled"`). This incident + was found by a user, not by alerting. Enabling `SENTRY_DSN` would have + surfaced the P2022 on the first occurrence. diff --git a/USER-PROVISIONING-SPEC.md b/USER-PROVISIONING-SPEC.md new file mode 100644 index 0000000..61487c9 --- /dev/null +++ b/USER-PROVISIONING-SPEC.md @@ -0,0 +1,115 @@ +# Build spec for Claude Code — operator-created users (buyers + staff) + +Run locally where you have the shell, `.env.local` with `CLERK_SECRET_KEY`, and can +run `npm run check`. This is security-sensitive (account + credential creation) — read +the whole spec, reuse the existing infra called out, and test before committing. + +## Goal +Let operators create accounts that the person can log in with later: +- **Front desk** → create a **buyer** with a full profile (a walk-in / offline buyer + who didn't purchase through the portal). They can log in with the email provided. +- **CEO (admin)** → add company staff to roles: **Marketer / Finance (accountant) / + Front desk (STAFF) / Legal**. +- For any created account, the operator **chooses the credential delivery**: + 1. **Invite / set-password link** (primary; works with any Clerk config). + 2. **Temporary password shown once on screen** (only when Clerk password auth is + enabled — gate behind a flag, default OFF). + +## Auth reality (do not fight this) +Auth is **Clerk**; the DB has no password column. Assume **password sign-in is OFF** +until confirmed in the Clerk Dashboard, so the **invite/set-password link is the default +and the only path enabled by default.** The temp-password path is implemented but gated +behind a new flag (`featureFlags.hasClerkPassword`, from env `CLERK_PASSWORD_ENABLED`, +default false) so it can be switched on later without code changes. + +## Reuse this existing infrastructure (don't reinvent) +- `src/lib/auth/clerk-user-sync.ts` — `syncAuthenticatedClerkUser` already **links a + placeholder** `clerkUserId` that starts with `manual:` to the real Clerk account on + first sign-in (matched by email). This is THE mechanism for operator-created users: + create the local `User` now with `clerkUserId = "manual:" + randomUUID()`, and Clerk + links it when they sign in. Confirm the email-match path and extend if needed. +- `src/modules/invitations/team-invitations.ts` — the invite create/accept pattern + (`createTeamMemberInvitation` / `acceptTeamMemberInvitation`, `TeamMemberInvitation` + table, TTL, audit). Generalize it to also cover **BUYER** and to link to a + pre-created profile. Prefer reusing your own invitation table over Clerk invitations + so the accept flow + local linkage you already have keeps working. +- `src/modules/admin/user-actions.ts` (`setUserRoleAction`) + `src/modules/admin/users.ts` + (`OPERATOR_ROLES`, `ROLE_LABELS`, `CompanyUserRow`) — reuse for role assignment; the + Marketer path already auto-provisions a `StaffProfile`. +- Buyer profile fields: the `Profile` model + `src/modules/kyc/service.ts` + (`getBuyerProfileRecord`) — reuse for the front-desk buyer form fields. +- `src/lib/audit/service.ts` (`writeAuditLog`), `@clerk/nextjs/server` (`clerkClient`). + +## Backend: `src/modules/provisioning/provision-user.ts` (new) +`provisionCompanyUser(input)`: +- Input: `{ companyId, email, fullName, phone?, role, branchId?, buyerProfile?, delivery, actor }` + where `role ∈ {BUYER, MARKETER, FINANCE, STAFF, LEGAL}`, `delivery ∈ {"invite","password"}`. +- **Authorize the actor**: front-desk (STAFF) may create **BUYER only**; ADMIN/SUPER_ADMIN + may create any of the roles above; never `SUPER_ADMIN`/`ADMIN` through this flow. +- **Validate**: normalize email; reject if a User with that email already exists in the + company; company-scope everything; full name ≥ 2 chars. +- **Create local records** in a transaction: + - `User` with `clerkUserId = "manual:" + randomUUID()`, email, firstName/lastName split + from fullName, phone, companyId, branchId, isActive true. + - Role via `Role`/`UserRole` upsert (reuse the pattern in `setUserRoleAction`). + - For BUYER: create the buyer `Profile` from `buyerProfile` fields. + - For MARKETER: `StaffProfile` (isAssignable) — mirror the existing auto-provision. +- **Delivery**: + - `"invite"` (default/only-enabled): create a `TeamMemberInvitation`-style record + (generalized to BUYER) and email the accept/set-password link (reuse the email + sender + template; add a buyer variant). Return `{ mode: "invite", email }`. + - `"password"` (only if `featureFlags.hasClerkPassword`): call + `clerkClient().users.createUser({ emailAddress:[email], password: , + publicMetadata:{ mustResetPassword:true, companyId } })`, set the local + `User.clerkUserId` to the real Clerk id (replace the placeholder), and return + `{ mode: "password", oneTimePassword: }`. If disabled, reject with a + clear message so the UI only offers "invite". +- **Password generation**: strong (≥14 chars, upper/lower/digit/symbol) that satisfies + Clerk's policy. **Never persist it, never put it in audit payloads or logs.** +- **Audit**: `writeAuditLog` the creation (email, role, delivery mode, actor) — NO password. + +## Force change on first login (password path only) +Set `publicMetadata.mustResetPassword = true` at creation. In the portal/admin post-auth +resolution (or middleware), if the signed-in Clerk user has `mustResetPassword`, redirect +to a change-password page until it's cleared; clear the flag after they change it. + +## UI +- **Front desk** — add an "Add buyer" entry on `/admin/front-desk` (or a new + `/admin/front-desk/new-buyer`). Full buyer form (name, email, phone, address/city/state, + occupation, notes) + a delivery choice (Invite link / Temp password — the latter only + shown when `hasClerkPassword`). Submit → server action → **result modal**. +- **CEO** — add an "Add person" button on the Users tab (`users-management.tsx`). Modal: + name, email, phone, role picker (Marketer / Finance / Front desk / Legal) + delivery + choice → same result modal. +- **Credential result modal** (shared): on `mode:"password"` show the email + one-time + password with a Copy button and a bold "This is shown once. Ask them to change it after + first login." note; on `mode:"invite"` show "Invite sent to ." Both re-fetch the + list so the new user appears. + +## Security guardrails (must all hold) +- Actor-role gating (front-desk → BUYER only) enforced **server-side**, not just UI. +- Company-scoped; email unique per company. +- One-time password shown once, never stored, never logged/audited. +- Invite tokens single-use + TTL (reuse existing). +- Force-change-on-first-login for the password path. +- Rate-limit / audit every creation. + +## Schema / migration +Aim for **no migration**: use the `manual:` placeholder clerkUserId + `publicMetadata` +for the reset flag + audit logs. If you must add a column (e.g. `User.createdByUserId`), +add a migration and follow `DEPLOY-PIPELINE.md` (`db:migrate:deploy` + `db:migrate:check`). + +## Testing (before committing) +1. `npm run check` green. +2. Unit tests for `provisionCompanyUser`: role gating (front-desk can't create staff), + email-uniqueness, password strength, delivery branching. +3. Manual E2E (dev): front-desk creates a buyer → invite path → accept/sign-in → buyer + lands in `/portal` with the profile populated. CEO creates a Marketer → appears in + Users with the role + a StaffProfile → assignable on `/admin/leads`. +4. If password auth is enabled in Clerk: create with temp password → sign in with it → + forced to change on first login. + +## Clerk config note +Temp-password path needs Clerk Dashboard → User & Authentication → **Email+Password ON** +and `CLERK_SECRET_KEY` set. Keep `CLERK_PASSWORD_ENABLED=false` until then; the invite +path needs neither and should ship first. diff --git a/e2e/add-person-role.spec.ts b/e2e/add-person-role.spec.ts new file mode 100644 index 0000000..8b64ef6 --- /dev/null +++ b/e2e/add-person-role.spec.ts @@ -0,0 +1,65 @@ +import { expect, test } from "@playwright/test"; + +/** + * Regression: /admin/users → "Add person" always submitted STAFF ("Front Desk"). + * + * The Role