From 77801b20fc2773ef7cdcb9b5f83f7a2f8b482ee5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andreas=20Fr=C3=B8yland?= <81354124+Andreas-Froyland@users.noreply.github.com> Date: Mon, 21 Sep 2026 09:42:53 +0200 Subject: [PATCH 1/4] feat(runner): guarantee a no-build CLI and add the Linux display preflight (Task 2.2, part 1) Node 22.18 or newer runs the TypeScript directly, so the CLI needs no build step. Fix the one construct that stopped the package loading that way (a constructor parameter property), enforce it with erasableSyntaxOnly, prove it with a test that loads the package under plain Node and expects no warning, raise the engines floor to 22.18 (verified against Node's changelog) and record the decision, including that a distributable would still need a build. Add classifyDisplay and a gatherer that reads the machine: a display is virtual only when the virtual server (Xvfb, Xvnc, Xdummy) serving that exact display is visible, real only when the session type says desktop, and otherwise unknown. It never decides from the operating system's name. Environment inspection now reports the display kind and adds a real-display capability that a virtual display does not satisfy. Co-Authored-By: Claude Sonnet 5 --- docs/decisions/tool-layout.md | 4 +- package.json | 2 +- packages/qa/package.json | 2 +- packages/qa/src/runner/display.ts | 64 ++++++++++++++ packages/qa/src/runner/environment.ts | 42 +++++++--- packages/qa/test/no-build.test.ts | 21 +++++ packages/qa/test/runner/display.test.ts | 92 +++++++++++++++++++++ packages/qa/test/runner/environment.test.ts | 39 +++++++++ packages/qa/test/runner/execute.test.ts | 15 ++++ packages/qa/tsconfig.json | 1 + 10 files changed, 265 insertions(+), 17 deletions(-) create mode 100644 packages/qa/src/runner/display.ts create mode 100644 packages/qa/test/no-build.test.ts create mode 100644 packages/qa/test/runner/display.test.ts diff --git a/docs/decisions/tool-layout.md b/docs/decisions/tool-layout.md index 5eb8475..64cb055 100644 --- a/docs/decisions/tool-layout.md +++ b/docs/decisions/tool-layout.md @@ -9,7 +9,7 @@ These defaults come from the plan's proposals and the Stage 0 results ([native a | Area | Decision | Basis | | --- | --- | --- | | Repository | `Frogbyte-io/release-qa`, public | Created and made public before Stage 0 | -| Runtime | Node.js 22. `.node-version` pins **22.23.2** exactly; `engines.node` is the floor `>=22.12.0` | The harness ran on Node 22.23.2 (Linux) and 24.13.0 (Windows). CI installs the version in `.node-version`; bump it deliberately | +| Runtime | Node.js 22. `.node-version` pins **22.23.2** exactly; `engines.node` is the floor `>=22.18.0` | The harness ran on Node 22.23.2 (Linux) and 24.13.0 (Windows). 22.18.0 is the first 22.x with type stripping on by default and without an experimental warning (per Node's changelog). CI installs the version in `.node-version`; bump it deliberately | | Package manager | npm workspaces, one root `package-lock.json` | Plan default; nothing in Stage 0 contradicted it | | Language and test | TypeScript 7.0.2, Vitest 5.0.1, `@types/node` 22.20.4, all pinned exactly | Verified together in this change: `npm run typecheck` and `npm test` pass, and a deliberate type error fails the typecheck | | Native driver (sample apps) | WebdriverIO 9.31.9 `remote()` API + external `tauri-driver` 2.0.6; Edge WebDriver on Windows, `WebKitWebDriver` on Linux | Proven on the sample: 60/60 attempts across three runs per platform | @@ -31,7 +31,7 @@ docs/decisions/ decision records `examples/tauri-smoke` stays outside the workspace on purpose. The packages tested in Stage 0 were built with its own `package-lock.json`, and hoisting its build tooling into a root lockfile would change that resolution without a re-test. Revisit when the sample is consumed by the runner's own tests. -`packages/qa/tsconfig.json` sets `noEmit`. The CLI needs a build step (emit or bundle), which Task 2.2 decides when there is a CLI to build. `bin`, `main` and `exports` are intentionally absent from `package.json` until then: the source is TypeScript that Node cannot load directly, so nothing should be able to import the package before it has a build output. +**No build step (decided in Task 2.2).** Node 22.18 or newer runs the TypeScript directly by stripping types, so the CLI is `node packages/qa/src/cli/bin.ts`. That only works while the source uses *erasable* syntax (no enums, namespaces or constructor parameter properties), which `erasableSyntaxOnly` in `tsconfig.json` enforces at typecheck time, and `test/no-build.test.ts` loads the package under plain Node on every CI runner. `packages/qa/tsconfig.json` keeps `noEmit`. This suits running from a checkout; **a published or installed distribution would still need a build or bundle**, and `bin`, `main` and `exports` stay out of `package.json` until that is decided. ## Stage 0 commands and where they go diff --git a/package.json b/package.json index 5306dbe..a356551 100644 --- a/package.json +++ b/package.json @@ -8,7 +8,7 @@ "apps/*" ], "engines": { - "node": ">=22.12.0" + "node": ">=22.18.0" }, "scripts": { "typecheck": "npm run typecheck --workspaces --if-present", diff --git a/packages/qa/package.json b/packages/qa/package.json index 80950ec..3dabdcf 100644 --- a/packages/qa/package.json +++ b/packages/qa/package.json @@ -5,7 +5,7 @@ "description": "Release QA runner, contracts, reports and GitHub integration", "type": "module", "engines": { - "node": ">=22.12.0" + "node": ">=22.18.0" }, "scripts": { "typecheck": "tsc -p tsconfig.json --noEmit", diff --git a/packages/qa/src/runner/display.ts b/packages/qa/src/runner/display.ts new file mode 100644 index 0000000..053da53 --- /dev/null +++ b/packages/qa/src/runner/display.ts @@ -0,0 +1,64 @@ +import { readdir, readFile } from 'node:fs/promises'; +import { platform as hostPlatform } from 'node:os'; +import { basename } from 'node:path'; + +/** + * `virtual`: a display server that is not a screen (Xvfb and friends). `real`: the machine's own graphical + * session. `unknown`: something answers on a display but nothing shows which. `none`: no display at all. + */ +export type DisplayKind = 'none' | 'virtual' | 'real' | 'unknown'; + +/** What the machine says about its graphical session. Gathered separately so classification stays pure. */ +export interface DisplayFacts { + platform: NodeJS.Platform; + env: Readonly>; + /** Command lines of the running processes (Linux only; empty elsewhere). */ + commandLines: readonly string[]; +} + +/** X servers that draw to memory or a remote viewer rather than to a screen. */ +const VIRTUAL_SERVERS = new Set(['Xvfb', 'Xvnc', 'Xdummy']); + +/** + * Never guesses: it names a display virtual only when it can see the virtual server that serves it, and real only + * when the session says it is a desktop session. Anything else that has a display is `unknown`, and the caller + * decides what to do with that. The operating system's name alone never decides. + */ +export function classifyDisplay(facts: DisplayFacts): { kind: DisplayKind; detail: string } { + if (facts.platform === 'win32') { + // Services run in a session named "Services" with no desktop; an interactive one is "Console" or "RDP-Tcp#n". + const session = facts.env.SESSIONNAME ?? ''; + return session !== '' && session !== 'Services' ? { kind: 'real', detail: `interactive session ${session}` } : { kind: 'none', detail: 'no interactive session' }; + } + if (facts.platform !== 'linux') return { kind: 'unknown', detail: `cannot tell on ${facts.platform}` }; + + const { DISPLAY: display, WAYLAND_DISPLAY: wayland, XDG_SESSION_TYPE: session } = facts.env; + if (!display && !wayland) return { kind: 'none', detail: 'neither DISPLAY nor WAYLAND_DISPLAY is set' }; + + if (display) { + for (const line of facts.commandLines) { + const [program = '', ...args] = line.trim().split(/\s+/); + const name = basename(program); + if (VIRTUAL_SERVERS.has(name) && args.includes(display)) return { kind: 'virtual', detail: `${name} serving ${display}` }; + } + } + if ((session === 'x11' && display) || (session === 'wayland' && (wayland || display))) return { kind: 'real', detail: `${session} desktop session` }; + return { kind: 'unknown', detail: `a display is set (${display ?? wayland}) but it is not evidently a desktop session or a virtual server` }; +} + +/** Reads what this machine says. Best effort: anything unreadable is simply absent from the facts. */ +export async function readDisplayFacts(): Promise { + const platform = hostPlatform(); + return { platform, env: process.env, commandLines: platform === 'linux' ? await linuxCommandLines() : [] }; +} + +async function linuxCommandLines(): Promise { + const entries = await readdir('/proc').catch(() => [] as string[]); + const lines = await Promise.all( + entries + .filter((name) => /^\d+$/.test(name)) + .slice(0, 5000) + .map((pid) => readFile(`/proc/${pid}/cmdline`, 'utf8').then((text) => text.split('\0').join(' ').trim(), () => '')), + ); + return lines.filter((line) => line !== ''); +} diff --git a/packages/qa/src/runner/environment.ts b/packages/qa/src/runner/environment.ts index ffa0c3f..c6ed6dd 100644 --- a/packages/qa/src/runner/environment.ts +++ b/packages/qa/src/runner/environment.ts @@ -3,26 +3,27 @@ import { readFile } from 'node:fs/promises'; import { arch, platform, release } from 'node:os'; import type { EnvironmentProfile } from '../model/project.ts'; import type { MeasuredEnvironment } from '../model/result.ts'; +import { classifyDisplay, readDisplayFacts, type DisplayKind } from './display.ts'; /** How the runner asks the machine what it can do. Injectable so tests need no display or sound card. */ export interface EnvironmentProbes { display(): Promise; audio(): Promise; + /** What kind of display it is, when the probe can tell. Without it a display is of unknown kind. */ + describeDisplay?(): Promise<{ kind: DisplayKind; detail: string }>; } /** - * Heuristics, not proof. `display` says a graphical session appears to be reachable; `audio` says the sound - * subsystem appears to be running. Neither checks that a particular device exists or that a virtual display is - * not standing in for a real one; a scenario that needs that must check it itself. + * Heuristics, not proof. `display` says a graphical session appears to be reachable and `describeDisplay` says + * whether it looks virtual or real, or admits it cannot tell; `audio` says the sound subsystem appears to be + * running. Neither checks that a particular device exists. */ export const defaultProbes: EnvironmentProbes = { async display() { - if (platform() === 'win32') { - // Services run in a non-interactive session, which is named "Services". - const session = process.env.SESSIONNAME; - return session !== undefined && session !== '' && session !== 'Services'; - } - return Boolean(process.env.DISPLAY || process.env.WAYLAND_DISPLAY); + return classifyDisplay(await readDisplayFacts()).kind !== 'none'; + }, + async describeDisplay() { + return classifyDisplay(await readDisplayFacts()); }, async audio() { if (platform() === 'win32') return (await run('sc', ['query', 'Audiosrv'])).includes('RUNNING'); @@ -39,6 +40,8 @@ function run(command: string, args: readonly string[]): Promise { export interface InspectedEnvironment { environment: MeasuredEnvironment; + /** What the machine's display is, for reports and `doctor`. */ + display: { kind: DisplayKind; detail: string }; /** Set when the machine is not what the profile asks for (operating system or architecture). */ profileMismatch?: string; } @@ -46,11 +49,14 @@ export interface InspectedEnvironment { const OS_NAMES: Record = { win32: 'windows', linux: 'linux', darwin: 'macos' }; const ARCH_NAMES: Record = { x64: 'x86_64', arm64: 'aarch64' }; -/** Measures this machine. A probe that throws means the capability is absent, never a crash of the run. */ +/** + * Measures this machine. A probe that throws means the capability is absent, never a crash of the run. A display + * is a `display` capability, and only a display that is evidently a desktop session is also a `real-display`. + */ export async function inspectEnvironment(profile: EnvironmentProfile, probes: EnvironmentProbes = defaultProbes, toolVersion = '0.0.0'): Promise { // Run inside a promise so a probe that throws before returning one is caught too, not only one that rejects. const has = (probe: () => Promise): Promise => Promise.resolve().then(probe).catch(() => false); - const [display, audio] = await Promise.all([has(() => probes.display()), has(() => probes.audio())]); + const [audio, display] = await Promise.all([has(() => probes.audio()), describeDisplay(probes)]); const os = OS_NAMES[platform()] ?? platform(); const architecture = ARCH_NAMES[arch()] ?? arch(); @@ -58,12 +64,22 @@ export async function inspectEnvironment(profile: EnvironmentProfile, probes: En os, osVersion: release(), arch: architecture, - capabilities: [...(audio ? ['audio'] : []), ...(display ? ['display'] : [])], + capabilities: [...(audio ? ['audio'] : []), ...(display.kind !== 'none' ? ['display'] : []), ...(display.kind === 'real' ? ['real-display'] : [])], toolVersion, }; const problems: string[] = []; if (os !== profile.os) problems.push(`this machine is ${os}, the profile expects ${profile.os}`); if (architecture !== profile.arch) problems.push(`this machine is ${architecture}, the profile expects ${profile.arch}`); - return problems.length === 0 ? { environment } : { environment, profileMismatch: problems.join('; ') }; + return problems.length === 0 ? { environment, display } : { environment, display, profileMismatch: problems.join('; ') }; +} + +/** A probe that can only say yes or no describes a display of unknown kind; one that throws describes none. */ +async function describeDisplay(probes: EnvironmentProbes): Promise<{ kind: DisplayKind; detail: string }> { + try { + if (probes.describeDisplay !== undefined) return await probes.describeDisplay(); + return (await probes.display()) ? { kind: 'unknown', detail: 'a display is present but its kind cannot be told' } : { kind: 'none', detail: 'no display' }; + } catch { + return { kind: 'none', detail: 'the display probe failed' }; + } } diff --git a/packages/qa/test/no-build.test.ts b/packages/qa/test/no-build.test.ts new file mode 100644 index 0000000..5cb50fd --- /dev/null +++ b/packages/qa/test/no-build.test.ts @@ -0,0 +1,21 @@ +import { execFile } from 'node:child_process'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { promisify } from 'node:util'; +import { expect, test } from 'vitest'; + +const run = promisify(execFile); +const packageDir = join(dirname(fileURLToPath(import.meta.url)), '..'); + +// The CLI has no build step: Node (22.18 or newer) runs the TypeScript directly by stripping types. That only works +// while the source uses erasable syntax alone (no enums, namespaces or constructor parameter properties), which the +// `erasableSyntaxOnly` compiler option enforces. The floor of 22.18 is when stripping became default and silent. +test('the package loads under Node type stripping, so the CLI needs no build, and it prints no warning', async () => { + const { stdout, stderr } = await run( + process.execPath, + ['-e', "import('./src/index.ts').then((m) => console.log(typeof m.executeScenario + ' ' + typeof m.evaluate + ' ' + typeof m.renderReport))"], + { cwd: packageDir }, + ); + expect(stdout.trim()).toBe('function function function'); + expect(stderr).toBe(''); +}); diff --git a/packages/qa/test/runner/display.test.ts b/packages/qa/test/runner/display.test.ts new file mode 100644 index 0000000..6baf1ac --- /dev/null +++ b/packages/qa/test/runner/display.test.ts @@ -0,0 +1,92 @@ +import { describe, expect, test } from 'vitest'; +import { classifyDisplay, readDisplayFacts, type DisplayFacts } from '../../src/runner/display.ts'; + +const linux = (env: Record, commandLines: string[] = []): DisplayFacts => ({ platform: 'linux', env, commandLines }); +const windows = (env: Record): DisplayFacts => ({ platform: 'win32', env, commandLines: [] }); + +describe('Linux', () => { + test('no display variables at all means no display', () => { + expect(classifyDisplay(linux({})).kind).toBe('none'); + expect(classifyDisplay(linux({ XDG_SESSION_TYPE: 'tty' })).kind).toBe('none'); + // ...and it is the variables that decide, not the platform. + expect(classifyDisplay(linux({ DISPLAY: ':0', XDG_SESSION_TYPE: 'x11' })).kind).not.toBe('none'); + }); + + test('a display served by Xvfb is virtual, and says which server', () => { + const result = classifyDisplay(linux({ DISPLAY: ':99' }, ['Xvfb :99 -screen 0 1280x1024x24 -nolisten tcp -auth /tmp/xvfb-run.x/Xauthority'])); + expect(result.kind).toBe('virtual'); + expect(result.detail).toContain('Xvfb'); + expect(result.detail).toContain(':99'); + }); + + test.each([['Xvnc :1 -geometry 1280x800'], ['/usr/bin/Xvfb :1 -screen 0 800x600x24'], ['Xdummy :1']])('%s is a virtual display server', (line) => { + expect(classifyDisplay(linux({ DISPLAY: ':1' }, [line])).kind).toBe('virtual'); + }); + + test('a virtual server is recognised even when the session type says tty (e.g. started over ssh)', () => { + expect(classifyDisplay(linux({ DISPLAY: ':99', XDG_SESSION_TYPE: 'tty' }, ['Xvfb :99 -screen 0 1024x768x24'])).kind).toBe('virtual'); + }); + + test('a virtual server on a different display number does not make this display virtual', () => { + expect(classifyDisplay(linux({ DISPLAY: ':99' }, ['Xvfb :98 -screen 0 1024x768x24'])).kind).toBe('unknown'); + }); + + test('a desktop session is real', () => { + expect(classifyDisplay(linux({ DISPLAY: ':0', XDG_SESSION_TYPE: 'x11' })).kind).toBe('real'); + expect(classifyDisplay(linux({ WAYLAND_DISPLAY: 'wayland-0', XDG_SESSION_TYPE: 'wayland' })).kind).toBe('real'); + }); + + test('Xwayland on a Wayland desktop is a real session, not a virtual display', () => { + const facts = linux({ DISPLAY: ':0', WAYLAND_DISPLAY: 'wayland-0', XDG_SESSION_TYPE: 'wayland' }, ['/usr/bin/Xwayland :0 -auth /run/user/1000/x']); + expect(classifyDisplay(facts).kind).toBe('real'); + }); + + test('a display that is neither a known virtual server nor a desktop session is reported as unknown, never guessed', () => { + // For example X forwarding over ssh: something is listening, but it is not evidently this machine's own desktop. + expect(classifyDisplay(linux({ DISPLAY: 'localhost:10.0', XDG_SESSION_TYPE: 'tty' })).kind).toBe('unknown'); + expect(classifyDisplay(linux({ DISPLAY: ':0' })).kind).toBe('unknown'); + expect(classifyDisplay(linux({ WAYLAND_DISPLAY: 'wayland-0' })).kind).toBe('unknown'); + }); + + test('a process merely mentioning Xvfb in its arguments does not make the display virtual', () => { + expect(classifyDisplay(linux({ DISPLAY: ':0', XDG_SESSION_TYPE: 'x11' }, ['/usr/bin/vim notes-about-Xvfb :0.txt'])).kind).toBe('real'); + }); +}); + +describe('Windows', () => { + test.each([['Console'], ['RDP-Tcp#0']])('the interactive session %s is real', (session) => { + expect(classifyDisplay(windows({ SESSIONNAME: session })).kind).toBe('real'); + }); + + test.each([['Services'], ['']])('the non-interactive session %j has no display', (session) => { + expect(classifyDisplay(windows({ SESSIONNAME: session })).kind).toBe('none'); + expect(classifyDisplay(windows({ SESSIONNAME: 'Console' })).kind).toBe('real'); + }); + + test('an unknown session name is not assumed to be a desktop', () => { + expect(classifyDisplay(windows({})).kind).toBe('none'); + expect(classifyDisplay(windows({ SESSIONNAME: 'Console' })).kind).toBe('real'); + }); +}); + +test('other platforms are reported as unknown rather than guessed', () => { + expect(classifyDisplay({ platform: 'darwin', env: {}, commandLines: [] }).kind).toBe('unknown'); +}); + +describe('reading this machine', () => { + test('reports the platform and the environment it really has', async () => { + const facts = await readDisplayFacts(); + expect(facts.platform).toBe(process.platform); + expect(facts.env.PATH ?? facts.env.Path).toBeDefined(); + }); + + test.skipIf(process.platform !== 'linux')('on Linux it reads the command lines of running processes from /proc, including this one', async () => { + const facts = await readDisplayFacts(); + expect(facts.commandLines.length).toBeGreaterThan(0); + expect(facts.commandLines.some((line) => line.includes(process.execPath) || /node/.test(line))).toBe(true); + }); + + test.skipIf(process.platform === 'linux')('off Linux it does not pretend to have process command lines', async () => { + expect((await readDisplayFacts()).commandLines).toEqual([]); + }); +}); diff --git a/packages/qa/test/runner/environment.test.ts b/packages/qa/test/runner/environment.test.ts index f6b15f5..b5eff03 100644 --- a/packages/qa/test/runner/environment.test.ts +++ b/packages/qa/test/runner/environment.test.ts @@ -53,6 +53,45 @@ describe('inspectEnvironment', () => { expect(profileMismatch).toContain(hostArch()); }); + describe('the kind of display', () => { + const withDisplay = (kind: 'none' | 'virtual' | 'real' | 'unknown', detail = 'test') => ({ + display: async () => kind !== 'none', + audio: async () => false, + describeDisplay: async () => ({ kind, detail }), + }); + + test.each([ + ['a real desktop', 'real', ['display', 'real-display']], + ['a virtual display', 'virtual', ['display']], + ['a display of unknown kind', 'unknown', ['display']], + ['no display', 'none', []], + ] as const)('%s gives the capabilities %j', async (_label, kind, capabilities) => { + const inspected = await inspectEnvironment(hostProfile(), withDisplay(kind, 'why')); + expect(inspected.environment.capabilities).toEqual(capabilities); + expect(inspected.display).toEqual({ kind, detail: 'why' }); + }); + + test('a probe that only says yes or no is a display of unknown kind, never a real one', async () => { + const inspected = await inspectEnvironment(hostProfile(), probes(true, false)); + expect(inspected.display.kind).toBe('unknown'); + expect(inspected.environment.capabilities).toEqual(['display']); + }); + + test('a display description that throws before it returns a promise is also no display', async () => { + const describeDisplay = (): Promise => { throw new Error('thrown synchronously'); }; + const inspected = await inspectEnvironment(hostProfile(), { display: async () => true, audio: async () => false, describeDisplay }); + expect(inspected.display.kind).toBe('none'); + expect(inspected.environment.capabilities).toEqual([]); + }); + + test('a display probe that throws is no display', async () => { + const broken = { display: async () => true, audio: async () => false, describeDisplay: async (): Promise => { throw new Error('boom'); } }; + const inspected = await inspectEnvironment(hostProfile(), broken); + expect(inspected.display.kind).toBe('none'); + expect(inspected.environment.capabilities).toEqual([]); + }); + }); + test('the default probes answer with a boolean and never throw, whatever this machine has', async () => { expect(typeof (await defaultProbes.display())).toBe('boolean'); expect(typeof (await defaultProbes.audio())).toBe('boolean'); diff --git a/packages/qa/test/runner/execute.test.ts b/packages/qa/test/runner/execute.test.ts index af6564f..23492d3 100644 --- a/packages/qa/test/runner/execute.test.ts +++ b/packages/qa/test/runner/execute.test.ts @@ -68,6 +68,21 @@ describe('blocked: prerequisites that are not met', () => { expect(result.missing).toEqual(['audio', 'display']); }); + test('a scenario that needs a real desktop is blocked on a virtual display, which does satisfy a plain display need', async () => { + const virtual = { display: async () => true, audio: async () => true, describeDisplay: async () => ({ kind: 'virtual' as const, detail: 'Xvfb on :99' }) }; + const needsReal = await arrange({ probes: virtual }); + const blocked = await executeScenario(needsReal.context, scenarioOf({ requirement: requirement({ key: `${hostOs()}/persistence`, capabilities: ['real-display'] }) })); + expect(blocked).toMatchObject({ outcome: 'blocked', reason: 'capability-missing', missing: ['real-display'] }); + expect(needsReal.calls).toEqual([]); + + const real = { ...virtual, describeDisplay: async () => ({ kind: 'real' as const, detail: 'x11 session' }) }; + const onRealDesktop = await arrange({ probes: real }); + expect((await executeScenario(onRealDesktop.context, scenarioOf({ requirement: requirement({ key: `${hostOs()}/persistence`, capabilities: ['real-display'] }) }))).outcome).toBe('passed'); + + const needsAny = await arrange({ probes: virtual }); + expect((await executeScenario(needsAny.context, scenarioOf({ requirement: requirement({ key: `${hostOs()}/persistence`, capabilities: ['display'] }) }))).outcome).toBe('passed'); + }); + test('does not need a capability the scenario never asked for', async () => { const { context } = await arrange({ probes: { display: async () => false, audio: async () => false } }); expect((await executeScenario(context, scenarioOf())).outcome).toBe('passed'); diff --git a/packages/qa/tsconfig.json b/packages/qa/tsconfig.json index 12aa7fe..6b68c46 100644 --- a/packages/qa/tsconfig.json +++ b/packages/qa/tsconfig.json @@ -9,6 +9,7 @@ "noUncheckedIndexedAccess": true, "exactOptionalPropertyTypes": true, "verbatimModuleSyntax": true, + "erasableSyntaxOnly": true, "isolatedModules": true, "allowImportingTsExtensions": true, "noEmit": true, From 00bb9845ad6554db2cb8adcb4acf70e210cc5d4d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andreas=20Fr=C3=B8yland?= <81354124+Andreas-Froyland@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:17:13 +0200 Subject: [PATCH 2/4] fix(test): stop racing a short phase deadline against a fixed sleep in the late-ownership tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit "a late spawn/own from a hook that was cut off is refused" assumed prerequisites and phase entry always finish in a few milliseconds (phaseMs: 30) before the hook's own 150 ms sleep elapses. On a loaded or slow CI machine that margin is not guaranteed: prerequisites itself can exceed the phase deadline, so the install hook is never entered and the late attempt never happens, leaving the test's "refused" assertion looking at undefined. Observed as a Windows CI failure; not reproducible locally even under CPU load, consistent with I/O/scheduling jitter specific to that runner. Both tests now wait for the install hook to actually start (eventually(() => calls.includes('install'))), then abort the run directly and let the hook's own abort listener trigger the late attempt — the same pattern the neighbouring "ignores cancellation" test already uses. This removes the wall-clock race entirely instead of loosening it. Verified the tests still catch the regression: neutering the ownership check they exercise makes both fail. Co-Authored-By: Claude Sonnet 5 --- .../qa/test/runner/execute-hardening.test.ts | 35 ++++++++++++------- 1 file changed, 23 insertions(+), 12 deletions(-) diff --git a/packages/qa/test/runner/execute-hardening.test.ts b/packages/qa/test/runner/execute-hardening.test.ts index a273014..9b62145 100644 --- a/packages/qa/test/runner/execute-hardening.test.ts +++ b/packages/qa/test/runner/execute-hardening.test.ts @@ -175,14 +175,19 @@ describe('hooks that outlive their phase', () => { expect(result.cleanup.ok).toBe(true); }); + // These two wait for the phase to actually be cut off (the abort reaching the hook's own signal) before the hook + // acts, rather than racing a short phase deadline against a fixed sleep: a race like that assumes prerequisites and + // phase entry reliably finish in a few milliseconds, which does not hold on a loaded or slow CI machine. test('a late spawn from a hook that was cut off is refused and starts nothing', async () => { let refused: unknown; + const calls: string[] = []; // The child would write this file if it were ever allowed to run; it lives in a scratch directory, never the repo. const markerFile = join(await makeTempDir('qa-late-'), 'late-spawn-marker'); - const { context, testRoot } = await arrange({ - lifecycle: lifecycleOf([], { + const { context, controller, testRoot } = await arrange({ + lifecycle: lifecycleOf(calls, { install: async (ctx) => { - await sleep(150); + calls.push('install'); + await new Promise((resolve) => ctx.signal.addEventListener('abort', () => resolve(), { once: true })); try { await ctx.spawn('late', process.execPath, ['-e', `require('fs').writeFileSync(${JSON.stringify(markerFile)}, 'x'); setInterval(() => {}, 1000)`], { stdio: 'ignore' }); } catch (error) { @@ -190,11 +195,13 @@ describe('hooks that outlive their phase', () => { } }, }), - timeouts: { phaseMs: 30, stepsMs: 2000, cleanupMs: 2000, abandonedGraceMs: 400 }, + timeouts: { phaseMs: 2000, stepsMs: 2000, cleanupMs: 2000, abandonedGraceMs: 400 }, }); - await executeScenario(context, scenarioOf()); - await sleep(500); + const running = executeScenario(context, scenarioOf()); + await eventually(() => calls.includes('install')); + controller.abort(); + await running; expect(refused).toBeInstanceOf(Error); expect(String((refused as Error).message)).toMatch(/stopped|no longer/i); @@ -205,10 +212,12 @@ describe('hooks that outlive their phase', () => { test('a late claim of ownership from a hook that was cut off is refused', async () => { let refused: unknown; let root = ''; - const { context, testRoot } = await arrange({ - lifecycle: lifecycleOf([], { + const calls: string[] = []; + const { context, controller, testRoot } = await arrange({ + lifecycle: lifecycleOf(calls, { install: async (ctx) => { - await sleep(150); + calls.push('install'); + await new Promise((resolve) => ctx.signal.addEventListener('abort', () => resolve(), { once: true })); try { await ctx.own({ kind: 'path', path: join(root, 'late'), label: 'late' }); } catch (error) { @@ -216,11 +225,13 @@ describe('hooks that outlive their phase', () => { } }, }), - timeouts: { phaseMs: 30, stepsMs: 2000, cleanupMs: 2000, abandonedGraceMs: 400 }, + timeouts: { phaseMs: 2000, stepsMs: 2000, cleanupMs: 2000, abandonedGraceMs: 400 }, }); root = testRoot; - await executeScenario(context, scenarioOf()); - await sleep(500); + const running = executeScenario(context, scenarioOf()); + await eventually(() => calls.includes('install')); + controller.abort(); + await running; expect(refused).toBeInstanceOf(Error); expect(await readLedger(testRoot)).toEqual([]); }); From 3c66a24483f291a840640830194889245e3fab73 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andreas=20Fr=C3=B8yland?= <81354124+Andreas-Froyland@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:21:55 +0200 Subject: [PATCH 3/4] fix(runner): classify a headless non-Linux host as no display, match Xvfb by display number, scan all processes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - On any platform that isn't Linux or Windows (in practice, macOS), no DISPLAY and no WAYLAND_DISPLAY at all is now "none", not "unknown": those variables aren't required by that platform's own GUI apps, so their presence is weak evidence at best, but their total absence must not grant a display capability the host cannot back up. - Matching a client's DISPLAY against the Xvfb/Xvnc/Xdummy process serving it now compares display numbers, ignoring any host prefix and any ".screen" suffix on either side: ":99.0" and ":99" name the same endpoint, and Xvfb's own argument never carries a screen number. - The scan of /proc for the serving virtual display server no longer stops after 5000 processes; missing the one process that matters would misclassify a virtual display as unknown or real. - Fixed a test that claimed to cover a rejecting display() probe but, because its fixture also provided describeDisplay(), only ever exercised that describeDisplay() rejecting — added the genuine case (no describeDisplay at all, display() itself rejects) and renamed the misleading one. - README and the tool-layout decision record now state the 22.18 floor and no-build guarantee accurately: no stale 22.12 mention, and no reference to a CLI entry path that does not exist yet. - Regenerated package-lock.json's engines metadata for the two workspace entries to match the raised floor (npm install; verified only those two lines changed). Co-Authored-By: Claude Sonnet 5 --- README.md | 2 +- docs/decisions/tool-layout.md | 2 +- package-lock.json | 4 ++-- packages/qa/src/runner/display.ts | 25 ++++++++++++++++++--- packages/qa/test/runner/display.test.ts | 25 +++++++++++++++++++-- packages/qa/test/runner/environment.test.ts | 9 +++++++- 6 files changed, 57 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index 7e03e1e..b9ca9a2 100644 --- a/README.md +++ b/README.md @@ -19,7 +19,7 @@ Stage 0 (proving the assumptions) is under way; the tool itself is not built yet ## Development -Requires Node.js 22.12 or newer. +Requires Node.js 22.18 or newer (the CLI needs no build step because Node strips types directly from that version on). ```sh npm ci diff --git a/docs/decisions/tool-layout.md b/docs/decisions/tool-layout.md index 64cb055..9b1466a 100644 --- a/docs/decisions/tool-layout.md +++ b/docs/decisions/tool-layout.md @@ -31,7 +31,7 @@ docs/decisions/ decision records `examples/tauri-smoke` stays outside the workspace on purpose. The packages tested in Stage 0 were built with its own `package-lock.json`, and hoisting its build tooling into a root lockfile would change that resolution without a re-test. Revisit when the sample is consumed by the runner's own tests. -**No build step (decided in Task 2.2).** Node 22.18 or newer runs the TypeScript directly by stripping types, so the CLI is `node packages/qa/src/cli/bin.ts`. That only works while the source uses *erasable* syntax (no enums, namespaces or constructor parameter properties), which `erasableSyntaxOnly` in `tsconfig.json` enforces at typecheck time, and `test/no-build.test.ts` loads the package under plain Node on every CI runner. `packages/qa/tsconfig.json` keeps `noEmit`. This suits running from a checkout; **a published or installed distribution would still need a build or bundle**, and `bin`, `main` and `exports` stay out of `package.json` until that is decided. +**No build step (decided in Task 2.2, part 1).** Node 22.18 or newer runs the TypeScript directly by stripping types, so a future CLI entry point can be run with a plain `node` invocation, needing no build step; that entry point does not exist yet (Task 2.2's next part adds `packages/qa/src/cli/`) and this record does not fix its path in advance. What is verified today is `packages/qa/src/index.ts`, which `test/no-build.test.ts` loads under plain Node on every CI runner and asserts prints no warning. This only works while the source uses *erasable* syntax (no enums, namespaces or constructor parameter properties), which `erasableSyntaxOnly` in `tsconfig.json` enforces at typecheck time. `packages/qa/tsconfig.json` keeps `noEmit`. This suits running from a checkout; **a published or installed distribution would still need a build or bundle**, and `bin`, `main` and `exports` stay out of `package.json` until that is decided. ## Stage 0 commands and where they go diff --git a/package-lock.json b/package-lock.json index a7f9044..0f2a68c 100644 --- a/package-lock.json +++ b/package-lock.json @@ -12,7 +12,7 @@ "apps/*" ], "engines": { - "node": ">=22.12.0" + "node": ">=22.18.0" } }, "apps/desktop": { @@ -1516,7 +1516,7 @@ "vitest": "5.0.1" }, "engines": { - "node": ">=22.12.0" + "node": ">=22.18.0" } } } diff --git a/packages/qa/src/runner/display.ts b/packages/qa/src/runner/display.ts index 053da53..3cf986d 100644 --- a/packages/qa/src/runner/display.ts +++ b/packages/qa/src/runner/display.ts @@ -30,22 +30,40 @@ export function classifyDisplay(facts: DisplayFacts): { kind: DisplayKind; detai const session = facts.env.SESSIONNAME ?? ''; return session !== '' && session !== 'Services' ? { kind: 'real', detail: `interactive session ${session}` } : { kind: 'none', detail: 'no interactive session' }; } - if (facts.platform !== 'linux') return { kind: 'unknown', detail: `cannot tell on ${facts.platform}` }; + if (facts.platform !== 'linux') { + // DISPLAY/WAYLAND_DISPLAY are not required by this platform's own native GUI apps, so their presence is only + // weak evidence and their absence is not proof either way — except that with no evidence at all, a headless + // host must not be handed a display capability it cannot back up. + const { DISPLAY: otherDisplay, WAYLAND_DISPLAY: otherWayland } = facts.env; + if (!otherDisplay && !otherWayland) return { kind: 'none', detail: `no display evidence on ${facts.platform}` }; + return { kind: 'unknown', detail: `cannot tell on ${facts.platform}` }; + } const { DISPLAY: display, WAYLAND_DISPLAY: wayland, XDG_SESSION_TYPE: session } = facts.env; if (!display && !wayland) return { kind: 'none', detail: 'neither DISPLAY nor WAYLAND_DISPLAY is set' }; if (display) { + const target = displayNumber(display); for (const line of facts.commandLines) { const [program = '', ...args] = line.trim().split(/\s+/); const name = basename(program); - if (VIRTUAL_SERVERS.has(name) && args.includes(display)) return { kind: 'virtual', detail: `${name} serving ${display}` }; + if (VIRTUAL_SERVERS.has(name) && target !== undefined && args.some((arg) => displayNumber(arg) === target)) return { kind: 'virtual', detail: `${name} serving ${display}` }; } } if ((session === 'x11' && display) || (session === 'wayland' && (wayland || display))) return { kind: 'real', detail: `${session} desktop session` }; return { kind: 'unknown', detail: `a display is set (${display ?? wayland}) but it is not evidently a desktop session or a virtual server` }; } +/** + * The `:N` display number a DISPLAY-like string names, ignoring any host prefix and any `.screen` suffix: Xvfb's own + * argument never carries a screen number (screens are configured separately, with `-screen`), but a client's + * DISPLAY may still name one explicitly, and ":99.0" is the same endpoint as ":99". + */ +function displayNumber(raw: string): string | undefined { + const match = /:(\d+)(?:\.\d+)?$/.exec(raw); + return match === null ? undefined : `:${match[1]}`; +} + /** Reads what this machine says. Best effort: anything unreadable is simply absent from the facts. */ export async function readDisplayFacts(): Promise { const platform = hostPlatform(); @@ -54,10 +72,11 @@ export async function readDisplayFacts(): Promise { async function linuxCommandLines(): Promise { const entries = await readdir('/proc').catch(() => [] as string[]); + // Every process is read, not just the first few thousand: missing the one X server that happens to be serving + // this display would misclassify a virtual display as unknown or real. const lines = await Promise.all( entries .filter((name) => /^\d+$/.test(name)) - .slice(0, 5000) .map((pid) => readFile(`/proc/${pid}/cmdline`, 'utf8').then((text) => text.split('\0').join(' ').trim(), () => '')), ); return lines.filter((line) => line !== ''); diff --git a/packages/qa/test/runner/display.test.ts b/packages/qa/test/runner/display.test.ts index 6baf1ac..c62feeb 100644 --- a/packages/qa/test/runner/display.test.ts +++ b/packages/qa/test/runner/display.test.ts @@ -31,6 +31,17 @@ describe('Linux', () => { expect(classifyDisplay(linux({ DISPLAY: ':99' }, ['Xvfb :98 -screen 0 1024x768x24'])).kind).toBe('unknown'); }); + test('an explicit screen number in DISPLAY still matches the Xvfb serving that display', () => { + // Xvfb's own argument never carries a screen number (screens are configured separately, with `-screen`); a + // client's DISPLAY may still name one explicitly (":99.0" is the same endpoint as ":99"). + expect(classifyDisplay(linux({ DISPLAY: ':99.0' }, ['Xvfb :99 -screen 0 1280x1024x24'])).kind).toBe('virtual'); + }); + + test('a screen number on either side of the comparison is ignored, but the display number must still match', () => { + expect(classifyDisplay(linux({ DISPLAY: ':99' }, ['Xvfb :99.0 -screen 0 1280x1024x24'])).kind).toBe('virtual'); + expect(classifyDisplay(linux({ DISPLAY: ':99.0' }, ['Xvfb :98.0 -screen 0 1280x1024x24'])).kind).toBe('unknown'); + }); + test('a desktop session is real', () => { expect(classifyDisplay(linux({ DISPLAY: ':0', XDG_SESSION_TYPE: 'x11' })).kind).toBe('real'); expect(classifyDisplay(linux({ WAYLAND_DISPLAY: 'wayland-0', XDG_SESSION_TYPE: 'wayland' })).kind).toBe('real'); @@ -69,8 +80,18 @@ describe('Windows', () => { }); }); -test('other platforms are reported as unknown rather than guessed', () => { - expect(classifyDisplay({ platform: 'darwin', env: {}, commandLines: [] }).kind).toBe('unknown'); +describe('other platforms', () => { + // DISPLAY/WAYLAND_DISPLAY are not required by this platform's own native GUI apps, so their presence or absence + // is weak evidence at best; a headless host here still reports none rather than being handed a display capability + // it cannot back up. + test('no display evidence at all is reported as none, not guessed as a working display', () => { + expect(classifyDisplay({ platform: 'darwin', env: {}, commandLines: [] }).kind).toBe('none'); + }); + + test('some display evidence, without a way to classify it further, is reported as unknown', () => { + expect(classifyDisplay({ platform: 'darwin', env: { DISPLAY: ':0' }, commandLines: [] }).kind).toBe('unknown'); + expect(classifyDisplay({ platform: 'darwin', env: { WAYLAND_DISPLAY: 'wayland-0' }, commandLines: [] }).kind).toBe('unknown'); + }); }); describe('reading this machine', () => { diff --git a/packages/qa/test/runner/environment.test.ts b/packages/qa/test/runner/environment.test.ts index b5eff03..0eeaae1 100644 --- a/packages/qa/test/runner/environment.test.ts +++ b/packages/qa/test/runner/environment.test.ts @@ -84,12 +84,19 @@ describe('inspectEnvironment', () => { expect(inspected.environment.capabilities).toEqual([]); }); - test('a display probe that throws is no display', async () => { + test('a display description that rejects is also no display', async () => { const broken = { display: async () => true, audio: async () => false, describeDisplay: async (): Promise => { throw new Error('boom'); } }; const inspected = await inspectEnvironment(hostProfile(), broken); expect(inspected.display.kind).toBe('none'); expect(inspected.environment.capabilities).toEqual([]); }); + + test('a display probe that rejects, with no describeDisplay to ask instead, is also no display', async () => { + const broken = { display: async (): Promise => { throw new Error('boom'); }, audio: async () => false }; + const inspected = await inspectEnvironment(hostProfile(), broken); + expect(inspected.display.kind).toBe('none'); + expect(inspected.environment.capabilities).toEqual([]); + }); }); test('the default probes answer with a boolean and never throw, whatever this machine has', async () => { From 3c4aad4b82c5e627b11e44014c9555352d7d2c1c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andreas=20Fr=C3=B8yland?= <81354124+Andreas-Froyland@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:35:04 +0200 Subject: [PATCH 4/4] fix(test): restore deadline-cutoff coverage for late-ownership tests, bound /proc read concurrency MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - The late-spawn/late-own tests were rewritten to wait for the hook to actually enter its phase and then let that phase's own 300 ms deadline (not an outer cancellation) cut it off. 300 ms, not the original 30 ms: prerequisites shares the same budget, measured at 1-4 ms idle, and 30 ms left too little margin for a loaded CI machine to finish prerequisites and enter install before the deadline fired — that margin, not the mechanism, was the actual cause of the Windows CI flakiness fixed a commit ago. Both tests now also assert the phase was reached and the result is a timeout, so the deadline path they are named for is genuinely exercised again. - linuxCommandLines reads /proc//cmdline in bounded batches of 256 instead of one unbounded Promise.all over every process, so a host with many thousands of processes cannot exhaust file descriptors while still being scanned in full. - Fixed a grammar error and an unprefixed path in the tool-layout decision record. Co-Authored-By: Claude Sonnet 5 --- docs/decisions/tool-layout.md | 2 +- packages/qa/src/runner/display.ts | 19 +++++++----- .../qa/test/runner/execute-hardening.test.ts | 30 +++++++++---------- 3 files changed, 28 insertions(+), 23 deletions(-) diff --git a/docs/decisions/tool-layout.md b/docs/decisions/tool-layout.md index 9b1466a..6daf16a 100644 --- a/docs/decisions/tool-layout.md +++ b/docs/decisions/tool-layout.md @@ -31,7 +31,7 @@ docs/decisions/ decision records `examples/tauri-smoke` stays outside the workspace on purpose. The packages tested in Stage 0 were built with its own `package-lock.json`, and hoisting its build tooling into a root lockfile would change that resolution without a re-test. Revisit when the sample is consumed by the runner's own tests. -**No build step (decided in Task 2.2, part 1).** Node 22.18 or newer runs the TypeScript directly by stripping types, so a future CLI entry point can be run with a plain `node` invocation, needing no build step; that entry point does not exist yet (Task 2.2's next part adds `packages/qa/src/cli/`) and this record does not fix its path in advance. What is verified today is `packages/qa/src/index.ts`, which `test/no-build.test.ts` loads under plain Node on every CI runner and asserts prints no warning. This only works while the source uses *erasable* syntax (no enums, namespaces or constructor parameter properties), which `erasableSyntaxOnly` in `tsconfig.json` enforces at typecheck time. `packages/qa/tsconfig.json` keeps `noEmit`. This suits running from a checkout; **a published or installed distribution would still need a build or bundle**, and `bin`, `main` and `exports` stay out of `package.json` until that is decided. +**No build step (decided in Task 2.2, part 1).** Node 22.18 or newer runs the TypeScript directly by stripping types, so a future CLI entry point can be run with a plain `node` invocation, needing no build step; that entry point does not exist yet (Task 2.2's next part adds `packages/qa/src/cli/`) and this record does not fix its path in advance. What is verified today is `packages/qa/src/index.ts`, which `packages/qa/test/no-build.test.ts` loads under plain Node on every CI runner and asserts that no warning is printed. This only works while the source uses *erasable* syntax (no enums, namespaces or constructor parameter properties), which `erasableSyntaxOnly` in `tsconfig.json` enforces at typecheck time. `packages/qa/tsconfig.json` keeps `noEmit`. This suits running from a checkout; **a published or installed distribution would still need a build or bundle**, and `bin`, `main` and `exports` stay out of `package.json` until that is decided. ## Stage 0 commands and where they go diff --git a/packages/qa/src/runner/display.ts b/packages/qa/src/runner/display.ts index 3cf986d..f6b979a 100644 --- a/packages/qa/src/runner/display.ts +++ b/packages/qa/src/runner/display.ts @@ -70,14 +70,19 @@ export async function readDisplayFacts(): Promise { return { platform, env: process.env, commandLines: platform === 'linux' ? await linuxCommandLines() : [] }; } +// Every process is read, not just the first few thousand: missing the one X server that happens to be serving this +// display would misclassify a virtual display as unknown or real. Read in bounded batches rather than opening every +// /proc//cmdline at once, so a host with many thousands of processes cannot exhaust file descriptors. +const CMDLINE_BATCH_SIZE = 256; + async function linuxCommandLines(): Promise { const entries = await readdir('/proc').catch(() => [] as string[]); - // Every process is read, not just the first few thousand: missing the one X server that happens to be serving - // this display would misclassify a virtual display as unknown or real. - const lines = await Promise.all( - entries - .filter((name) => /^\d+$/.test(name)) - .map((pid) => readFile(`/proc/${pid}/cmdline`, 'utf8').then((text) => text.split('\0').join(' ').trim(), () => '')), - ); + const pids = entries.filter((name) => /^\d+$/.test(name)); + const lines: string[] = []; + for (let start = 0; start < pids.length; start += CMDLINE_BATCH_SIZE) { + const batch = pids.slice(start, start + CMDLINE_BATCH_SIZE); + const read = await Promise.all(batch.map((pid) => readFile(`/proc/${pid}/cmdline`, 'utf8').then((text) => text.split('\0').join(' ').trim(), () => ''))); + lines.push(...read); + } return lines.filter((line) => line !== ''); } diff --git a/packages/qa/test/runner/execute-hardening.test.ts b/packages/qa/test/runner/execute-hardening.test.ts index 9b62145..c0d134a 100644 --- a/packages/qa/test/runner/execute-hardening.test.ts +++ b/packages/qa/test/runner/execute-hardening.test.ts @@ -175,15 +175,17 @@ describe('hooks that outlive their phase', () => { expect(result.cleanup.ok).toBe(true); }); - // These two wait for the phase to actually be cut off (the abort reaching the hook's own signal) before the hook - // acts, rather than racing a short phase deadline against a fixed sleep: a race like that assumes prerequisites and - // phase entry reliably finish in a few milliseconds, which does not hold on a loaded or slow CI machine. + // The hook waits for its own phase signal to abort, rather than for a fixed sleep to elapse, and the cutoff is the + // phase's own deadline (not an outer cancellation) so these still cover a hook actually outliving its phase. The + // deadline itself is 300 ms, not the original 30 ms: prerequisites shares the same budget (idle cost measured at + // 1-4 ms), and 30 ms left far too little margin for a loaded or slow CI machine to complete prerequisites and + // enter install before the deadline fired, which is exactly what made these two tests flaky on Windows CI. test('a late spawn from a hook that was cut off is refused and starts nothing', async () => { let refused: unknown; const calls: string[] = []; // The child would write this file if it were ever allowed to run; it lives in a scratch directory, never the repo. const markerFile = join(await makeTempDir('qa-late-'), 'late-spawn-marker'); - const { context, controller, testRoot } = await arrange({ + const { context, testRoot } = await arrange({ lifecycle: lifecycleOf(calls, { install: async (ctx) => { calls.push('install'); @@ -195,14 +197,13 @@ describe('hooks that outlive their phase', () => { } }, }), - timeouts: { phaseMs: 2000, stepsMs: 2000, cleanupMs: 2000, abandonedGraceMs: 400 }, + timeouts: { phaseMs: 300, stepsMs: 2000, cleanupMs: 2000, abandonedGraceMs: 400 }, }); - const running = executeScenario(context, scenarioOf()); - await eventually(() => calls.includes('install')); - controller.abort(); - await running; + const result = await executeScenario(context, scenarioOf()); + expect(calls).toContain('install'); // reached the phase whose deadline is under test, not cut off earlier + expect(result).toMatchObject({ outcome: 'interrupted', reason: 'timeout' }); expect(refused).toBeInstanceOf(Error); expect(String((refused as Error).message)).toMatch(/stopped|no longer/i); expect(await readLedger(testRoot)).toEqual([]); @@ -213,7 +214,7 @@ describe('hooks that outlive their phase', () => { let refused: unknown; let root = ''; const calls: string[] = []; - const { context, controller, testRoot } = await arrange({ + const { context, testRoot } = await arrange({ lifecycle: lifecycleOf(calls, { install: async (ctx) => { calls.push('install'); @@ -225,13 +226,12 @@ describe('hooks that outlive their phase', () => { } }, }), - timeouts: { phaseMs: 2000, stepsMs: 2000, cleanupMs: 2000, abandonedGraceMs: 400 }, + timeouts: { phaseMs: 300, stepsMs: 2000, cleanupMs: 2000, abandonedGraceMs: 400 }, }); root = testRoot; - const running = executeScenario(context, scenarioOf()); - await eventually(() => calls.includes('install')); - controller.abort(); - await running; + const result = await executeScenario(context, scenarioOf()); + expect(calls).toContain('install'); + expect(result).toMatchObject({ outcome: 'interrupted', reason: 'timeout' }); expect(refused).toBeInstanceOf(Error); expect(await readLedger(testRoot)).toEqual([]); });