From 1bfce92bc6c0a3182a91c8de3861782508acb853 Mon Sep 17 00:00:00 2001 From: macodev00 <273427913+macodev00@users.noreply.github.com> Date: Fri, 2 Oct 2026 06:41:25 +0000 Subject: [PATCH] fix(desktop): recover when WSL backend crashes during connect WSL-only mode kept the connecting splash up when the primary backend exited before it was ready, or stayed up and never answered. After three consecutive startup failures, show an error and use the in-memory Windows fallback for this launch. Fixes #14393 Co-authored-by: maco --- .../src/backend/DesktopBackendManager.test.ts | 461 +++++++++++ .../src/backend/DesktopBackendManager.ts | 729 +++++++++++------- .../src/backend/DesktopBackendPool.test.ts | 215 +++++- .../desktop/src/backend/DesktopBackendPool.ts | 28 + 4 files changed, 1172 insertions(+), 261 deletions(-) diff --git a/apps/desktop/src/backend/DesktopBackendManager.test.ts b/apps/desktop/src/backend/DesktopBackendManager.test.ts index 901d9f4a2708..a43cd6a0da37 100644 --- a/apps/desktop/src/backend/DesktopBackendManager.test.ts +++ b/apps/desktop/src/backend/DesktopBackendManager.test.ts @@ -126,6 +126,11 @@ interface MakeInstanceInput { readonly onPreflightFailed?: ( failure: DesktopBackendManager.PreflightFailure, ) => Effect.Effect; + readonly onStartupFailed?: ( + reason: string, + config: DesktopBackendManager.DesktopBackendStartConfig, + ) => Effect.Effect; + readonly readinessTimeout?: Duration.Duration; readonly config?: DesktopBackendManager.DesktopBackendStartConfig; readonly configResolve?: Effect.Effect< DesktopBackendManager.DesktopBackendStartConfig, @@ -185,6 +190,8 @@ function makeTestInstance(input: MakeInstanceInput) { ...(input.onReady ? { onReady: () => input.onReady! } : {}), ...(input.onShutdown ? { onShutdown: () => input.onShutdown! } : {}), ...(input.onPreflightFailed ? { onPreflightFailed: input.onPreflightFailed } : {}), + ...(input.onStartupFailed ? { onStartupFailed: input.onStartupFailed } : {}), + ...(input.readinessTimeout === undefined ? {} : { readinessTimeout: input.readinessTimeout }), }); return instance.pipe(Effect.provide(servicesLayer)); @@ -1399,6 +1406,460 @@ describe("DesktopBackendManager", () => { ), ); + it.effect( + "replaces a live run that never becomes reachable once startup failures hit the cap", + () => + Effect.scoped( + Effect.gen(function* () { + // wsl-only: the process stays up, but Windows cannot reach it (the + // iptables case). The hook swaps the resolved config to loopback. + const wslConfig: DesktopBackendManager.DesktopBackendStartConfig = { + ...baseConfig, + httpBaseUrl: new URL("http://172.17.0.1:3773"), + runningDistro: "Ubuntu-22.04", + }; + const useFallback = yield* Ref.make(false); + const startupFailures: string[] = []; + const spawnedUrls = yield* Queue.unbounded(); + const ready = yield* Deferred.make(); + let currentUrl = ""; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.gen(function* () { + const scope = yield* Scope.Scope; + const exited = yield* Deferred.make(); + yield* Queue.offer(spawnedUrls, currentUrl); + yield* Scope.addFinalizer(scope, Deferred.succeed(exited, void 0)); + return makeProcess({ + exitCode: Deferred.await(exited).pipe(Effect.as(ChildProcessSpawner.ExitCode(0))), + kill: () => Deferred.succeed(exited, void 0).pipe(Effect.asVoid), + }); + }), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + readinessTimeout: Duration.seconds(1), + configResolve: Ref.get(useFallback).pipe( + Effect.map((fallback) => (fallback ? baseConfig : wslConfig)), + Effect.tap((config) => + Effect.sync(() => { + currentUrl = config.httpBaseUrl.href; + }), + ), + ), + httpClientLayer: httpClientLayer((request) => + Effect.succeed( + responseForRequest(request, request.url.startsWith("http://127.0.0.1") ? 200 : 503), + ), + ), + onReady: Deferred.succeed(ready, void 0).pipe(Effect.asVoid), + onStartupFailed: (reason) => + Effect.sync(() => { + startupFailures.push(reason); + }).pipe(Effect.andThen(Ref.set(useFallback, true)), Effect.as(true)), + }); + + yield* instance.start; + assert.equal(yield* Queue.take(spawnedUrls), "http://172.17.0.1:3773/"); + + // Two unreachable readiness rounds are tolerated (slow WSL cold boot). + yield* TestClock.adjust(Duration.seconds(2)); + assert.deepEqual(startupFailures, []); + assert.equal(yield* Queue.size(spawnedUrls), 0); + + // The third round surfaces the failure once and swaps the run. + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(yield* Queue.take(spawnedUrls), "http://127.0.0.1:3773/"); + yield* Deferred.await(ready); + assert.equal(startupFailures.length, 1); + assert.equal((yield* instance.snapshot).ready, true); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("stopping the backend cancels a pending startup failure hook", () => + Effect.scoped( + Effect.gen(function* () { + const hookEntered = yield* Deferred.make(); + const releaseHook = yield* Deferred.make(); + let fallbackApplied = false; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.gen(function* () { + const scope = yield* Scope.Scope; + const exited = yield* Deferred.make(); + yield* Scope.addFinalizer(scope, Deferred.succeed(exited, void 0)); + return makeProcess({ + exitCode: Deferred.await(exited).pipe(Effect.as(ChildProcessSpawner.ExitCode(0))), + kill: () => Deferred.succeed(exited, void 0).pipe(Effect.asVoid), + }); + }), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + readinessTimeout: Duration.seconds(1), + httpClientLayer: httpClientLayer((request) => + Effect.succeed(responseForRequest(request, 503)), + ), + onStartupFailed: () => + Deferred.succeed(hookEntered, void 0).pipe( + Effect.andThen(Deferred.await(releaseHook)), + Effect.andThen( + Effect.sync(() => { + fallbackApplied = true; + }), + ), + Effect.as(true), + ), + }); + + yield* instance.start; + yield* TestClock.adjust(Duration.seconds(3)); + yield* Deferred.await(hookEntered); + + yield* instance.stop(); + yield* Deferred.succeed(releaseHook, void 0); + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(fallbackApplied, false); + assert.equal((yield* instance.snapshot).desiredRunning, false); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("keeps a backend that became ready while the startup failure hook was pending", () => + Effect.scoped( + Effect.gen(function* () { + const hookEntered = yield* Deferred.make(); + const releaseHook = yield* Deferred.make(); + const ready = yield* Deferred.make(); + let spawnCount = 0; + let reachable = false; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.gen(function* () { + spawnCount += 1; + const scope = yield* Scope.Scope; + const exited = yield* Deferred.make(); + yield* Scope.addFinalizer(scope, Deferred.succeed(exited, void 0)); + return makeProcess({ + exitCode: Deferred.await(exited).pipe(Effect.as(ChildProcessSpawner.ExitCode(0))), + kill: () => Deferred.succeed(exited, void 0).pipe(Effect.asVoid), + }); + }), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + readinessTimeout: Duration.seconds(1), + httpClientLayer: httpClientLayer((request) => + Effect.sync(() => responseForRequest(request, reachable ? 200 : 503)), + ), + onReady: Deferred.succeed(ready, void 0).pipe(Effect.asVoid), + onStartupFailed: () => + Deferred.succeed(hookEntered, void 0).pipe( + Effect.andThen(Deferred.await(releaseHook)), + Effect.as(true), + ), + }); + + yield* instance.start; + yield* TestClock.adjust(Duration.seconds(3)); + yield* Deferred.await(hookEntered); + + reachable = true; + yield* TestClock.adjust(Duration.seconds(1)); + yield* Deferred.await(ready); + + yield* Deferred.succeed(releaseHook, void 0); + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(spawnCount, 1); + assert.equal((yield* instance.snapshot).ready, true); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("stop is not held up by a pending startup failure hook after pre-ready exits", () => + Effect.scoped( + Effect.gen(function* () { + const hookEntered = yield* Deferred.make(); + const releaseHook = yield* Deferred.make(); + const exits = yield* Queue.unbounded(); + let fallbackApplied = false; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.succeed( + makeProcess({ + exitCode: Queue.offer(exits, void 0).pipe( + Effect.as(ChildProcessSpawner.ExitCode(1)), + ), + }), + ), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + httpClientLayer: httpClientLayer(() => Effect.never), + onStartupFailed: () => + Deferred.succeed(hookEntered, void 0).pipe( + Effect.andThen(Deferred.await(releaseHook)), + Effect.andThen( + Effect.sync(() => { + fallbackApplied = true; + }), + ), + Effect.as(true), + ), + }); + + yield* instance.start; + // Pre-ready exits restart after 0.5s and 1s; the third exit hits the cap. + yield* Queue.take(exits); + yield* TestClock.adjust(Duration.millis(500)); + yield* Queue.take(exits); + yield* TestClock.adjust(Duration.seconds(1)); + yield* Deferred.await(hookEntered); + + yield* instance.stop(); + yield* Deferred.succeed(releaseHook, void 0); + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(fallbackApplied, false); + assert.equal((yield* instance.snapshot).desiredRunning, false); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("does not restart a backend that was stopped while being replaced", () => + Effect.scoped( + Effect.gen(function* () { + const closing = yield* Queue.unbounded(); + const releaseClose = yield* Deferred.make(); + let spawnCount = 0; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.gen(function* () { + spawnCount += 1; + const scope = yield* Scope.Scope; + const exited = yield* Deferred.make(); + // Closing the run blocks until the test lets it finish. + yield* Scope.addFinalizer( + scope, + Queue.offer(closing, void 0).pipe( + Effect.andThen(Deferred.await(releaseClose)), + Effect.andThen(Deferred.succeed(exited, void 0)), + ), + ); + return makeProcess({ + exitCode: Deferred.await(exited).pipe(Effect.as(ChildProcessSpawner.ExitCode(0))), + }); + }), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + readinessTimeout: Duration.seconds(1), + httpClientLayer: httpClientLayer((request) => + Effect.succeed(responseForRequest(request, 503)), + ), + onStartupFailed: () => Effect.succeed(true), + }); + + yield* instance.start; + yield* TestClock.adjust(Duration.seconds(3)); + // The replacement is closing the old run; the user stops the backend now. + yield* Queue.take(closing); + const userStop = yield* Effect.forkChild(instance.stop()); + yield* TestClock.adjust(Duration.millis(1)); + yield* Deferred.succeed(releaseClose, void 0); + yield* Fiber.join(userStop); + yield* TestClock.adjust(Duration.seconds(1)); + + assert.equal(spawnCount, 1); + assert.equal((yield* instance.snapshot).desiredRunning, false); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("surfaces repeated exits before readiness and keeps retrying when declined", () => + Effect.scoped( + Effect.gen(function* () { + const starts = yield* Queue.unbounded(); + const startupFailures = yield* Queue.unbounded(); + let startCount = 0; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.sync(() => { + startCount += 1; + return makeProcess({ + exitCode: Queue.offer(starts, startCount).pipe( + Effect.as(ChildProcessSpawner.ExitCode(1)), + ), + }); + }), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + httpClientLayer: httpClientLayer(() => Effect.never), + onStartupFailed: (reason) => Queue.offer(startupFailures, reason).pipe(Effect.as(false)), + }); + + yield* instance.start; + assert.equal(yield* Queue.take(starts), 1); + yield* TestClock.adjust(Duration.millis(500)); + assert.equal(yield* Queue.take(starts), 2); + assert.equal(yield* Queue.size(startupFailures), 0); + + // The third exit before the backend ever became ready hits the cap. + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(yield* Queue.take(starts), 3); + yield* Queue.take(startupFailures); + + // Declined (a Windows primary): the restart loop carries on. + yield* TestClock.adjust(Duration.seconds(2)); + assert.equal(yield* Queue.take(starts), 4); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("waits for the startup failure hook before restarting a crashed run", () => + Effect.scoped( + Effect.gen(function* () { + const wslConfig: DesktopBackendManager.DesktopBackendStartConfig = { + ...baseConfig, + httpBaseUrl: new URL("http://172.17.0.1:3773"), + runningDistro: "Ubuntu-22.04", + }; + const useFallback = yield* Ref.make(false); + const hookEntered = yield* Deferred.make(); + const releaseHook = yield* Deferred.make(); + const spawnedUrls = yield* Queue.unbounded(); + const ready = yield* Deferred.make(); + let currentUrl = ""; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.sync(() => + makeProcess({ + exitCode: Effect.sync(() => currentUrl).pipe( + Effect.tap((url) => Queue.offer(spawnedUrls, url)), + Effect.flatMap((url) => + url.startsWith("http://127.0.0.1") + ? Effect.never + : Effect.succeed(ChildProcessSpawner.ExitCode(1)), + ), + ), + }), + ), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + configResolve: Ref.get(useFallback).pipe( + Effect.map((fallback) => (fallback ? baseConfig : wslConfig)), + Effect.tap((config) => + Effect.sync(() => { + currentUrl = config.httpBaseUrl.href; + }), + ), + ), + httpClientLayer: httpClientLayer((request) => + Effect.succeed( + responseForRequest(request, request.url.startsWith("http://127.0.0.1") ? 200 : 503), + ), + ), + onReady: Deferred.succeed(ready, void 0).pipe(Effect.asVoid), + onStartupFailed: () => + Deferred.succeed(hookEntered, void 0).pipe( + Effect.andThen(Deferred.await(releaseHook)), + Effect.andThen(Ref.set(useFallback, true)), + Effect.as(true), + ), + }); + + yield* instance.start; + assert.equal(yield* Queue.take(spawnedUrls), "http://172.17.0.1:3773/"); + yield* TestClock.adjust(Duration.millis(500)); + assert.equal(yield* Queue.take(spawnedUrls), "http://172.17.0.1:3773/"); + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(yield* Queue.take(spawnedUrls), "http://172.17.0.1:3773/"); + yield* Deferred.await(hookEntered); + + // The crashed run must not be replaced until the hook applies the fallback. + yield* TestClock.adjust(Duration.seconds(30)); + assert.equal(yield* Queue.size(spawnedUrls), 0); + + yield* Deferred.succeed(releaseHook, void 0); + yield* TestClock.adjust(Duration.seconds(2)); + assert.equal(yield* Queue.take(spawnedUrls), "http://127.0.0.1:3773/"); + yield* Deferred.await(ready); + assert.equal((yield* instance.snapshot).ready, true); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("does not surface exits that happen after the backend became ready", () => + Effect.scoped( + Effect.gen(function* () { + const readied = yield* Queue.unbounded(); + const exits = yield* Queue.unbounded(); + let startupFailures = 0; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.succeed( + makeProcess({ + // Each run crashes only after it has reported ready. + exitCode: Queue.take(readied).pipe(Effect.as(ChildProcessSpawner.ExitCode(1))), + }), + ), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + onReady: Queue.offer(readied, void 0).pipe(Effect.asVoid), + onStartupFailed: () => + Effect.sync(() => { + startupFailures += 1; + }).pipe(Effect.as(false)), + backendOutputLog: { + persistFailure: ({ details }) => Queue.offer(exits, details).pipe(Effect.asVoid), + }, + }); + + yield* instance.start; + for (let i = 0; i < 5; i++) { + yield* Queue.take(exits); + yield* TestClock.adjust(Duration.seconds(1)); + } + assert.equal(startupFailures, 0); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + it.effect("cancels a scheduled restart when start is requested manually", () => Effect.scoped( Effect.gen(function* () { diff --git a/apps/desktop/src/backend/DesktopBackendManager.ts b/apps/desktop/src/backend/DesktopBackendManager.ts index 6f1ea139bdca..548c43345a3b 100644 --- a/apps/desktop/src/backend/DesktopBackendManager.ts +++ b/apps/desktop/src/backend/DesktopBackendManager.ts @@ -61,6 +61,10 @@ const MAX_RESTART_DELAY = Duration.seconds(10); // failures may instead provide their own larger retryLimit when they should // self-heal for a while but must not leave the app connecting forever. const MAX_PREFLIGHT_FAILURE_ATTEMPTS = 5; +// Consecutive readiness timeouts or exits before the backend was ever ready. +// Three one-minute readiness rounds still leave room for a slow WSL cold boot. +// Past that, wsl-only mode can leave the splash and fall back for this launch. +const MAX_STARTUP_FAILURE_ATTEMPTS = 3; const DEFAULT_BACKEND_READINESS_TIMEOUT = Duration.minutes(1); const DEFAULT_BACKEND_READINESS_INTERVAL = Duration.millis(100); const DEFAULT_BACKEND_READINESS_REQUEST_TIMEOUT = Duration.seconds(1); @@ -299,6 +303,20 @@ export interface BackendInstanceSpec { // retries. Returns true when the callback changed configuration and the // manager should resolve once more; false stops the failed instance. readonly onPreflightFailed?: (failure: PreflightFailure) => Effect.Effect; + // Fired when MAX_STARTUP_FAILURE_ATTEMPTS consecutive startup failures + // accumulate after a clean preflight (readiness budgets that time out, or + // exits before the run was ever ready). `config` is the run that failed. + // Return true when configuration changed and the current run should be + // replaced; false leaves the normal restart and readiness loops in place. + // Exits after a successful ready do not count, so a backend that was healthy + // and later crashes keeps the existing restart loop. + readonly onStartupFailed?: ( + reason: string, + config: DesktopBackendStartConfig, + ) => Effect.Effect; + // Overrides the per-round readiness budget. Production uses the default; + // tests pass a short duration so the cap can be reached under TestClock. + readonly readinessTimeout?: Duration.Duration; } interface ActiveBackendRun { @@ -319,6 +337,18 @@ interface BackendManagerState { // Consecutive bounded/fatal preflight failures, reset on a clean or // unbounded-transient preflight. restartAttempt counts all restarts. readonly preflightFailureAttempt: number; + // Consecutive readiness timeouts and pre-ready exits, reset once ready. + readonly startupFailureAttempt: number; + // True from the moment the cap is claimed until the hook's follow-up + // finishes, so a later timeout cannot start a second hook. + readonly startupFailurePending: boolean; + // The in-flight onStartupFailed hook. stop() interrupts it so a blocked + // dialog cannot hold shutdown. The follow-up that replaces the run is a + // different fiber and is not cancelled here. + readonly startupFailureFiber: Option.Option>; + // Bumped by every stop(), so a replacement can tell an explicit stop + // happened while it was closing the failed run. + readonly stopGeneration: number; readonly restartFiber: Option.Option>; readonly nextRunId: number; } @@ -330,6 +360,10 @@ const initialState: BackendManagerState = { active: Option.none(), restartAttempt: 0, preflightFailureAttempt: 0, + startupFailureAttempt: 0, + startupFailurePending: false, + startupFailureFiber: Option.none(), + stopGeneration: 0, restartFiber: Option.none(), nextRunId: 1, }; @@ -688,302 +722,467 @@ export const makeBackendInstance = Effect.fn("makeBackendInstance")(function* ( }); }); - const start: Effect.Effect = Effect.suspend(() => - mutex.withPermits(1)( - Effect.gen(function* () { - const current = yield* Ref.get(state); - if (Option.isSome(current.active)) { - if (!current.desiredRunning) { - yield* Ref.update(state, (latest) => ({ - ...latest, - desiredRunning: true, - })); + const beginStart = (options?: { + readonly expectedStopGeneration?: number; + }): Effect.Effect => + Effect.suspend(() => + mutex.withPermits(1)( + Effect.gen(function* () { + const current = yield* Ref.get(state); + // Automatic recovery checks the generation under the same lock that + // sets desiredRunning, so a stop that lands after replaceRun's own + // stop() cannot be overwritten by the replacement start. + if ( + options?.expectedStopGeneration !== undefined && + current.stopGeneration !== options.expectedStopGeneration + ) { + return; } - return; - } - - if (current.ready) { - yield* spec.onShutdown?.() ?? Effect.void; - yield* Ref.update(state, (latest) => - latest.ready ? { ...latest, ready: false } : latest, - ); - } - const config = yield* spec.configResolve.pipe( - Effect.tapError((error) => - logInstanceError("failed to generate desktop backend configuration", { - cause: error.message, - }), - ), - Effect.option, - ); - if (Option.isNone(config)) { - if (current.desiredRunning) { - yield* scheduleRestart("failed to generate desktop backend configuration"); + if (Option.isSome(current.active)) { + if (!current.desiredRunning) { + yield* Ref.update(state, (latest) => ({ + ...latest, + desiredRunning: true, + })); + } + return; } - return; - } - const entryExists = yield* fileSystem - .exists(config.value.entryPath) - .pipe(Effect.orElseSucceed(() => false)); - - const resetFatalPreflightCounter = - !current.desiredRunning && current.preflightFailureAttempt > 0; - yield* cancelRestart; - yield* Ref.update(state, (latest) => ({ - ...latest, - desiredRunning: true, - ready: false, - config: Option.some(config.value), - preflightFailureAttempt: resetFatalPreflightCounter ? 0 : latest.preflightFailureAttempt, - })); - - const preflightFailure = config.value.preflightFailure; - if (Option.isSome(preflightFailure)) { - const { reason, fatal, retryLimit } = preflightFailure.value; - if (!fatal && retryLimit === undefined) { - // Transient (WSL cold-starting, wslpath while the VM boots). Keep - // retrying so the backend self-heals once WSL is ready. Reset a - // prior bounded/fatal streak because this is a different failure. + + if (current.ready) { + yield* spec.onShutdown?.() ?? Effect.void; yield* Ref.update(state, (latest) => - latest.preflightFailureAttempt === 0 - ? latest - : { ...latest, preflightFailureAttempt: 0 }, + latest.ready ? { ...latest, ready: false } : latest, ); - yield* scheduleRestart(reason); - return; } - const attemptLimit = retryLimit ?? MAX_PREFLIGHT_FAILURE_ATTEMPTS; - const attempt = yield* Ref.modify(state, (latest) => { - const next = latest.preflightFailureAttempt + 1; - return [next, { ...latest, preflightFailureAttempt: next }] as const; - }); - if (attempt > attemptLimit) { - // We already surfaced and asked for the Windows fallback, yet we're - // still resolving the WSL primary — the fallback didn't take (e.g. - // the settings write failed). Stop rather than loop forever. - yield* logInstanceError("backend preflight still failing after fallback; stopping", { - reason, - attempt, - }); - yield* Ref.update(state, (latest) => ({ - ...latest, - desiredRunning: false, - ready: false, - })); + const config = yield* spec.configResolve.pipe( + Effect.tapError((error) => + logInstanceError("failed to generate desktop backend configuration", { + cause: error.message, + }), + ), + Effect.option, + ); + if (Option.isNone(config)) { + if (current.desiredRunning) { + yield* scheduleRestart("failed to generate desktop backend configuration"); + } return; } - if (attempt === attemptLimit) { - // Fatal/bounded and out of retries. Surface the reason (onPreflightFailed, - // on the primary, shows a dialog and persists Windows mode), then - // schedule one more restart so the next resolve picks up the Windows - // primary and a window can open. - yield* logInstanceError( - "backend preflight failed repeatedly; surfacing and falling back", - { reason, attempt }, - ); - const shouldRestart = yield* ( - spec.onPreflightFailed?.(preflightFailure.value) ?? Effect.succeed(false) - ); - if (shouldRestart) { + const entryExists = yield* fileSystem + .exists(config.value.entryPath) + .pipe(Effect.orElseSucceed(() => false)); + + const resetFatalPreflightCounter = + !current.desiredRunning && current.preflightFailureAttempt > 0; + yield* cancelRestart; + yield* Ref.update(state, (latest) => ({ + ...latest, + desiredRunning: true, + ready: false, + config: Option.some(config.value), + preflightFailureAttempt: resetFatalPreflightCounter + ? 0 + : latest.preflightFailureAttempt, + // A user-driven start gets a fresh startup budget. Restart-loop + // starts keep the streak so crashes can reach the cap. + startupFailureAttempt: current.desiredRunning ? latest.startupFailureAttempt : 0, + })); + + const preflightFailure = config.value.preflightFailure; + if (Option.isSome(preflightFailure)) { + const { reason, fatal, retryLimit } = preflightFailure.value; + if (!fatal && retryLimit === undefined) { + // Transient (WSL cold-starting, wslpath while the VM boots). Keep + // retrying so the backend self-heals once WSL is ready. Reset a + // prior bounded/fatal streak because this is a different failure. + yield* Ref.update(state, (latest) => + latest.preflightFailureAttempt === 0 + ? latest + : { ...latest, preflightFailureAttempt: 0 }, + ); yield* scheduleRestart(reason); - } else { + return; + } + const attemptLimit = retryLimit ?? MAX_PREFLIGHT_FAILURE_ATTEMPTS; + const attempt = yield* Ref.modify(state, (latest) => { + const next = latest.preflightFailureAttempt + 1; + return [next, { ...latest, preflightFailureAttempt: next }] as const; + }); + if (attempt > attemptLimit) { + // We already surfaced and asked for the Windows fallback, yet we're + // still resolving the WSL primary — the fallback didn't take (e.g. + // the settings write failed). Stop rather than loop forever. + yield* logInstanceError("backend preflight still failing after fallback; stopping", { + reason, + attempt, + }); yield* Ref.update(state, (latest) => ({ ...latest, desiredRunning: false, ready: false, })); + return; + } + if (attempt === attemptLimit) { + // Fatal/bounded and out of retries. Surface the reason (onPreflightFailed, + // on the primary, shows a dialog and persists Windows mode), then + // schedule one more restart so the next resolve picks up the Windows + // primary and a window can open. + yield* logInstanceError( + "backend preflight failed repeatedly; surfacing and falling back", + { reason, attempt }, + ); + const shouldRestart = yield* ( + spec.onPreflightFailed?.(preflightFailure.value) ?? Effect.succeed(false) + ); + if (shouldRestart) { + yield* scheduleRestart(reason); + } else { + yield* Ref.update(state, (latest) => ({ + ...latest, + desiredRunning: false, + ready: false, + })); + } + return; } + yield* scheduleRestart(reason); return; } - yield* scheduleRestart(reason); - return; - } - // Clean preflight — reset the fatal counter so a later failure gets a - // fresh allowance. - yield* Ref.update(state, (latest) => - latest.preflightFailureAttempt === 0 ? latest : { ...latest, preflightFailureAttempt: 0 }, - ); + // Clean preflight — reset the fatal counter so a later failure gets a + // fresh allowance. + yield* Ref.update(state, (latest) => + latest.preflightFailureAttempt === 0 + ? latest + : { ...latest, preflightFailureAttempt: 0 }, + ); - if (!entryExists) { - yield* scheduleRestart(`missing server entry at ${config.value.entryPath}`); - return; - } + if (!entryExists) { + yield* scheduleRestart(`missing server entry at ${config.value.entryPath}`); + return; + } - const runScope = yield* Scope.make("sequential"); - const runId = yield* Ref.modify(state, (latest) => [ - latest.nextRunId, - { - ...latest, - active: Option.some({ - id: latest.nextRunId, - scope: runScope, - fiber: Option.none(), - pid: Option.none(), - exitObserved: false, - stopRequested: false, - } satisfies ActiveBackendRun), - nextRunId: latest.nextRunId + 1, - }, - ]); - - const finalizeRun = Effect.fn("desktop.backendInstance.finalizeRun")(function* ( - reason: string, - ) { - yield* mutex.withPermits(1)( - Effect.gen(function* () { - const { isCurrentRun, nextState, pid, exitObserved, stopRequested, wasReady } = - yield* Ref.modify( - state, - ( - latest, - ): readonly [ - { - readonly isCurrentRun: boolean; - readonly nextState: BackendManagerState; - readonly pid: Option.Option; - readonly exitObserved: boolean; - readonly stopRequested: boolean; - readonly wasReady: boolean; - }, - BackendManagerState, - ] => { - const currentRun = Option.getOrUndefined(latest.active); - if (currentRun?.id !== runId) { + const runScope = yield* Scope.make("sequential"); + const runId = yield* Ref.modify(state, (latest) => [ + latest.nextRunId, + { + ...latest, + active: Option.some({ + id: latest.nextRunId, + scope: runScope, + fiber: Option.none(), + pid: Option.none(), + exitObserved: false, + stopRequested: false, + } satisfies ActiveBackendRun), + nextRunId: latest.nextRunId + 1, + }, + ]); + + const finalizeRun = Effect.fn("desktop.backendInstance.finalizeRun")(function* ( + reason: string, + ) { + yield* mutex.withPermits(1)( + Effect.gen(function* () { + const { isCurrentRun, nextState, pid, exitObserved, stopRequested, wasReady } = + yield* Ref.modify( + state, + ( + latest, + ): readonly [ + { + readonly isCurrentRun: boolean; + readonly nextState: BackendManagerState; + readonly pid: Option.Option; + readonly exitObserved: boolean; + readonly stopRequested: boolean; + readonly wasReady: boolean; + }, + BackendManagerState, + ] => { + const currentRun = Option.getOrUndefined(latest.active); + if (currentRun?.id !== runId) { + return [ + { + isCurrentRun: false, + nextState: latest, + pid: Option.none(), + exitObserved: false, + stopRequested: false, + wasReady: false, + }, + latest, + ] as const; + } + + const next = { + ...latest, + active: Option.none(), + ready: false, + }; return [ { - isCurrentRun: false, - nextState: latest, - pid: Option.none(), - exitObserved: false, - stopRequested: false, - wasReady: false, + isCurrentRun: true, + nextState: next, + pid: currentRun.pid, + exitObserved: currentRun.exitObserved, + stopRequested: currentRun.stopRequested, + wasReady: latest.ready, }, - latest, + next, ] as const; + }, + ); + + if (isCurrentRun) { + yield* desktopTelemetryPublisher.removeControlSource(spec.id); + if (Option.isSome(pid)) { + if (exitObserved && !stopRequested) { + yield* backendOutputLog.persistFailure({ + details: `pid=${pid.value} ${reason}`, + }); + } else { + yield* backendOutputLog.discardSession; } + } + if (wasReady) { + yield* spec.onShutdown?.() ?? Effect.void; + } + } - const next = { - ...latest, - active: Option.none(), - ready: false, - }; - return [ - { - isCurrentRun: true, - nextState: next, - pid: currentRun.pid, - exitObserved: currentRun.exitObserved, - stopRequested: currentRun.stopRequested, - wasReady: latest.ready, - }, - next, - ] as const; - }, - ); - - if (isCurrentRun) { - yield* desktopTelemetryPublisher.removeControlSource(spec.id); - if (Option.isSome(pid)) { - if (exitObserved && !stopRequested) { - yield* backendOutputLog.persistFailure({ - details: `pid=${pid.value} ${reason}`, - }); - } else { - yield* backendOutputLog.discardSession; + // The process is already gone. When the startup cap is hit, the + // hook owns the next restart so a fallback it applies is visible + // to configResolve. Scheduling here would race that write. + let startupFailureClaimed = false; + if (isCurrentRun && exitObserved && !stopRequested && !wasReady) { + startupFailureClaimed = yield* claimStartupFailure; + if (startupFailureClaimed) { + yield* armStartupFailure(reason, config.value, Option.none()); } } - if (wasReady) { - yield* spec.onShutdown?.() ?? Effect.void; + + if (isCurrentRun && nextState.desiredRunning && !startupFailureClaimed) { + yield* scheduleRestart(reason); + } + }), + ); + }); + + const program = runBackendProcess({ + ...config.value, + ...(spec.readinessTimeout === undefined + ? {} + : { readinessTimeout: spec.readinessTimeout }), + desktopTelemetryStream: desktopTelemetryPublisher.encoded, + onDesktopTelemetryControl: (message) => + desktopTelemetryPublisher.handleControlForSource(spec.id, message), + onStarted: Effect.fn("desktop.backendInstance.onStarted")(function* (pid) { + yield* updateActiveRun(runId, (run) => ({ + ...run, + pid: Option.some(pid), + })); + yield* backendOutputLog.beginSession({ + details: `pid=${pid} port=${config.value.bootstrap.port} cwd=${config.value.cwd}`, + }); + }), + onExitObserved: () => + updateActiveRun(runId, (run) => ({ + ...run, + exitObserved: true, + })), + onReady: Effect.fn("desktop.backendInstance.onReady")(function* () { + const isCurrentRun = yield* Ref.modify(state, (latest) => { + const activeRun = Option.getOrUndefined(latest.active); + if (activeRun?.id !== runId) { + return [false, latest] as const; } + + return [ + true, + { + ...latest, + restartAttempt: 0, + startupFailureAttempt: 0, + ready: true, + }, + ] as const; + }); + if (!isCurrentRun) { + return; } - if (isCurrentRun && nextState.desiredRunning) { - yield* scheduleRestart(reason); + yield* spec.onReady?.(config.value.httpBaseUrl) ?? Effect.void; + if ( + config.value.runningDistro !== undefined && + config.value.wslRuntimeId !== undefined + ) { + yield* wslEnvironment.pruneRuntimes( + config.value.runningDistro, + config.value.wslRuntimeId, + ); } }), + onReadinessFailure: Effect.fn("desktop.backendInstance.onReadinessFailure")( + function* (error) { + yield* logInstanceWarning("backend readiness check failed during bootstrap", { + error: error.message, + }); + yield* backendOutputLog.persistFailureSnapshot({ + details: error.message, + }); + yield* mutex.withPermits(1)( + Effect.gen(function* () { + const current = yield* Ref.get(state); + if (Option.getOrUndefined(current.active)?.id !== runId) return; + if (!current.desiredRunning || current.ready) return; + if (!(yield* claimStartupFailure)) return; + yield* armStartupFailure(error.message, config.value, Option.some(runId)); + }), + ); + }, + ), + onOutput: (streamName, chunk) => backendOutputLog.writeOutputChunk(streamName, chunk), + }).pipe( + Effect.provideService(ChildProcessSpawner.ChildProcessSpawner, spawner), + Effect.provideService(HttpClient.HttpClient, httpClient), + Scope.provide(runScope), + Effect.matchEffect({ + onFailure: (error) => finalizeRun(error.message), + onSuccess: (exit) => finalizeRun(exit.reason), + }), + Effect.ensuring(Scope.close(runScope, Exit.void).pipe(Effect.ignore)), ); - }); - const program = runBackendProcess({ - ...config.value, - desktopTelemetryStream: desktopTelemetryPublisher.encoded, - onDesktopTelemetryControl: (message) => - desktopTelemetryPublisher.handleControlForSource(spec.id, message), - onStarted: Effect.fn("desktop.backendInstance.onStarted")(function* (pid) { - yield* updateActiveRun(runId, (run) => ({ - ...run, - pid: Option.some(pid), - })); - yield* backendOutputLog.beginSession({ - details: `pid=${pid} port=${config.value.bootstrap.port} cwd=${config.value.cwd}`, - }); - }), - onExitObserved: () => - updateActiveRun(runId, (run) => ({ - ...run, - exitObserved: true, - })), - onReady: Effect.fn("desktop.backendInstance.onReady")(function* () { - const isCurrentRun = yield* Ref.modify(state, (latest) => { - const activeRun = Option.getOrUndefined(latest.active); - if (activeRun?.id !== runId) { - return [false, latest] as const; - } + const fiber = yield* Effect.forkIn(program, parentScope); + yield* updateActiveRun(runId, (run) => ({ + ...run, + fiber: Option.some(fiber), + })); + }), + ), + ).pipe(Effect.withSpan("desktop.backendInstance.start", { attributes: { id: spec.id } })); + + const start: Effect.Effect = beginStart(); + + // Counts one pre-ready startup failure. True only when this failure reaches + // the cap and no hook is already in flight. + const claimStartupFailure: Effect.Effect = Ref.modify(state, (latest) => { + if ( + spec.onStartupFailed === undefined || + latest.ready || + latest.startupFailurePending || + Option.isSome(latest.startupFailureFiber) + ) { + return [false, latest] as const; + } + const next = latest.startupFailureAttempt + 1; + if (next < MAX_STARTUP_FAILURE_ATTEMPTS) { + return [false, { ...latest, startupFailureAttempt: next }] as const; + } + return [true, { ...latest, startupFailureAttempt: 0, startupFailurePending: true }] as const; + }); - return [ - true, - { - ...latest, - restartAttempt: 0, - ready: true, - }, - ] as const; - }); - if (!isCurrentRun) { - return; - } + const clearTrackedStartupFailure = (tracked: Fiber.Fiber) => + Ref.update(state, (latest) => + Option.isSome(latest.startupFailureFiber) && latest.startupFailureFiber.value === tracked + ? { + ...latest, + startupFailureFiber: Option.none>(), + startupFailurePending: false, + } + : latest, + ); - yield* spec.onReady?.(config.value.httpBaseUrl) ?? Effect.void; - if ( - config.value.runningDistro !== undefined && - config.value.wslRuntimeId !== undefined - ) { - yield* wslEnvironment.pruneRuntimes( - config.value.runningDistro, - config.value.wslRuntimeId, - ); - } - }), - onReadinessFailure: Effect.fn("desktop.backendInstance.onReadinessFailure")( - function* (error) { - yield* logInstanceWarning("backend readiness check failed during bootstrap", { - error: error.message, - }); - yield* backendOutputLog.persistFailureSnapshot({ - details: error.message, - }); - }, - ), - onOutput: (streamName, chunk) => backendOutputLog.writeOutputChunk(streamName, chunk), - }).pipe( - Effect.provideService(ChildProcessSpawner.ChildProcessSpawner, spawner), - Effect.provideService(HttpClient.HttpClient, httpClient), - Scope.provide(runScope), - Effect.matchEffect({ - onFailure: (error) => finalizeRun(error.message), - onSuccess: (exit) => finalizeRun(exit.reason), - }), - Effect.ensuring(Scope.close(runScope, Exit.void).pipe(Effect.ignore)), + // Caller holds the instance mutex. The hook waits for that mutex so it + // cannot run (or call back into start/stop) until the caller releases it. + const armStartupFailure = ( + reason: string, + failedConfig: DesktopBackendStartConfig, + runToReplace: Option.Option, + ): Effect.Effect => + Effect.gen(function* () { + const onStartupFailed = spec.onStartupFailed; + if (onStartupFailed === undefined) { + yield* Ref.update(state, (latest) => + latest.startupFailurePending ? { ...latest, startupFailurePending: false } : latest, ); + return; + } - const fiber = yield* Effect.forkIn(program, parentScope); - yield* updateActiveRun(runId, (run) => ({ - ...run, - fiber: Option.some(fiber), - })); - }), - ), - ).pipe(Effect.withSpan("desktop.backendInstance.start", { attributes: { id: spec.id } })); + const tracked: { fiber?: Fiber.Fiber } = {}; + let followUpStarted = false; + const fiber = yield* Effect.forkIn( + mutex + .withPermits(1)(Effect.void) + .pipe( + Effect.andThen( + Effect.gen(function* () { + const decision = yield* onStartupFailed(reason, failedConfig).pipe(Effect.exit); + if (Exit.isFailure(decision)) { + if (Cause.hasInterruptsOnly(decision.cause)) return; + yield* logInstanceError("desktop backend startup failure hook failed", { + cause: Cause.pretty(decision.cause), + }); + } + const shouldReplace = Exit.isSuccess(decision) && decision.value; + const trackedFiber = tracked.fiber; + if (trackedFiber === undefined) return; + // Detach the restart from this fiber. stop() interrupts the hook + // while a dialog is up; it must not cancel a replacement that + // already decided to proceed, and the replacement's own stop() + // must not interrupt itself. + yield* Effect.forkIn( + Effect.gen(function* () { + if (Option.isSome(runToReplace)) { + if (shouldReplace) { + yield* replaceRun(runToReplace.value); + } + return; + } + if (!(yield* Ref.get(state)).desiredRunning) return; + yield* scheduleRestart(reason); + }).pipe(Effect.ensuring(clearTrackedStartupFailure(trackedFiber))), + parentScope, + ); + followUpStarted = true; + }), + ), + Effect.ensuring( + Effect.suspend(() => { + const trackedFiber = tracked.fiber; + if (followUpStarted || trackedFiber === undefined) return Effect.void; + return clearTrackedStartupFailure(trackedFiber); + }), + ), + ), + parentScope, + ); + tracked.fiber = fiber; + yield* Ref.update(state, (latest) => ({ + ...latest, + startupFailureFiber: Option.some(fiber), + startupFailurePending: true, + })); + }); + + // Stops the unreachable run, then starts again so configResolve sees + // whatever onStartupFailed changed. A stop from anywhere else bumps + // stopGeneration and the replacement start no-ops. + const replaceRun = Effect.fn("desktop.backendInstance.replaceUnreadyRun")(function* ( + runId: number, + ) { + const current = yield* Ref.get(state); + if ( + Option.getOrUndefined(current.active)?.id !== runId || + !current.desiredRunning || + current.ready + ) { + return; + } + const expectedStopGeneration = current.stopGeneration + 1; + yield* stop(); + yield* beginStart({ expectedStopGeneration }); + }); const scheduleRestart = Effect.fn("desktop.backendInstance.scheduleRestart")(function* ( reason: string, @@ -1048,7 +1247,9 @@ export const makeBackendInstance = Effect.fn("makeBackendInstance")(function* ( const stop = Effect.fn("desktop.backendInstance.stop")(function* (options?: { readonly timeout?: Duration.Duration; }) { - const { active, restartFiber, notifyShutdown } = yield* mutex.withPermits(1)( + const { active, restartFiber, startupFailureFiber, notifyShutdown } = yield* mutex.withPermits( + 1, + )( Effect.gen(function* () { const result = yield* Ref.modify(state, (latest) => { const active = Option.map(latest.active, (run) => @@ -1058,6 +1259,7 @@ export const makeBackendInstance = Effect.fn("makeBackendInstance")(function* ( { active, restartFiber: latest.restartFiber, + startupFailureFiber: latest.startupFailureFiber, notifyShutdown: latest.ready, }, { @@ -1066,6 +1268,9 @@ export const makeBackendInstance = Effect.fn("makeBackendInstance")(function* ( ready: false, active, restartFiber: Option.none>(), + startupFailureFiber: Option.none>(), + startupFailurePending: false, + stopGeneration: latest.stopGeneration + 1, }, ] as const; }); @@ -1080,6 +1285,10 @@ export const makeBackendInstance = Effect.fn("makeBackendInstance")(function* ( onNone: () => Effect.void, onSome: (fiber) => Fiber.interrupt(fiber).pipe(Effect.asVoid), }); + yield* Option.match(startupFailureFiber, { + onNone: () => Effect.void, + onSome: (fiber) => Fiber.interrupt(fiber).pipe(Effect.asVoid), + }); yield* Option.match(active, { onNone: () => Effect.void, onSome: (run) => diff --git a/apps/desktop/src/backend/DesktopBackendPool.test.ts b/apps/desktop/src/backend/DesktopBackendPool.test.ts index 8373437ae312..3c495fdb9a85 100644 --- a/apps/desktop/src/backend/DesktopBackendPool.test.ts +++ b/apps/desktop/src/backend/DesktopBackendPool.test.ts @@ -1,12 +1,17 @@ import { assert, describe, it } from "@effect/vitest"; +import * as Deferred from "effect/Deferred"; import * as Duration from "effect/Duration"; import * as Effect from "effect/Effect"; import * as FileSystem from "effect/FileSystem"; import * as Layer from "effect/Layer"; import * as Option from "effect/Option"; +import * as Queue from "effect/Queue"; import * as Ref from "effect/Ref"; +import * as Scope from "effect/Scope"; +import * as Sink from "effect/Sink"; import * as Stream from "effect/Stream"; -import { HttpClient } from "effect/unstable/http"; +import * as TestClock from "effect/testing/TestClock"; +import { HttpClient, HttpClientRequest, HttpClientResponse } from "effect/unstable/http"; import { ChildProcessSpawner } from "effect/unstable/process"; import * as DesktopObservability from "../app/DesktopObservability.ts"; @@ -154,4 +159,212 @@ describe("DesktopBackendPool", () => { }), ), ); + + it.effect("uses Windows for this launch when a wsl-only primary keeps crashing on startup", () => + Effect.scoped( + Effect.gen(function* () { + const mode = yield* Ref.make<"wsl" | "windows">("wsl"); + const spawns = yield* Queue.unbounded<"wsl" | "windows">(); + const dialogs = yield* Queue.unbounded<{ + readonly title: string; + readonly content: string; + }>(); + const ready = yield* Deferred.make(); + const wslConfig: DesktopBackendStartConfig = { + executablePath: "/electron", + args: ["/server/bin.mjs"], + entryPath: "/server/bin.mjs", + cwd: "/server", + env: {}, + bootstrap: { + mode: "desktop", + noBrowser: true, + port: 3773, + t3Home: "/tmp/t3", + host: "127.0.0.1", + desktopBootstrapToken: "token", + tailscaleServeEnabled: false, + tailscaleServePort: 443, + desktopTelemetryFd: 4, + desktopTelemetryControlFd: 5, + }, + bootstrapDelivery: "fd3", + extendEnv: false, + httpBaseUrl: new URL("http://172.17.0.1:3773"), + captureOutput: false, + preflightFailure: Option.none(), + runningDistro: "Ubuntu-22.04", + }; + const { runningDistro: _ignoredRunningDistro, ...windowsBase } = wslConfig; + const windowsConfig: DesktopBackendStartConfig = { + ...windowsBase, + extendEnv: true, + httpBaseUrl: new URL("http://127.0.0.1:3773"), + }; + + const settingsLayer = DesktopAppSettings.layerTest({ + ...DesktopAppSettings.DEFAULT_DESKTOP_SETTINGS, + wslBackendEnabled: true, + wslOnly: true, + wslDistro: "Ubuntu-22.04", + }); + const configurationLayer = Layer.effect( + DesktopBackendConfiguration.DesktopBackendConfiguration, + Effect.gen(function* () { + const settings = yield* DesktopAppSettings.DesktopAppSettings; + return { + resolvePrimary: Effect.gen(function* () { + const current = yield* settings.get; + const next = current.wslOnly && current.wslBackendEnabled ? "wsl" : "windows"; + yield* Ref.set(mode, next); + return next === "wsl" ? wslConfig : windowsConfig; + }), + resolvePrimaryLabel: Effect.succeed("WSL (Ubuntu-22.04)"), + resolveWsl: () => Effect.die("unexpected WSL config resolve"), + } satisfies DesktopBackendConfiguration.DesktopBackendConfiguration["Service"]; + }), + ); + const stack = DesktopBackendPool.layer.pipe( + Layer.provide(configurationLayer), + Layer.provideMerge(settingsLayer), + Layer.provide( + Layer.mergeAll( + FileSystem.layerNoop({ exists: () => Effect.succeed(true) }), + Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.gen(function* () { + const current = yield* Ref.get(mode); + yield* Queue.offer(spawns, current); + if (current === "windows") { + const scope = yield* Scope.Scope; + const exited = yield* Deferred.make(); + yield* Scope.addFinalizer(scope, Deferred.succeed(exited, void 0)); + return exitedProcess( + Deferred.await(exited).pipe(Effect.as(ChildProcessSpawner.ExitCode(0))), + ); + } + return exitedProcess(Effect.succeed(ChildProcessSpawner.ExitCode(1))); + }), + ), + ), + Layer.succeed( + HttpClient.HttpClient, + HttpClient.make((request) => + Effect.succeed( + responseFor(request, request.url.includes("127.0.0.1") ? 200 : 503), + ), + ), + ), + Layer.succeed(DesktopObservability.DesktopBackendOutputLogFactory, { + forInstance: () => + Effect.succeed({ + beginSession: () => Effect.void, + writeOutputChunk: () => Effect.void, + persistFailureSnapshot: () => Effect.void, + persistFailure: () => Effect.void, + discardSession: Effect.void, + } satisfies DesktopObservability.DesktopBackendOutputLogShape), + } satisfies DesktopObservability.DesktopBackendOutputLogFactory["Service"]), + Layer.succeed(DesktopTelemetryPublisher.DesktopTelemetryPublisher, { + latest: Effect.succeedNone, + changes: Stream.empty, + encoded: Stream.empty, + handleControlForSource: () => Effect.void, + removeControlSource: () => Effect.void, + publishUpdateReport: () => Effect.void, + updateRequests: Stream.empty, + updateCommits: Stream.empty, + updateCancellations: Stream.empty, + }), + DesktopWslEnvironment.layerTest(), + Layer.succeed(ElectronDialog.ElectronDialog, { + pickFolder: () => Effect.die("unexpected folder picker"), + pickFiles: () => Effect.die("unexpected file picker"), + showMessageBox: () => Effect.die("unexpected message box"), + showErrorBox: (title, content) => + Queue.offer(dialogs, { title, content }).pipe(Effect.asVoid), + } satisfies ElectronDialog.ElectronDialog["Service"]), + Layer.succeed(DesktopWindow.DesktopWindow, { + createMain: Effect.die("unexpected window create"), + ensureMain: Effect.die("unexpected window ensure"), + revealOrCreateMain: Effect.die("unexpected window reveal"), + activate: Effect.die("unexpected window activate"), + createMainIfBackendReady: Effect.die("unexpected window create"), + showConnectingSplash: Effect.void, + handleBackendReady: () => Deferred.succeed(ready, void 0).pipe(Effect.asVoid), + handleBackendNotReady: Effect.void, + flushMainWindowBounds: Effect.void, + prepareCaptureReveal: Effect.void, + dispatchMenuAction: () => Effect.die("unexpected menu action"), + dispatchSnapShotEvent: () => Effect.void, + zoomMain: () => Effect.die("unexpected zoom"), + syncAppearance: Effect.void, + } satisfies DesktopWindow.DesktopWindow["Service"]), + ), + ), + ); + yield* Effect.gen(function* () { + const pool = yield* DesktopBackendPool.DesktopBackendPool; + const settings = yield* DesktopAppSettings.DesktopAppSettings; + const primary = yield* pool.primary; + + yield* primary.start; + assert.equal(yield* Queue.take(spawns), "wsl"); + yield* TestClock.adjust(Duration.millis(500)); + assert.equal(yield* Queue.take(spawns), "wsl"); + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(yield* Queue.take(spawns), "wsl"); + + const dialog = yield* Queue.take(dialogs); + assert.equal(dialog.title, "WSL backend isn't responding"); + assert.include(dialog.content, "Ubuntu-22.04"); + assert.include(dialog.content, "code=1"); + assert.include(dialog.content, "Windows backend for this launch"); + + let restartScheduled = false; + while (!restartScheduled) { + restartScheduled = (yield* primary.snapshot).restartScheduled; + if (!restartScheduled) { + yield* Effect.yieldNow; + } + } + yield* TestClock.adjust(Duration.seconds(2)); + assert.equal(yield* Queue.take(spawns), "windows"); + yield* Deferred.await(ready); + + const recovered = yield* settings.get; + assert.equal(recovered.wslOnly, false); + assert.equal(recovered.wslBackendEnabled, false); + assert.equal(recovered.wslDistro, "Ubuntu-22.04"); + assert.equal((yield* primary.snapshot).ready, true); + }).pipe(Effect.provide(stack)); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); }); + +function responseFor( + request: HttpClientRequest.HttpClientRequest, + status: number, +): HttpClientResponse.HttpClientResponse { + return HttpClientResponse.fromWeb(request, new Response(null, { status })); +} + +function exitedProcess( + exitCode: Effect.Effect, +): ChildProcessSpawner.ChildProcessHandle { + return ChildProcessSpawner.makeHandle({ + pid: ChildProcessSpawner.ProcessId(123), + stdout: Stream.empty, + stderr: Stream.empty, + all: Stream.empty, + exitCode, + isRunning: Effect.succeed(false), + kill: () => Effect.void, + stdin: Sink.drain, + getInputFd: () => Sink.drain, + getOutputFd: () => Stream.empty, + unref: Effect.succeed(Effect.void), + }); +} diff --git a/apps/desktop/src/backend/DesktopBackendPool.ts b/apps/desktop/src/backend/DesktopBackendPool.ts index 2930780d4102..69f81d9e1d34 100644 --- a/apps/desktop/src/backend/DesktopBackendPool.ts +++ b/apps/desktop/src/backend/DesktopBackendPool.ts @@ -277,6 +277,33 @@ export const layer = Layer.effect( }, ); + // Preflight passed, but the primary still never became ready: it keeps + // exiting, or it stays up and the readiness probe keeps timing out. The + // connecting splash has no controls, so wsl-only mode would sit there + // forever. Use Windows for this launch only — the same in-memory fallback + // as a bounded preflight failure — and try WSL again on the next launch. + // A run that already resolved to Windows (no distro) has nothing to fall + // back to and keeps its existing restart loop. + const handlePrimaryStartupFailure = Effect.fn("desktop.backendPool.primaryStartupFailed")( + function* (reason: string, config: DesktopBackendManager.DesktopBackendStartConfig) { + if (config.runningDistro === undefined) return false; + const detail = `The WSL backend (${config.runningDistro}) did not become ready (${reason}).`; + yield* logBackendPoolWarning( + "primary WSL backend did not become ready; using Windows for this launch", + { reason, distro: config.runningDistro }, + ); + // A failed dialog must not keep the splash up. + yield* electronDialog + .showErrorBox( + "WSL backend isn't responding", + `${detail}\n\nT3 Code will use the Windows backend for this launch and retry WSL the next time the app starts.`, + ) + .pipe(Effect.ignoreCause); + yield* appSettings.applyWslWindowsFallbackInMemory; + return true; + }, + ); + const primary = yield* DesktopBackendManager.makeBackendInstance({ id: DesktopBackendManager.PRIMARY_INSTANCE_ID, // Keep this lazy. The pool layer is initialized before startup loads @@ -301,6 +328,7 @@ export const layer = Layer.effect( ), onShutdown: () => desktopWindow.handleBackendNotReady, onPreflightFailed: handlePrimaryPreflightFailure, + onStartupFailed: handlePrimaryStartupFailure, }); const instancesRef = yield* SynchronizedRef.make<