diff --git a/apps/desktop/src/backend/DesktopBackendManager.test.ts b/apps/desktop/src/backend/DesktopBackendManager.test.ts index 901d9f4a2708..a43cd6a0da37 100644 --- a/apps/desktop/src/backend/DesktopBackendManager.test.ts +++ b/apps/desktop/src/backend/DesktopBackendManager.test.ts @@ -126,6 +126,11 @@ interface MakeInstanceInput { readonly onPreflightFailed?: ( failure: DesktopBackendManager.PreflightFailure, ) => Effect.Effect; + readonly onStartupFailed?: ( + reason: string, + config: DesktopBackendManager.DesktopBackendStartConfig, + ) => Effect.Effect; + readonly readinessTimeout?: Duration.Duration; readonly config?: DesktopBackendManager.DesktopBackendStartConfig; readonly configResolve?: Effect.Effect< DesktopBackendManager.DesktopBackendStartConfig, @@ -185,6 +190,8 @@ function makeTestInstance(input: MakeInstanceInput) { ...(input.onReady ? { onReady: () => input.onReady! } : {}), ...(input.onShutdown ? { onShutdown: () => input.onShutdown! } : {}), ...(input.onPreflightFailed ? { onPreflightFailed: input.onPreflightFailed } : {}), + ...(input.onStartupFailed ? { onStartupFailed: input.onStartupFailed } : {}), + ...(input.readinessTimeout === undefined ? {} : { readinessTimeout: input.readinessTimeout }), }); return instance.pipe(Effect.provide(servicesLayer)); @@ -1399,6 +1406,460 @@ describe("DesktopBackendManager", () => { ), ); + it.effect( + "replaces a live run that never becomes reachable once startup failures hit the cap", + () => + Effect.scoped( + Effect.gen(function* () { + // wsl-only: the process stays up, but Windows cannot reach it (the + // iptables case). The hook swaps the resolved config to loopback. + const wslConfig: DesktopBackendManager.DesktopBackendStartConfig = { + ...baseConfig, + httpBaseUrl: new URL("http://172.17.0.1:3773"), + runningDistro: "Ubuntu-22.04", + }; + const useFallback = yield* Ref.make(false); + const startupFailures: string[] = []; + const spawnedUrls = yield* Queue.unbounded(); + const ready = yield* Deferred.make(); + let currentUrl = ""; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.gen(function* () { + const scope = yield* Scope.Scope; + const exited = yield* Deferred.make(); + yield* Queue.offer(spawnedUrls, currentUrl); + yield* Scope.addFinalizer(scope, Deferred.succeed(exited, void 0)); + return makeProcess({ + exitCode: Deferred.await(exited).pipe(Effect.as(ChildProcessSpawner.ExitCode(0))), + kill: () => Deferred.succeed(exited, void 0).pipe(Effect.asVoid), + }); + }), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + readinessTimeout: Duration.seconds(1), + configResolve: Ref.get(useFallback).pipe( + Effect.map((fallback) => (fallback ? baseConfig : wslConfig)), + Effect.tap((config) => + Effect.sync(() => { + currentUrl = config.httpBaseUrl.href; + }), + ), + ), + httpClientLayer: httpClientLayer((request) => + Effect.succeed( + responseForRequest(request, request.url.startsWith("http://127.0.0.1") ? 200 : 503), + ), + ), + onReady: Deferred.succeed(ready, void 0).pipe(Effect.asVoid), + onStartupFailed: (reason) => + Effect.sync(() => { + startupFailures.push(reason); + }).pipe(Effect.andThen(Ref.set(useFallback, true)), Effect.as(true)), + }); + + yield* instance.start; + assert.equal(yield* Queue.take(spawnedUrls), "http://172.17.0.1:3773/"); + + // Two unreachable readiness rounds are tolerated (slow WSL cold boot). + yield* TestClock.adjust(Duration.seconds(2)); + assert.deepEqual(startupFailures, []); + assert.equal(yield* Queue.size(spawnedUrls), 0); + + // The third round surfaces the failure once and swaps the run. + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(yield* Queue.take(spawnedUrls), "http://127.0.0.1:3773/"); + yield* Deferred.await(ready); + assert.equal(startupFailures.length, 1); + assert.equal((yield* instance.snapshot).ready, true); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("stopping the backend cancels a pending startup failure hook", () => + Effect.scoped( + Effect.gen(function* () { + const hookEntered = yield* Deferred.make(); + const releaseHook = yield* Deferred.make(); + let fallbackApplied = false; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.gen(function* () { + const scope = yield* Scope.Scope; + const exited = yield* Deferred.make(); + yield* Scope.addFinalizer(scope, Deferred.succeed(exited, void 0)); + return makeProcess({ + exitCode: Deferred.await(exited).pipe(Effect.as(ChildProcessSpawner.ExitCode(0))), + kill: () => Deferred.succeed(exited, void 0).pipe(Effect.asVoid), + }); + }), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + readinessTimeout: Duration.seconds(1), + httpClientLayer: httpClientLayer((request) => + Effect.succeed(responseForRequest(request, 503)), + ), + onStartupFailed: () => + Deferred.succeed(hookEntered, void 0).pipe( + Effect.andThen(Deferred.await(releaseHook)), + Effect.andThen( + Effect.sync(() => { + fallbackApplied = true; + }), + ), + Effect.as(true), + ), + }); + + yield* instance.start; + yield* TestClock.adjust(Duration.seconds(3)); + yield* Deferred.await(hookEntered); + + yield* instance.stop(); + yield* Deferred.succeed(releaseHook, void 0); + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(fallbackApplied, false); + assert.equal((yield* instance.snapshot).desiredRunning, false); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("keeps a backend that became ready while the startup failure hook was pending", () => + Effect.scoped( + Effect.gen(function* () { + const hookEntered = yield* Deferred.make(); + const releaseHook = yield* Deferred.make(); + const ready = yield* Deferred.make(); + let spawnCount = 0; + let reachable = false; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.gen(function* () { + spawnCount += 1; + const scope = yield* Scope.Scope; + const exited = yield* Deferred.make(); + yield* Scope.addFinalizer(scope, Deferred.succeed(exited, void 0)); + return makeProcess({ + exitCode: Deferred.await(exited).pipe(Effect.as(ChildProcessSpawner.ExitCode(0))), + kill: () => Deferred.succeed(exited, void 0).pipe(Effect.asVoid), + }); + }), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + readinessTimeout: Duration.seconds(1), + httpClientLayer: httpClientLayer((request) => + Effect.sync(() => responseForRequest(request, reachable ? 200 : 503)), + ), + onReady: Deferred.succeed(ready, void 0).pipe(Effect.asVoid), + onStartupFailed: () => + Deferred.succeed(hookEntered, void 0).pipe( + Effect.andThen(Deferred.await(releaseHook)), + Effect.as(true), + ), + }); + + yield* instance.start; + yield* TestClock.adjust(Duration.seconds(3)); + yield* Deferred.await(hookEntered); + + reachable = true; + yield* TestClock.adjust(Duration.seconds(1)); + yield* Deferred.await(ready); + + yield* Deferred.succeed(releaseHook, void 0); + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(spawnCount, 1); + assert.equal((yield* instance.snapshot).ready, true); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("stop is not held up by a pending startup failure hook after pre-ready exits", () => + Effect.scoped( + Effect.gen(function* () { + const hookEntered = yield* Deferred.make(); + const releaseHook = yield* Deferred.make(); + const exits = yield* Queue.unbounded(); + let fallbackApplied = false; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.succeed( + makeProcess({ + exitCode: Queue.offer(exits, void 0).pipe( + Effect.as(ChildProcessSpawner.ExitCode(1)), + ), + }), + ), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + httpClientLayer: httpClientLayer(() => Effect.never), + onStartupFailed: () => + Deferred.succeed(hookEntered, void 0).pipe( + Effect.andThen(Deferred.await(releaseHook)), + Effect.andThen( + Effect.sync(() => { + fallbackApplied = true; + }), + ), + Effect.as(true), + ), + }); + + yield* instance.start; + // Pre-ready exits restart after 0.5s and 1s; the third exit hits the cap. + yield* Queue.take(exits); + yield* TestClock.adjust(Duration.millis(500)); + yield* Queue.take(exits); + yield* TestClock.adjust(Duration.seconds(1)); + yield* Deferred.await(hookEntered); + + yield* instance.stop(); + yield* Deferred.succeed(releaseHook, void 0); + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(fallbackApplied, false); + assert.equal((yield* instance.snapshot).desiredRunning, false); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("does not restart a backend that was stopped while being replaced", () => + Effect.scoped( + Effect.gen(function* () { + const closing = yield* Queue.unbounded(); + const releaseClose = yield* Deferred.make(); + let spawnCount = 0; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.gen(function* () { + spawnCount += 1; + const scope = yield* Scope.Scope; + const exited = yield* Deferred.make(); + // Closing the run blocks until the test lets it finish. + yield* Scope.addFinalizer( + scope, + Queue.offer(closing, void 0).pipe( + Effect.andThen(Deferred.await(releaseClose)), + Effect.andThen(Deferred.succeed(exited, void 0)), + ), + ); + return makeProcess({ + exitCode: Deferred.await(exited).pipe(Effect.as(ChildProcessSpawner.ExitCode(0))), + }); + }), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + readinessTimeout: Duration.seconds(1), + httpClientLayer: httpClientLayer((request) => + Effect.succeed(responseForRequest(request, 503)), + ), + onStartupFailed: () => Effect.succeed(true), + }); + + yield* instance.start; + yield* TestClock.adjust(Duration.seconds(3)); + // The replacement is closing the old run; the user stops the backend now. + yield* Queue.take(closing); + const userStop = yield* Effect.forkChild(instance.stop()); + yield* TestClock.adjust(Duration.millis(1)); + yield* Deferred.succeed(releaseClose, void 0); + yield* Fiber.join(userStop); + yield* TestClock.adjust(Duration.seconds(1)); + + assert.equal(spawnCount, 1); + assert.equal((yield* instance.snapshot).desiredRunning, false); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("surfaces repeated exits before readiness and keeps retrying when declined", () => + Effect.scoped( + Effect.gen(function* () { + const starts = yield* Queue.unbounded(); + const startupFailures = yield* Queue.unbounded(); + let startCount = 0; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.sync(() => { + startCount += 1; + return makeProcess({ + exitCode: Queue.offer(starts, startCount).pipe( + Effect.as(ChildProcessSpawner.ExitCode(1)), + ), + }); + }), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + httpClientLayer: httpClientLayer(() => Effect.never), + onStartupFailed: (reason) => Queue.offer(startupFailures, reason).pipe(Effect.as(false)), + }); + + yield* instance.start; + assert.equal(yield* Queue.take(starts), 1); + yield* TestClock.adjust(Duration.millis(500)); + assert.equal(yield* Queue.take(starts), 2); + assert.equal(yield* Queue.size(startupFailures), 0); + + // The third exit before the backend ever became ready hits the cap. + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(yield* Queue.take(starts), 3); + yield* Queue.take(startupFailures); + + // Declined (a Windows primary): the restart loop carries on. + yield* TestClock.adjust(Duration.seconds(2)); + assert.equal(yield* Queue.take(starts), 4); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("waits for the startup failure hook before restarting a crashed run", () => + Effect.scoped( + Effect.gen(function* () { + const wslConfig: DesktopBackendManager.DesktopBackendStartConfig = { + ...baseConfig, + httpBaseUrl: new URL("http://172.17.0.1:3773"), + runningDistro: "Ubuntu-22.04", + }; + const useFallback = yield* Ref.make(false); + const hookEntered = yield* Deferred.make(); + const releaseHook = yield* Deferred.make(); + const spawnedUrls = yield* Queue.unbounded(); + const ready = yield* Deferred.make(); + let currentUrl = ""; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.sync(() => + makeProcess({ + exitCode: Effect.sync(() => currentUrl).pipe( + Effect.tap((url) => Queue.offer(spawnedUrls, url)), + Effect.flatMap((url) => + url.startsWith("http://127.0.0.1") + ? Effect.never + : Effect.succeed(ChildProcessSpawner.ExitCode(1)), + ), + ), + }), + ), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + configResolve: Ref.get(useFallback).pipe( + Effect.map((fallback) => (fallback ? baseConfig : wslConfig)), + Effect.tap((config) => + Effect.sync(() => { + currentUrl = config.httpBaseUrl.href; + }), + ), + ), + httpClientLayer: httpClientLayer((request) => + Effect.succeed( + responseForRequest(request, request.url.startsWith("http://127.0.0.1") ? 200 : 503), + ), + ), + onReady: Deferred.succeed(ready, void 0).pipe(Effect.asVoid), + onStartupFailed: () => + Deferred.succeed(hookEntered, void 0).pipe( + Effect.andThen(Deferred.await(releaseHook)), + Effect.andThen(Ref.set(useFallback, true)), + Effect.as(true), + ), + }); + + yield* instance.start; + assert.equal(yield* Queue.take(spawnedUrls), "http://172.17.0.1:3773/"); + yield* TestClock.adjust(Duration.millis(500)); + assert.equal(yield* Queue.take(spawnedUrls), "http://172.17.0.1:3773/"); + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(yield* Queue.take(spawnedUrls), "http://172.17.0.1:3773/"); + yield* Deferred.await(hookEntered); + + // The crashed run must not be replaced until the hook applies the fallback. + yield* TestClock.adjust(Duration.seconds(30)); + assert.equal(yield* Queue.size(spawnedUrls), 0); + + yield* Deferred.succeed(releaseHook, void 0); + yield* TestClock.adjust(Duration.seconds(2)); + assert.equal(yield* Queue.take(spawnedUrls), "http://127.0.0.1:3773/"); + yield* Deferred.await(ready); + assert.equal((yield* instance.snapshot).ready, true); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + + it.effect("does not surface exits that happen after the backend became ready", () => + Effect.scoped( + Effect.gen(function* () { + const readied = yield* Queue.unbounded(); + const exits = yield* Queue.unbounded(); + let startupFailures = 0; + + const spawnerLayer = Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.succeed( + makeProcess({ + // Each run crashes only after it has reported ready. + exitCode: Queue.take(readied).pipe(Effect.as(ChildProcessSpawner.ExitCode(1))), + }), + ), + ), + ); + + const instance = yield* makeTestInstance({ + spawnerLayer, + onReady: Queue.offer(readied, void 0).pipe(Effect.asVoid), + onStartupFailed: () => + Effect.sync(() => { + startupFailures += 1; + }).pipe(Effect.as(false)), + backendOutputLog: { + persistFailure: ({ details }) => Queue.offer(exits, details).pipe(Effect.asVoid), + }, + }); + + yield* instance.start; + for (let i = 0; i < 5; i++) { + yield* Queue.take(exits); + yield* TestClock.adjust(Duration.seconds(1)); + } + assert.equal(startupFailures, 0); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); + it.effect("cancels a scheduled restart when start is requested manually", () => Effect.scoped( Effect.gen(function* () { diff --git a/apps/desktop/src/backend/DesktopBackendManager.ts b/apps/desktop/src/backend/DesktopBackendManager.ts index 6f1ea139bdca..548c43345a3b 100644 --- a/apps/desktop/src/backend/DesktopBackendManager.ts +++ b/apps/desktop/src/backend/DesktopBackendManager.ts @@ -61,6 +61,10 @@ const MAX_RESTART_DELAY = Duration.seconds(10); // failures may instead provide their own larger retryLimit when they should // self-heal for a while but must not leave the app connecting forever. const MAX_PREFLIGHT_FAILURE_ATTEMPTS = 5; +// Consecutive readiness timeouts or exits before the backend was ever ready. +// Three one-minute readiness rounds still leave room for a slow WSL cold boot. +// Past that, wsl-only mode can leave the splash and fall back for this launch. +const MAX_STARTUP_FAILURE_ATTEMPTS = 3; const DEFAULT_BACKEND_READINESS_TIMEOUT = Duration.minutes(1); const DEFAULT_BACKEND_READINESS_INTERVAL = Duration.millis(100); const DEFAULT_BACKEND_READINESS_REQUEST_TIMEOUT = Duration.seconds(1); @@ -299,6 +303,20 @@ export interface BackendInstanceSpec { // retries. Returns true when the callback changed configuration and the // manager should resolve once more; false stops the failed instance. readonly onPreflightFailed?: (failure: PreflightFailure) => Effect.Effect; + // Fired when MAX_STARTUP_FAILURE_ATTEMPTS consecutive startup failures + // accumulate after a clean preflight (readiness budgets that time out, or + // exits before the run was ever ready). `config` is the run that failed. + // Return true when configuration changed and the current run should be + // replaced; false leaves the normal restart and readiness loops in place. + // Exits after a successful ready do not count, so a backend that was healthy + // and later crashes keeps the existing restart loop. + readonly onStartupFailed?: ( + reason: string, + config: DesktopBackendStartConfig, + ) => Effect.Effect; + // Overrides the per-round readiness budget. Production uses the default; + // tests pass a short duration so the cap can be reached under TestClock. + readonly readinessTimeout?: Duration.Duration; } interface ActiveBackendRun { @@ -319,6 +337,18 @@ interface BackendManagerState { // Consecutive bounded/fatal preflight failures, reset on a clean or // unbounded-transient preflight. restartAttempt counts all restarts. readonly preflightFailureAttempt: number; + // Consecutive readiness timeouts and pre-ready exits, reset once ready. + readonly startupFailureAttempt: number; + // True from the moment the cap is claimed until the hook's follow-up + // finishes, so a later timeout cannot start a second hook. + readonly startupFailurePending: boolean; + // The in-flight onStartupFailed hook. stop() interrupts it so a blocked + // dialog cannot hold shutdown. The follow-up that replaces the run is a + // different fiber and is not cancelled here. + readonly startupFailureFiber: Option.Option>; + // Bumped by every stop(), so a replacement can tell an explicit stop + // happened while it was closing the failed run. + readonly stopGeneration: number; readonly restartFiber: Option.Option>; readonly nextRunId: number; } @@ -330,6 +360,10 @@ const initialState: BackendManagerState = { active: Option.none(), restartAttempt: 0, preflightFailureAttempt: 0, + startupFailureAttempt: 0, + startupFailurePending: false, + startupFailureFiber: Option.none(), + stopGeneration: 0, restartFiber: Option.none(), nextRunId: 1, }; @@ -688,302 +722,467 @@ export const makeBackendInstance = Effect.fn("makeBackendInstance")(function* ( }); }); - const start: Effect.Effect = Effect.suspend(() => - mutex.withPermits(1)( - Effect.gen(function* () { - const current = yield* Ref.get(state); - if (Option.isSome(current.active)) { - if (!current.desiredRunning) { - yield* Ref.update(state, (latest) => ({ - ...latest, - desiredRunning: true, - })); + const beginStart = (options?: { + readonly expectedStopGeneration?: number; + }): Effect.Effect => + Effect.suspend(() => + mutex.withPermits(1)( + Effect.gen(function* () { + const current = yield* Ref.get(state); + // Automatic recovery checks the generation under the same lock that + // sets desiredRunning, so a stop that lands after replaceRun's own + // stop() cannot be overwritten by the replacement start. + if ( + options?.expectedStopGeneration !== undefined && + current.stopGeneration !== options.expectedStopGeneration + ) { + return; } - return; - } - - if (current.ready) { - yield* spec.onShutdown?.() ?? Effect.void; - yield* Ref.update(state, (latest) => - latest.ready ? { ...latest, ready: false } : latest, - ); - } - const config = yield* spec.configResolve.pipe( - Effect.tapError((error) => - logInstanceError("failed to generate desktop backend configuration", { - cause: error.message, - }), - ), - Effect.option, - ); - if (Option.isNone(config)) { - if (current.desiredRunning) { - yield* scheduleRestart("failed to generate desktop backend configuration"); + if (Option.isSome(current.active)) { + if (!current.desiredRunning) { + yield* Ref.update(state, (latest) => ({ + ...latest, + desiredRunning: true, + })); + } + return; } - return; - } - const entryExists = yield* fileSystem - .exists(config.value.entryPath) - .pipe(Effect.orElseSucceed(() => false)); - - const resetFatalPreflightCounter = - !current.desiredRunning && current.preflightFailureAttempt > 0; - yield* cancelRestart; - yield* Ref.update(state, (latest) => ({ - ...latest, - desiredRunning: true, - ready: false, - config: Option.some(config.value), - preflightFailureAttempt: resetFatalPreflightCounter ? 0 : latest.preflightFailureAttempt, - })); - - const preflightFailure = config.value.preflightFailure; - if (Option.isSome(preflightFailure)) { - const { reason, fatal, retryLimit } = preflightFailure.value; - if (!fatal && retryLimit === undefined) { - // Transient (WSL cold-starting, wslpath while the VM boots). Keep - // retrying so the backend self-heals once WSL is ready. Reset a - // prior bounded/fatal streak because this is a different failure. + + if (current.ready) { + yield* spec.onShutdown?.() ?? Effect.void; yield* Ref.update(state, (latest) => - latest.preflightFailureAttempt === 0 - ? latest - : { ...latest, preflightFailureAttempt: 0 }, + latest.ready ? { ...latest, ready: false } : latest, ); - yield* scheduleRestart(reason); - return; } - const attemptLimit = retryLimit ?? MAX_PREFLIGHT_FAILURE_ATTEMPTS; - const attempt = yield* Ref.modify(state, (latest) => { - const next = latest.preflightFailureAttempt + 1; - return [next, { ...latest, preflightFailureAttempt: next }] as const; - }); - if (attempt > attemptLimit) { - // We already surfaced and asked for the Windows fallback, yet we're - // still resolving the WSL primary — the fallback didn't take (e.g. - // the settings write failed). Stop rather than loop forever. - yield* logInstanceError("backend preflight still failing after fallback; stopping", { - reason, - attempt, - }); - yield* Ref.update(state, (latest) => ({ - ...latest, - desiredRunning: false, - ready: false, - })); + const config = yield* spec.configResolve.pipe( + Effect.tapError((error) => + logInstanceError("failed to generate desktop backend configuration", { + cause: error.message, + }), + ), + Effect.option, + ); + if (Option.isNone(config)) { + if (current.desiredRunning) { + yield* scheduleRestart("failed to generate desktop backend configuration"); + } return; } - if (attempt === attemptLimit) { - // Fatal/bounded and out of retries. Surface the reason (onPreflightFailed, - // on the primary, shows a dialog and persists Windows mode), then - // schedule one more restart so the next resolve picks up the Windows - // primary and a window can open. - yield* logInstanceError( - "backend preflight failed repeatedly; surfacing and falling back", - { reason, attempt }, - ); - const shouldRestart = yield* ( - spec.onPreflightFailed?.(preflightFailure.value) ?? Effect.succeed(false) - ); - if (shouldRestart) { + const entryExists = yield* fileSystem + .exists(config.value.entryPath) + .pipe(Effect.orElseSucceed(() => false)); + + const resetFatalPreflightCounter = + !current.desiredRunning && current.preflightFailureAttempt > 0; + yield* cancelRestart; + yield* Ref.update(state, (latest) => ({ + ...latest, + desiredRunning: true, + ready: false, + config: Option.some(config.value), + preflightFailureAttempt: resetFatalPreflightCounter + ? 0 + : latest.preflightFailureAttempt, + // A user-driven start gets a fresh startup budget. Restart-loop + // starts keep the streak so crashes can reach the cap. + startupFailureAttempt: current.desiredRunning ? latest.startupFailureAttempt : 0, + })); + + const preflightFailure = config.value.preflightFailure; + if (Option.isSome(preflightFailure)) { + const { reason, fatal, retryLimit } = preflightFailure.value; + if (!fatal && retryLimit === undefined) { + // Transient (WSL cold-starting, wslpath while the VM boots). Keep + // retrying so the backend self-heals once WSL is ready. Reset a + // prior bounded/fatal streak because this is a different failure. + yield* Ref.update(state, (latest) => + latest.preflightFailureAttempt === 0 + ? latest + : { ...latest, preflightFailureAttempt: 0 }, + ); yield* scheduleRestart(reason); - } else { + return; + } + const attemptLimit = retryLimit ?? MAX_PREFLIGHT_FAILURE_ATTEMPTS; + const attempt = yield* Ref.modify(state, (latest) => { + const next = latest.preflightFailureAttempt + 1; + return [next, { ...latest, preflightFailureAttempt: next }] as const; + }); + if (attempt > attemptLimit) { + // We already surfaced and asked for the Windows fallback, yet we're + // still resolving the WSL primary — the fallback didn't take (e.g. + // the settings write failed). Stop rather than loop forever. + yield* logInstanceError("backend preflight still failing after fallback; stopping", { + reason, + attempt, + }); yield* Ref.update(state, (latest) => ({ ...latest, desiredRunning: false, ready: false, })); + return; + } + if (attempt === attemptLimit) { + // Fatal/bounded and out of retries. Surface the reason (onPreflightFailed, + // on the primary, shows a dialog and persists Windows mode), then + // schedule one more restart so the next resolve picks up the Windows + // primary and a window can open. + yield* logInstanceError( + "backend preflight failed repeatedly; surfacing and falling back", + { reason, attempt }, + ); + const shouldRestart = yield* ( + spec.onPreflightFailed?.(preflightFailure.value) ?? Effect.succeed(false) + ); + if (shouldRestart) { + yield* scheduleRestart(reason); + } else { + yield* Ref.update(state, (latest) => ({ + ...latest, + desiredRunning: false, + ready: false, + })); + } + return; } + yield* scheduleRestart(reason); return; } - yield* scheduleRestart(reason); - return; - } - // Clean preflight — reset the fatal counter so a later failure gets a - // fresh allowance. - yield* Ref.update(state, (latest) => - latest.preflightFailureAttempt === 0 ? latest : { ...latest, preflightFailureAttempt: 0 }, - ); + // Clean preflight — reset the fatal counter so a later failure gets a + // fresh allowance. + yield* Ref.update(state, (latest) => + latest.preflightFailureAttempt === 0 + ? latest + : { ...latest, preflightFailureAttempt: 0 }, + ); - if (!entryExists) { - yield* scheduleRestart(`missing server entry at ${config.value.entryPath}`); - return; - } + if (!entryExists) { + yield* scheduleRestart(`missing server entry at ${config.value.entryPath}`); + return; + } - const runScope = yield* Scope.make("sequential"); - const runId = yield* Ref.modify(state, (latest) => [ - latest.nextRunId, - { - ...latest, - active: Option.some({ - id: latest.nextRunId, - scope: runScope, - fiber: Option.none(), - pid: Option.none(), - exitObserved: false, - stopRequested: false, - } satisfies ActiveBackendRun), - nextRunId: latest.nextRunId + 1, - }, - ]); - - const finalizeRun = Effect.fn("desktop.backendInstance.finalizeRun")(function* ( - reason: string, - ) { - yield* mutex.withPermits(1)( - Effect.gen(function* () { - const { isCurrentRun, nextState, pid, exitObserved, stopRequested, wasReady } = - yield* Ref.modify( - state, - ( - latest, - ): readonly [ - { - readonly isCurrentRun: boolean; - readonly nextState: BackendManagerState; - readonly pid: Option.Option; - readonly exitObserved: boolean; - readonly stopRequested: boolean; - readonly wasReady: boolean; - }, - BackendManagerState, - ] => { - const currentRun = Option.getOrUndefined(latest.active); - if (currentRun?.id !== runId) { + const runScope = yield* Scope.make("sequential"); + const runId = yield* Ref.modify(state, (latest) => [ + latest.nextRunId, + { + ...latest, + active: Option.some({ + id: latest.nextRunId, + scope: runScope, + fiber: Option.none(), + pid: Option.none(), + exitObserved: false, + stopRequested: false, + } satisfies ActiveBackendRun), + nextRunId: latest.nextRunId + 1, + }, + ]); + + const finalizeRun = Effect.fn("desktop.backendInstance.finalizeRun")(function* ( + reason: string, + ) { + yield* mutex.withPermits(1)( + Effect.gen(function* () { + const { isCurrentRun, nextState, pid, exitObserved, stopRequested, wasReady } = + yield* Ref.modify( + state, + ( + latest, + ): readonly [ + { + readonly isCurrentRun: boolean; + readonly nextState: BackendManagerState; + readonly pid: Option.Option; + readonly exitObserved: boolean; + readonly stopRequested: boolean; + readonly wasReady: boolean; + }, + BackendManagerState, + ] => { + const currentRun = Option.getOrUndefined(latest.active); + if (currentRun?.id !== runId) { + return [ + { + isCurrentRun: false, + nextState: latest, + pid: Option.none(), + exitObserved: false, + stopRequested: false, + wasReady: false, + }, + latest, + ] as const; + } + + const next = { + ...latest, + active: Option.none(), + ready: false, + }; return [ { - isCurrentRun: false, - nextState: latest, - pid: Option.none(), - exitObserved: false, - stopRequested: false, - wasReady: false, + isCurrentRun: true, + nextState: next, + pid: currentRun.pid, + exitObserved: currentRun.exitObserved, + stopRequested: currentRun.stopRequested, + wasReady: latest.ready, }, - latest, + next, ] as const; + }, + ); + + if (isCurrentRun) { + yield* desktopTelemetryPublisher.removeControlSource(spec.id); + if (Option.isSome(pid)) { + if (exitObserved && !stopRequested) { + yield* backendOutputLog.persistFailure({ + details: `pid=${pid.value} ${reason}`, + }); + } else { + yield* backendOutputLog.discardSession; } + } + if (wasReady) { + yield* spec.onShutdown?.() ?? Effect.void; + } + } - const next = { - ...latest, - active: Option.none(), - ready: false, - }; - return [ - { - isCurrentRun: true, - nextState: next, - pid: currentRun.pid, - exitObserved: currentRun.exitObserved, - stopRequested: currentRun.stopRequested, - wasReady: latest.ready, - }, - next, - ] as const; - }, - ); - - if (isCurrentRun) { - yield* desktopTelemetryPublisher.removeControlSource(spec.id); - if (Option.isSome(pid)) { - if (exitObserved && !stopRequested) { - yield* backendOutputLog.persistFailure({ - details: `pid=${pid.value} ${reason}`, - }); - } else { - yield* backendOutputLog.discardSession; + // The process is already gone. When the startup cap is hit, the + // hook owns the next restart so a fallback it applies is visible + // to configResolve. Scheduling here would race that write. + let startupFailureClaimed = false; + if (isCurrentRun && exitObserved && !stopRequested && !wasReady) { + startupFailureClaimed = yield* claimStartupFailure; + if (startupFailureClaimed) { + yield* armStartupFailure(reason, config.value, Option.none()); } } - if (wasReady) { - yield* spec.onShutdown?.() ?? Effect.void; + + if (isCurrentRun && nextState.desiredRunning && !startupFailureClaimed) { + yield* scheduleRestart(reason); + } + }), + ); + }); + + const program = runBackendProcess({ + ...config.value, + ...(spec.readinessTimeout === undefined + ? {} + : { readinessTimeout: spec.readinessTimeout }), + desktopTelemetryStream: desktopTelemetryPublisher.encoded, + onDesktopTelemetryControl: (message) => + desktopTelemetryPublisher.handleControlForSource(spec.id, message), + onStarted: Effect.fn("desktop.backendInstance.onStarted")(function* (pid) { + yield* updateActiveRun(runId, (run) => ({ + ...run, + pid: Option.some(pid), + })); + yield* backendOutputLog.beginSession({ + details: `pid=${pid} port=${config.value.bootstrap.port} cwd=${config.value.cwd}`, + }); + }), + onExitObserved: () => + updateActiveRun(runId, (run) => ({ + ...run, + exitObserved: true, + })), + onReady: Effect.fn("desktop.backendInstance.onReady")(function* () { + const isCurrentRun = yield* Ref.modify(state, (latest) => { + const activeRun = Option.getOrUndefined(latest.active); + if (activeRun?.id !== runId) { + return [false, latest] as const; } + + return [ + true, + { + ...latest, + restartAttempt: 0, + startupFailureAttempt: 0, + ready: true, + }, + ] as const; + }); + if (!isCurrentRun) { + return; } - if (isCurrentRun && nextState.desiredRunning) { - yield* scheduleRestart(reason); + yield* spec.onReady?.(config.value.httpBaseUrl) ?? Effect.void; + if ( + config.value.runningDistro !== undefined && + config.value.wslRuntimeId !== undefined + ) { + yield* wslEnvironment.pruneRuntimes( + config.value.runningDistro, + config.value.wslRuntimeId, + ); } }), + onReadinessFailure: Effect.fn("desktop.backendInstance.onReadinessFailure")( + function* (error) { + yield* logInstanceWarning("backend readiness check failed during bootstrap", { + error: error.message, + }); + yield* backendOutputLog.persistFailureSnapshot({ + details: error.message, + }); + yield* mutex.withPermits(1)( + Effect.gen(function* () { + const current = yield* Ref.get(state); + if (Option.getOrUndefined(current.active)?.id !== runId) return; + if (!current.desiredRunning || current.ready) return; + if (!(yield* claimStartupFailure)) return; + yield* armStartupFailure(error.message, config.value, Option.some(runId)); + }), + ); + }, + ), + onOutput: (streamName, chunk) => backendOutputLog.writeOutputChunk(streamName, chunk), + }).pipe( + Effect.provideService(ChildProcessSpawner.ChildProcessSpawner, spawner), + Effect.provideService(HttpClient.HttpClient, httpClient), + Scope.provide(runScope), + Effect.matchEffect({ + onFailure: (error) => finalizeRun(error.message), + onSuccess: (exit) => finalizeRun(exit.reason), + }), + Effect.ensuring(Scope.close(runScope, Exit.void).pipe(Effect.ignore)), ); - }); - const program = runBackendProcess({ - ...config.value, - desktopTelemetryStream: desktopTelemetryPublisher.encoded, - onDesktopTelemetryControl: (message) => - desktopTelemetryPublisher.handleControlForSource(spec.id, message), - onStarted: Effect.fn("desktop.backendInstance.onStarted")(function* (pid) { - yield* updateActiveRun(runId, (run) => ({ - ...run, - pid: Option.some(pid), - })); - yield* backendOutputLog.beginSession({ - details: `pid=${pid} port=${config.value.bootstrap.port} cwd=${config.value.cwd}`, - }); - }), - onExitObserved: () => - updateActiveRun(runId, (run) => ({ - ...run, - exitObserved: true, - })), - onReady: Effect.fn("desktop.backendInstance.onReady")(function* () { - const isCurrentRun = yield* Ref.modify(state, (latest) => { - const activeRun = Option.getOrUndefined(latest.active); - if (activeRun?.id !== runId) { - return [false, latest] as const; - } + const fiber = yield* Effect.forkIn(program, parentScope); + yield* updateActiveRun(runId, (run) => ({ + ...run, + fiber: Option.some(fiber), + })); + }), + ), + ).pipe(Effect.withSpan("desktop.backendInstance.start", { attributes: { id: spec.id } })); + + const start: Effect.Effect = beginStart(); + + // Counts one pre-ready startup failure. True only when this failure reaches + // the cap and no hook is already in flight. + const claimStartupFailure: Effect.Effect = Ref.modify(state, (latest) => { + if ( + spec.onStartupFailed === undefined || + latest.ready || + latest.startupFailurePending || + Option.isSome(latest.startupFailureFiber) + ) { + return [false, latest] as const; + } + const next = latest.startupFailureAttempt + 1; + if (next < MAX_STARTUP_FAILURE_ATTEMPTS) { + return [false, { ...latest, startupFailureAttempt: next }] as const; + } + return [true, { ...latest, startupFailureAttempt: 0, startupFailurePending: true }] as const; + }); - return [ - true, - { - ...latest, - restartAttempt: 0, - ready: true, - }, - ] as const; - }); - if (!isCurrentRun) { - return; - } + const clearTrackedStartupFailure = (tracked: Fiber.Fiber) => + Ref.update(state, (latest) => + Option.isSome(latest.startupFailureFiber) && latest.startupFailureFiber.value === tracked + ? { + ...latest, + startupFailureFiber: Option.none>(), + startupFailurePending: false, + } + : latest, + ); - yield* spec.onReady?.(config.value.httpBaseUrl) ?? Effect.void; - if ( - config.value.runningDistro !== undefined && - config.value.wslRuntimeId !== undefined - ) { - yield* wslEnvironment.pruneRuntimes( - config.value.runningDistro, - config.value.wslRuntimeId, - ); - } - }), - onReadinessFailure: Effect.fn("desktop.backendInstance.onReadinessFailure")( - function* (error) { - yield* logInstanceWarning("backend readiness check failed during bootstrap", { - error: error.message, - }); - yield* backendOutputLog.persistFailureSnapshot({ - details: error.message, - }); - }, - ), - onOutput: (streamName, chunk) => backendOutputLog.writeOutputChunk(streamName, chunk), - }).pipe( - Effect.provideService(ChildProcessSpawner.ChildProcessSpawner, spawner), - Effect.provideService(HttpClient.HttpClient, httpClient), - Scope.provide(runScope), - Effect.matchEffect({ - onFailure: (error) => finalizeRun(error.message), - onSuccess: (exit) => finalizeRun(exit.reason), - }), - Effect.ensuring(Scope.close(runScope, Exit.void).pipe(Effect.ignore)), + // Caller holds the instance mutex. The hook waits for that mutex so it + // cannot run (or call back into start/stop) until the caller releases it. + const armStartupFailure = ( + reason: string, + failedConfig: DesktopBackendStartConfig, + runToReplace: Option.Option, + ): Effect.Effect => + Effect.gen(function* () { + const onStartupFailed = spec.onStartupFailed; + if (onStartupFailed === undefined) { + yield* Ref.update(state, (latest) => + latest.startupFailurePending ? { ...latest, startupFailurePending: false } : latest, ); + return; + } - const fiber = yield* Effect.forkIn(program, parentScope); - yield* updateActiveRun(runId, (run) => ({ - ...run, - fiber: Option.some(fiber), - })); - }), - ), - ).pipe(Effect.withSpan("desktop.backendInstance.start", { attributes: { id: spec.id } })); + const tracked: { fiber?: Fiber.Fiber } = {}; + let followUpStarted = false; + const fiber = yield* Effect.forkIn( + mutex + .withPermits(1)(Effect.void) + .pipe( + Effect.andThen( + Effect.gen(function* () { + const decision = yield* onStartupFailed(reason, failedConfig).pipe(Effect.exit); + if (Exit.isFailure(decision)) { + if (Cause.hasInterruptsOnly(decision.cause)) return; + yield* logInstanceError("desktop backend startup failure hook failed", { + cause: Cause.pretty(decision.cause), + }); + } + const shouldReplace = Exit.isSuccess(decision) && decision.value; + const trackedFiber = tracked.fiber; + if (trackedFiber === undefined) return; + // Detach the restart from this fiber. stop() interrupts the hook + // while a dialog is up; it must not cancel a replacement that + // already decided to proceed, and the replacement's own stop() + // must not interrupt itself. + yield* Effect.forkIn( + Effect.gen(function* () { + if (Option.isSome(runToReplace)) { + if (shouldReplace) { + yield* replaceRun(runToReplace.value); + } + return; + } + if (!(yield* Ref.get(state)).desiredRunning) return; + yield* scheduleRestart(reason); + }).pipe(Effect.ensuring(clearTrackedStartupFailure(trackedFiber))), + parentScope, + ); + followUpStarted = true; + }), + ), + Effect.ensuring( + Effect.suspend(() => { + const trackedFiber = tracked.fiber; + if (followUpStarted || trackedFiber === undefined) return Effect.void; + return clearTrackedStartupFailure(trackedFiber); + }), + ), + ), + parentScope, + ); + tracked.fiber = fiber; + yield* Ref.update(state, (latest) => ({ + ...latest, + startupFailureFiber: Option.some(fiber), + startupFailurePending: true, + })); + }); + + // Stops the unreachable run, then starts again so configResolve sees + // whatever onStartupFailed changed. A stop from anywhere else bumps + // stopGeneration and the replacement start no-ops. + const replaceRun = Effect.fn("desktop.backendInstance.replaceUnreadyRun")(function* ( + runId: number, + ) { + const current = yield* Ref.get(state); + if ( + Option.getOrUndefined(current.active)?.id !== runId || + !current.desiredRunning || + current.ready + ) { + return; + } + const expectedStopGeneration = current.stopGeneration + 1; + yield* stop(); + yield* beginStart({ expectedStopGeneration }); + }); const scheduleRestart = Effect.fn("desktop.backendInstance.scheduleRestart")(function* ( reason: string, @@ -1048,7 +1247,9 @@ export const makeBackendInstance = Effect.fn("makeBackendInstance")(function* ( const stop = Effect.fn("desktop.backendInstance.stop")(function* (options?: { readonly timeout?: Duration.Duration; }) { - const { active, restartFiber, notifyShutdown } = yield* mutex.withPermits(1)( + const { active, restartFiber, startupFailureFiber, notifyShutdown } = yield* mutex.withPermits( + 1, + )( Effect.gen(function* () { const result = yield* Ref.modify(state, (latest) => { const active = Option.map(latest.active, (run) => @@ -1058,6 +1259,7 @@ export const makeBackendInstance = Effect.fn("makeBackendInstance")(function* ( { active, restartFiber: latest.restartFiber, + startupFailureFiber: latest.startupFailureFiber, notifyShutdown: latest.ready, }, { @@ -1066,6 +1268,9 @@ export const makeBackendInstance = Effect.fn("makeBackendInstance")(function* ( ready: false, active, restartFiber: Option.none>(), + startupFailureFiber: Option.none>(), + startupFailurePending: false, + stopGeneration: latest.stopGeneration + 1, }, ] as const; }); @@ -1080,6 +1285,10 @@ export const makeBackendInstance = Effect.fn("makeBackendInstance")(function* ( onNone: () => Effect.void, onSome: (fiber) => Fiber.interrupt(fiber).pipe(Effect.asVoid), }); + yield* Option.match(startupFailureFiber, { + onNone: () => Effect.void, + onSome: (fiber) => Fiber.interrupt(fiber).pipe(Effect.asVoid), + }); yield* Option.match(active, { onNone: () => Effect.void, onSome: (run) => diff --git a/apps/desktop/src/backend/DesktopBackendPool.test.ts b/apps/desktop/src/backend/DesktopBackendPool.test.ts index 8373437ae312..3c495fdb9a85 100644 --- a/apps/desktop/src/backend/DesktopBackendPool.test.ts +++ b/apps/desktop/src/backend/DesktopBackendPool.test.ts @@ -1,12 +1,17 @@ import { assert, describe, it } from "@effect/vitest"; +import * as Deferred from "effect/Deferred"; import * as Duration from "effect/Duration"; import * as Effect from "effect/Effect"; import * as FileSystem from "effect/FileSystem"; import * as Layer from "effect/Layer"; import * as Option from "effect/Option"; +import * as Queue from "effect/Queue"; import * as Ref from "effect/Ref"; +import * as Scope from "effect/Scope"; +import * as Sink from "effect/Sink"; import * as Stream from "effect/Stream"; -import { HttpClient } from "effect/unstable/http"; +import * as TestClock from "effect/testing/TestClock"; +import { HttpClient, HttpClientRequest, HttpClientResponse } from "effect/unstable/http"; import { ChildProcessSpawner } from "effect/unstable/process"; import * as DesktopObservability from "../app/DesktopObservability.ts"; @@ -154,4 +159,212 @@ describe("DesktopBackendPool", () => { }), ), ); + + it.effect("uses Windows for this launch when a wsl-only primary keeps crashing on startup", () => + Effect.scoped( + Effect.gen(function* () { + const mode = yield* Ref.make<"wsl" | "windows">("wsl"); + const spawns = yield* Queue.unbounded<"wsl" | "windows">(); + const dialogs = yield* Queue.unbounded<{ + readonly title: string; + readonly content: string; + }>(); + const ready = yield* Deferred.make(); + const wslConfig: DesktopBackendStartConfig = { + executablePath: "/electron", + args: ["/server/bin.mjs"], + entryPath: "/server/bin.mjs", + cwd: "/server", + env: {}, + bootstrap: { + mode: "desktop", + noBrowser: true, + port: 3773, + t3Home: "/tmp/t3", + host: "127.0.0.1", + desktopBootstrapToken: "token", + tailscaleServeEnabled: false, + tailscaleServePort: 443, + desktopTelemetryFd: 4, + desktopTelemetryControlFd: 5, + }, + bootstrapDelivery: "fd3", + extendEnv: false, + httpBaseUrl: new URL("http://172.17.0.1:3773"), + captureOutput: false, + preflightFailure: Option.none(), + runningDistro: "Ubuntu-22.04", + }; + const { runningDistro: _ignoredRunningDistro, ...windowsBase } = wslConfig; + const windowsConfig: DesktopBackendStartConfig = { + ...windowsBase, + extendEnv: true, + httpBaseUrl: new URL("http://127.0.0.1:3773"), + }; + + const settingsLayer = DesktopAppSettings.layerTest({ + ...DesktopAppSettings.DEFAULT_DESKTOP_SETTINGS, + wslBackendEnabled: true, + wslOnly: true, + wslDistro: "Ubuntu-22.04", + }); + const configurationLayer = Layer.effect( + DesktopBackendConfiguration.DesktopBackendConfiguration, + Effect.gen(function* () { + const settings = yield* DesktopAppSettings.DesktopAppSettings; + return { + resolvePrimary: Effect.gen(function* () { + const current = yield* settings.get; + const next = current.wslOnly && current.wslBackendEnabled ? "wsl" : "windows"; + yield* Ref.set(mode, next); + return next === "wsl" ? wslConfig : windowsConfig; + }), + resolvePrimaryLabel: Effect.succeed("WSL (Ubuntu-22.04)"), + resolveWsl: () => Effect.die("unexpected WSL config resolve"), + } satisfies DesktopBackendConfiguration.DesktopBackendConfiguration["Service"]; + }), + ); + const stack = DesktopBackendPool.layer.pipe( + Layer.provide(configurationLayer), + Layer.provideMerge(settingsLayer), + Layer.provide( + Layer.mergeAll( + FileSystem.layerNoop({ exists: () => Effect.succeed(true) }), + Layer.succeed( + ChildProcessSpawner.ChildProcessSpawner, + ChildProcessSpawner.make(() => + Effect.gen(function* () { + const current = yield* Ref.get(mode); + yield* Queue.offer(spawns, current); + if (current === "windows") { + const scope = yield* Scope.Scope; + const exited = yield* Deferred.make(); + yield* Scope.addFinalizer(scope, Deferred.succeed(exited, void 0)); + return exitedProcess( + Deferred.await(exited).pipe(Effect.as(ChildProcessSpawner.ExitCode(0))), + ); + } + return exitedProcess(Effect.succeed(ChildProcessSpawner.ExitCode(1))); + }), + ), + ), + Layer.succeed( + HttpClient.HttpClient, + HttpClient.make((request) => + Effect.succeed( + responseFor(request, request.url.includes("127.0.0.1") ? 200 : 503), + ), + ), + ), + Layer.succeed(DesktopObservability.DesktopBackendOutputLogFactory, { + forInstance: () => + Effect.succeed({ + beginSession: () => Effect.void, + writeOutputChunk: () => Effect.void, + persistFailureSnapshot: () => Effect.void, + persistFailure: () => Effect.void, + discardSession: Effect.void, + } satisfies DesktopObservability.DesktopBackendOutputLogShape), + } satisfies DesktopObservability.DesktopBackendOutputLogFactory["Service"]), + Layer.succeed(DesktopTelemetryPublisher.DesktopTelemetryPublisher, { + latest: Effect.succeedNone, + changes: Stream.empty, + encoded: Stream.empty, + handleControlForSource: () => Effect.void, + removeControlSource: () => Effect.void, + publishUpdateReport: () => Effect.void, + updateRequests: Stream.empty, + updateCommits: Stream.empty, + updateCancellations: Stream.empty, + }), + DesktopWslEnvironment.layerTest(), + Layer.succeed(ElectronDialog.ElectronDialog, { + pickFolder: () => Effect.die("unexpected folder picker"), + pickFiles: () => Effect.die("unexpected file picker"), + showMessageBox: () => Effect.die("unexpected message box"), + showErrorBox: (title, content) => + Queue.offer(dialogs, { title, content }).pipe(Effect.asVoid), + } satisfies ElectronDialog.ElectronDialog["Service"]), + Layer.succeed(DesktopWindow.DesktopWindow, { + createMain: Effect.die("unexpected window create"), + ensureMain: Effect.die("unexpected window ensure"), + revealOrCreateMain: Effect.die("unexpected window reveal"), + activate: Effect.die("unexpected window activate"), + createMainIfBackendReady: Effect.die("unexpected window create"), + showConnectingSplash: Effect.void, + handleBackendReady: () => Deferred.succeed(ready, void 0).pipe(Effect.asVoid), + handleBackendNotReady: Effect.void, + flushMainWindowBounds: Effect.void, + prepareCaptureReveal: Effect.void, + dispatchMenuAction: () => Effect.die("unexpected menu action"), + dispatchSnapShotEvent: () => Effect.void, + zoomMain: () => Effect.die("unexpected zoom"), + syncAppearance: Effect.void, + } satisfies DesktopWindow.DesktopWindow["Service"]), + ), + ), + ); + yield* Effect.gen(function* () { + const pool = yield* DesktopBackendPool.DesktopBackendPool; + const settings = yield* DesktopAppSettings.DesktopAppSettings; + const primary = yield* pool.primary; + + yield* primary.start; + assert.equal(yield* Queue.take(spawns), "wsl"); + yield* TestClock.adjust(Duration.millis(500)); + assert.equal(yield* Queue.take(spawns), "wsl"); + yield* TestClock.adjust(Duration.seconds(1)); + assert.equal(yield* Queue.take(spawns), "wsl"); + + const dialog = yield* Queue.take(dialogs); + assert.equal(dialog.title, "WSL backend isn't responding"); + assert.include(dialog.content, "Ubuntu-22.04"); + assert.include(dialog.content, "code=1"); + assert.include(dialog.content, "Windows backend for this launch"); + + let restartScheduled = false; + while (!restartScheduled) { + restartScheduled = (yield* primary.snapshot).restartScheduled; + if (!restartScheduled) { + yield* Effect.yieldNow; + } + } + yield* TestClock.adjust(Duration.seconds(2)); + assert.equal(yield* Queue.take(spawns), "windows"); + yield* Deferred.await(ready); + + const recovered = yield* settings.get; + assert.equal(recovered.wslOnly, false); + assert.equal(recovered.wslBackendEnabled, false); + assert.equal(recovered.wslDistro, "Ubuntu-22.04"); + assert.equal((yield* primary.snapshot).ready, true); + }).pipe(Effect.provide(stack)); + }).pipe(Effect.provide(TestClock.layer())), + ), + ); }); + +function responseFor( + request: HttpClientRequest.HttpClientRequest, + status: number, +): HttpClientResponse.HttpClientResponse { + return HttpClientResponse.fromWeb(request, new Response(null, { status })); +} + +function exitedProcess( + exitCode: Effect.Effect, +): ChildProcessSpawner.ChildProcessHandle { + return ChildProcessSpawner.makeHandle({ + pid: ChildProcessSpawner.ProcessId(123), + stdout: Stream.empty, + stderr: Stream.empty, + all: Stream.empty, + exitCode, + isRunning: Effect.succeed(false), + kill: () => Effect.void, + stdin: Sink.drain, + getInputFd: () => Sink.drain, + getOutputFd: () => Stream.empty, + unref: Effect.succeed(Effect.void), + }); +} diff --git a/apps/desktop/src/backend/DesktopBackendPool.ts b/apps/desktop/src/backend/DesktopBackendPool.ts index 2930780d4102..69f81d9e1d34 100644 --- a/apps/desktop/src/backend/DesktopBackendPool.ts +++ b/apps/desktop/src/backend/DesktopBackendPool.ts @@ -277,6 +277,33 @@ export const layer = Layer.effect( }, ); + // Preflight passed, but the primary still never became ready: it keeps + // exiting, or it stays up and the readiness probe keeps timing out. The + // connecting splash has no controls, so wsl-only mode would sit there + // forever. Use Windows for this launch only — the same in-memory fallback + // as a bounded preflight failure — and try WSL again on the next launch. + // A run that already resolved to Windows (no distro) has nothing to fall + // back to and keeps its existing restart loop. + const handlePrimaryStartupFailure = Effect.fn("desktop.backendPool.primaryStartupFailed")( + function* (reason: string, config: DesktopBackendManager.DesktopBackendStartConfig) { + if (config.runningDistro === undefined) return false; + const detail = `The WSL backend (${config.runningDistro}) did not become ready (${reason}).`; + yield* logBackendPoolWarning( + "primary WSL backend did not become ready; using Windows for this launch", + { reason, distro: config.runningDistro }, + ); + // A failed dialog must not keep the splash up. + yield* electronDialog + .showErrorBox( + "WSL backend isn't responding", + `${detail}\n\nT3 Code will use the Windows backend for this launch and retry WSL the next time the app starts.`, + ) + .pipe(Effect.ignoreCause); + yield* appSettings.applyWslWindowsFallbackInMemory; + return true; + }, + ); + const primary = yield* DesktopBackendManager.makeBackendInstance({ id: DesktopBackendManager.PRIMARY_INSTANCE_ID, // Keep this lazy. The pool layer is initialized before startup loads @@ -301,6 +328,7 @@ export const layer = Layer.effect( ), onShutdown: () => desktopWindow.handleBackendNotReady, onPreflightFailed: handlePrimaryPreflightFailure, + onStartupFailed: handlePrimaryStartupFailure, }); const instancesRef = yield* SynchronizedRef.make<