From f7d17c120218900d2cfb3820e4d5d3f236457e79 Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Thu, 1 Oct 2026 09:51:36 -0400 Subject: [PATCH 1/6] build(local-ai): bump managed llama-server runtime to b11320 The RTX Spark recipe set published on 2026-09-30 raises its minimum llama.cpp build: the Qwen3.6-35B-A3B recipes require b11236, the Qwen3.8-27B DFlash recipes require b11229, and Qwen3.8 Flash-Next requires b11256. A single managed runtime is pinned for every recipe, so the pin moves to b11320, the newest build carrying all four Windows CUDA 13.4 artifacts, which clears every one of those minimums. Artifact sizes and digests are taken from the release and verified by downloading each llama-server archive and recomputing SHA-256. The CUDA runtime archives are byte-identical to the ones already pinned at b11026, so their digests are unchanged. b11026 moves into the retired set alongside b10655. It is a shipped runtime once this release goes out, so an installation recorded against it has to keep resolving its own receipt and stay launchable until setup upgrades it. --- .../Inference/Catalog/LlamaRuntimeCatalog.cs | 147 ++++++++++++++++-- .../LocalAiInstallRecoveryTests.cs | 4 +- .../LocalInferenceQualificationTests.cs | 27 ++++ 3 files changed, 161 insertions(+), 17 deletions(-) diff --git a/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs index 4ddccf335..90500efda 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs @@ -62,12 +62,12 @@ public LlamaRuntimeVariant( /// public static class LlamaRuntimeCatalog { - public const string ReleaseTag = "b11026"; - public const string ReleaseCommitSha = "b49650adb31f2e49a0d76113aeb1792134fd8413"; + public const string ReleaseTag = "b11320"; + public const string ReleaseCommitSha = "b8f96c3e82284028cb077811ed1666caac3c5bac"; public const string ServerExecutableName = "llama-server.exe"; public const string ServerImplementationLibraryName = "llama-server-impl.dll"; - public const string X64RuntimeId = "b11026-cuda13-x64"; - public const string Arm64RuntimeId = "b11026-cuda13-arm64"; + public const string X64RuntimeId = "b11320-cuda13-x64"; + public const string Arm64RuntimeId = "b11320-cuda13-arm64"; public static GitHubReleaseSource Source { get; } = new( "ggml-org/llama.cpp", @@ -85,13 +85,13 @@ public static class LlamaRuntimeCatalog new[] { RuntimeArtifact( - "llama-b11026-cuda13-x64", + "llama-b11320-cuda13-x64", ArtifactRole.RuntimeBinary, - "llama-b11026-bin-win-cuda-13.4-x64.zip", - 150_102_391, - "6799f0962d066c54aee3773f0e5efa0076e46418695c0f4f6d24a38e7007dfb1"), + "llama-b11320-bin-win-cuda-13.4-x64.zip", + 152_787_584, + "75afa9d56077ebbd7a6123e923e8b6e82d4f4b1e53947ce04ca14ccbbfbba9c0"), RuntimeArtifact( - "cudart-b11026-cuda13-x64", + "cudart-b11320-cuda13-x64", ArtifactRole.RuntimeDependency, "cudart-llama-bin-win-cuda-13.4-x64.zip", 423_535_356, @@ -106,13 +106,13 @@ public static class LlamaRuntimeCatalog new[] { RuntimeArtifact( - "llama-b11026-cuda13-arm64", + "llama-b11320-cuda13-arm64", ArtifactRole.RuntimeBinary, - "llama-b11026-bin-win-cuda-13.4-arm64.zip", - 142_993_717, - "d4a31d05b4fe997872020d81e8482e7712c7c254ec9c7ccdb9597ae2a31e6728"), + "llama-b11320-bin-win-cuda-13.4-arm64.zip", + 144_946_754, + "9fcf3fb79c7d107b2fc60cff6e5947b133c965ca071ba0f8b05f641308b8e660"), RuntimeArtifact( - "cudart-b11026-cuda13-arm64", + "cudart-b11320-cuda13-arm64", ArtifactRole.RuntimeDependency, "cudart-llama-bin-win-cuda-13.4-arm64.zip", 153_262_407, @@ -125,6 +125,15 @@ public static class LlamaRuntimeCatalog // managed installation recorded before the runtime bump keeps resolving its own // receipt and stays launchable until setup upgrades it. Pins are reproduced // exactly as they were installed; nothing is remapped. + private const string LegacyB11026ReleaseTag = "b11026"; + private const string LegacyB11026RuntimeIdX64 = "b11026-cuda13-x64"; + private const string LegacyB11026RuntimeIdArm64 = "b11026-cuda13-arm64"; + + private static GitHubReleaseSource LegacyB11026Source { get; } = new( + "ggml-org/llama.cpp", + LegacyB11026ReleaseTag, + "b49650adb31f2e49a0d76113aeb1792134fd8413"); + private const string LegacyB10655ReleaseTag = "b10655"; private const string LegacyB10655RuntimeIdX64 = "b10655-cuda13-x64"; private const string LegacyB10655RuntimeIdArm64 = "b10655-cuda13-arm64"; @@ -179,6 +188,50 @@ public static class LlamaRuntimeCatalog "5a40dc7c5fa3d0a80ceeba4f16f9e8d25d87bcf1399c9233588953c43436c33c"), }), LegacyB10655ReleaseTag), + new LlamaRuntimeVariant( + LegacyB11026RuntimeIdX64, + Architecture.X64, + new Version(13, 4), + Array.AsReadOnly( + new[] + { + LegacyB11026Artifact( + "llama-b11026-cuda13-x64", + ArtifactRole.RuntimeBinary, + "llama-b11026-bin-win-cuda-13.4-x64.zip", + 150_102_391, + "6799f0962d066c54aee3773f0e5efa0076e46418695c0f4f6d24a38e7007dfb1"), + LegacyB11026Artifact( + "cudart-b11026-cuda13-x64", + ArtifactRole.RuntimeDependency, + "cudart-llama-bin-win-cuda-13.4-x64.zip", + 423_535_356, + "738f8c251ac22b70c3ae6f83a10cf222725df0395246a2cf58f32bdb85fbe668"), + }), + LegacyB11026ReleaseTag, + requiredFiles: LegacyB11026X64RuntimeFiles()), + new LlamaRuntimeVariant( + LegacyB11026RuntimeIdArm64, + Architecture.Arm64, + new Version(13, 4), + Array.AsReadOnly( + new[] + { + LegacyB11026Artifact( + "llama-b11026-cuda13-arm64", + ArtifactRole.RuntimeBinary, + "llama-b11026-bin-win-cuda-13.4-arm64.zip", + 142_993_717, + "d4a31d05b4fe997872020d81e8482e7712c7c254ec9c7ccdb9597ae2a31e6728"), + LegacyB11026Artifact( + "cudart-b11026-cuda13-arm64", + ArtifactRole.RuntimeDependency, + "cudart-llama-bin-win-cuda-13.4-arm64.zip", + 153_262_407, + "642dcde8805b3e3165ca710a5443b3b4044b27d96bd3ee3132473988c9bcb774"), + }), + LegacyB11026ReleaseTag, + requiredFiles: LegacyB11026Arm64RuntimeFiles()), }); public static IReadOnlyList Variants => s_variants; @@ -200,6 +253,55 @@ public static class LlamaRuntimeCatalog ?? s_legacyVariants.SingleOrDefault(variant => string.Equals(variant.Id, id, StringComparison.Ordinal)); private static IReadOnlyList X64RuntimeFiles() => + [ + RuntimeFile("cublas64_13.dll", 54_942_320, "1119dbca0a808e0c8850bb4221e330daf0b5ddb348d62d37b6ac533854723df9"), + RuntimeFile("cublasLt64_13.dll", 492_752_496, "0f5bc315fef706b5626248ebd74e8a48a9e82fa2f0cad38330f5f70984913107"), + RuntimeFile("cudart64_13.dll", 551_024, "05bfafcb97bd53b0089568a96e1e5bd6921637bd0853ff7f5342bd31a6a16890"), + RuntimeFile("ggml-base.dll", 801_280, "777562364f0528eaadd7865faf0e6e0b17ff3904e50a18438caebc4bfd0790e3"), + RuntimeFile("ggml-cpu-alderlake.dll", 1_431_552, "6201c30f8b96e131654d196125c659d554ba478294c0703da0b7a20a89618122"), + RuntimeFile("ggml-cpu-cannonlake.dll", 1_657_856, "39a18866d22962cf20c604c93f877b351bd280d798d794c8e8905d03ca73c228"), + RuntimeFile("ggml-cpu-cascadelake.dll", 1_642_496, "e74f4902fa3bd58ccdfe75c36517a33574eeabb370e18b7ca803b9ee3f0e5113"), + RuntimeFile("ggml-cpu-cooperlake.dll", 1_643_520, "65bde4b5d95ff05523dbd3a403abbf47de569fae92708a38ebfd8abfa934a64c"), + RuntimeFile("ggml-cpu-haswell.dll", 1_436_160, "a54e0bb576b0b17453d0db649d0f367b59aeeaab9f5e394df501891dc7437142"), + RuntimeFile("ggml-cpu-icelake.dll", 1_649_152, "4734e2f5ad4a9003e0345d2dd785598fd8ea1af1f23408316b14aa54f2fbd855"), + RuntimeFile("ggml-cpu-ivybridge.dll", 1_307_136, "a6bc4be1e4cec08406fddead48feafc29dc5cc2ee3d4a30367335dc0ec277b8b"), + RuntimeFile("ggml-cpu-piledriver.dll", 1_310_208, "e86eb4d9c67a713b3a9faf7ffd430fc933d88fa993c150ba15fcc1ccc9edf334"), + RuntimeFile("ggml-cpu-sandybridge.dll", 1_286_656, "b2a4ac1f2a4391a707b23848a10326e42d86abf6e31ab3f1aa31eac8ff1a5f0c"), + RuntimeFile("ggml-cpu-sapphirerapids.dll", 1_920_000, "85b7c5163e8be5576e2bc87678c2a979c6d85b70be1b78317f0e7ee7324ff87e"), + RuntimeFile("ggml-cpu-skylakex.dll", 1_651_712, "572c006b7ab4df21f35f469ff7b20b9959a97097d2edb3f1fb28ee834c4c8f2f"), + RuntimeFile("ggml-cpu-sse42.dll", 916_480, "de669e22953f32567c0512199448fbd3259f96bd90b1d41967d42479a6d4f1be"), + RuntimeFile("ggml-cpu-x64.dll", 908_800, "95660d872ed83b6335708dbf539d1b15ca60ea5832a0aed59e9c6d38d6ed468e"), + RuntimeFile("ggml-cpu-zen4.dll", 1_649_664, "e75dea4fc377ffe75096cdca48cd26fc4777f4ace76c94483a2414c8c41b7b02"), + RuntimeFile("ggml-cuda.dll", 147_597_824, "af6121ddd035db3fddb90bdc7dbbef8bfb244db522157af459b9682f0d18934e"), + RuntimeFile("ggml-rpc.dll", 168_448, "29cd136e276a998b7d1ec43b5dd2d6d9dee837c38bbae73ffbc02e015584c740"), + RuntimeFile("ggml.dll", 79_872, "96a162e0be5a3c798af6b31a5d21c1bd04493914e7b11ae8f9212d0bf0996e6c"), + RuntimeFile("libomp.dll", 768_000, "a12116ba72d1d6820407cf30be23da04ce79d6bb8a71a5ee71759c5a1faa6f1c"), + RuntimeFile("llama-common.dll", 7_891_968, "a6a87f92224a11a562f5397554119882d81b9e0038ea98cf8aa4edc34f3d6d2d"), + RuntimeFile("llama-server-impl.dll", 8_946_688, "6dfeb749b21b6cd3f4a64d0319b75a090be3d79f222053239ac6a4527a5386ed"), + RuntimeFile("llama-server.exe", 9_216, "96bfefe33c2cf0f1f421cae2f9e1e4a385264c8d4d731f7b08ff880edbb4f133"), + RuntimeFile("llama.dll", 3_270_656, "cfbf1a2ac5d676ccb3c7685098877f832a90285ad21575faf5a71eefece7f759"), + RuntimeFile("mtmd.dll", 1_794_560, "c29eacab7497aa8b18951ea2e7466c8fd8867ff49d122322a203e8d6c7a87dfa"), + ]; + + private static IReadOnlyList Arm64RuntimeFiles() => + [ + RuntimeFile("cublas64_13.dll", 24_207_984, "49e8fa23d88ac0cbae27e200fc092b894dd55b4de5cb4293993986e29fb9b65a"), + RuntimeFile("cublasLt64_13.dll", 193_128_560, "daf579ae36bb3c85e2340c57a1994f634e07c6f55ef70dd4cdc26695de21c752"), + RuntimeFile("cudart64_13.dll", 606_832, "bd927ddf03823eeead7c8b261a8669d96764b89109311f4ab6accde8b7d97ec2"), + RuntimeFile("ggml-base.dll", 664_576, "4817c34746df9b1b519d3da9140e61752c4a852a4d1e9df63dabd8ae069bd03e"), + RuntimeFile("ggml-cpu.dll", 834_560, "2896d13bf7b05f80b0f4f1e4d1f85c19c3ebc207757117ea709a8e4dfe0914f7"), + RuntimeFile("ggml-cuda.dll", 145_212_928, "c877fcb29338139d502065d56bf93e1bbd8b3092ddc2dd032bd47b6395b2b8b3"), + RuntimeFile("ggml-rpc.dll", 155_648, "7b237414f5739105c41cf55cd4ad9e7c14f13ba417253d7bb3bfac961406a9cb"), + RuntimeFile("ggml.dll", 70_656, "09f47cabdce4c7efd13b68b5f4206b276516f89e09b946460b048c7eb5d215ff"), + RuntimeFile("libomp.dll", 764_928, "26caae17f29aaf2238f664375b663cd306d596bc9e36e780fa88356c51fe876a"), + RuntimeFile("llama-common.dll", 6_983_680, "fca38964f724eb153bb0d35926a397cfa010142cbeb6649f63d13a64b090bfc9"), + RuntimeFile("llama-server-impl.dll", 8_201_728, "cb751a06de6ac4fef398704c26807ff2a1740d6711d9f2ab7318ccb27d3e77f9"), + RuntimeFile("llama-server.exe", 9_728, "67a7789b17b2ec98ca87c079bf6686da7e5b9e56d4d9d6df8c3a08d4b654b6b5"), + RuntimeFile("llama.dll", 2_864_640, "9d18dfdd61e5f027a1763196f6c0d258c3e2307e6f4886da51679317dd779d8c"), + RuntimeFile("mtmd.dll", 1_507_328, "d47dc5f6441efa1de3176f60da2c5da28c4da411807be6c6b8a9c054f61cb37a"), + ]; + + private static IReadOnlyList LegacyB11026X64RuntimeFiles() => [ RuntimeFile("cublas64_13.dll", 54_942_320, "1119dbca0a808e0c8850bb4221e330daf0b5ddb348d62d37b6ac533854723df9"), RuntimeFile("cublasLt64_13.dll", 492_752_496, "0f5bc315fef706b5626248ebd74e8a48a9e82fa2f0cad38330f5f70984913107"), @@ -230,7 +332,7 @@ private static IReadOnlyList X64RuntimeFiles() => RuntimeFile("mtmd.dll", 1_772_032, "131d3ddce28051fd48ca07a0fa9128fb35996631afa516e43ecc5094b7aa402f"), ]; - private static IReadOnlyList Arm64RuntimeFiles() => + private static IReadOnlyList LegacyB11026Arm64RuntimeFiles() => [ RuntimeFile("cublas64_13.dll", 24_207_984, "49e8fa23d88ac0cbae27e200fc092b894dd55b4de5cb4293993986e29fb9b65a"), RuntimeFile("cublasLt64_13.dll", 193_128_560, "daf579ae36bb3c85e2340c57a1994f634e07c6f55ef70dd4cdc26695de21c752"), @@ -266,6 +368,21 @@ private static PinnedArtifact RuntimeArtifact( new Sha256Digest(sha256), LocalInferenceCatalogProvenance.NvidiaCair); + private static PinnedArtifact LegacyB11026Artifact( + string id, + ArtifactRole role, + string fileName, + long sizeBytes, + string sha256) => + new( + id, + role, + LegacyB11026Source, + fileName, + sizeBytes, + new Sha256Digest(sha256), + LocalInferenceCatalogProvenance.NvidiaCair); + private static PinnedArtifact LegacyB10655Artifact( string id, ArtifactRole role, diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs index 1210020b6..9276f3207 100644 --- a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs +++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs @@ -1433,7 +1433,7 @@ public async Task RuntimeUpgrade_MigratesModelAndRestoresOriginalReceiptOnFailur new UpgradeCheckpointStep("after-persist", ctx => { LocalAiResolvedInstall upgraded = Assert.IsType(ctx.LocalAiResolvedInstall); - Assert.Equal("b11026", upgraded.Manifest.EngineVersion); + Assert.Equal(LlamaRuntimeCatalog.ReleaseTag, upgraded.Manifest.EngineVersion); Assert.Equal(LocalAiInstallManifest.HubCacheReceiptSchemaVersion, upgraded.Manifest.SchemaVersion); Assert.Equal(cacheRoot, upgraded.Manifest.ModelCacheRoot); Assert.Equal(cachedModel, upgraded.ModelPath); @@ -1462,7 +1462,7 @@ public async Task RuntimeUpgrade_MigratesModelAndRestoresOriginalReceiptOnFailur if (failureStage is null) { Assert.Equal(newExecutable, persisted.ExecutablePath); - Assert.Equal("b11026", persisted.Manifest.EngineVersion); + Assert.Equal(LlamaRuntimeCatalog.ReleaseTag, persisted.Manifest.EngineVersion); } else { diff --git a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs index d1da88ca9..1b1790b9a 100644 --- a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs +++ b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs @@ -827,6 +827,8 @@ private static GpuInfo Gpu( [Theory] [InlineData("b10655-cuda13-x64", "b10655")] [InlineData("b10655-cuda13-arm64", "b10655")] + [InlineData("b11026-cuda13-x64", "b11026")] + [InlineData("b11026-cuda13-arm64", "b11026")] public void FindInstalled_ResolvesRetiredRuntimeSoExistingInstallsStayLaunchable( string runtimeId, string expectedReleaseTag) @@ -883,4 +885,29 @@ public void FindInstalled_RejectsUnknownRuntimeId() Assert.Null(LlamaRuntimeCatalog.FindInstalled("b00000-cuda13-x64")); Assert.Null(LlamaRuntimeCatalog.FindInstalled(null)); } + + /// + /// The receipt published alpha.78 actually writes, read back off an x64 install of + /// that build. Both halves have to resolve together: the reconciler looks the + /// runtime up by runtimeId and the model up by modelCatalogId, and + /// rejects the install outright if either lookup comes back empty. Pruning one of + /// these entries would strand every published install behind a recipe-mismatch + /// error, so pin the pair rather than the two ids separately. + /// + [Fact] + public void PublishedAlphaReceipt_StillResolvesAfterTheRuntimeBump() + { + LlamaRuntimeVariant? runtime = LlamaRuntimeCatalog.FindInstalled("b11026-cuda13-x64"); + LocalModelInfo? model = LocalModelCatalog.FindInstalled(LocalModelCatalog.Qwen38_27BModelId); + + Assert.NotNull(runtime); + Assert.Equal("b11026", runtime.ReleaseTag); + Assert.NotNull(model); + + // The model is still the current recommendation for a dGPU box, so only the + // runtime half is an upgrade. That asymmetry is the point: the install is + // reusable as-is, and the 16.5 GB weights never need re-acquiring. + Assert.NotEqual(LlamaRuntimeCatalog.ReleaseTag, runtime.ReleaseTag); + Assert.False(LocalModelCatalog.IsLegacy(LocalModelCatalog.Qwen38_27BModelId)); + } } From 19e0f9a3ccc12e452d83c382d3fc1b8942c1b361 Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Thu, 1 Oct 2026 09:55:50 -0400 Subject: [PATCH 2/6] feat(local-ai): move the RTX Spark 48GB recipe to the Q4_K_S quantization The 2026-09-30 recipe set replaces the 48GB-SKU Qwen3.6-35B-A3B quantization: UD-IQ4_XS becomes UD-Q4_K_S. Both files already exist at the Hugging Face revision this catalog pins, so only the artifact changes; the revision, run recipe, and 98,304-token context tier are untouched. The previous quantization is retired rather than rewritten in place. It ships as the 48GB default, so an existing receipt has to keep resolving its own pinned artifact and profile until setup upgrades it. Retired entries normally expose only the pre-profile native/F16 profile, which this model was never installed under, so the retired entry keeps the 48GB SKU's fixed tier instead. A regression covers that and fails without it. --- .../Inference/Catalog/LocalModelCatalog.cs | 52 +++++++++++++++---- .../Catalog/RtxSparkInferenceSelector.cs | 2 +- .../LocalInferenceQualificationTests.cs | 42 ++++++++++++++- 3 files changed, 83 insertions(+), 13 deletions(-) diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs index e15b59ba8..4a476f523 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs @@ -156,6 +156,12 @@ public static class LocalModelCatalog /// public const string Qwen9BModelId = "qwen3.5-9b-mtp-q4-k-m"; /// RTX Spark 48GB-SKU recipe. Never offered on the generic dGPU path; see RtxSparkInferenceSelector. + public const string Qwen35B_Q4KSModelId = "qwen3.6-35b-a3b-mtp-ud-q4-k-s"; + /// + /// Retired from new installs: the 2026-09-30 recipe set replaced this quantization + /// with . Retained only so an already-installed + /// managed receipt keeps resolving and launching across upgrade. + /// public const string Qwen35B_IQ4XSModelId = "qwen3.6-35b-a3b-mtp-ud-iq4-xs"; /// RTX Spark 128GB-SKU default recipe. Never offered on the generic dGPU path; see RtxSparkInferenceSelector. public const string Qwen38_27B_DFlashModelId = "qwen3.8-27b-dflash-ud-q4-k-m"; @@ -163,7 +169,7 @@ public static class LocalModelCatalog public const int IntermediateContextTokens = 196_608; public const int ReducedContextTokens = 131_072; public const int MinimumContextTokens = 65_536; - /// RTX Spark 48GB-SKU context tier (98,304 tokens); see . + /// RTX Spark 48GB-SKU context tier (98,304 tokens); see . public const int RtxSpark48GbContextTokens = 98_304; // Measured-conservative allowances for compute buffers, recurrent state, @@ -264,16 +270,16 @@ public static class LocalModelCatalog // comment) so the always-alternative-only ones can't still win by // tie-break/fallback ordering among themselves. new LocalModelInfo( - Qwen35B_IQ4XSModelId, - "Qwen3.6 35B-A3B (UD-IQ4_XS)", + Qwen35B_Q4KSModelId, + "Qwen3.6 35B-A3B (UD-Q4_K_S)", "Qwen3.6", - "UD-IQ4_XS", + "UD-Q4_K_S", ModelArtifact( - Qwen35B_IQ4XSModelId, + Qwen35B_Q4KSModelId, s_qwen35BSource, - "Qwen3.6-35B-A3B-UD-IQ4_XS.gguf", - 18_209_036_576, - "df27a780435b7b45c2597536112ea3cb091f8544c3d0c3318d9f4258b31f7adf"), + "Qwen3.6-35B-A3B-UD-Q4_K_S.gguf", + 21_388_319_008, + "2bee952b218e4a481430c59d8d3bdc7bae20bed0eb501326340c5fca7ae95d42"), Recipe( fullAttentionLayerCount: 10, keyValueHeadCount: 2, @@ -340,16 +346,42 @@ public static class LocalModelCatalog IsExplicitAlternative: false, SupportsVision: false, RecommendationPriority: 0), + new LocalModelInfo( + Qwen35B_IQ4XSModelId, + "Qwen3.6 35B-A3B (UD-IQ4_XS)", + "Qwen3.6", + "UD-IQ4_XS", + ModelArtifact( + Qwen35B_IQ4XSModelId, + s_qwen35BSource, + "Qwen3.6-35B-A3B-UD-IQ4_XS.gguf", + 18_209_036_576, + "df27a780435b7b45c2597536112ea3cb091f8544c3d0c3318d9f4258b31f7adf"), + Recipe( + fullAttentionLayerCount: 10, + keyValueHeadCount: 2, + temperature: 0.6, + speculativeDraftMaxTokens: 2), + IsDefault: false, + IsExplicitAlternative: false, + SupportsVision: false, + RecommendationPriority: 0), }); private static readonly IReadOnlyDictionary> s_profilesByModel = s_models .Select(model => (model, profiles: Array.AsReadOnly( - string.Equals(model.Id, Qwen35B_IQ4XSModelId, StringComparison.Ordinal) + string.Equals(model.Id, Qwen35B_Q4KSModelId, StringComparison.Ordinal) ? CreateRtxSpark48GbProfiles(model) : CreateProfiles(model)))) .Concat(s_legacyModels - .Select(model => (model, profiles: Array.AsReadOnly(CreateLegacyProfiles(model))))) + .Select(model => (model, profiles: Array.AsReadOnly( + // The retired 48GB-SKU quantization was only ever installed at that + // SKU's fixed tier, not the pre-profile native/F16 one, so it keeps + // the same profile set it was recorded under. + string.Equals(model.Id, Qwen35B_IQ4XSModelId, StringComparison.Ordinal) + ? CreateRtxSpark48GbProfiles(model) + : CreateLegacyProfiles(model))))) .ToDictionary( entry => entry.model.Id, entry => entry.profiles, diff --git a/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs index 38040b24b..8382a7309 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs @@ -38,7 +38,7 @@ internal static (LocalModelInfo Model, LocalInferenceRunProfile Profile)? Select return totalBytes switch { _ when totalBytes < s_boundary32_48 => null, // 32GB SKU: no local AI recommended - _ when totalBytes < s_boundary48_64 => Recipe(LocalModelCatalog.Qwen35B_IQ4XSModelId), + _ when totalBytes < s_boundary48_64 => Recipe(LocalModelCatalog.Qwen35B_Q4KSModelId), _ when totalBytes < s_boundary64_128 => Recipe(LocalModelCatalog.Qwen38_27BModelId, ReducedQ8_0ProfileId), _ => Recipe(LocalModelCatalog.Qwen38_27B_DFlashModelId), }; diff --git a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs index 1b1790b9a..9dd00449a 100644 --- a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs +++ b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs @@ -320,7 +320,7 @@ public void Evaluate_RoutesRuntimeByArchitectureWithoutGpuSkuPairing( // "Gb48" case below almost exactly. [Theory] [InlineData(30, null)] // 32GB SKU: no local AI recommended - [InlineData(45, LocalModelCatalog.Qwen35B_IQ4XSModelId)] // 48GB SKU -> 24GB recipe + [InlineData(45, LocalModelCatalog.Qwen35B_Q4KSModelId)] // 48GB SKU -> Qwen3.6-35B-A3B (Q4_K_S) [InlineData(62, LocalModelCatalog.Qwen38_27BModelId)] // 64GB SKU -> 28GB recipe [InlineData(120, LocalModelCatalog.Qwen38_27B_DFlashModelId)] // 128GB SKU -> 48GB recipe (default) public void Evaluate_RoutesRtxSparkByFixedSkuTable(long totalGiB, string? expectedModelId) @@ -369,7 +369,7 @@ public void Evaluate_SparkRecipeIsBoundToTheSparkGpuOnMixedHosts() Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", totalGiB: 80, freeGiB: 80))); Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); - Assert.Equal(LocalModelCatalog.Qwen35B_IQ4XSModelId, result.Plan?.Model.Id); + Assert.Equal(LocalModelCatalog.Qwen35B_Q4KSModelId, result.Plan?.Model.Id); Assert.Equal("GPU-spark", result.SelectedGpu?.StableId); Assert.Equal("GPU-spark", result.Plan?.BoundGpuStableId); } @@ -824,6 +824,44 @@ private static GpuInfo Gpu( CudaMajorVersion: 13, StableId: stableId); + /// + /// The retired 48GB-SKU quantization was only ever installed at that SKU's fixed + /// 98,304-token tier. Retiring it must keep that profile, not collapse it onto the + /// pre-profile native/F16 set, or an existing receipt stops resolving its profile. + /// + [Fact] + public void RetiredSpark48GbModel_KeepsTheProfileItWasInstalledUnder() + { + LocalModelInfo? retired = LocalModelCatalog.FindInstalled(LocalModelCatalog.Qwen35B_IQ4XSModelId); + + Assert.NotNull(retired); + Assert.True(LocalModelCatalog.IsLegacy(LocalModelCatalog.Qwen35B_IQ4XSModelId)); + + LocalInferenceRunProfile profile = Assert.Single(LocalModelCatalog.GetProfiles(retired)); + Assert.Equal(LocalModelCatalog.RtxSpark48GbContextTokens, profile.ContextTokens); + } + + /// + /// The 48GB SKU is picked from a fixed table, so no capacity fit-test backstops it. + /// Pin the recipe's required memory so a future quantization change cannot silently + /// grow past what a 48GB-SKU Spark (~48.6e9 bytes visible) can actually hold. + /// + [Fact] + public void Spark48GbRecipe_RequiredMemoryStaysWithinTheSku() + { + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware(RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark", 45, 45))); + + Assert.Equal(LocalModelCatalog.Qwen35B_Q4KSModelId, result.Plan!.Model.Id); + Assert.Equal(28_971_620_640L, result.RequiredTotalMemoryBytes); + + // 45 GiB (48.32e9) under-states what a real 48GB-SKU Spark reports through + // cuMemGetInfo (48.72e9 measured), so a fit here is the conservative check that + // keeps this test honest if the pinned figure above is ever raised. + Assert.NotNull(result.DetectedTotalMemoryBytes); + Assert.True(result.RequiredTotalMemoryBytes <= result.DetectedTotalMemoryBytes); + } + [Theory] [InlineData("b10655-cuda13-x64", "b10655")] [InlineData("b10655-cuda13-arm64", "b10655")] From a3eb74257116ed249a32ff5bc743a5b50865e80b Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Thu, 1 Oct 2026 09:57:37 -0400 Subject: [PATCH 3/6] feat(local-ai): align the offered Qwen3.6-35B recipe with MTP n=2 The 2026-09-30 recipe set runs every Qwen3.6-35B-A3B configuration with MTP n=2, including the 28GB and 30GB tiers the 64GB and 128GB SKUs offer. The already-offered UD-Q4_K_M entry inherited the catalog-wide default of 3 instead, so it launched with a draft depth no published recipe uses. Its pinned artifact already matches the new recipes byte for byte -- same revision, same file, same digest -- so only the draft depth changes, and the model stays reachable exactly as before through the existing explicit-alternative path. The sampling block published alongside these recipes is not adopted: every recipe in the set carries an identical string, including "speculative draft backend sampling: ON" on MTP-only entries that have no draft model, so it reads as shared boilerplate rather than per-model tuning. The 35B entries keep the temperature they ship with today. --- .../Inference/Catalog/LocalModelCatalog.cs | 3 ++- .../LocalInferenceQualificationTests.cs | 15 +++++++++++++++ 2 files changed, 17 insertions(+), 1 deletion(-) diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs index 4a476f523..51b7668ad 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs @@ -237,7 +237,8 @@ public static class LocalModelCatalog Recipe( fullAttentionLayerCount: 10, keyValueHeadCount: 2, - temperature: 0.6), + temperature: 0.6, + speculativeDraftMaxTokens: 2), IsDefault: false, IsExplicitAlternative: true, SupportsVision: false, diff --git a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs index 9dd00449a..f8bfcb5cc 100644 --- a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs +++ b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs @@ -824,6 +824,21 @@ private static GpuInfo Gpu( CudaMajorVersion: 13, StableId: stableId); + /// + /// Every Qwen3.6-35B-A3B recipe in the 2026-09-30 set runs MTP with n=2, not the + /// catalog-wide default of 3. Both offered quantizations have to agree with it. + /// + [Theory] + [InlineData(LocalModelCatalog.Qwen35BModelId)] + [InlineData(LocalModelCatalog.Qwen35B_Q4KSModelId)] + public void Qwen35BRecipes_UseTwoSpeculativeDraftTokens(string modelId) + { + LocalModelInfo model = LocalModelCatalog.Find(modelId)!; + + Assert.Equal(SpeculativeDecodingMode.DraftMtp, model.Recipe.SpeculativeDecoding); + Assert.Equal(2, model.Recipe.SpeculativeDraftMaxTokens); + } + /// /// The retired 48GB-SKU quantization was only ever installed at that SKU's fixed /// 98,304-token tier. Retiring it must keep that profile, not collapse it onto the From ed8b4be17c712ff5c909a3ecea8b11fe4b6b760c Mon Sep 17 00:00:00 2001 From: Dallin Romney Date: Fri, 2 Oct 2026 09:31:43 -0700 Subject: [PATCH 4/6] fix(local-ai): preserve onboarding for retired receipts --- .../Catalog/LocalInferenceEligibility.cs | 20 ++++++++++- .../Catalog/LocalInferenceSelector.cs | 24 ++++++++++++-- .../Inference/Catalog/LocalModelCatalog.cs | 4 +-- .../Services/SetupLocalAiHost.cs | 4 ++- .../LocalAiOnboardingTests.cs | 33 +++++++++++++++++-- .../LocalInferenceQualificationTests.cs | 20 +++++++++++ 6 files changed, 97 insertions(+), 8 deletions(-) diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs index cda634a2e..0911322b7 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs @@ -84,7 +84,25 @@ public static LocalInferenceEligibilityResult Evaluate( { ArgumentNullException.ThrowIfNull(hardware); - LocalInferenceSelectionResult selection = LocalInferenceSelector.Select(hardware, requestedModelId); + return Evaluate(hardware, LocalInferenceSelector.Select(hardware, requestedModelId)); + } + + /// + /// Evaluates the explicit model recorded by an existing installation receipt, + /// including a retired model that is no longer offered for fresh selection. + /// + public static LocalInferenceEligibilityResult EvaluateInstalled( + HostHardwareInfo hardware, + string installedModelId) + { + ArgumentNullException.ThrowIfNull(hardware); + return Evaluate(hardware, LocalInferenceSelector.SelectInstalled(hardware, installedModelId)); + } + + private static LocalInferenceEligibilityResult Evaluate( + HostHardwareInfo hardware, + LocalInferenceSelectionResult selection) + { if (!selection.IsSelected || selection.Plan is null) { return Unsupported( diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs index 61cbe6e85..52fafd40c 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs @@ -83,7 +83,25 @@ public static class LocalInferenceSelector { public static LocalInferenceSelectionResult Select( HostHardwareInfo hardware, - string? requestedModelId = null) + string? requestedModelId = null) => + Select(hardware, requestedModelId, includeRetiredInstalledModel: false); + + /// + /// Resolves an explicit model from an existing installation receipt. Retired models + /// remain valid here, but are never admitted by the fresh-selection overload. + /// + public static LocalInferenceSelectionResult SelectInstalled( + HostHardwareInfo hardware, + string installedModelId) + { + ArgumentException.ThrowIfNullOrWhiteSpace(installedModelId); + return Select(hardware, installedModelId, includeRetiredInstalledModel: true); + } + + private static LocalInferenceSelectionResult Select( + HostHardwareInfo hardware, + string? requestedModelId, + bool includeRetiredInstalledModel) { ArgumentNullException.ThrowIfNull(hardware); @@ -138,7 +156,9 @@ public static LocalInferenceSelectionResult Select( } else { - model = LocalModelCatalog.Find(requestedModelId); + model = includeRetiredInstalledModel + ? LocalModelCatalog.FindInstalled(requestedModelId) + : LocalModelCatalog.Find(requestedModelId); if (model is null) return LocalInferenceSelectionResult.Unsupported(LocalInferenceSelectionFailureCode.UnknownModel); if (sparkPick is { } recommended && diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs index 51b7668ad..d898188c5 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs @@ -448,9 +448,9 @@ public static IReadOnlyList GetProfiles(LocalModelInfo /// Resolves a model that an existing installation receipt may reference, /// including retired entries that are no longer offered for new installs. /// Use this only on installed-receipt validation, launch, and display - /// paths. Selection, recommendation, and eligibility must keep using + /// paths. Fresh selection, recommendation, and eligibility must keep using /// and so retired models are never - /// offered again. + /// offered again; receipt-aware eligibility may resolve the installed model. /// public static LocalModelInfo? FindInstalled(string? id) => Find(id) ?? diff --git a/src/OpenClaw.Tray.WinUI/Services/SetupLocalAiHost.cs b/src/OpenClaw.Tray.WinUI/Services/SetupLocalAiHost.cs index c00f7f55a..563265f70 100644 --- a/src/OpenClaw.Tray.WinUI/Services/SetupLocalAiHost.cs +++ b/src/OpenClaw.Tray.WinUI/Services/SetupLocalAiHost.cs @@ -120,7 +120,9 @@ public async Task ObserveAsync(CancellationToken ct, catch (Exception ex) when (ex is InvalidDataException or IOException or UnauthorizedAccessException) { damaged = true; } progress?.Report(LocalAiSetupStage.CheckingHardware); var hardware = await probeHardware(ct); - var eligibility = LocalInferenceEligibility.Evaluate(hardware, install?.Manifest.ModelCatalogId); + var eligibility = install is null + ? LocalInferenceEligibility.Evaluate(hardware) + : LocalInferenceEligibility.EvaluateInstalled(hardware, install.Manifest.ModelCatalogId); bool verified = false; if (install is not null) { diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiOnboardingTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiOnboardingTests.cs index 2fa6de6f5..1ae752d2e 100644 --- a/tests/OpenClaw.SetupEngine.Tests/LocalAiOnboardingTests.cs +++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiOnboardingTests.cs @@ -190,6 +190,35 @@ public async Task Observation_CancelsRefreshAndDiscardsStaleCallbacks() Assert.Equal([LocalAiSetupStage.CheckingHardware, LocalAiSetupStage.CheckingHardware], stages); } + [Theory] + [InlineData(LocalAiRuntimeState.Stopped, LocalAiOnboardingState.StartAndUse)] + [InlineData(LocalAiRuntimeState.Healthy, LocalAiOnboardingState.Use)] + [InlineData(LocalAiRuntimeState.Failed, LocalAiOnboardingState.Repair)] + public async Task Observation_RetainedSparkReceiptPreservesOnboardingActions( + LocalAiRuntimeState runtimeState, + LocalAiOnboardingState expected) + { + using var directory = new TempDirectory(); + var registry = Registry(directory.Path); + var install = Install(LocalModelCatalog.Qwen35B_IQ4XSModelId); + var runtime = new FakeRuntime(RuntimeSnapshot(install, runtimeState)); + var hardware = new HostHardwareInfo(Architecture.Arm64, 128L << 30, 80L << 30, + [new(GpuVendor.Nvidia, "NVIDIA RTX Spark N1X", 45L << 30, 45L << 30, + DriverVersion: "615.0", CudaMajorVersion: 13, StableId: "GPU-spark")], false); + var host = new SetupLocalAiHost( + () => Task.FromResult(new LocalAiSetupResolution(LocalAiSetupRoute.Recovery, + new("gateway", "Managed", 18789, install.Manifest.ModelCatalogId, install.Manifest.RequestedPort))), + () => registry, () => runtime, _ => Task.FromResult(install), + (_, _) => Task.FromResult(true), _ => Task.FromResult(hardware), + () => throw new InvalidOperationException("Observation must not mutate the Gateway.")); + + LocalAiOnboardingSnapshot snapshot = await host.ObserveAsync(CancellationToken.None); + + Assert.Equal(expected, snapshot.State); + Assert.True(snapshot.CanUse || snapshot.CanReview); + Assert.Equal(LocalModelCatalog.Qwen35B_IQ4XSModelId, snapshot.Eligibility!.Plan!.Model.Id); + } + [Fact] public async Task ClosingObservation_FencesCallbacksAndNeverMutates() { @@ -785,9 +814,9 @@ private static GatewayRegistry Registry(string directory) return registry; } - internal static LocalAiResolvedInstall Install() + internal static LocalAiResolvedInstall Install(string modelId = LocalModelCatalog.Qwen35BModelId) { - var model = LocalModelCatalog.FindInstalled(LocalModelCatalog.Qwen35BModelId)!; + var model = LocalModelCatalog.FindInstalled(modelId)!; var endpoint = new Uri("http://127.0.0.1:18803/v1"); return new(new LocalAiInstallManifest { diff --git a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs index f8bfcb5cc..7dff0709e 100644 --- a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs +++ b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs @@ -856,6 +856,26 @@ public void RetiredSpark48GbModel_KeepsTheProfileItWasInstalledUnder() Assert.Equal(LocalModelCatalog.RtxSpark48GbContextTokens, profile.ContextTokens); } + [Fact] + public void InstalledRetiredSpark48GbModel_RemainsEligibleWithoutRestoringFreshSelection() + { + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, + Gpu("NVIDIA RTX Spark N1X", "GPU-spark", 45, 45)); + + LocalInferenceEligibilityResult fresh = LocalInferenceEligibility.Evaluate( + hardware, + LocalModelCatalog.Qwen35B_IQ4XSModelId); + LocalInferenceEligibilityResult installed = LocalInferenceEligibility.EvaluateInstalled( + hardware, + LocalModelCatalog.Qwen35B_IQ4XSModelId); + + Assert.Equal(LocalInferenceSelectionFailureCode.UnknownModel, fresh.SelectionFailureCode); + Assert.True(installed.CanInstall); + Assert.Equal(LocalModelCatalog.Qwen35B_IQ4XSModelId, installed.Plan!.Model.Id); + Assert.Equal(LocalModelCatalog.RtxSpark48GbContextTokens, installed.Plan.Profile.ContextTokens); + } + /// /// The 48GB SKU is picked from a fixed table, so no capacity fit-test backstops it. /// Pin the recipe's required memory so a future quantization change cannot silently From d312ceda479a61190d792e37fc5848a1b3e6885c Mon Sep 17 00:00:00 2001 From: Dallin Romney Date: Fri, 2 Oct 2026 09:42:10 -0700 Subject: [PATCH 5/6] fix(local-ai): retain installed model through repair --- .../Controls/LocalAiSetupControl.xaml.cs | 10 ++++- .../SetupWindow.xaml.cs | 2 + .../LocalAiRecoveryPolicy.cs | 3 ++ src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs | 12 ++++-- src/OpenClaw.SetupEngine/SetupContext.cs | 6 +++ .../LocalAiPortHandoffTests.cs | 38 +++++++++++++++++++ .../SetupReviewOwnershipTests.cs | 1 + .../LocalAiSetupUxContractTests.cs | 5 ++- 8 files changed, 71 insertions(+), 6 deletions(-) diff --git a/src/OpenClaw.SetupEngine.UI/Controls/LocalAiSetupControl.xaml.cs b/src/OpenClaw.SetupEngine.UI/Controls/LocalAiSetupControl.xaml.cs index 7b807f570..104bf8046 100644 --- a/src/OpenClaw.SetupEngine.UI/Controls/LocalAiSetupControl.xaml.cs +++ b/src/OpenClaw.SetupEngine.UI/Controls/LocalAiSetupControl.xaml.cs @@ -125,7 +125,9 @@ private async Task InitializeLocalAiReviewAsync( if (_config.LocalAi.SelectedModelId is { } selectedModelId) { LocalInferenceEligibilityResult selectedEligibility = - LocalInferenceEligibility.Evaluate(_localAiHardware, selectedModelId); + _localAiRecoveryModelPinned + ? LocalInferenceEligibility.EvaluateInstalled(_localAiHardware, selectedModelId) + : LocalInferenceEligibility.Evaluate(_localAiHardware, selectedModelId); if (_localAiRecoveryModelPinned) { eligibility = selectedEligibility; @@ -137,7 +139,11 @@ private async Task InitializeLocalAiReviewAsync( } _config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? availability.Plan.Model.Id; - eligibility ??= LocalInferenceEligibility.Evaluate( + eligibility ??= _localAiRecoveryModelPinned + ? LocalInferenceEligibility.EvaluateInstalled( + _localAiHardware, + _config.LocalAi.SelectedModelId!) + : LocalInferenceEligibility.Evaluate( _localAiHardware, _config.LocalAi.SelectedModelId); if (_localAiRecoveryModelPinned && !eligibility.CanInstall) diff --git a/src/OpenClaw.SetupEngine.UI/SetupWindow.xaml.cs b/src/OpenClaw.SetupEngine.UI/SetupWindow.xaml.cs index f8869f9a5..418214cb6 100644 --- a/src/OpenClaw.SetupEngine.UI/SetupWindow.xaml.cs +++ b/src/OpenClaw.SetupEngine.UI/SetupWindow.xaml.cs @@ -330,6 +330,7 @@ public SetupWindow( if (!string.IsNullOrWhiteSpace(localAiRecoveryModelId)) { _config.LocalAi.SelectedModelId = localAiRecoveryModelId; + _config.LocalAi.InstalledReceiptModelId = localAiRecoveryModelId; _pinLocalAiRecoveryModel = true; } if (localAiRecoveryRequestedPort is { } requestedPort && @@ -602,6 +603,7 @@ private async Task ReviewLocalAiCoreAsync(LocalAiOnboardingSnapshot selection) _config.GatewayPort = target.GatewayPort; _config.GatewayUrl = null; _config.LocalAi.SelectedModelId = target.ModelCatalogId; + _config.LocalAi.InstalledReceiptModelId = target.ModelCatalogId; if (target.RequestedLocalAiPort is { } port) _config.LocalAi.Port = port; _config.LocalAi.Enabled = true; diff --git a/src/OpenClaw.SetupEngine/LocalAiRecoveryPolicy.cs b/src/OpenClaw.SetupEngine/LocalAiRecoveryPolicy.cs index 41aa60633..2771ad5c1 100644 --- a/src/OpenClaw.SetupEngine/LocalAiRecoveryPolicy.cs +++ b/src/OpenClaw.SetupEngine/LocalAiRecoveryPolicy.cs @@ -36,6 +36,7 @@ internal sealed record LocalAiRecoveryConfigurationBaseline( int GatewayPort, string? GatewayUrl, string? SelectedProfileId, + string? InstalledReceiptModelId, bool NetworkingConsent) { public static LocalAiRecoveryConfigurationBaseline Capture(SetupConfig config) => @@ -49,6 +50,7 @@ public static LocalAiRecoveryConfigurationBaseline Capture(SetupConfig config) = config.GatewayPort, config.GatewayUrl, config.LocalAi.SelectedProfileId, + config.LocalAi.InstalledReceiptModelId, config.LocalAi.WslMirroredNetworkingConsent); public void Restore(SetupConfig config) @@ -62,6 +64,7 @@ public void Restore(SetupConfig config) config.GatewayPort = GatewayPort; config.GatewayUrl = GatewayUrl; config.LocalAi.SelectedProfileId = SelectedProfileId; + config.LocalAi.InstalledReceiptModelId = InstalledReceiptModelId; config.LocalAi.WslMirroredNetworkingConsent = NetworkingConsent; } } diff --git a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs index e83cb8262..398dcf5a4 100644 --- a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs +++ b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs @@ -47,9 +47,15 @@ public override Task ExecuteAsync(SetupContext ctx, CancellationToke ex)); } - LocalInferenceEligibilityResult eligibility = LocalInferenceEligibility.Evaluate( - hardware, - ctx.Config.LocalAi.SelectedModelId); + string? selectedModelId = ctx.Config.LocalAi.SelectedModelId; + LocalInferenceEligibilityResult eligibility = + !string.IsNullOrWhiteSpace(selectedModelId) && + string.Equals( + selectedModelId, + ctx.Config.LocalAi.InstalledReceiptModelId, + StringComparison.OrdinalIgnoreCase) + ? LocalInferenceEligibility.EvaluateInstalled(hardware, selectedModelId) + : LocalInferenceEligibility.Evaluate(hardware, selectedModelId); ctx.LocalAiHardware = hardware; ctx.LocalAiEligibility = eligibility; ctx.Config.LocalAi.SelectedProfileId = eligibility.Plan?.Profile.Id; diff --git a/src/OpenClaw.SetupEngine/SetupContext.cs b/src/OpenClaw.SetupEngine/SetupContext.cs index 4293151bc..d7fe93dab 100644 --- a/src/OpenClaw.SetupEngine/SetupContext.cs +++ b/src/OpenClaw.SetupEngine/SetupContext.cs @@ -150,6 +150,12 @@ public sealed class LocalAiConfig /// Runtime-only effective profile selected from detected GPU capacity. [JsonIgnore] public string? SelectedProfileId { get; set; } + /// + /// Runtime-only model ID proven by the existing installation receipt for a pinned + /// recovery. It authorizes retired-catalog lookup only for that exact selection. + /// + [JsonIgnore] + public string? InstalledReceiptModelId { get; set; } /// Managed llama-server port. Zero selects a free loopback port during setup. public int Port { get; set; } public bool WslMirroredNetworkingConsent { get; set; } diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs index b5d4e2f1a..dabd5bd24 100644 --- a/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs +++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs @@ -34,6 +34,44 @@ public async Task Preflight_RejectsReservedPort80() Assert.Null(context.LocalAiPort); } + [Fact] + public async Task Preflight_RetainedSparkReceiptUsesInstalledModelCatalogDuringRepair() + { + SetupContext context = CreateContext(new LocalAiConfig + { + Enabled = true, + Port = 0, + SelectedModelId = LocalModelCatalog.Qwen35B_IQ4XSModelId, + InstalledReceiptModelId = LocalModelCatalog.Qwen35B_IQ4XSModelId, + }); + var step = new PreflightLocalAiHardwareStep(new FakeHardwareProbe(CreateSparkHardware())); + + StepResult result = await step.ExecuteAsync(context, CancellationToken.None); + + Assert.Equal(StepOutcome.Success, result.Outcome); + Assert.Equal(LocalModelCatalog.Qwen35B_IQ4XSModelId, context.LocalAiEligibility!.Plan!.Model.Id); + Assert.Equal(LocalModelCatalog.RtxSpark48GbContextTokens, context.LocalAiEligibility.Plan.Profile.ContextTokens); + } + + [Fact] + public async Task Preflight_RetiredSparkModelWithoutMatchingReceiptProofRemainsUnavailable() + { + SetupContext context = CreateContext(new LocalAiConfig + { + Enabled = true, + Port = 0, + SelectedModelId = LocalModelCatalog.Qwen35B_IQ4XSModelId, + }); + var step = new PreflightLocalAiHardwareStep(new FakeHardwareProbe(CreateSparkHardware())); + + StepResult result = await step.ExecuteAsync(context, CancellationToken.None); + + Assert.Equal(StepOutcome.FailedTerminal, result.Outcome); + Assert.Equal( + LocalInferenceSelectionFailureCode.UnknownModel, + context.LocalAiEligibility!.SelectionFailureCode); + } + [Fact] public async Task Preflight_StopsBeforeAnyDownloadWhenCapacityIsUnknown() { diff --git a/tests/OpenClaw.SetupEngine.Tests/SetupReviewOwnershipTests.cs b/tests/OpenClaw.SetupEngine.Tests/SetupReviewOwnershipTests.cs index c6c0fc360..5d9817f74 100644 --- a/tests/OpenClaw.SetupEngine.Tests/SetupReviewOwnershipTests.cs +++ b/tests/OpenClaw.SetupEngine.Tests/SetupReviewOwnershipTests.cs @@ -11,6 +11,7 @@ public void LocalAiReview_PreservesGenerationEligibilityAndPinnedRecovery() Assert.Contains("LocalInferenceEligibilityFailureCode.HardwareFactsIncomplete", source); Assert.Contains("TryApplyProbeFailure", source); Assert.Contains("if (_localAiRecoveryModelPinned)", source); + Assert.Contains("LocalInferenceEligibility.EvaluateInstalled(_localAiHardware, selectedModelId)", source); Assert.Contains("LocalAiToggle.IsEnabled = !_localAiRecoveryOnly", source); Assert.Contains("_localAiAvailability.CancelCurrent()", source); } diff --git a/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs b/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs index 22e33cf61..8a3031455 100644 --- a/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs +++ b/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs @@ -307,13 +307,16 @@ public void CapabilitiesReview_GatesOnDeviceEligibilityAndReconcilesStaleSelecte "if (!availability.CanInstall || availability.Plan is null || availability.SelectedGpu is null)", "hardwareReason = DescribeLocalAiUnavailable(availability);", "LocalInferenceEligibilityResult selectedEligibility =", + "LocalInferenceEligibility.EvaluateInstalled(_localAiHardware, selectedModelId)", "LocalInferenceEligibility.Evaluate(_localAiHardware, selectedModelId);", "if (_localAiRecoveryModelPinned)", "eligibility = selectedEligibility;", "else if (!selectedEligibility.CanInstall)", "_config.LocalAi.SelectedModelId = null;", "_config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? availability.Plan.Model.Id;", - "eligibility ??= LocalInferenceEligibility.Evaluate(", + "eligibility ??= _localAiRecoveryModelPinned", + "LocalInferenceEligibility.EvaluateInstalled(", + "LocalInferenceEligibility.Evaluate(", "_config.LocalAi.SelectedModelId);"); } From 5f2eee477cf7d7049a7d351e273419c7b17f7c0b Mon Sep 17 00:00:00 2001 From: Dallin Romney Date: Fri, 2 Oct 2026 16:04:59 -0700 Subject: [PATCH 6/6] refactor(local-ai): keep recovery internals private --- src/OpenClaw.SetupEngine/SetupContext.cs | 2 +- src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/OpenClaw.SetupEngine/SetupContext.cs b/src/OpenClaw.SetupEngine/SetupContext.cs index d7fe93dab..5c5f5ef05 100644 --- a/src/OpenClaw.SetupEngine/SetupContext.cs +++ b/src/OpenClaw.SetupEngine/SetupContext.cs @@ -155,7 +155,7 @@ public sealed class LocalAiConfig /// recovery. It authorizes retired-catalog lookup only for that exact selection. /// [JsonIgnore] - public string? InstalledReceiptModelId { get; set; } + internal string? InstalledReceiptModelId { get; set; } /// Managed llama-server port. Zero selects a free loopback port during setup. public int Port { get; set; } public bool WslMirroredNetworkingConsent { get; set; } diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs index 52fafd40c..b8bcf9840 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs @@ -90,7 +90,7 @@ public static LocalInferenceSelectionResult Select( /// Resolves an explicit model from an existing installation receipt. Retired models /// remain valid here, but are never admitted by the fresh-selection overload. /// - public static LocalInferenceSelectionResult SelectInstalled( + internal static LocalInferenceSelectionResult SelectInstalled( HostHardwareInfo hardware, string installedModelId) {