From 937828481a71dd3d4bd8343cff53dfed8d7f072a Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Wed, 23 Sep 2026 14:23:35 -0400 Subject: [PATCH 01/13] feat(local-ai): detect RTX Spark by driver-reported GPU name Adds GpuInfo.IsRtxSpark, additive only. Groundwork for routing RTX Spark's unified-memory SKUs through a fixed recipe table instead of the generic capacity fit-test. --- src/OpenClaw.Shared/Inference/HostHardwareInfo.cs | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs b/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs index b1beaa788..60c6b0813 100644 --- a/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs +++ b/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs @@ -52,7 +52,17 @@ public sealed record GpuInfo( long? FreeSharedGpuMemoryBytes = null, string? DriverVersion = null, int? CudaMajorVersion = null, - string? StableId = null); + string? StableId = null) +{ + /// + /// True for an RTX Spark unified-memory adapter (e.g. "NVIDIA RTX Spark + /// N1X"), identified by its driver-reported name. RTX Spark is routed + /// through a fixed SKU recipe table instead of the generic capacity + /// fit-test; see RtxSparkInferenceSelector. + /// + public bool IsRtxSpark => + Name.Contains("RTX Spark", StringComparison.OrdinalIgnoreCase); +} /// /// Snapshot of the host's inference-relevant hardware. Every probed field is From fb6086628a5f43fc34eec68dcbd674f37cd3b3cd Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Wed, 23 Sep 2026 14:23:43 -0400 Subject: [PATCH 02/13] feat(local-ai): route RTX Spark defaults through a fixed SKU table RTX Spark's total CUDA-visible memory identifies the physical memory SKU, not usable capacity the way it does on a discrete GPU, so the generic priority/fit-test default pick doesn't apply. Adds: - SpeculativeDecodingMode.None/DraftDFlash and an optional separate draft checkpoint (LocalModelRunRecipe.DraftWeights), needed because DFlash uses an independently pinned draft GGUF rather than an embedded MTP draft layer. - Three RTX Spark catalog recipes: Qwen3.6-35B-A3B IQ4_XS (48GB SKU), and Qwen3.8-27B Q4_K_M with DFlash n=7 (128GB SKU default). The 64GB-SKU recipe reuses the existing default model's ctx-131072-q8_0 profile -- no new catalog entry needed. - RtxSparkInferenceSelector, a fixed SKU-to-recipe table keyed off GpuVisibleMemoryBytes (empirically verified against a real 48GB RTX Spark unit). Wired into LocalInferenceSelector.Select only for the no-requested-model default path; explicit model requests and every non-Spark GPU keep using the untouched generic fit-test. - GetRequiredMemoryBytes/GetDraftKvCacheMemoryBytes now account for a DFlash recipe's separate draft weights and skip draft KV entirely for SpeculativeDecodingMode.None. --- .../Pages/CapabilitiesPage.xaml.cs | 17 ++- .../Catalog/LocalInferenceEligibility.cs | 39 ++++++- .../Catalog/LocalInferenceSelector.cs | 92 +++++++++++++-- .../Inference/Catalog/LocalModelCatalog.cs | 108 ++++++++++++++++-- .../Catalog/RtxSparkInferenceSelector.cs | 62 ++++++++++ .../Presentation/LocalAiPageViewModel.cs | 22 +++- 6 files changed, 317 insertions(+), 23 deletions(-) create mode 100644 src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs diff --git a/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs b/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs index 26855297d..0c0b0d6d9 100644 --- a/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs +++ b/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs @@ -330,9 +330,18 @@ private async Task InitializeLocalAiReviewAsync( ? deviceEligibility.Plan?.Model.Id : null; - if (!deviceEligibility.CanInstall || deviceEligibility.Plan is null || deviceEligibility.SelectedGpu is null) + // A SKU with no recommended default (RTX Spark 32 GB) still runs an already + // configured model. Gate availability on that configured selection when there is + // one, so rerunning setup does not switch Local AI off on a working machine. + // _localAiRecommendedModelId stays null so nothing is labelled Recommended. + LocalInferenceEligibilityResult availability = + LocalInferenceEligibility.EvaluateForConfiguredAvailability( + _localAiHardware, + _config!.LocalAi.SelectedModelId); + + if (!availability.CanInstall || availability.Plan is null || availability.SelectedGpu is null) { - hardwareReason = DescribeLocalAiUnavailable(deviceEligibility); + hardwareReason = DescribeLocalAiUnavailable(availability); } else { @@ -343,7 +352,7 @@ private async Task InitializeLocalAiReviewAsync( // model instead of leaving setup stuck on a known-incompatible selection. A // merely busy GPU (EligibleButBusy) is not reconciled away: the same model would // still work once the GPU frees up, and CanInstall already covers that case. - if (_config!.LocalAi.SelectedModelId is { } selectedModelId) + if (_config.LocalAi.SelectedModelId is { } selectedModelId) { LocalInferenceEligibilityResult selectedEligibility = LocalInferenceEligibility.Evaluate(_localAiHardware, selectedModelId); @@ -356,7 +365,7 @@ private async Task InitializeLocalAiReviewAsync( _config.LocalAi.SelectedModelId = null; } } - _config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? deviceEligibility.Plan.Model.Id; + _config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? availability.Plan.Model.Id; eligibility ??= LocalInferenceEligibility.Evaluate( _localAiHardware, diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs index 58581b1f5..cda634a2e 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs @@ -48,6 +48,36 @@ public static long GetRequiredMemoryBytes( LocalInferenceRunProfile profile) => LocalInferenceQualificationPolicy.GetRequiredMemoryBytes(model, profile); + /// + /// Device eligibility for deciding whether Local AI stays available, given the model + /// already configured on this machine. + /// + /// + /// A SKU with no recommended default is a statement about what to install by default, + /// not about what the device can run. Gating availability purely on the default pick + /// would switch Local AI off on a setup rerun for a machine that already has a working + /// configured model, so once a model is configured this reports on that model: a + /// selection that still passes the full capacity fit-test is retained, and one that does + /// not carries its own failure (unknown model, or the model name with its required and + /// detected memory) rather than the SKU's generic no-recommendation reason. With no model + /// configured, the device result stands and fresh setup is unchanged. + /// + public static LocalInferenceEligibilityResult EvaluateForConfiguredAvailability( + HostHardwareInfo hardware, + string? configuredModelId) + { + ArgumentNullException.ThrowIfNull(hardware); + LocalInferenceEligibilityResult device = Evaluate(hardware); + if (device.CanInstall || + device.SelectionFailureCode != LocalInferenceSelectionFailureCode.NotRecommendedForSku || + string.IsNullOrWhiteSpace(configuredModelId)) + { + return device; + } + + return Evaluate(hardware, configuredModelId); + } + public static LocalInferenceEligibilityResult Evaluate( HostHardwareInfo hardware, string? requestedModelId = null) @@ -64,7 +94,14 @@ public static LocalInferenceEligibilityResult Evaluate( LocalInferencePlan plan = selection.Plan; long requiredMemoryBytes = GetRequiredMemoryBytes(plan.Model, plan.Profile); - CandidateAssessment? selected = hardware.NvidiaGpus + // A plan bound to one adapter (an RTX Spark SKU recipe) must be assessed only + // against that adapter. Ranking every NVIDIA GPU here would let the recipe + // chosen for the Spark be reported against, and then launched on, a different + // GPU on a mixed host. + IEnumerable candidateGpus = plan.BoundGpuStableId is { Length: > 0 } boundId + ? hardware.NvidiaGpus.Where(gpu => string.Equals(gpu.StableId, boundId, StringComparison.Ordinal)) + : hardware.NvidiaGpus; + CandidateAssessment? selected = candidateGpus .Select(gpu => Assess(gpu, plan.Runtime, requiredMemoryBytes)) .OrderBy(candidate => StatusRank(candidate.Status)) .ThenBy(candidate => DefinitivenessRank(candidate.FailureCode)) diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs index bbc8f99c7..cd684e342 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs @@ -16,6 +16,8 @@ public enum LocalInferenceSelectionFailureCode RuntimeUnavailable = 1, NoNvidiaGpu = 2, UnknownModel = 3, + /// RTX Spark detected, but this memory SKU has no recommended local model. + NotRecommendedForSku = 4, } /// Whether a caller accepted the catalog default or named a model explicitly. @@ -26,11 +28,19 @@ public enum LocalInferenceModelSelectionOrigin } /// A complete, immutable native inference choice. +/// +/// The adapter this recipe was chosen for, when the choice is only valid on that +/// adapter. RTX Spark recipes come from a fixed per-device SKU table, so the recipe +/// and the GPU that runs it must be the same adapter; eligibility restricts its +/// candidates to this id. Null means any qualifying NVIDIA GPU may run the plan, +/// which is the generic discrete-GPU behavior. +/// public sealed record LocalInferencePlan( LlamaRuntimeVariant Runtime, LocalModelInfo Model, LocalInferenceRunProfile Profile, - LocalInferenceModelSelectionOrigin ModelSelectionOrigin); + LocalInferenceModelSelectionOrigin ModelSelectionOrigin, + string? BoundGpuStableId = null); /// The deterministic result of selecting from the pinned local inference catalog. public sealed record LocalInferenceSelectionResult @@ -60,7 +70,14 @@ internal static LocalInferenceSelectionResult Unsupported(LocalInferenceSelectio /// /// Pure selection from a hardware snapshot and optional model ID. The CPU /// architecture chooses only the native runtime. GPU names and CPU/GPU SKU -/// pairings are not part of qualification. +/// pairings are not part of qualification for discrete GPUs -- the one +/// deliberate exception is RTX Spark's default pick, routed through +/// because its unified-memory SKU +/// cannot be identified by capacity fit-testing alone (see NVIDIA's fixed +/// SKU-to-recipe table). An explicitly requested model ID uses the same +/// capacity fit-test on every GPU, except when the request is the Spark SKU's +/// own recommendation round-tripped through setup, which keeps that SKU's +/// pinned profile and adapter binding. /// public static class LocalInferenceSelector { @@ -81,9 +98,42 @@ public static LocalInferenceSelectionResult Select( LocalModelInfo? model; LocalInferenceRunProfile profile; LocalInferenceModelSelectionOrigin modelSelectionOrigin; + string? boundGpuStableId = null; + GpuInfo? sparkGpu = hardware.NvidiaGpus.FirstOrDefault( + gpu => gpu.IsRtxSpark && LocalInferenceQualificationPolicy.HasCompleteFacts(gpu)); + var sparkPick = sparkGpu is null + ? null + : RtxSparkInferenceSelector.SelectDefault(sparkGpu); if (string.IsNullOrWhiteSpace(requestedModelId)) { - (model, profile) = SelectDefaultModelAndProfile(hardware, runtime); + if (sparkPick is not null) + { + // Bind the plan to this adapter: the SKU table answers "what should + // THIS Spark run", so the recipe is only valid on the Spark that + // produced it, never on some other GPU eligibility might rank higher. + (model, profile) = sparkPick.Value; + boundGpuStableId = sparkGpu!.StableId; + } + else + { + // Either no Spark, or a Spark SKU with no recommended model. In the + // latter case the Spark is excluded rather than failing the whole + // host, so a discrete GPU alongside it can still qualify normally. + HostHardwareInfo genericHardware = sparkGpu is null + ? hardware + : hardware with + { + Gpus = hardware.Gpus.Where(gpu => !gpu.IsRtxSpark).ToArray(), + }; + if (!genericHardware.HasNvidiaGpu) + { + return LocalInferenceSelectionResult.Unsupported( + LocalInferenceSelectionFailureCode.NotRecommendedForSku); + } + + (model, profile) = SelectDefaultModelAndProfile(genericHardware, runtime); + } + modelSelectionOrigin = LocalInferenceModelSelectionOrigin.Default; } else @@ -91,13 +141,30 @@ public static LocalInferenceSelectionResult Select( model = LocalModelCatalog.Find(requestedModelId); if (model is null) return LocalInferenceSelectionResult.Unsupported(LocalInferenceSelectionFailureCode.UnknownModel); - profile = SelectBestFittingProfile(hardware, runtime, model) ?? - LocalModelCatalog.GetProfiles(model)[^1]; + if (sparkPick is { } recommended && + string.Equals(recommended.Model.Id, model.Id, StringComparison.OrdinalIgnoreCase)) + { + // Setup persists the recommended model id and passes it back here, so + // the SKU's own recommendation arrives as an explicit request. Re-deriving + // its profile through the generic fit-test would silently discard the + // pinned profile the SKU table specifies (the 64 GB tier's reduced context + // is not the largest that merely fits) and drop the adapter binding. + // A request for any other model is a real user override and still uses + // the generic fit-test below. + profile = recommended.Profile; + boundGpuStableId = sparkGpu!.StableId; + } + else + { + profile = SelectBestFittingProfile(hardware, runtime, model) ?? + LocalModelCatalog.GetProfiles(model)[^1]; + } + modelSelectionOrigin = LocalInferenceModelSelectionOrigin.Explicit; } return LocalInferenceSelectionResult.Selected( - new LocalInferencePlan(runtime, model, profile, modelSelectionOrigin)); + new LocalInferencePlan(runtime, model, profile, modelSelectionOrigin, boundGpuStableId)); } private static (LocalModelInfo Model, LocalInferenceRunProfile Profile) SelectDefaultModelAndProfile( @@ -149,9 +216,12 @@ public static long GetRequiredMemoryBytes( { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(profile); + long draftWeightsBytes = model.Recipe.DraftWeights?.SizeBytes ?? 0; return SaturatingAdd( SaturatingAdd( - SaturatingAdd(model.Weights.SizeBytes, GetKvCacheMemoryBytes(model.Recipe, profile)), + SaturatingAdd( + SaturatingAdd(model.Weights.SizeBytes, draftWeightsBytes), + GetKvCacheMemoryBytes(model.Recipe, profile)), GetDraftKvCacheMemoryBytes(model.Recipe, profile)), profile.RuntimeWorkspaceBytes); } @@ -182,8 +252,12 @@ internal static long GetDraftKvCacheMemoryBytes( ArgumentNullException.ThrowIfNull(recipe); ArgumentNullException.ThrowIfNull(profile); - // The pinned Qwen MTP artifacts contain one draft attention layer with - // the same KV head count and head dimension as the target model. + if (recipe.SpeculativeDecoding == SpeculativeDecodingMode.None) + return 0; + + // The pinned Qwen MTP and DFlash draft artifacts contain one draft + // attention layer with the same KV head count and head dimension as + // the target model. long bytesPerToken = SaturatingAdd( SaturatingMultiply( recipe.KeyValueHeadCount, diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs index 72a2a2391..e1f9dd9ac 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs @@ -13,6 +13,10 @@ public enum KvCachePrecision public enum SpeculativeDecodingMode { DraftMtp = 0, + /// No speculative decoding; the target model runs standalone. + None = 1, + /// Draft-flash decoding using a separate, independently pinned draft checkpoint. + DraftDFlash = 2, } /// Sampling values recommended for the model's thinking mode. @@ -38,8 +42,13 @@ public LocalModelRunRecipe( bool offloadAllLayers, SpeculativeDecodingMode speculativeDecoding, int speculativeDraftMaxTokens, - ModelSamplingPreset sampling) + ModelSamplingPreset sampling, + PinnedArtifact? draftWeights = null) { + if (speculativeDecoding == SpeculativeDecodingMode.DraftDFlash && draftWeights is null) + throw new ArgumentException("Draft-flash decoding requires a pinned draft checkpoint.", nameof(draftWeights)); + if (speculativeDecoding != SpeculativeDecodingMode.DraftDFlash && draftWeights is not null) + throw new ArgumentException("Only draft-flash decoding uses a separate draft checkpoint.", nameof(draftWeights)); if (batchTokens <= 0) throw new ArgumentOutOfRangeException(nameof(batchTokens)); if (microBatchTokens <= 0 || microBatchTokens > batchTokens) @@ -67,6 +76,7 @@ public LocalModelRunRecipe( SpeculativeDecoding = speculativeDecoding; SpeculativeDraftMaxTokens = speculativeDraftMaxTokens; Sampling = sampling; + DraftWeights = draftWeights; } public int BatchTokens { get; } @@ -80,6 +90,8 @@ public LocalModelRunRecipe( public SpeculativeDecodingMode SpeculativeDecoding { get; } public int SpeculativeDraftMaxTokens { get; } public ModelSamplingPreset Sampling { get; } + /// The independently pinned draft checkpoint, set only for . + public PinnedArtifact? DraftWeights { get; } } /// A downloadable GGUF model and its deterministic llama-server recipe. @@ -142,10 +154,16 @@ public static class LocalModelCatalog /// Qwen3.5 9B receipt keeps resolving and launching across upgrade. /// public const string Qwen9BModelId = "qwen3.5-9b-mtp-q4-k-m"; + /// RTX Spark 48GB-SKU recipe. Never offered on the generic dGPU path; see RtxSparkInferenceSelector. + public const string Qwen35B_IQ4XSModelId = "qwen3.6-35b-a3b-mtp-ud-iq4-xs"; + /// RTX Spark 128GB-SKU default recipe. Never offered on the generic dGPU path; see RtxSparkInferenceSelector. + public const string Qwen38_27B_DFlashModelId = "qwen3.8-27b-dflash-ud-q4-k-m"; public const int NativeContextTokens = 262_144; public const int IntermediateContextTokens = 196_608; public const int ReducedContextTokens = 131_072; public const int MinimumContextTokens = 65_536; + /// RTX Spark 48GB-SKU context tier (98,304 tokens); see . + public const int RtxSpark48GbContextTokens = 98_304; // Measured-conservative allowances for compute buffers, recurrent state, // CUDA graphs, allocator alignment, and miscellaneous backend allocations. @@ -172,6 +190,10 @@ public static class LocalModelCatalog "unsloth/Qwen3.5-9B-MTP-GGUF", "9716a636ee4bddc3fed678220b7a33dd2a4160ae"); + private static readonly HuggingFaceRevisionSource s_qwen38_27BDFlashDraftSource = new( + "z-lab/Qwen3.8-27B-DFlash2-GGUF", + "2d9571f8ce46e151f61c6499c99dee6079e1d610"); + private static readonly ReadOnlyCollection s_models = Array.AsReadOnly( new[] { @@ -232,6 +254,58 @@ public static class LocalModelCatalog IsExplicitAlternative: true, SupportsVision: false, RecommendationPriority: 200), + // RTX Spark SKU recipes below. RecommendationPriority: 0 keeps them + // out of the generic dGPU default pick; only RtxSparkInferenceSelector + // offers them, keyed off the detected Spark unified-memory SKU. + new LocalModelInfo( + Qwen35B_IQ4XSModelId, + "Qwen3.6 35B-A3B (UD-IQ4_XS)", + "Qwen3.6", + "UD-IQ4_XS", + ModelArtifact( + Qwen35B_IQ4XSModelId, + s_qwen35BSource, + "Qwen3.6-35B-A3B-UD-IQ4_XS.gguf", + 18_209_036_576, + "df27a780435b7b45c2597536112ea3cb091f8544c3d0c3318d9f4258b31f7adf"), + Recipe( + fullAttentionLayerCount: 10, + keyValueHeadCount: 2, + temperature: 0.6, + speculativeDraftMaxTokens: 2), + IsDefault: false, + IsExplicitAlternative: false, + SupportsVision: false, + RecommendationPriority: 0), + new LocalModelInfo( + Qwen38_27B_DFlashModelId, + "Qwen3.8 27B (UD-Q4_K_M, DFlash)", + "Qwen3.8", + "UD-Q4_K_M", + ModelArtifact( + Qwen38_27B_DFlashModelId, + s_qwen38_27BSource, + "Qwen3.8-27B-UD-Q4_K_M.gguf", + 16_464_440_224, + "322e194ff79741c7baa497c240f677f54b201b0efab44ca8e50f122b39123482"), + Recipe( + fullAttentionLayerCount: 16, + keyValueHeadCount: 4, + temperature: 1.0, + batchTokens: 4_096, + microBatchTokens: 512, + speculativeDecoding: SpeculativeDecodingMode.DraftDFlash, + speculativeDraftMaxTokens: 7, + draftWeights: ModelArtifact( + "qwen3.8-27b-dflash2-q4-k-m", + s_qwen38_27BDFlashDraftSource, + "Qwen3.8-27B-DFlash2-Q4_K_M.gguf", + 1_143_006_816, + "1a25c56858e1ebe93f2718ac1d49d1151f9323325c1bbfd6209370f4db131ebd")), + IsDefault: false, + IsExplicitAlternative: false, + SupportsVision: false, + RecommendationPriority: 0), }); // Retired from new installs and never offered, recommended, or selectable. @@ -264,7 +338,10 @@ public static class LocalModelCatalog private static readonly IReadOnlyDictionary> s_profilesByModel = s_models - .Select(model => (model, profiles: Array.AsReadOnly(CreateProfiles(model)))) + .Select(model => (model, profiles: Array.AsReadOnly( + string.Equals(model.Id, Qwen35B_IQ4XSModelId, StringComparison.Ordinal) + ? CreateRtxSpark48GbProfiles(model) + : CreateProfiles(model)))) .Concat(s_legacyModels .Select(model => (model, profiles: Array.AsReadOnly(CreateLegacyProfiles(model))))) .ToDictionary( @@ -367,6 +444,14 @@ private static LocalInferenceRunProfile[] CreateLegacyProfiles(LocalModelInfo mo Profile(model, NativeContextTokens, KvCachePrecision.F16), ]; + // The RTX Spark 48GB-SKU recipe launches at a single fixed context/KV + // tier (98,304 tokens, F16 KV) rather than the shared cross-product of + // tiers other models expose. + private static LocalInferenceRunProfile[] CreateRtxSpark48GbProfiles(LocalModelInfo model) => + [ + Profile(model, RtxSpark48GbContextTokens, KvCachePrecision.F16), + ]; + private static LocalInferenceRunProfile Profile( LocalModelInfo model, int contextTokens, @@ -377,6 +462,7 @@ private static LocalInferenceRunProfile Profile( NativeContextTokens => RuntimeWorkspaceReserveBytes, IntermediateContextTokens => IntermediateContextWorkspaceReserveBytes, ReducedContextTokens => ReducedContextWorkspaceReserveBytes, + RtxSpark48GbContextTokens => ReducedContextWorkspaceReserveBytes, MinimumContextTokens => MinimumContextWorkspaceReserveBytes, _ => throw new ArgumentOutOfRangeException(nameof(contextTokens)), }; @@ -408,23 +494,29 @@ private static PinnedArtifact ModelArtifact( private static LocalModelRunRecipe Recipe( int fullAttentionLayerCount, int keyValueHeadCount, - double temperature) => + double temperature, + int batchTokens = 4_096, + int microBatchTokens = 4_096, + SpeculativeDecodingMode speculativeDecoding = SpeculativeDecodingMode.DraftMtp, + int speculativeDraftMaxTokens = 3, + PinnedArtifact? draftWeights = null) => new( - batchTokens: 4_096, - microBatchTokens: 4_096, + batchTokens: batchTokens, + microBatchTokens: microBatchTokens, parallelRequests: 1, fullAttentionLayerCount: fullAttentionLayerCount, keyValueHeadCount: keyValueHeadCount, keyValueHeadDimension: 256, flashAttention: true, offloadAllLayers: true, - speculativeDecoding: SpeculativeDecodingMode.DraftMtp, - speculativeDraftMaxTokens: 3, + speculativeDecoding: speculativeDecoding, + speculativeDraftMaxTokens: speculativeDraftMaxTokens, sampling: new ModelSamplingPreset( Temperature: temperature, TopK: 20, TopP: 0.95, MinP: 0.0, RepetitionPenalty: 1.0, - PresencePenalty: 0.0)); + PresencePenalty: 0.0), + draftWeights: draftWeights); } diff --git a/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs new file mode 100644 index 000000000..38040b24b --- /dev/null +++ b/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs @@ -0,0 +1,62 @@ +namespace OpenClaw.Shared.Inference.Catalog; + +/// +/// Default-recipe routing for RTX Spark, a unified-memory SKU where "total +/// CUDA-visible memory" identifies the physical memory SKU rather than a +/// discrete GPU's VRAM. RTX Spark bypasses the generic priority/fit-test +/// default pick () in favor of a fixed +/// SKU-to-recipe table NVIDIA specified for this hardware. Runtime/driver +/// eligibility is still assessed uniformly afterward by +/// for both paths. +/// +internal static class RtxSparkInferenceSelector +{ + // Empirically confirmed on real RTX Spark hardware: a 48GB-SKU unit's + // cuMemGetInfo total reads ~48.59e9 bytes (~45.25 GiB), tracking the + // nominal decimal-GB SKU size closely. This is NOT what nvidia-smi's "FB + // Memory Usage" (NVML) or Windows' "Total Physical Memory" report on the + // same box -- both read far lower on Spark's unified-memory design and + // must never be used for SKU classification; only GpuInfo.GpuVisibleMemoryBytes + // (CudaHostHardwareProbe's cuMemGetInfo reading) is reliable here. + // Boundaries are geometric midpoints between nominal decimal-GB SKU + // sizes; only the 48GB boundary is hardware-verified today. + private const long DecimalGigabyte = 1_000_000_000L; + + private static readonly long s_boundary32_48 = GeometricMidpointBytes(32, 48); + private static readonly long s_boundary48_64 = GeometricMidpointBytes(48, 64); + private static readonly long s_boundary64_128 = GeometricMidpointBytes(64, 128); + + /// + /// Returns the default (model, profile) pick for a detected RTX Spark + /// GPU, or null when this SKU has no recommended local model. + /// + internal static (LocalModelInfo Model, LocalInferenceRunProfile Profile)? SelectDefault(GpuInfo sparkGpu) + { + ArgumentNullException.ThrowIfNull(sparkGpu); + long totalBytes = LocalInferenceQualificationPolicy.GetEffectiveTotalMemoryBytes(sparkGpu); + + return totalBytes switch + { + _ when totalBytes < s_boundary32_48 => null, // 32GB SKU: no local AI recommended + _ when totalBytes < s_boundary48_64 => Recipe(LocalModelCatalog.Qwen35B_IQ4XSModelId), + _ when totalBytes < s_boundary64_128 => Recipe(LocalModelCatalog.Qwen38_27BModelId, ReducedQ8_0ProfileId), + _ => Recipe(LocalModelCatalog.Qwen38_27B_DFlashModelId), + }; + } + + private const string ReducedQ8_0ProfileId = "ctx-131072-q8_0"; + + private static (LocalModelInfo, LocalInferenceRunProfile) Recipe(string modelId, string? profileId = null) + { + LocalModelInfo model = LocalModelCatalog.Find(modelId) + ?? throw new InvalidOperationException($"RTX Spark recipe '{modelId}' is missing from the catalog."); + LocalInferenceRunProfile profile = profileId is null + ? LocalModelCatalog.GetProfiles(model)[0] + : LocalModelCatalog.FindProfile(model, profileId) + ?? throw new InvalidOperationException($"RTX Spark recipe '{modelId}' is missing profile '{profileId}'."); + return (model, profile); + } + + private static long GeometricMidpointBytes(int lowerNominalGb, int upperNominalGb) => + (long)(Math.Sqrt((double)lowerNominalGb * upperNominalGb) * DecimalGigabyte); +} diff --git a/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs b/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs index c0a8a3fcd..eff8285d0 100644 --- a/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs +++ b/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs @@ -283,6 +283,9 @@ private async Task RefreshAvailabilityAsync(CancellationTokenSource cancellation // dispatched callback runs) would make a queued-but-not-yet-run callback's own // IsCurrentAvailabilityProbe guard fail against itself, silently dropping a real // asynchronous DispatcherQueue completion. + // Captured before the probe so the managed-install receipt is read on the caller's + // thread rather than on whatever thread resumes after the awaited probe. + string? installedModelId = _runtimeSnapshot.ModelId; try { HostHardwareInfo hardware = await Task.Run( @@ -292,7 +295,16 @@ private async Task RefreshAvailabilityAsync(CancellationTokenSource cancellation // not the currently selected/installed model. A selection-specific failure (unknown, // deprecated, or oversized model) must not report the device itself as unavailable // and block retry-setup from switching to a compatible catalog model. - LocalInferenceEligibilityResult eligibility = LocalInferenceEligibility.Evaluate(hardware); + // + // The one exception is a SKU that has no recommended default at all (RTX Spark + // 32 GB). That says nothing about whether this device can run what is already + // installed, so the managed-install receipt is taken into account: an installed + // model that still qualifies keeps this entry point available, which is what lets + // Retry Setup reach the manifest-aware recovery path and Change Model stay enabled. + // A machine with no managed receipt still reports unavailable, and a receipt whose + // model is unknown or no longer fits reports that model's own reason. + LocalInferenceEligibilityResult eligibility = + LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, installedModelId); if (eligibility.FailureCode == LocalInferenceEligibilityFailureCode.HardwareFactsIncomplete) { // Incomplete facts (a CUDA read that came back partial or transient) are @@ -520,8 +532,16 @@ private void ApplyOnUiThread(Action action) private void ApplyRuntimeSnapshot(LocalAiRuntimeSnapshot snapshot) { + string? previousModelId = _runtimeSnapshot.ModelId; _runtimeSnapshot = snapshot; OnPropertyChanged(null); + // Availability reads the managed-install receipt (see RefreshAvailabilityAsync), which the + // runtime refresh may only publish after that read has already happened. Recomputing when + // the model id changes is what keeps a first visit correct: otherwise a 32 GB Spark whose + // receipt arrives late stays pinned at NotRecommendedForSku, with Retry Setup and Change + // Model disabled and Recheck unavailable, until the page is left and reopened. + if (IsActive && !string.Equals(previousModelId, snapshot.ModelId, StringComparison.Ordinal)) + StartAvailabilityRefresh(); } private void ApplyGatewaySnapshot(GatewayConnectionSnapshot snapshot) { From b7d50a50138b711e21ba6b304ceeb3b34cd644db Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Wed, 23 Sep 2026 14:23:48 -0400 Subject: [PATCH 03/13] feat(local-ai): branch llama-server preset on speculative decoding mode BuildPreset hardcoded spec-type = draft-mtp unconditionally. Branches on LocalModelRunRecipe.SpeculativeDecoding so DraftDFlash recipes emit spec-type = draft-dflash plus spec-draft-model, and None emits no spec-* lines at all. DraftDFlash still requires a resolved draftModelPath; that path isn't threaded from the install manifest yet (the DFlash draft checkpoint isn't acquired/verified through HuggingFaceModelInstaller in this change), so BuildPreset fails closed with InvalidDataException rather than silently omitting the draft model. Follow-up work. --- .../LocalAi/LlamaServerRouterConfiguration.cs | 34 ++++++++++++++++--- 1 file changed, 29 insertions(+), 5 deletions(-) diff --git a/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs b/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs index 49f9d558e..ebbea9e35 100644 --- a/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs +++ b/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs @@ -84,7 +84,11 @@ private static LlamaServerRouterLaunchPlan BuildCore( .WithComparers(StringComparer.OrdinalIgnoreCase) .Add("CUDA_VISIBLE_DEVICES", manifest.SelectedGpuId), presetPath, - BuildPreset(model, profile, modelPath), + // TODO(rtx-spark-dflash): DraftDFlash recipes need their pinned + // draft checkpoint acquired and verified alongside the primary + // weights before a real path can be threaded through here; see + // BuildPreset's draftModelPath parameter. + BuildPreset(model, profile, modelPath, draftModelPath: null), model.Id); } @@ -150,12 +154,17 @@ internal static void ValidateArtifactReceipts( private static string BuildPreset( LocalModelInfo model, LocalInferenceRunProfile profile, - string modelPath) + string modelPath, + string? draftModelPath) { if (modelPath.IndexOfAny(['\r', '\n']) >= 0) throw new InvalidDataException("The managed model path cannot be represented safely in a llama-server preset."); + if (draftModelPath is not null && draftModelPath.IndexOfAny(['\r', '\n']) >= 0) + throw new InvalidDataException("The managed draft model path cannot be represented safely in a llama-server preset."); LocalModelRunRecipe recipe = model.Recipe; + if (recipe.SpeculativeDecoding == SpeculativeDecodingMode.DraftDFlash && draftModelPath is null) + throw new InvalidDataException("Draft-flash decoding requires a resolved draft model path."); ModelSamplingPreset sampling = recipe.Sampling; var preset = new StringBuilder(); preset.AppendLine("version = 1"); @@ -178,9 +187,24 @@ private static string BuildPreset( preset.AppendLine("main-gpu = 0"); preset.AppendLine("fit = off"); preset.AppendLine("load-mode = dio"); - preset.AppendLine("spec-type = draft-mtp"); - preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens)); - preset.AppendLine("spec-draft-backend-sampling = true"); + switch (recipe.SpeculativeDecoding) + { + case SpeculativeDecodingMode.DraftMtp: + preset.AppendLine("spec-type = draft-mtp"); + preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens)); + preset.AppendLine("spec-draft-backend-sampling = true"); + break; + case SpeculativeDecodingMode.DraftDFlash: + preset.AppendLine("spec-type = draft-dflash"); + preset.Append("spec-draft-model = ").AppendLine(draftModelPath); + preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens)); + preset.AppendLine("spec-draft-backend-sampling = true"); + break; + case SpeculativeDecodingMode.None: + break; + default: + throw new ArgumentOutOfRangeException(nameof(recipe.SpeculativeDecoding)); + } preset.Append("temperature = ").AppendLine(Invariant(sampling.Temperature)); preset.Append("top-k = ").AppendLine(Invariant(sampling.TopK)); preset.Append("top-p = ").AppendLine(Invariant(sampling.TopP)); From 92f1ec2faa352a06920765897424a82faf5a978f Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Wed, 23 Sep 2026 14:23:53 -0400 Subject: [PATCH 04/13] test(local-ai): cover RTX Spark SKU routing Adds SKU-boundary coverage for the new RtxSparkInferenceSelector (32/48/64/128GB) and a case proving a non-Spark GPU with Spark-sized memory still takes the generic path. Evaluate_RoutesRuntimeByArchitectureWithoutGpuSkuPairing used "NVIDIA RTX Spark N1X" as an arbitrary placeholder name to prove GPU name doesn't affect runtime routing. That's no longer SKU-irrelevant now that RTX Spark has real SKU routing, so the fixture GPU name is swapped to a generic dGPU. --- .../LocalInferenceQualificationTests.cs | 295 +++++++++++++++++- .../LocalAiSetupUxContractTests.cs | 25 +- .../Presentation/LocalAiPageViewModelTests.cs | 196 +++++++++++- 3 files changed, 504 insertions(+), 12 deletions(-) diff --git a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs index 471866ffc..fe9a929c4 100644 --- a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs +++ b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs @@ -298,7 +298,7 @@ UuidFailure is not null } [Theory] - [InlineData(RuntimeArchitecture.X64, "NVIDIA RTX Spark N1X", LlamaRuntimeCatalog.X64RuntimeId)] + [InlineData(RuntimeArchitecture.X64, "NVIDIA GeForce RTX 5080", LlamaRuntimeCatalog.X64RuntimeId)] [InlineData(RuntimeArchitecture.Arm64, "NVIDIA GeForce RTX 5090", LlamaRuntimeCatalog.Arm64RuntimeId)] public void Evaluate_RoutesRuntimeByArchitectureWithoutGpuSkuPairing( RuntimeArchitecture architecture, @@ -315,6 +315,265 @@ public void Evaluate_RoutesRuntimeByArchitectureWithoutGpuSkuPairing( Assert.Equal(KvCachePrecision.Q8_0, result.Plan?.Profile.KeyCachePrecision); } + // Boundaries verified against real hardware: a real 48GB-SKU RTX Spark + // reads ~45.25 GiB (48,585,498,624 bytes) via cuMemGetInfo, matching the + // "Gb48" case below almost exactly. + [Theory] + [InlineData(30, null)] // 32GB SKU: no local AI recommended + [InlineData(45, LocalModelCatalog.Qwen35B_IQ4XSModelId)] // 48GB SKU -> 24GB recipe + [InlineData(62, LocalModelCatalog.Qwen38_27BModelId)] // 64GB SKU -> 28GB recipe + [InlineData(120, LocalModelCatalog.Qwen38_27B_DFlashModelId)] // 128GB SKU -> 48GB recipe (default) + public void Evaluate_RoutesRtxSparkByFixedSkuTable(long totalGiB, string? expectedModelId) + { + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware(RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark", totalGiB, totalGiB))); + + if (expectedModelId is null) + { + Assert.Equal(LocalInferenceEligibilityStatus.Unsupported, result.Status); + Assert.Equal(LocalInferenceEligibilityFailureCode.CatalogSelectionFailed, result.FailureCode); + Assert.Equal(LocalInferenceSelectionFailureCode.NotRecommendedForSku, result.SelectionFailureCode); + Assert.Null(result.Plan); + } + else + { + Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); + Assert.Equal(expectedModelId, result.Plan?.Model.Id); + } + } + + [Fact] + public void Evaluate_Rtx5090WithSparkSizedMemoryIgnoresSkuTable() + { + // A non-Spark GPU that happens to have Spark-sized memory must still + // take the generic priority/fit-test path -- SKU routing is keyed + // strictly off the RTX Spark name, not memory size. + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware(RuntimeArchitecture.X64, Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", totalGiB: 45, freeGiB: 45))); + + Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); + Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id); + } + + [Fact] + public void Evaluate_SparkRecipeIsBoundToTheSparkGpuOnMixedHosts() + { + // The SKU table answers "what should THIS Spark run", so the recipe and the + // GPU that runs it must be the same adapter. A discrete GPU with more free + // memory must not win the eligibility ranking and end up running a recipe + // that was chosen for the Spark. + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware( + RuntimeArchitecture.Arm64, + Gpu("NVIDIA RTX Spark N1X", "GPU-spark", totalGiB: 45, freeGiB: 45), + Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", totalGiB: 80, freeGiB: 80))); + + Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); + Assert.Equal(LocalModelCatalog.Qwen35B_IQ4XSModelId, result.Plan?.Model.Id); + Assert.Equal("GPU-spark", result.SelectedGpu?.StableId); + Assert.Equal("GPU-spark", result.Plan?.BoundGpuStableId); + } + + [Fact] + public void Evaluate_UnrecommendedSparkSkuStillQualifiesADiscreteGpuOnTheSameHost() + { + // A 32 GB Spark has no recommended model, but that is a statement about the + // Spark, not about the host. An eligible discrete GPU beside it must still + // qualify through the generic path instead of the whole host being rejected. + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware( + RuntimeArchitecture.X64, + Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", totalGiB: 30, freeGiB: 30), + Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", totalGiB: 32, freeGiB: 32))); + + Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); + Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id); + Assert.Equal("GPU-5090", result.SelectedGpu?.StableId); + Assert.Null(result.Plan?.BoundGpuStableId); + } + + [Fact] + public void Evaluate_UnrecommendedSparkSkuAloneStillReportsNotRecommended() + { + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware(RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30))); + + Assert.Equal(LocalInferenceEligibilityStatus.Unsupported, result.Status); + Assert.Equal( + LocalInferenceSelectionFailureCode.NotRecommendedForSku, + result.SelectionFailureCode); + } + + [Theory] + [InlineData(45)] + [InlineData(62)] + [InlineData(120)] + public void Evaluate_SparkRecommendationRoundTrippedBySetupKeepsItsSkuProfile(long totalGiB) + { + // Normal setup persists the recommended model id and passes it back as an + // explicit request, so the recommendation must resolve identically both ways. + // Otherwise the SKU's pinned profile (the 64 GB tier's reduced context is not + // the largest that merely fits) is silently replaced by the generic fit-test. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, + Gpu("NVIDIA RTX Spark N1X", "GPU-spark", totalGiB, totalGiB)); + + LocalInferenceEligibilityResult recommended = LocalInferenceEligibility.Evaluate(hardware); + LocalInferenceEligibilityResult roundTripped = LocalInferenceEligibility.Evaluate( + hardware, + recommended.Plan!.Model.Id); + + Assert.Equal(recommended.Plan!.Model.Id, roundTripped.Plan?.Model.Id); + Assert.Equal(recommended.Plan!.Profile.Id, roundTripped.Plan?.Profile.Id); + Assert.Equal("GPU-spark", roundTripped.Plan?.BoundGpuStableId); + Assert.Equal(recommended.SelectedGpu?.StableId, roundTripped.SelectedGpu?.StableId); + } + + [Fact] + public void Evaluate_ExplicitNonRecommendedModelOnSparkStillUsesTheGenericFitTest() + { + // A real user override must not be forced onto the SKU recipe. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, + Gpu("NVIDIA RTX Spark N1X", "GPU-spark", 62, 62)); + + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + hardware, + LocalModelCatalog.Qwen27BModelId); + + Assert.Equal(LocalModelCatalog.Qwen27BModelId, result.Plan?.Model.Id); + Assert.Null(result.Plan?.BoundGpuStableId); + } + + [Fact] + public void EvaluateForConfiguredAvailability_KeepsAValidSavedModelOn32GbSpark() + { + // A 32 GB Spark has no recommended default, but it still runs a model that was + // already configured. Rerunning setup must not switch Local AI off on that machine. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30)); + + LocalInferenceEligibilityResult result = + LocalInferenceEligibility.EvaluateForConfiguredAvailability( + hardware, + LocalModelCatalog.Qwen38_27BModelId); + + Assert.True(result.CanInstall); + Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id); + Assert.NotNull(result.SelectedGpu); + } + + [Fact] + public void EvaluateForConfiguredAvailability_PreservesTheRecoveryPinnedModelAndProfileOn32GbSpark() + { + // Recovery pins the configured model and reuses its resolved plan. On a SKU with no + // recommended default that selection must survive the availability gate with the same + // model and the same profile the explicit path resolves, so a recovery rerun does not + // silently move an existing install to a different context or KV precision. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30)); + const string pinnedModelId = LocalModelCatalog.Qwen38_27BModelId; + + LocalInferenceEligibilityResult availability = + LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, pinnedModelId); + LocalInferenceEligibilityResult pinned = + LocalInferenceEligibility.Evaluate(hardware, pinnedModelId); + + Assert.True(availability.CanInstall); + Assert.Equal(pinnedModelId, availability.Plan?.Model.Id); + Assert.Equal(pinned.Plan?.Profile.Id, availability.Plan?.Profile.Id); + Assert.Equal(pinned.Plan?.Profile.ContextTokens, availability.Plan?.Profile.ContextTokens); + Assert.Equal(pinned.Plan?.Profile.KeyCachePrecision, availability.Plan?.Profile.KeyCachePrecision); + Assert.Equal(pinned.SelectedGpu?.StableId, availability.SelectedGpu?.StableId); + } + + [Fact] + public void EvaluateForConfiguredAvailability_WithNoSavedModelStillReportsNotRecommended() + { + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30)); + + LocalInferenceEligibilityResult result = + LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, configuredModelId: null); + + Assert.False(result.CanInstall); + Assert.Equal( + LocalInferenceSelectionFailureCode.NotRecommendedForSku, + result.SelectionFailureCode); + } + + [Fact] + public void EvaluateForConfiguredAvailability_UnknownSavedModelReportsThatModelsFailure() + { + // The reason must name what is wrong with the saved selection, not fall back to the + // SKU's generic no-recommendation message, or setup offers no path to recovery. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30)); + + LocalInferenceEligibilityResult result = + LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, "no-such-model-id"); + + Assert.False(result.CanInstall); + Assert.Equal(LocalInferenceSelectionFailureCode.UnknownModel, result.SelectionFailureCode); + Assert.Equal( + LocalInferenceUnavailableReasonKind.UnknownModel, + LocalInferenceEligibilityDiagnostics.GetUnavailableReason(result).Kind); + } + + [Fact] + public void EvaluateForConfiguredAvailability_OversizedSavedModelReportsCapacityNotSkuPolicy() + { + // A saved model that no longer fits must report the capacity shortfall, including the + // model name and the required and detected memory the setup page renders. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-sparkSmall", 12, 12)); + + LocalInferenceEligibilityResult result = + LocalInferenceEligibility.EvaluateForConfiguredAvailability( + hardware, + LocalModelCatalog.Qwen38_27BModelId); + + Assert.False(result.CanInstall); + Assert.Equal( + LocalInferenceEligibilityFailureCode.InsufficientGpuMemory, + result.FailureCode); + LocalInferenceUnavailableReason reason = + LocalInferenceEligibilityDiagnostics.GetUnavailableReason(result); + Assert.Equal(LocalInferenceUnavailableReasonKind.InsufficientGpuMemory, reason.Kind); + Assert.False(string.IsNullOrWhiteSpace(reason.ModelDisplayName)); + } + + [Fact] + public void EvaluateForConfiguredAvailability_LeavesNonSparkDevicesUnchanged() + { + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.X64, Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", 32, 32)); + + LocalInferenceEligibilityResult withSaved = + LocalInferenceEligibility.EvaluateForConfiguredAvailability( + hardware, LocalModelCatalog.Qwen27BModelId); + LocalInferenceEligibilityResult device = LocalInferenceEligibility.Evaluate(hardware); + + Assert.True(withSaved.CanInstall); + Assert.Equal(device.Plan?.Model.Id, withSaved.Plan?.Model.Id); + } + + [Fact] + public void Evaluate_HugeNonSparkGpuStillDefaultsToTheRecommendedModel() + { + // A priority-0, IsExplicitAlternative model must never win the generic + // default/fallback pick regardless of available memory. The guard in + // SelectDefaultModelAndProfile keeps that true as the catalog grows. + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware(RuntimeArchitecture.X64, Gpu("NVIDIA arbitrary huge adapter", "GPU-huge", totalGiB: 200, freeGiB: 200))); + + Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); + Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id); + Assert.DoesNotContain( + LocalModelCatalog.Models, + m => m.RecommendationPriority == 0 && m.IsExplicitAlternative && m.Id == result.Plan!.Model.Id); + } + [Fact] public void Evaluate_UnsetModelChoosesHighestPriorityModelThatFitsTotalCapacity() { @@ -565,4 +824,38 @@ private static GpuInfo Gpu( CudaMajorVersion: 13, StableId: stableId); + [Theory] + [InlineData("b10655-cuda13-x64", "b10655")] + [InlineData("b10655-cuda13-arm64", "b10655")] + public void FindInstalled_ResolvesRetiredRuntimeSoExistingInstallsStayLaunchable( + string runtimeId, + string expectedReleaseTag) + { + // An installation recorded before the runtime bump must keep resolving its own + // receipt, otherwise updating the app strands it until setup repairs it. + LlamaRuntimeVariant? installed = LlamaRuntimeCatalog.FindInstalled(runtimeId); + + Assert.NotNull(installed); + Assert.Equal(runtimeId, installed.Id); + Assert.Equal(expectedReleaseTag, installed.ReleaseTag); + Assert.False(string.Equals(LlamaRuntimeCatalog.ReleaseTag, installed.ReleaseTag, StringComparison.Ordinal)); + } + + [Fact] + public void RetiredRuntimeIsNeverOfferedForNewInstalls() + { + Assert.DoesNotContain( + LlamaRuntimeCatalog.Variants, + variant => variant.ReleaseTag != LlamaRuntimeCatalog.ReleaseTag); + Assert.All( + LlamaRuntimeCatalog.Variants, + variant => Assert.Equal(LlamaRuntimeCatalog.ReleaseTag, variant.ReleaseTag)); + } + + [Fact] + public void FindInstalled_RejectsUnknownRuntimeId() + { + Assert.Null(LlamaRuntimeCatalog.FindInstalled("b00000-cuda13-x64")); + Assert.Null(LlamaRuntimeCatalog.FindInstalled(null)); + } } diff --git a/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs b/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs index 1ade70466..c191edc98 100644 --- a/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs +++ b/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs @@ -249,13 +249,17 @@ public void CapabilitiesReview_InstallDistroCard_UsesSimplifiedCopy() } /// - /// The "is Local AI unavailable" gate and the recommended/selected model must be decided - /// from device-level eligibility (the best catalog model this hardware can run), not from - /// the currently configured SelectedModelId. A stale/removed model, or one that exists but - /// this hardware cannot run at all, must be reconciled to a valid one instead of making an - /// otherwise-capable device look unavailable or leaving setup on a known-incompatible model. - /// A merely busy GPU (EligibleButBusy) is not reconciled away: CanInstall covers that case + /// The recommended model must be decided from device-level eligibility (the best catalog + /// model this hardware can run), not from the currently configured SelectedModelId. A + /// stale/removed model, or one that exists but this hardware cannot run at all, must be + /// reconciled to a valid one instead of leaving setup on a known-incompatible model. A + /// merely busy GPU (EligibleButBusy) is not reconciled away: CanInstall covers that case /// and the same model would still work once the GPU frees up. + /// + /// The "is Local AI unavailable" gate additionally honours an already configured model. + /// A SKU with no recommended default (RTX Spark 32 GB) is a statement about what to + /// install by default, not about what the device can run, so a configured selection that + /// still passes the capacity fit-test keeps Local AI available on a setup rerun. /// [Fact] public void CapabilitiesReview_GatesOnDeviceEligibilityAndReconcilesStaleSelectedModel() @@ -272,15 +276,18 @@ public void CapabilitiesReview_GatesOnDeviceEligibilityAndReconcilesStaleSelecte AssertInOrder( method, "LocalInferenceEligibilityResult deviceEligibility = LocalInferenceEligibility.Evaluate(_localAiHardware);", - "if (!deviceEligibility.CanInstall || deviceEligibility.Plan is null || deviceEligibility.SelectedGpu is null)", - "hardwareReason = DescribeLocalAiUnavailable(deviceEligibility);", + "_localAiRecommendedModelId = deviceEligibility.CanInstall", + "LocalInferenceEligibility.EvaluateForConfiguredAvailability(", + "_config!.LocalAi.SelectedModelId);", + "if (!availability.CanInstall || availability.Plan is null || availability.SelectedGpu is null)", + "hardwareReason = DescribeLocalAiUnavailable(availability);", "LocalInferenceEligibilityResult selectedEligibility =", "LocalInferenceEligibility.Evaluate(_localAiHardware, selectedModelId);", "if (_localAiRecoveryModelPinned)", "eligibility = selectedEligibility;", "else if (!selectedEligibility.CanInstall)", "_config.LocalAi.SelectedModelId = null;", - "_config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? deviceEligibility.Plan.Model.Id;", + "_config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? availability.Plan.Model.Id;", "eligibility ??= LocalInferenceEligibility.Evaluate(", "_config.LocalAi.SelectedModelId);"); } diff --git a/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs b/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs index 3c9261513..50a859f9b 100644 --- a/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs +++ b/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs @@ -737,6 +737,185 @@ private static async Task WaitForConditionAsync(Func condition, TimeSpan t } } + /// + /// A 32 GB RTX Spark. The fixed SKU table gives this device no recommended default, which + /// is a statement about fresh installs rather than about what the device can run. + /// + private static HostHardwareInfo Create32GbSparkHardware() => + new( + Architecture.Arm64, + TotalPhysicalMemoryBytes: 128_000_000_000, + AvailablePhysicalMemoryBytes: 96_000_000_000, + Gpus: + [ + new GpuInfo( + GpuVendor.Nvidia, + "NVIDIA RTX Spark N1X", + GpuVisibleMemoryBytes: 30L * 1024 * 1024 * 1024, + FreeGpuVisibleMemoryBytes: 30L * 1024 * 1024 * 1024, + DriverVersion: "620.0", + CudaMajorVersion: 13, + StableId: "GPU-spark32"), + ], + VulkanAvailable: false); + + private static LocalAiPageViewModel CreateViewModel( + LocalAiRuntimeSnapshot snapshot, + HostHardwareInfo hardware, + out FakeAppCommands commands, + out PermissionsPageRuntimeSource gatewaySource) + { + var runtime = new FakeLocalAiRuntime(snapshot); + var runtimeHost = new FakePermissionsPageRuntimeHost + { + ConnectionSnapshot = GatewayConnectionSnapshot.Idle with + { + OperatorState = RoleConnectionState.Idle, + }, + }; + gatewaySource = new PermissionsPageRuntimeSource(runtimeHost); + commands = new FakeAppCommands(); + return new LocalAiPageViewModel( + runtime, + gatewaySource, + commands, + new RecordingUiDispatcher(), + new FixedHardwareProbe(hardware)); + } + + /// + /// A 32 GB Spark whose managed receipt names a model this hardware still runs keeps the + /// Local AI entry point available, so a broken or unverified runtime can reach Retry Setup + /// and an installed model can still be changed. + /// + [Fact] + public async Task Spark32Gb_WithValidManagedReceipt_KeepsRetrySetupReachable() + { + LocalAiRuntimeSnapshot snapshot = CreateInstalledSnapshot() with + { + ModelId = LocalModelCatalog.Qwen38_27BModelId, + State = LocalAiRuntimeState.Failed, + ModelEvidence = new LocalAiModelEvidence( + LocalAiModelAvailabilityState.Unknown, + DateTimeOffset.UtcNow), + }; + using var viewModel = CreateViewModel( + snapshot, Create32GbSparkHardware(), out _, out PermissionsPageRuntimeSource source); + using (source) + { + await ActivateAndWaitForAvailabilityAsync(viewModel); + + Assert.True(viewModel.IsAvailabilityKnown); + Assert.True(viewModel.IsLocalAiAvailable); + Assert.True(viewModel.IsSetupAvailable); + Assert.True(viewModel.CanRetrySetup); + } + } + + /// + /// The same 32 GB Spark with no managed installation still receives no default, so Local AI + /// stays unavailable for a fresh setup on that SKU. + /// + [Fact] + public async Task Spark32Gb_WithNoManagedReceipt_StaysUnavailable() + { + LocalAiRuntimeSnapshot snapshot = CreateInstalledSnapshot() with + { + ModelId = null, + State = LocalAiRuntimeState.NotInstalled, + Ownership = LocalAiOwnership.None, + ModelEvidence = new LocalAiModelEvidence( + LocalAiModelAvailabilityState.Unknown, + DateTimeOffset.UtcNow), + }; + using var viewModel = CreateViewModel( + snapshot, Create32GbSparkHardware(), out _, out PermissionsPageRuntimeSource source); + using (source) + { + await ActivateAndWaitForAvailabilityAsync(viewModel); + + Assert.True(viewModel.IsAvailabilityKnown); + Assert.False(viewModel.IsLocalAiAvailable); + Assert.False(viewModel.IsSetupAvailable); + Assert.False(viewModel.CanRetrySetup); + Assert.False(viewModel.CanChangeModel); + // NotRecommendedForSku has no dedicated reason kind, so this currently surfaces the + // generic Unknown copy. Asserted so the behavior is recorded rather than assumed. + Assert.Equal( + LocalInferenceUnavailableReasonKind.Unknown, + viewModel.LocalAiUnavailableReason?.Kind); + } + } + + /// + /// The runtime refresh can publish the managed receipt after the availability probe has + /// already read the earlier snapshot. Availability must be recomputed when that happens, + /// otherwise a first visit leaves a 32 GB Spark fixed at NotRecommendedForSku with Retry + /// Setup and Change Model disabled and Recheck unavailable, until the page is reopened. + /// + [Fact] + public async Task Spark32Gb_WhenReceiptArrivesAfterAvailability_RecomputesAndKeepsRetrySetupReachable() + { + LocalAiRuntimeSnapshot pending = CreateInstalledSnapshot() with + { + ModelId = null, + State = LocalAiRuntimeState.Failed, + ModelEvidence = new LocalAiModelEvidence( + LocalAiModelAvailabilityState.Unknown, + DateTimeOffset.UtcNow), + }; + var refreshGate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + var runtime = new FakeLocalAiRuntime(pending) + { + RefreshDelay = refreshGate.Task, + RefreshResult = pending with { ModelId = LocalModelCatalog.Qwen38_27BModelId }, + }; + using var gatewaySource = new PermissionsPageRuntimeSource(new FakePermissionsPageRuntimeHost()); + using var viewModel = new LocalAiPageViewModel( + runtime, + gatewaySource, + new FakeAppCommands(), + new RecordingUiDispatcher(), + new FixedHardwareProbe(Create32GbSparkHardware())); + + await ActivateAndWaitForAvailabilityAsync(viewModel); + + // The receipt has not been published yet, so the SKU alone leaves the entry point closed. + Assert.False(viewModel.IsLocalAiAvailable); + Assert.False(viewModel.CanRetrySetup); + + refreshGate.TrySetResult(); + await WaitForAsync(viewModel, () => viewModel.IsLocalAiAvailable); + + Assert.True(viewModel.IsSetupAvailable); + Assert.True(viewModel.CanRetrySetup); + } + + /// + /// A receipt naming a model this catalog no longer knows reports that model's own failure, + /// so the page explains what to fix instead of showing the SKU's generic message. + /// + [Fact] + public async Task Spark32Gb_WithUnknownReceiptModel_ReportsThatModelsReason() + { + LocalAiRuntimeSnapshot snapshot = CreateInstalledSnapshot() with + { + ModelId = "no-such-model-id", + }; + using var viewModel = CreateViewModel( + snapshot, Create32GbSparkHardware(), out _, out PermissionsPageRuntimeSource source); + using (source) + { + await ActivateAndWaitForAvailabilityAsync(viewModel); + + Assert.True(viewModel.IsAvailabilityKnown); + Assert.False(viewModel.IsLocalAiAvailable); + Assert.Equal( + LocalInferenceUnavailableReasonKind.UnknownModel, + viewModel.LocalAiUnavailableReason?.Kind); + } + } + private static HostHardwareInfo CreateQualifiedHardware() => new( Architecture.X64, @@ -954,8 +1133,21 @@ public Task RestartAsync(CancellationToken cancellationT return Task.FromResult(Snapshot); } - public Task RefreshAsync(CancellationToken cancellationToken = default) => - Task.FromResult(Snapshot); + /// Snapshot the refresh publishes, letting a test model a receipt that the runtime + /// only resolves after construction. + public LocalAiRuntimeSnapshot? RefreshResult { get; init; } + + /// Holds the refresh open until this completes, so a test can land the refresh + /// after the availability probe has already read the earlier snapshot. + public Task? RefreshDelay { get; init; } + + public async Task RefreshAsync(CancellationToken cancellationToken = default) + { + if (RefreshDelay is { } delay) + await delay.ConfigureAwait(false); + Snapshot = RefreshResult ?? Snapshot; + return Snapshot; + } public ValueTask DisposeAsync() => ValueTask.CompletedTask; } From e8ce55fca5c594bcb18d53285e1515c2c2ca2a90 Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Wed, 23 Sep 2026 15:22:54 -0400 Subject: [PATCH 05/13] feat(local-ai): account for the draft checkpoint in memory qualification LocalModelCatalog.AdditionalArtifacts() gives every acquirer, manifest, and launch path a single fixed ordering for a recipe's non-primary pinned artifacts. Today that is the DFlash draft checkpoint. GetRequiredMemoryBytes now includes the draft checkpoint's weights, so a DFlash recipe does not under-report the memory it needs during qualification. --- .../Inference/Catalog/LocalInferenceSelector.cs | 8 ++++---- .../Inference/Catalog/LocalModelCatalog.cs | 14 ++++++++++++++ 2 files changed, 18 insertions(+), 4 deletions(-) diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs index cd684e342..1142fc6db 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs @@ -216,12 +216,12 @@ public static long GetRequiredMemoryBytes( { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(profile); - long draftWeightsBytes = model.Recipe.DraftWeights?.SizeBytes ?? 0; + long weightsBytes = SaturatingAdd( + model.Weights.SizeBytes, + model.Recipe.DraftWeights?.SizeBytes ?? 0); return SaturatingAdd( SaturatingAdd( - SaturatingAdd( - SaturatingAdd(model.Weights.SizeBytes, draftWeightsBytes), - GetKvCacheMemoryBytes(model.Recipe, profile)), + SaturatingAdd(weightsBytes, GetKvCacheMemoryBytes(model.Recipe, profile)), GetDraftKvCacheMemoryBytes(model.Recipe, profile)), profile.RuntimeWorkspaceBytes); } diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs index e1f9dd9ac..96f31d96d 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs @@ -1,3 +1,4 @@ +using System.Collections.Immutable; using System.Collections.ObjectModel; namespace OpenClaw.Shared.Inference.Catalog; @@ -423,6 +424,19 @@ public static IReadOnlyList GetProfiles(LocalModelInfo /// True when the id resolves only to a retired catalog entry. public static bool IsLegacy(string? id) => Find(id) is null && FindInstalled(id) is not null; + /// + /// The recipe's additional pinned artifacts beyond its primary weights, in the + /// fixed order every acquirer, manifest, and launch path must agree on. Today that + /// is the DFlash draft checkpoint, when the recipe pins one. + /// + public static ImmutableArray AdditionalArtifacts(LocalModelInfo model) + { + ArgumentNullException.ThrowIfNull(model); + return model.Recipe.DraftWeights is { } draftWeights + ? [draftWeights] + : ImmutableArray.Empty; + } + private static LocalInferenceRunProfile[] CreateProfiles(LocalModelInfo model) => [ Profile(model, NativeContextTokens, KvCachePrecision.F16), From 685e8e0990205fed25d204a09d07aa4ccc88867b Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Wed, 23 Sep 2026 15:22:59 -0400 Subject: [PATCH 06/13] build(local-ai): bump managed llama-server runtime to b11026 b10655 (CUDA 13.3 on x64) predates DFlash draft-decoding support and the 96GB recipe's validated build. Bumping to b11026 puts both x64 and arm64 on CUDA 13.4 and covers every RTX Spark recipe with one pinned runtime, matching the existing single-global-pin design instead of adding per-recipe runtime routing. --- .../LocalAi/LlamaServerRouterConfiguration.cs | 8 +- .../LocalAiInstallReconciler.cs | 150 ++++++++++++++++-- src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs | 5 + .../Inference/Catalog/LlamaRuntimeCatalog.cs | 121 ++++++++++++-- .../LocalAiPortLifecycleTests.cs | 37 +++++ .../LocalAiInstallRecoveryTests.cs | 49 ++++++ 6 files changed, 340 insertions(+), 30 deletions(-) diff --git a/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs b/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs index ebbea9e35..03fead054 100644 --- a/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs +++ b/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs @@ -53,8 +53,10 @@ private static LlamaServerRouterLaunchPlan BuildCore( LocalAiInstallManifest manifest = install.Manifest; int port = listenPort ?? manifest.RequestedPort; LocalAiPortPolicy.Validate(port); - LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.Variants.SingleOrDefault( - candidate => string.Equals(candidate.Id, manifest.RuntimeId, StringComparison.Ordinal)) + // FindInstalled, not Variants: an installation recorded before the last + // runtime bump must keep launching until setup upgrades it, instead of being + // stranded the moment the catalog moves to a newer pinned release. + LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.FindInstalled(manifest.RuntimeId) ?? throw new InvalidDataException("The managed llama-server runtime is no longer qualified."); LocalModelInfo model = LocalModelCatalog.FindInstalled(manifest.ModelCatalogId) ?? throw new InvalidDataException("The managed local AI model is no longer qualified."); @@ -124,7 +126,7 @@ internal static void ValidateArtifactReceipts( { throw new InvalidDataException("The managed local AI architecture and runtime receipt do not match."); } - if (!string.Equals(manifest.EngineVersion, LlamaRuntimeCatalog.ReleaseTag, StringComparison.Ordinal) || + if (!string.Equals(manifest.EngineVersion, runtime.ReleaseTag, StringComparison.Ordinal) || !string.Equals(manifest.ModelAlias, model.Id, StringComparison.Ordinal)) { throw new InvalidDataException("The managed local AI model recipe receipt does not match the qualified catalog."); diff --git a/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs b/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs index ebcafdc94..fa2ec8b47 100644 --- a/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs +++ b/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs @@ -1,3 +1,4 @@ +using System.Collections.Immutable; using OpenClaw.Connection.LocalAi; using OpenClaw.Shared.Inference.Catalog; @@ -8,7 +9,8 @@ internal sealed record LocalAiReconcileResult( LocalAiResolvedInstall? ResolvedInstall, LlamaRuntimeInstallResult? RuntimeInstall, HuggingFaceModelInstallResult? ModelInstall, - LocalAiResolvedInstall? OriginalInstall = null) + LocalAiResolvedInstall? OriginalInstall = null, + ImmutableArray? AdditionalModelInstalls = null) { public static LocalAiReconcileResult NotInstalled { get; } = new(false, null, null, null); @@ -26,6 +28,18 @@ Task VerifyLegacyCompatibilityAsync( LocalAiPaths paths, PinnedArtifact artifact, CancellationToken cancellationToken); + + /// + /// Verifies one schema-5 additional model asset (a DFlash draft checkpoint, + /// or an extra split-GGUF shard) still matches its pinned receipt in the + /// shared hub cache. Additional assets have no legacy app-owned copy, so + /// unlike there is no separate schema-3 path. + /// + Task VerifyAdditionalAssetAsync( + LocalAiResolvedInstall install, + string cachedAssetPath, + PinnedArtifact artifact, + CancellationToken cancellationToken); } internal sealed class LocalAiModelFileVerifier : ILocalAiModelFileVerifier @@ -35,7 +49,7 @@ public async Task VerifyActiveAsync( PinnedArtifact artifact, CancellationToken cancellationToken) { - if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (!install.Manifest.UsesHubCache) { if (!File.Exists(install.ModelPath)) return false; @@ -62,7 +76,7 @@ public Task VerifyLegacyCompatibilityAsync( PinnedArtifact artifact, CancellationToken cancellationToken) { - if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (!install.Manifest.UsesHubCache) return Task.FromResult(true); string legacyModelPath = paths.ResolveContainedPath( @@ -75,6 +89,24 @@ public Task VerifyLegacyCompatibilityAsync( artifact, cancellationToken); } + + public async Task VerifyAdditionalAssetAsync( + LocalAiResolvedInstall install, + string cachedAssetPath, + PinnedArtifact artifact, + CancellationToken cancellationToken) + { + await using FileStream? verified = + await HuggingFaceHubCache.TryOpenVerifiedCacheFileAsync( + install.Manifest.ModelCacheRoot!, + cachedAssetPath, + artifact.SizeBytes, + artifact.Sha256, + progress: null, + cancellationToken) + .ConfigureAwait(false); + return verified is not null; + } } /// @@ -126,7 +158,7 @@ public async Task ReconcileAsync( if (install is null) return LocalAiReconcileResult.NotInstalled; LocalAiResolvedInstall originalInstall = install; - ValidateRecipeMatch(install, plan, selectedGpuId, localDataDirectory); + bool runtimeUpgradePending = ValidateRecipeMatch(install, plan, selectedGpuId, localDataDirectory); bool migrateLegacyGpuId = !string.Equals(install.Manifest.SelectedGpuId, selectedGpuId, StringComparison.Ordinal) && @@ -145,7 +177,27 @@ public async Task ReconcileAsync( plan.Model.Weights, cancellationToken) .ConfigureAwait(false); - bool modelIsValid = activeModelIsValid && legacyModelIsValid; + bool additionalAssetsAreValid = activeModelIsValid && await VerifyAdditionalAssetsAsync( + install, + plan.Model, + cancellationToken) + .ConfigureAwait(false); + bool modelIsValid = activeModelIsValid && legacyModelIsValid && additionalAssetsAreValid; + if (runtimeUpgradePending) + { + // The catalog moved to a newer pinned runtime. Drop only the runtime so the + // acquirer installs the new one, and keep the verified model and its extra + // assets so an upgrade does not re-download tens of GB. OriginalInstall lets + // the manifest step replace the existing receipt in place. + return new LocalAiReconcileResult( + Reused: false, + ResolvedInstall: null, + RuntimeInstall: null, + ModelInstall: modelIsValid ? CreateModelInstall(install, localDataDirectory) : null, + OriginalInstall: originalInstall, + AdditionalModelInstalls: modelIsValid ? CreateAdditionalModelInstalls(install) : null); + } + if (!inspection.IsValid || !modelIsValid) { if (!allowIncompleteInstallation) @@ -173,7 +225,8 @@ public async Task ReconcileAsync( ResolvedInstall: null, RuntimeInstall: inspection.IsValid ? CreateRuntimeInstall(install) : null, ModelInstall: modelIsValid ? CreateModelInstall(install, localDataDirectory) : null, - OriginalInstall: originalInstall); + OriginalInstall: originalInstall, + AdditionalModelInstalls: modelIsValid ? CreateAdditionalModelInstalls(install) : null); } install = await MigrateLegacyModelAsync( @@ -201,7 +254,8 @@ public async Task ReconcileAsync( install, CreateRuntimeInstall(install), CreateModelInstall(install, localDataDirectory), - OriginalInstall: allowIncompleteInstallation ? originalInstall : null); + OriginalInstall: allowIncompleteInstallation ? originalInstall : null, + AdditionalModelInstalls: CreateAdditionalModelInstalls(install)); } private static LlamaRuntimeInstallResult CreateRuntimeInstall(LocalAiResolvedInstall install) => @@ -222,7 +276,7 @@ private static HuggingFaceModelInstallResult CreateModelInstall( LocalAiResolvedInstall install, string localDataDirectory) { - if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (!install.Manifest.UsesHubCache) { return new HuggingFaceModelInstallResult( install.ModelPath, @@ -244,6 +298,60 @@ private static HuggingFaceModelInstallResult CreateModelInstall( LegacyCreatedThisRun: false); } + /// + /// Reconstructs the additional-asset install results a reused install's + /// manifest already proves verified, so a caller that skips re-acquiring + /// them (because just verified them) still has + /// a populated SetupContext.LocalAiAdditionalModelInstalls to persist. + /// + private static ImmutableArray CreateAdditionalModelInstalls( + LocalAiResolvedInstall install) + { + if (install.Manifest.AdditionalModelPaths.IsDefaultOrEmpty) + return ImmutableArray.Empty; + + var builder = ImmutableArray.CreateBuilder( + install.Manifest.AdditionalModelPathsOrEmpty.Length); + foreach (string cachedAssetPath in install.Manifest.AdditionalModelPathsOrEmpty) + { + builder.Add(new HuggingFaceAdditionalAssetInstallResult( + cachedAssetPath, + install.Manifest.ModelCacheRoot!, + HuggingFaceModelInstallDisposition.ReusedVerified, + CreatedThisRun: false)); + } + + return builder.MoveToImmutable(); + } + + private async Task VerifyAdditionalAssetsAsync( + LocalAiResolvedInstall install, + LocalModelInfo model, + CancellationToken cancellationToken) + { + ImmutableArray expected = LocalModelCatalog.AdditionalArtifacts(model); + if (expected.IsEmpty) + return install.Manifest.AdditionalModelAssetsOrEmpty.IsEmpty; + if (install.Manifest.AdditionalModelPathsOrEmpty.Length != expected.Length) + return false; + + for (int i = 0; i < expected.Length; i++) + { + if (!await _modelVerifier + .VerifyAdditionalAssetAsync( + install, + install.Manifest.AdditionalModelPathsOrEmpty[i], + expected[i], + cancellationToken) + .ConfigureAwait(false)) + { + return false; + } + } + + return true; + } + private async Task MigrateLegacyModelAsync( LocalAiResolvedInstall install, LocalAiPaths paths, @@ -271,26 +379,36 @@ private async Task MigrateLegacyModelAsync( .ConfigureAwait(false) ?? throw new InvalidDataException( "The Local AI installation manifest disappeared during cache migration."); - ValidateRecipeMatch(migrated, plan, selectedGpuId, localDataDirectory); + _ = ValidateRecipeMatch(migrated, plan, selectedGpuId, localDataDirectory); return migrated; } - private static void ValidateRecipeMatch( + /// + /// Validates the receipt against the runtime and recipe it actually recorded, and + /// reports whether the catalog has since moved to a newer pinned runtime. A version + /// difference is an expected upgrade, not a corrupt install, so it must not throw: + /// throwing here ends setup with an uninstall instruction instead of upgrading. + /// + private static bool ValidateRecipeMatch( LocalAiResolvedInstall install, LocalInferencePlan plan, string selectedGpuId, string localDataDirectory) { LocalAiInstallManifest manifest = install.Manifest; + LlamaRuntimeVariant installedRuntime = + LlamaRuntimeCatalog.FindInstalled(manifest.RuntimeId) ?? plan.Runtime; + bool runtimeUpgradePending = + !string.Equals(installedRuntime.Id, plan.Runtime.Id, StringComparison.Ordinal); string expectedArchitecture = plan.Runtime.Architecture switch { System.Runtime.InteropServices.Architecture.X64 => "x64", System.Runtime.InteropServices.Architecture.Arm64 => "arm64", _ => throw new InvalidDataException("The selected Local AI runtime architecture is unsupported."), }; - if (!string.Equals(manifest.EngineVersion, LlamaRuntimeCatalog.ReleaseTag, StringComparison.Ordinal) || + if (!string.Equals(manifest.EngineVersion, installedRuntime.ReleaseTag, StringComparison.Ordinal) || !string.Equals(manifest.Architecture, expectedArchitecture, StringComparison.Ordinal) || - !string.Equals(manifest.RuntimeId, plan.Runtime.Id, StringComparison.Ordinal) || + !string.Equals(manifest.RuntimeId, installedRuntime.Id, StringComparison.Ordinal) || !string.Equals(manifest.ModelCatalogId, plan.Model.Id, StringComparison.Ordinal) || manifest.ContextLength != plan.Profile.ContextTokens || manifest.KeyCachePrecision != plan.Profile.KeyCachePrecision || @@ -303,7 +421,7 @@ private static void ValidateRecipeMatch( "The existing managed Local AI installation does not match the selected runtime, GPU, and model recipe."); } - LocalAiComponentIdentity component = LlamaRuntimeInstaller.Component(plan.Runtime); + LocalAiComponentIdentity component = LlamaRuntimeInstaller.Component(installedRuntime); if (!LocalAiPathPolicy.TryResolve( localDataDirectory, component, @@ -327,11 +445,11 @@ private static void ValidateRecipeMatch( } LlamaServerRouterConfiguration.ValidateArtifactReceipts( manifest, - plan.Runtime, + installedRuntime, plan.Model); bool modelPathMatches; - if (manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (manifest.UsesHubCache) { modelPathMatches = !string.IsNullOrWhiteSpace(manifest.ModelCacheRoot) && @@ -372,6 +490,8 @@ private static void ValidateRecipeMatch( ? "The managed model path does not match the selected catalog recipe." : error); } + + return runtimeUpgradePending; } private static bool GpuIdsMatch(string persistedGpuId, string selectedGpuId) => diff --git a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs index fb0b437a8..07d53b5d0 100644 --- a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs +++ b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs @@ -287,6 +287,11 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati } if (!result.Reused) { + // A retained baseline receipt (gateway recovery, or a pending runtime + // upgrade) is what lets the manifest step replace the existing receipt + // instead of refusing because one is already present. + if (result.OriginalInstall is { } retainedReceipt) + ctx.LocalAiRecoveryOriginalInstall ??= retainedReceipt; ctx.LocalAiRuntimeInstall = result.RuntimeInstall; ctx.LocalAiModelInstall = result.ModelInstall; return StepResult.Skip(result.OriginalInstall is null diff --git a/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs index a0b6ee26a..9242951f0 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs @@ -10,7 +10,8 @@ public LlamaRuntimeVariant( string id, Architecture architecture, Version cudaVersion, - IReadOnlyList artifacts) + IReadOnlyList artifacts, + string? releaseTag = null) { ArgumentException.ThrowIfNullOrWhiteSpace(id); ArgumentNullException.ThrowIfNull(cudaVersion); @@ -32,12 +33,20 @@ public LlamaRuntimeVariant( Architecture = architecture; CudaVersion = cudaVersion; Artifacts = artifacts; + ReleaseTag = releaseTag ?? LlamaRuntimeCatalog.ReleaseTag; } public string Id { get; } public Architecture Architecture { get; } public Version CudaVersion { get; } public IReadOnlyList Artifacts { get; } + /// + /// The llama.cpp release this variant was pinned from. Equals + /// for the current runtime and + /// keeps its own older value for a retired one, so an installed receipt is + /// validated against the release it actually recorded. + /// + public string ReleaseTag { get; } public long TotalDownloadSizeBytes => Artifacts.Sum(artifact => artifact.SizeBytes); } @@ -47,11 +56,11 @@ public LlamaRuntimeVariant( /// public static class LlamaRuntimeCatalog { - public const string ReleaseTag = "b10655"; - public const string ReleaseCommitSha = "cb300598d5f90189cb69d2702f4930aaf99d32a2"; + public const string ReleaseTag = "b11026"; + public const string ReleaseCommitSha = "b49650adb31f2e49a0d76113aeb1792134fd8413"; public const string ServerExecutableName = "llama-server.exe"; - public const string X64RuntimeId = "b10655-cuda13-x64"; - public const string Arm64RuntimeId = "b10655-cuda13-arm64"; + public const string X64RuntimeId = "b11026-cuda13-x64"; + public const string Arm64RuntimeId = "b11026-cuda13-arm64"; public static GitHubReleaseSource Source { get; } = new( "ggml-org/llama.cpp", @@ -64,43 +73,103 @@ public static class LlamaRuntimeCatalog new LlamaRuntimeVariant( X64RuntimeId, Architecture.X64, - new Version(13, 3), + new Version(13, 4), + Array.AsReadOnly( + new[] + { + RuntimeArtifact( + "llama-b11026-cuda13-x64", + ArtifactRole.RuntimeBinary, + "llama-b11026-bin-win-cuda-13.4-x64.zip", + 150_102_391, + "6799f0962d066c54aee3773f0e5efa0076e46418695c0f4f6d24a38e7007dfb1"), + RuntimeArtifact( + "cudart-b11026-cuda13-x64", + ArtifactRole.RuntimeDependency, + "cudart-llama-bin-win-cuda-13.4-x64.zip", + 423_535_356, + "738f8c251ac22b70c3ae6f83a10cf222725df0395246a2cf58f32bdb85fbe668"), + })), + new LlamaRuntimeVariant( + Arm64RuntimeId, + Architecture.Arm64, + new Version(13, 4), Array.AsReadOnly( new[] { RuntimeArtifact( + "llama-b11026-cuda13-arm64", + ArtifactRole.RuntimeBinary, + "llama-b11026-bin-win-cuda-13.4-arm64.zip", + 142_993_717, + "d4a31d05b4fe997872020d81e8482e7712c7c254ec9c7ccdb9597ae2a31e6728"), + RuntimeArtifact( + "cudart-b11026-cuda13-arm64", + ArtifactRole.RuntimeDependency, + "cudart-llama-bin-win-cuda-13.4-arm64.zip", + 153_262_407, + "642dcde8805b3e3165ca710a5443b3b4044b27d96bd3ee3132473988c9bcb774"), + })), + }); + + // Retired from new installs and never offered or selected. These exist only so a + // managed installation recorded before the runtime bump keeps resolving its own + // receipt and stays launchable until setup upgrades it. Pins are reproduced + // exactly as they were installed; nothing is remapped. + private const string LegacyB10655ReleaseTag = "b10655"; + private const string LegacyB10655RuntimeIdX64 = "b10655-cuda13-x64"; + private const string LegacyB10655RuntimeIdArm64 = "b10655-cuda13-arm64"; + + private static GitHubReleaseSource LegacyB10655Source { get; } = new( + "ggml-org/llama.cpp", + LegacyB10655ReleaseTag, + "cb300598d5f90189cb69d2702f4930aaf99d32a2"); + + private static readonly ReadOnlyCollection s_legacyVariants = Array.AsReadOnly( + new[] + { + new LlamaRuntimeVariant( + LegacyB10655RuntimeIdX64, + Architecture.X64, + new Version(13, 3), + Array.AsReadOnly( + new[] + { + LegacyB10655Artifact( "llama-b10655-cuda13-x64", ArtifactRole.RuntimeBinary, "llama-b10655-bin-win-cuda-13.3-x64.zip", 146_478_045, "be61636141327b3ca4d437c17489fd69964838a31a5fe3e97400f0dcd9f669dc"), - RuntimeArtifact( + LegacyB10655Artifact( "cudart-b10655-cuda13-x64", ArtifactRole.RuntimeDependency, "cudart-llama-bin-win-cuda-13.3-x64.zip", 390_970_417, "1462a050eb4c684921ba51dcc4cc488a036674c3e73e9945ee705b854808d03e"), - })), + }), + LegacyB10655ReleaseTag), new LlamaRuntimeVariant( - Arm64RuntimeId, + LegacyB10655RuntimeIdArm64, Architecture.Arm64, new Version(13, 4), Array.AsReadOnly( new[] { - RuntimeArtifact( + LegacyB10655Artifact( "llama-b10655-cuda13-arm64", ArtifactRole.RuntimeBinary, "llama-b10655-bin-win-cuda-13.4-arm64.zip", 140_055_278, "567e61b4129e0d5b0580e5d3ea86b82ab5b6bee745ee02f69b58af799b49a582"), - RuntimeArtifact( + LegacyB10655Artifact( "cudart-b10655-cuda13-arm64", ArtifactRole.RuntimeDependency, "cudart-llama-bin-win-cuda-13.4-arm64.zip", 153_318_797, "5a40dc7c5fa3d0a80ceeba4f16f9e8d25d87bcf1399c9233588953c43436c33c"), - })), + }), + LegacyB10655ReleaseTag), }); public static IReadOnlyList Variants => s_variants; @@ -108,6 +177,19 @@ public static class LlamaRuntimeCatalog public static LlamaRuntimeVariant? Find(Architecture architecture) => s_variants.SingleOrDefault(variant => variant.Architecture == architecture); + /// + /// Resolves a runtime id from an already-installed receipt, including a retired + /// runtime from before the last version bump. Use this only on installed-receipt + /// validation and launch paths. Selection and acquisition must keep using + /// and so a retired runtime is never + /// installed again. + /// + public static LlamaRuntimeVariant? FindInstalled(string? id) => + string.IsNullOrWhiteSpace(id) + ? null + : s_variants.SingleOrDefault(variant => string.Equals(variant.Id, id, StringComparison.Ordinal)) + ?? s_legacyVariants.SingleOrDefault(variant => string.Equals(variant.Id, id, StringComparison.Ordinal)); + private static PinnedArtifact RuntimeArtifact( string id, ArtifactRole role, @@ -122,4 +204,19 @@ private static PinnedArtifact RuntimeArtifact( sizeBytes, new Sha256Digest(sha256), LocalInferenceCatalogProvenance.NvidiaCair); + + private static PinnedArtifact LegacyB10655Artifact( + string id, + ArtifactRole role, + string fileName, + long sizeBytes, + string sha256) => + new( + id, + role, + LegacyB10655Source, + fileName, + sizeBytes, + new Sha256Digest(sha256), + LocalInferenceCatalogProvenance.NvidiaCair); } diff --git a/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs b/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs index 936247e00..426494b59 100644 --- a/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs +++ b/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs @@ -2546,6 +2546,43 @@ private static LocalAiInstallManifest LegacyQwen9BManifest() }; } + /// + /// A managed install recorded before the llama-server runtime bump must keep + /// launching against its own pinned receipt. Updating the app must not strand an + /// installed model until a separate setup repair runs. + /// + [Fact] + public async Task Router_LaunchesRetiredRuntimeInstallAfterVersionBump() + { + using var temp = new TempDirectory("local-ai-legacy-runtime-"); + var paths = new LocalAiPaths(temp.Path); + LlamaRuntimeVariant retired = LlamaRuntimeCatalog.FindInstalled("b10655-cuda13-arm64")!; + LocalAiInstallManifest manifest = ValidManifest() with + { + EngineVersion = retired.ReleaseTag, + RuntimeId = retired.Id, + ExecutablePath = Path.Combine( + "engines", + $"llama-{retired.ReleaseTag}", + LlamaRuntimeCatalog.ServerExecutableName), + RuntimeAssets = retired.Artifacts.Select(artifact => new LocalAiAssetReceipt + { + FileName = Path.GetFileName(artifact.RelativePath), + SourceUrl = artifact.DownloadUri.AbsoluteUri, + SizeBytes = artifact.SizeBytes, + Sha256 = artifact.Sha256.Value, + }).ToImmutableArray(), + }; + var store = new LocalAiManifestStore(paths); + await store.SaveAsync(manifest); + + LocalAiResolvedInstall saved = (await store.LoadAsync())!; + LlamaServerRouterLaunchPlan launch = LlamaServerRouterConfiguration.Build(paths, saved); + + Assert.NotEqual(LlamaRuntimeCatalog.ReleaseTag, retired.ReleaseTag); + Assert.Equal("qwen3.6-35b-a3b-mtp-q4-k-m", launch.ModelAlias); + } + private static LocalAiInstallManifest ValidManifest() { LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.Find( diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs index fc353ea91..cdc53f34d 100644 --- a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs +++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs @@ -1145,6 +1145,55 @@ public async Task Reconciler_RecoveryRepairsMissingSchemaFourCompatibilityCopy() Assert.Null(result.ModelInstall); } + [Fact] + public async Task Reconciler_UpgradesRetiredRuntimeReceiptInsteadOfFailingSetup() + { + // An install recorded before the runtime bump must upgrade, not end setup with an + // uninstall instruction. The runtime is dropped so the acquirer installs the new + // pin; the verified model is kept so an upgrade does not re-download it. + using var temp = new TempDirectory(); + LocalInferencePlan plan = CatalogPlan(); + const string gpuId = "GPU-0"; + var paths = new LocalAiPaths(temp.Path); + LlamaRuntimeVariant retired = LlamaRuntimeCatalog.FindInstalled("b10655-cuda13-x64")!; + Assert.True(LocalAiPathPolicy.TryResolve( + temp.Path, + LlamaRuntimeInstaller.Component(retired), + out LocalAiSetupPaths retiredPaths, + out string retiredError), retiredError); + LocalAiInstallManifest manifest = CreateManifest(temp.Path, plan, gpuId) with + { + EngineVersion = retired.ReleaseTag, + RuntimeId = retired.Id, + ExecutablePath = Path.GetRelativePath( + paths.RootDirectory, + Path.Combine(retiredPaths.InstallDirectory, LlamaRuntimeCatalog.ServerExecutableName)), + RuntimeAssets = retired.Artifacts.Select(artifact => new LocalAiAssetReceipt + { + FileName = Path.GetFileName(artifact.RelativePath), + SourceUrl = artifact.DownloadUri.AbsoluteUri, + SizeBytes = artifact.SizeBytes, + Sha256 = artifact.Sha256.Value, + }).ToImmutableArray(), + }; + await new LocalAiManifestStore(paths).SaveAsync(manifest); + var reconciler = new LocalAiInstallReconciler( + new ValidRuntimeInspector(), + new AcceptingModelVerifier()); + + LocalAiReconcileResult result = await reconciler.ReconcileAsync( + temp.Path, + plan, + gpuId, + CancellationToken.None); + + Assert.False(result.Reused); + Assert.Null(result.RuntimeInstall); + Assert.NotNull(result.ModelInstall); + Assert.NotNull(result.OriginalInstall); + Assert.Equal(retired.ReleaseTag, result.OriginalInstall!.Manifest.EngineVersion); + } + [Fact] public async Task Reconciler_RejectsMigrationCacheRootInsideManagedInstallTree() { From 00aa95f858246706980fbe681b52c7a6e55de1e6 Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Wed, 23 Sep 2026 15:23:04 -0400 Subject: [PATCH 07/13] feat(local-ai): add schema-5 manifest support for additional model assets Extends LocalAiInstallManifest with AdditionalModelAssets/ AdditionalModelPaths (schema 5) so a recipe's DFlash draft checkpoint or extra split-GGUF shards can be recorded and re-verified alongside the primary weights receipt in the same Hugging Face hub cache. Schema 4 manifests are untouched -- the new fields are empty and absent from JSON unless a recipe actually pins additional assets. Each additional-asset receipt derives its own repository/revision from its own SourceUrl (not the primary ModelId) since the DFlash draft checkpoint is pinned from a different HF repo than the primary weights. Adds LocalAiInstallManifest.UsesHubCache so schema 4 and schema 5 are treated identically everywhere the manifest previously branched only on the exact HubCacheReceiptSchemaVersion value. AdditionalModelAssets/AdditionalModelPaths are left at their unset ImmutableArray default (not .Empty) so JsonIgnoreCondition.WhenWritingDefault actually omits them for schema-3/4 manifests -- .Empty is a distinct, non-default array instance the condition never matches, so writing it would have added new fields to every existing schema-4 receipt and broken older app builds' strict unknown-field rejection. UsesHubCache is marked [JsonIgnore] for the same reason: it's a derived read helper, not part of the persisted contract. ResolveAndValidate normalizes the unset default to .Empty immediately after load so every in-memory reader keeps using ordinary IsEmpty/Length calls safely. --- .../LocalAi/LocalAiManifest.cs | 155 +++++++++++++++++- .../LocalAiManifestMigrationTests.cs | 62 +++++++ 2 files changed, 215 insertions(+), 2 deletions(-) diff --git a/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs b/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs index ce3cbe886..895ab908c 100644 --- a/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs +++ b/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs @@ -148,8 +148,23 @@ public sealed record LocalAiInstallManifest /// the model must also verify the pinned content before use. /// public const int HubCacheReceiptSchemaVersion = 4; + /// + /// Schema-4 plus one or more additional model assets verified in the same + /// hub cache. Only recipes with a LocalModelRunRecipe.DraftWeights + /// pin use this; every existing single-asset recipe keeps writing schema 4. + /// + public const int AdditionalAssetsSchemaVersion = 5; public const string SupportedEngine = "llama-server"; + /// + /// True for schema 4 and schema 5, whose active model resolves through the + /// standard Hugging Face hub cache rather than the legacy app-owned copy. + /// Derived entirely from , so it must never be + /// persisted -- an older app build would reject it as an unknown field. + /// + [JsonIgnore] + public bool UsesHubCache => SchemaVersion is HubCacheReceiptSchemaVersion or AdditionalAssetsSchemaVersion; + public int SchemaVersion { get; init; } = CurrentSchemaVersion; public string Engine { get; init; } = SupportedEngine; public required string EngineVersion { get; init; } @@ -181,6 +196,47 @@ public sealed record LocalAiInstallManifest public required string ModelAlias { get; init; } public required LocalAiAssetReceipt ModelAsset { get; init; } /// + /// Schema-5 receipts for additional model assets verified in the hub + /// cache alongside (the draft checkpoint), in + /// catalog order. Left at its unset default + /// (not .Empty) for schema-3/4 manifests, since + /// ImmutableArray<T>.Empty is a distinct, non-default instance + /// that would not omit + /// -- writing it would break older app builds' strict unknown-field + /// rejection on an otherwise-unchanged schema-4 receipt. + /// normalizes the + /// unset default to .Empty for every in-memory reader. + /// + [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingDefault)] + public ImmutableArray AdditionalModelAssets { get; init; } + /// + /// Verified hub-cache paths parallel to . + /// Same unset-default-not-Empty rule; see that property's remarks. + /// + [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingDefault)] + public ImmutableArray AdditionalModelPaths { get; init; } + /// + /// with the unset default collapsed to + /// an empty array. Read through this, never the raw property: a schema-3/4 + /// or hand-edited manifest leaves the raw value at ImmutableArray's + /// default (null-backed) instance, where Length/IsEmpty throw + /// NullReferenceException instead of the intended InvalidDataException. + /// Normalizing on read rather than rewriting the record keeps the persisted + /// JSON byte-identical -- assigning .Empty back onto the manifest + /// would make the next save emit the field on an otherwise-untouched + /// schema-4 receipt. + /// + [JsonIgnore] + public ImmutableArray AdditionalModelAssetsOrEmpty => + AdditionalModelAssets.IsDefault ? ImmutableArray.Empty : AdditionalModelAssets; + /// + /// with the unset default collapsed to + /// an empty array; see . + /// + [JsonIgnore] + public ImmutableArray AdditionalModelPathsOrEmpty => + AdditionalModelPaths.IsDefault ? ImmutableArray.Empty : AdditionalModelPaths; + /// /// The requested listener port. Zero delegates allocation to llama-server so /// the child owns the port continuously from bind through startup. /// @@ -512,7 +568,8 @@ public LocalAiResolvedInstall ResolveAndValidate(LocalAiInstallManifest manifest ArgumentNullException.ThrowIfNull(manifest); if (manifest.SchemaVersion is not ( LocalAiInstallManifest.CurrentSchemaVersion or - LocalAiInstallManifest.HubCacheReceiptSchemaVersion)) + LocalAiInstallManifest.HubCacheReceiptSchemaVersion or + LocalAiInstallManifest.AdditionalAssetsSchemaVersion)) { throw new InvalidDataException($"Unsupported local AI manifest schema version {manifest.SchemaVersion}."); } @@ -566,10 +623,11 @@ LocalAiInstallManifest.CurrentSchemaVersion or ValidateHubCacheReceipt(manifest, provenance); string legacyModel = _paths.ResolveContainedPath(manifest.ModelPath, nameof(manifest.ModelPath)); ValidateModelPath(legacyModel, manifest.ModelAsset, "legacy-compatible"); - string model = manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion + string model = manifest.UsesHubCache ? ResolveHubCacheModelPath(manifest) : legacyModel; ValidateModelPath(model, manifest.ModelAsset, "active"); + ValidateAdditionalAssets(manifest); LocalAiPortPolicy.Validate(manifest.RequestedPort); LocalAiGatewayModelPolicy.ValidateFallbackModel(manifest.GatewayFallbackModel); @@ -746,6 +804,99 @@ private static void ValidateHubCacheReceipt( private static string ResolveHubCacheModelPath(LocalAiInstallManifest manifest) => WindowsPathSafety.NormalizePath(manifest.CachedModelPath!); + /// + /// Validates schema-5 additional assets (a DFlash draft checkpoint). Each + /// receipt derives its own repository and + /// revision from its own SourceUrl -- unlike the primary + /// , additional assets are + /// not required to share the primary model's repository (the DFlash + /// draft checkpoint is pinned from a different one). + /// + private static void ValidateAdditionalAssets(LocalAiInstallManifest manifest) + { + if (manifest.SchemaVersion != LocalAiInstallManifest.AdditionalAssetsSchemaVersion) + { + if (!manifest.AdditionalModelAssets.IsDefaultOrEmpty || !manifest.AdditionalModelPaths.IsDefaultOrEmpty) + { + throw new InvalidDataException( + "Only schema-5 local AI manifests may record additional model assets."); + } + return; + } + + if (manifest.AdditionalModelAssets.IsDefaultOrEmpty || + manifest.AdditionalModelAssetsOrEmpty.Length != manifest.AdditionalModelPathsOrEmpty.Length) + { + throw new InvalidDataException( + "A schema-5 local AI manifest must record a matching additional-asset receipt and cache path pair."); + } + + var seenFileNames = new HashSet(StringComparer.OrdinalIgnoreCase) { manifest.ModelAsset.FileName }; + for (int i = 0; i < manifest.AdditionalModelAssetsOrEmpty.Length; i++) + { + LocalAiAssetReceipt receipt = manifest.AdditionalModelAssetsOrEmpty[i]; + ValidateAssetReceipt(receipt, $"{nameof(manifest.AdditionalModelAssets)}[{i}]"); + if (!seenFileNames.Add(receipt.FileName)) + throw new InvalidDataException("The local AI manifest additional asset filenames must be unique."); + + HuggingFaceModelProvenance provenance = ParseHuggingFaceProvenance(receipt); + if (!HuggingFaceHubCache.TryGetSnapshotPaths( + manifest.ModelCacheRoot!, + provenance.RepositoryId, + provenance.Revision, + provenance.RelativePath, + out string expectedPath, + out _, + out string error) || + !string.Equals(manifest.AdditionalModelPathsOrEmpty[i], expectedPath, StringComparison.OrdinalIgnoreCase)) + { + throw new InvalidDataException( + string.IsNullOrWhiteSpace(error) + ? "The local AI manifest additional asset cache receipt is invalid." + : error); + } + } + } + + private static HuggingFaceModelProvenance ParseHuggingFaceProvenance(LocalAiAssetReceipt receipt) + { + var source = new Uri(receipt.SourceUrl, UriKind.Absolute); + if (!string.Equals(source.Scheme, Uri.UriSchemeHttps, StringComparison.OrdinalIgnoreCase) || + !string.Equals(source.Host, "huggingface.co", StringComparison.OrdinalIgnoreCase) || + source.Query is not ("" or "?download=true")) + { + throw new InvalidDataException("The local AI manifest asset source must be an immutable Hugging Face resolve URL."); + } + + string[] pathSegments = Uri.UnescapeDataString(source.AbsolutePath).Split( + '/', StringSplitOptions.RemoveEmptyEntries); + int resolveIndex = Array.IndexOf(pathSegments, "resolve"); + if (resolveIndex != 2 || pathSegments.Length < resolveIndex + 3) + { + throw new InvalidDataException("The local AI manifest asset source must be an immutable Hugging Face resolve URL."); + } + + string repositoryId = $"{pathSegments[0]}/{pathSegments[1]}"; + string revision = pathSegments[resolveIndex + 1]; + if (revision.Length != 40 || + revision.Any(character => character is not (>= '0' and <= '9' or >= 'a' and <= 'f'))) + { + throw new InvalidDataException( + "The local AI manifest asset revision must be a lowercase 40-character commit digest."); + } + + string[] relativeSegments = pathSegments[(resolveIndex + 2)..]; + string relativePath = string.Join('/', relativeSegments); + if (relativeSegments.Any(segment => !WindowsPathSafety.IsSafeSegment(segment)) || + !string.Equals(relativeSegments[^1], receipt.FileName, StringComparison.Ordinal)) + { + throw new InvalidDataException( + "The local AI manifest asset source must match its own repository, revision, and filename."); + } + + return new HuggingFaceModelProvenance(repositoryId, revision, relativePath); + } + internal sealed record HuggingFaceModelProvenance( string RepositoryId, string Revision, diff --git a/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs b/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs index 29381bf68..92596120d 100644 --- a/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs +++ b/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs @@ -28,6 +28,68 @@ public async Task Load_DoesNotTriggerCacheMigration() Assert.Equal(fixture.Content, await File.ReadAllBytesAsync(fixture.LegacyModelPath)); } + [Fact] + public async Task Save_SchemaFourManifestOmitsAdditionalAssetFieldsFromJson() + { + // A recipe with no additional assets (every recipe before this session, + // and most since) must keep writing the exact schema-4 shape an older + // app build already knows how to read. AdditionalModelAssets/Paths + // default to ImmutableArray's unset (not .Empty) value specifically + // so JsonIgnoreCondition.WhenWritingDefault omits them here, and + // UsesHubCache must never appear at all -- it's a derived read helper, + // not part of the persisted contract. + using var temp = new TempDirectory("local-ai-manifest-schema4-json-"); + var paths = new LocalAiPaths(temp.Combine("app-data")); + string legacyRelativePath = Path.Combine("models", "owner", "repository", Revision, "model.gguf"); + var manifest = new LocalAiInstallManifest + { + SchemaVersion = LocalAiInstallManifest.HubCacheReceiptSchemaVersion, + EngineVersion = "b1", + Architecture = "x64", + RuntimeId = "llama-server-test", + ModelCatalogId = "test-model", + SelectedGpuId = "GPU-TEST", + ExecutablePath = Path.Combine("engines", "llama-server.exe"), + RuntimeAssets = ImmutableArray.Create(new LocalAiAssetReceipt + { + FileName = "runtime.zip", + SourceUrl = "https://example.invalid/runtime.zip", + SizeBytes = 1, + Sha256 = new string('a', 64), + }), + ModelPath = legacyRelativePath, + ModelCacheRoot = temp.Combine("hf-cache"), + CachedModelPath = HuggingFaceHubCache.TryGetSnapshotPaths( + temp.Combine("hf-cache"), + RepositoryId, + Revision, + RelativeModelPath, + out string cachedModelPath, + out _, + out string pathError) + ? cachedModelPath + : throw new InvalidOperationException(pathError), + ModelId = $"{RepositoryId}@{Revision}", + ModelAlias = "test-model", + ModelAsset = new LocalAiAssetReceipt + { + FileName = "model.gguf", + SourceUrl = $"https://huggingface.co/{RepositoryId}/resolve/{Revision}/{RelativeModelPath}?download=true", + SizeBytes = 1, + Sha256 = new string('b', 64), + }, + ContextLength = 4096, + }; + var store = new LocalAiManifestStore(paths, () => temp.Combine("hf-cache")); + await store.SaveAsync(manifest); + + JsonObject persisted = (JsonNode.Parse(await File.ReadAllTextAsync(paths.ManifestPath)) as JsonObject)!; + + Assert.False(persisted.ContainsKey("additionalModelAssets")); + Assert.False(persisted.ContainsKey("additionalModelPaths")); + Assert.False(persisted.ContainsKey("usesHubCache")); + } + [Fact] public async Task Load_CopiesVerifiedLegacyWeightsAndRecordsTransitionalReceipt() { From d41a0fd2fdd4aeca9db6f384064ef838cc009bdd Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Wed, 23 Sep 2026 15:23:08 -0400 Subject: [PATCH 08/13] feat(local-ai): acquire additional model assets via the HF hub cache Adds HuggingFaceModelInstaller.InstallAdditionalAssetAsync, mirroring InstallAsync's resumable-download/verify/promote flow for a recipe's non-primary pinned artifacts (DFlash draft checkpoint, extra split-GGUF shards). Deliberately duplicated rather than refactored out of InstallAsync to avoid any risk to that heavily-tested primary weights path; the one thing it omits is the legacy app-owned compatibility copy, since every recipe using an additional asset is new since the hub cache became the primary store. --- ...gingFaceModelInstaller.AdditionalAssets.cs | 200 ++++++++++++++++++ .../HuggingFaceModelInstaller.cs | 23 +- 2 files changed, 222 insertions(+), 1 deletion(-) create mode 100644 src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.AdditionalAssets.cs diff --git a/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.AdditionalAssets.cs b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.AdditionalAssets.cs new file mode 100644 index 000000000..060e34dde --- /dev/null +++ b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.AdditionalAssets.cs @@ -0,0 +1,200 @@ +using OpenClaw.Connection.LocalAi; +using OpenClaw.Shared.Inference.Catalog; + +namespace OpenClaw.SetupEngine; + +/// +/// Acquisition for additional model assets (a DFlash draft checkpoint) +/// verified into the same standard Hugging Face hub cache uses for the +/// primary weights. Reuses the same resumable download, hash verification, +/// and safe-cache-directory promotion helpers; the one thing it deliberately +/// omits is the legacy app-owned compatibility copy, since every recipe that +/// pins an additional asset is new -- there is no pre-hub-cache install of +/// it to stay compatible with. +/// +internal sealed partial class HuggingFaceModelInstaller +{ + public async Task InstallAdditionalAssetAsync( + string localDataDirectory, + PinnedArtifact artifact, + IProgress? progress, + CancellationToken cancellationToken) + { + ArgumentException.ThrowIfNullOrWhiteSpace(localDataDirectory); + ArgumentNullException.ThrowIfNull(artifact); + if (artifact.Role != ArtifactRole.ModelWeights || artifact.Source is not HuggingFaceRevisionSource source) + { + throw new HuggingFaceModelInstallException( + "An additional Local AI model asset must be an immutable Hugging Face weights artifact."); + } + + string cacheRoot = _cacheRootResolver(); + if (!TryValidateCacheRootOwnershipBoundary(localDataDirectory, cacheRoot, out string cacheRootError)) + throw new HuggingFaceModelInstallException(cacheRootError); + if (!HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, + source.RepositoryId, + source.RevisionSha, + artifact.RelativePath, + out string modelPath, + out string partialPath, + out string pathError)) + { + throw new HuggingFaceModelInstallException(pathError); + } + + if (Directory.Exists(modelPath)) + throw new HuggingFaceModelInstallException("The managed Local AI model path is an existing directory."); + if (Directory.Exists(partialPath)) + throw new HuggingFaceModelInstallException("The managed Local AI partial model path is an existing directory."); + + await using (FileStream? verified = + await HuggingFaceHubCache.TryOpenVerifiedCacheFileAsync( + cacheRoot, + modelPath, + artifact.SizeBytes, + artifact.Sha256, + new VerificationProgress(this, progress, artifact.SizeBytes), + cancellationToken) + .ConfigureAwait(false)) + { + if (verified is not null) + { + return new HuggingFaceAdditionalAssetInstallResult( + modelPath, cacheRoot, HuggingFaceModelInstallDisposition.ReusedVerified, CreatedThisRun: false); + } + } + + if (PathEntryExists(modelPath)) + { + throw new HuggingFaceModelInstallException( + $"The Hugging Face cache destination '{modelPath}' is unsafe or does not match " + + "the pinned model. Remove it manually and retry setup."); + } + + string destinationDirectory = Path.GetDirectoryName(modelPath) + ?? throw new HuggingFaceModelInstallException( + "The Hugging Face cache destination has no parent directory."); + string destinationFileName = Path.GetFileName(modelPath); + string partialFileName = Path.GetFileName(partialPath); + LocalAiManifestMigration.SafeCacheDirectory? directory = null; + LocalAiManifestMigration.CacheMigrationFile? partial = null; + bool partialExistedBeforeInstall = false; + HuggingFaceModelInstallDisposition disposition = HuggingFaceModelInstallDisposition.Downloaded; + try + { + directory = LocalAiManifestMigration.SafeCacheDirectory.OpenOrCreate(cacheRoot, destinationDirectory); + partial = directory.TryOpenExisting(partialFileName); + partialExistedBeforeInstall = partial is not null; + + bool partialVerified = partial is not null && + partial.Stream.Length == artifact.SizeBytes && + await VerifyOpenFileAsync(partial.Stream, artifact, progress, cancellationToken).ConfigureAwait(false); + if (partial is not null && partial.Stream.Length >= artifact.SizeBytes && !partialVerified) + { + partial.Stream.SetLength(0); + partial.Stream.Position = 0; + } + + if (!partialVerified) + { + partial ??= directory.CreateNew(partialFileName); + if (partial.Stream.Length == 0 && + await TryCopyVerifiedBlobAsync( + cacheRoot, source.RepositoryId, artifact, partial.Stream, progress, cancellationToken) + .ConfigureAwait(false)) + { + disposition = HuggingFaceModelInstallDisposition.ReusedVerified; + } + else + { + await DownloadAndVerifyAsync(artifact, partial.Stream, progress, cancellationToken) + .ConfigureAwait(false); + } + } + + cancellationToken.ThrowIfCancellationRequested(); + LocalAiManifestMigration.CacheMigrationFile activePartial = partial + ?? throw new HuggingFaceModelInstallException("The Hugging Face cache partial was not created."); + if (!HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, + source.RepositoryId, + source.RevisionSha, + artifact.RelativePath, + out string revalidatedModelPath, + out string revalidatedPartialPath, + out pathError) || + !string.Equals(modelPath, revalidatedModelPath, StringComparison.OrdinalIgnoreCase) || + !string.Equals(partialPath, revalidatedPartialPath, StringComparison.OrdinalIgnoreCase)) + { + throw new HuggingFaceModelInstallException( + string.IsNullOrWhiteSpace(pathError) + ? "The Local AI model paths changed before promotion." + : pathError); + } + + if (!await VerifyOpenFileAsync(activePartial.Stream, artifact, progress, cancellationToken) + .ConfigureAwait(false)) + { + throw new HuggingFaceModelInstallException( + "The Hugging Face cache partial does not match the pinned model."); + } + + try + { + activePartial.Promote(destinationFileName); + } + catch (IOException ex) + { + if (partialExistedBeforeInstall) + activePartial.Commit(); + activePartial.Dispose(); + partial = null; + await using FileStream? winner = + await HuggingFaceHubCache.TryOpenVerifiedCacheFileAsync( + cacheRoot, + modelPath, + artifact.SizeBytes, + artifact.Sha256, + new VerificationProgress(this, progress, artifact.SizeBytes), + cancellationToken) + .ConfigureAwait(false); + if (winner is null) + { + throw new HuggingFaceModelInstallException( + "The Hugging Face cache destination changed before promotion.", ex); + } + + return new HuggingFaceAdditionalAssetInstallResult( + modelPath, cacheRoot, HuggingFaceModelInstallDisposition.ReusedVerified, CreatedThisRun: false); + } + + if (partialExistedBeforeInstall) + activePartial.Commit(); + directory.RequirePromotedFile(activePartial.Stream.SafeFileHandle, destinationFileName); + activePartial.Commit(); + return new HuggingFaceAdditionalAssetInstallResult(modelPath, cacheRoot, disposition, CreatedThisRun: true); + } + catch (OperationCanceledException) + { + partial?.Commit(); + throw; + } + catch (Exception exception) when ( + exception is IOException or HttpRequestException or TransientHuggingFaceModelInstallException) + { + partial?.Commit(); + throw; + } + catch (HuggingFaceModelInstallException) when (partialExistedBeforeInstall) + { + partial?.Commit(); + throw; + } + finally + { + partial?.Dispose(); + directory?.Dispose(); + } + } +} diff --git a/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs index 1e5d53fee..080e8ed22 100644 --- a/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs +++ b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs @@ -57,6 +57,13 @@ public TransientHuggingFaceModelInstallException(string message) } } +/// A verified additional model asset (DFlash draft checkpoint). +internal sealed record HuggingFaceAdditionalAssetInstallResult( + string ModelPath, + string CacheRoot, + HuggingFaceModelInstallDisposition Disposition, + bool CreatedThisRun); + internal interface IHuggingFaceModelAcquirer { Task InstallAsync( @@ -66,6 +73,20 @@ Task InstallAsync( IProgress? progress, CancellationToken cancellationToken); + /// + /// Verifies or acquires one additional model asset (a DFlash draft + /// checkpoint) into the same hub cache + /// uses for the primary weights. Unlike the + /// primary weights, additional assets have no legacy app-owned + /// compatibility copy -- every recipe that uses one is new since the hub + /// cache became the primary store. + /// + Task InstallAdditionalAssetAsync( + string localDataDirectory, + PinnedArtifact artifact, + IProgress? progress, + CancellationToken cancellationToken); + void RemoveInstalledModel(string localDataDirectory, HuggingFaceModelInstallResult install); void RemovePartialModel( @@ -80,7 +101,7 @@ void RemovePartialModel( /// left by process termination is resumed with an HTTP range request. Shared /// cache artifacts and resumable partials survive rollback and cancellation. /// -internal sealed class HuggingFaceModelInstaller : IHuggingFaceModelAcquirer +internal sealed partial class HuggingFaceModelInstaller : IHuggingFaceModelAcquirer { private const int BufferSize = 1024 * 1024; private const int ProgressIntervalBytes = 4 * 1024 * 1024; From 67d7f39c29ed84b9020814d4182a26193a34399d Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Wed, 23 Sep 2026 15:23:13 -0400 Subject: [PATCH 09/13] feat(local-ai): wire additional-asset acquisition into setup and reconcile AcquireLocalAiModelStep now downloads/verifies a recipe's additional assets (via LocalModelCatalog.AdditionalArtifacts) right after its primary weights, records the results on SetupContext, and rolls the context field back on failure -- the hub-cache artifacts themselves survive rollback the same way the primary weights' do, since there is no legacy copy to delete. PersistLocalAiManifestStep writes them into a schema-5 manifest (schema 4 unchanged for recipes with none), and leaves the two new manifest fields at their unset default rather than an explicitly-built empty array when there is nothing to add, so a plain schema-4 install keeps omitting them from JSON. LocalAiInstallReconciler gains VerifyAdditionalAssetAsync so a reused install re-verifies a recipe's additional assets against the hub cache, not just its primary weights, before treating the install as still valid. LocalAiReconcileResult now also carries the additional asset installs it just verified, reconstructed from the manifest's own already-verified receipts: recovery for a broken runtime (model and additional assets still valid) previously left SetupContext's additional-install list empty because AcquireLocalAiModelStep's reuse skip never re-runs acquisition, which made PersistLocalAiManifestStep hard-fail on the very installs this series adds an acquisition step for. The setup review consent screen now also lists every additional artifact (a DFlash draft checkpoint, or extra split-GGUF shards) as its own download line instead of only the primary weights. --- .../LocalAiInstallReconciler.cs | 4 +- src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs | 73 +++++++- src/OpenClaw.SetupEngine/SetupContext.cs | 8 + .../SetupReviewSummary.cs | 31 +++- .../LocalAiInstallRecoveryTests.cs | 156 ++++++++++++++++++ 5 files changed, 262 insertions(+), 10 deletions(-) diff --git a/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs b/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs index fa2ec8b47..051d61b71 100644 --- a/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs +++ b/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs @@ -30,8 +30,8 @@ Task VerifyLegacyCompatibilityAsync( CancellationToken cancellationToken); /// - /// Verifies one schema-5 additional model asset (a DFlash draft checkpoint, - /// or an extra split-GGUF shard) still matches its pinned receipt in the + /// Verifies one schema-5 additional model asset (a DFlash draft checkpoint) + /// still matches its pinned receipt in the /// shared hub cache. Additional assets have no legacy app-owned copy, so /// unlike there is no separate schema-3 path. /// diff --git a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs index 07d53b5d0..56135bea2 100644 --- a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs +++ b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs @@ -294,6 +294,8 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati ctx.LocalAiRecoveryOriginalInstall ??= retainedReceipt; ctx.LocalAiRuntimeInstall = result.RuntimeInstall; ctx.LocalAiModelInstall = result.ModelInstall; + ctx.LocalAiAdditionalModelInstalls = result.AdditionalModelInstalls + ?? ImmutableArray.Empty; return StepResult.Skip(result.OriginalInstall is null ? "No completed managed Local AI installation was found." : "The existing Local AI receipt was retained while incomplete assets are repaired."); @@ -302,6 +304,8 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati ctx.LocalAiResolvedInstall = result.ResolvedInstall; ctx.LocalAiRuntimeInstall = result.RuntimeInstall; ctx.LocalAiModelInstall = result.ModelInstall; + ctx.LocalAiAdditionalModelInstalls = result.AdditionalModelInstalls + ?? ImmutableArray.Empty; ctx.LocalAiPort = result.ResolvedInstall!.Manifest.RequestedPort; return StepResult.Ok("Reused the verified managed Local AI installation."); } @@ -442,6 +446,16 @@ public AcquireLocalAiModelStep() internal AcquireLocalAiModelStep(IHuggingFaceModelAcquirer acquirer) => _acquirer = acquirer ?? throw new ArgumentNullException(nameof(acquirer)); + /// + /// Additional artifacts a recipe needs beyond its primary weights, in the + /// fixed catalog order + /// defines. relies on that same + /// ordering to tell the draft checkpoint apart from a shard without a + /// separate "kind" tag on the receipt. + /// + internal static ImmutableArray AdditionalArtifacts(LocalModelInfo model) => + LocalModelCatalog.AdditionalArtifacts(model); + public override string Id => "acquire-local-ai-model"; public override string DisplayName => "Downloading Local AI model from Hugging Face"; public override bool CanRetry => false; @@ -481,6 +495,29 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati progress, linked.Token); ctx.LocalAiModelInstall = install; + + ImmutableArray additionalArtifacts = AdditionalArtifacts(plan.Model); + var additionalInstalls = ImmutableArray.CreateBuilder( + additionalArtifacts.Length); + foreach (PinnedArtifact artifact in additionalArtifacts) + { + var artifactProgress = new SynchronousProgress(value => + ctx.DetailProgress?.Report(new SetupDetailProgressEvent( + Id, + value.Phase == HuggingFaceModelInstallPhase.Verifying + ? $"Verifying {artifact.RelativePath}" + : $"Downloading {artifact.RelativePath}", + value.CompletedBytes, + value.TotalBytes, + SetupDetailProgressUnit.Bytes))); + additionalInstalls.Add(await _acquirer.InstallAdditionalAssetAsync( + ctx.LocalDataDir, + artifact, + artifactProgress, + linked.Token)); + } + ctx.LocalAiAdditionalModelInstalls = additionalInstalls.MoveToImmutable(); + string action = install.Disposition == HuggingFaceModelInstallDisposition.ReusedVerified ? "Verified existing" : "Downloaded"; @@ -513,6 +550,9 @@ public override Task RollbackAsync(SetupContext ctx, CancellationToken ct) _acquirer.RemoveInstalledModel(ctx.LocalDataDir, install); ctx.LocalAiModelInstall = null; } + // Additional assets have no legacy app-owned copy to remove; their hub-cache + // artifacts survive rollback the same way the primary weights' do. + ctx.LocalAiAdditionalModelInstalls = ImmutableArray.Empty; if (ctx.LocalAiEligibility?.Plan is { } plan) { _acquirer.RemovePartialModel( @@ -603,9 +643,33 @@ ctx.LocalAiRecoveryOriginalInstall is not null && return StepResult.Terminal(ex.Message, ex); } + ImmutableArray additionalArtifacts = AcquireLocalAiModelStep.AdditionalArtifacts(plan.Model); + if (additionalArtifacts.Length != ctx.LocalAiAdditionalModelInstalls.Length) + { + return StepResult.Terminal( + "The Local AI installation receipt requires a completed additional-asset acquisition step."); + } + + var additionalModelAssets = ImmutableArray.CreateBuilder(additionalArtifacts.Length); + var additionalModelPaths = ImmutableArray.CreateBuilder(additionalArtifacts.Length); + for (int i = 0; i < additionalArtifacts.Length; i++) + { + PinnedArtifact artifact = additionalArtifacts[i]; + additionalModelAssets.Add(new LocalAiAssetReceipt + { + FileName = Path.GetFileName(artifact.RelativePath), + SourceUrl = artifact.DownloadUri.AbsoluteUri, + SizeBytes = artifact.SizeBytes, + Sha256 = artifact.Sha256.Value, + }); + additionalModelPaths.Add(ctx.LocalAiAdditionalModelInstalls[i].ModelPath); + } + LocalAiInstallManifest manifest = new() { - SchemaVersion = LocalAiInstallManifest.HubCacheReceiptSchemaVersion, + SchemaVersion = additionalArtifacts.IsEmpty + ? LocalAiInstallManifest.HubCacheReceiptSchemaVersion + : LocalAiInstallManifest.AdditionalAssetsSchemaVersion, EngineVersion = LlamaRuntimeCatalog.ReleaseTag, Architecture = plan.Runtime.Architecture switch { @@ -630,6 +694,11 @@ ctx.LocalAiRecoveryOriginalInstall is not null && SizeBytes = plan.Model.Weights.SizeBytes, Sha256 = plan.Model.Weights.Sha256.Value, }, + // Leave these at their unset default (not an explicitly-built empty + // array) when there is nothing to add, so schema-4 manifests omit + // them from JSON entirely -- see the properties' remarks. + AdditionalModelAssets = additionalArtifacts.IsEmpty ? default : additionalModelAssets.MoveToImmutable(), + AdditionalModelPaths = additionalArtifacts.IsEmpty ? default : additionalModelPaths.MoveToImmutable(), RequestedPort = requestedPort, Endpoint = null, ContextLength = plan.Profile.ContextTokens, @@ -656,6 +725,8 @@ ctx.LocalAiRecoveryOriginalInstall is not null && ModelId = manifest.ModelId, ModelAlias = manifest.ModelAlias, ModelAsset = manifest.ModelAsset, + AdditionalModelAssets = manifest.AdditionalModelAssets, + AdditionalModelPaths = manifest.AdditionalModelPaths, RequestedPort = manifest.RequestedPort, Endpoint = null, ContextLength = manifest.ContextLength, diff --git a/src/OpenClaw.SetupEngine/SetupContext.cs b/src/OpenClaw.SetupEngine/SetupContext.cs index aad8f82dd..5cd0d7a94 100644 --- a/src/OpenClaw.SetupEngine/SetupContext.cs +++ b/src/OpenClaw.SetupEngine/SetupContext.cs @@ -1,3 +1,4 @@ +using System.Collections.Immutable; using System.Diagnostics.CodeAnalysis; using System.Text.Json; using System.Text.Json.Serialization; @@ -515,6 +516,13 @@ public Func>? public int? LocalAiPort { get; set; } internal LlamaRuntimeInstallResult? LocalAiRuntimeInstall { get; set; } internal HuggingFaceModelInstallResult? LocalAiModelInstall { get; set; } + /// + /// Verified additional model assets (a DFlash draft checkpoint and/or + /// in catalog order. Empty for every recipe + /// that has neither. + /// + internal ImmutableArray LocalAiAdditionalModelInstalls { get; set; } = + ImmutableArray.Empty; internal LocalAiResolvedInstall? LocalAiResolvedInstall { get; set; } internal LocalAiResolvedInstall? LocalAiRecoveryOriginalInstall { get; set; } internal bool LocalAiRecoveryProviderTransition { get; set; } diff --git a/src/OpenClaw.SetupEngine/SetupReviewSummary.cs b/src/OpenClaw.SetupEngine/SetupReviewSummary.cs index 0b9f1f696..10b31de8b 100644 --- a/src/OpenClaw.SetupEngine/SetupReviewSummary.cs +++ b/src/OpenClaw.SetupEngine/SetupReviewSummary.cs @@ -77,16 +77,33 @@ public static SetupReviewSummary Build(SetupConfig config, string? dataDir = nul LocalModelCatalog.Find(config.LocalAi.SelectedModelId) ?? LocalModelCatalog.Default; LocalInferenceRunProfile? localAiProfile = LocalModelCatalog.FindProfile(localAiModel, config.LocalAi.SelectedProfileId); - string[] localAiCommands = config.LocalAi.Enabled - ? - [ + string[] localAiCommands; + if (config.LocalAi.Enabled) + { + var commands = new List + { "download verified llama-server + CUDA runtime for Windows", $"download {localAiModel.Weights.RelativePath} from Hugging Face revision " + ((HuggingFaceRevisionSource)localAiModel.Weights.Source).RevisionSha, - $"llama-server router on dynamic 127.0.0.1 port; model loads on first request", - $"openclaw provider llamacpp -> /v1; primary llamacpp/{localAiModel.Id}", - ] - : []; + }; + // Additional pinned artifacts (a DFlash draft checkpoint) are + // separate downloads the user is consenting + // to alongside the primary weights -- list each one explicitly + // rather than letting the consent screen understate what's fetched. + foreach (PinnedArtifact artifact in LocalModelCatalog.AdditionalArtifacts(localAiModel)) + { + commands.Add( + $"download {artifact.RelativePath} from Hugging Face revision " + + ((HuggingFaceRevisionSource)artifact.Source).RevisionSha); + } + commands.Add("llama-server router on dynamic 127.0.0.1 port; model loads on first request"); + commands.Add($"openclaw provider llamacpp -> /v1; primary llamacpp/{localAiModel.Id}"); + localAiCommands = commands.ToArray(); + } + else + { + localAiCommands = []; + } var summary = new SetupReviewSummary( DistroTitle: $"Install {baseDistro.Replace('-', ' ')} in WSL", diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs index cdc53f34d..9773e560a 100644 --- a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs +++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs @@ -967,6 +967,11 @@ public void ArchiveDestination_ResolvesValidNestedEntry() public async Task Reconciler_ReusesOnlyMatchingManifestWithoutMutation() { using var temp = new TempDirectory(); + // Pin the hub cache to an empty directory. This asserts that a matching receipt is + // reused untouched; with the ambient user cache it would instead depend on whether + // that cache happens to already hold the default model, which legitimately triggers + // the schema-3 to schema-4 migration and rewrites the receipt. + using var environment = new EnvironmentScope("HF_HUB_CACHE", CacheRoot(temp.Path)); LocalInferencePlan plan = CatalogPlan(); const string gpuId = "GPU-0"; LocalAiInstallManifest manifest = CreateManifest(temp.Path, plan, gpuId); @@ -1145,6 +1150,103 @@ public async Task Reconciler_RecoveryRepairsMissingSchemaFourCompatibilityCopy() Assert.Null(result.ModelInstall); } + [Fact] + public async Task Reconciler_RecoveryWithValidModelStillPopulatesAdditionalAssetInstalls() + { + // Regression: recovery for a broken runtime (model + additional assets + // still verified valid) must give the caller everything it needs to + // persist a schema-5 manifest without re-downloading the already- + // verified draft checkpoint -- AcquireLocalAiModelStep's "reuse the + // verified model" skip only re-populates SetupContext from the + // reconcile result, it never re-runs acquisition itself. + using var temp = new TempDirectory(); + byte[] primaryBytes = "verified-dflash-primary"u8.ToArray(); + byte[] draftBytes = "verified-dflash-draft"u8.ToArray(); + var draftSource = new HuggingFaceRevisionSource("owner/draft-repo", new string('c', 40)); + var draftWeights = new PinnedArtifact( + "test-model-dflash-draft", + ArtifactRole.ModelWeights, + draftSource, + "draft.gguf", + draftBytes.Length, + new Sha256Digest(Sha256(draftBytes))); + LocalModelInfo model = CreateModelWithDraft(primaryBytes, draftWeights); + LlamaRuntimeVariant runtime = CreateRuntime( + CreateZip(("llama-server.exe", "server"u8.ToArray())), + CreateZip(("dependency.dll", "dependency"u8.ToArray()))); + var plan = new LocalInferencePlan( + runtime, + model, + new LocalInferenceRunProfile( + "test-profile", + 128, + KvCachePrecision.F16, + KvCachePrecision.F16, + KvCachePrecision.F16, + KvCachePrecision.F16, + runtimeWorkspaceBytes: 1), + LocalInferenceModelSelectionOrigin.Default); + var paths = new LocalAiPaths(temp.Path); + string cacheRoot = CacheRoot(temp.Path); + var primarySource = Assert.IsType(model.Weights.Source); + Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, + primarySource.RepositoryId, + primarySource.RevisionSha, + model.Weights.RelativePath, + out string cachedModelPath, + out _, + out string error), error); + Directory.CreateDirectory(Path.GetDirectoryName(cachedModelPath)!); + await File.WriteAllBytesAsync(cachedModelPath, primaryBytes); + Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, + draftSource.RepositoryId, + draftSource.RevisionSha, + draftWeights.RelativePath, + out string cachedDraftPath, + out _, + out error), error); + Directory.CreateDirectory(Path.GetDirectoryName(cachedDraftPath)!); + await File.WriteAllBytesAsync(cachedDraftPath, draftBytes); + + LocalAiInstallManifest manifest = CreateManifest(temp.Path, plan, "GPU-0") with + { + SchemaVersion = LocalAiInstallManifest.AdditionalAssetsSchemaVersion, + ModelCacheRoot = cacheRoot, + CachedModelPath = cachedModelPath, + AdditionalModelAssets = ImmutableArray.Create(new LocalAiAssetReceipt + { + FileName = "draft.gguf", + SourceUrl = draftWeights.DownloadUri.AbsoluteUri, + SizeBytes = draftWeights.SizeBytes, + Sha256 = draftWeights.Sha256.Value, + }), + AdditionalModelPaths = ImmutableArray.Create(cachedDraftPath), + }; + await new LocalAiManifestStore(paths, () => cacheRoot).SaveAsync(manifest); + + LocalAiReconcileResult result = await new LocalAiInstallReconciler( + new InvalidRuntimeInspector(), + new AcceptingModelVerifier(), + () => cacheRoot) + .ReconcileAsync( + temp.Path, + plan, + "GPU-0", + CancellationToken.None, + allowIncompleteInstallation: true); + + Assert.False(result.Reused); + Assert.Null(result.RuntimeInstall); + Assert.NotNull(result.ModelInstall); + ImmutableArray additionalInstalls = + result.AdditionalModelInstalls ?? ImmutableArray.Empty; + HuggingFaceAdditionalAssetInstallResult draftInstall = Assert.Single(additionalInstalls); + Assert.Equal(cachedDraftPath, draftInstall.ModelPath); + Assert.False(draftInstall.CreatedThisRun); + } + [Fact] public async Task Reconciler_UpgradesRetiredRuntimeReceiptInsteadOfFailingSetup() { @@ -1560,6 +1662,40 @@ private static LocalModelInfo CreateModel(byte[] bytes) SupportsVision: false); } + private static LocalModelInfo CreateModelWithDraft(byte[] primaryBytes, PinnedArtifact draftWeights) + { + var source = new HuggingFaceRevisionSource("owner/repo", new string('a', 40)); + var artifact = new PinnedArtifact( + "test-model-dflash", + ArtifactRole.ModelWeights, + source, + "model.gguf", + primaryBytes.Length, + new Sha256Digest(Sha256(primaryBytes))); + return new LocalModelInfo( + "test-model-dflash", + "Test model (DFlash)", + "Test", + "Q4", + artifact, + new LocalModelRunRecipe( + 128, + 128, + 1, + 1, + 1, + 128, + true, + true, + SpeculativeDecodingMode.DraftDFlash, + 1, + new ModelSamplingPreset(0.6, 20, 0.95, 0, 1, 0), + draftWeights), + IsDefault: true, + IsExplicitAlternative: false, + SupportsVision: false); + } + private static LlamaRuntimeVariant CreateRuntime(byte[] binaryZip, byte[] dependencyZip) { var source = new GitHubReleaseSource("owner/repo", "v1", new string('b', 40)); @@ -1780,6 +1916,14 @@ public Task InspectAsync( Task.FromResult(new LlamaRuntimeInspection(true, "valid", null)); } + private sealed class InvalidRuntimeInspector : ILlamaRuntimeInspector + { + public Task InspectAsync( + string installDirectory, + CancellationToken cancellationToken) => + Task.FromResult(new LlamaRuntimeInspection(false, "invalid", "simulated corrupted runtime")); + } + private sealed class AcceptingModelVerifier : ILocalAiModelFileVerifier { public Task VerifyActiveAsync( @@ -1792,6 +1936,12 @@ public Task VerifyLegacyCompatibilityAsync( LocalAiPaths paths, PinnedArtifact artifact, CancellationToken cancellationToken) => Task.FromResult(true); + + public Task VerifyAdditionalAssetAsync( + LocalAiResolvedInstall install, + string cachedAssetPath, + PinnedArtifact artifact, + CancellationToken cancellationToken) => Task.FromResult(true); } private sealed class RejectingModelVerifier : ILocalAiModelFileVerifier @@ -1806,6 +1956,12 @@ public Task VerifyLegacyCompatibilityAsync( LocalAiPaths paths, PinnedArtifact artifact, CancellationToken cancellationToken) => Task.FromResult(false); + + public Task VerifyAdditionalAssetAsync( + LocalAiResolvedInstall install, + string cachedAssetPath, + PinnedArtifact artifact, + CancellationToken cancellationToken) => Task.FromResult(false); } private sealed class CompleteModelRepairStep(string localDataDirectory, LocalInferencePlan plan) : SetupStep From 755a9ba8e05bc7f380a0e14defa05aa2453cf61b Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Thu, 24 Sep 2026 10:48:10 -0400 Subject: [PATCH 10/13] feat(local-ai): thread the verified DFlash draft identity into the launch preset BuildCore resolves the DFlash draft checkpoint and passes it to BuildPreset, so spec-draft-model is finally emitted for real. ValidateArtifactReceipts also checks additional-asset receipts against the catalog, as it already does for the primary weights and runtime artifacts. The path handed to llama-server is the handle-resolved physical path from the same verification that opened the file, not the persisted snapshot path. The hub cache hands out a handle-resolved path precisely so a snapshot-link replacement cannot change the file identity a native reader finally opens, which is how the primary model is already bound; the draft checkpoint now gets the same guarantee instead of re-deriving its path from its receipt. LlamaServerRuntimeService's schema checks use LocalAiInstallManifest.UsesHubCache so schema-5 installs are treated as hub-cache-backed, matching schema 4, instead of falling into the legacy schema-3 branch. --- .../LocalAi/LlamaServerRouterConfiguration.cs | 62 ++++++++++++-- .../LocalAi/LlamaServerRuntimeService.cs | 47 ++++++++++- .../LocalAiPortLifecycleTests.cs | 84 +++++++++++++++++++ 3 files changed, 183 insertions(+), 10 deletions(-) diff --git a/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs b/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs index 03fead054..33cbb2305 100644 --- a/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs +++ b/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs @@ -31,19 +31,27 @@ public static LlamaServerRouterLaunchPlan Build( LocalAiPaths paths, LocalAiResolvedInstall install, int? listenPort = null) => - BuildCore(paths, install, install.ModelPath, listenPort); + BuildCore(paths, install, install.ModelPath, verifiedDraftModelPath: null, listenPort); + /// + /// The draft checkpoint's handle-resolved physical path, from the same verification + /// that opened it. Passing the persisted snapshot path instead would let a + /// snapshot-link replacement change the file llama-server finally opens, which is + /// exactly what resolving the primary model through its own handle prevents. + /// internal static LlamaServerRouterLaunchPlan BuildForVerifiedRuntime( LocalAiPaths paths, LocalAiResolvedInstall install, string verifiedModelPath, + string? verifiedDraftModelPath, int? listenPort = null) => - BuildCore(paths, install, verifiedModelPath, listenPort); + BuildCore(paths, install, verifiedModelPath, verifiedDraftModelPath, listenPort); private static LlamaServerRouterLaunchPlan BuildCore( LocalAiPaths paths, LocalAiResolvedInstall install, string modelPath, + string? verifiedDraftModelPath, int? listenPort) { ArgumentNullException.ThrowIfNull(paths); @@ -62,6 +70,7 @@ private static LlamaServerRouterLaunchPlan BuildCore( ?? throw new InvalidDataException("The managed local AI model is no longer qualified."); LocalInferenceRunProfile profile = ResolveQualifiedReceipt(manifest, runtime, model); + string? draftModelPath = ResolveDraftModelPath(manifest, model, verifiedDraftModelPath); string presetPath = paths.ResolveContainedPath( Path.GetRelativePath(paths.RootDirectory, paths.RouterPresetPath), @@ -86,11 +95,7 @@ private static LlamaServerRouterLaunchPlan BuildCore( .WithComparers(StringComparer.OrdinalIgnoreCase) .Add("CUDA_VISIBLE_DEVICES", manifest.SelectedGpuId), presetPath, - // TODO(rtx-spark-dflash): DraftDFlash recipes need their pinned - // draft checkpoint acquired and verified alongside the primary - // weights before a real path can be threaded through here; see - // BuildPreset's draftModelPath parameter. - BuildPreset(model, profile, modelPath, draftModelPath: null), + BuildPreset(model, profile, modelPath, draftModelPath), model.Id); } @@ -151,6 +156,49 @@ internal static void ValidateArtifactReceipts( { throw new InvalidDataException("The managed model artifact receipt does not match the qualified catalog."); } + + ImmutableArray expectedAdditionalArtifacts = LocalModelCatalog.AdditionalArtifacts(model); + if (manifest.AdditionalModelAssetsOrEmpty.Length != expectedAdditionalArtifacts.Length || + manifest.AdditionalModelPathsOrEmpty.Length != expectedAdditionalArtifacts.Length) + { + throw new InvalidDataException( + "The managed additional model asset receipts do not match the qualified catalog."); + } + for (int i = 0; i < expectedAdditionalArtifacts.Length; i++) + { + PinnedArtifact artifact = expectedAdditionalArtifacts[i]; + LocalAiAssetReceipt receipt = manifest.AdditionalModelAssetsOrEmpty[i]; + if (!string.Equals(receipt.FileName, Path.GetFileName(artifact.RelativePath), StringComparison.Ordinal) || + receipt.SizeBytes != artifact.SizeBytes || + !string.Equals(receipt.Sha256, artifact.Sha256.Value, StringComparison.Ordinal) || + !string.Equals(receipt.SourceUrl, artifact.DownloadUri.AbsoluteUri, StringComparison.Ordinal)) + { + throw new InvalidDataException( + "The managed additional model asset receipts do not match the qualified catalog."); + } + } + } + + /// + /// The DFlash draft checkpoint's path for the preset. Prefers the handle-resolved + /// physical path supplied by the caller that verified and still holds the file, so a + /// snapshot-link replacement cannot change the identity llama-server opens. Falls + /// back to the persisted receipt path only for callers that do not verify first + /// (, used for inspection rather than launch). Null for recipes + /// with no separate draft checkpoint. Callers must validate the manifest via + /// first, which guarantees + /// AdditionalModelPaths has one entry per catalog-pinned artifact. + /// + private static string? ResolveDraftModelPath( + LocalAiInstallManifest manifest, + LocalModelInfo model, + string? verifiedDraftModelPath) + { + if (model.Recipe.DraftWeights is null) + return null; + return string.IsNullOrWhiteSpace(verifiedDraftModelPath) + ? manifest.AdditionalModelPathsOrEmpty[^1] + : verifiedDraftModelPath; } private static string BuildPreset( diff --git a/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs b/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs index ff5a0bad8..3828b4cc6 100644 --- a/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs +++ b/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs @@ -118,6 +118,7 @@ public sealed class LlamaServerRuntimeService : ILocalAiRuntime private LocalAiRuntimeSnapshot _snapshot; private ILocalAiManagedProcess? _managedProcess; private LocalAiVerifiedModelLease? _verifiedModel; + private readonly List _verifiedAdditionalAssets = []; private string? _runtimeModelPath; private LocalAiResolvedInstall? _install; private long _generation; @@ -378,6 +379,7 @@ private async Task EnsureStartedCoreAsync(CancellationTo _options.Paths, install, GetRuntimeModelPath(install), + GetRuntimeDraftModelPath(), requestedPort); await WritePresetAtomicallyAsync(launchPlan, cancellationToken).ConfigureAwait(false); } @@ -862,7 +864,7 @@ private async Task ValidateInstalledFilesAsync( DisposeVerifiedModelHandle(); ValidateInstalledFilesForStatus(install); - if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (!install.Manifest.UsesHubCache) { _runtimeModelPath = install.ModelPath; return; @@ -883,6 +885,33 @@ await _modelFileVerifier.TryOpenAsync( "The shared Hugging Face cache model is unsafe or no longer matches its receipt."); } + // Schema-5 extra assets (a DFlash draft checkpoint) are loaded natively + // by llama-server exactly like the primary weights, and they + // live in the same shared, user-writable hub cache. Rehash them here and hold + // the handles for the process lifetime, so a file swapped after setup cannot + // reach the loader with only a structural path check behind it. + foreach ((LocalAiAssetReceipt receipt, string cachedPath) in + install.Manifest.AdditionalModelAssetsOrEmpty + .Zip(install.Manifest.AdditionalModelPathsOrEmpty)) + { + LocalAiVerifiedModelLease? verifiedAsset = + await _modelFileVerifier.TryOpenAsync( + install.Manifest.ModelCacheRoot!, + cachedPath, + receipt.SizeBytes, + new Sha256Digest(receipt.Sha256), + cancellationToken) + .ConfigureAwait(false); + if (verifiedAsset is null) + { + DisposeVerifiedModelHandle(); + throw new InvalidDataException( + $"The shared Hugging Face cache asset '{receipt.FileName}' is unsafe or no longer matches its receipt."); + } + + _verifiedAdditionalAssets.Add(verifiedAsset); + } + _runtimeModelPath = _verifiedModel.ResolvedPath; } @@ -890,7 +919,7 @@ private static void ValidateInstalledFilesForStatus(LocalAiResolvedInstall insta { if (!File.Exists(install.ExecutablePath)) throw new InvalidDataException("The managed llama-server executable is missing."); - if (install.Manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (install.Manifest.UsesHubCache) { if (!File.Exists(install.ModelPath)) throw new InvalidDataException("The managed GGUF model is missing."); @@ -1666,14 +1695,26 @@ private void DisposeVerifiedModelHandle() { _verifiedModel?.Dispose(); _verifiedModel = null; + foreach (LocalAiVerifiedModelLease lease in _verifiedAdditionalAssets) + lease.Dispose(); + _verifiedAdditionalAssets.Clear(); _runtimeModelPath = null; } + /// + /// The handle-resolved path of the draft checkpoint this process verified and still + /// holds open, or null when the recipe has no additional assets. Additional assets are + /// verified in catalog order and the draft checkpoint is always last, matching + /// . + /// + private string? GetRuntimeDraftModelPath() => + _verifiedAdditionalAssets.Count == 0 ? null : _verifiedAdditionalAssets[^1].ResolvedPath; + private string GetRuntimeModelPath(LocalAiResolvedInstall install) { if (_runtimeModelPath is not null) return _runtimeModelPath; - if (install.Manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (install.Manifest.UsesHubCache) throw new InvalidOperationException("The verified shared-cache model identity is unavailable."); return install.ModelPath; } diff --git a/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs b/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs index 426494b59..97f86b893 100644 --- a/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs +++ b/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs @@ -2583,6 +2583,90 @@ public async Task Router_LaunchesRetiredRuntimeInstallAfterVersionBump() Assert.Equal("qwen3.6-35b-a3b-mtp-q4-k-m", launch.ModelAlias); } + /// + /// Schema-5 extra assets are loaded natively by llama-server from the shared, + /// user-writable hub cache. One that no longer matches its pinned digest must + /// stop startup, exactly like a tampered primary model does. + /// + [Fact] + public async Task Startup_FailsWhenAnAdditionalModelAssetNoLongerMatchesItsReceipt() + { + using var temp = new TempDirectory("local-ai-tampered-asset-"); + LocalAiPaths paths = await PrepareInstallAsync(temp); + var store = new LocalAiManifestStore(paths); + LocalAiResolvedInstall installed = (await store.LoadAsync())!; + string cacheRoot = temp.Combine("hf-cache"); + const string draftRepo = "z-lab/Qwen3.8-27B-DFlash2-GGUF"; + string draftRevision = new('c', 40); + Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, draftRepo, draftRevision, "draft.gguf", + out string draftPath, out _, out string error), error); + // The primary model's cached path must be the real hub-cache snapshot path + // for its own repository and revision, or the receipt fails validation + // before the additional-asset check under test is ever reached. + Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, + "unsloth/Qwen3.6-35B-A3B-MTP-GGUF", + "5bc3e238d916f48a861bac2f8a1990a0e9b7e98d", + "Qwen3.6-35B-A3B-UD-Q4_K_M.gguf", + out string cachedPrimaryPath, out _, out error), error); + Directory.CreateDirectory(Path.GetDirectoryName(cachedPrimaryPath)!); + await File.WriteAllTextAsync(cachedPrimaryPath, "primary"); + Directory.CreateDirectory(Path.GetDirectoryName(draftPath)!); + await File.WriteAllTextAsync(draftPath, "draft"); + LocalAiInstallManifest schemaFive = installed.Manifest with + { + SchemaVersion = LocalAiInstallManifest.AdditionalAssetsSchemaVersion, + ModelCacheRoot = cacheRoot, + CachedModelPath = cachedPrimaryPath, + AdditionalModelAssets = ImmutableArray.Create(new LocalAiAssetReceipt + { + FileName = "draft.gguf", + SourceUrl = $"https://huggingface.co/{draftRepo}/resolve/{draftRevision}/draft.gguf?download=true", + SizeBytes = 1_143_006_816, + Sha256 = new string('d', 64), + }), + AdditionalModelPaths = ImmutableArray.Create(draftPath), + }; + + var events = new SynchronizedEventLog(); + var platform = new FakePlatform(); + await using var runtime = CreateRuntime( + paths, + new FakeProcessHost(platform, events, selectedPort: 28_771), + platform, + new FakeClient(events), + new FakeLifecycle(events), + modelFileVerifier: new SelectiveModelFileVerifier( + cachedPrimaryPath, + rejectPath: draftPath)); + await store.SaveAsync(schemaFive); + + LocalAiRuntimeSnapshot snapshot = await runtime.EnsureStartedAsync(); + + Assert.Equal(LocalAiRuntimeState.Failed, snapshot.State); + Assert.Contains("draft.gguf", snapshot.Detail ?? string.Empty, StringComparison.Ordinal); + } + + /// Verifies the primary model but rejects one named additional asset. + private sealed class SelectiveModelFileVerifier(string resolvedPath, string rejectPath) + : ILocalAiModelFileVerifier + { + public Task TryOpenAsync( + string cacheRoot, + string candidatePath, + long expectedSizeBytes, + Sha256Digest expectedSha256, + CancellationToken cancellationToken) + { + cancellationToken.ThrowIfCancellationRequested(); + if (string.Equals(candidatePath, rejectPath, StringComparison.OrdinalIgnoreCase)) + return Task.FromResult(null); + return Task.FromResult( + new LocalAiVerifiedModelLease(new MemoryStream(), resolvedPath)); + } + } + private static LocalAiInstallManifest ValidManifest() { LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.Find( From 32cd79622048258cb4083d68eb87098ab6232771 Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Thu, 24 Sep 2026 10:48:10 -0400 Subject: [PATCH 11/13] fix(local-ai): rank, verify, and disclose models by total download size A recipe's pinned weights are not everything it downloads or loads: a DFlash recipe also pulls a separate draft checkpoint, and llama-server loads both. Ranking, the post-launch GPU-load sanity check, and the user-facing size disclosure all read the primary weights alone, so a DFlash recipe understated its footprint and the setup review omitted the draft checkpoint entirely. Adds LocalModelCatalog.TotalDownloadSizeBytes (weights plus draft checkpoint) and switches SelectDefaultModelAndProfile's tie-break and fallback, LocalAiGpuVerification's minimum-load-delta check, and the setup UI's model picker and detail text to use it. SelectDefaultModelAndProfile also excludes priority-0, explicit-alternative models from the generic default and fallback pool, so a model reachable only through a specific SKU can never win the generic pick as the catalog grows. --- .../Pages/CapabilitiesPage.xaml.cs | 4 ++-- .../LocalAiGpuVerification.cs | 5 +++- .../Catalog/LocalInferenceSelector.cs | 18 +++++++++++--- .../Inference/Catalog/LocalModelCatalog.cs | 24 ++++++++++++++++--- 4 files changed, 42 insertions(+), 9 deletions(-) diff --git a/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs b/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs index 0c0b0d6d9..59f8cf1cf 100644 --- a/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs +++ b/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs @@ -679,7 +679,7 @@ private void PopulateLocalAiModels() LocalAiModelSelector.Items.Add(new ComboBoxItem { Content = $"{SetupReviewSummaryBuilder.DisplayModelName(model)} " + - $"({FormatSize(model.Weights.SizeBytes)}, " + + $"({FormatSize(LocalModelCatalog.TotalDownloadSizeBytes(model))}, " + $"{FormatContext(plan.Profile.ContextTokens)}, " + $"{LocalModelCatalog.ToDisplayCacheType(plan.Profile.KeyCachePrecision)} KV)" + (isRecommended ? " (Recommended)" : string.Empty), @@ -789,7 +789,7 @@ private void UpdateLocalAiModelDetails() "loads on first request"; LocalAiModelDetailText.Text = $"{SetupReviewSummaryBuilder.DisplayModelName(plan.Model)}, " + - $"{FormatSize(plan.Model.Weights.SizeBytes)} from Hugging Face"; + $"{FormatSize(LocalModelCatalog.TotalDownloadSizeBytes(plan.Model))} from Hugging Face"; UpdatePrimaryButtonState(); } diff --git a/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs b/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs index d9b16207e..6316da3d4 100644 --- a/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs +++ b/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs @@ -3,6 +3,7 @@ using System.Text.RegularExpressions; using OpenClaw.Connection.LocalAi; using OpenClaw.Shared.Inference; +using OpenClaw.Shared.Inference.Catalog; namespace OpenClaw.SetupEngine; @@ -297,7 +298,9 @@ ctx.LocalAiInferenceVerification is null || throw new InvalidDataException( "llama-server loaded CUDA from outside the managed runtime directory."); } - long minimumDelta = Math.Max(512L * 1024 * 1024, plan.Model.Weights.SizeBytes / 2); + long minimumDelta = Math.Max( + 512L * 1024 * 1024, + LocalModelCatalog.TotalDownloadSizeBytes(plan.Model) / 2); if (!HasRequiredGpuLoadEvidence(evidence, minimumDelta)) { throw new InvalidDataException( diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs index 1142fc6db..61cbe6e85 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs @@ -171,16 +171,28 @@ private static (LocalModelInfo Model, LocalInferenceRunProfile Profile) SelectDe HostHardwareInfo hardware, LlamaRuntimeVariant runtime) { - foreach (LocalModelInfo candidate in LocalModelCatalog.Models + // A priority-0, explicit-alternative model (currently only the + // experimental 96GB Flash-Next recipe) is offered solely by + // RtxSparkInferenceSelector for its one intended SKU, never picked as + // a generic dGPU default or fallback -- excluded here so a large + // enough non-Spark GPU (or a Spark GPU IsRtxSpark fails to detect) + // can't land on it by tie-break/fallback ordering. Priority-0 models + // that aren't explicit alternatives (the other Spark-only recipes) + // keep their existing, unrelated reachability. + IEnumerable genericDefaultCandidates = LocalModelCatalog.Models + .Where(model => model.RecommendationPriority > 0 || !model.IsExplicitAlternative); + foreach (LocalModelInfo candidate in genericDefaultCandidates .OrderByDescending(model => model.RecommendationPriority) - .ThenByDescending(model => model.Weights.SizeBytes)) + .ThenByDescending(LocalModelCatalog.TotalDownloadSizeBytes)) { LocalInferenceRunProfile? profile = SelectBestFittingProfile(hardware, runtime, candidate); if (profile is not null) return (candidate, profile); } - LocalModelInfo fallback = LocalModelCatalog.Models.OrderBy(model => model.Weights.SizeBytes).First(); + LocalModelInfo fallback = genericDefaultCandidates + .OrderBy(LocalModelCatalog.TotalDownloadSizeBytes) + .First(); return (fallback, LocalModelCatalog.GetProfiles(fallback)[^1]); } diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs index 96f31d96d..e15b59ba8 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs @@ -255,9 +255,14 @@ public static class LocalModelCatalog IsExplicitAlternative: true, SupportsVision: false, RecommendationPriority: 200), - // RTX Spark SKU recipes below. RecommendationPriority: 0 keeps them - // out of the generic dGPU default pick; only RtxSparkInferenceSelector - // offers them, keyed off the detected Spark unified-memory SKU. + // RTX Spark SKU recipes below, offered by RtxSparkInferenceSelector + // keyed off the detected Spark unified-memory SKU. RecommendationPriority: 0 + // alone does not exclude a model from the generic dGPU default pick -- + // it only loses every tie-break against a positive-priority model that + // fits. LocalInferenceSelector.SelectDefaultModelAndProfile separately + // excludes IsExplicitAlternative models at this priority (see its + // comment) so the always-alternative-only ones can't still win by + // tie-break/fallback ordering among themselves. new LocalModelInfo( Qwen35B_IQ4XSModelId, "Qwen3.6 35B-A3B (UD-IQ4_XS)", @@ -424,6 +429,19 @@ public static IReadOnlyList GetProfiles(LocalModelInfo /// True when the id resolves only to a retired catalog entry. public static bool IsLegacy(string? id) => Find(id) is null && FindInstalled(id) is not null; + /// + /// Everything setup downloads and llama-server loads for this recipe: the pinned + /// weights plus the DFlash draft checkpoint, if any. Use this for ranking, capacity + /// checks, and user-facing download-size disclosure -- the draft checkpoint is a + /// separate pinned artifact, not part of the target model's own weights, but it is + /// still bytes the user consents to, setup fetches, and the runtime loads. + /// + public static long TotalDownloadSizeBytes(LocalModelInfo model) + { + ArgumentNullException.ThrowIfNull(model); + return model.Weights.SizeBytes + (model.Recipe.DraftWeights?.SizeBytes ?? 0); + } + /// /// The recipe's additional pinned artifacts beyond its primary weights, in the /// fixed order every acquirer, manifest, and launch path must agree on. Today that From 7e519933a3444ee0bfa45cffb723568b1c0f45c1 Mon Sep 17 00:00:00 2001 From: Joel Fernandes Date: Wed, 23 Sep 2026 15:38:50 -0400 Subject: [PATCH 12/13] fix(local-ai): defer to the generic path when RTX Spark memory facts are incomplete An RTX Spark GPU whose CUDA memory couldn't be read reads as 0 bytes, which the fixed SKU table treated as a legitimately-too-small SKU (NotRecommendedForSku) instead of the real problem: unreadable capacity. Select() now only takes the Spark SKU-routing branch when the Spark GPU's facts are actually complete, so an incomplete-facts Spark GPU falls through to the generic path and gets correctly diagnosed as HardwareFactsIncomplete, same as any other GPU. Also bumps LocalAiPortHandoffTests' Spark fixture from an arbitrary ~24GiB (a pre-SKU-routing placeholder, now below the smallest 32GB tier) to a real 48GB SKU's measured cuMemGetInfo total, so it again resolves to the 24GB recipe instead of "not recommended." --- .../OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs index 1bf2e8ea8..b5d4e2f1a 100644 --- a/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs +++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs @@ -139,8 +139,11 @@ private static SetupContext CreateContext(LocalAiConfig localAi, string? localDa new GpuInfo( GpuVendor.Nvidia, "NVIDIA RTX Spark N1X (6144-core Blackwell RTX GPU)", - GpuVisibleMemoryBytes: 25_702_694_912, - FreeGpuVisibleMemoryBytes: 25_702_694_912, + // A real 48GB-SKU RTX Spark's measured cuMemGetInfo total, so this + // fixture routes through the fixed SKU table instead of landing + // below the smallest (32GB) recommended tier. + GpuVisibleMemoryBytes: 48_585_498_624, + FreeGpuVisibleMemoryBytes: 48_585_498_624, DriverVersion: "616.00", CudaMajorVersion: 13, StableId: "GPU-SPARK"), From 7d25102ba653fd04af7356ea02c54f007e36ae2c Mon Sep 17 00:00:00 2001 From: Karen Lai <7976322+karkarl@users.noreply.github.com> Date: Tue, 29 Sep 2026 08:10:37 -0700 Subject: [PATCH 13/13] fix(local-ai): migrate and roll back retired runtime upgrades safely Validate shipped runtime paths against their own release, migrate verified legacy models before upgrade reuse, and restore the pre-upgrade receipt on rollback independently of gateway recovery. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> Copilot-Session: 9e59ac0b-0ccb-4614-bb44-088bb882d0ab --- docs/SETUP_ENGINE_REDESIGN.md | 11 + .../LlamaRuntimeInstaller.cs | 2 +- .../LocalAiInstallReconciler.cs | 13 ++ src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs | 52 +++-- src/OpenClaw.SetupEngine/SetupContext.cs | 1 + .../LocalAiInstallRecoveryTests.cs | 193 ++++++++++++++++-- 6 files changed, 239 insertions(+), 33 deletions(-) diff --git a/docs/SETUP_ENGINE_REDESIGN.md b/docs/SETUP_ENGINE_REDESIGN.md index 26ed96f5a..24e15a033 100644 --- a/docs/SETUP_ENGINE_REDESIGN.md +++ b/docs/SETUP_ENGINE_REDESIGN.md @@ -272,6 +272,17 @@ hub-cache snapshot as the active model path, while preserving the legacy compatibility path and the prior gateway fallback, install time, and rollback metadata. +Runtime upgrades validate the installed executable against its recorded runtime +release, not the current catalog release. A verified schema-3 model is migrated +to the hub cache before reuse by the new runtime, without downloading it again. +Normal setup keeps the pre-upgrade receipt separate from gateway recovery state. +If a later step fails, receipt persistence restores that baseline before runtime +acquisition removes the newly installed runtime. Reconciliation also restores +the baseline when setup fails after migration but before receipt persistence. +The old runtime, compatibility model, and verified shared-cache copy are retained. +Superseded runtime directories remain until explicit uninstall; upgrades do not +prune the rollback baseline. + Completed cache files and pre-existing resumable partials are shared state. Setup rollback and uninstall do not delete them. Unsafe links, reparse points, hard-linked partials, destination conflicts, receipt mismatches, and concurrent diff --git a/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs b/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs index f991f4e76..e9fb5b693 100644 --- a/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs +++ b/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs @@ -130,7 +130,7 @@ public async Task InstallAsync( internal static LocalAiComponentIdentity Component(LlamaRuntimeVariant runtime) => new( "llama-server", - LlamaRuntimeCatalog.ReleaseTag, + runtime.ReleaseTag, runtime.Architecture switch { Architecture.X64 => "win-x64", diff --git a/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs b/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs index 051d61b71..afea2bcc5 100644 --- a/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs +++ b/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs @@ -185,6 +185,19 @@ public async Task ReconcileAsync( bool modelIsValid = activeModelIsValid && legacyModelIsValid && additionalAssetsAreValid; if (runtimeUpgradePending) { + if (modelIsValid) + { + install = await MigrateLegacyModelAsync( + install, + paths, + localDataDirectory, + plan, + selectedGpuId, + migrationProgress, + cancellationToken) + .ConfigureAwait(false); + } + // The catalog moved to a newer pinned runtime. Drop only the runtime so the // acquirer installs the new one, and keep the verified model and its extra // assets so an upgrade does not re-download tens of GB. OriginalInstall lets diff --git a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs index 56135bea2..085e14040 100644 --- a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs +++ b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs @@ -287,11 +287,11 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati } if (!result.Reused) { - // A retained baseline receipt (gateway recovery, or a pending runtime - // upgrade) is what lets the manifest step replace the existing receipt - // instead of refusing because one is already present. - if (result.OriginalInstall is { } retainedReceipt) - ctx.LocalAiRecoveryOriginalInstall ??= retainedReceipt; + // Normal upgrades restore their receipt independently of the gateway + // recovery pipeline's endpoint-health and provider rollback guards. + if (ctx.LocalAiRecoveryOriginalInstall is null && + result.OriginalInstall is { } retainedReceipt) + ctx.LocalAiUpgradeOriginalInstall ??= retainedReceipt; ctx.LocalAiRuntimeInstall = result.RuntimeInstall; ctx.LocalAiModelInstall = result.ModelInstall; ctx.LocalAiAdditionalModelInstalls = result.AdditionalModelInstalls @@ -321,6 +321,11 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati ex); } } + + public override Task RollbackAsync(SetupContext ctx, CancellationToken ct) => + ctx.IsUninstalling + ? Task.CompletedTask + : PersistLocalAiManifestStep.RestoreUpgradeReceiptAsync(ctx, ct); } /// Installs the two pinned llama.cpp runtime archives as one atomic component. @@ -391,7 +396,7 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati progress, linked.Token); ctx.LocalAiRuntimeInstall = install; - return StepResult.Ok($"Installed llama-server {LlamaRuntimeCatalog.ReleaseTag}."); + return StepResult.Ok($"Installed llama-server {plan.Runtime.ReleaseTag}."); } catch (OperationCanceledException) when (ct.IsCancellationRequested) { @@ -599,10 +604,12 @@ ctx.LocalAiModelInstall is not return StepResult.Terminal(portError ?? "The requested Local AI port is invalid."); var paths = new LocalAiPaths(ctx.LocalDataDir); - bool replacesRecoveryReceipt = - ctx.LocalAiRecoveryOriginalInstall is not null && + LocalAiResolvedInstall? originalInstall = + ctx.LocalAiRecoveryOriginalInstall ?? ctx.LocalAiUpgradeOriginalInstall; + bool replacesExistingReceipt = + originalInstall is not null && File.Exists(paths.ManifestPath); - if (File.Exists(paths.ManifestPath) && !replacesRecoveryReceipt) + if (File.Exists(paths.ManifestPath) && !replacesExistingReceipt) return StepResult.Terminal("A managed Local AI installation receipt already exists."); LocalAiComponentIdentity component = LlamaRuntimeInstaller.Component(plan.Runtime); if (!LocalAiPathPolicy.TryResolve( @@ -670,7 +677,7 @@ ctx.LocalAiRecoveryOriginalInstall is not null && SchemaVersion = additionalArtifacts.IsEmpty ? LocalAiInstallManifest.HubCacheReceiptSchemaVersion : LocalAiInstallManifest.AdditionalAssetsSchemaVersion, - EngineVersion = LlamaRuntimeCatalog.ReleaseTag, + EngineVersion = plan.Runtime.ReleaseTag, Architecture = plan.Runtime.Architecture switch { Architecture.X64 => "x64", @@ -707,7 +714,7 @@ ctx.LocalAiRecoveryOriginalInstall is not null && DraftKeyCachePrecision = plan.Profile.DraftKeyCachePrecision, DraftValueCachePrecision = plan.Profile.DraftValueCachePrecision, }; - if (ctx.LocalAiRecoveryOriginalInstall is { } originalInstall) + if (originalInstall is not null) { manifest = originalInstall.Manifest with { @@ -742,7 +749,7 @@ ctx.LocalAiRecoveryOriginalInstall is not null && { await store.SaveAsync(manifest, ct); ctx.LocalAiResolvedInstall = store.ResolveAndValidate(manifest); - ctx.LocalAiManifestCreatedThisRun = !replacesRecoveryReceipt; + ctx.LocalAiManifestCreatedThisRun = !replacesExistingReceipt; return StepResult.Ok("Recorded the verified llama-server and Hugging Face installation."); } catch (Exception ex) when (ex is IOException or UnauthorizedAccessException or InvalidDataException) @@ -771,10 +778,17 @@ public override async Task RollbackAsync(SetupContext ctx, CancellationToken ct) ctx.LocalAiRuntimeInstall = null; ctx.LocalAiModelInstall = null; ctx.LocalAiResolvedInstall = null; + ctx.LocalAiUpgradeOriginalInstall = null; ctx.LocalAiManifestCreatedThisRun = false; return; } + if (ctx.LocalAiUpgradeOriginalInstall is not null) + { + await RestoreUpgradeReceiptAsync(ctx, ct); + return; + } + if (!ctx.LocalAiManifestCreatedThisRun) return; @@ -786,6 +800,20 @@ public override async Task RollbackAsync(SetupContext ctx, CancellationToken ct) ctx.LocalAiManifestCreatedThisRun = false; } + internal static async Task RestoreUpgradeReceiptAsync(SetupContext ctx, CancellationToken ct) + { + if (ctx.LocalAiUpgradeOriginalInstall is not { } originalInstall) + return; + + var paths = new LocalAiPaths(ctx.LocalDataDir); + var store = new LocalAiManifestStore(paths); + await store.SaveAsync(originalInstall.Manifest, ct); + File.Delete(paths.RouterPresetPath); + ctx.LocalAiResolvedInstall = store.ResolveAndValidate(originalInstall.Manifest); + ctx.LocalAiManifestCreatedThisRun = false; + ctx.LocalAiUpgradeOriginalInstall = null; + } + private static ImmutableArray BuildRuntimeReceipts( LlamaRuntimeVariant runtime, LlamaRuntimeInstallResult install) diff --git a/src/OpenClaw.SetupEngine/SetupContext.cs b/src/OpenClaw.SetupEngine/SetupContext.cs index 5cd0d7a94..4390a6f14 100644 --- a/src/OpenClaw.SetupEngine/SetupContext.cs +++ b/src/OpenClaw.SetupEngine/SetupContext.cs @@ -525,6 +525,7 @@ public Func>? ImmutableArray.Empty; internal LocalAiResolvedInstall? LocalAiResolvedInstall { get; set; } internal LocalAiResolvedInstall? LocalAiRecoveryOriginalInstall { get; set; } + internal LocalAiResolvedInstall? LocalAiUpgradeOriginalInstall { get; set; } internal bool LocalAiRecoveryProviderTransition { get; set; } internal bool LocalAiRecoveryReceiptRollbackAllowed { get; set; } internal bool LocalAiManifestCreatedThisRun { get; set; } diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs index 9773e560a..38381a515 100644 --- a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs +++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs @@ -1254,30 +1254,12 @@ public async Task Reconciler_UpgradesRetiredRuntimeReceiptInsteadOfFailingSetup( // uninstall instruction. The runtime is dropped so the acquirer installs the new // pin; the verified model is kept so an upgrade does not re-download it. using var temp = new TempDirectory(); + using var environment = new EnvironmentScope("HF_HUB_CACHE", CacheRoot(temp.Path)); LocalInferencePlan plan = CatalogPlan(); const string gpuId = "GPU-0"; var paths = new LocalAiPaths(temp.Path); LlamaRuntimeVariant retired = LlamaRuntimeCatalog.FindInstalled("b10655-cuda13-x64")!; - Assert.True(LocalAiPathPolicy.TryResolve( - temp.Path, - LlamaRuntimeInstaller.Component(retired), - out LocalAiSetupPaths retiredPaths, - out string retiredError), retiredError); - LocalAiInstallManifest manifest = CreateManifest(temp.Path, plan, gpuId) with - { - EngineVersion = retired.ReleaseTag, - RuntimeId = retired.Id, - ExecutablePath = Path.GetRelativePath( - paths.RootDirectory, - Path.Combine(retiredPaths.InstallDirectory, LlamaRuntimeCatalog.ServerExecutableName)), - RuntimeAssets = retired.Artifacts.Select(artifact => new LocalAiAssetReceipt - { - FileName = Path.GetFileName(artifact.RelativePath), - SourceUrl = artifact.DownloadUri.AbsoluteUri, - SizeBytes = artifact.SizeBytes, - Sha256 = artifact.Sha256.Value, - }).ToImmutableArray(), - }; + LocalAiInstallManifest manifest = CreateRetiredManifest(temp.Path, plan, gpuId); await new LocalAiManifestStore(paths).SaveAsync(manifest); var reconciler = new LocalAiInstallReconciler( new ValidRuntimeInspector(), @@ -1296,6 +1278,143 @@ public async Task Reconciler_UpgradesRetiredRuntimeReceiptInsteadOfFailingSetup( Assert.Equal(retired.ReleaseTag, result.OriginalInstall!.Manifest.EngineVersion); } + [Theory] + [InlineData(false, null)] + [InlineData(true, null)] + [InlineData(false, "after-reconcile")] + [InlineData(true, "after-reconcile")] + [InlineData(false, "after-persist")] + [InlineData(true, "after-persist")] + public async Task RuntimeUpgrade_MigratesModelAndRestoresOriginalReceiptOnFailure( + bool usesHubCache, + string? failureStage) + { + using var temp = new TempDirectory(); + string cacheRoot = CacheRoot(temp.Path); + byte[] modelBytes = "verified-upgrade-model"u8.ToArray(); + byte[] runtimeZip = CreateZip(("llama-server.exe", "new-server"u8.ToArray())); + byte[] dependencyZip = CreateZip(("dependency.dll", "dependency"u8.ToArray())); + LlamaRuntimeVariant runtime = CreateRuntime(runtimeZip, dependencyZip); + LocalInferencePlan plan = CatalogPlan() with + { + Runtime = runtime, + Model = CreateModel(modelBytes), + }; + var paths = new LocalAiPaths(temp.Path); + var store = new LocalAiManifestStore(paths, () => cacheRoot); + LocalAiInstallManifest manifest = CreateRetiredManifest(temp.Path, plan, "GPU-0") with + { + GatewayFallbackModel = "openai/gpt-5", + }; + string oldExecutable = paths.ResolveContainedPath(manifest.ExecutablePath, "executable"); + Directory.CreateDirectory(Path.GetDirectoryName(oldExecutable)!); + await File.WriteAllTextAsync(oldExecutable, "old-server"); + string legacyModel = paths.ResolveContainedPath(manifest.ModelPath, "model"); + Directory.CreateDirectory(Path.GetDirectoryName(legacyModel)!); + await File.WriteAllBytesAsync(legacyModel, modelBytes); + await store.SaveAsync(manifest); + if (usesHubCache) + manifest = (await store.MigrateLegacyModelToHubCacheAsync())!.Manifest; + byte[] originalReceipt = await File.ReadAllBytesAsync(paths.ManifestPath); + + var context = CreateContext(temp.Path, confirmDestructive: false); + context.Config.LocalAi.Enabled = true; + context.Config.RollbackOnFailure = true; + context.LocalAiPort = manifest.RequestedPort; + context.LocalAiEligibility = new LocalInferenceEligibilityResult( + LocalInferenceEligibilityStatus.Eligible, + LocalInferenceEligibilityFailureCode.None, + LocalInferenceSelectionFailureCode.None, + plan, + new GpuInfo(GpuVendor.Nvidia, "Test GPU", StableId: "GPU-0"), + RequiredTotalMemoryBytes: 0, + DetectedTotalMemoryBytes: 0, + RequiredFreeMemoryBytes: 0, + AvailableFreeMemoryBytes: 0); + int runtimeDownloads = 0; + using var runtimeClient = new HttpClient(new DelegateHandler(request => + { + runtimeDownloads++; + return new HttpResponseMessage(HttpStatusCode.OK) + { + Content = new ByteArrayContent( + request.RequestUri!.AbsolutePath.EndsWith("runtime.zip", StringComparison.Ordinal) + ? runtimeZip + : dependencyZip), + }; + })); + using var modelClient = new HttpClient(new DelegateHandler(_ => + throw new InvalidOperationException("Upgrade must not download the verified primary model."))); + var reconciler = new LocalAiInstallReconciler( + new ValidRuntimeInspector(), new LocalAiModelFileVerifier(), () => cacheRoot); + string? cachedModel = null; + string? newExecutable = null; + var pipeline = new SetupPipeline( + [ + new ReconcileLocalAiInstallationStep(reconciler), + new UpgradeCheckpointStep("after-reconcile", ctx => + { + Assert.Null(ctx.LocalAiRecoveryOriginalInstall); + Assert.Equal(manifest.SchemaVersion, ctx.LocalAiUpgradeOriginalInstall?.Manifest.SchemaVersion); + Assert.Equal(oldExecutable, ctx.LocalAiUpgradeOriginalInstall?.ExecutablePath); + Assert.Equal(cacheRoot, ctx.LocalAiModelInstall?.CacheRoot); + cachedModel = ctx.LocalAiModelInstall!.ModelPath; + return failureStage == "after-reconcile"; + }), + new AcquireLocalAiRuntimeStep(new LlamaRuntimeInstaller( + new LocalAiArtifactInstaller(runtimeClient), new ValidRuntimeInspector())), + new AcquireLocalAiModelStep(CreateModelInstaller(modelClient, temp.Path)), + new PersistLocalAiManifestStep(), + new UpgradeCheckpointStep("after-persist", ctx => + { + LocalAiResolvedInstall upgraded = Assert.IsType(ctx.LocalAiResolvedInstall); + Assert.Equal("b11026", upgraded.Manifest.EngineVersion); + Assert.Equal(LocalAiInstallManifest.HubCacheReceiptSchemaVersion, upgraded.Manifest.SchemaVersion); + Assert.Equal(cacheRoot, upgraded.Manifest.ModelCacheRoot); + Assert.Equal(cachedModel, upgraded.ModelPath); + Assert.Equal(manifest.InstalledAtUtc, upgraded.Manifest.InstalledAtUtc); + Assert.Equal(manifest.GatewayFallbackModel, upgraded.Manifest.GatewayFallbackModel); + Assert.Null(upgraded.Endpoint); + newExecutable = upgraded.ExecutablePath; + Assert.True(File.Exists(newExecutable)); + return failureStage == "after-persist"; + }), + ]); + + PipelineResult result = await pipeline.RunAsync(context); + + Assert.True( + result.Outcome == (failureStage is null ? PipelineOutcome.Success : PipelineOutcome.Failed), + result.Message); + Assert.Equal(failureStage, result.FailedStepId); + if (failureStage is not null) + Assert.Equal("Injected upgrade failure.", result.Message); + Assert.Equal(failureStage == "after-reconcile" ? 0 : 2, runtimeDownloads); + Assert.Equal(modelBytes, await File.ReadAllBytesAsync(cachedModel!)); + Assert.Equal(modelBytes, await File.ReadAllBytesAsync(legacyModel)); + Assert.Equal("old-server", await File.ReadAllTextAsync(oldExecutable)); + LocalAiResolvedInstall persisted = (await store.LoadAsync())!; + if (failureStage is null) + { + Assert.Equal(newExecutable, persisted.ExecutablePath); + Assert.Equal("b11026", persisted.Manifest.EngineVersion); + } + else + { + Assert.Equal(originalReceipt, await File.ReadAllBytesAsync(paths.ManifestPath)); + Assert.Equal(oldExecutable, persisted.ExecutablePath); + Assert.Null(context.LocalAiUpgradeOriginalInstall); + if (newExecutable is not null) + Assert.False(File.Exists(newExecutable)); + + LocalAiReconcileResult retry = await reconciler.ReconcileAsync( + temp.Path, plan, "GPU-0", CancellationToken.None); + Assert.False(retry.Reused); + Assert.Null(retry.RuntimeInstall); + Assert.Equal(cacheRoot, retry.ModelInstall?.CacheRoot); + } + } + [Fact] public async Task Reconciler_RejectsMigrationCacheRootInsideManagedInstallTree() { @@ -1629,6 +1748,28 @@ private static LocalAiInstallManifest CreateManifest( }; } + private static LocalAiInstallManifest CreateRetiredManifest( + string localDataDirectory, + LocalInferencePlan plan, + string gpuId) + { + LlamaRuntimeVariant retired = LlamaRuntimeCatalog.FindInstalled("b10655-cuda13-x64")!; + return CreateManifest(localDataDirectory, plan, gpuId) with + { + EngineVersion = "b10655", + RuntimeId = retired.Id, + // Use the shipped path, not the installer helper under test. + ExecutablePath = Path.Combine("engines", "llama-server", "b10655", "win-x64", "llama-server.exe"), + RuntimeAssets = retired.Artifacts.Select(artifact => new LocalAiAssetReceipt + { + FileName = Path.GetFileName(artifact.RelativePath), + SourceUrl = artifact.DownloadUri.AbsoluteUri, + SizeBytes = artifact.SizeBytes, + Sha256 = artifact.Sha256.Value, + }).ToImmutableArray(), + }; + } + private static LocalModelInfo CreateModel(byte[] bytes) { var source = new HuggingFaceRevisionSource("owner/repo", new string('a', 40)); @@ -2006,6 +2147,18 @@ public override Task ExecuteAsync(SetupContext ctx, CancellationToke } } + private sealed class UpgradeCheckpointStep( + string id, + Func shouldFail) : SetupStep + { + public override string Id => id; + public override string DisplayName => id; + public override bool CanRetry => false; + + public override Task ExecuteAsync(SetupContext ctx, CancellationToken ct) => + Task.FromResult(shouldFail(ctx) ? StepResult.Fail("Injected upgrade failure.") : StepResult.Ok("Checked.")); + } + private sealed class TempDirectory : IDisposable { public TempDirectory()