diff --git a/docs/SETUP_ENGINE_REDESIGN.md b/docs/SETUP_ENGINE_REDESIGN.md
index 26ed96f5a..24e15a033 100644
--- a/docs/SETUP_ENGINE_REDESIGN.md
+++ b/docs/SETUP_ENGINE_REDESIGN.md
@@ -272,6 +272,17 @@ hub-cache snapshot as the active model path, while preserving the legacy
compatibility path and the prior gateway fallback, install time, and rollback
metadata.
+Runtime upgrades validate the installed executable against its recorded runtime
+release, not the current catalog release. A verified schema-3 model is migrated
+to the hub cache before reuse by the new runtime, without downloading it again.
+Normal setup keeps the pre-upgrade receipt separate from gateway recovery state.
+If a later step fails, receipt persistence restores that baseline before runtime
+acquisition removes the newly installed runtime. Reconciliation also restores
+the baseline when setup fails after migration but before receipt persistence.
+The old runtime, compatibility model, and verified shared-cache copy are retained.
+Superseded runtime directories remain until explicit uninstall; upgrades do not
+prune the rollback baseline.
+
Completed cache files and pre-existing resumable partials are shared state.
Setup rollback and uninstall do not delete them. Unsafe links, reparse points,
hard-linked partials, destination conflicts, receipt mismatches, and concurrent
diff --git a/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs b/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs
index 49f9d558e..33cbb2305 100644
--- a/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs
+++ b/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs
@@ -31,19 +31,27 @@ public static LlamaServerRouterLaunchPlan Build(
LocalAiPaths paths,
LocalAiResolvedInstall install,
int? listenPort = null) =>
- BuildCore(paths, install, install.ModelPath, listenPort);
+ BuildCore(paths, install, install.ModelPath, verifiedDraftModelPath: null, listenPort);
+ ///
+ /// The draft checkpoint's handle-resolved physical path, from the same verification
+ /// that opened it. Passing the persisted snapshot path instead would let a
+ /// snapshot-link replacement change the file llama-server finally opens, which is
+ /// exactly what resolving the primary model through its own handle prevents.
+ ///
internal static LlamaServerRouterLaunchPlan BuildForVerifiedRuntime(
LocalAiPaths paths,
LocalAiResolvedInstall install,
string verifiedModelPath,
+ string? verifiedDraftModelPath,
int? listenPort = null) =>
- BuildCore(paths, install, verifiedModelPath, listenPort);
+ BuildCore(paths, install, verifiedModelPath, verifiedDraftModelPath, listenPort);
private static LlamaServerRouterLaunchPlan BuildCore(
LocalAiPaths paths,
LocalAiResolvedInstall install,
string modelPath,
+ string? verifiedDraftModelPath,
int? listenPort)
{
ArgumentNullException.ThrowIfNull(paths);
@@ -53,13 +61,16 @@ private static LlamaServerRouterLaunchPlan BuildCore(
LocalAiInstallManifest manifest = install.Manifest;
int port = listenPort ?? manifest.RequestedPort;
LocalAiPortPolicy.Validate(port);
- LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.Variants.SingleOrDefault(
- candidate => string.Equals(candidate.Id, manifest.RuntimeId, StringComparison.Ordinal))
+ // FindInstalled, not Variants: an installation recorded before the last
+ // runtime bump must keep launching until setup upgrades it, instead of being
+ // stranded the moment the catalog moves to a newer pinned release.
+ LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.FindInstalled(manifest.RuntimeId)
?? throw new InvalidDataException("The managed llama-server runtime is no longer qualified.");
LocalModelInfo model = LocalModelCatalog.FindInstalled(manifest.ModelCatalogId)
?? throw new InvalidDataException("The managed local AI model is no longer qualified.");
LocalInferenceRunProfile profile = ResolveQualifiedReceipt(manifest, runtime, model);
+ string? draftModelPath = ResolveDraftModelPath(manifest, model, verifiedDraftModelPath);
string presetPath = paths.ResolveContainedPath(
Path.GetRelativePath(paths.RootDirectory, paths.RouterPresetPath),
@@ -84,7 +95,7 @@ private static LlamaServerRouterLaunchPlan BuildCore(
.WithComparers(StringComparer.OrdinalIgnoreCase)
.Add("CUDA_VISIBLE_DEVICES", manifest.SelectedGpuId),
presetPath,
- BuildPreset(model, profile, modelPath),
+ BuildPreset(model, profile, modelPath, draftModelPath),
model.Id);
}
@@ -120,7 +131,7 @@ internal static void ValidateArtifactReceipts(
{
throw new InvalidDataException("The managed local AI architecture and runtime receipt do not match.");
}
- if (!string.Equals(manifest.EngineVersion, LlamaRuntimeCatalog.ReleaseTag, StringComparison.Ordinal) ||
+ if (!string.Equals(manifest.EngineVersion, runtime.ReleaseTag, StringComparison.Ordinal) ||
!string.Equals(manifest.ModelAlias, model.Id, StringComparison.Ordinal))
{
throw new InvalidDataException("The managed local AI model recipe receipt does not match the qualified catalog.");
@@ -145,17 +156,65 @@ internal static void ValidateArtifactReceipts(
{
throw new InvalidDataException("The managed model artifact receipt does not match the qualified catalog.");
}
+
+ ImmutableArray expectedAdditionalArtifacts = LocalModelCatalog.AdditionalArtifacts(model);
+ if (manifest.AdditionalModelAssetsOrEmpty.Length != expectedAdditionalArtifacts.Length ||
+ manifest.AdditionalModelPathsOrEmpty.Length != expectedAdditionalArtifacts.Length)
+ {
+ throw new InvalidDataException(
+ "The managed additional model asset receipts do not match the qualified catalog.");
+ }
+ for (int i = 0; i < expectedAdditionalArtifacts.Length; i++)
+ {
+ PinnedArtifact artifact = expectedAdditionalArtifacts[i];
+ LocalAiAssetReceipt receipt = manifest.AdditionalModelAssetsOrEmpty[i];
+ if (!string.Equals(receipt.FileName, Path.GetFileName(artifact.RelativePath), StringComparison.Ordinal) ||
+ receipt.SizeBytes != artifact.SizeBytes ||
+ !string.Equals(receipt.Sha256, artifact.Sha256.Value, StringComparison.Ordinal) ||
+ !string.Equals(receipt.SourceUrl, artifact.DownloadUri.AbsoluteUri, StringComparison.Ordinal))
+ {
+ throw new InvalidDataException(
+ "The managed additional model asset receipts do not match the qualified catalog.");
+ }
+ }
+ }
+
+ ///
+ /// The DFlash draft checkpoint's path for the preset. Prefers the handle-resolved
+ /// physical path supplied by the caller that verified and still holds the file, so a
+ /// snapshot-link replacement cannot change the identity llama-server opens. Falls
+ /// back to the persisted receipt path only for callers that do not verify first
+ /// (, used for inspection rather than launch). Null for recipes
+ /// with no separate draft checkpoint. Callers must validate the manifest via
+ /// first, which guarantees
+ /// AdditionalModelPaths has one entry per catalog-pinned artifact.
+ ///
+ private static string? ResolveDraftModelPath(
+ LocalAiInstallManifest manifest,
+ LocalModelInfo model,
+ string? verifiedDraftModelPath)
+ {
+ if (model.Recipe.DraftWeights is null)
+ return null;
+ return string.IsNullOrWhiteSpace(verifiedDraftModelPath)
+ ? manifest.AdditionalModelPathsOrEmpty[^1]
+ : verifiedDraftModelPath;
}
private static string BuildPreset(
LocalModelInfo model,
LocalInferenceRunProfile profile,
- string modelPath)
+ string modelPath,
+ string? draftModelPath)
{
if (modelPath.IndexOfAny(['\r', '\n']) >= 0)
throw new InvalidDataException("The managed model path cannot be represented safely in a llama-server preset.");
+ if (draftModelPath is not null && draftModelPath.IndexOfAny(['\r', '\n']) >= 0)
+ throw new InvalidDataException("The managed draft model path cannot be represented safely in a llama-server preset.");
LocalModelRunRecipe recipe = model.Recipe;
+ if (recipe.SpeculativeDecoding == SpeculativeDecodingMode.DraftDFlash && draftModelPath is null)
+ throw new InvalidDataException("Draft-flash decoding requires a resolved draft model path.");
ModelSamplingPreset sampling = recipe.Sampling;
var preset = new StringBuilder();
preset.AppendLine("version = 1");
@@ -178,9 +237,24 @@ private static string BuildPreset(
preset.AppendLine("main-gpu = 0");
preset.AppendLine("fit = off");
preset.AppendLine("load-mode = dio");
- preset.AppendLine("spec-type = draft-mtp");
- preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens));
- preset.AppendLine("spec-draft-backend-sampling = true");
+ switch (recipe.SpeculativeDecoding)
+ {
+ case SpeculativeDecodingMode.DraftMtp:
+ preset.AppendLine("spec-type = draft-mtp");
+ preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens));
+ preset.AppendLine("spec-draft-backend-sampling = true");
+ break;
+ case SpeculativeDecodingMode.DraftDFlash:
+ preset.AppendLine("spec-type = draft-dflash");
+ preset.Append("spec-draft-model = ").AppendLine(draftModelPath);
+ preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens));
+ preset.AppendLine("spec-draft-backend-sampling = true");
+ break;
+ case SpeculativeDecodingMode.None:
+ break;
+ default:
+ throw new ArgumentOutOfRangeException(nameof(recipe.SpeculativeDecoding));
+ }
preset.Append("temperature = ").AppendLine(Invariant(sampling.Temperature));
preset.Append("top-k = ").AppendLine(Invariant(sampling.TopK));
preset.Append("top-p = ").AppendLine(Invariant(sampling.TopP));
diff --git a/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs b/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs
index ff5a0bad8..3828b4cc6 100644
--- a/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs
+++ b/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs
@@ -118,6 +118,7 @@ public sealed class LlamaServerRuntimeService : ILocalAiRuntime
private LocalAiRuntimeSnapshot _snapshot;
private ILocalAiManagedProcess? _managedProcess;
private LocalAiVerifiedModelLease? _verifiedModel;
+ private readonly List _verifiedAdditionalAssets = [];
private string? _runtimeModelPath;
private LocalAiResolvedInstall? _install;
private long _generation;
@@ -378,6 +379,7 @@ private async Task EnsureStartedCoreAsync(CancellationTo
_options.Paths,
install,
GetRuntimeModelPath(install),
+ GetRuntimeDraftModelPath(),
requestedPort);
await WritePresetAtomicallyAsync(launchPlan, cancellationToken).ConfigureAwait(false);
}
@@ -862,7 +864,7 @@ private async Task ValidateInstalledFilesAsync(
DisposeVerifiedModelHandle();
ValidateInstalledFilesForStatus(install);
- if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion)
+ if (!install.Manifest.UsesHubCache)
{
_runtimeModelPath = install.ModelPath;
return;
@@ -883,6 +885,33 @@ await _modelFileVerifier.TryOpenAsync(
"The shared Hugging Face cache model is unsafe or no longer matches its receipt.");
}
+ // Schema-5 extra assets (a DFlash draft checkpoint) are loaded natively
+ // by llama-server exactly like the primary weights, and they
+ // live in the same shared, user-writable hub cache. Rehash them here and hold
+ // the handles for the process lifetime, so a file swapped after setup cannot
+ // reach the loader with only a structural path check behind it.
+ foreach ((LocalAiAssetReceipt receipt, string cachedPath) in
+ install.Manifest.AdditionalModelAssetsOrEmpty
+ .Zip(install.Manifest.AdditionalModelPathsOrEmpty))
+ {
+ LocalAiVerifiedModelLease? verifiedAsset =
+ await _modelFileVerifier.TryOpenAsync(
+ install.Manifest.ModelCacheRoot!,
+ cachedPath,
+ receipt.SizeBytes,
+ new Sha256Digest(receipt.Sha256),
+ cancellationToken)
+ .ConfigureAwait(false);
+ if (verifiedAsset is null)
+ {
+ DisposeVerifiedModelHandle();
+ throw new InvalidDataException(
+ $"The shared Hugging Face cache asset '{receipt.FileName}' is unsafe or no longer matches its receipt.");
+ }
+
+ _verifiedAdditionalAssets.Add(verifiedAsset);
+ }
+
_runtimeModelPath = _verifiedModel.ResolvedPath;
}
@@ -890,7 +919,7 @@ private static void ValidateInstalledFilesForStatus(LocalAiResolvedInstall insta
{
if (!File.Exists(install.ExecutablePath))
throw new InvalidDataException("The managed llama-server executable is missing.");
- if (install.Manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion)
+ if (install.Manifest.UsesHubCache)
{
if (!File.Exists(install.ModelPath))
throw new InvalidDataException("The managed GGUF model is missing.");
@@ -1666,14 +1695,26 @@ private void DisposeVerifiedModelHandle()
{
_verifiedModel?.Dispose();
_verifiedModel = null;
+ foreach (LocalAiVerifiedModelLease lease in _verifiedAdditionalAssets)
+ lease.Dispose();
+ _verifiedAdditionalAssets.Clear();
_runtimeModelPath = null;
}
+ ///
+ /// The handle-resolved path of the draft checkpoint this process verified and still
+ /// holds open, or null when the recipe has no additional assets. Additional assets are
+ /// verified in catalog order and the draft checkpoint is always last, matching
+ /// .
+ ///
+ private string? GetRuntimeDraftModelPath() =>
+ _verifiedAdditionalAssets.Count == 0 ? null : _verifiedAdditionalAssets[^1].ResolvedPath;
+
private string GetRuntimeModelPath(LocalAiResolvedInstall install)
{
if (_runtimeModelPath is not null)
return _runtimeModelPath;
- if (install.Manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion)
+ if (install.Manifest.UsesHubCache)
throw new InvalidOperationException("The verified shared-cache model identity is unavailable.");
return install.ModelPath;
}
diff --git a/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs b/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs
index ce3cbe886..895ab908c 100644
--- a/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs
+++ b/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs
@@ -148,8 +148,23 @@ public sealed record LocalAiInstallManifest
/// the model must also verify the pinned content before use.
///
public const int HubCacheReceiptSchemaVersion = 4;
+ ///
+ /// Schema-4 plus one or more additional model assets verified in the same
+ /// hub cache. Only recipes with a LocalModelRunRecipe.DraftWeights
+ /// pin use this; every existing single-asset recipe keeps writing schema 4.
+ ///
+ public const int AdditionalAssetsSchemaVersion = 5;
public const string SupportedEngine = "llama-server";
+ ///
+ /// True for schema 4 and schema 5, whose active model resolves through the
+ /// standard Hugging Face hub cache rather than the legacy app-owned copy.
+ /// Derived entirely from , so it must never be
+ /// persisted -- an older app build would reject it as an unknown field.
+ ///
+ [JsonIgnore]
+ public bool UsesHubCache => SchemaVersion is HubCacheReceiptSchemaVersion or AdditionalAssetsSchemaVersion;
+
public int SchemaVersion { get; init; } = CurrentSchemaVersion;
public string Engine { get; init; } = SupportedEngine;
public required string EngineVersion { get; init; }
@@ -181,6 +196,47 @@ public sealed record LocalAiInstallManifest
public required string ModelAlias { get; init; }
public required LocalAiAssetReceipt ModelAsset { get; init; }
///
+ /// Schema-5 receipts for additional model assets verified in the hub
+ /// cache alongside (the draft checkpoint), in
+ /// catalog order. Left at its unset default
+ /// (not .Empty) for schema-3/4 manifests, since
+ /// ImmutableArray<T>.Empty is a distinct, non-default instance
+ /// that would not omit
+ /// -- writing it would break older app builds' strict unknown-field
+ /// rejection on an otherwise-unchanged schema-4 receipt.
+ /// normalizes the
+ /// unset default to .Empty for every in-memory reader.
+ ///
+ [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingDefault)]
+ public ImmutableArray AdditionalModelAssets { get; init; }
+ ///
+ /// Verified hub-cache paths parallel to .
+ /// Same unset-default-not-Empty rule; see that property's remarks.
+ ///
+ [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingDefault)]
+ public ImmutableArray AdditionalModelPaths { get; init; }
+ ///
+ /// with the unset default collapsed to
+ /// an empty array. Read through this, never the raw property: a schema-3/4
+ /// or hand-edited manifest leaves the raw value at ImmutableArray's
+ /// default (null-backed) instance, where Length/IsEmpty throw
+ /// NullReferenceException instead of the intended InvalidDataException.
+ /// Normalizing on read rather than rewriting the record keeps the persisted
+ /// JSON byte-identical -- assigning .Empty back onto the manifest
+ /// would make the next save emit the field on an otherwise-untouched
+ /// schema-4 receipt.
+ ///
+ [JsonIgnore]
+ public ImmutableArray AdditionalModelAssetsOrEmpty =>
+ AdditionalModelAssets.IsDefault ? ImmutableArray.Empty : AdditionalModelAssets;
+ ///
+ /// with the unset default collapsed to
+ /// an empty array; see .
+ ///
+ [JsonIgnore]
+ public ImmutableArray AdditionalModelPathsOrEmpty =>
+ AdditionalModelPaths.IsDefault ? ImmutableArray.Empty : AdditionalModelPaths;
+ ///
/// The requested listener port. Zero delegates allocation to llama-server so
/// the child owns the port continuously from bind through startup.
///
@@ -512,7 +568,8 @@ public LocalAiResolvedInstall ResolveAndValidate(LocalAiInstallManifest manifest
ArgumentNullException.ThrowIfNull(manifest);
if (manifest.SchemaVersion is not (
LocalAiInstallManifest.CurrentSchemaVersion or
- LocalAiInstallManifest.HubCacheReceiptSchemaVersion))
+ LocalAiInstallManifest.HubCacheReceiptSchemaVersion or
+ LocalAiInstallManifest.AdditionalAssetsSchemaVersion))
{
throw new InvalidDataException($"Unsupported local AI manifest schema version {manifest.SchemaVersion}.");
}
@@ -566,10 +623,11 @@ LocalAiInstallManifest.CurrentSchemaVersion or
ValidateHubCacheReceipt(manifest, provenance);
string legacyModel = _paths.ResolveContainedPath(manifest.ModelPath, nameof(manifest.ModelPath));
ValidateModelPath(legacyModel, manifest.ModelAsset, "legacy-compatible");
- string model = manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion
+ string model = manifest.UsesHubCache
? ResolveHubCacheModelPath(manifest)
: legacyModel;
ValidateModelPath(model, manifest.ModelAsset, "active");
+ ValidateAdditionalAssets(manifest);
LocalAiPortPolicy.Validate(manifest.RequestedPort);
LocalAiGatewayModelPolicy.ValidateFallbackModel(manifest.GatewayFallbackModel);
@@ -746,6 +804,99 @@ private static void ValidateHubCacheReceipt(
private static string ResolveHubCacheModelPath(LocalAiInstallManifest manifest) =>
WindowsPathSafety.NormalizePath(manifest.CachedModelPath!);
+ ///
+ /// Validates schema-5 additional assets (a DFlash draft checkpoint). Each
+ /// receipt derives its own repository and
+ /// revision from its own SourceUrl -- unlike the primary
+ /// , additional assets are
+ /// not required to share the primary model's repository (the DFlash
+ /// draft checkpoint is pinned from a different one).
+ ///
+ private static void ValidateAdditionalAssets(LocalAiInstallManifest manifest)
+ {
+ if (manifest.SchemaVersion != LocalAiInstallManifest.AdditionalAssetsSchemaVersion)
+ {
+ if (!manifest.AdditionalModelAssets.IsDefaultOrEmpty || !manifest.AdditionalModelPaths.IsDefaultOrEmpty)
+ {
+ throw new InvalidDataException(
+ "Only schema-5 local AI manifests may record additional model assets.");
+ }
+ return;
+ }
+
+ if (manifest.AdditionalModelAssets.IsDefaultOrEmpty ||
+ manifest.AdditionalModelAssetsOrEmpty.Length != manifest.AdditionalModelPathsOrEmpty.Length)
+ {
+ throw new InvalidDataException(
+ "A schema-5 local AI manifest must record a matching additional-asset receipt and cache path pair.");
+ }
+
+ var seenFileNames = new HashSet(StringComparer.OrdinalIgnoreCase) { manifest.ModelAsset.FileName };
+ for (int i = 0; i < manifest.AdditionalModelAssetsOrEmpty.Length; i++)
+ {
+ LocalAiAssetReceipt receipt = manifest.AdditionalModelAssetsOrEmpty[i];
+ ValidateAssetReceipt(receipt, $"{nameof(manifest.AdditionalModelAssets)}[{i}]");
+ if (!seenFileNames.Add(receipt.FileName))
+ throw new InvalidDataException("The local AI manifest additional asset filenames must be unique.");
+
+ HuggingFaceModelProvenance provenance = ParseHuggingFaceProvenance(receipt);
+ if (!HuggingFaceHubCache.TryGetSnapshotPaths(
+ manifest.ModelCacheRoot!,
+ provenance.RepositoryId,
+ provenance.Revision,
+ provenance.RelativePath,
+ out string expectedPath,
+ out _,
+ out string error) ||
+ !string.Equals(manifest.AdditionalModelPathsOrEmpty[i], expectedPath, StringComparison.OrdinalIgnoreCase))
+ {
+ throw new InvalidDataException(
+ string.IsNullOrWhiteSpace(error)
+ ? "The local AI manifest additional asset cache receipt is invalid."
+ : error);
+ }
+ }
+ }
+
+ private static HuggingFaceModelProvenance ParseHuggingFaceProvenance(LocalAiAssetReceipt receipt)
+ {
+ var source = new Uri(receipt.SourceUrl, UriKind.Absolute);
+ if (!string.Equals(source.Scheme, Uri.UriSchemeHttps, StringComparison.OrdinalIgnoreCase) ||
+ !string.Equals(source.Host, "huggingface.co", StringComparison.OrdinalIgnoreCase) ||
+ source.Query is not ("" or "?download=true"))
+ {
+ throw new InvalidDataException("The local AI manifest asset source must be an immutable Hugging Face resolve URL.");
+ }
+
+ string[] pathSegments = Uri.UnescapeDataString(source.AbsolutePath).Split(
+ '/', StringSplitOptions.RemoveEmptyEntries);
+ int resolveIndex = Array.IndexOf(pathSegments, "resolve");
+ if (resolveIndex != 2 || pathSegments.Length < resolveIndex + 3)
+ {
+ throw new InvalidDataException("The local AI manifest asset source must be an immutable Hugging Face resolve URL.");
+ }
+
+ string repositoryId = $"{pathSegments[0]}/{pathSegments[1]}";
+ string revision = pathSegments[resolveIndex + 1];
+ if (revision.Length != 40 ||
+ revision.Any(character => character is not (>= '0' and <= '9' or >= 'a' and <= 'f')))
+ {
+ throw new InvalidDataException(
+ "The local AI manifest asset revision must be a lowercase 40-character commit digest.");
+ }
+
+ string[] relativeSegments = pathSegments[(resolveIndex + 2)..];
+ string relativePath = string.Join('/', relativeSegments);
+ if (relativeSegments.Any(segment => !WindowsPathSafety.IsSafeSegment(segment)) ||
+ !string.Equals(relativeSegments[^1], receipt.FileName, StringComparison.Ordinal))
+ {
+ throw new InvalidDataException(
+ "The local AI manifest asset source must match its own repository, revision, and filename.");
+ }
+
+ return new HuggingFaceModelProvenance(repositoryId, revision, relativePath);
+ }
+
internal sealed record HuggingFaceModelProvenance(
string RepositoryId,
string Revision,
diff --git a/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs b/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs
index 26855297d..59f8cf1cf 100644
--- a/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs
+++ b/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs
@@ -330,9 +330,18 @@ private async Task InitializeLocalAiReviewAsync(
? deviceEligibility.Plan?.Model.Id
: null;
- if (!deviceEligibility.CanInstall || deviceEligibility.Plan is null || deviceEligibility.SelectedGpu is null)
+ // A SKU with no recommended default (RTX Spark 32 GB) still runs an already
+ // configured model. Gate availability on that configured selection when there is
+ // one, so rerunning setup does not switch Local AI off on a working machine.
+ // _localAiRecommendedModelId stays null so nothing is labelled Recommended.
+ LocalInferenceEligibilityResult availability =
+ LocalInferenceEligibility.EvaluateForConfiguredAvailability(
+ _localAiHardware,
+ _config!.LocalAi.SelectedModelId);
+
+ if (!availability.CanInstall || availability.Plan is null || availability.SelectedGpu is null)
{
- hardwareReason = DescribeLocalAiUnavailable(deviceEligibility);
+ hardwareReason = DescribeLocalAiUnavailable(availability);
}
else
{
@@ -343,7 +352,7 @@ private async Task InitializeLocalAiReviewAsync(
// model instead of leaving setup stuck on a known-incompatible selection. A
// merely busy GPU (EligibleButBusy) is not reconciled away: the same model would
// still work once the GPU frees up, and CanInstall already covers that case.
- if (_config!.LocalAi.SelectedModelId is { } selectedModelId)
+ if (_config.LocalAi.SelectedModelId is { } selectedModelId)
{
LocalInferenceEligibilityResult selectedEligibility =
LocalInferenceEligibility.Evaluate(_localAiHardware, selectedModelId);
@@ -356,7 +365,7 @@ private async Task InitializeLocalAiReviewAsync(
_config.LocalAi.SelectedModelId = null;
}
}
- _config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? deviceEligibility.Plan.Model.Id;
+ _config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? availability.Plan.Model.Id;
eligibility ??= LocalInferenceEligibility.Evaluate(
_localAiHardware,
@@ -670,7 +679,7 @@ private void PopulateLocalAiModels()
LocalAiModelSelector.Items.Add(new ComboBoxItem
{
Content = $"{SetupReviewSummaryBuilder.DisplayModelName(model)} " +
- $"({FormatSize(model.Weights.SizeBytes)}, " +
+ $"({FormatSize(LocalModelCatalog.TotalDownloadSizeBytes(model))}, " +
$"{FormatContext(plan.Profile.ContextTokens)}, " +
$"{LocalModelCatalog.ToDisplayCacheType(plan.Profile.KeyCachePrecision)} KV)" +
(isRecommended ? " (Recommended)" : string.Empty),
@@ -780,7 +789,7 @@ private void UpdateLocalAiModelDetails()
"loads on first request";
LocalAiModelDetailText.Text =
$"{SetupReviewSummaryBuilder.DisplayModelName(plan.Model)}, " +
- $"{FormatSize(plan.Model.Weights.SizeBytes)} from Hugging Face";
+ $"{FormatSize(LocalModelCatalog.TotalDownloadSizeBytes(plan.Model))} from Hugging Face";
UpdatePrimaryButtonState();
}
diff --git a/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.AdditionalAssets.cs b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.AdditionalAssets.cs
new file mode 100644
index 000000000..060e34dde
--- /dev/null
+++ b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.AdditionalAssets.cs
@@ -0,0 +1,200 @@
+using OpenClaw.Connection.LocalAi;
+using OpenClaw.Shared.Inference.Catalog;
+
+namespace OpenClaw.SetupEngine;
+
+///
+/// Acquisition for additional model assets (a DFlash draft checkpoint)
+/// verified into the same standard Hugging Face hub cache uses for the
+/// primary weights. Reuses the same resumable download, hash verification,
+/// and safe-cache-directory promotion helpers; the one thing it deliberately
+/// omits is the legacy app-owned compatibility copy, since every recipe that
+/// pins an additional asset is new -- there is no pre-hub-cache install of
+/// it to stay compatible with.
+///
+internal sealed partial class HuggingFaceModelInstaller
+{
+ public async Task InstallAdditionalAssetAsync(
+ string localDataDirectory,
+ PinnedArtifact artifact,
+ IProgress? progress,
+ CancellationToken cancellationToken)
+ {
+ ArgumentException.ThrowIfNullOrWhiteSpace(localDataDirectory);
+ ArgumentNullException.ThrowIfNull(artifact);
+ if (artifact.Role != ArtifactRole.ModelWeights || artifact.Source is not HuggingFaceRevisionSource source)
+ {
+ throw new HuggingFaceModelInstallException(
+ "An additional Local AI model asset must be an immutable Hugging Face weights artifact.");
+ }
+
+ string cacheRoot = _cacheRootResolver();
+ if (!TryValidateCacheRootOwnershipBoundary(localDataDirectory, cacheRoot, out string cacheRootError))
+ throw new HuggingFaceModelInstallException(cacheRootError);
+ if (!HuggingFaceHubCache.TryGetSnapshotPaths(
+ cacheRoot,
+ source.RepositoryId,
+ source.RevisionSha,
+ artifact.RelativePath,
+ out string modelPath,
+ out string partialPath,
+ out string pathError))
+ {
+ throw new HuggingFaceModelInstallException(pathError);
+ }
+
+ if (Directory.Exists(modelPath))
+ throw new HuggingFaceModelInstallException("The managed Local AI model path is an existing directory.");
+ if (Directory.Exists(partialPath))
+ throw new HuggingFaceModelInstallException("The managed Local AI partial model path is an existing directory.");
+
+ await using (FileStream? verified =
+ await HuggingFaceHubCache.TryOpenVerifiedCacheFileAsync(
+ cacheRoot,
+ modelPath,
+ artifact.SizeBytes,
+ artifact.Sha256,
+ new VerificationProgress(this, progress, artifact.SizeBytes),
+ cancellationToken)
+ .ConfigureAwait(false))
+ {
+ if (verified is not null)
+ {
+ return new HuggingFaceAdditionalAssetInstallResult(
+ modelPath, cacheRoot, HuggingFaceModelInstallDisposition.ReusedVerified, CreatedThisRun: false);
+ }
+ }
+
+ if (PathEntryExists(modelPath))
+ {
+ throw new HuggingFaceModelInstallException(
+ $"The Hugging Face cache destination '{modelPath}' is unsafe or does not match " +
+ "the pinned model. Remove it manually and retry setup.");
+ }
+
+ string destinationDirectory = Path.GetDirectoryName(modelPath)
+ ?? throw new HuggingFaceModelInstallException(
+ "The Hugging Face cache destination has no parent directory.");
+ string destinationFileName = Path.GetFileName(modelPath);
+ string partialFileName = Path.GetFileName(partialPath);
+ LocalAiManifestMigration.SafeCacheDirectory? directory = null;
+ LocalAiManifestMigration.CacheMigrationFile? partial = null;
+ bool partialExistedBeforeInstall = false;
+ HuggingFaceModelInstallDisposition disposition = HuggingFaceModelInstallDisposition.Downloaded;
+ try
+ {
+ directory = LocalAiManifestMigration.SafeCacheDirectory.OpenOrCreate(cacheRoot, destinationDirectory);
+ partial = directory.TryOpenExisting(partialFileName);
+ partialExistedBeforeInstall = partial is not null;
+
+ bool partialVerified = partial is not null &&
+ partial.Stream.Length == artifact.SizeBytes &&
+ await VerifyOpenFileAsync(partial.Stream, artifact, progress, cancellationToken).ConfigureAwait(false);
+ if (partial is not null && partial.Stream.Length >= artifact.SizeBytes && !partialVerified)
+ {
+ partial.Stream.SetLength(0);
+ partial.Stream.Position = 0;
+ }
+
+ if (!partialVerified)
+ {
+ partial ??= directory.CreateNew(partialFileName);
+ if (partial.Stream.Length == 0 &&
+ await TryCopyVerifiedBlobAsync(
+ cacheRoot, source.RepositoryId, artifact, partial.Stream, progress, cancellationToken)
+ .ConfigureAwait(false))
+ {
+ disposition = HuggingFaceModelInstallDisposition.ReusedVerified;
+ }
+ else
+ {
+ await DownloadAndVerifyAsync(artifact, partial.Stream, progress, cancellationToken)
+ .ConfigureAwait(false);
+ }
+ }
+
+ cancellationToken.ThrowIfCancellationRequested();
+ LocalAiManifestMigration.CacheMigrationFile activePartial = partial
+ ?? throw new HuggingFaceModelInstallException("The Hugging Face cache partial was not created.");
+ if (!HuggingFaceHubCache.TryGetSnapshotPaths(
+ cacheRoot,
+ source.RepositoryId,
+ source.RevisionSha,
+ artifact.RelativePath,
+ out string revalidatedModelPath,
+ out string revalidatedPartialPath,
+ out pathError) ||
+ !string.Equals(modelPath, revalidatedModelPath, StringComparison.OrdinalIgnoreCase) ||
+ !string.Equals(partialPath, revalidatedPartialPath, StringComparison.OrdinalIgnoreCase))
+ {
+ throw new HuggingFaceModelInstallException(
+ string.IsNullOrWhiteSpace(pathError)
+ ? "The Local AI model paths changed before promotion."
+ : pathError);
+ }
+
+ if (!await VerifyOpenFileAsync(activePartial.Stream, artifact, progress, cancellationToken)
+ .ConfigureAwait(false))
+ {
+ throw new HuggingFaceModelInstallException(
+ "The Hugging Face cache partial does not match the pinned model.");
+ }
+
+ try
+ {
+ activePartial.Promote(destinationFileName);
+ }
+ catch (IOException ex)
+ {
+ if (partialExistedBeforeInstall)
+ activePartial.Commit();
+ activePartial.Dispose();
+ partial = null;
+ await using FileStream? winner =
+ await HuggingFaceHubCache.TryOpenVerifiedCacheFileAsync(
+ cacheRoot,
+ modelPath,
+ artifact.SizeBytes,
+ artifact.Sha256,
+ new VerificationProgress(this, progress, artifact.SizeBytes),
+ cancellationToken)
+ .ConfigureAwait(false);
+ if (winner is null)
+ {
+ throw new HuggingFaceModelInstallException(
+ "The Hugging Face cache destination changed before promotion.", ex);
+ }
+
+ return new HuggingFaceAdditionalAssetInstallResult(
+ modelPath, cacheRoot, HuggingFaceModelInstallDisposition.ReusedVerified, CreatedThisRun: false);
+ }
+
+ if (partialExistedBeforeInstall)
+ activePartial.Commit();
+ directory.RequirePromotedFile(activePartial.Stream.SafeFileHandle, destinationFileName);
+ activePartial.Commit();
+ return new HuggingFaceAdditionalAssetInstallResult(modelPath, cacheRoot, disposition, CreatedThisRun: true);
+ }
+ catch (OperationCanceledException)
+ {
+ partial?.Commit();
+ throw;
+ }
+ catch (Exception exception) when (
+ exception is IOException or HttpRequestException or TransientHuggingFaceModelInstallException)
+ {
+ partial?.Commit();
+ throw;
+ }
+ catch (HuggingFaceModelInstallException) when (partialExistedBeforeInstall)
+ {
+ partial?.Commit();
+ throw;
+ }
+ finally
+ {
+ partial?.Dispose();
+ directory?.Dispose();
+ }
+ }
+}
diff --git a/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs
index 1e5d53fee..080e8ed22 100644
--- a/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs
+++ b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs
@@ -57,6 +57,13 @@ public TransientHuggingFaceModelInstallException(string message)
}
}
+/// A verified additional model asset (DFlash draft checkpoint).
+internal sealed record HuggingFaceAdditionalAssetInstallResult(
+ string ModelPath,
+ string CacheRoot,
+ HuggingFaceModelInstallDisposition Disposition,
+ bool CreatedThisRun);
+
internal interface IHuggingFaceModelAcquirer
{
Task InstallAsync(
@@ -66,6 +73,20 @@ Task InstallAsync(
IProgress? progress,
CancellationToken cancellationToken);
+ ///
+ /// Verifies or acquires one additional model asset (a DFlash draft
+ /// checkpoint) into the same hub cache
+ /// uses for the primary weights. Unlike the
+ /// primary weights, additional assets have no legacy app-owned
+ /// compatibility copy -- every recipe that uses one is new since the hub
+ /// cache became the primary store.
+ ///
+ Task InstallAdditionalAssetAsync(
+ string localDataDirectory,
+ PinnedArtifact artifact,
+ IProgress? progress,
+ CancellationToken cancellationToken);
+
void RemoveInstalledModel(string localDataDirectory, HuggingFaceModelInstallResult install);
void RemovePartialModel(
@@ -80,7 +101,7 @@ void RemovePartialModel(
/// left by process termination is resumed with an HTTP range request. Shared
/// cache artifacts and resumable partials survive rollback and cancellation.
///
-internal sealed class HuggingFaceModelInstaller : IHuggingFaceModelAcquirer
+internal sealed partial class HuggingFaceModelInstaller : IHuggingFaceModelAcquirer
{
private const int BufferSize = 1024 * 1024;
private const int ProgressIntervalBytes = 4 * 1024 * 1024;
diff --git a/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs b/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs
index f991f4e76..e9fb5b693 100644
--- a/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs
+++ b/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs
@@ -130,7 +130,7 @@ public async Task InstallAsync(
internal static LocalAiComponentIdentity Component(LlamaRuntimeVariant runtime) =>
new(
"llama-server",
- LlamaRuntimeCatalog.ReleaseTag,
+ runtime.ReleaseTag,
runtime.Architecture switch
{
Architecture.X64 => "win-x64",
diff --git a/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs b/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs
index d9b16207e..6316da3d4 100644
--- a/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs
+++ b/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs
@@ -3,6 +3,7 @@
using System.Text.RegularExpressions;
using OpenClaw.Connection.LocalAi;
using OpenClaw.Shared.Inference;
+using OpenClaw.Shared.Inference.Catalog;
namespace OpenClaw.SetupEngine;
@@ -297,7 +298,9 @@ ctx.LocalAiInferenceVerification is null ||
throw new InvalidDataException(
"llama-server loaded CUDA from outside the managed runtime directory.");
}
- long minimumDelta = Math.Max(512L * 1024 * 1024, plan.Model.Weights.SizeBytes / 2);
+ long minimumDelta = Math.Max(
+ 512L * 1024 * 1024,
+ LocalModelCatalog.TotalDownloadSizeBytes(plan.Model) / 2);
if (!HasRequiredGpuLoadEvidence(evidence, minimumDelta))
{
throw new InvalidDataException(
diff --git a/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs b/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs
index ebcafdc94..afea2bcc5 100644
--- a/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs
+++ b/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs
@@ -1,3 +1,4 @@
+using System.Collections.Immutable;
using OpenClaw.Connection.LocalAi;
using OpenClaw.Shared.Inference.Catalog;
@@ -8,7 +9,8 @@ internal sealed record LocalAiReconcileResult(
LocalAiResolvedInstall? ResolvedInstall,
LlamaRuntimeInstallResult? RuntimeInstall,
HuggingFaceModelInstallResult? ModelInstall,
- LocalAiResolvedInstall? OriginalInstall = null)
+ LocalAiResolvedInstall? OriginalInstall = null,
+ ImmutableArray? AdditionalModelInstalls = null)
{
public static LocalAiReconcileResult NotInstalled { get; } =
new(false, null, null, null);
@@ -26,6 +28,18 @@ Task VerifyLegacyCompatibilityAsync(
LocalAiPaths paths,
PinnedArtifact artifact,
CancellationToken cancellationToken);
+
+ ///
+ /// Verifies one schema-5 additional model asset (a DFlash draft checkpoint)
+ /// still matches its pinned receipt in the
+ /// shared hub cache. Additional assets have no legacy app-owned copy, so
+ /// unlike there is no separate schema-3 path.
+ ///
+ Task VerifyAdditionalAssetAsync(
+ LocalAiResolvedInstall install,
+ string cachedAssetPath,
+ PinnedArtifact artifact,
+ CancellationToken cancellationToken);
}
internal sealed class LocalAiModelFileVerifier : ILocalAiModelFileVerifier
@@ -35,7 +49,7 @@ public async Task VerifyActiveAsync(
PinnedArtifact artifact,
CancellationToken cancellationToken)
{
- if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion)
+ if (!install.Manifest.UsesHubCache)
{
if (!File.Exists(install.ModelPath))
return false;
@@ -62,7 +76,7 @@ public Task VerifyLegacyCompatibilityAsync(
PinnedArtifact artifact,
CancellationToken cancellationToken)
{
- if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion)
+ if (!install.Manifest.UsesHubCache)
return Task.FromResult(true);
string legacyModelPath = paths.ResolveContainedPath(
@@ -75,6 +89,24 @@ public Task VerifyLegacyCompatibilityAsync(
artifact,
cancellationToken);
}
+
+ public async Task VerifyAdditionalAssetAsync(
+ LocalAiResolvedInstall install,
+ string cachedAssetPath,
+ PinnedArtifact artifact,
+ CancellationToken cancellationToken)
+ {
+ await using FileStream? verified =
+ await HuggingFaceHubCache.TryOpenVerifiedCacheFileAsync(
+ install.Manifest.ModelCacheRoot!,
+ cachedAssetPath,
+ artifact.SizeBytes,
+ artifact.Sha256,
+ progress: null,
+ cancellationToken)
+ .ConfigureAwait(false);
+ return verified is not null;
+ }
}
///
@@ -126,7 +158,7 @@ public async Task ReconcileAsync(
if (install is null)
return LocalAiReconcileResult.NotInstalled;
LocalAiResolvedInstall originalInstall = install;
- ValidateRecipeMatch(install, plan, selectedGpuId, localDataDirectory);
+ bool runtimeUpgradePending = ValidateRecipeMatch(install, plan, selectedGpuId, localDataDirectory);
bool migrateLegacyGpuId =
!string.Equals(install.Manifest.SelectedGpuId, selectedGpuId, StringComparison.Ordinal) &&
@@ -145,7 +177,40 @@ public async Task ReconcileAsync(
plan.Model.Weights,
cancellationToken)
.ConfigureAwait(false);
- bool modelIsValid = activeModelIsValid && legacyModelIsValid;
+ bool additionalAssetsAreValid = activeModelIsValid && await VerifyAdditionalAssetsAsync(
+ install,
+ plan.Model,
+ cancellationToken)
+ .ConfigureAwait(false);
+ bool modelIsValid = activeModelIsValid && legacyModelIsValid && additionalAssetsAreValid;
+ if (runtimeUpgradePending)
+ {
+ if (modelIsValid)
+ {
+ install = await MigrateLegacyModelAsync(
+ install,
+ paths,
+ localDataDirectory,
+ plan,
+ selectedGpuId,
+ migrationProgress,
+ cancellationToken)
+ .ConfigureAwait(false);
+ }
+
+ // The catalog moved to a newer pinned runtime. Drop only the runtime so the
+ // acquirer installs the new one, and keep the verified model and its extra
+ // assets so an upgrade does not re-download tens of GB. OriginalInstall lets
+ // the manifest step replace the existing receipt in place.
+ return new LocalAiReconcileResult(
+ Reused: false,
+ ResolvedInstall: null,
+ RuntimeInstall: null,
+ ModelInstall: modelIsValid ? CreateModelInstall(install, localDataDirectory) : null,
+ OriginalInstall: originalInstall,
+ AdditionalModelInstalls: modelIsValid ? CreateAdditionalModelInstalls(install) : null);
+ }
+
if (!inspection.IsValid || !modelIsValid)
{
if (!allowIncompleteInstallation)
@@ -173,7 +238,8 @@ public async Task ReconcileAsync(
ResolvedInstall: null,
RuntimeInstall: inspection.IsValid ? CreateRuntimeInstall(install) : null,
ModelInstall: modelIsValid ? CreateModelInstall(install, localDataDirectory) : null,
- OriginalInstall: originalInstall);
+ OriginalInstall: originalInstall,
+ AdditionalModelInstalls: modelIsValid ? CreateAdditionalModelInstalls(install) : null);
}
install = await MigrateLegacyModelAsync(
@@ -201,7 +267,8 @@ public async Task ReconcileAsync(
install,
CreateRuntimeInstall(install),
CreateModelInstall(install, localDataDirectory),
- OriginalInstall: allowIncompleteInstallation ? originalInstall : null);
+ OriginalInstall: allowIncompleteInstallation ? originalInstall : null,
+ AdditionalModelInstalls: CreateAdditionalModelInstalls(install));
}
private static LlamaRuntimeInstallResult CreateRuntimeInstall(LocalAiResolvedInstall install) =>
@@ -222,7 +289,7 @@ private static HuggingFaceModelInstallResult CreateModelInstall(
LocalAiResolvedInstall install,
string localDataDirectory)
{
- if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion)
+ if (!install.Manifest.UsesHubCache)
{
return new HuggingFaceModelInstallResult(
install.ModelPath,
@@ -244,6 +311,60 @@ private static HuggingFaceModelInstallResult CreateModelInstall(
LegacyCreatedThisRun: false);
}
+ ///
+ /// Reconstructs the additional-asset install results a reused install's
+ /// manifest already proves verified, so a caller that skips re-acquiring
+ /// them (because just verified them) still has
+ /// a populated SetupContext.LocalAiAdditionalModelInstalls to persist.
+ ///
+ private static ImmutableArray CreateAdditionalModelInstalls(
+ LocalAiResolvedInstall install)
+ {
+ if (install.Manifest.AdditionalModelPaths.IsDefaultOrEmpty)
+ return ImmutableArray.Empty;
+
+ var builder = ImmutableArray.CreateBuilder(
+ install.Manifest.AdditionalModelPathsOrEmpty.Length);
+ foreach (string cachedAssetPath in install.Manifest.AdditionalModelPathsOrEmpty)
+ {
+ builder.Add(new HuggingFaceAdditionalAssetInstallResult(
+ cachedAssetPath,
+ install.Manifest.ModelCacheRoot!,
+ HuggingFaceModelInstallDisposition.ReusedVerified,
+ CreatedThisRun: false));
+ }
+
+ return builder.MoveToImmutable();
+ }
+
+ private async Task VerifyAdditionalAssetsAsync(
+ LocalAiResolvedInstall install,
+ LocalModelInfo model,
+ CancellationToken cancellationToken)
+ {
+ ImmutableArray expected = LocalModelCatalog.AdditionalArtifacts(model);
+ if (expected.IsEmpty)
+ return install.Manifest.AdditionalModelAssetsOrEmpty.IsEmpty;
+ if (install.Manifest.AdditionalModelPathsOrEmpty.Length != expected.Length)
+ return false;
+
+ for (int i = 0; i < expected.Length; i++)
+ {
+ if (!await _modelVerifier
+ .VerifyAdditionalAssetAsync(
+ install,
+ install.Manifest.AdditionalModelPathsOrEmpty[i],
+ expected[i],
+ cancellationToken)
+ .ConfigureAwait(false))
+ {
+ return false;
+ }
+ }
+
+ return true;
+ }
+
private async Task MigrateLegacyModelAsync(
LocalAiResolvedInstall install,
LocalAiPaths paths,
@@ -271,26 +392,36 @@ private async Task MigrateLegacyModelAsync(
.ConfigureAwait(false)
?? throw new InvalidDataException(
"The Local AI installation manifest disappeared during cache migration.");
- ValidateRecipeMatch(migrated, plan, selectedGpuId, localDataDirectory);
+ _ = ValidateRecipeMatch(migrated, plan, selectedGpuId, localDataDirectory);
return migrated;
}
- private static void ValidateRecipeMatch(
+ ///
+ /// Validates the receipt against the runtime and recipe it actually recorded, and
+ /// reports whether the catalog has since moved to a newer pinned runtime. A version
+ /// difference is an expected upgrade, not a corrupt install, so it must not throw:
+ /// throwing here ends setup with an uninstall instruction instead of upgrading.
+ ///
+ private static bool ValidateRecipeMatch(
LocalAiResolvedInstall install,
LocalInferencePlan plan,
string selectedGpuId,
string localDataDirectory)
{
LocalAiInstallManifest manifest = install.Manifest;
+ LlamaRuntimeVariant installedRuntime =
+ LlamaRuntimeCatalog.FindInstalled(manifest.RuntimeId) ?? plan.Runtime;
+ bool runtimeUpgradePending =
+ !string.Equals(installedRuntime.Id, plan.Runtime.Id, StringComparison.Ordinal);
string expectedArchitecture = plan.Runtime.Architecture switch
{
System.Runtime.InteropServices.Architecture.X64 => "x64",
System.Runtime.InteropServices.Architecture.Arm64 => "arm64",
_ => throw new InvalidDataException("The selected Local AI runtime architecture is unsupported."),
};
- if (!string.Equals(manifest.EngineVersion, LlamaRuntimeCatalog.ReleaseTag, StringComparison.Ordinal) ||
+ if (!string.Equals(manifest.EngineVersion, installedRuntime.ReleaseTag, StringComparison.Ordinal) ||
!string.Equals(manifest.Architecture, expectedArchitecture, StringComparison.Ordinal) ||
- !string.Equals(manifest.RuntimeId, plan.Runtime.Id, StringComparison.Ordinal) ||
+ !string.Equals(manifest.RuntimeId, installedRuntime.Id, StringComparison.Ordinal) ||
!string.Equals(manifest.ModelCatalogId, plan.Model.Id, StringComparison.Ordinal) ||
manifest.ContextLength != plan.Profile.ContextTokens ||
manifest.KeyCachePrecision != plan.Profile.KeyCachePrecision ||
@@ -303,7 +434,7 @@ private static void ValidateRecipeMatch(
"The existing managed Local AI installation does not match the selected runtime, GPU, and model recipe.");
}
- LocalAiComponentIdentity component = LlamaRuntimeInstaller.Component(plan.Runtime);
+ LocalAiComponentIdentity component = LlamaRuntimeInstaller.Component(installedRuntime);
if (!LocalAiPathPolicy.TryResolve(
localDataDirectory,
component,
@@ -327,11 +458,11 @@ private static void ValidateRecipeMatch(
}
LlamaServerRouterConfiguration.ValidateArtifactReceipts(
manifest,
- plan.Runtime,
+ installedRuntime,
plan.Model);
bool modelPathMatches;
- if (manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion)
+ if (manifest.UsesHubCache)
{
modelPathMatches =
!string.IsNullOrWhiteSpace(manifest.ModelCacheRoot) &&
@@ -372,6 +503,8 @@ private static void ValidateRecipeMatch(
? "The managed model path does not match the selected catalog recipe."
: error);
}
+
+ return runtimeUpgradePending;
}
private static bool GpuIdsMatch(string persistedGpuId, string selectedGpuId) =>
diff --git a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs
index fb0b437a8..085e14040 100644
--- a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs
+++ b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs
@@ -287,8 +287,15 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati
}
if (!result.Reused)
{
+ // Normal upgrades restore their receipt independently of the gateway
+ // recovery pipeline's endpoint-health and provider rollback guards.
+ if (ctx.LocalAiRecoveryOriginalInstall is null &&
+ result.OriginalInstall is { } retainedReceipt)
+ ctx.LocalAiUpgradeOriginalInstall ??= retainedReceipt;
ctx.LocalAiRuntimeInstall = result.RuntimeInstall;
ctx.LocalAiModelInstall = result.ModelInstall;
+ ctx.LocalAiAdditionalModelInstalls = result.AdditionalModelInstalls
+ ?? ImmutableArray.Empty;
return StepResult.Skip(result.OriginalInstall is null
? "No completed managed Local AI installation was found."
: "The existing Local AI receipt was retained while incomplete assets are repaired.");
@@ -297,6 +304,8 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati
ctx.LocalAiResolvedInstall = result.ResolvedInstall;
ctx.LocalAiRuntimeInstall = result.RuntimeInstall;
ctx.LocalAiModelInstall = result.ModelInstall;
+ ctx.LocalAiAdditionalModelInstalls = result.AdditionalModelInstalls
+ ?? ImmutableArray.Empty;
ctx.LocalAiPort = result.ResolvedInstall!.Manifest.RequestedPort;
return StepResult.Ok("Reused the verified managed Local AI installation.");
}
@@ -312,6 +321,11 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati
ex);
}
}
+
+ public override Task RollbackAsync(SetupContext ctx, CancellationToken ct) =>
+ ctx.IsUninstalling
+ ? Task.CompletedTask
+ : PersistLocalAiManifestStep.RestoreUpgradeReceiptAsync(ctx, ct);
}
/// Installs the two pinned llama.cpp runtime archives as one atomic component.
@@ -382,7 +396,7 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati
progress,
linked.Token);
ctx.LocalAiRuntimeInstall = install;
- return StepResult.Ok($"Installed llama-server {LlamaRuntimeCatalog.ReleaseTag}.");
+ return StepResult.Ok($"Installed llama-server {plan.Runtime.ReleaseTag}.");
}
catch (OperationCanceledException) when (ct.IsCancellationRequested)
{
@@ -437,6 +451,16 @@ public AcquireLocalAiModelStep()
internal AcquireLocalAiModelStep(IHuggingFaceModelAcquirer acquirer) =>
_acquirer = acquirer ?? throw new ArgumentNullException(nameof(acquirer));
+ ///
+ /// Additional artifacts a recipe needs beyond its primary weights, in the
+ /// fixed catalog order
+ /// defines. relies on that same
+ /// ordering to tell the draft checkpoint apart from a shard without a
+ /// separate "kind" tag on the receipt.
+ ///
+ internal static ImmutableArray AdditionalArtifacts(LocalModelInfo model) =>
+ LocalModelCatalog.AdditionalArtifacts(model);
+
public override string Id => "acquire-local-ai-model";
public override string DisplayName => "Downloading Local AI model from Hugging Face";
public override bool CanRetry => false;
@@ -476,6 +500,29 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati
progress,
linked.Token);
ctx.LocalAiModelInstall = install;
+
+ ImmutableArray additionalArtifacts = AdditionalArtifacts(plan.Model);
+ var additionalInstalls = ImmutableArray.CreateBuilder(
+ additionalArtifacts.Length);
+ foreach (PinnedArtifact artifact in additionalArtifacts)
+ {
+ var artifactProgress = new SynchronousProgress(value =>
+ ctx.DetailProgress?.Report(new SetupDetailProgressEvent(
+ Id,
+ value.Phase == HuggingFaceModelInstallPhase.Verifying
+ ? $"Verifying {artifact.RelativePath}"
+ : $"Downloading {artifact.RelativePath}",
+ value.CompletedBytes,
+ value.TotalBytes,
+ SetupDetailProgressUnit.Bytes)));
+ additionalInstalls.Add(await _acquirer.InstallAdditionalAssetAsync(
+ ctx.LocalDataDir,
+ artifact,
+ artifactProgress,
+ linked.Token));
+ }
+ ctx.LocalAiAdditionalModelInstalls = additionalInstalls.MoveToImmutable();
+
string action = install.Disposition == HuggingFaceModelInstallDisposition.ReusedVerified
? "Verified existing"
: "Downloaded";
@@ -508,6 +555,9 @@ public override Task RollbackAsync(SetupContext ctx, CancellationToken ct)
_acquirer.RemoveInstalledModel(ctx.LocalDataDir, install);
ctx.LocalAiModelInstall = null;
}
+ // Additional assets have no legacy app-owned copy to remove; their hub-cache
+ // artifacts survive rollback the same way the primary weights' do.
+ ctx.LocalAiAdditionalModelInstalls = ImmutableArray.Empty;
if (ctx.LocalAiEligibility?.Plan is { } plan)
{
_acquirer.RemovePartialModel(
@@ -554,10 +604,12 @@ ctx.LocalAiModelInstall is not
return StepResult.Terminal(portError ?? "The requested Local AI port is invalid.");
var paths = new LocalAiPaths(ctx.LocalDataDir);
- bool replacesRecoveryReceipt =
- ctx.LocalAiRecoveryOriginalInstall is not null &&
+ LocalAiResolvedInstall? originalInstall =
+ ctx.LocalAiRecoveryOriginalInstall ?? ctx.LocalAiUpgradeOriginalInstall;
+ bool replacesExistingReceipt =
+ originalInstall is not null &&
File.Exists(paths.ManifestPath);
- if (File.Exists(paths.ManifestPath) && !replacesRecoveryReceipt)
+ if (File.Exists(paths.ManifestPath) && !replacesExistingReceipt)
return StepResult.Terminal("A managed Local AI installation receipt already exists.");
LocalAiComponentIdentity component = LlamaRuntimeInstaller.Component(plan.Runtime);
if (!LocalAiPathPolicy.TryResolve(
@@ -598,10 +650,34 @@ ctx.LocalAiRecoveryOriginalInstall is not null &&
return StepResult.Terminal(ex.Message, ex);
}
+ ImmutableArray additionalArtifacts = AcquireLocalAiModelStep.AdditionalArtifacts(plan.Model);
+ if (additionalArtifacts.Length != ctx.LocalAiAdditionalModelInstalls.Length)
+ {
+ return StepResult.Terminal(
+ "The Local AI installation receipt requires a completed additional-asset acquisition step.");
+ }
+
+ var additionalModelAssets = ImmutableArray.CreateBuilder(additionalArtifacts.Length);
+ var additionalModelPaths = ImmutableArray.CreateBuilder(additionalArtifacts.Length);
+ for (int i = 0; i < additionalArtifacts.Length; i++)
+ {
+ PinnedArtifact artifact = additionalArtifacts[i];
+ additionalModelAssets.Add(new LocalAiAssetReceipt
+ {
+ FileName = Path.GetFileName(artifact.RelativePath),
+ SourceUrl = artifact.DownloadUri.AbsoluteUri,
+ SizeBytes = artifact.SizeBytes,
+ Sha256 = artifact.Sha256.Value,
+ });
+ additionalModelPaths.Add(ctx.LocalAiAdditionalModelInstalls[i].ModelPath);
+ }
+
LocalAiInstallManifest manifest = new()
{
- SchemaVersion = LocalAiInstallManifest.HubCacheReceiptSchemaVersion,
- EngineVersion = LlamaRuntimeCatalog.ReleaseTag,
+ SchemaVersion = additionalArtifacts.IsEmpty
+ ? LocalAiInstallManifest.HubCacheReceiptSchemaVersion
+ : LocalAiInstallManifest.AdditionalAssetsSchemaVersion,
+ EngineVersion = plan.Runtime.ReleaseTag,
Architecture = plan.Runtime.Architecture switch
{
Architecture.X64 => "x64",
@@ -625,6 +701,11 @@ ctx.LocalAiRecoveryOriginalInstall is not null &&
SizeBytes = plan.Model.Weights.SizeBytes,
Sha256 = plan.Model.Weights.Sha256.Value,
},
+ // Leave these at their unset default (not an explicitly-built empty
+ // array) when there is nothing to add, so schema-4 manifests omit
+ // them from JSON entirely -- see the properties' remarks.
+ AdditionalModelAssets = additionalArtifacts.IsEmpty ? default : additionalModelAssets.MoveToImmutable(),
+ AdditionalModelPaths = additionalArtifacts.IsEmpty ? default : additionalModelPaths.MoveToImmutable(),
RequestedPort = requestedPort,
Endpoint = null,
ContextLength = plan.Profile.ContextTokens,
@@ -633,7 +714,7 @@ ctx.LocalAiRecoveryOriginalInstall is not null &&
DraftKeyCachePrecision = plan.Profile.DraftKeyCachePrecision,
DraftValueCachePrecision = plan.Profile.DraftValueCachePrecision,
};
- if (ctx.LocalAiRecoveryOriginalInstall is { } originalInstall)
+ if (originalInstall is not null)
{
manifest = originalInstall.Manifest with
{
@@ -651,6 +732,8 @@ ctx.LocalAiRecoveryOriginalInstall is not null &&
ModelId = manifest.ModelId,
ModelAlias = manifest.ModelAlias,
ModelAsset = manifest.ModelAsset,
+ AdditionalModelAssets = manifest.AdditionalModelAssets,
+ AdditionalModelPaths = manifest.AdditionalModelPaths,
RequestedPort = manifest.RequestedPort,
Endpoint = null,
ContextLength = manifest.ContextLength,
@@ -666,7 +749,7 @@ ctx.LocalAiRecoveryOriginalInstall is not null &&
{
await store.SaveAsync(manifest, ct);
ctx.LocalAiResolvedInstall = store.ResolveAndValidate(manifest);
- ctx.LocalAiManifestCreatedThisRun = !replacesRecoveryReceipt;
+ ctx.LocalAiManifestCreatedThisRun = !replacesExistingReceipt;
return StepResult.Ok("Recorded the verified llama-server and Hugging Face installation.");
}
catch (Exception ex) when (ex is IOException or UnauthorizedAccessException or InvalidDataException)
@@ -695,10 +778,17 @@ public override async Task RollbackAsync(SetupContext ctx, CancellationToken ct)
ctx.LocalAiRuntimeInstall = null;
ctx.LocalAiModelInstall = null;
ctx.LocalAiResolvedInstall = null;
+ ctx.LocalAiUpgradeOriginalInstall = null;
ctx.LocalAiManifestCreatedThisRun = false;
return;
}
+ if (ctx.LocalAiUpgradeOriginalInstall is not null)
+ {
+ await RestoreUpgradeReceiptAsync(ctx, ct);
+ return;
+ }
+
if (!ctx.LocalAiManifestCreatedThisRun)
return;
@@ -710,6 +800,20 @@ public override async Task RollbackAsync(SetupContext ctx, CancellationToken ct)
ctx.LocalAiManifestCreatedThisRun = false;
}
+ internal static async Task RestoreUpgradeReceiptAsync(SetupContext ctx, CancellationToken ct)
+ {
+ if (ctx.LocalAiUpgradeOriginalInstall is not { } originalInstall)
+ return;
+
+ var paths = new LocalAiPaths(ctx.LocalDataDir);
+ var store = new LocalAiManifestStore(paths);
+ await store.SaveAsync(originalInstall.Manifest, ct);
+ File.Delete(paths.RouterPresetPath);
+ ctx.LocalAiResolvedInstall = store.ResolveAndValidate(originalInstall.Manifest);
+ ctx.LocalAiManifestCreatedThisRun = false;
+ ctx.LocalAiUpgradeOriginalInstall = null;
+ }
+
private static ImmutableArray BuildRuntimeReceipts(
LlamaRuntimeVariant runtime,
LlamaRuntimeInstallResult install)
diff --git a/src/OpenClaw.SetupEngine/SetupContext.cs b/src/OpenClaw.SetupEngine/SetupContext.cs
index aad8f82dd..4390a6f14 100644
--- a/src/OpenClaw.SetupEngine/SetupContext.cs
+++ b/src/OpenClaw.SetupEngine/SetupContext.cs
@@ -1,3 +1,4 @@
+using System.Collections.Immutable;
using System.Diagnostics.CodeAnalysis;
using System.Text.Json;
using System.Text.Json.Serialization;
@@ -515,8 +516,16 @@ public Func>?
public int? LocalAiPort { get; set; }
internal LlamaRuntimeInstallResult? LocalAiRuntimeInstall { get; set; }
internal HuggingFaceModelInstallResult? LocalAiModelInstall { get; set; }
+ ///
+ /// Verified additional model assets (a DFlash draft checkpoint and/or
+ /// in catalog order. Empty for every recipe
+ /// that has neither.
+ ///
+ internal ImmutableArray LocalAiAdditionalModelInstalls { get; set; } =
+ ImmutableArray.Empty;
internal LocalAiResolvedInstall? LocalAiResolvedInstall { get; set; }
internal LocalAiResolvedInstall? LocalAiRecoveryOriginalInstall { get; set; }
+ internal LocalAiResolvedInstall? LocalAiUpgradeOriginalInstall { get; set; }
internal bool LocalAiRecoveryProviderTransition { get; set; }
internal bool LocalAiRecoveryReceiptRollbackAllowed { get; set; }
internal bool LocalAiManifestCreatedThisRun { get; set; }
diff --git a/src/OpenClaw.SetupEngine/SetupReviewSummary.cs b/src/OpenClaw.SetupEngine/SetupReviewSummary.cs
index 0b9f1f696..10b31de8b 100644
--- a/src/OpenClaw.SetupEngine/SetupReviewSummary.cs
+++ b/src/OpenClaw.SetupEngine/SetupReviewSummary.cs
@@ -77,16 +77,33 @@ public static SetupReviewSummary Build(SetupConfig config, string? dataDir = nul
LocalModelCatalog.Find(config.LocalAi.SelectedModelId) ?? LocalModelCatalog.Default;
LocalInferenceRunProfile? localAiProfile =
LocalModelCatalog.FindProfile(localAiModel, config.LocalAi.SelectedProfileId);
- string[] localAiCommands = config.LocalAi.Enabled
- ?
- [
+ string[] localAiCommands;
+ if (config.LocalAi.Enabled)
+ {
+ var commands = new List
+ {
"download verified llama-server + CUDA runtime for Windows",
$"download {localAiModel.Weights.RelativePath} from Hugging Face revision " +
((HuggingFaceRevisionSource)localAiModel.Weights.Source).RevisionSha,
- $"llama-server router on dynamic 127.0.0.1 port; model loads on first request",
- $"openclaw provider llamacpp -> /v1; primary llamacpp/{localAiModel.Id}",
- ]
- : [];
+ };
+ // Additional pinned artifacts (a DFlash draft checkpoint) are
+ // separate downloads the user is consenting
+ // to alongside the primary weights -- list each one explicitly
+ // rather than letting the consent screen understate what's fetched.
+ foreach (PinnedArtifact artifact in LocalModelCatalog.AdditionalArtifacts(localAiModel))
+ {
+ commands.Add(
+ $"download {artifact.RelativePath} from Hugging Face revision " +
+ ((HuggingFaceRevisionSource)artifact.Source).RevisionSha);
+ }
+ commands.Add("llama-server router on dynamic 127.0.0.1 port; model loads on first request");
+ commands.Add($"openclaw provider llamacpp -> /v1; primary llamacpp/{localAiModel.Id}");
+ localAiCommands = commands.ToArray();
+ }
+ else
+ {
+ localAiCommands = [];
+ }
var summary = new SetupReviewSummary(
DistroTitle: $"Install {baseDistro.Replace('-', ' ')} in WSL",
diff --git a/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs
index a0b6ee26a..9242951f0 100644
--- a/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs
+++ b/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs
@@ -10,7 +10,8 @@ public LlamaRuntimeVariant(
string id,
Architecture architecture,
Version cudaVersion,
- IReadOnlyList artifacts)
+ IReadOnlyList artifacts,
+ string? releaseTag = null)
{
ArgumentException.ThrowIfNullOrWhiteSpace(id);
ArgumentNullException.ThrowIfNull(cudaVersion);
@@ -32,12 +33,20 @@ public LlamaRuntimeVariant(
Architecture = architecture;
CudaVersion = cudaVersion;
Artifacts = artifacts;
+ ReleaseTag = releaseTag ?? LlamaRuntimeCatalog.ReleaseTag;
}
public string Id { get; }
public Architecture Architecture { get; }
public Version CudaVersion { get; }
public IReadOnlyList Artifacts { get; }
+ ///
+ /// The llama.cpp release this variant was pinned from. Equals
+ /// for the current runtime and
+ /// keeps its own older value for a retired one, so an installed receipt is
+ /// validated against the release it actually recorded.
+ ///
+ public string ReleaseTag { get; }
public long TotalDownloadSizeBytes => Artifacts.Sum(artifact => artifact.SizeBytes);
}
@@ -47,11 +56,11 @@ public LlamaRuntimeVariant(
///
public static class LlamaRuntimeCatalog
{
- public const string ReleaseTag = "b10655";
- public const string ReleaseCommitSha = "cb300598d5f90189cb69d2702f4930aaf99d32a2";
+ public const string ReleaseTag = "b11026";
+ public const string ReleaseCommitSha = "b49650adb31f2e49a0d76113aeb1792134fd8413";
public const string ServerExecutableName = "llama-server.exe";
- public const string X64RuntimeId = "b10655-cuda13-x64";
- public const string Arm64RuntimeId = "b10655-cuda13-arm64";
+ public const string X64RuntimeId = "b11026-cuda13-x64";
+ public const string Arm64RuntimeId = "b11026-cuda13-arm64";
public static GitHubReleaseSource Source { get; } = new(
"ggml-org/llama.cpp",
@@ -64,43 +73,103 @@ public static class LlamaRuntimeCatalog
new LlamaRuntimeVariant(
X64RuntimeId,
Architecture.X64,
- new Version(13, 3),
+ new Version(13, 4),
+ Array.AsReadOnly(
+ new[]
+ {
+ RuntimeArtifact(
+ "llama-b11026-cuda13-x64",
+ ArtifactRole.RuntimeBinary,
+ "llama-b11026-bin-win-cuda-13.4-x64.zip",
+ 150_102_391,
+ "6799f0962d066c54aee3773f0e5efa0076e46418695c0f4f6d24a38e7007dfb1"),
+ RuntimeArtifact(
+ "cudart-b11026-cuda13-x64",
+ ArtifactRole.RuntimeDependency,
+ "cudart-llama-bin-win-cuda-13.4-x64.zip",
+ 423_535_356,
+ "738f8c251ac22b70c3ae6f83a10cf222725df0395246a2cf58f32bdb85fbe668"),
+ })),
+ new LlamaRuntimeVariant(
+ Arm64RuntimeId,
+ Architecture.Arm64,
+ new Version(13, 4),
Array.AsReadOnly(
new[]
{
RuntimeArtifact(
+ "llama-b11026-cuda13-arm64",
+ ArtifactRole.RuntimeBinary,
+ "llama-b11026-bin-win-cuda-13.4-arm64.zip",
+ 142_993_717,
+ "d4a31d05b4fe997872020d81e8482e7712c7c254ec9c7ccdb9597ae2a31e6728"),
+ RuntimeArtifact(
+ "cudart-b11026-cuda13-arm64",
+ ArtifactRole.RuntimeDependency,
+ "cudart-llama-bin-win-cuda-13.4-arm64.zip",
+ 153_262_407,
+ "642dcde8805b3e3165ca710a5443b3b4044b27d96bd3ee3132473988c9bcb774"),
+ })),
+ });
+
+ // Retired from new installs and never offered or selected. These exist only so a
+ // managed installation recorded before the runtime bump keeps resolving its own
+ // receipt and stays launchable until setup upgrades it. Pins are reproduced
+ // exactly as they were installed; nothing is remapped.
+ private const string LegacyB10655ReleaseTag = "b10655";
+ private const string LegacyB10655RuntimeIdX64 = "b10655-cuda13-x64";
+ private const string LegacyB10655RuntimeIdArm64 = "b10655-cuda13-arm64";
+
+ private static GitHubReleaseSource LegacyB10655Source { get; } = new(
+ "ggml-org/llama.cpp",
+ LegacyB10655ReleaseTag,
+ "cb300598d5f90189cb69d2702f4930aaf99d32a2");
+
+ private static readonly ReadOnlyCollection s_legacyVariants = Array.AsReadOnly(
+ new[]
+ {
+ new LlamaRuntimeVariant(
+ LegacyB10655RuntimeIdX64,
+ Architecture.X64,
+ new Version(13, 3),
+ Array.AsReadOnly(
+ new[]
+ {
+ LegacyB10655Artifact(
"llama-b10655-cuda13-x64",
ArtifactRole.RuntimeBinary,
"llama-b10655-bin-win-cuda-13.3-x64.zip",
146_478_045,
"be61636141327b3ca4d437c17489fd69964838a31a5fe3e97400f0dcd9f669dc"),
- RuntimeArtifact(
+ LegacyB10655Artifact(
"cudart-b10655-cuda13-x64",
ArtifactRole.RuntimeDependency,
"cudart-llama-bin-win-cuda-13.3-x64.zip",
390_970_417,
"1462a050eb4c684921ba51dcc4cc488a036674c3e73e9945ee705b854808d03e"),
- })),
+ }),
+ LegacyB10655ReleaseTag),
new LlamaRuntimeVariant(
- Arm64RuntimeId,
+ LegacyB10655RuntimeIdArm64,
Architecture.Arm64,
new Version(13, 4),
Array.AsReadOnly(
new[]
{
- RuntimeArtifact(
+ LegacyB10655Artifact(
"llama-b10655-cuda13-arm64",
ArtifactRole.RuntimeBinary,
"llama-b10655-bin-win-cuda-13.4-arm64.zip",
140_055_278,
"567e61b4129e0d5b0580e5d3ea86b82ab5b6bee745ee02f69b58af799b49a582"),
- RuntimeArtifact(
+ LegacyB10655Artifact(
"cudart-b10655-cuda13-arm64",
ArtifactRole.RuntimeDependency,
"cudart-llama-bin-win-cuda-13.4-arm64.zip",
153_318_797,
"5a40dc7c5fa3d0a80ceeba4f16f9e8d25d87bcf1399c9233588953c43436c33c"),
- })),
+ }),
+ LegacyB10655ReleaseTag),
});
public static IReadOnlyList Variants => s_variants;
@@ -108,6 +177,19 @@ public static class LlamaRuntimeCatalog
public static LlamaRuntimeVariant? Find(Architecture architecture) =>
s_variants.SingleOrDefault(variant => variant.Architecture == architecture);
+ ///
+ /// Resolves a runtime id from an already-installed receipt, including a retired
+ /// runtime from before the last version bump. Use this only on installed-receipt
+ /// validation and launch paths. Selection and acquisition must keep using
+ /// and so a retired runtime is never
+ /// installed again.
+ ///
+ public static LlamaRuntimeVariant? FindInstalled(string? id) =>
+ string.IsNullOrWhiteSpace(id)
+ ? null
+ : s_variants.SingleOrDefault(variant => string.Equals(variant.Id, id, StringComparison.Ordinal))
+ ?? s_legacyVariants.SingleOrDefault(variant => string.Equals(variant.Id, id, StringComparison.Ordinal));
+
private static PinnedArtifact RuntimeArtifact(
string id,
ArtifactRole role,
@@ -122,4 +204,19 @@ private static PinnedArtifact RuntimeArtifact(
sizeBytes,
new Sha256Digest(sha256),
LocalInferenceCatalogProvenance.NvidiaCair);
+
+ private static PinnedArtifact LegacyB10655Artifact(
+ string id,
+ ArtifactRole role,
+ string fileName,
+ long sizeBytes,
+ string sha256) =>
+ new(
+ id,
+ role,
+ LegacyB10655Source,
+ fileName,
+ sizeBytes,
+ new Sha256Digest(sha256),
+ LocalInferenceCatalogProvenance.NvidiaCair);
}
diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs
index 58581b1f5..cda634a2e 100644
--- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs
+++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs
@@ -48,6 +48,36 @@ public static long GetRequiredMemoryBytes(
LocalInferenceRunProfile profile) =>
LocalInferenceQualificationPolicy.GetRequiredMemoryBytes(model, profile);
+ ///
+ /// Device eligibility for deciding whether Local AI stays available, given the model
+ /// already configured on this machine.
+ ///
+ ///
+ /// A SKU with no recommended default is a statement about what to install by default,
+ /// not about what the device can run. Gating availability purely on the default pick
+ /// would switch Local AI off on a setup rerun for a machine that already has a working
+ /// configured model, so once a model is configured this reports on that model: a
+ /// selection that still passes the full capacity fit-test is retained, and one that does
+ /// not carries its own failure (unknown model, or the model name with its required and
+ /// detected memory) rather than the SKU's generic no-recommendation reason. With no model
+ /// configured, the device result stands and fresh setup is unchanged.
+ ///
+ public static LocalInferenceEligibilityResult EvaluateForConfiguredAvailability(
+ HostHardwareInfo hardware,
+ string? configuredModelId)
+ {
+ ArgumentNullException.ThrowIfNull(hardware);
+ LocalInferenceEligibilityResult device = Evaluate(hardware);
+ if (device.CanInstall ||
+ device.SelectionFailureCode != LocalInferenceSelectionFailureCode.NotRecommendedForSku ||
+ string.IsNullOrWhiteSpace(configuredModelId))
+ {
+ return device;
+ }
+
+ return Evaluate(hardware, configuredModelId);
+ }
+
public static LocalInferenceEligibilityResult Evaluate(
HostHardwareInfo hardware,
string? requestedModelId = null)
@@ -64,7 +94,14 @@ public static LocalInferenceEligibilityResult Evaluate(
LocalInferencePlan plan = selection.Plan;
long requiredMemoryBytes = GetRequiredMemoryBytes(plan.Model, plan.Profile);
- CandidateAssessment? selected = hardware.NvidiaGpus
+ // A plan bound to one adapter (an RTX Spark SKU recipe) must be assessed only
+ // against that adapter. Ranking every NVIDIA GPU here would let the recipe
+ // chosen for the Spark be reported against, and then launched on, a different
+ // GPU on a mixed host.
+ IEnumerable candidateGpus = plan.BoundGpuStableId is { Length: > 0 } boundId
+ ? hardware.NvidiaGpus.Where(gpu => string.Equals(gpu.StableId, boundId, StringComparison.Ordinal))
+ : hardware.NvidiaGpus;
+ CandidateAssessment? selected = candidateGpus
.Select(gpu => Assess(gpu, plan.Runtime, requiredMemoryBytes))
.OrderBy(candidate => StatusRank(candidate.Status))
.ThenBy(candidate => DefinitivenessRank(candidate.FailureCode))
diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs
index bbc8f99c7..61cbe6e85 100644
--- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs
+++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs
@@ -16,6 +16,8 @@ public enum LocalInferenceSelectionFailureCode
RuntimeUnavailable = 1,
NoNvidiaGpu = 2,
UnknownModel = 3,
+ /// RTX Spark detected, but this memory SKU has no recommended local model.
+ NotRecommendedForSku = 4,
}
/// Whether a caller accepted the catalog default or named a model explicitly.
@@ -26,11 +28,19 @@ public enum LocalInferenceModelSelectionOrigin
}
/// A complete, immutable native inference choice.
+///
+/// The adapter this recipe was chosen for, when the choice is only valid on that
+/// adapter. RTX Spark recipes come from a fixed per-device SKU table, so the recipe
+/// and the GPU that runs it must be the same adapter; eligibility restricts its
+/// candidates to this id. Null means any qualifying NVIDIA GPU may run the plan,
+/// which is the generic discrete-GPU behavior.
+///
public sealed record LocalInferencePlan(
LlamaRuntimeVariant Runtime,
LocalModelInfo Model,
LocalInferenceRunProfile Profile,
- LocalInferenceModelSelectionOrigin ModelSelectionOrigin);
+ LocalInferenceModelSelectionOrigin ModelSelectionOrigin,
+ string? BoundGpuStableId = null);
/// The deterministic result of selecting from the pinned local inference catalog.
public sealed record LocalInferenceSelectionResult
@@ -60,7 +70,14 @@ internal static LocalInferenceSelectionResult Unsupported(LocalInferenceSelectio
///
/// Pure selection from a hardware snapshot and optional model ID. The CPU
/// architecture chooses only the native runtime. GPU names and CPU/GPU SKU
-/// pairings are not part of qualification.
+/// pairings are not part of qualification for discrete GPUs -- the one
+/// deliberate exception is RTX Spark's default pick, routed through
+/// because its unified-memory SKU
+/// cannot be identified by capacity fit-testing alone (see NVIDIA's fixed
+/// SKU-to-recipe table). An explicitly requested model ID uses the same
+/// capacity fit-test on every GPU, except when the request is the Spark SKU's
+/// own recommendation round-tripped through setup, which keeps that SKU's
+/// pinned profile and adapter binding.
///
public static class LocalInferenceSelector
{
@@ -81,9 +98,42 @@ public static LocalInferenceSelectionResult Select(
LocalModelInfo? model;
LocalInferenceRunProfile profile;
LocalInferenceModelSelectionOrigin modelSelectionOrigin;
+ string? boundGpuStableId = null;
+ GpuInfo? sparkGpu = hardware.NvidiaGpus.FirstOrDefault(
+ gpu => gpu.IsRtxSpark && LocalInferenceQualificationPolicy.HasCompleteFacts(gpu));
+ var sparkPick = sparkGpu is null
+ ? null
+ : RtxSparkInferenceSelector.SelectDefault(sparkGpu);
if (string.IsNullOrWhiteSpace(requestedModelId))
{
- (model, profile) = SelectDefaultModelAndProfile(hardware, runtime);
+ if (sparkPick is not null)
+ {
+ // Bind the plan to this adapter: the SKU table answers "what should
+ // THIS Spark run", so the recipe is only valid on the Spark that
+ // produced it, never on some other GPU eligibility might rank higher.
+ (model, profile) = sparkPick.Value;
+ boundGpuStableId = sparkGpu!.StableId;
+ }
+ else
+ {
+ // Either no Spark, or a Spark SKU with no recommended model. In the
+ // latter case the Spark is excluded rather than failing the whole
+ // host, so a discrete GPU alongside it can still qualify normally.
+ HostHardwareInfo genericHardware = sparkGpu is null
+ ? hardware
+ : hardware with
+ {
+ Gpus = hardware.Gpus.Where(gpu => !gpu.IsRtxSpark).ToArray(),
+ };
+ if (!genericHardware.HasNvidiaGpu)
+ {
+ return LocalInferenceSelectionResult.Unsupported(
+ LocalInferenceSelectionFailureCode.NotRecommendedForSku);
+ }
+
+ (model, profile) = SelectDefaultModelAndProfile(genericHardware, runtime);
+ }
+
modelSelectionOrigin = LocalInferenceModelSelectionOrigin.Default;
}
else
@@ -91,29 +141,58 @@ public static LocalInferenceSelectionResult Select(
model = LocalModelCatalog.Find(requestedModelId);
if (model is null)
return LocalInferenceSelectionResult.Unsupported(LocalInferenceSelectionFailureCode.UnknownModel);
- profile = SelectBestFittingProfile(hardware, runtime, model) ??
- LocalModelCatalog.GetProfiles(model)[^1];
+ if (sparkPick is { } recommended &&
+ string.Equals(recommended.Model.Id, model.Id, StringComparison.OrdinalIgnoreCase))
+ {
+ // Setup persists the recommended model id and passes it back here, so
+ // the SKU's own recommendation arrives as an explicit request. Re-deriving
+ // its profile through the generic fit-test would silently discard the
+ // pinned profile the SKU table specifies (the 64 GB tier's reduced context
+ // is not the largest that merely fits) and drop the adapter binding.
+ // A request for any other model is a real user override and still uses
+ // the generic fit-test below.
+ profile = recommended.Profile;
+ boundGpuStableId = sparkGpu!.StableId;
+ }
+ else
+ {
+ profile = SelectBestFittingProfile(hardware, runtime, model) ??
+ LocalModelCatalog.GetProfiles(model)[^1];
+ }
+
modelSelectionOrigin = LocalInferenceModelSelectionOrigin.Explicit;
}
return LocalInferenceSelectionResult.Selected(
- new LocalInferencePlan(runtime, model, profile, modelSelectionOrigin));
+ new LocalInferencePlan(runtime, model, profile, modelSelectionOrigin, boundGpuStableId));
}
private static (LocalModelInfo Model, LocalInferenceRunProfile Profile) SelectDefaultModelAndProfile(
HostHardwareInfo hardware,
LlamaRuntimeVariant runtime)
{
- foreach (LocalModelInfo candidate in LocalModelCatalog.Models
+ // A priority-0, explicit-alternative model (currently only the
+ // experimental 96GB Flash-Next recipe) is offered solely by
+ // RtxSparkInferenceSelector for its one intended SKU, never picked as
+ // a generic dGPU default or fallback -- excluded here so a large
+ // enough non-Spark GPU (or a Spark GPU IsRtxSpark fails to detect)
+ // can't land on it by tie-break/fallback ordering. Priority-0 models
+ // that aren't explicit alternatives (the other Spark-only recipes)
+ // keep their existing, unrelated reachability.
+ IEnumerable genericDefaultCandidates = LocalModelCatalog.Models
+ .Where(model => model.RecommendationPriority > 0 || !model.IsExplicitAlternative);
+ foreach (LocalModelInfo candidate in genericDefaultCandidates
.OrderByDescending(model => model.RecommendationPriority)
- .ThenByDescending(model => model.Weights.SizeBytes))
+ .ThenByDescending(LocalModelCatalog.TotalDownloadSizeBytes))
{
LocalInferenceRunProfile? profile = SelectBestFittingProfile(hardware, runtime, candidate);
if (profile is not null)
return (candidate, profile);
}
- LocalModelInfo fallback = LocalModelCatalog.Models.OrderBy(model => model.Weights.SizeBytes).First();
+ LocalModelInfo fallback = genericDefaultCandidates
+ .OrderBy(LocalModelCatalog.TotalDownloadSizeBytes)
+ .First();
return (fallback, LocalModelCatalog.GetProfiles(fallback)[^1]);
}
@@ -149,9 +228,12 @@ public static long GetRequiredMemoryBytes(
{
ArgumentNullException.ThrowIfNull(model);
ArgumentNullException.ThrowIfNull(profile);
+ long weightsBytes = SaturatingAdd(
+ model.Weights.SizeBytes,
+ model.Recipe.DraftWeights?.SizeBytes ?? 0);
return SaturatingAdd(
SaturatingAdd(
- SaturatingAdd(model.Weights.SizeBytes, GetKvCacheMemoryBytes(model.Recipe, profile)),
+ SaturatingAdd(weightsBytes, GetKvCacheMemoryBytes(model.Recipe, profile)),
GetDraftKvCacheMemoryBytes(model.Recipe, profile)),
profile.RuntimeWorkspaceBytes);
}
@@ -182,8 +264,12 @@ internal static long GetDraftKvCacheMemoryBytes(
ArgumentNullException.ThrowIfNull(recipe);
ArgumentNullException.ThrowIfNull(profile);
- // The pinned Qwen MTP artifacts contain one draft attention layer with
- // the same KV head count and head dimension as the target model.
+ if (recipe.SpeculativeDecoding == SpeculativeDecodingMode.None)
+ return 0;
+
+ // The pinned Qwen MTP and DFlash draft artifacts contain one draft
+ // attention layer with the same KV head count and head dimension as
+ // the target model.
long bytesPerToken = SaturatingAdd(
SaturatingMultiply(
recipe.KeyValueHeadCount,
diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs
index 72a2a2391..e15b59ba8 100644
--- a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs
+++ b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs
@@ -1,3 +1,4 @@
+using System.Collections.Immutable;
using System.Collections.ObjectModel;
namespace OpenClaw.Shared.Inference.Catalog;
@@ -13,6 +14,10 @@ public enum KvCachePrecision
public enum SpeculativeDecodingMode
{
DraftMtp = 0,
+ /// No speculative decoding; the target model runs standalone.
+ None = 1,
+ /// Draft-flash decoding using a separate, independently pinned draft checkpoint.
+ DraftDFlash = 2,
}
/// Sampling values recommended for the model's thinking mode.
@@ -38,8 +43,13 @@ public LocalModelRunRecipe(
bool offloadAllLayers,
SpeculativeDecodingMode speculativeDecoding,
int speculativeDraftMaxTokens,
- ModelSamplingPreset sampling)
+ ModelSamplingPreset sampling,
+ PinnedArtifact? draftWeights = null)
{
+ if (speculativeDecoding == SpeculativeDecodingMode.DraftDFlash && draftWeights is null)
+ throw new ArgumentException("Draft-flash decoding requires a pinned draft checkpoint.", nameof(draftWeights));
+ if (speculativeDecoding != SpeculativeDecodingMode.DraftDFlash && draftWeights is not null)
+ throw new ArgumentException("Only draft-flash decoding uses a separate draft checkpoint.", nameof(draftWeights));
if (batchTokens <= 0)
throw new ArgumentOutOfRangeException(nameof(batchTokens));
if (microBatchTokens <= 0 || microBatchTokens > batchTokens)
@@ -67,6 +77,7 @@ public LocalModelRunRecipe(
SpeculativeDecoding = speculativeDecoding;
SpeculativeDraftMaxTokens = speculativeDraftMaxTokens;
Sampling = sampling;
+ DraftWeights = draftWeights;
}
public int BatchTokens { get; }
@@ -80,6 +91,8 @@ public LocalModelRunRecipe(
public SpeculativeDecodingMode SpeculativeDecoding { get; }
public int SpeculativeDraftMaxTokens { get; }
public ModelSamplingPreset Sampling { get; }
+ /// The independently pinned draft checkpoint, set only for .
+ public PinnedArtifact? DraftWeights { get; }
}
/// A downloadable GGUF model and its deterministic llama-server recipe.
@@ -142,10 +155,16 @@ public static class LocalModelCatalog
/// Qwen3.5 9B receipt keeps resolving and launching across upgrade.
///
public const string Qwen9BModelId = "qwen3.5-9b-mtp-q4-k-m";
+ /// RTX Spark 48GB-SKU recipe. Never offered on the generic dGPU path; see RtxSparkInferenceSelector.
+ public const string Qwen35B_IQ4XSModelId = "qwen3.6-35b-a3b-mtp-ud-iq4-xs";
+ /// RTX Spark 128GB-SKU default recipe. Never offered on the generic dGPU path; see RtxSparkInferenceSelector.
+ public const string Qwen38_27B_DFlashModelId = "qwen3.8-27b-dflash-ud-q4-k-m";
public const int NativeContextTokens = 262_144;
public const int IntermediateContextTokens = 196_608;
public const int ReducedContextTokens = 131_072;
public const int MinimumContextTokens = 65_536;
+ /// RTX Spark 48GB-SKU context tier (98,304 tokens); see .
+ public const int RtxSpark48GbContextTokens = 98_304;
// Measured-conservative allowances for compute buffers, recurrent state,
// CUDA graphs, allocator alignment, and miscellaneous backend allocations.
@@ -172,6 +191,10 @@ public static class LocalModelCatalog
"unsloth/Qwen3.5-9B-MTP-GGUF",
"9716a636ee4bddc3fed678220b7a33dd2a4160ae");
+ private static readonly HuggingFaceRevisionSource s_qwen38_27BDFlashDraftSource = new(
+ "z-lab/Qwen3.8-27B-DFlash2-GGUF",
+ "2d9571f8ce46e151f61c6499c99dee6079e1d610");
+
private static readonly ReadOnlyCollection s_models = Array.AsReadOnly(
new[]
{
@@ -232,6 +255,63 @@ public static class LocalModelCatalog
IsExplicitAlternative: true,
SupportsVision: false,
RecommendationPriority: 200),
+ // RTX Spark SKU recipes below, offered by RtxSparkInferenceSelector
+ // keyed off the detected Spark unified-memory SKU. RecommendationPriority: 0
+ // alone does not exclude a model from the generic dGPU default pick --
+ // it only loses every tie-break against a positive-priority model that
+ // fits. LocalInferenceSelector.SelectDefaultModelAndProfile separately
+ // excludes IsExplicitAlternative models at this priority (see its
+ // comment) so the always-alternative-only ones can't still win by
+ // tie-break/fallback ordering among themselves.
+ new LocalModelInfo(
+ Qwen35B_IQ4XSModelId,
+ "Qwen3.6 35B-A3B (UD-IQ4_XS)",
+ "Qwen3.6",
+ "UD-IQ4_XS",
+ ModelArtifact(
+ Qwen35B_IQ4XSModelId,
+ s_qwen35BSource,
+ "Qwen3.6-35B-A3B-UD-IQ4_XS.gguf",
+ 18_209_036_576,
+ "df27a780435b7b45c2597536112ea3cb091f8544c3d0c3318d9f4258b31f7adf"),
+ Recipe(
+ fullAttentionLayerCount: 10,
+ keyValueHeadCount: 2,
+ temperature: 0.6,
+ speculativeDraftMaxTokens: 2),
+ IsDefault: false,
+ IsExplicitAlternative: false,
+ SupportsVision: false,
+ RecommendationPriority: 0),
+ new LocalModelInfo(
+ Qwen38_27B_DFlashModelId,
+ "Qwen3.8 27B (UD-Q4_K_M, DFlash)",
+ "Qwen3.8",
+ "UD-Q4_K_M",
+ ModelArtifact(
+ Qwen38_27B_DFlashModelId,
+ s_qwen38_27BSource,
+ "Qwen3.8-27B-UD-Q4_K_M.gguf",
+ 16_464_440_224,
+ "322e194ff79741c7baa497c240f677f54b201b0efab44ca8e50f122b39123482"),
+ Recipe(
+ fullAttentionLayerCount: 16,
+ keyValueHeadCount: 4,
+ temperature: 1.0,
+ batchTokens: 4_096,
+ microBatchTokens: 512,
+ speculativeDecoding: SpeculativeDecodingMode.DraftDFlash,
+ speculativeDraftMaxTokens: 7,
+ draftWeights: ModelArtifact(
+ "qwen3.8-27b-dflash2-q4-k-m",
+ s_qwen38_27BDFlashDraftSource,
+ "Qwen3.8-27B-DFlash2-Q4_K_M.gguf",
+ 1_143_006_816,
+ "1a25c56858e1ebe93f2718ac1d49d1151f9323325c1bbfd6209370f4db131ebd")),
+ IsDefault: false,
+ IsExplicitAlternative: false,
+ SupportsVision: false,
+ RecommendationPriority: 0),
});
// Retired from new installs and never offered, recommended, or selectable.
@@ -264,7 +344,10 @@ public static class LocalModelCatalog
private static readonly IReadOnlyDictionary>
s_profilesByModel = s_models
- .Select(model => (model, profiles: Array.AsReadOnly(CreateProfiles(model))))
+ .Select(model => (model, profiles: Array.AsReadOnly(
+ string.Equals(model.Id, Qwen35B_IQ4XSModelId, StringComparison.Ordinal)
+ ? CreateRtxSpark48GbProfiles(model)
+ : CreateProfiles(model))))
.Concat(s_legacyModels
.Select(model => (model, profiles: Array.AsReadOnly(CreateLegacyProfiles(model)))))
.ToDictionary(
@@ -346,6 +429,32 @@ public static IReadOnlyList GetProfiles(LocalModelInfo
/// True when the id resolves only to a retired catalog entry.
public static bool IsLegacy(string? id) => Find(id) is null && FindInstalled(id) is not null;
+ ///
+ /// Everything setup downloads and llama-server loads for this recipe: the pinned
+ /// weights plus the DFlash draft checkpoint, if any. Use this for ranking, capacity
+ /// checks, and user-facing download-size disclosure -- the draft checkpoint is a
+ /// separate pinned artifact, not part of the target model's own weights, but it is
+ /// still bytes the user consents to, setup fetches, and the runtime loads.
+ ///
+ public static long TotalDownloadSizeBytes(LocalModelInfo model)
+ {
+ ArgumentNullException.ThrowIfNull(model);
+ return model.Weights.SizeBytes + (model.Recipe.DraftWeights?.SizeBytes ?? 0);
+ }
+
+ ///
+ /// The recipe's additional pinned artifacts beyond its primary weights, in the
+ /// fixed order every acquirer, manifest, and launch path must agree on. Today that
+ /// is the DFlash draft checkpoint, when the recipe pins one.
+ ///
+ public static ImmutableArray AdditionalArtifacts(LocalModelInfo model)
+ {
+ ArgumentNullException.ThrowIfNull(model);
+ return model.Recipe.DraftWeights is { } draftWeights
+ ? [draftWeights]
+ : ImmutableArray.Empty;
+ }
+
private static LocalInferenceRunProfile[] CreateProfiles(LocalModelInfo model) =>
[
Profile(model, NativeContextTokens, KvCachePrecision.F16),
@@ -367,6 +476,14 @@ private static LocalInferenceRunProfile[] CreateLegacyProfiles(LocalModelInfo mo
Profile(model, NativeContextTokens, KvCachePrecision.F16),
];
+ // The RTX Spark 48GB-SKU recipe launches at a single fixed context/KV
+ // tier (98,304 tokens, F16 KV) rather than the shared cross-product of
+ // tiers other models expose.
+ private static LocalInferenceRunProfile[] CreateRtxSpark48GbProfiles(LocalModelInfo model) =>
+ [
+ Profile(model, RtxSpark48GbContextTokens, KvCachePrecision.F16),
+ ];
+
private static LocalInferenceRunProfile Profile(
LocalModelInfo model,
int contextTokens,
@@ -377,6 +494,7 @@ private static LocalInferenceRunProfile Profile(
NativeContextTokens => RuntimeWorkspaceReserveBytes,
IntermediateContextTokens => IntermediateContextWorkspaceReserveBytes,
ReducedContextTokens => ReducedContextWorkspaceReserveBytes,
+ RtxSpark48GbContextTokens => ReducedContextWorkspaceReserveBytes,
MinimumContextTokens => MinimumContextWorkspaceReserveBytes,
_ => throw new ArgumentOutOfRangeException(nameof(contextTokens)),
};
@@ -408,23 +526,29 @@ private static PinnedArtifact ModelArtifact(
private static LocalModelRunRecipe Recipe(
int fullAttentionLayerCount,
int keyValueHeadCount,
- double temperature) =>
+ double temperature,
+ int batchTokens = 4_096,
+ int microBatchTokens = 4_096,
+ SpeculativeDecodingMode speculativeDecoding = SpeculativeDecodingMode.DraftMtp,
+ int speculativeDraftMaxTokens = 3,
+ PinnedArtifact? draftWeights = null) =>
new(
- batchTokens: 4_096,
- microBatchTokens: 4_096,
+ batchTokens: batchTokens,
+ microBatchTokens: microBatchTokens,
parallelRequests: 1,
fullAttentionLayerCount: fullAttentionLayerCount,
keyValueHeadCount: keyValueHeadCount,
keyValueHeadDimension: 256,
flashAttention: true,
offloadAllLayers: true,
- speculativeDecoding: SpeculativeDecodingMode.DraftMtp,
- speculativeDraftMaxTokens: 3,
+ speculativeDecoding: speculativeDecoding,
+ speculativeDraftMaxTokens: speculativeDraftMaxTokens,
sampling: new ModelSamplingPreset(
Temperature: temperature,
TopK: 20,
TopP: 0.95,
MinP: 0.0,
RepetitionPenalty: 1.0,
- PresencePenalty: 0.0));
+ PresencePenalty: 0.0),
+ draftWeights: draftWeights);
}
diff --git a/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs
new file mode 100644
index 000000000..38040b24b
--- /dev/null
+++ b/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs
@@ -0,0 +1,62 @@
+namespace OpenClaw.Shared.Inference.Catalog;
+
+///
+/// Default-recipe routing for RTX Spark, a unified-memory SKU where "total
+/// CUDA-visible memory" identifies the physical memory SKU rather than a
+/// discrete GPU's VRAM. RTX Spark bypasses the generic priority/fit-test
+/// default pick () in favor of a fixed
+/// SKU-to-recipe table NVIDIA specified for this hardware. Runtime/driver
+/// eligibility is still assessed uniformly afterward by
+/// for both paths.
+///
+internal static class RtxSparkInferenceSelector
+{
+ // Empirically confirmed on real RTX Spark hardware: a 48GB-SKU unit's
+ // cuMemGetInfo total reads ~48.59e9 bytes (~45.25 GiB), tracking the
+ // nominal decimal-GB SKU size closely. This is NOT what nvidia-smi's "FB
+ // Memory Usage" (NVML) or Windows' "Total Physical Memory" report on the
+ // same box -- both read far lower on Spark's unified-memory design and
+ // must never be used for SKU classification; only GpuInfo.GpuVisibleMemoryBytes
+ // (CudaHostHardwareProbe's cuMemGetInfo reading) is reliable here.
+ // Boundaries are geometric midpoints between nominal decimal-GB SKU
+ // sizes; only the 48GB boundary is hardware-verified today.
+ private const long DecimalGigabyte = 1_000_000_000L;
+
+ private static readonly long s_boundary32_48 = GeometricMidpointBytes(32, 48);
+ private static readonly long s_boundary48_64 = GeometricMidpointBytes(48, 64);
+ private static readonly long s_boundary64_128 = GeometricMidpointBytes(64, 128);
+
+ ///
+ /// Returns the default (model, profile) pick for a detected RTX Spark
+ /// GPU, or null when this SKU has no recommended local model.
+ ///
+ internal static (LocalModelInfo Model, LocalInferenceRunProfile Profile)? SelectDefault(GpuInfo sparkGpu)
+ {
+ ArgumentNullException.ThrowIfNull(sparkGpu);
+ long totalBytes = LocalInferenceQualificationPolicy.GetEffectiveTotalMemoryBytes(sparkGpu);
+
+ return totalBytes switch
+ {
+ _ when totalBytes < s_boundary32_48 => null, // 32GB SKU: no local AI recommended
+ _ when totalBytes < s_boundary48_64 => Recipe(LocalModelCatalog.Qwen35B_IQ4XSModelId),
+ _ when totalBytes < s_boundary64_128 => Recipe(LocalModelCatalog.Qwen38_27BModelId, ReducedQ8_0ProfileId),
+ _ => Recipe(LocalModelCatalog.Qwen38_27B_DFlashModelId),
+ };
+ }
+
+ private const string ReducedQ8_0ProfileId = "ctx-131072-q8_0";
+
+ private static (LocalModelInfo, LocalInferenceRunProfile) Recipe(string modelId, string? profileId = null)
+ {
+ LocalModelInfo model = LocalModelCatalog.Find(modelId)
+ ?? throw new InvalidOperationException($"RTX Spark recipe '{modelId}' is missing from the catalog.");
+ LocalInferenceRunProfile profile = profileId is null
+ ? LocalModelCatalog.GetProfiles(model)[0]
+ : LocalModelCatalog.FindProfile(model, profileId)
+ ?? throw new InvalidOperationException($"RTX Spark recipe '{modelId}' is missing profile '{profileId}'.");
+ return (model, profile);
+ }
+
+ private static long GeometricMidpointBytes(int lowerNominalGb, int upperNominalGb) =>
+ (long)(Math.Sqrt((double)lowerNominalGb * upperNominalGb) * DecimalGigabyte);
+}
diff --git a/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs b/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs
index b1beaa788..60c6b0813 100644
--- a/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs
+++ b/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs
@@ -52,7 +52,17 @@ public sealed record GpuInfo(
long? FreeSharedGpuMemoryBytes = null,
string? DriverVersion = null,
int? CudaMajorVersion = null,
- string? StableId = null);
+ string? StableId = null)
+{
+ ///
+ /// True for an RTX Spark unified-memory adapter (e.g. "NVIDIA RTX Spark
+ /// N1X"), identified by its driver-reported name. RTX Spark is routed
+ /// through a fixed SKU recipe table instead of the generic capacity
+ /// fit-test; see RtxSparkInferenceSelector.
+ ///
+ public bool IsRtxSpark =>
+ Name.Contains("RTX Spark", StringComparison.OrdinalIgnoreCase);
+}
///
/// Snapshot of the host's inference-relevant hardware. Every probed field is
diff --git a/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs b/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs
index c0a8a3fcd..eff8285d0 100644
--- a/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs
+++ b/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs
@@ -283,6 +283,9 @@ private async Task RefreshAvailabilityAsync(CancellationTokenSource cancellation
// dispatched callback runs) would make a queued-but-not-yet-run callback's own
// IsCurrentAvailabilityProbe guard fail against itself, silently dropping a real
// asynchronous DispatcherQueue completion.
+ // Captured before the probe so the managed-install receipt is read on the caller's
+ // thread rather than on whatever thread resumes after the awaited probe.
+ string? installedModelId = _runtimeSnapshot.ModelId;
try
{
HostHardwareInfo hardware = await Task.Run(
@@ -292,7 +295,16 @@ private async Task RefreshAvailabilityAsync(CancellationTokenSource cancellation
// not the currently selected/installed model. A selection-specific failure (unknown,
// deprecated, or oversized model) must not report the device itself as unavailable
// and block retry-setup from switching to a compatible catalog model.
- LocalInferenceEligibilityResult eligibility = LocalInferenceEligibility.Evaluate(hardware);
+ //
+ // The one exception is a SKU that has no recommended default at all (RTX Spark
+ // 32 GB). That says nothing about whether this device can run what is already
+ // installed, so the managed-install receipt is taken into account: an installed
+ // model that still qualifies keeps this entry point available, which is what lets
+ // Retry Setup reach the manifest-aware recovery path and Change Model stay enabled.
+ // A machine with no managed receipt still reports unavailable, and a receipt whose
+ // model is unknown or no longer fits reports that model's own reason.
+ LocalInferenceEligibilityResult eligibility =
+ LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, installedModelId);
if (eligibility.FailureCode == LocalInferenceEligibilityFailureCode.HardwareFactsIncomplete)
{
// Incomplete facts (a CUDA read that came back partial or transient) are
@@ -520,8 +532,16 @@ private void ApplyOnUiThread(Action action)
private void ApplyRuntimeSnapshot(LocalAiRuntimeSnapshot snapshot)
{
+ string? previousModelId = _runtimeSnapshot.ModelId;
_runtimeSnapshot = snapshot;
OnPropertyChanged(null);
+ // Availability reads the managed-install receipt (see RefreshAvailabilityAsync), which the
+ // runtime refresh may only publish after that read has already happened. Recomputing when
+ // the model id changes is what keeps a first visit correct: otherwise a 32 GB Spark whose
+ // receipt arrives late stays pinned at NotRecommendedForSku, with Retry Setup and Change
+ // Model disabled and Recheck unavailable, until the page is left and reopened.
+ if (IsActive && !string.Equals(previousModelId, snapshot.ModelId, StringComparison.Ordinal))
+ StartAvailabilityRefresh();
}
private void ApplyGatewaySnapshot(GatewayConnectionSnapshot snapshot)
{
diff --git a/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs b/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs
index 29381bf68..92596120d 100644
--- a/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs
+++ b/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs
@@ -28,6 +28,68 @@ public async Task Load_DoesNotTriggerCacheMigration()
Assert.Equal(fixture.Content, await File.ReadAllBytesAsync(fixture.LegacyModelPath));
}
+ [Fact]
+ public async Task Save_SchemaFourManifestOmitsAdditionalAssetFieldsFromJson()
+ {
+ // A recipe with no additional assets (every recipe before this session,
+ // and most since) must keep writing the exact schema-4 shape an older
+ // app build already knows how to read. AdditionalModelAssets/Paths
+ // default to ImmutableArray's unset (not .Empty) value specifically
+ // so JsonIgnoreCondition.WhenWritingDefault omits them here, and
+ // UsesHubCache must never appear at all -- it's a derived read helper,
+ // not part of the persisted contract.
+ using var temp = new TempDirectory("local-ai-manifest-schema4-json-");
+ var paths = new LocalAiPaths(temp.Combine("app-data"));
+ string legacyRelativePath = Path.Combine("models", "owner", "repository", Revision, "model.gguf");
+ var manifest = new LocalAiInstallManifest
+ {
+ SchemaVersion = LocalAiInstallManifest.HubCacheReceiptSchemaVersion,
+ EngineVersion = "b1",
+ Architecture = "x64",
+ RuntimeId = "llama-server-test",
+ ModelCatalogId = "test-model",
+ SelectedGpuId = "GPU-TEST",
+ ExecutablePath = Path.Combine("engines", "llama-server.exe"),
+ RuntimeAssets = ImmutableArray.Create(new LocalAiAssetReceipt
+ {
+ FileName = "runtime.zip",
+ SourceUrl = "https://example.invalid/runtime.zip",
+ SizeBytes = 1,
+ Sha256 = new string('a', 64),
+ }),
+ ModelPath = legacyRelativePath,
+ ModelCacheRoot = temp.Combine("hf-cache"),
+ CachedModelPath = HuggingFaceHubCache.TryGetSnapshotPaths(
+ temp.Combine("hf-cache"),
+ RepositoryId,
+ Revision,
+ RelativeModelPath,
+ out string cachedModelPath,
+ out _,
+ out string pathError)
+ ? cachedModelPath
+ : throw new InvalidOperationException(pathError),
+ ModelId = $"{RepositoryId}@{Revision}",
+ ModelAlias = "test-model",
+ ModelAsset = new LocalAiAssetReceipt
+ {
+ FileName = "model.gguf",
+ SourceUrl = $"https://huggingface.co/{RepositoryId}/resolve/{Revision}/{RelativeModelPath}?download=true",
+ SizeBytes = 1,
+ Sha256 = new string('b', 64),
+ },
+ ContextLength = 4096,
+ };
+ var store = new LocalAiManifestStore(paths, () => temp.Combine("hf-cache"));
+ await store.SaveAsync(manifest);
+
+ JsonObject persisted = (JsonNode.Parse(await File.ReadAllTextAsync(paths.ManifestPath)) as JsonObject)!;
+
+ Assert.False(persisted.ContainsKey("additionalModelAssets"));
+ Assert.False(persisted.ContainsKey("additionalModelPaths"));
+ Assert.False(persisted.ContainsKey("usesHubCache"));
+ }
+
[Fact]
public async Task Load_CopiesVerifiedLegacyWeightsAndRecordsTransitionalReceipt()
{
diff --git a/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs b/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs
index 936247e00..97f86b893 100644
--- a/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs
+++ b/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs
@@ -2546,6 +2546,127 @@ private static LocalAiInstallManifest LegacyQwen9BManifest()
};
}
+ ///
+ /// A managed install recorded before the llama-server runtime bump must keep
+ /// launching against its own pinned receipt. Updating the app must not strand an
+ /// installed model until a separate setup repair runs.
+ ///
+ [Fact]
+ public async Task Router_LaunchesRetiredRuntimeInstallAfterVersionBump()
+ {
+ using var temp = new TempDirectory("local-ai-legacy-runtime-");
+ var paths = new LocalAiPaths(temp.Path);
+ LlamaRuntimeVariant retired = LlamaRuntimeCatalog.FindInstalled("b10655-cuda13-arm64")!;
+ LocalAiInstallManifest manifest = ValidManifest() with
+ {
+ EngineVersion = retired.ReleaseTag,
+ RuntimeId = retired.Id,
+ ExecutablePath = Path.Combine(
+ "engines",
+ $"llama-{retired.ReleaseTag}",
+ LlamaRuntimeCatalog.ServerExecutableName),
+ RuntimeAssets = retired.Artifacts.Select(artifact => new LocalAiAssetReceipt
+ {
+ FileName = Path.GetFileName(artifact.RelativePath),
+ SourceUrl = artifact.DownloadUri.AbsoluteUri,
+ SizeBytes = artifact.SizeBytes,
+ Sha256 = artifact.Sha256.Value,
+ }).ToImmutableArray(),
+ };
+ var store = new LocalAiManifestStore(paths);
+ await store.SaveAsync(manifest);
+
+ LocalAiResolvedInstall saved = (await store.LoadAsync())!;
+ LlamaServerRouterLaunchPlan launch = LlamaServerRouterConfiguration.Build(paths, saved);
+
+ Assert.NotEqual(LlamaRuntimeCatalog.ReleaseTag, retired.ReleaseTag);
+ Assert.Equal("qwen3.6-35b-a3b-mtp-q4-k-m", launch.ModelAlias);
+ }
+
+ ///
+ /// Schema-5 extra assets are loaded natively by llama-server from the shared,
+ /// user-writable hub cache. One that no longer matches its pinned digest must
+ /// stop startup, exactly like a tampered primary model does.
+ ///
+ [Fact]
+ public async Task Startup_FailsWhenAnAdditionalModelAssetNoLongerMatchesItsReceipt()
+ {
+ using var temp = new TempDirectory("local-ai-tampered-asset-");
+ LocalAiPaths paths = await PrepareInstallAsync(temp);
+ var store = new LocalAiManifestStore(paths);
+ LocalAiResolvedInstall installed = (await store.LoadAsync())!;
+ string cacheRoot = temp.Combine("hf-cache");
+ const string draftRepo = "z-lab/Qwen3.8-27B-DFlash2-GGUF";
+ string draftRevision = new('c', 40);
+ Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths(
+ cacheRoot, draftRepo, draftRevision, "draft.gguf",
+ out string draftPath, out _, out string error), error);
+ // The primary model's cached path must be the real hub-cache snapshot path
+ // for its own repository and revision, or the receipt fails validation
+ // before the additional-asset check under test is ever reached.
+ Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths(
+ cacheRoot,
+ "unsloth/Qwen3.6-35B-A3B-MTP-GGUF",
+ "5bc3e238d916f48a861bac2f8a1990a0e9b7e98d",
+ "Qwen3.6-35B-A3B-UD-Q4_K_M.gguf",
+ out string cachedPrimaryPath, out _, out error), error);
+ Directory.CreateDirectory(Path.GetDirectoryName(cachedPrimaryPath)!);
+ await File.WriteAllTextAsync(cachedPrimaryPath, "primary");
+ Directory.CreateDirectory(Path.GetDirectoryName(draftPath)!);
+ await File.WriteAllTextAsync(draftPath, "draft");
+ LocalAiInstallManifest schemaFive = installed.Manifest with
+ {
+ SchemaVersion = LocalAiInstallManifest.AdditionalAssetsSchemaVersion,
+ ModelCacheRoot = cacheRoot,
+ CachedModelPath = cachedPrimaryPath,
+ AdditionalModelAssets = ImmutableArray.Create(new LocalAiAssetReceipt
+ {
+ FileName = "draft.gguf",
+ SourceUrl = $"https://huggingface.co/{draftRepo}/resolve/{draftRevision}/draft.gguf?download=true",
+ SizeBytes = 1_143_006_816,
+ Sha256 = new string('d', 64),
+ }),
+ AdditionalModelPaths = ImmutableArray.Create(draftPath),
+ };
+
+ var events = new SynchronizedEventLog();
+ var platform = new FakePlatform();
+ await using var runtime = CreateRuntime(
+ paths,
+ new FakeProcessHost(platform, events, selectedPort: 28_771),
+ platform,
+ new FakeClient(events),
+ new FakeLifecycle(events),
+ modelFileVerifier: new SelectiveModelFileVerifier(
+ cachedPrimaryPath,
+ rejectPath: draftPath));
+ await store.SaveAsync(schemaFive);
+
+ LocalAiRuntimeSnapshot snapshot = await runtime.EnsureStartedAsync();
+
+ Assert.Equal(LocalAiRuntimeState.Failed, snapshot.State);
+ Assert.Contains("draft.gguf", snapshot.Detail ?? string.Empty, StringComparison.Ordinal);
+ }
+
+ /// Verifies the primary model but rejects one named additional asset.
+ private sealed class SelectiveModelFileVerifier(string resolvedPath, string rejectPath)
+ : ILocalAiModelFileVerifier
+ {
+ public Task TryOpenAsync(
+ string cacheRoot,
+ string candidatePath,
+ long expectedSizeBytes,
+ Sha256Digest expectedSha256,
+ CancellationToken cancellationToken)
+ {
+ cancellationToken.ThrowIfCancellationRequested();
+ if (string.Equals(candidatePath, rejectPath, StringComparison.OrdinalIgnoreCase))
+ return Task.FromResult(null);
+ return Task.FromResult(
+ new LocalAiVerifiedModelLease(new MemoryStream(), resolvedPath));
+ }
+ }
+
private static LocalAiInstallManifest ValidManifest()
{
LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.Find(
diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs
index fc353ea91..38381a515 100644
--- a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs
+++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs
@@ -967,6 +967,11 @@ public void ArchiveDestination_ResolvesValidNestedEntry()
public async Task Reconciler_ReusesOnlyMatchingManifestWithoutMutation()
{
using var temp = new TempDirectory();
+ // Pin the hub cache to an empty directory. This asserts that a matching receipt is
+ // reused untouched; with the ambient user cache it would instead depend on whether
+ // that cache happens to already hold the default model, which legitimately triggers
+ // the schema-3 to schema-4 migration and rewrites the receipt.
+ using var environment = new EnvironmentScope("HF_HUB_CACHE", CacheRoot(temp.Path));
LocalInferencePlan plan = CatalogPlan();
const string gpuId = "GPU-0";
LocalAiInstallManifest manifest = CreateManifest(temp.Path, plan, gpuId);
@@ -1145,6 +1150,271 @@ public async Task Reconciler_RecoveryRepairsMissingSchemaFourCompatibilityCopy()
Assert.Null(result.ModelInstall);
}
+ [Fact]
+ public async Task Reconciler_RecoveryWithValidModelStillPopulatesAdditionalAssetInstalls()
+ {
+ // Regression: recovery for a broken runtime (model + additional assets
+ // still verified valid) must give the caller everything it needs to
+ // persist a schema-5 manifest without re-downloading the already-
+ // verified draft checkpoint -- AcquireLocalAiModelStep's "reuse the
+ // verified model" skip only re-populates SetupContext from the
+ // reconcile result, it never re-runs acquisition itself.
+ using var temp = new TempDirectory();
+ byte[] primaryBytes = "verified-dflash-primary"u8.ToArray();
+ byte[] draftBytes = "verified-dflash-draft"u8.ToArray();
+ var draftSource = new HuggingFaceRevisionSource("owner/draft-repo", new string('c', 40));
+ var draftWeights = new PinnedArtifact(
+ "test-model-dflash-draft",
+ ArtifactRole.ModelWeights,
+ draftSource,
+ "draft.gguf",
+ draftBytes.Length,
+ new Sha256Digest(Sha256(draftBytes)));
+ LocalModelInfo model = CreateModelWithDraft(primaryBytes, draftWeights);
+ LlamaRuntimeVariant runtime = CreateRuntime(
+ CreateZip(("llama-server.exe", "server"u8.ToArray())),
+ CreateZip(("dependency.dll", "dependency"u8.ToArray())));
+ var plan = new LocalInferencePlan(
+ runtime,
+ model,
+ new LocalInferenceRunProfile(
+ "test-profile",
+ 128,
+ KvCachePrecision.F16,
+ KvCachePrecision.F16,
+ KvCachePrecision.F16,
+ KvCachePrecision.F16,
+ runtimeWorkspaceBytes: 1),
+ LocalInferenceModelSelectionOrigin.Default);
+ var paths = new LocalAiPaths(temp.Path);
+ string cacheRoot = CacheRoot(temp.Path);
+ var primarySource = Assert.IsType(model.Weights.Source);
+ Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths(
+ cacheRoot,
+ primarySource.RepositoryId,
+ primarySource.RevisionSha,
+ model.Weights.RelativePath,
+ out string cachedModelPath,
+ out _,
+ out string error), error);
+ Directory.CreateDirectory(Path.GetDirectoryName(cachedModelPath)!);
+ await File.WriteAllBytesAsync(cachedModelPath, primaryBytes);
+ Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths(
+ cacheRoot,
+ draftSource.RepositoryId,
+ draftSource.RevisionSha,
+ draftWeights.RelativePath,
+ out string cachedDraftPath,
+ out _,
+ out error), error);
+ Directory.CreateDirectory(Path.GetDirectoryName(cachedDraftPath)!);
+ await File.WriteAllBytesAsync(cachedDraftPath, draftBytes);
+
+ LocalAiInstallManifest manifest = CreateManifest(temp.Path, plan, "GPU-0") with
+ {
+ SchemaVersion = LocalAiInstallManifest.AdditionalAssetsSchemaVersion,
+ ModelCacheRoot = cacheRoot,
+ CachedModelPath = cachedModelPath,
+ AdditionalModelAssets = ImmutableArray.Create(new LocalAiAssetReceipt
+ {
+ FileName = "draft.gguf",
+ SourceUrl = draftWeights.DownloadUri.AbsoluteUri,
+ SizeBytes = draftWeights.SizeBytes,
+ Sha256 = draftWeights.Sha256.Value,
+ }),
+ AdditionalModelPaths = ImmutableArray.Create(cachedDraftPath),
+ };
+ await new LocalAiManifestStore(paths, () => cacheRoot).SaveAsync(manifest);
+
+ LocalAiReconcileResult result = await new LocalAiInstallReconciler(
+ new InvalidRuntimeInspector(),
+ new AcceptingModelVerifier(),
+ () => cacheRoot)
+ .ReconcileAsync(
+ temp.Path,
+ plan,
+ "GPU-0",
+ CancellationToken.None,
+ allowIncompleteInstallation: true);
+
+ Assert.False(result.Reused);
+ Assert.Null(result.RuntimeInstall);
+ Assert.NotNull(result.ModelInstall);
+ ImmutableArray additionalInstalls =
+ result.AdditionalModelInstalls ?? ImmutableArray.Empty;
+ HuggingFaceAdditionalAssetInstallResult draftInstall = Assert.Single(additionalInstalls);
+ Assert.Equal(cachedDraftPath, draftInstall.ModelPath);
+ Assert.False(draftInstall.CreatedThisRun);
+ }
+
+ [Fact]
+ public async Task Reconciler_UpgradesRetiredRuntimeReceiptInsteadOfFailingSetup()
+ {
+ // An install recorded before the runtime bump must upgrade, not end setup with an
+ // uninstall instruction. The runtime is dropped so the acquirer installs the new
+ // pin; the verified model is kept so an upgrade does not re-download it.
+ using var temp = new TempDirectory();
+ using var environment = new EnvironmentScope("HF_HUB_CACHE", CacheRoot(temp.Path));
+ LocalInferencePlan plan = CatalogPlan();
+ const string gpuId = "GPU-0";
+ var paths = new LocalAiPaths(temp.Path);
+ LlamaRuntimeVariant retired = LlamaRuntimeCatalog.FindInstalled("b10655-cuda13-x64")!;
+ LocalAiInstallManifest manifest = CreateRetiredManifest(temp.Path, plan, gpuId);
+ await new LocalAiManifestStore(paths).SaveAsync(manifest);
+ var reconciler = new LocalAiInstallReconciler(
+ new ValidRuntimeInspector(),
+ new AcceptingModelVerifier());
+
+ LocalAiReconcileResult result = await reconciler.ReconcileAsync(
+ temp.Path,
+ plan,
+ gpuId,
+ CancellationToken.None);
+
+ Assert.False(result.Reused);
+ Assert.Null(result.RuntimeInstall);
+ Assert.NotNull(result.ModelInstall);
+ Assert.NotNull(result.OriginalInstall);
+ Assert.Equal(retired.ReleaseTag, result.OriginalInstall!.Manifest.EngineVersion);
+ }
+
+ [Theory]
+ [InlineData(false, null)]
+ [InlineData(true, null)]
+ [InlineData(false, "after-reconcile")]
+ [InlineData(true, "after-reconcile")]
+ [InlineData(false, "after-persist")]
+ [InlineData(true, "after-persist")]
+ public async Task RuntimeUpgrade_MigratesModelAndRestoresOriginalReceiptOnFailure(
+ bool usesHubCache,
+ string? failureStage)
+ {
+ using var temp = new TempDirectory();
+ string cacheRoot = CacheRoot(temp.Path);
+ byte[] modelBytes = "verified-upgrade-model"u8.ToArray();
+ byte[] runtimeZip = CreateZip(("llama-server.exe", "new-server"u8.ToArray()));
+ byte[] dependencyZip = CreateZip(("dependency.dll", "dependency"u8.ToArray()));
+ LlamaRuntimeVariant runtime = CreateRuntime(runtimeZip, dependencyZip);
+ LocalInferencePlan plan = CatalogPlan() with
+ {
+ Runtime = runtime,
+ Model = CreateModel(modelBytes),
+ };
+ var paths = new LocalAiPaths(temp.Path);
+ var store = new LocalAiManifestStore(paths, () => cacheRoot);
+ LocalAiInstallManifest manifest = CreateRetiredManifest(temp.Path, plan, "GPU-0") with
+ {
+ GatewayFallbackModel = "openai/gpt-5",
+ };
+ string oldExecutable = paths.ResolveContainedPath(manifest.ExecutablePath, "executable");
+ Directory.CreateDirectory(Path.GetDirectoryName(oldExecutable)!);
+ await File.WriteAllTextAsync(oldExecutable, "old-server");
+ string legacyModel = paths.ResolveContainedPath(manifest.ModelPath, "model");
+ Directory.CreateDirectory(Path.GetDirectoryName(legacyModel)!);
+ await File.WriteAllBytesAsync(legacyModel, modelBytes);
+ await store.SaveAsync(manifest);
+ if (usesHubCache)
+ manifest = (await store.MigrateLegacyModelToHubCacheAsync())!.Manifest;
+ byte[] originalReceipt = await File.ReadAllBytesAsync(paths.ManifestPath);
+
+ var context = CreateContext(temp.Path, confirmDestructive: false);
+ context.Config.LocalAi.Enabled = true;
+ context.Config.RollbackOnFailure = true;
+ context.LocalAiPort = manifest.RequestedPort;
+ context.LocalAiEligibility = new LocalInferenceEligibilityResult(
+ LocalInferenceEligibilityStatus.Eligible,
+ LocalInferenceEligibilityFailureCode.None,
+ LocalInferenceSelectionFailureCode.None,
+ plan,
+ new GpuInfo(GpuVendor.Nvidia, "Test GPU", StableId: "GPU-0"),
+ RequiredTotalMemoryBytes: 0,
+ DetectedTotalMemoryBytes: 0,
+ RequiredFreeMemoryBytes: 0,
+ AvailableFreeMemoryBytes: 0);
+ int runtimeDownloads = 0;
+ using var runtimeClient = new HttpClient(new DelegateHandler(request =>
+ {
+ runtimeDownloads++;
+ return new HttpResponseMessage(HttpStatusCode.OK)
+ {
+ Content = new ByteArrayContent(
+ request.RequestUri!.AbsolutePath.EndsWith("runtime.zip", StringComparison.Ordinal)
+ ? runtimeZip
+ : dependencyZip),
+ };
+ }));
+ using var modelClient = new HttpClient(new DelegateHandler(_ =>
+ throw new InvalidOperationException("Upgrade must not download the verified primary model.")));
+ var reconciler = new LocalAiInstallReconciler(
+ new ValidRuntimeInspector(), new LocalAiModelFileVerifier(), () => cacheRoot);
+ string? cachedModel = null;
+ string? newExecutable = null;
+ var pipeline = new SetupPipeline(
+ [
+ new ReconcileLocalAiInstallationStep(reconciler),
+ new UpgradeCheckpointStep("after-reconcile", ctx =>
+ {
+ Assert.Null(ctx.LocalAiRecoveryOriginalInstall);
+ Assert.Equal(manifest.SchemaVersion, ctx.LocalAiUpgradeOriginalInstall?.Manifest.SchemaVersion);
+ Assert.Equal(oldExecutable, ctx.LocalAiUpgradeOriginalInstall?.ExecutablePath);
+ Assert.Equal(cacheRoot, ctx.LocalAiModelInstall?.CacheRoot);
+ cachedModel = ctx.LocalAiModelInstall!.ModelPath;
+ return failureStage == "after-reconcile";
+ }),
+ new AcquireLocalAiRuntimeStep(new LlamaRuntimeInstaller(
+ new LocalAiArtifactInstaller(runtimeClient), new ValidRuntimeInspector())),
+ new AcquireLocalAiModelStep(CreateModelInstaller(modelClient, temp.Path)),
+ new PersistLocalAiManifestStep(),
+ new UpgradeCheckpointStep("after-persist", ctx =>
+ {
+ LocalAiResolvedInstall upgraded = Assert.IsType(ctx.LocalAiResolvedInstall);
+ Assert.Equal("b11026", upgraded.Manifest.EngineVersion);
+ Assert.Equal(LocalAiInstallManifest.HubCacheReceiptSchemaVersion, upgraded.Manifest.SchemaVersion);
+ Assert.Equal(cacheRoot, upgraded.Manifest.ModelCacheRoot);
+ Assert.Equal(cachedModel, upgraded.ModelPath);
+ Assert.Equal(manifest.InstalledAtUtc, upgraded.Manifest.InstalledAtUtc);
+ Assert.Equal(manifest.GatewayFallbackModel, upgraded.Manifest.GatewayFallbackModel);
+ Assert.Null(upgraded.Endpoint);
+ newExecutable = upgraded.ExecutablePath;
+ Assert.True(File.Exists(newExecutable));
+ return failureStage == "after-persist";
+ }),
+ ]);
+
+ PipelineResult result = await pipeline.RunAsync(context);
+
+ Assert.True(
+ result.Outcome == (failureStage is null ? PipelineOutcome.Success : PipelineOutcome.Failed),
+ result.Message);
+ Assert.Equal(failureStage, result.FailedStepId);
+ if (failureStage is not null)
+ Assert.Equal("Injected upgrade failure.", result.Message);
+ Assert.Equal(failureStage == "after-reconcile" ? 0 : 2, runtimeDownloads);
+ Assert.Equal(modelBytes, await File.ReadAllBytesAsync(cachedModel!));
+ Assert.Equal(modelBytes, await File.ReadAllBytesAsync(legacyModel));
+ Assert.Equal("old-server", await File.ReadAllTextAsync(oldExecutable));
+ LocalAiResolvedInstall persisted = (await store.LoadAsync())!;
+ if (failureStage is null)
+ {
+ Assert.Equal(newExecutable, persisted.ExecutablePath);
+ Assert.Equal("b11026", persisted.Manifest.EngineVersion);
+ }
+ else
+ {
+ Assert.Equal(originalReceipt, await File.ReadAllBytesAsync(paths.ManifestPath));
+ Assert.Equal(oldExecutable, persisted.ExecutablePath);
+ Assert.Null(context.LocalAiUpgradeOriginalInstall);
+ if (newExecutable is not null)
+ Assert.False(File.Exists(newExecutable));
+
+ LocalAiReconcileResult retry = await reconciler.ReconcileAsync(
+ temp.Path, plan, "GPU-0", CancellationToken.None);
+ Assert.False(retry.Reused);
+ Assert.Null(retry.RuntimeInstall);
+ Assert.Equal(cacheRoot, retry.ModelInstall?.CacheRoot);
+ }
+ }
+
[Fact]
public async Task Reconciler_RejectsMigrationCacheRootInsideManagedInstallTree()
{
@@ -1478,6 +1748,28 @@ private static LocalAiInstallManifest CreateManifest(
};
}
+ private static LocalAiInstallManifest CreateRetiredManifest(
+ string localDataDirectory,
+ LocalInferencePlan plan,
+ string gpuId)
+ {
+ LlamaRuntimeVariant retired = LlamaRuntimeCatalog.FindInstalled("b10655-cuda13-x64")!;
+ return CreateManifest(localDataDirectory, plan, gpuId) with
+ {
+ EngineVersion = "b10655",
+ RuntimeId = retired.Id,
+ // Use the shipped path, not the installer helper under test.
+ ExecutablePath = Path.Combine("engines", "llama-server", "b10655", "win-x64", "llama-server.exe"),
+ RuntimeAssets = retired.Artifacts.Select(artifact => new LocalAiAssetReceipt
+ {
+ FileName = Path.GetFileName(artifact.RelativePath),
+ SourceUrl = artifact.DownloadUri.AbsoluteUri,
+ SizeBytes = artifact.SizeBytes,
+ Sha256 = artifact.Sha256.Value,
+ }).ToImmutableArray(),
+ };
+ }
+
private static LocalModelInfo CreateModel(byte[] bytes)
{
var source = new HuggingFaceRevisionSource("owner/repo", new string('a', 40));
@@ -1511,6 +1803,40 @@ private static LocalModelInfo CreateModel(byte[] bytes)
SupportsVision: false);
}
+ private static LocalModelInfo CreateModelWithDraft(byte[] primaryBytes, PinnedArtifact draftWeights)
+ {
+ var source = new HuggingFaceRevisionSource("owner/repo", new string('a', 40));
+ var artifact = new PinnedArtifact(
+ "test-model-dflash",
+ ArtifactRole.ModelWeights,
+ source,
+ "model.gguf",
+ primaryBytes.Length,
+ new Sha256Digest(Sha256(primaryBytes)));
+ return new LocalModelInfo(
+ "test-model-dflash",
+ "Test model (DFlash)",
+ "Test",
+ "Q4",
+ artifact,
+ new LocalModelRunRecipe(
+ 128,
+ 128,
+ 1,
+ 1,
+ 1,
+ 128,
+ true,
+ true,
+ SpeculativeDecodingMode.DraftDFlash,
+ 1,
+ new ModelSamplingPreset(0.6, 20, 0.95, 0, 1, 0),
+ draftWeights),
+ IsDefault: true,
+ IsExplicitAlternative: false,
+ SupportsVision: false);
+ }
+
private static LlamaRuntimeVariant CreateRuntime(byte[] binaryZip, byte[] dependencyZip)
{
var source = new GitHubReleaseSource("owner/repo", "v1", new string('b', 40));
@@ -1731,6 +2057,14 @@ public Task InspectAsync(
Task.FromResult(new LlamaRuntimeInspection(true, "valid", null));
}
+ private sealed class InvalidRuntimeInspector : ILlamaRuntimeInspector
+ {
+ public Task InspectAsync(
+ string installDirectory,
+ CancellationToken cancellationToken) =>
+ Task.FromResult(new LlamaRuntimeInspection(false, "invalid", "simulated corrupted runtime"));
+ }
+
private sealed class AcceptingModelVerifier : ILocalAiModelFileVerifier
{
public Task VerifyActiveAsync(
@@ -1743,6 +2077,12 @@ public Task VerifyLegacyCompatibilityAsync(
LocalAiPaths paths,
PinnedArtifact artifact,
CancellationToken cancellationToken) => Task.FromResult(true);
+
+ public Task VerifyAdditionalAssetAsync(
+ LocalAiResolvedInstall install,
+ string cachedAssetPath,
+ PinnedArtifact artifact,
+ CancellationToken cancellationToken) => Task.FromResult(true);
}
private sealed class RejectingModelVerifier : ILocalAiModelFileVerifier
@@ -1757,6 +2097,12 @@ public Task VerifyLegacyCompatibilityAsync(
LocalAiPaths paths,
PinnedArtifact artifact,
CancellationToken cancellationToken) => Task.FromResult(false);
+
+ public Task VerifyAdditionalAssetAsync(
+ LocalAiResolvedInstall install,
+ string cachedAssetPath,
+ PinnedArtifact artifact,
+ CancellationToken cancellationToken) => Task.FromResult(false);
}
private sealed class CompleteModelRepairStep(string localDataDirectory, LocalInferencePlan plan) : SetupStep
@@ -1801,6 +2147,18 @@ public override Task ExecuteAsync(SetupContext ctx, CancellationToke
}
}
+ private sealed class UpgradeCheckpointStep(
+ string id,
+ Func shouldFail) : SetupStep
+ {
+ public override string Id => id;
+ public override string DisplayName => id;
+ public override bool CanRetry => false;
+
+ public override Task ExecuteAsync(SetupContext ctx, CancellationToken ct) =>
+ Task.FromResult(shouldFail(ctx) ? StepResult.Fail("Injected upgrade failure.") : StepResult.Ok("Checked."));
+ }
+
private sealed class TempDirectory : IDisposable
{
public TempDirectory()
diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs
index 1bf2e8ea8..b5d4e2f1a 100644
--- a/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs
+++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs
@@ -139,8 +139,11 @@ private static SetupContext CreateContext(LocalAiConfig localAi, string? localDa
new GpuInfo(
GpuVendor.Nvidia,
"NVIDIA RTX Spark N1X (6144-core Blackwell RTX GPU)",
- GpuVisibleMemoryBytes: 25_702_694_912,
- FreeGpuVisibleMemoryBytes: 25_702_694_912,
+ // A real 48GB-SKU RTX Spark's measured cuMemGetInfo total, so this
+ // fixture routes through the fixed SKU table instead of landing
+ // below the smallest (32GB) recommended tier.
+ GpuVisibleMemoryBytes: 48_585_498_624,
+ FreeGpuVisibleMemoryBytes: 48_585_498_624,
DriverVersion: "616.00",
CudaMajorVersion: 13,
StableId: "GPU-SPARK"),
diff --git a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs
index 471866ffc..fe9a929c4 100644
--- a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs
+++ b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs
@@ -298,7 +298,7 @@ UuidFailure is not null
}
[Theory]
- [InlineData(RuntimeArchitecture.X64, "NVIDIA RTX Spark N1X", LlamaRuntimeCatalog.X64RuntimeId)]
+ [InlineData(RuntimeArchitecture.X64, "NVIDIA GeForce RTX 5080", LlamaRuntimeCatalog.X64RuntimeId)]
[InlineData(RuntimeArchitecture.Arm64, "NVIDIA GeForce RTX 5090", LlamaRuntimeCatalog.Arm64RuntimeId)]
public void Evaluate_RoutesRuntimeByArchitectureWithoutGpuSkuPairing(
RuntimeArchitecture architecture,
@@ -315,6 +315,265 @@ public void Evaluate_RoutesRuntimeByArchitectureWithoutGpuSkuPairing(
Assert.Equal(KvCachePrecision.Q8_0, result.Plan?.Profile.KeyCachePrecision);
}
+ // Boundaries verified against real hardware: a real 48GB-SKU RTX Spark
+ // reads ~45.25 GiB (48,585,498,624 bytes) via cuMemGetInfo, matching the
+ // "Gb48" case below almost exactly.
+ [Theory]
+ [InlineData(30, null)] // 32GB SKU: no local AI recommended
+ [InlineData(45, LocalModelCatalog.Qwen35B_IQ4XSModelId)] // 48GB SKU -> 24GB recipe
+ [InlineData(62, LocalModelCatalog.Qwen38_27BModelId)] // 64GB SKU -> 28GB recipe
+ [InlineData(120, LocalModelCatalog.Qwen38_27B_DFlashModelId)] // 128GB SKU -> 48GB recipe (default)
+ public void Evaluate_RoutesRtxSparkByFixedSkuTable(long totalGiB, string? expectedModelId)
+ {
+ LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate(
+ Hardware(RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark", totalGiB, totalGiB)));
+
+ if (expectedModelId is null)
+ {
+ Assert.Equal(LocalInferenceEligibilityStatus.Unsupported, result.Status);
+ Assert.Equal(LocalInferenceEligibilityFailureCode.CatalogSelectionFailed, result.FailureCode);
+ Assert.Equal(LocalInferenceSelectionFailureCode.NotRecommendedForSku, result.SelectionFailureCode);
+ Assert.Null(result.Plan);
+ }
+ else
+ {
+ Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status);
+ Assert.Equal(expectedModelId, result.Plan?.Model.Id);
+ }
+ }
+
+ [Fact]
+ public void Evaluate_Rtx5090WithSparkSizedMemoryIgnoresSkuTable()
+ {
+ // A non-Spark GPU that happens to have Spark-sized memory must still
+ // take the generic priority/fit-test path -- SKU routing is keyed
+ // strictly off the RTX Spark name, not memory size.
+ LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate(
+ Hardware(RuntimeArchitecture.X64, Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", totalGiB: 45, freeGiB: 45)));
+
+ Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status);
+ Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id);
+ }
+
+ [Fact]
+ public void Evaluate_SparkRecipeIsBoundToTheSparkGpuOnMixedHosts()
+ {
+ // The SKU table answers "what should THIS Spark run", so the recipe and the
+ // GPU that runs it must be the same adapter. A discrete GPU with more free
+ // memory must not win the eligibility ranking and end up running a recipe
+ // that was chosen for the Spark.
+ LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate(
+ Hardware(
+ RuntimeArchitecture.Arm64,
+ Gpu("NVIDIA RTX Spark N1X", "GPU-spark", totalGiB: 45, freeGiB: 45),
+ Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", totalGiB: 80, freeGiB: 80)));
+
+ Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status);
+ Assert.Equal(LocalModelCatalog.Qwen35B_IQ4XSModelId, result.Plan?.Model.Id);
+ Assert.Equal("GPU-spark", result.SelectedGpu?.StableId);
+ Assert.Equal("GPU-spark", result.Plan?.BoundGpuStableId);
+ }
+
+ [Fact]
+ public void Evaluate_UnrecommendedSparkSkuStillQualifiesADiscreteGpuOnTheSameHost()
+ {
+ // A 32 GB Spark has no recommended model, but that is a statement about the
+ // Spark, not about the host. An eligible discrete GPU beside it must still
+ // qualify through the generic path instead of the whole host being rejected.
+ LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate(
+ Hardware(
+ RuntimeArchitecture.X64,
+ Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", totalGiB: 30, freeGiB: 30),
+ Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", totalGiB: 32, freeGiB: 32)));
+
+ Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status);
+ Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id);
+ Assert.Equal("GPU-5090", result.SelectedGpu?.StableId);
+ Assert.Null(result.Plan?.BoundGpuStableId);
+ }
+
+ [Fact]
+ public void Evaluate_UnrecommendedSparkSkuAloneStillReportsNotRecommended()
+ {
+ LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate(
+ Hardware(RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30)));
+
+ Assert.Equal(LocalInferenceEligibilityStatus.Unsupported, result.Status);
+ Assert.Equal(
+ LocalInferenceSelectionFailureCode.NotRecommendedForSku,
+ result.SelectionFailureCode);
+ }
+
+ [Theory]
+ [InlineData(45)]
+ [InlineData(62)]
+ [InlineData(120)]
+ public void Evaluate_SparkRecommendationRoundTrippedBySetupKeepsItsSkuProfile(long totalGiB)
+ {
+ // Normal setup persists the recommended model id and passes it back as an
+ // explicit request, so the recommendation must resolve identically both ways.
+ // Otherwise the SKU's pinned profile (the 64 GB tier's reduced context is not
+ // the largest that merely fits) is silently replaced by the generic fit-test.
+ HostHardwareInfo hardware = Hardware(
+ RuntimeArchitecture.Arm64,
+ Gpu("NVIDIA RTX Spark N1X", "GPU-spark", totalGiB, totalGiB));
+
+ LocalInferenceEligibilityResult recommended = LocalInferenceEligibility.Evaluate(hardware);
+ LocalInferenceEligibilityResult roundTripped = LocalInferenceEligibility.Evaluate(
+ hardware,
+ recommended.Plan!.Model.Id);
+
+ Assert.Equal(recommended.Plan!.Model.Id, roundTripped.Plan?.Model.Id);
+ Assert.Equal(recommended.Plan!.Profile.Id, roundTripped.Plan?.Profile.Id);
+ Assert.Equal("GPU-spark", roundTripped.Plan?.BoundGpuStableId);
+ Assert.Equal(recommended.SelectedGpu?.StableId, roundTripped.SelectedGpu?.StableId);
+ }
+
+ [Fact]
+ public void Evaluate_ExplicitNonRecommendedModelOnSparkStillUsesTheGenericFitTest()
+ {
+ // A real user override must not be forced onto the SKU recipe.
+ HostHardwareInfo hardware = Hardware(
+ RuntimeArchitecture.Arm64,
+ Gpu("NVIDIA RTX Spark N1X", "GPU-spark", 62, 62));
+
+ LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate(
+ hardware,
+ LocalModelCatalog.Qwen27BModelId);
+
+ Assert.Equal(LocalModelCatalog.Qwen27BModelId, result.Plan?.Model.Id);
+ Assert.Null(result.Plan?.BoundGpuStableId);
+ }
+
+ [Fact]
+ public void EvaluateForConfiguredAvailability_KeepsAValidSavedModelOn32GbSpark()
+ {
+ // A 32 GB Spark has no recommended default, but it still runs a model that was
+ // already configured. Rerunning setup must not switch Local AI off on that machine.
+ HostHardwareInfo hardware = Hardware(
+ RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30));
+
+ LocalInferenceEligibilityResult result =
+ LocalInferenceEligibility.EvaluateForConfiguredAvailability(
+ hardware,
+ LocalModelCatalog.Qwen38_27BModelId);
+
+ Assert.True(result.CanInstall);
+ Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id);
+ Assert.NotNull(result.SelectedGpu);
+ }
+
+ [Fact]
+ public void EvaluateForConfiguredAvailability_PreservesTheRecoveryPinnedModelAndProfileOn32GbSpark()
+ {
+ // Recovery pins the configured model and reuses its resolved plan. On a SKU with no
+ // recommended default that selection must survive the availability gate with the same
+ // model and the same profile the explicit path resolves, so a recovery rerun does not
+ // silently move an existing install to a different context or KV precision.
+ HostHardwareInfo hardware = Hardware(
+ RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30));
+ const string pinnedModelId = LocalModelCatalog.Qwen38_27BModelId;
+
+ LocalInferenceEligibilityResult availability =
+ LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, pinnedModelId);
+ LocalInferenceEligibilityResult pinned =
+ LocalInferenceEligibility.Evaluate(hardware, pinnedModelId);
+
+ Assert.True(availability.CanInstall);
+ Assert.Equal(pinnedModelId, availability.Plan?.Model.Id);
+ Assert.Equal(pinned.Plan?.Profile.Id, availability.Plan?.Profile.Id);
+ Assert.Equal(pinned.Plan?.Profile.ContextTokens, availability.Plan?.Profile.ContextTokens);
+ Assert.Equal(pinned.Plan?.Profile.KeyCachePrecision, availability.Plan?.Profile.KeyCachePrecision);
+ Assert.Equal(pinned.SelectedGpu?.StableId, availability.SelectedGpu?.StableId);
+ }
+
+ [Fact]
+ public void EvaluateForConfiguredAvailability_WithNoSavedModelStillReportsNotRecommended()
+ {
+ HostHardwareInfo hardware = Hardware(
+ RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30));
+
+ LocalInferenceEligibilityResult result =
+ LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, configuredModelId: null);
+
+ Assert.False(result.CanInstall);
+ Assert.Equal(
+ LocalInferenceSelectionFailureCode.NotRecommendedForSku,
+ result.SelectionFailureCode);
+ }
+
+ [Fact]
+ public void EvaluateForConfiguredAvailability_UnknownSavedModelReportsThatModelsFailure()
+ {
+ // The reason must name what is wrong with the saved selection, not fall back to the
+ // SKU's generic no-recommendation message, or setup offers no path to recovery.
+ HostHardwareInfo hardware = Hardware(
+ RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30));
+
+ LocalInferenceEligibilityResult result =
+ LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, "no-such-model-id");
+
+ Assert.False(result.CanInstall);
+ Assert.Equal(LocalInferenceSelectionFailureCode.UnknownModel, result.SelectionFailureCode);
+ Assert.Equal(
+ LocalInferenceUnavailableReasonKind.UnknownModel,
+ LocalInferenceEligibilityDiagnostics.GetUnavailableReason(result).Kind);
+ }
+
+ [Fact]
+ public void EvaluateForConfiguredAvailability_OversizedSavedModelReportsCapacityNotSkuPolicy()
+ {
+ // A saved model that no longer fits must report the capacity shortfall, including the
+ // model name and the required and detected memory the setup page renders.
+ HostHardwareInfo hardware = Hardware(
+ RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-sparkSmall", 12, 12));
+
+ LocalInferenceEligibilityResult result =
+ LocalInferenceEligibility.EvaluateForConfiguredAvailability(
+ hardware,
+ LocalModelCatalog.Qwen38_27BModelId);
+
+ Assert.False(result.CanInstall);
+ Assert.Equal(
+ LocalInferenceEligibilityFailureCode.InsufficientGpuMemory,
+ result.FailureCode);
+ LocalInferenceUnavailableReason reason =
+ LocalInferenceEligibilityDiagnostics.GetUnavailableReason(result);
+ Assert.Equal(LocalInferenceUnavailableReasonKind.InsufficientGpuMemory, reason.Kind);
+ Assert.False(string.IsNullOrWhiteSpace(reason.ModelDisplayName));
+ }
+
+ [Fact]
+ public void EvaluateForConfiguredAvailability_LeavesNonSparkDevicesUnchanged()
+ {
+ HostHardwareInfo hardware = Hardware(
+ RuntimeArchitecture.X64, Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", 32, 32));
+
+ LocalInferenceEligibilityResult withSaved =
+ LocalInferenceEligibility.EvaluateForConfiguredAvailability(
+ hardware, LocalModelCatalog.Qwen27BModelId);
+ LocalInferenceEligibilityResult device = LocalInferenceEligibility.Evaluate(hardware);
+
+ Assert.True(withSaved.CanInstall);
+ Assert.Equal(device.Plan?.Model.Id, withSaved.Plan?.Model.Id);
+ }
+
+ [Fact]
+ public void Evaluate_HugeNonSparkGpuStillDefaultsToTheRecommendedModel()
+ {
+ // A priority-0, IsExplicitAlternative model must never win the generic
+ // default/fallback pick regardless of available memory. The guard in
+ // SelectDefaultModelAndProfile keeps that true as the catalog grows.
+ LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate(
+ Hardware(RuntimeArchitecture.X64, Gpu("NVIDIA arbitrary huge adapter", "GPU-huge", totalGiB: 200, freeGiB: 200)));
+
+ Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status);
+ Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id);
+ Assert.DoesNotContain(
+ LocalModelCatalog.Models,
+ m => m.RecommendationPriority == 0 && m.IsExplicitAlternative && m.Id == result.Plan!.Model.Id);
+ }
+
[Fact]
public void Evaluate_UnsetModelChoosesHighestPriorityModelThatFitsTotalCapacity()
{
@@ -565,4 +824,38 @@ private static GpuInfo Gpu(
CudaMajorVersion: 13,
StableId: stableId);
+ [Theory]
+ [InlineData("b10655-cuda13-x64", "b10655")]
+ [InlineData("b10655-cuda13-arm64", "b10655")]
+ public void FindInstalled_ResolvesRetiredRuntimeSoExistingInstallsStayLaunchable(
+ string runtimeId,
+ string expectedReleaseTag)
+ {
+ // An installation recorded before the runtime bump must keep resolving its own
+ // receipt, otherwise updating the app strands it until setup repairs it.
+ LlamaRuntimeVariant? installed = LlamaRuntimeCatalog.FindInstalled(runtimeId);
+
+ Assert.NotNull(installed);
+ Assert.Equal(runtimeId, installed.Id);
+ Assert.Equal(expectedReleaseTag, installed.ReleaseTag);
+ Assert.False(string.Equals(LlamaRuntimeCatalog.ReleaseTag, installed.ReleaseTag, StringComparison.Ordinal));
+ }
+
+ [Fact]
+ public void RetiredRuntimeIsNeverOfferedForNewInstalls()
+ {
+ Assert.DoesNotContain(
+ LlamaRuntimeCatalog.Variants,
+ variant => variant.ReleaseTag != LlamaRuntimeCatalog.ReleaseTag);
+ Assert.All(
+ LlamaRuntimeCatalog.Variants,
+ variant => Assert.Equal(LlamaRuntimeCatalog.ReleaseTag, variant.ReleaseTag));
+ }
+
+ [Fact]
+ public void FindInstalled_RejectsUnknownRuntimeId()
+ {
+ Assert.Null(LlamaRuntimeCatalog.FindInstalled("b00000-cuda13-x64"));
+ Assert.Null(LlamaRuntimeCatalog.FindInstalled(null));
+ }
}
diff --git a/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs b/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs
index 1ade70466..c191edc98 100644
--- a/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs
+++ b/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs
@@ -249,13 +249,17 @@ public void CapabilitiesReview_InstallDistroCard_UsesSimplifiedCopy()
}
///
- /// The "is Local AI unavailable" gate and the recommended/selected model must be decided
- /// from device-level eligibility (the best catalog model this hardware can run), not from
- /// the currently configured SelectedModelId. A stale/removed model, or one that exists but
- /// this hardware cannot run at all, must be reconciled to a valid one instead of making an
- /// otherwise-capable device look unavailable or leaving setup on a known-incompatible model.
- /// A merely busy GPU (EligibleButBusy) is not reconciled away: CanInstall covers that case
+ /// The recommended model must be decided from device-level eligibility (the best catalog
+ /// model this hardware can run), not from the currently configured SelectedModelId. A
+ /// stale/removed model, or one that exists but this hardware cannot run at all, must be
+ /// reconciled to a valid one instead of leaving setup on a known-incompatible model. A
+ /// merely busy GPU (EligibleButBusy) is not reconciled away: CanInstall covers that case
/// and the same model would still work once the GPU frees up.
+ ///
+ /// The "is Local AI unavailable" gate additionally honours an already configured model.
+ /// A SKU with no recommended default (RTX Spark 32 GB) is a statement about what to
+ /// install by default, not about what the device can run, so a configured selection that
+ /// still passes the capacity fit-test keeps Local AI available on a setup rerun.
///
[Fact]
public void CapabilitiesReview_GatesOnDeviceEligibilityAndReconcilesStaleSelectedModel()
@@ -272,15 +276,18 @@ public void CapabilitiesReview_GatesOnDeviceEligibilityAndReconcilesStaleSelecte
AssertInOrder(
method,
"LocalInferenceEligibilityResult deviceEligibility = LocalInferenceEligibility.Evaluate(_localAiHardware);",
- "if (!deviceEligibility.CanInstall || deviceEligibility.Plan is null || deviceEligibility.SelectedGpu is null)",
- "hardwareReason = DescribeLocalAiUnavailable(deviceEligibility);",
+ "_localAiRecommendedModelId = deviceEligibility.CanInstall",
+ "LocalInferenceEligibility.EvaluateForConfiguredAvailability(",
+ "_config!.LocalAi.SelectedModelId);",
+ "if (!availability.CanInstall || availability.Plan is null || availability.SelectedGpu is null)",
+ "hardwareReason = DescribeLocalAiUnavailable(availability);",
"LocalInferenceEligibilityResult selectedEligibility =",
"LocalInferenceEligibility.Evaluate(_localAiHardware, selectedModelId);",
"if (_localAiRecoveryModelPinned)",
"eligibility = selectedEligibility;",
"else if (!selectedEligibility.CanInstall)",
"_config.LocalAi.SelectedModelId = null;",
- "_config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? deviceEligibility.Plan.Model.Id;",
+ "_config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? availability.Plan.Model.Id;",
"eligibility ??= LocalInferenceEligibility.Evaluate(",
"_config.LocalAi.SelectedModelId);");
}
diff --git a/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs b/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs
index 3c9261513..50a859f9b 100644
--- a/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs
+++ b/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs
@@ -737,6 +737,185 @@ private static async Task WaitForConditionAsync(Func condition, TimeSpan t
}
}
+ ///
+ /// A 32 GB RTX Spark. The fixed SKU table gives this device no recommended default, which
+ /// is a statement about fresh installs rather than about what the device can run.
+ ///
+ private static HostHardwareInfo Create32GbSparkHardware() =>
+ new(
+ Architecture.Arm64,
+ TotalPhysicalMemoryBytes: 128_000_000_000,
+ AvailablePhysicalMemoryBytes: 96_000_000_000,
+ Gpus:
+ [
+ new GpuInfo(
+ GpuVendor.Nvidia,
+ "NVIDIA RTX Spark N1X",
+ GpuVisibleMemoryBytes: 30L * 1024 * 1024 * 1024,
+ FreeGpuVisibleMemoryBytes: 30L * 1024 * 1024 * 1024,
+ DriverVersion: "620.0",
+ CudaMajorVersion: 13,
+ StableId: "GPU-spark32"),
+ ],
+ VulkanAvailable: false);
+
+ private static LocalAiPageViewModel CreateViewModel(
+ LocalAiRuntimeSnapshot snapshot,
+ HostHardwareInfo hardware,
+ out FakeAppCommands commands,
+ out PermissionsPageRuntimeSource gatewaySource)
+ {
+ var runtime = new FakeLocalAiRuntime(snapshot);
+ var runtimeHost = new FakePermissionsPageRuntimeHost
+ {
+ ConnectionSnapshot = GatewayConnectionSnapshot.Idle with
+ {
+ OperatorState = RoleConnectionState.Idle,
+ },
+ };
+ gatewaySource = new PermissionsPageRuntimeSource(runtimeHost);
+ commands = new FakeAppCommands();
+ return new LocalAiPageViewModel(
+ runtime,
+ gatewaySource,
+ commands,
+ new RecordingUiDispatcher(),
+ new FixedHardwareProbe(hardware));
+ }
+
+ ///
+ /// A 32 GB Spark whose managed receipt names a model this hardware still runs keeps the
+ /// Local AI entry point available, so a broken or unverified runtime can reach Retry Setup
+ /// and an installed model can still be changed.
+ ///
+ [Fact]
+ public async Task Spark32Gb_WithValidManagedReceipt_KeepsRetrySetupReachable()
+ {
+ LocalAiRuntimeSnapshot snapshot = CreateInstalledSnapshot() with
+ {
+ ModelId = LocalModelCatalog.Qwen38_27BModelId,
+ State = LocalAiRuntimeState.Failed,
+ ModelEvidence = new LocalAiModelEvidence(
+ LocalAiModelAvailabilityState.Unknown,
+ DateTimeOffset.UtcNow),
+ };
+ using var viewModel = CreateViewModel(
+ snapshot, Create32GbSparkHardware(), out _, out PermissionsPageRuntimeSource source);
+ using (source)
+ {
+ await ActivateAndWaitForAvailabilityAsync(viewModel);
+
+ Assert.True(viewModel.IsAvailabilityKnown);
+ Assert.True(viewModel.IsLocalAiAvailable);
+ Assert.True(viewModel.IsSetupAvailable);
+ Assert.True(viewModel.CanRetrySetup);
+ }
+ }
+
+ ///
+ /// The same 32 GB Spark with no managed installation still receives no default, so Local AI
+ /// stays unavailable for a fresh setup on that SKU.
+ ///
+ [Fact]
+ public async Task Spark32Gb_WithNoManagedReceipt_StaysUnavailable()
+ {
+ LocalAiRuntimeSnapshot snapshot = CreateInstalledSnapshot() with
+ {
+ ModelId = null,
+ State = LocalAiRuntimeState.NotInstalled,
+ Ownership = LocalAiOwnership.None,
+ ModelEvidence = new LocalAiModelEvidence(
+ LocalAiModelAvailabilityState.Unknown,
+ DateTimeOffset.UtcNow),
+ };
+ using var viewModel = CreateViewModel(
+ snapshot, Create32GbSparkHardware(), out _, out PermissionsPageRuntimeSource source);
+ using (source)
+ {
+ await ActivateAndWaitForAvailabilityAsync(viewModel);
+
+ Assert.True(viewModel.IsAvailabilityKnown);
+ Assert.False(viewModel.IsLocalAiAvailable);
+ Assert.False(viewModel.IsSetupAvailable);
+ Assert.False(viewModel.CanRetrySetup);
+ Assert.False(viewModel.CanChangeModel);
+ // NotRecommendedForSku has no dedicated reason kind, so this currently surfaces the
+ // generic Unknown copy. Asserted so the behavior is recorded rather than assumed.
+ Assert.Equal(
+ LocalInferenceUnavailableReasonKind.Unknown,
+ viewModel.LocalAiUnavailableReason?.Kind);
+ }
+ }
+
+ ///
+ /// The runtime refresh can publish the managed receipt after the availability probe has
+ /// already read the earlier snapshot. Availability must be recomputed when that happens,
+ /// otherwise a first visit leaves a 32 GB Spark fixed at NotRecommendedForSku with Retry
+ /// Setup and Change Model disabled and Recheck unavailable, until the page is reopened.
+ ///
+ [Fact]
+ public async Task Spark32Gb_WhenReceiptArrivesAfterAvailability_RecomputesAndKeepsRetrySetupReachable()
+ {
+ LocalAiRuntimeSnapshot pending = CreateInstalledSnapshot() with
+ {
+ ModelId = null,
+ State = LocalAiRuntimeState.Failed,
+ ModelEvidence = new LocalAiModelEvidence(
+ LocalAiModelAvailabilityState.Unknown,
+ DateTimeOffset.UtcNow),
+ };
+ var refreshGate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously);
+ var runtime = new FakeLocalAiRuntime(pending)
+ {
+ RefreshDelay = refreshGate.Task,
+ RefreshResult = pending with { ModelId = LocalModelCatalog.Qwen38_27BModelId },
+ };
+ using var gatewaySource = new PermissionsPageRuntimeSource(new FakePermissionsPageRuntimeHost());
+ using var viewModel = new LocalAiPageViewModel(
+ runtime,
+ gatewaySource,
+ new FakeAppCommands(),
+ new RecordingUiDispatcher(),
+ new FixedHardwareProbe(Create32GbSparkHardware()));
+
+ await ActivateAndWaitForAvailabilityAsync(viewModel);
+
+ // The receipt has not been published yet, so the SKU alone leaves the entry point closed.
+ Assert.False(viewModel.IsLocalAiAvailable);
+ Assert.False(viewModel.CanRetrySetup);
+
+ refreshGate.TrySetResult();
+ await WaitForAsync(viewModel, () => viewModel.IsLocalAiAvailable);
+
+ Assert.True(viewModel.IsSetupAvailable);
+ Assert.True(viewModel.CanRetrySetup);
+ }
+
+ ///
+ /// A receipt naming a model this catalog no longer knows reports that model's own failure,
+ /// so the page explains what to fix instead of showing the SKU's generic message.
+ ///
+ [Fact]
+ public async Task Spark32Gb_WithUnknownReceiptModel_ReportsThatModelsReason()
+ {
+ LocalAiRuntimeSnapshot snapshot = CreateInstalledSnapshot() with
+ {
+ ModelId = "no-such-model-id",
+ };
+ using var viewModel = CreateViewModel(
+ snapshot, Create32GbSparkHardware(), out _, out PermissionsPageRuntimeSource source);
+ using (source)
+ {
+ await ActivateAndWaitForAvailabilityAsync(viewModel);
+
+ Assert.True(viewModel.IsAvailabilityKnown);
+ Assert.False(viewModel.IsLocalAiAvailable);
+ Assert.Equal(
+ LocalInferenceUnavailableReasonKind.UnknownModel,
+ viewModel.LocalAiUnavailableReason?.Kind);
+ }
+ }
+
private static HostHardwareInfo CreateQualifiedHardware() =>
new(
Architecture.X64,
@@ -954,8 +1133,21 @@ public Task RestartAsync(CancellationToken cancellationT
return Task.FromResult(Snapshot);
}
- public Task RefreshAsync(CancellationToken cancellationToken = default) =>
- Task.FromResult(Snapshot);
+ /// Snapshot the refresh publishes, letting a test model a receipt that the runtime
+ /// only resolves after construction.
+ public LocalAiRuntimeSnapshot? RefreshResult { get; init; }
+
+ /// Holds the refresh open until this completes, so a test can land the refresh
+ /// after the availability probe has already read the earlier snapshot.
+ public Task? RefreshDelay { get; init; }
+
+ public async Task RefreshAsync(CancellationToken cancellationToken = default)
+ {
+ if (RefreshDelay is { } delay)
+ await delay.ConfigureAwait(false);
+ Snapshot = RefreshResult ?? Snapshot;
+ return Snapshot;
+ }
public ValueTask DisposeAsync() => ValueTask.CompletedTask;
}