diff --git a/docs/SETUP_ENGINE_REDESIGN.md b/docs/SETUP_ENGINE_REDESIGN.md index 26ed96f5a..24e15a033 100644 --- a/docs/SETUP_ENGINE_REDESIGN.md +++ b/docs/SETUP_ENGINE_REDESIGN.md @@ -272,6 +272,17 @@ hub-cache snapshot as the active model path, while preserving the legacy compatibility path and the prior gateway fallback, install time, and rollback metadata. +Runtime upgrades validate the installed executable against its recorded runtime +release, not the current catalog release. A verified schema-3 model is migrated +to the hub cache before reuse by the new runtime, without downloading it again. +Normal setup keeps the pre-upgrade receipt separate from gateway recovery state. +If a later step fails, receipt persistence restores that baseline before runtime +acquisition removes the newly installed runtime. Reconciliation also restores +the baseline when setup fails after migration but before receipt persistence. +The old runtime, compatibility model, and verified shared-cache copy are retained. +Superseded runtime directories remain until explicit uninstall; upgrades do not +prune the rollback baseline. + Completed cache files and pre-existing resumable partials are shared state. Setup rollback and uninstall do not delete them. Unsafe links, reparse points, hard-linked partials, destination conflicts, receipt mismatches, and concurrent diff --git a/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs b/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs index 49f9d558e..33cbb2305 100644 --- a/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs +++ b/src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs @@ -31,19 +31,27 @@ public static LlamaServerRouterLaunchPlan Build( LocalAiPaths paths, LocalAiResolvedInstall install, int? listenPort = null) => - BuildCore(paths, install, install.ModelPath, listenPort); + BuildCore(paths, install, install.ModelPath, verifiedDraftModelPath: null, listenPort); + /// + /// The draft checkpoint's handle-resolved physical path, from the same verification + /// that opened it. Passing the persisted snapshot path instead would let a + /// snapshot-link replacement change the file llama-server finally opens, which is + /// exactly what resolving the primary model through its own handle prevents. + /// internal static LlamaServerRouterLaunchPlan BuildForVerifiedRuntime( LocalAiPaths paths, LocalAiResolvedInstall install, string verifiedModelPath, + string? verifiedDraftModelPath, int? listenPort = null) => - BuildCore(paths, install, verifiedModelPath, listenPort); + BuildCore(paths, install, verifiedModelPath, verifiedDraftModelPath, listenPort); private static LlamaServerRouterLaunchPlan BuildCore( LocalAiPaths paths, LocalAiResolvedInstall install, string modelPath, + string? verifiedDraftModelPath, int? listenPort) { ArgumentNullException.ThrowIfNull(paths); @@ -53,13 +61,16 @@ private static LlamaServerRouterLaunchPlan BuildCore( LocalAiInstallManifest manifest = install.Manifest; int port = listenPort ?? manifest.RequestedPort; LocalAiPortPolicy.Validate(port); - LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.Variants.SingleOrDefault( - candidate => string.Equals(candidate.Id, manifest.RuntimeId, StringComparison.Ordinal)) + // FindInstalled, not Variants: an installation recorded before the last + // runtime bump must keep launching until setup upgrades it, instead of being + // stranded the moment the catalog moves to a newer pinned release. + LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.FindInstalled(manifest.RuntimeId) ?? throw new InvalidDataException("The managed llama-server runtime is no longer qualified."); LocalModelInfo model = LocalModelCatalog.FindInstalled(manifest.ModelCatalogId) ?? throw new InvalidDataException("The managed local AI model is no longer qualified."); LocalInferenceRunProfile profile = ResolveQualifiedReceipt(manifest, runtime, model); + string? draftModelPath = ResolveDraftModelPath(manifest, model, verifiedDraftModelPath); string presetPath = paths.ResolveContainedPath( Path.GetRelativePath(paths.RootDirectory, paths.RouterPresetPath), @@ -84,7 +95,7 @@ private static LlamaServerRouterLaunchPlan BuildCore( .WithComparers(StringComparer.OrdinalIgnoreCase) .Add("CUDA_VISIBLE_DEVICES", manifest.SelectedGpuId), presetPath, - BuildPreset(model, profile, modelPath), + BuildPreset(model, profile, modelPath, draftModelPath), model.Id); } @@ -120,7 +131,7 @@ internal static void ValidateArtifactReceipts( { throw new InvalidDataException("The managed local AI architecture and runtime receipt do not match."); } - if (!string.Equals(manifest.EngineVersion, LlamaRuntimeCatalog.ReleaseTag, StringComparison.Ordinal) || + if (!string.Equals(manifest.EngineVersion, runtime.ReleaseTag, StringComparison.Ordinal) || !string.Equals(manifest.ModelAlias, model.Id, StringComparison.Ordinal)) { throw new InvalidDataException("The managed local AI model recipe receipt does not match the qualified catalog."); @@ -145,17 +156,65 @@ internal static void ValidateArtifactReceipts( { throw new InvalidDataException("The managed model artifact receipt does not match the qualified catalog."); } + + ImmutableArray expectedAdditionalArtifacts = LocalModelCatalog.AdditionalArtifacts(model); + if (manifest.AdditionalModelAssetsOrEmpty.Length != expectedAdditionalArtifacts.Length || + manifest.AdditionalModelPathsOrEmpty.Length != expectedAdditionalArtifacts.Length) + { + throw new InvalidDataException( + "The managed additional model asset receipts do not match the qualified catalog."); + } + for (int i = 0; i < expectedAdditionalArtifacts.Length; i++) + { + PinnedArtifact artifact = expectedAdditionalArtifacts[i]; + LocalAiAssetReceipt receipt = manifest.AdditionalModelAssetsOrEmpty[i]; + if (!string.Equals(receipt.FileName, Path.GetFileName(artifact.RelativePath), StringComparison.Ordinal) || + receipt.SizeBytes != artifact.SizeBytes || + !string.Equals(receipt.Sha256, artifact.Sha256.Value, StringComparison.Ordinal) || + !string.Equals(receipt.SourceUrl, artifact.DownloadUri.AbsoluteUri, StringComparison.Ordinal)) + { + throw new InvalidDataException( + "The managed additional model asset receipts do not match the qualified catalog."); + } + } + } + + /// + /// The DFlash draft checkpoint's path for the preset. Prefers the handle-resolved + /// physical path supplied by the caller that verified and still holds the file, so a + /// snapshot-link replacement cannot change the identity llama-server opens. Falls + /// back to the persisted receipt path only for callers that do not verify first + /// (, used for inspection rather than launch). Null for recipes + /// with no separate draft checkpoint. Callers must validate the manifest via + /// first, which guarantees + /// AdditionalModelPaths has one entry per catalog-pinned artifact. + /// + private static string? ResolveDraftModelPath( + LocalAiInstallManifest manifest, + LocalModelInfo model, + string? verifiedDraftModelPath) + { + if (model.Recipe.DraftWeights is null) + return null; + return string.IsNullOrWhiteSpace(verifiedDraftModelPath) + ? manifest.AdditionalModelPathsOrEmpty[^1] + : verifiedDraftModelPath; } private static string BuildPreset( LocalModelInfo model, LocalInferenceRunProfile profile, - string modelPath) + string modelPath, + string? draftModelPath) { if (modelPath.IndexOfAny(['\r', '\n']) >= 0) throw new InvalidDataException("The managed model path cannot be represented safely in a llama-server preset."); + if (draftModelPath is not null && draftModelPath.IndexOfAny(['\r', '\n']) >= 0) + throw new InvalidDataException("The managed draft model path cannot be represented safely in a llama-server preset."); LocalModelRunRecipe recipe = model.Recipe; + if (recipe.SpeculativeDecoding == SpeculativeDecodingMode.DraftDFlash && draftModelPath is null) + throw new InvalidDataException("Draft-flash decoding requires a resolved draft model path."); ModelSamplingPreset sampling = recipe.Sampling; var preset = new StringBuilder(); preset.AppendLine("version = 1"); @@ -178,9 +237,24 @@ private static string BuildPreset( preset.AppendLine("main-gpu = 0"); preset.AppendLine("fit = off"); preset.AppendLine("load-mode = dio"); - preset.AppendLine("spec-type = draft-mtp"); - preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens)); - preset.AppendLine("spec-draft-backend-sampling = true"); + switch (recipe.SpeculativeDecoding) + { + case SpeculativeDecodingMode.DraftMtp: + preset.AppendLine("spec-type = draft-mtp"); + preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens)); + preset.AppendLine("spec-draft-backend-sampling = true"); + break; + case SpeculativeDecodingMode.DraftDFlash: + preset.AppendLine("spec-type = draft-dflash"); + preset.Append("spec-draft-model = ").AppendLine(draftModelPath); + preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens)); + preset.AppendLine("spec-draft-backend-sampling = true"); + break; + case SpeculativeDecodingMode.None: + break; + default: + throw new ArgumentOutOfRangeException(nameof(recipe.SpeculativeDecoding)); + } preset.Append("temperature = ").AppendLine(Invariant(sampling.Temperature)); preset.Append("top-k = ").AppendLine(Invariant(sampling.TopK)); preset.Append("top-p = ").AppendLine(Invariant(sampling.TopP)); diff --git a/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs b/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs index ff5a0bad8..3828b4cc6 100644 --- a/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs +++ b/src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs @@ -118,6 +118,7 @@ public sealed class LlamaServerRuntimeService : ILocalAiRuntime private LocalAiRuntimeSnapshot _snapshot; private ILocalAiManagedProcess? _managedProcess; private LocalAiVerifiedModelLease? _verifiedModel; + private readonly List _verifiedAdditionalAssets = []; private string? _runtimeModelPath; private LocalAiResolvedInstall? _install; private long _generation; @@ -378,6 +379,7 @@ private async Task EnsureStartedCoreAsync(CancellationTo _options.Paths, install, GetRuntimeModelPath(install), + GetRuntimeDraftModelPath(), requestedPort); await WritePresetAtomicallyAsync(launchPlan, cancellationToken).ConfigureAwait(false); } @@ -862,7 +864,7 @@ private async Task ValidateInstalledFilesAsync( DisposeVerifiedModelHandle(); ValidateInstalledFilesForStatus(install); - if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (!install.Manifest.UsesHubCache) { _runtimeModelPath = install.ModelPath; return; @@ -883,6 +885,33 @@ await _modelFileVerifier.TryOpenAsync( "The shared Hugging Face cache model is unsafe or no longer matches its receipt."); } + // Schema-5 extra assets (a DFlash draft checkpoint) are loaded natively + // by llama-server exactly like the primary weights, and they + // live in the same shared, user-writable hub cache. Rehash them here and hold + // the handles for the process lifetime, so a file swapped after setup cannot + // reach the loader with only a structural path check behind it. + foreach ((LocalAiAssetReceipt receipt, string cachedPath) in + install.Manifest.AdditionalModelAssetsOrEmpty + .Zip(install.Manifest.AdditionalModelPathsOrEmpty)) + { + LocalAiVerifiedModelLease? verifiedAsset = + await _modelFileVerifier.TryOpenAsync( + install.Manifest.ModelCacheRoot!, + cachedPath, + receipt.SizeBytes, + new Sha256Digest(receipt.Sha256), + cancellationToken) + .ConfigureAwait(false); + if (verifiedAsset is null) + { + DisposeVerifiedModelHandle(); + throw new InvalidDataException( + $"The shared Hugging Face cache asset '{receipt.FileName}' is unsafe or no longer matches its receipt."); + } + + _verifiedAdditionalAssets.Add(verifiedAsset); + } + _runtimeModelPath = _verifiedModel.ResolvedPath; } @@ -890,7 +919,7 @@ private static void ValidateInstalledFilesForStatus(LocalAiResolvedInstall insta { if (!File.Exists(install.ExecutablePath)) throw new InvalidDataException("The managed llama-server executable is missing."); - if (install.Manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (install.Manifest.UsesHubCache) { if (!File.Exists(install.ModelPath)) throw new InvalidDataException("The managed GGUF model is missing."); @@ -1666,14 +1695,26 @@ private void DisposeVerifiedModelHandle() { _verifiedModel?.Dispose(); _verifiedModel = null; + foreach (LocalAiVerifiedModelLease lease in _verifiedAdditionalAssets) + lease.Dispose(); + _verifiedAdditionalAssets.Clear(); _runtimeModelPath = null; } + /// + /// The handle-resolved path of the draft checkpoint this process verified and still + /// holds open, or null when the recipe has no additional assets. Additional assets are + /// verified in catalog order and the draft checkpoint is always last, matching + /// . + /// + private string? GetRuntimeDraftModelPath() => + _verifiedAdditionalAssets.Count == 0 ? null : _verifiedAdditionalAssets[^1].ResolvedPath; + private string GetRuntimeModelPath(LocalAiResolvedInstall install) { if (_runtimeModelPath is not null) return _runtimeModelPath; - if (install.Manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (install.Manifest.UsesHubCache) throw new InvalidOperationException("The verified shared-cache model identity is unavailable."); return install.ModelPath; } diff --git a/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs b/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs index ce3cbe886..895ab908c 100644 --- a/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs +++ b/src/OpenClaw.Connection/LocalAi/LocalAiManifest.cs @@ -148,8 +148,23 @@ public sealed record LocalAiInstallManifest /// the model must also verify the pinned content before use. /// public const int HubCacheReceiptSchemaVersion = 4; + /// + /// Schema-4 plus one or more additional model assets verified in the same + /// hub cache. Only recipes with a LocalModelRunRecipe.DraftWeights + /// pin use this; every existing single-asset recipe keeps writing schema 4. + /// + public const int AdditionalAssetsSchemaVersion = 5; public const string SupportedEngine = "llama-server"; + /// + /// True for schema 4 and schema 5, whose active model resolves through the + /// standard Hugging Face hub cache rather than the legacy app-owned copy. + /// Derived entirely from , so it must never be + /// persisted -- an older app build would reject it as an unknown field. + /// + [JsonIgnore] + public bool UsesHubCache => SchemaVersion is HubCacheReceiptSchemaVersion or AdditionalAssetsSchemaVersion; + public int SchemaVersion { get; init; } = CurrentSchemaVersion; public string Engine { get; init; } = SupportedEngine; public required string EngineVersion { get; init; } @@ -181,6 +196,47 @@ public sealed record LocalAiInstallManifest public required string ModelAlias { get; init; } public required LocalAiAssetReceipt ModelAsset { get; init; } /// + /// Schema-5 receipts for additional model assets verified in the hub + /// cache alongside (the draft checkpoint), in + /// catalog order. Left at its unset default + /// (not .Empty) for schema-3/4 manifests, since + /// ImmutableArray<T>.Empty is a distinct, non-default instance + /// that would not omit + /// -- writing it would break older app builds' strict unknown-field + /// rejection on an otherwise-unchanged schema-4 receipt. + /// normalizes the + /// unset default to .Empty for every in-memory reader. + /// + [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingDefault)] + public ImmutableArray AdditionalModelAssets { get; init; } + /// + /// Verified hub-cache paths parallel to . + /// Same unset-default-not-Empty rule; see that property's remarks. + /// + [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingDefault)] + public ImmutableArray AdditionalModelPaths { get; init; } + /// + /// with the unset default collapsed to + /// an empty array. Read through this, never the raw property: a schema-3/4 + /// or hand-edited manifest leaves the raw value at ImmutableArray's + /// default (null-backed) instance, where Length/IsEmpty throw + /// NullReferenceException instead of the intended InvalidDataException. + /// Normalizing on read rather than rewriting the record keeps the persisted + /// JSON byte-identical -- assigning .Empty back onto the manifest + /// would make the next save emit the field on an otherwise-untouched + /// schema-4 receipt. + /// + [JsonIgnore] + public ImmutableArray AdditionalModelAssetsOrEmpty => + AdditionalModelAssets.IsDefault ? ImmutableArray.Empty : AdditionalModelAssets; + /// + /// with the unset default collapsed to + /// an empty array; see . + /// + [JsonIgnore] + public ImmutableArray AdditionalModelPathsOrEmpty => + AdditionalModelPaths.IsDefault ? ImmutableArray.Empty : AdditionalModelPaths; + /// /// The requested listener port. Zero delegates allocation to llama-server so /// the child owns the port continuously from bind through startup. /// @@ -512,7 +568,8 @@ public LocalAiResolvedInstall ResolveAndValidate(LocalAiInstallManifest manifest ArgumentNullException.ThrowIfNull(manifest); if (manifest.SchemaVersion is not ( LocalAiInstallManifest.CurrentSchemaVersion or - LocalAiInstallManifest.HubCacheReceiptSchemaVersion)) + LocalAiInstallManifest.HubCacheReceiptSchemaVersion or + LocalAiInstallManifest.AdditionalAssetsSchemaVersion)) { throw new InvalidDataException($"Unsupported local AI manifest schema version {manifest.SchemaVersion}."); } @@ -566,10 +623,11 @@ LocalAiInstallManifest.CurrentSchemaVersion or ValidateHubCacheReceipt(manifest, provenance); string legacyModel = _paths.ResolveContainedPath(manifest.ModelPath, nameof(manifest.ModelPath)); ValidateModelPath(legacyModel, manifest.ModelAsset, "legacy-compatible"); - string model = manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion + string model = manifest.UsesHubCache ? ResolveHubCacheModelPath(manifest) : legacyModel; ValidateModelPath(model, manifest.ModelAsset, "active"); + ValidateAdditionalAssets(manifest); LocalAiPortPolicy.Validate(manifest.RequestedPort); LocalAiGatewayModelPolicy.ValidateFallbackModel(manifest.GatewayFallbackModel); @@ -746,6 +804,99 @@ private static void ValidateHubCacheReceipt( private static string ResolveHubCacheModelPath(LocalAiInstallManifest manifest) => WindowsPathSafety.NormalizePath(manifest.CachedModelPath!); + /// + /// Validates schema-5 additional assets (a DFlash draft checkpoint). Each + /// receipt derives its own repository and + /// revision from its own SourceUrl -- unlike the primary + /// , additional assets are + /// not required to share the primary model's repository (the DFlash + /// draft checkpoint is pinned from a different one). + /// + private static void ValidateAdditionalAssets(LocalAiInstallManifest manifest) + { + if (manifest.SchemaVersion != LocalAiInstallManifest.AdditionalAssetsSchemaVersion) + { + if (!manifest.AdditionalModelAssets.IsDefaultOrEmpty || !manifest.AdditionalModelPaths.IsDefaultOrEmpty) + { + throw new InvalidDataException( + "Only schema-5 local AI manifests may record additional model assets."); + } + return; + } + + if (manifest.AdditionalModelAssets.IsDefaultOrEmpty || + manifest.AdditionalModelAssetsOrEmpty.Length != manifest.AdditionalModelPathsOrEmpty.Length) + { + throw new InvalidDataException( + "A schema-5 local AI manifest must record a matching additional-asset receipt and cache path pair."); + } + + var seenFileNames = new HashSet(StringComparer.OrdinalIgnoreCase) { manifest.ModelAsset.FileName }; + for (int i = 0; i < manifest.AdditionalModelAssetsOrEmpty.Length; i++) + { + LocalAiAssetReceipt receipt = manifest.AdditionalModelAssetsOrEmpty[i]; + ValidateAssetReceipt(receipt, $"{nameof(manifest.AdditionalModelAssets)}[{i}]"); + if (!seenFileNames.Add(receipt.FileName)) + throw new InvalidDataException("The local AI manifest additional asset filenames must be unique."); + + HuggingFaceModelProvenance provenance = ParseHuggingFaceProvenance(receipt); + if (!HuggingFaceHubCache.TryGetSnapshotPaths( + manifest.ModelCacheRoot!, + provenance.RepositoryId, + provenance.Revision, + provenance.RelativePath, + out string expectedPath, + out _, + out string error) || + !string.Equals(manifest.AdditionalModelPathsOrEmpty[i], expectedPath, StringComparison.OrdinalIgnoreCase)) + { + throw new InvalidDataException( + string.IsNullOrWhiteSpace(error) + ? "The local AI manifest additional asset cache receipt is invalid." + : error); + } + } + } + + private static HuggingFaceModelProvenance ParseHuggingFaceProvenance(LocalAiAssetReceipt receipt) + { + var source = new Uri(receipt.SourceUrl, UriKind.Absolute); + if (!string.Equals(source.Scheme, Uri.UriSchemeHttps, StringComparison.OrdinalIgnoreCase) || + !string.Equals(source.Host, "huggingface.co", StringComparison.OrdinalIgnoreCase) || + source.Query is not ("" or "?download=true")) + { + throw new InvalidDataException("The local AI manifest asset source must be an immutable Hugging Face resolve URL."); + } + + string[] pathSegments = Uri.UnescapeDataString(source.AbsolutePath).Split( + '/', StringSplitOptions.RemoveEmptyEntries); + int resolveIndex = Array.IndexOf(pathSegments, "resolve"); + if (resolveIndex != 2 || pathSegments.Length < resolveIndex + 3) + { + throw new InvalidDataException("The local AI manifest asset source must be an immutable Hugging Face resolve URL."); + } + + string repositoryId = $"{pathSegments[0]}/{pathSegments[1]}"; + string revision = pathSegments[resolveIndex + 1]; + if (revision.Length != 40 || + revision.Any(character => character is not (>= '0' and <= '9' or >= 'a' and <= 'f'))) + { + throw new InvalidDataException( + "The local AI manifest asset revision must be a lowercase 40-character commit digest."); + } + + string[] relativeSegments = pathSegments[(resolveIndex + 2)..]; + string relativePath = string.Join('/', relativeSegments); + if (relativeSegments.Any(segment => !WindowsPathSafety.IsSafeSegment(segment)) || + !string.Equals(relativeSegments[^1], receipt.FileName, StringComparison.Ordinal)) + { + throw new InvalidDataException( + "The local AI manifest asset source must match its own repository, revision, and filename."); + } + + return new HuggingFaceModelProvenance(repositoryId, revision, relativePath); + } + internal sealed record HuggingFaceModelProvenance( string RepositoryId, string Revision, diff --git a/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs b/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs index 26855297d..59f8cf1cf 100644 --- a/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs +++ b/src/OpenClaw.SetupEngine.UI/Pages/CapabilitiesPage.xaml.cs @@ -330,9 +330,18 @@ private async Task InitializeLocalAiReviewAsync( ? deviceEligibility.Plan?.Model.Id : null; - if (!deviceEligibility.CanInstall || deviceEligibility.Plan is null || deviceEligibility.SelectedGpu is null) + // A SKU with no recommended default (RTX Spark 32 GB) still runs an already + // configured model. Gate availability on that configured selection when there is + // one, so rerunning setup does not switch Local AI off on a working machine. + // _localAiRecommendedModelId stays null so nothing is labelled Recommended. + LocalInferenceEligibilityResult availability = + LocalInferenceEligibility.EvaluateForConfiguredAvailability( + _localAiHardware, + _config!.LocalAi.SelectedModelId); + + if (!availability.CanInstall || availability.Plan is null || availability.SelectedGpu is null) { - hardwareReason = DescribeLocalAiUnavailable(deviceEligibility); + hardwareReason = DescribeLocalAiUnavailable(availability); } else { @@ -343,7 +352,7 @@ private async Task InitializeLocalAiReviewAsync( // model instead of leaving setup stuck on a known-incompatible selection. A // merely busy GPU (EligibleButBusy) is not reconciled away: the same model would // still work once the GPU frees up, and CanInstall already covers that case. - if (_config!.LocalAi.SelectedModelId is { } selectedModelId) + if (_config.LocalAi.SelectedModelId is { } selectedModelId) { LocalInferenceEligibilityResult selectedEligibility = LocalInferenceEligibility.Evaluate(_localAiHardware, selectedModelId); @@ -356,7 +365,7 @@ private async Task InitializeLocalAiReviewAsync( _config.LocalAi.SelectedModelId = null; } } - _config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? deviceEligibility.Plan.Model.Id; + _config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? availability.Plan.Model.Id; eligibility ??= LocalInferenceEligibility.Evaluate( _localAiHardware, @@ -670,7 +679,7 @@ private void PopulateLocalAiModels() LocalAiModelSelector.Items.Add(new ComboBoxItem { Content = $"{SetupReviewSummaryBuilder.DisplayModelName(model)} " + - $"({FormatSize(model.Weights.SizeBytes)}, " + + $"({FormatSize(LocalModelCatalog.TotalDownloadSizeBytes(model))}, " + $"{FormatContext(plan.Profile.ContextTokens)}, " + $"{LocalModelCatalog.ToDisplayCacheType(plan.Profile.KeyCachePrecision)} KV)" + (isRecommended ? " (Recommended)" : string.Empty), @@ -780,7 +789,7 @@ private void UpdateLocalAiModelDetails() "loads on first request"; LocalAiModelDetailText.Text = $"{SetupReviewSummaryBuilder.DisplayModelName(plan.Model)}, " + - $"{FormatSize(plan.Model.Weights.SizeBytes)} from Hugging Face"; + $"{FormatSize(LocalModelCatalog.TotalDownloadSizeBytes(plan.Model))} from Hugging Face"; UpdatePrimaryButtonState(); } diff --git a/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.AdditionalAssets.cs b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.AdditionalAssets.cs new file mode 100644 index 000000000..060e34dde --- /dev/null +++ b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.AdditionalAssets.cs @@ -0,0 +1,200 @@ +using OpenClaw.Connection.LocalAi; +using OpenClaw.Shared.Inference.Catalog; + +namespace OpenClaw.SetupEngine; + +/// +/// Acquisition for additional model assets (a DFlash draft checkpoint) +/// verified into the same standard Hugging Face hub cache uses for the +/// primary weights. Reuses the same resumable download, hash verification, +/// and safe-cache-directory promotion helpers; the one thing it deliberately +/// omits is the legacy app-owned compatibility copy, since every recipe that +/// pins an additional asset is new -- there is no pre-hub-cache install of +/// it to stay compatible with. +/// +internal sealed partial class HuggingFaceModelInstaller +{ + public async Task InstallAdditionalAssetAsync( + string localDataDirectory, + PinnedArtifact artifact, + IProgress? progress, + CancellationToken cancellationToken) + { + ArgumentException.ThrowIfNullOrWhiteSpace(localDataDirectory); + ArgumentNullException.ThrowIfNull(artifact); + if (artifact.Role != ArtifactRole.ModelWeights || artifact.Source is not HuggingFaceRevisionSource source) + { + throw new HuggingFaceModelInstallException( + "An additional Local AI model asset must be an immutable Hugging Face weights artifact."); + } + + string cacheRoot = _cacheRootResolver(); + if (!TryValidateCacheRootOwnershipBoundary(localDataDirectory, cacheRoot, out string cacheRootError)) + throw new HuggingFaceModelInstallException(cacheRootError); + if (!HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, + source.RepositoryId, + source.RevisionSha, + artifact.RelativePath, + out string modelPath, + out string partialPath, + out string pathError)) + { + throw new HuggingFaceModelInstallException(pathError); + } + + if (Directory.Exists(modelPath)) + throw new HuggingFaceModelInstallException("The managed Local AI model path is an existing directory."); + if (Directory.Exists(partialPath)) + throw new HuggingFaceModelInstallException("The managed Local AI partial model path is an existing directory."); + + await using (FileStream? verified = + await HuggingFaceHubCache.TryOpenVerifiedCacheFileAsync( + cacheRoot, + modelPath, + artifact.SizeBytes, + artifact.Sha256, + new VerificationProgress(this, progress, artifact.SizeBytes), + cancellationToken) + .ConfigureAwait(false)) + { + if (verified is not null) + { + return new HuggingFaceAdditionalAssetInstallResult( + modelPath, cacheRoot, HuggingFaceModelInstallDisposition.ReusedVerified, CreatedThisRun: false); + } + } + + if (PathEntryExists(modelPath)) + { + throw new HuggingFaceModelInstallException( + $"The Hugging Face cache destination '{modelPath}' is unsafe or does not match " + + "the pinned model. Remove it manually and retry setup."); + } + + string destinationDirectory = Path.GetDirectoryName(modelPath) + ?? throw new HuggingFaceModelInstallException( + "The Hugging Face cache destination has no parent directory."); + string destinationFileName = Path.GetFileName(modelPath); + string partialFileName = Path.GetFileName(partialPath); + LocalAiManifestMigration.SafeCacheDirectory? directory = null; + LocalAiManifestMigration.CacheMigrationFile? partial = null; + bool partialExistedBeforeInstall = false; + HuggingFaceModelInstallDisposition disposition = HuggingFaceModelInstallDisposition.Downloaded; + try + { + directory = LocalAiManifestMigration.SafeCacheDirectory.OpenOrCreate(cacheRoot, destinationDirectory); + partial = directory.TryOpenExisting(partialFileName); + partialExistedBeforeInstall = partial is not null; + + bool partialVerified = partial is not null && + partial.Stream.Length == artifact.SizeBytes && + await VerifyOpenFileAsync(partial.Stream, artifact, progress, cancellationToken).ConfigureAwait(false); + if (partial is not null && partial.Stream.Length >= artifact.SizeBytes && !partialVerified) + { + partial.Stream.SetLength(0); + partial.Stream.Position = 0; + } + + if (!partialVerified) + { + partial ??= directory.CreateNew(partialFileName); + if (partial.Stream.Length == 0 && + await TryCopyVerifiedBlobAsync( + cacheRoot, source.RepositoryId, artifact, partial.Stream, progress, cancellationToken) + .ConfigureAwait(false)) + { + disposition = HuggingFaceModelInstallDisposition.ReusedVerified; + } + else + { + await DownloadAndVerifyAsync(artifact, partial.Stream, progress, cancellationToken) + .ConfigureAwait(false); + } + } + + cancellationToken.ThrowIfCancellationRequested(); + LocalAiManifestMigration.CacheMigrationFile activePartial = partial + ?? throw new HuggingFaceModelInstallException("The Hugging Face cache partial was not created."); + if (!HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, + source.RepositoryId, + source.RevisionSha, + artifact.RelativePath, + out string revalidatedModelPath, + out string revalidatedPartialPath, + out pathError) || + !string.Equals(modelPath, revalidatedModelPath, StringComparison.OrdinalIgnoreCase) || + !string.Equals(partialPath, revalidatedPartialPath, StringComparison.OrdinalIgnoreCase)) + { + throw new HuggingFaceModelInstallException( + string.IsNullOrWhiteSpace(pathError) + ? "The Local AI model paths changed before promotion." + : pathError); + } + + if (!await VerifyOpenFileAsync(activePartial.Stream, artifact, progress, cancellationToken) + .ConfigureAwait(false)) + { + throw new HuggingFaceModelInstallException( + "The Hugging Face cache partial does not match the pinned model."); + } + + try + { + activePartial.Promote(destinationFileName); + } + catch (IOException ex) + { + if (partialExistedBeforeInstall) + activePartial.Commit(); + activePartial.Dispose(); + partial = null; + await using FileStream? winner = + await HuggingFaceHubCache.TryOpenVerifiedCacheFileAsync( + cacheRoot, + modelPath, + artifact.SizeBytes, + artifact.Sha256, + new VerificationProgress(this, progress, artifact.SizeBytes), + cancellationToken) + .ConfigureAwait(false); + if (winner is null) + { + throw new HuggingFaceModelInstallException( + "The Hugging Face cache destination changed before promotion.", ex); + } + + return new HuggingFaceAdditionalAssetInstallResult( + modelPath, cacheRoot, HuggingFaceModelInstallDisposition.ReusedVerified, CreatedThisRun: false); + } + + if (partialExistedBeforeInstall) + activePartial.Commit(); + directory.RequirePromotedFile(activePartial.Stream.SafeFileHandle, destinationFileName); + activePartial.Commit(); + return new HuggingFaceAdditionalAssetInstallResult(modelPath, cacheRoot, disposition, CreatedThisRun: true); + } + catch (OperationCanceledException) + { + partial?.Commit(); + throw; + } + catch (Exception exception) when ( + exception is IOException or HttpRequestException or TransientHuggingFaceModelInstallException) + { + partial?.Commit(); + throw; + } + catch (HuggingFaceModelInstallException) when (partialExistedBeforeInstall) + { + partial?.Commit(); + throw; + } + finally + { + partial?.Dispose(); + directory?.Dispose(); + } + } +} diff --git a/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs index 1e5d53fee..080e8ed22 100644 --- a/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs +++ b/src/OpenClaw.SetupEngine/HuggingFaceModelInstaller.cs @@ -57,6 +57,13 @@ public TransientHuggingFaceModelInstallException(string message) } } +/// A verified additional model asset (DFlash draft checkpoint). +internal sealed record HuggingFaceAdditionalAssetInstallResult( + string ModelPath, + string CacheRoot, + HuggingFaceModelInstallDisposition Disposition, + bool CreatedThisRun); + internal interface IHuggingFaceModelAcquirer { Task InstallAsync( @@ -66,6 +73,20 @@ Task InstallAsync( IProgress? progress, CancellationToken cancellationToken); + /// + /// Verifies or acquires one additional model asset (a DFlash draft + /// checkpoint) into the same hub cache + /// uses for the primary weights. Unlike the + /// primary weights, additional assets have no legacy app-owned + /// compatibility copy -- every recipe that uses one is new since the hub + /// cache became the primary store. + /// + Task InstallAdditionalAssetAsync( + string localDataDirectory, + PinnedArtifact artifact, + IProgress? progress, + CancellationToken cancellationToken); + void RemoveInstalledModel(string localDataDirectory, HuggingFaceModelInstallResult install); void RemovePartialModel( @@ -80,7 +101,7 @@ void RemovePartialModel( /// left by process termination is resumed with an HTTP range request. Shared /// cache artifacts and resumable partials survive rollback and cancellation. /// -internal sealed class HuggingFaceModelInstaller : IHuggingFaceModelAcquirer +internal sealed partial class HuggingFaceModelInstaller : IHuggingFaceModelAcquirer { private const int BufferSize = 1024 * 1024; private const int ProgressIntervalBytes = 4 * 1024 * 1024; diff --git a/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs b/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs index f991f4e76..e9fb5b693 100644 --- a/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs +++ b/src/OpenClaw.SetupEngine/LlamaRuntimeInstaller.cs @@ -130,7 +130,7 @@ public async Task InstallAsync( internal static LocalAiComponentIdentity Component(LlamaRuntimeVariant runtime) => new( "llama-server", - LlamaRuntimeCatalog.ReleaseTag, + runtime.ReleaseTag, runtime.Architecture switch { Architecture.X64 => "win-x64", diff --git a/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs b/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs index d9b16207e..6316da3d4 100644 --- a/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs +++ b/src/OpenClaw.SetupEngine/LocalAiGpuVerification.cs @@ -3,6 +3,7 @@ using System.Text.RegularExpressions; using OpenClaw.Connection.LocalAi; using OpenClaw.Shared.Inference; +using OpenClaw.Shared.Inference.Catalog; namespace OpenClaw.SetupEngine; @@ -297,7 +298,9 @@ ctx.LocalAiInferenceVerification is null || throw new InvalidDataException( "llama-server loaded CUDA from outside the managed runtime directory."); } - long minimumDelta = Math.Max(512L * 1024 * 1024, plan.Model.Weights.SizeBytes / 2); + long minimumDelta = Math.Max( + 512L * 1024 * 1024, + LocalModelCatalog.TotalDownloadSizeBytes(plan.Model) / 2); if (!HasRequiredGpuLoadEvidence(evidence, minimumDelta)) { throw new InvalidDataException( diff --git a/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs b/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs index ebcafdc94..afea2bcc5 100644 --- a/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs +++ b/src/OpenClaw.SetupEngine/LocalAiInstallReconciler.cs @@ -1,3 +1,4 @@ +using System.Collections.Immutable; using OpenClaw.Connection.LocalAi; using OpenClaw.Shared.Inference.Catalog; @@ -8,7 +9,8 @@ internal sealed record LocalAiReconcileResult( LocalAiResolvedInstall? ResolvedInstall, LlamaRuntimeInstallResult? RuntimeInstall, HuggingFaceModelInstallResult? ModelInstall, - LocalAiResolvedInstall? OriginalInstall = null) + LocalAiResolvedInstall? OriginalInstall = null, + ImmutableArray? AdditionalModelInstalls = null) { public static LocalAiReconcileResult NotInstalled { get; } = new(false, null, null, null); @@ -26,6 +28,18 @@ Task VerifyLegacyCompatibilityAsync( LocalAiPaths paths, PinnedArtifact artifact, CancellationToken cancellationToken); + + /// + /// Verifies one schema-5 additional model asset (a DFlash draft checkpoint) + /// still matches its pinned receipt in the + /// shared hub cache. Additional assets have no legacy app-owned copy, so + /// unlike there is no separate schema-3 path. + /// + Task VerifyAdditionalAssetAsync( + LocalAiResolvedInstall install, + string cachedAssetPath, + PinnedArtifact artifact, + CancellationToken cancellationToken); } internal sealed class LocalAiModelFileVerifier : ILocalAiModelFileVerifier @@ -35,7 +49,7 @@ public async Task VerifyActiveAsync( PinnedArtifact artifact, CancellationToken cancellationToken) { - if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (!install.Manifest.UsesHubCache) { if (!File.Exists(install.ModelPath)) return false; @@ -62,7 +76,7 @@ public Task VerifyLegacyCompatibilityAsync( PinnedArtifact artifact, CancellationToken cancellationToken) { - if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (!install.Manifest.UsesHubCache) return Task.FromResult(true); string legacyModelPath = paths.ResolveContainedPath( @@ -75,6 +89,24 @@ public Task VerifyLegacyCompatibilityAsync( artifact, cancellationToken); } + + public async Task VerifyAdditionalAssetAsync( + LocalAiResolvedInstall install, + string cachedAssetPath, + PinnedArtifact artifact, + CancellationToken cancellationToken) + { + await using FileStream? verified = + await HuggingFaceHubCache.TryOpenVerifiedCacheFileAsync( + install.Manifest.ModelCacheRoot!, + cachedAssetPath, + artifact.SizeBytes, + artifact.Sha256, + progress: null, + cancellationToken) + .ConfigureAwait(false); + return verified is not null; + } } /// @@ -126,7 +158,7 @@ public async Task ReconcileAsync( if (install is null) return LocalAiReconcileResult.NotInstalled; LocalAiResolvedInstall originalInstall = install; - ValidateRecipeMatch(install, plan, selectedGpuId, localDataDirectory); + bool runtimeUpgradePending = ValidateRecipeMatch(install, plan, selectedGpuId, localDataDirectory); bool migrateLegacyGpuId = !string.Equals(install.Manifest.SelectedGpuId, selectedGpuId, StringComparison.Ordinal) && @@ -145,7 +177,40 @@ public async Task ReconcileAsync( plan.Model.Weights, cancellationToken) .ConfigureAwait(false); - bool modelIsValid = activeModelIsValid && legacyModelIsValid; + bool additionalAssetsAreValid = activeModelIsValid && await VerifyAdditionalAssetsAsync( + install, + plan.Model, + cancellationToken) + .ConfigureAwait(false); + bool modelIsValid = activeModelIsValid && legacyModelIsValid && additionalAssetsAreValid; + if (runtimeUpgradePending) + { + if (modelIsValid) + { + install = await MigrateLegacyModelAsync( + install, + paths, + localDataDirectory, + plan, + selectedGpuId, + migrationProgress, + cancellationToken) + .ConfigureAwait(false); + } + + // The catalog moved to a newer pinned runtime. Drop only the runtime so the + // acquirer installs the new one, and keep the verified model and its extra + // assets so an upgrade does not re-download tens of GB. OriginalInstall lets + // the manifest step replace the existing receipt in place. + return new LocalAiReconcileResult( + Reused: false, + ResolvedInstall: null, + RuntimeInstall: null, + ModelInstall: modelIsValid ? CreateModelInstall(install, localDataDirectory) : null, + OriginalInstall: originalInstall, + AdditionalModelInstalls: modelIsValid ? CreateAdditionalModelInstalls(install) : null); + } + if (!inspection.IsValid || !modelIsValid) { if (!allowIncompleteInstallation) @@ -173,7 +238,8 @@ public async Task ReconcileAsync( ResolvedInstall: null, RuntimeInstall: inspection.IsValid ? CreateRuntimeInstall(install) : null, ModelInstall: modelIsValid ? CreateModelInstall(install, localDataDirectory) : null, - OriginalInstall: originalInstall); + OriginalInstall: originalInstall, + AdditionalModelInstalls: modelIsValid ? CreateAdditionalModelInstalls(install) : null); } install = await MigrateLegacyModelAsync( @@ -201,7 +267,8 @@ public async Task ReconcileAsync( install, CreateRuntimeInstall(install), CreateModelInstall(install, localDataDirectory), - OriginalInstall: allowIncompleteInstallation ? originalInstall : null); + OriginalInstall: allowIncompleteInstallation ? originalInstall : null, + AdditionalModelInstalls: CreateAdditionalModelInstalls(install)); } private static LlamaRuntimeInstallResult CreateRuntimeInstall(LocalAiResolvedInstall install) => @@ -222,7 +289,7 @@ private static HuggingFaceModelInstallResult CreateModelInstall( LocalAiResolvedInstall install, string localDataDirectory) { - if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (!install.Manifest.UsesHubCache) { return new HuggingFaceModelInstallResult( install.ModelPath, @@ -244,6 +311,60 @@ private static HuggingFaceModelInstallResult CreateModelInstall( LegacyCreatedThisRun: false); } + /// + /// Reconstructs the additional-asset install results a reused install's + /// manifest already proves verified, so a caller that skips re-acquiring + /// them (because just verified them) still has + /// a populated SetupContext.LocalAiAdditionalModelInstalls to persist. + /// + private static ImmutableArray CreateAdditionalModelInstalls( + LocalAiResolvedInstall install) + { + if (install.Manifest.AdditionalModelPaths.IsDefaultOrEmpty) + return ImmutableArray.Empty; + + var builder = ImmutableArray.CreateBuilder( + install.Manifest.AdditionalModelPathsOrEmpty.Length); + foreach (string cachedAssetPath in install.Manifest.AdditionalModelPathsOrEmpty) + { + builder.Add(new HuggingFaceAdditionalAssetInstallResult( + cachedAssetPath, + install.Manifest.ModelCacheRoot!, + HuggingFaceModelInstallDisposition.ReusedVerified, + CreatedThisRun: false)); + } + + return builder.MoveToImmutable(); + } + + private async Task VerifyAdditionalAssetsAsync( + LocalAiResolvedInstall install, + LocalModelInfo model, + CancellationToken cancellationToken) + { + ImmutableArray expected = LocalModelCatalog.AdditionalArtifacts(model); + if (expected.IsEmpty) + return install.Manifest.AdditionalModelAssetsOrEmpty.IsEmpty; + if (install.Manifest.AdditionalModelPathsOrEmpty.Length != expected.Length) + return false; + + for (int i = 0; i < expected.Length; i++) + { + if (!await _modelVerifier + .VerifyAdditionalAssetAsync( + install, + install.Manifest.AdditionalModelPathsOrEmpty[i], + expected[i], + cancellationToken) + .ConfigureAwait(false)) + { + return false; + } + } + + return true; + } + private async Task MigrateLegacyModelAsync( LocalAiResolvedInstall install, LocalAiPaths paths, @@ -271,26 +392,36 @@ private async Task MigrateLegacyModelAsync( .ConfigureAwait(false) ?? throw new InvalidDataException( "The Local AI installation manifest disappeared during cache migration."); - ValidateRecipeMatch(migrated, plan, selectedGpuId, localDataDirectory); + _ = ValidateRecipeMatch(migrated, plan, selectedGpuId, localDataDirectory); return migrated; } - private static void ValidateRecipeMatch( + /// + /// Validates the receipt against the runtime and recipe it actually recorded, and + /// reports whether the catalog has since moved to a newer pinned runtime. A version + /// difference is an expected upgrade, not a corrupt install, so it must not throw: + /// throwing here ends setup with an uninstall instruction instead of upgrading. + /// + private static bool ValidateRecipeMatch( LocalAiResolvedInstall install, LocalInferencePlan plan, string selectedGpuId, string localDataDirectory) { LocalAiInstallManifest manifest = install.Manifest; + LlamaRuntimeVariant installedRuntime = + LlamaRuntimeCatalog.FindInstalled(manifest.RuntimeId) ?? plan.Runtime; + bool runtimeUpgradePending = + !string.Equals(installedRuntime.Id, plan.Runtime.Id, StringComparison.Ordinal); string expectedArchitecture = plan.Runtime.Architecture switch { System.Runtime.InteropServices.Architecture.X64 => "x64", System.Runtime.InteropServices.Architecture.Arm64 => "arm64", _ => throw new InvalidDataException("The selected Local AI runtime architecture is unsupported."), }; - if (!string.Equals(manifest.EngineVersion, LlamaRuntimeCatalog.ReleaseTag, StringComparison.Ordinal) || + if (!string.Equals(manifest.EngineVersion, installedRuntime.ReleaseTag, StringComparison.Ordinal) || !string.Equals(manifest.Architecture, expectedArchitecture, StringComparison.Ordinal) || - !string.Equals(manifest.RuntimeId, plan.Runtime.Id, StringComparison.Ordinal) || + !string.Equals(manifest.RuntimeId, installedRuntime.Id, StringComparison.Ordinal) || !string.Equals(manifest.ModelCatalogId, plan.Model.Id, StringComparison.Ordinal) || manifest.ContextLength != plan.Profile.ContextTokens || manifest.KeyCachePrecision != plan.Profile.KeyCachePrecision || @@ -303,7 +434,7 @@ private static void ValidateRecipeMatch( "The existing managed Local AI installation does not match the selected runtime, GPU, and model recipe."); } - LocalAiComponentIdentity component = LlamaRuntimeInstaller.Component(plan.Runtime); + LocalAiComponentIdentity component = LlamaRuntimeInstaller.Component(installedRuntime); if (!LocalAiPathPolicy.TryResolve( localDataDirectory, component, @@ -327,11 +458,11 @@ private static void ValidateRecipeMatch( } LlamaServerRouterConfiguration.ValidateArtifactReceipts( manifest, - plan.Runtime, + installedRuntime, plan.Model); bool modelPathMatches; - if (manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion) + if (manifest.UsesHubCache) { modelPathMatches = !string.IsNullOrWhiteSpace(manifest.ModelCacheRoot) && @@ -372,6 +503,8 @@ private static void ValidateRecipeMatch( ? "The managed model path does not match the selected catalog recipe." : error); } + + return runtimeUpgradePending; } private static bool GpuIdsMatch(string persistedGpuId, string selectedGpuId) => diff --git a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs index fb0b437a8..085e14040 100644 --- a/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs +++ b/src/OpenClaw.SetupEngine/LocalAiSetupSteps.cs @@ -287,8 +287,15 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati } if (!result.Reused) { + // Normal upgrades restore their receipt independently of the gateway + // recovery pipeline's endpoint-health and provider rollback guards. + if (ctx.LocalAiRecoveryOriginalInstall is null && + result.OriginalInstall is { } retainedReceipt) + ctx.LocalAiUpgradeOriginalInstall ??= retainedReceipt; ctx.LocalAiRuntimeInstall = result.RuntimeInstall; ctx.LocalAiModelInstall = result.ModelInstall; + ctx.LocalAiAdditionalModelInstalls = result.AdditionalModelInstalls + ?? ImmutableArray.Empty; return StepResult.Skip(result.OriginalInstall is null ? "No completed managed Local AI installation was found." : "The existing Local AI receipt was retained while incomplete assets are repaired."); @@ -297,6 +304,8 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati ctx.LocalAiResolvedInstall = result.ResolvedInstall; ctx.LocalAiRuntimeInstall = result.RuntimeInstall; ctx.LocalAiModelInstall = result.ModelInstall; + ctx.LocalAiAdditionalModelInstalls = result.AdditionalModelInstalls + ?? ImmutableArray.Empty; ctx.LocalAiPort = result.ResolvedInstall!.Manifest.RequestedPort; return StepResult.Ok("Reused the verified managed Local AI installation."); } @@ -312,6 +321,11 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati ex); } } + + public override Task RollbackAsync(SetupContext ctx, CancellationToken ct) => + ctx.IsUninstalling + ? Task.CompletedTask + : PersistLocalAiManifestStep.RestoreUpgradeReceiptAsync(ctx, ct); } /// Installs the two pinned llama.cpp runtime archives as one atomic component. @@ -382,7 +396,7 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati progress, linked.Token); ctx.LocalAiRuntimeInstall = install; - return StepResult.Ok($"Installed llama-server {LlamaRuntimeCatalog.ReleaseTag}."); + return StepResult.Ok($"Installed llama-server {plan.Runtime.ReleaseTag}."); } catch (OperationCanceledException) when (ct.IsCancellationRequested) { @@ -437,6 +451,16 @@ public AcquireLocalAiModelStep() internal AcquireLocalAiModelStep(IHuggingFaceModelAcquirer acquirer) => _acquirer = acquirer ?? throw new ArgumentNullException(nameof(acquirer)); + /// + /// Additional artifacts a recipe needs beyond its primary weights, in the + /// fixed catalog order + /// defines. relies on that same + /// ordering to tell the draft checkpoint apart from a shard without a + /// separate "kind" tag on the receipt. + /// + internal static ImmutableArray AdditionalArtifacts(LocalModelInfo model) => + LocalModelCatalog.AdditionalArtifacts(model); + public override string Id => "acquire-local-ai-model"; public override string DisplayName => "Downloading Local AI model from Hugging Face"; public override bool CanRetry => false; @@ -476,6 +500,29 @@ public override async Task ExecuteAsync(SetupContext ctx, Cancellati progress, linked.Token); ctx.LocalAiModelInstall = install; + + ImmutableArray additionalArtifacts = AdditionalArtifacts(plan.Model); + var additionalInstalls = ImmutableArray.CreateBuilder( + additionalArtifacts.Length); + foreach (PinnedArtifact artifact in additionalArtifacts) + { + var artifactProgress = new SynchronousProgress(value => + ctx.DetailProgress?.Report(new SetupDetailProgressEvent( + Id, + value.Phase == HuggingFaceModelInstallPhase.Verifying + ? $"Verifying {artifact.RelativePath}" + : $"Downloading {artifact.RelativePath}", + value.CompletedBytes, + value.TotalBytes, + SetupDetailProgressUnit.Bytes))); + additionalInstalls.Add(await _acquirer.InstallAdditionalAssetAsync( + ctx.LocalDataDir, + artifact, + artifactProgress, + linked.Token)); + } + ctx.LocalAiAdditionalModelInstalls = additionalInstalls.MoveToImmutable(); + string action = install.Disposition == HuggingFaceModelInstallDisposition.ReusedVerified ? "Verified existing" : "Downloaded"; @@ -508,6 +555,9 @@ public override Task RollbackAsync(SetupContext ctx, CancellationToken ct) _acquirer.RemoveInstalledModel(ctx.LocalDataDir, install); ctx.LocalAiModelInstall = null; } + // Additional assets have no legacy app-owned copy to remove; their hub-cache + // artifacts survive rollback the same way the primary weights' do. + ctx.LocalAiAdditionalModelInstalls = ImmutableArray.Empty; if (ctx.LocalAiEligibility?.Plan is { } plan) { _acquirer.RemovePartialModel( @@ -554,10 +604,12 @@ ctx.LocalAiModelInstall is not return StepResult.Terminal(portError ?? "The requested Local AI port is invalid."); var paths = new LocalAiPaths(ctx.LocalDataDir); - bool replacesRecoveryReceipt = - ctx.LocalAiRecoveryOriginalInstall is not null && + LocalAiResolvedInstall? originalInstall = + ctx.LocalAiRecoveryOriginalInstall ?? ctx.LocalAiUpgradeOriginalInstall; + bool replacesExistingReceipt = + originalInstall is not null && File.Exists(paths.ManifestPath); - if (File.Exists(paths.ManifestPath) && !replacesRecoveryReceipt) + if (File.Exists(paths.ManifestPath) && !replacesExistingReceipt) return StepResult.Terminal("A managed Local AI installation receipt already exists."); LocalAiComponentIdentity component = LlamaRuntimeInstaller.Component(plan.Runtime); if (!LocalAiPathPolicy.TryResolve( @@ -598,10 +650,34 @@ ctx.LocalAiRecoveryOriginalInstall is not null && return StepResult.Terminal(ex.Message, ex); } + ImmutableArray additionalArtifacts = AcquireLocalAiModelStep.AdditionalArtifacts(plan.Model); + if (additionalArtifacts.Length != ctx.LocalAiAdditionalModelInstalls.Length) + { + return StepResult.Terminal( + "The Local AI installation receipt requires a completed additional-asset acquisition step."); + } + + var additionalModelAssets = ImmutableArray.CreateBuilder(additionalArtifacts.Length); + var additionalModelPaths = ImmutableArray.CreateBuilder(additionalArtifacts.Length); + for (int i = 0; i < additionalArtifacts.Length; i++) + { + PinnedArtifact artifact = additionalArtifacts[i]; + additionalModelAssets.Add(new LocalAiAssetReceipt + { + FileName = Path.GetFileName(artifact.RelativePath), + SourceUrl = artifact.DownloadUri.AbsoluteUri, + SizeBytes = artifact.SizeBytes, + Sha256 = artifact.Sha256.Value, + }); + additionalModelPaths.Add(ctx.LocalAiAdditionalModelInstalls[i].ModelPath); + } + LocalAiInstallManifest manifest = new() { - SchemaVersion = LocalAiInstallManifest.HubCacheReceiptSchemaVersion, - EngineVersion = LlamaRuntimeCatalog.ReleaseTag, + SchemaVersion = additionalArtifacts.IsEmpty + ? LocalAiInstallManifest.HubCacheReceiptSchemaVersion + : LocalAiInstallManifest.AdditionalAssetsSchemaVersion, + EngineVersion = plan.Runtime.ReleaseTag, Architecture = plan.Runtime.Architecture switch { Architecture.X64 => "x64", @@ -625,6 +701,11 @@ ctx.LocalAiRecoveryOriginalInstall is not null && SizeBytes = plan.Model.Weights.SizeBytes, Sha256 = plan.Model.Weights.Sha256.Value, }, + // Leave these at their unset default (not an explicitly-built empty + // array) when there is nothing to add, so schema-4 manifests omit + // them from JSON entirely -- see the properties' remarks. + AdditionalModelAssets = additionalArtifacts.IsEmpty ? default : additionalModelAssets.MoveToImmutable(), + AdditionalModelPaths = additionalArtifacts.IsEmpty ? default : additionalModelPaths.MoveToImmutable(), RequestedPort = requestedPort, Endpoint = null, ContextLength = plan.Profile.ContextTokens, @@ -633,7 +714,7 @@ ctx.LocalAiRecoveryOriginalInstall is not null && DraftKeyCachePrecision = plan.Profile.DraftKeyCachePrecision, DraftValueCachePrecision = plan.Profile.DraftValueCachePrecision, }; - if (ctx.LocalAiRecoveryOriginalInstall is { } originalInstall) + if (originalInstall is not null) { manifest = originalInstall.Manifest with { @@ -651,6 +732,8 @@ ctx.LocalAiRecoveryOriginalInstall is not null && ModelId = manifest.ModelId, ModelAlias = manifest.ModelAlias, ModelAsset = manifest.ModelAsset, + AdditionalModelAssets = manifest.AdditionalModelAssets, + AdditionalModelPaths = manifest.AdditionalModelPaths, RequestedPort = manifest.RequestedPort, Endpoint = null, ContextLength = manifest.ContextLength, @@ -666,7 +749,7 @@ ctx.LocalAiRecoveryOriginalInstall is not null && { await store.SaveAsync(manifest, ct); ctx.LocalAiResolvedInstall = store.ResolveAndValidate(manifest); - ctx.LocalAiManifestCreatedThisRun = !replacesRecoveryReceipt; + ctx.LocalAiManifestCreatedThisRun = !replacesExistingReceipt; return StepResult.Ok("Recorded the verified llama-server and Hugging Face installation."); } catch (Exception ex) when (ex is IOException or UnauthorizedAccessException or InvalidDataException) @@ -695,10 +778,17 @@ public override async Task RollbackAsync(SetupContext ctx, CancellationToken ct) ctx.LocalAiRuntimeInstall = null; ctx.LocalAiModelInstall = null; ctx.LocalAiResolvedInstall = null; + ctx.LocalAiUpgradeOriginalInstall = null; ctx.LocalAiManifestCreatedThisRun = false; return; } + if (ctx.LocalAiUpgradeOriginalInstall is not null) + { + await RestoreUpgradeReceiptAsync(ctx, ct); + return; + } + if (!ctx.LocalAiManifestCreatedThisRun) return; @@ -710,6 +800,20 @@ public override async Task RollbackAsync(SetupContext ctx, CancellationToken ct) ctx.LocalAiManifestCreatedThisRun = false; } + internal static async Task RestoreUpgradeReceiptAsync(SetupContext ctx, CancellationToken ct) + { + if (ctx.LocalAiUpgradeOriginalInstall is not { } originalInstall) + return; + + var paths = new LocalAiPaths(ctx.LocalDataDir); + var store = new LocalAiManifestStore(paths); + await store.SaveAsync(originalInstall.Manifest, ct); + File.Delete(paths.RouterPresetPath); + ctx.LocalAiResolvedInstall = store.ResolveAndValidate(originalInstall.Manifest); + ctx.LocalAiManifestCreatedThisRun = false; + ctx.LocalAiUpgradeOriginalInstall = null; + } + private static ImmutableArray BuildRuntimeReceipts( LlamaRuntimeVariant runtime, LlamaRuntimeInstallResult install) diff --git a/src/OpenClaw.SetupEngine/SetupContext.cs b/src/OpenClaw.SetupEngine/SetupContext.cs index aad8f82dd..4390a6f14 100644 --- a/src/OpenClaw.SetupEngine/SetupContext.cs +++ b/src/OpenClaw.SetupEngine/SetupContext.cs @@ -1,3 +1,4 @@ +using System.Collections.Immutable; using System.Diagnostics.CodeAnalysis; using System.Text.Json; using System.Text.Json.Serialization; @@ -515,8 +516,16 @@ public Func>? public int? LocalAiPort { get; set; } internal LlamaRuntimeInstallResult? LocalAiRuntimeInstall { get; set; } internal HuggingFaceModelInstallResult? LocalAiModelInstall { get; set; } + /// + /// Verified additional model assets (a DFlash draft checkpoint and/or + /// in catalog order. Empty for every recipe + /// that has neither. + /// + internal ImmutableArray LocalAiAdditionalModelInstalls { get; set; } = + ImmutableArray.Empty; internal LocalAiResolvedInstall? LocalAiResolvedInstall { get; set; } internal LocalAiResolvedInstall? LocalAiRecoveryOriginalInstall { get; set; } + internal LocalAiResolvedInstall? LocalAiUpgradeOriginalInstall { get; set; } internal bool LocalAiRecoveryProviderTransition { get; set; } internal bool LocalAiRecoveryReceiptRollbackAllowed { get; set; } internal bool LocalAiManifestCreatedThisRun { get; set; } diff --git a/src/OpenClaw.SetupEngine/SetupReviewSummary.cs b/src/OpenClaw.SetupEngine/SetupReviewSummary.cs index 0b9f1f696..10b31de8b 100644 --- a/src/OpenClaw.SetupEngine/SetupReviewSummary.cs +++ b/src/OpenClaw.SetupEngine/SetupReviewSummary.cs @@ -77,16 +77,33 @@ public static SetupReviewSummary Build(SetupConfig config, string? dataDir = nul LocalModelCatalog.Find(config.LocalAi.SelectedModelId) ?? LocalModelCatalog.Default; LocalInferenceRunProfile? localAiProfile = LocalModelCatalog.FindProfile(localAiModel, config.LocalAi.SelectedProfileId); - string[] localAiCommands = config.LocalAi.Enabled - ? - [ + string[] localAiCommands; + if (config.LocalAi.Enabled) + { + var commands = new List + { "download verified llama-server + CUDA runtime for Windows", $"download {localAiModel.Weights.RelativePath} from Hugging Face revision " + ((HuggingFaceRevisionSource)localAiModel.Weights.Source).RevisionSha, - $"llama-server router on dynamic 127.0.0.1 port; model loads on first request", - $"openclaw provider llamacpp -> /v1; primary llamacpp/{localAiModel.Id}", - ] - : []; + }; + // Additional pinned artifacts (a DFlash draft checkpoint) are + // separate downloads the user is consenting + // to alongside the primary weights -- list each one explicitly + // rather than letting the consent screen understate what's fetched. + foreach (PinnedArtifact artifact in LocalModelCatalog.AdditionalArtifacts(localAiModel)) + { + commands.Add( + $"download {artifact.RelativePath} from Hugging Face revision " + + ((HuggingFaceRevisionSource)artifact.Source).RevisionSha); + } + commands.Add("llama-server router on dynamic 127.0.0.1 port; model loads on first request"); + commands.Add($"openclaw provider llamacpp -> /v1; primary llamacpp/{localAiModel.Id}"); + localAiCommands = commands.ToArray(); + } + else + { + localAiCommands = []; + } var summary = new SetupReviewSummary( DistroTitle: $"Install {baseDistro.Replace('-', ' ')} in WSL", diff --git a/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs index a0b6ee26a..9242951f0 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LlamaRuntimeCatalog.cs @@ -10,7 +10,8 @@ public LlamaRuntimeVariant( string id, Architecture architecture, Version cudaVersion, - IReadOnlyList artifacts) + IReadOnlyList artifacts, + string? releaseTag = null) { ArgumentException.ThrowIfNullOrWhiteSpace(id); ArgumentNullException.ThrowIfNull(cudaVersion); @@ -32,12 +33,20 @@ public LlamaRuntimeVariant( Architecture = architecture; CudaVersion = cudaVersion; Artifacts = artifacts; + ReleaseTag = releaseTag ?? LlamaRuntimeCatalog.ReleaseTag; } public string Id { get; } public Architecture Architecture { get; } public Version CudaVersion { get; } public IReadOnlyList Artifacts { get; } + /// + /// The llama.cpp release this variant was pinned from. Equals + /// for the current runtime and + /// keeps its own older value for a retired one, so an installed receipt is + /// validated against the release it actually recorded. + /// + public string ReleaseTag { get; } public long TotalDownloadSizeBytes => Artifacts.Sum(artifact => artifact.SizeBytes); } @@ -47,11 +56,11 @@ public LlamaRuntimeVariant( /// public static class LlamaRuntimeCatalog { - public const string ReleaseTag = "b10655"; - public const string ReleaseCommitSha = "cb300598d5f90189cb69d2702f4930aaf99d32a2"; + public const string ReleaseTag = "b11026"; + public const string ReleaseCommitSha = "b49650adb31f2e49a0d76113aeb1792134fd8413"; public const string ServerExecutableName = "llama-server.exe"; - public const string X64RuntimeId = "b10655-cuda13-x64"; - public const string Arm64RuntimeId = "b10655-cuda13-arm64"; + public const string X64RuntimeId = "b11026-cuda13-x64"; + public const string Arm64RuntimeId = "b11026-cuda13-arm64"; public static GitHubReleaseSource Source { get; } = new( "ggml-org/llama.cpp", @@ -64,43 +73,103 @@ public static class LlamaRuntimeCatalog new LlamaRuntimeVariant( X64RuntimeId, Architecture.X64, - new Version(13, 3), + new Version(13, 4), + Array.AsReadOnly( + new[] + { + RuntimeArtifact( + "llama-b11026-cuda13-x64", + ArtifactRole.RuntimeBinary, + "llama-b11026-bin-win-cuda-13.4-x64.zip", + 150_102_391, + "6799f0962d066c54aee3773f0e5efa0076e46418695c0f4f6d24a38e7007dfb1"), + RuntimeArtifact( + "cudart-b11026-cuda13-x64", + ArtifactRole.RuntimeDependency, + "cudart-llama-bin-win-cuda-13.4-x64.zip", + 423_535_356, + "738f8c251ac22b70c3ae6f83a10cf222725df0395246a2cf58f32bdb85fbe668"), + })), + new LlamaRuntimeVariant( + Arm64RuntimeId, + Architecture.Arm64, + new Version(13, 4), Array.AsReadOnly( new[] { RuntimeArtifact( + "llama-b11026-cuda13-arm64", + ArtifactRole.RuntimeBinary, + "llama-b11026-bin-win-cuda-13.4-arm64.zip", + 142_993_717, + "d4a31d05b4fe997872020d81e8482e7712c7c254ec9c7ccdb9597ae2a31e6728"), + RuntimeArtifact( + "cudart-b11026-cuda13-arm64", + ArtifactRole.RuntimeDependency, + "cudart-llama-bin-win-cuda-13.4-arm64.zip", + 153_262_407, + "642dcde8805b3e3165ca710a5443b3b4044b27d96bd3ee3132473988c9bcb774"), + })), + }); + + // Retired from new installs and never offered or selected. These exist only so a + // managed installation recorded before the runtime bump keeps resolving its own + // receipt and stays launchable until setup upgrades it. Pins are reproduced + // exactly as they were installed; nothing is remapped. + private const string LegacyB10655ReleaseTag = "b10655"; + private const string LegacyB10655RuntimeIdX64 = "b10655-cuda13-x64"; + private const string LegacyB10655RuntimeIdArm64 = "b10655-cuda13-arm64"; + + private static GitHubReleaseSource LegacyB10655Source { get; } = new( + "ggml-org/llama.cpp", + LegacyB10655ReleaseTag, + "cb300598d5f90189cb69d2702f4930aaf99d32a2"); + + private static readonly ReadOnlyCollection s_legacyVariants = Array.AsReadOnly( + new[] + { + new LlamaRuntimeVariant( + LegacyB10655RuntimeIdX64, + Architecture.X64, + new Version(13, 3), + Array.AsReadOnly( + new[] + { + LegacyB10655Artifact( "llama-b10655-cuda13-x64", ArtifactRole.RuntimeBinary, "llama-b10655-bin-win-cuda-13.3-x64.zip", 146_478_045, "be61636141327b3ca4d437c17489fd69964838a31a5fe3e97400f0dcd9f669dc"), - RuntimeArtifact( + LegacyB10655Artifact( "cudart-b10655-cuda13-x64", ArtifactRole.RuntimeDependency, "cudart-llama-bin-win-cuda-13.3-x64.zip", 390_970_417, "1462a050eb4c684921ba51dcc4cc488a036674c3e73e9945ee705b854808d03e"), - })), + }), + LegacyB10655ReleaseTag), new LlamaRuntimeVariant( - Arm64RuntimeId, + LegacyB10655RuntimeIdArm64, Architecture.Arm64, new Version(13, 4), Array.AsReadOnly( new[] { - RuntimeArtifact( + LegacyB10655Artifact( "llama-b10655-cuda13-arm64", ArtifactRole.RuntimeBinary, "llama-b10655-bin-win-cuda-13.4-arm64.zip", 140_055_278, "567e61b4129e0d5b0580e5d3ea86b82ab5b6bee745ee02f69b58af799b49a582"), - RuntimeArtifact( + LegacyB10655Artifact( "cudart-b10655-cuda13-arm64", ArtifactRole.RuntimeDependency, "cudart-llama-bin-win-cuda-13.4-arm64.zip", 153_318_797, "5a40dc7c5fa3d0a80ceeba4f16f9e8d25d87bcf1399c9233588953c43436c33c"), - })), + }), + LegacyB10655ReleaseTag), }); public static IReadOnlyList Variants => s_variants; @@ -108,6 +177,19 @@ public static class LlamaRuntimeCatalog public static LlamaRuntimeVariant? Find(Architecture architecture) => s_variants.SingleOrDefault(variant => variant.Architecture == architecture); + /// + /// Resolves a runtime id from an already-installed receipt, including a retired + /// runtime from before the last version bump. Use this only on installed-receipt + /// validation and launch paths. Selection and acquisition must keep using + /// and so a retired runtime is never + /// installed again. + /// + public static LlamaRuntimeVariant? FindInstalled(string? id) => + string.IsNullOrWhiteSpace(id) + ? null + : s_variants.SingleOrDefault(variant => string.Equals(variant.Id, id, StringComparison.Ordinal)) + ?? s_legacyVariants.SingleOrDefault(variant => string.Equals(variant.Id, id, StringComparison.Ordinal)); + private static PinnedArtifact RuntimeArtifact( string id, ArtifactRole role, @@ -122,4 +204,19 @@ private static PinnedArtifact RuntimeArtifact( sizeBytes, new Sha256Digest(sha256), LocalInferenceCatalogProvenance.NvidiaCair); + + private static PinnedArtifact LegacyB10655Artifact( + string id, + ArtifactRole role, + string fileName, + long sizeBytes, + string sha256) => + new( + id, + role, + LegacyB10655Source, + fileName, + sizeBytes, + new Sha256Digest(sha256), + LocalInferenceCatalogProvenance.NvidiaCair); } diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs index 58581b1f5..cda634a2e 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceEligibility.cs @@ -48,6 +48,36 @@ public static long GetRequiredMemoryBytes( LocalInferenceRunProfile profile) => LocalInferenceQualificationPolicy.GetRequiredMemoryBytes(model, profile); + /// + /// Device eligibility for deciding whether Local AI stays available, given the model + /// already configured on this machine. + /// + /// + /// A SKU with no recommended default is a statement about what to install by default, + /// not about what the device can run. Gating availability purely on the default pick + /// would switch Local AI off on a setup rerun for a machine that already has a working + /// configured model, so once a model is configured this reports on that model: a + /// selection that still passes the full capacity fit-test is retained, and one that does + /// not carries its own failure (unknown model, or the model name with its required and + /// detected memory) rather than the SKU's generic no-recommendation reason. With no model + /// configured, the device result stands and fresh setup is unchanged. + /// + public static LocalInferenceEligibilityResult EvaluateForConfiguredAvailability( + HostHardwareInfo hardware, + string? configuredModelId) + { + ArgumentNullException.ThrowIfNull(hardware); + LocalInferenceEligibilityResult device = Evaluate(hardware); + if (device.CanInstall || + device.SelectionFailureCode != LocalInferenceSelectionFailureCode.NotRecommendedForSku || + string.IsNullOrWhiteSpace(configuredModelId)) + { + return device; + } + + return Evaluate(hardware, configuredModelId); + } + public static LocalInferenceEligibilityResult Evaluate( HostHardwareInfo hardware, string? requestedModelId = null) @@ -64,7 +94,14 @@ public static LocalInferenceEligibilityResult Evaluate( LocalInferencePlan plan = selection.Plan; long requiredMemoryBytes = GetRequiredMemoryBytes(plan.Model, plan.Profile); - CandidateAssessment? selected = hardware.NvidiaGpus + // A plan bound to one adapter (an RTX Spark SKU recipe) must be assessed only + // against that adapter. Ranking every NVIDIA GPU here would let the recipe + // chosen for the Spark be reported against, and then launched on, a different + // GPU on a mixed host. + IEnumerable candidateGpus = plan.BoundGpuStableId is { Length: > 0 } boundId + ? hardware.NvidiaGpus.Where(gpu => string.Equals(gpu.StableId, boundId, StringComparison.Ordinal)) + : hardware.NvidiaGpus; + CandidateAssessment? selected = candidateGpus .Select(gpu => Assess(gpu, plan.Runtime, requiredMemoryBytes)) .OrderBy(candidate => StatusRank(candidate.Status)) .ThenBy(candidate => DefinitivenessRank(candidate.FailureCode)) diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs index bbc8f99c7..61cbe6e85 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalInferenceSelector.cs @@ -16,6 +16,8 @@ public enum LocalInferenceSelectionFailureCode RuntimeUnavailable = 1, NoNvidiaGpu = 2, UnknownModel = 3, + /// RTX Spark detected, but this memory SKU has no recommended local model. + NotRecommendedForSku = 4, } /// Whether a caller accepted the catalog default or named a model explicitly. @@ -26,11 +28,19 @@ public enum LocalInferenceModelSelectionOrigin } /// A complete, immutable native inference choice. +/// +/// The adapter this recipe was chosen for, when the choice is only valid on that +/// adapter. RTX Spark recipes come from a fixed per-device SKU table, so the recipe +/// and the GPU that runs it must be the same adapter; eligibility restricts its +/// candidates to this id. Null means any qualifying NVIDIA GPU may run the plan, +/// which is the generic discrete-GPU behavior. +/// public sealed record LocalInferencePlan( LlamaRuntimeVariant Runtime, LocalModelInfo Model, LocalInferenceRunProfile Profile, - LocalInferenceModelSelectionOrigin ModelSelectionOrigin); + LocalInferenceModelSelectionOrigin ModelSelectionOrigin, + string? BoundGpuStableId = null); /// The deterministic result of selecting from the pinned local inference catalog. public sealed record LocalInferenceSelectionResult @@ -60,7 +70,14 @@ internal static LocalInferenceSelectionResult Unsupported(LocalInferenceSelectio /// /// Pure selection from a hardware snapshot and optional model ID. The CPU /// architecture chooses only the native runtime. GPU names and CPU/GPU SKU -/// pairings are not part of qualification. +/// pairings are not part of qualification for discrete GPUs -- the one +/// deliberate exception is RTX Spark's default pick, routed through +/// because its unified-memory SKU +/// cannot be identified by capacity fit-testing alone (see NVIDIA's fixed +/// SKU-to-recipe table). An explicitly requested model ID uses the same +/// capacity fit-test on every GPU, except when the request is the Spark SKU's +/// own recommendation round-tripped through setup, which keeps that SKU's +/// pinned profile and adapter binding. /// public static class LocalInferenceSelector { @@ -81,9 +98,42 @@ public static LocalInferenceSelectionResult Select( LocalModelInfo? model; LocalInferenceRunProfile profile; LocalInferenceModelSelectionOrigin modelSelectionOrigin; + string? boundGpuStableId = null; + GpuInfo? sparkGpu = hardware.NvidiaGpus.FirstOrDefault( + gpu => gpu.IsRtxSpark && LocalInferenceQualificationPolicy.HasCompleteFacts(gpu)); + var sparkPick = sparkGpu is null + ? null + : RtxSparkInferenceSelector.SelectDefault(sparkGpu); if (string.IsNullOrWhiteSpace(requestedModelId)) { - (model, profile) = SelectDefaultModelAndProfile(hardware, runtime); + if (sparkPick is not null) + { + // Bind the plan to this adapter: the SKU table answers "what should + // THIS Spark run", so the recipe is only valid on the Spark that + // produced it, never on some other GPU eligibility might rank higher. + (model, profile) = sparkPick.Value; + boundGpuStableId = sparkGpu!.StableId; + } + else + { + // Either no Spark, or a Spark SKU with no recommended model. In the + // latter case the Spark is excluded rather than failing the whole + // host, so a discrete GPU alongside it can still qualify normally. + HostHardwareInfo genericHardware = sparkGpu is null + ? hardware + : hardware with + { + Gpus = hardware.Gpus.Where(gpu => !gpu.IsRtxSpark).ToArray(), + }; + if (!genericHardware.HasNvidiaGpu) + { + return LocalInferenceSelectionResult.Unsupported( + LocalInferenceSelectionFailureCode.NotRecommendedForSku); + } + + (model, profile) = SelectDefaultModelAndProfile(genericHardware, runtime); + } + modelSelectionOrigin = LocalInferenceModelSelectionOrigin.Default; } else @@ -91,29 +141,58 @@ public static LocalInferenceSelectionResult Select( model = LocalModelCatalog.Find(requestedModelId); if (model is null) return LocalInferenceSelectionResult.Unsupported(LocalInferenceSelectionFailureCode.UnknownModel); - profile = SelectBestFittingProfile(hardware, runtime, model) ?? - LocalModelCatalog.GetProfiles(model)[^1]; + if (sparkPick is { } recommended && + string.Equals(recommended.Model.Id, model.Id, StringComparison.OrdinalIgnoreCase)) + { + // Setup persists the recommended model id and passes it back here, so + // the SKU's own recommendation arrives as an explicit request. Re-deriving + // its profile through the generic fit-test would silently discard the + // pinned profile the SKU table specifies (the 64 GB tier's reduced context + // is not the largest that merely fits) and drop the adapter binding. + // A request for any other model is a real user override and still uses + // the generic fit-test below. + profile = recommended.Profile; + boundGpuStableId = sparkGpu!.StableId; + } + else + { + profile = SelectBestFittingProfile(hardware, runtime, model) ?? + LocalModelCatalog.GetProfiles(model)[^1]; + } + modelSelectionOrigin = LocalInferenceModelSelectionOrigin.Explicit; } return LocalInferenceSelectionResult.Selected( - new LocalInferencePlan(runtime, model, profile, modelSelectionOrigin)); + new LocalInferencePlan(runtime, model, profile, modelSelectionOrigin, boundGpuStableId)); } private static (LocalModelInfo Model, LocalInferenceRunProfile Profile) SelectDefaultModelAndProfile( HostHardwareInfo hardware, LlamaRuntimeVariant runtime) { - foreach (LocalModelInfo candidate in LocalModelCatalog.Models + // A priority-0, explicit-alternative model (currently only the + // experimental 96GB Flash-Next recipe) is offered solely by + // RtxSparkInferenceSelector for its one intended SKU, never picked as + // a generic dGPU default or fallback -- excluded here so a large + // enough non-Spark GPU (or a Spark GPU IsRtxSpark fails to detect) + // can't land on it by tie-break/fallback ordering. Priority-0 models + // that aren't explicit alternatives (the other Spark-only recipes) + // keep their existing, unrelated reachability. + IEnumerable genericDefaultCandidates = LocalModelCatalog.Models + .Where(model => model.RecommendationPriority > 0 || !model.IsExplicitAlternative); + foreach (LocalModelInfo candidate in genericDefaultCandidates .OrderByDescending(model => model.RecommendationPriority) - .ThenByDescending(model => model.Weights.SizeBytes)) + .ThenByDescending(LocalModelCatalog.TotalDownloadSizeBytes)) { LocalInferenceRunProfile? profile = SelectBestFittingProfile(hardware, runtime, candidate); if (profile is not null) return (candidate, profile); } - LocalModelInfo fallback = LocalModelCatalog.Models.OrderBy(model => model.Weights.SizeBytes).First(); + LocalModelInfo fallback = genericDefaultCandidates + .OrderBy(LocalModelCatalog.TotalDownloadSizeBytes) + .First(); return (fallback, LocalModelCatalog.GetProfiles(fallback)[^1]); } @@ -149,9 +228,12 @@ public static long GetRequiredMemoryBytes( { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(profile); + long weightsBytes = SaturatingAdd( + model.Weights.SizeBytes, + model.Recipe.DraftWeights?.SizeBytes ?? 0); return SaturatingAdd( SaturatingAdd( - SaturatingAdd(model.Weights.SizeBytes, GetKvCacheMemoryBytes(model.Recipe, profile)), + SaturatingAdd(weightsBytes, GetKvCacheMemoryBytes(model.Recipe, profile)), GetDraftKvCacheMemoryBytes(model.Recipe, profile)), profile.RuntimeWorkspaceBytes); } @@ -182,8 +264,12 @@ internal static long GetDraftKvCacheMemoryBytes( ArgumentNullException.ThrowIfNull(recipe); ArgumentNullException.ThrowIfNull(profile); - // The pinned Qwen MTP artifacts contain one draft attention layer with - // the same KV head count and head dimension as the target model. + if (recipe.SpeculativeDecoding == SpeculativeDecodingMode.None) + return 0; + + // The pinned Qwen MTP and DFlash draft artifacts contain one draft + // attention layer with the same KV head count and head dimension as + // the target model. long bytesPerToken = SaturatingAdd( SaturatingMultiply( recipe.KeyValueHeadCount, diff --git a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs index 72a2a2391..e15b59ba8 100644 --- a/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs +++ b/src/OpenClaw.Shared/Inference/Catalog/LocalModelCatalog.cs @@ -1,3 +1,4 @@ +using System.Collections.Immutable; using System.Collections.ObjectModel; namespace OpenClaw.Shared.Inference.Catalog; @@ -13,6 +14,10 @@ public enum KvCachePrecision public enum SpeculativeDecodingMode { DraftMtp = 0, + /// No speculative decoding; the target model runs standalone. + None = 1, + /// Draft-flash decoding using a separate, independently pinned draft checkpoint. + DraftDFlash = 2, } /// Sampling values recommended for the model's thinking mode. @@ -38,8 +43,13 @@ public LocalModelRunRecipe( bool offloadAllLayers, SpeculativeDecodingMode speculativeDecoding, int speculativeDraftMaxTokens, - ModelSamplingPreset sampling) + ModelSamplingPreset sampling, + PinnedArtifact? draftWeights = null) { + if (speculativeDecoding == SpeculativeDecodingMode.DraftDFlash && draftWeights is null) + throw new ArgumentException("Draft-flash decoding requires a pinned draft checkpoint.", nameof(draftWeights)); + if (speculativeDecoding != SpeculativeDecodingMode.DraftDFlash && draftWeights is not null) + throw new ArgumentException("Only draft-flash decoding uses a separate draft checkpoint.", nameof(draftWeights)); if (batchTokens <= 0) throw new ArgumentOutOfRangeException(nameof(batchTokens)); if (microBatchTokens <= 0 || microBatchTokens > batchTokens) @@ -67,6 +77,7 @@ public LocalModelRunRecipe( SpeculativeDecoding = speculativeDecoding; SpeculativeDraftMaxTokens = speculativeDraftMaxTokens; Sampling = sampling; + DraftWeights = draftWeights; } public int BatchTokens { get; } @@ -80,6 +91,8 @@ public LocalModelRunRecipe( public SpeculativeDecodingMode SpeculativeDecoding { get; } public int SpeculativeDraftMaxTokens { get; } public ModelSamplingPreset Sampling { get; } + /// The independently pinned draft checkpoint, set only for . + public PinnedArtifact? DraftWeights { get; } } /// A downloadable GGUF model and its deterministic llama-server recipe. @@ -142,10 +155,16 @@ public static class LocalModelCatalog /// Qwen3.5 9B receipt keeps resolving and launching across upgrade. /// public const string Qwen9BModelId = "qwen3.5-9b-mtp-q4-k-m"; + /// RTX Spark 48GB-SKU recipe. Never offered on the generic dGPU path; see RtxSparkInferenceSelector. + public const string Qwen35B_IQ4XSModelId = "qwen3.6-35b-a3b-mtp-ud-iq4-xs"; + /// RTX Spark 128GB-SKU default recipe. Never offered on the generic dGPU path; see RtxSparkInferenceSelector. + public const string Qwen38_27B_DFlashModelId = "qwen3.8-27b-dflash-ud-q4-k-m"; public const int NativeContextTokens = 262_144; public const int IntermediateContextTokens = 196_608; public const int ReducedContextTokens = 131_072; public const int MinimumContextTokens = 65_536; + /// RTX Spark 48GB-SKU context tier (98,304 tokens); see . + public const int RtxSpark48GbContextTokens = 98_304; // Measured-conservative allowances for compute buffers, recurrent state, // CUDA graphs, allocator alignment, and miscellaneous backend allocations. @@ -172,6 +191,10 @@ public static class LocalModelCatalog "unsloth/Qwen3.5-9B-MTP-GGUF", "9716a636ee4bddc3fed678220b7a33dd2a4160ae"); + private static readonly HuggingFaceRevisionSource s_qwen38_27BDFlashDraftSource = new( + "z-lab/Qwen3.8-27B-DFlash2-GGUF", + "2d9571f8ce46e151f61c6499c99dee6079e1d610"); + private static readonly ReadOnlyCollection s_models = Array.AsReadOnly( new[] { @@ -232,6 +255,63 @@ public static class LocalModelCatalog IsExplicitAlternative: true, SupportsVision: false, RecommendationPriority: 200), + // RTX Spark SKU recipes below, offered by RtxSparkInferenceSelector + // keyed off the detected Spark unified-memory SKU. RecommendationPriority: 0 + // alone does not exclude a model from the generic dGPU default pick -- + // it only loses every tie-break against a positive-priority model that + // fits. LocalInferenceSelector.SelectDefaultModelAndProfile separately + // excludes IsExplicitAlternative models at this priority (see its + // comment) so the always-alternative-only ones can't still win by + // tie-break/fallback ordering among themselves. + new LocalModelInfo( + Qwen35B_IQ4XSModelId, + "Qwen3.6 35B-A3B (UD-IQ4_XS)", + "Qwen3.6", + "UD-IQ4_XS", + ModelArtifact( + Qwen35B_IQ4XSModelId, + s_qwen35BSource, + "Qwen3.6-35B-A3B-UD-IQ4_XS.gguf", + 18_209_036_576, + "df27a780435b7b45c2597536112ea3cb091f8544c3d0c3318d9f4258b31f7adf"), + Recipe( + fullAttentionLayerCount: 10, + keyValueHeadCount: 2, + temperature: 0.6, + speculativeDraftMaxTokens: 2), + IsDefault: false, + IsExplicitAlternative: false, + SupportsVision: false, + RecommendationPriority: 0), + new LocalModelInfo( + Qwen38_27B_DFlashModelId, + "Qwen3.8 27B (UD-Q4_K_M, DFlash)", + "Qwen3.8", + "UD-Q4_K_M", + ModelArtifact( + Qwen38_27B_DFlashModelId, + s_qwen38_27BSource, + "Qwen3.8-27B-UD-Q4_K_M.gguf", + 16_464_440_224, + "322e194ff79741c7baa497c240f677f54b201b0efab44ca8e50f122b39123482"), + Recipe( + fullAttentionLayerCount: 16, + keyValueHeadCount: 4, + temperature: 1.0, + batchTokens: 4_096, + microBatchTokens: 512, + speculativeDecoding: SpeculativeDecodingMode.DraftDFlash, + speculativeDraftMaxTokens: 7, + draftWeights: ModelArtifact( + "qwen3.8-27b-dflash2-q4-k-m", + s_qwen38_27BDFlashDraftSource, + "Qwen3.8-27B-DFlash2-Q4_K_M.gguf", + 1_143_006_816, + "1a25c56858e1ebe93f2718ac1d49d1151f9323325c1bbfd6209370f4db131ebd")), + IsDefault: false, + IsExplicitAlternative: false, + SupportsVision: false, + RecommendationPriority: 0), }); // Retired from new installs and never offered, recommended, or selectable. @@ -264,7 +344,10 @@ public static class LocalModelCatalog private static readonly IReadOnlyDictionary> s_profilesByModel = s_models - .Select(model => (model, profiles: Array.AsReadOnly(CreateProfiles(model)))) + .Select(model => (model, profiles: Array.AsReadOnly( + string.Equals(model.Id, Qwen35B_IQ4XSModelId, StringComparison.Ordinal) + ? CreateRtxSpark48GbProfiles(model) + : CreateProfiles(model)))) .Concat(s_legacyModels .Select(model => (model, profiles: Array.AsReadOnly(CreateLegacyProfiles(model))))) .ToDictionary( @@ -346,6 +429,32 @@ public static IReadOnlyList GetProfiles(LocalModelInfo /// True when the id resolves only to a retired catalog entry. public static bool IsLegacy(string? id) => Find(id) is null && FindInstalled(id) is not null; + /// + /// Everything setup downloads and llama-server loads for this recipe: the pinned + /// weights plus the DFlash draft checkpoint, if any. Use this for ranking, capacity + /// checks, and user-facing download-size disclosure -- the draft checkpoint is a + /// separate pinned artifact, not part of the target model's own weights, but it is + /// still bytes the user consents to, setup fetches, and the runtime loads. + /// + public static long TotalDownloadSizeBytes(LocalModelInfo model) + { + ArgumentNullException.ThrowIfNull(model); + return model.Weights.SizeBytes + (model.Recipe.DraftWeights?.SizeBytes ?? 0); + } + + /// + /// The recipe's additional pinned artifacts beyond its primary weights, in the + /// fixed order every acquirer, manifest, and launch path must agree on. Today that + /// is the DFlash draft checkpoint, when the recipe pins one. + /// + public static ImmutableArray AdditionalArtifacts(LocalModelInfo model) + { + ArgumentNullException.ThrowIfNull(model); + return model.Recipe.DraftWeights is { } draftWeights + ? [draftWeights] + : ImmutableArray.Empty; + } + private static LocalInferenceRunProfile[] CreateProfiles(LocalModelInfo model) => [ Profile(model, NativeContextTokens, KvCachePrecision.F16), @@ -367,6 +476,14 @@ private static LocalInferenceRunProfile[] CreateLegacyProfiles(LocalModelInfo mo Profile(model, NativeContextTokens, KvCachePrecision.F16), ]; + // The RTX Spark 48GB-SKU recipe launches at a single fixed context/KV + // tier (98,304 tokens, F16 KV) rather than the shared cross-product of + // tiers other models expose. + private static LocalInferenceRunProfile[] CreateRtxSpark48GbProfiles(LocalModelInfo model) => + [ + Profile(model, RtxSpark48GbContextTokens, KvCachePrecision.F16), + ]; + private static LocalInferenceRunProfile Profile( LocalModelInfo model, int contextTokens, @@ -377,6 +494,7 @@ private static LocalInferenceRunProfile Profile( NativeContextTokens => RuntimeWorkspaceReserveBytes, IntermediateContextTokens => IntermediateContextWorkspaceReserveBytes, ReducedContextTokens => ReducedContextWorkspaceReserveBytes, + RtxSpark48GbContextTokens => ReducedContextWorkspaceReserveBytes, MinimumContextTokens => MinimumContextWorkspaceReserveBytes, _ => throw new ArgumentOutOfRangeException(nameof(contextTokens)), }; @@ -408,23 +526,29 @@ private static PinnedArtifact ModelArtifact( private static LocalModelRunRecipe Recipe( int fullAttentionLayerCount, int keyValueHeadCount, - double temperature) => + double temperature, + int batchTokens = 4_096, + int microBatchTokens = 4_096, + SpeculativeDecodingMode speculativeDecoding = SpeculativeDecodingMode.DraftMtp, + int speculativeDraftMaxTokens = 3, + PinnedArtifact? draftWeights = null) => new( - batchTokens: 4_096, - microBatchTokens: 4_096, + batchTokens: batchTokens, + microBatchTokens: microBatchTokens, parallelRequests: 1, fullAttentionLayerCount: fullAttentionLayerCount, keyValueHeadCount: keyValueHeadCount, keyValueHeadDimension: 256, flashAttention: true, offloadAllLayers: true, - speculativeDecoding: SpeculativeDecodingMode.DraftMtp, - speculativeDraftMaxTokens: 3, + speculativeDecoding: speculativeDecoding, + speculativeDraftMaxTokens: speculativeDraftMaxTokens, sampling: new ModelSamplingPreset( Temperature: temperature, TopK: 20, TopP: 0.95, MinP: 0.0, RepetitionPenalty: 1.0, - PresencePenalty: 0.0)); + PresencePenalty: 0.0), + draftWeights: draftWeights); } diff --git a/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs b/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs new file mode 100644 index 000000000..38040b24b --- /dev/null +++ b/src/OpenClaw.Shared/Inference/Catalog/RtxSparkInferenceSelector.cs @@ -0,0 +1,62 @@ +namespace OpenClaw.Shared.Inference.Catalog; + +/// +/// Default-recipe routing for RTX Spark, a unified-memory SKU where "total +/// CUDA-visible memory" identifies the physical memory SKU rather than a +/// discrete GPU's VRAM. RTX Spark bypasses the generic priority/fit-test +/// default pick () in favor of a fixed +/// SKU-to-recipe table NVIDIA specified for this hardware. Runtime/driver +/// eligibility is still assessed uniformly afterward by +/// for both paths. +/// +internal static class RtxSparkInferenceSelector +{ + // Empirically confirmed on real RTX Spark hardware: a 48GB-SKU unit's + // cuMemGetInfo total reads ~48.59e9 bytes (~45.25 GiB), tracking the + // nominal decimal-GB SKU size closely. This is NOT what nvidia-smi's "FB + // Memory Usage" (NVML) or Windows' "Total Physical Memory" report on the + // same box -- both read far lower on Spark's unified-memory design and + // must never be used for SKU classification; only GpuInfo.GpuVisibleMemoryBytes + // (CudaHostHardwareProbe's cuMemGetInfo reading) is reliable here. + // Boundaries are geometric midpoints between nominal decimal-GB SKU + // sizes; only the 48GB boundary is hardware-verified today. + private const long DecimalGigabyte = 1_000_000_000L; + + private static readonly long s_boundary32_48 = GeometricMidpointBytes(32, 48); + private static readonly long s_boundary48_64 = GeometricMidpointBytes(48, 64); + private static readonly long s_boundary64_128 = GeometricMidpointBytes(64, 128); + + /// + /// Returns the default (model, profile) pick for a detected RTX Spark + /// GPU, or null when this SKU has no recommended local model. + /// + internal static (LocalModelInfo Model, LocalInferenceRunProfile Profile)? SelectDefault(GpuInfo sparkGpu) + { + ArgumentNullException.ThrowIfNull(sparkGpu); + long totalBytes = LocalInferenceQualificationPolicy.GetEffectiveTotalMemoryBytes(sparkGpu); + + return totalBytes switch + { + _ when totalBytes < s_boundary32_48 => null, // 32GB SKU: no local AI recommended + _ when totalBytes < s_boundary48_64 => Recipe(LocalModelCatalog.Qwen35B_IQ4XSModelId), + _ when totalBytes < s_boundary64_128 => Recipe(LocalModelCatalog.Qwen38_27BModelId, ReducedQ8_0ProfileId), + _ => Recipe(LocalModelCatalog.Qwen38_27B_DFlashModelId), + }; + } + + private const string ReducedQ8_0ProfileId = "ctx-131072-q8_0"; + + private static (LocalModelInfo, LocalInferenceRunProfile) Recipe(string modelId, string? profileId = null) + { + LocalModelInfo model = LocalModelCatalog.Find(modelId) + ?? throw new InvalidOperationException($"RTX Spark recipe '{modelId}' is missing from the catalog."); + LocalInferenceRunProfile profile = profileId is null + ? LocalModelCatalog.GetProfiles(model)[0] + : LocalModelCatalog.FindProfile(model, profileId) + ?? throw new InvalidOperationException($"RTX Spark recipe '{modelId}' is missing profile '{profileId}'."); + return (model, profile); + } + + private static long GeometricMidpointBytes(int lowerNominalGb, int upperNominalGb) => + (long)(Math.Sqrt((double)lowerNominalGb * upperNominalGb) * DecimalGigabyte); +} diff --git a/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs b/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs index b1beaa788..60c6b0813 100644 --- a/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs +++ b/src/OpenClaw.Shared/Inference/HostHardwareInfo.cs @@ -52,7 +52,17 @@ public sealed record GpuInfo( long? FreeSharedGpuMemoryBytes = null, string? DriverVersion = null, int? CudaMajorVersion = null, - string? StableId = null); + string? StableId = null) +{ + /// + /// True for an RTX Spark unified-memory adapter (e.g. "NVIDIA RTX Spark + /// N1X"), identified by its driver-reported name. RTX Spark is routed + /// through a fixed SKU recipe table instead of the generic capacity + /// fit-test; see RtxSparkInferenceSelector. + /// + public bool IsRtxSpark => + Name.Contains("RTX Spark", StringComparison.OrdinalIgnoreCase); +} /// /// Snapshot of the host's inference-relevant hardware. Every probed field is diff --git a/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs b/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs index c0a8a3fcd..eff8285d0 100644 --- a/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs +++ b/src/OpenClaw.Tray.WinUI/Presentation/LocalAiPageViewModel.cs @@ -283,6 +283,9 @@ private async Task RefreshAvailabilityAsync(CancellationTokenSource cancellation // dispatched callback runs) would make a queued-but-not-yet-run callback's own // IsCurrentAvailabilityProbe guard fail against itself, silently dropping a real // asynchronous DispatcherQueue completion. + // Captured before the probe so the managed-install receipt is read on the caller's + // thread rather than on whatever thread resumes after the awaited probe. + string? installedModelId = _runtimeSnapshot.ModelId; try { HostHardwareInfo hardware = await Task.Run( @@ -292,7 +295,16 @@ private async Task RefreshAvailabilityAsync(CancellationTokenSource cancellation // not the currently selected/installed model. A selection-specific failure (unknown, // deprecated, or oversized model) must not report the device itself as unavailable // and block retry-setup from switching to a compatible catalog model. - LocalInferenceEligibilityResult eligibility = LocalInferenceEligibility.Evaluate(hardware); + // + // The one exception is a SKU that has no recommended default at all (RTX Spark + // 32 GB). That says nothing about whether this device can run what is already + // installed, so the managed-install receipt is taken into account: an installed + // model that still qualifies keeps this entry point available, which is what lets + // Retry Setup reach the manifest-aware recovery path and Change Model stay enabled. + // A machine with no managed receipt still reports unavailable, and a receipt whose + // model is unknown or no longer fits reports that model's own reason. + LocalInferenceEligibilityResult eligibility = + LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, installedModelId); if (eligibility.FailureCode == LocalInferenceEligibilityFailureCode.HardwareFactsIncomplete) { // Incomplete facts (a CUDA read that came back partial or transient) are @@ -520,8 +532,16 @@ private void ApplyOnUiThread(Action action) private void ApplyRuntimeSnapshot(LocalAiRuntimeSnapshot snapshot) { + string? previousModelId = _runtimeSnapshot.ModelId; _runtimeSnapshot = snapshot; OnPropertyChanged(null); + // Availability reads the managed-install receipt (see RefreshAvailabilityAsync), which the + // runtime refresh may only publish after that read has already happened. Recomputing when + // the model id changes is what keeps a first visit correct: otherwise a 32 GB Spark whose + // receipt arrives late stays pinned at NotRecommendedForSku, with Retry Setup and Change + // Model disabled and Recheck unavailable, until the page is left and reopened. + if (IsActive && !string.Equals(previousModelId, snapshot.ModelId, StringComparison.Ordinal)) + StartAvailabilityRefresh(); } private void ApplyGatewaySnapshot(GatewayConnectionSnapshot snapshot) { diff --git a/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs b/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs index 29381bf68..92596120d 100644 --- a/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs +++ b/tests/OpenClaw.Connection.Tests/LocalAiManifestMigrationTests.cs @@ -28,6 +28,68 @@ public async Task Load_DoesNotTriggerCacheMigration() Assert.Equal(fixture.Content, await File.ReadAllBytesAsync(fixture.LegacyModelPath)); } + [Fact] + public async Task Save_SchemaFourManifestOmitsAdditionalAssetFieldsFromJson() + { + // A recipe with no additional assets (every recipe before this session, + // and most since) must keep writing the exact schema-4 shape an older + // app build already knows how to read. AdditionalModelAssets/Paths + // default to ImmutableArray's unset (not .Empty) value specifically + // so JsonIgnoreCondition.WhenWritingDefault omits them here, and + // UsesHubCache must never appear at all -- it's a derived read helper, + // not part of the persisted contract. + using var temp = new TempDirectory("local-ai-manifest-schema4-json-"); + var paths = new LocalAiPaths(temp.Combine("app-data")); + string legacyRelativePath = Path.Combine("models", "owner", "repository", Revision, "model.gguf"); + var manifest = new LocalAiInstallManifest + { + SchemaVersion = LocalAiInstallManifest.HubCacheReceiptSchemaVersion, + EngineVersion = "b1", + Architecture = "x64", + RuntimeId = "llama-server-test", + ModelCatalogId = "test-model", + SelectedGpuId = "GPU-TEST", + ExecutablePath = Path.Combine("engines", "llama-server.exe"), + RuntimeAssets = ImmutableArray.Create(new LocalAiAssetReceipt + { + FileName = "runtime.zip", + SourceUrl = "https://example.invalid/runtime.zip", + SizeBytes = 1, + Sha256 = new string('a', 64), + }), + ModelPath = legacyRelativePath, + ModelCacheRoot = temp.Combine("hf-cache"), + CachedModelPath = HuggingFaceHubCache.TryGetSnapshotPaths( + temp.Combine("hf-cache"), + RepositoryId, + Revision, + RelativeModelPath, + out string cachedModelPath, + out _, + out string pathError) + ? cachedModelPath + : throw new InvalidOperationException(pathError), + ModelId = $"{RepositoryId}@{Revision}", + ModelAlias = "test-model", + ModelAsset = new LocalAiAssetReceipt + { + FileName = "model.gguf", + SourceUrl = $"https://huggingface.co/{RepositoryId}/resolve/{Revision}/{RelativeModelPath}?download=true", + SizeBytes = 1, + Sha256 = new string('b', 64), + }, + ContextLength = 4096, + }; + var store = new LocalAiManifestStore(paths, () => temp.Combine("hf-cache")); + await store.SaveAsync(manifest); + + JsonObject persisted = (JsonNode.Parse(await File.ReadAllTextAsync(paths.ManifestPath)) as JsonObject)!; + + Assert.False(persisted.ContainsKey("additionalModelAssets")); + Assert.False(persisted.ContainsKey("additionalModelPaths")); + Assert.False(persisted.ContainsKey("usesHubCache")); + } + [Fact] public async Task Load_CopiesVerifiedLegacyWeightsAndRecordsTransitionalReceipt() { diff --git a/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs b/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs index 936247e00..97f86b893 100644 --- a/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs +++ b/tests/OpenClaw.Connection.Tests/LocalAiPortLifecycleTests.cs @@ -2546,6 +2546,127 @@ private static LocalAiInstallManifest LegacyQwen9BManifest() }; } + /// + /// A managed install recorded before the llama-server runtime bump must keep + /// launching against its own pinned receipt. Updating the app must not strand an + /// installed model until a separate setup repair runs. + /// + [Fact] + public async Task Router_LaunchesRetiredRuntimeInstallAfterVersionBump() + { + using var temp = new TempDirectory("local-ai-legacy-runtime-"); + var paths = new LocalAiPaths(temp.Path); + LlamaRuntimeVariant retired = LlamaRuntimeCatalog.FindInstalled("b10655-cuda13-arm64")!; + LocalAiInstallManifest manifest = ValidManifest() with + { + EngineVersion = retired.ReleaseTag, + RuntimeId = retired.Id, + ExecutablePath = Path.Combine( + "engines", + $"llama-{retired.ReleaseTag}", + LlamaRuntimeCatalog.ServerExecutableName), + RuntimeAssets = retired.Artifacts.Select(artifact => new LocalAiAssetReceipt + { + FileName = Path.GetFileName(artifact.RelativePath), + SourceUrl = artifact.DownloadUri.AbsoluteUri, + SizeBytes = artifact.SizeBytes, + Sha256 = artifact.Sha256.Value, + }).ToImmutableArray(), + }; + var store = new LocalAiManifestStore(paths); + await store.SaveAsync(manifest); + + LocalAiResolvedInstall saved = (await store.LoadAsync())!; + LlamaServerRouterLaunchPlan launch = LlamaServerRouterConfiguration.Build(paths, saved); + + Assert.NotEqual(LlamaRuntimeCatalog.ReleaseTag, retired.ReleaseTag); + Assert.Equal("qwen3.6-35b-a3b-mtp-q4-k-m", launch.ModelAlias); + } + + /// + /// Schema-5 extra assets are loaded natively by llama-server from the shared, + /// user-writable hub cache. One that no longer matches its pinned digest must + /// stop startup, exactly like a tampered primary model does. + /// + [Fact] + public async Task Startup_FailsWhenAnAdditionalModelAssetNoLongerMatchesItsReceipt() + { + using var temp = new TempDirectory("local-ai-tampered-asset-"); + LocalAiPaths paths = await PrepareInstallAsync(temp); + var store = new LocalAiManifestStore(paths); + LocalAiResolvedInstall installed = (await store.LoadAsync())!; + string cacheRoot = temp.Combine("hf-cache"); + const string draftRepo = "z-lab/Qwen3.8-27B-DFlash2-GGUF"; + string draftRevision = new('c', 40); + Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, draftRepo, draftRevision, "draft.gguf", + out string draftPath, out _, out string error), error); + // The primary model's cached path must be the real hub-cache snapshot path + // for its own repository and revision, or the receipt fails validation + // before the additional-asset check under test is ever reached. + Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, + "unsloth/Qwen3.6-35B-A3B-MTP-GGUF", + "5bc3e238d916f48a861bac2f8a1990a0e9b7e98d", + "Qwen3.6-35B-A3B-UD-Q4_K_M.gguf", + out string cachedPrimaryPath, out _, out error), error); + Directory.CreateDirectory(Path.GetDirectoryName(cachedPrimaryPath)!); + await File.WriteAllTextAsync(cachedPrimaryPath, "primary"); + Directory.CreateDirectory(Path.GetDirectoryName(draftPath)!); + await File.WriteAllTextAsync(draftPath, "draft"); + LocalAiInstallManifest schemaFive = installed.Manifest with + { + SchemaVersion = LocalAiInstallManifest.AdditionalAssetsSchemaVersion, + ModelCacheRoot = cacheRoot, + CachedModelPath = cachedPrimaryPath, + AdditionalModelAssets = ImmutableArray.Create(new LocalAiAssetReceipt + { + FileName = "draft.gguf", + SourceUrl = $"https://huggingface.co/{draftRepo}/resolve/{draftRevision}/draft.gguf?download=true", + SizeBytes = 1_143_006_816, + Sha256 = new string('d', 64), + }), + AdditionalModelPaths = ImmutableArray.Create(draftPath), + }; + + var events = new SynchronizedEventLog(); + var platform = new FakePlatform(); + await using var runtime = CreateRuntime( + paths, + new FakeProcessHost(platform, events, selectedPort: 28_771), + platform, + new FakeClient(events), + new FakeLifecycle(events), + modelFileVerifier: new SelectiveModelFileVerifier( + cachedPrimaryPath, + rejectPath: draftPath)); + await store.SaveAsync(schemaFive); + + LocalAiRuntimeSnapshot snapshot = await runtime.EnsureStartedAsync(); + + Assert.Equal(LocalAiRuntimeState.Failed, snapshot.State); + Assert.Contains("draft.gguf", snapshot.Detail ?? string.Empty, StringComparison.Ordinal); + } + + /// Verifies the primary model but rejects one named additional asset. + private sealed class SelectiveModelFileVerifier(string resolvedPath, string rejectPath) + : ILocalAiModelFileVerifier + { + public Task TryOpenAsync( + string cacheRoot, + string candidatePath, + long expectedSizeBytes, + Sha256Digest expectedSha256, + CancellationToken cancellationToken) + { + cancellationToken.ThrowIfCancellationRequested(); + if (string.Equals(candidatePath, rejectPath, StringComparison.OrdinalIgnoreCase)) + return Task.FromResult(null); + return Task.FromResult( + new LocalAiVerifiedModelLease(new MemoryStream(), resolvedPath)); + } + } + private static LocalAiInstallManifest ValidManifest() { LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.Find( diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs index fc353ea91..38381a515 100644 --- a/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs +++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiInstallRecoveryTests.cs @@ -967,6 +967,11 @@ public void ArchiveDestination_ResolvesValidNestedEntry() public async Task Reconciler_ReusesOnlyMatchingManifestWithoutMutation() { using var temp = new TempDirectory(); + // Pin the hub cache to an empty directory. This asserts that a matching receipt is + // reused untouched; with the ambient user cache it would instead depend on whether + // that cache happens to already hold the default model, which legitimately triggers + // the schema-3 to schema-4 migration and rewrites the receipt. + using var environment = new EnvironmentScope("HF_HUB_CACHE", CacheRoot(temp.Path)); LocalInferencePlan plan = CatalogPlan(); const string gpuId = "GPU-0"; LocalAiInstallManifest manifest = CreateManifest(temp.Path, plan, gpuId); @@ -1145,6 +1150,271 @@ public async Task Reconciler_RecoveryRepairsMissingSchemaFourCompatibilityCopy() Assert.Null(result.ModelInstall); } + [Fact] + public async Task Reconciler_RecoveryWithValidModelStillPopulatesAdditionalAssetInstalls() + { + // Regression: recovery for a broken runtime (model + additional assets + // still verified valid) must give the caller everything it needs to + // persist a schema-5 manifest without re-downloading the already- + // verified draft checkpoint -- AcquireLocalAiModelStep's "reuse the + // verified model" skip only re-populates SetupContext from the + // reconcile result, it never re-runs acquisition itself. + using var temp = new TempDirectory(); + byte[] primaryBytes = "verified-dflash-primary"u8.ToArray(); + byte[] draftBytes = "verified-dflash-draft"u8.ToArray(); + var draftSource = new HuggingFaceRevisionSource("owner/draft-repo", new string('c', 40)); + var draftWeights = new PinnedArtifact( + "test-model-dflash-draft", + ArtifactRole.ModelWeights, + draftSource, + "draft.gguf", + draftBytes.Length, + new Sha256Digest(Sha256(draftBytes))); + LocalModelInfo model = CreateModelWithDraft(primaryBytes, draftWeights); + LlamaRuntimeVariant runtime = CreateRuntime( + CreateZip(("llama-server.exe", "server"u8.ToArray())), + CreateZip(("dependency.dll", "dependency"u8.ToArray()))); + var plan = new LocalInferencePlan( + runtime, + model, + new LocalInferenceRunProfile( + "test-profile", + 128, + KvCachePrecision.F16, + KvCachePrecision.F16, + KvCachePrecision.F16, + KvCachePrecision.F16, + runtimeWorkspaceBytes: 1), + LocalInferenceModelSelectionOrigin.Default); + var paths = new LocalAiPaths(temp.Path); + string cacheRoot = CacheRoot(temp.Path); + var primarySource = Assert.IsType(model.Weights.Source); + Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, + primarySource.RepositoryId, + primarySource.RevisionSha, + model.Weights.RelativePath, + out string cachedModelPath, + out _, + out string error), error); + Directory.CreateDirectory(Path.GetDirectoryName(cachedModelPath)!); + await File.WriteAllBytesAsync(cachedModelPath, primaryBytes); + Assert.True(HuggingFaceHubCache.TryGetSnapshotPaths( + cacheRoot, + draftSource.RepositoryId, + draftSource.RevisionSha, + draftWeights.RelativePath, + out string cachedDraftPath, + out _, + out error), error); + Directory.CreateDirectory(Path.GetDirectoryName(cachedDraftPath)!); + await File.WriteAllBytesAsync(cachedDraftPath, draftBytes); + + LocalAiInstallManifest manifest = CreateManifest(temp.Path, plan, "GPU-0") with + { + SchemaVersion = LocalAiInstallManifest.AdditionalAssetsSchemaVersion, + ModelCacheRoot = cacheRoot, + CachedModelPath = cachedModelPath, + AdditionalModelAssets = ImmutableArray.Create(new LocalAiAssetReceipt + { + FileName = "draft.gguf", + SourceUrl = draftWeights.DownloadUri.AbsoluteUri, + SizeBytes = draftWeights.SizeBytes, + Sha256 = draftWeights.Sha256.Value, + }), + AdditionalModelPaths = ImmutableArray.Create(cachedDraftPath), + }; + await new LocalAiManifestStore(paths, () => cacheRoot).SaveAsync(manifest); + + LocalAiReconcileResult result = await new LocalAiInstallReconciler( + new InvalidRuntimeInspector(), + new AcceptingModelVerifier(), + () => cacheRoot) + .ReconcileAsync( + temp.Path, + plan, + "GPU-0", + CancellationToken.None, + allowIncompleteInstallation: true); + + Assert.False(result.Reused); + Assert.Null(result.RuntimeInstall); + Assert.NotNull(result.ModelInstall); + ImmutableArray additionalInstalls = + result.AdditionalModelInstalls ?? ImmutableArray.Empty; + HuggingFaceAdditionalAssetInstallResult draftInstall = Assert.Single(additionalInstalls); + Assert.Equal(cachedDraftPath, draftInstall.ModelPath); + Assert.False(draftInstall.CreatedThisRun); + } + + [Fact] + public async Task Reconciler_UpgradesRetiredRuntimeReceiptInsteadOfFailingSetup() + { + // An install recorded before the runtime bump must upgrade, not end setup with an + // uninstall instruction. The runtime is dropped so the acquirer installs the new + // pin; the verified model is kept so an upgrade does not re-download it. + using var temp = new TempDirectory(); + using var environment = new EnvironmentScope("HF_HUB_CACHE", CacheRoot(temp.Path)); + LocalInferencePlan plan = CatalogPlan(); + const string gpuId = "GPU-0"; + var paths = new LocalAiPaths(temp.Path); + LlamaRuntimeVariant retired = LlamaRuntimeCatalog.FindInstalled("b10655-cuda13-x64")!; + LocalAiInstallManifest manifest = CreateRetiredManifest(temp.Path, plan, gpuId); + await new LocalAiManifestStore(paths).SaveAsync(manifest); + var reconciler = new LocalAiInstallReconciler( + new ValidRuntimeInspector(), + new AcceptingModelVerifier()); + + LocalAiReconcileResult result = await reconciler.ReconcileAsync( + temp.Path, + plan, + gpuId, + CancellationToken.None); + + Assert.False(result.Reused); + Assert.Null(result.RuntimeInstall); + Assert.NotNull(result.ModelInstall); + Assert.NotNull(result.OriginalInstall); + Assert.Equal(retired.ReleaseTag, result.OriginalInstall!.Manifest.EngineVersion); + } + + [Theory] + [InlineData(false, null)] + [InlineData(true, null)] + [InlineData(false, "after-reconcile")] + [InlineData(true, "after-reconcile")] + [InlineData(false, "after-persist")] + [InlineData(true, "after-persist")] + public async Task RuntimeUpgrade_MigratesModelAndRestoresOriginalReceiptOnFailure( + bool usesHubCache, + string? failureStage) + { + using var temp = new TempDirectory(); + string cacheRoot = CacheRoot(temp.Path); + byte[] modelBytes = "verified-upgrade-model"u8.ToArray(); + byte[] runtimeZip = CreateZip(("llama-server.exe", "new-server"u8.ToArray())); + byte[] dependencyZip = CreateZip(("dependency.dll", "dependency"u8.ToArray())); + LlamaRuntimeVariant runtime = CreateRuntime(runtimeZip, dependencyZip); + LocalInferencePlan plan = CatalogPlan() with + { + Runtime = runtime, + Model = CreateModel(modelBytes), + }; + var paths = new LocalAiPaths(temp.Path); + var store = new LocalAiManifestStore(paths, () => cacheRoot); + LocalAiInstallManifest manifest = CreateRetiredManifest(temp.Path, plan, "GPU-0") with + { + GatewayFallbackModel = "openai/gpt-5", + }; + string oldExecutable = paths.ResolveContainedPath(manifest.ExecutablePath, "executable"); + Directory.CreateDirectory(Path.GetDirectoryName(oldExecutable)!); + await File.WriteAllTextAsync(oldExecutable, "old-server"); + string legacyModel = paths.ResolveContainedPath(manifest.ModelPath, "model"); + Directory.CreateDirectory(Path.GetDirectoryName(legacyModel)!); + await File.WriteAllBytesAsync(legacyModel, modelBytes); + await store.SaveAsync(manifest); + if (usesHubCache) + manifest = (await store.MigrateLegacyModelToHubCacheAsync())!.Manifest; + byte[] originalReceipt = await File.ReadAllBytesAsync(paths.ManifestPath); + + var context = CreateContext(temp.Path, confirmDestructive: false); + context.Config.LocalAi.Enabled = true; + context.Config.RollbackOnFailure = true; + context.LocalAiPort = manifest.RequestedPort; + context.LocalAiEligibility = new LocalInferenceEligibilityResult( + LocalInferenceEligibilityStatus.Eligible, + LocalInferenceEligibilityFailureCode.None, + LocalInferenceSelectionFailureCode.None, + plan, + new GpuInfo(GpuVendor.Nvidia, "Test GPU", StableId: "GPU-0"), + RequiredTotalMemoryBytes: 0, + DetectedTotalMemoryBytes: 0, + RequiredFreeMemoryBytes: 0, + AvailableFreeMemoryBytes: 0); + int runtimeDownloads = 0; + using var runtimeClient = new HttpClient(new DelegateHandler(request => + { + runtimeDownloads++; + return new HttpResponseMessage(HttpStatusCode.OK) + { + Content = new ByteArrayContent( + request.RequestUri!.AbsolutePath.EndsWith("runtime.zip", StringComparison.Ordinal) + ? runtimeZip + : dependencyZip), + }; + })); + using var modelClient = new HttpClient(new DelegateHandler(_ => + throw new InvalidOperationException("Upgrade must not download the verified primary model."))); + var reconciler = new LocalAiInstallReconciler( + new ValidRuntimeInspector(), new LocalAiModelFileVerifier(), () => cacheRoot); + string? cachedModel = null; + string? newExecutable = null; + var pipeline = new SetupPipeline( + [ + new ReconcileLocalAiInstallationStep(reconciler), + new UpgradeCheckpointStep("after-reconcile", ctx => + { + Assert.Null(ctx.LocalAiRecoveryOriginalInstall); + Assert.Equal(manifest.SchemaVersion, ctx.LocalAiUpgradeOriginalInstall?.Manifest.SchemaVersion); + Assert.Equal(oldExecutable, ctx.LocalAiUpgradeOriginalInstall?.ExecutablePath); + Assert.Equal(cacheRoot, ctx.LocalAiModelInstall?.CacheRoot); + cachedModel = ctx.LocalAiModelInstall!.ModelPath; + return failureStage == "after-reconcile"; + }), + new AcquireLocalAiRuntimeStep(new LlamaRuntimeInstaller( + new LocalAiArtifactInstaller(runtimeClient), new ValidRuntimeInspector())), + new AcquireLocalAiModelStep(CreateModelInstaller(modelClient, temp.Path)), + new PersistLocalAiManifestStep(), + new UpgradeCheckpointStep("after-persist", ctx => + { + LocalAiResolvedInstall upgraded = Assert.IsType(ctx.LocalAiResolvedInstall); + Assert.Equal("b11026", upgraded.Manifest.EngineVersion); + Assert.Equal(LocalAiInstallManifest.HubCacheReceiptSchemaVersion, upgraded.Manifest.SchemaVersion); + Assert.Equal(cacheRoot, upgraded.Manifest.ModelCacheRoot); + Assert.Equal(cachedModel, upgraded.ModelPath); + Assert.Equal(manifest.InstalledAtUtc, upgraded.Manifest.InstalledAtUtc); + Assert.Equal(manifest.GatewayFallbackModel, upgraded.Manifest.GatewayFallbackModel); + Assert.Null(upgraded.Endpoint); + newExecutable = upgraded.ExecutablePath; + Assert.True(File.Exists(newExecutable)); + return failureStage == "after-persist"; + }), + ]); + + PipelineResult result = await pipeline.RunAsync(context); + + Assert.True( + result.Outcome == (failureStage is null ? PipelineOutcome.Success : PipelineOutcome.Failed), + result.Message); + Assert.Equal(failureStage, result.FailedStepId); + if (failureStage is not null) + Assert.Equal("Injected upgrade failure.", result.Message); + Assert.Equal(failureStage == "after-reconcile" ? 0 : 2, runtimeDownloads); + Assert.Equal(modelBytes, await File.ReadAllBytesAsync(cachedModel!)); + Assert.Equal(modelBytes, await File.ReadAllBytesAsync(legacyModel)); + Assert.Equal("old-server", await File.ReadAllTextAsync(oldExecutable)); + LocalAiResolvedInstall persisted = (await store.LoadAsync())!; + if (failureStage is null) + { + Assert.Equal(newExecutable, persisted.ExecutablePath); + Assert.Equal("b11026", persisted.Manifest.EngineVersion); + } + else + { + Assert.Equal(originalReceipt, await File.ReadAllBytesAsync(paths.ManifestPath)); + Assert.Equal(oldExecutable, persisted.ExecutablePath); + Assert.Null(context.LocalAiUpgradeOriginalInstall); + if (newExecutable is not null) + Assert.False(File.Exists(newExecutable)); + + LocalAiReconcileResult retry = await reconciler.ReconcileAsync( + temp.Path, plan, "GPU-0", CancellationToken.None); + Assert.False(retry.Reused); + Assert.Null(retry.RuntimeInstall); + Assert.Equal(cacheRoot, retry.ModelInstall?.CacheRoot); + } + } + [Fact] public async Task Reconciler_RejectsMigrationCacheRootInsideManagedInstallTree() { @@ -1478,6 +1748,28 @@ private static LocalAiInstallManifest CreateManifest( }; } + private static LocalAiInstallManifest CreateRetiredManifest( + string localDataDirectory, + LocalInferencePlan plan, + string gpuId) + { + LlamaRuntimeVariant retired = LlamaRuntimeCatalog.FindInstalled("b10655-cuda13-x64")!; + return CreateManifest(localDataDirectory, plan, gpuId) with + { + EngineVersion = "b10655", + RuntimeId = retired.Id, + // Use the shipped path, not the installer helper under test. + ExecutablePath = Path.Combine("engines", "llama-server", "b10655", "win-x64", "llama-server.exe"), + RuntimeAssets = retired.Artifacts.Select(artifact => new LocalAiAssetReceipt + { + FileName = Path.GetFileName(artifact.RelativePath), + SourceUrl = artifact.DownloadUri.AbsoluteUri, + SizeBytes = artifact.SizeBytes, + Sha256 = artifact.Sha256.Value, + }).ToImmutableArray(), + }; + } + private static LocalModelInfo CreateModel(byte[] bytes) { var source = new HuggingFaceRevisionSource("owner/repo", new string('a', 40)); @@ -1511,6 +1803,40 @@ private static LocalModelInfo CreateModel(byte[] bytes) SupportsVision: false); } + private static LocalModelInfo CreateModelWithDraft(byte[] primaryBytes, PinnedArtifact draftWeights) + { + var source = new HuggingFaceRevisionSource("owner/repo", new string('a', 40)); + var artifact = new PinnedArtifact( + "test-model-dflash", + ArtifactRole.ModelWeights, + source, + "model.gguf", + primaryBytes.Length, + new Sha256Digest(Sha256(primaryBytes))); + return new LocalModelInfo( + "test-model-dflash", + "Test model (DFlash)", + "Test", + "Q4", + artifact, + new LocalModelRunRecipe( + 128, + 128, + 1, + 1, + 1, + 128, + true, + true, + SpeculativeDecodingMode.DraftDFlash, + 1, + new ModelSamplingPreset(0.6, 20, 0.95, 0, 1, 0), + draftWeights), + IsDefault: true, + IsExplicitAlternative: false, + SupportsVision: false); + } + private static LlamaRuntimeVariant CreateRuntime(byte[] binaryZip, byte[] dependencyZip) { var source = new GitHubReleaseSource("owner/repo", "v1", new string('b', 40)); @@ -1731,6 +2057,14 @@ public Task InspectAsync( Task.FromResult(new LlamaRuntimeInspection(true, "valid", null)); } + private sealed class InvalidRuntimeInspector : ILlamaRuntimeInspector + { + public Task InspectAsync( + string installDirectory, + CancellationToken cancellationToken) => + Task.FromResult(new LlamaRuntimeInspection(false, "invalid", "simulated corrupted runtime")); + } + private sealed class AcceptingModelVerifier : ILocalAiModelFileVerifier { public Task VerifyActiveAsync( @@ -1743,6 +2077,12 @@ public Task VerifyLegacyCompatibilityAsync( LocalAiPaths paths, PinnedArtifact artifact, CancellationToken cancellationToken) => Task.FromResult(true); + + public Task VerifyAdditionalAssetAsync( + LocalAiResolvedInstall install, + string cachedAssetPath, + PinnedArtifact artifact, + CancellationToken cancellationToken) => Task.FromResult(true); } private sealed class RejectingModelVerifier : ILocalAiModelFileVerifier @@ -1757,6 +2097,12 @@ public Task VerifyLegacyCompatibilityAsync( LocalAiPaths paths, PinnedArtifact artifact, CancellationToken cancellationToken) => Task.FromResult(false); + + public Task VerifyAdditionalAssetAsync( + LocalAiResolvedInstall install, + string cachedAssetPath, + PinnedArtifact artifact, + CancellationToken cancellationToken) => Task.FromResult(false); } private sealed class CompleteModelRepairStep(string localDataDirectory, LocalInferencePlan plan) : SetupStep @@ -1801,6 +2147,18 @@ public override Task ExecuteAsync(SetupContext ctx, CancellationToke } } + private sealed class UpgradeCheckpointStep( + string id, + Func shouldFail) : SetupStep + { + public override string Id => id; + public override string DisplayName => id; + public override bool CanRetry => false; + + public override Task ExecuteAsync(SetupContext ctx, CancellationToken ct) => + Task.FromResult(shouldFail(ctx) ? StepResult.Fail("Injected upgrade failure.") : StepResult.Ok("Checked.")); + } + private sealed class TempDirectory : IDisposable { public TempDirectory() diff --git a/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs b/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs index 1bf2e8ea8..b5d4e2f1a 100644 --- a/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs +++ b/tests/OpenClaw.SetupEngine.Tests/LocalAiPortHandoffTests.cs @@ -139,8 +139,11 @@ private static SetupContext CreateContext(LocalAiConfig localAi, string? localDa new GpuInfo( GpuVendor.Nvidia, "NVIDIA RTX Spark N1X (6144-core Blackwell RTX GPU)", - GpuVisibleMemoryBytes: 25_702_694_912, - FreeGpuVisibleMemoryBytes: 25_702_694_912, + // A real 48GB-SKU RTX Spark's measured cuMemGetInfo total, so this + // fixture routes through the fixed SKU table instead of landing + // below the smallest (32GB) recommended tier. + GpuVisibleMemoryBytes: 48_585_498_624, + FreeGpuVisibleMemoryBytes: 48_585_498_624, DriverVersion: "616.00", CudaMajorVersion: 13, StableId: "GPU-SPARK"), diff --git a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs index 471866ffc..fe9a929c4 100644 --- a/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs +++ b/tests/OpenClaw.Shared.Tests/LocalInferenceQualificationTests.cs @@ -298,7 +298,7 @@ UuidFailure is not null } [Theory] - [InlineData(RuntimeArchitecture.X64, "NVIDIA RTX Spark N1X", LlamaRuntimeCatalog.X64RuntimeId)] + [InlineData(RuntimeArchitecture.X64, "NVIDIA GeForce RTX 5080", LlamaRuntimeCatalog.X64RuntimeId)] [InlineData(RuntimeArchitecture.Arm64, "NVIDIA GeForce RTX 5090", LlamaRuntimeCatalog.Arm64RuntimeId)] public void Evaluate_RoutesRuntimeByArchitectureWithoutGpuSkuPairing( RuntimeArchitecture architecture, @@ -315,6 +315,265 @@ public void Evaluate_RoutesRuntimeByArchitectureWithoutGpuSkuPairing( Assert.Equal(KvCachePrecision.Q8_0, result.Plan?.Profile.KeyCachePrecision); } + // Boundaries verified against real hardware: a real 48GB-SKU RTX Spark + // reads ~45.25 GiB (48,585,498,624 bytes) via cuMemGetInfo, matching the + // "Gb48" case below almost exactly. + [Theory] + [InlineData(30, null)] // 32GB SKU: no local AI recommended + [InlineData(45, LocalModelCatalog.Qwen35B_IQ4XSModelId)] // 48GB SKU -> 24GB recipe + [InlineData(62, LocalModelCatalog.Qwen38_27BModelId)] // 64GB SKU -> 28GB recipe + [InlineData(120, LocalModelCatalog.Qwen38_27B_DFlashModelId)] // 128GB SKU -> 48GB recipe (default) + public void Evaluate_RoutesRtxSparkByFixedSkuTable(long totalGiB, string? expectedModelId) + { + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware(RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark", totalGiB, totalGiB))); + + if (expectedModelId is null) + { + Assert.Equal(LocalInferenceEligibilityStatus.Unsupported, result.Status); + Assert.Equal(LocalInferenceEligibilityFailureCode.CatalogSelectionFailed, result.FailureCode); + Assert.Equal(LocalInferenceSelectionFailureCode.NotRecommendedForSku, result.SelectionFailureCode); + Assert.Null(result.Plan); + } + else + { + Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); + Assert.Equal(expectedModelId, result.Plan?.Model.Id); + } + } + + [Fact] + public void Evaluate_Rtx5090WithSparkSizedMemoryIgnoresSkuTable() + { + // A non-Spark GPU that happens to have Spark-sized memory must still + // take the generic priority/fit-test path -- SKU routing is keyed + // strictly off the RTX Spark name, not memory size. + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware(RuntimeArchitecture.X64, Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", totalGiB: 45, freeGiB: 45))); + + Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); + Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id); + } + + [Fact] + public void Evaluate_SparkRecipeIsBoundToTheSparkGpuOnMixedHosts() + { + // The SKU table answers "what should THIS Spark run", so the recipe and the + // GPU that runs it must be the same adapter. A discrete GPU with more free + // memory must not win the eligibility ranking and end up running a recipe + // that was chosen for the Spark. + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware( + RuntimeArchitecture.Arm64, + Gpu("NVIDIA RTX Spark N1X", "GPU-spark", totalGiB: 45, freeGiB: 45), + Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", totalGiB: 80, freeGiB: 80))); + + Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); + Assert.Equal(LocalModelCatalog.Qwen35B_IQ4XSModelId, result.Plan?.Model.Id); + Assert.Equal("GPU-spark", result.SelectedGpu?.StableId); + Assert.Equal("GPU-spark", result.Plan?.BoundGpuStableId); + } + + [Fact] + public void Evaluate_UnrecommendedSparkSkuStillQualifiesADiscreteGpuOnTheSameHost() + { + // A 32 GB Spark has no recommended model, but that is a statement about the + // Spark, not about the host. An eligible discrete GPU beside it must still + // qualify through the generic path instead of the whole host being rejected. + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware( + RuntimeArchitecture.X64, + Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", totalGiB: 30, freeGiB: 30), + Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", totalGiB: 32, freeGiB: 32))); + + Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); + Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id); + Assert.Equal("GPU-5090", result.SelectedGpu?.StableId); + Assert.Null(result.Plan?.BoundGpuStableId); + } + + [Fact] + public void Evaluate_UnrecommendedSparkSkuAloneStillReportsNotRecommended() + { + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware(RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30))); + + Assert.Equal(LocalInferenceEligibilityStatus.Unsupported, result.Status); + Assert.Equal( + LocalInferenceSelectionFailureCode.NotRecommendedForSku, + result.SelectionFailureCode); + } + + [Theory] + [InlineData(45)] + [InlineData(62)] + [InlineData(120)] + public void Evaluate_SparkRecommendationRoundTrippedBySetupKeepsItsSkuProfile(long totalGiB) + { + // Normal setup persists the recommended model id and passes it back as an + // explicit request, so the recommendation must resolve identically both ways. + // Otherwise the SKU's pinned profile (the 64 GB tier's reduced context is not + // the largest that merely fits) is silently replaced by the generic fit-test. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, + Gpu("NVIDIA RTX Spark N1X", "GPU-spark", totalGiB, totalGiB)); + + LocalInferenceEligibilityResult recommended = LocalInferenceEligibility.Evaluate(hardware); + LocalInferenceEligibilityResult roundTripped = LocalInferenceEligibility.Evaluate( + hardware, + recommended.Plan!.Model.Id); + + Assert.Equal(recommended.Plan!.Model.Id, roundTripped.Plan?.Model.Id); + Assert.Equal(recommended.Plan!.Profile.Id, roundTripped.Plan?.Profile.Id); + Assert.Equal("GPU-spark", roundTripped.Plan?.BoundGpuStableId); + Assert.Equal(recommended.SelectedGpu?.StableId, roundTripped.SelectedGpu?.StableId); + } + + [Fact] + public void Evaluate_ExplicitNonRecommendedModelOnSparkStillUsesTheGenericFitTest() + { + // A real user override must not be forced onto the SKU recipe. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, + Gpu("NVIDIA RTX Spark N1X", "GPU-spark", 62, 62)); + + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + hardware, + LocalModelCatalog.Qwen27BModelId); + + Assert.Equal(LocalModelCatalog.Qwen27BModelId, result.Plan?.Model.Id); + Assert.Null(result.Plan?.BoundGpuStableId); + } + + [Fact] + public void EvaluateForConfiguredAvailability_KeepsAValidSavedModelOn32GbSpark() + { + // A 32 GB Spark has no recommended default, but it still runs a model that was + // already configured. Rerunning setup must not switch Local AI off on that machine. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30)); + + LocalInferenceEligibilityResult result = + LocalInferenceEligibility.EvaluateForConfiguredAvailability( + hardware, + LocalModelCatalog.Qwen38_27BModelId); + + Assert.True(result.CanInstall); + Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id); + Assert.NotNull(result.SelectedGpu); + } + + [Fact] + public void EvaluateForConfiguredAvailability_PreservesTheRecoveryPinnedModelAndProfileOn32GbSpark() + { + // Recovery pins the configured model and reuses its resolved plan. On a SKU with no + // recommended default that selection must survive the availability gate with the same + // model and the same profile the explicit path resolves, so a recovery rerun does not + // silently move an existing install to a different context or KV precision. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30)); + const string pinnedModelId = LocalModelCatalog.Qwen38_27BModelId; + + LocalInferenceEligibilityResult availability = + LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, pinnedModelId); + LocalInferenceEligibilityResult pinned = + LocalInferenceEligibility.Evaluate(hardware, pinnedModelId); + + Assert.True(availability.CanInstall); + Assert.Equal(pinnedModelId, availability.Plan?.Model.Id); + Assert.Equal(pinned.Plan?.Profile.Id, availability.Plan?.Profile.Id); + Assert.Equal(pinned.Plan?.Profile.ContextTokens, availability.Plan?.Profile.ContextTokens); + Assert.Equal(pinned.Plan?.Profile.KeyCachePrecision, availability.Plan?.Profile.KeyCachePrecision); + Assert.Equal(pinned.SelectedGpu?.StableId, availability.SelectedGpu?.StableId); + } + + [Fact] + public void EvaluateForConfiguredAvailability_WithNoSavedModelStillReportsNotRecommended() + { + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30)); + + LocalInferenceEligibilityResult result = + LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, configuredModelId: null); + + Assert.False(result.CanInstall); + Assert.Equal( + LocalInferenceSelectionFailureCode.NotRecommendedForSku, + result.SelectionFailureCode); + } + + [Fact] + public void EvaluateForConfiguredAvailability_UnknownSavedModelReportsThatModelsFailure() + { + // The reason must name what is wrong with the saved selection, not fall back to the + // SKU's generic no-recommendation message, or setup offers no path to recovery. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-spark32", 30, 30)); + + LocalInferenceEligibilityResult result = + LocalInferenceEligibility.EvaluateForConfiguredAvailability(hardware, "no-such-model-id"); + + Assert.False(result.CanInstall); + Assert.Equal(LocalInferenceSelectionFailureCode.UnknownModel, result.SelectionFailureCode); + Assert.Equal( + LocalInferenceUnavailableReasonKind.UnknownModel, + LocalInferenceEligibilityDiagnostics.GetUnavailableReason(result).Kind); + } + + [Fact] + public void EvaluateForConfiguredAvailability_OversizedSavedModelReportsCapacityNotSkuPolicy() + { + // A saved model that no longer fits must report the capacity shortfall, including the + // model name and the required and detected memory the setup page renders. + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.Arm64, Gpu("NVIDIA RTX Spark N1X", "GPU-sparkSmall", 12, 12)); + + LocalInferenceEligibilityResult result = + LocalInferenceEligibility.EvaluateForConfiguredAvailability( + hardware, + LocalModelCatalog.Qwen38_27BModelId); + + Assert.False(result.CanInstall); + Assert.Equal( + LocalInferenceEligibilityFailureCode.InsufficientGpuMemory, + result.FailureCode); + LocalInferenceUnavailableReason reason = + LocalInferenceEligibilityDiagnostics.GetUnavailableReason(result); + Assert.Equal(LocalInferenceUnavailableReasonKind.InsufficientGpuMemory, reason.Kind); + Assert.False(string.IsNullOrWhiteSpace(reason.ModelDisplayName)); + } + + [Fact] + public void EvaluateForConfiguredAvailability_LeavesNonSparkDevicesUnchanged() + { + HostHardwareInfo hardware = Hardware( + RuntimeArchitecture.X64, Gpu("NVIDIA GeForce RTX 5090", "GPU-5090", 32, 32)); + + LocalInferenceEligibilityResult withSaved = + LocalInferenceEligibility.EvaluateForConfiguredAvailability( + hardware, LocalModelCatalog.Qwen27BModelId); + LocalInferenceEligibilityResult device = LocalInferenceEligibility.Evaluate(hardware); + + Assert.True(withSaved.CanInstall); + Assert.Equal(device.Plan?.Model.Id, withSaved.Plan?.Model.Id); + } + + [Fact] + public void Evaluate_HugeNonSparkGpuStillDefaultsToTheRecommendedModel() + { + // A priority-0, IsExplicitAlternative model must never win the generic + // default/fallback pick regardless of available memory. The guard in + // SelectDefaultModelAndProfile keeps that true as the catalog grows. + LocalInferenceEligibilityResult result = LocalInferenceEligibility.Evaluate( + Hardware(RuntimeArchitecture.X64, Gpu("NVIDIA arbitrary huge adapter", "GPU-huge", totalGiB: 200, freeGiB: 200))); + + Assert.Equal(LocalInferenceEligibilityStatus.Eligible, result.Status); + Assert.Equal(LocalModelCatalog.Qwen38_27BModelId, result.Plan?.Model.Id); + Assert.DoesNotContain( + LocalModelCatalog.Models, + m => m.RecommendationPriority == 0 && m.IsExplicitAlternative && m.Id == result.Plan!.Model.Id); + } + [Fact] public void Evaluate_UnsetModelChoosesHighestPriorityModelThatFitsTotalCapacity() { @@ -565,4 +824,38 @@ private static GpuInfo Gpu( CudaMajorVersion: 13, StableId: stableId); + [Theory] + [InlineData("b10655-cuda13-x64", "b10655")] + [InlineData("b10655-cuda13-arm64", "b10655")] + public void FindInstalled_ResolvesRetiredRuntimeSoExistingInstallsStayLaunchable( + string runtimeId, + string expectedReleaseTag) + { + // An installation recorded before the runtime bump must keep resolving its own + // receipt, otherwise updating the app strands it until setup repairs it. + LlamaRuntimeVariant? installed = LlamaRuntimeCatalog.FindInstalled(runtimeId); + + Assert.NotNull(installed); + Assert.Equal(runtimeId, installed.Id); + Assert.Equal(expectedReleaseTag, installed.ReleaseTag); + Assert.False(string.Equals(LlamaRuntimeCatalog.ReleaseTag, installed.ReleaseTag, StringComparison.Ordinal)); + } + + [Fact] + public void RetiredRuntimeIsNeverOfferedForNewInstalls() + { + Assert.DoesNotContain( + LlamaRuntimeCatalog.Variants, + variant => variant.ReleaseTag != LlamaRuntimeCatalog.ReleaseTag); + Assert.All( + LlamaRuntimeCatalog.Variants, + variant => Assert.Equal(LlamaRuntimeCatalog.ReleaseTag, variant.ReleaseTag)); + } + + [Fact] + public void FindInstalled_RejectsUnknownRuntimeId() + { + Assert.Null(LlamaRuntimeCatalog.FindInstalled("b00000-cuda13-x64")); + Assert.Null(LlamaRuntimeCatalog.FindInstalled(null)); + } } diff --git a/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs b/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs index 1ade70466..c191edc98 100644 --- a/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs +++ b/tests/OpenClaw.Tray.Tests/LocalAiSetupUxContractTests.cs @@ -249,13 +249,17 @@ public void CapabilitiesReview_InstallDistroCard_UsesSimplifiedCopy() } /// - /// The "is Local AI unavailable" gate and the recommended/selected model must be decided - /// from device-level eligibility (the best catalog model this hardware can run), not from - /// the currently configured SelectedModelId. A stale/removed model, or one that exists but - /// this hardware cannot run at all, must be reconciled to a valid one instead of making an - /// otherwise-capable device look unavailable or leaving setup on a known-incompatible model. - /// A merely busy GPU (EligibleButBusy) is not reconciled away: CanInstall covers that case + /// The recommended model must be decided from device-level eligibility (the best catalog + /// model this hardware can run), not from the currently configured SelectedModelId. A + /// stale/removed model, or one that exists but this hardware cannot run at all, must be + /// reconciled to a valid one instead of leaving setup on a known-incompatible model. A + /// merely busy GPU (EligibleButBusy) is not reconciled away: CanInstall covers that case /// and the same model would still work once the GPU frees up. + /// + /// The "is Local AI unavailable" gate additionally honours an already configured model. + /// A SKU with no recommended default (RTX Spark 32 GB) is a statement about what to + /// install by default, not about what the device can run, so a configured selection that + /// still passes the capacity fit-test keeps Local AI available on a setup rerun. /// [Fact] public void CapabilitiesReview_GatesOnDeviceEligibilityAndReconcilesStaleSelectedModel() @@ -272,15 +276,18 @@ public void CapabilitiesReview_GatesOnDeviceEligibilityAndReconcilesStaleSelecte AssertInOrder( method, "LocalInferenceEligibilityResult deviceEligibility = LocalInferenceEligibility.Evaluate(_localAiHardware);", - "if (!deviceEligibility.CanInstall || deviceEligibility.Plan is null || deviceEligibility.SelectedGpu is null)", - "hardwareReason = DescribeLocalAiUnavailable(deviceEligibility);", + "_localAiRecommendedModelId = deviceEligibility.CanInstall", + "LocalInferenceEligibility.EvaluateForConfiguredAvailability(", + "_config!.LocalAi.SelectedModelId);", + "if (!availability.CanInstall || availability.Plan is null || availability.SelectedGpu is null)", + "hardwareReason = DescribeLocalAiUnavailable(availability);", "LocalInferenceEligibilityResult selectedEligibility =", "LocalInferenceEligibility.Evaluate(_localAiHardware, selectedModelId);", "if (_localAiRecoveryModelPinned)", "eligibility = selectedEligibility;", "else if (!selectedEligibility.CanInstall)", "_config.LocalAi.SelectedModelId = null;", - "_config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? deviceEligibility.Plan.Model.Id;", + "_config.LocalAi.SelectedModelId ??= _localAiRecommendedModelId ?? availability.Plan.Model.Id;", "eligibility ??= LocalInferenceEligibility.Evaluate(", "_config.LocalAi.SelectedModelId);"); } diff --git a/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs b/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs index 3c9261513..50a859f9b 100644 --- a/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs +++ b/tests/OpenClaw.Tray.Tests/Presentation/LocalAiPageViewModelTests.cs @@ -737,6 +737,185 @@ private static async Task WaitForConditionAsync(Func condition, TimeSpan t } } + /// + /// A 32 GB RTX Spark. The fixed SKU table gives this device no recommended default, which + /// is a statement about fresh installs rather than about what the device can run. + /// + private static HostHardwareInfo Create32GbSparkHardware() => + new( + Architecture.Arm64, + TotalPhysicalMemoryBytes: 128_000_000_000, + AvailablePhysicalMemoryBytes: 96_000_000_000, + Gpus: + [ + new GpuInfo( + GpuVendor.Nvidia, + "NVIDIA RTX Spark N1X", + GpuVisibleMemoryBytes: 30L * 1024 * 1024 * 1024, + FreeGpuVisibleMemoryBytes: 30L * 1024 * 1024 * 1024, + DriverVersion: "620.0", + CudaMajorVersion: 13, + StableId: "GPU-spark32"), + ], + VulkanAvailable: false); + + private static LocalAiPageViewModel CreateViewModel( + LocalAiRuntimeSnapshot snapshot, + HostHardwareInfo hardware, + out FakeAppCommands commands, + out PermissionsPageRuntimeSource gatewaySource) + { + var runtime = new FakeLocalAiRuntime(snapshot); + var runtimeHost = new FakePermissionsPageRuntimeHost + { + ConnectionSnapshot = GatewayConnectionSnapshot.Idle with + { + OperatorState = RoleConnectionState.Idle, + }, + }; + gatewaySource = new PermissionsPageRuntimeSource(runtimeHost); + commands = new FakeAppCommands(); + return new LocalAiPageViewModel( + runtime, + gatewaySource, + commands, + new RecordingUiDispatcher(), + new FixedHardwareProbe(hardware)); + } + + /// + /// A 32 GB Spark whose managed receipt names a model this hardware still runs keeps the + /// Local AI entry point available, so a broken or unverified runtime can reach Retry Setup + /// and an installed model can still be changed. + /// + [Fact] + public async Task Spark32Gb_WithValidManagedReceipt_KeepsRetrySetupReachable() + { + LocalAiRuntimeSnapshot snapshot = CreateInstalledSnapshot() with + { + ModelId = LocalModelCatalog.Qwen38_27BModelId, + State = LocalAiRuntimeState.Failed, + ModelEvidence = new LocalAiModelEvidence( + LocalAiModelAvailabilityState.Unknown, + DateTimeOffset.UtcNow), + }; + using var viewModel = CreateViewModel( + snapshot, Create32GbSparkHardware(), out _, out PermissionsPageRuntimeSource source); + using (source) + { + await ActivateAndWaitForAvailabilityAsync(viewModel); + + Assert.True(viewModel.IsAvailabilityKnown); + Assert.True(viewModel.IsLocalAiAvailable); + Assert.True(viewModel.IsSetupAvailable); + Assert.True(viewModel.CanRetrySetup); + } + } + + /// + /// The same 32 GB Spark with no managed installation still receives no default, so Local AI + /// stays unavailable for a fresh setup on that SKU. + /// + [Fact] + public async Task Spark32Gb_WithNoManagedReceipt_StaysUnavailable() + { + LocalAiRuntimeSnapshot snapshot = CreateInstalledSnapshot() with + { + ModelId = null, + State = LocalAiRuntimeState.NotInstalled, + Ownership = LocalAiOwnership.None, + ModelEvidence = new LocalAiModelEvidence( + LocalAiModelAvailabilityState.Unknown, + DateTimeOffset.UtcNow), + }; + using var viewModel = CreateViewModel( + snapshot, Create32GbSparkHardware(), out _, out PermissionsPageRuntimeSource source); + using (source) + { + await ActivateAndWaitForAvailabilityAsync(viewModel); + + Assert.True(viewModel.IsAvailabilityKnown); + Assert.False(viewModel.IsLocalAiAvailable); + Assert.False(viewModel.IsSetupAvailable); + Assert.False(viewModel.CanRetrySetup); + Assert.False(viewModel.CanChangeModel); + // NotRecommendedForSku has no dedicated reason kind, so this currently surfaces the + // generic Unknown copy. Asserted so the behavior is recorded rather than assumed. + Assert.Equal( + LocalInferenceUnavailableReasonKind.Unknown, + viewModel.LocalAiUnavailableReason?.Kind); + } + } + + /// + /// The runtime refresh can publish the managed receipt after the availability probe has + /// already read the earlier snapshot. Availability must be recomputed when that happens, + /// otherwise a first visit leaves a 32 GB Spark fixed at NotRecommendedForSku with Retry + /// Setup and Change Model disabled and Recheck unavailable, until the page is reopened. + /// + [Fact] + public async Task Spark32Gb_WhenReceiptArrivesAfterAvailability_RecomputesAndKeepsRetrySetupReachable() + { + LocalAiRuntimeSnapshot pending = CreateInstalledSnapshot() with + { + ModelId = null, + State = LocalAiRuntimeState.Failed, + ModelEvidence = new LocalAiModelEvidence( + LocalAiModelAvailabilityState.Unknown, + DateTimeOffset.UtcNow), + }; + var refreshGate = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + var runtime = new FakeLocalAiRuntime(pending) + { + RefreshDelay = refreshGate.Task, + RefreshResult = pending with { ModelId = LocalModelCatalog.Qwen38_27BModelId }, + }; + using var gatewaySource = new PermissionsPageRuntimeSource(new FakePermissionsPageRuntimeHost()); + using var viewModel = new LocalAiPageViewModel( + runtime, + gatewaySource, + new FakeAppCommands(), + new RecordingUiDispatcher(), + new FixedHardwareProbe(Create32GbSparkHardware())); + + await ActivateAndWaitForAvailabilityAsync(viewModel); + + // The receipt has not been published yet, so the SKU alone leaves the entry point closed. + Assert.False(viewModel.IsLocalAiAvailable); + Assert.False(viewModel.CanRetrySetup); + + refreshGate.TrySetResult(); + await WaitForAsync(viewModel, () => viewModel.IsLocalAiAvailable); + + Assert.True(viewModel.IsSetupAvailable); + Assert.True(viewModel.CanRetrySetup); + } + + /// + /// A receipt naming a model this catalog no longer knows reports that model's own failure, + /// so the page explains what to fix instead of showing the SKU's generic message. + /// + [Fact] + public async Task Spark32Gb_WithUnknownReceiptModel_ReportsThatModelsReason() + { + LocalAiRuntimeSnapshot snapshot = CreateInstalledSnapshot() with + { + ModelId = "no-such-model-id", + }; + using var viewModel = CreateViewModel( + snapshot, Create32GbSparkHardware(), out _, out PermissionsPageRuntimeSource source); + using (source) + { + await ActivateAndWaitForAvailabilityAsync(viewModel); + + Assert.True(viewModel.IsAvailabilityKnown); + Assert.False(viewModel.IsLocalAiAvailable); + Assert.Equal( + LocalInferenceUnavailableReasonKind.UnknownModel, + viewModel.LocalAiUnavailableReason?.Kind); + } + } + private static HostHardwareInfo CreateQualifiedHardware() => new( Architecture.X64, @@ -954,8 +1133,21 @@ public Task RestartAsync(CancellationToken cancellationT return Task.FromResult(Snapshot); } - public Task RefreshAsync(CancellationToken cancellationToken = default) => - Task.FromResult(Snapshot); + /// Snapshot the refresh publishes, letting a test model a receipt that the runtime + /// only resolves after construction. + public LocalAiRuntimeSnapshot? RefreshResult { get; init; } + + /// Holds the refresh open until this completes, so a test can land the refresh + /// after the availability probe has already read the earlier snapshot. + public Task? RefreshDelay { get; init; } + + public async Task RefreshAsync(CancellationToken cancellationToken = default) + { + if (RefreshDelay is { } delay) + await delay.ConfigureAwait(false); + Snapshot = RefreshResult ?? Snapshot; + return Snapshot; + } public ValueTask DisposeAsync() => ValueTask.CompletedTask; }