diff --git a/go/internal/httpapi/smith_autonomy_handlers.go b/go/internal/httpapi/smith_autonomy_handlers.go index c89eede..aaacaf1 100644 --- a/go/internal/httpapi/smith_autonomy_handlers.go +++ b/go/internal/httpapi/smith_autonomy_handlers.go @@ -13,16 +13,19 @@ package httpapi // does, the same "always the safe direction" posture reject/checkpoint-abort // use elsewhere in this API. Because the decision depends on the request // BODY (which must be parsed first), it can't be expressed as a route-level -// requireAssurance(...) middleware wrap the way every other step-up-gated -// route in this file is — so this handler evaluates the policy directly, -// duplicating requireAssurance's core check rather than complicating that -// shared middleware with a body-dependent special case. +// requireAssurance(...)/requireStrictAssurance(...) middleware wrap the way +// every other step-up-gated route in this file is — so this handler calls +// the shared evaluateAssurance helper directly on the escalating path. +// +// It deliberately uses the requireStrictAssurance shape (no bearer bypass), +// not requireAssurance's blanket one: escalating smith's standing autonomy +// is at least as sensitive as key management (#37), since it's the switch +// that lets smith execute system-modifying procedures unattended. import ( "context" "encoding/json" "net/http" - "time" "github.com/jsaigou/the-forge/internal/authz" "github.com/jsaigou/the-forge/internal/smith" @@ -128,26 +131,25 @@ func (s *Server) handleSmithAutonomyPut(w http.ResponseWriter, r *http.Request) writeError(w, http.StatusUnauthorized, "Authentication required") return } - if ident.KeyID == "" && s.deps.PolicyStore != nil { - pctx, cancel := context.WithTimeout(ctx, 2*time.Second) - policy, err := s.deps.PolicyStore.Load(pctx) - cancel() - if err != nil { - writeError(w, http.StatusInternalServerError, "policy load failed") - return - } - ttl := s.deps.StepUpTTL - if ttl == 0 { - ttl = authz.DefaultStepUpTTL - } - eval := authz.NewPolicyEvaluator(policy, ttl, time.Now) - decision := eval.Evaluate(authz.ResourceActionSmithAutonomy, ident.Assurance, ident.AssuranceAt) - if !decision.Allowed { - writeJSON(w, http.StatusForbidden, map[string]string{ - "error": "step_up_required", "required": string(decision.Required), "resource": decision.Resource, - }) - return - } + // No bearer bypass here — unlike requireAssurance's blanket one for + // ordinary a0/MCP traffic, escalating smith's standing autonomy is at + // least as sensitive as key management (#37's requireStrictAssurance + // carve-out), since it's the switch that lets smith execute + // system-modifying procedures unattended. Every identity is + // evaluated through the same evaluateAssurance path requireAssurance/ + // requireStrictAssurance use; a bearer identity carries no session + // assurance, so it always fails a password-or-above resource here — + // only a stepped-up browser session can flip this switch on. + allowed, decision, err := s.evaluateAssurance(ctx, ident, authz.ResourceActionSmithAutonomy) + if err != nil { + writeError(w, http.StatusInternalServerError, "policy load failed") + return + } + if !allowed { + writeJSON(w, http.StatusForbidden, map[string]string{ + "error": "step_up_required", "required": string(decision.Required), "resource": decision.Resource, + }) + return } } diff --git a/go/internal/httpapi/smith_autonomy_handlers_test.go b/go/internal/httpapi/smith_autonomy_handlers_test.go index 8ce4a78..011fc09 100644 --- a/go/internal/httpapi/smith_autonomy_handlers_test.go +++ b/go/internal/httpapi/smith_autonomy_handlers_test.go @@ -163,3 +163,48 @@ func TestHandleSmithAutonomy_EscalationRequiresStepUp(t *testing.T) { t.Errorf("policy.Enabled = true after refused escalations, want false") } } + +// TestHandleSmithAutonomy_BearerKeyCannotEscalateWithoutStepUp is the +// regression test for the smith-autonomy analogue of issue #37: before this +// fix, the escalating branch only ever evaluated step-up when +// ident.KeyID == "" (a session), so ANY bearer forge key with RoleOperator+ +// — even one never granted a human step-up — could flip smith's standing +// autonomy on and let it start executing system-modifying procedures +// unattended. A non-escalating PUT (disabling) must still work over bearer, +// same as #37's self-rotation carve-out preserved ordinary bearer use. +func TestHandleSmithAutonomy_BearerKeyCannotEscalateWithoutStepUp(t *testing.T) { + s, auth, _ := serverWithSmithActionsAuth(t) + token, err := auth.MintKey(t.Context(), authz.KindForge, "admin-bot", "", authz.RoleAdmin, "", time.Time{}) + if err != nil { + t.Fatalf("MintKey: %v", err) + } + + req := httptest.NewRequest("PUT", "/api/v1/smith/autonomy", bytes.NewBufferString(`{"enabled":true}`)) + req.Header.Set("Authorization", "Bearer "+token) + req.Header.Set("Content-Type", "application/json") + rec := httptest.NewRecorder() + s.Handler().ServeHTTP(rec, req) + if rec.Code != http.StatusForbidden { + t.Fatalf("bearer PUT enabled:true = %d, want 403; body=%s", rec.Code, rec.Body.String()) + } + var errResp map[string]string + json.NewDecoder(rec.Body).Decode(&errResp) + if errResp["error"] != "step_up_required" || errResp["resource"] != authz.ResourceActionSmithAutonomy { + t.Errorf("errResp = %+v, want step_up_required for action.smith.autonomy", errResp) + } + + // A non-escalating PUT (disabling) over the same bearer key still works. + req = httptest.NewRequest("PUT", "/api/v1/smith/autonomy", bytes.NewBufferString(`{"enabled":false}`)) + req.Header.Set("Authorization", "Bearer "+token) + req.Header.Set("Content-Type", "application/json") + rec = httptest.NewRecorder() + s.Handler().ServeHTTP(rec, req) + if rec.Code != http.StatusOK { + t.Fatalf("bearer PUT enabled:false = %d, want 200; body=%s", rec.Code, rec.Body.String()) + } + + final := s.deps.Smith.AutonomyPolicy(context.Background()) + if final.Enabled { + t.Errorf("policy.Enabled = true after a refused bearer escalation, want false") + } +} diff --git a/go/internal/registry/registry.go b/go/internal/registry/registry.go index 0f25ed5..4b8971a 100644 --- a/go/internal/registry/registry.go +++ b/go/internal/registry/registry.go @@ -381,7 +381,11 @@ func (s *catalogSnapshot) resolveLogos(cfgLogo, cfgLogoDark string, mdl store.Mo } // resolveModalities (Sprint J1) narrows a model's architectural modalities -// to what one specific config can actually deliver: +// to what one specific config can actually deliver. The precedence itself +// now lives in store.ResolveModalities (2026-09-13, so a0's /v1/models and +// this package's cards can never disagree) — this is a thin adapter that +// resolves the mmproj-missing lookup against this snapshot's artifact map +// and re-shapes the result into this package's []ModalityGap wire type: // // 1. "text" is always enabled — every config can do plain text. // 2. An explicit cfg.Modalities override wins verbatim, even an empty one @@ -393,37 +397,12 @@ func (s *catalogSnapshot) resolveLogos(cfgLogo, cfgLogoDark string, mdl store.Mo // unavailable too ("mmproj file missing on disk") rather than silently // claiming a capability the config can't currently serve. func (s *catalogSnapshot) resolveModalities(c store.Config, mdl store.Model) (enabled []string, unavailable []ModalityGap) { - nonText := func(mods []string) []string { - out := make([]string, 0, len(mods)) - for _, m := range mods { - if m != "text" { - out = append(out, m) - } - } - return out - } - - if c.Modalities != nil { - enabled = append([]string{"text"}, nonText(*c.Modalities)...) - return enabled, nil - } - - if c.MMProjArtifactID == 0 { - for _, m := range nonText(mdl.Modalities) { - unavailable = append(unavailable, ModalityGap{ID: m, Reason: "no mmproj linked"}) - } - return []string{"text"}, unavailable - } - - if a, ok := s.artifactByID[c.MMProjArtifactID]; ok && a.Missing { - for _, m := range nonText(mdl.Modalities) { - unavailable = append(unavailable, ModalityGap{ID: m, Reason: "mmproj file missing on disk"}) - } - return []string{"text"}, unavailable + a, ok := s.artifactByID[c.MMProjArtifactID] + res := store.ResolveModalities(c, mdl, ok && a.Missing) + for _, m := range res.Unavailable { + unavailable = append(unavailable, ModalityGap{ID: m, Reason: res.Reason}) } - - enabled = append([]string{"text"}, nonText(mdl.Modalities)...) - return enabled, nil + return res.Enabled, unavailable } // benchesFor unions a model's benchmarks (capability scores — intrinsic to diff --git a/go/internal/router/catalog.go b/go/internal/router/catalog.go index 449705f..dfd9c32 100644 --- a/go/internal/router/catalog.go +++ b/go/internal/router/catalog.go @@ -226,6 +226,22 @@ type ModelsResponse struct { Data []ModelEntry `json:"data"` } +// ModelArchitecture is an OpenRouter/llama-swap-style inline modality block +// (2026-09-13), shaped specifically for the OpenCode discovery plugin's +// "llama-swap" modelInfoFormat, which reads architecture.input_modalities / +// architecture.output_modalities off this same /v1/models response with no +// second request. +// +// VOCABULARY TRAP: that reader only recognizes +// text|image|audio|video|pdf and silently drops anything else — the +// catalog's own vocabulary is text|vision|audio, so "vision" MUST be +// mapped to "image" here (see openCodeModalities) or vision support +// disappears from OpenCode with no error anywhere. +type ModelArchitecture struct { + InputModalities []string `json:"input_modalities,omitempty"` + OutputModalities []string `json:"output_modalities,omitempty"` +} + // ModelEntry is one entry in the /v1/models list. type ModelEntry struct { ID string `json:"id"` @@ -233,6 +249,176 @@ type ModelEntry struct { Created int64 `json:"created"` OwnedBy string `json:"owned_by"` ContextLength int `json:"context_length,omitempty"` + + // Name is the catalog Model's display name, emitted ONLY when it is + // unambiguous across the whole listing — two Configs of one Model + // (e.g. gemma4-26b-a4b / gemma4-26b-a4b-nothink) would otherwise render + // as two identically-named entries in a consumer's model picker. Empty + // means the consumer falls back to formatting the id itself. See + // assignUniqueDisplayNames. + Name string `json:"name,omitempty"` + + // Architecture carries input/output modalities for this entry. nil + // (omitted from the JSON) means "unknown", NOT "text only" — a catalog + // read failure must never make a0 actively claim a vision-capable + // model can't see. See modalitySnapshot. + Architecture *ModelArchitecture `json:"architecture,omitempty"` +} + +// modalitySnapshot holds the extra catalog reads /v1/models needs to answer +// "what input can this entry actually accept" — separate from the +// offerings/configs reads BuildModelsResponse already does, since neither +// of those rows carries a Model directly (an Offering has ModelID; a +// Config has VariantID, one hop further from Model via Variant). ok is +// false if ANY read failed — callers then omit `architecture` entirely +// (a read failure must never be read as "text only", see ModelEntry.Architecture). +type modalitySnapshot struct { + ok bool + modelByID map[int64]store.Model + modelIDByVar map[int64]int64 // Variant.ID -> Variant.ModelID + mmprojMissing map[int64]bool // Artifact.ID -> Missing (mmproj artifacts only) +} + +// loadModalitySnapshot reads ListModels + ListVariants + ListArtifacts — +// three small full-table reads (tens of rows each, live) inside +// BuildModelsResponse's existing 2s context budget. /v1/models is a listing +// endpoint, not the chat hot path. +func loadModalitySnapshot(ctx context.Context, storeCat store.Catalog) modalitySnapshot { + models, err := storeCat.ListModels(ctx) + if err != nil { + return modalitySnapshot{} + } + variants, err := storeCat.ListVariants(ctx) + if err != nil { + return modalitySnapshot{} + } + artifacts, err := storeCat.ListArtifacts(ctx) + if err != nil { + return modalitySnapshot{} + } + + s := modalitySnapshot{ + ok: true, + modelByID: make(map[int64]store.Model, len(models)), + modelIDByVar: make(map[int64]int64, len(variants)), + mmprojMissing: make(map[int64]bool, len(artifacts)), + } + for _, m := range models { + s.modelByID[m.ID] = m + } + for _, v := range variants { + s.modelIDByVar[v.ID] = v.ModelID + } + for _, a := range artifacts { + if a.ArtifactType == "mmproj" { + s.mmprojMissing[a.ID] = a.Missing + } + } + return s +} + +// offeringModalities returns the model-level architectural modalities for a +// remote offering. store.Offering has no modality column of its own — +// remote-provider vision can only come from the joined Model via o.ModelID; +// there is no mmproj narrowing for a remote provider (that's a local-config +// concept). nil when the snapshot is unusable or the model is unknown. +func (s modalitySnapshot) offeringModalities(o store.Offering) []string { + if !s.ok { + return nil + } + mdl, ok := s.modelByID[o.ModelID] + if !ok { + return nil + } + return append([]string{"text"}, nonTextModalities(mdl.Modalities)...) +} + +// configModalities returns what this specific Config can actually deliver, +// via store.ResolveModalities — the same precedence the PWA's ConfigCard +// shows (registry.resolveModalities), so a0 and the dashboard can never +// disagree. nil when the snapshot is unusable or the config's model can't +// be resolved. +func (s modalitySnapshot) configModalities(c store.Config) []string { + if !s.ok { + return nil + } + modelID, ok := s.modelIDByVar[c.VariantID] + if !ok { + return nil + } + mdl, ok := s.modelByID[modelID] + if !ok { + return nil + } + missing := s.mmprojMissing[c.MMProjArtifactID] // false for id 0 / unknown id, correctly + res := store.ResolveModalities(c, mdl, missing) + return res.Enabled +} + +// nonTextModalities drops "text" out of a modality list (offeringModalities +// re-adds it explicitly first, mirroring store.ResolveModalities' own +// convention of always leading with "text"). +func nonTextModalities(mods []string) []string { + out := make([]string, 0, len(mods)) + for _, m := range mods { + if m != "text" { + out = append(out, m) + } + } + return out +} + +// openCodeModalities maps the catalog's own vocabulary (text|vision|audio) +// to the input-modality names the OpenCode discovery plugin's "llama-swap" +// enricher actually recognizes (text|image|audio|video|pdf — verified +// against the plugin's real source, not just its docs). "vision" becomes +// "image"; anything unrecognized is dropped rather than passed through, so +// a future catalog modality can't silently leak an unrecognized token onto +// the wire. nil in -> nil out. +func openCodeModalities(storeMods []string) []string { + if storeMods == nil { + return nil + } + out := make([]string, 0, len(storeMods)) + for _, m := range storeMods { + switch m { + case "text", "audio": + out = append(out, m) + case "vision": + out = append(out, "image") + } + } + return out +} + +// architectureFor wraps openCodeModalities into the wire type. Output +// modalities are always ["text"] — nothing a0 routes emits image or audio +// on the chat-completions path. nil in -> nil out (propagates "unknown"). +func architectureFor(storeMods []string) *ModelArchitecture { + input := openCodeModalities(storeMods) + if input == nil { + return nil + } + return &ModelArchitecture{InputModalities: input, OutputModalities: []string{"text"}} +} + +// assignUniqueDisplayNames sets Name on each entry whose candidate name is +// both non-empty and unique across the whole listing, and leaves entries +// whose candidate collides (or is empty/unknown) at their zero value so the +// consumer falls back to formatting the id itself (D4 — see +// ModelEntry.Name). candidates is index-parallel to data. +func assignUniqueDisplayNames(data []ModelEntry, candidates []string) { + counts := make(map[string]int, len(candidates)) + for _, name := range candidates { + if name != "" { + counts[name]++ + } + } + for i, name := range candidates { + if name != "" && counts[name] == 1 { + data[i].Name = name + } + } } // BuildModelsResponse builds an OpenAI-shaped /v1/models payload — one entry @@ -254,13 +440,26 @@ type ModelEntry struct { // providers' wire names still route as aliases converging on the same // primary). providers nil (provider store unwired) skips the enablement // filter — pre-0032 behavior for skeleton mode. +// +// Modalities (2026-09-13): each entry's Architecture is single-sourced from +// store.ResolveModalities (the same rule the PWA's cards use), never +// computed twice. A read failure anywhere in loadModalitySnapshot means +// every entry ships with Architecture == nil ("unknown") rather than a0 +// actively asserting a vision-capable model is text-only — see +// modalitySnapshot and ModelEntry.Architecture. func BuildModelsResponse(ctx context.Context, storeCat store.Catalog, providers []store.ProviderRow) ModelsResponse { ctx, cancel := context.WithTimeout(ctx, 2*time.Second) defer cancel() data := make([]ModelEntry, 0) + var nameCandidates []string seen := map[string]bool{} + var modSnap modalitySnapshot + if storeCat != nil { + modSnap = loadModalitySnapshot(ctx, storeCat) + } + // Store-backed Offerings (MODEL CATALOG Phase 2). Selection (which // offering is each model group's primary) is shared with offeringChain // and the routing-preview endpoint — see select.go — so what's listed @@ -318,7 +517,9 @@ func BuildModelsResponse(ctx context.Context, storeCat store.Catalog, providers if o.ContextLength > 0 { entry.ContextLength = o.ContextLength } + entry.Architecture = architectureFor(modSnap.offeringModalities(o)) data = append(data, entry) + nameCandidates = append(nameCandidates, modSnap.modelByID[o.ModelID].Name) seen[o.WireModel] = true } } @@ -351,12 +552,16 @@ func BuildModelsResponse(ctx context.Context, storeCat store.Catalog, providers if c.NCtx > 0 { entry.ContextLength = c.NCtx } + entry.Architecture = architectureFor(modSnap.configModalities(c)) data = append(data, entry) + modelID := modSnap.modelIDByVar[c.VariantID] + nameCandidates = append(nameCandidates, modSnap.modelByID[modelID].Name) seen[c.Name] = true } } } + assignUniqueDisplayNames(data, nameCandidates) return ModelsResponse{Object: "list", Data: data} } diff --git a/go/internal/router/catalog_test.go b/go/internal/router/catalog_test.go index fc9fea6..b091bd6 100644 --- a/go/internal/router/catalog_test.go +++ b/go/internal/router/catalog_test.go @@ -5,8 +5,10 @@ package router import ( "context" "encoding/json" + "errors" "net/http" "net/http/httptest" + "strings" "sync/atomic" "testing" "time" @@ -270,15 +272,29 @@ func TestBuildModelsResponse_NilStoreCatalog(t *testing.T) { } } +// seedConfigOpts widens seedCatalogConfig for the modality tests below +// without touching any of its existing call sites (variadic, all fields +// optional/zero-valued by default). +type seedConfigOpts struct { + modelModalities []string // nil -> store default (["text"]) + withMMProj bool // create an mmproj artifact and link it + mmprojMissing bool // only meaningful if withMMProj + cfgModalities *[]string // nil -> derive; non-nil (incl. empty) -> explicit override +} + // seedCatalogConfig creates the minimal Model → Variant → Artifact(weight) → // Config chain needed to satisfy the configs table's FK constraints // (name, variant_id, weight_artifact_id, engine_id all NOT NULL), mirroring // TestCatalogFullRoundTrip in internal/store/catalog_test.go. Returns the new // config's ID. -func seedCatalogConfig(t *testing.T, cat store.Catalog, name string, nCtx int, visibility string) int64 { +func seedCatalogConfig(t *testing.T, cat store.Catalog, name string, nCtx int, visibility string, opts ...seedConfigOpts) int64 { t.Helper() + var o seedConfigOpts + if len(opts) > 0 { + o = opts[0] + } ctx := context.Background() - mdlID, err := cat.CreateModel(ctx, store.Model{Name: name}) + mdlID, err := cat.CreateModel(ctx, store.Model{Name: name, Modalities: o.modelModalities}) if err != nil { t.Fatalf("CreateModel: %v", err) } @@ -305,9 +321,20 @@ func seedCatalogConfig(t *testing.T, cat store.Catalog, name string, nCtx int, v if err != nil { t.Fatalf("EngineByName: %v", err) } + var mmprojID int64 + if o.withMMProj { + mmprojID, err = cat.CreateArtifact(ctx, store.Artifact{ + VariantID: varID, FormatID: f.ID, FilePath: name + "-mmproj.gguf", + IsAuxiliary: true, ArtifactType: "mmproj", Missing: o.mmprojMissing, + }) + if err != nil { + t.Fatalf("CreateArtifact (mmproj): %v", err) + } + } cfgID, err := cat.CreateConfig(ctx, store.Config{ Name: name, VariantID: varID, WeightArtifactID: weightID, - EngineID: eng.ID, NCtx: nCtx, Visibility: visibility, + EngineID: eng.ID, MMProjArtifactID: mmprojID, NCtx: nCtx, Visibility: visibility, + Modalities: o.cfgModalities, }) if err != nil { t.Fatalf("CreateConfig: %v", err) @@ -382,3 +409,315 @@ func TestBuildModelsResponse_CatalogConfigs(t *testing.T) { t.Errorf("dedup 'gemma4-31b' owned_by: got %q, want 'deepseek' (offering wins)", got) } } + +// ── /v1/models modalities (2026-09-13) ──────────────────────────────────── + +func TestBuildModelsResponse_ConfigVisionModalities(t *testing.T) { + db, err := store.Open(":memory:") + if err != nil { + t.Fatal(err) + } + defer db.Close() + cat := db.Catalog() + + seedCatalogConfig(t, cat, "gemma4-vision", 262144, "visible", seedConfigOpts{ + modelModalities: []string{"text", "vision"}, + withMMProj: true, + }) + + resp := BuildModelsResponse(context.Background(), cat, nil) + e := mustFindEntry(t, resp, "gemma4-vision") + if e.Architecture == nil { + t.Fatal("Architecture is nil, want a vision-capable entry") + } + if !stringSlicesEqualCatalog(e.Architecture.InputModalities, []string{"text", "image"}) { + t.Errorf("InputModalities = %v, want [text image]", e.Architecture.InputModalities) + } + if !stringSlicesEqualCatalog(e.Architecture.OutputModalities, []string{"text"}) { + t.Errorf("OutputModalities = %v, want [text]", e.Architecture.OutputModalities) + } +} + +func TestBuildModelsResponse_ConfigNoMMProjIsTextOnly(t *testing.T) { + db, err := store.Open(":memory:") + if err != nil { + t.Fatal(err) + } + defer db.Close() + cat := db.Catalog() + + seedCatalogConfig(t, cat, "vision-model-no-mmproj", 8192, "visible", seedConfigOpts{ + modelModalities: []string{"text", "vision"}, + withMMProj: false, + }) + + resp := BuildModelsResponse(context.Background(), cat, nil) + e := mustFindEntry(t, resp, "vision-model-no-mmproj") + if e.Architecture == nil { + t.Fatal("Architecture is nil, want an explicit text-only entry") + } + if !stringSlicesEqualCatalog(e.Architecture.InputModalities, []string{"text"}) { + t.Errorf("InputModalities = %v, want [text] (no mmproj linked)", e.Architecture.InputModalities) + } +} + +func TestBuildModelsResponse_ConfigMMProjMissingIsTextOnly(t *testing.T) { + db, err := store.Open(":memory:") + if err != nil { + t.Fatal(err) + } + defer db.Close() + cat := db.Catalog() + + seedCatalogConfig(t, cat, "vision-model-missing-mmproj", 8192, "visible", seedConfigOpts{ + modelModalities: []string{"text", "vision"}, + withMMProj: true, + mmprojMissing: true, + }) + + resp := BuildModelsResponse(context.Background(), cat, nil) + e := mustFindEntry(t, resp, "vision-model-missing-mmproj") + if e.Architecture == nil || !stringSlicesEqualCatalog(e.Architecture.InputModalities, []string{"text"}) { + t.Errorf("Architecture = %+v, want text-only (mmproj file missing on disk)", e.Architecture) + } +} + +func TestBuildModelsResponse_ConfigExplicitOverrideWins(t *testing.T) { + db, err := store.Open(":memory:") + if err != nil { + t.Fatal(err) + } + defer db.Close() + cat := db.Catalog() + + visionOverride := []string{"text", "vision"} + seedCatalogConfig(t, cat, "override-forces-vision", 8192, "visible", seedConfigOpts{ + modelModalities: nil, // text-only model + withMMProj: false, + cfgModalities: &visionOverride, + }) + + emptyOverride := []string{} + seedCatalogConfig(t, cat, "override-forces-text-only", 8192, "visible", seedConfigOpts{ + modelModalities: []string{"text", "vision"}, + withMMProj: true, + cfgModalities: &emptyOverride, + }) + + resp := BuildModelsResponse(context.Background(), cat, nil) + + visEntry := mustFindEntry(t, resp, "override-forces-vision") + if visEntry.Architecture == nil || !stringSlicesEqualCatalog(visEntry.Architecture.InputModalities, []string{"text", "image"}) { + t.Errorf("override-forces-vision Architecture = %+v, want [text image]", visEntry.Architecture) + } + + textEntry := mustFindEntry(t, resp, "override-forces-text-only") + if textEntry.Architecture == nil || !stringSlicesEqualCatalog(textEntry.Architecture.InputModalities, []string{"text"}) { + t.Errorf("override-forces-text-only Architecture = %+v, want [text] (empty override still wins)", textEntry.Architecture) + } +} + +func TestBuildModelsResponse_OfferingModalitiesFromModel(t *testing.T) { + db, err := store.Open(":memory:") + if err != nil { + t.Fatal(err) + } + defer db.Close() + ctx := context.Background() + cat := db.Catalog() + + db.Routing().SaveProvider(ctx, store.ProviderRow{ + Name: "deepseek", APIKey: "sk-test", Enabled: true, CreatedAt: time.Now(), + }) + deepseek, _, _ := db.Routing().ProviderByName(ctx, "deepseek") + + visionMdl, _ := cat.CreateModel(ctx, store.Model{Name: "deepseek-vision-model", Modalities: []string{"text", "vision"}}) + cat.CreateOffering(ctx, store.Offering{ + ModelID: visionMdl, ProviderID: deepseek.ID, + WireModel: "deepseek-vision-offering", Enabled: true, + }) + + textMdl, _ := cat.CreateModel(ctx, store.Model{Name: "deepseek-text-model"}) + cat.CreateOffering(ctx, store.Offering{ + ModelID: textMdl, ProviderID: deepseek.ID, + WireModel: "deepseek-text-offering", Enabled: true, + }) + + rows, _ := db.Routing().Providers(ctx) + resp := BuildModelsResponse(ctx, cat, rows) + + vis := mustFindEntry(t, resp, "deepseek-vision-offering") + if vis.Architecture == nil || !stringSlicesEqualCatalog(vis.Architecture.InputModalities, []string{"text", "image"}) { + t.Errorf("deepseek-vision-offering Architecture = %+v, want [text image]", vis.Architecture) + } + + txt := mustFindEntry(t, resp, "deepseek-text-offering") + if txt.Architecture == nil || !stringSlicesEqualCatalog(txt.Architecture.InputModalities, []string{"text"}) { + t.Errorf("deepseek-text-offering Architecture = %+v, want [text]", txt.Architecture) + } +} + +// TestBuildModelsResponse_VisionIsNeverOnTheWire is the single most +// important test in this group: the OpenCode discovery plugin's +// "llama-swap" modality parser only recognizes text|image|audio|video|pdf +// and silently drops anything else. If a0 ever regresses to emitting the +// catalog's native "vision" token instead of mapping it to "image", this +// test is the only thing that catches it — the JSON would still be +// perfectly valid, just silently ignored downstream. +func TestBuildModelsResponse_VisionIsNeverOnTheWire(t *testing.T) { + db, err := store.Open(":memory:") + if err != nil { + t.Fatal(err) + } + defer db.Close() + cat := db.Catalog() + + // Deliberately avoid "vision" anywhere in the config/model NAME itself + // (the id/name are also in the JSON) so the only possible source of the + // literal string "vision" in the response is an un-mapped modality token. + seedCatalogConfig(t, cat, "attach-wire-check", 8192, "visible", seedConfigOpts{ + modelModalities: []string{"text", "vision"}, + withMMProj: true, + }) + + resp := BuildModelsResponse(context.Background(), cat, nil) + body, err := json.Marshal(resp) + if err != nil { + t.Fatalf("marshal: %v", err) + } + if strings.Contains(string(body), "vision") { + t.Errorf("response JSON contains the literal string \"vision\" — must be mapped to \"image\": %s", body) + } + if !strings.Contains(string(body), "image") { + t.Errorf("response JSON never contains \"image\" for a vision-capable model: %s", body) + } +} + +func TestBuildModelsResponse_DisplayNameUniqueOnly(t *testing.T) { + db, err := store.Open(":memory:") + if err != nil { + t.Fatal(err) + } + defer db.Close() + ctx := context.Background() + cat := db.Catalog() + + // Two Configs sharing one Model (mirrors gemma4-26b-a4b / + // gemma4-26b-a4b-nothink both pointing at "Gemma 4 26B A4B (MTP)") — + // both must come back with an empty Name (D4), never a duplicated one. + mdlID, err := cat.CreateModel(ctx, store.Model{Name: "Shared Model"}) + if err != nil { + t.Fatalf("CreateModel: %v", err) + } + seedSiblingConfig(t, cat, mdlID, "shared-config-a", "visible") + seedSiblingConfig(t, cat, mdlID, "shared-config-b", "visible") + + // A third config with its own, unshared Model must get its Name. + seedCatalogConfig(t, cat, "solo-config", 8192, "visible") + + resp := BuildModelsResponse(ctx, cat, nil) + + if got := mustFindEntry(t, resp, "shared-config-a").Name; got != "" { + t.Errorf("shared-config-a Name = %q, want empty (ambiguous)", got) + } + if got := mustFindEntry(t, resp, "shared-config-b").Name; got != "" { + t.Errorf("shared-config-b Name = %q, want empty (ambiguous)", got) + } + if got := mustFindEntry(t, resp, "solo-config").Name; got != "solo-config" { + t.Errorf("solo-config Name = %q, want %q (unique)", got, "solo-config") + } +} + +// seedSiblingConfig creates a second Config pointed at an EXISTING model — +// the two-configs-one-model shape assignUniqueDisplayNames needs to be +// tested against, which seedCatalogConfig (one model per config) can't +// produce on its own. +func seedSiblingConfig(t *testing.T, cat store.Catalog, modelID int64, name, visibility string) int64 { + t.Helper() + ctx := context.Background() + varID, err := cat.CreateVariant(ctx, store.Variant{ModelID: modelID, Name: name}) + if err != nil { + t.Fatalf("CreateVariant: %v", err) + } + f, err := cat.FormatByName(ctx, "GGUF") + if err != nil { + t.Fatalf("FormatByName: %v", err) + } + weightID, err := cat.CreateArtifact(ctx, store.Artifact{ + VariantID: varID, FormatID: f.ID, FilePath: name + ".gguf", ArtifactType: "weight", + }) + if err != nil { + t.Fatalf("CreateArtifact: %v", err) + } + eng, err := cat.EngineByName(ctx, "llama.cpp") + if err != nil { + t.Fatalf("EngineByName: %v", err) + } + cfgID, err := cat.CreateConfig(ctx, store.Config{ + Name: name, VariantID: varID, WeightArtifactID: weightID, + EngineID: eng.ID, Visibility: visibility, + }) + if err != nil { + t.Fatalf("CreateConfig: %v", err) + } + return cfgID +} + +// errListModelsCatalog wraps a real store.Catalog and injects a ListModels +// failure — everything else passes through untouched via embedding, so no +// method list needs duplicating as the interface grows. +type errListModelsCatalog struct{ store.Catalog } + +func (errListModelsCatalog) ListModels(context.Context) ([]store.Model, error) { + return nil, errors.New("boom") +} + +// TestBuildModelsResponse_ModalityReadErrorOmitsArchitecture is the D5 +// guard: a catalog read failure must make a0 omit Architecture entirely +// (unknown), never assert text-only — that would silently disable real +// vision during a transient DB hiccup. +func TestBuildModelsResponse_ModalityReadErrorOmitsArchitecture(t *testing.T) { + db, err := store.Open(":memory:") + if err != nil { + t.Fatal(err) + } + defer db.Close() + cat := db.Catalog() + + seedCatalogConfig(t, cat, "vision-during-outage", 8192, "visible", seedConfigOpts{ + modelModalities: []string{"text", "vision"}, + withMMProj: true, + }) + + resp := BuildModelsResponse(context.Background(), errListModelsCatalog{cat}, nil) + e := mustFindEntry(t, resp, "vision-during-outage") + if e.Architecture != nil { + t.Errorf("Architecture = %+v, want nil (unknown) during a catalog read failure", e.Architecture) + } + if e.ContextLength != 8192 { + t.Errorf("ContextLength = %d, want 8192 — listing itself must survive a modality-read failure", e.ContextLength) + } +} + +func mustFindEntry(t *testing.T, resp ModelsResponse, id string) ModelEntry { + t.Helper() + for _, e := range resp.Data { + if e.ID == id { + return e + } + } + t.Fatalf("entry %q not found in /v1/models response: %+v", id, resp.Data) + return ModelEntry{} +} + +func stringSlicesEqualCatalog(a, b []string) bool { + if len(a) != len(b) { + return false + } + for i := range a { + if a[i] != b[i] { + return false + } + } + return true +} diff --git a/go/internal/router/router_test.go b/go/internal/router/router_test.go index 771dfe9..5871400 100644 --- a/go/internal/router/router_test.go +++ b/go/internal/router/router_test.go @@ -190,7 +190,10 @@ func TestModels_List(t *testing.T) { t.Fatal(err) } defer db.Close() - seedCatalogConfig(t, db.Catalog(), "gemma4-26b-mtp", 8192, "visible") + seedCatalogConfig(t, db.Catalog(), "gemma4-26b-mtp", 8192, "visible", seedConfigOpts{ + modelModalities: []string{"text", "vision"}, + withMMProj: true, + }) srv := NewWithDeps(Deps{Cfg: testCfg(nil, nil), StoreCatalog: db.Catalog(), Auth: &stubAuth{validToken: "x"}}) @@ -216,6 +219,13 @@ func TestModels_List(t *testing.T) { if resp.Data[0].ContextLength != 8192 { t.Errorf("context_length = %d, want 8192", resp.Data[0].ContextLength) } + // End-to-end through the real HTTP handler (not just BuildModelsResponse + // directly, per catalog_test.go's coverage): a vision-wired config's + // JSON carries the OpenCode-plugin-compatible architecture block. + arch := resp.Data[0].Architecture + if arch == nil || len(arch.InputModalities) != 2 || arch.InputModalities[0] != "text" || arch.InputModalities[1] != "image" { + t.Errorf("architecture.input_modalities = %+v, want [text image]", arch) + } } func TestModels_UnloadedConfigStillListed(t *testing.T) { diff --git a/go/internal/store/db_test.go b/go/internal/store/db_test.go index 01abb18..fcb0273 100644 --- a/go/internal/store/db_test.go +++ b/go/internal/store/db_test.go @@ -19,8 +19,8 @@ func TestMigrateFresh(t *testing.T) { ).Scan(&version); err != nil { t.Fatalf("read version: %v", err) } - if version != 78 { - t.Fatalf("schema version = %d, want 78", version) + if version != 81 { + t.Fatalf("schema version = %d, want 81", version) } // Every Contract 3 table (0001) plus the Sprint 0 §0.11 polish tables diff --git a/go/internal/store/migration_0079_test.go b/go/internal/store/migration_0079_test.go new file mode 100644 index 0000000..7fbc5dc --- /dev/null +++ b/go/internal/store/migration_0079_test.go @@ -0,0 +1,193 @@ +// SPDX-License-Identifier: Apache-2.0 + +package store + +import ( + "database/sql" + "testing" +) + +func apply0079(t *testing.T, sqlDB *sql.DB) { + t.Helper() + body, err := migrationsFS.ReadFile("migrations/0079_a0_vision_modalities.sql") + if err != nil { + t.Fatalf("read 0079: %v", err) + } + if _, err := sqlDB.Exec(string(body)); err != nil { + t.Fatalf("apply 0079: %v", err) + } +} + +func TestMigration0079_DeepSeekFlashGetsVision(t *testing.T) { + sqlDB := openThrough(t, 78) + defer sqlDB.Close() + + if _, err := sqlDB.Exec(`INSERT INTO router_providers (name, api_key, created_at) VALUES ('deepseek', 'sk-x', 0)`); err != nil { + t.Fatalf("seed router_providers: %v", err) + } + var providerID int64 + if err := sqlDB.QueryRow(`SELECT id FROM router_providers WHERE name = 'deepseek'`).Scan(&providerID); err != nil { + t.Fatalf("resolve provider id: %v", err) + } + + if _, err := sqlDB.Exec(`INSERT INTO models (family_id, name) VALUES (NULL, 'DeepSeek V4.1 Flash')`); err != nil { + t.Fatalf("seed models: %v", err) + } + var modelID int64 + if err := sqlDB.QueryRow(`SELECT id FROM models WHERE name = 'DeepSeek V4.1 Flash'`).Scan(&modelID); err != nil { + t.Fatalf("resolve model id: %v", err) + } + + // A second, unrelated model+offering must be left untouched. + if _, err := sqlDB.Exec(`INSERT INTO models (family_id, name) VALUES (NULL, 'DeepSeek V4 Pro')`); err != nil { + t.Fatalf("seed second model: %v", err) + } + var otherModelID int64 + if err := sqlDB.QueryRow(`SELECT id FROM models WHERE name = 'DeepSeek V4 Pro'`).Scan(&otherModelID); err != nil { + t.Fatalf("resolve other model id: %v", err) + } + + if _, err := sqlDB.Exec( + `INSERT INTO offerings (model_id, provider_id, wire_model, price_in_per_1m, price_out_per_1m, currency, enabled, priority) + VALUES (?, ?, 'deepseek-flash', 0.14, 0.28, 'USD', 1, 100)`, + modelID, providerID, + ); err != nil { + t.Fatalf("seed offerings: %v", err) + } + if _, err := sqlDB.Exec( + `INSERT INTO offerings (model_id, provider_id, wire_model, price_in_per_1m, price_out_per_1m, currency, enabled, priority) + VALUES (?, ?, 'deepseek-v4-pro', 0.5, 1.5, 'USD', 1, 100)`, + otherModelID, providerID, + ); err != nil { + t.Fatalf("seed second offering: %v", err) + } + + apply0079(t, sqlDB) + + var mods string + if err := sqlDB.QueryRow(`SELECT modalities FROM models WHERE id = ?`, modelID).Scan(&mods); err != nil { + t.Fatalf("read model modalities: %v", err) + } + if mods != `["text","vision"]` { + t.Errorf("deepseek-flash model modalities = %q, want [\"text\",\"vision\"]", mods) + } + + var otherMods string + if err := sqlDB.QueryRow(`SELECT modalities FROM models WHERE id = ?`, otherModelID).Scan(&otherMods); err != nil { + t.Fatalf("read other model modalities: %v", err) + } + if otherMods != `["text"]` { + t.Errorf("unrelated deepseek-v4-pro model modalities changed to %q, want unchanged [\"text\"]", otherMods) + } + + // Idempotence: a second application changes nothing further (row is no + // longer '["text"]', so the guard no-ops). + apply0079(t, sqlDB) + var modsAgain string + if err := sqlDB.QueryRow(`SELECT modalities FROM models WHERE id = ?`, modelID).Scan(&modsAgain); err != nil { + t.Fatalf("read model modalities after re-apply: %v", err) + } + if modsAgain != `["text","vision"]` { + t.Errorf("modalities changed on re-apply: %q", modsAgain) + } +} + +func TestMigration0079_Qwen38FlashNextGetsOverride(t *testing.T) { + sqlDB := openThrough(t, 78) + defer sqlDB.Close() + + if _, err := sqlDB.Exec(`PRAGMA foreign_keys=OFF`); err != nil { + t.Fatalf("pragma off: %v", err) + } + if _, err := sqlDB.Exec( + `INSERT INTO configs (name, variant_id, weight_artifact_id, engine_id, extra_args) + VALUES ('qwen38-flash-next', 1, 1, 1, '["--no-mmap","--mmproj","/opt/forge/models/qwen3.8-flash-next/mmproj/mmproj-Qwen3.8-Flash-Next-f16.gguf"]')`, + ); err != nil { + t.Fatalf("seed config: %v", err) + } + if _, err := sqlDB.Exec(`PRAGMA foreign_keys=ON`); err != nil { + t.Fatalf("pragma on: %v", err) + } + + apply0079(t, sqlDB) + + var mods any + if err := sqlDB.QueryRow(`SELECT modalities FROM configs WHERE name = 'qwen38-flash-next'`).Scan(&mods); err != nil { + t.Fatalf("read config modalities: %v", err) + } + modsStr, ok := mods.(string) + if !ok || modsStr != `["text","vision"]` { + t.Errorf("qwen38-flash-next config modalities = %v, want [\"text\",\"vision\"]", mods) + } + + // Idempotence. + apply0079(t, sqlDB) + var modsAgain string + if err := sqlDB.QueryRow(`SELECT modalities FROM configs WHERE name = 'qwen38-flash-next'`).Scan(&modsAgain); err != nil { + t.Fatalf("read config modalities after re-apply: %v", err) + } + if modsAgain != `["text","vision"]` { + t.Errorf("modalities changed on re-apply: %q", modsAgain) + } +} + +func TestMigration0079_NoMMProjFlagIsNoOp(t *testing.T) { + sqlDB := openThrough(t, 78) + defer sqlDB.Close() + + if _, err := sqlDB.Exec(`PRAGMA foreign_keys=OFF`); err != nil { + t.Fatalf("pragma off: %v", err) + } + // Same name, but no --mmproj in extra_args -- the EXISTS guard must + // leave this alone (this config genuinely has no wired vision). + if _, err := sqlDB.Exec( + `INSERT INTO configs (name, variant_id, weight_artifact_id, engine_id, extra_args) + VALUES ('qwen38-flash-next', 1, 1, 1, '["--no-mmap","--jinja"]')`, + ); err != nil { + t.Fatalf("seed config: %v", err) + } + if _, err := sqlDB.Exec(`PRAGMA foreign_keys=ON`); err != nil { + t.Fatalf("pragma on: %v", err) + } + + apply0079(t, sqlDB) + + var mods any + if err := sqlDB.QueryRow(`SELECT modalities FROM configs WHERE name = 'qwen38-flash-next'`).Scan(&mods); err != nil { + t.Fatalf("read config modalities: %v", err) + } + if mods != nil { + t.Errorf("config with no --mmproj flag got modalities = %v, want NULL (no-op)", mods) + } +} + +func TestMigration0079_ExistingOverrideNotClobbered(t *testing.T) { + sqlDB := openThrough(t, 78) + defer sqlDB.Close() + + if _, err := sqlDB.Exec(`PRAGMA foreign_keys=OFF`); err != nil { + t.Fatalf("pragma off: %v", err) + } + // Has --mmproj AND an operator already set an explicit override (e.g. + // asserting text-only despite the flag, for whatever reason) -- the IS + // NULL guard must never clobber that. + if _, err := sqlDB.Exec( + `INSERT INTO configs (name, variant_id, weight_artifact_id, engine_id, extra_args, modalities) + VALUES ('qwen38-flash-next', 1, 1, 1, '["--mmproj","/x.gguf"]', '["text"]')`, + ); err != nil { + t.Fatalf("seed config: %v", err) + } + if _, err := sqlDB.Exec(`PRAGMA foreign_keys=ON`); err != nil { + t.Fatalf("pragma on: %v", err) + } + + apply0079(t, sqlDB) + + var mods string + if err := sqlDB.QueryRow(`SELECT modalities FROM configs WHERE name = 'qwen38-flash-next'`).Scan(&mods); err != nil { + t.Fatalf("read config modalities: %v", err) + } + if mods != `["text"]` { + t.Errorf("operator override clobbered: modalities = %q, want unchanged [\"text\"]", mods) + } +} diff --git a/go/internal/store/migration_0080_test.go b/go/internal/store/migration_0080_test.go new file mode 100644 index 0000000..e9e56c4 --- /dev/null +++ b/go/internal/store/migration_0080_test.go @@ -0,0 +1,68 @@ +// SPDX-License-Identifier: Apache-2.0 + +package store + +import ( + "database/sql" + "testing" +) + +// TestMigration0080_RestoresRegressedMTPVision covers the real gap found +// live while verifying 0079's deploy: two of 0028's three named models +// ('Gemma 4 26B A4B (MTP)', 'Qwen3.6 35B MTP') had silently regressed back +// to modalities = '[]' by some later catalog edit; the third ('Qwen3.6 35B +// (Aggressive)') had not. 0080 restores the first two and must never touch +// an unrelated model, a model already correct, or a model that's genuinely +// been set to '[]' deliberately by an operator after this migration ships. +func TestMigration0080_RestoresRegressedMTPVision(t *testing.T) { + sqlDB := openThrough(t, 79) + defer sqlDB.Close() + + seed := func(name, modalities string) int64 { + res, err := sqlDB.Exec(`INSERT INTO models (family_id, name, modalities) VALUES (NULL, ?, ?)`, name, modalities) + if err != nil { + t.Fatalf("seed model %q: %v", name, err) + } + id, _ := res.LastInsertId() + return id + } + + regressedGemma := seed("Gemma 4 26B A4B (MTP)", `[]`) + regressedQwenMTP := seed("Qwen3.6 35B MTP", `[]`) + alreadyCorrect := seed("Qwen3.6 35B (Aggressive)", `["text","vision"]`) + unrelated := seed("Some Other Model", `[]`) + + apply0080(t, sqlDB) + + assertModalities := func(id int64, want string) { + t.Helper() + var got string + if err := sqlDB.QueryRow(`SELECT modalities FROM models WHERE id = ?`, id).Scan(&got); err != nil { + t.Fatalf("read model %d: %v", id, err) + } + if got != want { + t.Errorf("model %d modalities = %q, want %q", id, got, want) + } + } + + assertModalities(regressedGemma, `["text","vision"]`) + assertModalities(regressedQwenMTP, `["text","vision"]`) + assertModalities(alreadyCorrect, `["text","vision"]`) // untouched, already correct + assertModalities(unrelated, `[]`) // untouched, not a named target + + // Idempotence. + apply0080(t, sqlDB) + assertModalities(regressedGemma, `["text","vision"]`) + assertModalities(regressedQwenMTP, `["text","vision"]`) +} + +func apply0080(t *testing.T, sqlDB *sql.DB) { + t.Helper() + body, err := migrationsFS.ReadFile("migrations/0080_restore_mtp_vision_modalities.sql") + if err != nil { + t.Fatalf("read 0080: %v", err) + } + if _, err := sqlDB.Exec(string(body)); err != nil { + t.Fatalf("apply 0080: %v", err) + } +} diff --git a/go/internal/store/migrations/0079_a0_vision_modalities.sql b/go/internal/store/migrations/0079_a0_vision_modalities.sql new file mode 100644 index 0000000..ad9d312 --- /dev/null +++ b/go/internal/store/migrations/0079_a0_vision_modalities.sql @@ -0,0 +1,44 @@ +-- SPDX-License-Identifier: Apache-2.0 +-- Schema v79 (a0 model-discovery sprint, 2026-09-13). Two real modality +-- facts the catalog was never told, both now load-bearing: a0's /v1/models +-- is becoming the single source of truth for OpenCode's model capabilities +-- (see BuildModelsResponse/store.ResolveModalities), so a model whose +-- vision wiring is invisible to that resolution rule is now advertised to +-- every consumer as text-only, not just displayed inconsistently on one +-- page as before. +-- +-- Both UPDATEs follow the guarded, no-op-safe idiom already used by +-- 0028_modalities.sql / 0020_gemma_logo.sql — a no-op on any DB shaped +-- differently than the one this was written against. + +-- (1) DeepSeek V4.1-Flash (wire model deepseek-flash) is natively +-- multimodal, but its `models` row still carries the '["text"]' default +-- from 0028, which only ever set vision on three LOCAL models. `offerings` +-- has no modality column of its own (see store.Offering) -- the joined +-- `models` row via offerings.model_id is the ONLY source a0 has for a +-- remote provider's vision support. Keyed through offerings.wire_model +-- rather than models.name so a display-name edit can't make this guard +-- miss. +UPDATE models SET modalities = '["text","vision"]' +WHERE id IN (SELECT model_id FROM offerings WHERE wire_model = 'deepseek-flash') + AND modalities = '["text"]'; + +-- (2) qwen38-flash-next's mmproj was wired 2026-09-12 as a raw +-- `--mmproj ` appended to configs.extra_args, not via +-- mmproj_artifact_id (no artifact-creation HTTP route exists yet -- see +-- progress.md's 2026-09-12 entry). Nothing parses --mmproj out of +-- extra_args anywhere in Go, so store.ResolveModalities sees +-- MMProjArtifactID == 0 and correctly-by-its-own-rule-but-wrongly-in-fact +-- concludes "text only". An explicit configs.modalities override is the +-- modelled way to say otherwise. The EXISTS guard makes this a no-op on +-- any DB where that flag isn't actually present; the IS NULL guard never +-- clobbers an operator's own override. +-- +-- Remove this override if/when a real mmproj artifact row is registered +-- for this config (i.e. once store.CreateArtifact is wired to httpapi and +-- someone points mmproj_artifact_id at it instead) -- at that point the +-- derive rule owns the fact correctly on its own again. +UPDATE configs SET modalities = '["text","vision"]' +WHERE name = 'qwen38-flash-next' + AND modalities IS NULL + AND EXISTS (SELECT 1 FROM json_each(configs.extra_args) WHERE value = '--mmproj'); diff --git a/go/internal/store/migrations/0080_restore_mtp_vision_modalities.sql b/go/internal/store/migrations/0080_restore_mtp_vision_modalities.sql new file mode 100644 index 0000000..b76d46a --- /dev/null +++ b/go/internal/store/migrations/0080_restore_mtp_vision_modalities.sql @@ -0,0 +1,24 @@ +-- SPDX-License-Identifier: Apache-2.0 +-- Schema v80 (a0 model-discovery sprint, 2026-09-13, found live while +-- verifying 0079's deploy). `0028_modalities.sql` set +-- modalities = '["text","vision"]' on three named models: 'Gemma 4 26B A4B +-- (MTP)', 'Qwen3.6 35B (Aggressive)', and 'Qwen3.6 35B MTP'. A live audit +-- of the real ForgeHost catalog today (cross-referencing every visible config +-- with a real, non-missing mmproj against its model's stored modalities) +-- found only the middle one still carries it — 'Gemma 4 26B A4B (MTP)' and +-- 'Qwen3.6 35B MTP' are both back to modalities = '[]', silently regressed +-- by some later catalog edit between 0028 (2026-08-05) and now (root cause +-- not chased down — likely a model-editing/family-editing form save that +-- doesn't round-trip a column it doesn't render, the same class of bug +-- Sprint D found for capability scores). This is why gemma4-26b-a4b, +-- gemma4-26b-a4b-nothink, and qwen36-35b-a3b were still advertising +-- text-only on /v1/models even after 0079 shipped the resolution-rule +-- wiring — the code was correctly reading genuinely wrong data. +-- +-- Guarded the same way as 0028/0079: a no-op wherever these exact names +-- don't exist or already carry a non-empty modalities value (never +-- overwrites an operator's own deliberate '[]' — if one exists, whoever +-- set it should be asked, not silently overridden a second time). +UPDATE models SET modalities = '["text","vision"]' +WHERE name IN ('Gemma 4 26B A4B (MTP)', 'Qwen3.6 35B MTP') + AND modalities = '[]'; diff --git a/go/internal/store/migrations/0081_qwen_genealogy_logo.sql b/go/internal/store/migrations/0081_qwen_genealogy_logo.sql new file mode 100644 index 0000000..f53fc3e --- /dev/null +++ b/go/internal/store/migrations/0081_qwen_genealogy_logo.sql @@ -0,0 +1,17 @@ +-- SPDX-License-Identifier: Apache-2.0 +-- Schema v81 (operator feedback, 2026-09-13). Data fix, not a schema +-- change: the "Qwen" genealogy's uploaded logo is Alibaba's corporate mark, +-- not Qwen's own — every Qwen 3.5+ model inherits it via +-- registry.resolveLogos (0040_icon_inheritance_takeover.sql's model-level +-- clear left these models with logo='', correctly deferring to the parent, +-- but the parent mark itself was wrong). web/src/assets/icons/manifest.ts +-- already vendors a real, distinct "qwen" brand mark (used by +-- creatorIcon.ts's CREATOR_ALIASES for any model whose bare `creator` is +-- "qwen") — this points the genealogy at that same slug instead of the +-- uploaded image, matching the DeepSeek/Gemma/Kimi/Ornith genealogies, +-- which already reference manifest slugs rather than raw uploads. +-- Guarded on the row still holding an uploaded (data URI) image, so this is +-- a no-op against a DB where the logo was never wrong, already fixed, or +-- deliberately set to something else. +UPDATE genealogies SET logo = 'qwen', logo_dark = '' +WHERE name = 'Qwen' AND logo LIKE 'data:%'; diff --git a/go/internal/store/modalities.go b/go/internal/store/modalities.go new file mode 100644 index 0000000..444aeff --- /dev/null +++ b/go/internal/store/modalities.go @@ -0,0 +1,69 @@ +// SPDX-License-Identifier: Apache-2.0 +package store + +// Modality-gap reasons (Sprint J1 vocabulary, verbatim — these strings are +// wire-facing via registry.ModalityGap.Reason). +const ( + ModalityReasonNoMMProj = "no mmproj linked" + ModalityReasonMMProjMissing = "mmproj file missing on disk" +) + +// ModalityResolution is the outcome of narrowing one Model's architectural +// modalities to what one specific Config can actually deliver. +type ModalityResolution struct { + Enabled []string // always begins with "text"; never nil + Unavailable []string // model modalities this config can't deliver; nil when none + Reason string // "" iff Unavailable is empty +} + +// ResolveModalities is the single source of truth for the Sprint J1 +// precedence (moved here from registry.resolveModalities so a0's +// /v1/models and the PWA's cards can never disagree): +// +// 1. "text" is always enabled. +// 2. An explicit c.Modalities override wins verbatim, even an empty one +// (an operator asserting "text only" despite a capable model/mmproj). +// 3. MMProjArtifactID == 0 → text only; every other model-level modality +// is unavailable, reason ModalityReasonNoMMProj. +// 4. mmprojMissing → text only, reason ModalityReasonMMProjMissing. +// 5. Otherwise inherit mdl.Modalities. +// +// mmprojMissing means "the linked mmproj artifact row exists AND is +// flagged Missing". An unknown/zero artifact id must be passed as false, +// which preserves the pre-extraction `ok && a.Missing` behavior exactly. +func ResolveModalities(c Config, mdl Model, mmprojMissing bool) ModalityResolution { + nonText := func(mods []string) []string { + out := make([]string, 0, len(mods)) + for _, m := range mods { + if m != "text" { + out = append(out, m) + } + } + return out + } + + if c.Modalities != nil { + return ModalityResolution{Enabled: append([]string{"text"}, nonText(*c.Modalities)...)} + } + + if c.MMProjArtifactID == 0 { + return newTextOnlyResolution(nonText(mdl.Modalities), ModalityReasonNoMMProj) + } + + if mmprojMissing { + return newTextOnlyResolution(nonText(mdl.Modalities), ModalityReasonMMProjMissing) + } + + return ModalityResolution{Enabled: append([]string{"text"}, nonText(mdl.Modalities)...)} +} + +// newTextOnlyResolution builds the "text only" result for the no-mmproj / +// mmproj-missing branches. Reason is only set when there's an actual gap to +// report — a model with no non-text modalities to begin with has nothing +// unavailable, so Unavailable/Reason both stay zero-valued. +func newTextOnlyResolution(gaps []string, reason string) ModalityResolution { + if len(gaps) == 0 { + return ModalityResolution{Enabled: []string{"text"}} + } + return ModalityResolution{Enabled: []string{"text"}, Unavailable: gaps, Reason: reason} +} diff --git a/go/internal/store/modalities_test.go b/go/internal/store/modalities_test.go new file mode 100644 index 0000000..268bde5 --- /dev/null +++ b/go/internal/store/modalities_test.go @@ -0,0 +1,103 @@ +// SPDX-License-Identifier: Apache-2.0 + +package store_test + +import ( + "reflect" + "testing" + + "github.com/jsaigou/the-forge/internal/store" +) + +func TestResolveModalities(t *testing.T) { + visionModel := store.Model{Modalities: []string{"text", "vision"}} + textOnlyModel := store.Model{Modalities: []string{"text"}} + emptyModalitiesModel := store.Model{Modalities: nil} + + cases := []struct { + name string + cfg store.Config + mdl store.Model + mmprojMissing bool + wantEnabled []string + wantUnavail []string + wantReason string + }{ + { + name: "explicit override wins verbatim", + cfg: store.Config{MMProjArtifactID: 0, Modalities: ptr([]string{"text", "vision"})}, + mdl: textOnlyModel, // override should win even though the model itself is text-only + wantEnabled: []string{"text", "vision"}, + }, + { + name: "explicit EMPTY override forces text-only", + cfg: store.Config{MMProjArtifactID: 42, Modalities: ptr([]string{})}, + mdl: visionModel, // override should win even though model+mmproj would otherwise allow vision + wantEnabled: []string{"text"}, + }, + { + name: "no mmproj linked is text-only with a gap", + cfg: store.Config{MMProjArtifactID: 0, Modalities: nil}, + mdl: visionModel, + wantEnabled: []string{"text"}, + wantUnavail: []string{"vision"}, + wantReason: store.ModalityReasonNoMMProj, + }, + { + name: "no mmproj linked, text-only model has no gap to report", + cfg: store.Config{MMProjArtifactID: 0, Modalities: nil}, + mdl: textOnlyModel, + wantEnabled: []string{"text"}, + }, + { + name: "mmproj missing on disk is text-only with a gap", + cfg: store.Config{MMProjArtifactID: 7, Modalities: nil}, + mdl: visionModel, + mmprojMissing: true, + wantEnabled: []string{"text"}, + wantUnavail: []string{"vision"}, + wantReason: store.ModalityReasonMMProjMissing, + }, + { + name: "mmproj linked and present inherits model modalities", + cfg: store.Config{MMProjArtifactID: 7, Modalities: nil}, + mdl: visionModel, + wantEnabled: []string{"text", "vision"}, + }, + { + name: "model with nil modalities resolves to text only", + cfg: store.Config{MMProjArtifactID: 7, Modalities: nil}, + mdl: emptyModalitiesModel, + wantEnabled: []string{"text"}, + }, + { + name: "unknown/zero artifact id must be treated as not-missing", + cfg: store.Config{MMProjArtifactID: 0, Modalities: nil}, + mdl: visionModel, + // mmprojMissing=true here would be a caller bug (MMProjArtifactID + // is 0, so there's no artifact to be missing) — confirms the + // "no mmproj linked" branch is checked first regardless. + mmprojMissing: true, + wantEnabled: []string{"text"}, + wantUnavail: []string{"vision"}, + wantReason: store.ModalityReasonNoMMProj, + }, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + res := store.ResolveModalities(tc.cfg, tc.mdl, tc.mmprojMissing) + if !reflect.DeepEqual(res.Enabled, tc.wantEnabled) { + t.Errorf("Enabled = %v, want %v", res.Enabled, tc.wantEnabled) + } + if !reflect.DeepEqual(res.Unavailable, tc.wantUnavail) && !(len(res.Unavailable) == 0 && len(tc.wantUnavail) == 0) { + t.Errorf("Unavailable = %v, want %v", res.Unavailable, tc.wantUnavail) + } + if res.Reason != tc.wantReason { + t.Errorf("Reason = %q, want %q", res.Reason, tc.wantReason) + } + }) + } +} + +func ptr[T any](v T) *T { return &v } diff --git a/web/package-lock.json b/web/package-lock.json index e00f6b8..98c3a87 100644 --- a/web/package-lock.json +++ b/web/package-lock.json @@ -2982,9 +2982,9 @@ } }, "node_modules/baseline-browser-mapping": { - "version": "2.10.44", - "resolved": "https://registry.npmjs.org/baseline-browser-mapping/-/baseline-browser-mapping-2.10.44.tgz", - "integrity": "sha512-T3ghW+sl/ZJ8w1v/yQx3qvJ9040DWoLBz8JT/CILbAKcFyG9b2MRe75v6W5uXjv6uH1lumK2Kv46y2zSkcej0Q==", + "version": "2.11.22", + "resolved": "https://registry.npmjs.org/baseline-browser-mapping/-/baseline-browser-mapping-2.11.22.tgz", + "integrity": "sha512-pWc4w51fBFd7mav43/zKRC+RI6f4yfzQoVlfvE8dECePyfkn1bzLp01Fj0QACcyCZyFhiEMyD2qScfKRWgWibA==", "license": "Apache-2.0", "bin": { "baseline-browser-mapping": "dist/cli.cjs" @@ -3006,9 +3006,9 @@ } }, "node_modules/browserslist": { - "version": "4.28.6", - "resolved": "https://registry.npmjs.org/browserslist/-/browserslist-4.28.6.tgz", - "integrity": "sha512-FQBYNK15VMslhLHpA7+n+n1GOlF1kId2xcCg7/j95f24AOF6VDYMNH4mFxF7KuaTdv627faazpOAjFzMrfJOUw==", + "version": "4.28.9", + "resolved": "https://registry.npmjs.org/browserslist/-/browserslist-4.28.9.tgz", + "integrity": "sha512-EWazOblFYUvlGZcfGhPUPmYh3nikUxBVb+y9MJun5f3hBi812X+8MSQTujLBtgK3cf51fJWbWfOjyeO954d+Eg==", "funding": [ { "type": "opencollective", @@ -3025,11 +3025,11 @@ ], "license": "MIT", "dependencies": { - "baseline-browser-mapping": "^2.10.42", - "caniuse-lite": "^1.0.30001803", - "electron-to-chromium": "^1.5.389", - "node-releases": "^2.0.51", - "update-browserslist-db": "^1.2.3" + "baseline-browser-mapping": "^2.11.20", + "caniuse-lite": "^1.0.30001810", + "electron-to-chromium": "^1.5.420", + "node-releases": "^2.0.54", + "update-browserslist-db": "^1.3.2" }, "bin": { "browserslist": "cli.js" @@ -3092,9 +3092,9 @@ } }, "node_modules/caniuse-lite": { - "version": "1.0.30001806", - "resolved": "https://registry.npmjs.org/caniuse-lite/-/caniuse-lite-1.0.30001806.tgz", - "integrity": "sha512-72Cuvd95zbSYPKq6Fhg8eDJRlzgWDf7/mtoZv6Qe/DYNCEBdNxoA3+rZAU2ZhGCpZlns3EssFavaZomckT5Uuw==", + "version": "1.0.30001810", + "resolved": "https://registry.npmjs.org/caniuse-lite/-/caniuse-lite-1.0.30001810.tgz", + "integrity": "sha512-TITQPUkaz+aVk5GL6NhOdwk1aEaNTSDPsGFWrTuhKGtjTF70jL/Oht2W4c6rXUe5fu7Ie19VIahAXHIIiWWNeg==", "funding": [ { "type": "opencollective", @@ -3325,9 +3325,9 @@ } }, "node_modules/electron-to-chromium": { - "version": "1.5.394", - "resolved": "https://registry.npmjs.org/electron-to-chromium/-/electron-to-chromium-1.5.394.tgz", - "integrity": "sha512-Wmt2Gm0o8JWBuGgmc4XZ0u9s1RaCRqhxP47phplmfg04+qypTUurpeJGP45A7Fhv7jdrrVH44PLlR9qXo37cVQ==", + "version": "1.5.427", + "resolved": "https://registry.npmjs.org/electron-to-chromium/-/electron-to-chromium-1.5.427.tgz", + "integrity": "sha512-n14zb3FdsChZ2BNobqNHAJMcP3ifFv4paox2LvCrfVAQcqGiSURgbJl+PfMpHVCNFkStnNc+RRVtPBTVW5PDgw==", "license": "ISC" }, "node_modules/es-abstract": { @@ -4774,9 +4774,9 @@ } }, "node_modules/node-releases": { - "version": "2.0.51", - "resolved": "https://registry.npmjs.org/node-releases/-/node-releases-2.0.51.tgz", - "integrity": "sha512-wRNIrw4DmVLKQlbgOMdkMx27Wrpzes2hh5Jtbi2bjPd+4wJstWIqP5A+lscnqbm0xxmT5Bpg8Lec5ItEBwx6BQ==", + "version": "2.0.55", + "resolved": "https://registry.npmjs.org/node-releases/-/node-releases-2.0.55.tgz", + "integrity": "sha512-mIrE/Cw9y+9Au6dS5vDKDhQza9YvG6w+ZrS6X+ZzA7yFW/soAeaups4Qzn1bL6g5FVy8WtP79+0j82oPIbqRjQ==", "license": "MIT", "engines": { "node": ">=18" @@ -5895,9 +5895,9 @@ } }, "node_modules/update-browserslist-db": { - "version": "1.2.3", - "resolved": "https://registry.npmjs.org/update-browserslist-db/-/update-browserslist-db-1.2.3.tgz", - "integrity": "sha512-Js0m9cx+qOgDxo0eMiFGEueWztz+d4+M3rGlmKPT+T4IS/jP4ylw3Nwpu6cpTTP8R1MAC1kF4VbdLt3ARf209w==", + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/update-browserslist-db/-/update-browserslist-db-1.3.3.tgz", + "integrity": "sha512-pJ2sYawQS0R/WI928Gj5GlPhTGzbMelq0+4INtSYNDV9ErKJcX6xjGWkoG/VnB3dpUm00zALaqkrUD77pO5TDQ==", "funding": [ { "type": "opencollective", diff --git a/web/src/components/RoutingTree.tsx b/web/src/components/RoutingTree.tsx index f2fc3ba..a082577 100644 --- a/web/src/components/RoutingTree.tsx +++ b/web/src/components/RoutingTree.tsx @@ -29,8 +29,10 @@ // compressing teal solid + glow compressing, resource health ok // degraded amber solid + glow compressing but restart-looping or // leaking memory — still passing traffic -// bypassed slate solid operator passthrough (global or this +// bypassed blue solid operator passthrough (global or this // proxy) — direct by deliberate choice +// (own --route-bypass token, not +// --reserved — see theme.css) // autobypass orange dashed compressor DOWN; Sprint 8's per-request // auto-bypass is routing these straight // to the real upstream, uncompressed @@ -43,10 +45,28 @@ // // readOnly (the Dashboard copy) suppresses the admin-only long-press // enable/disable interaction; Settings → Routing keeps it. +// +// Model layout toggle (operator feedback 2026-09-13): providers (left) are +// always alphabetical — that's an invariant, not a mode. Models (right) can +// either stay plain A-Z (default, unchanged from before this toggle existed) +// or be regrouped to cluster near their primary provider's row — "primary" +// mirrors offeringPreference.ts's own rule (lowest priority among enabled +// offerings of enabled providers, same tiebreak), so grouping can never +// disagree with the "preferred" badge shown elsewhere. Grouping is a pure +// display reorder — it doesn't change any link's color/state/priority. +// +// Legend placement: moved off its own full-width row (operator feedback +// 2026-09-13) into the diagram's bottom-left corner, which — since real +// deployments always have far fewer providers than models — sits empty +// below the last provider row. It renders after the node maps (last in DOM) +// so it paints above the SVG's link curves, which do pass through that +// corner; a panel backdrop keeps it legible over them. import { useEffect, useRef, useState } from "react"; import { Icon } from "./Icon"; import { creatorIconSlug } from "../lib/creatorIcon"; +import { groupOfferingsByModel } from "../lib/offeringPreference"; import { providerIconSlug } from "../lib/providerPresets"; +import { RangeToggle } from "./RangeToggle"; import { useCatalogModels, useCatalogOfferings, @@ -57,11 +77,18 @@ import { } from "../lib/queries"; import type { CatalogOffering, CompressorProxy, InfraService } from "../lib/types"; +const LAYOUT_MODES = [ + { key: "az", label: "A–Z" }, + { key: "group", label: "Group" }, +] as const; +type LayoutMode = (typeof LAYOUT_MODES)[number]["key"]; + const COMPRESSOR_ROW_PREFIX = "Compressor ("; type LinkVisual = { cls: string; color: string; opacity: number; dash?: string }; export function RoutingTree({ readOnly = false }: { readOnly?: boolean }) { + const [layoutMode, setLayoutMode] = useState("az"); const offerings = useCatalogOfferings(); const models = useCatalogModels(); const providers = useProviders(); @@ -70,6 +97,13 @@ export function RoutingTree({ readOnly = false }: { readOnly?: boolean }) { const infraServices = useInfraServices(); const offeringList = offerings.data ?? []; + // Dashboard feedback (2026-09-13): the read-only Dashboard copy hides + // disabled offerings (and, transitively, any provider/model left with + // nothing enabled) entirely rather than showing them grayed out — an + // at-a-glance operations view shouldn't include inert routes. Settings → + // Routing (readOnly=false) keeps everything visible, since seeing and + // managing what's disabled is the point of that view. + const visibleOfferings = readOnly ? offeringList.filter((o) => o.enabled) : offeringList; const modelList = models.data ?? []; const providerList = providers.data?.providers ?? []; const providerByName = new Map(providerList.map((p) => [p.name, p])); @@ -111,10 +145,32 @@ export function RoutingTree({ readOnly = false }: { readOnly?: boolean }) { return ""; } - const providerNames = [...new Set(offeringList.map((o) => o.provider))].sort(); - const modelIds = [...new Set(offeringList.map((o) => o.model_id))].sort((a, b) => { + // Providers are always alphabetical — an invariant, not part of the + // layout toggle below (see file header). + const providerNames = [...new Set(visibleOfferings.map((o) => o.provider))].sort(); + const providerIndex = new Map(providerNames.map((n, i) => [n, i])); + // "Group" mode's per-model primary provider — same rule as + // offeringPreference.ts's preferredOfferingIds (lowest priority among + // enabled offerings of enabled providers, tiebreak provider name then id), + // so a model's group can never disagree with the "preferred" badge shown + // elsewhere. Falls back to the group's first entry (any offering) for a + // model with nothing routable, so it still lands in a deterministic spot + // instead of being excluded from grouping. + const providerEnabled = (name: string) => providerByName.get(name)?.enabled ?? true; + const offeringsByModel = groupOfferingsByModel(visibleOfferings); + function primaryProvider(modelId: number): string { + const group = offeringsByModel.get(modelId) ?? []; + const routable = group.find((o) => o.enabled && providerEnabled(o.provider)); + return (routable ?? group[0])?.provider ?? ""; + } + const modelIds = [...new Set(visibleOfferings.map((o) => o.model_id))].sort((a, b) => { const na = modelById.get(a)?.name ?? `#${a}`; const nb = modelById.get(b)?.name ?? `#${b}`; + if (layoutMode === "group") { + const ia = providerIndex.get(primaryProvider(a)) ?? Number.MAX_SAFE_INTEGER; + const ib = providerIndex.get(primaryProvider(b)) ?? Number.MAX_SAFE_INTEGER; + if (ia !== ib) return ia - ib; + } return na.localeCompare(nb); }); @@ -125,7 +181,17 @@ export function RoutingTree({ readOnly = false }: { readOnly?: boolean }) { const RIGHT_X = 532; const W = 700; const rows = Math.max(providerNames.length, modelIds.length); - const H = Math.max(rows * ROW_H + PAD * 2, 80); + // Legend sits below the last provider row (see the legend div's own + // comment) rather than pinned to the diagram's bottom edge — operator + // feedback 2026-09-13: anchoring to the bottom made it climb up over the + // provider column whenever models outnumbered providers by enough that + // the empty gap below the last provider row was shorter than the legend + // itself. LEGEND_H is a fixed estimate (7 two-column entries + 1 + // full-width note), reserved in H so the legend never overflows past the + // diagram's own bottom edge either. + const LEGEND_TOP = PAD + providerNames.length * ROW_H + 8; + const LEGEND_H = 118; + const H = Math.max(rows * ROW_H + PAD * 2, 80, LEGEND_TOP + LEGEND_H); const midX = (LEFT_X + RIGHT_X) / 2; const providerY = new Map(providerNames.map((n, i) => [n, PAD + i * ROW_H + ROW_H / 2])); @@ -187,7 +253,7 @@ export function RoutingTree({ readOnly = false }: { readOnly?: boolean }) { if (!service) return { cls: "direct", color: "var(--ok)", opacity: 0.9 }; // Operator passthrough (global or per-proxy) — direct by deliberate // choice, distinct from Sprint 8's failure-driven auto-bypass above. - if (state === "bypassed") return { cls: "bypassed", color: "var(--reserved)", opacity: 0.8 }; + if (state === "bypassed") return { cls: "bypassed", color: "var(--route-bypass)", opacity: 0.8 }; // Still compressing, but the process is restart-looping or leaking // (Sprint 4's resource health) — traffic passes, health is not ok. if (health === "restarting" || health === "memory_growth") { @@ -201,27 +267,20 @@ export function RoutingTree({ readOnly = false }: { readOnly?: boolean }) { if (offeringList.length === 0) { return
No offerings configured — nothing to map yet.
; } + if (visibleOfferings.length === 0) { + return
All offerings are currently disabled.
; + } const lpressCircumference = 2 * Math.PI * 17; return ( <> -
- direct - compressing - compressing · degraded - operator bypass - auto-bypassed (compressor down) - failing (compressor + upstream down) - provider unreachable - disabled - - thicker = higher priority{readOnly ? "" : " · hold a provider node to enable/disable"} - +
+
- {offeringList.map((o) => { + {visibleOfferings.map((o) => { const py = providerY.get(o.provider); const my = modelY.get(o.model_id); if (py === undefined || my === undefined) return null; @@ -281,6 +340,29 @@ export function RoutingTree({ readOnly = false }: { readOnly?: boolean }) {
); })} + {/* Anchored below the last provider row (LEGEND_TOP, computed above + H), after the node maps so it paints above the SVG's link curves + (see file header). "direct" (no compressor anywhere in the path) + is dropped from this legend: with the shared "external" proxy + covering every provider that has no dedicated one, every + currently-enabled route always has SOME compressor in its path + on this deployment, so that line/color never actually appears — + it'd be pure noise. linkState's `direct` case itself stays (a + defensive fallback if that ever stops being true, e.g. the + operator disables the shared external proxy), it's just no + longer documented in the key since it can't happen today. */} +
+ compressing + compressing · degraded + operator bypass + auto-bypassed (compressor down) + failing (compressor + upstream down) + provider unreachable + disabled + + thicker = higher priority{readOnly ? "" : " · hold a provider node to enable/disable"} + +
); diff --git a/web/src/components/charts/ActivityHeatmap.tsx b/web/src/components/charts/ActivityHeatmap.tsx index 4ce6e1e..31ab1e9 100644 --- a/web/src/components/charts/ActivityHeatmap.tsx +++ b/web/src/components/charts/ActivityHeatmap.tsx @@ -55,33 +55,75 @@ function levelFor(tokens: number, max: number): number { return 1; } -export function ActivityHeatmap({ days, colors = DEFAULT_COLORS, ariaLabel = "Token activity by day, last 12 weeks" }: { days: HeatmapCell[]; colors?: string[]; ariaLabel?: string }) { - if (days.length === 0) { +// One cross-fadeable data source for the grid — ActivityHeatmapWidget's +// auto-cycle passes two of these (the outgoing scope at opacity 1→0, the +// incoming scope at opacity 0→1) so the swap is a true crossfade, not a +// fade-to-empty-then-fade-in. A caller not cross-fading just passes one. +export interface HeatmapLayer { + days: HeatmapCell[]; + colors?: string[]; + opacity: number; +} + +export function ActivityHeatmap({ + layers, + ariaLabel = "Token activity by day", + stretch = false, + fadeMs = 250, +}: { + // Grid geometry (dates, columns, month labels) is derived from + // layers[0].days — every layer is expected to share the same date range + // (true for the widget's All/Local/External scopes, which all come from + // the same window, just different per-day fields), only cell colors + // differ per layer. + layers: HeatmapLayer[]; + ariaLabel?: string; + // stretch (operator feedback 2026-09-13): the grid's natural pixel width + // (LEFT_PAD + columns*STEP) is normally just a ceiling — maxWidth:100% + // only ever shrinks it to fit a narrower card, never grows it to fill a + // wider one, so a card sized for the widget's own section reads as mostly + // empty. width:100% instead always fills the container; height:auto plus + // the SVG's own viewBox (no preserveAspectRatio override) keeps every + // cell square and proportional either way — this only changes whether the + // grid is allowed to scale UP, not how cells are laid out. + stretch?: boolean; + // fadeMs: how long each layer's own opacity transition takes. With two + // layers cross-fading simultaneously (one 1→0, one 0→1 at the same time) + // this is the whole transition's duration, not half of it. + fadeMs?: number; +}) { + const referenceDays = layers[0]?.days ?? []; + if (referenceDays.length === 0) { return
No activity data yet.
; } - const maxTokens = Math.max(0, ...days.map((d) => d.tokens)); - const firstDate = parseDay(days[0].date); + const firstDate = parseDay(referenceDays[0].date); const firstSunday = new Date(firstDate); firstSunday.setUTCDate(firstSunday.getUTCDate() - firstSunday.getUTCDay()); type Cell = { x: number; y: number; day: HeatmapCell; date: Date }; - const cells: Cell[] = []; + function buildCells(days: HeatmapCell[]): Cell[] { + return days.map((day) => { + const date = parseDay(day.date); + const weekday = date.getUTCDay(); + const col = Math.round((date.getTime() - firstSunday.getTime()) / (7 * 86400000)); + return { x: LEFT_PAD + col * STEP, y: TOP_PAD + weekday * STEP, day, date }; + }); + } + + // Month labels + overall size come from the reference layer only — every + // layer shares the same dates, so this never needs to be computed twice. let maxCol = 0; let lastMonthLabeled = -1; const monthLabels: { x: number; text: string }[] = []; - - for (const day of days) { + for (const day of referenceDays) { const date = parseDay(day.date); const weekday = date.getUTCDay(); const col = Math.round((date.getTime() - firstSunday.getTime()) / (7 * 86400000)); maxCol = Math.max(maxCol, col); - const x = LEFT_PAD + col * STEP; - const y = TOP_PAD + weekday * STEP; - cells.push({ x, y, day, date }); if (weekday === 0 && date.getUTCMonth() !== lastMonthLabeled) { lastMonthLabeled = date.getUTCMonth(); - monthLabels.push({ x, text: MONTH_NAMES[date.getUTCMonth()] }); + monthLabels.push({ x: LEFT_PAD + col * STEP, text: MONTH_NAMES[date.getUTCMonth()] }); } } @@ -94,7 +136,7 @@ export function ActivityHeatmap({ days, colors = DEFAULT_COLORS, ariaLabel = "To viewBox={`0 0 ${width} ${height}`} width={width} height={height} - style={{ maxWidth: "100%", height: "auto", display: "block" }} + style={stretch ? { width: "100%", height: "auto", display: "block" } : { maxWidth: "100%", height: "auto", display: "block" }} role="img" aria-label={ariaLabel} > @@ -114,15 +156,24 @@ export function ActivityHeatmap({ days, colors = DEFAULT_COLORS, ariaLabel = "To {label} ))} - {cells.map((c) => { - const level = levelFor(c.day.tokens, maxTokens); - const dateLabel = c.date.toLocaleDateString([], { month: "short", day: "numeric", year: "numeric", timeZone: "UTC" }); + {layers.map((layer, i) => { + const cells = buildCells(layer.days); + const colors = layer.colors ?? DEFAULT_COLORS; + const maxTokens = Math.max(0, ...layer.days.map((d) => d.tokens)); return ( - - - {dateLabel}: {formatTokens(c.day.tokens)} tokens, {c.day.requests} request{c.day.requests === 1 ? "" : "s"} - - + + {cells.map((c) => { + const level = levelFor(c.day.tokens, maxTokens); + const dateLabel = c.date.toLocaleDateString([], { month: "short", day: "numeric", year: "numeric", timeZone: "UTC" }); + return ( + + + {dateLabel}: {formatTokens(c.day.tokens)} tokens, {c.day.requests} request{c.day.requests === 1 ? "" : "s"} + + + ); + })} + ); })} diff --git a/web/src/components/widgets/ActivityHeatmapWidget.tsx b/web/src/components/widgets/ActivityHeatmapWidget.tsx index 0e3b997..ff14cdd 100644 --- a/web/src/components/widgets/ActivityHeatmapWidget.tsx +++ b/web/src/components/widgets/ActivityHeatmapWidget.tsx @@ -1,4 +1,4 @@ -import { useState } from "react"; +import { useEffect, useState } from "react"; import { useUsageHeatmap } from "../../lib/queries"; import { ActivityHeatmap, sequentialRamp } from "../charts/ActivityHeatmap"; import { RangeToggle } from "../RangeToggle"; @@ -18,32 +18,91 @@ const HEATMAP_SCOPE_BASE_HEX: Partial> = { external: "#b100e8", }; +// Auto-cycle (operator feedback 2026-09-13): steps through All → Local → +// External → All on its own, TRUE cross-fading over FADE_MS — not a +// fade-out-then-fade-in. Two "slots" (index 0 and 1) each hold one scope's +// data; exactly one is active (opacity 1) at rest. Every CYCLE_MS +// (= PAUSE_MS + FADE_MS — the interval timer fires once per pause-then-fade +// cycle, not once per fade alone) the INACTIVE slot's data is swapped to +// the next scope (invisible at that instant, so the swap itself is never +// seen) and the two opacities flip simultaneously, so the outgoing scope's +// cells and the incoming scope's cells visibly overlap and cross-fade for +// the whole FADE_MS, holding static for PAUSE_MS in between. A manual +// RangeToggle click overwrites the active slot's data directly (no opacity +// change, so no transition — an instant jump) and the cycle just continues +// forward from there. +const PAUSE_MS = 3000; +const FADE_MS = 3000; +const CYCLE_MS = PAUSE_MS + FADE_MS; + +type CycleState = { active: 0 | 1; scopes: [HeatmapScopeKey, HeatmapScopeKey] }; + +function nextScopeAfter(scope: HeatmapScopeKey): HeatmapScopeKey { + const idx = HEATMAP_SCOPES.findIndex((s) => s.key === scope); + return HEATMAP_SCOPES[(idx + 1) % HEATMAP_SCOPES.length].key; +} + // Widget "activity-heatmap" (see lib/widgetRegistry.ts). Extracted verbatim // from Dashboard's Overview tab, Phase 5 (2026-08-12). export function ActivityHeatmapWidget() { - const heatmap = useUsageHeatmap("84d"); - const [heatmapScope, setHeatmapScope] = useState("all"); - - const heatmapDays = heatmap.data?.days.map((d) => ({ - date: d.date, - tokens: heatmapScope === "local" ? d.tokens_local : heatmapScope === "external" ? d.tokens_external : d.tokens, - requests: heatmapScope === "local" ? d.requests_local : heatmapScope === "external" ? d.requests_external : d.requests, - })); - const heatmapScopeHex = HEATMAP_SCOPE_BASE_HEX[heatmapScope]; - const heatmapColors = heatmapScopeHex ? sequentialRamp(heatmapScopeHex) : undefined; - const heatmapScopeLabel = HEATMAP_SCOPES.find((sc) => sc.key === heatmapScope)!.label; + const heatmap = useUsageHeatmap("365d"); + const [cycle, setCycle] = useState({ active: 0, scopes: ["all", "all"] }); + + useEffect(() => { + const intervalId = setInterval(() => { + setCycle((prev) => { + const nextActive: 0 | 1 = prev.active === 0 ? 1 : 0; + const scopes: [HeatmapScopeKey, HeatmapScopeKey] = [...prev.scopes]; + scopes[nextActive] = nextScopeAfter(prev.scopes[prev.active]); + return { active: nextActive, scopes }; + }); + }, CYCLE_MS); + return () => clearInterval(intervalId); + }, []); + + function handleManualChange(scope: HeatmapScopeKey) { + setCycle((prev) => { + const scopes: [HeatmapScopeKey, HeatmapScopeKey] = [...prev.scopes]; + scopes[prev.active] = scope; + return { ...prev, scopes }; + }); + } + + function daysForScope(scope: HeatmapScopeKey) { + return (heatmap.data?.days ?? []).map((d) => ({ + date: d.date, + tokens: scope === "local" ? d.tokens_local : scope === "external" ? d.tokens_external : d.tokens, + requests: scope === "local" ? d.requests_local : scope === "external" ? d.requests_external : d.requests, + })); + } + function colorsForScope(scope: HeatmapScopeKey) { + const hex = HEATMAP_SCOPE_BASE_HEX[scope]; + return hex ? sequentialRamp(hex) : undefined; + } + + const activeScope = cycle.scopes[cycle.active]; + const activeScopeLabel = HEATMAP_SCOPES.find((sc) => sc.key === activeScope)!.label; + const hasData = (heatmap.data?.days.length ?? 0) > 0; + const layers = [0, 1].map((i) => { + const scope = cycle.scopes[i as 0 | 1]; + return { + days: daysForScope(scope), + colors: colorsForScope(scope), + opacity: cycle.active === i ? 1 : 0, + }; + }); return ( <>
- Token activity · last 12 weeks - + Token activity · last year +
{heatmap.isLoading ? (
Loading activity…
- ) : heatmapDays && heatmapDays.length > 0 ? ( - + ) : hasData ? ( + ) : (
No activity data yet.
)} diff --git a/web/src/styles/theme.css b/web/src/styles/theme.css index add779f..fc571d8 100644 --- a/web/src/styles/theme.css +++ b/web/src/styles/theme.css @@ -34,6 +34,12 @@ light reads clearly distinct from Temp (green→fuchsia) and GPU (orange), and is less saturated than the previously-rejected #3A86FF. */ --ring-vram: #6FA3C8; + /* RoutingTree "operator bypass" link — was var(--reserved), which is a + shared multi-use token (slot-reservation badges, card borders) and, at + the low saturation it needs for those roles, read as gray rather than + blue on this diagram (operator feedback 2026-09-13). Its own token so + recoloring it never ripples into --reserved's other consumers. */ + --route-bypass: #3a8fe0; --heat-deep: color-mix(in srgb, var(--heat) 55%, black); --tab-active: #7c5cff; --load-btn-bg: var(--heat); --load-btn-text: #100a06; @@ -117,6 +123,7 @@ --reserved: color-mix(in srgb, var(--cool) 65%, white); --ok: #17a34a; --ring-vram: #33658A; + --route-bypass: #1a5fb4; --tab-active: #00b4d8; --navy-electric: #1421c4; --load-btn-bg: var(--navy-electric); --load-btn-text: #fff; @@ -157,6 +164,7 @@ light reads clearly distinct from Temp (green→fuchsia) and GPU (orange), and is less saturated than the previously-rejected #3A86FF. */ --ring-vram: #6FA3C8; + --route-bypass: #3a8fe0; --heat-deep: color-mix(in srgb, var(--heat) 55%, black); --tab-active: #7c5cff; --load-btn-bg: var(--heat); --load-btn-text: #100a06; @@ -188,6 +196,7 @@ --reserved: color-mix(in srgb, var(--cool) 65%, white); --ok: #17a34a; --ring-vram: #33658A; + --route-bypass: #1a5fb4; --tab-active: #00b4d8; --navy-electric: #1421c4; --load-btn-bg: var(--navy-electric); --load-btn-text: #fff; @@ -1248,13 +1257,24 @@ h1,h2,h3 { text-wrap: balance; margin: 0; } marks BOTH compressing states (healthy teal, degraded amber). */ .rtree-link.compressing { filter: drop-shadow(0 0 5px var(--cool)); } .rtree-link.degraded { filter: drop-shadow(0 0 5px var(--heat-2)); } -.rtree-legend { display: flex; align-items: center; flex-wrap: wrap; gap: 14px; margin: 0 0 12px; font-size: 11px; color: var(--text-dim); } +/* Anchored below the last provider row inside .rtree, `top` set inline per + the live provider count (operator feedback 2026-09-13 — pinning this to + `bottom:0` let it climb up over the provider column whenever the model + column was tall enough to leave a gap there shorter than the legend). + A 2-column grid (was a single stacked column) roughly halves the + legend's own height so it fits a typical few-provider gap; a panel + backdrop keeps it legible since the SVG's link curves do pass behind it. */ +.rtree-legend { + position: absolute; left: 0; max-width: min(100%, 340px); + display: grid; grid-template-columns: repeat(2, minmax(0, 1fr)); gap: 4px 16px; + padding: 8px 10px; border-radius: 8px; + background: var(--panel-2); font-size: 11px; color: var(--text-dim); +} .rtree-lg { display: inline-flex; align-items: center; gap: 6px; } -.rtree-lg-line { display: inline-block; width: 26px; height: 0; border-top: 3px solid; border-radius: 2px; } -.rtree-lg-line.direct { border-top-color: var(--ok); } +.rtree-lg-line { display: inline-block; width: 26px; height: 0; border-top: 3px solid; border-radius: 2px; flex: none; } .rtree-lg-line.compressing { border-top-color: var(--cool); filter: drop-shadow(0 0 3px var(--cool)); } .rtree-lg-line.degraded { border-top-color: var(--heat-2); filter: drop-shadow(0 0 3px var(--heat-2)); } -.rtree-lg-line.bypassed { border-top-color: var(--reserved); } +.rtree-lg-line.bypassed { border-top-color: var(--route-bypass); } .rtree-lg-line.autobypass { border-top-color: var(--heat); border-top-style: dashed; border-top-width: 2px; } .rtree-lg-line.failing { border-top-color: var(--crit); } .rtree-lg-line.disabled { border-top-color: var(--text-mute); opacity: .3; } @@ -1266,8 +1286,8 @@ h1,h2,h3 { text-wrap: balance; margin: 0; } glance in both themes. A repeating-linear-gradient dash is rendered crisply at any size, unlike the browser's border-dash algorithm at small scales. */ .rtree-lg-line.unreachable { height: 3px; border-top: none; background-image: repeating-linear-gradient(to right, var(--crit) 0 5px, transparent 5px 9px); } -.rtree-lg-note { color: var(--text-mute); font-size: 10.5px; margin-left: auto; } -@media (max-width: 560px) { .rtree { overflow-x: auto; } .rtree-lg-note { margin-left: 0; width: 100%; } } +.rtree-lg-note { grid-column: 1 / -1; color: var(--text-mute); font-size: 10.5px; } +@media (max-width: 560px) { .rtree { overflow-x: auto; } } @media (prefers-reduced-motion: reduce) { .rtree-lpress circle { animation: none; } }