diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp index c23a9c13503..37a7064d88e 100644 --- a/src/llama-graph.cpp +++ b/src/llama-graph.cpp @@ -2469,7 +2469,17 @@ ggml_tensor * llm_graph_context::build_moe_ffn( ean.reserve(n_ean_il); for (uint32_t i = 0; i < n_ean_il; ++i) { ggml_tensor * slot = ggml_view_2d(ctx0, experts, n_embd, n_tokens, experts->nb[2], i*experts->nb[1]); - ggml_tensor * norm = ggml_sqrt(ctx0, ggml_sum_rows(ctx0, ggml_sqr(ctx0, slot))); // [1, n_tokens] + // RULING 1 (hydra_vortex#806): `slot` is a strided ggml_view_2d + // into [n_embd, n_expert_used, n_tokens], so ggml_sqr(slot) hits + // ggml_cuda_op_sqr's contiguity assert and aborts the server at + // load warmup whenever HYDRA_EAN_STATS=1. Materialize ONE + // contiguous copy (single-copy fallback when the view is not + // already contiguous) and assert build-time contiguity of the + // sqr input. Entirely inside the env gate: unset HYDRA_EAN_STATS + // still adds no ops — byte-identical OFF graph. + ggml_tensor * slot_c = ggml_is_contiguous(slot) ? slot : ggml_cont(ctx0, slot); + GGML_ASSERT(ggml_is_contiguous(slot_c)); + ggml_tensor * norm = ggml_sqrt(ctx0, ggml_sum_rows(ctx0, ggml_sqr(ctx0, slot_c))); // [1, n_tokens] ggml_build_forward_expand(gf, norm); ean.push_back(norm); } diff --git a/tools/expert-atlas/README.md b/tools/expert-atlas/README.md index e69ef3c512c..b0e7b666000 100644 --- a/tools/expert-atlas/README.md +++ b/tools/expert-atlas/README.md @@ -40,7 +40,9 @@ production stats path once it lands. - `analyze.py` — statistics port (mean share, p(c|e), spec = 1 − H/log C, replication gate) emitting the observability-tier `experts.json` + provenance - `export_pinfile.py` — `experts.json`/`expert-ranks.json` → pin file - (`L `, hot-first, wrap-aware). Engine-agnostic: `--expect-engine-id` + (`L `, hot-first, wrap-aware). Format contract; consumer not + present in this tree (see `tools/atlas/placement_proposal.py:11-13` in the + parent repo — no in-tree pinfile parser). Engine-agnostic: `--expect-engine-id` refusal + optional `--check-url` live geometry match. Round-trip verified byte-identical to the parent repo artifact - `validate.py` — leave-one-prompt-out validation (LLM pop review #175): the diff --git a/tools/expert-atlas/export_pinfile.py b/tools/expert-atlas/export_pinfile.py index c56cf6d4aee..80a9fc29e2e 100755 --- a/tools/expert-atlas/export_pinfile.py +++ b/tools/expert-atlas/export_pinfile.py @@ -6,8 +6,9 @@ # discipline, colibri route_trace.h: histories/artifacts from another engine # are never consumed). # -# Pin-file format (llama-context.cpp hydra_cpu_init; identical parser in -# ggml-cuda.cu): +# Pin-file format contract; consumer not present in this tree (see +# tools/atlas/placement_proposal.py:11-13 in the parent repo — "nothing in +# prod reads a pin file"; there is no in-tree pinfile parser): # - '#' lines and blank lines are skipped # - each pin line: "L "; il must be in [0, 256) # - ORDER IS SIGNIFICANT: ids are hot-first (design §4 contract) diff --git a/tools/server/server-atlas.cpp b/tools/server/server-atlas.cpp index a79f9692312..bbdfdef275c 100644 --- a/tools/server/server-atlas.cpp +++ b/tools/server/server-atlas.cpp @@ -458,6 +458,33 @@ void reset() { g_cap_embd = 0; } +// hydra_vortex#806 (RULING 3): scope-aware reset for POST /atlas/reset — +// kept separate from reset() (the per-probe protocol) so EAN-only resets +// never wipe expert heat counts/seq/turn/capture state, and scope "all" +// stays a true full reset incl. EAN. "ean" zeroes the accumulators in place +// (sizes preserved: /experts keeps serving "ean" with cells=0 until the +// next decode refills it), so the next prompt's EAN matches that same +// prompt on a fresh server (per-prompt replay, no cross-prompt leak). +bool reset_scope(const std::string & scope) { + if (scope == "all") { + reset(); + return true; + } + if (scope != "ean") { + return false; + } + std::lock_guard lock(g_mtx); + if (g_geom) { + const size_t cells = (size_t) g_geom->rows * (size_t) g_geom->cols; + g_ean_gxn.assign(cells, 0.0); + g_ean_nsel.assign(cells, 0); + } else { + g_ean_gxn.clear(); + g_ean_nsel.clear(); + } + return true; +} + // hydra #785: one-shot load-time capture. Breakdown split mirrors // common/memory_breakdown_print (fit.cpp:938-965): host buft → ram, // device buft → vram. Device totals via ggml_backend_dev_memory; host diff --git a/tools/server/server-atlas.h b/tools/server/server-atlas.h index ed844815543..26060325b8c 100644 --- a/tools/server/server-atlas.h +++ b/tools/server/server-atlas.h @@ -67,6 +67,15 @@ void reset(); // ":"); absent until counted. Thread-safe. bool ean_enabled(); void accumulate_ean(int grid_row, const int32_t * ids, const float * gates, const float * norms, int n, int cols); +// hydra_vortex#806 (RULING 3): scope-aware reset behind the new +// POST /atlas/reset?scope=ean|all route — deliberately separate from +// reset() (per-probe protocol) so the two are never conflated. +// "ean" -> zeroes ONLY the EAN accumulators (g_ean_gxn gate*norm sums + +// g_ean_nsel per-cell sample counts); expert heat counts, seq, +// turn window and capture sidecar survive. +// "all" -> full reset including EAN (reset() semantics). +// Returns false on an unknown scope (caller rejects with 400). Thread-safe. +bool reset_scope(const std::string & scope); // Allocation snapshot for honest tier/hwinfo reporting on engine /health // (#785). Captured once at model-load time (sleep-safe: get_health reads the // stored copy, never ctx_server). Tiers are device/host buffer splits from diff --git a/tools/server/server.cpp b/tools/server/server.cpp index 432893a04ef..f0b8a4afed2 100644 --- a/tools/server/server.cpp +++ b/tools/server/server.cpp @@ -428,6 +428,25 @@ int llama_server(common_params & params, int argc, char ** argv) { return res; })); + // hydra_vortex#806 (RULING 3): per-prompt EAN reset WITHOUT the + // /capture/reset surface above (that route 503s unless capture is on and + // the gate harness + atlas-web depend on its semantics). + // POST /atlas/reset?scope=ean — zero only the EAN accumulators + // (g_ean_gxn / g_ean_nsel), leaving expert heat counts/seq untouched. + // POST /atlas/reset?scope=all — full reset including EAN. + // 200 + small JSON on success; 400 on unknown scope. + ctx_http.post("/atlas/reset", ex_wrapper([](const server_http_req & req) { + auto res = std::make_unique(); + const std::string scope = req.get_param("scope"); + if (!hydra_atlas::reset_scope(scope)) { + res->status = 400; + res->data = safe_json_to_str(json{{"error", "scope must be 'ean' or 'all'"}}); + return res; + } + res->data = safe_json_to_str(json{{"ok", true}, {"scope", scope}}); + return res; + })); + // hydra task-16ec332378: engine-hosted Stage-C atlas artifacts (design // §C/D owner ruling: the fork ships the two atlas files; atlas-web proxies // GET {engine}/experts.json and propagates the status verbatim — engine