From 17ae9f93fed4c65be703475e09128fd6c2547316 Mon Sep 17 00:00:00 2001 From: Xore Date: Fri, 25 Sep 2026 21:30:04 +0200 Subject: [PATCH] fix(ghidra): raise the Ghidra triage/benchmark output cap 512 -> 4096 The 512 cap truncated answers rather than limiting them. Measured 2026-09-25 on the Tier A corpus: qwen2.5-coder:7b-instruct-q4_K_M was cut off on 14/14 cases at 512, still 1/17 at 2048 (done_reason: length on file_write_persist, 5127 chars), and 0/17 at 4096. #2694 had already recorded the same cap truncating injection conclusions in 23/30 Tier B answers, so any baseline recorded at 512 is suspect in the same direction -- and the base model is penalised hardest because it writes longer answers. Two sites, because they are independent defaults: - approved-models.json: ghidra qualification_request and runtime_request. CI's 'Ghidra output cap matches the approved runtime request' pins the worker constant to this value, so leaving it would have kept the live worker truncating while the manifest claimed otherwise. - ghidra-worker.py: TRIAGE_OUTPUT_TOKENS default. This is the production triage path, not only the benchmark. revdeck.qualification_request also moves to 4096; it is the benchmark's own measurement slot, and largest measured prompt is 2008 tokens so 2008 + 4096 = 6104 < its 8192 context. sessions stays at 512. That slot is llm-worker session classification, not reverse engineering, and llm-worker clamps LLM_OUTPUT_TOKENS to max=2048 with an 8192 context -- the manifest value must stay representable there. Its 512 was not a truncation defect. Refs #1804, #1947, #2694 --- analysis/ghidra/models/approved-models.json | 6 +++--- analysis/ghidra/worker/ghidra-worker.py | 2 +- analysis/ghidra/worker/tests/test_ghidra_worker.py | 2 +- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/analysis/ghidra/models/approved-models.json b/analysis/ghidra/models/approved-models.json index 98cfb6ef0..f99d27a11 100644 --- a/analysis/ghidra/models/approved-models.json +++ b/analysis/ghidra/models/approved-models.json @@ -68,7 +68,7 @@ "concurrency": 1, "context_tokens": 32768, "keep_alive": "10m", - "output_tokens": 512, + "output_tokens": 4096, "seed": 144, "temperature": 0, "thinking": false @@ -77,7 +77,7 @@ "concurrency": 1, "context_tokens": 32768, "keep_alive": "30m", - "output_tokens": 512, + "output_tokens": 4096, "seed": 144, "temperature": 0, "thinking": false @@ -123,7 +123,7 @@ "concurrency": 1, "context_tokens": 8192, "keep_alive": "10m", - "output_tokens": 512, + "output_tokens": 4096, "seed": 144, "temperature": 0, "thinking": false diff --git a/analysis/ghidra/worker/ghidra-worker.py b/analysis/ghidra/worker/ghidra-worker.py index 48d69beca..fa4ca6e6e 100755 --- a/analysis/ghidra/worker/ghidra-worker.py +++ b/analysis/ghidra/worker/ghidra-worker.py @@ -219,7 +219,7 @@ def _reassert_dir_perms(path: Path) -> None: "GHIDRA_TRIAGE_API_BASE", "http://127.0.0.1:11434/v1").rstrip("/") TRIAGE_MODEL = os.environ.get("GHIDRA_TRIAGE_MODEL", "qwen3:14b") TRIAGE_TIMEOUT = int(os.environ.get("GHIDRA_TRIAGE_TIMEOUT", "300")) -TRIAGE_OUTPUT_TOKENS = int(os.environ.get("GHIDRA_TRIAGE_OUTPUT_TOKENS", "512")) +TRIAGE_OUTPUT_TOKENS = int(os.environ.get("GHIDRA_TRIAGE_OUTPUT_TOKENS", "4096")) TRIAGE_SEED = int(os.environ.get("GHIDRA_TRIAGE_SEED", "144")) # #2646: temperature 0 and a fixed seed do not make two runs agree. Ollama diff --git a/analysis/ghidra/worker/tests/test_ghidra_worker.py b/analysis/ghidra/worker/tests/test_ghidra_worker.py index ec477e51e..b08052d57 100644 --- a/analysis/ghidra/worker/tests/test_ghidra_worker.py +++ b/analysis/ghidra/worker/tests/test_ghidra_worker.py @@ -700,7 +700,7 @@ def test_triage(ghidra, model, truncating): "the system prompt names the evidence as untrusted") check(all(p.get("reasoning_effort") == "none" for p in ModelStub.prompts), "bounded triage disables hidden reasoning") - check(all(p.get("max_tokens") == 512 for p in ModelStub.prompts), + check(all(p.get("max_tokens") == 4096 for p in ModelStub.prompts), "the approved output cap is sent to the model") check(all(p.get("seed") == 144 for p in ModelStub.prompts), "the approved deterministic seed is sent to the model")