From 69eaf03672f44d1bdbc77f80c859af1f4c6418b5 Mon Sep 17 00:00:00 2001 From: Jeongkyu Shin Date: Fri, 2 Oct 2026 14:44:03 +0900 Subject: [PATCH 1/7] update(speculative): gate CUDA prompt-lookup drafting on shadow probes On GB10 prompt lookup ran prose 3-21% below plain decoding (#2091). A synchronous verify forward there costs, in pipelined steps, 1.14-1.35 at width 2, 1.41-1.62 at width 3, and 2.5-3.6 from width 5 up (widths 5-7 cost more than 8, where `qmm_sm80` takes over), and every drafted round also drained the pipeline. The PR #2074 governor kept probing prose with such blocks every 4-32 rounds. `DraftPolicy::Gated`, the default on CUDA builds, uses only a narrow block (2 proposals) or the full one, starts narrow and on probation, and never ends a pause on a timer: while paused it keeps looking up and resumes only when two proposed tokens in a row come true unverified (`ShadowProbe`), costing a hash probe instead of a verify forward. `DraftPolicy::Graded` keeps PR #2074's rules unchanged and stays the default elsewhere, since they were tuned on Apple Silicon and no Metal measurement backs a change. `--prompt-lookup-policy auto|graded|gated` selects either for A/B, and the summary line reports `shadow_confirmations`. `examples/verify_width_cost.rs` measures the per-width cost table quoted in `DraftPolicy`. Tests: gated start, probation, timer-free pause, narrow/full widths, shadow settling, policy defaults, CLI parsing; the rollback parity matrix now runs both policies and requires a shadow-confirmed resume. Five deliberate mutations of the gated logic each fail them. Refs #2091 --- examples/verify_width_cost.rs | 117 ++++++++ src/lib.rs | 2 +- .../src/speculative/prompt_lookup.rs | 261 +++++++++++++++--- .../src/speculative/prompt_lookup_tests.rs | 157 ++++++++++- src/main.rs | 39 +++ src/main_tests.rs | 40 +++ 6 files changed, 579 insertions(+), 37 deletions(-) create mode 100644 examples/verify_width_cost.rs diff --git a/examples/verify_width_cost.rs b/examples/verify_width_cost.rs new file mode 100644 index 000000000..409143a6c --- /dev/null +++ b/examples/verify_width_cost.rs @@ -0,0 +1,117 @@ +// Copyright 2025-2026 Lablup Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! What a speculative verify block costs at each width, in pipelined decode +//! steps (issue #2091). +//! +//! Loads a model, prefills a fixed 300-token prompt, then times: +//! +//! - a pipelined one-token step, as `CxxGenerator` runs it (the next step is +//! submitted from the still-lazy argmax before the host reads it); +//! - a synchronous forward of `w` tokens plus the argmax and its host read, +//! for `w` in 1..=10, trimming `w - 1` positions afterwards so the cache +//! grows one token per round like a rejected verify. +//! +//! Each width reports its median over 40 rounds and its ratio to the +//! pipelined step. `DraftPolicy` in `speculative/prompt_lookup.rs` quotes +//! these ratios for GB10; rerun this after an MLX pin bump that moves the +//! quantized-matmul kernel boundaries. +//! +//! ```bash +//! cargo build --release --features cuda --example verify_width_cost +//! gpu-lock run --tag verify-width -- \ +//! target/release/examples/verify_width_cost models/mlx/qwen3-8b-4bit +//! ``` +fn main() { + use mlxcel::generate::LanguageModel; + use std::path::Path; + use std::time::Instant; + + let path = std::env::args().nth(1).expect("model path"); + let (model, _) = mlxcel::load_model(Path::new(&path)).unwrap(); + let prompt: Vec = (0..300).map(|i| 1000 + (i * 37) % 5000).collect(); + + let median = |mut v: Vec| { + v.sort_by(|a, b| a.partial_cmp(b).unwrap()); + v[v.len() / 2] + }; + + let fresh = || { + let mut caches = model.make_caches(); + let input = mlxcel_core::from_slice_i32(&prompt, &[1, prompt.len() as i32]); + let logits = model.forward(&input, &mut caches, None); + mlxcel_core::eval(&logits); + caches + }; + + // Pipelined one-token steps, as CxxGenerator runs them. + let pipelined = |n: usize| { + let mut caches = fresh(); + let mut y = mlxcel_core::from_slice_i32(&[1234], &[1, 1]); + let start = Instant::now(); + let mut prev = mlxcel_core::reshape(&y, &[1, 1]); + for _ in 0..n { + let logits = model.forward(&y, &mut caches, None); + let tok = mlxcel_core::argmax_last_axis(&logits); + let next = mlxcel_core::reshape(&tok, &[1, 1]); + mlxcel_core::async_eval(&next); + let _ = mlxcel_core::item_i32(&prev); + prev = mlxcel_core::reshape(&next, &[1, 1]); + y = next; + } + let _ = mlxcel_core::item_i32(&prev); + start.elapsed().as_secs_f64() * 1000.0 / n as f64 + }; + + // Synchronous width-w forwards; keep one token per round like a rejected + // verify, so the cache grows the way the decode loop grows it. + let sync_width = |w: usize, n: usize| { + let mut caches = fresh(); + let mut times = Vec::new(); + for i in 0..n + 5 { + let tokens: Vec = (0..w) + .map(|j| 2000 + ((i * 7 + j * 13) % 3000) as i32) + .collect(); + let start = Instant::now(); + let input = mlxcel_core::from_slice_i32(&tokens, &[1, w as i32]); + let logits = model.forward(&input, &mut caches, None); + let am = mlxcel_core::argmax_last_axis(&logits); + mlxcel_core::eval(&am); + let _ = mlxcel_core::array_to_raw_bytes(&am); + for c in caches.iter_mut() { + c.trim(w as i32 - 1); + } + if i >= 5 { + times.push(start.elapsed().as_secs_f64() * 1000.0); + } + } + median(times) + }; + + for _ in 0..2 { + let _ = pipelined(20); + for w in 1..=10 { + let _ = sync_width(w, 3); + } + } + let pipe = median((0..5).map(|_| pipelined(60)).collect()); + println!("pipelined step: {pipe:.3} ms"); + for w in 1..=10 { + let t = sync_width(w, 40); + println!( + "sync width {w:2}: {t:.3} ms = {:.2} pipelined steps", + t / pipe + ); + } +} diff --git a/src/lib.rs b/src/lib.rs index 985d29c1f..94e0f7138 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -80,7 +80,7 @@ pub use mlxcel_core::generate::{ }; pub use mlxcel_core::speculative::SpeculativeGenerator; pub use mlxcel_core::speculative::prompt_lookup::{ - PromptLookupConfig, PromptLookupGenerator, prompt_lookup_unsupported_reason, + DraftPolicy, PromptLookupConfig, PromptLookupGenerator, prompt_lookup_unsupported_reason, supports_prompt_lookup, }; #[cfg(feature = "xla-diagnostics")] diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs index 56d90e81f..c70e22b84 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs @@ -53,9 +53,12 @@ //! two and a half one-token steps. On prose, where lookup matches a common //! token and the target rarely agrees, proposing every round made decoding a //! quarter slower than plain decode. [`DraftGovernor`] keeps the block short -//! while recent rounds land few tokens and, after repeated misses, stops -//! proposing for a few rounds (doubling up to a cap) before trying again. -//! Edits keep landing five or six tokens a round, so they keep the full block. +//! while recent rounds land few tokens and stops proposing after repeated +//! misses. Edits keep landing five or six tokens a round, so they keep the full +//! block. How it shortens and when it resumes is a [`DraftPolicy`], because the +//! cost of a verify width depends on the backend's kernels: the +//! [`DraftPolicy::Graded`] rules were tuned on an M4 Pro, the +//! [`DraftPolicy::Gated`] rules on GB10. //! //! ## Acceptance //! @@ -145,6 +148,9 @@ pub struct PromptLookupConfig { /// Shorten or pause proposals while they stop landing, see /// [`DraftGovernor`]. `false` proposes up to `max_draft` every round. pub adaptive: bool, + /// How the adaptive governor sizes blocks and decides when to resume. + /// Ignored when `adaptive` is `false`. + pub policy: DraftPolicy, } impl Default for PromptLookupConfig { @@ -154,6 +160,49 @@ impl Default for PromptLookupConfig { ngram_min: DEFAULT_NGRAM_MIN, max_draft: DEFAULT_MAX_DRAFT, adaptive: true, + policy: DraftPolicy::default_for_backend(), + } + } +} + +/// How [`DraftGovernor`] sizes verify blocks and decides when to resume +/// proposing after a run of misses. +/// +/// The two policies answer one question differently: how much a drafted round +/// that lands nothing costs against a plain step. On an M4 Pro a wide block +/// is cheap, so [`Self::Graded`] grades the block with the recent acceptance +/// and probes again after a short pause. On GB10 (CUDA, affine 4-bit, MLX pin +/// `81ba1c6a`) a synchronous verify forward measured, in pipelined one-token +/// steps for Qwen3-1.7B / Qwen3-8B: width 2 at 1.35 / 1.14, width 3 at +/// 1.62 / 1.41, width 4 at 2.26 / 1.85, widths 5 to 7 rising to 3.59 / 3.28, +/// and width 8 and up flat at 3.34 / 2.51 (`qmv`'s multirow instantiations at +/// 2, 4 and 8 rows, then `qmm_sm80` from 8 rows). Widths 5 to 7 cost more than +/// 8, and a miss also drains the pipeline, so [`Self::Gated`] uses only a +/// narrow or a full block and spends no verify forward to find out whether a +/// copy has resumed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DraftPolicy { + /// PR #2074's rules: the block is twice the recent accepted average plus + /// two, and a pause after `MISSES_BEFORE_COOLDOWN` (3) misses lasts + /// `MIN_COOLDOWN` (4) rounds, doubling up to `MAX_COOLDOWN` (32), before the + /// next drafted round probes again. + Graded, + /// Narrow (`GATED_NARROW_DRAFT`, 2 proposals) until a narrow block lands + /// whole, then full (`max_draft`). While paused it keeps looking up and + /// checks each proposal against the tokens decoding emits next; proposing + /// resumes only once `SHADOW_CONFIRM` (2) proposed tokens in a row came + /// true, and a resumed round that lands nothing pauses again at once. + Gated, +} + +impl DraftPolicy { + /// [`Self::Gated`] on CUDA builds, where it was measured, and + /// [`Self::Graded`] everywhere else, where it was tuned. + pub const fn default_for_backend() -> Self { + if cfg!(feature = "cuda") { + Self::Gated + } else { + Self::Graded } } } @@ -237,13 +286,29 @@ pub fn supports_prompt_lookup(model: &M) -> bool { prompt_lookup_unsupported_reason(model).is_none() } -/// Smallest pause, in rounds, after proposals stop landing. +/// Smallest pause, in rounds, after proposals stop landing ([`DraftPolicy::Graded`]). const MIN_COOLDOWN: usize = 4; -/// Largest pause. Kept short so an edit that writes a new passage and then -/// resumes copying is back on full blocks within a line or two. +/// Largest pause ([`DraftPolicy::Graded`]). Kept short so an edit that writes +/// a new passage and then resumes copying is back on full blocks within a line +/// or two. const MAX_COOLDOWN: usize = 32; /// Consecutive drafted rounds that land nothing before a pause. const MISSES_BEFORE_COOLDOWN: usize = 3; +/// Proposals in a [`DraftPolicy::Gated`] narrow block (verify width 3, the +/// widest block below the 4-row multirow instantiation on CUDA). +const GATED_NARROW_DRAFT: usize = 2; +/// Accepted-proposal average at which [`DraftPolicy::Gated`] switches from the +/// narrow to the full block. A narrow block that lands whole lifts the average +/// from its starting value to exactly this, so one clean narrow round is +/// enough; a full block that lands nothing twice drops it back. +const GATED_FULL_AT: f64 = 1.5; +/// Accepted-proposal average a [`DraftPolicy::Gated`] governor starts from and +/// resumes with: below [`GATED_FULL_AT`], so the first block is narrow. +const GATED_START_EMA: f64 = 1.0; +/// Proposed tokens in a row a paused [`DraftPolicy::Gated`] governor must see +/// come true before it proposes again. One is not enough on prose: the token +/// after a two-token match is often a common one that happens to recur. +pub(crate) const SHADOW_CONFIRM: usize = 2; /// Consecutive rounds without a proposal before the decode loop pipelines. /// /// A pipelined step is submitted before the host knows the token it follows, @@ -256,48 +321,81 @@ const PLAIN_ROUNDS_BEFORE_PIPELINE: usize = 2; /// Decides, round by round, how many lookup tokens to propose. /// /// Tracks an exponential moving average of the proposals each drafted round -/// landed and caps the next block at twice that plus two, so a run of full -/// matches keeps the full block and a run of one-token landings verifies -/// five-wide blocks instead of eight. Lookup landings are all-or-nothing more -/// often than not (a copy either resumes or it does not), so the cap leaves -/// room past the average. After [`MISSES_BEFORE_COOLDOWN`] drafted rounds in -/// a row land nothing, it stops proposing for a cooldown that doubles on every -/// repeat up to [`MAX_COOLDOWN`] and resets once a round lands a token. +/// landed. Under [`DraftPolicy::Graded`] it caps the next block at twice that +/// plus two, so a run of full matches keeps the full block and a run of +/// one-token landings verifies five-wide blocks instead of eight (lookup +/// landings are all-or-nothing more often than not, so the cap leaves room +/// past the average); after [`MISSES_BEFORE_COOLDOWN`] drafted rounds in a row +/// land nothing, it stops proposing for a cooldown that doubles on every repeat +/// up to [`MAX_COOLDOWN`] and resets once a round lands a token. Under +/// [`DraftPolicy::Gated`] the block is narrow or full, and a pause has no +/// length: it ends when [`Self::shadow_confirmed`] reports a proposal that came +/// true without being verified. #[derive(Debug, Clone)] pub(crate) struct DraftGovernor { max_draft: usize, adaptive: bool, + policy: DraftPolicy, accepted_ema: f64, misses: usize, + /// [`DraftPolicy::Graded`]: paused rounds left, and the next pause length. cooldown: usize, next_cooldown: usize, + /// [`DraftPolicy::Gated`]: proposing at all, and whether the next miss + /// pauses outright because this run of drafted rounds has landed nothing + /// since it resumed. + drafting: bool, + probation: bool, } impl DraftGovernor { pub(crate) fn new(config: &PromptLookupConfig) -> Self { + let accepted_ema = match config.policy { + // Optimistic start: the first match in an edit should get the + // whole block. + DraftPolicy::Graded => config.max_draft as f64, + // A first block that misses on prose costs a narrow verify, not + // a full one; an edit earns the full block one round later. + DraftPolicy::Gated => GATED_START_EMA, + }; Self { max_draft: config.max_draft, adaptive: config.adaptive, - // Optimistic start: the first match in an edit should get the - // whole block. - accepted_ema: config.max_draft as f64, + policy: config.policy, + accepted_ema, misses: 0, cooldown: 0, next_cooldown: MIN_COOLDOWN, + drafting: true, + // The first drafted round of a reply has no evidence behind it. + probation: true, } } /// Proposals allowed this round; `0` means run a plain decode step. A - /// paused round counts down the pause. + /// paused [`DraftPolicy::Graded`] round counts down the pause. pub(crate) fn budget(&mut self) -> usize { if !self.adaptive { return self.max_draft; } - if self.cooldown > 0 { - self.cooldown -= 1; - return 0; + match self.policy { + DraftPolicy::Graded => { + if self.cooldown > 0 { + self.cooldown -= 1; + return 0; + } + ((2.0 * self.accepted_ema).round() as usize + 2).clamp(1, self.max_draft) + } + DraftPolicy::Gated => { + if !self.drafting { + 0 + } else if self.accepted_ema >= GATED_FULL_AT { + self.max_draft + } else { + GATED_NARROW_DRAFT.min(self.max_draft) + } + } } - ((2.0 * self.accepted_ema).round() as usize + 2).clamp(1, self.max_draft) } /// Record a drafted round that landed `accepted` proposals. @@ -306,17 +404,78 @@ impl DraftGovernor { return; } self.accepted_ema = 0.5 * self.accepted_ema + 0.5 * accepted as f64; - if accepted > 0 { - self.misses = 0; - self.next_cooldown = MIN_COOLDOWN; - return; + match self.policy { + DraftPolicy::Graded => { + if accepted > 0 { + self.misses = 0; + self.next_cooldown = MIN_COOLDOWN; + return; + } + self.misses += 1; + if self.misses >= MISSES_BEFORE_COOLDOWN { + self.misses = 0; + self.cooldown = self.next_cooldown; + self.next_cooldown = (self.next_cooldown * 2).min(MAX_COOLDOWN); + } + } + DraftPolicy::Gated => { + if accepted > 0 { + self.misses = 0; + self.probation = false; + return; + } + self.misses += 1; + if self.probation || self.misses >= MISSES_BEFORE_COOLDOWN { + self.drafting = false; + self.misses = 0; + } + } } - self.misses += 1; - if self.misses >= MISSES_BEFORE_COOLDOWN { + } + + /// Whether the decode loop should keep looking up while this governor is + /// paused and report proposals that come true via + /// [`Self::shadow_confirmed`]. + pub(crate) fn probes_while_paused(&self) -> bool { + self.adaptive && self.policy == DraftPolicy::Gated && !self.drafting + } + + /// A paused [`DraftPolicy::Gated`] governor saw a lookup proposal come + /// true for [`SHADOW_CONFIRM`] tokens: resume with a narrow block, on + /// probation. + pub(crate) fn shadow_confirmed(&mut self) { + if self.probes_while_paused() { + self.drafting = true; + self.probation = true; self.misses = 0; - self.cooldown = self.next_cooldown; - self.next_cooldown = (self.next_cooldown * 2).min(MAX_COOLDOWN); + self.accepted_ema = GATED_START_EMA; + } + } +} + +/// A lookup proposal made while proposals are paused, tested against the +/// tokens decoding goes on to emit instead of against a verify forward. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct ShadowProbe { + /// Context length when the proposal was made: `proposal[i]` claims the + /// token that lands at `context[start + i]`. + pub(crate) start: usize, + pub(crate) proposal: Vec, +} + +impl ShadowProbe { + /// `Some(true)` once the first [`SHADOW_CONFIRM`] proposed tokens all came + /// true, `Some(false)` at the first one that did not, and `None` while + /// decoding has not emitted that far yet. + pub(crate) fn settle(&self, context: &[i32]) -> Option { + for (i, &claimed) in self.proposal.iter().take(SHADOW_CONFIRM).enumerate() { + match context.get(self.start + i) { + None => return None, + Some(&emitted) if emitted != claimed => return Some(false), + Some(_) => {} + } } + Some(true) } } @@ -433,13 +592,18 @@ pub struct PromptLookupStats { pub rounds: usize, /// Rounds whose lookup found a match and verified a draft block. pub drafted_rounds: usize, - /// One-token rounds run without looking up a proposal, because the - /// [`DraftGovernor`] paused proposals or `max_tokens` left no room. + /// One-token rounds that verified no proposal because the + /// [`DraftGovernor`] paused proposals or `max_tokens` left no room. A + /// paused [`DraftPolicy::Gated`] round still looks up, to test the + /// proposal against what decoding emits next. pub paused_rounds: usize, /// Tokens proposed across all rounds. pub proposed_draft_tokens: usize, /// Proposals the target accepted. pub accepted_draft_tokens: usize, + /// Times a paused [`DraftPolicy::Gated`] governor resumed because a + /// lookup proposal came true without being verified. + pub shadow_confirmations: usize, } impl PromptLookupStats { @@ -459,7 +623,8 @@ impl PromptLookupStats { pub fn summary_line(&self, generated_tokens: usize) -> String { format!( "[Prompt lookup] rounds={} drafted_rounds={} paused_rounds={} proposed={} \ - accepted={} acceptance_rate={:.4} tokens_per_forward={:.4}", + accepted={} acceptance_rate={:.4} tokens_per_forward={:.4} \ + shadow_confirmations={}", self.rounds, self.drafted_rounds, self.paused_rounds, @@ -467,6 +632,7 @@ impl PromptLookupStats { self.accepted_draft_tokens, self.acceptance_rate(), self.tokens_per_forward(generated_tokens), + self.shadow_confirmations, ) } } @@ -701,13 +867,44 @@ impl PromptLookupGenerator { let mut in_flight: Option> = None; // Rounds in a row, this one included, whose lookup proposed nothing. let mut plain_streak = 0usize; + // A proposal looked up while a `DraftPolicy::Gated` governor is + // paused, waiting for decoding to show whether it comes true. Long + // enough to settle even when `max_draft` is shorter. + let mut shadow: Option = None; + let shadow_config = PromptLookupConfig { + max_draft: self.config.max_draft.max(SHADOW_CONFIRM), + ..self.config + }; while !done { + if let Some(confirmed) = shadow + .as_ref() + .and_then(|probe| probe.settle(&self.context)) + { + shadow = None; + if confirmed { + governor.shadow_confirmed(); + self.stats.shadow_confirmations += 1; + } + } // Never propose past `max_tokens`: every accepted proposal is // emitted, and the round also emits one target token. let remaining = max_tokens - self.generated_tokens.len(); let budget = governor.budget().min(remaining.saturating_sub(1)); let draft = if budget == 0 { + // Paused: look up anyway and let the next tokens decoding + // emits say whether proposing would have paid, at the cost of + // a host-side hash probe instead of a verify forward. + if shadow.is_none() && governor.probes_while_paused() { + index.extend(&self.context); + let proposal = index.find(&self.context, &shadow_config); + if proposal.len() >= SHADOW_CONFIRM { + shadow = Some(ShadowProbe { + start: self.context.len(), + proposal, + }); + } + } Vec::new() } else { index.extend(&self.context); diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs index 57c9aea1b..2da778fd6 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs @@ -20,6 +20,14 @@ fn cfg(ngram_max: usize, ngram_min: usize, max_draft: usize) -> PromptLookupConf ngram_min, max_draft, adaptive: true, + policy: DraftPolicy::Graded, + } +} + +fn gated(max_draft: usize) -> PromptLookupConfig { + PromptLookupConfig { + policy: DraftPolicy::Gated, + ..cfg(3, 2, max_draft) } } @@ -97,6 +105,7 @@ fn tokens_per_forward_excludes_the_prefill_token() { paused_rounds: 0, proposed_draft_tokens: 28, accepted_draft_tokens: 20, + shadow_confirmations: 0, }; // 25 generated: 1 from prefill, 24 over 4 rounds. assert!((stats.tokens_per_forward(25) - 6.0).abs() < 1e-9); @@ -438,8 +447,8 @@ fn induction_parity( let prompt = lcg_tokens(seed, 48, INDUCTION_VOCAB as u64); let expected = reference(model, &prompt); assert_eq!(expected.len(), 96, "seed {seed}: the reference ran short"); - for config in [ - PromptLookupConfig::default(), + let bases = [ + cfg(3, 2, 7), cfg(3, 1, 7), cfg(4, 2, 3), // Six-token matches are rare in a random prompt and appear @@ -449,9 +458,15 @@ fn induction_parity( cfg(6, 6, 7), PromptLookupConfig { adaptive: false, - ..PromptLookupConfig::default() + ..cfg(3, 2, 7) }, - ] { + ]; + let policies = [DraftPolicy::Graded, DraftPolicy::Gated]; + for config in policies.iter().flat_map(|&policy| { + bases + .iter() + .map(move |&base| PromptLookupConfig { policy, ..base }) + }) { let mut generator = PromptLookupGenerator::new(config); let (tokens, _) = generator.generate(model, &prompt, 96, sampling); assert_eq!( @@ -465,6 +480,7 @@ fn induction_parity( total.paused_rounds += stats.paused_rounds; total.proposed_draft_tokens += stats.proposed_draft_tokens; total.accepted_draft_tokens += stats.accepted_draft_tokens; + total.shadow_confirmations += stats.shadow_confirmations; } } } @@ -489,6 +505,10 @@ fn assert_both_regimes(totals: &[PromptLookupStats; 2]) { missing.paused_rounds > 0, "the missing model never paused the governor: {missing:?}" ); + assert!( + totals.iter().any(|stats| stats.shadow_confirmations > 0), + "no paused gated governor ever resumed on a confirmed proposal: {totals:?}" + ); } /// The rollback contract: after every verify round the caches must hold @@ -636,3 +656,132 @@ fn warmup_runs_every_verify_width_and_rolls_each_back() { assert_eq!(*model.widths.borrow(), vec![5, 2, 3, 4, 5]); assert_eq!(generator.caches[0].offset, prompt.len() as i32); } + +#[test] +fn gated_governor_starts_narrow_on_probation() { + let mut g = DraftGovernor::new(&gated(7)); + assert_eq!(g.budget(), GATED_NARROW_DRAFT); + // The first round of a reply has nothing behind it: one miss pauses. + g.record(0); + assert_eq!(g.budget(), 0); + assert!(g.probes_while_paused()); +} + +#[test] +fn gated_pause_has_no_length_and_ends_on_a_confirmed_proposal() { + let mut g = DraftGovernor::new(&gated(7)); + g.record(0); + for _ in 0..1000 { + assert_eq!(g.budget(), 0, "a gated pause must not expire on its own"); + } + g.shadow_confirmed(); + assert!(!g.probes_while_paused()); + assert_eq!(g.budget(), GATED_NARROW_DRAFT, "it resumes narrow"); + // Resumed on probation: one more miss pauses again at once. + g.record(0); + assert_eq!(g.budget(), 0); +} + +#[test] +fn gated_block_widens_after_a_whole_narrow_block_and_narrows_after_misses() { + let mut g = DraftGovernor::new(&gated(7)); + g.record(GATED_NARROW_DRAFT); + assert_eq!( + g.budget(), + 7, + "a narrow block that landed whole earns the full one" + ); + g.record(5); + assert_eq!(g.budget(), 7); + // Off probation now, so misses shrink the block before they pause it. + g.record(0); + g.record(0); + assert_eq!(g.budget(), GATED_NARROW_DRAFT); + g.record(0); + assert_eq!(g.budget(), 0, "the third miss in a row pauses"); +} + +#[test] +fn gated_block_never_uses_the_widths_between_narrow_and_full() { + let mut g = DraftGovernor::new(&gated(7)); + for accepted in [2, 7, 3, 1, 0, 1, 2, 6, 0, 4, 1, 2, 2, 0, 5] { + let budget = g.budget(); + assert!( + budget == 0 || budget == GATED_NARROW_DRAFT || budget == 7, + "budget {budget}" + ); + if budget > 0 { + g.record(accepted.min(budget)); + } else { + g.shadow_confirmed(); + } + } +} + +#[test] +fn gated_respects_a_max_draft_below_the_narrow_block() { + let mut g = DraftGovernor::new(&gated(1)); + assert_eq!(g.budget(), 1); + g.record(1); + assert_eq!(g.budget(), 1); +} + +#[test] +fn shadow_confirmation_is_ignored_unless_paused_and_gated() { + let mut g = DraftGovernor::new(&gated(7)); + g.record(2); + g.shadow_confirmed(); + assert_eq!(g.budget(), 7, "a drafting governor keeps its full block"); + + let mut graded = DraftGovernor::new(&cfg(3, 2, 7)); + for _ in 0..MISSES_BEFORE_COOLDOWN { + graded.record(0); + } + assert!(!graded.probes_while_paused()); + graded.shadow_confirmed(); + assert_eq!( + pause_len(&mut graded), + MIN_COOLDOWN, + "graded pauses are unchanged" + ); + + let mut fixed = DraftGovernor::new(&PromptLookupConfig { + adaptive: false, + ..gated(7) + }); + fixed.record(0); + assert_eq!(fixed.budget(), 7); + assert!(!fixed.probes_while_paused()); +} + +#[test] +fn shadow_probe_settles_on_the_first_two_emitted_tokens() { + let probe = ShadowProbe { + start: 3, + proposal: vec![7, 8, 9], + }; + assert_eq!(probe.settle(&[1, 2, 3]), None, "nothing emitted yet"); + assert_eq!(probe.settle(&[1, 2, 3, 7]), None, "one token is not enough"); + assert_eq!(probe.settle(&[1, 2, 3, 7, 8]), Some(true)); + assert_eq!( + probe.settle(&[1, 2, 3, 7, 8, 0]), + Some(true), + "only the first two count" + ); + assert_eq!( + probe.settle(&[1, 2, 3, 6]), + Some(false), + "settles at the first miss" + ); + assert_eq!(probe.settle(&[1, 2, 3, 7, 6]), Some(false)); +} + +#[test] +fn default_policy_follows_the_build() { + let expected = if cfg!(feature = "cuda") { + DraftPolicy::Gated + } else { + DraftPolicy::Graded + }; + assert_eq!(PromptLookupConfig::default().policy, expected); +} diff --git a/src/main.rs b/src/main.rs index 728591cb3..2cd8c1caa 100644 --- a/src/main.rs +++ b/src/main.rs @@ -414,6 +414,43 @@ pub(crate) struct PromptLookupOptions { /// proposals while they stop landing (for A/B measurement) #[arg(long, requires = "prompt_lookup")] pub(crate) prompt_lookup_no_adaptive: bool, + + /// How proposals are shortened and resumed: `gated` (narrow or full + /// blocks, resume only after a looked-up proposal comes true unverified; + /// tuned on CUDA), `graded` (block follows recent acceptance, probe again + /// after a short pause; tuned on Apple Silicon), or `auto` (`gated` on + /// CUDA builds, `graded` otherwise) + #[arg( + long, + value_enum, + requires = "prompt_lookup", + default_value_t = PromptLookupPolicyArg::Auto, + value_name = "POLICY" + )] + pub(crate) prompt_lookup_policy: PromptLookupPolicyArg, +} + +/// `--prompt-lookup-policy` values; see [`mlxcel::DraftPolicy`]. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, clap::ValueEnum)] +pub(crate) enum PromptLookupPolicyArg { + /// The build's default: `gated` on CUDA, `graded` elsewhere. + #[default] + Auto, + /// Block follows recent acceptance; probe again after a short pause. + Graded, + /// Narrow or full blocks; resume only after an unverified proposal came + /// true. + Gated, +} + +impl PromptLookupPolicyArg { + fn resolve(self) -> mlxcel::DraftPolicy { + match self { + Self::Auto => mlxcel::DraftPolicy::default_for_backend(), + Self::Graded => mlxcel::DraftPolicy::Graded, + Self::Gated => mlxcel::DraftPolicy::Gated, + } + } } impl Default for PromptLookupOptions { @@ -425,6 +462,7 @@ impl Default for PromptLookupOptions { prompt_lookup_ngram_min: config.ngram_min, prompt_lookup_max_draft: config.max_draft, prompt_lookup_no_adaptive: !config.adaptive, + prompt_lookup_policy: PromptLookupPolicyArg::Auto, } } } @@ -436,6 +474,7 @@ impl PromptLookupOptions { ngram_min: self.prompt_lookup_ngram_min, max_draft: self.prompt_lookup_max_draft, adaptive: !self.prompt_lookup_no_adaptive, + policy: self.prompt_lookup_policy.resolve(), } } } diff --git a/src/main_tests.rs b/src/main_tests.rs index 95b84a959..81daed508 100644 --- a/src/main_tests.rs +++ b/src/main_tests.rs @@ -128,6 +128,8 @@ fn prompt_lookup_flags_parse_into_the_generator_config() { "--prompt-lookup-max-draft", "5", "--prompt-lookup-no-adaptive", + "--prompt-lookup-policy", + "graded", ]); let Commands::Generate(args) = cli.command else { panic!("expected generate command"); @@ -140,10 +142,47 @@ fn prompt_lookup_flags_parse_into_the_generator_config() { ngram_min: 1, max_draft: 5, adaptive: false, + policy: mlxcel::DraftPolicy::Graded, } ); } +#[test] +fn prompt_lookup_policy_auto_follows_the_build_and_names_override_it() { + let pick = |args: &'static [&'static str]| { + let Commands::Generate(args) = parse_cli(args).command else { + panic!("expected generate command"); + }; + args.prompt_lookup.config().policy + }; + assert_eq!( + pick(&[ + "mlxcel", + "generate", + "-m", + "m", + "-p", + "hi", + "--prompt-lookup" + ]), + mlxcel::DraftPolicy::default_for_backend() + ); + assert_eq!( + pick(&[ + "mlxcel", + "generate", + "-m", + "m", + "-p", + "hi", + "--prompt-lookup", + "--prompt-lookup-policy", + "gated" + ]), + mlxcel::DraftPolicy::Gated + ); +} + /// A tuning flag without `--prompt-lookup` would otherwise be accepted and /// silently do nothing. #[test] @@ -153,6 +192,7 @@ fn prompt_lookup_tuning_flags_require_prompt_lookup() { &["--prompt-lookup-ngram-min", "1"][..], &["--prompt-lookup-max-draft", "5"][..], &["--prompt-lookup-no-adaptive"][..], + &["--prompt-lookup-policy", "gated"][..], ] { let mut args = vec!["mlxcel", "generate", "-m", "models/foo", "-p", "hi"]; args.extend_from_slice(flag); From 43dc07556786bd7e4e217cd13bec9ad1d91667d1 Mon Sep 17 00:00:00 2001 From: Jeongkyu Shin Date: Fri, 2 Oct 2026 14:56:37 +0900 Subject: [PATCH 2/7] update(speculative): pipeline paused gated rounds, pin probe alignment Review fixes for #2092: - A paused `DraftPolicy::Gated` governor now pipelines its plain rounds at once instead of after two synchronous ones: it cannot propose until a shadow probe settles, which takes at least two emitted tokens, so pipelining delays nothing. - Shadow lookups ask for exactly the two tokens a probe settles on. - Rollback parity totals are kept per policy, and Graded must never report a shadow confirmation. - New exact test: a script model copies its prompt, breaks off for one token and resumes; the gated loop must confirm the resumed copy exactly once, pipelined and on the synchronous history-sampler path. A probe recorded one position early or late now fails it. - `DraftPolicy` is `#[non_exhaustive]`, its doc separates the budget from the verify width and states the kernel boundaries once, the end-of-decode trace reports `shadow_confirmations`, and `verify_width_cost` refuses models whose caches cannot be trimmed. Refs #2091 --- examples/verify_width_cost.rs | 7 + .../src/speculative/prompt_lookup.rs | 31 ++-- .../src/speculative/prompt_lookup_tests.rs | 160 +++++++++++++++--- 3 files changed, 159 insertions(+), 39 deletions(-) diff --git a/examples/verify_width_cost.rs b/examples/verify_width_cost.rs index 409143a6c..862a9cac9 100644 --- a/examples/verify_width_cost.rs +++ b/examples/verify_width_cost.rs @@ -40,6 +40,13 @@ fn main() { let path = std::env::args().nth(1).expect("model path"); let (model, _) = mlxcel::load_model(Path::new(&path)).unwrap(); + // The rounds below roll a rejected block back with a cache trim; on a + // model that cannot do that the cache would grow `w` tokens per round and + // skew every ratio without saying so. + assert!( + mlxcel::supports_prompt_lookup(&model), + "{path}: verify blocks cannot be rolled back on this model" + ); let prompt: Vec = (0..300).map(|i| 1000 + (i * 37) % 5000).collect(); let median = |mut v: Vec| { diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs index c70e22b84..4e2e9d0f9 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs @@ -175,12 +175,16 @@ impl Default for PromptLookupConfig { /// `81ba1c6a`) a synchronous verify forward measured, in pipelined one-token /// steps for Qwen3-1.7B / Qwen3-8B: width 2 at 1.35 / 1.14, width 3 at /// 1.62 / 1.41, width 4 at 2.26 / 1.85, widths 5 to 7 rising to 3.59 / 3.28, -/// and width 8 and up flat at 3.34 / 2.51 (`qmv`'s multirow instantiations at -/// 2, 4 and 8 rows, then `qmm_sm80` from 8 rows). Widths 5 to 7 cost more than -/// 8, and a miss also drains the pipeline, so [`Self::Gated`] uses only a -/// narrow or a full block and spends no verify forward to find out whether a -/// copy has resumed. +/// and width 8 and up flat at 3.34 / 2.51 (`examples/verify_width_cost.rs`). +/// Below 8 rows the affine path runs `qmv`'s multirow kernel, instantiated at +/// 2, 4 and 8 accumulator rows, so 5 to 7 rows pay for the 8-row +/// instantiation; from 8 rows it switches to `qmm_sm80`, which is cheaper than +/// that. A miss also drains the pipeline. So [`Self::Gated`] budgets only a +/// narrow or a full block (a lookup near the end of the context, or the +/// `max_tokens` limit, can still return fewer tokens) and spends no verify +/// forward to find out whether a copy has resumed. #[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[non_exhaustive] pub enum DraftPolicy { /// PR #2074's rules: the block is twice the recent accepted average plus /// two, and a pause after `MISSES_BEFORE_COOLDOWN` (3) misses lasts @@ -300,7 +304,8 @@ const GATED_NARROW_DRAFT: usize = 2; /// Accepted-proposal average at which [`DraftPolicy::Gated`] switches from the /// narrow to the full block. A narrow block that lands whole lifts the average /// from its starting value to exactly this, so one clean narrow round is -/// enough; a full block that lands nothing twice drops it back. +/// enough, and the first full block that lands nothing drops it back below; +/// once full blocks have landed several tokens, it takes a run of misses. const GATED_FULL_AT: f64 = 1.5; /// Accepted-proposal average a [`DraftPolicy::Gated`] governor starts from and /// resumes with: below [`GATED_FULL_AT`], so the first block is narrow. @@ -315,7 +320,10 @@ pub(crate) const SHADOW_CONFIRM: usize = 2; /// so it cannot carry a proposal: the first proposal after a pipelined run /// waits one step. Edits that miss for a token or two at each changed field /// (a JSON id, a renamed identifier) would pay that step at every field, so -/// only a run of plain rounds switches to pipelining. +/// only a run of plain rounds switches to pipelining. A paused +/// [`DraftPolicy::Gated`] governor pipelines at once: it cannot propose until +/// a shadow probe settles, which takes at least [`SHADOW_CONFIRM`] emitted +/// tokens, so there is no proposal for a pipelined step to delay. const PLAIN_ROUNDS_BEFORE_PIPELINE: usize = 2; /// Decides, round by round, how many lookup tokens to propose. @@ -868,11 +876,11 @@ impl PromptLookupGenerator { // Rounds in a row, this one included, whose lookup proposed nothing. let mut plain_streak = 0usize; // A proposal looked up while a `DraftPolicy::Gated` governor is - // paused, waiting for decoding to show whether it comes true. Long - // enough to settle even when `max_draft` is shorter. + // paused, waiting for decoding to show whether it comes true. Exactly + // as long as settling reads, whatever `max_draft` is. let mut shadow: Option = None; let shadow_config = PromptLookupConfig { - max_draft: self.config.max_draft.max(SHADOW_CONFIRM), + max_draft: SHADOW_CONFIRM, ..self.config }; @@ -944,7 +952,7 @@ impl PromptLookupGenerator { if draft.is_empty() && pipeline && remaining > 1 - && plain_streak > PLAIN_ROUNDS_BEFORE_PIPELINE + && (plain_streak > PLAIN_ROUNDS_BEFORE_PIPELINE || governor.probes_while_paused()) { let input = ffi::from_slice_i32(&[current_token], &[1, 1]); let logits = model.forward(&input, &mut self.caches, None); @@ -1046,6 +1054,7 @@ impl PromptLookupGenerator { drafted_rounds = self.stats.drafted_rounds, proposed_draft_tokens = self.stats.proposed_draft_tokens, accepted_draft_tokens = self.stats.accepted_draft_tokens, + shadow_confirmations = self.stats.shadow_confirmations, generated_tokens = self.generated_tokens.len(), "prompt-lookup decode finished" ); diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs index 2da778fd6..d43e15042 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs @@ -436,9 +436,10 @@ fn sequential_reference( /// Run prompt lookup on both [`INDUCTION_MODELS`] over several prompts and /// configurations, assert every reply equals `reference(model, prompt)`, and -/// return the summed acceptance accounting per model. +/// return the summed acceptance accounting per model, for `policy` only. fn induction_parity( sampling: &SamplingConfig, + policy: DraftPolicy, reference: impl Fn(&InductionModel, &[i32]) -> Vec, ) -> [PromptLookupStats; 2] { let mut totals = [PromptLookupStats::default(); 2]; @@ -461,12 +462,10 @@ fn induction_parity( ..cfg(3, 2, 7) }, ]; - let policies = [DraftPolicy::Graded, DraftPolicy::Gated]; - for config in policies.iter().flat_map(|&policy| { - bases - .iter() - .map(move |&base| PromptLookupConfig { policy, ..base }) - }) { + for config in bases + .iter() + .map(|&base| PromptLookupConfig { policy, ..base }) + { let mut generator = PromptLookupGenerator::new(config); let (tokens, _) = generator.generate(model, &prompt, 96, sampling); assert_eq!( @@ -487,9 +486,11 @@ fn induction_parity( totals } -/// Both acceptance regimes showed up: blocks that land and blocks that are -/// trimmed, and governor pauses long enough for the loop to pipeline. -fn assert_both_regimes(totals: &[PromptLookupStats; 2]) { +/// Both acceptance regimes showed up under `policy`: blocks that land and +/// blocks that are trimmed, governor pauses long enough for the loop to +/// pipeline, and, for [`DraftPolicy::Gated`] only, resumes on a confirmed +/// proposal. +fn assert_both_regimes(policy: DraftPolicy, totals: &[PromptLookupStats; 2]) { for stats in totals { assert!( stats.accepted_draft_tokens > 0, @@ -505,10 +506,14 @@ fn assert_both_regimes(totals: &[PromptLookupStats; 2]) { missing.paused_rounds > 0, "the missing model never paused the governor: {missing:?}" ); - assert!( - totals.iter().any(|stats| stats.shadow_confirmations > 0), - "no paused gated governor ever resumed on a confirmed proposal: {totals:?}" - ); + let confirmations: usize = totals.iter().map(|stats| stats.shadow_confirmations).sum(); + match policy { + DraftPolicy::Gated => assert!( + confirmations > 0, + "no paused gated governor ever resumed on a confirmed proposal: {totals:?}" + ), + _ => assert_eq!(confirmations, 0, "only a gated governor probes: {totals:?}"), + } } /// The rollback contract: after every verify round the caches must hold @@ -518,16 +523,19 @@ fn assert_both_regimes(totals: &[PromptLookupStats; 2]) { #[test] fn rollback_keeps_a_cache_dependent_target_identical_to_plain_decoding() { let sampling = SamplingConfig::greedy(); - let totals = induction_parity(&sampling, |model, prompt| { - let plain = crate::generate::CxxGenerator::new(1).generate(model, prompt, 96, &sampling); - assert_eq!( - plain, - sequential_reference(model, prompt, 96, &sampling), - "greedy plain decoding is the sequential reference" - ); - plain - }); - assert_both_regimes(&totals); + for policy in [DraftPolicy::Graded, DraftPolicy::Gated] { + let totals = induction_parity(&sampling, policy, |model, prompt| { + let plain = + crate::generate::CxxGenerator::new(1).generate(model, prompt, 96, &sampling); + assert_eq!( + plain, + sequential_reference(model, prompt, 96, &sampling), + "greedy plain decoding is the sequential reference" + ); + plain + }); + assert_both_regimes(policy, &totals); + } } /// Same contract on the per-position sampler path: a repetition penalty @@ -542,10 +550,12 @@ fn rollback_matches_plain_decoding_under_a_repetition_penalty() { ..SamplingConfig::greedy() }; assert!(sampling.needs_token_history()); - let totals = induction_parity(&sampling, |model, prompt| { - sequential_reference(model, prompt, 96, &sampling) - }); - assert_both_regimes(&totals); + for policy in [DraftPolicy::Graded, DraftPolicy::Gated] { + let totals = induction_parity(&sampling, policy, |model, prompt| { + sequential_reference(model, prompt, 96, &sampling) + }); + assert_both_regimes(policy, &totals); + } } /// A dense-cache target that nonetheless owns its sequence state, the shape @@ -785,3 +795,97 @@ fn default_policy_follows_the_build() { }; assert_eq!(PromptLookupConfig::default().policy, expected); } + +/// Vocabulary of [`ScriptModel`]. +const SCRIPT_VOCAB: usize = 1024; + +/// Target that emits a fixed script by absolute position: the logits after +/// position `p` peak at `script[p + 1]`. Every token is unique except where +/// the script copies its own prompt, so a lookup proposal is right exactly +/// when it is aligned with the copy. +struct ScriptModel { + script: Vec, +} + +impl LanguageModel for ScriptModel { + fn forward( + &self, + input_ids: &ffi::MlxArray, + caches: &mut [KVCache], + _mask: Option<&ffi::MlxArray>, + ) -> UniquePtr { + let seq_len = ffi::array_shape(input_ids)[1]; + let offset = caches[0].offset as usize; + let kv = || ffi::ones(&[1, 1, seq_len, 1], crate::dtype::FLOAT32); + caches[0].update(kv(), kv()); + let mut logits = vec![0.0f32; seq_len as usize * SCRIPT_VOCAB]; + for row in 0..seq_len as usize { + let next = self.script[offset + row + 1] as usize; + logits[row * SCRIPT_VOCAB + next] = 10.0; + } + ffi::from_slice_f32(&logits, &[1, seq_len, SCRIPT_VOCAB as i32]) + } + + fn make_caches(&self) -> Vec { + vec![KVCache::new()] + } + + fn num_layers(&self) -> usize { + 1 + } + + fn eos_token_ids(&self) -> Vec { + Vec::new() + } +} + +/// A reply that starts copying its prompt, breaks off for one token, and +/// resumes the copy: the first drafted round misses on probation, the +/// governor pauses, and a shadow probe must notice the resumed copy. +/// +/// Prompt `100..=119`; reply `300, 100, 101, 999, 102, 103, ..., 119`. After +/// `100 101` the lookup proposes `102 103`, the target emits `999`, and that +/// probation miss pauses. Two rounds later `102 103` matches again and the +/// probe made there proposes `104 105`, which come true, so drafting resumes +/// and copies the rest. A probe recorded one position early or late compares +/// `104` with `103` or `105` and never confirms. +fn copy_break_copy_script() -> (Vec, usize) { + let prompt: Vec = (100..120).collect(); + let mut script = prompt.clone(); + script.extend([300, 100, 101, 999]); + script.extend(102..120); + // Room for the model to look one position past the last emitted token. + script.push(0); + (script, prompt.len()) +} + +#[test] +fn gated_shadow_probe_confirms_exactly_the_resumed_copy() { + let (script, prompt_len) = copy_break_copy_script(); + let model = ScriptModel { + script: script.clone(), + }; + let expected = &script[prompt_len..script.len() - 1]; + let penalty = SamplingConfig { + repetition_penalty: 1.1, + penalty_last_n: 8, + ..SamplingConfig::greedy() + }; + // Pipelined plain rounds, and the synchronous path a history-reading + // sampler forces. + for sampling in [SamplingConfig::greedy(), penalty] { + let mut generator = PromptLookupGenerator::new(gated(7)); + let (tokens, _) = + generator.generate(&model, &script[..prompt_len], expected.len(), &sampling); + assert_eq!(tokens, expected, "{sampling:?}"); + let stats = generator.stats(); + assert_eq!( + stats.shadow_confirmations, 1, + "the resumed copy is confirmed once: {stats:?}" + ); + assert!( + stats.drafted_rounds >= 3 && stats.accepted_draft_tokens >= 10, + "drafting resumed and copied the rest: {stats:?}" + ); + } +} From 34c627d77cdbc54d7cc39cc24fad68dec03f1c75 Mon Sep 17 00:00:00 2001 From: Jeongkyu Shin Date: Fri, 2 Oct 2026 15:35:39 +0900 Subject: [PATCH 3/7] update(speculative): end gated probation only on a whole narrow block The first GB10 matrix left Qwen3-8B's email at 0.96x of plain decoding with 6 drafted rounds, 3 landed tokens and no pause: rounds that landed one stray token reset the miss count and lifted probation, so the misses between them were never caught. Probation now ends only when a round lands a whole narrow block (2 proposals); one landed token still resets the miss count but leaves probation on, so the next miss pauses. Refs #2091 --- .../src/speculative/prompt_lookup.rs | 15 ++++++++---- .../src/speculative/prompt_lookup_tests.rs | 23 +++++++++++++++++++ 2 files changed, 34 insertions(+), 4 deletions(-) diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs index 4e2e9d0f9..4982e4423 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs @@ -195,7 +195,8 @@ pub enum DraftPolicy { /// whole, then full (`max_draft`). While paused it keeps looking up and /// checks each proposal against the tokens decoding emits next; proposing /// resumes only once `SHADOW_CONFIRM` (2) proposed tokens in a row came - /// true, and a resumed round that lands nothing pauses again at once. + /// true, and until a round lands a whole narrow block, a round that lands + /// nothing pauses again at once. Gated, } @@ -350,8 +351,8 @@ pub(crate) struct DraftGovernor { cooldown: usize, next_cooldown: usize, /// [`DraftPolicy::Gated`]: proposing at all, and whether the next miss - /// pauses outright because this run of drafted rounds has landed nothing - /// since it resumed. + /// pauses outright because no round since the last resume has landed a + /// whole narrow block. drafting: bool, probation: bool, } @@ -427,9 +428,15 @@ impl DraftGovernor { } } DraftPolicy::Gated => { + // Probation ends on a round that lands a whole narrow block: + // one stray token is what prose lands too, and on GB10 a run + // of such rounds (8B email: 6 drafted, 3 tokens landed) cost + // more than it emitted. + if accepted >= GATED_NARROW_DRAFT.min(self.max_draft) { + self.probation = false; + } if accepted > 0 { self.misses = 0; - self.probation = false; return; } self.misses += 1; diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs index d43e15042..28e61cf4b 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs @@ -889,3 +889,26 @@ fn gated_shadow_probe_confirms_exactly_the_resumed_copy() { ); } } + +#[test] +fn gated_probation_ends_only_on_a_whole_narrow_block() { + let mut g = DraftGovernor::new(&gated(7)); + // One stray token is not evidence of a copy: still on probation. + g.record(1); + assert_eq!(g.budget(), GATED_NARROW_DRAFT); + g.record(0); + assert_eq!( + g.budget(), + 0, + "a miss on probation pauses even after a partial round" + ); + + let mut g = DraftGovernor::new(&gated(7)); + g.record(GATED_NARROW_DRAFT); + g.record(0); + assert_eq!( + g.budget(), + GATED_NARROW_DRAFT, + "off probation, one miss narrows the block instead of pausing" + ); +} From 7224525eb94ac0c9dc2f6c1d8e91e0fcc22f05c4 Mon Sep 17 00:00:00 2001 From: Jeongkyu Shin Date: Fri, 2 Oct 2026 16:08:46 +0900 Subject: [PATCH 4/7] update(speculative): earn the gated full block with two clean narrows The second GB10 matrix left Qwen3-8B's email at 0.97x. A round-by-round trace showed why: the reply restates a short fragment of the prompt, one narrow block landed whole, the governor switched to the full block, and the next two full blocks (width 8, about 2.5 pipelined steps on 8B) landed nothing as the fragment ended. `GATED_FULL_AT` rises from 1.5 to 1.75, so two narrow blocks in a row must land whole (four confirmed tokens) before a full block is spent. Edits keep full blocks once they land several tokens, since the average then stays above the threshold through a miss. Refs #2091 --- .../src/speculative/prompt_lookup.rs | 17 ++++++++++------- .../src/speculative/prompt_lookup_tests.rs | 11 +++++++++-- 2 files changed, 19 insertions(+), 9 deletions(-) diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs index 4982e4423..29d4e7d05 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs @@ -191,8 +191,8 @@ pub enum DraftPolicy { /// `MIN_COOLDOWN` (4) rounds, doubling up to `MAX_COOLDOWN` (32), before the /// next drafted round probes again. Graded, - /// Narrow (`GATED_NARROW_DRAFT`, 2 proposals) until a narrow block lands - /// whole, then full (`max_draft`). While paused it keeps looking up and + /// Narrow (`GATED_NARROW_DRAFT`, 2 proposals) until two narrow blocks in a + /// row land whole, then full (`max_draft`). While paused it keeps looking up and /// checks each proposal against the tokens decoding emits next; proposing /// resumes only once `SHADOW_CONFIRM` (2) proposed tokens in a row came /// true, and until a round lands a whole narrow block, a round that lands @@ -303,11 +303,14 @@ const MISSES_BEFORE_COOLDOWN: usize = 3; /// widest block below the 4-row multirow instantiation on CUDA). const GATED_NARROW_DRAFT: usize = 2; /// Accepted-proposal average at which [`DraftPolicy::Gated`] switches from the -/// narrow to the full block. A narrow block that lands whole lifts the average -/// from its starting value to exactly this, so one clean narrow round is -/// enough, and the first full block that lands nothing drops it back below; -/// once full blocks have landed several tokens, it takes a run of misses. -const GATED_FULL_AT: f64 = 1.5; +/// narrow to the full block. From the starting average, two narrow blocks in +/// a row that land whole (four confirmed tokens) reach exactly this; one is +/// not enough, because prose also restates short fragments of the prompt, and +/// on GB10 a full block that misses costs about two narrow ones (Qwen3-8B +/// email: two full blocks landed nothing right after one clean narrow block). +/// Once full blocks have landed several tokens, it takes a run of misses to +/// fall back below. +const GATED_FULL_AT: f64 = 1.75; /// Accepted-proposal average a [`DraftPolicy::Gated`] governor starts from and /// resumes with: below [`GATED_FULL_AT`], so the first block is narrow. const GATED_START_EMA: f64 = 1.0; diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs index 28e61cf4b..204c6edc0 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs @@ -693,13 +693,19 @@ fn gated_pause_has_no_length_and_ends_on_a_confirmed_proposal() { } #[test] -fn gated_block_widens_after_a_whole_narrow_block_and_narrows_after_misses() { +fn gated_block_widens_after_two_whole_narrow_blocks_and_narrows_after_misses() { let mut g = DraftGovernor::new(&gated(7)); g.record(GATED_NARROW_DRAFT); + assert_eq!( + g.budget(), + GATED_NARROW_DRAFT, + "one clean narrow block is not enough: prose restates short fragments" + ); + g.record(GATED_NARROW_DRAFT); assert_eq!( g.budget(), 7, - "a narrow block that landed whole earns the full one" + "two narrow blocks in a row that landed whole earn the full one" ); g.record(5); assert_eq!(g.budget(), 7); @@ -740,6 +746,7 @@ fn gated_respects_a_max_draft_below_the_narrow_block() { fn shadow_confirmation_is_ignored_unless_paused_and_gated() { let mut g = DraftGovernor::new(&gated(7)); g.record(2); + g.record(2); g.shadow_confirmed(); assert_eq!(g.budget(), 7, "a drafting governor keeps its full block"); From d678c940d4e640da8037c8dc8d393f5c38f572f1 Mon Sep 17 00:00:00 2001 From: Jeongkyu Shin Date: Fri, 2 Oct 2026 16:39:31 +0900 Subject: [PATCH 5/7] update(speculative): restore the gated full block after one clean narrow Reverts 7224525e. On the third GB10 matrix the higher `GATED_FULL_AT` left Qwen3-8B's email where it was (0.97x, graded 0.97x) and changed the verify schedule enough that the Qwen3-1.7B and Qwen3-4B emails, byte-identical to plain decoding at 1.5 and under Graded, diverged from it (near-tie flips in the multi-token verify). No measured gain, a lost parity criterion: back to 1.5. Refs #2091 --- .../src/speculative/prompt_lookup.rs | 17 +++++++---------- .../src/speculative/prompt_lookup_tests.rs | 11 ++--------- 2 files changed, 9 insertions(+), 19 deletions(-) diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs index 29d4e7d05..4982e4423 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs @@ -191,8 +191,8 @@ pub enum DraftPolicy { /// `MIN_COOLDOWN` (4) rounds, doubling up to `MAX_COOLDOWN` (32), before the /// next drafted round probes again. Graded, - /// Narrow (`GATED_NARROW_DRAFT`, 2 proposals) until two narrow blocks in a - /// row land whole, then full (`max_draft`). While paused it keeps looking up and + /// Narrow (`GATED_NARROW_DRAFT`, 2 proposals) until a narrow block lands + /// whole, then full (`max_draft`). While paused it keeps looking up and /// checks each proposal against the tokens decoding emits next; proposing /// resumes only once `SHADOW_CONFIRM` (2) proposed tokens in a row came /// true, and until a round lands a whole narrow block, a round that lands @@ -303,14 +303,11 @@ const MISSES_BEFORE_COOLDOWN: usize = 3; /// widest block below the 4-row multirow instantiation on CUDA). const GATED_NARROW_DRAFT: usize = 2; /// Accepted-proposal average at which [`DraftPolicy::Gated`] switches from the -/// narrow to the full block. From the starting average, two narrow blocks in -/// a row that land whole (four confirmed tokens) reach exactly this; one is -/// not enough, because prose also restates short fragments of the prompt, and -/// on GB10 a full block that misses costs about two narrow ones (Qwen3-8B -/// email: two full blocks landed nothing right after one clean narrow block). -/// Once full blocks have landed several tokens, it takes a run of misses to -/// fall back below. -const GATED_FULL_AT: f64 = 1.75; +/// narrow to the full block. A narrow block that lands whole lifts the average +/// from its starting value to exactly this, so one clean narrow round is +/// enough, and the first full block that lands nothing drops it back below; +/// once full blocks have landed several tokens, it takes a run of misses. +const GATED_FULL_AT: f64 = 1.5; /// Accepted-proposal average a [`DraftPolicy::Gated`] governor starts from and /// resumes with: below [`GATED_FULL_AT`], so the first block is narrow. const GATED_START_EMA: f64 = 1.0; diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs index 204c6edc0..28e61cf4b 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup_tests.rs @@ -693,19 +693,13 @@ fn gated_pause_has_no_length_and_ends_on_a_confirmed_proposal() { } #[test] -fn gated_block_widens_after_two_whole_narrow_blocks_and_narrows_after_misses() { +fn gated_block_widens_after_a_whole_narrow_block_and_narrows_after_misses() { let mut g = DraftGovernor::new(&gated(7)); g.record(GATED_NARROW_DRAFT); - assert_eq!( - g.budget(), - GATED_NARROW_DRAFT, - "one clean narrow block is not enough: prose restates short fragments" - ); - g.record(GATED_NARROW_DRAFT); assert_eq!( g.budget(), 7, - "two narrow blocks in a row that landed whole earn the full one" + "a narrow block that landed whole earns the full one" ); g.record(5); assert_eq!(g.budget(), 7); @@ -746,7 +740,6 @@ fn gated_respects_a_max_draft_below_the_narrow_block() { fn shadow_confirmation_is_ignored_unless_paused_and_gated() { let mut g = DraftGovernor::new(&gated(7)); g.record(2); - g.record(2); g.shadow_confirmed(); assert_eq!(g.budget(), 7, "a drafting governor keeps its full block"); From 51edf12ce94b6bedac5bf4af67c1d79018209bd2 Mon Sep 17 00:00:00 2001 From: Jeongkyu Shin Date: Fri, 2 Oct 2026 16:46:14 +0900 Subject: [PATCH 6/7] docs(benchmarks): record the prompt-lookup policy matrix on GB10 Adds `docs/benchmark_results/prompt-lookup-governor-gb10-2026-10-02.md` with the interleaved harness, prompts, raw results of all three matrices, the load log, and the per-width verify cost on four models, and links it from `docs/benchmarks.md`. `DraftPolicy`'s doc now quotes the committed release-build ratios instead of the earlier test-fast probe. Final matrix (`34c627d7`, greedy, median of 3, plain/graded/gated interleaved): story 0.98x to 1.00x on all four models (graded 0.83x to 0.93x); write 0.98x to 1.02x except Qwen3-8B at 0.97x (graded 0.98x in the same run); edit and summary gains kept within 4% of graded or raised; every reply graded keeps byte-identical to plain decoding stays identical under gated. Refs #2091 --- .../harness/ab.py | 86 ++ .../harness/prompts/edit_fn.txt | 27 + .../harness/prompts/edit_json.txt | 9 + .../harness/prompts/story.txt | 1 + .../harness/prompts/summary.txt | 9 + .../harness/prompts/write.txt | 1 + .../results.json | 1142 +++++++++++++++++ .../run1-probation-on-any-token.jsonl | 20 + .../run2-final.jsonl | 20 + .../run2-load.log | 38 + .../run3-full-at-1.75.jsonl | 20 + .../verify_width_cost.txt | 49 + .../prompt-lookup-governor-gb10-2026-10-02.md | 121 ++ docs/benchmarks.md | 8 + .../src/speculative/prompt_lookup.rs | 22 +- 15 files changed, 1564 insertions(+), 9 deletions(-) create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/ab.py create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/edit_fn.txt create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/edit_json.txt create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/story.txt create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/summary.txt create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/write.txt create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/results.json create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run1-probation-on-any-token.jsonl create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run2-final.jsonl create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run2-load.log create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run3-full-at-1.75.jsonl create mode 100644 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/verify_width_cost.txt create mode 100644 docs/benchmark_results/prompt-lookup-governor-gb10-2026-10-02.md diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/ab.py b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/ab.py new file mode 100644 index 000000000..9e52e0888 --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/ab.py @@ -0,0 +1,86 @@ +#!/usr/bin/env python3 +"""Interleaved A/B for issue #2091: plain decoding against prompt lookup arms. + +Run from the repository root. Each repetition runs every arm back to back +(round-robin), so slow drift in machine load hits all arms alike. The rate is +the CLI's whole-call `[Generated N tokens in Xs = Y tok/s]`, prefill +included, after `mlxcel generate`'s own in-process warmup. Writes one JSON +line per case to stdout, the first repetition's reply text per arm for the +greedy parity check, and `results.json`. + +Environment: + MLXCEL_BIN binary under test (default target/release/mlxcel) + MLXCEL_MODELS_ROOT checkpoint directory (default models/mlx) + CASES model:prompt:max_tokens,... (prompts/ holds the texts) + ARMS JSON {arm: [extra args] or null for plain} + REPS repetitions per arm (default 3) + BENCH_OUT output directory +""" +import json, os, re, statistics, subprocess, sys, time +from pathlib import Path + +HERE = Path(__file__).resolve().parent +BIN = os.environ.get("MLXCEL_BIN", "target/release/mlxcel") +MODELS_ROOT = Path(os.environ.get("MLXCEL_MODELS_ROOT", "models/mlx")) +REPS = int(os.environ.get("REPS", "3")) +OUT = Path(os.environ.get("BENCH_OUT", "out-ab")) +GEN_RE = re.compile(r"\[Generated (\d+) tokens in ([\d.]+)s = ([\d.]+) tok/s\]") +PL_RE = re.compile(r"\[Prompt lookup\] (.*)") + +# name -> extra args (None = plain decoding) +ARMS = json.loads(os.environ.get("ARMS", json.dumps({ + "plain": None, + "graded": ["--prompt-lookup", "--prompt-lookup-policy", "graded"], + "gated": ["--prompt-lookup", "--prompt-lookup-policy", "gated"], +}))) +CASES = [c.split(":") for c in os.environ.get("CASES", "qwen3-1.7b-4bit:write:400").split(",")] + + +def prompt_text(model, name): + text = (HERE / "prompts" / f"{name}.txt").read_text().strip() + return text + " /no_think" if model.startswith("qwen3") else text + + +def run(model, prompt, n, extra): + cmd = [BIN, "generate", "-m", str(MODELS_ROOT / model), "-p", prompt_text(model, prompt), "-n", n, "--temp", "0"] + (extra or []) + p = subprocess.run(cmd, capture_output=True, text=True) + if p.returncode != 0: + raise RuntimeError(f"{cmd}\n{p.stderr[-1500:]}") + m = GEN_RE.search(p.stdout) + st = p.stdout.find("Generating...\n") + body = p.stdout[st + 14 if st >= 0 else 0 : m.start()] + pl = PL_RE.search(p.stderr) + stats = dict(kv.split("=") for kv in pl.group(1).split()) if pl else {} + return float(m.group(3)), int(m.group(1)), body, stats + + +def main(): + OUT.mkdir(exist_ok=True) + results = [] + for model, prompt, n in CASES: + rates = {a: [] for a in ARMS} + info = {} + for rep in range(REPS): + for arm, extra in ARMS.items(): + r, toks, body, stats = run(model, prompt, n, extra) + rates[arm].append(r) + if rep == 0: + info[arm] = {"tokens": toks, "stats": stats} + (OUT / f"{model}.{prompt}.{arm}.txt").write_text(body) + row = {"model": model, "prompt": prompt, "n": n} + base = statistics.median(rates["plain"]) if "plain" in rates else None + for arm in ARMS: + med = statistics.median(rates[arm]) + row[arm] = {"median": med, "rates": rates[arm], **info[arm]} + if base and arm != "plain": + row[arm]["x"] = med / base + a = (OUT / f"{model}.{prompt}.plain.txt").read_text() + b = (OUT / f"{model}.{prompt}.{arm}.txt").read_text() + row[arm]["parity"] = "identical" if a == b else f"diverges@{next((i for i,(x,y) in enumerate(zip(a,b)) if x!=y), min(len(a),len(b)))}" + results.append(row) + print(json.dumps(row), flush=True) + (OUT / "results.json").write_text(json.dumps(results, indent=1)) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/edit_fn.txt b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/edit_fn.txt new file mode 100644 index 000000000..f4ac24531 --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/edit_fn.txt @@ -0,0 +1,27 @@ +Rename the variable `total` to `running_sum` everywhere in this Python function. Reply with the complete updated function only, no explanation. + +```python +def summarize_orders(orders, tax_rate=0.08, discount_threshold=100.0): + total = 0.0 + count = 0 + skipped = [] + for order in orders: + if order.get("status") != "paid": + skipped.append(order["id"]) + continue + amount = float(order["amount"]) + if amount > discount_threshold: + amount *= 0.95 + total += amount + count += 1 + tax = total * tax_rate + total_with_tax = total + tax + average = total / count if count else 0.0 + return { + "total": round(total, 2), + "tax": round(tax, 2), + "total_with_tax": round(total_with_tax, 2), + "average": round(average, 2), + "skipped": skipped, + } +``` diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/edit_json.txt b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/edit_json.txt new file mode 100644 index 000000000..e6b0b34cd --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/edit_json.txt @@ -0,0 +1,9 @@ +Add a field "active": true to every object in this JSON array, right after the "email" field. Reply with the complete updated JSON array only, no explanation. + +[ + {"id": 101, "name": "Alice Kim", "email": "alice.kim@example.com", "role": "engineer", "team": "platform", "joined": "2021-03-14"}, + {"id": 102, "name": "Brian Lee", "email": "brian.lee@example.com", "role": "designer", "team": "product", "joined": "2022-07-01"}, + {"id": 103, "name": "Chloe Park", "email": "chloe.park@example.com", "role": "manager", "team": "platform", "joined": "2019-11-20"}, + {"id": 104, "name": "Daniel Choi", "email": "daniel.choi@example.com", "role": "engineer", "team": "infra", "joined": "2023-01-09"}, + {"id": 105, "name": "Eunji Han", "email": "eunji.han@example.com", "role": "analyst", "team": "data", "joined": "2020-05-27"} +] diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/story.txt b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/story.txt new file mode 100644 index 000000000..9faab1460 --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/story.txt @@ -0,0 +1 @@ +Write a short story of about 600 words about a lighthouse keeper who finds a message in a bottle. Use vivid but plain language. diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/summary.txt b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/summary.txt new file mode 100644 index 000000000..f0b6c332e --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/summary.txt @@ -0,0 +1,9 @@ +Summarize these meeting notes as a short bulleted list. Reuse the wording of the notes where you can. + +Meeting notes, platform team weekly sync: +- The staging cluster migration to the new node pool finished on Tuesday; two services still pin the old image tag and need a redeploy before Friday. +- Latency on the search endpoint regressed by about 30 percent after the cache change; Daniel will roll back the cache change and open a ticket to profile the cold path. +- The on-call rotation for next month is published; Chloe asked everyone to confirm their shifts by Wednesday. +- Budget review: the GPU reservation is under-used at night, so we will try scheduling batch evaluation jobs between midnight and six in the morning. +- Hiring: two onsite interviews for the infrastructure engineer role are scheduled for next week, and Brian will prepare the system design question. +- Action items: redeploy the two pinned services, roll back the cache change, confirm on-call shifts, draft the night batch schedule, prepare the interview question. diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/write.txt b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/write.txt new file mode 100644 index 000000000..16a2bb881 --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/prompts/write.txt @@ -0,0 +1 @@ +Write a short, friendly email to a colleague asking to move our Thursday planning meeting to Friday afternoon, because I have a customer visit on Thursday. Keep it under 150 words. diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/results.json b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/results.json new file mode 100644 index 000000000..6df6c965e --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/results.json @@ -0,0 +1,1142 @@ +[ + { + "model": "qwen3-1.7b-4bit", + "prompt": "edit_fn", + "n": "400", + "plain": { + "median": 179.62, + "rates": [ + 179.62, + 179.63, + 177.86 + ], + "tokens": 210, + "stats": {} + }, + "graded": { + "median": 209.04, + "rates": [ + 209.04, + 206.65, + 216.54 + ], + "tokens": 210, + "stats": { + "rounds": "63", + "drafted_rounds": "41", + "paused_rounds": "0", + "proposed": "259", + "accepted": "147", + "acceptance_rate": "0.5676", + "tokens_per_forward": "3.3175", + "shadow_confirmations": "0" + }, + "x": 1.1637902238058122, + "parity": "identical" + }, + "gated": { + "median": 218.28, + "rates": [ + 218.28, + 223.95, + 217.41 + ], + "tokens": 210, + "stats": { + "rounds": "63", + "drafted_rounds": "41", + "paused_rounds": "0", + "proposed": "252", + "accepted": "147", + "acceptance_rate": "0.5833", + "tokens_per_forward": "3.3175", + "shadow_confirmations": "0" + }, + "x": 1.2152321567754147, + "parity": "identical" + } + }, + { + "model": "qwen3-1.7b-4bit", + "prompt": "edit_json", + "n": "400", + "plain": { + "median": 178.32, + "rates": [ + 177.55, + 178.32, + 179.76 + ], + "tokens": 309, + "stats": {} + }, + "graded": { + "median": 247.8, + "rates": [ + 247.8, + 241.83, + 248.24 + ], + "tokens": 309, + "stats": { + "rounds": "66", + "drafted_rounds": "51", + "paused_rounds": "0", + "proposed": "354", + "accepted": "243", + "acceptance_rate": "0.6864", + "tokens_per_forward": "4.6667", + "shadow_confirmations": "0" + }, + "x": 1.3896366083445493, + "parity": "identical" + }, + "gated": { + "median": 251.13, + "rates": [ + 251.13, + 249.62, + 256.75 + ], + "tokens": 309, + "stats": { + "rounds": "66", + "drafted_rounds": "51", + "paused_rounds": "0", + "proposed": "352", + "accepted": "243", + "acceptance_rate": "0.6903", + "tokens_per_forward": "4.6667", + "shadow_confirmations": "0" + }, + "x": 1.4083109017496636, + "parity": "identical" + } + }, + { + "model": "qwen3-1.7b-4bit", + "prompt": "summary", + "n": "400", + "plain": { + "median": 167.02, + "rates": [ + 167.02, + 165.99, + 167.77 + ], + "tokens": 115, + "stats": {} + }, + "graded": { + "median": 136.28, + "rates": [ + 134.71, + 136.28, + 137.93 + ], + "tokens": 115, + "stats": { + "rounds": "99", + "drafted_rounds": "24", + "paused_rounds": "8", + "proposed": "95", + "accepted": "16", + "acceptance_rate": "0.1684", + "tokens_per_forward": "1.1515", + "shadow_confirmations": "0" + }, + "x": 0.8159501856065141, + "parity": "identical" + }, + "gated": { + "median": 166.14, + "rates": [ + 166.14, + 166.1, + 168.48 + ], + "tokens": 115, + "stats": { + "rounds": "115", + "drafted_rounds": "4", + "paused_rounds": "101", + "proposed": "8", + "accepted": "1", + "acceptance_rate": "0.1250", + "tokens_per_forward": "0.9913", + "shadow_confirmations": "2" + }, + "x": 0.99473116991977, + "parity": "identical" + } + }, + { + "model": "qwen3-1.7b-4bit", + "prompt": "write", + "n": "400", + "plain": { + "median": 181.93, + "rates": [ + 181.93, + 180.03, + 182.23 + ], + "tokens": 72, + "stats": {} + }, + "graded": { + "median": 194.73, + "rates": [ + 194.73, + 197.39, + 193.45 + ], + "tokens": 72, + "stats": { + "rounds": "62", + "drafted_rounds": "2", + "paused_rounds": "0", + "proposed": "14", + "accepted": "11", + "acceptance_rate": "0.7857", + "tokens_per_forward": "1.1452", + "shadow_confirmations": "0" + }, + "x": 1.0703567306106743, + "parity": "identical" + }, + "gated": { + "median": 184.94, + "rates": [ + 184.94, + 185.79, + 182.31 + ], + "tokens": 72, + "stats": { + "rounds": "64", + "drafted_rounds": "4", + "paused_rounds": "0", + "proposed": "18", + "accepted": "9", + "acceptance_rate": "0.5000", + "tokens_per_forward": "1.1094", + "shadow_confirmations": "0" + }, + "x": 1.0165448249326663, + "parity": "identical" + } + }, + { + "model": "qwen3-1.7b-4bit", + "prompt": "story", + "n": "800", + "plain": { + "median": 187.22, + "rates": [ + 187.64, + 185.19, + 187.22 + ], + "tokens": 451, + "stats": {} + }, + "graded": { + "median": 156.3, + "rates": [ + 156.18, + 156.85, + 156.3 + ], + "tokens": 328, + "stats": { + "rounds": "323", + "drafted_rounds": "30", + "paused_rounds": "32", + "proposed": "89", + "accepted": "6", + "acceptance_rate": "0.0674", + "tokens_per_forward": "1.0124", + "shadow_confirmations": "0" + }, + "x": 0.8348467044119219, + "parity": "diverges@536" + }, + "gated": { + "median": 184.01, + "rates": [ + 178.69, + 185.86, + 184.01 + ], + "tokens": 501, + "stats": { + "rounds": "492", + "drafted_rounds": "18", + "paused_rounds": "407", + "proposed": "51", + "accepted": "9", + "acceptance_rate": "0.1765", + "tokens_per_forward": "1.0163", + "shadow_confirmations": "10" + }, + "x": 0.9828543958978742, + "parity": "diverges@1149" + } + }, + { + "model": "qwen3-4b-4bit", + "prompt": "edit_fn", + "n": "400", + "plain": { + "median": 83.22, + "rates": [ + 83.22, + 83.0, + 83.59 + ], + "tokens": 207, + "stats": {} + }, + "graded": { + "median": 105.96, + "rates": [ + 106.67, + 103.18, + 105.96 + ], + "tokens": 207, + "stats": { + "rounds": "57", + "drafted_rounds": "36", + "paused_rounds": "0", + "proposed": "243", + "accepted": "150", + "acceptance_rate": "0.6173", + "tokens_per_forward": "3.6140", + "shadow_confirmations": "0" + }, + "x": 1.2732516222062005, + "parity": "identical" + }, + "gated": { + "median": 115.49, + "rates": [ + 115.49, + 110.35, + 117.26 + ], + "tokens": 207, + "stats": { + "rounds": "55", + "drafted_rounds": "34", + "paused_rounds": "0", + "proposed": "233", + "accepted": "152", + "acceptance_rate": "0.6524", + "tokens_per_forward": "3.7455", + "shadow_confirmations": "0" + }, + "x": 1.3877673636145158, + "parity": "identical" + } + }, + { + "model": "qwen3-4b-4bit", + "prompt": "edit_json", + "n": "400", + "plain": { + "median": 82.75, + "rates": [ + 83.61, + 82.35, + 82.75 + ], + "tokens": 308, + "stats": {} + }, + "graded": { + "median": 141.49, + "rates": [ + 141.49, + 142.17, + 134.76 + ], + "tokens": 309, + "stats": { + "rounds": "52", + "drafted_rounds": "45", + "paused_rounds": "0", + "proposed": "313", + "accepted": "257", + "acceptance_rate": "0.8211", + "tokens_per_forward": "5.9231", + "shadow_confirmations": "0" + }, + "x": 1.7098489425981873, + "parity": "diverges@1346" + }, + "gated": { + "median": 135.47, + "rates": [ + 138.49, + 135.47, + 133.64 + ], + "tokens": 308, + "stats": { + "rounds": "55", + "drafted_rounds": "46", + "paused_rounds": "0", + "proposed": "317", + "accepted": "253", + "acceptance_rate": "0.7981", + "tokens_per_forward": "5.5818", + "shadow_confirmations": "0" + }, + "x": 1.6370996978851964, + "parity": "identical" + } + }, + { + "model": "qwen3-4b-4bit", + "prompt": "summary", + "n": "400", + "plain": { + "median": 82.19, + "rates": [ + 81.93, + 82.72, + 82.19 + ], + "tokens": 185, + "stats": {} + }, + "graded": { + "median": 113.66, + "rates": [ + 114.68, + 112.3, + 113.66 + ], + "tokens": 185, + "stats": { + "rounds": "44", + "drafted_rounds": "31", + "paused_rounds": "0", + "proposed": "212", + "accepted": "141", + "acceptance_rate": "0.6651", + "tokens_per_forward": "4.1818", + "shadow_confirmations": "0" + }, + "x": 1.3828932960214138, + "parity": "identical" + }, + "gated": { + "median": 116.83, + "rates": [ + 116.83, + 115.47, + 117.44 + ], + "tokens": 185, + "stats": { + "rounds": "43", + "drafted_rounds": "30", + "paused_rounds": "0", + "proposed": "200", + "accepted": "142", + "acceptance_rate": "0.7100", + "tokens_per_forward": "4.2791", + "shadow_confirmations": "0" + }, + "x": 1.4214624650200756, + "parity": "identical" + } + }, + { + "model": "qwen3-4b-4bit", + "prompt": "write", + "n": "400", + "plain": { + "median": 85.9, + "rates": [ + 85.9, + 86.44, + 85.56 + ], + "tokens": 140, + "stats": {} + }, + "graded": { + "median": 80.98, + "rates": [ + 80.98, + 80.76, + 81.37 + ], + "tokens": 140, + "stats": { + "rounds": "138", + "drafted_rounds": "3", + "paused_rounds": "0", + "proposed": "21", + "accepted": "3", + "acceptance_rate": "0.1429", + "tokens_per_forward": "1.0072", + "shadow_confirmations": "0" + }, + "x": 0.9427240977881257, + "parity": "identical" + }, + "gated": { + "median": 84.17, + "rates": [ + 84.35, + 83.88, + 84.17 + ], + "tokens": 140, + "stats": { + "rounds": "138", + "drafted_rounds": "3", + "paused_rounds": "0", + "proposed": "11", + "accepted": "3", + "acceptance_rate": "0.2727", + "tokens_per_forward": "1.0072", + "shadow_confirmations": "0" + }, + "x": 0.979860302677532, + "parity": "identical" + } + }, + { + "model": "qwen3-4b-4bit", + "prompt": "story", + "n": "800", + "plain": { + "median": 85.41, + "rates": [ + 85.44, + 83.6, + 85.41 + ], + "tokens": 792, + "stats": {} + }, + "graded": { + "median": 74.35, + "rates": [ + 74.14, + 74.35, + 75.27 + ], + "tokens": 730, + "stats": { + "rounds": "715", + "drafted_rounds": "70", + "paused_rounds": "152", + "proposed": "187", + "accepted": "15", + "acceptance_rate": "0.0802", + "tokens_per_forward": "1.0196", + "shadow_confirmations": "0" + }, + "x": 0.8705069663973773, + "parity": "diverges@659" + }, + "gated": { + "median": 83.46, + "rates": [ + 83.46, + 82.95, + 84.27 + ], + "tokens": 796, + "stats": { + "rounds": "794", + "drafted_rounds": "12", + "paused_rounds": "689", + "proposed": "29", + "accepted": "3", + "acceptance_rate": "0.1034", + "tokens_per_forward": "1.0013", + "shadow_confirmations": "7" + }, + "x": 0.9771689497716894, + "parity": "diverges@1474" + } + }, + { + "model": "qwen3-8b-4bit", + "prompt": "edit_fn", + "n": "400", + "plain": { + "median": 50.71, + "rates": [ + 50.8, + 50.66, + 50.71 + ], + "tokens": 207, + "stats": {} + }, + "graded": { + "median": 71.69, + "rates": [ + 70.9, + 71.69, + 71.81 + ], + "tokens": 207, + "stats": { + "rounds": "57", + "drafted_rounds": "36", + "paused_rounds": "0", + "proposed": "243", + "accepted": "150", + "acceptance_rate": "0.6173", + "tokens_per_forward": "3.6140", + "shadow_confirmations": "0" + }, + "x": 1.4137251035298757, + "parity": "identical" + }, + "gated": { + "median": 76.38, + "rates": [ + 78.43, + 75.65, + 76.38 + ], + "tokens": 207, + "stats": { + "rounds": "55", + "drafted_rounds": "34", + "paused_rounds": "0", + "proposed": "233", + "accepted": "152", + "acceptance_rate": "0.6524", + "tokens_per_forward": "3.7455", + "shadow_confirmations": "0" + }, + "x": 1.506211792545849, + "parity": "identical" + } + }, + { + "model": "qwen3-8b-4bit", + "prompt": "edit_json", + "n": "400", + "plain": { + "median": 49.99, + "rates": [ + 49.69, + 49.99, + 50.78 + ], + "tokens": 309, + "stats": {} + }, + "graded": { + "median": 95.19, + "rates": [ + 96.72, + 95.19, + 94.09 + ], + "tokens": 309, + "stats": { + "rounds": "52", + "drafted_rounds": "45", + "paused_rounds": "0", + "proposed": "313", + "accepted": "257", + "acceptance_rate": "0.8211", + "tokens_per_forward": "5.9231", + "shadow_confirmations": "0" + }, + "x": 1.9041808361672332, + "parity": "identical" + }, + "gated": { + "median": 95.4, + "rates": [ + 95.56, + 95.4, + 94.46 + ], + "tokens": 309, + "stats": { + "rounds": "53", + "drafted_rounds": "46", + "paused_rounds": "0", + "proposed": "317", + "accepted": "256", + "acceptance_rate": "0.8076", + "tokens_per_forward": "5.8113", + "shadow_confirmations": "0" + }, + "x": 1.908381676335267, + "parity": "identical" + } + }, + { + "model": "qwen3-8b-4bit", + "prompt": "summary", + "n": "400", + "plain": { + "median": 49.64, + "rates": [ + 49.28, + 50.74, + 49.64 + ], + "tokens": 185, + "stats": {} + }, + "graded": { + "median": 78.84, + "rates": [ + 77.43, + 78.84, + 78.92 + ], + "tokens": 185, + "stats": { + "rounds": "44", + "drafted_rounds": "31", + "paused_rounds": "0", + "proposed": "212", + "accepted": "141", + "acceptance_rate": "0.6651", + "tokens_per_forward": "4.1818", + "shadow_confirmations": "0" + }, + "x": 1.5882352941176472, + "parity": "identical" + }, + "gated": { + "median": 80.39, + "rates": [ + 79.67, + 80.39, + 81.32 + ], + "tokens": 185, + "stats": { + "rounds": "43", + "drafted_rounds": "30", + "paused_rounds": "0", + "proposed": "200", + "accepted": "142", + "acceptance_rate": "0.7100", + "tokens_per_forward": "4.2791", + "shadow_confirmations": "0" + }, + "x": 1.619460112812248, + "parity": "identical" + } + }, + { + "model": "qwen3-8b-4bit", + "prompt": "write", + "n": "400", + "plain": { + "median": 50.67, + "rates": [ + 50.67, + 50.63, + 51.16 + ], + "tokens": 141, + "stats": {} + }, + "graded": { + "median": 49.51, + "rates": [ + 49.44, + 49.51, + 49.86 + ], + "tokens": 131, + "stats": { + "rounds": "123", + "drafted_rounds": "4", + "paused_rounds": "0", + "proposed": "28", + "accepted": "9", + "acceptance_rate": "0.3214", + "tokens_per_forward": "1.0569", + "shadow_confirmations": "0" + }, + "x": 0.9771067692914939, + "parity": "diverges@525" + }, + "gated": { + "median": 49.13, + "rates": [ + 48.89, + 49.41, + 49.13 + ], + "tokens": 133, + "stats": { + "rounds": "127", + "drafted_rounds": "6", + "paused_rounds": "0", + "proposed": "27", + "accepted": "7", + "acceptance_rate": "0.2593", + "tokens_per_forward": "1.0394", + "shadow_confirmations": "0" + }, + "x": 0.9696072626800869, + "parity": "diverges@543" + } + }, + { + "model": "qwen3-8b-4bit", + "prompt": "story", + "n": "800", + "plain": { + "median": 52.24, + "rates": [ + 52.34, + 52.05, + 52.24 + ], + "tokens": 644, + "stats": {} + }, + "graded": { + "median": 47.44, + "rates": [ + 47.48, + 46.66, + 47.44 + ], + "tokens": 699, + "stats": { + "rounds": "679", + "drafted_rounds": "54", + "paused_rounds": "76", + "proposed": "165", + "accepted": "21", + "acceptance_rate": "0.1273", + "tokens_per_forward": "1.0280", + "shadow_confirmations": "0" + }, + "x": 0.908116385911179, + "parity": "diverges@475" + }, + "gated": { + "median": 51.14, + "rates": [ + 51.53, + 50.7, + 51.14 + ], + "tokens": 689, + "stats": { + "rounds": "684", + "drafted_rounds": "13", + "paused_rounds": "560", + "proposed": "36", + "accepted": "6", + "acceptance_rate": "0.1667", + "tokens_per_forward": "1.0058", + "shadow_confirmations": "7" + }, + "x": 0.9789433384379785, + "parity": "diverges@1479" + } + }, + { + "model": "meta-llama-3.1-8b-instruct-4bit", + "prompt": "edit_fn", + "n": "400", + "plain": { + "median": 51.14, + "rates": [ + 51.66, + 51.14, + 51.13 + ], + "tokens": 199, + "stats": {} + }, + "graded": { + "median": 78.83, + "rates": [ + 78.52, + 79.61, + 78.83 + ], + "tokens": 199, + "stats": { + "rounds": "51", + "drafted_rounds": "35", + "paused_rounds": "0", + "proposed": "237", + "accepted": "148", + "acceptance_rate": "0.6245", + "tokens_per_forward": "3.8824", + "shadow_confirmations": "0" + }, + "x": 1.541454829878764, + "parity": "identical" + }, + "gated": { + "median": 85.01, + "rates": [ + 85.34, + 84.34, + 85.01 + ], + "tokens": 199, + "stats": { + "rounds": "50", + "drafted_rounds": "34", + "paused_rounds": "0", + "proposed": "228", + "accepted": "149", + "acceptance_rate": "0.6535", + "tokens_per_forward": "3.9600", + "shadow_confirmations": "0" + }, + "x": 1.6622995698083693, + "parity": "identical" + } + }, + { + "model": "meta-llama-3.1-8b-instruct-4bit", + "prompt": "edit_json", + "n": "400", + "plain": { + "median": 51.28, + "rates": [ + 51.2, + 51.39, + 51.28 + ], + "tokens": 275, + "stats": {} + }, + "graded": { + "median": 103.54, + "rates": [ + 103.54, + 102.93, + 103.55 + ], + "tokens": 275, + "stats": { + "rounds": "45", + "drafted_rounds": "43", + "paused_rounds": "0", + "proposed": "298", + "accepted": "230", + "acceptance_rate": "0.7718", + "tokens_per_forward": "6.0889", + "shadow_confirmations": "0" + }, + "x": 2.0191107644305775, + "parity": "identical" + }, + "gated": { + "median": 100.88, + "rates": [ + 102.05, + 100.88, + 99.96 + ], + "tokens": 275, + "stats": { + "rounds": "47", + "drafted_rounds": "45", + "paused_rounds": "0", + "proposed": "305", + "accepted": "228", + "acceptance_rate": "0.7475", + "tokens_per_forward": "5.8298", + "shadow_confirmations": "0" + }, + "x": 1.9672386895475817, + "parity": "identical" + } + }, + { + "model": "meta-llama-3.1-8b-instruct-4bit", + "prompt": "summary", + "n": "400", + "plain": { + "median": 49.8, + "rates": [ + 50.26, + 49.46, + 49.8 + ], + "tokens": 121, + "stats": {} + }, + "graded": { + "median": 46.57, + "rates": [ + 47.03, + 46.57, + 46.21 + ], + "tokens": 121, + "stats": { + "rounds": "83", + "drafted_rounds": "23", + "paused_rounds": "0", + "proposed": "127", + "accepted": "38", + "acceptance_rate": "0.2992", + "tokens_per_forward": "1.4458", + "shadow_confirmations": "0" + }, + "x": 0.935140562248996, + "parity": "identical" + }, + "gated": { + "median": 50.96, + "rates": [ + 51.26, + 50.71, + 50.96 + ], + "tokens": 121, + "stats": { + "rounds": "89", + "drafted_rounds": "29", + "paused_rounds": "0", + "proposed": "93", + "accepted": "32", + "acceptance_rate": "0.3441", + "tokens_per_forward": "1.3483", + "shadow_confirmations": "0" + }, + "x": 1.023293172690763, + "parity": "identical" + } + }, + { + "model": "meta-llama-3.1-8b-instruct-4bit", + "prompt": "write", + "n": "400", + "plain": { + "median": 50.94, + "rates": [ + 50.56, + 50.94, + 51.07 + ], + "tokens": 131, + "stats": {} + }, + "graded": { + "median": 49.14, + "rates": [ + 49.14, + 48.96, + 49.2 + ], + "tokens": 133, + "stats": { + "rounds": "129", + "drafted_rounds": "4", + "paused_rounds": "0", + "proposed": "26", + "accepted": "5", + "acceptance_rate": "0.1923", + "tokens_per_forward": "1.0233", + "shadow_confirmations": "0" + }, + "x": 0.9646643109540637, + "parity": "diverges@606" + }, + "gated": { + "median": 50.31, + "rates": [ + 50.31, + 49.01, + 51.28 + ], + "tokens": 119, + "stats": { + "rounds": "115", + "drafted_rounds": "4", + "paused_rounds": "0", + "proposed": "13", + "accepted": "5", + "acceptance_rate": "0.3846", + "tokens_per_forward": "1.0261", + "shadow_confirmations": "0" + }, + "x": 0.9876325088339224, + "parity": "diverges@560" + } + }, + { + "model": "meta-llama-3.1-8b-instruct-4bit", + "prompt": "story", + "n": "800", + "plain": { + "median": 51.58, + "rates": [ + 51.58, + 51.56, + 52.64 + ], + "tokens": 764, + "stats": {} + }, + "graded": { + "median": 47.92, + "rates": [ + 47.8, + 48.2, + 47.92 + ], + "tokens": 720, + "stats": { + "rounds": "694", + "drafted_rounds": "42", + "paused_rounds": "24", + "proposed": "151", + "accepted": "26", + "acceptance_rate": "0.1722", + "tokens_per_forward": "1.0360", + "shadow_confirmations": "0" + }, + "x": 0.9290422644435828, + "parity": "diverges@668" + }, + "gated": { + "median": 51.7, + "rates": [ + 51.7, + 51.59, + 51.86 + ], + "tokens": 764, + "stats": { + "rounds": "761", + "drafted_rounds": "13", + "paused_rounds": "715", + "proposed": "31", + "accepted": "4", + "acceptance_rate": "0.1290", + "tokens_per_forward": "1.0026", + "shadow_confirmations": "7" + }, + "x": 1.0023264831329974, + "parity": "identical" + } + } +] \ No newline at end of file diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run1-probation-on-any-token.jsonl b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run1-probation-on-any-token.jsonl new file mode 100644 index 000000000..075a2f280 --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run1-probation-on-any-token.jsonl @@ -0,0 +1,20 @@ +{"model": "qwen3-1.7b-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 179.19, "rates": [157.46, 179.19, 179.45], "tokens": 210, "stats": {}}, "graded": {"median": 206.91, "rates": [206.91, 210.59, 199.12], "tokens": 210, "stats": {"rounds": "63", "drafted_rounds": "41", "paused_rounds": "0", "proposed": "259", "accepted": "147", "acceptance_rate": "0.5676", "tokens_per_forward": "3.3175", "shadow_confirmations": "0"}, "x": 1.1546961325966851, "parity": "identical"}, "gated": {"median": 216.81, "rates": [221.61, 216.81, 215.13], "tokens": 210, "stats": {"rounds": "63", "drafted_rounds": "41", "paused_rounds": "0", "proposed": "252", "accepted": "147", "acceptance_rate": "0.5833", "tokens_per_forward": "3.3175", "shadow_confirmations": "0"}, "x": 1.2099447513812156, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 180.52, "rates": [165.1, 181.99, 180.52], "tokens": 309, "stats": {}}, "graded": {"median": 243.78, "rates": [246.31, 228.25, 243.78], "tokens": 309, "stats": {"rounds": "66", "drafted_rounds": "51", "paused_rounds": "0", "proposed": "354", "accepted": "243", "acceptance_rate": "0.6864", "tokens_per_forward": "4.6667", "shadow_confirmations": "0"}, "x": 1.3504320850875249, "parity": "identical"}, "gated": {"median": 241.39, "rates": [241.39, 244.64, 238.92], "tokens": 309, "stats": {"rounds": "66", "drafted_rounds": "51", "paused_rounds": "0", "proposed": "352", "accepted": "243", "acceptance_rate": "0.6903", "tokens_per_forward": "4.6667", "shadow_confirmations": "0"}, "x": 1.3371925548415686, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "summary", "n": "400", "plain": {"median": 168.97, "rates": [168.97, 169.16, 168.58], "tokens": 115, "stats": {}}, "graded": {"median": 134.77, "rates": [136.42, 134.77, 132.33], "tokens": 115, "stats": {"rounds": "99", "drafted_rounds": "24", "paused_rounds": "8", "proposed": "95", "accepted": "16", "acceptance_rate": "0.1684", "tokens_per_forward": "1.1515", "shadow_confirmations": "0"}, "x": 0.7975972066047228, "parity": "identical"}, "gated": {"median": 164.92, "rates": [164.92, 164.87, 166.14], "tokens": 115, "stats": {"rounds": "115", "drafted_rounds": "6", "paused_rounds": "95", "proposed": "12", "accepted": "1", "acceptance_rate": "0.0833", "tokens_per_forward": "0.9913", "shadow_confirmations": "2"}, "x": 0.9760312481505592, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "write", "n": "400", "plain": {"median": 180.43, "rates": [180.36, 180.43, 180.53], "tokens": 72, "stats": {}}, "graded": {"median": 189.85, "rates": [189.55, 189.9, 189.85], "tokens": 72, "stats": {"rounds": "62", "drafted_rounds": "2", "paused_rounds": "0", "proposed": "14", "accepted": "11", "acceptance_rate": "0.7857", "tokens_per_forward": "1.1452", "shadow_confirmations": "0"}, "x": 1.0522086127584105, "parity": "identical"}, "gated": {"median": 182.08, "rates": [180.58, 182.3, 182.08], "tokens": 72, "stats": {"rounds": "64", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "18", "accepted": "9", "acceptance_rate": "0.5000", "tokens_per_forward": "1.1094", "shadow_confirmations": "0"}, "x": 1.0091448207060911, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "story", "n": "800", "plain": {"median": 188.23, "rates": [188.68, 186.12, 188.23], "tokens": 382, "stats": {}}, "graded": {"median": 155.75, "rates": [158.0, 155.75, 154.17], "tokens": 378, "stats": {"rounds": "371", "drafted_rounds": "35", "paused_rounds": "56", "proposed": "97", "accepted": "7", "acceptance_rate": "0.0722", "tokens_per_forward": "1.0162", "shadow_confirmations": "0"}, "x": 0.8274451468947565, "parity": "diverges@536"}, "gated": {"median": 184.47, "rates": [184.47, 183.21, 185.66], "tokens": 424, "stats": {"rounds": "423", "drafted_rounds": "8", "paused_rounds": "357", "proposed": "16", "accepted": "1", "acceptance_rate": "0.0625", "tokens_per_forward": "1.0000", "shadow_confirmations": "5"}, "x": 0.9800244381873241, "parity": "diverges@1149"}} +{"model": "qwen3-4b-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 83.62, "rates": [83.62, 84.11, 83.55], "tokens": 207, "stats": {}}, "graded": {"median": 105.74, "rates": [105.33, 105.74, 106.42], "tokens": 207, "stats": {"rounds": "57", "drafted_rounds": "36", "paused_rounds": "0", "proposed": "243", "accepted": "150", "acceptance_rate": "0.6173", "tokens_per_forward": "3.6140", "shadow_confirmations": "0"}, "x": 1.264530016742406, "parity": "identical"}, "gated": {"median": 106.57, "rates": [106.57, 105.79, 111.12], "tokens": 207, "stats": {"rounds": "55", "drafted_rounds": "34", "paused_rounds": "0", "proposed": "233", "accepted": "152", "acceptance_rate": "0.6524", "tokens_per_forward": "3.7455", "shadow_confirmations": "0"}, "x": 1.2744558718010044, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 83.93, "rates": [84.26, 83.74, 83.93], "tokens": 308, "stats": {}}, "graded": {"median": 132.15, "rates": [131.61, 135.34, 132.15], "tokens": 309, "stats": {"rounds": "52", "drafted_rounds": "45", "paused_rounds": "0", "proposed": "313", "accepted": "257", "acceptance_rate": "0.8211", "tokens_per_forward": "5.9231", "shadow_confirmations": "0"}, "x": 1.5745263910401524, "parity": "diverges@1346"}, "gated": {"median": 129.52, "rates": [129.09, 133.08, 129.52], "tokens": 308, "stats": {"rounds": "55", "drafted_rounds": "46", "paused_rounds": "0", "proposed": "317", "accepted": "253", "acceptance_rate": "0.7981", "tokens_per_forward": "5.5818", "shadow_confirmations": "0"}, "x": 1.5431907541999286, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "summary", "n": "400", "plain": {"median": 83.48, "rates": [83.45, 83.48, 83.74], "tokens": 185, "stats": {}}, "graded": {"median": 112.13, "rates": [112.13, 110.07, 112.87], "tokens": 185, "stats": {"rounds": "44", "drafted_rounds": "31", "paused_rounds": "0", "proposed": "212", "accepted": "141", "acceptance_rate": "0.6651", "tokens_per_forward": "4.1818", "shadow_confirmations": "0"}, "x": 1.3431959750838522, "parity": "identical"}, "gated": {"median": 112.91, "rates": [112.91, 117.14, 111.69], "tokens": 185, "stats": {"rounds": "43", "drafted_rounds": "30", "paused_rounds": "0", "proposed": "200", "accepted": "142", "acceptance_rate": "0.7100", "tokens_per_forward": "4.2791", "shadow_confirmations": "0"}, "x": 1.3525395304264494, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "write", "n": "400", "plain": {"median": 85.96, "rates": [86.51, 85.96, 85.51], "tokens": 140, "stats": {}}, "graded": {"median": 80.59, "rates": [80.59, 80.58, 80.98], "tokens": 140, "stats": {"rounds": "138", "drafted_rounds": "3", "paused_rounds": "0", "proposed": "21", "accepted": "3", "acceptance_rate": "0.1429", "tokens_per_forward": "1.0072", "shadow_confirmations": "0"}, "x": 0.9375290832945558, "parity": "identical"}, "gated": {"median": 84.47, "rates": [84.21, 84.53, 84.47], "tokens": 140, "stats": {"rounds": "138", "drafted_rounds": "3", "paused_rounds": "0", "proposed": "11", "accepted": "3", "acceptance_rate": "0.2727", "tokens_per_forward": "1.0072", "shadow_confirmations": "0"}, "x": 0.9826663564448581, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "story", "n": "800", "plain": {"median": 85.62, "rates": [84.39, 85.86, 85.62], "tokens": 785, "stats": {}}, "graded": {"median": 74.73, "rates": [74.39, 74.73, 75.47], "tokens": 767, "stats": {"rounds": "742", "drafted_rounds": "90", "paused_rounds": "160", "proposed": "249", "accepted": "26", "acceptance_rate": "0.1044", "tokens_per_forward": "1.0323", "shadow_confirmations": "0"}, "x": 0.8728100911002102, "parity": "diverges@659"}, "gated": {"median": 84.77, "rates": [84.84, 84.27, 84.77], "tokens": 796, "stats": {"rounds": "794", "drafted_rounds": "14", "paused_rounds": "679", "proposed": "33", "accepted": "3", "acceptance_rate": "0.0909", "tokens_per_forward": "1.0013", "shadow_confirmations": "7"}, "x": 0.9900724129876196, "parity": "diverges@1700"}} +{"model": "qwen3-8b-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 50.96, "rates": [50.7, 51.06, 50.96], "tokens": 207, "stats": {}}, "graded": {"median": 70.33, "rates": [69.92, 71.35, 70.33], "tokens": 207, "stats": {"rounds": "57", "drafted_rounds": "36", "paused_rounds": "0", "proposed": "243", "accepted": "150", "acceptance_rate": "0.6173", "tokens_per_forward": "3.6140", "shadow_confirmations": "0"}, "x": 1.3801020408163265, "parity": "identical"}, "gated": {"median": 76.0, "rates": [76.0, 76.2, 75.82], "tokens": 207, "stats": {"rounds": "55", "drafted_rounds": "34", "paused_rounds": "0", "proposed": "233", "accepted": "152", "acceptance_rate": "0.6524", "tokens_per_forward": "3.7455", "shadow_confirmations": "0"}, "x": 1.4913657770800628, "parity": "identical"}} +{"model": "qwen3-8b-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 50.95, "rates": [50.98, 50.93, 50.95], "tokens": 309, "stats": {}}, "graded": {"median": 94.77, "rates": [92.95, 94.77, 94.79], "tokens": 309, "stats": {"rounds": "52", "drafted_rounds": "45", "paused_rounds": "0", "proposed": "313", "accepted": "257", "acceptance_rate": "0.8211", "tokens_per_forward": "5.9231", "shadow_confirmations": "0"}, "x": 1.8600588812561334, "parity": "identical"}, "gated": {"median": 93.76, "rates": [95.11, 93.76, 93.51], "tokens": 309, "stats": {"rounds": "53", "drafted_rounds": "46", "paused_rounds": "0", "proposed": "317", "accepted": "256", "acceptance_rate": "0.8076", "tokens_per_forward": "5.8113", "shadow_confirmations": "0"}, "x": 1.840235525024534, "parity": "identical"}} +{"model": "qwen3-8b-4bit", "prompt": "summary", "n": "400", "plain": {"median": 50.79, "rates": [50.75, 50.79, 50.79], "tokens": 185, "stats": {}}, "graded": {"median": 77.13, "rates": [77.03, 78.09, 77.13], "tokens": 185, "stats": {"rounds": "44", "drafted_rounds": "31", "paused_rounds": "0", "proposed": "212", "accepted": "141", "acceptance_rate": "0.6651", "tokens_per_forward": "4.1818", "shadow_confirmations": "0"}, "x": 1.518606024808033, "parity": "identical"}, "gated": {"median": 80.2, "rates": [81.5, 80.2, 80.12], "tokens": 185, "stats": {"rounds": "43", "drafted_rounds": "30", "paused_rounds": "0", "proposed": "200", "accepted": "142", "acceptance_rate": "0.7100", "tokens_per_forward": "4.2791", "shadow_confirmations": "0"}, "x": 1.5790509942902147, "parity": "identical"}} +{"model": "qwen3-8b-4bit", "prompt": "write", "n": "400", "plain": {"median": 51.23, "rates": [51.23, 51.28, 51.08], "tokens": 141, "stats": {}}, "graded": {"median": 49.55, "rates": [49.82, 49.3, 49.55], "tokens": 131, "stats": {"rounds": "123", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "28", "accepted": "9", "acceptance_rate": "0.3214", "tokens_per_forward": "1.0569", "shadow_confirmations": "0"}, "x": 0.9672067148155378, "parity": "diverges@525"}, "gated": {"median": 49.25, "rates": [49.28, 49.11, 49.25], "tokens": 133, "stats": {"rounds": "127", "drafted_rounds": "6", "paused_rounds": "0", "proposed": "27", "accepted": "7", "acceptance_rate": "0.2593", "tokens_per_forward": "1.0394", "shadow_confirmations": "0"}, "x": 0.9613507710325981, "parity": "diverges@543"}} +{"model": "qwen3-8b-4bit", "prompt": "story", "n": "800", "plain": {"median": 52.23, "rates": [52.23, 52.15, 52.25], "tokens": 645, "stats": {}}, "graded": {"median": 47.32, "rates": [47.32, 46.65, 47.4], "tokens": 733, "stats": {"rounds": "712", "drafted_rounds": "59", "paused_rounds": "137", "proposed": "179", "accepted": "22", "acceptance_rate": "0.1229", "tokens_per_forward": "1.0281", "shadow_confirmations": "0"}, "x": 0.9059927244878423, "parity": "diverges@475"}, "gated": {"median": 51.26, "rates": [51.79, 51.25, 51.26], "tokens": 651, "stats": {"rounds": "652", "drafted_rounds": "6", "paused_rounds": "550", "proposed": "12", "accepted": "0", "acceptance_rate": "0.0000", "tokens_per_forward": "0.9969", "shadow_confirmations": "5"}, "x": 0.9814282979130768, "parity": "diverges@1362"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 51.63, "rates": [47.63, 51.64, 51.63], "tokens": 199, "stats": {}}, "graded": {"median": 78.75, "rates": [78.75, 79.76, 78.68], "tokens": 199, "stats": {"rounds": "51", "drafted_rounds": "35", "paused_rounds": "0", "proposed": "237", "accepted": "148", "acceptance_rate": "0.6245", "tokens_per_forward": "3.8824", "shadow_confirmations": "0"}, "x": 1.52527600232423, "parity": "identical"}, "gated": {"median": 84.66, "rates": [84.66, 85.39, 84.02], "tokens": 199, "stats": {"rounds": "50", "drafted_rounds": "34", "paused_rounds": "0", "proposed": "228", "accepted": "149", "acceptance_rate": "0.6535", "tokens_per_forward": "3.9600", "shadow_confirmations": "0"}, "x": 1.639744334689134, "parity": "identical"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 51.2, "rates": [49.73, 51.42, 51.2], "tokens": 275, "stats": {}}, "graded": {"median": 101.38, "rates": [101.72, 100.89, 101.38], "tokens": 275, "stats": {"rounds": "45", "drafted_rounds": "43", "paused_rounds": "0", "proposed": "298", "accepted": "230", "acceptance_rate": "0.7718", "tokens_per_forward": "6.0889", "shadow_confirmations": "0"}, "x": 1.980078125, "parity": "identical"}, "gated": {"median": 99.55, "rates": [99.59, 99.55, 98.32], "tokens": 275, "stats": {"rounds": "47", "drafted_rounds": "45", "paused_rounds": "0", "proposed": "305", "accepted": "228", "acceptance_rate": "0.7475", "tokens_per_forward": "5.8298", "shadow_confirmations": "0"}, "x": 1.9443359374999998, "parity": "identical"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "summary", "n": "400", "plain": {"median": 50.36, "rates": [50.43, 50.36, 50.34], "tokens": 121, "stats": {}}, "graded": {"median": 46.91, "rates": [47.03, 46.91, 46.29], "tokens": 121, "stats": {"rounds": "83", "drafted_rounds": "23", "paused_rounds": "0", "proposed": "127", "accepted": "38", "acceptance_rate": "0.2992", "tokens_per_forward": "1.4458", "shadow_confirmations": "0"}, "x": 0.9314932486100079, "parity": "identical"}, "gated": {"median": 51.32, "rates": [51.32, 51.45, 51.25], "tokens": 121, "stats": {"rounds": "89", "drafted_rounds": "29", "paused_rounds": "0", "proposed": "93", "accepted": "32", "acceptance_rate": "0.3441", "tokens_per_forward": "1.3483", "shadow_confirmations": "0"}, "x": 1.0190627482128674, "parity": "identical"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "write", "n": "400", "plain": {"median": 51.95, "rates": [51.95, 51.98, 51.88], "tokens": 131, "stats": {}}, "graded": {"median": 50.1, "rates": [50.1, 49.93, 50.12], "tokens": 133, "stats": {"rounds": "129", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "26", "accepted": "5", "acceptance_rate": "0.1923", "tokens_per_forward": "1.0233", "shadow_confirmations": "0"}, "x": 0.9643888354186718, "parity": "diverges@606"}, "gated": {"median": 51.78, "rates": [51.67, 51.78, 51.79], "tokens": 119, "stats": {"rounds": "115", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "13", "accepted": "5", "acceptance_rate": "0.3846", "tokens_per_forward": "1.0261", "shadow_confirmations": "0"}, "x": 0.9967276227141482, "parity": "diverges@560"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "story", "n": "800", "plain": {"median": 52.16, "rates": [52.16, 52.85, 52.13], "tokens": 764, "stats": {}}, "graded": {"median": 48.3, "rates": [49.43, 48.3, 47.63], "tokens": 698, "stats": {"rounds": "684", "drafted_rounds": "36", "paused_rounds": "24", "proposed": "119", "accepted": "15", "acceptance_rate": "0.1261", "tokens_per_forward": "1.0190", "shadow_confirmations": "0"}, "x": 0.9259969325153374, "parity": "diverges@668"}, "gated": {"median": 51.88, "rates": [51.88, 51.91, 51.22], "tokens": 764, "stats": {"rounds": "761", "drafted_rounds": "17", "paused_rounds": "673", "proposed": "39", "accepted": "4", "acceptance_rate": "0.1026", "tokens_per_forward": "1.0026", "shadow_confirmations": "7"}, "x": 0.9946319018404909, "parity": "identical"}} diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run2-final.jsonl b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run2-final.jsonl new file mode 100644 index 000000000..7d8b34619 --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run2-final.jsonl @@ -0,0 +1,20 @@ +{"model": "qwen3-1.7b-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 179.62, "rates": [179.62, 179.63, 177.86], "tokens": 210, "stats": {}}, "graded": {"median": 209.04, "rates": [209.04, 206.65, 216.54], "tokens": 210, "stats": {"rounds": "63", "drafted_rounds": "41", "paused_rounds": "0", "proposed": "259", "accepted": "147", "acceptance_rate": "0.5676", "tokens_per_forward": "3.3175", "shadow_confirmations": "0"}, "x": 1.1637902238058122, "parity": "identical"}, "gated": {"median": 218.28, "rates": [218.28, 223.95, 217.41], "tokens": 210, "stats": {"rounds": "63", "drafted_rounds": "41", "paused_rounds": "0", "proposed": "252", "accepted": "147", "acceptance_rate": "0.5833", "tokens_per_forward": "3.3175", "shadow_confirmations": "0"}, "x": 1.2152321567754147, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 178.32, "rates": [177.55, 178.32, 179.76], "tokens": 309, "stats": {}}, "graded": {"median": 247.8, "rates": [247.8, 241.83, 248.24], "tokens": 309, "stats": {"rounds": "66", "drafted_rounds": "51", "paused_rounds": "0", "proposed": "354", "accepted": "243", "acceptance_rate": "0.6864", "tokens_per_forward": "4.6667", "shadow_confirmations": "0"}, "x": 1.3896366083445493, "parity": "identical"}, "gated": {"median": 251.13, "rates": [251.13, 249.62, 256.75], "tokens": 309, "stats": {"rounds": "66", "drafted_rounds": "51", "paused_rounds": "0", "proposed": "352", "accepted": "243", "acceptance_rate": "0.6903", "tokens_per_forward": "4.6667", "shadow_confirmations": "0"}, "x": 1.4083109017496636, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "summary", "n": "400", "plain": {"median": 167.02, "rates": [167.02, 165.99, 167.77], "tokens": 115, "stats": {}}, "graded": {"median": 136.28, "rates": [134.71, 136.28, 137.93], "tokens": 115, "stats": {"rounds": "99", "drafted_rounds": "24", "paused_rounds": "8", "proposed": "95", "accepted": "16", "acceptance_rate": "0.1684", "tokens_per_forward": "1.1515", "shadow_confirmations": "0"}, "x": 0.8159501856065141, "parity": "identical"}, "gated": {"median": 166.14, "rates": [166.14, 166.1, 168.48], "tokens": 115, "stats": {"rounds": "115", "drafted_rounds": "4", "paused_rounds": "101", "proposed": "8", "accepted": "1", "acceptance_rate": "0.1250", "tokens_per_forward": "0.9913", "shadow_confirmations": "2"}, "x": 0.99473116991977, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "write", "n": "400", "plain": {"median": 181.93, "rates": [181.93, 180.03, 182.23], "tokens": 72, "stats": {}}, "graded": {"median": 194.73, "rates": [194.73, 197.39, 193.45], "tokens": 72, "stats": {"rounds": "62", "drafted_rounds": "2", "paused_rounds": "0", "proposed": "14", "accepted": "11", "acceptance_rate": "0.7857", "tokens_per_forward": "1.1452", "shadow_confirmations": "0"}, "x": 1.0703567306106743, "parity": "identical"}, "gated": {"median": 184.94, "rates": [184.94, 185.79, 182.31], "tokens": 72, "stats": {"rounds": "64", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "18", "accepted": "9", "acceptance_rate": "0.5000", "tokens_per_forward": "1.1094", "shadow_confirmations": "0"}, "x": 1.0165448249326663, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "story", "n": "800", "plain": {"median": 187.22, "rates": [187.64, 185.19, 187.22], "tokens": 451, "stats": {}}, "graded": {"median": 156.3, "rates": [156.18, 156.85, 156.3], "tokens": 328, "stats": {"rounds": "323", "drafted_rounds": "30", "paused_rounds": "32", "proposed": "89", "accepted": "6", "acceptance_rate": "0.0674", "tokens_per_forward": "1.0124", "shadow_confirmations": "0"}, "x": 0.8348467044119219, "parity": "diverges@536"}, "gated": {"median": 184.01, "rates": [178.69, 185.86, 184.01], "tokens": 501, "stats": {"rounds": "492", "drafted_rounds": "18", "paused_rounds": "407", "proposed": "51", "accepted": "9", "acceptance_rate": "0.1765", "tokens_per_forward": "1.0163", "shadow_confirmations": "10"}, "x": 0.9828543958978742, "parity": "diverges@1149"}} +{"model": "qwen3-4b-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 83.22, "rates": [83.22, 83.0, 83.59], "tokens": 207, "stats": {}}, "graded": {"median": 105.96, "rates": [106.67, 103.18, 105.96], "tokens": 207, "stats": {"rounds": "57", "drafted_rounds": "36", "paused_rounds": "0", "proposed": "243", "accepted": "150", "acceptance_rate": "0.6173", "tokens_per_forward": "3.6140", "shadow_confirmations": "0"}, "x": 1.2732516222062005, "parity": "identical"}, "gated": {"median": 115.49, "rates": [115.49, 110.35, 117.26], "tokens": 207, "stats": {"rounds": "55", "drafted_rounds": "34", "paused_rounds": "0", "proposed": "233", "accepted": "152", "acceptance_rate": "0.6524", "tokens_per_forward": "3.7455", "shadow_confirmations": "0"}, "x": 1.3877673636145158, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 82.75, "rates": [83.61, 82.35, 82.75], "tokens": 308, "stats": {}}, "graded": {"median": 141.49, "rates": [141.49, 142.17, 134.76], "tokens": 309, "stats": {"rounds": "52", "drafted_rounds": "45", "paused_rounds": "0", "proposed": "313", "accepted": "257", "acceptance_rate": "0.8211", "tokens_per_forward": "5.9231", "shadow_confirmations": "0"}, "x": 1.7098489425981873, "parity": "diverges@1346"}, "gated": {"median": 135.47, "rates": [138.49, 135.47, 133.64], "tokens": 308, "stats": {"rounds": "55", "drafted_rounds": "46", "paused_rounds": "0", "proposed": "317", "accepted": "253", "acceptance_rate": "0.7981", "tokens_per_forward": "5.5818", "shadow_confirmations": "0"}, "x": 1.6370996978851964, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "summary", "n": "400", "plain": {"median": 82.19, "rates": [81.93, 82.72, 82.19], "tokens": 185, "stats": {}}, "graded": {"median": 113.66, "rates": [114.68, 112.3, 113.66], "tokens": 185, "stats": {"rounds": "44", "drafted_rounds": "31", "paused_rounds": "0", "proposed": "212", "accepted": "141", "acceptance_rate": "0.6651", "tokens_per_forward": "4.1818", "shadow_confirmations": "0"}, "x": 1.3828932960214138, "parity": "identical"}, "gated": {"median": 116.83, "rates": [116.83, 115.47, 117.44], "tokens": 185, "stats": {"rounds": "43", "drafted_rounds": "30", "paused_rounds": "0", "proposed": "200", "accepted": "142", "acceptance_rate": "0.7100", "tokens_per_forward": "4.2791", "shadow_confirmations": "0"}, "x": 1.4214624650200756, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "write", "n": "400", "plain": {"median": 85.9, "rates": [85.9, 86.44, 85.56], "tokens": 140, "stats": {}}, "graded": {"median": 80.98, "rates": [80.98, 80.76, 81.37], "tokens": 140, "stats": {"rounds": "138", "drafted_rounds": "3", "paused_rounds": "0", "proposed": "21", "accepted": "3", "acceptance_rate": "0.1429", "tokens_per_forward": "1.0072", "shadow_confirmations": "0"}, "x": 0.9427240977881257, "parity": "identical"}, "gated": {"median": 84.17, "rates": [84.35, 83.88, 84.17], "tokens": 140, "stats": {"rounds": "138", "drafted_rounds": "3", "paused_rounds": "0", "proposed": "11", "accepted": "3", "acceptance_rate": "0.2727", "tokens_per_forward": "1.0072", "shadow_confirmations": "0"}, "x": 0.979860302677532, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "story", "n": "800", "plain": {"median": 85.41, "rates": [85.44, 83.6, 85.41], "tokens": 792, "stats": {}}, "graded": {"median": 74.35, "rates": [74.14, 74.35, 75.27], "tokens": 730, "stats": {"rounds": "715", "drafted_rounds": "70", "paused_rounds": "152", "proposed": "187", "accepted": "15", "acceptance_rate": "0.0802", "tokens_per_forward": "1.0196", "shadow_confirmations": "0"}, "x": 0.8705069663973773, "parity": "diverges@659"}, "gated": {"median": 83.46, "rates": [83.46, 82.95, 84.27], "tokens": 796, "stats": {"rounds": "794", "drafted_rounds": "12", "paused_rounds": "689", "proposed": "29", "accepted": "3", "acceptance_rate": "0.1034", "tokens_per_forward": "1.0013", "shadow_confirmations": "7"}, "x": 0.9771689497716894, "parity": "diverges@1474"}} +{"model": "qwen3-8b-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 50.71, "rates": [50.8, 50.66, 50.71], "tokens": 207, "stats": {}}, "graded": {"median": 71.69, "rates": [70.9, 71.69, 71.81], "tokens": 207, "stats": {"rounds": "57", "drafted_rounds": "36", "paused_rounds": "0", "proposed": "243", "accepted": "150", "acceptance_rate": "0.6173", "tokens_per_forward": "3.6140", "shadow_confirmations": "0"}, "x": 1.4137251035298757, "parity": "identical"}, "gated": {"median": 76.38, "rates": [78.43, 75.65, 76.38], "tokens": 207, "stats": {"rounds": "55", "drafted_rounds": "34", "paused_rounds": "0", "proposed": "233", "accepted": "152", "acceptance_rate": "0.6524", "tokens_per_forward": "3.7455", "shadow_confirmations": "0"}, "x": 1.506211792545849, "parity": "identical"}} +{"model": "qwen3-8b-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 49.99, "rates": [49.69, 49.99, 50.78], "tokens": 309, "stats": {}}, "graded": {"median": 95.19, "rates": [96.72, 95.19, 94.09], "tokens": 309, "stats": {"rounds": "52", "drafted_rounds": "45", "paused_rounds": "0", "proposed": "313", "accepted": "257", "acceptance_rate": "0.8211", "tokens_per_forward": "5.9231", "shadow_confirmations": "0"}, "x": 1.9041808361672332, "parity": "identical"}, "gated": {"median": 95.4, "rates": [95.56, 95.4, 94.46], "tokens": 309, "stats": {"rounds": "53", "drafted_rounds": "46", "paused_rounds": "0", "proposed": "317", "accepted": "256", "acceptance_rate": "0.8076", "tokens_per_forward": "5.8113", "shadow_confirmations": "0"}, "x": 1.908381676335267, "parity": "identical"}} +{"model": "qwen3-8b-4bit", "prompt": "summary", "n": "400", "plain": {"median": 49.64, "rates": [49.28, 50.74, 49.64], "tokens": 185, "stats": {}}, "graded": {"median": 78.84, "rates": [77.43, 78.84, 78.92], "tokens": 185, "stats": {"rounds": "44", "drafted_rounds": "31", "paused_rounds": "0", "proposed": "212", "accepted": "141", "acceptance_rate": "0.6651", "tokens_per_forward": "4.1818", "shadow_confirmations": "0"}, "x": 1.5882352941176472, "parity": "identical"}, "gated": {"median": 80.39, "rates": [79.67, 80.39, 81.32], "tokens": 185, "stats": {"rounds": "43", "drafted_rounds": "30", "paused_rounds": "0", "proposed": "200", "accepted": "142", "acceptance_rate": "0.7100", "tokens_per_forward": "4.2791", "shadow_confirmations": "0"}, "x": 1.619460112812248, "parity": "identical"}} +{"model": "qwen3-8b-4bit", "prompt": "write", "n": "400", "plain": {"median": 50.67, "rates": [50.67, 50.63, 51.16], "tokens": 141, "stats": {}}, "graded": {"median": 49.51, "rates": [49.44, 49.51, 49.86], "tokens": 131, "stats": {"rounds": "123", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "28", "accepted": "9", "acceptance_rate": "0.3214", "tokens_per_forward": "1.0569", "shadow_confirmations": "0"}, "x": 0.9771067692914939, "parity": "diverges@525"}, "gated": {"median": 49.13, "rates": [48.89, 49.41, 49.13], "tokens": 133, "stats": {"rounds": "127", "drafted_rounds": "6", "paused_rounds": "0", "proposed": "27", "accepted": "7", "acceptance_rate": "0.2593", "tokens_per_forward": "1.0394", "shadow_confirmations": "0"}, "x": 0.9696072626800869, "parity": "diverges@543"}} +{"model": "qwen3-8b-4bit", "prompt": "story", "n": "800", "plain": {"median": 52.24, "rates": [52.34, 52.05, 52.24], "tokens": 644, "stats": {}}, "graded": {"median": 47.44, "rates": [47.48, 46.66, 47.44], "tokens": 699, "stats": {"rounds": "679", "drafted_rounds": "54", "paused_rounds": "76", "proposed": "165", "accepted": "21", "acceptance_rate": "0.1273", "tokens_per_forward": "1.0280", "shadow_confirmations": "0"}, "x": 0.908116385911179, "parity": "diverges@475"}, "gated": {"median": 51.14, "rates": [51.53, 50.7, 51.14], "tokens": 689, "stats": {"rounds": "684", "drafted_rounds": "13", "paused_rounds": "560", "proposed": "36", "accepted": "6", "acceptance_rate": "0.1667", "tokens_per_forward": "1.0058", "shadow_confirmations": "7"}, "x": 0.9789433384379785, "parity": "diverges@1479"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 51.14, "rates": [51.66, 51.14, 51.13], "tokens": 199, "stats": {}}, "graded": {"median": 78.83, "rates": [78.52, 79.61, 78.83], "tokens": 199, "stats": {"rounds": "51", "drafted_rounds": "35", "paused_rounds": "0", "proposed": "237", "accepted": "148", "acceptance_rate": "0.6245", "tokens_per_forward": "3.8824", "shadow_confirmations": "0"}, "x": 1.541454829878764, "parity": "identical"}, "gated": {"median": 85.01, "rates": [85.34, 84.34, 85.01], "tokens": 199, "stats": {"rounds": "50", "drafted_rounds": "34", "paused_rounds": "0", "proposed": "228", "accepted": "149", "acceptance_rate": "0.6535", "tokens_per_forward": "3.9600", "shadow_confirmations": "0"}, "x": 1.6622995698083693, "parity": "identical"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 51.28, "rates": [51.2, 51.39, 51.28], "tokens": 275, "stats": {}}, "graded": {"median": 103.54, "rates": [103.54, 102.93, 103.55], "tokens": 275, "stats": {"rounds": "45", "drafted_rounds": "43", "paused_rounds": "0", "proposed": "298", "accepted": "230", "acceptance_rate": "0.7718", "tokens_per_forward": "6.0889", "shadow_confirmations": "0"}, "x": 2.0191107644305775, "parity": "identical"}, "gated": {"median": 100.88, "rates": [102.05, 100.88, 99.96], "tokens": 275, "stats": {"rounds": "47", "drafted_rounds": "45", "paused_rounds": "0", "proposed": "305", "accepted": "228", "acceptance_rate": "0.7475", "tokens_per_forward": "5.8298", "shadow_confirmations": "0"}, "x": 1.9672386895475817, "parity": "identical"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "summary", "n": "400", "plain": {"median": 49.8, "rates": [50.26, 49.46, 49.8], "tokens": 121, "stats": {}}, "graded": {"median": 46.57, "rates": [47.03, 46.57, 46.21], "tokens": 121, "stats": {"rounds": "83", "drafted_rounds": "23", "paused_rounds": "0", "proposed": "127", "accepted": "38", "acceptance_rate": "0.2992", "tokens_per_forward": "1.4458", "shadow_confirmations": "0"}, "x": 0.935140562248996, "parity": "identical"}, "gated": {"median": 50.96, "rates": [51.26, 50.71, 50.96], "tokens": 121, "stats": {"rounds": "89", "drafted_rounds": "29", "paused_rounds": "0", "proposed": "93", "accepted": "32", "acceptance_rate": "0.3441", "tokens_per_forward": "1.3483", "shadow_confirmations": "0"}, "x": 1.023293172690763, "parity": "identical"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "write", "n": "400", "plain": {"median": 50.94, "rates": [50.56, 50.94, 51.07], "tokens": 131, "stats": {}}, "graded": {"median": 49.14, "rates": [49.14, 48.96, 49.2], "tokens": 133, "stats": {"rounds": "129", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "26", "accepted": "5", "acceptance_rate": "0.1923", "tokens_per_forward": "1.0233", "shadow_confirmations": "0"}, "x": 0.9646643109540637, "parity": "diverges@606"}, "gated": {"median": 50.31, "rates": [50.31, 49.01, 51.28], "tokens": 119, "stats": {"rounds": "115", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "13", "accepted": "5", "acceptance_rate": "0.3846", "tokens_per_forward": "1.0261", "shadow_confirmations": "0"}, "x": 0.9876325088339224, "parity": "diverges@560"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "story", "n": "800", "plain": {"median": 51.58, "rates": [51.58, 51.56, 52.64], "tokens": 764, "stats": {}}, "graded": {"median": 47.92, "rates": [47.8, 48.2, 47.92], "tokens": 720, "stats": {"rounds": "694", "drafted_rounds": "42", "paused_rounds": "24", "proposed": "151", "accepted": "26", "acceptance_rate": "0.1722", "tokens_per_forward": "1.0360", "shadow_confirmations": "0"}, "x": 0.9290422644435828, "parity": "diverges@668"}, "gated": {"median": 51.7, "rates": [51.7, 51.59, 51.86], "tokens": 764, "stats": {"rounds": "761", "drafted_rounds": "13", "paused_rounds": "715", "proposed": "31", "accepted": "4", "acceptance_rate": "0.1290", "tokens_per_forward": "1.0026", "shadow_confirmations": "7"}, "x": 1.0023264831329974, "parity": "identical"}} diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run2-load.log b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run2-load.log new file mode 100644 index 000000000..42c1d17e4 --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run2-load.log @@ -0,0 +1,38 @@ +15:42:24 3.30 2.51 2.76 5/1159 938062 +15:42:54 3.59 2.65 2.80 6/1184 938834 +15:43:24 4.00 2.84 2.86 7/1180 939623 +15:43:54 3.78 2.91 2.88 5/1146 940268 +15:44:24 3.20 2.85 2.86 4/1135 940990 +15:44:54 3.19 2.88 2.88 4/1161 941578 +15:45:25 3.16 2.90 2.88 3/1168 942103 +15:45:55 3.61 3.04 2.93 4/1141 943602 +15:46:25 3.11 2.98 2.91 5/1148 944380 +15:46:55 2.66 2.89 2.88 2/1119 944944 +15:47:25 2.26 2.75 2.84 2/1122 946064 +15:47:55 2.24 2.70 2.82 4/1121 961327 +15:48:25 3.55 2.96 2.90 6/1184 973065 +15:48:55 3.19 2.93 2.89 4/1187 974147 +15:49:25 2.72 2.84 2.86 12/1160 974596 +15:49:55 2.38 2.74 2.83 4/1156 975045 +15:50:25 3.36 2.94 2.89 5/1214 976844 +15:50:55 3.62 3.05 2.93 5/1211 977221 +15:51:25 4.49 3.30 3.02 3/1155 979145 +15:51:55 3.80 3.24 3.01 5/1155 979889 +15:52:25 3.09 3.12 2.97 4/1121 980381 +15:52:55 2.19 2.90 2.90 5/1120 980733 +15:53:25 1.66 2.70 2.84 1/1123 980975 +15:53:55 1.49 2.55 2.78 5/1150 981372 +15:54:25 2.43 2.68 2.82 3/1175 982088 +15:54:55 1.80 2.50 2.75 1/1127 982377 +15:55:25 1.65 2.39 2.71 3/1120 982726 +15:55:55 1.61 2.31 2.67 5/1147 983970 +15:56:25 1.93 2.32 2.66 2/1146 984375 +15:56:55 1.95 2.29 2.64 3/1137 984734 +15:57:25 2.08 2.28 2.62 6/1148 985271 +15:57:55 3.10 2.51 2.69 6/1200 986187 +15:58:25 4.28 2.85 2.80 6/1202 987396 +15:58:55 4.46 3.02 2.86 4/1193 987846 +15:59:25 4.42 3.16 2.91 5/1148 988615 +15:59:55 3.54 3.07 2.89 6/1217 989337 +16:00:25 2.87 2.95 2.85 2/1210 989965 +16:00:55 2.29 2.79 2.80 3/1219 990271 diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run3-full-at-1.75.jsonl b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run3-full-at-1.75.jsonl new file mode 100644 index 000000000..6b010125e --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/run3-full-at-1.75.jsonl @@ -0,0 +1,20 @@ +{"model": "qwen3-1.7b-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 176.5, "rates": [174.85, 177.91, 176.5], "tokens": 210, "stats": {}}, "graded": {"median": 207.78, "rates": [207.71, 209.61, 207.78], "tokens": 210, "stats": {"rounds": "63", "drafted_rounds": "41", "paused_rounds": "0", "proposed": "259", "accepted": "147", "acceptance_rate": "0.5676", "tokens_per_forward": "3.3175", "shadow_confirmations": "0"}, "x": 1.1772237960339944, "parity": "identical"}, "gated": {"median": 211.64, "rates": [213.85, 211.64, 210.22], "tokens": 210, "stats": {"rounds": "77", "drafted_rounds": "55", "paused_rounds": "0", "proposed": "225", "accepted": "133", "acceptance_rate": "0.5911", "tokens_per_forward": "2.7143", "shadow_confirmations": "0"}, "x": 1.1990934844192633, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 178.85, "rates": [179.01, 177.26, 178.85], "tokens": 309, "stats": {}}, "graded": {"median": 244.96, "rates": [243.86, 244.96, 254.99], "tokens": 309, "stats": {"rounds": "66", "drafted_rounds": "51", "paused_rounds": "0", "proposed": "354", "accepted": "243", "acceptance_rate": "0.6864", "tokens_per_forward": "4.6667", "shadow_confirmations": "0"}, "x": 1.3696393625943528, "parity": "identical"}, "gated": {"median": 252.05, "rates": [253.88, 250.4, 252.05], "tokens": 309, "stats": {"rounds": "66", "drafted_rounds": "51", "paused_rounds": "0", "proposed": "347", "accepted": "243", "acceptance_rate": "0.7003", "tokens_per_forward": "4.6667", "shadow_confirmations": "0"}, "x": 1.4092815208275091, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "summary", "n": "400", "plain": {"median": 165.88, "rates": [166.31, 165.88, 164.6], "tokens": 115, "stats": {}}, "graded": {"median": 136.37, "rates": [132.75, 136.37, 137.4], "tokens": 115, "stats": {"rounds": "99", "drafted_rounds": "24", "paused_rounds": "8", "proposed": "95", "accepted": "16", "acceptance_rate": "0.1684", "tokens_per_forward": "1.1515", "shadow_confirmations": "0"}, "x": 0.8221003134796239, "parity": "identical"}, "gated": {"median": 164.99, "rates": [165.19, 164.99, 164.25], "tokens": 115, "stats": {"rounds": "115", "drafted_rounds": "4", "paused_rounds": "101", "proposed": "8", "accepted": "1", "acceptance_rate": "0.1250", "tokens_per_forward": "0.9913", "shadow_confirmations": "2"}, "x": 0.9946346756691585, "parity": "identical"}} +{"model": "qwen3-1.7b-4bit", "prompt": "write", "n": "400", "plain": {"median": 179.62, "rates": [180.29, 176.56, 179.62], "tokens": 72, "stats": {}}, "graded": {"median": 193.89, "rates": [193.89, 194.36, 192.81], "tokens": 72, "stats": {"rounds": "62", "drafted_rounds": "2", "paused_rounds": "0", "proposed": "14", "accepted": "11", "acceptance_rate": "0.7857", "tokens_per_forward": "1.1452", "shadow_confirmations": "0"}, "x": 1.0794454960472106, "parity": "identical"}, "gated": {"median": 184.53, "rates": [184.53, 184.76, 183.26], "tokens": 70, "stats": {"rounds": "63", "drafted_rounds": "5", "paused_rounds": "0", "proposed": "15", "accepted": "8", "acceptance_rate": "0.5333", "tokens_per_forward": "1.0952", "shadow_confirmations": "0"}, "x": 1.027335486026055, "parity": "diverges@372"}} +{"model": "qwen3-1.7b-4bit", "prompt": "story", "n": "800", "plain": {"median": 184.03, "rates": [185.71, 184.03, 181.15], "tokens": 385, "stats": {}}, "graded": {"median": 157.67, "rates": [157.67, 157.67, 155.3], "tokens": 378, "stats": {"rounds": "371", "drafted_rounds": "35", "paused_rounds": "56", "proposed": "97", "accepted": "7", "acceptance_rate": "0.0722", "tokens_per_forward": "1.0162", "shadow_confirmations": "0"}, "x": 0.8567624843775471, "parity": "diverges@536"}, "gated": {"median": 183.77, "rates": [183.77, 181.23, 184.24], "tokens": 430, "stats": {"rounds": "431", "drafted_rounds": "4", "paused_rounds": "371", "proposed": "8", "accepted": "0", "acceptance_rate": "0.0000", "tokens_per_forward": "0.9954", "shadow_confirmations": "3"}, "x": 0.9985871868717058, "parity": "diverges@1075"}} +{"model": "qwen3-4b-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 82.24, "rates": [83.2, 82.24, 81.56], "tokens": 207, "stats": {}}, "graded": {"median": 106.42, "rates": [101.84, 109.72, 106.42], "tokens": 207, "stats": {"rounds": "57", "drafted_rounds": "36", "paused_rounds": "0", "proposed": "243", "accepted": "150", "acceptance_rate": "0.6173", "tokens_per_forward": "3.6140", "shadow_confirmations": "0"}, "x": 1.2940175097276265, "parity": "identical"}, "gated": {"median": 109.29, "rates": [108.92, 109.29, 111.19], "tokens": 207, "stats": {"rounds": "60", "drafted_rounds": "39", "paused_rounds": "0", "proposed": "228", "accepted": "147", "acceptance_rate": "0.6447", "tokens_per_forward": "3.4333", "shadow_confirmations": "0"}, "x": 1.3289153696498055, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 81.46, "rates": [80.66, 81.46, 82.45], "tokens": 308, "stats": {}}, "graded": {"median": 143.31, "rates": [143.45, 143.31, 136.42], "tokens": 309, "stats": {"rounds": "52", "drafted_rounds": "45", "paused_rounds": "0", "proposed": "313", "accepted": "257", "acceptance_rate": "0.8211", "tokens_per_forward": "5.9231", "shadow_confirmations": "0"}, "x": 1.7592683525656765, "parity": "diverges@1346"}, "gated": {"median": 141.49, "rates": [139.44, 143.34, 141.49], "tokens": 308, "stats": {"rounds": "55", "drafted_rounds": "46", "paused_rounds": "0", "proposed": "307", "accepted": "253", "acceptance_rate": "0.8241", "tokens_per_forward": "5.5818", "shadow_confirmations": "0"}, "x": 1.736926098698748, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "summary", "n": "400", "plain": {"median": 81.36, "rates": [81.36, 80.42, 81.53], "tokens": 185, "stats": {}}, "graded": {"median": 115.46, "rates": [119.01, 110.63, 115.46], "tokens": 185, "stats": {"rounds": "44", "drafted_rounds": "31", "paused_rounds": "0", "proposed": "212", "accepted": "141", "acceptance_rate": "0.6651", "tokens_per_forward": "4.1818", "shadow_confirmations": "0"}, "x": 1.4191248770894789, "parity": "identical"}, "gated": {"median": 114.33, "rates": [109.96, 116.04, 114.33], "tokens": 185, "stats": {"rounds": "45", "drafted_rounds": "32", "paused_rounds": "0", "proposed": "204", "accepted": "140", "acceptance_rate": "0.6863", "tokens_per_forward": "4.0889", "shadow_confirmations": "0"}, "x": 1.40523598820059, "parity": "identical"}} +{"model": "qwen3-4b-4bit", "prompt": "write", "n": "400", "plain": {"median": 85.0, "rates": [85.0, 84.64, 85.27], "tokens": 140, "stats": {}}, "graded": {"median": 80.37, "rates": [80.4, 80.13, 80.37], "tokens": 140, "stats": {"rounds": "138", "drafted_rounds": "3", "paused_rounds": "0", "proposed": "21", "accepted": "3", "acceptance_rate": "0.1429", "tokens_per_forward": "1.0072", "shadow_confirmations": "0"}, "x": 0.945529411764706, "parity": "identical"}, "gated": {"median": 84.87, "rates": [84.87, 84.4, 84.98], "tokens": 139, "stats": {"rounds": "137", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "8", "accepted": "3", "acceptance_rate": "0.3750", "tokens_per_forward": "1.0073", "shadow_confirmations": "0"}, "x": 0.9984705882352942, "parity": "diverges@722"}} +{"model": "qwen3-4b-4bit", "prompt": "story", "n": "800", "plain": {"median": 83.97, "rates": [85.5, 83.97, 83.94], "tokens": 725, "stats": {}}, "graded": {"median": 74.0, "rates": [74.0, 73.92, 74.37], "tokens": 800, "stats": {"rounds": "780", "drafted_rounds": "91", "paused_rounds": "185", "proposed": "235", "accepted": "19", "acceptance_rate": "0.0809", "tokens_per_forward": "1.0244", "shadow_confirmations": "0"}, "x": 0.8812671192092414, "parity": "diverges@659"}, "gated": {"median": 83.08, "rates": [84.12, 82.02, 83.08], "tokens": 787, "stats": {"rounds": "785", "drafted_rounds": "9", "paused_rounds": "683", "proposed": "18", "accepted": "3", "acceptance_rate": "0.1667", "tokens_per_forward": "1.0013", "shadow_confirmations": "4"}, "x": 0.9894009765392402, "parity": "diverges@1267"}} +{"model": "qwen3-8b-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 49.5, "rates": [49.5, 50.0, 48.64], "tokens": 207, "stats": {}}, "graded": {"median": 71.45, "rates": [71.46, 71.32, 71.45], "tokens": 207, "stats": {"rounds": "57", "drafted_rounds": "36", "paused_rounds": "0", "proposed": "243", "accepted": "150", "acceptance_rate": "0.6173", "tokens_per_forward": "3.6140", "shadow_confirmations": "0"}, "x": 1.4434343434343435, "parity": "identical"}, "gated": {"median": 75.48, "rates": [74.98, 76.09, 75.48], "tokens": 207, "stats": {"rounds": "60", "drafted_rounds": "39", "paused_rounds": "0", "proposed": "228", "accepted": "147", "acceptance_rate": "0.6447", "tokens_per_forward": "3.4333", "shadow_confirmations": "0"}, "x": 1.524848484848485, "parity": "identical"}} +{"model": "qwen3-8b-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 49.8, "rates": [49.8, 50.03, 49.07], "tokens": 309, "stats": {}}, "graded": {"median": 97.9, "rates": [99.13, 97.9, 96.6], "tokens": 309, "stats": {"rounds": "52", "drafted_rounds": "45", "paused_rounds": "0", "proposed": "313", "accepted": "257", "acceptance_rate": "0.8211", "tokens_per_forward": "5.9231", "shadow_confirmations": "0"}, "x": 1.9658634538152613, "parity": "identical"}, "gated": {"median": 97.89, "rates": [96.15, 98.84, 97.89], "tokens": 309, "stats": {"rounds": "53", "drafted_rounds": "46", "paused_rounds": "0", "proposed": "307", "accepted": "256", "acceptance_rate": "0.8339", "tokens_per_forward": "5.8113", "shadow_confirmations": "0"}, "x": 1.9656626506024097, "parity": "identical"}} +{"model": "qwen3-8b-4bit", "prompt": "summary", "n": "400", "plain": {"median": 49.32, "rates": [49.32, 49.58, 48.11], "tokens": 185, "stats": {}}, "graded": {"median": 80.05, "rates": [79.87, 80.05, 82.71], "tokens": 185, "stats": {"rounds": "44", "drafted_rounds": "31", "paused_rounds": "0", "proposed": "212", "accepted": "141", "acceptance_rate": "0.6651", "tokens_per_forward": "4.1818", "shadow_confirmations": "0"}, "x": 1.623073803730738, "parity": "identical"}, "gated": {"median": 80.48, "rates": [80.57, 80.44, 80.48], "tokens": 185, "stats": {"rounds": "45", "drafted_rounds": "32", "paused_rounds": "0", "proposed": "204", "accepted": "140", "acceptance_rate": "0.6863", "tokens_per_forward": "4.0889", "shadow_confirmations": "0"}, "x": 1.6317923763179238, "parity": "identical"}} +{"model": "qwen3-8b-4bit", "prompt": "write", "n": "400", "plain": {"median": 49.34, "rates": [48.58, 49.39, 49.34], "tokens": 141, "stats": {}}, "graded": {"median": 47.67, "rates": [47.21, 47.67, 48.09], "tokens": 131, "stats": {"rounds": "123", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "28", "accepted": "9", "acceptance_rate": "0.3214", "tokens_per_forward": "1.0569", "shadow_confirmations": "0"}, "x": 0.9661532225374949, "parity": "diverges@525"}, "gated": {"median": 47.98, "rates": [47.98, 49.02, 47.86], "tokens": 131, "stats": {"rounds": "126", "drafted_rounds": "7", "paused_rounds": "0", "proposed": "19", "accepted": "6", "acceptance_rate": "0.3158", "tokens_per_forward": "1.0317", "shadow_confirmations": "0"}, "x": 0.9724361572760436, "parity": "diverges@525"}} +{"model": "qwen3-8b-4bit", "prompt": "story", "n": "800", "plain": {"median": 50.58, "rates": [49.92, 50.58, 51.31], "tokens": 686, "stats": {}}, "graded": {"median": 46.34, "rates": [45.22, 46.34, 47.06], "tokens": 640, "stats": {"rounds": "623", "drafted_rounds": "49", "paused_rounds": "77", "proposed": "149", "accepted": "17", "acceptance_rate": "0.1141", "tokens_per_forward": "1.0257", "shadow_confirmations": "0"}, "x": 0.9161724001581654, "parity": "diverges@475"}, "gated": {"median": 50.07, "rates": [49.3, 50.07, 50.82], "tokens": 680, "stats": {"rounds": "681", "drafted_rounds": "6", "paused_rounds": "577", "proposed": "12", "accepted": "0", "acceptance_rate": "0.0000", "tokens_per_forward": "0.9971", "shadow_confirmations": "5"}, "x": 0.9899169632265719, "parity": "diverges@1225"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "edit_fn", "n": "400", "plain": {"median": 49.69, "rates": [49.69, 49.22, 50.69], "tokens": 199, "stats": {}}, "graded": {"median": 80.33, "rates": [80.1, 80.37, 80.33], "tokens": 199, "stats": {"rounds": "51", "drafted_rounds": "35", "paused_rounds": "0", "proposed": "237", "accepted": "148", "acceptance_rate": "0.6245", "tokens_per_forward": "3.8824", "shadow_confirmations": "0"}, "x": 1.6166230629905414, "parity": "identical"}, "gated": {"median": 84.18, "rates": [83.14, 85.19, 84.18], "tokens": 199, "stats": {"rounds": "54", "drafted_rounds": "38", "paused_rounds": "0", "proposed": "221", "accepted": "145", "acceptance_rate": "0.6561", "tokens_per_forward": "3.6667", "shadow_confirmations": "0"}, "x": 1.6941034413362852, "parity": "identical"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "edit_json", "n": "400", "plain": {"median": 50.82, "rates": [50.65, 50.82, 50.9], "tokens": 275, "stats": {}}, "graded": {"median": 105.19, "rates": [104.79, 105.45, 105.19], "tokens": 275, "stats": {"rounds": "45", "drafted_rounds": "43", "paused_rounds": "0", "proposed": "298", "accepted": "230", "acceptance_rate": "0.7718", "tokens_per_forward": "6.0889", "shadow_confirmations": "0"}, "x": 2.069854388036206, "parity": "identical"}, "gated": {"median": 107.33, "rates": [108.03, 106.92, 107.33], "tokens": 275, "stats": {"rounds": "47", "drafted_rounds": "45", "paused_rounds": "0", "proposed": "290", "accepted": "228", "acceptance_rate": "0.7862", "tokens_per_forward": "5.8298", "shadow_confirmations": "0"}, "x": 2.1119637937819755, "parity": "identical"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "summary", "n": "400", "plain": {"median": 49.97, "rates": [49.54, 49.97, 50.11], "tokens": 121, "stats": {}}, "graded": {"median": 46.77, "rates": [46.77, 47.21, 44.14], "tokens": 121, "stats": {"rounds": "83", "drafted_rounds": "23", "paused_rounds": "0", "proposed": "127", "accepted": "38", "acceptance_rate": "0.2992", "tokens_per_forward": "1.4458", "shadow_confirmations": "0"}, "x": 0.9359615769461678, "parity": "identical"}, "gated": {"median": 55.46, "rates": [54.67, 55.64, 55.46], "tokens": 121, "stats": {"rounds": "89", "drafted_rounds": "29", "paused_rounds": "0", "proposed": "63", "accepted": "32", "acceptance_rate": "0.5079", "tokens_per_forward": "1.3483", "shadow_confirmations": "0"}, "x": 1.109865919551731, "parity": "identical"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "write", "n": "400", "plain": {"median": 51.56, "rates": [51.38, 51.56, 51.77], "tokens": 131, "stats": {}}, "graded": {"median": 49.74, "rates": [49.66, 49.74, 50.11], "tokens": 133, "stats": {"rounds": "129", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "26", "accepted": "5", "acceptance_rate": "0.1923", "tokens_per_forward": "1.0233", "shadow_confirmations": "0"}, "x": 0.9647013188518231, "parity": "diverges@606"}, "gated": {"median": 52.17, "rates": [51.96, 52.39, 52.17], "tokens": 119, "stats": {"rounds": "115", "drafted_rounds": "4", "paused_rounds": "0", "proposed": "8", "accepted": "5", "acceptance_rate": "0.6250", "tokens_per_forward": "1.0261", "shadow_confirmations": "0"}, "x": 1.0118308766485649, "parity": "diverges@560"}} +{"model": "meta-llama-3.1-8b-instruct-4bit", "prompt": "story", "n": "800", "plain": {"median": 52.17, "rates": [52.17, 51.57, 52.45], "tokens": 751, "stats": {}}, "graded": {"median": 48.81, "rates": [48.36, 48.81, 49.08], "tokens": 698, "stats": {"rounds": "684", "drafted_rounds": "36", "paused_rounds": "24", "proposed": "119", "accepted": "15", "acceptance_rate": "0.1261", "tokens_per_forward": "1.0190", "shadow_confirmations": "0"}, "x": 0.9355951696377228, "parity": "diverges@668"}, "gated": {"median": 51.56, "rates": [51.52, 51.56, 51.67], "tokens": 751, "stats": {"rounds": "750", "drafted_rounds": "10", "paused_rounds": "729", "proposed": "20", "accepted": "2", "acceptance_rate": "0.1000", "tokens_per_forward": "1.0000", "shadow_confirmations": "7"}, "x": 0.9883074563925628, "parity": "identical"}} diff --git a/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/verify_width_cost.txt b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/verify_width_cost.txt new file mode 100644 index 000000000..f0823167e --- /dev/null +++ b/docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/verify_width_cost.txt @@ -0,0 +1,49 @@ +# examples/verify_width_cost on GB10, release build, 2026-10-02 16:40 KST, load at start: 1.29 2.66 3.91 +## qwen3-1.7b-4bit +pipelined step: 5.174 ms +sync width 1: 6.343 ms = 1.23 pipelined steps +sync width 2: 7.112 ms = 1.37 pipelined steps +sync width 3: 9.085 ms = 1.76 pipelined steps +sync width 4: 12.546 ms = 2.43 pipelined steps +sync width 5: 15.731 ms = 3.04 pipelined steps +sync width 6: 17.966 ms = 3.47 pipelined steps +sync width 7: 19.875 ms = 3.84 pipelined steps +sync width 8: 20.250 ms = 3.91 pipelined steps +sync width 9: 20.143 ms = 3.89 pipelined steps +sync width 10: 20.814 ms = 4.02 pipelined steps +## qwen3-4b-4bit +pipelined step: 11.566 ms +sync width 1: 12.842 ms = 1.11 pipelined steps +sync width 2: 14.505 ms = 1.25 pipelined steps +sync width 3: 18.626 ms = 1.61 pipelined steps +sync width 4: 25.173 ms = 2.18 pipelined steps +sync width 5: 34.199 ms = 2.96 pipelined steps +sync width 6: 38.732 ms = 3.35 pipelined steps +sync width 7: 44.027 ms = 3.81 pipelined steps +sync width 8: 45.642 ms = 3.95 pipelined steps +sync width 9: 45.773 ms = 3.96 pipelined steps +sync width 10: 46.305 ms = 4.00 pipelined steps +## qwen3-8b-4bit +pipelined step: 18.998 ms +sync width 1: 20.397 ms = 1.07 pipelined steps +sync width 2: 21.941 ms = 1.15 pipelined steps +sync width 3: 30.278 ms = 1.59 pipelined steps +sync width 4: 39.829 ms = 2.10 pipelined steps +sync width 5: 54.370 ms = 2.86 pipelined steps +sync width 6: 64.075 ms = 3.37 pipelined steps +sync width 7: 73.161 ms = 3.85 pipelined steps +sync width 8: 62.687 ms = 3.30 pipelined steps +sync width 9: 62.338 ms = 3.28 pipelined steps +sync width 10: 62.267 ms = 3.28 pipelined steps +## meta-llama-3.1-8b-instruct-4bit +pipelined step: 18.773 ms +sync width 1: 20.311 ms = 1.08 pipelined steps +sync width 2: 21.823 ms = 1.16 pipelined steps +sync width 3: 30.139 ms = 1.61 pipelined steps +sync width 4: 39.042 ms = 2.08 pipelined steps +sync width 5: 52.522 ms = 2.80 pipelined steps +sync width 6: 60.868 ms = 3.24 pipelined steps +sync width 7: 69.254 ms = 3.69 pipelined steps +sync width 8: 55.426 ms = 2.95 pipelined steps +sync width 9: 55.387 ms = 2.95 pipelined steps +sync width 10: 55.712 ms = 2.97 pipelined steps diff --git a/docs/benchmark_results/prompt-lookup-governor-gb10-2026-10-02.md b/docs/benchmark_results/prompt-lookup-governor-gb10-2026-10-02.md new file mode 100644 index 000000000..88d14376d --- /dev/null +++ b/docs/benchmark_results/prompt-lookup-governor-gb10-2026-10-02.md @@ -0,0 +1,121 @@ +# Prompt-lookup drafting policy on GB10, 2026-10-02 + +Issue #2091, PR #2092. Prompt lookup (`mlxcel generate --prompt-lookup`, PR #2074) slowed prose 3 to 21% on GB10 while speeding up replies that copy their prompt. This record measures what a verify block costs per width on GB10, compares PR #2074's governor (`--prompt-lookup-policy graded`) with the CUDA policy this change adds and makes the CUDA default (`gated`), and checks greedy parity. + +Raw data under [`data/prompt-lookup-governor-gb10-2026-10-02/`](data/prompt-lookup-governor-gb10-2026-10-02/): + +- `harness/ab.py` and `harness/prompts/`: the interleaved A/B driver and the five prompts; +- `results.json`, `run2-final.jsonl`, `run2-load.log`: the final matrix below, one line per case with every repetition's rate and the arms' `[Prompt lookup]` counters, and the load average sampled every 30 s during it; +- `run1-probation-on-any-token.jsonl`, `run3-full-at-1.75.jsonl`: the two other full matrices, on variants that were not kept (see "How the policy was settled"); +- `verify_width_cost.txt`: `examples/verify_width_cost` on all four models. + +## Environment + +| Item | Value | +|------|-------| +| **Hardware** | NVIDIA GB10 (DGX Spark, sm_121), 20 CPU cores, 121 GiB unified memory | +| **OS / driver** | Linux 7.0.0-1019-nvidia, driver 580.178.04, CUDA 13.0 (V13.0.88) | +| **mlxcel** | `34c627d7` on `update/issue-2091-prompt-lookup-cuda-governor`, `cargo build --release --features cuda --bin mlxcel` | +| **MLX pin** | `81ba1c6a` | +| **Toolchain** | Rust 1.97.1 | +| **Checkpoints** | `models/mlx/qwen3-1.7b-4bit`, `qwen3-4b-4bit`, `qwen3-8b-4bit`, `meta-llama-3.1-8b-instruct-4bit` (affine 4-bit) | + +The host is shared: other sessions' builds ran intermittently. The GPU was held under `gpu-lock` for every run and no CI job was running. Load average over the final matrix: median 3.1, range 1.5 to 4.5 (38 samples). The interleaving below is what makes that tolerable: every repetition runs all three arms back to back, so drift lands on every arm alike. + +## Verify width cost + +`examples/verify_width_cost` prefills a fixed 300-token prompt, then times a pipelined one-token step (submitted from the still-lazy argmax before the host reads it, as `CxxGenerator` does) and a synchronous forward of `w` tokens plus the argmax and its host read, trimming `w - 1` positions per round so the cache grows like a rejected verify. Median of 40 rounds per width, in pipelined steps: + +| Width | 1 | 2 | 3 | 4 | 5 | 6 | 7 | 8 | 10 | +|---|---|---|---|---|---|---|---|---|---| +| Qwen3-1.7B | 1.23 | 1.37 | 1.76 | 2.43 | 3.04 | 3.47 | 3.84 | 3.91 | 4.02 | +| Qwen3-4B | 1.11 | 1.25 | 1.61 | 2.18 | 2.96 | 3.35 | 3.81 | 3.95 | 4.00 | +| Qwen3-8B | 1.07 | 1.15 | 1.59 | 2.10 | 2.86 | 3.37 | 3.85 | 3.30 | 3.28 | +| Llama-3.1-8B | 1.08 | 1.16 | 1.61 | 2.08 | 2.80 | 3.24 | 3.69 | 2.95 | 2.97 | + +Width 1 is the synchronous step itself: 7 to 23% over a pipelined one, more on the smaller models, where the host's share of a step is larger. Below 8 rows the affine path runs `qmv`'s multirow kernel, instantiated at 2, 4 and 8 accumulator rows (`dispatch_multirow_width`), so 5 to 7 rows pay for the 8-row instantiation; from 8 rows `quantized.cpp` switches to `qmm_sm80`, which on the 8B models is cheaper than the 7-row multirow launch. PR #2074's default block is 7 proposals, verify width 8. + +A drafted round in the PR #2074 loop also drains the pipeline: a proposal found while a plain step is in flight waits for that step to be read, the verify runs synchronously, and two synchronous plain rounds follow before pipelining resumes. On prose, where proposals rarely land, every drafted round is close to a pure loss of its verify cost plus those syncs. + +## The two policies + +`graded` (PR #2074, unchanged, still the default on Metal and ROCm): the block is twice the recent accepted average plus two, capped at 7; after three drafted rounds in a row land nothing it pauses 4 rounds, doubling to 32, and then probes again with a drafted round. + +`gated` (new, the default on CUDA builds): + +- the block is narrow (2 proposals, verify width 3) or full (7); it switches to full once a narrow block lands whole, and back when the accepted average falls; +- a reply starts narrow and on probation, and probation ends only when a round lands a whole narrow block; a round that lands nothing while on probation pauses; +- a pause has no timer: while paused, every round still looks the proposal up and checks it against the tokens decoding emits next, and drafting resumes only after two proposed tokens in a row came true (a "shadow confirmation"), on probation again; paused rounds pipeline at once. + +## Final matrix + +Greedy (`--temp 0`), `-n 400` (story `-n 800`), whole-call tok/s from the CLI's `[Generated ...]` line (prefill included, after `mlxcel generate`'s own warmup). Three repetitions per arm, interleaved; the cell is the median arm rate over the median plain rate of the same case. + +Plain decoding, tok/s: + +| Model | edit_fn | edit_json | summary | write | story | +|---|---|---|---|---|---| +| Qwen3-1.7B | 179.6 | 178.3 | 167.0 | 181.9 | 187.2 | +| Qwen3-4B | 83.2 | 82.8 | 82.2 | 85.9 | 85.4 | +| Qwen3-8B | 50.7 | 50.0 | 49.6 | 50.7 | 52.2 | +| Llama-3.1-8B | 51.1 | 51.3 | 49.8 | 50.9 | 51.6 | + +Prompt lookup over plain: + +| Model | policy | edit_fn | edit_json | summary | write | story | +|---|---|---|---|---|---|---| +| Qwen3-1.7B | graded | 1.16x | 1.39x | 0.82x | 1.07x | 0.83x | +| Qwen3-1.7B | gated | 1.22x | 1.41x | 0.99x | 1.02x | 0.98x | +| Qwen3-4B | graded | 1.27x | 1.71x | 1.38x | 0.94x | 0.87x | +| Qwen3-4B | gated | 1.39x | 1.64x | 1.42x | 0.98x | 0.98x | +| Qwen3-8B | graded | 1.41x | 1.90x | 1.59x | 0.98x | 0.91x | +| Qwen3-8B | gated | 1.51x | 1.91x | 1.62x | 0.97x | 0.98x | +| Llama-3.1-8B | graded | 1.54x | 2.02x | 0.94x | 0.96x | 0.93x | +| Llama-3.1-8B | gated | 1.66x | 1.97x | 1.02x | 0.99x | 1.00x | + +The `graded` rows reproduce the issue's baseline, measured on PR #2074's merged head, to within 1 to 3 points (1.7B story 0.83x against 0.84x, 8B edit_json 1.90x against 1.81x). Per-repetition spread within an arm was under 2% for most cases; the largest was 6.3%, one repetition of one case. + +Against the issue's criteria: every story row reaches 0.98x or more; three of four write rows do; Qwen3-8B's write row stays at 0.97x (0.98x under `graded` in the same run, 0.96x to 0.97x for `gated` across the three matrices). The edit and summary rows keep their gains: the largest drop against `graded` is Qwen3-4B edit_json, 1.71x to 1.64x (4%); most rise, because narrow blocks are cheaper per landed token on GB10 than full ones. + +Counters and greedy parity (reply text against plain decoding, first repetition): + +| Model | prompt | graded drafted/rounds | gated drafted/rounds | gated shadow confirmations | graded vs plain | gated vs plain | +|---|---|---|---|---|---|---| +| Qwen3-1.7B | edit_fn | 41/63 | 41/63 | 0 | identical | identical | +| Qwen3-1.7B | edit_json | 51/66 | 51/66 | 0 | identical | identical | +| Qwen3-1.7B | summary | 24/99 | 4/115 | 2 | identical | identical | +| Qwen3-1.7B | write | 2/62 | 4/64 | 0 | identical | identical | +| Qwen3-1.7B | story | 30/323 | 18/492 | 10 | diverges at char 536 | diverges at char 1149 | +| Qwen3-4B | edit_fn | 36/57 | 34/55 | 0 | identical | identical | +| Qwen3-4B | edit_json | 45/52 | 46/55 | 0 | diverges at char 1346 | identical | +| Qwen3-4B | summary | 31/44 | 30/43 | 0 | identical | identical | +| Qwen3-4B | write | 3/138 | 3/138 | 0 | identical | identical | +| Qwen3-4B | story | 70/715 | 12/794 | 7 | diverges at char 659 | diverges at char 1474 | +| Qwen3-8B | edit_fn | 36/57 | 34/55 | 0 | identical | identical | +| Qwen3-8B | edit_json | 45/52 | 46/53 | 0 | identical | identical | +| Qwen3-8B | summary | 31/44 | 30/43 | 0 | identical | identical | +| Qwen3-8B | write | 4/123 | 6/127 | 0 | diverges at char 525 | diverges at char 543 | +| Qwen3-8B | story | 54/679 | 13/684 | 7 | diverges at char 475 | diverges at char 1479 | +| Llama-3.1-8B | edit_fn | 35/51 | 34/50 | 0 | identical | identical | +| Llama-3.1-8B | edit_json | 43/45 | 45/47 | 0 | identical | identical | +| Llama-3.1-8B | summary | 23/83 | 29/89 | 0 | identical | identical | +| Llama-3.1-8B | write | 4/129 | 4/115 | 0 | diverges at char 606 | diverges at char 560 | +| Llama-3.1-8B | story | 42/694 | 13/761 | 7 | diverges at char 668 | identical | + +Every reply that `graded` keeps byte-identical to plain decoding, `gated` keeps identical too. The divergences are the documented near-tie class (a multi-token verify forward rounds differently from the one-token path); with fewer verify rounds, `gated` reaches them later or not at all. Story rounds outnumber `graded`'s because a paused `gated` loop emits one token per round where a `graded` drafted round sometimes emitted two. + +## The remaining row + +A round-by-round trace of Qwen3-8B's email (`gated`, debug build) shows where its 3% goes. The reply restates a short fragment of the prompt around tokens 45 to 55; lookup finds it, a narrow block lands whole, the block widens, and two full blocks (width 8, 3.3 pipelined steps on this model) land nothing as the fragment ends. `graded` spends its loss on the same fragment. Requiring two clean narrow blocks before a full one (`GATED_FULL_AT` 1.75, the third matrix) left this row at 0.97x and made the Qwen3-1.7B and Qwen3-4B emails diverge from plain decoding, so it was reverted. What would remove the remaining cost is the round-loop change the issue lists first: verify from the in-flight token instead of draining it, and drop the two synchronous rounds after a drafted round. It changes where verify blocks start, and with it which near-ties flip, so it needs its own parity pass. + +## How the policy was settled + +Three full matrices, same binary build procedure, interleaved arms: + +1. `43dc0755`, probation ended by any landed token (load median 1.2): story 0.98x to 1.00x, Qwen3-8B write 0.96x with 6 drafted rounds, 3 landed tokens and no pause: single landed tokens kept lifting probation. Fixed in `34c627d7`. +2. `34c627d7`, probation ends on a whole narrow block (load median 3.1): the final matrix above. +3. `7224525e`, full block after two whole narrow blocks (load median 4.9): Qwen3-8B write unchanged at 0.97x, two write rows lost parity. Reverted. + +## Metal + +No Apple Silicon machine was reachable from the measuring host. `graded` stays the default everywhere except CUDA builds and makes exactly PR #2074's decisions (same governor arithmetic, no shadow lookups), so the Apple Silicon numbers in PR #2074 still describe Metal. `--prompt-lookup-policy gated` is available there for anyone who wants to measure it. diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 5c82ff175..a237b8f4a 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -149,6 +149,14 @@ BENCH_MEM_OVERHEAD_FACTOR=1.209 ./scripts/bench_decode.sh all --cooldown 30 --bi # transcribe it into benchmarks/{backend}_{hw}_spec_{date}.csv. ./target/release/speculative_bench --sweep --max-tokens 128 +# Prompt lookup (`generate --prompt-lookup`): plain decoding against both +# drafting policies, arms interleaved per repetition, plus the verify-width +# cost the policies are built on. Method and GB10 results: +# docs/benchmark_results/prompt-lookup-governor-gb10-2026-10-02.md. +CASES=qwen3-8b-4bit:edit_json:400,qwen3-8b-4bit:story:800 REPS=3 BENCH_OUT=out-ab \ + python3 docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/harness/ab.py +cargo run --release --features cuda --example verify_width_cost -- models/mlx/qwen3-8b-4bit + # Batched serving ladder, against a server started with # --parallel 4 --max-batch-prefill 4. Writes no CSV; transcribe into # benchmarks/{backend}_{hw}_batch_{date}.csv. diff --git a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs index 4982e4423..b351b0e13 100644 --- a/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs +++ b/src/lib/mlxcel-core/src/speculative/prompt_lookup.rs @@ -172,17 +172,21 @@ impl Default for PromptLookupConfig { /// that lands nothing costs against a plain step. On an M4 Pro a wide block /// is cheap, so [`Self::Graded`] grades the block with the recent acceptance /// and probes again after a short pause. On GB10 (CUDA, affine 4-bit, MLX pin -/// `81ba1c6a`) a synchronous verify forward measured, in pipelined one-token -/// steps for Qwen3-1.7B / Qwen3-8B: width 2 at 1.35 / 1.14, width 3 at -/// 1.62 / 1.41, width 4 at 2.26 / 1.85, widths 5 to 7 rising to 3.59 / 3.28, -/// and width 8 and up flat at 3.34 / 2.51 (`examples/verify_width_cost.rs`). +/// `81ba1c6a`, release build) a synchronous verify forward measured, in +/// pipelined one-token steps for Qwen3-1.7B / Qwen3-8B: width 2 at +/// 1.37 / 1.15, width 3 at 1.76 / 1.59, width 4 at 2.43 / 2.10, width 7 at +/// 3.84 / 3.85, and width 8 and up at 3.91 / 3.30 +/// (`examples/verify_width_cost.rs`; all four benchmarked models in +/// `docs/benchmark_results/data/prompt-lookup-governor-gb10-2026-10-02/`). /// Below 8 rows the affine path runs `qmv`'s multirow kernel, instantiated at /// 2, 4 and 8 accumulator rows, so 5 to 7 rows pay for the 8-row -/// instantiation; from 8 rows it switches to `qmm_sm80`, which is cheaper than -/// that. A miss also drains the pipeline. So [`Self::Gated`] budgets only a -/// narrow or a full block (a lookup near the end of the context, or the -/// `max_tokens` limit, can still return fewer tokens) and spends no verify -/// forward to find out whether a copy has resumed. +/// instantiation; from 8 rows it switches to `qmm_sm80`. A block between +/// narrow and full therefore buys a few more proposals for close to a full +/// block's price (on the 8B models, more than a full block's), and a miss +/// also drains the pipeline. So [`Self::Gated`] budgets only a narrow or a +/// full block (a lookup near the end of the context, or the `max_tokens` +/// limit, can still return fewer tokens) and spends no verify forward to find +/// out whether a copy has resumed. #[derive(Debug, Clone, Copy, PartialEq, Eq)] #[non_exhaustive] pub enum DraftPolicy { From cd82927abb4a48aec3b48db76d0723d3070c3c5b Mon Sep 17 00:00:00 2001 From: Jeongkyu Shin Date: Fri, 2 Oct 2026 16:50:20 +0900 Subject: [PATCH 7/7] docs: add technical report for PR #2092 --- ...t-lookup-cuda-shadow-gating-20261002.en.md | 94 +++++++++++++++++++ ...t-lookup-cuda-shadow-gating-20261002.ko.md | 94 +++++++++++++++++++ 2 files changed, 188 insertions(+) create mode 100644 TECHNICAL_REPORTS/2092-prompt-lookup-cuda-shadow-gating-20261002.en.md create mode 100644 TECHNICAL_REPORTS/2092-prompt-lookup-cuda-shadow-gating-20261002.ko.md diff --git a/TECHNICAL_REPORTS/2092-prompt-lookup-cuda-shadow-gating-20261002.en.md b/TECHNICAL_REPORTS/2092-prompt-lookup-cuda-shadow-gating-20261002.en.md new file mode 100644 index 000000000..631cf67de --- /dev/null +++ b/TECHNICAL_REPORTS/2092-prompt-lookup-cuda-shadow-gating-20261002.en.md @@ -0,0 +1,94 @@ +# Technical Report: PR #2092 - Gate CUDA prompt-lookup drafting on shadow probes + +**Date**: 2026-10-02 + +**Status**: Implemented and measured on GB10; pending merge. Issue #2091 stays open: one of its eight write and story targets (Qwen3-8B email, 0.97x against 0.98x) is not met, and Apple Silicon was not measured. + +**Languages**: Rust (prompt-lookup governor and decode loop, CLI flag, tests, one example), Python (benchmark harness), Markdown and JSON (benchmark record) + +**Risk Level**: Low. Only CUDA builds change behavior; the Metal and ROCm path makes exactly PR #2074's decisions. Output distribution is unchanged on every path (the acceptance rule is untouched); greedy output can differ from plain decoding only at near-ties, as before. + +## Executive Summary + +Prompt lookup (`mlxcel generate --prompt-lookup`, PR #2074) slowed prose 3 to 21% on GB10 because its governor, tuned on an M4 Pro, kept probing prose with verify blocks that are expensive on CUDA. This PR adds a second drafting policy, `DraftPolicy::Gated`, made the default on CUDA builds: it uses only a narrow or a full block, and it never spends a verify forward to find out whether a copy has resumed; while paused it checks the lookup's proposal against the tokens decoding emits next ("shadow probe"). On GB10, stories go from 0.83x-0.93x to 0.98x-1.00x of plain decoding, edit and summary gains are kept or raised, and every reply that stayed byte-identical to plain decoding still does. + +## 1. Problem Statement + +PR #2074's governor (now `DraftPolicy::Graded`) sizes the block as twice the recent accepted average plus two, up to 7 proposals (verify width 8), and after three drafted rounds land nothing it pauses 4 rounds, doubling to 32, then probes again with a drafted round. Each drafted round in the loop pays three costs: the verify forward itself, a drained pipeline (a proposal found while a plain step is in flight waits for that step to be read), and two synchronous plain rounds before pipelining resumes. On prose, where proposals rarely land, the probes are nearly pure loss. + +How much loss depends on the backend, which the original tuning could not see. `examples/verify_width_cost.rs`, added here, measures a synchronous verify forward in pipelined one-token steps. On GB10 (release build, MLX pin `81ba1c6a`): + +| Width | 2 | 3 | 4 | 7 | 8 | +|---|---|---|---|---|---| +| Qwen3-1.7B | 1.37 | 1.76 | 2.43 | 3.84 | 3.91 | +| Qwen3-8B | 1.15 | 1.59 | 2.10 | 3.85 | 3.30 | + +Below 8 rows the affine path runs `qmv`'s multirow kernel, which `dispatch_multirow_width` instantiates at 2, 4 and 8 accumulator rows, so 5 to 7 rows pay for the 8-row instantiation; from 8 rows `quantized.cpp` switches to `qmm_sm80`. A block between narrow and full buys a few more proposals for nearly a full block's price, and on the 8B models for more than it. + +## 2. Change Summary + +- `DraftPolicy { Graded, Gated }` (`#[non_exhaustive]`) on `PromptLookupConfig`; `DraftPolicy::default_for_backend()` picks `Gated` when the crate is built with `cuda`, `Graded` otherwise. +- `DraftGovernor` keeps the Graded arithmetic byte-for-byte and adds the Gated state machine: `drafting`, `probation`, the shared accepted-proposal EMA, narrow (`GATED_NARROW_DRAFT` = 2) or full (`max_draft`) budget, `GATED_FULL_AT` = 1.5, `GATED_START_EMA` = 1.0. +- `ShadowProbe { start, proposal }` and `settle(&context) -> Option`; the decode loop settles the pending probe at the top of each round, looks a new one up on paused rounds (two tokens, `SHADOW_CONFIRM`), and calls `governor.shadow_confirmed()` when it comes true. A paused Gated loop pipelines its plain rounds at once. +- `PromptLookupStats::shadow_confirmations`, printed on the `[Prompt lookup]` line and in the end-of-decode trace. +- CLI: `--prompt-lookup-policy auto|graded|gated` (requires `--prompt-lookup`), `PromptLookupPolicyArg::resolve`, `mlxcel::DraftPolicy` re-export. +- Tests: Gated start, probation, timer-free pause, widths, max-draft below the narrow block, shadow settling, policy defaults, CLI parsing; the rollback parity matrix runs per policy; an exact script-model test pins probe alignment. +- `docs/benchmark_results/prompt-lookup-governor-gb10-2026-10-02.md` with the harness, prompts, three matrices, the load log and the width costs; `docs/benchmarks.md` links it. + +## 3. How Gated Decides + +1. A reply starts drafting, narrow, on probation. The first drafted round of a reply has no evidence behind it. +2. A drafted round updates the EMA. When the EMA reaches 1.5, the block is full; a narrow block that lands whole from the starting EMA gets there in one round. +3. Probation ends only on a round that lands a whole narrow block. A round that lands nothing while on probation pauses at once; off probation, three misses in a row pause. +4. While paused, `budget()` returns 0 indefinitely. Each round, if no probe is pending, the loop looks up a two-token proposal at the current context length and records it. When decoding has emitted those two positions, the probe settles: both came true, so drafting resumes (narrow, on probation, EMA reset); otherwise it is dropped and a new one is made. +5. Because nothing can be proposed until a probe settles, which takes at least two emitted tokens, paused rounds pipeline immediately rather than after two synchronous ones. + +The probe's `start` is the context length when the lookup ran. In the pipelined path the context ends at `current_token` while `next` is in flight, so `proposal[0]` claims `next`; in the synchronous path it claims the token the round's forward will produce. Both line up with the next emitted token, which the script-model test pins: a probe recorded one position early or late never confirms there. + +## 4. Measured Results + +Final matrix on `34c627d7` (the shipped tree), greedy, `-n 400` (story `-n 800`), whole-call tok/s including prefill, median of 3, plain/graded/gated interleaved per repetition; load average median 3.1 on a shared host. + +| Model | edit_fn | edit_json | summary | write | story | +|---|---|---|---|---|---| +| Qwen3-1.7B graded / gated | 1.16 / 1.22 | 1.39 / 1.41 | 0.82 / 0.99 | 1.07 / 1.02 | 0.83 / 0.98 | +| Qwen3-4B graded / gated | 1.27 / 1.39 | 1.71 / 1.64 | 1.38 / 1.42 | 0.94 / 0.98 | 0.87 / 0.98 | +| Qwen3-8B graded / gated | 1.41 / 1.51 | 1.90 / 1.91 | 1.59 / 1.62 | 0.98 / 0.97 | 0.91 / 0.98 | +| Llama-3.1-8B graded / gated | 1.54 / 1.66 | 2.02 / 1.97 | 0.94 / 1.02 | 0.96 / 0.99 | 0.93 / 1.00 | + +Drafted rounds on stories fall from 30-70 to 12-18; shadow probes resume drafting 7 to 10 times per story. Edit rows often rise because a narrow block lands more tokens per unit of verify cost on GB10 than a full one. The largest edit/summary drop against Graded is Qwen3-4B edit_json (4%). + +## 5. Technical Decisions + +**Per-backend default instead of one retuned governor.** Graded was tuned on an M4 Pro, where a wide block is cheap; Gated was measured on GB10. No Apple Silicon machine was reachable, so changing Metal's behavior would have been unmeasured. Keeping Graded's decisions exact on Metal makes "no Metal regression" true by construction, and `--prompt-lookup-policy` lets anyone measure Gated there. The cost is two policies to maintain. + +**Shadow probes instead of timed probes.** Graded learns whether a copy resumed by paying a verify forward; Gated learns it from a hash lookup and the tokens decoding was going to emit anyway. Requiring two confirmed tokens (not one) matters on prose, where the token after a two-token match is often a common one that recurs. + +**Governor only; the round loop's verify path untouched.** The issue listed a pipeline-preserving verify (start the verify from the in-flight token) as its first direction. It changes where verify blocks start, and with that which near-ties flip, so it would put the greedy-parity criterion at risk for a measured gap of about one point on one row. The issue's own decision rule prefers the smaller change to the round loop. + +**Rejected after measurement: a higher full-block threshold.** Requiring two clean narrow blocks before a full one (`GATED_FULL_AT` 1.75) aimed at the Qwen3-8B email, where a short restated fragment of the prompt widened the block just before the fragment ended. The third matrix showed the row unchanged at 0.97x and two email rows newly diverging from plain decoding, so it was reverted (`d678c940`). + +## 6. Validation + +- `speculative::prompt_lookup`: 39 tests pass. Deliberate mutations each fail them: probation ignored, a timed Gated pause, one-token confirmation, no shadow lookups, a full first block, probe start one position early or late, and probation lifted by a single landed token. +- CLI tests in the `mlxcel` bin: 96 pass; `dead_doc_pointers` passes; clippy `-D warnings` on lib, tests, bins and the example, and fmt, are clean. +- Real checkpoints on GB10: the matrix above, plus a round-by-round trace of the Qwen3-8B email from a debug build. + +## 7. Residual Risks and What Was Not Verified + +- **Qwen3-8B email at 0.97x** (0.96x to 0.97x across three matrices; Graded 0.97x to 0.98x). The trace attributes it to a restated prompt fragment; the pipeline-preserving verify is the remaining lever. +- **Metal not measured.** Safe by construction for the default; Gated on Metal is untested. +- **ROCm** keeps Graded, also unmeasured for prompt lookup. +- **Other CUDA architectures and NVFP4 targets** inherit Gated. The multirow and `qmm_sm80` boundaries are CUDA-wide on the affine path; NVFP4 runs `fp_qmv`, whose costs per width were not measured. +- **Shared host.** The final matrix ran at load median 3.1; arm interleaving spreads drift, and the largest within-arm spread was 6.3% (one repetition). + +## 8. Learning Points + +- **Measure the cost curve before tuning a speculation policy.** The governor's question is "is a miss cheap enough to probe with?", and on CUDA the answer is a step function of block width set by kernel instantiations, not a smooth curve. `examples/verify_width_cost.rs` answers it in a minute per model; rerun it after an MLX pin bump that moves the `qmv` or `qmm_sm80` boundaries. +- **A free signal beats a paid probe.** Decoding emits the tokens a proposal would have claimed anyway; comparing them costs a hash lookup. The same pattern applies to any drafter whose proposal can be produced without the target. +- **Speculative schedules move greedy near-ties.** Any change to which positions run in a multi-token verify can flip a near-tie, so "byte-identical where it was" has to be measured per row after every schedule change, as the reverted threshold showed. + +## 9. Related + +- Issue #2091 (this work), PR #2074 (prompt lookup, Graded governor), #2090 (plain decoding's stale penalty history, found during #2074). +- `src/cli/draft_block_policy.rs`: the same kernel-boundary reasoning for DFlash block width on GB10. diff --git a/TECHNICAL_REPORTS/2092-prompt-lookup-cuda-shadow-gating-20261002.ko.md b/TECHNICAL_REPORTS/2092-prompt-lookup-cuda-shadow-gating-20261002.ko.md new file mode 100644 index 000000000..f68a0c47c --- /dev/null +++ b/TECHNICAL_REPORTS/2092-prompt-lookup-cuda-shadow-gating-20261002.ko.md @@ -0,0 +1,94 @@ +# 기술 보고서: PR #2092 - CUDA prompt lookup 드래프팅을 그림자 탐침으로 게이팅 + +**작성일**: 2026-10-02 + +**상태**: GB10에서 구현하고 측정했으며 머지 대기 중이다. 이슈 #2091은 열어 둔다. write·story 목표 여덟 개 가운데 하나(Qwen3-8B 이메일, 0.98x 목표에 0.97x)를 채우지 못했고 Apple Silicon은 측정하지 못했다. + +**언어**: Rust(prompt lookup 거버너와 디코드 루프, CLI 플래그, 테스트, 예제 하나), Python(벤치마크 하니스), Markdown·JSON(벤치마크 기록) + +**위험도**: 낮음. 동작이 바뀌는 것은 CUDA 빌드뿐이고, Metal과 ROCm 경로는 PR #2074와 똑같은 결정을 내린다. 수락 규칙을 건드리지 않았으므로 어느 경로에서든 출력 분포는 그대로이고, greedy 출력이 plain 디코딩과 달라질 수 있는 곳은 예전처럼 near-tie뿐이다. + +## 요약 + +Prompt lookup(`mlxcel generate --prompt-lookup`, PR #2074)은 GB10에서 산문을 3~21% 느리게 만들었다. M4 Pro에서 튜닝한 거버너가 CUDA에서 비싼 검증 블록으로 산문을 계속 찔러 봤기 때문이다. 이 PR은 두 번째 드래프팅 정책 `DraftPolicy::Gated`를 추가하고 CUDA 빌드의 기본값으로 삼는다. 이 정책은 좁은 블록과 꽉 찬 블록만 쓰고, 복사가 다시 시작됐는지 알아보려고 검증 forward를 쓰지 않는다. 멈춰 있는 동안에는 lookup이 내놓은 제안을 디코딩이 실제로 내보내는 다음 토큰과 대조한다(그림자 탐침). GB10에서 story는 plain 디코딩 대비 0.83x~0.93x에서 0.98x~1.00x가 됐고, 편집과 요약의 이득은 유지되거나 커졌으며, plain 디코딩과 바이트 단위로 같던 응답은 모두 그대로 같다. + +## 1. 문제 정의 + +PR #2074의 거버너(이제 `DraftPolicy::Graded`)는 블록 크기를 최근 수락 평균의 두 배에 2를 더한 값으로 잡는다(최대 제안 7개, 검증 폭 8). 드래프트 라운드가 세 번 연속 아무것도 못 맞히면 4라운드를 쉬고, 그 길이를 32까지 두 배씩 늘린 뒤 다시 드래프트 라운드로 찔러 본다. 루프에서 드래프트 라운드 하나는 세 가지 비용을 낸다. 검증 forward 자체, 비워지는 파이프라인(plain 스텝이 진행 중일 때 제안이 나오면 그 스텝을 읽을 때까지 기다린다), 그리고 파이프라이닝이 재개되기 전에 도는 동기 plain 라운드 두 개다. 제안이 거의 맞지 않는 산문에서는 이 찔러 보기가 거의 순손실이다. + +손실의 크기는 백엔드에 따라 다르고, 원래 튜닝은 이것을 볼 수 없었다. 이번에 추가한 `examples/verify_width_cost.rs`는 동기 검증 forward 비용을 파이프라인된 1토큰 스텝 단위로 잰다. GB10(release 빌드, MLX 핀 `81ba1c6a`)에서 잰 값은 이렇다. + +| 폭 | 2 | 3 | 4 | 7 | 8 | +|---|---|---|---|---|---| +| Qwen3-1.7B | 1.37 | 1.76 | 2.43 | 3.84 | 3.91 | +| Qwen3-8B | 1.15 | 1.59 | 2.10 | 3.85 | 3.30 | + +8행 미만에서 affine 경로는 `qmv`의 multirow 커널을 쓰는데, `dispatch_multirow_width`가 이 커널을 누산기 2·4·8행으로만 인스턴스화하므로 5~7행은 8행 인스턴스 비용을 그대로 낸다. 8행부터는 `quantized.cpp`가 `qmm_sm80`으로 넘어간다. 그래서 좁은 블록과 꽉 찬 블록 사이의 크기는 제안 몇 개를 더 얻는 대가로 거의 꽉 찬 블록만큼 내고, 8B 모델에서는 그보다 더 낸다. + +## 2. 변경 요약 + +- `PromptLookupConfig`에 `DraftPolicy { Graded, Gated }`(`#[non_exhaustive]`)를 추가했다. `DraftPolicy::default_for_backend()`는 크레이트가 `cuda`로 빌드되면 `Gated`, 아니면 `Graded`를 고른다. +- `DraftGovernor`는 Graded의 산술을 바이트 단위로 그대로 두고 Gated 상태 기계를 더했다. 상태는 `drafting`, `probation`, 공유하는 수락 EMA이고, 예산은 좁은 블록(`GATED_NARROW_DRAFT` = 2)이나 꽉 찬 블록(`max_draft`) 중 하나다. `GATED_FULL_AT` = 1.5, `GATED_START_EMA` = 1.0. +- `ShadowProbe { start, proposal }`와 `settle(&context) -> Option`. 디코드 루프는 라운드마다 맨 앞에서 대기 중인 탐침을 판정하고, 멈춘 라운드에서는 새 탐침(토큰 두 개, `SHADOW_CONFIRM`)을 만들며, 탐침이 맞으면 `governor.shadow_confirmed()`를 부른다. 멈춘 Gated 루프는 plain 라운드를 곧바로 파이프라인한다. +- `PromptLookupStats::shadow_confirmations`를 `[Prompt lookup]` 줄과 디코드 종료 trace에 찍는다. +- CLI: `--prompt-lookup-policy auto|graded|gated`(`--prompt-lookup` 필요), `PromptLookupPolicyArg::resolve`, `mlxcel::DraftPolicy` 재수출. +- 테스트: Gated의 시작 상태, probation, 타이머 없는 정지, 블록 폭, 좁은 블록보다 작은 max-draft, 탐침 판정, 정책 기본값, CLI 파싱. 롤백 패리티 매트릭스는 정책별로 따로 돌고, 스크립트 모델로 탐침 정렬을 정확히 고정하는 테스트를 더했다. +- `docs/benchmark_results/prompt-lookup-governor-gb10-2026-10-02.md`에 하니스, 프롬프트, 매트릭스 세 벌, 부하 로그, 폭별 비용을 담았고 `docs/benchmarks.md`에서 링크한다. + +## 3. Gated가 결정하는 방식 + +1. 응답은 좁은 블록, probation 상태로 드래프팅을 시작한다. 응답의 첫 드래프트 라운드에는 근거가 없기 때문이다. +2. 드래프트 라운드마다 EMA를 갱신하고, EMA가 1.5에 이르면 블록을 꽉 채운다. 시작 EMA에서 좁은 블록 하나가 통째로 맞으면 한 라운드 만에 거기 닿는다. +3. probation은 좁은 블록 하나를 통째로 맞힌 라운드에서만 풀린다. probation 중에 아무것도 못 맞힌 라운드는 곧바로 멈추고, probation이 풀린 뒤에는 세 번 연속 빗나가야 멈춘다. +4. 멈춰 있는 동안 `budget()`은 계속 0을 돌려준다. 루프는 대기 중인 탐침이 없으면 라운드마다 현재 컨텍스트 길이에서 토큰 두 개짜리 제안을 찾아 기록해 둔다. 디코딩이 그 두 자리를 내보내면 탐침을 판정하는데, 둘 다 맞았으면 드래프팅을 재개하고(좁은 블록, 다시 probation, EMA 초기화) 아니면 버리고 새로 만든다. +5. 탐침이 판정되기 전에는 아무것도 제안할 수 없고 판정에는 최소 두 토큰이 필요하므로, 멈춘 라운드는 동기 라운드 두 개를 기다리지 않고 곧바로 파이프라인한다. + +탐침의 `start`는 lookup을 돌린 시점의 컨텍스트 길이다. 파이프라인 경로에서는 `next`가 진행 중이라 컨텍스트가 `current_token`에서 끝나므로 `proposal[0]`은 `next`를 가리키고, 동기 경로에서는 그 라운드의 forward가 만들 토큰을 가리킨다. 어느 쪽이든 다음에 내보낼 토큰과 자리가 맞으며, 스크립트 모델 테스트가 이것을 고정한다. 한 자리 이르거나 늦게 기록된 탐침은 그 테스트에서 끝내 확인되지 않는다. + +## 4. 측정 결과 + +최종 매트릭스는 `34c627d7`(실제로 나가는 트리)에서 돌렸다. greedy, `-n 400`(story는 `-n 800`), prefill을 포함한 전체 호출 tok/s, 3회 중앙값이고 반복마다 plain·graded·gated를 번갈아 돌렸다. 공유 호스트였고 부하 평균의 중앙값은 3.1이었다. + +| 모델 | edit_fn | edit_json | summary | write | story | +|---|---|---|---|---|---| +| Qwen3-1.7B graded / gated | 1.16 / 1.22 | 1.39 / 1.41 | 0.82 / 0.99 | 1.07 / 1.02 | 0.83 / 0.98 | +| Qwen3-4B graded / gated | 1.27 / 1.39 | 1.71 / 1.64 | 1.38 / 1.42 | 0.94 / 0.98 | 0.87 / 0.98 | +| Qwen3-8B graded / gated | 1.41 / 1.51 | 1.90 / 1.91 | 1.59 / 1.62 | 0.98 / 0.97 | 0.91 / 0.98 | +| Llama-3.1-8B graded / gated | 1.54 / 1.66 | 2.02 / 1.97 | 0.94 / 1.02 | 0.96 / 0.99 | 0.93 / 1.00 | + +story의 드래프트 라운드는 30~70개에서 12~18개로 줄었고, 그림자 탐침이 story마다 7~10번 드래프팅을 재개시켰다. 편집 행이 자주 오르는 이유는 GB10에서 좁은 블록이 검증 비용당 맞히는 토큰이 꽉 찬 블록보다 많기 때문이다. 편집·요약 행 가운데 Graded 대비 가장 크게 떨어진 것은 Qwen3-4B edit_json(4%)이다. + +## 5. 기술적 선택과 그 이유 + +**거버너 하나를 다시 튜닝하지 않고 백엔드별 기본값을 둔다.** Graded는 넓은 블록이 싼 M4 Pro에서 튜닝했고 Gated는 GB10에서 측정했다. Apple Silicon 머신에 접근할 수 없었으므로 Metal 동작을 바꾸면 측정 없는 변경이 된다. Metal에서 Graded의 결정을 그대로 두면 'Metal 회귀 없음'이 구조적으로 참이 되고, 누구든 `--prompt-lookup-policy`로 Metal에서 Gated를 측정할 수 있다. 대가는 정책 두 개를 유지하는 일이다. + +**시간으로 찔러 보지 않고 그림자 탐침을 쓴다.** Graded는 복사가 재개됐는지 알려고 검증 forward를 지불하고, Gated는 해시 조회 한 번과 어차피 디코딩이 내보낼 토큰으로 알아낸다. 확인 토큰을 하나가 아니라 둘로 잡은 것은 산문 때문이다. 두 토큰 일치 다음에 오는 토큰은 흔히 자주 반복되는 평범한 토큰이라 하나만으로는 우연히 맞는다. + +**거버너만 바꾸고 라운드 루프의 검증 경로는 건드리지 않는다.** 이슈가 첫 방향으로 꼽은 것은 파이프라인을 보존하는 검증, 곧 진행 중인 토큰에서 검증을 시작하는 방식이었다. 이 방식은 검증 블록이 시작되는 자리를 바꾸고 그에 따라 뒤집히는 near-tie도 바꾸므로, 한 행에서 측정된 1포인트 남짓의 격차를 위해 greedy 패리티 기준을 위험에 빠뜨린다. 이슈의 결정 규칙도 라운드 루프를 덜 바꾸는 쪽을 택하라고 한다. + +**측정 후 기각: 꽉 찬 블록의 문턱을 높이는 안.** 꽉 찬 블록 전에 깨끗한 좁은 블록 두 개를 요구하는 안(`GATED_FULL_AT` 1.75)은 Qwen3-8B 이메일을 겨냥했다. 그 응답에서는 프롬프트의 짧은 구절을 되풀이하는 부분이 구절이 끝나기 직전에 블록을 넓혔다. 세 번째 매트릭스에서 이 행은 0.97x로 그대로였고 이메일 행 두 개가 plain 디코딩과 새로 갈라졌으므로 되돌렸다(`d678c940`). + +## 6. 검증 + +- `speculative::prompt_lookup`: 테스트 39개 통과. 일부러 넣은 변이마다 테스트가 실패했다. probation 무시, 시간 기반 Gated 정지, 한 토큰 확인, 그림자 lookup 없음, 꽉 찬 첫 블록, 탐침 시작 위치 한 칸 이름·늦음, 맞힌 토큰 하나로 풀리는 probation이 그것이다. +- `mlxcel` 바이너리의 CLI 테스트 96개 통과, `dead_doc_pointers` 통과, lib·tests·bins·예제에 대한 clippy `-D warnings`와 fmt 모두 깨끗하다. +- GB10 실제 체크포인트: 위 매트릭스와, 디버그 빌드로 뽑은 Qwen3-8B 이메일의 라운드별 추적. + +## 7. 남은 위험과 검증하지 못한 것 + +- **Qwen3-8B 이메일 0.97x**(매트릭스 세 벌에서 0.96x~0.97x, Graded는 0.97x~0.98x). 추적으로 보면 프롬프트 구절을 되풀이하는 부분 때문이고, 남은 수단은 파이프라인을 보존하는 검증이다. +- **Metal 미측정.** 기본값은 구조적으로 안전하지만 Metal에서의 Gated는 시험하지 않았다. +- **ROCm**은 Graded를 유지하며, prompt lookup으로는 역시 측정하지 않았다. +- **다른 CUDA 아키텍처와 NVFP4 대상**도 Gated를 물려받는다. affine 경로의 multirow와 `qmm_sm80` 경계는 CUDA 전반에 공통이지만, NVFP4는 `fp_qmv`를 타며 그 폭별 비용은 재지 않았다. +- **공유 호스트.** 최종 매트릭스는 부하 평균 중앙값 3.1에서 돌았다. 팔을 번갈아 돌려 흔들림을 고르게 나눴고, 한 팔 안의 최대 편차는 6.3%(반복 한 번)였다. + +## 8. 학습 포인트 + +- **추측 정책을 튜닝하기 전에 비용 곡선부터 잰다.** 거버너가 묻는 것은 '빗나감이 찔러 볼 만큼 싼가'이고, CUDA에서 그 답은 매끄러운 곡선이 아니라 커널 인스턴스화가 정하는 블록 폭의 계단 함수다. `examples/verify_width_cost.rs`가 모델당 1분 안에 답을 준다. `qmv`나 `qmm_sm80` 경계를 옮기는 MLX 핀 갱신 뒤에는 다시 돌려야 한다. +- **공짜 신호가 돈 드는 탐침보다 낫다.** 제안이 주장했을 토큰은 디코딩이 어차피 내보내고, 그것과 대조하는 비용은 해시 조회 한 번이다. 대상 모델 없이 제안을 만들 수 있는 드래프터라면 같은 방식을 쓸 수 있다. +- **추측 스케줄은 greedy near-tie를 움직인다.** 어느 자리를 다중 토큰 검증으로 돌리는지가 바뀌면 near-tie가 뒤집힐 수 있다. 그래서 '같던 곳은 그대로 같다'는 스케줄을 바꿀 때마다 행별로 측정해야 하고, 되돌린 문턱 변경이 그 예다. + +## 9. 관련 항목 + +- 이슈 #2091(이번 작업), PR #2074(prompt lookup과 Graded 거버너), #2090(#2074 작업 중 발견한 plain 디코딩의 낡은 페널티 이력). +- `src/cli/draft_block_policy.rs`: GB10에서 DFlash 블록 폭을 같은 커널 경계 논리로 정한 곳.