From b13439cbc73690cbc6765ef910fe4c054209959d Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 28 May 2026 17:49:59 +0000 Subject: [PATCH] feat(forge): SkillOpt-style epoch training with mini-batch validation Adds neural-network-style training controls to the forge loop, inspired by microsoft/SkillOpt: iterate eval -> learn -> validate over multiple epochs, each validating a candidate patch on a rotating mini-batch of golden cases, keeping a single best_skill snapshot promoted only after a full-set gate. Also feeds real execution telemetry (skill_executions) into auto-learn as a trajectory-driven failure signal for root-cause analysis and patch proposals. https://claude.ai/code/session_01EntkmBiYh381pKqvvSFBkg --- src/lib/skills/forge-loop.functions.ts | 338 ++++++++++++++++++------- src/lib/skills/pipelines.server.ts | 23 +- src/routes/admin.packages.tsx | 10 +- 3 files changed, 273 insertions(+), 98 deletions(-) diff --git a/src/lib/skills/forge-loop.functions.ts b/src/lib/skills/forge-loop.functions.ts index 92bad1f4..609ca80e 100644 --- a/src/lib/skills/forge-loop.functions.ts +++ b/src/lib/skills/forge-loop.functions.ts @@ -12,8 +12,52 @@ import { createFeedbackRequest } from "./feedback.functions"; const LoopInput = z.object({ package_slug: z.string(), hotswap: z.boolean().default(false), + // SkillOpt-style training controls: epochs iterate the eval→learn→re-eval + // loop, each epoch validating a patch on a rotating mini-batch of golden + // cases. The single best snapshot across all epochs is promoted at the end. + epochs: z.number().int().min(1).max(6).default(1), + batch_size: z.number().int().min(1).max(20).default(8), + candidates_per_gen: z.number().int().min(1).max(4).default(2), }); +/** Summarize real agent runs into the trajectory-driven learning signal. */ +async function buildExecutionSignal(supabase: any, packageId: string) { + const since = new Date(Date.now() - 30 * 24 * 60 * 60 * 1000).toISOString(); + const { data: rows } = await supabase + .from("skill_executions") + .select("success, error_kind, model") + .eq("package_id", packageId) + .gte("created_at", since) + .limit(2000); + const execs = (rows as Array<{ success: boolean; error_kind: string | null; model: string | null }>) || []; + if (execs.length === 0) return null; + const failures = execs.filter((e) => !e.success).length; + const errorCounts = new Map(); + const modelStats = new Map(); + for (const e of execs) { + if (e.error_kind) errorCounts.set(e.error_kind, (errorCounts.get(e.error_kind) ?? 0) + 1); + const m = e.model ?? "unknown"; + const s = modelStats.get(m) ?? { runs: 0, ok: 0 }; + s.runs += 1; + if (e.success) s.ok += 1; + modelStats.set(m, s); + } + return { + total: execs.length, + failures, + success_rate: Math.round(((execs.length - failures) / execs.length) * 100) / 100, + top_error_kinds: [...errorCounts.entries()] + .sort((a, b) => b[1] - a[1]) + .slice(0, 6) + .map(([kind, count]) => ({ kind, count })), + by_model: [...modelStats.entries()].map(([model, s]) => ({ + model, + runs: s.runs, + success_rate: Math.round((s.ok / s.runs) * 100) / 100, + })), + }; +} + export const runForgeLoop = createServerFn({ method: "POST" }) .middleware([requireSupabaseAuth]) .inputValidator((d: unknown) => LoopInput.parse(d)) @@ -44,7 +88,28 @@ export const runForgeLoop = createServerFn({ method: "POST" }) const goldenCases = (goldenRows as Array<{ title: string; input: string; expected_output: string; label_pass: boolean; label_source: string }>) || []; - // 1) Evaluate current + // Shared learning signals (fetched once for the whole training run). + const { data: metrics } = await supabase + .from("package_metrics_daily") + .select("*") + .eq("package_id", pkg.id) + .order("day", { ascending: false }) + .limit(30); + const { data: learnings } = await supabase + .from("learnings") + .select("kind, evidence, suggested_patch, weight, created_at") + .eq("package_id", pkg.id) + .order("created_at", { ascending: false }) + .limit(80); + const { data: feedback } = await supabase + .from("package_feedback") + .select("source, rating, sentiment, comments, agent_model") + .eq("package_id", pkg.id) + .order("created_at", { ascending: false }) + .limit(120); + const executions = await buildExecutionSignal(supabase, pkg.id); + + // 1) Baseline evaluation on the full golden set. const before = await evaluatorPipeline({ pkg: { name: pkg.name, type: pkg.type }, version: { @@ -74,109 +139,189 @@ export const runForgeLoop = createServerFn({ method: "POST" }) judge_calibration: before.judgeCalibration, }); - // 2) Auto-learn proposal - const { data: metrics } = await supabase - .from("package_metrics_daily") - .select("*") - .eq("package_id", pkg.id) - .order("day", { ascending: false }) - .limit(30); - const { data: learnings } = await supabase - .from("learnings") - .select("kind, evidence, suggested_patch, weight, created_at") - .eq("package_id", pkg.id) - .order("created_at", { ascending: false }) - .limit(80); + // best_skill snapshot — the running champion across all epochs. + type Snapshot = { + system_prompt: string; + rules: unknown; + examples: unknown[]; + score: number; + from_version: string; + rationale: string; + }; + let best: Snapshot = { + system_prompt: ver.system_prompt, + rules: ver.rules, + examples: Array.isArray(ver.examples) ? (ver.examples as unknown[]) : [], + score: before.evaluation.overall_score, + from_version: ver.version, + rationale: "baseline", + }; - const { data: feedback } = await supabase - .from("package_feedback") - .select("source, rating, sentiment, comments, agent_model") - .eq("package_id", pkg.id) - .order("created_at", { ascending: false }) - .limit(120); + // Mini-batch the golden cases per epoch (rotating window) so each epoch + // validates on a different slice — like SGD mini-batches over a dataset. + const batchSize = Math.min(data.batch_size, Math.max(1, goldenCases.length || data.batch_size)); + const miniBatch = (epoch: number) => { + if (goldenCases.length === 0) return []; + const start = (epoch * batchSize) % goldenCases.length; + const slice = goldenCases.slice(start, start + batchSize); + return slice.length >= batchSize ? slice : [...slice, ...goldenCases.slice(0, batchSize - slice.length)]; + }; - const learn = await autoLearnPipeline({ - pkg: { name: pkg.name, type: pkg.type }, - version: { version: ver.version, system_prompt: ver.system_prompt, rules: ver.rules, examples: ver.examples }, - metrics: metrics || [], - learnings: (learnings || []) as Array<{ - kind: string; - evidence: unknown; - suggested_patch: string | null; - weight: number; - created_at: string; - }>, - feedback: (feedback || []) as Array<{ - source: string; - rating: number | null; - sentiment: string | null; - comments: string | null; - agent_model: string | null; - }>, - }); + // Auto-learn proposal is generated fresh each epoch from the running + // champion, so improvements compound. With multiple epochs we run a single + // generation per epoch (the epoch loop is the outer search). + let lastLearn: Awaited> | null = null; + const epochHistory: Array<{ + epoch: number; + batch: number; + regression: boolean; + candidate_score: number | null; + accepted: boolean; + best_score: number; + rationale: string; + }> = []; - // 3) Hot-swap (only if requested AND no regression AND verdict suggests change is needed) - let newVersion: { id: string; version: string } | null = null; - let after: Awaited> | null = null; + for (let epoch = 0; epoch < data.epochs; epoch++) { + const learn = await autoLearnPipeline({ + pkg: { name: pkg.name, type: pkg.type }, + version: { + version: best.from_version, + system_prompt: best.system_prompt, + rules: best.rules, + examples: best.examples, + }, + metrics: metrics || [], + learnings: (learnings || []) as Array<{ + kind: string; + evidence: unknown; + suggested_patch: string | null; + weight: number; + created_at: string; + }>, + feedback: (feedback || []) as Array<{ + source: string; + rating: number | null; + sentiment: string | null; + comments: string | null; + agent_model: string | null; + }>, + executions, + generations: data.epochs > 1 ? 1 : 2, + candidatesPerGen: data.candidates_per_gen, + }); + lastLearn = learn; - const shouldSwap = - data.hotswap && - !learn.regression && - (before.evaluation.verdict !== "ship" || before.evaluation.overall_score < 90); + if (learn.regression) { + epochHistory.push({ + epoch, + batch: batchSize, + regression: true, + candidate_score: null, + accepted: false, + best_score: best.score, + rationale: learn.patch.rationale, + }); + continue; + } - if (shouldSwap) { + // Form the candidate and validate it on this epoch's mini-batch. const mergedRules = JSON.parse( - JSON.stringify({ ...(ver.rules as object), ...(learn.patch.patched_rules as object) }) + JSON.stringify({ ...(best.rules as object), ...(learn.patch.patched_rules as object) }) ); - const existingExamples = Array.isArray(ver.examples) ? (ver.examples as unknown[]) : []; - const mergedExamples = JSON.parse(JSON.stringify([...existingExamples, ...learn.patch.new_examples])); - const { data: nv } = await supabase - .from("package_versions") - .insert({ - package_id: pkg.id, - version: learn.patch.next_version, - status: "beta", - notes: `Forge loop hot-swap · ${learn.patch.rationale}`.slice(0, 500), + const mergedExamples = JSON.parse(JSON.stringify([...best.examples, ...learn.patch.new_examples])); + const batch = miniBatch(epoch); + const candEval = await evaluatorPipeline({ + pkg: { name: pkg.name, type: pkg.type }, + version: { system_prompt: learn.patch.patched_system_prompt, rules: mergedRules, examples: mergedExamples }, + goldenCases: batch.length > 0 ? batch : undefined, + }); + const candScore = candEval.evaluation.overall_score; + const accepted = candScore > best.score; + if (accepted) { + best = { system_prompt: learn.patch.patched_system_prompt, rules: mergedRules, examples: mergedExamples, - compatibility: ver.compatibility, - parent_version_id: ver.id, - }) - .select() - .single(); - newVersion = nv as { id: string; version: string }; + score: candScore, + from_version: learn.patch.next_version, + rationale: learn.patch.rationale, + }; + } + epochHistory.push({ + epoch, + batch: batch.length, + regression: false, + candidate_score: candScore, + accepted, + best_score: best.score, + rationale: learn.patch.rationale, + }); + } - // 4) Re-evaluate the new version + // 3) Promote the champion (validation gate on the FULL golden set). + let newVersion: { id: string; version: string } | null = null; + let after: Awaited> | null = null; + const improved = best.from_version !== ver.version; + const shouldSwap = + data.hotswap && + improved && + (before.evaluation.verdict !== "ship" || before.evaluation.overall_score < 90); + + if (shouldSwap) { + // Final validation on the full golden set before persisting. after = await evaluatorPipeline({ pkg: { name: pkg.name, type: pkg.type }, - version: { - system_prompt: learn.patch.patched_system_prompt, - rules: mergedRules, - examples: mergedExamples, - }, + version: { system_prompt: best.system_prompt, rules: best.rules, examples: best.examples }, goldenCases, }); - await supabase.from("package_evaluations").insert({ - package_id: pkg.id, - version_id: newVersion.id, - triggered_by: userId, - trigger_kind: "loop:after", - overall_score: after.evaluation.overall_score, - precision_score: after.evaluation.precision, - health_score: after.evaluation.health, - hallucination_rate: after.evaluation.hallucination_rate, - safety_score: after.evaluation.safety, - verdict: after.evaluation.verdict, - strengths: after.evaluation.strengths, - weaknesses: after.evaluation.weaknesses, - improvement_actions: after.evaluation.improvement_actions, - example_results: after.evaluation.example_results, - adversarial_results: after.adversarial, - pipeline_stages: after.stages, - judge_calibration: after.judgeCalibration, - evolution_trace: { evolution: learn.evolution, feedback_summary: learn.feedback_summary }, - }); + // Hard gate: never deploy a champion that regresses on the full set. + if (after.evaluation.overall_score >= before.evaluation.overall_score) { + const { data: nv } = await supabase + .from("package_versions") + .insert({ + package_id: pkg.id, + version: best.from_version, + status: "beta", + notes: `Forge loop (${data.epochs} epochs) · ${best.rationale}`.slice(0, 500), + system_prompt: best.system_prompt, + rules: best.rules, + examples: best.examples, + compatibility: ver.compatibility, + parent_version_id: ver.id, + }) + .select() + .single(); + newVersion = nv as { id: string; version: string }; + + await supabase.from("package_evaluations").insert({ + package_id: pkg.id, + version_id: newVersion.id, + triggered_by: userId, + trigger_kind: "loop:after", + overall_score: after.evaluation.overall_score, + precision_score: after.evaluation.precision, + health_score: after.evaluation.health, + hallucination_rate: after.evaluation.hallucination_rate, + safety_score: after.evaluation.safety, + verdict: after.evaluation.verdict, + strengths: after.evaluation.strengths, + weaknesses: after.evaluation.weaknesses, + improvement_actions: after.evaluation.improvement_actions, + example_results: after.evaluation.example_results, + adversarial_results: after.adversarial, + pipeline_stages: after.stages, + judge_calibration: after.judgeCalibration, + evolution_trace: { + evolution: lastLearn?.evolution, + feedback_summary: lastLearn?.feedback_summary, + epoch_history: epochHistory, + executions, + }, + }); + } else { + // Champion failed the full-set gate — discard it. + newVersion = null; + } } const feedback_request = await createFeedbackRequest(supabase, { @@ -192,14 +337,23 @@ export const runForgeLoop = createServerFn({ method: "POST" }) package: pkg, base_version: ver, before: before.evaluation, - patch: learn.patch, - regression: learn.regression, + patch: lastLearn?.patch ?? null, + regression: lastLearn?.regression ?? true, hotswapped: !!newVersion, new_version: newVersion, after: after?.evaluation ?? null, + training: { + epochs: data.epochs, + batch_size: batchSize, + candidates_per_gen: data.candidates_per_gen, + best_score: best.score, + baseline_score: before.evaluation.overall_score, + executions_used: !!executions, + epoch_history: epochHistory, + }, stages: { evaluate: before.stages, - learn: learn.stages, + learn: lastLearn?.stages ?? [], re_evaluate: after?.stages ?? [], }, feedback_request, diff --git a/src/lib/skills/pipelines.server.ts b/src/lib/skills/pipelines.server.ts index db450ced..4660f24b 100644 --- a/src/lib/skills/pipelines.server.ts +++ b/src/lib/skills/pipelines.server.ts @@ -864,12 +864,26 @@ function summarizeFeedback(rows: FeedbackRow[], llmWeight = 2.0, humanWeight = 1 const ELITE_SIZE = 2; +/** + * Execution telemetry signal — derived from real agent runs (skill_executions). + * This is the trajectory-driven signal: failures observed in production carry + * the strongest evidence about what the skill gets wrong in the wild. + */ +export type ExecutionSignal = { + total: number; + failures: number; + success_rate: number | null; + top_error_kinds: Array<{ kind: string; count: number }>; + by_model: Array<{ model: string; runs: number; success_rate: number }>; +}; + export async function autoLearnPipeline(opts: { pkg: { name: string; type: string }; version: { version: string; system_prompt: string; rules: unknown; examples: unknown }; metrics: unknown[]; learnings: Array<{ kind: string; evidence: unknown; suggested_patch: string | null; weight: number; created_at: string }>; feedback?: FeedbackRow[]; + executions?: ExecutionSignal | null; generations?: number; candidatesPerGen?: number; }) { @@ -877,6 +891,11 @@ export async function autoLearnPipeline(opts: { const generations = Math.max(1, Math.min(4, opts.generations ?? 2)); const candidatesPerGen = Math.max(1, Math.min(4, opts.candidatesPerGen ?? 2)); const feedbackSummary = summarizeFeedback(opts.feedback ?? []); + const exec = opts.executions ?? null; + const executionBlock = + exec && exec.total > 0 + ? `\n\nEXECUTION TELEMETRY (real agent runs — strongest failure evidence; ${exec.total} runs, success_rate=${exec.success_rate}):\ntop_error_kinds: ${JSON.stringify(exec.top_error_kinds)}\nby_model: ${JSON.stringify(exec.by_model)}` + : ""; const feedbackBlock = feedbackSummary.count > 0 ? `\n\nCUSTOMER FEEDBACK (LLM feedback weighted ${feedbackSummary.llm_weight}× vs human ${feedbackSummary.human_weight}×; ${feedbackSummary.llm_count}/${feedbackSummary.count} from agents):\nweighted_avg_rating=${feedbackSummary.weighted_avg_rating}\nsentiment(weighted)=${JSON.stringify(feedbackSummary.sentiment_weighted)}\ntop_comments(LLM-first):\n${feedbackSummary.top_comments.map((c) => `- ${c}`).join("\n")}` @@ -899,7 +918,7 @@ export async function autoLearnPipeline(opts: { const { experimental_output } = await generateText({ model: getGatewayModel(DEEP), system: "You are SkillForge Root-Cause Analyst. From learnings, metrics AND customer feedback, isolate root causes (not symptoms). Treat LLM-agent feedback as the strongest signal (the agent directly observed the failure). For each cause, identify which package aspect to change. Output strict JSON.", - prompt: `LEARNINGS (last ${opts.learnings.length}):\n${JSON.stringify(opts.learnings).slice(0, 8000)}\n\nMETRICS:\n${JSON.stringify(opts.metrics).slice(0, 3000)}${feedbackBlock}`, + prompt: `LEARNINGS (last ${opts.learnings.length}):\n${JSON.stringify(opts.learnings).slice(0, 8000)}\n\nMETRICS:\n${JSON.stringify(opts.metrics).slice(0, 3000)}${feedbackBlock}${executionBlock}`, experimental_output: Output.object({ schema: RootCauseSchema }), }); rootCauses = experimental_output.root_causes; @@ -1000,7 +1019,7 @@ ROOT CAUSES: ${JSON.stringify(rootCauses)} CLUSTERS: -${JSON.stringify(clusters)}${feedbackBlock}`; +${JSON.stringify(clusters)}${feedbackBlock}${executionBlock}`; const scoreCandidate = async (gen: number, patch: z.infer): Promise => { const ab: AbRow[] = await Promise.all( diff --git a/src/routes/admin.packages.tsx b/src/routes/admin.packages.tsx index 61ae1fd6..8d55f4ef 100644 --- a/src/routes/admin.packages.tsx +++ b/src/routes/admin.packages.tsx @@ -191,10 +191,12 @@ function PackagesPage() { ))} -
-

Patch rationale

-

{loopResult.data.patch.rationale}

-
+ {loopResult.data.patch && ( +
+

Patch rationale

+

{loopResult.data.patch.rationale}

+
+ )}
Pipeline stages