Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
338 changes: 246 additions & 92 deletions src/lib/skills/forge-loop.functions.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,8 +12,52 @@ import { createFeedbackRequest } from "./feedback.functions";
const LoopInput = z.object({
package_slug: z.string(),
hotswap: z.boolean().default(false),
// SkillOpt-style training controls: epochs iterate the eval→learn→re-eval
// loop, each epoch validating a patch on a rotating mini-batch of golden
// cases. The single best snapshot across all epochs is promoted at the end.
epochs: z.number().int().min(1).max(6).default(1),
batch_size: z.number().int().min(1).max(20).default(8),
candidates_per_gen: z.number().int().min(1).max(4).default(2),
});

/** Summarize real agent runs into the trajectory-driven learning signal. */
async function buildExecutionSignal(supabase: any, packageId: string) {
const since = new Date(Date.now() - 30 * 24 * 60 * 60 * 1000).toISOString();
const { data: rows } = await supabase
.from("skill_executions")
.select("success, error_kind, model")
.eq("package_id", packageId)
.gte("created_at", since)
.limit(2000);
const execs = (rows as Array<{ success: boolean; error_kind: string | null; model: string | null }>) || [];
if (execs.length === 0) return null;
const failures = execs.filter((e) => !e.success).length;
const errorCounts = new Map<string, number>();
const modelStats = new Map<string, { runs: number; ok: number }>();
for (const e of execs) {
if (e.error_kind) errorCounts.set(e.error_kind, (errorCounts.get(e.error_kind) ?? 0) + 1);
const m = e.model ?? "unknown";
const s = modelStats.get(m) ?? { runs: 0, ok: 0 };
s.runs += 1;
if (e.success) s.ok += 1;
modelStats.set(m, s);
}
return {
total: execs.length,
failures,
success_rate: Math.round(((execs.length - failures) / execs.length) * 100) / 100,
top_error_kinds: [...errorCounts.entries()]
.sort((a, b) => b[1] - a[1])
.slice(0, 6)
.map(([kind, count]) => ({ kind, count })),
by_model: [...modelStats.entries()].map(([model, s]) => ({
model,
runs: s.runs,
success_rate: Math.round((s.ok / s.runs) * 100) / 100,
})),
};
}

export const runForgeLoop = createServerFn({ method: "POST" })
.middleware([requireSupabaseAuth])
.inputValidator((d: unknown) => LoopInput.parse(d))
Expand Down Expand Up @@ -44,7 +88,28 @@ export const runForgeLoop = createServerFn({ method: "POST" })
const goldenCases =
(goldenRows as Array<{ title: string; input: string; expected_output: string; label_pass: boolean; label_source: string }>) || [];

// 1) Evaluate current
// Shared learning signals (fetched once for the whole training run).
const { data: metrics } = await supabase
.from("package_metrics_daily")
.select("*")
.eq("package_id", pkg.id)
.order("day", { ascending: false })
.limit(30);
const { data: learnings } = await supabase
.from("learnings")
.select("kind, evidence, suggested_patch, weight, created_at")
.eq("package_id", pkg.id)
.order("created_at", { ascending: false })
.limit(80);
const { data: feedback } = await supabase
.from("package_feedback")
.select("source, rating, sentiment, comments, agent_model")
.eq("package_id", pkg.id)
.order("created_at", { ascending: false })
.limit(120);
const executions = await buildExecutionSignal(supabase, pkg.id);

// 1) Baseline evaluation on the full golden set.
const before = await evaluatorPipeline({
pkg: { name: pkg.name, type: pkg.type },
version: {
Expand Down Expand Up @@ -74,109 +139,189 @@ export const runForgeLoop = createServerFn({ method: "POST" })
judge_calibration: before.judgeCalibration,
});

// 2) Auto-learn proposal
const { data: metrics } = await supabase
.from("package_metrics_daily")
.select("*")
.eq("package_id", pkg.id)
.order("day", { ascending: false })
.limit(30);
const { data: learnings } = await supabase
.from("learnings")
.select("kind, evidence, suggested_patch, weight, created_at")
.eq("package_id", pkg.id)
.order("created_at", { ascending: false })
.limit(80);
// best_skill snapshot — the running champion across all epochs.
type Snapshot = {
system_prompt: string;
rules: unknown;
examples: unknown[];
score: number;
from_version: string;
rationale: string;
};
let best: Snapshot = {
system_prompt: ver.system_prompt,
rules: ver.rules,
examples: Array.isArray(ver.examples) ? (ver.examples as unknown[]) : [],
score: before.evaluation.overall_score,
from_version: ver.version,
rationale: "baseline",
};

const { data: feedback } = await supabase
.from("package_feedback")
.select("source, rating, sentiment, comments, agent_model")
.eq("package_id", pkg.id)
.order("created_at", { ascending: false })
.limit(120);
// Mini-batch the golden cases per epoch (rotating window) so each epoch
// validates on a different slice — like SGD mini-batches over a dataset.
const batchSize = Math.min(data.batch_size, Math.max(1, goldenCases.length || data.batch_size));
const miniBatch = (epoch: number) => {
if (goldenCases.length === 0) return [];
const start = (epoch * batchSize) % goldenCases.length;
const slice = goldenCases.slice(start, start + batchSize);
return slice.length >= batchSize ? slice : [...slice, ...goldenCases.slice(0, batchSize - slice.length)];
};

const learn = await autoLearnPipeline({
pkg: { name: pkg.name, type: pkg.type },
version: { version: ver.version, system_prompt: ver.system_prompt, rules: ver.rules, examples: ver.examples },
metrics: metrics || [],
learnings: (learnings || []) as Array<{
kind: string;
evidence: unknown;
suggested_patch: string | null;
weight: number;
created_at: string;
}>,
feedback: (feedback || []) as Array<{
source: string;
rating: number | null;
sentiment: string | null;
comments: string | null;
agent_model: string | null;
}>,
});
// Auto-learn proposal is generated fresh each epoch from the running
// champion, so improvements compound. With multiple epochs we run a single
// generation per epoch (the epoch loop is the outer search).
let lastLearn: Awaited<ReturnType<typeof autoLearnPipeline>> | null = null;
const epochHistory: Array<{
epoch: number;
batch: number;
regression: boolean;
candidate_score: number | null;
accepted: boolean;
best_score: number;
rationale: string;
}> = [];

// 3) Hot-swap (only if requested AND no regression AND verdict suggests change is needed)
let newVersion: { id: string; version: string } | null = null;
let after: Awaited<ReturnType<typeof evaluatorPipeline>> | null = null;
for (let epoch = 0; epoch < data.epochs; epoch++) {
const learn = await autoLearnPipeline({
pkg: { name: pkg.name, type: pkg.type },
version: {
version: best.from_version,
system_prompt: best.system_prompt,
rules: best.rules,
examples: best.examples,
},
metrics: metrics || [],
learnings: (learnings || []) as Array<{
kind: string;
evidence: unknown;
suggested_patch: string | null;
weight: number;
created_at: string;
}>,
feedback: (feedback || []) as Array<{
source: string;
rating: number | null;
sentiment: string | null;
comments: string | null;
agent_model: string | null;
}>,
executions,
generations: data.epochs > 1 ? 1 : 2,
candidatesPerGen: data.candidates_per_gen,
});
lastLearn = learn;

const shouldSwap =
data.hotswap &&
!learn.regression &&
(before.evaluation.verdict !== "ship" || before.evaluation.overall_score < 90);
if (learn.regression) {
epochHistory.push({
epoch,
batch: batchSize,
regression: true,
candidate_score: null,
accepted: false,
best_score: best.score,
rationale: learn.patch.rationale,
});
continue;
}

if (shouldSwap) {
// Form the candidate and validate it on this epoch's mini-batch.
const mergedRules = JSON.parse(
JSON.stringify({ ...(ver.rules as object), ...(learn.patch.patched_rules as object) })
JSON.stringify({ ...(best.rules as object), ...(learn.patch.patched_rules as object) })
);
const existingExamples = Array.isArray(ver.examples) ? (ver.examples as unknown[]) : [];
const mergedExamples = JSON.parse(JSON.stringify([...existingExamples, ...learn.patch.new_examples]));
const { data: nv } = await supabase
.from("package_versions")
.insert({
package_id: pkg.id,
version: learn.patch.next_version,
status: "beta",
notes: `Forge loop hot-swap · ${learn.patch.rationale}`.slice(0, 500),
const mergedExamples = JSON.parse(JSON.stringify([...best.examples, ...learn.patch.new_examples]));
const batch = miniBatch(epoch);
const candEval = await evaluatorPipeline({
pkg: { name: pkg.name, type: pkg.type },
version: { system_prompt: learn.patch.patched_system_prompt, rules: mergedRules, examples: mergedExamples },
goldenCases: batch.length > 0 ? batch : undefined,
});
const candScore = candEval.evaluation.overall_score;
const accepted = candScore > best.score;
if (accepted) {
best = {
system_prompt: learn.patch.patched_system_prompt,
rules: mergedRules,
examples: mergedExamples,
compatibility: ver.compatibility,
parent_version_id: ver.id,
})
.select()
.single();
newVersion = nv as { id: string; version: string };
score: candScore,
from_version: learn.patch.next_version,
rationale: learn.patch.rationale,
};
}
epochHistory.push({
epoch,
batch: batch.length,
regression: false,
candidate_score: candScore,
accepted,
best_score: best.score,
rationale: learn.patch.rationale,
});
}

// 4) Re-evaluate the new version
// 3) Promote the champion (validation gate on the FULL golden set).
let newVersion: { id: string; version: string } | null = null;
let after: Awaited<ReturnType<typeof evaluatorPipeline>> | null = null;
const improved = best.from_version !== ver.version;
const shouldSwap =
data.hotswap &&
improved &&
(before.evaluation.verdict !== "ship" || before.evaluation.overall_score < 90);

if (shouldSwap) {
// Final validation on the full golden set before persisting.
after = await evaluatorPipeline({
pkg: { name: pkg.name, type: pkg.type },
version: {
system_prompt: learn.patch.patched_system_prompt,
rules: mergedRules,
examples: mergedExamples,
},
version: { system_prompt: best.system_prompt, rules: best.rules, examples: best.examples },
goldenCases,
});
await supabase.from("package_evaluations").insert({
package_id: pkg.id,
version_id: newVersion.id,
triggered_by: userId,
trigger_kind: "loop:after",
overall_score: after.evaluation.overall_score,
precision_score: after.evaluation.precision,
health_score: after.evaluation.health,
hallucination_rate: after.evaluation.hallucination_rate,
safety_score: after.evaluation.safety,
verdict: after.evaluation.verdict,
strengths: after.evaluation.strengths,
weaknesses: after.evaluation.weaknesses,
improvement_actions: after.evaluation.improvement_actions,
example_results: after.evaluation.example_results,
adversarial_results: after.adversarial,
pipeline_stages: after.stages,
judge_calibration: after.judgeCalibration,
evolution_trace: { evolution: learn.evolution, feedback_summary: learn.feedback_summary },
});
// Hard gate: never deploy a champion that regresses on the full set.
if (after.evaluation.overall_score >= before.evaluation.overall_score) {
const { data: nv } = await supabase
.from("package_versions")
.insert({
package_id: pkg.id,
version: best.from_version,
status: "beta",
notes: `Forge loop (${data.epochs} epochs) · ${best.rationale}`.slice(0, 500),
system_prompt: best.system_prompt,
rules: best.rules,
examples: best.examples,
compatibility: ver.compatibility,
parent_version_id: ver.id,
})
.select()
.single();
newVersion = nv as { id: string; version: string };

await supabase.from("package_evaluations").insert({
package_id: pkg.id,
version_id: newVersion.id,
triggered_by: userId,
trigger_kind: "loop:after",
overall_score: after.evaluation.overall_score,
precision_score: after.evaluation.precision,
health_score: after.evaluation.health,
hallucination_rate: after.evaluation.hallucination_rate,
safety_score: after.evaluation.safety,
verdict: after.evaluation.verdict,
strengths: after.evaluation.strengths,
weaknesses: after.evaluation.weaknesses,
improvement_actions: after.evaluation.improvement_actions,
example_results: after.evaluation.example_results,
adversarial_results: after.adversarial,
pipeline_stages: after.stages,
judge_calibration: after.judgeCalibration,
evolution_trace: {
evolution: lastLearn?.evolution,
feedback_summary: lastLearn?.feedback_summary,
epoch_history: epochHistory,
executions,
},
});
} else {
// Champion failed the full-set gate — discard it.
newVersion = null;
}
}

const feedback_request = await createFeedbackRequest(supabase, {
Expand All @@ -192,14 +337,23 @@ export const runForgeLoop = createServerFn({ method: "POST" })
package: pkg,
base_version: ver,
before: before.evaluation,
patch: learn.patch,
regression: learn.regression,
patch: lastLearn?.patch ?? null,
regression: lastLearn?.regression ?? true,
hotswapped: !!newVersion,
new_version: newVersion,
after: after?.evaluation ?? null,
training: {
epochs: data.epochs,
batch_size: batchSize,
candidates_per_gen: data.candidates_per_gen,
best_score: best.score,
baseline_score: before.evaluation.overall_score,
executions_used: !!executions,
epoch_history: epochHistory,
},
stages: {
evaluate: before.stages,
learn: learn.stages,
learn: lastLearn?.stages ?? [],
re_evaluate: after?.stages ?? [],
},
feedback_request,
Expand Down
Loading
Loading