Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
81 changes: 76 additions & 5 deletions crates/flowproof-cli/src/agent_flow.rs
Original file line number Diff line number Diff line change
Expand Up @@ -832,6 +832,18 @@ pub fn containment(spec: &FlowSpec) -> Containment {
/// egress lane to store (record) or discard (replay). Fails - so record mints
/// no trace and replay fails the flow - when `assert_no_egress` cannot be
/// certified or was violated.
/// The tier to REPORT: what the run ACHIEVED where it decided one, and the
/// pre-run prediction only where it did not.
///
/// One definition, used by the certification path and the reporting path
/// alike. Them disagreeing is the failure mode worth designing out: a report
/// that says "not contained" over a run `assert_no_egress` certified would
/// leave an auditor with two artifacts describing the same run differently,
/// and no way to tell which one is lying.
pub fn achieved_tier(run: &AgentRun, spec_tier: &Containment) -> Containment {
run.containment.clone().unwrap_or_else(|| spec_tier.clone())
}
Comment on lines +843 to +845

fn check_egress(
plan: &Plan,
run: &AgentRun,
Expand All @@ -843,7 +855,7 @@ fn check_egress(
// fail to become contained after the probe said yes, and certifying on
// the probe's optimism would be the false green of #300 and #301 arriving
// by prediction rather than by silence.
let containment = run.containment.as_ref().unwrap_or(spec_tier);
let containment = &achieved_tier(run, spec_tier);
// `assert_no_egress` is a CAPABILITY claim: it can only certify where
// containment is actually enforced. There is no bypass flag.
if plan.assert_no_egress && !containment.is_enforced() {
Expand Down Expand Up @@ -1297,7 +1309,22 @@ fn warn_unprotected_tools(spec: &FlowSpec, plan: &Plan, phase: &str) {
}
}

pub fn record(spec: &FlowSpec, out: &Path) -> Result<(), String> {
pub fn record(spec: &FlowSpec, out: &Path) -> (Containment, Result<(), String>) {
let mut achieved = None;
let outcome = record_inner(spec, out, &mut achieved);
(achieved.unwrap_or_else(|| containment(spec)), outcome)
}

/// The body. Takes `achieved` as an out-parameter rather than returning the
/// tier, because the tier becomes known PART WAY THROUGH and every `?` before
/// that point would otherwise have to name a tier it does not have yet. A run
/// that failed before it started reports the prediction, which is the honest
/// answer for a run that achieved nothing.
fn record_inner(
spec: &FlowSpec,
out: &Path,
achieved: &mut Option<Containment>,
) -> Result<(), String> {
let mut plan = plan(spec)?;
warn_unprotected_tools(spec, &plan, "record");
// Set up the MCP boundary BEFORE the agent starts: write the plans and
Expand Down Expand Up @@ -1339,6 +1366,7 @@ pub fn record(spec: &FlowSpec, out: &Path) -> Result<(), String> {
if let Some(warning) = egress_warning(&plan, &tier) {
eprintln!("{warning}");
}
*achieved = Some(achieved_tier(&run, &tier));
let egress = check_egress(&plan, &run, &tier)?;
// The secret-leak scan runs BEFORE the trace is minted: a leak fails the
// run so NO trace is written. That doubles as a store-guard - a secret
Expand All @@ -1360,7 +1388,18 @@ pub fn record(spec: &FlowSpec, out: &Path) -> Result<(), String> {
/// Replay an `app: agent` flow: serve the recorded cassette, run the
/// agent, and check that the trajectory reproduced and the assertions
/// still hold.
pub fn replay(spec: &FlowSpec, trace_path: &Path) -> Result<(), String> {
pub fn replay(spec: &FlowSpec, trace_path: &Path) -> (Containment, Result<(), String>) {
let mut achieved = None;
let outcome = replay_inner(spec, trace_path, &mut achieved);
(achieved.unwrap_or_else(|| containment(spec)), outcome)
}

/// The body; see [`record_inner`] for why the tier leaves by out-parameter.
fn replay_inner(
spec: &FlowSpec,
trace_path: &Path,
achieved: &mut Option<Containment>,
) -> Result<(), String> {
let mut plan = plan(spec)?;
warn_unprotected_tools(spec, &plan, "replay");
let raw = std::fs::read_to_string(trace_path)
Expand Down Expand Up @@ -1406,6 +1445,7 @@ pub fn replay(spec: &FlowSpec, trace_path: &Path) -> Result<(), String> {
if let Some(warning) = egress_warning(&plan, &tier) {
eprintln!("{warning}");
}
*achieved = Some(achieved_tier(&run, &tier));
check_egress(&plan, &run, &tier)?;
// Re-scan the recorded corpus for declared secrets by the SAME mechanism
// as record, so an unchanged system replays the same verdict. The corpus
Expand Down Expand Up @@ -1800,6 +1840,33 @@ mod tests {
}
}

/// The REPORT follows the run, not the probe.
///
/// `check_egress` already certified on this rule. The report line did not,
/// so a Windows run that WAS contained printed "not contained" on stdout
/// and in `--json`, beside a trace lane that said the opposite. Two
/// artifacts describing one run differently is worse than either being
/// wrong alone: an auditor has no way to tell which one is lying.
#[test]
fn the_reported_tier_is_the_one_the_run_achieved() {
let mut run = egress_run(vec![]);
run.containment = Some(Containment::Enforced);
let predicted = Containment::NotContained("this host cannot enforce".into());
assert_eq!(achieved_tier(&run, &predicted), Containment::Enforced);
}

/// And the other direction, which is NOT implied by the one above: where
/// the run decided no tier - a `url:` service, a flow engaging no egress,
/// a platform with no mechanism - the prediction stands. A run that
/// determined nothing must not be read as one that achieved nothing
/// either; those are different sentences.
#[test]
fn a_run_that_decided_no_tier_reports_the_prediction() {
let run = egress_run(vec![]);
let predicted = Containment::NotContained("a service flowproof did not start".into());
assert_eq!(achieved_tier(&run, &predicted), predicted);
}

/// `assert_no_egress` is a CAPABILITY claim: on any tier that is not
/// enforced it fails outright, with no bypass, rather than passing
/// vacuously.
Expand Down Expand Up @@ -2275,7 +2342,9 @@ mod tests {
))
.expect("spec parses");

replay(&spec, &trace).expect("driver-blind replay via url passes");
replay(&spec, &trace)
.1
.expect("driver-blind replay via url passes");
handle.join().ok();
}

Expand All @@ -2291,7 +2360,9 @@ mod tests {
))
.expect("spec parses");

let why = replay(&spec, &trace).expect_err("mispointed service must fail");
let why = replay(&spec, &trace)
.1
.expect_err("mispointed service must fail");
handle.join().ok();
assert!(why.contains("made 0 model calls"), "{why}");
assert!(
Expand Down
26 changes: 20 additions & 6 deletions crates/flowproof-cli/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -298,11 +298,18 @@ fn cmd_record(
// The containment tier prints on EVERY agent run, on every platform,
// pass or fail - computed before the run so it shows even when
// recording errors out.
let tier = agent_flow::containment(&spec);
let predicted = agent_flow::containment(&spec);
if !json {
println!("{}", predicted.report_line());
}
let (tier, outcome) = agent_flow::record(&spec, &out);
Comment on lines +301 to +305
// Reprinted only when the RUN decided a different tier than the probe
// predicted. On Linux they agree by construction, so this is silent;
// on Windows the run is the authority and the line above was a guess.
if !json && tier != predicted {
println!("{}", tier.report_line());
}
agent_flow::record(&spec, &out)?;
outcome?;
if json {
println!(
"{}",
Expand Down Expand Up @@ -706,10 +713,14 @@ fn run_agent_flow_in_suite(
}
// The containment tier prints on every agent run, pass or fail - the
// single-spec path does the same, and a suite must not hide it.
let predicted = agent_flow::containment(spec);
if !json {
println!("{}", agent_flow::containment(spec).report_line());
println!("{}", predicted.report_line());
}
let (tier, outcome) = agent_flow::replay(spec, trace_path);
if !json && tier != predicted {
println!("{}", tier.report_line());
}
let outcome = agent_flow::replay(spec, trace_path);
if let Some(cmd) = &manifest.after_each {
run_hook(cmd, spec_path, "after_each")?;
}
Expand Down Expand Up @@ -1284,11 +1295,14 @@ fn cmd_run(
run_hook(cmd, spec_path, "before_each")?;
}
// The containment tier prints on EVERY agent run, pass or fail.
let tier = agent_flow::containment(&spec);
let predicted = agent_flow::containment(&spec);
if !json {
println!("{}", predicted.report_line());
}
let (tier, outcome) = agent_flow::replay(&spec, &trace_path);
if !json && tier != predicted {
println!("{}", tier.report_line());
}
let outcome = agent_flow::replay(&spec, &trace_path);
if let Some(cmd) = manifest.as_ref().and_then(|m| m.after_each.as_ref()) {
run_hook(cmd, spec_path, "after_each")?;
}
Expand Down
Loading