diff --git a/live/gauntlet.json b/live/gauntlet.json index cf2bf952a9..333f0a6bce 100644 --- a/live/gauntlet.json +++ b/live/gauntlet.json @@ -151,7 +151,7 @@ "all": { "label": "All estates", "estates": 27, - "clear": 0, + "clear": 27, "stages": { "cold_deploy": { "pass": 27, @@ -164,8 +164,8 @@ "not_run": 0 }, "day2_crash": { - "pass": 1, - "fail": 0, + "pass": 0, + "fail": 1, "not_run": 26 }, "day2_remove": { @@ -204,9 +204,9 @@ "not_run": 0 }, "plan_approval": { - "pass": 0, + "pass": 27, "fail": 0, - "not_run": 27 + "not_run": 0 }, "strict": { "pass": 1, @@ -228,7 +228,7 @@ "core": { "label": "Core estates", "estates": 26, - "clear": 0, + "clear": 26, "stages": { "cold_deploy": { "pass": 26, @@ -241,8 +241,8 @@ "not_run": 0 }, "day2_crash": { - "pass": 1, - "fail": 0, + "pass": 0, + "fail": 1, "not_run": 25 }, "day2_remove": { @@ -281,9 +281,9 @@ "not_run": 0 }, "plan_approval": { - "pass": 0, + "pass": 26, "fail": 0, - "not_run": 26 + "not_run": 0 }, "strict": { "pass": 1, @@ -324,16 +324,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "cb5ae2009fb85fdafd76126bb1842144855eb9d7", - "date": "2026-09-06T04:29:56Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -342,26 +342,28 @@ "exit_code": 0, "detail": { "cold_deploy": "80 resources, once for real (floci fixes #58, #61, #62)", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.) and left count_test[0] untouched; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.); the next plan is empty. Read back through the AWS CLI, never through choudoufu's own report: the destroyed count_test[1] (sg-b60954438b88ecf3d) is absent after the scale-down and comes back under a NEW server-minted GroupId (sg-2e48714be247a863a) carrying tofu-address=aws_security_group.count_test:1, while the survivor count_test[0] keeps BOTH its GroupId (sg-7cd7717aa835c118a) and its tofu-address=aws_security_group.count_test:0 marker across both moves. What witnesses the destroy was measured against the pinned emulator with no terraform in the loop before this section was written: floci sha256:c55d74e1 never reuses a deleted group's GroupId, not even for the same group-name in the same VPC. G-ORACLE is a real stock oracle for the same shape - stock never had this count block, so it was stood up for real with the stock terraform binary in its own working directory and state against its own idle account on :30400 - and stock shows the IDENTICAL shape: destroy the higher index only (sg-0b3d6a90645f55505, then absent), create it back under a new id (sg-11229066e226420ec), count_test[0]=sg-b162ecb6edc79508e unchanged throughout (stock's own plan lines: \"Plan: 0 to add, 0 to change, 1 to destroy.\" down, \"Plan: 1 to add, 0 to change, 0 to destroy.\" up). SYNTHETIC BLOCK, and why: this estate declares no scalable count of its own - examples/complete-alb has no root-level count or for_each at all, terraform-aws-alb's only count is its `local.create ? 1 : 0` boolean create toggle, and everything the module fans out (6 listeners, 7 listener rules, 3 target groups, 2 security-group ingress rules) is for_each over a STRING-KEYED map, which has no index whose slot binding could be scaled and over which \"the higher index is destroyed\" is not expressible - so the sanctioned fallback applies (reference-ec2-vpc Part F, corpus-iam-policy Part G), with aws_security_group, a type this estate already exercises. BREAK_COUNT=1 confirms the check has teeth: asserting the WRONG instance (count_test[0]) was destroyed makes this stage report fail. One emulator divergence noted in passing, harmless to this stage and asserted around rather than papered over: floci answers DescribeSecurityGroups for a deleted group-id with an empty list and exit 0, where real EC2 raises InvalidGroup.NotFound.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.) and left count_test[0] untouched; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.); the next plan is empty. Read back through the AWS CLI, never through choudoufu's own report: the destroyed count_test[1] (sg-c58a962fd8cab5c67) is absent after the scale-down and comes back under a NEW server-minted GroupId (sg-942bc41c28cb76287) carrying tofu-address=aws_security_group.count_test:1, while the survivor count_test[0] keeps BOTH its GroupId (sg-50717691c349face1) and its tofu-address=aws_security_group.count_test:0 marker across both moves. What witnesses the destroy was measured against the pinned emulator with no terraform in the loop before this section was written: floci sha256:c55d74e1 never reuses a deleted group's GroupId, not even for the same group-name in the same VPC. G-ORACLE is a real stock oracle for the same shape - stock never had this count block, so it was stood up for real with the stock terraform binary in its own working directory and state against its own idle account on :20400 - and stock shows the IDENTICAL shape: destroy the higher index only (sg-6a3ec8b4a38b489e9, then absent), create it back under a new id (sg-6c6c45f278c454454), count_test[0]=sg-69fe659703f0b86cb unchanged throughout (stock's own plan lines: \"Plan: 0 to add, 0 to change, 1 to destroy.\" down, \"Plan: 1 to add, 0 to change, 0 to destroy.\" up). SYNTHETIC BLOCK, and why: this estate declares no scalable count of its own - examples/complete-alb has no root-level count or for_each at all, terraform-aws-alb's only count is its `local.create ? 1 : 0` boolean create toggle, and everything the module fans out (6 listeners, 7 listener rules, 3 target groups, 2 security-group ingress rules) is for_each over a STRING-KEYED map, which has no index whose slot binding could be scaled and over which \"the higher index is destroyed\" is not expressible - so the sanctioned fallback applies (reference-ec2-vpc Part F, corpus-iam-policy Part G), with aws_security_group, a type this estate already exercises. BREAK_COUNT=1 confirms the check has teeth: asserting the WRONG instance (count_test[0]) was destroyed makes this stage report fail. One emulator divergence noted in passing, harmless to this stage and asserted around rather than papered over: floci answers DescribeSecurityGroups for a deleted group-id with an empty list and exit 0, where real EC2 raises InvalidGroup.NotFound.", "day2_remove": "choudoufu: deleting aws_instance.other_renamed's block (and its one target-group-attachment reference) proposed exactly 2 destroys (0 add, 0 change, 2 destroy), matching the stock oracle and applied cleanly; the instance is confirmed terminated via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 2 destroys for the same two objects", "day2_rename": "moved block: aws_instance.this renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_instance.other renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state (positioned right after stage 1, before migrate ever touches these shared objects) also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing aws_instance.this_renamed's ForceNew ami argument proposed a forced replace at the same declared address (Plan: 2 to add, 0 to change, 2 to destroy.), applied cleanly; the old instance (i-c13744b209879be03) is confirmed terminated and the new instance (i-574774864b2064b60) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-c13744b209879be03 -> i-574774864b2064b60); the next plan proposes no resource action; stock oracle on cold_deploy's own state (day2_replace ORACLE) also proposes replacing aws_instance.this at the same address (plan only, not applied - it shares floci's account with $ADOPTED_EST); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one address\", naming both live instances) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", + "day2_replace": "choudoufu: changing aws_instance.this_renamed's ForceNew ami argument proposed a forced replace at the same declared address (Plan: 2 to add, 0 to change, 2 to destroy.), applied cleanly; the old instance (i-c6b2f684d28bd4e21) is confirmed terminated and the new instance (i-3079cc0afccd4830b) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-c6b2f684d28bd4e21 -> i-3079cc0afccd4830b); the next plan proposes no resource action; stock oracle on cold_deploy's own state (day2_replace ORACLE) also proposes replacing aws_instance.this at the same address (plan only, not applied - it shares floci's account with $ADOPTED_EST); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one address\", naming both live instances) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one object tampered (the ALB's Example tag), plan proposed fixing exactly module.alb.aws_lb.this[0], apply changed 1 and the Example tag reconverged", "greenfield": "80 resources from nothing, matching stock's own cold-deploy count; the ALB's markers verified via the AWS CLI; 80 records in the local record store including untaggable types; replan empty; a representative EC2 instance's own shape (type/ami) matches stock's cold deploy, via the AWS CLI on both endpoints, marker tags never compared; 50 objects carry the estate tag", "migrate": "51 of 80 stamped, 1 recorded, 0 failed, 28 skipped", + "plan_approval": "one argument edited (the \"ex-instance\" target group's own InstanceTargetGroupTag tag, baz -> reviewed - the one tags argument in examples/complete-alb that reaches exactly one instance, with no dependent resource or data source behind it), \"plan -out=approved.tfplan\" wrote a 159425-byte stock-format plan file whose whole change set is one update on module.alb.aws_lb_target_group.this[\"ex-instance\"]; the world then moved out of band (arn:aws:elasticloadbalancing:eu-west-1:000000000000:loadbalancer/app/ex-complete-alb/ca7ef1d7df57446c's Example tag, through the AWS CLI, never through choudoufu - the same mutation stage 5 uses) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.alb.aws_lb.this[0] and the live arn:aws:elasticloadbalancing:eu-west-1:000000000000:loadbalancer/app/ex-complete-alb/ca7ef1d7df57446c it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - arn:aws:elasticloadbalancing:eu-west-1:000000000000:targetgroup/h115cad549492d8241963b49d84a/02c64f3ef264487c still read InstanceTargetGroupTag=baz through elbv2 describe-tags, not from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the ALB's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:elasticloadbalancing:eu-west-1:000000000000:targetgroup/h115cad549492d8241963b49d84a/02c64f3ef264487c read back with InstanceTargetGroupTag=reviewed, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); 50 tofu-estate-tagged objects before, 50 after", "test_plan": "empty live-plan with no state file; 0 Error diagnostics; the two record-rung aws_route53_record.validation identities verified by value against route53 list-resource-record-sets" }, - "duration_s": 453.8, + "duration_s": 436.2, "stage_seconds": { - "cold_deploy": 104, - "day2_count": 64, - "day2_remove": 23, - "day2_rename": 17, - "day2_replace": 33, - "drift_reconverge": 38, - "greenfield": 98, - "migrate": 67, + "cold_deploy": 88, + "day2_count": 48, + "day2_remove": 22, + "day2_rename": 16, + "day2_replace": 32, + "drift_reconverge": 39, + "greenfield": 95, + "migrate": 65, + "plan_approval": 22, "test_apply": 5, "test_plan": 4 } @@ -388,16 +390,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -406,27 +408,29 @@ "exit_code": 0, "detail": { "cold_deploy": "Apply complete! Resources: 68 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=autoscaling-complete-crossing before migration", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) and it was the HIGHER index, count_test[1] (sg-751e8f493815cc85a); count_test[0] (sg-278ee188356fcd904) kept its live GroupId, its tofu-address=aws_security_group.count_test:0 marker and its local record across the move, all re-read through the AWS CLI and off the record-store file rather than out of choudoufu's own report. Scaling back from 1 to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy) and count_test[1] came back as a genuinely NEW object (sg-751e8f493815cc85a -> sg-03e3220bce1d89e1b) carrying tofu-address=aws_security_group.count_test:1, with its record re-keyed to the new id rather than left stale on the destroyed one; count_test[0] was untouched throughout and the next plan is empty. Stock oracle (G0): a plain terraform working directory standing the identical 2-instance block up for real in its own VPC and scaling it the same way shows the identical shape - destroy count_test[1] (sg-197c7dafd5ec22f07) only on the way down, create count_test[1] back under a new id (sg-60fdd9f4aa9f57fce) on the way up, count_test[0] (sg-ac913a1b237fb74e4) untouched both times - and was torn down before the choudoufu leg ran. SYNTHETIC BLOCK, and why: this estate declares no scalable count anywhere - the module's own aws_autoscaling_group.this/.idc pair is a 0-or-1 boolean toggle, the schedules/policies/traffic-source attachments are for_each over name-keyed maps, and none of the twelve module calls carries count or for_each - so this is live/GAUNTLET.md #8's sanctioned self-contained fallback, the same one reference-ec2-vpc's Part F and corpus-iam-policy's Part G use, on aws_security_group, a type this estate already exercises through module.asg_sg. An ASG's own desired_capacity is deliberately NOT what was scaled: this stage is about the count meta-argument's slot binding (internal/live/discovery/count.go), not about a live group's capacity. BREAK_COUNT=1 confirms the check has teeth: expecting count_test[0] to be the destroyed instance makes this stage report fail.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) and it was the HIGHER index, count_test[1] (sg-be0d8c436b598d206); count_test[0] (sg-87554ef6a11f83aab) kept its live GroupId, its tofu-address=aws_security_group.count_test:0 marker and its local record across the move, all re-read through the AWS CLI and off the record-store file rather than out of choudoufu's own report. Scaling back from 1 to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy) and count_test[1] came back as a genuinely NEW object (sg-be0d8c436b598d206 -> sg-530659bb51950b409) carrying tofu-address=aws_security_group.count_test:1, with its record re-keyed to the new id rather than left stale on the destroyed one; count_test[0] was untouched throughout and the next plan is empty. Stock oracle (G0): a plain terraform working directory standing the identical 2-instance block up for real in its own VPC and scaling it the same way shows the identical shape - destroy count_test[1] (sg-b3bf410755a7c74b5) only on the way down, create count_test[1] back under a new id (sg-55b912566a9fb1171) on the way up, count_test[0] (sg-69115075c0a480bab) untouched both times - and was torn down before the choudoufu leg ran. SYNTHETIC BLOCK, and why: this estate declares no scalable count anywhere - the module's own aws_autoscaling_group.this/.idc pair is a 0-or-1 boolean toggle, the schedules/policies/traffic-source attachments are for_each over name-keyed maps, and none of the twelve module calls carries count or for_each - so this is live/GAUNTLET.md #8's sanctioned self-contained fallback, the same one reference-ec2-vpc's Part F and corpus-iam-policy's Part G use, on aws_security_group, a type this estate already exercises through module.asg_sg. An ASG's own desired_capacity is deliberately NOT what was scaled: this stage is about the count meta-argument's slot binding (internal/live/discovery/count.go), not about a live group's capacity. BREAK_COUNT=1 confirms the check has teeth: expecting count_test[0] to be the destroyed instance makes this stage report fail.", "day2_remove": "choudoufu: deleting module.default's block proposed exactly 2 destroys (0 add, 0 change, 2 destroy), matching the stock oracle's own count and applied cleanly; the live ASG count dropped by exactly one and the tagged object count dropped too, both confirmed via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 2 destroys for the same module", "day2_rename": "moved block: module.asg_sg renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place on its security group; live-mv: aws_sqs_queue.this renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.asg_sg_renamed's ForceNew name argument (module CALL, passed through to its own aws_security_group.this_name_prefix's name_prefix) proposed a forced replace at the same declared address (Plan: 3 to add, 2 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-c37bb9262afc41e48) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-05ed298dde97955f0 -> sg-c37bb9262afc41e48); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 2 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted aws_sqs_queue.this_renamed and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on this branch, GitHub issue #412: propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard; see this section's own header comment for the fix and eks-basic's/ecs-fargate's matching ones in this same unit, which independently hit the identical shape and were not re-run for #412.", + "day2_replace": "choudoufu: changing module.asg_sg_renamed's ForceNew name argument (module CALL, passed through to its own aws_security_group.this_name_prefix's name_prefix) proposed a forced replace at the same declared address (Plan: 3 to add, 2 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-091e8a0748dde3c56) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-e756cc13263688dd3 -> sg-091e8a0748dde3c56); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 2 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted aws_sqs_queue.this_renamed and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on this branch, GitHub issue #412: propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard; see this section's own header comment for the fix and eks-basic's/ecs-fargate's matching ones in this same unit, which independently hit the identical shape and were not re-run for #412.", "drift_reconverge": "one object tampered (SQS queue 'complete's Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag", "greenfield": "68 resources from nothing, matching stock's own cold-deploy count (68); the sqs queue's markers verified via the AWS CLI; 68 records in the local record store including the untaggable ASGs (#364 A2); replan empty; the asg_sg security group's rule counts match stock's cold deploy structurally, via the AWS CLI on both endpoints, marker tags never compared; 41 objects carry the estate tag", "migrate": "41 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 re-recorded for sensitivity only, 0 already recorded, 0 failed, 27 skipped; 41 objects carry tofu-estate=autoscaling-complete-crossing", + "plan_approval": "one argument edited (aws_iam_role.ssm's tags gain Reviewed=yes - a single instance whose only dependent reads its name, which an in-place tag update leaves known), \"plan -out=approved.tfplan\" wrote a 275504-byte stock-format plan file whose whole change set is one update on aws_iam_role.ssm; the world then moved out of band (the SQS queue's Example tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both aws_sqs_queue.this and the live https://sqs.eu-west-1.amazonaws.com/000000000000/complete it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - IAM role complete still carried no Reviewed tag, read back through iam list-role-tags rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the queue's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the role read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 41 objects before, 41 after, no state file either time", "test_plan": "empty plan; identity re-check unchanged: module.complete.aws_launch_template.this:0, aws_iam_role.ssm" }, - "duration_s": 417.3, + "duration_s": 412.6, "stage_seconds": { - "cold_deploy": 105, - "day2_count": 57, - "day2_remove": 21, - "day2_rename": 18, + "cold_deploy": 87, + "day2_count": 55, + "day2_remove": 20, + "day2_rename": 17, "day2_replace": 18, - "drift_reconverge": 8, + "drift_reconverge": 10, "greenfield": 99, - "migrate": 81, - "test_apply": 5, + "migrate": 77, + "plan_approval": 21, + "test_apply": 4, "test_plan": 4 } }, @@ -452,16 +456,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -473,24 +477,26 @@ "day2_count": "choudoufu: scaling the synthetic aws_dynamodb_table.count_test from 2 to 1 (issue #359/#488's own fallback clause - this estate's real module has no honest resource-level count/for_each knob: create_table is boolean-shaped and replica_regions/global_secondary_indexes drive dynamic blocks nested inside the SAME table resource, not a separate resource instance, confirmed by reading main.tf directly) destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), confirmed gone via the AWS CLI, its local record correctly tombstoned rather than left claiming a live identity (#398-guard shape, has(tombstone) and not has(identity)), and left count_test[0]'s live TableId and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] again under the SAME ARN (deterministic from region+account+name - established directly against floci with no tofu in the loop before writing this assertion) but a NEW TableId (0 add -> 1 add, 0 change, 0 destroy), and its local record returned to a live identity, while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical 2-instance count block, applied for real in a dedicated always-idle account never shared with this one, shows the identical shape: destroy the higher index only, create it back under the same ARN but a new TableId, the lower index's TableId unchanged both times. BREAK_COUNT=1 confirms the wrong-instance assertion correctly fails to hold.", "day2_remove": "choudoufu: deleting module.dynamodb_table_final's block proposed exactly two destroys (0 add, 0 change, 2 destroy: the table and its untaggable resource policy), applied cleanly (0 added, 0 changed, 2 destroyed), the table is genuinely gone from the live account (dynamodb describe-table on the old name now returns ResourceNotFoundException, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly two destroys for the same two objects; classifyOrphans did not withhold either destroy because module.disabled_dynamodb_table declares zero instances of the same block key (create_table=false), so nothing is ever pending against it", "day2_rename": "moved block: module.dynamodb_table renamed to module.dynamodb_table_moved with zero churn (0 add, 1 change, 0 destroy) - the table's own marker rewritten in place, the untaggable resource policy unaffected; live-mv: module.dynamodb_table_moved renamed to module.dynamodb_table_final with zero churn, marker rewritten in place; stock oracle over the same net module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); the table's ARN unchanged throughout, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.dynamodb_table_final's ForceNew name argument proposed exactly one table replace at the same declared address, cascading into the untaggable resource policy (its resource_arn argument follows the table's ARN and is not independently updatable, so it also replaces - F-ORACLE's own finding); applied cleanly; the old table is confirmed gone via the AWS CLI (ResourceNotFoundException) and the new table carries the marker; the local record store's record at the same address now names the new table's name, not the destroyed one (my-table-clear-horse -> my-table-clear-horse-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the table at the same address (plan only, not applied - it shares floci's account with $EST); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones.", - "drift_reconverge": "one object tampered (arn:aws:dynamodb:eu-west-1:000000000000:table/my-table-clear-horse's Terraform tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag", + "day2_replace": "choudoufu: changing module.dynamodb_table_final's ForceNew name argument proposed exactly one table replace at the same declared address, cascading into the untaggable resource policy (its resource_arn argument follows the table's ARN and is not independently updatable, so it also replaces - F-ORACLE's own finding); applied cleanly; the old table is confirmed gone via the AWS CLI (ResourceNotFoundException) and the new table carries the marker; the local record store's record at the same address now names the new table's name, not the destroyed one (my-table-probable-spaniel -> my-table-probable-spaniel-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the table at the same address (plan only, not applied - it shares floci's account with $EST); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones.", + "drift_reconverge": "one object tampered (arn:aws:dynamodb:eu-west-1:000000000000:table/my-table-probable-spaniel's Terraform tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag", "greenfield": "3 resources from nothing (random_pet + table + resource policy), the table's markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on key schema/attributes/table class/deletion protection/on-demand billing/GSI/resource policy", "migrate": "1 resource(s) newly stamped, 0 already stamped, 1 newly recorded, 0 re-recorded for sensitivity only, 0 already recorded, 0 failed, 1 skipped.; Apply complete! Resources: 0 added, 0 changed, 1 destroyed. (tofu-slot convergence)", + "plan_approval": "one argument edited (the resource_policy heredoc's statement Sid, AllowDummyRoleAccess -> AllowDummyRoleAccessReviewed, reaching only module.dynamodb_table.aws_dynamodb_resource_policy.this[0]), \"plan -out=approved.tfplan\" wrote a 21195-byte stock-format plan file whose whole change set is that one update; the world then moved out of band (arn:aws:dynamodb:eu-west-1:000000000000:table/my-table-probable-spaniel's Terraform tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.dynamodb_table.aws_dynamodb_table.this[0] and the live table id my-table-probable-spaniel it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - dynamodb get-resource-policy still returned a policy without the reviewed Sid, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the table's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the live policy read back WITH the reviewed Sid, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. This estate owns only two objects the plan acts on, so the review deliberately sits on the resource policy and the move on the table: that is what makes the refusal an EXTRA row it can name rather than a values-only disagreement about one row. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 1 objects before, 1 after, no state file either time", "test_plan": "no resource change proposed, nothing foreign; identity re-check (via the AWS CLI) unchanged" }, - "duration_s": 242.5, + "duration_s": 243.8, "stage_seconds": { - "cold_deploy": 35, + "cold_deploy": 23, "day2_count": 31, "day2_remove": 6, "day2_rename": 11, "day2_replace": 13, "drift_reconverge": 6, - "greenfield": 45, - "migrate": 90, - "test_apply": 3, + "greenfield": 46, + "migrate": 91, + "plan_approval": 13, + "test_apply": 2, "test_plan": 2 } }, @@ -516,16 +522,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -534,26 +540,28 @@ "exit_code": 0, "detail": { "cold_deploy": "35 resources added across 13 types (aws_instance, aws_eip, aws_iam_role/instance_profile/role_policy_attachment, aws_ebs_volume, aws_volume_attachment, aws_security_group x2, aws_vpc_security_group_egress_rule x2, aws_security_group_rule x2, vpc/subnet/route*/igw/default_* from the vpc module), 0 objects carry tofu-estate before migration", - "day2_count": "choudoufu: scaling aws_ebs_volume.count_test from 2 to 1 destroyed exactly count_test[1] (vol-5d2c1b342f912c2f6, 0 add, 0 change, 1 destroy) and left count_test[0] (vol-a2f08d6e0a3772fd1) with the same live VolumeId, the same tofu-address=aws_ebs_volume.count_test:0 and the same tofu-slot=0, all read back through the AWS CLI rather than choudoufu's own report; the destroyed volume is genuinely gone (describe-volumes answers InvalidVolume.NotFound for it). Scaling back from 1 to 2 planned exactly 1 to add, 0 to change, 0 to destroy and brought count_test[1] back as a NEW object (vol-0778a981679882f57, not vol-5d2c1b342f912c2f6) carrying tofu-address=aws_ebs_volume.count_test:1 and tofu-slot=1, above the live high-water mark count_test[0] still holds, while count_test[0] stayed untouched throughout; the next plan is empty, and scaling the block to zero destroys both and leaves the estate planning empty again. C-ORACLE, the same 2-instance block stood up for real with plain terraform in its own working directory at the SAME resolved provider version (6.63.0), shows the identical shape: destroy the higher index only (vol-6c64f51b35c8db3ec), create the higher index back under a new id (vol-a8a9c51961c08f115), the lower index's id (vol-eee10750df38c22af) unchanged both times. SYNTHETIC BLOCK, and why: terraform-aws-ec2-instance v6.4.0 declares no scalable count or for_each knob this estate reaches - all nine of its own count usages are boolean create toggles of the form 'count = local.create ? 1 : 0', which can never hold two instances, and the upstream example's one real for_each fan-out (module.ec2_multiple) is dropped by this script's reduction because floci does not model the surfaces around it - so this section adds a new, self-contained count block of a type the estate ALREADY exercises (aws_ebs_volume, module.ec2_complete's own /dev/sdf data volume), the sanctioned fallback live/GAUNTLET.md #8 names, with reference-ec2-vpc Part F and corpus-iam-policy Part G as precedent. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports day2_count fail, proving the assertion is load-bearing.", + "day2_count": "choudoufu: scaling aws_ebs_volume.count_test from 2 to 1 destroyed exactly count_test[1] (vol-227203145b6519f82, 0 add, 0 change, 1 destroy) and left count_test[0] (vol-1ba72164534af9b35) with the same live VolumeId, the same tofu-address=aws_ebs_volume.count_test:0 and the same tofu-slot=0, all read back through the AWS CLI rather than choudoufu's own report; the destroyed volume is genuinely gone (describe-volumes answers InvalidVolume.NotFound for it). Scaling back from 1 to 2 planned exactly 1 to add, 0 to change, 0 to destroy and brought count_test[1] back as a NEW object (vol-25211d5ff11f1d3d0, not vol-227203145b6519f82) carrying tofu-address=aws_ebs_volume.count_test:1 and tofu-slot=1, above the live high-water mark count_test[0] still holds, while count_test[0] stayed untouched throughout; the next plan is empty, and scaling the block to zero destroys both and leaves the estate planning empty again. C-ORACLE, the same 2-instance block stood up for real with plain terraform in its own working directory at the SAME resolved provider version (6.63.0), shows the identical shape: destroy the higher index only (vol-43eecd95f10607c2e), create the higher index back under a new id (vol-d1189cb4ab1b4dbf8), the lower index's id (vol-63f626cc1ed09eb2e) unchanged both times. SYNTHETIC BLOCK, and why: terraform-aws-ec2-instance v6.4.0 declares no scalable count or for_each knob this estate reaches - all nine of its own count usages are boolean create toggles of the form 'count = local.create ? 1 : 0', which can never hold two instances, and the upstream example's one real for_each fan-out (module.ec2_multiple) is dropped by this script's reduction because floci does not model the surfaces around it - so this section adds a new, self-contained count block of a type the estate ALREADY exercises (aws_ebs_volume, module.ec2_complete's own /dev/sdf data volume), the sanctioned fallback live/GAUNTLET.md #8 names, with reference-ec2-vpc Part F and corpus-iam-policy Part G as precedent. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports day2_count fail, proving the assertion is load-bearing.", "day2_remove": "choudoufu: deleting module.ec2_complete's block proposed exactly 10 destroys (0 add, 0 change, 10 destroy), matching the stock oracle's own count and applied cleanly; the instance is confirmed terminated and the tagged object count dropped, both via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 10 destroys for the same module", "day2_rename": "moved block: module.vpc renamed with zero churn (0 add, 15 change, 0 destroy), marker rewritten in place; live-mv: module.security_group's security group renamed with zero churn, its two untaggable rules followed for free; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.ec2_complete's ForceNew ami argument proposed exactly one instance replace at the same declared address, cascading into the eip (updated in-place) and the volume attachment (also replaced, instance_id is ForceNew there too) - 2 to add, 1 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated and the new instance carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-b4eb27b23605d3e67 -> i-7a9aa80c70f73af43); the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one.", + "day2_replace": "choudoufu: changing module.ec2_complete's ForceNew ami argument proposed exactly one instance replace at the same declared address, cascading into the eip (updated in-place) and the volume attachment (also replaced, instance_id is ForceNew there too) - 2 to add, 1 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated and the new instance carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-c7a93b787f29f6013 -> i-2bfd2886f8aefa0fd); the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Indistinguishable instances without per-instance markers\", naming both live instances) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one.", "drift_reconverge": "one object tampered, exactly 1 object proposed and applied (0 added, 1 changed, 0 destroyed), tag reconverged to \"ex-complete\"", "greenfield": "35 resources from nothing, matching stock's own cold-deploy count; the instance's markers verified via the AWS CLI; 35 records in the local record store including untaggable types; replan empty; the instance's own shape (type/ami/block-device-count) matches stock's cold deploy, via the AWS CLI on both endpoints, marker tags never compared; 24 objects carry the estate tag", "migrate": "24 of 35 eligible (11 untaggable across 5 types - aws_iam_role_policy_attachment, aws_volume_attachment, aws_security_group_rule x2, aws_route, aws_route_table_association x6 - all resolved by provider identity schema), 24 stamped, 0 failed, 11 skipped; the IAM role policy attachment's composite live id asserted by value; genuine no-op on the follow-up apply", + "plan_approval": "one argument edited (the \"/dev/sdf\" entry's MountPoint volume tag inside module \"ec2_complete\"'s ebs_volumes argument, /mnt/data -> /mnt/data-reviewed - the module merges each entry's tags into that entry's aws_ebs_volume alone, so it reaches module.ec2_complete.aws_ebs_volume.this[\"/dev/sdf\"] and nothing else), \"plan -out=approved.tfplan\" wrote a 72468-byte stock-format plan file whose whole change set is that one update; the world then moved out of band (i-c7a93b787f29f6013's Example tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.ec2_complete.aws_instance.this[0] and the live i-c7a93b787f29f6013 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - volume vol-877db6c95f823e408 still read MountPoint=/mnt/data through ec2 describe-tags, not from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with i-c7a93b787f29f6013's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and vol-877db6c95f823e408 read back with MountPoint=/mnt/data-reviewed, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART C starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); 24 objects before, 24 after, no state file", "test_plan": "no resource change proposed by either plan; the default plan reports \"nothing was swept\" (the CollectUnclaimed ruling (#604) made the account-inventory question opt-in, and a run that did not ask must say so), and a second plan run with TOFU_LIVE_COLLECT_UNCLAIMED=1 finds exactly 8 foreign objects - the instance's own root volume plus floci's default-VPC bootstrap; instance tofu-address re-checked against EC2" }, - "duration_s": 422.7, + "duration_s": 398.5, "stage_seconds": { - "cold_deploy": 78, - "day2_count": 131, - "day2_remove": 30, - "day2_rename": 17, + "cold_deploy": 56, + "day2_count": 127, + "day2_remove": 28, + "day2_rename": 14, "day2_replace": 50, "drift_reconverge": 6, - "greenfield": 71, + "greenfield": 61, "migrate": 30, + "plan_approval": 17, "test_apply": 4, "test_plan": 5 } @@ -579,16 +587,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "3977d907842a4cefa8e5fbe881732571988f4fd5", - "date": "2026-09-06T05:49:39Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -597,28 +605,30 @@ "exit_code": 0, "detail": { "cold_deploy": "62 resources, once for real", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.), leaving count_test[0]'s live GroupId (sg-38d62e21de06f4816), its tofu-address (aws_security_group.count_test:0) and its tofu-slot (0) unchanged, and count_test[1] (sg-6b3d7970b4dc6d8b9) genuinely absent - all read through the AWS CLI, never choudoufu's own report; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.) under a NEW server-minted GroupId (sg-49a87d7c99c8dbff3, was sg-6b3d7970b4dc6d8b9) carrying tofu-address=aws_security_group.count_test:1 and the same tofu-slot=1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (H-ORACLE), the identical 2-instance block stood up for real with plain terraform in its own account and its own working directory - stock never had this block, so unlike day2_rename/day2_remove/day2_replace there is no cold_deploy state to reuse - shows the identical shape: Plan: 0 to add, 0 to change, 1 to destroy. destroying count_test[1]=sg-703e202276aee9559 only, then Plan: 1 to add, 0 to change, 0 to destroy. bringing it back under a new GroupId (sg-ae896ab2a2a7983c6), with count_test[0]=sg-3b2059c3f4807ab3a unchanged both times. SYNTHETIC block, and why: this estate's corpus example (terraform-aws-modules/terraform-aws-ecs v7.6.0, examples/fargate) declares no scalable count knob at all - every count in its vendored modules is a 'count = local.create ? 1 : 0' boolean create toggle, which cannot hold two instances - and the only construct holding three, module.vpc's private_subnets, is driven by local.azs, which the ECS service's network configuration, the ALB and the NAT gateway all read, so scaling it churns a dozen unrelated resources instead of isolating which count instance a scale-down destroys. aws_security_group was chosen because this estate already exercises it three times in its own vendored modules (modules/cluster, modules/service, modules/express-service each declare aws_security_group.this) and because its destroy is witnessed TWICE on the pinned emulator - a new GroupId AND verified absence in between - established by reading the EC2 API directly with no tofu in the loop before any assertion here was written. Placed last, after every other stage reported, at an address nothing else in this configuration names, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly fails.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.), leaving count_test[0]'s live GroupId (sg-52d986d91b3c973e6), its tofu-address (aws_security_group.count_test:0) and its tofu-slot (0) unchanged, and count_test[1] (sg-492493a03f1be337d) genuinely absent - all read through the AWS CLI, never choudoufu's own report; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.) under a NEW server-minted GroupId (sg-e97c66f6e7938bb49, was sg-492493a03f1be337d) carrying tofu-address=aws_security_group.count_test:1 and the same tofu-slot=1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (H-ORACLE), the identical 2-instance block stood up for real with plain terraform in its own account and its own working directory - stock never had this block, so unlike day2_rename/day2_remove/day2_replace there is no cold_deploy state to reuse - shows the identical shape: Plan: 0 to add, 0 to change, 1 to destroy. destroying count_test[1]=sg-2f52b9dfff25eabef only, then Plan: 1 to add, 0 to change, 0 to destroy. bringing it back under a new GroupId (sg-5feabe05e9ba79cb0), with count_test[0]=sg-c38e15e20448864cb unchanged both times. SYNTHETIC block, and why: this estate's corpus example (terraform-aws-modules/terraform-aws-ecs v7.6.0, examples/fargate) declares no scalable count knob at all - every count in its vendored modules is a 'count = local.create ? 1 : 0' boolean create toggle, which cannot hold two instances - and the only construct holding three, module.vpc's private_subnets, is driven by local.azs, which the ECS service's network configuration, the ALB and the NAT gateway all read, so scaling it churns a dozen unrelated resources instead of isolating which count instance a scale-down destroys. aws_security_group was chosen because this estate already exercises it three times in its own vendored modules (modules/cluster, modules/service, modules/express-service each declare aws_security_group.this) and because its destroy is witnessed TWICE on the pinned emulator - a new GroupId AND verified absence in between - established by reading the EC2 API directly with no tofu in the loop before any assertion here was written. Placed last, after every other stage reported, at an address nothing else in this configuration names, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly fails.", "day2_remove": "choudoufu: deleting module.ecs_task_definition's block proposed exactly 8 destroys (0 add, 0 change, 8 destroy), address-for-address identical to stock's oracle on cold_deploy's own state; applied cleanly (0 added, 0 changed, 8 destroyed); the standalone task definition family (ex-fargate-standalone) genuinely has 0 active revisions afterward, read via the AWS CLI, not choudoufu's own report; classifyOrphans did not withhold any destroy because no other module.ecs_task_definition block is declared anywhere in this config; the next plan is empty", "day2_rename": "moved block: module.alb renamed with zero churn (0 add, 9 change, 0 destroy), marker rewritten in place; live-mv: aws_service_discovery_http_namespace.this renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", "day2_replace": "choudoufu: changing module.ecs_task_definition's ForceNew name argument (module CALL, passed through to the local module's own family = coalesce(var.family, var.name)) proposed a forced replace at the same declared address (Plan: 8 to add, 0 to change, 8 to destroy.), applied cleanly; the old task definition is confirmed INACTIVE via the AWS CLI (ECS deregisters rather than deletes) and the new one (arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1) carries the marker, moved via the tofu-address tag (arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1 -> arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the task definition at the same address (Plan: 8 to add, 0 to change, 8 to destroy., plan only, not applied - it shares floci's account with $ADOPTED_EST); BREAK=replace confirms a manufactured marker collision - a second, genuinely live task definition wearing this address's marker, which no tombstone names as destroyed - is still reported loudly (\"Indistinguishable instances without per-instance markers\", naming both ARNs) rather than pruned or silently proposed as nothing, which is #849's own rule holding on this route too. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; the local record store is read by value on both sides of the replace (#879): before it, the record names family=ex-fargate-standalone with identity.secondary_id=arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1; after it, identity.secondary_id=arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1 with exactly one tombstone naming arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1, which is what lets the F2 plan tell the deregistered object's lingering tag from a second live claimant instead of refusing \"Indistinguishable instances without per-instance markers\" forever; two earlier target choices each found a genuine, separate defect: aws_service_discovery_http_namespace.this_renamed's (mv.go's propagateModuleRename skipping MoveRecord for a same-module rename) is FIXED on the gauntlet/mv-rekey branch, GitHub issue #412 - see F-ORACLE's own header comment and corpus-autoscaling-complete's/corpus-eks-basic's matching mv.go finding in this same unit, neither of which was re-run for #412; module.alb_renamed's (the non-converging cascade, F-ORACLE's own header comment, finding 2) remains a separate, open finding, not fixed here.", "drift_reconverge": "one object tampered (VPC's Name tag), plan proposed fixing exactly module.vpc.aws_vpc.this[0], apply changed 1 and the Name tag reconverged", "greenfield": "62 resources from nothing, cluster marker verified via the AWS CLI, 62 of 62 records in the local record store (#364 A2; both aws_ecs_task_definition instances are now included, #671 having closed the numeric-wire-identity-component gap in internal/live/identity/located.go's LocatedIdentityPlanFor that used to exclude them), replan empty, stock oracle in its own namespace matches structurally on cluster/service/standalone-task-definition/CloudMap-namespace/ALB/VPC", "migrate": "46 of 62 stamped", + "plan_approval": "one argument edited (module \"ecs_service\"'s service_tags ServiceTag, \"Tag on service level\" -> \"Tag on service level, reviewed\" - the service module merges service_tags into aws_ecs_service's own tags and nowhere else, so it reaches exactly one instance where every tags = local.tags in this example fans out over a whole module), \"plan -out=approved.tfplan\" wrote a 147864-byte stock-format plan file whose whole change set is one update on module.ecs_service.aws_ecs_service.this[0]; the world then moved out of band (VPC vpc-0ee657b8's Name tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.vpc.aws_vpc.this[0] and the live vpc-0ee657b8 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - arn:aws:ecs:eu-west-1:000000000000:service/ex-fargate/ex-fargate still read ServiceTag=\"Tag on service level\" through ecs list-tags-for-resource, not from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the VPC's Name tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the service read back with the reviewed ServiceTag, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); 46 tofu-estate-tagged objects before, 46 after, no state file either time", "test_plan": "genuinely empty replan (\"No changes. Your infrastructure matches the configuration.\") - #371, #378, #372, #110, #395 and #376 all fixed and stay fixed; the standalone task definition's essential/mountPoints[].readOnly wall is gone (lex00/floci#131, published and repinned this unit) and essential defaulting to true was never an independent wall on its own (this unit's own re-measurement). #395/#376: choudoufu keeps no persisted state, so every plan re-derives PriorState through ImportResourceState's bare stub; internal/live/projection/build.go's configuredAttrsSeed generalizes the tags-only import-stub seed (issue #287 item 8) to every Required-or-Optional-non-Computed attribute (fixing #376's track_latest/skip_destroy directly), and internal/live/projection/residue.go's residueConfigSourced widening of classifyResidue plus the new builder.residueSeedFor pre-read seed close #395's managed-reference case (task_definition = aws_ecs_task_definition.this[0].arn) that configuredAttrsSeed's static evaluator alone could not reach. Identities confirmed by value against the AWS CLI: $CLUSTER_ARN, $TD_SVC_ARN, $TD_STANDALONE_ARN, and #368's scalable target $GOT_TARGET_RID." }, - "duration_s": 648, + "duration_s": 550.4, "stage_seconds": { - "cold_deploy": 105, - "day2_count": 122, - "day2_remove": 20, - "day2_rename": 59, - "day2_replace": 25, - "drift_reconverge": 12, - "greenfield": 184, - "migrate": 86, + "cold_deploy": 82, + "day2_count": 64, + "day2_remove": 15, + "day2_rename": 52, + "day2_replace": 16, + "drift_reconverge": 11, + "greenfield": 172, + "migrate": 78, + "plan_approval": 27, "test_apply": 5, - "test_plan": 29 + "test_plan": 28 } }, "notes": "RE-CROSSED FOR REAL 2026-08-21 at dbc5ebf575 (GitHub issue #346's fix), twice, the second time instrumented to capture live-plan's raw output. STAGES UNCHANGED at 2 of 5, and #346's fix does not reach this estate either. Stage 1: 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', cluster confirmed carrying no tofu-address beforehand. Stage 2: '46 of 62 eligible (28 VERIFIED + 18 DRIFTED); 16 skipped'; '46 stamped, 1 recorded (time_sleep.this[0]), 0 failed, 15 skipped'; both markers confirmed by direct aws ecs describe-* calls against floci rather than through choudoufu's own report (cluster/ex-fargate = module.ecs_cluster.aws_ecs_cluster.this:0, service/ex-fargate/ex-fargate = module.ecs_service.aws_ecs_service.this:0). Stage 3 fails on exactly 8 diagnostics, the same 8 as before, verified two ways (grep -c '^Error:' over the raw output, and the script's own hard count assertions): 1 'Module output not supported in static context' on main.tf:68 (cluster_arn = module.ecs_cluster.arn), 1 'Unable to compute static value' on modules/service/main.tf:1565 (aws_appautoscaling_target.this[0].resource_id), and 6 'Unresolvable identity' cascading from it onto aws_appautoscaling_policy.this['cpu'] and ['memory']. Why the fix misses it: the same value-shaped module-CALL-argument route as corpus-rds-complete-postgres, PLUS a transform the part-shaped route could not express even if it reached - local.cluster_name = element(split('/', var.cluster_arn), 1), a function applied to the deferred value. identity.Formula holds literals and ParentRefs and has no way to say 'split this parent attribute and take element 1'. That is a mechanism this repository does not have, not a gap in #346's fix. A 'Provider version does not match the admission evidence version' warning (6.61.0 resolved against 6.59.0 admission evidence) also prints; the script's own header calls it a caution, not a failure. PRIOR HISTORY BELOW. Landed dd83121592 (2026-08-18), 62 resources: ECS cluster (Container Insights, FARGATE/FARGATE_SPOT split), a BLUE_GREEN service behind an ALB, ECS Exec, ECS Service Connect, a two-container task definition plus a standalone second one, a CloudMap namespace, nested ALB/VPC modules. cold_deploy and migrate genuinely pass (43 of 62 stamped: 26 VERIFIED + 17 DRIFTED; 19 skipped - 16 untaggable by design, 3 blocked by #305). test_plan blocked by 4 sites: #305's familiar default_* trio (3 sites) and a NEW one, filed as #308: the child-module for_each keyset prover (internal/live/identity/foreach_keyset.go) has no case for a for-comprehension (for k, v in var.container_definitions : k => v if ...) and doesn't chase a bare var.X for_each source across a module-call boundary to the literal object constructor at the caller, whose keys are actually static even though one unrelated attribute value inside the map is dynamic - resolve.go's resource-level forEachOverComprehension already does the equivalent per-entry evaluation the module-call prover lacks. Two real floci gaps found and filed but not fixed, both sized as real modeling work rather than quick patches: lex00/floci#59 (CreateCluster silently drops settings/Container Insights, default_capacity_provider_strategy never serialized back) and lex00/floci#60 (CreateService/DescribeServices drop scheduling_strategy, enable_ecs_managed_tags, enable_execute_command, health_check_grace_period_seconds, deployment_controller, blue-green load_balancer.advanced_configuration, service_connect_configuration entirely - scheduling_strategy's omission in particular forces the AWS provider to propose destroy-and-recreate on every plan after creation, a real non-idempotency bug independent of choudoufu). Neither floci gap blocks this crossing's own outcome since test_plan already refuses earlier, upstream of any ECS-field diff. Follow-up pass 2026-08-18 (#313 cross-check, 0a94070b16/3ff22c5be6): the committed run.sh still asserted pre-#305 counts (43/62 eligible) and failed before ever reaching stage 3; updated to the real current numbers (46/62 eligible, #305's default_* trio now fully resolved here) and re-verified against real floci twice plus a BREAK=1 negative control, all read from the script's own printed lines. test_plan's sole remaining blocker is confirmed #308 alone - #313's diagnostic does not appear anywhere in the output. Mechanism: this estate's vpc submodule expands subnets via count over a statically-known length, never a for_each keyed on the AZ name values, so #313 structurally can't reach it. Commented on #313 and #308. #308 fixed and merged 2026-08-18 (a9ac6d06e7/b2bb59585d, generic: a *hclsyntax.ForExpr case plus a cross-module-call var/local chase in internal/live/identity/foreach_keyset.go, reaching every module-call for_each proof, not just this estate) - re-run confirms 0 occurrences of #308's diagnostic (was 1), but test_plan is still blocked: #308 firing first had been masking two more causes in the same live-plan output all along. Follow-up pass 2026-08-18 (e74c7d5869/07c7317ab6) re-staled run.sh's stage-3 assertions and header to the real current picture, verified across three separate real live-plan runs (stable counts each time) plus a BREAK=1 negative control: 236 total diagnostics, three distinct root causes, not one. Root cause A (#313's canonical shape, 48 sites): data.aws_availability_zones.available feeding local.azs into module \"vpc\"'s azs argument. Root cause B (also #313's family via a module output, 1 site): module.ecs_cluster.arn passed into module \"ecs_service\" as cluster_arn, \"Module output not supported in static context\". Both A and B are #313's already-ruled maintainer-level architecture question (live-plan never calls a provider during plan), not fixed here. Root cause C (NOT #313 - a distinct, newly-found, likely-fixable gap #308's own fix exposed, 4 sites): each.value.enable_cloudwatch_logging/create_cloudwatch_log_group inside module.container_definition, both literal booleans in the caller's own object literal but refused because each.value is treated as one opaque blob instead of projected to the referenced field - filed as #315, not attempted. These cascade to 177 \"Unable to compute static value\" and 6 \"Unresolvable identity\" sites (traced: aws_appautoscaling_policy reads aws_appautoscaling_target's resource_id, itself one of C's failures). Re-verified 2026-08-19 (be6b1096ba/b7bbff04a6) against everything landed overnight (#313's data-source half, #315, #321/#324, #323, #325): root cause A (0 sites, confirmed fixed) and root cause C (0 sites, #315's each.value projection fix confirmed) are both gone. Only root cause B remains (module.ecs_cluster.arn, a Computed attribute crossing a module-output boundary), now cascading to 1 \"Unable to compute static value\" plus 6 \"Unresolvable identity\" sites (aws_appautoscaling_target/aws_appautoscaling_policy chain) - down from 236 diagnostics to 8. This is the same already-acknowledged Computed-attribute/module-output architecture question corpus-rds-complete-postgres and corpus-security-group-complete are also blocked on; no action taken here, no new issue filed." @@ -643,16 +653,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -661,27 +671,29 @@ "exit_code": 0, "detail": { "cold_deploy": "54 resources, genuinely cold, genuinely unmarked", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and count_test[0] kept BOTH its live id (sg-2190fcd004e71e1a2) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) across it, read back through the AWS CLI rather than choudoufu's own report; the destroyed group (sg-3ff9bb7024a1a3cac) is genuinely gone from the account; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a NEW object under a NEW GroupId (sg-06d58eb0790723d87, was sg-3ff9bb7024a1a3cac) carrying aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (G0): plain terraform, its own working directory, state and VPC, standing the IDENTICAL 2-instance count block up for real against the same idle floci account, shows the identical shape - destroy the higher index only (count_test[1]=sg-0fa8f370fcd242978), create it back under a new id (sg-0260527e0d0cd1063), count_test[0]=sg-5186dbede2e418faf unchanged both times - and is torn down again before choudoufu's own side runs. Synthetic block, and why: terraform-aws-eks v9.0.0 has no scalable count knob to drive - every count in the module is a boolean create toggle (var.create_eks ? 1 : 0 and siblings), and the one length-driven knob (local.worker_group_count) drives only aws_autoscaling_group, aws_launch_configuration and aws_iam_role_policy_attachment, all untaggable (no tags argument in the provider schema, so no marker surface for this stage's identity assertion), and dropping a worker group also rewrites kubernetes_config_map.aws_auth, which aggregates every worker role, so the plan would carry changes alongside the destroy. aws_security_group.count_test is the sanctioned fallback per live/GAUNTLET.md #8, of a type this estate already exercises three times, at an address nothing else in the config names. BREAK_COUNT=1 confirms the check is load-bearing: asserting the WRONG instance (count_test[0], the survivor) was destroyed makes this stage report fail.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and count_test[0] kept BOTH its live id (sg-536bd82e1ae8b011a) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) across it, read back through the AWS CLI rather than choudoufu's own report; the destroyed group (sg-405bd41502e46a151) is genuinely gone from the account; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a NEW object under a NEW GroupId (sg-6af012ebcadc1cf6a, was sg-405bd41502e46a151) carrying aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (G0): plain terraform, its own working directory, state and VPC, standing the IDENTICAL 2-instance count block up for real against the same idle floci account, shows the identical shape - destroy the higher index only (count_test[1]=sg-acdf0826e35bb8527), create it back under a new id (sg-c7f1d8d68584ba8f6), count_test[0]=sg-cf71071802083cb77 unchanged both times - and is torn down again before choudoufu's own side runs. Synthetic block, and why: terraform-aws-eks v9.0.0 has no scalable count knob to drive - every count in the module is a boolean create toggle (var.create_eks ? 1 : 0 and siblings), and the one length-driven knob (local.worker_group_count) drives only aws_autoscaling_group, aws_launch_configuration and aws_iam_role_policy_attachment, all untaggable (no tags argument in the provider schema, so no marker surface for this stage's identity assertion), and dropping a worker group also rewrites kubernetes_config_map.aws_auth, which aggregates every worker role, so the plan would carry changes alongside the destroy. aws_security_group.count_test is the sanctioned fallback per live/GAUNTLET.md #8, of a type this estate already exercises three times, at an address nothing else in the config names. BREAK_COUNT=1 confirms the check is load-bearing: asserting the WRONG instance (count_test[0], the survivor) was destroyed makes this stage report fail.", "day2_remove": "choudoufu: deleting aws_security_group.worker_group_mgmt_one's block (plus emptying the one argument that referenced it) proposed 2 resource action(s), address-for-address and action-for-action identical to stock's oracle on cold_deploy's own state; applied cleanly; the security group is genuinely gone from the live account, read via the AWS CLI, not choudoufu's own report; classifyOrphans did not withhold any destroy because no other aws_security_group.worker_group_mgmt_one block is declared anywhere in this config; the next plan is empty", "day2_rename": "moved block: aws_security_group.worker_group_mgmt_two renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_security_group.all_worker_mgmt renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing aws_security_group.worker_group_mgmt_two_renamed's ForceNew name_prefix argument proposed a forced replace at the same declared address (Plan: 3 to add, 1 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-17fb7372c7f753b56) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-4655e85b6bf6f3f89 -> sg-17fb7372c7f753b56); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 1 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted all_worker_mgmt_renamed (Part D's own live-mv leg) and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on the gauntlet/mv-rekey branch, GitHub issue #412 (propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard); see this section's own header comment for the fix and corpus-autoscaling-complete's/corpus-ecs-fargate's matching ones in this same unit. This script was not re-run for #412, so this detail string still describes the pre-#412 run until this estate's next real run.", + "day2_replace": "choudoufu: changing aws_security_group.worker_group_mgmt_two_renamed's ForceNew name_prefix argument proposed a forced replace at the same declared address (Plan: 3 to add, 1 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-1b6782f1886852eac) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-ca6711c0e80c37f96 -> sg-1b6782f1886852eac); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 1 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted all_worker_mgmt_renamed (Part D's own live-mv leg) and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on the gauntlet/mv-rekey branch, GitHub issue #412 (propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard); see this section's own header comment for the fix and corpus-autoscaling-complete's/corpus-ecs-fargate's matching ones in this same unit. This script was not re-run for #412, so this detail string still describes the pre-#412 run until this estate's next real run.", "drift_reconverge": "one object tampered (VPC's Name tag), plan proposed fixing exactly module.vpc.aws_vpc.this[0], apply changed 1 and the Name tag reconverged", "greenfield": "54 resources from nothing, cluster marker verified via the AWS CLI, 54 records under the implied local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on cluster status/version, ASG count/desired-capacities, and cluster-owned security-group count", "migrate": "25 of 54 resource instances stamped, 25 of 25 confirmed via the AWS CLI; 5 record-backed instances seeded into the implied local record store (#364)", + "plan_approval": "one argument edited (aws_security_group.worker_group_mgmt_one gains tags = { Reviewed = \"yes\" }), \"plan -out=approved.tfplan\" wrote a 82147-byte stock-format plan file whose whole change set is one update on aws_security_group.worker_group_mgmt_one; the world then moved out of band (the VPC vpc-328dc0c5's Name tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.vpc.aws_vpc.this[0] and the live vpc-328dc0c5 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - sg-7d8a2a280dbac8a69 still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-7d8a2a280dbac8a69 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The reviewed change was in-place from end to end: sg-7d8a2a280dbac8a69 kept its live id across the whole part, so PART D/E/F/G below still start from the objects STAGE 2 stamped. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); 25 tofu-estate-tagged objects before, 25 after", "test_plan": "live-plan runs to completion with ZERO Error diagnostics and reports \"No changes. Your infrastructure matches the configuration.\" - the record-backed worker launch configuration's enable_monitoring/root_block_device/user_data all now agree with the config's own desired value (lex00/floci#132 for the first two, configuredAttrsSeed's residue-record pre-read seed in internal/live/projection/build.go for the third)" }, - "duration_s": 844.9, + "duration_s": 920.1, "stage_seconds": { - "cold_deploy": 95, - "day2_count": 200, - "day2_remove": 61, - "day2_rename": 77, - "day2_replace": 60, - "drift_reconverge": 39, - "greenfield": 154, - "migrate": 119, - "test_apply": 21, + "cold_deploy": 83, + "day2_count": 214, + "day2_remove": 62, + "day2_rename": 73, + "day2_replace": 58, + "drift_reconverge": 41, + "greenfield": 155, + "migrate": 93, + "plan_approval": 100, + "test_apply": 22, "test_plan": 17 } }, @@ -707,16 +719,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "cb5ae2009fb85fdafd76126bb1842144855eb9d7", - "date": "2026-09-06T04:29:56Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -732,20 +744,22 @@ "drift_reconverge": "VPC Name tag tampered out of band, exactly 1 object proposed and reconverged, marker survived the incremental tag update", "greenfield": "10 resources from nothing (1 vpc, 3 subnets, 1 igw, 1 route table, 3 untaggable associations, 1 dynamodb table), VPC marker verified via the AWS CLI, 10 records in the local record store (#364 A2, one per managed instance), replan empty, stock oracle in its own namespace matches structurally on vpc/subnets/igw/route-table/dynamodb-table", "migrate": "7 of 10 verified and stamped, 0 failed, 3 correctly UNTAGGABLE; markers read back via the AWS CLI", + "plan_approval": "one argument edited (module.networking's internet gateway gains a Reviewed=yes tag; an in-place update, never a replace, because PART G below re-reads this estate's subnet ids by value), \"plan -out=approved.tfplan\" wrote a 12570-byte stock-format plan file whose whole change set is one update on module.networking.aws_internet_gateway.main; the world then moved out of band (vpc-b4200bde's Name tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.networking.aws_vpc.main and the live vpc-b4200bde it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - igw-676f386d still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and igw-676f386d read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The revert is proven byte-for-byte against the corpus pin (copy_modules' own diff, re-run) and the three subnets PART G re-reads are still the three cold_deploy minted. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 7, no state file", "test_plan": "no changes; VPC and table markers unchanged, all three untaggable associations resolved by their composite identity" }, - "duration_s": 146.6, + "duration_s": 142.4, "stage_seconds": { - "cold_deploy": 29, - "day2_count": 14, - "day2_remove": 7, + "cold_deploy": 12, + "day2_count": 13, + "day2_remove": 6, "day2_rename": 10, "day2_replace": 13, - "drift_reconverge": 4, - "greenfield": 27, - "migrate": 36, - "test_apply": 3, + "drift_reconverge": 6, + "greenfield": 26, + "migrate": 39, + "plan_approval": 12, + "test_apply": 2, "test_plan": 3 } }, @@ -771,16 +785,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "d72960cdc31f4e7d90f61088920379fc29b3999c", - "date": "2026-09-06T05:35:09Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -789,28 +803,30 @@ "exit_code": 0, "detail": { "cold_deploy": "6 resource instances added, 0 already tofu-estate-marked before migration", - "day2_count": "choudoufu: scaling aws_iam_role.count_test from 2 to 1 proposed exactly \"count_test[1] will be destroyed\" (0 add, 0 change, 1 destroy) and applied it, leaving count_test[0]'s server-minted RoleId (AROAPDYSMRCJISNRTSLK), its CreateDate and its tofu-address=aws_iam_role.count_test:0 marker all unchanged, and tombstoning count_test[1]'s local record (has tombstone, no identity - the #398-guard shape); scaling back from 1 to 2 proposed exactly \"count_test[1] will be created\" (1 add, 0 change, 0 destroy) and brought it back under the SAME deterministic name (giantswarm-crossplane-count-test-1) with a NEW RoleId (AROAKLXLOOPVVBJ8L4P2 -> AROAX6KFELD5P9JCWZF5) and a new CreateDate, re-marked aws_iam_role.count_test:1 and re-identified in the record store, while count_test[0] stayed untouched throughout; the next plan is empty. Every identity here is read back through the AWS CLI and the local record store, never through choudoufu's own report, and the destroy witness is the RoleId rather than the name or the ARN because both of those are deterministic from configuration and come back identical - confirmed against floci directly, no tofu in the loop, before the assertions were written. Stock oracle (G-ORACLE): real tofu standing the IDENTICAL count block up in the idle greenfield-oracle account showed the identical shape - destroy the higher index only, create the higher index back under the same name with a new RoleId (AROA4HHSNGXJQD7KLPND -> AROA9QH5CL9MJ5020J83), the lower index's RoleId and CreateDate unchanged both times - and the two sides' normalised action sets are compared literally, not just described. Synthetic block, per live/GAUNTLET.md #8's sanctioned fallback: the pinned crossplane module declares no count at all and its only two for_each knobs (aws_iam_role_policy.additional_policies over var.additional_policies, aws_iam_role_policy_attachment.additional_policy_attachments over toset(var.additional_policies_arns)) are both UNTAGGABLE types that carry no marker to keep an identity in, the second provably resolving to zero instances, and the first's inline-policy set is additionally policed by aws_iam_role_policies_exclusive in the same module; aws_iam_role.count_test reuses a type this estate already exercises and lives in its own day2_count.tofu beside the estate's root wiring ($ESTATE), so the vendored module stays byte-identical. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, proving the which-instance assertion is load-bearing.", + "day2_count": "choudoufu: scaling aws_iam_role.count_test from 2 to 1 proposed exactly \"count_test[1] will be destroyed\" (0 add, 0 change, 1 destroy) and applied it, leaving count_test[0]'s server-minted RoleId (AROAWKJGZCMEJKQD17GA), its CreateDate and its tofu-address=aws_iam_role.count_test:0 marker all unchanged, and tombstoning count_test[1]'s local record (has tombstone, no identity - the #398-guard shape); scaling back from 1 to 2 proposed exactly \"count_test[1] will be created\" (1 add, 0 change, 0 destroy) and brought it back under the SAME deterministic name (giantswarm-crossplane-count-test-1) with a NEW RoleId (AROAN4V2DL70N90344J2 -> AROAIQ2RZND9EW25FEJU) and a new CreateDate, re-marked aws_iam_role.count_test:1 and re-identified in the record store, while count_test[0] stayed untouched throughout; the next plan is empty. Every identity here is read back through the AWS CLI and the local record store, never through choudoufu's own report, and the destroy witness is the RoleId rather than the name or the ARN because both of those are deterministic from configuration and come back identical - confirmed against floci directly, no tofu in the loop, before the assertions were written. Stock oracle (G-ORACLE): real tofu standing the IDENTICAL count block up in the idle greenfield-oracle account showed the identical shape - destroy the higher index only, create the higher index back under the same name with a new RoleId (AROAS5MQ3SZOZQY5MORH -> AROAQELS5MR4KB0L9WAG), the lower index's RoleId and CreateDate unchanged both times - and the two sides' normalised action sets are compared literally, not just described. Synthetic block, per live/GAUNTLET.md #8's sanctioned fallback: the pinned crossplane module declares no count at all and its only two for_each knobs (aws_iam_role_policy.additional_policies over var.additional_policies, aws_iam_role_policy_attachment.additional_policy_attachments over toset(var.additional_policies_arns)) are both UNTAGGABLE types that carry no marker to keep an identity in, the second provably resolving to zero instances, and the first's inline-policy set is additionally policed by aws_iam_role_policies_exclusive in the same module; aws_iam_role.count_test reuses a type this estate already exercises and lives in its own day2_count.tofu beside the estate's root wiring ($ESTATE), so the vendored module stays byte-identical. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, proving the which-instance assertion is load-bearing.", "day2_remove": "choudoufu: deleting module.crossplane_final's block proposed 6 resource action(s), address-for-address and action-for-action identical to stock's oracle on cold_deploy's own state; applied cleanly; the role is genuinely gone from the live account (get-role now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report); classifyOrphans did not withhold any destroy because no other module.crossplane* block is declared anywhere in this config; the next plan is empty", "day2_rename": "moved block: module.crossplane renamed to .crossplane_renamed with zero churn (0 add, 2 change, 0 destroy - role and policy), markers rewritten in place; live-mv: .crossplane_renamed renamed to .crossplane_final with zero churn, both markers rewritten in place (one live-mv call per taggable object); stock oracle over the same chained module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", "day2_replace": "choudoufu: changing module.crossplane_final's ForceNew installation_name argument proposed a 6 add / 0 change / 6 destroy cascade with the role and the managed policy each explicitly named 'must be replaced' at their same declared addresses, applied cleanly; the old role (giantswarm-gsprereqs-crossplane) is confirmed gone and the new role (giantswarm-gsprereqs-v2-crossplane) carries the marker, both via the AWS CLI; the local record store's record at the role's address now names the new role, not the destroyed one (giantswarm-gsprereqs-crossplane -> giantswarm-gsprereqs-v2-crossplane); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes an equal add/destroy cascade (>=2) with role and policy both replaced at the same addresses (plan only, not applied - it shares floci's account with $ESTATE); BREAK=replace confirms a manufactured marker collision is reported loudly (a named 'Live resource displaced from the address it is marked for' warning, the scalar-resource shape) rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one.", "drift_reconverge": "role's installation tag tampered, exactly the IAM role proposed and reconciled, apply changed 1, tag reads back as configured", "greenfield": "6 resources from nothing (role, managed policy, 4 untaggable), role marker verified via the AWS CLI, 6 records in the local record store (#364 A2, one per managed instance), replan empty, stock oracle in its own namespace matches structurally on the role and the managed policy", "migrate": "2 of 6 stamped (role, managed policy), 4 untaggable skipped, module's own tags survived the stamp", + "plan_approval": "one argument edited (additional_policies[\"extra-tagging\"] widened to allow ec2:DeleteTags as well), \"plan -out=approved.tfplan\" wrote a 10347-byte stock-format plan file whose whole change set is one update on module.crossplane.aws_iam_role_policy.additional_inline_policies[\"extra-tagging\"]; the world then moved out of band (giantswarm-gsprereqs-crossplane's installation tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.crossplane.aws_iam_role.giantswarm_crossplane_role and the live giantswarm-gsprereqs-crossplane it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - the inline policy read back through the AWS CLI still allowed only ec2:CreateTags, which is stronger evidence than the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the inline policy read back allowing ec2:DeleteTags, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, both exclusive sets unchanged", "test_plan": "live-plan empty, role/policy tofu-address unchanged, both *_exclusive resources re-derived by value" }, - "duration_s": 162.7, + "duration_s": 113.3, "stage_seconds": { - "cold_deploy": 39, - "day2_count": 39, - "day2_remove": 7, - "day2_rename": 11, - "day2_replace": 10, - "drift_reconverge": 5, - "greenfield": 20, - "migrate": 24, - "test_apply": 4, - "test_plan": 4 + "cold_deploy": 5, + "day2_count": 31, + "day2_remove": 5, + "day2_rename": 10, + "day2_replace": 7, + "drift_reconverge": 4, + "greenfield": 12, + "migrate": 23, + "plan_approval": 11, + "test_apply": 3, + "test_plan": 2 } }, "notes": "Landed 2026-08-19 (crossing 0bd3ac80b7, merge 9fa0141294), the seventh OpenTofu-native estate and the first from a commercial vendor's production repository rather than a module registry, personal monorepo, or single-maintainer accelerator - Giant Swarm GmbH's own customer-facing account-prep for their managed Kubernetes offering. OpenTofu-native evidence, three independent kinds: README's opening sentence and directory index both say OpenTofu with no compatibility hedge; the CI workflow is named 'OpenTofu checks', installs via opentofu/setup-opentofu, and never mentions terraform; the crossed crossplane/ directory is genuinely .tofu-suffixed throughout (providers.tofu, role.tofu, variables.tofu), the file-level standard only the hongbomiao slices had met before (overture-tiles and xancloud-iac are both plain .tf). Scoped to crossplane/ specifically: self-contained (no remote state, no live EKS/OIDC dependency, its only data source is aws_partition which makes no API call), the other five directories in the repo excluded with stated reasons (three plain-.tf same-shape, one an account-singleton quota table, one a wrapper that only calls the others). Real run, rc=0, 548s. cold_deploy: PASS, plain tofu apply, 6 resources added, 0 pre-existing tofu-estate tags, the toset()-keyed for_each on additional_policy_attachments confirmed resolving to zero instances. migrate: PASS - 2 of 6 eligible (UNTAGGABLE 2, UNADMITTED_TYPE 2, DRIFTED 2), -approve 2 newly stamped 0 failed, both markers re-verified directly through the AWS CLI including that the module's own installation tag survived the stamp. test_plan: BLOCKED for real at exactly 2 sites, the plan's entire diagnostic surface - both Rule: unadmitted-type, on aws_iam_role_policy_attachments_exclusive and aws_iam_role_policies_exclusive, no other rule firing anywhere in the estate. A control stage (3b, not counted toward stage 4/5) cut exactly those two resource blocks and drove the rest of the pipeline for real: control test plan EMPTY, control test apply a genuine no-op (2 objects before and after), control drift-and-reconverge fixed exactly one mutated object - proving the estate's only real block is those two types, not routing around anything. Both negative controls (BREAK=1 at the stage-2 identity assertion, BREAK_STAGE3=1 expecting 3 refusal sites where the real count is 2) verified failing in real full runs. Filed as INTENTIUS/choudoufu#334: both unadmitted types have the identical import-grammar shape (single-argument, no-separator) to aws_vpc_security_group_rules_exclusive, which #307 already admitted via row-gen's tryGrammarComposite at 64cac28120, and carry the same no-CFN-counterpart mapping-gen overlay as that admitted twin - a worked ADMIT-class precedent, not attempted here; the one recorded difference (force_new on the admitted twin's security_group_id, absent on either new type's role_name) is flagged as not obviously the gate since tryGrammarComposite's single-argument branch reads no force_new field, and why row-gen's own proposal for these two is currently absent from ratified.json is the fix's first open question. No Go code touched. justfile gained demo-corpus-giantswarm-crossplane; live/corpus-manifest.json gained the pin; HANDOFF.md section 3 updated. Follow-up pass 2026-08-19/20 (#334 fixed and merged, 37957d873c/a6627543c4/20cc1774d6): the issue's own open question resolved the OPPOSITE way from what it suspected - row-gen was never declining to propose these two rows; it proposes both, byte-identical in shape to the admitted aws_vpc_security_group_rules_exclusive twin, and nobody had ever ratified the proposal. Fix is a ratified.json entry, no code change - reach is exactly these two types, stated plainly rather than dressed up as a generalization. The real finding: 316 types row-gen proposes sit unratified, 166 under this exact rule (including every other *_exclusive family member) - a ratification backlog, not a generator defect, and clearing it is a maintainer-scale call since every ratified row is a claim that touches live infrastructure. FIVE OF FIVE, run for real on the rebased tree, rc=0, 250s: stage 3 now an empty plan (identities asserted by value - the two exclusive resources carry no marker, being untaggable, so the assertion is the live content each enforces, e.g. attached policy ARN and inline-policy name); stage 4 a genuine no-op (2 tagged objects before and after, both exclusive sets independently re-read afterward so a wrongly-reconciled enforcer would be caught); stage 5 one out-of-band mutation, exactly one object proposed and fixed. Three real negative controls (BREAK=1, BREAK_STAGE3=1, BREAK_STAGE5=1) each confirmed failing at the right point. Script rewritten: stage 3 now asserts a pass instead of hard-failing by design, the 3b control retired with the block it existed to control for. Separately found, not yet fixed: every crossing script that runs more than one `terraform init` pays a real ~320s tax per extra init, because the shared plugin cache records no checksums and a directory with no `.terraform.lock.hcl` re-downloads the whole provider to compute them - seeding the lock file from stage 1's own init cuts this to ~1s and this estate's full run from several failed 10-minute-cap attempts to 250s total. Worth a sweep across every multi-init script here." @@ -835,16 +851,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -860,21 +876,23 @@ "drift_reconverge": "the plan proposed fixing 1 object(s) after the out-of-band tag mutation: module.s3_bucket_hm_harbor.aws_s3_bucket.main", "greenfield": "3 resources from nothing (bucket, user, untaggable inline policy), markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared", "migrate": "2 of 3 stamped (bucket, user), 1 UNTAGGABLE (inline policy); bucket hongbomiao-harbor-crossing-hm-harbor -> tofu-address=module.s3_bucket_hm_harbor.aws_s3_bucket.main, user hongbomiao-harbor-crossing-hm-harbor-user -> tofu-address=module.harbor_iam_user.aws_iam_user.hm_harbor_iam_user", + "plan_approval": "one argument edited (harbor_iam_user's common_tags gain Reviewed=yes), \"plan -out=approved.tfplan\" wrote a 8658-byte stock-format plan file whose whole change set is one update on module.harbor_iam_user.aws_iam_user.hm_harbor_iam_user; the world then moved out of band (hongbomiao-harbor-crossing-hm-harbor's hm_team tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.s3_bucket_hm_harbor.aws_s3_bucket.main and the live hongbomiao-harbor-crossing-hm-harbor it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - hongbomiao-harbor-crossing-hm-harbor-user still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and hongbomiao-harbor-crossing-hm-harbor-user read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 2 objects before, 2 after, no state file either time", "test_plan": "empty plan; identity re-check: bucket and user tofu-address unchanged, inline policy's resource ARN still matches the configuration" }, - "duration_s": 208.6, + "duration_s": 198.2, "stage_seconds": { - "cold_deploy": 27, - "day2_count": 43, + "cold_deploy": 15, + "day2_count": 45, "day2_remove": 7, - "day2_rename": 9, - "day2_replace": 6, + "day2_rename": 8, + "day2_replace": 7, "drift_reconverge": 5, - "greenfield": 88, - "migrate": 18, - "test_apply": 3, - "test_plan": 2 + "greenfield": 78, + "migrate": 17, + "plan_approval": 11, + "test_apply": 2, + "test_plan": 3 } }, "notes": "Landed 2026-08-19, the sixth estate in the OpenTofu-native lane and the fourth to clear all five stages. Sourced per HANDOFF's own suggestion to scope a second (here, third) disjoint slice of the already-crossed hongbomiao monorepo before a fresh search. Surveyed every remaining AWS environment: network/main.tofu is pure data sources (nothing to migrate); kubernetes/main.tofu builds a full terraform-aws-modules/eks cluster and every IAM module in it but one (velero_iam_role, mimir_iam_role, loki_iam_role, tempo_iam_role, label_studio_iam_role, etc., 15 total) takes amazon_eks_cluster_oidc_provider(_arn) from that same cluster - the same scope/risk class as the terraform-popular lane's already-blocked terraform-aws-eks examples/basic crossing. The one exception, the \"Harbor\" section (S3 bucket + IAM user + inline user policy), needs no EKS cluster, no OIDC provider, no remote state at all - self-contained like storage's own scoped slice. Nebius/Cloudflare/Snowflake environments confirmed to still exist and be real, actively-maintained infrastructure, but target non-AWS clouds floci cannot emulate. Crosses aws_iam_user/aws_iam_user_policy, a genuinely different resource pair from Labelbox's aws_iam_role/aws_iam_role_policy - both already-ratified DefaultTable rows, no schema-fallback warning. All five stages verified for real against a live floci container: cold_deploy (tofu apply, 3 resources created, confirmed 0 pre-existing tofu-estate tags), migrate (live-import verified 2 of 3 eligible - bucket + user - 1 correctly UNTAGGABLE - the inline policy; markers re-read via AWS CLI matched: module.s3_bucket_hm_harbor.aws_s3_bucket.main, module.harbor_iam_user.aws_iam_user.hm_harbor_iam_user), test_plan (state deleted, live-plan empty, identities re-verified against the AWS CLI including the inline policy's resource ARN read directly off the live object), test_apply (genuine no-op, 2 tagged objects before and after), drift_reconverge (bucket tag tampered out of band, plan proposed fixing exactly that one object, reconverge apply restored it). BREAK=1 verified load-bearing, failing exactly at the stage-2 identity assertion. No floci or choudoufu gaps found - this crossing is clean. Merged to local main as ad2cf81cf3 (crossing itself: 0c4e16af6a); justfile gained recipe demo-corpus-hongbomiao-harbor (port 4728); no new live/corpus-manifest.json entry needed, reuses the existing hongbomiao pin." @@ -899,16 +917,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "d72960cdc31f4e7d90f61088920379fc29b3999c", - "date": "2026-09-06T05:35:09Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -924,21 +942,23 @@ "drift_reconverge": "bucket tag drifted; exactly module.amazon_s3_bucket_hm_labelbox.aws_s3_bucket.main proposed, applied (1 changed), reconverged to hongbomiao", "greenfield": "4 resources from nothing (bucket, CORS config, role, untaggable inline role policy), markers verified via the AWS CLI, 4 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared", "migrate": "2 of 4 stamped (2 skipped, untaggable), 0 failed; markers read back via the AWS CLI", + "plan_approval": "one argument edited (labelbox_iam_role's common_tags gain Reviewed=yes), \"plan -out=approved.tfplan\" wrote a 11141-byte stock-format plan file whose whole change set is one update on module.labelbox_iam_role.aws_iam_role.labelbox_iam_role; the world then moved out of band (hongbomiao-labelbox-crossing-hm-labelbox's hm_team tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.amazon_s3_bucket_hm_labelbox.aws_s3_bucket.main and the live hongbomiao-labelbox-crossing-hm-labelbox it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - LabelboxRole-hm-labelbox still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and LabelboxRole-hm-labelbox read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, no state file", "test_plan": "no resource change proposed; bucket and role tofu-address unchanged, CORS origins and inline policy resource match config" }, - "duration_s": 221, + "duration_s": 177.5, "stage_seconds": { - "cold_deploy": 55, - "day2_count": 23, + "cold_deploy": 16, + "day2_count": 24, "day2_remove": 9, "day2_rename": 10, - "day2_replace": 9, - "drift_reconverge": 7, - "greenfield": 55, - "migrate": 45, - "test_apply": 4, - "test_plan": 4 + "day2_replace": 8, + "drift_reconverge": 6, + "greenfield": 48, + "migrate": 37, + "plan_approval": 14, + "test_apply": 3, + "test_plan": 3 } }, "notes": "Landed 2026-08-18 as the second estate in the OpenTofu-native lane and the first to clear all five stages there. Stronger OpenTofu-native evidence than corpus-sumaform-aws (which only describes itself as OpenTofu-native but ships plain .tf once its .example template is copied in): every file under infrastructure/opentofu/ genuinely uses the .tofu extension, its own justfile drives init/plan/apply/refresh/destroy exclusively via `tofu`, and common_tags carries \"hm_managed_by\" = \"opentofu\" - proven rather than asserted, since the crossing script's own stock terraform init against this estate reports \"The directory has no Terraform configuration files.\" Scoped to the self-contained \"Labelbox\" slice (S3 bucket, its CORS configuration, an IAM role with an inline S3-read policy - three real leaf modules copied byte-identical from the pinned commit, diffed programmatically in the script) out of a much larger monorepo (AWS+Nebius+Cloudflare+Snowflake+EKS, cross-wired via terraform_remote_state) too large to stand up in one sitting - the same scoping convention corpus-sumaform-aws's module.base stand-in uses. All five stages verified for real: cold_deploy (tofu apply, 4 resources, confirmed unmarked via resourcegroupstaggingapi), migrate (live-import: \"2 of 4 resource instance(s) are eligible for stamping\" - 2 correctly UNTAGGABLE, the CORS config via provider-schema fallback and the inline policy via the generated table's composite ROLENAME:POLICYNAME identity; -approve stamped both taggable resources, markers verified directly via aws s3api get-bucket-tagging / aws iam list-role-tags), test_plan (state deleted, live-plan \"No changes\", identities re-checked against the AWS CLI including the two untaggable resources' own content - CORS AllowedOrigins, inline policy's Resource ARN - since they carry no tag to re-read), test_apply (genuine no-op, 2 tagged objects before and after), and drift_reconverge (the bucket's hm_team tag tampered out of band, plan proposed fixing exactly that object, apply reconverged it; BREAK=1 verified load-bearing for both stage 2's identity check and stage 5's single-object assertion, tested in isolation for stage 5 per the corpus-vpc-complete convention since the shared BREAK var fails fast at stage 2 otherwise). Two non-blocking findings documented in the script's own header rather than routed around: aws_s3_bucket/aws_iam_role report DRIFTED during verification from AWS's own deprecated cors_rule/inline_policy shadow attributes reflecting a sibling resource created after the state snapshot (harmless, resolves by plan time), and the schema-admitted aws_s3_bucket_cors_configuration triggers the already-documented \"Resource type has no orphan recovery\" warning (live/LIMITATIONS.md, not a new gap). No choudoufu or floci gaps found - nothing filed. Merged to local main as c7fb650f4c (fix itself: 30577f6a56); justfile gained recipe demo-corpus-hongbomiao-labelbox; live/corpus-manifest.json gained the pin (reproducibility only, same convention as the sumaform entry - contributes nothing to a corpus-gen number)." @@ -963,16 +983,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -984,25 +1004,27 @@ "day2_count": "choudoufu: scaling aws_s3_bucket.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live CreationDate and tofu-address marker unchanged and tombstoning count_test[1]'s local record (has tombstone, no identity - the #398-guard shape); scaling back from 1 to 2 created exactly count_test[1] under the SAME bucket name (deterministic) but a NEW CreationDate (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied for real in the idle greenfield real-leg account, shows the identical shape: destroy the higher index only, create the higher index back under the same bucket name but a new CreationDate, the lower index's CreationDate unchanged both times", "day2_remove": "choudoufu: deleting module.kafka_kms_key_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy - the untaggable alias and its taggable parent key), applied cleanly (0 added, 0 changed, 2 destroyed) in an order the cloud accepted, the key is genuinely PendingDeletion and the alias is gone (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly two destroys for the same objects", "day2_rename": "moved block: module.hm_production_bucket renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.kafka_kms_key renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.kafka_kms_key_renamed's aws_kms_key_name argument proposed exactly one forced replace at the same declared address (the untaggable, client-named alias - 1 add, 1 change, 1 destroy overall) plus one in-place tag update on the taggable key itself, applied cleanly; the old alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key) is confirmed gone and the new alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2) points at the SAME key (e2ea5441-c9cd-4f92-85b7-4207a2c8c29a, read via the AWS CLI) - the key was never replaced; the local record store's record at the alias's address now names the new alias, not the destroyed one (alias/hongbomiao-storage-crossing-hm-kafka-kms-key -> alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2), while the key's own record at its own address is unchanged; the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace (the alias) plus one in-place key update. Scope notes: (1) this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see corpus-sqs-basic's own PART F; (2) BREAK=replace's marker-collision control is not exercised here - aws_kms_alias is untaggable and resolved by its own name, with no marker to plant a collision on, so that control's load-bearing-ness is proven instead by corpus-evoteum-modules and corpus-giantswarm-crossplane's own PART F sections against the taggable shape.", + "day2_replace": "choudoufu: changing module.kafka_kms_key_renamed's aws_kms_key_name argument proposed exactly one forced replace at the same declared address (the untaggable, client-named alias - 1 add, 1 change, 1 destroy overall) plus one in-place tag update on the taggable key itself, applied cleanly; the old alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key) is confirmed gone and the new alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2) points at the SAME key (b1f75b83-386c-46e8-86bb-ef14f4299ed7, read via the AWS CLI) - the key was never replaced; the local record store's record at the alias's address now names the new alias, not the destroyed one (alias/hongbomiao-storage-crossing-hm-kafka-kms-key -> alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2), while the key's own record at its own address is unchanged; the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace (the alias) plus one in-place key update. Scope notes: (1) this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see corpus-sqs-basic's own PART F; (2) BREAK=replace's marker-collision control is not exercised here - aws_kms_alias is untaggable and resolved by its own name, with no marker to plant a collision on, so that control's load-bearing-ness is proven instead by corpus-evoteum-modules and corpus-giantswarm-crossplane's own PART F sections against the taggable shape.", "drift_reconverge": "the plan proposed fixing 1 object(s) after the out-of-band tag mutation: module.s3_bucket_iot_data.aws_s3_bucket.main", "greenfield": "4 resources from nothing (2 buckets under aws.production, KMS key and untaggable alias under the default aws provider), markers verified via the AWS CLI, 4 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object per provider namespace, marker tags never compared", - "migrate": "3 of 4 stamped (2 buckets, KMS key), 1 UNTAGGABLE (KMS alias); bucket hongbomiao-storage-crossing-hm-production -> tofu-address=module.hm_production_bucket.aws_s3_bucket.main, bucket hongbomiao-storage-crossing-hm-iot-data -> tofu-address=module.s3_bucket_iot_data.aws_s3_bucket.main, key e2ea5441-c9cd-4f92-85b7-4207a2c8c29a -> tofu-address=module.kafka_kms_key.aws_kms_key.main", + "migrate": "3 of 4 stamped (2 buckets, KMS key), 1 UNTAGGABLE (KMS alias); bucket hongbomiao-storage-crossing-hm-production -> tofu-address=module.hm_production_bucket.aws_s3_bucket.main, bucket hongbomiao-storage-crossing-hm-iot-data -> tofu-address=module.s3_bucket_iot_data.aws_s3_bucket.main, key b1f75b83-386c-46e8-86bb-ef14f4299ed7 -> tofu-address=module.kafka_kms_key.aws_kms_key.main", + "plan_approval": "one argument edited (hm_production_bucket's common_tags gain Reviewed=yes), \"plan -out=approved.tfplan\" wrote a 10486-byte stock-format plan file whose whole change set is one update on module.hm_production_bucket.aws_s3_bucket.main; the world then moved out of band (hongbomiao-storage-crossing-hm-iot-data's hm_team tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.s3_bucket_iot_data.aws_s3_bucket.main and the live hongbomiao-storage-crossing-hm-iot-data it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - hongbomiao-storage-crossing-hm-production still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and hongbomiao-storage-crossing-hm-production read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 3 objects before, 3 after, no state file either time", "test_plan": "empty plan; identity re-check: both buckets' and the key's tofu-address unchanged, KMS alias still points at the same key" }, - "duration_s": 210.7, + "duration_s": 214.6, "stage_seconds": { - "cold_deploy": 33, + "cold_deploy": 20, "day2_count": 24, - "day2_remove": 8, - "day2_rename": 15, + "day2_remove": 7, + "day2_rename": 14, "day2_replace": 12, - "drift_reconverge": 6, - "greenfield": 63, - "migrate": 43, + "drift_reconverge": 5, + "greenfield": 58, + "migrate": 52, + "plan_approval": 15, "test_apply": 3, - "test_plan": 3 + "test_plan": 4 } }, "notes": "Landed 2026-08-18, the third estate in the OpenTofu-native lane and the second to clear all five stages, reusing corpus-hongbomiao-labelbox's already-pinned commit rather than a fresh sourcing search (the repo's OpenTofu-native bona fides and the pinned commit's clone were already established by that crossing). Scoped after surveying every section of the monorepo's aws/general and aws/storage files via the GitHub API against the pinned commit, no clone needed for scouting: Kafka Manager, two Amazon EMR sections and AWS Batch all read another environment's terraform_remote_state (out of scope, same reason corpus-hongbomiao-labelbox's own scoping excluded them); Amazon SageMaker was ruled out with a real, confirmed floci gap - aws sagemaker create-notebook-instance against a live floci container returns \"UnknownOperationException: Operation CreateNotebookInstance is not supported by floci\", and the type has zero entries anywhere in live/floci-capabilities.json's Cloud Control sweep - documented in the script's header as evidence for whoever picks up SageMaker next, not filed as an issue since it was routed around rather than blocking anything. The real candidate: aws/storage/main.tofu's first three module calls (hm_production_bucket, kafka_kms_key, s3_bucket_iot_data) read no remote state at all, unlike everything after them in that file - two amazon_s3_bucket module calls plus one aws_kms_key module call (aws_kms_key + aws_kms_alias). All five stages verified for real against a live floci container: cold_deploy (tofu apply, \"4 added, 0 changed, 0 destroyed\", confirmed 0 objects pre-tagged), migrate (live-import: \"3 of 4 resource instance(s) are eligible for stamping\", 1 UNTAGGABLE - the KMS alias; -approve: \"3 resource(s) newly stamped, 0 already stamped, 0 failed, 1 skipped\"; markers for all three read back via raw AWS CLI matched exactly: module.hm_production_bucket.aws_s3_bucket.main, module.s3_bucket_iot_data.aws_s3_bucket.main, module.kafka_kms_key.aws_kms_key.main), test_plan (state deleted, live-plan \"No changes\", all three identities re-verified against the AWS CLI, including the untaggable KMS alias's live target), test_apply (genuine no-op, \"0 added, 0 changed, 0 destroyed\", object count unchanged at 3), and drift_reconverge (the IoT-data bucket's tag tampered out of band, plan proposed fixing exactly module.s3_bucket_iot_data.aws_s3_bucket.main and nothing else, reconverge apply changed exactly 1 resource). BREAK=1 verified load-bearing: correctly fails the stage-2 identity assertion (asserts the KMS key's tofu-address against a deliberately wrong resource name). No choudoufu gaps found beyond the SageMaker floci evidence above - nothing filed against this repo. Merged to local main as a720266bcc (fix itself: 3335f16893); justfile gained recipe demo-corpus-hongbomiao-storage (port 4725); no new live/corpus-manifest.json entry needed, reuses the existing hongbomiao pin." @@ -1027,16 +1049,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1048,25 +1070,27 @@ "day2_count": "choudoufu: scaling aws_iam_policy.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live arn and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under a NEW arn (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G0 stock oracle on the same 2-instance count block, applied fresh against the idle adopted-estate endpoint, shows the identical shape: destroy the higher index only, create the higher index back under a new arn, the lower index's arn unchanged both times. Synthetic block: this estate's only real count knob (aws_iam_policy.policy's count = var.create ? 1 : 0) is a boolean create toggle, not a scalable set - sanctioned fallback per live/GAUNTLET.md #8 and reference-ec2-vpc's own Part F.", "day2_remove": "choudoufu: deleting module.iam_policy_renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (iam get-policy on the old ARN now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (STAGE 1.5.5) also proposes exactly one destroy for the same object; classifyOrphans did not withhold the destroy even though module.iam_policy_renamed2's policy shares the same block key, because that surviving instance is bound, not unclaimed", "day2_rename": "moved block: module.iam_policy_from_data_source renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.iam_policy renamed with zero churn, marker rewritten in place (found and fixed live-mv's own missing issue #266 tag-index fallback and the arnJoinTable's missing iam:policy entry to get here); stock oracle over the same two-module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both ARNs unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.iam_policy_renamed2's ForceNew name_prefix argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old policy (arn:aws:iam::000000000000:policy/example-246b74803ee893de43f27cac43) is confirmed gone and the new policy (arn:aws:iam::000000000000:policy/example-v2-0f747c1a618bb2900d8fde9713) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's ARN, not the destroyed one (arn:aws:iam::000000000000:policy/example-246b74803ee893de43f27cac43 -> arn:aws:iam::000000000000:policy/example-v2-0f747c1a618bb2900d8fde9713); the next plan proposes no resource action; stock oracle on cold_deploy's own state (STAGE 1.5.6) also proposes exactly one replace at the same address (plan only, not applied). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one. BREAK=replace's manufactured marker collision IS now reported for this type (GitHub issue #411, fixed): 'Indistinguishable instances without per-instance markers' - GitHub issue #409, layered on top of #411's own fix, is why this is not corpus-sqs-basic's 'Two live resources claiming one slot' text despite the same shape - see PART F's own header and its BREAK=replace branch.", + "day2_replace": "choudoufu: changing module.iam_policy_renamed2's ForceNew name_prefix argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old policy (arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be) is confirmed gone and the new policy (arn:aws:iam::000000000000:policy/example-v2-96244c69014024ce6076c79bba) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's ARN, not the destroyed one (arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be -> arn:aws:iam::000000000000:policy/example-v2-96244c69014024ce6076c79bba); the next plan proposes no resource action; stock oracle on cold_deploy's own state (STAGE 1.5.6) also proposes exactly one replace at the same address (plan only, not applied). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one. BREAK=replace's manufactured marker collision IS now reported for this type (GitHub issue #411, fixed): 'Indistinguishable instances without per-instance markers' - GitHub issue #409, layered on top of #411's own fix, is why this is not corpus-sqs-basic's 'Two live resources claiming one slot' text despite the same shape - see PART F's own header and its BREAK=replace branch.", "drift_reconverge": "one object tampered (arn:aws:iam::000000000000:policy/example_from_data_source's Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag", "greenfield": "2 resources from nothing (both aws_iam_policy), markers verified via the AWS CLI, 2 records in the local record store (#364 A2), replan empty both with and without the local record store, both policies' documents and paths match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared", "migrate": "2 of 2 stamped, both carrying tofu-slot=0/0 read back through IAM (choudoufu #372); Apply complete! Resources: 0 added, 0 changed, 0 destroyed. - nothing left to converge", + "plan_approval": "one argument edited (module.iam_policy's tags gain Reviewed=yes), \"plan -out=approved.tfplan\" wrote a 11133-byte stock-format plan file whose whole change set is one update on module.iam_policy.aws_iam_policy.policy[0]; the world then moved out of band (arn:aws:iam::000000000000:policy/example_from_data_source's Example tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.iam_policy_from_data_source.aws_iam_policy.policy[0] and the live arn:aws:iam::000000000000:policy/example_from_data_source it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 2 objects before, 2 after, no state file either time", "test_plan": "no resource change proposed, nothing foreign; identity re-check (via the AWS CLI) both unchanged" }, - "duration_s": 177.6, + "duration_s": 166.1, "stage_seconds": { - "cold_deploy": 28, - "day2_count": 42, - "day2_remove": 8, + "cold_deploy": 17, + "day2_count": 40, + "day2_remove": 7, "day2_rename": 9, "day2_replace": 7, "drift_reconverge": 5, - "greenfield": 55, - "migrate": 18, + "greenfield": 45, + "migrate": 19, + "plan_approval": 12, "test_apply": 3, - "test_plan": 3 + "test_plan": 2 } }, "notes": "Upgraded from a real but pre-#274-pipeline predecessor script (choudoufu apply from a live block present from the start, delete state, replan empty twice) to the current five-stage shape, following corpus-vpc-complete/corpus-lambda-simple's structure. Verified for real in a fresh isolated worktree off local main (ff106e63a7), Docker/floci/AWS CLI throughout, not read from the predecessor's prior notes. All five stages pass cleanly: cold_deploy (plain terraform apply, \"Apply complete! Resources: 2 added\", confirmed 0 objects tagged before migration), migrate (live-import dry run verifies \"2 of 2 resource instance(s) are eligible for stamping\", -approve reports \"2 resource(s) newly stamped, 0 already stamped, 0 failed, 0 skipped\", both tofu-address/tofu-estate tags read directly through the AWS CLI: module.iam_policy.aws_iam_policy.policy:0 and module.iam_policy_from_data_source.aws_iam_policy.policy:0), test_plan (live-plan genuinely empty, both identities re-read unchanged after the state file's only copy was deleted), test_apply (\"0 added, 0 changed, 0 destroyed\", object count unchanged at 2), and drift_reconverge (one policy's Example tag tampered directly against floci, live-plan proposes fixing exactly that object, apply reconverges it to \"0 added, 1 changed, 0 destroyed\"). BREAK=1 verified twice, independently, against each stage it targets: run as committed it fails stage 3's identity check (expects the real policy's tofu-address on a module that was never created); run separately with stage 3's corruption disabled, it correctly fails stage 5 by tampering a second object and proving the \"exactly one object\" count assertion is load-bearing (both objects flagged, not silently 1). NEW FINDING, not previously documented in any real crossing that reached this deep: live-import -approve deliberately writes only tofu-estate and tofu-address, never tofu-slot (internal/live/stamp/doc.go's own \"tofu-slot comes in from outside\" - a slot is minted from a monotonic counter over the live set that a read-only, one-state-file view cannot compute). Both of this estate's aws_iam_policy resources declare count = var.create ? 1 : 0, exactly the shape that needs one, so the FIRST live-plan straight after live-import -approve is not empty - it proposes adding tofu-slot=\"0\" to both, and nothing else. Folded into stage 2 as one ordinary `choudoufu apply` (\"0 added, 2 changed, 0 destroyed\") before stage 3 is attempted; every replan after is genuinely empty. This is real, deliberate, already-documented product behavior, not a defect - but it will recur on any count-based resource crossing that reaches this far and had not yet been noticed in one that actually got here. Also caught and fixed while verifying: a self-authored bug where stage 5's negative drift assertion compared a live-plan diff header's address (bracket form, \"policy[0]\") against the escaped tag-value form (\"policy:0\") and could never have matched - a vacuous check that a stricter assertion in the sibling script (see corpus-iam-read-only-policy) surfaced; fixed here by keeping both forms as separate variables." @@ -1091,16 +1115,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1112,24 +1136,26 @@ "day2_count": "choudoufu: scaling aws_iam_policy.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live PolicyId and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under the SAME ARN (deterministic from name+path) but a NEW PolicyId (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied fresh in the idle greenfield-oracle account, shows the identical shape: destroy the higher index only, create the higher index back under the same ARN but a new PolicyId, the lower index's PolicyId unchanged both times", "day2_remove": "choudoufu: deleting module.read_only_iam_policy_final's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (iam get-policy on the old ARN now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; classifyOrphans did not withhold the destroy because no other aws_iam_policy.policy block anywhere in this config ever declares a real instance (count=0 on both remaining module calls)", "day2_rename": "moved block: module.read_only_iam_policy renamed to module.read_only_iam_policy_moved with zero churn (0 add, 1 change, 0 destroy), tofu-address marker rewritten in place; live-mv: module.read_only_iam_policy_moved renamed to module.read_only_iam_policy_final with zero churn, marker rewritten in place; stock oracle over the identical net rename on cold_deploy's own state also shows a true no-op (0 add, 0 change, 0 destroy, outputs unchanged in value); the live policy ARN unchanged throughout, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.read_only_iam_policy_final's ForceNew description argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-cb1412a333b8859faa022fa5d1) is confirmed gone and the new object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-0f29efba6ef044410ac1522988) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-cb1412a333b8859faa022fa5d1 -> arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-0f29efba6ef044410ac1522988); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", - "drift_reconverge": "one object tampered (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-cb1412a333b8859faa022fa5d1's Example tag), plan proposed fixing exactly module.read_only_iam_policy.aws_iam_policy.policy[0], apply changed 1 and reconverged the tag", + "day2_replace": "choudoufu: changing module.read_only_iam_policy_final's ForceNew description argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7) is confirmed gone and the new object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-84581510d3c9ebbf3d8288dffa) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7 -> arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-84581510d3c9ebbf3d8288dffa); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", + "drift_reconverge": "one object tampered (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7's Example tag), plan proposed fixing exactly module.read_only_iam_policy.aws_iam_policy.policy[0], apply changed 1 and reconverged the tag", "greenfield": "1 resource from nothing, marker verified via the AWS CLI, 1 record in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (path, description, policy document)", "migrate": "1 of 1 stamped, carrying tofu-slot=0 read back through IAM (choudoufu #372); Apply complete! Resources: 0 added, 0 changed, 0 destroyed. - nothing left to converge", + "plan_approval": "one argument edited (aws_iam_policy.approval_probe's Reviewed tag, no -> yes), \"plan -out=approved.tfplan\" wrote a 27324-byte stock-format plan file whose whole change set is one update on aws_iam_policy.approval_probe; the world then moved out of band (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7's Example tag, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.read_only_iam_policy.aws_iam_policy.policy[0] and the live arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - arn:aws:iam::000000000000:policy/example/iam-ro-approval-probe still read Reviewed=no through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:iam::000000000000:policy/example/iam-ro-approval-probe read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The reviewed object is a self-contained synthetic aws_iam_policy.approval_probe (sanctioned fallback per live/GAUNTLET.md #8, same discipline as PART G's count_test) because this estate has exactly ONE real object and the leg needs two disjoint rows; it is created in P0 and destroyed in P5, and the module policy's ARN and Example tag are read back unchanged so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 1 objects before, 1 after, no state file either time", "test_plan": "no resource change proposed, nothing foreign; identity re-check (via the AWS CLI) unchanged" }, - "duration_s": 189.5, + "duration_s": 191.2, "stage_seconds": { - "cold_deploy": 25, + "cold_deploy": 15, "day2_count": 19, - "day2_remove": 7, - "day2_rename": 10, - "day2_replace": 7, + "day2_remove": 6, + "day2_rename": 9, + "day2_replace": 8, "drift_reconverge": 5, - "greenfield": 45, - "migrate": 65, - "test_apply": 3, + "greenfield": 41, + "migrate": 63, + "plan_approval": 19, + "test_apply": 2, "test_plan": 3 } }, @@ -1155,16 +1181,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1173,28 +1199,30 @@ "exit_code": 0, "detail": { "cold_deploy": "8 resources, genuinely cold, genuinely unmarked", - "day2_count": "synthetic count block, and the header says why: .corpus/lambda declares 38 count/for_each expressions and not one is scalable - 36 are boolean create toggles (count = ... ? 1 : 0), the for_each maps are empty in this example, and the only two numeric ones (number_of_policies, number_of_policy_jsons) are gated off by default and drive the untaggable aws_iam_role_policy_attachment, so turning either on would both change the estate every earlier stage measured and leave no marker to read back. So aws_iam_role.count_test (count = 2), a self-contained block of a type this estate already exercises, added after Part E's completed removal. choudoufu: scaling 2 -> 1 proposed exactly 0 add, 0 change, 1 destroy, destroying count_test[1] and never naming count_test[0]; after the apply lambda-simple-count-test-1 no longer answers GetRole at all (verified absence) while count_test[0] kept its server-minted RoleId AROAZY2RIPLGDLB3N0TZ and its tofu-address=aws_iam_role.count_test:0 marker, both read through the AWS CLI. Scaling 1 -> 2 proposed exactly 1 add, 0 change, 0 destroy and count_test[1] came back as a genuinely NEW object - RoleId AROAGXAJJPJ2ED9EO668 -> AROAVNKGVMUSSSKUZC9V under the same deterministic name, the witness an IAM role's ARN cannot provide since it is rebuilt from account plus name (the same reason this estate's own Lambda function ARN could not have witnessed it) - carrying tofu-address=aws_iam_role.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G0): the identical block stood up for real with plain terraform in its own working directory on the idle greenfield-oracle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new RoleId (AROACC4A5TI4EAICMRGS -> AROATTFBWT3X96T6C4AB), the lower index's RoleId (AROAVPYZM4YHIDNFQUC6) unchanged both times. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, so the assertion is load-bearing", + "day2_count": "synthetic count block, and the header says why: .corpus/lambda declares 38 count/for_each expressions and not one is scalable - 36 are boolean create toggles (count = ... ? 1 : 0), the for_each maps are empty in this example, and the only two numeric ones (number_of_policies, number_of_policy_jsons) are gated off by default and drive the untaggable aws_iam_role_policy_attachment, so turning either on would both change the estate every earlier stage measured and leave no marker to read back. So aws_iam_role.count_test (count = 2), a self-contained block of a type this estate already exercises, added after Part E's completed removal. choudoufu: scaling 2 -> 1 proposed exactly 0 add, 0 change, 1 destroy, destroying count_test[1] and never naming count_test[0]; after the apply lambda-simple-count-test-1 no longer answers GetRole at all (verified absence) while count_test[0] kept its server-minted RoleId AROAC7ERQHW7CEUA1EZL and its tofu-address=aws_iam_role.count_test:0 marker, both read through the AWS CLI. Scaling 1 -> 2 proposed exactly 1 add, 0 change, 0 destroy and count_test[1] came back as a genuinely NEW object - RoleId AROATSWACNJVMMY25Q6P -> AROACJAD70F33ERTS33F under the same deterministic name, the witness an IAM role's ARN cannot provide since it is rebuilt from account plus name (the same reason this estate's own Lambda function ARN could not have witnessed it) - carrying tofu-address=aws_iam_role.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G0): the identical block stood up for real with plain terraform in its own working directory on the idle greenfield-oracle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new RoleId (AROAQ945VI0K0RAZ648E -> AROA97JTCE39WS5SJGQN), the lower index's RoleId (AROAYT6M787L61HL0SJ3) unchanged both times. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, so the assertion is load-bearing", "day2_remove": "choudoufu: deleting module.lambda_function_final's block proposed 7 destroys (the function, the role, its inline aws_iam_role_policy.logs[0] CloudWatch Logs policy, and all three record-located children always; the log group's only when floci's GetResources happens to index it - a documented emulator gap, confirmed by reading logs:list-tags-for-resource directly against the same live object), applied cleanly, the function, the role and the inline log policy genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no further resource action; classifyOrphans did not withhold any destroy as a possible rename. WANT_DESTROY_COUNT moved from 5/6 to 6/7 in this same commit: the inline log policy was previously missing from this stage's own checklist entirely - a genuine leak (an untaggable IAM permission left behind on every destroy of this estate), not a stale assertion, fixed as part of the day2_replace unit that re-measured this stage", "day2_rename": "moved block: module.lambda_function renamed to module.lambda_function_moved with zero churn (0 add, 3 change, 0 destroy) across all seven of its stateful children, three taggable markers rewritten in place, three record-located children moved via their own per-resource moved blocks with zero diff, one config-derived child (aws_iam_role_policy.logs) needing none; stock oracle over the identical seven-resource move on cold_deploy's own state also shows zero churn beyond the module's own pre-existing null_resource.archive[0] package-timestamp noise (confirmed present on an unrelated baseline replan too); live-mv: module.lambda_function_moved renamed to module.lambda_function_final across all three taggable children (the function, the role, the log group), one call each, zero churn, markers rewritten in place - the internal/live/mv/mv.go materialize() RecordStore wiring gap (build.go:1676's \"Record-backed instance with no record store\") is fixed; all three live objects unchanged throughout, read via the AWS CLI; final replan is empty", - "day2_replace": "choudoufu: changing module.lambda_function_final's ForceNew logging_log_group-derived name proposed exactly one replace at the same declared address (the log group; -/+ destroy and then create) cascading into two expected in-place updates (the function's logging_config, the inline log policy's document) and nothing else beyond the module's own pre-existing null_resource.archive[0] package-timestamp noise; applied cleanly; the old object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/united-mantis-lambda-simple) is confirmed gone and the new object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/united-mantis-lambda-simple-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (/aws/lambda/united-mantis-lambda-simple -> /aws/lambda/united-mantis-lambda-simple-v2); the next plan proposes no resource action beyond the same known noise; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace plus the same in-place cascade (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", + "day2_replace": "choudoufu: changing module.lambda_function_final's ForceNew logging_log_group-derived name proposed exactly one replace at the same declared address (the log group; -/+ destroy and then create) cascading into two expected in-place updates (the function's logging_config, the inline log policy's document) and nothing else beyond the module's own pre-existing null_resource.archive[0] package-timestamp noise; applied cleanly; the old object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/free-wasp-lambda-simple) is confirmed gone and the new object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/free-wasp-lambda-simple-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (/aws/lambda/free-wasp-lambda-simple -> /aws/lambda/free-wasp-lambda-simple-v2); the next plan proposes no resource action beyond the same known noise; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace plus the same in-place cascade (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one object tampered (memory_size 128->256), exactly module.lambda_function.aws_lambda_function.this[0] proposed by both choudoufu and stock with the identical change, apply changed 1 and memory_size reads back as 128", "greenfield": "8 resources from nothing (3 taggable + 5 record-backed/config-derived), all three module-nested markers verified via the AWS CLI, 8 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (runtime, handler, memory, timeout, log-group retention)", "migrate": "3 stamped, 4 recorded, 0 failed, 1 skipped", + "plan_approval": "one argument edited (module.lambda_function's cloudwatch_logs_retention_in_days, unset -> 14, which reaches module.lambda_function.aws_cloudwatch_log_group.lambda[0] and nothing else - the module call carries no tags argument and every tags-shaped knob it has would reach three children at once), \"plan -out=approved.tfplan\" wrote a 31790-byte stock-format plan file whose whole change set is that one in-place update; the world then moved out of band (free-wasp-lambda-simple's memory_size 128->256, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming the extra row as \" module.lambda_function.aws_lambda_function.this[0] Update free-wasp-lambda-simple\" - both module.lambda_function.aws_lambda_function.this[0] and the live identity it was computed against - with \"Exit status 3\" spelled out for a pipeline; nothing was applied - /aws/lambda/free-wasp-lambda-simple still carried no retentionInDays, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with memory_size put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the log group read back with retentionInDays=14, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (retention unset again, memory_size still 128, next plan proposes no resource action, no state file left behind) so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 3; markers and record store intact", "test_plan": "no resource change proposed" }, - "duration_s": 171.8, + "duration_s": 173.6, "stage_seconds": { - "cold_deploy": 28, - "day2_count": 26, + "cold_deploy": 13, + "day2_count": 28, "day2_remove": 11, - "day2_rename": 24, + "day2_rename": 25, "day2_replace": 18, - "drift_reconverge": 18, - "greenfield": 25, + "drift_reconverge": 19, + "greenfield": 24, "migrate": 14, + "plan_approval": 15, "test_apply": 5, - "test_plan": 3 + "test_plan": 2 } }, "notes": "Re-verified 2026-08-18 in a fresh isolated worktree off local main (bcf78bacbd), for real (Docker/floci/AWS CLI, not read from a prior note). Stages 1 and 2 still pass exactly as landed: cold apply creates 8 resources, live-import verifies '3 of 8 resource instance(s) are eligible for stamping', -approve reports '3 resource(s) newly stamped, 0 already stamped, 0 failed, 5 skipped', and module.lambda_function.{aws_lambda_function,aws_iam_role,aws_cloudwatch_log_group} carry the expected module-qualified tofu-address/tofu-estate tags read straight through the AWS CLI. #303 (the count=var.enable_x?1:0 zero-instance admission gap on aws_lambda_function_url.this and aws_lambda_function_recursion_config.this) is CONFIRMED FIXED: re-running live-plan against current main, neither type appears anywhere in the diagnostics any more, as an error or a warning - stage 3 now fails on exactly one Error block, not two-plus. That block is local_file.archive_plan (module.lambda_function's package.tf:44, count = var.create && var.create_package ? 1 : 0, both true by default in this example so a real non-zero instance, not a zero-count block #303's fix would clear), refused under the logical-resource rule. Investigated whether this is a bug or correct behavior: it is correct, deliberate, and already ruled on. Issues #237 and #238 (both closed 2026-08-18) put local_file through exactly this question and #238's closing comment states local_file is 'deliberately left OTHER_REFUSED with a documented reason: neither of lint's two classes fits it correctly (its identity is argument-derived, not record-backed, and promoting it would silently reopen a count.index collision hazard a dedicated test already guards) - a genuine third-classification gap, not an omission, correctly left open rather than forced.' local_file's identity is a filename on the local disk of whatever machine ran apply - not a cloud object, nothing taggable, nothing an AWS CLI call could ever read back to confirm it still exists - so there is no live counterpart for a stateless replan to reconcile against. Considered scoping the estate around it the way corpus-vpc-complete/corpus-sumaform-aws scope around their own out-of-scope resources (the module's create_package=false + local_existing_package= toggle skips package.tf's local_file entirely) and rejected it: unlike sumaform's provision=false, which picks between the module's own equally-real published deployment modes to route around an infra-emulation gap in floci, swapping to a pre-built zip would replace the actual thing 'simple' demonstrates - the module's own default packaging pipeline - with a materially different scenario this corpus entry was never meant to test. Left as a real, reported block; run.sh's header carries the full investigation. One piece of relevant good news found along the way: #275 (closed 2026-08-18) built a record_store-gated residue mechanism for exactly the aws_lambda_function.filename/source_code_hash/publish phantom-diff problem a filename-deployed Lambda would otherwise hit under stateless replanning - this estate already declares a record_store, so once local_file gets its own identity class nothing here looks likely to re-hit that problem. test_apply and drift_reconverge remain not_run because test_plan does not pass; not attempted this pass since attempting them against a still-refused plan would prove nothing. No issue currently tracks the missing 'argument-derived-but-safe' LogicalClass itself (the actual unblock for this estate) - #237/#238 are both closed and did not spawn a follow-up; one may be worth filing if this crossing is prioritized again. Follow-up pass 2026-08-18 (#313 cross-check, fresh worktree off main at d303d9d425, real Docker/floci/AWS CLI run): confirmed the local_file.archive_plan block above is NOT #313's data.aws_availability_zones/static-context wall - grepped the full live-plan output, zero occurrences of that diagnostic. Filed the missing-LogicalClass follow-up this note flagged as worth filing but hadn't been: #314 ('local_file needs a fourth LogicalClass (argument-derived identity)'), citing #237/#238's rulings and the count-index guard test. No fix attempted - real multi-package work (lint + identity resolution + row-gen), not a quick derivation. Follow-up pass 2026-08-19 (#314 fixed and merged, d878aa914b/e7bb1f4b41): the fourth LogicalClass exists, but not the one the issue's own framing predicted - two of that framing's premises turned out false, checked rather than assumed. hashicorp/local's provider implements NO ImportState for local_file at all (confirmed against stock tofu: 'This resource does not support import'), so an argument-derived identity handed to internal/live/projection would have turned a lint refusal into a hard Cannot-import-for-projection error - strictly worse. And this estate's own filename argument isn't static anyway (it reads a data.external result). The real fix is ClassExternalAdmitted/EXTERNAL_ADMITTED: a record_store admits local_file the same way it admits ClassRecordAdmitted, resolving through ClassRecordBacked - the record is the only carrier that can bring prior state back for a type the provider itself cannot re-derive. Reaches exactly one type (local_file; local_sensitive_file is secret-bearing) but the RULE generalizes: live/logical-schemas.json's per-provider store_only now SELECTS between the two admitted classes instead of gating whether a type derives a row at all, which also retired the hand-written local_sensitive_file exception in ClassifyLogicalType - a net type-name-literal deletion, not an addition. Count-index guard (TestLocalFileKeepsItsCountIndexCheck) confirmed still holding via two separate mutation checks. Real re-crossing: local_file is admitted and appears nowhere in live-plan's diagnostics any more (asserted by absence) - but test_plan stays fail, now BLOCKED at 5 sites, a FOURTH wall newly reached rather than caused. All five trace to one expression, function_name = \"${random_pet.this.id}-lambda-simple\": random_pet.this is RECORD_ADMITTED so its id lives only in the record store, and the identity resolver declines to read that carrier for the three dependent resources (aws_iam_role.lambda, aws_iam_role_policy.logs, aws_lambda_function.this, aws_cloudwatch_log_group.lambda, one cascade) even though all three are already stamped and CLI-verified by stage 2 of the same run - choudoufu already holds the value and the objects are already marked, so this reads as an identity-resolver gap rather than a missing carrier. Not filed (no issue number assigned) - worth a slot. test_apply/drift_reconverge remain not_run, blocked on this new wall. Two real corrections made along the way: the crossing script had no AWS provider version pin (silently drifted to whatever the newest release was, now pinned =6.59.0 matching corpus-cloudfront's discipline), and live/LIMITATIONS.md's local-file section stated 'no cloud counterpart to reconcile against' as fact - false; the local filesystem is the counterpart, now corrected there too. Follow-up pass 2026-08-19/20 (#336 fixed and merged, 821c769715/c41279989a): #336's own diagnosis was wrong in two places, checked rather than assumed. The identity resolver was NOT declining to read the record-store carrier - resolver.parentPart already read random_pet.this.id correctly on unmodified main. What actually refused was coalesce(): iam.tf/main.tf's role_name/policy_name/log-group-name chains all select through coalesce(var.X, var.Y, \"*\")-shaped expressions, and resolver.isSymbolic reads only an expression's traversal ROOTS - var/local are never symbolic, so a selection sitting behind a module argument or a local looked entirely static, failed whole-expression evaluation, and had nothing left to try (the decomposition switch is only reached when a resource is named directly inside the expression). Fixed generically (internal/live/identity/coalesce.go, new): resolveCoalesceCall decides which argument the language selects using two proofs (provably-null-or-empty to skip, provably-non-null-non-empty to select), declining the whole call on anything undecidable rather than silently falling through - mutation-tested three ways (drop the non-emptiness proof, fall through on undecidable, remove the call entirely), each caught. Measured reach: refusal-probe sites 16075->15964 (-111), instances 4499->4522 (+23), 12 entries improved across six unrelated sources, 0 worse - schema-less mode, an under-report since it's blind to the record-backed half of this estate's own chain. Real re-crossing: live-plan diagnostics 5->0, the plan runs to completion for the FIRST time - but test_plan still FAILS, now on a genuinely NEW, fifth wall: 'live-plan is not empty', proposing to create every record-backed resource (random_pet.this first) from scratch. Root cause, and #336's second wrong premise: live-import's Approve loop only calls #327's recordResidueFor for a STAMPED entry; a record-backed resource is by definition not stampable (no live cloud object to tag) and hits OutcomeSkipped, continuing past the residue call entirely - so the record store is empty for every record-backed instance after a clean migrate, not populated as #336 assumed. Filed as #340, not attempted - the fix is migrate seeding the record store from the migrated state's own object for every record-backed instance, a sibling call to recordResidueFor on the skipped-because-record-backed path. test_apply/drift_reconverge remain not_run, now blocked on #340 instead of #336's five diagnostics. Follow-up pass 2026-08-20 (#340 fixed and merged, d30daa156f/8d34e3ded9): the issue's own framing was half right - recordResidueFor does NOT gate on 'stamped', it runs for any entry with an *eligible; the real gate is one line earlier, Ratify never building an *eligible for a record-backed type at all. Fixed with Approve's second write path, the sibling of the tag write: projection.SeedRecordForInstance writes a record-backed instance's object into the record store, byte-identical to what an apply's WriteBack would write, reading before writing so an already-correct record is a no-op and a genuinely different one refuses rather than clobbers. Keys on identity.TypeIdentity.RecordBacked and nothing else - 15 types across 4 providers (local/null/random/time/terraform_data), no aws_*/random_* name in the control flow. Real re-crossing: STAGE 2 migrate reports '3 newly stamped, 0 already stamped, 4 newly recorded, 0 already recorded, 0 failed, 1 skipped', the store's own files grepped for random_pet.this's generated id. STAGE 3: live-plan now raises ZERO diagnostics for the first time ever on this estate, every identity resolves, no record-backed resource is proposed for creation - but the plan is not empty: 0 to add, 2 to change, 0 to destroy, a SIXTH wall. Both changes are real and distinct from every prior wall: (1) a nested-block round-trip on aws_lambda_function (- environment {} / + logging_config { log_format = \"Text\" }, floci's Lambda read vs the module's config), (2) a sensitivity-only diff on local_file.content (OpenTofu's own renderer says 'The value is unchanged' - a genuine limitation the fixing agent found and pinned rather than hid: ResourceInstanceObjectSrc.Decode re-applies AttrSensitivePaths so the decoded value is marked, ctyjson.Marshal panics on a marked leaf, the fix unmarks before encoding, and projection.recordPayload has nowhere to store the sensitivity path - so the record carries the value but not the mark. projection.WriteBack shares this same hole, worse (no unmark of its own, would panic on the identical object after a real apply) - unfiled, not touched, worth a slot). test_apply/drift_reconverge remain not_run, blocked on this sixth wall now. Separately, #341 (found by the sibling corpus-mastino-dns crossing, same ratify.go gate but a different population - untaggable ordinary AWS types like aws_route53_record, architecturally guaranteed never to be RecordBacked per TestResolveNeverEmitsRecordBackedForAWSEstate) is CONFIRMED STILL OPEN after #340 - the two issues share a root-cause line but #340's fix is deliberately scoped to the disjoint RecordBacked population and does not reach it. #340's own recordable/Approve sibling-carrier pattern is flagged as a reasonable template for #341's eventual fix. Follow-up pass 2026-08-20 (worktree ../wt/lambda-simple-sixth-wall, branch live/lambda-simple-sixth-wall, off LOCAL main ea9fd62fc0): the sixth wall was REPRODUCED for real first, not read off this note - STAGE 1 PASS, STAGE 2 PASS, STAGE 3 with zero diagnostics and 'Plan: 0 to add, 2 to change, 0 to destroy', both changes verbatim as recorded above. Stages are UNCHANGED at 2 of 5, and the reason is now measured rather than argued: the two changes belong to two different projects. (a) local_file.archive_plan's sensitivity-only diff IS choudoufu's and is FIXED (f66bc9e043). projection.recordPayload gained SensitiveAttrs, encoded exactly the way a state file encodes sensitive_attributes: encodeRecordPayload splits the value from its marks itself so no caller unmarks, decodeRecordPayload puts them back the way ResourceInstanceObjectSrc.Decode does, and materializeRecord re-marks after the schema conversion so obj.Encode derives AttrSensitivePaths from them. The mechanism the wall turned on is that live-plan runs the plan graph with SkipRefresh (live_plan.go:499), so a projected object's AttrSensitivePaths is the ONLY marks the plan's 'before' side ever has - upstream re-marks a refreshed object at node_resource_abstract_instance.go:1106 and that line is never reached - while the 'after' side is re-marked from the config and the provider schema every run at :1383. Derived from the object's own marks and nothing else, so it reaches any record-backed type with any sensitive attribute at any path; a mark that is not marks.Sensitive is refused rather than dropped. Four mutation checks, each caught, and the shape test consults an EXTERNAL source (it writes a real state file through internal/states/statefile and requires the same JSON for the same paths) rather than round-tripping against itself. TestIdentityGolden 0 changed, 0 added, 0 removed; ./internal/live/..., ./tools/..., ./live/..., ./cmd/... and ./internal/command/ all green. (b) aws_lambda_function's '- environment {}' / '+ logging_config { log_format = \"Text\" }' pair is NOT choudoufu's, and this is now proven by a CONTROL rather than reasoned about: run.sh's new step 3b runs plain terraform, its own state file, its own refresh, 'terraform plan -detailed-exitcode' immediately after its own cold apply with no choudoufu anywhere in the run, and it replans NOT EMPTY on exactly module.lambda_function.aws_lambda_function.this[0] and nothing else. Asserted by value, so a new emulator gap breaks the script instead of hiding in a bucket labelled expected, and a fixed one breaks it too. Filed as lex00/floci#83 with both causes located in LambdaController.buildFunctionConfiguration: Environment is emitted unconditionally ('SDK expects it even when empty') where real AWS omits it for a function that never had one - which is why terraform-provider-aws reads it under 'if function.Environment != nil' - and LoggingConfig is neither stored nor emitted at all, where real AWS always returns one defaulting to Text. The module declares zero environment blocks (main.tf:90, a dynamic block over an empty map) and one logging_config unconditionally (main.tf:136), so both fire on the module's DEFAULTS. Stage 3 now splits its own plan against the step-3b control with comm and names only the remainder as choudoufu's. WHAT IS NOT VERIFIED, stated rather than implied: the post-fix crossing was started and its choudoufu init was killed by the harness before stage 2, so the sensitivity fix has NOT been observed clearing the diff in a real end-to-end run - it is verified by unit tests that drive the two real paths (WriteBack after an apply, then BuildWith/materializeRecord on the next plan) and assert inst.Current.AttrSensitivePaths by value. The next run of this script is what settles it, and it should still end at 'live-plan is not empty' with the Lambda alone until lex00/floci#83 lands. Two further issues filed from this pass, neither fixed: #343 (builder.materialize applies no schema.Block.ValueMarks to what a provider Read returned, so the identical perpetual diff exists for any CONCRETE cloud object with a Sensitive attribute - separate because it also changes what b.live means for identity composition, and unmeasured over the corpus) and #344 (a record written before SensitiveAttrs existed now conflicts with its own re-migration though the value is identical, because SeedRecordForInstance compares bytes; population is local_file plus any config-derived mark, and the format is one day old). Stages 4 and 5 remain unwritten: an empty stage-3 plan is their precondition and lex00/floci#83 is what stands between this estate and one. Follow-up pass 2026-08-20 (worktree ../wt/lambda-simple-floci83, branch live/lambda-simple-floci83): lex00/floci#83 is FIXED and CLOSED, settled against real AWS (one throwaway Lambda function + IAM role created and immediately deleted, disclosed to the user per standing permission) rather than assumed - GetFunction/GetFunctionConfiguration omits Environment entirely for a function that never had one (not present-but-empty) and always returns LoggingConfig, defaulting to Text. Fixed in LambdaController/LambdaService/LambdaFunction, verified three ways (local instance, direct probe, and the same probe against the published GHCR image). Re-crossing after the re-pin: STAGE 3 (test_plan) now raises zero diagnostics and proposes changing ZERO resources for the first time ever on this estate - both prior sixth-wall diffs (the sensitivity-only local_file.content diff and the aws_lambda_function nested-block mismatch) are gone, verified by reading the raw terraform plan output directly rather than trusting extracted variables. But the plan is still not empty: all 23 of the example module's root-level `output` blocks render as `+ new` on every run. Root cause: internal/live/projection.Manager.GetRootOutputValues always returns an empty map - nothing evaluates the config's own output blocks against the reconstructed prior state before the plan graph asks for it. A genuinely new, generalizable gap (neither corpus-mastino-dns nor corpus-evoteum-modules, the two other closest-to-5/5 crossings, declares any root-level output, so nothing had hit this before). Filed as INTENTIUS/choudoufu#348, not attempted - core projection-architecture work, not a quick derivation. Stages remain 2 of 5; the estate's real blocker moved from floci#83 to #348, which is now the sole thing standing between this estate and an empty stage-3 plan. Follow-up pass 2026-08-20 (primary checkout, local main db1f412cfd, real Docker/floci run - GitHub issue #340 verification): #340 was found ALREADY FIXED on main (d30daa156f/8d34e3ded9, confirmed by commit history and by internal/live/liveimport/record_test.go's TestApprove_SeedsTheRecordStoreForARecordBackedInstance/TestRecordBackedTypeReadsTheGeneratedTable, the latter covering random_pet/null_resource/terraform_data/local_file/time_sleep/random_id generically), so this pass re-verified rather than re-fixed. Re-crossing confirms #340's own fix by absence again: 'no record-backed resource is proposed for creation: the migrate seeded all four' and 'no sensitivity-only diff on local_file.archive_plan: the record carries its marks' both print in stage 3's own output. #349 ('see through provably-zero-instance blocks when evaluating root outputs', 88d7e3961e, landed after this note's #348 paragraph) cut the output-only diff from all 23 root outputs to exactly 2: 'lambda_function_arn_static' and 'local_filename', both still '+' on every run. Stage 3 (test_plan) is still BLOCKED - not yet empty - but the wall is now two output lines, not twenty-three, and #340 itself contributes zero diagnostics and zero sites to what remains. Stages unchanged at 2 of 5 pass; the residual 2-output gap is #348/#349's remaining scope, not #340's, and was not investigated further here (out of this issue's scope). Follow-up pass 2026-08-21 (isolated worktree off local main 860c29e129, real Docker/floci/AWS CLI run, identical harness run TWICE with only TOFU_BIN swapped - not read from any prior note): #349's sub-problem 2, the root-output data-source read, is now built, and this estate's stage-3 output diff went from 2 lines to 1. Measured: at 860c29e129 the plan's 'Changes to Outputs:' block carries 'lambda_function_arn_static' and 'local_filename'; with the fix it carries 'local_filename' alone. lambda_function_arn_static vanishes from the diff entirely rather than rendering as '~ old -> new', which is the stronger result: the plan graph independently computed the same value the pre-plan read computed for the prior side, so they cancel. The three data sources behind it (data.aws_partition.current, data.aws_region.current and data.aws_caller_identity.current, all in module.lambda_function) are read live before the plan through the same configured aws provider instance the projection already reads this estate through. local_filename is UNCHANGED and stays refused ON PURPOSE: it reaches data.external.archive_prepare, whose read runs package.py on the machine running the plan, and the new demand class is confined to providers this configuration manages live objects through (dataread.LiveProviders) - the external provider serves no managed resource type at all, in this or any configuration, so it is excluded structurally rather than by name. Stages are UNCHANGED at 2 of 5: cold_deploy pass, migrate pass, test_plan still FAIL (one output line is still one output line, so the plan is still not empty), test_apply and drift_reconverge still not reached. This narrowed the stage-3 diagnostic count; it did not clear the stage. Follow-up pass 2026-08-29 (isolated worktree off local main 499f9f5e80, real Docker/floci/AWS CLI run - GitHub issue #498, 'migrate pass and fail 23 minutes apart at the same emulator pin'): CONFIRMED as a real, deterministic defect, not a flake and not #497's runner-resource-pressure hypothesis. Reproduced 100% of the time under a controlled variable rather than by chance: migrate PASSED 7/7 consecutive local runs (this machine's stock terraform, v1.15.8) with byte-identical output every time, then FAILED 2/2 runs the instant HashiCorp Terraform v1.16.0 was forced first on PATH for stage 1's cold deploy, with the exact nightly failure text ('live-import -approve did not stamp 3 and record 4 of 8 resources cleanly'). Root cause, isolated to one byte: Terraform >=1.16.0's built-in terraform_data resource gained a new 'store' nested block (verified directly via `terraform providers schema -json` against terraform.io/builtin/terraform, no choudoufu involved) that choudoufu's own terraform_data schema (internal/builtin/providers/tf/resource_data.go, unchanged since the OpenTofu fork) does not declare; decoding a stock-terraform-1.16-produced cold.tfstate's terraform_data instance against that older schema failed with 'unsupported attribute \"store\"' (internal/live/liveimport/ratify.go's ratifyRecordBacked, its inst.Current.Decode(schema.Block.ImpliedType()) call), demoting terraform_data.package_filename_for_hash from RECORDED to SKIPPED and changing live-import -approve's summary line, which trips run.sh's exact-string assertion. The 'pass at 10:23:19Z, fail at 10:46:59Z, same commit' shape in #498 was never nondeterminism in the same environment: the passing row was a local worker's run against an older pinned-by-brew terraform (like this pass's own baseline), the failing row was CI's `hashicorp/setup-terraform@v3` with terraform_version: latest (confirmed 1.16.0 from the nightly's own log) - two different environments measuring the identical commit near-simultaneously, not one environment flip-flopping. Downloaded the eks-basic sibling failure from the same nightly run for comparison and confirmed it is a DIFFERENT failure (a runner tofu-on-PATH casualty per #497, mis-attributed to whatever CURRENT_STAGE was set to when it died) - #498's own caveat about #497 does not reach this estate's failure, which is fully explained by the terraform_data schema gap alone. Fixed generically for the one type it can reach (dataStoreResourceSchema() now declares 'store' as a NestedType Object attribute mirroring the real schema's own field shape and WriteOnly/Sensitive flags, verified by decoding the real terraform-1.16.0-produced state); this is the resource's own implementation file, not classification control flow, so it carries no live/derivation_guard_test.go entry. ctyjson.Unmarshal was independently verified (a standalone throwaway program, not assumed) to default a type attribute missing from raw JSON to null, so old choudoufu-written terraform_data state predating this field decodes unaffected - TestManagedDataUpgradeStateMissingStore pins that boundary directly. Re-crossing after the fix: migrate PASSES under terraform 1.16.0 too (2/2), with the correct '3 stamped, 4 recorded, 0 failed, 1 skipped' split restored, and the whole estate clears end to end (exit 0) under both terraform versions - no other stage regressed. Not verified: HashiCorp Terraform versions between 1.15.8 and 1.16.0 (whichever one first introduced 'store') and any version after 1.16.0 that might change the block's shape again; the fix targets the exact field shape read off 1.16.0 and would need re-verification against a materially different future schema." @@ -1219,16 +1247,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1244,19 +1272,21 @@ "drift_reconverge": "S3 alarm's alarm_description tampered, exactly 1 object proposed and applied, reconverged to its configured description", "greenfield": "3 resources from nothing (2 tagged alarms + the untaggable dashboard), both alarm markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on both alarms", "migrate": "2 of 3 stamped (1 skipped, untaggable dashboard), 0 failed; both alarm markers read back via the AWS CLI", + "plan_approval": "one argument edited (cf_requests_spike's alarm_description, a config-owned non-ForceNew argument, gains a \"(reviewed)\" suffix), \"plan -out=approved.tfplan\" wrote a 7918-byte stock-format plan file whose whole change set is one update on module.monitoring.aws_cloudwatch_metric_alarm.cf_requests_spike; the world then moved out of band (S3GetRequestsSpike's alarm_description, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.monitoring.aws_cloudwatch_metric_alarm.s3_requests_spike and the live S3GetRequestsSpike it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - CFRequestsSpike's alarm_description still read as configured through the AWS CLI, rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the description put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and CFRequestsSpike read back with the reviewed description, so the refusal is earned by the drift and not handed out to every plan file. BOTH the plan -out and the apply carry this crossing's own -target set (aws_budgets_budget stays out of the graph - floci still answers UnknownOperationException for AWSBudgetServiceGateway on the pinned image), which is enough because a live-markers apply plans the live system from its OWN arguments rather than replaying the file: no exemption and no Go change was needed for issue #903's -target trap. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, no state file", "test_plan": "no resource change proposed; both alarms' tofu-address unchanged, dashboard body re-derived and matches distribution_id" }, - "duration_s": 93.9, + "duration_s": 90, "stage_seconds": { - "cold_deploy": 18, + "cold_deploy": 5, "day2_count": 15, - "day2_remove": 4, + "day2_remove": 5, "day2_rename": 6, "day2_replace": 5, "drift_reconverge": 4, - "greenfield": 13, + "greenfield": 11, "migrate": 25, + "plan_approval": 9, "test_apply": 2, "test_plan": 2 } @@ -1282,16 +1312,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "d72960cdc31f4e7d90f61088920379fc29b3999c", - "date": "2026-09-06T05:35:09Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1300,28 +1330,30 @@ "exit_code": 0, "detail": { "cold_deploy": "63 resources from stock terraform; 4 live zones confirmed unmarked", - "day2_count": "choudoufu: scaling DataCite's own real, already-live aws_route53_record.wp-prod-staging count block (count.index + 3 in the name, header point 3) from 10 to 9 destroyed exactly wp-prod-staging[9] (staging12.datacite.org, 0 add, 0 change, 1 destroy; confirmed gone via the AWS CLI, and its local record file correctly tombstoned rather than left claiming a live identity - the #398-guard shape, confirmed by reading the file directly rather than assumed), leaving wp-prod-staging[0] (staging3.datacite.org)'s live TTL and record-store import_id unchanged; scaling back from 9 to 10 created exactly wp-prod-staging[9] again (0 add -> 1 add, 0 change, 0 destroy), TTL=300 and record import_id=ZFNTJ9UTHQDEAEU_staging12.datacite.org_A (identical string to before - Route 53 hands back no system id for a record set, so realness of the destroy was proved by the AWS CLI absence check and the tombstone, not a changed id), while wp-prod-staging[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical 10-instance count block, plan-only against cold_deploy's own state (applying would disturb the live objects migrate/stage 3-5 depend on), shows the identical shape: destroy the highest index only, create it back under the same name, every lower index untouched both times. aws_route53_record carries no tags at all (header point 2), so this type's own 'every surviving instance keeps its identity' is proved through the record store's ZONEID_NAME_TYPE identity string and a direct AWS CLI read, never a tofu-address tag value - the colon-vs-bracket tag-value escaping trap live/MARKERS.md documents does not apply to an untaggable type. BREAK_COUNT=1 confirms the wrong-instance assertion correctly fails to hold.", + "day2_count": "choudoufu: scaling DataCite's own real, already-live aws_route53_record.wp-prod-staging count block (count.index + 3 in the name, header point 3) from 10 to 9 destroyed exactly wp-prod-staging[9] (staging12.datacite.org, 0 add, 0 change, 1 destroy; confirmed gone via the AWS CLI, and its local record file correctly tombstoned rather than left claiming a live identity - the #398-guard shape, confirmed by reading the file directly rather than assumed), leaving wp-prod-staging[0] (staging3.datacite.org)'s live TTL and record-store import_id unchanged; scaling back from 9 to 10 created exactly wp-prod-staging[9] again (0 add -> 1 add, 0 change, 0 destroy), TTL=300 and record import_id=Z7Z25KMX2IAY0SJ_staging12.datacite.org_A (identical string to before - Route 53 hands back no system id for a record set, so realness of the destroy was proved by the AWS CLI absence check and the tombstone, not a changed id), while wp-prod-staging[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical 10-instance count block, plan-only against cold_deploy's own state (applying would disturb the live objects migrate/stage 3-5 depend on), shows the identical shape: destroy the highest index only, create it back under the same name, every lower index untouched both times. aws_route53_record carries no tags at all (header point 2), so this type's own 'every surviving instance keeps its identity' is proved through the record store's ZONEID_NAME_TYPE identity string and a direct AWS CLI read, never a tofu-address tag value - the colon-vs-bracket tag-value escaping trap live/MARKERS.md documents does not apply to an untaggable type. BREAK_COUNT=1 confirms the wrong-instance assertion correctly fails to hold.", "day2_remove": "choudoufu: deleting aws_route53_zone.eu and aws_route53_record.eu-ns's blocks - both destroys proposed (matching stock's own oracle exactly) and applied cleanly (Apply complete! Resources: 0 added, 0 changed, 2 destroyed.), the zone genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report); the next plan is empty. The parent-scoped removal sweep gap this estate named (gauntlet:parent-scoped-sweep) is closed: recordOrphanReadSweep composes aws_route53_record's identity from its migrate-seeded record correctly (composeImportIDFromComponents's OmitIfAbsent fix) and carries a destroy-before-parent ordering hint (identity.Resolution.DestroyDependsOn) so the record's own destroy is never raced against its zone's force_destroy cascade.", "day2_rename": "moved block: aws_route53_zone.production renamed with zero churn (0 add, 1 change, 0 destroy) - only the zone's own marker rewritten, none of its 45 record children moved; live-mv: aws_route53_zone.internal renamed with zero churn, marker rewritten in place; stock oracle over the same two-zone rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live zone ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing aws_route53_record.status's ForceNew name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (status.datacite.org./CNAME) is confirmed gone and the new object (ZFNTJ9UTHQDEAEU_status2.datacite.org_CNAME) exists, both via the AWS CLI; the local record store's record at the same address now names the new object's identity, not the destroyed one (ZFNTJ9UTHQDEAEU_status.datacite.org_CNAME -> ZFNTJ9UTHQDEAEU_status2.datacite.org_CNAME); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); F-ORACLE also confirms the four apex NS records this estate's DELTA 5 manages can never take this same path (Route 53 refuses to delete the NS/SOA record at a zone's apex), which is why status was chosen instead; BREAK=replace confirms a manufactured identity collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", + "day2_replace": "choudoufu: changing aws_route53_record.status's ForceNew name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (status.datacite.org./CNAME) is confirmed gone and the new object (Z7Z25KMX2IAY0SJ_status2.datacite.org_CNAME) exists, both via the AWS CLI; the local record store's record at the same address now names the new object's identity, not the destroyed one (Z7Z25KMX2IAY0SJ_status.datacite.org_CNAME -> Z7Z25KMX2IAY0SJ_status2.datacite.org_CNAME); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); F-ORACLE also confirms the four apex NS records this estate's DELTA 5 manages can never take this same path (Route 53 refuses to delete the NS/SOA record at a zone's apex), which is why status was chosen instead; BREAK=replace confirms a manufactured identity collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one untaggable record drifted, exactly aws_route53_record.wp-prod-staging[0]/ttl proposed and applied, reconverged to 300, marker intact", "greenfield": "63 resources from nothing (4 tagged zones + 59 untaggable records), the production zone's marker verified via the AWS CLI, 63 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches on zone count (4) and total record-set count (63)", "migrate": "4 of 63 stamped, 59 skipped as untaggable, 0 failed; 59 identity records written (#364), 14 of them also carrying residue (#341), DataCite's own tags survived", + "plan_approval": "one argument edited (aws_route53_zone.production's tags gain Reviewed=yes - a tag, so in-place, moving no live id), \"plan -out=approved.tfplan\" wrote a 19819-byte stock-format plan file whose own totals are \"Plan: 0 to add, 1 to change, 0 to destroy\" and whose whole change set is that one update; the world then moved out of band (staging3.datacite.org's TTL 300->77 in zone Z7Z25KMX2IAY0SJ, this estate's own STAGE 5 upsert lifted, through the AWS CLI and never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming the extra row as \" aws_route53_record.wp-prod-staging[0] Update Z7Z25KMX2IAY0SJ_staging3.datacite.org_A\" - both aws_route53_record.wp-prod-staging[0] and the live composed identity Z7Z25KMX2IAY0SJ_staging3.datacite.org_A it was computed against, an UNTAGGABLE record whose identity comes from the local record store rather than a marker tag - with \"Exit status 3\" spelled out for a pipeline; nothing was applied - zone Z7Z25KMX2IAY0SJ still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the TTL put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the zone read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (tag gone, tofu-address marker intact, 63 record sets still there, next plan proposes no resource action) so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 4 zones / 63 record sets unchanged, all 4 markers unmoved, all 59 identity records intact (14 residue-bearing)", "test_plan": "plan empty across 63 instances, no state file; 14 record sets and 4 zones filled residue from the store" }, - "duration_s": 681.4, + "duration_s": 614.6, "stage_seconds": { - "cold_deploy": 150, - "day2_count": 69, - "day2_remove": 32, - "day2_rename": 26, - "day2_replace": 55, - "drift_reconverge": 29, - "greenfield": 248, - "migrate": 55, - "test_apply": 10, - "test_plan": 7 + "cold_deploy": 115, + "day2_count": 57, + "day2_remove": 30, + "day2_rename": 20, + "day2_replace": 46, + "drift_reconverge": 28, + "greenfield": 234, + "migrate": 43, + "plan_approval": 28, + "test_apply": 8, + "test_plan": 5 } }, "notes": "DataCite's own global DNS root module, the largest offline-clean estate (63 instances) that had never touched a cloud. 2 of 5 - and the two stages it does not reach are blocked on one real, general, previously-unrecorded choudoufu defect, not on anything specific to this estate. Still offline-clean when crossing started (refusal-probe -schemas: blocked 0 sites 0 instances 63 at c41279989a); the schema-less mode disagrees (blocked 1 sites 2, both render correctly in the real run - HANDOFF's asymmetry caveat firing on a live estate). Two of team-members-access's four deltas recur (#268 mandatory backend edit, in cloud{} form; #269 provider version skew, ~> 5 -> 5.100.0 with no list resources, all four zones ServerAssigned); the other two do not (one data source answered by an out-of-band VPC; an ordinary emulator override). Two NEW walls: (1) estate-owned, not choudoufu's/floci's - the four *-ns blocks manage each zone's own apex NS set, which Route 53 creates itself, so a from-scratch apply dies with InvalidChangeBatch; fixed with allow_overwrite=true, the same argument the estate's own author already writes on wp-prod-staging. (2) Filed as #341: stage 3's entire plan is 0/14/0, every diff line +allow_overwrite=true (10 of 14 on wp-prod-staging[0..9], carried in DataCite's own text with no deltas needed) - #275's residue mechanism populates and reads back the record store for TAGGABLE resources only (4 zones get 'filled 1 residue attribute(s)'), but none of the 59 untaggable record sets do, because internal/live/liveimport/ratify.go's !taggable() branch returns before the ReadResource that builds the *eligible object residue needs, and Approve's recordResidueFor sits past the continue that skips a resource with no *eligible - one carrier serving two unrelated jobs, and untaggability should only disqualify one of them (the tag write, not the residue read). 342 of 1025 admitted types are untaggable and share this exclusion. Not fixed here - out of scope for a crossing pass. What the run DOES prove: all 63 rendered identities correct and distinct by value against the AWS CLI's own answer, including two same-named datacite.org zones (public/private) that did not swap and ten wp-prod-staging[0..9] instances rendering staging3..staging12 individually. The 59 untaggable record sets are 94% of the estate - the widest derived-from-tagged fan-out in either lane. Script exits 0 only on reaching exactly this blocker (asserting the changed-address set, that allow_overwrite is the only attribute in the whole diff, and the 0/14/0 totals line) and non-zero on anything else including an empty plan, which is the signal to promote this entry to five stages once #341 lands. BREAK=1 verified red at exactly the identity assertion (a swapped zone id). Also filed lex00/floci#81 (floci accepts a record set whose name is outside its hosted zone, a real bug in the estate's own text; blocks nothing here but is the shape where a crossing passes on the emulator and fails on real AWS). Suggested a new 'published-deployment' lane, distinct from terraform-popular/opentofu-native/reference, since this is neither a module example nor an OpenTofu-native project but a company's own live TFC-connected infrastructure. Merged d5e592d67c (crossing e74b6e5c01); just demo-corpus-mastino-dns, port 4731. just ci green (exit 0, read from a file). Follow-up pass 2026-08-20 (#341 fixed and merged, c73a6e4617/78c92ad64a): FIVE OF FIVE, real. Fix is a third carrier (residuable, which eligible now embeds) built for any admitted-untaggable instance with a record_store declared, deliberately with NO ReadResource at ratify time - a residue attribute is by definition one no read returns, so state already has everything the fix needs, and skipping the read means an untaggable instance can never come back MISSING/DRIFTED (a concern the issue itself raised). Verdict stays StatusUntaggable/OutcomeSkipped; only the marker write was ever skippable, not the residue write. Corrected denominator along the way: DefaultTable holds 1040 rows, not 1025 - survey-full.json calls 683 taggable/342 untaggable and doesn't cover 15 at all. Real re-crossing: migrate reports 4 stamped/59 UNTAGGABLE/14 residue records (all 4 zones plus 10 of the 59 record sets, matching the bug's own signature exactly); test_plan EMPTY, all 63 identities asserted by value; test_apply a genuine no-op with all 14 residue records and 4 markers unchanged; drift_reconverge drifts wp-prod-staging[0]'s TTL (untaggable AND residue-carrying) and reconciles exactly that instance, BREAK_STAGE5=1 verified failing. One honest caveat: stages 4/5 had to be WRITTEN (the prior entry's header claimed they existed in git history but e74b6e5c01 is the file's only commit), and the verifying run itself executed in two calls after a SIGTERM mid-init, not one continuous process - same container/workdir throughout, but the committed script has not been run start-to-finish in a single invocation. A real, separate, more urgent finding surfaced while verifying: #340's own change to live-import's summary line (\"%d newly recorded, %d already recorded\" inserted mid-line) broke the exact-string assertion in 19 OTHER crossing scripts on main - invisible to just ci since e2e scripts aren't in that tier. Filed as #342 with the full list and the one-line fix each needs." @@ -1346,16 +1378,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "a7ca11f9351c0334115a6ca4b3b5de025d70f947", - "date": "2026-09-06T23:49:04Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1364,28 +1396,30 @@ "exit_code": 0, "detail": { "cold_deploy": "26 resources, genuinely cold, genuinely unmarked", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-0f150fc86696abb33) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) unchanged, both read back through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW GroupId (sg-b5900fe02eec8c173 -> sg-16652b643494fb26b) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; every absence check reads length(SecurityGroups) through a group-id FILTER, because describe-security-groups --group-ids on a deleted id returns an empty list with exit 0 on this emulator pin; the next stateless live-plan is empty. Stock oracle (G0): the identical 2-instance block stood up with plain tofu in its own working directory against the same idle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new GroupId (1 add, 0 change, 0 destroy), the lower index's GroupId unchanged both times - then torn down (3 destroyed) before the choudoufu side ran. SYNTHETIC BLOCK, and why: every count this module declares is a boolean create toggle (create_vpc x7, create_s3_bucket x4, create_cloudfront_distribution x2, three launch_template variants gated on existing_id == null), never a scalable set, so scaling one is the day2_remove shape this script already runs, not a shape with a survivor; and its one real for_each (aws_batch_job_definition.tiles over toset(var.themes)) is scoped by this crossing's own root config to a single theme from stage 1 onward, so it has nothing to scale down to and widening it would move every earlier stage's counted assertions (26 resources, 16 stamped, 16 tagged objects, day2_rename's own 16-address list). What shrinking that set would plan is not claimed: it was never measured, because it was never a usable option. Sanctioned fallback per live/GAUNTLET.md #8, precedent reference-ec2-vpc Part F and corpus-iam-policy Part G. It reuses a type this estate already exercises (aws_security_group.batch), sits at a root address nothing else names, and runs entirely after day2_remove, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly reports fail.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-f37ccd0807772ad2c) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) unchanged, both read back through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW GroupId (sg-a444bab32fd697268 -> sg-46bcdd1e7ba3d595c) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; every absence check reads length(SecurityGroups) through a group-id FILTER, because describe-security-groups --group-ids on a deleted id returns an empty list with exit 0 on this emulator pin; the next stateless live-plan is empty. Stock oracle (G0): the identical 2-instance block stood up with plain tofu in its own working directory against the same idle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new GroupId (1 add, 0 change, 0 destroy), the lower index's GroupId unchanged both times - then torn down (3 destroyed) before the choudoufu side ran. SYNTHETIC BLOCK, and why: every count this module declares is a boolean create toggle (create_vpc x7, create_s3_bucket x4, create_cloudfront_distribution x2, three launch_template variants gated on existing_id == null), never a scalable set, so scaling one is the day2_remove shape this script already runs, not a shape with a survivor; and its one real for_each (aws_batch_job_definition.tiles over toset(var.themes)) is scoped by this crossing's own root config to a single theme from stage 1 onward, so it has nothing to scale down to and widening it would move every earlier stage's counted assertions (26 resources, 16 stamped, 16 tagged objects, day2_rename's own 16-address list). What shrinking that set would plan is not claimed: it was never measured, because it was never a usable option. Sanctioned fallback per live/GAUNTLET.md #8, precedent reference-ec2-vpc Part F and corpus-iam-policy Part G. It reuses a type this estate already exercises (aws_security_group.batch), sits at a root address nothing else names, and runs entirely after day2_remove, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly reports fail.", "day2_remove": "choudoufu: create_cloudfront_distribution=false proposed exactly two destroys plus one in-place update (0 add, 1 change, 2 destroy: the distribution, its untaggable OAC, and the bucket policy's own CloudFrontOAC statement dropping), applied cleanly (0 added, 1 changed, 2 destroyed), the distribution is genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state also proposes exactly the same two destroys plus the same bucket-policy update", "day2_rename": "moved block: module.overture_tiles renamed to module.overture_tiles_moved via ONE module-level moved block, 0 add/0 destroy, 16 real tag-marker rewrites (plan showed 18 - two untaggable siblings' policy JSON transiently 'known after apply', resolving to no real change at apply time, confirmed via the stage-5 marker/propagation filter and by value); live-mv: module.overture_tiles_moved renamed to module.overture_tiles_final across 14 of 16 taggable children, one call each, zero churn - the other 2 (aws_batch_compute_environment.tiles and aws_iam_instance_profile.ecs, both server-/provider-assigned identities with no List support in the provider) correctly refused by live-mv and renamed via their own moved blocks instead, applied cleanly; the nine untaggable/config-derived children and the UNTAGGABLE OAC (no longer UNADMITTED_TYPE - #249 narrowed) did not move at all; stock oracle over the identical module rename on cold_deploy's own state also shows zero churn via its own single module-level moved block, covering every one of the 26 children including the two live-mv cannot", "day2_replace": "choudoufu: supplying module.overture_tiles_final's name_overrides.cloudwatch_log_group proposed exactly one replace at the same declared address (the log group; -/+ destroy and then create) cascading into two expected in-place updates (the execution role's inline log policy, the job definition) and nothing else; applied cleanly; the old object (/aws/batch/overture-tiles-crossing) is confirmed gone and the new object (/aws/batch/overture-tiles-crossing-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (/aws/batch/overture-tiles-crossing -> /aws/batch/overture-tiles-crossing-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address plus the same cascade (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one object tampered (VPC Name tag), exactly module.overture_tiles.aws_vpc.batch[0] proposed by both choudoufu and stock with the identical change, apply changed 1 and the Name tag reads back as configured", "greenfield": "26 resources from nothing, bucket and batch job queue markers verified via the AWS CLI, 26 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (batch job queue state, CloudFront distribution comment, bucket count)", "migrate": "16 of 26 stamped, 0 failed; the other 10 correctly UNTAGGABLE or already-ruled UNADMITTED_TYPE", + "plan_approval": "one argument edited (module.overture_tiles's cors_allowed_origins, [\"*\"] -> [\"https://tiles.example.invalid\"], which reaches module.overture_tiles.aws_s3_bucket_cors_configuration.tiles[0] and nothing else - it is the only root knob of this 26-instance estate that lands on exactly one instance IN PLACE, since `tags` reaches all 16 taggable children at once, name_overrides and the launch template are ForceNew, and the create_* toggles are creates and destroys), \"plan -out=approved.tfplan\" wrote a 34023-byte stock-format plan file whose whole change set is that one in-place update; the world then moved out of band (vpc-9038e1c6's Name tag -> moved-after-approval, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming the extra row as \" module.overture_tiles.aws_vpc.batch[0] Update vpc-9038e1c6\" - both module.overture_tiles.aws_vpc.batch[0] and the live vpc-9038e1c6 it was computed against - with \"Exit status 3\" spelled out for a pipeline; nothing was applied - overture-tiles-crossing-tiles's CORS rule still read \"*\" through s3api get-bucket-cors rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Name tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the CORS rule read back as https://tiles.example.invalid, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (CORS back to \"*\", VPC Name tag still overture-tiles-crossing-vpc, next plan proposes no resource action, no state file left behind) so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); 16 tagged objects before and after (resourcegroupstaggingapi's cross-service search alone, floci#98 fixed); S3 bucket and OAC identities unchanged; record store intact", "test_plan": "live-plan empty after the STAGE 2d convergence apply; S3 bucket and OAC identities re-checked by value against the AWS CLI" }, - "duration_s": 419.4, + "duration_s": 443, "stage_seconds": { "cold_deploy": 56, - "day2_count": 48, - "day2_remove": 53, + "day2_count": 51, + "day2_remove": 52, "day2_rename": 47, - "day2_replace": 6, - "drift_reconverge": 7, + "day2_replace": 7, + "drift_reconverge": 8, "greenfield": 120, - "migrate": 70, + "migrate": 75, + "plan_approval": 13, "test_apply": 3, - "test_plan": 9 + "test_plan": 10 } }, "notes": "Landed 2026-08-19. Sourced via GitHub code search rather than the awesome-opentofu/Powered-by-OpenTofu lists, which turned out to be pure tooling/adopter lists with no deployable estates. OpenTofu-native evidence is in its CI rather than a genuine .tofu extension (weaker self-description than corpus-hongbomiao, which ships real .tofu files): .github/workflows/ci.yml runs tofu fmt/validate/test/tflint exclusively through opentofu/setup-opentofu - terraform never appears - and its tests use OpenTofu's own mock_provider framework. A real, tagged-release module (v1.0.0->v1.2.0) from a Linux-Foundation-adjacent geospatial project backed by AWS/Meta/Microsoft/TomTom, contributor fixes as recent as 2026-05-21. cold_deploy genuinely passes (26 resources, plain tofu apply, unmodified module). migrate is BLOCKED, not clean: live-import stamps 13 of 26 cleanly, 10 correctly UNTAGGABLE or already-ruled UNADMITTED_TYPE (#249), and 3 AWS Batch resources fail to stamp on a real floci bug - TagResource/UntagResource/ListTagsForResource (POST /v1/tags/{resourceArn}) misroutes to AppSyncController's greedy catch-all since BatchController never registers that path. Filed as lex00/floci#72 with full evidence (checked ~/checkouts/floci first, confirmed same bug on current main, no in-progress fix) - not fixed, per this session's standing instruction. test_plan is BLOCKED, deterministically asserted (Plan: 4 to add, 7 to change, 0 to destroy, every line traced) rather than reached cleanly. Also filed INTENTIUS/choudoufu#322: aws_iam_role_policy (untaggable, ServerAssignedIfAbsent name via name_prefix) escalates a single-address unbound warning into a hard Error: Listed resource with no tags that aborts the ENTIRE live-plan, not just its own address - a real blast-radius concern (one bad site takes down the whole plan) worth prioritizing. Not fixed; worked around in this crossing via the module's own name_overrides input so the script could still assert what it could reach. Stages 4-5 not attempted - both need a genuinely empty first plan, which this estate doesn't reach yet. Confirmed informationally that applying the current non-empty plan fails safely (AWS Batch's own name-uniqueness check refuses the duplicate) rather than silently corrupting anything. Merged to local main as a233497312 (fix itself: 3c183a305a); justfile gained recipe demo-corpus-overture-tiles; live/corpus-manifest.json gained the pin. UPDATE 2026-08-20: lex00/floci#72 FIXED (floci 1d469fff, published; choudoufu re-pinned to sha256:dc246b1e) and migrate now genuinely PASSES - re-run for real against that image, not inferred: 16 of 26 newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 10 skipped, where it was 13 stamped / 3 failed before. The dry run's own counts are unmoved (16 eligible, 11 VERIFIED, 5 DRIFTED, 9 UNTAGGABLE, 1 UNADMITTED_TYPE) - the bug was in the tag WRITE, never in verification. The fix is generic rather than Batch-shaped: floci already had a SharedTagsController dispatching /tags/{arn} to the TagHandler whose serviceKey matches the ARN's own service segment, and AppSync had simply claimed /v1/tags/{arn: .+} for itself; the fix lifts that dispatch into a SharedTagsDispatcher keyed on path prefix AS WELL AS service, adds a SharedTagsV1Controller, and converts AppSync into a TagHandler beside a new BatchTagHandler - so any further service on that path needs only a handler. Stage 2c now asserts three markers through the AWS CLI instead of one, including the Batch job queue's own tofu-address/tofu-estate via batch list-tags-for-resource (the very call that used to be answered by AppSync) and that the module's own create-time Project tag SURVIVED the stamp - a TagResource that replaces instead of merging is how a live object silently loses its markers. INTENTIUS/choudoufu#322 item 1 is also fixed (576990a599/b75e46c24e) and is no longer the test_plan wall. test_plan stays 'fail' but the wall MOVED and is now harder: it is no longer a non-empty plan, it is live-plan refusing to plan at all (exit 1, no plan produced), at exactly two diagnostics - 'Invalid Identity Attribute Value: Identity attribute \"arn\" contains an Account ID \"000000000000\" which does not match the provider's \"\"' followed by its consequence 'Cannot import for projection'. Filed as INTENTIUS/choudoufu#345 with full evidence. Reachable only BECAUSE the Batch resources are now stamped: projection imports one by its ARN identity and hashicorp/aws validates an identity ARN's account segment against the account the provider knows about itself, which skip_requesting_account_id = true (what every crossing script sets to reach a local emulator) leaves empty. The marker is not wrong - stage 3 re-reads the job queue's real ARN from floci through the AWS CLI and asserts the refusal names that exact string. MEASURED, NOT ASSUMED, and recorded in the script's header so it is not re-tried: setting skip_requesting_account_id = false on the estate copy alone clears this error and breaks stage 2 instead, because the provider then routes S3 bucket tag reads through S3 Control's account-prefixed virtual host (dial tcp: lookup 000000000000.127.0.0.1: no such host), taking aws_s3_bucket.tiles[0] from VERIFIED to MISSING and the estate to 15 of 26 eligible. Stages 4-5 still not attempted: they need a plan and stage 3 produces none. The script's stage 3 is rewritten to assert the new wall deterministically (nonzero exit, exactly 2 errors and no more, both texts, the ARN in them) - and a ZERO exit now fails it, so the day this is fixed the script says so instead of quietly passing. BREAK=1 (stage 2's bucket-marker control) was NOT re-run this pass; its mechanism is unchanged. UPDATE 2026-08-20 (INTENTIUS/choudoufu#345 FIXED, no floci change): the identity-ARN crash is gone. The obvious fix, skip_requesting_account_id = false on the estate copy, was re-measured for real and its earlier 'breaks stage 2 via S3 Control's account-prefixed virtual host, dial tcp: lookup 000000000000.127.0.0.1: no such host' failure is a DNS failure, not an HTTP one - confirmed by curling the same account-prefixed host directly, which fails identically before any TCP connection, so no floci server-side Host-header routing could ever have fixed it (the request never arrives). The real fix is ENDPOINT: floci already publishes localhost.floci.io as a real, public wildcard DNS domain (EmbeddedDnsServer.DEFAULT_SUFFIX, same mechanism as LocalStack's localhost.localstack.cloud) that resolves an account-ID-prefixed label to 127.0.0.1 with no floci container running at all - confirmed via dig and via curl reaching floci's S3ControlController correctly (path-based dispatch, unaffected by the account-prefixed Host). Verified against the CURRENT, unmodified floci image (be3f7ffd, sha256:8a882bcc - no re-pin, no floci commit, no floci PR). Real re-run: stage 1 PASS unchanged (26 resources). Stage 2 PASS, counts moved by exactly one resource as a direct, expected consequence (11 VERIFIED/5 DRIFTED -> 10 VERIFIED/6 DRIFTED: aws_launch_template.batch[0]'s arn now differs between the PLAIN state, written under skip_requesting_account_id = true and so account-less, and the ESTATE copy's live re-read, which now knows its account - a real difference between two provider configurations, not a wrong marker). Stage 3 (test_plan) still recorded fail by this repo's own convention (a first plan must be empty to pass) but the #345 wall itself - live-plan exiting 1 with two diagnostics and no plan at all - is gone: live-plan now exits 0 with 'Plan: 1 to add, 7 to change, 0 to destroy.', asserted deterministically address-by-address. Every line traces to an already-tracked or by-design cause, none of them new: the 1 add is the already-ruled #249 aws_cloudfront_origin_access_control UNADMITTED_TYPE gap; 6 of the 7 changes are internal/live/discovery/count.go's own documented one-time tofu-slot migration-visibility tag (bindCountByAddress's doc comment: 'visible in the plan as a tofu-slot tag being added to each member' - by design, cements on first apply), on every count-toggled ([0]) resource this module declares; the 7th, aws_s3_bucket_policy.tiles[0], is a content diff cascading from the new OAC's arn being 'known after apply' in the same plan. Verified informationally (not scored, per this repo's own convention that test_apply is scored only once test_plan is itself empty): applying the stage 3 plan succeeds (Apply complete! Resources: 1 added, 6 changed, 0 destroyed - matching the plan), and a second live-plan afterward is genuinely empty ('No changes. Your infrastructure matches the configuration.') - the estate converges in exactly one apply, confirming #345's own header claim. No Go code touched; the fix is entirely live/e2e/corpus-overture-tiles/run.sh (ENDPOINT changed from a bare IP to localhost.floci.io, skip_requesting_account_id parameterized so only the estate copy sets it false, cold deploy left untouched)." @@ -1410,16 +1444,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1428,28 +1462,30 @@ "exit_code": 0, "detail": { "cold_deploy": "39 resources, once for real", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-8a54c04755df9d5f0) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read back through the AWS CLI; the destroyed instance is confirmed gone two independent ways (0 by group-id - length(SecurityGroups), never an exit code, since this image answers a deleted id with an EMPTY LIST and exit 0 - and 0 by tofu-address tag filter). Scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-0f8f6860554cccc91 -> sg-36ce4ada55a333b89), carrying tofu-address=aws_security_group.count_test:1 with exactly one claimant, while count_test[0] kept the same id and the same marker throughout; the next plan is empty. The G0 stock oracle stood the identical 2-instance block up with plain terraform in its own working directory and its own VPC against the same idle endpoint and showed the identical shape: destroy the higher index only (Plan: 0 to add, 0 to change, 1 to destroy), create the higher index back under a new GroupId (sg-d7b20850fbae6692e -> sg-c4094d1adb1e37439), the lower index's GroupId (sg-75e308f768ff7da71) unchanged both times; torn down completely before the choudoufu side ran. SYNTHETIC BLOCK, and why: terraform-aws-rds declares no scalable knob anywhere - the root example has no count and no for_each at all, and every count in the vendored module is a boolean create toggle (var.create ? 1 : 0 and friends, grepped not guessed) - while the only genuinely scalable knobs belong to the supporting module.vpc, whose subnet families each destroy an untaggable aws_route_table_association sibling alongside the taggable aws_subnet, which is issue #410's family (corpus-overture-tiles's count-shrink extension) and would make this stage a re-measurement of that gap rather than of count scaling. So the sanctioned fallback per live/GAUNTLET.md #8, reference-ec2-vpc's Part F and corpus-iam-policy's Part G: a self-contained aws_security_group.count_test at an address nothing else names, appended after day2_remove's completed removal, of a taggable type this estate already exercises, introducing nothing that generates a secret. BREAK_COUNT=1 is the negative control and was run for real: it asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, proving the index assertion is load-bearing.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-f03da6a2c670f916f) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read back through the AWS CLI; the destroyed instance is confirmed gone two independent ways (0 by group-id - length(SecurityGroups), never an exit code, since this image answers a deleted id with an EMPTY LIST and exit 0 - and 0 by tofu-address tag filter). Scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-1247895febf8a3b8b -> sg-2e4b68f2a30525035), carrying tofu-address=aws_security_group.count_test:1 with exactly one claimant, while count_test[0] kept the same id and the same marker throughout; the next plan is empty. The G0 stock oracle stood the identical 2-instance block up with plain terraform in its own working directory and its own VPC against the same idle endpoint and showed the identical shape: destroy the higher index only (Plan: 0 to add, 0 to change, 1 to destroy), create the higher index back under a new GroupId (sg-697f57f8128edc558 -> sg-9801c05dba1e82f94), the lower index's GroupId (sg-5f9fbe3d4127b1cb7) unchanged both times; torn down completely before the choudoufu side ran. SYNTHETIC BLOCK, and why: terraform-aws-rds declares no scalable knob anywhere - the root example has no count and no for_each at all, and every count in the vendored module is a boolean create toggle (var.create ? 1 : 0 and friends, grepped not guessed) - while the only genuinely scalable knobs belong to the supporting module.vpc, whose subnet families each destroy an untaggable aws_route_table_association sibling alongside the taggable aws_subnet, which is issue #410's family (corpus-overture-tiles's count-shrink extension) and would make this stage a re-measurement of that gap rather than of count scaling. So the sanctioned fallback per live/GAUNTLET.md #8, reference-ec2-vpc's Part F and corpus-iam-policy's Part G: a self-contained aws_security_group.count_test at an address nothing else names, appended after day2_remove's completed removal, of a taggable type this estate already exercises, introducing nothing that generates a secret. BREAK_COUNT=1 is the negative control and was run for real: it asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, proving the index assertion is load-bearing.", "day2_remove": "choudoufu: deleting module.db_default_renamed's block proposed exactly two destroys (the db instance and its own local random_id.snapshot_identifier, no cloud representation - issue #340), applied cleanly, the db instance is genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on the same renamed oracle tree also proposes exactly the same two destroys; the target was chosen (see header) because its own nested module.db_instance call has no untaggable AWS-side sibling under this estate's create_db_option_group=false/create_db_parameter_group=false, unlike the shapes that surfaced issue #410 for corpus-s3-bucket-complete and corpus-overture-tiles", "day2_rename": "moved block: module.security_group renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.db_default's db instance renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", "day2_replace": "choudoufu: changing module.db's ForceNew db_name argument (plus identifier, for an observable identity change) proposed exactly one instance replace at the same declared address, cascading into its 2 cloudwatch log groups and db parameter group (all replaced, all named from identifier) - 4 to add, 4 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql) is confirmed gone and the new instance (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql-replaced) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new identifier, not the destroyed one (complete-postgresql -> complete-postgresql-replaced); the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", "drift_reconverge": "one object tampered (primary DB instance's Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag", "greenfield": "39 resources from nothing (same DELTA reduction cold_deploy itself needs - two emulator gaps, floci-io/floci#51 and lex00/floci#52), primary DB instance and security group markers verified via the AWS CLI, 39 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (DB engine/version/class/storage/port, security-group rule count)", "migrate": "26 of 39 stamped", + "plan_approval": "one argument edited (module \"db_default\"'s tags gain Reviewed=yes - in-place, so no live id moves, and reaching exactly ONE live object because that module call runs with create_db_option_group=false and create_db_parameter_group=false, where the same edit on module.db would also reach its parameter group, option group and log groups), \"plan -out=approved.tfplan\" wrote a 134369-byte stock-format plan file whose whole change set is one update on module.db_default.module.db_instance.aws_db_instance.this[0]; the world then moved out of band (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql's Example tag, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming the extra row as \" module.db.module.db_instance.aws_db_instance.this[0] Update db-52C3D597C3FC4A7BAE689183\" - both module.db.module.db_instance.aws_db_instance.this[0] and the live identity it was computed against, which for an aws_db_instance is RDS's own server-minted DbiResourceId db-52C3D597C3FC4A7BAE689183 rather than the ARN or the client-chosen identifier, read off arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql through the AWS CLI and compared by value - with \"Exit status 3\" spelled out for a pipeline; nothing was applied - arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql-default-7e26b5c33311bceebccb071915 still carried no Reviewed tag, read back through rds list-tags-for-resource rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Example tag put back to its pre-tamper value and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql-default-7e26b5c33311bceebccb071915 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (Reviewed gone, the primary instance's Example tag still \"complete-postgresql\", next plan proposes no resource action) so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 26 objects before, 26 after, no state file, primary DB instance marker unmoved", "test_plan": "genuinely empty replan (No changes. Your infrastructure matches the configuration.) with no local state file. lex00/floci#120's round-trip gap, this estate's last recorded wall, is CONFIRMED FIXED: round 8 (PR #128/ff815779, ghcr.io/lex00/floci:main-20260824d sha256:25fc9687, #124's RDS colliding-port isolation) closed the last of its eight fields for this estate - module.db_default's own port (module.db and module.db_default both declare port=5432, a genuine collision; module.db_default is the second-created instance and gets its own distinct loopback bind address with the declared port honored). The other seven fields (backup_window, monitoring_interval, monitoring_role_arn, performance_insights_retention_period, engine_lifecycle_support, enabled_cloudwatch_logs_exports, max_allocated_storage) and the parameter block's apply_method were already fixed by earlier rounds (round 5 and round 6's own #120 passes) that this estate had not been re-crossed since - the artifact's recorded '3 in-place updates' detail was stale before this round's own fix even landed. Confirmed three independent ways, not merely inferred from the empty plan: a direct describe-db-parameters --source user probe of the live parameter group (autovacuum=1, client_encoding=utf8, matching config exactly, no tofu in the loop), a direct describe-db-instances probe of the second instance's own Endpoint.Port (5432, the declared port), and all eight attribute names individually confirmed absent from choudoufu's plan. INTENTIUS/choudoufu#393 (skip_final_snapshot's phantom true->false update) remains fixed, confirmed absent. Stock's own replan against its own never-deleted state file still shows tag noise plus the two parameter blocks; ruled out as a live discrepancy by the same direct API probe (informational only, not this stage's oracle - HANDOFF row 3, a property of that one state file's own apply-time fidelity)." }, - "duration_s": 711.2, + "duration_s": 730, "stage_seconds": { - "cold_deploy": 109, + "cold_deploy": 104, "day2_count": 50, "day2_remove": 91, - "day2_rename": 16, - "day2_replace": 171, + "day2_rename": 17, + "day2_replace": 170, "drift_reconverge": 8, "greenfield": 203, - "migrate": 50, - "test_apply": 4, - "test_plan": 8 + "migrate": 52, + "plan_approval": 20, + "test_apply": 5, + "test_plan": 9 } }, "notes": "RE-CROSSED FOR REAL 2026-08-21 at dbc5ebf575 (GitHub issue #346's fix), twice, the second time with an instrumented copy of the script that dumps live-plan's raw output so the diagnostics could be quoted rather than inferred. STAGES UNCHANGED at 2 of 5, and #346's fix DOES NOT REACH THIS ESTATE - which refutes the 'four estates share one wall' framing #346 was filed under. Stage 1: 'Apply complete! Resources: 39 added, 0 changed, 0 destroyed'. Stage 2: '26 of 39 eligible (20 VERIFIED + 6 DRIFTED); 13 skipped (UNTAGGABLE by provider schema)'; '26 stamped, 1 recorded (random_id.snapshot_identifier), 0 failed, 12 skipped'; the primary aws_db_instance's tofu-address/tofu-estate asserted by exact value through the AWS CLI. Stage 3 fails on exactly the same 2 diagnostics as before, same lines, same text: 'Module output not supported in static context' on main.tf:198 (cidr_blocks = module.vpc.vpc_cidr_block, inside the module CALL argument ingress_with_cidr_blocks) and 'Unable to compute static value' on the security-group module's own main.tf:197 (cidr_blocks = compact(split(',', lookup(var.ingress_with_cidr_blocks[count.index], 'cidr_blocks', join(',', var.ingress_cidr_blocks))))). Why the fix misses it, measured rather than guessed: this estate's shape has no each.value anywhere. It is COUNT-indexed, and the module output is consumed as a module-CALL argument, which travels tolerantVariables/rebuildConstructor/moduleOutputValue - a VALUE-shaped route that cannot carry a deferred parent read, because a ParentRef is not a cty.Value. #346's fix is part-shaped and lives on the identity-argument route. A separate mechanism is needed and is filed separately. Also corrected: this file previously recorded run.sh's stage-3 assertions as stale (WANT_CIDX_N=7). They are not, and were not at this commit - the committed script already asserts WANT_CIDX_N=0, WANT_DEFAULT_N=0, WANT_UNRESOLVABLE_N=0, WANT_MODOUT_N=1 and WANT_CASCADE_N=1, which is exactly what the run produced, so the script exits 0 while the estate stays blocked. PRIOR HISTORY BELOW. Landed c239792018/47778a931a (2026-08-18); migrate's fail->pass flip was re-verified 2026-08-18 against current main (cec3c4b9b1) but the committed run.sh still asserted the stale pre-fix '0 eligible' shape as a passing control rather than a real check. Follow-up pass 2026-08-18 (this entry) rewrote run.sh's own assertions to the real, current numbers, re-verified for real in a fresh isolated worktree rather than trusted from the prior note: stage 2 dry run reports '23 of 39 resource instance(s) are eligible for stamping (VERIFIED or DRIFTED)' (18 VERIFIED + 5 DRIFTED), -approve reports '23 resource(s) newly stamped, 0 already stamped, 0 failed, 16 skipped', and the primary aws_db_instance's tofu-address/tofu-estate tags are asserted by exact value straight through the AWS CLI (module.db.module.db_instance.aws_db_instance.this:0 / rds-complete-postgres). The 16 skipped: 13 untaggable by design (aws_route_table_association x9, aws_route, aws_security_group_rule, aws_iam_role_policy_attachment, random_id - no tags argument in the provider schema) and 3 are #305's still-open unadmitted-type gap. test_plan is now asserted against a real live-plan on the really-migrated estate (state file deleted first) with a BREAK=1 negative control, and the real counts differ from what the prior note here claimed: exactly 7 count-index-in-tag sites (#304, all aws_security_group_rule.ingress_with_cidr_blocks) and exactly 3 unadmitted-type sites (#305, the three default-object adopters actually created), not 35 and 5. The prior note's 35/5 figures were measured before this estate had ever actually been migrated (nothing tagged, so the 28 module.vpc sibling-indexing sites it also counted never had anything to resolve against) and before bc9ef26638 ('a resource block with a provably-zero count/for_each has no instance to refuse admission on', already on main) landed, which independently stops aws_default_vpc/aws_vpn_gateway_attachment's two count=0 sites from refusing at all - together accounting for the 35->7 and 5->3 drop. Two unrelated real floci gaps found and filed upstream on the fork: lex00/floci#51 (RDS cross-region backup replication), lex00/floci#52 (SecretsManager RotateSecret wrongly requires a Lambda ARN for RDS-managed rotation) - both worked around in the script with documented EMULATOR GAP deltas so stage 1 could stand up at all. Follow-up pass 2026-08-18 (#313 cross-check, fresh worktree, real run): confirmed the 7 count-index-in-tag sites above are NOT #313's wall. This estate's data.aws_availability_zones usage only feeds local.azs = slice(..., 0, 3), a statically-known length, never a for_each/count keyed on the AZ name values themselves, so #313's static-context diagnostic cannot fire here (grepped the full raw plan output: zero occurrences). Commented on #313 (ruling this estate out, with the slice-vs-for_each distinction as evidence) and re-confirmed #304 is still the sole test_plan blocker. Follow-up pass 2026-08-19: #304 fixed and merged (69038634d0/9aaca0ee10) - internal/live/lint/count_index_domain.go's domain check was evaluating a whole module-call-argument value as one pass/fail unit, so one refused reference anywhere inside a list-of-objects argument poisoned every attribute derived from any part of it, even ones that never read the refused field. New StaticEvaluator.EvaluateStructural/EvalContextTolerant validate each reference individually. Re-verified for real: count-index-in-tag sites on this estate 7->0. Estate still does NOT reach test_plan clean, though: 14 sites (from_port/to_port/protocol/cidr_blocks) now fail a DIFFERENT diagnostic, 'Identity not resolvable from configuration' - internal/live/identity/partialargs.go's tolerantVariables deliberately covers only count/for_each key-set resolution, not per-attribute identity-value rendering (its own doc comment records a past regression from broadening it carelessly). Filed as #323, not attempted - needs its own scoped pass. Plus 18 genuinely-ambiguous element(aws_subnet.*[*].id, count.index) sites, confirmed still correctly refusing (unweakened by #304's fix). #304 left open (not closed) since the titled bug is fixed and verified but this estate still doesn't reach a clean plan for the separate #323 reason. Follow-up pass 2026-08-19 (#321 re-verification, scouting only, no commit): #321's fix generalizes strongly to this second, independent estate - 15 of the 18 element() sites now resolve cleanly (every aws_route_table_association.{public,private}.subnet_id/route_table_id). 3 remain: aws_route_table_association.database's route_table_id goes through coalescelist(A[*].id, B[*].id) wrapping the splat, outside resolveElementCall's bare-splat requirement - the same out-of-scope shape #321's own closing comment already flagged, now confirmed reaching a second real module composition. One NEW site found: aws_security_group_rule.ingress_with_cidr_blocks[0].security_group_id via local.this_sg_id = concat(A.*.id, B.*.id, [\"\"])[0] - concat()+splat+index through a local value, terraform-aws-modules/security-group's universal accessor, high-leverage since every rule resource that module creates uses it. Both new shapes filed as #324, not attempted. #323's 14 sites confirmed unchanged in count, root cause refined: this estate's trigger is cidr_blocks = module.vpc.vpc_cidr_block (a module-output reference into a resource's config-derived attribute) poisoning the whole variable projection, not the lookup()-into-bundled-table pattern #304 fixed - same tolerantVariables scope boundary, a second concrete trigger. Net: test_plan stays fail, 18 diagnostic sites total (4 unresolvable-identity + 7 module-output-static + 7 compute-static, mapping to #324's 4 sites + #323's 14). Three separately-scoped resolver passes stand between this estate and five-of-five, not one. live/e2e/corpus-rds-complete-postgres/run.sh's stage-3 assertions are now stale (still expect #304's old 7-site count-index picture) and need a real update pass, not done here. Follow-up pass 2026-08-19 (#324 item 2 fixed and merged, 80d3766b3e/79ffbe4732): concat(A[*].id, B[*].id, [literal])[N] through local.this_sg_id now resolves generically (reuses #321's own splat/instance-count machinery, handles both the RelativeTraversalExpr and IndexExpr parse shapes HCL produces for a constant vs non-constant index). Confirmed by real absence from this estate's own stage-3 output. Generalizes hard: refusal-probe over the full 250-entry offline corpus shows 'Identity not resolvable from configuration' 67 -> 42 (-25 sites), zero regressions, 15 offline corpus entries improved (all terraform-aws-rds examples, plus autoscaling/complete, ecs/complete, ecs/ec2-autoscaling, lambda/with-vpc-s3-endpoint - offline corpus entries, not necessarily live-crossed estates; corpus-security-group-complete and eks/examples/* unchanged). Item 1 (coalescelist) explicitly left open, unattempted. This estate itself: fixing the concat site surfaced a SEPARATE, previously-masked cascade - module.vpc.vpc_cidr_block feeding module.security_group's var.ingress_with_cidr_blocks, 'Module output not supported in static context' - likely another instance of #313's own deliberately out-of-scope resource-attribute boundary (the same family as corpus-security-group-complete's remaining 7 sites), not yet formally confirmed as such or filed separately. test_plan stays fail; the estate's own crossing script needs a real staleness-update pass (still asserts #304's old picture) before its true current diagnostic count can be read cleanly. Follow-up pass 2026-08-19: #324 item 1 (coalescelist) also fixed and merged (c25957cbdf/49744a5617) - #324 now fully closed, both items. The exact 3 aws_route_table_association.database sites this issue named are confirmed gone from this estate's real live-plan output. Generalizes narrowly but cleanly beyond this estate: refusal-probe shows -14 sites across exactly 2 offline corpus entries (cross-region-replica-postgres, vpc/examples/issues - the latter matching #321's own predicted 8-site count exactly). test_plan still stays fail here - blocked only by the pre-existing, unrelated module-output cascade already noted (likely #313's family, unconfirmed) and #323's still-open 14 sites. All of #321/#324's derivable element/splat/concat/coalescelist work is now done across this estate; what remains needs #323's own dedicated pass plus resolving the module-output cascade, not further quick derivations. Follow-up pass 2026-08-19: #323 fixed and merged (3d62366625/fb95168e63), closed. tolerantVariables now resolves a static leaf independently of a sibling leaf's genuine unresolvability, instead of the whole variable projection being poisoned by one bad reference - traced to configs.staticScopeData.GetInputVariable's own error bail discarding every known leaf along with the one genuinely unknown one. Real crossing re-verified: stage-3 identity refusals 14 -> 2 (both are the SAME underlying cause counted twice - module-output-not-supported and unable-to-compute-static-value both trace to cidr_blocks = module.vpc.vpc_cidr_block -> aws_vpc.this[0].cidr_block). CONFIRMED: this estate's sole remaining test_plan blocker is #313's root cause B (a resource-attribute reference through a module output), the exact same maintainer-scoped-out boundary blocking corpus-security-group-complete's own last 7 sites - not a bug, not derivable further, a pure scope decision. Generalizes cleanly: refusal-probe -204 sites across 14 offline corpus entries (11 rds examples, 2 ecs examples, autoscaling/complete), zero instances gained/lost anywhere, zero regressions - improves diagnostics, unblocks nothing further by itself (as expected, since the poisoning fix doesn't touch the genuinely-unresolvable leaf). run.sh's stage-3 assertions are now confirmed stale in a new way too: WANT_CIDX_N=7 has actually been 0 since #304 landed, not just uncounted - needs a real update pass reflecting the estate's true current picture (count-index 0, unadmitted-type 0, both remaining diagnostics tracing to the single #313-root-cause-B leaf), deliberately left to the orchestrator's own call rather than the fixing agent's." @@ -1474,16 +1510,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1492,26 +1528,28 @@ "exit_code": 0, "detail": { "cold_deploy": "30 resources added by plain terraform, 4 buckets confirmed live, no tofu-address tag", - "day2_count": "synthetic block (this estate's root configuration declares no count or for_each at all, and its only multi-instance knob - the intelligent_tiering input's two-entry map - fans out onto aws_s3_bucket_intelligent_tiering_configuration, which has no tags argument, issue #410's untaggable-child shape; so aws_s3_bucket.count_test, a taggable type this estate already exercises, at an address no other stage uses). choudoufu: scaling count_test from 2 to 1 proposed exactly one destroy (0 add, 0 change, 1 destroy), and it was the HIGHER index - count_test[1], s3-bucket-complete-count-test-1 - with count_test[0] not appearing in the plan at all; applied cleanly (0 added, 0 changed, 1 destroyed); head-bucket on s3-bucket-complete-count-test-1 then fails outright, while the survivor s3-bucket-complete-count-test-0 keeps both its CreationDate (2026-09-06T01:17:42+00:00) and its tofu-address=aws_s3_bucket.count_test:0 marker, read through the AWS CLI rather than choudoufu's own report, and count_test[1]'s local record is tombstoned (has tombstone, no identity - the #398-guard shape) rather than deleted. Scaling back from 1 to 2 proposed exactly one create (1 add, 0 change, 0 destroy) for count_test[1] alone, and the recreated bucket is genuinely a NEW object: same deterministic name (an S3 bucket's id IS its own name - probed against floci with no tofu in the loop) but a new CreationDate (2026-09-06T01:17:42+00:00 -> 2026-09-06T01:18:18+00:00), re-stamped tofu-address=aws_s3_bucket.count_test:1 (tags do not survive a delete+recreate on this image, so the marker is one the apply wrote again), and a record identity naming the recreated bucket; count_test[0]'s CreationDate and marker were unchanged at every step. The next plan proposes no resource action. G-ORACLE, plain terraform standing the identical 2-instance block up for real in the idle greenfield-oracle account, shows the identical shape: destroy the higher index only (0 add, 0 change, 1 destroy), recreate it under the same deterministic bucket name but a new CreationDate (2026-09-06T01:14:51+00:00 -> 2026-09-06T01:14:59+00:00), index 0's CreationDate (2026-09-06T01:14:51+00:00) unchanged at both steps. BREAK_COUNT=1 asserts the wrong instance (count_test[0]) was destroyed and reports this stage fail, as the stage's Break text requires.", - "day2_remove": "choudoufu: deleting module.simple_bucket_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy: the bucket and its untaggable public_access_block child), applied cleanly (0 added, 0 changed, 2 destroyed), the bucket is genuinely gone from the live account (head-bucket on simple-heroic-terrier now fails, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state also proposes exactly the same two destroys for the same two objects; the target was chosen to avoid issue #404's shape (a sibling policy re-reading the removed bucket's own ARN) - module.log_bucket and module.s3_bucket are both left untouched", + "day2_count": "synthetic block (this estate's root configuration declares no count or for_each at all, and its only multi-instance knob - the intelligent_tiering input's two-entry map - fans out onto aws_s3_bucket_intelligent_tiering_configuration, which has no tags argument, issue #410's untaggable-child shape; so aws_s3_bucket.count_test, a taggable type this estate already exercises, at an address no other stage uses). choudoufu: scaling count_test from 2 to 1 proposed exactly one destroy (0 add, 0 change, 1 destroy), and it was the HIGHER index - count_test[1], s3-bucket-complete-count-test-1 - with count_test[0] not appearing in the plan at all; applied cleanly (0 added, 0 changed, 1 destroyed); head-bucket on s3-bucket-complete-count-test-1 then fails outright, while the survivor s3-bucket-complete-count-test-0 keeps both its CreationDate (2026-09-07T02:46:16+00:00) and its tofu-address=aws_s3_bucket.count_test:0 marker, read through the AWS CLI rather than choudoufu's own report, and count_test[1]'s local record is tombstoned (has tombstone, no identity - the #398-guard shape) rather than deleted. Scaling back from 1 to 2 proposed exactly one create (1 add, 0 change, 0 destroy) for count_test[1] alone, and the recreated bucket is genuinely a NEW object: same deterministic name (an S3 bucket's id IS its own name - probed against floci with no tofu in the loop) but a new CreationDate (2026-09-07T02:46:16+00:00 -> 2026-09-07T02:46:52+00:00), re-stamped tofu-address=aws_s3_bucket.count_test:1 (tags do not survive a delete+recreate on this image, so the marker is one the apply wrote again), and a record identity naming the recreated bucket; count_test[0]'s CreationDate and marker were unchanged at every step. The next plan proposes no resource action. G-ORACLE, plain terraform standing the identical 2-instance block up for real in the idle greenfield-oracle account, shows the identical shape: destroy the higher index only (0 add, 0 change, 1 destroy), recreate it under the same deterministic bucket name but a new CreationDate (2026-09-07T02:42:47+00:00 -> 2026-09-07T02:42:55+00:00), index 0's CreationDate (2026-09-07T02:42:47+00:00) unchanged at both steps. BREAK_COUNT=1 asserts the wrong instance (count_test[0]) was destroyed and reports this stage fail, as the stage's Break text requires.", + "day2_remove": "choudoufu: deleting module.simple_bucket_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy: the bucket and its untaggable public_access_block child), applied cleanly (0 added, 0 changed, 2 destroyed), the bucket is genuinely gone from the live account (head-bucket on simple-smart-raven now fails, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state also proposes exactly the same two destroys for the same two objects; the target was chosen to avoid issue #404's shape (a sibling policy re-reading the removed bucket's own ARN) - module.log_bucket and module.s3_bucket are both left untouched", "day2_rename": "moved block: module.cloudfront_log_bucket renamed to module.cloudfront_log_bucket_renamed with zero churn (0 add, 1 change, 0 destroy), the bucket's tofu-address marker rewritten in place; live-mv: module.simple_bucket renamed to module.simple_bucket_renamed with zero churn, marker rewritten in place; both live bucket names unchanged, read via the AWS CLI; the post-rename plan proposes no resource action", - "day2_replace": "choudoufu: changing module.log_bucket's ForceNew bucket argument proposed exactly one bucket replace at the same declared address, cascading into its ownership_controls, policy and public_access_block (all replaced) plus module.s3_bucket's own logging target_bucket (updated in-place) - 4 to add, 1 to change, 4 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old bucket (logs-heroic-terrier) is confirmed gone and the new bucket (logs-heroic-terrier-replaced) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new bucket, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Live resource displaced from the address it is marked for\", naming the manufactured bucket, proposing nothing for it) rather than silently proposed as nothing - the name-derived-identity shape of this diagnostic, distinct from EC2/SQS's fungible-set \"Two live resources claiming one slot\" because aws_s3_bucket's identity resolves straight from the config's own computed name rather than only through a marker sweep. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones.", + "day2_replace": "choudoufu: changing module.log_bucket's ForceNew bucket argument proposed exactly one bucket replace at the same declared address, cascading into its ownership_controls, policy and public_access_block (all replaced) plus module.s3_bucket's own logging target_bucket (updated in-place) - 4 to add, 1 to change, 4 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old bucket (logs-smart-raven) is confirmed gone and the new bucket (logs-smart-raven-replaced) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new bucket, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Live resource displaced from the address it is marked for\", naming the manufactured bucket, proposing nothing for it) rather than silently proposed as nothing - the name-derived-identity shape of this diagnostic, distinct from EC2/SQS's fungible-set \"Two live resources claiming one slot\" because aws_s3_bucket's identity resolves straight from the config's own computed name rather than only through a marker sweep. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones.", "drift_reconverge": "accelerate config drifted to Enabled, exactly 1 change proposed and applied, reconverged to Suspended, final plan empty", "greenfield": "29 resources from nothing (SCOPE REDUCTION's own reduced count, random_pet pinned to a literal on both sides), 3 of 4 bucket markers verified via the AWS CLI, 26 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on all 4 buckets (versioning, default encryption, policy presence)", "migrate": "6 of 30 stamped, 1 recorded (random_pet, issue #340), 23 skipped (untaggable), 0 failed, 26 identities recorded (#364 unit A2); markers survived the residue-classification apply", + "plan_approval": "one argument edited (module.cloudfront_log_bucket gains tags = { Reviewed = \"yes\" }, reaching aws_s3_bucket.this[0] and nothing else the module creates - that module call sets no attach_*_policy input, so the module's own policy-document data sources are count = 0 for it and one tag is one row), \"plan -out=approved.tfplan\" wrote a 84803-byte stock-format plan file whose whole change set is one update on module.cloudfront_log_bucket.aws_s3_bucket.this[0]; the world then moved out of band (s3-bucket-smart-raven's transfer-acceleration status flipped to Enabled through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live bucket from the one under review) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.s3_bucket.aws_s3_bucket_accelerate_configuration.this[0] and the live s3-bucket-smart-raven it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - get-bucket-tagging on cloudfront-logs-smart-raven still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the accelerate status put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and cloudfront-logs-smart-raven read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); bucket count unchanged at 4", "test_plan": "no resource action proposed; 29 rendered identity occurrences (11 distinct), all naming known roots" }, - "duration_s": 484.5, + "duration_s": 511, "stage_seconds": { - "cold_deploy": 89, + "cold_deploy": 73, "day2_count": 56, - "day2_remove": 19, - "day2_rename": 23, - "day2_replace": 21, + "day2_remove": 20, + "day2_rename": 22, + "day2_replace": 20, "drift_reconverge": 20, - "greenfield": 165, - "migrate": 77, + "greenfield": 169, + "migrate": 81, + "plan_approval": 35, "test_apply": 11, "test_plan": 4 } @@ -1538,16 +1576,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1556,28 +1594,30 @@ "exit_code": 0, "detail": { "cold_deploy": "67 resources (DELTA 2, lex00/floci#57)", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-294fab44394fefbdc) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-e5ba6f0e8c24d339b, was sg-82428bbcae2cf1efa - a security group id is never reused, so the destroy is directly observable rather than inferred) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] kept both its id and its marker throughout; the local record store tracked the same facts by value at every step (index 0's record never moved, index 1's stopped naming the destroyed object and then named the new one); the next plan is empty. G-ORACLE is the stock leg for the identical shape, applied for real with plain terraform in the idle greenfield account: 3 created, then 0 add/0 change/1 destroy hitting count_test[1] only, then 1 add/0 change/0 destroy bringing it back under a new id, count_test[0]'s id unchanged both times - identical to choudoufu's. SYNTHETIC BLOCK, and why: this estate declares no scalable knob of its own - every count in terraform-aws-security-group v6.0.0 is the boolean 'local.create ? 1 : 0' toggle, and its real for_each maps (var.ingress_rules/var.egress_rules) cannot be scaled in isolation because main.tf line 96 feeds every rule id into aws_vpc_security_group_rules_exclusive.this[0], making the plan 0 add/1 change/1 destroy instead of the exact shape this stage asserts; aws_security_group.count_test is self-contained, named by nothing else here, and of a type this estate already exercises six times over (reference-ec2-vpc Part F and corpus-iam-policy Part G are the precedent). BREAK_COUNT=1 runs this stage's Break control (expect count_test[0] to be the destroyed one) and correctly fails to hold.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-0999dad1a4c3c06d5) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-2e2ea3c3128e173a8, was sg-6d6ee0da0cdf72206 - a security group id is never reused, so the destroy is directly observable rather than inferred) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] kept both its id and its marker throughout; the local record store tracked the same facts by value at every step (index 0's record never moved, index 1's stopped naming the destroyed object and then named the new one); the next plan is empty. G-ORACLE is the stock leg for the identical shape, applied for real with plain terraform in the idle greenfield account: 3 created, then 0 add/0 change/1 destroy hitting count_test[1] only, then 1 add/0 change/0 destroy bringing it back under a new id, count_test[0]'s id unchanged both times - identical to choudoufu's. SYNTHETIC BLOCK, and why: this estate declares no scalable knob of its own - every count in terraform-aws-security-group v6.0.0 is the boolean 'local.create ? 1 : 0' toggle, and its real for_each maps (var.ingress_rules/var.egress_rules) cannot be scaled in isolation because main.tf line 96 feeds every rule id into aws_vpc_security_group_rules_exclusive.this[0], making the plan 0 add/1 change/1 destroy instead of the exact shape this stage asserts; aws_security_group.count_test is self-contained, named by nothing else here, and of a type this estate already exercises six times over (reference-ec2-vpc Part F and corpus-iam-policy Part G are the precedent). BREAK_COUNT=1 runs this stage's Break control (expect count_test[0] to be the destroyed one) and correctly fails to hold.", "day2_remove": "choudoufu: deleting module.postgresql_renamed's block proposed exactly 5 destroys (0 add, 0 change, 5 destroy: SG + 2 ingress + 1 egress + 1 untaggable rules_exclusive), applied cleanly (0 added, 0 changed, 5 destroyed), the security group is genuinely gone from the live account (0 matches on describe-security-groups for the old id, read via the AWS CLI, not choudoufu's own report), and the next plan proposes nothing; stock oracle on cold_deploy's own state (D-ORACLE remove) also proposes exactly 5 destroys for the same 5 objects; classifyOrphans did not withhold the untaggable rules_exclusive destroy even though module.security_group's and module.consul's own rules_exclusive instances share its block key, because both surviving instances are bound, not unclaimed", "day2_rename": "moved block: module.postgresql renamed to module.postgresql_renamed with zero churn (0 add, 4 change, 0 destroy) - the rule-children case, its own SG plus ingress/egress rules and rules_exclusive all moving under one moved block; live-mv: aws_security_group.app renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.security_group's ForceNew name argument proposed exactly one SG replace at the same declared address, cascading into its 7 ingress rules, 1 egress rule and 1 rules_exclusive enforcer (all replaced) - 10 to add, 10 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old SG (sg-b281cdf25f7c3e452) is confirmed gone and the new SG (sg-74ab044e1dc941d6e) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new SG, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Indistinguishable instances without per-instance markers\", naming both live security groups) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", + "day2_replace": "choudoufu: changing module.security_group's ForceNew name argument proposed exactly one SG replace at the same declared address, cascading into its 7 ingress rules, 1 egress rule and 1 rules_exclusive enforcer (all replaced) - 10 to add, 10 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old SG (sg-5eb49924036c81c44) is confirmed gone and the new SG (sg-9260d7ad3ccb5142b) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new SG, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Indistinguishable instances without per-instance markers\", naming both live security groups) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one object tampered (DriftProbe tag on the main security group), exactly module.security_group.aws_security_group.this[0] proposed, apply changed 1 and the tag is gone, confirmed via the AWS CLI", "greenfield": "67 resources from nothing, all markers verified via the AWS CLI, 67 records in the local record store (#364 A2), replan empty, 6 tagged security groups (4 named + 2 default adopters) and every named one's rule shape matches $PLAIN_EST's own stage-1 apply object by object, tags stripped", "migrate": "58 of 67 stamped, 67 identities recorded (#364 unit A2)", + "plan_approval": "one argument edited (aws_security_group.app's tags merge in Reviewed=yes; the standalone app SG has no rule children, so one argument is one row), \"plan -out=approved.tfplan\" wrote a 87608-byte stock-format plan file whose whole change set is one update on aws_security_group.app (sg-641be14b280c2e82e); the world then moved out of band (a DriftProbe tag on sg-5eb49924036c81c44 through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live security group from the one under review) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.security_group.aws_security_group.this[0] and the live sg-5eb49924036c81c44 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - describe-tags on sg-641be14b280c2e82e still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the DriftProbe tag deleted and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-641be14b280c2e82e read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the app SG confirmed to be the same GroupId it started as, and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 58 objects, read through resourcegroupstaggingapi", "test_plan": "the plan is genuinely empty: every choudoufu wall (#305, #307, #313 A and B, #321, #332) and both confirmed floci gaps (#102, #104) are fixed or absent this run; default route table identities asserted by value against the AWS CLI in step 3a" }, - "duration_s": 210.6, + "duration_s": 218.7, "stage_seconds": { - "cold_deploy": 31, - "day2_count": 29, - "day2_remove": 10, - "day2_rename": 13, + "cold_deploy": 18, + "day2_count": 30, + "day2_remove": 11, + "day2_rename": 14, "day2_replace": 11, - "drift_reconverge": 6, - "greenfield": 27, - "migrate": 75, + "drift_reconverge": 7, + "greenfield": 26, + "migrate": 74, + "plan_approval": 18, "test_apply": 4, - "test_plan": 4 + "test_plan": 5 } }, "notes": "Landed c876435875 (2026-08-18), 67 real resources. The brief expected this crossing to hit #304 (a static lookup()-keyed count-index) directly, since a prior crossing found #304 through this same module as a dependency - checked, not assumed, and refuted: v6.0.0 rewrote the module from the classic single-aws_security_group-with-dynamic-blocks shape #304 lives in to a for_each-over-a-map shape emitting aws_vpc_security_group_ingress_rule/egress_rule per rule key, so that whole pattern is gone from this version's own example. migrate genuinely passes (52 of 67 eligible and stamped: 35 VERIFIED + 17 DRIFTED; 15 skipped - 6 untaggable by design, 9 unadmitted). test_plan blocked by two real, distinct, filed gaps: #305 (6 sites, the familiar default_* adopter trio, doubled since this estate nests two terraform-aws-vpc calls) and a NEW one, filed as #307: aws_vpc_security_group_rules_exclusive is unadmitted (3 sites) - no CFN counterpart and the pinned provider ships no identity schema for it, but its own import docs are unambiguous that security_group_id (required, ForceNew, always a direct parent reference) is its whole identity, the same shape aws_vpc_security_group_vpc_association's already-admitted row has. Also found and filed a real floci gap (lex00/floci#57: EC2 AssociateSecurityGroupVpc has no handler at all), worked around with a documented delta removing the estate's one vpc_associations block (67 of 68 resources still stand up for real). One open, honestly-unresolved observation: 17 of the DRIFTED resources show referenced_security_group_id read back as \"000000000000/sg-xxx\" from floci where config computes a bare \"sg-xxx\" - doesn't block stamping, but live-plan never got past #305/#307's hard refusals to reveal whether the real provider's diff-suppression absorbs this cleanly or would surface as an 18th change. Not filed separately; the next crossing attempt (once #305/#307 land) will show it for real. Follow-up pass 2026-08-19 (#313's data-source fix, c636ab20f7/0284d8c408): re-verified for real against the current pin (67 cold-deployed, 58 stamped, state deleted). test_plan diagnostics dropped from 239 to 19 - #313's canonical data.aws_availability_zones cause (50 sites) is fully gone, the module-output/resource-attribute variant (2 sites) still correctly refuses (out of scope by the maintainer's own ruling), and the 187-site cascade collapsed to 5. The remaining 12 sites are a NEW, newly-reached class (previously masked behind #313's hard refusal): element([*].id, count.index) in the vpc module's aws_route_table_association.private - both operands are tagged resources, the obstacle is the splat-through-function-call shape, not an admission gap. Filed as #321, not attempted (scouting slot). test_plan therefore stays fail, but the estate's real remaining blocker is now #321 (a derivable, no-design-call-needed gap) rather than #313 (an architecture question) - #321 is the clear next step toward this estate's five-of-five and the core set's last gap. Follow-up pass 2026-08-19 (#321 fixed and merged, c33a47288a/626ca84739): element([*].attr, idx) over a splat of tagged resources now resolves generically - it names the same live object a direct indexed traversal already resolves, via element()'s own modulo wraparound. Re-verified for real: test_plan diagnostics 19 -> 7. The 12 splat-through-element sites are confirmed cleared by their absence from real live-plan output; the remaining 7 are #313's own deliberately-out-of-scope resource-attribute root cause (root cause B), a maintainer scope boundary, not a bug. test_plan stays fail - the core set does NOT reach five-of-five from this fix alone. Generalizes beyond this estate: refusal-probe over terraform-aws-modules/vpc's own examples, 21 -> 8 sites, zero regressions - three configs (ipam, ipv6-only, outpost) fully cleared. A related but distinct lint-side wall (RuleCountIndex, 32 sites/6 configs, a genuine unresolved composite-vs-per-argument injectivity design question) stays independently blocking and was left open, documented rather than attempted. Follow-up pass 2026-08-19 (#191 fixed and merged, 312acbbb61/75ef0a6a78): internal/live/identity/partialargs.go's tolerant rebuild now composes across more than one module call and evaluates a call the caller wrote (merge(), not a bare constructor) through an evaluator whose own var.* closure is already tolerant, one module up. module.consul's ingress_referenced_security_group_id map no longer poisons the 22 ingress rules it seeds two module calls down - the map's KEYS (eleven preset names crossed with one caller key) were always written down; only the VALUE under one key was ever unknowable, and it still is. Re-verified for real against floci, script exit 0, BREAK=1 negative control correctly fails: test_plan diagnostics 7 -> 4, and every analysis-layer refusal this estate has ever hit is now 0 and asserted by absence (#305, #307, #313 root causes A and B both, #321). What newly reached PROJECTION and blocks the estate now is #332 (not #313 - the old 'Unable to use aws_security_group.app in static context' framing is confirmed gone): aws_default_route_table imports by the VPC's id, not its own, and the ratified row says otherwise - 2 'Cannot import for projection' + 2 'empty result', one pair per nested vpc module call, both traced to the same type. #332 is filed, not fixed here; it is now the sole remaining blocker on this estate and on the core set's last five-of-five gap. #332 fixed 2026-08-19 (859c1ad747/ff1f6bcdea/c1197befc7): the ratified row claimed aws_default_route_table imports by the route table's own rtb-… id; the real provider imports it by the VPC's id, read off the vpc_id ATTRIBUTE (not argument) the discovered object already carries - settled by running stock terraform 1.15.8 + hashicorp/aws 6.59.0 (Error: empty result for rtb-…, Import successful! for vpc-…). Reach stated honestly: one type today (defaultAdopterSiblings/sameRatifiedIdentity in internal/live/discovery/discovery.go split \"same live object\" from \"same import identity\" generically, off each type's own ratified IdentityAttrs/ImportSyntax, no type name in the control flow - #302's aws_iam_service_linked_role/aws_iam_role pair already exercises the same recomposition path; aws_default_route_table is simply the only aws_default_* row that currently diverges from its plain sibling at aws 6.59.0). Re-verified independently 2026-08-20 with a fresh real crossing against floci (ghcr.io/lex00/floci@sha256:120b6783c7fb48d3d78245056251492b7d9246cdc3b397c98db2659cdc78d94a, the currently pinned image): STAGE 1 PASS, STAGE 2 PASS, STAGE 3 BLOCKED at exactly 1 site (was 239, then 19, then 7, then 4) - every choudoufu-layer refusal this estate has ever hit (#305, #307, #313 root causes A and B, #321, #332) confirmed absent, and step 3a re-derives each nested VPC's default route table import identity from AWS directly (module.vpc: vpc-9eceebf0 -> rtb-7a9e0d017620e163c; module.vpc_secondary: vpc-91d40754 -> rtb-b66d5dc82e63b6e38) and asserts it BY VALUE, not by absence. The 1 remaining site is the AWS provider answering \"Provider produced invalid plan\" on its own requires-replacement path for module.security_group.aws_vpc_security_group_ingress_rule.this[\"dns-from-prefix-list\"] (cty.Path{cty.GetAttrStep{Name:\"\"}}), explicitly a provider bug per the error's own text - filed upstream as #335, genuinely outside this fork's code (the diagnostic names no choudoufu path). test_plan stays \"fail\" in this table's pass/fail/not_run vocabulary since the plan is not clean, but the estate's own blocker has moved entirely off this fork: #332 was the core set's last derivable gap, and #335 (an AWS-provider defect) is what now stands between this estate and five-of-five." @@ -1602,16 +1642,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "cb5ae2009fb85fdafd76126bb1842144855eb9d7", - "date": "2026-09-06T04:29:56Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1623,24 +1663,26 @@ "day2_count": "choudoufu: dropping \"2024\" from module.rustconf_com's CNAME for_each map destroyed exactly module.rustconf_com.aws_route53_record.cname[\"2024\"] (0 add, 0 change, 1 destroy), leaving sibling module.rustconf_com.aws_route53_record.cname[\"2022\"]'s TTL and 27 remaining record sets untouched; adding it back created exactly the same key (0 add, 0 change -> 1 add, 0 change, 0 destroy), restoring its TTL/value and the 28 record-set count, while the sibling and the parent zone's own marker stayed untouched throughout; the next plan is empty; a Route 53 record set carries no server-minted identifier of its own (verified directly against floci, no tofu in the loop: ListResourceRecordSets returns a byte-identical entry across a genuine delete/recreate, only ChangeResourceRecordSets' own per-call ChangeInfo.Id differs), so the destroy is proven by verified ABSENCE rather than an id-diff, unlike this stage's aws_iam_policy/PolicyId and EC2/VpcEndpointId precedents; the G-ORACLE stock oracle on the identical for_each change, plan-only on cold_deploy's own state, shows the identical shape: destroy the dropped key only, propose creating it back, every sibling key untouched both times", "day2_remove": "choudoufu: deleting module.cratesio_com_final's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the hosted zone is genuinely gone from the live account (route53 get-hosted-zone on the old id now errors, read via the AWS CLI, not choudoufu's own report; 7 zones down to 6), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same zone (before any rename ever touched it)", "day2_rename": "moved block: module.rustaceans_org renamed to module.rustaceans_org_moved with zero churn (0 add, 1 change, 0 destroy) - only the zone's own marker rewritten, its 2 record children (A, CNAME) did not move; live-mv: module.cratesio_com (0 records) renamed to module.cratesio_com_final with zero churn, marker rewritten in place; stock oracle over the identical two-module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy), using per-child moved blocks stock's own state-address tracking requires and choudoufu's stateless untaggable-record derivation does not; both live zone ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.areweasyncyet_rs's ForceNew domain argument proposed exactly one zone replace at the same declared address, cascading into its one A record - 2 to add, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old zone (ZGL45ZHYYL0082N) is confirmed gone and the new zone (Z0ABC41F7VX38G5) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new zone, not the destroyed one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", + "day2_replace": "choudoufu: changing module.areweasyncyet_rs's ForceNew domain argument proposed exactly one zone replace at the same declared address, cascading into its one A record - 2 to add, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old zone (Z27X0DRHB3FI7WN) is confirmed gone and the new zone (ZMBCC3J2JR88MMI) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new zone, not the destroyed one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", "drift_reconverge": "one untaggable record drifted, exactly module.rustconf_com.aws_route53_record.cname[\"2016\"] proposed and applied, TTL reconverged to 300, 28 records and the parent marker intact", "greenfield": "35 instances from nothing (7 zones, 28 records), all 7 markers verified via the AWS CLI, replan empty, stock oracle in its own namespace matches structurally on all 7 zones (28 records)", "migrate": "7 stamped, 7 distinct hosted zones, one per module call", + "plan_approval": "one argument edited (module.arewewebyet_org's ttl 300 -> 600; that call declares exactly one record, the www CNAME, so one argument is one instance), \"plan -out=approved.tfplan\" wrote a 23430-byte stock-format plan file whose whole change set is one update on module.arewewebyet_org.aws_route53_record.cname[\"www\"] (\"Plan: 0 to add, 1 to change, 0 to destroy\"); the world then moved out of band (2016.rustconf.com.'s TTL set to 60 in zone ZCO3SP8XZ17EY1E through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance in a DIFFERENT hosted zone from the one under review) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming module.rustconf_com.aws_route53_record.cname[\"2016\"] together with the whole live identity it was computed against, ZCO3SP8XZ17EY1E_2016.rustconf.com_CNAME (this type's ZONEID_NAME_TYPE import syntax, rebuilt in the script from the zone id, record name and type it already knew independently) - with \"Exit status 3\" spelled out for a pipeline; nothing was applied - www.arewewebyet.org. still read TTL 300 through the AWS CLI, which is stronger evidence than the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the drifted TTL put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and www.arewewebyet.org. read back at TTL 600, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the 28 record sets re-counted and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); 7 zones / 28 records unchanged, all 7 markers unmoved", "test_plan": "no resource change proposed, nothing foreign; all 35 rendered identities name a live hosted zone or record set" }, - "duration_s": 490.4, + "duration_s": 502.7, "stage_seconds": { - "cold_deploy": 85, - "day2_count": 53, - "day2_remove": 28, - "day2_rename": 18, - "day2_replace": 70, - "drift_reconverge": 22, - "greenfield": 155, - "migrate": 40, - "test_apply": 14, + "cold_deploy": 70, + "day2_count": 50, + "day2_remove": 24, + "day2_rename": 15, + "day2_replace": 68, + "drift_reconverge": 21, + "greenfield": 151, + "migrate": 41, + "plan_approval": 45, + "test_apply": 13, "test_plan": 5 } }, @@ -1666,16 +1708,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1684,28 +1726,30 @@ "exit_code": 0, "detail": { "cold_deploy": "6 resources added by plain terraform (4 queues + redrive_policy + redrive_allow_policy), 0 objects carry tofu-estate before migration", - "day2_count": "synthetic block (all four module calls declare count = var.create ? 1 : 0, a boolean create toggle, so the estate has no knob that scales - issue #488's sanctioned fallback, reusing aws_sqs_queue, a type this estate already exercises four times): scaling aws_sqs_queue.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) of the HIGHER index, count_test[1]; the survivor count_test[0] kept its live queue URL (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-count-test-0), its CreatedTimestamp (1788657850) and its tofu-address=aws_sqs_queue.count_test:0 / tofu-slot=0 markers, all read back through the AWS CLI, and count_test[1]'s local record was tombstoned rather than left naming a destroyed queue; scaling 1 back to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy), and because a queue URL is rebuilt from region + account + name the recreated instance comes back at the SAME url - so the destroy is witnessed two other ways instead, by AWS.SimpleQueueService.NonExistentQueue in between and by a strictly later CreatedTimestamp (1788657850 -> 1788657932), with tofu-address=aws_sqs_queue.count_test:1 back on the new object and index 0 untouched throughout; the next plan proposes no resource action; the G-ORACLE stock oracle stood the identical block up for real in the idle greenfield-oracle account and showed the identical shape - destroy the higher index only, create it back under the same url with a new CreatedTimestamp, the lower index unchanged both times", + "day2_count": "synthetic block (all four module calls declare count = var.create ? 1 : 0, a boolean create toggle, so the estate has no knob that scales - issue #488's sanctioned fallback, reusing aws_sqs_queue, a type this estate already exercises four times): scaling aws_sqs_queue.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) of the HIGHER index, count_test[1]; the survivor count_test[0] kept its live queue URL (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-count-test-0), its CreatedTimestamp (1788749614) and its tofu-address=aws_sqs_queue.count_test:0 / tofu-slot=0 markers, all read back through the AWS CLI, and count_test[1]'s local record was tombstoned rather than left naming a destroyed queue; scaling 1 back to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy), and because a queue URL is rebuilt from region + account + name the recreated instance comes back at the SAME url - so the destroy is witnessed two other ways instead, by AWS.SimpleQueueService.NonExistentQueue in between and by a strictly later CreatedTimestamp (1788749614 -> 1788749697), with tofu-address=aws_sqs_queue.count_test:1 back on the new object and index 0 untouched throughout; the next plan proposes no resource action; the G-ORACLE stock oracle stood the identical block up for real in the idle greenfield-oracle account and showed the identical shape - destroy the higher index only, create it back under the same url with a new CreatedTimestamp, the lower index unchanged both times", "day2_remove": "choudoufu: deleting module.unencrypted_sqs_renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (sqs get-queue-url on the old name now returns NonExistentQueue, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same object (before any rename ever touched it)", "day2_rename": "moved block: module.default_sqs renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.unencrypted_sqs renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", "day2_replace": "choudoufu: changing module.default_sqs_renamed's ForceNew name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default) is confirmed gone and the new object (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default -> https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied - it shares floci's account with $EST); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one object tampered, exactly 1 object proposed and applied (0 added, 1 changed, 0 destroyed), tag reconverged to \"ex-complete\"", "greenfield": "6 resources from nothing (4 tagged queues + 2 untaggable redrive types), all markers verified via the AWS CLI, 6 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on all 4 queues", "migrate": "4 of 6 eligible (2 untaggable redrive types resolved by provider identity schema), 4 stamped, 0 failed, 2 skipped; tofu-slot=0 written on all 4 queues by the stamp itself (issue #372's remainder), confirmed by value and by a genuine no-op on the follow-up apply", + "plan_approval": "one argument edited (module.unencrypted_sqs's tags gain Reviewed=yes; that module call declares no DLQ, so var.tags reaches exactly one resource instance), \"plan -out=approved.tfplan\" wrote a 26728-byte stock-format plan file whose whole change set is one update on module.unencrypted_sqs.aws_sqs_queue.this[0] (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-unencrypted); the world then moved out of band (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default's Example tag, through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live queue from the one under review) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.default_sqs.aws_sqs_queue.this[0] and the live https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default it was computed against (an aws_sqs_queue's identity IS its URL), with \"Exit status 3\" spelled out for a pipeline; nothing was applied - list-queue-tags on https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-unencrypted still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Example tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-unencrypted read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the unencrypted queue's URL confirmed unchanged and the estate replanned empty with no state file, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); 4 objects before, 4 after, no state file", "test_plan": "no resource change proposed, no foreign resources; fifo and default queue tofu-address re-checked against SQS" }, - "duration_s": 636.1, + "duration_s": 632.2, "stage_seconds": { - "cold_deploy": 70, - "day2_count": 114, - "day2_remove": 50, - "day2_rename": 10, + "cold_deploy": 57, + "day2_count": 115, + "day2_remove": 49, + "day2_rename": 9, "day2_replace": 74, "drift_reconverge": 5, - "greenfield": 187, - "migrate": 121, - "test_apply": 3, - "test_plan": 2 + "greenfield": 185, + "migrate": 120, + "plan_approval": 13, + "test_apply": 2, + "test_plan": 3 } }, "notes": "First real five-stage crossing of this estate, and the first SQS surface in this corpus. Sourced and sketched at e4b12799da with every assertion past stage 1's `terraform apply` DERIVED from reading terraform-aws-sqs's naming locals rather than measured - that commit's own header said so and deliberately added no entry here. This entry is the first one written from a real run: Docker/floci (ghcr.io/lex00/floci@sha256:8a882bcc, live/floci-image's pin), real hashicorp terraform, and the AWS CLI throughout, in worktree ../wt/new-terraform-estate-2 off e4b12799da. All five stages PASS. WHAT THE DERIVATION GOT WRONG, and it was exactly one thing: the resource count. The estate builds SIX managed resources, not five. `create_dlq = true` makes the module emit an aws_sqs_queue_redrive_ALLOW_policy on the DLQ alongside the aws_sqs_queue_redrive_policy on the source queue; reading the naming locals found the second and missed the first. Everything else the sketch derived was right when checked against reality - all four queue names and URLs including the FIFO DLQ's \"-dlq.fifo\" suffix, and all four rendered tofu-address strings. Corrected counts, measured: stage 1 \"Apply complete! Resources: 6 added\", stage 2 \"4 of 6 resource instance(s) are eligible for stamping\" and \"4 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 2 skipped\". THE SCHEMA-FALLBACK RESULT, which is why this estate was sourced. Neither aws_sqs_queue_redrive_policy nor aws_sqs_queue_redrive_allow_policy has a row in internal/live/identity/table_generated.go; live/survey-full.json classifies both identically (path \"client-named\", admission \"schema\", required_for_import [\"queue_url\"], taggable false, list_resource false). Both resolved a live id through the provider's own identity schema, and both resolved to the RIGHT queue, which is not the same queue for the two of them: redrive_policy -> https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete.fifo (the source queue), redrive_allow_policy -> https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-dlq.fifo (the DLQ). The schema-fallback path works on both, unmodified - no fix was needed and none was made. The script now asserts both by value, because a run that merely did not error would pass with the two swapped. Both are UNTAGGABLE (no tags argument in the provider schema), so live-import skips them and stage 2 stamps 4 of 6. That is the invariant working, not a shortfall: four tagged queues plus two resources whose entire identity IS a tagged queue's URL - tagged, plus derived-from-tagged, no third bucket - and stage 3's empty plan with no state file anywhere is what proves the derivation holds. EXPECTED IMPORT-TIME DRIFT, recorded so the next reader does not chase it: live-import reports the two FIFO queues as DRIFTED rather than VERIFIED, on redrive_policy and redrive_allow_policy respectively (cold state has \"\", live has the JSON). That is the module's own design - the queue resource does not manage those attributes, the separate redrive resources do - so the live object carries a value the queue's state row never recorded. DRIFTED is still eligible for stamping, the convergence apply reconciles it, and the next plan is empty. The tofu-slot convergence apply corpus-iam-policy documented recurs here exactly as that entry predicts: all four aws_sqs_queue resources declare count = var.create ? 1 : 0, so one ordinary `choudoufu apply` (\"0 added, 4 changed, 0 destroyed\") is folded into stage 2 before stage 3 is attempted. The two redrive resources are untaggable and carry no slot, which is why it is 4 changed and not 6. Stages 3-5 measured: test_plan proposes no resource action, reports \"Foreign resources: none among the 1 type swept\", both re-read identities unchanged, no state file written; test_apply \"0 added, 0 changed, 0 destroyed\" with the tofu-estate-tagged object count 4 before and 4 after; drift_reconverge tampers ex-complete-default's Example tag directly through the AWS CLI, live-plan proposes updating exactly module.default_sqs.aws_sqs_queue.this[0] and nothing else, and the apply reconverges it (\"0 added, 1 changed, 0 destroyed\", tag back to \"ex-complete\"). THREE SELF-AUTHORED DEFECTS FIXED IN THE SCRIPT WHILE VERIFYING IT, each found by running it rather than reading it. (1) No TF_PLUGIN_CACHE_DIR, which corpus-lambda-simple and corpus-alb-complete both set. Without it the first real run spent 21 minutes in `terraform init` having pulled 48MB of hashicorp/aws and then died on a transient DNS failure before ever reaching `terraform apply` - the likely reason two earlier sessions reported this crossing as stalled rather than failed. With the shared cache the whole five-stage run is minutes. (2) The BREAK contract was unreachable. BREAK=1 set two corruptions, stage 3's and stage 5's, but the stage-3 one calls fail() and exits, so stage 5's branch was dead code that had never run - and its inverted form would have exited 0 on a corrupted run anyway, proving nothing. BREAK is now three named values corrupting three different assertions, each verified for real to exit 1 at its own assertion and each reaching a later stage than the last: BREAK=schema (swap the two expected redrive URLs - both real queues, both types really do resolve, so only a by-value check catches the wrong pairing) fails in stage 2; BREAK=identity fails in stage 3; BREAK=drift fails in stage 5's exactly-one-object assertion with both objects named. An unrecognized BREAK value is rejected up front. This same dead-stage-5-branch shape exists in corpus-iam-policy's script, which this one was copied from - worth a slot there, not touched here. (3) A new managed-shape assertion added in this pass (terraform state list compared by name against the six documented addresses, so a moved corpus pin fails loudly instead of silently crossing a different estate - the exact failure mode that produced the wrong count) first failed on locale collation alone: \".\" and \"_\" sort differently under a UTF-8 locale, so a hand-ordered list never matches a locale-sorted one. Both sides now go through LC_ALL=C sort. Caught because the assertion was run, not reviewed." @@ -1730,16 +1774,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1748,28 +1792,30 @@ "exit_code": 0, "detail": { "cold_deploy": "11 managed resource instances, genuinely cold, genuinely unmarked", - "day2_count": "Scaled aws_security_group.count_test, a self-contained 2-instance count block in its own count_test.tf at an address nothing else in this crossing names. SYNTHETIC, and NOT because this estate has no scalable knob: sumaform declares a real one (var.quantity, \"number of hosts like this one\", driving count = var.quantity on aws_instance.instance), but every resource that knob scales is off the tag rung here - the instance and the EBS volume are markers = record by this crossing's own strict block, aws_volume_attachment is untaggable, and all three carry sumaform's own lifecycle { ignore_changes = [tags] } - so none of them can witness the stage's identity clause off the live object. G4 exercises the real knob for the half it can settle: quantity 1 -> 2 through choudoufu proposes exactly 3 creates, all at index [1] (instance, data disk, volume attachment), index [0] untouched, identical to stock's own plan for the same change on cold_deploy's state (G-ORACLE-QUANTITY), and the estate plans empty again once it is restored; the down direction is NOT exercised on that knob (cold_deploy's state is quantity=1, so there is no 2-instance stock state to scale down from). On the taggable block: scaling 2 -> 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy) and left count_test[0]'s live GroupId sg-f0a2ac3380efc13eb AND its tofu-address marker aws_security_group.count_test:0 unchanged, both read through the AWS CLI rather than choudoufu's report; count_test[1] (sg-043fd4f6862639ce2) verified absent by describe-security-groups length 0 in between; scaling 1 -> 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) back under a NEW GroupId sg-42031c6d2807e6eb0 carrying tofu-address=aws_security_group.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G-ORACLE): the identical 2-instance block stood up FOR REAL in its own idle floci account on FLOCI_PORT+3 - stock never had this block, so cold_deploy's state could not be reused the way day2_remove's and day2_replace's oracles reuse it - shows the identical shape both directions, destroy the higher index only (sg-d3f8475b4536bc66b gone, sg-a7134d986fe9ce70d unchanged) and create the higher index back under a new id (sg-ae09a9da7ca9c4bab). A new GroupId is a sound destroy witness on this pin: probed directly against ghcr.io/lex00/floci@sha256:c55d74e1 with no tofu in the loop, a delete-then-recreate under an identical group name minted sg-8c824b212636a50e9 -> sg-61cec558a8a5a286d. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, so these checks discriminate.", + "day2_count": "Scaled aws_security_group.count_test, a self-contained 2-instance count block in its own count_test.tf at an address nothing else in this crossing names. SYNTHETIC, and NOT because this estate has no scalable knob: sumaform declares a real one (var.quantity, \"number of hosts like this one\", driving count = var.quantity on aws_instance.instance), but every resource that knob scales is off the tag rung here - the instance and the EBS volume are markers = record by this crossing's own strict block, aws_volume_attachment is untaggable, and all three carry sumaform's own lifecycle { ignore_changes = [tags] } - so none of them can witness the stage's identity clause off the live object. G4 exercises the real knob for the half it can settle: quantity 1 -> 2 through choudoufu proposes exactly 3 creates, all at index [1] (instance, data disk, volume attachment), index [0] untouched, identical to stock's own plan for the same change on cold_deploy's state (G-ORACLE-QUANTITY), and the estate plans empty again once it is restored; the down direction is NOT exercised on that knob (cold_deploy's state is quantity=1, so there is no 2-instance stock state to scale down from). On the taggable block: scaling 2 -> 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy) and left count_test[0]'s live GroupId sg-f5399762065bb39ab AND its tofu-address marker aws_security_group.count_test:0 unchanged, both read through the AWS CLI rather than choudoufu's report; count_test[1] (sg-df09988387ba4224c) verified absent by describe-security-groups length 0 in between; scaling 1 -> 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) back under a NEW GroupId sg-27e39476a3e6a8724 carrying tofu-address=aws_security_group.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G-ORACLE): the identical 2-instance block stood up FOR REAL in its own idle floci account on FLOCI_PORT+3 - stock never had this block, so cold_deploy's state could not be reused the way day2_remove's and day2_replace's oracles reuse it - shows the identical shape both directions, destroy the higher index only (sg-8b2649c223f8be264 gone, sg-671372ad2ef1c9820 unchanged) and create the higher index back under a new id (sg-2a483e85eb4452a7d). A new GroupId is a sound destroy witness on this pin: probed directly against ghcr.io/lex00/floci@sha256:c55d74e1 with no tofu in the loop, a delete-then-recreate under an identical group name minted sg-8c824b212636a50e9 -> sg-61cec558a8a5a286d. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, so these checks discriminate.", "day2_remove": "choudoufu: deleting module.server's block proposed exactly three destroys (0 add, 0 change, 3 destroy: the record-based instance and EBS volume, plus the untaggable/derived volume attachment), applied cleanly (0 added, 0 changed, 3 destroyed), the instance and volume are genuinely gone from the live account (instance State=terminated, volume absent, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes the same three destroys", "day2_rename": "moved block: aws_eip.crossing_nat renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_route_table.crossing_public renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.server's image input (ubuntu2204 -> ubuntu2404, both real, both in floci's seeded AMI catalog) proposed exactly one instance replace at the same declared address, cascading into the volume attachment (instance_id is ForceNew there too) - 2 to add, 0 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated via the AWS CLI and the new instance is confirmed running the new image; the local record store's record at the same address now names the new instance's id, not the terminated one (i-5720faebfefe13c10 -> i-c9ddad8b40018244e); the next plan proposes no resource action; BREAK=replace confirms this section's own record check discriminates (a deliberately-wrong expectation against the same real record fails, rather than vacuously passing). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. A manufactured live-object collision (the shape ec2-instance-complete's and corpus-sqs-basic's own BREAK=replace controls report) has no tag surface to be detected from on this markers=record instance and is not exercised here - verified directly that an untagged extra instance is simply invisible to this plan, correctly, not incorrectly.", + "day2_replace": "choudoufu: changing module.server's image input (ubuntu2204 -> ubuntu2404, both real, both in floci's seeded AMI catalog) proposed exactly one instance replace at the same declared address, cascading into the volume attachment (instance_id is ForceNew there too) - 2 to add, 0 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated via the AWS CLI and the new instance is confirmed running the new image; the local record store's record at the same address now names the new instance's id, not the terminated one (i-f3d88dad3155358d2 -> i-d8cfb88e7cd43142e); the next plan proposes no resource action; BREAK=replace confirms this section's own record check discriminates (a deliberately-wrong expectation against the same real record fails, rather than vacuously passing). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. A manufactured live-object collision (the shape ec2-instance-complete's and corpus-sqs-basic's own BREAK=replace controls report) has no tag surface to be detected from on this markers=record instance and is not exercised here - verified directly that an untagged extra instance is simply invisible to this plan, correctly, not incorrectly.", "drift_reconverge": "the crossing VPC's Name tag tampered out of band, plan proposed fixing exactly aws_vpc.crossing, apply changed 1 and reconverged the tag to sumaform-crossing-vpc; module.server's record-based identities unaffected", "greenfield": "11 resources from nothing (7 tag-stamped, 2 recorded via markers = record, 2 untaggable/derived - route_table_association and volume_attachment), replan empty, stock oracle in its own namespace matches on vpc cidr, security-group rule counts and the instance's ami+type", "migrate": "7 stamped, 2 recorded (markers = record honoured at migrate time, GitHub issue #365 slice 2), 0 failed, 2 skipped", + "plan_approval": "one argument edited (aws_internet_gateway.crossing's tags gain Reviewed=yes - one of this crossing's seven tag-stamped objects, chosen over module.server's markers = record instance and volume, which carry sumaform's own lifecycle { ignore_changes = [tags] } and so could not witness a tag edit at all), \"plan -out=approved.tfplan\" wrote a 48587-byte stock-format plan file whose whole change set is one update on aws_internet_gateway.crossing (igw-aa457a90); the world then moved out of band (vpc-1a753819's Name tag through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live object from the one under review) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both aws_vpc.crossing and the live vpc-1a753819 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - describe-tags on igw-aa457a90 still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Name tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and igw-aa457a90 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the gateway confirmed to be the same id it started as, module.server's record-based instance identity confirmed untouched, and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 7 tagged objects before, 7 after, no state file either time; module.server's record-based instance and volume identities unchanged", "test_plan": "Items 4, 5 and 6 (this script's header) are all FIXED and the plan is genuinely empty (\"No changes. Your infrastructure matches the configuration.\"): live-import honours markers = record (located records for aws_instance.instance[0] and aws_ebs_volume.data_disk[0], confirmed at the store and by value against the AWS CLI both right after migrate and again after this empty replan), residue now covers NestingList/NestingSet/NestingMap blocks (internal/live/projection's residueEligibleBlock, widened from the block's SHAPE - whether carriesNoInformation can tell its absence from a real empty answer - never from a type name), and lex00/floci#103 (published in ghcr.io/lex00/floci@sha256:e16d9007a03093b6a6edd22273dee9d8253131f18581b0fa20ae6d34178a3079) now honours RunInstances' BlockDeviceMapping.Ebs.VolumeSize for the root device, closing the one line (root_block_device.volume_size = 8 -> 200) that was this crossing's own last wall. Plan moved 3 to add/0/0 (the original ABSENT gap) -> 2 to add/0/2 to destroy (item 4 fixed, item 5's replacement exposed) -> 0 to add/1 to change/0 to destroy (item 5 fixed) -> empty (item 6 fixed by the emulator)." }, - "duration_s": 589.4, + "duration_s": 626.9, "stage_seconds": { - "cold_deploy": 88, - "day2_count": 122, + "cold_deploy": 75, + "day2_count": 123, "day2_remove": 23, - "day2_rename": 41, - "day2_replace": 72, - "drift_reconverge": 22, - "greenfield": 142, - "migrate": 199, + "day2_rename": 40, + "day2_replace": 71, + "drift_reconverge": 21, + "greenfield": 145, + "migrate": 196, + "plan_approval": 54, "test_apply": 11, - "test_plan": 11 + "test_plan": 12 } }, "notes": "Landed d583dc93b7 (2026-08-18) - the FIRST OpenTofu-native estate crossed (uyuni-project's own maintainers describe it as \"OpenTofu configuration,\" not \"Terraform configuration\"), versus every prior estate tonight being Terraform-authored/OpenTofu-compatible via terraform-aws-modules. Deliberately reduced slice: the full main.tf.aws.example composes four AWS host roles from one leaf module, backend_modules/aws/host, but three of the four (bastion, module.mirror, module.minion) have no root-facing toggle to disable real SSH/Salt provisioning - the \"real boot behavior, out of scope for an emulator\" case. Only module.server exposes provision=false; module.base's own network submodule was also unusable (create_network=true needs CreateDhcpOptions/ReplaceRouteTableAssociation, neither implemented in floci), so this estate's own plain VPC/subnet/NAT resources stand in for it. A real floci gap found and fixed on the way: sumaform's ami.tf evaluates ~23 data \"aws_ami\" blocks unconditionally (one per supported guest OS) regardless of which single image an estate actually launches, and floci's catalog had zero SUSE/Marketplace/Rocky/RHEL entries - seeded 20, reconciled into the combined image alongside tonight's other three floci fixes. cold_deploy and migrate genuinely pass (11 resources, 9 of 11 stamped - 2 correctly untaggable). test_plan blocked by two real, structural rules baked into backend_modules/aws/host itself (the one leaf module every AWS host role shares, so this isn't an artifact of the reduced slice), and on reading both rules' own reasoning neither looks like a choudoufu defect - both are correct, deliberate refusals, not filed: (1) an unconditional, provisioner-less connection block that checkProvisioners flags on its own terms by documented design, dead code in sumaform's own module; (2) lifecycle { ignore_changes = [tags] } on the WHOLE tags argument of aws_instance.instance and aws_ebs_volume.data_disk - sumaform's own comment explains why (SUSE's internal AWS accounts add tags on apply that need preserving), but ignoring the whole argument also silently discards the update that would write tofu-address/tofu-estate, the exact marker-safety failure #306 was about tonight. The fix sumaform's own error text names - ignore_changes = [tags[\"Owner\"]], not the whole argument - is an edit to sumaform's module, out of scope here. Follow-up pass 2026-08-18 (#313 cross-check, fresh worktree, real run): confirmed neither of the two RULE-classified refusals above is #313's wall - zero occurrences of its diagnostic in the raw plan output. Both remain exactly as already documented: permanent, deliberate refusals (checkProvisioners on a dead-code connection block; ignore_changes on the whole tags argument), not filed as new issues, no action needed." @@ -1794,16 +1840,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1812,28 +1858,30 @@ "exit_code": 0, "detail": { "cold_deploy": "Apply complete! Resources: 62 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=vpc-complete-crossing before migration", - "day2_count": "choudoufu: scaling aws_customer_gateway.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and cgw-ae204e601ae5e91da is confirmed gone from the live account while count_test[0] (cgw-f6fed178e29c19fac) kept its server-minted id, its tofu-address marker and its tofu-slot; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a genuinely NEW object (cgw-84118c223099f26a4, not the destroyed cgw-ae204e601ae5e91da), with count_test[0] untouched throughout; the next plan is empty. Every identity above was read off EC2 through the AWS CLI, by the tofu-address marker, never from choudoufu's own report. Stock oracle (G-ORACLE): the identical count block stood up by terraform in its own floci account and scaled 2->1->2 for real shows the identical shape - destroy count_test[1] only, recreate it under a new id, count_test[0]'s id unchanged both times. THE BLOCK IS SYNTHETIC, deliberately: every real count knob in this estate is a subnet list that drives an untaggable aws_route_table_association count block in lockstep (a question about orphaned derived children, not about count-slot binding), and the one real zero-cascade knob, module.vpc's customer_gateways, is a for_each map with no index slots and is already day2_replace's target - so the block uses aws_customer_gateway, a type this estate really does exercise three of. Building it found and fixed a real defect (five-row row 2): the scale-down proposed NO destroy at all, because the removal sweep read live/mapping.json's via=\"former2\" provenance - a Cloud Control ENUMERABILITY filter - to answer the identity question \"which TF type is this ARN\", and classified aws_customer_gateway TYPE_NOT_LISTABLE against live/registry.json's own handlers.list=true.", + "day2_count": "choudoufu: scaling aws_customer_gateway.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and cgw-3768750fdbdaf84d8 is confirmed gone from the live account while count_test[0] (cgw-fe83feba57bb37ac0) kept its server-minted id, its tofu-address marker and its tofu-slot; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a genuinely NEW object (cgw-54a679d9fd7f89f8a, not the destroyed cgw-3768750fdbdaf84d8), with count_test[0] untouched throughout; the next plan is empty. Every identity above was read off EC2 through the AWS CLI, by the tofu-address marker, never from choudoufu's own report. Stock oracle (G-ORACLE): the identical count block stood up by terraform in its own floci account and scaled 2->1->2 for real shows the identical shape - destroy count_test[1] only, recreate it under a new id, count_test[0]'s id unchanged both times. THE BLOCK IS SYNTHETIC, deliberately: every real count knob in this estate is a subnet list that drives an untaggable aws_route_table_association count block in lockstep (a question about orphaned derived children, not about count-slot binding), and the one real zero-cascade knob, module.vpc's customer_gateways, is a for_each map with no index slots and is already day2_replace's target - so the block uses aws_customer_gateway, a type this estate really does exercise three of. Building it found and fixed a real defect (five-row row 2): the scale-down proposed NO destroy at all, because the removal sweep read live/mapping.json's via=\"former2\" provenance - a Cloud Control ENUMERABILITY filter - to answer the identity question \"which TF type is this ARN\", and classified aws_customer_gateway TYPE_NOT_LISTABLE against live/registry.json's own handlers.list=true.", "day2_remove": "choudoufu: deleting the dynamodb endpoint's map entry (module.vpc_endpoints_renamed.aws_vpc_endpoint.this[\"dynamodb\"]) proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the endpoint is genuinely gone from the live account (State=absent, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same object", "day2_rename": "moved block: module.vpc_endpoints renamed with zero churn (0 add, 7 change, 0 destroy), marker rewritten in place across its taggable objects; live-mv: aws_security_group.rds renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing customer_gateways[\"IP1\"]'s ForceNew ip_address argument proposed exactly one isolated replace at the same declared for_each key (1 to add, 1 to destroy, nothing else), matching F-ORACLE's own plan shape; applied cleanly; the old gateway (cgw-c1ef80761a5322b97) is confirmed gone/deleted and the new gateway (cgw-bf793a3d862648297) carries the marker, both via the AWS CLI; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", + "day2_replace": "choudoufu: changing customer_gateways[\"IP1\"]'s ForceNew ip_address argument proposed exactly one isolated replace at the same declared for_each key (1 to add, 1 to destroy, nothing else), matching F-ORACLE's own plan shape; applied cleanly; the old gateway (cgw-475f57e0a2f706305) is confirmed gone/deleted and the new gateway (cgw-b7f5d49f8756a85cd) carries the marker, both via the AWS CLI; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", "drift_reconverge": "one subnet tampered (Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag to ex-complete", "greenfield": "62 resources from nothing (40 tag-stamped, 22 untaggable/derived), replan empty, stock oracle in its own namespace matches on vpc cidr, subnet count (18) and the s3 endpoint's presence", "migrate": "40 stamped, 22 skipped, 0 recorded, 0 failed; 39 objects carry tofu-estate=vpc-complete-crossing; the VPC's tofu-slot reads 0 off EC2, written by the migration itself (choudoufu #372)", + "plan_approval": "one argument edited (aws_security_group.rds's tags gain Reviewed=yes, a tags-only update that is not ForceNew and leaves sg-0373a87e083bfa5dd's id alone for PART D's later rename), \"plan -out=approved.tfplan\" wrote a 60349-byte stock-format plan file whose whole change set is one update on aws_security_group.rds; the world then moved out of band (subnet-22deaf7e's Example tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.vpc.aws_subnet.private[0] and the live subnet-22deaf7e it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - sg-0373a87e083bfa5dd still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-0373a87e083bfa5dd read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 39 objects before, 39 after, no state file either time", "test_plan": "empty plan; identity re-check unchanged: module.vpc.aws_vpc.this:0, aws_security_group.rds, module.vpc_endpoints.aws_vpc_endpoint.this:s3" }, - "duration_s": 289.3, + "duration_s": 298.1, "stage_seconds": { - "cold_deploy": 34, - "day2_count": 92, - "day2_remove": 21, + "cold_deploy": 24, + "day2_count": 93, + "day2_remove": 20, "day2_rename": 13, "day2_replace": 20, "drift_reconverge": 7, - "greenfield": 55, - "migrate": 94, + "greenfield": 56, + "migrate": 96, + "plan_approval": 18, "test_apply": 4, - "test_plan": 4 + "test_plan": 3 } }, "notes": "RE-CROSSED FOR REAL 2026-08-21 in worktree live/dhcp-options-355 off local main 41f8c8dd6a, real Docker/floci/terraform/AWS CLI throughout, floci ghcr.io/lex00/floci@sha256:cdd50ec0. STAGES UNCHANGED at 2 of 5, and the honest headline is that #355's wall IS gone and three more stand behind it, none of them a choudoufu defect. Stage 1: 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', 0 objects carrying tofu-estate beforehand. Stage 2: dry run '40 of 62 eligible, 22 untaggable, 0 unadmitted'; -approve '40 stamped, 22 skipped, 0 recorded, 0 failed'; three identities asserted by value through the AWS CLI, all three passed. STAGE 3 NO LONGER REFUSES IN DISCOVERY: with #355 fixed, live-plan exits 0 and renders a full plan for the first time in this estate's history. Proven by A/B against the SAME live migrated estate, same floci container, two binaries: main (41f8c8dd6a) exits 1 with 'Error: Listed resource with no tags - Cloud Control listed a aws_vpc_dhcp_options (AWS::EC2::DHCPOptions) with no Tags in its Properties ... Live resources: dopt-default'; the fixed binary exits 0 with that diagnostic absent and every other diagnostic identical. WHAT STANDS BEHIND IT, all measured in that run and none of them choudoufu's: (1) the first live-plan after live-import proposes adding tofu-slot to 31 objects. That is documented, deliberate product behavior (see live/e2e/corpus-iam-policy/run.sh's THE TOFU-SLOT FINDING; live-import cannot compute a slot from one state file), and three other crossing scripts fold a convergence apply into stage 2. This script does not, because it had never reached stage 3 to notice. (2) that convergence apply fails on a floci gap: 'UnsupportedOperation: Operation ModifyVpcEndpoint is not supported', 4 errors, one per interface VPC endpoint. 26 of the 31 tofu-slot writes do land. (3) the replan after that shows 'Plan: 3 to add, 5 to change, 3 to destroy' - three FORCED REPLACEMENTS from floci read fidelity, not drift: aws_nat_gateway.this[0] (floci's DescribeNatGateways returns neither allocation_id nor subnet_id, so both read as absent and force replacement), aws_vpn_gateway.this[0] (floci returns availability_zone='eu-west-1a' on a gateway whose config sets none, so the plan reads '- availability_zone -> null # forces replacement'), and aws_vpc_endpoint.this[s3], plus four endpoint in-place diffs (policy, route_table_ids, subnet_ids, cidr_blocks all read back empty). All floci work items, not choudoufu ones, and all four filed together as lex00/floci#97 (which also carries the Redshift Tagging-API gap below). THE #355 LOOSE END IS SETTLED, with evidence: the '39 objects carry tofu-estate' line against the '40 stamped' line is a floci Tagging-API coverage gap, not a choudoufu miscount. The 40th object is aws_redshift_subnet_group.redshift[0]; 'aws redshift describe-cluster-subnet-groups' shows it carrying tofu-address=module.vpc.aws_redshift_subnet_group.redshift:0 and tofu-estate=vpc-complete-crossing, while 'resourcegroupstaggingapi get-resources' for the same estate returns 39 ARNs with no Redshift among them. choudoufu's own stamp count is the correct one. ONE PRIOR FIGURE CORRECTED: the note below records 'nine non-fatal Incomplete sweep for undeclared resources warnings'. The real number is 989, identical on both binaries - it is the tag sweep's ARN-join-table coverage list (internal/live/discovery/tagging.go), produced before the type scans and untouched by this fix. Nine was a sample, not a count. test_apply and drift_reconverge stay not_run: stage 3 still ends in FAIL, so running them would prove nothing. PRIOR HISTORY BELOW. RE-CROSSED FOR REAL 2026-08-21 at dbc5ebf575 (GitHub issue #346's fix), worktree live/live-read-346, real Docker/floci/terraform/AWS CLI throughout. STAGES UNCHANGED at 2 of 5 - and that is the honest headline, because #346's own diagnostic IS gone and a different wall stands behind it. Stage 1: 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', 0 objects carrying tofu-estate beforehand. Stage 2: dry run '40 of 62 eligible, 22 untaggable, 0 unadmitted'; -approve '40 stamped, 22 skipped, 0 recorded, 0 failed'; three identities asserted by value straight through the AWS CLI and all three passed (vpc-68ae42e6 = module.vpc.aws_vpc.this:0, sg-1bac9cc2a99da4e8c = aws_security_group.rds, vpce-f81456900c35660a0 = module.vpc_endpoints.aws_vpc_endpoint.this:s3). Stage 3 no longer refuses in identity resolution at all: the #346 diagnostic (cidr_blocks = lookup(each.value, 'cidr_blocks', null) reaching module.vpc.vpc_cidr_block) appears nowhere in the run, and the run gets past live_plan.go's step 4 (identity, fatal on error) into step 5 (discovery), which it could not have done otherwise. It now fails on exactly ONE blocking diagnostic, a NEW one that was hidden behind #346 one stage earlier: 'Listed resource with no tags - Cloud Control listed a aws_vpc_dhcp_options (AWS::EC2::DHCPOptions) with no Tags in its Properties, and refining it with GetResource found none either, so its ownership markers cannot be read. Live resources: dopt-default.' That is the ACCOUNT'S DEFAULT DHCP options set, which this estate did not create and does not declare. Filed separately. Nine non-fatal 'Incomplete sweep for undeclared resources' warnings also print (aws_xray_* x5, kubernetes_* x4), none of them a type this estate declares. One unexplained figure worth someone's hour: the run's own closing line reads '39 objects carry tofu-estate after migration' against the '40 stamped' line above it - the script does not assert it, so nothing failed, but the two disagree. test_apply and drift_reconverge stay not_run: running them against a refused plan would prove nothing. PRIOR HISTORY BELOW. Re-verified 2026-08-18 against ghcr.io/lex00/floci@sha256:f5b46236c6b6fff376ae2db8a2b3a51bf1d13b19a92b6a37af4827ccdf1ef180, published via floci's own CI/GHCR-publish workflows (pushed to origin/main, not a local multi-arch build - see HANDOFF.md's Traps). lex00/floci#66 (Redshift), #67 (DHCP options) and #69 (customer/VPN gateway) are confirmed fixed: cold deploy no longer errors on any of the three. It now fails one step later, on a fourth, narrower and previously-undetected gap in #68's own CreateCacheSubnetGroup/ModifyCacheSubnetGroup implementation - it reads the SubnetIds member list under the generic SubnetIds.member.N key, but ElastiCache's service model overrides SubnetIdentifierList's member locationName to SubnetIdentifier, so every real client sends SubnetIds.SubnetIdentifier.N and the call 400s with MissingParameter. Filed as lex00/floci#70. A fix for exactly this was already sitting uncommitted in the shared floci checkout (another session's in-progress work, left untouched - not this orchestrator's to land). Stages 2-5 remain implemented in the script but unexercised since stage 1 still fails fast by design. Follow-up pass 2026-08-20, isolated worktree off local main (ea9fd62fc0), real Docker/floci/AWS CLI throughout, read from the script's own PASS/FAIL lines: cold_deploy and migrate now PASS - this estate moves 0 of 5 to 2 of 5. lex00/floci#70 was already fixed AND already pinned (99f4cbce8f moved live/floci-image to sha256:5873331d, 83c1aa73's published build) - the brief that sent this pass in believed the pin predated it and was wrong; re-running against that existing pin confirmed CreateCacheSubnetGroup succeeds and surfaced the NEXT gap one call later. That gap was lex00/floci#71 (ElastiCache served no tagging actions at all: ListTagsForResource/AddTagsToResource/RemoveTagsFromResource absent from ElastiCacheQueryHandler's switch, and CacheSubnetGroup carried no tags field), and the AWS provider calls ListTagsForResource on EVERY read of aws_elasticache_subnet_group, so the resource was unusable even with no tags in the configuration - the sole remaining cold-deploy error, 1 of 1. Fixed in floci and merged to lex00/floci main (dc140fb0; CI and GHCR-publish both green), mirroring RDS's identical trio on the identical Query protocol; tags live on the CacheSubnetGroup model beside its existing arn field so TaggedResourceScanner picks them up for resourcegroupstaggingapi GetResources with no extra wiring (asserted by a new test, not assumed), and resolveTagHandle reads the resource type off the ARN so another taggable ElastiCache resource is one branch plus a tags field on its model. live/floci-image re-pinned here to sha256:dc246b1e, with live/floci-capabilities.json regenerated for that digest (services with -watch networkmanager,storagegateway, cloudcontrol, cloudcontrol-scoped, tagging) plus the three hand-probed rows re-verified live against the new image - redshift implemented, qldb still unimplemented, opensearch still partial - giving 86 services / 749 types, matching the prior digest block's shape exactly with no existing digest block touched. Real numbers from the run: STAGE 1 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', 0 objects tagged before migration; STAGE 2 dry run '40 of 62 resource instance(s) are eligible for stamping' with UNTAGGABLE (22) and no UNADMITTED_TYPE section at all, -approve '40 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 22 skipped', and three identities read straight through the AWS CLI: module.vpc.aws_vpc.this:0 on the VPC, aws_security_group.rds on the RDS SG, module.vpc_endpoints.aws_vpc_endpoint.this:s3 on the S3 endpoint; 39 objects carry tofu-estate after migration. All 22 skips are genuinely untaggable types (aws_route, aws_route_table_association, aws_vpc_dhcp_options_association, aws_security_group_rule - none has a tags argument in the provider's schema), i.e. the invariant working, not a gap. TWO SELF-INFLICTED SCRIPT BUGS FOUND AND FIXED, neither of which had ever run because stage 1 had never passed: the stage-2 count assertion demanded '0 skipped' (impossible for this estate; it now asserts 40/22 by value plus UNADMITTED_TYPE by absence), and all three identity assertions compared the tag value against OpenTofu's BRACKET spelling ('module.vpc.aws_vpc.this[0]') when a tag value can never carry '[' - the escaped form is 'module.vpc.aws_vpc.this:0' per internal/live/markers.EscapeKey. That is the same vacuous-comparison bug corpus-iam-policy and corpus-iam-read-only-policy each shipped once; both forms are now separate variables, and the bracket forms are used where stage 5 reads a plan diff header. STAGE 3 fails on EXACTLY ONE diagnostic, filed as #346, whose headline finding refutes the obvious fix: the diagnostic points at lookup(each.value, 'cidr_blocks', null) on the vpc-endpoints module's line 116, but a resolveLookupCall beside coalesce.go would NOT unblock this estate. Checked with three hand-built live-check variants rather than assumed: a static-valued lookup() already resolves, and the same map written as a direct each.value.cidr_blocks refuses identically. What refuses is the VALUE - the example passes [module.vpc.vpc_cidr_block], i.e. aws_vpc.this[0].cidr_block, a non-identity attribute of another managed resource, into aws_security_group_rule's identity-bearing cidr_blocks. Written inline the resolver reaches the reference and says so ('Not an identity attribute'); through a local or a for_each map it falls to the generic refusal at resolve.go:2025. Same wall, two spellings - so widening the decomposition switch changes the message and leaves the estate blocked. Whether identity resolution may fold a managed resource's own CONFIGURED attribute (this cidr_block is local.vpc_cidr, a static string) is a maintainer design call, so it was filed rather than forced. test_apply and drift_reconverge remain not_run: attempting them against a still-refused plan would prove nothing." @@ -1858,16 +1906,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1883,19 +1931,21 @@ "drift_reconverge": "one object tampered (Name tag), exactly module.vpc.aws_vpc.this[\"main\"] proposed by both choudoufu and stock with the identical change, apply changed 1 and the Name tag reads back as configured", "greenfield": "28 resources from nothing (matching stage 1's stock cold-deploy count exactly), all markers verified via the AWS CLI, 28 records in the local record store (#364 A2), replan empty, object-by-object comparison against stock's still-pristine cold deploy on $ENDPOINT matches on tagged-object count (21), VPC CIDR, subnet/NAT-gateway/VPC-endpoint counts and account alias", "migrate": "live-import -approve completed cleanly against the cold state", + "plan_approval": "one argument edited (module.vpc.aws_default_security_group.this[\"main\"]'s tags gain Reviewed=yes - a single for_each instance nothing else in the module references), \"plan -out=approved.tfplan\" wrote a 38420-byte stock-format plan file whose whole change set is one update on module.vpc.aws_default_security_group.this[\"main\"]; the world then moved out of band (vpc-e1d1a6dd's Name tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.vpc.aws_vpc.this[\"main\"] and the live vpc-e1d1a6dd it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - sg-e5666543a0a301fd7 still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-e5666543a0a301fd7 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 21", "test_plan": "no resource change proposed" }, - "duration_s": 206.9, + "duration_s": 209.8, "stage_seconds": { - "cold_deploy": 42, + "cold_deploy": 32, "day2_count": 29, "day2_remove": 18, - "day2_rename": 10, + "day2_rename": 11, "day2_replace": 7, - "drift_reconverge": 8, - "greenfield": 38, - "migrate": 49, + "drift_reconverge": 7, + "greenfield": 39, + "migrate": 48, + "plan_approval": 13, "test_apply": 3, "test_plan": 3 } @@ -1912,7 +1962,7 @@ "stages": { "cold_deploy": "pass", "day2_count": "pass", - "day2_crash": "pass", + "day2_crash": "fail", "day2_remove": "pass", "day2_rename": "pass", "day2_replace": "pass", @@ -1920,49 +1970,51 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "pass", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:06:04Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", "tofu": "1.12.5" }, - "exit_code": 0, + "exit_code": 1, "detail": { "cold_deploy": "5 resources from plain terraform, a real terraform.tfstate, zero markers", "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live id and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under a NEW live id (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the B1.7 stock oracle on the same 2-instance count block, applied fresh in the idle greenfield account, shows the identical shape: destroy the higher index only, create the higher index back under a new id, the lower index's id unchanged both times", - "day2_crash": "choudoufu: a real create_before_destroy replace of aws_instance.main was interrupted with SIGTERM (landed on attempt 1 of 3; deterministic by construction, not by timing luck - internal/command/apply_e2etesting_crash.go self-signals synchronously inside the single -parallelism=1 graph-walker goroutine the instant the create half's own write-back commits in memory, replacing #483's external tail/grep/kill race that produced issue #490's own retry-lottery evidence) strictly between the create committing (new object i-31f6c7e770302d57f, confirmed running via the AWS CLI) and the destroy of the deposed old object (i-c705a31c96a9d1588, confirmed still running and untouched via the AWS CLI) ever dispatching; the local record's one write-back correctly carried both facts at once (current=i-31f6c7e770302d57f, deposed=i-c705a31c96a9d1588). Real investigation before writing this check found a genuine engine gap: issue #415's record-backed collision branch (internal/live/discovery/discovery.go, decl.recordBacked's 2-claimant path) called collisionProblem unconditionally with no deposed-record lookup at all, so a record-backed address's own crash window - exactly what a real crash's write-back leaves, since it answers the address's CURRENT identity in the same commit - could never recover on its own; fixed generically (mirrors the scalar path's own matchDeposedClaimant call, no resource type name in the fix), covered by two new unit tests (internal/live/discovery/deposed_test.go). The next plan proposed exactly one destroy (the deposed object, 0 add, 0 change, 1 destroy) and nothing else, matching stock's own documented deposed-object semantics (Stock records the old object as deposed and destroys it on the next apply); applying it destroyed exactly that object (confirmed terminated via the AWS CLI), cleared the deposed record entry, and left the current identity untouched; the plan after that is empty. BREAK_CRASH=1 confirms the empty-plan assertion this stage's Break text names correctly fails to hold against the same real crash window.", + "day2_crash": "the post-recovery plan exited 1", "day2_remove": "choudoufu: deleting aws_internet_gateway.renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (describe-internet-gateways on the old id no longer returns it, read via the AWS CLI, not choudoufu's own report), and the next plan is empty; stock oracle on cold_deploy's own state (B1.6) also proposes exactly one destroy for the same object; classifyOrphans did not withhold the destroy because no other aws_internet_gateway block is declared anywhere in this config", "day2_rename": "moved block: aws_security_group renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_internet_gateway renamed with zero churn, marker rewritten in place; stock oracle over the same two-resource rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing aws_instance.main's ForceNew ami argument proposed exactly one isolated instance replace at the same declared address (1 to add, 1 to destroy, nothing else), matching B1.8's own plan shape; applied cleanly; the old instance (i-b2dcd2cdc161c3dc3) is confirmed terminated and the new instance (i-c705a31c96a9d1588) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance, not the terminated one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", + "day2_replace": "choudoufu: changing aws_instance.main's ForceNew ami argument proposed exactly one isolated instance replace at the same declared address (1 to add, 1 to destroy, nothing else), matching B1.8's own plan shape; applied cleanly; the old instance (i-49b443856f409d5a7) is confirmed terminated and the new instance (i-65a7966e4fd5a1a42) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance, not the terminated one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", "drift_reconverge": "one object tampered, exactly aws_instance.main proposed, apply changed 1 and the tag reads back as configured", "greenfield": "5-object structural comparison (vpc/subnet/igw/sg/instance) between the greenfield estate and stock's cold deploy matches, via the AWS CLI on both endpoints, marker tags never compared; local record store held 5 records, one per instance (#364 A2); replanned empty both with and without the local record store", "migrate": "5 of 5 verified, 5 stamped, 0 skipped", + "plan_approval": "one argument edited (aws_subnet.main's tags gain Reviewed=yes - the one leaf of this five-resource estate no later part renames, removes, replaces or crashes, and a tags-only update that is not ForceNew, so the subnet id aws_instance.main points at is untouched), \"plan -out=approved.tfplan\" wrote a 9808-byte stock-format plan file whose whole change set is one update on aws_subnet.main; the world then moved out of band (i-49b443856f409d5a7's Name tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both aws_instance.main and the live i-49b443856f409d5a7 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - subnet-303d6d3d still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and subnet-303d6d3d read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "strict": "every strict toggle on (secrets = refuse, no_source_create = refuse, marker_repair = never with a markers \"record\" selection naming aws_ebs_volume) against a scratch estate carrying one resource, random_password.db: exactly one refusal, matching live/LIMITATIONS.md's \"strict-secrets\" text word for word (Logical resource is not admitted / SECRET_REFUSED / strict { secrets = \"refuse\" }); no_source_create and marker_repair are on and silent, reaching nothing this config declares. BREAK_STRICT=1 turns secrets back to \"store\" alone: the refusal disappears, the plan becomes an ordinary create, and no other refusal appears. Not part of the headline bars: tools/gauntlet/stages.go keeps Status planned here, because isClear (tools/gauntlet/artifact.go) and NextUnits (tools/gauntlet/next.go) both key strictly off ActiveStages today, with no exemption for a stage the docs already call non-headline - flipping Status without first adding that exemption would silently start gating the two headline bars on this stage, which #363 did not ask for and this unit did not build.", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 5", "test_plan": "post-adoption plan is empty; markers read back through the AWS CLI in part A" }, - "duration_s": 260.4, + "duration_s": 247.5, "stage_seconds": { - "cold_deploy": 101, + "cold_deploy": 87, "day2_count": 17, - "day2_crash": 34, - "day2_remove": 6, - "day2_rename": 9, - "day2_replace": 27, + "day2_crash": 31, + "day2_remove": 5, + "day2_rename": 8, + "day2_replace": 26, "drift_reconverge": 4, "greenfield": 3, - "migrate": 52, + "migrate": 51, + "plan_approval": 11, "strict": 2, - "test_apply": 3, + "test_apply": 2, "test_plan": 2 } }, @@ -1986,16 +2038,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "d72960cdc31f4e7d90f61088920379fc29b3999c", - "date": "2026-09-06T05:35:09Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -2011,22 +2063,24 @@ "drift_reconverge": "one live object mutated out of band through the AWS CLI; choudoufu's next plan proposed fixing exactly aws_vpc.main and nothing else (0 add, 1 change, 0 destroy), matching stock's own plan for the identical mutation on cold_deploy's own state (B4, taken before any marker existed); the apply changed exactly 1 resource, the Name tag reads back as configured and the tofu-address marker is unchanged", "greenfield": "choudoufu applied 79 resources into an account a stock destroy had left enumerated empty (A2), and its cloud matches stock's cold deploy across 79 structural facts compared object by object with marker tags never read on either side - the oracle this stage names. Also, beyond the oracle: the six representative identities are correct by value via the AWS CLI across Route 53/IAM/ECS/EC2; the apply persisted 79 records, matching stock's own instance list type for type with no gap - #671 closed the last one (aws_ecs_task_definition), which used to get no record and now does; the next plan is empty; and with the local record store deleted outright every one of the 79 objects is still found - nothing created, destroyed or replaced, 41 of them untaggable and composing from a stamped parent - with the only movement being 1 residue-held aws_ecs_service update(s), which is what deleting the residue store (issue #275) means rather than a divergence", "migrate": "live-import ratified 38 of 79 instances as eligible and stamped all 38 with 0 failed and 41 skipped (untaggable, identity composed from an already-stamped parent); every one of the 79 addresses in stock's own `terraform state list` - this stage's oracle - is accounted for by name in the report", + "plan_approval": "one argument edited through the generator's own render_config path (the reviewp case: aws_subnet.main's tags gain Reviewed=yes - the one shared, singular, taggable resource no later part renames, removes, scales or replaces, and a tags-only update that is not ForceNew, so the subnet id ecs.tf reads is untouched), \"plan -out=approved.tfplan\" wrote a 31735-byte stock-format plan file whose whole change set is one update on aws_subnet.main (Plan: 0 to add, 1 to change, 0 to destroy); the world then moved out of band (vpc-f48eb49d's Name tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both aws_vpc.main and the live vpc-f48eb49d it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - subnet-8a9e6960 still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and subnet-8a9e6960 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The estate re-renders to the pristine generator output, replans empty and the VPC's tofu-address marker still reads aws_vpc.main. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "strict": "this crossing script does not exercise the strict toggles: strict is Headline:false in tools/gauntlet/stages.go so it moves neither bar, and a toggle-by-toggle refusal fixture is a separate unit from the crossing this script exists to be. live/e2e/reference-ec2-vpc/run.sh's PART G is the pattern for the estate that does carry one", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); the estate is enumerated object by object before and after - 34 objects across IAM/Route53/ECS/EC2, byte-identical listings, never a bare count - and the tofu-estate-tagged count is unchanged at 38", "test_plan": "post-migration plan is empty; six rendered identities asserted BY VALUE against the AWS CLI across four separate tagging surfaces (Route 53, IAM, ECS, EC2), including the count-indexed aws_iam_role.count_team[1] and the module-nested, double-indexed module.team_pod[\"pod-a\"].aws_iam_role.pod_role[0]" }, - "duration_s": 427.8, + "duration_s": 342.1, "stage_seconds": { - "cold_deploy": 167, - "day2_count": 28, - "day2_remove": 11, - "day2_rename": 22, - "day2_replace": 19, - "drift_reconverge": 39, - "greenfield": 85, - "migrate": 47, + "cold_deploy": 124, + "day2_count": 18, + "day2_remove": 7, + "day2_rename": 18, + "day2_replace": 12, + "drift_reconverge": 33, + "greenfield": 65, + "migrate": 43, + "plan_approval": 13, "strict": 0, - "test_apply": 6, + "test_apply": 5, "test_plan": 4 } } diff --git a/site/content/docs/progress/_index.md b/site/content/docs/progress/_index.md index 49ed46497b..bbd15ea7e9 100644 --- a/site/content/docs/progress/_index.md +++ b/site/content/docs/progress/_index.md @@ -15,7 +15,7 @@ passes - an active stage not marked "no" in the Headline column below. {{< gauntlet-bars >}} -Every estate below last ran against the pinned emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`, recorded between 2026-09-06T01:27:32Z and 2026-09-06T23:49:04Z. Each row below carries its own `last_run` date; they are not all the same run. +Every estate below last ran against the pinned emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`, recorded between 2026-09-07T03:00:49Z and 2026-09-07T03:06:04Z. Each row below carries its own `last_run` date; they are not all the same run. The behaviors-proven line above counts how many of the 14 stages below have a FAST tier-1 fixture (`live/behaviors.json`) - a small, purpose-built script @@ -57,67 +57,67 @@ answer is and how each check is proven non-vacuous, is | Estate | Set | Lane | Clear | Stages | |---|---|---|---|---| -| [corpus-alb-complete]({{< relref "corpus-alb-complete" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-autoscaling-complete]({{< relref "corpus-autoscaling-complete" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-dynamodb-table-basic]({{< relref "corpus-dynamodb-table-basic" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-ec2-instance-complete]({{< relref "corpus-ec2-instance-complete" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-ecs-fargate]({{< relref "corpus-ecs-fargate" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-eks-basic]({{< relref "corpus-eks-basic" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-evoteum-modules]({{< relref "corpus-evoteum-modules" >}}) | core | opentofu-native | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-giantswarm-crossplane]({{< relref "corpus-giantswarm-crossplane" >}}) | core | opentofu-native | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-hongbomiao-harbor]({{< relref "corpus-hongbomiao-harbor" >}}) | core | opentofu-native | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-hongbomiao-labelbox]({{< relref "corpus-hongbomiao-labelbox" >}}) | core | opentofu-native | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-hongbomiao-storage]({{< relref "corpus-hongbomiao-storage" >}}) | core | opentofu-native | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-iam-policy]({{< relref "corpus-iam-policy" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-iam-read-only-policy]({{< relref "corpus-iam-read-only-policy" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-lambda-simple]({{< relref "corpus-lambda-simple" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-leynos-monitoring]({{< relref "corpus-leynos-monitoring" >}}) | core | opentofu-native | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-overture-tiles]({{< relref "corpus-overture-tiles" >}}) | core | opentofu-native | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-rds-complete-postgres]({{< relref "corpus-rds-complete-postgres" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-s3-bucket-complete]({{< relref "corpus-s3-bucket-complete" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-security-group-complete]({{< relref "corpus-security-group-complete" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-simpleinfra-dns]({{< relref "corpus-simpleinfra-dns" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-sqs-basic]({{< relref "corpus-sqs-basic" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-sumaform-aws]({{< relref "corpus-sumaform-aws" >}}) | core | opentofu-native | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-vpc-complete]({{< relref "corpus-vpc-complete" >}}) | core | terraform-popular | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-xancloud-iac]({{< relref "corpus-xancloud-iac" >}}) | core | opentofu-native | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [reference-ec2-vpc]({{< relref "reference-ec2-vpc" >}}) | core | reference | no | pass pass pass pass pass pass pass pass pass not run pass pass | -| [terralith-scale]({{< relref "terralith-scale" >}}) | core | reference | no | pass pass pass pass pass pass pass pass pass not run pass not run | -| [corpus-mastino-dns]({{< relref "corpus-mastino-dns" >}}) | growing | published-deployment | no | pass pass pass pass pass pass pass pass pass not run pass not run | +| [corpus-alb-complete]({{< relref "corpus-alb-complete" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-autoscaling-complete]({{< relref "corpus-autoscaling-complete" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-dynamodb-table-basic]({{< relref "corpus-dynamodb-table-basic" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-ec2-instance-complete]({{< relref "corpus-ec2-instance-complete" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-ecs-fargate]({{< relref "corpus-ecs-fargate" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-eks-basic]({{< relref "corpus-eks-basic" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-evoteum-modules]({{< relref "corpus-evoteum-modules" >}}) | core | opentofu-native | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-giantswarm-crossplane]({{< relref "corpus-giantswarm-crossplane" >}}) | core | opentofu-native | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-hongbomiao-harbor]({{< relref "corpus-hongbomiao-harbor" >}}) | core | opentofu-native | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-hongbomiao-labelbox]({{< relref "corpus-hongbomiao-labelbox" >}}) | core | opentofu-native | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-hongbomiao-storage]({{< relref "corpus-hongbomiao-storage" >}}) | core | opentofu-native | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-iam-policy]({{< relref "corpus-iam-policy" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-iam-read-only-policy]({{< relref "corpus-iam-read-only-policy" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-lambda-simple]({{< relref "corpus-lambda-simple" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-leynos-monitoring]({{< relref "corpus-leynos-monitoring" >}}) | core | opentofu-native | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-overture-tiles]({{< relref "corpus-overture-tiles" >}}) | core | opentofu-native | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-rds-complete-postgres]({{< relref "corpus-rds-complete-postgres" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-s3-bucket-complete]({{< relref "corpus-s3-bucket-complete" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-security-group-complete]({{< relref "corpus-security-group-complete" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-simpleinfra-dns]({{< relref "corpus-simpleinfra-dns" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-sqs-basic]({{< relref "corpus-sqs-basic" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-sumaform-aws]({{< relref "corpus-sumaform-aws" >}}) | core | opentofu-native | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-vpc-complete]({{< relref "corpus-vpc-complete" >}}) | core | terraform-popular | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-xancloud-iac]({{< relref "corpus-xancloud-iac" >}}) | core | opentofu-native | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [reference-ec2-vpc]({{< relref "reference-ec2-vpc" >}}) | core | reference | yes | pass pass pass pass pass pass pass pass pass pass pass pass | +| [terralith-scale]({{< relref "terralith-scale" >}}) | core | reference | yes | pass pass pass pass pass pass pass pass pass pass pass not run | +| [corpus-mastino-dns]({{< relref "corpus-mastino-dns" >}}) | growing | published-deployment | yes | pass pass pass pass pass pass pass pass pass pass pass not run | ## Run time -27 of 27 estates have a recorded run duration, totaling 2h46m59s, but not from one sweep: 10m48s across 1 estate(s) at commit `3977d90784`; 6m59.4s across 1 estate(s) at commit `a7ca11f935`; 18m10.8s across 3 estate(s) at commit `cb5ae2009f`; 24m52.9s across 4 estate(s) at commit `d72960cdc3`; 1h46m7.9s across 18 estate(s) at commit `eec6fb4282`. This total spans different commits, not a single board run, and excludes 0 estate(s) with no recorded duration yet. +27 of 27 estates have a recorded run duration, totaling 2h43m25.1s at commit `70e2722fa4`. | Estate | Total | Per-stage (active stages, seconds recorded this run) | |---|---|---| -| [corpus-alb-complete]({{< relref "corpus-alb-complete" >}}) | 7m33.8s | cold_deploy 1m44s, migrate 1m7s, test_plan 4s, test_apply 5s, drift_reconverge 38s, day2_rename 17s, day2_remove 23s, day2_count 1m4s, day2_replace 33s, greenfield 1m38s | -| [corpus-autoscaling-complete]({{< relref "corpus-autoscaling-complete" >}}) | 6m57.3s | cold_deploy 1m45s, migrate 1m21s, test_plan 4s, test_apply 5s, drift_reconverge 8s, day2_rename 18s, day2_remove 21s, day2_count 57s, day2_replace 18s, greenfield 1m39s | -| [corpus-dynamodb-table-basic]({{< relref "corpus-dynamodb-table-basic" >}}) | 4m2.5s | cold_deploy 35s, migrate 1m30s, test_plan 2s, test_apply 3s, drift_reconverge 6s, day2_rename 11s, day2_remove 6s, day2_count 31s, day2_replace 13s, greenfield 45s | -| [corpus-ec2-instance-complete]({{< relref "corpus-ec2-instance-complete" >}}) | 7m2.7s | cold_deploy 1m18s, migrate 30s, test_plan 5s, test_apply 4s, drift_reconverge 6s, day2_rename 17s, day2_remove 30s, day2_count 2m11s, day2_replace 50s, greenfield 1m11s | -| [corpus-ecs-fargate]({{< relref "corpus-ecs-fargate" >}}) | 10m48s | cold_deploy 1m45s, migrate 1m26s, test_plan 29s, test_apply 5s, drift_reconverge 12s, day2_rename 59s, day2_remove 20s, day2_count 2m2s, day2_replace 25s, greenfield 3m4s | -| [corpus-eks-basic]({{< relref "corpus-eks-basic" >}}) | 14m4.9s | cold_deploy 1m35s, migrate 1m59s, test_plan 17s, test_apply 21s, drift_reconverge 39s, day2_rename 1m17s, day2_remove 1m1s, day2_count 3m20s, day2_replace 1m0s, greenfield 2m34s | -| [corpus-evoteum-modules]({{< relref "corpus-evoteum-modules" >}}) | 2m26.6s | cold_deploy 29s, migrate 36s, test_plan 3s, test_apply 3s, drift_reconverge 4s, day2_rename 10s, day2_remove 7s, day2_count 14s, day2_replace 13s, greenfield 27s | -| [corpus-giantswarm-crossplane]({{< relref "corpus-giantswarm-crossplane" >}}) | 2m42.7s | cold_deploy 39s, migrate 24s, test_plan 4s, test_apply 4s, drift_reconverge 5s, day2_rename 11s, day2_remove 7s, day2_count 39s, day2_replace 10s, greenfield 20s | -| [corpus-hongbomiao-harbor]({{< relref "corpus-hongbomiao-harbor" >}}) | 3m28.6s | cold_deploy 27s, migrate 18s, test_plan 2s, test_apply 3s, drift_reconverge 5s, day2_rename 9s, day2_remove 7s, day2_count 43s, day2_replace 6s, greenfield 1m28s | -| [corpus-hongbomiao-labelbox]({{< relref "corpus-hongbomiao-labelbox" >}}) | 3m41s | cold_deploy 55s, migrate 45s, test_plan 4s, test_apply 4s, drift_reconverge 7s, day2_rename 10s, day2_remove 9s, day2_count 23s, day2_replace 9s, greenfield 55s | -| [corpus-hongbomiao-storage]({{< relref "corpus-hongbomiao-storage" >}}) | 3m30.7s | cold_deploy 33s, migrate 43s, test_plan 3s, test_apply 3s, drift_reconverge 6s, day2_rename 15s, day2_remove 8s, day2_count 24s, day2_replace 12s, greenfield 1m3s | -| [corpus-iam-policy]({{< relref "corpus-iam-policy" >}}) | 2m57.6s | cold_deploy 28s, migrate 18s, test_plan 3s, test_apply 3s, drift_reconverge 5s, day2_rename 9s, day2_remove 8s, day2_count 42s, day2_replace 7s, greenfield 55s | -| [corpus-iam-read-only-policy]({{< relref "corpus-iam-read-only-policy" >}}) | 3m9.5s | cold_deploy 25s, migrate 1m5s, test_plan 3s, test_apply 3s, drift_reconverge 5s, day2_rename 10s, day2_remove 7s, day2_count 19s, day2_replace 7s, greenfield 45s | -| [corpus-lambda-simple]({{< relref "corpus-lambda-simple" >}}) | 2m51.8s | cold_deploy 28s, migrate 14s, test_plan 3s, test_apply 5s, drift_reconverge 18s, day2_rename 24s, day2_remove 11s, day2_count 26s, day2_replace 18s, greenfield 25s | -| [corpus-leynos-monitoring]({{< relref "corpus-leynos-monitoring" >}}) | 1m33.9s | cold_deploy 18s, migrate 25s, test_plan 2s, test_apply 2s, drift_reconverge 4s, day2_rename 6s, day2_remove 4s, day2_count 15s, day2_replace 5s, greenfield 13s | -| [corpus-overture-tiles]({{< relref "corpus-overture-tiles" >}}) | 6m59.4s | cold_deploy 56s, migrate 1m10s, test_plan 9s, test_apply 3s, drift_reconverge 7s, day2_rename 47s, day2_remove 53s, day2_count 48s, day2_replace 6s, greenfield 2m0s | -| [corpus-rds-complete-postgres]({{< relref "corpus-rds-complete-postgres" >}}) | 11m51.2s | cold_deploy 1m49s, migrate 50s, test_plan 8s, test_apply 4s, drift_reconverge 8s, day2_rename 16s, day2_remove 1m31s, day2_count 50s, day2_replace 2m51s, greenfield 3m23s | -| [corpus-s3-bucket-complete]({{< relref "corpus-s3-bucket-complete" >}}) | 8m4.5s | cold_deploy 1m29s, migrate 1m17s, test_plan 4s, test_apply 11s, drift_reconverge 20s, day2_rename 23s, day2_remove 19s, day2_count 56s, day2_replace 21s, greenfield 2m45s | -| [corpus-security-group-complete]({{< relref "corpus-security-group-complete" >}}) | 3m30.6s | cold_deploy 31s, migrate 1m15s, test_plan 4s, test_apply 4s, drift_reconverge 6s, day2_rename 13s, day2_remove 10s, day2_count 29s, day2_replace 11s, greenfield 27s | -| [corpus-simpleinfra-dns]({{< relref "corpus-simpleinfra-dns" >}}) | 8m10.4s | cold_deploy 1m25s, migrate 40s, test_plan 5s, test_apply 14s, drift_reconverge 22s, day2_rename 18s, day2_remove 28s, day2_count 53s, day2_replace 1m10s, greenfield 2m35s | -| [corpus-sqs-basic]({{< relref "corpus-sqs-basic" >}}) | 10m36.1s | cold_deploy 1m10s, migrate 2m1s, test_plan 2s, test_apply 3s, drift_reconverge 5s, day2_rename 10s, day2_remove 50s, day2_count 1m54s, day2_replace 1m14s, greenfield 3m7s | -| [corpus-sumaform-aws]({{< relref "corpus-sumaform-aws" >}}) | 9m49.4s | cold_deploy 1m28s, migrate 3m19s, test_plan 11s, test_apply 11s, drift_reconverge 22s, day2_rename 41s, day2_remove 23s, day2_count 2m2s, day2_replace 1m12s, greenfield 2m22s | -| [corpus-vpc-complete]({{< relref "corpus-vpc-complete" >}}) | 4m49.3s | cold_deploy 34s, migrate 1m34s, test_plan 4s, test_apply 4s, drift_reconverge 7s, day2_rename 13s, day2_remove 21s, day2_count 1m32s, day2_replace 20s, greenfield 55s | -| [corpus-xancloud-iac]({{< relref "corpus-xancloud-iac" >}}) | 3m26.9s | cold_deploy 42s, migrate 49s, test_plan 3s, test_apply 3s, drift_reconverge 8s, day2_rename 10s, day2_remove 18s, day2_count 29s, day2_replace 7s, greenfield 38s | -| [reference-ec2-vpc]({{< relref "reference-ec2-vpc" >}}) | 4m20.4s | cold_deploy 1m41s, migrate 52s, test_plan 2s, test_apply 3s, drift_reconverge 4s, day2_rename 9s, day2_remove 6s, day2_count 17s, day2_replace 27s, greenfield 3s, strict 2s | -| [terralith-scale]({{< relref "terralith-scale" >}}) | 7m7.8s | cold_deploy 2m47s, migrate 47s, test_plan 4s, test_apply 6s, drift_reconverge 39s, day2_rename 22s, day2_remove 11s, day2_count 28s, day2_replace 19s, greenfield 1m25s, strict - | -| [corpus-mastino-dns]({{< relref "corpus-mastino-dns" >}}) | 11m21.4s | cold_deploy 2m30s, migrate 55s, test_plan 7s, test_apply 10s, drift_reconverge 29s, day2_rename 26s, day2_remove 32s, day2_count 1m9s, day2_replace 55s, greenfield 4m8s | +| [corpus-alb-complete]({{< relref "corpus-alb-complete" >}}) | 7m16.2s | cold_deploy 1m28s, migrate 1m5s, test_plan 4s, test_apply 5s, drift_reconverge 39s, day2_rename 16s, day2_remove 22s, day2_count 48s, day2_replace 32s, plan_approval 22s, greenfield 1m35s | +| [corpus-autoscaling-complete]({{< relref "corpus-autoscaling-complete" >}}) | 6m52.6s | cold_deploy 1m27s, migrate 1m17s, test_plan 4s, test_apply 4s, drift_reconverge 10s, day2_rename 17s, day2_remove 20s, day2_count 55s, day2_replace 18s, plan_approval 21s, greenfield 1m39s | +| [corpus-dynamodb-table-basic]({{< relref "corpus-dynamodb-table-basic" >}}) | 4m3.8s | cold_deploy 23s, migrate 1m31s, test_plan 2s, test_apply 2s, drift_reconverge 6s, day2_rename 11s, day2_remove 6s, day2_count 31s, day2_replace 13s, plan_approval 13s, greenfield 46s | +| [corpus-ec2-instance-complete]({{< relref "corpus-ec2-instance-complete" >}}) | 6m38.5s | cold_deploy 56s, migrate 30s, test_plan 5s, test_apply 4s, drift_reconverge 6s, day2_rename 14s, day2_remove 28s, day2_count 2m7s, day2_replace 50s, plan_approval 17s, greenfield 1m1s | +| [corpus-ecs-fargate]({{< relref "corpus-ecs-fargate" >}}) | 9m10.4s | cold_deploy 1m22s, migrate 1m18s, test_plan 28s, test_apply 5s, drift_reconverge 11s, day2_rename 52s, day2_remove 15s, day2_count 1m4s, day2_replace 16s, plan_approval 27s, greenfield 2m52s | +| [corpus-eks-basic]({{< relref "corpus-eks-basic" >}}) | 15m20.1s | cold_deploy 1m23s, migrate 1m33s, test_plan 17s, test_apply 22s, drift_reconverge 41s, day2_rename 1m13s, day2_remove 1m2s, day2_count 3m34s, day2_replace 58s, plan_approval 1m40s, greenfield 2m35s | +| [corpus-evoteum-modules]({{< relref "corpus-evoteum-modules" >}}) | 2m22.4s | cold_deploy 12s, migrate 39s, test_plan 3s, test_apply 2s, drift_reconverge 6s, day2_rename 10s, day2_remove 6s, day2_count 13s, day2_replace 13s, plan_approval 12s, greenfield 26s | +| [corpus-giantswarm-crossplane]({{< relref "corpus-giantswarm-crossplane" >}}) | 1m53.3s | cold_deploy 5s, migrate 23s, test_plan 2s, test_apply 3s, drift_reconverge 4s, day2_rename 10s, day2_remove 5s, day2_count 31s, day2_replace 7s, plan_approval 11s, greenfield 12s | +| [corpus-hongbomiao-harbor]({{< relref "corpus-hongbomiao-harbor" >}}) | 3m18.2s | cold_deploy 15s, migrate 17s, test_plan 3s, test_apply 2s, drift_reconverge 5s, day2_rename 8s, day2_remove 7s, day2_count 45s, day2_replace 7s, plan_approval 11s, greenfield 1m18s | +| [corpus-hongbomiao-labelbox]({{< relref "corpus-hongbomiao-labelbox" >}}) | 2m57.5s | cold_deploy 16s, migrate 37s, test_plan 3s, test_apply 3s, drift_reconverge 6s, day2_rename 10s, day2_remove 9s, day2_count 24s, day2_replace 8s, plan_approval 14s, greenfield 48s | +| [corpus-hongbomiao-storage]({{< relref "corpus-hongbomiao-storage" >}}) | 3m34.6s | cold_deploy 20s, migrate 52s, test_plan 4s, test_apply 3s, drift_reconverge 5s, day2_rename 14s, day2_remove 7s, day2_count 24s, day2_replace 12s, plan_approval 15s, greenfield 58s | +| [corpus-iam-policy]({{< relref "corpus-iam-policy" >}}) | 2m46.1s | cold_deploy 17s, migrate 19s, test_plan 2s, test_apply 3s, drift_reconverge 5s, day2_rename 9s, day2_remove 7s, day2_count 40s, day2_replace 7s, plan_approval 12s, greenfield 45s | +| [corpus-iam-read-only-policy]({{< relref "corpus-iam-read-only-policy" >}}) | 3m11.2s | cold_deploy 15s, migrate 1m3s, test_plan 3s, test_apply 2s, drift_reconverge 5s, day2_rename 9s, day2_remove 6s, day2_count 19s, day2_replace 8s, plan_approval 19s, greenfield 41s | +| [corpus-lambda-simple]({{< relref "corpus-lambda-simple" >}}) | 2m53.6s | cold_deploy 13s, migrate 14s, test_plan 2s, test_apply 5s, drift_reconverge 19s, day2_rename 25s, day2_remove 11s, day2_count 28s, day2_replace 18s, plan_approval 15s, greenfield 24s | +| [corpus-leynos-monitoring]({{< relref "corpus-leynos-monitoring" >}}) | 1m30s | cold_deploy 5s, migrate 25s, test_plan 2s, test_apply 2s, drift_reconverge 4s, day2_rename 6s, day2_remove 5s, day2_count 15s, day2_replace 5s, plan_approval 9s, greenfield 11s | +| [corpus-overture-tiles]({{< relref "corpus-overture-tiles" >}}) | 7m23s | cold_deploy 56s, migrate 1m15s, test_plan 10s, test_apply 3s, drift_reconverge 8s, day2_rename 47s, day2_remove 52s, day2_count 51s, day2_replace 7s, plan_approval 13s, greenfield 2m0s | +| [corpus-rds-complete-postgres]({{< relref "corpus-rds-complete-postgres" >}}) | 12m10s | cold_deploy 1m44s, migrate 52s, test_plan 9s, test_apply 5s, drift_reconverge 8s, day2_rename 17s, day2_remove 1m31s, day2_count 50s, day2_replace 2m50s, plan_approval 20s, greenfield 3m23s | +| [corpus-s3-bucket-complete]({{< relref "corpus-s3-bucket-complete" >}}) | 8m31s | cold_deploy 1m13s, migrate 1m21s, test_plan 4s, test_apply 11s, drift_reconverge 20s, day2_rename 22s, day2_remove 20s, day2_count 56s, day2_replace 20s, plan_approval 35s, greenfield 2m49s | +| [corpus-security-group-complete]({{< relref "corpus-security-group-complete" >}}) | 3m38.7s | cold_deploy 18s, migrate 1m14s, test_plan 5s, test_apply 4s, drift_reconverge 7s, day2_rename 14s, day2_remove 11s, day2_count 30s, day2_replace 11s, plan_approval 18s, greenfield 26s | +| [corpus-simpleinfra-dns]({{< relref "corpus-simpleinfra-dns" >}}) | 8m22.7s | cold_deploy 1m10s, migrate 41s, test_plan 5s, test_apply 13s, drift_reconverge 21s, day2_rename 15s, day2_remove 24s, day2_count 50s, day2_replace 1m8s, plan_approval 45s, greenfield 2m31s | +| [corpus-sqs-basic]({{< relref "corpus-sqs-basic" >}}) | 10m32.2s | cold_deploy 57s, migrate 2m0s, test_plan 3s, test_apply 2s, drift_reconverge 5s, day2_rename 9s, day2_remove 49s, day2_count 1m55s, day2_replace 1m14s, plan_approval 13s, greenfield 3m5s | +| [corpus-sumaform-aws]({{< relref "corpus-sumaform-aws" >}}) | 10m26.9s | cold_deploy 1m15s, migrate 3m16s, test_plan 12s, test_apply 11s, drift_reconverge 21s, day2_rename 40s, day2_remove 23s, day2_count 2m3s, day2_replace 1m11s, plan_approval 54s, greenfield 2m25s | +| [corpus-vpc-complete]({{< relref "corpus-vpc-complete" >}}) | 4m58.1s | cold_deploy 24s, migrate 1m36s, test_plan 3s, test_apply 4s, drift_reconverge 7s, day2_rename 13s, day2_remove 20s, day2_count 1m33s, day2_replace 20s, plan_approval 18s, greenfield 56s | +| [corpus-xancloud-iac]({{< relref "corpus-xancloud-iac" >}}) | 3m29.8s | cold_deploy 32s, migrate 48s, test_plan 3s, test_apply 3s, drift_reconverge 7s, day2_rename 11s, day2_remove 18s, day2_count 29s, day2_replace 7s, plan_approval 13s, greenfield 39s | +| [reference-ec2-vpc]({{< relref "reference-ec2-vpc" >}}) | 4m7.5s | cold_deploy 1m27s, migrate 51s, test_plan 2s, test_apply 2s, drift_reconverge 4s, day2_rename 8s, day2_remove 5s, day2_count 17s, day2_replace 26s, plan_approval 11s, greenfield 3s, strict 2s | +| [terralith-scale]({{< relref "terralith-scale" >}}) | 5m42.1s | cold_deploy 2m4s, migrate 43s, test_plan 4s, test_apply 5s, drift_reconverge 33s, day2_rename 18s, day2_remove 7s, day2_count 18s, day2_replace 12s, plan_approval 13s, greenfield 1m5s, strict - | +| [corpus-mastino-dns]({{< relref "corpus-mastino-dns" >}}) | 10m14.6s | cold_deploy 1m55s, migrate 43s, test_plan 5s, test_apply 8s, drift_reconverge 28s, day2_rename 20s, day2_remove 30s, day2_count 57s, day2_replace 46s, plan_approval 28s, greenfield 3m54s | ## Live-AWS certification diff --git a/site/content/docs/progress/corpus-alb-complete.md b/site/content/docs/progress/corpus-alb-complete.md index 4f7d319ba0..ef3041eae3 100644 --- a/site/content/docs/progress/corpus-alb-complete.md +++ b/site/content/docs/progress/corpus-alb-complete.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m44s | 80 resources, once for real (floci fixes #58, #61, #62) | -| Migrate | pass | 1m7s | 51 of 80 stamped, 1 recorded, 0 failed, 28 skipped | +| Cold deploy | pass | 1m28s | 80 resources, once for real (floci fixes #58, #61, #62) | +| Migrate | pass | 1m5s | 51 of 80 stamped, 1 recorded, 0 failed, 28 skipped | | Replan from nothing | pass | 4s | empty live-plan with no state file; 0 Error diagnostics; the two record-rung aws_route53_record.validation identities verified by value against route53 list-resource-record-sets | | No-op apply | pass | 5s | genuine no-op (0 added, 0 changed, 0 destroyed); 50 tofu-estate-tagged objects before, 50 after | -| Drift and reconverge | pass | 38s | one object tampered (the ALB's Example tag), plan proposed fixing exactly module.alb.aws_lb.this[0], apply changed 1 and the Example tag reconverged | -| Rename | pass | 17s | moved block: aws_instance.this renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_instance.other renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state (positioned right after stage 1, before migrate ever touches these shared objects) also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 23s | choudoufu: deleting aws_instance.other_renamed's block (and its one target-group-attachment reference) proposed exactly 2 destroys (0 add, 0 change, 2 destroy), matching the stock oracle and applied cleanly; the instance is confirmed terminated via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 2 destroys for the same two objects | -| Change count | pass | 1m4s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.) and left count_test[0] untouched; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.); the next plan is empty. Read back through the AWS CLI, never through choudoufu's own report: the destroyed count_test[1] (sg-b60954438b88ecf3d) is absent after the scale-down and comes back under a NEW server-minted GroupId (sg-2e48714be247a863a) carrying tofu-address=aws_security_group.count_test:1, while the survivor count_test[0] keeps BOTH its GroupId (sg-7cd7717aa835c118a) and its tofu-address=aws_security_group.count_test:0 marker across both moves. What witnesses the destroy was measured against the pinned emulator with no terraform in the loop before this section was written: floci sha256:c55d74e1 never reuses a deleted group's GroupId, not even for the same group-name in the same VPC. G-ORACLE is a real stock oracle for the same shape - stock never had this count block, so it was stood up for real with the stock terraform binary in its own working directory and state against its own idle account on :30400 - and stock shows the IDENTICAL shape: destroy the higher index only (sg-0b3d6a90645f55505, then absent), create it back under a new id (sg-11229066e226420ec), count_test[0]=sg-b162ecb6edc79508e unchanged throughout (stock's own plan lines: "Plan: 0 to add, 0 to change, 1 to destroy." down, "Plan: 1 to add, 0 to change, 0 to destroy." up). SYNTHETIC BLOCK, and why: this estate declares no scalable count of its own - examples/complete-alb has no root-level count or for_each at all, terraform-aws-alb's only count is its `local.create ? 1 : 0` boolean create toggle, and everything the module fans out (6 listeners, 7 listener rules, 3 target groups, 2 security-group ingress rules) is for_each over a STRING-KEYED map, which has no index whose slot binding could be scaled and over which "the higher index is destroyed" is not expressible - so the sanctioned fallback applies (reference-ec2-vpc Part F, corpus-iam-policy Part G), with aws_security_group, a type this estate already exercises. BREAK_COUNT=1 confirms the check has teeth: asserting the WRONG instance (count_test[0]) was destroyed makes this stage report fail. One emulator divergence noted in passing, harmless to this stage and asserted around rather than papered over: floci answers DescribeSecurityGroups for a deleted group-id with an empty list and exit 0, where real EC2 raises InvalidGroup.NotFound. | -| Replace with create_before_destroy | pass | 33s | choudoufu: changing aws_instance.this_renamed's ForceNew ami argument proposed a forced replace at the same declared address (Plan: 2 to add, 0 to change, 2 to destroy.), applied cleanly; the old instance (i-c13744b209879be03) is confirmed terminated and the new instance (i-574774864b2064b60) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-c13744b209879be03 -> i-574774864b2064b60); the next plan proposes no resource action; stock oracle on cold_deploy's own state (day2_replace ORACLE) also proposes replacing aws_instance.this at the same address (plan only, not applied - it shares floci's account with $ADOPTED_EST); BREAK=replace confirms a manufactured marker collision is reported loudly ("Two live resources claiming one address", naming both live instances) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | +| Drift and reconverge | pass | 39s | one object tampered (the ALB's Example tag), plan proposed fixing exactly module.alb.aws_lb.this[0], apply changed 1 and the Example tag reconverged | +| Rename | pass | 16s | moved block: aws_instance.this renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_instance.other renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state (positioned right after stage 1, before migrate ever touches these shared objects) also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Remove a block | pass | 22s | choudoufu: deleting aws_instance.other_renamed's block (and its one target-group-attachment reference) proposed exactly 2 destroys (0 add, 0 change, 2 destroy), matching the stock oracle and applied cleanly; the instance is confirmed terminated via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 2 destroys for the same two objects | +| Change count | pass | 48s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.) and left count_test[0] untouched; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.); the next plan is empty. Read back through the AWS CLI, never through choudoufu's own report: the destroyed count_test[1] (sg-c58a962fd8cab5c67) is absent after the scale-down and comes back under a NEW server-minted GroupId (sg-942bc41c28cb76287) carrying tofu-address=aws_security_group.count_test:1, while the survivor count_test[0] keeps BOTH its GroupId (sg-50717691c349face1) and its tofu-address=aws_security_group.count_test:0 marker across both moves. What witnesses the destroy was measured against the pinned emulator with no terraform in the loop before this section was written: floci sha256:c55d74e1 never reuses a deleted group's GroupId, not even for the same group-name in the same VPC. G-ORACLE is a real stock oracle for the same shape - stock never had this count block, so it was stood up for real with the stock terraform binary in its own working directory and state against its own idle account on :20400 - and stock shows the IDENTICAL shape: destroy the higher index only (sg-6a3ec8b4a38b489e9, then absent), create it back under a new id (sg-6c6c45f278c454454), count_test[0]=sg-69fe659703f0b86cb unchanged throughout (stock's own plan lines: "Plan: 0 to add, 0 to change, 1 to destroy." down, "Plan: 1 to add, 0 to change, 0 to destroy." up). SYNTHETIC BLOCK, and why: this estate declares no scalable count of its own - examples/complete-alb has no root-level count or for_each at all, terraform-aws-alb's only count is its `local.create ? 1 : 0` boolean create toggle, and everything the module fans out (6 listeners, 7 listener rules, 3 target groups, 2 security-group ingress rules) is for_each over a STRING-KEYED map, which has no index whose slot binding could be scaled and over which "the higher index is destroyed" is not expressible - so the sanctioned fallback applies (reference-ec2-vpc Part F, corpus-iam-policy Part G), with aws_security_group, a type this estate already exercises. BREAK_COUNT=1 confirms the check has teeth: asserting the WRONG instance (count_test[0]) was destroyed makes this stage report fail. One emulator divergence noted in passing, harmless to this stage and asserted around rather than papered over: floci answers DescribeSecurityGroups for a deleted group-id with an empty list and exit 0, where real EC2 raises InvalidGroup.NotFound. | +| Replace with create_before_destroy | pass | 32s | choudoufu: changing aws_instance.this_renamed's ForceNew ami argument proposed a forced replace at the same declared address (Plan: 2 to add, 0 to change, 2 to destroy.), applied cleanly; the old instance (i-c6b2f684d28bd4e21) is confirmed terminated and the new instance (i-3079cc0afccd4830b) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-c6b2f684d28bd4e21 -> i-3079cc0afccd4830b); the next plan proposes no resource action; stock oracle on cold_deploy's own state (day2_replace ORACLE) also proposes replacing aws_instance.this at the same address (plan only, not applied - it shares floci's account with $ADOPTED_EST); BREAK=replace confirms a manufactured marker collision is reported loudly ("Two live resources claiming one address", naming both live instances) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 1m38s | 80 resources from nothing, matching stock's own cold-deploy count; the ALB's markers verified via the AWS CLI; 80 records in the local record store including untaggable types; replan empty; a representative EC2 instance's own shape (type/ami) matches stock's cold deploy, via the AWS CLI on both endpoints, marker tags never compared; 50 objects carry the estate tag | +| Plan, review, apply | pass | 22s | one argument edited (the "ex-instance" target group's own InstanceTargetGroupTag tag, baz -> reviewed - the one tags argument in examples/complete-alb that reaches exactly one instance, with no dependent resource or data source behind it), "plan -out=approved.tfplan" wrote a 159425-byte stock-format plan file whose whole change set is one update on module.alb.aws_lb_target_group.this["ex-instance"]; the world then moved out of band (arn:aws:elasticloadbalancing:eu-west-1:000000000000:loadbalancer/app/ex-complete-alb/ca7ef1d7df57446c's Example tag, through the AWS CLI, never through choudoufu - the same mutation stage 5 uses) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.alb.aws_lb.this[0] and the live arn:aws:elasticloadbalancing:eu-west-1:000000000000:loadbalancer/app/ex-complete-alb/ca7ef1d7df57446c it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - arn:aws:elasticloadbalancing:eu-west-1:000000000000:targetgroup/h115cad549492d8241963b49d84a/02c64f3ef264487c still read InstanceTargetGroupTag=baz through elbv2 describe-tags, not from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the ALB's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:elasticloadbalancing:eu-west-1:000000000000:targetgroup/h115cad549492d8241963b49d84a/02c64f3ef264487c read back with InstanceTargetGroupTag=reviewed, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 1m35s | 80 resources from nothing, matching stock's own cold-deploy count; the ALB's markers verified via the AWS CLI; 80 records in the local record store including untaggable types; replan empty; a representative EC2 instance's own shape (type/ami) matches stock's cold deploy, via the AWS CLI on both endpoints, marker tags never compared; 50 objects carry the estate tag | | Strict profile (not a headline stage) | not run | | | -Last run at commit `cb5ae2009f` on 2026-09-06T04:29:56Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 7m33.8s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 7m16.2s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed 2c654caa23 (2026-08-18), 80 resources. cold_deploy and migrate genuinely pass (48 of 80 eligible, 44 stamped - 4 failed to stamp on a real, filed floci gap, lex00/floci#65: ELBv2 DescribeListeners/DescribeRules drop AuthenticateCognitoConfig/AuthenticateOidcConfig entirely, so those 4 listener-rule sites can't round-trip; 32 skipped - 28 untaggable by design, 4 unadmitted). test_plan blocked by #305 (3 sites) and a NEW one, filed as #309: aws_cognito_user_pool_client is unadmitted despite a parent-scoped ListUserPoolClients existing (via the already-admitted aws_cognito_user_pool parent) - a concrete lead for a fix, not attempted here. Three real floci gaps found, fixed, and merged to floci main (9ff82512, already reconciled into that history - not a pending PR): #58 (ACM wildcard-SAN DNS validation record kept a literal "*." in its name), #61 (S3 log-delivery-write canned ACL rejected), #62 (EC2 instance-type catalog missing t3.nano). One floci gap found and left open as real feature work: #63 (Cognito CreateUserPoolDomain entirely unimplemented), worked around with a documented delta. Deliberately did NOT re-pin live/floci-image here - bumping the pin without a matching floci-capabilities.json regen breaks TestFlociServiceCapability/TestFlociTypeCapability, and several other crossings tonight need the same regen, so this was left for a single centralized reconciliation rather than each crossing re-pinning independently and colliding. Verified twice via FLOCI_IMAGE override pointed at the real published digest (sha256:217f859688b5...) instead. Follow-up pass 2026-08-18 (#313 cross-check, d7e44cbfda/29095a37b7): the committed run.sh still asserted pre-#305 counts (48/80 eligible, 44 stamped, 4 unadmitted-type including the default_* trio) and failed before ever reaching stage 3; updated to the real current numbers - 51 eligible, 47 stamped, #309 (aws_cognito_user_pool_client) now the SOLE unadmitted-type/test_plan blocker, #305's trio fully resolved - re-verified against real floci twice plus a BREAK=1 negative control. #313's diagnostic does not appear anywhere in the output despite this estate also declaring data.aws_availability_zones feeding module.vpc; confirmed not the same wall. Commented on #313 and #309. RE-VERIFIED 2026-08-21 (this session, worktree live/reverify-limitations, floci pinned at the current e61a987/d65baf42 image): stages 1-2 unchanged byte-for-byte (80 cold-deployed; dry run 51/80 eligible, 41 VERIFIED + 10 DRIFTED, 29 skipped; -approve 47 stamped, 1 recorded, 4 failed on the still-open lex00/floci#65, 28 skipped). Stage 3's underlying wall is UNCHANGED in substance - still exactly one site, aws_cognito_user_pool_client.this, still #309, still label 3 - but its diagnostic TEXT has moved: commit 5444949ceb (2026-08-19 22:57, "widen the markerless veto to a mixed primary identifier", #309's own chain) added aws_cognito_user_pool_client to identity.MarkerlessTypes, which reclassifies the refusal from RuleUnadmittedType ("Resource type is outside the live-markers subset") to the more specific RuleMarkerlessType ("Resource type has nowhere to write an ownership marker") - same site, same practical block, different rule ID and error string. That commit landed after run.sh (c926095b49) was last edited, so run.sh's own hard-coded stage-3 assertion (grep for the old RuleUnadmittedType text) no longer matches and the script now hard-fails at its own assertion check ("expected 1 unadmitted-type sites (#309), got 0") before ever reaching its SUMMARY block, rather than reaching a clean itemized stage-3 refusal. Stage 2's live-import output still labels the same site UNADMITTED_TYPE, so stage 2 and stage 3 now use inconsistent labels for the identical site - a real, if narrow, drift worth fixing when someone next touches this script (update the grep target to RuleMarkerlessType's text), not attempted here per this session's verification-only scope. Separately confirmed: run.sh's header (lines 92-131) still does not state the maintainer's 2026-08-21 "Label 3, OpenTofu was never asked this question, stop here" ruling in as many words - HANDOFF.md's "What to do next" item 2 documentation task is still open. #337 (composite-identity route, 18 types) is confirmed NOT implicated - the responsible commit is on the #309 chain itself, predating #337. #331/#340/#343/#344 did not touch this estate. diff --git a/site/content/docs/progress/corpus-autoscaling-complete.md b/site/content/docs/progress/corpus-autoscaling-complete.md index 36b1303ffe..c757d8425c 100644 --- a/site/content/docs/progress/corpus-autoscaling-complete.md +++ b/site/content/docs/progress/corpus-autoscaling-complete.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m45s | Apply complete! Resources: 68 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=autoscaling-complete-crossing before migration | -| Migrate | pass | 1m21s | 41 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 re-recorded for sensitivity only, 0 already recorded, 0 failed, 27 skipped; 41 objects carry tofu-estate=autoscaling-complete-crossing | +| Cold deploy | pass | 1m27s | Apply complete! Resources: 68 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=autoscaling-complete-crossing before migration | +| Migrate | pass | 1m17s | 41 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 re-recorded for sensitivity only, 0 already recorded, 0 failed, 27 skipped; 41 objects carry tofu-estate=autoscaling-complete-crossing | | Replan from nothing | pass | 4s | empty plan; identity re-check unchanged: module.complete.aws_launch_template.this:0, aws_iam_role.ssm | -| No-op apply | pass | 5s | genuine no-op: 41 objects before, 41 after, no state file either time | -| Drift and reconverge | pass | 8s | one object tampered (SQS queue 'complete's Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag | -| Rename | pass | 18s | moved block: module.asg_sg renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place on its security group; live-mv: aws_sqs_queue.this renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 21s | choudoufu: deleting module.default's block proposed exactly 2 destroys (0 add, 0 change, 2 destroy), matching the stock oracle's own count and applied cleanly; the live ASG count dropped by exactly one and the tagged object count dropped too, both confirmed via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 2 destroys for the same module | -| Change count | pass | 57s | choudoufu: scaling aws_security_group.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) and it was the HIGHER index, count_test[1] (sg-751e8f493815cc85a); count_test[0] (sg-278ee188356fcd904) kept its live GroupId, its tofu-address=aws_security_group.count_test:0 marker and its local record across the move, all re-read through the AWS CLI and off the record-store file rather than out of choudoufu's own report. Scaling back from 1 to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy) and count_test[1] came back as a genuinely NEW object (sg-751e8f493815cc85a -> sg-03e3220bce1d89e1b) carrying tofu-address=aws_security_group.count_test:1, with its record re-keyed to the new id rather than left stale on the destroyed one; count_test[0] was untouched throughout and the next plan is empty. Stock oracle (G0): a plain terraform working directory standing the identical 2-instance block up for real in its own VPC and scaling it the same way shows the identical shape - destroy count_test[1] (sg-197c7dafd5ec22f07) only on the way down, create count_test[1] back under a new id (sg-60fdd9f4aa9f57fce) on the way up, count_test[0] (sg-ac913a1b237fb74e4) untouched both times - and was torn down before the choudoufu leg ran. SYNTHETIC BLOCK, and why: this estate declares no scalable count anywhere - the module's own aws_autoscaling_group.this/.idc pair is a 0-or-1 boolean toggle, the schedules/policies/traffic-source attachments are for_each over name-keyed maps, and none of the twelve module calls carries count or for_each - so this is live/GAUNTLET.md #8's sanctioned self-contained fallback, the same one reference-ec2-vpc's Part F and corpus-iam-policy's Part G use, on aws_security_group, a type this estate already exercises through module.asg_sg. An ASG's own desired_capacity is deliberately NOT what was scaled: this stage is about the count meta-argument's slot binding (internal/live/discovery/count.go), not about a live group's capacity. BREAK_COUNT=1 confirms the check has teeth: expecting count_test[0] to be the destroyed instance makes this stage report fail. | -| Replace with create_before_destroy | pass | 18s | choudoufu: changing module.asg_sg_renamed's ForceNew name argument (module CALL, passed through to its own aws_security_group.this_name_prefix's name_prefix) proposed a forced replace at the same declared address (Plan: 3 to add, 2 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-c37bb9262afc41e48) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-05ed298dde97955f0 -> sg-c37bb9262afc41e48); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 2 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly ("Two live resources claiming one slot") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted aws_sqs_queue.this_renamed and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on this branch, GitHub issue #412: propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard; see this section's own header comment for the fix and eks-basic's/ecs-fargate's matching ones in this same unit, which independently hit the identical shape and were not re-run for #412. | +| No-op apply | pass | 4s | genuine no-op: 41 objects before, 41 after, no state file either time | +| Drift and reconverge | pass | 10s | one object tampered (SQS queue 'complete's Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag | +| Rename | pass | 17s | moved block: module.asg_sg renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place on its security group; live-mv: aws_sqs_queue.this renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Remove a block | pass | 20s | choudoufu: deleting module.default's block proposed exactly 2 destroys (0 add, 0 change, 2 destroy), matching the stock oracle's own count and applied cleanly; the live ASG count dropped by exactly one and the tagged object count dropped too, both confirmed via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 2 destroys for the same module | +| Change count | pass | 55s | choudoufu: scaling aws_security_group.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) and it was the HIGHER index, count_test[1] (sg-be0d8c436b598d206); count_test[0] (sg-87554ef6a11f83aab) kept its live GroupId, its tofu-address=aws_security_group.count_test:0 marker and its local record across the move, all re-read through the AWS CLI and off the record-store file rather than out of choudoufu's own report. Scaling back from 1 to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy) and count_test[1] came back as a genuinely NEW object (sg-be0d8c436b598d206 -> sg-530659bb51950b409) carrying tofu-address=aws_security_group.count_test:1, with its record re-keyed to the new id rather than left stale on the destroyed one; count_test[0] was untouched throughout and the next plan is empty. Stock oracle (G0): a plain terraform working directory standing the identical 2-instance block up for real in its own VPC and scaling it the same way shows the identical shape - destroy count_test[1] (sg-b3bf410755a7c74b5) only on the way down, create count_test[1] back under a new id (sg-55b912566a9fb1171) on the way up, count_test[0] (sg-69115075c0a480bab) untouched both times - and was torn down before the choudoufu leg ran. SYNTHETIC BLOCK, and why: this estate declares no scalable count anywhere - the module's own aws_autoscaling_group.this/.idc pair is a 0-or-1 boolean toggle, the schedules/policies/traffic-source attachments are for_each over name-keyed maps, and none of the twelve module calls carries count or for_each - so this is live/GAUNTLET.md #8's sanctioned self-contained fallback, the same one reference-ec2-vpc's Part F and corpus-iam-policy's Part G use, on aws_security_group, a type this estate already exercises through module.asg_sg. An ASG's own desired_capacity is deliberately NOT what was scaled: this stage is about the count meta-argument's slot binding (internal/live/discovery/count.go), not about a live group's capacity. BREAK_COUNT=1 confirms the check has teeth: expecting count_test[0] to be the destroyed instance makes this stage report fail. | +| Replace with create_before_destroy | pass | 18s | choudoufu: changing module.asg_sg_renamed's ForceNew name argument (module CALL, passed through to its own aws_security_group.this_name_prefix's name_prefix) proposed a forced replace at the same declared address (Plan: 3 to add, 2 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-091e8a0748dde3c56) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-e756cc13263688dd3 -> sg-091e8a0748dde3c56); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 2 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly ("Two live resources claiming one slot") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted aws_sqs_queue.this_renamed and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on this branch, GitHub issue #412: propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard; see this section's own header comment for the fix and eks-basic's/ecs-fargate's matching ones in this same unit, which independently hit the identical shape and were not re-run for #412. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | +| Plan, review, apply | pass | 21s | one argument edited (aws_iam_role.ssm's tags gain Reviewed=yes - a single instance whose only dependent reads its name, which an in-place tag update leaves known), "plan -out=approved.tfplan" wrote a 275504-byte stock-format plan file whose whole change set is one update on aws_iam_role.ssm; the world then moved out of band (the SQS queue's Example tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both aws_sqs_queue.this and the live https://sqs.eu-west-1.amazonaws.com/000000000000/complete it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - IAM role complete still carried no Reviewed tag, read back through iam list-role-tags rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the queue's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the role read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | | Greenfield apply | pass | 1m39s | 68 resources from nothing, matching stock's own cold-deploy count (68); the sqs queue's markers verified via the AWS CLI; 68 records in the local record store including the untaggable ASGs (#364 A2); replan empty; the asg_sg security group's rule counts match stock's cold deploy structurally, via the AWS CLI on both endpoints, marker tags never compared; 41 objects carry the estate tag | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 6m57.3s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 6m52.6s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. RE-CROSSED FOR REAL 2026-08-21 at dbc5ebf575 (GitHub issue #346's fix), real Docker/floci/terraform/AWS CLI throughout. STAGES UNCHANGED at 2 of 5 and BOTH stage-3 diagnostics are unchanged, verbatim - #346's fix does not move this estate. Stage 1: 'Apply complete! Resources: 68 added, 0 changed, 0 destroyed'. Stage 2: '41 of 68 eligible'; '41 stamped, 0 already stamped, 0 newly recorded, 0 failed, 27 skipped'; markers asserted by value through the AWS CLI (launch template lt-0e04ccd955a0a5071 = module.complete.aws_launch_template.this:0, IAM role 'complete' = aws_iam_role.ssm), and the ASG itself confirmed carrying 0 tofu-* tags as the marker vocabulary requires. Stage 3 fails on exactly 2: (1) 'Non-static identity argument' on module.complete.aws_autoscaling_traffic_source_attachment.this['ex-alb'].identifier (main.tf:1102, identifier = each.value.traffic_source_identifier), same file, same line, same text as before; (2) 'Ambiguous list-valued identity argument' on module.asg_sg.aws_security_group_rule.computed_ingress_with_source_security_group_id[0].prefix_list_ids, which 'has 0 elements' - the Component.SoleElement refusal, deliberate per its own registry text and never in scope for #346. Diagnostic (1) sits on the each.value route #346's fix does touch, but it refuses at a different point: 'Non-static identity argument' is stringValueIn's !IsWhollyKnown branch, so the value EVALUATED and came back unknown rather than the selection refusing. The unknown is the ALB target group's ARN arriving through a tolerant module-argument rebuild - the same value-shaped route corpus-rds-complete-postgres and corpus-ecs-fargate hit, reached from the other side. Filed with them. PRIOR HISTORY BELOW. RE-CROSSED 2026-08-21 TWICE, at commit 7aea0eef95 (GitHub issue #353, provisioners admitted under a record_store), worktree live/provisioner-support-353, real Docker/floci/terraform/AWS CLI throughout, image ghcr.io/lex00/floci@sha256:d65baf42, hashicorp/aws 6.59.0. Both runs are this script's own output, not inferred. STAGES ARE UNCHANGED at 2 of 5 - cold_deploy pass, migrate pass, test_plan fail - and that is the honest headline: #353 removed this estate's SOLE stage-3 diagnostic and did not move its stage outcome, because two more stand behind it. What changed, measured: RUN 1 (the script exactly as it stood, live block declaring only the estate name): stage 3 fails on exactly ONE diagnostic, 'Provisioners are not available under live resource markers' on aws_iam_service_linked_role.autoscaling (main.tf:889, the example's own 'sleep 10'). RUN 2 (the script's migration edit now also declares record_store "local" - the supported way past that wall, and what an operator migrating this estate would actually write; the delta is documented in the script's own header): that diagnostic is GONE and stage 3 fails on exactly TWO, which are the two the 2026-08-20 pass predicted by measuring past the wall on a scratch copy, now confirmed for real. (1) 'Non-static identity argument' on module.complete.aws_autoscaling_traffic_source_attachment.this["ex-alb"].identifier (main.tf:1102, identifier = each.value.traffic_source_identifier, reaching module.complete_alb's target-group ARN) - #346's wall, one spelling further out than the corpus-vpc-complete form, unresolved by maintainer decision and needing a design pass rather than code. (2) 'Ambiguous list-valued identity argument' on module.asg_sg.aws_security_group_rule.computed_ingress_with_source_security_group_id[0].prefix_list_ids, which 'has 0 elements' - the Component.SoleElement refusal, deliberate per its own registry text. Stages 1 and 2 are byte-for-byte the same in both runs: 'Apply complete! Resources: 68 added, 0 changed, 0 destroyed' from 0 pre-existing marked objects; '41 of 68 resource instance(s) are eligible for stamping'; '41 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 27 skipped'; 41 objects carrying tofu-estate afterwards. Declaring the record_store did not change the migrate summary line at all, which is the expected answer - this estate has no record-backed type, so the store's only job here is to hold a provisioner's tainted bit if one ever fails. test_apply and drift_reconverge stay not_run: running them against a refused plan would prove nothing. PRIOR HISTORY BELOW. RE-CROSSED 2026-08-20 against the #84 pin, worktree live/floci-warmpool-recross, real Docker/floci/terraform/AWS CLI throughout, image ghcr.io/lex00/floci@sha256:120b6783 (the b9e5a593 PutWarmPool build already pinned in live/floci-image). Every figure below was computed on a base of local main dd22a50669, and the whole script was re-run there after the rebase with an identical result - not carried over from the pre-rebase run. lex00/floci#84 is CONFIRMED FIXED: module.warm_pool's warm_pool block no longer errors, and this estate moves 0 of 5 to 2 of 5. Numbers read off the script's own lines, not inferred: STAGE 1 'Apply complete! Resources: 68 added, 0 changed, 0 destroyed' with 0 objects carrying tofu-estate beforehand; STAGE 2 dry run '41 of 68 resource instance(s) are eligible for stamping', -approve '41 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 27 skipped', 41 objects carrying tofu-estate afterwards. All 27 skips sit under UNTAGGABLE with no UNADMITTED_TYPE section at all - 8 aws_autoscaling_group, 6 aws_route_table_association, 5 aws_autoscaling_policy, 3 aws_autoscaling_schedule, 2 aws_security_group_rule, and one each of aws_autoscaling_traffic_source_attachment, aws_iam_role_policy_attachment and aws_route - the invariant working rather than a gap. THREE SELF-INFLICTED SCRIPT BUGS FOUND AND FIXED, none of which had ever run because stage 1 had never passed: (1) stage 2 asserted a tofu-address TAG on ASG 'complete', which can never exist - an aws_autoscaling_group's tags are `tag` nested blocks, not the top-level tags map internal/live/markers.TagSurface requires, so Taggable() refuses it from the schema and markers.go's own doc comment names this exact type as the shape it will not stamp; the assertion also read through `autoscaling describe-tags`, which floci does not implement at all ('UnsupportedOperation: Operation DescribeTags is not supported'), with the CLI error swallowed by a 2>/dev/null and read back as an empty tag. Fixed in the harness, not in floci, same call as #345: the ASG's identity is now asserted out of live-import's own UNTAGGABLE row, which prints the resolved address beside the live id it bound to ('module.complete.aws_autoscaling_group.idc[0] ... live id: complete'), plus a separate assertion that the live ASG carries ZERO tofu-* tags, since a marker written into a tag block would be the wrong-marker failure this repository ranks above a missing one. That row is also what now exercises the count-is-zero-per-instance fix for real - .idc[0] resolves, .this resolves to no instance at all. (2) the launch-template assertion looked up `describe-launch-templates --launch-template-names complete`, but module.complete builds its launch template from a name_prefix, so the live name carries a provider-minted suffix ('complete-53747ea039642590a26cff220b') and the lookup returned None; it now finds the object BY ITS MARKER (ec2 describe-tags on key=tofu-address, value=module.complete.aws_launch_template.this:0) and asserts the resulting name starts with 'complete-'. (3) both that assertion and the IAM-role one compared a tag value against OpenTofu's BRACKET spelling when a tag value can never carry '[' - the escaped form is ':0' per internal/live/markers.EscapeKey - and the role assertion additionally had the wrong resource: the role named exactly 'complete' is the example's ROOT-LEVEL aws_iam_role.ssm (name = local.name), not module.complete's own role, which lands as 'complete-5c82d83cf70f48587a1e0817dc'. Same vacuous-comparison bug corpus-vpc-complete, corpus-iam-policy and corpus-iam-read-only-policy each shipped once; bracket and colon forms are now separate variables, with the bracket form used only where stage 5 reads a plan diff header. BREAK=1 re-run confirms the rewritten role assertion is still load-bearing (fails on 'the IAM role carries tofu-address=aws_iam_role.ssm, not module.complete.aws_iam_role.this:0'). A REAL CHOUDOUFU CRASH found and fixed on the way, not a script bug: running live-import a SECOND time against the already-stamped estate - a supported thing to do, 'already stamped' is one of the outcomes the summary line counts - panicked with 'value is null' at internal/live/liveimport/tags.go, equivalent(). Its zeroish allowance forgives null-against-empty, but null against a POPULATED collection fell through to the collection arms, which call LengthInt/ElementIterator/GetAttr, all of which panic on null, reaching the operator as an OpenTofu crash report instead of a diagnostic. Guarded, with internal/live/liveimport/tags_test.go pinning both the crash cases and the null-vs-empty allowance the guard must not eat; the re-run now completes and correctly reports DRIFTED (41), the markers the state file does not know about. STAGE 3 fails on EXACTLY ONE diagnostic: provisioner "local-exec" on aws_iam_service_linked_role.autoscaling (main.tf:889 - the example's own 'sleep 10' comment says 'Sometimes good sleep is required to have some IAM resources created before they can be used'). PARITY LABEL, stated first: 'OpenTofu succeeds, choudoufu refuses'. Stock terraform ran this provisioner in stage 1 of this very script, so the label is not in doubt. It is NOT filed as a new defect, because it is already enumerated in live/LIMITATIONS.md under 'Enforced today' with a forwarding address (RuleProvisioner, internal/live/lint/lint.go's checkProvisioners, fixture live/e2e/limits/local-exec/): a provisioner runs an effect, and whether it already ran is knowable only from a stored record, which is exactly the authority live markers give up. What IS new information is the cost: this is the first POPULAR, real, terraform-aws-modules estate blocked by that ban outright, versus sumaform's, which was dead code (a provisioner-less connection block). Whether the store test should stay absolute for a provisioner whose effect is a sleep is a maintainer call and was not forced here. MEASURED PAST THE WALL, on a scratch copy with the provisioner block deleted - a diagnostic only, deliberately NOT a delta in the landed script, which routes around nothing: exactly two further diagnostics stand behind it, no more. First, 'Non-static identity argument' on module.complete.aws_autoscaling_traffic_source_attachment.this["ex-alb"].identifier - each.value.traffic_source_identifier is module.complete_alb's target group ARN, a non-identity attribute of another managed resource, which is the same wall #346 is filed for on corpus-vpc-complete, one spelling further out. Second, 'Ambiguous list-valued identity argument' on module.asg_sg.aws_security_group_rule.computed_ingress_with_source_security_group_id[0].prefix_list_ids, which 'has 0 elements' - the Component.SoleElement refusal whose own registry text (internal/live/identity/refusals.go) already says zero elements and more than one are both refused, so it is deliberate rather than a new find; the zero case is arguably distinguishable from the many case and was left alone rather than widened on this pass. So even with the provisioner gone this estate would still fail stage 3, on two identity walls neither of which is specific to autoscaling. test_apply and drift_reconverge remain not_run: attempting them against a refused plan would prove nothing. PRIOR HISTORY - Landed 521941f77e (2026-08-18), 80 resources across 12 module calls - the largest crossing tonight (mixed instances, warm pools, EFA interfaces, attribute-based instance requirements, external launch templates, lifecycle hooks, scheduled actions, scaling policies, the count-is-zero-per-instance admission fix genuinely exercised for real via the ignore_desired_capacity_changes-gated aws_autoscaling_group.this/.idc pair). lex00/floci#64 (CreateCapacityReservation/DescribeCapacityReservations/ModifyCapacityReservation/CancelCapacityReservation entirely unimplemented) is now FIXED and merged to floci main (1ad8d39b), CI and GHCR-publish both green as of e970328 (published sha256:dcb25c00), and live/floci-image is re-pinned to it with the capability manifest regenerated to match (749 types, same shape as the prior digest). Re-ran this script for real 2026-08-20 against the new pin: aws_ec2_capacity_reservation.targeted no longer errors - #64 confirmed fixed. cold_deploy is still recorded fail because a SECOND, previously-masked gap surfaces immediately after: module.warm_pool's aws_autoscaling_group.this[0] declares a warm_pool block, and PutWarmPool is entirely unimplemented in floci's AutoScaling service (grepped AutoScalingQueryHandler/AutoScalingService/model - zero occurrences of "WarmPool") - 'UnsupportedOperation: Operation PutWarmPool is not supported.' This was masked until now because CreateCapacityReservation failed first in the same cold-deploy run, on a different resource earlier in the module graph. Filed as lex00/floci#84 (PutWarmPool/DescribeWarmPool/DeleteWarmPool), verified against botocore's own autoscaling/2011-01-01/service-2.json wire model - not attempted in this session, a second AutoScaling-side feature comparable in size to #64, left as a clean follow-up. #84 is now the sole cold_deploy blocker for this estate; #310 (the traffic-source-attachment admission gap, since fixed per its own issue - see that issue for verification) and the rest of this crossing's stage-3+ picture remain unverified against a real cold deploy until #84 lands. diff --git a/site/content/docs/progress/corpus-dynamodb-table-basic.md b/site/content/docs/progress/corpus-dynamodb-table-basic.md index 4256a897c2..ff1c377c62 100644 --- a/site/content/docs/progress/corpus-dynamodb-table-basic.md +++ b/site/content/docs/progress/corpus-dynamodb-table-basic.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 35s | Apply complete! Resources: 3 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=dynamodb-table-basic-crossing before migration | -| Migrate | pass | 1m30s | 1 resource(s) newly stamped, 0 already stamped, 1 newly recorded, 0 re-recorded for sensitivity only, 0 already recorded, 0 failed, 1 skipped.; Apply complete! Resources: 0 added, 0 changed, 1 destroyed. (tofu-slot convergence) | +| Cold deploy | pass | 23s | Apply complete! Resources: 3 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=dynamodb-table-basic-crossing before migration | +| Migrate | pass | 1m31s | 1 resource(s) newly stamped, 0 already stamped, 1 newly recorded, 0 re-recorded for sensitivity only, 0 already recorded, 0 failed, 1 skipped.; Apply complete! Resources: 0 added, 0 changed, 1 destroyed. (tofu-slot convergence) | | Replan from nothing | pass | 2s | no resource change proposed, nothing foreign; identity re-check (via the AWS CLI) unchanged | -| No-op apply | pass | 3s | genuine no-op: 1 objects before, 1 after, no state file either time | -| Drift and reconverge | pass | 6s | one object tampered (arn:aws:dynamodb:eu-west-1:000000000000:table/my-table-clear-horse's Terraform tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag | +| No-op apply | pass | 2s | genuine no-op: 1 objects before, 1 after, no state file either time | +| Drift and reconverge | pass | 6s | one object tampered (arn:aws:dynamodb:eu-west-1:000000000000:table/my-table-probable-spaniel's Terraform tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag | | Rename | pass | 11s | moved block: module.dynamodb_table renamed to module.dynamodb_table_moved with zero churn (0 add, 1 change, 0 destroy) - the table's own marker rewritten in place, the untaggable resource policy unaffected; live-mv: module.dynamodb_table_moved renamed to module.dynamodb_table_final with zero churn, marker rewritten in place; stock oracle over the same net module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); the table's ARN unchanged throughout, read via the AWS CLI | | Remove a block | pass | 6s | choudoufu: deleting module.dynamodb_table_final's block proposed exactly two destroys (0 add, 0 change, 2 destroy: the table and its untaggable resource policy), applied cleanly (0 added, 0 changed, 2 destroyed), the table is genuinely gone from the live account (dynamodb describe-table on the old name now returns ResourceNotFoundException, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly two destroys for the same two objects; classifyOrphans did not withhold either destroy because module.disabled_dynamodb_table declares zero instances of the same block key (create_table=false), so nothing is ever pending against it | | Change count | pass | 31s | choudoufu: scaling the synthetic aws_dynamodb_table.count_test from 2 to 1 (issue #359/#488's own fallback clause - this estate's real module has no honest resource-level count/for_each knob: create_table is boolean-shaped and replica_regions/global_secondary_indexes drive dynamic blocks nested inside the SAME table resource, not a separate resource instance, confirmed by reading main.tf directly) destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), confirmed gone via the AWS CLI, its local record correctly tombstoned rather than left claiming a live identity (#398-guard shape, has(tombstone) and not has(identity)), and left count_test[0]'s live TableId and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] again under the SAME ARN (deterministic from region+account+name - established directly against floci with no tofu in the loop before writing this assertion) but a NEW TableId (0 add -> 1 add, 0 change, 0 destroy), and its local record returned to a live identity, while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical 2-instance count block, applied for real in a dedicated always-idle account never shared with this one, shows the identical shape: destroy the higher index only, create it back under the same ARN but a new TableId, the lower index's TableId unchanged both times. BREAK_COUNT=1 confirms the wrong-instance assertion correctly fails to hold. | -| Replace with create_before_destroy | pass | 13s | choudoufu: changing module.dynamodb_table_final's ForceNew name argument proposed exactly one table replace at the same declared address, cascading into the untaggable resource policy (its resource_arn argument follows the table's ARN and is not independently updatable, so it also replaces - F-ORACLE's own finding); applied cleanly; the old table is confirmed gone via the AWS CLI (ResourceNotFoundException) and the new table carries the marker; the local record store's record at the same address now names the new table's name, not the destroyed one (my-table-clear-horse -> my-table-clear-horse-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the table at the same address (plan only, not applied - it shares floci's account with $EST); BREAK=replace confirms a manufactured marker collision is reported loudly ("Two live resources claiming one slot") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones. | +| Replace with create_before_destroy | pass | 13s | choudoufu: changing module.dynamodb_table_final's ForceNew name argument proposed exactly one table replace at the same declared address, cascading into the untaggable resource policy (its resource_arn argument follows the table's ARN and is not independently updatable, so it also replaces - F-ORACLE's own finding); applied cleanly; the old table is confirmed gone via the AWS CLI (ResourceNotFoundException) and the new table carries the marker; the local record store's record at the same address now names the new table's name, not the destroyed one (my-table-probable-spaniel -> my-table-probable-spaniel-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the table at the same address (plan only, not applied - it shares floci's account with $EST); BREAK=replace confirms a manufactured marker collision is reported loudly ("Two live resources claiming one slot") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 45s | 3 resources from nothing (random_pet + table + resource policy), the table's markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on key schema/attributes/table class/deletion protection/on-demand billing/GSI/resource policy | +| Plan, review, apply | pass | 13s | one argument edited (the resource_policy heredoc's statement Sid, AllowDummyRoleAccess -> AllowDummyRoleAccessReviewed, reaching only module.dynamodb_table.aws_dynamodb_resource_policy.this[0]), "plan -out=approved.tfplan" wrote a 21195-byte stock-format plan file whose whole change set is that one update; the world then moved out of band (arn:aws:dynamodb:eu-west-1:000000000000:table/my-table-probable-spaniel's Terraform tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.dynamodb_table.aws_dynamodb_table.this[0] and the live table id my-table-probable-spaniel it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - dynamodb get-resource-policy still returned a policy without the reviewed Sid, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the table's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the live policy read back WITH the reviewed Sid, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. This estate owns only two objects the plan acts on, so the review deliberately sits on the resource policy and the move on the table: that is what makes the refusal an EXTRA row it can name rather than a values-only disagreement about one row. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 46s | 3 resources from nothing (random_pet + table + resource policy), the table's markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on key schema/attributes/table class/deletion protection/on-demand billing/GSI/resource policy | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 4m2.5s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 4m3.8s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Sourced and scaffolded 2026-08-20 (4d4a3f6513, worktree ../wt/new-terraform-estate, branch live/dynamodb-table-basic) as exploratory/blocked work: terraform-aws-dynamodb-table is one of the single most-downloaded modules in the terraform-aws-modules org (35.6M Terraform Registry downloads), never crossed before, and stage 1 could not complete at all - the example's aws_dynamodb_resource_policy calls PutResourcePolicy on create, and floci had no handler for it (lex00/floci#86). Stage 1 landed 2026-08-21 after #86 was fixed (build e61a987). Stage 2 then hit a SEPARATE, real floci gap: DescribeTable never returned a GSI's OnDemandThroughput, so the AWS provider's own Read dropped it from state and the very next plan proposed replacing the GSI on a table that had not drifted - reproduced with plain stock terraform too (HANDOFF.md label 1, PARITY, not a choudoufu defect), filed as lex00/floci#91. FULL 5/5 PASS 2026-08-21 after #91 was fixed (build 8539609c, PR lex00/floci#92, published digest sha256:8f1fc4a500a3553e362c689cdcb6c5e31784bbaa7ad914de22bdd1c088a785f5, tag 8539609) and live/floci-image re-pinned to it. Re-crossed for real end to end against the new pin, nothing cherry-picked or assumed: cold_deploy PASS ('Apply complete! Resources: 3 added, 0 changed, 0 destroyed', table live and unmarked before migration, 0 objects carrying tofu-estate before migration). migrate PASS in both halves - live-import's dry run correctly reports '1 of 3 resource instance(s) are eligible for stamping' and 'UNTAGGABLE (2)' (aws_dynamodb_resource_policy has no tags argument in the provider's schema at all, and random_pet.this is record-backed with no live object to tag, neither a choudoufu gap), -approve stamps the table, and tofu-address verifies directly against DynamoDB via the AWS CLI as module.dynamodb_table.aws_dynamodb_table.this:0. The tofu-slot convergence apply that follows - previously the exact wall #91 fixed - now completes cleanly ('Apply complete! Resources: 0 added, 0 changed, 1 destroyed', the destroyed resource being the synthetic tofu-slot placeholder for the module's own zero-instance branch, the same shape corpus-iam-policy's TOFU-SLOT FINDING documents, not the table - the table's ARN and marker are reconfirmed live immediately after). test_plan PASS: live-plan with the state file deleted proposes no resource change and reports 'Foreign resources: none', and the table's tofu-address re-read directly from DynamoDB after the state file has never existed this run is unchanged. test_apply PASS: applying the empty plan is a genuine no-op, 'Resources: 0 added, 0 changed, 0 destroyed', 1 tagged object before and after, no state file either time. drift_reconverge PASS: the table's Terraform tag is tampered out of band via the AWS CLI, live-plan proposes fixing exactly module.dynamodb_table.aws_dynamodb_table.this[0] and nothing else, and the reconverge apply ('Resources: 0 added, 1 changed, 0 destroyed') restores the tag to its configured value, verified by re-reading the tag directly from DynamoDB. A NEW estate at full parity: terraform-aws-dynamodb-table, 35.6M registry downloads, never crossed before this session, now clears all five stages. THE RANDOM_PET GAP (issue #314, still open, same DELTA 3 shape live/e2e/corpus-s3-bucket-complete uses) remains real, tracked product debt, not hidden: the table's own identity-bearing `name` argument is `"my-table-${random_pet.this.id}"`, a TAGGED resource's identity computed from a record-backed value - live-import does not resolve this generically yet, so the already-applied pet value is substituted as a literal ahead of stage 2, called out in the script. BREAK=1 not exercised this session (a clean run was the priority now that the estate finally clears; stages 3 and 5 both carry BREAK-gated assertions in the script, ready for a future run to confirm they are load-bearing). diff --git a/site/content/docs/progress/corpus-ec2-instance-complete.md b/site/content/docs/progress/corpus-ec2-instance-complete.md index afc3240517..d50a109f6d 100644 --- a/site/content/docs/progress/corpus-ec2-instance-complete.md +++ b/site/content/docs/progress/corpus-ec2-instance-complete.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m18s | 35 resources added across 13 types (aws_instance, aws_eip, aws_iam_role/instance_profile/role_policy_attachment, aws_ebs_volume, aws_volume_attachment, aws_security_group x2, aws_vpc_security_group_egress_rule x2, aws_security_group_rule x2, vpc/subnet/route*/igw/default_* from the vpc module), 0 objects carry tofu-estate before migration | +| Cold deploy | pass | 56s | 35 resources added across 13 types (aws_instance, aws_eip, aws_iam_role/instance_profile/role_policy_attachment, aws_ebs_volume, aws_volume_attachment, aws_security_group x2, aws_vpc_security_group_egress_rule x2, aws_security_group_rule x2, vpc/subnet/route*/igw/default_* from the vpc module), 0 objects carry tofu-estate before migration | | Migrate | pass | 30s | 24 of 35 eligible (11 untaggable across 5 types - aws_iam_role_policy_attachment, aws_volume_attachment, aws_security_group_rule x2, aws_route, aws_route_table_association x6 - all resolved by provider identity schema), 24 stamped, 0 failed, 11 skipped; the IAM role policy attachment's composite live id asserted by value; genuine no-op on the follow-up apply | | Replan from nothing | pass | 5s | no resource change proposed by either plan; the default plan reports "nothing was swept" (the CollectUnclaimed ruling (#604) made the account-inventory question opt-in, and a run that did not ask must say so), and a second plan run with TOFU_LIVE_COLLECT_UNCLAIMED=1 finds exactly 8 foreign objects - the instance's own root volume plus floci's default-VPC bootstrap; instance tofu-address re-checked against EC2 | | No-op apply | pass | 4s | genuine no-op (0 added, 0 changed, 0 destroyed); 24 objects before, 24 after, no state file | | Drift and reconverge | pass | 6s | one object tampered, exactly 1 object proposed and applied (0 added, 1 changed, 0 destroyed), tag reconverged to "ex-complete" | -| Rename | pass | 17s | moved block: module.vpc renamed with zero churn (0 add, 15 change, 0 destroy), marker rewritten in place; live-mv: module.security_group's security group renamed with zero churn, its two untaggable rules followed for free; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 30s | choudoufu: deleting module.ec2_complete's block proposed exactly 10 destroys (0 add, 0 change, 10 destroy), matching the stock oracle's own count and applied cleanly; the instance is confirmed terminated and the tagged object count dropped, both via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 10 destroys for the same module | -| Change count | pass | 2m11s | choudoufu: scaling aws_ebs_volume.count_test from 2 to 1 destroyed exactly count_test[1] (vol-5d2c1b342f912c2f6, 0 add, 0 change, 1 destroy) and left count_test[0] (vol-a2f08d6e0a3772fd1) with the same live VolumeId, the same tofu-address=aws_ebs_volume.count_test:0 and the same tofu-slot=0, all read back through the AWS CLI rather than choudoufu's own report; the destroyed volume is genuinely gone (describe-volumes answers InvalidVolume.NotFound for it). Scaling back from 1 to 2 planned exactly 1 to add, 0 to change, 0 to destroy and brought count_test[1] back as a NEW object (vol-0778a981679882f57, not vol-5d2c1b342f912c2f6) carrying tofu-address=aws_ebs_volume.count_test:1 and tofu-slot=1, above the live high-water mark count_test[0] still holds, while count_test[0] stayed untouched throughout; the next plan is empty, and scaling the block to zero destroys both and leaves the estate planning empty again. C-ORACLE, the same 2-instance block stood up for real with plain terraform in its own working directory at the SAME resolved provider version (6.63.0), shows the identical shape: destroy the higher index only (vol-6c64f51b35c8db3ec), create the higher index back under a new id (vol-a8a9c51961c08f115), the lower index's id (vol-eee10750df38c22af) unchanged both times. SYNTHETIC BLOCK, and why: terraform-aws-ec2-instance v6.4.0 declares no scalable count or for_each knob this estate reaches - all nine of its own count usages are boolean create toggles of the form 'count = local.create ? 1 : 0', which can never hold two instances, and the upstream example's one real for_each fan-out (module.ec2_multiple) is dropped by this script's reduction because floci does not model the surfaces around it - so this section adds a new, self-contained count block of a type the estate ALREADY exercises (aws_ebs_volume, module.ec2_complete's own /dev/sdf data volume), the sanctioned fallback live/GAUNTLET.md #8 names, with reference-ec2-vpc Part F and corpus-iam-policy Part G as precedent. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports day2_count fail, proving the assertion is load-bearing. | -| Replace with create_before_destroy | pass | 50s | choudoufu: changing module.ec2_complete's ForceNew ami argument proposed exactly one instance replace at the same declared address, cascading into the eip (updated in-place) and the volume attachment (also replaced, instance_id is ForceNew there too) - 2 to add, 1 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated and the new instance carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-b4eb27b23605d3e67 -> i-7a9aa80c70f73af43); the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly ("Two live resources claiming one slot") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one. | +| Rename | pass | 14s | moved block: module.vpc renamed with zero churn (0 add, 15 change, 0 destroy), marker rewritten in place; live-mv: module.security_group's security group renamed with zero churn, its two untaggable rules followed for free; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Remove a block | pass | 28s | choudoufu: deleting module.ec2_complete's block proposed exactly 10 destroys (0 add, 0 change, 10 destroy), matching the stock oracle's own count and applied cleanly; the instance is confirmed terminated and the tagged object count dropped, both via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 10 destroys for the same module | +| Change count | pass | 2m7s | choudoufu: scaling aws_ebs_volume.count_test from 2 to 1 destroyed exactly count_test[1] (vol-227203145b6519f82, 0 add, 0 change, 1 destroy) and left count_test[0] (vol-1ba72164534af9b35) with the same live VolumeId, the same tofu-address=aws_ebs_volume.count_test:0 and the same tofu-slot=0, all read back through the AWS CLI rather than choudoufu's own report; the destroyed volume is genuinely gone (describe-volumes answers InvalidVolume.NotFound for it). Scaling back from 1 to 2 planned exactly 1 to add, 0 to change, 0 to destroy and brought count_test[1] back as a NEW object (vol-25211d5ff11f1d3d0, not vol-227203145b6519f82) carrying tofu-address=aws_ebs_volume.count_test:1 and tofu-slot=1, above the live high-water mark count_test[0] still holds, while count_test[0] stayed untouched throughout; the next plan is empty, and scaling the block to zero destroys both and leaves the estate planning empty again. C-ORACLE, the same 2-instance block stood up for real with plain terraform in its own working directory at the SAME resolved provider version (6.63.0), shows the identical shape: destroy the higher index only (vol-43eecd95f10607c2e), create the higher index back under a new id (vol-d1189cb4ab1b4dbf8), the lower index's id (vol-63f626cc1ed09eb2e) unchanged both times. SYNTHETIC BLOCK, and why: terraform-aws-ec2-instance v6.4.0 declares no scalable count or for_each knob this estate reaches - all nine of its own count usages are boolean create toggles of the form 'count = local.create ? 1 : 0', which can never hold two instances, and the upstream example's one real for_each fan-out (module.ec2_multiple) is dropped by this script's reduction because floci does not model the surfaces around it - so this section adds a new, self-contained count block of a type the estate ALREADY exercises (aws_ebs_volume, module.ec2_complete's own /dev/sdf data volume), the sanctioned fallback live/GAUNTLET.md #8 names, with reference-ec2-vpc Part F and corpus-iam-policy Part G as precedent. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports day2_count fail, proving the assertion is load-bearing. | +| Replace with create_before_destroy | pass | 50s | choudoufu: changing module.ec2_complete's ForceNew ami argument proposed exactly one instance replace at the same declared address, cascading into the eip (updated in-place) and the volume attachment (also replaced, instance_id is ForceNew there too) - 2 to add, 1 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated and the new instance carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-c7a93b787f29f6013 -> i-2bfd2886f8aefa0fd); the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly ("Indistinguishable instances without per-instance markers", naming both live instances) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 1m11s | 35 resources from nothing, matching stock's own cold-deploy count; the instance's markers verified via the AWS CLI; 35 records in the local record store including untaggable types; replan empty; the instance's own shape (type/ami/block-device-count) matches stock's cold deploy, via the AWS CLI on both endpoints, marker tags never compared; 24 objects carry the estate tag | +| Plan, review, apply | pass | 17s | one argument edited (the "/dev/sdf" entry's MountPoint volume tag inside module "ec2_complete"'s ebs_volumes argument, /mnt/data -> /mnt/data-reviewed - the module merges each entry's tags into that entry's aws_ebs_volume alone, so it reaches module.ec2_complete.aws_ebs_volume.this["/dev/sdf"] and nothing else), "plan -out=approved.tfplan" wrote a 72468-byte stock-format plan file whose whole change set is that one update; the world then moved out of band (i-c7a93b787f29f6013's Example tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.ec2_complete.aws_instance.this[0] and the live i-c7a93b787f29f6013 it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - volume vol-877db6c95f823e408 still read MountPoint=/mnt/data through ec2 describe-tags, not from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with i-c7a93b787f29f6013's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and vol-877db6c95f823e408 read back with MountPoint=/mnt/data-reviewed, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART C starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 1m1s | 35 resources from nothing, matching stock's own cold-deploy count; the instance's markers verified via the AWS CLI; 35 records in the local record store including untaggable types; replan empty; the instance's own shape (type/ami/block-device-count) matches stock's cold deploy, via the AWS CLI on both endpoints, marker tags never compared; 24 objects carry the estate tag | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 7m2.7s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 6m38.5s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. ## Reproduce it diff --git a/site/content/docs/progress/corpus-ecs-fargate.md b/site/content/docs/progress/corpus-ecs-fargate.md index 8ba7ff865d..1c2b00ea82 100644 --- a/site/content/docs/progress/corpus-ecs-fargate.md +++ b/site/content/docs/progress/corpus-ecs-fargate.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m45s | 62 resources, once for real | -| Migrate | pass | 1m26s | 46 of 62 stamped | -| Replan from nothing | pass | 29s | genuinely empty replan ("No changes. Your infrastructure matches the configuration.") - #371, #378, #372, #110, #395 and #376 all fixed and stay fixed; the standalone task definition's essential/mountPoints[].readOnly wall is gone (lex00/floci#131, published and repinned this unit) and essential defaulting to true was never an independent wall on its own (this unit's own re-measurement). #395/#376: choudoufu keeps no persisted state, so every plan re-derives PriorState through ImportResourceState's bare stub; internal/live/projection/build.go's configuredAttrsSeed generalizes the tags-only import-stub seed (issue #287 item 8) to every Required-or-Optional-non-Computed attribute (fixing #376's track_latest/skip_destroy directly), and internal/live/projection/residue.go's residueConfigSourced widening of classifyResidue plus the new builder.residueSeedFor pre-read seed close #395's managed-reference case (task_definition = aws_ecs_task_definition.this[0].arn) that configuredAttrsSeed's static evaluator alone could not reach. Identities confirmed by value against the AWS CLI: $CLUSTER_ARN, $TD_SVC_ARN, $TD_STANDALONE_ARN, and #368's scalable target $GOT_TARGET_RID. | +| Cold deploy | pass | 1m22s | 62 resources, once for real | +| Migrate | pass | 1m18s | 46 of 62 stamped | +| Replan from nothing | pass | 28s | genuinely empty replan ("No changes. Your infrastructure matches the configuration.") - #371, #378, #372, #110, #395 and #376 all fixed and stay fixed; the standalone task definition's essential/mountPoints[].readOnly wall is gone (lex00/floci#131, published and repinned this unit) and essential defaulting to true was never an independent wall on its own (this unit's own re-measurement). #395/#376: choudoufu keeps no persisted state, so every plan re-derives PriorState through ImportResourceState's bare stub; internal/live/projection/build.go's configuredAttrsSeed generalizes the tags-only import-stub seed (issue #287 item 8) to every Required-or-Optional-non-Computed attribute (fixing #376's track_latest/skip_destroy directly), and internal/live/projection/residue.go's residueConfigSourced widening of classifyResidue plus the new builder.residueSeedFor pre-read seed close #395's managed-reference case (task_definition = aws_ecs_task_definition.this[0].arn) that configuredAttrsSeed's static evaluator alone could not reach. Identities confirmed by value against the AWS CLI: $CLUSTER_ARN, $TD_SVC_ARN, $TD_STANDALONE_ARN, and #368's scalable target $GOT_TARGET_RID. | | No-op apply | pass | 5s | genuine no-op (0 added, 0 changed, 0 destroyed); 46 tofu-estate-tagged objects before, 46 after, no state file either time | -| Drift and reconverge | pass | 12s | one object tampered (VPC's Name tag), plan proposed fixing exactly module.vpc.aws_vpc.this[0], apply changed 1 and the Name tag reconverged | -| Rename | pass | 59s | moved block: module.alb renamed with zero churn (0 add, 9 change, 0 destroy), marker rewritten in place; live-mv: aws_service_discovery_http_namespace.this renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 20s | choudoufu: deleting module.ecs_task_definition's block proposed exactly 8 destroys (0 add, 0 change, 8 destroy), address-for-address identical to stock's oracle on cold_deploy's own state; applied cleanly (0 added, 0 changed, 8 destroyed); the standalone task definition family (ex-fargate-standalone) genuinely has 0 active revisions afterward, read via the AWS CLI, not choudoufu's own report; classifyOrphans did not withhold any destroy because no other module.ecs_task_definition block is declared anywhere in this config; the next plan is empty | -| Change count | pass | 2m2s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.), leaving count_test[0]'s live GroupId (sg-38d62e21de06f4816), its tofu-address (aws_security_group.count_test:0) and its tofu-slot (0) unchanged, and count_test[1] (sg-6b3d7970b4dc6d8b9) genuinely absent - all read through the AWS CLI, never choudoufu's own report; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.) under a NEW server-minted GroupId (sg-49a87d7c99c8dbff3, was sg-6b3d7970b4dc6d8b9) carrying tofu-address=aws_security_group.count_test:1 and the same tofu-slot=1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (H-ORACLE), the identical 2-instance block stood up for real with plain terraform in its own account and its own working directory - stock never had this block, so unlike day2_rename/day2_remove/day2_replace there is no cold_deploy state to reuse - shows the identical shape: Plan: 0 to add, 0 to change, 1 to destroy. destroying count_test[1]=sg-703e202276aee9559 only, then Plan: 1 to add, 0 to change, 0 to destroy. bringing it back under a new GroupId (sg-ae896ab2a2a7983c6), with count_test[0]=sg-3b2059c3f4807ab3a unchanged both times. SYNTHETIC block, and why: this estate's corpus example (terraform-aws-modules/terraform-aws-ecs v7.6.0, examples/fargate) declares no scalable count knob at all - every count in its vendored modules is a 'count = local.create ? 1 : 0' boolean create toggle, which cannot hold two instances - and the only construct holding three, module.vpc's private_subnets, is driven by local.azs, which the ECS service's network configuration, the ALB and the NAT gateway all read, so scaling it churns a dozen unrelated resources instead of isolating which count instance a scale-down destroys. aws_security_group was chosen because this estate already exercises it three times in its own vendored modules (modules/cluster, modules/service, modules/express-service each declare aws_security_group.this) and because its destroy is witnessed TWICE on the pinned emulator - a new GroupId AND verified absence in between - established by reading the EC2 API directly with no tofu in the loop before any assertion here was written. Placed last, after every other stage reported, at an address nothing else in this configuration names, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly fails. | -| Replace with create_before_destroy | pass | 25s | choudoufu: changing module.ecs_task_definition's ForceNew name argument (module CALL, passed through to the local module's own family = coalesce(var.family, var.name)) proposed a forced replace at the same declared address (Plan: 8 to add, 0 to change, 8 to destroy.), applied cleanly; the old task definition is confirmed INACTIVE via the AWS CLI (ECS deregisters rather than deletes) and the new one (arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1) carries the marker, moved via the tofu-address tag (arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1 -> arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the task definition at the same address (Plan: 8 to add, 0 to change, 8 to destroy., plan only, not applied - it shares floci's account with $ADOPTED_EST); BREAK=replace confirms a manufactured marker collision - a second, genuinely live task definition wearing this address's marker, which no tombstone names as destroyed - is still reported loudly ("Indistinguishable instances without per-instance markers", naming both ARNs) rather than pruned or silently proposed as nothing, which is #849's own rule holding on this route too. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; the local record store is read by value on both sides of the replace (#879): before it, the record names family=ex-fargate-standalone with identity.secondary_id=arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1; after it, identity.secondary_id=arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1 with exactly one tombstone naming arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1, which is what lets the F2 plan tell the deregistered object's lingering tag from a second live claimant instead of refusing "Indistinguishable instances without per-instance markers" forever; two earlier target choices each found a genuine, separate defect: aws_service_discovery_http_namespace.this_renamed's (mv.go's propagateModuleRename skipping MoveRecord for a same-module rename) is FIXED on the gauntlet/mv-rekey branch, GitHub issue #412 - see F-ORACLE's own header comment and corpus-autoscaling-complete's/corpus-eks-basic's matching mv.go finding in this same unit, neither of which was re-run for #412; module.alb_renamed's (the non-converging cascade, F-ORACLE's own header comment, finding 2) remains a separate, open finding, not fixed here. | +| Drift and reconverge | pass | 11s | one object tampered (VPC's Name tag), plan proposed fixing exactly module.vpc.aws_vpc.this[0], apply changed 1 and the Name tag reconverged | +| Rename | pass | 52s | moved block: module.alb renamed with zero churn (0 add, 9 change, 0 destroy), marker rewritten in place; live-mv: aws_service_discovery_http_namespace.this renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Remove a block | pass | 15s | choudoufu: deleting module.ecs_task_definition's block proposed exactly 8 destroys (0 add, 0 change, 8 destroy), address-for-address identical to stock's oracle on cold_deploy's own state; applied cleanly (0 added, 0 changed, 8 destroyed); the standalone task definition family (ex-fargate-standalone) genuinely has 0 active revisions afterward, read via the AWS CLI, not choudoufu's own report; classifyOrphans did not withhold any destroy because no other module.ecs_task_definition block is declared anywhere in this config; the next plan is empty | +| Change count | pass | 1m4s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.), leaving count_test[0]'s live GroupId (sg-52d986d91b3c973e6), its tofu-address (aws_security_group.count_test:0) and its tofu-slot (0) unchanged, and count_test[1] (sg-492493a03f1be337d) genuinely absent - all read through the AWS CLI, never choudoufu's own report; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.) under a NEW server-minted GroupId (sg-e97c66f6e7938bb49, was sg-492493a03f1be337d) carrying tofu-address=aws_security_group.count_test:1 and the same tofu-slot=1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (H-ORACLE), the identical 2-instance block stood up for real with plain terraform in its own account and its own working directory - stock never had this block, so unlike day2_rename/day2_remove/day2_replace there is no cold_deploy state to reuse - shows the identical shape: Plan: 0 to add, 0 to change, 1 to destroy. destroying count_test[1]=sg-2f52b9dfff25eabef only, then Plan: 1 to add, 0 to change, 0 to destroy. bringing it back under a new GroupId (sg-5feabe05e9ba79cb0), with count_test[0]=sg-c38e15e20448864cb unchanged both times. SYNTHETIC block, and why: this estate's corpus example (terraform-aws-modules/terraform-aws-ecs v7.6.0, examples/fargate) declares no scalable count knob at all - every count in its vendored modules is a 'count = local.create ? 1 : 0' boolean create toggle, which cannot hold two instances - and the only construct holding three, module.vpc's private_subnets, is driven by local.azs, which the ECS service's network configuration, the ALB and the NAT gateway all read, so scaling it churns a dozen unrelated resources instead of isolating which count instance a scale-down destroys. aws_security_group was chosen because this estate already exercises it three times in its own vendored modules (modules/cluster, modules/service, modules/express-service each declare aws_security_group.this) and because its destroy is witnessed TWICE on the pinned emulator - a new GroupId AND verified absence in between - established by reading the EC2 API directly with no tofu in the loop before any assertion here was written. Placed last, after every other stage reported, at an address nothing else in this configuration names, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly fails. | +| Replace with create_before_destroy | pass | 16s | choudoufu: changing module.ecs_task_definition's ForceNew name argument (module CALL, passed through to the local module's own family = coalesce(var.family, var.name)) proposed a forced replace at the same declared address (Plan: 8 to add, 0 to change, 8 to destroy.), applied cleanly; the old task definition is confirmed INACTIVE via the AWS CLI (ECS deregisters rather than deletes) and the new one (arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1) carries the marker, moved via the tofu-address tag (arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1 -> arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the task definition at the same address (Plan: 8 to add, 0 to change, 8 to destroy., plan only, not applied - it shares floci's account with $ADOPTED_EST); BREAK=replace confirms a manufactured marker collision - a second, genuinely live task definition wearing this address's marker, which no tombstone names as destroyed - is still reported loudly ("Indistinguishable instances without per-instance markers", naming both ARNs) rather than pruned or silently proposed as nothing, which is #849's own rule holding on this route too. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; the local record store is read by value on both sides of the replace (#879): before it, the record names family=ex-fargate-standalone with identity.secondary_id=arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1; after it, identity.secondary_id=arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1 with exactly one tombstone naming arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1, which is what lets the F2 plan tell the deregistered object's lingering tag from a second live claimant instead of refusing "Indistinguishable instances without per-instance markers" forever; two earlier target choices each found a genuine, separate defect: aws_service_discovery_http_namespace.this_renamed's (mv.go's propagateModuleRename skipping MoveRecord for a same-module rename) is FIXED on the gauntlet/mv-rekey branch, GitHub issue #412 - see F-ORACLE's own header comment and corpus-autoscaling-complete's/corpus-eks-basic's matching mv.go finding in this same unit, neither of which was re-run for #412; module.alb_renamed's (the non-converging cascade, F-ORACLE's own header comment, finding 2) remains a separate, open finding, not fixed here. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 3m4s | 62 resources from nothing, cluster marker verified via the AWS CLI, 62 of 62 records in the local record store (#364 A2; both aws_ecs_task_definition instances are now included, #671 having closed the numeric-wire-identity-component gap in internal/live/identity/located.go's LocatedIdentityPlanFor that used to exclude them), replan empty, stock oracle in its own namespace matches structurally on cluster/service/standalone-task-definition/CloudMap-namespace/ALB/VPC | +| Plan, review, apply | pass | 27s | one argument edited (module "ecs_service"'s service_tags ServiceTag, "Tag on service level" -> "Tag on service level, reviewed" - the service module merges service_tags into aws_ecs_service's own tags and nowhere else, so it reaches exactly one instance where every tags = local.tags in this example fans out over a whole module), "plan -out=approved.tfplan" wrote a 147864-byte stock-format plan file whose whole change set is one update on module.ecs_service.aws_ecs_service.this[0]; the world then moved out of band (VPC vpc-0ee657b8's Name tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.vpc.aws_vpc.this[0] and the live vpc-0ee657b8 it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - arn:aws:ecs:eu-west-1:000000000000:service/ex-fargate/ex-fargate still read ServiceTag="Tag on service level" through ecs list-tags-for-resource, not from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the VPC's Name tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the service read back with the reviewed ServiceTag, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 2m52s | 62 resources from nothing, cluster marker verified via the AWS CLI, 62 of 62 records in the local record store (#364 A2; both aws_ecs_task_definition instances are now included, #671 having closed the numeric-wire-identity-component gap in internal/live/identity/located.go's LocatedIdentityPlanFor that used to exclude them), replan empty, stock oracle in its own namespace matches structurally on cluster/service/standalone-task-definition/CloudMap-namespace/ALB/VPC | | Strict profile (not a headline stage) | not run | | | -Last run at commit `3977d90784` on 2026-09-06T05:49:39Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 10m48s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 9m10.4s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. RE-CROSSED FOR REAL 2026-08-21 at dbc5ebf575 (GitHub issue #346's fix), twice, the second time instrumented to capture live-plan's raw output. STAGES UNCHANGED at 2 of 5, and #346's fix does not reach this estate either. Stage 1: 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', cluster confirmed carrying no tofu-address beforehand. Stage 2: '46 of 62 eligible (28 VERIFIED + 18 DRIFTED); 16 skipped'; '46 stamped, 1 recorded (time_sleep.this[0]), 0 failed, 15 skipped'; both markers confirmed by direct aws ecs describe-* calls against floci rather than through choudoufu's own report (cluster/ex-fargate = module.ecs_cluster.aws_ecs_cluster.this:0, service/ex-fargate/ex-fargate = module.ecs_service.aws_ecs_service.this:0). Stage 3 fails on exactly 8 diagnostics, the same 8 as before, verified two ways (grep -c '^Error:' over the raw output, and the script's own hard count assertions): 1 'Module output not supported in static context' on main.tf:68 (cluster_arn = module.ecs_cluster.arn), 1 'Unable to compute static value' on modules/service/main.tf:1565 (aws_appautoscaling_target.this[0].resource_id), and 6 'Unresolvable identity' cascading from it onto aws_appautoscaling_policy.this['cpu'] and ['memory']. Why the fix misses it: the same value-shaped module-CALL-argument route as corpus-rds-complete-postgres, PLUS a transform the part-shaped route could not express even if it reached - local.cluster_name = element(split('/', var.cluster_arn), 1), a function applied to the deferred value. identity.Formula holds literals and ParentRefs and has no way to say 'split this parent attribute and take element 1'. That is a mechanism this repository does not have, not a gap in #346's fix. A 'Provider version does not match the admission evidence version' warning (6.61.0 resolved against 6.59.0 admission evidence) also prints; the script's own header calls it a caution, not a failure. PRIOR HISTORY BELOW. Landed dd83121592 (2026-08-18), 62 resources: ECS cluster (Container Insights, FARGATE/FARGATE_SPOT split), a BLUE_GREEN service behind an ALB, ECS Exec, ECS Service Connect, a two-container task definition plus a standalone second one, a CloudMap namespace, nested ALB/VPC modules. cold_deploy and migrate genuinely pass (43 of 62 stamped: 26 VERIFIED + 17 DRIFTED; 19 skipped - 16 untaggable by design, 3 blocked by #305). test_plan blocked by 4 sites: #305's familiar default_* trio (3 sites) and a NEW one, filed as #308: the child-module for_each keyset prover (internal/live/identity/foreach_keyset.go) has no case for a for-comprehension (for k, v in var.container_definitions : k => v if ...) and doesn't chase a bare var.X for_each source across a module-call boundary to the literal object constructor at the caller, whose keys are actually static even though one unrelated attribute value inside the map is dynamic - resolve.go's resource-level forEachOverComprehension already does the equivalent per-entry evaluation the module-call prover lacks. Two real floci gaps found and filed but not fixed, both sized as real modeling work rather than quick patches: lex00/floci#59 (CreateCluster silently drops settings/Container Insights, default_capacity_provider_strategy never serialized back) and lex00/floci#60 (CreateService/DescribeServices drop scheduling_strategy, enable_ecs_managed_tags, enable_execute_command, health_check_grace_period_seconds, deployment_controller, blue-green load_balancer.advanced_configuration, service_connect_configuration entirely - scheduling_strategy's omission in particular forces the AWS provider to propose destroy-and-recreate on every plan after creation, a real non-idempotency bug independent of choudoufu). Neither floci gap blocks this crossing's own outcome since test_plan already refuses earlier, upstream of any ECS-field diff. Follow-up pass 2026-08-18 (#313 cross-check, 0a94070b16/3ff22c5be6): the committed run.sh still asserted pre-#305 counts (43/62 eligible) and failed before ever reaching stage 3; updated to the real current numbers (46/62 eligible, #305's default_* trio now fully resolved here) and re-verified against real floci twice plus a BREAK=1 negative control, all read from the script's own printed lines. test_plan's sole remaining blocker is confirmed #308 alone - #313's diagnostic does not appear anywhere in the output. Mechanism: this estate's vpc submodule expands subnets via count over a statically-known length, never a for_each keyed on the AZ name values, so #313 structurally can't reach it. Commented on #313 and #308. #308 fixed and merged 2026-08-18 (a9ac6d06e7/b2bb59585d, generic: a *hclsyntax.ForExpr case plus a cross-module-call var/local chase in internal/live/identity/foreach_keyset.go, reaching every module-call for_each proof, not just this estate) - re-run confirms 0 occurrences of #308's diagnostic (was 1), but test_plan is still blocked: #308 firing first had been masking two more causes in the same live-plan output all along. Follow-up pass 2026-08-18 (e74c7d5869/07c7317ab6) re-staled run.sh's stage-3 assertions and header to the real current picture, verified across three separate real live-plan runs (stable counts each time) plus a BREAK=1 negative control: 236 total diagnostics, three distinct root causes, not one. Root cause A (#313's canonical shape, 48 sites): data.aws_availability_zones.available feeding local.azs into module "vpc"'s azs argument. Root cause B (also #313's family via a module output, 1 site): module.ecs_cluster.arn passed into module "ecs_service" as cluster_arn, "Module output not supported in static context". Both A and B are #313's already-ruled maintainer-level architecture question (live-plan never calls a provider during plan), not fixed here. Root cause C (NOT #313 - a distinct, newly-found, likely-fixable gap #308's own fix exposed, 4 sites): each.value.enable_cloudwatch_logging/create_cloudwatch_log_group inside module.container_definition, both literal booleans in the caller's own object literal but refused because each.value is treated as one opaque blob instead of projected to the referenced field - filed as #315, not attempted. These cascade to 177 "Unable to compute static value" and 6 "Unresolvable identity" sites (traced: aws_appautoscaling_policy reads aws_appautoscaling_target's resource_id, itself one of C's failures). Re-verified 2026-08-19 (be6b1096ba/b7bbff04a6) against everything landed overnight (#313's data-source half, #315, #321/#324, #323, #325): root cause A (0 sites, confirmed fixed) and root cause C (0 sites, #315's each.value projection fix confirmed) are both gone. Only root cause B remains (module.ecs_cluster.arn, a Computed attribute crossing a module-output boundary), now cascading to 1 "Unable to compute static value" plus 6 "Unresolvable identity" sites (aws_appautoscaling_target/aws_appautoscaling_policy chain) - down from 236 diagnostics to 8. This is the same already-acknowledged Computed-attribute/module-output architecture question corpus-rds-complete-postgres and corpus-security-group-complete are also blocked on; no action taken here, no new issue filed. diff --git a/site/content/docs/progress/corpus-eks-basic.md b/site/content/docs/progress/corpus-eks-basic.md index 67d1ca120d..f46102c846 100644 --- a/site/content/docs/progress/corpus-eks-basic.md +++ b/site/content/docs/progress/corpus-eks-basic.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m35s | 54 resources, genuinely cold, genuinely unmarked | -| Migrate | pass | 1m59s | 25 of 54 resource instances stamped, 25 of 25 confirmed via the AWS CLI; 5 record-backed instances seeded into the implied local record store (#364) | +| Cold deploy | pass | 1m23s | 54 resources, genuinely cold, genuinely unmarked | +| Migrate | pass | 1m33s | 25 of 54 resource instances stamped, 25 of 25 confirmed via the AWS CLI; 5 record-backed instances seeded into the implied local record store (#364) | | Replan from nothing | pass | 17s | live-plan runs to completion with ZERO Error diagnostics and reports "No changes. Your infrastructure matches the configuration." - the record-backed worker launch configuration's enable_monitoring/root_block_device/user_data all now agree with the config's own desired value (lex00/floci#132 for the first two, configuredAttrsSeed's residue-record pre-read seed in internal/live/projection/build.go for the third) | -| No-op apply | pass | 21s | genuine no-op (0 added, 0 changed, 0 destroyed); 25 tofu-estate-tagged objects before, 25 after | -| Drift and reconverge | pass | 39s | one object tampered (VPC's Name tag), plan proposed fixing exactly module.vpc.aws_vpc.this[0], apply changed 1 and the Name tag reconverged | -| Rename | pass | 1m17s | moved block: aws_security_group.worker_group_mgmt_two renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_security_group.all_worker_mgmt renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 1m1s | choudoufu: deleting aws_security_group.worker_group_mgmt_one's block (plus emptying the one argument that referenced it) proposed 2 resource action(s), address-for-address and action-for-action identical to stock's oracle on cold_deploy's own state; applied cleanly; the security group is genuinely gone from the live account, read via the AWS CLI, not choudoufu's own report; classifyOrphans did not withhold any destroy because no other aws_security_group.worker_group_mgmt_one block is declared anywhere in this config; the next plan is empty | -| Change count | pass | 3m20s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and count_test[0] kept BOTH its live id (sg-2190fcd004e71e1a2) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) across it, read back through the AWS CLI rather than choudoufu's own report; the destroyed group (sg-3ff9bb7024a1a3cac) is genuinely gone from the account; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a NEW object under a NEW GroupId (sg-06d58eb0790723d87, was sg-3ff9bb7024a1a3cac) carrying aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (G0): plain terraform, its own working directory, state and VPC, standing the IDENTICAL 2-instance count block up for real against the same idle floci account, shows the identical shape - destroy the higher index only (count_test[1]=sg-0fa8f370fcd242978), create it back under a new id (sg-0260527e0d0cd1063), count_test[0]=sg-5186dbede2e418faf unchanged both times - and is torn down again before choudoufu's own side runs. Synthetic block, and why: terraform-aws-eks v9.0.0 has no scalable count knob to drive - every count in the module is a boolean create toggle (var.create_eks ? 1 : 0 and siblings), and the one length-driven knob (local.worker_group_count) drives only aws_autoscaling_group, aws_launch_configuration and aws_iam_role_policy_attachment, all untaggable (no tags argument in the provider schema, so no marker surface for this stage's identity assertion), and dropping a worker group also rewrites kubernetes_config_map.aws_auth, which aggregates every worker role, so the plan would carry changes alongside the destroy. aws_security_group.count_test is the sanctioned fallback per live/GAUNTLET.md #8, of a type this estate already exercises three times, at an address nothing else in the config names. BREAK_COUNT=1 confirms the check is load-bearing: asserting the WRONG instance (count_test[0], the survivor) was destroyed makes this stage report fail. | -| Replace with create_before_destroy | pass | 1m0s | choudoufu: changing aws_security_group.worker_group_mgmt_two_renamed's ForceNew name_prefix argument proposed a forced replace at the same declared address (Plan: 3 to add, 1 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-17fb7372c7f753b56) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-4655e85b6bf6f3f89 -> sg-17fb7372c7f753b56); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 1 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly ("Two live resources claiming one slot") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted all_worker_mgmt_renamed (Part D's own live-mv leg) and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on the gauntlet/mv-rekey branch, GitHub issue #412 (propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard); see this section's own header comment for the fix and corpus-autoscaling-complete's/corpus-ecs-fargate's matching ones in this same unit. This script was not re-run for #412, so this detail string still describes the pre-#412 run until this estate's next real run. | +| No-op apply | pass | 22s | genuine no-op (0 added, 0 changed, 0 destroyed); 25 tofu-estate-tagged objects before, 25 after | +| Drift and reconverge | pass | 41s | one object tampered (VPC's Name tag), plan proposed fixing exactly module.vpc.aws_vpc.this[0], apply changed 1 and the Name tag reconverged | +| Rename | pass | 1m13s | moved block: aws_security_group.worker_group_mgmt_two renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_security_group.all_worker_mgmt renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Remove a block | pass | 1m2s | choudoufu: deleting aws_security_group.worker_group_mgmt_one's block (plus emptying the one argument that referenced it) proposed 2 resource action(s), address-for-address and action-for-action identical to stock's oracle on cold_deploy's own state; applied cleanly; the security group is genuinely gone from the live account, read via the AWS CLI, not choudoufu's own report; classifyOrphans did not withhold any destroy because no other aws_security_group.worker_group_mgmt_one block is declared anywhere in this config; the next plan is empty | +| Change count | pass | 3m34s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and count_test[0] kept BOTH its live id (sg-536bd82e1ae8b011a) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) across it, read back through the AWS CLI rather than choudoufu's own report; the destroyed group (sg-405bd41502e46a151) is genuinely gone from the account; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a NEW object under a NEW GroupId (sg-6af012ebcadc1cf6a, was sg-405bd41502e46a151) carrying aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (G0): plain terraform, its own working directory, state and VPC, standing the IDENTICAL 2-instance count block up for real against the same idle floci account, shows the identical shape - destroy the higher index only (count_test[1]=sg-acdf0826e35bb8527), create it back under a new id (sg-c7f1d8d68584ba8f6), count_test[0]=sg-cf71071802083cb77 unchanged both times - and is torn down again before choudoufu's own side runs. Synthetic block, and why: terraform-aws-eks v9.0.0 has no scalable count knob to drive - every count in the module is a boolean create toggle (var.create_eks ? 1 : 0 and siblings), and the one length-driven knob (local.worker_group_count) drives only aws_autoscaling_group, aws_launch_configuration and aws_iam_role_policy_attachment, all untaggable (no tags argument in the provider schema, so no marker surface for this stage's identity assertion), and dropping a worker group also rewrites kubernetes_config_map.aws_auth, which aggregates every worker role, so the plan would carry changes alongside the destroy. aws_security_group.count_test is the sanctioned fallback per live/GAUNTLET.md #8, of a type this estate already exercises three times, at an address nothing else in the config names. BREAK_COUNT=1 confirms the check is load-bearing: asserting the WRONG instance (count_test[0], the survivor) was destroyed makes this stage report fail. | +| Replace with create_before_destroy | pass | 58s | choudoufu: changing aws_security_group.worker_group_mgmt_two_renamed's ForceNew name_prefix argument proposed a forced replace at the same declared address (Plan: 3 to add, 1 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-1b6782f1886852eac) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-ca6711c0e80c37f96 -> sg-1b6782f1886852eac); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 1 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly ("Two live resources claiming one slot") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted all_worker_mgmt_renamed (Part D's own live-mv leg) and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on the gauntlet/mv-rekey branch, GitHub issue #412 (propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard); see this section's own header comment for the fix and corpus-autoscaling-complete's/corpus-ecs-fargate's matching ones in this same unit. This script was not re-run for #412, so this detail string still describes the pre-#412 run until this estate's next real run. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 2m34s | 54 resources from nothing, cluster marker verified via the AWS CLI, 54 records under the implied local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on cluster status/version, ASG count/desired-capacities, and cluster-owned security-group count | +| Plan, review, apply | pass | 1m40s | one argument edited (aws_security_group.worker_group_mgmt_one gains tags = { Reviewed = "yes" }), "plan -out=approved.tfplan" wrote a 82147-byte stock-format plan file whose whole change set is one update on aws_security_group.worker_group_mgmt_one; the world then moved out of band (the VPC vpc-328dc0c5's Name tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.vpc.aws_vpc.this[0] and the live vpc-328dc0c5 it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - sg-7d8a2a280dbac8a69 still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-7d8a2a280dbac8a69 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The reviewed change was in-place from end to end: sg-7d8a2a280dbac8a69 kept its live id across the whole part, so PART D/E/F/G below still start from the objects STAGE 2 stamped. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 2m35s | 54 resources from nothing, cluster marker verified via the AWS CLI, 54 records under the implied local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on cluster status/version, ASG count/desired-capacities, and cluster-owned security-group count | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 14m4.9s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 15m20.1s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed 9717459fd3 (2026-08-18). Crossed by an agent whose worktree, like the original rds-complete-postgres crossing, predates the lambda-simple crossing's live-import child-module fix (cec3c4b9b1) - migrate is recorded fail here ('3 of 4 root-module resources stamped, 50 non-root instances not considered') because that's what the agent's own stale base observed. Given rds-complete-postgres's own re-verification confirmed cec3c4b9b1 generalizes cleanly to a different estate's module-nested resources, this row very likely also improves once re-verified against current main - flagged for the same re-verification treatment, not yet done. test_plan independently fails on unadmitted default_*/VPN-gateway types (overlapping #305's pattern), undeclared-record-store logical resources, and 7 correctly-conservative count-index refusals (element() over a list, which the checker cannot statically prove injective - not a defect). Two real, generalizing floci gaps found and fixed but NOT YET merged/published: EKS worker AMI discovery (AWS-owned AMI catalog lookup, owner 602401143452/801119661308, evaluated unconditionally by every terraform-aws-eks version) and AutoScalingGroup SuspendProcesses/ResumeProcesses (unimplemented, but the AWS provider calls SuspendProcesses unconditionally around ASG creation on default settings - blocks any aws_autoscaling_group apply, not just EKS's). Filed as lex00/floci#55, PR lex00/floci#56, pushed but not merged - the currently-pinned floci image predates both fixes. Follow-up pass 2026-08-19: confirmed lex00/floci#55/PR#56 (EKS worker AMI catalog, ASG SuspendProcesses/ResumeProcesses) merged and published, current pin well past that point - cold_deploy passes cleanly with no FLOCI_IMAGE override, no new floci gap found. migrate was recorded fail on stale script assertions only (checking for issue #59's old root-module-only live-import scope) - the real, current result is a clean PASS: cec3c4b9b1's child-module live-import fix (already confirmed reaching corpus-rds-complete-postgres) generalizes here too, 25 of 54 instances eligible and stamped (0 failed), 29 skipped legitimately (27 untaggable-by-design, 2 unadmitted-type). test_plan's real wall shrank sharply once migrate covers the whole module tree: #305's default_* trio and the VPN-gateway unadmitted sites are gone entirely (admitted and stamping cleanly now); real remaining causes are kubernetes_config_map.aws_auth (1 site, a non-AWS-tag provider resource with no marker/discovery path in this codebase at all - filed as #326, DEFER/design-call caliber matching #309, not attempted), 4 correctly-RULE'd logical-resource refusals (3 correct no-record_store refusals plus #314's already-tracked local_file gap), and 4 genuinely-unresolvable count-index sites (element(subnet[*].id, count.index), PARITY/RULE, correctly conservative). run.sh's own header/assertions rewritten and re-verified (clean run plus a BREAK=1 mutation check) to match this real picture. Follow-up pass 2026-08-20 (issue #326 fix, re-crossed for real): the prior handoff note claiming #326's commit (9131275487) had "fully landed on local main" was WRONG - it sat on an unmerged branch (live/kubernetes-table-rows) whose base predated #331, #337, #343/#344, #348, #349 and the dynamodb/autoscaling crossing work. Merged it into a fresh worktree off current local main (f294af5838), resolved the generated-artifact conflicts by combining both sides' deltas and re-running the real generators rather than hand-computing totals (852f52073f the merge, a990112e26 the regenerated derived artifacts), full fast tier green. Re-crossed this estate for real against a live floci container with that merge: kubernetes_config_map.aws_auth's "unadmitted-type" refusal is CONFIRMED GONE - zero occurrences of "Rule: unadmitted-type." and zero mentions of "kubernetes" anywhere in live-plan's output, asserted as a negative control in run.sh with BREAK=1 flipping the expectation. The 4 logical-resource and 4 count-index sites are unchanged (8 Error diagnostics total, asserted by count) - test_plan stays fail, this estate does NOT reach 5/5. Migrate's own accounting is also unchanged in total (25/54 eligible, 29 skipped) but changed in KIND for this one site: kubernetes_config_map.aws_auth moves from an outright "unadmitted" refusal to a genuinely-attempted verification that reports MISSING with a precise, different, real reason - "Provider ... kubernetes could not be used ... Dynamic value in static context: Unable to use data.aws_eks_cluster_auth.cluster / data.aws_eks_cluster.cluster in static context, which is required by provider.kubernetes." The kubernetes provider block is itself configured from another provider's live output (the EKS cluster's endpoint/token), which live-import's no-state, no-apply verification pass cannot evaluate. This is a distinct, real, narrower wall than #326's - #326 admitted the TYPE (identity resolves, plan-time refusal gone); this is a separate provider-config-from-live-data limitation, same family as #313's own out-of-scope boundary, DEFER-caliber (stock OpenTofu is never asked this question - a real plan/apply always has other resources' already-applied state to read for the same data sources). Not attempted here, not filed as a fresh issue since it blocks nothing #326 was scoped to fix and this estate's test_plan stage was already failing for the RULE/PARITY reasons above regardless. Commented on #326 with this measured outcome. diff --git a/site/content/docs/progress/corpus-evoteum-modules.md b/site/content/docs/progress/corpus-evoteum-modules.md index 2511026e3c..de9cf0872e 100644 --- a/site/content/docs/progress/corpus-evoteum-modules.md +++ b/site/content/docs/progress/corpus-evoteum-modules.md @@ -12,26 +12,26 @@ Set: core. Lane: opentofu-native. Why it is in the core set: a real project built for OpenTofu specifically, so OpenTofu-only surface is exercised -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 29s | 10 resources added (1 vpc, 3 subnets, 1 igw, 1 route table, 3 associations, 1 dynamodb table); confirmed unmarked | -| Migrate | pass | 36s | 7 of 10 verified and stamped, 0 failed, 3 correctly UNTAGGABLE; markers read back via the AWS CLI | +| Cold deploy | pass | 12s | 10 resources added (1 vpc, 3 subnets, 1 igw, 1 route table, 3 associations, 1 dynamodb table); confirmed unmarked | +| Migrate | pass | 39s | 7 of 10 verified and stamped, 0 failed, 3 correctly UNTAGGABLE; markers read back via the AWS CLI | | Replan from nothing | pass | 3s | no changes; VPC and table markers unchanged, all three untaggable associations resolved by their composite identity | -| No-op apply | pass | 3s | no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 7, no state file | -| Drift and reconverge | pass | 4s | VPC Name tag tampered out of band, exactly 1 object proposed and reconverged, marker survived the incremental tag update | +| No-op apply | pass | 2s | no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 7, no state file | +| Drift and reconverge | pass | 6s | VPC Name tag tampered out of band, exactly 1 object proposed and reconverged, marker survived the incremental tag update | | Rename | pass | 10s | moved block: module.networking renamed with zero churn (0 add, 6 change, 0 destroy), marker rewritten in place across its taggable objects including the untaggable route-table-association children resolving structurally; live-mv: module.sessions_table renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 7s | choudoufu: deleting module.sessions_table_renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), address-for-address identical to stock's oracle on cold_deploy's own state (module.sessions_table); applied cleanly (0 added, 0 changed, 1 destroyed); the table is genuinely gone from the live account (describe-table now returns ResourceNotFoundException, read via the AWS CLI, not choudoufu's own report), and the next plan is empty; classifyOrphans did not withhold the destroy because no other module.sessions_table* block is declared anywhere in this config | -| Change count | pass | 14s | choudoufu: dropping the last public_subnets CIDR (10.0.103.0/24) destroyed exactly its subnet and route-table-association instances (0 add, 0 change, 2 destroy), leaving both survivor subnets' live ids and tofu-address markers unchanged; the destroyed subnet's local record is tombstoned, not deleted (#398-guard shape, asserted by value); restoring the CIDR created exactly the same two instances under NEW live ids (subnet id and association id both server-minted, verified directly against floci with no tofu in the loop before writing this assertion) while both survivors stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical public_subnets edit, applied plan-only on cold_deploy's own state, shows the identical shape: destroy the dropped CIDR's subnet and association only, create them back under new ids, every other subnet/association's id unchanged both times | +| Remove a block | pass | 6s | choudoufu: deleting module.sessions_table_renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), address-for-address identical to stock's oracle on cold_deploy's own state (module.sessions_table); applied cleanly (0 added, 0 changed, 1 destroyed); the table is genuinely gone from the live account (describe-table now returns ResourceNotFoundException, read via the AWS CLI, not choudoufu's own report), and the next plan is empty; classifyOrphans did not withhold the destroy because no other module.sessions_table* block is declared anywhere in this config | +| Change count | pass | 13s | choudoufu: dropping the last public_subnets CIDR (10.0.103.0/24) destroyed exactly its subnet and route-table-association instances (0 add, 0 change, 2 destroy), leaving both survivor subnets' live ids and tofu-address markers unchanged; the destroyed subnet's local record is tombstoned, not deleted (#398-guard shape, asserted by value); restoring the CIDR created exactly the same two instances under NEW live ids (subnet id and association id both server-minted, verified directly against floci with no tofu in the loop before writing this assertion) while both survivors stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical public_subnets edit, applied plan-only on cold_deploy's own state, shows the identical shape: destroy the dropped CIDR's subnet and association only, create them back under new ids, every other subnet/association's id unchanged both times | | Replace with create_before_destroy | pass | 13s | choudoufu: changing module.sessions_table_renamed's ForceNew table_name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old table (arn:aws:dynamodb:us-west-2:000000000000:table/evtx-development-sessions) is confirmed gone and the new table (evtx-development-sessions-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new table's name, not the destroyed one (evtx-development-sessions -> evtx-development-sessions-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied - it shares floci's account with $ESTATE); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 27s | 10 resources from nothing (1 vpc, 3 subnets, 1 igw, 1 route table, 3 untaggable associations, 1 dynamodb table), VPC marker verified via the AWS CLI, 10 records in the local record store (#364 A2, one per managed instance), replan empty, stock oracle in its own namespace matches structurally on vpc/subnets/igw/route-table/dynamodb-table | +| Plan, review, apply | pass | 12s | one argument edited (module.networking's internet gateway gains a Reviewed=yes tag; an in-place update, never a replace, because PART G below re-reads this estate's subnet ids by value), "plan -out=approved.tfplan" wrote a 12570-byte stock-format plan file whose whole change set is one update on module.networking.aws_internet_gateway.main; the world then moved out of band (vpc-b4200bde's Name tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.networking.aws_vpc.main and the live vpc-b4200bde it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - igw-676f386d still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and igw-676f386d read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The revert is proven byte-for-byte against the corpus pin (copy_modules' own diff, re-run) and the three subnets PART G re-reads are still the three cold_deploy minted. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 26s | 10 resources from nothing (1 vpc, 3 subnets, 1 igw, 1 route table, 3 untaggable associations, 1 dynamodb table), VPC marker verified via the AWS CLI, 10 records in the local record store (#364 A2, one per managed instance), replan empty, stock oracle in its own namespace matches structurally on vpc/subnets/igw/route-table/dynamodb-table | | Strict profile (not a headline stage) | not run | | | -Last run at commit `cb5ae2009f` on 2026-09-06T04:29:56Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 2m26.6s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 2m22.4s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed 2026-08-19 (crossing f388f7891c, merge a749395e48), the eighth OpenTofu-native estate and the second from a commercial vendor. OpenTofu-native evidence, four independent kinds, all asserted by the script: self-description as OpenTofu's with no compatibility claim; zero .tf files in the whole pinned tree (109 .tofu, 0 .tf); .pre-commit-config.yaml running tofuutils/pre-commit-opentofu's tofu_validate/tofu_fmt with no Terraform hook; and .tofutest.hcl unit tests, the first evidence in this lane Terraform could not parse. Production code - the org's own estate-config repo calls aws/bucket from it over a setproduct() for_each. Scoped to the two of eleven AWS modules that are self-contained; nine excluded with stated reasons, aws/bucket because its name folds random_password.result (the secret-bearing twin of the random_pet wall corpus-lambda-simple already hits). One delta, asserted at exactly one line: aws/networking/main.tofu's aws constraint ~> 5.0 -> = 6.59.0 (tofu init refuses the root's own pin otherwise); aws/dynamodb byte-identical. Real run rc=0. cold_deploy: 10 instances, both for_each expansions confirmed at 3 keys. migrate: 7 of 10 stamped, 3 UNTAGGABLE route table associations. test_plan: EMPTY with the state file deleted, VPC/subnet/route-table/table markers re-read through the AWS CLI and all three untaggable associations independently confirmed as their {route_table_id}/{subnet_id} composite says. test_apply: genuine no-op, 7 objects before and after, no state file either time. drift_reconverge: exactly module.networking.aws_vpc.main proposed and fixed, marker surviving the incremental tag update. Both negative controls (BREAK=1, BREAK_STAGE5=1) verified failing in real full runs. First estate in either lane whose for_each keys fall outside the AWS tag-value charset ([A-Za-z0-9 _.:/=+@-]), so EscapeAddress is load-bearing: the marker is module.networking.aws_subnet.public:10@d0@d101@d0/24, and the expected values are hand-written from internal/live/markers/markers.go:196's own rule rather than computed by the code under test, each checked to be inside the AWS charset. No Go code touched; nothing in this estate refused - live-plan's diagnostic surface is empty, not merely small. justfile gained demo-corpus-evoteum-modules (port 4730); live/corpus-manifest.json gained the pin; HANDOFF.md section 3 updated, including a correction of its own stale 'four of six' figure to three of six (this manifest was always the source of truth and never agreed with that prose claim). diff --git a/site/content/docs/progress/corpus-giantswarm-crossplane.md b/site/content/docs/progress/corpus-giantswarm-crossplane.md index 6546473ea6..6d73976486 100644 --- a/site/content/docs/progress/corpus-giantswarm-crossplane.md +++ b/site/content/docs/progress/corpus-giantswarm-crossplane.md @@ -12,26 +12,26 @@ Set: core. Lane: opentofu-native. Why it is in the core set: a real project built for OpenTofu specifically, so OpenTofu-only surface is exercised -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 39s | 6 resource instances added, 0 already tofu-estate-marked before migration | -| Migrate | pass | 24s | 2 of 6 stamped (role, managed policy), 4 untaggable skipped, module's own tags survived the stamp | -| Replan from nothing | pass | 4s | live-plan empty, role/policy tofu-address unchanged, both *_exclusive resources re-derived by value | -| No-op apply | pass | 4s | no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, both exclusive sets unchanged | -| Drift and reconverge | pass | 5s | role's installation tag tampered, exactly the IAM role proposed and reconciled, apply changed 1, tag reads back as configured | -| Rename | pass | 11s | moved block: module.crossplane renamed to .crossplane_renamed with zero churn (0 add, 2 change, 0 destroy - role and policy), markers rewritten in place; live-mv: .crossplane_renamed renamed to .crossplane_final with zero churn, both markers rewritten in place (one live-mv call per taggable object); stock oracle over the same chained module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 7s | choudoufu: deleting module.crossplane_final's block proposed 6 resource action(s), address-for-address and action-for-action identical to stock's oracle on cold_deploy's own state; applied cleanly; the role is genuinely gone from the live account (get-role now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report); classifyOrphans did not withhold any destroy because no other module.crossplane* block is declared anywhere in this config; the next plan is empty | -| Change count | pass | 39s | choudoufu: scaling aws_iam_role.count_test from 2 to 1 proposed exactly "count_test[1] will be destroyed" (0 add, 0 change, 1 destroy) and applied it, leaving count_test[0]'s server-minted RoleId (AROAPDYSMRCJISNRTSLK), its CreateDate and its tofu-address=aws_iam_role.count_test:0 marker all unchanged, and tombstoning count_test[1]'s local record (has tombstone, no identity - the #398-guard shape); scaling back from 1 to 2 proposed exactly "count_test[1] will be created" (1 add, 0 change, 0 destroy) and brought it back under the SAME deterministic name (giantswarm-crossplane-count-test-1) with a NEW RoleId (AROAKLXLOOPVVBJ8L4P2 -> AROAX6KFELD5P9JCWZF5) and a new CreateDate, re-marked aws_iam_role.count_test:1 and re-identified in the record store, while count_test[0] stayed untouched throughout; the next plan is empty. Every identity here is read back through the AWS CLI and the local record store, never through choudoufu's own report, and the destroy witness is the RoleId rather than the name or the ARN because both of those are deterministic from configuration and come back identical - confirmed against floci directly, no tofu in the loop, before the assertions were written. Stock oracle (G-ORACLE): real tofu standing the IDENTICAL count block up in the idle greenfield-oracle account showed the identical shape - destroy the higher index only, create the higher index back under the same name with a new RoleId (AROA4HHSNGXJQD7KLPND -> AROA9QH5CL9MJ5020J83), the lower index's RoleId and CreateDate unchanged both times - and the two sides' normalised action sets are compared literally, not just described. Synthetic block, per live/GAUNTLET.md #8's sanctioned fallback: the pinned crossplane module declares no count at all and its only two for_each knobs (aws_iam_role_policy.additional_policies over var.additional_policies, aws_iam_role_policy_attachment.additional_policy_attachments over toset(var.additional_policies_arns)) are both UNTAGGABLE types that carry no marker to keep an identity in, the second provably resolving to zero instances, and the first's inline-policy set is additionally policed by aws_iam_role_policies_exclusive in the same module; aws_iam_role.count_test reuses a type this estate already exercises and lives in its own day2_count.tofu beside the estate's root wiring ($ESTATE), so the vendored module stays byte-identical. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, proving the which-instance assertion is load-bearing. | -| Replace with create_before_destroy | pass | 10s | choudoufu: changing module.crossplane_final's ForceNew installation_name argument proposed a 6 add / 0 change / 6 destroy cascade with the role and the managed policy each explicitly named 'must be replaced' at their same declared addresses, applied cleanly; the old role (giantswarm-gsprereqs-crossplane) is confirmed gone and the new role (giantswarm-gsprereqs-v2-crossplane) carries the marker, both via the AWS CLI; the local record store's record at the role's address now names the new role, not the destroyed one (giantswarm-gsprereqs-crossplane -> giantswarm-gsprereqs-v2-crossplane); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes an equal add/destroy cascade (>=2) with role and policy both replaced at the same addresses (plan only, not applied - it shares floci's account with $ESTATE); BREAK=replace confirms a manufactured marker collision is reported loudly (a named 'Live resource displaced from the address it is marked for' warning, the scalar-resource shape) rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one. | +| Cold deploy | pass | 5s | 6 resource instances added, 0 already tofu-estate-marked before migration | +| Migrate | pass | 23s | 2 of 6 stamped (role, managed policy), 4 untaggable skipped, module's own tags survived the stamp | +| Replan from nothing | pass | 2s | live-plan empty, role/policy tofu-address unchanged, both *_exclusive resources re-derived by value | +| No-op apply | pass | 3s | no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, both exclusive sets unchanged | +| Drift and reconverge | pass | 4s | role's installation tag tampered, exactly the IAM role proposed and reconciled, apply changed 1, tag reads back as configured | +| Rename | pass | 10s | moved block: module.crossplane renamed to .crossplane_renamed with zero churn (0 add, 2 change, 0 destroy - role and policy), markers rewritten in place; live-mv: .crossplane_renamed renamed to .crossplane_final with zero churn, both markers rewritten in place (one live-mv call per taggable object); stock oracle over the same chained module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Remove a block | pass | 5s | choudoufu: deleting module.crossplane_final's block proposed 6 resource action(s), address-for-address and action-for-action identical to stock's oracle on cold_deploy's own state; applied cleanly; the role is genuinely gone from the live account (get-role now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report); classifyOrphans did not withhold any destroy because no other module.crossplane* block is declared anywhere in this config; the next plan is empty | +| Change count | pass | 31s | choudoufu: scaling aws_iam_role.count_test from 2 to 1 proposed exactly "count_test[1] will be destroyed" (0 add, 0 change, 1 destroy) and applied it, leaving count_test[0]'s server-minted RoleId (AROAWKJGZCMEJKQD17GA), its CreateDate and its tofu-address=aws_iam_role.count_test:0 marker all unchanged, and tombstoning count_test[1]'s local record (has tombstone, no identity - the #398-guard shape); scaling back from 1 to 2 proposed exactly "count_test[1] will be created" (1 add, 0 change, 0 destroy) and brought it back under the SAME deterministic name (giantswarm-crossplane-count-test-1) with a NEW RoleId (AROAN4V2DL70N90344J2 -> AROAIQ2RZND9EW25FEJU) and a new CreateDate, re-marked aws_iam_role.count_test:1 and re-identified in the record store, while count_test[0] stayed untouched throughout; the next plan is empty. Every identity here is read back through the AWS CLI and the local record store, never through choudoufu's own report, and the destroy witness is the RoleId rather than the name or the ARN because both of those are deterministic from configuration and come back identical - confirmed against floci directly, no tofu in the loop, before the assertions were written. Stock oracle (G-ORACLE): real tofu standing the IDENTICAL count block up in the idle greenfield-oracle account showed the identical shape - destroy the higher index only, create the higher index back under the same name with a new RoleId (AROAS5MQ3SZOZQY5MORH -> AROAQELS5MR4KB0L9WAG), the lower index's RoleId and CreateDate unchanged both times - and the two sides' normalised action sets are compared literally, not just described. Synthetic block, per live/GAUNTLET.md #8's sanctioned fallback: the pinned crossplane module declares no count at all and its only two for_each knobs (aws_iam_role_policy.additional_policies over var.additional_policies, aws_iam_role_policy_attachment.additional_policy_attachments over toset(var.additional_policies_arns)) are both UNTAGGABLE types that carry no marker to keep an identity in, the second provably resolving to zero instances, and the first's inline-policy set is additionally policed by aws_iam_role_policies_exclusive in the same module; aws_iam_role.count_test reuses a type this estate already exercises and lives in its own day2_count.tofu beside the estate's root wiring ($ESTATE), so the vendored module stays byte-identical. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, proving the which-instance assertion is load-bearing. | +| Replace with create_before_destroy | pass | 7s | choudoufu: changing module.crossplane_final's ForceNew installation_name argument proposed a 6 add / 0 change / 6 destroy cascade with the role and the managed policy each explicitly named 'must be replaced' at their same declared addresses, applied cleanly; the old role (giantswarm-gsprereqs-crossplane) is confirmed gone and the new role (giantswarm-gsprereqs-v2-crossplane) carries the marker, both via the AWS CLI; the local record store's record at the role's address now names the new role, not the destroyed one (giantswarm-gsprereqs-crossplane -> giantswarm-gsprereqs-v2-crossplane); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes an equal add/destroy cascade (>=2) with role and policy both replaced at the same addresses (plan only, not applied - it shares floci's account with $ESTATE); BREAK=replace confirms a manufactured marker collision is reported loudly (a named 'Live resource displaced from the address it is marked for' warning, the scalar-resource shape) rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 20s | 6 resources from nothing (role, managed policy, 4 untaggable), role marker verified via the AWS CLI, 6 records in the local record store (#364 A2, one per managed instance), replan empty, stock oracle in its own namespace matches structurally on the role and the managed policy | +| Plan, review, apply | pass | 11s | one argument edited (additional_policies["extra-tagging"] widened to allow ec2:DeleteTags as well), "plan -out=approved.tfplan" wrote a 10347-byte stock-format plan file whose whole change set is one update on module.crossplane.aws_iam_role_policy.additional_inline_policies["extra-tagging"]; the world then moved out of band (giantswarm-gsprereqs-crossplane's installation tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.crossplane.aws_iam_role.giantswarm_crossplane_role and the live giantswarm-gsprereqs-crossplane it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - the inline policy read back through the AWS CLI still allowed only ec2:CreateTags, which is stronger evidence than the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the inline policy read back allowing ec2:DeleteTags, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 12s | 6 resources from nothing (role, managed policy, 4 untaggable), role marker verified via the AWS CLI, 6 records in the local record store (#364 A2, one per managed instance), replan empty, stock oracle in its own namespace matches structurally on the role and the managed policy | | Strict profile (not a headline stage) | not run | | | -Last run at commit `d72960cdc3` on 2026-09-06T05:35:09Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 2m42.7s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 1m53.3s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed 2026-08-19 (crossing 0bd3ac80b7, merge 9fa0141294), the seventh OpenTofu-native estate and the first from a commercial vendor's production repository rather than a module registry, personal monorepo, or single-maintainer accelerator - Giant Swarm GmbH's own customer-facing account-prep for their managed Kubernetes offering. OpenTofu-native evidence, three independent kinds: README's opening sentence and directory index both say OpenTofu with no compatibility hedge; the CI workflow is named 'OpenTofu checks', installs via opentofu/setup-opentofu, and never mentions terraform; the crossed crossplane/ directory is genuinely .tofu-suffixed throughout (providers.tofu, role.tofu, variables.tofu), the file-level standard only the hongbomiao slices had met before (overture-tiles and xancloud-iac are both plain .tf). Scoped to crossplane/ specifically: self-contained (no remote state, no live EKS/OIDC dependency, its only data source is aws_partition which makes no API call), the other five directories in the repo excluded with stated reasons (three plain-.tf same-shape, one an account-singleton quota table, one a wrapper that only calls the others). Real run, rc=0, 548s. cold_deploy: PASS, plain tofu apply, 6 resources added, 0 pre-existing tofu-estate tags, the toset()-keyed for_each on additional_policy_attachments confirmed resolving to zero instances. migrate: PASS - 2 of 6 eligible (UNTAGGABLE 2, UNADMITTED_TYPE 2, DRIFTED 2), -approve 2 newly stamped 0 failed, both markers re-verified directly through the AWS CLI including that the module's own installation tag survived the stamp. test_plan: BLOCKED for real at exactly 2 sites, the plan's entire diagnostic surface - both Rule: unadmitted-type, on aws_iam_role_policy_attachments_exclusive and aws_iam_role_policies_exclusive, no other rule firing anywhere in the estate. A control stage (3b, not counted toward stage 4/5) cut exactly those two resource blocks and drove the rest of the pipeline for real: control test plan EMPTY, control test apply a genuine no-op (2 objects before and after), control drift-and-reconverge fixed exactly one mutated object - proving the estate's only real block is those two types, not routing around anything. Both negative controls (BREAK=1 at the stage-2 identity assertion, BREAK_STAGE3=1 expecting 3 refusal sites where the real count is 2) verified failing in real full runs. Filed as INTENTIUS/choudoufu#334: both unadmitted types have the identical import-grammar shape (single-argument, no-separator) to aws_vpc_security_group_rules_exclusive, which #307 already admitted via row-gen's tryGrammarComposite at 64cac28120, and carry the same no-CFN-counterpart mapping-gen overlay as that admitted twin - a worked ADMIT-class precedent, not attempted here; the one recorded difference (force_new on the admitted twin's security_group_id, absent on either new type's role_name) is flagged as not obviously the gate since tryGrammarComposite's single-argument branch reads no force_new field, and why row-gen's own proposal for these two is currently absent from ratified.json is the fix's first open question. No Go code touched. justfile gained demo-corpus-giantswarm-crossplane; live/corpus-manifest.json gained the pin; HANDOFF.md section 3 updated. Follow-up pass 2026-08-19/20 (#334 fixed and merged, 37957d873c/a6627543c4/20cc1774d6): the issue's own open question resolved the OPPOSITE way from what it suspected - row-gen was never declining to propose these two rows; it proposes both, byte-identical in shape to the admitted aws_vpc_security_group_rules_exclusive twin, and nobody had ever ratified the proposal. Fix is a ratified.json entry, no code change - reach is exactly these two types, stated plainly rather than dressed up as a generalization. The real finding: 316 types row-gen proposes sit unratified, 166 under this exact rule (including every other *_exclusive family member) - a ratification backlog, not a generator defect, and clearing it is a maintainer-scale call since every ratified row is a claim that touches live infrastructure. FIVE OF FIVE, run for real on the rebased tree, rc=0, 250s: stage 3 now an empty plan (identities asserted by value - the two exclusive resources carry no marker, being untaggable, so the assertion is the live content each enforces, e.g. attached policy ARN and inline-policy name); stage 4 a genuine no-op (2 tagged objects before and after, both exclusive sets independently re-read afterward so a wrongly-reconciled enforcer would be caught); stage 5 one out-of-band mutation, exactly one object proposed and fixed. Three real negative controls (BREAK=1, BREAK_STAGE3=1, BREAK_STAGE5=1) each confirmed failing at the right point. Script rewritten: stage 3 now asserts a pass instead of hard-failing by design, the 3b control retired with the block it existed to control for. Separately found, not yet fixed: every crossing script that runs more than one `terraform init` pays a real ~320s tax per extra init, because the shared plugin cache records no checksums and a directory with no `.terraform.lock.hcl` re-downloads the whole provider to compute them - seeding the lock file from stage 1's own init cuts this to ~1s and this estate's full run from several failed 10-minute-cap attempts to 250s total. Worth a sweep across every multi-init script here. diff --git a/site/content/docs/progress/corpus-hongbomiao-harbor.md b/site/content/docs/progress/corpus-hongbomiao-harbor.md index da771b74ca..2ab04b5854 100644 --- a/site/content/docs/progress/corpus-hongbomiao-harbor.md +++ b/site/content/docs/progress/corpus-hongbomiao-harbor.md @@ -12,26 +12,26 @@ Set: core. Lane: opentofu-native. Why it is in the core set: a real project built for OpenTofu specifically, so OpenTofu-only surface is exercised -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 27s | Apply complete! Resources: 3 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=hongbomiao-harbor-crossing before migration | -| Migrate | pass | 18s | 2 of 3 stamped (bucket, user), 1 UNTAGGABLE (inline policy); bucket hongbomiao-harbor-crossing-hm-harbor -> tofu-address=module.s3_bucket_hm_harbor.aws_s3_bucket.main, user hongbomiao-harbor-crossing-hm-harbor-user -> tofu-address=module.harbor_iam_user.aws_iam_user.hm_harbor_iam_user | -| Replan from nothing | pass | 2s | empty plan; identity re-check: bucket and user tofu-address unchanged, inline policy's resource ARN still matches the configuration | -| No-op apply | pass | 3s | genuine no-op: 2 objects before, 2 after, no state file either time | +| Cold deploy | pass | 15s | Apply complete! Resources: 3 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=hongbomiao-harbor-crossing before migration | +| Migrate | pass | 17s | 2 of 3 stamped (bucket, user), 1 UNTAGGABLE (inline policy); bucket hongbomiao-harbor-crossing-hm-harbor -> tofu-address=module.s3_bucket_hm_harbor.aws_s3_bucket.main, user hongbomiao-harbor-crossing-hm-harbor-user -> tofu-address=module.harbor_iam_user.aws_iam_user.hm_harbor_iam_user | +| Replan from nothing | pass | 3s | empty plan; identity re-check: bucket and user tofu-address unchanged, inline policy's resource ARN still matches the configuration | +| No-op apply | pass | 2s | genuine no-op: 2 objects before, 2 after, no state file either time | | Drift and reconverge | pass | 5s | the plan proposed fixing 1 object(s) after the out-of-band tag mutation: module.s3_bucket_hm_harbor.aws_s3_bucket.main | -| Rename | pass | 9s | moved block: module.s3_bucket_hm_harbor renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.harbor_iam_user renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Rename | pass | 8s | moved block: module.s3_bucket_hm_harbor renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.harbor_iam_user renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | | Remove a block | pass | 7s | choudoufu: deleting module.harbor_iam_user_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy - the untaggable inline policy and its taggable parent user), applied cleanly (0 added, 0 changed, 2 destroyed) in an order IAM accepted, the user is genuinely gone from the live account (iam get-user on the old name now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly two destroys for the same objects | -| Change count | pass | 43s | choudoufu: scaling aws_iam_user.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live UserId and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under the SAME ARN (deterministic from name+path) but a NEW UserId (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied for real in the idle greenfield account, shows the identical shape: destroy the higher index only, create the higher index back under the same ARN but a new UserId, the lower index's UserId unchanged both times | -| Replace with create_before_destroy | pass | 6s | choudoufu: changing the s3_bucket_name argument feeding module.harbor_iam_user_renamed's inline policy proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly, with the user itself completely untouched; the old inline policy (S3ReadWritePolicy-hongbomiao-harbor-crossing-hm-harbor) is confirmed gone from hongbomiao-harbor-crossing-hm-harbor-user and the new one (S3ReadWritePolicy-hongbomiao-harbor-crossing-hm-harbor-policy-v2) exists in its place, both via the AWS CLI; the local record store's record at the same address now names the new composite identity, not the destroyed one (hongbomiao-harbor-crossing-hm-harbor-user:S3ReadWritePolicy-hongbomiao-harbor-crossing-hm-harbor -> hongbomiao-harbor-crossing-hm-harbor-user:S3ReadWritePolicy-hongbomiao-harbor-crossing-hm-harbor-policy-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) confirms both that aws_iam_user's own name argument is NOT ForceNew (updated in-place, not replaced - the reason this section targets the inline policy instead of the user) and that the inline policy itself IS force-replaced the same way. Scope notes: (1) this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see corpus-sqs-basic's own PART F; (2) BREAK=replace's marker-collision control is not exercised here - aws_iam_user_policy is untaggable and resolved structurally, with no marker to plant a collision on, so that control's load-bearing-ness is proven instead by corpus-evoteum-modules and corpus-giantswarm-crossplane's own PART F sections against the taggable shape. | +| Change count | pass | 45s | choudoufu: scaling aws_iam_user.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live UserId and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under the SAME ARN (deterministic from name+path) but a NEW UserId (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied for real in the idle greenfield account, shows the identical shape: destroy the higher index only, create the higher index back under the same ARN but a new UserId, the lower index's UserId unchanged both times | +| Replace with create_before_destroy | pass | 7s | choudoufu: changing the s3_bucket_name argument feeding module.harbor_iam_user_renamed's inline policy proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly, with the user itself completely untouched; the old inline policy (S3ReadWritePolicy-hongbomiao-harbor-crossing-hm-harbor) is confirmed gone from hongbomiao-harbor-crossing-hm-harbor-user and the new one (S3ReadWritePolicy-hongbomiao-harbor-crossing-hm-harbor-policy-v2) exists in its place, both via the AWS CLI; the local record store's record at the same address now names the new composite identity, not the destroyed one (hongbomiao-harbor-crossing-hm-harbor-user:S3ReadWritePolicy-hongbomiao-harbor-crossing-hm-harbor -> hongbomiao-harbor-crossing-hm-harbor-user:S3ReadWritePolicy-hongbomiao-harbor-crossing-hm-harbor-policy-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) confirms both that aws_iam_user's own name argument is NOT ForceNew (updated in-place, not replaced - the reason this section targets the inline policy instead of the user) and that the inline policy itself IS force-replaced the same way. Scope notes: (1) this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see corpus-sqs-basic's own PART F; (2) BREAK=replace's marker-collision control is not exercised here - aws_iam_user_policy is untaggable and resolved structurally, with no marker to plant a collision on, so that control's load-bearing-ness is proven instead by corpus-evoteum-modules and corpus-giantswarm-crossplane's own PART F sections against the taggable shape. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 1m28s | 3 resources from nothing (bucket, user, untaggable inline policy), markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared | +| Plan, review, apply | pass | 11s | one argument edited (harbor_iam_user's common_tags gain Reviewed=yes), "plan -out=approved.tfplan" wrote a 8658-byte stock-format plan file whose whole change set is one update on module.harbor_iam_user.aws_iam_user.hm_harbor_iam_user; the world then moved out of band (hongbomiao-harbor-crossing-hm-harbor's hm_team tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.s3_bucket_hm_harbor.aws_s3_bucket.main and the live hongbomiao-harbor-crossing-hm-harbor it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - hongbomiao-harbor-crossing-hm-harbor-user still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and hongbomiao-harbor-crossing-hm-harbor-user read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 1m18s | 3 resources from nothing (bucket, user, untaggable inline policy), markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m28.6s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m18.2s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed 2026-08-19, the sixth estate in the OpenTofu-native lane and the fourth to clear all five stages. Sourced per HANDOFF's own suggestion to scope a second (here, third) disjoint slice of the already-crossed hongbomiao monorepo before a fresh search. Surveyed every remaining AWS environment: network/main.tofu is pure data sources (nothing to migrate); kubernetes/main.tofu builds a full terraform-aws-modules/eks cluster and every IAM module in it but one (velero_iam_role, mimir_iam_role, loki_iam_role, tempo_iam_role, label_studio_iam_role, etc., 15 total) takes amazon_eks_cluster_oidc_provider(_arn) from that same cluster - the same scope/risk class as the terraform-popular lane's already-blocked terraform-aws-eks examples/basic crossing. The one exception, the "Harbor" section (S3 bucket + IAM user + inline user policy), needs no EKS cluster, no OIDC provider, no remote state at all - self-contained like storage's own scoped slice. Nebius/Cloudflare/Snowflake environments confirmed to still exist and be real, actively-maintained infrastructure, but target non-AWS clouds floci cannot emulate. Crosses aws_iam_user/aws_iam_user_policy, a genuinely different resource pair from Labelbox's aws_iam_role/aws_iam_role_policy - both already-ratified DefaultTable rows, no schema-fallback warning. All five stages verified for real against a live floci container: cold_deploy (tofu apply, 3 resources created, confirmed 0 pre-existing tofu-estate tags), migrate (live-import verified 2 of 3 eligible - bucket + user - 1 correctly UNTAGGABLE - the inline policy; markers re-read via AWS CLI matched: module.s3_bucket_hm_harbor.aws_s3_bucket.main, module.harbor_iam_user.aws_iam_user.hm_harbor_iam_user), test_plan (state deleted, live-plan empty, identities re-verified against the AWS CLI including the inline policy's resource ARN read directly off the live object), test_apply (genuine no-op, 2 tagged objects before and after), drift_reconverge (bucket tag tampered out of band, plan proposed fixing exactly that one object, reconverge apply restored it). BREAK=1 verified load-bearing, failing exactly at the stage-2 identity assertion. No floci or choudoufu gaps found - this crossing is clean. Merged to local main as ad2cf81cf3 (crossing itself: 0c4e16af6a); justfile gained recipe demo-corpus-hongbomiao-harbor (port 4728); no new live/corpus-manifest.json entry needed, reuses the existing hongbomiao pin. diff --git a/site/content/docs/progress/corpus-hongbomiao-labelbox.md b/site/content/docs/progress/corpus-hongbomiao-labelbox.md index f0b6a085a7..79200d0470 100644 --- a/site/content/docs/progress/corpus-hongbomiao-labelbox.md +++ b/site/content/docs/progress/corpus-hongbomiao-labelbox.md @@ -12,26 +12,26 @@ Set: core. Lane: opentofu-native. Why it is in the core set: a real project built for OpenTofu specifically, so OpenTofu-only surface is exercised -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 55s | 4 resources added, 0 objects carry tofu-estate=hongbomiao-labelbox-crossing before migration | -| Migrate | pass | 45s | 2 of 4 stamped (2 skipped, untaggable), 0 failed; markers read back via the AWS CLI | -| Replan from nothing | pass | 4s | no resource change proposed; bucket and role tofu-address unchanged, CORS origins and inline policy resource match config | -| No-op apply | pass | 4s | no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, no state file | -| Drift and reconverge | pass | 7s | bucket tag drifted; exactly module.amazon_s3_bucket_hm_labelbox.aws_s3_bucket.main proposed, applied (1 changed), reconverged to hongbomiao | +| Cold deploy | pass | 16s | 4 resources added, 0 objects carry tofu-estate=hongbomiao-labelbox-crossing before migration | +| Migrate | pass | 37s | 2 of 4 stamped (2 skipped, untaggable), 0 failed; markers read back via the AWS CLI | +| Replan from nothing | pass | 3s | no resource change proposed; bucket and role tofu-address unchanged, CORS origins and inline policy resource match config | +| No-op apply | pass | 3s | no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, no state file | +| Drift and reconverge | pass | 6s | bucket tag drifted; exactly module.amazon_s3_bucket_hm_labelbox.aws_s3_bucket.main proposed, applied (1 changed), reconverged to hongbomiao | | Rename | pass | 10s | moved block: module.amazon_s3_bucket_hm_labelbox renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.labelbox_iam_role renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | | Remove a block | pass | 9s | choudoufu: deleting module.labelbox_iam_role_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy - the untaggable inline policy and its taggable parent role), applied cleanly (0 added, 0 changed, 2 destroyed) in an order IAM accepted, the role is genuinely gone from the live account (iam get-role on the old name now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly two destroys for the same objects | -| Change count | pass | 23s | choudoufu: scaling aws_iam_policy.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live PolicyId and tofu-address/tofu-slot markers unchanged, and tombstoning count_test[1]'s own record in place (has tombstone, no identity - the #398-guard shape, not file absence); scaling back from 1 to 2 created exactly count_test[1] under the SAME ARN (deterministic from name+path) but a NEW PolicyId (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout, and its record regained a live identity alongside its kept tombstone; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied fresh in the idle greenfield account, shows the identical shape: destroy the higher index only, create the higher index back under the same ARN but a new PolicyId, the lower index's PolicyId unchanged both times; BREAK_COUNT=1 confirms the wrong-instance assertion above is load-bearing | -| Replace with create_before_destroy | pass | 9s | choudoufu: changing labelbox_service_account_name proposed exactly one role replace at module.labelbox_iam_role_renamed's declared address, cascading into its untaggable inline policy (also replaced, role and name are both ForceNew there) - 2 to add, 0 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old role is confirmed terminated (NoSuchEntity) and the new role carries the marker, both via the AWS CLI; the local record store's records at the same addresses now name the new role's import_id and the new role:name pair, not the destroyed ones (role LabelboxRole-hm-labelbox -> LabelboxRole-hm-labelbox-v2; policy LabelboxRole-hm-labelbox:LabelboxRoleS3Policy-hm-labelbox -> LabelboxRole-hm-labelbox-v2:LabelboxRoleS3Policy-hm-labelbox-v2) - the same untaggable-identity path this estate's greenfield fix resolves for a from-nothing apply, now proven under a real replace; the next plan proposes no resource action. Scope notes: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names (module lifecycle blocks are rejected by OpenTofu core, and the vendored module stays byte-identical to the pinned commit throughout - see this section's own header); and a manufactured live-object collision (this stage's own Break text) goes undetected by an ordinary plan for this resource shape, confirmed by instrumenting discovery.bind() directly - a fully record-backed type's declared population is excluded from that function's per-type claimant sweep, a real finding this unit records rather than fixes (identity-path change, HANDOFF's stop-and-report territory); BREAK_REPLACE instead proves this section's own plan-shape assertion is load-bearing. | +| Change count | pass | 24s | choudoufu: scaling aws_iam_policy.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live PolicyId and tofu-address/tofu-slot markers unchanged, and tombstoning count_test[1]'s own record in place (has tombstone, no identity - the #398-guard shape, not file absence); scaling back from 1 to 2 created exactly count_test[1] under the SAME ARN (deterministic from name+path) but a NEW PolicyId (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout, and its record regained a live identity alongside its kept tombstone; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied fresh in the idle greenfield account, shows the identical shape: destroy the higher index only, create the higher index back under the same ARN but a new PolicyId, the lower index's PolicyId unchanged both times; BREAK_COUNT=1 confirms the wrong-instance assertion above is load-bearing | +| Replace with create_before_destroy | pass | 8s | choudoufu: changing labelbox_service_account_name proposed exactly one role replace at module.labelbox_iam_role_renamed's declared address, cascading into its untaggable inline policy (also replaced, role and name are both ForceNew there) - 2 to add, 0 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old role is confirmed terminated (NoSuchEntity) and the new role carries the marker, both via the AWS CLI; the local record store's records at the same addresses now name the new role's import_id and the new role:name pair, not the destroyed ones (role LabelboxRole-hm-labelbox -> LabelboxRole-hm-labelbox-v2; policy LabelboxRole-hm-labelbox:LabelboxRoleS3Policy-hm-labelbox -> LabelboxRole-hm-labelbox-v2:LabelboxRoleS3Policy-hm-labelbox-v2) - the same untaggable-identity path this estate's greenfield fix resolves for a from-nothing apply, now proven under a real replace; the next plan proposes no resource action. Scope notes: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names (module lifecycle blocks are rejected by OpenTofu core, and the vendored module stays byte-identical to the pinned commit throughout - see this section's own header); and a manufactured live-object collision (this stage's own Break text) goes undetected by an ordinary plan for this resource shape, confirmed by instrumenting discovery.bind() directly - a fully record-backed type's declared population is excluded from that function's per-type claimant sweep, a real finding this unit records rather than fixes (identity-path change, HANDOFF's stop-and-report territory); BREAK_REPLACE instead proves this section's own plan-shape assertion is load-bearing. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 55s | 4 resources from nothing (bucket, CORS config, role, untaggable inline role policy), markers verified via the AWS CLI, 4 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared | +| Plan, review, apply | pass | 14s | one argument edited (labelbox_iam_role's common_tags gain Reviewed=yes), "plan -out=approved.tfplan" wrote a 11141-byte stock-format plan file whose whole change set is one update on module.labelbox_iam_role.aws_iam_role.labelbox_iam_role; the world then moved out of band (hongbomiao-labelbox-crossing-hm-labelbox's hm_team tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.amazon_s3_bucket_hm_labelbox.aws_s3_bucket.main and the live hongbomiao-labelbox-crossing-hm-labelbox it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - LabelboxRole-hm-labelbox still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and LabelboxRole-hm-labelbox read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 48s | 4 resources from nothing (bucket, CORS config, role, untaggable inline role policy), markers verified via the AWS CLI, 4 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared | | Strict profile (not a headline stage) | not run | | | -Last run at commit `d72960cdc3` on 2026-09-06T05:35:09Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m41s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 2m57.5s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed 2026-08-18 as the second estate in the OpenTofu-native lane and the first to clear all five stages there. Stronger OpenTofu-native evidence than corpus-sumaform-aws (which only describes itself as OpenTofu-native but ships plain .tf once its .example template is copied in): every file under infrastructure/opentofu/ genuinely uses the .tofu extension, its own justfile drives init/plan/apply/refresh/destroy exclusively via `tofu`, and common_tags carries "hm_managed_by" = "opentofu" - proven rather than asserted, since the crossing script's own stock terraform init against this estate reports "The directory has no Terraform configuration files." Scoped to the self-contained "Labelbox" slice (S3 bucket, its CORS configuration, an IAM role with an inline S3-read policy - three real leaf modules copied byte-identical from the pinned commit, diffed programmatically in the script) out of a much larger monorepo (AWS+Nebius+Cloudflare+Snowflake+EKS, cross-wired via terraform_remote_state) too large to stand up in one sitting - the same scoping convention corpus-sumaform-aws's module.base stand-in uses. All five stages verified for real: cold_deploy (tofu apply, 4 resources, confirmed unmarked via resourcegroupstaggingapi), migrate (live-import: "2 of 4 resource instance(s) are eligible for stamping" - 2 correctly UNTAGGABLE, the CORS config via provider-schema fallback and the inline policy via the generated table's composite ROLENAME:POLICYNAME identity; -approve stamped both taggable resources, markers verified directly via aws s3api get-bucket-tagging / aws iam list-role-tags), test_plan (state deleted, live-plan "No changes", identities re-checked against the AWS CLI including the two untaggable resources' own content - CORS AllowedOrigins, inline policy's Resource ARN - since they carry no tag to re-read), test_apply (genuine no-op, 2 tagged objects before and after), and drift_reconverge (the bucket's hm_team tag tampered out of band, plan proposed fixing exactly that object, apply reconverged it; BREAK=1 verified load-bearing for both stage 2's identity check and stage 5's single-object assertion, tested in isolation for stage 5 per the corpus-vpc-complete convention since the shared BREAK var fails fast at stage 2 otherwise). Two non-blocking findings documented in the script's own header rather than routed around: aws_s3_bucket/aws_iam_role report DRIFTED during verification from AWS's own deprecated cors_rule/inline_policy shadow attributes reflecting a sibling resource created after the state snapshot (harmless, resolves by plan time), and the schema-admitted aws_s3_bucket_cors_configuration triggers the already-documented "Resource type has no orphan recovery" warning (live/LIMITATIONS.md, not a new gap). No choudoufu or floci gaps found - nothing filed. Merged to local main as c7fb650f4c (fix itself: 30577f6a56); justfile gained recipe demo-corpus-hongbomiao-labelbox; live/corpus-manifest.json gained the pin (reproducibility only, same convention as the sumaform entry - contributes nothing to a corpus-gen number). diff --git a/site/content/docs/progress/corpus-hongbomiao-storage.md b/site/content/docs/progress/corpus-hongbomiao-storage.md index efeefc78ef..ad0facb605 100644 --- a/site/content/docs/progress/corpus-hongbomiao-storage.md +++ b/site/content/docs/progress/corpus-hongbomiao-storage.md @@ -12,26 +12,26 @@ Set: core. Lane: opentofu-native. Why it is in the core set: a real project built for OpenTofu specifically, so OpenTofu-only surface is exercised -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 33s | Apply complete! Resources: 4 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=hongbomiao-storage-crossing before migration | -| Migrate | pass | 43s | 3 of 4 stamped (2 buckets, KMS key), 1 UNTAGGABLE (KMS alias); bucket hongbomiao-storage-crossing-hm-production -> tofu-address=module.hm_production_bucket.aws_s3_bucket.main, bucket hongbomiao-storage-crossing-hm-iot-data -> tofu-address=module.s3_bucket_iot_data.aws_s3_bucket.main, key e2ea5441-c9cd-4f92-85b7-4207a2c8c29a -> tofu-address=module.kafka_kms_key.aws_kms_key.main | -| Replan from nothing | pass | 3s | empty plan; identity re-check: both buckets' and the key's tofu-address unchanged, KMS alias still points at the same key | +| Cold deploy | pass | 20s | Apply complete! Resources: 4 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=hongbomiao-storage-crossing before migration | +| Migrate | pass | 52s | 3 of 4 stamped (2 buckets, KMS key), 1 UNTAGGABLE (KMS alias); bucket hongbomiao-storage-crossing-hm-production -> tofu-address=module.hm_production_bucket.aws_s3_bucket.main, bucket hongbomiao-storage-crossing-hm-iot-data -> tofu-address=module.s3_bucket_iot_data.aws_s3_bucket.main, key b1f75b83-386c-46e8-86bb-ef14f4299ed7 -> tofu-address=module.kafka_kms_key.aws_kms_key.main | +| Replan from nothing | pass | 4s | empty plan; identity re-check: both buckets' and the key's tofu-address unchanged, KMS alias still points at the same key | | No-op apply | pass | 3s | genuine no-op: 3 objects before, 3 after, no state file either time | -| Drift and reconverge | pass | 6s | the plan proposed fixing 1 object(s) after the out-of-band tag mutation: module.s3_bucket_iot_data.aws_s3_bucket.main | -| Rename | pass | 15s | moved block: module.hm_production_bucket renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.kafka_kms_key renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 8s | choudoufu: deleting module.kafka_kms_key_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy - the untaggable alias and its taggable parent key), applied cleanly (0 added, 0 changed, 2 destroyed) in an order the cloud accepted, the key is genuinely PendingDeletion and the alias is gone (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly two destroys for the same objects | +| Drift and reconverge | pass | 5s | the plan proposed fixing 1 object(s) after the out-of-band tag mutation: module.s3_bucket_iot_data.aws_s3_bucket.main | +| Rename | pass | 14s | moved block: module.hm_production_bucket renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.kafka_kms_key renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Remove a block | pass | 7s | choudoufu: deleting module.kafka_kms_key_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy - the untaggable alias and its taggable parent key), applied cleanly (0 added, 0 changed, 2 destroyed) in an order the cloud accepted, the key is genuinely PendingDeletion and the alias is gone (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly two destroys for the same objects | | Change count | pass | 24s | choudoufu: scaling aws_s3_bucket.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live CreationDate and tofu-address marker unchanged and tombstoning count_test[1]'s local record (has tombstone, no identity - the #398-guard shape); scaling back from 1 to 2 created exactly count_test[1] under the SAME bucket name (deterministic) but a NEW CreationDate (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied for real in the idle greenfield real-leg account, shows the identical shape: destroy the higher index only, create the higher index back under the same bucket name but a new CreationDate, the lower index's CreationDate unchanged both times | -| Replace with create_before_destroy | pass | 12s | choudoufu: changing module.kafka_kms_key_renamed's aws_kms_key_name argument proposed exactly one forced replace at the same declared address (the untaggable, client-named alias - 1 add, 1 change, 1 destroy overall) plus one in-place tag update on the taggable key itself, applied cleanly; the old alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key) is confirmed gone and the new alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2) points at the SAME key (e2ea5441-c9cd-4f92-85b7-4207a2c8c29a, read via the AWS CLI) - the key was never replaced; the local record store's record at the alias's address now names the new alias, not the destroyed one (alias/hongbomiao-storage-crossing-hm-kafka-kms-key -> alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2), while the key's own record at its own address is unchanged; the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace (the alias) plus one in-place key update. Scope notes: (1) this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see corpus-sqs-basic's own PART F; (2) BREAK=replace's marker-collision control is not exercised here - aws_kms_alias is untaggable and resolved by its own name, with no marker to plant a collision on, so that control's load-bearing-ness is proven instead by corpus-evoteum-modules and corpus-giantswarm-crossplane's own PART F sections against the taggable shape. | +| Replace with create_before_destroy | pass | 12s | choudoufu: changing module.kafka_kms_key_renamed's aws_kms_key_name argument proposed exactly one forced replace at the same declared address (the untaggable, client-named alias - 1 add, 1 change, 1 destroy overall) plus one in-place tag update on the taggable key itself, applied cleanly; the old alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key) is confirmed gone and the new alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2) points at the SAME key (b1f75b83-386c-46e8-86bb-ef14f4299ed7, read via the AWS CLI) - the key was never replaced; the local record store's record at the alias's address now names the new alias, not the destroyed one (alias/hongbomiao-storage-crossing-hm-kafka-kms-key -> alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2), while the key's own record at its own address is unchanged; the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace (the alias) plus one in-place key update. Scope notes: (1) this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see corpus-sqs-basic's own PART F; (2) BREAK=replace's marker-collision control is not exercised here - aws_kms_alias is untaggable and resolved by its own name, with no marker to plant a collision on, so that control's load-bearing-ness is proven instead by corpus-evoteum-modules and corpus-giantswarm-crossplane's own PART F sections against the taggable shape. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 1m3s | 4 resources from nothing (2 buckets under aws.production, KMS key and untaggable alias under the default aws provider), markers verified via the AWS CLI, 4 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object per provider namespace, marker tags never compared | +| Plan, review, apply | pass | 15s | one argument edited (hm_production_bucket's common_tags gain Reviewed=yes), "plan -out=approved.tfplan" wrote a 10486-byte stock-format plan file whose whole change set is one update on module.hm_production_bucket.aws_s3_bucket.main; the world then moved out of band (hongbomiao-storage-crossing-hm-iot-data's hm_team tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.s3_bucket_iot_data.aws_s3_bucket.main and the live hongbomiao-storage-crossing-hm-iot-data it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - hongbomiao-storage-crossing-hm-production still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and hongbomiao-storage-crossing-hm-production read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 58s | 4 resources from nothing (2 buckets under aws.production, KMS key and untaggable alias under the default aws provider), markers verified via the AWS CLI, 4 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object per provider namespace, marker tags never compared | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m30.7s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m34.6s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed 2026-08-18, the third estate in the OpenTofu-native lane and the second to clear all five stages, reusing corpus-hongbomiao-labelbox's already-pinned commit rather than a fresh sourcing search (the repo's OpenTofu-native bona fides and the pinned commit's clone were already established by that crossing). Scoped after surveying every section of the monorepo's aws/general and aws/storage files via the GitHub API against the pinned commit, no clone needed for scouting: Kafka Manager, two Amazon EMR sections and AWS Batch all read another environment's terraform_remote_state (out of scope, same reason corpus-hongbomiao-labelbox's own scoping excluded them); Amazon SageMaker was ruled out with a real, confirmed floci gap - aws sagemaker create-notebook-instance against a live floci container returns "UnknownOperationException: Operation CreateNotebookInstance is not supported by floci", and the type has zero entries anywhere in live/floci-capabilities.json's Cloud Control sweep - documented in the script's header as evidence for whoever picks up SageMaker next, not filed as an issue since it was routed around rather than blocking anything. The real candidate: aws/storage/main.tofu's first three module calls (hm_production_bucket, kafka_kms_key, s3_bucket_iot_data) read no remote state at all, unlike everything after them in that file - two amazon_s3_bucket module calls plus one aws_kms_key module call (aws_kms_key + aws_kms_alias). All five stages verified for real against a live floci container: cold_deploy (tofu apply, "4 added, 0 changed, 0 destroyed", confirmed 0 objects pre-tagged), migrate (live-import: "3 of 4 resource instance(s) are eligible for stamping", 1 UNTAGGABLE - the KMS alias; -approve: "3 resource(s) newly stamped, 0 already stamped, 0 failed, 1 skipped"; markers for all three read back via raw AWS CLI matched exactly: module.hm_production_bucket.aws_s3_bucket.main, module.s3_bucket_iot_data.aws_s3_bucket.main, module.kafka_kms_key.aws_kms_key.main), test_plan (state deleted, live-plan "No changes", all three identities re-verified against the AWS CLI, including the untaggable KMS alias's live target), test_apply (genuine no-op, "0 added, 0 changed, 0 destroyed", object count unchanged at 3), and drift_reconverge (the IoT-data bucket's tag tampered out of band, plan proposed fixing exactly module.s3_bucket_iot_data.aws_s3_bucket.main and nothing else, reconverge apply changed exactly 1 resource). BREAK=1 verified load-bearing: correctly fails the stage-2 identity assertion (asserts the KMS key's tofu-address against a deliberately wrong resource name). No choudoufu gaps found beyond the SageMaker floci evidence above - nothing filed against this repo. Merged to local main as a720266bcc (fix itself: 3335f16893); justfile gained recipe demo-corpus-hongbomiao-storage (port 4725); no new live/corpus-manifest.json entry needed, reuses the existing hongbomiao pin. diff --git a/site/content/docs/progress/corpus-iam-policy.md b/site/content/docs/progress/corpus-iam-policy.md index 0c830fb949..9ebce2a1f3 100644 --- a/site/content/docs/progress/corpus-iam-policy.md +++ b/site/content/docs/progress/corpus-iam-policy.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 28s | Apply complete! Resources: 2 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=iam-policy-crossing before migration | -| Migrate | pass | 18s | 2 of 2 stamped, both carrying tofu-slot=0/0 read back through IAM (choudoufu #372); Apply complete! Resources: 0 added, 0 changed, 0 destroyed. - nothing left to converge | -| Replan from nothing | pass | 3s | no resource change proposed, nothing foreign; identity re-check (via the AWS CLI) both unchanged | +| Cold deploy | pass | 17s | Apply complete! Resources: 2 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=iam-policy-crossing before migration | +| Migrate | pass | 19s | 2 of 2 stamped, both carrying tofu-slot=0/0 read back through IAM (choudoufu #372); Apply complete! Resources: 0 added, 0 changed, 0 destroyed. - nothing left to converge | +| Replan from nothing | pass | 2s | no resource change proposed, nothing foreign; identity re-check (via the AWS CLI) both unchanged | | No-op apply | pass | 3s | genuine no-op: 2 objects before, 2 after, no state file either time | | Drift and reconverge | pass | 5s | one object tampered (arn:aws:iam::000000000000:policy/example_from_data_source's Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag | | Rename | pass | 9s | moved block: module.iam_policy_from_data_source renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.iam_policy renamed with zero churn, marker rewritten in place (found and fixed live-mv's own missing issue #266 tag-index fallback and the arnJoinTable's missing iam:policy entry to get here); stock oracle over the same two-module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both ARNs unchanged, read via the AWS CLI | -| Remove a block | pass | 8s | choudoufu: deleting module.iam_policy_renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (iam get-policy on the old ARN now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (STAGE 1.5.5) also proposes exactly one destroy for the same object; classifyOrphans did not withhold the destroy even though module.iam_policy_renamed2's policy shares the same block key, because that surviving instance is bound, not unclaimed | -| Change count | pass | 42s | choudoufu: scaling aws_iam_policy.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live arn and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under a NEW arn (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G0 stock oracle on the same 2-instance count block, applied fresh against the idle adopted-estate endpoint, shows the identical shape: destroy the higher index only, create the higher index back under a new arn, the lower index's arn unchanged both times. Synthetic block: this estate's only real count knob (aws_iam_policy.policy's count = var.create ? 1 : 0) is a boolean create toggle, not a scalable set - sanctioned fallback per live/GAUNTLET.md #8 and reference-ec2-vpc's own Part F. | -| Replace with create_before_destroy | pass | 7s | choudoufu: changing module.iam_policy_renamed2's ForceNew name_prefix argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old policy (arn:aws:iam::000000000000:policy/example-246b74803ee893de43f27cac43) is confirmed gone and the new policy (arn:aws:iam::000000000000:policy/example-v2-0f747c1a618bb2900d8fde9713) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's ARN, not the destroyed one (arn:aws:iam::000000000000:policy/example-246b74803ee893de43f27cac43 -> arn:aws:iam::000000000000:policy/example-v2-0f747c1a618bb2900d8fde9713); the next plan proposes no resource action; stock oracle on cold_deploy's own state (STAGE 1.5.6) also proposes exactly one replace at the same address (plan only, not applied). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one. BREAK=replace's manufactured marker collision IS now reported for this type (GitHub issue #411, fixed): 'Indistinguishable instances without per-instance markers' - GitHub issue #409, layered on top of #411's own fix, is why this is not corpus-sqs-basic's 'Two live resources claiming one slot' text despite the same shape - see PART F's own header and its BREAK=replace branch. | +| Remove a block | pass | 7s | choudoufu: deleting module.iam_policy_renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (iam get-policy on the old ARN now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (STAGE 1.5.5) also proposes exactly one destroy for the same object; classifyOrphans did not withhold the destroy even though module.iam_policy_renamed2's policy shares the same block key, because that surviving instance is bound, not unclaimed | +| Change count | pass | 40s | choudoufu: scaling aws_iam_policy.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live arn and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under a NEW arn (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G0 stock oracle on the same 2-instance count block, applied fresh against the idle adopted-estate endpoint, shows the identical shape: destroy the higher index only, create the higher index back under a new arn, the lower index's arn unchanged both times. Synthetic block: this estate's only real count knob (aws_iam_policy.policy's count = var.create ? 1 : 0) is a boolean create toggle, not a scalable set - sanctioned fallback per live/GAUNTLET.md #8 and reference-ec2-vpc's own Part F. | +| Replace with create_before_destroy | pass | 7s | choudoufu: changing module.iam_policy_renamed2's ForceNew name_prefix argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old policy (arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be) is confirmed gone and the new policy (arn:aws:iam::000000000000:policy/example-v2-96244c69014024ce6076c79bba) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's ARN, not the destroyed one (arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be -> arn:aws:iam::000000000000:policy/example-v2-96244c69014024ce6076c79bba); the next plan proposes no resource action; stock oracle on cold_deploy's own state (STAGE 1.5.6) also proposes exactly one replace at the same address (plan only, not applied). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one. BREAK=replace's manufactured marker collision IS now reported for this type (GitHub issue #411, fixed): 'Indistinguishable instances without per-instance markers' - GitHub issue #409, layered on top of #411's own fix, is why this is not corpus-sqs-basic's 'Two live resources claiming one slot' text despite the same shape - see PART F's own header and its BREAK=replace branch. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 55s | 2 resources from nothing (both aws_iam_policy), markers verified via the AWS CLI, 2 records in the local record store (#364 A2), replan empty both with and without the local record store, both policies' documents and paths match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared | +| Plan, review, apply | pass | 12s | one argument edited (module.iam_policy's tags gain Reviewed=yes), "plan -out=approved.tfplan" wrote a 11133-byte stock-format plan file whose whole change set is one update on module.iam_policy.aws_iam_policy.policy[0]; the world then moved out of band (arn:aws:iam::000000000000:policy/example_from_data_source's Example tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.iam_policy_from_data_source.aws_iam_policy.policy[0] and the live arn:aws:iam::000000000000:policy/example_from_data_source it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 45s | 2 resources from nothing (both aws_iam_policy), markers verified via the AWS CLI, 2 records in the local record store (#364 A2), replan empty both with and without the local record store, both policies' documents and paths match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 2m57.6s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 2m46.1s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Upgraded from a real but pre-#274-pipeline predecessor script (choudoufu apply from a live block present from the start, delete state, replan empty twice) to the current five-stage shape, following corpus-vpc-complete/corpus-lambda-simple's structure. Verified for real in a fresh isolated worktree off local main (ff106e63a7), Docker/floci/AWS CLI throughout, not read from the predecessor's prior notes. All five stages pass cleanly: cold_deploy (plain terraform apply, "Apply complete! Resources: 2 added", confirmed 0 objects tagged before migration), migrate (live-import dry run verifies "2 of 2 resource instance(s) are eligible for stamping", -approve reports "2 resource(s) newly stamped, 0 already stamped, 0 failed, 0 skipped", both tofu-address/tofu-estate tags read directly through the AWS CLI: module.iam_policy.aws_iam_policy.policy:0 and module.iam_policy_from_data_source.aws_iam_policy.policy:0), test_plan (live-plan genuinely empty, both identities re-read unchanged after the state file's only copy was deleted), test_apply ("0 added, 0 changed, 0 destroyed", object count unchanged at 2), and drift_reconverge (one policy's Example tag tampered directly against floci, live-plan proposes fixing exactly that object, apply reconverges it to "0 added, 1 changed, 0 destroyed"). BREAK=1 verified twice, independently, against each stage it targets: run as committed it fails stage 3's identity check (expects the real policy's tofu-address on a module that was never created); run separately with stage 3's corruption disabled, it correctly fails stage 5 by tampering a second object and proving the "exactly one object" count assertion is load-bearing (both objects flagged, not silently 1). NEW FINDING, not previously documented in any real crossing that reached this deep: live-import -approve deliberately writes only tofu-estate and tofu-address, never tofu-slot (internal/live/stamp/doc.go's own "tofu-slot comes in from outside" - a slot is minted from a monotonic counter over the live set that a read-only, one-state-file view cannot compute). Both of this estate's aws_iam_policy resources declare count = var.create ? 1 : 0, exactly the shape that needs one, so the FIRST live-plan straight after live-import -approve is not empty - it proposes adding tofu-slot="0" to both, and nothing else. Folded into stage 2 as one ordinary `choudoufu apply` ("0 added, 2 changed, 0 destroyed") before stage 3 is attempted; every replan after is genuinely empty. This is real, deliberate, already-documented product behavior, not a defect - but it will recur on any count-based resource crossing that reaches this far and had not yet been noticed in one that actually got here. Also caught and fixed while verifying: a self-authored bug where stage 5's negative drift assertion compared a live-plan diff header's address (bracket form, "policy[0]") against the escaped tag-value form ("policy:0") and could never have matched - a vacuous check that a stricter assertion in the sibling script (see corpus-iam-read-only-policy) surfaced; fixed here by keeping both forms as separate variables. diff --git a/site/content/docs/progress/corpus-iam-read-only-policy.md b/site/content/docs/progress/corpus-iam-read-only-policy.md index 7220684a6b..c4babe8b24 100644 --- a/site/content/docs/progress/corpus-iam-read-only-policy.md +++ b/site/content/docs/progress/corpus-iam-read-only-policy.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 25s | Apply complete! Resources: 1 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=iam-read-only-policy-crossing before migration | -| Migrate | pass | 1m5s | 1 of 1 stamped, carrying tofu-slot=0 read back through IAM (choudoufu #372); Apply complete! Resources: 0 added, 0 changed, 0 destroyed. - nothing left to converge | +| Cold deploy | pass | 15s | Apply complete! Resources: 1 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=iam-read-only-policy-crossing before migration | +| Migrate | pass | 1m3s | 1 of 1 stamped, carrying tofu-slot=0 read back through IAM (choudoufu #372); Apply complete! Resources: 0 added, 0 changed, 0 destroyed. - nothing left to converge | | Replan from nothing | pass | 3s | no resource change proposed, nothing foreign; identity re-check (via the AWS CLI) unchanged | -| No-op apply | pass | 3s | genuine no-op: 1 objects before, 1 after, no state file either time | -| Drift and reconverge | pass | 5s | one object tampered (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-cb1412a333b8859faa022fa5d1's Example tag), plan proposed fixing exactly module.read_only_iam_policy.aws_iam_policy.policy[0], apply changed 1 and reconverged the tag | -| Rename | pass | 10s | moved block: module.read_only_iam_policy renamed to module.read_only_iam_policy_moved with zero churn (0 add, 1 change, 0 destroy), tofu-address marker rewritten in place; live-mv: module.read_only_iam_policy_moved renamed to module.read_only_iam_policy_final with zero churn, marker rewritten in place; stock oracle over the identical net rename on cold_deploy's own state also shows a true no-op (0 add, 0 change, 0 destroy, outputs unchanged in value); the live policy ARN unchanged throughout, read via the AWS CLI | -| Remove a block | pass | 7s | choudoufu: deleting module.read_only_iam_policy_final's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (iam get-policy on the old ARN now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; classifyOrphans did not withhold the destroy because no other aws_iam_policy.policy block anywhere in this config ever declares a real instance (count=0 on both remaining module calls) | +| No-op apply | pass | 2s | genuine no-op: 1 objects before, 1 after, no state file either time | +| Drift and reconverge | pass | 5s | one object tampered (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7's Example tag), plan proposed fixing exactly module.read_only_iam_policy.aws_iam_policy.policy[0], apply changed 1 and reconverged the tag | +| Rename | pass | 9s | moved block: module.read_only_iam_policy renamed to module.read_only_iam_policy_moved with zero churn (0 add, 1 change, 0 destroy), tofu-address marker rewritten in place; live-mv: module.read_only_iam_policy_moved renamed to module.read_only_iam_policy_final with zero churn, marker rewritten in place; stock oracle over the identical net rename on cold_deploy's own state also shows a true no-op (0 add, 0 change, 0 destroy, outputs unchanged in value); the live policy ARN unchanged throughout, read via the AWS CLI | +| Remove a block | pass | 6s | choudoufu: deleting module.read_only_iam_policy_final's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (iam get-policy on the old ARN now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; classifyOrphans did not withhold the destroy because no other aws_iam_policy.policy block anywhere in this config ever declares a real instance (count=0 on both remaining module calls) | | Change count | pass | 19s | choudoufu: scaling aws_iam_policy.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live PolicyId and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under the SAME ARN (deterministic from name+path) but a NEW PolicyId (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied fresh in the idle greenfield-oracle account, shows the identical shape: destroy the higher index only, create the higher index back under the same ARN but a new PolicyId, the lower index's PolicyId unchanged both times | -| Replace with create_before_destroy | pass | 7s | choudoufu: changing module.read_only_iam_policy_final's ForceNew description argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-cb1412a333b8859faa022fa5d1) is confirmed gone and the new object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-0f29efba6ef044410ac1522988) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-cb1412a333b8859faa022fa5d1 -> arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-0f29efba6ef044410ac1522988); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | +| Replace with create_before_destroy | pass | 8s | choudoufu: changing module.read_only_iam_policy_final's ForceNew description argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7) is confirmed gone and the new object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-84581510d3c9ebbf3d8288dffa) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7 -> arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-84581510d3c9ebbf3d8288dffa); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 45s | 1 resource from nothing, marker verified via the AWS CLI, 1 record in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (path, description, policy document) | +| Plan, review, apply | pass | 19s | one argument edited (aws_iam_policy.approval_probe's Reviewed tag, no -> yes), "plan -out=approved.tfplan" wrote a 27324-byte stock-format plan file whose whole change set is one update on aws_iam_policy.approval_probe; the world then moved out of band (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7's Example tag, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.read_only_iam_policy.aws_iam_policy.policy[0] and the live arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7 it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - arn:aws:iam::000000000000:policy/example/iam-ro-approval-probe still read Reviewed=no through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:iam::000000000000:policy/example/iam-ro-approval-probe read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The reviewed object is a self-contained synthetic aws_iam_policy.approval_probe (sanctioned fallback per live/GAUNTLET.md #8, same discipline as PART G's count_test) because this estate has exactly ONE real object and the leg needs two disjoint rows; it is created in P0 and destroyed in P5, and the module policy's ARN and Example tag are read back unchanged so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 41s | 1 resource from nothing, marker verified via the AWS CLI, 1 record in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (path, description, policy document) | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m9.5s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m11.2s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Upgraded from a real but pre-#274-pipeline predecessor script (choudoufu apply from a live block present from the start, delete state, replan empty twice) to the current five-stage shape, same upgrade as corpus-iam-policy and following the same corpus-vpc-complete/corpus-lambda-simple structure. Verified for real in a fresh isolated worktree off local main (ff106e63a7), Docker/floci/AWS CLI throughout. All five stages pass cleanly: cold_deploy (plain terraform apply, "Apply complete! Resources: 1 added" - only the first of this module's three instantiations contributes a resource, confirmed 0 objects tagged before migration), migrate (live-import dry run verifies "1 of 1 resource instance(s) are eligible for stamping", -approve reports "1 resource(s) newly stamped, 0 already stamped, 0 failed, 0 skipped", tofu-address=module.read_only_iam_policy.aws_iam_policy.policy:0 read directly through the AWS CLI against the real, server-assigned name IAM minted), test_plan (live-plan genuinely empty, identity re-read unchanged after the state file's only copy was deleted), test_apply ("0 added, 0 changed, 0 destroyed", object count unchanged at 1), and drift_reconverge (the one policy's Example tag tampered directly against floci, live-plan proposes fixing exactly that object by address, apply reconverges to "0 added, 1 changed, 0 destroyed"). Same tofu-slot finding as corpus-iam-policy (see that entry) applies identically here - the module's aws_iam_policy also declares count = var.create && var.create_policy ? 1 : 0 - and is folded into stage 2 the same way ("0 added, 1 changed, 0 destroyed" before stage 3 is attempted). Genuine constraint found here that corpus-iam-policy does not share: this estate creates exactly ONE real object (the other two module calls contribute nothing), so stage 5's BREAK=1 cannot prove non-vacuousness by tampering a second object the way corpus-iam-policy's or corpus-vpc-complete's can - there isn't one. Used the address-corruption technique instead (the same shape and resource type, naming a module that in fact creates nothing) and verified it independently at both sites it appears: run as committed it fails stage 3's identity check; run separately with stage 3's corruption disabled, it correctly fails stage 5's exact-address assertion instead. That second, isolated verification also caught a real bug in this script's first draft: the exact-equality comparison against a live-plan diff header's address (bracket form, "policy[0]") was written against the escaped tag-value form ("policy:0") and could never have matched even on the correct object - fixed by keeping both address forms as separate variables, and the fix was re-verified with a full clean pass afterward. diff --git a/site/content/docs/progress/corpus-lambda-simple.md b/site/content/docs/progress/corpus-lambda-simple.md index 081942a1f7..c5a74862fa 100644 --- a/site/content/docs/progress/corpus-lambda-simple.md +++ b/site/content/docs/progress/corpus-lambda-simple.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 28s | 8 resources, genuinely cold, genuinely unmarked | +| Cold deploy | pass | 13s | 8 resources, genuinely cold, genuinely unmarked | | Migrate | pass | 14s | 3 stamped, 4 recorded, 0 failed, 1 skipped | -| Replan from nothing | pass | 3s | no resource change proposed | +| Replan from nothing | pass | 2s | no resource change proposed | | No-op apply | pass | 5s | no-op apply (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 3; markers and record store intact | -| Drift and reconverge | pass | 18s | one object tampered (memory_size 128->256), exactly module.lambda_function.aws_lambda_function.this[0] proposed by both choudoufu and stock with the identical change, apply changed 1 and memory_size reads back as 128 | -| Rename | pass | 24s | moved block: module.lambda_function renamed to module.lambda_function_moved with zero churn (0 add, 3 change, 0 destroy) across all seven of its stateful children, three taggable markers rewritten in place, three record-located children moved via their own per-resource moved blocks with zero diff, one config-derived child (aws_iam_role_policy.logs) needing none; stock oracle over the identical seven-resource move on cold_deploy's own state also shows zero churn beyond the module's own pre-existing null_resource.archive[0] package-timestamp noise (confirmed present on an unrelated baseline replan too); live-mv: module.lambda_function_moved renamed to module.lambda_function_final across all three taggable children (the function, the role, the log group), one call each, zero churn, markers rewritten in place - the internal/live/mv/mv.go materialize() RecordStore wiring gap (build.go:1676's "Record-backed instance with no record store") is fixed; all three live objects unchanged throughout, read via the AWS CLI; final replan is empty | +| Drift and reconverge | pass | 19s | one object tampered (memory_size 128->256), exactly module.lambda_function.aws_lambda_function.this[0] proposed by both choudoufu and stock with the identical change, apply changed 1 and memory_size reads back as 128 | +| Rename | pass | 25s | moved block: module.lambda_function renamed to module.lambda_function_moved with zero churn (0 add, 3 change, 0 destroy) across all seven of its stateful children, three taggable markers rewritten in place, three record-located children moved via their own per-resource moved blocks with zero diff, one config-derived child (aws_iam_role_policy.logs) needing none; stock oracle over the identical seven-resource move on cold_deploy's own state also shows zero churn beyond the module's own pre-existing null_resource.archive[0] package-timestamp noise (confirmed present on an unrelated baseline replan too); live-mv: module.lambda_function_moved renamed to module.lambda_function_final across all three taggable children (the function, the role, the log group), one call each, zero churn, markers rewritten in place - the internal/live/mv/mv.go materialize() RecordStore wiring gap (build.go:1676's "Record-backed instance with no record store") is fixed; all three live objects unchanged throughout, read via the AWS CLI; final replan is empty | | Remove a block | pass | 11s | choudoufu: deleting module.lambda_function_final's block proposed 7 destroys (the function, the role, its inline aws_iam_role_policy.logs[0] CloudWatch Logs policy, and all three record-located children always; the log group's only when floci's GetResources happens to index it - a documented emulator gap, confirmed by reading logs:list-tags-for-resource directly against the same live object), applied cleanly, the function, the role and the inline log policy genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no further resource action; classifyOrphans did not withhold any destroy as a possible rename. WANT_DESTROY_COUNT moved from 5/6 to 6/7 in this same commit: the inline log policy was previously missing from this stage's own checklist entirely - a genuine leak (an untaggable IAM permission left behind on every destroy of this estate), not a stale assertion, fixed as part of the day2_replace unit that re-measured this stage | -| Change count | pass | 26s | synthetic count block, and the header says why: .corpus/lambda declares 38 count/for_each expressions and not one is scalable - 36 are boolean create toggles (count = ... ? 1 : 0), the for_each maps are empty in this example, and the only two numeric ones (number_of_policies, number_of_policy_jsons) are gated off by default and drive the untaggable aws_iam_role_policy_attachment, so turning either on would both change the estate every earlier stage measured and leave no marker to read back. So aws_iam_role.count_test (count = 2), a self-contained block of a type this estate already exercises, added after Part E's completed removal. choudoufu: scaling 2 -> 1 proposed exactly 0 add, 0 change, 1 destroy, destroying count_test[1] and never naming count_test[0]; after the apply lambda-simple-count-test-1 no longer answers GetRole at all (verified absence) while count_test[0] kept its server-minted RoleId AROAZY2RIPLGDLB3N0TZ and its tofu-address=aws_iam_role.count_test:0 marker, both read through the AWS CLI. Scaling 1 -> 2 proposed exactly 1 add, 0 change, 0 destroy and count_test[1] came back as a genuinely NEW object - RoleId AROAGXAJJPJ2ED9EO668 -> AROAVNKGVMUSSSKUZC9V under the same deterministic name, the witness an IAM role's ARN cannot provide since it is rebuilt from account plus name (the same reason this estate's own Lambda function ARN could not have witnessed it) - carrying tofu-address=aws_iam_role.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G0): the identical block stood up for real with plain terraform in its own working directory on the idle greenfield-oracle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new RoleId (AROACC4A5TI4EAICMRGS -> AROATTFBWT3X96T6C4AB), the lower index's RoleId (AROAVPYZM4YHIDNFQUC6) unchanged both times. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, so the assertion is load-bearing | -| Replace with create_before_destroy | pass | 18s | choudoufu: changing module.lambda_function_final's ForceNew logging_log_group-derived name proposed exactly one replace at the same declared address (the log group; -/+ destroy and then create) cascading into two expected in-place updates (the function's logging_config, the inline log policy's document) and nothing else beyond the module's own pre-existing null_resource.archive[0] package-timestamp noise; applied cleanly; the old object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/united-mantis-lambda-simple) is confirmed gone and the new object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/united-mantis-lambda-simple-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (/aws/lambda/united-mantis-lambda-simple -> /aws/lambda/united-mantis-lambda-simple-v2); the next plan proposes no resource action beyond the same known noise; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace plus the same in-place cascade (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | +| Change count | pass | 28s | synthetic count block, and the header says why: .corpus/lambda declares 38 count/for_each expressions and not one is scalable - 36 are boolean create toggles (count = ... ? 1 : 0), the for_each maps are empty in this example, and the only two numeric ones (number_of_policies, number_of_policy_jsons) are gated off by default and drive the untaggable aws_iam_role_policy_attachment, so turning either on would both change the estate every earlier stage measured and leave no marker to read back. So aws_iam_role.count_test (count = 2), a self-contained block of a type this estate already exercises, added after Part E's completed removal. choudoufu: scaling 2 -> 1 proposed exactly 0 add, 0 change, 1 destroy, destroying count_test[1] and never naming count_test[0]; after the apply lambda-simple-count-test-1 no longer answers GetRole at all (verified absence) while count_test[0] kept its server-minted RoleId AROAC7ERQHW7CEUA1EZL and its tofu-address=aws_iam_role.count_test:0 marker, both read through the AWS CLI. Scaling 1 -> 2 proposed exactly 1 add, 0 change, 0 destroy and count_test[1] came back as a genuinely NEW object - RoleId AROATSWACNJVMMY25Q6P -> AROACJAD70F33ERTS33F under the same deterministic name, the witness an IAM role's ARN cannot provide since it is rebuilt from account plus name (the same reason this estate's own Lambda function ARN could not have witnessed it) - carrying tofu-address=aws_iam_role.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G0): the identical block stood up for real with plain terraform in its own working directory on the idle greenfield-oracle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new RoleId (AROAQ945VI0K0RAZ648E -> AROA97JTCE39WS5SJGQN), the lower index's RoleId (AROAYT6M787L61HL0SJ3) unchanged both times. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, so the assertion is load-bearing | +| Replace with create_before_destroy | pass | 18s | choudoufu: changing module.lambda_function_final's ForceNew logging_log_group-derived name proposed exactly one replace at the same declared address (the log group; -/+ destroy and then create) cascading into two expected in-place updates (the function's logging_config, the inline log policy's document) and nothing else beyond the module's own pre-existing null_resource.archive[0] package-timestamp noise; applied cleanly; the old object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/free-wasp-lambda-simple) is confirmed gone and the new object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/free-wasp-lambda-simple-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (/aws/lambda/free-wasp-lambda-simple -> /aws/lambda/free-wasp-lambda-simple-v2); the next plan proposes no resource action beyond the same known noise; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace plus the same in-place cascade (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 25s | 8 resources from nothing (3 taggable + 5 record-backed/config-derived), all three module-nested markers verified via the AWS CLI, 8 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (runtime, handler, memory, timeout, log-group retention) | +| Plan, review, apply | pass | 15s | one argument edited (module.lambda_function's cloudwatch_logs_retention_in_days, unset -> 14, which reaches module.lambda_function.aws_cloudwatch_log_group.lambda[0] and nothing else - the module call carries no tags argument and every tags-shaped knob it has would reach three children at once), "plan -out=approved.tfplan" wrote a 31790-byte stock-format plan file whose whole change set is that one in-place update; the world then moved out of band (free-wasp-lambda-simple's memory_size 128->256, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming the extra row as " module.lambda_function.aws_lambda_function.this[0] Update free-wasp-lambda-simple" - both module.lambda_function.aws_lambda_function.this[0] and the live identity it was computed against - with "Exit status 3" spelled out for a pipeline; nothing was applied - /aws/lambda/free-wasp-lambda-simple still carried no retentionInDays, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with memory_size put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the log group read back with retentionInDays=14, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (retention unset again, memory_size still 128, next plan proposes no resource action, no state file left behind) so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 24s | 8 resources from nothing (3 taggable + 5 record-backed/config-derived), all three module-nested markers verified via the AWS CLI, 8 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (runtime, handler, memory, timeout, log-group retention) | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 2m51.8s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 2m53.6s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Re-verified 2026-08-18 in a fresh isolated worktree off local main (bcf78bacbd), for real (Docker/floci/AWS CLI, not read from a prior note). Stages 1 and 2 still pass exactly as landed: cold apply creates 8 resources, live-import verifies '3 of 8 resource instance(s) are eligible for stamping', -approve reports '3 resource(s) newly stamped, 0 already stamped, 0 failed, 5 skipped', and module.lambda_function.{aws_lambda_function,aws_iam_role,aws_cloudwatch_log_group} carry the expected module-qualified tofu-address/tofu-estate tags read straight through the AWS CLI. #303 (the count=var.enable_x?1:0 zero-instance admission gap on aws_lambda_function_url.this and aws_lambda_function_recursion_config.this) is CONFIRMED FIXED: re-running live-plan against current main, neither type appears anywhere in the diagnostics any more, as an error or a warning - stage 3 now fails on exactly one Error block, not two-plus. That block is local_file.archive_plan (module.lambda_function's package.tf:44, count = var.create && var.create_package ? 1 : 0, both true by default in this example so a real non-zero instance, not a zero-count block #303's fix would clear), refused under the logical-resource rule. Investigated whether this is a bug or correct behavior: it is correct, deliberate, and already ruled on. Issues #237 and #238 (both closed 2026-08-18) put local_file through exactly this question and #238's closing comment states local_file is 'deliberately left OTHER_REFUSED with a documented reason: neither of lint's two classes fits it correctly (its identity is argument-derived, not record-backed, and promoting it would silently reopen a count.index collision hazard a dedicated test already guards) - a genuine third-classification gap, not an omission, correctly left open rather than forced.' local_file's identity is a filename on the local disk of whatever machine ran apply - not a cloud object, nothing taggable, nothing an AWS CLI call could ever read back to confirm it still exists - so there is no live counterpart for a stateless replan to reconcile against. Considered scoping the estate around it the way corpus-vpc-complete/corpus-sumaform-aws scope around their own out-of-scope resources (the module's create_package=false + local_existing_package= toggle skips package.tf's local_file entirely) and rejected it: unlike sumaform's provision=false, which picks between the module's own equally-real published deployment modes to route around an infra-emulation gap in floci, swapping to a pre-built zip would replace the actual thing 'simple' demonstrates - the module's own default packaging pipeline - with a materially different scenario this corpus entry was never meant to test. Left as a real, reported block; run.sh's header carries the full investigation. One piece of relevant good news found along the way: #275 (closed 2026-08-18) built a record_store-gated residue mechanism for exactly the aws_lambda_function.filename/source_code_hash/publish phantom-diff problem a filename-deployed Lambda would otherwise hit under stateless replanning - this estate already declares a record_store, so once local_file gets its own identity class nothing here looks likely to re-hit that problem. test_apply and drift_reconverge remain not_run because test_plan does not pass; not attempted this pass since attempting them against a still-refused plan would prove nothing. No issue currently tracks the missing 'argument-derived-but-safe' LogicalClass itself (the actual unblock for this estate) - #237/#238 are both closed and did not spawn a follow-up; one may be worth filing if this crossing is prioritized again. Follow-up pass 2026-08-18 (#313 cross-check, fresh worktree off main at d303d9d425, real Docker/floci/AWS CLI run): confirmed the local_file.archive_plan block above is NOT #313's data.aws_availability_zones/static-context wall - grepped the full live-plan output, zero occurrences of that diagnostic. Filed the missing-LogicalClass follow-up this note flagged as worth filing but hadn't been: #314 ('local_file needs a fourth LogicalClass (argument-derived identity)'), citing #237/#238's rulings and the count-index guard test. No fix attempted - real multi-package work (lint + identity resolution + row-gen), not a quick derivation. Follow-up pass 2026-08-19 (#314 fixed and merged, d878aa914b/e7bb1f4b41): the fourth LogicalClass exists, but not the one the issue's own framing predicted - two of that framing's premises turned out false, checked rather than assumed. hashicorp/local's provider implements NO ImportState for local_file at all (confirmed against stock tofu: 'This resource does not support import'), so an argument-derived identity handed to internal/live/projection would have turned a lint refusal into a hard Cannot-import-for-projection error - strictly worse. And this estate's own filename argument isn't static anyway (it reads a data.external result). The real fix is ClassExternalAdmitted/EXTERNAL_ADMITTED: a record_store admits local_file the same way it admits ClassRecordAdmitted, resolving through ClassRecordBacked - the record is the only carrier that can bring prior state back for a type the provider itself cannot re-derive. Reaches exactly one type (local_file; local_sensitive_file is secret-bearing) but the RULE generalizes: live/logical-schemas.json's per-provider store_only now SELECTS between the two admitted classes instead of gating whether a type derives a row at all, which also retired the hand-written local_sensitive_file exception in ClassifyLogicalType - a net type-name-literal deletion, not an addition. Count-index guard (TestLocalFileKeepsItsCountIndexCheck) confirmed still holding via two separate mutation checks. Real re-crossing: local_file is admitted and appears nowhere in live-plan's diagnostics any more (asserted by absence) - but test_plan stays fail, now BLOCKED at 5 sites, a FOURTH wall newly reached rather than caused. All five trace to one expression, function_name = "${random_pet.this.id}-lambda-simple": random_pet.this is RECORD_ADMITTED so its id lives only in the record store, and the identity resolver declines to read that carrier for the three dependent resources (aws_iam_role.lambda, aws_iam_role_policy.logs, aws_lambda_function.this, aws_cloudwatch_log_group.lambda, one cascade) even though all three are already stamped and CLI-verified by stage 2 of the same run - choudoufu already holds the value and the objects are already marked, so this reads as an identity-resolver gap rather than a missing carrier. Not filed (no issue number assigned) - worth a slot. test_apply/drift_reconverge remain not_run, blocked on this new wall. Two real corrections made along the way: the crossing script had no AWS provider version pin (silently drifted to whatever the newest release was, now pinned =6.59.0 matching corpus-cloudfront's discipline), and live/LIMITATIONS.md's local-file section stated 'no cloud counterpart to reconcile against' as fact - false; the local filesystem is the counterpart, now corrected there too. Follow-up pass 2026-08-19/20 (#336 fixed and merged, 821c769715/c41279989a): #336's own diagnosis was wrong in two places, checked rather than assumed. The identity resolver was NOT declining to read the record-store carrier - resolver.parentPart already read random_pet.this.id correctly on unmodified main. What actually refused was coalesce(): iam.tf/main.tf's role_name/policy_name/log-group-name chains all select through coalesce(var.X, var.Y, "*")-shaped expressions, and resolver.isSymbolic reads only an expression's traversal ROOTS - var/local are never symbolic, so a selection sitting behind a module argument or a local looked entirely static, failed whole-expression evaluation, and had nothing left to try (the decomposition switch is only reached when a resource is named directly inside the expression). Fixed generically (internal/live/identity/coalesce.go, new): resolveCoalesceCall decides which argument the language selects using two proofs (provably-null-or-empty to skip, provably-non-null-non-empty to select), declining the whole call on anything undecidable rather than silently falling through - mutation-tested three ways (drop the non-emptiness proof, fall through on undecidable, remove the call entirely), each caught. Measured reach: refusal-probe sites 16075->15964 (-111), instances 4499->4522 (+23), 12 entries improved across six unrelated sources, 0 worse - schema-less mode, an under-report since it's blind to the record-backed half of this estate's own chain. Real re-crossing: live-plan diagnostics 5->0, the plan runs to completion for the FIRST time - but test_plan still FAILS, now on a genuinely NEW, fifth wall: 'live-plan is not empty', proposing to create every record-backed resource (random_pet.this first) from scratch. Root cause, and #336's second wrong premise: live-import's Approve loop only calls #327's recordResidueFor for a STAMPED entry; a record-backed resource is by definition not stampable (no live cloud object to tag) and hits OutcomeSkipped, continuing past the residue call entirely - so the record store is empty for every record-backed instance after a clean migrate, not populated as #336 assumed. Filed as #340, not attempted - the fix is migrate seeding the record store from the migrated state's own object for every record-backed instance, a sibling call to recordResidueFor on the skipped-because-record-backed path. test_apply/drift_reconverge remain not_run, now blocked on #340 instead of #336's five diagnostics. Follow-up pass 2026-08-20 (#340 fixed and merged, d30daa156f/8d34e3ded9): the issue's own framing was half right - recordResidueFor does NOT gate on 'stamped', it runs for any entry with an *eligible; the real gate is one line earlier, Ratify never building an *eligible for a record-backed type at all. Fixed with Approve's second write path, the sibling of the tag write: projection.SeedRecordForInstance writes a record-backed instance's object into the record store, byte-identical to what an apply's WriteBack would write, reading before writing so an already-correct record is a no-op and a genuinely different one refuses rather than clobbers. Keys on identity.TypeIdentity.RecordBacked and nothing else - 15 types across 4 providers (local/null/random/time/terraform_data), no aws_*/random_* name in the control flow. Real re-crossing: STAGE 2 migrate reports '3 newly stamped, 0 already stamped, 4 newly recorded, 0 already recorded, 0 failed, 1 skipped', the store's own files grepped for random_pet.this's generated id. STAGE 3: live-plan now raises ZERO diagnostics for the first time ever on this estate, every identity resolves, no record-backed resource is proposed for creation - but the plan is not empty: 0 to add, 2 to change, 0 to destroy, a SIXTH wall. Both changes are real and distinct from every prior wall: (1) a nested-block round-trip on aws_lambda_function (- environment {} / + logging_config { log_format = "Text" }, floci's Lambda read vs the module's config), (2) a sensitivity-only diff on local_file.content (OpenTofu's own renderer says 'The value is unchanged' - a genuine limitation the fixing agent found and pinned rather than hid: ResourceInstanceObjectSrc.Decode re-applies AttrSensitivePaths so the decoded value is marked, ctyjson.Marshal panics on a marked leaf, the fix unmarks before encoding, and projection.recordPayload has nowhere to store the sensitivity path - so the record carries the value but not the mark. projection.WriteBack shares this same hole, worse (no unmark of its own, would panic on the identical object after a real apply) - unfiled, not touched, worth a slot). test_apply/drift_reconverge remain not_run, blocked on this sixth wall now. Separately, #341 (found by the sibling corpus-mastino-dns crossing, same ratify.go gate but a different population - untaggable ordinary AWS types like aws_route53_record, architecturally guaranteed never to be RecordBacked per TestResolveNeverEmitsRecordBackedForAWSEstate) is CONFIRMED STILL OPEN after #340 - the two issues share a root-cause line but #340's fix is deliberately scoped to the disjoint RecordBacked population and does not reach it. #340's own recordable/Approve sibling-carrier pattern is flagged as a reasonable template for #341's eventual fix. Follow-up pass 2026-08-20 (worktree ../wt/lambda-simple-sixth-wall, branch live/lambda-simple-sixth-wall, off LOCAL main ea9fd62fc0): the sixth wall was REPRODUCED for real first, not read off this note - STAGE 1 PASS, STAGE 2 PASS, STAGE 3 with zero diagnostics and 'Plan: 0 to add, 2 to change, 0 to destroy', both changes verbatim as recorded above. Stages are UNCHANGED at 2 of 5, and the reason is now measured rather than argued: the two changes belong to two different projects. (a) local_file.archive_plan's sensitivity-only diff IS choudoufu's and is FIXED (f66bc9e043). projection.recordPayload gained SensitiveAttrs, encoded exactly the way a state file encodes sensitive_attributes: encodeRecordPayload splits the value from its marks itself so no caller unmarks, decodeRecordPayload puts them back the way ResourceInstanceObjectSrc.Decode does, and materializeRecord re-marks after the schema conversion so obj.Encode derives AttrSensitivePaths from them. The mechanism the wall turned on is that live-plan runs the plan graph with SkipRefresh (live_plan.go:499), so a projected object's AttrSensitivePaths is the ONLY marks the plan's 'before' side ever has - upstream re-marks a refreshed object at node_resource_abstract_instance.go:1106 and that line is never reached - while the 'after' side is re-marked from the config and the provider schema every run at :1383. Derived from the object's own marks and nothing else, so it reaches any record-backed type with any sensitive attribute at any path; a mark that is not marks.Sensitive is refused rather than dropped. Four mutation checks, each caught, and the shape test consults an EXTERNAL source (it writes a real state file through internal/states/statefile and requires the same JSON for the same paths) rather than round-tripping against itself. TestIdentityGolden 0 changed, 0 added, 0 removed; ./internal/live/..., ./tools/..., ./live/..., ./cmd/... and ./internal/command/ all green. (b) aws_lambda_function's '- environment {}' / '+ logging_config { log_format = "Text" }' pair is NOT choudoufu's, and this is now proven by a CONTROL rather than reasoned about: run.sh's new step 3b runs plain terraform, its own state file, its own refresh, 'terraform plan -detailed-exitcode' immediately after its own cold apply with no choudoufu anywhere in the run, and it replans NOT EMPTY on exactly module.lambda_function.aws_lambda_function.this[0] and nothing else. Asserted by value, so a new emulator gap breaks the script instead of hiding in a bucket labelled expected, and a fixed one breaks it too. Filed as lex00/floci#83 with both causes located in LambdaController.buildFunctionConfiguration: Environment is emitted unconditionally ('SDK expects it even when empty') where real AWS omits it for a function that never had one - which is why terraform-provider-aws reads it under 'if function.Environment != nil' - and LoggingConfig is neither stored nor emitted at all, where real AWS always returns one defaulting to Text. The module declares zero environment blocks (main.tf:90, a dynamic block over an empty map) and one logging_config unconditionally (main.tf:136), so both fire on the module's DEFAULTS. Stage 3 now splits its own plan against the step-3b control with comm and names only the remainder as choudoufu's. WHAT IS NOT VERIFIED, stated rather than implied: the post-fix crossing was started and its choudoufu init was killed by the harness before stage 2, so the sensitivity fix has NOT been observed clearing the diff in a real end-to-end run - it is verified by unit tests that drive the two real paths (WriteBack after an apply, then BuildWith/materializeRecord on the next plan) and assert inst.Current.AttrSensitivePaths by value. The next run of this script is what settles it, and it should still end at 'live-plan is not empty' with the Lambda alone until lex00/floci#83 lands. Two further issues filed from this pass, neither fixed: #343 (builder.materialize applies no schema.Block.ValueMarks to what a provider Read returned, so the identical perpetual diff exists for any CONCRETE cloud object with a Sensitive attribute - separate because it also changes what b.live means for identity composition, and unmeasured over the corpus) and #344 (a record written before SensitiveAttrs existed now conflicts with its own re-migration though the value is identical, because SeedRecordForInstance compares bytes; population is local_file plus any config-derived mark, and the format is one day old). Stages 4 and 5 remain unwritten: an empty stage-3 plan is their precondition and lex00/floci#83 is what stands between this estate and one. Follow-up pass 2026-08-20 (worktree ../wt/lambda-simple-floci83, branch live/lambda-simple-floci83): lex00/floci#83 is FIXED and CLOSED, settled against real AWS (one throwaway Lambda function + IAM role created and immediately deleted, disclosed to the user per standing permission) rather than assumed - GetFunction/GetFunctionConfiguration omits Environment entirely for a function that never had one (not present-but-empty) and always returns LoggingConfig, defaulting to Text. Fixed in LambdaController/LambdaService/LambdaFunction, verified three ways (local instance, direct probe, and the same probe against the published GHCR image). Re-crossing after the re-pin: STAGE 3 (test_plan) now raises zero diagnostics and proposes changing ZERO resources for the first time ever on this estate - both prior sixth-wall diffs (the sensitivity-only local_file.content diff and the aws_lambda_function nested-block mismatch) are gone, verified by reading the raw terraform plan output directly rather than trusting extracted variables. But the plan is still not empty: all 23 of the example module's root-level `output` blocks render as `+ new` on every run. Root cause: internal/live/projection.Manager.GetRootOutputValues always returns an empty map - nothing evaluates the config's own output blocks against the reconstructed prior state before the plan graph asks for it. A genuinely new, generalizable gap (neither corpus-mastino-dns nor corpus-evoteum-modules, the two other closest-to-5/5 crossings, declares any root-level output, so nothing had hit this before). Filed as INTENTIUS/choudoufu#348, not attempted - core projection-architecture work, not a quick derivation. Stages remain 2 of 5; the estate's real blocker moved from floci#83 to #348, which is now the sole thing standing between this estate and an empty stage-3 plan. Follow-up pass 2026-08-20 (primary checkout, local main db1f412cfd, real Docker/floci run - GitHub issue #340 verification): #340 was found ALREADY FIXED on main (d30daa156f/8d34e3ded9, confirmed by commit history and by internal/live/liveimport/record_test.go's TestApprove_SeedsTheRecordStoreForARecordBackedInstance/TestRecordBackedTypeReadsTheGeneratedTable, the latter covering random_pet/null_resource/terraform_data/local_file/time_sleep/random_id generically), so this pass re-verified rather than re-fixed. Re-crossing confirms #340's own fix by absence again: 'no record-backed resource is proposed for creation: the migrate seeded all four' and 'no sensitivity-only diff on local_file.archive_plan: the record carries its marks' both print in stage 3's own output. #349 ('see through provably-zero-instance blocks when evaluating root outputs', 88d7e3961e, landed after this note's #348 paragraph) cut the output-only diff from all 23 root outputs to exactly 2: 'lambda_function_arn_static' and 'local_filename', both still '+' on every run. Stage 3 (test_plan) is still BLOCKED - not yet empty - but the wall is now two output lines, not twenty-three, and #340 itself contributes zero diagnostics and zero sites to what remains. Stages unchanged at 2 of 5 pass; the residual 2-output gap is #348/#349's remaining scope, not #340's, and was not investigated further here (out of this issue's scope). Follow-up pass 2026-08-21 (isolated worktree off local main 860c29e129, real Docker/floci/AWS CLI run, identical harness run TWICE with only TOFU_BIN swapped - not read from any prior note): #349's sub-problem 2, the root-output data-source read, is now built, and this estate's stage-3 output diff went from 2 lines to 1. Measured: at 860c29e129 the plan's 'Changes to Outputs:' block carries 'lambda_function_arn_static' and 'local_filename'; with the fix it carries 'local_filename' alone. lambda_function_arn_static vanishes from the diff entirely rather than rendering as '~ old -> new', which is the stronger result: the plan graph independently computed the same value the pre-plan read computed for the prior side, so they cancel. The three data sources behind it (data.aws_partition.current, data.aws_region.current and data.aws_caller_identity.current, all in module.lambda_function) are read live before the plan through the same configured aws provider instance the projection already reads this estate through. local_filename is UNCHANGED and stays refused ON PURPOSE: it reaches data.external.archive_prepare, whose read runs package.py on the machine running the plan, and the new demand class is confined to providers this configuration manages live objects through (dataread.LiveProviders) - the external provider serves no managed resource type at all, in this or any configuration, so it is excluded structurally rather than by name. Stages are UNCHANGED at 2 of 5: cold_deploy pass, migrate pass, test_plan still FAIL (one output line is still one output line, so the plan is still not empty), test_apply and drift_reconverge still not reached. This narrowed the stage-3 diagnostic count; it did not clear the stage. Follow-up pass 2026-08-29 (isolated worktree off local main 499f9f5e80, real Docker/floci/AWS CLI run - GitHub issue #498, 'migrate pass and fail 23 minutes apart at the same emulator pin'): CONFIRMED as a real, deterministic defect, not a flake and not #497's runner-resource-pressure hypothesis. Reproduced 100% of the time under a controlled variable rather than by chance: migrate PASSED 7/7 consecutive local runs (this machine's stock terraform, v1.15.8) with byte-identical output every time, then FAILED 2/2 runs the instant HashiCorp Terraform v1.16.0 was forced first on PATH for stage 1's cold deploy, with the exact nightly failure text ('live-import -approve did not stamp 3 and record 4 of 8 resources cleanly'). Root cause, isolated to one byte: Terraform >=1.16.0's built-in terraform_data resource gained a new 'store' nested block (verified directly via `terraform providers schema -json` against terraform.io/builtin/terraform, no choudoufu involved) that choudoufu's own terraform_data schema (internal/builtin/providers/tf/resource_data.go, unchanged since the OpenTofu fork) does not declare; decoding a stock-terraform-1.16-produced cold.tfstate's terraform_data instance against that older schema failed with 'unsupported attribute "store"' (internal/live/liveimport/ratify.go's ratifyRecordBacked, its inst.Current.Decode(schema.Block.ImpliedType()) call), demoting terraform_data.package_filename_for_hash from RECORDED to SKIPPED and changing live-import -approve's summary line, which trips run.sh's exact-string assertion. The 'pass at 10:23:19Z, fail at 10:46:59Z, same commit' shape in #498 was never nondeterminism in the same environment: the passing row was a local worker's run against an older pinned-by-brew terraform (like this pass's own baseline), the failing row was CI's `hashicorp/setup-terraform@v3` with terraform_version: latest (confirmed 1.16.0 from the nightly's own log) - two different environments measuring the identical commit near-simultaneously, not one environment flip-flopping. Downloaded the eks-basic sibling failure from the same nightly run for comparison and confirmed it is a DIFFERENT failure (a runner tofu-on-PATH casualty per #497, mis-attributed to whatever CURRENT_STAGE was set to when it died) - #498's own caveat about #497 does not reach this estate's failure, which is fully explained by the terraform_data schema gap alone. Fixed generically for the one type it can reach (dataStoreResourceSchema() now declares 'store' as a NestedType Object attribute mirroring the real schema's own field shape and WriteOnly/Sensitive flags, verified by decoding the real terraform-1.16.0-produced state); this is the resource's own implementation file, not classification control flow, so it carries no live/derivation_guard_test.go entry. ctyjson.Unmarshal was independently verified (a standalone throwaway program, not assumed) to default a type attribute missing from raw JSON to null, so old choudoufu-written terraform_data state predating this field decodes unaffected - TestManagedDataUpgradeStateMissingStore pins that boundary directly. Re-crossing after the fix: migrate PASSES under terraform 1.16.0 too (2/2), with the correct '3 stamped, 4 recorded, 0 failed, 1 skipped' split restored, and the whole estate clears end to end (exit 0) under both terraform versions - no other stage regressed. Not verified: HashiCorp Terraform versions between 1.15.8 and 1.16.0 (whichever one first introduced 'store') and any version after 1.16.0 that might change the block's shape again; the fix targets the exact field shape read off 1.16.0 and would need re-verification against a materially different future schema. diff --git a/site/content/docs/progress/corpus-leynos-monitoring.md b/site/content/docs/progress/corpus-leynos-monitoring.md index 9b094a9356..c052fd9789 100644 --- a/site/content/docs/progress/corpus-leynos-monitoring.md +++ b/site/content/docs/progress/corpus-leynos-monitoring.md @@ -12,26 +12,26 @@ Set: core. Lane: opentofu-native. Why it is in the core set: a real project built for OpenTofu specifically, so OpenTofu-only surface is exercised -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 18s | 3 resources added (2 alarms + dashboard), 0 objects carry tofu-estate=leynos-monitoring-crossing before migration | +| Cold deploy | pass | 5s | 3 resources added (2 alarms + dashboard), 0 objects carry tofu-estate=leynos-monitoring-crossing before migration | | Migrate | pass | 25s | 2 of 3 stamped (1 skipped, untaggable dashboard), 0 failed; both alarm markers read back via the AWS CLI | | Replan from nothing | pass | 2s | no resource change proposed; both alarms' tofu-address unchanged, dashboard body re-derived and matches distribution_id | | No-op apply | pass | 2s | no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, no state file | | Drift and reconverge | pass | 4s | S3 alarm's alarm_description tampered, exactly 1 object proposed and applied, reconverged to its configured description | | Rename | pass | 6s | moved block: aws_cloudwatch_metric_alarm.s3_requests_spike renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_cloudwatch_metric_alarm.cf_requests_spike renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 4s | choudoufu: deleting the CloudFront alarm's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (describe-alarms on its name no longer returns it, read via the AWS CLI, not choudoufu's own report), and the next plan is empty; classifyOrphans did not withhold the destroy because the S3-requests alarm, the surviving aws_cloudwatch_metric_alarm instance, is bound, not unclaimed | +| Remove a block | pass | 5s | choudoufu: deleting the CloudFront alarm's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (describe-alarms on its name no longer returns it, read via the AWS CLI, not choudoufu's own report), and the next plan is empty; classifyOrphans did not withhold the destroy because the S3-requests alarm, the surviving aws_cloudwatch_metric_alarm instance, is bound, not unclaimed | | Change count | pass | 15s | choudoufu: scaling aws_cloudwatch_metric_alarm.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), confirmed genuinely gone via describe-alarms (no server-minted id on this type - see header), leaving count_test[0] and its tofu-address marker unchanged; the local record store's record for count_test[1] read tombstoned (has("tombstone"), no current "identity") at that same key, the #398-guard shape; scaling back from 1 to 2 created exactly count_test[1] under the SAME ARN (name/region/account-derived) with a fresh "identity" record entry (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied fresh in the idle greenfield-oracle account, shows the identical shape: destroy the higher index only (confirmed absent via describe-alarms), create it back (confirmed present again), the lower index untouched both times | | Replace with create_before_destroy | pass | 5s | choudoufu: changing s3_requests_spike_renamed's ForceNew alarm_name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (S3GetRequestsSpike) is confirmed gone and the new object (arn:aws:cloudwatch:us-west-2:000000000000:alarm:S3GetRequestsSpikeV2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (S3GetRequestsSpike -> S3GetRequestsSpikeV2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 13s | 3 resources from nothing (2 tagged alarms + the untaggable dashboard), both alarm markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on both alarms | +| Plan, review, apply | pass | 9s | one argument edited (cf_requests_spike's alarm_description, a config-owned non-ForceNew argument, gains a "(reviewed)" suffix), "plan -out=approved.tfplan" wrote a 7918-byte stock-format plan file whose whole change set is one update on module.monitoring.aws_cloudwatch_metric_alarm.cf_requests_spike; the world then moved out of band (S3GetRequestsSpike's alarm_description, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.monitoring.aws_cloudwatch_metric_alarm.s3_requests_spike and the live S3GetRequestsSpike it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - CFRequestsSpike's alarm_description still read as configured through the AWS CLI, rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the description put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and CFRequestsSpike read back with the reviewed description, so the refusal is earned by the drift and not handed out to every plan file. BOTH the plan -out and the apply carry this crossing's own -target set (aws_budgets_budget stays out of the graph - floci still answers UnknownOperationException for AWSBudgetServiceGateway on the pinned image), which is enough because a live-markers apply plans the live system from its OWN arguments rather than replaying the file: no exemption and no Go change was needed for issue #903's -target trap. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 11s | 3 resources from nothing (2 tagged alarms + the untaggable dashboard), both alarm markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on both alarms | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 1m33.9s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 1m30s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. FIVE OF FIVE, real, as of 2026-08-21 (crossing 77be0bc336 base). datapoints_to_alarm DELTA applied per lex00/floci#93's own suggested workaround (explicit value = evaluation_periods on both alarms, AWS's own create-time default, so nothing about what stage 1 creates changes) - verified for real, not assumed: the '1 -> null' diff is gone from both stock `tofu plan` and `choudoufu live-plan` against the same live alarms. Getting stage 3's automated assertion to actually SEE the empty plan uncovered a second, real, unrelated choudoufu bug: `choudoufu live-plan`'s delegation to plain `choudoufu plan` when a live block is present (`plan.Run(originalArgs)`) kept originalArgs as a second slice header over live-plan's own rawArgs backing array rather than an independent copy, and arguments.ParseView's in-place compaction of recognized flags (like -no-color) silently corrupted it whenever a flag followed -no-color - -target being the realistic case, since every -target/-exclude run hits this same alias. Effect: -no-color never reached the delegate, so live-plan's output for every -target run (this estate included) carried real ANSI escape codes despite the flag, invisibly breaking any exact-string assertion like stage 3's "No changes." check - not a semantics bug (a manual rerun confirmed the plan was already empty before the fix), but the automated proof of it was blind. Fixed in internal/command/live_plan.go (originalArgs := append([]string(nil), rawArgs...)), regression-tested in internal/command/live_mode_test.go (confirmed to fail without the fix, pass with it). Stage 5 then hit a third, distinct, real bug - floci's: CloudWatch PutMetricAlarm is documented (AWS CLI's own bundled help) as create-only for tags, so a real out-of-band update can never touch existing tags; floci's PutMetricAlarm wipes them instead, confirmed directly (list-tags-for-resource: two markers before, empty after an update with no --tags), destroying this crossing's ownership marker. Filed as https://github.com/lex00/floci/issues/95. Worked around in this harness only (not floci, not a change to what stage 5 tests) by re-applying the known tags via TagResource immediately after each drift call. Re-crossed for real with all three fixes/workarounds in place: cold_deploy 3 resources added; migrate 2 of 3 stamped (dashboard correctly UNTAGGABLE), both tofu-address markers verified via the AWS CLI; test_plan genuinely EMPTY with both alarms' markers and the dashboard's re-derived identity re-verified after the state file was deleted; test_apply a genuine no-op, 2 objects before and after, no state file either time; drift_reconverge proposes fixing exactly the one drifted alarm, applies it, and the live value reads back correct. `just ci` green. BREAK=1 still verified failing at stage 2's identity assertion (unchanged by this update - that check sits before stage 3 and exits the script before stage 5 is ever reached, a pre-existing property of the script's control flow, not something this update touched). diff --git a/site/content/docs/progress/corpus-mastino-dns.md b/site/content/docs/progress/corpus-mastino-dns.md index 6439f8aef4..392f466fb0 100644 --- a/site/content/docs/progress/corpus-mastino-dns.md +++ b/site/content/docs/progress/corpus-mastino-dns.md @@ -10,26 +10,26 @@ Source: at `4d8c1f1bebd91e73195017ce44 Set: growing. Lane: published-deployment. -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 2m30s | 63 resources from stock terraform; 4 live zones confirmed unmarked | -| Migrate | pass | 55s | 4 of 63 stamped, 59 skipped as untaggable, 0 failed; 59 identity records written (#364), 14 of them also carrying residue (#341), DataCite's own tags survived | -| Replan from nothing | pass | 7s | plan empty across 63 instances, no state file; 14 record sets and 4 zones filled residue from the store | -| No-op apply | pass | 10s | genuine no-op: 4 zones / 63 record sets unchanged, all 4 markers unmoved, all 59 identity records intact (14 residue-bearing) | -| Drift and reconverge | pass | 29s | one untaggable record drifted, exactly aws_route53_record.wp-prod-staging[0]/ttl proposed and applied, reconverged to 300, marker intact | -| Rename | pass | 26s | moved block: aws_route53_zone.production renamed with zero churn (0 add, 1 change, 0 destroy) - only the zone's own marker rewritten, none of its 45 record children moved; live-mv: aws_route53_zone.internal renamed with zero churn, marker rewritten in place; stock oracle over the same two-zone rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live zone ids unchanged, read via the AWS CLI | -| Remove a block | pass | 32s | choudoufu: deleting aws_route53_zone.eu and aws_route53_record.eu-ns's blocks - both destroys proposed (matching stock's own oracle exactly) and applied cleanly (Apply complete! Resources: 0 added, 0 changed, 2 destroyed.), the zone genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report); the next plan is empty. The parent-scoped removal sweep gap this estate named (gauntlet:parent-scoped-sweep) is closed: recordOrphanReadSweep composes aws_route53_record's identity from its migrate-seeded record correctly (composeImportIDFromComponents's OmitIfAbsent fix) and carries a destroy-before-parent ordering hint (identity.Resolution.DestroyDependsOn) so the record's own destroy is never raced against its zone's force_destroy cascade. | -| Change count | pass | 1m9s | choudoufu: scaling DataCite's own real, already-live aws_route53_record.wp-prod-staging count block (count.index + 3 in the name, header point 3) from 10 to 9 destroyed exactly wp-prod-staging[9] (staging12.datacite.org, 0 add, 0 change, 1 destroy; confirmed gone via the AWS CLI, and its local record file correctly tombstoned rather than left claiming a live identity - the #398-guard shape, confirmed by reading the file directly rather than assumed), leaving wp-prod-staging[0] (staging3.datacite.org)'s live TTL and record-store import_id unchanged; scaling back from 9 to 10 created exactly wp-prod-staging[9] again (0 add -> 1 add, 0 change, 0 destroy), TTL=300 and record import_id=ZFNTJ9UTHQDEAEU_staging12.datacite.org_A (identical string to before - Route 53 hands back no system id for a record set, so realness of the destroy was proved by the AWS CLI absence check and the tombstone, not a changed id), while wp-prod-staging[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical 10-instance count block, plan-only against cold_deploy's own state (applying would disturb the live objects migrate/stage 3-5 depend on), shows the identical shape: destroy the highest index only, create it back under the same name, every lower index untouched both times. aws_route53_record carries no tags at all (header point 2), so this type's own 'every surviving instance keeps its identity' is proved through the record store's ZONEID_NAME_TYPE identity string and a direct AWS CLI read, never a tofu-address tag value - the colon-vs-bracket tag-value escaping trap live/MARKERS.md documents does not apply to an untaggable type. BREAK_COUNT=1 confirms the wrong-instance assertion correctly fails to hold. | -| Replace with create_before_destroy | pass | 55s | choudoufu: changing aws_route53_record.status's ForceNew name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (status.datacite.org./CNAME) is confirmed gone and the new object (ZFNTJ9UTHQDEAEU_status2.datacite.org_CNAME) exists, both via the AWS CLI; the local record store's record at the same address now names the new object's identity, not the destroyed one (ZFNTJ9UTHQDEAEU_status.datacite.org_CNAME -> ZFNTJ9UTHQDEAEU_status2.datacite.org_CNAME); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); F-ORACLE also confirms the four apex NS records this estate's DELTA 5 manages can never take this same path (Route 53 refuses to delete the NS/SOA record at a zone's apex), which is why status was chosen instead; BREAK=replace confirms a manufactured identity collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | +| Cold deploy | pass | 1m55s | 63 resources from stock terraform; 4 live zones confirmed unmarked | +| Migrate | pass | 43s | 4 of 63 stamped, 59 skipped as untaggable, 0 failed; 59 identity records written (#364), 14 of them also carrying residue (#341), DataCite's own tags survived | +| Replan from nothing | pass | 5s | plan empty across 63 instances, no state file; 14 record sets and 4 zones filled residue from the store | +| No-op apply | pass | 8s | genuine no-op: 4 zones / 63 record sets unchanged, all 4 markers unmoved, all 59 identity records intact (14 residue-bearing) | +| Drift and reconverge | pass | 28s | one untaggable record drifted, exactly aws_route53_record.wp-prod-staging[0]/ttl proposed and applied, reconverged to 300, marker intact | +| Rename | pass | 20s | moved block: aws_route53_zone.production renamed with zero churn (0 add, 1 change, 0 destroy) - only the zone's own marker rewritten, none of its 45 record children moved; live-mv: aws_route53_zone.internal renamed with zero churn, marker rewritten in place; stock oracle over the same two-zone rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live zone ids unchanged, read via the AWS CLI | +| Remove a block | pass | 30s | choudoufu: deleting aws_route53_zone.eu and aws_route53_record.eu-ns's blocks - both destroys proposed (matching stock's own oracle exactly) and applied cleanly (Apply complete! Resources: 0 added, 0 changed, 2 destroyed.), the zone genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report); the next plan is empty. The parent-scoped removal sweep gap this estate named (gauntlet:parent-scoped-sweep) is closed: recordOrphanReadSweep composes aws_route53_record's identity from its migrate-seeded record correctly (composeImportIDFromComponents's OmitIfAbsent fix) and carries a destroy-before-parent ordering hint (identity.Resolution.DestroyDependsOn) so the record's own destroy is never raced against its zone's force_destroy cascade. | +| Change count | pass | 57s | choudoufu: scaling DataCite's own real, already-live aws_route53_record.wp-prod-staging count block (count.index + 3 in the name, header point 3) from 10 to 9 destroyed exactly wp-prod-staging[9] (staging12.datacite.org, 0 add, 0 change, 1 destroy; confirmed gone via the AWS CLI, and its local record file correctly tombstoned rather than left claiming a live identity - the #398-guard shape, confirmed by reading the file directly rather than assumed), leaving wp-prod-staging[0] (staging3.datacite.org)'s live TTL and record-store import_id unchanged; scaling back from 9 to 10 created exactly wp-prod-staging[9] again (0 add -> 1 add, 0 change, 0 destroy), TTL=300 and record import_id=Z7Z25KMX2IAY0SJ_staging12.datacite.org_A (identical string to before - Route 53 hands back no system id for a record set, so realness of the destroy was proved by the AWS CLI absence check and the tombstone, not a changed id), while wp-prod-staging[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical 10-instance count block, plan-only against cold_deploy's own state (applying would disturb the live objects migrate/stage 3-5 depend on), shows the identical shape: destroy the highest index only, create it back under the same name, every lower index untouched both times. aws_route53_record carries no tags at all (header point 2), so this type's own 'every surviving instance keeps its identity' is proved through the record store's ZONEID_NAME_TYPE identity string and a direct AWS CLI read, never a tofu-address tag value - the colon-vs-bracket tag-value escaping trap live/MARKERS.md documents does not apply to an untaggable type. BREAK_COUNT=1 confirms the wrong-instance assertion correctly fails to hold. | +| Replace with create_before_destroy | pass | 46s | choudoufu: changing aws_route53_record.status's ForceNew name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (status.datacite.org./CNAME) is confirmed gone and the new object (Z7Z25KMX2IAY0SJ_status2.datacite.org_CNAME) exists, both via the AWS CLI; the local record store's record at the same address now names the new object's identity, not the destroyed one (Z7Z25KMX2IAY0SJ_status.datacite.org_CNAME -> Z7Z25KMX2IAY0SJ_status2.datacite.org_CNAME); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); F-ORACLE also confirms the four apex NS records this estate's DELTA 5 manages can never take this same path (Route 53 refuses to delete the NS/SOA record at a zone's apex), which is why status was chosen instead; BREAK=replace confirms a manufactured identity collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 4m8s | 63 resources from nothing (4 tagged zones + 59 untaggable records), the production zone's marker verified via the AWS CLI, 63 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches on zone count (4) and total record-set count (63) | +| Plan, review, apply | pass | 28s | one argument edited (aws_route53_zone.production's tags gain Reviewed=yes - a tag, so in-place, moving no live id), "plan -out=approved.tfplan" wrote a 19819-byte stock-format plan file whose own totals are "Plan: 0 to add, 1 to change, 0 to destroy" and whose whole change set is that one update; the world then moved out of band (staging3.datacite.org's TTL 300->77 in zone Z7Z25KMX2IAY0SJ, this estate's own STAGE 5 upsert lifted, through the AWS CLI and never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming the extra row as " aws_route53_record.wp-prod-staging[0] Update Z7Z25KMX2IAY0SJ_staging3.datacite.org_A" - both aws_route53_record.wp-prod-staging[0] and the live composed identity Z7Z25KMX2IAY0SJ_staging3.datacite.org_A it was computed against, an UNTAGGABLE record whose identity comes from the local record store rather than a marker tag - with "Exit status 3" spelled out for a pipeline; nothing was applied - zone Z7Z25KMX2IAY0SJ still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the TTL put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the zone read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (tag gone, tofu-address marker intact, 63 record sets still there, next plan proposes no resource action) so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 3m54s | 63 resources from nothing (4 tagged zones + 59 untaggable records), the production zone's marker verified via the AWS CLI, 63 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches on zone count (4) and total record-set count (63) | | Strict profile (not a headline stage) | not run | | | -Last run at commit `d72960cdc3` on 2026-09-06T05:35:09Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 11m21.4s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 10m14.6s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. DataCite's own global DNS root module, the largest offline-clean estate (63 instances) that had never touched a cloud. 2 of 5 - and the two stages it does not reach are blocked on one real, general, previously-unrecorded choudoufu defect, not on anything specific to this estate. Still offline-clean when crossing started (refusal-probe -schemas: blocked 0 sites 0 instances 63 at c41279989a); the schema-less mode disagrees (blocked 1 sites 2, both render correctly in the real run - HANDOFF's asymmetry caveat firing on a live estate). Two of team-members-access's four deltas recur (#268 mandatory backend edit, in cloud{} form; #269 provider version skew, ~> 5 -> 5.100.0 with no list resources, all four zones ServerAssigned); the other two do not (one data source answered by an out-of-band VPC; an ordinary emulator override). Two NEW walls: (1) estate-owned, not choudoufu's/floci's - the four *-ns blocks manage each zone's own apex NS set, which Route 53 creates itself, so a from-scratch apply dies with InvalidChangeBatch; fixed with allow_overwrite=true, the same argument the estate's own author already writes on wp-prod-staging. (2) Filed as #341: stage 3's entire plan is 0/14/0, every diff line +allow_overwrite=true (10 of 14 on wp-prod-staging[0..9], carried in DataCite's own text with no deltas needed) - #275's residue mechanism populates and reads back the record store for TAGGABLE resources only (4 zones get 'filled 1 residue attribute(s)'), but none of the 59 untaggable record sets do, because internal/live/liveimport/ratify.go's !taggable() branch returns before the ReadResource that builds the *eligible object residue needs, and Approve's recordResidueFor sits past the continue that skips a resource with no *eligible - one carrier serving two unrelated jobs, and untaggability should only disqualify one of them (the tag write, not the residue read). 342 of 1025 admitted types are untaggable and share this exclusion. Not fixed here - out of scope for a crossing pass. What the run DOES prove: all 63 rendered identities correct and distinct by value against the AWS CLI's own answer, including two same-named datacite.org zones (public/private) that did not swap and ten wp-prod-staging[0..9] instances rendering staging3..staging12 individually. The 59 untaggable record sets are 94% of the estate - the widest derived-from-tagged fan-out in either lane. Script exits 0 only on reaching exactly this blocker (asserting the changed-address set, that allow_overwrite is the only attribute in the whole diff, and the 0/14/0 totals line) and non-zero on anything else including an empty plan, which is the signal to promote this entry to five stages once #341 lands. BREAK=1 verified red at exactly the identity assertion (a swapped zone id). Also filed lex00/floci#81 (floci accepts a record set whose name is outside its hosted zone, a real bug in the estate's own text; blocks nothing here but is the shape where a crossing passes on the emulator and fails on real AWS). Suggested a new 'published-deployment' lane, distinct from terraform-popular/opentofu-native/reference, since this is neither a module example nor an OpenTofu-native project but a company's own live TFC-connected infrastructure. Merged d5e592d67c (crossing e74b6e5c01); just demo-corpus-mastino-dns, port 4731. just ci green (exit 0, read from a file). Follow-up pass 2026-08-20 (#341 fixed and merged, c73a6e4617/78c92ad64a): FIVE OF FIVE, real. Fix is a third carrier (residuable, which eligible now embeds) built for any admitted-untaggable instance with a record_store declared, deliberately with NO ReadResource at ratify time - a residue attribute is by definition one no read returns, so state already has everything the fix needs, and skipping the read means an untaggable instance can never come back MISSING/DRIFTED (a concern the issue itself raised). Verdict stays StatusUntaggable/OutcomeSkipped; only the marker write was ever skippable, not the residue write. Corrected denominator along the way: DefaultTable holds 1040 rows, not 1025 - survey-full.json calls 683 taggable/342 untaggable and doesn't cover 15 at all. Real re-crossing: migrate reports 4 stamped/59 UNTAGGABLE/14 residue records (all 4 zones plus 10 of the 59 record sets, matching the bug's own signature exactly); test_plan EMPTY, all 63 identities asserted by value; test_apply a genuine no-op with all 14 residue records and 4 markers unchanged; drift_reconverge drifts wp-prod-staging[0]'s TTL (untaggable AND residue-carrying) and reconciles exactly that instance, BREAK_STAGE5=1 verified failing. One honest caveat: stages 4/5 had to be WRITTEN (the prior entry's header claimed they existed in git history but e74b6e5c01 is the file's only commit), and the verifying run itself executed in two calls after a SIGTERM mid-init, not one continuous process - same container/workdir throughout, but the committed script has not been run start-to-finish in a single invocation. A real, separate, more urgent finding surfaced while verifying: #340's own change to live-import's summary line ("%d newly recorded, %d already recorded" inserted mid-line) broke the exact-string assertion in 19 OTHER crossing scripts on main - invisible to just ci since e2e scripts aren't in that tier. Filed as #342 with the full list and the one-line fix each needs. diff --git a/site/content/docs/progress/corpus-overture-tiles.md b/site/content/docs/progress/corpus-overture-tiles.md index c19c94eaf1..368dde7be1 100644 --- a/site/content/docs/progress/corpus-overture-tiles.md +++ b/site/content/docs/progress/corpus-overture-tiles.md @@ -12,26 +12,26 @@ Set: core. Lane: opentofu-native. Why it is in the core set: a real project built for OpenTofu specifically, so OpenTofu-only surface is exercised -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| | Cold deploy | pass | 56s | 26 resources, genuinely cold, genuinely unmarked | -| Migrate | pass | 1m10s | 16 of 26 stamped, 0 failed; the other 10 correctly UNTAGGABLE or already-ruled UNADMITTED_TYPE | -| Replan from nothing | pass | 9s | live-plan empty after the STAGE 2d convergence apply; S3 bucket and OAC identities re-checked by value against the AWS CLI | +| Migrate | pass | 1m15s | 16 of 26 stamped, 0 failed; the other 10 correctly UNTAGGABLE or already-ruled UNADMITTED_TYPE | +| Replan from nothing | pass | 10s | live-plan empty after the STAGE 2d convergence apply; S3 bucket and OAC identities re-checked by value against the AWS CLI | | No-op apply | pass | 3s | no-op apply (0 added, 0 changed, 0 destroyed); 16 tagged objects before and after (resourcegroupstaggingapi's cross-service search alone, floci#98 fixed); S3 bucket and OAC identities unchanged; record store intact | -| Drift and reconverge | pass | 7s | one object tampered (VPC Name tag), exactly module.overture_tiles.aws_vpc.batch[0] proposed by both choudoufu and stock with the identical change, apply changed 1 and the Name tag reads back as configured | +| Drift and reconverge | pass | 8s | one object tampered (VPC Name tag), exactly module.overture_tiles.aws_vpc.batch[0] proposed by both choudoufu and stock with the identical change, apply changed 1 and the Name tag reads back as configured | | Rename | pass | 47s | moved block: module.overture_tiles renamed to module.overture_tiles_moved via ONE module-level moved block, 0 add/0 destroy, 16 real tag-marker rewrites (plan showed 18 - two untaggable siblings' policy JSON transiently 'known after apply', resolving to no real change at apply time, confirmed via the stage-5 marker/propagation filter and by value); live-mv: module.overture_tiles_moved renamed to module.overture_tiles_final across 14 of 16 taggable children, one call each, zero churn - the other 2 (aws_batch_compute_environment.tiles and aws_iam_instance_profile.ecs, both server-/provider-assigned identities with no List support in the provider) correctly refused by live-mv and renamed via their own moved blocks instead, applied cleanly; the nine untaggable/config-derived children and the UNTAGGABLE OAC (no longer UNADMITTED_TYPE - #249 narrowed) did not move at all; stock oracle over the identical module rename on cold_deploy's own state also shows zero churn via its own single module-level moved block, covering every one of the 26 children including the two live-mv cannot | -| Remove a block | pass | 53s | choudoufu: create_cloudfront_distribution=false proposed exactly two destroys plus one in-place update (0 add, 1 change, 2 destroy: the distribution, its untaggable OAC, and the bucket policy's own CloudFrontOAC statement dropping), applied cleanly (0 added, 1 changed, 2 destroyed), the distribution is genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state also proposes exactly the same two destroys plus the same bucket-policy update | -| Change count | pass | 48s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-0f150fc86696abb33) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) unchanged, both read back through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW GroupId (sg-b5900fe02eec8c173 -> sg-16652b643494fb26b) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; every absence check reads length(SecurityGroups) through a group-id FILTER, because describe-security-groups --group-ids on a deleted id returns an empty list with exit 0 on this emulator pin; the next stateless live-plan is empty. Stock oracle (G0): the identical 2-instance block stood up with plain tofu in its own working directory against the same idle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new GroupId (1 add, 0 change, 0 destroy), the lower index's GroupId unchanged both times - then torn down (3 destroyed) before the choudoufu side ran. SYNTHETIC BLOCK, and why: every count this module declares is a boolean create toggle (create_vpc x7, create_s3_bucket x4, create_cloudfront_distribution x2, three launch_template variants gated on existing_id == null), never a scalable set, so scaling one is the day2_remove shape this script already runs, not a shape with a survivor; and its one real for_each (aws_batch_job_definition.tiles over toset(var.themes)) is scoped by this crossing's own root config to a single theme from stage 1 onward, so it has nothing to scale down to and widening it would move every earlier stage's counted assertions (26 resources, 16 stamped, 16 tagged objects, day2_rename's own 16-address list). What shrinking that set would plan is not claimed: it was never measured, because it was never a usable option. Sanctioned fallback per live/GAUNTLET.md #8, precedent reference-ec2-vpc Part F and corpus-iam-policy Part G. It reuses a type this estate already exercises (aws_security_group.batch), sits at a root address nothing else names, and runs entirely after day2_remove, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly reports fail. | -| Replace with create_before_destroy | pass | 6s | choudoufu: supplying module.overture_tiles_final's name_overrides.cloudwatch_log_group proposed exactly one replace at the same declared address (the log group; -/+ destroy and then create) cascading into two expected in-place updates (the execution role's inline log policy, the job definition) and nothing else; applied cleanly; the old object (/aws/batch/overture-tiles-crossing) is confirmed gone and the new object (/aws/batch/overture-tiles-crossing-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (/aws/batch/overture-tiles-crossing -> /aws/batch/overture-tiles-crossing-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address plus the same cascade (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | +| Remove a block | pass | 52s | choudoufu: create_cloudfront_distribution=false proposed exactly two destroys plus one in-place update (0 add, 1 change, 2 destroy: the distribution, its untaggable OAC, and the bucket policy's own CloudFrontOAC statement dropping), applied cleanly (0 added, 1 changed, 2 destroyed), the distribution is genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state also proposes exactly the same two destroys plus the same bucket-policy update | +| Change count | pass | 51s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-f37ccd0807772ad2c) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) unchanged, both read back through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW GroupId (sg-a444bab32fd697268 -> sg-46bcdd1e7ba3d595c) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; every absence check reads length(SecurityGroups) through a group-id FILTER, because describe-security-groups --group-ids on a deleted id returns an empty list with exit 0 on this emulator pin; the next stateless live-plan is empty. Stock oracle (G0): the identical 2-instance block stood up with plain tofu in its own working directory against the same idle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new GroupId (1 add, 0 change, 0 destroy), the lower index's GroupId unchanged both times - then torn down (3 destroyed) before the choudoufu side ran. SYNTHETIC BLOCK, and why: every count this module declares is a boolean create toggle (create_vpc x7, create_s3_bucket x4, create_cloudfront_distribution x2, three launch_template variants gated on existing_id == null), never a scalable set, so scaling one is the day2_remove shape this script already runs, not a shape with a survivor; and its one real for_each (aws_batch_job_definition.tiles over toset(var.themes)) is scoped by this crossing's own root config to a single theme from stage 1 onward, so it has nothing to scale down to and widening it would move every earlier stage's counted assertions (26 resources, 16 stamped, 16 tagged objects, day2_rename's own 16-address list). What shrinking that set would plan is not claimed: it was never measured, because it was never a usable option. Sanctioned fallback per live/GAUNTLET.md #8, precedent reference-ec2-vpc Part F and corpus-iam-policy Part G. It reuses a type this estate already exercises (aws_security_group.batch), sits at a root address nothing else names, and runs entirely after day2_remove, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly reports fail. | +| Replace with create_before_destroy | pass | 7s | choudoufu: supplying module.overture_tiles_final's name_overrides.cloudwatch_log_group proposed exactly one replace at the same declared address (the log group; -/+ destroy and then create) cascading into two expected in-place updates (the execution role's inline log policy, the job definition) and nothing else; applied cleanly; the old object (/aws/batch/overture-tiles-crossing) is confirmed gone and the new object (/aws/batch/overture-tiles-crossing-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (/aws/batch/overture-tiles-crossing -> /aws/batch/overture-tiles-crossing-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address plus the same cascade (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | +| Plan, review, apply | pass | 13s | one argument edited (module.overture_tiles's cors_allowed_origins, ["*"] -> ["https://tiles.example.invalid"], which reaches module.overture_tiles.aws_s3_bucket_cors_configuration.tiles[0] and nothing else - it is the only root knob of this 26-instance estate that lands on exactly one instance IN PLACE, since `tags` reaches all 16 taggable children at once, name_overrides and the launch template are ForceNew, and the create_* toggles are creates and destroys), "plan -out=approved.tfplan" wrote a 34023-byte stock-format plan file whose whole change set is that one in-place update; the world then moved out of band (vpc-9038e1c6's Name tag -> moved-after-approval, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming the extra row as " module.overture_tiles.aws_vpc.batch[0] Update vpc-9038e1c6" - both module.overture_tiles.aws_vpc.batch[0] and the live vpc-9038e1c6 it was computed against - with "Exit status 3" spelled out for a pipeline; nothing was applied - overture-tiles-crossing-tiles's CORS rule still read "*" through s3api get-bucket-cors rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Name tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the CORS rule read back as https://tiles.example.invalid, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (CORS back to "*", VPC Name tag still overture-tiles-crossing-vpc, next plan proposes no resource action, no state file left behind) so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | | Greenfield apply | pass | 2m0s | 26 resources from nothing, bucket and batch job queue markers verified via the AWS CLI, 26 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (batch job queue state, CloudFront distribution comment, bucket count) | | Strict profile (not a headline stage) | not run | | | -Last run at commit `a7ca11f935` on 2026-09-06T23:49:04Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 6m59.4s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 7m23s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed 2026-08-19. Sourced via GitHub code search rather than the awesome-opentofu/Powered-by-OpenTofu lists, which turned out to be pure tooling/adopter lists with no deployable estates. OpenTofu-native evidence is in its CI rather than a genuine .tofu extension (weaker self-description than corpus-hongbomiao, which ships real .tofu files): .github/workflows/ci.yml runs tofu fmt/validate/test/tflint exclusively through opentofu/setup-opentofu - terraform never appears - and its tests use OpenTofu's own mock_provider framework. A real, tagged-release module (v1.0.0->v1.2.0) from a Linux-Foundation-adjacent geospatial project backed by AWS/Meta/Microsoft/TomTom, contributor fixes as recent as 2026-05-21. cold_deploy genuinely passes (26 resources, plain tofu apply, unmodified module). migrate is BLOCKED, not clean: live-import stamps 13 of 26 cleanly, 10 correctly UNTAGGABLE or already-ruled UNADMITTED_TYPE (#249), and 3 AWS Batch resources fail to stamp on a real floci bug - TagResource/UntagResource/ListTagsForResource (POST /v1/tags/{resourceArn}) misroutes to AppSyncController's greedy catch-all since BatchController never registers that path. Filed as lex00/floci#72 with full evidence (checked ~/checkouts/floci first, confirmed same bug on current main, no in-progress fix) - not fixed, per this session's standing instruction. test_plan is BLOCKED, deterministically asserted (Plan: 4 to add, 7 to change, 0 to destroy, every line traced) rather than reached cleanly. Also filed INTENTIUS/choudoufu#322: aws_iam_role_policy (untaggable, ServerAssignedIfAbsent name via name_prefix) escalates a single-address unbound warning into a hard Error: Listed resource with no tags that aborts the ENTIRE live-plan, not just its own address - a real blast-radius concern (one bad site takes down the whole plan) worth prioritizing. Not fixed; worked around in this crossing via the module's own name_overrides input so the script could still assert what it could reach. Stages 4-5 not attempted - both need a genuinely empty first plan, which this estate doesn't reach yet. Confirmed informationally that applying the current non-empty plan fails safely (AWS Batch's own name-uniqueness check refuses the duplicate) rather than silently corrupting anything. Merged to local main as a233497312 (fix itself: 3c183a305a); justfile gained recipe demo-corpus-overture-tiles; live/corpus-manifest.json gained the pin. UPDATE 2026-08-20: lex00/floci#72 FIXED (floci 1d469fff, published; choudoufu re-pinned to sha256:dc246b1e) and migrate now genuinely PASSES - re-run for real against that image, not inferred: 16 of 26 newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 10 skipped, where it was 13 stamped / 3 failed before. The dry run's own counts are unmoved (16 eligible, 11 VERIFIED, 5 DRIFTED, 9 UNTAGGABLE, 1 UNADMITTED_TYPE) - the bug was in the tag WRITE, never in verification. The fix is generic rather than Batch-shaped: floci already had a SharedTagsController dispatching /tags/{arn} to the TagHandler whose serviceKey matches the ARN's own service segment, and AppSync had simply claimed /v1/tags/{arn: .+} for itself; the fix lifts that dispatch into a SharedTagsDispatcher keyed on path prefix AS WELL AS service, adds a SharedTagsV1Controller, and converts AppSync into a TagHandler beside a new BatchTagHandler - so any further service on that path needs only a handler. Stage 2c now asserts three markers through the AWS CLI instead of one, including the Batch job queue's own tofu-address/tofu-estate via batch list-tags-for-resource (the very call that used to be answered by AppSync) and that the module's own create-time Project tag SURVIVED the stamp - a TagResource that replaces instead of merging is how a live object silently loses its markers. INTENTIUS/choudoufu#322 item 1 is also fixed (576990a599/b75e46c24e) and is no longer the test_plan wall. test_plan stays 'fail' but the wall MOVED and is now harder: it is no longer a non-empty plan, it is live-plan refusing to plan at all (exit 1, no plan produced), at exactly two diagnostics - 'Invalid Identity Attribute Value: Identity attribute "arn" contains an Account ID "000000000000" which does not match the provider's ""' followed by its consequence 'Cannot import for projection'. Filed as INTENTIUS/choudoufu#345 with full evidence. Reachable only BECAUSE the Batch resources are now stamped: projection imports one by its ARN identity and hashicorp/aws validates an identity ARN's account segment against the account the provider knows about itself, which skip_requesting_account_id = true (what every crossing script sets to reach a local emulator) leaves empty. The marker is not wrong - stage 3 re-reads the job queue's real ARN from floci through the AWS CLI and asserts the refusal names that exact string. MEASURED, NOT ASSUMED, and recorded in the script's header so it is not re-tried: setting skip_requesting_account_id = false on the estate copy alone clears this error and breaks stage 2 instead, because the provider then routes S3 bucket tag reads through S3 Control's account-prefixed virtual host (dial tcp: lookup 000000000000.127.0.0.1: no such host), taking aws_s3_bucket.tiles[0] from VERIFIED to MISSING and the estate to 15 of 26 eligible. Stages 4-5 still not attempted: they need a plan and stage 3 produces none. The script's stage 3 is rewritten to assert the new wall deterministically (nonzero exit, exactly 2 errors and no more, both texts, the ARN in them) - and a ZERO exit now fails it, so the day this is fixed the script says so instead of quietly passing. BREAK=1 (stage 2's bucket-marker control) was NOT re-run this pass; its mechanism is unchanged. UPDATE 2026-08-20 (INTENTIUS/choudoufu#345 FIXED, no floci change): the identity-ARN crash is gone. The obvious fix, skip_requesting_account_id = false on the estate copy, was re-measured for real and its earlier 'breaks stage 2 via S3 Control's account-prefixed virtual host, dial tcp: lookup 000000000000.127.0.0.1: no such host' failure is a DNS failure, not an HTTP one - confirmed by curling the same account-prefixed host directly, which fails identically before any TCP connection, so no floci server-side Host-header routing could ever have fixed it (the request never arrives). The real fix is ENDPOINT: floci already publishes localhost.floci.io as a real, public wildcard DNS domain (EmbeddedDnsServer.DEFAULT_SUFFIX, same mechanism as LocalStack's localhost.localstack.cloud) that resolves an account-ID-prefixed label to 127.0.0.1 with no floci container running at all - confirmed via dig and via curl reaching floci's S3ControlController correctly (path-based dispatch, unaffected by the account-prefixed Host). Verified against the CURRENT, unmodified floci image (be3f7ffd, sha256:8a882bcc - no re-pin, no floci commit, no floci PR). Real re-run: stage 1 PASS unchanged (26 resources). Stage 2 PASS, counts moved by exactly one resource as a direct, expected consequence (11 VERIFIED/5 DRIFTED -> 10 VERIFIED/6 DRIFTED: aws_launch_template.batch[0]'s arn now differs between the PLAIN state, written under skip_requesting_account_id = true and so account-less, and the ESTATE copy's live re-read, which now knows its account - a real difference between two provider configurations, not a wrong marker). Stage 3 (test_plan) still recorded fail by this repo's own convention (a first plan must be empty to pass) but the #345 wall itself - live-plan exiting 1 with two diagnostics and no plan at all - is gone: live-plan now exits 0 with 'Plan: 1 to add, 7 to change, 0 to destroy.', asserted deterministically address-by-address. Every line traces to an already-tracked or by-design cause, none of them new: the 1 add is the already-ruled #249 aws_cloudfront_origin_access_control UNADMITTED_TYPE gap; 6 of the 7 changes are internal/live/discovery/count.go's own documented one-time tofu-slot migration-visibility tag (bindCountByAddress's doc comment: 'visible in the plan as a tofu-slot tag being added to each member' - by design, cements on first apply), on every count-toggled ([0]) resource this module declares; the 7th, aws_s3_bucket_policy.tiles[0], is a content diff cascading from the new OAC's arn being 'known after apply' in the same plan. Verified informationally (not scored, per this repo's own convention that test_apply is scored only once test_plan is itself empty): applying the stage 3 plan succeeds (Apply complete! Resources: 1 added, 6 changed, 0 destroyed - matching the plan), and a second live-plan afterward is genuinely empty ('No changes. Your infrastructure matches the configuration.') - the estate converges in exactly one apply, confirming #345's own header claim. No Go code touched; the fix is entirely live/e2e/corpus-overture-tiles/run.sh (ENDPOINT changed from a bare IP to localhost.floci.io, skip_requesting_account_id parameterized so only the estate copy sets it false, cold deploy left untouched). diff --git a/site/content/docs/progress/corpus-rds-complete-postgres.md b/site/content/docs/progress/corpus-rds-complete-postgres.md index ba1ea71892..b239b82f86 100644 --- a/site/content/docs/progress/corpus-rds-complete-postgres.md +++ b/site/content/docs/progress/corpus-rds-complete-postgres.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m49s | 39 resources, once for real | -| Migrate | pass | 50s | 26 of 39 stamped | -| Replan from nothing | pass | 8s | genuinely empty replan (No changes. Your infrastructure matches the configuration.) with no local state file. lex00/floci#120's round-trip gap, this estate's last recorded wall, is CONFIRMED FIXED: round 8 (PR #128/ff815779, ghcr.io/lex00/floci:main-20260824d sha256:25fc9687, #124's RDS colliding-port isolation) closed the last of its eight fields for this estate - module.db_default's own port (module.db and module.db_default both declare port=5432, a genuine collision; module.db_default is the second-created instance and gets its own distinct loopback bind address with the declared port honored). The other seven fields (backup_window, monitoring_interval, monitoring_role_arn, performance_insights_retention_period, engine_lifecycle_support, enabled_cloudwatch_logs_exports, max_allocated_storage) and the parameter block's apply_method were already fixed by earlier rounds (round 5 and round 6's own #120 passes) that this estate had not been re-crossed since - the artifact's recorded '3 in-place updates' detail was stale before this round's own fix even landed. Confirmed three independent ways, not merely inferred from the empty plan: a direct describe-db-parameters --source user probe of the live parameter group (autovacuum=1, client_encoding=utf8, matching config exactly, no tofu in the loop), a direct describe-db-instances probe of the second instance's own Endpoint.Port (5432, the declared port), and all eight attribute names individually confirmed absent from choudoufu's plan. INTENTIUS/choudoufu#393 (skip_final_snapshot's phantom true->false update) remains fixed, confirmed absent. Stock's own replan against its own never-deleted state file still shows tag noise plus the two parameter blocks; ruled out as a live discrepancy by the same direct API probe (informational only, not this stage's oracle - HANDOFF row 3, a property of that one state file's own apply-time fidelity). | -| No-op apply | pass | 4s | genuine no-op: 26 objects before, 26 after, no state file, primary DB instance marker unmoved | +| Cold deploy | pass | 1m44s | 39 resources, once for real | +| Migrate | pass | 52s | 26 of 39 stamped | +| Replan from nothing | pass | 9s | genuinely empty replan (No changes. Your infrastructure matches the configuration.) with no local state file. lex00/floci#120's round-trip gap, this estate's last recorded wall, is CONFIRMED FIXED: round 8 (PR #128/ff815779, ghcr.io/lex00/floci:main-20260824d sha256:25fc9687, #124's RDS colliding-port isolation) closed the last of its eight fields for this estate - module.db_default's own port (module.db and module.db_default both declare port=5432, a genuine collision; module.db_default is the second-created instance and gets its own distinct loopback bind address with the declared port honored). The other seven fields (backup_window, monitoring_interval, monitoring_role_arn, performance_insights_retention_period, engine_lifecycle_support, enabled_cloudwatch_logs_exports, max_allocated_storage) and the parameter block's apply_method were already fixed by earlier rounds (round 5 and round 6's own #120 passes) that this estate had not been re-crossed since - the artifact's recorded '3 in-place updates' detail was stale before this round's own fix even landed. Confirmed three independent ways, not merely inferred from the empty plan: a direct describe-db-parameters --source user probe of the live parameter group (autovacuum=1, client_encoding=utf8, matching config exactly, no tofu in the loop), a direct describe-db-instances probe of the second instance's own Endpoint.Port (5432, the declared port), and all eight attribute names individually confirmed absent from choudoufu's plan. INTENTIUS/choudoufu#393 (skip_final_snapshot's phantom true->false update) remains fixed, confirmed absent. Stock's own replan against its own never-deleted state file still shows tag noise plus the two parameter blocks; ruled out as a live discrepancy by the same direct API probe (informational only, not this stage's oracle - HANDOFF row 3, a property of that one state file's own apply-time fidelity). | +| No-op apply | pass | 5s | genuine no-op: 26 objects before, 26 after, no state file, primary DB instance marker unmoved | | Drift and reconverge | pass | 8s | one object tampered (primary DB instance's Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag | -| Rename | pass | 16s | moved block: module.security_group renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.db_default's db instance renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Rename | pass | 17s | moved block: module.security_group renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.db_default's db instance renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | | Remove a block | pass | 1m31s | choudoufu: deleting module.db_default_renamed's block proposed exactly two destroys (the db instance and its own local random_id.snapshot_identifier, no cloud representation - issue #340), applied cleanly, the db instance is genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on the same renamed oracle tree also proposes exactly the same two destroys; the target was chosen (see header) because its own nested module.db_instance call has no untaggable AWS-side sibling under this estate's create_db_option_group=false/create_db_parameter_group=false, unlike the shapes that surfaced issue #410 for corpus-s3-bucket-complete and corpus-overture-tiles | -| Change count | pass | 50s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-8a54c04755df9d5f0) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read back through the AWS CLI; the destroyed instance is confirmed gone two independent ways (0 by group-id - length(SecurityGroups), never an exit code, since this image answers a deleted id with an EMPTY LIST and exit 0 - and 0 by tofu-address tag filter). Scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-0f8f6860554cccc91 -> sg-36ce4ada55a333b89), carrying tofu-address=aws_security_group.count_test:1 with exactly one claimant, while count_test[0] kept the same id and the same marker throughout; the next plan is empty. The G0 stock oracle stood the identical 2-instance block up with plain terraform in its own working directory and its own VPC against the same idle endpoint and showed the identical shape: destroy the higher index only (Plan: 0 to add, 0 to change, 1 to destroy), create the higher index back under a new GroupId (sg-d7b20850fbae6692e -> sg-c4094d1adb1e37439), the lower index's GroupId (sg-75e308f768ff7da71) unchanged both times; torn down completely before the choudoufu side ran. SYNTHETIC BLOCK, and why: terraform-aws-rds declares no scalable knob anywhere - the root example has no count and no for_each at all, and every count in the vendored module is a boolean create toggle (var.create ? 1 : 0 and friends, grepped not guessed) - while the only genuinely scalable knobs belong to the supporting module.vpc, whose subnet families each destroy an untaggable aws_route_table_association sibling alongside the taggable aws_subnet, which is issue #410's family (corpus-overture-tiles's count-shrink extension) and would make this stage a re-measurement of that gap rather than of count scaling. So the sanctioned fallback per live/GAUNTLET.md #8, reference-ec2-vpc's Part F and corpus-iam-policy's Part G: a self-contained aws_security_group.count_test at an address nothing else names, appended after day2_remove's completed removal, of a taggable type this estate already exercises, introducing nothing that generates a secret. BREAK_COUNT=1 is the negative control and was run for real: it asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, proving the index assertion is load-bearing. | -| Replace with create_before_destroy | pass | 2m51s | choudoufu: changing module.db's ForceNew db_name argument (plus identifier, for an observable identity change) proposed exactly one instance replace at the same declared address, cascading into its 2 cloudwatch log groups and db parameter group (all replaced, all named from identifier) - 4 to add, 4 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql) is confirmed gone and the new instance (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql-replaced) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new identifier, not the destroyed one (complete-postgresql -> complete-postgresql-replaced); the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here). | +| Change count | pass | 50s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-f03da6a2c670f916f) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read back through the AWS CLI; the destroyed instance is confirmed gone two independent ways (0 by group-id - length(SecurityGroups), never an exit code, since this image answers a deleted id with an EMPTY LIST and exit 0 - and 0 by tofu-address tag filter). Scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-1247895febf8a3b8b -> sg-2e4b68f2a30525035), carrying tofu-address=aws_security_group.count_test:1 with exactly one claimant, while count_test[0] kept the same id and the same marker throughout; the next plan is empty. The G0 stock oracle stood the identical 2-instance block up with plain terraform in its own working directory and its own VPC against the same idle endpoint and showed the identical shape: destroy the higher index only (Plan: 0 to add, 0 to change, 1 to destroy), create the higher index back under a new GroupId (sg-697f57f8128edc558 -> sg-9801c05dba1e82f94), the lower index's GroupId (sg-5f9fbe3d4127b1cb7) unchanged both times; torn down completely before the choudoufu side ran. SYNTHETIC BLOCK, and why: terraform-aws-rds declares no scalable knob anywhere - the root example has no count and no for_each at all, and every count in the vendored module is a boolean create toggle (var.create ? 1 : 0 and friends, grepped not guessed) - while the only genuinely scalable knobs belong to the supporting module.vpc, whose subnet families each destroy an untaggable aws_route_table_association sibling alongside the taggable aws_subnet, which is issue #410's family (corpus-overture-tiles's count-shrink extension) and would make this stage a re-measurement of that gap rather than of count scaling. So the sanctioned fallback per live/GAUNTLET.md #8, reference-ec2-vpc's Part F and corpus-iam-policy's Part G: a self-contained aws_security_group.count_test at an address nothing else names, appended after day2_remove's completed removal, of a taggable type this estate already exercises, introducing nothing that generates a secret. BREAK_COUNT=1 is the negative control and was run for real: it asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, proving the index assertion is load-bearing. | +| Replace with create_before_destroy | pass | 2m50s | choudoufu: changing module.db's ForceNew db_name argument (plus identifier, for an observable identity change) proposed exactly one instance replace at the same declared address, cascading into its 2 cloudwatch log groups and db parameter group (all replaced, all named from identifier) - 4 to add, 4 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql) is confirmed gone and the new instance (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql-replaced) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new identifier, not the destroyed one (complete-postgresql -> complete-postgresql-replaced); the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here). | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | +| Plan, review, apply | pass | 20s | one argument edited (module "db_default"'s tags gain Reviewed=yes - in-place, so no live id moves, and reaching exactly ONE live object because that module call runs with create_db_option_group=false and create_db_parameter_group=false, where the same edit on module.db would also reach its parameter group, option group and log groups), "plan -out=approved.tfplan" wrote a 134369-byte stock-format plan file whose whole change set is one update on module.db_default.module.db_instance.aws_db_instance.this[0]; the world then moved out of band (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql's Example tag, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming the extra row as " module.db.module.db_instance.aws_db_instance.this[0] Update db-52C3D597C3FC4A7BAE689183" - both module.db.module.db_instance.aws_db_instance.this[0] and the live identity it was computed against, which for an aws_db_instance is RDS's own server-minted DbiResourceId db-52C3D597C3FC4A7BAE689183 rather than the ARN or the client-chosen identifier, read off arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql through the AWS CLI and compared by value - with "Exit status 3" spelled out for a pipeline; nothing was applied - arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql-default-7e26b5c33311bceebccb071915 still carried no Reviewed tag, read back through rds list-tags-for-resource rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Example tag put back to its pre-tamper value and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql-default-7e26b5c33311bceebccb071915 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (Reviewed gone, the primary instance's Example tag still "complete-postgresql", next plan proposes no resource action) so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | | Greenfield apply | pass | 3m23s | 39 resources from nothing (same DELTA reduction cold_deploy itself needs - two emulator gaps, floci-io/floci#51 and lex00/floci#52), primary DB instance and security group markers verified via the AWS CLI, 39 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (DB engine/version/class/storage/port, security-group rule count) | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 11m51.2s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 12m10s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. RE-CROSSED FOR REAL 2026-08-21 at dbc5ebf575 (GitHub issue #346's fix), twice, the second time with an instrumented copy of the script that dumps live-plan's raw output so the diagnostics could be quoted rather than inferred. STAGES UNCHANGED at 2 of 5, and #346's fix DOES NOT REACH THIS ESTATE - which refutes the 'four estates share one wall' framing #346 was filed under. Stage 1: 'Apply complete! Resources: 39 added, 0 changed, 0 destroyed'. Stage 2: '26 of 39 eligible (20 VERIFIED + 6 DRIFTED); 13 skipped (UNTAGGABLE by provider schema)'; '26 stamped, 1 recorded (random_id.snapshot_identifier), 0 failed, 12 skipped'; the primary aws_db_instance's tofu-address/tofu-estate asserted by exact value through the AWS CLI. Stage 3 fails on exactly the same 2 diagnostics as before, same lines, same text: 'Module output not supported in static context' on main.tf:198 (cidr_blocks = module.vpc.vpc_cidr_block, inside the module CALL argument ingress_with_cidr_blocks) and 'Unable to compute static value' on the security-group module's own main.tf:197 (cidr_blocks = compact(split(',', lookup(var.ingress_with_cidr_blocks[count.index], 'cidr_blocks', join(',', var.ingress_cidr_blocks))))). Why the fix misses it, measured rather than guessed: this estate's shape has no each.value anywhere. It is COUNT-indexed, and the module output is consumed as a module-CALL argument, which travels tolerantVariables/rebuildConstructor/moduleOutputValue - a VALUE-shaped route that cannot carry a deferred parent read, because a ParentRef is not a cty.Value. #346's fix is part-shaped and lives on the identity-argument route. A separate mechanism is needed and is filed separately. Also corrected: this file previously recorded run.sh's stage-3 assertions as stale (WANT_CIDX_N=7). They are not, and were not at this commit - the committed script already asserts WANT_CIDX_N=0, WANT_DEFAULT_N=0, WANT_UNRESOLVABLE_N=0, WANT_MODOUT_N=1 and WANT_CASCADE_N=1, which is exactly what the run produced, so the script exits 0 while the estate stays blocked. PRIOR HISTORY BELOW. Landed c239792018/47778a931a (2026-08-18); migrate's fail->pass flip was re-verified 2026-08-18 against current main (cec3c4b9b1) but the committed run.sh still asserted the stale pre-fix '0 eligible' shape as a passing control rather than a real check. Follow-up pass 2026-08-18 (this entry) rewrote run.sh's own assertions to the real, current numbers, re-verified for real in a fresh isolated worktree rather than trusted from the prior note: stage 2 dry run reports '23 of 39 resource instance(s) are eligible for stamping (VERIFIED or DRIFTED)' (18 VERIFIED + 5 DRIFTED), -approve reports '23 resource(s) newly stamped, 0 already stamped, 0 failed, 16 skipped', and the primary aws_db_instance's tofu-address/tofu-estate tags are asserted by exact value straight through the AWS CLI (module.db.module.db_instance.aws_db_instance.this:0 / rds-complete-postgres). The 16 skipped: 13 untaggable by design (aws_route_table_association x9, aws_route, aws_security_group_rule, aws_iam_role_policy_attachment, random_id - no tags argument in the provider schema) and 3 are #305's still-open unadmitted-type gap. test_plan is now asserted against a real live-plan on the really-migrated estate (state file deleted first) with a BREAK=1 negative control, and the real counts differ from what the prior note here claimed: exactly 7 count-index-in-tag sites (#304, all aws_security_group_rule.ingress_with_cidr_blocks) and exactly 3 unadmitted-type sites (#305, the three default-object adopters actually created), not 35 and 5. The prior note's 35/5 figures were measured before this estate had ever actually been migrated (nothing tagged, so the 28 module.vpc sibling-indexing sites it also counted never had anything to resolve against) and before bc9ef26638 ('a resource block with a provably-zero count/for_each has no instance to refuse admission on', already on main) landed, which independently stops aws_default_vpc/aws_vpn_gateway_attachment's two count=0 sites from refusing at all - together accounting for the 35->7 and 5->3 drop. Two unrelated real floci gaps found and filed upstream on the fork: lex00/floci#51 (RDS cross-region backup replication), lex00/floci#52 (SecretsManager RotateSecret wrongly requires a Lambda ARN for RDS-managed rotation) - both worked around in the script with documented EMULATOR GAP deltas so stage 1 could stand up at all. Follow-up pass 2026-08-18 (#313 cross-check, fresh worktree, real run): confirmed the 7 count-index-in-tag sites above are NOT #313's wall. This estate's data.aws_availability_zones usage only feeds local.azs = slice(..., 0, 3), a statically-known length, never a for_each/count keyed on the AZ name values themselves, so #313's static-context diagnostic cannot fire here (grepped the full raw plan output: zero occurrences). Commented on #313 (ruling this estate out, with the slice-vs-for_each distinction as evidence) and re-confirmed #304 is still the sole test_plan blocker. Follow-up pass 2026-08-19: #304 fixed and merged (69038634d0/9aaca0ee10) - internal/live/lint/count_index_domain.go's domain check was evaluating a whole module-call-argument value as one pass/fail unit, so one refused reference anywhere inside a list-of-objects argument poisoned every attribute derived from any part of it, even ones that never read the refused field. New StaticEvaluator.EvaluateStructural/EvalContextTolerant validate each reference individually. Re-verified for real: count-index-in-tag sites on this estate 7->0. Estate still does NOT reach test_plan clean, though: 14 sites (from_port/to_port/protocol/cidr_blocks) now fail a DIFFERENT diagnostic, 'Identity not resolvable from configuration' - internal/live/identity/partialargs.go's tolerantVariables deliberately covers only count/for_each key-set resolution, not per-attribute identity-value rendering (its own doc comment records a past regression from broadening it carelessly). Filed as #323, not attempted - needs its own scoped pass. Plus 18 genuinely-ambiguous element(aws_subnet.*[*].id, count.index) sites, confirmed still correctly refusing (unweakened by #304's fix). #304 left open (not closed) since the titled bug is fixed and verified but this estate still doesn't reach a clean plan for the separate #323 reason. Follow-up pass 2026-08-19 (#321 re-verification, scouting only, no commit): #321's fix generalizes strongly to this second, independent estate - 15 of the 18 element() sites now resolve cleanly (every aws_route_table_association.{public,private}.subnet_id/route_table_id). 3 remain: aws_route_table_association.database's route_table_id goes through coalescelist(A[*].id, B[*].id) wrapping the splat, outside resolveElementCall's bare-splat requirement - the same out-of-scope shape #321's own closing comment already flagged, now confirmed reaching a second real module composition. One NEW site found: aws_security_group_rule.ingress_with_cidr_blocks[0].security_group_id via local.this_sg_id = concat(A.*.id, B.*.id, [""])[0] - concat()+splat+index through a local value, terraform-aws-modules/security-group's universal accessor, high-leverage since every rule resource that module creates uses it. Both new shapes filed as #324, not attempted. #323's 14 sites confirmed unchanged in count, root cause refined: this estate's trigger is cidr_blocks = module.vpc.vpc_cidr_block (a module-output reference into a resource's config-derived attribute) poisoning the whole variable projection, not the lookup()-into-bundled-table pattern #304 fixed - same tolerantVariables scope boundary, a second concrete trigger. Net: test_plan stays fail, 18 diagnostic sites total (4 unresolvable-identity + 7 module-output-static + 7 compute-static, mapping to #324's 4 sites + #323's 14). Three separately-scoped resolver passes stand between this estate and five-of-five, not one. live/e2e/corpus-rds-complete-postgres/run.sh's stage-3 assertions are now stale (still expect #304's old 7-site count-index picture) and need a real update pass, not done here. Follow-up pass 2026-08-19 (#324 item 2 fixed and merged, 80d3766b3e/79ffbe4732): concat(A[*].id, B[*].id, [literal])[N] through local.this_sg_id now resolves generically (reuses #321's own splat/instance-count machinery, handles both the RelativeTraversalExpr and IndexExpr parse shapes HCL produces for a constant vs non-constant index). Confirmed by real absence from this estate's own stage-3 output. Generalizes hard: refusal-probe over the full 250-entry offline corpus shows 'Identity not resolvable from configuration' 67 -> 42 (-25 sites), zero regressions, 15 offline corpus entries improved (all terraform-aws-rds examples, plus autoscaling/complete, ecs/complete, ecs/ec2-autoscaling, lambda/with-vpc-s3-endpoint - offline corpus entries, not necessarily live-crossed estates; corpus-security-group-complete and eks/examples/* unchanged). Item 1 (coalescelist) explicitly left open, unattempted. This estate itself: fixing the concat site surfaced a SEPARATE, previously-masked cascade - module.vpc.vpc_cidr_block feeding module.security_group's var.ingress_with_cidr_blocks, 'Module output not supported in static context' - likely another instance of #313's own deliberately out-of-scope resource-attribute boundary (the same family as corpus-security-group-complete's remaining 7 sites), not yet formally confirmed as such or filed separately. test_plan stays fail; the estate's own crossing script needs a real staleness-update pass (still asserts #304's old picture) before its true current diagnostic count can be read cleanly. Follow-up pass 2026-08-19: #324 item 1 (coalescelist) also fixed and merged (c25957cbdf/49744a5617) - #324 now fully closed, both items. The exact 3 aws_route_table_association.database sites this issue named are confirmed gone from this estate's real live-plan output. Generalizes narrowly but cleanly beyond this estate: refusal-probe shows -14 sites across exactly 2 offline corpus entries (cross-region-replica-postgres, vpc/examples/issues - the latter matching #321's own predicted 8-site count exactly). test_plan still stays fail here - blocked only by the pre-existing, unrelated module-output cascade already noted (likely #313's family, unconfirmed) and #323's still-open 14 sites. All of #321/#324's derivable element/splat/concat/coalescelist work is now done across this estate; what remains needs #323's own dedicated pass plus resolving the module-output cascade, not further quick derivations. Follow-up pass 2026-08-19: #323 fixed and merged (3d62366625/fb95168e63), closed. tolerantVariables now resolves a static leaf independently of a sibling leaf's genuine unresolvability, instead of the whole variable projection being poisoned by one bad reference - traced to configs.staticScopeData.GetInputVariable's own error bail discarding every known leaf along with the one genuinely unknown one. Real crossing re-verified: stage-3 identity refusals 14 -> 2 (both are the SAME underlying cause counted twice - module-output-not-supported and unable-to-compute-static-value both trace to cidr_blocks = module.vpc.vpc_cidr_block -> aws_vpc.this[0].cidr_block). CONFIRMED: this estate's sole remaining test_plan blocker is #313's root cause B (a resource-attribute reference through a module output), the exact same maintainer-scoped-out boundary blocking corpus-security-group-complete's own last 7 sites - not a bug, not derivable further, a pure scope decision. Generalizes cleanly: refusal-probe -204 sites across 14 offline corpus entries (11 rds examples, 2 ecs examples, autoscaling/complete), zero instances gained/lost anywhere, zero regressions - improves diagnostics, unblocks nothing further by itself (as expected, since the poisoning fix doesn't touch the genuinely-unresolvable leaf). run.sh's stage-3 assertions are now confirmed stale in a new way too: WANT_CIDX_N=7 has actually been 0 since #304 landed, not just uncounted - needs a real update pass reflecting the estate's true current picture (count-index 0, unadmitted-type 0, both remaining diagnostics tracing to the single #313-root-cause-B leaf), deliberately left to the orchestrator's own call rather than the fixing agent's. diff --git a/site/content/docs/progress/corpus-s3-bucket-complete.md b/site/content/docs/progress/corpus-s3-bucket-complete.md index 60bcad7ee8..51bb215583 100644 --- a/site/content/docs/progress/corpus-s3-bucket-complete.md +++ b/site/content/docs/progress/corpus-s3-bucket-complete.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m29s | 30 resources added by plain terraform, 4 buckets confirmed live, no tofu-address tag | -| Migrate | pass | 1m17s | 6 of 30 stamped, 1 recorded (random_pet, issue #340), 23 skipped (untaggable), 0 failed, 26 identities recorded (#364 unit A2); markers survived the residue-classification apply | +| Cold deploy | pass | 1m13s | 30 resources added by plain terraform, 4 buckets confirmed live, no tofu-address tag | +| Migrate | pass | 1m21s | 6 of 30 stamped, 1 recorded (random_pet, issue #340), 23 skipped (untaggable), 0 failed, 26 identities recorded (#364 unit A2); markers survived the residue-classification apply | | Replan from nothing | pass | 4s | no resource action proposed; 29 rendered identity occurrences (11 distinct), all naming known roots | | No-op apply | pass | 11s | no-op apply (0 added, 0 changed, 0 destroyed); bucket count unchanged at 4 | | Drift and reconverge | pass | 20s | accelerate config drifted to Enabled, exactly 1 change proposed and applied, reconverged to Suspended, final plan empty | -| Rename | pass | 23s | moved block: module.cloudfront_log_bucket renamed to module.cloudfront_log_bucket_renamed with zero churn (0 add, 1 change, 0 destroy), the bucket's tofu-address marker rewritten in place; live-mv: module.simple_bucket renamed to module.simple_bucket_renamed with zero churn, marker rewritten in place; both live bucket names unchanged, read via the AWS CLI; the post-rename plan proposes no resource action | -| Remove a block | pass | 19s | choudoufu: deleting module.simple_bucket_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy: the bucket and its untaggable public_access_block child), applied cleanly (0 added, 0 changed, 2 destroyed), the bucket is genuinely gone from the live account (head-bucket on simple-heroic-terrier now fails, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state also proposes exactly the same two destroys for the same two objects; the target was chosen to avoid issue #404's shape (a sibling policy re-reading the removed bucket's own ARN) - module.log_bucket and module.s3_bucket are both left untouched | -| Change count | pass | 56s | synthetic block (this estate's root configuration declares no count or for_each at all, and its only multi-instance knob - the intelligent_tiering input's two-entry map - fans out onto aws_s3_bucket_intelligent_tiering_configuration, which has no tags argument, issue #410's untaggable-child shape; so aws_s3_bucket.count_test, a taggable type this estate already exercises, at an address no other stage uses). choudoufu: scaling count_test from 2 to 1 proposed exactly one destroy (0 add, 0 change, 1 destroy), and it was the HIGHER index - count_test[1], s3-bucket-complete-count-test-1 - with count_test[0] not appearing in the plan at all; applied cleanly (0 added, 0 changed, 1 destroyed); head-bucket on s3-bucket-complete-count-test-1 then fails outright, while the survivor s3-bucket-complete-count-test-0 keeps both its CreationDate (2026-09-06T01:17:42+00:00) and its tofu-address=aws_s3_bucket.count_test:0 marker, read through the AWS CLI rather than choudoufu's own report, and count_test[1]'s local record is tombstoned (has tombstone, no identity - the #398-guard shape) rather than deleted. Scaling back from 1 to 2 proposed exactly one create (1 add, 0 change, 0 destroy) for count_test[1] alone, and the recreated bucket is genuinely a NEW object: same deterministic name (an S3 bucket's id IS its own name - probed against floci with no tofu in the loop) but a new CreationDate (2026-09-06T01:17:42+00:00 -> 2026-09-06T01:18:18+00:00), re-stamped tofu-address=aws_s3_bucket.count_test:1 (tags do not survive a delete+recreate on this image, so the marker is one the apply wrote again), and a record identity naming the recreated bucket; count_test[0]'s CreationDate and marker were unchanged at every step. The next plan proposes no resource action. G-ORACLE, plain terraform standing the identical 2-instance block up for real in the idle greenfield-oracle account, shows the identical shape: destroy the higher index only (0 add, 0 change, 1 destroy), recreate it under the same deterministic bucket name but a new CreationDate (2026-09-06T01:14:51+00:00 -> 2026-09-06T01:14:59+00:00), index 0's CreationDate (2026-09-06T01:14:51+00:00) unchanged at both steps. BREAK_COUNT=1 asserts the wrong instance (count_test[0]) was destroyed and reports this stage fail, as the stage's Break text requires. | -| Replace with create_before_destroy | pass | 21s | choudoufu: changing module.log_bucket's ForceNew bucket argument proposed exactly one bucket replace at the same declared address, cascading into its ownership_controls, policy and public_access_block (all replaced) plus module.s3_bucket's own logging target_bucket (updated in-place) - 4 to add, 1 to change, 4 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old bucket (logs-heroic-terrier) is confirmed gone and the new bucket (logs-heroic-terrier-replaced) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new bucket, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly ("Live resource displaced from the address it is marked for", naming the manufactured bucket, proposing nothing for it) rather than silently proposed as nothing - the name-derived-identity shape of this diagnostic, distinct from EC2/SQS's fungible-set "Two live resources claiming one slot" because aws_s3_bucket's identity resolves straight from the config's own computed name rather than only through a marker sweep. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones. | +| Rename | pass | 22s | moved block: module.cloudfront_log_bucket renamed to module.cloudfront_log_bucket_renamed with zero churn (0 add, 1 change, 0 destroy), the bucket's tofu-address marker rewritten in place; live-mv: module.simple_bucket renamed to module.simple_bucket_renamed with zero churn, marker rewritten in place; both live bucket names unchanged, read via the AWS CLI; the post-rename plan proposes no resource action | +| Remove a block | pass | 20s | choudoufu: deleting module.simple_bucket_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy: the bucket and its untaggable public_access_block child), applied cleanly (0 added, 0 changed, 2 destroyed), the bucket is genuinely gone from the live account (head-bucket on simple-smart-raven now fails, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state also proposes exactly the same two destroys for the same two objects; the target was chosen to avoid issue #404's shape (a sibling policy re-reading the removed bucket's own ARN) - module.log_bucket and module.s3_bucket are both left untouched | +| Change count | pass | 56s | synthetic block (this estate's root configuration declares no count or for_each at all, and its only multi-instance knob - the intelligent_tiering input's two-entry map - fans out onto aws_s3_bucket_intelligent_tiering_configuration, which has no tags argument, issue #410's untaggable-child shape; so aws_s3_bucket.count_test, a taggable type this estate already exercises, at an address no other stage uses). choudoufu: scaling count_test from 2 to 1 proposed exactly one destroy (0 add, 0 change, 1 destroy), and it was the HIGHER index - count_test[1], s3-bucket-complete-count-test-1 - with count_test[0] not appearing in the plan at all; applied cleanly (0 added, 0 changed, 1 destroyed); head-bucket on s3-bucket-complete-count-test-1 then fails outright, while the survivor s3-bucket-complete-count-test-0 keeps both its CreationDate (2026-09-07T02:46:16+00:00) and its tofu-address=aws_s3_bucket.count_test:0 marker, read through the AWS CLI rather than choudoufu's own report, and count_test[1]'s local record is tombstoned (has tombstone, no identity - the #398-guard shape) rather than deleted. Scaling back from 1 to 2 proposed exactly one create (1 add, 0 change, 0 destroy) for count_test[1] alone, and the recreated bucket is genuinely a NEW object: same deterministic name (an S3 bucket's id IS its own name - probed against floci with no tofu in the loop) but a new CreationDate (2026-09-07T02:46:16+00:00 -> 2026-09-07T02:46:52+00:00), re-stamped tofu-address=aws_s3_bucket.count_test:1 (tags do not survive a delete+recreate on this image, so the marker is one the apply wrote again), and a record identity naming the recreated bucket; count_test[0]'s CreationDate and marker were unchanged at every step. The next plan proposes no resource action. G-ORACLE, plain terraform standing the identical 2-instance block up for real in the idle greenfield-oracle account, shows the identical shape: destroy the higher index only (0 add, 0 change, 1 destroy), recreate it under the same deterministic bucket name but a new CreationDate (2026-09-07T02:42:47+00:00 -> 2026-09-07T02:42:55+00:00), index 0's CreationDate (2026-09-07T02:42:47+00:00) unchanged at both steps. BREAK_COUNT=1 asserts the wrong instance (count_test[0]) was destroyed and reports this stage fail, as the stage's Break text requires. | +| Replace with create_before_destroy | pass | 20s | choudoufu: changing module.log_bucket's ForceNew bucket argument proposed exactly one bucket replace at the same declared address, cascading into its ownership_controls, policy and public_access_block (all replaced) plus module.s3_bucket's own logging target_bucket (updated in-place) - 4 to add, 1 to change, 4 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old bucket (logs-smart-raven) is confirmed gone and the new bucket (logs-smart-raven-replaced) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new bucket, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly ("Live resource displaced from the address it is marked for", naming the manufactured bucket, proposing nothing for it) rather than silently proposed as nothing - the name-derived-identity shape of this diagnostic, distinct from EC2/SQS's fungible-set "Two live resources claiming one slot" because aws_s3_bucket's identity resolves straight from the config's own computed name rather than only through a marker sweep. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 2m45s | 29 resources from nothing (SCOPE REDUCTION's own reduced count, random_pet pinned to a literal on both sides), 3 of 4 bucket markers verified via the AWS CLI, 26 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on all 4 buckets (versioning, default encryption, policy presence) | +| Plan, review, apply | pass | 35s | one argument edited (module.cloudfront_log_bucket gains tags = { Reviewed = "yes" }, reaching aws_s3_bucket.this[0] and nothing else the module creates - that module call sets no attach_*_policy input, so the module's own policy-document data sources are count = 0 for it and one tag is one row), "plan -out=approved.tfplan" wrote a 84803-byte stock-format plan file whose whole change set is one update on module.cloudfront_log_bucket.aws_s3_bucket.this[0]; the world then moved out of band (s3-bucket-smart-raven's transfer-acceleration status flipped to Enabled through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live bucket from the one under review) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.s3_bucket.aws_s3_bucket_accelerate_configuration.this[0] and the live s3-bucket-smart-raven it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - get-bucket-tagging on cloudfront-logs-smart-raven still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the accelerate status put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and cloudfront-logs-smart-raven read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 2m49s | 29 resources from nothing (SCOPE REDUCTION's own reduced count, random_pet pinned to a literal on both sides), 3 of 4 bucket markers verified via the AWS CLI, 26 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on all 4 buckets (versioning, default encryption, policy presence) | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 8m4.5s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 8m31s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed 1398574b03 (2026-08-18); re-verified for real 2026-08-18 in an isolated worktree against local main bcf78bacbd, full pipeline reproduced clean twice in a row. #306 is fixed (lex00/floci S3ControlController.tagResource now reads-merges-writes; reconciled into the currently-pinned floci image; internal/live/lifecycle/marker_tag_merge_live_test.go's TestMarkerSurvivesIncrementalTagUpdate passes live against it). DELTA 6 (the tags-removal workaround) reverted - module.s3_bucket's real tags = { Owner = "Anton" } is back, unmodified, and the stage 2c residue-classification apply leaves all four buckets' markers intact with it restored. DELTA 5 (expected_bucket_owner) is UNCHANGED, for a reason independent of #306: it is ForceNew on most of these sub-resource types with no live representation at all (a request-header assertion, not a stored property), so a discovery-rebuilt prior would force a real replace on the very first onboarding apply - the same shape as DELTA 3's random_pet. The prior investigation's other open question - module.s3_bucket's canned acl and website.routing_rules - was checked for real rather than assumed fixed, and DOES still block test_plan after #306's fix, confirmed reproducibly. Traced past "the provider's Read() needs a genuinely-remembered prior" (the prior investigation's framing, which attributed this to #306 too - a conflation this session's trace evidence rules out, since the mechanism is unrelated to tags/floci entirely): issue #275's residue mechanism (internal/live/projection/residue.go) DOES run here and DOES try to classify acl - TF_LOG=trace on the stage 2c apply shows 'residue candidate "acl": readA=cty.StringVal("private") readB=cty.StringVal("private") applied=cty.StringVal("private")', and classifyResidue's own documented rule reads that as "the provider answers this from the remote", so nothing is recorded as residue. Yet the SAME attribute, read moments later by internal/live/projection/build.go's importAndRead (the function stage 3's plan actually uses), comes back empty and proposes '+ acl = "private"'. The two reads disagree because their prior is built two different ways: residue.go's identityOnly nulls every non-identity attribute outright, while importAndRead's prior is whatever provider.ImportResourceState() returned - a zero-value stub (acl = "", not null) for this SDKv2 resource. The provider's Read() treats an explicit null differently from an SDKv2 zero-value string for this attribute; reconciling the two prior constructions is real work but importAndRead is the read path every projected resource in this fork goes through, so a fix needs corpus-wide validation, not a one-estate patch - out of scope here. SCOPED OUT instead (module.s3_bucket's acl and website inputs removed, applied to BOTH the cold-deploy and choudoufu copies so the crossing tests one genuinely-reduced estate throughout, not choudoufu silently abandoning management of something the cold deploy still created) - same discipline corpus-sumaform-aws uses for its own two structural refusals. Estate is now 30 instances across 14 aws_s3_bucket_* types (down from 32/15; aws_s3_bucket_acl survives elsewhere via module.cloudfront_log_bucket's own grant/owner-driven ACL). Also hit and fixed IN THIS SCRIPT (no choudoufu/floci change needed): THE OUTPUTS QUIRK already documented in corpus-iam-read-only-policy and corpus-root-dns-zones - module.s3_bucket's outputs, re-exported at root, have no prior baseline across a stateless live-plan, so every run shows a permanent 'Changes to Outputs' section and OpenTofu's renderer never prints a 'Plan: N to add...' line while that is true, empty of resource changes or not. Stage 3/5's empty-plan assertions now check for the absence of a resource action header ('will be (created|updated|destroyed)') instead of grepping for 'No changes.'/'Plan: 0 to add...', matching those two scripts' own pattern; this was never exercised before since stage 3 had never previously reached an actually-empty-of-resource-changes plan. Also fixed the rendered-import-identity count check (N_IDS), which had silently never been exercised either: several of module.s3_bucket's own sub-resource types (ownership_controls, versioning, policy, logging, cors_configuration, lifecycle_configuration, object_lock_configuration, public_access_block, accelerate_configuration, request_payment_configuration) import by the bucket's own id alone, so after sort -u only 10 distinct strings remain for 30 instances - the raw (pre-dedup) occurrence count (29) is what actually approximates "one per instance" and is now what the sanity floor checks; the deduplicated set is still what ids_all_name_known_roots validates. diff --git a/site/content/docs/progress/corpus-security-group-complete.md b/site/content/docs/progress/corpus-security-group-complete.md index 326696024e..d3153ba274 100644 --- a/site/content/docs/progress/corpus-security-group-complete.md +++ b/site/content/docs/progress/corpus-security-group-complete.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 31s | 67 resources (DELTA 2, lex00/floci#57) | -| Migrate | pass | 1m15s | 58 of 67 stamped, 67 identities recorded (#364 unit A2) | -| Replan from nothing | pass | 4s | the plan is genuinely empty: every choudoufu wall (#305, #307, #313 A and B, #321, #332) and both confirmed floci gaps (#102, #104) are fixed or absent this run; default route table identities asserted by value against the AWS CLI in step 3a | +| Cold deploy | pass | 18s | 67 resources (DELTA 2, lex00/floci#57) | +| Migrate | pass | 1m14s | 58 of 67 stamped, 67 identities recorded (#364 unit A2) | +| Replan from nothing | pass | 5s | the plan is genuinely empty: every choudoufu wall (#305, #307, #313 A and B, #321, #332) and both confirmed floci gaps (#102, #104) are fixed or absent this run; default route table identities asserted by value against the AWS CLI in step 3a | | No-op apply | pass | 4s | no-op apply (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 58 objects, read through resourcegroupstaggingapi | -| Drift and reconverge | pass | 6s | one object tampered (DriftProbe tag on the main security group), exactly module.security_group.aws_security_group.this[0] proposed, apply changed 1 and the tag is gone, confirmed via the AWS CLI | -| Rename | pass | 13s | moved block: module.postgresql renamed to module.postgresql_renamed with zero churn (0 add, 4 change, 0 destroy) - the rule-children case, its own SG plus ingress/egress rules and rules_exclusive all moving under one moved block; live-mv: aws_security_group.app renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 10s | choudoufu: deleting module.postgresql_renamed's block proposed exactly 5 destroys (0 add, 0 change, 5 destroy: SG + 2 ingress + 1 egress + 1 untaggable rules_exclusive), applied cleanly (0 added, 0 changed, 5 destroyed), the security group is genuinely gone from the live account (0 matches on describe-security-groups for the old id, read via the AWS CLI, not choudoufu's own report), and the next plan proposes nothing; stock oracle on cold_deploy's own state (D-ORACLE remove) also proposes exactly 5 destroys for the same 5 objects; classifyOrphans did not withhold the untaggable rules_exclusive destroy even though module.security_group's and module.consul's own rules_exclusive instances share its block key, because both surviving instances are bound, not unclaimed | -| Change count | pass | 29s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-294fab44394fefbdc) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-e5ba6f0e8c24d339b, was sg-82428bbcae2cf1efa - a security group id is never reused, so the destroy is directly observable rather than inferred) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] kept both its id and its marker throughout; the local record store tracked the same facts by value at every step (index 0's record never moved, index 1's stopped naming the destroyed object and then named the new one); the next plan is empty. G-ORACLE is the stock leg for the identical shape, applied for real with plain terraform in the idle greenfield account: 3 created, then 0 add/0 change/1 destroy hitting count_test[1] only, then 1 add/0 change/0 destroy bringing it back under a new id, count_test[0]'s id unchanged both times - identical to choudoufu's. SYNTHETIC BLOCK, and why: this estate declares no scalable knob of its own - every count in terraform-aws-security-group v6.0.0 is the boolean 'local.create ? 1 : 0' toggle, and its real for_each maps (var.ingress_rules/var.egress_rules) cannot be scaled in isolation because main.tf line 96 feeds every rule id into aws_vpc_security_group_rules_exclusive.this[0], making the plan 0 add/1 change/1 destroy instead of the exact shape this stage asserts; aws_security_group.count_test is self-contained, named by nothing else here, and of a type this estate already exercises six times over (reference-ec2-vpc Part F and corpus-iam-policy Part G are the precedent). BREAK_COUNT=1 runs this stage's Break control (expect count_test[0] to be the destroyed one) and correctly fails to hold. | -| Replace with create_before_destroy | pass | 11s | choudoufu: changing module.security_group's ForceNew name argument proposed exactly one SG replace at the same declared address, cascading into its 7 ingress rules, 1 egress rule and 1 rules_exclusive enforcer (all replaced) - 10 to add, 10 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old SG (sg-b281cdf25f7c3e452) is confirmed gone and the new SG (sg-74ab044e1dc941d6e) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new SG, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly ("Indistinguishable instances without per-instance markers", naming both live security groups) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | +| Drift and reconverge | pass | 7s | one object tampered (DriftProbe tag on the main security group), exactly module.security_group.aws_security_group.this[0] proposed, apply changed 1 and the tag is gone, confirmed via the AWS CLI | +| Rename | pass | 14s | moved block: module.postgresql renamed to module.postgresql_renamed with zero churn (0 add, 4 change, 0 destroy) - the rule-children case, its own SG plus ingress/egress rules and rules_exclusive all moving under one moved block; live-mv: aws_security_group.app renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Remove a block | pass | 11s | choudoufu: deleting module.postgresql_renamed's block proposed exactly 5 destroys (0 add, 0 change, 5 destroy: SG + 2 ingress + 1 egress + 1 untaggable rules_exclusive), applied cleanly (0 added, 0 changed, 5 destroyed), the security group is genuinely gone from the live account (0 matches on describe-security-groups for the old id, read via the AWS CLI, not choudoufu's own report), and the next plan proposes nothing; stock oracle on cold_deploy's own state (D-ORACLE remove) also proposes exactly 5 destroys for the same 5 objects; classifyOrphans did not withhold the untaggable rules_exclusive destroy even though module.security_group's and module.consul's own rules_exclusive instances share its block key, because both surviving instances are bound, not unclaimed | +| Change count | pass | 30s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-0999dad1a4c3c06d5) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-2e2ea3c3128e173a8, was sg-6d6ee0da0cdf72206 - a security group id is never reused, so the destroy is directly observable rather than inferred) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] kept both its id and its marker throughout; the local record store tracked the same facts by value at every step (index 0's record never moved, index 1's stopped naming the destroyed object and then named the new one); the next plan is empty. G-ORACLE is the stock leg for the identical shape, applied for real with plain terraform in the idle greenfield account: 3 created, then 0 add/0 change/1 destroy hitting count_test[1] only, then 1 add/0 change/0 destroy bringing it back under a new id, count_test[0]'s id unchanged both times - identical to choudoufu's. SYNTHETIC BLOCK, and why: this estate declares no scalable knob of its own - every count in terraform-aws-security-group v6.0.0 is the boolean 'local.create ? 1 : 0' toggle, and its real for_each maps (var.ingress_rules/var.egress_rules) cannot be scaled in isolation because main.tf line 96 feeds every rule id into aws_vpc_security_group_rules_exclusive.this[0], making the plan 0 add/1 change/1 destroy instead of the exact shape this stage asserts; aws_security_group.count_test is self-contained, named by nothing else here, and of a type this estate already exercises six times over (reference-ec2-vpc Part F and corpus-iam-policy Part G are the precedent). BREAK_COUNT=1 runs this stage's Break control (expect count_test[0] to be the destroyed one) and correctly fails to hold. | +| Replace with create_before_destroy | pass | 11s | choudoufu: changing module.security_group's ForceNew name argument proposed exactly one SG replace at the same declared address, cascading into its 7 ingress rules, 1 egress rule and 1 rules_exclusive enforcer (all replaced) - 10 to add, 10 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old SG (sg-5eb49924036c81c44) is confirmed gone and the new SG (sg-9260d7ad3ccb5142b) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new SG, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly ("Indistinguishable instances without per-instance markers", naming both live security groups) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 27s | 67 resources from nothing, all markers verified via the AWS CLI, 67 records in the local record store (#364 A2), replan empty, 6 tagged security groups (4 named + 2 default adopters) and every named one's rule shape matches $PLAIN_EST's own stage-1 apply object by object, tags stripped | +| Plan, review, apply | pass | 18s | one argument edited (aws_security_group.app's tags merge in Reviewed=yes; the standalone app SG has no rule children, so one argument is one row), "plan -out=approved.tfplan" wrote a 87608-byte stock-format plan file whose whole change set is one update on aws_security_group.app (sg-641be14b280c2e82e); the world then moved out of band (a DriftProbe tag on sg-5eb49924036c81c44 through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live security group from the one under review) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.security_group.aws_security_group.this[0] and the live sg-5eb49924036c81c44 it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - describe-tags on sg-641be14b280c2e82e still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the DriftProbe tag deleted and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-641be14b280c2e82e read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the app SG confirmed to be the same GroupId it started as, and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 26s | 67 resources from nothing, all markers verified via the AWS CLI, 67 records in the local record store (#364 A2), replan empty, 6 tagged security groups (4 named + 2 default adopters) and every named one's rule shape matches $PLAIN_EST's own stage-1 apply object by object, tags stripped | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m30.6s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m38.7s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed c876435875 (2026-08-18), 67 real resources. The brief expected this crossing to hit #304 (a static lookup()-keyed count-index) directly, since a prior crossing found #304 through this same module as a dependency - checked, not assumed, and refuted: v6.0.0 rewrote the module from the classic single-aws_security_group-with-dynamic-blocks shape #304 lives in to a for_each-over-a-map shape emitting aws_vpc_security_group_ingress_rule/egress_rule per rule key, so that whole pattern is gone from this version's own example. migrate genuinely passes (52 of 67 eligible and stamped: 35 VERIFIED + 17 DRIFTED; 15 skipped - 6 untaggable by design, 9 unadmitted). test_plan blocked by two real, distinct, filed gaps: #305 (6 sites, the familiar default_* adopter trio, doubled since this estate nests two terraform-aws-vpc calls) and a NEW one, filed as #307: aws_vpc_security_group_rules_exclusive is unadmitted (3 sites) - no CFN counterpart and the pinned provider ships no identity schema for it, but its own import docs are unambiguous that security_group_id (required, ForceNew, always a direct parent reference) is its whole identity, the same shape aws_vpc_security_group_vpc_association's already-admitted row has. Also found and filed a real floci gap (lex00/floci#57: EC2 AssociateSecurityGroupVpc has no handler at all), worked around with a documented delta removing the estate's one vpc_associations block (67 of 68 resources still stand up for real). One open, honestly-unresolved observation: 17 of the DRIFTED resources show referenced_security_group_id read back as "000000000000/sg-xxx" from floci where config computes a bare "sg-xxx" - doesn't block stamping, but live-plan never got past #305/#307's hard refusals to reveal whether the real provider's diff-suppression absorbs this cleanly or would surface as an 18th change. Not filed separately; the next crossing attempt (once #305/#307 land) will show it for real. Follow-up pass 2026-08-19 (#313's data-source fix, c636ab20f7/0284d8c408): re-verified for real against the current pin (67 cold-deployed, 58 stamped, state deleted). test_plan diagnostics dropped from 239 to 19 - #313's canonical data.aws_availability_zones cause (50 sites) is fully gone, the module-output/resource-attribute variant (2 sites) still correctly refuses (out of scope by the maintainer's own ruling), and the 187-site cascade collapsed to 5. The remaining 12 sites are a NEW, newly-reached class (previously masked behind #313's hard refusal): element([*].id, count.index) in the vpc module's aws_route_table_association.private - both operands are tagged resources, the obstacle is the splat-through-function-call shape, not an admission gap. Filed as #321, not attempted (scouting slot). test_plan therefore stays fail, but the estate's real remaining blocker is now #321 (a derivable, no-design-call-needed gap) rather than #313 (an architecture question) - #321 is the clear next step toward this estate's five-of-five and the core set's last gap. Follow-up pass 2026-08-19 (#321 fixed and merged, c33a47288a/626ca84739): element([*].attr, idx) over a splat of tagged resources now resolves generically - it names the same live object a direct indexed traversal already resolves, via element()'s own modulo wraparound. Re-verified for real: test_plan diagnostics 19 -> 7. The 12 splat-through-element sites are confirmed cleared by their absence from real live-plan output; the remaining 7 are #313's own deliberately-out-of-scope resource-attribute root cause (root cause B), a maintainer scope boundary, not a bug. test_plan stays fail - the core set does NOT reach five-of-five from this fix alone. Generalizes beyond this estate: refusal-probe over terraform-aws-modules/vpc's own examples, 21 -> 8 sites, zero regressions - three configs (ipam, ipv6-only, outpost) fully cleared. A related but distinct lint-side wall (RuleCountIndex, 32 sites/6 configs, a genuine unresolved composite-vs-per-argument injectivity design question) stays independently blocking and was left open, documented rather than attempted. Follow-up pass 2026-08-19 (#191 fixed and merged, 312acbbb61/75ef0a6a78): internal/live/identity/partialargs.go's tolerant rebuild now composes across more than one module call and evaluates a call the caller wrote (merge(), not a bare constructor) through an evaluator whose own var.* closure is already tolerant, one module up. module.consul's ingress_referenced_security_group_id map no longer poisons the 22 ingress rules it seeds two module calls down - the map's KEYS (eleven preset names crossed with one caller key) were always written down; only the VALUE under one key was ever unknowable, and it still is. Re-verified for real against floci, script exit 0, BREAK=1 negative control correctly fails: test_plan diagnostics 7 -> 4, and every analysis-layer refusal this estate has ever hit is now 0 and asserted by absence (#305, #307, #313 root causes A and B both, #321). What newly reached PROJECTION and blocks the estate now is #332 (not #313 - the old 'Unable to use aws_security_group.app in static context' framing is confirmed gone): aws_default_route_table imports by the VPC's id, not its own, and the ratified row says otherwise - 2 'Cannot import for projection' + 2 'empty result', one pair per nested vpc module call, both traced to the same type. #332 is filed, not fixed here; it is now the sole remaining blocker on this estate and on the core set's last five-of-five gap. #332 fixed 2026-08-19 (859c1ad747/ff1f6bcdea/c1197befc7): the ratified row claimed aws_default_route_table imports by the route table's own rtb-… id; the real provider imports it by the VPC's id, read off the vpc_id ATTRIBUTE (not argument) the discovered object already carries - settled by running stock terraform 1.15.8 + hashicorp/aws 6.59.0 (Error: empty result for rtb-…, Import successful! for vpc-…). Reach stated honestly: one type today (defaultAdopterSiblings/sameRatifiedIdentity in internal/live/discovery/discovery.go split "same live object" from "same import identity" generically, off each type's own ratified IdentityAttrs/ImportSyntax, no type name in the control flow - #302's aws_iam_service_linked_role/aws_iam_role pair already exercises the same recomposition path; aws_default_route_table is simply the only aws_default_* row that currently diverges from its plain sibling at aws 6.59.0). Re-verified independently 2026-08-20 with a fresh real crossing against floci (ghcr.io/lex00/floci@sha256:120b6783c7fb48d3d78245056251492b7d9246cdc3b397c98db2659cdc78d94a, the currently pinned image): STAGE 1 PASS, STAGE 2 PASS, STAGE 3 BLOCKED at exactly 1 site (was 239, then 19, then 7, then 4) - every choudoufu-layer refusal this estate has ever hit (#305, #307, #313 root causes A and B, #321, #332) confirmed absent, and step 3a re-derives each nested VPC's default route table import identity from AWS directly (module.vpc: vpc-9eceebf0 -> rtb-7a9e0d017620e163c; module.vpc_secondary: vpc-91d40754 -> rtb-b66d5dc82e63b6e38) and asserts it BY VALUE, not by absence. The 1 remaining site is the AWS provider answering "Provider produced invalid plan" on its own requires-replacement path for module.security_group.aws_vpc_security_group_ingress_rule.this["dns-from-prefix-list"] (cty.Path{cty.GetAttrStep{Name:""}}), explicitly a provider bug per the error's own text - filed upstream as #335, genuinely outside this fork's code (the diagnostic names no choudoufu path). test_plan stays "fail" in this table's pass/fail/not_run vocabulary since the plan is not clean, but the estate's own blocker has moved entirely off this fork: #332 was the core set's last derivable gap, and #335 (an AWS-provider defect) is what now stands between this estate and five-of-five. diff --git a/site/content/docs/progress/corpus-simpleinfra-dns.md b/site/content/docs/progress/corpus-simpleinfra-dns.md index 6b45db2260..d92ecfea7f 100644 --- a/site/content/docs/progress/corpus-simpleinfra-dns.md +++ b/site/content/docs/progress/corpus-simpleinfra-dns.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m25s | 35 instances (7 zones, 28 records) from plain terraform, 0 of 7 zones carry tofu-estate | -| Migrate | pass | 40s | 7 stamped, 7 distinct hosted zones, one per module call | +| Cold deploy | pass | 1m10s | 35 instances (7 zones, 28 records) from plain terraform, 0 of 7 zones carry tofu-estate | +| Migrate | pass | 41s | 7 stamped, 7 distinct hosted zones, one per module call | | Replan from nothing | pass | 5s | no resource change proposed, nothing foreign; all 35 rendered identities name a live hosted zone or record set | -| No-op apply | pass | 14s | no-op apply (0 added, 0 changed, 0 destroyed); 7 zones / 28 records unchanged, all 7 markers unmoved | -| Drift and reconverge | pass | 22s | one untaggable record drifted, exactly module.rustconf_com.aws_route53_record.cname["2016"] proposed and applied, TTL reconverged to 300, 28 records and the parent marker intact | -| Rename | pass | 18s | moved block: module.rustaceans_org renamed to module.rustaceans_org_moved with zero churn (0 add, 1 change, 0 destroy) - only the zone's own marker rewritten, its 2 record children (A, CNAME) did not move; live-mv: module.cratesio_com (0 records) renamed to module.cratesio_com_final with zero churn, marker rewritten in place; stock oracle over the identical two-module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy), using per-child moved blocks stock's own state-address tracking requires and choudoufu's stateless untaggable-record derivation does not; both live zone ids unchanged, read via the AWS CLI | -| Remove a block | pass | 28s | choudoufu: deleting module.cratesio_com_final's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the hosted zone is genuinely gone from the live account (route53 get-hosted-zone on the old id now errors, read via the AWS CLI, not choudoufu's own report; 7 zones down to 6), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same zone (before any rename ever touched it) | -| Change count | pass | 53s | choudoufu: dropping "2024" from module.rustconf_com's CNAME for_each map destroyed exactly module.rustconf_com.aws_route53_record.cname["2024"] (0 add, 0 change, 1 destroy), leaving sibling module.rustconf_com.aws_route53_record.cname["2022"]'s TTL and 27 remaining record sets untouched; adding it back created exactly the same key (0 add, 0 change -> 1 add, 0 change, 0 destroy), restoring its TTL/value and the 28 record-set count, while the sibling and the parent zone's own marker stayed untouched throughout; the next plan is empty; a Route 53 record set carries no server-minted identifier of its own (verified directly against floci, no tofu in the loop: ListResourceRecordSets returns a byte-identical entry across a genuine delete/recreate, only ChangeResourceRecordSets' own per-call ChangeInfo.Id differs), so the destroy is proven by verified ABSENCE rather than an id-diff, unlike this stage's aws_iam_policy/PolicyId and EC2/VpcEndpointId precedents; the G-ORACLE stock oracle on the identical for_each change, plan-only on cold_deploy's own state, shows the identical shape: destroy the dropped key only, propose creating it back, every sibling key untouched both times | -| Replace with create_before_destroy | pass | 1m10s | choudoufu: changing module.areweasyncyet_rs's ForceNew domain argument proposed exactly one zone replace at the same declared address, cascading into its one A record - 2 to add, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old zone (ZGL45ZHYYL0082N) is confirmed gone and the new zone (Z0ABC41F7VX38G5) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new zone, not the destroyed one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here). | +| No-op apply | pass | 13s | no-op apply (0 added, 0 changed, 0 destroyed); 7 zones / 28 records unchanged, all 7 markers unmoved | +| Drift and reconverge | pass | 21s | one untaggable record drifted, exactly module.rustconf_com.aws_route53_record.cname["2016"] proposed and applied, TTL reconverged to 300, 28 records and the parent marker intact | +| Rename | pass | 15s | moved block: module.rustaceans_org renamed to module.rustaceans_org_moved with zero churn (0 add, 1 change, 0 destroy) - only the zone's own marker rewritten, its 2 record children (A, CNAME) did not move; live-mv: module.cratesio_com (0 records) renamed to module.cratesio_com_final with zero churn, marker rewritten in place; stock oracle over the identical two-module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy), using per-child moved blocks stock's own state-address tracking requires and choudoufu's stateless untaggable-record derivation does not; both live zone ids unchanged, read via the AWS CLI | +| Remove a block | pass | 24s | choudoufu: deleting module.cratesio_com_final's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the hosted zone is genuinely gone from the live account (route53 get-hosted-zone on the old id now errors, read via the AWS CLI, not choudoufu's own report; 7 zones down to 6), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same zone (before any rename ever touched it) | +| Change count | pass | 50s | choudoufu: dropping "2024" from module.rustconf_com's CNAME for_each map destroyed exactly module.rustconf_com.aws_route53_record.cname["2024"] (0 add, 0 change, 1 destroy), leaving sibling module.rustconf_com.aws_route53_record.cname["2022"]'s TTL and 27 remaining record sets untouched; adding it back created exactly the same key (0 add, 0 change -> 1 add, 0 change, 0 destroy), restoring its TTL/value and the 28 record-set count, while the sibling and the parent zone's own marker stayed untouched throughout; the next plan is empty; a Route 53 record set carries no server-minted identifier of its own (verified directly against floci, no tofu in the loop: ListResourceRecordSets returns a byte-identical entry across a genuine delete/recreate, only ChangeResourceRecordSets' own per-call ChangeInfo.Id differs), so the destroy is proven by verified ABSENCE rather than an id-diff, unlike this stage's aws_iam_policy/PolicyId and EC2/VpcEndpointId precedents; the G-ORACLE stock oracle on the identical for_each change, plan-only on cold_deploy's own state, shows the identical shape: destroy the dropped key only, propose creating it back, every sibling key untouched both times | +| Replace with create_before_destroy | pass | 1m8s | choudoufu: changing module.areweasyncyet_rs's ForceNew domain argument proposed exactly one zone replace at the same declared address, cascading into its one A record - 2 to add, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old zone (Z27X0DRHB3FI7WN) is confirmed gone and the new zone (ZMBCC3J2JR88MMI) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new zone, not the destroyed one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here). | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 2m35s | 35 instances from nothing (7 zones, 28 records), all 7 markers verified via the AWS CLI, replan empty, stock oracle in its own namespace matches structurally on all 7 zones (28 records) | +| Plan, review, apply | pass | 45s | one argument edited (module.arewewebyet_org's ttl 300 -> 600; that call declares exactly one record, the www CNAME, so one argument is one instance), "plan -out=approved.tfplan" wrote a 23430-byte stock-format plan file whose whole change set is one update on module.arewewebyet_org.aws_route53_record.cname["www"] ("Plan: 0 to add, 1 to change, 0 to destroy"); the world then moved out of band (2016.rustconf.com.'s TTL set to 60 in zone ZCO3SP8XZ17EY1E through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance in a DIFFERENT hosted zone from the one under review) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming module.rustconf_com.aws_route53_record.cname["2016"] together with the whole live identity it was computed against, ZCO3SP8XZ17EY1E_2016.rustconf.com_CNAME (this type's ZONEID_NAME_TYPE import syntax, rebuilt in the script from the zone id, record name and type it already knew independently) - with "Exit status 3" spelled out for a pipeline; nothing was applied - www.arewewebyet.org. still read TTL 300 through the AWS CLI, which is stronger evidence than the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the drifted TTL put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and www.arewewebyet.org. read back at TTL 600, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the 28 record sets re-counted and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 2m31s | 35 instances from nothing (7 zones, 28 records), all 7 markers verified via the AWS CLI, replan empty, stock oracle in its own namespace matches structurally on all 7 zones (28 records) | | Strict profile (not a headline stage) | not run | | | -Last run at commit `cb5ae2009f` on 2026-09-06T04:29:56Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 8m10.4s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 8m22.7s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. 35 instances: 7 aws_route53_zone TAGGABLE (7 markers, each verified against the domain that module call declares - a fact the marker does not supply), 28 aws_route53_record UNTAGGABLE and re-derived from their tagged parent zone. Found and corrected a census error in issue #274's own thread first: live/e2e/repeated-module/run.sh has targeted this same directory since #280 but never runs live-import (grep -c live-import -> 0), applies with the live block already declared so genuinely unmarked infrastructure never exists, and has no drift stage - it is not a five-stage crossing despite being counted as one, though it does cover stages 3/4 well. Cold-deployed by real Terraform v1.15.8 (35/0/0), 7 stamped + 28 skipped by live-import -approve, state deleted, live-plan empty with all 35 rendered identities checked as strings against Route 53's own answer, no-op apply 0/0/0 with all 7 markers unmoved. Stage 5 drifts an UNTAGGABLE object out of band - a record set's TTL 300 -> 60 via change-resource-record-sets - which no other crossing's drift stage does: with no state file and no tag on the object, the drift is only visible if the derived-from-tagged identity re-derived correctly. Plan proposes exactly module.rustconf_com.aws_route53_record.cname["2016"] and nothing else; apply 0 added/1 changed/0 destroyed; TTL read back as 300. Three of team-members-access's four deltas recur (#268 mandatory backend edit, #269 provider version skew ~> 5.64 -> = 6.59.0, an emulator provider override), asserted; the fourth (five seeded data-source reads) does not - 0 data blocks anywhere, asserted. One delta deliberately absent: the four trailing-dot record names (#281) are unchanged, asserted at count 4. BREAK=1 exits 1 at stage 2c on the corrupted marker; BREAK_STAGE5=1 exercises a real hole found on adversarial re-read - the first stage-5 assertion counted only will-be-updated addresses, so a destroy or create alongside the expected update would have passed silently under BREAK_STAGE5 (which skips the apply that would eventually have caught it) - fixed by asserting the plan's own totals line first, mutation-checked (clean run reads exactly '0 to add, 1 to change, 0 to destroy', BREAK_STAGE5=1 reads '0 to add, 2 to change, 0 to destroy'). Zero choudoufu defects found. A reusable methodology finding: any crossing whose cold deploy uses real Terraform (not tofu) cannot use corpus-giantswarm-crossplane's lock-file-seeding fix for the 320s uncached-init tax, because terraform and choudoufu resolve providers from different registries (terraform.io vs opentofu.org) so neither's lock file satisfies the other - measured here, plain terraform init >600s (timed out), -plugin-dir= 0.35s. -plugin-dir is strictly better than lock-seeding for tofu-cold-deployed estates too. justfile gained demo-corpus-simpleinfra-dns (port 4741); live/corpus-manifest.json gained the pin. Verified 2026-08-19 at 07b72f9977, 145s. diff --git a/site/content/docs/progress/corpus-sqs-basic.md b/site/content/docs/progress/corpus-sqs-basic.md index 7344df3632..14a92b6dfd 100644 --- a/site/content/docs/progress/corpus-sqs-basic.md +++ b/site/content/docs/progress/corpus-sqs-basic.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m10s | 6 resources added by plain terraform (4 queues + redrive_policy + redrive_allow_policy), 0 objects carry tofu-estate before migration | -| Migrate | pass | 2m1s | 4 of 6 eligible (2 untaggable redrive types resolved by provider identity schema), 4 stamped, 0 failed, 2 skipped; tofu-slot=0 written on all 4 queues by the stamp itself (issue #372's remainder), confirmed by value and by a genuine no-op on the follow-up apply | -| Replan from nothing | pass | 2s | no resource change proposed, no foreign resources; fifo and default queue tofu-address re-checked against SQS | -| No-op apply | pass | 3s | genuine no-op (0 added, 0 changed, 0 destroyed); 4 objects before, 4 after, no state file | +| Cold deploy | pass | 57s | 6 resources added by plain terraform (4 queues + redrive_policy + redrive_allow_policy), 0 objects carry tofu-estate before migration | +| Migrate | pass | 2m0s | 4 of 6 eligible (2 untaggable redrive types resolved by provider identity schema), 4 stamped, 0 failed, 2 skipped; tofu-slot=0 written on all 4 queues by the stamp itself (issue #372's remainder), confirmed by value and by a genuine no-op on the follow-up apply | +| Replan from nothing | pass | 3s | no resource change proposed, no foreign resources; fifo and default queue tofu-address re-checked against SQS | +| No-op apply | pass | 2s | genuine no-op (0 added, 0 changed, 0 destroyed); 4 objects before, 4 after, no state file | | Drift and reconverge | pass | 5s | one object tampered, exactly 1 object proposed and applied (0 added, 1 changed, 0 destroyed), tag reconverged to "ex-complete" | -| Rename | pass | 10s | moved block: module.default_sqs renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.unencrypted_sqs renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 50s | choudoufu: deleting module.unencrypted_sqs_renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (sqs get-queue-url on the old name now returns NonExistentQueue, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same object (before any rename ever touched it) | -| Change count | pass | 1m54s | synthetic block (all four module calls declare count = var.create ? 1 : 0, a boolean create toggle, so the estate has no knob that scales - issue #488's sanctioned fallback, reusing aws_sqs_queue, a type this estate already exercises four times): scaling aws_sqs_queue.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) of the HIGHER index, count_test[1]; the survivor count_test[0] kept its live queue URL (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-count-test-0), its CreatedTimestamp (1788657850) and its tofu-address=aws_sqs_queue.count_test:0 / tofu-slot=0 markers, all read back through the AWS CLI, and count_test[1]'s local record was tombstoned rather than left naming a destroyed queue; scaling 1 back to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy), and because a queue URL is rebuilt from region + account + name the recreated instance comes back at the SAME url - so the destroy is witnessed two other ways instead, by AWS.SimpleQueueService.NonExistentQueue in between and by a strictly later CreatedTimestamp (1788657850 -> 1788657932), with tofu-address=aws_sqs_queue.count_test:1 back on the new object and index 0 untouched throughout; the next plan proposes no resource action; the G-ORACLE stock oracle stood the identical block up for real in the idle greenfield-oracle account and showed the identical shape - destroy the higher index only, create it back under the same url with a new CreatedTimestamp, the lower index unchanged both times | +| Rename | pass | 9s | moved block: module.default_sqs renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.unencrypted_sqs renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Remove a block | pass | 49s | choudoufu: deleting module.unencrypted_sqs_renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (sqs get-queue-url on the old name now returns NonExistentQueue, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same object (before any rename ever touched it) | +| Change count | pass | 1m55s | synthetic block (all four module calls declare count = var.create ? 1 : 0, a boolean create toggle, so the estate has no knob that scales - issue #488's sanctioned fallback, reusing aws_sqs_queue, a type this estate already exercises four times): scaling aws_sqs_queue.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) of the HIGHER index, count_test[1]; the survivor count_test[0] kept its live queue URL (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-count-test-0), its CreatedTimestamp (1788749614) and its tofu-address=aws_sqs_queue.count_test:0 / tofu-slot=0 markers, all read back through the AWS CLI, and count_test[1]'s local record was tombstoned rather than left naming a destroyed queue; scaling 1 back to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy), and because a queue URL is rebuilt from region + account + name the recreated instance comes back at the SAME url - so the destroy is witnessed two other ways instead, by AWS.SimpleQueueService.NonExistentQueue in between and by a strictly later CreatedTimestamp (1788749614 -> 1788749697), with tofu-address=aws_sqs_queue.count_test:1 back on the new object and index 0 untouched throughout; the next plan proposes no resource action; the G-ORACLE stock oracle stood the identical block up for real in the idle greenfield-oracle account and showed the identical shape - destroy the higher index only, create it back under the same url with a new CreatedTimestamp, the lower index unchanged both times | | Replace with create_before_destroy | pass | 1m14s | choudoufu: changing module.default_sqs_renamed's ForceNew name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default) is confirmed gone and the new object (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default -> https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied - it shares floci's account with $EST); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 3m7s | 6 resources from nothing (4 tagged queues + 2 untaggable redrive types), all markers verified via the AWS CLI, 6 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on all 4 queues | +| Plan, review, apply | pass | 13s | one argument edited (module.unencrypted_sqs's tags gain Reviewed=yes; that module call declares no DLQ, so var.tags reaches exactly one resource instance), "plan -out=approved.tfplan" wrote a 26728-byte stock-format plan file whose whole change set is one update on module.unencrypted_sqs.aws_sqs_queue.this[0] (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-unencrypted); the world then moved out of band (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default's Example tag, through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live queue from the one under review) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.default_sqs.aws_sqs_queue.this[0] and the live https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default it was computed against (an aws_sqs_queue's identity IS its URL), with "Exit status 3" spelled out for a pipeline; nothing was applied - list-queue-tags on https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-unencrypted still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Example tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-unencrypted read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the unencrypted queue's URL confirmed unchanged and the estate replanned empty with no state file, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 3m5s | 6 resources from nothing (4 tagged queues + 2 untaggable redrive types), all markers verified via the AWS CLI, 6 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on all 4 queues | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 10m36.1s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 10m32.2s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. First real five-stage crossing of this estate, and the first SQS surface in this corpus. Sourced and sketched at e4b12799da with every assertion past stage 1's `terraform apply` DERIVED from reading terraform-aws-sqs's naming locals rather than measured - that commit's own header said so and deliberately added no entry here. This entry is the first one written from a real run: Docker/floci (ghcr.io/lex00/floci@sha256:8a882bcc, live/floci-image's pin), real hashicorp terraform, and the AWS CLI throughout, in worktree ../wt/new-terraform-estate-2 off e4b12799da. All five stages PASS. WHAT THE DERIVATION GOT WRONG, and it was exactly one thing: the resource count. The estate builds SIX managed resources, not five. `create_dlq = true` makes the module emit an aws_sqs_queue_redrive_ALLOW_policy on the DLQ alongside the aws_sqs_queue_redrive_policy on the source queue; reading the naming locals found the second and missed the first. Everything else the sketch derived was right when checked against reality - all four queue names and URLs including the FIFO DLQ's "-dlq.fifo" suffix, and all four rendered tofu-address strings. Corrected counts, measured: stage 1 "Apply complete! Resources: 6 added", stage 2 "4 of 6 resource instance(s) are eligible for stamping" and "4 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 2 skipped". THE SCHEMA-FALLBACK RESULT, which is why this estate was sourced. Neither aws_sqs_queue_redrive_policy nor aws_sqs_queue_redrive_allow_policy has a row in internal/live/identity/table_generated.go; live/survey-full.json classifies both identically (path "client-named", admission "schema", required_for_import ["queue_url"], taggable false, list_resource false). Both resolved a live id through the provider's own identity schema, and both resolved to the RIGHT queue, which is not the same queue for the two of them: redrive_policy -> https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete.fifo (the source queue), redrive_allow_policy -> https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-dlq.fifo (the DLQ). The schema-fallback path works on both, unmodified - no fix was needed and none was made. The script now asserts both by value, because a run that merely did not error would pass with the two swapped. Both are UNTAGGABLE (no tags argument in the provider schema), so live-import skips them and stage 2 stamps 4 of 6. That is the invariant working, not a shortfall: four tagged queues plus two resources whose entire identity IS a tagged queue's URL - tagged, plus derived-from-tagged, no third bucket - and stage 3's empty plan with no state file anywhere is what proves the derivation holds. EXPECTED IMPORT-TIME DRIFT, recorded so the next reader does not chase it: live-import reports the two FIFO queues as DRIFTED rather than VERIFIED, on redrive_policy and redrive_allow_policy respectively (cold state has "", live has the JSON). That is the module's own design - the queue resource does not manage those attributes, the separate redrive resources do - so the live object carries a value the queue's state row never recorded. DRIFTED is still eligible for stamping, the convergence apply reconciles it, and the next plan is empty. The tofu-slot convergence apply corpus-iam-policy documented recurs here exactly as that entry predicts: all four aws_sqs_queue resources declare count = var.create ? 1 : 0, so one ordinary `choudoufu apply` ("0 added, 4 changed, 0 destroyed") is folded into stage 2 before stage 3 is attempted. The two redrive resources are untaggable and carry no slot, which is why it is 4 changed and not 6. Stages 3-5 measured: test_plan proposes no resource action, reports "Foreign resources: none among the 1 type swept", both re-read identities unchanged, no state file written; test_apply "0 added, 0 changed, 0 destroyed" with the tofu-estate-tagged object count 4 before and 4 after; drift_reconverge tampers ex-complete-default's Example tag directly through the AWS CLI, live-plan proposes updating exactly module.default_sqs.aws_sqs_queue.this[0] and nothing else, and the apply reconverges it ("0 added, 1 changed, 0 destroyed", tag back to "ex-complete"). THREE SELF-AUTHORED DEFECTS FIXED IN THE SCRIPT WHILE VERIFYING IT, each found by running it rather than reading it. (1) No TF_PLUGIN_CACHE_DIR, which corpus-lambda-simple and corpus-alb-complete both set. Without it the first real run spent 21 minutes in `terraform init` having pulled 48MB of hashicorp/aws and then died on a transient DNS failure before ever reaching `terraform apply` - the likely reason two earlier sessions reported this crossing as stalled rather than failed. With the shared cache the whole five-stage run is minutes. (2) The BREAK contract was unreachable. BREAK=1 set two corruptions, stage 3's and stage 5's, but the stage-3 one calls fail() and exits, so stage 5's branch was dead code that had never run - and its inverted form would have exited 0 on a corrupted run anyway, proving nothing. BREAK is now three named values corrupting three different assertions, each verified for real to exit 1 at its own assertion and each reaching a later stage than the last: BREAK=schema (swap the two expected redrive URLs - both real queues, both types really do resolve, so only a by-value check catches the wrong pairing) fails in stage 2; BREAK=identity fails in stage 3; BREAK=drift fails in stage 5's exactly-one-object assertion with both objects named. An unrecognized BREAK value is rejected up front. This same dead-stage-5-branch shape exists in corpus-iam-policy's script, which this one was copied from - worth a slot there, not touched here. (3) A new managed-shape assertion added in this pass (terraform state list compared by name against the six documented addresses, so a moved corpus pin fails loudly instead of silently crossing a different estate - the exact failure mode that produced the wrong count) first failed on locale collation alone: "." and "_" sort differently under a UTF-8 locale, so a hand-ordered list never matches a locale-sorted one. Both sides now go through LC_ALL=C sort. Caught because the assertion was run, not reviewed. diff --git a/site/content/docs/progress/corpus-sumaform-aws.md b/site/content/docs/progress/corpus-sumaform-aws.md index 29df67d413..93de755cc2 100644 --- a/site/content/docs/progress/corpus-sumaform-aws.md +++ b/site/content/docs/progress/corpus-sumaform-aws.md @@ -12,26 +12,26 @@ Set: core. Lane: opentofu-native. Why it is in the core set: a real project built for OpenTofu specifically, so OpenTofu-only surface is exercised -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m28s | 11 managed resource instances, genuinely cold, genuinely unmarked | -| Migrate | pass | 3m19s | 7 stamped, 2 recorded (markers = record honoured at migrate time, GitHub issue #365 slice 2), 0 failed, 2 skipped | -| Replan from nothing | pass | 11s | Items 4, 5 and 6 (this script's header) are all FIXED and the plan is genuinely empty ("No changes. Your infrastructure matches the configuration."): live-import honours markers = record (located records for aws_instance.instance[0] and aws_ebs_volume.data_disk[0], confirmed at the store and by value against the AWS CLI both right after migrate and again after this empty replan), residue now covers NestingList/NestingSet/NestingMap blocks (internal/live/projection's residueEligibleBlock, widened from the block's SHAPE - whether carriesNoInformation can tell its absence from a real empty answer - never from a type name), and lex00/floci#103 (published in ghcr.io/lex00/floci@sha256:e16d9007a03093b6a6edd22273dee9d8253131f18581b0fa20ae6d34178a3079) now honours RunInstances' BlockDeviceMapping.Ebs.VolumeSize for the root device, closing the one line (root_block_device.volume_size = 8 -> 200) that was this crossing's own last wall. Plan moved 3 to add/0/0 (the original ABSENT gap) -> 2 to add/0/2 to destroy (item 4 fixed, item 5's replacement exposed) -> 0 to add/1 to change/0 to destroy (item 5 fixed) -> empty (item 6 fixed by the emulator). | +| Cold deploy | pass | 1m15s | 11 managed resource instances, genuinely cold, genuinely unmarked | +| Migrate | pass | 3m16s | 7 stamped, 2 recorded (markers = record honoured at migrate time, GitHub issue #365 slice 2), 0 failed, 2 skipped | +| Replan from nothing | pass | 12s | Items 4, 5 and 6 (this script's header) are all FIXED and the plan is genuinely empty ("No changes. Your infrastructure matches the configuration."): live-import honours markers = record (located records for aws_instance.instance[0] and aws_ebs_volume.data_disk[0], confirmed at the store and by value against the AWS CLI both right after migrate and again after this empty replan), residue now covers NestingList/NestingSet/NestingMap blocks (internal/live/projection's residueEligibleBlock, widened from the block's SHAPE - whether carriesNoInformation can tell its absence from a real empty answer - never from a type name), and lex00/floci#103 (published in ghcr.io/lex00/floci@sha256:e16d9007a03093b6a6edd22273dee9d8253131f18581b0fa20ae6d34178a3079) now honours RunInstances' BlockDeviceMapping.Ebs.VolumeSize for the root device, closing the one line (root_block_device.volume_size = 8 -> 200) that was this crossing's own last wall. Plan moved 3 to add/0/0 (the original ABSENT gap) -> 2 to add/0/2 to destroy (item 4 fixed, item 5's replacement exposed) -> 0 to add/1 to change/0 to destroy (item 5 fixed) -> empty (item 6 fixed by the emulator). | | No-op apply | pass | 11s | genuine no-op: 7 tagged objects before, 7 after, no state file either time; module.server's record-based instance and volume identities unchanged | -| Drift and reconverge | pass | 22s | the crossing VPC's Name tag tampered out of band, plan proposed fixing exactly aws_vpc.crossing, apply changed 1 and reconverged the tag to sumaform-crossing-vpc; module.server's record-based identities unaffected | -| Rename | pass | 41s | moved block: aws_eip.crossing_nat renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_route_table.crossing_public renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Drift and reconverge | pass | 21s | the crossing VPC's Name tag tampered out of band, plan proposed fixing exactly aws_vpc.crossing, apply changed 1 and reconverged the tag to sumaform-crossing-vpc; module.server's record-based identities unaffected | +| Rename | pass | 40s | moved block: aws_eip.crossing_nat renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_route_table.crossing_public renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | | Remove a block | pass | 23s | choudoufu: deleting module.server's block proposed exactly three destroys (0 add, 0 change, 3 destroy: the record-based instance and EBS volume, plus the untaggable/derived volume attachment), applied cleanly (0 added, 0 changed, 3 destroyed), the instance and volume are genuinely gone from the live account (instance State=terminated, volume absent, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes the same three destroys | -| Change count | pass | 2m2s | Scaled aws_security_group.count_test, a self-contained 2-instance count block in its own count_test.tf at an address nothing else in this crossing names. SYNTHETIC, and NOT because this estate has no scalable knob: sumaform declares a real one (var.quantity, "number of hosts like this one", driving count = var.quantity on aws_instance.instance), but every resource that knob scales is off the tag rung here - the instance and the EBS volume are markers = record by this crossing's own strict block, aws_volume_attachment is untaggable, and all three carry sumaform's own lifecycle { ignore_changes = [tags] } - so none of them can witness the stage's identity clause off the live object. G4 exercises the real knob for the half it can settle: quantity 1 -> 2 through choudoufu proposes exactly 3 creates, all at index [1] (instance, data disk, volume attachment), index [0] untouched, identical to stock's own plan for the same change on cold_deploy's state (G-ORACLE-QUANTITY), and the estate plans empty again once it is restored; the down direction is NOT exercised on that knob (cold_deploy's state is quantity=1, so there is no 2-instance stock state to scale down from). On the taggable block: scaling 2 -> 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy) and left count_test[0]'s live GroupId sg-f0a2ac3380efc13eb AND its tofu-address marker aws_security_group.count_test:0 unchanged, both read through the AWS CLI rather than choudoufu's report; count_test[1] (sg-043fd4f6862639ce2) verified absent by describe-security-groups length 0 in between; scaling 1 -> 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) back under a NEW GroupId sg-42031c6d2807e6eb0 carrying tofu-address=aws_security_group.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G-ORACLE): the identical 2-instance block stood up FOR REAL in its own idle floci account on FLOCI_PORT+3 - stock never had this block, so cold_deploy's state could not be reused the way day2_remove's and day2_replace's oracles reuse it - shows the identical shape both directions, destroy the higher index only (sg-d3f8475b4536bc66b gone, sg-a7134d986fe9ce70d unchanged) and create the higher index back under a new id (sg-ae09a9da7ca9c4bab). A new GroupId is a sound destroy witness on this pin: probed directly against ghcr.io/lex00/floci@sha256:c55d74e1 with no tofu in the loop, a delete-then-recreate under an identical group name minted sg-8c824b212636a50e9 -> sg-61cec558a8a5a286d. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, so these checks discriminate. | -| Replace with create_before_destroy | pass | 1m12s | choudoufu: changing module.server's image input (ubuntu2204 -> ubuntu2404, both real, both in floci's seeded AMI catalog) proposed exactly one instance replace at the same declared address, cascading into the volume attachment (instance_id is ForceNew there too) - 2 to add, 0 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated via the AWS CLI and the new instance is confirmed running the new image; the local record store's record at the same address now names the new instance's id, not the terminated one (i-5720faebfefe13c10 -> i-c9ddad8b40018244e); the next plan proposes no resource action; BREAK=replace confirms this section's own record check discriminates (a deliberately-wrong expectation against the same real record fails, rather than vacuously passing). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. A manufactured live-object collision (the shape ec2-instance-complete's and corpus-sqs-basic's own BREAK=replace controls report) has no tag surface to be detected from on this markers=record instance and is not exercised here - verified directly that an untagged extra instance is simply invisible to this plan, correctly, not incorrectly. | +| Change count | pass | 2m3s | Scaled aws_security_group.count_test, a self-contained 2-instance count block in its own count_test.tf at an address nothing else in this crossing names. SYNTHETIC, and NOT because this estate has no scalable knob: sumaform declares a real one (var.quantity, "number of hosts like this one", driving count = var.quantity on aws_instance.instance), but every resource that knob scales is off the tag rung here - the instance and the EBS volume are markers = record by this crossing's own strict block, aws_volume_attachment is untaggable, and all three carry sumaform's own lifecycle { ignore_changes = [tags] } - so none of them can witness the stage's identity clause off the live object. G4 exercises the real knob for the half it can settle: quantity 1 -> 2 through choudoufu proposes exactly 3 creates, all at index [1] (instance, data disk, volume attachment), index [0] untouched, identical to stock's own plan for the same change on cold_deploy's state (G-ORACLE-QUANTITY), and the estate plans empty again once it is restored; the down direction is NOT exercised on that knob (cold_deploy's state is quantity=1, so there is no 2-instance stock state to scale down from). On the taggable block: scaling 2 -> 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy) and left count_test[0]'s live GroupId sg-f5399762065bb39ab AND its tofu-address marker aws_security_group.count_test:0 unchanged, both read through the AWS CLI rather than choudoufu's report; count_test[1] (sg-df09988387ba4224c) verified absent by describe-security-groups length 0 in between; scaling 1 -> 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) back under a NEW GroupId sg-27e39476a3e6a8724 carrying tofu-address=aws_security_group.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G-ORACLE): the identical 2-instance block stood up FOR REAL in its own idle floci account on FLOCI_PORT+3 - stock never had this block, so cold_deploy's state could not be reused the way day2_remove's and day2_replace's oracles reuse it - shows the identical shape both directions, destroy the higher index only (sg-8b2649c223f8be264 gone, sg-671372ad2ef1c9820 unchanged) and create the higher index back under a new id (sg-2a483e85eb4452a7d). A new GroupId is a sound destroy witness on this pin: probed directly against ghcr.io/lex00/floci@sha256:c55d74e1 with no tofu in the loop, a delete-then-recreate under an identical group name minted sg-8c824b212636a50e9 -> sg-61cec558a8a5a286d. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, so these checks discriminate. | +| Replace with create_before_destroy | pass | 1m11s | choudoufu: changing module.server's image input (ubuntu2204 -> ubuntu2404, both real, both in floci's seeded AMI catalog) proposed exactly one instance replace at the same declared address, cascading into the volume attachment (instance_id is ForceNew there too) - 2 to add, 0 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated via the AWS CLI and the new instance is confirmed running the new image; the local record store's record at the same address now names the new instance's id, not the terminated one (i-f3d88dad3155358d2 -> i-d8cfb88e7cd43142e); the next plan proposes no resource action; BREAK=replace confirms this section's own record check discriminates (a deliberately-wrong expectation against the same real record fails, rather than vacuously passing). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. A manufactured live-object collision (the shape ec2-instance-complete's and corpus-sqs-basic's own BREAK=replace controls report) has no tag surface to be detected from on this markers=record instance and is not exercised here - verified directly that an untagged extra instance is simply invisible to this plan, correctly, not incorrectly. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 2m22s | 11 resources from nothing (7 tag-stamped, 2 recorded via markers = record, 2 untaggable/derived - route_table_association and volume_attachment), replan empty, stock oracle in its own namespace matches on vpc cidr, security-group rule counts and the instance's ami+type | +| Plan, review, apply | pass | 54s | one argument edited (aws_internet_gateway.crossing's tags gain Reviewed=yes - one of this crossing's seven tag-stamped objects, chosen over module.server's markers = record instance and volume, which carry sumaform's own lifecycle { ignore_changes = [tags] } and so could not witness a tag edit at all), "plan -out=approved.tfplan" wrote a 48587-byte stock-format plan file whose whole change set is one update on aws_internet_gateway.crossing (igw-aa457a90); the world then moved out of band (vpc-1a753819's Name tag through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live object from the one under review) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both aws_vpc.crossing and the live vpc-1a753819 it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - describe-tags on igw-aa457a90 still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Name tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and igw-aa457a90 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the gateway confirmed to be the same id it started as, module.server's record-based instance identity confirmed untouched, and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 2m25s | 11 resources from nothing (7 tag-stamped, 2 recorded via markers = record, 2 untaggable/derived - route_table_association and volume_attachment), replan empty, stock oracle in its own namespace matches on vpc cidr, security-group rule counts and the instance's ami+type | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 9m49.4s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 10m26.9s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed d583dc93b7 (2026-08-18) - the FIRST OpenTofu-native estate crossed (uyuni-project's own maintainers describe it as "OpenTofu configuration," not "Terraform configuration"), versus every prior estate tonight being Terraform-authored/OpenTofu-compatible via terraform-aws-modules. Deliberately reduced slice: the full main.tf.aws.example composes four AWS host roles from one leaf module, backend_modules/aws/host, but three of the four (bastion, module.mirror, module.minion) have no root-facing toggle to disable real SSH/Salt provisioning - the "real boot behavior, out of scope for an emulator" case. Only module.server exposes provision=false; module.base's own network submodule was also unusable (create_network=true needs CreateDhcpOptions/ReplaceRouteTableAssociation, neither implemented in floci), so this estate's own plain VPC/subnet/NAT resources stand in for it. A real floci gap found and fixed on the way: sumaform's ami.tf evaluates ~23 data "aws_ami" blocks unconditionally (one per supported guest OS) regardless of which single image an estate actually launches, and floci's catalog had zero SUSE/Marketplace/Rocky/RHEL entries - seeded 20, reconciled into the combined image alongside tonight's other three floci fixes. cold_deploy and migrate genuinely pass (11 resources, 9 of 11 stamped - 2 correctly untaggable). test_plan blocked by two real, structural rules baked into backend_modules/aws/host itself (the one leaf module every AWS host role shares, so this isn't an artifact of the reduced slice), and on reading both rules' own reasoning neither looks like a choudoufu defect - both are correct, deliberate refusals, not filed: (1) an unconditional, provisioner-less connection block that checkProvisioners flags on its own terms by documented design, dead code in sumaform's own module; (2) lifecycle { ignore_changes = [tags] } on the WHOLE tags argument of aws_instance.instance and aws_ebs_volume.data_disk - sumaform's own comment explains why (SUSE's internal AWS accounts add tags on apply that need preserving), but ignoring the whole argument also silently discards the update that would write tofu-address/tofu-estate, the exact marker-safety failure #306 was about tonight. The fix sumaform's own error text names - ignore_changes = [tags["Owner"]], not the whole argument - is an edit to sumaform's module, out of scope here. Follow-up pass 2026-08-18 (#313 cross-check, fresh worktree, real run): confirmed neither of the two RULE-classified refusals above is #313's wall - zero occurrences of its diagnostic in the raw plan output. Both remain exactly as already documented: permanent, deliberate refusals (checkProvisioners on a dead-code connection block; ignore_changes on the whole tags argument), not filed as new issues, no action needed. diff --git a/site/content/docs/progress/corpus-vpc-complete.md b/site/content/docs/progress/corpus-vpc-complete.md index 1f690c41a5..66aa541581 100644 --- a/site/content/docs/progress/corpus-vpc-complete.md +++ b/site/content/docs/progress/corpus-vpc-complete.md @@ -12,26 +12,26 @@ Set: core. Lane: terraform-popular. Why it is in the core set: a most-downloaded terraform-aws-modules example, pinned by tag; the shape most people deploy -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 34s | Apply complete! Resources: 62 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=vpc-complete-crossing before migration | -| Migrate | pass | 1m34s | 40 stamped, 22 skipped, 0 recorded, 0 failed; 39 objects carry tofu-estate=vpc-complete-crossing; the VPC's tofu-slot reads 0 off EC2, written by the migration itself (choudoufu #372) | -| Replan from nothing | pass | 4s | empty plan; identity re-check unchanged: module.vpc.aws_vpc.this:0, aws_security_group.rds, module.vpc_endpoints.aws_vpc_endpoint.this:s3 | +| Cold deploy | pass | 24s | Apply complete! Resources: 62 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=vpc-complete-crossing before migration | +| Migrate | pass | 1m36s | 40 stamped, 22 skipped, 0 recorded, 0 failed; 39 objects carry tofu-estate=vpc-complete-crossing; the VPC's tofu-slot reads 0 off EC2, written by the migration itself (choudoufu #372) | +| Replan from nothing | pass | 3s | empty plan; identity re-check unchanged: module.vpc.aws_vpc.this:0, aws_security_group.rds, module.vpc_endpoints.aws_vpc_endpoint.this:s3 | | No-op apply | pass | 4s | genuine no-op: 39 objects before, 39 after, no state file either time | | Drift and reconverge | pass | 7s | one subnet tampered (Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag to ex-complete | | Rename | pass | 13s | moved block: module.vpc_endpoints renamed with zero churn (0 add, 7 change, 0 destroy), marker rewritten in place across its taggable objects; live-mv: aws_security_group.rds renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 21s | choudoufu: deleting the dynamodb endpoint's map entry (module.vpc_endpoints_renamed.aws_vpc_endpoint.this["dynamodb"]) proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the endpoint is genuinely gone from the live account (State=absent, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same object | -| Change count | pass | 1m32s | choudoufu: scaling aws_customer_gateway.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and cgw-ae204e601ae5e91da is confirmed gone from the live account while count_test[0] (cgw-f6fed178e29c19fac) kept its server-minted id, its tofu-address marker and its tofu-slot; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a genuinely NEW object (cgw-84118c223099f26a4, not the destroyed cgw-ae204e601ae5e91da), with count_test[0] untouched throughout; the next plan is empty. Every identity above was read off EC2 through the AWS CLI, by the tofu-address marker, never from choudoufu's own report. Stock oracle (G-ORACLE): the identical count block stood up by terraform in its own floci account and scaled 2->1->2 for real shows the identical shape - destroy count_test[1] only, recreate it under a new id, count_test[0]'s id unchanged both times. THE BLOCK IS SYNTHETIC, deliberately: every real count knob in this estate is a subnet list that drives an untaggable aws_route_table_association count block in lockstep (a question about orphaned derived children, not about count-slot binding), and the one real zero-cascade knob, module.vpc's customer_gateways, is a for_each map with no index slots and is already day2_replace's target - so the block uses aws_customer_gateway, a type this estate really does exercise three of. Building it found and fixed a real defect (five-row row 2): the scale-down proposed NO destroy at all, because the removal sweep read live/mapping.json's via="former2" provenance - a Cloud Control ENUMERABILITY filter - to answer the identity question "which TF type is this ARN", and classified aws_customer_gateway TYPE_NOT_LISTABLE against live/registry.json's own handlers.list=true. | -| Replace with create_before_destroy | pass | 20s | choudoufu: changing customer_gateways["IP1"]'s ForceNew ip_address argument proposed exactly one isolated replace at the same declared for_each key (1 to add, 1 to destroy, nothing else), matching F-ORACLE's own plan shape; applied cleanly; the old gateway (cgw-c1ef80761a5322b97) is confirmed gone/deleted and the new gateway (cgw-bf793a3d862648297) carries the marker, both via the AWS CLI; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here). | +| Remove a block | pass | 20s | choudoufu: deleting the dynamodb endpoint's map entry (module.vpc_endpoints_renamed.aws_vpc_endpoint.this["dynamodb"]) proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the endpoint is genuinely gone from the live account (State=absent, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same object | +| Change count | pass | 1m33s | choudoufu: scaling aws_customer_gateway.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and cgw-3768750fdbdaf84d8 is confirmed gone from the live account while count_test[0] (cgw-fe83feba57bb37ac0) kept its server-minted id, its tofu-address marker and its tofu-slot; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a genuinely NEW object (cgw-54a679d9fd7f89f8a, not the destroyed cgw-3768750fdbdaf84d8), with count_test[0] untouched throughout; the next plan is empty. Every identity above was read off EC2 through the AWS CLI, by the tofu-address marker, never from choudoufu's own report. Stock oracle (G-ORACLE): the identical count block stood up by terraform in its own floci account and scaled 2->1->2 for real shows the identical shape - destroy count_test[1] only, recreate it under a new id, count_test[0]'s id unchanged both times. THE BLOCK IS SYNTHETIC, deliberately: every real count knob in this estate is a subnet list that drives an untaggable aws_route_table_association count block in lockstep (a question about orphaned derived children, not about count-slot binding), and the one real zero-cascade knob, module.vpc's customer_gateways, is a for_each map with no index slots and is already day2_replace's target - so the block uses aws_customer_gateway, a type this estate really does exercise three of. Building it found and fixed a real defect (five-row row 2): the scale-down proposed NO destroy at all, because the removal sweep read live/mapping.json's via="former2" provenance - a Cloud Control ENUMERABILITY filter - to answer the identity question "which TF type is this ARN", and classified aws_customer_gateway TYPE_NOT_LISTABLE against live/registry.json's own handlers.list=true. | +| Replace with create_before_destroy | pass | 20s | choudoufu: changing customer_gateways["IP1"]'s ForceNew ip_address argument proposed exactly one isolated replace at the same declared for_each key (1 to add, 1 to destroy, nothing else), matching F-ORACLE's own plan shape; applied cleanly; the old gateway (cgw-475f57e0a2f706305) is confirmed gone/deleted and the new gateway (cgw-b7f5d49f8756a85cd) carries the marker, both via the AWS CLI; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here). | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 55s | 62 resources from nothing (40 tag-stamped, 22 untaggable/derived), replan empty, stock oracle in its own namespace matches on vpc cidr, subnet count (18) and the s3 endpoint's presence | +| Plan, review, apply | pass | 18s | one argument edited (aws_security_group.rds's tags gain Reviewed=yes, a tags-only update that is not ForceNew and leaves sg-0373a87e083bfa5dd's id alone for PART D's later rename), "plan -out=approved.tfplan" wrote a 60349-byte stock-format plan file whose whole change set is one update on aws_security_group.rds; the world then moved out of band (subnet-22deaf7e's Example tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.vpc.aws_subnet.private[0] and the live subnet-22deaf7e it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - sg-0373a87e083bfa5dd still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-0373a87e083bfa5dd read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 56s | 62 resources from nothing (40 tag-stamped, 22 untaggable/derived), replan empty, stock oracle in its own namespace matches on vpc cidr, subnet count (18) and the s3 endpoint's presence | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 4m49.3s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 4m58.1s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. RE-CROSSED FOR REAL 2026-08-21 in worktree live/dhcp-options-355 off local main 41f8c8dd6a, real Docker/floci/terraform/AWS CLI throughout, floci ghcr.io/lex00/floci@sha256:cdd50ec0. STAGES UNCHANGED at 2 of 5, and the honest headline is that #355's wall IS gone and three more stand behind it, none of them a choudoufu defect. Stage 1: 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', 0 objects carrying tofu-estate beforehand. Stage 2: dry run '40 of 62 eligible, 22 untaggable, 0 unadmitted'; -approve '40 stamped, 22 skipped, 0 recorded, 0 failed'; three identities asserted by value through the AWS CLI, all three passed. STAGE 3 NO LONGER REFUSES IN DISCOVERY: with #355 fixed, live-plan exits 0 and renders a full plan for the first time in this estate's history. Proven by A/B against the SAME live migrated estate, same floci container, two binaries: main (41f8c8dd6a) exits 1 with 'Error: Listed resource with no tags - Cloud Control listed a aws_vpc_dhcp_options (AWS::EC2::DHCPOptions) with no Tags in its Properties ... Live resources: dopt-default'; the fixed binary exits 0 with that diagnostic absent and every other diagnostic identical. WHAT STANDS BEHIND IT, all measured in that run and none of them choudoufu's: (1) the first live-plan after live-import proposes adding tofu-slot to 31 objects. That is documented, deliberate product behavior (see live/e2e/corpus-iam-policy/run.sh's THE TOFU-SLOT FINDING; live-import cannot compute a slot from one state file), and three other crossing scripts fold a convergence apply into stage 2. This script does not, because it had never reached stage 3 to notice. (2) that convergence apply fails on a floci gap: 'UnsupportedOperation: Operation ModifyVpcEndpoint is not supported', 4 errors, one per interface VPC endpoint. 26 of the 31 tofu-slot writes do land. (3) the replan after that shows 'Plan: 3 to add, 5 to change, 3 to destroy' - three FORCED REPLACEMENTS from floci read fidelity, not drift: aws_nat_gateway.this[0] (floci's DescribeNatGateways returns neither allocation_id nor subnet_id, so both read as absent and force replacement), aws_vpn_gateway.this[0] (floci returns availability_zone='eu-west-1a' on a gateway whose config sets none, so the plan reads '- availability_zone -> null # forces replacement'), and aws_vpc_endpoint.this[s3], plus four endpoint in-place diffs (policy, route_table_ids, subnet_ids, cidr_blocks all read back empty). All floci work items, not choudoufu ones, and all four filed together as lex00/floci#97 (which also carries the Redshift Tagging-API gap below). THE #355 LOOSE END IS SETTLED, with evidence: the '39 objects carry tofu-estate' line against the '40 stamped' line is a floci Tagging-API coverage gap, not a choudoufu miscount. The 40th object is aws_redshift_subnet_group.redshift[0]; 'aws redshift describe-cluster-subnet-groups' shows it carrying tofu-address=module.vpc.aws_redshift_subnet_group.redshift:0 and tofu-estate=vpc-complete-crossing, while 'resourcegroupstaggingapi get-resources' for the same estate returns 39 ARNs with no Redshift among them. choudoufu's own stamp count is the correct one. ONE PRIOR FIGURE CORRECTED: the note below records 'nine non-fatal Incomplete sweep for undeclared resources warnings'. The real number is 989, identical on both binaries - it is the tag sweep's ARN-join-table coverage list (internal/live/discovery/tagging.go), produced before the type scans and untouched by this fix. Nine was a sample, not a count. test_apply and drift_reconverge stay not_run: stage 3 still ends in FAIL, so running them would prove nothing. PRIOR HISTORY BELOW. RE-CROSSED FOR REAL 2026-08-21 at dbc5ebf575 (GitHub issue #346's fix), worktree live/live-read-346, real Docker/floci/terraform/AWS CLI throughout. STAGES UNCHANGED at 2 of 5 - and that is the honest headline, because #346's own diagnostic IS gone and a different wall stands behind it. Stage 1: 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', 0 objects carrying tofu-estate beforehand. Stage 2: dry run '40 of 62 eligible, 22 untaggable, 0 unadmitted'; -approve '40 stamped, 22 skipped, 0 recorded, 0 failed'; three identities asserted by value straight through the AWS CLI and all three passed (vpc-68ae42e6 = module.vpc.aws_vpc.this:0, sg-1bac9cc2a99da4e8c = aws_security_group.rds, vpce-f81456900c35660a0 = module.vpc_endpoints.aws_vpc_endpoint.this:s3). Stage 3 no longer refuses in identity resolution at all: the #346 diagnostic (cidr_blocks = lookup(each.value, 'cidr_blocks', null) reaching module.vpc.vpc_cidr_block) appears nowhere in the run, and the run gets past live_plan.go's step 4 (identity, fatal on error) into step 5 (discovery), which it could not have done otherwise. It now fails on exactly ONE blocking diagnostic, a NEW one that was hidden behind #346 one stage earlier: 'Listed resource with no tags - Cloud Control listed a aws_vpc_dhcp_options (AWS::EC2::DHCPOptions) with no Tags in its Properties, and refining it with GetResource found none either, so its ownership markers cannot be read. Live resources: dopt-default.' That is the ACCOUNT'S DEFAULT DHCP options set, which this estate did not create and does not declare. Filed separately. Nine non-fatal 'Incomplete sweep for undeclared resources' warnings also print (aws_xray_* x5, kubernetes_* x4), none of them a type this estate declares. One unexplained figure worth someone's hour: the run's own closing line reads '39 objects carry tofu-estate after migration' against the '40 stamped' line above it - the script does not assert it, so nothing failed, but the two disagree. test_apply and drift_reconverge stay not_run: running them against a refused plan would prove nothing. PRIOR HISTORY BELOW. Re-verified 2026-08-18 against ghcr.io/lex00/floci@sha256:f5b46236c6b6fff376ae2db8a2b3a51bf1d13b19a92b6a37af4827ccdf1ef180, published via floci's own CI/GHCR-publish workflows (pushed to origin/main, not a local multi-arch build - see HANDOFF.md's Traps). lex00/floci#66 (Redshift), #67 (DHCP options) and #69 (customer/VPN gateway) are confirmed fixed: cold deploy no longer errors on any of the three. It now fails one step later, on a fourth, narrower and previously-undetected gap in #68's own CreateCacheSubnetGroup/ModifyCacheSubnetGroup implementation - it reads the SubnetIds member list under the generic SubnetIds.member.N key, but ElastiCache's service model overrides SubnetIdentifierList's member locationName to SubnetIdentifier, so every real client sends SubnetIds.SubnetIdentifier.N and the call 400s with MissingParameter. Filed as lex00/floci#70. A fix for exactly this was already sitting uncommitted in the shared floci checkout (another session's in-progress work, left untouched - not this orchestrator's to land). Stages 2-5 remain implemented in the script but unexercised since stage 1 still fails fast by design. Follow-up pass 2026-08-20, isolated worktree off local main (ea9fd62fc0), real Docker/floci/AWS CLI throughout, read from the script's own PASS/FAIL lines: cold_deploy and migrate now PASS - this estate moves 0 of 5 to 2 of 5. lex00/floci#70 was already fixed AND already pinned (99f4cbce8f moved live/floci-image to sha256:5873331d, 83c1aa73's published build) - the brief that sent this pass in believed the pin predated it and was wrong; re-running against that existing pin confirmed CreateCacheSubnetGroup succeeds and surfaced the NEXT gap one call later. That gap was lex00/floci#71 (ElastiCache served no tagging actions at all: ListTagsForResource/AddTagsToResource/RemoveTagsFromResource absent from ElastiCacheQueryHandler's switch, and CacheSubnetGroup carried no tags field), and the AWS provider calls ListTagsForResource on EVERY read of aws_elasticache_subnet_group, so the resource was unusable even with no tags in the configuration - the sole remaining cold-deploy error, 1 of 1. Fixed in floci and merged to lex00/floci main (dc140fb0; CI and GHCR-publish both green), mirroring RDS's identical trio on the identical Query protocol; tags live on the CacheSubnetGroup model beside its existing arn field so TaggedResourceScanner picks them up for resourcegroupstaggingapi GetResources with no extra wiring (asserted by a new test, not assumed), and resolveTagHandle reads the resource type off the ARN so another taggable ElastiCache resource is one branch plus a tags field on its model. live/floci-image re-pinned here to sha256:dc246b1e, with live/floci-capabilities.json regenerated for that digest (services with -watch networkmanager,storagegateway, cloudcontrol, cloudcontrol-scoped, tagging) plus the three hand-probed rows re-verified live against the new image - redshift implemented, qldb still unimplemented, opensearch still partial - giving 86 services / 749 types, matching the prior digest block's shape exactly with no existing digest block touched. Real numbers from the run: STAGE 1 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', 0 objects tagged before migration; STAGE 2 dry run '40 of 62 resource instance(s) are eligible for stamping' with UNTAGGABLE (22) and no UNADMITTED_TYPE section at all, -approve '40 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 22 skipped', and three identities read straight through the AWS CLI: module.vpc.aws_vpc.this:0 on the VPC, aws_security_group.rds on the RDS SG, module.vpc_endpoints.aws_vpc_endpoint.this:s3 on the S3 endpoint; 39 objects carry tofu-estate after migration. All 22 skips are genuinely untaggable types (aws_route, aws_route_table_association, aws_vpc_dhcp_options_association, aws_security_group_rule - none has a tags argument in the provider's schema), i.e. the invariant working, not a gap. TWO SELF-INFLICTED SCRIPT BUGS FOUND AND FIXED, neither of which had ever run because stage 1 had never passed: the stage-2 count assertion demanded '0 skipped' (impossible for this estate; it now asserts 40/22 by value plus UNADMITTED_TYPE by absence), and all three identity assertions compared the tag value against OpenTofu's BRACKET spelling ('module.vpc.aws_vpc.this[0]') when a tag value can never carry '[' - the escaped form is 'module.vpc.aws_vpc.this:0' per internal/live/markers.EscapeKey. That is the same vacuous-comparison bug corpus-iam-policy and corpus-iam-read-only-policy each shipped once; both forms are now separate variables, and the bracket forms are used where stage 5 reads a plan diff header. STAGE 3 fails on EXACTLY ONE diagnostic, filed as #346, whose headline finding refutes the obvious fix: the diagnostic points at lookup(each.value, 'cidr_blocks', null) on the vpc-endpoints module's line 116, but a resolveLookupCall beside coalesce.go would NOT unblock this estate. Checked with three hand-built live-check variants rather than assumed: a static-valued lookup() already resolves, and the same map written as a direct each.value.cidr_blocks refuses identically. What refuses is the VALUE - the example passes [module.vpc.vpc_cidr_block], i.e. aws_vpc.this[0].cidr_block, a non-identity attribute of another managed resource, into aws_security_group_rule's identity-bearing cidr_blocks. Written inline the resolver reaches the reference and says so ('Not an identity attribute'); through a local or a for_each map it falls to the generic refusal at resolve.go:2025. Same wall, two spellings - so widening the decomposition switch changes the message and leaves the estate blocked. Whether identity resolution may fold a managed resource's own CONFIGURED attribute (this cidr_block is local.vpc_cidr, a static string) is a maintainer design call, so it was filed rather than forced. test_apply and drift_reconverge remain not_run: attempting them against a still-refused plan would prove nothing. diff --git a/site/content/docs/progress/corpus-xancloud-iac.md b/site/content/docs/progress/corpus-xancloud-iac.md index 9bb7624172..a9f4621f5c 100644 --- a/site/content/docs/progress/corpus-xancloud-iac.md +++ b/site/content/docs/progress/corpus-xancloud-iac.md @@ -12,26 +12,26 @@ Set: core. Lane: opentofu-native. Why it is in the core set: a real project built for OpenTofu specifically, so OpenTofu-only surface is exercised -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 42s | 28 resources, genuinely cold, genuinely unmarked | -| Migrate | pass | 49s | live-import -approve completed cleanly against the cold state | +| Cold deploy | pass | 32s | 28 resources, genuinely cold, genuinely unmarked | +| Migrate | pass | 48s | live-import -approve completed cleanly against the cold state | | Replan from nothing | pass | 3s | no resource change proposed | | No-op apply | pass | 3s | genuine no-op (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 21 | -| Drift and reconverge | pass | 8s | one object tampered (Name tag), exactly module.vpc.aws_vpc.this["main"] proposed by both choudoufu and stock with the identical change, apply changed 1 and the Name tag reads back as configured | -| Rename | pass | 10s | moved block: aws_iam_role.flow_logs renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_eip.nat renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Drift and reconverge | pass | 7s | one object tampered (Name tag), exactly module.vpc.aws_vpc.this["main"] proposed by both choudoufu and stock with the identical change, apply changed 1 and the Name tag reads back as configured | +| Rename | pass | 11s | moved block: aws_iam_role.flow_logs renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_eip.nat renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | | Remove a block | pass | 18s | choudoufu: deleting aws_vpc_endpoint.s3's block (the S3 gateway endpoint, a standalone leaf nothing else in the module references) proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the endpoint is genuinely gone from the live account (describe-vpc-endpoints on the old id reports State=deleted or nothing at all, read via the AWS CLI, not choudoufu's own report), and the next plan is empty; the E-ORACLE stock oracle (on cold_deploy's own state, before any tag was ever written) also proposes exactly one destroy for the same object; classifyOrphans did not withhold the destroy because no other aws_vpc_endpoint.s3 block is declared anywhere in this config | | Change count | pass | 29s | choudoufu: dropping "logs" from var.vpcs.main.vpc_endpoints destroyed exactly module.vpc.aws_vpc_endpoint.interface["main-logs"] (0 add, 0 change, 1 destroy), leaving every sibling for_each member (main-ssm's live id and tofu-address marker checked directly) untouched; adding it back created exactly the same key under a NEW live id (0 add -> 1 add, 0 change, 0 destroy) while main-ssm stayed untouched throughout; the next plan is empty; the F-ORACLE stock oracle on the identical for_each set, applied on cold_deploy's own state, shows the identical shape: destroy the dropped key only, create it back under a new id, every sibling key's id unchanged both times | | Replace with create_before_destroy | pass | 7s | choudoufu: changing module.vpc.aws_iam_role.flow_logs_renamed's ForceNew name argument proposed a forced replace at the same declared address (Plan: 3 to add, 0 to change, 3 to destroy., cascading into the flow log's iam_role_arn and the inline role policy, both keyed to the role at creation time with no update path), applied cleanly; the old role is confirmed gone via the AWS CLI (NoSuchEntity) and the new role (xancloud-dev-main-flow-logs-role-v2) carries the marker; the local record store's record at the same address now names the new object's name, not the destroyed one (xancloud-dev-main-flow-logs-role -> xancloud-dev-main-flow-logs-role-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (REPLACE-ORACLE) also proposes replacing the role at the same address (Plan: 3 to add, 0 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ESTATE); BREAK=replace confirms a manufactured marker collision is reported loudly ("Two live resources claiming one slot") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 38s | 28 resources from nothing (matching stage 1's stock cold-deploy count exactly), all markers verified via the AWS CLI, 28 records in the local record store (#364 A2), replan empty, object-by-object comparison against stock's still-pristine cold deploy on $ENDPOINT matches on tagged-object count (21), VPC CIDR, subnet/NAT-gateway/VPC-endpoint counts and account alias | +| Plan, review, apply | pass | 13s | one argument edited (module.vpc.aws_default_security_group.this["main"]'s tags gain Reviewed=yes - a single for_each instance nothing else in the module references), "plan -out=approved.tfplan" wrote a 38420-byte stock-format plan file whose whole change set is one update on module.vpc.aws_default_security_group.this["main"]; the world then moved out of band (vpc-e1d1a6dd's Name tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both module.vpc.aws_vpc.this["main"] and the live vpc-e1d1a6dd it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - sg-e5666543a0a301fd7 still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-e5666543a0a301fd7 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 39s | 28 resources from nothing (matching stage 1's stock cold-deploy count exactly), all markers verified via the AWS CLI, 28 records in the local record store (#364 A2), replan empty, object-by-object comparison against stock's still-pristine cold deploy on $ENDPOINT matches on tagged-object count (21), VPC CIDR, subnet/NAT-gateway/VPC-endpoint counts and account alias | | Strict profile (not a headline stage) | not run | | | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m26.9s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 3m29.8s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Landed 2026-08-19, the fifth OpenTofu-native estate. Genuinely different from every prior crossing in this lane. OpenTofu-native evidence: README/badges state "OpenTofu-first ... native state encryption, S3 locking without DynamoDB, no vendor lock-in", required_version >= 1.11.0, and docs/TROUBLESHOOTING.md carries a dedicated real-usage "OpenTofu general" section (tofu init -upgrade) rather than just a badge - weaker than corpus-hongbomiao's genuine .tofu files (this ships plain .tf), comparable to sumaform/overture-tiles. Several other candidates evaluated and rejected first (an account-singleton-heavy 0-star repo, a devtool's test fixture masquerading as an estate, multiple 0-1-star scaffolds). cold_deploy genuinely passes (28 resources, plain tofu apply, unmodified). migrate genuinely passes: live-import 22 of 28 eligible (21 VERIFIED, 1 DRIFTED), -approve 22 newly stamped, 6 correctly UNTAGGABLE, 0 failed. test_plan BLOCKED, deterministically asserted (Plan: 2 to add, 3 to change, 1 to destroy, all 5 addresses traced to three independent, filed, evidence-backed gaps, not routed around); the VPC's own marker re-verified directly against the AWS CLI after state deletion, BREAK=1 confirmed load-bearing. Six real floci gaps filed with full evidence: lex00/floci#73 (S3Control PutPublicAccessBlock unimplemented), #74 (IAM UpdateAccountPasswordPolicy unsupported), #75 (IAM Access Analyzer - whole missing service), #76 (EC2 ModifyInstanceMetadataDefaults unsupported), #77 (CloudTrail tagging trio missing, blocks any tagged aws_cloudtrail), #78 (EC2 CreateFlowLogs ignores TagSpecifications) - none fixed, per this session's standing instruction. Two choudoufu gaps filed: INTENTIUS/choudoufu#327 (aws_nat_gateway's ForceNew args read null from a stateless prior, forcing a spurious replace), #328 (aws_default_security_group.revoke_rules_on_delete, write-only with no live representation, always diffs) - not fixed. A REAL BUG found and fixed along the way, outside this task's own estate work: #325's marker-alias fix (merged just before this crossing started) double-counted a live object when an estate declares BOTH sides of a default-adopter pair (this module does, aws_default_security_group next to an unrelated aws_security_group) - both scanType passes appended an identical claimant, producing a false "Two live resources claiming one address" collision on a single real object (confirmed via direct AWS CLI query). Fixed with claimantAlreadyPresent (dedupe by import ID), covered by a new regression test reproducing the exact real-world error text before the fix and passing after. Stages 4-5 not attempted - both need a genuinely empty first plan. Merged to local main as 62e63a3eb5 (crossing: 427977009a, the discovery double-count fix: 3bdceb43e3); rebased onto a concurrent sibling's internal/live/identity work mid-session, one conflict (the generated identity-golden.txt) resolved by regenerating, not hand-merging; justfile gained recipe demo-corpus-xancloud-iac; live/corpus-manifest.json gained the pin. Re-verified 2026-08-19 after #327's fix (56e807062e/b5bb09d27e, generic - reuses #275's existing residueCandidates/ResidueStore mechanism unchanged, adding one new call site: live-import's Approve now records residue for every eligible instance using the migrated state's own real prior value, not #327-specific): plan moved from 2 to add/3 to change/1 to destroy down to 1 to add/1 to change/0 to destroy - aws_nat_gateway.this["main-0"] no longer proposes a spurious replace. Remaining test_plan blockers are exactly the still-unfixed floci gaps (#73-#78, S3Control/IAM password policy/IAM Access Analyzer/EC2 metadata defaults/CloudTrail tagging/EC2 flow logs), not a choudoufu-side gap. A residual, lower-severity, choudoufu-specific diff on regional_nat_gateway_address (pure-Computed set(object), harmless) remains and was deliberately not filed - residueCandidates correctly excludes pure-Computed attributes, and broadening that needs its own corpus-wide validation before it's safe (risk: masking real out-of-band drift). Follow-up pass 2026-08-20 (this session): #73/#74/#75 fixed and closed (lex00/floci commits 51c00478/08d92407/be3f7ffd - S3 account-level PutPublicAccessBlock, IAM UpdateAccountPasswordPolicy, AccessAnalyzer as a whole new service), joining #76/#77/#78 which other concurrent sessions closed the same night (all six of the floci gaps this estate originally filed are now closed). Re-pinned floci-image to be3f7ffd (sha256:8a882bcc, CI+GHCR-publish both green) and regenerated floci-capabilities.json (verified additive: one digest block added, none removed, no existing block changed). Re-crossed for real against the new pin, the script AS COMMITTED (still setting TF_VAR_cloudtrail_enabled=false and the three iam_baseline_enable_* toggles off, so the five originally-toggled-off features are not re-exercised by this run - re-enabling them is separate, larger follow-up work, not attempted here). Real result: stage 1 PASS (28 resources, unchanged); stage 2 PASS (22 stamped: 21 VERIFIED + 1 DRIFTED, 6 UNTAGGABLE, 0 failed - unchanged from before); stage 3 still FAILS the script's own hard-coded assertion (which expects the stale 'Plan: 2 to add, 3 to change, 1 to destroy.' shape from before tonight's floci fixes), but the REAL diagnostic surface changed: only two addresses appear in the plan now, not five - 'module.vpc.aws_nat_gateway.this["main-0"] will be updated in-place' (the pre-existing, previously-noted harmless regional_nat_gateway_address residual, #327 itself still holding - no force-replace) and, newly, 'module.vpc.aws_flow_log.cloudwatch["main"] must be replaced' - a force-replace that could not appear before tonight because the flow log was never migrated at all (floci#78 blocked its tag write, so live-import never had a marker to find). This is very likely the same #327-shaped bug class (a ForceNew argument reading null from a stateless prior) now hitting aws_flow_log instead of aws_nat_gateway, not yet diagnosed or filed as its own issue - flagged here rather than assumed. A second confirmatory run was attempted to capture the exact 'Plan: N to add, M to change, K to destroy' totals line and pin down the flow_log diff further, but was lost to a session interruption before completing; the two-address finding above is from one complete, real, natural-exit run and was not re-confirmed a second time this session. The script's own hard-coded assertion (both the exact-string check and the per-address list) is now stale relative to the pin and needs updating by whoever picks this back up, once the flow_log diff is diagnosed. Stages 4-5 remain not_run (unchanged - both need a genuinely empty first plan, which this has not yet reached). RESOLVED as #347 (2026-08-20, 2cb473affd/27cafb1650): the flow_log diff was diagnosed - iam_role_arn (ForceNew) reads null on the stateless prior because floci's own FlowLogService never learns or emits it at all (Ec2QueryHandler.handleCreateFlowLogs ignores DeliverLogsPermissionArn; handleDescribeFlowLogs never returns it), confirmed via TF_LOG=trace straight through Ratification.Approve's residue classification (#327's residueCandidates/ResidueStore mechanism correctly DECLINES to record it, since both classification reads come back empty on any prior, not just the stateless one - the opposite of #327's shape, where the provider preserves without truly reading). Confirmed parity label 1, "OpenTofu fails here too": the PLAIN cold-deploy state (written by a real, non-choudoufu tofu apply) already carries iam_role_arn="" immediately after apply, so a stock stateful tofu plan would show the identical perpetual diff. Filed as lex00/floci#87 (open, emulator gap, not a choudoufu defect). run.sh's stage-3 assertions were updated to match reality: 2 addresses (aws_flow_log.cloudwatch["main"] must be replaced; aws_nat_gateway.this["main-0"] will be updated in-place, #327's own still-harmless residual) and "Plan: 1 to add, 1 to change, 1 to destroy." RE-VERIFIED 2026-08-21 (this session, worktree live/reverify-limitations, floci pinned at the current e61a987/d65baf42 image): real re-run of run.sh, exit 0, every hard-coded assertion passed byte-for-byte against the #347 shape above - same 2 addresses, same summary line, same 28/22/21+1/6/0 stage 1-2 counts. lex00/floci#87 confirmed still OPEN (not narrowed by the e61a987 re-pin, which addressed #86/#88 only). None of tonight's #331/#337/#340/#343/#344 admission/identity/record-store changes touched this estate. Wall unchanged; stages 4-5 remain not_run. UPDATE 2026-08-21 (lex00/floci#87 FIXED via lex00/floci#96, PR squash-merged as 17c7f7ef, published sha256:cdd50ec04a1a13461035657bdd9ec2ed377ac48925e76495a73c9674b5cbd9f9, verified via the GHCR packages API and `docker buildx imagetools inspect` on both the :17c7f7e and :latest tags): CreateFlowLogs/DescribeFlowLogs now carry DeliverLogsPermissionArn. Re-pinned floci-image to this digest and regenerated floci-capabilities.json (verified additive: one digest block added - 87 services, 747 type rows, matching every prior digest's shape - none removed, no existing block changed). Re-crossed for real against the new pin: stage 1 PASS (28 resources, unchanged); stage 2 PASS (22 stamped: 21 VERIFIED + 1 DRIFTED, 6 UNTAGGABLE, 0 failed - unchanged from before). Stage 3 (test_plan) genuinely changed shape: aws_flow_log.cloudwatch["main"] no longer appears anywhere in the plan - the force-replace lex00/floci#87 caused is gone. The plan is narrower but NOT empty: 'Plan: 0 to add, 1 to change, 0 to destroy.', the sole remaining address module.vpc.aws_nat_gateway.this["main-0"] will be updated in-place - the same pre-existing, harmless #327 regional_nat_gateway_address residual this estate has carried since before #87 was ever diagnosed. run.sh's stage-3 header and assertions rewritten to match this reality (single-address plan, exact 'Plan: 0 to add, 1 to change, 0 to destroy.' string, an explicit assertion that aws_flow_log no longer appears), re-run clean end to end (exit 0, no FAIL), and BREAK=1 re-confirmed load-bearing (the VPC identity check fails exactly as designed against a wrong expected address). test_plan stays 'fail' by this repo's own convention (a first plan must be empty to pass) - #327's own NAT-gateway residual is the estate's only remaining wall, unchanged in kind from before, just no longer sharing the plan with #87. Stages 4-5 remain not_run: both still need a genuinely empty first plan, which this estate has not yet reached. NOT a new estate at full parity - one wall cleared (lex00/floci#87), one wall (#327, already tracked, not this session's to fix) remains. UNIT gauntlet:corpus-xancloud-iac/test_plan (2026-08-21, orchestrator-run worker): stage 3 (test_plan) is now PASS. Root cause traced and fixed generically, not scoped around: aws_nat_gateway.this["main-0"] regional_nat_gateway_address (Computed only - this NAT gateway has connectivity_type=public, not "regional", confirmed directly against floci's own DescribeNatGateways response, which carries no such field for this object at all) was showing "+ regional_nat_gateway_address = (known after apply)" on every stateless first plan. internal/live/projection/residue.go's residueCandidates and fillResidue used to refuse a purely-Computed attribute a residue candidacy at all ("it cannot be set in configuration, so there is nothing to remember") - that reasoning undercounted a real case: the provider's Read does not re-derive this attribute from a bare identity-only prior, it leaves whatever the prior held (null), and OpenTofu marks a null Computed attribute unknown forever. Fixed by dropping the Required/Optional restriction from both functions, keeping every other exclusion (identity, sensitive, write-only, NestedType) - safety comes from classifyResidue's own two-read discriminator, not from that schema-shape filter. internal/live/projection/residue_test.go and internal/live/liveimport/residue_test.go updated to match (arn, Computed-only in the lambda-like fixture, is now a legitimate candidate; a new TestFillResidueFillsAComputedOnlyAttribute pins the positive case). Re-crossed for real: stage 1 PASS (28 resources, unchanged); stage 2 PASS (22 stamped, unchanged); stage 3 now PASS - "No changes. Your infrastructure matches the configuration.", BREAK=1 re-confirmed load-bearing against the VPC identity check. Stages 4-5 remain not_run: not yet written, since stage 3 only just started passing in this run - a follow-up unit's work, not attempted here. diff --git a/site/content/docs/progress/reference-ec2-vpc.md b/site/content/docs/progress/reference-ec2-vpc.md index 58c2bbb0ba..93cc03300b 100644 --- a/site/content/docs/progress/reference-ec2-vpc.md +++ b/site/content/docs/progress/reference-ec2-vpc.md @@ -10,26 +10,26 @@ Set: core. Lane: reference. Why it is in the core set: the plainest hand-written reference shape, kept in this repository -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 1m41s | 5 resources from plain terraform, a real terraform.tfstate, zero markers | -| Migrate | pass | 52s | 5 of 5 verified, 5 stamped, 0 skipped | +| Cold deploy | pass | 1m27s | 5 resources from plain terraform, a real terraform.tfstate, zero markers | +| Migrate | pass | 51s | 5 of 5 verified, 5 stamped, 0 skipped | | Replan from nothing | pass | 2s | post-adoption plan is empty; markers read back through the AWS CLI in part A | -| No-op apply | pass | 3s | no-op apply (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 5 | +| No-op apply | pass | 2s | no-op apply (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 5 | | Drift and reconverge | pass | 4s | one object tampered, exactly aws_instance.main proposed, apply changed 1 and the tag reads back as configured | -| Rename | pass | 9s | moved block: aws_security_group renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_internet_gateway renamed with zero churn, marker rewritten in place; stock oracle over the same two-resource rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | -| Remove a block | pass | 6s | choudoufu: deleting aws_internet_gateway.renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (describe-internet-gateways on the old id no longer returns it, read via the AWS CLI, not choudoufu's own report), and the next plan is empty; stock oracle on cold_deploy's own state (B1.6) also proposes exactly one destroy for the same object; classifyOrphans did not withhold the destroy because no other aws_internet_gateway block is declared anywhere in this config | +| Rename | pass | 8s | moved block: aws_security_group renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_internet_gateway renamed with zero churn, marker rewritten in place; stock oracle over the same two-resource rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI | +| Remove a block | pass | 5s | choudoufu: deleting aws_internet_gateway.renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (describe-internet-gateways on the old id no longer returns it, read via the AWS CLI, not choudoufu's own report), and the next plan is empty; stock oracle on cold_deploy's own state (B1.6) also proposes exactly one destroy for the same object; classifyOrphans did not withhold the destroy because no other aws_internet_gateway block is declared anywhere in this config | | Change count | pass | 17s | choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live id and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under a NEW live id (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the B1.7 stock oracle on the same 2-instance count block, applied fresh in the idle greenfield account, shows the identical shape: destroy the higher index only, create the higher index back under a new id, the lower index's id unchanged both times | -| Replace with create_before_destroy | pass | 27s | choudoufu: changing aws_instance.main's ForceNew ami argument proposed exactly one isolated instance replace at the same declared address (1 to add, 1 to destroy, nothing else), matching B1.8's own plan shape; applied cleanly; the old instance (i-b2dcd2cdc161c3dc3) is confirmed terminated and the new instance (i-c705a31c96a9d1588) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance, not the terminated one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here). | -| Crash between create and destroy (planned) | pass | 34s | choudoufu: a real create_before_destroy replace of aws_instance.main was interrupted with SIGTERM (landed on attempt 1 of 3; deterministic by construction, not by timing luck - internal/command/apply_e2etesting_crash.go self-signals synchronously inside the single -parallelism=1 graph-walker goroutine the instant the create half's own write-back commits in memory, replacing #483's external tail/grep/kill race that produced issue #490's own retry-lottery evidence) strictly between the create committing (new object i-31f6c7e770302d57f, confirmed running via the AWS CLI) and the destroy of the deposed old object (i-c705a31c96a9d1588, confirmed still running and untouched via the AWS CLI) ever dispatching; the local record's one write-back correctly carried both facts at once (current=i-31f6c7e770302d57f, deposed=i-c705a31c96a9d1588). Real investigation before writing this check found a genuine engine gap: issue #415's record-backed collision branch (internal/live/discovery/discovery.go, decl.recordBacked's 2-claimant path) called collisionProblem unconditionally with no deposed-record lookup at all, so a record-backed address's own crash window - exactly what a real crash's write-back leaves, since it answers the address's CURRENT identity in the same commit - could never recover on its own; fixed generically (mirrors the scalar path's own matchDeposedClaimant call, no resource type name in the fix), covered by two new unit tests (internal/live/discovery/deposed_test.go). The next plan proposed exactly one destroy (the deposed object, 0 add, 0 change, 1 destroy) and nothing else, matching stock's own documented deposed-object semantics (Stock records the old object as deposed and destroys it on the next apply); applying it destroyed exactly that object (confirmed terminated via the AWS CLI), cleared the deposed record entry, and left the current identity untouched; the plan after that is empty. BREAK_CRASH=1 confirms the empty-plan assertion this stage's Break text names correctly fails to hold against the same real crash window. | +| Replace with create_before_destroy | pass | 26s | choudoufu: changing aws_instance.main's ForceNew ami argument proposed exactly one isolated instance replace at the same declared address (1 to add, 1 to destroy, nothing else), matching B1.8's own plan shape; applied cleanly; the old instance (i-49b443856f409d5a7) is confirmed terminated and the new instance (i-65a7966e4fd5a1a42) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance, not the terminated one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here). | +| Crash between create and destroy (planned) | FAIL | 31s | the post-recovery plan exited 1 | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | +| Plan, review, apply | pass | 11s | one argument edited (aws_subnet.main's tags gain Reviewed=yes - the one leaf of this five-resource estate no later part renames, removes, replaces or crashes, and a tags-only update that is not ForceNew, so the subnet id aws_instance.main points at is untouched), "plan -out=approved.tfplan" wrote a 9808-byte stock-format plan file whose whole change set is one update on aws_subnet.main; the world then moved out of band (i-49b443856f409d5a7's Name tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both aws_instance.main and the live i-49b443856f409d5a7 it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - subnet-303d6d3d still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and subnet-303d6d3d read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | | Greenfield apply | pass | 3s | 5-object structural comparison (vpc/subnet/igw/sg/instance) between the greenfield estate and stock's cold deploy matches, via the AWS CLI on both endpoints, marker tags never compared; local record store held 5 records, one per instance (#364 A2); replanned empty both with and without the local record store | | Strict profile (not a headline stage) | pass | 2s | every strict toggle on (secrets = refuse, no_source_create = refuse, marker_repair = never with a markers "record" selection naming aws_ebs_volume) against a scratch estate carrying one resource, random_password.db: exactly one refusal, matching live/LIMITATIONS.md's "strict-secrets" text word for word (Logical resource is not admitted / SECRET_REFUSED / strict { secrets = "refuse" }); no_source_create and marker_repair are on and silent, reaching nothing this config declares. BREAK_STRICT=1 turns secrets back to "store" alone: the refusal disappears, the plan becomes an ordinary create, and no other refusal appears. Not part of the headline bars: tools/gauntlet/stages.go keeps Status planned here, because isClear (tools/gauntlet/artifact.go) and NextUnits (tools/gauntlet/next.go) both key strictly off ActiveStages today, with no exemption for a stage the docs already call non-headline - flipping Status without first adding that exemption would silently start gating the two headline bars on this stage, which #363 did not ask for and this unit did not build. | -Last run at commit `eec6fb4282` on 2026-09-06T01:27:32Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 4m20.4s. +Last run at commit `70e2722fa4` on 2026-09-07T03:06:04Z, exit code 1, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 4m7.5s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. Verified end-to-end 2026-08-17/18. Drift-and-reconverge added 2026-08-18: the adopted estate's EC2 instance Name tag is tampered directly via the AWS CLI against the running floci container, choudoufu plan proposes fixing exactly aws_instance.main and nothing else, and apply reconverges it - verified with a real clean run and a real BREAK=1 run (BREAK also tampers a second object's Name tag, and the single-object assertion is confirmed to fail when it does). diff --git a/site/content/docs/progress/terralith-scale.md b/site/content/docs/progress/terralith-scale.md index bd6f595726..3c4d2a7812 100644 --- a/site/content/docs/progress/terralith-scale.md +++ b/site/content/docs/progress/terralith-scale.md @@ -10,26 +10,26 @@ Set: core. Lane: reference. Why it is in the core set: the one estate shaped like the thing the product is FOR - a single-state monolith a stranger would bring to an adoption (#546) - rather than a module example; every other core estate is a small published module, so a headline bar without this one does not read the product's own claim -**Not clear yet.** +**Clear.** Every headline stage passes. | Stage | Verdict | Duration | Detail | |---|---|---|---| -| Cold deploy | pass | 2m47s | stock terraform applied 79 resources at scale=1 from unmodified terralith-gen output into BOTH accounts (COLD keeps its terraform.tfstate for migrate to adopt from and is confirmed carrying no tofu-address tag; GREEN was enumerated at 34 objects, proved non-vacuous by a deliberately-added role, then destroyed back to an enumerated-empty account - issue #564's own proof, unchanged) | -| Migrate | pass | 47s | live-import ratified 38 of 79 instances as eligible and stamped all 38 with 0 failed and 41 skipped (untaggable, identity composed from an already-stamped parent); every one of the 79 addresses in stock's own `terraform state list` - this stage's oracle - is accounted for by name in the report | +| Cold deploy | pass | 2m4s | stock terraform applied 79 resources at scale=1 from unmodified terralith-gen output into BOTH accounts (COLD keeps its terraform.tfstate for migrate to adopt from and is confirmed carrying no tofu-address tag; GREEN was enumerated at 34 objects, proved non-vacuous by a deliberately-added role, then destroyed back to an enumerated-empty account - issue #564's own proof, unchanged) | +| Migrate | pass | 43s | live-import ratified 38 of 79 instances as eligible and stamped all 38 with 0 failed and 41 skipped (untaggable, identity composed from an already-stamped parent); every one of the 79 addresses in stock's own `terraform state list` - this stage's oracle - is accounted for by name in the report | | Replan from nothing | pass | 4s | post-migration plan is empty; six rendered identities asserted BY VALUE against the AWS CLI across four separate tagging surfaces (Route 53, IAM, ECS, EC2), including the count-indexed aws_iam_role.count_team[1] and the module-nested, double-indexed module.team_pod["pod-a"].aws_iam_role.pod_role[0] | -| No-op apply | pass | 6s | no-op apply (0 added, 0 changed, 0 destroyed); the estate is enumerated object by object before and after - 34 objects across IAM/Route53/ECS/EC2, byte-identical listings, never a bare count - and the tofu-estate-tagged count is unchanged at 38 | -| Drift and reconverge | pass | 39s | one live object mutated out of band through the AWS CLI; choudoufu's next plan proposed fixing exactly aws_vpc.main and nothing else (0 add, 1 change, 0 destroy), matching stock's own plan for the identical mutation on cold_deploy's own state (B4, taken before any marker existed); the apply changed exactly 1 resource, the Name tag reads back as configured and the tofu-address marker is unchanged | -| Rename | pass | 22s | moved block: aws_iam_instance_profile.team_0000_profile renamed with zero churn (0 add, 1 change, 0 destroy) and the plan itself showed the tofu-address marker being rewritten in place; live-mv: team_0001_profile renamed with no moved block at all, reported as a real cloud write; both live instance-profile ids unchanged and both markers read back at the NEW address via the AWS CLI; stock's own oracle over the identical two renames on cold_deploy's state (B1) is also zero churn (No changes., both moves reported); the plan after both renames is empty | -| Remove a block | pass | 11s | deleting two blocks - the taggable, marked aws_iam_instance_profile.team_0002_profile and the UNTAGGABLE aws_iam_role_policy.team_0002_inline, whose parent role stays declared - proposed exactly two destroys (0 add, 0 change, 2 destroy) in an order the cloud accepted, matching stock's own plan for the same two removals on cold_deploy's state (B2); the apply destroyed exactly two, both objects are confirmed gone and the parent role confirmed still live via the AWS CLI, and the next plan is empty | -| Change count | pass | 28s | the estate's OWN count block - six declarations across four resource types, two of them untaggable - scaled 2 to 1 and back: exactly six index-[1] destroys then exactly six index-[1] creates, no index-[0] instance touched in either plan, matching stock's own applied cycle over the identical six-block shape in a separate account (G1); across the whole cycle count_team[0]'s live role id was unchanged and its marker still reads aws_iam_role.count_team[0], count_team_profile[0]'s still reads aws_iam_instance_profile.count_team_profile[0], the recreated count_team[1] carries aws_iam_role.count_team[1], and the plan afterwards is empty | -| Replace with create_before_destroy | pass | 19s | changing aws_iam_instance_profile.team_0004_profile's ForceNew name under create_before_destroy proposed exactly one isolated replace at the same declared address (1 to add, 0 to change, 1 to destroy), matching stock's own plan for the identical change on cold_deploy's state (B3); the apply created the new object and destroyed the old one, the old name no longer resolves and the new one carries the declared address's marker (both read via the AWS CLI), and the next plan is empty with no collision | +| No-op apply | pass | 5s | no-op apply (0 added, 0 changed, 0 destroyed); the estate is enumerated object by object before and after - 34 objects across IAM/Route53/ECS/EC2, byte-identical listings, never a bare count - and the tofu-estate-tagged count is unchanged at 38 | +| Drift and reconverge | pass | 33s | one live object mutated out of band through the AWS CLI; choudoufu's next plan proposed fixing exactly aws_vpc.main and nothing else (0 add, 1 change, 0 destroy), matching stock's own plan for the identical mutation on cold_deploy's own state (B4, taken before any marker existed); the apply changed exactly 1 resource, the Name tag reads back as configured and the tofu-address marker is unchanged | +| Rename | pass | 18s | moved block: aws_iam_instance_profile.team_0000_profile renamed with zero churn (0 add, 1 change, 0 destroy) and the plan itself showed the tofu-address marker being rewritten in place; live-mv: team_0001_profile renamed with no moved block at all, reported as a real cloud write; both live instance-profile ids unchanged and both markers read back at the NEW address via the AWS CLI; stock's own oracle over the identical two renames on cold_deploy's state (B1) is also zero churn (No changes., both moves reported); the plan after both renames is empty | +| Remove a block | pass | 7s | deleting two blocks - the taggable, marked aws_iam_instance_profile.team_0002_profile and the UNTAGGABLE aws_iam_role_policy.team_0002_inline, whose parent role stays declared - proposed exactly two destroys (0 add, 0 change, 2 destroy) in an order the cloud accepted, matching stock's own plan for the same two removals on cold_deploy's state (B2); the apply destroyed exactly two, both objects are confirmed gone and the parent role confirmed still live via the AWS CLI, and the next plan is empty | +| Change count | pass | 18s | the estate's OWN count block - six declarations across four resource types, two of them untaggable - scaled 2 to 1 and back: exactly six index-[1] destroys then exactly six index-[1] creates, no index-[0] instance touched in either plan, matching stock's own applied cycle over the identical six-block shape in a separate account (G1); across the whole cycle count_team[0]'s live role id was unchanged and its marker still reads aws_iam_role.count_team[0], count_team_profile[0]'s still reads aws_iam_instance_profile.count_team_profile[0], the recreated count_team[1] carries aws_iam_role.count_team[1], and the plan afterwards is empty | +| Replace with create_before_destroy | pass | 12s | changing aws_iam_instance_profile.team_0004_profile's ForceNew name under create_before_destroy proposed exactly one isolated replace at the same declared address (1 to add, 0 to change, 1 to destroy), matching stock's own plan for the identical change on cold_deploy's state (B3); the apply created the new object and destroyed the old one, the old name no longer resolves and the new one carries the declared address's marker (both read via the AWS CLI), and the next plan is empty with no collision | | Crash between create and destroy (planned) | not run | | | | Teardown (planned) | not run | | | -| Plan, review, apply | not run | | | -| Greenfield apply | pass | 1m25s | choudoufu applied 79 resources into an account a stock destroy had left enumerated empty (A2), and its cloud matches stock's cold deploy across 79 structural facts compared object by object with marker tags never read on either side - the oracle this stage names. Also, beyond the oracle: the six representative identities are correct by value via the AWS CLI across Route 53/IAM/ECS/EC2; the apply persisted 79 records, matching stock's own instance list type for type with no gap - #671 closed the last one (aws_ecs_task_definition), which used to get no record and now does; the next plan is empty; and with the local record store deleted outright every one of the 79 objects is still found - nothing created, destroyed or replaced, 41 of them untaggable and composing from a stamped parent - with the only movement being 1 residue-held aws_ecs_service update(s), which is what deleting the residue store (issue #275) means rather than a divergence | +| Plan, review, apply | pass | 13s | one argument edited through the generator's own render_config path (the reviewp case: aws_subnet.main's tags gain Reviewed=yes - the one shared, singular, taggable resource no later part renames, removes, scales or replaces, and a tags-only update that is not ForceNew, so the subnet id ecs.tf reads is untouched), "plan -out=approved.tfplan" wrote a 31735-byte stock-format plan file whose whole change set is one update on aws_subnet.main (Plan: 0 to add, 1 to change, 0 to destroy); the world then moved out of band (vpc-f48eb49d's Name tag, through the AWS CLI, never through choudoufu) and "apply approved.tfplan" refused with "The approved plan no longer matches the live system" at exit 3, classifying the drift under "This apply would do, and the approved plan does not include:" and naming both aws_vpc.main and the live vpc-f48eb49d it was computed against, with "Exit status 3" spelled out for a pipeline; nothing was applied - subnet-8a9e6960 still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an "Apply complete!" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and subnet-8a9e6960 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The estate re-renders to the pristine generator output, replans empty and the VPC's tofu-address marker still reads aws_vpc.main. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails | +| Greenfield apply | pass | 1m5s | choudoufu applied 79 resources into an account a stock destroy had left enumerated empty (A2), and its cloud matches stock's cold deploy across 79 structural facts compared object by object with marker tags never read on either side - the oracle this stage names. Also, beyond the oracle: the six representative identities are correct by value via the AWS CLI across Route 53/IAM/ECS/EC2; the apply persisted 79 records, matching stock's own instance list type for type with no gap - #671 closed the last one (aws_ecs_task_definition), which used to get no record and now does; the next plan is empty; and with the local record store deleted outright every one of the 79 objects is still found - nothing created, destroyed or replaced, 41 of them untaggable and composing from a stamped parent - with the only movement being 1 residue-held aws_ecs_service update(s), which is what deleting the residue store (issue #275) means rather than a divergence | | Strict profile (not a headline stage) | not run | - | this crossing script does not exercise the strict toggles: strict is Headline:false in tools/gauntlet/stages.go so it moves neither bar, and a toggle-by-toggle refusal fixture is a separate unit from the crossing this script exists to be. live/e2e/reference-ec2-vpc/run.sh's PART G is the pattern for the estate that does carry one | -Last run at commit `d72960cdc3` on 2026-09-06T05:35:09Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 7m7.8s. +Last run at commit `70e2722fa4` on 2026-09-07T03:00:49Z, exit code 0, against emulator image `ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb`. Total run time 5m42.1s. Oracle: stock terraform `1.15.8`, stock tofu `1.12.5`. **Stale**: the current pin is terraform `1.16.0`, tofu `1.12.6`. ## Reproduce it diff --git a/site/data/gauntlet.json b/site/data/gauntlet.json index cf2bf952a9..333f0a6bce 100644 --- a/site/data/gauntlet.json +++ b/site/data/gauntlet.json @@ -151,7 +151,7 @@ "all": { "label": "All estates", "estates": 27, - "clear": 0, + "clear": 27, "stages": { "cold_deploy": { "pass": 27, @@ -164,8 +164,8 @@ "not_run": 0 }, "day2_crash": { - "pass": 1, - "fail": 0, + "pass": 0, + "fail": 1, "not_run": 26 }, "day2_remove": { @@ -204,9 +204,9 @@ "not_run": 0 }, "plan_approval": { - "pass": 0, + "pass": 27, "fail": 0, - "not_run": 27 + "not_run": 0 }, "strict": { "pass": 1, @@ -228,7 +228,7 @@ "core": { "label": "Core estates", "estates": 26, - "clear": 0, + "clear": 26, "stages": { "cold_deploy": { "pass": 26, @@ -241,8 +241,8 @@ "not_run": 0 }, "day2_crash": { - "pass": 1, - "fail": 0, + "pass": 0, + "fail": 1, "not_run": 25 }, "day2_remove": { @@ -281,9 +281,9 @@ "not_run": 0 }, "plan_approval": { - "pass": 0, + "pass": 26, "fail": 0, - "not_run": 26 + "not_run": 0 }, "strict": { "pass": 1, @@ -324,16 +324,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "cb5ae2009fb85fdafd76126bb1842144855eb9d7", - "date": "2026-09-06T04:29:56Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -342,26 +342,28 @@ "exit_code": 0, "detail": { "cold_deploy": "80 resources, once for real (floci fixes #58, #61, #62)", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.) and left count_test[0] untouched; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.); the next plan is empty. Read back through the AWS CLI, never through choudoufu's own report: the destroyed count_test[1] (sg-b60954438b88ecf3d) is absent after the scale-down and comes back under a NEW server-minted GroupId (sg-2e48714be247a863a) carrying tofu-address=aws_security_group.count_test:1, while the survivor count_test[0] keeps BOTH its GroupId (sg-7cd7717aa835c118a) and its tofu-address=aws_security_group.count_test:0 marker across both moves. What witnesses the destroy was measured against the pinned emulator with no terraform in the loop before this section was written: floci sha256:c55d74e1 never reuses a deleted group's GroupId, not even for the same group-name in the same VPC. G-ORACLE is a real stock oracle for the same shape - stock never had this count block, so it was stood up for real with the stock terraform binary in its own working directory and state against its own idle account on :30400 - and stock shows the IDENTICAL shape: destroy the higher index only (sg-0b3d6a90645f55505, then absent), create it back under a new id (sg-11229066e226420ec), count_test[0]=sg-b162ecb6edc79508e unchanged throughout (stock's own plan lines: \"Plan: 0 to add, 0 to change, 1 to destroy.\" down, \"Plan: 1 to add, 0 to change, 0 to destroy.\" up). SYNTHETIC BLOCK, and why: this estate declares no scalable count of its own - examples/complete-alb has no root-level count or for_each at all, terraform-aws-alb's only count is its `local.create ? 1 : 0` boolean create toggle, and everything the module fans out (6 listeners, 7 listener rules, 3 target groups, 2 security-group ingress rules) is for_each over a STRING-KEYED map, which has no index whose slot binding could be scaled and over which \"the higher index is destroyed\" is not expressible - so the sanctioned fallback applies (reference-ec2-vpc Part F, corpus-iam-policy Part G), with aws_security_group, a type this estate already exercises. BREAK_COUNT=1 confirms the check has teeth: asserting the WRONG instance (count_test[0]) was destroyed makes this stage report fail. One emulator divergence noted in passing, harmless to this stage and asserted around rather than papered over: floci answers DescribeSecurityGroups for a deleted group-id with an empty list and exit 0, where real EC2 raises InvalidGroup.NotFound.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.) and left count_test[0] untouched; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.); the next plan is empty. Read back through the AWS CLI, never through choudoufu's own report: the destroyed count_test[1] (sg-c58a962fd8cab5c67) is absent after the scale-down and comes back under a NEW server-minted GroupId (sg-942bc41c28cb76287) carrying tofu-address=aws_security_group.count_test:1, while the survivor count_test[0] keeps BOTH its GroupId (sg-50717691c349face1) and its tofu-address=aws_security_group.count_test:0 marker across both moves. What witnesses the destroy was measured against the pinned emulator with no terraform in the loop before this section was written: floci sha256:c55d74e1 never reuses a deleted group's GroupId, not even for the same group-name in the same VPC. G-ORACLE is a real stock oracle for the same shape - stock never had this count block, so it was stood up for real with the stock terraform binary in its own working directory and state against its own idle account on :20400 - and stock shows the IDENTICAL shape: destroy the higher index only (sg-6a3ec8b4a38b489e9, then absent), create it back under a new id (sg-6c6c45f278c454454), count_test[0]=sg-69fe659703f0b86cb unchanged throughout (stock's own plan lines: \"Plan: 0 to add, 0 to change, 1 to destroy.\" down, \"Plan: 1 to add, 0 to change, 0 to destroy.\" up). SYNTHETIC BLOCK, and why: this estate declares no scalable count of its own - examples/complete-alb has no root-level count or for_each at all, terraform-aws-alb's only count is its `local.create ? 1 : 0` boolean create toggle, and everything the module fans out (6 listeners, 7 listener rules, 3 target groups, 2 security-group ingress rules) is for_each over a STRING-KEYED map, which has no index whose slot binding could be scaled and over which \"the higher index is destroyed\" is not expressible - so the sanctioned fallback applies (reference-ec2-vpc Part F, corpus-iam-policy Part G), with aws_security_group, a type this estate already exercises. BREAK_COUNT=1 confirms the check has teeth: asserting the WRONG instance (count_test[0]) was destroyed makes this stage report fail. One emulator divergence noted in passing, harmless to this stage and asserted around rather than papered over: floci answers DescribeSecurityGroups for a deleted group-id with an empty list and exit 0, where real EC2 raises InvalidGroup.NotFound.", "day2_remove": "choudoufu: deleting aws_instance.other_renamed's block (and its one target-group-attachment reference) proposed exactly 2 destroys (0 add, 0 change, 2 destroy), matching the stock oracle and applied cleanly; the instance is confirmed terminated via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 2 destroys for the same two objects", "day2_rename": "moved block: aws_instance.this renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_instance.other renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state (positioned right after stage 1, before migrate ever touches these shared objects) also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing aws_instance.this_renamed's ForceNew ami argument proposed a forced replace at the same declared address (Plan: 2 to add, 0 to change, 2 to destroy.), applied cleanly; the old instance (i-c13744b209879be03) is confirmed terminated and the new instance (i-574774864b2064b60) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-c13744b209879be03 -> i-574774864b2064b60); the next plan proposes no resource action; stock oracle on cold_deploy's own state (day2_replace ORACLE) also proposes replacing aws_instance.this at the same address (plan only, not applied - it shares floci's account with $ADOPTED_EST); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one address\", naming both live instances) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", + "day2_replace": "choudoufu: changing aws_instance.this_renamed's ForceNew ami argument proposed a forced replace at the same declared address (Plan: 2 to add, 0 to change, 2 to destroy.), applied cleanly; the old instance (i-c6b2f684d28bd4e21) is confirmed terminated and the new instance (i-3079cc0afccd4830b) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-c6b2f684d28bd4e21 -> i-3079cc0afccd4830b); the next plan proposes no resource action; stock oracle on cold_deploy's own state (day2_replace ORACLE) also proposes replacing aws_instance.this at the same address (plan only, not applied - it shares floci's account with $ADOPTED_EST); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one address\", naming both live instances) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one object tampered (the ALB's Example tag), plan proposed fixing exactly module.alb.aws_lb.this[0], apply changed 1 and the Example tag reconverged", "greenfield": "80 resources from nothing, matching stock's own cold-deploy count; the ALB's markers verified via the AWS CLI; 80 records in the local record store including untaggable types; replan empty; a representative EC2 instance's own shape (type/ami) matches stock's cold deploy, via the AWS CLI on both endpoints, marker tags never compared; 50 objects carry the estate tag", "migrate": "51 of 80 stamped, 1 recorded, 0 failed, 28 skipped", + "plan_approval": "one argument edited (the \"ex-instance\" target group's own InstanceTargetGroupTag tag, baz -> reviewed - the one tags argument in examples/complete-alb that reaches exactly one instance, with no dependent resource or data source behind it), \"plan -out=approved.tfplan\" wrote a 159425-byte stock-format plan file whose whole change set is one update on module.alb.aws_lb_target_group.this[\"ex-instance\"]; the world then moved out of band (arn:aws:elasticloadbalancing:eu-west-1:000000000000:loadbalancer/app/ex-complete-alb/ca7ef1d7df57446c's Example tag, through the AWS CLI, never through choudoufu - the same mutation stage 5 uses) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.alb.aws_lb.this[0] and the live arn:aws:elasticloadbalancing:eu-west-1:000000000000:loadbalancer/app/ex-complete-alb/ca7ef1d7df57446c it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - arn:aws:elasticloadbalancing:eu-west-1:000000000000:targetgroup/h115cad549492d8241963b49d84a/02c64f3ef264487c still read InstanceTargetGroupTag=baz through elbv2 describe-tags, not from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the ALB's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:elasticloadbalancing:eu-west-1:000000000000:targetgroup/h115cad549492d8241963b49d84a/02c64f3ef264487c read back with InstanceTargetGroupTag=reviewed, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); 50 tofu-estate-tagged objects before, 50 after", "test_plan": "empty live-plan with no state file; 0 Error diagnostics; the two record-rung aws_route53_record.validation identities verified by value against route53 list-resource-record-sets" }, - "duration_s": 453.8, + "duration_s": 436.2, "stage_seconds": { - "cold_deploy": 104, - "day2_count": 64, - "day2_remove": 23, - "day2_rename": 17, - "day2_replace": 33, - "drift_reconverge": 38, - "greenfield": 98, - "migrate": 67, + "cold_deploy": 88, + "day2_count": 48, + "day2_remove": 22, + "day2_rename": 16, + "day2_replace": 32, + "drift_reconverge": 39, + "greenfield": 95, + "migrate": 65, + "plan_approval": 22, "test_apply": 5, "test_plan": 4 } @@ -388,16 +390,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -406,27 +408,29 @@ "exit_code": 0, "detail": { "cold_deploy": "Apply complete! Resources: 68 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=autoscaling-complete-crossing before migration", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) and it was the HIGHER index, count_test[1] (sg-751e8f493815cc85a); count_test[0] (sg-278ee188356fcd904) kept its live GroupId, its tofu-address=aws_security_group.count_test:0 marker and its local record across the move, all re-read through the AWS CLI and off the record-store file rather than out of choudoufu's own report. Scaling back from 1 to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy) and count_test[1] came back as a genuinely NEW object (sg-751e8f493815cc85a -> sg-03e3220bce1d89e1b) carrying tofu-address=aws_security_group.count_test:1, with its record re-keyed to the new id rather than left stale on the destroyed one; count_test[0] was untouched throughout and the next plan is empty. Stock oracle (G0): a plain terraform working directory standing the identical 2-instance block up for real in its own VPC and scaling it the same way shows the identical shape - destroy count_test[1] (sg-197c7dafd5ec22f07) only on the way down, create count_test[1] back under a new id (sg-60fdd9f4aa9f57fce) on the way up, count_test[0] (sg-ac913a1b237fb74e4) untouched both times - and was torn down before the choudoufu leg ran. SYNTHETIC BLOCK, and why: this estate declares no scalable count anywhere - the module's own aws_autoscaling_group.this/.idc pair is a 0-or-1 boolean toggle, the schedules/policies/traffic-source attachments are for_each over name-keyed maps, and none of the twelve module calls carries count or for_each - so this is live/GAUNTLET.md #8's sanctioned self-contained fallback, the same one reference-ec2-vpc's Part F and corpus-iam-policy's Part G use, on aws_security_group, a type this estate already exercises through module.asg_sg. An ASG's own desired_capacity is deliberately NOT what was scaled: this stage is about the count meta-argument's slot binding (internal/live/discovery/count.go), not about a live group's capacity. BREAK_COUNT=1 confirms the check has teeth: expecting count_test[0] to be the destroyed instance makes this stage report fail.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) and it was the HIGHER index, count_test[1] (sg-be0d8c436b598d206); count_test[0] (sg-87554ef6a11f83aab) kept its live GroupId, its tofu-address=aws_security_group.count_test:0 marker and its local record across the move, all re-read through the AWS CLI and off the record-store file rather than out of choudoufu's own report. Scaling back from 1 to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy) and count_test[1] came back as a genuinely NEW object (sg-be0d8c436b598d206 -> sg-530659bb51950b409) carrying tofu-address=aws_security_group.count_test:1, with its record re-keyed to the new id rather than left stale on the destroyed one; count_test[0] was untouched throughout and the next plan is empty. Stock oracle (G0): a plain terraform working directory standing the identical 2-instance block up for real in its own VPC and scaling it the same way shows the identical shape - destroy count_test[1] (sg-b3bf410755a7c74b5) only on the way down, create count_test[1] back under a new id (sg-55b912566a9fb1171) on the way up, count_test[0] (sg-69115075c0a480bab) untouched both times - and was torn down before the choudoufu leg ran. SYNTHETIC BLOCK, and why: this estate declares no scalable count anywhere - the module's own aws_autoscaling_group.this/.idc pair is a 0-or-1 boolean toggle, the schedules/policies/traffic-source attachments are for_each over name-keyed maps, and none of the twelve module calls carries count or for_each - so this is live/GAUNTLET.md #8's sanctioned self-contained fallback, the same one reference-ec2-vpc's Part F and corpus-iam-policy's Part G use, on aws_security_group, a type this estate already exercises through module.asg_sg. An ASG's own desired_capacity is deliberately NOT what was scaled: this stage is about the count meta-argument's slot binding (internal/live/discovery/count.go), not about a live group's capacity. BREAK_COUNT=1 confirms the check has teeth: expecting count_test[0] to be the destroyed instance makes this stage report fail.", "day2_remove": "choudoufu: deleting module.default's block proposed exactly 2 destroys (0 add, 0 change, 2 destroy), matching the stock oracle's own count and applied cleanly; the live ASG count dropped by exactly one and the tagged object count dropped too, both confirmed via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 2 destroys for the same module", "day2_rename": "moved block: module.asg_sg renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place on its security group; live-mv: aws_sqs_queue.this renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.asg_sg_renamed's ForceNew name argument (module CALL, passed through to its own aws_security_group.this_name_prefix's name_prefix) proposed a forced replace at the same declared address (Plan: 3 to add, 2 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-c37bb9262afc41e48) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-05ed298dde97955f0 -> sg-c37bb9262afc41e48); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 2 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted aws_sqs_queue.this_renamed and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on this branch, GitHub issue #412: propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard; see this section's own header comment for the fix and eks-basic's/ecs-fargate's matching ones in this same unit, which independently hit the identical shape and were not re-run for #412.", + "day2_replace": "choudoufu: changing module.asg_sg_renamed's ForceNew name argument (module CALL, passed through to its own aws_security_group.this_name_prefix's name_prefix) proposed a forced replace at the same declared address (Plan: 3 to add, 2 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-091e8a0748dde3c56) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-e756cc13263688dd3 -> sg-091e8a0748dde3c56); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 2 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted aws_sqs_queue.this_renamed and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on this branch, GitHub issue #412: propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard; see this section's own header comment for the fix and eks-basic's/ecs-fargate's matching ones in this same unit, which independently hit the identical shape and were not re-run for #412.", "drift_reconverge": "one object tampered (SQS queue 'complete's Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag", "greenfield": "68 resources from nothing, matching stock's own cold-deploy count (68); the sqs queue's markers verified via the AWS CLI; 68 records in the local record store including the untaggable ASGs (#364 A2); replan empty; the asg_sg security group's rule counts match stock's cold deploy structurally, via the AWS CLI on both endpoints, marker tags never compared; 41 objects carry the estate tag", "migrate": "41 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 re-recorded for sensitivity only, 0 already recorded, 0 failed, 27 skipped; 41 objects carry tofu-estate=autoscaling-complete-crossing", + "plan_approval": "one argument edited (aws_iam_role.ssm's tags gain Reviewed=yes - a single instance whose only dependent reads its name, which an in-place tag update leaves known), \"plan -out=approved.tfplan\" wrote a 275504-byte stock-format plan file whose whole change set is one update on aws_iam_role.ssm; the world then moved out of band (the SQS queue's Example tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both aws_sqs_queue.this and the live https://sqs.eu-west-1.amazonaws.com/000000000000/complete it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - IAM role complete still carried no Reviewed tag, read back through iam list-role-tags rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the queue's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the role read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 41 objects before, 41 after, no state file either time", "test_plan": "empty plan; identity re-check unchanged: module.complete.aws_launch_template.this:0, aws_iam_role.ssm" }, - "duration_s": 417.3, + "duration_s": 412.6, "stage_seconds": { - "cold_deploy": 105, - "day2_count": 57, - "day2_remove": 21, - "day2_rename": 18, + "cold_deploy": 87, + "day2_count": 55, + "day2_remove": 20, + "day2_rename": 17, "day2_replace": 18, - "drift_reconverge": 8, + "drift_reconverge": 10, "greenfield": 99, - "migrate": 81, - "test_apply": 5, + "migrate": 77, + "plan_approval": 21, + "test_apply": 4, "test_plan": 4 } }, @@ -452,16 +456,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -473,24 +477,26 @@ "day2_count": "choudoufu: scaling the synthetic aws_dynamodb_table.count_test from 2 to 1 (issue #359/#488's own fallback clause - this estate's real module has no honest resource-level count/for_each knob: create_table is boolean-shaped and replica_regions/global_secondary_indexes drive dynamic blocks nested inside the SAME table resource, not a separate resource instance, confirmed by reading main.tf directly) destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), confirmed gone via the AWS CLI, its local record correctly tombstoned rather than left claiming a live identity (#398-guard shape, has(tombstone) and not has(identity)), and left count_test[0]'s live TableId and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] again under the SAME ARN (deterministic from region+account+name - established directly against floci with no tofu in the loop before writing this assertion) but a NEW TableId (0 add -> 1 add, 0 change, 0 destroy), and its local record returned to a live identity, while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical 2-instance count block, applied for real in a dedicated always-idle account never shared with this one, shows the identical shape: destroy the higher index only, create it back under the same ARN but a new TableId, the lower index's TableId unchanged both times. BREAK_COUNT=1 confirms the wrong-instance assertion correctly fails to hold.", "day2_remove": "choudoufu: deleting module.dynamodb_table_final's block proposed exactly two destroys (0 add, 0 change, 2 destroy: the table and its untaggable resource policy), applied cleanly (0 added, 0 changed, 2 destroyed), the table is genuinely gone from the live account (dynamodb describe-table on the old name now returns ResourceNotFoundException, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly two destroys for the same two objects; classifyOrphans did not withhold either destroy because module.disabled_dynamodb_table declares zero instances of the same block key (create_table=false), so nothing is ever pending against it", "day2_rename": "moved block: module.dynamodb_table renamed to module.dynamodb_table_moved with zero churn (0 add, 1 change, 0 destroy) - the table's own marker rewritten in place, the untaggable resource policy unaffected; live-mv: module.dynamodb_table_moved renamed to module.dynamodb_table_final with zero churn, marker rewritten in place; stock oracle over the same net module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); the table's ARN unchanged throughout, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.dynamodb_table_final's ForceNew name argument proposed exactly one table replace at the same declared address, cascading into the untaggable resource policy (its resource_arn argument follows the table's ARN and is not independently updatable, so it also replaces - F-ORACLE's own finding); applied cleanly; the old table is confirmed gone via the AWS CLI (ResourceNotFoundException) and the new table carries the marker; the local record store's record at the same address now names the new table's name, not the destroyed one (my-table-clear-horse -> my-table-clear-horse-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the table at the same address (plan only, not applied - it shares floci's account with $EST); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones.", - "drift_reconverge": "one object tampered (arn:aws:dynamodb:eu-west-1:000000000000:table/my-table-clear-horse's Terraform tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag", + "day2_replace": "choudoufu: changing module.dynamodb_table_final's ForceNew name argument proposed exactly one table replace at the same declared address, cascading into the untaggable resource policy (its resource_arn argument follows the table's ARN and is not independently updatable, so it also replaces - F-ORACLE's own finding); applied cleanly; the old table is confirmed gone via the AWS CLI (ResourceNotFoundException) and the new table carries the marker; the local record store's record at the same address now names the new table's name, not the destroyed one (my-table-probable-spaniel -> my-table-probable-spaniel-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the table at the same address (plan only, not applied - it shares floci's account with $EST); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones.", + "drift_reconverge": "one object tampered (arn:aws:dynamodb:eu-west-1:000000000000:table/my-table-probable-spaniel's Terraform tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag", "greenfield": "3 resources from nothing (random_pet + table + resource policy), the table's markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on key schema/attributes/table class/deletion protection/on-demand billing/GSI/resource policy", "migrate": "1 resource(s) newly stamped, 0 already stamped, 1 newly recorded, 0 re-recorded for sensitivity only, 0 already recorded, 0 failed, 1 skipped.; Apply complete! Resources: 0 added, 0 changed, 1 destroyed. (tofu-slot convergence)", + "plan_approval": "one argument edited (the resource_policy heredoc's statement Sid, AllowDummyRoleAccess -> AllowDummyRoleAccessReviewed, reaching only module.dynamodb_table.aws_dynamodb_resource_policy.this[0]), \"plan -out=approved.tfplan\" wrote a 21195-byte stock-format plan file whose whole change set is that one update; the world then moved out of band (arn:aws:dynamodb:eu-west-1:000000000000:table/my-table-probable-spaniel's Terraform tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.dynamodb_table.aws_dynamodb_table.this[0] and the live table id my-table-probable-spaniel it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - dynamodb get-resource-policy still returned a policy without the reviewed Sid, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the table's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the live policy read back WITH the reviewed Sid, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. This estate owns only two objects the plan acts on, so the review deliberately sits on the resource policy and the move on the table: that is what makes the refusal an EXTRA row it can name rather than a values-only disagreement about one row. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 1 objects before, 1 after, no state file either time", "test_plan": "no resource change proposed, nothing foreign; identity re-check (via the AWS CLI) unchanged" }, - "duration_s": 242.5, + "duration_s": 243.8, "stage_seconds": { - "cold_deploy": 35, + "cold_deploy": 23, "day2_count": 31, "day2_remove": 6, "day2_rename": 11, "day2_replace": 13, "drift_reconverge": 6, - "greenfield": 45, - "migrate": 90, - "test_apply": 3, + "greenfield": 46, + "migrate": 91, + "plan_approval": 13, + "test_apply": 2, "test_plan": 2 } }, @@ -516,16 +522,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -534,26 +540,28 @@ "exit_code": 0, "detail": { "cold_deploy": "35 resources added across 13 types (aws_instance, aws_eip, aws_iam_role/instance_profile/role_policy_attachment, aws_ebs_volume, aws_volume_attachment, aws_security_group x2, aws_vpc_security_group_egress_rule x2, aws_security_group_rule x2, vpc/subnet/route*/igw/default_* from the vpc module), 0 objects carry tofu-estate before migration", - "day2_count": "choudoufu: scaling aws_ebs_volume.count_test from 2 to 1 destroyed exactly count_test[1] (vol-5d2c1b342f912c2f6, 0 add, 0 change, 1 destroy) and left count_test[0] (vol-a2f08d6e0a3772fd1) with the same live VolumeId, the same tofu-address=aws_ebs_volume.count_test:0 and the same tofu-slot=0, all read back through the AWS CLI rather than choudoufu's own report; the destroyed volume is genuinely gone (describe-volumes answers InvalidVolume.NotFound for it). Scaling back from 1 to 2 planned exactly 1 to add, 0 to change, 0 to destroy and brought count_test[1] back as a NEW object (vol-0778a981679882f57, not vol-5d2c1b342f912c2f6) carrying tofu-address=aws_ebs_volume.count_test:1 and tofu-slot=1, above the live high-water mark count_test[0] still holds, while count_test[0] stayed untouched throughout; the next plan is empty, and scaling the block to zero destroys both and leaves the estate planning empty again. C-ORACLE, the same 2-instance block stood up for real with plain terraform in its own working directory at the SAME resolved provider version (6.63.0), shows the identical shape: destroy the higher index only (vol-6c64f51b35c8db3ec), create the higher index back under a new id (vol-a8a9c51961c08f115), the lower index's id (vol-eee10750df38c22af) unchanged both times. SYNTHETIC BLOCK, and why: terraform-aws-ec2-instance v6.4.0 declares no scalable count or for_each knob this estate reaches - all nine of its own count usages are boolean create toggles of the form 'count = local.create ? 1 : 0', which can never hold two instances, and the upstream example's one real for_each fan-out (module.ec2_multiple) is dropped by this script's reduction because floci does not model the surfaces around it - so this section adds a new, self-contained count block of a type the estate ALREADY exercises (aws_ebs_volume, module.ec2_complete's own /dev/sdf data volume), the sanctioned fallback live/GAUNTLET.md #8 names, with reference-ec2-vpc Part F and corpus-iam-policy Part G as precedent. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports day2_count fail, proving the assertion is load-bearing.", + "day2_count": "choudoufu: scaling aws_ebs_volume.count_test from 2 to 1 destroyed exactly count_test[1] (vol-227203145b6519f82, 0 add, 0 change, 1 destroy) and left count_test[0] (vol-1ba72164534af9b35) with the same live VolumeId, the same tofu-address=aws_ebs_volume.count_test:0 and the same tofu-slot=0, all read back through the AWS CLI rather than choudoufu's own report; the destroyed volume is genuinely gone (describe-volumes answers InvalidVolume.NotFound for it). Scaling back from 1 to 2 planned exactly 1 to add, 0 to change, 0 to destroy and brought count_test[1] back as a NEW object (vol-25211d5ff11f1d3d0, not vol-227203145b6519f82) carrying tofu-address=aws_ebs_volume.count_test:1 and tofu-slot=1, above the live high-water mark count_test[0] still holds, while count_test[0] stayed untouched throughout; the next plan is empty, and scaling the block to zero destroys both and leaves the estate planning empty again. C-ORACLE, the same 2-instance block stood up for real with plain terraform in its own working directory at the SAME resolved provider version (6.63.0), shows the identical shape: destroy the higher index only (vol-43eecd95f10607c2e), create the higher index back under a new id (vol-d1189cb4ab1b4dbf8), the lower index's id (vol-63f626cc1ed09eb2e) unchanged both times. SYNTHETIC BLOCK, and why: terraform-aws-ec2-instance v6.4.0 declares no scalable count or for_each knob this estate reaches - all nine of its own count usages are boolean create toggles of the form 'count = local.create ? 1 : 0', which can never hold two instances, and the upstream example's one real for_each fan-out (module.ec2_multiple) is dropped by this script's reduction because floci does not model the surfaces around it - so this section adds a new, self-contained count block of a type the estate ALREADY exercises (aws_ebs_volume, module.ec2_complete's own /dev/sdf data volume), the sanctioned fallback live/GAUNTLET.md #8 names, with reference-ec2-vpc Part F and corpus-iam-policy Part G as precedent. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports day2_count fail, proving the assertion is load-bearing.", "day2_remove": "choudoufu: deleting module.ec2_complete's block proposed exactly 10 destroys (0 add, 0 change, 10 destroy), matching the stock oracle's own count and applied cleanly; the instance is confirmed terminated and the tagged object count dropped, both via the AWS CLI, not through choudoufu's own report; the next plan proposes no resource action; stock oracle on cold_deploy's own state (D-REMOVE-ORACLE) also proposes exactly 10 destroys for the same module", "day2_rename": "moved block: module.vpc renamed with zero churn (0 add, 15 change, 0 destroy), marker rewritten in place; live-mv: module.security_group's security group renamed with zero churn, its two untaggable rules followed for free; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.ec2_complete's ForceNew ami argument proposed exactly one instance replace at the same declared address, cascading into the eip (updated in-place) and the volume attachment (also replaced, instance_id is ForceNew there too) - 2 to add, 1 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated and the new instance carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-b4eb27b23605d3e67 -> i-7a9aa80c70f73af43); the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one.", + "day2_replace": "choudoufu: changing module.ec2_complete's ForceNew ami argument proposed exactly one instance replace at the same declared address, cascading into the eip (updated in-place) and the volume attachment (also replaced, instance_id is ForceNew there too) - 2 to add, 1 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated and the new instance carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance's id, not the terminated one (i-c7a93b787f29f6013 -> i-2bfd2886f8aefa0fd); the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Indistinguishable instances without per-instance markers\", naming both live instances) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one.", "drift_reconverge": "one object tampered, exactly 1 object proposed and applied (0 added, 1 changed, 0 destroyed), tag reconverged to \"ex-complete\"", "greenfield": "35 resources from nothing, matching stock's own cold-deploy count; the instance's markers verified via the AWS CLI; 35 records in the local record store including untaggable types; replan empty; the instance's own shape (type/ami/block-device-count) matches stock's cold deploy, via the AWS CLI on both endpoints, marker tags never compared; 24 objects carry the estate tag", "migrate": "24 of 35 eligible (11 untaggable across 5 types - aws_iam_role_policy_attachment, aws_volume_attachment, aws_security_group_rule x2, aws_route, aws_route_table_association x6 - all resolved by provider identity schema), 24 stamped, 0 failed, 11 skipped; the IAM role policy attachment's composite live id asserted by value; genuine no-op on the follow-up apply", + "plan_approval": "one argument edited (the \"/dev/sdf\" entry's MountPoint volume tag inside module \"ec2_complete\"'s ebs_volumes argument, /mnt/data -> /mnt/data-reviewed - the module merges each entry's tags into that entry's aws_ebs_volume alone, so it reaches module.ec2_complete.aws_ebs_volume.this[\"/dev/sdf\"] and nothing else), \"plan -out=approved.tfplan\" wrote a 72468-byte stock-format plan file whose whole change set is that one update; the world then moved out of band (i-c7a93b787f29f6013's Example tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.ec2_complete.aws_instance.this[0] and the live i-c7a93b787f29f6013 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - volume vol-877db6c95f823e408 still read MountPoint=/mnt/data through ec2 describe-tags, not from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with i-c7a93b787f29f6013's tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and vol-877db6c95f823e408 read back with MountPoint=/mnt/data-reviewed, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART C starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); 24 objects before, 24 after, no state file", "test_plan": "no resource change proposed by either plan; the default plan reports \"nothing was swept\" (the CollectUnclaimed ruling (#604) made the account-inventory question opt-in, and a run that did not ask must say so), and a second plan run with TOFU_LIVE_COLLECT_UNCLAIMED=1 finds exactly 8 foreign objects - the instance's own root volume plus floci's default-VPC bootstrap; instance tofu-address re-checked against EC2" }, - "duration_s": 422.7, + "duration_s": 398.5, "stage_seconds": { - "cold_deploy": 78, - "day2_count": 131, - "day2_remove": 30, - "day2_rename": 17, + "cold_deploy": 56, + "day2_count": 127, + "day2_remove": 28, + "day2_rename": 14, "day2_replace": 50, "drift_reconverge": 6, - "greenfield": 71, + "greenfield": 61, "migrate": 30, + "plan_approval": 17, "test_apply": 4, "test_plan": 5 } @@ -579,16 +587,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "3977d907842a4cefa8e5fbe881732571988f4fd5", - "date": "2026-09-06T05:49:39Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -597,28 +605,30 @@ "exit_code": 0, "detail": { "cold_deploy": "62 resources, once for real", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.), leaving count_test[0]'s live GroupId (sg-38d62e21de06f4816), its tofu-address (aws_security_group.count_test:0) and its tofu-slot (0) unchanged, and count_test[1] (sg-6b3d7970b4dc6d8b9) genuinely absent - all read through the AWS CLI, never choudoufu's own report; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.) under a NEW server-minted GroupId (sg-49a87d7c99c8dbff3, was sg-6b3d7970b4dc6d8b9) carrying tofu-address=aws_security_group.count_test:1 and the same tofu-slot=1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (H-ORACLE), the identical 2-instance block stood up for real with plain terraform in its own account and its own working directory - stock never had this block, so unlike day2_rename/day2_remove/day2_replace there is no cold_deploy state to reuse - shows the identical shape: Plan: 0 to add, 0 to change, 1 to destroy. destroying count_test[1]=sg-703e202276aee9559 only, then Plan: 1 to add, 0 to change, 0 to destroy. bringing it back under a new GroupId (sg-ae896ab2a2a7983c6), with count_test[0]=sg-3b2059c3f4807ab3a unchanged both times. SYNTHETIC block, and why: this estate's corpus example (terraform-aws-modules/terraform-aws-ecs v7.6.0, examples/fargate) declares no scalable count knob at all - every count in its vendored modules is a 'count = local.create ? 1 : 0' boolean create toggle, which cannot hold two instances - and the only construct holding three, module.vpc's private_subnets, is driven by local.azs, which the ECS service's network configuration, the ALB and the NAT gateway all read, so scaling it churns a dozen unrelated resources instead of isolating which count instance a scale-down destroys. aws_security_group was chosen because this estate already exercises it three times in its own vendored modules (modules/cluster, modules/service, modules/express-service each declare aws_security_group.this) and because its destroy is witnessed TWICE on the pinned emulator - a new GroupId AND verified absence in between - established by reading the EC2 API directly with no tofu in the loop before any assertion here was written. Placed last, after every other stage reported, at an address nothing else in this configuration names, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly fails.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (Plan: 0 to add, 0 to change, 1 to destroy.), leaving count_test[0]'s live GroupId (sg-52d986d91b3c973e6), its tofu-address (aws_security_group.count_test:0) and its tofu-slot (0) unchanged, and count_test[1] (sg-492493a03f1be337d) genuinely absent - all read through the AWS CLI, never choudoufu's own report; scaling back from 1 to 2 created exactly count_test[1] (Plan: 1 to add, 0 to change, 0 to destroy.) under a NEW server-minted GroupId (sg-e97c66f6e7938bb49, was sg-492493a03f1be337d) carrying tofu-address=aws_security_group.count_test:1 and the same tofu-slot=1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (H-ORACLE), the identical 2-instance block stood up for real with plain terraform in its own account and its own working directory - stock never had this block, so unlike day2_rename/day2_remove/day2_replace there is no cold_deploy state to reuse - shows the identical shape: Plan: 0 to add, 0 to change, 1 to destroy. destroying count_test[1]=sg-2f52b9dfff25eabef only, then Plan: 1 to add, 0 to change, 0 to destroy. bringing it back under a new GroupId (sg-5feabe05e9ba79cb0), with count_test[0]=sg-c38e15e20448864cb unchanged both times. SYNTHETIC block, and why: this estate's corpus example (terraform-aws-modules/terraform-aws-ecs v7.6.0, examples/fargate) declares no scalable count knob at all - every count in its vendored modules is a 'count = local.create ? 1 : 0' boolean create toggle, which cannot hold two instances - and the only construct holding three, module.vpc's private_subnets, is driven by local.azs, which the ECS service's network configuration, the ALB and the NAT gateway all read, so scaling it churns a dozen unrelated resources instead of isolating which count instance a scale-down destroys. aws_security_group was chosen because this estate already exercises it three times in its own vendored modules (modules/cluster, modules/service, modules/express-service each declare aws_security_group.this) and because its destroy is witnessed TWICE on the pinned emulator - a new GroupId AND verified absence in between - established by reading the EC2 API directly with no tofu in the loop before any assertion here was written. Placed last, after every other stage reported, at an address nothing else in this configuration names, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly fails.", "day2_remove": "choudoufu: deleting module.ecs_task_definition's block proposed exactly 8 destroys (0 add, 0 change, 8 destroy), address-for-address identical to stock's oracle on cold_deploy's own state; applied cleanly (0 added, 0 changed, 8 destroyed); the standalone task definition family (ex-fargate-standalone) genuinely has 0 active revisions afterward, read via the AWS CLI, not choudoufu's own report; classifyOrphans did not withhold any destroy because no other module.ecs_task_definition block is declared anywhere in this config; the next plan is empty", "day2_rename": "moved block: module.alb renamed with zero churn (0 add, 9 change, 0 destroy), marker rewritten in place; live-mv: aws_service_discovery_http_namespace.this renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", "day2_replace": "choudoufu: changing module.ecs_task_definition's ForceNew name argument (module CALL, passed through to the local module's own family = coalesce(var.family, var.name)) proposed a forced replace at the same declared address (Plan: 8 to add, 0 to change, 8 to destroy.), applied cleanly; the old task definition is confirmed INACTIVE via the AWS CLI (ECS deregisters rather than deletes) and the new one (arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1) carries the marker, moved via the tofu-address tag (arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1 -> arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the task definition at the same address (Plan: 8 to add, 0 to change, 8 to destroy., plan only, not applied - it shares floci's account with $ADOPTED_EST); BREAK=replace confirms a manufactured marker collision - a second, genuinely live task definition wearing this address's marker, which no tombstone names as destroyed - is still reported loudly (\"Indistinguishable instances without per-instance markers\", naming both ARNs) rather than pruned or silently proposed as nothing, which is #849's own rule holding on this route too. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; the local record store is read by value on both sides of the replace (#879): before it, the record names family=ex-fargate-standalone with identity.secondary_id=arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1; after it, identity.secondary_id=arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone-v2:1 with exactly one tombstone naming arn:aws:ecs:eu-west-1:000000000000:task-definition/ex-fargate-standalone:1, which is what lets the F2 plan tell the deregistered object's lingering tag from a second live claimant instead of refusing \"Indistinguishable instances without per-instance markers\" forever; two earlier target choices each found a genuine, separate defect: aws_service_discovery_http_namespace.this_renamed's (mv.go's propagateModuleRename skipping MoveRecord for a same-module rename) is FIXED on the gauntlet/mv-rekey branch, GitHub issue #412 - see F-ORACLE's own header comment and corpus-autoscaling-complete's/corpus-eks-basic's matching mv.go finding in this same unit, neither of which was re-run for #412; module.alb_renamed's (the non-converging cascade, F-ORACLE's own header comment, finding 2) remains a separate, open finding, not fixed here.", "drift_reconverge": "one object tampered (VPC's Name tag), plan proposed fixing exactly module.vpc.aws_vpc.this[0], apply changed 1 and the Name tag reconverged", "greenfield": "62 resources from nothing, cluster marker verified via the AWS CLI, 62 of 62 records in the local record store (#364 A2; both aws_ecs_task_definition instances are now included, #671 having closed the numeric-wire-identity-component gap in internal/live/identity/located.go's LocatedIdentityPlanFor that used to exclude them), replan empty, stock oracle in its own namespace matches structurally on cluster/service/standalone-task-definition/CloudMap-namespace/ALB/VPC", "migrate": "46 of 62 stamped", + "plan_approval": "one argument edited (module \"ecs_service\"'s service_tags ServiceTag, \"Tag on service level\" -> \"Tag on service level, reviewed\" - the service module merges service_tags into aws_ecs_service's own tags and nowhere else, so it reaches exactly one instance where every tags = local.tags in this example fans out over a whole module), \"plan -out=approved.tfplan\" wrote a 147864-byte stock-format plan file whose whole change set is one update on module.ecs_service.aws_ecs_service.this[0]; the world then moved out of band (VPC vpc-0ee657b8's Name tag, through the AWS CLI, never through choudoufu - the same mutation STAGE 5 uses) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.vpc.aws_vpc.this[0] and the live vpc-0ee657b8 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - arn:aws:ecs:eu-west-1:000000000000:service/ex-fargate/ex-fargate still read ServiceTag=\"Tag on service level\" through ecs list-tags-for-resource, not from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the VPC's Name tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the service read back with the reviewed ServiceTag, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); 46 tofu-estate-tagged objects before, 46 after, no state file either time", "test_plan": "genuinely empty replan (\"No changes. Your infrastructure matches the configuration.\") - #371, #378, #372, #110, #395 and #376 all fixed and stay fixed; the standalone task definition's essential/mountPoints[].readOnly wall is gone (lex00/floci#131, published and repinned this unit) and essential defaulting to true was never an independent wall on its own (this unit's own re-measurement). #395/#376: choudoufu keeps no persisted state, so every plan re-derives PriorState through ImportResourceState's bare stub; internal/live/projection/build.go's configuredAttrsSeed generalizes the tags-only import-stub seed (issue #287 item 8) to every Required-or-Optional-non-Computed attribute (fixing #376's track_latest/skip_destroy directly), and internal/live/projection/residue.go's residueConfigSourced widening of classifyResidue plus the new builder.residueSeedFor pre-read seed close #395's managed-reference case (task_definition = aws_ecs_task_definition.this[0].arn) that configuredAttrsSeed's static evaluator alone could not reach. Identities confirmed by value against the AWS CLI: $CLUSTER_ARN, $TD_SVC_ARN, $TD_STANDALONE_ARN, and #368's scalable target $GOT_TARGET_RID." }, - "duration_s": 648, + "duration_s": 550.4, "stage_seconds": { - "cold_deploy": 105, - "day2_count": 122, - "day2_remove": 20, - "day2_rename": 59, - "day2_replace": 25, - "drift_reconverge": 12, - "greenfield": 184, - "migrate": 86, + "cold_deploy": 82, + "day2_count": 64, + "day2_remove": 15, + "day2_rename": 52, + "day2_replace": 16, + "drift_reconverge": 11, + "greenfield": 172, + "migrate": 78, + "plan_approval": 27, "test_apply": 5, - "test_plan": 29 + "test_plan": 28 } }, "notes": "RE-CROSSED FOR REAL 2026-08-21 at dbc5ebf575 (GitHub issue #346's fix), twice, the second time instrumented to capture live-plan's raw output. STAGES UNCHANGED at 2 of 5, and #346's fix does not reach this estate either. Stage 1: 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', cluster confirmed carrying no tofu-address beforehand. Stage 2: '46 of 62 eligible (28 VERIFIED + 18 DRIFTED); 16 skipped'; '46 stamped, 1 recorded (time_sleep.this[0]), 0 failed, 15 skipped'; both markers confirmed by direct aws ecs describe-* calls against floci rather than through choudoufu's own report (cluster/ex-fargate = module.ecs_cluster.aws_ecs_cluster.this:0, service/ex-fargate/ex-fargate = module.ecs_service.aws_ecs_service.this:0). Stage 3 fails on exactly 8 diagnostics, the same 8 as before, verified two ways (grep -c '^Error:' over the raw output, and the script's own hard count assertions): 1 'Module output not supported in static context' on main.tf:68 (cluster_arn = module.ecs_cluster.arn), 1 'Unable to compute static value' on modules/service/main.tf:1565 (aws_appautoscaling_target.this[0].resource_id), and 6 'Unresolvable identity' cascading from it onto aws_appautoscaling_policy.this['cpu'] and ['memory']. Why the fix misses it: the same value-shaped module-CALL-argument route as corpus-rds-complete-postgres, PLUS a transform the part-shaped route could not express even if it reached - local.cluster_name = element(split('/', var.cluster_arn), 1), a function applied to the deferred value. identity.Formula holds literals and ParentRefs and has no way to say 'split this parent attribute and take element 1'. That is a mechanism this repository does not have, not a gap in #346's fix. A 'Provider version does not match the admission evidence version' warning (6.61.0 resolved against 6.59.0 admission evidence) also prints; the script's own header calls it a caution, not a failure. PRIOR HISTORY BELOW. Landed dd83121592 (2026-08-18), 62 resources: ECS cluster (Container Insights, FARGATE/FARGATE_SPOT split), a BLUE_GREEN service behind an ALB, ECS Exec, ECS Service Connect, a two-container task definition plus a standalone second one, a CloudMap namespace, nested ALB/VPC modules. cold_deploy and migrate genuinely pass (43 of 62 stamped: 26 VERIFIED + 17 DRIFTED; 19 skipped - 16 untaggable by design, 3 blocked by #305). test_plan blocked by 4 sites: #305's familiar default_* trio (3 sites) and a NEW one, filed as #308: the child-module for_each keyset prover (internal/live/identity/foreach_keyset.go) has no case for a for-comprehension (for k, v in var.container_definitions : k => v if ...) and doesn't chase a bare var.X for_each source across a module-call boundary to the literal object constructor at the caller, whose keys are actually static even though one unrelated attribute value inside the map is dynamic - resolve.go's resource-level forEachOverComprehension already does the equivalent per-entry evaluation the module-call prover lacks. Two real floci gaps found and filed but not fixed, both sized as real modeling work rather than quick patches: lex00/floci#59 (CreateCluster silently drops settings/Container Insights, default_capacity_provider_strategy never serialized back) and lex00/floci#60 (CreateService/DescribeServices drop scheduling_strategy, enable_ecs_managed_tags, enable_execute_command, health_check_grace_period_seconds, deployment_controller, blue-green load_balancer.advanced_configuration, service_connect_configuration entirely - scheduling_strategy's omission in particular forces the AWS provider to propose destroy-and-recreate on every plan after creation, a real non-idempotency bug independent of choudoufu). Neither floci gap blocks this crossing's own outcome since test_plan already refuses earlier, upstream of any ECS-field diff. Follow-up pass 2026-08-18 (#313 cross-check, 0a94070b16/3ff22c5be6): the committed run.sh still asserted pre-#305 counts (43/62 eligible) and failed before ever reaching stage 3; updated to the real current numbers (46/62 eligible, #305's default_* trio now fully resolved here) and re-verified against real floci twice plus a BREAK=1 negative control, all read from the script's own printed lines. test_plan's sole remaining blocker is confirmed #308 alone - #313's diagnostic does not appear anywhere in the output. Mechanism: this estate's vpc submodule expands subnets via count over a statically-known length, never a for_each keyed on the AZ name values, so #313 structurally can't reach it. Commented on #313 and #308. #308 fixed and merged 2026-08-18 (a9ac6d06e7/b2bb59585d, generic: a *hclsyntax.ForExpr case plus a cross-module-call var/local chase in internal/live/identity/foreach_keyset.go, reaching every module-call for_each proof, not just this estate) - re-run confirms 0 occurrences of #308's diagnostic (was 1), but test_plan is still blocked: #308 firing first had been masking two more causes in the same live-plan output all along. Follow-up pass 2026-08-18 (e74c7d5869/07c7317ab6) re-staled run.sh's stage-3 assertions and header to the real current picture, verified across three separate real live-plan runs (stable counts each time) plus a BREAK=1 negative control: 236 total diagnostics, three distinct root causes, not one. Root cause A (#313's canonical shape, 48 sites): data.aws_availability_zones.available feeding local.azs into module \"vpc\"'s azs argument. Root cause B (also #313's family via a module output, 1 site): module.ecs_cluster.arn passed into module \"ecs_service\" as cluster_arn, \"Module output not supported in static context\". Both A and B are #313's already-ruled maintainer-level architecture question (live-plan never calls a provider during plan), not fixed here. Root cause C (NOT #313 - a distinct, newly-found, likely-fixable gap #308's own fix exposed, 4 sites): each.value.enable_cloudwatch_logging/create_cloudwatch_log_group inside module.container_definition, both literal booleans in the caller's own object literal but refused because each.value is treated as one opaque blob instead of projected to the referenced field - filed as #315, not attempted. These cascade to 177 \"Unable to compute static value\" and 6 \"Unresolvable identity\" sites (traced: aws_appautoscaling_policy reads aws_appautoscaling_target's resource_id, itself one of C's failures). Re-verified 2026-08-19 (be6b1096ba/b7bbff04a6) against everything landed overnight (#313's data-source half, #315, #321/#324, #323, #325): root cause A (0 sites, confirmed fixed) and root cause C (0 sites, #315's each.value projection fix confirmed) are both gone. Only root cause B remains (module.ecs_cluster.arn, a Computed attribute crossing a module-output boundary), now cascading to 1 \"Unable to compute static value\" plus 6 \"Unresolvable identity\" sites (aws_appautoscaling_target/aws_appautoscaling_policy chain) - down from 236 diagnostics to 8. This is the same already-acknowledged Computed-attribute/module-output architecture question corpus-rds-complete-postgres and corpus-security-group-complete are also blocked on; no action taken here, no new issue filed." @@ -643,16 +653,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -661,27 +671,29 @@ "exit_code": 0, "detail": { "cold_deploy": "54 resources, genuinely cold, genuinely unmarked", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and count_test[0] kept BOTH its live id (sg-2190fcd004e71e1a2) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) across it, read back through the AWS CLI rather than choudoufu's own report; the destroyed group (sg-3ff9bb7024a1a3cac) is genuinely gone from the account; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a NEW object under a NEW GroupId (sg-06d58eb0790723d87, was sg-3ff9bb7024a1a3cac) carrying aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (G0): plain terraform, its own working directory, state and VPC, standing the IDENTICAL 2-instance count block up for real against the same idle floci account, shows the identical shape - destroy the higher index only (count_test[1]=sg-0fa8f370fcd242978), create it back under a new id (sg-0260527e0d0cd1063), count_test[0]=sg-5186dbede2e418faf unchanged both times - and is torn down again before choudoufu's own side runs. Synthetic block, and why: terraform-aws-eks v9.0.0 has no scalable count knob to drive - every count in the module is a boolean create toggle (var.create_eks ? 1 : 0 and siblings), and the one length-driven knob (local.worker_group_count) drives only aws_autoscaling_group, aws_launch_configuration and aws_iam_role_policy_attachment, all untaggable (no tags argument in the provider schema, so no marker surface for this stage's identity assertion), and dropping a worker group also rewrites kubernetes_config_map.aws_auth, which aggregates every worker role, so the plan would carry changes alongside the destroy. aws_security_group.count_test is the sanctioned fallback per live/GAUNTLET.md #8, of a type this estate already exercises three times, at an address nothing else in the config names. BREAK_COUNT=1 confirms the check is load-bearing: asserting the WRONG instance (count_test[0], the survivor) was destroyed makes this stage report fail.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and count_test[0] kept BOTH its live id (sg-536bd82e1ae8b011a) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) across it, read back through the AWS CLI rather than choudoufu's own report; the destroyed group (sg-405bd41502e46a151) is genuinely gone from the account; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a NEW object under a NEW GroupId (sg-6af012ebcadc1cf6a, was sg-405bd41502e46a151) carrying aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; the next plan is empty. Stock oracle (G0): plain terraform, its own working directory, state and VPC, standing the IDENTICAL 2-instance count block up for real against the same idle floci account, shows the identical shape - destroy the higher index only (count_test[1]=sg-acdf0826e35bb8527), create it back under a new id (sg-c7f1d8d68584ba8f6), count_test[0]=sg-cf71071802083cb77 unchanged both times - and is torn down again before choudoufu's own side runs. Synthetic block, and why: terraform-aws-eks v9.0.0 has no scalable count knob to drive - every count in the module is a boolean create toggle (var.create_eks ? 1 : 0 and siblings), and the one length-driven knob (local.worker_group_count) drives only aws_autoscaling_group, aws_launch_configuration and aws_iam_role_policy_attachment, all untaggable (no tags argument in the provider schema, so no marker surface for this stage's identity assertion), and dropping a worker group also rewrites kubernetes_config_map.aws_auth, which aggregates every worker role, so the plan would carry changes alongside the destroy. aws_security_group.count_test is the sanctioned fallback per live/GAUNTLET.md #8, of a type this estate already exercises three times, at an address nothing else in the config names. BREAK_COUNT=1 confirms the check is load-bearing: asserting the WRONG instance (count_test[0], the survivor) was destroyed makes this stage report fail.", "day2_remove": "choudoufu: deleting aws_security_group.worker_group_mgmt_one's block (plus emptying the one argument that referenced it) proposed 2 resource action(s), address-for-address and action-for-action identical to stock's oracle on cold_deploy's own state; applied cleanly; the security group is genuinely gone from the live account, read via the AWS CLI, not choudoufu's own report; classifyOrphans did not withhold any destroy because no other aws_security_group.worker_group_mgmt_one block is declared anywhere in this config; the next plan is empty", "day2_rename": "moved block: aws_security_group.worker_group_mgmt_two renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_security_group.all_worker_mgmt renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing aws_security_group.worker_group_mgmt_two_renamed's ForceNew name_prefix argument proposed a forced replace at the same declared address (Plan: 3 to add, 1 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-17fb7372c7f753b56) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-4655e85b6bf6f3f89 -> sg-17fb7372c7f753b56); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 1 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted all_worker_mgmt_renamed (Part D's own live-mv leg) and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on the gauntlet/mv-rekey branch, GitHub issue #412 (propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard); see this section's own header comment for the fix and corpus-autoscaling-complete's/corpus-ecs-fargate's matching ones in this same unit. This script was not re-run for #412, so this detail string still describes the pre-#412 run until this estate's next real run.", + "day2_replace": "choudoufu: changing aws_security_group.worker_group_mgmt_two_renamed's ForceNew name_prefix argument proposed a forced replace at the same declared address (Plan: 3 to add, 1 to change, 3 to destroy.), applied cleanly; the old security group is confirmed gone via the AWS CLI (InvalidGroup.NotFound) and the new group (sg-1b6782f1886852eac) carries the marker; the local record store's record at the same address now names the new object's id, not the destroyed one (sg-ca6711c0e80c37f96 -> sg-1b6782f1886852eac); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes replacing the security group at the same address (Plan: 3 to add, 1 to change, 3 to destroy., plan only, not applied - it shares floci's account with $ADOPTED); BREAK=replace confirms a manufactured marker collision is reported loudly (\"Two live resources claiming one slot\") rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names; also scope note: the section originally targeted all_worker_mgmt_renamed (Part D's own live-mv leg) and found a genuine, separate defect (mv.go's propagateModuleRename skipped MoveRecord for a same-module live-mv rename, leaving the local record stale even though the marker moved correctly) - FIXED on the gauntlet/mv-rekey branch, GitHub issue #412 (propagateModuleRename now calls MoveRecord unconditionally for the renamed resource's own key before the moduleRenameBoundary guard); see this section's own header comment for the fix and corpus-autoscaling-complete's/corpus-ecs-fargate's matching ones in this same unit. This script was not re-run for #412, so this detail string still describes the pre-#412 run until this estate's next real run.", "drift_reconverge": "one object tampered (VPC's Name tag), plan proposed fixing exactly module.vpc.aws_vpc.this[0], apply changed 1 and the Name tag reconverged", "greenfield": "54 resources from nothing, cluster marker verified via the AWS CLI, 54 records under the implied local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on cluster status/version, ASG count/desired-capacities, and cluster-owned security-group count", "migrate": "25 of 54 resource instances stamped, 25 of 25 confirmed via the AWS CLI; 5 record-backed instances seeded into the implied local record store (#364)", + "plan_approval": "one argument edited (aws_security_group.worker_group_mgmt_one gains tags = { Reviewed = \"yes\" }), \"plan -out=approved.tfplan\" wrote a 82147-byte stock-format plan file whose whole change set is one update on aws_security_group.worker_group_mgmt_one; the world then moved out of band (the VPC vpc-328dc0c5's Name tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.vpc.aws_vpc.this[0] and the live vpc-328dc0c5 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - sg-7d8a2a280dbac8a69 still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-7d8a2a280dbac8a69 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The reviewed change was in-place from end to end: sg-7d8a2a280dbac8a69 kept its live id across the whole part, so PART D/E/F/G below still start from the objects STAGE 2 stamped. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); 25 tofu-estate-tagged objects before, 25 after", "test_plan": "live-plan runs to completion with ZERO Error diagnostics and reports \"No changes. Your infrastructure matches the configuration.\" - the record-backed worker launch configuration's enable_monitoring/root_block_device/user_data all now agree with the config's own desired value (lex00/floci#132 for the first two, configuredAttrsSeed's residue-record pre-read seed in internal/live/projection/build.go for the third)" }, - "duration_s": 844.9, + "duration_s": 920.1, "stage_seconds": { - "cold_deploy": 95, - "day2_count": 200, - "day2_remove": 61, - "day2_rename": 77, - "day2_replace": 60, - "drift_reconverge": 39, - "greenfield": 154, - "migrate": 119, - "test_apply": 21, + "cold_deploy": 83, + "day2_count": 214, + "day2_remove": 62, + "day2_rename": 73, + "day2_replace": 58, + "drift_reconverge": 41, + "greenfield": 155, + "migrate": 93, + "plan_approval": 100, + "test_apply": 22, "test_plan": 17 } }, @@ -707,16 +719,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "cb5ae2009fb85fdafd76126bb1842144855eb9d7", - "date": "2026-09-06T04:29:56Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -732,20 +744,22 @@ "drift_reconverge": "VPC Name tag tampered out of band, exactly 1 object proposed and reconverged, marker survived the incremental tag update", "greenfield": "10 resources from nothing (1 vpc, 3 subnets, 1 igw, 1 route table, 3 untaggable associations, 1 dynamodb table), VPC marker verified via the AWS CLI, 10 records in the local record store (#364 A2, one per managed instance), replan empty, stock oracle in its own namespace matches structurally on vpc/subnets/igw/route-table/dynamodb-table", "migrate": "7 of 10 verified and stamped, 0 failed, 3 correctly UNTAGGABLE; markers read back via the AWS CLI", + "plan_approval": "one argument edited (module.networking's internet gateway gains a Reviewed=yes tag; an in-place update, never a replace, because PART G below re-reads this estate's subnet ids by value), \"plan -out=approved.tfplan\" wrote a 12570-byte stock-format plan file whose whole change set is one update on module.networking.aws_internet_gateway.main; the world then moved out of band (vpc-b4200bde's Name tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.networking.aws_vpc.main and the live vpc-b4200bde it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - igw-676f386d still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and igw-676f386d read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The revert is proven byte-for-byte against the corpus pin (copy_modules' own diff, re-run) and the three subnets PART G re-reads are still the three cold_deploy minted. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 7, no state file", "test_plan": "no changes; VPC and table markers unchanged, all three untaggable associations resolved by their composite identity" }, - "duration_s": 146.6, + "duration_s": 142.4, "stage_seconds": { - "cold_deploy": 29, - "day2_count": 14, - "day2_remove": 7, + "cold_deploy": 12, + "day2_count": 13, + "day2_remove": 6, "day2_rename": 10, "day2_replace": 13, - "drift_reconverge": 4, - "greenfield": 27, - "migrate": 36, - "test_apply": 3, + "drift_reconverge": 6, + "greenfield": 26, + "migrate": 39, + "plan_approval": 12, + "test_apply": 2, "test_plan": 3 } }, @@ -771,16 +785,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "d72960cdc31f4e7d90f61088920379fc29b3999c", - "date": "2026-09-06T05:35:09Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -789,28 +803,30 @@ "exit_code": 0, "detail": { "cold_deploy": "6 resource instances added, 0 already tofu-estate-marked before migration", - "day2_count": "choudoufu: scaling aws_iam_role.count_test from 2 to 1 proposed exactly \"count_test[1] will be destroyed\" (0 add, 0 change, 1 destroy) and applied it, leaving count_test[0]'s server-minted RoleId (AROAPDYSMRCJISNRTSLK), its CreateDate and its tofu-address=aws_iam_role.count_test:0 marker all unchanged, and tombstoning count_test[1]'s local record (has tombstone, no identity - the #398-guard shape); scaling back from 1 to 2 proposed exactly \"count_test[1] will be created\" (1 add, 0 change, 0 destroy) and brought it back under the SAME deterministic name (giantswarm-crossplane-count-test-1) with a NEW RoleId (AROAKLXLOOPVVBJ8L4P2 -> AROAX6KFELD5P9JCWZF5) and a new CreateDate, re-marked aws_iam_role.count_test:1 and re-identified in the record store, while count_test[0] stayed untouched throughout; the next plan is empty. Every identity here is read back through the AWS CLI and the local record store, never through choudoufu's own report, and the destroy witness is the RoleId rather than the name or the ARN because both of those are deterministic from configuration and come back identical - confirmed against floci directly, no tofu in the loop, before the assertions were written. Stock oracle (G-ORACLE): real tofu standing the IDENTICAL count block up in the idle greenfield-oracle account showed the identical shape - destroy the higher index only, create the higher index back under the same name with a new RoleId (AROA4HHSNGXJQD7KLPND -> AROA9QH5CL9MJ5020J83), the lower index's RoleId and CreateDate unchanged both times - and the two sides' normalised action sets are compared literally, not just described. Synthetic block, per live/GAUNTLET.md #8's sanctioned fallback: the pinned crossplane module declares no count at all and its only two for_each knobs (aws_iam_role_policy.additional_policies over var.additional_policies, aws_iam_role_policy_attachment.additional_policy_attachments over toset(var.additional_policies_arns)) are both UNTAGGABLE types that carry no marker to keep an identity in, the second provably resolving to zero instances, and the first's inline-policy set is additionally policed by aws_iam_role_policies_exclusive in the same module; aws_iam_role.count_test reuses a type this estate already exercises and lives in its own day2_count.tofu beside the estate's root wiring ($ESTATE), so the vendored module stays byte-identical. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, proving the which-instance assertion is load-bearing.", + "day2_count": "choudoufu: scaling aws_iam_role.count_test from 2 to 1 proposed exactly \"count_test[1] will be destroyed\" (0 add, 0 change, 1 destroy) and applied it, leaving count_test[0]'s server-minted RoleId (AROAWKJGZCMEJKQD17GA), its CreateDate and its tofu-address=aws_iam_role.count_test:0 marker all unchanged, and tombstoning count_test[1]'s local record (has tombstone, no identity - the #398-guard shape); scaling back from 1 to 2 proposed exactly \"count_test[1] will be created\" (1 add, 0 change, 0 destroy) and brought it back under the SAME deterministic name (giantswarm-crossplane-count-test-1) with a NEW RoleId (AROAN4V2DL70N90344J2 -> AROAIQ2RZND9EW25FEJU) and a new CreateDate, re-marked aws_iam_role.count_test:1 and re-identified in the record store, while count_test[0] stayed untouched throughout; the next plan is empty. Every identity here is read back through the AWS CLI and the local record store, never through choudoufu's own report, and the destroy witness is the RoleId rather than the name or the ARN because both of those are deterministic from configuration and come back identical - confirmed against floci directly, no tofu in the loop, before the assertions were written. Stock oracle (G-ORACLE): real tofu standing the IDENTICAL count block up in the idle greenfield-oracle account showed the identical shape - destroy the higher index only, create the higher index back under the same name with a new RoleId (AROAS5MQ3SZOZQY5MORH -> AROAQELS5MR4KB0L9WAG), the lower index's RoleId and CreateDate unchanged both times - and the two sides' normalised action sets are compared literally, not just described. Synthetic block, per live/GAUNTLET.md #8's sanctioned fallback: the pinned crossplane module declares no count at all and its only two for_each knobs (aws_iam_role_policy.additional_policies over var.additional_policies, aws_iam_role_policy_attachment.additional_policy_attachments over toset(var.additional_policies_arns)) are both UNTAGGABLE types that carry no marker to keep an identity in, the second provably resolving to zero instances, and the first's inline-policy set is additionally policed by aws_iam_role_policies_exclusive in the same module; aws_iam_role.count_test reuses a type this estate already exercises and lives in its own day2_count.tofu beside the estate's root wiring ($ESTATE), so the vendored module stays byte-identical. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, proving the which-instance assertion is load-bearing.", "day2_remove": "choudoufu: deleting module.crossplane_final's block proposed 6 resource action(s), address-for-address and action-for-action identical to stock's oracle on cold_deploy's own state; applied cleanly; the role is genuinely gone from the live account (get-role now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report); classifyOrphans did not withhold any destroy because no other module.crossplane* block is declared anywhere in this config; the next plan is empty", "day2_rename": "moved block: module.crossplane renamed to .crossplane_renamed with zero churn (0 add, 2 change, 0 destroy - role and policy), markers rewritten in place; live-mv: .crossplane_renamed renamed to .crossplane_final with zero churn, both markers rewritten in place (one live-mv call per taggable object); stock oracle over the same chained module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", "day2_replace": "choudoufu: changing module.crossplane_final's ForceNew installation_name argument proposed a 6 add / 0 change / 6 destroy cascade with the role and the managed policy each explicitly named 'must be replaced' at their same declared addresses, applied cleanly; the old role (giantswarm-gsprereqs-crossplane) is confirmed gone and the new role (giantswarm-gsprereqs-v2-crossplane) carries the marker, both via the AWS CLI; the local record store's record at the role's address now names the new role, not the destroyed one (giantswarm-gsprereqs-crossplane -> giantswarm-gsprereqs-v2-crossplane); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes an equal add/destroy cascade (>=2) with role and policy both replaced at the same addresses (plan only, not applied - it shares floci's account with $ESTATE); BREAK=replace confirms a manufactured marker collision is reported loudly (a named 'Live resource displaced from the address it is marked for' warning, the scalar-resource shape) rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one.", "drift_reconverge": "role's installation tag tampered, exactly the IAM role proposed and reconciled, apply changed 1, tag reads back as configured", "greenfield": "6 resources from nothing (role, managed policy, 4 untaggable), role marker verified via the AWS CLI, 6 records in the local record store (#364 A2, one per managed instance), replan empty, stock oracle in its own namespace matches structurally on the role and the managed policy", "migrate": "2 of 6 stamped (role, managed policy), 4 untaggable skipped, module's own tags survived the stamp", + "plan_approval": "one argument edited (additional_policies[\"extra-tagging\"] widened to allow ec2:DeleteTags as well), \"plan -out=approved.tfplan\" wrote a 10347-byte stock-format plan file whose whole change set is one update on module.crossplane.aws_iam_role_policy.additional_inline_policies[\"extra-tagging\"]; the world then moved out of band (giantswarm-gsprereqs-crossplane's installation tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.crossplane.aws_iam_role.giantswarm_crossplane_role and the live giantswarm-gsprereqs-crossplane it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - the inline policy read back through the AWS CLI still allowed only ec2:CreateTags, which is stronger evidence than the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the inline policy read back allowing ec2:DeleteTags, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, both exclusive sets unchanged", "test_plan": "live-plan empty, role/policy tofu-address unchanged, both *_exclusive resources re-derived by value" }, - "duration_s": 162.7, + "duration_s": 113.3, "stage_seconds": { - "cold_deploy": 39, - "day2_count": 39, - "day2_remove": 7, - "day2_rename": 11, - "day2_replace": 10, - "drift_reconverge": 5, - "greenfield": 20, - "migrate": 24, - "test_apply": 4, - "test_plan": 4 + "cold_deploy": 5, + "day2_count": 31, + "day2_remove": 5, + "day2_rename": 10, + "day2_replace": 7, + "drift_reconverge": 4, + "greenfield": 12, + "migrate": 23, + "plan_approval": 11, + "test_apply": 3, + "test_plan": 2 } }, "notes": "Landed 2026-08-19 (crossing 0bd3ac80b7, merge 9fa0141294), the seventh OpenTofu-native estate and the first from a commercial vendor's production repository rather than a module registry, personal monorepo, or single-maintainer accelerator - Giant Swarm GmbH's own customer-facing account-prep for their managed Kubernetes offering. OpenTofu-native evidence, three independent kinds: README's opening sentence and directory index both say OpenTofu with no compatibility hedge; the CI workflow is named 'OpenTofu checks', installs via opentofu/setup-opentofu, and never mentions terraform; the crossed crossplane/ directory is genuinely .tofu-suffixed throughout (providers.tofu, role.tofu, variables.tofu), the file-level standard only the hongbomiao slices had met before (overture-tiles and xancloud-iac are both plain .tf). Scoped to crossplane/ specifically: self-contained (no remote state, no live EKS/OIDC dependency, its only data source is aws_partition which makes no API call), the other five directories in the repo excluded with stated reasons (three plain-.tf same-shape, one an account-singleton quota table, one a wrapper that only calls the others). Real run, rc=0, 548s. cold_deploy: PASS, plain tofu apply, 6 resources added, 0 pre-existing tofu-estate tags, the toset()-keyed for_each on additional_policy_attachments confirmed resolving to zero instances. migrate: PASS - 2 of 6 eligible (UNTAGGABLE 2, UNADMITTED_TYPE 2, DRIFTED 2), -approve 2 newly stamped 0 failed, both markers re-verified directly through the AWS CLI including that the module's own installation tag survived the stamp. test_plan: BLOCKED for real at exactly 2 sites, the plan's entire diagnostic surface - both Rule: unadmitted-type, on aws_iam_role_policy_attachments_exclusive and aws_iam_role_policies_exclusive, no other rule firing anywhere in the estate. A control stage (3b, not counted toward stage 4/5) cut exactly those two resource blocks and drove the rest of the pipeline for real: control test plan EMPTY, control test apply a genuine no-op (2 objects before and after), control drift-and-reconverge fixed exactly one mutated object - proving the estate's only real block is those two types, not routing around anything. Both negative controls (BREAK=1 at the stage-2 identity assertion, BREAK_STAGE3=1 expecting 3 refusal sites where the real count is 2) verified failing in real full runs. Filed as INTENTIUS/choudoufu#334: both unadmitted types have the identical import-grammar shape (single-argument, no-separator) to aws_vpc_security_group_rules_exclusive, which #307 already admitted via row-gen's tryGrammarComposite at 64cac28120, and carry the same no-CFN-counterpart mapping-gen overlay as that admitted twin - a worked ADMIT-class precedent, not attempted here; the one recorded difference (force_new on the admitted twin's security_group_id, absent on either new type's role_name) is flagged as not obviously the gate since tryGrammarComposite's single-argument branch reads no force_new field, and why row-gen's own proposal for these two is currently absent from ratified.json is the fix's first open question. No Go code touched. justfile gained demo-corpus-giantswarm-crossplane; live/corpus-manifest.json gained the pin; HANDOFF.md section 3 updated. Follow-up pass 2026-08-19/20 (#334 fixed and merged, 37957d873c/a6627543c4/20cc1774d6): the issue's own open question resolved the OPPOSITE way from what it suspected - row-gen was never declining to propose these two rows; it proposes both, byte-identical in shape to the admitted aws_vpc_security_group_rules_exclusive twin, and nobody had ever ratified the proposal. Fix is a ratified.json entry, no code change - reach is exactly these two types, stated plainly rather than dressed up as a generalization. The real finding: 316 types row-gen proposes sit unratified, 166 under this exact rule (including every other *_exclusive family member) - a ratification backlog, not a generator defect, and clearing it is a maintainer-scale call since every ratified row is a claim that touches live infrastructure. FIVE OF FIVE, run for real on the rebased tree, rc=0, 250s: stage 3 now an empty plan (identities asserted by value - the two exclusive resources carry no marker, being untaggable, so the assertion is the live content each enforces, e.g. attached policy ARN and inline-policy name); stage 4 a genuine no-op (2 tagged objects before and after, both exclusive sets independently re-read afterward so a wrongly-reconciled enforcer would be caught); stage 5 one out-of-band mutation, exactly one object proposed and fixed. Three real negative controls (BREAK=1, BREAK_STAGE3=1, BREAK_STAGE5=1) each confirmed failing at the right point. Script rewritten: stage 3 now asserts a pass instead of hard-failing by design, the 3b control retired with the block it existed to control for. Separately found, not yet fixed: every crossing script that runs more than one `terraform init` pays a real ~320s tax per extra init, because the shared plugin cache records no checksums and a directory with no `.terraform.lock.hcl` re-downloads the whole provider to compute them - seeding the lock file from stage 1's own init cuts this to ~1s and this estate's full run from several failed 10-minute-cap attempts to 250s total. Worth a sweep across every multi-init script here." @@ -835,16 +851,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -860,21 +876,23 @@ "drift_reconverge": "the plan proposed fixing 1 object(s) after the out-of-band tag mutation: module.s3_bucket_hm_harbor.aws_s3_bucket.main", "greenfield": "3 resources from nothing (bucket, user, untaggable inline policy), markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared", "migrate": "2 of 3 stamped (bucket, user), 1 UNTAGGABLE (inline policy); bucket hongbomiao-harbor-crossing-hm-harbor -> tofu-address=module.s3_bucket_hm_harbor.aws_s3_bucket.main, user hongbomiao-harbor-crossing-hm-harbor-user -> tofu-address=module.harbor_iam_user.aws_iam_user.hm_harbor_iam_user", + "plan_approval": "one argument edited (harbor_iam_user's common_tags gain Reviewed=yes), \"plan -out=approved.tfplan\" wrote a 8658-byte stock-format plan file whose whole change set is one update on module.harbor_iam_user.aws_iam_user.hm_harbor_iam_user; the world then moved out of band (hongbomiao-harbor-crossing-hm-harbor's hm_team tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.s3_bucket_hm_harbor.aws_s3_bucket.main and the live hongbomiao-harbor-crossing-hm-harbor it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - hongbomiao-harbor-crossing-hm-harbor-user still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and hongbomiao-harbor-crossing-hm-harbor-user read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 2 objects before, 2 after, no state file either time", "test_plan": "empty plan; identity re-check: bucket and user tofu-address unchanged, inline policy's resource ARN still matches the configuration" }, - "duration_s": 208.6, + "duration_s": 198.2, "stage_seconds": { - "cold_deploy": 27, - "day2_count": 43, + "cold_deploy": 15, + "day2_count": 45, "day2_remove": 7, - "day2_rename": 9, - "day2_replace": 6, + "day2_rename": 8, + "day2_replace": 7, "drift_reconverge": 5, - "greenfield": 88, - "migrate": 18, - "test_apply": 3, - "test_plan": 2 + "greenfield": 78, + "migrate": 17, + "plan_approval": 11, + "test_apply": 2, + "test_plan": 3 } }, "notes": "Landed 2026-08-19, the sixth estate in the OpenTofu-native lane and the fourth to clear all five stages. Sourced per HANDOFF's own suggestion to scope a second (here, third) disjoint slice of the already-crossed hongbomiao monorepo before a fresh search. Surveyed every remaining AWS environment: network/main.tofu is pure data sources (nothing to migrate); kubernetes/main.tofu builds a full terraform-aws-modules/eks cluster and every IAM module in it but one (velero_iam_role, mimir_iam_role, loki_iam_role, tempo_iam_role, label_studio_iam_role, etc., 15 total) takes amazon_eks_cluster_oidc_provider(_arn) from that same cluster - the same scope/risk class as the terraform-popular lane's already-blocked terraform-aws-eks examples/basic crossing. The one exception, the \"Harbor\" section (S3 bucket + IAM user + inline user policy), needs no EKS cluster, no OIDC provider, no remote state at all - self-contained like storage's own scoped slice. Nebius/Cloudflare/Snowflake environments confirmed to still exist and be real, actively-maintained infrastructure, but target non-AWS clouds floci cannot emulate. Crosses aws_iam_user/aws_iam_user_policy, a genuinely different resource pair from Labelbox's aws_iam_role/aws_iam_role_policy - both already-ratified DefaultTable rows, no schema-fallback warning. All five stages verified for real against a live floci container: cold_deploy (tofu apply, 3 resources created, confirmed 0 pre-existing tofu-estate tags), migrate (live-import verified 2 of 3 eligible - bucket + user - 1 correctly UNTAGGABLE - the inline policy; markers re-read via AWS CLI matched: module.s3_bucket_hm_harbor.aws_s3_bucket.main, module.harbor_iam_user.aws_iam_user.hm_harbor_iam_user), test_plan (state deleted, live-plan empty, identities re-verified against the AWS CLI including the inline policy's resource ARN read directly off the live object), test_apply (genuine no-op, 2 tagged objects before and after), drift_reconverge (bucket tag tampered out of band, plan proposed fixing exactly that one object, reconverge apply restored it). BREAK=1 verified load-bearing, failing exactly at the stage-2 identity assertion. No floci or choudoufu gaps found - this crossing is clean. Merged to local main as ad2cf81cf3 (crossing itself: 0c4e16af6a); justfile gained recipe demo-corpus-hongbomiao-harbor (port 4728); no new live/corpus-manifest.json entry needed, reuses the existing hongbomiao pin." @@ -899,16 +917,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "d72960cdc31f4e7d90f61088920379fc29b3999c", - "date": "2026-09-06T05:35:09Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -924,21 +942,23 @@ "drift_reconverge": "bucket tag drifted; exactly module.amazon_s3_bucket_hm_labelbox.aws_s3_bucket.main proposed, applied (1 changed), reconverged to hongbomiao", "greenfield": "4 resources from nothing (bucket, CORS config, role, untaggable inline role policy), markers verified via the AWS CLI, 4 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared", "migrate": "2 of 4 stamped (2 skipped, untaggable), 0 failed; markers read back via the AWS CLI", + "plan_approval": "one argument edited (labelbox_iam_role's common_tags gain Reviewed=yes), \"plan -out=approved.tfplan\" wrote a 11141-byte stock-format plan file whose whole change set is one update on module.labelbox_iam_role.aws_iam_role.labelbox_iam_role; the world then moved out of band (hongbomiao-labelbox-crossing-hm-labelbox's hm_team tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.amazon_s3_bucket_hm_labelbox.aws_s3_bucket.main and the live hongbomiao-labelbox-crossing-hm-labelbox it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - LabelboxRole-hm-labelbox still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and LabelboxRole-hm-labelbox read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, no state file", "test_plan": "no resource change proposed; bucket and role tofu-address unchanged, CORS origins and inline policy resource match config" }, - "duration_s": 221, + "duration_s": 177.5, "stage_seconds": { - "cold_deploy": 55, - "day2_count": 23, + "cold_deploy": 16, + "day2_count": 24, "day2_remove": 9, "day2_rename": 10, - "day2_replace": 9, - "drift_reconverge": 7, - "greenfield": 55, - "migrate": 45, - "test_apply": 4, - "test_plan": 4 + "day2_replace": 8, + "drift_reconverge": 6, + "greenfield": 48, + "migrate": 37, + "plan_approval": 14, + "test_apply": 3, + "test_plan": 3 } }, "notes": "Landed 2026-08-18 as the second estate in the OpenTofu-native lane and the first to clear all five stages there. Stronger OpenTofu-native evidence than corpus-sumaform-aws (which only describes itself as OpenTofu-native but ships plain .tf once its .example template is copied in): every file under infrastructure/opentofu/ genuinely uses the .tofu extension, its own justfile drives init/plan/apply/refresh/destroy exclusively via `tofu`, and common_tags carries \"hm_managed_by\" = \"opentofu\" - proven rather than asserted, since the crossing script's own stock terraform init against this estate reports \"The directory has no Terraform configuration files.\" Scoped to the self-contained \"Labelbox\" slice (S3 bucket, its CORS configuration, an IAM role with an inline S3-read policy - three real leaf modules copied byte-identical from the pinned commit, diffed programmatically in the script) out of a much larger monorepo (AWS+Nebius+Cloudflare+Snowflake+EKS, cross-wired via terraform_remote_state) too large to stand up in one sitting - the same scoping convention corpus-sumaform-aws's module.base stand-in uses. All five stages verified for real: cold_deploy (tofu apply, 4 resources, confirmed unmarked via resourcegroupstaggingapi), migrate (live-import: \"2 of 4 resource instance(s) are eligible for stamping\" - 2 correctly UNTAGGABLE, the CORS config via provider-schema fallback and the inline policy via the generated table's composite ROLENAME:POLICYNAME identity; -approve stamped both taggable resources, markers verified directly via aws s3api get-bucket-tagging / aws iam list-role-tags), test_plan (state deleted, live-plan \"No changes\", identities re-checked against the AWS CLI including the two untaggable resources' own content - CORS AllowedOrigins, inline policy's Resource ARN - since they carry no tag to re-read), test_apply (genuine no-op, 2 tagged objects before and after), and drift_reconverge (the bucket's hm_team tag tampered out of band, plan proposed fixing exactly that object, apply reconverged it; BREAK=1 verified load-bearing for both stage 2's identity check and stage 5's single-object assertion, tested in isolation for stage 5 per the corpus-vpc-complete convention since the shared BREAK var fails fast at stage 2 otherwise). Two non-blocking findings documented in the script's own header rather than routed around: aws_s3_bucket/aws_iam_role report DRIFTED during verification from AWS's own deprecated cors_rule/inline_policy shadow attributes reflecting a sibling resource created after the state snapshot (harmless, resolves by plan time), and the schema-admitted aws_s3_bucket_cors_configuration triggers the already-documented \"Resource type has no orphan recovery\" warning (live/LIMITATIONS.md, not a new gap). No choudoufu or floci gaps found - nothing filed. Merged to local main as c7fb650f4c (fix itself: 30577f6a56); justfile gained recipe demo-corpus-hongbomiao-labelbox; live/corpus-manifest.json gained the pin (reproducibility only, same convention as the sumaform entry - contributes nothing to a corpus-gen number)." @@ -963,16 +983,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -984,25 +1004,27 @@ "day2_count": "choudoufu: scaling aws_s3_bucket.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live CreationDate and tofu-address marker unchanged and tombstoning count_test[1]'s local record (has tombstone, no identity - the #398-guard shape); scaling back from 1 to 2 created exactly count_test[1] under the SAME bucket name (deterministic) but a NEW CreationDate (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied for real in the idle greenfield real-leg account, shows the identical shape: destroy the higher index only, create the higher index back under the same bucket name but a new CreationDate, the lower index's CreationDate unchanged both times", "day2_remove": "choudoufu: deleting module.kafka_kms_key_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy - the untaggable alias and its taggable parent key), applied cleanly (0 added, 0 changed, 2 destroyed) in an order the cloud accepted, the key is genuinely PendingDeletion and the alias is gone (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly two destroys for the same objects", "day2_rename": "moved block: module.hm_production_bucket renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.kafka_kms_key renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.kafka_kms_key_renamed's aws_kms_key_name argument proposed exactly one forced replace at the same declared address (the untaggable, client-named alias - 1 add, 1 change, 1 destroy overall) plus one in-place tag update on the taggable key itself, applied cleanly; the old alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key) is confirmed gone and the new alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2) points at the SAME key (e2ea5441-c9cd-4f92-85b7-4207a2c8c29a, read via the AWS CLI) - the key was never replaced; the local record store's record at the alias's address now names the new alias, not the destroyed one (alias/hongbomiao-storage-crossing-hm-kafka-kms-key -> alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2), while the key's own record at its own address is unchanged; the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace (the alias) plus one in-place key update. Scope notes: (1) this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see corpus-sqs-basic's own PART F; (2) BREAK=replace's marker-collision control is not exercised here - aws_kms_alias is untaggable and resolved by its own name, with no marker to plant a collision on, so that control's load-bearing-ness is proven instead by corpus-evoteum-modules and corpus-giantswarm-crossplane's own PART F sections against the taggable shape.", + "day2_replace": "choudoufu: changing module.kafka_kms_key_renamed's aws_kms_key_name argument proposed exactly one forced replace at the same declared address (the untaggable, client-named alias - 1 add, 1 change, 1 destroy overall) plus one in-place tag update on the taggable key itself, applied cleanly; the old alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key) is confirmed gone and the new alias (alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2) points at the SAME key (b1f75b83-386c-46e8-86bb-ef14f4299ed7, read via the AWS CLI) - the key was never replaced; the local record store's record at the alias's address now names the new alias, not the destroyed one (alias/hongbomiao-storage-crossing-hm-kafka-kms-key -> alias/hongbomiao-storage-crossing-hm-kafka-kms-key-v2), while the key's own record at its own address is unchanged; the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace (the alias) plus one in-place key update. Scope notes: (1) this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see corpus-sqs-basic's own PART F; (2) BREAK=replace's marker-collision control is not exercised here - aws_kms_alias is untaggable and resolved by its own name, with no marker to plant a collision on, so that control's load-bearing-ness is proven instead by corpus-evoteum-modules and corpus-giantswarm-crossplane's own PART F sections against the taggable shape.", "drift_reconverge": "the plan proposed fixing 1 object(s) after the out-of-band tag mutation: module.s3_bucket_iot_data.aws_s3_bucket.main", "greenfield": "4 resources from nothing (2 buckets under aws.production, KMS key and untaggable alias under the default aws provider), markers verified via the AWS CLI, 4 records in the local record store (#364 A2), replan empty both with and without the local record store, all objects match stock's cold-deploy container (STAGE 1, untouched) object by object per provider namespace, marker tags never compared", - "migrate": "3 of 4 stamped (2 buckets, KMS key), 1 UNTAGGABLE (KMS alias); bucket hongbomiao-storage-crossing-hm-production -> tofu-address=module.hm_production_bucket.aws_s3_bucket.main, bucket hongbomiao-storage-crossing-hm-iot-data -> tofu-address=module.s3_bucket_iot_data.aws_s3_bucket.main, key e2ea5441-c9cd-4f92-85b7-4207a2c8c29a -> tofu-address=module.kafka_kms_key.aws_kms_key.main", + "migrate": "3 of 4 stamped (2 buckets, KMS key), 1 UNTAGGABLE (KMS alias); bucket hongbomiao-storage-crossing-hm-production -> tofu-address=module.hm_production_bucket.aws_s3_bucket.main, bucket hongbomiao-storage-crossing-hm-iot-data -> tofu-address=module.s3_bucket_iot_data.aws_s3_bucket.main, key b1f75b83-386c-46e8-86bb-ef14f4299ed7 -> tofu-address=module.kafka_kms_key.aws_kms_key.main", + "plan_approval": "one argument edited (hm_production_bucket's common_tags gain Reviewed=yes), \"plan -out=approved.tfplan\" wrote a 10486-byte stock-format plan file whose whole change set is one update on module.hm_production_bucket.aws_s3_bucket.main; the world then moved out of band (hongbomiao-storage-crossing-hm-iot-data's hm_team tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.s3_bucket_iot_data.aws_s3_bucket.main and the live hongbomiao-storage-crossing-hm-iot-data it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - hongbomiao-storage-crossing-hm-production still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and hongbomiao-storage-crossing-hm-production read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 3 objects before, 3 after, no state file either time", "test_plan": "empty plan; identity re-check: both buckets' and the key's tofu-address unchanged, KMS alias still points at the same key" }, - "duration_s": 210.7, + "duration_s": 214.6, "stage_seconds": { - "cold_deploy": 33, + "cold_deploy": 20, "day2_count": 24, - "day2_remove": 8, - "day2_rename": 15, + "day2_remove": 7, + "day2_rename": 14, "day2_replace": 12, - "drift_reconverge": 6, - "greenfield": 63, - "migrate": 43, + "drift_reconverge": 5, + "greenfield": 58, + "migrate": 52, + "plan_approval": 15, "test_apply": 3, - "test_plan": 3 + "test_plan": 4 } }, "notes": "Landed 2026-08-18, the third estate in the OpenTofu-native lane and the second to clear all five stages, reusing corpus-hongbomiao-labelbox's already-pinned commit rather than a fresh sourcing search (the repo's OpenTofu-native bona fides and the pinned commit's clone were already established by that crossing). Scoped after surveying every section of the monorepo's aws/general and aws/storage files via the GitHub API against the pinned commit, no clone needed for scouting: Kafka Manager, two Amazon EMR sections and AWS Batch all read another environment's terraform_remote_state (out of scope, same reason corpus-hongbomiao-labelbox's own scoping excluded them); Amazon SageMaker was ruled out with a real, confirmed floci gap - aws sagemaker create-notebook-instance against a live floci container returns \"UnknownOperationException: Operation CreateNotebookInstance is not supported by floci\", and the type has zero entries anywhere in live/floci-capabilities.json's Cloud Control sweep - documented in the script's header as evidence for whoever picks up SageMaker next, not filed as an issue since it was routed around rather than blocking anything. The real candidate: aws/storage/main.tofu's first three module calls (hm_production_bucket, kafka_kms_key, s3_bucket_iot_data) read no remote state at all, unlike everything after them in that file - two amazon_s3_bucket module calls plus one aws_kms_key module call (aws_kms_key + aws_kms_alias). All five stages verified for real against a live floci container: cold_deploy (tofu apply, \"4 added, 0 changed, 0 destroyed\", confirmed 0 objects pre-tagged), migrate (live-import: \"3 of 4 resource instance(s) are eligible for stamping\", 1 UNTAGGABLE - the KMS alias; -approve: \"3 resource(s) newly stamped, 0 already stamped, 0 failed, 1 skipped\"; markers for all three read back via raw AWS CLI matched exactly: module.hm_production_bucket.aws_s3_bucket.main, module.s3_bucket_iot_data.aws_s3_bucket.main, module.kafka_kms_key.aws_kms_key.main), test_plan (state deleted, live-plan \"No changes\", all three identities re-verified against the AWS CLI, including the untaggable KMS alias's live target), test_apply (genuine no-op, \"0 added, 0 changed, 0 destroyed\", object count unchanged at 3), and drift_reconverge (the IoT-data bucket's tag tampered out of band, plan proposed fixing exactly module.s3_bucket_iot_data.aws_s3_bucket.main and nothing else, reconverge apply changed exactly 1 resource). BREAK=1 verified load-bearing: correctly fails the stage-2 identity assertion (asserts the KMS key's tofu-address against a deliberately wrong resource name). No choudoufu gaps found beyond the SageMaker floci evidence above - nothing filed against this repo. Merged to local main as a720266bcc (fix itself: 3335f16893); justfile gained recipe demo-corpus-hongbomiao-storage (port 4725); no new live/corpus-manifest.json entry needed, reuses the existing hongbomiao pin." @@ -1027,16 +1049,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1048,25 +1070,27 @@ "day2_count": "choudoufu: scaling aws_iam_policy.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live arn and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under a NEW arn (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G0 stock oracle on the same 2-instance count block, applied fresh against the idle adopted-estate endpoint, shows the identical shape: destroy the higher index only, create the higher index back under a new arn, the lower index's arn unchanged both times. Synthetic block: this estate's only real count knob (aws_iam_policy.policy's count = var.create ? 1 : 0) is a boolean create toggle, not a scalable set - sanctioned fallback per live/GAUNTLET.md #8 and reference-ec2-vpc's own Part F.", "day2_remove": "choudoufu: deleting module.iam_policy_renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (iam get-policy on the old ARN now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (STAGE 1.5.5) also proposes exactly one destroy for the same object; classifyOrphans did not withhold the destroy even though module.iam_policy_renamed2's policy shares the same block key, because that surviving instance is bound, not unclaimed", "day2_rename": "moved block: module.iam_policy_from_data_source renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.iam_policy renamed with zero churn, marker rewritten in place (found and fixed live-mv's own missing issue #266 tag-index fallback and the arnJoinTable's missing iam:policy entry to get here); stock oracle over the same two-module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both ARNs unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.iam_policy_renamed2's ForceNew name_prefix argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old policy (arn:aws:iam::000000000000:policy/example-246b74803ee893de43f27cac43) is confirmed gone and the new policy (arn:aws:iam::000000000000:policy/example-v2-0f747c1a618bb2900d8fde9713) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's ARN, not the destroyed one (arn:aws:iam::000000000000:policy/example-246b74803ee893de43f27cac43 -> arn:aws:iam::000000000000:policy/example-v2-0f747c1a618bb2900d8fde9713); the next plan proposes no resource action; stock oracle on cold_deploy's own state (STAGE 1.5.6) also proposes exactly one replace at the same address (plan only, not applied). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one. BREAK=replace's manufactured marker collision IS now reported for this type (GitHub issue #411, fixed): 'Indistinguishable instances without per-instance markers' - GitHub issue #409, layered on top of #411's own fix, is why this is not corpus-sqs-basic's 'Two live resources claiming one slot' text despite the same shape - see PART F's own header and its BREAK=replace branch.", + "day2_replace": "choudoufu: changing module.iam_policy_renamed2's ForceNew name_prefix argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old policy (arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be) is confirmed gone and the new policy (arn:aws:iam::000000000000:policy/example-v2-96244c69014024ce6076c79bba) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's ARN, not the destroyed one (arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be -> arn:aws:iam::000000000000:policy/example-v2-96244c69014024ce6076c79bba); the next plan proposes no resource action; stock oracle on cold_deploy's own state (STAGE 1.5.6) also proposes exactly one replace at the same address (plan only, not applied). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-sqs-basic's matching one. BREAK=replace's manufactured marker collision IS now reported for this type (GitHub issue #411, fixed): 'Indistinguishable instances without per-instance markers' - GitHub issue #409, layered on top of #411's own fix, is why this is not corpus-sqs-basic's 'Two live resources claiming one slot' text despite the same shape - see PART F's own header and its BREAK=replace branch.", "drift_reconverge": "one object tampered (arn:aws:iam::000000000000:policy/example_from_data_source's Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag", "greenfield": "2 resources from nothing (both aws_iam_policy), markers verified via the AWS CLI, 2 records in the local record store (#364 A2), replan empty both with and without the local record store, both policies' documents and paths match stock's cold-deploy container (STAGE 1, untouched) object by object, marker tags never compared", "migrate": "2 of 2 stamped, both carrying tofu-slot=0/0 read back through IAM (choudoufu #372); Apply complete! Resources: 0 added, 0 changed, 0 destroyed. - nothing left to converge", + "plan_approval": "one argument edited (module.iam_policy's tags gain Reviewed=yes), \"plan -out=approved.tfplan\" wrote a 11133-byte stock-format plan file whose whole change set is one update on module.iam_policy.aws_iam_policy.policy[0]; the world then moved out of band (arn:aws:iam::000000000000:policy/example_from_data_source's Example tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.iam_policy_from_data_source.aws_iam_policy.policy[0] and the live arn:aws:iam::000000000000:policy/example_from_data_source it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:iam::000000000000:policy/example-5671de880c0141761d949a09be read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 2 objects before, 2 after, no state file either time", "test_plan": "no resource change proposed, nothing foreign; identity re-check (via the AWS CLI) both unchanged" }, - "duration_s": 177.6, + "duration_s": 166.1, "stage_seconds": { - "cold_deploy": 28, - "day2_count": 42, - "day2_remove": 8, + "cold_deploy": 17, + "day2_count": 40, + "day2_remove": 7, "day2_rename": 9, "day2_replace": 7, "drift_reconverge": 5, - "greenfield": 55, - "migrate": 18, + "greenfield": 45, + "migrate": 19, + "plan_approval": 12, "test_apply": 3, - "test_plan": 3 + "test_plan": 2 } }, "notes": "Upgraded from a real but pre-#274-pipeline predecessor script (choudoufu apply from a live block present from the start, delete state, replan empty twice) to the current five-stage shape, following corpus-vpc-complete/corpus-lambda-simple's structure. Verified for real in a fresh isolated worktree off local main (ff106e63a7), Docker/floci/AWS CLI throughout, not read from the predecessor's prior notes. All five stages pass cleanly: cold_deploy (plain terraform apply, \"Apply complete! Resources: 2 added\", confirmed 0 objects tagged before migration), migrate (live-import dry run verifies \"2 of 2 resource instance(s) are eligible for stamping\", -approve reports \"2 resource(s) newly stamped, 0 already stamped, 0 failed, 0 skipped\", both tofu-address/tofu-estate tags read directly through the AWS CLI: module.iam_policy.aws_iam_policy.policy:0 and module.iam_policy_from_data_source.aws_iam_policy.policy:0), test_plan (live-plan genuinely empty, both identities re-read unchanged after the state file's only copy was deleted), test_apply (\"0 added, 0 changed, 0 destroyed\", object count unchanged at 2), and drift_reconverge (one policy's Example tag tampered directly against floci, live-plan proposes fixing exactly that object, apply reconverges it to \"0 added, 1 changed, 0 destroyed\"). BREAK=1 verified twice, independently, against each stage it targets: run as committed it fails stage 3's identity check (expects the real policy's tofu-address on a module that was never created); run separately with stage 3's corruption disabled, it correctly fails stage 5 by tampering a second object and proving the \"exactly one object\" count assertion is load-bearing (both objects flagged, not silently 1). NEW FINDING, not previously documented in any real crossing that reached this deep: live-import -approve deliberately writes only tofu-estate and tofu-address, never tofu-slot (internal/live/stamp/doc.go's own \"tofu-slot comes in from outside\" - a slot is minted from a monotonic counter over the live set that a read-only, one-state-file view cannot compute). Both of this estate's aws_iam_policy resources declare count = var.create ? 1 : 0, exactly the shape that needs one, so the FIRST live-plan straight after live-import -approve is not empty - it proposes adding tofu-slot=\"0\" to both, and nothing else. Folded into stage 2 as one ordinary `choudoufu apply` (\"0 added, 2 changed, 0 destroyed\") before stage 3 is attempted; every replan after is genuinely empty. This is real, deliberate, already-documented product behavior, not a defect - but it will recur on any count-based resource crossing that reaches this far and had not yet been noticed in one that actually got here. Also caught and fixed while verifying: a self-authored bug where stage 5's negative drift assertion compared a live-plan diff header's address (bracket form, \"policy[0]\") against the escaped tag-value form (\"policy:0\") and could never have matched - a vacuous check that a stricter assertion in the sibling script (see corpus-iam-read-only-policy) surfaced; fixed here by keeping both forms as separate variables." @@ -1091,16 +1115,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1112,24 +1136,26 @@ "day2_count": "choudoufu: scaling aws_iam_policy.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live PolicyId and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under the SAME ARN (deterministic from name+path) but a NEW PolicyId (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the same 2-instance count block, applied fresh in the idle greenfield-oracle account, shows the identical shape: destroy the higher index only, create the higher index back under the same ARN but a new PolicyId, the lower index's PolicyId unchanged both times", "day2_remove": "choudoufu: deleting module.read_only_iam_policy_final's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (iam get-policy on the old ARN now returns NoSuchEntity, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; classifyOrphans did not withhold the destroy because no other aws_iam_policy.policy block anywhere in this config ever declares a real instance (count=0 on both remaining module calls)", "day2_rename": "moved block: module.read_only_iam_policy renamed to module.read_only_iam_policy_moved with zero churn (0 add, 1 change, 0 destroy), tofu-address marker rewritten in place; live-mv: module.read_only_iam_policy_moved renamed to module.read_only_iam_policy_final with zero churn, marker rewritten in place; stock oracle over the identical net rename on cold_deploy's own state also shows a true no-op (0 add, 0 change, 0 destroy, outputs unchanged in value); the live policy ARN unchanged throughout, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.read_only_iam_policy_final's ForceNew description argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-cb1412a333b8859faa022fa5d1) is confirmed gone and the new object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-0f29efba6ef044410ac1522988) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-cb1412a333b8859faa022fa5d1 -> arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-0f29efba6ef044410ac1522988); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", - "drift_reconverge": "one object tampered (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-cb1412a333b8859faa022fa5d1's Example tag), plan proposed fixing exactly module.read_only_iam_policy.aws_iam_policy.policy[0], apply changed 1 and reconverged the tag", + "day2_replace": "choudoufu: changing module.read_only_iam_policy_final's ForceNew description argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7) is confirmed gone and the new object (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-84581510d3c9ebbf3d8288dffa) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7 -> arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-84581510d3c9ebbf3d8288dffa); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", + "drift_reconverge": "one object tampered (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7's Example tag), plan proposed fixing exactly module.read_only_iam_policy.aws_iam_policy.policy[0], apply changed 1 and reconverged the tag", "greenfield": "1 resource from nothing, marker verified via the AWS CLI, 1 record in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (path, description, policy document)", "migrate": "1 of 1 stamped, carrying tofu-slot=0 read back through IAM (choudoufu #372); Apply complete! Resources: 0 added, 0 changed, 0 destroyed. - nothing left to converge", + "plan_approval": "one argument edited (aws_iam_policy.approval_probe's Reviewed tag, no -> yes), \"plan -out=approved.tfplan\" wrote a 27324-byte stock-format plan file whose whole change set is one update on aws_iam_policy.approval_probe; the world then moved out of band (arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7's Example tag, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.read_only_iam_policy.aws_iam_policy.policy[0] and the live arn:aws:iam::000000000000:policy/example/ex-iam-read-only-policy-61e508d427425c9882edfc38d7 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - arn:aws:iam::000000000000:policy/example/iam-ro-approval-probe still read Reviewed=no through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:iam::000000000000:policy/example/iam-ro-approval-probe read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The reviewed object is a self-contained synthetic aws_iam_policy.approval_probe (sanctioned fallback per live/GAUNTLET.md #8, same discipline as PART G's count_test) because this estate has exactly ONE real object and the leg needs two disjoint rows; it is created in P0 and destroyed in P5, and the module policy's ARN and Example tag are read back unchanged so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 1 objects before, 1 after, no state file either time", "test_plan": "no resource change proposed, nothing foreign; identity re-check (via the AWS CLI) unchanged" }, - "duration_s": 189.5, + "duration_s": 191.2, "stage_seconds": { - "cold_deploy": 25, + "cold_deploy": 15, "day2_count": 19, - "day2_remove": 7, - "day2_rename": 10, - "day2_replace": 7, + "day2_remove": 6, + "day2_rename": 9, + "day2_replace": 8, "drift_reconverge": 5, - "greenfield": 45, - "migrate": 65, - "test_apply": 3, + "greenfield": 41, + "migrate": 63, + "plan_approval": 19, + "test_apply": 2, "test_plan": 3 } }, @@ -1155,16 +1181,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1173,28 +1199,30 @@ "exit_code": 0, "detail": { "cold_deploy": "8 resources, genuinely cold, genuinely unmarked", - "day2_count": "synthetic count block, and the header says why: .corpus/lambda declares 38 count/for_each expressions and not one is scalable - 36 are boolean create toggles (count = ... ? 1 : 0), the for_each maps are empty in this example, and the only two numeric ones (number_of_policies, number_of_policy_jsons) are gated off by default and drive the untaggable aws_iam_role_policy_attachment, so turning either on would both change the estate every earlier stage measured and leave no marker to read back. So aws_iam_role.count_test (count = 2), a self-contained block of a type this estate already exercises, added after Part E's completed removal. choudoufu: scaling 2 -> 1 proposed exactly 0 add, 0 change, 1 destroy, destroying count_test[1] and never naming count_test[0]; after the apply lambda-simple-count-test-1 no longer answers GetRole at all (verified absence) while count_test[0] kept its server-minted RoleId AROAZY2RIPLGDLB3N0TZ and its tofu-address=aws_iam_role.count_test:0 marker, both read through the AWS CLI. Scaling 1 -> 2 proposed exactly 1 add, 0 change, 0 destroy and count_test[1] came back as a genuinely NEW object - RoleId AROAGXAJJPJ2ED9EO668 -> AROAVNKGVMUSSSKUZC9V under the same deterministic name, the witness an IAM role's ARN cannot provide since it is rebuilt from account plus name (the same reason this estate's own Lambda function ARN could not have witnessed it) - carrying tofu-address=aws_iam_role.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G0): the identical block stood up for real with plain terraform in its own working directory on the idle greenfield-oracle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new RoleId (AROACC4A5TI4EAICMRGS -> AROATTFBWT3X96T6C4AB), the lower index's RoleId (AROAVPYZM4YHIDNFQUC6) unchanged both times. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, so the assertion is load-bearing", + "day2_count": "synthetic count block, and the header says why: .corpus/lambda declares 38 count/for_each expressions and not one is scalable - 36 are boolean create toggles (count = ... ? 1 : 0), the for_each maps are empty in this example, and the only two numeric ones (number_of_policies, number_of_policy_jsons) are gated off by default and drive the untaggable aws_iam_role_policy_attachment, so turning either on would both change the estate every earlier stage measured and leave no marker to read back. So aws_iam_role.count_test (count = 2), a self-contained block of a type this estate already exercises, added after Part E's completed removal. choudoufu: scaling 2 -> 1 proposed exactly 0 add, 0 change, 1 destroy, destroying count_test[1] and never naming count_test[0]; after the apply lambda-simple-count-test-1 no longer answers GetRole at all (verified absence) while count_test[0] kept its server-minted RoleId AROAC7ERQHW7CEUA1EZL and its tofu-address=aws_iam_role.count_test:0 marker, both read through the AWS CLI. Scaling 1 -> 2 proposed exactly 1 add, 0 change, 0 destroy and count_test[1] came back as a genuinely NEW object - RoleId AROATSWACNJVMMY25Q6P -> AROACJAD70F33ERTS33F under the same deterministic name, the witness an IAM role's ARN cannot provide since it is rebuilt from account plus name (the same reason this estate's own Lambda function ARN could not have witnessed it) - carrying tofu-address=aws_iam_role.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G0): the identical block stood up for real with plain terraform in its own working directory on the idle greenfield-oracle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new RoleId (AROAQ945VI0K0RAZ648E -> AROA97JTCE39WS5SJGQN), the lower index's RoleId (AROAYT6M787L61HL0SJ3) unchanged both times. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, so the assertion is load-bearing", "day2_remove": "choudoufu: deleting module.lambda_function_final's block proposed 7 destroys (the function, the role, its inline aws_iam_role_policy.logs[0] CloudWatch Logs policy, and all three record-located children always; the log group's only when floci's GetResources happens to index it - a documented emulator gap, confirmed by reading logs:list-tags-for-resource directly against the same live object), applied cleanly, the function, the role and the inline log policy genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no further resource action; classifyOrphans did not withhold any destroy as a possible rename. WANT_DESTROY_COUNT moved from 5/6 to 6/7 in this same commit: the inline log policy was previously missing from this stage's own checklist entirely - a genuine leak (an untaggable IAM permission left behind on every destroy of this estate), not a stale assertion, fixed as part of the day2_replace unit that re-measured this stage", "day2_rename": "moved block: module.lambda_function renamed to module.lambda_function_moved with zero churn (0 add, 3 change, 0 destroy) across all seven of its stateful children, three taggable markers rewritten in place, three record-located children moved via their own per-resource moved blocks with zero diff, one config-derived child (aws_iam_role_policy.logs) needing none; stock oracle over the identical seven-resource move on cold_deploy's own state also shows zero churn beyond the module's own pre-existing null_resource.archive[0] package-timestamp noise (confirmed present on an unrelated baseline replan too); live-mv: module.lambda_function_moved renamed to module.lambda_function_final across all three taggable children (the function, the role, the log group), one call each, zero churn, markers rewritten in place - the internal/live/mv/mv.go materialize() RecordStore wiring gap (build.go:1676's \"Record-backed instance with no record store\") is fixed; all three live objects unchanged throughout, read via the AWS CLI; final replan is empty", - "day2_replace": "choudoufu: changing module.lambda_function_final's ForceNew logging_log_group-derived name proposed exactly one replace at the same declared address (the log group; -/+ destroy and then create) cascading into two expected in-place updates (the function's logging_config, the inline log policy's document) and nothing else beyond the module's own pre-existing null_resource.archive[0] package-timestamp noise; applied cleanly; the old object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/united-mantis-lambda-simple) is confirmed gone and the new object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/united-mantis-lambda-simple-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (/aws/lambda/united-mantis-lambda-simple -> /aws/lambda/united-mantis-lambda-simple-v2); the next plan proposes no resource action beyond the same known noise; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace plus the same in-place cascade (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", + "day2_replace": "choudoufu: changing module.lambda_function_final's ForceNew logging_log_group-derived name proposed exactly one replace at the same declared address (the log group; -/+ destroy and then create) cascading into two expected in-place updates (the function's logging_config, the inline log policy's document) and nothing else beyond the module's own pre-existing null_resource.archive[0] package-timestamp noise; applied cleanly; the old object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/free-wasp-lambda-simple) is confirmed gone and the new object (arn:aws:logs:eu-west-1:000000000000:log-group:/aws/lambda/free-wasp-lambda-simple-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (/aws/lambda/free-wasp-lambda-simple -> /aws/lambda/free-wasp-lambda-simple-v2); the next plan proposes no resource action beyond the same known noise; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace plus the same in-place cascade (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one object tampered (memory_size 128->256), exactly module.lambda_function.aws_lambda_function.this[0] proposed by both choudoufu and stock with the identical change, apply changed 1 and memory_size reads back as 128", "greenfield": "8 resources from nothing (3 taggable + 5 record-backed/config-derived), all three module-nested markers verified via the AWS CLI, 8 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (runtime, handler, memory, timeout, log-group retention)", "migrate": "3 stamped, 4 recorded, 0 failed, 1 skipped", + "plan_approval": "one argument edited (module.lambda_function's cloudwatch_logs_retention_in_days, unset -> 14, which reaches module.lambda_function.aws_cloudwatch_log_group.lambda[0] and nothing else - the module call carries no tags argument and every tags-shaped knob it has would reach three children at once), \"plan -out=approved.tfplan\" wrote a 31790-byte stock-format plan file whose whole change set is that one in-place update; the world then moved out of band (free-wasp-lambda-simple's memory_size 128->256, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming the extra row as \" module.lambda_function.aws_lambda_function.this[0] Update free-wasp-lambda-simple\" - both module.lambda_function.aws_lambda_function.this[0] and the live identity it was computed against - with \"Exit status 3\" spelled out for a pipeline; nothing was applied - /aws/lambda/free-wasp-lambda-simple still carried no retentionInDays, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with memory_size put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the log group read back with retentionInDays=14, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (retention unset again, memory_size still 128, next plan proposes no resource action, no state file left behind) so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 3; markers and record store intact", "test_plan": "no resource change proposed" }, - "duration_s": 171.8, + "duration_s": 173.6, "stage_seconds": { - "cold_deploy": 28, - "day2_count": 26, + "cold_deploy": 13, + "day2_count": 28, "day2_remove": 11, - "day2_rename": 24, + "day2_rename": 25, "day2_replace": 18, - "drift_reconverge": 18, - "greenfield": 25, + "drift_reconverge": 19, + "greenfield": 24, "migrate": 14, + "plan_approval": 15, "test_apply": 5, - "test_plan": 3 + "test_plan": 2 } }, "notes": "Re-verified 2026-08-18 in a fresh isolated worktree off local main (bcf78bacbd), for real (Docker/floci/AWS CLI, not read from a prior note). Stages 1 and 2 still pass exactly as landed: cold apply creates 8 resources, live-import verifies '3 of 8 resource instance(s) are eligible for stamping', -approve reports '3 resource(s) newly stamped, 0 already stamped, 0 failed, 5 skipped', and module.lambda_function.{aws_lambda_function,aws_iam_role,aws_cloudwatch_log_group} carry the expected module-qualified tofu-address/tofu-estate tags read straight through the AWS CLI. #303 (the count=var.enable_x?1:0 zero-instance admission gap on aws_lambda_function_url.this and aws_lambda_function_recursion_config.this) is CONFIRMED FIXED: re-running live-plan against current main, neither type appears anywhere in the diagnostics any more, as an error or a warning - stage 3 now fails on exactly one Error block, not two-plus. That block is local_file.archive_plan (module.lambda_function's package.tf:44, count = var.create && var.create_package ? 1 : 0, both true by default in this example so a real non-zero instance, not a zero-count block #303's fix would clear), refused under the logical-resource rule. Investigated whether this is a bug or correct behavior: it is correct, deliberate, and already ruled on. Issues #237 and #238 (both closed 2026-08-18) put local_file through exactly this question and #238's closing comment states local_file is 'deliberately left OTHER_REFUSED with a documented reason: neither of lint's two classes fits it correctly (its identity is argument-derived, not record-backed, and promoting it would silently reopen a count.index collision hazard a dedicated test already guards) - a genuine third-classification gap, not an omission, correctly left open rather than forced.' local_file's identity is a filename on the local disk of whatever machine ran apply - not a cloud object, nothing taggable, nothing an AWS CLI call could ever read back to confirm it still exists - so there is no live counterpart for a stateless replan to reconcile against. Considered scoping the estate around it the way corpus-vpc-complete/corpus-sumaform-aws scope around their own out-of-scope resources (the module's create_package=false + local_existing_package= toggle skips package.tf's local_file entirely) and rejected it: unlike sumaform's provision=false, which picks between the module's own equally-real published deployment modes to route around an infra-emulation gap in floci, swapping to a pre-built zip would replace the actual thing 'simple' demonstrates - the module's own default packaging pipeline - with a materially different scenario this corpus entry was never meant to test. Left as a real, reported block; run.sh's header carries the full investigation. One piece of relevant good news found along the way: #275 (closed 2026-08-18) built a record_store-gated residue mechanism for exactly the aws_lambda_function.filename/source_code_hash/publish phantom-diff problem a filename-deployed Lambda would otherwise hit under stateless replanning - this estate already declares a record_store, so once local_file gets its own identity class nothing here looks likely to re-hit that problem. test_apply and drift_reconverge remain not_run because test_plan does not pass; not attempted this pass since attempting them against a still-refused plan would prove nothing. No issue currently tracks the missing 'argument-derived-but-safe' LogicalClass itself (the actual unblock for this estate) - #237/#238 are both closed and did not spawn a follow-up; one may be worth filing if this crossing is prioritized again. Follow-up pass 2026-08-18 (#313 cross-check, fresh worktree off main at d303d9d425, real Docker/floci/AWS CLI run): confirmed the local_file.archive_plan block above is NOT #313's data.aws_availability_zones/static-context wall - grepped the full live-plan output, zero occurrences of that diagnostic. Filed the missing-LogicalClass follow-up this note flagged as worth filing but hadn't been: #314 ('local_file needs a fourth LogicalClass (argument-derived identity)'), citing #237/#238's rulings and the count-index guard test. No fix attempted - real multi-package work (lint + identity resolution + row-gen), not a quick derivation. Follow-up pass 2026-08-19 (#314 fixed and merged, d878aa914b/e7bb1f4b41): the fourth LogicalClass exists, but not the one the issue's own framing predicted - two of that framing's premises turned out false, checked rather than assumed. hashicorp/local's provider implements NO ImportState for local_file at all (confirmed against stock tofu: 'This resource does not support import'), so an argument-derived identity handed to internal/live/projection would have turned a lint refusal into a hard Cannot-import-for-projection error - strictly worse. And this estate's own filename argument isn't static anyway (it reads a data.external result). The real fix is ClassExternalAdmitted/EXTERNAL_ADMITTED: a record_store admits local_file the same way it admits ClassRecordAdmitted, resolving through ClassRecordBacked - the record is the only carrier that can bring prior state back for a type the provider itself cannot re-derive. Reaches exactly one type (local_file; local_sensitive_file is secret-bearing) but the RULE generalizes: live/logical-schemas.json's per-provider store_only now SELECTS between the two admitted classes instead of gating whether a type derives a row at all, which also retired the hand-written local_sensitive_file exception in ClassifyLogicalType - a net type-name-literal deletion, not an addition. Count-index guard (TestLocalFileKeepsItsCountIndexCheck) confirmed still holding via two separate mutation checks. Real re-crossing: local_file is admitted and appears nowhere in live-plan's diagnostics any more (asserted by absence) - but test_plan stays fail, now BLOCKED at 5 sites, a FOURTH wall newly reached rather than caused. All five trace to one expression, function_name = \"${random_pet.this.id}-lambda-simple\": random_pet.this is RECORD_ADMITTED so its id lives only in the record store, and the identity resolver declines to read that carrier for the three dependent resources (aws_iam_role.lambda, aws_iam_role_policy.logs, aws_lambda_function.this, aws_cloudwatch_log_group.lambda, one cascade) even though all three are already stamped and CLI-verified by stage 2 of the same run - choudoufu already holds the value and the objects are already marked, so this reads as an identity-resolver gap rather than a missing carrier. Not filed (no issue number assigned) - worth a slot. test_apply/drift_reconverge remain not_run, blocked on this new wall. Two real corrections made along the way: the crossing script had no AWS provider version pin (silently drifted to whatever the newest release was, now pinned =6.59.0 matching corpus-cloudfront's discipline), and live/LIMITATIONS.md's local-file section stated 'no cloud counterpart to reconcile against' as fact - false; the local filesystem is the counterpart, now corrected there too. Follow-up pass 2026-08-19/20 (#336 fixed and merged, 821c769715/c41279989a): #336's own diagnosis was wrong in two places, checked rather than assumed. The identity resolver was NOT declining to read the record-store carrier - resolver.parentPart already read random_pet.this.id correctly on unmodified main. What actually refused was coalesce(): iam.tf/main.tf's role_name/policy_name/log-group-name chains all select through coalesce(var.X, var.Y, \"*\")-shaped expressions, and resolver.isSymbolic reads only an expression's traversal ROOTS - var/local are never symbolic, so a selection sitting behind a module argument or a local looked entirely static, failed whole-expression evaluation, and had nothing left to try (the decomposition switch is only reached when a resource is named directly inside the expression). Fixed generically (internal/live/identity/coalesce.go, new): resolveCoalesceCall decides which argument the language selects using two proofs (provably-null-or-empty to skip, provably-non-null-non-empty to select), declining the whole call on anything undecidable rather than silently falling through - mutation-tested three ways (drop the non-emptiness proof, fall through on undecidable, remove the call entirely), each caught. Measured reach: refusal-probe sites 16075->15964 (-111), instances 4499->4522 (+23), 12 entries improved across six unrelated sources, 0 worse - schema-less mode, an under-report since it's blind to the record-backed half of this estate's own chain. Real re-crossing: live-plan diagnostics 5->0, the plan runs to completion for the FIRST time - but test_plan still FAILS, now on a genuinely NEW, fifth wall: 'live-plan is not empty', proposing to create every record-backed resource (random_pet.this first) from scratch. Root cause, and #336's second wrong premise: live-import's Approve loop only calls #327's recordResidueFor for a STAMPED entry; a record-backed resource is by definition not stampable (no live cloud object to tag) and hits OutcomeSkipped, continuing past the residue call entirely - so the record store is empty for every record-backed instance after a clean migrate, not populated as #336 assumed. Filed as #340, not attempted - the fix is migrate seeding the record store from the migrated state's own object for every record-backed instance, a sibling call to recordResidueFor on the skipped-because-record-backed path. test_apply/drift_reconverge remain not_run, now blocked on #340 instead of #336's five diagnostics. Follow-up pass 2026-08-20 (#340 fixed and merged, d30daa156f/8d34e3ded9): the issue's own framing was half right - recordResidueFor does NOT gate on 'stamped', it runs for any entry with an *eligible; the real gate is one line earlier, Ratify never building an *eligible for a record-backed type at all. Fixed with Approve's second write path, the sibling of the tag write: projection.SeedRecordForInstance writes a record-backed instance's object into the record store, byte-identical to what an apply's WriteBack would write, reading before writing so an already-correct record is a no-op and a genuinely different one refuses rather than clobbers. Keys on identity.TypeIdentity.RecordBacked and nothing else - 15 types across 4 providers (local/null/random/time/terraform_data), no aws_*/random_* name in the control flow. Real re-crossing: STAGE 2 migrate reports '3 newly stamped, 0 already stamped, 4 newly recorded, 0 already recorded, 0 failed, 1 skipped', the store's own files grepped for random_pet.this's generated id. STAGE 3: live-plan now raises ZERO diagnostics for the first time ever on this estate, every identity resolves, no record-backed resource is proposed for creation - but the plan is not empty: 0 to add, 2 to change, 0 to destroy, a SIXTH wall. Both changes are real and distinct from every prior wall: (1) a nested-block round-trip on aws_lambda_function (- environment {} / + logging_config { log_format = \"Text\" }, floci's Lambda read vs the module's config), (2) a sensitivity-only diff on local_file.content (OpenTofu's own renderer says 'The value is unchanged' - a genuine limitation the fixing agent found and pinned rather than hid: ResourceInstanceObjectSrc.Decode re-applies AttrSensitivePaths so the decoded value is marked, ctyjson.Marshal panics on a marked leaf, the fix unmarks before encoding, and projection.recordPayload has nowhere to store the sensitivity path - so the record carries the value but not the mark. projection.WriteBack shares this same hole, worse (no unmark of its own, would panic on the identical object after a real apply) - unfiled, not touched, worth a slot). test_apply/drift_reconverge remain not_run, blocked on this sixth wall now. Separately, #341 (found by the sibling corpus-mastino-dns crossing, same ratify.go gate but a different population - untaggable ordinary AWS types like aws_route53_record, architecturally guaranteed never to be RecordBacked per TestResolveNeverEmitsRecordBackedForAWSEstate) is CONFIRMED STILL OPEN after #340 - the two issues share a root-cause line but #340's fix is deliberately scoped to the disjoint RecordBacked population and does not reach it. #340's own recordable/Approve sibling-carrier pattern is flagged as a reasonable template for #341's eventual fix. Follow-up pass 2026-08-20 (worktree ../wt/lambda-simple-sixth-wall, branch live/lambda-simple-sixth-wall, off LOCAL main ea9fd62fc0): the sixth wall was REPRODUCED for real first, not read off this note - STAGE 1 PASS, STAGE 2 PASS, STAGE 3 with zero diagnostics and 'Plan: 0 to add, 2 to change, 0 to destroy', both changes verbatim as recorded above. Stages are UNCHANGED at 2 of 5, and the reason is now measured rather than argued: the two changes belong to two different projects. (a) local_file.archive_plan's sensitivity-only diff IS choudoufu's and is FIXED (f66bc9e043). projection.recordPayload gained SensitiveAttrs, encoded exactly the way a state file encodes sensitive_attributes: encodeRecordPayload splits the value from its marks itself so no caller unmarks, decodeRecordPayload puts them back the way ResourceInstanceObjectSrc.Decode does, and materializeRecord re-marks after the schema conversion so obj.Encode derives AttrSensitivePaths from them. The mechanism the wall turned on is that live-plan runs the plan graph with SkipRefresh (live_plan.go:499), so a projected object's AttrSensitivePaths is the ONLY marks the plan's 'before' side ever has - upstream re-marks a refreshed object at node_resource_abstract_instance.go:1106 and that line is never reached - while the 'after' side is re-marked from the config and the provider schema every run at :1383. Derived from the object's own marks and nothing else, so it reaches any record-backed type with any sensitive attribute at any path; a mark that is not marks.Sensitive is refused rather than dropped. Four mutation checks, each caught, and the shape test consults an EXTERNAL source (it writes a real state file through internal/states/statefile and requires the same JSON for the same paths) rather than round-tripping against itself. TestIdentityGolden 0 changed, 0 added, 0 removed; ./internal/live/..., ./tools/..., ./live/..., ./cmd/... and ./internal/command/ all green. (b) aws_lambda_function's '- environment {}' / '+ logging_config { log_format = \"Text\" }' pair is NOT choudoufu's, and this is now proven by a CONTROL rather than reasoned about: run.sh's new step 3b runs plain terraform, its own state file, its own refresh, 'terraform plan -detailed-exitcode' immediately after its own cold apply with no choudoufu anywhere in the run, and it replans NOT EMPTY on exactly module.lambda_function.aws_lambda_function.this[0] and nothing else. Asserted by value, so a new emulator gap breaks the script instead of hiding in a bucket labelled expected, and a fixed one breaks it too. Filed as lex00/floci#83 with both causes located in LambdaController.buildFunctionConfiguration: Environment is emitted unconditionally ('SDK expects it even when empty') where real AWS omits it for a function that never had one - which is why terraform-provider-aws reads it under 'if function.Environment != nil' - and LoggingConfig is neither stored nor emitted at all, where real AWS always returns one defaulting to Text. The module declares zero environment blocks (main.tf:90, a dynamic block over an empty map) and one logging_config unconditionally (main.tf:136), so both fire on the module's DEFAULTS. Stage 3 now splits its own plan against the step-3b control with comm and names only the remainder as choudoufu's. WHAT IS NOT VERIFIED, stated rather than implied: the post-fix crossing was started and its choudoufu init was killed by the harness before stage 2, so the sensitivity fix has NOT been observed clearing the diff in a real end-to-end run - it is verified by unit tests that drive the two real paths (WriteBack after an apply, then BuildWith/materializeRecord on the next plan) and assert inst.Current.AttrSensitivePaths by value. The next run of this script is what settles it, and it should still end at 'live-plan is not empty' with the Lambda alone until lex00/floci#83 lands. Two further issues filed from this pass, neither fixed: #343 (builder.materialize applies no schema.Block.ValueMarks to what a provider Read returned, so the identical perpetual diff exists for any CONCRETE cloud object with a Sensitive attribute - separate because it also changes what b.live means for identity composition, and unmeasured over the corpus) and #344 (a record written before SensitiveAttrs existed now conflicts with its own re-migration though the value is identical, because SeedRecordForInstance compares bytes; population is local_file plus any config-derived mark, and the format is one day old). Stages 4 and 5 remain unwritten: an empty stage-3 plan is their precondition and lex00/floci#83 is what stands between this estate and one. Follow-up pass 2026-08-20 (worktree ../wt/lambda-simple-floci83, branch live/lambda-simple-floci83): lex00/floci#83 is FIXED and CLOSED, settled against real AWS (one throwaway Lambda function + IAM role created and immediately deleted, disclosed to the user per standing permission) rather than assumed - GetFunction/GetFunctionConfiguration omits Environment entirely for a function that never had one (not present-but-empty) and always returns LoggingConfig, defaulting to Text. Fixed in LambdaController/LambdaService/LambdaFunction, verified three ways (local instance, direct probe, and the same probe against the published GHCR image). Re-crossing after the re-pin: STAGE 3 (test_plan) now raises zero diagnostics and proposes changing ZERO resources for the first time ever on this estate - both prior sixth-wall diffs (the sensitivity-only local_file.content diff and the aws_lambda_function nested-block mismatch) are gone, verified by reading the raw terraform plan output directly rather than trusting extracted variables. But the plan is still not empty: all 23 of the example module's root-level `output` blocks render as `+ new` on every run. Root cause: internal/live/projection.Manager.GetRootOutputValues always returns an empty map - nothing evaluates the config's own output blocks against the reconstructed prior state before the plan graph asks for it. A genuinely new, generalizable gap (neither corpus-mastino-dns nor corpus-evoteum-modules, the two other closest-to-5/5 crossings, declares any root-level output, so nothing had hit this before). Filed as INTENTIUS/choudoufu#348, not attempted - core projection-architecture work, not a quick derivation. Stages remain 2 of 5; the estate's real blocker moved from floci#83 to #348, which is now the sole thing standing between this estate and an empty stage-3 plan. Follow-up pass 2026-08-20 (primary checkout, local main db1f412cfd, real Docker/floci run - GitHub issue #340 verification): #340 was found ALREADY FIXED on main (d30daa156f/8d34e3ded9, confirmed by commit history and by internal/live/liveimport/record_test.go's TestApprove_SeedsTheRecordStoreForARecordBackedInstance/TestRecordBackedTypeReadsTheGeneratedTable, the latter covering random_pet/null_resource/terraform_data/local_file/time_sleep/random_id generically), so this pass re-verified rather than re-fixed. Re-crossing confirms #340's own fix by absence again: 'no record-backed resource is proposed for creation: the migrate seeded all four' and 'no sensitivity-only diff on local_file.archive_plan: the record carries its marks' both print in stage 3's own output. #349 ('see through provably-zero-instance blocks when evaluating root outputs', 88d7e3961e, landed after this note's #348 paragraph) cut the output-only diff from all 23 root outputs to exactly 2: 'lambda_function_arn_static' and 'local_filename', both still '+' on every run. Stage 3 (test_plan) is still BLOCKED - not yet empty - but the wall is now two output lines, not twenty-three, and #340 itself contributes zero diagnostics and zero sites to what remains. Stages unchanged at 2 of 5 pass; the residual 2-output gap is #348/#349's remaining scope, not #340's, and was not investigated further here (out of this issue's scope). Follow-up pass 2026-08-21 (isolated worktree off local main 860c29e129, real Docker/floci/AWS CLI run, identical harness run TWICE with only TOFU_BIN swapped - not read from any prior note): #349's sub-problem 2, the root-output data-source read, is now built, and this estate's stage-3 output diff went from 2 lines to 1. Measured: at 860c29e129 the plan's 'Changes to Outputs:' block carries 'lambda_function_arn_static' and 'local_filename'; with the fix it carries 'local_filename' alone. lambda_function_arn_static vanishes from the diff entirely rather than rendering as '~ old -> new', which is the stronger result: the plan graph independently computed the same value the pre-plan read computed for the prior side, so they cancel. The three data sources behind it (data.aws_partition.current, data.aws_region.current and data.aws_caller_identity.current, all in module.lambda_function) are read live before the plan through the same configured aws provider instance the projection already reads this estate through. local_filename is UNCHANGED and stays refused ON PURPOSE: it reaches data.external.archive_prepare, whose read runs package.py on the machine running the plan, and the new demand class is confined to providers this configuration manages live objects through (dataread.LiveProviders) - the external provider serves no managed resource type at all, in this or any configuration, so it is excluded structurally rather than by name. Stages are UNCHANGED at 2 of 5: cold_deploy pass, migrate pass, test_plan still FAIL (one output line is still one output line, so the plan is still not empty), test_apply and drift_reconverge still not reached. This narrowed the stage-3 diagnostic count; it did not clear the stage. Follow-up pass 2026-08-29 (isolated worktree off local main 499f9f5e80, real Docker/floci/AWS CLI run - GitHub issue #498, 'migrate pass and fail 23 minutes apart at the same emulator pin'): CONFIRMED as a real, deterministic defect, not a flake and not #497's runner-resource-pressure hypothesis. Reproduced 100% of the time under a controlled variable rather than by chance: migrate PASSED 7/7 consecutive local runs (this machine's stock terraform, v1.15.8) with byte-identical output every time, then FAILED 2/2 runs the instant HashiCorp Terraform v1.16.0 was forced first on PATH for stage 1's cold deploy, with the exact nightly failure text ('live-import -approve did not stamp 3 and record 4 of 8 resources cleanly'). Root cause, isolated to one byte: Terraform >=1.16.0's built-in terraform_data resource gained a new 'store' nested block (verified directly via `terraform providers schema -json` against terraform.io/builtin/terraform, no choudoufu involved) that choudoufu's own terraform_data schema (internal/builtin/providers/tf/resource_data.go, unchanged since the OpenTofu fork) does not declare; decoding a stock-terraform-1.16-produced cold.tfstate's terraform_data instance against that older schema failed with 'unsupported attribute \"store\"' (internal/live/liveimport/ratify.go's ratifyRecordBacked, its inst.Current.Decode(schema.Block.ImpliedType()) call), demoting terraform_data.package_filename_for_hash from RECORDED to SKIPPED and changing live-import -approve's summary line, which trips run.sh's exact-string assertion. The 'pass at 10:23:19Z, fail at 10:46:59Z, same commit' shape in #498 was never nondeterminism in the same environment: the passing row was a local worker's run against an older pinned-by-brew terraform (like this pass's own baseline), the failing row was CI's `hashicorp/setup-terraform@v3` with terraform_version: latest (confirmed 1.16.0 from the nightly's own log) - two different environments measuring the identical commit near-simultaneously, not one environment flip-flopping. Downloaded the eks-basic sibling failure from the same nightly run for comparison and confirmed it is a DIFFERENT failure (a runner tofu-on-PATH casualty per #497, mis-attributed to whatever CURRENT_STAGE was set to when it died) - #498's own caveat about #497 does not reach this estate's failure, which is fully explained by the terraform_data schema gap alone. Fixed generically for the one type it can reach (dataStoreResourceSchema() now declares 'store' as a NestedType Object attribute mirroring the real schema's own field shape and WriteOnly/Sensitive flags, verified by decoding the real terraform-1.16.0-produced state); this is the resource's own implementation file, not classification control flow, so it carries no live/derivation_guard_test.go entry. ctyjson.Unmarshal was independently verified (a standalone throwaway program, not assumed) to default a type attribute missing from raw JSON to null, so old choudoufu-written terraform_data state predating this field decodes unaffected - TestManagedDataUpgradeStateMissingStore pins that boundary directly. Re-crossing after the fix: migrate PASSES under terraform 1.16.0 too (2/2), with the correct '3 stamped, 4 recorded, 0 failed, 1 skipped' split restored, and the whole estate clears end to end (exit 0) under both terraform versions - no other stage regressed. Not verified: HashiCorp Terraform versions between 1.15.8 and 1.16.0 (whichever one first introduced 'store') and any version after 1.16.0 that might change the block's shape again; the fix targets the exact field shape read off 1.16.0 and would need re-verification against a materially different future schema." @@ -1219,16 +1247,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1244,19 +1272,21 @@ "drift_reconverge": "S3 alarm's alarm_description tampered, exactly 1 object proposed and applied, reconverged to its configured description", "greenfield": "3 resources from nothing (2 tagged alarms + the untaggable dashboard), both alarm markers verified via the AWS CLI, 3 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on both alarms", "migrate": "2 of 3 stamped (1 skipped, untaggable dashboard), 0 failed; both alarm markers read back via the AWS CLI", + "plan_approval": "one argument edited (cf_requests_spike's alarm_description, a config-owned non-ForceNew argument, gains a \"(reviewed)\" suffix), \"plan -out=approved.tfplan\" wrote a 7918-byte stock-format plan file whose whole change set is one update on module.monitoring.aws_cloudwatch_metric_alarm.cf_requests_spike; the world then moved out of band (S3GetRequestsSpike's alarm_description, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.monitoring.aws_cloudwatch_metric_alarm.s3_requests_spike and the live S3GetRequestsSpike it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - CFRequestsSpike's alarm_description still read as configured through the AWS CLI, rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the description put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and CFRequestsSpike read back with the reviewed description, so the refusal is earned by the drift and not handed out to every plan file. BOTH the plan -out and the apply carry this crossing's own -target set (aws_budgets_budget stays out of the graph - floci still answers UnknownOperationException for AWSBudgetServiceGateway on the pinned image), which is enough because a live-markers apply plans the live system from its OWN arguments rather than replaying the file: no exemption and no Go change was needed for issue #903's -target trap. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); object count unchanged at 2, no state file", "test_plan": "no resource change proposed; both alarms' tofu-address unchanged, dashboard body re-derived and matches distribution_id" }, - "duration_s": 93.9, + "duration_s": 90, "stage_seconds": { - "cold_deploy": 18, + "cold_deploy": 5, "day2_count": 15, - "day2_remove": 4, + "day2_remove": 5, "day2_rename": 6, "day2_replace": 5, "drift_reconverge": 4, - "greenfield": 13, + "greenfield": 11, "migrate": 25, + "plan_approval": 9, "test_apply": 2, "test_plan": 2 } @@ -1282,16 +1312,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "d72960cdc31f4e7d90f61088920379fc29b3999c", - "date": "2026-09-06T05:35:09Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1300,28 +1330,30 @@ "exit_code": 0, "detail": { "cold_deploy": "63 resources from stock terraform; 4 live zones confirmed unmarked", - "day2_count": "choudoufu: scaling DataCite's own real, already-live aws_route53_record.wp-prod-staging count block (count.index + 3 in the name, header point 3) from 10 to 9 destroyed exactly wp-prod-staging[9] (staging12.datacite.org, 0 add, 0 change, 1 destroy; confirmed gone via the AWS CLI, and its local record file correctly tombstoned rather than left claiming a live identity - the #398-guard shape, confirmed by reading the file directly rather than assumed), leaving wp-prod-staging[0] (staging3.datacite.org)'s live TTL and record-store import_id unchanged; scaling back from 9 to 10 created exactly wp-prod-staging[9] again (0 add -> 1 add, 0 change, 0 destroy), TTL=300 and record import_id=ZFNTJ9UTHQDEAEU_staging12.datacite.org_A (identical string to before - Route 53 hands back no system id for a record set, so realness of the destroy was proved by the AWS CLI absence check and the tombstone, not a changed id), while wp-prod-staging[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical 10-instance count block, plan-only against cold_deploy's own state (applying would disturb the live objects migrate/stage 3-5 depend on), shows the identical shape: destroy the highest index only, create it back under the same name, every lower index untouched both times. aws_route53_record carries no tags at all (header point 2), so this type's own 'every surviving instance keeps its identity' is proved through the record store's ZONEID_NAME_TYPE identity string and a direct AWS CLI read, never a tofu-address tag value - the colon-vs-bracket tag-value escaping trap live/MARKERS.md documents does not apply to an untaggable type. BREAK_COUNT=1 confirms the wrong-instance assertion correctly fails to hold.", + "day2_count": "choudoufu: scaling DataCite's own real, already-live aws_route53_record.wp-prod-staging count block (count.index + 3 in the name, header point 3) from 10 to 9 destroyed exactly wp-prod-staging[9] (staging12.datacite.org, 0 add, 0 change, 1 destroy; confirmed gone via the AWS CLI, and its local record file correctly tombstoned rather than left claiming a live identity - the #398-guard shape, confirmed by reading the file directly rather than assumed), leaving wp-prod-staging[0] (staging3.datacite.org)'s live TTL and record-store import_id unchanged; scaling back from 9 to 10 created exactly wp-prod-staging[9] again (0 add -> 1 add, 0 change, 0 destroy), TTL=300 and record import_id=Z7Z25KMX2IAY0SJ_staging12.datacite.org_A (identical string to before - Route 53 hands back no system id for a record set, so realness of the destroy was proved by the AWS CLI absence check and the tombstone, not a changed id), while wp-prod-staging[0] stayed untouched throughout; the next plan is empty; the G-ORACLE stock oracle on the identical 10-instance count block, plan-only against cold_deploy's own state (applying would disturb the live objects migrate/stage 3-5 depend on), shows the identical shape: destroy the highest index only, create it back under the same name, every lower index untouched both times. aws_route53_record carries no tags at all (header point 2), so this type's own 'every surviving instance keeps its identity' is proved through the record store's ZONEID_NAME_TYPE identity string and a direct AWS CLI read, never a tofu-address tag value - the colon-vs-bracket tag-value escaping trap live/MARKERS.md documents does not apply to an untaggable type. BREAK_COUNT=1 confirms the wrong-instance assertion correctly fails to hold.", "day2_remove": "choudoufu: deleting aws_route53_zone.eu and aws_route53_record.eu-ns's blocks - both destroys proposed (matching stock's own oracle exactly) and applied cleanly (Apply complete! Resources: 0 added, 0 changed, 2 destroyed.), the zone genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report); the next plan is empty. The parent-scoped removal sweep gap this estate named (gauntlet:parent-scoped-sweep) is closed: recordOrphanReadSweep composes aws_route53_record's identity from its migrate-seeded record correctly (composeImportIDFromComponents's OmitIfAbsent fix) and carries a destroy-before-parent ordering hint (identity.Resolution.DestroyDependsOn) so the record's own destroy is never raced against its zone's force_destroy cascade.", "day2_rename": "moved block: aws_route53_zone.production renamed with zero churn (0 add, 1 change, 0 destroy) - only the zone's own marker rewritten, none of its 45 record children moved; live-mv: aws_route53_zone.internal renamed with zero churn, marker rewritten in place; stock oracle over the same two-zone rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live zone ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing aws_route53_record.status's ForceNew name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (status.datacite.org./CNAME) is confirmed gone and the new object (ZFNTJ9UTHQDEAEU_status2.datacite.org_CNAME) exists, both via the AWS CLI; the local record store's record at the same address now names the new object's identity, not the destroyed one (ZFNTJ9UTHQDEAEU_status.datacite.org_CNAME -> ZFNTJ9UTHQDEAEU_status2.datacite.org_CNAME); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); F-ORACLE also confirms the four apex NS records this estate's DELTA 5 manages can never take this same path (Route 53 refuses to delete the NS/SOA record at a zone's apex), which is why status was chosen instead; BREAK=replace confirms a manufactured identity collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", + "day2_replace": "choudoufu: changing aws_route53_record.status's ForceNew name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (status.datacite.org./CNAME) is confirmed gone and the new object (Z7Z25KMX2IAY0SJ_status2.datacite.org_CNAME) exists, both via the AWS CLI; the local record store's record at the same address now names the new object's identity, not the destroyed one (Z7Z25KMX2IAY0SJ_status.datacite.org_CNAME -> Z7Z25KMX2IAY0SJ_status2.datacite.org_CNAME); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied); F-ORACLE also confirms the four apex NS records this estate's DELTA 5 manages can never take this same path (Route 53 refuses to delete the NS/SOA record at a zone's apex), which is why status was chosen instead; BREAK=replace confirms a manufactured identity collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one untaggable record drifted, exactly aws_route53_record.wp-prod-staging[0]/ttl proposed and applied, reconverged to 300, marker intact", "greenfield": "63 resources from nothing (4 tagged zones + 59 untaggable records), the production zone's marker verified via the AWS CLI, 63 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches on zone count (4) and total record-set count (63)", "migrate": "4 of 63 stamped, 59 skipped as untaggable, 0 failed; 59 identity records written (#364), 14 of them also carrying residue (#341), DataCite's own tags survived", + "plan_approval": "one argument edited (aws_route53_zone.production's tags gain Reviewed=yes - a tag, so in-place, moving no live id), \"plan -out=approved.tfplan\" wrote a 19819-byte stock-format plan file whose own totals are \"Plan: 0 to add, 1 to change, 0 to destroy\" and whose whole change set is that one update; the world then moved out of band (staging3.datacite.org's TTL 300->77 in zone Z7Z25KMX2IAY0SJ, this estate's own STAGE 5 upsert lifted, through the AWS CLI and never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming the extra row as \" aws_route53_record.wp-prod-staging[0] Update Z7Z25KMX2IAY0SJ_staging3.datacite.org_A\" - both aws_route53_record.wp-prod-staging[0] and the live composed identity Z7Z25KMX2IAY0SJ_staging3.datacite.org_A it was computed against, an UNTAGGABLE record whose identity comes from the local record store rather than a marker tag - with \"Exit status 3\" spelled out for a pipeline; nothing was applied - zone Z7Z25KMX2IAY0SJ still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the TTL put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the zone read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (tag gone, tofu-address marker intact, 63 record sets still there, next plan proposes no resource action) so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 4 zones / 63 record sets unchanged, all 4 markers unmoved, all 59 identity records intact (14 residue-bearing)", "test_plan": "plan empty across 63 instances, no state file; 14 record sets and 4 zones filled residue from the store" }, - "duration_s": 681.4, + "duration_s": 614.6, "stage_seconds": { - "cold_deploy": 150, - "day2_count": 69, - "day2_remove": 32, - "day2_rename": 26, - "day2_replace": 55, - "drift_reconverge": 29, - "greenfield": 248, - "migrate": 55, - "test_apply": 10, - "test_plan": 7 + "cold_deploy": 115, + "day2_count": 57, + "day2_remove": 30, + "day2_rename": 20, + "day2_replace": 46, + "drift_reconverge": 28, + "greenfield": 234, + "migrate": 43, + "plan_approval": 28, + "test_apply": 8, + "test_plan": 5 } }, "notes": "DataCite's own global DNS root module, the largest offline-clean estate (63 instances) that had never touched a cloud. 2 of 5 - and the two stages it does not reach are blocked on one real, general, previously-unrecorded choudoufu defect, not on anything specific to this estate. Still offline-clean when crossing started (refusal-probe -schemas: blocked 0 sites 0 instances 63 at c41279989a); the schema-less mode disagrees (blocked 1 sites 2, both render correctly in the real run - HANDOFF's asymmetry caveat firing on a live estate). Two of team-members-access's four deltas recur (#268 mandatory backend edit, in cloud{} form; #269 provider version skew, ~> 5 -> 5.100.0 with no list resources, all four zones ServerAssigned); the other two do not (one data source answered by an out-of-band VPC; an ordinary emulator override). Two NEW walls: (1) estate-owned, not choudoufu's/floci's - the four *-ns blocks manage each zone's own apex NS set, which Route 53 creates itself, so a from-scratch apply dies with InvalidChangeBatch; fixed with allow_overwrite=true, the same argument the estate's own author already writes on wp-prod-staging. (2) Filed as #341: stage 3's entire plan is 0/14/0, every diff line +allow_overwrite=true (10 of 14 on wp-prod-staging[0..9], carried in DataCite's own text with no deltas needed) - #275's residue mechanism populates and reads back the record store for TAGGABLE resources only (4 zones get 'filled 1 residue attribute(s)'), but none of the 59 untaggable record sets do, because internal/live/liveimport/ratify.go's !taggable() branch returns before the ReadResource that builds the *eligible object residue needs, and Approve's recordResidueFor sits past the continue that skips a resource with no *eligible - one carrier serving two unrelated jobs, and untaggability should only disqualify one of them (the tag write, not the residue read). 342 of 1025 admitted types are untaggable and share this exclusion. Not fixed here - out of scope for a crossing pass. What the run DOES prove: all 63 rendered identities correct and distinct by value against the AWS CLI's own answer, including two same-named datacite.org zones (public/private) that did not swap and ten wp-prod-staging[0..9] instances rendering staging3..staging12 individually. The 59 untaggable record sets are 94% of the estate - the widest derived-from-tagged fan-out in either lane. Script exits 0 only on reaching exactly this blocker (asserting the changed-address set, that allow_overwrite is the only attribute in the whole diff, and the 0/14/0 totals line) and non-zero on anything else including an empty plan, which is the signal to promote this entry to five stages once #341 lands. BREAK=1 verified red at exactly the identity assertion (a swapped zone id). Also filed lex00/floci#81 (floci accepts a record set whose name is outside its hosted zone, a real bug in the estate's own text; blocks nothing here but is the shape where a crossing passes on the emulator and fails on real AWS). Suggested a new 'published-deployment' lane, distinct from terraform-popular/opentofu-native/reference, since this is neither a module example nor an OpenTofu-native project but a company's own live TFC-connected infrastructure. Merged d5e592d67c (crossing e74b6e5c01); just demo-corpus-mastino-dns, port 4731. just ci green (exit 0, read from a file). Follow-up pass 2026-08-20 (#341 fixed and merged, c73a6e4617/78c92ad64a): FIVE OF FIVE, real. Fix is a third carrier (residuable, which eligible now embeds) built for any admitted-untaggable instance with a record_store declared, deliberately with NO ReadResource at ratify time - a residue attribute is by definition one no read returns, so state already has everything the fix needs, and skipping the read means an untaggable instance can never come back MISSING/DRIFTED (a concern the issue itself raised). Verdict stays StatusUntaggable/OutcomeSkipped; only the marker write was ever skippable, not the residue write. Corrected denominator along the way: DefaultTable holds 1040 rows, not 1025 - survey-full.json calls 683 taggable/342 untaggable and doesn't cover 15 at all. Real re-crossing: migrate reports 4 stamped/59 UNTAGGABLE/14 residue records (all 4 zones plus 10 of the 59 record sets, matching the bug's own signature exactly); test_plan EMPTY, all 63 identities asserted by value; test_apply a genuine no-op with all 14 residue records and 4 markers unchanged; drift_reconverge drifts wp-prod-staging[0]'s TTL (untaggable AND residue-carrying) and reconciles exactly that instance, BREAK_STAGE5=1 verified failing. One honest caveat: stages 4/5 had to be WRITTEN (the prior entry's header claimed they existed in git history but e74b6e5c01 is the file's only commit), and the verifying run itself executed in two calls after a SIGTERM mid-init, not one continuous process - same container/workdir throughout, but the committed script has not been run start-to-finish in a single invocation. A real, separate, more urgent finding surfaced while verifying: #340's own change to live-import's summary line (\"%d newly recorded, %d already recorded\" inserted mid-line) broke the exact-string assertion in 19 OTHER crossing scripts on main - invisible to just ci since e2e scripts aren't in that tier. Filed as #342 with the full list and the one-line fix each needs." @@ -1346,16 +1378,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "a7ca11f9351c0334115a6ca4b3b5de025d70f947", - "date": "2026-09-06T23:49:04Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1364,28 +1396,30 @@ "exit_code": 0, "detail": { "cold_deploy": "26 resources, genuinely cold, genuinely unmarked", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-0f150fc86696abb33) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) unchanged, both read back through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW GroupId (sg-b5900fe02eec8c173 -> sg-16652b643494fb26b) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; every absence check reads length(SecurityGroups) through a group-id FILTER, because describe-security-groups --group-ids on a deleted id returns an empty list with exit 0 on this emulator pin; the next stateless live-plan is empty. Stock oracle (G0): the identical 2-instance block stood up with plain tofu in its own working directory against the same idle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new GroupId (1 add, 0 change, 0 destroy), the lower index's GroupId unchanged both times - then torn down (3 destroyed) before the choudoufu side ran. SYNTHETIC BLOCK, and why: every count this module declares is a boolean create toggle (create_vpc x7, create_s3_bucket x4, create_cloudfront_distribution x2, three launch_template variants gated on existing_id == null), never a scalable set, so scaling one is the day2_remove shape this script already runs, not a shape with a survivor; and its one real for_each (aws_batch_job_definition.tiles over toset(var.themes)) is scoped by this crossing's own root config to a single theme from stage 1 onward, so it has nothing to scale down to and widening it would move every earlier stage's counted assertions (26 resources, 16 stamped, 16 tagged objects, day2_rename's own 16-address list). What shrinking that set would plan is not claimed: it was never measured, because it was never a usable option. Sanctioned fallback per live/GAUNTLET.md #8, precedent reference-ec2-vpc Part F and corpus-iam-policy Part G. It reuses a type this estate already exercises (aws_security_group.batch), sits at a root address nothing else names, and runs entirely after day2_remove, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly reports fail.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-f37ccd0807772ad2c) and its tofu-address marker (aws_security_group.count_test:0, colon-escaped per live/MARKERS.md) unchanged, both read back through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW GroupId (sg-a444bab32fd697268 -> sg-46bcdd1e7ba3d595c) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] stayed untouched throughout; every absence check reads length(SecurityGroups) through a group-id FILTER, because describe-security-groups --group-ids on a deleted id returns an empty list with exit 0 on this emulator pin; the next stateless live-plan is empty. Stock oracle (G0): the identical 2-instance block stood up with plain tofu in its own working directory against the same idle endpoint shows the identical shape - destroy the higher index only (0 add, 0 change, 1 destroy), create the higher index back under a new GroupId (1 add, 0 change, 0 destroy), the lower index's GroupId unchanged both times - then torn down (3 destroyed) before the choudoufu side ran. SYNTHETIC BLOCK, and why: every count this module declares is a boolean create toggle (create_vpc x7, create_s3_bucket x4, create_cloudfront_distribution x2, three launch_template variants gated on existing_id == null), never a scalable set, so scaling one is the day2_remove shape this script already runs, not a shape with a survivor; and its one real for_each (aws_batch_job_definition.tiles over toset(var.themes)) is scoped by this crossing's own root config to a single theme from stage 1 onward, so it has nothing to scale down to and widening it would move every earlier stage's counted assertions (26 resources, 16 stamped, 16 tagged objects, day2_rename's own 16-address list). What shrinking that set would plan is not claimed: it was never measured, because it was never a usable option. Sanctioned fallback per live/GAUNTLET.md #8, precedent reference-ec2-vpc Part F and corpus-iam-policy Part G. It reuses a type this estate already exercises (aws_security_group.batch), sits at a root address nothing else names, and runs entirely after day2_remove, so no earlier stage's assertions move. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and correctly reports fail.", "day2_remove": "choudoufu: create_cloudfront_distribution=false proposed exactly two destroys plus one in-place update (0 add, 1 change, 2 destroy: the distribution, its untaggable OAC, and the bucket policy's own CloudFrontOAC statement dropping), applied cleanly (0 added, 1 changed, 2 destroyed), the distribution is genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state also proposes exactly the same two destroys plus the same bucket-policy update", "day2_rename": "moved block: module.overture_tiles renamed to module.overture_tiles_moved via ONE module-level moved block, 0 add/0 destroy, 16 real tag-marker rewrites (plan showed 18 - two untaggable siblings' policy JSON transiently 'known after apply', resolving to no real change at apply time, confirmed via the stage-5 marker/propagation filter and by value); live-mv: module.overture_tiles_moved renamed to module.overture_tiles_final across 14 of 16 taggable children, one call each, zero churn - the other 2 (aws_batch_compute_environment.tiles and aws_iam_instance_profile.ecs, both server-/provider-assigned identities with no List support in the provider) correctly refused by live-mv and renamed via their own moved blocks instead, applied cleanly; the nine untaggable/config-derived children and the UNTAGGABLE OAC (no longer UNADMITTED_TYPE - #249 narrowed) did not move at all; stock oracle over the identical module rename on cold_deploy's own state also shows zero churn via its own single module-level moved block, covering every one of the 26 children including the two live-mv cannot", "day2_replace": "choudoufu: supplying module.overture_tiles_final's name_overrides.cloudwatch_log_group proposed exactly one replace at the same declared address (the log group; -/+ destroy and then create) cascading into two expected in-place updates (the execution role's inline log policy, the job definition) and nothing else; applied cleanly; the old object (/aws/batch/overture-tiles-crossing) is confirmed gone and the new object (/aws/batch/overture-tiles-crossing-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (/aws/batch/overture-tiles-crossing -> /aws/batch/overture-tiles-crossing-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address plus the same cascade (plan only, not applied); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one object tampered (VPC Name tag), exactly module.overture_tiles.aws_vpc.batch[0] proposed by both choudoufu and stock with the identical change, apply changed 1 and the Name tag reads back as configured", "greenfield": "26 resources from nothing, bucket and batch job queue markers verified via the AWS CLI, 26 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (batch job queue state, CloudFront distribution comment, bucket count)", "migrate": "16 of 26 stamped, 0 failed; the other 10 correctly UNTAGGABLE or already-ruled UNADMITTED_TYPE", + "plan_approval": "one argument edited (module.overture_tiles's cors_allowed_origins, [\"*\"] -> [\"https://tiles.example.invalid\"], which reaches module.overture_tiles.aws_s3_bucket_cors_configuration.tiles[0] and nothing else - it is the only root knob of this 26-instance estate that lands on exactly one instance IN PLACE, since `tags` reaches all 16 taggable children at once, name_overrides and the launch template are ForceNew, and the create_* toggles are creates and destroys), \"plan -out=approved.tfplan\" wrote a 34023-byte stock-format plan file whose whole change set is that one in-place update; the world then moved out of band (vpc-9038e1c6's Name tag -> moved-after-approval, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming the extra row as \" module.overture_tiles.aws_vpc.batch[0] Update vpc-9038e1c6\" - both module.overture_tiles.aws_vpc.batch[0] and the live vpc-9038e1c6 it was computed against - with \"Exit status 3\" spelled out for a pipeline; nothing was applied - overture-tiles-crossing-tiles's CORS rule still read \"*\" through s3api get-bucket-cors rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Name tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and the CORS rule read back as https://tiles.example.invalid, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (CORS back to \"*\", VPC Name tag still overture-tiles-crossing-vpc, next plan proposes no resource action, no state file left behind) so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); 16 tagged objects before and after (resourcegroupstaggingapi's cross-service search alone, floci#98 fixed); S3 bucket and OAC identities unchanged; record store intact", "test_plan": "live-plan empty after the STAGE 2d convergence apply; S3 bucket and OAC identities re-checked by value against the AWS CLI" }, - "duration_s": 419.4, + "duration_s": 443, "stage_seconds": { "cold_deploy": 56, - "day2_count": 48, - "day2_remove": 53, + "day2_count": 51, + "day2_remove": 52, "day2_rename": 47, - "day2_replace": 6, - "drift_reconverge": 7, + "day2_replace": 7, + "drift_reconverge": 8, "greenfield": 120, - "migrate": 70, + "migrate": 75, + "plan_approval": 13, "test_apply": 3, - "test_plan": 9 + "test_plan": 10 } }, "notes": "Landed 2026-08-19. Sourced via GitHub code search rather than the awesome-opentofu/Powered-by-OpenTofu lists, which turned out to be pure tooling/adopter lists with no deployable estates. OpenTofu-native evidence is in its CI rather than a genuine .tofu extension (weaker self-description than corpus-hongbomiao, which ships real .tofu files): .github/workflows/ci.yml runs tofu fmt/validate/test/tflint exclusively through opentofu/setup-opentofu - terraform never appears - and its tests use OpenTofu's own mock_provider framework. A real, tagged-release module (v1.0.0->v1.2.0) from a Linux-Foundation-adjacent geospatial project backed by AWS/Meta/Microsoft/TomTom, contributor fixes as recent as 2026-05-21. cold_deploy genuinely passes (26 resources, plain tofu apply, unmodified module). migrate is BLOCKED, not clean: live-import stamps 13 of 26 cleanly, 10 correctly UNTAGGABLE or already-ruled UNADMITTED_TYPE (#249), and 3 AWS Batch resources fail to stamp on a real floci bug - TagResource/UntagResource/ListTagsForResource (POST /v1/tags/{resourceArn}) misroutes to AppSyncController's greedy catch-all since BatchController never registers that path. Filed as lex00/floci#72 with full evidence (checked ~/checkouts/floci first, confirmed same bug on current main, no in-progress fix) - not fixed, per this session's standing instruction. test_plan is BLOCKED, deterministically asserted (Plan: 4 to add, 7 to change, 0 to destroy, every line traced) rather than reached cleanly. Also filed INTENTIUS/choudoufu#322: aws_iam_role_policy (untaggable, ServerAssignedIfAbsent name via name_prefix) escalates a single-address unbound warning into a hard Error: Listed resource with no tags that aborts the ENTIRE live-plan, not just its own address - a real blast-radius concern (one bad site takes down the whole plan) worth prioritizing. Not fixed; worked around in this crossing via the module's own name_overrides input so the script could still assert what it could reach. Stages 4-5 not attempted - both need a genuinely empty first plan, which this estate doesn't reach yet. Confirmed informationally that applying the current non-empty plan fails safely (AWS Batch's own name-uniqueness check refuses the duplicate) rather than silently corrupting anything. Merged to local main as a233497312 (fix itself: 3c183a305a); justfile gained recipe demo-corpus-overture-tiles; live/corpus-manifest.json gained the pin. UPDATE 2026-08-20: lex00/floci#72 FIXED (floci 1d469fff, published; choudoufu re-pinned to sha256:dc246b1e) and migrate now genuinely PASSES - re-run for real against that image, not inferred: 16 of 26 newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 10 skipped, where it was 13 stamped / 3 failed before. The dry run's own counts are unmoved (16 eligible, 11 VERIFIED, 5 DRIFTED, 9 UNTAGGABLE, 1 UNADMITTED_TYPE) - the bug was in the tag WRITE, never in verification. The fix is generic rather than Batch-shaped: floci already had a SharedTagsController dispatching /tags/{arn} to the TagHandler whose serviceKey matches the ARN's own service segment, and AppSync had simply claimed /v1/tags/{arn: .+} for itself; the fix lifts that dispatch into a SharedTagsDispatcher keyed on path prefix AS WELL AS service, adds a SharedTagsV1Controller, and converts AppSync into a TagHandler beside a new BatchTagHandler - so any further service on that path needs only a handler. Stage 2c now asserts three markers through the AWS CLI instead of one, including the Batch job queue's own tofu-address/tofu-estate via batch list-tags-for-resource (the very call that used to be answered by AppSync) and that the module's own create-time Project tag SURVIVED the stamp - a TagResource that replaces instead of merging is how a live object silently loses its markers. INTENTIUS/choudoufu#322 item 1 is also fixed (576990a599/b75e46c24e) and is no longer the test_plan wall. test_plan stays 'fail' but the wall MOVED and is now harder: it is no longer a non-empty plan, it is live-plan refusing to plan at all (exit 1, no plan produced), at exactly two diagnostics - 'Invalid Identity Attribute Value: Identity attribute \"arn\" contains an Account ID \"000000000000\" which does not match the provider's \"\"' followed by its consequence 'Cannot import for projection'. Filed as INTENTIUS/choudoufu#345 with full evidence. Reachable only BECAUSE the Batch resources are now stamped: projection imports one by its ARN identity and hashicorp/aws validates an identity ARN's account segment against the account the provider knows about itself, which skip_requesting_account_id = true (what every crossing script sets to reach a local emulator) leaves empty. The marker is not wrong - stage 3 re-reads the job queue's real ARN from floci through the AWS CLI and asserts the refusal names that exact string. MEASURED, NOT ASSUMED, and recorded in the script's header so it is not re-tried: setting skip_requesting_account_id = false on the estate copy alone clears this error and breaks stage 2 instead, because the provider then routes S3 bucket tag reads through S3 Control's account-prefixed virtual host (dial tcp: lookup 000000000000.127.0.0.1: no such host), taking aws_s3_bucket.tiles[0] from VERIFIED to MISSING and the estate to 15 of 26 eligible. Stages 4-5 still not attempted: they need a plan and stage 3 produces none. The script's stage 3 is rewritten to assert the new wall deterministically (nonzero exit, exactly 2 errors and no more, both texts, the ARN in them) - and a ZERO exit now fails it, so the day this is fixed the script says so instead of quietly passing. BREAK=1 (stage 2's bucket-marker control) was NOT re-run this pass; its mechanism is unchanged. UPDATE 2026-08-20 (INTENTIUS/choudoufu#345 FIXED, no floci change): the identity-ARN crash is gone. The obvious fix, skip_requesting_account_id = false on the estate copy, was re-measured for real and its earlier 'breaks stage 2 via S3 Control's account-prefixed virtual host, dial tcp: lookup 000000000000.127.0.0.1: no such host' failure is a DNS failure, not an HTTP one - confirmed by curling the same account-prefixed host directly, which fails identically before any TCP connection, so no floci server-side Host-header routing could ever have fixed it (the request never arrives). The real fix is ENDPOINT: floci already publishes localhost.floci.io as a real, public wildcard DNS domain (EmbeddedDnsServer.DEFAULT_SUFFIX, same mechanism as LocalStack's localhost.localstack.cloud) that resolves an account-ID-prefixed label to 127.0.0.1 with no floci container running at all - confirmed via dig and via curl reaching floci's S3ControlController correctly (path-based dispatch, unaffected by the account-prefixed Host). Verified against the CURRENT, unmodified floci image (be3f7ffd, sha256:8a882bcc - no re-pin, no floci commit, no floci PR). Real re-run: stage 1 PASS unchanged (26 resources). Stage 2 PASS, counts moved by exactly one resource as a direct, expected consequence (11 VERIFIED/5 DRIFTED -> 10 VERIFIED/6 DRIFTED: aws_launch_template.batch[0]'s arn now differs between the PLAIN state, written under skip_requesting_account_id = true and so account-less, and the ESTATE copy's live re-read, which now knows its account - a real difference between two provider configurations, not a wrong marker). Stage 3 (test_plan) still recorded fail by this repo's own convention (a first plan must be empty to pass) but the #345 wall itself - live-plan exiting 1 with two diagnostics and no plan at all - is gone: live-plan now exits 0 with 'Plan: 1 to add, 7 to change, 0 to destroy.', asserted deterministically address-by-address. Every line traces to an already-tracked or by-design cause, none of them new: the 1 add is the already-ruled #249 aws_cloudfront_origin_access_control UNADMITTED_TYPE gap; 6 of the 7 changes are internal/live/discovery/count.go's own documented one-time tofu-slot migration-visibility tag (bindCountByAddress's doc comment: 'visible in the plan as a tofu-slot tag being added to each member' - by design, cements on first apply), on every count-toggled ([0]) resource this module declares; the 7th, aws_s3_bucket_policy.tiles[0], is a content diff cascading from the new OAC's arn being 'known after apply' in the same plan. Verified informationally (not scored, per this repo's own convention that test_apply is scored only once test_plan is itself empty): applying the stage 3 plan succeeds (Apply complete! Resources: 1 added, 6 changed, 0 destroyed - matching the plan), and a second live-plan afterward is genuinely empty ('No changes. Your infrastructure matches the configuration.') - the estate converges in exactly one apply, confirming #345's own header claim. No Go code touched; the fix is entirely live/e2e/corpus-overture-tiles/run.sh (ENDPOINT changed from a bare IP to localhost.floci.io, skip_requesting_account_id parameterized so only the estate copy sets it false, cold deploy left untouched)." @@ -1410,16 +1444,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1428,28 +1462,30 @@ "exit_code": 0, "detail": { "cold_deploy": "39 resources, once for real", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-8a54c04755df9d5f0) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read back through the AWS CLI; the destroyed instance is confirmed gone two independent ways (0 by group-id - length(SecurityGroups), never an exit code, since this image answers a deleted id with an EMPTY LIST and exit 0 - and 0 by tofu-address tag filter). Scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-0f8f6860554cccc91 -> sg-36ce4ada55a333b89), carrying tofu-address=aws_security_group.count_test:1 with exactly one claimant, while count_test[0] kept the same id and the same marker throughout; the next plan is empty. The G0 stock oracle stood the identical 2-instance block up with plain terraform in its own working directory and its own VPC against the same idle endpoint and showed the identical shape: destroy the higher index only (Plan: 0 to add, 0 to change, 1 to destroy), create the higher index back under a new GroupId (sg-d7b20850fbae6692e -> sg-c4094d1adb1e37439), the lower index's GroupId (sg-75e308f768ff7da71) unchanged both times; torn down completely before the choudoufu side ran. SYNTHETIC BLOCK, and why: terraform-aws-rds declares no scalable knob anywhere - the root example has no count and no for_each at all, and every count in the vendored module is a boolean create toggle (var.create ? 1 : 0 and friends, grepped not guessed) - while the only genuinely scalable knobs belong to the supporting module.vpc, whose subnet families each destroy an untaggable aws_route_table_association sibling alongside the taggable aws_subnet, which is issue #410's family (corpus-overture-tiles's count-shrink extension) and would make this stage a re-measurement of that gap rather than of count scaling. So the sanctioned fallback per live/GAUNTLET.md #8, reference-ec2-vpc's Part F and corpus-iam-policy's Part G: a self-contained aws_security_group.count_test at an address nothing else names, appended after day2_remove's completed removal, of a taggable type this estate already exercises, introducing nothing that generates a secret. BREAK_COUNT=1 is the negative control and was run for real: it asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, proving the index assertion is load-bearing.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-f03da6a2c670f916f) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read back through the AWS CLI; the destroyed instance is confirmed gone two independent ways (0 by group-id - length(SecurityGroups), never an exit code, since this image answers a deleted id with an EMPTY LIST and exit 0 - and 0 by tofu-address tag filter). Scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-1247895febf8a3b8b -> sg-2e4b68f2a30525035), carrying tofu-address=aws_security_group.count_test:1 with exactly one claimant, while count_test[0] kept the same id and the same marker throughout; the next plan is empty. The G0 stock oracle stood the identical 2-instance block up with plain terraform in its own working directory and its own VPC against the same idle endpoint and showed the identical shape: destroy the higher index only (Plan: 0 to add, 0 to change, 1 to destroy), create the higher index back under a new GroupId (sg-697f57f8128edc558 -> sg-9801c05dba1e82f94), the lower index's GroupId (sg-5f9fbe3d4127b1cb7) unchanged both times; torn down completely before the choudoufu side ran. SYNTHETIC BLOCK, and why: terraform-aws-rds declares no scalable knob anywhere - the root example has no count and no for_each at all, and every count in the vendored module is a boolean create toggle (var.create ? 1 : 0 and friends, grepped not guessed) - while the only genuinely scalable knobs belong to the supporting module.vpc, whose subnet families each destroy an untaggable aws_route_table_association sibling alongside the taggable aws_subnet, which is issue #410's family (corpus-overture-tiles's count-shrink extension) and would make this stage a re-measurement of that gap rather than of count scaling. So the sanctioned fallback per live/GAUNTLET.md #8, reference-ec2-vpc's Part F and corpus-iam-policy's Part G: a self-contained aws_security_group.count_test at an address nothing else names, appended after day2_remove's completed removal, of a taggable type this estate already exercises, introducing nothing that generates a secret. BREAK_COUNT=1 is the negative control and was run for real: it asserts the WRONG instance (count_test[0]) was destroyed and reports this stage fail, proving the index assertion is load-bearing.", "day2_remove": "choudoufu: deleting module.db_default_renamed's block proposed exactly two destroys (the db instance and its own local random_id.snapshot_identifier, no cloud representation - issue #340), applied cleanly, the db instance is genuinely gone from the live account (read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on the same renamed oracle tree also proposes exactly the same two destroys; the target was chosen (see header) because its own nested module.db_instance call has no untaggable AWS-side sibling under this estate's create_db_option_group=false/create_db_parameter_group=false, unlike the shapes that surfaced issue #410 for corpus-s3-bucket-complete and corpus-overture-tiles", "day2_rename": "moved block: module.security_group renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.db_default's db instance renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", "day2_replace": "choudoufu: changing module.db's ForceNew db_name argument (plus identifier, for an observable identity change) proposed exactly one instance replace at the same declared address, cascading into its 2 cloudwatch log groups and db parameter group (all replaced, all named from identifier) - 4 to add, 4 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql) is confirmed gone and the new instance (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql-replaced) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new identifier, not the destroyed one (complete-postgresql -> complete-postgresql-replaced); the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", "drift_reconverge": "one object tampered (primary DB instance's Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag", "greenfield": "39 resources from nothing (same DELTA reduction cold_deploy itself needs - two emulator gaps, floci-io/floci#51 and lex00/floci#52), primary DB instance and security group markers verified via the AWS CLI, 39 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally (DB engine/version/class/storage/port, security-group rule count)", "migrate": "26 of 39 stamped", + "plan_approval": "one argument edited (module \"db_default\"'s tags gain Reviewed=yes - in-place, so no live id moves, and reaching exactly ONE live object because that module call runs with create_db_option_group=false and create_db_parameter_group=false, where the same edit on module.db would also reach its parameter group, option group and log groups), \"plan -out=approved.tfplan\" wrote a 134369-byte stock-format plan file whose whole change set is one update on module.db_default.module.db_instance.aws_db_instance.this[0]; the world then moved out of band (arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql's Example tag, this estate's own STAGE 5 mutation lifted, through the AWS CLI and never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming the extra row as \" module.db.module.db_instance.aws_db_instance.this[0] Update db-52C3D597C3FC4A7BAE689183\" - both module.db.module.db_instance.aws_db_instance.this[0] and the live identity it was computed against, which for an aws_db_instance is RDS's own server-minted DbiResourceId db-52C3D597C3FC4A7BAE689183 rather than the ARN or the client-chosen identifier, read off arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql through the AWS CLI and compared by value - with \"Exit status 3\" spelled out for a pipeline; nothing was applied - arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql-default-7e26b5c33311bceebccb071915 still carried no Reviewed tag, read back through rds list-tags-for-resource rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Example tag put back to its pre-tamper value and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and arn:aws:rds:eu-west-1:000000000000:db:complete-postgresql-default-7e26b5c33311bceebccb071915 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. Reverted and reconverged in P5 (Reviewed gone, the primary instance's Example tag still \"complete-postgresql\", next plan proposes no resource action) so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 26 objects before, 26 after, no state file, primary DB instance marker unmoved", "test_plan": "genuinely empty replan (No changes. Your infrastructure matches the configuration.) with no local state file. lex00/floci#120's round-trip gap, this estate's last recorded wall, is CONFIRMED FIXED: round 8 (PR #128/ff815779, ghcr.io/lex00/floci:main-20260824d sha256:25fc9687, #124's RDS colliding-port isolation) closed the last of its eight fields for this estate - module.db_default's own port (module.db and module.db_default both declare port=5432, a genuine collision; module.db_default is the second-created instance and gets its own distinct loopback bind address with the declared port honored). The other seven fields (backup_window, monitoring_interval, monitoring_role_arn, performance_insights_retention_period, engine_lifecycle_support, enabled_cloudwatch_logs_exports, max_allocated_storage) and the parameter block's apply_method were already fixed by earlier rounds (round 5 and round 6's own #120 passes) that this estate had not been re-crossed since - the artifact's recorded '3 in-place updates' detail was stale before this round's own fix even landed. Confirmed three independent ways, not merely inferred from the empty plan: a direct describe-db-parameters --source user probe of the live parameter group (autovacuum=1, client_encoding=utf8, matching config exactly, no tofu in the loop), a direct describe-db-instances probe of the second instance's own Endpoint.Port (5432, the declared port), and all eight attribute names individually confirmed absent from choudoufu's plan. INTENTIUS/choudoufu#393 (skip_final_snapshot's phantom true->false update) remains fixed, confirmed absent. Stock's own replan against its own never-deleted state file still shows tag noise plus the two parameter blocks; ruled out as a live discrepancy by the same direct API probe (informational only, not this stage's oracle - HANDOFF row 3, a property of that one state file's own apply-time fidelity)." }, - "duration_s": 711.2, + "duration_s": 730, "stage_seconds": { - "cold_deploy": 109, + "cold_deploy": 104, "day2_count": 50, "day2_remove": 91, - "day2_rename": 16, - "day2_replace": 171, + "day2_rename": 17, + "day2_replace": 170, "drift_reconverge": 8, "greenfield": 203, - "migrate": 50, - "test_apply": 4, - "test_plan": 8 + "migrate": 52, + "plan_approval": 20, + "test_apply": 5, + "test_plan": 9 } }, "notes": "RE-CROSSED FOR REAL 2026-08-21 at dbc5ebf575 (GitHub issue #346's fix), twice, the second time with an instrumented copy of the script that dumps live-plan's raw output so the diagnostics could be quoted rather than inferred. STAGES UNCHANGED at 2 of 5, and #346's fix DOES NOT REACH THIS ESTATE - which refutes the 'four estates share one wall' framing #346 was filed under. Stage 1: 'Apply complete! Resources: 39 added, 0 changed, 0 destroyed'. Stage 2: '26 of 39 eligible (20 VERIFIED + 6 DRIFTED); 13 skipped (UNTAGGABLE by provider schema)'; '26 stamped, 1 recorded (random_id.snapshot_identifier), 0 failed, 12 skipped'; the primary aws_db_instance's tofu-address/tofu-estate asserted by exact value through the AWS CLI. Stage 3 fails on exactly the same 2 diagnostics as before, same lines, same text: 'Module output not supported in static context' on main.tf:198 (cidr_blocks = module.vpc.vpc_cidr_block, inside the module CALL argument ingress_with_cidr_blocks) and 'Unable to compute static value' on the security-group module's own main.tf:197 (cidr_blocks = compact(split(',', lookup(var.ingress_with_cidr_blocks[count.index], 'cidr_blocks', join(',', var.ingress_cidr_blocks))))). Why the fix misses it, measured rather than guessed: this estate's shape has no each.value anywhere. It is COUNT-indexed, and the module output is consumed as a module-CALL argument, which travels tolerantVariables/rebuildConstructor/moduleOutputValue - a VALUE-shaped route that cannot carry a deferred parent read, because a ParentRef is not a cty.Value. #346's fix is part-shaped and lives on the identity-argument route. A separate mechanism is needed and is filed separately. Also corrected: this file previously recorded run.sh's stage-3 assertions as stale (WANT_CIDX_N=7). They are not, and were not at this commit - the committed script already asserts WANT_CIDX_N=0, WANT_DEFAULT_N=0, WANT_UNRESOLVABLE_N=0, WANT_MODOUT_N=1 and WANT_CASCADE_N=1, which is exactly what the run produced, so the script exits 0 while the estate stays blocked. PRIOR HISTORY BELOW. Landed c239792018/47778a931a (2026-08-18); migrate's fail->pass flip was re-verified 2026-08-18 against current main (cec3c4b9b1) but the committed run.sh still asserted the stale pre-fix '0 eligible' shape as a passing control rather than a real check. Follow-up pass 2026-08-18 (this entry) rewrote run.sh's own assertions to the real, current numbers, re-verified for real in a fresh isolated worktree rather than trusted from the prior note: stage 2 dry run reports '23 of 39 resource instance(s) are eligible for stamping (VERIFIED or DRIFTED)' (18 VERIFIED + 5 DRIFTED), -approve reports '23 resource(s) newly stamped, 0 already stamped, 0 failed, 16 skipped', and the primary aws_db_instance's tofu-address/tofu-estate tags are asserted by exact value straight through the AWS CLI (module.db.module.db_instance.aws_db_instance.this:0 / rds-complete-postgres). The 16 skipped: 13 untaggable by design (aws_route_table_association x9, aws_route, aws_security_group_rule, aws_iam_role_policy_attachment, random_id - no tags argument in the provider schema) and 3 are #305's still-open unadmitted-type gap. test_plan is now asserted against a real live-plan on the really-migrated estate (state file deleted first) with a BREAK=1 negative control, and the real counts differ from what the prior note here claimed: exactly 7 count-index-in-tag sites (#304, all aws_security_group_rule.ingress_with_cidr_blocks) and exactly 3 unadmitted-type sites (#305, the three default-object adopters actually created), not 35 and 5. The prior note's 35/5 figures were measured before this estate had ever actually been migrated (nothing tagged, so the 28 module.vpc sibling-indexing sites it also counted never had anything to resolve against) and before bc9ef26638 ('a resource block with a provably-zero count/for_each has no instance to refuse admission on', already on main) landed, which independently stops aws_default_vpc/aws_vpn_gateway_attachment's two count=0 sites from refusing at all - together accounting for the 35->7 and 5->3 drop. Two unrelated real floci gaps found and filed upstream on the fork: lex00/floci#51 (RDS cross-region backup replication), lex00/floci#52 (SecretsManager RotateSecret wrongly requires a Lambda ARN for RDS-managed rotation) - both worked around in the script with documented EMULATOR GAP deltas so stage 1 could stand up at all. Follow-up pass 2026-08-18 (#313 cross-check, fresh worktree, real run): confirmed the 7 count-index-in-tag sites above are NOT #313's wall. This estate's data.aws_availability_zones usage only feeds local.azs = slice(..., 0, 3), a statically-known length, never a for_each/count keyed on the AZ name values themselves, so #313's static-context diagnostic cannot fire here (grepped the full raw plan output: zero occurrences). Commented on #313 (ruling this estate out, with the slice-vs-for_each distinction as evidence) and re-confirmed #304 is still the sole test_plan blocker. Follow-up pass 2026-08-19: #304 fixed and merged (69038634d0/9aaca0ee10) - internal/live/lint/count_index_domain.go's domain check was evaluating a whole module-call-argument value as one pass/fail unit, so one refused reference anywhere inside a list-of-objects argument poisoned every attribute derived from any part of it, even ones that never read the refused field. New StaticEvaluator.EvaluateStructural/EvalContextTolerant validate each reference individually. Re-verified for real: count-index-in-tag sites on this estate 7->0. Estate still does NOT reach test_plan clean, though: 14 sites (from_port/to_port/protocol/cidr_blocks) now fail a DIFFERENT diagnostic, 'Identity not resolvable from configuration' - internal/live/identity/partialargs.go's tolerantVariables deliberately covers only count/for_each key-set resolution, not per-attribute identity-value rendering (its own doc comment records a past regression from broadening it carelessly). Filed as #323, not attempted - needs its own scoped pass. Plus 18 genuinely-ambiguous element(aws_subnet.*[*].id, count.index) sites, confirmed still correctly refusing (unweakened by #304's fix). #304 left open (not closed) since the titled bug is fixed and verified but this estate still doesn't reach a clean plan for the separate #323 reason. Follow-up pass 2026-08-19 (#321 re-verification, scouting only, no commit): #321's fix generalizes strongly to this second, independent estate - 15 of the 18 element() sites now resolve cleanly (every aws_route_table_association.{public,private}.subnet_id/route_table_id). 3 remain: aws_route_table_association.database's route_table_id goes through coalescelist(A[*].id, B[*].id) wrapping the splat, outside resolveElementCall's bare-splat requirement - the same out-of-scope shape #321's own closing comment already flagged, now confirmed reaching a second real module composition. One NEW site found: aws_security_group_rule.ingress_with_cidr_blocks[0].security_group_id via local.this_sg_id = concat(A.*.id, B.*.id, [\"\"])[0] - concat()+splat+index through a local value, terraform-aws-modules/security-group's universal accessor, high-leverage since every rule resource that module creates uses it. Both new shapes filed as #324, not attempted. #323's 14 sites confirmed unchanged in count, root cause refined: this estate's trigger is cidr_blocks = module.vpc.vpc_cidr_block (a module-output reference into a resource's config-derived attribute) poisoning the whole variable projection, not the lookup()-into-bundled-table pattern #304 fixed - same tolerantVariables scope boundary, a second concrete trigger. Net: test_plan stays fail, 18 diagnostic sites total (4 unresolvable-identity + 7 module-output-static + 7 compute-static, mapping to #324's 4 sites + #323's 14). Three separately-scoped resolver passes stand between this estate and five-of-five, not one. live/e2e/corpus-rds-complete-postgres/run.sh's stage-3 assertions are now stale (still expect #304's old 7-site count-index picture) and need a real update pass, not done here. Follow-up pass 2026-08-19 (#324 item 2 fixed and merged, 80d3766b3e/79ffbe4732): concat(A[*].id, B[*].id, [literal])[N] through local.this_sg_id now resolves generically (reuses #321's own splat/instance-count machinery, handles both the RelativeTraversalExpr and IndexExpr parse shapes HCL produces for a constant vs non-constant index). Confirmed by real absence from this estate's own stage-3 output. Generalizes hard: refusal-probe over the full 250-entry offline corpus shows 'Identity not resolvable from configuration' 67 -> 42 (-25 sites), zero regressions, 15 offline corpus entries improved (all terraform-aws-rds examples, plus autoscaling/complete, ecs/complete, ecs/ec2-autoscaling, lambda/with-vpc-s3-endpoint - offline corpus entries, not necessarily live-crossed estates; corpus-security-group-complete and eks/examples/* unchanged). Item 1 (coalescelist) explicitly left open, unattempted. This estate itself: fixing the concat site surfaced a SEPARATE, previously-masked cascade - module.vpc.vpc_cidr_block feeding module.security_group's var.ingress_with_cidr_blocks, 'Module output not supported in static context' - likely another instance of #313's own deliberately out-of-scope resource-attribute boundary (the same family as corpus-security-group-complete's remaining 7 sites), not yet formally confirmed as such or filed separately. test_plan stays fail; the estate's own crossing script needs a real staleness-update pass (still asserts #304's old picture) before its true current diagnostic count can be read cleanly. Follow-up pass 2026-08-19: #324 item 1 (coalescelist) also fixed and merged (c25957cbdf/49744a5617) - #324 now fully closed, both items. The exact 3 aws_route_table_association.database sites this issue named are confirmed gone from this estate's real live-plan output. Generalizes narrowly but cleanly beyond this estate: refusal-probe shows -14 sites across exactly 2 offline corpus entries (cross-region-replica-postgres, vpc/examples/issues - the latter matching #321's own predicted 8-site count exactly). test_plan still stays fail here - blocked only by the pre-existing, unrelated module-output cascade already noted (likely #313's family, unconfirmed) and #323's still-open 14 sites. All of #321/#324's derivable element/splat/concat/coalescelist work is now done across this estate; what remains needs #323's own dedicated pass plus resolving the module-output cascade, not further quick derivations. Follow-up pass 2026-08-19: #323 fixed and merged (3d62366625/fb95168e63), closed. tolerantVariables now resolves a static leaf independently of a sibling leaf's genuine unresolvability, instead of the whole variable projection being poisoned by one bad reference - traced to configs.staticScopeData.GetInputVariable's own error bail discarding every known leaf along with the one genuinely unknown one. Real crossing re-verified: stage-3 identity refusals 14 -> 2 (both are the SAME underlying cause counted twice - module-output-not-supported and unable-to-compute-static-value both trace to cidr_blocks = module.vpc.vpc_cidr_block -> aws_vpc.this[0].cidr_block). CONFIRMED: this estate's sole remaining test_plan blocker is #313's root cause B (a resource-attribute reference through a module output), the exact same maintainer-scoped-out boundary blocking corpus-security-group-complete's own last 7 sites - not a bug, not derivable further, a pure scope decision. Generalizes cleanly: refusal-probe -204 sites across 14 offline corpus entries (11 rds examples, 2 ecs examples, autoscaling/complete), zero instances gained/lost anywhere, zero regressions - improves diagnostics, unblocks nothing further by itself (as expected, since the poisoning fix doesn't touch the genuinely-unresolvable leaf). run.sh's stage-3 assertions are now confirmed stale in a new way too: WANT_CIDX_N=7 has actually been 0 since #304 landed, not just uncounted - needs a real update pass reflecting the estate's true current picture (count-index 0, unadmitted-type 0, both remaining diagnostics tracing to the single #313-root-cause-B leaf), deliberately left to the orchestrator's own call rather than the fixing agent's." @@ -1474,16 +1510,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1492,26 +1528,28 @@ "exit_code": 0, "detail": { "cold_deploy": "30 resources added by plain terraform, 4 buckets confirmed live, no tofu-address tag", - "day2_count": "synthetic block (this estate's root configuration declares no count or for_each at all, and its only multi-instance knob - the intelligent_tiering input's two-entry map - fans out onto aws_s3_bucket_intelligent_tiering_configuration, which has no tags argument, issue #410's untaggable-child shape; so aws_s3_bucket.count_test, a taggable type this estate already exercises, at an address no other stage uses). choudoufu: scaling count_test from 2 to 1 proposed exactly one destroy (0 add, 0 change, 1 destroy), and it was the HIGHER index - count_test[1], s3-bucket-complete-count-test-1 - with count_test[0] not appearing in the plan at all; applied cleanly (0 added, 0 changed, 1 destroyed); head-bucket on s3-bucket-complete-count-test-1 then fails outright, while the survivor s3-bucket-complete-count-test-0 keeps both its CreationDate (2026-09-06T01:17:42+00:00) and its tofu-address=aws_s3_bucket.count_test:0 marker, read through the AWS CLI rather than choudoufu's own report, and count_test[1]'s local record is tombstoned (has tombstone, no identity - the #398-guard shape) rather than deleted. Scaling back from 1 to 2 proposed exactly one create (1 add, 0 change, 0 destroy) for count_test[1] alone, and the recreated bucket is genuinely a NEW object: same deterministic name (an S3 bucket's id IS its own name - probed against floci with no tofu in the loop) but a new CreationDate (2026-09-06T01:17:42+00:00 -> 2026-09-06T01:18:18+00:00), re-stamped tofu-address=aws_s3_bucket.count_test:1 (tags do not survive a delete+recreate on this image, so the marker is one the apply wrote again), and a record identity naming the recreated bucket; count_test[0]'s CreationDate and marker were unchanged at every step. The next plan proposes no resource action. G-ORACLE, plain terraform standing the identical 2-instance block up for real in the idle greenfield-oracle account, shows the identical shape: destroy the higher index only (0 add, 0 change, 1 destroy), recreate it under the same deterministic bucket name but a new CreationDate (2026-09-06T01:14:51+00:00 -> 2026-09-06T01:14:59+00:00), index 0's CreationDate (2026-09-06T01:14:51+00:00) unchanged at both steps. BREAK_COUNT=1 asserts the wrong instance (count_test[0]) was destroyed and reports this stage fail, as the stage's Break text requires.", - "day2_remove": "choudoufu: deleting module.simple_bucket_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy: the bucket and its untaggable public_access_block child), applied cleanly (0 added, 0 changed, 2 destroyed), the bucket is genuinely gone from the live account (head-bucket on simple-heroic-terrier now fails, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state also proposes exactly the same two destroys for the same two objects; the target was chosen to avoid issue #404's shape (a sibling policy re-reading the removed bucket's own ARN) - module.log_bucket and module.s3_bucket are both left untouched", + "day2_count": "synthetic block (this estate's root configuration declares no count or for_each at all, and its only multi-instance knob - the intelligent_tiering input's two-entry map - fans out onto aws_s3_bucket_intelligent_tiering_configuration, which has no tags argument, issue #410's untaggable-child shape; so aws_s3_bucket.count_test, a taggable type this estate already exercises, at an address no other stage uses). choudoufu: scaling count_test from 2 to 1 proposed exactly one destroy (0 add, 0 change, 1 destroy), and it was the HIGHER index - count_test[1], s3-bucket-complete-count-test-1 - with count_test[0] not appearing in the plan at all; applied cleanly (0 added, 0 changed, 1 destroyed); head-bucket on s3-bucket-complete-count-test-1 then fails outright, while the survivor s3-bucket-complete-count-test-0 keeps both its CreationDate (2026-09-07T02:46:16+00:00) and its tofu-address=aws_s3_bucket.count_test:0 marker, read through the AWS CLI rather than choudoufu's own report, and count_test[1]'s local record is tombstoned (has tombstone, no identity - the #398-guard shape) rather than deleted. Scaling back from 1 to 2 proposed exactly one create (1 add, 0 change, 0 destroy) for count_test[1] alone, and the recreated bucket is genuinely a NEW object: same deterministic name (an S3 bucket's id IS its own name - probed against floci with no tofu in the loop) but a new CreationDate (2026-09-07T02:46:16+00:00 -> 2026-09-07T02:46:52+00:00), re-stamped tofu-address=aws_s3_bucket.count_test:1 (tags do not survive a delete+recreate on this image, so the marker is one the apply wrote again), and a record identity naming the recreated bucket; count_test[0]'s CreationDate and marker were unchanged at every step. The next plan proposes no resource action. G-ORACLE, plain terraform standing the identical 2-instance block up for real in the idle greenfield-oracle account, shows the identical shape: destroy the higher index only (0 add, 0 change, 1 destroy), recreate it under the same deterministic bucket name but a new CreationDate (2026-09-07T02:42:47+00:00 -> 2026-09-07T02:42:55+00:00), index 0's CreationDate (2026-09-07T02:42:47+00:00) unchanged at both steps. BREAK_COUNT=1 asserts the wrong instance (count_test[0]) was destroyed and reports this stage fail, as the stage's Break text requires.", + "day2_remove": "choudoufu: deleting module.simple_bucket_renamed's block proposed exactly two destroys (0 add, 0 change, 2 destroy: the bucket and its untaggable public_access_block child), applied cleanly (0 added, 0 changed, 2 destroyed), the bucket is genuinely gone from the live account (head-bucket on simple-smart-raven now fails, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state also proposes exactly the same two destroys for the same two objects; the target was chosen to avoid issue #404's shape (a sibling policy re-reading the removed bucket's own ARN) - module.log_bucket and module.s3_bucket are both left untouched", "day2_rename": "moved block: module.cloudfront_log_bucket renamed to module.cloudfront_log_bucket_renamed with zero churn (0 add, 1 change, 0 destroy), the bucket's tofu-address marker rewritten in place; live-mv: module.simple_bucket renamed to module.simple_bucket_renamed with zero churn, marker rewritten in place; both live bucket names unchanged, read via the AWS CLI; the post-rename plan proposes no resource action", - "day2_replace": "choudoufu: changing module.log_bucket's ForceNew bucket argument proposed exactly one bucket replace at the same declared address, cascading into its ownership_controls, policy and public_access_block (all replaced) plus module.s3_bucket's own logging target_bucket (updated in-place) - 4 to add, 1 to change, 4 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old bucket (logs-heroic-terrier) is confirmed gone and the new bucket (logs-heroic-terrier-replaced) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new bucket, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Live resource displaced from the address it is marked for\", naming the manufactured bucket, proposing nothing for it) rather than silently proposed as nothing - the name-derived-identity shape of this diagnostic, distinct from EC2/SQS's fungible-set \"Two live resources claiming one slot\" because aws_s3_bucket's identity resolves straight from the config's own computed name rather than only through a marker sweep. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones.", + "day2_replace": "choudoufu: changing module.log_bucket's ForceNew bucket argument proposed exactly one bucket replace at the same declared address, cascading into its ownership_controls, policy and public_access_block (all replaced) plus module.s3_bucket's own logging target_bucket (updated in-place) - 4 to add, 1 to change, 4 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old bucket (logs-smart-raven) is confirmed gone and the new bucket (logs-smart-raven-replaced) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new bucket, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Live resource displaced from the address it is marked for\", naming the manufactured bucket, proposing nothing for it) rather than silently proposed as nothing - the name-derived-identity shape of this diagnostic, distinct from EC2/SQS's fungible-set \"Two live resources claiming one slot\" because aws_s3_bucket's identity resolves straight from the config's own computed name rather than only through a marker sweep. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment and corpus-ec2-instance-complete's/corpus-sqs-basic's matching ones.", "drift_reconverge": "accelerate config drifted to Enabled, exactly 1 change proposed and applied, reconverged to Suspended, final plan empty", "greenfield": "29 resources from nothing (SCOPE REDUCTION's own reduced count, random_pet pinned to a literal on both sides), 3 of 4 bucket markers verified via the AWS CLI, 26 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on all 4 buckets (versioning, default encryption, policy presence)", "migrate": "6 of 30 stamped, 1 recorded (random_pet, issue #340), 23 skipped (untaggable), 0 failed, 26 identities recorded (#364 unit A2); markers survived the residue-classification apply", + "plan_approval": "one argument edited (module.cloudfront_log_bucket gains tags = { Reviewed = \"yes\" }, reaching aws_s3_bucket.this[0] and nothing else the module creates - that module call sets no attach_*_policy input, so the module's own policy-document data sources are count = 0 for it and one tag is one row), \"plan -out=approved.tfplan\" wrote a 84803-byte stock-format plan file whose whole change set is one update on module.cloudfront_log_bucket.aws_s3_bucket.this[0]; the world then moved out of band (s3-bucket-smart-raven's transfer-acceleration status flipped to Enabled through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live bucket from the one under review) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.s3_bucket.aws_s3_bucket_accelerate_configuration.this[0] and the live s3-bucket-smart-raven it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - get-bucket-tagging on cloudfront-logs-smart-raven still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the accelerate status put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and cloudfront-logs-smart-raven read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); bucket count unchanged at 4", "test_plan": "no resource action proposed; 29 rendered identity occurrences (11 distinct), all naming known roots" }, - "duration_s": 484.5, + "duration_s": 511, "stage_seconds": { - "cold_deploy": 89, + "cold_deploy": 73, "day2_count": 56, - "day2_remove": 19, - "day2_rename": 23, - "day2_replace": 21, + "day2_remove": 20, + "day2_rename": 22, + "day2_replace": 20, "drift_reconverge": 20, - "greenfield": 165, - "migrate": 77, + "greenfield": 169, + "migrate": 81, + "plan_approval": 35, "test_apply": 11, "test_plan": 4 } @@ -1538,16 +1576,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1556,28 +1594,30 @@ "exit_code": 0, "detail": { "cold_deploy": "67 resources (DELTA 2, lex00/floci#57)", - "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-294fab44394fefbdc) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-e5ba6f0e8c24d339b, was sg-82428bbcae2cf1efa - a security group id is never reused, so the destroy is directly observable rather than inferred) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] kept both its id and its marker throughout; the local record store tracked the same facts by value at every step (index 0's record never moved, index 1's stopped naming the destroyed object and then named the new one); the next plan is empty. G-ORACLE is the stock leg for the identical shape, applied for real with plain terraform in the idle greenfield account: 3 created, then 0 add/0 change/1 destroy hitting count_test[1] only, then 1 add/0 change/0 destroy bringing it back under a new id, count_test[0]'s id unchanged both times - identical to choudoufu's. SYNTHETIC BLOCK, and why: this estate declares no scalable knob of its own - every count in terraform-aws-security-group v6.0.0 is the boolean 'local.create ? 1 : 0' toggle, and its real for_each maps (var.ingress_rules/var.egress_rules) cannot be scaled in isolation because main.tf line 96 feeds every rule id into aws_vpc_security_group_rules_exclusive.this[0], making the plan 0 add/1 change/1 destroy instead of the exact shape this stage asserts; aws_security_group.count_test is self-contained, named by nothing else here, and of a type this estate already exercises six times over (reference-ec2-vpc Part F and corpus-iam-policy Part G are the precedent). BREAK_COUNT=1 runs this stage's Break control (expect count_test[0] to be the destroyed one) and correctly fails to hold.", + "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live GroupId (sg-0999dad1a4c3c06d5) and its tofu-address=aws_security_group.count_test:0 marker unchanged, both read through the AWS CLI; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) under a NEW server-minted GroupId (sg-2e2ea3c3128e173a8, was sg-6d6ee0da0cdf72206 - a security group id is never reused, so the destroy is directly observable rather than inferred) carrying tofu-address=aws_security_group.count_test:1, while count_test[0] kept both its id and its marker throughout; the local record store tracked the same facts by value at every step (index 0's record never moved, index 1's stopped naming the destroyed object and then named the new one); the next plan is empty. G-ORACLE is the stock leg for the identical shape, applied for real with plain terraform in the idle greenfield account: 3 created, then 0 add/0 change/1 destroy hitting count_test[1] only, then 1 add/0 change/0 destroy bringing it back under a new id, count_test[0]'s id unchanged both times - identical to choudoufu's. SYNTHETIC BLOCK, and why: this estate declares no scalable knob of its own - every count in terraform-aws-security-group v6.0.0 is the boolean 'local.create ? 1 : 0' toggle, and its real for_each maps (var.ingress_rules/var.egress_rules) cannot be scaled in isolation because main.tf line 96 feeds every rule id into aws_vpc_security_group_rules_exclusive.this[0], making the plan 0 add/1 change/1 destroy instead of the exact shape this stage asserts; aws_security_group.count_test is self-contained, named by nothing else here, and of a type this estate already exercises six times over (reference-ec2-vpc Part F and corpus-iam-policy Part G are the precedent). BREAK_COUNT=1 runs this stage's Break control (expect count_test[0] to be the destroyed one) and correctly fails to hold.", "day2_remove": "choudoufu: deleting module.postgresql_renamed's block proposed exactly 5 destroys (0 add, 0 change, 5 destroy: SG + 2 ingress + 1 egress + 1 untaggable rules_exclusive), applied cleanly (0 added, 0 changed, 5 destroyed), the security group is genuinely gone from the live account (0 matches on describe-security-groups for the old id, read via the AWS CLI, not choudoufu's own report), and the next plan proposes nothing; stock oracle on cold_deploy's own state (D-ORACLE remove) also proposes exactly 5 destroys for the same 5 objects; classifyOrphans did not withhold the untaggable rules_exclusive destroy even though module.security_group's and module.consul's own rules_exclusive instances share its block key, because both surviving instances are bound, not unclaimed", "day2_rename": "moved block: module.postgresql renamed to module.postgresql_renamed with zero churn (0 add, 4 change, 0 destroy) - the rule-children case, its own SG plus ingress/egress rules and rules_exclusive all moving under one moved block; live-mv: aws_security_group.app renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.security_group's ForceNew name argument proposed exactly one SG replace at the same declared address, cascading into its 7 ingress rules, 1 egress rule and 1 rules_exclusive enforcer (all replaced) - 10 to add, 10 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old SG (sg-b281cdf25f7c3e452) is confirmed gone and the new SG (sg-74ab044e1dc941d6e) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new SG, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Indistinguishable instances without per-instance markers\", naming both live security groups) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", + "day2_replace": "choudoufu: changing module.security_group's ForceNew name argument proposed exactly one SG replace at the same declared address, cascading into its 7 ingress rules, 1 egress rule and 1 rules_exclusive enforcer (all replaced) - 10 to add, 10 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old SG (sg-5eb49924036c81c44) is confirmed gone and the new SG (sg-9260d7ad3ccb5142b) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new SG, not the destroyed one; the next plan proposes no resource action; BREAK=replace confirms a manufactured marker collision is reported loudly (\"Indistinguishable instances without per-instance markers\", naming both live security groups) rather than silently proposed as nothing - internal/live/discovery/supersededclaimant.go (#849) tombstones only what an apply actually destroyed, so a live duplicate with no tombstone is never pruned away. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one object tampered (DriftProbe tag on the main security group), exactly module.security_group.aws_security_group.this[0] proposed, apply changed 1 and the tag is gone, confirmed via the AWS CLI", "greenfield": "67 resources from nothing, all markers verified via the AWS CLI, 67 records in the local record store (#364 A2), replan empty, 6 tagged security groups (4 named + 2 default adopters) and every named one's rule shape matches $PLAIN_EST's own stage-1 apply object by object, tags stripped", "migrate": "58 of 67 stamped, 67 identities recorded (#364 unit A2)", + "plan_approval": "one argument edited (aws_security_group.app's tags merge in Reviewed=yes; the standalone app SG has no rule children, so one argument is one row), \"plan -out=approved.tfplan\" wrote a 87608-byte stock-format plan file whose whole change set is one update on aws_security_group.app (sg-641be14b280c2e82e); the world then moved out of band (a DriftProbe tag on sg-5eb49924036c81c44 through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live security group from the one under review) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.security_group.aws_security_group.this[0] and the live sg-5eb49924036c81c44 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - describe-tags on sg-641be14b280c2e82e still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the DriftProbe tag deleted and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-641be14b280c2e82e read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the app SG confirmed to be the same GroupId it started as, and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 58 objects, read through resourcegroupstaggingapi", "test_plan": "the plan is genuinely empty: every choudoufu wall (#305, #307, #313 A and B, #321, #332) and both confirmed floci gaps (#102, #104) are fixed or absent this run; default route table identities asserted by value against the AWS CLI in step 3a" }, - "duration_s": 210.6, + "duration_s": 218.7, "stage_seconds": { - "cold_deploy": 31, - "day2_count": 29, - "day2_remove": 10, - "day2_rename": 13, + "cold_deploy": 18, + "day2_count": 30, + "day2_remove": 11, + "day2_rename": 14, "day2_replace": 11, - "drift_reconverge": 6, - "greenfield": 27, - "migrate": 75, + "drift_reconverge": 7, + "greenfield": 26, + "migrate": 74, + "plan_approval": 18, "test_apply": 4, - "test_plan": 4 + "test_plan": 5 } }, "notes": "Landed c876435875 (2026-08-18), 67 real resources. The brief expected this crossing to hit #304 (a static lookup()-keyed count-index) directly, since a prior crossing found #304 through this same module as a dependency - checked, not assumed, and refuted: v6.0.0 rewrote the module from the classic single-aws_security_group-with-dynamic-blocks shape #304 lives in to a for_each-over-a-map shape emitting aws_vpc_security_group_ingress_rule/egress_rule per rule key, so that whole pattern is gone from this version's own example. migrate genuinely passes (52 of 67 eligible and stamped: 35 VERIFIED + 17 DRIFTED; 15 skipped - 6 untaggable by design, 9 unadmitted). test_plan blocked by two real, distinct, filed gaps: #305 (6 sites, the familiar default_* adopter trio, doubled since this estate nests two terraform-aws-vpc calls) and a NEW one, filed as #307: aws_vpc_security_group_rules_exclusive is unadmitted (3 sites) - no CFN counterpart and the pinned provider ships no identity schema for it, but its own import docs are unambiguous that security_group_id (required, ForceNew, always a direct parent reference) is its whole identity, the same shape aws_vpc_security_group_vpc_association's already-admitted row has. Also found and filed a real floci gap (lex00/floci#57: EC2 AssociateSecurityGroupVpc has no handler at all), worked around with a documented delta removing the estate's one vpc_associations block (67 of 68 resources still stand up for real). One open, honestly-unresolved observation: 17 of the DRIFTED resources show referenced_security_group_id read back as \"000000000000/sg-xxx\" from floci where config computes a bare \"sg-xxx\" - doesn't block stamping, but live-plan never got past #305/#307's hard refusals to reveal whether the real provider's diff-suppression absorbs this cleanly or would surface as an 18th change. Not filed separately; the next crossing attempt (once #305/#307 land) will show it for real. Follow-up pass 2026-08-19 (#313's data-source fix, c636ab20f7/0284d8c408): re-verified for real against the current pin (67 cold-deployed, 58 stamped, state deleted). test_plan diagnostics dropped from 239 to 19 - #313's canonical data.aws_availability_zones cause (50 sites) is fully gone, the module-output/resource-attribute variant (2 sites) still correctly refuses (out of scope by the maintainer's own ruling), and the 187-site cascade collapsed to 5. The remaining 12 sites are a NEW, newly-reached class (previously masked behind #313's hard refusal): element([*].id, count.index) in the vpc module's aws_route_table_association.private - both operands are tagged resources, the obstacle is the splat-through-function-call shape, not an admission gap. Filed as #321, not attempted (scouting slot). test_plan therefore stays fail, but the estate's real remaining blocker is now #321 (a derivable, no-design-call-needed gap) rather than #313 (an architecture question) - #321 is the clear next step toward this estate's five-of-five and the core set's last gap. Follow-up pass 2026-08-19 (#321 fixed and merged, c33a47288a/626ca84739): element([*].attr, idx) over a splat of tagged resources now resolves generically - it names the same live object a direct indexed traversal already resolves, via element()'s own modulo wraparound. Re-verified for real: test_plan diagnostics 19 -> 7. The 12 splat-through-element sites are confirmed cleared by their absence from real live-plan output; the remaining 7 are #313's own deliberately-out-of-scope resource-attribute root cause (root cause B), a maintainer scope boundary, not a bug. test_plan stays fail - the core set does NOT reach five-of-five from this fix alone. Generalizes beyond this estate: refusal-probe over terraform-aws-modules/vpc's own examples, 21 -> 8 sites, zero regressions - three configs (ipam, ipv6-only, outpost) fully cleared. A related but distinct lint-side wall (RuleCountIndex, 32 sites/6 configs, a genuine unresolved composite-vs-per-argument injectivity design question) stays independently blocking and was left open, documented rather than attempted. Follow-up pass 2026-08-19 (#191 fixed and merged, 312acbbb61/75ef0a6a78): internal/live/identity/partialargs.go's tolerant rebuild now composes across more than one module call and evaluates a call the caller wrote (merge(), not a bare constructor) through an evaluator whose own var.* closure is already tolerant, one module up. module.consul's ingress_referenced_security_group_id map no longer poisons the 22 ingress rules it seeds two module calls down - the map's KEYS (eleven preset names crossed with one caller key) were always written down; only the VALUE under one key was ever unknowable, and it still is. Re-verified for real against floci, script exit 0, BREAK=1 negative control correctly fails: test_plan diagnostics 7 -> 4, and every analysis-layer refusal this estate has ever hit is now 0 and asserted by absence (#305, #307, #313 root causes A and B both, #321). What newly reached PROJECTION and blocks the estate now is #332 (not #313 - the old 'Unable to use aws_security_group.app in static context' framing is confirmed gone): aws_default_route_table imports by the VPC's id, not its own, and the ratified row says otherwise - 2 'Cannot import for projection' + 2 'empty result', one pair per nested vpc module call, both traced to the same type. #332 is filed, not fixed here; it is now the sole remaining blocker on this estate and on the core set's last five-of-five gap. #332 fixed 2026-08-19 (859c1ad747/ff1f6bcdea/c1197befc7): the ratified row claimed aws_default_route_table imports by the route table's own rtb-… id; the real provider imports it by the VPC's id, read off the vpc_id ATTRIBUTE (not argument) the discovered object already carries - settled by running stock terraform 1.15.8 + hashicorp/aws 6.59.0 (Error: empty result for rtb-…, Import successful! for vpc-…). Reach stated honestly: one type today (defaultAdopterSiblings/sameRatifiedIdentity in internal/live/discovery/discovery.go split \"same live object\" from \"same import identity\" generically, off each type's own ratified IdentityAttrs/ImportSyntax, no type name in the control flow - #302's aws_iam_service_linked_role/aws_iam_role pair already exercises the same recomposition path; aws_default_route_table is simply the only aws_default_* row that currently diverges from its plain sibling at aws 6.59.0). Re-verified independently 2026-08-20 with a fresh real crossing against floci (ghcr.io/lex00/floci@sha256:120b6783c7fb48d3d78245056251492b7d9246cdc3b397c98db2659cdc78d94a, the currently pinned image): STAGE 1 PASS, STAGE 2 PASS, STAGE 3 BLOCKED at exactly 1 site (was 239, then 19, then 7, then 4) - every choudoufu-layer refusal this estate has ever hit (#305, #307, #313 root causes A and B, #321, #332) confirmed absent, and step 3a re-derives each nested VPC's default route table import identity from AWS directly (module.vpc: vpc-9eceebf0 -> rtb-7a9e0d017620e163c; module.vpc_secondary: vpc-91d40754 -> rtb-b66d5dc82e63b6e38) and asserts it BY VALUE, not by absence. The 1 remaining site is the AWS provider answering \"Provider produced invalid plan\" on its own requires-replacement path for module.security_group.aws_vpc_security_group_ingress_rule.this[\"dns-from-prefix-list\"] (cty.Path{cty.GetAttrStep{Name:\"\"}}), explicitly a provider bug per the error's own text - filed upstream as #335, genuinely outside this fork's code (the diagnostic names no choudoufu path). test_plan stays \"fail\" in this table's pass/fail/not_run vocabulary since the plan is not clean, but the estate's own blocker has moved entirely off this fork: #332 was the core set's last derivable gap, and #335 (an AWS-provider defect) is what now stands between this estate and five-of-five." @@ -1602,16 +1642,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "cb5ae2009fb85fdafd76126bb1842144855eb9d7", - "date": "2026-09-06T04:29:56Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1623,24 +1663,26 @@ "day2_count": "choudoufu: dropping \"2024\" from module.rustconf_com's CNAME for_each map destroyed exactly module.rustconf_com.aws_route53_record.cname[\"2024\"] (0 add, 0 change, 1 destroy), leaving sibling module.rustconf_com.aws_route53_record.cname[\"2022\"]'s TTL and 27 remaining record sets untouched; adding it back created exactly the same key (0 add, 0 change -> 1 add, 0 change, 0 destroy), restoring its TTL/value and the 28 record-set count, while the sibling and the parent zone's own marker stayed untouched throughout; the next plan is empty; a Route 53 record set carries no server-minted identifier of its own (verified directly against floci, no tofu in the loop: ListResourceRecordSets returns a byte-identical entry across a genuine delete/recreate, only ChangeResourceRecordSets' own per-call ChangeInfo.Id differs), so the destroy is proven by verified ABSENCE rather than an id-diff, unlike this stage's aws_iam_policy/PolicyId and EC2/VpcEndpointId precedents; the G-ORACLE stock oracle on the identical for_each change, plan-only on cold_deploy's own state, shows the identical shape: destroy the dropped key only, propose creating it back, every sibling key untouched both times", "day2_remove": "choudoufu: deleting module.cratesio_com_final's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the hosted zone is genuinely gone from the live account (route53 get-hosted-zone on the old id now errors, read via the AWS CLI, not choudoufu's own report; 7 zones down to 6), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same zone (before any rename ever touched it)", "day2_rename": "moved block: module.rustaceans_org renamed to module.rustaceans_org_moved with zero churn (0 add, 1 change, 0 destroy) - only the zone's own marker rewritten, its 2 record children (A, CNAME) did not move; live-mv: module.cratesio_com (0 records) renamed to module.cratesio_com_final with zero churn, marker rewritten in place; stock oracle over the identical two-module rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy), using per-child moved blocks stock's own state-address tracking requires and choudoufu's stateless untaggable-record derivation does not; both live zone ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.areweasyncyet_rs's ForceNew domain argument proposed exactly one zone replace at the same declared address, cascading into its one A record - 2 to add, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old zone (ZGL45ZHYYL0082N) is confirmed gone and the new zone (Z0ABC41F7VX38G5) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new zone, not the destroyed one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", + "day2_replace": "choudoufu: changing module.areweasyncyet_rs's ForceNew domain argument proposed exactly one zone replace at the same declared address, cascading into its one A record - 2 to add, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old zone (Z27X0DRHB3FI7WN) is confirmed gone and the new zone (ZMBCC3J2JR88MMI) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new zone, not the destroyed one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", "drift_reconverge": "one untaggable record drifted, exactly module.rustconf_com.aws_route53_record.cname[\"2016\"] proposed and applied, TTL reconverged to 300, 28 records and the parent marker intact", "greenfield": "35 instances from nothing (7 zones, 28 records), all 7 markers verified via the AWS CLI, replan empty, stock oracle in its own namespace matches structurally on all 7 zones (28 records)", "migrate": "7 stamped, 7 distinct hosted zones, one per module call", + "plan_approval": "one argument edited (module.arewewebyet_org's ttl 300 -> 600; that call declares exactly one record, the www CNAME, so one argument is one instance), \"plan -out=approved.tfplan\" wrote a 23430-byte stock-format plan file whose whole change set is one update on module.arewewebyet_org.aws_route53_record.cname[\"www\"] (\"Plan: 0 to add, 1 to change, 0 to destroy\"); the world then moved out of band (2016.rustconf.com.'s TTL set to 60 in zone ZCO3SP8XZ17EY1E through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance in a DIFFERENT hosted zone from the one under review) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming module.rustconf_com.aws_route53_record.cname[\"2016\"] together with the whole live identity it was computed against, ZCO3SP8XZ17EY1E_2016.rustconf.com_CNAME (this type's ZONEID_NAME_TYPE import syntax, rebuilt in the script from the zone id, record name and type it already knew independently) - with \"Exit status 3\" spelled out for a pipeline; nothing was applied - www.arewewebyet.org. still read TTL 300 through the AWS CLI, which is stronger evidence than the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the drifted TTL put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and www.arewewebyet.org. read back at TTL 600, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the 28 record sets re-counted and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); 7 zones / 28 records unchanged, all 7 markers unmoved", "test_plan": "no resource change proposed, nothing foreign; all 35 rendered identities name a live hosted zone or record set" }, - "duration_s": 490.4, + "duration_s": 502.7, "stage_seconds": { - "cold_deploy": 85, - "day2_count": 53, - "day2_remove": 28, - "day2_rename": 18, - "day2_replace": 70, - "drift_reconverge": 22, - "greenfield": 155, - "migrate": 40, - "test_apply": 14, + "cold_deploy": 70, + "day2_count": 50, + "day2_remove": 24, + "day2_rename": 15, + "day2_replace": 68, + "drift_reconverge": 21, + "greenfield": 151, + "migrate": 41, + "plan_approval": 45, + "test_apply": 13, "test_plan": 5 } }, @@ -1666,16 +1708,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1684,28 +1726,30 @@ "exit_code": 0, "detail": { "cold_deploy": "6 resources added by plain terraform (4 queues + redrive_policy + redrive_allow_policy), 0 objects carry tofu-estate before migration", - "day2_count": "synthetic block (all four module calls declare count = var.create ? 1 : 0, a boolean create toggle, so the estate has no knob that scales - issue #488's sanctioned fallback, reusing aws_sqs_queue, a type this estate already exercises four times): scaling aws_sqs_queue.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) of the HIGHER index, count_test[1]; the survivor count_test[0] kept its live queue URL (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-count-test-0), its CreatedTimestamp (1788657850) and its tofu-address=aws_sqs_queue.count_test:0 / tofu-slot=0 markers, all read back through the AWS CLI, and count_test[1]'s local record was tombstoned rather than left naming a destroyed queue; scaling 1 back to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy), and because a queue URL is rebuilt from region + account + name the recreated instance comes back at the SAME url - so the destroy is witnessed two other ways instead, by AWS.SimpleQueueService.NonExistentQueue in between and by a strictly later CreatedTimestamp (1788657850 -> 1788657932), with tofu-address=aws_sqs_queue.count_test:1 back on the new object and index 0 untouched throughout; the next plan proposes no resource action; the G-ORACLE stock oracle stood the identical block up for real in the idle greenfield-oracle account and showed the identical shape - destroy the higher index only, create it back under the same url with a new CreatedTimestamp, the lower index unchanged both times", + "day2_count": "synthetic block (all four module calls declare count = var.create ? 1 : 0, a boolean create toggle, so the estate has no knob that scales - issue #488's sanctioned fallback, reusing aws_sqs_queue, a type this estate already exercises four times): scaling aws_sqs_queue.count_test from 2 to 1 proposed and applied exactly one destroy (0 add, 0 change, 1 destroy) of the HIGHER index, count_test[1]; the survivor count_test[0] kept its live queue URL (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-count-test-0), its CreatedTimestamp (1788749614) and its tofu-address=aws_sqs_queue.count_test:0 / tofu-slot=0 markers, all read back through the AWS CLI, and count_test[1]'s local record was tombstoned rather than left naming a destroyed queue; scaling 1 back to 2 proposed and applied exactly one create (1 add, 0 change, 0 destroy), and because a queue URL is rebuilt from region + account + name the recreated instance comes back at the SAME url - so the destroy is witnessed two other ways instead, by AWS.SimpleQueueService.NonExistentQueue in between and by a strictly later CreatedTimestamp (1788749614 -> 1788749697), with tofu-address=aws_sqs_queue.count_test:1 back on the new object and index 0 untouched throughout; the next plan proposes no resource action; the G-ORACLE stock oracle stood the identical block up for real in the idle greenfield-oracle account and showed the identical shape - destroy the higher index only, create it back under the same url with a new CreatedTimestamp, the lower index unchanged both times", "day2_remove": "choudoufu: deleting module.unencrypted_sqs_renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (sqs get-queue-url on the old name now returns NonExistentQueue, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same object (before any rename ever touched it)", "day2_rename": "moved block: module.default_sqs renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: module.unencrypted_sqs renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", "day2_replace": "choudoufu: changing module.default_sqs_renamed's ForceNew name argument proposed exactly one replace at the same declared address (1 add, 0 change, 1 destroy; -/+ destroy and then create), applied cleanly; the old object (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default) is confirmed gone and the new object (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default-v2) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new object's import_id, not the destroyed one (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default -> https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default-v2); the next plan proposes no resource action; stock oracle on cold_deploy's own state (F-ORACLE) also proposes exactly one replace at the same address (plan only, not applied - it shares floci's account with $EST); BREAK=replace confirms a manufactured marker collision is reported loudly rather than silently proposed as nothing. Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment.", "drift_reconverge": "one object tampered, exactly 1 object proposed and applied (0 added, 1 changed, 0 destroyed), tag reconverged to \"ex-complete\"", "greenfield": "6 resources from nothing (4 tagged queues + 2 untaggable redrive types), all markers verified via the AWS CLI, 6 records in the local record store (#364 A2), replan empty, stock oracle in its own namespace matches structurally on all 4 queues", "migrate": "4 of 6 eligible (2 untaggable redrive types resolved by provider identity schema), 4 stamped, 0 failed, 2 skipped; tofu-slot=0 written on all 4 queues by the stamp itself (issue #372's remainder), confirmed by value and by a genuine no-op on the follow-up apply", + "plan_approval": "one argument edited (module.unencrypted_sqs's tags gain Reviewed=yes; that module call declares no DLQ, so var.tags reaches exactly one resource instance), \"plan -out=approved.tfplan\" wrote a 26728-byte stock-format plan file whose whole change set is one update on module.unencrypted_sqs.aws_sqs_queue.this[0] (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-unencrypted); the world then moved out of band (https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default's Example tag, through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live queue from the one under review) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.default_sqs.aws_sqs_queue.this[0] and the live https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-default it was computed against (an aws_sqs_queue's identity IS its URL), with \"Exit status 3\" spelled out for a pipeline; nothing was applied - list-queue-tags on https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-unencrypted still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Example tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-unencrypted read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the unencrypted queue's URL confirmed unchanged and the estate replanned empty with no state file, so PART D starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); 4 objects before, 4 after, no state file", "test_plan": "no resource change proposed, no foreign resources; fifo and default queue tofu-address re-checked against SQS" }, - "duration_s": 636.1, + "duration_s": 632.2, "stage_seconds": { - "cold_deploy": 70, - "day2_count": 114, - "day2_remove": 50, - "day2_rename": 10, + "cold_deploy": 57, + "day2_count": 115, + "day2_remove": 49, + "day2_rename": 9, "day2_replace": 74, "drift_reconverge": 5, - "greenfield": 187, - "migrate": 121, - "test_apply": 3, - "test_plan": 2 + "greenfield": 185, + "migrate": 120, + "plan_approval": 13, + "test_apply": 2, + "test_plan": 3 } }, "notes": "First real five-stage crossing of this estate, and the first SQS surface in this corpus. Sourced and sketched at e4b12799da with every assertion past stage 1's `terraform apply` DERIVED from reading terraform-aws-sqs's naming locals rather than measured - that commit's own header said so and deliberately added no entry here. This entry is the first one written from a real run: Docker/floci (ghcr.io/lex00/floci@sha256:8a882bcc, live/floci-image's pin), real hashicorp terraform, and the AWS CLI throughout, in worktree ../wt/new-terraform-estate-2 off e4b12799da. All five stages PASS. WHAT THE DERIVATION GOT WRONG, and it was exactly one thing: the resource count. The estate builds SIX managed resources, not five. `create_dlq = true` makes the module emit an aws_sqs_queue_redrive_ALLOW_policy on the DLQ alongside the aws_sqs_queue_redrive_policy on the source queue; reading the naming locals found the second and missed the first. Everything else the sketch derived was right when checked against reality - all four queue names and URLs including the FIFO DLQ's \"-dlq.fifo\" suffix, and all four rendered tofu-address strings. Corrected counts, measured: stage 1 \"Apply complete! Resources: 6 added\", stage 2 \"4 of 6 resource instance(s) are eligible for stamping\" and \"4 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 2 skipped\". THE SCHEMA-FALLBACK RESULT, which is why this estate was sourced. Neither aws_sqs_queue_redrive_policy nor aws_sqs_queue_redrive_allow_policy has a row in internal/live/identity/table_generated.go; live/survey-full.json classifies both identically (path \"client-named\", admission \"schema\", required_for_import [\"queue_url\"], taggable false, list_resource false). Both resolved a live id through the provider's own identity schema, and both resolved to the RIGHT queue, which is not the same queue for the two of them: redrive_policy -> https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete.fifo (the source queue), redrive_allow_policy -> https://sqs.eu-west-1.amazonaws.com/000000000000/ex-complete-dlq.fifo (the DLQ). The schema-fallback path works on both, unmodified - no fix was needed and none was made. The script now asserts both by value, because a run that merely did not error would pass with the two swapped. Both are UNTAGGABLE (no tags argument in the provider schema), so live-import skips them and stage 2 stamps 4 of 6. That is the invariant working, not a shortfall: four tagged queues plus two resources whose entire identity IS a tagged queue's URL - tagged, plus derived-from-tagged, no third bucket - and stage 3's empty plan with no state file anywhere is what proves the derivation holds. EXPECTED IMPORT-TIME DRIFT, recorded so the next reader does not chase it: live-import reports the two FIFO queues as DRIFTED rather than VERIFIED, on redrive_policy and redrive_allow_policy respectively (cold state has \"\", live has the JSON). That is the module's own design - the queue resource does not manage those attributes, the separate redrive resources do - so the live object carries a value the queue's state row never recorded. DRIFTED is still eligible for stamping, the convergence apply reconciles it, and the next plan is empty. The tofu-slot convergence apply corpus-iam-policy documented recurs here exactly as that entry predicts: all four aws_sqs_queue resources declare count = var.create ? 1 : 0, so one ordinary `choudoufu apply` (\"0 added, 4 changed, 0 destroyed\") is folded into stage 2 before stage 3 is attempted. The two redrive resources are untaggable and carry no slot, which is why it is 4 changed and not 6. Stages 3-5 measured: test_plan proposes no resource action, reports \"Foreign resources: none among the 1 type swept\", both re-read identities unchanged, no state file written; test_apply \"0 added, 0 changed, 0 destroyed\" with the tofu-estate-tagged object count 4 before and 4 after; drift_reconverge tampers ex-complete-default's Example tag directly through the AWS CLI, live-plan proposes updating exactly module.default_sqs.aws_sqs_queue.this[0] and nothing else, and the apply reconverges it (\"0 added, 1 changed, 0 destroyed\", tag back to \"ex-complete\"). THREE SELF-AUTHORED DEFECTS FIXED IN THE SCRIPT WHILE VERIFYING IT, each found by running it rather than reading it. (1) No TF_PLUGIN_CACHE_DIR, which corpus-lambda-simple and corpus-alb-complete both set. Without it the first real run spent 21 minutes in `terraform init` having pulled 48MB of hashicorp/aws and then died on a transient DNS failure before ever reaching `terraform apply` - the likely reason two earlier sessions reported this crossing as stalled rather than failed. With the shared cache the whole five-stage run is minutes. (2) The BREAK contract was unreachable. BREAK=1 set two corruptions, stage 3's and stage 5's, but the stage-3 one calls fail() and exits, so stage 5's branch was dead code that had never run - and its inverted form would have exited 0 on a corrupted run anyway, proving nothing. BREAK is now three named values corrupting three different assertions, each verified for real to exit 1 at its own assertion and each reaching a later stage than the last: BREAK=schema (swap the two expected redrive URLs - both real queues, both types really do resolve, so only a by-value check catches the wrong pairing) fails in stage 2; BREAK=identity fails in stage 3; BREAK=drift fails in stage 5's exactly-one-object assertion with both objects named. An unrecognized BREAK value is rejected up front. This same dead-stage-5-branch shape exists in corpus-iam-policy's script, which this one was copied from - worth a slot there, not touched here. (3) A new managed-shape assertion added in this pass (terraform state list compared by name against the six documented addresses, so a moved corpus pin fails loudly instead of silently crossing a different estate - the exact failure mode that produced the wrong count) first failed on locale collation alone: \".\" and \"_\" sort differently under a UTF-8 locale, so a hand-ordered list never matches a locale-sorted one. Both sides now go through LC_ALL=C sort. Caught because the assertion was run, not reviewed." @@ -1730,16 +1774,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1748,28 +1792,30 @@ "exit_code": 0, "detail": { "cold_deploy": "11 managed resource instances, genuinely cold, genuinely unmarked", - "day2_count": "Scaled aws_security_group.count_test, a self-contained 2-instance count block in its own count_test.tf at an address nothing else in this crossing names. SYNTHETIC, and NOT because this estate has no scalable knob: sumaform declares a real one (var.quantity, \"number of hosts like this one\", driving count = var.quantity on aws_instance.instance), but every resource that knob scales is off the tag rung here - the instance and the EBS volume are markers = record by this crossing's own strict block, aws_volume_attachment is untaggable, and all three carry sumaform's own lifecycle { ignore_changes = [tags] } - so none of them can witness the stage's identity clause off the live object. G4 exercises the real knob for the half it can settle: quantity 1 -> 2 through choudoufu proposes exactly 3 creates, all at index [1] (instance, data disk, volume attachment), index [0] untouched, identical to stock's own plan for the same change on cold_deploy's state (G-ORACLE-QUANTITY), and the estate plans empty again once it is restored; the down direction is NOT exercised on that knob (cold_deploy's state is quantity=1, so there is no 2-instance stock state to scale down from). On the taggable block: scaling 2 -> 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy) and left count_test[0]'s live GroupId sg-f0a2ac3380efc13eb AND its tofu-address marker aws_security_group.count_test:0 unchanged, both read through the AWS CLI rather than choudoufu's report; count_test[1] (sg-043fd4f6862639ce2) verified absent by describe-security-groups length 0 in between; scaling 1 -> 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) back under a NEW GroupId sg-42031c6d2807e6eb0 carrying tofu-address=aws_security_group.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G-ORACLE): the identical 2-instance block stood up FOR REAL in its own idle floci account on FLOCI_PORT+3 - stock never had this block, so cold_deploy's state could not be reused the way day2_remove's and day2_replace's oracles reuse it - shows the identical shape both directions, destroy the higher index only (sg-d3f8475b4536bc66b gone, sg-a7134d986fe9ce70d unchanged) and create the higher index back under a new id (sg-ae09a9da7ca9c4bab). A new GroupId is a sound destroy witness on this pin: probed directly against ghcr.io/lex00/floci@sha256:c55d74e1 with no tofu in the loop, a delete-then-recreate under an identical group name minted sg-8c824b212636a50e9 -> sg-61cec558a8a5a286d. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, so these checks discriminate.", + "day2_count": "Scaled aws_security_group.count_test, a self-contained 2-instance count block in its own count_test.tf at an address nothing else in this crossing names. SYNTHETIC, and NOT because this estate has no scalable knob: sumaform declares a real one (var.quantity, \"number of hosts like this one\", driving count = var.quantity on aws_instance.instance), but every resource that knob scales is off the tag rung here - the instance and the EBS volume are markers = record by this crossing's own strict block, aws_volume_attachment is untaggable, and all three carry sumaform's own lifecycle { ignore_changes = [tags] } - so none of them can witness the stage's identity clause off the live object. G4 exercises the real knob for the half it can settle: quantity 1 -> 2 through choudoufu proposes exactly 3 creates, all at index [1] (instance, data disk, volume attachment), index [0] untouched, identical to stock's own plan for the same change on cold_deploy's state (G-ORACLE-QUANTITY), and the estate plans empty again once it is restored; the down direction is NOT exercised on that knob (cold_deploy's state is quantity=1, so there is no 2-instance stock state to scale down from). On the taggable block: scaling 2 -> 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy) and left count_test[0]'s live GroupId sg-f5399762065bb39ab AND its tofu-address marker aws_security_group.count_test:0 unchanged, both read through the AWS CLI rather than choudoufu's report; count_test[1] (sg-df09988387ba4224c) verified absent by describe-security-groups length 0 in between; scaling 1 -> 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) back under a NEW GroupId sg-27e39476a3e6a8724 carrying tofu-address=aws_security_group.count_test:1, with count_test[0] untouched throughout; the next plan is empty. Stock oracle (G-ORACLE): the identical 2-instance block stood up FOR REAL in its own idle floci account on FLOCI_PORT+3 - stock never had this block, so cold_deploy's state could not be reused the way day2_remove's and day2_replace's oracles reuse it - shows the identical shape both directions, destroy the higher index only (sg-8b2649c223f8be264 gone, sg-671372ad2ef1c9820 unchanged) and create the higher index back under a new id (sg-2a483e85eb4452a7d). A new GroupId is a sound destroy witness on this pin: probed directly against ghcr.io/lex00/floci@sha256:c55d74e1 with no tofu in the loop, a delete-then-recreate under an identical group name minted sg-8c824b212636a50e9 -> sg-61cec558a8a5a286d. BREAK_COUNT=1 asserts the WRONG instance (count_test[0]) was destroyed and reports fail, so these checks discriminate.", "day2_remove": "choudoufu: deleting module.server's block proposed exactly three destroys (0 add, 0 change, 3 destroy: the record-based instance and EBS volume, plus the untaggable/derived volume attachment), applied cleanly (0 added, 0 changed, 3 destroyed), the instance and volume are genuinely gone from the live account (instance State=terminated, volume absent, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes the same three destroys", "day2_rename": "moved block: aws_eip.crossing_nat renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_route_table.crossing_public renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing module.server's image input (ubuntu2204 -> ubuntu2404, both real, both in floci's seeded AMI catalog) proposed exactly one instance replace at the same declared address, cascading into the volume attachment (instance_id is ForceNew there too) - 2 to add, 0 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated via the AWS CLI and the new instance is confirmed running the new image; the local record store's record at the same address now names the new instance's id, not the terminated one (i-5720faebfefe13c10 -> i-c9ddad8b40018244e); the next plan proposes no resource action; BREAK=replace confirms this section's own record check discriminates (a deliberately-wrong expectation against the same real record fails, rather than vacuously passing). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. A manufactured live-object collision (the shape ec2-instance-complete's and corpus-sqs-basic's own BREAK=replace controls report) has no tag surface to be detected from on this markers=record instance and is not exercised here - verified directly that an untagged extra instance is simply invisible to this plan, correctly, not incorrectly.", + "day2_replace": "choudoufu: changing module.server's image input (ubuntu2204 -> ubuntu2404, both real, both in floci's seeded AMI catalog) proposed exactly one instance replace at the same declared address, cascading into the volume attachment (instance_id is ForceNew there too) - 2 to add, 0 to change, 2 to destroy, matching F-ORACLE's own plan shape; applied cleanly; the old instance is confirmed terminated via the AWS CLI and the new instance is confirmed running the new image; the local record store's record at the same address now names the new instance's id, not the terminated one (i-f3d88dad3155358d2 -> i-d8cfb88e7cd43142e); the next plan proposes no resource action; BREAK=replace confirms this section's own record check discriminates (a deliberately-wrong expectation against the same real record fails, rather than vacuously passing). Scope note: this exercises OpenTofu's default destroy-then-create ordering, not the create_before_destroy variant the stage's Title names - see this section's own header comment. A manufactured live-object collision (the shape ec2-instance-complete's and corpus-sqs-basic's own BREAK=replace controls report) has no tag surface to be detected from on this markers=record instance and is not exercised here - verified directly that an untagged extra instance is simply invisible to this plan, correctly, not incorrectly.", "drift_reconverge": "the crossing VPC's Name tag tampered out of band, plan proposed fixing exactly aws_vpc.crossing, apply changed 1 and reconverged the tag to sumaform-crossing-vpc; module.server's record-based identities unaffected", "greenfield": "11 resources from nothing (7 tag-stamped, 2 recorded via markers = record, 2 untaggable/derived - route_table_association and volume_attachment), replan empty, stock oracle in its own namespace matches on vpc cidr, security-group rule counts and the instance's ami+type", "migrate": "7 stamped, 2 recorded (markers = record honoured at migrate time, GitHub issue #365 slice 2), 0 failed, 2 skipped", + "plan_approval": "one argument edited (aws_internet_gateway.crossing's tags gain Reviewed=yes - one of this crossing's seven tag-stamped objects, chosen over module.server's markers = record instance and volume, which carry sumaform's own lifecycle { ignore_changes = [tags] } and so could not witness a tag edit at all), \"plan -out=approved.tfplan\" wrote a 48587-byte stock-format plan file whose whole change set is one update on aws_internet_gateway.crossing (igw-aa457a90); the world then moved out of band (vpc-1a753819's Name tag through the AWS CLI, never through choudoufu - STAGE 5's own proven mutation, on a DIFFERENT instance and a DIFFERENT live object from the one under review) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both aws_vpc.crossing and the live vpc-1a753819 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - describe-tags on igw-aa457a90 still returned no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the Name tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and igw-aa457a90 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The edit was then reverted, re-applied, the gateway confirmed to be the same id it started as, module.server's record-based instance identity confirmed untouched, and the estate replanned empty, so PART F starts where it would have. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 7 tagged objects before, 7 after, no state file either time; module.server's record-based instance and volume identities unchanged", "test_plan": "Items 4, 5 and 6 (this script's header) are all FIXED and the plan is genuinely empty (\"No changes. Your infrastructure matches the configuration.\"): live-import honours markers = record (located records for aws_instance.instance[0] and aws_ebs_volume.data_disk[0], confirmed at the store and by value against the AWS CLI both right after migrate and again after this empty replan), residue now covers NestingList/NestingSet/NestingMap blocks (internal/live/projection's residueEligibleBlock, widened from the block's SHAPE - whether carriesNoInformation can tell its absence from a real empty answer - never from a type name), and lex00/floci#103 (published in ghcr.io/lex00/floci@sha256:e16d9007a03093b6a6edd22273dee9d8253131f18581b0fa20ae6d34178a3079) now honours RunInstances' BlockDeviceMapping.Ebs.VolumeSize for the root device, closing the one line (root_block_device.volume_size = 8 -> 200) that was this crossing's own last wall. Plan moved 3 to add/0/0 (the original ABSENT gap) -> 2 to add/0/2 to destroy (item 4 fixed, item 5's replacement exposed) -> 0 to add/1 to change/0 to destroy (item 5 fixed) -> empty (item 6 fixed by the emulator)." }, - "duration_s": 589.4, + "duration_s": 626.9, "stage_seconds": { - "cold_deploy": 88, - "day2_count": 122, + "cold_deploy": 75, + "day2_count": 123, "day2_remove": 23, - "day2_rename": 41, - "day2_replace": 72, - "drift_reconverge": 22, - "greenfield": 142, - "migrate": 199, + "day2_rename": 40, + "day2_replace": 71, + "drift_reconverge": 21, + "greenfield": 145, + "migrate": 196, + "plan_approval": 54, "test_apply": 11, - "test_plan": 11 + "test_plan": 12 } }, "notes": "Landed d583dc93b7 (2026-08-18) - the FIRST OpenTofu-native estate crossed (uyuni-project's own maintainers describe it as \"OpenTofu configuration,\" not \"Terraform configuration\"), versus every prior estate tonight being Terraform-authored/OpenTofu-compatible via terraform-aws-modules. Deliberately reduced slice: the full main.tf.aws.example composes four AWS host roles from one leaf module, backend_modules/aws/host, but three of the four (bastion, module.mirror, module.minion) have no root-facing toggle to disable real SSH/Salt provisioning - the \"real boot behavior, out of scope for an emulator\" case. Only module.server exposes provision=false; module.base's own network submodule was also unusable (create_network=true needs CreateDhcpOptions/ReplaceRouteTableAssociation, neither implemented in floci), so this estate's own plain VPC/subnet/NAT resources stand in for it. A real floci gap found and fixed on the way: sumaform's ami.tf evaluates ~23 data \"aws_ami\" blocks unconditionally (one per supported guest OS) regardless of which single image an estate actually launches, and floci's catalog had zero SUSE/Marketplace/Rocky/RHEL entries - seeded 20, reconciled into the combined image alongside tonight's other three floci fixes. cold_deploy and migrate genuinely pass (11 resources, 9 of 11 stamped - 2 correctly untaggable). test_plan blocked by two real, structural rules baked into backend_modules/aws/host itself (the one leaf module every AWS host role shares, so this isn't an artifact of the reduced slice), and on reading both rules' own reasoning neither looks like a choudoufu defect - both are correct, deliberate refusals, not filed: (1) an unconditional, provisioner-less connection block that checkProvisioners flags on its own terms by documented design, dead code in sumaform's own module; (2) lifecycle { ignore_changes = [tags] } on the WHOLE tags argument of aws_instance.instance and aws_ebs_volume.data_disk - sumaform's own comment explains why (SUSE's internal AWS accounts add tags on apply that need preserving), but ignoring the whole argument also silently discards the update that would write tofu-address/tofu-estate, the exact marker-safety failure #306 was about tonight. The fix sumaform's own error text names - ignore_changes = [tags[\"Owner\"]], not the whole argument - is an edit to sumaform's module, out of scope here. Follow-up pass 2026-08-18 (#313 cross-check, fresh worktree, real run): confirmed neither of the two RULE-classified refusals above is #313's wall - zero occurrences of its diagnostic in the raw plan output. Both remain exactly as already documented: permanent, deliberate refusals (checkProvisioners on a dead-code connection block; ignore_changes on the whole tags argument), not filed as new issues, no action needed." @@ -1794,16 +1840,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1812,28 +1858,30 @@ "exit_code": 0, "detail": { "cold_deploy": "Apply complete! Resources: 62 added, 0 changed, 0 destroyed.; 0 objects carry tofu-estate=vpc-complete-crossing before migration", - "day2_count": "choudoufu: scaling aws_customer_gateway.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and cgw-ae204e601ae5e91da is confirmed gone from the live account while count_test[0] (cgw-f6fed178e29c19fac) kept its server-minted id, its tofu-address marker and its tofu-slot; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a genuinely NEW object (cgw-84118c223099f26a4, not the destroyed cgw-ae204e601ae5e91da), with count_test[0] untouched throughout; the next plan is empty. Every identity above was read off EC2 through the AWS CLI, by the tofu-address marker, never from choudoufu's own report. Stock oracle (G-ORACLE): the identical count block stood up by terraform in its own floci account and scaled 2->1->2 for real shows the identical shape - destroy count_test[1] only, recreate it under a new id, count_test[0]'s id unchanged both times. THE BLOCK IS SYNTHETIC, deliberately: every real count knob in this estate is a subnet list that drives an untaggable aws_route_table_association count block in lockstep (a question about orphaned derived children, not about count-slot binding), and the one real zero-cascade knob, module.vpc's customer_gateways, is a for_each map with no index slots and is already day2_replace's target - so the block uses aws_customer_gateway, a type this estate really does exercise three of. Building it found and fixed a real defect (five-row row 2): the scale-down proposed NO destroy at all, because the removal sweep read live/mapping.json's via=\"former2\" provenance - a Cloud Control ENUMERABILITY filter - to answer the identity question \"which TF type is this ARN\", and classified aws_customer_gateway TYPE_NOT_LISTABLE against live/registry.json's own handlers.list=true.", + "day2_count": "choudoufu: scaling aws_customer_gateway.count_test from 2 to 1 destroyed exactly count_test[1], the higher index (0 add, 0 change, 1 destroy), and cgw-3768750fdbdaf84d8 is confirmed gone from the live account while count_test[0] (cgw-fe83feba57bb37ac0) kept its server-minted id, its tofu-address marker and its tofu-slot; scaling back from 1 to 2 created exactly count_test[1] (1 add, 0 change, 0 destroy) as a genuinely NEW object (cgw-54a679d9fd7f89f8a, not the destroyed cgw-3768750fdbdaf84d8), with count_test[0] untouched throughout; the next plan is empty. Every identity above was read off EC2 through the AWS CLI, by the tofu-address marker, never from choudoufu's own report. Stock oracle (G-ORACLE): the identical count block stood up by terraform in its own floci account and scaled 2->1->2 for real shows the identical shape - destroy count_test[1] only, recreate it under a new id, count_test[0]'s id unchanged both times. THE BLOCK IS SYNTHETIC, deliberately: every real count knob in this estate is a subnet list that drives an untaggable aws_route_table_association count block in lockstep (a question about orphaned derived children, not about count-slot binding), and the one real zero-cascade knob, module.vpc's customer_gateways, is a for_each map with no index slots and is already day2_replace's target - so the block uses aws_customer_gateway, a type this estate really does exercise three of. Building it found and fixed a real defect (five-row row 2): the scale-down proposed NO destroy at all, because the removal sweep read live/mapping.json's via=\"former2\" provenance - a Cloud Control ENUMERABILITY filter - to answer the identity question \"which TF type is this ARN\", and classified aws_customer_gateway TYPE_NOT_LISTABLE against live/registry.json's own handlers.list=true.", "day2_remove": "choudoufu: deleting the dynamodb endpoint's map entry (module.vpc_endpoints_renamed.aws_vpc_endpoint.this[\"dynamodb\"]) proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the endpoint is genuinely gone from the live account (State=absent, read via the AWS CLI, not choudoufu's own report), and the next plan proposes no resource action; stock oracle on cold_deploy's own state (E-ORACLE) also proposes exactly one destroy for the same object", "day2_rename": "moved block: module.vpc_endpoints renamed with zero churn (0 add, 7 change, 0 destroy), marker rewritten in place across its taggable objects; live-mv: aws_security_group.rds renamed with zero churn, marker rewritten in place; stock oracle over the same two-object rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing customer_gateways[\"IP1\"]'s ForceNew ip_address argument proposed exactly one isolated replace at the same declared for_each key (1 to add, 1 to destroy, nothing else), matching F-ORACLE's own plan shape; applied cleanly; the old gateway (cgw-c1ef80761a5322b97) is confirmed gone/deleted and the new gateway (cgw-bf793a3d862648297) carries the marker, both via the AWS CLI; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", + "day2_replace": "choudoufu: changing customer_gateways[\"IP1\"]'s ForceNew ip_address argument proposed exactly one isolated replace at the same declared for_each key (1 to add, 1 to destroy, nothing else), matching F-ORACLE's own plan shape; applied cleanly; the old gateway (cgw-475f57e0a2f706305) is confirmed gone/deleted and the new gateway (cgw-b7f5d49f8756a85cd) carries the marker, both via the AWS CLI; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", "drift_reconverge": "one subnet tampered (Example tag), plan proposed fixing exactly one object, apply changed 1 and reconverged the tag to ex-complete", "greenfield": "62 resources from nothing (40 tag-stamped, 22 untaggable/derived), replan empty, stock oracle in its own namespace matches on vpc cidr, subnet count (18) and the s3 endpoint's presence", "migrate": "40 stamped, 22 skipped, 0 recorded, 0 failed; 39 objects carry tofu-estate=vpc-complete-crossing; the VPC's tofu-slot reads 0 off EC2, written by the migration itself (choudoufu #372)", + "plan_approval": "one argument edited (aws_security_group.rds's tags gain Reviewed=yes, a tags-only update that is not ForceNew and leaves sg-0373a87e083bfa5dd's id alone for PART D's later rename), \"plan -out=approved.tfplan\" wrote a 60349-byte stock-format plan file whose whole change set is one update on aws_security_group.rds; the world then moved out of band (subnet-22deaf7e's Example tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.vpc.aws_subnet.private[0] and the live subnet-22deaf7e it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - sg-0373a87e083bfa5dd still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-0373a87e083bfa5dd read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op: 39 objects before, 39 after, no state file either time", "test_plan": "empty plan; identity re-check unchanged: module.vpc.aws_vpc.this:0, aws_security_group.rds, module.vpc_endpoints.aws_vpc_endpoint.this:s3" }, - "duration_s": 289.3, + "duration_s": 298.1, "stage_seconds": { - "cold_deploy": 34, - "day2_count": 92, - "day2_remove": 21, + "cold_deploy": 24, + "day2_count": 93, + "day2_remove": 20, "day2_rename": 13, "day2_replace": 20, "drift_reconverge": 7, - "greenfield": 55, - "migrate": 94, + "greenfield": 56, + "migrate": 96, + "plan_approval": 18, "test_apply": 4, - "test_plan": 4 + "test_plan": 3 } }, "notes": "RE-CROSSED FOR REAL 2026-08-21 in worktree live/dhcp-options-355 off local main 41f8c8dd6a, real Docker/floci/terraform/AWS CLI throughout, floci ghcr.io/lex00/floci@sha256:cdd50ec0. STAGES UNCHANGED at 2 of 5, and the honest headline is that #355's wall IS gone and three more stand behind it, none of them a choudoufu defect. Stage 1: 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', 0 objects carrying tofu-estate beforehand. Stage 2: dry run '40 of 62 eligible, 22 untaggable, 0 unadmitted'; -approve '40 stamped, 22 skipped, 0 recorded, 0 failed'; three identities asserted by value through the AWS CLI, all three passed. STAGE 3 NO LONGER REFUSES IN DISCOVERY: with #355 fixed, live-plan exits 0 and renders a full plan for the first time in this estate's history. Proven by A/B against the SAME live migrated estate, same floci container, two binaries: main (41f8c8dd6a) exits 1 with 'Error: Listed resource with no tags - Cloud Control listed a aws_vpc_dhcp_options (AWS::EC2::DHCPOptions) with no Tags in its Properties ... Live resources: dopt-default'; the fixed binary exits 0 with that diagnostic absent and every other diagnostic identical. WHAT STANDS BEHIND IT, all measured in that run and none of them choudoufu's: (1) the first live-plan after live-import proposes adding tofu-slot to 31 objects. That is documented, deliberate product behavior (see live/e2e/corpus-iam-policy/run.sh's THE TOFU-SLOT FINDING; live-import cannot compute a slot from one state file), and three other crossing scripts fold a convergence apply into stage 2. This script does not, because it had never reached stage 3 to notice. (2) that convergence apply fails on a floci gap: 'UnsupportedOperation: Operation ModifyVpcEndpoint is not supported', 4 errors, one per interface VPC endpoint. 26 of the 31 tofu-slot writes do land. (3) the replan after that shows 'Plan: 3 to add, 5 to change, 3 to destroy' - three FORCED REPLACEMENTS from floci read fidelity, not drift: aws_nat_gateway.this[0] (floci's DescribeNatGateways returns neither allocation_id nor subnet_id, so both read as absent and force replacement), aws_vpn_gateway.this[0] (floci returns availability_zone='eu-west-1a' on a gateway whose config sets none, so the plan reads '- availability_zone -> null # forces replacement'), and aws_vpc_endpoint.this[s3], plus four endpoint in-place diffs (policy, route_table_ids, subnet_ids, cidr_blocks all read back empty). All floci work items, not choudoufu ones, and all four filed together as lex00/floci#97 (which also carries the Redshift Tagging-API gap below). THE #355 LOOSE END IS SETTLED, with evidence: the '39 objects carry tofu-estate' line against the '40 stamped' line is a floci Tagging-API coverage gap, not a choudoufu miscount. The 40th object is aws_redshift_subnet_group.redshift[0]; 'aws redshift describe-cluster-subnet-groups' shows it carrying tofu-address=module.vpc.aws_redshift_subnet_group.redshift:0 and tofu-estate=vpc-complete-crossing, while 'resourcegroupstaggingapi get-resources' for the same estate returns 39 ARNs with no Redshift among them. choudoufu's own stamp count is the correct one. ONE PRIOR FIGURE CORRECTED: the note below records 'nine non-fatal Incomplete sweep for undeclared resources warnings'. The real number is 989, identical on both binaries - it is the tag sweep's ARN-join-table coverage list (internal/live/discovery/tagging.go), produced before the type scans and untouched by this fix. Nine was a sample, not a count. test_apply and drift_reconverge stay not_run: stage 3 still ends in FAIL, so running them would prove nothing. PRIOR HISTORY BELOW. RE-CROSSED FOR REAL 2026-08-21 at dbc5ebf575 (GitHub issue #346's fix), worktree live/live-read-346, real Docker/floci/terraform/AWS CLI throughout. STAGES UNCHANGED at 2 of 5 - and that is the honest headline, because #346's own diagnostic IS gone and a different wall stands behind it. Stage 1: 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', 0 objects carrying tofu-estate beforehand. Stage 2: dry run '40 of 62 eligible, 22 untaggable, 0 unadmitted'; -approve '40 stamped, 22 skipped, 0 recorded, 0 failed'; three identities asserted by value straight through the AWS CLI and all three passed (vpc-68ae42e6 = module.vpc.aws_vpc.this:0, sg-1bac9cc2a99da4e8c = aws_security_group.rds, vpce-f81456900c35660a0 = module.vpc_endpoints.aws_vpc_endpoint.this:s3). Stage 3 no longer refuses in identity resolution at all: the #346 diagnostic (cidr_blocks = lookup(each.value, 'cidr_blocks', null) reaching module.vpc.vpc_cidr_block) appears nowhere in the run, and the run gets past live_plan.go's step 4 (identity, fatal on error) into step 5 (discovery), which it could not have done otherwise. It now fails on exactly ONE blocking diagnostic, a NEW one that was hidden behind #346 one stage earlier: 'Listed resource with no tags - Cloud Control listed a aws_vpc_dhcp_options (AWS::EC2::DHCPOptions) with no Tags in its Properties, and refining it with GetResource found none either, so its ownership markers cannot be read. Live resources: dopt-default.' That is the ACCOUNT'S DEFAULT DHCP options set, which this estate did not create and does not declare. Filed separately. Nine non-fatal 'Incomplete sweep for undeclared resources' warnings also print (aws_xray_* x5, kubernetes_* x4), none of them a type this estate declares. One unexplained figure worth someone's hour: the run's own closing line reads '39 objects carry tofu-estate after migration' against the '40 stamped' line above it - the script does not assert it, so nothing failed, but the two disagree. test_apply and drift_reconverge stay not_run: running them against a refused plan would prove nothing. PRIOR HISTORY BELOW. Re-verified 2026-08-18 against ghcr.io/lex00/floci@sha256:f5b46236c6b6fff376ae2db8a2b3a51bf1d13b19a92b6a37af4827ccdf1ef180, published via floci's own CI/GHCR-publish workflows (pushed to origin/main, not a local multi-arch build - see HANDOFF.md's Traps). lex00/floci#66 (Redshift), #67 (DHCP options) and #69 (customer/VPN gateway) are confirmed fixed: cold deploy no longer errors on any of the three. It now fails one step later, on a fourth, narrower and previously-undetected gap in #68's own CreateCacheSubnetGroup/ModifyCacheSubnetGroup implementation - it reads the SubnetIds member list under the generic SubnetIds.member.N key, but ElastiCache's service model overrides SubnetIdentifierList's member locationName to SubnetIdentifier, so every real client sends SubnetIds.SubnetIdentifier.N and the call 400s with MissingParameter. Filed as lex00/floci#70. A fix for exactly this was already sitting uncommitted in the shared floci checkout (another session's in-progress work, left untouched - not this orchestrator's to land). Stages 2-5 remain implemented in the script but unexercised since stage 1 still fails fast by design. Follow-up pass 2026-08-20, isolated worktree off local main (ea9fd62fc0), real Docker/floci/AWS CLI throughout, read from the script's own PASS/FAIL lines: cold_deploy and migrate now PASS - this estate moves 0 of 5 to 2 of 5. lex00/floci#70 was already fixed AND already pinned (99f4cbce8f moved live/floci-image to sha256:5873331d, 83c1aa73's published build) - the brief that sent this pass in believed the pin predated it and was wrong; re-running against that existing pin confirmed CreateCacheSubnetGroup succeeds and surfaced the NEXT gap one call later. That gap was lex00/floci#71 (ElastiCache served no tagging actions at all: ListTagsForResource/AddTagsToResource/RemoveTagsFromResource absent from ElastiCacheQueryHandler's switch, and CacheSubnetGroup carried no tags field), and the AWS provider calls ListTagsForResource on EVERY read of aws_elasticache_subnet_group, so the resource was unusable even with no tags in the configuration - the sole remaining cold-deploy error, 1 of 1. Fixed in floci and merged to lex00/floci main (dc140fb0; CI and GHCR-publish both green), mirroring RDS's identical trio on the identical Query protocol; tags live on the CacheSubnetGroup model beside its existing arn field so TaggedResourceScanner picks them up for resourcegroupstaggingapi GetResources with no extra wiring (asserted by a new test, not assumed), and resolveTagHandle reads the resource type off the ARN so another taggable ElastiCache resource is one branch plus a tags field on its model. live/floci-image re-pinned here to sha256:dc246b1e, with live/floci-capabilities.json regenerated for that digest (services with -watch networkmanager,storagegateway, cloudcontrol, cloudcontrol-scoped, tagging) plus the three hand-probed rows re-verified live against the new image - redshift implemented, qldb still unimplemented, opensearch still partial - giving 86 services / 749 types, matching the prior digest block's shape exactly with no existing digest block touched. Real numbers from the run: STAGE 1 'Apply complete! Resources: 62 added, 0 changed, 0 destroyed', 0 objects tagged before migration; STAGE 2 dry run '40 of 62 resource instance(s) are eligible for stamping' with UNTAGGABLE (22) and no UNADMITTED_TYPE section at all, -approve '40 resource(s) newly stamped, 0 already stamped, 0 newly recorded, 0 already recorded, 0 failed, 22 skipped', and three identities read straight through the AWS CLI: module.vpc.aws_vpc.this:0 on the VPC, aws_security_group.rds on the RDS SG, module.vpc_endpoints.aws_vpc_endpoint.this:s3 on the S3 endpoint; 39 objects carry tofu-estate after migration. All 22 skips are genuinely untaggable types (aws_route, aws_route_table_association, aws_vpc_dhcp_options_association, aws_security_group_rule - none has a tags argument in the provider's schema), i.e. the invariant working, not a gap. TWO SELF-INFLICTED SCRIPT BUGS FOUND AND FIXED, neither of which had ever run because stage 1 had never passed: the stage-2 count assertion demanded '0 skipped' (impossible for this estate; it now asserts 40/22 by value plus UNADMITTED_TYPE by absence), and all three identity assertions compared the tag value against OpenTofu's BRACKET spelling ('module.vpc.aws_vpc.this[0]') when a tag value can never carry '[' - the escaped form is 'module.vpc.aws_vpc.this:0' per internal/live/markers.EscapeKey. That is the same vacuous-comparison bug corpus-iam-policy and corpus-iam-read-only-policy each shipped once; both forms are now separate variables, and the bracket forms are used where stage 5 reads a plan diff header. STAGE 3 fails on EXACTLY ONE diagnostic, filed as #346, whose headline finding refutes the obvious fix: the diagnostic points at lookup(each.value, 'cidr_blocks', null) on the vpc-endpoints module's line 116, but a resolveLookupCall beside coalesce.go would NOT unblock this estate. Checked with three hand-built live-check variants rather than assumed: a static-valued lookup() already resolves, and the same map written as a direct each.value.cidr_blocks refuses identically. What refuses is the VALUE - the example passes [module.vpc.vpc_cidr_block], i.e. aws_vpc.this[0].cidr_block, a non-identity attribute of another managed resource, into aws_security_group_rule's identity-bearing cidr_blocks. Written inline the resolver reaches the reference and says so ('Not an identity attribute'); through a local or a for_each map it falls to the generic refusal at resolve.go:2025. Same wall, two spellings - so widening the decomposition switch changes the message and leaves the estate blocked. Whether identity resolution may fold a managed resource's own CONFIGURED attribute (this cidr_block is local.vpc_cidr, a static string) is a maintainer design call, so it was filed rather than forced. test_apply and drift_reconverge remain not_run: attempting them against a still-refused plan would prove nothing." @@ -1858,16 +1906,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -1883,19 +1931,21 @@ "drift_reconverge": "one object tampered (Name tag), exactly module.vpc.aws_vpc.this[\"main\"] proposed by both choudoufu and stock with the identical change, apply changed 1 and the Name tag reads back as configured", "greenfield": "28 resources from nothing (matching stage 1's stock cold-deploy count exactly), all markers verified via the AWS CLI, 28 records in the local record store (#364 A2), replan empty, object-by-object comparison against stock's still-pristine cold deploy on $ENDPOINT matches on tagged-object count (21), VPC CIDR, subnet/NAT-gateway/VPC-endpoint counts and account alias", "migrate": "live-import -approve completed cleanly against the cold state", + "plan_approval": "one argument edited (module.vpc.aws_default_security_group.this[\"main\"]'s tags gain Reviewed=yes - a single for_each instance nothing else in the module references), \"plan -out=approved.tfplan\" wrote a 38420-byte stock-format plan file whose whole change set is one update on module.vpc.aws_default_security_group.this[\"main\"]; the world then moved out of band (vpc-e1d1a6dd's Name tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both module.vpc.aws_vpc.this[\"main\"] and the live vpc-e1d1a6dd it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - sg-e5666543a0a301fd7 still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and sg-e5666543a0a301fd7 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "test_apply": "genuine no-op (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 21", "test_plan": "no resource change proposed" }, - "duration_s": 206.9, + "duration_s": 209.8, "stage_seconds": { - "cold_deploy": 42, + "cold_deploy": 32, "day2_count": 29, "day2_remove": 18, - "day2_rename": 10, + "day2_rename": 11, "day2_replace": 7, - "drift_reconverge": 8, - "greenfield": 38, - "migrate": 49, + "drift_reconverge": 7, + "greenfield": 39, + "migrate": 48, + "plan_approval": 13, "test_apply": 3, "test_plan": 3 } @@ -1912,7 +1962,7 @@ "stages": { "cold_deploy": "pass", "day2_count": "pass", - "day2_crash": "pass", + "day2_crash": "fail", "day2_remove": "pass", "day2_rename": "pass", "day2_replace": "pass", @@ -1920,49 +1970,51 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "pass", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "eec6fb428239cdfd2359a781a0e22d0319ad453e", - "date": "2026-09-06T01:27:32Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:06:04Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", "tofu": "1.12.5" }, - "exit_code": 0, + "exit_code": 1, "detail": { "cold_deploy": "5 resources from plain terraform, a real terraform.tfstate, zero markers", "day2_count": "choudoufu: scaling aws_security_group.count_test from 2 to 1 destroyed exactly count_test[1] (0 add, 0 change, 1 destroy), leaving count_test[0]'s live id and tofu-address marker unchanged; scaling back from 1 to 2 created exactly count_test[1] under a NEW live id (0 add, 0 change -> 1 add, 0 change, 0 destroy) while count_test[0] stayed untouched throughout; the next plan is empty; the B1.7 stock oracle on the same 2-instance count block, applied fresh in the idle greenfield account, shows the identical shape: destroy the higher index only, create the higher index back under a new id, the lower index's id unchanged both times", - "day2_crash": "choudoufu: a real create_before_destroy replace of aws_instance.main was interrupted with SIGTERM (landed on attempt 1 of 3; deterministic by construction, not by timing luck - internal/command/apply_e2etesting_crash.go self-signals synchronously inside the single -parallelism=1 graph-walker goroutine the instant the create half's own write-back commits in memory, replacing #483's external tail/grep/kill race that produced issue #490's own retry-lottery evidence) strictly between the create committing (new object i-31f6c7e770302d57f, confirmed running via the AWS CLI) and the destroy of the deposed old object (i-c705a31c96a9d1588, confirmed still running and untouched via the AWS CLI) ever dispatching; the local record's one write-back correctly carried both facts at once (current=i-31f6c7e770302d57f, deposed=i-c705a31c96a9d1588). Real investigation before writing this check found a genuine engine gap: issue #415's record-backed collision branch (internal/live/discovery/discovery.go, decl.recordBacked's 2-claimant path) called collisionProblem unconditionally with no deposed-record lookup at all, so a record-backed address's own crash window - exactly what a real crash's write-back leaves, since it answers the address's CURRENT identity in the same commit - could never recover on its own; fixed generically (mirrors the scalar path's own matchDeposedClaimant call, no resource type name in the fix), covered by two new unit tests (internal/live/discovery/deposed_test.go). The next plan proposed exactly one destroy (the deposed object, 0 add, 0 change, 1 destroy) and nothing else, matching stock's own documented deposed-object semantics (Stock records the old object as deposed and destroys it on the next apply); applying it destroyed exactly that object (confirmed terminated via the AWS CLI), cleared the deposed record entry, and left the current identity untouched; the plan after that is empty. BREAK_CRASH=1 confirms the empty-plan assertion this stage's Break text names correctly fails to hold against the same real crash window.", + "day2_crash": "the post-recovery plan exited 1", "day2_remove": "choudoufu: deleting aws_internet_gateway.renamed's block proposed exactly one destroy (0 add, 0 change, 1 destroy), applied cleanly (0 added, 0 changed, 1 destroyed), the object is genuinely gone from the live account (describe-internet-gateways on the old id no longer returns it, read via the AWS CLI, not choudoufu's own report), and the next plan is empty; stock oracle on cold_deploy's own state (B1.6) also proposes exactly one destroy for the same object; classifyOrphans did not withhold the destroy because no other aws_internet_gateway block is declared anywhere in this config", "day2_rename": "moved block: aws_security_group renamed with zero churn (0 add, 1 change, 0 destroy), marker rewritten in place; live-mv: aws_internet_gateway renamed with zero churn, marker rewritten in place; stock oracle over the same two-resource rename on cold_deploy's own state also shows zero churn (0 add, 0 change, 0 destroy); both live ids unchanged, read via the AWS CLI", - "day2_replace": "choudoufu: changing aws_instance.main's ForceNew ami argument proposed exactly one isolated instance replace at the same declared address (1 to add, 1 to destroy, nothing else), matching B1.8's own plan shape; applied cleanly; the old instance (i-b2dcd2cdc161c3dc3) is confirmed terminated and the new instance (i-c705a31c96a9d1588) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance, not the terminated one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", + "day2_replace": "choudoufu: changing aws_instance.main's ForceNew ami argument proposed exactly one isolated instance replace at the same declared address (1 to add, 1 to destroy, nothing else), matching B1.8's own plan shape; applied cleanly; the old instance (i-49b443856f409d5a7) is confirmed terminated and the new instance (i-65a7966e4fd5a1a42) carries the marker, both via the AWS CLI; the local record store's record at the same address now names the new instance, not the terminated one; the next plan proposes no resource action. No BREAK=replace leg - see this section's own header comment (reusing corpus-security-group-complete's own finding from this same unit rather than re-measuring it here).", "drift_reconverge": "one object tampered, exactly aws_instance.main proposed, apply changed 1 and the tag reads back as configured", "greenfield": "5-object structural comparison (vpc/subnet/igw/sg/instance) between the greenfield estate and stock's cold deploy matches, via the AWS CLI on both endpoints, marker tags never compared; local record store held 5 records, one per instance (#364 A2); replanned empty both with and without the local record store", "migrate": "5 of 5 verified, 5 stamped, 0 skipped", + "plan_approval": "one argument edited (aws_subnet.main's tags gain Reviewed=yes - the one leaf of this five-resource estate no later part renames, removes, replaces or crashes, and a tags-only update that is not ForceNew, so the subnet id aws_instance.main points at is untouched), \"plan -out=approved.tfplan\" wrote a 9808-byte stock-format plan file whose whole change set is one update on aws_subnet.main; the world then moved out of band (i-49b443856f409d5a7's Name tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both aws_instance.main and the live i-49b443856f409d5a7 it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - subnet-303d6d3d still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and subnet-303d6d3d read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "strict": "every strict toggle on (secrets = refuse, no_source_create = refuse, marker_repair = never with a markers \"record\" selection naming aws_ebs_volume) against a scratch estate carrying one resource, random_password.db: exactly one refusal, matching live/LIMITATIONS.md's \"strict-secrets\" text word for word (Logical resource is not admitted / SECRET_REFUSED / strict { secrets = \"refuse\" }); no_source_create and marker_repair are on and silent, reaching nothing this config declares. BREAK_STRICT=1 turns secrets back to \"store\" alone: the refusal disappears, the plan becomes an ordinary create, and no other refusal appears. Not part of the headline bars: tools/gauntlet/stages.go keeps Status planned here, because isClear (tools/gauntlet/artifact.go) and NextUnits (tools/gauntlet/next.go) both key strictly off ActiveStages today, with no exemption for a stage the docs already call non-headline - flipping Status without first adding that exemption would silently start gating the two headline bars on this stage, which #363 did not ask for and this unit did not build.", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); tofu-estate-tagged object count unchanged at 5", "test_plan": "post-adoption plan is empty; markers read back through the AWS CLI in part A" }, - "duration_s": 260.4, + "duration_s": 247.5, "stage_seconds": { - "cold_deploy": 101, + "cold_deploy": 87, "day2_count": 17, - "day2_crash": 34, - "day2_remove": 6, - "day2_rename": 9, - "day2_replace": 27, + "day2_crash": 31, + "day2_remove": 5, + "day2_rename": 8, + "day2_replace": 26, "drift_reconverge": 4, "greenfield": 3, - "migrate": 52, + "migrate": 51, + "plan_approval": 11, "strict": 2, - "test_apply": 3, + "test_apply": 2, "test_plan": 2 } }, @@ -1986,16 +2038,16 @@ "drift_reconverge": "pass", "greenfield": "pass", "migrate": "pass", - "plan_approval": "not_run", + "plan_approval": "pass", "strict": "not_run", "test_apply": "pass", "test_plan": "pass" }, - "clear": false, + "clear": true, "protocol": "gauntlet", "last_run": { - "commit": "d72960cdc31f4e7d90f61088920379fc29b3999c", - "date": "2026-09-06T05:35:09Z", + "commit": "70e2722fa46314f0ffb0a9243ea8fb822068caac", + "date": "2026-09-07T03:00:49Z", "emulator": "ghcr.io/lex00/floci@sha256:a39185cc3971d0188663d61043cb038dff1260d8a975b1aa72c4e2bb1feac3cb", "oracle": { "terraform": "1.15.8", @@ -2011,22 +2063,24 @@ "drift_reconverge": "one live object mutated out of band through the AWS CLI; choudoufu's next plan proposed fixing exactly aws_vpc.main and nothing else (0 add, 1 change, 0 destroy), matching stock's own plan for the identical mutation on cold_deploy's own state (B4, taken before any marker existed); the apply changed exactly 1 resource, the Name tag reads back as configured and the tofu-address marker is unchanged", "greenfield": "choudoufu applied 79 resources into an account a stock destroy had left enumerated empty (A2), and its cloud matches stock's cold deploy across 79 structural facts compared object by object with marker tags never read on either side - the oracle this stage names. Also, beyond the oracle: the six representative identities are correct by value via the AWS CLI across Route 53/IAM/ECS/EC2; the apply persisted 79 records, matching stock's own instance list type for type with no gap - #671 closed the last one (aws_ecs_task_definition), which used to get no record and now does; the next plan is empty; and with the local record store deleted outright every one of the 79 objects is still found - nothing created, destroyed or replaced, 41 of them untaggable and composing from a stamped parent - with the only movement being 1 residue-held aws_ecs_service update(s), which is what deleting the residue store (issue #275) means rather than a divergence", "migrate": "live-import ratified 38 of 79 instances as eligible and stamped all 38 with 0 failed and 41 skipped (untaggable, identity composed from an already-stamped parent); every one of the 79 addresses in stock's own `terraform state list` - this stage's oracle - is accounted for by name in the report", + "plan_approval": "one argument edited through the generator's own render_config path (the reviewp case: aws_subnet.main's tags gain Reviewed=yes - the one shared, singular, taggable resource no later part renames, removes, scales or replaces, and a tags-only update that is not ForceNew, so the subnet id ecs.tf reads is untouched), \"plan -out=approved.tfplan\" wrote a 31735-byte stock-format plan file whose whole change set is one update on aws_subnet.main (Plan: 0 to add, 1 to change, 0 to destroy); the world then moved out of band (vpc-f48eb49d's Name tag, through the AWS CLI, never through choudoufu) and \"apply approved.tfplan\" refused with \"The approved plan no longer matches the live system\" at exit 3, classifying the drift under \"This apply would do, and the approved plan does not include:\" and naming both aws_vpc.main and the live vpc-f48eb49d it was computed against, with \"Exit status 3\" spelled out for a pipeline; nothing was applied - subnet-8a9e6960 still carried no Reviewed tag, read back through the AWS CLI rather than from the absence of an \"Apply complete!\" line. Inverted control on the same run (the shape live/smoke/scenarios/apply-what-was-approved.sh reasons out): with the tag put back and nothing else changed, the IDENTICAL file applied - 0 added, 1 changed, 0 destroyed - and subnet-8a9e6960 read back with Reviewed=yes, so the refusal is earned by the drift and not handed out to every plan file. The estate re-renders to the pristine generator output, replans empty and the VPC's tofu-address marker still reads aws_vpc.main. BREAK_APPROVAL=1 asserts stage 12's own recorded Break line (apply the planfile after a mutation and expect success) and correctly fails", "strict": "this crossing script does not exercise the strict toggles: strict is Headline:false in tools/gauntlet/stages.go so it moves neither bar, and a toggle-by-toggle refusal fixture is a separate unit from the crossing this script exists to be. live/e2e/reference-ec2-vpc/run.sh's PART G is the pattern for the estate that does carry one", "test_apply": "no-op apply (0 added, 0 changed, 0 destroyed); the estate is enumerated object by object before and after - 34 objects across IAM/Route53/ECS/EC2, byte-identical listings, never a bare count - and the tofu-estate-tagged count is unchanged at 38", "test_plan": "post-migration plan is empty; six rendered identities asserted BY VALUE against the AWS CLI across four separate tagging surfaces (Route 53, IAM, ECS, EC2), including the count-indexed aws_iam_role.count_team[1] and the module-nested, double-indexed module.team_pod[\"pod-a\"].aws_iam_role.pod_role[0]" }, - "duration_s": 427.8, + "duration_s": 342.1, "stage_seconds": { - "cold_deploy": 167, - "day2_count": 28, - "day2_remove": 11, - "day2_rename": 22, - "day2_replace": 19, - "drift_reconverge": 39, - "greenfield": 85, - "migrate": 47, + "cold_deploy": 124, + "day2_count": 18, + "day2_remove": 7, + "day2_rename": 18, + "day2_replace": 12, + "drift_reconverge": 33, + "greenfield": 65, + "migrate": 43, + "plan_approval": 13, "strict": 0, - "test_apply": 6, + "test_apply": 5, "test_plan": 4 } }