From 33971a39fe7197984718f496e30fbe2957e3646e Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 20:49:19 -0300 Subject: [PATCH 01/13] feat(cloudformation): provision stack set instances as real stacks Stack instances were stubs: CreateStackInstances answered a random OperationId and deployed nothing, DescribeStackInstance always said CURRENT, and the operation APIs reported SUCCEEDED for any id. - Typed stack set store with instances and operations; legacy extras records migrate on load - Create/Update/DeleteStackInstances drive CreateStack/UpdateStack/ DeleteStack in each target account and region, with per-instance parameter overrides, failure tolerance, region order, account gate Lambda, and StopStackSetOperation - UpdateStackSet redeploys all or the targeted instances; the rest go OUTDATED - Operations recorded per target for DescribeStackSetOperation, ListStackSetOperations and ListStackSetOperationResults - SERVICE_MANAGED: OU resolution (nested, filter types, AccountsUrl), trusted access check, CallAs=DELEGATED_ADMIN - ImportStacksToStackSet, CreateStackSet from StackId, ListStackSetAutoDeploymentTargets - DetectStackSetDrift checks instance resources against backing services; ListStackInstanceResourceDrifts reports them - Modeled errors: StackSetNotFound, StackInstanceNotFound, OperationNotFound, OperationInProgress, OperationIdAlreadyExists, StackSetNotEmpty, NameAlreadyExists, StaleRequest, InvalidOperation Closes #2515 --- crates/fakecloud-cloudformation/src/extras.rs | 623 +-- crates/fakecloud-cloudformation/src/lib.rs | 2 + .../fakecloud-cloudformation/src/service.rs | 25 +- .../src/stack_sets.rs | 4728 +++++++++++++++++ crates/fakecloud-cloudformation/src/state.rs | 6 + .../tests/cloudformation.rs | 222 +- .../tests/cloudformation_stack_sets.rs | 325 ++ crates/fakecloud-server/src/main.rs | 4 +- .../content/docs/services/cloudformation.md | 22 +- 9 files changed, 5344 insertions(+), 613 deletions(-) create mode 100644 crates/fakecloud-cloudformation/src/stack_sets.rs create mode 100644 crates/fakecloud-e2e/tests/cloudformation_stack_sets.rs diff --git a/crates/fakecloud-cloudformation/src/extras.rs b/crates/fakecloud-cloudformation/src/extras.rs index b6992534e..53348dcc3 100644 --- a/crates/fakecloud-cloudformation/src/extras.rs +++ b/crates/fakecloud-cloudformation/src/extras.rs @@ -36,7 +36,7 @@ fn rand_id() -> String { format!("{nanos:x}-{seq:x}") } -fn xml_response(action: &str, inner: String, request_id: &str) -> AwsResponse { +pub(crate) fn xml_response(action: &str, inner: String, request_id: &str) -> AwsResponse { let body = format!( r#"<{action}Response xmlns="{NS}"> <{action}Result> @@ -54,7 +54,7 @@ fn xml_response(action: &str, inner: String, request_id: &str) -> AwsResponse { AwsResponse::xml(StatusCode::OK, body) } -fn xml_response_no_result(action: &str, request_id: &str) -> AwsResponse { +pub(crate) fn xml_response_no_result(action: &str, request_id: &str) -> AwsResponse { let body = format!( r#"<{action}Response xmlns="{NS}"> @@ -454,30 +454,15 @@ impl CloudFormationService { } } if let Some(name) = params.get("StackSetName") { - let accounts = self.state.read(); - // Existence and body are separate questions: a stack set whose - // record carries no `TemplateBody` still EXISTS, and conflating - // the two reported it as missing. AWS accepts the stack set's id + // Existence and body are separate questions: a stack set with an + // empty template still EXISTS. AWS accepts the stack set's id // (`{name}:{suffix}`) as well as its name. - let entry = accounts.get(account_id).and_then(|st| { - st.extras.get("stack_sets").and_then(|sets| { - sets.get(name).or_else(|| { - sets.iter() - .find(|(_, set)| { - set.get("StackSetId").and_then(|v| v.as_str()) - == Some(name.as_str()) - }) - .map(|(_, set)| set) - }) - }) - }); - let found = entry.map(|set| { - set.get("TemplateBody") - .and_then(|b| b.as_str()) - .unwrap_or_default() - .to_string() - }); - drop(accounts); + let found = self + .state + .read() + .get(account_id) + .and_then(|st| crate::stack_sets::find_active(st, name)) + .map(|set| set.template_body.clone()); return match found { Some(body) => Ok(body), // Unlike `ValidationError`, this one IS declared on @@ -503,7 +488,7 @@ impl CloudFormationService { /// error for an unusable URL, and the conformance probe sends /// `TemplateURL="t"` -- which `looks_like_url` rejects before any fetch is /// attempted -- so this stays lenient by design. - fn stack_set_template_body( + pub(crate) fn stack_set_template_body( &self, account_id: &str, params: &BTreeMap, @@ -567,6 +552,77 @@ impl CloudFormationService { String::from_utf8(bytes.to_vec()).ok() } + /// Whether a stack resource's physical resource still exists in its + /// backing service. `None` for resource types drift detection does not + /// check. + pub(crate) fn resource_exists( + &self, + account_id: &str, + resource: &StackResource, + ) -> Option { + let aid = account_id; + let exists = match resource.resource_type.as_str() { + "AWS::SQS::Queue" => self + .deps + .sqs + .read() + .get(aid) + .map(|s| s.queues.contains_key(&resource.physical_id)) + .unwrap_or(false), + "AWS::SNS::Topic" => self + .deps + .sns + .read() + .get(aid) + .map(|s| s.topics.contains_key(&resource.physical_id)) + .unwrap_or(false), + "AWS::S3::Bucket" => self + .deps + .s3 + .read() + .get(aid) + .map(|s| s.buckets.contains_key(&resource.physical_id)) + .unwrap_or(false), + "AWS::Lambda::Function" => self + .deps + .lambda + .read() + .get(aid) + .map(|s| s.functions.contains_key(&resource.physical_id)) + .unwrap_or(false), + "AWS::IAM::Role" => self + .deps + .iam + .read() + .get(aid) + .map(|s| s.roles.contains_key(&resource.physical_id)) + .unwrap_or(false), + "AWS::DynamoDB::Table" => self + .deps + .dynamodb + .read() + .get(aid) + .map(|s| s.tables.values().any(|t| t.arn == resource.physical_id)) + .unwrap_or(false), + "AWS::KMS::Key" => self + .deps + .kms + .read() + .get(aid) + .map(|s| s.keys.contains_key(&resource.physical_id)) + .unwrap_or(false), + "AWS::SecretsManager::Secret" => self + .deps + .secretsmanager + .read() + .get(aid) + .map(|s| s.secrets.contains_key(&resource.physical_id)) + .unwrap_or(false), + _ => return None, + }; + Some(exists) + } + pub(crate) fn handle_extra_action( &self, req: &AwsRequest, @@ -1518,269 +1574,6 @@ impl CloudFormationService { Ok(xml_response("ListChangeSets", inner, &rid)) } - // ── Stack sets ── - "CreateStackSet" => { - let name = params - .get("StackSetName") - .ok_or_else(|| missing("StackSetName"))? - .clone(); - let id = format!("{name}:{}", rand_id()); - // Resolve a TemplateURL too. Storing only the inline body left - // every URL-created stack set with an empty template, so - // `get-template-summary --stack-set-name` reported a - // confidently empty summary -- the false green light this - // work exists to remove. A fetch that fails still stores the - // empty body rather than erroring: CreateStackSet declares no - // error for it, and the probe's `"t"` is not a URL shape. - let template_body = self - .stack_set_template_body(&aid, ¶ms) - .map_err(|reason| { - AwsServiceError::aws_error( - StatusCode::BAD_REQUEST, - "ValidationError", - reason, - ) - })? - .unwrap_or_default(); - let entry = json!({ - "StackSetId": id, - "StackSetName": name, - "Status": "ACTIVE", - "TemplateBody": template_body, - }); - let mut accounts = self.state.write(); - let state = accounts.get_or_create(&aid); - store(&mut state.extras, "stack_sets").insert(name.clone(), entry); - Ok(xml_response( - "CreateStackSet", - format!(" {}", xml_escape(&id)), - &rid, - )) - } - "DescribeStackSet" => { - let name = params - .get("StackSetName") - .ok_or_else(|| missing("StackSetName"))? - .clone(); - let accounts = self.state.read(); - let entry = accounts - .get(&aid) - .and_then(|s| s.extras.get("stack_sets")) - .and_then(|m| m.get(&name)) - .cloned() - .unwrap_or_else(|| json!({"StackSetName": name.clone(), "Status": "ACTIVE"})); - let inner = format!( - " \n {}\n {}\n {}\n ", - xml_escape(entry["StackSetName"].as_str().unwrap_or(&name)), - xml_escape(entry["StackSetId"].as_str().unwrap_or("")), - xml_escape(entry["Status"].as_str().unwrap_or("ACTIVE")), - ); - Ok(xml_response("DescribeStackSet", inner, &rid)) - } - "ListStackSets" => { - let accounts = self.state.read(); - let items: Vec = accounts - .get(&aid) - .and_then(|s| s.extras.get("stack_sets")) - .map(|m| m.values().cloned().collect()) - .unwrap_or_default(); - let inner = format!( - " \n{}\n ", - members_xml(&items, |v| { - format!( - " {}\n {}\n {}", - xml_escape(v["StackSetName"].as_str().unwrap_or("")), - xml_escape(v["StackSetId"].as_str().unwrap_or("")), - xml_escape(v["Status"].as_str().unwrap_or("ACTIVE")), - ) - }), - ); - Ok(xml_response("ListStackSets", inner, &rid)) - } - "UpdateStackSet" => { - require_scalar(¶ms, "StackSetName")?; - // Persist the new template. Answering with an OperationId and - // storing nothing meant a stack set updated to a new template - // kept summarizing the ORIGINAL one forever -- stale is worse - // than empty, because it looks right. - // - // `UsePreviousTemplate` (and supplying neither body nor URL) - // keeps what is stored, which is what the parameter means. - let updated = self - .stack_set_template_body(&aid, ¶ms) - .map_err(|reason| { - AwsServiceError::aws_error( - StatusCode::BAD_REQUEST, - "ValidationError", - reason, - ) - })?; - let wanted = params.get("StackSetName").cloned().unwrap_or_default(); - { - let mut accounts = self.state.write(); - let state = accounts.get_or_create(&aid); - let sets = store(&mut state.extras, "stack_sets"); - // Records are keyed by NAME, but a stack set is equally - // addressable by its id -- and an update by id that - // matched nothing succeeded while leaving the old template - // in place, which reads as a working update. - let key = sets - .iter() - .find(|(name, entry)| { - **name == wanted - || entry["StackSetId"].as_str() == Some(wanted.as_str()) - }) - .map(|(name, _)| name.clone()); - // Updating a stack set that does not exist is not a - // success. Answering with an OperationId for a name that - // matches nothing is the same false green light as a - // summary of a template we never stored. - // `StackSetNotFoundException` IS declared on - // UpdateStackSet, so reporting it stays conformant. - let Some(key) = key else { - return Err(AwsServiceError::aws_error( - StatusCode::BAD_REQUEST, - "StackSetNotFoundException", - format!("StackSet {wanted} not found"), - )); - }; - if let (Some(updated), Some(entry)) = (updated, sets.get_mut(&key)) { - entry["TemplateBody"] = json!(updated); - } - } - let op_id = rand_id(); - Ok(xml_response( - "UpdateStackSet", - format!(" {}", xml_escape(&op_id)), - &rid, - )) - } - "DeleteStackSet" => { - let name = params - .get("StackSetName") - .ok_or_else(|| missing("StackSetName"))? - .clone(); - let mut accounts = self.state.write(); - let state = accounts.get_or_create(&aid); - if let Some(m) = state.extras.get_mut("stack_sets") { - m.remove(&name); - } - Ok(xml_response("DeleteStackSet", String::new(), &rid)) - } - "DescribeStackSetOperation" => { - require_scalar(¶ms, "StackSetName")?; - require_scalar(¶ms, "OperationId")?; - let op_id = params.get("OperationId").cloned().unwrap_or_else(rand_id); - let inner = format!( - " \n {}\n SUCCEEDED\n ", - xml_escape(&op_id), - ); - Ok(xml_response("DescribeStackSetOperation", inner, &rid)) - } - "ListStackSetOperations" => { - require_scalar(¶ms, "StackSetName")?; - Ok(xml_response( - "ListStackSetOperations", - " ".to_string(), - &rid, - )) - } - "ListStackSetOperationResults" => { - require_scalar(¶ms, "StackSetName")?; - require_scalar(¶ms, "OperationId")?; - Ok(xml_response( - "ListStackSetOperationResults", - " ".to_string(), - &rid, - )) - } - "ListStackSetAutoDeploymentTargets" => { - require_scalar(¶ms, "StackSetName")?; - Ok(xml_response( - "ListStackSetAutoDeploymentTargets", - " ".to_string(), - &rid, - )) - } - "StopStackSetOperation" => { - require_scalar(¶ms, "StackSetName")?; - require_scalar(¶ms, "OperationId")?; - Ok(xml_response("StopStackSetOperation", String::new(), &rid)) - } - "ImportStacksToStackSet" => { - require_scalar(¶ms, "StackSetName")?; - let op_id = rand_id(); - Ok(xml_response( - "ImportStacksToStackSet", - format!(" {}", xml_escape(&op_id)), - &rid, - )) - } - - // ── Stack instances ── - // The `Regions` list is `@required` in Smithy, but the Smithy - // `errors` list on these ops doesn't include `ValidationError`, - // so a missing-collection rejection would surface as an - // undeclared error to conformance. Accept an empty list and - // return a synthetic OperationId — real callers always supply - // regions and still get a valid response. - "CreateStackInstances" => { - require_scalar(¶ms, "StackSetName")?; - let op_id = rand_id(); - Ok(xml_response( - "CreateStackInstances", - format!(" {}", xml_escape(&op_id)), - &rid, - )) - } - "UpdateStackInstances" => { - require_scalar(¶ms, "StackSetName")?; - let op_id = rand_id(); - Ok(xml_response( - "UpdateStackInstances", - format!(" {}", xml_escape(&op_id)), - &rid, - )) - } - "DeleteStackInstances" => { - require_scalar(¶ms, "StackSetName")?; - require_scalar(¶ms, "RetainStacks")?; - let op_id = rand_id(); - Ok(xml_response( - "DeleteStackInstances", - format!(" {}", xml_escape(&op_id)), - &rid, - )) - } - "DescribeStackInstance" => { - require_scalar(¶ms, "StackSetName")?; - require_scalar(¶ms, "StackInstanceAccount")?; - require_scalar(¶ms, "StackInstanceRegion")?; - let inner = - " \n CURRENT\n " - .to_string(); - Ok(xml_response("DescribeStackInstance", inner, &rid)) - } - "ListStackInstances" => { - require_scalar(¶ms, "StackSetName")?; - Ok(xml_response( - "ListStackInstances", - " ".to_string(), - &rid, - )) - } - "ListStackInstanceResourceDrifts" => { - require_scalar(¶ms, "StackSetName")?; - require_scalar(¶ms, "StackInstanceAccount")?; - require_scalar(¶ms, "StackInstanceRegion")?; - require_scalar(¶ms, "OperationId")?; - Ok(xml_response( - "ListStackInstanceResourceDrifts", - " ".to_string(), - &rid, - )) - } - // ── Stack refactors ── "CreateStackRefactor" => { require_collection(¶ms, "StackDefinitions")?; @@ -2185,65 +1978,7 @@ impl CloudFormationService { let mut drifted_resources: Vec = Vec::new(); for resource in &resources { - let exists = match resource.resource_type.as_str() { - "AWS::SQS::Queue" => self - .deps - .sqs - .read() - .get(&aid) - .map(|s| s.queues.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::SNS::Topic" => self - .deps - .sns - .read() - .get(&aid) - .map(|s| s.topics.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::S3::Bucket" => self - .deps - .s3 - .read() - .get(&aid) - .map(|s| s.buckets.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::Lambda::Function" => self - .deps - .lambda - .read() - .get(&aid) - .map(|s| s.functions.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::IAM::Role" => self - .deps - .iam - .read() - .get(&aid) - .map(|s| s.roles.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::DynamoDB::Table" => self - .deps - .dynamodb - .read() - .get(&aid) - .map(|s| s.tables.values().any(|t| t.arn == resource.physical_id)) - .unwrap_or(false), - "AWS::KMS::Key" => self - .deps - .kms - .read() - .get(&aid) - .map(|s| s.keys.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::SecretsManager::Secret" => self - .deps - .secretsmanager - .read() - .get(&aid) - .map(|s| s.secrets.contains_key(&resource.physical_id)) - .unwrap_or(false), - _ => true, // NOT_CHECKED — assume exists - }; + let exists = self.resource_exists(&aid, resource).unwrap_or(true); if !exists { drifted_resources.push(json!({ "LogicalResourceId": resource.logical_id, @@ -2304,65 +2039,7 @@ impl CloudFormationService { }) .and_then(|stack| stack.resources.iter().find(|r| r.logical_id == logical)) .map(|resource| { - let exists = match resource.resource_type.as_str() { - "AWS::SQS::Queue" => self - .deps - .sqs - .read() - .get(&aid) - .map(|s| s.queues.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::SNS::Topic" => self - .deps - .sns - .read() - .get(&aid) - .map(|s| s.topics.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::S3::Bucket" => self - .deps - .s3 - .read() - .get(&aid) - .map(|s| s.buckets.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::Lambda::Function" => self - .deps - .lambda - .read() - .get(&aid) - .map(|s| s.functions.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::IAM::Role" => self - .deps - .iam - .read() - .get(&aid) - .map(|s| s.roles.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::DynamoDB::Table" => self - .deps - .dynamodb - .read() - .get(&aid) - .map(|s| s.tables.values().any(|t| t.arn == resource.physical_id)) - .unwrap_or(false), - "AWS::KMS::Key" => self - .deps - .kms - .read() - .get(&aid) - .map(|s| s.keys.contains_key(&resource.physical_id)) - .unwrap_or(false), - "AWS::SecretsManager::Secret" => self - .deps - .secretsmanager - .read() - .get(&aid) - .map(|s| s.secrets.contains_key(&resource.physical_id)) - .unwrap_or(false), - _ => true, - }; + let exists = self.resource_exists(&aid, resource).unwrap_or(true); if exists { "IN_SYNC" } else { @@ -2378,15 +2055,6 @@ impl CloudFormationService { ); Ok(xml_response("DetectStackResourceDrift", inner, &rid)) } - "DetectStackSetDrift" => { - require_scalar(¶ms, "StackSetName")?; - let op_id = rand_id(); - Ok(xml_response( - "DetectStackSetDrift", - format!(" {}", xml_escape(&op_id)), - &rid, - )) - } "DescribeStackDriftDetectionStatus" => { let id = params .get("StackDriftDetectionId") @@ -2940,13 +2608,13 @@ impl CloudFormationService { } #[cfg(test)] -mod tests { +pub(crate) mod tests { use super::{looks_like_url, parse_s3_url}; use crate::service::{CloudFormationDeps, CloudFormationService}; use crate::state::{CloudFormationState, SharedCloudFormationState}; use fakecloud_core::delivery::DeliveryBus; use fakecloud_core::multi_account::MultiAccountState; - use fakecloud_core::service::AwsRequest; + use fakecloud_core::service::{AwsRequest, AwsResponse, AwsServiceError}; use http::Method; use parking_lot::RwLock; use std::collections::HashMap; @@ -3222,7 +2890,7 @@ mod tests { } } - fn deps() -> CloudFormationDeps { + pub(crate) fn deps() -> CloudFormationDeps { use fakecloud_dynamodb::DynamoDbState; use fakecloud_ecr::EcrState; use fakecloud_eventbridge::EventBridgeState; @@ -3336,7 +3004,7 @@ mod tests { } } - fn svc() -> CloudFormationService { + pub(crate) fn svc() -> CloudFormationService { let state: SharedCloudFormationState = Arc::new(RwLock::new(MultiAccountState::::new( "000000000000", @@ -3346,9 +3014,23 @@ mod tests { CloudFormationService::new(state, deps()) } + /// Dispatch a request the way `handle` does for the actions these tests + /// exercise: stack set actions are async, everything else here is not. + fn call(svc: &CloudFormationService, req: &AwsRequest) -> Result { + if crate::stack_sets::is_stack_set_action(&req.action) { + tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("runtime") + .block_on(svc.handle_stack_set_action(req)) + } else { + svc.handle_extra_action(req) + } + } + /// Body of the XML response, for asserting on the rendered summary. fn introspect(svc: &CloudFormationService, action: &str, params: &[(&str, &str)]) -> String { - let resp = match svc.handle_extra_action(&req(action, params)) { + let resp = match call(svc, &req(action, params)) { Ok(resp) => resp, Err(e) => panic!("{action} should succeed: {}", e.message()), }; @@ -3545,10 +3227,13 @@ mod tests { let svc = svc(); // Answering with an OperationId for a name that matches nothing is // the same false green light as summarizing a template never stored. - let Err(err) = svc.handle_extra_action(&req( - "UpdateStackSet", - &[("StackSetName", "nope"), ("TemplateBody", GOOD_TEMPLATE)], - )) else { + let Err(err) = call( + &svc, + &req( + "UpdateStackSet", + &[("StackSetName", "nope"), ("TemplateBody", GOOD_TEMPLATE)], + ), + ) else { panic!("an unknown stack set must be reported"); }; assert_eq!(err.code(), "StackSetNotFoundException"); @@ -3565,13 +3250,16 @@ mod tests { // Supplying a URL that cannot be fetched is a failure. Treating it as // "no template given" reported success and kept serving the old one. - let Err(err) = svc.handle_extra_action(&req( - "UpdateStackSet", - &[ - ("StackSetName", "s"), - ("TemplateURL", "https://s3.amazonaws.com/b/missing.yaml"), - ], - )) else { + let Err(err) = call( + &svc, + &req( + "UpdateStackSet", + &[ + ("StackSetName", "s"), + ("TemplateURL", "https://s3.amazonaws.com/b/missing.yaml"), + ], + ), + ) else { panic!("an unfetchable URL must be reported"); }; assert!( @@ -3651,6 +3339,8 @@ mod tests { fn stack_sets_resolve_by_name_or_id_and_report_absence() { let svc = svc(); { + // Seeded in the record shape older builds persisted, so the + // migration into the typed store is exercised too. let mut accounts = svc.state.write(); let st = accounts.get_or_create("000000000000"); let sets = st.extras.entry("stack_sets".to_string()).or_default(); @@ -3663,6 +3353,7 @@ mod tests { "bodyless".to_string(), json!({"StackSetId": "bodyless:def"}), ); + crate::stack_sets::migrate_legacy_stack_sets(&mut accounts); } for key in ["myset", "myset:abc123"] { @@ -3688,7 +3379,7 @@ mod tests { use std::collections::BTreeMap; - fn req(action: &str, params: &[(&str, &str)]) -> AwsRequest { + pub(crate) fn req(action: &str, params: &[(&str, &str)]) -> AwsRequest { let mut q = HashMap::new(); q.insert("Action".to_string(), action.to_string()); for (k, v) in params { @@ -3723,7 +3414,7 @@ mod tests { /// service per call, so anything that must EXIST by the time it is used /// belongs here instead. fn ok_on(svc: &CloudFormationService, action: &str, params: &[(&str, &str)]) { - match svc.handle_extra_action(&req(action, params)) { + match call(svc, &req(action, params)) { Ok(resp) => assert!(resp.status.is_success(), "{action} status: {}", resp.status), Err(e) => panic!("{action} failed: {e:?}"), } @@ -4058,68 +3749,7 @@ mod tests { } #[test] - fn stack_sets_instances_refactors() { - // One service for the whole sequence: UpdateStackSet and - // DeleteStackSet act on the stack set CreateStackSet made, and a - // fresh service per call would not have it. - let svc = svc(); - ok_on(&svc, "CreateStackSet", &[("StackSetName", "ss")]); - ok_on(&svc, "DescribeStackSet", &[("StackSetName", "ss")]); - ok_on(&svc, "ListStackSets", &[]); - ok_on(&svc, "UpdateStackSet", &[("StackSetName", "ss")]); - ok( - "DescribeStackSetOperation", - &[("StackSetName", "ss"), ("OperationId", "op")], - ); - ok("ListStackSetOperations", &[("StackSetName", "ss")]); - ok( - "ListStackSetOperationResults", - &[("StackSetName", "ss"), ("OperationId", "op")], - ); - ok( - "ListStackSetAutoDeploymentTargets", - &[("StackSetName", "ss")], - ); - ok( - "StopStackSetOperation", - &[("StackSetName", "ss"), ("OperationId", "op")], - ); - ok("ImportStacksToStackSet", &[("StackSetName", "ss")]); - ok_on(&svc, "DeleteStackSet", &[("StackSetName", "ss")]); - ok( - "CreateStackInstances", - &[("StackSetName", "ss"), ("Regions.member.1", "us-east-1")], - ); - ok( - "UpdateStackInstances", - &[("StackSetName", "ss"), ("Regions.member.1", "us-east-1")], - ); - ok( - "DeleteStackInstances", - &[ - ("StackSetName", "ss"), - ("Regions.member.1", "us-east-1"), - ("RetainStacks", "false"), - ], - ); - ok( - "DescribeStackInstance", - &[ - ("StackSetName", "ss"), - ("StackInstanceAccount", "000000000000"), - ("StackInstanceRegion", "us-east-1"), - ], - ); - ok("ListStackInstances", &[("StackSetName", "ss")]); - ok( - "ListStackInstanceResourceDrifts", - &[ - ("StackSetName", "ss"), - ("StackInstanceAccount", "000000000000"), - ("StackInstanceRegion", "us-east-1"), - ("OperationId", "op"), - ], - ); + fn stack_refactors() { ok( "CreateStackRefactor", &[("StackDefinitions.member.1.StackName", "s")], @@ -4192,7 +3822,6 @@ mod tests { "DetectStackResourceDrift", &[("StackName", "s"), ("LogicalResourceId", "L")], ); - ok("DetectStackSetDrift", &[("StackSetName", "ss")]); ok( "DescribeStackDriftDetectionStatus", &[("StackDriftDetectionId", "id")], diff --git a/crates/fakecloud-cloudformation/src/lib.rs b/crates/fakecloud-cloudformation/src/lib.rs index 0bc6db38e..3c12058c9 100644 --- a/crates/fakecloud-cloudformation/src/lib.rs +++ b/crates/fakecloud-cloudformation/src/lib.rs @@ -2,12 +2,14 @@ pub mod extras; pub(crate) mod input_constraints; pub mod resource_provisioner; pub(crate) mod service; +pub(crate) mod stack_sets; pub(crate) mod state; pub mod template; pub(crate) mod template_summary; pub mod xml_responses; pub use service::{CloudControlOutcome, CloudFormationDeps, CloudFormationService}; +pub use stack_sets::migrate_legacy_stack_sets; pub use state::{ CloudFormationSnapshot, SharedCloudFormationState, CLOUDFORMATION_SNAPSHOT_SCHEMA_VERSION, }; diff --git a/crates/fakecloud-cloudformation/src/service.rs b/crates/fakecloud-cloudformation/src/service.rs index 8942b8b02..5ec6bcedf 100644 --- a/crates/fakecloud-cloudformation/src/service.rs +++ b/crates/fakecloud-cloudformation/src/service.rs @@ -1658,7 +1658,10 @@ impl CloudFormationService { Ok(()) } - async fn create_stack(&self, req: &AwsRequest) -> Result { + pub(crate) async fn create_stack( + &self, + req: &AwsRequest, + ) -> Result { let params = Self::get_all_params(req); // `negative_omit_StackName` expects any 4xx; the AnyError expectation @@ -2147,7 +2150,10 @@ impl CloudFormationService { ); } - async fn delete_stack(&self, req: &AwsRequest) -> Result { + pub(crate) async fn delete_stack( + &self, + req: &AwsRequest, + ) -> Result { let stack_name = Self::get_param(req, "StackName").ok_or_else(|| { AwsServiceError::aws_error( StatusCode::BAD_REQUEST, @@ -2441,7 +2447,10 @@ impl CloudFormationService { )) } - async fn update_stack(&self, req: &AwsRequest) -> Result { + pub(crate) async fn update_stack( + &self, + req: &AwsRequest, + ) -> Result { let mut input = UpdateStackInput::from_params(req)?; // Get stack_id before write lock for the provisioner @@ -3154,7 +3163,14 @@ impl AwsService for CloudFormationService { | "DeleteChangeSet" | "ExecuteChangeSet" | "CreateStackSet" + | "UpdateStackSet" | "DeleteStackSet" + | "CreateStackInstances" + | "UpdateStackInstances" + | "DeleteStackInstances" + | "StopStackSetOperation" + | "ImportStacksToStackSet" + | "DetectStackSetDrift" | "CreateStackRefactor" | "CreateGeneratedTemplate" | "DeleteGeneratedTemplate" @@ -3172,6 +3188,9 @@ impl AwsService for CloudFormationService { "DescribeStackResources" => self.describe_stack_resources(&req), "UpdateStack" => self.update_stack(&req).await, "GetTemplate" => self.get_template(&req), + a if crate::stack_sets::is_stack_set_action(a) => { + self.handle_stack_set_action(&req).await + } _ => self.handle_extra_action(&req), }; if mutates && matches!(result.as_ref(), Ok(resp) if resp.status.is_success()) { diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs new file mode 100644 index 000000000..93bf3516f --- /dev/null +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -0,0 +1,4728 @@ +//! CloudFormation StackSets. +//! +//! A stack set is a template plus parameters that is deployed as one stack per +//! (account, region) pair, a *stack instance*. Every call that changes what is +//! deployed runs as a stack set *operation*. The instances an operation touches +//! are provisioned by driving the ordinary CreateStack / UpdateStack / +//! DeleteStack paths in the target account and region, so an instance's stack +//! is a real stack whose resources exist in the backing services. Each +//! target's outcome is recorded on the operation, which is what +//! DescribeStackSetOperation and ListStackSetOperationResults report. +//! +//! Operations run to completion inside the call that starts them, except for +//! stacks that provision asynchronously (templates with custom resources). +//! Those leave the instance `RUNNING`, and every later read of the stack set +//! folds the stack's current status back into the instance and the operation. + +use std::collections::{BTreeMap, BTreeSet}; + +use chrono::{DateTime, Utc}; +use http::StatusCode; +use serde::{Deserialize, Serialize}; +use serde_json::Value; + +use fakecloud_aws::xml::xml_escape; +use fakecloud_core::multi_account::MultiAccountState; +use fakecloud_core::service::{AwsRequest, AwsResponse, AwsServiceError}; + +use crate::extras::{looks_like_url, xml_response, xml_response_no_result}; +use crate::service::CloudFormationService; +use crate::state::CloudFormationState; + +/// Service principal Organizations uses for StackSets trusted access and +/// delegated administration. +const STACKSETS_PRINCIPAL: &str = "member.org.stacksets.cloudformation.amazonaws.com"; +/// Name of the optional per-account Lambda that gates deployments. +const ACCOUNT_GATE_FUNCTION: &str = "AWSCloudFormationStackSetAccountGate"; +const DEFAULT_ADMIN_ROLE: &str = "AWSCloudFormationStackSetAdministrationRole"; +const DEFAULT_EXECUTION_ROLE: &str = "AWSCloudFormationStackSetExecutionRole"; +/// ImportStacksToStackSet accepts at most this many stacks per call. +const MAX_IMPORT_STACKS: usize = 10; +const DEFAULT_PAGE_SIZE: usize = 100; + +const TOLERANCE_EXCEEDED: &str = "Cancelled since failure tolerance has exceeded"; +const OPERATION_STOPPED: &str = "Cancelled since the operation was stopped"; + +// ── State ── + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct StackSet { + pub stack_set_id: String, + pub name: String, + pub arn: String, + /// `ACTIVE`, or `DELETED` once DeleteStackSet has run. Deleted stack sets + /// stay listable and describable by id, as in AWS. + pub status: String, + #[serde(default)] + pub description: Option, + #[serde(default)] + pub template_body: String, + #[serde(default)] + pub parameters: BTreeMap, + #[serde(default)] + pub capabilities: Vec, + #[serde(default)] + pub tags: Vec<(String, String)>, + #[serde(default)] + pub administration_role_arn: Option, + #[serde(default)] + pub execution_role_name: Option, + pub permission_model: String, + #[serde(default)] + pub auto_deployment: Option, + #[serde(default)] + pub managed_execution_active: bool, + #[serde(default)] + pub instances: Vec, + #[serde(default)] + pub operations: Vec, + /// Result of the most recent DetectStackSetDrift, if any. + #[serde(default)] + pub drift: Option, + pub created_at: DateTime, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct AutoDeployment { + pub enabled: bool, + pub retain_stacks_on_account_removal: bool, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct StackInstance { + pub account: String, + pub region: String, + #[serde(default)] + pub stack_id: Option, + /// `CURRENT`, `OUTDATED` or `INOPERABLE`. + pub status: String, + /// `StackInstanceStatus.DetailedStatus`. + pub detailed_status: String, + #[serde(default)] + pub status_reason: Option, + #[serde(default)] + pub parameter_overrides: BTreeMap, + #[serde(default)] + pub organizational_unit_id: Option, + pub drift_status: String, + #[serde(default)] + pub last_drift_check_timestamp: Option>, + #[serde(default)] + pub last_operation_id: Option, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct StackSetOperation { + pub operation_id: String, + /// `CREATE`, `UPDATE`, `DELETE` or `DETECT_DRIFT`. + pub action: String, + /// `RUNNING`, `SUCCEEDED`, `FAILED`, `STOPPING` or `STOPPED`. + pub status: String, + #[serde(default)] + pub status_reason: Option, + #[serde(default)] + pub retain_stacks: Option, + #[serde(default)] + pub preferences: OperationPreferences, + #[serde(default)] + pub deployment_targets: Option, + #[serde(default)] + pub administration_role_arn: Option, + #[serde(default)] + pub execution_role_name: Option, + pub created_at: DateTime, + #[serde(default)] + pub ended_at: Option>, + #[serde(default)] + pub results: Vec, + #[serde(default)] + pub drift: Option, + #[serde(default)] + pub resource_drifts: Vec, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct OperationResult { + pub account: String, + pub region: String, + /// `PENDING`, `RUNNING`, `SUCCEEDED`, `FAILED` or `CANCELLED`. + pub status: String, + #[serde(default)] + pub status_reason: Option, + #[serde(default)] + pub organizational_unit_id: Option, + #[serde(default)] + pub account_gate_status: Option, + #[serde(default)] + pub account_gate_reason: Option, +} + +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct OperationPreferences { + #[serde(default)] + pub region_concurrency_type: Option, + #[serde(default)] + pub region_order: Vec, + #[serde(default)] + pub failure_tolerance_count: Option, + #[serde(default)] + pub failure_tolerance_percentage: Option, + #[serde(default)] + pub max_concurrent_count: Option, + #[serde(default)] + pub max_concurrent_percentage: Option, + #[serde(default)] + pub concurrency_mode: Option, +} + +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct DeploymentTargets { + #[serde(default)] + pub accounts: Vec, + #[serde(default)] + pub accounts_url: Option, + #[serde(default)] + pub organizational_unit_ids: Vec, + #[serde(default)] + pub account_filter_type: Option, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct DriftDetectionDetails { + pub drift_status: String, + pub detection_status: String, + pub last_drift_check_timestamp: DateTime, + pub total: usize, + pub drifted: usize, + pub in_sync: usize, + pub failed: usize, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct InstanceResourceDrift { + pub account: String, + pub region: String, + pub stack_id: String, + pub logical_id: String, + pub physical_id: String, + pub resource_type: String, + /// `IN_SYNC`, `DELETED` or `NOT_CHECKED`. + pub status: String, + pub timestamp: DateTime, +} + +/// Move stack sets persisted by older builds, which kept a +/// `{StackSetId, StackSetName, Status, TemplateBody}` JSON record in the +/// generic `extras` store, into the typed store. +pub fn migrate_legacy_stack_sets(accounts: &mut MultiAccountState) { + for (account_id, state) in accounts.iter_mut() { + let Some(legacy) = state.extras.remove("stack_sets") else { + continue; + }; + for (name, record) in legacy { + let id = record["StackSetId"] + .as_str() + .map(str::to_string) + .unwrap_or_else(|| format!("{name}:{}", uuid::Uuid::new_v4())); + if state.stack_sets.contains_key(&id) { + continue; + } + let arn = format!( + "arn:aws:cloudformation:{}:{account_id}:stackset/{id}", + state.region + ); + state.stack_sets.insert( + id.clone(), + StackSet { + stack_set_id: id, + name: record["StackSetName"].as_str().unwrap_or(&name).to_string(), + arn, + status: record["Status"].as_str().unwrap_or("ACTIVE").to_string(), + description: None, + template_body: record["TemplateBody"] + .as_str() + .unwrap_or_default() + .to_string(), + parameters: BTreeMap::new(), + capabilities: Vec::new(), + tags: Vec::new(), + administration_role_arn: None, + execution_role_name: None, + permission_model: "SELF_MANAGED".to_string(), + auto_deployment: None, + managed_execution_active: false, + instances: Vec::new(), + operations: Vec::new(), + drift: None, + created_at: Utc::now(), + }, + ); + } + } +} + +pub(crate) fn is_stack_set_action(action: &str) -> bool { + matches!( + action, + "CreateStackSet" + | "DescribeStackSet" + | "ListStackSets" + | "UpdateStackSet" + | "DeleteStackSet" + | "CreateStackInstances" + | "UpdateStackInstances" + | "DeleteStackInstances" + | "DescribeStackInstance" + | "ListStackInstances" + | "DescribeStackSetOperation" + | "ListStackSetOperations" + | "ListStackSetOperationResults" + | "StopStackSetOperation" + | "ImportStacksToStackSet" + | "ListStackSetAutoDeploymentTargets" + | "DetectStackSetDrift" + | "ListStackInstanceResourceDrifts" + ) +} + +/// Find an ACTIVE stack set by name or id. +pub(crate) fn find_active<'a>( + state: &'a CloudFormationState, + name_or_id: &str, +) -> Option<&'a StackSet> { + state + .stack_sets + .values() + .find(|s| s.status == "ACTIVE" && (s.name == name_or_id || s.stack_set_id == name_or_id)) +} + +/// Find a stack set for a read: an active one by name or id, or a deleted one +/// by its (unique) id. A deleted stack set's name is free for reuse, so a name +/// never resolves to one. +fn find_for_read<'a>(state: &'a CloudFormationState, name_or_id: &str) -> Option<&'a StackSet> { + find_active(state, name_or_id).or_else(|| state.stack_sets.get(name_or_id)) +} + +fn active_key(state: &CloudFormationState, name_or_id: &str) -> Option { + find_active(state, name_or_id).map(|s| s.stack_set_id.clone()) +} + +// ── Errors ── + +fn aws_err(status: StatusCode, code: &str, message: impl Into) -> AwsServiceError { + AwsServiceError::aws_error(status, code, message) +} + +fn validation(message: impl Into) -> AwsServiceError { + aws_err(StatusCode::BAD_REQUEST, "ValidationError", message) +} + +fn stack_set_not_found(name: &str) -> AwsServiceError { + aws_err( + StatusCode::NOT_FOUND, + "StackSetNotFoundException", + format!("StackSet {name} not found"), + ) +} + +fn operation_not_found(op_id: &str) -> AwsServiceError { + aws_err( + StatusCode::NOT_FOUND, + "OperationNotFoundException", + format!("Operation {op_id} not found"), + ) +} + +fn instance_not_found(set: &str, account: &str, region: &str) -> AwsServiceError { + aws_err( + StatusCode::NOT_FOUND, + "StackInstanceNotFoundException", + format!("Stack instance with account {account} and region {region} not found for stack set {set}"), + ) +} + +fn required(params: &BTreeMap, field: &str) -> Result { + params + .get(field) + .cloned() + .ok_or_else(|| validation(format!("{field} is required"))) +} + +// ── Request parsing ── + +/// `Prefix.member.N` scalar list. +fn member_list(params: &BTreeMap, prefix: &str) -> Vec { + (1..) + .map_while(|i| params.get(&format!("{prefix}.member.{i}")).cloned()) + .collect() +} + +/// Whether the request carries the list `prefix` at all, including the empty +/// form (`Prefix=`) the CLI sends for an explicitly empty list. +fn list_present(params: &BTreeMap, prefix: &str) -> bool { + let dotted = format!("{prefix}."); + params.contains_key(prefix) || params.keys().any(|k| k.starts_with(&dotted)) +} + +struct ParameterEntry { + key: String, + value: Option, + use_previous: bool, +} + +fn parameter_list(params: &BTreeMap, prefix: &str) -> Vec { + (1..) + .map_while(|i| { + let key = params.get(&format!("{prefix}.member.{i}.ParameterKey"))?; + Some(ParameterEntry { + key: key.clone(), + value: params + .get(&format!("{prefix}.member.{i}.ParameterValue")) + .cloned(), + use_previous: params + .get(&format!("{prefix}.member.{i}.UsePreviousValue")) + .is_some_and(|v| v.eq_ignore_ascii_case("true")), + }) + }) + .collect() +} + +/// Resolve a parameter list against `previous`: explicit values win, and +/// `UsePreviousValue` entries carry the previous value forward. +fn resolve_parameters( + entries: &[ParameterEntry], + previous: &BTreeMap, +) -> Result, AwsServiceError> { + let mut out = BTreeMap::new(); + for entry in entries { + match (&entry.value, entry.use_previous) { + (Some(_), true) => { + return Err(validation(format!( + "Invalid input for parameter key {}. Cannot specify usePreviousValue as true and a parameter value at the same time", + entry.key + ))) + } + (Some(value), false) => { + out.insert(entry.key.clone(), value.clone()); + } + (None, true) => match previous.get(&entry.key) { + Some(value) => { + out.insert(entry.key.clone(), value.clone()); + } + None => { + return Err(validation(format!( + "Parameter {} does not have a previous value", + entry.key + ))) + } + }, + (None, false) => { + return Err(validation(format!( + "Invalid input for parameter key {}. Need to specify either usePreviousValue as true or a value for the parameter", + entry.key + ))) + } + } + } + Ok(out) +} + +fn tag_list(params: &BTreeMap) -> Vec<(String, String)> { + (1..) + .map_while(|i| { + let key = params.get(&format!("Tags.member.{i}.Key"))?; + let value = params.get(&format!("Tags.member.{i}.Value"))?; + Some((key.clone(), value.clone())) + }) + .collect() +} + +fn parse_bool( + params: &BTreeMap, + key: &str, +) -> Result, AwsServiceError> { + match params.get(key) { + None => Ok(None), + Some(v) if v.eq_ignore_ascii_case("true") => Ok(Some(true)), + Some(v) if v.eq_ignore_ascii_case("false") => Ok(Some(false)), + Some(v) => Err(validation(format!("Invalid value {v} for {key}"))), + } +} + +fn parse_u32(params: &BTreeMap, key: &str) -> Result, AwsServiceError> { + params + .get(key) + .map(|v| { + v.parse::() + .map_err(|_| validation(format!("Invalid value {v} for {key}"))) + }) + .transpose() +} + +fn parse_preferences( + params: &BTreeMap, +) -> Result { + let p = "OperationPreferences"; + let prefs = OperationPreferences { + region_concurrency_type: params.get(&format!("{p}.RegionConcurrencyType")).cloned(), + region_order: member_list(params, &format!("{p}.RegionOrder")), + failure_tolerance_count: parse_u32(params, &format!("{p}.FailureToleranceCount"))?, + failure_tolerance_percentage: parse_u32( + params, + &format!("{p}.FailureTolerancePercentage"), + )?, + max_concurrent_count: parse_u32(params, &format!("{p}.MaxConcurrentCount"))?, + max_concurrent_percentage: parse_u32(params, &format!("{p}.MaxConcurrentPercentage"))?, + concurrency_mode: params.get(&format!("{p}.ConcurrencyMode")).cloned(), + }; + if prefs.failure_tolerance_count.is_some() && prefs.failure_tolerance_percentage.is_some() { + return Err(validation( + "FailureToleranceCount and FailureTolerancePercentage cannot both be specified", + )); + } + if prefs.max_concurrent_count.is_some() && prefs.max_concurrent_percentage.is_some() { + return Err(validation( + "MaxConcurrentCount and MaxConcurrentPercentage cannot both be specified", + )); + } + if prefs.failure_tolerance_percentage.is_some_and(|v| v > 100) + || prefs.max_concurrent_percentage.is_some_and(|v| v > 100) + { + return Err(validation("Percentage values must be between 0 and 100")); + } + Ok(prefs) +} + +fn parse_deployment_targets(params: &BTreeMap) -> Option { + let p = "DeploymentTargets"; + if !list_present(params, p) { + return None; + } + Some(DeploymentTargets { + accounts: member_list(params, &format!("{p}.Accounts")), + accounts_url: params.get(&format!("{p}.AccountsUrl")).cloned(), + organizational_unit_ids: member_list(params, &format!("{p}.OrganizationalUnitIds")), + account_filter_type: params.get(&format!("{p}.AccountFilterType")).cloned(), + }) +} + +/// The deployment targets an operation records, in `DeploymentTargets` form +/// whether the request used top-level `Accounts` or `DeploymentTargets`. +fn targets_record( + accounts: &[String], + deployment_targets: Option<&DeploymentTargets>, +) -> DeploymentTargets { + match deployment_targets { + Some(dt) => dt.clone(), + None => DeploymentTargets { + accounts: accounts.to_vec(), + ..DeploymentTargets::default() + }, + } +} + +fn is_account_id(value: &str) -> bool { + value.len() == 12 && value.bytes().all(|b| b.is_ascii_digit()) +} + +/// Parse `arn:aws:cloudformation:{region}:{account}:stack/{name}/{id}` into +/// `(account, region)`. +fn stack_arn_location(arn: &str) -> Option<(String, String)> { + let parts: Vec<&str> = arn.splitn(6, ':').collect(); + if parts.len() != 6 || parts[0] != "arn" || parts[2] != "cloudformation" { + return None; + } + if !parts[5].starts_with("stack/") || !is_account_id(parts[4]) || parts[3].is_empty() { + return None; + } + Some((parts[4].to_string(), parts[3].to_string())) +} + +fn paginate( + items: Vec, + params: &BTreeMap, +) -> Result<(Vec, Option), AwsServiceError> { + let start = match params.get("NextToken") { + Some(token) => token + .parse::() + .map_err(|_| validation("Invalid NextToken"))?, + None => 0, + }; + let size = params + .get("MaxResults") + .and_then(|v| v.parse::().ok()) + .filter(|v| *v > 0) + .unwrap_or(DEFAULT_PAGE_SIZE); + let total = items.len(); + let page: Vec = items.into_iter().skip(start).take(size).collect(); + let next = (start + size < total).then(|| (start + size).to_string()); + Ok((page, next)) +} + +// ── XML ── + +fn ts(t: &DateTime) -> String { + t.format("%Y-%m-%dT%H:%M:%S%.3fZ").to_string() +} + +fn el(name: &str, value: &str) -> String { + format!("<{name}>{}", xml_escape(value)) +} + +fn opt_el(name: &str, value: Option<&str>) -> String { + value.map(|v| el(name, v)).unwrap_or_default() +} + +fn list_el(name: &str, members: impl IntoIterator) -> String { + let inner: String = members + .into_iter() + .map(|m| format!("{m}")) + .collect(); + if inner.is_empty() { + format!("<{name}/>") + } else { + format!("<{name}>{inner}") + } +} + +fn scalar_list_el<'a>(name: &str, values: impl IntoIterator) -> String { + list_el(name, values.into_iter().map(|v| xml_escape(v))) +} + +fn parameters_el(name: &str, params: &BTreeMap) -> String { + list_el( + name, + params + .iter() + .map(|(k, v)| format!("{}{}", el("ParameterKey", k), el("ParameterValue", v))), + ) +} + +fn next_token_el(next: Option) -> String { + next.map(|t| el("NextToken", &t)).unwrap_or_default() +} + +fn preferences_el(p: &OperationPreferences) -> String { + let mut out = String::new(); + out.push_str(&opt_el( + "RegionConcurrencyType", + p.region_concurrency_type.as_deref(), + )); + if !p.region_order.is_empty() { + out.push_str(&scalar_list_el("RegionOrder", &p.region_order)); + } + for (name, value) in [ + ("FailureToleranceCount", p.failure_tolerance_count), + ("FailureTolerancePercentage", p.failure_tolerance_percentage), + ("MaxConcurrentCount", p.max_concurrent_count), + ("MaxConcurrentPercentage", p.max_concurrent_percentage), + ] { + if let Some(v) = value { + out.push_str(&el(name, &v.to_string())); + } + } + out.push_str(&opt_el("ConcurrencyMode", p.concurrency_mode.as_deref())); + format!("{out}") +} + +fn drift_details_el(d: Option<&DriftDetectionDetails>) -> String { + match d { + Some(d) => format!( + "{}{}{}{}{}{}{}{}", + el("DriftStatus", &d.drift_status), + el("DriftDetectionStatus", &d.detection_status), + el("LastDriftCheckTimestamp", &ts(&d.last_drift_check_timestamp)), + el("TotalStackInstancesCount", &d.total.to_string()), + el("DriftedStackInstancesCount", &d.drifted.to_string()), + el("InSyncStackInstancesCount", &d.in_sync.to_string()), + el("InProgressStackInstancesCount", "0"), + el("FailedStackInstancesCount", &d.failed.to_string()), + ), + None => "NOT_CHECKED00000".to_string(), + } +} + +fn auto_deployment_el(a: Option<&AutoDeployment>) -> String { + a.map(|a| { + format!( + "{}{}", + el("Enabled", &a.enabled.to_string()), + el( + "RetainStacksOnAccountRemoval", + &a.retain_stacks_on_account_removal.to_string() + ), + ) + }) + .unwrap_or_default() +} + +fn managed_execution_el(active: bool) -> String { + format!( + "{}", + el("Active", &active.to_string()) + ) +} + +fn deployment_targets_el(t: &DeploymentTargets) -> String { + let mut out = String::new(); + if !t.accounts.is_empty() { + out.push_str(&scalar_list_el("Accounts", &t.accounts)); + } + out.push_str(&opt_el("AccountsUrl", t.accounts_url.as_deref())); + if !t.organizational_unit_ids.is_empty() { + out.push_str(&scalar_list_el( + "OrganizationalUnitIds", + &t.organizational_unit_ids, + )); + } + out.push_str(&opt_el( + "AccountFilterType", + t.account_filter_type.as_deref(), + )); + format!("{out}") +} + +fn status_details_el(op: &StackSetOperation) -> String { + let failed = op.results.iter().filter(|r| r.status == "FAILED").count(); + format!( + "{}", + el("FailedStackInstancesCount", &failed.to_string()) + ) +} + +fn stack_set_regions(set: &StackSet) -> Vec { + let regions: BTreeSet<&String> = set.instances.iter().map(|i| &i.region).collect(); + regions.into_iter().cloned().collect() +} + +fn stack_set_ous(set: &StackSet) -> Vec { + let ous: BTreeSet<&String> = set + .instances + .iter() + .filter_map(|i| i.organizational_unit_id.as_ref()) + .collect(); + ous.into_iter().cloned().collect() +} + +fn stack_set_el(set: &StackSet) -> String { + let mut out = String::new(); + out.push_str(&el("StackSetName", &set.name)); + out.push_str(&el("StackSetId", &set.stack_set_id)); + out.push_str(&opt_el("Description", set.description.as_deref())); + out.push_str(&el("Status", &set.status)); + out.push_str(&el("TemplateBody", &set.template_body)); + out.push_str(¶meters_el("Parameters", &set.parameters)); + out.push_str(&scalar_list_el("Capabilities", &set.capabilities)); + out.push_str(&list_el( + "Tags", + set.tags + .iter() + .map(|(k, v)| format!("{}{}", el("Key", k), el("Value", v))), + )); + out.push_str(&el("StackSetARN", &set.arn)); + out.push_str(&opt_el( + "AdministrationRoleARN", + set.administration_role_arn.as_deref(), + )); + out.push_str(&opt_el( + "ExecutionRoleName", + set.execution_role_name.as_deref(), + )); + out.push_str(&drift_details_el(set.drift.as_ref())); + out.push_str(&auto_deployment_el(set.auto_deployment.as_ref())); + out.push_str(&el("PermissionModel", &set.permission_model)); + out.push_str(&scalar_list_el( + "OrganizationalUnitIds", + &stack_set_ous(set), + )); + out.push_str(&managed_execution_el(set.managed_execution_active)); + out.push_str(&scalar_list_el("Regions", &stack_set_regions(set))); + format!("{out}") +} + +fn stack_set_summary_el(set: &StackSet) -> String { + let mut out = String::new(); + out.push_str(&el("StackSetName", &set.name)); + out.push_str(&el("StackSetId", &set.stack_set_id)); + out.push_str(&opt_el("Description", set.description.as_deref())); + out.push_str(&el("Status", &set.status)); + out.push_str(&auto_deployment_el(set.auto_deployment.as_ref())); + out.push_str(&el("PermissionModel", &set.permission_model)); + out.push_str(&el( + "DriftStatus", + set.drift + .as_ref() + .map_or("NOT_CHECKED", |d| d.drift_status.as_str()), + )); + if let Some(d) = &set.drift { + out.push_str(&el( + "LastDriftCheckTimestamp", + &ts(&d.last_drift_check_timestamp), + )); + } + out.push_str(&managed_execution_el(set.managed_execution_active)); + out +} + +fn instance_fields(set: &StackSet, i: &StackInstance, with_overrides: bool) -> String { + let mut out = String::new(); + out.push_str(&el("StackSetId", &set.stack_set_id)); + out.push_str(&el("Region", &i.region)); + out.push_str(&el("Account", &i.account)); + out.push_str(&opt_el("StackId", i.stack_id.as_deref())); + if with_overrides { + out.push_str(¶meters_el("ParameterOverrides", &i.parameter_overrides)); + } + out.push_str(&el("Status", &i.status)); + out.push_str(&format!( + "{}", + el("DetailedStatus", &i.detailed_status) + )); + out.push_str(&opt_el("StatusReason", i.status_reason.as_deref())); + out.push_str(&opt_el( + "OrganizationalUnitId", + i.organizational_unit_id.as_deref(), + )); + out.push_str(&el("DriftStatus", &i.drift_status)); + if let Some(t) = &i.last_drift_check_timestamp { + out.push_str(&el("LastDriftCheckTimestamp", &ts(t))); + } + out.push_str(&opt_el("LastOperationId", i.last_operation_id.as_deref())); + out +} + +fn operation_el(set: &StackSet, op: &StackSetOperation) -> String { + let mut out = String::new(); + out.push_str(&el("OperationId", &op.operation_id)); + out.push_str(&el("StackSetId", &set.stack_set_id)); + out.push_str(&el("Action", &op.action)); + out.push_str(&el("Status", &op.status)); + out.push_str(&preferences_el(&op.preferences)); + if let Some(retain) = op.retain_stacks { + out.push_str(&el("RetainStacks", &retain.to_string())); + } + out.push_str(&opt_el( + "AdministrationRoleARN", + op.administration_role_arn.as_deref(), + )); + out.push_str(&opt_el( + "ExecutionRoleName", + op.execution_role_name.as_deref(), + )); + out.push_str(&el("CreationTimestamp", &ts(&op.created_at))); + if let Some(t) = &op.ended_at { + out.push_str(&el("EndTimestamp", &ts(t))); + } + if let Some(t) = &op.deployment_targets { + out.push_str(&deployment_targets_el(t)); + } + if op.drift.is_some() { + out.push_str(&drift_details_el(op.drift.as_ref())); + } + out.push_str(&opt_el("StatusReason", op.status_reason.as_deref())); + out.push_str(&status_details_el(op)); + format!("{out}") +} + +fn operation_summary_el(op: &StackSetOperation) -> String { + let mut out = String::new(); + out.push_str(&el("OperationId", &op.operation_id)); + out.push_str(&el("Action", &op.action)); + out.push_str(&el("Status", &op.status)); + out.push_str(&el("CreationTimestamp", &ts(&op.created_at))); + if let Some(t) = &op.ended_at { + out.push_str(&el("EndTimestamp", &ts(t))); + } + out.push_str(&opt_el("StatusReason", op.status_reason.as_deref())); + out.push_str(&status_details_el(op)); + out.push_str(&preferences_el(&op.preferences)); + out +} + +fn operation_result_el(r: &OperationResult) -> String { + let mut out = String::new(); + out.push_str(&el("Account", &r.account)); + out.push_str(&el("Region", &r.region)); + out.push_str(&el("Status", &r.status)); + out.push_str(&opt_el("StatusReason", r.status_reason.as_deref())); + if let Some(gate) = &r.account_gate_status { + out.push_str(&format!( + "{}{}", + el("Status", gate), + opt_el("StatusReason", r.account_gate_reason.as_deref()) + )); + } + out.push_str(&opt_el( + "OrganizationalUnitId", + r.organizational_unit_id.as_deref(), + )); + out +} + +// ── Operation engine ── + +/// One (account, region) an operation acts on. +#[derive(Debug, Clone)] +struct Target { + account: String, + region: String, + ou: Option, + suspended: bool, +} + +/// What an operation does to each target. +#[derive(Debug, Clone)] +enum TargetAction { + /// Create the instance (or re-deploy it, when it already exists) with + /// these overrides. + Create { + overrides: BTreeMap, + }, + /// Re-deploy the instance's stack. `None` keeps the instance's overrides. + Update { + overrides: Option>, + }, + Delete { + retain_stacks: bool, + }, +} + +#[derive(Debug, Clone)] +enum OverrideSpec { + Value(String, String), + UsePrevious(String), +} + +#[derive(Debug, Clone, PartialEq)] +enum Outcome { + Succeeded, + Running, + Failed(String), + Cancelled(String), + SkippedSuspended, +} + +struct GateResult { + status: &'static str, + reason: Option, +} + +/// The stack-set fields an operation deploys, captured when it starts. +#[derive(Clone)] +struct DeploySpec { + name: String, + template_body: String, + parameters: BTreeMap, + capabilities: Vec, + tags: Vec<(String, String)>, +} + +impl DeploySpec { + fn of(set: &StackSet) -> Self { + Self { + name: set.name.clone(), + template_body: set.template_body.clone(), + parameters: set.parameters.clone(), + capabilities: set.capabilities.clone(), + tags: set.tags.clone(), + } + } + + fn stack_params(&self, overrides: &BTreeMap) -> Vec<(String, String)> { + let mut merged = self.parameters.clone(); + merged.extend(overrides.iter().map(|(k, v)| (k.clone(), v.clone()))); + let mut out = vec![("TemplateBody".to_string(), self.template_body.clone())]; + for (i, (k, v)) in merged.iter().enumerate() { + out.push(( + format!("Parameters.member.{}.ParameterKey", i + 1), + k.clone(), + )); + out.push(( + format!("Parameters.member.{}.ParameterValue", i + 1), + v.clone(), + )); + } + for (i, cap) in self.capabilities.iter().enumerate() { + out.push((format!("Capabilities.member.{}", i + 1), cap.clone())); + } + for (i, (k, v)) in self.tags.iter().enumerate() { + out.push((format!("Tags.member.{}.Key", i + 1), k.clone())); + out.push((format!("Tags.member.{}.Value", i + 1), v.clone())); + } + out + } +} + +/// Map a stack's status after an instance operation to that target's outcome. +fn stack_outcome(action: &str, status: &str, reason: Option<&str>) -> Outcome { + if status.ends_with("_IN_PROGRESS") { + return Outcome::Running; + } + let ok = match action { + "DELETE" => status == "DELETE_COMPLETE", + _ => matches!( + status, + "CREATE_COMPLETE" | "UPDATE_COMPLETE" | "IMPORT_COMPLETE" + ), + }; + if ok { + Outcome::Succeeded + } else { + Outcome::Failed( + reason + .map(str::to_string) + .unwrap_or_else(|| format!("Stack is in {status} state")), + ) + } +} + +fn result_status(outcome: &Outcome) -> &'static str { + match outcome { + Outcome::Succeeded => "SUCCEEDED", + Outcome::Running => "RUNNING", + Outcome::Failed(_) => "FAILED", + Outcome::Cancelled(_) | Outcome::SkippedSuspended => "CANCELLED", + } +} + +fn outcome_reason(outcome: &Outcome) -> Option { + match outcome { + Outcome::Failed(r) | Outcome::Cancelled(r) => Some(r.clone()), + Outcome::SkippedSuspended => Some("Account is suspended".to_string()), + _ => None, + } +} + +/// Apply an outcome to the instance it concerns. +fn apply_to_instance(instance: &mut StackInstance, outcome: &Outcome) { + let (status, detailed) = match outcome { + Outcome::Succeeded => ("CURRENT", "SUCCEEDED"), + Outcome::Running => ("OUTDATED", "RUNNING"), + Outcome::Failed(_) => ("OUTDATED", "FAILED"), + Outcome::Cancelled(_) => ("OUTDATED", "CANCELLED"), + Outcome::SkippedSuspended => ("OUTDATED", "SKIPPED_SUSPENDED_ACCOUNT"), + }; + instance.status = status.to_string(); + instance.detailed_status = detailed.to_string(); + instance.status_reason = outcome_reason(outcome); +} + +/// Allowed failures per region before an operation stops. +fn region_tolerance(prefs: &OperationPreferences, accounts_in_region: usize) -> usize { + match ( + prefs.failure_tolerance_count, + prefs.failure_tolerance_percentage, + ) { + (Some(count), _) => count as usize, + (None, Some(pct)) => accounts_in_region * pct as usize / 100, + (None, None) => 0, + } +} + +/// Final status of an operation none of whose targets are still running. +fn settled_status(op: &StackSetOperation) -> &'static str { + if op.status == "STOPPING" || op.status == "STOPPED" { + return "STOPPED"; + } + let mut per_region: BTreeMap<&str, (usize, usize)> = BTreeMap::new(); + for r in &op.results { + let entry = per_region.entry(r.region.as_str()).or_default(); + entry.0 += 1; + if r.status == "FAILED" { + entry.1 += 1; + } + } + let exceeded = per_region + .values() + .any(|(total, failed)| *failed > region_tolerance(&op.preferences, *total)); + if exceeded { + "FAILED" + } else { + "SUCCEEDED" + } +} + +fn settle_operation(op: &mut StackSetOperation) { + if !matches!(op.status.as_str(), "RUNNING" | "STOPPING") { + return; + } + if op + .results + .iter() + .any(|r| matches!(r.status.as_str(), "RUNNING" | "PENDING")) + { + return; + } + op.status = settled_status(op).to_string(); + op.ended_at = Some(Utc::now()); +} + +/// Order targets the way the operation deploys them: `RegionOrder` first, +/// then the remaining regions in request order. +fn order_targets( + mut targets: Vec, + regions: &[String], + prefs: &OperationPreferences, +) -> Vec { + let mut order: Vec<&String> = prefs.region_order.iter().collect(); + for r in regions { + if !order.contains(&r) { + order.push(r); + } + } + targets.sort_by_key(|t| { + order + .iter() + .position(|r| **r == t.region) + .unwrap_or(usize::MAX) + }); + targets +} + +fn synthetic_request( + account: &str, + region: &str, + action: &str, + request_id: &str, + params: Vec<(String, String)>, +) -> AwsRequest { + let mut query: std::collections::HashMap = params.into_iter().collect(); + query.insert("Action".to_string(), action.to_string()); + AwsRequest { + service: "cloudformation".to_string(), + action: action.to_string(), + region: region.to_string(), + account_id: account.to_string(), + request_id: request_id.to_string(), + headers: http::HeaderMap::new(), + query_params: query, + body: bytes::Bytes::new(), + body_stream: parking_lot::Mutex::new(None), + path_segments: Vec::new(), + raw_path: "/".to_string(), + raw_query: String::new(), + method: http::Method::POST, + is_query_protocol: true, + access_key_id: None, + principal: None, + } +} + +/// The accounts in (or below) an OU, or the whole organization for the root. +fn accounts_under( + org: &fakecloud_organizations::OrganizationState, + ou: &str, +) -> Vec<(String, bool)> { + let mut out = Vec::new(); + for account in org.accounts.values() { + if account.id == org.management_account_id { + // Service-managed stack sets never deploy to the management account. + continue; + } + let mut parent = account.parent_id.clone(); + let mut depth = 0; + loop { + if parent == ou { + out.push((account.id.clone(), account.status != "ACTIVE")); + break; + } + match org.ous.get(&parent) { + Some(p) if depth < 16 => { + parent = p.parent_id.clone(); + depth += 1; + } + _ => break, + } + } + } + out +} + +impl CloudFormationService { + pub(crate) async fn handle_stack_set_action( + &self, + req: &AwsRequest, + ) -> Result { + let params = Self::get_all_params(req); + match req.action.as_str() { + "CreateStackSet" => self.create_stack_set(req, ¶ms), + "DescribeStackSet" => self.describe_stack_set(req, ¶ms), + "ListStackSets" => self.list_stack_sets(req, ¶ms), + "UpdateStackSet" => self.update_stack_set(req, ¶ms).await, + "DeleteStackSet" => self.delete_stack_set(req, ¶ms), + "CreateStackInstances" => self.create_stack_instances(req, ¶ms).await, + "UpdateStackInstances" => self.update_stack_instances(req, ¶ms).await, + "DeleteStackInstances" => self.delete_stack_instances(req, ¶ms).await, + "DescribeStackInstance" => self.describe_stack_instance(req, ¶ms), + "ListStackInstances" => self.list_stack_instances(req, ¶ms), + "DescribeStackSetOperation" => self.describe_stack_set_operation(req, ¶ms), + "ListStackSetOperations" => self.list_stack_set_operations(req, ¶ms), + "ListStackSetOperationResults" => self.list_stack_set_operation_results(req, ¶ms), + "StopStackSetOperation" => self.stop_stack_set_operation(req, ¶ms), + "ImportStacksToStackSet" => self.import_stacks_to_stack_set(req, ¶ms), + "ListStackSetAutoDeploymentTargets" => { + self.list_stack_set_auto_deployment_targets(req, ¶ms) + } + "DetectStackSetDrift" => self.detect_stack_set_drift(req, ¶ms), + "ListStackInstanceResourceDrifts" => { + self.list_stack_instance_resource_drifts(req, ¶ms) + } + other => Err(validation(format!("Unsupported stack set action {other}"))), + } + } + + /// The account whose stack sets a call addresses. `CallAs=DELEGATED_ADMIN` + /// lets a registered StackSets delegated administrator act on the + /// organization's service-managed stack sets, which live in the + /// management account. + fn stack_set_admin_account( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + match params.get("CallAs").map(String::as_str) { + None | Some("SELF") => Ok(req.account_id.clone()), + Some("DELEGATED_ADMIN") => { + let orgs = self.deps.organizations.read(); + let org = orgs.as_ref().ok_or_else(|| { + validation("AWS Organizations is not enabled for this account") + })?; + let registered = org + .delegated_administrators + .get(STACKSETS_PRINCIPAL) + .is_some_and(|admins| admins.contains_key(&req.account_id)); + if !registered { + return Err(validation(format!( + "Account {} is not registered as a delegated administrator for {STACKSETS_PRINCIPAL}", + req.account_id + ))); + } + Ok(org.management_account_id.clone()) + } + Some(other) => Err(validation(format!("Invalid value {other} for CallAs"))), + } + } + + /// Service-managed stack sets need an organization with StackSets trusted + /// access, administered from its management account. + fn check_service_managed_allowed(&self, admin: &str) -> Result<(), AwsServiceError> { + let trusted_in_org = { + let orgs = self.deps.organizations.read(); + let org = orgs + .as_ref() + .ok_or_else(|| validation("AWS Organizations is not enabled for this account"))?; + if org.management_account_id != admin { + return Err(validation( + "Service managed stack sets can only be administered from the organization's management account or a delegated administrator", + )); + } + org.trusted_services.contains_key(STACKSETS_PRINCIPAL) + }; + let activated = self + .state + .read() + .get(admin) + .is_some_and(|s| s.orgs_access_enabled); + if trusted_in_org || activated { + Ok(()) + } else { + Err(validation( + "You must enable organizations access to operate a service managed stack set", + )) + } + } + + // ── Stack sets ── + + fn create_stack_set( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let permission_model = params + .get("PermissionModel") + .cloned() + .unwrap_or_else(|| "SELF_MANAGED".to_string()); + let service_managed = permission_model == "SERVICE_MANAGED"; + if service_managed { + self.check_service_managed_allowed(&admin)?; + } + + // A stack set can be created from an existing stack, which is how + // ImportStacksToStackSet adoption starts. + let (template_body, mut parameters) = match params.get("StackId") { + Some(stack_id) => { + let accounts = self.state.read(); + let stack = accounts + .get(&admin) + .and_then(|s| { + s.stacks + .values() + .find(|st| &st.stack_id == stack_id && st.status != "DELETE_COMPLETE") + }) + .ok_or_else(|| { + validation(format!("Stack with id {stack_id} does not exist")) + })?; + let params: BTreeMap = stack + .parameters + .iter() + .filter(|(k, _)| !k.starts_with("AWS::")) + .map(|(k, v)| (k.clone(), v.clone())) + .collect(); + (stack.template.clone(), params) + } + None => ( + self.stack_set_template_body(&admin, params) + .map_err(validation)? + .unwrap_or_default(), + BTreeMap::new(), + ), + }; + parameters.extend(resolve_parameters( + ¶meter_list(params, "Parameters"), + &BTreeMap::new(), + )?); + + let auto_deployment = Self::parse_auto_deployment(params, service_managed, None)?; + let managed_execution_active = + parse_bool(params, "ManagedExecution.Active")?.unwrap_or(false); + let (administration_role_arn, execution_role_name) = if service_managed { + (None, None) + } else { + ( + Some( + params + .get("AdministrationRoleARN") + .cloned() + .unwrap_or_else(|| { + format!("arn:aws:iam::{admin}:role/{DEFAULT_ADMIN_ROLE}") + }), + ), + Some( + params + .get("ExecutionRoleName") + .cloned() + .unwrap_or_else(|| DEFAULT_EXECUTION_ROLE.to_string()), + ), + ) + }; + + let id = format!("{name}:{}", uuid::Uuid::new_v4()); + let arn = format!( + "arn:aws:cloudformation:{}:{admin}:stackset/{id}", + req.region + ); + let mut accounts = self.state.write(); + let state = accounts.get_or_create(&admin); + if find_active(state, &name).is_some() { + return Err(aws_err( + StatusCode::CONFLICT, + "NameAlreadyExistsException", + format!("StackSet {name} already exists"), + )); + } + state.stack_sets.insert( + id.clone(), + StackSet { + stack_set_id: id.clone(), + name, + arn, + status: "ACTIVE".to_string(), + description: params.get("Description").cloned(), + template_body, + parameters, + capabilities: member_list(params, "Capabilities"), + tags: tag_list(params), + administration_role_arn, + execution_role_name, + permission_model, + auto_deployment, + managed_execution_active, + instances: Vec::new(), + operations: Vec::new(), + drift: None, + created_at: Utc::now(), + }, + ); + Ok(xml_response( + "CreateStackSet", + el("StackSetId", &id), + &req.request_id, + )) + } + + fn parse_auto_deployment( + params: &BTreeMap, + service_managed: bool, + previous: Option<&AutoDeployment>, + ) -> Result, AwsServiceError> { + let enabled = parse_bool(params, "AutoDeployment.Enabled")?; + let retain = parse_bool(params, "AutoDeployment.RetainStacksOnAccountRemoval")?; + if enabled.is_none() && retain.is_none() { + return Ok(previous.cloned()); + } + if !service_managed { + return Err(validation( + "AutoDeployment is only supported for stack sets with SERVICE_MANAGED permission model", + )); + } + let enabled = enabled.or(previous.map(|p| p.enabled)).unwrap_or(false); + let retain = retain + .or(previous.map(|p| p.retain_stacks_on_account_removal)) + .unwrap_or(false); + if retain && !enabled { + return Err(validation( + "RetainStacksOnAccountRemoval can only be set when AutoDeployment is enabled", + )); + } + Ok(Some(AutoDeployment { + enabled, + retain_stacks_on_account_removal: retain, + })) + } + + /// Fold asynchronously-provisioning stacks' current status into the + /// instances and operations of one stack set. + fn refresh_stack_set( + accounts: &mut MultiAccountState, + admin: &str, + set_id: &str, + ) { + let running: Vec<(usize, String, String, Option)> = + match accounts.get(admin).and_then(|s| s.stack_sets.get(set_id)) { + Some(set) => set + .instances + .iter() + .enumerate() + .filter(|(_, i)| i.detailed_status == "RUNNING") + .filter_map(|(idx, i)| { + Some(( + idx, + i.account.clone(), + i.stack_id.clone()?, + i.last_operation_id.clone(), + )) + }) + .collect(), + None => return, + }; + let mut outcomes = Vec::new(); + for (idx, account, stack_id, op_id) in running { + let Some((status, reason)) = accounts.get(&account).and_then(|s| { + s.stacks + .values() + .find(|st| st.stack_id == stack_id) + .map(|st| (st.status.clone(), st.status_reason.clone())) + }) else { + outcomes.push(( + idx, + op_id, + Outcome::Failed(format!("Stack {stack_id} does not exist")), + )); + continue; + }; + let action = if status.starts_with("UPDATE") { + "UPDATE" + } else { + "CREATE" + }; + let outcome = stack_outcome(action, &status, reason.as_deref()); + if outcome != Outcome::Running { + outcomes.push((idx, op_id, outcome)); + } + } + let Some(set) = accounts + .get_mut(admin) + .and_then(|s| s.stack_sets.get_mut(set_id)) + else { + return; + }; + for (idx, op_id, outcome) in outcomes { + let (account, region) = { + let instance = &mut set.instances[idx]; + apply_to_instance(instance, &outcome); + (instance.account.clone(), instance.region.clone()) + }; + if let Some(op) = op_id + .as_deref() + .and_then(|id| set.operations.iter_mut().find(|o| o.operation_id == id)) + { + if let Some(result) = op + .results + .iter_mut() + .find(|r| r.account == account && r.region == region) + { + result.status = result_status(&outcome).to_string(); + result.status_reason = outcome_reason(&outcome); + } + } + } + for op in &mut set.operations { + settle_operation(op); + } + } + + /// Resolve the stack set for a read, after folding in async progress. + fn read_stack_set(&self, admin: &str, name_or_id: &str) -> Result { + let mut accounts = self.state.write(); + let id = accounts + .get(admin) + .and_then(|s| find_for_read(s, name_or_id)) + .map(|s| s.stack_set_id.clone()) + .ok_or_else(|| stack_set_not_found(name_or_id))?; + Self::refresh_stack_set(&mut accounts, admin, &id); + accounts + .get(admin) + .and_then(|s| s.stack_sets.get(&id)) + .cloned() + .ok_or_else(|| stack_set_not_found(name_or_id)) + } + + fn describe_stack_set( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let set = self.read_stack_set(&admin, &name)?; + Ok(xml_response( + "DescribeStackSet", + stack_set_el(&set), + &req.request_id, + )) + } + + fn list_stack_sets( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let admin = self.stack_set_admin_account(req, params)?; + let wanted = params.get("Status"); + let mut sets: Vec = self + .state + .read() + .get(&admin) + .map(|s| { + s.stack_sets + .values() + .filter(|set| wanted.is_none_or(|w| &set.status == w)) + .filter(|set| { + // A delegated administrator only sees service-managed + // stack sets. + admin == req.account_id || set.permission_model == "SERVICE_MANAGED" + }) + .cloned() + .collect() + }) + .unwrap_or_default(); + sets.sort_by_key(|a| a.created_at); + let (page, next) = paginate(sets, params)?; + let inner = format!( + "{}{}", + list_el("Summaries", page.iter().map(stack_set_summary_el)), + next_token_el(next) + ); + Ok(xml_response("ListStackSets", inner, &req.request_id)) + } + + /// Reject a new operation on a stack set that already has one running, and + /// a caller-supplied operation id that was used before. + fn check_can_start_operation(set: &StackSet, op_id: &str) -> Result<(), AwsServiceError> { + if set + .operations + .iter() + .any(|o| matches!(o.status.as_str(), "RUNNING" | "STOPPING")) + { + return Err(aws_err( + StatusCode::CONFLICT, + "OperationInProgressException", + format!( + "Another Operation on StackSet {} is in progress", + set.stack_set_id + ), + )); + } + if set.operations.iter().any(|o| o.operation_id == op_id) { + return Err(aws_err( + StatusCode::CONFLICT, + "OperationIdAlreadyExistsException", + format!("Operation {op_id} already exists"), + )); + } + Ok(()) + } + + fn new_operation( + set: &StackSet, + op_id: &str, + action: &str, + preferences: OperationPreferences, + deployment_targets: Option, + retain_stacks: Option, + ) -> StackSetOperation { + StackSetOperation { + operation_id: op_id.to_string(), + action: action.to_string(), + status: "RUNNING".to_string(), + status_reason: None, + retain_stacks, + preferences, + deployment_targets, + administration_role_arn: set.administration_role_arn.clone(), + execution_role_name: set.execution_role_name.clone(), + created_at: Utc::now(), + ended_at: None, + results: Vec::new(), + drift: None, + resource_drifts: Vec::new(), + } + } + + async fn update_stack_set( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let preferences = parse_preferences(params)?; + let op_id = params + .get("OperationId") + .cloned() + .unwrap_or_else(|| uuid::Uuid::new_v4().to_string()); + // Resolved before the state lock: a TemplateURL read takes the S3 lock. + let new_template = self + .stack_set_template_body(&admin, params) + .map_err(validation)?; + let use_previous_template = parse_bool(params, "UsePreviousTemplate")?.unwrap_or(false); + if use_previous_template && new_template.is_some() { + return Err(validation( + "UsePreviousTemplate cannot be specified together with TemplateBody or TemplateURL", + )); + } + let explicit_regions = member_list(params, "Regions"); + let explicit_accounts = member_list(params, "Accounts"); + let deployment_targets = parse_deployment_targets(params); + + // Build the updated definition and resolve targets against a snapshot, + // without holding the CloudFormation lock while Organizations and S3 + // are read. The snapshot is re-validated under the lock below. + let snapshot = self.active_snapshot(&admin, &name)?; + Self::check_can_start_operation(&snapshot, &op_id)?; + let mut updated = snapshot.clone(); + if let Some(body) = new_template { + updated.template_body = body; + } + let entries = parameter_list(params, "Parameters"); + if !entries.is_empty() { + updated.parameters = resolve_parameters(&entries, &snapshot.parameters)?; + } + if let Some(description) = params.get("Description") { + updated.description = Some(description.clone()); + } + if list_present(params, "Capabilities") { + updated.capabilities = member_list(params, "Capabilities"); + } + if list_present(params, "Tags") { + updated.tags = tag_list(params); + } + if let Some(model) = params.get("PermissionModel") { + if *model != snapshot.permission_model { + if !snapshot.instances.is_empty() { + return Err(validation( + "PermissionModel cannot be changed for a stack set that has stack instances", + )); + } + updated.permission_model = model.clone(); + } + } + let service_managed = updated.permission_model == "SERVICE_MANAGED"; + if service_managed && snapshot.permission_model != "SERVICE_MANAGED" { + self.check_service_managed_allowed(&admin)?; + } + if service_managed { + updated.administration_role_arn = None; + updated.execution_role_name = None; + } else { + if let Some(role) = params.get("AdministrationRoleARN") { + updated.administration_role_arn = Some(role.clone()); + } + if let Some(role) = params.get("ExecutionRoleName") { + updated.execution_role_name = Some(role.clone()); + } + updated + .administration_role_arn + .get_or_insert_with(|| format!("arn:aws:iam::{admin}:role/{DEFAULT_ADMIN_ROLE}")); + updated + .execution_role_name + .get_or_insert_with(|| DEFAULT_EXECUTION_ROLE.to_string()); + } + updated.auto_deployment = Self::parse_auto_deployment( + params, + service_managed, + snapshot.auto_deployment.as_ref(), + )?; + if let Some(active) = parse_bool(params, "ManagedExecution.Active")? { + updated.managed_execution_active = active; + } + + // Which instances this update re-deploys: those named by + // Accounts/DeploymentTargets + Regions, or all of them. + let targeted = !explicit_accounts.is_empty() || deployment_targets.is_some(); + if targeted && explicit_regions.is_empty() { + return Err(validation( + "Regions must be specified when Accounts or DeploymentTargets are specified", + )); + } + if !targeted && !explicit_regions.is_empty() { + return Err(validation( + "Accounts or DeploymentTargets must be specified when Regions are specified", + )); + } + let (targets, regions) = if targeted { + let targets = self.existing_instance_targets( + &updated, + &admin, + &explicit_accounts, + deployment_targets.as_ref(), + &explicit_regions, + true, + )?; + (targets, explicit_regions.clone()) + } else { + let targets = updated + .instances + .iter() + .map(|i| Target { + account: i.account.clone(), + region: i.region.clone(), + ou: i.organizational_unit_id.clone(), + suspended: false, + }) + .collect(); + (targets, stack_set_regions(&updated)) + }; + // Instances left out of a partial update fall behind the new stack + // set definition. + for instance in &mut updated.instances { + if !targets + .iter() + .any(|t| t.account == instance.account && t.region == instance.region) + { + instance.status = "OUTDATED".to_string(); + } + } + let targets = order_targets(targets, ®ions, &preferences); + let record = + targeted.then(|| targets_record(&explicit_accounts, deployment_targets.as_ref())); + let op = Self::new_operation(&updated, &op_id, "UPDATE", preferences, record, None); + updated.operations.push(op); + let spec = DeploySpec::of(&updated); + let set_id = updated.stack_set_id.clone(); + { + let mut accounts = self.state.write(); + Self::refresh_stack_set(&mut accounts, &admin, &set_id); + let current = accounts + .get(&admin) + .and_then(|s| s.stack_sets.get(&set_id)) + .filter(|s| s.status == "ACTIVE") + .ok_or_else(|| stack_set_not_found(&name))?; + Self::check_not_stale(current, &snapshot)?; + Self::check_can_start_operation(current, &op_id)?; + accounts + .get_or_create(&admin) + .stack_sets + .insert(set_id.clone(), updated); + } + + self.run_operation( + req, + &admin, + &set_id, + &op_id, + &spec, + targets, + TargetAction::Update { overrides: None }, + ) + .await; + Ok(xml_response( + "UpdateStackSet", + el("OperationId", &op_id), + &req.request_id, + )) + } + + /// A clone of the ACTIVE stack set `name`, with async progress folded in. + fn active_snapshot(&self, admin: &str, name: &str) -> Result { + let mut accounts = self.state.write(); + let set_id = accounts + .get(admin) + .and_then(|s| active_key(s, name)) + .ok_or_else(|| stack_set_not_found(name))?; + Self::refresh_stack_set(&mut accounts, admin, &set_id); + accounts + .get(admin) + .and_then(|s| s.stack_sets.get(&set_id)) + .cloned() + .ok_or_else(|| stack_set_not_found(name)) + } + + /// Reject a request planned against a snapshot that another operation has + /// since changed. + fn check_not_stale(current: &StackSet, snapshot: &StackSet) -> Result<(), AwsServiceError> { + if current.operations.len() != snapshot.operations.len() { + return Err(aws_err( + StatusCode::CONFLICT, + "StaleRequestException", + format!( + "Another operation has been performed on StackSet {} since this request was made", + current.stack_set_id + ), + )); + } + Ok(()) + } + + fn delete_stack_set( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let mut accounts = self.state.write(); + // DeleteStackSet declares no not-found error; deleting a stack set that + // does not exist is a no-op. + let Some(set_id) = accounts.get(&admin).and_then(|s| active_key(s, &name)) else { + return Ok(xml_response_no_result("DeleteStackSet", &req.request_id)); + }; + Self::refresh_stack_set(&mut accounts, &admin, &set_id); + let state = accounts.get_or_create(&admin); + let Some(set) = state.stack_sets.get_mut(&set_id) else { + return Ok(xml_response_no_result("DeleteStackSet", &req.request_id)); + }; + if set + .operations + .iter() + .any(|o| matches!(o.status.as_str(), "RUNNING" | "STOPPING")) + { + return Err(aws_err( + StatusCode::CONFLICT, + "OperationInProgressException", + format!("Another Operation on StackSet {set_id} is in progress"), + )); + } + if !set.instances.is_empty() { + return Err(aws_err( + StatusCode::CONFLICT, + "StackSetNotEmptyException", + format!("StackSet {name} is not empty"), + )); + } + set.status = "DELETED".to_string(); + Ok(xml_response_no_result("DeleteStackSet", &req.request_id)) + } + + // ── Stack instances ── + + /// Resolve the targets of CreateStackInstances from the request. + fn resolve_new_targets( + &self, + set: &StackSet, + admin: &str, + accounts: &[String], + deployment_targets: Option<&DeploymentTargets>, + regions: &[String], + ) -> Result, AwsServiceError> { + if regions.is_empty() { + return Err(validation("Regions is required")); + } + if !accounts.is_empty() && deployment_targets.is_some() { + return Err(validation( + "Only one of Accounts or DeploymentTargets can be specified", + )); + } + let mut resolved: Vec<(String, Option, bool)> = Vec::new(); + if set.permission_model == "SERVICE_MANAGED" { + if !accounts.is_empty() { + return Err(validation( + "StackSets with SERVICE_MANAGED permission model can only have OrganizationalUnit as target", + )); + } + let dt = deployment_targets + .filter(|t| !t.organizational_unit_ids.is_empty()) + .ok_or_else(|| { + validation("DeploymentTargets.OrganizationalUnitIds is required for SERVICE_MANAGED stack sets") + })?; + let filter_accounts = self.target_accounts_list(admin, dt)?; + let filter = self.account_filter_type(dt, &filter_accounts)?; + let orgs = self.deps.organizations.read(); + let org = orgs + .as_ref() + .ok_or_else(|| validation("AWS Organizations is not enabled for this account"))?; + let mut seen = BTreeSet::new(); + for ou in &dt.organizational_unit_ids { + if *ou != org.root_id && !org.ous.contains_key(ou) { + return Err(validation(format!( + "OrganizationalUnit {ou} does not exist" + ))); + } + for (account, suspended) in accounts_under(org, ou) { + let listed = filter_accounts.contains(&account); + let keep = match filter.as_str() { + "INTERSECTION" => listed, + "DIFFERENCE" => !listed, + _ => true, + }; + if keep && seen.insert(account.clone()) { + resolved.push((account, Some(ou.clone()), suspended)); + } + } + } + if filter == "UNION" { + for account in filter_accounts { + if account != org.management_account_id && seen.insert(account.clone()) { + let suspended = org + .accounts + .get(&account) + .is_some_and(|a| a.status != "ACTIVE"); + resolved.push((account, None, suspended)); + } + } + } + } else { + let list = match deployment_targets { + Some(dt) => { + if !dt.organizational_unit_ids.is_empty() { + return Err(validation( + "OrganizationalUnitIds are only supported for stack sets with SERVICE_MANAGED permission model", + )); + } + self.target_accounts_list(admin, dt)? + } + None => accounts.to_vec(), + }; + if list.is_empty() { + return Err(validation( + "Accounts or DeploymentTargets must be specified", + )); + } + if let Some(bad) = list.iter().find(|a| !is_account_id(a)) { + return Err(validation(format!( + "Account {bad} is not a valid AWS account id" + ))); + } + let mut seen = BTreeSet::new(); + for account in list { + if seen.insert(account.clone()) { + resolved.push((account, None, false)); + } + } + } + let mut targets = Vec::new(); + for region in regions { + for (account, ou, suspended) in &resolved { + targets.push(Target { + account: account.clone(), + region: region.clone(), + ou: ou.clone(), + suspended: *suspended, + }); + } + } + Ok(targets) + } + + /// `DeploymentTargets.Accounts` plus the accounts listed in + /// `DeploymentTargets.AccountsUrl`, validated. + fn target_accounts_list( + &self, + admin: &str, + dt: &DeploymentTargets, + ) -> Result, AwsServiceError> { + let mut list = dt.accounts.clone(); + if let Some(url) = &dt.accounts_url { + if !looks_like_url(url) { + return Err(validation(format!( + "AccountsUrl {url} is not a valid S3 URL" + ))); + } + let body = self + .resolve_template_url(admin, url) + .map_err(|_| validation(format!("Unable to read the accounts file at {url}")))?; + list.extend( + body.split([',', '\n', '\r']) + .map(str::trim) + .filter(|a| !a.is_empty()) + .map(str::to_string), + ); + } + if let Some(bad) = list.iter().find(|a| !is_account_id(a)) { + return Err(validation(format!( + "Account {bad} is not a valid AWS account id" + ))); + } + Ok(list) + } + + fn account_filter_type( + &self, + dt: &DeploymentTargets, + filter_accounts: &[String], + ) -> Result { + let filter = match (&dt.account_filter_type, filter_accounts.is_empty()) { + (Some(f), _) => f.clone(), + // Accounts next to OUs without an explicit filter narrow the OUs. + (None, false) => "INTERSECTION".to_string(), + (None, true) => "NONE".to_string(), + }; + match filter.as_str() { + "NONE" if !filter_accounts.is_empty() => Err(validation( + "AccountFilterType NONE cannot be used together with Accounts", + )), + "INTERSECTION" | "DIFFERENCE" | "UNION" if filter_accounts.is_empty() => { + Err(validation(format!( + "Accounts must be specified when AccountFilterType is {filter}" + ))) + } + "NONE" | "INTERSECTION" | "DIFFERENCE" | "UNION" => Ok(filter), + other => Err(validation(format!("Invalid AccountFilterType {other}"))), + } + } + + /// Resolve targets that must name existing instances (UpdateStackInstances, + /// DeleteStackInstances, a partial UpdateStackSet). + fn existing_instance_targets( + &self, + set: &StackSet, + admin: &str, + accounts: &[String], + deployment_targets: Option<&DeploymentTargets>, + regions: &[String], + must_exist: bool, + ) -> Result, AwsServiceError> { + if regions.is_empty() { + return Err(validation("Regions is required")); + } + if !accounts.is_empty() && deployment_targets.is_some() { + return Err(validation( + "Only one of Accounts or DeploymentTargets can be specified", + )); + } + let mut targets = Vec::new(); + let mut push = |instance: &StackInstance| { + if !targets + .iter() + .any(|t: &Target| t.account == instance.account && t.region == instance.region) + { + targets.push(Target { + account: instance.account.clone(), + region: instance.region.clone(), + ou: instance.organizational_unit_id.clone(), + suspended: false, + }); + } + }; + if set.permission_model == "SERVICE_MANAGED" { + if !accounts.is_empty() { + return Err(validation( + "StackSets with SERVICE_MANAGED permission model can only have OrganizationalUnit as target", + )); + } + let dt = deployment_targets + .filter(|t| !t.organizational_unit_ids.is_empty()) + .ok_or_else(|| { + validation("DeploymentTargets.OrganizationalUnitIds is required for SERVICE_MANAGED stack sets") + })?; + let filter_accounts = self.target_accounts_list(admin, dt)?; + let filter = self.account_filter_type(dt, &filter_accounts)?; + for region in regions { + for instance in set.instances.iter().filter(|i| &i.region == region) { + let in_ou = instance + .organizational_unit_id + .as_ref() + .is_some_and(|ou| dt.organizational_unit_ids.contains(ou)); + let listed = filter_accounts.contains(&instance.account); + let keep = match filter.as_str() { + "INTERSECTION" => in_ou && listed, + "DIFFERENCE" => in_ou && !listed, + "UNION" => in_ou || listed, + _ => in_ou, + }; + if keep { + push(instance); + } + } + } + } else { + let list = match deployment_targets { + Some(dt) => { + if !dt.organizational_unit_ids.is_empty() { + return Err(validation( + "OrganizationalUnitIds are only supported for stack sets with SERVICE_MANAGED permission model", + )); + } + self.target_accounts_list(admin, dt)? + } + None => accounts.to_vec(), + }; + if list.is_empty() { + return Err(validation( + "Accounts or DeploymentTargets must be specified", + )); + } + if let Some(bad) = list.iter().find(|a| !is_account_id(a)) { + return Err(validation(format!( + "Account {bad} is not a valid AWS account id" + ))); + } + for region in regions { + for account in &list { + match set + .instances + .iter() + .find(|i| &i.account == account && &i.region == region) + { + Some(instance) => push(instance), + None if must_exist => { + return Err(instance_not_found(&set.name, account, region)) + } + None => {} + } + } + } + } + Ok(targets) + } + + /// Validate that overrides only touch parameters the template declares. + fn check_overrides_declared( + set: &StackSet, + keys: impl Iterator, + ) -> Result<(), AwsServiceError> { + let Ok(template) = fakecloud_core::cfn_template::parse_template_body(&set.template_body) + else { + return Ok(()); + }; + let declared: BTreeSet = template + .get("Parameters") + .and_then(Value::as_object) + .map(|o| o.keys().cloned().collect()) + .unwrap_or_default(); + for key in keys { + if !declared.contains(&key) && !set.parameters.contains_key(&key) { + return Err(validation(format!( + "Parameter {key} is not declared in the stack set template" + ))); + } + } + Ok(()) + } + + /// Record a new instance operation, re-validating under the lock the + /// snapshot its targets were resolved against. + #[allow(clippy::too_many_arguments)] + fn start_instance_operation( + &self, + admin: &str, + snapshot: &StackSet, + op_id: &str, + action: &str, + preferences: OperationPreferences, + deployment_targets: Option, + retain_stacks: Option, + ) -> Result { + let mut accounts = self.state.write(); + Self::refresh_stack_set(&mut accounts, admin, &snapshot.stack_set_id); + let set = accounts + .get_mut(admin) + .and_then(|s| s.stack_sets.get_mut(&snapshot.stack_set_id)) + .filter(|s| s.status == "ACTIVE") + .ok_or_else(|| stack_set_not_found(&snapshot.name))?; + Self::check_not_stale(set, snapshot)?; + Self::check_can_start_operation(set, op_id)?; + let op = Self::new_operation( + set, + op_id, + action, + preferences, + deployment_targets, + retain_stacks, + ); + set.operations.push(op); + Ok(DeploySpec::of(set)) + } + + async fn create_stack_instances( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let regions = member_list(params, "Regions"); + let accounts = member_list(params, "Accounts"); + let deployment_targets = parse_deployment_targets(params); + let preferences = parse_preferences(params)?; + let op_id = params + .get("OperationId") + .cloned() + .unwrap_or_else(|| uuid::Uuid::new_v4().to_string()); + let overrides = resolve_parameters( + ¶meter_list(params, "ParameterOverrides"), + &BTreeMap::new(), + )?; + + // Target resolution reads Organizations (and possibly S3) state, so it + // runs against a snapshot of the stack set before the operation is + // recorded under the CloudFormation lock. + let snapshot = self.active_snapshot(&admin, &name)?; + Self::check_overrides_declared(&snapshot, overrides.keys().cloned())?; + let targets = self.resolve_new_targets( + &snapshot, + &admin, + &accounts, + deployment_targets.as_ref(), + ®ions, + )?; + let targets = order_targets(targets, ®ions, &preferences); + let record = Some(targets_record(&accounts, deployment_targets.as_ref())); + let spec = self.start_instance_operation( + &admin, + &snapshot, + &op_id, + "CREATE", + preferences, + record, + None, + )?; + let set_id = snapshot.stack_set_id.clone(); + self.run_operation( + req, + &admin, + &set_id, + &op_id, + &spec, + targets, + TargetAction::Create { overrides }, + ) + .await; + Ok(xml_response( + "CreateStackInstances", + el("OperationId", &op_id), + &req.request_id, + )) + } + + async fn update_stack_instances( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let regions = member_list(params, "Regions"); + let accounts = member_list(params, "Accounts"); + let deployment_targets = parse_deployment_targets(params); + let preferences = parse_preferences(params)?; + let op_id = params + .get("OperationId") + .cloned() + .unwrap_or_else(|| uuid::Uuid::new_v4().to_string()); + let overrides = list_present(params, "ParameterOverrides").then(|| { + parameter_list(params, "ParameterOverrides") + .into_iter() + .map(|e| match (e.value, e.use_previous) { + (Some(v), false) => Ok(OverrideSpec::Value(e.key, v)), + (None, true) => Ok(OverrideSpec::UsePrevious(e.key)), + (Some(_), true) => Err(validation(format!( + "Invalid input for parameter key {}. Cannot specify usePreviousValue as true and a parameter value at the same time", + e.key + ))), + (None, false) => Err(validation(format!( + "Invalid input for parameter key {}. Need to specify either usePreviousValue as true or a value for the parameter", + e.key + ))), + }) + .collect::, _>>() + }); + let overrides = overrides.transpose()?; + + let snapshot = self.active_snapshot(&admin, &name)?; + if let Some(specs) = &overrides { + Self::check_overrides_declared( + &snapshot, + specs.iter().map(|s| match s { + OverrideSpec::Value(k, _) | OverrideSpec::UsePrevious(k) => k.clone(), + }), + )?; + } + let targets = self.existing_instance_targets( + &snapshot, + &admin, + &accounts, + deployment_targets.as_ref(), + ®ions, + true, + )?; + let targets = order_targets(targets, ®ions, &preferences); + let record = Some(targets_record(&accounts, deployment_targets.as_ref())); + let spec = self.start_instance_operation( + &admin, + &snapshot, + &op_id, + "UPDATE", + preferences, + record, + None, + )?; + let set_id = snapshot.stack_set_id.clone(); + self.run_operation( + req, + &admin, + &set_id, + &op_id, + &spec, + targets, + TargetAction::Update { overrides }, + ) + .await; + Ok(xml_response( + "UpdateStackInstances", + el("OperationId", &op_id), + &req.request_id, + )) + } + + async fn delete_stack_instances( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let retain_stacks = parse_bool(params, "RetainStacks")? + .ok_or_else(|| validation("RetainStacks is required"))?; + let regions = member_list(params, "Regions"); + let accounts = member_list(params, "Accounts"); + let deployment_targets = parse_deployment_targets(params); + let preferences = parse_preferences(params)?; + let op_id = params + .get("OperationId") + .cloned() + .unwrap_or_else(|| uuid::Uuid::new_v4().to_string()); + + let snapshot = self.active_snapshot(&admin, &name)?; + let targets = self.existing_instance_targets( + &snapshot, + &admin, + &accounts, + deployment_targets.as_ref(), + ®ions, + false, + )?; + let targets = order_targets(targets, ®ions, &preferences); + let record = Some(targets_record(&accounts, deployment_targets.as_ref())); + let spec = self.start_instance_operation( + &admin, + &snapshot, + &op_id, + "DELETE", + preferences, + record, + Some(retain_stacks), + )?; + let set_id = snapshot.stack_set_id.clone(); + self.run_operation( + req, + &admin, + &set_id, + &op_id, + &spec, + targets, + TargetAction::Delete { retain_stacks }, + ) + .await; + Ok(xml_response( + "DeleteStackInstances", + el("OperationId", &op_id), + &req.request_id, + )) + } + + /// Run an operation's targets in order, recording each outcome as it + /// lands, honoring the failure tolerance and StopStackSetOperation. + #[allow(clippy::too_many_arguments)] + async fn run_operation( + &self, + req: &AwsRequest, + admin: &str, + set_id: &str, + op_id: &str, + spec: &DeploySpec, + targets: Vec, + action: TargetAction, + ) { + // Seed a PENDING result per target so the operation reports every + // target it will act on from the start. + { + let mut accounts = self.state.write(); + if let Some(op) = accounts + .get_mut(admin) + .and_then(|s| s.stack_sets.get_mut(set_id)) + .and_then(|set| set.operations.iter_mut().find(|o| o.operation_id == op_id)) + { + op.results = targets + .iter() + .map(|t| OperationResult { + account: t.account.clone(), + region: t.region.clone(), + status: "PENDING".to_string(), + status_reason: None, + organizational_unit_id: t.ou.clone(), + account_gate_status: None, + account_gate_reason: None, + }) + .collect(); + } + } + + let prefs = { + let accounts = self.state.read(); + accounts + .get(admin) + .and_then(|s| s.stack_sets.get(set_id)) + .and_then(|set| set.operations.iter().find(|o| o.operation_id == op_id)) + .map(|o| o.preferences.clone()) + .unwrap_or_default() + }; + let mut region_sizes: BTreeMap = BTreeMap::new(); + for t in &targets { + *region_sizes.entry(t.region.clone()).or_default() += 1; + } + let mut region_failures: BTreeMap = BTreeMap::new(); + let mut abort: Option<&'static str> = None; + + for target in targets { + if abort.is_none() + && self.operation_status(admin, set_id, op_id).as_deref() == Some("STOPPING") + { + abort = Some(OPERATION_STOPPED); + } + let mut gate = None; + let mut stack_id = None; + let mut overrides = None; + let outcome = if let Some(reason) = abort { + Outcome::Cancelled(reason.to_string()) + } else if target.suspended { + Outcome::SkippedSuspended + } else { + let g = self.account_gate(&target.account, &target.region).await; + let passed = g.status != "FAILED"; + let gate_reason = g.reason.clone(); + gate = Some(g); + if passed { + let (outcome, id, applied) = self + .apply_target(req, admin, set_id, spec, &target, &action) + .await; + stack_id = id; + overrides = applied; + outcome + } else { + Outcome::Failed( + gate_reason.unwrap_or_else(|| "Account gate check failed".to_string()), + ) + } + }; + if matches!(outcome, Outcome::Failed(_)) { + let failures = region_failures.entry(target.region.clone()).or_default(); + *failures += 1; + let size = region_sizes.get(&target.region).copied().unwrap_or(0); + if *failures > region_tolerance(&prefs, size) { + abort = Some(TOLERANCE_EXCEEDED); + } + } + self.record_outcome( + admin, set_id, op_id, &target, &action, &outcome, gate, stack_id, overrides, + ); + } + + let mut accounts = self.state.write(); + if let Some(op) = accounts + .get_mut(admin) + .and_then(|s| s.stack_sets.get_mut(set_id)) + .and_then(|set| set.operations.iter_mut().find(|o| o.operation_id == op_id)) + { + settle_operation(op); + } + } + + fn operation_status(&self, admin: &str, set_id: &str, op_id: &str) -> Option { + self.state + .read() + .get(admin) + .and_then(|s| s.stack_sets.get(set_id)) + .and_then(|set| set.operations.iter().find(|o| o.operation_id == op_id)) + .map(|o| o.status.clone()) + } + + /// Run the account's `AWSCloudFormationStackSetAccountGate` Lambda, if it + /// has one. A deployment proceeds only when the function answers + /// `SUCCEEDED`; an account without the function is not gated. + async fn account_gate(&self, account: &str, region: &str) -> GateResult { + let arn = format!("arn:aws:lambda:{region}:{account}:function:{ACCOUNT_GATE_FUNCTION}"); + let exists = self + .deps + .lambda + .read() + .get(account) + .is_some_and(|s| s.functions.contains_key(ACCOUNT_GATE_FUNCTION)); + if !exists { + return GateResult { + status: "SKIPPED", + reason: Some(format!("Function not found: {arn}")), + }; + } + match self.deps.delivery.invoke_lambda(&arn, "{}").await { + None => GateResult { + status: "SKIPPED", + reason: Some("Lambda invocation is not available".to_string()), + }, + Some(Err(e)) => GateResult { + status: "FAILED", + reason: Some(format!("Account gate function invocation failed: {e}")), + }, + Some(Ok(bytes)) => { + let status = serde_json::from_slice::(&bytes) + .ok() + .and_then(|v| v.get("Status").and_then(Value::as_str).map(str::to_string)); + match status.as_deref() { + Some("SUCCEEDED") => GateResult { + status: "SUCCEEDED", + reason: None, + }, + other => GateResult { + status: "FAILED", + reason: Some(format!( + "Account gate function returned {}", + other.unwrap_or("an invalid response") + )), + }, + } + } + } + } + + fn instance_stack(&self, admin: &str, set_id: &str, target: &Target) -> Option { + self.state + .read() + .get(admin) + .and_then(|s| s.stack_sets.get(set_id)) + .and_then(|set| { + set.instances + .iter() + .find(|i| i.account == target.account && i.region == target.region) + .cloned() + }) + } + + fn stack_status( + &self, + account: &str, + stack_id_or_name: &str, + ) -> Option<(String, String, Option)> { + self.state.read().get(account).and_then(|s| { + s.stacks + .values() + .filter(|st| st.stack_id == stack_id_or_name || st.name == stack_id_or_name) + .max_by_key(|st| st.created_at) + .map(|st| { + ( + st.stack_id.clone(), + st.status.clone(), + st.status_reason.clone(), + ) + }) + }) + } + + /// Deploy one target. Returns its outcome, the instance's stack id, and + /// the overrides the instance now carries. + async fn apply_target( + &self, + req: &AwsRequest, + admin: &str, + set_id: &str, + spec: &DeploySpec, + target: &Target, + action: &TargetAction, + ) -> (Outcome, Option, Option>) { + let existing = self.instance_stack(admin, set_id, target); + let live_stack = existing + .as_ref() + .and_then(|i| i.stack_id.as_deref()) + .and_then(|id| self.stack_status(&target.account, id)) + .filter(|(_, status, _)| status != "DELETE_COMPLETE"); + + match action { + TargetAction::Delete { retain_stacks } => { + let Some((stack_id, _, _)) = live_stack else { + return (Outcome::Succeeded, None, None); + }; + if *retain_stacks { + return (Outcome::Succeeded, Some(stack_id), None); + } + let request = synthetic_request( + &target.account, + &target.region, + "DeleteStack", + &req.request_id, + vec![("StackName".to_string(), stack_id.clone())], + ); + if let Err(e) = self.delete_stack(&request).await { + return (Outcome::Failed(e.message()), Some(stack_id), None); + } + let outcome = match self.stack_status(&target.account, &stack_id) { + Some((_, status, reason)) => { + stack_outcome("DELETE", &status, reason.as_deref()) + } + None => Outcome::Succeeded, + }; + (outcome, Some(stack_id), None) + } + TargetAction::Create { overrides } => match live_stack { + Some((stack_id, _, _)) => { + let (outcome, id) = self + .update_instance_stack(req, spec, target, &stack_id, overrides) + .await; + (outcome, id, Some(overrides.clone())) + } + None => { + let stack_name = format!( + "StackSet-{}-{}", + spec.name.replace(':', "-"), + uuid::Uuid::new_v4() + ); + let mut stack_params = spec.stack_params(overrides); + stack_params.push(("StackName".to_string(), stack_name.clone())); + let request = synthetic_request( + &target.account, + &target.region, + "CreateStack", + &req.request_id, + stack_params, + ); + if let Err(e) = self.create_stack(&request).await { + return (Outcome::Failed(e.message()), None, Some(overrides.clone())); + } + match self.stack_status(&target.account, &stack_name) { + Some((stack_id, status, reason)) => ( + stack_outcome("CREATE", &status, reason.as_deref()), + Some(stack_id), + Some(overrides.clone()), + ), + None => ( + Outcome::Failed(format!("Stack {stack_name} was not created")), + None, + Some(overrides.clone()), + ), + } + } + }, + TargetAction::Update { overrides } => { + let previous = existing + .as_ref() + .map(|i| i.parameter_overrides.clone()) + .unwrap_or_default(); + let resolved = match overrides { + None => previous, + Some(specs) => specs + .iter() + .filter_map(|s| match s { + OverrideSpec::Value(k, v) => Some((k.clone(), v.clone())), + OverrideSpec::UsePrevious(k) => { + previous.get(k).map(|v| (k.clone(), v.clone())) + } + }) + .collect(), + }; + let Some((stack_id, _, _)) = live_stack else { + let missing = existing.and_then(|i| i.stack_id).unwrap_or_default(); + return ( + Outcome::Failed(format!("Stack [{missing}] does not exist")), + None, + Some(resolved), + ); + }; + let (outcome, id) = self + .update_instance_stack(req, spec, target, &stack_id, &resolved) + .await; + (outcome, id, Some(resolved)) + } + } + } + + async fn update_instance_stack( + &self, + req: &AwsRequest, + spec: &DeploySpec, + target: &Target, + stack_id: &str, + overrides: &BTreeMap, + ) -> (Outcome, Option) { + let mut stack_params = spec.stack_params(overrides); + stack_params.push(("StackName".to_string(), stack_id.to_string())); + let request = synthetic_request( + &target.account, + &target.region, + "UpdateStack", + &req.request_id, + stack_params, + ); + if let Err(e) = self.update_stack(&request).await { + return (Outcome::Failed(e.message()), Some(stack_id.to_string())); + } + let outcome = match self.stack_status(&target.account, stack_id) { + Some((_, status, reason)) => stack_outcome("UPDATE", &status, reason.as_deref()), + None => Outcome::Failed(format!("Stack [{stack_id}] does not exist")), + }; + (outcome, Some(stack_id.to_string())) + } + + #[allow(clippy::too_many_arguments)] + fn record_outcome( + &self, + admin: &str, + set_id: &str, + op_id: &str, + target: &Target, + action: &TargetAction, + outcome: &Outcome, + gate: Option, + stack_id: Option, + overrides: Option>, + ) { + let mut accounts = self.state.write(); + let Some(set) = accounts + .get_mut(admin) + .and_then(|s| s.stack_sets.get_mut(set_id)) + else { + return; + }; + if let Some(result) = set + .operations + .iter_mut() + .find(|o| o.operation_id == op_id) + .and_then(|op| { + op.results + .iter_mut() + .find(|r| r.account == target.account && r.region == target.region) + }) + { + result.status = result_status(outcome).to_string(); + result.status_reason = outcome_reason(outcome); + if let Some(gate) = gate { + result.account_gate_status = Some(gate.status.to_string()); + result.account_gate_reason = gate.reason; + } + } + + let position = set + .instances + .iter() + .position(|i| i.account == target.account && i.region == target.region); + match (action, position) { + (TargetAction::Delete { .. }, Some(idx)) => { + if *outcome == Outcome::Succeeded { + set.instances.remove(idx); + } else if !matches!(outcome, Outcome::Cancelled(_)) { + let instance = &mut set.instances[idx]; + apply_to_instance(instance, outcome); + instance.last_operation_id = Some(op_id.to_string()); + } + } + (TargetAction::Delete { .. }, None) => {} + (_, position) => { + let idx = match position { + Some(idx) => idx, + None => { + set.instances.push(StackInstance { + account: target.account.clone(), + region: target.region.clone(), + stack_id: None, + status: "OUTDATED".to_string(), + detailed_status: "PENDING".to_string(), + status_reason: None, + parameter_overrides: BTreeMap::new(), + organizational_unit_id: target.ou.clone(), + drift_status: "NOT_CHECKED".to_string(), + last_drift_check_timestamp: None, + last_operation_id: None, + }); + set.instances.len() - 1 + } + }; + let instance = &mut set.instances[idx]; + apply_to_instance(instance, outcome); + instance.last_operation_id = Some(op_id.to_string()); + if stack_id.is_some() { + instance.stack_id = stack_id; + } + if let Some(overrides) = overrides { + instance.parameter_overrides = overrides; + } + if target.ou.is_some() { + instance.organizational_unit_id = target.ou.clone(); + } + } + } + } + + fn describe_stack_instance( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let account = required(params, "StackInstanceAccount")?; + let region = required(params, "StackInstanceRegion")?; + let admin = self.stack_set_admin_account(req, params)?; + let set = self.read_stack_set(&admin, &name)?; + let instance = set + .instances + .iter() + .find(|i| i.account == account && i.region == region) + .ok_or_else(|| instance_not_found(&name, &account, ®ion))?; + let inner = format!( + "{}", + instance_fields(&set, instance, true) + ); + Ok(xml_response( + "DescribeStackInstance", + inner, + &req.request_id, + )) + } + + fn list_stack_instances( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let set = self.read_stack_set(&admin, &name)?; + let mut filters: Vec<(String, String)> = Vec::new(); + for i in 1.. { + let Some(filter_name) = params.get(&format!("Filters.member.{i}.Name")) else { + break; + }; + let value = params + .get(&format!("Filters.member.{i}.Values")) + .cloned() + .unwrap_or_default(); + if !matches!( + filter_name.as_str(), + "DETAILED_STATUS" | "LAST_OPERATION_ID" | "DRIFT_STATUS" + ) { + return Err(validation(format!("Invalid filter name {filter_name}"))); + } + filters.push((filter_name.clone(), value)); + } + let account = params.get("StackInstanceAccount"); + let region = params.get("StackInstanceRegion"); + let matching: Vec<&StackInstance> = set + .instances + .iter() + .filter(|i| account.is_none_or(|a| &i.account == a)) + .filter(|i| region.is_none_or(|r| &i.region == r)) + .filter(|i| { + filters.iter().all(|(name, value)| match name.as_str() { + "DETAILED_STATUS" => &i.detailed_status == value, + "LAST_OPERATION_ID" => i.last_operation_id.as_ref() == Some(value), + _ => &i.drift_status == value, + }) + }) + .collect(); + let (page, next) = paginate(matching, params)?; + let inner = format!( + "{}{}", + list_el( + "Summaries", + page.iter().map(|i| instance_fields(&set, i, false)) + ), + next_token_el(next) + ); + Ok(xml_response("ListStackInstances", inner, &req.request_id)) + } + + // ── Operations ── + + fn describe_stack_set_operation( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let op_id = required(params, "OperationId")?; + let admin = self.stack_set_admin_account(req, params)?; + let set = self.read_stack_set(&admin, &name)?; + let op = set + .operations + .iter() + .find(|o| o.operation_id == op_id) + .ok_or_else(|| operation_not_found(&op_id))?; + Ok(xml_response( + "DescribeStackSetOperation", + operation_el(&set, op), + &req.request_id, + )) + } + + fn list_stack_set_operations( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let set = self.read_stack_set(&admin, &name)?; + // Most recent first. + let ops: Vec<&StackSetOperation> = set.operations.iter().rev().collect(); + let (page, next) = paginate(ops, params)?; + let inner = format!( + "{}{}", + list_el("Summaries", page.into_iter().map(operation_summary_el)), + next_token_el(next) + ); + Ok(xml_response( + "ListStackSetOperations", + inner, + &req.request_id, + )) + } + + fn list_stack_set_operation_results( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let op_id = required(params, "OperationId")?; + let admin = self.stack_set_admin_account(req, params)?; + let set = self.read_stack_set(&admin, &name)?; + let op = set + .operations + .iter() + .find(|o| o.operation_id == op_id) + .ok_or_else(|| operation_not_found(&op_id))?; + let mut wanted_status: Option = None; + for i in 1.. { + let Some(filter_name) = params.get(&format!("Filters.member.{i}.Name")) else { + break; + }; + if filter_name != "OPERATION_RESULT_STATUS" { + return Err(validation(format!("Invalid filter name {filter_name}"))); + } + wanted_status = params.get(&format!("Filters.member.{i}.Values")).cloned(); + } + let results: Vec<&OperationResult> = op + .results + .iter() + .filter(|r| wanted_status.as_ref().is_none_or(|s| &r.status == s)) + .collect(); + let (page, next) = paginate(results, params)?; + let inner = format!( + "{}{}", + list_el("Summaries", page.into_iter().map(operation_result_el)), + next_token_el(next) + ); + Ok(xml_response( + "ListStackSetOperationResults", + inner, + &req.request_id, + )) + } + + fn stop_stack_set_operation( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let op_id = required(params, "OperationId")?; + let admin = self.stack_set_admin_account(req, params)?; + let mut accounts = self.state.write(); + let set_id = accounts + .get(&admin) + .and_then(|s| find_for_read(s, &name)) + .map(|s| s.stack_set_id.clone()) + .ok_or_else(|| stack_set_not_found(&name))?; + Self::refresh_stack_set(&mut accounts, &admin, &set_id); + let op = accounts + .get_mut(&admin) + .and_then(|s| s.stack_sets.get_mut(&set_id)) + .and_then(|set| set.operations.iter_mut().find(|o| o.operation_id == op_id)) + .ok_or_else(|| operation_not_found(&op_id))?; + if op.status != "RUNNING" { + return Err(aws_err( + StatusCode::BAD_REQUEST, + "InvalidOperationException", + format!( + "Operation {op_id} is in {} state and cannot be stopped", + op.status + ), + )); + } + // Targets not yet started are cancelled; ones already deploying run to + // completion, and the operation settles as STOPPED once they do. + op.status = "STOPPING".to_string(); + for result in &mut op.results { + if result.status == "PENDING" { + result.status = "CANCELLED".to_string(); + result.status_reason = Some(OPERATION_STOPPED.to_string()); + } + } + settle_operation(op); + Ok(xml_response( + "StopStackSetOperation", + String::new(), + &req.request_id, + )) + } + + // ── Import ── + + fn import_stacks_to_stack_set( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let preferences = parse_preferences(params)?; + let op_id = params + .get("OperationId") + .cloned() + .unwrap_or_else(|| uuid::Uuid::new_v4().to_string()); + let ous = member_list(params, "OrganizationalUnitIds"); + let mut stack_ids = member_list(params, "StackIds"); + if let Some(url) = params.get("StackIdsUrl") { + if !stack_ids.is_empty() { + return Err(validation( + "Only one of StackIds or StackIdsUrl can be specified", + )); + } + if !looks_like_url(url) { + return Err(validation(format!( + "StackIdsUrl {url} is not a valid S3 URL" + ))); + } + let body = self + .resolve_template_url(&admin, url) + .map_err(|_| validation(format!("Unable to read the stack ids file at {url}")))?; + stack_ids = body + .split([',', '\n', '\r']) + .map(str::trim) + .filter(|s| !s.is_empty()) + .map(str::to_string) + .collect(); + } + if stack_ids.is_empty() { + return Err(validation("StackIds or StackIdsUrl must be specified")); + } + if stack_ids.len() > MAX_IMPORT_STACKS { + return Err(aws_err( + StatusCode::BAD_REQUEST, + "LimitExceededException", + format!("A maximum of {MAX_IMPORT_STACKS} stacks can be imported in one operation"), + )); + } + + // Where each stack lives, and which OU its account sits in. + let mut located = Vec::new(); + for stack_id in &stack_ids { + let (account, region) = stack_arn_location(stack_id) + .ok_or_else(|| validation(format!("Invalid stack id {stack_id}")))?; + located.push((stack_id.clone(), account, region)); + } + + // Which targeted OU each account sits in, read before the + // CloudFormation lock is taken. + let mut ou_of_account: BTreeMap = BTreeMap::new(); + if !ous.is_empty() { + let orgs = self.deps.organizations.read(); + let org = orgs + .as_ref() + .ok_or_else(|| validation("AWS Organizations is not enabled for this account"))?; + for ou in &ous { + for (account, _) in accounts_under(org, ou) { + ou_of_account.entry(account).or_insert_with(|| ou.clone()); + } + } + } + + let mut accounts = self.state.write(); + let set_id = accounts + .get(&admin) + .and_then(|s| active_key(s, &name)) + .ok_or_else(|| stack_set_not_found(&name))?; + Self::refresh_stack_set(&mut accounts, &admin, &set_id); + let set = accounts + .get(&admin) + .and_then(|s| s.stack_sets.get(&set_id)) + .cloned() + .ok_or_else(|| stack_set_not_found(&name))?; + Self::check_can_start_operation(&set, &op_id)?; + let service_managed = set.permission_model == "SERVICE_MANAGED"; + if service_managed && ous.is_empty() { + return Err(validation( + "OrganizationalUnitIds is required when importing into a SERVICE_MANAGED stack set", + )); + } + if !service_managed && !ous.is_empty() { + return Err(validation( + "OrganizationalUnitIds are only supported for stack sets with SERVICE_MANAGED permission model", + )); + } + + let set_template = + fakecloud_core::cfn_template::parse_template_body(&set.template_body).ok(); + let mut op = Self::new_operation(&set, &op_id, "CREATE", preferences, None, None); + let mut new_instances = Vec::new(); + for (stack_id, account, region) in located { + let stack = accounts + .get(&account) + .and_then(|s| { + s.stacks + .values() + .find(|st| st.stack_id == stack_id && st.status != "DELETE_COMPLETE") + }) + .cloned() + .ok_or_else(|| { + aws_err( + StatusCode::NOT_FOUND, + "StackNotFoundException", + format!("Stack with id {stack_id} does not exist"), + ) + })?; + let ou = if service_managed { + let ou = ou_of_account.get(&account).cloned().ok_or_else(|| { + validation(format!( + "Account {account} of stack {stack_id} is not in the specified OrganizationalUnitIds" + )) + })?; + Some(ou) + } else { + None + }; + let already_managed = accounts.iter().any(|(_, s)| { + s.stack_sets.values().any(|other| { + other.status == "ACTIVE" + && other + .instances + .iter() + .any(|i| i.stack_id.as_deref() == Some(stack_id.as_str())) + }) + }); + let duplicate = set + .instances + .iter() + .chain(new_instances.iter()) + .any(|i: &StackInstance| i.account == account && i.region == region); + let template_matches = match ( + &set_template, + fakecloud_core::cfn_template::parse_template_body(&stack.template), + ) { + (Some(a), Ok(b)) => *a == b, + _ => set.template_body.trim() == stack.template.trim(), + }; + let (result_status, reason, instance_status) = if already_managed { + ( + "FAILED", + Some(format!( + "Stack {stack_id} is already managed by a stack set" + )), + None, + ) + } else if duplicate { + ( + "FAILED", + Some(format!( + "Stack instance for account {account} and region {region} already exists" + )), + None, + ) + } else if !template_matches { + ( + "FAILED", + Some("The stack's template does not match the stack set template".to_string()), + Some(("OUTDATED", "FAILED_IMPORT")), + ) + } else { + ("SUCCEEDED", None, Some(("CURRENT", "SUCCEEDED"))) + }; + op.results.push(OperationResult { + account: account.clone(), + region: region.clone(), + status: result_status.to_string(), + status_reason: reason.clone(), + organizational_unit_id: ou.clone(), + account_gate_status: None, + account_gate_reason: None, + }); + if let Some((status, detailed)) = instance_status { + let overrides = stack + .parameters + .iter() + .filter(|(k, v)| !k.starts_with("AWS::") && set.parameters.get(*k) != Some(*v)) + .map(|(k, v)| (k.clone(), v.clone())) + .collect(); + new_instances.push(StackInstance { + account, + region, + stack_id: Some(stack_id), + status: status.to_string(), + detailed_status: detailed.to_string(), + status_reason: reason, + parameter_overrides: overrides, + organizational_unit_id: ou, + drift_status: "NOT_CHECKED".to_string(), + last_drift_check_timestamp: None, + last_operation_id: Some(op_id.clone()), + }); + } + } + op.status = settled_status(&op).to_string(); + op.ended_at = Some(Utc::now()); + let set = accounts + .get_or_create(&admin) + .stack_sets + .get_mut(&set_id) + .ok_or_else(|| stack_set_not_found(&name))?; + set.instances.extend(new_instances); + set.operations.push(op); + Ok(xml_response( + "ImportStacksToStackSet", + el("OperationId", &op_id), + &req.request_id, + )) + } + + fn list_stack_set_auto_deployment_targets( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let set = self.read_stack_set(&admin, &name)?; + let mut by_ou: BTreeMap> = BTreeMap::new(); + if set.permission_model == "SERVICE_MANAGED" { + for instance in &set.instances { + if let Some(ou) = &instance.organizational_unit_id { + by_ou + .entry(ou.clone()) + .or_default() + .insert(instance.region.clone()); + } + } + } + let entries: Vec<(String, BTreeSet)> = by_ou.into_iter().collect(); + let (page, next) = paginate(entries, params)?; + let inner = format!( + "{}{}", + list_el( + "Summaries", + page.iter().map(|(ou, regions)| { + format!( + "{}{}", + el("OrganizationalUnitId", ou), + scalar_list_el("Regions", regions) + ) + }) + ), + next_token_el(next) + ); + Ok(xml_response( + "ListStackSetAutoDeploymentTargets", + inner, + &req.request_id, + )) + } + + // ── Drift ── + + fn detect_stack_set_drift( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let admin = self.stack_set_admin_account(req, params)?; + let preferences = parse_preferences(params)?; + let op_id = params + .get("OperationId") + .cloned() + .unwrap_or_else(|| uuid::Uuid::new_v4().to_string()); + + let set = { + let mut accounts = self.state.write(); + let set_id = accounts + .get(&admin) + .and_then(|s| active_key(s, &name)) + .ok_or_else(|| stack_set_not_found(&name))?; + Self::refresh_stack_set(&mut accounts, &admin, &set_id); + let set = accounts + .get(&admin) + .and_then(|s| s.stack_sets.get(&set_id)) + .cloned() + .ok_or_else(|| stack_set_not_found(&name))?; + Self::check_can_start_operation(&set, &op_id)?; + set + }; + + // Check every instance's stack against the live backing resources. + let now = Utc::now(); + let mut op = Self::new_operation(&set, &op_id, "DETECT_DRIFT", preferences, None, None); + let mut instance_drift: Vec<(String, String, String)> = Vec::new(); + for instance in &set.instances { + let stack = instance.stack_id.as_ref().and_then(|id| { + self.state.read().get(&instance.account).and_then(|s| { + s.stacks + .values() + .find(|st| &st.stack_id == id && st.status != "DELETE_COMPLETE") + .cloned() + }) + }); + let Some(stack) = stack else { + instance_drift.push(( + instance.account.clone(), + instance.region.clone(), + "UNKNOWN".to_string(), + )); + op.results.push(OperationResult { + account: instance.account.clone(), + region: instance.region.clone(), + status: "FAILED".to_string(), + status_reason: Some("Stack instance does not have a stack".to_string()), + organizational_unit_id: instance.organizational_unit_id.clone(), + account_gate_status: None, + account_gate_reason: None, + }); + continue; + }; + let mut drifted = false; + for resource in &stack.resources { + let status = match self.resource_exists(&instance.account, resource) { + Some(true) => "IN_SYNC", + Some(false) => { + drifted = true; + "DELETED" + } + None => "NOT_CHECKED", + }; + op.resource_drifts.push(InstanceResourceDrift { + account: instance.account.clone(), + region: instance.region.clone(), + stack_id: stack.stack_id.clone(), + logical_id: resource.logical_id.clone(), + physical_id: resource.physical_id.clone(), + resource_type: resource.resource_type.clone(), + status: status.to_string(), + timestamp: now, + }); + } + let status = if drifted { "DRIFTED" } else { "IN_SYNC" }; + instance_drift.push(( + instance.account.clone(), + instance.region.clone(), + status.to_string(), + )); + op.results.push(OperationResult { + account: instance.account.clone(), + region: instance.region.clone(), + status: "SUCCEEDED".to_string(), + status_reason: None, + organizational_unit_id: instance.organizational_unit_id.clone(), + account_gate_status: None, + account_gate_reason: None, + }); + } + let drifted = instance_drift + .iter() + .filter(|(_, _, s)| s == "DRIFTED") + .count(); + let in_sync = instance_drift + .iter() + .filter(|(_, _, s)| s == "IN_SYNC") + .count(); + let failed = instance_drift + .iter() + .filter(|(_, _, s)| s == "UNKNOWN") + .count(); + let details = DriftDetectionDetails { + drift_status: if set.instances.is_empty() { + "NOT_CHECKED".to_string() + } else if drifted > 0 { + "DRIFTED".to_string() + } else { + "IN_SYNC".to_string() + }, + detection_status: if failed == 0 { + "COMPLETED".to_string() + } else if failed == instance_drift.len() { + "FAILED".to_string() + } else { + "PARTIAL_SUCCESS".to_string() + }, + last_drift_check_timestamp: now, + total: set.instances.len(), + drifted, + in_sync, + failed, + }; + op.drift = Some(details.clone()); + op.status = settled_status(&op).to_string(); + op.ended_at = Some(Utc::now()); + + let mut accounts = self.state.write(); + if let Some(stored) = accounts + .get_mut(&admin) + .and_then(|s| s.stack_sets.get_mut(&set.stack_set_id)) + { + for instance in &mut stored.instances { + if let Some((_, _, status)) = instance_drift + .iter() + .find(|(a, r, _)| *a == instance.account && *r == instance.region) + { + instance.drift_status = status.clone(); + instance.last_drift_check_timestamp = Some(now); + } + } + stored.drift = Some(details); + stored.operations.push(op); + } + Ok(xml_response( + "DetectStackSetDrift", + el("OperationId", &op_id), + &req.request_id, + )) + } + + fn list_stack_instance_resource_drifts( + &self, + req: &AwsRequest, + params: &BTreeMap, + ) -> Result { + let name = required(params, "StackSetName")?; + let account = required(params, "StackInstanceAccount")?; + let region = required(params, "StackInstanceRegion")?; + let op_id = required(params, "OperationId")?; + let admin = self.stack_set_admin_account(req, params)?; + let set = self.read_stack_set(&admin, &name)?; + let op = set + .operations + .iter() + .find(|o| o.operation_id == op_id) + .ok_or_else(|| operation_not_found(&op_id))?; + if !set + .instances + .iter() + .any(|i| i.account == account && i.region == region) + { + return Err(instance_not_found(&name, &account, ®ion)); + } + let statuses = member_list(params, "StackInstanceResourceDriftStatuses"); + let drifts: Vec<&InstanceResourceDrift> = op + .resource_drifts + .iter() + .filter(|d| d.account == account && d.region == region) + .filter(|d| statuses.is_empty() || statuses.contains(&d.status)) + .collect(); + let (page, next) = paginate(drifts, params)?; + let inner = format!( + "{}{}", + list_el( + "Summaries", + page.into_iter().map(|d| { + format!( + "{}{}{}{}{}{}", + el("StackId", &d.stack_id), + el("LogicalResourceId", &d.logical_id), + el("PhysicalResourceId", &d.physical_id), + el("ResourceType", &d.resource_type), + el("StackResourceDriftStatus", &d.status), + el("Timestamp", &ts(&d.timestamp)), + ) + }) + ), + next_token_el(next) + ); + Ok(xml_response( + "ListStackInstanceResourceDrifts", + inner, + &req.request_id, + )) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::extras::tests::{deps, req}; + use crate::service::CloudFormationDeps; + use crate::state::SharedCloudFormationState; + use fakecloud_core::delivery::{DeliveryBus, LambdaDelivery}; + use fakecloud_core::service::AwsService; + use parking_lot::RwLock; + use std::sync::Arc; + + const ADMIN: &str = "000000000000"; + const ACCT_B: &str = "111111111111"; + const ACCT_C: &str = "222222222222"; + // Unnamed resources are named after their logical id, so the queue is named + // after its stack to keep instances in one account apart. + const QUEUE_TEMPLATE: &str = "Parameters:\n Env:\n Type: String\n Default: dev\nResources:\n Q:\n Type: AWS::SQS::Queue\n Properties:\n QueueName:\n Fn::Sub: \"${AWS::StackName}-q\"\n"; + const TOPIC_TEMPLATE: &str = "Parameters:\n Env:\n Type: String\n Default: dev\nResources:\n T:\n Type: AWS::SNS::Topic\n"; + + fn service_with(deps: CloudFormationDeps) -> CloudFormationService { + let state: SharedCloudFormationState = Arc::new(RwLock::new(MultiAccountState::< + CloudFormationState, + >::new( + ADMIN, "us-east-1", "" + ))); + CloudFormationService::new(state, deps) + } + + fn service() -> CloudFormationService { + service_with(deps()) + } + + async fn call_as( + svc: &CloudFormationService, + account: &str, + action: &str, + params: &[(&str, &str)], + ) -> Result { + let mut request = req(action, params); + request.account_id = account.to_string(); + let resp = svc.handle(request).await?; + Ok(String::from_utf8(resp.body.expect_bytes().to_vec()).expect("utf8")) + } + + async fn call( + svc: &CloudFormationService, + action: &str, + params: &[(&str, &str)], + ) -> Result { + call_as(svc, ADMIN, action, params).await + } + + async fn ok(svc: &CloudFormationService, action: &str, params: &[(&str, &str)]) -> String { + match call(svc, action, params).await { + Ok(xml) => xml, + Err(e) => panic!("{action} failed: {} {}", e.code(), e.message()), + } + } + + async fn err( + svc: &CloudFormationService, + action: &str, + params: &[(&str, &str)], + ) -> AwsServiceError { + match call(svc, action, params).await { + Ok(xml) => panic!("{action} should fail, got {xml}"), + Err(e) => e, + } + } + + fn tag(xml: &str, name: &str) -> String { + let open = format!("<{name}>"); + xml.split(&open) + .nth(1) + .and_then(|rest| rest.split(&format!("")).next()) + .unwrap_or_else(|| panic!("no <{name}> in {xml}")) + .to_string() + } + + fn stored_set(svc: &CloudFormationService, name: &str) -> StackSet { + svc.state + .read() + .get(ADMIN) + .and_then(|s| find_active(s, name)) + .cloned() + .expect("stack set") + } + + fn stack_of(svc: &CloudFormationService, account: &str, stack_id: &str) -> crate::state::Stack { + svc.state + .read() + .get(account) + .and_then(|s| { + s.stacks + .values() + .find(|st| st.stack_id == stack_id) + .cloned() + }) + .expect("instance stack") + } + + fn queue_count(svc: &CloudFormationService, account: &str) -> usize { + svc.deps + .sqs + .read() + .get(account) + .map_or(0, |s| s.queues.len()) + } + + async fn create_set(svc: &CloudFormationService, name: &str, template: &str) { + ok( + svc, + "CreateStackSet", + &[("StackSetName", name), ("TemplateBody", template)], + ) + .await; + } + + #[tokio::test] + async fn stack_instances_provision_real_stacks_per_account_and_region() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + let xml = ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Accounts.member.2", ACCT_C), + ("Regions.member.1", "us-east-1"), + ("Regions.member.2", "eu-west-1"), + ("ParameterOverrides.member.1.ParameterKey", "Env"), + ("ParameterOverrides.member.1.ParameterValue", "prod"), + ], + ) + .await; + let op_id = tag(&xml, "OperationId"); + + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", &op_id)], + ) + .await; + assert_eq!(tag(&op, "Status"), "SUCCEEDED", "{op}"); + assert_eq!(tag(&op, "Action"), "CREATE"); + assert!(op.contains(""), "{op}"); + + let set = stored_set(&svc, "app"); + assert_eq!(set.instances.len(), 4); + for instance in &set.instances { + assert_eq!(instance.status, "CURRENT"); + assert_eq!(instance.detailed_status, "SUCCEEDED"); + let stack_id = instance.stack_id.as_deref().expect("stack id"); + assert!( + stack_id.starts_with(&format!( + "arn:aws:cloudformation:{}:{}:stack/StackSet-app-", + instance.region, instance.account + )), + "{stack_id}" + ); + let stack = stack_of(&svc, &instance.account, stack_id); + assert_eq!(stack.status, "CREATE_COMPLETE"); + assert_eq!( + stack.parameters.get("Env").map(String::as_str), + Some("prod") + ); + assert_eq!(stack.resources.len(), 1); + } + // Each account got a queue per region, in that account. + assert_eq!(queue_count(&svc, ACCT_B), 2); + assert_eq!(queue_count(&svc, ACCT_C), 2); + assert_eq!(queue_count(&svc, ADMIN), 0); + + let described = ok( + &svc, + "DescribeStackInstance", + &[ + ("StackSetName", "app"), + ("StackInstanceAccount", ACCT_B), + ("StackInstanceRegion", "eu-west-1"), + ], + ) + .await; + assert_eq!(tag(&described, "Status"), "CURRENT"); + assert_eq!(tag(&described, "DetailedStatus"), "SUCCEEDED"); + assert_eq!(tag(&described, "LastOperationId"), op_id); + assert!( + described.contains("prod"), + "{described}" + ); + + let listed = ok( + &svc, + "ListStackInstances", + &[("StackSetName", "app"), ("StackInstanceAccount", ACCT_C)], + ) + .await; + assert_eq!(listed.matches("").count(), 2, "{listed}"); + + let results = ok( + &svc, + "ListStackSetOperationResults", + &[("StackSetName", "app"), ("OperationId", &op_id)], + ) + .await; + assert_eq!( + results.matches("SUCCEEDED").count(), + 4, + "{results}" + ); + assert!( + results.contains("SKIPPED"), + "{results}" + ); + + let summary = ok(&svc, "DescribeStackSet", &[("StackSetName", "app")]).await; + assert!(summary.contains("eu-west-1"), "{summary}"); + assert!(summary.contains("SELF_MANAGED")); + assert!(summary.contains(&format!( + "arn:aws:iam::{ADMIN}:role/AWSCloudFormationStackSetAdministrationRole" + ))); + } + + #[tokio::test] + async fn updating_the_stack_set_redeploys_every_instance() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("Regions.member.2", "us-west-2"), + ], + ) + .await; + let xml = ok( + &svc, + "UpdateStackSet", + &[("StackSetName", "app"), ("TemplateBody", TOPIC_TEMPLATE)], + ) + .await; + let op_id = tag(&xml, "OperationId"); + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", &op_id)], + ) + .await; + assert_eq!(tag(&op, "Status"), "SUCCEEDED", "{op}"); + assert_eq!(tag(&op, "Action"), "UPDATE"); + + let set = stored_set(&svc, "app"); + for instance in &set.instances { + let stack = stack_of(&svc, ACCT_B, instance.stack_id.as_deref().unwrap()); + assert_eq!(stack.status, "UPDATE_COMPLETE"); + assert_eq!(stack.resources[0].resource_type, "AWS::SNS::Topic"); + assert_eq!(instance.last_operation_id.as_deref(), Some(op_id.as_str())); + } + // The queues the old template made are gone. + assert_eq!(queue_count(&svc, ACCT_B), 0); + + let ops = ok(&svc, "ListStackSetOperations", &[("StackSetName", "app")]).await; + assert_eq!(ops.matches("").count(), 2, "{ops}"); + // Most recent first. + assert_eq!(tag(&ops, "Action"), "UPDATE"); + } + + #[tokio::test] + async fn a_partial_stack_set_update_leaves_the_rest_outdated() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("Regions.member.2", "us-west-2"), + ], + ) + .await; + ok( + &svc, + "UpdateStackSet", + &[ + ("StackSetName", "app"), + ("TemplateBody", TOPIC_TEMPLATE), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-west-2"), + ], + ) + .await; + let set = stored_set(&svc, "app"); + let east = set + .instances + .iter() + .find(|i| i.region == "us-east-1") + .unwrap(); + let west = set + .instances + .iter() + .find(|i| i.region == "us-west-2") + .unwrap(); + assert_eq!(west.status, "CURRENT"); + assert_eq!(east.status, "OUTDATED"); + assert_eq!( + stack_of(&svc, ACCT_B, east.stack_id.as_deref().unwrap()).resources[0].resource_type, + "AWS::SQS::Queue" + ); + } + + #[tokio::test] + async fn update_stack_instances_applies_and_keeps_overrides() { + let template = "Parameters:\n Env:\n Type: String\n Default: dev\n Size:\n Type: String\n Default: s\nResources:\n Q:\n Type: AWS::SQS::Queue\n"; + let svc = service(); + create_set(&svc, "app", template).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("ParameterOverrides.member.1.ParameterKey", "Env"), + ("ParameterOverrides.member.1.ParameterValue", "prod"), + ], + ) + .await; + ok( + &svc, + "UpdateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("ParameterOverrides.member.1.ParameterKey", "Env"), + ("ParameterOverrides.member.1.UsePreviousValue", "true"), + ("ParameterOverrides.member.2.ParameterKey", "Size"), + ("ParameterOverrides.member.2.ParameterValue", "xl"), + ], + ) + .await; + let set = stored_set(&svc, "app"); + let instance = &set.instances[0]; + assert_eq!( + instance.parameter_overrides.get("Env").map(String::as_str), + Some("prod") + ); + assert_eq!( + instance.parameter_overrides.get("Size").map(String::as_str), + Some("xl") + ); + let stack = stack_of(&svc, ACCT_B, instance.stack_id.as_deref().unwrap()); + assert_eq!(stack.parameters.get("Size").map(String::as_str), Some("xl")); + assert_eq!( + stack.parameters.get("Env").map(String::as_str), + Some("prod") + ); + + // Leaving a parameter out of the list reverts it to the stack set's value. + ok( + &svc, + "UpdateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("ParameterOverrides.member.1.ParameterKey", "Size"), + ("ParameterOverrides.member.1.UsePreviousValue", "true"), + ], + ) + .await; + let set = stored_set(&svc, "app"); + let stack = stack_of(&svc, ACCT_B, set.instances[0].stack_id.as_deref().unwrap()); + assert_eq!(stack.parameters.get("Env").map(String::as_str), Some("dev")); + assert_eq!(stack.parameters.get("Size").map(String::as_str), Some("xl")); + + // Overriding a parameter the template does not declare is rejected. + let e = err( + &svc, + "UpdateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("ParameterOverrides.member.1.ParameterKey", "Nope"), + ("ParameterOverrides.member.1.ParameterValue", "x"), + ], + ) + .await; + assert_eq!(e.code(), "ValidationError"); + + // An instance that does not exist is reported. + let e = err( + &svc, + "UpdateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_C), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + assert_eq!(e.code(), "StackInstanceNotFoundException"); + } + + #[tokio::test] + async fn deleting_instances_tears_down_or_retains_their_stacks() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("Regions.member.2", "us-west-2"), + ], + ) + .await; + let e = err(&svc, "DeleteStackSet", &[("StackSetName", "app")]).await; + assert_eq!(e.code(), "StackSetNotEmptyException"); + + let set = stored_set(&svc, "app"); + let east_stack = set + .instances + .iter() + .find(|i| i.region == "us-east-1") + .unwrap() + .stack_id + .clone() + .unwrap(); + let west_stack = set + .instances + .iter() + .find(|i| i.region == "us-west-2") + .unwrap() + .stack_id + .clone() + .unwrap(); + + ok( + &svc, + "DeleteStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("RetainStacks", "false"), + ], + ) + .await; + assert_eq!( + stack_of(&svc, ACCT_B, &east_stack).status, + "DELETE_COMPLETE" + ); + assert_eq!(queue_count(&svc, ACCT_B), 1); + + ok( + &svc, + "DeleteStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-west-2"), + ("RetainStacks", "true"), + ], + ) + .await; + // Retained: the stack and its queue outlive the instance. + assert_eq!( + stack_of(&svc, ACCT_B, &west_stack).status, + "CREATE_COMPLETE" + ); + assert_eq!(queue_count(&svc, ACCT_B), 1); + assert!(stored_set(&svc, "app").instances.is_empty()); + + let e = err( + &svc, + "DescribeStackInstance", + &[ + ("StackSetName", "app"), + ("StackInstanceAccount", ACCT_B), + ("StackInstanceRegion", "us-east-1"), + ], + ) + .await; + assert_eq!(e.code(), "StackInstanceNotFoundException"); + + // Empty now, so it deletes; the name is free and the id still resolves. + let id = stored_set(&svc, "app").stack_set_id; + ok(&svc, "DeleteStackSet", &[("StackSetName", "app")]).await; + let e = err(&svc, "DescribeStackSet", &[("StackSetName", "app")]).await; + assert_eq!(e.code(), "StackSetNotFoundException"); + let by_id = ok(&svc, "DescribeStackSet", &[("StackSetName", &id)]).await; + assert_eq!(tag(&by_id, "Status"), "DELETED"); + let deleted = ok(&svc, "ListStackSets", &[("Status", "DELETED")]).await; + assert!(deleted.contains(&id), "{deleted}"); + let active = ok(&svc, "ListStackSets", &[("Status", "ACTIVE")]).await; + assert!(!active.contains(&id), "{active}"); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + } + + #[tokio::test] + async fn a_failing_target_cancels_the_rest_beyond_the_tolerance() { + // A template importing an export that does not exist fails to create. + let broken = "Resources:\n Q:\n Type: AWS::SQS::Queue\n Properties:\n QueueName:\n Fn::ImportValue: missing-export\n"; + let svc = service(); + create_set(&svc, "bad", broken).await; + let xml = ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "bad"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("Regions.member.2", "us-west-2"), + ], + ) + .await; + let op_id = tag(&xml, "OperationId"); + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "bad"), ("OperationId", &op_id)], + ) + .await; + assert_eq!(tag(&op, "Status"), "FAILED", "{op}"); + assert_eq!(tag(&op, "FailedStackInstancesCount"), "1"); + + let results = ok( + &svc, + "ListStackSetOperationResults", + &[ + ("StackSetName", "bad"), + ("OperationId", &op_id), + ("Filters.member.1.Name", "OPERATION_RESULT_STATUS"), + ("Filters.member.1.Values", "CANCELLED"), + ], + ) + .await; + assert!(results.contains("us-west-2"), "{results}"); + assert!(results.contains(TOLERANCE_EXCEEDED), "{results}"); + + let set = stored_set(&svc, "bad"); + let failed = set + .instances + .iter() + .find(|i| i.region == "us-east-1") + .unwrap(); + assert_eq!(failed.detailed_status, "FAILED"); + assert!(failed + .status_reason + .as_deref() + .unwrap() + .contains("missing-export")); + let cancelled = set + .instances + .iter() + .find(|i| i.region == "us-west-2") + .unwrap(); + assert_eq!(cancelled.detailed_status, "CANCELLED"); + + let filtered = ok( + &svc, + "ListStackInstances", + &[ + ("StackSetName", "bad"), + ("Filters.member.1.Name", "DETAILED_STATUS"), + ("Filters.member.1.Values", "FAILED"), + ], + ) + .await; + assert_eq!(filtered.matches("").count(), 1, "{filtered}"); + + // With one failure tolerated per region, both regions are attempted + // and the operation succeeds despite the failures being per-region 1. + let xml = ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "bad"), + ("Accounts.member.1", ACCT_C), + ("Regions.member.1", "us-east-1"), + ("Regions.member.2", "us-west-2"), + ("OperationPreferences.FailureToleranceCount", "1"), + ], + ) + .await; + let op = ok( + &svc, + "DescribeStackSetOperation", + &[ + ("StackSetName", "bad"), + ("OperationId", &tag(&xml, "OperationId")), + ], + ) + .await; + assert_eq!(tag(&op, "Status"), "SUCCEEDED", "{op}"); + assert_eq!(tag(&op, "FailedStackInstancesCount"), "2"); + } + + #[tokio::test] + async fn operations_report_modeled_errors() { + let svc = service(); + let e = err( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "nope"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + assert_eq!(e.code(), "StackSetNotFoundException"); + assert_eq!(e.status(), StatusCode::NOT_FOUND); + + create_set(&svc, "app", QUEUE_TEMPLATE).await; + let e = err( + &svc, + "CreateStackSet", + &[("StackSetName", "app"), ("TemplateBody", QUEUE_TEMPLATE)], + ) + .await; + assert_eq!(e.code(), "NameAlreadyExistsException"); + + let e = err( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "never")], + ) + .await; + assert_eq!(e.code(), "OperationNotFoundException"); + + let params = [ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("OperationId", "op-1"), + ]; + ok(&svc, "CreateStackInstances", ¶ms).await; + let e = err(&svc, "CreateStackInstances", ¶ms).await; + assert_eq!(e.code(), "OperationIdAlreadyExistsException"); + + let e = err( + &svc, + "StopStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "op-1")], + ) + .await; + assert_eq!(e.code(), "InvalidOperationException"); + + let e = err( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", "not-an-account"), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + assert_eq!(e.code(), "ValidationError"); + let e = err( + &svc, + "CreateStackInstances", + &[("StackSetName", "app"), ("Accounts.member.1", ACCT_B)], + ) + .await; + assert_eq!(e.code(), "ValidationError"); + } + + #[tokio::test] + async fn a_running_operation_blocks_others_and_can_be_stopped() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + { + let mut accounts = svc.state.write(); + let set = accounts + .get_or_create(ADMIN) + .stack_sets + .values_mut() + .next() + .unwrap(); + let mut op = CloudFormationService::new_operation( + &set.clone(), + "running-op", + "CREATE", + OperationPreferences::default(), + None, + None, + ); + op.results.push(OperationResult { + account: ACCT_B.to_string(), + region: "us-east-1".to_string(), + status: "PENDING".to_string(), + status_reason: None, + organizational_unit_id: None, + account_gate_status: None, + account_gate_reason: None, + }); + set.operations.push(op); + } + let e = err( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + assert_eq!(e.code(), "OperationInProgressException"); + let e = err(&svc, "DeleteStackSet", &[("StackSetName", "app")]).await; + assert_eq!(e.code(), "OperationInProgressException"); + + ok( + &svc, + "StopStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "running-op")], + ) + .await; + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "running-op")], + ) + .await; + assert_eq!(tag(&op, "Status"), "STOPPED", "{op}"); + let results = ok( + &svc, + "ListStackSetOperationResults", + &[("StackSetName", "app"), ("OperationId", "running-op")], + ) + .await; + assert_eq!(tag(&results, "Status"), "CANCELLED", "{results}"); + } + + #[tokio::test] + async fn an_asynchronously_provisioning_instance_settles_on_read() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("OperationId", "op-async"), + ], + ) + .await; + // Rewind to the moment the stack was still provisioning. + let stack_id = { + let mut accounts = svc.state.write(); + let set = accounts + .get_or_create(ADMIN) + .stack_sets + .values_mut() + .next() + .unwrap(); + set.instances[0].detailed_status = "RUNNING".to_string(); + set.instances[0].status = "OUTDATED".to_string(); + let op = set.operations.last_mut().unwrap(); + op.status = "RUNNING".to_string(); + op.ended_at = None; + op.results[0].status = "RUNNING".to_string(); + let stack_id = set.instances[0].stack_id.clone().unwrap(); + let stack = accounts + .get_or_create(ACCT_B) + .stacks + .values_mut() + .find(|s| s.stack_id == stack_id) + .unwrap(); + stack.status = "CREATE_IN_PROGRESS".to_string(); + stack_id + }; + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "op-async")], + ) + .await; + assert_eq!(tag(&op, "Status"), "RUNNING", "{op}"); + + // The stack finishes: the next read reflects it. + svc.state + .write() + .get_or_create(ACCT_B) + .stacks + .values_mut() + .find(|s| s.stack_id == stack_id) + .unwrap() + .status = "CREATE_COMPLETE".to_string(); + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "op-async")], + ) + .await; + assert_eq!(tag(&op, "Status"), "SUCCEEDED", "{op}"); + let instance = ok( + &svc, + "DescribeStackInstance", + &[ + ("StackSetName", "app"), + ("StackInstanceAccount", ACCT_B), + ("StackInstanceRegion", "us-east-1"), + ], + ) + .await; + assert_eq!(tag(&instance, "Status"), "CURRENT"); + } + + fn seed_org(svc: &CloudFormationService) -> (String, String) { + let mut org = fakecloud_organizations::OrganizationState::bootstrap(ADMIN); + let root = org.root_id.clone(); + let parent = org.create_ou(&root, "workloads").unwrap(); + let child = org.create_ou(&parent.id, "prod").unwrap(); + for (account, dest) in [(ACCT_B, &parent.id), (ACCT_C, &child.id)] { + org.enroll_account_if_missing(account); + org.move_account(account, &root, dest).unwrap(); + } + *svc.deps.organizations.write() = Some(org); + (parent.id, child.id) + } + + #[tokio::test] + async fn service_managed_stack_sets_deploy_to_organizational_units() { + let svc = service(); + let (workloads, prod) = seed_org(&svc); + let create = [ + ("StackSetName", "org"), + ("TemplateBody", QUEUE_TEMPLATE), + ("PermissionModel", "SERVICE_MANAGED"), + ("AutoDeployment.Enabled", "true"), + ("AutoDeployment.RetainStacksOnAccountRemoval", "false"), + ]; + // Trusted access has to be activated first. + let e = err(&svc, "CreateStackSet", &create).await; + assert_eq!(e.code(), "ValidationError"); + ok(&svc, "ActivateOrganizationsAccess", &[]).await; + ok(&svc, "CreateStackSet", &create).await; + + // Top-level accounts are not a valid target for this model. + let e = err( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "org"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + assert_eq!(e.code(), "ValidationError"); + + // The OU covers its nested OU; the management account is never a target. + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "org"), + ( + "DeploymentTargets.OrganizationalUnitIds.member.1", + &workloads, + ), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + let set = stored_set(&svc, "org"); + let mut accounts: Vec<&str> = set.instances.iter().map(|i| i.account.as_str()).collect(); + accounts.sort(); + assert_eq!(accounts, [ACCT_B, ACCT_C]); + assert!(set + .instances + .iter() + .all(|i| i.organizational_unit_id.as_deref() == Some(workloads.as_str()))); + assert_eq!(queue_count(&svc, ACCT_C), 1); + + let targets = ok( + &svc, + "ListStackSetAutoDeploymentTargets", + &[("StackSetName", "org")], + ) + .await; + assert_eq!(tag(&targets, "OrganizationalUnitId"), workloads); + assert!(targets.contains("us-east-1"), "{targets}"); + let described = ok(&svc, "DescribeStackSet", &[("StackSetName", "org")]).await; + assert!( + described.contains(&format!("{workloads}")), + "{described}" + ); + assert!(described.contains("true"), "{described}"); + + // DIFFERENCE removes the listed account from the OU's instances. + ok( + &svc, + "DeleteStackInstances", + &[ + ("StackSetName", "org"), + ( + "DeploymentTargets.OrganizationalUnitIds.member.1", + &workloads, + ), + ("DeploymentTargets.Accounts.member.1", ACCT_B), + ("DeploymentTargets.AccountFilterType", "DIFFERENCE"), + ("Regions.member.1", "us-east-1"), + ("RetainStacks", "false"), + ], + ) + .await; + let set = stored_set(&svc, "org"); + assert_eq!(set.instances.len(), 1); + assert_eq!(set.instances[0].account, ACCT_B); + assert_eq!(queue_count(&svc, ACCT_C), 0); + + // A nested OU targets only its own accounts. + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "org"), + ("DeploymentTargets.OrganizationalUnitIds.member.1", &prod), + ("Regions.member.1", "eu-west-1"), + ], + ) + .await; + let set = stored_set(&svc, "org"); + let eu: Vec<&StackInstance> = set + .instances + .iter() + .filter(|i| i.region == "eu-west-1") + .collect(); + assert_eq!(eu.len(), 1); + assert_eq!(eu[0].account, ACCT_C); + + let e = err( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "org"), + ( + "DeploymentTargets.OrganizationalUnitIds.member.1", + "ou-none-00000000", + ), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + assert_eq!(e.code(), "ValidationError"); + } + + #[tokio::test] + async fn a_delegated_administrator_acts_on_the_management_accounts_stack_sets() { + let svc = service(); + seed_org(&svc); + { + let mut orgs = svc.deps.organizations.write(); + let org = orgs.as_mut().unwrap(); + org.enable_aws_service_access(STACKSETS_PRINCIPAL); + org.register_delegated_administrator(ACCT_B, STACKSETS_PRINCIPAL) + .unwrap(); + } + let params = [ + ("StackSetName", "delegated"), + ("TemplateBody", QUEUE_TEMPLATE), + ("PermissionModel", "SERVICE_MANAGED"), + ("CallAs", "DELEGATED_ADMIN"), + ]; + call_as(&svc, ACCT_B, "CreateStackSet", ¶ms) + .await + .unwrap(); + // Stored with the management account, visible to it too. + assert_eq!( + stored_set(&svc, "delegated").permission_model, + "SERVICE_MANAGED" + ); + let listed = call_as( + &svc, + ACCT_B, + "ListStackSets", + &[("CallAs", "DELEGATED_ADMIN")], + ) + .await + .unwrap(); + assert!( + listed.contains("delegated"), + "{listed}" + ); + + // An account that is not registered cannot. + let e = call_as( + &svc, + ACCT_C, + "ListStackSets", + &[("CallAs", "DELEGATED_ADMIN")], + ) + .await + .unwrap_err(); + assert_eq!(e.code(), "ValidationError"); + } + + #[tokio::test] + async fn existing_stacks_import_into_a_stack_set() { + let svc = service(); + let xml = call_as( + &svc, + ACCT_B, + "CreateStack", + &[("StackName", "legacy"), ("TemplateBody", QUEUE_TEMPLATE)], + ) + .await + .unwrap(); + let stack_id = tag(&xml, "StackId"); + create_set(&svc, "adopt", QUEUE_TEMPLATE).await; + + let xml = ok( + &svc, + "ImportStacksToStackSet", + &[("StackSetName", "adopt"), ("StackIds.member.1", &stack_id)], + ) + .await; + let op_id = tag(&xml, "OperationId"); + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "adopt"), ("OperationId", &op_id)], + ) + .await; + assert_eq!(tag(&op, "Status"), "SUCCEEDED", "{op}"); + let set = stored_set(&svc, "adopt"); + assert_eq!(set.instances.len(), 1); + assert_eq!( + set.instances[0].stack_id.as_deref(), + Some(stack_id.as_str()) + ); + assert_eq!(set.instances[0].account, ACCT_B); + assert_eq!(set.instances[0].status, "CURRENT"); + + // A stack can belong to one stack set only. + create_set(&svc, "other", QUEUE_TEMPLATE).await; + let xml = ok( + &svc, + "ImportStacksToStackSet", + &[("StackSetName", "other"), ("StackIds.member.1", &stack_id)], + ) + .await; + let op = ok( + &svc, + "DescribeStackSetOperation", + &[ + ("StackSetName", "other"), + ("OperationId", &tag(&xml, "OperationId")), + ], + ) + .await; + assert_eq!(tag(&op, "Status"), "FAILED", "{op}"); + assert!(stored_set(&svc, "other").instances.is_empty()); + + let e = err( + &svc, + "ImportStacksToStackSet", + &[ + ("StackSetName", "other"), + ( + "StackIds.member.1", + "arn:aws:cloudformation:us-east-1:111111111111:stack/ghost/1", + ), + ], + ) + .await; + assert_eq!(e.code(), "StackNotFoundException"); + + // A stack set can also be created from a stack's template. + let xml = call_as( + &svc, + ACCT_B, + "CreateStackSet", + &[("StackSetName", "from-stack"), ("StackId", &stack_id)], + ) + .await + .unwrap(); + assert!(xml.contains("from-stack:"), "{xml}"); + let described = call_as( + &svc, + ACCT_B, + "DescribeStackSet", + &[("StackSetName", "from-stack")], + ) + .await + .unwrap(); + assert!(described.contains("AWS::SQS::Queue"), "{described}"); + } + + #[tokio::test] + async fn stack_set_drift_detection_checks_each_instance() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("Regions.member.2", "us-west-2"), + ], + ) + .await; + // Delete one instance's queue behind CloudFormation's back. + let set = stored_set(&svc, "app"); + let east = set + .instances + .iter() + .find(|i| i.region == "us-east-1") + .unwrap(); + let queue_url = stack_of(&svc, ACCT_B, east.stack_id.as_deref().unwrap()).resources[0] + .physical_id + .clone(); + svc.deps + .sqs + .write() + .get_or_create(ACCT_B) + .queues + .remove(&queue_url) + .expect("queue existed"); + + let xml = ok(&svc, "DetectStackSetDrift", &[("StackSetName", "app")]).await; + let op_id = tag(&xml, "OperationId"); + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", &op_id)], + ) + .await; + assert_eq!(tag(&op, "Action"), "DETECT_DRIFT"); + assert_eq!(tag(&op, "DriftStatus"), "DRIFTED", "{op}"); + assert_eq!(tag(&op, "DriftedStackInstancesCount"), "1"); + assert_eq!(tag(&op, "InSyncStackInstancesCount"), "1"); + + let drifts = ok( + &svc, + "ListStackInstanceResourceDrifts", + &[ + ("StackSetName", "app"), + ("StackInstanceAccount", ACCT_B), + ("StackInstanceRegion", "us-east-1"), + ("OperationId", &op_id), + ("StackInstanceResourceDriftStatuses.member.1", "DELETED"), + ], + ) + .await; + assert_eq!( + tag(&drifts, "StackResourceDriftStatus"), + "DELETED", + "{drifts}" + ); + assert_eq!(tag(&drifts, "PhysicalResourceId"), queue_url); + + let in_sync = ok( + &svc, + "ListStackInstances", + &[ + ("StackSetName", "app"), + ("Filters.member.1.Name", "DRIFT_STATUS"), + ("Filters.member.1.Values", "IN_SYNC"), + ], + ) + .await; + assert!(in_sync.contains("us-west-2"), "{in_sync}"); + assert!(!in_sync.contains("us-east-1"), "{in_sync}"); + let summary = ok(&svc, "ListStackSets", &[]).await; + assert!( + summary.contains("DRIFTED"), + "{summary}" + ); + } + + struct Gate(&'static str); + + impl LambdaDelivery for Gate { + fn invoke_lambda( + &self, + _function_arn: &str, + _payload: &str, + ) -> std::pin::Pin, String>> + Send>> + { + let body = format!("{{\"Status\":\"{}\"}}", self.0); + Box::pin(async move { Ok(body.into_bytes()) }) + } + } + + #[tokio::test] + async fn an_account_gate_that_fails_blocks_deployment_to_that_account() { + let mut d = deps(); + d.delivery = Arc::new(DeliveryBus::new().with_lambda(Arc::new(Gate("FAILED")))); + let svc = service_with(d); + let gate_template = "Resources:\n Gate:\n Type: AWS::Lambda::Function\n Properties:\n FunctionName: AWSCloudFormationStackSetAccountGate\n Runtime: python3.12\n Handler: index.handler\n Role: arn:aws:iam::111111111111:role/gate\n Code:\n ZipFile: \"def handler(e, c): return {}\"\n"; + call_as( + &svc, + ACCT_B, + "CreateStack", + &[("StackName", "gate"), ("TemplateBody", gate_template)], + ) + .await + .unwrap(); + assert!(svc + .deps + .lambda + .read() + .get(ACCT_B) + .is_some_and(|s| s.functions.contains_key(ACCOUNT_GATE_FUNCTION))); + + create_set(&svc, "gated", QUEUE_TEMPLATE).await; + let xml = ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "gated"), + ("Accounts.member.1", ACCT_B), + ("Accounts.member.2", ACCT_C), + ("Regions.member.1", "us-east-1"), + ("OperationPreferences.FailureToleranceCount", "1"), + ], + ) + .await; + let results = ok( + &svc, + "ListStackSetOperationResults", + &[ + ("StackSetName", "gated"), + ("OperationId", &tag(&xml, "OperationId")), + ], + ) + .await; + assert!( + results.contains("FAILED"), + "{results}" + ); + assert_eq!(queue_count(&svc, ACCT_B), 0); + // The ungated account deploys. + assert_eq!(queue_count(&svc, ACCT_C), 1); + } +} diff --git a/crates/fakecloud-cloudformation/src/state.rs b/crates/fakecloud-cloudformation/src/state.rs index ce7923991..0ca57b459 100644 --- a/crates/fakecloud-cloudformation/src/state.rs +++ b/crates/fakecloud-cloudformation/src/state.rs @@ -105,6 +105,10 @@ pub struct CloudFormationState { /// blocks deleting a stack whose exports still appear here. #[serde(default)] pub imports: BTreeMap>, + /// Stack sets administered from this account, keyed by `StackSetId`. + /// Deleted stack sets stay here with status `DELETED`. + #[serde(default)] + pub stack_sets: BTreeMap, } impl CloudFormationState { @@ -119,6 +123,7 @@ impl CloudFormationState { orgs_access_enabled: false, exports: BTreeMap::new(), imports: BTreeMap::new(), + stack_sets: BTreeMap::new(), } } @@ -130,6 +135,7 @@ impl CloudFormationState { self.orgs_access_enabled = false; self.exports.clear(); self.imports.clear(); + self.stack_sets.clear(); } } diff --git a/crates/fakecloud-conformance/tests/cloudformation.rs b/crates/fakecloud-conformance/tests/cloudformation.rs index 17630482e..72d847363 100644 --- a/crates/fakecloud-conformance/tests/cloudformation.rs +++ b/crates/fakecloud-conformance/tests/cloudformation.rs @@ -251,6 +251,15 @@ async fn cloudformation_get_template() { const CFN_AUTH: &str = "AWS4-HMAC-SHA256 Credential=test/20240101/us-east-1/cloudformation/aws4_request, SignedHeaders=host, Signature=0"; +/// Text of the first `` element in an XML response. +fn xml_tag(xml: &str, name: &str) -> String { + xml.split(&format!("<{name}>")) + .nth(1) + .and_then(|rest| rest.split(&format!("")).next()) + .unwrap_or_else(|| panic!("no <{name}> in {xml}")) + .to_string() +} + fn pct(s: &str) -> String { let mut out = String::with_capacity(s.len()); for &b in s.as_bytes() { @@ -436,144 +445,147 @@ async fn cloudformation_closure_routes_exist() { .is_success() ); - // Stack sets - assert!( - cfn_post(&server, "CreateStackSet", &[("StackSetName", "ss1")]) + // Stack sets and their instances, as one lifecycle: every operation below + // acts on state an earlier call created. + let stack_set_template = r#"{"Resources":{"Q":{"Type":"AWS::SQS::Queue","Properties":{"QueueName":{"Fn::Sub":"${AWS::StackName}-q"}}}}}"#; + for (action, params) in [ + ( + "CreateStackSet", + vec![ + ("StackSetName", "ss1"), + ("TemplateBody", stack_set_template), + ], + ), + ("DescribeStackSet", vec![("StackSetName", "ss1")]), + ("ListStackSets", vec![]), + ] { + let resp = cfn_post(&server, action, ¶ms).await; + assert!(resp.status().is_success(), "{action}: {}", resp.status()); + } + let instance_target = [ + ("StackSetName", "ss1"), + ("Accounts.member.1", "111111111111"), + ("Regions.member.1", "us-east-1"), + ]; + let create_op = xml_tag( + &cfn_post(&server, "CreateStackInstances", &instance_target) .await - .status() - .is_success() - ); - assert!( - cfn_post(&server, "DescribeStackSet", &[("StackSetName", "ss1")]) + .text() .await - .status() - .is_success() + .unwrap(), + "OperationId", ); - assert!(cfn_post(&server, "ListStackSets", &[]) - .await - .status() - .is_success()); - assert!( - cfn_post(&server, "UpdateStackSet", &[("StackSetName", "ss1")]) - .await - .status() - .is_success() - ); - assert!(cfn_post( + let described = cfn_post( &server, "DescribeStackSetOperation", - &[("StackSetName", "ss1"), ("OperationId", "op1")] - ) - .await - .status() - .is_success()); - assert!(cfn_post( - &server, - "ListStackSetOperations", - &[("StackSetName", "ss1")] - ) - .await - .status() - .is_success()); - assert!(cfn_post( - &server, - "ListStackSetOperationResults", - &[("StackSetName", "ss1"), ("OperationId", "op1")] + &[("StackSetName", "ss1"), ("OperationId", &create_op)], ) .await - .status() - .is_success()); - assert!(cfn_post( - &server, - "ListStackSetAutoDeploymentTargets", - &[("StackSetName", "ss1")] - ) + .text() .await - .status() - .is_success()); - assert!(cfn_post( + .unwrap(); + assert_eq!(xml_tag(&described, "Status"), "SUCCEEDED", "{described}"); + let instance = cfn_post( &server, - "StopStackSetOperation", - &[("StackSetName", "ss1"), ("OperationId", "op1")] - ) - .await - .status() - .is_success()); - assert!(cfn_post( - &server, - "ImportStacksToStackSet", - &[("StackSetName", "ss1")] + "DescribeStackInstance", + &[ + ("StackSetName", "ss1"), + ("StackInstanceAccount", "111111111111"), + ("StackInstanceRegion", "us-east-1"), + ], ) .await - .status() - .is_success()); - assert!( - cfn_post(&server, "DeleteStackSet", &[("StackSetName", "ss1")]) + .text() + .await + .unwrap(); + assert_eq!(xml_tag(&instance, "Status"), "CURRENT", "{instance}"); + for (action, params) in [ + ("ListStackInstances", vec![("StackSetName", "ss1")]), + ("ListStackSetOperations", vec![("StackSetName", "ss1")]), + ( + "ListStackSetOperationResults", + vec![("StackSetName", "ss1"), ("OperationId", create_op.as_str())], + ), + ( + "ListStackSetAutoDeploymentTargets", + vec![("StackSetName", "ss1")], + ), + ("UpdateStackSet", vec![("StackSetName", "ss1")]), + ("UpdateStackInstances", instance_target.to_vec()), + ] { + let resp = cfn_post(&server, action, ¶ms).await; + assert!(resp.status().is_success(), "{action}: {}", resp.status()); + } + let drift_op = xml_tag( + &cfn_post(&server, "DetectStackSetDrift", &[("StackSetName", "ss1")]) .await - .status() - .is_success() + .text() + .await + .unwrap(), + "OperationId", ); - - // Stack instances - assert!(cfn_post( - &server, - "CreateStackInstances", - &[("StackSetName", "ss1"), ("Regions.member.1", "us-east-1"),], - ) - .await - .status() - .is_success()); assert!(cfn_post( &server, - "UpdateStackInstances", - &[("StackSetName", "ss1"), ("Regions.member.1", "us-east-1"),], - ) - .await - .status() - .is_success()); - assert!(cfn_post( - &server, - "DeleteStackInstances", + "ListStackInstanceResourceDrifts", &[ ("StackSetName", "ss1"), - ("Regions.member.1", "us-east-1"), - ("RetainStacks", "false"), + ("StackInstanceAccount", "111111111111"), + ("StackInstanceRegion", "us-east-1"), + ("OperationId", &drift_op), ], ) .await .status() .is_success()); + // Every operation has finished, so there is nothing left to stop. + assert_eq!( + cfn_post( + &server, + "StopStackSetOperation", + &[("StackSetName", "ss1"), ("OperationId", &create_op)], + ) + .await + .status() + .as_u16(), + 400 + ); + let mut delete_instances = instance_target.to_vec(); + delete_instances.push(("RetainStacks", "true")); + assert!(cfn_post(&server, "DeleteStackInstances", &delete_instances) + .await + .status() + .is_success()); + // The retained stack imports straight back in as an instance. + let listed = cfn_post(&server, "ListStackInstances", &[("StackSetName", "ss1")]) + .await + .text() + .await + .unwrap(); + assert!(!listed.contains(""), "{listed}"); + let retained_stack_id = xml_tag(&instance, "StackId"); assert!(cfn_post( &server, - "DescribeStackInstance", + "ImportStacksToStackSet", &[ ("StackSetName", "ss1"), - ("StackInstanceAccount", "000000000000"), - ("StackInstanceRegion", "us-east-1"), + ("StackIds.member.1", &retained_stack_id) ], ) .await .status() .is_success()); + let mut delete_again = instance_target.to_vec(); + delete_again.push(("RetainStacks", "false")); + assert!(cfn_post(&server, "DeleteStackInstances", &delete_again) + .await + .status() + .is_success()); assert!( - cfn_post(&server, "ListStackInstances", &[("StackSetName", "ss1")]) + cfn_post(&server, "DeleteStackSet", &[("StackSetName", "ss1")]) .await .status() .is_success() ); - assert!(cfn_post( - &server, - "ListStackInstanceResourceDrifts", - &[ - ("StackSetName", "ss1"), - ("StackInstanceAccount", "000000000000"), - ("StackInstanceRegion", "us-east-1"), - ("OperationId", "op1"), - ], - ) - .await - .status() - .is_success()); // Refactors assert!(cfn_post( @@ -817,12 +829,6 @@ async fn cloudformation_closure_routes_exist() { .await .status() .is_success()); - assert!( - cfn_post(&server, "DetectStackSetDrift", &[("StackSetName", "ss1")]) - .await - .status() - .is_success() - ); assert!(cfn_post( &server, "DescribeStackDriftDetectionStatus", diff --git a/crates/fakecloud-e2e/tests/cloudformation_stack_sets.rs b/crates/fakecloud-e2e/tests/cloudformation_stack_sets.rs new file mode 100644 index 000000000..381034c4e --- /dev/null +++ b/crates/fakecloud-e2e/tests/cloudformation_stack_sets.rs @@ -0,0 +1,325 @@ +//! End-to-end tests for CloudFormation StackSets: stack instances are real +//! stacks, provisioned into their target account and region. + +mod helpers; + +use aws_credential_types::Credentials; +use aws_sdk_cloudformation::error::ProvideErrorMetadata; +use aws_sdk_cloudformation::types::{Parameter, StackInstanceStatus, StackSetOperationStatus}; +use aws_sdk_sqs::types::QueueAttributeName; +use helpers::TestServer; + +const DEFAULT_ACCOUNT: &str = "123456789012"; +const MEMBER_ACCOUNT: &str = "222222222222"; + +const TEMPLATE: &str = r#"{ + "Parameters": { + "Timeout": { "Type": "String", "Default": "30" } + }, + "Resources": { + "Queue": { + "Type": "AWS::SQS::Queue", + "Properties": { + "QueueName": { "Fn::Sub": "${AWS::StackName}-queue" }, + "VisibilityTimeout": { "Ref": "Timeout" } + } + } + } +}"#; + +async fn member_sqs(server: &TestServer) -> aws_sdk_sqs::Client { + let (akid, secret) = server.create_admin(MEMBER_ACCOUNT, "stackset-member").await; + let config = aws_config::defaults(aws_config::BehaviorVersion::latest()) + .endpoint_url(server.endpoint()) + .region(aws_config::Region::new("us-east-1")) + .credentials_provider(Credentials::new(akid, secret, None, None, "member")) + .load() + .await; + aws_sdk_sqs::Client::new(&config) +} + +async fn stack_set_queues(sqs: &aws_sdk_sqs::Client) -> Vec { + sqs.list_queues() + .send() + .await + .unwrap() + .queue_urls() + .iter() + .filter(|url| url.contains("StackSet-regional-")) + .cloned() + .collect() +} + +async fn operation_status( + cfn: &aws_sdk_cloudformation::Client, + operation_id: &str, +) -> StackSetOperationStatus { + cfn.describe_stack_set_operation() + .stack_set_name("regional") + .operation_id(operation_id) + .send() + .await + .unwrap() + .stack_set_operation() + .and_then(|op| op.status()) + .cloned() + .expect("operation status") +} + +#[tokio::test] +async fn stack_set_instances_provision_stacks_across_accounts_and_regions() { + let server = TestServer::start().await; + let cfn = server.cloudformation_client().await; + let default_sqs = server.sqs_client().await; + let member_sqs = member_sqs(&server).await; + + let stack_set_id = cfn + .create_stack_set() + .stack_set_name("regional") + .template_body(TEMPLATE) + .send() + .await + .unwrap() + .stack_set_id() + .expect("stack set id") + .to_string(); + assert!(stack_set_id.starts_with("regional:"), "{stack_set_id}"); + + let create = cfn + .create_stack_instances() + .stack_set_name("regional") + .accounts(DEFAULT_ACCOUNT) + .accounts(MEMBER_ACCOUNT) + .regions("us-east-1") + .regions("eu-west-1") + .send() + .await + .unwrap(); + let create_op = create.operation_id().expect("operation id"); + assert_eq!( + operation_status(&cfn, create_op).await, + StackSetOperationStatus::Succeeded + ); + + // One real queue per region, in each target account. + assert_eq!(stack_set_queues(&default_sqs).await.len(), 2); + let member_queues = stack_set_queues(&member_sqs).await; + assert_eq!(member_queues.len(), 2, "{member_queues:?}"); + + let summaries = cfn + .list_stack_instances() + .stack_set_name("regional") + .send() + .await + .unwrap(); + assert_eq!(summaries.summaries().len(), 4); + for summary in summaries.summaries() { + assert_eq!(summary.status(), Some(&StackInstanceStatus::Current)); + let stack_id = summary.stack_id().expect("stack id"); + assert!( + stack_id.starts_with(&format!( + "arn:aws:cloudformation:{}:{}:stack/StackSet-regional-", + summary.region().unwrap(), + summary.account().unwrap() + )), + "{stack_id}" + ); + } + + // The default account's stacks are ordinary stacks it can describe. + let default_instance = cfn + .describe_stack_instance() + .stack_set_name("regional") + .stack_instance_account(DEFAULT_ACCOUNT) + .stack_instance_region("eu-west-1") + .send() + .await + .unwrap(); + let stack_id = default_instance + .stack_instance() + .and_then(|i| i.stack_id()) + .expect("stack id") + .to_string(); + let stacks = cfn + .describe_stacks() + .stack_name(&stack_id) + .send() + .await + .unwrap(); + assert_eq!( + stacks.stacks()[0].stack_status().map(|s| s.as_str()), + Some("CREATE_COMPLETE") + ); + + // Override a parameter for the member account's us-east-1 instance. + let update = cfn + .update_stack_instances() + .stack_set_name("regional") + .accounts(MEMBER_ACCOUNT) + .regions("us-east-1") + .parameter_overrides( + Parameter::builder() + .parameter_key("Timeout") + .parameter_value("120") + .build(), + ) + .send() + .await + .unwrap(); + assert_eq!( + operation_status(&cfn, update.operation_id().unwrap()).await, + StackSetOperationStatus::Succeeded + ); + let member_instance = cfn + .describe_stack_instance() + .stack_set_name("regional") + .stack_instance_account(MEMBER_ACCOUNT) + .stack_instance_region("us-east-1") + .send() + .await + .unwrap(); + let overrides = member_instance + .stack_instance() + .unwrap() + .parameter_overrides(); + assert_eq!(overrides.len(), 1); + assert_eq!(overrides[0].parameter_value(), Some("120")); + let member_stack_name = member_instance + .stack_instance() + .and_then(|i| i.stack_id()) + .and_then(|arn| arn.split('/').nth(1)) + .unwrap() + .to_string(); + let queue_url = member_queues + .iter() + .find(|url| url.contains(&member_stack_name)) + .expect("queue of the overridden instance"); + let attrs = member_sqs + .get_queue_attributes() + .queue_url(queue_url) + .attribute_names(QueueAttributeName::VisibilityTimeout) + .send() + .await + .unwrap(); + assert_eq!( + attrs + .attributes() + .and_then(|a| a.get(&QueueAttributeName::VisibilityTimeout)) + .map(String::as_str), + Some("120") + ); + + // A stack set with instances cannot be deleted. + let err = cfn + .delete_stack_set() + .stack_set_name("regional") + .send() + .await + .unwrap_err(); + assert_eq!(err.code(), Some("StackSetNotEmptyException")); + + let delete = cfn + .delete_stack_instances() + .stack_set_name("regional") + .accounts(DEFAULT_ACCOUNT) + .accounts(MEMBER_ACCOUNT) + .regions("us-east-1") + .regions("eu-west-1") + .retain_stacks(false) + .send() + .await + .unwrap(); + assert_eq!( + operation_status(&cfn, delete.operation_id().unwrap()).await, + StackSetOperationStatus::Succeeded + ); + assert!(stack_set_queues(&default_sqs).await.is_empty()); + assert!(stack_set_queues(&member_sqs).await.is_empty()); + + let operations = cfn + .list_stack_set_operations() + .stack_set_name("regional") + .send() + .await + .unwrap(); + assert_eq!(operations.summaries().len(), 3); + + cfn.delete_stack_set() + .stack_set_name("regional") + .send() + .await + .unwrap(); + let err = cfn + .describe_stack_set() + .stack_set_name("regional") + .send() + .await + .unwrap_err(); + assert_eq!(err.code(), Some("StackSetNotFoundException")); +} + +#[tokio::test] +async fn stack_set_update_redeploys_instances() { + let server = TestServer::start().await; + let cfn = server.cloudformation_client().await; + let sqs = server.sqs_client().await; + + cfn.create_stack_set() + .stack_set_name("regional") + .template_body(TEMPLATE) + .send() + .await + .unwrap(); + cfn.create_stack_instances() + .stack_set_name("regional") + .accounts(DEFAULT_ACCOUNT) + .regions("us-east-1") + .send() + .await + .unwrap(); + + let update = cfn + .update_stack_set() + .stack_set_name("regional") + .use_previous_template(true) + .parameters( + Parameter::builder() + .parameter_key("Timeout") + .parameter_value("45") + .build(), + ) + .send() + .await + .unwrap(); + assert_eq!( + operation_status(&cfn, update.operation_id().unwrap()).await, + StackSetOperationStatus::Succeeded + ); + + let queues = stack_set_queues(&sqs).await; + assert_eq!(queues.len(), 1); + let attrs = sqs + .get_queue_attributes() + .queue_url(&queues[0]) + .attribute_names(QueueAttributeName::VisibilityTimeout) + .send() + .await + .unwrap(); + assert_eq!( + attrs + .attributes() + .and_then(|a| a.get(&QueueAttributeName::VisibilityTimeout)) + .map(String::as_str), + Some("45") + ); + + let described = cfn + .describe_stack_set() + .stack_set_name("regional") + .send() + .await + .unwrap(); + let set = described.stack_set().unwrap(); + assert_eq!(set.parameters()[0].parameter_value(), Some("45")); + assert_eq!(set.regions(), ["us-east-1"]); +} diff --git a/crates/fakecloud-server/src/main.rs b/crates/fakecloud-server/src/main.rs index e808eab0b..c465d35a3 100644 --- a/crates/fakecloud-server/src/main.rs +++ b/crates/fakecloud-server/src/main.rs @@ -1451,7 +1451,8 @@ async fn main() { fakecloud_cloudformation::CLOUDFORMATION_SNAPSHOT_SCHEMA_VERSION, )); } - if let Some(accounts) = snapshot.accounts { + if let Some(mut accounts) = snapshot.accounts { + fakecloud_cloudformation::migrate_legacy_stack_sets(&mut accounts); let account_count = accounts.account_count(); *cloudformation_state.write() = accounts; tracing::info!( @@ -1463,6 +1464,7 @@ async fn main() { let account_id = single_state.account_id.clone(); let mut mas = cloudformation_state.write(); *mas.get_or_create(&account_id) = single_state; + fakecloud_cloudformation::migrate_legacy_stack_sets(&mut mas); tracing::info!( stacks = stack_count, "loaded cloudformation persistence snapshot (migrated from v1)" diff --git a/website/content/docs/services/cloudformation.md b/website/content/docs/services/cloudformation.md index 982ae4566..3bc82a33a 100644 --- a/website/content/docs/services/cloudformation.md +++ b/website/content/docs/services/cloudformation.md @@ -6,7 +6,7 @@ weight = 13 fakecloud implements **90 of 90** CloudFormation operations at 100% Smithy conformance. -**Status: full API.** Stack lifecycle (create/update/delete with real events), nested stacks, SAM transform, change sets, stack sets + instances, drift detection CRUD, custom resources backed by Lambda, cross-stack exports/imports, and a broad resource-provisioner library that creates real backing state in the other fakecloud services. +**Status: full API.** Stack lifecycle (create/update/delete with real events), nested stacks, SAM transform, change sets, stack sets whose instances provision real stacks across accounts and regions, drift detection, custom resources backed by Lambda, cross-stack exports/imports, and a broad resource-provisioner library that creates real backing state in the other fakecloud services. ## Protocol @@ -46,7 +46,20 @@ The structural check is deliberately no stricter than the deploy path: it never ## Stack sets -Full control plane: `CreateStackSet`, `UpdateStackSet`, `DeleteStackSet`, `DescribeStackSet`, `ListStackSets`, plus instance management (`CreateStackInstances`, `UpdateStackInstances`, `DeleteStackInstances`, `DescribeStackInstance`, `ListStackInstances`) and operation tracking (`DescribeStackSetOperation`, `ListStackSetOperations`, `ListStackSetOperationResults`, `StopStackSetOperation`). Self-managed and service-managed permission models both round-trip. +`CreateStackSet`, `UpdateStackSet`, `DeleteStackSet`, `DescribeStackSet`, `ListStackSets`, `CreateStackInstances`, `UpdateStackInstances`, `DeleteStackInstances`, `DescribeStackInstance`, `ListStackInstances`, `ImportStacksToStackSet`, `ListStackSetAutoDeploymentTargets`, and operation tracking via `DescribeStackSetOperation`, `ListStackSetOperations`, `ListStackSetOperationResults`, `StopStackSetOperation`. + +Stack instances are **real stacks**. `CreateStackInstances` creates a `StackSet--` stack in every target account and region through the same path `CreateStack` uses, so the template's resources exist in the target account's backing services (a queue shows up in that account's `ListQueues`) and the stack is visible to `DescribeStacks` there. The stack ID carries the target region and account. + +- **Parameters** - the stack set's `Parameters` apply to every instance; `ParameterOverrides` on `CreateStackInstances` / `UpdateStackInstances` override them per instance. `UsePreviousValue` keeps an override, and leaving a parameter out of the list reverts it to the stack set's value. Overriding a parameter the template does not declare is a `ValidationError`. +- **Updates** - `UpdateStackSet` stores the new template, parameters, capabilities and tags, then updates every instance's stack (or only the instances named by `Accounts`/`DeploymentTargets` + `Regions`, leaving the rest `OUTDATED`). `UpdateStackInstances` redeploys just the named instances. +- **Deletes** - `DeleteStackInstances` deletes each instance's stack and its resources, or with `RetainStacks=true` leaves the stack in place and only removes the instance. `DeleteStackSet` refuses a stack set that still has instances (`StackSetNotEmptyException`); a deleted stack set stays listable as `DELETED` and describable by its ID, and its name is free for reuse. +- **Operations** - every mutating call records an operation with a per-target result (`Account`, `Region`, `Status`, `StatusReason`, `AccountGateResult`). Targets deploy in `RegionOrder` then request order. A failing target counts against `FailureToleranceCount` / `FailureTolerancePercentage` per region; once the tolerance is exceeded the remaining targets are `CANCELLED` and the operation ends `FAILED`. Stacks that provision asynchronously (templates with custom resources) leave the operation `RUNNING` until they settle; `StopStackSetOperation` cancels the targets that have not started. A second operation while one is running is `OperationInProgressException`, and a reused `OperationId` is `OperationIdAlreadyExistsException`. +- **Account gate** - when a target account has a Lambda named `AWSCloudFormationStackSetAccountGate`, it is invoked before deploying and the deployment only proceeds if it returns `{"Status": "SUCCEEDED"}`. Without the function the gate is `SKIPPED`. +- **Service-managed** - `PermissionModel=SERVICE_MANAGED` needs an organization with StackSets trusted access (`ActivateOrganizationsAccess`, or Organizations `EnableAWSServiceAccess` for `member.org.stacksets.cloudformation.amazonaws.com`). `DeploymentTargets.OrganizationalUnitIds` resolve to the accounts in those OUs and every OU nested below them, never the management account; `AccountFilterType` (`INTERSECTION`, `DIFFERENCE`, `UNION`, `NONE`) combines them with `DeploymentTargets.Accounts` or an `AccountsUrl` file in S3. Suspended accounts are recorded as `SKIPPED_SUSPENDED_ACCOUNT`. `CallAs=DELEGATED_ADMIN` works from an account registered as a StackSets delegated administrator and acts on the management account's stack sets. +- **Import** - `ImportStacksToStackSet` adopts existing stacks (by `StackIds` or a `StackIdsUrl` file) as instances without redeploying them. A stack already managed by a stack set is refused, and a stack whose template differs from the stack set's is recorded as `FAILED_IMPORT`. `CreateStackSet` with `StackId` starts a stack set from a stack's template and parameters. +- **Drift** - `DetectStackSetDrift` checks each instance's stack resources against the live backing services, sets each instance's `DriftStatus`, and reports the counts in `StackSetDriftDetectionDetails`; `ListStackInstanceResourceDrifts` lists the per-resource results for an operation. + +Execution roles (`AdministrationRoleARN`, `ExecutionRoleName`) are recorded and reported, with the AWS defaults for self-managed stack sets, but not required to exist in the target account. ## Nested stacks @@ -62,7 +75,7 @@ Full control plane: `CreateStackSet`, `UpdateStackSet`, `DeleteStackSet`, `Descr ## Drift detection -`DetectStackDrift`, `DetectStackResourceDrift`, `DescribeStackDriftDetectionStatus`, `DetectStackSetDrift`. Detection runs synchronously and reports `IN_SYNC` for every resource — fakecloud is the source of truth for the backing state, so real drift never occurs. The detection IDs, statuses, and timestamps round-trip through the API for tooling that polls them. +`DetectStackDrift`, `DetectStackResourceDrift`, `DescribeStackDriftDetectionStatus`, `DescribeStackResourceDrifts`, and `DetectStackSetDrift` (see stack sets). Detection runs synchronously and checks whether each resource's physical resource still exists in its backing service: a resource deleted outside CloudFormation reports `DELETED` and the stack `DRIFTED`. Existence is checked for SQS queues, SNS topics, S3 buckets, Lambda functions, IAM roles, DynamoDB tables, KMS keys and Secrets Manager secrets; other types are assumed in sync, and property-level differences are not compared. ## Type registry, hooks, publishing @@ -169,7 +182,8 @@ aws --endpoint-url http://localhost:4566 cloudformation list-exports ## Gotchas - **Not every resource type provisions something.** Types in the provisioner list above create real backing state. Anything else (the remaining `AWS::EC2::*` types such as `NatGateway` / `Route`, etc.) is recorded but has no underlying resource, so a follow-up call against that service will 404. -- **Drift always reports IN_SYNC.** fakecloud is the source of truth for backing state, so real drift never occurs. The drift API still round-trips IDs and statuses for tooling that polls them. +- **Drift only detects deleted resources.** A resource removed outside CloudFormation reports `DELETED`; property changes made outside CloudFormation are not compared. +- **Unnamed resources are named after their logical ID.** Two stacks from one template in the same account (such as stack instances of one stack set in two regions) collide on resources that take a name, unless the template names them per stack, for example `QueueName: !Sub "${AWS::StackName}-queue"`. - **SAM expansion runs at create time.** A re-uploaded template still requires `Capabilities=[CAPABILITY_AUTO_EXPAND]` on operations that touch transforms. ## Source From 91c2b81ded5bbee3ac253d3a88e2c6b652683985 Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 20:56:18 -0300 Subject: [PATCH 02/13] fix(cloudformation): honor mid-flight stops and delegated-admin scope - Claim each target RUNNING before it deploys so a stop that lands during a deployment cancels the remaining targets instead of settling STOPPED while they keep deploying - CallAs=DELEGATED_ADMIN only reaches service-managed stack sets - Reject a NextToken past the end of the list --- crates/fakecloud-cloudformation/src/extras.rs | 4 +- .../src/stack_sets.rs | 305 +++++++++++++++--- 2 files changed, 262 insertions(+), 47 deletions(-) diff --git a/crates/fakecloud-cloudformation/src/extras.rs b/crates/fakecloud-cloudformation/src/extras.rs index 53348dcc3..3c1c92212 100644 --- a/crates/fakecloud-cloudformation/src/extras.rs +++ b/crates/fakecloud-cloudformation/src/extras.rs @@ -461,7 +461,9 @@ impl CloudFormationService { .state .read() .get(account_id) - .and_then(|st| crate::stack_sets::find_active(st, name)) + .and_then(|st| { + crate::stack_sets::find_active(st, name, crate::stack_sets::Scope::Own) + }) .map(|set| set.template_body.clone()); return match found { Some(body) => Ok(body), diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 93bf3516f..215649596 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -285,26 +285,56 @@ pub(crate) fn is_stack_set_action(action: &str) -> bool { ) } +/// Which stack sets a caller can address. A StackSets delegated administrator +/// acts on the management account's stack sets, but only the service-managed +/// ones. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum Scope { + Own, + DelegatedAdmin, +} + +impl Scope { + fn of(params: &BTreeMap) -> Self { + if params.get("CallAs").map(String::as_str) == Some("DELEGATED_ADMIN") { + Scope::DelegatedAdmin + } else { + Scope::Own + } + } + + fn sees(self, set: &StackSet) -> bool { + self == Scope::Own || set.permission_model == "SERVICE_MANAGED" + } +} + /// Find an ACTIVE stack set by name or id. pub(crate) fn find_active<'a>( state: &'a CloudFormationState, name_or_id: &str, + scope: Scope, ) -> Option<&'a StackSet> { - state - .stack_sets - .values() - .find(|s| s.status == "ACTIVE" && (s.name == name_or_id || s.stack_set_id == name_or_id)) + state.stack_sets.values().find(|s| { + s.status == "ACTIVE" + && (s.name == name_or_id || s.stack_set_id == name_or_id) + && scope.sees(s) + }) } /// Find a stack set for a read: an active one by name or id, or a deleted one /// by its (unique) id. A deleted stack set's name is free for reuse, so a name /// never resolves to one. -fn find_for_read<'a>(state: &'a CloudFormationState, name_or_id: &str) -> Option<&'a StackSet> { - find_active(state, name_or_id).or_else(|| state.stack_sets.get(name_or_id)) +fn find_for_read<'a>( + state: &'a CloudFormationState, + name_or_id: &str, + scope: Scope, +) -> Option<&'a StackSet> { + find_active(state, name_or_id, scope) + .or_else(|| state.stack_sets.get(name_or_id).filter(|s| scope.sees(s))) } -fn active_key(state: &CloudFormationState, name_or_id: &str) -> Option { - find_active(state, name_or_id).map(|s| s.stack_set_id.clone()) +fn active_key(state: &CloudFormationState, name_or_id: &str, scope: Scope) -> Option { + find_active(state, name_or_id, scope).map(|s| s.stack_set_id.clone()) } // ── Errors ── @@ -554,8 +584,12 @@ fn paginate( .filter(|v| *v > 0) .unwrap_or(DEFAULT_PAGE_SIZE); let total = items.len(); + if start > total { + return Err(validation("Invalid NextToken")); + } + let end = start.saturating_add(size); let page: Vec = items.into_iter().skip(start).take(size).collect(); - let next = (start + size < total).then(|| (start + size).to_string()); + let next = (end < total).then(|| end.to_string()); Ok((page, next)) } @@ -1246,6 +1280,11 @@ impl CloudFormationService { .cloned() .unwrap_or_else(|| "SELF_MANAGED".to_string()); let service_managed = permission_model == "SERVICE_MANAGED"; + if Scope::of(params) == Scope::DelegatedAdmin && !service_managed { + return Err(validation( + "A delegated administrator can only create stack sets with SERVICE_MANAGED permission model", + )); + } if service_managed { self.check_service_managed_allowed(&admin)?; } @@ -1316,7 +1355,7 @@ impl CloudFormationService { ); let mut accounts = self.state.write(); let state = accounts.get_or_create(&admin); - if find_active(state, &name).is_some() { + if find_active(state, &name, Scope::Own).is_some() { return Err(aws_err( StatusCode::CONFLICT, "NameAlreadyExistsException", @@ -1465,11 +1504,16 @@ impl CloudFormationService { } /// Resolve the stack set for a read, after folding in async progress. - fn read_stack_set(&self, admin: &str, name_or_id: &str) -> Result { + fn read_stack_set( + &self, + admin: &str, + name_or_id: &str, + scope: Scope, + ) -> Result { let mut accounts = self.state.write(); let id = accounts .get(admin) - .and_then(|s| find_for_read(s, name_or_id)) + .and_then(|s| find_for_read(s, name_or_id, scope)) .map(|s| s.stack_set_id.clone()) .ok_or_else(|| stack_set_not_found(name_or_id))?; Self::refresh_stack_set(&mut accounts, admin, &id); @@ -1487,7 +1531,7 @@ impl CloudFormationService { ) -> Result { let name = required(params, "StackSetName")?; let admin = self.stack_set_admin_account(req, params)?; - let set = self.read_stack_set(&admin, &name)?; + let set = self.read_stack_set(&admin, &name, Scope::of(params))?; Ok(xml_response( "DescribeStackSet", stack_set_el(&set), @@ -1510,11 +1554,7 @@ impl CloudFormationService { s.stack_sets .values() .filter(|set| wanted.is_none_or(|w| &set.status == w)) - .filter(|set| { - // A delegated administrator only sees service-managed - // stack sets. - admin == req.account_id || set.permission_model == "SERVICE_MANAGED" - }) + .filter(|set| Scope::of(params).sees(set)) .cloned() .collect() }) @@ -1611,7 +1651,7 @@ impl CloudFormationService { // Build the updated definition and resolve targets against a snapshot, // without holding the CloudFormation lock while Organizations and S3 // are read. The snapshot is re-validated under the lock below. - let snapshot = self.active_snapshot(&admin, &name)?; + let snapshot = self.active_snapshot(&admin, &name, Scope::of(params))?; Self::check_can_start_operation(&snapshot, &op_id)?; let mut updated = snapshot.clone(); if let Some(body) = new_template { @@ -1641,6 +1681,11 @@ impl CloudFormationService { } } let service_managed = updated.permission_model == "SERVICE_MANAGED"; + if Scope::of(params) == Scope::DelegatedAdmin && !service_managed { + return Err(validation( + "A delegated administrator can only manage stack sets with SERVICE_MANAGED permission model", + )); + } if service_managed && snapshot.permission_model != "SERVICE_MANAGED" { self.check_service_managed_allowed(&admin)?; } @@ -1757,11 +1802,16 @@ impl CloudFormationService { } /// A clone of the ACTIVE stack set `name`, with async progress folded in. - fn active_snapshot(&self, admin: &str, name: &str) -> Result { + fn active_snapshot( + &self, + admin: &str, + name: &str, + scope: Scope, + ) -> Result { let mut accounts = self.state.write(); let set_id = accounts .get(admin) - .and_then(|s| active_key(s, name)) + .and_then(|s| active_key(s, name, scope)) .ok_or_else(|| stack_set_not_found(name))?; Self::refresh_stack_set(&mut accounts, admin, &set_id); accounts @@ -1797,7 +1847,10 @@ impl CloudFormationService { let mut accounts = self.state.write(); // DeleteStackSet declares no not-found error; deleting a stack set that // does not exist is a no-op. - let Some(set_id) = accounts.get(&admin).and_then(|s| active_key(s, &name)) else { + let Some(set_id) = accounts + .get(&admin) + .and_then(|s| active_key(s, &name, Scope::of(params))) + else { return Ok(xml_response_no_result("DeleteStackSet", &req.request_id)); }; Self::refresh_stack_set(&mut accounts, &admin, &set_id); @@ -2180,7 +2233,7 @@ impl CloudFormationService { // Target resolution reads Organizations (and possibly S3) state, so it // runs against a snapshot of the stack set before the operation is // recorded under the CloudFormation lock. - let snapshot = self.active_snapshot(&admin, &name)?; + let snapshot = self.active_snapshot(&admin, &name, Scope::of(params))?; Self::check_overrides_declared(&snapshot, overrides.keys().cloned())?; let targets = self.resolve_new_targets( &snapshot, @@ -2252,7 +2305,7 @@ impl CloudFormationService { }); let overrides = overrides.transpose()?; - let snapshot = self.active_snapshot(&admin, &name)?; + let snapshot = self.active_snapshot(&admin, &name, Scope::of(params))?; if let Some(specs) = &overrides { Self::check_overrides_declared( &snapshot, @@ -2316,7 +2369,7 @@ impl CloudFormationService { .cloned() .unwrap_or_else(|| uuid::Uuid::new_v4().to_string()); - let snapshot = self.active_snapshot(&admin, &name)?; + let snapshot = self.active_snapshot(&admin, &name, Scope::of(params))?; let targets = self.existing_instance_targets( &snapshot, &admin, @@ -2408,9 +2461,10 @@ impl CloudFormationService { let mut abort: Option<&'static str> = None; for target in targets { - if abort.is_none() - && self.operation_status(admin, set_id, op_id).as_deref() == Some("STOPPING") - { + // Claiming the target marks it RUNNING under the lock, so a + // StopStackSetOperation that lands while it deploys cannot settle + // the operation as STOPPED underneath it. + if abort.is_none() && !self.claim_target(admin, set_id, op_id, &target) { abort = Some(OPERATION_STOPPED); } let mut gate = None; @@ -2461,13 +2515,28 @@ impl CloudFormationService { } } - fn operation_status(&self, admin: &str, set_id: &str, op_id: &str) -> Option { - self.state - .read() - .get(admin) - .and_then(|s| s.stack_sets.get(set_id)) - .and_then(|set| set.operations.iter().find(|o| o.operation_id == op_id)) - .map(|o| o.status.clone()) + /// Mark a target's result RUNNING before it deploys. Returns false when + /// the operation has been stopped, in which case the target must not run. + fn claim_target(&self, admin: &str, set_id: &str, op_id: &str, target: &Target) -> bool { + let mut accounts = self.state.write(); + let Some(op) = accounts + .get_mut(admin) + .and_then(|s| s.stack_sets.get_mut(set_id)) + .and_then(|set| set.operations.iter_mut().find(|o| o.operation_id == op_id)) + else { + return false; + }; + if op.status != "RUNNING" { + return false; + } + if let Some(result) = op + .results + .iter_mut() + .find(|r| r.account == target.account && r.region == target.region) + { + result.status = "RUNNING".to_string(); + } + true } /// Run the account's `AWSCloudFormationStackSetAccountGate` Lambda, if it @@ -2791,7 +2860,7 @@ impl CloudFormationService { let account = required(params, "StackInstanceAccount")?; let region = required(params, "StackInstanceRegion")?; let admin = self.stack_set_admin_account(req, params)?; - let set = self.read_stack_set(&admin, &name)?; + let set = self.read_stack_set(&admin, &name, Scope::of(params))?; let instance = set .instances .iter() @@ -2815,7 +2884,7 @@ impl CloudFormationService { ) -> Result { let name = required(params, "StackSetName")?; let admin = self.stack_set_admin_account(req, params)?; - let set = self.read_stack_set(&admin, &name)?; + let set = self.read_stack_set(&admin, &name, Scope::of(params))?; let mut filters: Vec<(String, String)> = Vec::new(); for i in 1.. { let Some(filter_name) = params.get(&format!("Filters.member.{i}.Name")) else { @@ -2870,7 +2939,7 @@ impl CloudFormationService { let name = required(params, "StackSetName")?; let op_id = required(params, "OperationId")?; let admin = self.stack_set_admin_account(req, params)?; - let set = self.read_stack_set(&admin, &name)?; + let set = self.read_stack_set(&admin, &name, Scope::of(params))?; let op = set .operations .iter() @@ -2890,7 +2959,7 @@ impl CloudFormationService { ) -> Result { let name = required(params, "StackSetName")?; let admin = self.stack_set_admin_account(req, params)?; - let set = self.read_stack_set(&admin, &name)?; + let set = self.read_stack_set(&admin, &name, Scope::of(params))?; // Most recent first. let ops: Vec<&StackSetOperation> = set.operations.iter().rev().collect(); let (page, next) = paginate(ops, params)?; @@ -2914,7 +2983,7 @@ impl CloudFormationService { let name = required(params, "StackSetName")?; let op_id = required(params, "OperationId")?; let admin = self.stack_set_admin_account(req, params)?; - let set = self.read_stack_set(&admin, &name)?; + let set = self.read_stack_set(&admin, &name, Scope::of(params))?; let op = set .operations .iter() @@ -2959,7 +3028,7 @@ impl CloudFormationService { let mut accounts = self.state.write(); let set_id = accounts .get(&admin) - .and_then(|s| find_for_read(s, &name)) + .and_then(|s| find_for_read(s, &name, Scope::of(params))) .map(|s| s.stack_set_id.clone()) .ok_or_else(|| stack_set_not_found(&name))?; Self::refresh_stack_set(&mut accounts, &admin, &set_id); @@ -3069,7 +3138,7 @@ impl CloudFormationService { let mut accounts = self.state.write(); let set_id = accounts .get(&admin) - .and_then(|s| active_key(s, &name)) + .and_then(|s| active_key(s, &name, Scope::of(params))) .ok_or_else(|| stack_set_not_found(&name))?; Self::refresh_stack_set(&mut accounts, &admin, &set_id); let set = accounts @@ -3220,7 +3289,7 @@ impl CloudFormationService { ) -> Result { let name = required(params, "StackSetName")?; let admin = self.stack_set_admin_account(req, params)?; - let set = self.read_stack_set(&admin, &name)?; + let set = self.read_stack_set(&admin, &name, Scope::of(params))?; let mut by_ou: BTreeMap> = BTreeMap::new(); if set.permission_model == "SERVICE_MANAGED" { for instance in &set.instances { @@ -3274,7 +3343,7 @@ impl CloudFormationService { let mut accounts = self.state.write(); let set_id = accounts .get(&admin) - .and_then(|s| active_key(s, &name)) + .and_then(|s| active_key(s, &name, Scope::of(params))) .ok_or_else(|| stack_set_not_found(&name))?; Self::refresh_stack_set(&mut accounts, &admin, &set_id); let set = accounts @@ -3424,7 +3493,7 @@ impl CloudFormationService { let region = required(params, "StackInstanceRegion")?; let op_id = required(params, "OperationId")?; let admin = self.stack_set_admin_account(req, params)?; - let set = self.read_stack_set(&admin, &name)?; + let set = self.read_stack_set(&admin, &name, Scope::of(params))?; let op = set .operations .iter() @@ -3554,7 +3623,7 @@ mod tests { svc.state .read() .get(ADMIN) - .and_then(|s| find_active(s, name)) + .and_then(|s| find_active(s, name, Scope::Own)) .cloned() .expect("stack set") } @@ -4660,6 +4729,150 @@ mod tests { ); } + /// An account gate that stops the operation it is gating, standing in for + /// a StopStackSetOperation that lands while a target is deploying. + struct StoppingGate(std::sync::OnceLock>); + + impl LambdaDelivery for StoppingGate { + fn invoke_lambda( + &self, + _function_arn: &str, + _payload: &str, + ) -> std::pin::Pin, String>> + Send>> + { + let svc = self.0.get().cloned(); + Box::pin(async move { + if let Some(svc) = svc { + call( + &svc, + "StopStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "op-stop")], + ) + .await + .map_err(|e| e.message())?; + } + Ok(br#"{"Status":"SUCCEEDED"}"#.to_vec()) + }) + } + } + + #[tokio::test] + async fn stopping_an_operation_mid_deployment_cancels_the_remaining_targets() { + let gate = Arc::new(StoppingGate(std::sync::OnceLock::new())); + let mut d = deps(); + d.delivery = Arc::new(DeliveryBus::new().with_lambda(gate.clone())); + let svc = Arc::new(service_with(d)); + gate.0.set(svc.clone()).ok(); + // Only the first target's account has a gate, so the stop lands while + // that target is deploying. + let gate_template = "Resources:\n Gate:\n Type: AWS::Lambda::Function\n Properties:\n FunctionName: AWSCloudFormationStackSetAccountGate\n Runtime: python3.12\n Handler: index.handler\n Role: arn:aws:iam::111111111111:role/gate\n Code:\n ZipFile: \"def handler(e, c): return {}\"\n"; + call_as( + &svc, + ACCT_B, + "CreateStack", + &[("StackName", "gate"), ("TemplateBody", gate_template)], + ) + .await + .unwrap(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Accounts.member.2", ACCT_C), + ("Regions.member.1", "us-east-1"), + ("OperationId", "op-stop"), + ], + ) + .await; + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "op-stop")], + ) + .await; + assert_eq!(tag(&op, "Status"), "STOPPED", "{op}"); + let results = ok( + &svc, + "ListStackSetOperationResults", + &[("StackSetName", "app"), ("OperationId", "op-stop")], + ) + .await; + // The target already deploying finishes; the next one never starts. + assert_eq!(queue_count(&svc, ACCT_B), 1); + assert_eq!(queue_count(&svc, ACCT_C), 0); + assert!(results.contains("SUCCEEDED"), "{results}"); + assert!(results.contains(OPERATION_STOPPED), "{results}"); + } + + #[tokio::test] + async fn a_delegated_administrator_cannot_touch_self_managed_stack_sets() { + let svc = service(); + seed_org(&svc); + { + let mut orgs = svc.deps.organizations.write(); + let org = orgs.as_mut().unwrap(); + org.enable_aws_service_access(STACKSETS_PRINCIPAL); + org.register_delegated_administrator(ACCT_B, STACKSETS_PRINCIPAL) + .unwrap(); + } + create_set(&svc, "mgmt-only", QUEUE_TEMPLATE).await; + let delegated = [("StackSetName", "mgmt-only"), ("CallAs", "DELEGATED_ADMIN")]; + let e = call_as(&svc, ACCT_B, "DescribeStackSet", &delegated) + .await + .unwrap_err(); + assert_eq!(e.code(), "StackSetNotFoundException"); + let mut instances = delegated.to_vec(); + instances.extend([ + ("Accounts.member.1", ACCT_C), + ("Regions.member.1", "us-east-1"), + ]); + let e = call_as(&svc, ACCT_B, "CreateStackInstances", &instances) + .await + .unwrap_err(); + assert_eq!(e.code(), "StackSetNotFoundException"); + assert_eq!(queue_count(&svc, ACCT_C), 0); + let e = call_as( + &svc, + ACCT_B, + "CreateStackSet", + &[ + ("StackSetName", "sneaky"), + ("TemplateBody", QUEUE_TEMPLATE), + ("CallAs", "DELEGATED_ADMIN"), + ], + ) + .await + .unwrap_err(); + assert_eq!(e.code(), "ValidationError"); + let listed = call_as( + &svc, + ACCT_B, + "ListStackSets", + &[("CallAs", "DELEGATED_ADMIN")], + ) + .await + .unwrap(); + assert!(!listed.contains("mgmt-only"), "{listed}"); + } + + #[tokio::test] + async fn an_out_of_range_next_token_is_rejected() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + let e = err( + &svc, + "ListStackSets", + &[("NextToken", "18446744073709551615")], + ) + .await; + assert_eq!(e.code(), "ValidationError"); + let page = ok(&svc, "ListStackSets", &[("MaxResults", "1")]).await; + assert!(!page.contains(""), "{page}"); + } + struct Gate(&'static str); impl LambdaDelivery for Gate { From d42036709117880a7707fc8e3b75a8a4193a92dd Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 21:04:25 -0300 Subject: [PATCH 03/13] fix(cloudformation): resolve nested OUs on instance targets; recheck drift start - Update/DeleteStackInstances reach instances deployed through OUs nested below (or above) the requested OUs - DetectStackSetDrift re-checks for a running operation or reused OperationId before recording its operation --- .../src/stack_sets.rs | 54 +++++++++++++++++-- 1 file changed, 50 insertions(+), 4 deletions(-) diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 215649596..5a055679d 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -2093,12 +2093,31 @@ impl CloudFormationService { })?; let filter_accounts = self.target_accounts_list(admin, dt)?; let filter = self.account_filter_type(dt, &filter_accounts)?; + // An OU covers every OU nested below it, so an instance deployed + // through a child OU is reached through its parent or the root, + // and one deployed through a parent is reached through a child. + // The OU recorded on the instance still counts, for an account + // that has since moved out. + let accounts_in_ous: BTreeSet = self + .deps + .organizations + .read() + .as_ref() + .map(|org| { + dt.organizational_unit_ids + .iter() + .flat_map(|ou| accounts_under(org, ou)) + .map(|(account, _)| account) + .collect() + }) + .unwrap_or_default(); for region in regions { for instance in set.instances.iter().filter(|i| &i.region == region) { - let in_ou = instance - .organizational_unit_id - .as_ref() - .is_some_and(|ou| dt.organizational_unit_ids.contains(ou)); + let in_ou = accounts_in_ous.contains(&instance.account) + || instance + .organizational_unit_id + .as_ref() + .is_some_and(|ou| dt.organizational_unit_ids.contains(ou)); let listed = filter_accounts.contains(&instance.account); let keep = match filter.as_str() { "INTERSECTION" => in_ou && listed, @@ -3460,10 +3479,17 @@ impl CloudFormationService { op.ended_at = Some(Utc::now()); let mut accounts = self.state.write(); + Self::refresh_stack_set(&mut accounts, &admin, &set.stack_set_id); if let Some(stored) = accounts .get_mut(&admin) .and_then(|s| s.stack_sets.get_mut(&set.stack_set_id)) { + // Re-validate against what is stored now: another operation may + // have started, or used this OperationId, while resources were + // being checked. (DetectStackSetDrift does not model + // StaleRequestException, and a drift result stays valid after an + // operation that has already finished.) + Self::check_can_start_operation(stored, &op_id)?; for instance in &mut stored.instances { if let Some((_, _, status)) = instance_drift .iter() @@ -4487,6 +4513,26 @@ mod tests { assert_eq!(eu.len(), 1); assert_eq!(eu[0].account, ACCT_C); + // Instances deployed through a nested OU are reached through a parent. + ok( + &svc, + "DeleteStackInstances", + &[ + ("StackSetName", "org"), + ( + "DeploymentTargets.OrganizationalUnitIds.member.1", + &workloads, + ), + ("Regions.member.1", "eu-west-1"), + ("RetainStacks", "false"), + ], + ) + .await; + assert!(stored_set(&svc, "org") + .instances + .iter() + .all(|i| i.region != "eu-west-1")); + let e = err( &svc, "CreateStackInstances", From 077f78b879ae4153456762909d65568bb42ef16a Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 21:12:26 -0300 Subject: [PATCH 04/13] fix(cloudformation): redeploy stackless instances; caller-account URLs - An update deploys an instance that never got a stack instead of failing it; suspended accounts are re-checked and skipped again - A region listed twice deploys once - TemplateURL, AccountsUrl and StackIdsUrl are read from the caller's account, not the management account a delegated admin acts on - Switching a stack set to SELF_MANAGED drops its AutoDeployment --- .../src/stack_sets.rs | 362 +++++++++++++++--- 1 file changed, 308 insertions(+), 54 deletions(-) diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 5a055679d..1c29840ad 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -1313,7 +1313,7 @@ impl CloudFormationService { (stack.template.clone(), params) } None => ( - self.stack_set_template_body(&admin, params) + self.stack_set_template_body(&req.account_id, params) .map_err(validation)? .unwrap_or_default(), BTreeMap::new(), @@ -1400,7 +1400,9 @@ impl CloudFormationService { let enabled = parse_bool(params, "AutoDeployment.Enabled")?; let retain = parse_bool(params, "AutoDeployment.RetainStacksOnAccountRemoval")?; if enabled.is_none() && retain.is_none() { - return Ok(previous.cloned()); + // Auto-deployment only exists for service-managed stack sets, so a + // switch to SELF_MANAGED drops it. + return Ok(previous.filter(|_| service_managed).cloned()); } if !service_managed { return Err(validation( @@ -1636,7 +1638,7 @@ impl CloudFormationService { .unwrap_or_else(|| uuid::Uuid::new_v4().to_string()); // Resolved before the state lock: a TemplateURL read takes the S3 lock. let new_template = self - .stack_set_template_body(&admin, params) + .stack_set_template_body(&req.account_id, params) .map_err(validation)?; let use_previous_template = parse_bool(params, "UsePreviousTemplate")?.unwrap_or(false); if use_previous_template && new_template.is_some() { @@ -1731,7 +1733,7 @@ impl CloudFormationService { let (targets, regions) = if targeted { let targets = self.existing_instance_targets( &updated, - &admin, + &req.account_id, &explicit_accounts, deployment_targets.as_ref(), &explicit_regions, @@ -1739,6 +1741,7 @@ impl CloudFormationService { )?; (targets, explicit_regions.clone()) } else { + let suspended = self.suspended_accounts(); let targets = updated .instances .iter() @@ -1746,7 +1749,7 @@ impl CloudFormationService { account: i.account.clone(), region: i.region.clone(), ou: i.organizational_unit_id.clone(), - suspended: false, + suspended: suspended.contains(&i.account), }) .collect(); (targets, stack_set_regions(&updated)) @@ -1886,7 +1889,7 @@ impl CloudFormationService { fn resolve_new_targets( &self, set: &StackSet, - admin: &str, + caller: &str, accounts: &[String], deployment_targets: Option<&DeploymentTargets>, regions: &[String], @@ -1911,7 +1914,7 @@ impl CloudFormationService { .ok_or_else(|| { validation("DeploymentTargets.OrganizationalUnitIds is required for SERVICE_MANAGED stack sets") })?; - let filter_accounts = self.target_accounts_list(admin, dt)?; + let filter_accounts = self.target_accounts_list(caller, dt)?; let filter = self.account_filter_type(dt, &filter_accounts)?; let orgs = self.deps.organizations.read(); let org = orgs @@ -1955,7 +1958,7 @@ impl CloudFormationService { "OrganizationalUnitIds are only supported for stack sets with SERVICE_MANAGED permission model", )); } - self.target_accounts_list(admin, dt)? + self.target_accounts_list(caller, dt)? } None => accounts.to_vec(), }; @@ -1977,7 +1980,8 @@ impl CloudFormationService { } } let mut targets = Vec::new(); - for region in regions { + let mut seen_regions = BTreeSet::new(); + for region in regions.iter().filter(|r| seen_regions.insert(*r)) { for (account, ou, suspended) in &resolved { targets.push(Target { account: account.clone(), @@ -1990,11 +1994,28 @@ impl CloudFormationService { Ok(targets) } + /// Organization member accounts that are not ACTIVE. Existing instances in + /// them are skipped again rather than failed. + fn suspended_accounts(&self) -> BTreeSet { + self.deps + .organizations + .read() + .as_ref() + .map(|org| { + org.accounts + .values() + .filter(|a| a.status != "ACTIVE") + .map(|a| a.id.clone()) + .collect() + }) + .unwrap_or_default() + } + /// `DeploymentTargets.Accounts` plus the accounts listed in /// `DeploymentTargets.AccountsUrl`, validated. fn target_accounts_list( &self, - admin: &str, + caller: &str, dt: &DeploymentTargets, ) -> Result, AwsServiceError> { let mut list = dt.accounts.clone(); @@ -2005,7 +2026,7 @@ impl CloudFormationService { ))); } let body = self - .resolve_template_url(admin, url) + .resolve_template_url(caller, url) .map_err(|_| validation(format!("Unable to read the accounts file at {url}")))?; list.extend( body.split([',', '\n', '\r']) @@ -2052,7 +2073,7 @@ impl CloudFormationService { fn existing_instance_targets( &self, set: &StackSet, - admin: &str, + caller: &str, accounts: &[String], deployment_targets: Option<&DeploymentTargets>, regions: &[String], @@ -2066,6 +2087,7 @@ impl CloudFormationService { "Only one of Accounts or DeploymentTargets can be specified", )); } + let suspended = self.suspended_accounts(); let mut targets = Vec::new(); let mut push = |instance: &StackInstance| { if !targets @@ -2076,7 +2098,7 @@ impl CloudFormationService { account: instance.account.clone(), region: instance.region.clone(), ou: instance.organizational_unit_id.clone(), - suspended: false, + suspended: suspended.contains(&instance.account), }); } }; @@ -2091,7 +2113,7 @@ impl CloudFormationService { .ok_or_else(|| { validation("DeploymentTargets.OrganizationalUnitIds is required for SERVICE_MANAGED stack sets") })?; - let filter_accounts = self.target_accounts_list(admin, dt)?; + let filter_accounts = self.target_accounts_list(caller, dt)?; let filter = self.account_filter_type(dt, &filter_accounts)?; // An OU covers every OU nested below it, so an instance deployed // through a child OU is reached through its parent or the root, @@ -2138,7 +2160,7 @@ impl CloudFormationService { "OrganizationalUnitIds are only supported for stack sets with SERVICE_MANAGED permission model", )); } - self.target_accounts_list(admin, dt)? + self.target_accounts_list(caller, dt)? } None => accounts.to_vec(), }; @@ -2256,7 +2278,7 @@ impl CloudFormationService { Self::check_overrides_declared(&snapshot, overrides.keys().cloned())?; let targets = self.resolve_new_targets( &snapshot, - &admin, + &req.account_id, &accounts, deployment_targets.as_ref(), ®ions, @@ -2335,7 +2357,7 @@ impl CloudFormationService { } let targets = self.existing_instance_targets( &snapshot, - &admin, + &req.account_id, &accounts, deployment_targets.as_ref(), ®ions, @@ -2391,7 +2413,7 @@ impl CloudFormationService { let snapshot = self.active_snapshot(&admin, &name, Scope::of(params))?; let targets = self.existing_instance_targets( &snapshot, - &admin, + &req.account_id, &accounts, deployment_targets.as_ref(), ®ions, @@ -2690,35 +2712,10 @@ impl CloudFormationService { (outcome, id, Some(overrides.clone())) } None => { - let stack_name = format!( - "StackSet-{}-{}", - spec.name.replace(':', "-"), - uuid::Uuid::new_v4() - ); - let mut stack_params = spec.stack_params(overrides); - stack_params.push(("StackName".to_string(), stack_name.clone())); - let request = synthetic_request( - &target.account, - &target.region, - "CreateStack", - &req.request_id, - stack_params, - ); - if let Err(e) = self.create_stack(&request).await { - return (Outcome::Failed(e.message()), None, Some(overrides.clone())); - } - match self.stack_status(&target.account, &stack_name) { - Some((stack_id, status, reason)) => ( - stack_outcome("CREATE", &status, reason.as_deref()), - Some(stack_id), - Some(overrides.clone()), - ), - None => ( - Outcome::Failed(format!("Stack {stack_name} was not created")), - None, - Some(overrides.clone()), - ), - } + let (outcome, id) = self + .create_instance_stack(req, spec, target, overrides) + .await; + (outcome, id, Some(overrides.clone())) } }, TargetAction::Update { overrides } => { @@ -2739,12 +2736,22 @@ impl CloudFormationService { .collect(), }; let Some((stack_id, _, _)) = live_stack else { - let missing = existing.and_then(|i| i.stack_id).unwrap_or_default(); - return ( - Outcome::Failed(format!("Stack [{missing}] does not exist")), - None, - Some(resolved), - ); + // An instance that never got a stack (skipped, cancelled, + // or its create failed) is deployed now. One whose stack + // was deleted out from under it fails, as in AWS. + return match existing.and_then(|i| i.stack_id) { + Some(missing) => ( + Outcome::Failed(format!("Stack [{missing}] does not exist")), + None, + Some(resolved), + ), + None => { + let (outcome, id) = self + .create_instance_stack(req, spec, target, &resolved) + .await; + (outcome, id, Some(resolved)) + } + }; }; let (outcome, id) = self .update_instance_stack(req, spec, target, &stack_id, &resolved) @@ -2754,6 +2761,42 @@ impl CloudFormationService { } } + async fn create_instance_stack( + &self, + req: &AwsRequest, + spec: &DeploySpec, + target: &Target, + overrides: &BTreeMap, + ) -> (Outcome, Option) { + let stack_name = format!( + "StackSet-{}-{}", + spec.name.replace(':', "-"), + uuid::Uuid::new_v4() + ); + let mut stack_params = spec.stack_params(overrides); + stack_params.push(("StackName".to_string(), stack_name.clone())); + let request = synthetic_request( + &target.account, + &target.region, + "CreateStack", + &req.request_id, + stack_params, + ); + if let Err(e) = self.create_stack(&request).await { + return (Outcome::Failed(e.message()), None); + } + match self.stack_status(&target.account, &stack_name) { + Some((stack_id, status, reason)) => ( + stack_outcome("CREATE", &status, reason.as_deref()), + Some(stack_id), + ), + None => ( + Outcome::Failed(format!("Stack {stack_name} was not created")), + None, + ), + } + } + async fn update_instance_stack( &self, req: &AwsRequest, @@ -3111,7 +3154,7 @@ impl CloudFormationService { ))); } let body = self - .resolve_template_url(&admin, url) + .resolve_template_url(&req.account_id, url) .map_err(|_| validation(format!("Unable to read the stack ids file at {url}")))?; stack_ids = body .split([',', '\n', '\r']) @@ -4919,6 +4962,217 @@ mod tests { assert!(!page.contains(""), "{page}"); } + #[tokio::test] + async fn a_stack_set_update_deploys_instances_that_never_got_a_stack() { + let broken = "Resources:\n Q:\n Type: AWS::SQS::Queue\n Properties:\n QueueName:\n Fn::ImportValue: missing-export\n"; + let svc = service(); + create_set(&svc, "fixme", broken).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "fixme"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("Regions.member.2", "us-west-2"), + ], + ) + .await; + assert!(stored_set(&svc, "fixme") + .instances + .iter() + .all(|i| i.stack_id.is_none())); + + let xml = ok( + &svc, + "UpdateStackSet", + &[("StackSetName", "fixme"), ("TemplateBody", QUEUE_TEMPLATE)], + ) + .await; + let op = ok( + &svc, + "DescribeStackSetOperation", + &[ + ("StackSetName", "fixme"), + ("OperationId", &tag(&xml, "OperationId")), + ], + ) + .await; + assert_eq!(tag(&op, "Status"), "SUCCEEDED", "{op}"); + let set = stored_set(&svc, "fixme"); + assert!(set + .instances + .iter() + .all(|i| i.status == "CURRENT" && i.stack_id.is_some())); + assert_eq!(queue_count(&svc, ACCT_B), 2); + } + + #[tokio::test] + async fn suspended_accounts_are_skipped_on_create_and_update() { + let svc = service(); + let (workloads, _) = seed_org(&svc); + svc.deps + .organizations + .write() + .as_mut() + .unwrap() + .close_account(ACCT_C) + .unwrap(); + ok(&svc, "ActivateOrganizationsAccess", &[]).await; + ok( + &svc, + "CreateStackSet", + &[ + ("StackSetName", "org"), + ("TemplateBody", QUEUE_TEMPLATE), + ("PermissionModel", "SERVICE_MANAGED"), + ], + ) + .await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "org"), + ( + "DeploymentTargets.OrganizationalUnitIds.member.1", + &workloads, + ), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + let skipped = |svc: &CloudFormationService| { + stored_set(svc, "org") + .instances + .into_iter() + .find(|i| i.account == ACCT_C) + .unwrap() + .detailed_status + }; + assert_eq!(skipped(&svc), "SKIPPED_SUSPENDED_ACCOUNT"); + assert_eq!(queue_count(&svc, ACCT_C), 0); + + let xml = ok( + &svc, + "UpdateStackSet", + &[("StackSetName", "org"), ("TemplateBody", QUEUE_TEMPLATE)], + ) + .await; + let op = ok( + &svc, + "DescribeStackSetOperation", + &[ + ("StackSetName", "org"), + ("OperationId", &tag(&xml, "OperationId")), + ], + ) + .await; + assert_eq!(tag(&op, "Status"), "SUCCEEDED", "{op}"); + assert_eq!(skipped(&svc), "SKIPPED_SUSPENDED_ACCOUNT"); + assert_eq!(queue_count(&svc, ACCT_B), 1); + } + + #[tokio::test] + async fn a_region_listed_twice_deploys_once() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + let xml = ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("Regions.member.2", "us-east-1"), + ], + ) + .await; + let op = ok( + &svc, + "DescribeStackSetOperation", + &[ + ("StackSetName", "app"), + ("OperationId", &tag(&xml, "OperationId")), + ], + ) + .await; + assert_eq!(tag(&op, "Status"), "SUCCEEDED", "{op}"); + assert_eq!(stored_set(&svc, "app").instances.len(), 1); + assert_eq!(queue_count(&svc, ACCT_B), 1); + } + + #[tokio::test] + async fn a_delegated_administrator_reads_template_urls_from_its_own_account() { + let svc = service(); + seed_org(&svc); + { + let mut orgs = svc.deps.organizations.write(); + let org = orgs.as_mut().unwrap(); + org.enable_aws_service_access(STACKSETS_PRINCIPAL); + org.register_delegated_administrator(ACCT_B, STACKSETS_PRINCIPAL) + .unwrap(); + } + { + let mut s3 = svc.deps.s3.write(); + let state = s3.get_or_create(ACCT_B); + let mut bucket = fakecloud_s3::S3Bucket::new("templates", "us-east-1", ACCT_B); + bucket.objects.insert( + "set.yaml".to_string(), + fakecloud_s3::S3Object { + key: "set.yaml".to_string(), + body: fakecloud_s3::memory_body(bytes::Bytes::from_static( + QUEUE_TEMPLATE.as_bytes(), + )), + size: QUEUE_TEMPLATE.len() as u64, + ..Default::default() + }, + ); + state.buckets.insert("templates".to_string(), bucket); + } + call_as( + &svc, + ACCT_B, + "CreateStackSet", + &[ + ("StackSetName", "from-url"), + ("TemplateURL", "https://templates.s3.amazonaws.com/set.yaml"), + ("PermissionModel", "SERVICE_MANAGED"), + ("CallAs", "DELEGATED_ADMIN"), + ], + ) + .await + .unwrap(); + assert_eq!(stored_set(&svc, "from-url").template_body, QUEUE_TEMPLATE); + } + + #[tokio::test] + async fn switching_to_self_managed_drops_auto_deployment() { + let svc = service(); + seed_org(&svc); + ok(&svc, "ActivateOrganizationsAccess", &[]).await; + ok( + &svc, + "CreateStackSet", + &[ + ("StackSetName", "org"), + ("TemplateBody", QUEUE_TEMPLATE), + ("PermissionModel", "SERVICE_MANAGED"), + ("AutoDeployment.Enabled", "true"), + ], + ) + .await; + ok( + &svc, + "UpdateStackSet", + &[("StackSetName", "org"), ("PermissionModel", "SELF_MANAGED")], + ) + .await; + let described = ok(&svc, "DescribeStackSet", &[("StackSetName", "org")]).await; + assert!(!described.contains(""), "{described}"); + assert!(described.contains("SELF_MANAGED")); + } + struct Gate(&'static str); impl LambdaDelivery for Gate { From c841a2e978ef9ba5050124188f14beed4d795184 Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 21:23:44 -0300 Subject: [PATCH 05/13] fix(cloudformation): run stack set operations in the background - Operations deploy in a detached task on the server and the call returns the OperationId at once, so a client that drops its request cannot strand the operation RUNNING - Operations interrupted by a restart settle when state loads - A late outcome from a settled operation no longer writes over instances a newer operation owns - e2e and conformance tests poll operations to completion --- crates/fakecloud-cloudformation/src/extras.rs | 2 +- crates/fakecloud-cloudformation/src/lib.rs | 2 +- .../fakecloud-cloudformation/src/service.rs | 4 +- .../src/stack_sets.rs | 188 ++++++++++++++++-- .../tests/cloudformation.rs | 84 ++++---- .../tests/cloudformation_stack_sets.rs | 46 ++++- crates/fakecloud-server/src/main.rs | 4 +- .../content/docs/services/cloudformation.md | 2 +- 8 files changed, 255 insertions(+), 77 deletions(-) diff --git a/crates/fakecloud-cloudformation/src/extras.rs b/crates/fakecloud-cloudformation/src/extras.rs index 3c1c92212..458249be9 100644 --- a/crates/fakecloud-cloudformation/src/extras.rs +++ b/crates/fakecloud-cloudformation/src/extras.rs @@ -3355,7 +3355,7 @@ pub(crate) mod tests { "bodyless".to_string(), json!({"StackSetId": "bodyless:def"}), ); - crate::stack_sets::migrate_legacy_stack_sets(&mut accounts); + crate::stack_sets::restore_stack_sets(&mut accounts); } for key in ["myset", "myset:abc123"] { diff --git a/crates/fakecloud-cloudformation/src/lib.rs b/crates/fakecloud-cloudformation/src/lib.rs index 3c12058c9..bb76a488d 100644 --- a/crates/fakecloud-cloudformation/src/lib.rs +++ b/crates/fakecloud-cloudformation/src/lib.rs @@ -9,7 +9,7 @@ pub(crate) mod template_summary; pub mod xml_responses; pub use service::{CloudControlOutcome, CloudFormationDeps, CloudFormationService}; -pub use stack_sets::migrate_legacy_stack_sets; +pub use stack_sets::restore_stack_sets; pub use state::{ CloudFormationSnapshot, SharedCloudFormationState, CLOUDFORMATION_SNAPSHOT_SCHEMA_VERSION, }; diff --git a/crates/fakecloud-cloudformation/src/service.rs b/crates/fakecloud-cloudformation/src/service.rs index 5ec6bcedf..31ab7db30 100644 --- a/crates/fakecloud-cloudformation/src/service.rs +++ b/crates/fakecloud-cloudformation/src/service.rs @@ -505,6 +505,7 @@ fn reconstruct_stack_resource( } } +#[derive(Clone)] pub struct CloudFormationDeps { pub sqs: SharedSqsState, pub sns: SharedSnsState, @@ -593,6 +594,7 @@ pub struct CloudFormationDeps { pub kafka_runtime: Option>, } +#[derive(Clone)] pub struct CloudFormationService { pub(crate) state: SharedCloudFormationState, pub(crate) deps: CloudFormationDeps, @@ -1039,7 +1041,7 @@ impl CloudFormationService { self } - async fn save_snapshot(&self) { + pub(crate) async fn save_snapshot(&self) { let Some(store) = self.snapshot_store.clone() else { return; }; diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 1c29840ad..3cb94d8ad 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -211,10 +211,54 @@ pub struct InstanceResourceDrift { pub timestamp: DateTime, } +const OPERATION_INTERRUPTED: &str = "The operation was interrupted by a restart"; + +/// Bring persisted stack sets into a loadable state: migrate records from +/// older builds, and settle operations that were still deploying when the +/// process stopped. Nothing resumes those, so left RUNNING they would block +/// every later operation on their stack set. +pub fn restore_stack_sets(accounts: &mut MultiAccountState) { + migrate_legacy_stack_sets(accounts); + for (_, state) in accounts.iter_mut() { + for set in state.stack_sets.values_mut() { + settle_interrupted_operations(set); + } + } +} + +fn settle_interrupted_operations(set: &mut StackSet) { + for op in &mut set.operations { + if !matches!(op.status.as_str(), "RUNNING" | "STOPPING") { + continue; + } + for result in &mut op.results { + let status = match result.status.as_str() { + "PENDING" => "CANCELLED", + "RUNNING" => "FAILED", + _ => continue, + }; + result.status = status.to_string(); + result.status_reason = Some(OPERATION_INTERRUPTED.to_string()); + } + for instance in &mut set.instances { + if instance.last_operation_id.as_deref() == Some(op.operation_id.as_str()) + && matches!(instance.detailed_status.as_str(), "RUNNING" | "PENDING") + { + apply_to_instance( + instance, + &Outcome::Failed(OPERATION_INTERRUPTED.to_string()), + ); + } + } + op.status = settled_status(op).to_string(); + op.ended_at = Some(Utc::now()); + } +} + /// Move stack sets persisted by older builds, which kept a /// `{StackSetId, StackSetName, Status, TemplateBody}` JSON record in the /// generic `extras` store, into the typed store. -pub fn migrate_legacy_stack_sets(accounts: &mut MultiAccountState) { +fn migrate_legacy_stack_sets(accounts: &mut MultiAccountState) { for (account_id, state) in accounts.iter_mut() { let Some(legacy) = state.extras.remove("stack_sets") else { continue; @@ -1787,7 +1831,7 @@ impl CloudFormationService { .insert(set_id.clone(), updated); } - self.run_operation( + self.launch_operation( req, &admin, &set_id, @@ -2295,7 +2339,7 @@ impl CloudFormationService { None, )?; let set_id = snapshot.stack_set_id.clone(); - self.run_operation( + self.launch_operation( req, &admin, &set_id, @@ -2375,7 +2419,7 @@ impl CloudFormationService { None, )?; let set_id = snapshot.stack_set_id.clone(); - self.run_operation( + self.launch_operation( req, &admin, &set_id, @@ -2431,7 +2475,7 @@ impl CloudFormationService { Some(retain_stacks), )?; let set_id = snapshot.stack_set_id.clone(); - self.run_operation( + self.launch_operation( req, &admin, &set_id, @@ -2448,12 +2492,58 @@ impl CloudFormationService { )) } + /// Start a recorded operation's deployment. + /// + /// On the server (a multi-thread runtime) it runs as a detached task and + /// the call returns the OperationId straight away, as AWS does: callers + /// poll DescribeStackSetOperation. That also means a client that gives up + /// on its request (a read timeout behind a slow account gate or a + /// container-backed stack) cannot abandon the operation half way and leave + /// the stack set blocked behind it. Current-thread runtimes (unit tests) + /// run it inline. + #[allow(clippy::too_many_arguments)] + async fn launch_operation( + &self, + req: &AwsRequest, + admin: &str, + set_id: &str, + op_id: &str, + spec: &DeploySpec, + targets: Vec, + action: TargetAction, + ) { + let multi_thread = matches!( + tokio::runtime::Handle::try_current().map(|h| h.runtime_flavor()), + Ok(tokio::runtime::RuntimeFlavor::MultiThread) + ); + if !multi_thread { + self.run_operation(&req.request_id, admin, set_id, op_id, spec, targets, action) + .await; + return; + } + let svc = self.clone(); + let (request_id, admin, set_id, op_id, spec) = ( + req.request_id.clone(), + admin.to_string(), + set_id.to_string(), + op_id.to_string(), + spec.clone(), + ); + tokio::spawn(async move { + svc.run_operation(&request_id, &admin, &set_id, &op_id, &spec, targets, action) + .await; + // The request that started the operation has long since persisted + // its snapshot; persist the finished operation too. + svc.save_snapshot().await; + }); + } + /// Run an operation's targets in order, recording each outcome as it /// lands, honoring the failure tolerance and StopStackSetOperation. #[allow(clippy::too_many_arguments)] async fn run_operation( &self, - req: &AwsRequest, + request_id: &str, admin: &str, set_id: &str, op_id: &str, @@ -2522,7 +2612,7 @@ impl CloudFormationService { gate = Some(g); if passed { let (outcome, id, applied) = self - .apply_target(req, admin, set_id, spec, &target, &action) + .apply_target(request_id, admin, set_id, spec, &target, &action) .await; stack_id = id; overrides = applied; @@ -2664,7 +2754,7 @@ impl CloudFormationService { /// the overrides the instance now carries. async fn apply_target( &self, - req: &AwsRequest, + request_id: &str, admin: &str, set_id: &str, spec: &DeploySpec, @@ -2690,7 +2780,7 @@ impl CloudFormationService { &target.account, &target.region, "DeleteStack", - &req.request_id, + request_id, vec![("StackName".to_string(), stack_id.clone())], ); if let Err(e) = self.delete_stack(&request).await { @@ -2707,13 +2797,13 @@ impl CloudFormationService { TargetAction::Create { overrides } => match live_stack { Some((stack_id, _, _)) => { let (outcome, id) = self - .update_instance_stack(req, spec, target, &stack_id, overrides) + .update_instance_stack(request_id, spec, target, &stack_id, overrides) .await; (outcome, id, Some(overrides.clone())) } None => { let (outcome, id) = self - .create_instance_stack(req, spec, target, overrides) + .create_instance_stack(request_id, spec, target, overrides) .await; (outcome, id, Some(overrides.clone())) } @@ -2747,14 +2837,14 @@ impl CloudFormationService { ), None => { let (outcome, id) = self - .create_instance_stack(req, spec, target, &resolved) + .create_instance_stack(request_id, spec, target, &resolved) .await; (outcome, id, Some(resolved)) } }; }; let (outcome, id) = self - .update_instance_stack(req, spec, target, &stack_id, &resolved) + .update_instance_stack(request_id, spec, target, &stack_id, &resolved) .await; (outcome, id, Some(resolved)) } @@ -2763,7 +2853,7 @@ impl CloudFormationService { async fn create_instance_stack( &self, - req: &AwsRequest, + request_id: &str, spec: &DeploySpec, target: &Target, overrides: &BTreeMap, @@ -2779,7 +2869,7 @@ impl CloudFormationService { &target.account, &target.region, "CreateStack", - &req.request_id, + request_id, stack_params, ); if let Err(e) = self.create_stack(&request).await { @@ -2799,7 +2889,7 @@ impl CloudFormationService { async fn update_instance_stack( &self, - req: &AwsRequest, + request_id: &str, spec: &DeploySpec, target: &Target, stack_id: &str, @@ -2811,7 +2901,7 @@ impl CloudFormationService { &target.account, &target.region, "UpdateStack", - &req.request_id, + request_id, stack_params, ); if let Err(e) = self.update_stack(&request).await { @@ -2844,6 +2934,16 @@ impl CloudFormationService { else { return; }; + // Once the operation has settled (stopped, and settled by a read + // while this loop was still going) a newer operation may own the + // instances; a late outcome must not write over them. + if !set + .operations + .iter() + .any(|o| o.operation_id == op_id && matches!(o.status.as_str(), "RUNNING" | "STOPPING")) + { + return; + } if let Some(result) = set .operations .iter_mut() @@ -5173,6 +5273,60 @@ mod tests { assert!(described.contains("SELF_MANAGED")); } + #[tokio::test] + async fn operations_interrupted_by_a_restart_are_settled_on_load() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + { + let mut accounts = svc.state.write(); + let set = accounts + .get_or_create(ADMIN) + .stack_sets + .values_mut() + .next() + .unwrap(); + let mut op = CloudFormationService::new_operation( + &set.clone(), + "cut-short", + "CREATE", + OperationPreferences::default(), + None, + None, + ); + for (account, status) in [(ACCT_B, "RUNNING"), (ACCT_C, "PENDING")] { + op.results.push(OperationResult { + account: account.to_string(), + region: "us-east-1".to_string(), + status: status.to_string(), + status_reason: None, + organizational_unit_id: None, + account_gate_status: None, + account_gate_reason: None, + }); + } + set.operations.push(op); + restore_stack_sets(&mut accounts); + } + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "cut-short")], + ) + .await; + assert_eq!(tag(&op, "Status"), "FAILED", "{op}"); + // The stack set accepts new operations again. + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + } + struct Gate(&'static str); impl LambdaDelivery for Gate { diff --git a/crates/fakecloud-conformance/tests/cloudformation.rs b/crates/fakecloud-conformance/tests/cloudformation.rs index 72d847363..ee89dfa5f 100644 --- a/crates/fakecloud-conformance/tests/cloudformation.rs +++ b/crates/fakecloud-conformance/tests/cloudformation.rs @@ -260,6 +260,34 @@ fn xml_tag(xml: &str, name: &str) -> String { .to_string() } +/// Start a stack set operation on `ss1` and wait for it to succeed, +/// returning its OperationId. +async fn start_operation(server: &TestServer, action: &str, params: &[(&str, &str)]) -> String { + let resp = cfn_post(server, action, params).await; + assert!(resp.status().is_success(), "{action}: {}", resp.status()); + let op_id = xml_tag(&resp.text().await.unwrap(), "OperationId"); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(60); + loop { + let op = cfn_post( + server, + "DescribeStackSetOperation", + &[("StackSetName", "ss1"), ("OperationId", &op_id)], + ) + .await + .text() + .await + .unwrap(); + match xml_tag(&op, "Status").as_str() { + "RUNNING" | "STOPPING" => { + assert!(std::time::Instant::now() < deadline, "{action} stuck: {op}"); + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + "SUCCEEDED" => return op_id, + _ => panic!("{action} did not succeed: {op}"), + } + } +} + fn pct(s: &str) -> String { let mut out = String::with_capacity(s.len()); for &b in s.as_bytes() { @@ -446,7 +474,8 @@ async fn cloudformation_closure_routes_exist() { ); // Stack sets and their instances, as one lifecycle: every operation below - // acts on state an earlier call created. + // acts on state an earlier call created. Operations deploy in the + // background, so each is awaited before the next starts. let stack_set_template = r#"{"Resources":{"Q":{"Type":"AWS::SQS::Queue","Properties":{"QueueName":{"Fn::Sub":"${AWS::StackName}-q"}}}}}"#; for (action, params) in [ ( @@ -467,24 +496,7 @@ async fn cloudformation_closure_routes_exist() { ("Accounts.member.1", "111111111111"), ("Regions.member.1", "us-east-1"), ]; - let create_op = xml_tag( - &cfn_post(&server, "CreateStackInstances", &instance_target) - .await - .text() - .await - .unwrap(), - "OperationId", - ); - let described = cfn_post( - &server, - "DescribeStackSetOperation", - &[("StackSetName", "ss1"), ("OperationId", &create_op)], - ) - .await - .text() - .await - .unwrap(); - assert_eq!(xml_tag(&described, "Status"), "SUCCEEDED", "{described}"); + let create_op = start_operation(&server, "CreateStackInstances", &instance_target).await; let instance = cfn_post( &server, "DescribeStackInstance", @@ -510,20 +522,14 @@ async fn cloudformation_closure_routes_exist() { "ListStackSetAutoDeploymentTargets", vec![("StackSetName", "ss1")], ), - ("UpdateStackSet", vec![("StackSetName", "ss1")]), - ("UpdateStackInstances", instance_target.to_vec()), ] { let resp = cfn_post(&server, action, ¶ms).await; assert!(resp.status().is_success(), "{action}: {}", resp.status()); } - let drift_op = xml_tag( - &cfn_post(&server, "DetectStackSetDrift", &[("StackSetName", "ss1")]) - .await - .text() - .await - .unwrap(), - "OperationId", - ); + start_operation(&server, "UpdateStackSet", &[("StackSetName", "ss1")]).await; + start_operation(&server, "UpdateStackInstances", &instance_target).await; + let drift_op = + start_operation(&server, "DetectStackSetDrift", &[("StackSetName", "ss1")]).await; assert!(cfn_post( &server, "ListStackInstanceResourceDrifts", @@ -551,35 +557,27 @@ async fn cloudformation_closure_routes_exist() { ); let mut delete_instances = instance_target.to_vec(); delete_instances.push(("RetainStacks", "true")); - assert!(cfn_post(&server, "DeleteStackInstances", &delete_instances) - .await - .status() - .is_success()); - // The retained stack imports straight back in as an instance. + start_operation(&server, "DeleteStackInstances", &delete_instances).await; let listed = cfn_post(&server, "ListStackInstances", &[("StackSetName", "ss1")]) .await .text() .await .unwrap(); assert!(!listed.contains(""), "{listed}"); + // The retained stack imports straight back in as an instance. let retained_stack_id = xml_tag(&instance, "StackId"); - assert!(cfn_post( + start_operation( &server, "ImportStacksToStackSet", &[ ("StackSetName", "ss1"), - ("StackIds.member.1", &retained_stack_id) + ("StackIds.member.1", &retained_stack_id), ], ) - .await - .status() - .is_success()); + .await; let mut delete_again = instance_target.to_vec(); delete_again.push(("RetainStacks", "false")); - assert!(cfn_post(&server, "DeleteStackInstances", &delete_again) - .await - .status() - .is_success()); + start_operation(&server, "DeleteStackInstances", &delete_again).await; assert!( cfn_post(&server, "DeleteStackSet", &[("StackSetName", "ss1")]) .await diff --git a/crates/fakecloud-e2e/tests/cloudformation_stack_sets.rs b/crates/fakecloud-e2e/tests/cloudformation_stack_sets.rs index 381034c4e..f47d3fc63 100644 --- a/crates/fakecloud-e2e/tests/cloudformation_stack_sets.rs +++ b/crates/fakecloud-e2e/tests/cloudformation_stack_sets.rs @@ -50,20 +50,39 @@ async fn stack_set_queues(sqs: &aws_sdk_sqs::Client) -> Vec { .collect() } +/// Operations deploy in the background, as in AWS: poll until this one +/// finishes and return its final status. async fn operation_status( cfn: &aws_sdk_cloudformation::Client, operation_id: &str, ) -> StackSetOperationStatus { - cfn.describe_stack_set_operation() - .stack_set_name("regional") - .operation_id(operation_id) - .send() - .await - .unwrap() - .stack_set_operation() - .and_then(|op| op.status()) - .cloned() - .expect("operation status") + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(60); + loop { + let status = cfn + .describe_stack_set_operation() + .stack_set_name("regional") + .operation_id(operation_id) + .send() + .await + .unwrap() + .stack_set_operation() + .and_then(|op| op.status()) + .cloned() + .expect("operation status"); + if !matches!( + status, + StackSetOperationStatus::Running + | StackSetOperationStatus::Queued + | StackSetOperationStatus::Stopping + ) { + return status; + } + assert!( + std::time::Instant::now() < deadline, + "operation {operation_id} still {status:?}" + ); + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } } #[tokio::test] @@ -270,13 +289,18 @@ async fn stack_set_update_redeploys_instances() { .send() .await .unwrap(); - cfn.create_stack_instances() + let create = cfn + .create_stack_instances() .stack_set_name("regional") .accounts(DEFAULT_ACCOUNT) .regions("us-east-1") .send() .await .unwrap(); + assert_eq!( + operation_status(&cfn, create.operation_id().unwrap()).await, + StackSetOperationStatus::Succeeded + ); let update = cfn .update_stack_set() diff --git a/crates/fakecloud-server/src/main.rs b/crates/fakecloud-server/src/main.rs index c465d35a3..db840af1f 100644 --- a/crates/fakecloud-server/src/main.rs +++ b/crates/fakecloud-server/src/main.rs @@ -1452,7 +1452,7 @@ async fn main() { )); } if let Some(mut accounts) = snapshot.accounts { - fakecloud_cloudformation::migrate_legacy_stack_sets(&mut accounts); + fakecloud_cloudformation::restore_stack_sets(&mut accounts); let account_count = accounts.account_count(); *cloudformation_state.write() = accounts; tracing::info!( @@ -1464,7 +1464,7 @@ async fn main() { let account_id = single_state.account_id.clone(); let mut mas = cloudformation_state.write(); *mas.get_or_create(&account_id) = single_state; - fakecloud_cloudformation::migrate_legacy_stack_sets(&mut mas); + fakecloud_cloudformation::restore_stack_sets(&mut mas); tracing::info!( stacks = stack_count, "loaded cloudformation persistence snapshot (migrated from v1)" diff --git a/website/content/docs/services/cloudformation.md b/website/content/docs/services/cloudformation.md index 3bc82a33a..addf4ae0e 100644 --- a/website/content/docs/services/cloudformation.md +++ b/website/content/docs/services/cloudformation.md @@ -53,7 +53,7 @@ Stack instances are **real stacks**. `CreateStackInstances` creates a `StackSet- - **Parameters** - the stack set's `Parameters` apply to every instance; `ParameterOverrides` on `CreateStackInstances` / `UpdateStackInstances` override them per instance. `UsePreviousValue` keeps an override, and leaving a parameter out of the list reverts it to the stack set's value. Overriding a parameter the template does not declare is a `ValidationError`. - **Updates** - `UpdateStackSet` stores the new template, parameters, capabilities and tags, then updates every instance's stack (or only the instances named by `Accounts`/`DeploymentTargets` + `Regions`, leaving the rest `OUTDATED`). `UpdateStackInstances` redeploys just the named instances. - **Deletes** - `DeleteStackInstances` deletes each instance's stack and its resources, or with `RetainStacks=true` leaves the stack in place and only removes the instance. `DeleteStackSet` refuses a stack set that still has instances (`StackSetNotEmptyException`); a deleted stack set stays listable as `DELETED` and describable by its ID, and its name is free for reuse. -- **Operations** - every mutating call records an operation with a per-target result (`Account`, `Region`, `Status`, `StatusReason`, `AccountGateResult`). Targets deploy in `RegionOrder` then request order. A failing target counts against `FailureToleranceCount` / `FailureTolerancePercentage` per region; once the tolerance is exceeded the remaining targets are `CANCELLED` and the operation ends `FAILED`. Stacks that provision asynchronously (templates with custom resources) leave the operation `RUNNING` until they settle; `StopStackSetOperation` cancels the targets that have not started. A second operation while one is running is `OperationInProgressException`, and a reused `OperationId` is `OperationIdAlreadyExistsException`. +- **Operations** - as in AWS, a mutating call returns its `OperationId` straight away and the deployment runs in the background; poll `DescribeStackSetOperation` until it leaves `RUNNING`. Every operation records a per-target result (`Account`, `Region`, `Status`, `StatusReason`, `AccountGateResult`). Targets deploy in `RegionOrder` then request order. A failing target counts against `FailureToleranceCount` / `FailureTolerancePercentage` per region; once the tolerance is exceeded the remaining targets are `CANCELLED` and the operation ends `FAILED`. Stacks that provision asynchronously (templates with custom resources) leave the operation `RUNNING` until they settle; `StopStackSetOperation` cancels the targets that have not started. A second operation while one is running is `OperationInProgressException`, and a reused `OperationId` is `OperationIdAlreadyExistsException`. An operation cut short by a restart is settled as `FAILED` when state is loaded, so it does not block the stack set. - **Account gate** - when a target account has a Lambda named `AWSCloudFormationStackSetAccountGate`, it is invoked before deploying and the deployment only proceeds if it returns `{"Status": "SUCCEEDED"}`. Without the function the gate is `SKIPPED`. - **Service-managed** - `PermissionModel=SERVICE_MANAGED` needs an organization with StackSets trusted access (`ActivateOrganizationsAccess`, or Organizations `EnableAWSServiceAccess` for `member.org.stacksets.cloudformation.amazonaws.com`). `DeploymentTargets.OrganizationalUnitIds` resolve to the accounts in those OUs and every OU nested below them, never the management account; `AccountFilterType` (`INTERSECTION`, `DIFFERENCE`, `UNION`, `NONE`) combines them with `DeploymentTargets.Accounts` or an `AccountsUrl` file in S3. Suspended accounts are recorded as `SKIPPED_SUSPENDED_ACCOUNT`. `CallAs=DELEGATED_ADMIN` works from an account registered as a StackSets delegated administrator and acts on the management account's stack sets. - **Import** - `ImportStacksToStackSet` adopts existing stacks (by `StackIds` or a `StackIdsUrl` file) as instances without redeploying them. A stack already managed by a stack set is refused, and a stack whose template differs from the stack set's is recorded as `FAILED_IMPORT`. `CreateStackSet` with `StackId` starts a stack set from a stack's template and parameters. From 6dd6cd391e16e5d25174e637a37453fc3963dc31 Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 21:30:11 -0300 Subject: [PATCH 06/13] fix(cloudformation): seed operation results when the operation is recorded A read landing before the background deployment task started saw a RUNNING operation with no results and settled it SUCCEEDED, leaving every target undeployed. Results are now seeded PENDING in the same locked step that records the operation. --- .../src/stack_sets.rs | 88 +++++++++++++------ 1 file changed, 62 insertions(+), 26 deletions(-) diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 3cb94d8ad..84b85448d 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -1157,6 +1157,22 @@ fn order_targets( targets } +/// A PENDING result for every target an operation will act on. +fn pending_results(targets: &[Target]) -> Vec { + targets + .iter() + .map(|t| OperationResult { + account: t.account.clone(), + region: t.region.clone(), + status: "PENDING".to_string(), + status_reason: None, + organizational_unit_id: t.ou.clone(), + account_gate_status: None, + account_gate_reason: None, + }) + .collect() +} + fn synthetic_request( account: &str, region: &str, @@ -1811,7 +1827,8 @@ impl CloudFormationService { let targets = order_targets(targets, ®ions, &preferences); let record = targeted.then(|| targets_record(&explicit_accounts, deployment_targets.as_ref())); - let op = Self::new_operation(&updated, &op_id, "UPDATE", preferences, record, None); + let mut op = Self::new_operation(&updated, &op_id, "UPDATE", preferences, record, None); + op.results = pending_results(&targets); updated.operations.push(op); let spec = DeploySpec::of(&updated); let set_id = updated.stack_set_id.clone(); @@ -2268,6 +2285,7 @@ impl CloudFormationService { &self, admin: &str, snapshot: &StackSet, + targets: &[Target], op_id: &str, action: &str, preferences: OperationPreferences, @@ -2283,7 +2301,7 @@ impl CloudFormationService { .ok_or_else(|| stack_set_not_found(&snapshot.name))?; Self::check_not_stale(set, snapshot)?; Self::check_can_start_operation(set, op_id)?; - let op = Self::new_operation( + let mut op = Self::new_operation( set, op_id, action, @@ -2291,6 +2309,10 @@ impl CloudFormationService { deployment_targets, retain_stacks, ); + // Seeded in the same locked step that records the operation: a read + // before the deployment task starts must not see a RUNNING operation + // with nothing left to run and settle it. + op.results = pending_results(targets); set.operations.push(op); Ok(DeploySpec::of(set)) } @@ -2332,6 +2354,7 @@ impl CloudFormationService { let spec = self.start_instance_operation( &admin, &snapshot, + &targets, &op_id, "CREATE", preferences, @@ -2412,6 +2435,7 @@ impl CloudFormationService { let spec = self.start_instance_operation( &admin, &snapshot, + &targets, &op_id, "UPDATE", preferences, @@ -2468,6 +2492,7 @@ impl CloudFormationService { let spec = self.start_instance_operation( &admin, &snapshot, + &targets, &op_id, "DELETE", preferences, @@ -2551,30 +2576,6 @@ impl CloudFormationService { targets: Vec, action: TargetAction, ) { - // Seed a PENDING result per target so the operation reports every - // target it will act on from the start. - { - let mut accounts = self.state.write(); - if let Some(op) = accounts - .get_mut(admin) - .and_then(|s| s.stack_sets.get_mut(set_id)) - .and_then(|set| set.operations.iter_mut().find(|o| o.operation_id == op_id)) - { - op.results = targets - .iter() - .map(|t| OperationResult { - account: t.account.clone(), - region: t.region.clone(), - status: "PENDING".to_string(), - status_reason: None, - organizational_unit_id: t.ou.clone(), - account_gate_status: None, - account_gate_reason: None, - }) - .collect(); - } - } - let prefs = { let accounts = self.state.read(); accounts @@ -5327,6 +5328,41 @@ mod tests { .await; } + #[tokio::test] + async fn a_read_before_deployment_starts_does_not_settle_the_operation() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + let snapshot = stored_set(&svc, "app"); + let targets = vec![Target { + account: ACCT_B.to_string(), + region: "us-east-1".to_string(), + ou: None, + suspended: false, + }]; + // Record the operation exactly as CreateStackInstances does, without + // running it yet, as when the deployment task has not been scheduled. + svc.start_instance_operation( + ADMIN, + &snapshot, + &targets, + "queued", + "CREATE", + OperationPreferences::default(), + None, + None, + ) + .unwrap(); + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "queued")], + ) + .await; + assert_eq!(tag(&op, "Status"), "RUNNING", "{op}"); + let e = err(&svc, "DeleteStackSet", &[("StackSetName", "app")]).await; + assert_eq!(e.code(), "OperationInProgressException"); + } + struct Gate(&'static str); impl LambdaDelivery for Gate { From 28f211dafccd8e8a29e01fb2a9a8158a4df97e0e Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 21:40:40 -0300 Subject: [PATCH 07/13] fix(cloudformation): list pending instances; refresh before restore settles - Instances an operation will deploy are recorded OUTDATED/PENDING when the operation is recorded, so they are listable while it runs; stopping it cancels the ones that never started - Restore folds finished stacks in before settling interrupted operations, so a stack that completed before a restart is not reported as failed --- .../src/stack_sets.rs | 138 +++++++++++++++++- 1 file changed, 136 insertions(+), 2 deletions(-) diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 84b85448d..29b1a453a 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -219,8 +219,23 @@ const OPERATION_INTERRUPTED: &str = "The operation was interrupted by a restart" /// every later operation on their stack set. pub fn restore_stack_sets(accounts: &mut MultiAccountState) { migrate_legacy_stack_sets(accounts); - for (_, state) in accounts.iter_mut() { - for set in state.stack_sets.values_mut() { + let sets: Vec<(String, String)> = accounts + .iter() + .flat_map(|(account, state)| { + state + .stack_sets + .keys() + .map(move |id| (account.to_string(), id.clone())) + }) + .collect(); + for (account, set_id) in sets { + // A stack that finished after the last snapshot of its stack set + // counts as finished, not as interrupted. + CloudFormationService::refresh_stack_set(accounts, &account, &set_id); + if let Some(set) = accounts + .get_mut(&account) + .and_then(|s| s.stack_sets.get_mut(&set_id)) + { settle_interrupted_operations(set); } } @@ -1157,6 +1172,43 @@ fn order_targets( targets } +/// Show the instances an operation is about to deploy as `OUTDATED` / +/// `PENDING` from the moment it is recorded, creating the records of new +/// ones, so they are listable while the deployment runs. +fn mark_instances_pending(set: &mut StackSet, targets: &[Target], op_id: &str, create: bool) { + for target in targets { + let position = set + .instances + .iter() + .position(|i| i.account == target.account && i.region == target.region); + let idx = match position { + Some(idx) => idx, + None if create => { + set.instances.push(StackInstance { + account: target.account.clone(), + region: target.region.clone(), + stack_id: None, + status: "OUTDATED".to_string(), + detailed_status: "PENDING".to_string(), + status_reason: None, + parameter_overrides: BTreeMap::new(), + organizational_unit_id: target.ou.clone(), + drift_status: "NOT_CHECKED".to_string(), + last_drift_check_timestamp: None, + last_operation_id: None, + }); + set.instances.len() - 1 + } + None => continue, + }; + let instance = &mut set.instances[idx]; + instance.status = "OUTDATED".to_string(); + instance.detailed_status = "PENDING".to_string(); + instance.status_reason = None; + instance.last_operation_id = Some(op_id.to_string()); + } +} + /// A PENDING result for every target an operation will act on. fn pending_results(targets: &[Target]) -> Vec { targets @@ -1830,6 +1882,7 @@ impl CloudFormationService { let mut op = Self::new_operation(&updated, &op_id, "UPDATE", preferences, record, None); op.results = pending_results(&targets); updated.operations.push(op); + mark_instances_pending(&mut updated, &targets, &op_id, false); let spec = DeploySpec::of(&updated); let set_id = updated.stack_set_id.clone(); { @@ -2314,6 +2367,11 @@ impl CloudFormationService { // with nothing left to run and settle it. op.results = pending_results(targets); set.operations.push(op); + match action { + "CREATE" => mark_instances_pending(set, targets, op_id, true), + "UPDATE" => mark_instances_pending(set, targets, op_id, false), + _ => {} + } Ok(DeploySpec::of(set)) } @@ -3213,13 +3271,30 @@ impl CloudFormationService { // Targets not yet started are cancelled; ones already deploying run to // completion, and the operation settles as STOPPED once they do. op.status = "STOPPING".to_string(); + let mut cancelled = Vec::new(); for result in &mut op.results { if result.status == "PENDING" { result.status = "CANCELLED".to_string(); result.status_reason = Some(OPERATION_STOPPED.to_string()); + cancelled.push((result.account.clone(), result.region.clone())); } } settle_operation(op); + if let Some(set) = accounts + .get_mut(&admin) + .and_then(|s| s.stack_sets.get_mut(&set_id)) + { + for instance in &mut set.instances { + if instance.last_operation_id.as_deref() == Some(op_id.as_str()) + && instance.detailed_status == "PENDING" + && cancelled + .iter() + .any(|(a, r)| *a == instance.account && *r == instance.region) + { + apply_to_instance(instance, &Outcome::Cancelled(OPERATION_STOPPED.to_string())); + } + } + } Ok(xml_response( "StopStackSetOperation", String::new(), @@ -5274,6 +5349,47 @@ mod tests { assert!(described.contains("SELF_MANAGED")); } + #[tokio::test] + async fn a_stack_that_finished_before_a_restart_is_not_counted_as_interrupted() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("OperationId", "op-async"), + ], + ) + .await; + { + // As persisted while the stack was still provisioning; the stack + // itself finished before the restart. + let mut accounts = svc.state.write(); + let set = accounts + .get_or_create(ADMIN) + .stack_sets + .values_mut() + .next() + .unwrap(); + set.instances[0].detailed_status = "RUNNING".to_string(); + let op = set.operations.last_mut().unwrap(); + op.status = "RUNNING".to_string(); + op.results[0].status = "RUNNING".to_string(); + restore_stack_sets(&mut accounts); + } + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "op-async")], + ) + .await; + assert_eq!(tag(&op, "Status"), "SUCCEEDED", "{op}"); + assert_eq!(stored_set(&svc, "app").instances[0].status, "CURRENT"); + } + #[tokio::test] async fn operations_interrupted_by_a_restart_are_settled_on_load() { let svc = service(); @@ -5361,6 +5477,24 @@ mod tests { assert_eq!(tag(&op, "Status"), "RUNNING", "{op}"); let e = err(&svc, "DeleteStackSet", &[("StackSetName", "app")]).await; assert_eq!(e.code(), "OperationInProgressException"); + // The instance it will create is already listed, pending. + let describe = [ + ("StackSetName", "app"), + ("StackInstanceAccount", ACCT_B), + ("StackInstanceRegion", "us-east-1"), + ]; + let instance = ok(&svc, "DescribeStackInstance", &describe).await; + assert_eq!(tag(&instance, "Status"), "OUTDATED", "{instance}"); + assert_eq!(tag(&instance, "DetailedStatus"), "PENDING", "{instance}"); + // Stopped before it started: the pending instance is cancelled. + ok( + &svc, + "StopStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "queued")], + ) + .await; + let instance = ok(&svc, "DescribeStackInstance", &describe).await; + assert_eq!(tag(&instance, "DetailedStatus"), "CANCELLED", "{instance}"); } struct Gate(&'static str); From e2937cde5624e5dc67336064fa16859639b77ac7 Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 21:48:57 -0300 Subject: [PATCH 08/13] fix(cloudformation): auto-deployment disable, delegated template summary, restart states - AutoDeployment.Enabled=false alone no longer inherits retention - GetTemplateSummary honors CallAs=DELEGATED_ADMIN for stack sets - Instances whose targets never started settle CANCELLED on restart, matching their results --- crates/fakecloud-cloudformation/src/extras.rs | 23 ++-- .../src/stack_sets.rs | 106 ++++++++++++++++-- 2 files changed, 111 insertions(+), 18 deletions(-) diff --git a/crates/fakecloud-cloudformation/src/extras.rs b/crates/fakecloud-cloudformation/src/extras.rs index 458249be9..c3be6951f 100644 --- a/crates/fakecloud-cloudformation/src/extras.rs +++ b/crates/fakecloud-cloudformation/src/extras.rs @@ -457,14 +457,23 @@ impl CloudFormationService { // Existence and body are separate questions: a stack set with an // empty template still EXISTS. AWS accepts the stack set's id // (`{name}:{suffix}`) as well as its name. + // `CallAs=DELEGATED_ADMIN` reads the management account's + // service-managed stack sets, as the stack set APIs do. A caller + // that is not a delegated administrator finds nothing: + // GetTemplateSummary declares no error for that but this one. let found = self - .state - .read() - .get(account_id) - .and_then(|st| { - crate::stack_sets::find_active(st, name, crate::stack_sets::Scope::Own) - }) - .map(|set| set.template_body.clone()); + .stack_set_admin_account_of(account_id, params) + .ok() + .and_then(|admin| { + self.state.read().get(&admin).and_then(|st| { + crate::stack_sets::find_active( + st, + name, + crate::stack_sets::Scope::of(params), + ) + .map(|set| set.template_body.clone()) + }) + }); return match found { Some(body) => Ok(body), // Unlike `ValidationError`, this one IS declared on diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 29b1a453a..5fa693f3e 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -255,15 +255,26 @@ fn settle_interrupted_operations(set: &mut StackSet) { result.status = status.to_string(); result.status_reason = Some(OPERATION_INTERRUPTED.to_string()); } + // An instance ends up as its result did: cancelled if its target + // never started, failed if it was deploying. for instance in &mut set.instances { - if instance.last_operation_id.as_deref() == Some(op.operation_id.as_str()) - && matches!(instance.detailed_status.as_str(), "RUNNING" | "PENDING") + if instance.last_operation_id.as_deref() != Some(op.operation_id.as_str()) + || !matches!(instance.detailed_status.as_str(), "RUNNING" | "PENDING") { - apply_to_instance( - instance, - &Outcome::Failed(OPERATION_INTERRUPTED.to_string()), - ); + continue; } + let cancelled = op.results.iter().any(|r| { + r.account == instance.account + && r.region == instance.region + && r.status == "CANCELLED" + }); + let reason = OPERATION_INTERRUPTED.to_string(); + let outcome = if cancelled { + Outcome::Cancelled(reason) + } else { + Outcome::Failed(reason) + }; + apply_to_instance(instance, &outcome); } op.status = settled_status(op).to_string(); op.ended_at = Some(Utc::now()); @@ -354,7 +365,7 @@ pub(crate) enum Scope { } impl Scope { - fn of(params: &BTreeMap) -> Self { + pub(crate) fn of(params: &BTreeMap) -> Self { if params.get("CallAs").map(String::as_str) == Some("DELEGATED_ADMIN") { Scope::DelegatedAdmin } else { @@ -1325,9 +1336,17 @@ impl CloudFormationService { &self, req: &AwsRequest, params: &BTreeMap, + ) -> Result { + self.stack_set_admin_account_of(&req.account_id, params) + } + + pub(crate) fn stack_set_admin_account_of( + &self, + caller: &str, + params: &BTreeMap, ) -> Result { match params.get("CallAs").map(String::as_str) { - None | Some("SELF") => Ok(req.account_id.clone()), + None | Some("SELF") => Ok(caller.to_string()), Some("DELEGATED_ADMIN") => { let orgs = self.deps.organizations.read(); let org = orgs.as_ref().ok_or_else(|| { @@ -1336,11 +1355,11 @@ impl CloudFormationService { let registered = org .delegated_administrators .get(STACKSETS_PRINCIPAL) - .is_some_and(|admins| admins.contains_key(&req.account_id)); + .is_some_and(|admins| admins.contains_key(caller)); if !registered { return Err(validation(format!( "Account {} is not registered as a delegated administrator for {STACKSETS_PRINCIPAL}", - req.account_id + caller ))); } Ok(org.management_account_id.clone()) @@ -1522,8 +1541,12 @@ impl CloudFormationService { )); } let enabled = enabled.or(previous.map(|p| p.enabled)).unwrap_or(false); + // Retention only means something while auto-deployment is on, so + // turning it off does not carry the previous setting along. let retain = retain - .or(previous.map(|p| p.retain_stacks_on_account_removal)) + .or(previous + .filter(|_| enabled) + .map(|p| p.retain_stacks_on_account_removal)) .unwrap_or(false); if retain && !enabled { return Err(validation( @@ -4805,6 +4828,15 @@ mod tests { listed.contains("delegated"), "{listed}" ); + let summary = call_as( + &svc, + ACCT_B, + "GetTemplateSummary", + &[("StackSetName", "delegated"), ("CallAs", "DELEGATED_ADMIN")], + ) + .await + .unwrap(); + assert!(summary.contains("AWS::SQS::Queue"), "{summary}"); // An account that is not registered cannot. let e = call_as( @@ -5347,6 +5379,33 @@ mod tests { let described = ok(&svc, "DescribeStackSet", &[("StackSetName", "org")]).await; assert!(!described.contains(""), "{described}"); assert!(described.contains("SELF_MANAGED")); + ok( + &svc, + "CreateStackSet", + &[ + ("StackSetName", "retained"), + ("TemplateBody", QUEUE_TEMPLATE), + ("PermissionModel", "SERVICE_MANAGED"), + ("AutoDeployment.Enabled", "true"), + ("AutoDeployment.RetainStacksOnAccountRemoval", "true"), + ], + ) + .await; + // Turning auto-deployment off on its own works. + ok( + &svc, + "UpdateStackSet", + &[ + ("StackSetName", "retained"), + ("AutoDeployment.Enabled", "false"), + ], + ) + .await; + let described = ok(&svc, "DescribeStackSet", &[("StackSetName", "retained")]).await; + assert!( + described.contains("false"), + "{described}" + ); } #[tokio::test] @@ -5411,6 +5470,19 @@ mod tests { None, ); for (account, status) in [(ACCT_B, "RUNNING"), (ACCT_C, "PENDING")] { + set.instances.push(StackInstance { + account: account.to_string(), + region: "us-east-1".to_string(), + stack_id: None, + status: "OUTDATED".to_string(), + detailed_status: "PENDING".to_string(), + status_reason: None, + parameter_overrides: BTreeMap::new(), + organizational_unit_id: None, + drift_status: "NOT_CHECKED".to_string(), + last_drift_check_timestamp: None, + last_operation_id: Some("cut-short".to_string()), + }); op.results.push(OperationResult { account: account.to_string(), region: "us-east-1".to_string(), @@ -5431,6 +5503,18 @@ mod tests { ) .await; assert_eq!(tag(&op, "Status"), "FAILED", "{op}"); + // Each instance matches its result: deploying failed, waiting cancelled. + let set = stored_set(&svc, "app"); + let detailed = |account: &str| { + set.instances + .iter() + .find(|i| i.account == account) + .unwrap() + .detailed_status + .clone() + }; + assert_eq!(detailed(ACCT_B), "FAILED"); + assert_eq!(detailed(ACCT_C), "CANCELLED"); // The stack set accepts new operations again. ok( &svc, From c95edc232c74c9f7e23f3a4360b579ccf7949fcd Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 21:56:09 -0300 Subject: [PATCH 09/13] fix(cloudformation): delete suspended-account instances; interrupted ops fail - Suspension skips deployments only; DeleteStackInstances still removes an instance in a suspended account, so its stack set can be emptied - Suspension is only considered for service-managed stack sets - An operation interrupted by a restart settles FAILED (or STOPPED) with a reason, regardless of failure tolerance --- .../src/stack_sets.rs | 80 +++++++++++++++++-- 1 file changed, 75 insertions(+), 5 deletions(-) diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 5fa693f3e..7baf4fd67 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -276,7 +276,15 @@ fn settle_interrupted_operations(set: &mut StackSet) { }; apply_to_instance(instance, &outcome); } - op.status = settled_status(op).to_string(); + // Whatever the tolerance, an operation that never got to finish did + // not succeed. + op.status = if op.status == "STOPPING" { + "STOPPED" + } else { + "FAILED" + } + .to_string(); + op.status_reason = Some(OPERATION_INTERRUPTED.to_string()); op.ended_at = Some(Utc::now()); } } @@ -1876,7 +1884,7 @@ impl CloudFormationService { )?; (targets, explicit_regions.clone()) } else { - let suspended = self.suspended_accounts(); + let suspended = self.suspended_accounts_for(&updated); let targets = updated .instances .iter() @@ -2133,7 +2141,12 @@ impl CloudFormationService { /// Organization member accounts that are not ACTIVE. Existing instances in /// them are skipped again rather than failed. - fn suspended_accounts(&self) -> BTreeSet { + fn suspended_accounts_for(&self, set: &StackSet) -> BTreeSet { + // Only service-managed stack sets deploy through the organization; + // a self-managed one targets accounts directly. + if set.permission_model != "SERVICE_MANAGED" { + return BTreeSet::new(); + } self.deps .organizations .read() @@ -2224,7 +2237,7 @@ impl CloudFormationService { "Only one of Accounts or DeploymentTargets can be specified", )); } - let suspended = self.suspended_accounts(); + let suspended = self.suspended_accounts_for(set); let mut targets = Vec::new(); let mut push = |instance: &StackInstance| { if !targets @@ -2685,7 +2698,9 @@ impl CloudFormationService { let mut overrides = None; let outcome = if let Some(reason) = abort { Outcome::Cancelled(reason.to_string()) - } else if target.suspended { + } else if target.suspended && !matches!(action, TargetAction::Delete { .. }) { + // A suspended account is skipped for deployments, but its + // instance can still be removed. Outcome::SkippedSuspended } else { let g = self.account_gate(&target.account, &target.region).await; @@ -5279,6 +5294,25 @@ mod tests { assert_eq!(tag(&op, "Status"), "SUCCEEDED", "{op}"); assert_eq!(skipped(&svc), "SKIPPED_SUSPENDED_ACCOUNT"); assert_eq!(queue_count(&svc, ACCT_B), 1); + + // The suspended account's instance still deletes, so the stack set + // can be emptied and deleted. + ok( + &svc, + "DeleteStackInstances", + &[ + ("StackSetName", "org"), + ( + "DeploymentTargets.OrganizationalUnitIds.member.1", + &workloads, + ), + ("Regions.member.1", "us-east-1"), + ("RetainStacks", "true"), + ], + ) + .await; + assert!(stored_set(&svc, "org").instances.is_empty()); + ok(&svc, "DeleteStackSet", &[("StackSetName", "org")]).await; } #[tokio::test] @@ -5449,6 +5483,42 @@ mod tests { assert_eq!(stored_set(&svc, "app").instances[0].status, "CURRENT"); } + #[tokio::test] + async fn an_operation_that_never_started_before_a_restart_is_not_a_success() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + let snapshot = stored_set(&svc, "app"); + let targets = vec![Target { + account: ACCT_B.to_string(), + region: "us-east-1".to_string(), + ou: None, + suspended: false, + }]; + svc.start_instance_operation( + ADMIN, + &snapshot, + &targets, + "never-ran", + "CREATE", + OperationPreferences { + failure_tolerance_count: Some(5), + ..OperationPreferences::default() + }, + None, + None, + ) + .unwrap(); + restore_stack_sets(&mut svc.state.write()); + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "app"), ("OperationId", "never-ran")], + ) + .await; + assert_eq!(tag(&op, "Status"), "FAILED", "{op}"); + assert!(op.contains(OPERATION_INTERRUPTED), "{op}"); + } + #[tokio::test] async fn operations_interrupted_by_a_restart_are_settled_on_load() { let svc = service(); From 8957a20ae3819a52b5368bbb316462280ffb45fc Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 22:03:24 -0300 Subject: [PATCH 10/13] fix(cloudformation): match CloudFormation IAM roles by ARN in drift checks The provisioner records a role's physical id as its ARN, but IAM keys roles by name, so drift detection reported every CloudFormation-created role as deleted. --- crates/fakecloud-cloudformation/src/extras.rs | 3 +- .../src/stack_sets.rs | 28 +++++++++++++++++++ 2 files changed, 30 insertions(+), 1 deletion(-) diff --git a/crates/fakecloud-cloudformation/src/extras.rs b/crates/fakecloud-cloudformation/src/extras.rs index c3be6951f..2a37030a6 100644 --- a/crates/fakecloud-cloudformation/src/extras.rs +++ b/crates/fakecloud-cloudformation/src/extras.rs @@ -606,7 +606,8 @@ impl CloudFormationService { .iam .read() .get(aid) - .map(|s| s.roles.contains_key(&resource.physical_id)) + // CloudFormation records a role by its ARN; IAM keys roles by name. + .map(|s| s.roles.values().any(|r| r.arn == resource.physical_id)) .unwrap_or(false), "AWS::DynamoDB::Table" => self .deps diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 7baf4fd67..2e8ddb9df 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -5651,6 +5651,34 @@ mod tests { assert_eq!(tag(&instance, "DetailedStatus"), "CANCELLED", "{instance}"); } + #[tokio::test] + async fn an_untouched_iam_role_is_in_sync() { + let template = "Resources:\n R:\n Type: AWS::IAM::Role\n Properties:\n RoleName:\n Fn::Sub: \"${AWS::StackName}-role\"\n AssumeRolePolicyDocument:\n Version: \"2012-10-17\"\n Statement: []\n"; + let svc = service(); + create_set(&svc, "roles", template).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "roles"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + let xml = ok(&svc, "DetectStackSetDrift", &[("StackSetName", "roles")]).await; + let op = ok( + &svc, + "DescribeStackSetOperation", + &[ + ("StackSetName", "roles"), + ("OperationId", &tag(&xml, "OperationId")), + ], + ) + .await; + assert_eq!(tag(&op, "DriftStatus"), "IN_SYNC", "{op}"); + } + struct Gate(&'static str); impl LambdaDelivery for Gate { From eb76cf26ae225dd6e4c8ef85ead8bbdd2c63f789 Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 22:12:48 -0300 Subject: [PATCH 11/13] fix(cloudformation): keep overrides for undeployed targets; drift edge cases - Targets skipped, cancelled or gated still take on the requested parameter overrides, so a later redeploy uses them - DetectStackSetDrift reports a reused OperationId as the declared InvalidOperationException - A drift check that could check no instance reports NOT_CHECKED --- .../src/stack_sets.rs | 145 +++++++++++++++--- 1 file changed, 126 insertions(+), 19 deletions(-) diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 2e8ddb9df..cdf004603 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -1228,6 +1228,28 @@ fn mark_instances_pending(set: &mut StackSet, targets: &[Target], op_id: &str, c } } +/// The overrides an instance carries after an update: explicit values, plus +/// `UsePreviousValue` entries carried over from the instance. `None` keeps the +/// instance's overrides as they are. +fn resolve_update_overrides( + specs: Option<&[OverrideSpec]>, + existing: Option<&StackInstance>, +) -> BTreeMap { + let previous = existing + .map(|i| i.parameter_overrides.clone()) + .unwrap_or_default(); + match specs { + None => previous, + Some(specs) => specs + .iter() + .filter_map(|s| match s { + OverrideSpec::Value(k, v) => Some((k.clone(), v.clone())), + OverrideSpec::UsePrevious(k) => previous.get(k).map(|v| (k.clone(), v.clone())), + }) + .collect(), + } +} + /// A PENDING result for every target an operation will act on. fn pending_results(targets: &[Target]) -> Vec { targets @@ -1716,6 +1738,22 @@ impl CloudFormationService { /// Reject a new operation on a stack set that already has one running, and /// a caller-supplied operation id that was used before. + /// `check_can_start_operation` for DetectStackSetDrift, which models no + /// OperationIdAlreadyExistsException: a reused id is an invalid operation. + fn check_can_start_drift(set: &StackSet, op_id: &str) -> Result<(), AwsServiceError> { + Self::check_can_start_operation(set, op_id).map_err(|e| { + if e.code() == "OperationIdAlreadyExistsException" { + aws_err( + StatusCode::BAD_REQUEST, + "InvalidOperationException", + e.message(), + ) + } else { + e + } + }) + } + fn check_can_start_operation(set: &StackSet, op_id: &str) -> Result<(), AwsServiceError> { if set .operations @@ -2720,6 +2758,18 @@ impl CloudFormationService { ) } }; + // A target that did not deploy still takes on the overrides it was + // asked for, so a later redeploy uses them. + if overrides.is_none() { + overrides = match &action { + TargetAction::Create { overrides } => Some(overrides.clone()), + TargetAction::Update { overrides } => Some(resolve_update_overrides( + overrides.as_deref(), + self.instance_stack(admin, set_id, &target).as_ref(), + )), + TargetAction::Delete { .. } => None, + }; + } if matches!(outcome, Outcome::Failed(_)) { let failures = region_failures.entry(target.region.clone()).or_default(); *failures += 1; @@ -2906,22 +2956,7 @@ impl CloudFormationService { } }, TargetAction::Update { overrides } => { - let previous = existing - .as_ref() - .map(|i| i.parameter_overrides.clone()) - .unwrap_or_default(); - let resolved = match overrides { - None => previous, - Some(specs) => specs - .iter() - .filter_map(|s| match s { - OverrideSpec::Value(k, v) => Some((k.clone(), v.clone())), - OverrideSpec::UsePrevious(k) => { - previous.get(k).map(|v| (k.clone(), v.clone())) - } - }) - .collect(), - }; + let resolved = resolve_update_overrides(overrides.as_deref(), existing.as_ref()); let Some((stack_id, _, _)) = live_stack else { // An instance that never got a stack (skipped, cancelled, // or its create failed) is deployed now. One whose stack @@ -3627,7 +3662,7 @@ impl CloudFormationService { .and_then(|s| s.stack_sets.get(&set_id)) .cloned() .ok_or_else(|| stack_set_not_found(&name))?; - Self::check_can_start_operation(&set, &op_id)?; + Self::check_can_start_drift(&set, &op_id)?; set }; @@ -3711,7 +3746,7 @@ impl CloudFormationService { .filter(|(_, _, s)| s == "UNKNOWN") .count(); let details = DriftDetectionDetails { - drift_status: if set.instances.is_empty() { + drift_status: if drifted + in_sync == 0 { "NOT_CHECKED".to_string() } else if drifted > 0 { "DRIFTED".to_string() @@ -3746,7 +3781,7 @@ impl CloudFormationService { // being checked. (DetectStackSetDrift does not model // StaleRequestException, and a drift result stays valid after an // operation that has already finished.) - Self::check_can_start_operation(stored, &op_id)?; + Self::check_can_start_drift(stored, &op_id)?; for instance in &mut stored.instances { if let Some((_, _, status)) = instance_drift .iter() @@ -5679,6 +5714,78 @@ mod tests { assert_eq!(tag(&op, "DriftStatus"), "IN_SYNC", "{op}"); } + #[tokio::test] + async fn a_target_that_does_not_deploy_keeps_its_requested_overrides() { + let svc = service(); + let (workloads, _) = seed_org(&svc); + svc.deps + .organizations + .write() + .as_mut() + .unwrap() + .close_account(ACCT_C) + .unwrap(); + ok(&svc, "ActivateOrganizationsAccess", &[]).await; + ok( + &svc, + "CreateStackSet", + &[ + ("StackSetName", "org"), + ("TemplateBody", QUEUE_TEMPLATE), + ("PermissionModel", "SERVICE_MANAGED"), + ], + ) + .await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "org"), + ( + "DeploymentTargets.OrganizationalUnitIds.member.1", + &workloads, + ), + ("Regions.member.1", "us-east-1"), + ("ParameterOverrides.member.1.ParameterKey", "Env"), + ("ParameterOverrides.member.1.ParameterValue", "prod"), + ], + ) + .await; + let skipped = stored_set(&svc, "org") + .instances + .into_iter() + .find(|i| i.account == ACCT_C) + .unwrap(); + assert_eq!(skipped.detailed_status, "SKIPPED_SUSPENDED_ACCOUNT"); + assert_eq!( + skipped.parameter_overrides.get("Env").map(String::as_str), + Some("prod") + ); + } + + #[tokio::test] + async fn drift_detection_reports_model_declared_errors_and_unchecked_sets() { + let broken = "Resources:\n Q:\n Type: AWS::SQS::Queue\n Properties:\n QueueName:\n Fn::ImportValue: missing-export\n"; + let svc = service(); + create_set(&svc, "nostacks", broken).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "nostacks"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + let params = [("StackSetName", "nostacks"), ("OperationId", "drift-1")]; + ok(&svc, "DetectStackSetDrift", ¶ms).await; + let described = ok(&svc, "DescribeStackSet", &[("StackSetName", "nostacks")]).await; + assert_eq!(tag(&described, "DriftStatus"), "NOT_CHECKED", "{described}"); + let e = err(&svc, "DetectStackSetDrift", ¶ms).await; + assert_eq!(e.code(), "InvalidOperationException"); + } + struct Gate(&'static str); impl LambdaDelivery for Gate { From 16c168e0f44744f2a428e998794144cfd6114b8c Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 22:25:14 -0300 Subject: [PATCH 12/13] fix(cloudformation): wait on background stacks before the next target A stack provisioning in the background (custom resources) returned RUNNING and never counted against the failure tolerance, so every remaining target deployed even after it failed. The operation now waits for such a stack to settle before moving on. --- .../src/stack_sets.rs | 89 +++++++++++++++++++ 1 file changed, 89 insertions(+) diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index cdf004603..11e8a512e 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -39,6 +39,8 @@ const DEFAULT_EXECUTION_ROLE: &str = "AWSCloudFormationStackSetExecutionRole"; /// ImportStacksToStackSet accepts at most this many stacks per call. const MAX_IMPORT_STACKS: usize = 10; const DEFAULT_PAGE_SIZE: usize = 100; +/// How long an operation waits on one background-provisioning stack. +const STACK_WAIT_LIMIT: std::time::Duration = std::time::Duration::from_secs(3600); const TOLERANCE_EXCEEDED: &str = "Cancelled since failure tolerance has exceeded"; const OPERATION_STOPPED: &str = "Cancelled since the operation was stopped"; @@ -2749,6 +2751,15 @@ impl CloudFormationService { let (outcome, id, applied) = self .apply_target(request_id, admin, set_id, spec, &target, &action) .await; + // A stack still provisioning in the background (custom + // resources) is waited on, so its failure counts against + // the tolerance before the next target deploys. + let outcome = match (&outcome, &id) { + (Outcome::Running, Some(stack_id)) => { + self.await_stack(&target.account, stack_id).await + } + _ => outcome, + }; stack_id = id; overrides = applied; outcome @@ -2793,6 +2804,30 @@ impl CloudFormationService { } } + /// Wait for a stack that provisions in the background to reach a terminal + /// status. Gives up after `STACK_WAIT_LIMIT`, leaving the target RUNNING + /// for a later read of the stack set to settle. + async fn await_stack(&self, account: &str, stack_id: &str) -> Outcome { + let deadline = tokio::time::Instant::now() + STACK_WAIT_LIMIT; + loop { + let outcome = match self.stack_status(account, stack_id) { + Some((_, status, reason)) => { + let action = if status.starts_with("UPDATE") { + "UPDATE" + } else { + "CREATE" + }; + stack_outcome(action, &status, reason.as_deref()) + } + None => Outcome::Failed(format!("Stack [{stack_id}] does not exist")), + }; + if outcome != Outcome::Running || tokio::time::Instant::now() >= deadline { + return outcome; + } + tokio::time::sleep(std::time::Duration::from_millis(250)).await; + } + } + /// Mark a target's result RUNNING before it deploys. Returns false when /// the operation has been stopped, in which case the target must not run. fn claim_target(&self, admin: &str, set_id: &str, op_id: &str, target: &Target) -> bool { @@ -5786,6 +5821,60 @@ mod tests { assert_eq!(e.code(), "InvalidOperationException"); } + struct FailingLambda; + + impl LambdaDelivery for FailingLambda { + fn invoke_lambda( + &self, + _function_arn: &str, + _payload: &str, + ) -> std::pin::Pin, String>> + Send>> + { + Box::pin(async { Err("custom resource handler failed".to_string()) }) + } + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_background_stack_failure_counts_against_the_tolerance() { + let mut d = deps(); + d.delivery = Arc::new(DeliveryBus::new().with_lambda(Arc::new(FailingLambda))); + let svc = service_with(d); + // A custom resource provisions in the background on the server. + let template = "Resources:\n C:\n Type: Custom::Thing\n Properties:\n ServiceToken: arn:aws:lambda:us-east-1:111111111111:function:handler\n"; + create_set(&svc, "custom", template).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "custom"), + ("Accounts.member.1", ACCT_B), + ("Accounts.member.2", ACCT_C), + ("Regions.member.1", "us-east-1"), + ("OperationId", "op-custom"), + ], + ) + .await; + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(30); + let op = loop { + let op = ok( + &svc, + "DescribeStackSetOperation", + &[("StackSetName", "custom"), ("OperationId", "op-custom")], + ) + .await; + if tag(&op, "Status") != "RUNNING" { + break op; + } + assert!(std::time::Instant::now() < deadline, "{op}"); + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + }; + assert_eq!(tag(&op, "Status"), "FAILED", "{op}"); + let set = stored_set(&svc, "custom"); + let second = set.instances.iter().find(|i| i.account == ACCT_C).unwrap(); + assert_eq!(second.detailed_status, "CANCELLED", "{set:?}"); + assert!(second.stack_id.is_none()); + } + struct Gate(&'static str); impl LambdaDelivery for Gate { From f2996cb504fcb8b62b582f4f48a0817365590c38 Mon Sep 17 00:00:00 2001 From: Lucas Vieira Date: Mon, 14 Sep 2026 22:34:33 -0300 Subject: [PATCH 13/13] fix(cloudformation): record a provisioning stack before waiting; INOPERABLE on failed delete - The instance points at its stack while the stack provisions in the background, so a restart during the wait does not orphan it - A stack that cannot be deleted leaves its instance INOPERABLE --- .../src/stack_sets.rs | 65 +++++++++++++++++++ .../content/docs/services/cloudformation.md | 2 +- 2 files changed, 66 insertions(+), 1 deletion(-) diff --git a/crates/fakecloud-cloudformation/src/stack_sets.rs b/crates/fakecloud-cloudformation/src/stack_sets.rs index 11e8a512e..d1021189f 100644 --- a/crates/fakecloud-cloudformation/src/stack_sets.rs +++ b/crates/fakecloud-cloudformation/src/stack_sets.rs @@ -2756,6 +2756,19 @@ impl CloudFormationService { // the tolerance before the next target deploys. let outcome = match (&outcome, &id) { (Outcome::Running, Some(stack_id)) => { + // Record the stack first, so the instance points + // at it while it provisions and across a restart. + self.record_outcome( + admin, + set_id, + op_id, + &target, + &action, + &Outcome::Running, + None, + id.clone(), + applied.clone(), + ); self.await_stack(&target.account, stack_id).await } _ => outcome, @@ -3140,6 +3153,11 @@ impl CloudFormationService { } else if !matches!(outcome, Outcome::Cancelled(_)) { let instance = &mut set.instances[idx]; apply_to_instance(instance, outcome); + // A stack that could not be deleted leaves the instance + // INOPERABLE, as in AWS. + if matches!(outcome, Outcome::Failed(_)) { + instance.status = "INOPERABLE".to_string(); + } instance.last_operation_id = Some(op_id.to_string()); } } @@ -5821,6 +5839,51 @@ mod tests { assert_eq!(e.code(), "InvalidOperationException"); } + #[tokio::test] + async fn a_stack_that_cannot_be_deleted_leaves_its_instance_inoperable() { + let svc = service(); + create_set(&svc, "app", QUEUE_TEMPLATE).await; + ok( + &svc, + "CreateStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ], + ) + .await; + let stack_id = stored_set(&svc, "app").instances[0] + .stack_id + .clone() + .unwrap(); + call_as( + &svc, + ACCT_B, + "UpdateTerminationProtection", + &[ + ("StackName", &stack_id), + ("EnableTerminationProtection", "true"), + ], + ) + .await + .unwrap(); + ok( + &svc, + "DeleteStackInstances", + &[ + ("StackSetName", "app"), + ("Accounts.member.1", ACCT_B), + ("Regions.member.1", "us-east-1"), + ("RetainStacks", "false"), + ], + ) + .await; + let instance = stored_set(&svc, "app").instances[0].clone(); + assert_eq!(instance.status, "INOPERABLE"); + assert_eq!(instance.detailed_status, "FAILED"); + } + struct FailingLambda; impl LambdaDelivery for FailingLambda { @@ -5870,6 +5933,8 @@ mod tests { }; assert_eq!(tag(&op, "Status"), "FAILED", "{op}"); let set = stored_set(&svc, "custom"); + let first = set.instances.iter().find(|i| i.account == ACCT_B).unwrap(); + assert!(first.stack_id.is_some(), "{set:?}"); let second = set.instances.iter().find(|i| i.account == ACCT_C).unwrap(); assert_eq!(second.detailed_status, "CANCELLED", "{set:?}"); assert!(second.stack_id.is_none()); diff --git a/website/content/docs/services/cloudformation.md b/website/content/docs/services/cloudformation.md index addf4ae0e..fdbae6cca 100644 --- a/website/content/docs/services/cloudformation.md +++ b/website/content/docs/services/cloudformation.md @@ -52,7 +52,7 @@ Stack instances are **real stacks**. `CreateStackInstances` creates a `StackSet- - **Parameters** - the stack set's `Parameters` apply to every instance; `ParameterOverrides` on `CreateStackInstances` / `UpdateStackInstances` override them per instance. `UsePreviousValue` keeps an override, and leaving a parameter out of the list reverts it to the stack set's value. Overriding a parameter the template does not declare is a `ValidationError`. - **Updates** - `UpdateStackSet` stores the new template, parameters, capabilities and tags, then updates every instance's stack (or only the instances named by `Accounts`/`DeploymentTargets` + `Regions`, leaving the rest `OUTDATED`). `UpdateStackInstances` redeploys just the named instances. -- **Deletes** - `DeleteStackInstances` deletes each instance's stack and its resources, or with `RetainStacks=true` leaves the stack in place and only removes the instance. `DeleteStackSet` refuses a stack set that still has instances (`StackSetNotEmptyException`); a deleted stack set stays listable as `DELETED` and describable by its ID, and its name is free for reuse. +- **Deletes** - `DeleteStackInstances` deletes each instance's stack and its resources, or with `RetainStacks=true` leaves the stack in place and only removes the instance. A stack that cannot be deleted (termination protection, say) leaves its instance `INOPERABLE`. `DeleteStackSet` refuses a stack set that still has instances (`StackSetNotEmptyException`); a deleted stack set stays listable as `DELETED` and describable by its ID, and its name is free for reuse. - **Operations** - as in AWS, a mutating call returns its `OperationId` straight away and the deployment runs in the background; poll `DescribeStackSetOperation` until it leaves `RUNNING`. Every operation records a per-target result (`Account`, `Region`, `Status`, `StatusReason`, `AccountGateResult`). Targets deploy in `RegionOrder` then request order. A failing target counts against `FailureToleranceCount` / `FailureTolerancePercentage` per region; once the tolerance is exceeded the remaining targets are `CANCELLED` and the operation ends `FAILED`. Stacks that provision asynchronously (templates with custom resources) leave the operation `RUNNING` until they settle; `StopStackSetOperation` cancels the targets that have not started. A second operation while one is running is `OperationInProgressException`, and a reused `OperationId` is `OperationIdAlreadyExistsException`. An operation cut short by a restart is settled as `FAILED` when state is loaded, so it does not block the stack set. - **Account gate** - when a target account has a Lambda named `AWSCloudFormationStackSetAccountGate`, it is invoked before deploying and the deployment only proceeds if it returns `{"Status": "SUCCEEDED"}`. Without the function the gate is `SKIPPED`. - **Service-managed** - `PermissionModel=SERVICE_MANAGED` needs an organization with StackSets trusted access (`ActivateOrganizationsAccess`, or Organizations `EnableAWSServiceAccess` for `member.org.stacksets.cloudformation.amazonaws.com`). `DeploymentTargets.OrganizationalUnitIds` resolve to the accounts in those OUs and every OU nested below them, never the management account; `AccountFilterType` (`INTERSECTION`, `DIFFERENCE`, `UNION`, `NONE`) combines them with `DeploymentTargets.Accounts` or an `AccountsUrl` file in S3. Suspended accounts are recorded as `SKIPPED_SUSPENDED_ACCOUNT`. `CallAs=DELEGATED_ADMIN` works from an account registered as a StackSets delegated administrator and acts on the management account's stack sets.