diff --git a/src/anvil/providers/aws/tasks/remove_cloudformation_stack.py b/src/anvil/providers/aws/tasks/remove_cloudformation_stack.py new file mode 100644 index 0000000..3a04c85 --- /dev/null +++ b/src/anvil/providers/aws/tasks/remove_cloudformation_stack.py @@ -0,0 +1,261 @@ +from __future__ import annotations + +import logging + +from botocore.exceptions import ClientError, WaiterError + +from anvil.actions import ActionRecorder +from anvil.providers.tasks._task_helpers import metadata_bool +from anvil.task_errors import TaskExecutionError + +__LOGGER__ = logging.getLogger(__name__) +TASK_SCOPE = "target" + +# Bounded wait for DELETE_COMPLETE. 30 * 20s = 10 minutes. +_DELETE_WAITER_CONFIG = {"Delay": 20, "MaxAttempts": 30} + + +def _force_retain_stuck_resources(metadata: dict[str, object]) -> bool: + """Return whether to retry a failed delete by retaining stuck resources.""" + + return metadata_bool( + task_name="remove_cloudformation_stack", + metadata=metadata, + key="force_retain_stuck_resources", + default=False, + ) + + +def _delete_failed_logical_ids(cfn_client, stack_id: str) -> list[str]: + """Return the logical IDs of resources currently stuck in DELETE_FAILED.""" + + resources = cfn_client.describe_stack_resources(StackName=stack_id)[ + "StackResources" + ] + return [ + resource["LogicalResourceId"] + for resource in resources + if resource.get("ResourceStatus") == "DELETE_FAILED" + ] + + +def _selected_stack_ids( + metadata: dict[str, object], *, execution_target_id: str +) -> list[str]: + """Return this account's exact stack identifiers from a shared account map. + + ``metadata.stack_ids_by_account`` is one mapping shared by every account a + target resolves; each invocation looks up only its own + ``execution_target_id`` entry, so one target/one metadata block can carry + a different, explicit stack selection per account without matching or + discovery. An account present in the target's ``include`` list but absent + from the map is treated as "nothing to do" for that account. + """ + + stack_map = metadata.get("stack_ids_by_account") + if not isinstance(stack_map, dict) or not stack_map: + raise RuntimeError( + "remove_cloudformation_stack requires metadata.stack_ids_by_account " + "to be a non-empty mapping of account ID to an array of stack IDs" + ) + + account_stack_ids = stack_map.get(execution_target_id, []) + if not isinstance(account_stack_ids, list) or not all( + isinstance(item, str) and item.strip() for item in account_stack_ids + ): + raise RuntimeError( + "remove_cloudformation_stack requires " + f"metadata.stack_ids_by_account['{execution_target_id}'] to be an " + "array of non-empty strings" + ) + return [item.strip() for item in account_stack_ids] + + +def _describe_stack(cfn_client, stack_id: str) -> dict[str, object] | None: + """Return one stack's description, or None if it no longer exists. + + ``describe_stacks`` keeps returning a deleted stack's record (with + ``StackStatus: DELETE_COMPLETE``) for a period after deletion when + queried by exact ARN -- it only raises "does not exist" once that + record ages out. A stack already in ``DELETE_COMPLETE`` is treated the + same as one that raised "does not exist": there's nothing left to do. + """ + + try: + stacks = cfn_client.describe_stacks(StackName=stack_id)["Stacks"] + except ClientError as error: + if "does not exist" in error.response["Error"].get("Message", ""): + return None + raise + stack = stacks[0] if stacks else None + if stack is not None and stack.get("StackStatus") == "DELETE_COMPLETE": + return None + return stack + + +def run( + *, + provider: str, + execution_target_id: str, + execution_target_name: str, + execution_target_type: str, + region: str, + session, + dry_run: bool, + metadata: dict[str, object], + dependency_data: dict[str, object], + actions: ActionRecorder, +) -> dict[str, object]: + """Delete one or more explicitly named CloudFormation stacks. + + This target-scoped AWS task deletes the exact stacks assigned to the + current account in ``metadata.stack_ids_by_account`` -- full stack ARNs + are recommended over bare names, since an ARN can never resolve to the + wrong stack even if a name is reused later. The same mapping is shared + across every account a target resolves; each invocation only reads its + own ``execution_target_id`` entry, so one target can carry a distinct, + explicit stack selection per account without any name matching or + discovery. It is meant for decommissioning specific, already identified + stacks -- for example, an orphaned or superseded StackSet-managed stack + instance whose resources block a replacement deployment. In dry-run mode + it reports the current status of each stack without deleting anything. + When not in dry-run mode, it waits (bounded) for each deletion to reach + ``DELETE_COMPLETE``. + + Metadata: + stack_ids_by_account: Required non-empty mapping of AWS account ID to + an array of exact stack names or ARNs to remove in that account. + An account missing from the mapping is a no-op. + force_retain_stuck_resources: Optional boolean, defaults to false. + When the initial delete fails, retries once with + ``RetainResources`` set to whatever logical IDs are currently + reported as ``DELETE_FAILED`` -- for example, a custom resource + this account doesn't own or control. This makes the *stack* + delete unconditionally, but any retained resource is left + behind, still existing, no longer managed by any stack. Use this + only when leaving that specific resource behind is acceptable. + + Args: + provider: Provider name for the current execution target. + execution_target_id: Target AWS account ID. + execution_target_name: Friendly name for the target account. + execution_target_type: Provider target type. + region: Current AWS region. + session: Boto3 session scoped to the current region. + dry_run: Whether execution is running in dry-run mode. + metadata: Task metadata containing the exact stack selectors. + dependency_data: Runtime data selected from declared task dependencies. + actions: Action recorder provided by the engine. + + Returns: + A payload with ``removed_stacks``, ``skipped_stacks``, and + ``failed_stacks`` (empty in dry-run mode, since nothing is removed). + Each removed stack includes ``retained_resources``: the logical IDs + (if any) left behind because of ``force_retain_stuck_resources``. + + Raises: + RuntimeError: If ``metadata.stack_ids_by_account`` is missing or invalid. + botocore.exceptions.ClientError: If an unexpected AWS API error occurs. + TaskExecutionError: If one or more selected stacks fail to be removed. + """ + stack_ids = _selected_stack_ids(metadata, execution_target_id=execution_target_id) + force_retain = _force_retain_stuck_resources(metadata) + cfn_client = session.client("cloudformation") + + planned_count = 0 + removed: list[dict[str, object]] = [] + skipped: list[dict[str, object]] = [] + failed: list[dict[str, object]] = [] + + for stack_id in stack_ids: + try: + stack = _describe_stack(cfn_client, stack_id) + if stack is None: + skipped.append({"stack_id": stack_id, "reason": "not_found"}) + __LOGGER__.info(f"Stack '{stack_id}' does not exist; skipping") + continue + + stack_name = stack["StackName"] + stack_status = stack["StackStatus"] + + if dry_run: + planned_count += 1 + __LOGGER__.info( + f"(dry-run) Would delete stack '{stack_name}' " + f"(currently {stack_status})" + ) + continue + + stack_result: dict[str, object] = { + "stack_id": stack_id, + "stack_name": stack_name, + "stack_status": stack_status, + } + + cfn_client.delete_stack(StackName=stack_id) + __LOGGER__.info(f"Delete initiated for stack '{stack_name}'") + retained_resources: list[str] = [] + try: + cfn_client.get_waiter("stack_delete_complete").wait( + StackName=stack_id, WaiterConfig=_DELETE_WAITER_CONFIG + ) + except WaiterError as error: + if not force_retain: + raise RuntimeError( + f"Timed out or failed waiting for '{stack_name}' to " + f"delete: {error}" + ) from error + + retained_resources = _delete_failed_logical_ids(cfn_client, stack_id) + if not retained_resources: + raise RuntimeError( + f"'{stack_name}' failed to delete and no DELETE_FAILED " + f"resources were found to retain: {error}" + ) from error + + __LOGGER__.warning( + f"'{stack_name}' failed to delete; retrying while retaining " + f"stuck resource(s): {retained_resources}" + ) + cfn_client.delete_stack( + StackName=stack_id, RetainResources=retained_resources + ) + try: + cfn_client.get_waiter("stack_delete_complete").wait( + StackName=stack_id, WaiterConfig=_DELETE_WAITER_CONFIG + ) + except WaiterError as retry_error: + raise RuntimeError( + f"'{stack_name}' still failed to delete after retaining " + f"{retained_resources}: {retry_error}" + ) from retry_error + + stack_result["retained_resources"] = retained_resources + removed.append(stack_result) + if retained_resources: + __LOGGER__.warning( + f"Deleted stack '{stack_name}', retaining: {retained_resources}" + ) + else: + __LOGGER__.info(f"Deleted stack '{stack_name}'") + except (ClientError, RuntimeError) as error: + failed.append({"stack_id": stack_id, "error": str(error)}) + __LOGGER__.warning(f"Failed to remove stack '{stack_id}': {error}") + + result: dict[str, object] = { + "removed_stacks": removed, + "skipped_stacks": skipped, + "failed_stacks": failed, + } + if dry_run: + actions.record(f"(dry-run) Would remove {planned_count} stack(s)") + else: + actions.record(f"Removed {len(removed)} stack(s)") + + if failed: + raise TaskExecutionError( + f"remove_cloudformation_stack failed to remove {len(failed)} of " + f"{len(stack_ids)} selected stack(s)", + partial_result=result, + ) + return result diff --git a/src/anvil/providers/aws/tasks/remove_iam_role.py b/src/anvil/providers/aws/tasks/remove_iam_role.py new file mode 100644 index 0000000..591f911 --- /dev/null +++ b/src/anvil/providers/aws/tasks/remove_iam_role.py @@ -0,0 +1,252 @@ +from __future__ import annotations + +import logging +from typing import Any + +from botocore.exceptions import ClientError + +from anvil.actions import ActionRecorder +from anvil.providers.tasks._task_helpers import metadata_string_array +from anvil.task_errors import TaskExecutionError + +__LOGGER__ = logging.getLogger(__name__) +TASK_SCOPE = "target" + + +def _list_paginated_role_resources( + iam_client, *, operation_name: str, result_key: str, role_name: str +) -> list[Any]: + """Return every page of one IAM role resource collection.""" + + try: + paginator = iam_client.get_paginator(operation_name) + resources: list[Any] = [] + for page in paginator.paginate(RoleName=role_name): + resources.extend(page.get(result_key, [])) + return resources + except ClientError as error: + if error.response["Error"]["Code"] == "NoSuchEntity": + return [] + raise + + +def cleanup_role_resources( + iam_client, role_name: str, dry_run: bool, actions: ActionRecorder +) -> int: + """Remove or plan removal of resources attached to one IAM role. + + Returns: + The number of attached resources discovered for removal. + """ + + resource_count = 0 + + # Instance Profiles + instance_profiles = _list_paginated_role_resources( + iam_client, + operation_name="list_instance_profiles_for_role", + result_key="InstanceProfiles", + role_name=role_name, + ) + for profile in instance_profiles: + resource_count += 1 + profile_name = profile["InstanceProfileName"] + if dry_run: + __LOGGER__.debug( + f"(dry-run) Would remove role from instance profile: {profile_name}" + ) + else: + iam_client.remove_role_from_instance_profile( + InstanceProfileName=profile_name, RoleName=role_name + ) + __LOGGER__.debug(f"Removed role from instance profile: {profile_name}") + + # Attached Policies + attached_policies = _list_paginated_role_resources( + iam_client, + operation_name="list_attached_role_policies", + result_key="AttachedPolicies", + role_name=role_name, + ) + for policy in attached_policies: + resource_count += 1 + arn = policy["PolicyArn"] + if dry_run: + __LOGGER__.debug(f"(dry-run) Would detach policy: {arn}") + else: + iam_client.detach_role_policy(RoleName=role_name, PolicyArn=arn) + __LOGGER__.debug(f"Detached policy: {arn}") + + # Inline Policies + inline_policy_names = _list_paginated_role_resources( + iam_client, + operation_name="list_role_policies", + result_key="PolicyNames", + role_name=role_name, + ) + for name in inline_policy_names: + resource_count += 1 + if dry_run: + __LOGGER__.debug(f"(dry-run) Would delete inline policy: {name}") + else: + iam_client.delete_role_policy(RoleName=role_name, PolicyName=name) + __LOGGER__.debug(f"Deleted inline policy: {name}") + + # Tags + tags = _list_paginated_role_resources( + iam_client, + operation_name="list_role_tags", + result_key="Tags", + role_name=role_name, + ) + tag_keys = [tag["Key"] for tag in tags] + if tag_keys: + resource_count += len(tag_keys) + if dry_run: + __LOGGER__.debug(f"(dry-run) Would remove tags: {tag_keys}") + else: + iam_client.untag_role(RoleName=role_name, TagKeys=tag_keys) + __LOGGER__.debug(f"Removed tags: {tag_keys}") + + # Permissions Boundary + try: + role_detail = iam_client.get_role(RoleName=role_name)["Role"] + except ClientError as error: + if error.response["Error"]["Code"] == "NoSuchEntity": + role_detail = {} + else: + raise + + if role_detail.get("PermissionsBoundary"): + resource_count += 1 + if dry_run: + __LOGGER__.debug("(dry-run) Would delete permissions boundary") + else: + iam_client.delete_role_permissions_boundary(RoleName=role_name) + __LOGGER__.debug("Deleted permissions boundary") + + return resource_count + + +def _selected_roles(metadata: dict[str, object]) -> list[str]: + """Return the required, normalized IAM role selectors.""" + + return metadata_string_array( + task_name="remove_iam_role", metadata=metadata, key="roles", required=True + ) + + +def _role_exists(iam_client, role_name: str) -> bool: + """Return whether an IAM role currently exists.""" + + try: + iam_client.get_role(RoleName=role_name) + except ClientError as error: + if error.response["Error"]["Code"] == "NoSuchEntity": + return False + raise + return True + + +def run( + *, + provider: str, + execution_target_id: str, + execution_target_name: str, + execution_target_type: str, + region: str, + session, + dry_run: bool, + metadata: dict[str, object], + dependency_data: dict[str, object], + actions: ActionRecorder, +) -> dict[str, object]: + """Remove selected IAM roles after cleaning their attached resources. + + This target-scoped AWS task runs once per resolved account and removes + instance profile associations, attached managed policies, inline + policies, tags, and any permissions boundary, then deletes each IAM + role. In dry-run mode it reports planned deletions without mutating IAM. + + Metadata: + roles: Required non-empty array of IAM role names to remove. + + Args: + provider: Provider name for the current execution target. + execution_target_id: Target AWS account ID. + execution_target_name: Friendly name for the target account. + execution_target_type: Provider target type. + region: Current AWS region. + session: Boto3 session scoped to the current region. + dry_run: Whether execution is running in dry-run mode. + metadata: Task metadata containing IAM role selectors. + dependency_data: Runtime data selected from declared task dependencies. + actions: Action recorder provided by the engine. + + Returns: + A payload containing planned, removed, skipped, and failed IAM roles + plus discovered attached-resource counts. + + Raises: + RuntimeError: If ``metadata.roles`` is missing or invalid. + botocore.exceptions.ClientError: If an unexpected AWS API error occurs. + TaskExecutionError: If one or more selected roles fail to be removed. + """ + role_names = _selected_roles(metadata) + iam_client = session.client("iam") + planned: list[dict[str, object]] = [] + removed: list[dict[str, object]] = [] + skipped: list[dict[str, object]] = [] + failed: list[dict[str, object]] = [] + + for role_name in role_names: + try: + if not _role_exists(iam_client, role_name): + skipped.append({"role_name": role_name, "reason": "not_found"}) + __LOGGER__.info(f"IAM role '{role_name}' does not exist; skipping") + continue + + resource_count = cleanup_role_resources( + iam_client=iam_client, + role_name=role_name, + dry_run=dry_run, + actions=actions, + ) + role_result: dict[str, object] = { + "role_name": role_name, + "attached_resource_count": resource_count, + } + if dry_run: + planned.append(role_result) + __LOGGER__.info(f"(dry-run) Would remove IAM role '{role_name}'") + else: + iam_client.delete_role(RoleName=role_name) + removed.append(role_result) + __LOGGER__.info(f"Removed IAM role '{role_name}'") + except ClientError as error: + failed.append({"role_name": role_name, "error": str(error)}) + __LOGGER__.warning(f"Failed to remove IAM role '{role_name}': {error}") + + result: dict[str, object] = { + "selected_count": len(role_names), + "planned_count": len(planned), + "removed_count": len(removed), + "skipped_count": len(skipped), + "failed_count": len(failed), + "planned_roles": planned, + "removed_roles": removed, + "skipped_roles": skipped, + "failed_roles": failed, + } + if dry_run: + actions.record(f"(dry-run) Would remove {len(planned)} IAM role(s)") + else: + actions.record(f"Removed {len(removed)} IAM role(s)") + + if failed: + raise TaskExecutionError( + f"remove_iam_role failed to remove {len(failed)} of " + f"{len(role_names)} selected role(s)", + partial_result=result, + ) + return result