@@ -471,6 +471,130 @@ async def rollback(self, job_id: str, payload: Dict[str, Any]) -> Dict[str, Any]
471471 self .logger .error (f"Rollback failed: { e } " )
472472 return {"status" : "failed" , "error" : str (e )}
473473
474+ @xray_recorder .capture ("destroy_deployment" ) # type: ignore[reportCallIssue]
475+ async def destroy_deployment (
476+ self ,
477+ job_id : str ,
478+ aws_config : Dict [str , Any ],
479+ ) -> Dict [str , Any ]:
480+ """Destroy a job's deployed infrastructure via `terraform destroy`,
481+ reusing the persisted remote state in TF_STATE_BUCKET (see
482+ _write_remote_state_backend). Falls back to reporting "no state" for
483+ deployments made before remote state was enabled, rather than
484+ pretending to succeed. On failure, describes what's still live in
485+ AWS (mirrors rollback()/the /monitoring endpoint) so the caller can
486+ tell the user precisely what to clean up manually -- per the
487+ feature/destroy-deployment design decisions.
488+ """
489+ workspace_dir = self ._workspace_dir (job_id )
490+ tf_dir = workspace_dir / "terraform"
491+ tf_dir .mkdir (parents = True , exist_ok = True )
492+ self ._write_remote_state_backend (tf_dir , job_id = job_id )
493+
494+ if not (tf_dir / "backend.tf" ).exists ():
495+ self .logger .warning (f"[{ job_id } ] Destroy: TF_STATE_BUCKET not configured, cannot recover state" )
496+ return {
497+ "status" : "no_state" ,
498+ "job_id" : job_id ,
499+ "message" : (
500+ "TF_STATE_BUCKET is not configured; this deployment has no "
501+ "recoverable Terraform state. Manual AWS cleanup required."
502+ ),
503+ "remaining_resources" : await self ._describe_remaining_resources (job_id , aws_config ),
504+ }
505+
506+ tf_runner = TerraformRunner (tf_dir )
507+
508+ if not tf_runner .init ():
509+ self .logger .error (f"[{ job_id } ] Destroy: terraform init failed" )
510+ return {
511+ "status" : "failed" ,
512+ "job_id" : job_id ,
513+ "error" : "terraform_init_failed" ,
514+ "message" : "Could not initialize Terraform with the remote state backend." ,
515+ "remaining_resources" : await self ._describe_remaining_resources (job_id , aws_config ),
516+ }
517+
518+ # A job with no prior real `apply` (or one applied before
519+ # TF_STATE_BUCKET existed) has no state key in S3 -- init() still
520+ # succeeds against an empty backend, so check for an actual state
521+ # explicitly rather than trusting init() alone.
522+ state_output = tf_runner .output ()
523+ if not state_output :
524+ self .logger .info (f"[{ job_id } ] Destroy: no Terraform state found in remote backend" )
525+ return {
526+ "status" : "no_state" ,
527+ "job_id" : job_id ,
528+ "message" : (
529+ "No Terraform state found for this job (deployed before "
530+ "remote state was enabled, or state was lost). Nothing to "
531+ "destroy via Terraform -- check AWS directly for leftover "
532+ "resources."
533+ ),
534+ "remaining_resources" : await self ._describe_remaining_resources (job_id , aws_config ),
535+ }
536+
537+ try :
538+ destroyed = tf_runner .destroy ()
539+ except Exception as exc :
540+ self .logger .error (f"[{ job_id } ] terraform destroy failed after retries: { exc } " )
541+ return {
542+ "status" : "partial_failure" ,
543+ "job_id" : job_id ,
544+ "error" : str (exc ),
545+ "remaining_resources" : await self ._describe_remaining_resources (job_id , aws_config ),
546+ }
547+
548+ if not destroyed :
549+ return {
550+ "status" : "partial_failure" ,
551+ "job_id" : job_id ,
552+ "error" : "terraform destroy did not report success" ,
553+ "remaining_resources" : await self ._describe_remaining_resources (job_id , aws_config ),
554+ }
555+
556+ self .logger .info (f"[{ job_id } ] Destroy successful" )
557+ return {"status" : "success" , "job_id" : job_id }
558+
559+ async def _describe_remaining_resources (
560+ self , job_id : str , aws_config : Dict [str , Any ]
561+ ) -> Dict [str , Any ]:
562+ """Best-effort snapshot of what's still live in AWS after a failed,
563+ partial, or unrecoverable-state destroy -- so the caller can tell the
564+ user precisely what to clean up manually and where (mirrors the read
565+ pattern already used by rollback() and the /monitoring endpoint)."""
566+ region = aws_config .get ("region" ) or "us-east-1"
567+ cluster = aws_config .get ("ecs_cluster" )
568+ service_name = aws_config .get ("service_name" )
569+ remaining : Dict [str , Any ] = {
570+ "ecs_service" : None ,
571+ "target_groups" : [],
572+ "error" : None ,
573+ }
574+ if not cluster or not service_name :
575+ remaining ["error" ] = "no ecs_cluster/service_name available to check"
576+ return remaining
577+ try :
578+ aws = AWSClient (region = region )
579+ ecs = aws .ecs ()
580+ desc = ecs .describe_services (cluster = cluster , services = [service_name ])
581+ services = desc .get ("services" ) or []
582+ if services and services [0 ].get ("status" ) != "INACTIVE" :
583+ svc = services [0 ]
584+ remaining ["ecs_service" ] = {
585+ "cluster" : cluster ,
586+ "service_name" : service_name ,
587+ "status" : svc .get ("status" ),
588+ "running_count" : svc .get ("runningCount" ),
589+ }
590+ elbv2 = aws .session .client ("elbv2" , region_name = region , config = RETRY_CONFIG )
591+ for lb in svc .get ("loadBalancers" , []):
592+ tg_arn = lb .get ("targetGroupArn" )
593+ if tg_arn :
594+ remaining ["target_groups" ].append (tg_arn )
595+ except Exception as exc : # noqa: BLE001 - best-effort diagnostic, never raise
596+ remaining ["error" ] = str (exc )
597+ return remaining
474598 @xray_recorder .capture ("rollback_deployment" ) # type: ignore[reportCallIssue]
475599 async def rollback_deployment (
476600 self ,
@@ -720,7 +844,7 @@ def _prepare_docker_config(self, docker_config: str) -> str:
720844 Writes an empty-credsStore config.json (avoids macOS credential-helper
721845 hangs / `osxkeychain` prompts) and, critically, symlinks the real
722846 ``~/.docker/cli-plugins`` (buildx et al.) into it. Pointing
723- DOCKER_CONFIG at a throwaway directory hides the CLI plugin directory —
847+ DOCKER_CONFIG at a throwaway directory hides the CLI plugin directory —
724848 ``docker buildx`` then fails with "unknown command: docker buildx"
725849 (reproduced: ``docker buildx build --load`` runs fine from an
726850 interactive shell but errors identically from the backend subprocess
0 commit comments