Skip to content

Commit 2c3121a

Browse files
author
hbibbensalem
committed
feat: implement destroy_deployment with S3 remote state backend
- Add TF_STATE_BUCKET/DYNAMODB support (bucket + table created on shared AWS account) - TerraformRunner.destroy() now retries with 10s/30s/60s backoff (AWS eventual consistency) - TerraformRunner.init() adds -input=false and -migrate-state -force-copy (fixes silent hangs and interactive prompts under non-interactive docker exec) - DeployOpsAgent.destroy_deployment() + _describe_remaining_resources() for post-failure diagnostics - New status 'destroyed' on Deployment model + Alembic migration - POST /api/jobs/{job_id}/destroy: owner-only auth, server-side service_name confirmation, no false success on failure - Bind-mount alembic/ in docker-compose.yml so future migrations are visible without rebuilding the image - Verified end-to-end: real terraform destroy succeeded against AWS, confirmed via ECS describe
1 parent f157185 commit 2c3121a

6 files changed

Lines changed: 294 additions & 8 deletions

File tree

Lines changed: 40 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,40 @@
1+
"""0002 add destroyed deployment status
2+
3+
Revision ID: b7f3a9c1d2e4
4+
Revises: 848480a9e6e6
5+
Create Date: 2026-08-18 00:00:00.000000
6+
7+
"""
8+
from typing import Sequence, Union
9+
10+
from alembic import op
11+
import sqlalchemy as sa
12+
13+
14+
# revision identifiers, used by Alembic.
15+
revision: str = 'b7f3a9c1d2e4'
16+
down_revision: Union[str, None] = '848480a9e6e6'
17+
branch_labels: Union[str, Sequence[str], None] = None
18+
depends_on: Union[str, Sequence[str], None] = None
19+
20+
21+
def upgrade() -> None:
22+
# 'destroyed' is distinct from 'rolled_back': a rollback leaves a
23+
# previous ECS task-definition revision running, a destroy leaves
24+
# nothing -- feature/destroy-deployment needs its own terminal status
25+
# rather than overloading 'rolled_back'.
26+
op.drop_constraint('ck_deployments_status', 'deployments', type_='check')
27+
op.create_check_constraint(
28+
'ck_deployments_status',
29+
'deployments',
30+
"status IN ('pending', 'applying', 'succeeded', 'failed', 'rolled_back', 'destroyed')",
31+
)
32+
33+
34+
def downgrade() -> None:
35+
op.drop_constraint('ck_deployments_status', 'deployments', type_='check')
36+
op.create_check_constraint(
37+
'ck_deployments_status',
38+
'deployments',
39+
"status IN ('pending', 'applying', 'succeeded', 'failed', 'rolled_back')",
40+
)

‎infrastructure/docker-compose.yml‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -70,14 +70,15 @@ services:
7070
REDIS_URL: redis://redis:6379/2
7171
QDRANT_URL: http://qdrant:6333
7272
# Persisted across container recreation (unlike bare /tmp), so a
73-
# terraform.tfstate from a real apply survives a backend rebuild —
73+
# terraform.tfstate from a real apply survives a backend rebuild —
7474
# lost one on 2026-08-12 mid-session, had to hand-delete AWS resources.
7575
DEPLOYOPS_WORKSPACE_ROOT: /data/deployops
7676
ports:
7777
- "${BACKEND_PORT:-8000}:8000"
7878
volumes:
7979
- deployops_workspace:/data/deployops
8080
- ../src:/app/src
81+
- ../alembic:/app/alembic
8182
- /var/run/docker.sock:/var/run/docker.sock
8283
depends_on:
8384
postgres:

‎src/agents/deployops/agent.py‎

Lines changed: 125 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -471,6 +471,130 @@ async def rollback(self, job_id: str, payload: Dict[str, Any]) -> Dict[str, Any]
471471
self.logger.error(f"Rollback failed: {e}")
472472
return {"status": "failed", "error": str(e)}
473473

474+
@xray_recorder.capture("destroy_deployment") # type: ignore[reportCallIssue]
475+
async def destroy_deployment(
476+
self,
477+
job_id: str,
478+
aws_config: Dict[str, Any],
479+
) -> Dict[str, Any]:
480+
"""Destroy a job's deployed infrastructure via `terraform destroy`,
481+
reusing the persisted remote state in TF_STATE_BUCKET (see
482+
_write_remote_state_backend). Falls back to reporting "no state" for
483+
deployments made before remote state was enabled, rather than
484+
pretending to succeed. On failure, describes what's still live in
485+
AWS (mirrors rollback()/the /monitoring endpoint) so the caller can
486+
tell the user precisely what to clean up manually -- per the
487+
feature/destroy-deployment design decisions.
488+
"""
489+
workspace_dir = self._workspace_dir(job_id)
490+
tf_dir = workspace_dir / "terraform"
491+
tf_dir.mkdir(parents=True, exist_ok=True)
492+
self._write_remote_state_backend(tf_dir, job_id=job_id)
493+
494+
if not (tf_dir / "backend.tf").exists():
495+
self.logger.warning(f"[{job_id}] Destroy: TF_STATE_BUCKET not configured, cannot recover state")
496+
return {
497+
"status": "no_state",
498+
"job_id": job_id,
499+
"message": (
500+
"TF_STATE_BUCKET is not configured; this deployment has no "
501+
"recoverable Terraform state. Manual AWS cleanup required."
502+
),
503+
"remaining_resources": await self._describe_remaining_resources(job_id, aws_config),
504+
}
505+
506+
tf_runner = TerraformRunner(tf_dir)
507+
508+
if not tf_runner.init():
509+
self.logger.error(f"[{job_id}] Destroy: terraform init failed")
510+
return {
511+
"status": "failed",
512+
"job_id": job_id,
513+
"error": "terraform_init_failed",
514+
"message": "Could not initialize Terraform with the remote state backend.",
515+
"remaining_resources": await self._describe_remaining_resources(job_id, aws_config),
516+
}
517+
518+
# A job with no prior real `apply` (or one applied before
519+
# TF_STATE_BUCKET existed) has no state key in S3 -- init() still
520+
# succeeds against an empty backend, so check for an actual state
521+
# explicitly rather than trusting init() alone.
522+
state_output = tf_runner.output()
523+
if not state_output:
524+
self.logger.info(f"[{job_id}] Destroy: no Terraform state found in remote backend")
525+
return {
526+
"status": "no_state",
527+
"job_id": job_id,
528+
"message": (
529+
"No Terraform state found for this job (deployed before "
530+
"remote state was enabled, or state was lost). Nothing to "
531+
"destroy via Terraform -- check AWS directly for leftover "
532+
"resources."
533+
),
534+
"remaining_resources": await self._describe_remaining_resources(job_id, aws_config),
535+
}
536+
537+
try:
538+
destroyed = tf_runner.destroy()
539+
except Exception as exc:
540+
self.logger.error(f"[{job_id}] terraform destroy failed after retries: {exc}")
541+
return {
542+
"status": "partial_failure",
543+
"job_id": job_id,
544+
"error": str(exc),
545+
"remaining_resources": await self._describe_remaining_resources(job_id, aws_config),
546+
}
547+
548+
if not destroyed:
549+
return {
550+
"status": "partial_failure",
551+
"job_id": job_id,
552+
"error": "terraform destroy did not report success",
553+
"remaining_resources": await self._describe_remaining_resources(job_id, aws_config),
554+
}
555+
556+
self.logger.info(f"[{job_id}] Destroy successful")
557+
return {"status": "success", "job_id": job_id}
558+
559+
async def _describe_remaining_resources(
560+
self, job_id: str, aws_config: Dict[str, Any]
561+
) -> Dict[str, Any]:
562+
"""Best-effort snapshot of what's still live in AWS after a failed,
563+
partial, or unrecoverable-state destroy -- so the caller can tell the
564+
user precisely what to clean up manually and where (mirrors the read
565+
pattern already used by rollback() and the /monitoring endpoint)."""
566+
region = aws_config.get("region") or "us-east-1"
567+
cluster = aws_config.get("ecs_cluster")
568+
service_name = aws_config.get("service_name")
569+
remaining: Dict[str, Any] = {
570+
"ecs_service": None,
571+
"target_groups": [],
572+
"error": None,
573+
}
574+
if not cluster or not service_name:
575+
remaining["error"] = "no ecs_cluster/service_name available to check"
576+
return remaining
577+
try:
578+
aws = AWSClient(region=region)
579+
ecs = aws.ecs()
580+
desc = ecs.describe_services(cluster=cluster, services=[service_name])
581+
services = desc.get("services") or []
582+
if services and services[0].get("status") != "INACTIVE":
583+
svc = services[0]
584+
remaining["ecs_service"] = {
585+
"cluster": cluster,
586+
"service_name": service_name,
587+
"status": svc.get("status"),
588+
"running_count": svc.get("runningCount"),
589+
}
590+
elbv2 = aws.session.client("elbv2", region_name=region, config=RETRY_CONFIG)
591+
for lb in svc.get("loadBalancers", []):
592+
tg_arn = lb.get("targetGroupArn")
593+
if tg_arn:
594+
remaining["target_groups"].append(tg_arn)
595+
except Exception as exc: # noqa: BLE001 - best-effort diagnostic, never raise
596+
remaining["error"] = str(exc)
597+
return remaining
474598
@xray_recorder.capture("rollback_deployment") # type: ignore[reportCallIssue]
475599
async def rollback_deployment(
476600
self,
@@ -720,7 +844,7 @@ def _prepare_docker_config(self, docker_config: str) -> str:
720844
Writes an empty-credsStore config.json (avoids macOS credential-helper
721845
hangs / `osxkeychain` prompts) and, critically, symlinks the real
722846
``~/.docker/cli-plugins`` (buildx et al.) into it. Pointing
723-
DOCKER_CONFIG at a throwaway directory hides the CLI plugin directory —
847+
DOCKER_CONFIG at a throwaway directory hides the CLI plugin directory —
724848
``docker buildx`` then fails with "unknown command: docker buildx"
725849
(reproduced: ``docker buildx build --load`` runs fine from an
726850
interactive shell but errors identically from the backend subprocess

‎src/backend/api/jobs.py‎

Lines changed: 96 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -346,7 +346,7 @@ def edit_artifacts(
346346
),
347347
)
348348

349-
# Validate every edit up front so a batch fails atomically — never a
349+
# Validate every edit up front so a batch fails atomically — never a
350350
# partial write of broken files.
351351
for edit in body.files:
352352
if not allowed_file_path(edit.file_path):
@@ -520,6 +520,101 @@ def rollback_job(
520520
}
521521

522522

523+
class DestroyRequest(BaseModel):
524+
confirm_service_name: str
525+
526+
527+
def _extract_aws_config(deployment: models.Deployment) -> dict:
528+
"""Resolve ecs_cluster/service_name/region from a deployment's
529+
infrastructure_json, trying the DeployOps-native `aws_config` shape
530+
first and falling back to the `terraform_outputs` shape (with the same
531+
suffix-derivation workaround as the /monitoring endpoint) for older
532+
deployments where ecs_cluster_name was stored empty."""
533+
infra = deployment.infrastructure_json or {}
534+
aws_config = infra.get("aws_config") or {}
535+
region = aws_config.get("region") or deployment.aws_region or "us-east-1"
536+
ecs_cluster = aws_config.get("ecs_cluster")
537+
service_name = aws_config.get("service_name")
538+
539+
if not ecs_cluster or not service_name:
540+
tf_outputs = infra.get("terraform_outputs") or {}
541+
service_name = service_name or tf_outputs.get("service_name")
542+
ecs_cluster = ecs_cluster or tf_outputs.get("ecs_cluster_name")
543+
if not ecs_cluster and service_name and "-" in service_name:
544+
suffix = service_name.rsplit("-", 1)[-1]
545+
ecs_cluster = f"devguard-cluster-{suffix}"
546+
547+
return {
548+
"region": region,
549+
"ecs_cluster": ecs_cluster,
550+
"service_name": service_name,
551+
}
552+
553+
554+
@router.post("/{job_id}/destroy")
555+
def destroy_job(
556+
job_id: str,
557+
body: DestroyRequest,
558+
db: Session = Depends(get_db),
559+
current_user: models.User = Depends(get_current_user),
560+
):
561+
"""Destroy a job's deployed infrastructure. Requires the caller to type
562+
the exact service_name as confirmation (feature/destroy-deployment
563+
decision #2) -- enforced server-side, not just via a disabled UI button.
564+
"""
565+
run = _get_owned_run(db, job_id, current_user.id)
566+
567+
deployment = (
568+
db.query(models.Deployment)
569+
.filter(models.Deployment.run_id == job_id)
570+
.order_by(models.Deployment.created_at.desc())
571+
.first()
572+
)
573+
if not deployment:
574+
raise HTTPException(status_code=404, detail="No deployment found for this job")
575+
576+
aws_config = _extract_aws_config(deployment)
577+
service_name = aws_config.get("service_name")
578+
579+
if not service_name:
580+
raise HTTPException(
581+
status_code=400,
582+
detail="Deployment has no service_name on record; cannot confirm destroy target.",
583+
)
584+
585+
if body.confirm_service_name != service_name:
586+
raise HTTPException(
587+
status_code=400,
588+
detail=f"Confirmation text does not match. Expected the service name '{service_name}'.",
589+
)
590+
591+
try:
592+
from src.agents.deployops.agent import DeployOpsAgent
593+
594+
agent = DeployOpsAgent()
595+
result = asyncio.run(agent.destroy_deployment(job_id=job_id, aws_config=aws_config))
596+
except Exception as exc:
597+
logger.error(f"Destroy failed for {job_id}: {exc}", exc_info=True)
598+
raise HTTPException(status_code=500, detail=f"Destroy failed: {exc}") from exc
599+
600+
status = result.get("status")
601+
if status == "success":
602+
deployment.status = "destroyed"
603+
db.commit()
604+
if run.status not in ("completed", "failed", "rejected"):
605+
run.status = "destroyed"
606+
run.completed_at = datetime.utcnow()
607+
db.commit()
608+
publish_results_ready(job_id)
609+
return {"job_id": job_id, "status": "destroyed", "result": result}
610+
611+
# no_state / failed / partial_failure: don't touch deployment.status --
612+
# leaving it as-is is more honest than silently marking it destroyed
613+
# when resources may still be live. remaining_resources lets the UI
614+
# show exactly what's left and where (feature/destroy-deployment
615+
# decision #3).
616+
return {"job_id": job_id, "status": status, "result": result}
617+
523618
@router.get("/{job_id}/deployments/revisions")
524619
def list_deployment_revisions(
525620
job_id: str,

‎src/backend/models.py‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -85,6 +85,7 @@ class DeploymentStatus(str, PyEnum):
8585
SUCCEEDED = "succeeded"
8686
FAILED = "failed"
8787
ROLLED_BACK = "rolled_back"
88+
DESTROYED = "destroyed"
8889

8990

9091
class NotificationType(str, PyEnum):
@@ -317,7 +318,7 @@ class Deployment(Base):
317318
name="ck_deployments_environment"
318319
),
319320
CheckConstraint(
320-
"status IN ('pending', 'applying', 'succeeded', 'failed', 'rolled_back')",
321+
"status IN ('pending', 'applying', 'succeeded', 'failed', 'rolled_back', 'destroyed')",
321322
name="ck_deployments_status"
322323
),
323324
Index("idx_deployments_run_id", "run_id"),

‎src/lib/terraform/runner.py‎

Lines changed: 29 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -12,15 +12,24 @@ def __init__(self, working_dir: Path):
1212
self.working_dir = working_dir
1313
self.working_dir.mkdir(parents=True, exist_ok=True)
1414

15-
def _retry_with_backoff(self, func: Callable[[], Any], max_attempts: int = 3, base_delay: float = 2.0) -> Any:
15+
def _retry_with_backoff(
16+
self,
17+
func: Callable[[], Any],
18+
max_attempts: int = 3,
19+
base_delay: float = 2.0,
20+
delays: Optional[List[float]] = None,
21+
) -> Any:
1622
last_exception = None
1723
for attempt in range(1, max_attempts + 1):
1824
try:
1925
return func()
2026
except Exception as e:
2127
last_exception = e
2228
if attempt < max_attempts:
23-
delay = base_delay * (2 ** (attempt - 1))
29+
if delays:
30+
delay = delays[min(attempt - 1, len(delays) - 1)]
31+
else:
32+
delay = base_delay * (2 ** (attempt - 1))
2433
logger.warning(f"Attempt {attempt} failed: {e}. Retrying in {delay}s...")
2534
time.sleep(delay)
2635
else:
@@ -60,7 +69,18 @@ def _run(self, cmd: List[str]) -> subprocess.CompletedProcess:
6069

6170
def init(self) -> bool:
6271
def _init():
63-
result = self._run(["init"])
72+
# -input=false: a backend change (e.g. first init against the new
73+
# S3 remote state) prompts interactively to migrate existing
74+
# state; under a non-interactive docker exec that has no stdin,
75+
# the prompt just hangs forever instead of erroring -- confirmed
76+
# live during feature/destroy-deployment testing.
77+
# -migrate-state -force-copy: a workspace with pre-existing local
78+
# state (DEPLOYOPS_WORKSPACE_ROOT is a persisted volume, so state
79+
# from before TF_STATE_BUCKET was configured can still be on disk)
80+
# triggers a migration prompt when the backend changes to S3;
81+
# -force-copy auto-approves it non-interactively instead of
82+
# hanging or erroring under -input=false.
83+
result = self._run(["init", "-input=false", "-migrate-state", "-force-copy"])
6484
if result.returncode != 0:
6585
raise RuntimeError(f"init failed: {result.stderr}")
6686
return True
@@ -110,7 +130,12 @@ def _destroy():
110130
if result.returncode != 0:
111131
raise RuntimeError(f"destroy failed: {result.stderr}")
112132
return True
113-
return self._retry_with_backoff(_destroy)
133+
# Destroy failures are often AWS eventual-consistency issues (e.g. a
134+
# lingering ENI blocking security-group deletion with
135+
# DependencyViolation) that clear on the order of tens of seconds,
136+
# not the 2s/4s/8s used for apply/init/plan -- so destroy gets a
137+
# slower, explicit backoff (feature/destroy-deployment decision).
138+
return self._retry_with_backoff(_destroy, max_attempts=4, delays=[10.0, 30.0, 60.0])
114139

115140
def output(self) -> Dict:
116141
result = self._run(["output", "-json"])

0 commit comments

Comments
 (0)