From d06f06c2ff6bf88fe05eca749e3a748ba34418cd Mon Sep 17 00:00:00 2001 From: Cristian Magherusan-Stanciu Date: Thu, 28 May 2026 20:36:19 +0200 Subject: [PATCH 1/3] sec(ci): gate terraform force-unlock behind explicit input (closes #438) Unconditionally deleting the S3 state lock before every terraform init allowed concurrent dispatch runs to race and corrupt state. Remove the automatic deletion and replace it with a gated step that only runs when the operator explicitly sets clear_stale_lock=true on workflow_dispatch. The existing failure/cancellation cleanup step is unchanged. Add runbooks/terraform-stuck-lock.md with diagnosis and recovery steps. --- .github/workflows/deploy-aws-fargate.yml | 27 +++++++-- runbooks/terraform-stuck-lock.md | 75 ++++++++++++++++++++++++ 2 files changed, 98 insertions(+), 4 deletions(-) create mode 100644 runbooks/terraform-stuck-lock.md diff --git a/.github/workflows/deploy-aws-fargate.yml b/.github/workflows/deploy-aws-fargate.yml index a08a764d4..7577f60cb 100644 --- a/.github/workflows/deploy-aws-fargate.yml +++ b/.github/workflows/deploy-aws-fargate.yml @@ -12,8 +12,12 @@ # - ECR_REPOSITORY: ECR repository name # # Triggered by: -# - Manual workflow dispatch only (push trigger disabled — Fargate is not the primary compute platform) +# - Manual workflow dispatch only (push trigger disabled; Fargate is not the primary compute platform) # - Workflow call from deploy-all.yml +# +# Lock cleanup is intentionally manual. If a prior run crashed and left a stuck +# lock, re-run this workflow with clear_stale_lock=true (workflow_dispatch only). +# See runbooks/terraform-stuck-lock.md for diagnosis and manual recovery steps. name: Deploy to AWS Fargate @@ -33,6 +37,11 @@ on: required: true type: choice options: [dev, staging, prod] + clear_stale_lock: + description: 'Force-clear a stale S3 state lock from a prior crashed run (use only when you know the prior run is dead)' + required: false + type: boolean + default: false workflow_call: inputs: environment: @@ -105,18 +114,28 @@ jobs: with: terraform_version: ${{ env.TF_VERSION }} - - name: Terraform Init + - name: Clear stale state lock (operator-triggered only) + if: ${{ inputs.clear_stale_lock == true }} env: TF_BACKEND: ${{ secrets.TF_BACKEND_AWS }} ENVIRONMENT: ${{ needs.prepare.outputs.environment }} run: | printf '%s\nkey = "github-fargate-%s/terraform.tfstate"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend - # Break any stale state lock from a previous failed run BUCKET=$(grep -E '^\s*bucket\s*=' /tmp/backend.tfbackend 2>/dev/null | tr -d ' "' | cut -d= -f2) if [ -n "$BUCKET" ]; then LOCK_KEY="github-fargate-${ENVIRONMENT}/terraform.tfstate.tflock" - aws s3 rm "s3://${BUCKET}/${LOCK_KEY}" 2>/dev/null || true + echo "Removing stale lock: s3://${BUCKET}/${LOCK_KEY}" + aws s3 rm "s3://${BUCKET}/${LOCK_KEY}" 2>/dev/null || echo "No lock file found (already clean)" + else + echo "Could not parse bucket from backend config; skipping lock clear" fi + + - name: Terraform Init + env: + TF_BACKEND: ${{ secrets.TF_BACKEND_AWS }} + ENVIRONMENT: ${{ needs.prepare.outputs.environment }} + run: | + printf '%s\nkey = "github-fargate-%s/terraform.tfstate"\n' "$TF_BACKEND" "$ENVIRONMENT" > /tmp/backend.tfbackend cd terraform/environments/aws terraform init -backend-config=/tmp/backend.tfbackend diff --git a/runbooks/terraform-stuck-lock.md b/runbooks/terraform-stuck-lock.md new file mode 100644 index 000000000..238a7551e --- /dev/null +++ b/runbooks/terraform-stuck-lock.md @@ -0,0 +1,75 @@ +# Runbook: Terraform Stuck State Lock + +## Symptoms + +A Terraform workflow fails with an error similar to: + +``` +Error: Error acquiring the state lock +... +Lock Info: + ID: + Operation: OperationTypeApply + ... +``` + +or the S3 backend reports the `.tflock` object already exists from a prior run. + +## When this happens + +A lock file is left behind when a CI run is interrupted (runner killed, job +cancelled mid-apply, network error during apply). The lock is NOT automatically +cleared on normal failure; the "Release state lock on failure" step handles that +for clean failures and cancellations within the same run. A lock persisting +across runs means a previous job exited abnormally before that cleanup step +ran. + +## Safe recovery steps + +Before clearing the lock, confirm the prior run that owns it is truly dead: + +1. Note the `Lock Info.ID` from the error output. +2. In GitHub Actions, find the run that created the lock (check the timestamp in + `Lock Info.Created`). Verify its status is "Failed" or "Cancelled" -- not + still in progress. +3. If the owning run is still running, do NOT clear the lock. Wait for it to + finish or cancel it explicitly first. + +## Clearing the lock via workflow dispatch + +Once you have confirmed the owning run is dead: + +1. Go to **Actions > Deploy to AWS Fargate > Run workflow**. +2. Select the affected environment. +3. Check **"Force-clear a stale S3 state lock from a prior crashed run"**. +4. Click **Run workflow**. + +The `Clear stale state lock (operator-triggered only)` step will remove the +`.tflock` object from S3 and then proceed with a normal deploy. + +## Manual recovery (AWS CLI) + +If you prefer to clear the lock outside CI: + +```bash +# Replace and with the actual backend bucket and environment name +BUCKET= +ENV= +LOCK_KEY="github-fargate-${ENV}/terraform.tfstate.tflock" + +aws s3 ls "s3://${BUCKET}/${LOCK_KEY}" # confirm it exists +aws s3 rm "s3://${BUCKET}/${LOCK_KEY}" # remove it +``` + +Then re-run the deployment workflow normally (without `clear_stale_lock`). + +## Why locks are not cleared automatically on every run + +Unconditionally deleting the lock before `terraform init` would allow two +concurrent deploys to race: the second run could remove the first run's active +lock, enabling parallel state writes and potential state corruption. The lock +must only be cleared when you know the holder is no longer active. + +`deploy-aws-lambda.yml` follows the same pattern: cleanup only runs in the +`if: failure() || cancelled()` step of the same job, which covers the +within-run case; cross-run stuck locks require explicit operator action. From 6715e24c0a1b3197b1602f66bbc36848a794839d Mon Sep 17 00:00:00 2001 From: Cristian Magherusan-Stanciu Date: Mon, 1 Jun 2026 19:22:55 +0200 Subject: [PATCH 2/3] fix(ci): add language tag to fenced code block in terraform-stuck-lock.md markdownlint MD040 requires fenced code blocks to specify a language. The error-output block at line 7 had a bare ``` opener; add "text" tag. --- runbooks/terraform-stuck-lock.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/runbooks/terraform-stuck-lock.md b/runbooks/terraform-stuck-lock.md index 238a7551e..9fc238a17 100644 --- a/runbooks/terraform-stuck-lock.md +++ b/runbooks/terraform-stuck-lock.md @@ -4,7 +4,7 @@ A Terraform workflow fails with an error similar to: -``` +```text Error: Error acquiring the state lock ... Lock Info: From a344ca18d3b06c60d8cbf1fd58ad886c1f37c608 Mon Sep 17 00:00:00 2001 From: Cristian Magherusan-Stanciu Date: Fri, 19 Jun 2026 23:40:28 +0200 Subject: [PATCH 3/3] sec(ci): drop unsafe failure-path state-lock delete (closes #438) The "Release state lock on failure" step deleted the S3 tflock on any failed/cancelled run. A run that fails because it could not acquire the lock would delete the lock a concurrent run is still actively holding, re-introducing the state-corruption race this PR aims to close. Remove the automatic failure-path unlock. Terraform already releases its own lock on a clean apply error; a lock surviving a run means the run died abnormally, which requires operator confirmation via clear_stale_lock before clearing. Update the runbook accordingly. --- .github/workflows/deploy-aws-fargate.yml | 19 ++++++-------- runbooks/terraform-stuck-lock.md | 32 ++++++++++++++---------- 2 files changed, 27 insertions(+), 24 deletions(-) diff --git a/.github/workflows/deploy-aws-fargate.yml b/.github/workflows/deploy-aws-fargate.yml index 7577f60cb..4db561048 100644 --- a/.github/workflows/deploy-aws-fargate.yml +++ b/.github/workflows/deploy-aws-fargate.yml @@ -157,17 +157,14 @@ jobs: cd terraform/environments/aws terraform apply -auto-approve tfplan - - name: Release state lock on failure - if: failure() || cancelled() - env: - ENVIRONMENT: ${{ needs.prepare.outputs.environment }} - run: | - BUCKET=$(grep -E '^\s*bucket\s*=' /tmp/backend.tfbackend 2>/dev/null | tr -d ' "' | cut -d= -f2) - if [ -n "$BUCKET" ]; then - LOCK_KEY="github-fargate-${ENVIRONMENT}/terraform.tfstate.tflock" - echo "Releasing S3 state lock: s3://${BUCKET}/${LOCK_KEY}" - aws s3 rm "s3://${BUCKET}/${LOCK_KEY}" 2>/dev/null || echo "No lock file found (already clean)" - fi + # No automatic state-lock release on failure. A blanket failure-path + # delete is itself a state-corruption risk: a run that fails *because* + # it could not acquire the lock would delete the lock held by another + # run that is still actively applying. Terraform already releases its + # own lock on a clean apply error; a lock that survives a run means the + # run died abnormally, which requires operator confirmation that the + # owning run is dead before clearing it via the clear_stale_lock input. + # See runbooks/terraform-stuck-lock.md. - name: Get outputs id: outputs diff --git a/runbooks/terraform-stuck-lock.md b/runbooks/terraform-stuck-lock.md index 9fc238a17..9ba286b2c 100644 --- a/runbooks/terraform-stuck-lock.md +++ b/runbooks/terraform-stuck-lock.md @@ -18,11 +18,13 @@ or the S3 backend reports the `.tflock` object already exists from a prior run. ## When this happens A lock file is left behind when a CI run is interrupted (runner killed, job -cancelled mid-apply, network error during apply). The lock is NOT automatically -cleared on normal failure; the "Release state lock on failure" step handles that -for clean failures and cancellations within the same run. A lock persisting -across runs means a previous job exited abnormally before that cleanup step -ran. +canceled mid-apply, network error during apply). The workflow does NOT +auto-delete the lock on failure: a blanket failure-path delete would itself +risk state corruption, because a run that fails *because* it could not acquire +the lock would delete the lock another run is still actively holding. Terraform +releases its own lock on a clean apply error, so a lock that survives a run +means that run died abnormally; clearing it requires operator confirmation that +the owning run is dead. ## Safe recovery steps @@ -63,13 +65,17 @@ aws s3 rm "s3://${BUCKET}/${LOCK_KEY}" # remove it Then re-run the deployment workflow normally (without `clear_stale_lock`). -## Why locks are not cleared automatically on every run +## Why locks are not cleared automatically -Unconditionally deleting the lock before `terraform init` would allow two -concurrent deploys to race: the second run could remove the first run's active -lock, enabling parallel state writes and potential state corruption. The lock -must only be cleared when you know the holder is no longer active. +Any unconditional lock delete (before `terraform init`, or in a blanket +failure-path cleanup step) lets two concurrent deploys race: one run could +remove another run's active lock, enabling parallel state writes and potential +state corruption. A failure-path delete is especially dangerous because a run +that fails *because* it lost the race to acquire the lock would then delete the +winner's live lock. The lock must only be cleared when you know the holder is no +longer active, which is why clearing is gated behind the operator-triggered +`clear_stale_lock` input rather than an automatic step. -`deploy-aws-lambda.yml` follows the same pattern: cleanup only runs in the -`if: failure() || cancelled()` step of the same job, which covers the -within-run case; cross-run stuck locks require explicit operator action. +Terraform releases its own lock on a clean apply error within the same run, so +this gating only affects the rarer case where a run dies abnormally before that +release happens.