Skip to content
4 changes: 4 additions & 0 deletions .github/workflows/deploy-agentcore.yml
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,10 @@ jobs:
steps:
- uses: actions/checkout@v4

- uses: hashicorp/setup-terraform@v3
with:
terraform_wrapper: false

- name: Select the branch's stack (fail-closed)
id: sel
run: |
Expand Down
59 changes: 32 additions & 27 deletions .github/workflows/deploy-web.yml
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,11 @@ on:
image_sha:
description: "Full commit SHA whose web-<sha> image to promote and roll (default: the dispatched ref's HEAD)"
required: false
build:
description: "Build & push web-<HEAD sha> first, then roll it (for a stack whose image was never built / was lost). Mutually exclusive with image_sha."
type: boolean
required: false
default: false

permissions:
contents: read
Expand All @@ -50,7 +55,7 @@ permissions:
jobs:
build:
name: Build & push (arm64)
if: github.event_name == 'push'
if: github.event_name == 'push' || (github.event_name == 'workflow_dispatch' && inputs.build)
runs-on: sample-awsops
concurrency:
group: deploy-web-build-${{ github.ref_name }}
Expand All @@ -59,11 +64,8 @@ jobs:
BRANCH: ${{ github.ref_name }}
MAIN_ROLE: ${{ secrets.AWS_CI_BUILD_ROLE_ARN }}
DEV_ROLE: ${{ secrets.AWS_CI_BUILD_DEV_ROLE_ARN }}
MAIN_BACKEND_B64: ${{ secrets.TF_BACKEND_HCL }}
MAIN_TFVARS_B64: ${{ secrets.TF_TFVARS }}
DEV_BACKEND_B64: ${{ secrets.TF_BACKEND_HCL_DEV }}
DEV_TFVARS_B64: ${{ secrets.TF_TFVARS_DEV }}
USER_BACKEND_B64: ${{ secrets[format('TF_BACKEND_HCL_PREVIEW_{0}', github.ref_name)] }}
USER_TFVARS_B64: ${{ secrets[format('TF_TFVARS_PREVIEW_{0}', github.ref_name)] }}
steps:
- uses: actions/checkout@v4
Expand All @@ -86,39 +88,33 @@ jobs:
role-to-assume: ${{ steps.sel.outputs.role }}
aws-region: ap-northeast-2

- name: Restore terraform.foundation backend (read-only — just need `output`)
working-directory: terraform/foundation
# The build job needs exactly one stack fact — the web ECR repository —
# and derives it WITHOUT reading Terraform state (the ci-build role is
# ECR-push-only by design; no S3 tfstate access): repo name is
# "${project}-web" (terraform/foundation/ecr.tf), project comes from the
# stack's tfvars blob, registry from the ECR login below.
- name: Resolve the stack's project name from tfvars
id: stack
run: |
case "$BRANCH" in
main) B="$MAIN_BACKEND_B64"; V="$MAIN_TFVARS_B64";;
dev) B="$DEV_BACKEND_B64"; V="$DEV_TFVARS_B64";;
atomoh|ssminji|whchoi) B="$USER_BACKEND_B64"; V="$USER_TFVARS_B64";;
main) V="$MAIN_TFVARS_B64";;
dev) V="$DEV_TFVARS_B64";;
atomoh|ssminji|whchoi) V="$USER_TFVARS_B64";;
*) echo "::error::unexpected branch '$BRANCH'"; exit 1;;
esac
if [ -z "$B" ] || [ -z "$V" ]; then
echo "::error::TF backend secrets for '$BRANCH' are not set — provision its stack first (docs/runbooks/branch-strategy.md). No cross-branch fallback, by design."
if [ -z "$V" ]; then
echo "::error::TF tfvars secret for '$BRANCH' is not set — provision its stack first (docs/runbooks/branch-strategy.md). No cross-branch fallback, by design."
exit 1
fi
echo "$B" | base64 -d > backend.hcl
echo "$V" | base64 -d > terraform.tfvars
terraform init -backend-config=backend.hcl -input=false

- name: Resolve ECR repo URI
id: tf
working-directory: terraform/foundation
run: echo "ecr_web_uri=$(terraform output -raw ecr_web_uri)" >> "$GITHUB_OUTPUT"

# The runner is persistent and shared — never leave the restored stack
# config behind, whatever the outcome.
- name: Clean restored terraform config off the runner
if: always()
working-directory: terraform/foundation
run: rm -f terraform.tfvars backend.hcl
PROJECT=$(echo "$V" | base64 -d | sed -nE 's/^[[:space:]]*project[[:space:]]*=[[:space:]]*"([^"]+)".*/\1/p' | head -1)
PROJECT="${PROJECT:-awsops-v2}" # variables.tf default
echo "project=$PROJECT" >> "$GITHUB_OUTPUT"

- uses: docker/setup-qemu-action@v3
- uses: docker/setup-buildx-action@v3

- name: Login to Amazon ECR
id: ecr
uses: aws-actions/amazon-ecr-login@v2

- name: Copy CHANGELOG into build context
Expand All @@ -134,7 +130,7 @@ jobs:
push: true
# ONLY the immutable per-commit tag — :web-latest is written solely
# by the deploy job's pin step.
tags: ${{ steps.tf.outputs.ecr_web_uri }}:web-${{ github.sha }}
tags: ${{ steps.ecr.outputs.registry }}/${{ steps.stack.outputs.project }}-web:web-${{ github.sha }}

deploy:
name: Roll ECS service
Expand Down Expand Up @@ -163,6 +159,10 @@ jobs:
steps:
- uses: actions/checkout@v4

- uses: hashicorp/setup-terraform@v3
with:
terraform_wrapper: false

- name: Select the branch's stack (fail-closed — no cross-branch fallback)
id: sel
run: |
Expand Down Expand Up @@ -228,7 +228,12 @@ jobs:
# has deployed since; prefer the explicit dispatch.)
env:
PIN_SHA: ${{ inputs.image_sha || github.sha }}
DISPATCH_BUILD: ${{ inputs.build }}
DISPATCH_IMAGE_SHA: ${{ inputs.image_sha }}
run: |
if [ "$DISPATCH_BUILD" = "true" ] && [ -n "$DISPATCH_IMAGE_SHA" ]; then
echo "::error::'build' and 'image_sha' are mutually exclusive — a fresh build is always HEAD's sha."; exit 1
fi
MANIFEST=$(aws ecr batch-get-image \
--repository-name "${{ steps.tf.outputs.ecr_repo }}" \
--image-ids imageTag="web-${PIN_SHA}" \
Expand Down
17 changes: 15 additions & 2 deletions .github/workflows/terraform.yml
Original file line number Diff line number Diff line change
Expand Up @@ -26,7 +26,9 @@ on:
paths: ["terraform/foundation/**"]
push:
branches: [main, dev, atomoh, ssminji, whchoi]
paths: ["terraform/foundation/**"]
# The workflow file itself is in the path filter so a workflow change
# self-tests with a real plan on the branch it lands on.
paths: ["terraform/foundation/**", ".github/workflows/terraform.yml"]
workflow_dispatch:
inputs:
plan_run_id:
Expand Down Expand Up @@ -61,6 +63,10 @@ jobs:
steps:
- uses: actions/checkout@v4

- uses: hashicorp/setup-terraform@v3
with:
terraform_wrapper: false

- name: Configure AWS credentials (OIDC -> ci-terraform-plan)
uses: aws-actions/configure-aws-credentials@v4
with:
Expand Down Expand Up @@ -107,7 +113,10 @@ jobs:
# is deliberately one shared credential across stacks; admin users are per-stack
# and NOT created by CI (create_admin_user defaults to false).
TF_VAR_demo_password: ${{ secrets.TF_VAR_DEMO_PASSWORD }}
run: terraform plan -out=tfplan -input=false
# -lock=false: the plan role is ReadOnlyAccess by design and cannot write
# the S3 lock object (use_lockfile → <key>.tflock, s3:PutObject). A
# read-only plan needs no lock; apply (deployer role) still locks.
run: terraform plan -out=tfplan -input=false -lock=false

# A plan file embeds every input variable value in plaintext (sensitive
# ones included), and artifacts on a public repo are downloadable by any
Expand Down Expand Up @@ -154,6 +163,10 @@ jobs:
steps:
- uses: actions/checkout@v4

- uses: hashicorp/setup-terraform@v3
with:
terraform_wrapper: false

- name: Select the branch's stack (fail-closed)
id: sel
run: |
Expand Down
16 changes: 12 additions & 4 deletions docs/reference/01-edge-network.md
Original file line number Diff line number Diff line change
Expand Up @@ -32,9 +32,14 @@ viewer ──TLS──> CloudFront ──TLS (https-only:443)──> VPC Origin
CloudFront cert's existing Route53 CNAMEs). The ALB forwards to a `target_type = "ip"`
target group on the Fargate container port (`3000`), health check path `/api/health`.
- **ALB security group** allows **443 only from the CloudFront managed SG
`CloudFront-VPCOrigins-Service-SG`**, looked up via a `data "aws_security_group"` with two
filters: `group-name` + `vpc-id` (= `local.vpc_id`). The earlier broad VPC-CIDR :443 rule was
dropped — a VPC-CIDR-only rule causes a persistent 504.
`CloudFront-VPCOrigins-Service-SG`**, looked up via a plural `data "aws_security_groups"` with
two filters: `group-name` + `vpc-id` (= `local.vpc_id`). The earlier broad VPC-CIDR :443 rule
was dropped — a VPC-CIDR-only rule causes a persistent 504. **Fresh-VPC bootstrap:** that
managed SG only appears once the first VPC origin in the VPC exists (this stack's own), so on a
brand-new VPC the lookup is empty and the ALB SG has **no 443 ingress at all** (a `check` block
warns; deliberately no CIDR fallback — it would serve no CloudFront traffic and only open an
unauthenticated in-VPC path); the next plan/apply after the VPC origin exists adds the
managed-SG rule in place. Bootstrap mode is a plan-able state, not a serving state.
- **VPC: new-or-reuse via `create_network`.** `true` (default) builds a new VPC
(`10.20.0.0/16` default), 2 public + 2 private subnets, IGW, NAT, route tables. `false`
reuses an existing VPC (`existing_vpc_id` + `existing_private_subnet_ids`, no `ec2:Create*`,
Expand Down Expand Up @@ -87,7 +92,10 @@ The 504 → 200 root cause (reuse-critical — re-read before changing the edge)
validated through the CloudFront cert's existing Route53 CNAME records.
2. **ALB SG must allow 443 from `CloudFront-VPCOrigins-Service-SG`.** A broad VPC-CIDR-only :443
ingress rule produces a **persistent 504** — CloudFront's VPC Origin ENIs are reached via that
managed SG, not by CIDR. Reference it with a `data` lookup filtered on `group-name` + `vpc-id`.
managed SG, not by CIDR. Reference it with a plural `data "aws_security_groups"` lookup filtered
on `group-name` + `vpc-id` — plural, because on a brand-new VPC the SG does not exist until this
stack's own VPC origin is created (the singular lookup hard-fails every plan); the ALB SG has no
443 ingress until then, and a second apply adds the managed-SG rule.
3. **A VPC Origin's protocol cannot update in-place** while attached to a distribution (409
`CannotUpdateEntityWhileInUse`). Use `lifecycle { create_before_destroy = true }` + a **distinct
name** (e.g. `*-alb-origin-tls`) + a `terraform apply -replace` so Terraform stands up the new
Expand Down
23 changes: 23 additions & 0 deletions docs/runbooks/dev-repo-setup.md
Original file line number Diff line number Diff line change
Expand Up @@ -117,6 +117,29 @@ alarm subscriber). Stacks provisioned before this policy carried a TF-managed
admin user: the first post-merge plan proposes destroying it — that removal
is intentional (recreate via the CLI above when the stack actually needs an
admin).

⚠️ Identity caveat: app ownership (reports, chat threads, …) is keyed by the
Cognito `sub`, which is minted per user object — deleting and recreating a
user yields a NEW `sub`, so rows owned by the old identity do not follow it.
Only the legacy verified-email read path bridges some tables. Two distinct
situations:

- **TF-managed admin about to be destroyed by the config removal above**: the
apply WILL delete the user object; disabling cannot stop a planned destroy.
To keep the identity (and its `sub`) alive on a stack with real user-owned
data, detach it from state BEFORE the first post-merge apply:
`terraform -chdir=terraform/foundation state rm 'aws_cognito_user.admin'` —
Terraform then forgets the
resource without touching the live user. Skip this on stacks with nothing
to preserve and let the apply delete it.
- **Manually-provisioned users** (the CLI flow above): to revoke access,
prefer `admin-disable-user` over delete/recreate — deletion is identity
loss.

(TF-관리 admin은 다음 apply가 반드시 삭제합니다 — 보존하려면 apply 전에
`terraform -chdir=terraform/foundation state rm 'aws_cognito_user.admin'`으로
상태에서만 떼어냅니다.
disable은 삭제를 막지 못하며, 수동 생성 사용자에 대한 접근 차단 수단입니다.)
The plan artifact is a covered channel too: a tfplan embeds every variable
value in plaintext and public-repo artifacts are downloadable by anyone, so
the plan job encrypts it with the `TF_PLAN_ENC_KEY` secret (fail-closed) and
Expand Down
42 changes: 35 additions & 7 deletions terraform/foundation/workload.tf
Original file line number Diff line number Diff line change
Expand Up @@ -556,7 +556,17 @@ resource "aws_ecs_task_definition" "web" {

# CloudFront provisions VPC Origin ENIs into a managed SG ("CloudFront-VPCOrigins-Service-SG").
# The ALB allows ONLY that SG on 443 (review #3: dropped the broad VPC-CIDR rule).
data "aws_security_group" "cf_vpc_origin" {
#
# Fresh-VPC bootstrap: that managed SG only appears once the FIRST VPC origin in
# the VPC exists — which is this stack's own aws_cloudfront_vpc_origin, which in
# turn needs this ALB. The singular data source hard-failed the plan on any new
# VPC ("no matching EC2 Security Group found"; the original live env never hit
# it because it reused a VPC that already had another stack's VPC origin). The
# plural lookup returns an empty list instead, and the ALB simply has NO 443
# ingress while the SG is absent (no CIDR fallback — see the SG below); the next
# plan after the VPC origin exists finds the SG and adds the rule in place. The
# check block below surfaces the bootstrap state as a warning.
data "aws_security_groups" "cf_vpc_origin" {
filter {
name = "group-name"
values = ["CloudFront-VPCOrigins-Service-SG"]
Expand All @@ -567,6 +577,17 @@ data "aws_security_group" "cf_vpc_origin" {
}
}

locals {
cf_vpc_origin_sg_id = length(data.aws_security_groups.cf_vpc_origin.ids) > 0 ? data.aws_security_groups.cf_vpc_origin.ids[0] : null
}

check "cf_vpc_origin_sg_present" {
assert {
condition = local.cf_vpc_origin_sg_id != null
error_message = "CloudFront-VPCOrigins-Service-SG is not in this VPC yet — the ALB SG has NO 443 ingress (bootstrap). Expected on a brand-new VPC before the first apply creates the VPC origin; run plan/apply once more afterwards to add the managed-SG rule, or the edge stays 504."
}
}

resource "aws_security_group" "alb" {
name = "${var.project}-alb-sg"
# description kept verbatim to match the existing SG (AWS SG description is
Expand All @@ -576,13 +597,20 @@ resource "aws_security_group" "alb" {
description = "Internal ALB - reachable from within the VPC (CloudFront VPC Origin ENIs)"
vpc_id = local.vpc_id

ingress {
description = "HTTPS from CloudFront VPC Origin managed SG"
from_port = 443
to_port = 443
protocol = "tcp"
security_groups = [data.aws_security_group.cf_vpc_origin.id]
dynamic "ingress" {
for_each = local.cf_vpc_origin_sg_id != null ? [local.cf_vpc_origin_sg_id] : []
content {
description = "HTTPS from CloudFront VPC Origin managed SG"
from_port = 443
to_port = 443
protocol = "tcp"
security_groups = [ingress.value]
}
}
# While the managed SG is absent there is deliberately NO ingress rule at all:
# a VPC-CIDR fallback would serve no CloudFront traffic (ENIs are matched by
# the managed SG, not CIDR) and would only open an unauthenticated in-VPC path
# to the app. Nothing can reach the ALB until the second apply adds the rule.
egress {
from_port = 0
to_port = 0
Expand Down
Loading